sonilo-cli 0.12.0__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/PKG-INFO +45 -5
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/README.md +43 -3
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/pyproject.toml +2 -2
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/src/sonilo_cli/__init__.py +1 -1
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/src/sonilo_cli/__main__.py +143 -2
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/test_cli.py +261 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/.gitignore +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/LICENSE +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/src/sonilo_cli/credentials.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/src/sonilo_cli/login.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/__init__.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/conftest.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/test_context7.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/test_credentials.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/test_login.py +0 -0
- {sonilo_cli-0.12.0 → sonilo_cli-0.14.0}/tests/test_smoke.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: sonilo-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Summary: Command-line interface for the Sonilo API: generate music and sound effects from text or video
|
|
5
5
|
Project-URL: Repository, https://github.com/sonilo-ai/sonilo-python
|
|
6
6
|
Author: Sonilo AI
|
|
@@ -8,7 +8,7 @@ License-Expression: MIT
|
|
|
8
8
|
License-File: LICENSE
|
|
9
9
|
Keywords: ai,cli,music,sfx,sonilo,text-to-music,video-to-music
|
|
10
10
|
Requires-Python: >=3.9
|
|
11
|
-
Requires-Dist: sonilo<0.
|
|
11
|
+
Requires-Dist: sonilo<0.16,>=0.15.0
|
|
12
12
|
Provides-Extra: dev
|
|
13
13
|
Requires-Dist: pytest>=8; extra == 'dev'
|
|
14
14
|
Requires-Dist: respx>=0.21; extra == 'dev'
|
|
@@ -98,6 +98,8 @@ production sign-in coexist without overwriting each other.
|
|
|
98
98
|
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
99
99
|
# ducks the existing music bed under the voice; a video voice comes back
|
|
100
100
|
# as a new .mp4 with the ducked mix muxed in
|
|
101
|
+
sonilo video-analysis --video clip.mp4 --variants 2
|
|
102
|
+
# prints a creative brief as JSON; generates nothing
|
|
101
103
|
sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
|
|
102
104
|
# writes dubbed.es.mp4 and dubbed.fr.mp4
|
|
103
105
|
sonilo tasks get <task-id>
|
|
@@ -106,8 +108,8 @@ production sign-in coexist without overwriting each other.
|
|
|
106
108
|
### Notes
|
|
107
109
|
|
|
108
110
|
- `text-to-music` / `video-to-music` stream a short `.m4a` by default. `--format wav`,
|
|
109
|
-
`--preserve-speech`, `--variants` above 1, and the legacy alias `--isolate-vocals`
|
|
110
|
-
to the async submit-and-poll path.
|
|
111
|
+
`--preserve-speech`, `--variants` above 1, `--stems`, and the legacy alias `--isolate-vocals`
|
|
112
|
+
each switch to the async submit-and-poll path.
|
|
111
113
|
- `text-to-sfx` / `video-to-sfx` are always async; `--format` accepts `wav|mp3|aac|flac`.
|
|
112
114
|
- Output defaults to `./output.<ext>`; override with `--output`.
|
|
113
115
|
|
|
@@ -171,6 +173,27 @@ not sent at all and the API's own 0.5 default applies; `--prompt-influence 0` is
|
|
|
171
173
|
|
|
172
174
|
sonilo video-to-music --video clip.mp4 --prompt "tense synths" --prompt-influence 0.8
|
|
173
175
|
|
|
176
|
+
### Music stems
|
|
177
|
+
|
|
178
|
+
`--stems` on `text-to-music` and `video-to-music` also splits the generated music into four
|
|
179
|
+
stems — drums, bass, vocals, other — saved next to the main output with the stem name spliced
|
|
180
|
+
before the extension (`take.m4a` → `take.drums.m4a`, and per variant with `--variants` above 1:
|
|
181
|
+
`take.0.drums.m4a`). It is free of charge, and forces the async path. On `video-to-music` it
|
|
182
|
+
splits the *generated* music, never the video's own audio.
|
|
183
|
+
|
|
184
|
+
sonilo text-to-music --prompt "warm lo-fi piano" --duration 60 --stems --output take.m4a
|
|
185
|
+
# writes take.m4a, take.drums.m4a, take.bass.m4a, take.vocals.m4a, take.other.m4a
|
|
186
|
+
|
|
187
|
+
- Separation runs **after** generation and typically adds 2-6 minutes to the wait; the CLI
|
|
188
|
+
waits up to 2400 seconds on these runs (covering the separation service's own 30-minute
|
|
189
|
+
ceiling, the way dubbing's `--timeout` covers its backend's). If the wait still times out,
|
|
190
|
+
the task keeps running server-side — resume it with `sonilo tasks wait <task-id>`.
|
|
191
|
+
- Separation can also come up short without failing the run: streams that did not separate are
|
|
192
|
+
reported on stderr (the API's `stems_error`), while the stems that did come back are still
|
|
193
|
+
saved — a partial result is not an error exit, and the main output is always written.
|
|
194
|
+
- Not the same flag as `--stem` on the sound commands, which saves layers those endpoints
|
|
195
|
+
already return; `--stems` *requests* a separation the API would not otherwise run.
|
|
196
|
+
|
|
174
197
|
### Scored video
|
|
175
198
|
|
|
176
199
|
`video-to-video-music` and `video-to-video-sfx` are the video-out counterparts of `video-to-music`
|
|
@@ -241,6 +264,23 @@ call instead.)
|
|
|
241
264
|
there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
|
|
242
265
|
- Each input is capped at 360 seconds server-side.
|
|
243
266
|
|
|
267
|
+
### Video analysis
|
|
268
|
+
|
|
269
|
+
`video-analysis` analyzes a video and prints a **creative brief** for scoring it. It is the one
|
|
270
|
+
command that produces no media file — nothing is generated:
|
|
271
|
+
|
|
272
|
+
sonilo video-analysis --video clip.mp4 --prompt "focus on the chase" --variants 2
|
|
273
|
+
|
|
274
|
+
- The brief goes to **stdout as JSON** so it can be piped into another tool: `segments` (a
|
|
275
|
+
time-aligned section plan) and `variations` (one ready-to-use generation prompt each). Pass
|
|
276
|
+
`--output brief.json` to write it to a file instead.
|
|
277
|
+
- `--variants` is 1-5 (default 1) and is **billed per brief**.
|
|
278
|
+
- Source videos may be at most 600 seconds long, and billing has a 10-second floor.
|
|
279
|
+
- Feed a variation's prompt straight into the next command:
|
|
280
|
+
|
|
281
|
+
sonilo video-analysis --video clip.mp4 --output brief.json
|
|
282
|
+
sonilo video-to-music --video clip.mp4 --prompt "$(jq -r '.variations[0].prompt' brief.json)"
|
|
283
|
+
|
|
244
284
|
### Dubbing
|
|
245
285
|
|
|
246
286
|
`dubbing` dubs a video into one or more target languages in a single async call:
|
|
@@ -267,7 +307,7 @@ required:
|
|
|
267
307
|
|
|
268
308
|
| Free runs | Endpoints |
|
|
269
309
|
| --- | --- |
|
|
270
|
-
| 2 each | text-to-music, text-to-sfx, audio-ducking |
|
|
310
|
+
| 2 each | text-to-music, text-to-sfx, audio-ducking, video-analysis |
|
|
271
311
|
| 1 each | video-to-music, video-to-sfx, video-to-video-music, video-to-video-sfx, video-to-sound, video-to-video-sound |
|
|
272
312
|
| 0 | dubbing |
|
|
273
313
|
|
|
@@ -82,6 +82,8 @@ production sign-in coexist without overwriting each other.
|
|
|
82
82
|
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
83
83
|
# ducks the existing music bed under the voice; a video voice comes back
|
|
84
84
|
# as a new .mp4 with the ducked mix muxed in
|
|
85
|
+
sonilo video-analysis --video clip.mp4 --variants 2
|
|
86
|
+
# prints a creative brief as JSON; generates nothing
|
|
85
87
|
sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
|
|
86
88
|
# writes dubbed.es.mp4 and dubbed.fr.mp4
|
|
87
89
|
sonilo tasks get <task-id>
|
|
@@ -90,8 +92,8 @@ production sign-in coexist without overwriting each other.
|
|
|
90
92
|
### Notes
|
|
91
93
|
|
|
92
94
|
- `text-to-music` / `video-to-music` stream a short `.m4a` by default. `--format wav`,
|
|
93
|
-
`--preserve-speech`, `--variants` above 1, and the legacy alias `--isolate-vocals`
|
|
94
|
-
to the async submit-and-poll path.
|
|
95
|
+
`--preserve-speech`, `--variants` above 1, `--stems`, and the legacy alias `--isolate-vocals`
|
|
96
|
+
each switch to the async submit-and-poll path.
|
|
95
97
|
- `text-to-sfx` / `video-to-sfx` are always async; `--format` accepts `wav|mp3|aac|flac`.
|
|
96
98
|
- Output defaults to `./output.<ext>`; override with `--output`.
|
|
97
99
|
|
|
@@ -155,6 +157,27 @@ not sent at all and the API's own 0.5 default applies; `--prompt-influence 0` is
|
|
|
155
157
|
|
|
156
158
|
sonilo video-to-music --video clip.mp4 --prompt "tense synths" --prompt-influence 0.8
|
|
157
159
|
|
|
160
|
+
### Music stems
|
|
161
|
+
|
|
162
|
+
`--stems` on `text-to-music` and `video-to-music` also splits the generated music into four
|
|
163
|
+
stems — drums, bass, vocals, other — saved next to the main output with the stem name spliced
|
|
164
|
+
before the extension (`take.m4a` → `take.drums.m4a`, and per variant with `--variants` above 1:
|
|
165
|
+
`take.0.drums.m4a`). It is free of charge, and forces the async path. On `video-to-music` it
|
|
166
|
+
splits the *generated* music, never the video's own audio.
|
|
167
|
+
|
|
168
|
+
sonilo text-to-music --prompt "warm lo-fi piano" --duration 60 --stems --output take.m4a
|
|
169
|
+
# writes take.m4a, take.drums.m4a, take.bass.m4a, take.vocals.m4a, take.other.m4a
|
|
170
|
+
|
|
171
|
+
- Separation runs **after** generation and typically adds 2-6 minutes to the wait; the CLI
|
|
172
|
+
waits up to 2400 seconds on these runs (covering the separation service's own 30-minute
|
|
173
|
+
ceiling, the way dubbing's `--timeout` covers its backend's). If the wait still times out,
|
|
174
|
+
the task keeps running server-side — resume it with `sonilo tasks wait <task-id>`.
|
|
175
|
+
- Separation can also come up short without failing the run: streams that did not separate are
|
|
176
|
+
reported on stderr (the API's `stems_error`), while the stems that did come back are still
|
|
177
|
+
saved — a partial result is not an error exit, and the main output is always written.
|
|
178
|
+
- Not the same flag as `--stem` on the sound commands, which saves layers those endpoints
|
|
179
|
+
already return; `--stems` *requests* a separation the API would not otherwise run.
|
|
180
|
+
|
|
158
181
|
### Scored video
|
|
159
182
|
|
|
160
183
|
`video-to-video-music` and `video-to-video-sfx` are the video-out counterparts of `video-to-music`
|
|
@@ -225,6 +248,23 @@ call instead.)
|
|
|
225
248
|
there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
|
|
226
249
|
- Each input is capped at 360 seconds server-side.
|
|
227
250
|
|
|
251
|
+
### Video analysis
|
|
252
|
+
|
|
253
|
+
`video-analysis` analyzes a video and prints a **creative brief** for scoring it. It is the one
|
|
254
|
+
command that produces no media file — nothing is generated:
|
|
255
|
+
|
|
256
|
+
sonilo video-analysis --video clip.mp4 --prompt "focus on the chase" --variants 2
|
|
257
|
+
|
|
258
|
+
- The brief goes to **stdout as JSON** so it can be piped into another tool: `segments` (a
|
|
259
|
+
time-aligned section plan) and `variations` (one ready-to-use generation prompt each). Pass
|
|
260
|
+
`--output brief.json` to write it to a file instead.
|
|
261
|
+
- `--variants` is 1-5 (default 1) and is **billed per brief**.
|
|
262
|
+
- Source videos may be at most 600 seconds long, and billing has a 10-second floor.
|
|
263
|
+
- Feed a variation's prompt straight into the next command:
|
|
264
|
+
|
|
265
|
+
sonilo video-analysis --video clip.mp4 --output brief.json
|
|
266
|
+
sonilo video-to-music --video clip.mp4 --prompt "$(jq -r '.variations[0].prompt' brief.json)"
|
|
267
|
+
|
|
228
268
|
### Dubbing
|
|
229
269
|
|
|
230
270
|
`dubbing` dubs a video into one or more target languages in a single async call:
|
|
@@ -251,7 +291,7 @@ required:
|
|
|
251
291
|
|
|
252
292
|
| Free runs | Endpoints |
|
|
253
293
|
| --- | --- |
|
|
254
|
-
| 2 each | text-to-music, text-to-sfx, audio-ducking |
|
|
294
|
+
| 2 each | text-to-music, text-to-sfx, audio-ducking, video-analysis |
|
|
255
295
|
| 1 each | video-to-music, video-to-sfx, video-to-video-music, video-to-video-sfx, video-to-sound, video-to-video-sound |
|
|
256
296
|
| 0 | dubbing |
|
|
257
297
|
|
|
@@ -4,13 +4,13 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "sonilo-cli"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.14.0"
|
|
8
8
|
description = "Command-line interface for the Sonilo API: generate music and sound effects from text or video"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
11
11
|
requires-python = ">=3.9"
|
|
12
12
|
authors = [{ name = "Sonilo AI" }]
|
|
13
|
-
dependencies = ["sonilo>=0.
|
|
13
|
+
dependencies = ["sonilo>=0.15.0,<0.16"]
|
|
14
14
|
keywords = ["sonilo", "cli", "music", "sfx", "text-to-music", "video-to-music", "ai"]
|
|
15
15
|
|
|
16
16
|
[project.urls]
|
|
@@ -11,6 +11,7 @@ from urllib.parse import urlparse
|
|
|
11
11
|
|
|
12
12
|
from sonilo import Sonilo
|
|
13
13
|
from sonilo.errors import APIError, SoniloError
|
|
14
|
+
from sonilo.resources.tasks import DEFAULT_WAIT_TIMEOUT
|
|
14
15
|
|
|
15
16
|
from sonilo_cli import __version__, credentials
|
|
16
17
|
from sonilo_cli.login import LoginError, cmd_login, cmd_logout, cmd_whoami
|
|
@@ -291,10 +292,53 @@ def _save_music_variants(result: Any, out: str) -> None:
|
|
|
291
292
|
_wrote(path, path.stat().st_size)
|
|
292
293
|
|
|
293
294
|
|
|
295
|
+
# Matched to the separation service's own ceiling on top of the generation
|
|
296
|
+
# wait: separation runs after generation, typically adds 2-6 minutes and gives
|
|
297
|
+
# up after 30 (1800s), so the SDK's generic DEFAULT_WAIT_TIMEOUT of 600s would
|
|
298
|
+
# abandon a stems task the user has already been charged the generation for —
|
|
299
|
+
# the same reasoning as DUBBING_WAIT_TIMEOUT below.
|
|
300
|
+
STEMS_WAIT_TIMEOUT = 2400.0
|
|
301
|
+
|
|
302
|
+
_MUSIC_STEMS = ("drums", "bass", "vocals", "other")
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _save_music_stems(result: Any, out: str) -> None:
|
|
306
|
+
"""Save the separated stems of an async music result next to `out`,
|
|
307
|
+
`take.m4a` becoming `take.drums.m4a` etc. — the same transform --stem
|
|
308
|
+
applies on the sound commands. With several variants the stem files pick
|
|
309
|
+
up the entry's own stream_index (`take.0.drums.m4a`), matching the
|
|
310
|
+
indexed file _save_music_variants wrote for that stream.
|
|
311
|
+
|
|
312
|
+
Entries are matched by stream_index, never list position — the list can
|
|
313
|
+
be shorter than `audio`. `stems_error` is a warning, not a failure, and
|
|
314
|
+
never suppresses the partial stems that did come back: the music itself
|
|
315
|
+
succeeded and separation is free, so incomplete stems must not turn the
|
|
316
|
+
whole command into an error exit."""
|
|
317
|
+
if result.stems_error:
|
|
318
|
+
sys.stderr.write(f"sonilo: stem separation incomplete: {result.stems_error}\n")
|
|
319
|
+
entries = result.stems or []
|
|
320
|
+
if not entries:
|
|
321
|
+
if not result.stems_error:
|
|
322
|
+
sys.stderr.write("sonilo: task succeeded but returned no stems\n")
|
|
323
|
+
return
|
|
324
|
+
multi = len(result.audio or []) > 1
|
|
325
|
+
for entry in entries:
|
|
326
|
+
base = _variant_path(out, entry.stream_index) if multi else out
|
|
327
|
+
for stem in _MUSIC_STEMS:
|
|
328
|
+
media = getattr(entry, stem, None)
|
|
329
|
+
if media is None:
|
|
330
|
+
continue
|
|
331
|
+
path = result.save_stem(
|
|
332
|
+
_stem_path(base, stem, media),
|
|
333
|
+
which=stem, stream_index=entry.stream_index,
|
|
334
|
+
)
|
|
335
|
+
_wrote(path, path.stat().st_size)
|
|
336
|
+
|
|
337
|
+
|
|
294
338
|
def cmd_text_to_music(client: Sonilo, args: argparse.Namespace) -> None:
|
|
295
339
|
fmt = args.format
|
|
296
340
|
multi = args.variants is not None and args.variants > 1
|
|
297
|
-
use_async = args.use_async or fmt != "m4a" or multi
|
|
341
|
+
use_async = args.use_async or fmt != "m4a" or multi or args.stems
|
|
298
342
|
out = _music_output(args, fmt)
|
|
299
343
|
segments = _segments(args)
|
|
300
344
|
if use_async:
|
|
@@ -304,8 +348,12 @@ def cmd_text_to_music(client: Sonilo, args: argparse.Namespace) -> None:
|
|
|
304
348
|
segments=segments,
|
|
305
349
|
output_format=fmt if fmt != "m4a" else None,
|
|
306
350
|
variants_num=args.variants,
|
|
351
|
+
stems=True if args.stems else None,
|
|
352
|
+
timeout=STEMS_WAIT_TIMEOUT if args.stems else DEFAULT_WAIT_TIMEOUT,
|
|
307
353
|
)
|
|
308
354
|
_save_music_variants(result, out)
|
|
355
|
+
if args.stems:
|
|
356
|
+
_save_music_stems(result, out)
|
|
309
357
|
else:
|
|
310
358
|
track = client.text_to_music.generate(
|
|
311
359
|
prompt=args.prompt, duration=args.duration, segments=segments
|
|
@@ -318,7 +366,8 @@ def cmd_video_to_music(client: Sonilo, args: argparse.Namespace) -> None:
|
|
|
318
366
|
fmt = args.format
|
|
319
367
|
multi = args.variants is not None and args.variants > 1
|
|
320
368
|
use_async = (
|
|
321
|
-
args.use_async or fmt != "m4a" or args.isolate_vocals or args.preserve_speech
|
|
369
|
+
args.use_async or fmt != "m4a" or args.isolate_vocals or args.preserve_speech
|
|
370
|
+
or multi or args.stems
|
|
322
371
|
)
|
|
323
372
|
out = _music_output(args, fmt)
|
|
324
373
|
segments = _segments(args)
|
|
@@ -333,8 +382,12 @@ def cmd_video_to_music(client: Sonilo, args: argparse.Namespace) -> None:
|
|
|
333
382
|
output_format=fmt if fmt != "m4a" else None,
|
|
334
383
|
variants_num=args.variants,
|
|
335
384
|
prompt_influence=args.prompt_influence,
|
|
385
|
+
stems=True if args.stems else None,
|
|
386
|
+
timeout=STEMS_WAIT_TIMEOUT if args.stems else DEFAULT_WAIT_TIMEOUT,
|
|
336
387
|
)
|
|
337
388
|
_save_music_variants(result, out)
|
|
389
|
+
if args.stems:
|
|
390
|
+
_save_music_stems(result, out)
|
|
338
391
|
else:
|
|
339
392
|
# prompt_influence rides the streaming path too — it is a generation
|
|
340
393
|
# parameter, not a finalize-time one, so it never forces async.
|
|
@@ -510,6 +563,51 @@ def cmd_audio_ducking(client: Sonilo, args: argparse.Namespace) -> None:
|
|
|
510
563
|
_wrote(path, path.stat().st_size)
|
|
511
564
|
|
|
512
565
|
|
|
566
|
+
def _analysis_payload(result: Any) -> dict:
|
|
567
|
+
"""Flatten a VideoAnalysisResult back into the API's own envelope shape.
|
|
568
|
+
|
|
569
|
+
Deliberately re-emits the wire format rather than dumping the dataclass:
|
|
570
|
+
this output is meant to be piped into another tool (or read by an agent),
|
|
571
|
+
and it should look identical to what GET /v1/tasks returned. None-valued
|
|
572
|
+
accounting fields are dropped so a brief stays readable.
|
|
573
|
+
"""
|
|
574
|
+
payload: dict = {
|
|
575
|
+
"task_id": result.task_id,
|
|
576
|
+
"status": result.status,
|
|
577
|
+
"segments": [
|
|
578
|
+
{"start": s.start, "end": s.end, "label": s.label, "prompt": s.prompt}
|
|
579
|
+
for s in result.segments
|
|
580
|
+
],
|
|
581
|
+
"variations": [{"prompt": v.prompt} for v in result.variations],
|
|
582
|
+
}
|
|
583
|
+
for key in ("variants_num", "duration_seconds", "cost"):
|
|
584
|
+
value = getattr(result, key, None)
|
|
585
|
+
if value is not None:
|
|
586
|
+
payload[key] = value
|
|
587
|
+
return payload
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def cmd_video_analysis(client: Sonilo, args: argparse.Namespace) -> None:
|
|
591
|
+
"""video-analysis is the one command that produces no media file. The
|
|
592
|
+
brief goes to stdout so it can be piped straight into the next tool;
|
|
593
|
+
--output is the opt-in for keeping a copy on disk."""
|
|
594
|
+
result = client.video_analysis.analyze(
|
|
595
|
+
video=args.video, video_url=args.video_url,
|
|
596
|
+
prompt=args.prompt, variants_num=args.variants,
|
|
597
|
+
timeout=args.timeout,
|
|
598
|
+
)
|
|
599
|
+
if not result.variations:
|
|
600
|
+
_fail("task succeeded but returned no creative brief")
|
|
601
|
+
payload = _analysis_payload(result)
|
|
602
|
+
if args.output is None:
|
|
603
|
+
_print_json(payload)
|
|
604
|
+
return
|
|
605
|
+
path = Path(args.output)
|
|
606
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
607
|
+
path.write_text(json.dumps(payload, indent=2) + "\n")
|
|
608
|
+
_wrote(path, path.stat().st_size)
|
|
609
|
+
|
|
610
|
+
|
|
513
611
|
# Matched to the dubbing backend's own ceiling: it polls its pipeline for up
|
|
514
612
|
# to 7200s (2 hours), so anything shorter abandons a job the user has already
|
|
515
613
|
# been charged for. The SDK's generic DEFAULT_WAIT_TIMEOUT of 600s is far too
|
|
@@ -630,6 +728,21 @@ def _add_variants(parser: argparse.ArgumentParser) -> None:
|
|
|
630
728
|
)
|
|
631
729
|
|
|
632
730
|
|
|
731
|
+
def _add_stems(parser: argparse.ArgumentParser) -> None:
|
|
732
|
+
# Only the two music-generation commands take this — the API accepts it
|
|
733
|
+
# nowhere else. Not to be confused with --stem on the sound commands,
|
|
734
|
+
# which saves layers those endpoints already return; --stems *requests*
|
|
735
|
+
# a separation the API would not otherwise run.
|
|
736
|
+
parser.add_argument(
|
|
737
|
+
"--stems", action="store_true",
|
|
738
|
+
help="Also split the generated music into drums/bass/vocals/other "
|
|
739
|
+
"stems, saved next to the output (output.drums.m4a, ...). Free "
|
|
740
|
+
"of charge. Forces async, and separation typically adds 2-6 "
|
|
741
|
+
"minutes to the wait. Streams that fail to separate are "
|
|
742
|
+
"reported on stderr; the rest are still saved.",
|
|
743
|
+
)
|
|
744
|
+
|
|
745
|
+
|
|
633
746
|
def build_parser() -> argparse.ArgumentParser:
|
|
634
747
|
parser = _Parser(prog="sonilo", description="Command-line interface for the Sonilo API")
|
|
635
748
|
parser.add_argument("--version", action="version", version=__version__)
|
|
@@ -683,6 +796,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
683
796
|
p_t2m.add_argument("--async", dest="use_async", action="store_true",
|
|
684
797
|
help="Submit and poll instead of streaming.")
|
|
685
798
|
_add_variants(p_t2m)
|
|
799
|
+
_add_stems(p_t2m)
|
|
686
800
|
p_t2m.set_defaults(func=cmd_text_to_music)
|
|
687
801
|
|
|
688
802
|
p_v2m = sub.add_parser("video-to-music", help="Generate music matched to a video")
|
|
@@ -705,6 +819,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
705
819
|
help="Submit and poll instead of streaming.")
|
|
706
820
|
_add_variants(p_v2m)
|
|
707
821
|
_add_prompt_influence(p_v2m)
|
|
822
|
+
_add_stems(p_v2m)
|
|
708
823
|
p_v2m.set_defaults(func=cmd_video_to_music)
|
|
709
824
|
|
|
710
825
|
p_t2s = sub.add_parser("text-to-sfx", help="Generate a sound effect from a text prompt")
|
|
@@ -858,6 +973,32 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
858
973
|
)
|
|
859
974
|
p_duck.set_defaults(func=cmd_audio_ducking)
|
|
860
975
|
|
|
976
|
+
p_va = sub.add_parser(
|
|
977
|
+
"video-analysis",
|
|
978
|
+
help="Analyze a video and print a creative brief for scoring it",
|
|
979
|
+
)
|
|
980
|
+
_add_global(p_va)
|
|
981
|
+
_add_video_source(p_va)
|
|
982
|
+
p_va.add_argument(
|
|
983
|
+
"--prompt", default=None,
|
|
984
|
+
help="Optional guidance for the analysis, e.g. 'focus on the chase'.",
|
|
985
|
+
)
|
|
986
|
+
p_va.add_argument(
|
|
987
|
+
"--variants", type=int, default=None,
|
|
988
|
+
help="How many independent briefs to author for the same video (1-5). "
|
|
989
|
+
"Billed per brief. Default: 1",
|
|
990
|
+
)
|
|
991
|
+
p_va.add_argument(
|
|
992
|
+
"--output", default=None,
|
|
993
|
+
help="Write the brief to this .json file instead of printing it to stdout.",
|
|
994
|
+
)
|
|
995
|
+
p_va.add_argument(
|
|
996
|
+
"--timeout", type=float, default=600.0,
|
|
997
|
+
help="Give up waiting after this many seconds. Default: 600. A timed-out "
|
|
998
|
+
"task may still finish — resume it with `sonilo tasks get <task-id>`.",
|
|
999
|
+
)
|
|
1000
|
+
p_va.set_defaults(func=cmd_video_analysis)
|
|
1001
|
+
|
|
861
1002
|
p_dub = sub.add_parser("dubbing", help="Dub a video into other languages")
|
|
862
1003
|
_add_global(p_dub)
|
|
863
1004
|
_add_video_source(p_dub)
|
|
@@ -1346,3 +1346,264 @@ def test_empty_env_var_falls_through_to_the_credential(monkeypatch):
|
|
|
1346
1346
|
)
|
|
1347
1347
|
main(["account"])
|
|
1348
1348
|
assert route.calls.last.request.headers["authorization"] == "Bearer sk-stored"
|
|
1349
|
+
|
|
1350
|
+
|
|
1351
|
+
ANALYSIS_ACK = {"task_id": "va1", "status": "processing"}
|
|
1352
|
+
ANALYSIS_BODY = {
|
|
1353
|
+
"task_id": "va1",
|
|
1354
|
+
"type": "video_analysis",
|
|
1355
|
+
"status": "succeeded",
|
|
1356
|
+
"variants_num": 2,
|
|
1357
|
+
"segments": [
|
|
1358
|
+
{"start": 0, "end": 12, "label": "intro", "prompt": "sparse piano"},
|
|
1359
|
+
],
|
|
1360
|
+
"variations": [
|
|
1361
|
+
{"prompt": "cinematic strings, 90bpm"},
|
|
1362
|
+
{"prompt": "lo-fi hip hop, warm keys"},
|
|
1363
|
+
],
|
|
1364
|
+
"duration_seconds": 30.0,
|
|
1365
|
+
"cost": 0.24,
|
|
1366
|
+
}
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
def _mock_analysis():
|
|
1370
|
+
respx.post(f"{BASE}/v1/video-analysis").mock(
|
|
1371
|
+
return_value=httpx.Response(202, json=ANALYSIS_ACK)
|
|
1372
|
+
)
|
|
1373
|
+
respx.get(f"{BASE}/v1/tasks/va1").mock(
|
|
1374
|
+
return_value=httpx.Response(200, json=ANALYSIS_BODY)
|
|
1375
|
+
)
|
|
1376
|
+
|
|
1377
|
+
|
|
1378
|
+
@respx.mock
|
|
1379
|
+
def test_video_analysis_prints_the_brief_as_json(capsys):
|
|
1380
|
+
"""The result is a brief, not a file: it goes to stdout so it can be
|
|
1381
|
+
piped, and nothing is written to disk unless --output says so."""
|
|
1382
|
+
_mock_analysis()
|
|
1383
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4"])
|
|
1384
|
+
out = json.loads(capsys.readouterr().out)
|
|
1385
|
+
assert out["task_id"] == "va1"
|
|
1386
|
+
assert out["segments"] == [
|
|
1387
|
+
{"start": 0, "end": 12, "label": "intro", "prompt": "sparse piano"}
|
|
1388
|
+
]
|
|
1389
|
+
assert [v["prompt"] for v in out["variations"]] == [
|
|
1390
|
+
"cinematic strings, 90bpm",
|
|
1391
|
+
"lo-fi hip hop, warm keys",
|
|
1392
|
+
]
|
|
1393
|
+
|
|
1394
|
+
|
|
1395
|
+
@respx.mock
|
|
1396
|
+
def test_video_analysis_sends_prompt_and_variants():
|
|
1397
|
+
_mock_analysis()
|
|
1398
|
+
route = respx.routes[0]
|
|
1399
|
+
run([
|
|
1400
|
+
"video-analysis",
|
|
1401
|
+
"--video-url", "https://x/v.mp4",
|
|
1402
|
+
"--prompt", "focus on the chase",
|
|
1403
|
+
"--variants", "2",
|
|
1404
|
+
])
|
|
1405
|
+
body = unquote_plus(route.calls.last.request.content.decode())
|
|
1406
|
+
assert "prompt=focus on the chase" in body
|
|
1407
|
+
assert "variants_num=2" in body
|
|
1408
|
+
|
|
1409
|
+
|
|
1410
|
+
@respx.mock
|
|
1411
|
+
def test_video_analysis_omits_unset_optionals():
|
|
1412
|
+
_mock_analysis()
|
|
1413
|
+
route = respx.routes[0]
|
|
1414
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4"])
|
|
1415
|
+
body = unquote_plus(route.calls.last.request.content.decode())
|
|
1416
|
+
assert "prompt" not in body
|
|
1417
|
+
assert "variants_num" not in body
|
|
1418
|
+
|
|
1419
|
+
|
|
1420
|
+
@respx.mock
|
|
1421
|
+
def test_video_analysis_output_writes_the_brief_to_a_file(tmp_path, capsys):
|
|
1422
|
+
_mock_analysis()
|
|
1423
|
+
out = tmp_path / "brief.json"
|
|
1424
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4", "--output", str(out)])
|
|
1425
|
+
written = json.loads(out.read_text())
|
|
1426
|
+
assert written["variations"][0]["prompt"] == "cinematic strings, 90bpm"
|
|
1427
|
+
# With --output the brief goes to the file, not to stdout.
|
|
1428
|
+
assert "Wrote" in capsys.readouterr().out
|
|
1429
|
+
|
|
1430
|
+
|
|
1431
|
+
def test_video_analysis_requires_a_video_source(capsys):
|
|
1432
|
+
with pytest.raises(SystemExit) as exc:
|
|
1433
|
+
main(["--api-key", "sk-test", "video-analysis"])
|
|
1434
|
+
assert exc.value.code == 1
|
|
1435
|
+
# Asserted on the message, not just the exit code: an unknown command
|
|
1436
|
+
# also exits 1, so the code alone would pass before the command exists.
|
|
1437
|
+
assert "--video" in capsys.readouterr().err
|
|
1438
|
+
|
|
1439
|
+
|
|
1440
|
+
# ---------- --stems (music stem separation) ----------
|
|
1441
|
+
|
|
1442
|
+
|
|
1443
|
+
def _stems_entry(i, base="https://r2.example.com"):
|
|
1444
|
+
return {
|
|
1445
|
+
"stream_index": i,
|
|
1446
|
+
"drums": {"url": f"{base}/s{i}.drums.m4a"},
|
|
1447
|
+
"bass": {"url": f"{base}/s{i}.bass.m4a"},
|
|
1448
|
+
"vocals": {"url": f"{base}/s{i}.vocals.m4a"},
|
|
1449
|
+
"other": {"url": f"{base}/s{i}.other.m4a"},
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1452
|
+
|
|
1453
|
+
def _mock_stem_downloads(*indices):
|
|
1454
|
+
for i in indices:
|
|
1455
|
+
for stem in ("drums", "bass", "vocals", "other"):
|
|
1456
|
+
respx.get(f"https://r2.example.com/s{i}.{stem}.m4a").mock(
|
|
1457
|
+
return_value=httpx.Response(200, content=f"{stem}{i}".encode())
|
|
1458
|
+
)
|
|
1459
|
+
|
|
1460
|
+
|
|
1461
|
+
@respx.mock
|
|
1462
|
+
def test_text_to_music_stems_forces_async_and_saves_stem_files(tmp_path, capsys):
|
|
1463
|
+
submit = respx.post(f"{BASE}/v1/text-to-music").mock(
|
|
1464
|
+
return_value=httpx.Response(200, json={"task_id": "ts1", "status": "processing"})
|
|
1465
|
+
)
|
|
1466
|
+
respx.get(f"{BASE}/v1/tasks/ts1").mock(
|
|
1467
|
+
return_value=httpx.Response(200, json={
|
|
1468
|
+
"task_id": "ts1", "type": "text_to_music", "status": "succeeded",
|
|
1469
|
+
"audio": [{"stream_index": 0, "url": "https://r2.example.com/ts1.m4a"}],
|
|
1470
|
+
"stems": [_stems_entry(0)],
|
|
1471
|
+
})
|
|
1472
|
+
)
|
|
1473
|
+
respx.get("https://r2.example.com/ts1.m4a").mock(
|
|
1474
|
+
return_value=httpx.Response(200, content=b"MIX")
|
|
1475
|
+
)
|
|
1476
|
+
_mock_stem_downloads(0)
|
|
1477
|
+
out = tmp_path / "take.m4a"
|
|
1478
|
+
run(["text-to-music", "--prompt", "lofi", "--duration", "10",
|
|
1479
|
+
"--stems", "--output", str(out)])
|
|
1480
|
+
# --stems must force the async submit-and-poll path, same as --format wav.
|
|
1481
|
+
assert submit.called
|
|
1482
|
+
body = submit.calls.last.request.content.decode()
|
|
1483
|
+
assert "stems=true" in body
|
|
1484
|
+
assert out.read_bytes() == b"MIX"
|
|
1485
|
+
for stem in ("drums", "bass", "vocals", "other"):
|
|
1486
|
+
assert (tmp_path / f"take.{stem}.m4a").read_bytes() == f"{stem}0".encode()
|
|
1487
|
+
# A clean separation warns about nothing.
|
|
1488
|
+
assert capsys.readouterr().err == ""
|
|
1489
|
+
|
|
1490
|
+
|
|
1491
|
+
@respx.mock
|
|
1492
|
+
def test_text_to_music_omits_stems_when_flag_unset(tmp_path):
|
|
1493
|
+
# Async for another reason (--format wav): the field must stay off the
|
|
1494
|
+
# wire entirely rather than pinning an explicit false.
|
|
1495
|
+
submit = respx.post(f"{BASE}/v1/text-to-music").mock(
|
|
1496
|
+
return_value=httpx.Response(200, json={"task_id": "tn1", "status": "processing"})
|
|
1497
|
+
)
|
|
1498
|
+
respx.get(f"{BASE}/v1/tasks/tn1").mock(
|
|
1499
|
+
return_value=httpx.Response(200, json={
|
|
1500
|
+
"task_id": "tn1", "type": "text_to_music", "status": "succeeded",
|
|
1501
|
+
"audio": [{"stream_index": 0, "url": "https://r2.example.com/tn1.wav"}],
|
|
1502
|
+
})
|
|
1503
|
+
)
|
|
1504
|
+
respx.get("https://r2.example.com/tn1.wav").mock(
|
|
1505
|
+
return_value=httpx.Response(200, content=b"RIF")
|
|
1506
|
+
)
|
|
1507
|
+
run(["text-to-music", "--prompt", "lofi", "--duration", "10",
|
|
1508
|
+
"--format", "wav", "--output", str(tmp_path / "t.wav")])
|
|
1509
|
+
assert b"stems" not in submit.calls.last.request.content
|
|
1510
|
+
|
|
1511
|
+
|
|
1512
|
+
@respx.mock
|
|
1513
|
+
def test_video_to_music_stems_partial_failure_warns_but_saves_the_rest(tmp_path, capsys):
|
|
1514
|
+
"""stems_error accompanies a PARTIAL stems list: the warning goes to
|
|
1515
|
+
stderr, the stems that DID come back are still written (named by their
|
|
1516
|
+
own stream_index, not list position), and the run is not an error —
|
|
1517
|
+
the music itself succeeded and separation is free."""
|
|
1518
|
+
submit = respx.post(f"{BASE}/v1/video-to-music").mock(
|
|
1519
|
+
return_value=httpx.Response(200, json={"task_id": "vs1", "status": "processing"})
|
|
1520
|
+
)
|
|
1521
|
+
respx.get(f"{BASE}/v1/tasks/vs1").mock(
|
|
1522
|
+
return_value=httpx.Response(200, json={
|
|
1523
|
+
"task_id": "vs1", "type": "video_to_music", "status": "succeeded",
|
|
1524
|
+
"variants_num": 2,
|
|
1525
|
+
"audio": [
|
|
1526
|
+
{"stream_index": 0, "url": "https://r2.example.com/vs1.0.m4a"},
|
|
1527
|
+
{"stream_index": 1, "url": "https://r2.example.com/vs1.1.m4a"},
|
|
1528
|
+
],
|
|
1529
|
+
# Only stream 1 separated; positional lookup would misfile these.
|
|
1530
|
+
"stems": [_stems_entry(1)],
|
|
1531
|
+
"stems_error": "stream 0 failed to separate",
|
|
1532
|
+
})
|
|
1533
|
+
)
|
|
1534
|
+
respx.get("https://r2.example.com/vs1.0.m4a").mock(
|
|
1535
|
+
return_value=httpx.Response(200, content=b"A0")
|
|
1536
|
+
)
|
|
1537
|
+
respx.get("https://r2.example.com/vs1.1.m4a").mock(
|
|
1538
|
+
return_value=httpx.Response(200, content=b"A1")
|
|
1539
|
+
)
|
|
1540
|
+
_mock_stem_downloads(1)
|
|
1541
|
+
out = tmp_path / "take.m4a"
|
|
1542
|
+
run(["video-to-music", "--video-url", "http://x/y.mp4",
|
|
1543
|
+
"--variants", "2", "--stems", "--output", str(out)])
|
|
1544
|
+
assert "stems=true" in submit.calls.last.request.content.decode()
|
|
1545
|
+
assert (tmp_path / "take.0.m4a").read_bytes() == b"A0"
|
|
1546
|
+
assert (tmp_path / "take.1.m4a").read_bytes() == b"A1"
|
|
1547
|
+
# Stem files carry stream 1's index; stream 0 has none.
|
|
1548
|
+
assert (tmp_path / "take.1.drums.m4a").read_bytes() == b"drums1"
|
|
1549
|
+
assert (tmp_path / "take.1.other.m4a").read_bytes() == b"other1"
|
|
1550
|
+
assert not (tmp_path / "take.0.drums.m4a").exists()
|
|
1551
|
+
err = capsys.readouterr().err
|
|
1552
|
+
assert "stream 0 failed to separate" in err
|
|
1553
|
+
|
|
1554
|
+
|
|
1555
|
+
def test_stems_passes_the_long_timeout_and_tri_state(tmp_path, monkeypatch):
|
|
1556
|
+
"""--stems switches the wait to STEMS_WAIT_TIMEOUT — the SDK's 600s
|
|
1557
|
+
default would abandon a separation that legitimately runs up to 30
|
|
1558
|
+
minutes past generation — while a stems-less async run keeps the
|
|
1559
|
+
default and sends no stems field at all."""
|
|
1560
|
+
from sonilo.resources.text_to_music import TextToMusic
|
|
1561
|
+
from sonilo.types import MusicAudioMedia, MusicResult
|
|
1562
|
+
|
|
1563
|
+
calls = []
|
|
1564
|
+
|
|
1565
|
+
def fake_generate_async(self, **kwargs):
|
|
1566
|
+
calls.append(kwargs)
|
|
1567
|
+
return MusicResult(
|
|
1568
|
+
task_id="t", status="succeeded",
|
|
1569
|
+
audio=[MusicAudioMedia(stream_index=0, url="https://r2.example.com/t.m4a")],
|
|
1570
|
+
stems=kwargs.get("stems") and [],
|
|
1571
|
+
stems_error="separation skipped" if kwargs.get("stems") else None,
|
|
1572
|
+
)
|
|
1573
|
+
|
|
1574
|
+
monkeypatch.setattr(TextToMusic, "generate_async", fake_generate_async)
|
|
1575
|
+
with respx.mock:
|
|
1576
|
+
respx.get("https://r2.example.com/t.m4a").mock(
|
|
1577
|
+
return_value=httpx.Response(200, content=b"A")
|
|
1578
|
+
)
|
|
1579
|
+
run(["text-to-music", "--prompt", "x", "--duration", "10",
|
|
1580
|
+
"--stems", "--output", str(tmp_path / "a.m4a")])
|
|
1581
|
+
run(["text-to-music", "--prompt", "x", "--duration", "10",
|
|
1582
|
+
"--format", "wav", "--output", str(tmp_path / "b.wav")])
|
|
1583
|
+
assert calls[0]["stems"] is True
|
|
1584
|
+
assert calls[0]["timeout"] == 2400.0
|
|
1585
|
+
assert calls[1]["stems"] is None
|
|
1586
|
+
assert calls[1]["timeout"] == 600.0
|
|
1587
|
+
|
|
1588
|
+
|
|
1589
|
+
def test_stems_warns_when_none_come_back(tmp_path, monkeypatch, capsys):
|
|
1590
|
+
from sonilo.resources.text_to_music import TextToMusic
|
|
1591
|
+
from sonilo.types import MusicAudioMedia, MusicResult
|
|
1592
|
+
|
|
1593
|
+
def fake_generate_async(self, **kwargs):
|
|
1594
|
+
return MusicResult(
|
|
1595
|
+
task_id="t", status="succeeded",
|
|
1596
|
+
audio=[MusicAudioMedia(stream_index=0, url="https://r2.example.com/t.m4a")],
|
|
1597
|
+
)
|
|
1598
|
+
|
|
1599
|
+
monkeypatch.setattr(TextToMusic, "generate_async", fake_generate_async)
|
|
1600
|
+
with respx.mock:
|
|
1601
|
+
respx.get("https://r2.example.com/t.m4a").mock(
|
|
1602
|
+
return_value=httpx.Response(200, content=b"A")
|
|
1603
|
+
)
|
|
1604
|
+
run(["text-to-music", "--prompt", "x", "--duration", "10",
|
|
1605
|
+
"--stems", "--output", str(tmp_path / "a.m4a")])
|
|
1606
|
+
# The main output was written and the absence of stems is a warning,
|
|
1607
|
+
# never a failure exit.
|
|
1608
|
+
assert (tmp_path / "a.m4a").read_bytes() == b"A"
|
|
1609
|
+
assert "no stems" in capsys.readouterr().err
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|