sonilo-cli 0.11.0__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/PKG-INFO +43 -3
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/README.md +41 -1
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/pyproject.toml +2 -2
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/src/sonilo_cli/__init__.py +1 -1
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/src/sonilo_cli/__main__.py +139 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/test_cli.py +173 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/.gitignore +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/LICENSE +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/src/sonilo_cli/credentials.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/src/sonilo_cli/login.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/__init__.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/conftest.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/test_context7.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/test_credentials.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/test_login.py +0 -0
- {sonilo_cli-0.11.0 → sonilo_cli-0.13.0}/tests/test_smoke.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: sonilo-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Command-line interface for the Sonilo API: generate music and sound effects from text or video
|
|
5
5
|
Project-URL: Repository, https://github.com/sonilo-ai/sonilo-python
|
|
6
6
|
Author: Sonilo AI
|
|
@@ -8,7 +8,7 @@ License-Expression: MIT
|
|
|
8
8
|
License-File: LICENSE
|
|
9
9
|
Keywords: ai,cli,music,sfx,sonilo,text-to-music,video-to-music
|
|
10
10
|
Requires-Python: >=3.9
|
|
11
|
-
Requires-Dist: sonilo<0.
|
|
11
|
+
Requires-Dist: sonilo<0.15,>=0.14.0
|
|
12
12
|
Provides-Extra: dev
|
|
13
13
|
Requires-Dist: pytest>=8; extra == 'dev'
|
|
14
14
|
Requires-Dist: respx>=0.21; extra == 'dev'
|
|
@@ -95,6 +95,11 @@ production sign-in coexist without overwriting each other.
|
|
|
95
95
|
sonilo video-to-video-music --video clip.mp4 --prompt "tense synths" --output scored.mp4
|
|
96
96
|
sonilo video-to-video-sfx --video clip.mp4 --segments @segments.json --output scored.mp4
|
|
97
97
|
sonilo video-to-video-sound --video clip.mp4 --music-prompt "tense synths"
|
|
98
|
+
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
99
|
+
# ducks the existing music bed under the voice; a video voice comes back
|
|
100
|
+
# as a new .mp4 with the ducked mix muxed in
|
|
101
|
+
sonilo video-analysis --video clip.mp4 --variants 2
|
|
102
|
+
# prints a creative brief as JSON; generates nothing
|
|
98
103
|
sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
|
|
99
104
|
# writes dubbed.es.mp4 and dubbed.fr.mp4
|
|
100
105
|
sonilo tasks get <task-id>
|
|
@@ -220,6 +225,41 @@ they differ only in what comes back: `video-to-sound` writes the mixed **audio**
|
|
|
220
225
|
or ducking altered the music bed.
|
|
221
226
|
- Both also take `--variants` — see [Variants](#variants) above.
|
|
222
227
|
|
|
228
|
+
### Audio ducking
|
|
229
|
+
|
|
230
|
+
`audio-ducking` mixes an **existing** music bed under an **existing** voice track — nothing is
|
|
231
|
+
generated, so reach for it when the music is fixed or external. (When the music is being generated
|
|
232
|
+
for the same clip anyway, `video-to-sound` or `video-to-music` duck internally as part of that one
|
|
233
|
+
call instead.)
|
|
234
|
+
|
|
235
|
+
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
236
|
+
|
|
237
|
+
- Exactly one of `--voice` / `--voice-url` and one of `--music` / `--music-url`; a local file and a
|
|
238
|
+
URL mix freely across the two inputs.
|
|
239
|
+
- The **voice** may be audio or video: a video's own audio track becomes the voice, and the ducked
|
|
240
|
+
mix is muxed back into a new video, so the result is a `.mp4` instead of a `.wav`. The default
|
|
241
|
+
`--output` name follows what came back (`output.wav` or `output.mp4`).
|
|
242
|
+
- The **music** must be audio (`wav, mp3, m4a, aac, ogg, flac`). The API does not detect a video
|
|
243
|
+
there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
|
|
244
|
+
- Each input is capped at 360 seconds server-side.
|
|
245
|
+
|
|
246
|
+
### Video analysis
|
|
247
|
+
|
|
248
|
+
`video-analysis` analyzes a video and prints a **creative brief** for scoring it. It is the one
|
|
249
|
+
command that produces no media file — nothing is generated:
|
|
250
|
+
|
|
251
|
+
sonilo video-analysis --video clip.mp4 --prompt "focus on the chase" --variants 2
|
|
252
|
+
|
|
253
|
+
- The brief goes to **stdout as JSON** so it can be piped into another tool: `segments` (a
|
|
254
|
+
time-aligned section plan) and `variations` (one ready-to-use generation prompt each). Pass
|
|
255
|
+
`--output brief.json` to write it to a file instead.
|
|
256
|
+
- `--variants` is 1-5 (default 1) and is **billed per brief**.
|
|
257
|
+
- Source videos may be at most 600 seconds long, and billing has a 10-second floor.
|
|
258
|
+
- Feed a variation's prompt straight into the next command:
|
|
259
|
+
|
|
260
|
+
sonilo video-analysis --video clip.mp4 --output brief.json
|
|
261
|
+
sonilo video-to-music --video clip.mp4 --prompt "$(jq -r '.variations[0].prompt' brief.json)"
|
|
262
|
+
|
|
223
263
|
### Dubbing
|
|
224
264
|
|
|
225
265
|
`dubbing` dubs a video into one or more target languages in a single async call:
|
|
@@ -246,7 +286,7 @@ required:
|
|
|
246
286
|
|
|
247
287
|
| Free runs | Endpoints |
|
|
248
288
|
| --- | --- |
|
|
249
|
-
| 2 each | text-to-music, text-to-sfx, audio-ducking |
|
|
289
|
+
| 2 each | text-to-music, text-to-sfx, audio-ducking, video-analysis |
|
|
250
290
|
| 1 each | video-to-music, video-to-sfx, video-to-video-music, video-to-video-sfx, video-to-sound, video-to-video-sound |
|
|
251
291
|
| 0 | dubbing |
|
|
252
292
|
|
|
@@ -79,6 +79,11 @@ production sign-in coexist without overwriting each other.
|
|
|
79
79
|
sonilo video-to-video-music --video clip.mp4 --prompt "tense synths" --output scored.mp4
|
|
80
80
|
sonilo video-to-video-sfx --video clip.mp4 --segments @segments.json --output scored.mp4
|
|
81
81
|
sonilo video-to-video-sound --video clip.mp4 --music-prompt "tense synths"
|
|
82
|
+
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
83
|
+
# ducks the existing music bed under the voice; a video voice comes back
|
|
84
|
+
# as a new .mp4 with the ducked mix muxed in
|
|
85
|
+
sonilo video-analysis --video clip.mp4 --variants 2
|
|
86
|
+
# prints a creative brief as JSON; generates nothing
|
|
82
87
|
sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
|
|
83
88
|
# writes dubbed.es.mp4 and dubbed.fr.mp4
|
|
84
89
|
sonilo tasks get <task-id>
|
|
@@ -204,6 +209,41 @@ they differ only in what comes back: `video-to-sound` writes the mixed **audio**
|
|
|
204
209
|
or ducking altered the music bed.
|
|
205
210
|
- Both also take `--variants` — see [Variants](#variants) above.
|
|
206
211
|
|
|
212
|
+
### Audio ducking
|
|
213
|
+
|
|
214
|
+
`audio-ducking` mixes an **existing** music bed under an **existing** voice track — nothing is
|
|
215
|
+
generated, so reach for it when the music is fixed or external. (When the music is being generated
|
|
216
|
+
for the same clip anyway, `video-to-sound` or `video-to-music` duck internally as part of that one
|
|
217
|
+
call instead.)
|
|
218
|
+
|
|
219
|
+
sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
|
|
220
|
+
|
|
221
|
+
- Exactly one of `--voice` / `--voice-url` and one of `--music` / `--music-url`; a local file and a
|
|
222
|
+
URL mix freely across the two inputs.
|
|
223
|
+
- The **voice** may be audio or video: a video's own audio track becomes the voice, and the ducked
|
|
224
|
+
mix is muxed back into a new video, so the result is a `.mp4` instead of a `.wav`. The default
|
|
225
|
+
`--output` name follows what came back (`output.wav` or `output.mp4`).
|
|
226
|
+
- The **music** must be audio (`wav, mp3, m4a, aac, ogg, flac`). The API does not detect a video
|
|
227
|
+
there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
|
|
228
|
+
- Each input is capped at 360 seconds server-side.
|
|
229
|
+
|
|
230
|
+
### Video analysis
|
|
231
|
+
|
|
232
|
+
`video-analysis` analyzes a video and prints a **creative brief** for scoring it. It is the one
|
|
233
|
+
command that produces no media file — nothing is generated:
|
|
234
|
+
|
|
235
|
+
sonilo video-analysis --video clip.mp4 --prompt "focus on the chase" --variants 2
|
|
236
|
+
|
|
237
|
+
- The brief goes to **stdout as JSON** so it can be piped into another tool: `segments` (a
|
|
238
|
+
time-aligned section plan) and `variations` (one ready-to-use generation prompt each). Pass
|
|
239
|
+
`--output brief.json` to write it to a file instead.
|
|
240
|
+
- `--variants` is 1-5 (default 1) and is **billed per brief**.
|
|
241
|
+
- Source videos may be at most 600 seconds long, and billing has a 10-second floor.
|
|
242
|
+
- Feed a variation's prompt straight into the next command:
|
|
243
|
+
|
|
244
|
+
sonilo video-analysis --video clip.mp4 --output brief.json
|
|
245
|
+
sonilo video-to-music --video clip.mp4 --prompt "$(jq -r '.variations[0].prompt' brief.json)"
|
|
246
|
+
|
|
207
247
|
### Dubbing
|
|
208
248
|
|
|
209
249
|
`dubbing` dubs a video into one or more target languages in a single async call:
|
|
@@ -230,7 +270,7 @@ required:
|
|
|
230
270
|
|
|
231
271
|
| Free runs | Endpoints |
|
|
232
272
|
| --- | --- |
|
|
233
|
-
| 2 each | text-to-music, text-to-sfx, audio-ducking |
|
|
273
|
+
| 2 each | text-to-music, text-to-sfx, audio-ducking, video-analysis |
|
|
234
274
|
| 1 each | video-to-music, video-to-sfx, video-to-video-music, video-to-video-sfx, video-to-sound, video-to-video-sound |
|
|
235
275
|
| 0 | dubbing |
|
|
236
276
|
|
|
@@ -4,13 +4,13 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "sonilo-cli"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.13.0"
|
|
8
8
|
description = "Command-line interface for the Sonilo API: generate music and sound effects from text or video"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
11
11
|
requires-python = ">=3.9"
|
|
12
12
|
authors = [{ name = "Sonilo AI" }]
|
|
13
|
-
dependencies = ["sonilo>=0.
|
|
13
|
+
dependencies = ["sonilo>=0.14.0,<0.15"]
|
|
14
14
|
keywords = ["sonilo", "cli", "music", "sfx", "text-to-music", "video-to-music", "ai"]
|
|
15
15
|
|
|
16
16
|
[project.urls]
|
|
@@ -474,6 +474,87 @@ def cmd_video_to_video_sfx(client: Sonilo, args: argparse.Namespace) -> None:
|
|
|
474
474
|
)
|
|
475
475
|
|
|
476
476
|
|
|
477
|
+
# The music-bed extensions audio-ducking accepts for a local --music file —
|
|
478
|
+
# the MCP server's _AUDIO_EXTS, kept identical so the two surfaces accept and
|
|
479
|
+
# reject the same files. A whitelist (not a video-extension blacklist) because
|
|
480
|
+
# the failure it guards against is silent: the API never probes the music
|
|
481
|
+
# input for a video stream, so a video sent there is mishandled without an
|
|
482
|
+
# error. The voice input needs no such guard — it may legitimately be audio
|
|
483
|
+
# or video, and the API probes it.
|
|
484
|
+
_MUSIC_AUDIO_EXTS = (".wav", ".mp3", ".m4a", ".aac", ".ogg", ".flac")
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def cmd_audio_ducking(client: Sonilo, args: argparse.Namespace) -> None:
|
|
488
|
+
if args.music is not None and Path(args.music).suffix.lower() not in _MUSIC_AUDIO_EXTS:
|
|
489
|
+
_fail(
|
|
490
|
+
f"--music must be an audio file ({', '.join(_MUSIC_AUDIO_EXTS)}) — "
|
|
491
|
+
"the API does not detect a video here and would mishandle it. "
|
|
492
|
+
"The voice input is the one that may be a video."
|
|
493
|
+
)
|
|
494
|
+
result = client.audio_ducking.generate(
|
|
495
|
+
voice=args.voice,
|
|
496
|
+
voice_url=args.voice_url,
|
|
497
|
+
music=args.music,
|
|
498
|
+
music_url=args.music_url,
|
|
499
|
+
)
|
|
500
|
+
# Default output name follows what actually came back: a .wav, or a .mp4
|
|
501
|
+
# (ducked mix re-muxed in) when the voice input was a video. An explicit
|
|
502
|
+
# --output is used verbatim, same as the video-out commands.
|
|
503
|
+
out = args.output
|
|
504
|
+
if out is None:
|
|
505
|
+
ext = Path(urlparse(result.output_url or "").path).suffix
|
|
506
|
+
if not ext:
|
|
507
|
+
ext = ".mp4" if result.output_type == "video" else ".wav"
|
|
508
|
+
out = f"output{ext}"
|
|
509
|
+
path = result.save(out)
|
|
510
|
+
_wrote(path, path.stat().st_size)
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def _analysis_payload(result: Any) -> dict:
|
|
514
|
+
"""Flatten a VideoAnalysisResult back into the API's own envelope shape.
|
|
515
|
+
|
|
516
|
+
Deliberately re-emits the wire format rather than dumping the dataclass:
|
|
517
|
+
this output is meant to be piped into another tool (or read by an agent),
|
|
518
|
+
and it should look identical to what GET /v1/tasks returned. None-valued
|
|
519
|
+
accounting fields are dropped so a brief stays readable.
|
|
520
|
+
"""
|
|
521
|
+
payload: dict = {
|
|
522
|
+
"task_id": result.task_id,
|
|
523
|
+
"status": result.status,
|
|
524
|
+
"segments": [
|
|
525
|
+
{"start": s.start, "end": s.end, "label": s.label, "prompt": s.prompt}
|
|
526
|
+
for s in result.segments
|
|
527
|
+
],
|
|
528
|
+
"variations": [{"prompt": v.prompt} for v in result.variations],
|
|
529
|
+
}
|
|
530
|
+
for key in ("variants_num", "duration_seconds", "cost"):
|
|
531
|
+
value = getattr(result, key, None)
|
|
532
|
+
if value is not None:
|
|
533
|
+
payload[key] = value
|
|
534
|
+
return payload
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def cmd_video_analysis(client: Sonilo, args: argparse.Namespace) -> None:
|
|
538
|
+
"""video-analysis is the one command that produces no media file. The
|
|
539
|
+
brief goes to stdout so it can be piped straight into the next tool;
|
|
540
|
+
--output is the opt-in for keeping a copy on disk."""
|
|
541
|
+
result = client.video_analysis.analyze(
|
|
542
|
+
video=args.video, video_url=args.video_url,
|
|
543
|
+
prompt=args.prompt, variants_num=args.variants,
|
|
544
|
+
timeout=args.timeout,
|
|
545
|
+
)
|
|
546
|
+
if not result.variations:
|
|
547
|
+
_fail("task succeeded but returned no creative brief")
|
|
548
|
+
payload = _analysis_payload(result)
|
|
549
|
+
if args.output is None:
|
|
550
|
+
_print_json(payload)
|
|
551
|
+
return
|
|
552
|
+
path = Path(args.output)
|
|
553
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
554
|
+
path.write_text(json.dumps(payload, indent=2) + "\n")
|
|
555
|
+
_wrote(path, path.stat().st_size)
|
|
556
|
+
|
|
557
|
+
|
|
477
558
|
# Matched to the dubbing backend's own ceiling: it polls its pipeline for up
|
|
478
559
|
# to 7200s (2 hours), so anything shorter abandons a job the user has already
|
|
479
560
|
# been charged for. The SDK's generic DEFAULT_WAIT_TIMEOUT of 600s is far too
|
|
@@ -790,6 +871,64 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
790
871
|
_add_variants(p_v2vsd)
|
|
791
872
|
p_v2vsd.set_defaults(func=cmd_video_to_video_sound)
|
|
792
873
|
|
|
874
|
+
p_duck = sub.add_parser(
|
|
875
|
+
"audio-ducking", help="Duck an existing music bed under a voice track"
|
|
876
|
+
)
|
|
877
|
+
_add_global(p_duck)
|
|
878
|
+
voice_group = p_duck.add_mutually_exclusive_group(required=True)
|
|
879
|
+
voice_group.add_argument(
|
|
880
|
+
"--voice", default=None,
|
|
881
|
+
help="Local voice track. May be audio or video: a video's own audio "
|
|
882
|
+
"track becomes the voice, and the ducked mix is muxed back into "
|
|
883
|
+
"a new video.",
|
|
884
|
+
)
|
|
885
|
+
voice_group.add_argument(
|
|
886
|
+
"--voice-url", dest="voice_url", default=None,
|
|
887
|
+
help="Remote voice audio/video URL.",
|
|
888
|
+
)
|
|
889
|
+
music_group = p_duck.add_mutually_exclusive_group(required=True)
|
|
890
|
+
music_group.add_argument(
|
|
891
|
+
"--music", default=None,
|
|
892
|
+
help="Local music bed. Audio only (wav, mp3, m4a, aac, ogg, flac) — "
|
|
893
|
+
"the API does not detect a video here, so the CLI rejects one.",
|
|
894
|
+
)
|
|
895
|
+
music_group.add_argument(
|
|
896
|
+
"--music-url", dest="music_url", default=None,
|
|
897
|
+
help="Remote music audio URL.",
|
|
898
|
+
)
|
|
899
|
+
p_duck.add_argument(
|
|
900
|
+
"--output", default=None,
|
|
901
|
+
help="Where to save the result. Default: output.wav, or output.mp4 "
|
|
902
|
+
"when the voice input was a video.",
|
|
903
|
+
)
|
|
904
|
+
p_duck.set_defaults(func=cmd_audio_ducking)
|
|
905
|
+
|
|
906
|
+
p_va = sub.add_parser(
|
|
907
|
+
"video-analysis",
|
|
908
|
+
help="Analyze a video and print a creative brief for scoring it",
|
|
909
|
+
)
|
|
910
|
+
_add_global(p_va)
|
|
911
|
+
_add_video_source(p_va)
|
|
912
|
+
p_va.add_argument(
|
|
913
|
+
"--prompt", default=None,
|
|
914
|
+
help="Optional guidance for the analysis, e.g. 'focus on the chase'.",
|
|
915
|
+
)
|
|
916
|
+
p_va.add_argument(
|
|
917
|
+
"--variants", type=int, default=None,
|
|
918
|
+
help="How many independent briefs to author for the same video (1-5). "
|
|
919
|
+
"Billed per brief. Default: 1",
|
|
920
|
+
)
|
|
921
|
+
p_va.add_argument(
|
|
922
|
+
"--output", default=None,
|
|
923
|
+
help="Write the brief to this .json file instead of printing it to stdout.",
|
|
924
|
+
)
|
|
925
|
+
p_va.add_argument(
|
|
926
|
+
"--timeout", type=float, default=600.0,
|
|
927
|
+
help="Give up waiting after this many seconds. Default: 600. A timed-out "
|
|
928
|
+
"task may still finish — resume it with `sonilo tasks get <task-id>`.",
|
|
929
|
+
)
|
|
930
|
+
p_va.set_defaults(func=cmd_video_analysis)
|
|
931
|
+
|
|
793
932
|
p_dub = sub.add_parser("dubbing", help="Dub a video into other languages")
|
|
794
933
|
_add_global(p_dub)
|
|
795
934
|
_add_video_source(p_dub)
|
|
@@ -523,6 +523,90 @@ def test_video_to_video_sound_requires_a_video_source():
|
|
|
523
523
|
assert exc.value.code == 1
|
|
524
524
|
|
|
525
525
|
|
|
526
|
+
# --- audio-ducking --------------------------------------------------------
|
|
527
|
+
#
|
|
528
|
+
# Same flat output_url envelope as video-to-sound, no stems. The result is a
|
|
529
|
+
# .wav, or a .mp4 (ducked mix re-muxed in) when the voice input was a video.
|
|
530
|
+
|
|
531
|
+
DUCKING_SUCCESS_BODY = {
|
|
532
|
+
"task_id": "ad1",
|
|
533
|
+
"type": "audio_ducking",
|
|
534
|
+
"status": "succeeded",
|
|
535
|
+
"output_url": "https://r2.example.com/ducked.wav",
|
|
536
|
+
"output_type": "audio",
|
|
537
|
+
"output_bytes": 5,
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
@respx.mock
|
|
542
|
+
def test_audio_ducking_saves_output(tmp_path):
|
|
543
|
+
route = respx.post(f"{BASE}/v1/audio-ducking").mock(
|
|
544
|
+
return_value=httpx.Response(202, json={"task_id": "ad1", "status": "processing"})
|
|
545
|
+
)
|
|
546
|
+
respx.get(f"{BASE}/v1/tasks/ad1").mock(
|
|
547
|
+
return_value=httpx.Response(200, json=DUCKING_SUCCESS_BODY)
|
|
548
|
+
)
|
|
549
|
+
respx.get("https://r2.example.com/ducked.wav").mock(
|
|
550
|
+
return_value=httpx.Response(200, content=b"DUCKED")
|
|
551
|
+
)
|
|
552
|
+
out = tmp_path / "mix.wav"
|
|
553
|
+
run(["audio-ducking", "--voice-url", "https://x/v.wav",
|
|
554
|
+
"--music-url", "https://x/m.wav", "--output", str(out)])
|
|
555
|
+
assert route.called
|
|
556
|
+
content = route.calls[0].request.content
|
|
557
|
+
assert b"voice_url" in content and b"music_url" in content
|
|
558
|
+
assert out.read_bytes() == b"DUCKED"
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
@respx.mock
|
|
562
|
+
def test_audio_ducking_default_output_follows_result_type(tmp_path, monkeypatch):
|
|
563
|
+
# A video voice input comes back as an mp4; the default output name must
|
|
564
|
+
# follow the result, not assume audio.
|
|
565
|
+
monkeypatch.chdir(tmp_path)
|
|
566
|
+
respx.post(f"{BASE}/v1/audio-ducking").mock(
|
|
567
|
+
return_value=httpx.Response(202, json={"task_id": "ad2", "status": "processing"})
|
|
568
|
+
)
|
|
569
|
+
respx.get(f"{BASE}/v1/tasks/ad2").mock(
|
|
570
|
+
return_value=httpx.Response(200, json={
|
|
571
|
+
**DUCKING_SUCCESS_BODY, "task_id": "ad2",
|
|
572
|
+
"output_url": "https://r2.example.com/ducked.mp4",
|
|
573
|
+
"output_type": "video",
|
|
574
|
+
})
|
|
575
|
+
)
|
|
576
|
+
respx.get("https://r2.example.com/ducked.mp4").mock(
|
|
577
|
+
return_value=httpx.Response(200, content=b"MP4DATA")
|
|
578
|
+
)
|
|
579
|
+
run(["audio-ducking", "--voice-url", "https://x/v.mp4",
|
|
580
|
+
"--music-url", "https://x/m.wav"])
|
|
581
|
+
assert (tmp_path / "output.mp4").read_bytes() == b"MP4DATA"
|
|
582
|
+
|
|
583
|
+
|
|
584
|
+
def test_audio_ducking_requires_a_voice_source():
|
|
585
|
+
with pytest.raises(SystemExit) as exc:
|
|
586
|
+
run(["audio-ducking", "--music-url", "https://x/m.wav"])
|
|
587
|
+
assert exc.value.code == 1
|
|
588
|
+
|
|
589
|
+
|
|
590
|
+
def test_audio_ducking_requires_a_music_source():
|
|
591
|
+
with pytest.raises(SystemExit) as exc:
|
|
592
|
+
run(["audio-ducking", "--voice-url", "https://x/v.wav"])
|
|
593
|
+
assert exc.value.code == 1
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
@respx.mock
|
|
597
|
+
def test_audio_ducking_rejects_a_video_music_file(tmp_path, capsys):
|
|
598
|
+
# The API never probes the music input for a video stream, so a video
|
|
599
|
+
# there is mishandled silently — the CLI must refuse before any request.
|
|
600
|
+
music = tmp_path / "background.mp4"
|
|
601
|
+
music.write_bytes(b"m")
|
|
602
|
+
with pytest.raises(SystemExit) as exc:
|
|
603
|
+
run(["audio-ducking", "--voice-url", "https://x/v.wav",
|
|
604
|
+
"--music", str(music)])
|
|
605
|
+
assert exc.value.code == 1
|
|
606
|
+
assert "audio" in capsys.readouterr().err
|
|
607
|
+
assert not respx.calls
|
|
608
|
+
|
|
609
|
+
|
|
526
610
|
def test_cli_identifies_itself_not_the_sdk():
|
|
527
611
|
"""CLI traffic must be separable from direct SDK use in analytics."""
|
|
528
612
|
import sonilo_cli
|
|
@@ -1262,3 +1346,92 @@ def test_empty_env_var_falls_through_to_the_credential(monkeypatch):
|
|
|
1262
1346
|
)
|
|
1263
1347
|
main(["account"])
|
|
1264
1348
|
assert route.calls.last.request.headers["authorization"] == "Bearer sk-stored"
|
|
1349
|
+
|
|
1350
|
+
|
|
1351
|
+
ANALYSIS_ACK = {"task_id": "va1", "status": "processing"}
|
|
1352
|
+
ANALYSIS_BODY = {
|
|
1353
|
+
"task_id": "va1",
|
|
1354
|
+
"type": "video_analysis",
|
|
1355
|
+
"status": "succeeded",
|
|
1356
|
+
"variants_num": 2,
|
|
1357
|
+
"segments": [
|
|
1358
|
+
{"start": 0, "end": 12, "label": "intro", "prompt": "sparse piano"},
|
|
1359
|
+
],
|
|
1360
|
+
"variations": [
|
|
1361
|
+
{"prompt": "cinematic strings, 90bpm"},
|
|
1362
|
+
{"prompt": "lo-fi hip hop, warm keys"},
|
|
1363
|
+
],
|
|
1364
|
+
"duration_seconds": 30.0,
|
|
1365
|
+
"cost": 0.24,
|
|
1366
|
+
}
|
|
1367
|
+
|
|
1368
|
+
|
|
1369
|
+
def _mock_analysis():
|
|
1370
|
+
respx.post(f"{BASE}/v1/video-analysis").mock(
|
|
1371
|
+
return_value=httpx.Response(202, json=ANALYSIS_ACK)
|
|
1372
|
+
)
|
|
1373
|
+
respx.get(f"{BASE}/v1/tasks/va1").mock(
|
|
1374
|
+
return_value=httpx.Response(200, json=ANALYSIS_BODY)
|
|
1375
|
+
)
|
|
1376
|
+
|
|
1377
|
+
|
|
1378
|
+
@respx.mock
|
|
1379
|
+
def test_video_analysis_prints_the_brief_as_json(capsys):
|
|
1380
|
+
"""The result is a brief, not a file: it goes to stdout so it can be
|
|
1381
|
+
piped, and nothing is written to disk unless --output says so."""
|
|
1382
|
+
_mock_analysis()
|
|
1383
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4"])
|
|
1384
|
+
out = json.loads(capsys.readouterr().out)
|
|
1385
|
+
assert out["task_id"] == "va1"
|
|
1386
|
+
assert out["segments"] == [
|
|
1387
|
+
{"start": 0, "end": 12, "label": "intro", "prompt": "sparse piano"}
|
|
1388
|
+
]
|
|
1389
|
+
assert [v["prompt"] for v in out["variations"]] == [
|
|
1390
|
+
"cinematic strings, 90bpm",
|
|
1391
|
+
"lo-fi hip hop, warm keys",
|
|
1392
|
+
]
|
|
1393
|
+
|
|
1394
|
+
|
|
1395
|
+
@respx.mock
|
|
1396
|
+
def test_video_analysis_sends_prompt_and_variants():
|
|
1397
|
+
_mock_analysis()
|
|
1398
|
+
route = respx.routes[0]
|
|
1399
|
+
run([
|
|
1400
|
+
"video-analysis",
|
|
1401
|
+
"--video-url", "https://x/v.mp4",
|
|
1402
|
+
"--prompt", "focus on the chase",
|
|
1403
|
+
"--variants", "2",
|
|
1404
|
+
])
|
|
1405
|
+
body = unquote_plus(route.calls.last.request.content.decode())
|
|
1406
|
+
assert "prompt=focus on the chase" in body
|
|
1407
|
+
assert "variants_num=2" in body
|
|
1408
|
+
|
|
1409
|
+
|
|
1410
|
+
@respx.mock
|
|
1411
|
+
def test_video_analysis_omits_unset_optionals():
|
|
1412
|
+
_mock_analysis()
|
|
1413
|
+
route = respx.routes[0]
|
|
1414
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4"])
|
|
1415
|
+
body = unquote_plus(route.calls.last.request.content.decode())
|
|
1416
|
+
assert "prompt" not in body
|
|
1417
|
+
assert "variants_num" not in body
|
|
1418
|
+
|
|
1419
|
+
|
|
1420
|
+
@respx.mock
|
|
1421
|
+
def test_video_analysis_output_writes_the_brief_to_a_file(tmp_path, capsys):
|
|
1422
|
+
_mock_analysis()
|
|
1423
|
+
out = tmp_path / "brief.json"
|
|
1424
|
+
run(["video-analysis", "--video-url", "https://x/v.mp4", "--output", str(out)])
|
|
1425
|
+
written = json.loads(out.read_text())
|
|
1426
|
+
assert written["variations"][0]["prompt"] == "cinematic strings, 90bpm"
|
|
1427
|
+
# With --output the brief goes to the file, not to stdout.
|
|
1428
|
+
assert "Wrote" in capsys.readouterr().out
|
|
1429
|
+
|
|
1430
|
+
|
|
1431
|
+
def test_video_analysis_requires_a_video_source(capsys):
|
|
1432
|
+
with pytest.raises(SystemExit) as exc:
|
|
1433
|
+
main(["--api-key", "sk-test", "video-analysis"])
|
|
1434
|
+
assert exc.value.code == 1
|
|
1435
|
+
# Asserted on the message, not just the exit code: an unknown command
|
|
1436
|
+
# also exits 1, so the code alone would pass before the command exists.
|
|
1437
|
+
assert "--video" in capsys.readouterr().err
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|