sonilo-cli 0.11.0__tar.gz → 0.12.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,19 @@
1
+ Metadata-Version: 2.5
2
+ Name: sonilo-cli
3
+ Version: 0.12.0
4
+ Summary: Command-line interface for the Sonilo API: generate music and sound effects from text or video
5
+ Project-URL: Repository, https://github.com/sonilo-ai/sonilo-python
6
+ Author: Sonilo AI
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: ai,cli,music,sfx,sonilo,text-to-music,video-to-music
10
+ Requires-Python: >=3.9
11
+ Requires-Dist: sonilo<0.14,>=0.13.0
12
+ Provides-Extra: dev
13
+ Requires-Dist: pytest>=8; extra == 'dev'
14
+ Requires-Dist: respx>=0.21; extra == 'dev'
15
+ Description-Content-Type: text/markdown
16
+
1
17
  # sonilo-cli
2
18
 
3
19
  Command-line interface for the [Sonilo API](https://github.com/sonilo-ai/sonilo-python) — generate music and sound effects from text or video.
@@ -79,6 +95,9 @@ production sign-in coexist without overwriting each other.
79
95
  sonilo video-to-video-music --video clip.mp4 --prompt "tense synths" --output scored.mp4
80
96
  sonilo video-to-video-sfx --video clip.mp4 --segments @segments.json --output scored.mp4
81
97
  sonilo video-to-video-sound --video clip.mp4 --music-prompt "tense synths"
98
+ sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
99
+ # ducks the existing music bed under the voice; a video voice comes back
100
+ # as a new .mp4 with the ducked mix muxed in
82
101
  sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
83
102
  # writes dubbed.es.mp4 and dubbed.fr.mp4
84
103
  sonilo tasks get <task-id>
@@ -204,6 +223,24 @@ they differ only in what comes back: `video-to-sound` writes the mixed **audio**
204
223
  or ducking altered the music bed.
205
224
  - Both also take `--variants` — see [Variants](#variants) above.
206
225
 
226
+ ### Audio ducking
227
+
228
+ `audio-ducking` mixes an **existing** music bed under an **existing** voice track — nothing is
229
+ generated, so reach for it when the music is fixed or external. (When the music is being generated
230
+ for the same clip anyway, `video-to-sound` or `video-to-music` duck internally as part of that one
231
+ call instead.)
232
+
233
+ sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
234
+
235
+ - Exactly one of `--voice` / `--voice-url` and one of `--music` / `--music-url`; a local file and a
236
+ URL mix freely across the two inputs.
237
+ - The **voice** may be audio or video: a video's own audio track becomes the voice, and the ducked
238
+ mix is muxed back into a new video, so the result is a `.mp4` instead of a `.wav`. The default
239
+ `--output` name follows what came back (`output.wav` or `output.mp4`).
240
+ - The **music** must be audio (`wav, mp3, m4a, aac, ogg, flac`). The API does not detect a video
241
+ there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
242
+ - Each input is capped at 360 seconds server-side.
243
+
207
244
  ### Dubbing
208
245
 
209
246
  `dubbing` dubs a video into one or more target languages in a single async call:
@@ -1,19 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: sonilo-cli
3
- Version: 0.11.0
4
- Summary: Command-line interface for the Sonilo API: generate music and sound effects from text or video
5
- Project-URL: Repository, https://github.com/sonilo-ai/sonilo-python
6
- Author: Sonilo AI
7
- License-Expression: MIT
8
- License-File: LICENSE
9
- Keywords: ai,cli,music,sfx,sonilo,text-to-music,video-to-music
10
- Requires-Python: >=3.9
11
- Requires-Dist: sonilo<0.13,>=0.12.0
12
- Provides-Extra: dev
13
- Requires-Dist: pytest>=8; extra == 'dev'
14
- Requires-Dist: respx>=0.21; extra == 'dev'
15
- Description-Content-Type: text/markdown
16
-
17
1
  # sonilo-cli
18
2
 
19
3
  Command-line interface for the [Sonilo API](https://github.com/sonilo-ai/sonilo-python) — generate music and sound effects from text or video.
@@ -95,6 +79,9 @@ production sign-in coexist without overwriting each other.
95
79
  sonilo video-to-video-music --video clip.mp4 --prompt "tense synths" --output scored.mp4
96
80
  sonilo video-to-video-sfx --video clip.mp4 --segments @segments.json --output scored.mp4
97
81
  sonilo video-to-video-sound --video clip.mp4 --music-prompt "tense synths"
82
+ sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
83
+ # ducks the existing music bed under the voice; a video voice comes back
84
+ # as a new .mp4 with the ducked mix muxed in
98
85
  sonilo dubbing --video-url https://example.com/clip.mp4 --languages es,fr --output dubbed.mp4
99
86
  # writes dubbed.es.mp4 and dubbed.fr.mp4
100
87
  sonilo tasks get <task-id>
@@ -220,6 +207,24 @@ they differ only in what comes back: `video-to-sound` writes the mixed **audio**
220
207
  or ducking altered the music bed.
221
208
  - Both also take `--variants` — see [Variants](#variants) above.
222
209
 
210
+ ### Audio ducking
211
+
212
+ `audio-ducking` mixes an **existing** music bed under an **existing** voice track — nothing is
213
+ generated, so reach for it when the music is fixed or external. (When the music is being generated
214
+ for the same clip anyway, `video-to-sound` or `video-to-music` duck internally as part of that one
215
+ call instead.)
216
+
217
+ sonilo audio-ducking --voice interview.mp4 --music-url https://example.com/bed.wav
218
+
219
+ - Exactly one of `--voice` / `--voice-url` and one of `--music` / `--music-url`; a local file and a
220
+ URL mix freely across the two inputs.
221
+ - The **voice** may be audio or video: a video's own audio track becomes the voice, and the ducked
222
+ mix is muxed back into a new video, so the result is a `.mp4` instead of a `.wav`. The default
223
+ `--output` name follows what came back (`output.wav` or `output.mp4`).
224
+ - The **music** must be audio (`wav, mp3, m4a, aac, ogg, flac`). The API does not detect a video
225
+ there, so the CLI rejects a local video file up front rather than let it be mishandled silently.
226
+ - Each input is capped at 360 seconds server-side.
227
+
223
228
  ### Dubbing
224
229
 
225
230
  `dubbing` dubs a video into one or more target languages in a single async call:
@@ -4,13 +4,13 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "sonilo-cli"
7
- version = "0.11.0"
7
+ version = "0.12.0"
8
8
  description = "Command-line interface for the Sonilo API: generate music and sound effects from text or video"
9
9
  readme = "README.md"
10
10
  license = "MIT"
11
11
  requires-python = ">=3.9"
12
12
  authors = [{ name = "Sonilo AI" }]
13
- dependencies = ["sonilo>=0.12.0,<0.13"]
13
+ dependencies = ["sonilo>=0.13.0,<0.14"]
14
14
  keywords = ["sonilo", "cli", "music", "sfx", "text-to-music", "video-to-music", "ai"]
15
15
 
16
16
  [project.urls]
@@ -1,3 +1,3 @@
1
- __version__ = "0.11.0"
1
+ __version__ = "0.12.0"
2
2
 
3
3
  __all__ = ["__version__"]
@@ -474,6 +474,42 @@ def cmd_video_to_video_sfx(client: Sonilo, args: argparse.Namespace) -> None:
474
474
  )
475
475
 
476
476
 
477
+ # The music-bed extensions audio-ducking accepts for a local --music file —
478
+ # the MCP server's _AUDIO_EXTS, kept identical so the two surfaces accept and
479
+ # reject the same files. A whitelist (not a video-extension blacklist) because
480
+ # the failure it guards against is silent: the API never probes the music
481
+ # input for a video stream, so a video sent there is mishandled without an
482
+ # error. The voice input needs no such guard — it may legitimately be audio
483
+ # or video, and the API probes it.
484
+ _MUSIC_AUDIO_EXTS = (".wav", ".mp3", ".m4a", ".aac", ".ogg", ".flac")
485
+
486
+
487
+ def cmd_audio_ducking(client: Sonilo, args: argparse.Namespace) -> None:
488
+ if args.music is not None and Path(args.music).suffix.lower() not in _MUSIC_AUDIO_EXTS:
489
+ _fail(
490
+ f"--music must be an audio file ({', '.join(_MUSIC_AUDIO_EXTS)}) — "
491
+ "the API does not detect a video here and would mishandle it. "
492
+ "The voice input is the one that may be a video."
493
+ )
494
+ result = client.audio_ducking.generate(
495
+ voice=args.voice,
496
+ voice_url=args.voice_url,
497
+ music=args.music,
498
+ music_url=args.music_url,
499
+ )
500
+ # Default output name follows what actually came back: a .wav, or a .mp4
501
+ # (ducked mix re-muxed in) when the voice input was a video. An explicit
502
+ # --output is used verbatim, same as the video-out commands.
503
+ out = args.output
504
+ if out is None:
505
+ ext = Path(urlparse(result.output_url or "").path).suffix
506
+ if not ext:
507
+ ext = ".mp4" if result.output_type == "video" else ".wav"
508
+ out = f"output{ext}"
509
+ path = result.save(out)
510
+ _wrote(path, path.stat().st_size)
511
+
512
+
477
513
  # Matched to the dubbing backend's own ceiling: it polls its pipeline for up
478
514
  # to 7200s (2 hours), so anything shorter abandons a job the user has already
479
515
  # been charged for. The SDK's generic DEFAULT_WAIT_TIMEOUT of 600s is far too
@@ -790,6 +826,38 @@ def build_parser() -> argparse.ArgumentParser:
790
826
  _add_variants(p_v2vsd)
791
827
  p_v2vsd.set_defaults(func=cmd_video_to_video_sound)
792
828
 
829
+ p_duck = sub.add_parser(
830
+ "audio-ducking", help="Duck an existing music bed under a voice track"
831
+ )
832
+ _add_global(p_duck)
833
+ voice_group = p_duck.add_mutually_exclusive_group(required=True)
834
+ voice_group.add_argument(
835
+ "--voice", default=None,
836
+ help="Local voice track. May be audio or video: a video's own audio "
837
+ "track becomes the voice, and the ducked mix is muxed back into "
838
+ "a new video.",
839
+ )
840
+ voice_group.add_argument(
841
+ "--voice-url", dest="voice_url", default=None,
842
+ help="Remote voice audio/video URL.",
843
+ )
844
+ music_group = p_duck.add_mutually_exclusive_group(required=True)
845
+ music_group.add_argument(
846
+ "--music", default=None,
847
+ help="Local music bed. Audio only (wav, mp3, m4a, aac, ogg, flac) — "
848
+ "the API does not detect a video here, so the CLI rejects one.",
849
+ )
850
+ music_group.add_argument(
851
+ "--music-url", dest="music_url", default=None,
852
+ help="Remote music audio URL.",
853
+ )
854
+ p_duck.add_argument(
855
+ "--output", default=None,
856
+ help="Where to save the result. Default: output.wav, or output.mp4 "
857
+ "when the voice input was a video.",
858
+ )
859
+ p_duck.set_defaults(func=cmd_audio_ducking)
860
+
793
861
  p_dub = sub.add_parser("dubbing", help="Dub a video into other languages")
794
862
  _add_global(p_dub)
795
863
  _add_video_source(p_dub)
@@ -523,6 +523,90 @@ def test_video_to_video_sound_requires_a_video_source():
523
523
  assert exc.value.code == 1
524
524
 
525
525
 
526
+ # --- audio-ducking --------------------------------------------------------
527
+ #
528
+ # Same flat output_url envelope as video-to-sound, no stems. The result is a
529
+ # .wav, or a .mp4 (ducked mix re-muxed in) when the voice input was a video.
530
+
531
+ DUCKING_SUCCESS_BODY = {
532
+ "task_id": "ad1",
533
+ "type": "audio_ducking",
534
+ "status": "succeeded",
535
+ "output_url": "https://r2.example.com/ducked.wav",
536
+ "output_type": "audio",
537
+ "output_bytes": 5,
538
+ }
539
+
540
+
541
+ @respx.mock
542
+ def test_audio_ducking_saves_output(tmp_path):
543
+ route = respx.post(f"{BASE}/v1/audio-ducking").mock(
544
+ return_value=httpx.Response(202, json={"task_id": "ad1", "status": "processing"})
545
+ )
546
+ respx.get(f"{BASE}/v1/tasks/ad1").mock(
547
+ return_value=httpx.Response(200, json=DUCKING_SUCCESS_BODY)
548
+ )
549
+ respx.get("https://r2.example.com/ducked.wav").mock(
550
+ return_value=httpx.Response(200, content=b"DUCKED")
551
+ )
552
+ out = tmp_path / "mix.wav"
553
+ run(["audio-ducking", "--voice-url", "https://x/v.wav",
554
+ "--music-url", "https://x/m.wav", "--output", str(out)])
555
+ assert route.called
556
+ content = route.calls[0].request.content
557
+ assert b"voice_url" in content and b"music_url" in content
558
+ assert out.read_bytes() == b"DUCKED"
559
+
560
+
561
+ @respx.mock
562
+ def test_audio_ducking_default_output_follows_result_type(tmp_path, monkeypatch):
563
+ # A video voice input comes back as an mp4; the default output name must
564
+ # follow the result, not assume audio.
565
+ monkeypatch.chdir(tmp_path)
566
+ respx.post(f"{BASE}/v1/audio-ducking").mock(
567
+ return_value=httpx.Response(202, json={"task_id": "ad2", "status": "processing"})
568
+ )
569
+ respx.get(f"{BASE}/v1/tasks/ad2").mock(
570
+ return_value=httpx.Response(200, json={
571
+ **DUCKING_SUCCESS_BODY, "task_id": "ad2",
572
+ "output_url": "https://r2.example.com/ducked.mp4",
573
+ "output_type": "video",
574
+ })
575
+ )
576
+ respx.get("https://r2.example.com/ducked.mp4").mock(
577
+ return_value=httpx.Response(200, content=b"MP4DATA")
578
+ )
579
+ run(["audio-ducking", "--voice-url", "https://x/v.mp4",
580
+ "--music-url", "https://x/m.wav"])
581
+ assert (tmp_path / "output.mp4").read_bytes() == b"MP4DATA"
582
+
583
+
584
+ def test_audio_ducking_requires_a_voice_source():
585
+ with pytest.raises(SystemExit) as exc:
586
+ run(["audio-ducking", "--music-url", "https://x/m.wav"])
587
+ assert exc.value.code == 1
588
+
589
+
590
+ def test_audio_ducking_requires_a_music_source():
591
+ with pytest.raises(SystemExit) as exc:
592
+ run(["audio-ducking", "--voice-url", "https://x/v.wav"])
593
+ assert exc.value.code == 1
594
+
595
+
596
+ @respx.mock
597
+ def test_audio_ducking_rejects_a_video_music_file(tmp_path, capsys):
598
+ # The API never probes the music input for a video stream, so a video
599
+ # there is mishandled silently — the CLI must refuse before any request.
600
+ music = tmp_path / "background.mp4"
601
+ music.write_bytes(b"m")
602
+ with pytest.raises(SystemExit) as exc:
603
+ run(["audio-ducking", "--voice-url", "https://x/v.wav",
604
+ "--music", str(music)])
605
+ assert exc.value.code == 1
606
+ assert "audio" in capsys.readouterr().err
607
+ assert not respx.calls
608
+
609
+
526
610
  def test_cli_identifies_itself_not_the_sdk():
527
611
  """CLI traffic must be separable from direct SDK use in analytics."""
528
612
  import sonilo_cli
File without changes
File without changes