rvcbench 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rvcbench/__init__.py +17 -0
- rvcbench/adapter.py +98 -0
- rvcbench/adversary/__init__.py +0 -0
- rvcbench/adversary/bark_voice_clone_ots.py +146 -0
- rvcbench/adversary/base_adversary.py +205 -0
- rvcbench/adversary/bertvit2_ots.py +231 -0
- rvcbench/adversary/bertvits2_finetune.py +53 -0
- rvcbench/adversary/cosyvoice_ots.py +156 -0
- rvcbench/adversary/dots_tts_ots.py +133 -0
- rvcbench/adversary/f5_tts_ots.py +176 -0
- rvcbench/adversary/fireredtts2_ots.py +118 -0
- rvcbench/adversary/fish_audio_s2_server_ots.py +188 -0
- rvcbench/adversary/fishspeech_ots.py +317 -0
- rvcbench/adversary/fishspeech_s2_ots.py +140 -0
- rvcbench/adversary/glmtts_ots.py +109 -0
- rvcbench/adversary/glowtts_ots.py +169 -0
- rvcbench/adversary/higgs_audio_ots.py +151 -0
- rvcbench/adversary/index_tts_ots.py +162 -0
- rvcbench/adversary/kimi_audio_ots.py +214 -0
- rvcbench/adversary/maskgct_ots.py +166 -0
- rvcbench/adversary/mgm_omni_ots.py +182 -0
- rvcbench/adversary/moss_tts_ots.py +141 -0
- rvcbench/adversary/moss_ttsd_ots.py +175 -0
- rvcbench/adversary/openai_speech_server_ots.py +355 -0
- rvcbench/adversary/openvoice_ots.py +149 -0
- rvcbench/adversary/ozspeech_ots.py +213 -0
- rvcbench/adversary/playdiffusion_ots.py +162 -0
- rvcbench/adversary/qwen3_omni_ots.py +254 -0
- rvcbench/adversary/qwen3_tts_ots.py +237 -0
- rvcbench/adversary/smoke.py +16 -0
- rvcbench/adversary/sparktts_ots.py +94 -0
- rvcbench/adversary/styletts2_ots.py +101 -0
- rvcbench/adversary/vall_e_ots.py +140 -0
- rvcbench/adversary/vibevoice_ots.py +179 -0
- rvcbench/adversary/voxcpm_ots.py +155 -0
- rvcbench/adversary/xtts_ots.py +281 -0
- rvcbench/adversary/zipvoice_ots.py +161 -0
- rvcbench/adversary/zonos2_ots.py +149 -0
- rvcbench/benchmark/__init__.py +1 -0
- rvcbench/benchmark/artifacts.py +252 -0
- rvcbench/benchmark/backends.py +278 -0
- rvcbench/benchmark/cli.py +268 -0
- rvcbench/benchmark/comparability.py +54 -0
- rvcbench/benchmark/comparison.py +114 -0
- rvcbench/benchmark/denoise_stage.py +160 -0
- rvcbench/benchmark/enkidu_stage.py +158 -0
- rvcbench/benchmark/fingerprints.py +208 -0
- rvcbench/benchmark/model_assets.py +369 -0
- rvcbench/benchmark/model_catalog.json +194 -0
- rvcbench/benchmark/noise_replay.py +166 -0
- rvcbench/benchmark/reference_stages.py +100 -0
- rvcbench/benchmark/registry.py +113 -0
- rvcbench/benchmark/reproduction.py +232 -0
- rvcbench/benchmark/runner.py +298 -0
- rvcbench/benchmark/source_audit.py +62 -0
- rvcbench/benchmark/submission.py +510 -0
- rvcbench/benchmark/subsets.py +40 -0
- rvcbench/benchmark/timing.py +214 -0
- rvcbench/configs/__init__.py +0 -0
- rvcbench/configs/adversary/bertvits2_ots.yaml +18 -0
- rvcbench/configs/dataset/aishell.yaml +32 -0
- rvcbench/configs/dataset/background_clean.yaml +22 -0
- rvcbench/configs/dataset/background_noise.yaml +30 -0
- rvcbench/configs/dataset/bilingual_uedin.yaml +32 -0
- rvcbench/configs/dataset/dummy.yaml +16 -0
- rvcbench/configs/dataset/french.yaml +32 -0
- rvcbench/configs/dataset/iemocap.yaml +27 -0
- rvcbench/configs/dataset/libritts.yaml +24 -0
- rvcbench/configs/dataset/long_librispeech.yaml +22 -0
- rvcbench/configs/dataset/multispeaker_libri.yaml +22 -0
- rvcbench/configs/dataset/robotcall.yaml +21 -0
- rvcbench/configs/dataset/test.yaml +18 -0
- rvcbench/configs/dataset/vctk.yaml +21 -0
- rvcbench/configs/dataset/vctk_text_robust.yaml +21 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts.yaml +44 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts_em.yaml +44 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts_enkidu.yaml +44 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts_grnoise.yaml +44 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts_safespeech.yaml +44 -0
- rvcbench/configs/denoise/denoiser_dns64_on_protected_libritts_spec.yaml +44 -0
- rvcbench/configs/dummy_on_dummy.yaml +17 -0
- rvcbench/configs/em_on_libritts.yaml +61 -0
- rvcbench/configs/enkidu_on_libritts.yaml +50 -0
- rvcbench/configs/grnoise_on_libritts.yaml +40 -0
- rvcbench/configs/model/BertVits2.yaml +58 -0
- rvcbench/configs/model/OZSpeech.yaml +7 -0
- rvcbench/configs/model/SpeakerRecognition.yaml +3 -0
- rvcbench/configs/model/default.yaml +3 -0
- rvcbench/configs/ots_vc/clean/aishell/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/aishell/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/aishell/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/aishell/eval_only.yaml +27 -0
- rvcbench/configs/ots_vc/clean/aishell/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/aishell/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/aishell/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/aishell/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/aishell/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/aishell/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/aishell/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/aishell/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/aishell/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/aishell/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/aishell/moss_ttsd_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/aishell/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/aishell/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/aishell/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/aishell/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/aishell/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/aishell/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/aishell/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/aishell/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/background/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/background/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/background/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/background/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/background/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/background/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/background/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/background/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/background/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/background/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/background/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/background/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/background/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/background/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/background_clean/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/background_clean/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background_clean/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/background_clean/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/background_clean/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background_clean/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background_clean/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/background_clean/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background_clean/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/background_clean/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/background_clean/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background_clean/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background_clean/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/background_clean/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/background_clean/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/background_clean/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/background_clean/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/background_clean/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/background_clean/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/background_clean/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/background_clean/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/background_clean/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/background_clean/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/higgs_audio_ots.yaml +76 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/bilingual_uedin/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/french/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/french/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/french/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/french/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/french/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/french/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/french/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/french/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/french/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/french/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/french/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/french/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/french/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/french/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/french/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/french/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/french/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/french/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/french/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/french/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/french/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/french/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/french/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/libritts/bark_voice_clone_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/libritts/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/libritts/cosyvoice_ots.yaml +38 -0
- rvcbench/configs/ots_vc/clean/libritts/custom_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/libritts/dots_tts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/libritts/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/libritts/f5_tts_ots.yaml +48 -0
- rvcbench/configs/ots_vc/clean/libritts/fireredtts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/libritts/fireredtts2_short10_retry20_ots.yaml +11 -0
- rvcbench/configs/ots_vc/clean/libritts/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/libritts/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/libritts/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/libritts/glmtts_ots.yaml +36 -0
- rvcbench/configs/ots_vc/clean/libritts/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/libritts/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/libritts/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/libritts/index_tts_ots.yaml +54 -0
- rvcbench/configs/ots_vc/clean/libritts/kimi_audio_ots.yaml +44 -0
- rvcbench/configs/ots_vc/clean/libritts/maskgct_ots.yaml +49 -0
- rvcbench/configs/ots_vc/clean/libritts/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/libritts/moss_tts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/libritts/moss_ttsd_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/libritts/openvoice_ots.yaml +60 -0
- rvcbench/configs/ots_vc/clean/libritts/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/libritts/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/libritts/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/libritts/qwen3_tts_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/libritts/sparktts_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/libritts/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/libritts/vall_e_amphion_ots.yaml +11 -0
- rvcbench/configs/ots_vc/clean/libritts/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/libritts/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/libritts/voxcpm_05_legacy_ots.yaml +13 -0
- rvcbench/configs/ots_vc/clean/libritts/voxcpm_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/libritts/xtts_ots.yaml +50 -0
- rvcbench/configs/ots_vc/clean/libritts/zipvoice_ots.yaml +52 -0
- rvcbench/configs/ots_vc/clean/libritts/zonos2_ots.yaml +39 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/README.md +26 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/long_librispeech/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/README.md +26 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/higgs_audio_ots.yaml +76 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/multispeaker_libri/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/robotcall/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/robotcall/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/robotcall/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/robotcall/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/robotcall/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/robotcall/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/robotcall/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/robotcall/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/robotcall/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/robotcall/higgs_audio_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/robotcall/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/robotcall/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/robotcall/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/robotcall/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/robotcall/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/robotcall/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/robotcall/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/robotcall/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/robotcall/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/robotcall/styletts2_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/robotcall/vall_e_amphion_ots.yaml +11 -0
- rvcbench/configs/ots_vc/clean/robotcall/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/robotcall/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/robotcall/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/test.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/vctk/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/vctk/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/vctk/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/vctk/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/vctk/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/vctk/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/clean/vctk/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/vctk/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/vctk/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/vctk/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/vctk/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/vctk/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/vctk/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/vctk/styletts2_ots.yaml +38 -0
- rvcbench/configs/ots_vc/clean/vctk/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/vctk/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/cosyvoice_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/dots_tts_ots.yaml +32 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/eval_only.yaml +26 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/fish_audio_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/fishspeech_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/fishspeech_s2_ots.yaml +46 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/glmtts_ots.yaml +37 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/glowtts_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/higgs_tts3_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/kimi_audio_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/mgm_omni_ots.yaml +41 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/moss_tts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/moss_ttsd_ots.yaml +40 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/ozspeech_ots.yaml +34 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/playdiffusion_ots.yaml +47 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/qwen3_omni_ots.yaml +58 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/sparktts_ots.yaml +33 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/styletts2_ots.yaml +38 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/vall_e_ots.yaml +35 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/vibevoice_ots.yaml +43 -0
- rvcbench/configs/ots_vc/clean/vctk_text_robust/zonos2_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/denoiser_dns64_on_spec_libritts/zonos2_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/em/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/em/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/em/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/em/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/em/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/em/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/em/glowtts_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/em/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/em/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/em/kimi_audio_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/em/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/em/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/em/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/em/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/em/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/em/qwen3_omni_ots.yaml +56 -0
- rvcbench/configs/ots_vc/protection/em/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/em/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/em/vall_e_ots.yaml +33 -0
- rvcbench/configs/ots_vc/protection/em/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/em/zonos2_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/enkidu/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/enkidu/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/enkidu/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/enkidu/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/enkidu/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/enkidu/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/enkidu/glowtts_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/enkidu/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/enkidu/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/enkidu/kimi_audio_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/enkidu/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/enkidu/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/enkidu/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/enkidu/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/enkidu/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/enkidu/qwen3_omni_ots.yaml +56 -0
- rvcbench/configs/ots_vc/protection/enkidu/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/enkidu/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/enkidu/vall_e_ots.yaml +33 -0
- rvcbench/configs/ots_vc/protection/enkidu/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/enkidu/zonos2_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/glowtts_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/kimi_audio_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/qwen3_omni_ots.yaml +56 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/vall_e_ots.yaml +33 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/gaussian_noise/zonos2_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/safespeech/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/safespeech/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/safespeech/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/safespeech/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/safespeech/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/safespeech/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/safespeech/glowtts_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/safespeech/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/safespeech/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/safespeech/kimi_audio_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/safespeech/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/safespeech/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/safespeech/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/safespeech/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/safespeech/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/safespeech/qwen3_omni_ots.yaml +56 -0
- rvcbench/configs/ots_vc/protection/safespeech/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/safespeech/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/safespeech/vall_e_ots.yaml +33 -0
- rvcbench/configs/ots_vc/protection/safespeech/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/safespeech/zonos2_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/spec/bert_ots.yaml +28 -0
- rvcbench/configs/ots_vc/protection/spec/cosyvoice_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/spec/dots_tts_ots.yaml +30 -0
- rvcbench/configs/ots_vc/protection/spec/fish_audio_s2_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/spec/fishspeech_ots.yaml +44 -0
- rvcbench/configs/ots_vc/protection/spec/glmtts_ots.yaml +34 -0
- rvcbench/configs/ots_vc/protection/spec/glowtts_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/spec/higgs_audio_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/spec/higgs_tts3_ots.yaml +41 -0
- rvcbench/configs/ots_vc/protection/spec/kimi_audio_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/spec/mgm_omni_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/spec/moss_tts_ots.yaml +29 -0
- rvcbench/configs/ots_vc/protection/spec/moss_ttsd_ots.yaml +38 -0
- rvcbench/configs/ots_vc/protection/spec/ozspeech_ots.yaml +32 -0
- rvcbench/configs/ots_vc/protection/spec/playdiffusion_ots.yaml +45 -0
- rvcbench/configs/ots_vc/protection/spec/qwen3_omni_ots.yaml +56 -0
- rvcbench/configs/ots_vc/protection/spec/sparktts_ots.yaml +31 -0
- rvcbench/configs/ots_vc/protection/spec/styletts2_ots.yaml +35 -0
- rvcbench/configs/ots_vc/protection/spec/vall_e_ots.yaml +33 -0
- rvcbench/configs/ots_vc/protection/spec/vibevoice_ots.yaml +40 -0
- rvcbench/configs/ots_vc/protection/spec/zonos2_ots.yaml +30 -0
- rvcbench/configs/ozspeech_ots.yaml +31 -0
- rvcbench/configs/safespeech_on_libritts.yaml +61 -0
- rvcbench/configs/safespeech_on_libritts_eval_only.yaml +28 -0
- rvcbench/configs/spec_on_libritts.yaml +61 -0
- rvcbench/datasets/__init__.py +9 -0
- rvcbench/datasets/audio_only.py +226 -0
- rvcbench/datasets/build_canonical_manifests.py +93 -0
- rvcbench/datasets/build_vctk_compression_dataset_from_manifest.py +319 -0
- rvcbench/datasets/data_utils.py +886 -0
- rvcbench/datasets/deepfake_preprocess.py +202 -0
- rvcbench/datasets/feature_embedder/__init__.py +0 -0
- rvcbench/datasets/feature_embedder/base_embedder.py +14 -0
- rvcbench/datasets/feature_embedder/stft.py +14 -0
- rvcbench/datasets/manifest_utils.py +393 -0
- rvcbench/datasets/mel_preprocessing.py +168 -0
- rvcbench/datasets/text_features.py +23 -0
- rvcbench/datasets/top10_longest_by_speaker_manifest.csv +41 -0
- rvcbench/datasets/zero_shot.py +74 -0
- rvcbench/entrypoints/__init__.py +1 -0
- rvcbench/entrypoints/denoise.py +349 -0
- rvcbench/entrypoints/protect.py +176 -0
- rvcbench/entrypoints/vc.py +79 -0
- rvcbench/entrypoints/vc_protect.py +174 -0
- rvcbench/evaluation/__init__.py +0 -0
- rvcbench/evaluation/assets.py +20 -0
- rvcbench/evaluation/audio_io.py +27 -0
- rvcbench/evaluation/bootstrap.py +112 -0
- rvcbench/evaluation/fidelity.py +946 -0
- rvcbench/evaluation/generation.py +1111 -0
- rvcbench/evaluation/pipeline.py +274 -0
- rvcbench/evaluation/scorers/__init__.py +31 -0
- rvcbench/evaluation/scorers/auxiliary.py +75 -0
- rvcbench/evaluation/scorers/emotion.py +85 -0
- rvcbench/evaluation/scorers/mcd.py +18 -0
- rvcbench/evaluation/scorers/speaker.py +35 -0
- rvcbench/evaluation/scorers/stoi.py +38 -0
- rvcbench/evaluation/scorers/text.py +91 -0
- rvcbench/evaluation/scorers/wer.py +54 -0
- rvcbench/evaluation/setup.py +98 -0
- rvcbench/losses/__init__.py +1 -0
- rvcbench/losses/bertvits2_loss.py +251 -0
- rvcbench/metrics.py +132 -0
- rvcbench/models/__init__.py +9 -0
- rvcbench/models/acoustic_system_model.py +508 -0
- rvcbench/models/antifake_model.py +730 -0
- rvcbench/models/bark_voice_clone/__init__.py +1 -0
- rvcbench/models/bark_voice_clone/generator.py +201 -0
- rvcbench/models/bark_voice_clone/loading.py +59 -0
- rvcbench/models/bert/bert_models.json +6 -0
- rvcbench/models/bertvits2_model.py +468 -0
- rvcbench/models/cosyvoice/__init__.py +8 -0
- rvcbench/models/cosyvoice/assets.py +80 -0
- rvcbench/models/cosyvoice/generator.py +297 -0
- rvcbench/models/dns64_kernel.py +50 -0
- rvcbench/models/dots_tts/__init__.py +3 -0
- rvcbench/models/dots_tts/generator.py +153 -0
- rvcbench/models/enkidu_model.py +27 -0
- rvcbench/models/f5_tts/__init__.py +5 -0
- rvcbench/models/f5_tts/generator.py +208 -0
- rvcbench/models/fireredtts2/__init__.py +8 -0
- rvcbench/models/fireredtts2/generator.py +241 -0
- rvcbench/models/glmtts/__init__.py +5 -0
- rvcbench/models/glmtts/synthesizer.py +365 -0
- rvcbench/models/higgs_audio/__init__.py +3 -0
- rvcbench/models/higgs_audio/generator.py +349 -0
- rvcbench/models/index_tts/__init__.py +5 -0
- rvcbench/models/index_tts/generator.py +322 -0
- rvcbench/models/kimi_audio/__init__.py +3 -0
- rvcbench/models/kimi_audio/generator.py +129 -0
- rvcbench/models/maskgct/__init__.py +5 -0
- rvcbench/models/maskgct/assets.py +34 -0
- rvcbench/models/maskgct/generator.py +362 -0
- rvcbench/models/mgm_omni/__init__.py +5 -0
- rvcbench/models/mgm_omni/generator.py +309 -0
- rvcbench/models/model.py +121 -0
- rvcbench/models/modules/__init__.py +0 -0
- rvcbench/models/modules/bertvits2_module.py +1078 -0
- rvcbench/models/modules/modules.py +922 -0
- rvcbench/models/modules/transforms.py +209 -0
- rvcbench/models/moss_tts/__init__.py +3 -0
- rvcbench/models/moss_tts/generator.py +140 -0
- rvcbench/models/moss_ttsd/__init__.py +8 -0
- rvcbench/models/moss_ttsd/generator.py +310 -0
- rvcbench/models/openvoice/__init__.py +8 -0
- rvcbench/models/openvoice/assets.py +15 -0
- rvcbench/models/openvoice/generator.py +360 -0
- rvcbench/models/openvoice/loading.py +32 -0
- rvcbench/models/openvoice/text_models.py +92 -0
- rvcbench/models/openvoice/text_resources.py +41 -0
- rvcbench/models/ozspeech/__init__.py +3 -0
- rvcbench/models/ozspeech/synthesizer.py +292 -0
- rvcbench/models/ozspeech_model.py +110 -0
- rvcbench/models/playdiffusion/__init__.py +3 -0
- rvcbench/models/playdiffusion/generator.py +175 -0
- rvcbench/models/qwen3_omni/__init__.py +3 -0
- rvcbench/models/qwen3_omni/generator.py +166 -0
- rvcbench/models/qwen3_tts/__init__.py +8 -0
- rvcbench/models/qwen3_tts/generator.py +169 -0
- rvcbench/models/sparktts/__init__.py +8 -0
- rvcbench/models/sparktts/assets.py +87 -0
- rvcbench/models/sparktts/generator.py +174 -0
- rvcbench/models/stable_codec_activations.py +40 -0
- rvcbench/models/styletts2/__init__.py +3 -0
- rvcbench/models/styletts2/loading.py +50 -0
- rvcbench/models/styletts2/synthesizer.py +376 -0
- rvcbench/models/text/__init__.py +54 -0
- rvcbench/models/text/bert_utils.py +12 -0
- rvcbench/models/text/chinese.py +206 -0
- rvcbench/models/text/chinese_bert.py +119 -0
- rvcbench/models/text/cleaner.py +28 -0
- rvcbench/models/text/cmudict.rep +129530 -0
- rvcbench/models/text/cmudict_cache.pickle +0 -0
- rvcbench/models/text/english.py +494 -0
- rvcbench/models/text/english_bert_mock.py +61 -0
- rvcbench/models/text/japanese.py +720 -0
- rvcbench/models/text/japanese_bert.py +65 -0
- rvcbench/models/text/opencpop-strict.txt +429 -0
- rvcbench/models/text/symbols.py +187 -0
- rvcbench/models/text/tone_sandhi.py +776 -0
- rvcbench/models/valle/__init__.py +3 -0
- rvcbench/models/valle/amphion.py +75 -0
- rvcbench/models/valle/generator.py +114 -0
- rvcbench/models/vibevoice/__init__.py +3 -0
- rvcbench/models/vibevoice/generator.py +259 -0
- rvcbench/models/voxcpm/__init__.py +8 -0
- rvcbench/models/voxcpm/generator.py +182 -0
- rvcbench/models/worker_protocol.py +47 -0
- rvcbench/models/xtts/__init__.py +8 -0
- rvcbench/models/xtts/assets.py +22 -0
- rvcbench/models/xtts/generator.py +244 -0
- rvcbench/models/zipvoice/__init__.py +5 -0
- rvcbench/models/zipvoice/assets.py +24 -0
- rvcbench/models/zipvoice/generator.py +331 -0
- rvcbench/models/zonos2/__init__.py +3 -0
- rvcbench/models/zonos2/generator.py +215 -0
- rvcbench/monotonic_align.py +55 -0
- rvcbench/protection/__init__.py +36 -0
- rvcbench/protection/antifake.py +991 -0
- rvcbench/protection/attackvc.py +77 -0
- rvcbench/protection/base_protector.py +79 -0
- rvcbench/protection/dummy.py +19 -0
- rvcbench/protection/em.py +242 -0
- rvcbench/protection/enkidu.py +276 -0
- rvcbench/protection/random_noise.py +104 -0
- rvcbench/protection/safespeech/original_code/LICENSE +21 -0
- rvcbench/protection/safespeech/original_code/QuickStart.ipynb +724 -0
- rvcbench/protection/safespeech/original_code/README.md +176 -0
- rvcbench/protection/safespeech/original_code/asr.py +64 -0
- rvcbench/protection/safespeech/original_code/bert_gen.py +101 -0
- rvcbench/protection/safespeech/original_code/download_models.py +53 -0
- rvcbench/protection/safespeech/original_code/evaluate.py +297 -0
- rvcbench/protection/safespeech/original_code/evaluation/evallists/BERT_VITS2_SPEC_LibriTTS_text.txt +52 -0
- rvcbench/protection/safespeech/original_code/preprocess_text.py +42 -0
- rvcbench/protection/safespeech/original_code/protect.py +380 -0
- rvcbench/protection/safespeech/original_code/requirements.txt +46 -0
- rvcbench/protection/safespeech/original_code/save_audio.py +137 -0
- rvcbench/protection/safespeech/original_code/toolbox.py +149 -0
- rvcbench/protection/safespeech/original_code/train.py +407 -0
- rvcbench/protection/safespeech/protector.py +84 -0
- rvcbench/protection/safespeech_use_wrapper.py +183 -0
- rvcbench/suites/__init__.py +0 -0
- rvcbench/suites/core_v1/adv-clean.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-clean.selection.json +290 -0
- rvcbench/suites/core_v1/adv-enkidu.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-enkidu.selection.json +290 -0
- rvcbench/suites/core_v1/adv-gaussian.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-gaussian.selection.json +290 -0
- rvcbench/suites/core_v1/adv-pop.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-pop.selection.json +290 -0
- rvcbench/suites/core_v1/adv-safespeech.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-safespeech.selection.json +290 -0
- rvcbench/suites/core_v1/adv-spec.metadata.json +342 -0
- rvcbench/suites/core_v1/adv-spec.selection.json +290 -0
- rvcbench/suites/core_v1/antiprotect-spec.metadata.json +342 -0
- rvcbench/suites/core_v1/antiprotect-spec.selection.json +290 -0
- rvcbench/suites/core_v1/audioshift.metadata.json +458 -0
- rvcbench/suites/core_v1/audioshift.selection.json +346 -0
- rvcbench/suites/core_v1/background-clean.metadata.json +342 -0
- rvcbench/suites/core_v1/background-clean.selection.json +290 -0
- rvcbench/suites/core_v1/background.metadata.json +342 -0
- rvcbench/suites/core_v1/background.selection.json +290 -0
- rvcbench/suites/core_v1/chinese.metadata.json +386 -0
- rvcbench/suites/core_v1/chinese.selection.json +346 -0
- rvcbench/suites/core_v1/crosslingual.metadata.json +410 -0
- rvcbench/suites/core_v1/crosslingual.selection.json +346 -0
- rvcbench/suites/core_v1/english-libritts.metadata.json +410 -0
- rvcbench/suites/core_v1/english-libritts.selection.json +346 -0
- rvcbench/suites/core_v1/french.metadata.json +386 -0
- rvcbench/suites/core_v1/french.selection.json +346 -0
- rvcbench/suites/core_v1/longaudio.metadata.json +410 -0
- rvcbench/suites/core_v1/longaudio.selection.json +346 -0
- rvcbench/suites/core_v1/longtext.metadata.json +322 -0
- rvcbench/suites/core_v1/longtext.selection.json +290 -0
- rvcbench/suites/core_v1/multispeaker-clean.metadata.json +434 -0
- rvcbench/suites/core_v1/multispeaker-clean.selection.json +346 -0
- rvcbench/suites/core_v1/multispeaker.metadata.json +434 -0
- rvcbench/suites/core_v1/multispeaker.selection.json +346 -0
- rvcbench/suites/core_v1/suite.json +402 -0
- rvcbench/suites/core_v1/textshift-hallucination.metadata.json +386 -0
- rvcbench/suites/core_v1/textshift-hallucination.selection.json +346 -0
- rvcbench/suites/core_v1/textshift-scam-standard.metadata.json +342 -0
- rvcbench/suites/core_v1/textshift-scam-standard.selection.json +290 -0
- rvcbench/suites/core_v1/textshift-scam.metadata.json +342 -0
- rvcbench/suites/core_v1/textshift-scam.selection.json +290 -0
- rvcbench/suites/core_v1/textshift-standard.metadata.json +386 -0
- rvcbench/suites/core_v1/textshift-standard.selection.json +346 -0
- rvcbench/suites/full_v1/audioshift.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/audioshift.selection.json.gz +0 -0
- rvcbench/suites/full_v1/background-clean.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/background-clean.selection.json.gz +0 -0
- rvcbench/suites/full_v1/background.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/background.selection.json.gz +0 -0
- rvcbench/suites/full_v1/chinese.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/chinese.selection.json.gz +0 -0
- rvcbench/suites/full_v1/crosslingual.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/crosslingual.selection.json.gz +0 -0
- rvcbench/suites/full_v1/english-libritts.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/english-libritts.selection.json.gz +0 -0
- rvcbench/suites/full_v1/french.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/french.selection.json.gz +0 -0
- rvcbench/suites/full_v1/longaudio.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/longaudio.selection.json.gz +0 -0
- rvcbench/suites/full_v1/longtext.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/longtext.selection.json.gz +0 -0
- rvcbench/suites/full_v1/multispeaker-clean.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/multispeaker-clean.selection.json.gz +0 -0
- rvcbench/suites/full_v1/multispeaker.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/multispeaker.selection.json.gz +0 -0
- rvcbench/suites/full_v1/suite.json +334 -0
- rvcbench/suites/full_v1/textshift-hallucination.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-hallucination.selection.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-scam-standard.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-scam-standard.selection.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-scam.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-scam.selection.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-standard.metadata.json.gz +0 -0
- rvcbench/suites/full_v1/textshift-standard.selection.json.gz +0 -0
- rvcbench/suites/onboarding_v1/libritts16_v1.metadata.json +258 -0
- rvcbench/suites/onboarding_v1/libritts16_v1.selection.json +246 -0
- rvcbench/suites/onboarding_v1/robotcall20_v1.metadata.json +322 -0
- rvcbench/suites/onboarding_v1/robotcall20_v1.selection.json +301 -0
- rvcbench/suites/onboarding_v1/suite.json +23 -0
- rvcbench/suites/onboarding_v1/vctk16_v1.metadata.json +258 -0
- rvcbench/suites/onboarding_v1/vctk16_v1.selection.json +245 -0
- rvcbench/trainers/__init__.py +1 -0
- rvcbench/trainers/bertvits2_trainer.py +239 -0
- rvcbench/utils/__init__.py +0 -0
- rvcbench/utils/commons.py +231 -0
- rvcbench/utils/env.py +42 -0
- rvcbench/utils/hub.py +19 -0
- rvcbench/utils/logger.py +69 -0
- rvcbench/utils/runtime_errors.py +8 -0
- rvcbench/utils/seeding.py +63 -0
- rvcbench/workflows/__init__.py +0 -0
- rvcbench/workflows/vc.py +358 -0
- rvcbench-2.0.0.dist-info/METADATA +295 -0
- rvcbench-2.0.0.dist-info/RECORD +743 -0
- rvcbench-2.0.0.dist-info/WHEEL +5 -0
- rvcbench-2.0.0.dist-info/entry_points.txt +2 -0
- rvcbench-2.0.0.dist-info/licenses/LICENSE +121 -0
- rvcbench-2.0.0.dist-info/licenses/src/rvcbench/protection/safespeech/original_code/LICENSE +21 -0
- rvcbench-2.0.0.dist-info/top_level.txt +1 -0
rvcbench/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""RVCBench: a benchmark for voice cloning robustness and audio protection."""
|
|
2
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
3
|
+
|
|
4
|
+
try:
|
|
5
|
+
__version__ = version("rvcbench")
|
|
6
|
+
except PackageNotFoundError: # source checkout that has not been installed
|
|
7
|
+
__version__ = "0+unknown"
|
|
8
|
+
|
|
9
|
+
__all__ = ["VoiceCloningAdapter", "__version__"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def __getattr__(name):
|
|
13
|
+
# Imported on demand so that `import rvcbench` stays free of heavy dependencies.
|
|
14
|
+
if name == "VoiceCloningAdapter":
|
|
15
|
+
from rvcbench.adapter import VoiceCloningAdapter
|
|
16
|
+
return VoiceCloningAdapter
|
|
17
|
+
raise AttributeError(f"module 'rvcbench' has no attribute {name!r}")
|
rvcbench/adapter.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Public base class for evaluating your own voice cloning model with RVCBench.
|
|
2
|
+
|
|
3
|
+
Subclass :class:`VoiceCloningAdapter`, implement :meth:`clone`, and select the class
|
|
4
|
+
with ``vc.adapter=your_package.your_module:YourAdapter``. The benchmark runner handles
|
|
5
|
+
sample selection, seeding, output naming, run records and scoring.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import time
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Optional, Tuple
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import soundfile as sf
|
|
15
|
+
|
|
16
|
+
from rvcbench.adversary.base_adversary import BaseAdversary
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class VoiceCloningAdapter(BaseAdversary):
|
|
20
|
+
"""Zero-shot voice cloning model driven one utterance at a time.
|
|
21
|
+
|
|
22
|
+
``self.config`` is the ``adversary`` block of the run config, ``self.device`` the
|
|
23
|
+
requested device and ``self.logger`` a standard logger.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
#: Label recorded with each synthesis time. Describe what :meth:`clone` measures.
|
|
27
|
+
timing_scope = "external_adapter_clone_call_excluding_output_write"
|
|
28
|
+
|
|
29
|
+
def __init__(self, config, dataset_config, device, logger):
|
|
30
|
+
super().__init__(config, device)
|
|
31
|
+
self.dataset_config = dataset_config
|
|
32
|
+
self.logger = logger
|
|
33
|
+
self._loaded = False
|
|
34
|
+
|
|
35
|
+
def load(self) -> None:
|
|
36
|
+
"""Load weights and other resources once per run. Optional."""
|
|
37
|
+
|
|
38
|
+
def clone(self, *, text: str, reference_audio: Path, reference_text: str,
|
|
39
|
+
language: Optional[str]) -> Tuple[np.ndarray, int]:
|
|
40
|
+
"""Speak ``text`` in the voice of ``reference_audio``.
|
|
41
|
+
|
|
42
|
+
Return ``(waveform, sample_rate)``: a mono float waveform in ``[-1, 1]`` and its
|
|
43
|
+
sample rate in Hz. Raise an exception when the utterance cannot be generated;
|
|
44
|
+
the runner records the failure for that sample and continues.
|
|
45
|
+
"""
|
|
46
|
+
raise NotImplementedError
|
|
47
|
+
|
|
48
|
+
def unload(self) -> None:
|
|
49
|
+
"""Release resources at the end of the run. Optional."""
|
|
50
|
+
|
|
51
|
+
# -- runner integration -------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
def _ensure_model(self) -> None:
|
|
54
|
+
if not self._loaded:
|
|
55
|
+
self.load()
|
|
56
|
+
self._loaded = True
|
|
57
|
+
|
|
58
|
+
def close(self) -> None:
|
|
59
|
+
try:
|
|
60
|
+
if self._loaded:
|
|
61
|
+
self.unload()
|
|
62
|
+
finally:
|
|
63
|
+
self._loaded = False
|
|
64
|
+
super().close()
|
|
65
|
+
|
|
66
|
+
def attack(self, *, output_path, dataset, protected_audio_path=None):
|
|
67
|
+
del protected_audio_path # protected references already replace sample.prompt_path
|
|
68
|
+
self._ensure_model()
|
|
69
|
+
output_dir = Path(output_path).resolve()
|
|
70
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
self._init_synthesis_timings(output_dir)
|
|
72
|
+
try:
|
|
73
|
+
for sample in dataset.get_zero_shot_samples(max_samples=self.config.get("max_samples")):
|
|
74
|
+
reference = self._resolve_prompt_path(sample)
|
|
75
|
+
if reference is None:
|
|
76
|
+
raise FileNotFoundError(f"Missing reference audio: {sample.prompt_path}")
|
|
77
|
+
text = (sample.target_text or "").strip()
|
|
78
|
+
if not text:
|
|
79
|
+
raise ValueError("Target text is empty")
|
|
80
|
+
started = time.perf_counter()
|
|
81
|
+
waveform, sample_rate = self.clone(
|
|
82
|
+
text=text, reference_audio=reference,
|
|
83
|
+
reference_text=(sample.prompt_text or "").strip(),
|
|
84
|
+
language=sample.target_language or sample.prompt_language)
|
|
85
|
+
elapsed = time.perf_counter() - started
|
|
86
|
+
waveform = np.asarray(waveform, dtype=np.float32)
|
|
87
|
+
if waveform.ndim == 2 and 1 in waveform.shape:
|
|
88
|
+
waveform = waveform.reshape(-1)
|
|
89
|
+
if waveform.ndim != 1 or not waveform.size or not np.isfinite(waveform).all():
|
|
90
|
+
raise ValueError("clone() must return nonempty finite mono audio")
|
|
91
|
+
if int(sample_rate) <= 0:
|
|
92
|
+
raise ValueError("clone() must return a positive sample rate")
|
|
93
|
+
path = (self._speaker_output_dir(output_dir, str(sample.speaker_id))
|
|
94
|
+
/ self._cloned_filename(sample, sample.index))
|
|
95
|
+
sf.write(str(path), np.clip(waveform, -1.0, 1.0), int(sample_rate))
|
|
96
|
+
self._record_synthesis_timing(path, elapsed)
|
|
97
|
+
finally:
|
|
98
|
+
self._flush_synthesis_timings()
|
|
File without changes
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
import time
|
|
5
|
+
from typing import TYPE_CHECKING, Optional
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
import soundfile as sf
|
|
9
|
+
from hydra.utils import to_absolute_path
|
|
10
|
+
|
|
11
|
+
from .base_adversary import BaseAdversary
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from rvcbench.models.bark_voice_clone import BarkVoiceCloneGenerator
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class BarkVoiceCloneZeroShotAdversary(BaseAdversary):
|
|
18
|
+
"""Runs the Bark voice cloning pipeline for zero-shot attacks."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, config, dataset_config, device, logger):
|
|
21
|
+
super().__init__(config, device)
|
|
22
|
+
self.dataset_config = dataset_config
|
|
23
|
+
self.logger = logger
|
|
24
|
+
|
|
25
|
+
self.code_path = Path(to_absolute_path(self.config.code_path)).resolve()
|
|
26
|
+
self.models_dir = self._resolve_optional_path(self.config.get("models_dir"))
|
|
27
|
+
self.cache_dir = self._resolve_optional_path(self.config.get("cache_dir"))
|
|
28
|
+
self.prompt_cache_dir = self._resolve_optional_path(self.config.get("prompt_cache_dir"))
|
|
29
|
+
self.hubert_checkpoint = self._resolve_optional_path(self.config.get("hubert_checkpoint"))
|
|
30
|
+
self.hubert_tokenizer = self._resolve_optional_path(self.config.get("hubert_tokenizer"))
|
|
31
|
+
|
|
32
|
+
self.reference_assignment = str(
|
|
33
|
+
self.config.get("reference_assignment", "round_robin")
|
|
34
|
+
).lower().strip()
|
|
35
|
+
self.max_samples = self.config.get("max_samples")
|
|
36
|
+
self.seed = self.config.get("seed")
|
|
37
|
+
self.default_prompt_text = str(
|
|
38
|
+
self.config.get(
|
|
39
|
+
"default_prompt_text",
|
|
40
|
+
"Here is a short sample of the desired voice.",
|
|
41
|
+
)
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
self._generator: Optional[BarkVoiceCloneGenerator] = None
|
|
45
|
+
|
|
46
|
+
# ------------------------------------------------------------------
|
|
47
|
+
# Helpers
|
|
48
|
+
# ------------------------------------------------------------------
|
|
49
|
+
def _resolve_optional_path(self, value) -> Optional[Path]:
|
|
50
|
+
if value in (None, ""):
|
|
51
|
+
return None
|
|
52
|
+
resolved = Path(to_absolute_path(str(value)))
|
|
53
|
+
return resolved.resolve()
|
|
54
|
+
|
|
55
|
+
def _coerce_optional(self, key: str, caster):
|
|
56
|
+
value = self.config.get(key)
|
|
57
|
+
if value in (None, ""):
|
|
58
|
+
return None
|
|
59
|
+
try:
|
|
60
|
+
return caster(value)
|
|
61
|
+
except (TypeError, ValueError):
|
|
62
|
+
if self.logger is not None:
|
|
63
|
+
self.logger.warning(
|
|
64
|
+
"[BarkVC] Failed to cast '%s' value '%s'; ignoring.",
|
|
65
|
+
key,
|
|
66
|
+
value,
|
|
67
|
+
)
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
def _ensure_generator(self) -> None:
|
|
71
|
+
if self._generator is not None:
|
|
72
|
+
return
|
|
73
|
+
from rvcbench.models.bark_voice_clone import BarkVoiceCloneGenerator, BarkVoiceCloneGeneratorConfig
|
|
74
|
+
|
|
75
|
+
generator_config = BarkVoiceCloneGeneratorConfig(
|
|
76
|
+
code_path=self.code_path,
|
|
77
|
+
models_dir=self.models_dir,
|
|
78
|
+
cache_dir=self.cache_dir,
|
|
79
|
+
prompt_cache_dir=self.prompt_cache_dir,
|
|
80
|
+
hubert_checkpoint=self.hubert_checkpoint,
|
|
81
|
+
hubert_tokenizer=self.hubert_tokenizer,
|
|
82
|
+
text_tokenizer_path=self._resolve_optional_path(self.config.get('text_tokenizer_path')),
|
|
83
|
+
text_temperature=float(self.config.get("text_temperature", 0.7)),
|
|
84
|
+
text_top_k=self._coerce_optional("text_top_k", int),
|
|
85
|
+
text_top_p=self._coerce_optional("text_top_p", float),
|
|
86
|
+
coarse_temperature=float(self.config.get("coarse_temperature", 0.7)),
|
|
87
|
+
coarse_top_k=self._coerce_optional("coarse_top_k", int),
|
|
88
|
+
coarse_top_p=self._coerce_optional("coarse_top_p", float),
|
|
89
|
+
fine_temperature=float(self.config.get("fine_temperature", 0.5)),
|
|
90
|
+
semantic_use_kv_cache=bool(self.config.get("semantic_use_kv_cache", True)),
|
|
91
|
+
coarse_use_kv_cache=bool(self.config.get("coarse_use_kv_cache", True)),
|
|
92
|
+
silent=bool(self.config.get("silent", True)),
|
|
93
|
+
force_reload_models=bool(self.config.get("force_reload_models", False)),
|
|
94
|
+
max_prompt_seconds=self._coerce_optional("max_prompt_seconds", float),
|
|
95
|
+
hubert_layer=int(self.config.get('hubert_layer', 9)),
|
|
96
|
+
)
|
|
97
|
+
self._generator = BarkVoiceCloneGenerator(generator_config, self.device, self.logger)
|
|
98
|
+
|
|
99
|
+
# ------------------------------------------------------------------
|
|
100
|
+
# Public API
|
|
101
|
+
# ------------------------------------------------------------------
|
|
102
|
+
def generate_sample(self, sample, *, output_dir):
|
|
103
|
+
reference_path = self._resolve_prompt_path(sample)
|
|
104
|
+
if reference_path is None:
|
|
105
|
+
raise FileNotFoundError(f"Missing Bark reference audio: {sample.prompt_path}")
|
|
106
|
+
text = (sample.target_text or "").strip()
|
|
107
|
+
if not text:
|
|
108
|
+
raise ValueError('Bark requires nonempty target text')
|
|
109
|
+
self._ensure_generator()
|
|
110
|
+
self._generator.ensure_model()
|
|
111
|
+
seed = self._sample_seed(sample)
|
|
112
|
+
if seed is not None:
|
|
113
|
+
from rvcbench.utils.seeding import configure_seeds
|
|
114
|
+
configure_seeds(seed, logger=None)
|
|
115
|
+
self._log_clone_request("BarkVC", 0, 1, str(sample.speaker_id), reference_path,
|
|
116
|
+
text, prompt_transcript=sample.prompt_text or self.default_prompt_text)
|
|
117
|
+
started = time.perf_counter()
|
|
118
|
+
audio, sample_rate = self._generator.generate(
|
|
119
|
+
text=text, prompt_audio=reference_path, sample_index=sample.index)
|
|
120
|
+
elapsed = time.perf_counter() - started
|
|
121
|
+
audio = np.asarray(audio, dtype=np.float32)
|
|
122
|
+
if audio.ndim != 1 or not audio.size or not np.isfinite(audio).all():
|
|
123
|
+
raise ValueError('Bark returned invalid mono audio')
|
|
124
|
+
path = self._speaker_output_dir(Path(output_dir).resolve(), str(sample.speaker_id)) / self._cloned_filename(sample, sample.index)
|
|
125
|
+
sf.write(path, audio, sample_rate)
|
|
126
|
+
return path, elapsed
|
|
127
|
+
|
|
128
|
+
def attack(self, *, output_path, dataset, protected_audio_path=None):
|
|
129
|
+
del protected_audio_path
|
|
130
|
+
output_dir = Path(output_path).resolve()
|
|
131
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
132
|
+
self._init_synthesis_timings(output_dir)
|
|
133
|
+
samples = dataset.get_zero_shot_samples(max_samples=self.max_samples)
|
|
134
|
+
if not samples:
|
|
135
|
+
raise RuntimeError("No zero-shot samples available for Bark voice cloning adversary.")
|
|
136
|
+
self._log_attack_plan("BarkVC", samples, self._count_available_prompts(samples))
|
|
137
|
+
completed = 0
|
|
138
|
+
try:
|
|
139
|
+
for sample in samples:
|
|
140
|
+
path, elapsed = self.generate_sample(sample, output_dir=output_dir)
|
|
141
|
+
self._record_synthesis_timing(path, elapsed)
|
|
142
|
+
completed += 1
|
|
143
|
+
finally:
|
|
144
|
+
self._flush_synthesis_timings()
|
|
145
|
+
if self.logger:
|
|
146
|
+
self.logger.info("[BarkVC] Generated %d/%d utterances.", completed, len(samples))
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
import csv
|
|
5
|
+
import re
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Optional, Sequence, TYPE_CHECKING
|
|
8
|
+
|
|
9
|
+
if TYPE_CHECKING:
|
|
10
|
+
from rvcbench.datasets.data_utils import AllSpeakerData
|
|
11
|
+
from rvcbench.datasets.data_utils import ZeroShotSample
|
|
12
|
+
|
|
13
|
+
class BaseAdversary(ABC):
|
|
14
|
+
"""Abstract Base Class for all malicious adversaries."""
|
|
15
|
+
def __init__(self, config, device):
|
|
16
|
+
# Hydra sometimes provides the full run config; extract the adversary block if present.
|
|
17
|
+
adversary_cfg = getattr(config, "adversary", None)
|
|
18
|
+
self.config = adversary_cfg or config
|
|
19
|
+
self.device = device
|
|
20
|
+
self._synthesis_timing_records = []
|
|
21
|
+
self._synthesis_timing_path: Optional[Path] = None
|
|
22
|
+
self._managed_lifetime = False
|
|
23
|
+
|
|
24
|
+
def _sample_seed(self, sample):
|
|
25
|
+
seed = self.config.get('seed')
|
|
26
|
+
return None if seed is None else int(seed) + int(sample.index)
|
|
27
|
+
|
|
28
|
+
def prepare(self):
|
|
29
|
+
"""Load adapter resources once so setup failures are run-level failures."""
|
|
30
|
+
self._managed_lifetime = True
|
|
31
|
+
for name in ('_ensure_generator', '_ensure_synthesizer', '_ensure_model', '_ensure_imports'):
|
|
32
|
+
hook = getattr(self, name, None)
|
|
33
|
+
if hook is not None:
|
|
34
|
+
hook()
|
|
35
|
+
break
|
|
36
|
+
for name in ('_generator', '_synthesizer'):
|
|
37
|
+
resource = getattr(self, name, None)
|
|
38
|
+
ensure = getattr(resource, 'ensure_model', None)
|
|
39
|
+
if ensure is not None:
|
|
40
|
+
ensure()
|
|
41
|
+
|
|
42
|
+
def close(self) -> None:
|
|
43
|
+
"""Release owned generators, including persistent subprocess workers."""
|
|
44
|
+
errors = []
|
|
45
|
+
for name in ('_generator', '_synthesizer'):
|
|
46
|
+
resource = getattr(self, name, None)
|
|
47
|
+
if resource is None:
|
|
48
|
+
continue
|
|
49
|
+
try:
|
|
50
|
+
close = getattr(resource, 'close', None)
|
|
51
|
+
if close is not None:
|
|
52
|
+
close()
|
|
53
|
+
except Exception as exc:
|
|
54
|
+
errors.append(exc)
|
|
55
|
+
else:
|
|
56
|
+
setattr(self, name, None)
|
|
57
|
+
if errors:
|
|
58
|
+
raise RuntimeError('Failed to close adapter resources') from errors[0]
|
|
59
|
+
self._managed_lifetime = False
|
|
60
|
+
|
|
61
|
+
def _speaker_slug(self, speaker_id: str) -> str:
|
|
62
|
+
"""Return a filesystem-safe identifier for the supplied speaker identifier."""
|
|
63
|
+
token = str(speaker_id or "").strip()
|
|
64
|
+
slug = re.sub(r"[^0-9A-Za-z_.-]", "_", token)
|
|
65
|
+
return slug or "unknown"
|
|
66
|
+
|
|
67
|
+
def _speaker_output_dir(self, base_dir: Path, speaker_id: str) -> Path:
|
|
68
|
+
"""Ensure the per-speaker directory exists and return its path."""
|
|
69
|
+
speaker_dir = Path(base_dir) / self._speaker_slug(speaker_id)
|
|
70
|
+
speaker_dir.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
return speaker_dir
|
|
72
|
+
|
|
73
|
+
def _resolve_prompt_path(self, sample) -> Optional[Path]:
|
|
74
|
+
raw_path = getattr(sample, "prompt_path", None)
|
|
75
|
+
if raw_path in (None, ""):
|
|
76
|
+
return None
|
|
77
|
+
path = Path(str(raw_path))
|
|
78
|
+
try:
|
|
79
|
+
candidate = path if path.is_absolute() else path.resolve(strict=False)
|
|
80
|
+
except Exception:
|
|
81
|
+
candidate = path
|
|
82
|
+
if candidate.exists():
|
|
83
|
+
return candidate
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
def _count_available_prompts(self, samples: Sequence["ZeroShotSample"]) -> int:
|
|
87
|
+
return sum(1 for sample in samples if self._resolve_prompt_path(sample) is not None)
|
|
88
|
+
|
|
89
|
+
def _preview_text(self, text: Optional[str], limit: int = 120) -> str:
|
|
90
|
+
snippet = re.sub(r"\s+", " ", (text or "").strip())
|
|
91
|
+
if len(snippet) > limit:
|
|
92
|
+
return snippet[: limit - 1] + "…"
|
|
93
|
+
return snippet or "<empty>"
|
|
94
|
+
|
|
95
|
+
def _describe_audio(self, path: Optional[Path]) -> str:
|
|
96
|
+
if path is None:
|
|
97
|
+
return "<none>"
|
|
98
|
+
try:
|
|
99
|
+
import soundfile as sf # Local import to avoid hard dependency when unused
|
|
100
|
+
|
|
101
|
+
with sf.SoundFile(str(path)) as handle:
|
|
102
|
+
duration = handle.frames / handle.samplerate if handle.samplerate else 0.0
|
|
103
|
+
return f"{path.name} (sr={handle.samplerate}, dur={duration:.2f}s)"
|
|
104
|
+
except Exception:
|
|
105
|
+
return path.name
|
|
106
|
+
|
|
107
|
+
def _log_attack_plan(
|
|
108
|
+
self,
|
|
109
|
+
model_label: str,
|
|
110
|
+
samples: Sequence["ZeroShotSample"],
|
|
111
|
+
prompt_count: int,
|
|
112
|
+
) -> None:
|
|
113
|
+
logger = getattr(self, "logger", None)
|
|
114
|
+
if not logger:
|
|
115
|
+
return
|
|
116
|
+
total = len(samples)
|
|
117
|
+
speakers = [str(sample.speaker_id) for sample in samples]
|
|
118
|
+
unique_speakers = sorted(set(speakers))
|
|
119
|
+
logger.info(
|
|
120
|
+
"[%s] Preparing %d utterances across %d speakers. Prompt audios available: %d.",
|
|
121
|
+
model_label,
|
|
122
|
+
total,
|
|
123
|
+
len(unique_speakers),
|
|
124
|
+
prompt_count,
|
|
125
|
+
)
|
|
126
|
+
if unique_speakers:
|
|
127
|
+
preview = ", ".join(unique_speakers[:8])
|
|
128
|
+
if len(unique_speakers) > 8:
|
|
129
|
+
preview += ", …"
|
|
130
|
+
logger.debug("[%s] Speaker roster: %s", model_label, preview)
|
|
131
|
+
|
|
132
|
+
def _log_clone_request(
|
|
133
|
+
self,
|
|
134
|
+
model_label: str,
|
|
135
|
+
index: int,
|
|
136
|
+
total: int,
|
|
137
|
+
speaker_id: str,
|
|
138
|
+
prompt_path: Optional[Path],
|
|
139
|
+
spoken_text: Optional[str],
|
|
140
|
+
*,
|
|
141
|
+
prompt_transcript: Optional[str] = None,
|
|
142
|
+
) -> None:
|
|
143
|
+
logger = getattr(self, "logger", None)
|
|
144
|
+
if not logger:
|
|
145
|
+
return
|
|
146
|
+
prompt_desc = self._describe_audio(prompt_path)
|
|
147
|
+
text_preview = self._preview_text(spoken_text)
|
|
148
|
+
message = (
|
|
149
|
+
f"[{model_label}] [{index + 1}/{total}] speaker={speaker_id} "
|
|
150
|
+
f"prompt={prompt_desc} text=\"{text_preview}\""
|
|
151
|
+
)
|
|
152
|
+
if prompt_transcript:
|
|
153
|
+
message += f" transcript=\"{self._preview_text(prompt_transcript)}\""
|
|
154
|
+
logger.info(message)
|
|
155
|
+
|
|
156
|
+
def _cloned_filename(self, sample, idx: int, suffix: str = "cloned") -> str:
|
|
157
|
+
from rvcbench.benchmark.artifacts import output_path
|
|
158
|
+
return output_path(Path('.'), sample, suffix).name
|
|
159
|
+
|
|
160
|
+
def _init_synthesis_timings(self, output_dir: Path) -> None:
|
|
161
|
+
self._synthesis_timing_records = []
|
|
162
|
+
self._synthesis_timing_path = Path(output_dir) / "synthesis_timings.csv"
|
|
163
|
+
|
|
164
|
+
def _record_synthesis_timing(self, generated_path: Path, elapsed_sec: Optional[float]) -> None:
|
|
165
|
+
if elapsed_sec is None:
|
|
166
|
+
return
|
|
167
|
+
if self._synthesis_timing_path is None:
|
|
168
|
+
return
|
|
169
|
+
try:
|
|
170
|
+
resolved = generated_path.resolve()
|
|
171
|
+
except Exception:
|
|
172
|
+
resolved = generated_path
|
|
173
|
+
self._synthesis_timing_records.append(
|
|
174
|
+
{
|
|
175
|
+
"generated_path": str(resolved),
|
|
176
|
+
"synthesis_time_sec": float(elapsed_sec),
|
|
177
|
+
"timing_scope": getattr(self, "timing_scope", "adapter_reported_synthesis"),
|
|
178
|
+
}
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
def _flush_synthesis_timings(self) -> None:
|
|
182
|
+
if not self._synthesis_timing_path or not self._synthesis_timing_records:
|
|
183
|
+
return
|
|
184
|
+
self._synthesis_timing_path.parent.mkdir(parents=True, exist_ok=True)
|
|
185
|
+
with open(self._synthesis_timing_path, "w", encoding="utf-8", newline="") as csvfile:
|
|
186
|
+
writer = csv.DictWriter(
|
|
187
|
+
csvfile,
|
|
188
|
+
fieldnames=["generated_path", "synthesis_time_sec", "timing_scope"],
|
|
189
|
+
)
|
|
190
|
+
writer.writeheader()
|
|
191
|
+
writer.writerows(self._synthesis_timing_records)
|
|
192
|
+
logger = getattr(self, "logger", None)
|
|
193
|
+
if logger is not None:
|
|
194
|
+
logger.info("Saved per-sample synthesis timings to %s", self._synthesis_timing_path)
|
|
195
|
+
|
|
196
|
+
@abstractmethod
|
|
197
|
+
def attack(
|
|
198
|
+
self,
|
|
199
|
+
*,
|
|
200
|
+
output_path: str,
|
|
201
|
+
dataset: "AllSpeakerData",
|
|
202
|
+
protected_audio_path: Optional[str] = None,
|
|
203
|
+
) -> None:
|
|
204
|
+
"""Generate adversarial audio for the provided dataset."""
|
|
205
|
+
pass
|