lfeats 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lfeats-0.2.0 → lfeats-0.2.1}/.gitignore +1 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/PKG-INFO +21 -5
- {lfeats-0.2.0 → lfeats-0.2.1}/README.md +17 -2
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/cli.py +4 -25
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/types.py +57 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/__init__.py +4 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/base.py +52 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/ecapa_tdnn.py +4 -2
- lfeats-0.2.1/lfeats/models/higgs_audio.py +136 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/next_tdnn.py +4 -2
- lfeats-0.2.1/lfeats/models/redimnet.py +106 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/x_vector.py +4 -2
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/__init__.py +2 -0
- lfeats-0.2.1/lfeats/resamplers/scipy.py +90 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/io.py +7 -2
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/validation.py +27 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/version.py +1 -1
- {lfeats-0.2.0 → lfeats-0.2.1}/pyproject.toml +3 -2
- {lfeats-0.2.0 → lfeats-0.2.1}/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/extractor.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/resampler.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/contentvec.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/data2vec.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/data2vec2.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/emotion2vec.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/emotion2vec_plus.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/hubert.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/manager.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/r_spin.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/r_vector.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/spidr.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/spin.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/sslzip.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/unispeech_sat.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/wav2vec2.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/wavlm.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/whisper.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/base.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/lilfilter.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/manager.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/soxr.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/torchaudio.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/checkpoint_utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/config.yaml +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/dictionary.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/modality.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/text_compressor.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/configs.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/constants.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/initialize.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/file_io.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/incremental_decoding_utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/meters.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec2.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec_audio.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/audio.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/base.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/modules.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_decoder.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_encoder.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_incremental_decoder.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_model.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/hubert.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/wav2vec2.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/ema_module.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fairseq_dropout.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fp32_group_norm.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gelu.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gumbel_vector_quantizer.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/layer_norm.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/multihead_attention.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/quant_noise.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/same_pad.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/transpose_last.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/quantization_utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/registry.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/audio_pretraining.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/fairseq_task.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/hubert_pretraining.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tokenizer.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/SpeakerNet.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/vap_bn_tanh_fc_bn.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7_cyclical_lr_step.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_light_C256_B3_K65.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/main.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/NeXt_TDNN.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt_light.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/mel_transform.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/model.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/wavlm_config.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/convert.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/hubert_model.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/wav2vec2_model.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/WavLM.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/modules.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/download.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/dataio.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/encoder.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/preprocess.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/classifiers.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/interfaces.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/features.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ECAPA_TDNN.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ResNet.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/Xvector.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/CNN.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/containers.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/linear.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/normalization.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/pooling.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/features.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/_workarounds.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/autocast.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/checkpoints.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/distributed.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/fetching.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/filter_analysis.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/logger.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/parameter_transfer.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/run_opts.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/base.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/spin.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/dnn.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/hubert.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/swav_vq_dis.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/wavlm.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/model_utils.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/padding.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/LICENSE +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/drop.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/helpers.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/mlp.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/__init__.py +0 -0
- {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/paths.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lfeats
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A unified interface to extract hidden representations from speech foundation models
|
|
5
5
|
Project-URL: Documentation, https://takenori-y.github.io/lfeats/stable/
|
|
6
6
|
Project-URL: Source, https://github.com/takenori-y/lfeats
|
|
@@ -28,11 +28,12 @@ Requires-Dist: onnxruntime>=1.19.0
|
|
|
28
28
|
Requires-Dist: parfive>=2.1.0
|
|
29
29
|
Requires-Dist: platformdirs>=2.0.0
|
|
30
30
|
Requires-Dist: requests>=2.27.0
|
|
31
|
+
Requires-Dist: scipy>=1.9.2
|
|
31
32
|
Requires-Dist: soundfile>=0.10.2
|
|
32
33
|
Requires-Dist: soxr>=0.4.0
|
|
33
34
|
Requires-Dist: torch>=2.6.0
|
|
34
35
|
Requires-Dist: torchaudio>=2.6.0
|
|
35
|
-
Requires-Dist: transformers>=
|
|
36
|
+
Requires-Dist: transformers>=5.3.0
|
|
36
37
|
Provides-Extra: dev
|
|
37
38
|
Requires-Dist: build; extra == 'dev'
|
|
38
39
|
Requires-Dist: matplotlib; extra == 'dev'
|
|
@@ -40,7 +41,7 @@ Requires-Dist: mdformat; extra == 'dev'
|
|
|
40
41
|
Requires-Dist: numpydoc; extra == 'dev'
|
|
41
42
|
Requires-Dist: pkginfo; extra == 'dev'
|
|
42
43
|
Requires-Dist: pydata-sphinx-theme; extra == 'dev'
|
|
43
|
-
Requires-Dist: pyright; extra == 'dev'
|
|
44
|
+
Requires-Dist: pyright<=1.1.408; extra == 'dev'
|
|
44
45
|
Requires-Dist: pytest; extra == 'dev'
|
|
45
46
|
Requires-Dist: pytest-cov; extra == 'dev'
|
|
46
47
|
Requires-Dist: ruff; extra == 'dev'
|
|
@@ -143,6 +144,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
143
144
|
| | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
|
|
144
145
|
| | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
|
|
145
146
|
|
|
147
|
+
### Token-Level Features
|
|
148
|
+
|
|
149
|
+
| Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
|
|
150
|
+
| :--- | :--- | ---: | ---: | :---: | :---: | :---: |
|
|
151
|
+
| `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
|
|
152
|
+
|
|
146
153
|
### Utterance-Level Features
|
|
147
154
|
|
|
148
155
|
| Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
|
|
@@ -151,7 +158,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
151
158
|
| `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
|
|
152
159
|
| | `base` | 0 | 192 | | | |
|
|
153
160
|
| | `base-v2` | 0 | 192 | | | |
|
|
154
|
-
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/
|
|
161
|
+
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
|
|
162
|
+
| `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
|
|
163
|
+
| | `b1` | 0 | 192 | | | |
|
|
164
|
+
| | `b2` | 0 | 192 | | | |
|
|
165
|
+
| | `b3` | 0 | 192 | | | |
|
|
166
|
+
| | `b4` | 0 | 192 | | | |
|
|
167
|
+
| | `b5` | 0 | 192 | | | |
|
|
168
|
+
| | `b6` | 0 | 192 | | | |
|
|
155
169
|
| `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
|
|
156
170
|
|
|
157
171
|
> [!IMPORTANT]
|
|
@@ -163,12 +177,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
163
177
|
| Resampler Type | Quality Preset | Source | License |
|
|
164
178
|
| :--- | :--- | :---: | :--- |
|
|
165
179
|
| `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
|
|
180
|
+
| `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
|
|
181
|
+
| | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
|
|
166
182
|
| `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
|
|
167
183
|
| | `low` | | |
|
|
168
184
|
| | `medium` | | |
|
|
169
185
|
| | `high` | | |
|
|
170
186
|
| | `very-high` | | |
|
|
171
|
-
| `torchaudio` | `kaiser-fast` | [
|
|
187
|
+
| `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
|
|
172
188
|
| | `kaiser-best` | | |
|
|
173
189
|
|
|
174
190
|
## Examples
|
|
@@ -93,6 +93,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
93
93
|
| | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
|
|
94
94
|
| | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
|
|
95
95
|
|
|
96
|
+
### Token-Level Features
|
|
97
|
+
|
|
98
|
+
| Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
|
|
99
|
+
| :--- | :--- | ---: | ---: | :---: | :---: | :---: |
|
|
100
|
+
| `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
|
|
101
|
+
|
|
96
102
|
### Utterance-Level Features
|
|
97
103
|
|
|
98
104
|
| Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
|
|
@@ -101,7 +107,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
101
107
|
| `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
|
|
102
108
|
| | `base` | 0 | 192 | | | |
|
|
103
109
|
| | `base-v2` | 0 | 192 | | | |
|
|
104
|
-
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/
|
|
110
|
+
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
|
|
111
|
+
| `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
|
|
112
|
+
| | `b1` | 0 | 192 | | | |
|
|
113
|
+
| | `b2` | 0 | 192 | | | |
|
|
114
|
+
| | `b3` | 0 | 192 | | | |
|
|
115
|
+
| | `b4` | 0 | 192 | | | |
|
|
116
|
+
| | `b5` | 0 | 192 | | | |
|
|
117
|
+
| | `b6` | 0 | 192 | | | |
|
|
105
118
|
| `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
|
|
106
119
|
|
|
107
120
|
> [!IMPORTANT]
|
|
@@ -113,12 +126,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
113
126
|
| Resampler Type | Quality Preset | Source | License |
|
|
114
127
|
| :--- | :--- | :---: | :--- |
|
|
115
128
|
| `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
|
|
129
|
+
| `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
|
|
130
|
+
| | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
|
|
116
131
|
| `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
|
|
117
132
|
| | `low` | | |
|
|
118
133
|
| | `medium` | | |
|
|
119
134
|
| | `high` | | |
|
|
120
135
|
| | `very-high` | | |
|
|
121
|
-
| `torchaudio` | `kaiser-fast` | [
|
|
136
|
+
| `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
|
|
122
137
|
| | `kaiser-best` | | |
|
|
123
138
|
|
|
124
139
|
## Examples
|
|
@@ -147,9 +147,6 @@ def main() -> None:
|
|
|
147
147
|
datefmt="%Y-%m-%d %H:%M:%S",
|
|
148
148
|
)
|
|
149
149
|
|
|
150
|
-
import numpy as np
|
|
151
|
-
import torch
|
|
152
|
-
|
|
153
150
|
import lfeats
|
|
154
151
|
from lfeats.utils.io import load_audio
|
|
155
152
|
|
|
@@ -170,7 +167,8 @@ def main() -> None:
|
|
|
170
167
|
raise ValueError(f"Invalid source: {args.source}")
|
|
171
168
|
|
|
172
169
|
if len(input_files) == 0:
|
|
173
|
-
|
|
170
|
+
logging.info(f"No audio files found in the source: {args.source}")
|
|
171
|
+
sys.exit(0)
|
|
174
172
|
logger.info(f"Found {len(input_files)} audio files to process.")
|
|
175
173
|
|
|
176
174
|
# Parse the layers argument.
|
|
@@ -216,7 +214,7 @@ def main() -> None:
|
|
|
216
214
|
path = Path(input_file).parent
|
|
217
215
|
# Remove the root part of the path.
|
|
218
216
|
dirs = path.relative_to(path.anchor).parts
|
|
219
|
-
if args.subdir_offset
|
|
217
|
+
if args.subdir_offset > len(dirs):
|
|
220
218
|
logger.error(
|
|
221
219
|
f"Subdir offset {args.subdir_offset} is too large for file: "
|
|
222
220
|
f"{input_file}. Skipping."
|
|
@@ -252,26 +250,7 @@ def main() -> None:
|
|
|
252
250
|
num_errors += 1
|
|
253
251
|
continue
|
|
254
252
|
|
|
255
|
-
|
|
256
|
-
result = {
|
|
257
|
-
"features": features.array,
|
|
258
|
-
"source": features.source,
|
|
259
|
-
"layers": features.layers,
|
|
260
|
-
}
|
|
261
|
-
np.savez_compressed(output_file, **result)
|
|
262
|
-
elif args.output_format == "pt":
|
|
263
|
-
result = {
|
|
264
|
-
"features": features.tensor.cpu(),
|
|
265
|
-
"source": features.source,
|
|
266
|
-
"layers": features.layers,
|
|
267
|
-
}
|
|
268
|
-
torch.save(result, output_file)
|
|
269
|
-
elif args.output_format == "float":
|
|
270
|
-
features.array.tofile(output_file)
|
|
271
|
-
elif args.output_format == "double":
|
|
272
|
-
features.array.astype(np.float64).tofile(output_file)
|
|
273
|
-
else:
|
|
274
|
-
raise ValueError(f"Unsupported output format: {args.output_format}")
|
|
253
|
+
features.tofile(output_file, double=args.output_format == "double")
|
|
275
254
|
|
|
276
255
|
if num_errors > 0:
|
|
277
256
|
logger.error(f"{num_errors} files were skipped due to errors.")
|
|
@@ -9,6 +9,7 @@ from dataclasses import dataclass
|
|
|
9
9
|
from enum import Enum
|
|
10
10
|
|
|
11
11
|
import numpy as np
|
|
12
|
+
import soundfile as sf
|
|
12
13
|
import torch
|
|
13
14
|
import torch.nn.functional as F
|
|
14
15
|
|
|
@@ -164,6 +165,28 @@ class Audio(Container):
|
|
|
164
165
|
"""
|
|
165
166
|
return self.data.shape[1]
|
|
166
167
|
|
|
168
|
+
def tofile(self, path: str) -> None:
|
|
169
|
+
"""Save the audio to a file.
|
|
170
|
+
|
|
171
|
+
Parameters
|
|
172
|
+
----------
|
|
173
|
+
path : str
|
|
174
|
+
The path to save the audio to.
|
|
175
|
+
|
|
176
|
+
"""
|
|
177
|
+
ext = path.split(".")[-1].lower()
|
|
178
|
+
if ext in ("wav", "flac"):
|
|
179
|
+
sf.write(path, self.array.T, self.sample_rate)
|
|
180
|
+
elif ext == "npz":
|
|
181
|
+
np.savez_compressed(path, samples=self.array, sample_rate=self.sample_rate)
|
|
182
|
+
elif ext == "pt":
|
|
183
|
+
torch.save(
|
|
184
|
+
{"samples": self.tensor.cpu(), "sample_rate": self.sample_rate},
|
|
185
|
+
path,
|
|
186
|
+
)
|
|
187
|
+
else:
|
|
188
|
+
self.array.tofile(path)
|
|
189
|
+
|
|
167
190
|
def normalize(self, eps: float = 1e-5) -> Audio:
|
|
168
191
|
"""Normalize the audio samples to have zero mean and unit variance.
|
|
169
192
|
|
|
@@ -237,6 +260,40 @@ class Features(Container):
|
|
|
237
260
|
"""
|
|
238
261
|
return self.data.shape[1]
|
|
239
262
|
|
|
263
|
+
def tofile(self, path: str, double: bool = False) -> None:
|
|
264
|
+
"""Save the features to a file.
|
|
265
|
+
|
|
266
|
+
Parameters
|
|
267
|
+
----------
|
|
268
|
+
path : str
|
|
269
|
+
The path to save the features to.
|
|
270
|
+
|
|
271
|
+
double : bool, optional
|
|
272
|
+
Whether to save the features in double precision instead of single one.
|
|
273
|
+
|
|
274
|
+
"""
|
|
275
|
+
ext = path.split(".")[-1].lower()
|
|
276
|
+
if ext == "npz":
|
|
277
|
+
np.savez_compressed(
|
|
278
|
+
path,
|
|
279
|
+
features=self.array.astype(np.float64 if double else np.float32),
|
|
280
|
+
source=self.source,
|
|
281
|
+
layers=self.layers or [],
|
|
282
|
+
)
|
|
283
|
+
elif ext == "pt":
|
|
284
|
+
torch.save(
|
|
285
|
+
{
|
|
286
|
+
"features": self.tensor.cpu().to(
|
|
287
|
+
torch.float64 if double else torch.float32
|
|
288
|
+
),
|
|
289
|
+
"source": self.source,
|
|
290
|
+
"layers": self.layers,
|
|
291
|
+
},
|
|
292
|
+
path,
|
|
293
|
+
)
|
|
294
|
+
else:
|
|
295
|
+
self.array.astype(np.float64 if double else np.float32).tofile(path)
|
|
296
|
+
|
|
240
297
|
def trim(self, start: int, end: int) -> Features:
|
|
241
298
|
"""Trim the features along the time dimension.
|
|
242
299
|
|
|
@@ -9,11 +9,13 @@ from .data2vec2 import Data2Vec2Model
|
|
|
9
9
|
from .ecapa_tdnn import EcapaTDNNModel
|
|
10
10
|
from .emotion2vec import Emotion2VecModel
|
|
11
11
|
from .emotion2vec_plus import Emotion2VecPlusModel
|
|
12
|
+
from .higgs_audio import HiggsAudioTokenizerModel
|
|
12
13
|
from .hubert import HuBERTModel
|
|
13
14
|
from .manager import ModelManager
|
|
14
15
|
from .next_tdnn import NeXtTDNNModel
|
|
15
16
|
from .r_spin import RSpinModel
|
|
16
17
|
from .r_vector import RVectorModel
|
|
18
|
+
from .redimnet import ReDimNetModel
|
|
17
19
|
from .spidr import SpidRModel
|
|
18
20
|
from .spin import SpinModel
|
|
19
21
|
from .sslzip import SSLZipModel
|
|
@@ -30,10 +32,12 @@ MODEL_MAP = {
|
|
|
30
32
|
"ecapa-tdnn": EcapaTDNNModel,
|
|
31
33
|
"emotion2vec": Emotion2VecModel,
|
|
32
34
|
"emotion2vec+": Emotion2VecPlusModel,
|
|
35
|
+
"higgs-audio": HiggsAudioTokenizerModel,
|
|
33
36
|
"hubert": HuBERTModel,
|
|
34
37
|
"next-tdnn": NeXtTDNNModel,
|
|
35
38
|
"r-spin": RSpinModel,
|
|
36
39
|
"r-vector": RVectorModel,
|
|
40
|
+
"redimnet": ReDimNetModel,
|
|
37
41
|
"spidr": SpidRModel,
|
|
38
42
|
"spin": SpinModel,
|
|
39
43
|
"sslzip": SSLZipModel,
|
|
@@ -256,6 +256,58 @@ class FrameLevelFeatureModel(BaseModel):
|
|
|
256
256
|
return Granularity.FRAME
|
|
257
257
|
|
|
258
258
|
|
|
259
|
+
class TokenLevelFeatureModel(BaseModel):
|
|
260
|
+
"""An abstract base class for frame-level feature extraction models."""
|
|
261
|
+
|
|
262
|
+
@property
|
|
263
|
+
def num_layers(self) -> int:
|
|
264
|
+
"""Get the number of layers in the model.
|
|
265
|
+
|
|
266
|
+
Returns
|
|
267
|
+
-------
|
|
268
|
+
out : int
|
|
269
|
+
The number of layers.
|
|
270
|
+
|
|
271
|
+
"""
|
|
272
|
+
return 0
|
|
273
|
+
|
|
274
|
+
@property
|
|
275
|
+
def frame_shift(self) -> int:
|
|
276
|
+
"""Get the frame shift of the model.
|
|
277
|
+
|
|
278
|
+
Returns
|
|
279
|
+
-------
|
|
280
|
+
out : int
|
|
281
|
+
The frame shift in samples.
|
|
282
|
+
|
|
283
|
+
"""
|
|
284
|
+
return int(40.0 * self.sample_rate / 1000)
|
|
285
|
+
|
|
286
|
+
@property
|
|
287
|
+
def center_offset(self) -> int:
|
|
288
|
+
"""Get the center offset of the model.
|
|
289
|
+
|
|
290
|
+
Returns
|
|
291
|
+
-------
|
|
292
|
+
out : int
|
|
293
|
+
The center offset in samples.
|
|
294
|
+
|
|
295
|
+
"""
|
|
296
|
+
return 0
|
|
297
|
+
|
|
298
|
+
@property
|
|
299
|
+
def granularity(self) -> Granularity:
|
|
300
|
+
"""Get the granularity of the features extracted by the model.
|
|
301
|
+
|
|
302
|
+
Returns
|
|
303
|
+
-------
|
|
304
|
+
out : str
|
|
305
|
+
The granularity of the features.
|
|
306
|
+
|
|
307
|
+
"""
|
|
308
|
+
return Granularity.FRAME
|
|
309
|
+
|
|
310
|
+
|
|
259
311
|
class UtteranceLevelFeatureModel(BaseModel):
|
|
260
312
|
"""An abstract base class for utterance-level feature extraction models."""
|
|
261
313
|
|
|
@@ -10,7 +10,7 @@ import torch
|
|
|
10
10
|
from ..interfaces.types import Audio, Features
|
|
11
11
|
from ..utils.io import silence_hf_hub
|
|
12
12
|
from ..utils.paths import setup_third_party_path
|
|
13
|
-
from ..utils.validation import validate_enum
|
|
13
|
+
from ..utils.validation import validate_enum, validate_length
|
|
14
14
|
from .base import UtteranceLevelFeatureModel
|
|
15
15
|
|
|
16
16
|
|
|
@@ -108,6 +108,8 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
|
|
|
108
108
|
raise RuntimeError("Model is not loaded. Call 'load' method first.")
|
|
109
109
|
|
|
110
110
|
with torch.inference_mode():
|
|
111
|
-
|
|
111
|
+
inputs = audio.tensor.to(self.device)
|
|
112
|
+
inputs = validate_length(inputs, 640)
|
|
113
|
+
vectors = self.model.encode_batch(inputs)
|
|
112
114
|
|
|
113
115
|
return Features(data=vectors, source=self.model_id)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Copyright (c) 2026 Takenori Yoshimura
|
|
2
|
+
# Released under the MIT License.
|
|
3
|
+
|
|
4
|
+
"""A module for the Higgs Audio tokenizer."""
|
|
5
|
+
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import torch
|
|
10
|
+
|
|
11
|
+
from ..interfaces.types import Audio, Features
|
|
12
|
+
from ..utils.io import silence_transformers
|
|
13
|
+
from ..utils.validation import validate_enum
|
|
14
|
+
from .base import TokenLevelFeatureModel
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class HiggsAudioTokenizerVariant(str, Enum):
|
|
18
|
+
"""Enumeration of supported Higgs Audio tokenizer variants."""
|
|
19
|
+
|
|
20
|
+
V2 = "v2"
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def model_name(self) -> str:
|
|
24
|
+
"""Return the model name corresponding to the variant.
|
|
25
|
+
|
|
26
|
+
Returns
|
|
27
|
+
-------
|
|
28
|
+
out : str
|
|
29
|
+
The model name corresponding to the variant.
|
|
30
|
+
|
|
31
|
+
"""
|
|
32
|
+
return f"eustlb/higgs-audio-{self.value}-tokenizer"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class HiggsAudioTokenizerModel(TokenLevelFeatureModel):
|
|
36
|
+
"""A class for the Higgs Audio tokenizer model."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
|
|
39
|
+
"""Initialize the Higgs Audio tokenizer model.
|
|
40
|
+
|
|
41
|
+
Parameters
|
|
42
|
+
----------
|
|
43
|
+
variant : str | None, optional
|
|
44
|
+
The variant of the model to use.
|
|
45
|
+
|
|
46
|
+
device : str, optional
|
|
47
|
+
The device to run the model on (e.g., 'cpu' or 'cuda').
|
|
48
|
+
|
|
49
|
+
"""
|
|
50
|
+
super().__init__(variant, device)
|
|
51
|
+
|
|
52
|
+
self.variant = validate_enum(
|
|
53
|
+
variant, HiggsAudioTokenizerVariant, HiggsAudioTokenizerVariant.V2
|
|
54
|
+
)
|
|
55
|
+
self._model_id = f"higgs-audio-{self.variant.value}"
|
|
56
|
+
|
|
57
|
+
self.feature_extractor = None
|
|
58
|
+
|
|
59
|
+
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
60
|
+
"""Load the model from the specified directory.
|
|
61
|
+
|
|
62
|
+
Parameters
|
|
63
|
+
----------
|
|
64
|
+
model_dir : str
|
|
65
|
+
The directory where the model checkpoint will be stored.
|
|
66
|
+
|
|
67
|
+
quiet : bool, optional
|
|
68
|
+
Whether to suppress output during the loading process.
|
|
69
|
+
|
|
70
|
+
"""
|
|
71
|
+
if self.model is not None:
|
|
72
|
+
return
|
|
73
|
+
|
|
74
|
+
from transformers import AutoFeatureExtractor, HiggsAudioV2TokenizerModel
|
|
75
|
+
|
|
76
|
+
with silence_transformers(quiet):
|
|
77
|
+
self.feature_extractor = AutoFeatureExtractor.from_pretrained(
|
|
78
|
+
self.variant.model_name, cache_dir=model_dir
|
|
79
|
+
)
|
|
80
|
+
self.model = HiggsAudioV2TokenizerModel.from_pretrained(
|
|
81
|
+
self.variant.model_name, cache_dir=model_dir
|
|
82
|
+
)
|
|
83
|
+
self.model.eval()
|
|
84
|
+
self.model.to(self.device) # type: ignore
|
|
85
|
+
|
|
86
|
+
def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
|
|
87
|
+
"""Extract features from the input audio using the model.
|
|
88
|
+
|
|
89
|
+
Parameters
|
|
90
|
+
----------
|
|
91
|
+
audio : Audio
|
|
92
|
+
The input audio data with shape (B, T).
|
|
93
|
+
|
|
94
|
+
layers : list[int]
|
|
95
|
+
The layer(s) from which to extract features.
|
|
96
|
+
|
|
97
|
+
Returns
|
|
98
|
+
-------
|
|
99
|
+
out : Features
|
|
100
|
+
The extracted features.
|
|
101
|
+
|
|
102
|
+
Raises
|
|
103
|
+
------
|
|
104
|
+
RuntimeError
|
|
105
|
+
If the model is not loaded.
|
|
106
|
+
|
|
107
|
+
"""
|
|
108
|
+
if self.feature_extractor is None or self.model is None:
|
|
109
|
+
raise RuntimeError("Model not loaded. Call 'load' method first.")
|
|
110
|
+
|
|
111
|
+
with torch.inference_mode():
|
|
112
|
+
inputs = self.feature_extractor(
|
|
113
|
+
raw_audio=[x for x in audio.array],
|
|
114
|
+
sampling_rate=self.feature_extractor.sampling_rate,
|
|
115
|
+
return_tensors="pt",
|
|
116
|
+
).to(self.device)
|
|
117
|
+
|
|
118
|
+
encoder_outputs: Any = self.model.encode(inputs["input_values"])
|
|
119
|
+
indices = encoder_outputs.audio_codes # (B, Q, N)
|
|
120
|
+
indices = indices.transpose(0, 1)
|
|
121
|
+
vectors = self.model.quantizer.decode(indices) # (B, D, N)
|
|
122
|
+
vectors = vectors.transpose(1, 2)
|
|
123
|
+
|
|
124
|
+
return Features(data=vectors, source=self.model_id)
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def sample_rate(self) -> int:
|
|
128
|
+
"""Get the sample rate required by the model.
|
|
129
|
+
|
|
130
|
+
Returns
|
|
131
|
+
-------
|
|
132
|
+
out : int
|
|
133
|
+
The sample rate in Hz.
|
|
134
|
+
|
|
135
|
+
"""
|
|
136
|
+
return 24000
|
|
@@ -11,7 +11,7 @@ import torch
|
|
|
11
11
|
from ..interfaces.types import Audio, Features
|
|
12
12
|
from ..utils.io import download_file
|
|
13
13
|
from ..utils.paths import setup_third_party_path
|
|
14
|
-
from ..utils.validation import validate_enum
|
|
14
|
+
from ..utils.validation import validate_enum, validate_length
|
|
15
15
|
from .base import UtteranceLevelFeatureModel
|
|
16
16
|
|
|
17
17
|
|
|
@@ -140,7 +140,9 @@ class NeXtTDNNModel(UtteranceLevelFeatureModel):
|
|
|
140
140
|
raise RuntimeError("Model is not loaded. Call 'load' method first.")
|
|
141
141
|
|
|
142
142
|
with torch.inference_mode():
|
|
143
|
-
|
|
143
|
+
inputs = audio.tensor.to(self.device)
|
|
144
|
+
inputs = validate_length(inputs, 640)
|
|
145
|
+
vectors = self.model(inputs)
|
|
144
146
|
vectors = vectors.unsqueeze(1)
|
|
145
147
|
|
|
146
148
|
return Features(data=vectors, source=self.model_id)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# Copyright (c) 2026 Takenori Yoshimura
|
|
2
|
+
# Released under the MIT License.
|
|
3
|
+
|
|
4
|
+
"""A module for the ReDimNet model."""
|
|
5
|
+
|
|
6
|
+
from enum import Enum
|
|
7
|
+
|
|
8
|
+
import torch
|
|
9
|
+
|
|
10
|
+
from ..interfaces.types import Audio, Features
|
|
11
|
+
from ..utils.io import safe_torch_hub_load
|
|
12
|
+
from ..utils.validation import validate_enum, validate_length
|
|
13
|
+
from .base import UtteranceLevelFeatureModel
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ReDimNetVariant(str, Enum):
|
|
17
|
+
"""Enumeration of supported ReDimNet model variants."""
|
|
18
|
+
|
|
19
|
+
B0 = "b0"
|
|
20
|
+
B1 = "b1"
|
|
21
|
+
B2 = "b2"
|
|
22
|
+
B3 = "b3"
|
|
23
|
+
B4 = "b4"
|
|
24
|
+
B5 = "b5"
|
|
25
|
+
B6 = "b6"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ReDimNetModel(UtteranceLevelFeatureModel):
|
|
29
|
+
"""A class for the ReDimNet model."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
|
|
32
|
+
"""Initialize the ReDimNet model.
|
|
33
|
+
|
|
34
|
+
Parameters
|
|
35
|
+
----------
|
|
36
|
+
variant : str | None, optional
|
|
37
|
+
The variant of the model to use.
|
|
38
|
+
|
|
39
|
+
device : str, optional
|
|
40
|
+
The device to run the model on (e.g., 'cpu' or 'cuda').
|
|
41
|
+
|
|
42
|
+
"""
|
|
43
|
+
super().__init__(variant, device)
|
|
44
|
+
|
|
45
|
+
self.variant = validate_enum(variant, ReDimNetVariant, ReDimNetVariant.B2)
|
|
46
|
+
self._model_id = f"redimnet-{self.variant.value}"
|
|
47
|
+
|
|
48
|
+
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
49
|
+
"""Load the model from the specified directory.
|
|
50
|
+
|
|
51
|
+
Parameters
|
|
52
|
+
----------
|
|
53
|
+
model_dir : str
|
|
54
|
+
The directory where the model checkpoint will be stored.
|
|
55
|
+
|
|
56
|
+
quiet : bool, optional
|
|
57
|
+
Whether to suppress output during the loading process.
|
|
58
|
+
|
|
59
|
+
"""
|
|
60
|
+
if self.model is not None:
|
|
61
|
+
return
|
|
62
|
+
|
|
63
|
+
self.model = safe_torch_hub_load(
|
|
64
|
+
"IDRnD/ReDimNet",
|
|
65
|
+
"ReDimNet",
|
|
66
|
+
model_dir,
|
|
67
|
+
quiet=quiet,
|
|
68
|
+
model_name=self.variant.value,
|
|
69
|
+
train_type="ft_lm",
|
|
70
|
+
dataset="vox2",
|
|
71
|
+
)
|
|
72
|
+
self.model.eval()
|
|
73
|
+
self.model.to(self.device)
|
|
74
|
+
|
|
75
|
+
def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
|
|
76
|
+
"""Extract features from the input audio using the model.
|
|
77
|
+
|
|
78
|
+
Parameters
|
|
79
|
+
----------
|
|
80
|
+
audio : Audio
|
|
81
|
+
The input audio data with shape (B, T).
|
|
82
|
+
|
|
83
|
+
layers : list[int]
|
|
84
|
+
The layer(s) from which to extract features.
|
|
85
|
+
|
|
86
|
+
Returns
|
|
87
|
+
-------
|
|
88
|
+
out : Features
|
|
89
|
+
The extracted features.
|
|
90
|
+
|
|
91
|
+
Raises
|
|
92
|
+
------
|
|
93
|
+
RuntimeError
|
|
94
|
+
If the model is not loaded.
|
|
95
|
+
|
|
96
|
+
"""
|
|
97
|
+
if self.model is None:
|
|
98
|
+
raise RuntimeError("Model is not loaded. Call 'load' method first.")
|
|
99
|
+
|
|
100
|
+
with torch.inference_mode():
|
|
101
|
+
inputs = audio.tensor.to(self.device)
|
|
102
|
+
inputs = validate_length(inputs, 320)
|
|
103
|
+
vectors = self.model(inputs)
|
|
104
|
+
vectors = vectors.unsqueeze(1)
|
|
105
|
+
|
|
106
|
+
return Features(data=vectors, source=self.model_id)
|
|
@@ -10,7 +10,7 @@ import torch
|
|
|
10
10
|
from ..interfaces.types import Audio, Features
|
|
11
11
|
from ..utils.io import silence_hf_hub
|
|
12
12
|
from ..utils.paths import setup_third_party_path
|
|
13
|
-
from ..utils.validation import validate_enum
|
|
13
|
+
from ..utils.validation import validate_enum, validate_length
|
|
14
14
|
from .base import UtteranceLevelFeatureModel
|
|
15
15
|
|
|
16
16
|
|
|
@@ -108,6 +108,8 @@ class XVectorModel(UtteranceLevelFeatureModel):
|
|
|
108
108
|
raise RuntimeError("Model is not loaded. Call 'load' method first.")
|
|
109
109
|
|
|
110
110
|
with torch.inference_mode():
|
|
111
|
-
|
|
111
|
+
inputs = audio.tensor.to(self.device)
|
|
112
|
+
inputs = validate_length(inputs, 640)
|
|
113
|
+
vectors = self.model.encode_batch(inputs)
|
|
112
114
|
|
|
113
115
|
return Features(data=vectors, source=self.model_id)
|
|
@@ -5,11 +5,13 @@
|
|
|
5
5
|
|
|
6
6
|
from .lilfilter import LilFilterResampler
|
|
7
7
|
from .manager import ResamplerManager
|
|
8
|
+
from .scipy import ScipyResampler
|
|
8
9
|
from .soxr import SoxrResampler
|
|
9
10
|
from .torchaudio import TorchAudioResampler
|
|
10
11
|
|
|
11
12
|
RESAMPLER_MAP = {
|
|
12
13
|
"lilfilter": LilFilterResampler,
|
|
14
|
+
"scipy": ScipyResampler,
|
|
13
15
|
"soxr": SoxrResampler,
|
|
14
16
|
"torchaudio": TorchAudioResampler,
|
|
15
17
|
}
|