lfeats 0.1.4__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lfeats-0.1.4 → lfeats-0.2.1}/.gitignore +1 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/PKG-INFO +22 -5
- {lfeats-0.1.4 → lfeats-0.2.1}/README.md +17 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/cli.py +4 -25
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/extractor.py +12 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/resampler.py +11 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/types.py +57 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/__init__.py +4 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/base.py +71 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/contentvec.py +0 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/data2vec.py +0 -3
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/data2vec2.py +0 -3
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/ecapa_tdnn.py +5 -5
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/emotion2vec.py +1 -3
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/emotion2vec_plus.py +2 -4
- lfeats-0.2.1/lfeats/models/higgs_audio.py +136 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/hubert.py +0 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/manager.py +13 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/next_tdnn.py +4 -4
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/r_spin.py +0 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/r_vector.py +1 -3
- lfeats-0.2.1/lfeats/models/redimnet.py +106 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/spidr.py +4 -8
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/spin.py +3 -4
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/sslzip.py +14 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/unispeech_sat.py +0 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/wav2vec2.py +0 -3
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/wavlm.py +0 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/whisper.py +0 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/x_vector.py +5 -5
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/__init__.py +2 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/base.py +11 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/lilfilter.py +13 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/manager.py +13 -0
- lfeats-0.2.1/lfeats/resamplers/scipy.py +90 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/torchaudio.py +14 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/download.py +24 -10
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/distributed.py +1 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/io.py +50 -6
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/validation.py +27 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/version.py +1 -1
- {lfeats-0.1.4 → lfeats-0.2.1}/pyproject.toml +4 -2
- {lfeats-0.1.4 → lfeats-0.2.1}/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/soxr.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/checkpoint_utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/config.yaml +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/dictionary.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/modality.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/text_compressor.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/configs.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/constants.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/initialize.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/file_io.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/incremental_decoding_utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/meters.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec2.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec_audio.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/audio.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/base.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/modules.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_decoder.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_encoder.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_incremental_decoder.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_model.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/hubert.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/wav2vec2.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/ema_module.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fairseq_dropout.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fp32_group_norm.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gelu.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gumbel_vector_quantizer.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/layer_norm.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/multihead_attention.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/quant_noise.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/same_pad.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/transpose_last.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/quantization_utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/registry.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/audio_pretraining.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/fairseq_task.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/hubert_pretraining.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tokenizer.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/SpeakerNet.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/vap_bn_tanh_fc_bn.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7_cyclical_lr_step.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_light_C256_B3_K65.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/main.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/NeXt_TDNN.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt_light.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/mel_transform.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/model.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/wavlm_config.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/convert.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/hubert_model.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/wav2vec2_model.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/WavLM.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/modules.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/dataio.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/encoder.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/preprocess.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/classifiers.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/interfaces.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/features.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ECAPA_TDNN.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ResNet.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/Xvector.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/CNN.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/containers.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/linear.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/normalization.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/pooling.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/features.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/_workarounds.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/autocast.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/checkpoints.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/fetching.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/filter_analysis.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/logger.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/parameter_transfer.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/run_opts.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/base.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/spin.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/dnn.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/hubert.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/swav_vq_dis.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/wavlm.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/model_utils.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/padding.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/LICENSE +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/drop.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/helpers.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/mlp.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/__init__.py +0 -0
- {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/paths.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lfeats
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A unified interface to extract hidden representations from speech foundation models
|
|
5
5
|
Project-URL: Documentation, https://takenori-y.github.io/lfeats/stable/
|
|
6
6
|
Project-URL: Source, https://github.com/takenori-y/lfeats
|
|
@@ -18,6 +18,7 @@ Classifier: Programming Language :: Python :: 3.12
|
|
|
18
18
|
Classifier: Programming Language :: Python :: 3.13
|
|
19
19
|
Classifier: Programming Language :: Python :: 3.14
|
|
20
20
|
Requires-Python: >=3.10
|
|
21
|
+
Requires-Dist: filelock>=3.10.0
|
|
21
22
|
Requires-Dist: huggingface-hub>=0.23.0
|
|
22
23
|
Requires-Dist: hydra-core>=1.3.0
|
|
23
24
|
Requires-Dist: hyperpyyaml>=0.0.1
|
|
@@ -27,11 +28,12 @@ Requires-Dist: onnxruntime>=1.19.0
|
|
|
27
28
|
Requires-Dist: parfive>=2.1.0
|
|
28
29
|
Requires-Dist: platformdirs>=2.0.0
|
|
29
30
|
Requires-Dist: requests>=2.27.0
|
|
31
|
+
Requires-Dist: scipy>=1.9.2
|
|
30
32
|
Requires-Dist: soundfile>=0.10.2
|
|
31
33
|
Requires-Dist: soxr>=0.4.0
|
|
32
34
|
Requires-Dist: torch>=2.6.0
|
|
33
35
|
Requires-Dist: torchaudio>=2.6.0
|
|
34
|
-
Requires-Dist: transformers>=
|
|
36
|
+
Requires-Dist: transformers>=5.3.0
|
|
35
37
|
Provides-Extra: dev
|
|
36
38
|
Requires-Dist: build; extra == 'dev'
|
|
37
39
|
Requires-Dist: matplotlib; extra == 'dev'
|
|
@@ -39,7 +41,7 @@ Requires-Dist: mdformat; extra == 'dev'
|
|
|
39
41
|
Requires-Dist: numpydoc; extra == 'dev'
|
|
40
42
|
Requires-Dist: pkginfo; extra == 'dev'
|
|
41
43
|
Requires-Dist: pydata-sphinx-theme; extra == 'dev'
|
|
42
|
-
Requires-Dist: pyright; extra == 'dev'
|
|
44
|
+
Requires-Dist: pyright<=1.1.408; extra == 'dev'
|
|
43
45
|
Requires-Dist: pytest; extra == 'dev'
|
|
44
46
|
Requires-Dist: pytest-cov; extra == 'dev'
|
|
45
47
|
Requires-Dist: ruff; extra == 'dev'
|
|
@@ -142,6 +144,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
142
144
|
| | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
|
|
143
145
|
| | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
|
|
144
146
|
|
|
147
|
+
### Token-Level Features
|
|
148
|
+
|
|
149
|
+
| Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
|
|
150
|
+
| :--- | :--- | ---: | ---: | :---: | :---: | :---: |
|
|
151
|
+
| `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
|
|
152
|
+
|
|
145
153
|
### Utterance-Level Features
|
|
146
154
|
|
|
147
155
|
| Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
|
|
@@ -150,7 +158,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
150
158
|
| `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
|
|
151
159
|
| | `base` | 0 | 192 | | | |
|
|
152
160
|
| | `base-v2` | 0 | 192 | | | |
|
|
153
|
-
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/
|
|
161
|
+
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
|
|
162
|
+
| `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
|
|
163
|
+
| | `b1` | 0 | 192 | | | |
|
|
164
|
+
| | `b2` | 0 | 192 | | | |
|
|
165
|
+
| | `b3` | 0 | 192 | | | |
|
|
166
|
+
| | `b4` | 0 | 192 | | | |
|
|
167
|
+
| | `b5` | 0 | 192 | | | |
|
|
168
|
+
| | `b6` | 0 | 192 | | | |
|
|
154
169
|
| `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
|
|
155
170
|
|
|
156
171
|
> [!IMPORTANT]
|
|
@@ -162,12 +177,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
162
177
|
| Resampler Type | Quality Preset | Source | License |
|
|
163
178
|
| :--- | :--- | :---: | :--- |
|
|
164
179
|
| `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
|
|
180
|
+
| `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
|
|
181
|
+
| | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
|
|
165
182
|
| `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
|
|
166
183
|
| | `low` | | |
|
|
167
184
|
| | `medium` | | |
|
|
168
185
|
| | `high` | | |
|
|
169
186
|
| | `very-high` | | |
|
|
170
|
-
| `torchaudio` | `kaiser-fast` | [
|
|
187
|
+
| `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
|
|
171
188
|
| | `kaiser-best` | | |
|
|
172
189
|
|
|
173
190
|
## Examples
|
|
@@ -93,6 +93,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
93
93
|
| | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
|
|
94
94
|
| | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
|
|
95
95
|
|
|
96
|
+
### Token-Level Features
|
|
97
|
+
|
|
98
|
+
| Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
|
|
99
|
+
| :--- | :--- | ---: | ---: | :---: | :---: | :---: |
|
|
100
|
+
| `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
|
|
101
|
+
|
|
96
102
|
### Utterance-Level Features
|
|
97
103
|
|
|
98
104
|
| Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
|
|
@@ -101,7 +107,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
101
107
|
| `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
|
|
102
108
|
| | `base` | 0 | 192 | | | |
|
|
103
109
|
| | `base-v2` | 0 | 192 | | | |
|
|
104
|
-
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/
|
|
110
|
+
| `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
|
|
111
|
+
| `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
|
|
112
|
+
| | `b1` | 0 | 192 | | | |
|
|
113
|
+
| | `b2` | 0 | 192 | | | |
|
|
114
|
+
| | `b3` | 0 | 192 | | | |
|
|
115
|
+
| | `b4` | 0 | 192 | | | |
|
|
116
|
+
| | `b5` | 0 | 192 | | | |
|
|
117
|
+
| | `b6` | 0 | 192 | | | |
|
|
105
118
|
| `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
|
|
106
119
|
|
|
107
120
|
> [!IMPORTANT]
|
|
@@ -113,12 +126,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
|
|
|
113
126
|
| Resampler Type | Quality Preset | Source | License |
|
|
114
127
|
| :--- | :--- | :---: | :--- |
|
|
115
128
|
| `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
|
|
129
|
+
| `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
|
|
130
|
+
| | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
|
|
116
131
|
| `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
|
|
117
132
|
| | `low` | | |
|
|
118
133
|
| | `medium` | | |
|
|
119
134
|
| | `high` | | |
|
|
120
135
|
| | `very-high` | | |
|
|
121
|
-
| `torchaudio` | `kaiser-fast` | [
|
|
136
|
+
| `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
|
|
122
137
|
| | `kaiser-best` | | |
|
|
123
138
|
|
|
124
139
|
## Examples
|
|
@@ -147,9 +147,6 @@ def main() -> None:
|
|
|
147
147
|
datefmt="%Y-%m-%d %H:%M:%S",
|
|
148
148
|
)
|
|
149
149
|
|
|
150
|
-
import numpy as np
|
|
151
|
-
import torch
|
|
152
|
-
|
|
153
150
|
import lfeats
|
|
154
151
|
from lfeats.utils.io import load_audio
|
|
155
152
|
|
|
@@ -170,7 +167,8 @@ def main() -> None:
|
|
|
170
167
|
raise ValueError(f"Invalid source: {args.source}")
|
|
171
168
|
|
|
172
169
|
if len(input_files) == 0:
|
|
173
|
-
|
|
170
|
+
logging.info(f"No audio files found in the source: {args.source}")
|
|
171
|
+
sys.exit(0)
|
|
174
172
|
logger.info(f"Found {len(input_files)} audio files to process.")
|
|
175
173
|
|
|
176
174
|
# Parse the layers argument.
|
|
@@ -216,7 +214,7 @@ def main() -> None:
|
|
|
216
214
|
path = Path(input_file).parent
|
|
217
215
|
# Remove the root part of the path.
|
|
218
216
|
dirs = path.relative_to(path.anchor).parts
|
|
219
|
-
if args.subdir_offset
|
|
217
|
+
if args.subdir_offset > len(dirs):
|
|
220
218
|
logger.error(
|
|
221
219
|
f"Subdir offset {args.subdir_offset} is too large for file: "
|
|
222
220
|
f"{input_file}. Skipping."
|
|
@@ -252,26 +250,7 @@ def main() -> None:
|
|
|
252
250
|
num_errors += 1
|
|
253
251
|
continue
|
|
254
252
|
|
|
255
|
-
|
|
256
|
-
result = {
|
|
257
|
-
"features": features.array,
|
|
258
|
-
"source": features.source,
|
|
259
|
-
"layers": features.layers,
|
|
260
|
-
}
|
|
261
|
-
np.savez_compressed(output_file, **result)
|
|
262
|
-
elif args.output_format == "pt":
|
|
263
|
-
result = {
|
|
264
|
-
"features": features.tensor.cpu(),
|
|
265
|
-
"source": features.source,
|
|
266
|
-
"layers": features.layers,
|
|
267
|
-
}
|
|
268
|
-
torch.save(result, output_file)
|
|
269
|
-
elif args.output_format == "float":
|
|
270
|
-
features.array.tofile(output_file)
|
|
271
|
-
elif args.output_format == "double":
|
|
272
|
-
features.array.astype(np.float64).tofile(output_file)
|
|
273
|
-
else:
|
|
274
|
-
raise ValueError(f"Unsupported output format: {args.output_format}")
|
|
253
|
+
features.tofile(output_file, double=args.output_format == "double")
|
|
275
254
|
|
|
276
255
|
if num_errors > 0:
|
|
277
256
|
logger.error(f"{num_errors} files were skipped due to errors.")
|
|
@@ -90,6 +90,18 @@ class Extractor:
|
|
|
90
90
|
"""
|
|
91
91
|
self.model_manager.get_model().load(self.cache_dir, quiet)
|
|
92
92
|
|
|
93
|
+
def to(self, device: str) -> None:
|
|
94
|
+
"""Move the model to the specified device.
|
|
95
|
+
|
|
96
|
+
Parameters
|
|
97
|
+
----------
|
|
98
|
+
device : str
|
|
99
|
+
The device to move the model to (e.g., 'cpu' or 'cuda').
|
|
100
|
+
|
|
101
|
+
"""
|
|
102
|
+
self.model_manager.to(device)
|
|
103
|
+
self.resampler_manager.to(device)
|
|
104
|
+
|
|
93
105
|
def __call__(
|
|
94
106
|
self,
|
|
95
107
|
source: np.ndarray | torch.Tensor | Audio,
|
|
@@ -45,6 +45,17 @@ class Resampler:
|
|
|
45
45
|
f"Supported resamplers are: {[k for k in RESAMPLER_MAP.keys()]}"
|
|
46
46
|
) from e
|
|
47
47
|
|
|
48
|
+
def to(self, device: str) -> None:
|
|
49
|
+
"""Move the resampler to the specified device.
|
|
50
|
+
|
|
51
|
+
Parameters
|
|
52
|
+
----------
|
|
53
|
+
device : str
|
|
54
|
+
The device to move the resampler to (e.g., 'cpu' or 'cuda').
|
|
55
|
+
|
|
56
|
+
"""
|
|
57
|
+
self.resampler_manager.to(device)
|
|
58
|
+
|
|
48
59
|
def __call__(
|
|
49
60
|
self,
|
|
50
61
|
source: np.ndarray | torch.Tensor | Audio,
|
|
@@ -9,6 +9,7 @@ from dataclasses import dataclass
|
|
|
9
9
|
from enum import Enum
|
|
10
10
|
|
|
11
11
|
import numpy as np
|
|
12
|
+
import soundfile as sf
|
|
12
13
|
import torch
|
|
13
14
|
import torch.nn.functional as F
|
|
14
15
|
|
|
@@ -164,6 +165,28 @@ class Audio(Container):
|
|
|
164
165
|
"""
|
|
165
166
|
return self.data.shape[1]
|
|
166
167
|
|
|
168
|
+
def tofile(self, path: str) -> None:
|
|
169
|
+
"""Save the audio to a file.
|
|
170
|
+
|
|
171
|
+
Parameters
|
|
172
|
+
----------
|
|
173
|
+
path : str
|
|
174
|
+
The path to save the audio to.
|
|
175
|
+
|
|
176
|
+
"""
|
|
177
|
+
ext = path.split(".")[-1].lower()
|
|
178
|
+
if ext in ("wav", "flac"):
|
|
179
|
+
sf.write(path, self.array.T, self.sample_rate)
|
|
180
|
+
elif ext == "npz":
|
|
181
|
+
np.savez_compressed(path, samples=self.array, sample_rate=self.sample_rate)
|
|
182
|
+
elif ext == "pt":
|
|
183
|
+
torch.save(
|
|
184
|
+
{"samples": self.tensor.cpu(), "sample_rate": self.sample_rate},
|
|
185
|
+
path,
|
|
186
|
+
)
|
|
187
|
+
else:
|
|
188
|
+
self.array.tofile(path)
|
|
189
|
+
|
|
167
190
|
def normalize(self, eps: float = 1e-5) -> Audio:
|
|
168
191
|
"""Normalize the audio samples to have zero mean and unit variance.
|
|
169
192
|
|
|
@@ -237,6 +260,40 @@ class Features(Container):
|
|
|
237
260
|
"""
|
|
238
261
|
return self.data.shape[1]
|
|
239
262
|
|
|
263
|
+
def tofile(self, path: str, double: bool = False) -> None:
|
|
264
|
+
"""Save the features to a file.
|
|
265
|
+
|
|
266
|
+
Parameters
|
|
267
|
+
----------
|
|
268
|
+
path : str
|
|
269
|
+
The path to save the features to.
|
|
270
|
+
|
|
271
|
+
double : bool, optional
|
|
272
|
+
Whether to save the features in double precision instead of single one.
|
|
273
|
+
|
|
274
|
+
"""
|
|
275
|
+
ext = path.split(".")[-1].lower()
|
|
276
|
+
if ext == "npz":
|
|
277
|
+
np.savez_compressed(
|
|
278
|
+
path,
|
|
279
|
+
features=self.array.astype(np.float64 if double else np.float32),
|
|
280
|
+
source=self.source,
|
|
281
|
+
layers=self.layers or [],
|
|
282
|
+
)
|
|
283
|
+
elif ext == "pt":
|
|
284
|
+
torch.save(
|
|
285
|
+
{
|
|
286
|
+
"features": self.tensor.cpu().to(
|
|
287
|
+
torch.float64 if double else torch.float32
|
|
288
|
+
),
|
|
289
|
+
"source": self.source,
|
|
290
|
+
"layers": self.layers,
|
|
291
|
+
},
|
|
292
|
+
path,
|
|
293
|
+
)
|
|
294
|
+
else:
|
|
295
|
+
self.array.astype(np.float64 if double else np.float32).tofile(path)
|
|
296
|
+
|
|
240
297
|
def trim(self, start: int, end: int) -> Features:
|
|
241
298
|
"""Trim the features along the time dimension.
|
|
242
299
|
|
|
@@ -9,11 +9,13 @@ from .data2vec2 import Data2Vec2Model
|
|
|
9
9
|
from .ecapa_tdnn import EcapaTDNNModel
|
|
10
10
|
from .emotion2vec import Emotion2VecModel
|
|
11
11
|
from .emotion2vec_plus import Emotion2VecPlusModel
|
|
12
|
+
from .higgs_audio import HiggsAudioTokenizerModel
|
|
12
13
|
from .hubert import HuBERTModel
|
|
13
14
|
from .manager import ModelManager
|
|
14
15
|
from .next_tdnn import NeXtTDNNModel
|
|
15
16
|
from .r_spin import RSpinModel
|
|
16
17
|
from .r_vector import RVectorModel
|
|
18
|
+
from .redimnet import ReDimNetModel
|
|
17
19
|
from .spidr import SpidRModel
|
|
18
20
|
from .spin import SpinModel
|
|
19
21
|
from .sslzip import SSLZipModel
|
|
@@ -30,10 +32,12 @@ MODEL_MAP = {
|
|
|
30
32
|
"ecapa-tdnn": EcapaTDNNModel,
|
|
31
33
|
"emotion2vec": Emotion2VecModel,
|
|
32
34
|
"emotion2vec+": Emotion2VecPlusModel,
|
|
35
|
+
"higgs-audio": HiggsAudioTokenizerModel,
|
|
33
36
|
"hubert": HuBERTModel,
|
|
34
37
|
"next-tdnn": NeXtTDNNModel,
|
|
35
38
|
"r-spin": RSpinModel,
|
|
36
39
|
"r-vector": RVectorModel,
|
|
40
|
+
"redimnet": ReDimNetModel,
|
|
37
41
|
"spidr": SpidRModel,
|
|
38
42
|
"spin": SpinModel,
|
|
39
43
|
"sslzip": SSLZipModel,
|
|
@@ -25,6 +25,7 @@ class BaseModel(ABC):
|
|
|
25
25
|
"""
|
|
26
26
|
self.device = device
|
|
27
27
|
|
|
28
|
+
self.model = None
|
|
28
29
|
self._model_id = None # To be defined in subclasses
|
|
29
30
|
|
|
30
31
|
@abstractmethod
|
|
@@ -42,6 +43,24 @@ class BaseModel(ABC):
|
|
|
42
43
|
"""
|
|
43
44
|
raise NotImplementedError
|
|
44
45
|
|
|
46
|
+
def to(self, device: str) -> None:
|
|
47
|
+
"""Move the model to the specified device.
|
|
48
|
+
|
|
49
|
+
Parameters
|
|
50
|
+
----------
|
|
51
|
+
device : str
|
|
52
|
+
The device to move the model to (e.g., 'cpu' or 'cuda').
|
|
53
|
+
|
|
54
|
+
"""
|
|
55
|
+
self.device = device
|
|
56
|
+
if self.model is not None and hasattr(self.model, "to"):
|
|
57
|
+
self.model.to(device)
|
|
58
|
+
if hasattr(self.model, "device"):
|
|
59
|
+
try:
|
|
60
|
+
self.model.device = device
|
|
61
|
+
except AttributeError:
|
|
62
|
+
pass
|
|
63
|
+
|
|
45
64
|
def extract_features(self, audio: Audio, layers: list[int]) -> Features:
|
|
46
65
|
"""Extract features from the input audio data.
|
|
47
66
|
|
|
@@ -237,6 +256,58 @@ class FrameLevelFeatureModel(BaseModel):
|
|
|
237
256
|
return Granularity.FRAME
|
|
238
257
|
|
|
239
258
|
|
|
259
|
+
class TokenLevelFeatureModel(BaseModel):
|
|
260
|
+
"""An abstract base class for frame-level feature extraction models."""
|
|
261
|
+
|
|
262
|
+
@property
|
|
263
|
+
def num_layers(self) -> int:
|
|
264
|
+
"""Get the number of layers in the model.
|
|
265
|
+
|
|
266
|
+
Returns
|
|
267
|
+
-------
|
|
268
|
+
out : int
|
|
269
|
+
The number of layers.
|
|
270
|
+
|
|
271
|
+
"""
|
|
272
|
+
return 0
|
|
273
|
+
|
|
274
|
+
@property
|
|
275
|
+
def frame_shift(self) -> int:
|
|
276
|
+
"""Get the frame shift of the model.
|
|
277
|
+
|
|
278
|
+
Returns
|
|
279
|
+
-------
|
|
280
|
+
out : int
|
|
281
|
+
The frame shift in samples.
|
|
282
|
+
|
|
283
|
+
"""
|
|
284
|
+
return int(40.0 * self.sample_rate / 1000)
|
|
285
|
+
|
|
286
|
+
@property
|
|
287
|
+
def center_offset(self) -> int:
|
|
288
|
+
"""Get the center offset of the model.
|
|
289
|
+
|
|
290
|
+
Returns
|
|
291
|
+
-------
|
|
292
|
+
out : int
|
|
293
|
+
The center offset in samples.
|
|
294
|
+
|
|
295
|
+
"""
|
|
296
|
+
return 0
|
|
297
|
+
|
|
298
|
+
@property
|
|
299
|
+
def granularity(self) -> Granularity:
|
|
300
|
+
"""Get the granularity of the features extracted by the model.
|
|
301
|
+
|
|
302
|
+
Returns
|
|
303
|
+
-------
|
|
304
|
+
out : str
|
|
305
|
+
The granularity of the features.
|
|
306
|
+
|
|
307
|
+
"""
|
|
308
|
+
return Granularity.FRAME
|
|
309
|
+
|
|
310
|
+
|
|
240
311
|
class UtteranceLevelFeatureModel(BaseModel):
|
|
241
312
|
"""An abstract base class for utterance-level feature extraction models."""
|
|
242
313
|
|
|
@@ -58,9 +58,6 @@ class Data2VecModel(FrameLevelFeatureModel):
|
|
|
58
58
|
self.variant = validate_enum(variant, Data2VecVariant, Data2VecVariant.BASE)
|
|
59
59
|
self._model_id = f"data2vec-{self.variant.value}"
|
|
60
60
|
|
|
61
|
-
self.processor = None
|
|
62
|
-
self.model = None
|
|
63
|
-
|
|
64
61
|
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
65
62
|
"""Load the model from the specified directory.
|
|
66
63
|
|
|
@@ -58,9 +58,6 @@ class Data2Vec2Model(FrameLevelFeatureModel):
|
|
|
58
58
|
self.variant = validate_enum(variant, Data2Vec2Variant, Data2Vec2Variant.BASE)
|
|
59
59
|
self._model_id = f"data2vec2-{self.variant.value}"
|
|
60
60
|
|
|
61
|
-
self.processor = None
|
|
62
|
-
self.model = None
|
|
63
|
-
|
|
64
61
|
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
65
62
|
"""Load the model from the specified directory.
|
|
66
63
|
|
|
@@ -10,7 +10,7 @@ import torch
|
|
|
10
10
|
from ..interfaces.types import Audio, Features
|
|
11
11
|
from ..utils.io import silence_hf_hub
|
|
12
12
|
from ..utils.paths import setup_third_party_path
|
|
13
|
-
from ..utils.validation import validate_enum
|
|
13
|
+
from ..utils.validation import validate_enum, validate_length
|
|
14
14
|
from .base import UtteranceLevelFeatureModel
|
|
15
15
|
|
|
16
16
|
|
|
@@ -40,8 +40,6 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
|
|
|
40
40
|
self.variant = validate_enum(variant, EcapaTDNNVariant, EcapaTDNNVariant.BASE)
|
|
41
41
|
self._model_id = f"ecapa-tdnn-{self.variant.value}"
|
|
42
42
|
|
|
43
|
-
self.model = None
|
|
44
|
-
|
|
45
43
|
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
46
44
|
"""Load the model from the specified directory.
|
|
47
45
|
|
|
@@ -78,11 +76,11 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
|
|
|
78
76
|
self.model = EncoderClassifier.from_hparams(
|
|
79
77
|
source="speechbrain/spkrec-ecapa-voxceleb",
|
|
80
78
|
fetch_config=fetch_config,
|
|
79
|
+
run_opts={"device": self.device},
|
|
81
80
|
)
|
|
82
81
|
if self.model is None:
|
|
83
82
|
raise RuntimeError("Failed to load the model.")
|
|
84
83
|
self.model.eval()
|
|
85
|
-
self.model.to(self.device)
|
|
86
84
|
|
|
87
85
|
def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
|
|
88
86
|
"""Extract features from the input audio using the model.
|
|
@@ -110,6 +108,8 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
|
|
|
110
108
|
raise RuntimeError("Model is not loaded. Call 'load' method first.")
|
|
111
109
|
|
|
112
110
|
with torch.inference_mode():
|
|
113
|
-
|
|
111
|
+
inputs = audio.tensor.to(self.device)
|
|
112
|
+
inputs = validate_length(inputs, 640)
|
|
113
|
+
vectors = self.model.encode_batch(inputs)
|
|
114
114
|
|
|
115
115
|
return Features(data=vectors, source=self.model_id)
|
|
@@ -55,8 +55,6 @@ class Emotion2VecModel(FrameLevelFeatureModel):
|
|
|
55
55
|
)
|
|
56
56
|
self._model_id = f"emotion2vec-{self.variant.value}"
|
|
57
57
|
|
|
58
|
-
self.model = None
|
|
59
|
-
|
|
60
58
|
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
61
59
|
"""Load the model from the specified directory.
|
|
62
60
|
|
|
@@ -78,7 +76,7 @@ class Emotion2VecModel(FrameLevelFeatureModel):
|
|
|
78
76
|
repo_id=repo_id,
|
|
79
77
|
filename=filename,
|
|
80
78
|
repo_type="model",
|
|
81
|
-
|
|
79
|
+
cache_dir=model_dir,
|
|
82
80
|
)
|
|
83
81
|
|
|
84
82
|
setup_third_party_path()
|
|
@@ -58,8 +58,6 @@ class Emotion2VecPlusModel(FrameLevelFeatureModel):
|
|
|
58
58
|
)
|
|
59
59
|
self._model_id = f"emotion2vec+-{self.variant.value}"
|
|
60
60
|
|
|
61
|
-
self.model = None
|
|
62
|
-
|
|
63
61
|
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
64
62
|
"""Load the model from the specified directory.
|
|
65
63
|
|
|
@@ -81,13 +79,13 @@ class Emotion2VecPlusModel(FrameLevelFeatureModel):
|
|
|
81
79
|
repo_id=repo_id,
|
|
82
80
|
filename="model.pt",
|
|
83
81
|
repo_type="model",
|
|
84
|
-
|
|
82
|
+
cache_dir=os.path.join(model_dir, sanitize(self.model_id)),
|
|
85
83
|
)
|
|
86
84
|
config = hf_hub_download(
|
|
87
85
|
repo_id=repo_id,
|
|
88
86
|
filename="config.yaml",
|
|
89
87
|
repo_type="model",
|
|
90
|
-
|
|
88
|
+
cache_dir=os.path.join(model_dir, sanitize(self.model_id)),
|
|
91
89
|
)
|
|
92
90
|
|
|
93
91
|
from hyperpyyaml import load_hyperpyyaml
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Copyright (c) 2026 Takenori Yoshimura
|
|
2
|
+
# Released under the MIT License.
|
|
3
|
+
|
|
4
|
+
"""A module for the Higgs Audio tokenizer."""
|
|
5
|
+
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import torch
|
|
10
|
+
|
|
11
|
+
from ..interfaces.types import Audio, Features
|
|
12
|
+
from ..utils.io import silence_transformers
|
|
13
|
+
from ..utils.validation import validate_enum
|
|
14
|
+
from .base import TokenLevelFeatureModel
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class HiggsAudioTokenizerVariant(str, Enum):
|
|
18
|
+
"""Enumeration of supported Higgs Audio tokenizer variants."""
|
|
19
|
+
|
|
20
|
+
V2 = "v2"
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def model_name(self) -> str:
|
|
24
|
+
"""Return the model name corresponding to the variant.
|
|
25
|
+
|
|
26
|
+
Returns
|
|
27
|
+
-------
|
|
28
|
+
out : str
|
|
29
|
+
The model name corresponding to the variant.
|
|
30
|
+
|
|
31
|
+
"""
|
|
32
|
+
return f"eustlb/higgs-audio-{self.value}-tokenizer"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class HiggsAudioTokenizerModel(TokenLevelFeatureModel):
|
|
36
|
+
"""A class for the Higgs Audio tokenizer model."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
|
|
39
|
+
"""Initialize the Higgs Audio tokenizer model.
|
|
40
|
+
|
|
41
|
+
Parameters
|
|
42
|
+
----------
|
|
43
|
+
variant : str | None, optional
|
|
44
|
+
The variant of the model to use.
|
|
45
|
+
|
|
46
|
+
device : str, optional
|
|
47
|
+
The device to run the model on (e.g., 'cpu' or 'cuda').
|
|
48
|
+
|
|
49
|
+
"""
|
|
50
|
+
super().__init__(variant, device)
|
|
51
|
+
|
|
52
|
+
self.variant = validate_enum(
|
|
53
|
+
variant, HiggsAudioTokenizerVariant, HiggsAudioTokenizerVariant.V2
|
|
54
|
+
)
|
|
55
|
+
self._model_id = f"higgs-audio-{self.variant.value}"
|
|
56
|
+
|
|
57
|
+
self.feature_extractor = None
|
|
58
|
+
|
|
59
|
+
def load(self, model_dir: str, quiet: bool = False) -> None:
|
|
60
|
+
"""Load the model from the specified directory.
|
|
61
|
+
|
|
62
|
+
Parameters
|
|
63
|
+
----------
|
|
64
|
+
model_dir : str
|
|
65
|
+
The directory where the model checkpoint will be stored.
|
|
66
|
+
|
|
67
|
+
quiet : bool, optional
|
|
68
|
+
Whether to suppress output during the loading process.
|
|
69
|
+
|
|
70
|
+
"""
|
|
71
|
+
if self.model is not None:
|
|
72
|
+
return
|
|
73
|
+
|
|
74
|
+
from transformers import AutoFeatureExtractor, HiggsAudioV2TokenizerModel
|
|
75
|
+
|
|
76
|
+
with silence_transformers(quiet):
|
|
77
|
+
self.feature_extractor = AutoFeatureExtractor.from_pretrained(
|
|
78
|
+
self.variant.model_name, cache_dir=model_dir
|
|
79
|
+
)
|
|
80
|
+
self.model = HiggsAudioV2TokenizerModel.from_pretrained(
|
|
81
|
+
self.variant.model_name, cache_dir=model_dir
|
|
82
|
+
)
|
|
83
|
+
self.model.eval()
|
|
84
|
+
self.model.to(self.device) # type: ignore
|
|
85
|
+
|
|
86
|
+
def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
|
|
87
|
+
"""Extract features from the input audio using the model.
|
|
88
|
+
|
|
89
|
+
Parameters
|
|
90
|
+
----------
|
|
91
|
+
audio : Audio
|
|
92
|
+
The input audio data with shape (B, T).
|
|
93
|
+
|
|
94
|
+
layers : list[int]
|
|
95
|
+
The layer(s) from which to extract features.
|
|
96
|
+
|
|
97
|
+
Returns
|
|
98
|
+
-------
|
|
99
|
+
out : Features
|
|
100
|
+
The extracted features.
|
|
101
|
+
|
|
102
|
+
Raises
|
|
103
|
+
------
|
|
104
|
+
RuntimeError
|
|
105
|
+
If the model is not loaded.
|
|
106
|
+
|
|
107
|
+
"""
|
|
108
|
+
if self.feature_extractor is None or self.model is None:
|
|
109
|
+
raise RuntimeError("Model not loaded. Call 'load' method first.")
|
|
110
|
+
|
|
111
|
+
with torch.inference_mode():
|
|
112
|
+
inputs = self.feature_extractor(
|
|
113
|
+
raw_audio=[x for x in audio.array],
|
|
114
|
+
sampling_rate=self.feature_extractor.sampling_rate,
|
|
115
|
+
return_tensors="pt",
|
|
116
|
+
).to(self.device)
|
|
117
|
+
|
|
118
|
+
encoder_outputs: Any = self.model.encode(inputs["input_values"])
|
|
119
|
+
indices = encoder_outputs.audio_codes # (B, Q, N)
|
|
120
|
+
indices = indices.transpose(0, 1)
|
|
121
|
+
vectors = self.model.quantizer.decode(indices) # (B, D, N)
|
|
122
|
+
vectors = vectors.transpose(1, 2)
|
|
123
|
+
|
|
124
|
+
return Features(data=vectors, source=self.model_id)
|
|
125
|
+
|
|
126
|
+
@property
|
|
127
|
+
def sample_rate(self) -> int:
|
|
128
|
+
"""Get the sample rate required by the model.
|
|
129
|
+
|
|
130
|
+
Returns
|
|
131
|
+
-------
|
|
132
|
+
out : int
|
|
133
|
+
The sample rate in Hz.
|
|
134
|
+
|
|
135
|
+
"""
|
|
136
|
+
return 24000
|