lfeats 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. {lfeats-0.2.0 → lfeats-0.2.1}/.gitignore +1 -0
  2. {lfeats-0.2.0 → lfeats-0.2.1}/PKG-INFO +21 -5
  3. {lfeats-0.2.0 → lfeats-0.2.1}/README.md +17 -2
  4. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/cli.py +4 -25
  5. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/types.py +57 -0
  6. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/__init__.py +4 -0
  7. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/base.py +52 -0
  8. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/ecapa_tdnn.py +4 -2
  9. lfeats-0.2.1/lfeats/models/higgs_audio.py +136 -0
  10. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/next_tdnn.py +4 -2
  11. lfeats-0.2.1/lfeats/models/redimnet.py +106 -0
  12. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/x_vector.py +4 -2
  13. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/__init__.py +2 -0
  14. lfeats-0.2.1/lfeats/resamplers/scipy.py +90 -0
  15. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/io.py +7 -2
  16. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/validation.py +27 -0
  17. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/version.py +1 -1
  18. {lfeats-0.2.0 → lfeats-0.2.1}/pyproject.toml +3 -2
  19. {lfeats-0.2.0 → lfeats-0.2.1}/LICENSE +0 -0
  20. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/__init__.py +0 -0
  21. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/__init__.py +0 -0
  22. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/extractor.py +0 -0
  23. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/resampler.py +0 -0
  24. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/interfaces/utils.py +0 -0
  25. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/contentvec.py +0 -0
  26. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/data2vec.py +0 -0
  27. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/data2vec2.py +0 -0
  28. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/emotion2vec.py +0 -0
  29. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/emotion2vec_plus.py +0 -0
  30. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/hubert.py +0 -0
  31. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/manager.py +0 -0
  32. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/r_spin.py +0 -0
  33. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/r_vector.py +0 -0
  34. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/spidr.py +0 -0
  35. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/spin.py +0 -0
  36. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/sslzip.py +0 -0
  37. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/unispeech_sat.py +0 -0
  38. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/wav2vec2.py +0 -0
  39. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/wavlm.py +0 -0
  40. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/models/whisper.py +0 -0
  41. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/base.py +0 -0
  42. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/lilfilter.py +0 -0
  43. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/manager.py +0 -0
  44. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/soxr.py +0 -0
  45. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/resamplers/torchaudio.py +0 -0
  46. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/__init__.py +0 -0
  47. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/LICENSE +0 -0
  48. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/__init__.py +0 -0
  49. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/checkpoint_utils.py +0 -0
  50. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/__init__.py +0 -0
  51. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/config.yaml +0 -0
  52. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/__init__.py +0 -0
  53. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/dictionary.py +0 -0
  54. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/modality.py +0 -0
  55. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/text_compressor.py +0 -0
  56. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/__init__.py +0 -0
  57. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/configs.py +0 -0
  58. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/constants.py +0 -0
  59. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/initialize.py +0 -0
  60. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/utils.py +0 -0
  61. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/file_io.py +0 -0
  62. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/incremental_decoding_utils.py +0 -0
  63. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/__init__.py +0 -0
  64. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/meters.py +0 -0
  65. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/__init__.py +0 -0
  66. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/__init__.py +0 -0
  67. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec2.py +0 -0
  68. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec_audio.py +0 -0
  69. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/__init__.py +0 -0
  70. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/audio.py +0 -0
  71. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/base.py +0 -0
  72. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/modules.py +0 -0
  73. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_decoder.py +0 -0
  74. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_encoder.py +0 -0
  75. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_incremental_decoder.py +0 -0
  76. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_model.py +0 -0
  77. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/__init__.py +0 -0
  78. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/hubert.py +0 -0
  79. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/__init__.py +0 -0
  80. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/utils.py +0 -0
  81. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/wav2vec2.py +0 -0
  82. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/__init__.py +0 -0
  83. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/ema_module.py +0 -0
  84. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fairseq_dropout.py +0 -0
  85. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fp32_group_norm.py +0 -0
  86. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gelu.py +0 -0
  87. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gumbel_vector_quantizer.py +0 -0
  88. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/layer_norm.py +0 -0
  89. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/multihead_attention.py +0 -0
  90. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/quant_noise.py +0 -0
  91. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/same_pad.py +0 -0
  92. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/transpose_last.py +0 -0
  93. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/quantization_utils.py +0 -0
  94. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/registry.py +0 -0
  95. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/__init__.py +0 -0
  96. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/audio_pretraining.py +0 -0
  97. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/fairseq_task.py +0 -0
  98. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/hubert_pretraining.py +0 -0
  99. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/tokenizer.py +0 -0
  100. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/fairseq/utils.py +0 -0
  101. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/LICENSE +0 -0
  102. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/SpeakerNet.py +0 -0
  103. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/__init__.py +0 -0
  104. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/__init__.py +0 -0
  105. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/vap_bn_tanh_fc_bn.py +0 -0
  106. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7.py +0 -0
  107. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7_cyclical_lr_step.py +0 -0
  108. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_light_C256_B3_K65.py +0 -0
  109. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/__init__.py +0 -0
  110. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/main.py +0 -0
  111. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/NeXt_TDNN.py +0 -0
  112. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt.py +0 -0
  113. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt_light.py +0 -0
  114. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/__init__.py +0 -0
  115. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/utils.py +0 -0
  116. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/__init__.py +0 -0
  117. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/mel_transform.py +0 -0
  118. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/LICENSE +0 -0
  119. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/__init__.py +0 -0
  120. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/model.py +0 -0
  121. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/rspin/wavlm_config.py +0 -0
  122. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/LICENSE +0 -0
  123. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/__init__.py +0 -0
  124. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/__init__.py +0 -0
  125. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/__init__.py +0 -0
  126. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/convert.py +0 -0
  127. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/hubert_model.py +0 -0
  128. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/utils.py +0 -0
  129. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/__init__.py +0 -0
  130. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/wav2vec2_model.py +0 -0
  131. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/WavLM.py +0 -0
  132. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/__init__.py +0 -0
  133. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/modules.py +0 -0
  134. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/__init__.py +0 -0
  135. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/download.py +0 -0
  136. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/LICENSE +0 -0
  137. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/__init__.py +0 -0
  138. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/__init__.py +0 -0
  139. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/dataio.py +0 -0
  140. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/encoder.py +0 -0
  141. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/preprocess.py +0 -0
  142. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/__init__.py +0 -0
  143. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/classifiers.py +0 -0
  144. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/interfaces.py +0 -0
  145. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/__init__.py +0 -0
  146. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/features.py +0 -0
  147. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ECAPA_TDNN.py +0 -0
  148. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ResNet.py +0 -0
  149. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/Xvector.py +0 -0
  150. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/__init__.py +0 -0
  151. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/CNN.py +0 -0
  152. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/containers.py +0 -0
  153. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/linear.py +0 -0
  154. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/normalization.py +0 -0
  155. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/pooling.py +0 -0
  156. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/__init__.py +0 -0
  157. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/features.py +0 -0
  158. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/__init__.py +0 -0
  159. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/_workarounds.py +0 -0
  160. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/autocast.py +0 -0
  161. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/checkpoints.py +0 -0
  162. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/distributed.py +0 -0
  163. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/fetching.py +0 -0
  164. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/filter_analysis.py +0 -0
  165. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/logger.py +0 -0
  166. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/parameter_transfer.py +0 -0
  167. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/run_opts.py +0 -0
  168. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/LICENSE +0 -0
  169. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/__init__.py +0 -0
  170. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/__init__.py +0 -0
  171. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/base.py +0 -0
  172. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/model/spin.py +0 -0
  173. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/__init__.py +0 -0
  174. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/dnn.py +0 -0
  175. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/hubert.py +0 -0
  176. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/swav_vq_dis.py +0 -0
  177. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/nn/wavlm.py +0 -0
  178. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/__init__.py +0 -0
  179. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/model_utils.py +0 -0
  180. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/spin/util/padding.py +0 -0
  181. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/LICENSE +0 -0
  182. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/__init__.py +0 -0
  183. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/__init__.py +0 -0
  184. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/drop.py +0 -0
  185. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/helpers.py +0 -0
  186. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/third_party/timm/layers/mlp.py +0 -0
  187. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/__init__.py +0 -0
  188. {lfeats-0.2.0 → lfeats-0.2.1}/lfeats/utils/paths.py +0 -0
@@ -11,6 +11,7 @@ build/
11
11
 
12
12
  # tests
13
13
  tests/outputs/
14
+ *.wav
14
15
 
15
16
  # tools
16
17
  tools/**/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: lfeats
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: A unified interface to extract hidden representations from speech foundation models
5
5
  Project-URL: Documentation, https://takenori-y.github.io/lfeats/stable/
6
6
  Project-URL: Source, https://github.com/takenori-y/lfeats
@@ -28,11 +28,12 @@ Requires-Dist: onnxruntime>=1.19.0
28
28
  Requires-Dist: parfive>=2.1.0
29
29
  Requires-Dist: platformdirs>=2.0.0
30
30
  Requires-Dist: requests>=2.27.0
31
+ Requires-Dist: scipy>=1.9.2
31
32
  Requires-Dist: soundfile>=0.10.2
32
33
  Requires-Dist: soxr>=0.4.0
33
34
  Requires-Dist: torch>=2.6.0
34
35
  Requires-Dist: torchaudio>=2.6.0
35
- Requires-Dist: transformers>=4.30.0
36
+ Requires-Dist: transformers>=5.3.0
36
37
  Provides-Extra: dev
37
38
  Requires-Dist: build; extra == 'dev'
38
39
  Requires-Dist: matplotlib; extra == 'dev'
@@ -40,7 +41,7 @@ Requires-Dist: mdformat; extra == 'dev'
40
41
  Requires-Dist: numpydoc; extra == 'dev'
41
42
  Requires-Dist: pkginfo; extra == 'dev'
42
43
  Requires-Dist: pydata-sphinx-theme; extra == 'dev'
43
- Requires-Dist: pyright; extra == 'dev'
44
+ Requires-Dist: pyright<=1.1.408; extra == 'dev'
44
45
  Requires-Dist: pytest; extra == 'dev'
45
46
  Requires-Dist: pytest-cov; extra == 'dev'
46
47
  Requires-Dist: ruff; extra == 'dev'
@@ -143,6 +144,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
143
144
  | | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
144
145
  | | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
145
146
 
147
+ ### Token-Level Features
148
+
149
+ | Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
150
+ | :--- | :--- | ---: | ---: | :---: | :---: | :---: |
151
+ | `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
152
+
146
153
  ### Utterance-Level Features
147
154
 
148
155
  | Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
@@ -151,7 +158,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
151
158
  | `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
152
159
  | | `base` | 0 | 192 | | | |
153
160
  | | `base-v2` | 0 | 192 | | | |
154
- | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/pdf/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
161
+ | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
162
+ | `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
163
+ | | `b1` | 0 | 192 | | | |
164
+ | | `b2` | 0 | 192 | | | |
165
+ | | `b3` | 0 | 192 | | | |
166
+ | | `b4` | 0 | 192 | | | |
167
+ | | `b5` | 0 | 192 | | | |
168
+ | | `b6` | 0 | 192 | | | |
155
169
  | `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
156
170
 
157
171
  > [!IMPORTANT]
@@ -163,12 +177,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
163
177
  | Resampler Type | Quality Preset | Source | License |
164
178
  | :--- | :--- | :---: | :--- |
165
179
  | `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
180
+ | `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
181
+ | | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
166
182
  | `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
167
183
  | | `low` | | |
168
184
  | | `medium` | | |
169
185
  | | `high` | | |
170
186
  | | `very-high` | | |
171
- | `torchaudio` | `kaiser-fast` | [GitHub](https://github.com/pytorch/audio) | BSD 2-Clause |
187
+ | `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
172
188
  | | `kaiser-best` | | |
173
189
 
174
190
  ## Examples
@@ -93,6 +93,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
93
93
  | | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
94
94
  | | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
95
95
 
96
+ ### Token-Level Features
97
+
98
+ | Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
99
+ | :--- | :--- | ---: | ---: | :---: | :---: | :---: |
100
+ | `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
101
+
96
102
  ### Utterance-Level Features
97
103
 
98
104
  | Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
@@ -101,7 +107,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
101
107
  | `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
102
108
  | | `base` | 0 | 192 | | | |
103
109
  | | `base-v2` | 0 | 192 | | | |
104
- | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/pdf/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
110
+ | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
111
+ | `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
112
+ | | `b1` | 0 | 192 | | | |
113
+ | | `b2` | 0 | 192 | | | |
114
+ | | `b3` | 0 | 192 | | | |
115
+ | | `b4` | 0 | 192 | | | |
116
+ | | `b5` | 0 | 192 | | | |
117
+ | | `b6` | 0 | 192 | | | |
105
118
  | `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
106
119
 
107
120
  > [!IMPORTANT]
@@ -113,12 +126,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
113
126
  | Resampler Type | Quality Preset | Source | License |
114
127
  | :--- | :--- | :---: | :--- |
115
128
  | `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
129
+ | `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
130
+ | | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
116
131
  | `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
117
132
  | | `low` | | |
118
133
  | | `medium` | | |
119
134
  | | `high` | | |
120
135
  | | `very-high` | | |
121
- | `torchaudio` | `kaiser-fast` | [GitHub](https://github.com/pytorch/audio) | BSD 2-Clause |
136
+ | `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
122
137
  | | `kaiser-best` | | |
123
138
 
124
139
  ## Examples
@@ -147,9 +147,6 @@ def main() -> None:
147
147
  datefmt="%Y-%m-%d %H:%M:%S",
148
148
  )
149
149
 
150
- import numpy as np
151
- import torch
152
-
153
150
  import lfeats
154
151
  from lfeats.utils.io import load_audio
155
152
 
@@ -170,7 +167,8 @@ def main() -> None:
170
167
  raise ValueError(f"Invalid source: {args.source}")
171
168
 
172
169
  if len(input_files) == 0:
173
- raise ValueError(f"No audio files found in the source: {args.source}")
170
+ logging.info(f"No audio files found in the source: {args.source}")
171
+ sys.exit(0)
174
172
  logger.info(f"Found {len(input_files)} audio files to process.")
175
173
 
176
174
  # Parse the layers argument.
@@ -216,7 +214,7 @@ def main() -> None:
216
214
  path = Path(input_file).parent
217
215
  # Remove the root part of the path.
218
216
  dirs = path.relative_to(path.anchor).parts
219
- if args.subdir_offset >= len(dirs):
217
+ if args.subdir_offset > len(dirs):
220
218
  logger.error(
221
219
  f"Subdir offset {args.subdir_offset} is too large for file: "
222
220
  f"{input_file}. Skipping."
@@ -252,26 +250,7 @@ def main() -> None:
252
250
  num_errors += 1
253
251
  continue
254
252
 
255
- if args.output_format == "npz":
256
- result = {
257
- "features": features.array,
258
- "source": features.source,
259
- "layers": features.layers,
260
- }
261
- np.savez_compressed(output_file, **result)
262
- elif args.output_format == "pt":
263
- result = {
264
- "features": features.tensor.cpu(),
265
- "source": features.source,
266
- "layers": features.layers,
267
- }
268
- torch.save(result, output_file)
269
- elif args.output_format == "float":
270
- features.array.tofile(output_file)
271
- elif args.output_format == "double":
272
- features.array.astype(np.float64).tofile(output_file)
273
- else:
274
- raise ValueError(f"Unsupported output format: {args.output_format}")
253
+ features.tofile(output_file, double=args.output_format == "double")
275
254
 
276
255
  if num_errors > 0:
277
256
  logger.error(f"{num_errors} files were skipped due to errors.")
@@ -9,6 +9,7 @@ from dataclasses import dataclass
9
9
  from enum import Enum
10
10
 
11
11
  import numpy as np
12
+ import soundfile as sf
12
13
  import torch
13
14
  import torch.nn.functional as F
14
15
 
@@ -164,6 +165,28 @@ class Audio(Container):
164
165
  """
165
166
  return self.data.shape[1]
166
167
 
168
+ def tofile(self, path: str) -> None:
169
+ """Save the audio to a file.
170
+
171
+ Parameters
172
+ ----------
173
+ path : str
174
+ The path to save the audio to.
175
+
176
+ """
177
+ ext = path.split(".")[-1].lower()
178
+ if ext in ("wav", "flac"):
179
+ sf.write(path, self.array.T, self.sample_rate)
180
+ elif ext == "npz":
181
+ np.savez_compressed(path, samples=self.array, sample_rate=self.sample_rate)
182
+ elif ext == "pt":
183
+ torch.save(
184
+ {"samples": self.tensor.cpu(), "sample_rate": self.sample_rate},
185
+ path,
186
+ )
187
+ else:
188
+ self.array.tofile(path)
189
+
167
190
  def normalize(self, eps: float = 1e-5) -> Audio:
168
191
  """Normalize the audio samples to have zero mean and unit variance.
169
192
 
@@ -237,6 +260,40 @@ class Features(Container):
237
260
  """
238
261
  return self.data.shape[1]
239
262
 
263
+ def tofile(self, path: str, double: bool = False) -> None:
264
+ """Save the features to a file.
265
+
266
+ Parameters
267
+ ----------
268
+ path : str
269
+ The path to save the features to.
270
+
271
+ double : bool, optional
272
+ Whether to save the features in double precision instead of single one.
273
+
274
+ """
275
+ ext = path.split(".")[-1].lower()
276
+ if ext == "npz":
277
+ np.savez_compressed(
278
+ path,
279
+ features=self.array.astype(np.float64 if double else np.float32),
280
+ source=self.source,
281
+ layers=self.layers or [],
282
+ )
283
+ elif ext == "pt":
284
+ torch.save(
285
+ {
286
+ "features": self.tensor.cpu().to(
287
+ torch.float64 if double else torch.float32
288
+ ),
289
+ "source": self.source,
290
+ "layers": self.layers,
291
+ },
292
+ path,
293
+ )
294
+ else:
295
+ self.array.astype(np.float64 if double else np.float32).tofile(path)
296
+
240
297
  def trim(self, start: int, end: int) -> Features:
241
298
  """Trim the features along the time dimension.
242
299
 
@@ -9,11 +9,13 @@ from .data2vec2 import Data2Vec2Model
9
9
  from .ecapa_tdnn import EcapaTDNNModel
10
10
  from .emotion2vec import Emotion2VecModel
11
11
  from .emotion2vec_plus import Emotion2VecPlusModel
12
+ from .higgs_audio import HiggsAudioTokenizerModel
12
13
  from .hubert import HuBERTModel
13
14
  from .manager import ModelManager
14
15
  from .next_tdnn import NeXtTDNNModel
15
16
  from .r_spin import RSpinModel
16
17
  from .r_vector import RVectorModel
18
+ from .redimnet import ReDimNetModel
17
19
  from .spidr import SpidRModel
18
20
  from .spin import SpinModel
19
21
  from .sslzip import SSLZipModel
@@ -30,10 +32,12 @@ MODEL_MAP = {
30
32
  "ecapa-tdnn": EcapaTDNNModel,
31
33
  "emotion2vec": Emotion2VecModel,
32
34
  "emotion2vec+": Emotion2VecPlusModel,
35
+ "higgs-audio": HiggsAudioTokenizerModel,
33
36
  "hubert": HuBERTModel,
34
37
  "next-tdnn": NeXtTDNNModel,
35
38
  "r-spin": RSpinModel,
36
39
  "r-vector": RVectorModel,
40
+ "redimnet": ReDimNetModel,
37
41
  "spidr": SpidRModel,
38
42
  "spin": SpinModel,
39
43
  "sslzip": SSLZipModel,
@@ -256,6 +256,58 @@ class FrameLevelFeatureModel(BaseModel):
256
256
  return Granularity.FRAME
257
257
 
258
258
 
259
+ class TokenLevelFeatureModel(BaseModel):
260
+ """An abstract base class for frame-level feature extraction models."""
261
+
262
+ @property
263
+ def num_layers(self) -> int:
264
+ """Get the number of layers in the model.
265
+
266
+ Returns
267
+ -------
268
+ out : int
269
+ The number of layers.
270
+
271
+ """
272
+ return 0
273
+
274
+ @property
275
+ def frame_shift(self) -> int:
276
+ """Get the frame shift of the model.
277
+
278
+ Returns
279
+ -------
280
+ out : int
281
+ The frame shift in samples.
282
+
283
+ """
284
+ return int(40.0 * self.sample_rate / 1000)
285
+
286
+ @property
287
+ def center_offset(self) -> int:
288
+ """Get the center offset of the model.
289
+
290
+ Returns
291
+ -------
292
+ out : int
293
+ The center offset in samples.
294
+
295
+ """
296
+ return 0
297
+
298
+ @property
299
+ def granularity(self) -> Granularity:
300
+ """Get the granularity of the features extracted by the model.
301
+
302
+ Returns
303
+ -------
304
+ out : str
305
+ The granularity of the features.
306
+
307
+ """
308
+ return Granularity.FRAME
309
+
310
+
259
311
  class UtteranceLevelFeatureModel(BaseModel):
260
312
  """An abstract base class for utterance-level feature extraction models."""
261
313
 
@@ -10,7 +10,7 @@ import torch
10
10
  from ..interfaces.types import Audio, Features
11
11
  from ..utils.io import silence_hf_hub
12
12
  from ..utils.paths import setup_third_party_path
13
- from ..utils.validation import validate_enum
13
+ from ..utils.validation import validate_enum, validate_length
14
14
  from .base import UtteranceLevelFeatureModel
15
15
 
16
16
 
@@ -108,6 +108,8 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
108
108
  raise RuntimeError("Model is not loaded. Call 'load' method first.")
109
109
 
110
110
  with torch.inference_mode():
111
- vectors = self.model.encode_batch(audio.tensor.to(self.device)) # (B, N, D)
111
+ inputs = audio.tensor.to(self.device)
112
+ inputs = validate_length(inputs, 640)
113
+ vectors = self.model.encode_batch(inputs)
112
114
 
113
115
  return Features(data=vectors, source=self.model_id)
@@ -0,0 +1,136 @@
1
+ # Copyright (c) 2026 Takenori Yoshimura
2
+ # Released under the MIT License.
3
+
4
+ """A module for the Higgs Audio tokenizer."""
5
+
6
+ from enum import Enum
7
+ from typing import Any
8
+
9
+ import torch
10
+
11
+ from ..interfaces.types import Audio, Features
12
+ from ..utils.io import silence_transformers
13
+ from ..utils.validation import validate_enum
14
+ from .base import TokenLevelFeatureModel
15
+
16
+
17
+ class HiggsAudioTokenizerVariant(str, Enum):
18
+ """Enumeration of supported Higgs Audio tokenizer variants."""
19
+
20
+ V2 = "v2"
21
+
22
+ @property
23
+ def model_name(self) -> str:
24
+ """Return the model name corresponding to the variant.
25
+
26
+ Returns
27
+ -------
28
+ out : str
29
+ The model name corresponding to the variant.
30
+
31
+ """
32
+ return f"eustlb/higgs-audio-{self.value}-tokenizer"
33
+
34
+
35
+ class HiggsAudioTokenizerModel(TokenLevelFeatureModel):
36
+ """A class for the Higgs Audio tokenizer model."""
37
+
38
+ def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
39
+ """Initialize the Higgs Audio tokenizer model.
40
+
41
+ Parameters
42
+ ----------
43
+ variant : str | None, optional
44
+ The variant of the model to use.
45
+
46
+ device : str, optional
47
+ The device to run the model on (e.g., 'cpu' or 'cuda').
48
+
49
+ """
50
+ super().__init__(variant, device)
51
+
52
+ self.variant = validate_enum(
53
+ variant, HiggsAudioTokenizerVariant, HiggsAudioTokenizerVariant.V2
54
+ )
55
+ self._model_id = f"higgs-audio-{self.variant.value}"
56
+
57
+ self.feature_extractor = None
58
+
59
+ def load(self, model_dir: str, quiet: bool = False) -> None:
60
+ """Load the model from the specified directory.
61
+
62
+ Parameters
63
+ ----------
64
+ model_dir : str
65
+ The directory where the model checkpoint will be stored.
66
+
67
+ quiet : bool, optional
68
+ Whether to suppress output during the loading process.
69
+
70
+ """
71
+ if self.model is not None:
72
+ return
73
+
74
+ from transformers import AutoFeatureExtractor, HiggsAudioV2TokenizerModel
75
+
76
+ with silence_transformers(quiet):
77
+ self.feature_extractor = AutoFeatureExtractor.from_pretrained(
78
+ self.variant.model_name, cache_dir=model_dir
79
+ )
80
+ self.model = HiggsAudioV2TokenizerModel.from_pretrained(
81
+ self.variant.model_name, cache_dir=model_dir
82
+ )
83
+ self.model.eval()
84
+ self.model.to(self.device) # type: ignore
85
+
86
+ def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
87
+ """Extract features from the input audio using the model.
88
+
89
+ Parameters
90
+ ----------
91
+ audio : Audio
92
+ The input audio data with shape (B, T).
93
+
94
+ layers : list[int]
95
+ The layer(s) from which to extract features.
96
+
97
+ Returns
98
+ -------
99
+ out : Features
100
+ The extracted features.
101
+
102
+ Raises
103
+ ------
104
+ RuntimeError
105
+ If the model is not loaded.
106
+
107
+ """
108
+ if self.feature_extractor is None or self.model is None:
109
+ raise RuntimeError("Model not loaded. Call 'load' method first.")
110
+
111
+ with torch.inference_mode():
112
+ inputs = self.feature_extractor(
113
+ raw_audio=[x for x in audio.array],
114
+ sampling_rate=self.feature_extractor.sampling_rate,
115
+ return_tensors="pt",
116
+ ).to(self.device)
117
+
118
+ encoder_outputs: Any = self.model.encode(inputs["input_values"])
119
+ indices = encoder_outputs.audio_codes # (B, Q, N)
120
+ indices = indices.transpose(0, 1)
121
+ vectors = self.model.quantizer.decode(indices) # (B, D, N)
122
+ vectors = vectors.transpose(1, 2)
123
+
124
+ return Features(data=vectors, source=self.model_id)
125
+
126
+ @property
127
+ def sample_rate(self) -> int:
128
+ """Get the sample rate required by the model.
129
+
130
+ Returns
131
+ -------
132
+ out : int
133
+ The sample rate in Hz.
134
+
135
+ """
136
+ return 24000
@@ -11,7 +11,7 @@ import torch
11
11
  from ..interfaces.types import Audio, Features
12
12
  from ..utils.io import download_file
13
13
  from ..utils.paths import setup_third_party_path
14
- from ..utils.validation import validate_enum
14
+ from ..utils.validation import validate_enum, validate_length
15
15
  from .base import UtteranceLevelFeatureModel
16
16
 
17
17
 
@@ -140,7 +140,9 @@ class NeXtTDNNModel(UtteranceLevelFeatureModel):
140
140
  raise RuntimeError("Model is not loaded. Call 'load' method first.")
141
141
 
142
142
  with torch.inference_mode():
143
- vectors = self.model(audio.tensor.to(self.device))
143
+ inputs = audio.tensor.to(self.device)
144
+ inputs = validate_length(inputs, 640)
145
+ vectors = self.model(inputs)
144
146
  vectors = vectors.unsqueeze(1)
145
147
 
146
148
  return Features(data=vectors, source=self.model_id)
@@ -0,0 +1,106 @@
1
+ # Copyright (c) 2026 Takenori Yoshimura
2
+ # Released under the MIT License.
3
+
4
+ """A module for the ReDimNet model."""
5
+
6
+ from enum import Enum
7
+
8
+ import torch
9
+
10
+ from ..interfaces.types import Audio, Features
11
+ from ..utils.io import safe_torch_hub_load
12
+ from ..utils.validation import validate_enum, validate_length
13
+ from .base import UtteranceLevelFeatureModel
14
+
15
+
16
+ class ReDimNetVariant(str, Enum):
17
+ """Enumeration of supported ReDimNet model variants."""
18
+
19
+ B0 = "b0"
20
+ B1 = "b1"
21
+ B2 = "b2"
22
+ B3 = "b3"
23
+ B4 = "b4"
24
+ B5 = "b5"
25
+ B6 = "b6"
26
+
27
+
28
+ class ReDimNetModel(UtteranceLevelFeatureModel):
29
+ """A class for the ReDimNet model."""
30
+
31
+ def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
32
+ """Initialize the ReDimNet model.
33
+
34
+ Parameters
35
+ ----------
36
+ variant : str | None, optional
37
+ The variant of the model to use.
38
+
39
+ device : str, optional
40
+ The device to run the model on (e.g., 'cpu' or 'cuda').
41
+
42
+ """
43
+ super().__init__(variant, device)
44
+
45
+ self.variant = validate_enum(variant, ReDimNetVariant, ReDimNetVariant.B2)
46
+ self._model_id = f"redimnet-{self.variant.value}"
47
+
48
+ def load(self, model_dir: str, quiet: bool = False) -> None:
49
+ """Load the model from the specified directory.
50
+
51
+ Parameters
52
+ ----------
53
+ model_dir : str
54
+ The directory where the model checkpoint will be stored.
55
+
56
+ quiet : bool, optional
57
+ Whether to suppress output during the loading process.
58
+
59
+ """
60
+ if self.model is not None:
61
+ return
62
+
63
+ self.model = safe_torch_hub_load(
64
+ "IDRnD/ReDimNet",
65
+ "ReDimNet",
66
+ model_dir,
67
+ quiet=quiet,
68
+ model_name=self.variant.value,
69
+ train_type="ft_lm",
70
+ dataset="vox2",
71
+ )
72
+ self.model.eval()
73
+ self.model.to(self.device)
74
+
75
+ def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
76
+ """Extract features from the input audio using the model.
77
+
78
+ Parameters
79
+ ----------
80
+ audio : Audio
81
+ The input audio data with shape (B, T).
82
+
83
+ layers : list[int]
84
+ The layer(s) from which to extract features.
85
+
86
+ Returns
87
+ -------
88
+ out : Features
89
+ The extracted features.
90
+
91
+ Raises
92
+ ------
93
+ RuntimeError
94
+ If the model is not loaded.
95
+
96
+ """
97
+ if self.model is None:
98
+ raise RuntimeError("Model is not loaded. Call 'load' method first.")
99
+
100
+ with torch.inference_mode():
101
+ inputs = audio.tensor.to(self.device)
102
+ inputs = validate_length(inputs, 320)
103
+ vectors = self.model(inputs)
104
+ vectors = vectors.unsqueeze(1)
105
+
106
+ return Features(data=vectors, source=self.model_id)
@@ -10,7 +10,7 @@ import torch
10
10
  from ..interfaces.types import Audio, Features
11
11
  from ..utils.io import silence_hf_hub
12
12
  from ..utils.paths import setup_third_party_path
13
- from ..utils.validation import validate_enum
13
+ from ..utils.validation import validate_enum, validate_length
14
14
  from .base import UtteranceLevelFeatureModel
15
15
 
16
16
 
@@ -108,6 +108,8 @@ class XVectorModel(UtteranceLevelFeatureModel):
108
108
  raise RuntimeError("Model is not loaded. Call 'load' method first.")
109
109
 
110
110
  with torch.inference_mode():
111
- vectors = self.model.encode_batch(audio.tensor.to(self.device))
111
+ inputs = audio.tensor.to(self.device)
112
+ inputs = validate_length(inputs, 640)
113
+ vectors = self.model.encode_batch(inputs)
112
114
 
113
115
  return Features(data=vectors, source=self.model_id)
@@ -5,11 +5,13 @@
5
5
 
6
6
  from .lilfilter import LilFilterResampler
7
7
  from .manager import ResamplerManager
8
+ from .scipy import ScipyResampler
8
9
  from .soxr import SoxrResampler
9
10
  from .torchaudio import TorchAudioResampler
10
11
 
11
12
  RESAMPLER_MAP = {
12
13
  "lilfilter": LilFilterResampler,
14
+ "scipy": ScipyResampler,
13
15
  "soxr": SoxrResampler,
14
16
  "torchaudio": TorchAudioResampler,
15
17
  }