lfeats 0.1.4__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. {lfeats-0.1.4 → lfeats-0.2.1}/.gitignore +1 -0
  2. {lfeats-0.1.4 → lfeats-0.2.1}/PKG-INFO +22 -5
  3. {lfeats-0.1.4 → lfeats-0.2.1}/README.md +17 -2
  4. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/cli.py +4 -25
  5. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/extractor.py +12 -0
  6. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/resampler.py +11 -0
  7. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/types.py +57 -0
  8. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/__init__.py +4 -0
  9. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/base.py +71 -0
  10. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/contentvec.py +0 -2
  11. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/data2vec.py +0 -3
  12. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/data2vec2.py +0 -3
  13. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/ecapa_tdnn.py +5 -5
  14. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/emotion2vec.py +1 -3
  15. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/emotion2vec_plus.py +2 -4
  16. lfeats-0.2.1/lfeats/models/higgs_audio.py +136 -0
  17. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/hubert.py +0 -2
  18. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/manager.py +13 -0
  19. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/next_tdnn.py +4 -4
  20. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/r_spin.py +0 -2
  21. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/r_vector.py +1 -3
  22. lfeats-0.2.1/lfeats/models/redimnet.py +106 -0
  23. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/spidr.py +4 -8
  24. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/spin.py +3 -4
  25. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/sslzip.py +14 -1
  26. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/unispeech_sat.py +0 -2
  27. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/wav2vec2.py +0 -3
  28. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/wavlm.py +0 -2
  29. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/whisper.py +0 -1
  30. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/models/x_vector.py +5 -5
  31. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/__init__.py +2 -0
  32. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/base.py +11 -0
  33. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/lilfilter.py +13 -1
  34. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/manager.py +13 -0
  35. lfeats-0.2.1/lfeats/resamplers/scipy.py +90 -0
  36. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/torchaudio.py +14 -1
  37. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/download.py +24 -10
  38. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/distributed.py +1 -1
  39. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/io.py +50 -6
  40. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/validation.py +27 -0
  41. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/version.py +1 -1
  42. {lfeats-0.1.4 → lfeats-0.2.1}/pyproject.toml +4 -2
  43. {lfeats-0.1.4 → lfeats-0.2.1}/LICENSE +0 -0
  44. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/__init__.py +0 -0
  45. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/__init__.py +0 -0
  46. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/interfaces/utils.py +0 -0
  47. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/resamplers/soxr.py +0 -0
  48. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/__init__.py +0 -0
  49. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/LICENSE +0 -0
  50. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/__init__.py +0 -0
  51. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/checkpoint_utils.py +0 -0
  52. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/__init__.py +0 -0
  53. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/config/config.yaml +0 -0
  54. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/__init__.py +0 -0
  55. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/dictionary.py +0 -0
  56. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/modality.py +0 -0
  57. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/data/text_compressor.py +0 -0
  58. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/__init__.py +0 -0
  59. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/configs.py +0 -0
  60. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/constants.py +0 -0
  61. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/initialize.py +0 -0
  62. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/dataclass/utils.py +0 -0
  63. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/file_io.py +0 -0
  64. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/incremental_decoding_utils.py +0 -0
  65. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/__init__.py +0 -0
  66. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/logging/meters.py +0 -0
  67. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/__init__.py +0 -0
  68. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/__init__.py +0 -0
  69. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec2.py +0 -0
  70. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/data2vec_audio.py +0 -0
  71. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/__init__.py +0 -0
  72. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/audio.py +0 -0
  73. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/base.py +0 -0
  74. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/data2vec/modalities/modules.py +0 -0
  75. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_decoder.py +0 -0
  76. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_encoder.py +0 -0
  77. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_incremental_decoder.py +0 -0
  78. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/fairseq_model.py +0 -0
  79. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/__init__.py +0 -0
  80. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/hubert/hubert.py +0 -0
  81. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/__init__.py +0 -0
  82. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/utils.py +0 -0
  83. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/models/wav2vec/wav2vec2.py +0 -0
  84. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/__init__.py +0 -0
  85. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/ema_module.py +0 -0
  86. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fairseq_dropout.py +0 -0
  87. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/fp32_group_norm.py +0 -0
  88. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gelu.py +0 -0
  89. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/gumbel_vector_quantizer.py +0 -0
  90. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/layer_norm.py +0 -0
  91. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/multihead_attention.py +0 -0
  92. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/quant_noise.py +0 -0
  93. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/same_pad.py +0 -0
  94. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/modules/transpose_last.py +0 -0
  95. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/quantization_utils.py +0 -0
  96. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/registry.py +0 -0
  97. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/__init__.py +0 -0
  98. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/audio_pretraining.py +0 -0
  99. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/fairseq_task.py +0 -0
  100. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tasks/hubert_pretraining.py +0 -0
  101. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/tokenizer.py +0 -0
  102. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/fairseq/utils.py +0 -0
  103. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/LICENSE +0 -0
  104. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/SpeakerNet.py +0 -0
  105. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/__init__.py +0 -0
  106. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/__init__.py +0 -0
  107. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/aggregation/vap_bn_tanh_fc_bn.py +0 -0
  108. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7.py +0 -0
  109. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_C256_B3_K65_7_cyclical_lr_step.py +0 -0
  110. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/NeXt_TDNN_light_C256_B3_K65.py +0 -0
  111. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/configs/__init__.py +0 -0
  112. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/main.py +0 -0
  113. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/NeXt_TDNN.py +0 -0
  114. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt.py +0 -0
  115. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/TSConvNeXt_light.py +0 -0
  116. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/__init__.py +0 -0
  117. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/models/utils.py +0 -0
  118. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/__init__.py +0 -0
  119. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/next_tdnn_asv/preprocessing/mel_transform.py +0 -0
  120. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/LICENSE +0 -0
  121. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/__init__.py +0 -0
  122. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/model.py +0 -0
  123. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/rspin/wavlm_config.py +0 -0
  124. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/LICENSE +0 -0
  125. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/__init__.py +0 -0
  126. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/__init__.py +0 -0
  127. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/__init__.py +0 -0
  128. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/convert.py +0 -0
  129. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/hubert/hubert_model.py +0 -0
  130. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/utils.py +0 -0
  131. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/__init__.py +0 -0
  132. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wav2vec2/wav2vec2_model.py +0 -0
  133. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/WavLM.py +0 -0
  134. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/__init__.py +0 -0
  135. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/upstream/wavlm/modules.py +0 -0
  136. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/s3prl/util/__init__.py +0 -0
  137. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/LICENSE +0 -0
  138. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/__init__.py +0 -0
  139. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/__init__.py +0 -0
  140. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/dataio.py +0 -0
  141. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/encoder.py +0 -0
  142. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/dataio/preprocess.py +0 -0
  143. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/__init__.py +0 -0
  144. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/classifiers.py +0 -0
  145. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/inference/interfaces.py +0 -0
  146. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/__init__.py +0 -0
  147. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/features.py +0 -0
  148. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ECAPA_TDNN.py +0 -0
  149. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/ResNet.py +0 -0
  150. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/Xvector.py +0 -0
  151. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/lobes/models/__init__.py +0 -0
  152. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/CNN.py +0 -0
  153. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/containers.py +0 -0
  154. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/linear.py +0 -0
  155. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/normalization.py +0 -0
  156. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/nnet/pooling.py +0 -0
  157. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/__init__.py +0 -0
  158. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/processing/features.py +0 -0
  159. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/__init__.py +0 -0
  160. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/_workarounds.py +0 -0
  161. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/autocast.py +0 -0
  162. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/checkpoints.py +0 -0
  163. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/fetching.py +0 -0
  164. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/filter_analysis.py +0 -0
  165. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/logger.py +0 -0
  166. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/parameter_transfer.py +0 -0
  167. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/speechbrain/utils/run_opts.py +0 -0
  168. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/LICENSE +0 -0
  169. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/__init__.py +0 -0
  170. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/__init__.py +0 -0
  171. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/base.py +0 -0
  172. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/model/spin.py +0 -0
  173. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/__init__.py +0 -0
  174. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/dnn.py +0 -0
  175. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/hubert.py +0 -0
  176. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/swav_vq_dis.py +0 -0
  177. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/nn/wavlm.py +0 -0
  178. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/__init__.py +0 -0
  179. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/model_utils.py +0 -0
  180. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/spin/util/padding.py +0 -0
  181. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/LICENSE +0 -0
  182. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/__init__.py +0 -0
  183. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/__init__.py +0 -0
  184. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/drop.py +0 -0
  185. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/helpers.py +0 -0
  186. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/third_party/timm/layers/mlp.py +0 -0
  187. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/__init__.py +0 -0
  188. {lfeats-0.1.4 → lfeats-0.2.1}/lfeats/utils/paths.py +0 -0
@@ -11,6 +11,7 @@ build/
11
11
 
12
12
  # tests
13
13
  tests/outputs/
14
+ *.wav
14
15
 
15
16
  # tools
16
17
  tools/**/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: lfeats
3
- Version: 0.1.4
3
+ Version: 0.2.1
4
4
  Summary: A unified interface to extract hidden representations from speech foundation models
5
5
  Project-URL: Documentation, https://takenori-y.github.io/lfeats/stable/
6
6
  Project-URL: Source, https://github.com/takenori-y/lfeats
@@ -18,6 +18,7 @@ Classifier: Programming Language :: Python :: 3.12
18
18
  Classifier: Programming Language :: Python :: 3.13
19
19
  Classifier: Programming Language :: Python :: 3.14
20
20
  Requires-Python: >=3.10
21
+ Requires-Dist: filelock>=3.10.0
21
22
  Requires-Dist: huggingface-hub>=0.23.0
22
23
  Requires-Dist: hydra-core>=1.3.0
23
24
  Requires-Dist: hyperpyyaml>=0.0.1
@@ -27,11 +28,12 @@ Requires-Dist: onnxruntime>=1.19.0
27
28
  Requires-Dist: parfive>=2.1.0
28
29
  Requires-Dist: platformdirs>=2.0.0
29
30
  Requires-Dist: requests>=2.27.0
31
+ Requires-Dist: scipy>=1.9.2
30
32
  Requires-Dist: soundfile>=0.10.2
31
33
  Requires-Dist: soxr>=0.4.0
32
34
  Requires-Dist: torch>=2.6.0
33
35
  Requires-Dist: torchaudio>=2.6.0
34
- Requires-Dist: transformers>=4.30.0
36
+ Requires-Dist: transformers>=5.3.0
35
37
  Provides-Extra: dev
36
38
  Requires-Dist: build; extra == 'dev'
37
39
  Requires-Dist: matplotlib; extra == 'dev'
@@ -39,7 +41,7 @@ Requires-Dist: mdformat; extra == 'dev'
39
41
  Requires-Dist: numpydoc; extra == 'dev'
40
42
  Requires-Dist: pkginfo; extra == 'dev'
41
43
  Requires-Dist: pydata-sphinx-theme; extra == 'dev'
42
- Requires-Dist: pyright; extra == 'dev'
44
+ Requires-Dist: pyright<=1.1.408; extra == 'dev'
43
45
  Requires-Dist: pytest; extra == 'dev'
44
46
  Requires-Dist: pytest-cov; extra == 'dev'
45
47
  Requires-Dist: ruff; extra == 'dev'
@@ -142,6 +144,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
142
144
  | | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
143
145
  | | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
144
146
 
147
+ ### Token-Level Features
148
+
149
+ | Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
150
+ | :--- | :--- | ---: | ---: | :---: | :---: | :---: |
151
+ | `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
152
+
145
153
  ### Utterance-Level Features
146
154
 
147
155
  | Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
@@ -150,7 +158,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
150
158
  | `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
151
159
  | | `base` | 0 | 192 | | | |
152
160
  | | `base-v2` | 0 | 192 | | | |
153
- | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/pdf/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
161
+ | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
162
+ | `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
163
+ | | `b1` | 0 | 192 | | | |
164
+ | | `b2` | 0 | 192 | | | |
165
+ | | `b3` | 0 | 192 | | | |
166
+ | | `b4` | 0 | 192 | | | |
167
+ | | `b5` | 0 | 192 | | | |
168
+ | | `b6` | 0 | 192 | | | |
154
169
  | `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
155
170
 
156
171
  > [!IMPORTANT]
@@ -162,12 +177,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
162
177
  | Resampler Type | Quality Preset | Source | License |
163
178
  | :--- | :--- | :---: | :--- |
164
179
  | `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
180
+ | `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
181
+ | | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
165
182
  | `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
166
183
  | | `low` | | |
167
184
  | | `medium` | | |
168
185
  | | `high` | | |
169
186
  | | `very-high` | | |
170
- | `torchaudio` | `kaiser-fast` | [GitHub](https://github.com/pytorch/audio) | BSD 2-Clause |
187
+ | `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
171
188
  | | `kaiser-best` | | |
172
189
 
173
190
  ## Examples
@@ -93,6 +93,12 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
93
93
  | | `large-v2` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v2) |
94
94
  | | `large-v3` | 32 | 1280 | | | [🤗](https://huggingface.co/openai/whisper-large-v3) |
95
95
 
96
+ ### Token-Level Features
97
+
98
+ | Model Name | Model Variant | Hop Size [ms] | Dimension | Paper | Source | Model Hub |
99
+ | :--- | :--- | ---: | ---: | :---: | :---: | :---: |
100
+ | `higgs-audio` | `v2` | 40 | 1024 | | [GitHub](https://github.com/boson-ai/higgs-audio) | [🤗](https://huggingface.co/eustlb/higgs-audio-v2-tokenizer) |
101
+
96
102
  ### Utterance-Level Features
97
103
 
98
104
  | Model Name | Model Variant | Layers | Dimension | Paper | Source | Model Hub |
@@ -101,7 +107,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
101
107
  | `next-tdnn` | `light` | 0 | 192 | [arXiv](https://arxiv.org/abs/2312.08603) | [GitHub](https://github.com/dmlguq456/NeXt_TDNN_ASV) | |
102
108
  | | `base` | 0 | 192 | | | |
103
109
  | | `base-v2` | 0 | 192 | | | |
104
- | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/pdf/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
110
+ | `r-vector` | `base` | 0 | 256 | [arXiv](https://arxiv.org/abs/1910.12592) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-resnet-voxceleb) |
111
+ | `redimnet` | `b0` | 0 | 192 | [arXiv](https://arxiv.org/abs/2407.18223) | [GitHub](https://github.com/IDRnD/redimnet) | |
112
+ | | `b1` | 0 | 192 | | | |
113
+ | | `b2` | 0 | 192 | | | |
114
+ | | `b3` | 0 | 192 | | | |
115
+ | | `b4` | 0 | 192 | | | |
116
+ | | `b5` | 0 | 192 | | | |
117
+ | | `b6` | 0 | 192 | | | |
105
118
  | `x-vector` | `base` | 0 | 512 | [IEEE](https://ieeexplore.ieee.org/document/8461375) | [GitHub](https://github.com/speechbrain/speechbrain) | [🤗](https://huggingface.co/speechbrain/spkrec-xvect-voxceleb) |
106
119
 
107
120
  > [!IMPORTANT]
@@ -113,12 +126,14 @@ pip install git+https://github.com/takenori-y/lfeats.git@master
113
126
  | Resampler Type | Quality Preset | Source | License |
114
127
  | :--- | :--- | :---: | :--- |
115
128
  | `lilfilter` | `base` | [GitHub](https://github.com/danpovey/filtering) | MIT |
129
+ | `scipy` | `fft` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample.html) | BSD 3-Clause |
130
+ | | `poly` | [Document](https://docs.scipy.org/doc/scipy/reference/generated/scipy.signal.resample_poly.html) | |
116
131
  | `soxr` | `quick` | [GitHub](https://github.com/dofuuz/python-soxr) | LGPL v2.1+ |
117
132
  | | `low` | | |
118
133
  | | `medium` | | |
119
134
  | | `high` | | |
120
135
  | | `very-high` | | |
121
- | `torchaudio` | `kaiser-fast` | [GitHub](https://github.com/pytorch/audio) | BSD 2-Clause |
136
+ | `torchaudio` | `kaiser-fast` | [Document](https://docs.pytorch.org/audio/stable/tutorials/audio_resampling_tutorial.html) | BSD 2-Clause |
122
137
  | | `kaiser-best` | | |
123
138
 
124
139
  ## Examples
@@ -147,9 +147,6 @@ def main() -> None:
147
147
  datefmt="%Y-%m-%d %H:%M:%S",
148
148
  )
149
149
 
150
- import numpy as np
151
- import torch
152
-
153
150
  import lfeats
154
151
  from lfeats.utils.io import load_audio
155
152
 
@@ -170,7 +167,8 @@ def main() -> None:
170
167
  raise ValueError(f"Invalid source: {args.source}")
171
168
 
172
169
  if len(input_files) == 0:
173
- raise ValueError(f"No audio files found in the source: {args.source}")
170
+ logging.info(f"No audio files found in the source: {args.source}")
171
+ sys.exit(0)
174
172
  logger.info(f"Found {len(input_files)} audio files to process.")
175
173
 
176
174
  # Parse the layers argument.
@@ -216,7 +214,7 @@ def main() -> None:
216
214
  path = Path(input_file).parent
217
215
  # Remove the root part of the path.
218
216
  dirs = path.relative_to(path.anchor).parts
219
- if args.subdir_offset >= len(dirs):
217
+ if args.subdir_offset > len(dirs):
220
218
  logger.error(
221
219
  f"Subdir offset {args.subdir_offset} is too large for file: "
222
220
  f"{input_file}. Skipping."
@@ -252,26 +250,7 @@ def main() -> None:
252
250
  num_errors += 1
253
251
  continue
254
252
 
255
- if args.output_format == "npz":
256
- result = {
257
- "features": features.array,
258
- "source": features.source,
259
- "layers": features.layers,
260
- }
261
- np.savez_compressed(output_file, **result)
262
- elif args.output_format == "pt":
263
- result = {
264
- "features": features.tensor.cpu(),
265
- "source": features.source,
266
- "layers": features.layers,
267
- }
268
- torch.save(result, output_file)
269
- elif args.output_format == "float":
270
- features.array.tofile(output_file)
271
- elif args.output_format == "double":
272
- features.array.astype(np.float64).tofile(output_file)
273
- else:
274
- raise ValueError(f"Unsupported output format: {args.output_format}")
253
+ features.tofile(output_file, double=args.output_format == "double")
275
254
 
276
255
  if num_errors > 0:
277
256
  logger.error(f"{num_errors} files were skipped due to errors.")
@@ -90,6 +90,18 @@ class Extractor:
90
90
  """
91
91
  self.model_manager.get_model().load(self.cache_dir, quiet)
92
92
 
93
+ def to(self, device: str) -> None:
94
+ """Move the model to the specified device.
95
+
96
+ Parameters
97
+ ----------
98
+ device : str
99
+ The device to move the model to (e.g., 'cpu' or 'cuda').
100
+
101
+ """
102
+ self.model_manager.to(device)
103
+ self.resampler_manager.to(device)
104
+
93
105
  def __call__(
94
106
  self,
95
107
  source: np.ndarray | torch.Tensor | Audio,
@@ -45,6 +45,17 @@ class Resampler:
45
45
  f"Supported resamplers are: {[k for k in RESAMPLER_MAP.keys()]}"
46
46
  ) from e
47
47
 
48
+ def to(self, device: str) -> None:
49
+ """Move the resampler to the specified device.
50
+
51
+ Parameters
52
+ ----------
53
+ device : str
54
+ The device to move the resampler to (e.g., 'cpu' or 'cuda').
55
+
56
+ """
57
+ self.resampler_manager.to(device)
58
+
48
59
  def __call__(
49
60
  self,
50
61
  source: np.ndarray | torch.Tensor | Audio,
@@ -9,6 +9,7 @@ from dataclasses import dataclass
9
9
  from enum import Enum
10
10
 
11
11
  import numpy as np
12
+ import soundfile as sf
12
13
  import torch
13
14
  import torch.nn.functional as F
14
15
 
@@ -164,6 +165,28 @@ class Audio(Container):
164
165
  """
165
166
  return self.data.shape[1]
166
167
 
168
+ def tofile(self, path: str) -> None:
169
+ """Save the audio to a file.
170
+
171
+ Parameters
172
+ ----------
173
+ path : str
174
+ The path to save the audio to.
175
+
176
+ """
177
+ ext = path.split(".")[-1].lower()
178
+ if ext in ("wav", "flac"):
179
+ sf.write(path, self.array.T, self.sample_rate)
180
+ elif ext == "npz":
181
+ np.savez_compressed(path, samples=self.array, sample_rate=self.sample_rate)
182
+ elif ext == "pt":
183
+ torch.save(
184
+ {"samples": self.tensor.cpu(), "sample_rate": self.sample_rate},
185
+ path,
186
+ )
187
+ else:
188
+ self.array.tofile(path)
189
+
167
190
  def normalize(self, eps: float = 1e-5) -> Audio:
168
191
  """Normalize the audio samples to have zero mean and unit variance.
169
192
 
@@ -237,6 +260,40 @@ class Features(Container):
237
260
  """
238
261
  return self.data.shape[1]
239
262
 
263
+ def tofile(self, path: str, double: bool = False) -> None:
264
+ """Save the features to a file.
265
+
266
+ Parameters
267
+ ----------
268
+ path : str
269
+ The path to save the features to.
270
+
271
+ double : bool, optional
272
+ Whether to save the features in double precision instead of single one.
273
+
274
+ """
275
+ ext = path.split(".")[-1].lower()
276
+ if ext == "npz":
277
+ np.savez_compressed(
278
+ path,
279
+ features=self.array.astype(np.float64 if double else np.float32),
280
+ source=self.source,
281
+ layers=self.layers or [],
282
+ )
283
+ elif ext == "pt":
284
+ torch.save(
285
+ {
286
+ "features": self.tensor.cpu().to(
287
+ torch.float64 if double else torch.float32
288
+ ),
289
+ "source": self.source,
290
+ "layers": self.layers,
291
+ },
292
+ path,
293
+ )
294
+ else:
295
+ self.array.astype(np.float64 if double else np.float32).tofile(path)
296
+
240
297
  def trim(self, start: int, end: int) -> Features:
241
298
  """Trim the features along the time dimension.
242
299
 
@@ -9,11 +9,13 @@ from .data2vec2 import Data2Vec2Model
9
9
  from .ecapa_tdnn import EcapaTDNNModel
10
10
  from .emotion2vec import Emotion2VecModel
11
11
  from .emotion2vec_plus import Emotion2VecPlusModel
12
+ from .higgs_audio import HiggsAudioTokenizerModel
12
13
  from .hubert import HuBERTModel
13
14
  from .manager import ModelManager
14
15
  from .next_tdnn import NeXtTDNNModel
15
16
  from .r_spin import RSpinModel
16
17
  from .r_vector import RVectorModel
18
+ from .redimnet import ReDimNetModel
17
19
  from .spidr import SpidRModel
18
20
  from .spin import SpinModel
19
21
  from .sslzip import SSLZipModel
@@ -30,10 +32,12 @@ MODEL_MAP = {
30
32
  "ecapa-tdnn": EcapaTDNNModel,
31
33
  "emotion2vec": Emotion2VecModel,
32
34
  "emotion2vec+": Emotion2VecPlusModel,
35
+ "higgs-audio": HiggsAudioTokenizerModel,
33
36
  "hubert": HuBERTModel,
34
37
  "next-tdnn": NeXtTDNNModel,
35
38
  "r-spin": RSpinModel,
36
39
  "r-vector": RVectorModel,
40
+ "redimnet": ReDimNetModel,
37
41
  "spidr": SpidRModel,
38
42
  "spin": SpinModel,
39
43
  "sslzip": SSLZipModel,
@@ -25,6 +25,7 @@ class BaseModel(ABC):
25
25
  """
26
26
  self.device = device
27
27
 
28
+ self.model = None
28
29
  self._model_id = None # To be defined in subclasses
29
30
 
30
31
  @abstractmethod
@@ -42,6 +43,24 @@ class BaseModel(ABC):
42
43
  """
43
44
  raise NotImplementedError
44
45
 
46
+ def to(self, device: str) -> None:
47
+ """Move the model to the specified device.
48
+
49
+ Parameters
50
+ ----------
51
+ device : str
52
+ The device to move the model to (e.g., 'cpu' or 'cuda').
53
+
54
+ """
55
+ self.device = device
56
+ if self.model is not None and hasattr(self.model, "to"):
57
+ self.model.to(device)
58
+ if hasattr(self.model, "device"):
59
+ try:
60
+ self.model.device = device
61
+ except AttributeError:
62
+ pass
63
+
45
64
  def extract_features(self, audio: Audio, layers: list[int]) -> Features:
46
65
  """Extract features from the input audio data.
47
66
 
@@ -237,6 +256,58 @@ class FrameLevelFeatureModel(BaseModel):
237
256
  return Granularity.FRAME
238
257
 
239
258
 
259
+ class TokenLevelFeatureModel(BaseModel):
260
+ """An abstract base class for frame-level feature extraction models."""
261
+
262
+ @property
263
+ def num_layers(self) -> int:
264
+ """Get the number of layers in the model.
265
+
266
+ Returns
267
+ -------
268
+ out : int
269
+ The number of layers.
270
+
271
+ """
272
+ return 0
273
+
274
+ @property
275
+ def frame_shift(self) -> int:
276
+ """Get the frame shift of the model.
277
+
278
+ Returns
279
+ -------
280
+ out : int
281
+ The frame shift in samples.
282
+
283
+ """
284
+ return int(40.0 * self.sample_rate / 1000)
285
+
286
+ @property
287
+ def center_offset(self) -> int:
288
+ """Get the center offset of the model.
289
+
290
+ Returns
291
+ -------
292
+ out : int
293
+ The center offset in samples.
294
+
295
+ """
296
+ return 0
297
+
298
+ @property
299
+ def granularity(self) -> Granularity:
300
+ """Get the granularity of the features extracted by the model.
301
+
302
+ Returns
303
+ -------
304
+ out : str
305
+ The granularity of the features.
306
+
307
+ """
308
+ return Granularity.FRAME
309
+
310
+
240
311
  class UtteranceLevelFeatureModel(BaseModel):
241
312
  """An abstract base class for utterance-level feature extraction models."""
242
313
 
@@ -68,8 +68,6 @@ class ContentVecModel(FrameLevelFeatureModel):
68
68
  )
69
69
  self._model_id = f"contentvec-{self.variant.value}"
70
70
 
71
- self.model = None
72
-
73
71
  def load(self, model_dir: str, quiet: bool = False) -> None:
74
72
  """Load the model from the specified directory.
75
73
 
@@ -58,9 +58,6 @@ class Data2VecModel(FrameLevelFeatureModel):
58
58
  self.variant = validate_enum(variant, Data2VecVariant, Data2VecVariant.BASE)
59
59
  self._model_id = f"data2vec-{self.variant.value}"
60
60
 
61
- self.processor = None
62
- self.model = None
63
-
64
61
  def load(self, model_dir: str, quiet: bool = False) -> None:
65
62
  """Load the model from the specified directory.
66
63
 
@@ -58,9 +58,6 @@ class Data2Vec2Model(FrameLevelFeatureModel):
58
58
  self.variant = validate_enum(variant, Data2Vec2Variant, Data2Vec2Variant.BASE)
59
59
  self._model_id = f"data2vec2-{self.variant.value}"
60
60
 
61
- self.processor = None
62
- self.model = None
63
-
64
61
  def load(self, model_dir: str, quiet: bool = False) -> None:
65
62
  """Load the model from the specified directory.
66
63
 
@@ -10,7 +10,7 @@ import torch
10
10
  from ..interfaces.types import Audio, Features
11
11
  from ..utils.io import silence_hf_hub
12
12
  from ..utils.paths import setup_third_party_path
13
- from ..utils.validation import validate_enum
13
+ from ..utils.validation import validate_enum, validate_length
14
14
  from .base import UtteranceLevelFeatureModel
15
15
 
16
16
 
@@ -40,8 +40,6 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
40
40
  self.variant = validate_enum(variant, EcapaTDNNVariant, EcapaTDNNVariant.BASE)
41
41
  self._model_id = f"ecapa-tdnn-{self.variant.value}"
42
42
 
43
- self.model = None
44
-
45
43
  def load(self, model_dir: str, quiet: bool = False) -> None:
46
44
  """Load the model from the specified directory.
47
45
 
@@ -78,11 +76,11 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
78
76
  self.model = EncoderClassifier.from_hparams(
79
77
  source="speechbrain/spkrec-ecapa-voxceleb",
80
78
  fetch_config=fetch_config,
79
+ run_opts={"device": self.device},
81
80
  )
82
81
  if self.model is None:
83
82
  raise RuntimeError("Failed to load the model.")
84
83
  self.model.eval()
85
- self.model.to(self.device)
86
84
 
87
85
  def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
88
86
  """Extract features from the input audio using the model.
@@ -110,6 +108,8 @@ class EcapaTDNNModel(UtteranceLevelFeatureModel):
110
108
  raise RuntimeError("Model is not loaded. Call 'load' method first.")
111
109
 
112
110
  with torch.inference_mode():
113
- vectors = self.model.encode_batch(audio.tensor.to(self.device)) # (B, N, D)
111
+ inputs = audio.tensor.to(self.device)
112
+ inputs = validate_length(inputs, 640)
113
+ vectors = self.model.encode_batch(inputs)
114
114
 
115
115
  return Features(data=vectors, source=self.model_id)
@@ -55,8 +55,6 @@ class Emotion2VecModel(FrameLevelFeatureModel):
55
55
  )
56
56
  self._model_id = f"emotion2vec-{self.variant.value}"
57
57
 
58
- self.model = None
59
-
60
58
  def load(self, model_dir: str, quiet: bool = False) -> None:
61
59
  """Load the model from the specified directory.
62
60
 
@@ -78,7 +76,7 @@ class Emotion2VecModel(FrameLevelFeatureModel):
78
76
  repo_id=repo_id,
79
77
  filename=filename,
80
78
  repo_type="model",
81
- local_dir=model_dir,
79
+ cache_dir=model_dir,
82
80
  )
83
81
 
84
82
  setup_third_party_path()
@@ -58,8 +58,6 @@ class Emotion2VecPlusModel(FrameLevelFeatureModel):
58
58
  )
59
59
  self._model_id = f"emotion2vec+-{self.variant.value}"
60
60
 
61
- self.model = None
62
-
63
61
  def load(self, model_dir: str, quiet: bool = False) -> None:
64
62
  """Load the model from the specified directory.
65
63
 
@@ -81,13 +79,13 @@ class Emotion2VecPlusModel(FrameLevelFeatureModel):
81
79
  repo_id=repo_id,
82
80
  filename="model.pt",
83
81
  repo_type="model",
84
- local_dir=os.path.join(model_dir, sanitize(self.model_id)),
82
+ cache_dir=os.path.join(model_dir, sanitize(self.model_id)),
85
83
  )
86
84
  config = hf_hub_download(
87
85
  repo_id=repo_id,
88
86
  filename="config.yaml",
89
87
  repo_type="model",
90
- local_dir=os.path.join(model_dir, sanitize(self.model_id)),
88
+ cache_dir=os.path.join(model_dir, sanitize(self.model_id)),
91
89
  )
92
90
 
93
91
  from hyperpyyaml import load_hyperpyyaml
@@ -0,0 +1,136 @@
1
+ # Copyright (c) 2026 Takenori Yoshimura
2
+ # Released under the MIT License.
3
+
4
+ """A module for the Higgs Audio tokenizer."""
5
+
6
+ from enum import Enum
7
+ from typing import Any
8
+
9
+ import torch
10
+
11
+ from ..interfaces.types import Audio, Features
12
+ from ..utils.io import silence_transformers
13
+ from ..utils.validation import validate_enum
14
+ from .base import TokenLevelFeatureModel
15
+
16
+
17
+ class HiggsAudioTokenizerVariant(str, Enum):
18
+ """Enumeration of supported Higgs Audio tokenizer variants."""
19
+
20
+ V2 = "v2"
21
+
22
+ @property
23
+ def model_name(self) -> str:
24
+ """Return the model name corresponding to the variant.
25
+
26
+ Returns
27
+ -------
28
+ out : str
29
+ The model name corresponding to the variant.
30
+
31
+ """
32
+ return f"eustlb/higgs-audio-{self.value}-tokenizer"
33
+
34
+
35
+ class HiggsAudioTokenizerModel(TokenLevelFeatureModel):
36
+ """A class for the Higgs Audio tokenizer model."""
37
+
38
+ def __init__(self, variant: str | None = None, device: str = "cpu") -> None:
39
+ """Initialize the Higgs Audio tokenizer model.
40
+
41
+ Parameters
42
+ ----------
43
+ variant : str | None, optional
44
+ The variant of the model to use.
45
+
46
+ device : str, optional
47
+ The device to run the model on (e.g., 'cpu' or 'cuda').
48
+
49
+ """
50
+ super().__init__(variant, device)
51
+
52
+ self.variant = validate_enum(
53
+ variant, HiggsAudioTokenizerVariant, HiggsAudioTokenizerVariant.V2
54
+ )
55
+ self._model_id = f"higgs-audio-{self.variant.value}"
56
+
57
+ self.feature_extractor = None
58
+
59
+ def load(self, model_dir: str, quiet: bool = False) -> None:
60
+ """Load the model from the specified directory.
61
+
62
+ Parameters
63
+ ----------
64
+ model_dir : str
65
+ The directory where the model checkpoint will be stored.
66
+
67
+ quiet : bool, optional
68
+ Whether to suppress output during the loading process.
69
+
70
+ """
71
+ if self.model is not None:
72
+ return
73
+
74
+ from transformers import AutoFeatureExtractor, HiggsAudioV2TokenizerModel
75
+
76
+ with silence_transformers(quiet):
77
+ self.feature_extractor = AutoFeatureExtractor.from_pretrained(
78
+ self.variant.model_name, cache_dir=model_dir
79
+ )
80
+ self.model = HiggsAudioV2TokenizerModel.from_pretrained(
81
+ self.variant.model_name, cache_dir=model_dir
82
+ )
83
+ self.model.eval()
84
+ self.model.to(self.device) # type: ignore
85
+
86
+ def extract_features_impl(self, audio: Audio, layers: list[int]) -> Features:
87
+ """Extract features from the input audio using the model.
88
+
89
+ Parameters
90
+ ----------
91
+ audio : Audio
92
+ The input audio data with shape (B, T).
93
+
94
+ layers : list[int]
95
+ The layer(s) from which to extract features.
96
+
97
+ Returns
98
+ -------
99
+ out : Features
100
+ The extracted features.
101
+
102
+ Raises
103
+ ------
104
+ RuntimeError
105
+ If the model is not loaded.
106
+
107
+ """
108
+ if self.feature_extractor is None or self.model is None:
109
+ raise RuntimeError("Model not loaded. Call 'load' method first.")
110
+
111
+ with torch.inference_mode():
112
+ inputs = self.feature_extractor(
113
+ raw_audio=[x for x in audio.array],
114
+ sampling_rate=self.feature_extractor.sampling_rate,
115
+ return_tensors="pt",
116
+ ).to(self.device)
117
+
118
+ encoder_outputs: Any = self.model.encode(inputs["input_values"])
119
+ indices = encoder_outputs.audio_codes # (B, Q, N)
120
+ indices = indices.transpose(0, 1)
121
+ vectors = self.model.quantizer.decode(indices) # (B, D, N)
122
+ vectors = vectors.transpose(1, 2)
123
+
124
+ return Features(data=vectors, source=self.model_id)
125
+
126
+ @property
127
+ def sample_rate(self) -> int:
128
+ """Get the sample rate required by the model.
129
+
130
+ Returns
131
+ -------
132
+ out : int
133
+ The sample rate in Hz.
134
+
135
+ """
136
+ return 24000