asrfront 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. asrfront-0.1.0/.clang-format +5 -0
  2. asrfront-0.1.0/.github/workflows/ci.yml +88 -0
  3. asrfront-0.1.0/.github/workflows/wheels.yml +89 -0
  4. asrfront-0.1.0/.gitignore +10 -0
  5. asrfront-0.1.0/.python-version +1 -0
  6. asrfront-0.1.0/CMakeLists.txt +113 -0
  7. asrfront-0.1.0/LICENSE +21 -0
  8. asrfront-0.1.0/PKG-INFO +144 -0
  9. asrfront-0.1.0/README.md +120 -0
  10. asrfront-0.1.0/bench/bench_fft.c +55 -0
  11. asrfront-0.1.0/bench/bench_frontend.c +91 -0
  12. asrfront-0.1.0/bench/compare.py +289 -0
  13. asrfront-0.1.0/csrc/common.c +25 -0
  14. asrfront-0.1.0/csrc/fft.c +276 -0
  15. asrfront-0.1.0/csrc/fft.h +50 -0
  16. asrfront-0.1.0/csrc/include/asrfront.h +100 -0
  17. asrfront-0.1.0/csrc/kernels.c +47 -0
  18. asrfront-0.1.0/csrc/kernels.h +60 -0
  19. asrfront-0.1.0/csrc/kernels_impl.c +497 -0
  20. asrfront-0.1.0/csrc/melbank.c +186 -0
  21. asrfront-0.1.0/csrc/melbank.h +40 -0
  22. asrfront-0.1.0/csrc/sensevoice.c +237 -0
  23. asrfront-0.1.0/csrc/whisper.c +143 -0
  24. asrfront-0.1.0/docs/benchmark.md +180 -0
  25. asrfront-0.1.0/docs/sensevoice.md +286 -0
  26. asrfront-0.1.0/docs/whisper.md +222 -0
  27. asrfront-0.1.0/plan.md +314 -0
  28. asrfront-0.1.0/pyproject.toml +105 -0
  29. asrfront-0.1.0/src/asrfront/__init__.py +6 -0
  30. asrfront-0.1.0/src/asrfront/_audio.py +93 -0
  31. asrfront-0.1.0/src/asrfront/_ext.cpp +214 -0
  32. asrfront-0.1.0/src/asrfront/_ext.pyi +56 -0
  33. asrfront-0.1.0/src/asrfront/py.typed +0 -0
  34. asrfront-0.1.0/src/asrfront/sensevoice.py +192 -0
  35. asrfront-0.1.0/src/asrfront/whisper.py +114 -0
  36. asrfront-0.1.0/tests/c/test_api.c +87 -0
  37. asrfront-0.1.0/tests/c/test_fft.c +186 -0
  38. asrfront-0.1.0/tests/data/README.md +6 -0
  39. asrfront-0.1.0/tests/data/jfk.wav +0 -0
  40. asrfront-0.1.0/tests/data/whisper_mel_filters.npz +0 -0
  41. asrfront-0.1.0/tests/python/audio_signals.py +61 -0
  42. asrfront-0.1.0/tests/python/conftest.py +10 -0
  43. asrfront-0.1.0/tests/python/helpers.py +22 -0
  44. asrfront-0.1.0/tests/python/reference.py +314 -0
  45. asrfront-0.1.0/tests/python/test_api.py +218 -0
  46. asrfront-0.1.0/tests/python/test_kernels.py +46 -0
  47. asrfront-0.1.0/tests/python/test_reference.py +186 -0
  48. asrfront-0.1.0/tests/python/test_sensevoice.py +260 -0
  49. asrfront-0.1.0/tests/python/test_whisper.py +118 -0
  50. asrfront-0.1.0/uv.lock +3277 -0
@@ -0,0 +1,5 @@
1
+ BasedOnStyle: LLVM
2
+ IndentWidth: 4
3
+ ColumnLimit: 100
4
+ PointerAlignment: Right
5
+ AllowShortFunctionsOnASingleLine: Inline
@@ -0,0 +1,88 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [master, main]
6
+ pull_request:
7
+ workflow_dispatch:
8
+
9
+ concurrency:
10
+ group: ${{ github.workflow }}-${{ github.ref }}
11
+ cancel-in-progress: true
12
+
13
+ jobs:
14
+ lint:
15
+ runs-on: ubuntu-latest
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: astral-sh/setup-uv@v6
19
+ - name: Install tools (no native build needed)
20
+ run: uv sync --no-install-project
21
+ - run: uv run --no-sync ruff check .
22
+ - run: uv run --no-sync ruff format --check .
23
+ - run: uv run --no-sync mypy --strict src/asrfront
24
+ - name: clang-format
25
+ run: >-
26
+ uv run --no-sync clang-format --dry-run --Werror
27
+ csrc/*.c csrc/*.h csrc/include/*.h tests/c/*.c bench/*.c src/asrfront/_ext.cpp
28
+
29
+ test:
30
+ name: test (${{ matrix.os }}, py${{ matrix.python }})
31
+ runs-on: ${{ matrix.os }}
32
+ strategy:
33
+ fail-fast: false
34
+ matrix:
35
+ os: [ubuntu-latest, ubuntu-24.04-arm, windows-latest]
36
+ python: ["3.9", "3.13"]
37
+ env:
38
+ UV_PYTHON: ${{ matrix.python }}
39
+ steps:
40
+ - uses: actions/checkout@v4
41
+ - uses: astral-sh/setup-uv@v6
42
+ - name: Build and install (editable)
43
+ run: uv sync
44
+ - name: Python tests
45
+ run: uv run --no-sync pytest -q
46
+ - name: Python tests with the plain-C kernels (MSVC code path)
47
+ run: uv run --no-sync pytest -q tests/python/test_whisper.py tests/python/test_sensevoice.py
48
+ env:
49
+ ASRFRONT_KERNELS: scalar
50
+ - name: C tests
51
+ run: |
52
+ uv run --no-sync cmake -S . -B build/ctest -DASRFRONT_BUILD_TESTS=ON -DCMAKE_BUILD_TYPE=Release
53
+ uv run --no-sync cmake --build build/ctest --config Release
54
+ uv run --no-sync ctest --test-dir build/ctest -C Release --output-on-failure
55
+
56
+ reference:
57
+ # Compare against the original float32 libraries (torch, torchaudio).
58
+ runs-on: ubuntu-latest
59
+ steps:
60
+ - uses: actions/checkout@v4
61
+ - uses: astral-sh/setup-uv@v6
62
+ - run: uv sync --group ref
63
+ - run: uv run --no-sync pytest -q
64
+
65
+ sanitizers:
66
+ runs-on: ubuntu-latest
67
+ env:
68
+ SAN_FLAGS: -fsanitize=address,undefined -fno-sanitize-recover=all -fno-omit-frame-pointer
69
+ steps:
70
+ - uses: actions/checkout@v4
71
+ - uses: astral-sh/setup-uv@v6
72
+ - name: C tests under ASan/UBSan
73
+ run: |
74
+ uv sync --no-install-project
75
+ uv run --no-sync cmake -S . -B build/asan -G Ninja -DASRFRONT_BUILD_TESTS=ON \
76
+ -DCMAKE_BUILD_TYPE=RelWithDebInfo -DCMAKE_C_FLAGS="$SAN_FLAGS"
77
+ uv run --no-sync cmake --build build/asan
78
+ uv run --no-sync ctest --test-dir build/asan --output-on-failure
79
+ - name: Python tests under ASan/UBSan
80
+ run: |
81
+ uv pip install --no-build-isolation scikit-build-core nanobind
82
+ uv pip install --no-build-isolation -e . \
83
+ -C cmake.build-type=RelWithDebInfo -C build-dir=build/asan-py \
84
+ -C "cmake.define.CMAKE_C_FLAGS=$SAN_FLAGS" -C "cmake.define.CMAKE_CXX_FLAGS=$SAN_FLAGS" \
85
+ -C "cmake.define.CMAKE_SHARED_LINKER_FLAGS=-fsanitize=address,undefined"
86
+ LD_PRELOAD="$(gcc -print-file-name=libasan.so) $(gcc -print-file-name=libubsan.so)" \
87
+ ASAN_OPTIONS=detect_leaks=0 \
88
+ .venv/bin/python -m pytest -q
@@ -0,0 +1,89 @@
1
+ name: Wheels
2
+
3
+ # Release: bump the version (pyproject.toml + csrc/include/asrfront.h), commit, then
4
+ # git tag vX.Y.Z && git push origin vX.Y.Z
5
+ # builds the sdist and all wheels, tests them and uploads them to PyPI.
6
+
7
+ on:
8
+ workflow_dispatch:
9
+ inputs:
10
+ testpypi:
11
+ description: "Upload the built wheels and sdist to TestPyPI"
12
+ type: boolean
13
+ default: false
14
+ push:
15
+ tags: ["v*"]
16
+
17
+ jobs:
18
+ sdist:
19
+ runs-on: ubuntu-latest
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+ - name: Check that the tag matches the package version
23
+ if: startsWith(github.ref, 'refs/tags/v')
24
+ run: |
25
+ version=$(python3 -c "import tomllib; print(tomllib.load(open('pyproject.toml', 'rb'))['project']['version'])")
26
+ if [ "v$version" != "$GITHUB_REF_NAME" ]; then
27
+ echo "::error::tag $GITHUB_REF_NAME does not match pyproject.toml version $version"
28
+ exit 1
29
+ fi
30
+ - uses: astral-sh/setup-uv@v6
31
+ - run: uv build --sdist --out-dir dist
32
+ - uses: actions/upload-artifact@v4
33
+ with:
34
+ name: sdist
35
+ path: dist/*.tar.gz
36
+
37
+ wheels:
38
+ name: wheels (${{ matrix.os }})
39
+ runs-on: ${{ matrix.os }}
40
+ strategy:
41
+ fail-fast: false
42
+ matrix:
43
+ os: [ubuntu-latest, ubuntu-24.04-arm, windows-latest]
44
+ steps:
45
+ - uses: actions/checkout@v4
46
+ - uses: astral-sh/setup-uv@v6
47
+ - name: Build and test wheels (config in pyproject.toml [tool.cibuildwheel])
48
+ run: uvx --from "cibuildwheel>=3,<4" cibuildwheel --output-dir wheelhouse
49
+ - uses: actions/upload-artifact@v4
50
+ with:
51
+ name: wheels-${{ matrix.os }}
52
+ path: wheelhouse/*.whl
53
+
54
+ publish-testpypi:
55
+ # Rehearsal: Actions -> Wheels -> Run workflow with "testpypi" checked.
56
+ # Trusted publishing: register this repository / wheels.yml / environment "testpypi"
57
+ # as a (pending) publisher on test.pypi.org first.
58
+ if: github.event_name == 'workflow_dispatch' && inputs.testpypi
59
+ needs: [sdist, wheels]
60
+ runs-on: ubuntu-latest
61
+ environment: testpypi
62
+ permissions:
63
+ id-token: write
64
+ steps:
65
+ - uses: actions/download-artifact@v4
66
+ with:
67
+ path: dist
68
+ merge-multiple: true
69
+ - uses: pypa/gh-action-pypi-publish@release/v1
70
+ with:
71
+ repository-url: https://test.pypi.org/legacy/
72
+ skip-existing: true # re-running for the same version is not an error
73
+
74
+ publish:
75
+ # Pushing a v* tag uploads to PyPI (only if the tag matches the version and every wheel
76
+ # built and passed its tests). Trusted publishing: register this repository /
77
+ # wheels.yml / environment "pypi" as a (pending) publisher on pypi.org first.
78
+ if: github.event_name == 'push' && startsWith(github.ref, 'refs/tags/v')
79
+ needs: [sdist, wheels]
80
+ runs-on: ubuntu-latest
81
+ environment: pypi
82
+ permissions:
83
+ id-token: write
84
+ steps:
85
+ - uses: actions/download-artifact@v4
86
+ with:
87
+ path: dist
88
+ merge-multiple: true
89
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,10 @@
1
+ build/
2
+ dist/
3
+ *.egg-info/
4
+ __pycache__/
5
+ *.so
6
+ *.pyd
7
+ .venv/
8
+ .pytest_cache/
9
+ .ruff_cache/
10
+ wheelhouse/
@@ -0,0 +1 @@
1
+ 3.12
@@ -0,0 +1,113 @@
1
+ cmake_minimum_required(VERSION 3.18...3.30)
2
+ project(asrfront LANGUAGES C)
3
+
4
+ # scikit-build-core sets SKBUILD=2, which option() would not treat as ON.
5
+ if(SKBUILD)
6
+ set(_asrfront_python_default ON)
7
+ else()
8
+ set(_asrfront_python_default OFF)
9
+ endif()
10
+ option(ASRFRONT_BUILD_PYTHON "Build the nanobind Python extension" ${_asrfront_python_default})
11
+ option(ASRFRONT_BUILD_TESTS "Build C unit tests" OFF)
12
+ option(ASRFRONT_BUILD_BENCH "Build C benchmarks" OFF)
13
+
14
+ set(CMAKE_C_STANDARD 11)
15
+ set(CMAKE_C_STANDARD_REQUIRED ON)
16
+ set(CMAKE_C_EXTENSIONS OFF)
17
+ if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
18
+ set(CMAKE_BUILD_TYPE Release)
19
+ endif()
20
+
21
+ # ---------------------------------------------------------------------------
22
+ # C core: pure C11, links only against libm. No -ffast-math (reproducibility).
23
+ # ---------------------------------------------------------------------------
24
+ function(asrfront_c_options target)
25
+ target_include_directories(${target} PUBLIC csrc/include PRIVATE csrc)
26
+ set_target_properties(${target} PROPERTIES POSITION_INDEPENDENT_CODE ON)
27
+ if(MSVC)
28
+ target_compile_options(${target} PRIVATE /W4)
29
+ else()
30
+ target_compile_options(${target} PRIVATE -Wall -Wextra -Wpedantic -Wshadow)
31
+ endif()
32
+ endfunction()
33
+
34
+ # SIMD kernels: the same source compiled per ISA, selected at runtime (kernels.c).
35
+ # No FMA flags anywhere, so every variant gives bit-identical results.
36
+ add_library(asrfront_kernels_generic OBJECT csrc/kernels_impl.c)
37
+ target_compile_definitions(asrfront_kernels_generic PRIVATE AF_ISA=generic)
38
+ asrfront_c_options(asrfront_kernels_generic)
39
+ set(ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_generic>)
40
+
41
+ # Plain-C variant (no vector extensions), i.e. the MSVC code path; selectable with
42
+ # ASRFRONT_KERNELS=scalar so it is tested on every platform.
43
+ add_library(asrfront_kernels_scalar OBJECT csrc/kernels_impl.c)
44
+ target_compile_definitions(asrfront_kernels_scalar PRIVATE AF_ISA=scalar AF_NO_VECTOR_EXT=1)
45
+ asrfront_c_options(asrfront_kernels_scalar)
46
+ list(APPEND ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_scalar>)
47
+
48
+ if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64|x64)$")
49
+ add_library(asrfront_kernels_avx2 OBJECT csrc/kernels_impl.c)
50
+ target_compile_definitions(asrfront_kernels_avx2 PRIVATE AF_ISA=avx2)
51
+ target_compile_options(asrfront_kernels_avx2 PRIVATE $<IF:$<C_COMPILER_ID:MSVC>,/arch:AVX2,-mavx2>)
52
+ asrfront_c_options(asrfront_kernels_avx2)
53
+ list(APPEND ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_avx2>)
54
+ set(ASRFRONT_HAVE_AVX2 ON)
55
+ endif()
56
+
57
+ add_library(asrfront_core STATIC
58
+ csrc/common.c
59
+ csrc/fft.c
60
+ csrc/kernels.c
61
+ csrc/melbank.c
62
+ csrc/whisper.c
63
+ csrc/sensevoice.c
64
+ ${ASRFRONT_KERNEL_OBJECTS}
65
+ )
66
+ asrfront_c_options(asrfront_core)
67
+ if(ASRFRONT_HAVE_AVX2)
68
+ target_compile_definitions(asrfront_core PRIVATE AF_HAVE_AVX2=1)
69
+ endif()
70
+ if(UNIX)
71
+ target_link_libraries(asrfront_core PUBLIC m)
72
+ endif()
73
+
74
+ # ---------------------------------------------------------------------------
75
+ # Python extension
76
+ # ---------------------------------------------------------------------------
77
+ if(ASRFRONT_BUILD_PYTHON)
78
+ enable_language(CXX)
79
+ set(CMAKE_CXX_STANDARD 17)
80
+ set(CMAKE_CXX_STANDARD_REQUIRED ON)
81
+ # SKBUILD_SABI_COMPONENT is "Development.SABIModule" when building a stable-ABI wheel.
82
+ find_package(Python 3.9 REQUIRED COMPONENTS Interpreter Development.Module
83
+ ${SKBUILD_SABI_COMPONENT})
84
+ find_package(nanobind CONFIG REQUIRED)
85
+ nanobind_add_module(_ext STABLE_ABI NB_STATIC src/asrfront/_ext.cpp)
86
+ target_include_directories(_ext PRIVATE csrc) # internal headers (test hooks)
87
+ target_link_libraries(_ext PRIVATE asrfront_core)
88
+ install(TARGETS _ext LIBRARY DESTINATION asrfront)
89
+ endif()
90
+
91
+ # ---------------------------------------------------------------------------
92
+ # C unit tests
93
+ # ---------------------------------------------------------------------------
94
+ if(ASRFRONT_BUILD_TESTS)
95
+ enable_testing()
96
+ foreach(t test_api test_fft)
97
+ add_executable(${t} tests/c/${t}.c)
98
+ target_include_directories(${t} PRIVATE csrc) # internal headers
99
+ target_link_libraries(${t} PRIVATE asrfront_core)
100
+ add_test(NAME ${t} COMMAND ${t})
101
+ endforeach()
102
+ endif()
103
+
104
+ # ---------------------------------------------------------------------------
105
+ # C benchmarks
106
+ # ---------------------------------------------------------------------------
107
+ if(ASRFRONT_BUILD_BENCH)
108
+ foreach(b bench_fft bench_frontend)
109
+ add_executable(${b} bench/${b}.c)
110
+ target_include_directories(${b} PRIVATE csrc)
111
+ target_link_libraries(${b} PRIVATE asrfront_core)
112
+ endforeach()
113
+ endif()
asrfront-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sinjin Jeong
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,144 @@
1
+ Metadata-Version: 2.1
2
+ Name: asrfront
3
+ Version: 0.1.0
4
+ Summary: Super-fast, dependency-free C frontends (log-mel, kaldi fbank) for Whisper and SenseVoice ASR models
5
+ Keywords: asr,speech,whisper,sensevoice,fbank,mel-spectrogram,kaldi
6
+ Author: sjjeong94
7
+ License: MIT
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: Intended Audience :: Science/Research
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Operating System :: Microsoft :: Windows
13
+ Classifier: Operating System :: POSIX :: Linux
14
+ Classifier: Programming Language :: C
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Typing :: Typed
19
+ Project-URL: Homepage, https://github.com/sjjeong94/asrfront
20
+ Project-URL: Issues, https://github.com/sjjeong94/asrfront/issues
21
+ Requires-Python: >=3.9
22
+ Requires-Dist: numpy>=1.21
23
+ Description-Content-Type: text/markdown
24
+
25
+ # asrfront
26
+
27
+ Super-fast frontend (feature extraction) for ASR models such as **Whisper** and
28
+ **SenseVoiceSmall**.
29
+
30
+ - native C core with no dependencies (own FFT, SIMD kernels with runtime CPU dispatch)
31
+ - Python binding based on [nanobind](https://github.com/wjakob/nanobind), zero-copy NumPy output
32
+ - numerically matches the reference implementations (see [Accuracy](#accuracy))
33
+
34
+ ```python
35
+ import asrfront
36
+
37
+ mel = asrfront.whisper.log_mel(audio) # (80, 3000)
38
+ feat = asrfront.sensevoice.fbank(audio, cmvn="am.mvn") # (T', 560)
39
+ ```
40
+
41
+ ## Install
42
+
43
+ ```bash
44
+ pip install asrfront
45
+ ```
46
+
47
+ Wheels: Linux (x86_64, aarch64), Windows (x86_64); CPython 3.9+
48
+ (one `abi3` wheel covers 3.12 and later). Building from source needs a C11/C++17 compiler.
49
+
50
+ ## Usage
51
+
52
+ `audio` is a 16 kHz mono waveform: a float array in [-1, 1], int16 PCM, a list, or a CPU
53
+ torch tensor. Resampling and decoding are out of scope.
54
+
55
+ ### Whisper
56
+
57
+ ```python
58
+ from asrfront import whisper
59
+
60
+ mel = whisper.log_mel(audio) # (80, 3000): pad/trim to 30 s like whisper
61
+ mel = whisper.log_mel(audio, n_mels=128) # large-v3
62
+ mel = whisper.log_mel(audio, pad_or_trim=False) # (80, len(audio) // 160)
63
+
64
+ batch = whisper.log_mel_batch([a1, a2, a3]) # (3, 80, 3000), multi-threaded
65
+ ```
66
+
67
+ Equivalent to `whisper.log_mel_spectrogram(whisper.pad_or_trim(audio), n_mels)`.
68
+
69
+ ### SenseVoice
70
+
71
+ ```python
72
+ from asrfront import sensevoice
73
+
74
+ feat = sensevoice.fbank(audio, cmvn="path/to/am.mvn") # (ceil(T / 6), 560)
75
+ feats, lengths = sensevoice.fbank_batch([a1, a2], cmvn="path/to/am.mvn") # zero-padded
76
+ raw = sensevoice.fbank_raw(audio) # (T, 80) kaldi fbank only
77
+ shift, scale = sensevoice.load_cmvn("path/to/am.mvn")
78
+ ```
79
+
80
+ Equivalent to FunASR's `WavFrontend` (`torchaudio.compliance.kaldi.fbank` on the int16-scaled
81
+ waveform with a Hamming window, `dither=0`, then LFR `m=7, n=6` and CMVN from `am.mvn`).
82
+ `dither=` and `seed=` are available; the RNG is reseeded on every call so identical inputs
83
+ give identical outputs.
84
+
85
+ ### Threads
86
+
87
+ All native calls release the GIL. The `*_batch` functions spread items over a thread pool
88
+ (`n_threads=None` uses every available core); each Python thread uses its own native handle.
89
+
90
+ ## Performance
91
+
92
+ 30 s of audio, single thread, Intel i5-14400F (AVX2), best of N runs
93
+ (`uv run --group ref --group bench python bench/compare.py`):
94
+
95
+ | | asrfront | reference | speed-up |
96
+ |---|---|---|---|
97
+ | Whisper log-mel (80 × 3000) | 0.85–0.95 ms | openai-whisper `log_mel_spectrogram` (torch) 8.6–9.6 ms | ~10x |
98
+ | | | librosa STFT + mel (numpy) 6.6–7.0 ms | ~7.7x |
99
+ | SenseVoice fbank + LFR + CMVN | 1.3–1.4 ms | FunASR `WavFrontend` (torchaudio) 12–14 ms | ~9–10x |
100
+ | kaldi fbank only | 1.3 ms | `torchaudio.compliance.kaldi.fbank` 11–12 ms | ~9x |
101
+
102
+ Full results (thread scaling, input lengths, kernel variants, methodology):
103
+ [`docs/benchmark.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/benchmark.md).
104
+
105
+ Kernels process 8 frames at a time in SIMD lanes. On x86-64 an AVX2 build is selected at
106
+ runtime when available (`asrfront._ext.kernel_isa()`); `ASRFRONT_KERNELS=generic` forces the
107
+ baseline build. All variants give bit-identical results. Batch of 32 × 30 s: ~7.5 ms with
108
+ 16 threads.
109
+
110
+ ## Accuracy
111
+
112
+ Maximum absolute error against float64 reference implementations
113
+ ([`tests/python/reference.py`](https://github.com/sjjeong94/asrfront/blob/main/tests/python/reference.py)) over 16 test signals (silence,
114
+ tones, chirp, noise, clipping, DC offset, real speech, short and > 30 s inputs):
115
+
116
+ | | worst case | speech | target |
117
+ |---|---|---|---|
118
+ | Whisper log-mel (80 and 128 mels) | 4.0e-5 | 1.3e-5 | 1e-4 |
119
+ | kaldi fbank (log domain) | 7.7e-4 | 2.0e-4 | 1e-3 |
120
+ | SenseVoice features with SenseVoiceSmall's `am.mvn` | 1.1e-4 | | |
121
+
122
+ The whisper mel filters are bit-identical to whisper's `mel_filters.npz`. Against the
123
+ original float32 libraries: ≤ 1e-4 vs. whisper (torch) and ≤ 2e-3 vs. torchaudio's kaldi
124
+ fbank (torchaudio builds its mel banks in float32).
125
+
126
+ ## Development
127
+
128
+ ```bash
129
+ uv sync # create .venv and build the extension (needs a C compiler)
130
+ uv run pytest # tests; add --group ref for torch/torchaudio comparisons
131
+ uv run --group ref pytest
132
+ uv run cmake -S . -B build/ctest -G Ninja -DASRFRONT_BUILD_TESTS=ON -DASRFRONT_BUILD_BENCH=ON
133
+ uv run cmake --build build/ctest && uv run ctest --test-dir build/ctest
134
+ ```
135
+
136
+ The C API is in [`csrc/include/asrfront.h`](https://github.com/sjjeong94/asrfront/blob/main/csrc/include/asrfront.h). The preprocessing of each
137
+ model is documented step by step in [`docs/whisper.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/whisper.md) and
138
+ [`docs/sensevoice.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/sensevoice.md); see [`plan.md`](https://github.com/sjjeong94/asrfront/blob/main/plan.md) for design and
139
+ implementation notes.
140
+
141
+ ## License
142
+
143
+ MIT. Test data in `tests/data` comes from [openai/whisper](https://github.com/openai/whisper)
144
+ (MIT); see [`tests/data/README.md`](https://github.com/sjjeong94/asrfront/blob/main/tests/data/README.md).
@@ -0,0 +1,120 @@
1
+ # asrfront
2
+
3
+ Super-fast frontend (feature extraction) for ASR models such as **Whisper** and
4
+ **SenseVoiceSmall**.
5
+
6
+ - native C core with no dependencies (own FFT, SIMD kernels with runtime CPU dispatch)
7
+ - Python binding based on [nanobind](https://github.com/wjakob/nanobind), zero-copy NumPy output
8
+ - numerically matches the reference implementations (see [Accuracy](#accuracy))
9
+
10
+ ```python
11
+ import asrfront
12
+
13
+ mel = asrfront.whisper.log_mel(audio) # (80, 3000)
14
+ feat = asrfront.sensevoice.fbank(audio, cmvn="am.mvn") # (T', 560)
15
+ ```
16
+
17
+ ## Install
18
+
19
+ ```bash
20
+ pip install asrfront
21
+ ```
22
+
23
+ Wheels: Linux (x86_64, aarch64), Windows (x86_64); CPython 3.9+
24
+ (one `abi3` wheel covers 3.12 and later). Building from source needs a C11/C++17 compiler.
25
+
26
+ ## Usage
27
+
28
+ `audio` is a 16 kHz mono waveform: a float array in [-1, 1], int16 PCM, a list, or a CPU
29
+ torch tensor. Resampling and decoding are out of scope.
30
+
31
+ ### Whisper
32
+
33
+ ```python
34
+ from asrfront import whisper
35
+
36
+ mel = whisper.log_mel(audio) # (80, 3000): pad/trim to 30 s like whisper
37
+ mel = whisper.log_mel(audio, n_mels=128) # large-v3
38
+ mel = whisper.log_mel(audio, pad_or_trim=False) # (80, len(audio) // 160)
39
+
40
+ batch = whisper.log_mel_batch([a1, a2, a3]) # (3, 80, 3000), multi-threaded
41
+ ```
42
+
43
+ Equivalent to `whisper.log_mel_spectrogram(whisper.pad_or_trim(audio), n_mels)`.
44
+
45
+ ### SenseVoice
46
+
47
+ ```python
48
+ from asrfront import sensevoice
49
+
50
+ feat = sensevoice.fbank(audio, cmvn="path/to/am.mvn") # (ceil(T / 6), 560)
51
+ feats, lengths = sensevoice.fbank_batch([a1, a2], cmvn="path/to/am.mvn") # zero-padded
52
+ raw = sensevoice.fbank_raw(audio) # (T, 80) kaldi fbank only
53
+ shift, scale = sensevoice.load_cmvn("path/to/am.mvn")
54
+ ```
55
+
56
+ Equivalent to FunASR's `WavFrontend` (`torchaudio.compliance.kaldi.fbank` on the int16-scaled
57
+ waveform with a Hamming window, `dither=0`, then LFR `m=7, n=6` and CMVN from `am.mvn`).
58
+ `dither=` and `seed=` are available; the RNG is reseeded on every call so identical inputs
59
+ give identical outputs.
60
+
61
+ ### Threads
62
+
63
+ All native calls release the GIL. The `*_batch` functions spread items over a thread pool
64
+ (`n_threads=None` uses every available core); each Python thread uses its own native handle.
65
+
66
+ ## Performance
67
+
68
+ 30 s of audio, single thread, Intel i5-14400F (AVX2), best of N runs
69
+ (`uv run --group ref --group bench python bench/compare.py`):
70
+
71
+ | | asrfront | reference | speed-up |
72
+ |---|---|---|---|
73
+ | Whisper log-mel (80 × 3000) | 0.85–0.95 ms | openai-whisper `log_mel_spectrogram` (torch) 8.6–9.6 ms | ~10x |
74
+ | | | librosa STFT + mel (numpy) 6.6–7.0 ms | ~7.7x |
75
+ | SenseVoice fbank + LFR + CMVN | 1.3–1.4 ms | FunASR `WavFrontend` (torchaudio) 12–14 ms | ~9–10x |
76
+ | kaldi fbank only | 1.3 ms | `torchaudio.compliance.kaldi.fbank` 11–12 ms | ~9x |
77
+
78
+ Full results (thread scaling, input lengths, kernel variants, methodology):
79
+ [`docs/benchmark.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/benchmark.md).
80
+
81
+ Kernels process 8 frames at a time in SIMD lanes. On x86-64 an AVX2 build is selected at
82
+ runtime when available (`asrfront._ext.kernel_isa()`); `ASRFRONT_KERNELS=generic` forces the
83
+ baseline build. All variants give bit-identical results. Batch of 32 × 30 s: ~7.5 ms with
84
+ 16 threads.
85
+
86
+ ## Accuracy
87
+
88
+ Maximum absolute error against float64 reference implementations
89
+ ([`tests/python/reference.py`](https://github.com/sjjeong94/asrfront/blob/main/tests/python/reference.py)) over 16 test signals (silence,
90
+ tones, chirp, noise, clipping, DC offset, real speech, short and > 30 s inputs):
91
+
92
+ | | worst case | speech | target |
93
+ |---|---|---|---|
94
+ | Whisper log-mel (80 and 128 mels) | 4.0e-5 | 1.3e-5 | 1e-4 |
95
+ | kaldi fbank (log domain) | 7.7e-4 | 2.0e-4 | 1e-3 |
96
+ | SenseVoice features with SenseVoiceSmall's `am.mvn` | 1.1e-4 | | |
97
+
98
+ The whisper mel filters are bit-identical to whisper's `mel_filters.npz`. Against the
99
+ original float32 libraries: ≤ 1e-4 vs. whisper (torch) and ≤ 2e-3 vs. torchaudio's kaldi
100
+ fbank (torchaudio builds its mel banks in float32).
101
+
102
+ ## Development
103
+
104
+ ```bash
105
+ uv sync # create .venv and build the extension (needs a C compiler)
106
+ uv run pytest # tests; add --group ref for torch/torchaudio comparisons
107
+ uv run --group ref pytest
108
+ uv run cmake -S . -B build/ctest -G Ninja -DASRFRONT_BUILD_TESTS=ON -DASRFRONT_BUILD_BENCH=ON
109
+ uv run cmake --build build/ctest && uv run ctest --test-dir build/ctest
110
+ ```
111
+
112
+ The C API is in [`csrc/include/asrfront.h`](https://github.com/sjjeong94/asrfront/blob/main/csrc/include/asrfront.h). The preprocessing of each
113
+ model is documented step by step in [`docs/whisper.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/whisper.md) and
114
+ [`docs/sensevoice.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/sensevoice.md); see [`plan.md`](https://github.com/sjjeong94/asrfront/blob/main/plan.md) for design and
115
+ implementation notes.
116
+
117
+ ## License
118
+
119
+ MIT. Test data in `tests/data` comes from [openai/whisper](https://github.com/openai/whisper)
120
+ (MIT); see [`tests/data/README.md`](https://github.com/sjjeong94/asrfront/blob/main/tests/data/README.md).
@@ -0,0 +1,55 @@
1
+ /* Real FFT throughput: ns per transform for the frontend sizes. */
2
+ #include <stdio.h>
3
+ #include <stdlib.h>
4
+ #include <time.h>
5
+
6
+ #include "asrfront.h"
7
+ #include "fft.h"
8
+
9
+ static double now_ns(void) {
10
+ struct timespec ts;
11
+ timespec_get(&ts, TIME_UTC);
12
+ return (double)ts.tv_sec * 1e9 + (double)ts.tv_nsec;
13
+ }
14
+
15
+ static void bench(int n, int iters) {
16
+ af_rfft *p = NULL;
17
+ if (af_rfft_create(n, &p) != AF_OK) {
18
+ fprintf(stderr, "create failed for n=%d\n", n);
19
+ exit(1);
20
+ }
21
+ float *x = malloc(n * sizeof(float));
22
+ float *pw = malloc((n / 2 + 1) * sizeof(float));
23
+ for (int i = 0; i < n; i++)
24
+ x[i] = (float)((i * 7919) % 1000) / 1000.0f - 0.5f;
25
+
26
+ volatile float sink = 0.0f;
27
+ for (int i = 0; i < iters / 10; i++) { /* warm-up */
28
+ af_rfft_power(p, x, pw);
29
+ sink += pw[1];
30
+ }
31
+ double best = 1e30;
32
+ for (int rep = 0; rep < 5; rep++) {
33
+ double t0 = now_ns();
34
+ for (int i = 0; i < iters; i++) {
35
+ x[0] = (float)i; /* defeat hoisting */
36
+ af_rfft_power(p, x, pw);
37
+ sink += pw[1];
38
+ }
39
+ double dt = (now_ns() - t0) / iters;
40
+ best = dt < best ? dt : best;
41
+ }
42
+ printf("rfft+power n=%4d: %8.1f ns/frame (%6.3f ms per 3000 frames = 30 s)\n", n, best,
43
+ best * 3000 / 1e6);
44
+ (void)sink;
45
+ free(x);
46
+ free(pw);
47
+ af_rfft_destroy(p);
48
+ }
49
+
50
+ int main(int argc, char **argv) {
51
+ int iters = argc > 1 ? atoi(argv[1]) : 200000;
52
+ bench(400, iters);
53
+ bench(512, iters);
54
+ return 0;
55
+ }