asrfront 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- asrfront-0.1.0/.clang-format +5 -0
- asrfront-0.1.0/.github/workflows/ci.yml +88 -0
- asrfront-0.1.0/.github/workflows/wheels.yml +89 -0
- asrfront-0.1.0/.gitignore +10 -0
- asrfront-0.1.0/.python-version +1 -0
- asrfront-0.1.0/CMakeLists.txt +113 -0
- asrfront-0.1.0/LICENSE +21 -0
- asrfront-0.1.0/PKG-INFO +144 -0
- asrfront-0.1.0/README.md +120 -0
- asrfront-0.1.0/bench/bench_fft.c +55 -0
- asrfront-0.1.0/bench/bench_frontend.c +91 -0
- asrfront-0.1.0/bench/compare.py +289 -0
- asrfront-0.1.0/csrc/common.c +25 -0
- asrfront-0.1.0/csrc/fft.c +276 -0
- asrfront-0.1.0/csrc/fft.h +50 -0
- asrfront-0.1.0/csrc/include/asrfront.h +100 -0
- asrfront-0.1.0/csrc/kernels.c +47 -0
- asrfront-0.1.0/csrc/kernels.h +60 -0
- asrfront-0.1.0/csrc/kernels_impl.c +497 -0
- asrfront-0.1.0/csrc/melbank.c +186 -0
- asrfront-0.1.0/csrc/melbank.h +40 -0
- asrfront-0.1.0/csrc/sensevoice.c +237 -0
- asrfront-0.1.0/csrc/whisper.c +143 -0
- asrfront-0.1.0/docs/benchmark.md +180 -0
- asrfront-0.1.0/docs/sensevoice.md +286 -0
- asrfront-0.1.0/docs/whisper.md +222 -0
- asrfront-0.1.0/plan.md +314 -0
- asrfront-0.1.0/pyproject.toml +105 -0
- asrfront-0.1.0/src/asrfront/__init__.py +6 -0
- asrfront-0.1.0/src/asrfront/_audio.py +93 -0
- asrfront-0.1.0/src/asrfront/_ext.cpp +214 -0
- asrfront-0.1.0/src/asrfront/_ext.pyi +56 -0
- asrfront-0.1.0/src/asrfront/py.typed +0 -0
- asrfront-0.1.0/src/asrfront/sensevoice.py +192 -0
- asrfront-0.1.0/src/asrfront/whisper.py +114 -0
- asrfront-0.1.0/tests/c/test_api.c +87 -0
- asrfront-0.1.0/tests/c/test_fft.c +186 -0
- asrfront-0.1.0/tests/data/README.md +6 -0
- asrfront-0.1.0/tests/data/jfk.wav +0 -0
- asrfront-0.1.0/tests/data/whisper_mel_filters.npz +0 -0
- asrfront-0.1.0/tests/python/audio_signals.py +61 -0
- asrfront-0.1.0/tests/python/conftest.py +10 -0
- asrfront-0.1.0/tests/python/helpers.py +22 -0
- asrfront-0.1.0/tests/python/reference.py +314 -0
- asrfront-0.1.0/tests/python/test_api.py +218 -0
- asrfront-0.1.0/tests/python/test_kernels.py +46 -0
- asrfront-0.1.0/tests/python/test_reference.py +186 -0
- asrfront-0.1.0/tests/python/test_sensevoice.py +260 -0
- asrfront-0.1.0/tests/python/test_whisper.py +118 -0
- asrfront-0.1.0/uv.lock +3277 -0
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [master, main]
|
|
6
|
+
pull_request:
|
|
7
|
+
workflow_dispatch:
|
|
8
|
+
|
|
9
|
+
concurrency:
|
|
10
|
+
group: ${{ github.workflow }}-${{ github.ref }}
|
|
11
|
+
cancel-in-progress: true
|
|
12
|
+
|
|
13
|
+
jobs:
|
|
14
|
+
lint:
|
|
15
|
+
runs-on: ubuntu-latest
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
- uses: astral-sh/setup-uv@v6
|
|
19
|
+
- name: Install tools (no native build needed)
|
|
20
|
+
run: uv sync --no-install-project
|
|
21
|
+
- run: uv run --no-sync ruff check .
|
|
22
|
+
- run: uv run --no-sync ruff format --check .
|
|
23
|
+
- run: uv run --no-sync mypy --strict src/asrfront
|
|
24
|
+
- name: clang-format
|
|
25
|
+
run: >-
|
|
26
|
+
uv run --no-sync clang-format --dry-run --Werror
|
|
27
|
+
csrc/*.c csrc/*.h csrc/include/*.h tests/c/*.c bench/*.c src/asrfront/_ext.cpp
|
|
28
|
+
|
|
29
|
+
test:
|
|
30
|
+
name: test (${{ matrix.os }}, py${{ matrix.python }})
|
|
31
|
+
runs-on: ${{ matrix.os }}
|
|
32
|
+
strategy:
|
|
33
|
+
fail-fast: false
|
|
34
|
+
matrix:
|
|
35
|
+
os: [ubuntu-latest, ubuntu-24.04-arm, windows-latest]
|
|
36
|
+
python: ["3.9", "3.13"]
|
|
37
|
+
env:
|
|
38
|
+
UV_PYTHON: ${{ matrix.python }}
|
|
39
|
+
steps:
|
|
40
|
+
- uses: actions/checkout@v4
|
|
41
|
+
- uses: astral-sh/setup-uv@v6
|
|
42
|
+
- name: Build and install (editable)
|
|
43
|
+
run: uv sync
|
|
44
|
+
- name: Python tests
|
|
45
|
+
run: uv run --no-sync pytest -q
|
|
46
|
+
- name: Python tests with the plain-C kernels (MSVC code path)
|
|
47
|
+
run: uv run --no-sync pytest -q tests/python/test_whisper.py tests/python/test_sensevoice.py
|
|
48
|
+
env:
|
|
49
|
+
ASRFRONT_KERNELS: scalar
|
|
50
|
+
- name: C tests
|
|
51
|
+
run: |
|
|
52
|
+
uv run --no-sync cmake -S . -B build/ctest -DASRFRONT_BUILD_TESTS=ON -DCMAKE_BUILD_TYPE=Release
|
|
53
|
+
uv run --no-sync cmake --build build/ctest --config Release
|
|
54
|
+
uv run --no-sync ctest --test-dir build/ctest -C Release --output-on-failure
|
|
55
|
+
|
|
56
|
+
reference:
|
|
57
|
+
# Compare against the original float32 libraries (torch, torchaudio).
|
|
58
|
+
runs-on: ubuntu-latest
|
|
59
|
+
steps:
|
|
60
|
+
- uses: actions/checkout@v4
|
|
61
|
+
- uses: astral-sh/setup-uv@v6
|
|
62
|
+
- run: uv sync --group ref
|
|
63
|
+
- run: uv run --no-sync pytest -q
|
|
64
|
+
|
|
65
|
+
sanitizers:
|
|
66
|
+
runs-on: ubuntu-latest
|
|
67
|
+
env:
|
|
68
|
+
SAN_FLAGS: -fsanitize=address,undefined -fno-sanitize-recover=all -fno-omit-frame-pointer
|
|
69
|
+
steps:
|
|
70
|
+
- uses: actions/checkout@v4
|
|
71
|
+
- uses: astral-sh/setup-uv@v6
|
|
72
|
+
- name: C tests under ASan/UBSan
|
|
73
|
+
run: |
|
|
74
|
+
uv sync --no-install-project
|
|
75
|
+
uv run --no-sync cmake -S . -B build/asan -G Ninja -DASRFRONT_BUILD_TESTS=ON \
|
|
76
|
+
-DCMAKE_BUILD_TYPE=RelWithDebInfo -DCMAKE_C_FLAGS="$SAN_FLAGS"
|
|
77
|
+
uv run --no-sync cmake --build build/asan
|
|
78
|
+
uv run --no-sync ctest --test-dir build/asan --output-on-failure
|
|
79
|
+
- name: Python tests under ASan/UBSan
|
|
80
|
+
run: |
|
|
81
|
+
uv pip install --no-build-isolation scikit-build-core nanobind
|
|
82
|
+
uv pip install --no-build-isolation -e . \
|
|
83
|
+
-C cmake.build-type=RelWithDebInfo -C build-dir=build/asan-py \
|
|
84
|
+
-C "cmake.define.CMAKE_C_FLAGS=$SAN_FLAGS" -C "cmake.define.CMAKE_CXX_FLAGS=$SAN_FLAGS" \
|
|
85
|
+
-C "cmake.define.CMAKE_SHARED_LINKER_FLAGS=-fsanitize=address,undefined"
|
|
86
|
+
LD_PRELOAD="$(gcc -print-file-name=libasan.so) $(gcc -print-file-name=libubsan.so)" \
|
|
87
|
+
ASAN_OPTIONS=detect_leaks=0 \
|
|
88
|
+
.venv/bin/python -m pytest -q
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
name: Wheels
|
|
2
|
+
|
|
3
|
+
# Release: bump the version (pyproject.toml + csrc/include/asrfront.h), commit, then
|
|
4
|
+
# git tag vX.Y.Z && git push origin vX.Y.Z
|
|
5
|
+
# builds the sdist and all wheels, tests them and uploads them to PyPI.
|
|
6
|
+
|
|
7
|
+
on:
|
|
8
|
+
workflow_dispatch:
|
|
9
|
+
inputs:
|
|
10
|
+
testpypi:
|
|
11
|
+
description: "Upload the built wheels and sdist to TestPyPI"
|
|
12
|
+
type: boolean
|
|
13
|
+
default: false
|
|
14
|
+
push:
|
|
15
|
+
tags: ["v*"]
|
|
16
|
+
|
|
17
|
+
jobs:
|
|
18
|
+
sdist:
|
|
19
|
+
runs-on: ubuntu-latest
|
|
20
|
+
steps:
|
|
21
|
+
- uses: actions/checkout@v4
|
|
22
|
+
- name: Check that the tag matches the package version
|
|
23
|
+
if: startsWith(github.ref, 'refs/tags/v')
|
|
24
|
+
run: |
|
|
25
|
+
version=$(python3 -c "import tomllib; print(tomllib.load(open('pyproject.toml', 'rb'))['project']['version'])")
|
|
26
|
+
if [ "v$version" != "$GITHUB_REF_NAME" ]; then
|
|
27
|
+
echo "::error::tag $GITHUB_REF_NAME does not match pyproject.toml version $version"
|
|
28
|
+
exit 1
|
|
29
|
+
fi
|
|
30
|
+
- uses: astral-sh/setup-uv@v6
|
|
31
|
+
- run: uv build --sdist --out-dir dist
|
|
32
|
+
- uses: actions/upload-artifact@v4
|
|
33
|
+
with:
|
|
34
|
+
name: sdist
|
|
35
|
+
path: dist/*.tar.gz
|
|
36
|
+
|
|
37
|
+
wheels:
|
|
38
|
+
name: wheels (${{ matrix.os }})
|
|
39
|
+
runs-on: ${{ matrix.os }}
|
|
40
|
+
strategy:
|
|
41
|
+
fail-fast: false
|
|
42
|
+
matrix:
|
|
43
|
+
os: [ubuntu-latest, ubuntu-24.04-arm, windows-latest]
|
|
44
|
+
steps:
|
|
45
|
+
- uses: actions/checkout@v4
|
|
46
|
+
- uses: astral-sh/setup-uv@v6
|
|
47
|
+
- name: Build and test wheels (config in pyproject.toml [tool.cibuildwheel])
|
|
48
|
+
run: uvx --from "cibuildwheel>=3,<4" cibuildwheel --output-dir wheelhouse
|
|
49
|
+
- uses: actions/upload-artifact@v4
|
|
50
|
+
with:
|
|
51
|
+
name: wheels-${{ matrix.os }}
|
|
52
|
+
path: wheelhouse/*.whl
|
|
53
|
+
|
|
54
|
+
publish-testpypi:
|
|
55
|
+
# Rehearsal: Actions -> Wheels -> Run workflow with "testpypi" checked.
|
|
56
|
+
# Trusted publishing: register this repository / wheels.yml / environment "testpypi"
|
|
57
|
+
# as a (pending) publisher on test.pypi.org first.
|
|
58
|
+
if: github.event_name == 'workflow_dispatch' && inputs.testpypi
|
|
59
|
+
needs: [sdist, wheels]
|
|
60
|
+
runs-on: ubuntu-latest
|
|
61
|
+
environment: testpypi
|
|
62
|
+
permissions:
|
|
63
|
+
id-token: write
|
|
64
|
+
steps:
|
|
65
|
+
- uses: actions/download-artifact@v4
|
|
66
|
+
with:
|
|
67
|
+
path: dist
|
|
68
|
+
merge-multiple: true
|
|
69
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
70
|
+
with:
|
|
71
|
+
repository-url: https://test.pypi.org/legacy/
|
|
72
|
+
skip-existing: true # re-running for the same version is not an error
|
|
73
|
+
|
|
74
|
+
publish:
|
|
75
|
+
# Pushing a v* tag uploads to PyPI (only if the tag matches the version and every wheel
|
|
76
|
+
# built and passed its tests). Trusted publishing: register this repository /
|
|
77
|
+
# wheels.yml / environment "pypi" as a (pending) publisher on pypi.org first.
|
|
78
|
+
if: github.event_name == 'push' && startsWith(github.ref, 'refs/tags/v')
|
|
79
|
+
needs: [sdist, wheels]
|
|
80
|
+
runs-on: ubuntu-latest
|
|
81
|
+
environment: pypi
|
|
82
|
+
permissions:
|
|
83
|
+
id-token: write
|
|
84
|
+
steps:
|
|
85
|
+
- uses: actions/download-artifact@v4
|
|
86
|
+
with:
|
|
87
|
+
path: dist
|
|
88
|
+
merge-multiple: true
|
|
89
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.12
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
cmake_minimum_required(VERSION 3.18...3.30)
|
|
2
|
+
project(asrfront LANGUAGES C)
|
|
3
|
+
|
|
4
|
+
# scikit-build-core sets SKBUILD=2, which option() would not treat as ON.
|
|
5
|
+
if(SKBUILD)
|
|
6
|
+
set(_asrfront_python_default ON)
|
|
7
|
+
else()
|
|
8
|
+
set(_asrfront_python_default OFF)
|
|
9
|
+
endif()
|
|
10
|
+
option(ASRFRONT_BUILD_PYTHON "Build the nanobind Python extension" ${_asrfront_python_default})
|
|
11
|
+
option(ASRFRONT_BUILD_TESTS "Build C unit tests" OFF)
|
|
12
|
+
option(ASRFRONT_BUILD_BENCH "Build C benchmarks" OFF)
|
|
13
|
+
|
|
14
|
+
set(CMAKE_C_STANDARD 11)
|
|
15
|
+
set(CMAKE_C_STANDARD_REQUIRED ON)
|
|
16
|
+
set(CMAKE_C_EXTENSIONS OFF)
|
|
17
|
+
if(NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
|
|
18
|
+
set(CMAKE_BUILD_TYPE Release)
|
|
19
|
+
endif()
|
|
20
|
+
|
|
21
|
+
# ---------------------------------------------------------------------------
|
|
22
|
+
# C core: pure C11, links only against libm. No -ffast-math (reproducibility).
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
function(asrfront_c_options target)
|
|
25
|
+
target_include_directories(${target} PUBLIC csrc/include PRIVATE csrc)
|
|
26
|
+
set_target_properties(${target} PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
|
27
|
+
if(MSVC)
|
|
28
|
+
target_compile_options(${target} PRIVATE /W4)
|
|
29
|
+
else()
|
|
30
|
+
target_compile_options(${target} PRIVATE -Wall -Wextra -Wpedantic -Wshadow)
|
|
31
|
+
endif()
|
|
32
|
+
endfunction()
|
|
33
|
+
|
|
34
|
+
# SIMD kernels: the same source compiled per ISA, selected at runtime (kernels.c).
|
|
35
|
+
# No FMA flags anywhere, so every variant gives bit-identical results.
|
|
36
|
+
add_library(asrfront_kernels_generic OBJECT csrc/kernels_impl.c)
|
|
37
|
+
target_compile_definitions(asrfront_kernels_generic PRIVATE AF_ISA=generic)
|
|
38
|
+
asrfront_c_options(asrfront_kernels_generic)
|
|
39
|
+
set(ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_generic>)
|
|
40
|
+
|
|
41
|
+
# Plain-C variant (no vector extensions), i.e. the MSVC code path; selectable with
|
|
42
|
+
# ASRFRONT_KERNELS=scalar so it is tested on every platform.
|
|
43
|
+
add_library(asrfront_kernels_scalar OBJECT csrc/kernels_impl.c)
|
|
44
|
+
target_compile_definitions(asrfront_kernels_scalar PRIVATE AF_ISA=scalar AF_NO_VECTOR_EXT=1)
|
|
45
|
+
asrfront_c_options(asrfront_kernels_scalar)
|
|
46
|
+
list(APPEND ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_scalar>)
|
|
47
|
+
|
|
48
|
+
if(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64|x64)$")
|
|
49
|
+
add_library(asrfront_kernels_avx2 OBJECT csrc/kernels_impl.c)
|
|
50
|
+
target_compile_definitions(asrfront_kernels_avx2 PRIVATE AF_ISA=avx2)
|
|
51
|
+
target_compile_options(asrfront_kernels_avx2 PRIVATE $<IF:$<C_COMPILER_ID:MSVC>,/arch:AVX2,-mavx2>)
|
|
52
|
+
asrfront_c_options(asrfront_kernels_avx2)
|
|
53
|
+
list(APPEND ASRFRONT_KERNEL_OBJECTS $<TARGET_OBJECTS:asrfront_kernels_avx2>)
|
|
54
|
+
set(ASRFRONT_HAVE_AVX2 ON)
|
|
55
|
+
endif()
|
|
56
|
+
|
|
57
|
+
add_library(asrfront_core STATIC
|
|
58
|
+
csrc/common.c
|
|
59
|
+
csrc/fft.c
|
|
60
|
+
csrc/kernels.c
|
|
61
|
+
csrc/melbank.c
|
|
62
|
+
csrc/whisper.c
|
|
63
|
+
csrc/sensevoice.c
|
|
64
|
+
${ASRFRONT_KERNEL_OBJECTS}
|
|
65
|
+
)
|
|
66
|
+
asrfront_c_options(asrfront_core)
|
|
67
|
+
if(ASRFRONT_HAVE_AVX2)
|
|
68
|
+
target_compile_definitions(asrfront_core PRIVATE AF_HAVE_AVX2=1)
|
|
69
|
+
endif()
|
|
70
|
+
if(UNIX)
|
|
71
|
+
target_link_libraries(asrfront_core PUBLIC m)
|
|
72
|
+
endif()
|
|
73
|
+
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
# Python extension
|
|
76
|
+
# ---------------------------------------------------------------------------
|
|
77
|
+
if(ASRFRONT_BUILD_PYTHON)
|
|
78
|
+
enable_language(CXX)
|
|
79
|
+
set(CMAKE_CXX_STANDARD 17)
|
|
80
|
+
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
|
81
|
+
# SKBUILD_SABI_COMPONENT is "Development.SABIModule" when building a stable-ABI wheel.
|
|
82
|
+
find_package(Python 3.9 REQUIRED COMPONENTS Interpreter Development.Module
|
|
83
|
+
${SKBUILD_SABI_COMPONENT})
|
|
84
|
+
find_package(nanobind CONFIG REQUIRED)
|
|
85
|
+
nanobind_add_module(_ext STABLE_ABI NB_STATIC src/asrfront/_ext.cpp)
|
|
86
|
+
target_include_directories(_ext PRIVATE csrc) # internal headers (test hooks)
|
|
87
|
+
target_link_libraries(_ext PRIVATE asrfront_core)
|
|
88
|
+
install(TARGETS _ext LIBRARY DESTINATION asrfront)
|
|
89
|
+
endif()
|
|
90
|
+
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
# C unit tests
|
|
93
|
+
# ---------------------------------------------------------------------------
|
|
94
|
+
if(ASRFRONT_BUILD_TESTS)
|
|
95
|
+
enable_testing()
|
|
96
|
+
foreach(t test_api test_fft)
|
|
97
|
+
add_executable(${t} tests/c/${t}.c)
|
|
98
|
+
target_include_directories(${t} PRIVATE csrc) # internal headers
|
|
99
|
+
target_link_libraries(${t} PRIVATE asrfront_core)
|
|
100
|
+
add_test(NAME ${t} COMMAND ${t})
|
|
101
|
+
endforeach()
|
|
102
|
+
endif()
|
|
103
|
+
|
|
104
|
+
# ---------------------------------------------------------------------------
|
|
105
|
+
# C benchmarks
|
|
106
|
+
# ---------------------------------------------------------------------------
|
|
107
|
+
if(ASRFRONT_BUILD_BENCH)
|
|
108
|
+
foreach(b bench_fft bench_frontend)
|
|
109
|
+
add_executable(${b} bench/${b}.c)
|
|
110
|
+
target_include_directories(${b} PRIVATE csrc)
|
|
111
|
+
target_link_libraries(${b} PRIVATE asrfront_core)
|
|
112
|
+
endforeach()
|
|
113
|
+
endif()
|
asrfront-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sinjin Jeong
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
asrfront-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: asrfront
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Super-fast, dependency-free C frontends (log-mel, kaldi fbank) for Whisper and SenseVoice ASR models
|
|
5
|
+
Keywords: asr,speech,whisper,sensevoice,fbank,mel-spectrogram,kaldi
|
|
6
|
+
Author: sjjeong94
|
|
7
|
+
License: MIT
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Intended Audience :: Science/Research
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
13
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
14
|
+
Classifier: Programming Language :: C
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Project-URL: Homepage, https://github.com/sjjeong94/asrfront
|
|
20
|
+
Project-URL: Issues, https://github.com/sjjeong94/asrfront/issues
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: numpy>=1.21
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# asrfront
|
|
26
|
+
|
|
27
|
+
Super-fast frontend (feature extraction) for ASR models such as **Whisper** and
|
|
28
|
+
**SenseVoiceSmall**.
|
|
29
|
+
|
|
30
|
+
- native C core with no dependencies (own FFT, SIMD kernels with runtime CPU dispatch)
|
|
31
|
+
- Python binding based on [nanobind](https://github.com/wjakob/nanobind), zero-copy NumPy output
|
|
32
|
+
- numerically matches the reference implementations (see [Accuracy](#accuracy))
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
import asrfront
|
|
36
|
+
|
|
37
|
+
mel = asrfront.whisper.log_mel(audio) # (80, 3000)
|
|
38
|
+
feat = asrfront.sensevoice.fbank(audio, cmvn="am.mvn") # (T', 560)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install asrfront
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Wheels: Linux (x86_64, aarch64), Windows (x86_64); CPython 3.9+
|
|
48
|
+
(one `abi3` wheel covers 3.12 and later). Building from source needs a C11/C++17 compiler.
|
|
49
|
+
|
|
50
|
+
## Usage
|
|
51
|
+
|
|
52
|
+
`audio` is a 16 kHz mono waveform: a float array in [-1, 1], int16 PCM, a list, or a CPU
|
|
53
|
+
torch tensor. Resampling and decoding are out of scope.
|
|
54
|
+
|
|
55
|
+
### Whisper
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
from asrfront import whisper
|
|
59
|
+
|
|
60
|
+
mel = whisper.log_mel(audio) # (80, 3000): pad/trim to 30 s like whisper
|
|
61
|
+
mel = whisper.log_mel(audio, n_mels=128) # large-v3
|
|
62
|
+
mel = whisper.log_mel(audio, pad_or_trim=False) # (80, len(audio) // 160)
|
|
63
|
+
|
|
64
|
+
batch = whisper.log_mel_batch([a1, a2, a3]) # (3, 80, 3000), multi-threaded
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Equivalent to `whisper.log_mel_spectrogram(whisper.pad_or_trim(audio), n_mels)`.
|
|
68
|
+
|
|
69
|
+
### SenseVoice
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from asrfront import sensevoice
|
|
73
|
+
|
|
74
|
+
feat = sensevoice.fbank(audio, cmvn="path/to/am.mvn") # (ceil(T / 6), 560)
|
|
75
|
+
feats, lengths = sensevoice.fbank_batch([a1, a2], cmvn="path/to/am.mvn") # zero-padded
|
|
76
|
+
raw = sensevoice.fbank_raw(audio) # (T, 80) kaldi fbank only
|
|
77
|
+
shift, scale = sensevoice.load_cmvn("path/to/am.mvn")
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Equivalent to FunASR's `WavFrontend` (`torchaudio.compliance.kaldi.fbank` on the int16-scaled
|
|
81
|
+
waveform with a Hamming window, `dither=0`, then LFR `m=7, n=6` and CMVN from `am.mvn`).
|
|
82
|
+
`dither=` and `seed=` are available; the RNG is reseeded on every call so identical inputs
|
|
83
|
+
give identical outputs.
|
|
84
|
+
|
|
85
|
+
### Threads
|
|
86
|
+
|
|
87
|
+
All native calls release the GIL. The `*_batch` functions spread items over a thread pool
|
|
88
|
+
(`n_threads=None` uses every available core); each Python thread uses its own native handle.
|
|
89
|
+
|
|
90
|
+
## Performance
|
|
91
|
+
|
|
92
|
+
30 s of audio, single thread, Intel i5-14400F (AVX2), best of N runs
|
|
93
|
+
(`uv run --group ref --group bench python bench/compare.py`):
|
|
94
|
+
|
|
95
|
+
| | asrfront | reference | speed-up |
|
|
96
|
+
|---|---|---|---|
|
|
97
|
+
| Whisper log-mel (80 × 3000) | 0.85–0.95 ms | openai-whisper `log_mel_spectrogram` (torch) 8.6–9.6 ms | ~10x |
|
|
98
|
+
| | | librosa STFT + mel (numpy) 6.6–7.0 ms | ~7.7x |
|
|
99
|
+
| SenseVoice fbank + LFR + CMVN | 1.3–1.4 ms | FunASR `WavFrontend` (torchaudio) 12–14 ms | ~9–10x |
|
|
100
|
+
| kaldi fbank only | 1.3 ms | `torchaudio.compliance.kaldi.fbank` 11–12 ms | ~9x |
|
|
101
|
+
|
|
102
|
+
Full results (thread scaling, input lengths, kernel variants, methodology):
|
|
103
|
+
[`docs/benchmark.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/benchmark.md).
|
|
104
|
+
|
|
105
|
+
Kernels process 8 frames at a time in SIMD lanes. On x86-64 an AVX2 build is selected at
|
|
106
|
+
runtime when available (`asrfront._ext.kernel_isa()`); `ASRFRONT_KERNELS=generic` forces the
|
|
107
|
+
baseline build. All variants give bit-identical results. Batch of 32 × 30 s: ~7.5 ms with
|
|
108
|
+
16 threads.
|
|
109
|
+
|
|
110
|
+
## Accuracy
|
|
111
|
+
|
|
112
|
+
Maximum absolute error against float64 reference implementations
|
|
113
|
+
([`tests/python/reference.py`](https://github.com/sjjeong94/asrfront/blob/main/tests/python/reference.py)) over 16 test signals (silence,
|
|
114
|
+
tones, chirp, noise, clipping, DC offset, real speech, short and > 30 s inputs):
|
|
115
|
+
|
|
116
|
+
| | worst case | speech | target |
|
|
117
|
+
|---|---|---|---|
|
|
118
|
+
| Whisper log-mel (80 and 128 mels) | 4.0e-5 | 1.3e-5 | 1e-4 |
|
|
119
|
+
| kaldi fbank (log domain) | 7.7e-4 | 2.0e-4 | 1e-3 |
|
|
120
|
+
| SenseVoice features with SenseVoiceSmall's `am.mvn` | 1.1e-4 | | |
|
|
121
|
+
|
|
122
|
+
The whisper mel filters are bit-identical to whisper's `mel_filters.npz`. Against the
|
|
123
|
+
original float32 libraries: ≤ 1e-4 vs. whisper (torch) and ≤ 2e-3 vs. torchaudio's kaldi
|
|
124
|
+
fbank (torchaudio builds its mel banks in float32).
|
|
125
|
+
|
|
126
|
+
## Development
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
uv sync # create .venv and build the extension (needs a C compiler)
|
|
130
|
+
uv run pytest # tests; add --group ref for torch/torchaudio comparisons
|
|
131
|
+
uv run --group ref pytest
|
|
132
|
+
uv run cmake -S . -B build/ctest -G Ninja -DASRFRONT_BUILD_TESTS=ON -DASRFRONT_BUILD_BENCH=ON
|
|
133
|
+
uv run cmake --build build/ctest && uv run ctest --test-dir build/ctest
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The C API is in [`csrc/include/asrfront.h`](https://github.com/sjjeong94/asrfront/blob/main/csrc/include/asrfront.h). The preprocessing of each
|
|
137
|
+
model is documented step by step in [`docs/whisper.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/whisper.md) and
|
|
138
|
+
[`docs/sensevoice.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/sensevoice.md); see [`plan.md`](https://github.com/sjjeong94/asrfront/blob/main/plan.md) for design and
|
|
139
|
+
implementation notes.
|
|
140
|
+
|
|
141
|
+
## License
|
|
142
|
+
|
|
143
|
+
MIT. Test data in `tests/data` comes from [openai/whisper](https://github.com/openai/whisper)
|
|
144
|
+
(MIT); see [`tests/data/README.md`](https://github.com/sjjeong94/asrfront/blob/main/tests/data/README.md).
|
asrfront-0.1.0/README.md
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
# asrfront
|
|
2
|
+
|
|
3
|
+
Super-fast frontend (feature extraction) for ASR models such as **Whisper** and
|
|
4
|
+
**SenseVoiceSmall**.
|
|
5
|
+
|
|
6
|
+
- native C core with no dependencies (own FFT, SIMD kernels with runtime CPU dispatch)
|
|
7
|
+
- Python binding based on [nanobind](https://github.com/wjakob/nanobind), zero-copy NumPy output
|
|
8
|
+
- numerically matches the reference implementations (see [Accuracy](#accuracy))
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
import asrfront
|
|
12
|
+
|
|
13
|
+
mel = asrfront.whisper.log_mel(audio) # (80, 3000)
|
|
14
|
+
feat = asrfront.sensevoice.fbank(audio, cmvn="am.mvn") # (T', 560)
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Install
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install asrfront
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Wheels: Linux (x86_64, aarch64), Windows (x86_64); CPython 3.9+
|
|
24
|
+
(one `abi3` wheel covers 3.12 and later). Building from source needs a C11/C++17 compiler.
|
|
25
|
+
|
|
26
|
+
## Usage
|
|
27
|
+
|
|
28
|
+
`audio` is a 16 kHz mono waveform: a float array in [-1, 1], int16 PCM, a list, or a CPU
|
|
29
|
+
torch tensor. Resampling and decoding are out of scope.
|
|
30
|
+
|
|
31
|
+
### Whisper
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from asrfront import whisper
|
|
35
|
+
|
|
36
|
+
mel = whisper.log_mel(audio) # (80, 3000): pad/trim to 30 s like whisper
|
|
37
|
+
mel = whisper.log_mel(audio, n_mels=128) # large-v3
|
|
38
|
+
mel = whisper.log_mel(audio, pad_or_trim=False) # (80, len(audio) // 160)
|
|
39
|
+
|
|
40
|
+
batch = whisper.log_mel_batch([a1, a2, a3]) # (3, 80, 3000), multi-threaded
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Equivalent to `whisper.log_mel_spectrogram(whisper.pad_or_trim(audio), n_mels)`.
|
|
44
|
+
|
|
45
|
+
### SenseVoice
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from asrfront import sensevoice
|
|
49
|
+
|
|
50
|
+
feat = sensevoice.fbank(audio, cmvn="path/to/am.mvn") # (ceil(T / 6), 560)
|
|
51
|
+
feats, lengths = sensevoice.fbank_batch([a1, a2], cmvn="path/to/am.mvn") # zero-padded
|
|
52
|
+
raw = sensevoice.fbank_raw(audio) # (T, 80) kaldi fbank only
|
|
53
|
+
shift, scale = sensevoice.load_cmvn("path/to/am.mvn")
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Equivalent to FunASR's `WavFrontend` (`torchaudio.compliance.kaldi.fbank` on the int16-scaled
|
|
57
|
+
waveform with a Hamming window, `dither=0`, then LFR `m=7, n=6` and CMVN from `am.mvn`).
|
|
58
|
+
`dither=` and `seed=` are available; the RNG is reseeded on every call so identical inputs
|
|
59
|
+
give identical outputs.
|
|
60
|
+
|
|
61
|
+
### Threads
|
|
62
|
+
|
|
63
|
+
All native calls release the GIL. The `*_batch` functions spread items over a thread pool
|
|
64
|
+
(`n_threads=None` uses every available core); each Python thread uses its own native handle.
|
|
65
|
+
|
|
66
|
+
## Performance
|
|
67
|
+
|
|
68
|
+
30 s of audio, single thread, Intel i5-14400F (AVX2), best of N runs
|
|
69
|
+
(`uv run --group ref --group bench python bench/compare.py`):
|
|
70
|
+
|
|
71
|
+
| | asrfront | reference | speed-up |
|
|
72
|
+
|---|---|---|---|
|
|
73
|
+
| Whisper log-mel (80 × 3000) | 0.85–0.95 ms | openai-whisper `log_mel_spectrogram` (torch) 8.6–9.6 ms | ~10x |
|
|
74
|
+
| | | librosa STFT + mel (numpy) 6.6–7.0 ms | ~7.7x |
|
|
75
|
+
| SenseVoice fbank + LFR + CMVN | 1.3–1.4 ms | FunASR `WavFrontend` (torchaudio) 12–14 ms | ~9–10x |
|
|
76
|
+
| kaldi fbank only | 1.3 ms | `torchaudio.compliance.kaldi.fbank` 11–12 ms | ~9x |
|
|
77
|
+
|
|
78
|
+
Full results (thread scaling, input lengths, kernel variants, methodology):
|
|
79
|
+
[`docs/benchmark.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/benchmark.md).
|
|
80
|
+
|
|
81
|
+
Kernels process 8 frames at a time in SIMD lanes. On x86-64 an AVX2 build is selected at
|
|
82
|
+
runtime when available (`asrfront._ext.kernel_isa()`); `ASRFRONT_KERNELS=generic` forces the
|
|
83
|
+
baseline build. All variants give bit-identical results. Batch of 32 × 30 s: ~7.5 ms with
|
|
84
|
+
16 threads.
|
|
85
|
+
|
|
86
|
+
## Accuracy
|
|
87
|
+
|
|
88
|
+
Maximum absolute error against float64 reference implementations
|
|
89
|
+
([`tests/python/reference.py`](https://github.com/sjjeong94/asrfront/blob/main/tests/python/reference.py)) over 16 test signals (silence,
|
|
90
|
+
tones, chirp, noise, clipping, DC offset, real speech, short and > 30 s inputs):
|
|
91
|
+
|
|
92
|
+
| | worst case | speech | target |
|
|
93
|
+
|---|---|---|---|
|
|
94
|
+
| Whisper log-mel (80 and 128 mels) | 4.0e-5 | 1.3e-5 | 1e-4 |
|
|
95
|
+
| kaldi fbank (log domain) | 7.7e-4 | 2.0e-4 | 1e-3 |
|
|
96
|
+
| SenseVoice features with SenseVoiceSmall's `am.mvn` | 1.1e-4 | | |
|
|
97
|
+
|
|
98
|
+
The whisper mel filters are bit-identical to whisper's `mel_filters.npz`. Against the
|
|
99
|
+
original float32 libraries: ≤ 1e-4 vs. whisper (torch) and ≤ 2e-3 vs. torchaudio's kaldi
|
|
100
|
+
fbank (torchaudio builds its mel banks in float32).
|
|
101
|
+
|
|
102
|
+
## Development
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
uv sync # create .venv and build the extension (needs a C compiler)
|
|
106
|
+
uv run pytest # tests; add --group ref for torch/torchaudio comparisons
|
|
107
|
+
uv run --group ref pytest
|
|
108
|
+
uv run cmake -S . -B build/ctest -G Ninja -DASRFRONT_BUILD_TESTS=ON -DASRFRONT_BUILD_BENCH=ON
|
|
109
|
+
uv run cmake --build build/ctest && uv run ctest --test-dir build/ctest
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
The C API is in [`csrc/include/asrfront.h`](https://github.com/sjjeong94/asrfront/blob/main/csrc/include/asrfront.h). The preprocessing of each
|
|
113
|
+
model is documented step by step in [`docs/whisper.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/whisper.md) and
|
|
114
|
+
[`docs/sensevoice.md`](https://github.com/sjjeong94/asrfront/blob/main/docs/sensevoice.md); see [`plan.md`](https://github.com/sjjeong94/asrfront/blob/main/plan.md) for design and
|
|
115
|
+
implementation notes.
|
|
116
|
+
|
|
117
|
+
## License
|
|
118
|
+
|
|
119
|
+
MIT. Test data in `tests/data` comes from [openai/whisper](https://github.com/openai/whisper)
|
|
120
|
+
(MIT); see [`tests/data/README.md`](https://github.com/sjjeong94/asrfront/blob/main/tests/data/README.md).
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/* Real FFT throughput: ns per transform for the frontend sizes. */
|
|
2
|
+
#include <stdio.h>
|
|
3
|
+
#include <stdlib.h>
|
|
4
|
+
#include <time.h>
|
|
5
|
+
|
|
6
|
+
#include "asrfront.h"
|
|
7
|
+
#include "fft.h"
|
|
8
|
+
|
|
9
|
+
static double now_ns(void) {
|
|
10
|
+
struct timespec ts;
|
|
11
|
+
timespec_get(&ts, TIME_UTC);
|
|
12
|
+
return (double)ts.tv_sec * 1e9 + (double)ts.tv_nsec;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
static void bench(int n, int iters) {
|
|
16
|
+
af_rfft *p = NULL;
|
|
17
|
+
if (af_rfft_create(n, &p) != AF_OK) {
|
|
18
|
+
fprintf(stderr, "create failed for n=%d\n", n);
|
|
19
|
+
exit(1);
|
|
20
|
+
}
|
|
21
|
+
float *x = malloc(n * sizeof(float));
|
|
22
|
+
float *pw = malloc((n / 2 + 1) * sizeof(float));
|
|
23
|
+
for (int i = 0; i < n; i++)
|
|
24
|
+
x[i] = (float)((i * 7919) % 1000) / 1000.0f - 0.5f;
|
|
25
|
+
|
|
26
|
+
volatile float sink = 0.0f;
|
|
27
|
+
for (int i = 0; i < iters / 10; i++) { /* warm-up */
|
|
28
|
+
af_rfft_power(p, x, pw);
|
|
29
|
+
sink += pw[1];
|
|
30
|
+
}
|
|
31
|
+
double best = 1e30;
|
|
32
|
+
for (int rep = 0; rep < 5; rep++) {
|
|
33
|
+
double t0 = now_ns();
|
|
34
|
+
for (int i = 0; i < iters; i++) {
|
|
35
|
+
x[0] = (float)i; /* defeat hoisting */
|
|
36
|
+
af_rfft_power(p, x, pw);
|
|
37
|
+
sink += pw[1];
|
|
38
|
+
}
|
|
39
|
+
double dt = (now_ns() - t0) / iters;
|
|
40
|
+
best = dt < best ? dt : best;
|
|
41
|
+
}
|
|
42
|
+
printf("rfft+power n=%4d: %8.1f ns/frame (%6.3f ms per 3000 frames = 30 s)\n", n, best,
|
|
43
|
+
best * 3000 / 1e6);
|
|
44
|
+
(void)sink;
|
|
45
|
+
free(x);
|
|
46
|
+
free(pw);
|
|
47
|
+
af_rfft_destroy(p);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
int main(int argc, char **argv) {
|
|
51
|
+
int iters = argc > 1 ? atoi(argv[1]) : 200000;
|
|
52
|
+
bench(400, iters);
|
|
53
|
+
bench(512, iters);
|
|
54
|
+
return 0;
|
|
55
|
+
}
|