audio-prep-pipeline 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audio_prep_pipeline-0.1.0/.github/workflows/ci.yml +45 -0
- audio_prep_pipeline-0.1.0/.gitignore +30 -0
- audio_prep_pipeline-0.1.0/.pre-commit-config.yaml +24 -0
- audio_prep_pipeline-0.1.0/LICENSE +21 -0
- audio_prep_pipeline-0.1.0/Makefile +31 -0
- audio_prep_pipeline-0.1.0/PKG-INFO +276 -0
- audio_prep_pipeline-0.1.0/README.md +237 -0
- audio_prep_pipeline-0.1.0/docs/api/chunker.md +43 -0
- audio_prep_pipeline-0.1.0/docs/api/config.md +23 -0
- audio_prep_pipeline-0.1.0/docs/api/converter.md +56 -0
- audio_prep_pipeline-0.1.0/docs/api/exceptions.md +24 -0
- audio_prep_pipeline-0.1.0/docs/api/manifest.md +39 -0
- audio_prep_pipeline-0.1.0/docs/api/validator.md +30 -0
- audio_prep_pipeline-0.1.0/docs/architecture.md +36 -0
- audio_prep_pipeline-0.1.0/docs/cli.md +83 -0
- audio_prep_pipeline-0.1.0/docs/configuration.md +62 -0
- audio_prep_pipeline-0.1.0/docs/development/contributing.md +34 -0
- audio_prep_pipeline-0.1.0/docs/development/roadmap.md +19 -0
- audio_prep_pipeline-0.1.0/docs/development/testing.md +46 -0
- audio_prep_pipeline-0.1.0/docs/getting-started.md +76 -0
- audio_prep_pipeline-0.1.0/docs/index.md +67 -0
- audio_prep_pipeline-0.1.0/docs/installation.md +77 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/chunking.md +56 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/conversion.md +51 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/discovery.md +44 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/manifest.md +49 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/overview.md +35 -0
- audio_prep_pipeline-0.1.0/docs/pipeline/validation.md +35 -0
- audio_prep_pipeline-0.1.0/docs/quickstart.md +59 -0
- audio_prep_pipeline-0.1.0/mkdocs.yml +56 -0
- audio_prep_pipeline-0.1.0/pyproject.toml +108 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/__init__.py +47 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/chunker.py +465 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/cli.py +204 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/config.py +40 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/converter.py +243 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/exceptions.py +32 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/manifest.py +121 -0
- audio_prep_pipeline-0.1.0/src/audio_prep/validator.py +120 -0
- audio_prep_pipeline-0.1.0/tests/__init__.py +0 -0
- audio_prep_pipeline-0.1.0/tests/conftest.py +95 -0
- audio_prep_pipeline-0.1.0/tests/test_chunker.py +418 -0
- audio_prep_pipeline-0.1.0/tests/test_cli.py +278 -0
- audio_prep_pipeline-0.1.0/tests/test_config.py +40 -0
- audio_prep_pipeline-0.1.0/tests/test_converter.py +321 -0
- audio_prep_pipeline-0.1.0/tests/test_manifest.py +117 -0
- audio_prep_pipeline-0.1.0/tests/test_validator.py +112 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
lint:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
- uses: actions/setup-python@v5
|
|
14
|
+
with:
|
|
15
|
+
python-version: "3.12"
|
|
16
|
+
- name: Install dev dependencies
|
|
17
|
+
run: pip install -e ".[dev]"
|
|
18
|
+
- name: Ruff lint
|
|
19
|
+
run: ruff check .
|
|
20
|
+
- name: Ruff format check
|
|
21
|
+
run: ruff format --check .
|
|
22
|
+
- name: mypy
|
|
23
|
+
run: mypy src
|
|
24
|
+
|
|
25
|
+
test:
|
|
26
|
+
runs-on: ubuntu-latest
|
|
27
|
+
strategy:
|
|
28
|
+
matrix:
|
|
29
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
30
|
+
steps:
|
|
31
|
+
- uses: actions/checkout@v4
|
|
32
|
+
- uses: actions/setup-python@v5
|
|
33
|
+
with:
|
|
34
|
+
python-version: ${{ matrix.python-version }}
|
|
35
|
+
- name: Install ffmpeg
|
|
36
|
+
run: sudo apt-get update && sudo apt-get install -y ffmpeg
|
|
37
|
+
- name: Install package + dev dependencies
|
|
38
|
+
run: pip install -e ".[dev]"
|
|
39
|
+
- name: Run tests with coverage
|
|
40
|
+
run: pytest --cov-report=xml
|
|
41
|
+
- name: Upload coverage report
|
|
42
|
+
uses: actions/upload-artifact@v4
|
|
43
|
+
with:
|
|
44
|
+
name: coverage-${{ matrix.python-version }}
|
|
45
|
+
path: coverage.xml
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*.egg-info/
|
|
4
|
+
.eggs/
|
|
5
|
+
build/
|
|
6
|
+
dist/
|
|
7
|
+
|
|
8
|
+
# mkdocs deploy
|
|
9
|
+
site/
|
|
10
|
+
|
|
11
|
+
.venv/
|
|
12
|
+
venv/
|
|
13
|
+
|
|
14
|
+
.pytest_cache/
|
|
15
|
+
.mypy_cache/
|
|
16
|
+
.ruff_cache/
|
|
17
|
+
.coverage
|
|
18
|
+
htmlcov/
|
|
19
|
+
|
|
20
|
+
# Local data / outputs -- never commit raw or converted audio
|
|
21
|
+
data/
|
|
22
|
+
*.wav
|
|
23
|
+
*.flac
|
|
24
|
+
*.mp3
|
|
25
|
+
manifest.jsonl
|
|
26
|
+
|
|
27
|
+
.DS_Store
|
|
28
|
+
.vscode/
|
|
29
|
+
.idea/
|
|
30
|
+
results/
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
repos:
|
|
2
|
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
3
|
+
rev: v0.6.9
|
|
4
|
+
hooks:
|
|
5
|
+
- id: ruff
|
|
6
|
+
args: [--fix]
|
|
7
|
+
- id: ruff-format
|
|
8
|
+
|
|
9
|
+
- repo: https://github.com/pre-commit/mirrors-mypy
|
|
10
|
+
rev: v1.11.2
|
|
11
|
+
hooks:
|
|
12
|
+
- id: mypy
|
|
13
|
+
additional_dependencies: [soundfile]
|
|
14
|
+
args: [--config-file=pyproject.toml]
|
|
15
|
+
|
|
16
|
+
- repo: https://github.com/pre-commit/pre-commit-hooks
|
|
17
|
+
rev: v4.6.0
|
|
18
|
+
hooks:
|
|
19
|
+
- id: trailing-whitespace
|
|
20
|
+
- id: end-of-file-fixer
|
|
21
|
+
- id: check-yaml
|
|
22
|
+
- id: check-toml
|
|
23
|
+
- id: check-added-large-files
|
|
24
|
+
args: [--maxkb=2048]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 nattkorat
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
.PHONY: install lint format typecheck test docs docs-serve check clean
|
|
2
|
+
|
|
3
|
+
install:
|
|
4
|
+
pip install -e ".[dev]"
|
|
5
|
+
pre-commit install
|
|
6
|
+
|
|
7
|
+
lint:
|
|
8
|
+
ruff check .
|
|
9
|
+
|
|
10
|
+
format:
|
|
11
|
+
ruff format .
|
|
12
|
+
ruff check --fix .
|
|
13
|
+
|
|
14
|
+
typecheck:
|
|
15
|
+
mypy src
|
|
16
|
+
|
|
17
|
+
test:
|
|
18
|
+
pytest
|
|
19
|
+
|
|
20
|
+
docs:
|
|
21
|
+
mkdocs build --strict
|
|
22
|
+
|
|
23
|
+
docs-serve:
|
|
24
|
+
mkdocs serve
|
|
25
|
+
|
|
26
|
+
# Everything CI runs, in one command -- run this before opening a PR.
|
|
27
|
+
check: lint typecheck test
|
|
28
|
+
|
|
29
|
+
clean:
|
|
30
|
+
rm -rf .pytest_cache .mypy_cache .ruff_cache .coverage htmlcov build dist *.egg-info
|
|
31
|
+
find . -type d -name __pycache__ -exec rm -rf {} +
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: audio-prep-pipeline
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Convert raw MP3 audio into pretraining-ready WAV/FLAC, with validation and dataset manifest output.
|
|
5
|
+
Project-URL: Homepage, https://github.com/nattkorat/audio-prep-pipeline
|
|
6
|
+
Project-URL: Repository, https://github.com/nattkorat/audio-prep-pipeline
|
|
7
|
+
Project-URL: Issues, https://github.com/nattkorat/audio-prep-pipeline/issues
|
|
8
|
+
Project-URL: Documentation, https://github.com/nattkorat/audio-prep-pipeline/tree/main/docs
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: audio,ffmpeg,preprocessing,pretraining,speech,vad
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Requires-Dist: numpy<2.3,>=1.24
|
|
25
|
+
Requires-Dist: soundfile>=0.12
|
|
26
|
+
Provides-Extra: chunking
|
|
27
|
+
Requires-Dist: silero-vad>=6.2.1; extra == 'chunking'
|
|
28
|
+
Requires-Dist: torch>=2.0.0; extra == 'chunking'
|
|
29
|
+
Requires-Dist: tqdm>=4.66.0; extra == 'chunking'
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest-cov>=5.0; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
35
|
+
Provides-Extra: docs
|
|
36
|
+
Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
|
|
37
|
+
Requires-Dist: mkdocs>=1.6; extra == 'docs'
|
|
38
|
+
Description-Content-Type: text/markdown
|
|
39
|
+
|
|
40
|
+
# audio-prep
|
|
41
|
+
|
|
42
|
+
[](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml)
|
|
43
|
+
[](https://pypi.org/project/audio-prep-pipeline/)
|
|
44
|
+

|
|
45
|
+

|
|
46
|
+

|
|
47
|
+

|
|
48
|
+

|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
Convert source audio into pretraining-ready WAV/FLAC, with validation and a
|
|
52
|
+
dataset manifest as the handoff artifact to the pretraining pipeline. Discovery
|
|
53
|
+
supports common FFmpeg-readable audio/video containers by default and can scan
|
|
54
|
+
all files with `--extensions all`.
|
|
55
|
+
|
|
56
|
+
Defaults: **16 kHz, mono** — the standard input spec for Wav2Vec2 / XLS-R
|
|
57
|
+
style self-supervised speech pretraining. Override via CLI flags if a
|
|
58
|
+
different downstream model needs something else.
|
|
59
|
+
|
|
60
|
+
## Repo layout
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
audio-prep-pipeline/
|
|
64
|
+
├── src/audio_prep/
|
|
65
|
+
│ ├── config.py # ConversionConfig: target spec + behavior knobs
|
|
66
|
+
│ ├── converter.py # find_audio_files, convert_file, convert_batch
|
|
67
|
+
│ ├── validator.py # probe_duration, validate_output
|
|
68
|
+
│ ├── chunker.py # (optional) ChunkConfig, chunk_file, chunk_batch -- Silero VAD speech chunking
|
|
69
|
+
│ ├── manifest.py # build_manifest, write_manifest (JSONL output)
|
|
70
|
+
│ ├── exceptions.py # ConversionError, ProbeError, ChunkingError
|
|
71
|
+
│ └── cli.py # `audio-prep convert ...` / `audio-prep chunk ...` entry point
|
|
72
|
+
├── tests/
|
|
73
|
+
│ ├── conftest.py # synthetic-audio fixtures (no binary files checked in)
|
|
74
|
+
│ ├── test_converter.py
|
|
75
|
+
│ ├── test_validator.py
|
|
76
|
+
│ ├── test_chunker.py # fake-detector tests, no real VAD model needed
|
|
77
|
+
│ ├── test_manifest.py
|
|
78
|
+
│ └── test_cli.py
|
|
79
|
+
├── .github/workflows/ci.yml # lint + typecheck + test matrix (3.10-3.12)
|
|
80
|
+
├── .pre-commit-config.yaml # ruff + mypy + basic hygiene hooks, runs on every commit
|
|
81
|
+
├── pyproject.toml # deps, ruff config, mypy config, pytest config
|
|
82
|
+
└── Makefile # `make check` runs everything CI runs, locally
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## Commands
|
|
86
|
+
|
|
87
|
+
`audio-prep` has two independent subcommands. Neither depends on the other
|
|
88
|
+
running first -- both scan `--input-dir` for supported source files directly.
|
|
89
|
+
|
|
90
|
+
- **`audio-prep convert`** - conversion only: `ffmpeg` resample/remix/re-encode
|
|
91
|
+
into the target WAV/FLAC spec, then validation, then an optional manifest.
|
|
92
|
+
Does not chunk.
|
|
93
|
+
- **`audio-prep chunk`** - chunking only: Silero VAD speech chunking straight
|
|
94
|
+
from source audio, with its own resample/format/manifest options. Does not convert
|
|
95
|
+
or validate.
|
|
96
|
+
|
|
97
|
+
### `convert` pipeline stages
|
|
98
|
+
|
|
99
|
+
1. **discovery** (`find_audio_files`) — recursively find supported source files
|
|
100
|
+
under an input directory.
|
|
101
|
+
2. **conversion** (`convert_file` / `convert_batch`) — shell out to `ffmpeg` to
|
|
102
|
+
resample/remix/re-encode into the target WAV/FLAC spec. Mirrors the input
|
|
103
|
+
directory's subfolder structure on output. Runs in a process pool since
|
|
104
|
+
each conversion is an independent subprocess call.
|
|
105
|
+
3. **validation** (`validate_output`) — re-opens each converted file with
|
|
106
|
+
`soundfile` and checks it actually matches the requested sample rate,
|
|
107
|
+
channel count, and minimum duration. This catches the case where ffmpeg
|
|
108
|
+
exits 0 but silently produced something degenerate.
|
|
109
|
+
4. **manifest** (`build_manifest` / `write_manifest`) — JSONL file, one row
|
|
110
|
+
per source file, recording `status` (`ok` / `conversion_failed` /
|
|
111
|
+
`validation_failed`), output path, duration, sample rate, and any error.
|
|
112
|
+
|
|
113
|
+
Conversion failures don't abort the batch — a bad file in a 50,000-file
|
|
114
|
+
corpus shows up as one `conversion_failed` row in the manifest, not a crashed
|
|
115
|
+
job three hours in.
|
|
116
|
+
|
|
117
|
+
### `chunk` pipeline stages
|
|
118
|
+
|
|
119
|
+
1. **discovery** (`find_audio_files`) — recursively find supported source files
|
|
120
|
+
under an input directory.
|
|
121
|
+
2. **chunking** (`chunk_file` / `chunk_batch`) — runs Silero VAD over each
|
|
122
|
+
file (decoding/resampling via ffmpeg) and splits it into speech-only
|
|
123
|
+
chunks bounded by a `[min, max]` duration window, so silence-heavy source
|
|
124
|
+
recordings don't waste pretraining compute. Needs the `chunking` extra
|
|
125
|
+
(`torch` + `silero-vad`, see Setup below).
|
|
126
|
+
3. **manifest** (`build_chunk_manifest` / `write_manifest`), optional — JSONL
|
|
127
|
+
file, one row per source file, recording `status` (`ok` /
|
|
128
|
+
`chunking_failed`), chunk count, and chunk paths.
|
|
129
|
+
|
|
130
|
+
Chunking failures (e.g. no speech detected) work the same way: `chunk_batch`
|
|
131
|
+
returns a `ChunkResult` per file instead of raising.
|
|
132
|
+
|
|
133
|
+
## Setup
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
# ffmpeg is a system dependency, not a pip package
|
|
137
|
+
sudo apt-get install ffmpeg # or: brew install ffmpeg
|
|
138
|
+
|
|
139
|
+
pip install audio-prep-pipeline
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Install directly from GitHub:
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
pip install "audio-prep-pipeline @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
For local development:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
make install # pip install -e ".[dev]" + pre-commit install
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
Chunking needs an extra install -- `make install` alone does not pull in
|
|
155
|
+
`torch`/`silero-vad`/`tqdm`, since `convert` doesn't need them:
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
pip install -e ".[chunking]"
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Or from GitHub:
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
pip install "audio-prep-pipeline[chunking] @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
The first `chunk` run needs either the `silero-vad` pip package installed, or
|
|
168
|
+
network access so `torch.hub` can download `snakers4/silero-vad` once (it's
|
|
169
|
+
then cached under `~/.cache/torch/hub`). If neither is available, pass
|
|
170
|
+
`--allow-energy-fallback` to use a lower-quality offline detector instead of
|
|
171
|
+
failing.
|
|
172
|
+
|
|
173
|
+
## Usage
|
|
174
|
+
|
|
175
|
+
### `audio-prep convert`
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
audio-prep convert \
|
|
179
|
+
--input-dir data/raw_mp3 \
|
|
180
|
+
--output-dir data/wav16k \
|
|
181
|
+
--format wav \
|
|
182
|
+
--sample-rate 16000 \
|
|
183
|
+
--workers 8 \
|
|
184
|
+
--manifest data/manifest.jsonl
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
| Flag | Default | Meaning |
|
|
188
|
+
|---|---|---|
|
|
189
|
+
| `--input-dir` | *(required)* | directory of source audio files |
|
|
190
|
+
| `--output-dir` | *(required)* | where converted output is written |
|
|
191
|
+
| `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
|
|
192
|
+
| `--format` | `wav` | output format (`wav` or `flac`) |
|
|
193
|
+
| `--sample-rate` | 16000 | target sample rate |
|
|
194
|
+
| `--channels` | 1 | target channel count |
|
|
195
|
+
| `--workers` | 4 | parallel conversion workers |
|
|
196
|
+
| `--min-duration-sec` | 0.5 | validation fails files shorter than this |
|
|
197
|
+
| `--overwrite` | off | re-convert even if output already exists and passes validation |
|
|
198
|
+
| `--normalize-loudness` | off | apply EBU R128 loudness normalization (-23 LUFS) |
|
|
199
|
+
| `--manifest` | none | path to write a JSONL manifest |
|
|
200
|
+
|
|
201
|
+
### `audio-prep chunk`
|
|
202
|
+
|
|
203
|
+
Independent of `convert` -- scans `--input-dir` for supported source files and runs VAD
|
|
204
|
+
chunking directly against them, decoding (and resampling, if `--sample-rate`
|
|
205
|
+
doesn't match the source) via ffmpeg:
|
|
206
|
+
|
|
207
|
+
```bash
|
|
208
|
+
audio-prep chunk \
|
|
209
|
+
--input-dir data/raw_mp3 \
|
|
210
|
+
--output-dir data/chunks \
|
|
211
|
+
--sample-rate 16000 \
|
|
212
|
+
--format flac \
|
|
213
|
+
--min-duration-sec 5 \
|
|
214
|
+
--max-duration-sec 20 \
|
|
215
|
+
--workers 4 \
|
|
216
|
+
--manifest data/chunk_manifest.jsonl
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
| Flag | Default | Meaning |
|
|
220
|
+
|---|---|---|
|
|
221
|
+
| `--input-dir` | *(required)* | directory of source audio files to scan and chunk |
|
|
222
|
+
| `--output-dir` | `<input-dir>/chunks` | where chunks are written |
|
|
223
|
+
| `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
|
|
224
|
+
| `--format` | `wav` | output chunk format (`wav` or `flac`) |
|
|
225
|
+
| `--sample-rate` | 16000 | resample (via ffmpeg) to this rate before chunking if the source doesn't already match it |
|
|
226
|
+
| `--min-duration-sec` | 5.0 | drop chunks shorter than this |
|
|
227
|
+
| `--max-duration-sec` | 20.0 | split longer speech into windows this size |
|
|
228
|
+
| `--workers` | 4 | parallel chunking workers |
|
|
229
|
+
| `--overwrite` | off | re-chunk even if valid output already exists |
|
|
230
|
+
| `--allow-energy-fallback` | off | fall back to a low-quality energy detector if Silero can't load, instead of raising |
|
|
231
|
+
| `--manifest` | none | path to write a JSONL chunk manifest (source file, status, chunk count/paths) |
|
|
232
|
+
|
|
233
|
+
`chunk_file` itself doesn't care about source extension -- it decodes
|
|
234
|
+
whatever path it's given -- so the Python API can also chunk an existing
|
|
235
|
+
WAV/FLAC corpus (e.g. `convert_batch` output) by passing `source_files`
|
|
236
|
+
explicitly instead of relying on `chunk_batch` discovery:
|
|
237
|
+
|
|
238
|
+
```python
|
|
239
|
+
from audio_prep import ConversionConfig, convert_batch, build_manifest, validate_output
|
|
240
|
+
from audio_prep import ChunkConfig, chunk_batch
|
|
241
|
+
|
|
242
|
+
config = ConversionConfig(output_format="wav", sample_rate=16_000, channels=1, num_workers=8)
|
|
243
|
+
results = convert_batch("data/raw_mp3", "data/wav16k", config)
|
|
244
|
+
validations = {r.output: validate_output(r.output, config) for r in results if r.success}
|
|
245
|
+
records = build_manifest(results, validations)
|
|
246
|
+
|
|
247
|
+
# source_files bypasses chunk_batch discovery, so this works
|
|
248
|
+
# directly against the already-converted WAV output above.
|
|
249
|
+
valid_outputs = [path for path, v in validations.items() if v.valid]
|
|
250
|
+
chunk_config = ChunkConfig(min_duration_sec=5, max_duration_sec=20, num_workers=4)
|
|
251
|
+
chunk_results = chunk_batch("data/wav16k", "data/wav16k/chunks", chunk_config, source_files=valid_outputs)
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
## Development workflow
|
|
255
|
+
|
|
256
|
+
```bash
|
|
257
|
+
make format # ruff format + autofix
|
|
258
|
+
make lint # ruff check
|
|
259
|
+
make typecheck # mypy --strict
|
|
260
|
+
make test # pytest with coverage
|
|
261
|
+
make check # all of the above -- run this before opening a PR
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
`pre-commit` (installed via `make install`) runs ruff + mypy + basic hygiene
|
|
265
|
+
checks automatically on every commit. CI (`.github/workflows/ci.yml`) re-runs
|
|
266
|
+
the same checks plus the full test matrix across Python 3.10–3.12 on every
|
|
267
|
+
push and PR.
|
|
268
|
+
|
|
269
|
+
## Extending this
|
|
270
|
+
|
|
271
|
+
Natural next additions, in roughly the order they'd come up:
|
|
272
|
+
|
|
273
|
+
- **Streaming manifest writes** for very large corpora, instead of holding
|
|
274
|
+
all `ConversionResult`s in memory before writing.
|
|
275
|
+
|
|
276
|
+
**Note**: This is the template, that you have to extend from. `Main` branch is protected so you have to create another branch to work on.
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
# audio-prep
|
|
2
|
+
|
|
3
|
+
[](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/audio-prep-pipeline/)
|
|
5
|
+

|
|
6
|
+

|
|
7
|
+

|
|
8
|
+

|
|
9
|
+

|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
Convert source audio into pretraining-ready WAV/FLAC, with validation and a
|
|
13
|
+
dataset manifest as the handoff artifact to the pretraining pipeline. Discovery
|
|
14
|
+
supports common FFmpeg-readable audio/video containers by default and can scan
|
|
15
|
+
all files with `--extensions all`.
|
|
16
|
+
|
|
17
|
+
Defaults: **16 kHz, mono** — the standard input spec for Wav2Vec2 / XLS-R
|
|
18
|
+
style self-supervised speech pretraining. Override via CLI flags if a
|
|
19
|
+
different downstream model needs something else.
|
|
20
|
+
|
|
21
|
+
## Repo layout
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
audio-prep-pipeline/
|
|
25
|
+
├── src/audio_prep/
|
|
26
|
+
│ ├── config.py # ConversionConfig: target spec + behavior knobs
|
|
27
|
+
│ ├── converter.py # find_audio_files, convert_file, convert_batch
|
|
28
|
+
│ ├── validator.py # probe_duration, validate_output
|
|
29
|
+
│ ├── chunker.py # (optional) ChunkConfig, chunk_file, chunk_batch -- Silero VAD speech chunking
|
|
30
|
+
│ ├── manifest.py # build_manifest, write_manifest (JSONL output)
|
|
31
|
+
│ ├── exceptions.py # ConversionError, ProbeError, ChunkingError
|
|
32
|
+
│ └── cli.py # `audio-prep convert ...` / `audio-prep chunk ...` entry point
|
|
33
|
+
├── tests/
|
|
34
|
+
│ ├── conftest.py # synthetic-audio fixtures (no binary files checked in)
|
|
35
|
+
│ ├── test_converter.py
|
|
36
|
+
│ ├── test_validator.py
|
|
37
|
+
│ ├── test_chunker.py # fake-detector tests, no real VAD model needed
|
|
38
|
+
│ ├── test_manifest.py
|
|
39
|
+
│ └── test_cli.py
|
|
40
|
+
├── .github/workflows/ci.yml # lint + typecheck + test matrix (3.10-3.12)
|
|
41
|
+
├── .pre-commit-config.yaml # ruff + mypy + basic hygiene hooks, runs on every commit
|
|
42
|
+
├── pyproject.toml # deps, ruff config, mypy config, pytest config
|
|
43
|
+
└── Makefile # `make check` runs everything CI runs, locally
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Commands
|
|
47
|
+
|
|
48
|
+
`audio-prep` has two independent subcommands. Neither depends on the other
|
|
49
|
+
running first -- both scan `--input-dir` for supported source files directly.
|
|
50
|
+
|
|
51
|
+
- **`audio-prep convert`** - conversion only: `ffmpeg` resample/remix/re-encode
|
|
52
|
+
into the target WAV/FLAC spec, then validation, then an optional manifest.
|
|
53
|
+
Does not chunk.
|
|
54
|
+
- **`audio-prep chunk`** - chunking only: Silero VAD speech chunking straight
|
|
55
|
+
from source audio, with its own resample/format/manifest options. Does not convert
|
|
56
|
+
or validate.
|
|
57
|
+
|
|
58
|
+
### `convert` pipeline stages
|
|
59
|
+
|
|
60
|
+
1. **discovery** (`find_audio_files`) — recursively find supported source files
|
|
61
|
+
under an input directory.
|
|
62
|
+
2. **conversion** (`convert_file` / `convert_batch`) — shell out to `ffmpeg` to
|
|
63
|
+
resample/remix/re-encode into the target WAV/FLAC spec. Mirrors the input
|
|
64
|
+
directory's subfolder structure on output. Runs in a process pool since
|
|
65
|
+
each conversion is an independent subprocess call.
|
|
66
|
+
3. **validation** (`validate_output`) — re-opens each converted file with
|
|
67
|
+
`soundfile` and checks it actually matches the requested sample rate,
|
|
68
|
+
channel count, and minimum duration. This catches the case where ffmpeg
|
|
69
|
+
exits 0 but silently produced something degenerate.
|
|
70
|
+
4. **manifest** (`build_manifest` / `write_manifest`) — JSONL file, one row
|
|
71
|
+
per source file, recording `status` (`ok` / `conversion_failed` /
|
|
72
|
+
`validation_failed`), output path, duration, sample rate, and any error.
|
|
73
|
+
|
|
74
|
+
Conversion failures don't abort the batch — a bad file in a 50,000-file
|
|
75
|
+
corpus shows up as one `conversion_failed` row in the manifest, not a crashed
|
|
76
|
+
job three hours in.
|
|
77
|
+
|
|
78
|
+
### `chunk` pipeline stages
|
|
79
|
+
|
|
80
|
+
1. **discovery** (`find_audio_files`) — recursively find supported source files
|
|
81
|
+
under an input directory.
|
|
82
|
+
2. **chunking** (`chunk_file` / `chunk_batch`) — runs Silero VAD over each
|
|
83
|
+
file (decoding/resampling via ffmpeg) and splits it into speech-only
|
|
84
|
+
chunks bounded by a `[min, max]` duration window, so silence-heavy source
|
|
85
|
+
recordings don't waste pretraining compute. Needs the `chunking` extra
|
|
86
|
+
(`torch` + `silero-vad`, see Setup below).
|
|
87
|
+
3. **manifest** (`build_chunk_manifest` / `write_manifest`), optional — JSONL
|
|
88
|
+
file, one row per source file, recording `status` (`ok` /
|
|
89
|
+
`chunking_failed`), chunk count, and chunk paths.
|
|
90
|
+
|
|
91
|
+
Chunking failures (e.g. no speech detected) work the same way: `chunk_batch`
|
|
92
|
+
returns a `ChunkResult` per file instead of raising.
|
|
93
|
+
|
|
94
|
+
## Setup
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
# ffmpeg is a system dependency, not a pip package
|
|
98
|
+
sudo apt-get install ffmpeg # or: brew install ffmpeg
|
|
99
|
+
|
|
100
|
+
pip install audio-prep-pipeline
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Install directly from GitHub:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
pip install "audio-prep-pipeline @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
For local development:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
make install # pip install -e ".[dev]" + pre-commit install
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Chunking needs an extra install -- `make install` alone does not pull in
|
|
116
|
+
`torch`/`silero-vad`/`tqdm`, since `convert` doesn't need them:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
pip install -e ".[chunking]"
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Or from GitHub:
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
pip install "audio-prep-pipeline[chunking] @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
The first `chunk` run needs either the `silero-vad` pip package installed, or
|
|
129
|
+
network access so `torch.hub` can download `snakers4/silero-vad` once (it's
|
|
130
|
+
then cached under `~/.cache/torch/hub`). If neither is available, pass
|
|
131
|
+
`--allow-energy-fallback` to use a lower-quality offline detector instead of
|
|
132
|
+
failing.
|
|
133
|
+
|
|
134
|
+
## Usage
|
|
135
|
+
|
|
136
|
+
### `audio-prep convert`
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
audio-prep convert \
|
|
140
|
+
--input-dir data/raw_mp3 \
|
|
141
|
+
--output-dir data/wav16k \
|
|
142
|
+
--format wav \
|
|
143
|
+
--sample-rate 16000 \
|
|
144
|
+
--workers 8 \
|
|
145
|
+
--manifest data/manifest.jsonl
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
| Flag | Default | Meaning |
|
|
149
|
+
|---|---|---|
|
|
150
|
+
| `--input-dir` | *(required)* | directory of source audio files |
|
|
151
|
+
| `--output-dir` | *(required)* | where converted output is written |
|
|
152
|
+
| `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
|
|
153
|
+
| `--format` | `wav` | output format (`wav` or `flac`) |
|
|
154
|
+
| `--sample-rate` | 16000 | target sample rate |
|
|
155
|
+
| `--channels` | 1 | target channel count |
|
|
156
|
+
| `--workers` | 4 | parallel conversion workers |
|
|
157
|
+
| `--min-duration-sec` | 0.5 | validation fails files shorter than this |
|
|
158
|
+
| `--overwrite` | off | re-convert even if output already exists and passes validation |
|
|
159
|
+
| `--normalize-loudness` | off | apply EBU R128 loudness normalization (-23 LUFS) |
|
|
160
|
+
| `--manifest` | none | path to write a JSONL manifest |
|
|
161
|
+
|
|
162
|
+
### `audio-prep chunk`
|
|
163
|
+
|
|
164
|
+
Independent of `convert` -- scans `--input-dir` for supported source files and runs VAD
|
|
165
|
+
chunking directly against them, decoding (and resampling, if `--sample-rate`
|
|
166
|
+
doesn't match the source) via ffmpeg:
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
audio-prep chunk \
|
|
170
|
+
--input-dir data/raw_mp3 \
|
|
171
|
+
--output-dir data/chunks \
|
|
172
|
+
--sample-rate 16000 \
|
|
173
|
+
--format flac \
|
|
174
|
+
--min-duration-sec 5 \
|
|
175
|
+
--max-duration-sec 20 \
|
|
176
|
+
--workers 4 \
|
|
177
|
+
--manifest data/chunk_manifest.jsonl
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
| Flag | Default | Meaning |
|
|
181
|
+
|---|---|---|
|
|
182
|
+
| `--input-dir` | *(required)* | directory of source audio files to scan and chunk |
|
|
183
|
+
| `--output-dir` | `<input-dir>/chunks` | where chunks are written |
|
|
184
|
+
| `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
|
|
185
|
+
| `--format` | `wav` | output chunk format (`wav` or `flac`) |
|
|
186
|
+
| `--sample-rate` | 16000 | resample (via ffmpeg) to this rate before chunking if the source doesn't already match it |
|
|
187
|
+
| `--min-duration-sec` | 5.0 | drop chunks shorter than this |
|
|
188
|
+
| `--max-duration-sec` | 20.0 | split longer speech into windows this size |
|
|
189
|
+
| `--workers` | 4 | parallel chunking workers |
|
|
190
|
+
| `--overwrite` | off | re-chunk even if valid output already exists |
|
|
191
|
+
| `--allow-energy-fallback` | off | fall back to a low-quality energy detector if Silero can't load, instead of raising |
|
|
192
|
+
| `--manifest` | none | path to write a JSONL chunk manifest (source file, status, chunk count/paths) |
|
|
193
|
+
|
|
194
|
+
`chunk_file` itself doesn't care about source extension -- it decodes
|
|
195
|
+
whatever path it's given -- so the Python API can also chunk an existing
|
|
196
|
+
WAV/FLAC corpus (e.g. `convert_batch` output) by passing `source_files`
|
|
197
|
+
explicitly instead of relying on `chunk_batch` discovery:
|
|
198
|
+
|
|
199
|
+
```python
|
|
200
|
+
from audio_prep import ConversionConfig, convert_batch, build_manifest, validate_output
|
|
201
|
+
from audio_prep import ChunkConfig, chunk_batch
|
|
202
|
+
|
|
203
|
+
config = ConversionConfig(output_format="wav", sample_rate=16_000, channels=1, num_workers=8)
|
|
204
|
+
results = convert_batch("data/raw_mp3", "data/wav16k", config)
|
|
205
|
+
validations = {r.output: validate_output(r.output, config) for r in results if r.success}
|
|
206
|
+
records = build_manifest(results, validations)
|
|
207
|
+
|
|
208
|
+
# source_files bypasses chunk_batch discovery, so this works
|
|
209
|
+
# directly against the already-converted WAV output above.
|
|
210
|
+
valid_outputs = [path for path, v in validations.items() if v.valid]
|
|
211
|
+
chunk_config = ChunkConfig(min_duration_sec=5, max_duration_sec=20, num_workers=4)
|
|
212
|
+
chunk_results = chunk_batch("data/wav16k", "data/wav16k/chunks", chunk_config, source_files=valid_outputs)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
## Development workflow
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
make format # ruff format + autofix
|
|
219
|
+
make lint # ruff check
|
|
220
|
+
make typecheck # mypy --strict
|
|
221
|
+
make test # pytest with coverage
|
|
222
|
+
make check # all of the above -- run this before opening a PR
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
`pre-commit` (installed via `make install`) runs ruff + mypy + basic hygiene
|
|
226
|
+
checks automatically on every commit. CI (`.github/workflows/ci.yml`) re-runs
|
|
227
|
+
the same checks plus the full test matrix across Python 3.10–3.12 on every
|
|
228
|
+
push and PR.
|
|
229
|
+
|
|
230
|
+
## Extending this
|
|
231
|
+
|
|
232
|
+
Natural next additions, in roughly the order they'd come up:
|
|
233
|
+
|
|
234
|
+
- **Streaming manifest writes** for very large corpora, instead of holding
|
|
235
|
+
all `ConversionResult`s in memory before writing.
|
|
236
|
+
|
|
237
|
+
**Note**: This is the template, that you have to extend from. `Main` branch is protected so you have to create another branch to work on.
|