audio-prep-pipeline 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. audio_prep_pipeline-0.1.0/.github/workflows/ci.yml +45 -0
  2. audio_prep_pipeline-0.1.0/.gitignore +30 -0
  3. audio_prep_pipeline-0.1.0/.pre-commit-config.yaml +24 -0
  4. audio_prep_pipeline-0.1.0/LICENSE +21 -0
  5. audio_prep_pipeline-0.1.0/Makefile +31 -0
  6. audio_prep_pipeline-0.1.0/PKG-INFO +276 -0
  7. audio_prep_pipeline-0.1.0/README.md +237 -0
  8. audio_prep_pipeline-0.1.0/docs/api/chunker.md +43 -0
  9. audio_prep_pipeline-0.1.0/docs/api/config.md +23 -0
  10. audio_prep_pipeline-0.1.0/docs/api/converter.md +56 -0
  11. audio_prep_pipeline-0.1.0/docs/api/exceptions.md +24 -0
  12. audio_prep_pipeline-0.1.0/docs/api/manifest.md +39 -0
  13. audio_prep_pipeline-0.1.0/docs/api/validator.md +30 -0
  14. audio_prep_pipeline-0.1.0/docs/architecture.md +36 -0
  15. audio_prep_pipeline-0.1.0/docs/cli.md +83 -0
  16. audio_prep_pipeline-0.1.0/docs/configuration.md +62 -0
  17. audio_prep_pipeline-0.1.0/docs/development/contributing.md +34 -0
  18. audio_prep_pipeline-0.1.0/docs/development/roadmap.md +19 -0
  19. audio_prep_pipeline-0.1.0/docs/development/testing.md +46 -0
  20. audio_prep_pipeline-0.1.0/docs/getting-started.md +76 -0
  21. audio_prep_pipeline-0.1.0/docs/index.md +67 -0
  22. audio_prep_pipeline-0.1.0/docs/installation.md +77 -0
  23. audio_prep_pipeline-0.1.0/docs/pipeline/chunking.md +56 -0
  24. audio_prep_pipeline-0.1.0/docs/pipeline/conversion.md +51 -0
  25. audio_prep_pipeline-0.1.0/docs/pipeline/discovery.md +44 -0
  26. audio_prep_pipeline-0.1.0/docs/pipeline/manifest.md +49 -0
  27. audio_prep_pipeline-0.1.0/docs/pipeline/overview.md +35 -0
  28. audio_prep_pipeline-0.1.0/docs/pipeline/validation.md +35 -0
  29. audio_prep_pipeline-0.1.0/docs/quickstart.md +59 -0
  30. audio_prep_pipeline-0.1.0/mkdocs.yml +56 -0
  31. audio_prep_pipeline-0.1.0/pyproject.toml +108 -0
  32. audio_prep_pipeline-0.1.0/src/audio_prep/__init__.py +47 -0
  33. audio_prep_pipeline-0.1.0/src/audio_prep/chunker.py +465 -0
  34. audio_prep_pipeline-0.1.0/src/audio_prep/cli.py +204 -0
  35. audio_prep_pipeline-0.1.0/src/audio_prep/config.py +40 -0
  36. audio_prep_pipeline-0.1.0/src/audio_prep/converter.py +243 -0
  37. audio_prep_pipeline-0.1.0/src/audio_prep/exceptions.py +32 -0
  38. audio_prep_pipeline-0.1.0/src/audio_prep/manifest.py +121 -0
  39. audio_prep_pipeline-0.1.0/src/audio_prep/validator.py +120 -0
  40. audio_prep_pipeline-0.1.0/tests/__init__.py +0 -0
  41. audio_prep_pipeline-0.1.0/tests/conftest.py +95 -0
  42. audio_prep_pipeline-0.1.0/tests/test_chunker.py +418 -0
  43. audio_prep_pipeline-0.1.0/tests/test_cli.py +278 -0
  44. audio_prep_pipeline-0.1.0/tests/test_config.py +40 -0
  45. audio_prep_pipeline-0.1.0/tests/test_converter.py +321 -0
  46. audio_prep_pipeline-0.1.0/tests/test_manifest.py +117 -0
  47. audio_prep_pipeline-0.1.0/tests/test_validator.py +112 -0
@@ -0,0 +1,45 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ lint:
10
+ runs-on: ubuntu-latest
11
+ steps:
12
+ - uses: actions/checkout@v4
13
+ - uses: actions/setup-python@v5
14
+ with:
15
+ python-version: "3.12"
16
+ - name: Install dev dependencies
17
+ run: pip install -e ".[dev]"
18
+ - name: Ruff lint
19
+ run: ruff check .
20
+ - name: Ruff format check
21
+ run: ruff format --check .
22
+ - name: mypy
23
+ run: mypy src
24
+
25
+ test:
26
+ runs-on: ubuntu-latest
27
+ strategy:
28
+ matrix:
29
+ python-version: ["3.10", "3.11", "3.12"]
30
+ steps:
31
+ - uses: actions/checkout@v4
32
+ - uses: actions/setup-python@v5
33
+ with:
34
+ python-version: ${{ matrix.python-version }}
35
+ - name: Install ffmpeg
36
+ run: sudo apt-get update && sudo apt-get install -y ffmpeg
37
+ - name: Install package + dev dependencies
38
+ run: pip install -e ".[dev]"
39
+ - name: Run tests with coverage
40
+ run: pytest --cov-report=xml
41
+ - name: Upload coverage report
42
+ uses: actions/upload-artifact@v4
43
+ with:
44
+ name: coverage-${{ matrix.python-version }}
45
+ path: coverage.xml
@@ -0,0 +1,30 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .eggs/
5
+ build/
6
+ dist/
7
+
8
+ # mkdocs deploy
9
+ site/
10
+
11
+ .venv/
12
+ venv/
13
+
14
+ .pytest_cache/
15
+ .mypy_cache/
16
+ .ruff_cache/
17
+ .coverage
18
+ htmlcov/
19
+
20
+ # Local data / outputs -- never commit raw or converted audio
21
+ data/
22
+ *.wav
23
+ *.flac
24
+ *.mp3
25
+ manifest.jsonl
26
+
27
+ .DS_Store
28
+ .vscode/
29
+ .idea/
30
+ results/
@@ -0,0 +1,24 @@
1
+ repos:
2
+ - repo: https://github.com/astral-sh/ruff-pre-commit
3
+ rev: v0.6.9
4
+ hooks:
5
+ - id: ruff
6
+ args: [--fix]
7
+ - id: ruff-format
8
+
9
+ - repo: https://github.com/pre-commit/mirrors-mypy
10
+ rev: v1.11.2
11
+ hooks:
12
+ - id: mypy
13
+ additional_dependencies: [soundfile]
14
+ args: [--config-file=pyproject.toml]
15
+
16
+ - repo: https://github.com/pre-commit/pre-commit-hooks
17
+ rev: v4.6.0
18
+ hooks:
19
+ - id: trailing-whitespace
20
+ - id: end-of-file-fixer
21
+ - id: check-yaml
22
+ - id: check-toml
23
+ - id: check-added-large-files
24
+ args: [--maxkb=2048]
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 nattkorat
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,31 @@
1
+ .PHONY: install lint format typecheck test docs docs-serve check clean
2
+
3
+ install:
4
+ pip install -e ".[dev]"
5
+ pre-commit install
6
+
7
+ lint:
8
+ ruff check .
9
+
10
+ format:
11
+ ruff format .
12
+ ruff check --fix .
13
+
14
+ typecheck:
15
+ mypy src
16
+
17
+ test:
18
+ pytest
19
+
20
+ docs:
21
+ mkdocs build --strict
22
+
23
+ docs-serve:
24
+ mkdocs serve
25
+
26
+ # Everything CI runs, in one command -- run this before opening a PR.
27
+ check: lint typecheck test
28
+
29
+ clean:
30
+ rm -rf .pytest_cache .mypy_cache .ruff_cache .coverage htmlcov build dist *.egg-info
31
+ find . -type d -name __pycache__ -exec rm -rf {} +
@@ -0,0 +1,276 @@
1
+ Metadata-Version: 2.4
2
+ Name: audio-prep-pipeline
3
+ Version: 0.1.0
4
+ Summary: Convert raw MP3 audio into pretraining-ready WAV/FLAC, with validation and dataset manifest output.
5
+ Project-URL: Homepage, https://github.com/nattkorat/audio-prep-pipeline
6
+ Project-URL: Repository, https://github.com/nattkorat/audio-prep-pipeline
7
+ Project-URL: Issues, https://github.com/nattkorat/audio-prep-pipeline/issues
8
+ Project-URL: Documentation, https://github.com/nattkorat/audio-prep-pipeline/tree/main/docs
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: audio,ffmpeg,preprocessing,pretraining,speech,vad
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Multimedia :: Sound/Audio
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Requires-Python: >=3.10
24
+ Requires-Dist: numpy<2.3,>=1.24
25
+ Requires-Dist: soundfile>=0.12
26
+ Provides-Extra: chunking
27
+ Requires-Dist: silero-vad>=6.2.1; extra == 'chunking'
28
+ Requires-Dist: torch>=2.0.0; extra == 'chunking'
29
+ Requires-Dist: tqdm>=4.66.0; extra == 'chunking'
30
+ Provides-Extra: dev
31
+ Requires-Dist: mypy>=1.10; extra == 'dev'
32
+ Requires-Dist: pytest-cov>=5.0; extra == 'dev'
33
+ Requires-Dist: pytest>=8.0; extra == 'dev'
34
+ Requires-Dist: ruff>=0.6; extra == 'dev'
35
+ Provides-Extra: docs
36
+ Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
37
+ Requires-Dist: mkdocs>=1.6; extra == 'docs'
38
+ Description-Content-Type: text/markdown
39
+
40
+ # audio-prep
41
+
42
+ [![CI](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml/badge.svg)](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml)
43
+ [![PyPI](https://img.shields.io/pypi/v/audio-prep-pipeline.svg)](https://pypi.org/project/audio-prep-pipeline/)
44
+ ![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue)
45
+ ![License: MIT](https://img.shields.io/badge/license-MIT-green)
46
+ ![Code style: Ruff](https://img.shields.io/badge/code%20style-ruff-46a2f1)
47
+ ![Type checked: mypy](https://img.shields.io/badge/type%20checked-mypy-blue)
48
+ ![Requires FFmpeg](https://img.shields.io/badge/requires-FFmpeg-007808)
49
+
50
+
51
+ Convert source audio into pretraining-ready WAV/FLAC, with validation and a
52
+ dataset manifest as the handoff artifact to the pretraining pipeline. Discovery
53
+ supports common FFmpeg-readable audio/video containers by default and can scan
54
+ all files with `--extensions all`.
55
+
56
+ Defaults: **16 kHz, mono** — the standard input spec for Wav2Vec2 / XLS-R
57
+ style self-supervised speech pretraining. Override via CLI flags if a
58
+ different downstream model needs something else.
59
+
60
+ ## Repo layout
61
+
62
+ ```
63
+ audio-prep-pipeline/
64
+ ├── src/audio_prep/
65
+ │ ├── config.py # ConversionConfig: target spec + behavior knobs
66
+ │ ├── converter.py # find_audio_files, convert_file, convert_batch
67
+ │ ├── validator.py # probe_duration, validate_output
68
+ │ ├── chunker.py # (optional) ChunkConfig, chunk_file, chunk_batch -- Silero VAD speech chunking
69
+ │ ├── manifest.py # build_manifest, write_manifest (JSONL output)
70
+ │ ├── exceptions.py # ConversionError, ProbeError, ChunkingError
71
+ │ └── cli.py # `audio-prep convert ...` / `audio-prep chunk ...` entry point
72
+ ├── tests/
73
+ │ ├── conftest.py # synthetic-audio fixtures (no binary files checked in)
74
+ │ ├── test_converter.py
75
+ │ ├── test_validator.py
76
+ │ ├── test_chunker.py # fake-detector tests, no real VAD model needed
77
+ │ ├── test_manifest.py
78
+ │ └── test_cli.py
79
+ ├── .github/workflows/ci.yml # lint + typecheck + test matrix (3.10-3.12)
80
+ ├── .pre-commit-config.yaml # ruff + mypy + basic hygiene hooks, runs on every commit
81
+ ├── pyproject.toml # deps, ruff config, mypy config, pytest config
82
+ └── Makefile # `make check` runs everything CI runs, locally
83
+ ```
84
+
85
+ ## Commands
86
+
87
+ `audio-prep` has two independent subcommands. Neither depends on the other
88
+ running first -- both scan `--input-dir` for supported source files directly.
89
+
90
+ - **`audio-prep convert`** - conversion only: `ffmpeg` resample/remix/re-encode
91
+ into the target WAV/FLAC spec, then validation, then an optional manifest.
92
+ Does not chunk.
93
+ - **`audio-prep chunk`** - chunking only: Silero VAD speech chunking straight
94
+ from source audio, with its own resample/format/manifest options. Does not convert
95
+ or validate.
96
+
97
+ ### `convert` pipeline stages
98
+
99
+ 1. **discovery** (`find_audio_files`) — recursively find supported source files
100
+ under an input directory.
101
+ 2. **conversion** (`convert_file` / `convert_batch`) — shell out to `ffmpeg` to
102
+ resample/remix/re-encode into the target WAV/FLAC spec. Mirrors the input
103
+ directory's subfolder structure on output. Runs in a process pool since
104
+ each conversion is an independent subprocess call.
105
+ 3. **validation** (`validate_output`) — re-opens each converted file with
106
+ `soundfile` and checks it actually matches the requested sample rate,
107
+ channel count, and minimum duration. This catches the case where ffmpeg
108
+ exits 0 but silently produced something degenerate.
109
+ 4. **manifest** (`build_manifest` / `write_manifest`) — JSONL file, one row
110
+ per source file, recording `status` (`ok` / `conversion_failed` /
111
+ `validation_failed`), output path, duration, sample rate, and any error.
112
+
113
+ Conversion failures don't abort the batch — a bad file in a 50,000-file
114
+ corpus shows up as one `conversion_failed` row in the manifest, not a crashed
115
+ job three hours in.
116
+
117
+ ### `chunk` pipeline stages
118
+
119
+ 1. **discovery** (`find_audio_files`) — recursively find supported source files
120
+ under an input directory.
121
+ 2. **chunking** (`chunk_file` / `chunk_batch`) — runs Silero VAD over each
122
+ file (decoding/resampling via ffmpeg) and splits it into speech-only
123
+ chunks bounded by a `[min, max]` duration window, so silence-heavy source
124
+ recordings don't waste pretraining compute. Needs the `chunking` extra
125
+ (`torch` + `silero-vad`, see Setup below).
126
+ 3. **manifest** (`build_chunk_manifest` / `write_manifest`), optional — JSONL
127
+ file, one row per source file, recording `status` (`ok` /
128
+ `chunking_failed`), chunk count, and chunk paths.
129
+
130
+ Chunking failures (e.g. no speech detected) work the same way: `chunk_batch`
131
+ returns a `ChunkResult` per file instead of raising.
132
+
133
+ ## Setup
134
+
135
+ ```bash
136
+ # ffmpeg is a system dependency, not a pip package
137
+ sudo apt-get install ffmpeg # or: brew install ffmpeg
138
+
139
+ pip install audio-prep-pipeline
140
+ ```
141
+
142
+ Install directly from GitHub:
143
+
144
+ ```bash
145
+ pip install "audio-prep-pipeline @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
146
+ ```
147
+
148
+ For local development:
149
+
150
+ ```bash
151
+ make install # pip install -e ".[dev]" + pre-commit install
152
+ ```
153
+
154
+ Chunking needs an extra install -- `make install` alone does not pull in
155
+ `torch`/`silero-vad`/`tqdm`, since `convert` doesn't need them:
156
+
157
+ ```bash
158
+ pip install -e ".[chunking]"
159
+ ```
160
+
161
+ Or from GitHub:
162
+
163
+ ```bash
164
+ pip install "audio-prep-pipeline[chunking] @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
165
+ ```
166
+
167
+ The first `chunk` run needs either the `silero-vad` pip package installed, or
168
+ network access so `torch.hub` can download `snakers4/silero-vad` once (it's
169
+ then cached under `~/.cache/torch/hub`). If neither is available, pass
170
+ `--allow-energy-fallback` to use a lower-quality offline detector instead of
171
+ failing.
172
+
173
+ ## Usage
174
+
175
+ ### `audio-prep convert`
176
+
177
+ ```bash
178
+ audio-prep convert \
179
+ --input-dir data/raw_mp3 \
180
+ --output-dir data/wav16k \
181
+ --format wav \
182
+ --sample-rate 16000 \
183
+ --workers 8 \
184
+ --manifest data/manifest.jsonl
185
+ ```
186
+
187
+ | Flag | Default | Meaning |
188
+ |---|---|---|
189
+ | `--input-dir` | *(required)* | directory of source audio files |
190
+ | `--output-dir` | *(required)* | where converted output is written |
191
+ | `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
192
+ | `--format` | `wav` | output format (`wav` or `flac`) |
193
+ | `--sample-rate` | 16000 | target sample rate |
194
+ | `--channels` | 1 | target channel count |
195
+ | `--workers` | 4 | parallel conversion workers |
196
+ | `--min-duration-sec` | 0.5 | validation fails files shorter than this |
197
+ | `--overwrite` | off | re-convert even if output already exists and passes validation |
198
+ | `--normalize-loudness` | off | apply EBU R128 loudness normalization (-23 LUFS) |
199
+ | `--manifest` | none | path to write a JSONL manifest |
200
+
201
+ ### `audio-prep chunk`
202
+
203
+ Independent of `convert` -- scans `--input-dir` for supported source files and runs VAD
204
+ chunking directly against them, decoding (and resampling, if `--sample-rate`
205
+ doesn't match the source) via ffmpeg:
206
+
207
+ ```bash
208
+ audio-prep chunk \
209
+ --input-dir data/raw_mp3 \
210
+ --output-dir data/chunks \
211
+ --sample-rate 16000 \
212
+ --format flac \
213
+ --min-duration-sec 5 \
214
+ --max-duration-sec 20 \
215
+ --workers 4 \
216
+ --manifest data/chunk_manifest.jsonl
217
+ ```
218
+
219
+ | Flag | Default | Meaning |
220
+ |---|---|---|
221
+ | `--input-dir` | *(required)* | directory of source audio files to scan and chunk |
222
+ | `--output-dir` | `<input-dir>/chunks` | where chunks are written |
223
+ | `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
224
+ | `--format` | `wav` | output chunk format (`wav` or `flac`) |
225
+ | `--sample-rate` | 16000 | resample (via ffmpeg) to this rate before chunking if the source doesn't already match it |
226
+ | `--min-duration-sec` | 5.0 | drop chunks shorter than this |
227
+ | `--max-duration-sec` | 20.0 | split longer speech into windows this size |
228
+ | `--workers` | 4 | parallel chunking workers |
229
+ | `--overwrite` | off | re-chunk even if valid output already exists |
230
+ | `--allow-energy-fallback` | off | fall back to a low-quality energy detector if Silero can't load, instead of raising |
231
+ | `--manifest` | none | path to write a JSONL chunk manifest (source file, status, chunk count/paths) |
232
+
233
+ `chunk_file` itself doesn't care about source extension -- it decodes
234
+ whatever path it's given -- so the Python API can also chunk an existing
235
+ WAV/FLAC corpus (e.g. `convert_batch` output) by passing `source_files`
236
+ explicitly instead of relying on `chunk_batch` discovery:
237
+
238
+ ```python
239
+ from audio_prep import ConversionConfig, convert_batch, build_manifest, validate_output
240
+ from audio_prep import ChunkConfig, chunk_batch
241
+
242
+ config = ConversionConfig(output_format="wav", sample_rate=16_000, channels=1, num_workers=8)
243
+ results = convert_batch("data/raw_mp3", "data/wav16k", config)
244
+ validations = {r.output: validate_output(r.output, config) for r in results if r.success}
245
+ records = build_manifest(results, validations)
246
+
247
+ # source_files bypasses chunk_batch discovery, so this works
248
+ # directly against the already-converted WAV output above.
249
+ valid_outputs = [path for path, v in validations.items() if v.valid]
250
+ chunk_config = ChunkConfig(min_duration_sec=5, max_duration_sec=20, num_workers=4)
251
+ chunk_results = chunk_batch("data/wav16k", "data/wav16k/chunks", chunk_config, source_files=valid_outputs)
252
+ ```
253
+
254
+ ## Development workflow
255
+
256
+ ```bash
257
+ make format # ruff format + autofix
258
+ make lint # ruff check
259
+ make typecheck # mypy --strict
260
+ make test # pytest with coverage
261
+ make check # all of the above -- run this before opening a PR
262
+ ```
263
+
264
+ `pre-commit` (installed via `make install`) runs ruff + mypy + basic hygiene
265
+ checks automatically on every commit. CI (`.github/workflows/ci.yml`) re-runs
266
+ the same checks plus the full test matrix across Python 3.10–3.12 on every
267
+ push and PR.
268
+
269
+ ## Extending this
270
+
271
+ Natural next additions, in roughly the order they'd come up:
272
+
273
+ - **Streaming manifest writes** for very large corpora, instead of holding
274
+ all `ConversionResult`s in memory before writing.
275
+
276
+ **Note**: This is the template, that you have to extend from. `Main` branch is protected so you have to create another branch to work on.
@@ -0,0 +1,237 @@
1
+ # audio-prep
2
+
3
+ [![CI](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml/badge.svg)](https://github.com/nattkorat/audio-prep-pipeline/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/audio-prep-pipeline.svg)](https://pypi.org/project/audio-prep-pipeline/)
5
+ ![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue)
6
+ ![License: MIT](https://img.shields.io/badge/license-MIT-green)
7
+ ![Code style: Ruff](https://img.shields.io/badge/code%20style-ruff-46a2f1)
8
+ ![Type checked: mypy](https://img.shields.io/badge/type%20checked-mypy-blue)
9
+ ![Requires FFmpeg](https://img.shields.io/badge/requires-FFmpeg-007808)
10
+
11
+
12
+ Convert source audio into pretraining-ready WAV/FLAC, with validation and a
13
+ dataset manifest as the handoff artifact to the pretraining pipeline. Discovery
14
+ supports common FFmpeg-readable audio/video containers by default and can scan
15
+ all files with `--extensions all`.
16
+
17
+ Defaults: **16 kHz, mono** — the standard input spec for Wav2Vec2 / XLS-R
18
+ style self-supervised speech pretraining. Override via CLI flags if a
19
+ different downstream model needs something else.
20
+
21
+ ## Repo layout
22
+
23
+ ```
24
+ audio-prep-pipeline/
25
+ ├── src/audio_prep/
26
+ │ ├── config.py # ConversionConfig: target spec + behavior knobs
27
+ │ ├── converter.py # find_audio_files, convert_file, convert_batch
28
+ │ ├── validator.py # probe_duration, validate_output
29
+ │ ├── chunker.py # (optional) ChunkConfig, chunk_file, chunk_batch -- Silero VAD speech chunking
30
+ │ ├── manifest.py # build_manifest, write_manifest (JSONL output)
31
+ │ ├── exceptions.py # ConversionError, ProbeError, ChunkingError
32
+ │ └── cli.py # `audio-prep convert ...` / `audio-prep chunk ...` entry point
33
+ ├── tests/
34
+ │ ├── conftest.py # synthetic-audio fixtures (no binary files checked in)
35
+ │ ├── test_converter.py
36
+ │ ├── test_validator.py
37
+ │ ├── test_chunker.py # fake-detector tests, no real VAD model needed
38
+ │ ├── test_manifest.py
39
+ │ └── test_cli.py
40
+ ├── .github/workflows/ci.yml # lint + typecheck + test matrix (3.10-3.12)
41
+ ├── .pre-commit-config.yaml # ruff + mypy + basic hygiene hooks, runs on every commit
42
+ ├── pyproject.toml # deps, ruff config, mypy config, pytest config
43
+ └── Makefile # `make check` runs everything CI runs, locally
44
+ ```
45
+
46
+ ## Commands
47
+
48
+ `audio-prep` has two independent subcommands. Neither depends on the other
49
+ running first -- both scan `--input-dir` for supported source files directly.
50
+
51
+ - **`audio-prep convert`** - conversion only: `ffmpeg` resample/remix/re-encode
52
+ into the target WAV/FLAC spec, then validation, then an optional manifest.
53
+ Does not chunk.
54
+ - **`audio-prep chunk`** - chunking only: Silero VAD speech chunking straight
55
+ from source audio, with its own resample/format/manifest options. Does not convert
56
+ or validate.
57
+
58
+ ### `convert` pipeline stages
59
+
60
+ 1. **discovery** (`find_audio_files`) — recursively find supported source files
61
+ under an input directory.
62
+ 2. **conversion** (`convert_file` / `convert_batch`) — shell out to `ffmpeg` to
63
+ resample/remix/re-encode into the target WAV/FLAC spec. Mirrors the input
64
+ directory's subfolder structure on output. Runs in a process pool since
65
+ each conversion is an independent subprocess call.
66
+ 3. **validation** (`validate_output`) — re-opens each converted file with
67
+ `soundfile` and checks it actually matches the requested sample rate,
68
+ channel count, and minimum duration. This catches the case where ffmpeg
69
+ exits 0 but silently produced something degenerate.
70
+ 4. **manifest** (`build_manifest` / `write_manifest`) — JSONL file, one row
71
+ per source file, recording `status` (`ok` / `conversion_failed` /
72
+ `validation_failed`), output path, duration, sample rate, and any error.
73
+
74
+ Conversion failures don't abort the batch — a bad file in a 50,000-file
75
+ corpus shows up as one `conversion_failed` row in the manifest, not a crashed
76
+ job three hours in.
77
+
78
+ ### `chunk` pipeline stages
79
+
80
+ 1. **discovery** (`find_audio_files`) — recursively find supported source files
81
+ under an input directory.
82
+ 2. **chunking** (`chunk_file` / `chunk_batch`) — runs Silero VAD over each
83
+ file (decoding/resampling via ffmpeg) and splits it into speech-only
84
+ chunks bounded by a `[min, max]` duration window, so silence-heavy source
85
+ recordings don't waste pretraining compute. Needs the `chunking` extra
86
+ (`torch` + `silero-vad`, see Setup below).
87
+ 3. **manifest** (`build_chunk_manifest` / `write_manifest`), optional — JSONL
88
+ file, one row per source file, recording `status` (`ok` /
89
+ `chunking_failed`), chunk count, and chunk paths.
90
+
91
+ Chunking failures (e.g. no speech detected) work the same way: `chunk_batch`
92
+ returns a `ChunkResult` per file instead of raising.
93
+
94
+ ## Setup
95
+
96
+ ```bash
97
+ # ffmpeg is a system dependency, not a pip package
98
+ sudo apt-get install ffmpeg # or: brew install ffmpeg
99
+
100
+ pip install audio-prep-pipeline
101
+ ```
102
+
103
+ Install directly from GitHub:
104
+
105
+ ```bash
106
+ pip install "audio-prep-pipeline @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
107
+ ```
108
+
109
+ For local development:
110
+
111
+ ```bash
112
+ make install # pip install -e ".[dev]" + pre-commit install
113
+ ```
114
+
115
+ Chunking needs an extra install -- `make install` alone does not pull in
116
+ `torch`/`silero-vad`/`tqdm`, since `convert` doesn't need them:
117
+
118
+ ```bash
119
+ pip install -e ".[chunking]"
120
+ ```
121
+
122
+ Or from GitHub:
123
+
124
+ ```bash
125
+ pip install "audio-prep-pipeline[chunking] @ git+https://github.com/nattkorat/audio-prep-pipeline.git"
126
+ ```
127
+
128
+ The first `chunk` run needs either the `silero-vad` pip package installed, or
129
+ network access so `torch.hub` can download `snakers4/silero-vad` once (it's
130
+ then cached under `~/.cache/torch/hub`). If neither is available, pass
131
+ `--allow-energy-fallback` to use a lower-quality offline detector instead of
132
+ failing.
133
+
134
+ ## Usage
135
+
136
+ ### `audio-prep convert`
137
+
138
+ ```bash
139
+ audio-prep convert \
140
+ --input-dir data/raw_mp3 \
141
+ --output-dir data/wav16k \
142
+ --format wav \
143
+ --sample-rate 16000 \
144
+ --workers 8 \
145
+ --manifest data/manifest.jsonl
146
+ ```
147
+
148
+ | Flag | Default | Meaning |
149
+ |---|---|---|
150
+ | `--input-dir` | *(required)* | directory of source audio files |
151
+ | `--output-dir` | *(required)* | where converted output is written |
152
+ | `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
153
+ | `--format` | `wav` | output format (`wav` or `flac`) |
154
+ | `--sample-rate` | 16000 | target sample rate |
155
+ | `--channels` | 1 | target channel count |
156
+ | `--workers` | 4 | parallel conversion workers |
157
+ | `--min-duration-sec` | 0.5 | validation fails files shorter than this |
158
+ | `--overwrite` | off | re-convert even if output already exists and passes validation |
159
+ | `--normalize-loudness` | off | apply EBU R128 loudness normalization (-23 LUFS) |
160
+ | `--manifest` | none | path to write a JSONL manifest |
161
+
162
+ ### `audio-prep chunk`
163
+
164
+ Independent of `convert` -- scans `--input-dir` for supported source files and runs VAD
165
+ chunking directly against them, decoding (and resampling, if `--sample-rate`
166
+ doesn't match the source) via ffmpeg:
167
+
168
+ ```bash
169
+ audio-prep chunk \
170
+ --input-dir data/raw_mp3 \
171
+ --output-dir data/chunks \
172
+ --sample-rate 16000 \
173
+ --format flac \
174
+ --min-duration-sec 5 \
175
+ --max-duration-sec 20 \
176
+ --workers 4 \
177
+ --manifest data/chunk_manifest.jsonl
178
+ ```
179
+
180
+ | Flag | Default | Meaning |
181
+ |---|---|---|
182
+ | `--input-dir` | *(required)* | directory of source audio files to scan and chunk |
183
+ | `--output-dir` | `<input-dir>/chunks` | where chunks are written |
184
+ | `--extensions` | common FFmpeg audio/video extensions | comma-separated source extensions, or `all` to pass every regular file to FFmpeg |
185
+ | `--format` | `wav` | output chunk format (`wav` or `flac`) |
186
+ | `--sample-rate` | 16000 | resample (via ffmpeg) to this rate before chunking if the source doesn't already match it |
187
+ | `--min-duration-sec` | 5.0 | drop chunks shorter than this |
188
+ | `--max-duration-sec` | 20.0 | split longer speech into windows this size |
189
+ | `--workers` | 4 | parallel chunking workers |
190
+ | `--overwrite` | off | re-chunk even if valid output already exists |
191
+ | `--allow-energy-fallback` | off | fall back to a low-quality energy detector if Silero can't load, instead of raising |
192
+ | `--manifest` | none | path to write a JSONL chunk manifest (source file, status, chunk count/paths) |
193
+
194
+ `chunk_file` itself doesn't care about source extension -- it decodes
195
+ whatever path it's given -- so the Python API can also chunk an existing
196
+ WAV/FLAC corpus (e.g. `convert_batch` output) by passing `source_files`
197
+ explicitly instead of relying on `chunk_batch` discovery:
198
+
199
+ ```python
200
+ from audio_prep import ConversionConfig, convert_batch, build_manifest, validate_output
201
+ from audio_prep import ChunkConfig, chunk_batch
202
+
203
+ config = ConversionConfig(output_format="wav", sample_rate=16_000, channels=1, num_workers=8)
204
+ results = convert_batch("data/raw_mp3", "data/wav16k", config)
205
+ validations = {r.output: validate_output(r.output, config) for r in results if r.success}
206
+ records = build_manifest(results, validations)
207
+
208
+ # source_files bypasses chunk_batch discovery, so this works
209
+ # directly against the already-converted WAV output above.
210
+ valid_outputs = [path for path, v in validations.items() if v.valid]
211
+ chunk_config = ChunkConfig(min_duration_sec=5, max_duration_sec=20, num_workers=4)
212
+ chunk_results = chunk_batch("data/wav16k", "data/wav16k/chunks", chunk_config, source_files=valid_outputs)
213
+ ```
214
+
215
+ ## Development workflow
216
+
217
+ ```bash
218
+ make format # ruff format + autofix
219
+ make lint # ruff check
220
+ make typecheck # mypy --strict
221
+ make test # pytest with coverage
222
+ make check # all of the above -- run this before opening a PR
223
+ ```
224
+
225
+ `pre-commit` (installed via `make install`) runs ruff + mypy + basic hygiene
226
+ checks automatically on every commit. CI (`.github/workflows/ci.yml`) re-runs
227
+ the same checks plus the full test matrix across Python 3.10–3.12 on every
228
+ push and PR.
229
+
230
+ ## Extending this
231
+
232
+ Natural next additions, in roughly the order they'd come up:
233
+
234
+ - **Streaming manifest writes** for very large corpora, instead of holding
235
+ all `ConversionResult`s in memory before writing.
236
+
237
+ **Note**: This is the template, that you have to extend from. `Main` branch is protected so you have to create another branch to work on.