dharma-transcribe 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dharma_transcribe-0.1.0/.env.example +44 -0
- dharma_transcribe-0.1.0/.github/FUNDING.yml +2 -0
- dharma_transcribe-0.1.0/.github/workflows/ci.yml +50 -0
- dharma_transcribe-0.1.0/.github/workflows/publish.yml +38 -0
- dharma_transcribe-0.1.0/.gitignore +19 -0
- dharma_transcribe-0.1.0/.pre-commit-config.yaml +13 -0
- dharma_transcribe-0.1.0/CHANGELOG.md +41 -0
- dharma_transcribe-0.1.0/CONTRIBUTING.md +85 -0
- dharma_transcribe-0.1.0/LICENSE +21 -0
- dharma_transcribe-0.1.0/PKG-INFO +227 -0
- dharma_transcribe-0.1.0/README.md +191 -0
- dharma_transcribe-0.1.0/build_anki_deck.py +466 -0
- dharma_transcribe-0.1.0/pyproject.toml +91 -0
- dharma_transcribe-0.1.0/requirements.txt +10 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/__init__.py +18 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/__main__.py +5 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/align.py +85 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/cli.py +136 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/config.py +133 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/diarize.py +58 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/gpu.py +64 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/ingest.py +95 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/llm_correct.py +294 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/output.py +188 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/pipeline.py +128 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/tibetan_second_pass.py +134 -0
- dharma_transcribe-0.1.0/src/dharma_transcribe/transcribe.py +75 -0
- dharma_transcribe-0.1.0/src/test_e2e_llm.py +29 -0
- dharma_transcribe-0.1.0/tests/conftest.py +111 -0
- dharma_transcribe-0.1.0/tests/fixtures/README.md +17 -0
- dharma_transcribe-0.1.0/tests/test_config.py +104 -0
- dharma_transcribe-0.1.0/tests/test_e2e.py +43 -0
- dharma_transcribe-0.1.0/tests/test_gpu.py +47 -0
- dharma_transcribe-0.1.0/tests/test_imports.py +86 -0
- dharma_transcribe-0.1.0/tests/test_ingest.py +68 -0
- dharma_transcribe-0.1.0/tests/test_llm_correct.py +144 -0
- dharma_transcribe-0.1.0/tests/test_output.py +156 -0
- dharma_transcribe-0.1.0/tests/test_pipeline.py +151 -0
- dharma_transcribe-0.1.0/tests/test_tibetan_second_pass.py +44 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Dharma Transcription Pipeline — Environment Variables
|
|
2
|
+
# Copy to .env and fill in your values. See .gitignore — .env is never committed.
|
|
3
|
+
|
|
4
|
+
# --- Required for LLM correction (Stage 6) ---
|
|
5
|
+
# Any OpenAI-compatible API endpoint. Examples:
|
|
6
|
+
# Synthetic.new: https://api.example.com/v1
|
|
7
|
+
# OpenAI: https://api.openai.com/v1
|
|
8
|
+
# Ollama (local): http://localhost:11434/v1
|
|
9
|
+
# vLLM (local): http://localhost:8000/v1
|
|
10
|
+
# LM Studio: http://localhost:1234/v1
|
|
11
|
+
DHARMA_LLM_API_URL=https://api.example.com/v1
|
|
12
|
+
DHARMA_LLM_API_KEY=your-api-key-here
|
|
13
|
+
DHARMA_LLM_MODEL=hf:openai/gpt-oss-120b
|
|
14
|
+
|
|
15
|
+
# --- Required for diarization (Stage 4) ---
|
|
16
|
+
# HuggingFace token for pyannote speaker-diarization-community-1 model.
|
|
17
|
+
# Without this, diarization is skipped (transcription still completes).
|
|
18
|
+
# Get one at: https://huggingface.co/settings/tokens
|
|
19
|
+
HF_TOKEN=hf_your_token_here
|
|
20
|
+
|
|
21
|
+
# --- Optional overrides ---
|
|
22
|
+
# Compute device: "cuda" (GPU) or "cpu"
|
|
23
|
+
# DHARMA_DEVICE=cuda
|
|
24
|
+
|
|
25
|
+
# WhisperX model size: large-v3, large-v2, medium, small, etc.
|
|
26
|
+
# DHARMA_WHISPER_MODEL=large-v3
|
|
27
|
+
|
|
28
|
+
# Compute type: int8, float16, int8_float16, etc.
|
|
29
|
+
# DHARMA_COMPUTE_TYPE=int8
|
|
30
|
+
|
|
31
|
+
# Batch size for WhisperX transcription
|
|
32
|
+
# DHARMA_BATCH_SIZE=4
|
|
33
|
+
|
|
34
|
+
# Tibetan second-pass model (OpenPecha dharma-trained)
|
|
35
|
+
# DHARMA_TIBETAN_MODEL=openpecha/op-whisper_small-ft-v2
|
|
36
|
+
|
|
37
|
+
# Output directory (defaults to ./output)
|
|
38
|
+
# DHARMA_OUTPUT_DIR=/path/to/output
|
|
39
|
+
|
|
40
|
+
# Log directory (defaults to ./logs)
|
|
41
|
+
# DHARMA_LOG_DIR=/path/to/logs
|
|
42
|
+
|
|
43
|
+
# Confidence threshold for review queue (0.0-1.0)
|
|
44
|
+
# DHARMA_CONFIDENCE_THRESHOLD=0.5
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main]
|
|
8
|
+
|
|
9
|
+
permissions:
|
|
10
|
+
contents: read
|
|
11
|
+
|
|
12
|
+
jobs:
|
|
13
|
+
lint:
|
|
14
|
+
runs-on: ubuntu-latest
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: actions/setup-python@v5
|
|
18
|
+
with:
|
|
19
|
+
python-version: "3.10"
|
|
20
|
+
- run: pip install ruff
|
|
21
|
+
- run: ruff check src/ tests/
|
|
22
|
+
- run: ruff format --check src/ tests/
|
|
23
|
+
|
|
24
|
+
typecheck:
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
steps:
|
|
27
|
+
- uses: actions/checkout@v4
|
|
28
|
+
- uses: actions/setup-python@v5
|
|
29
|
+
with:
|
|
30
|
+
python-version: "3.10"
|
|
31
|
+
- run: pip install mypy
|
|
32
|
+
- run: mypy src/dharma_transcribe/
|
|
33
|
+
|
|
34
|
+
test:
|
|
35
|
+
runs-on: ubuntu-latest
|
|
36
|
+
strategy:
|
|
37
|
+
matrix:
|
|
38
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
39
|
+
steps:
|
|
40
|
+
- uses: actions/checkout@v4
|
|
41
|
+
- uses: actions/setup-python@v5
|
|
42
|
+
with:
|
|
43
|
+
python-version: ${{ matrix.python-version }}
|
|
44
|
+
- run: pip install -e ".[dev]"
|
|
45
|
+
- run: pytest tests/ --ignore=tests/test_e2e.py --cov=dharma_transcribe --cov-report=xml --cov-report=term-missing
|
|
46
|
+
- uses: actions/upload-artifact@v4
|
|
47
|
+
if: matrix.python-version == '3.10'
|
|
48
|
+
with:
|
|
49
|
+
name: coverage-report
|
|
50
|
+
path: coverage.xml
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
# Trusted publishing via OIDC — no API tokens needed.
|
|
12
|
+
# Configure at: https://pypi.org/manage/account/publishing/
|
|
13
|
+
jobs:
|
|
14
|
+
build:
|
|
15
|
+
runs-on: ubuntu-latest
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
- uses: actions/setup-python@v5
|
|
19
|
+
with:
|
|
20
|
+
python-version: "3.10"
|
|
21
|
+
- run: pip install build
|
|
22
|
+
- run: python -m build
|
|
23
|
+
- uses: actions/upload-artifact@v4
|
|
24
|
+
with:
|
|
25
|
+
name: dist
|
|
26
|
+
path: dist/
|
|
27
|
+
|
|
28
|
+
publish:
|
|
29
|
+
needs: build
|
|
30
|
+
runs-on: ubuntu-latest
|
|
31
|
+
permissions:
|
|
32
|
+
id-token: write # Required for trusted publishing
|
|
33
|
+
steps:
|
|
34
|
+
- uses: actions/download-artifact@v4
|
|
35
|
+
with:
|
|
36
|
+
name: dist
|
|
37
|
+
path: dist/
|
|
38
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
venv/
|
|
2
|
+
.venv/
|
|
3
|
+
output/wav/
|
|
4
|
+
output/json/
|
|
5
|
+
output/srt/
|
|
6
|
+
output/vtt/
|
|
7
|
+
output/txt/
|
|
8
|
+
output/review/
|
|
9
|
+
output/corrections/
|
|
10
|
+
output/manifest.json
|
|
11
|
+
__pycache__/
|
|
12
|
+
*.pyc
|
|
13
|
+
.env
|
|
14
|
+
models/
|
|
15
|
+
logs/
|
|
16
|
+
tests/test_e2e_llm.py
|
|
17
|
+
output/
|
|
18
|
+
tests/test_e2e_llm.py
|
|
19
|
+
tests/test_e2e_llm.py
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
repos:
|
|
2
|
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
3
|
+
rev: v0.4.0
|
|
4
|
+
hooks:
|
|
5
|
+
- id: ruff
|
|
6
|
+
args: [--fix]
|
|
7
|
+
- id: ruff-format
|
|
8
|
+
- repo: https://github.com/pre-commit/mirrors-mypy
|
|
9
|
+
rev: v1.10.0
|
|
10
|
+
hooks:
|
|
11
|
+
- id: mypy
|
|
12
|
+
additional_dependencies: []
|
|
13
|
+
args: [--ignore-missing-imports]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
- GitHub Actions CI pipeline (lint, typecheck, test on Python 3.10/3.11/3.12)
|
|
12
|
+
- PyPI publishing workflow (trusted publishing via OIDC)
|
|
13
|
+
- Pre-commit hooks configuration (ruff + mypy)
|
|
14
|
+
- `.env.example` with all configurable environment variables
|
|
15
|
+
- `CONTRIBUTING.md` with development setup and testing guide
|
|
16
|
+
- `tests/fixtures/` directory with README for test audio samples
|
|
17
|
+
|
|
18
|
+
## [0.1.0] - 2026-08-23
|
|
19
|
+
|
|
20
|
+
### Added
|
|
21
|
+
- 7-stage pipeline architecture: Ingest → WhisperX → Alignment → Diarization → Tibetan Second-Pass → LLM Correction → Output
|
|
22
|
+
- WhisperX large-v3 primary transcription with automatic language detection
|
|
23
|
+
- Per-language forced alignment (wav2vec2 for en, ja, bo, sa)
|
|
24
|
+
- Speaker diarization via pyannote speaker-diarization-community-1
|
|
25
|
+
- Tibetan second-pass transcription using OpenPecha op-whisper_small-ft-v2 (dharma-trained)
|
|
26
|
+
- LLM post-correction via any OpenAI-compatible API (cloud or local)
|
|
27
|
+
- Output formats: JSON, SRT, VTT, TXT, review queue
|
|
28
|
+
- Corrections dictionary for deterministic pre-LLM fixes
|
|
29
|
+
- Idempotent manifest for batch processing
|
|
30
|
+
- VRAM management with model flushing between stages (6 GB consumer GPU support)
|
|
31
|
+
- `--device cpu` flag for CPU-only mode
|
|
32
|
+
- `--skip-llm` flag to skip LLM correction
|
|
33
|
+
- Environment-variable-based configuration (no hardcoded secrets)
|
|
34
|
+
- `src/` package layout with `pyproject.toml` (hatchling build backend)
|
|
35
|
+
- 54 unit tests covering all modules
|
|
36
|
+
- MIT license
|
|
37
|
+
|
|
38
|
+
### Security
|
|
39
|
+
- Git history scrubbed of all API keys, URLs, and private paths
|
|
40
|
+
- All sensitive values read from environment variables
|
|
41
|
+
- No credentials in source code or commit history
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
## Development Setup
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
git clone https://github.com/guan-tends/dharma-transcribe.git
|
|
7
|
+
cd dharma-transcribe
|
|
8
|
+
python3 -m venv venv
|
|
9
|
+
source venv/bin/activate
|
|
10
|
+
pip install -e ".[dev]"
|
|
11
|
+
pre-commit install
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## Workflow
|
|
15
|
+
|
|
16
|
+
1. Create a branch: `git checkout -b feature/your-feature`
|
|
17
|
+
2. Make changes. Commit early, commit often.
|
|
18
|
+
3. Run checks before pushing:
|
|
19
|
+
```bash
|
|
20
|
+
ruff check src tests
|
|
21
|
+
ruff format --check src tests
|
|
22
|
+
mypy src/dharma_transcribe
|
|
23
|
+
pytest
|
|
24
|
+
```
|
|
25
|
+
4. Push and open a PR.
|
|
26
|
+
|
|
27
|
+
## Testing
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
# Run all tests
|
|
31
|
+
pytest
|
|
32
|
+
|
|
33
|
+
# Run without GPU tests
|
|
34
|
+
pytest -m "not gpu"
|
|
35
|
+
|
|
36
|
+
# Run with coverage
|
|
37
|
+
pytest --cov=dharma_transcribe --cov-report=term-missing
|
|
38
|
+
|
|
39
|
+
# Run a single module
|
|
40
|
+
pytest tests/test_config.py
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### Adding Test Fixtures
|
|
44
|
+
|
|
45
|
+
Place small audio samples in `tests/fixtures/`. A fixture should be:
|
|
46
|
+
- Short (5–30 seconds)
|
|
47
|
+
- Royalty-free or your own recording
|
|
48
|
+
- Named descriptively: `english_sample.wav`, `tibetan_sample.wav`
|
|
49
|
+
|
|
50
|
+
Generate a test WAV with ffmpeg:
|
|
51
|
+
```bash
|
|
52
|
+
ffmpeg -f lavfi -i "sine=frequency=440:duration=5" -ar 16000 -ac 1 tests/fixtures/tone_5s.wav
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Code Style
|
|
56
|
+
|
|
57
|
+
- **Linter**: ruff (replaces flake8 + isort + black)
|
|
58
|
+
- **Formatter**: ruff format
|
|
59
|
+
- **Type checking**: mypy (non-strict, `ignore_missing_imports = true`)
|
|
60
|
+
- **Line length**: 100 characters
|
|
61
|
+
- **Python**: 3.10+ (uses `str | None` union syntax)
|
|
62
|
+
|
|
63
|
+
## Project Structure
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
src/dharma_transcribe/ # Package source
|
|
67
|
+
__init__.py # Package metadata
|
|
68
|
+
cli.py # CLI entry point (argparse)
|
|
69
|
+
pipeline.py # Stage orchestration
|
|
70
|
+
config.py # Environment-based configuration
|
|
71
|
+
ingest.py # Stage 1: audio extraction
|
|
72
|
+
transcribe.py # Stage 2: WhisperX ASR
|
|
73
|
+
align.py # Stage 3: forced alignment
|
|
74
|
+
diarize.py # Stage 4: speaker diarization
|
|
75
|
+
tibetan_second_pass.py # Stage 5: Tibetan re-transcription
|
|
76
|
+
llm_correct.py # Stage 6: LLM post-correction
|
|
77
|
+
output.py # Stage 7: multi-format output
|
|
78
|
+
gpu.py # GPU memory utilities
|
|
79
|
+
tests/ # Test suite
|
|
80
|
+
conftest.py # Shared fixtures
|
|
81
|
+
test_*.py # Unit and integration tests
|
|
82
|
+
fixtures/ # Test audio samples
|
|
83
|
+
pyproject.toml # Project metadata, tool config
|
|
84
|
+
.env.example # Environment variable template
|
|
85
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 David Newman & Guan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dharma-transcribe
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Headless multilingual transcription pipeline for Buddhist dharma teachings
|
|
5
|
+
Project-URL: Homepage, https://github.com/guan-tends/dharma-transcribe
|
|
6
|
+
Project-URL: Repository, https://github.com/guan-tends/dharma-transcribe
|
|
7
|
+
Author: David Newman & Guan
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: asr,buddhist,dharma,multilingual,sanskrit,speech-to-text,tibetan,transcription,whisper,whisperx
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
19
|
+
Classifier: Topic :: Religion
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Requires-Dist: openai>=1.0
|
|
22
|
+
Requires-Dist: requests>=2.28
|
|
23
|
+
Requires-Dist: tiktoken>=0.5
|
|
24
|
+
Requires-Dist: torch>=2.0
|
|
25
|
+
Requires-Dist: torchaudio>=2.0
|
|
26
|
+
Requires-Dist: transformers>=4.40
|
|
27
|
+
Requires-Dist: whisperx>=3.8.0
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
30
|
+
Requires-Dist: pre-commit>=3.7; extra == 'dev'
|
|
31
|
+
Requires-Dist: pytest-cov>=4.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest-mock>=3.12; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.4; extra == 'dev'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# Dharma Transcription Pipeline
|
|
38
|
+
|
|
39
|
+
A headless multilingual transcription pipeline designed for Buddhist dharma teachings. Handles mixed-language audio (Tibetan, Sanskrit, English, Japanese) with per-language forced alignment, speaker diarization, a Tibetan second-pass using dharma-trained models, and optional LLM post-correction.
|
|
40
|
+
|
|
41
|
+
**Privacy-first**: all processing runs locally. Sacred content never touches a cloud unless you explicitly configure LLM correction with a cloud API.
|
|
42
|
+
|
|
43
|
+
## Architecture — 7 Stages
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
Audio/Video → Ingest → WhisperX → Alignment → Diarization → Tibetan 2nd-Pass → LLM Correction → Output
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
| Stage | Purpose | Model | GPU Memory |
|
|
50
|
+
|-------|---------|-------|-----------|
|
|
51
|
+
| 1. Ingest | ffmpeg extract to 16kHz mono WAV, checksum idempotency | ffmpeg | — |
|
|
52
|
+
| 2. Transcription | Primary ASR with auto language detection | WhisperX large-v3 (int8) | ~4 GB |
|
|
53
|
+
| 3. Alignment | Word-level timestamps per language | wav2vec2 (per-language) | ~1–2 GB |
|
|
54
|
+
| 4. Diarization | Speaker identification | pyannote diarization-community-1 | ~1 GB |
|
|
55
|
+
| 5. Tibetan 2nd-Pass | Re-transcribe Tibetan segments with dharma-trained model | OpenPecha op-whisper_small-ft-v2 | ~1 GB |
|
|
56
|
+
| 6. LLM Correction | Post-correction with dharma domain knowledge | Any OpenAI-compatible API | (cloud or local) |
|
|
57
|
+
| 7. Output | JSON, SRT, VTT, TXT, review queue | — | — |
|
|
58
|
+
|
|
59
|
+
Stages run serially with GPU memory flushing between each — designed for 6 GB consumer GPUs.
|
|
60
|
+
|
|
61
|
+
## Setup
|
|
62
|
+
|
|
63
|
+
### Prerequisites
|
|
64
|
+
|
|
65
|
+
- Python 3.10+
|
|
66
|
+
- ffmpeg + ffprobe (system install)
|
|
67
|
+
- NVIDIA GPU with CUDA (optional — `--device cpu` works for all stages)
|
|
68
|
+
- HuggingFace token (for diarization only — transcription works without it)
|
|
69
|
+
|
|
70
|
+
### Installation
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
# Clone
|
|
74
|
+
git clone https://github.com/guan-tends/dharma-transcribe.git
|
|
75
|
+
cd dharma-transcribe
|
|
76
|
+
|
|
77
|
+
# Create virtual environment
|
|
78
|
+
python3 -m venv venv
|
|
79
|
+
source venv/bin/activate
|
|
80
|
+
|
|
81
|
+
# Install with pip
|
|
82
|
+
pip install -e ".[dev]"
|
|
83
|
+
|
|
84
|
+
# Or install from PyPI (when published)
|
|
85
|
+
pip install dharma-transcribe[dev]
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### GPU Setup (optional but recommended)
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
# Install PyTorch with CUDA support (adjust cuXXX for your CUDA version)
|
|
92
|
+
pip install torch torchaudio --index-url https://download.pytorch.org/whl/cu128
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### HuggingFace Token (for diarization)
|
|
96
|
+
|
|
97
|
+
The pipeline needs a HuggingFace token to download the pyannote diarization model. Without it, diarization is skipped — transcription still completes normally.
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
# Get a token: https://huggingface.co/settings/tokens
|
|
101
|
+
# Also accept the model license: https://huggingface.co/pyannote/speaker-diarization-community-1
|
|
102
|
+
|
|
103
|
+
export HF_TOKEN=hf_your_token_here
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### LLM Correction Configuration (optional)
|
|
107
|
+
|
|
108
|
+
Stage 6 uses any OpenAI-compatible API. Configure via environment variables:
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
# Cloud API example (Synthetic.new)
|
|
112
|
+
export DHARMA_LLM_API_URL=https://api.example.com/v1
|
|
113
|
+
export DHARMA_LLM_API_KEY=your-key
|
|
114
|
+
export DHARMA_LLM_MODEL=hf:openai/gpt-oss-120b
|
|
115
|
+
|
|
116
|
+
# Local model example (Ollama)
|
|
117
|
+
ollama pull qwen3.5:9b
|
|
118
|
+
export DHARMA_LLM_API_URL=http://localhost:11434/v1
|
|
119
|
+
export DHARMA_LLM_API_KEY=ollama
|
|
120
|
+
export DHARMA_LLM_MODEL=qwen3.5:9b
|
|
121
|
+
|
|
122
|
+
# vLLM example (local GPU)
|
|
123
|
+
# Start vLLM server: vllm serve Qwen/Qwen3.5-9B
|
|
124
|
+
export DHARMA_LLM_API_URL=http://localhost:8000/v1
|
|
125
|
+
export DHARMA_LLM_API_KEY=none
|
|
126
|
+
export DHARMA_LLM_MODEL=Qwen/Qwen3.5-9B
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Or skip LLM correction entirely: `--skip-llm`
|
|
130
|
+
|
|
131
|
+
## Usage
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
# Single file
|
|
135
|
+
dharma-transcribe /path/to/teaching.mp4
|
|
136
|
+
|
|
137
|
+
# Directory (batch — finds all audio/video recursively)
|
|
138
|
+
dharma-transcribe /path/to/recordings/
|
|
139
|
+
|
|
140
|
+
# Skip LLM correction (ASR only, faster)
|
|
141
|
+
dharma-transcribe /path/to/teaching.mp4 --skip-llm
|
|
142
|
+
|
|
143
|
+
# CPU-only mode (no GPU required)
|
|
144
|
+
dharma-transcribe /path/to/teaching.mp4 --device cpu --skip-llm
|
|
145
|
+
|
|
146
|
+
# With explicit HF token
|
|
147
|
+
dharma-transcribe /path/to/teaching.mp4 --hf-token $HF_TOKEN
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
### CLI Flags
|
|
151
|
+
|
|
152
|
+
| Flag | Default | Description |
|
|
153
|
+
|------|---------|-------------|
|
|
154
|
+
| `input` (positional) | — | File or directory to process |
|
|
155
|
+
| `--source-dir` | — | Default source directory |
|
|
156
|
+
| `--skip-llm` | off | Skip LLM correction stage |
|
|
157
|
+
| `--hf-token` | `$HF_TOKEN` | HuggingFace token for diarization |
|
|
158
|
+
| `--device` | `cuda` | Compute device: `cuda` or `cpu` |
|
|
159
|
+
|
|
160
|
+
### Environment Variables
|
|
161
|
+
|
|
162
|
+
See `.env.example` for the full list. Key variables:
|
|
163
|
+
|
|
164
|
+
| Variable | Default | Description |
|
|
165
|
+
|----------|---------|-------------|
|
|
166
|
+
| `DHARMA_LLM_API_URL` | (empty) | OpenAI-compatible API endpoint |
|
|
167
|
+
| `DHARMA_LLM_API_KEY` | (empty) | API key for LLM correction |
|
|
168
|
+
| `DHARMA_LLM_MODEL` | (empty) | Model name for LLM correction |
|
|
169
|
+
| `HF_TOKEN` | (empty) | HuggingFace token for diarization |
|
|
170
|
+
| `DHARMA_DEVICE` | `cuda` | Compute device |
|
|
171
|
+
| `DHARMA_WHISPER_MODEL` | `large-v3` | WhisperX model size |
|
|
172
|
+
| `DHARMA_OUTPUT_DIR` | `./output` | Output directory |
|
|
173
|
+
| `DHARMA_TIBETAN_MODEL` | `openpecha/op-whisper_small-ft-v2` | Tibetan second-pass model |
|
|
174
|
+
|
|
175
|
+
## Output Formats
|
|
176
|
+
|
|
177
|
+
Each processed file generates:
|
|
178
|
+
|
|
179
|
+
- **JSON** — full structured transcript with word-level timestamps, speaker labels, confidence scores
|
|
180
|
+
- **SRT** — subtitle file with speaker labels
|
|
181
|
+
- **VTT** — WebVTT subtitles with speaker tags
|
|
182
|
+
- **TXT** — plain text reading copy
|
|
183
|
+
- **Review queue** — JSON of low-confidence segments for manual review
|
|
184
|
+
|
|
185
|
+
Outputs are written to `output/{json,srt,vtt,txt,review}/`.
|
|
186
|
+
|
|
187
|
+
## Corrections Dictionary
|
|
188
|
+
|
|
189
|
+
`output/corrections/corrections.json` — case-insensitive string replacement applied before LLM correction. Grows from manual review.
|
|
190
|
+
|
|
191
|
+
```json
|
|
192
|
+
{
|
|
193
|
+
"corrections": [
|
|
194
|
+
{"pattern": "bodichita", "replacement": "bodhicitta"},
|
|
195
|
+
{"pattern": "shun yata", "replacement": "shunyata"}
|
|
196
|
+
]
|
|
197
|
+
}
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
## Idempotency
|
|
201
|
+
|
|
202
|
+
The manifest (`output/manifest.json`) tracks processed files. Re-running the pipeline skips completed files. Delete the manifest to reprocess everything.
|
|
203
|
+
|
|
204
|
+
## VRAM Management
|
|
205
|
+
|
|
206
|
+
Designed for consumer GPUs (6 GB+). Stages run serially with `gc.collect()` + `torch.cuda.empty_cache()` between each. No CPU fallback needed — the GPU is flushed fully before loading the next model.
|
|
207
|
+
|
|
208
|
+
## Why This Exists
|
|
209
|
+
|
|
210
|
+
Most transcription tools handle single languages. Dharma teachings commonly mix Tibetan, Sanskrit, English, and sometimes Japanese in a single recording. This pipeline:
|
|
211
|
+
|
|
212
|
+
1. **Detects language per segment** — not per file
|
|
213
|
+
2. **Aligns each language separately** — wav2vec2 models for bo, sa, en, ja
|
|
214
|
+
3. **Re-transcribes Tibetan** — OpenPecha's model (trained on Garchen Rinpoche's teachings) often outperforms WhisperX on Tibetan
|
|
215
|
+
4. **Corrects with dharma knowledge** — LLM post-correction knows bodhicitta from bodichita
|
|
216
|
+
|
|
217
|
+
## License
|
|
218
|
+
|
|
219
|
+
MIT — see [LICENSE](LICENSE).
|
|
220
|
+
|
|
221
|
+
## Support
|
|
222
|
+
|
|
223
|
+
If this pipeline helps preserve dharma teachings, consider supporting its continued development:
|
|
224
|
+
|
|
225
|
+
- **Solana**: `Eu8wQcW68TKMs1a6eqzZu8znzU52QLqQugAMG8uCD6y6`
|
|
226
|
+
- **EVM** (Ethereum / Base / Arbitrum / Optimism / Polygon): `0x2733ff7c865C56d565a99BE1DC11B81cc76850A5`
|
|
227
|
+
- **XRP Ledger**: `r4X6e7McAQj7e8vBCeued1RYu4mCJrREDG`
|