dharma-transcribe 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. dharma_transcribe-0.1.0/.env.example +44 -0
  2. dharma_transcribe-0.1.0/.github/FUNDING.yml +2 -0
  3. dharma_transcribe-0.1.0/.github/workflows/ci.yml +50 -0
  4. dharma_transcribe-0.1.0/.github/workflows/publish.yml +38 -0
  5. dharma_transcribe-0.1.0/.gitignore +19 -0
  6. dharma_transcribe-0.1.0/.pre-commit-config.yaml +13 -0
  7. dharma_transcribe-0.1.0/CHANGELOG.md +41 -0
  8. dharma_transcribe-0.1.0/CONTRIBUTING.md +85 -0
  9. dharma_transcribe-0.1.0/LICENSE +21 -0
  10. dharma_transcribe-0.1.0/PKG-INFO +227 -0
  11. dharma_transcribe-0.1.0/README.md +191 -0
  12. dharma_transcribe-0.1.0/build_anki_deck.py +466 -0
  13. dharma_transcribe-0.1.0/pyproject.toml +91 -0
  14. dharma_transcribe-0.1.0/requirements.txt +10 -0
  15. dharma_transcribe-0.1.0/src/dharma_transcribe/__init__.py +18 -0
  16. dharma_transcribe-0.1.0/src/dharma_transcribe/__main__.py +5 -0
  17. dharma_transcribe-0.1.0/src/dharma_transcribe/align.py +85 -0
  18. dharma_transcribe-0.1.0/src/dharma_transcribe/cli.py +136 -0
  19. dharma_transcribe-0.1.0/src/dharma_transcribe/config.py +133 -0
  20. dharma_transcribe-0.1.0/src/dharma_transcribe/diarize.py +58 -0
  21. dharma_transcribe-0.1.0/src/dharma_transcribe/gpu.py +64 -0
  22. dharma_transcribe-0.1.0/src/dharma_transcribe/ingest.py +95 -0
  23. dharma_transcribe-0.1.0/src/dharma_transcribe/llm_correct.py +294 -0
  24. dharma_transcribe-0.1.0/src/dharma_transcribe/output.py +188 -0
  25. dharma_transcribe-0.1.0/src/dharma_transcribe/pipeline.py +128 -0
  26. dharma_transcribe-0.1.0/src/dharma_transcribe/tibetan_second_pass.py +134 -0
  27. dharma_transcribe-0.1.0/src/dharma_transcribe/transcribe.py +75 -0
  28. dharma_transcribe-0.1.0/src/test_e2e_llm.py +29 -0
  29. dharma_transcribe-0.1.0/tests/conftest.py +111 -0
  30. dharma_transcribe-0.1.0/tests/fixtures/README.md +17 -0
  31. dharma_transcribe-0.1.0/tests/test_config.py +104 -0
  32. dharma_transcribe-0.1.0/tests/test_e2e.py +43 -0
  33. dharma_transcribe-0.1.0/tests/test_gpu.py +47 -0
  34. dharma_transcribe-0.1.0/tests/test_imports.py +86 -0
  35. dharma_transcribe-0.1.0/tests/test_ingest.py +68 -0
  36. dharma_transcribe-0.1.0/tests/test_llm_correct.py +144 -0
  37. dharma_transcribe-0.1.0/tests/test_output.py +156 -0
  38. dharma_transcribe-0.1.0/tests/test_pipeline.py +151 -0
  39. dharma_transcribe-0.1.0/tests/test_tibetan_second_pass.py +44 -0
@@ -0,0 +1,44 @@
1
+ # Dharma Transcription Pipeline — Environment Variables
2
+ # Copy to .env and fill in your values. See .gitignore — .env is never committed.
3
+
4
+ # --- Required for LLM correction (Stage 6) ---
5
+ # Any OpenAI-compatible API endpoint. Examples:
6
+ # Synthetic.new: https://api.example.com/v1
7
+ # OpenAI: https://api.openai.com/v1
8
+ # Ollama (local): http://localhost:11434/v1
9
+ # vLLM (local): http://localhost:8000/v1
10
+ # LM Studio: http://localhost:1234/v1
11
+ DHARMA_LLM_API_URL=https://api.example.com/v1
12
+ DHARMA_LLM_API_KEY=your-api-key-here
13
+ DHARMA_LLM_MODEL=hf:openai/gpt-oss-120b
14
+
15
+ # --- Required for diarization (Stage 4) ---
16
+ # HuggingFace token for pyannote speaker-diarization-community-1 model.
17
+ # Without this, diarization is skipped (transcription still completes).
18
+ # Get one at: https://huggingface.co/settings/tokens
19
+ HF_TOKEN=hf_your_token_here
20
+
21
+ # --- Optional overrides ---
22
+ # Compute device: "cuda" (GPU) or "cpu"
23
+ # DHARMA_DEVICE=cuda
24
+
25
+ # WhisperX model size: large-v3, large-v2, medium, small, etc.
26
+ # DHARMA_WHISPER_MODEL=large-v3
27
+
28
+ # Compute type: int8, float16, int8_float16, etc.
29
+ # DHARMA_COMPUTE_TYPE=int8
30
+
31
+ # Batch size for WhisperX transcription
32
+ # DHARMA_BATCH_SIZE=4
33
+
34
+ # Tibetan second-pass model (OpenPecha dharma-trained)
35
+ # DHARMA_TIBETAN_MODEL=openpecha/op-whisper_small-ft-v2
36
+
37
+ # Output directory (defaults to ./output)
38
+ # DHARMA_OUTPUT_DIR=/path/to/output
39
+
40
+ # Log directory (defaults to ./logs)
41
+ # DHARMA_LOG_DIR=/path/to/logs
42
+
43
+ # Confidence threshold for review queue (0.0-1.0)
44
+ # DHARMA_CONFIDENCE_THRESHOLD=0.5
@@ -0,0 +1,2 @@
1
+ # Cryptocurrency donations — see README Support section for addresses
2
+ custom: ["https://github.com/guan-tends#support"]
@@ -0,0 +1,50 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+ branches: [main]
8
+
9
+ permissions:
10
+ contents: read
11
+
12
+ jobs:
13
+ lint:
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: actions/setup-python@v5
18
+ with:
19
+ python-version: "3.10"
20
+ - run: pip install ruff
21
+ - run: ruff check src/ tests/
22
+ - run: ruff format --check src/ tests/
23
+
24
+ typecheck:
25
+ runs-on: ubuntu-latest
26
+ steps:
27
+ - uses: actions/checkout@v4
28
+ - uses: actions/setup-python@v5
29
+ with:
30
+ python-version: "3.10"
31
+ - run: pip install mypy
32
+ - run: mypy src/dharma_transcribe/
33
+
34
+ test:
35
+ runs-on: ubuntu-latest
36
+ strategy:
37
+ matrix:
38
+ python-version: ["3.10", "3.11", "3.12"]
39
+ steps:
40
+ - uses: actions/checkout@v4
41
+ - uses: actions/setup-python@v5
42
+ with:
43
+ python-version: ${{ matrix.python-version }}
44
+ - run: pip install -e ".[dev]"
45
+ - run: pytest tests/ --ignore=tests/test_e2e.py --cov=dharma_transcribe --cov-report=xml --cov-report=term-missing
46
+ - uses: actions/upload-artifact@v4
47
+ if: matrix.python-version == '3.10'
48
+ with:
49
+ name: coverage-report
50
+ path: coverage.xml
@@ -0,0 +1,38 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ push:
5
+ tags:
6
+ - "v*"
7
+
8
+ permissions:
9
+ contents: read
10
+
11
+ # Trusted publishing via OIDC — no API tokens needed.
12
+ # Configure at: https://pypi.org/manage/account/publishing/
13
+ jobs:
14
+ build:
15
+ runs-on: ubuntu-latest
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: actions/setup-python@v5
19
+ with:
20
+ python-version: "3.10"
21
+ - run: pip install build
22
+ - run: python -m build
23
+ - uses: actions/upload-artifact@v4
24
+ with:
25
+ name: dist
26
+ path: dist/
27
+
28
+ publish:
29
+ needs: build
30
+ runs-on: ubuntu-latest
31
+ permissions:
32
+ id-token: write # Required for trusted publishing
33
+ steps:
34
+ - uses: actions/download-artifact@v4
35
+ with:
36
+ name: dist
37
+ path: dist/
38
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,19 @@
1
+ venv/
2
+ .venv/
3
+ output/wav/
4
+ output/json/
5
+ output/srt/
6
+ output/vtt/
7
+ output/txt/
8
+ output/review/
9
+ output/corrections/
10
+ output/manifest.json
11
+ __pycache__/
12
+ *.pyc
13
+ .env
14
+ models/
15
+ logs/
16
+ tests/test_e2e_llm.py
17
+ output/
18
+ tests/test_e2e_llm.py
19
+ tests/test_e2e_llm.py
@@ -0,0 +1,13 @@
1
+ repos:
2
+ - repo: https://github.com/astral-sh/ruff-pre-commit
3
+ rev: v0.4.0
4
+ hooks:
5
+ - id: ruff
6
+ args: [--fix]
7
+ - id: ruff-format
8
+ - repo: https://github.com/pre-commit/mirrors-mypy
9
+ rev: v1.10.0
10
+ hooks:
11
+ - id: mypy
12
+ additional_dependencies: []
13
+ args: [--ignore-missing-imports]
@@ -0,0 +1,41 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ### Added
11
+ - GitHub Actions CI pipeline (lint, typecheck, test on Python 3.10/3.11/3.12)
12
+ - PyPI publishing workflow (trusted publishing via OIDC)
13
+ - Pre-commit hooks configuration (ruff + mypy)
14
+ - `.env.example` with all configurable environment variables
15
+ - `CONTRIBUTING.md` with development setup and testing guide
16
+ - `tests/fixtures/` directory with README for test audio samples
17
+
18
+ ## [0.1.0] - 2026-08-23
19
+
20
+ ### Added
21
+ - 7-stage pipeline architecture: Ingest → WhisperX → Alignment → Diarization → Tibetan Second-Pass → LLM Correction → Output
22
+ - WhisperX large-v3 primary transcription with automatic language detection
23
+ - Per-language forced alignment (wav2vec2 for en, ja, bo, sa)
24
+ - Speaker diarization via pyannote speaker-diarization-community-1
25
+ - Tibetan second-pass transcription using OpenPecha op-whisper_small-ft-v2 (dharma-trained)
26
+ - LLM post-correction via any OpenAI-compatible API (cloud or local)
27
+ - Output formats: JSON, SRT, VTT, TXT, review queue
28
+ - Corrections dictionary for deterministic pre-LLM fixes
29
+ - Idempotent manifest for batch processing
30
+ - VRAM management with model flushing between stages (6 GB consumer GPU support)
31
+ - `--device cpu` flag for CPU-only mode
32
+ - `--skip-llm` flag to skip LLM correction
33
+ - Environment-variable-based configuration (no hardcoded secrets)
34
+ - `src/` package layout with `pyproject.toml` (hatchling build backend)
35
+ - 54 unit tests covering all modules
36
+ - MIT license
37
+
38
+ ### Security
39
+ - Git history scrubbed of all API keys, URLs, and private paths
40
+ - All sensitive values read from environment variables
41
+ - No credentials in source code or commit history
@@ -0,0 +1,85 @@
1
+ # Contributing
2
+
3
+ ## Development Setup
4
+
5
+ ```bash
6
+ git clone https://github.com/guan-tends/dharma-transcribe.git
7
+ cd dharma-transcribe
8
+ python3 -m venv venv
9
+ source venv/bin/activate
10
+ pip install -e ".[dev]"
11
+ pre-commit install
12
+ ```
13
+
14
+ ## Workflow
15
+
16
+ 1. Create a branch: `git checkout -b feature/your-feature`
17
+ 2. Make changes. Commit early, commit often.
18
+ 3. Run checks before pushing:
19
+ ```bash
20
+ ruff check src tests
21
+ ruff format --check src tests
22
+ mypy src/dharma_transcribe
23
+ pytest
24
+ ```
25
+ 4. Push and open a PR.
26
+
27
+ ## Testing
28
+
29
+ ```bash
30
+ # Run all tests
31
+ pytest
32
+
33
+ # Run without GPU tests
34
+ pytest -m "not gpu"
35
+
36
+ # Run with coverage
37
+ pytest --cov=dharma_transcribe --cov-report=term-missing
38
+
39
+ # Run a single module
40
+ pytest tests/test_config.py
41
+ ```
42
+
43
+ ### Adding Test Fixtures
44
+
45
+ Place small audio samples in `tests/fixtures/`. A fixture should be:
46
+ - Short (5–30 seconds)
47
+ - Royalty-free or your own recording
48
+ - Named descriptively: `english_sample.wav`, `tibetan_sample.wav`
49
+
50
+ Generate a test WAV with ffmpeg:
51
+ ```bash
52
+ ffmpeg -f lavfi -i "sine=frequency=440:duration=5" -ar 16000 -ac 1 tests/fixtures/tone_5s.wav
53
+ ```
54
+
55
+ ## Code Style
56
+
57
+ - **Linter**: ruff (replaces flake8 + isort + black)
58
+ - **Formatter**: ruff format
59
+ - **Type checking**: mypy (non-strict, `ignore_missing_imports = true`)
60
+ - **Line length**: 100 characters
61
+ - **Python**: 3.10+ (uses `str | None` union syntax)
62
+
63
+ ## Project Structure
64
+
65
+ ```
66
+ src/dharma_transcribe/ # Package source
67
+ __init__.py # Package metadata
68
+ cli.py # CLI entry point (argparse)
69
+ pipeline.py # Stage orchestration
70
+ config.py # Environment-based configuration
71
+ ingest.py # Stage 1: audio extraction
72
+ transcribe.py # Stage 2: WhisperX ASR
73
+ align.py # Stage 3: forced alignment
74
+ diarize.py # Stage 4: speaker diarization
75
+ tibetan_second_pass.py # Stage 5: Tibetan re-transcription
76
+ llm_correct.py # Stage 6: LLM post-correction
77
+ output.py # Stage 7: multi-format output
78
+ gpu.py # GPU memory utilities
79
+ tests/ # Test suite
80
+ conftest.py # Shared fixtures
81
+ test_*.py # Unit and integration tests
82
+ fixtures/ # Test audio samples
83
+ pyproject.toml # Project metadata, tool config
84
+ .env.example # Environment variable template
85
+ ```
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 David Newman & Guan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,227 @@
1
+ Metadata-Version: 2.5
2
+ Name: dharma-transcribe
3
+ Version: 0.1.0
4
+ Summary: Headless multilingual transcription pipeline for Buddhist dharma teachings
5
+ Project-URL: Homepage, https://github.com/guan-tends/dharma-transcribe
6
+ Project-URL: Repository, https://github.com/guan-tends/dharma-transcribe
7
+ Author: David Newman & Guan
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: asr,buddhist,dharma,multilingual,sanskrit,speech-to-text,tibetan,transcription,whisper,whisperx
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
19
+ Classifier: Topic :: Religion
20
+ Requires-Python: >=3.10
21
+ Requires-Dist: openai>=1.0
22
+ Requires-Dist: requests>=2.28
23
+ Requires-Dist: tiktoken>=0.5
24
+ Requires-Dist: torch>=2.0
25
+ Requires-Dist: torchaudio>=2.0
26
+ Requires-Dist: transformers>=4.40
27
+ Requires-Dist: whisperx>=3.8.0
28
+ Provides-Extra: dev
29
+ Requires-Dist: mypy>=1.10; extra == 'dev'
30
+ Requires-Dist: pre-commit>=3.7; extra == 'dev'
31
+ Requires-Dist: pytest-cov>=4.0; extra == 'dev'
32
+ Requires-Dist: pytest-mock>=3.12; extra == 'dev'
33
+ Requires-Dist: pytest>=8.0; extra == 'dev'
34
+ Requires-Dist: ruff>=0.4; extra == 'dev'
35
+ Description-Content-Type: text/markdown
36
+
37
+ # Dharma Transcription Pipeline
38
+
39
+ A headless multilingual transcription pipeline designed for Buddhist dharma teachings. Handles mixed-language audio (Tibetan, Sanskrit, English, Japanese) with per-language forced alignment, speaker diarization, a Tibetan second-pass using dharma-trained models, and optional LLM post-correction.
40
+
41
+ **Privacy-first**: all processing runs locally. Sacred content never touches a cloud unless you explicitly configure LLM correction with a cloud API.
42
+
43
+ ## Architecture — 7 Stages
44
+
45
+ ```
46
+ Audio/Video → Ingest → WhisperX → Alignment → Diarization → Tibetan 2nd-Pass → LLM Correction → Output
47
+ ```
48
+
49
+ | Stage | Purpose | Model | GPU Memory |
50
+ |-------|---------|-------|-----------|
51
+ | 1. Ingest | ffmpeg extract to 16kHz mono WAV, checksum idempotency | ffmpeg | — |
52
+ | 2. Transcription | Primary ASR with auto language detection | WhisperX large-v3 (int8) | ~4 GB |
53
+ | 3. Alignment | Word-level timestamps per language | wav2vec2 (per-language) | ~1–2 GB |
54
+ | 4. Diarization | Speaker identification | pyannote diarization-community-1 | ~1 GB |
55
+ | 5. Tibetan 2nd-Pass | Re-transcribe Tibetan segments with dharma-trained model | OpenPecha op-whisper_small-ft-v2 | ~1 GB |
56
+ | 6. LLM Correction | Post-correction with dharma domain knowledge | Any OpenAI-compatible API | (cloud or local) |
57
+ | 7. Output | JSON, SRT, VTT, TXT, review queue | — | — |
58
+
59
+ Stages run serially with GPU memory flushing between each — designed for 6 GB consumer GPUs.
60
+
61
+ ## Setup
62
+
63
+ ### Prerequisites
64
+
65
+ - Python 3.10+
66
+ - ffmpeg + ffprobe (system install)
67
+ - NVIDIA GPU with CUDA (optional — `--device cpu` works for all stages)
68
+ - HuggingFace token (for diarization only — transcription works without it)
69
+
70
+ ### Installation
71
+
72
+ ```bash
73
+ # Clone
74
+ git clone https://github.com/guan-tends/dharma-transcribe.git
75
+ cd dharma-transcribe
76
+
77
+ # Create virtual environment
78
+ python3 -m venv venv
79
+ source venv/bin/activate
80
+
81
+ # Install with pip
82
+ pip install -e ".[dev]"
83
+
84
+ # Or install from PyPI (when published)
85
+ pip install dharma-transcribe[dev]
86
+ ```
87
+
88
+ ### GPU Setup (optional but recommended)
89
+
90
+ ```bash
91
+ # Install PyTorch with CUDA support (adjust cuXXX for your CUDA version)
92
+ pip install torch torchaudio --index-url https://download.pytorch.org/whl/cu128
93
+ ```
94
+
95
+ ### HuggingFace Token (for diarization)
96
+
97
+ The pipeline needs a HuggingFace token to download the pyannote diarization model. Without it, diarization is skipped — transcription still completes normally.
98
+
99
+ ```bash
100
+ # Get a token: https://huggingface.co/settings/tokens
101
+ # Also accept the model license: https://huggingface.co/pyannote/speaker-diarization-community-1
102
+
103
+ export HF_TOKEN=hf_your_token_here
104
+ ```
105
+
106
+ ### LLM Correction Configuration (optional)
107
+
108
+ Stage 6 uses any OpenAI-compatible API. Configure via environment variables:
109
+
110
+ ```bash
111
+ # Cloud API example (Synthetic.new)
112
+ export DHARMA_LLM_API_URL=https://api.example.com/v1
113
+ export DHARMA_LLM_API_KEY=your-key
114
+ export DHARMA_LLM_MODEL=hf:openai/gpt-oss-120b
115
+
116
+ # Local model example (Ollama)
117
+ ollama pull qwen3.5:9b
118
+ export DHARMA_LLM_API_URL=http://localhost:11434/v1
119
+ export DHARMA_LLM_API_KEY=ollama
120
+ export DHARMA_LLM_MODEL=qwen3.5:9b
121
+
122
+ # vLLM example (local GPU)
123
+ # Start vLLM server: vllm serve Qwen/Qwen3.5-9B
124
+ export DHARMA_LLM_API_URL=http://localhost:8000/v1
125
+ export DHARMA_LLM_API_KEY=none
126
+ export DHARMA_LLM_MODEL=Qwen/Qwen3.5-9B
127
+ ```
128
+
129
+ Or skip LLM correction entirely: `--skip-llm`
130
+
131
+ ## Usage
132
+
133
+ ```bash
134
+ # Single file
135
+ dharma-transcribe /path/to/teaching.mp4
136
+
137
+ # Directory (batch — finds all audio/video recursively)
138
+ dharma-transcribe /path/to/recordings/
139
+
140
+ # Skip LLM correction (ASR only, faster)
141
+ dharma-transcribe /path/to/teaching.mp4 --skip-llm
142
+
143
+ # CPU-only mode (no GPU required)
144
+ dharma-transcribe /path/to/teaching.mp4 --device cpu --skip-llm
145
+
146
+ # With explicit HF token
147
+ dharma-transcribe /path/to/teaching.mp4 --hf-token $HF_TOKEN
148
+ ```
149
+
150
+ ### CLI Flags
151
+
152
+ | Flag | Default | Description |
153
+ |------|---------|-------------|
154
+ | `input` (positional) | — | File or directory to process |
155
+ | `--source-dir` | — | Default source directory |
156
+ | `--skip-llm` | off | Skip LLM correction stage |
157
+ | `--hf-token` | `$HF_TOKEN` | HuggingFace token for diarization |
158
+ | `--device` | `cuda` | Compute device: `cuda` or `cpu` |
159
+
160
+ ### Environment Variables
161
+
162
+ See `.env.example` for the full list. Key variables:
163
+
164
+ | Variable | Default | Description |
165
+ |----------|---------|-------------|
166
+ | `DHARMA_LLM_API_URL` | (empty) | OpenAI-compatible API endpoint |
167
+ | `DHARMA_LLM_API_KEY` | (empty) | API key for LLM correction |
168
+ | `DHARMA_LLM_MODEL` | (empty) | Model name for LLM correction |
169
+ | `HF_TOKEN` | (empty) | HuggingFace token for diarization |
170
+ | `DHARMA_DEVICE` | `cuda` | Compute device |
171
+ | `DHARMA_WHISPER_MODEL` | `large-v3` | WhisperX model size |
172
+ | `DHARMA_OUTPUT_DIR` | `./output` | Output directory |
173
+ | `DHARMA_TIBETAN_MODEL` | `openpecha/op-whisper_small-ft-v2` | Tibetan second-pass model |
174
+
175
+ ## Output Formats
176
+
177
+ Each processed file generates:
178
+
179
+ - **JSON** — full structured transcript with word-level timestamps, speaker labels, confidence scores
180
+ - **SRT** — subtitle file with speaker labels
181
+ - **VTT** — WebVTT subtitles with speaker tags
182
+ - **TXT** — plain text reading copy
183
+ - **Review queue** — JSON of low-confidence segments for manual review
184
+
185
+ Outputs are written to `output/{json,srt,vtt,txt,review}/`.
186
+
187
+ ## Corrections Dictionary
188
+
189
+ `output/corrections/corrections.json` — case-insensitive string replacement applied before LLM correction. Grows from manual review.
190
+
191
+ ```json
192
+ {
193
+ "corrections": [
194
+ {"pattern": "bodichita", "replacement": "bodhicitta"},
195
+ {"pattern": "shun yata", "replacement": "shunyata"}
196
+ ]
197
+ }
198
+ ```
199
+
200
+ ## Idempotency
201
+
202
+ The manifest (`output/manifest.json`) tracks processed files. Re-running the pipeline skips completed files. Delete the manifest to reprocess everything.
203
+
204
+ ## VRAM Management
205
+
206
+ Designed for consumer GPUs (6 GB+). Stages run serially with `gc.collect()` + `torch.cuda.empty_cache()` between each. No CPU fallback needed — the GPU is flushed fully before loading the next model.
207
+
208
+ ## Why This Exists
209
+
210
+ Most transcription tools handle single languages. Dharma teachings commonly mix Tibetan, Sanskrit, English, and sometimes Japanese in a single recording. This pipeline:
211
+
212
+ 1. **Detects language per segment** — not per file
213
+ 2. **Aligns each language separately** — wav2vec2 models for bo, sa, en, ja
214
+ 3. **Re-transcribes Tibetan** — OpenPecha's model (trained on Garchen Rinpoche's teachings) often outperforms WhisperX on Tibetan
215
+ 4. **Corrects with dharma knowledge** — LLM post-correction knows bodhicitta from bodichita
216
+
217
+ ## License
218
+
219
+ MIT — see [LICENSE](LICENSE).
220
+
221
+ ## Support
222
+
223
+ If this pipeline helps preserve dharma teachings, consider supporting its continued development:
224
+
225
+ - **Solana**: `Eu8wQcW68TKMs1a6eqzZu8znzU52QLqQugAMG8uCD6y6`
226
+ - **EVM** (Ethereum / Base / Arbitrum / Optimism / Polygon): `0x2733ff7c865C56d565a99BE1DC11B81cc76850A5`
227
+ - **XRP Ledger**: `r4X6e7McAQj7e8vBCeued1RYu4mCJrREDG`