afrispeech-synth 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,23 @@
1
+ """afrispeech-synth — synthetic speech datasets for African languages.
2
+
3
+ from afrispeech_synth import RunConfig, run
4
+
5
+ config = RunConfig(language="twi", sources=["corpus:twi"])
6
+ config.select.max_sentences = 500
7
+ run(config)
8
+ """
9
+ from .config import RunConfig, SelectConfig, TTSConfig, PackageConfig, load, from_dict
10
+ from .lang import Language, LanguageNotSupported, resolve
11
+ from .normalise import Normaliser
12
+ from .pipeline import run, stage_sources, stage_select, stage_synthesise, stage_package
13
+ from .select import Selection
14
+
15
+ __version__ = "0.1.0"
16
+
17
+ __all__ = [
18
+ "RunConfig", "SelectConfig", "TTSConfig", "PackageConfig",
19
+ "load", "from_dict", "run",
20
+ "stage_sources", "stage_select", "stage_synthesise", "stage_package",
21
+ "Language", "LanguageNotSupported", "resolve", "Normaliser", "Selection",
22
+ "__version__",
23
+ ]
@@ -0,0 +1,268 @@
1
+ """Generate the dataset card.
2
+
3
+ Provenance is the whole point: a synthetic dataset is only trustworthy if a
4
+ reader can see which text it came from, how sentences were chosen, which G2P
5
+ produced the transcripts and which model and voices spoke them. All of that is
6
+ already in the run config, so the card is generated rather than hand-written.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import datetime
11
+ from typing import Optional
12
+
13
+ from .lang import Language
14
+ from .normalise import describe as describe_g2p
15
+
16
+ TEMPLATE = """---
17
+ language:
18
+ - {lang_code}
19
+ license: cc-by-4.0
20
+ task_categories:
21
+ - text-to-speech
22
+ - automatic-speech-recognition
23
+ tags:
24
+ - audio
25
+ - tts
26
+ - synthetic
27
+ - african-languages
28
+ - {lang_slug}
29
+ pretty_name: {title}
30
+ configs:
31
+ - config_name: default
32
+ data_files:
33
+ - split: train
34
+ path: data/*.parquet
35
+ ---
36
+
37
+ # {title}
38
+
39
+ Synthetic speech for **{lang_name}** ({lang_code}), generated with
40
+ [`afrispeech-synth`](https://github.com/AfriSpeech/afrispeech-synth).
41
+
42
+ {counts}
43
+
44
+ > **Synthetic data.** Every clip here was generated by {tts_credit}, not recorded
45
+ > from a speaker. It is meant to bootstrap and supplement TTS/ASR training for a
46
+ > low-resource language, not to replace recorded speech. The audio carries whatever
47
+ > accent and pronunciation the model has for this language — check a sample by ear
48
+ > before training on it, and see {terms_note}
49
+
50
+ ## How it was built
51
+
52
+ | Stage | What ran |
53
+ |---|---|
54
+ | Source text | {sources} |
55
+ | Selection | {selection} |
56
+ | Normalisation | {normalisation} |
57
+ | Synthesis | {synthesis} |
58
+
59
+ ### 1. Source text
60
+
61
+ {sources_detail}
62
+
63
+ ### 2. Sentence selection
64
+
65
+ {selection_detail}
66
+
67
+ ### 3. Normalisation
68
+
69
+ {normalisation_detail}
70
+
71
+ ### 4. Synthesis
72
+
73
+ Audio was generated with **{tts_model}** (backend `{tts_backend}`), voice(s)
74
+ **{voices}**, prompted with:
75
+
76
+ ```text
77
+ {prompt_example}
78
+ ```
79
+
80
+ ## Dataset structure
81
+
82
+ - `data/train-*.parquet` — audio bytes embedded inline (WAV, {sample_rate} Hz mono),
83
+ so the dataset viewer plays each clip next to its text.
84
+ - `metadata.jsonl` — one record per clip: `index`, `text`, `normalised_text`,
85
+ `voice`, `shard`, `file_name`.
86
+ - `sentences.txt` — the selected source sentences, one per line.
87
+
88
+ | Column | Description |
89
+ |---|---|
90
+ | `audio` | Generated speech, WAV @ {sample_rate} Hz mono |
91
+ | `text` | The original sentence |
92
+ | `normalised_text` | The transcript actually given to the TTS model |
93
+ | `voice` | Which voice spoke this clip |
94
+
95
+ ## Reproducing it
96
+
97
+ ```bash
98
+ pip install afrispeech-synth
99
+ afrispeech-synth run config.yaml
100
+ ```
101
+
102
+ ```yaml
103
+ {config_yaml}```
104
+
105
+ ## Acknowledgements and terms
106
+
107
+ The audio was generated by {tts_credit}. Clips are model output and are subject to
108
+ {terms_link} — check them before redistributing or training on this data.
109
+ Phonemisation is [africa-g2p](https://github.com/AfriSpeech/africa-g2p);
110
+ text and language metadata come from
111
+ [africa-corpus-builder](https://github.com/AfriSpeech/africa-corpus-builder) and
112
+ [afriso](https://github.com/AfriSpeech/afriso).
113
+
114
+ ## License
115
+
116
+ Transcripts and the dataset structure: CC-BY-4.0. The audio is model output — see the
117
+ terms above. Source text keeps the licence of its own corpus, linked above.
118
+
119
+ ---
120
+
121
+ Built with [afrispeech-synth](https://github.com/AfriSpeech/afrispeech-synth) ·
122
+ [africa-g2p](https://github.com/AfriSpeech/africa-g2p) ·
123
+ [africa-corpus-builder](https://github.com/AfriSpeech/africa-corpus-builder) ·
124
+ [afriso](https://github.com/AfriSpeech/afriso) — generated {date}.
125
+ """
126
+
127
+ PROVIDERS = {
128
+ "gemini": {
129
+ "credit": "**[Google Gemini TTS](https://ai.google.dev/gemini-api/docs/speech-generation)**",
130
+ "terms": "[Google's Gemini API terms](https://ai.google.dev/gemini-api/terms)",
131
+ },
132
+ }
133
+
134
+
135
+ def _provider(backend: str, model: str) -> dict:
136
+ known = PROVIDERS.get(backend)
137
+ if known:
138
+ return known
139
+ return {"credit": f"the `{backend}` TTS backend (`{model}`)",
140
+ "terms": "your TTS provider's terms"}
141
+
142
+
143
+ SOURCE_LINKS = {
144
+ "corpus": ("africa-corpus-builder",
145
+ "https://huggingface.co/datasets/AfriSpeech/africa-corpus"),
146
+ "hf": ("HuggingFace dataset", "https://huggingface.co/datasets/{target}"),
147
+ "file": ("local file", None),
148
+ }
149
+
150
+
151
+ def _source_detail(uri: str) -> str:
152
+ from .sources import _parse
153
+ scheme, target, column, qualifier = _parse(uri)
154
+ if scheme == "corpus":
155
+ cap = f", capped at {qualifier} sentences" if qualifier else ""
156
+ return (f"- **{target}** monolingual text from "
157
+ f"[`AfriSpeech/africa-corpus`](https://huggingface.co/datasets/AfriSpeech/africa-corpus) "
158
+ f"via [africa-corpus-builder](https://github.com/AfriSpeech/africa-corpus-builder){cap}.")
159
+ if scheme == "hf":
160
+ split = f", split `{qualifier}`" if qualifier else ", all splits"
161
+ return (f"- Column `{column}` of "
162
+ f"[`{target}`](https://huggingface.co/datasets/{target}){split}.")
163
+ return f"- Local file `{target}`" + (f", column `{column}`" if column else "") + "."
164
+
165
+
166
+ def _selection_detail(select_config, selection) -> str:
167
+ if select_config.cover == "none":
168
+ return ("No coverage selection — every sentence within the length filter was kept"
169
+ f" ({select_config.min_chars}–{select_config.max_chars} characters).")
170
+ unit = "phoneme" if select_config.cover == "phoneme" else "word"
171
+ detail = (
172
+ f"Greedy **set cover** over {unit} units: the smallest set of sentences such that every "
173
+ f"{unit} in the corpus appears at least once. This is what keeps a synthetic corpus "
174
+ f"small without leaving sounds unheard — every sentence dropped is an API call saved, "
175
+ f"and every {unit} kept is one the model gets to learn.\n\n"
176
+ f"Sentences were filtered to {select_config.min_chars}–{select_config.max_chars} characters"
177
+ )
178
+ if select_config.min_freq > 1:
179
+ detail += f", and only {unit}s occurring at least {select_config.min_freq} times were targeted"
180
+ detail += "."
181
+ if selection is not None and selection.total_units:
182
+ detail += (f"\n\n**{selection.covered:,} of {selection.total_units:,} {unit} units covered "
183
+ f"({selection.coverage:.1%}) by {len(selection.sentences):,} sentences.**")
184
+ return detail
185
+
186
+
187
+ def _normalisation_detail(language: Language, mode: str) -> str:
188
+ if mode == "none":
189
+ return "None — the raw source text was sent to the TTS model unchanged."
190
+ if mode == "universal":
191
+ return (
192
+ "Each sentence was rewritten in **universal graphemes** with "
193
+ "[`africa-g2p`](https://github.com/AfriSpeech/africa-g2p):\n\n"
194
+ "```python\n"
195
+ "from africa_g2p import GraphemeConverter, UNIVERSAL\n"
196
+ f"GraphemeConverter({language.g2p_code!r}, UNIVERSAL).convert(text)\n"
197
+ "```\n\n"
198
+ "Universal graphemes write every phoneme with the letter most African languages "
199
+ "use for it — `ɔ` becomes `o`, `ɛ` becomes `e` — so the same sound is spelled the "
200
+ "same way across languages, in plain letters rather than IPA symbols. The result "
201
+ "is stored as `normalised_text` and is the transcript the TTS model was given."
202
+ )
203
+ if mode == "ipa":
204
+ return (
205
+ "Each sentence was converted to **IPA** with "
206
+ "[`africa-g2p`](https://github.com/AfriSpeech/africa-g2p):\n\n"
207
+ "```python\n"
208
+ "from africa_g2p import AfricaPipeline\n"
209
+ f"AfricaPipeline(lang={language.g2p_code!r}, output='ipa').run(text)\n"
210
+ "```\n\n"
211
+ "IPA puts every language on one shared symbol inventory, which is what you want "
212
+ "when a single model is trained across several languages."
213
+ )
214
+ return (
215
+ "Each sentence was rewritten into **native-orthography phoneme units** with "
216
+ "[`africa-g2p`](https://github.com/AfriSpeech/africa-g2p):\n\n"
217
+ "```python\n"
218
+ "from africa_g2p import AfricaPipeline\n"
219
+ f"AfricaPipeline(lang={language.g2p_code!r}).run(text)\n"
220
+ "```\n\n"
221
+ "Multigraphs (`ny`, `kp`, `gb`, …) stay whole and the text stays inside the language's "
222
+ "own inventory, which trains TTS better than IPA for a single language. The result is "
223
+ "stored as `normalised_text` and is the transcript the TTS model was actually given."
224
+ )
225
+
226
+
227
+ def render(config, language: Language, clips: int, selection=None,
228
+ title: Optional[str] = None) -> str:
229
+ import yaml
230
+
231
+ from .synth import render_prompt
232
+
233
+ title = title or f"{language.name} Synthetic Speech"
234
+ sources = ", ".join(f"`{uri}`" for uri in config.sources) or "—"
235
+ counts = f"**{clips:,} clips** · {language.name} (`{language.code}`)"
236
+ if language.family:
237
+ counts += f" · {language.family}"
238
+
239
+ safe_config = config.to_dict()
240
+ safe_config["tts"].pop("api_key_env", None)
241
+
242
+ provider = _provider(config.tts.backend, config.tts.model)
243
+
244
+ return TEMPLATE.format(
245
+ tts_credit=provider["credit"],
246
+ terms_note=provider["terms"] + " for how the audio may be used.",
247
+ terms_link=provider["terms"],
248
+ lang_code=language.code,
249
+ lang_slug=language.name.lower().replace(" ", "-"),
250
+ lang_name=language.name,
251
+ title=title,
252
+ counts=counts,
253
+ sources=sources,
254
+ selection=(f"greedy set cover ({config.select.cover})"
255
+ if config.select.cover != "none" else "length filter only"),
256
+ normalisation=describe_g2p(language, config.normalise) or "none",
257
+ synthesis=f"{config.tts.backend} / {config.tts.model}",
258
+ sources_detail="\n".join(_source_detail(uri) for uri in config.sources) or "—",
259
+ selection_detail=_selection_detail(config.select, selection),
260
+ normalisation_detail=_normalisation_detail(language, config.normalise),
261
+ tts_model=config.tts.model,
262
+ tts_backend=config.tts.backend,
263
+ voices=", ".join(config.tts.voices),
264
+ prompt_example=render_prompt(config.tts, language, "<normalised_text>"),
265
+ sample_rate=config.tts.sample_rate,
266
+ config_yaml=yaml.safe_dump(safe_config, sort_keys=False, allow_unicode=True),
267
+ date=datetime.date.today().isoformat(),
268
+ )
@@ -0,0 +1,308 @@
1
+ """Command line interface.
2
+
3
+ afrispeech-synth run config.yaml # the whole pipeline
4
+ afrispeech-synth run config.yaml --dry-run # select sentences only
5
+ afrispeech-synth select --lang twi --source corpus:twi --max-sentences 2000
6
+ afrispeech-synth synth config.yaml # resume synthesis only
7
+ afrispeech-synth package config.yaml # rebuild parquet from the work dir
8
+ afrispeech-synth push config.yaml --repo AfriSpeech/twi-synthetic-speech
9
+ afrispeech-synth langs --search yor
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import argparse
14
+ import os
15
+ import sys
16
+
17
+ from . import __version__
18
+ from . import card as card_module
19
+ from . import config as config_module
20
+ from . import package as package_module
21
+ from . import pipeline
22
+ from . import publish as publish_module
23
+ from . import tts as tts_registry
24
+ from .lang import resolve
25
+ from .normalise import Normaliser
26
+
27
+
28
+ def _config_from_args(args) -> config_module.RunConfig:
29
+ """A config file, CLI flags, or both — flags always win."""
30
+ if getattr(args, "config", None):
31
+ config = config_module.load(args.config)
32
+ else:
33
+ config = config_module.RunConfig()
34
+
35
+ overrides = {}
36
+ for flag, target in [
37
+ ("lang", "language"), ("out", "out"), ("work", "work"),
38
+ ("normalise", "normalise"),
39
+ ("cover", "select.cover"), ("min_freq", "select.min_freq"),
40
+ ("max_sentences", "select.max_sentences"),
41
+ ("min_chars", "select.min_chars"), ("max_chars", "select.max_chars"),
42
+ ("backend", "tts.backend"), ("model", "tts.model"),
43
+ ("concurrency", "tts.concurrency"), ("rpm", "tts.rpm"),
44
+ ("repo", "package.push_to"),
45
+ ]:
46
+ value = getattr(args, flag, None)
47
+ if value is not None:
48
+ overrides[target] = value
49
+ if getattr(args, "voices", None):
50
+ overrides["tts.voices"] = [v.strip() for v in args.voices.split(",") if v.strip()]
51
+ if getattr(args, "source", None):
52
+ config.sources = list(args.source)
53
+ if getattr(args, "format", None):
54
+ overrides["package.formats"] = [f.strip() for f in args.format.split(",") if f.strip()]
55
+
56
+ config_module.apply_overrides(config, overrides)
57
+ if not config.sources and getattr(args, "_needs_sources", True):
58
+ config.sources = [f"corpus:{resolve(config.language).code}"]
59
+ return config
60
+
61
+
62
+ def cmd_run(args) -> int:
63
+ config = _config_from_args(args)
64
+ pipeline.run(config, resume=not args.no_resume, dry_run=args.dry_run)
65
+ return 0
66
+
67
+
68
+ def cmd_select(args) -> int:
69
+ config = _config_from_args(args)
70
+ language = resolve(config.language)
71
+ normaliser = Normaliser(language, config.normalise)
72
+ sentences = pipeline.stage_sources(config, language)
73
+ selection = pipeline.stage_select(config, language, sentences, normaliser)
74
+ pipeline._save_sentences(config, selection.sentences, selection)
75
+ out = os.path.join(config.work_dir, pipeline.SENTENCES_FILE)
76
+ print(f"\n{len(selection.sentences)} sentences -> {out}")
77
+ return 0
78
+
79
+
80
+ def cmd_synth(args) -> int:
81
+ config = _config_from_args(args)
82
+ language = resolve(config.language)
83
+ sentences = pipeline.load_sentences(config)
84
+ if sentences is None:
85
+ print(f"No {pipeline.SENTENCES_FILE} in {config.work_dir}. Run `select` first, "
86
+ f"or use `run` to do everything.", file=sys.stderr)
87
+ return 1
88
+ normaliser = Normaliser(language, config.normalise)
89
+ pipeline.stage_synthesise(config, language, sentences, normaliser,
90
+ resume=not args.no_resume)
91
+ return 0
92
+
93
+
94
+ def cmd_package(args) -> int:
95
+ config = _config_from_args(args)
96
+ language = resolve(config.language)
97
+ pipeline.stage_package(config, language)
98
+ return 0
99
+
100
+
101
+ def cmd_push(args) -> int:
102
+ config = _config_from_args(args)
103
+ repo = args.repo or config.package.push_to
104
+ if not repo:
105
+ print("No target repo. Pass --repo org/name or set package.push_to.", file=sys.stderr)
106
+ return 1
107
+ publish_module.push(config.out, repo, private=config.package.private)
108
+ return 0
109
+
110
+
111
+ def cmd_card(args) -> int:
112
+ config = _config_from_args(args)
113
+ language = resolve(config.language)
114
+ clips = len(package_module.Workspace(config.work_dir).records()) if \
115
+ os.path.exists(config.work_dir) else 0
116
+ print(card_module.render(config, language, clips))
117
+ return 0
118
+
119
+
120
+ STATUS_HELP = {
121
+ "ready": "text + G2P available — just name the language",
122
+ "bring text": "G2P available; supply text with a file: or hf: source",
123
+ "no g2p": "text available; run with --normalise none",
124
+ }
125
+
126
+
127
+ def cmd_langs(args) -> int:
128
+ from . import coverage
129
+
130
+ catalogue = coverage.load()
131
+ entries = catalogue.search(args.search) if args.search else catalogue.sorted()
132
+ if args.ready:
133
+ entries = [e for e in entries if e.ready]
134
+
135
+ for entry in entries:
136
+ print(f"{entry.code:<6} {entry.name[:34]:<35} {entry.status:<11} {entry.family}")
137
+
138
+ counts = catalogue.counts()
139
+ print(f"\n{len(entries)} shown.", file=sys.stderr)
140
+ print(f" {counts['ready']:>4} ready {STATUS_HELP['ready']}", file=sys.stderr)
141
+ print(f" {counts['g2p'] - counts['ready']:>4} bring text {STATUS_HELP['bring text']}",
142
+ file=sys.stderr)
143
+ print(f" {counts['text'] - counts['ready']:>4} no g2p {STATUS_HELP['no g2p']}",
144
+ file=sys.stderr)
145
+ if not counts["text"]:
146
+ print("\n (africa-corpus-builder not found, so no language shows as ready — "
147
+ "see the README to install it.)", file=sys.stderr)
148
+ return 0
149
+
150
+
151
+ def cmd_samples(args) -> int:
152
+ from . import samples as samples_module
153
+
154
+ config = _config_from_args(args)
155
+ codes = [c.strip() for c in args.langs.split(",") if c.strip()] if args.langs else None
156
+ built = samples_module.build(config, args.dir, codes=codes, limit=args.limit,
157
+ resume=not args.no_resume)
158
+ print(f"\n{len(built)} samples in {args.dir}. Next: afrispeech-synth space "
159
+ f"--dir {args.dir} --repo org/name")
160
+ return 0
161
+
162
+
163
+ def cmd_space(args) -> int:
164
+ from . import samples as samples_module
165
+ from . import space as space_module
166
+
167
+ built = samples_module.load_manifest(args.dir)
168
+ if not built:
169
+ print(f"No samples.json in {args.dir}. Run `samples` first.", file=sys.stderr)
170
+ return 1
171
+ space_module.build(built, args.dir, title=args.title)
172
+ if args.repo:
173
+ space_module.push(args.dir, args.repo, private=args.private)
174
+ else:
175
+ print(f"\nOpen {os.path.join(args.dir, 'index.html')} to preview, then re-run "
176
+ f"with --repo org/name to publish.")
177
+ return 0
178
+
179
+
180
+ def cmd_voices(args) -> int:
181
+ from .voices import GEMINI_VOICES
182
+
183
+ print("Gemini TTS voices — set any of these in `tts.voices`, "
184
+ "or several to rotate speakers:\n")
185
+ for name, character in GEMINI_VOICES.items():
186
+ print(f" {name:<16} {character}")
187
+ print(f"\n{len(GEMINI_VOICES)} voices. Example: --voices Zephyr,Kore,Sulafat",
188
+ file=sys.stderr)
189
+ return 0
190
+
191
+
192
+ def cmd_init(args) -> int:
193
+ language = resolve(args.lang)
194
+ config = config_module.RunConfig(
195
+ language=language.code,
196
+ sources=[f"corpus:{language.code}"],
197
+ out=args.out or f"out/{language.code}",
198
+ )
199
+ config.select.max_sentences = 2000
200
+ config.tts.context = f"speak in {language.name} accent"
201
+ path = args.config or f"{language.code}.yaml"
202
+ config_module.dump(config, path)
203
+ print(f"Wrote {path}. Edit it, then: afrispeech-synth run {path}")
204
+ return 0
205
+
206
+
207
+ def build_parser() -> argparse.ArgumentParser:
208
+ parser = argparse.ArgumentParser(
209
+ prog="afrispeech-synth",
210
+ description="Generate synthetic speech datasets for African languages.",
211
+ )
212
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
213
+ subparsers = parser.add_subparsers(dest="command", required=True)
214
+
215
+ def add_common(sub, with_config=True):
216
+ if with_config:
217
+ sub.add_argument("config", nargs="?", help="YAML run config")
218
+ sub.add_argument("--lang", help="Language name or code (e.g. Twi, twi, yor)")
219
+ sub.add_argument("--source", action="append",
220
+ help="Text source URI; repeatable (corpus:twi, hf:org/ds#col, file:x.txt)")
221
+ sub.add_argument("--out", help="Output directory")
222
+ sub.add_argument("--work", help="Work directory (default: <out>/work)")
223
+ sub.add_argument("--normalise", choices=["grapheme", "universal", "ipa", "none"])
224
+ sub.add_argument("--cover", choices=["phoneme", "word", "none"])
225
+ sub.add_argument("--min-freq", type=int)
226
+ sub.add_argument("--max-sentences", type=int)
227
+ sub.add_argument("--min-chars", type=int)
228
+ sub.add_argument("--max-chars", type=int)
229
+ sub.add_argument("--backend", choices=tts_registry.available())
230
+ sub.add_argument("--model")
231
+ sub.add_argument("--voices", help="Comma-separated voice names, round-robined")
232
+ sub.add_argument("--concurrency", type=int)
233
+ sub.add_argument("--rpm", type=int)
234
+ sub.add_argument("--format", help="Comma-separated: parquet,ljspeech")
235
+ sub.add_argument("--repo", help="HuggingFace dataset repo to push to")
236
+ return sub
237
+
238
+ run_parser = add_common(subparsers.add_parser("run", help="Run the whole pipeline"))
239
+ run_parser.add_argument("--dry-run", action="store_true",
240
+ help="Select sentences and stop, without calling the TTS API")
241
+ run_parser.add_argument("--no-resume", action="store_true",
242
+ help="Ignore existing work and start over")
243
+ run_parser.set_defaults(func=cmd_run)
244
+
245
+ add_common(subparsers.add_parser("select", help="Source + select sentences only")
246
+ ).set_defaults(func=cmd_select)
247
+
248
+ synth_parser = add_common(subparsers.add_parser("synth", help="Synthesise selected sentences"))
249
+ synth_parser.add_argument("--no-resume", action="store_true")
250
+ synth_parser.set_defaults(func=cmd_synth)
251
+
252
+ add_common(subparsers.add_parser("package", help="Build parquet + manifest + card")
253
+ ).set_defaults(func=cmd_package)
254
+ add_common(subparsers.add_parser("push", help="Upload a packaged dataset to the Hub")
255
+ ).set_defaults(func=cmd_push)
256
+ add_common(subparsers.add_parser("card", help="Print the dataset card")
257
+ ).set_defaults(func=cmd_card)
258
+
259
+ init_parser = subparsers.add_parser("init", help="Write a starter config for a language")
260
+ init_parser.add_argument("lang", help="Language name or code")
261
+ init_parser.add_argument("--config", help="Config path to write")
262
+ init_parser.add_argument("--out", help="Output directory to record in the config")
263
+ init_parser.set_defaults(func=cmd_init)
264
+
265
+ langs_parser = subparsers.add_parser(
266
+ "langs", help="List languages and what each one needs to run")
267
+ langs_parser.add_argument("--search", help="Filter by code or name")
268
+ langs_parser.add_argument("--ready", action="store_true",
269
+ help="Only languages that run with no text of your own")
270
+ langs_parser.set_defaults(func=cmd_langs)
271
+
272
+ voices_parser = subparsers.add_parser("voices", help="List the TTS voices you can choose")
273
+ voices_parser.set_defaults(func=cmd_voices)
274
+
275
+ samples_parser = add_common(
276
+ subparsers.add_parser("samples", help="Generate one sample clip per language"))
277
+ samples_parser.add_argument("--dir", default="space",
278
+ help="Where samples and the gallery are built (default: space/)")
279
+ samples_parser.add_argument("--langs", help="Comma-separated codes (default: every ready language)")
280
+ samples_parser.add_argument("--limit", type=int, help="Stop after this many languages")
281
+ samples_parser.add_argument("--no-resume", action="store_true")
282
+ samples_parser.set_defaults(func=cmd_samples)
283
+
284
+ space_parser = subparsers.add_parser(
285
+ "space", help="Build the samples gallery page, and optionally push it as a HF Space")
286
+ space_parser.add_argument("--dir", default="space", help="Directory holding samples.json")
287
+ space_parser.add_argument("--repo", help="HuggingFace Space to push to, e.g. AfriSpeech/samples")
288
+ space_parser.add_argument("--title", default="African Speech Samples")
289
+ space_parser.add_argument("--private", action="store_true")
290
+ space_parser.set_defaults(func=cmd_space)
291
+
292
+ return parser
293
+
294
+
295
+ def main(argv=None) -> int:
296
+ args = build_parser().parse_args(argv)
297
+ try:
298
+ return args.func(args)
299
+ except KeyboardInterrupt:
300
+ print("\nInterrupted. Re-run the same command to resume.", file=sys.stderr)
301
+ return 130
302
+ except Exception as exc:
303
+ print(f"error: {exc}", file=sys.stderr)
304
+ return 1
305
+
306
+
307
+ if __name__ == "__main__":
308
+ sys.exit(main())