lexphon 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lexphon/__init__.py ADDED
@@ -0,0 +1,32 @@
1
+ """Lexicon-driven phonemization on top of G2Lex."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ._version import __version__
6
+ from .engine import Phonemizer
7
+ from .errors import (
8
+ CatalogError,
9
+ DataIntegrityError,
10
+ LexiconNotInstalledError,
11
+ LexiconNotUsableError,
12
+ LexphonError,
13
+ UnknownWordError,
14
+ UnsupportedAlphabetError,
15
+ )
16
+ from .models import PhonemizationResult, PronunciationToken
17
+ from .store import DataStore
18
+
19
+ __all__ = [
20
+ "CatalogError",
21
+ "DataIntegrityError",
22
+ "DataStore",
23
+ "LexiconNotInstalledError",
24
+ "LexiconNotUsableError",
25
+ "LexphonError",
26
+ "PhonemizationResult",
27
+ "Phonemizer",
28
+ "PronunciationToken",
29
+ "UnknownWordError",
30
+ "UnsupportedAlphabetError",
31
+ "__version__",
32
+ ]
lexphon/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
lexphon/_version.py ADDED
@@ -0,0 +1,72 @@
1
+ """Small dependency-free Git-derived version helper used by setuptools."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ import subprocess
8
+ from email.parser import Parser
9
+ from pathlib import Path
10
+
11
+ _FALLBACK = "0.1.0"
12
+ _TAG = re.compile(
13
+ r"^v?(?P<base>\d+\.\d+\.\d+)(?:-(?P<count>\d+)-g(?P<sha>[0-9a-f]+))?(?P<dirty>-dirty)?$"
14
+ )
15
+
16
+
17
+ def _sdist_version(root: Path) -> str | None:
18
+ try:
19
+ metadata = Parser().parsestr(
20
+ (root / "PKG-INFO").read_text(encoding="utf-8"), headersonly=True
21
+ )
22
+ except (OSError, UnicodeError):
23
+ return None
24
+ if metadata.get("Name") != "lexphon":
25
+ return None
26
+ value = metadata.get("Version")
27
+ return value.strip() if value and value.strip() else None
28
+
29
+
30
+ def get_version() -> str:
31
+ override = os.environ.get("LEXPHON_VERSION")
32
+ if override:
33
+ return override
34
+ root = Path(__file__).resolve().parents[1]
35
+ try:
36
+ text = subprocess.check_output(
37
+ [
38
+ "git",
39
+ "-C",
40
+ str(root),
41
+ "describe",
42
+ "--tags",
43
+ "--long",
44
+ "--dirty",
45
+ "--match",
46
+ "v[0-9]*",
47
+ ],
48
+ stderr=subprocess.DEVNULL,
49
+ text=True,
50
+ timeout=2,
51
+ ).strip()
52
+ except (OSError, subprocess.SubprocessError):
53
+ text = ""
54
+ match = _TAG.match(text)
55
+ if not match:
56
+ return _sdist_version(root) or _FALLBACK
57
+ base = match.group("base")
58
+ count = int(match.group("count") or 0)
59
+ sha = match.group("sha")
60
+ dirty = bool(match.group("dirty"))
61
+ if count == 0 and not dirty:
62
+ return base
63
+ local = []
64
+ if sha:
65
+ local.append(f"g{sha}")
66
+ if dirty:
67
+ local.append("dirty")
68
+ version = f"{base}.post{count}" if count else f"{base}.dev0"
69
+ return version + ("+" + ".".join(local) if local else "")
70
+
71
+
72
+ __version__ = get_version()
lexphon/alphabets.py ADDED
@@ -0,0 +1,109 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ import unicodedata
5
+
6
+ from .errors import UnsupportedAlphabetError
7
+
8
+ _ARPA_CONSONANTS = {
9
+ "B": "b",
10
+ "CH": "tʃ",
11
+ "D": "d",
12
+ "DH": "ð",
13
+ "F": "f",
14
+ "G": "ɡ",
15
+ "HH": "h",
16
+ "JH": "dʒ",
17
+ "K": "k",
18
+ "L": "l",
19
+ "M": "m",
20
+ "N": "n",
21
+ "NG": "ŋ",
22
+ "P": "p",
23
+ "R": "ɹ",
24
+ "S": "s",
25
+ "SH": "ʃ",
26
+ "T": "t",
27
+ "TH": "θ",
28
+ "V": "v",
29
+ "W": "w",
30
+ "Y": "j",
31
+ "Z": "z",
32
+ "ZH": "ʒ",
33
+ }
34
+ _ARPA_VOWELS = {
35
+ "AA": "ɑ",
36
+ "AE": "æ",
37
+ "AH": "ʌ",
38
+ "AO": "ɔ",
39
+ "AW": "aʊ",
40
+ "AY": "aɪ",
41
+ "EH": "ɛ",
42
+ "ER": "ɝ",
43
+ "EY": "eɪ",
44
+ "IH": "ɪ",
45
+ "IY": "i",
46
+ "OW": "oʊ",
47
+ "OY": "ɔɪ",
48
+ "UH": "ʊ",
49
+ "UW": "u",
50
+ "AX": "ə",
51
+ "AXR": "ɚ",
52
+ "IX": "ɨ",
53
+ }
54
+ _ARPA_TOKEN = re.compile(r"^(?P<phoneme>[A-Z]+)(?P<stress>[012])?$")
55
+
56
+
57
+ def arpabet_to_ipa(value: str) -> str:
58
+ """Convert CMU-style ARPABET to deterministic broad IPA."""
59
+ if not isinstance(value, str) or not value.strip():
60
+ raise UnsupportedAlphabetError("ARPABET pronunciation must contain phone tokens")
61
+ segments: list[str] = []
62
+ vowel_indices: list[int] = []
63
+ stressed: list[tuple[int, str]] = []
64
+ for raw in value.split():
65
+ match = _ARPA_TOKEN.fullmatch(raw.upper())
66
+ if not match:
67
+ raise UnsupportedAlphabetError(f"invalid ARPABET token: {raw!r}")
68
+ symbol = match.group("phoneme")
69
+ stress = match.group("stress")
70
+ if symbol in _ARPA_VOWELS:
71
+ ipa = _ARPA_VOWELS[symbol]
72
+ if symbol == "AH" and stress == "0":
73
+ ipa = "ə"
74
+ elif symbol == "ER" and stress == "0":
75
+ ipa = "ɚ"
76
+ vowel_ordinal = len(vowel_indices)
77
+ if stress in {"1", "2"}:
78
+ stressed.append((vowel_ordinal, "ˈ" if stress == "1" else "ˌ"))
79
+ vowel_indices.append(len(segments))
80
+ segments.append(ipa)
81
+ elif symbol in _ARPA_CONSONANTS:
82
+ if stress is not None:
83
+ raise UnsupportedAlphabetError(f"stress on ARPABET consonant: {raw!r}")
84
+ segments.append(_ARPA_CONSONANTS[symbol])
85
+ else:
86
+ raise UnsupportedAlphabetError(f"unsupported ARPABET phoneme: {symbol}")
87
+
88
+ insertions: dict[int, list[str]] = {}
89
+ for vowel_ordinal, marker in stressed:
90
+ position = 0 if vowel_ordinal == 0 else vowel_indices[vowel_ordinal - 1] + 1
91
+ insertions.setdefault(position, []).append(marker)
92
+
93
+ output: list[str] = []
94
+ for index, segment in enumerate(segments):
95
+ output.extend(insertions.get(index, ()))
96
+ output.append(segment)
97
+ output.extend(insertions.get(len(segments), ()))
98
+ return unicodedata.normalize("NFC", "".join(output))
99
+
100
+
101
+ def to_ipa(value: str, encoding: str) -> str:
102
+ if not isinstance(value, str) or not value:
103
+ raise UnsupportedAlphabetError("pronunciation must be a non-empty string")
104
+ key = encoding.casefold().replace("-", "")
105
+ if key in {"ipa", "unicodeipa"}:
106
+ return unicodedata.normalize("NFC", value)
107
+ if key in {"arpabet", "cmu", "cmudict"}:
108
+ return arpabet_to_ipa(value)
109
+ raise UnsupportedAlphabetError(f"unsupported pronunciation encoding: {encoding!r}")
lexphon/catalog.py ADDED
@@ -0,0 +1,181 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ import re
6
+ import urllib.request
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+ from typing import Any
10
+ from urllib.parse import urlparse
11
+
12
+ from .errors import CatalogError
13
+
14
+ DEFAULT_CATALOG_URL = (
15
+ "https://raw.githubusercontent.com/buchwandler/g2lex-data/main/catalog/catalog.json"
16
+ )
17
+ _HASH_RE = re.compile(r"^[0-9a-f]{64}$")
18
+ _ID_RE = re.compile(r"^[a-z]{2,3}(?:-[a-z0-9]{2,8})*:[a-z0-9][a-z0-9._-]*$")
19
+ _LOCALE_RE = re.compile(r"^[a-z]{2,3}(?:-[a-z0-9]{2,8})*$")
20
+ _SUPPORTED_KINDS = {"pronunciation", "membership"}
21
+ _SUPPORTED_ENCODINGS = {"ipa", "arpabet", "none"}
22
+
23
+
24
+ def _require_text(value: dict[str, Any], key: str, label: str) -> str:
25
+ result = value.get(key)
26
+ if not isinstance(result, str) or not result:
27
+ raise CatalogError(f"{label} field {key!r} must be a non-empty string")
28
+ return result
29
+
30
+
31
+ def _require_hash(value: object, label: str) -> str:
32
+ if not isinstance(value, str) or not _HASH_RE.fullmatch(value):
33
+ raise CatalogError(f"{label} must be a lowercase SHA-256 hash")
34
+ return value
35
+
36
+
37
+ def _require_size(value: object, label: str) -> int:
38
+ if not isinstance(value, int) or isinstance(value, bool) or value <= 0:
39
+ raise CatalogError(f"{label} must be a positive integer")
40
+ return value
41
+
42
+
43
+ def _validate_reference(value: object, label: str, *, require_size: bool) -> dict[str, Any]:
44
+ if not isinstance(value, dict):
45
+ raise CatalogError(f"artifact {label} must be an object")
46
+ url = _require_text(value, "url", f"artifact {label}")
47
+ parsed = urlparse(url)
48
+ if parsed.scheme not in {"http", "https", "file"} and not parsed.scheme == "":
49
+ raise CatalogError(f"artifact {label} has unsupported URL scheme: {parsed.scheme!r}")
50
+ _require_text(value, "name", f"artifact {label}")
51
+ if Path(value["name"]).name != value["name"] or value["name"] in {".", ".."}:
52
+ raise CatalogError(f"artifact {label} name must be a plain filename")
53
+ _require_hash(value.get("sha256"), f"artifact {label}.sha256")
54
+ if require_size and "size" not in value:
55
+ raise CatalogError(f"artifact {label} requires size")
56
+ if "size" in value:
57
+ _require_size(value["size"], f"artifact {label}.size")
58
+ return value
59
+
60
+
61
+ @dataclass(frozen=True, slots=True)
62
+ class CatalogArtifact:
63
+ id: str
64
+ language: str
65
+ name: str
66
+ display_name: str
67
+ kind: str
68
+ phoneme_encoding: str
69
+ data_version: str
70
+ release_tag: str
71
+ manifest: dict[str, Any]
72
+ asset: dict[str, Any]
73
+ source: dict[str, Any]
74
+
75
+ @classmethod
76
+ def from_dict(cls, value: dict[str, Any]) -> CatalogArtifact:
77
+ if not isinstance(value, dict):
78
+ raise CatalogError("catalog artifacts must be objects")
79
+ identifier = _require_text(value, "id", "artifact")
80
+ if not _ID_RE.fullmatch(identifier):
81
+ raise CatalogError(f"artifact id is not a valid logical ID: {identifier!r}")
82
+ language = _require_text(value, "language", f"artifact {identifier}")
83
+ normalized_language = language.casefold().replace("_", "-")
84
+ id_language = identifier.split(":", 1)[0]
85
+ if not _LOCALE_RE.fullmatch(normalized_language) or normalized_language != id_language:
86
+ raise CatalogError(f"artifact {identifier} has inconsistent language metadata")
87
+ name = _require_text(value, "name", f"artifact {identifier}")
88
+ if Path(name).name != name:
89
+ raise CatalogError(f"artifact {identifier} name must be a plain name")
90
+ kind = _require_text(value, "kind", f"artifact {identifier}")
91
+ if kind not in _SUPPORTED_KINDS:
92
+ raise CatalogError(f"artifact {identifier} has unsupported kind: {kind!r}")
93
+ encoding = _require_text(value, "phoneme_encoding", f"artifact {identifier}").casefold()
94
+ if encoding not in _SUPPORTED_ENCODINGS:
95
+ raise CatalogError(
96
+ f"artifact {identifier} has unsupported phoneme_encoding: {encoding!r}"
97
+ )
98
+ if kind == "membership" and encoding != "none":
99
+ raise CatalogError(f"membership artifact {identifier} must use phoneme_encoding 'none'")
100
+ if kind == "pronunciation" and encoding == "none":
101
+ raise CatalogError(
102
+ f"pronunciation artifact {identifier} cannot use phoneme_encoding 'none'"
103
+ )
104
+ data_version = _require_text(value, "data_version", f"artifact {identifier}")
105
+ release_tag = _require_text(value, "release_tag", f"artifact {identifier}")
106
+ manifest = _validate_reference(value.get("manifest"), "manifest", require_size=False)
107
+ asset = _validate_reference(value.get("asset"), "asset", require_size=True)
108
+ if asset.get("format") not in {None, "g2lex.lexicon.v1", "g2lex"}:
109
+ raise CatalogError(f"artifact {identifier} has unsupported asset format")
110
+ source = value.get("source", {})
111
+ if not isinstance(source, dict):
112
+ raise CatalogError(f"artifact {identifier} source must be an object")
113
+ return cls(
114
+ id=identifier,
115
+ language=language,
116
+ name=name,
117
+ display_name=str(value.get("display_name") or name),
118
+ kind=kind,
119
+ phoneme_encoding=encoding,
120
+ data_version=data_version,
121
+ release_tag=release_tag,
122
+ manifest=manifest,
123
+ asset=asset,
124
+ source=source,
125
+ )
126
+
127
+
128
+ @dataclass(frozen=True, slots=True)
129
+ class Catalog:
130
+ catalog_version: int
131
+ runtime_contract: str
132
+ artifacts: tuple[CatalogArtifact, ...]
133
+
134
+ @classmethod
135
+ def from_dict(cls, value: dict[str, Any]) -> Catalog:
136
+ if not isinstance(value, dict):
137
+ raise CatalogError("catalog root must be an object")
138
+ if value.get("catalog_version") != 1:
139
+ raise CatalogError(f"unsupported catalog_version: {value.get('catalog_version')!r}")
140
+ contract = value.get("runtime_contract")
141
+ if contract != "g2lex-data.catalog.v1":
142
+ raise CatalogError(f"unsupported runtime_contract: {contract!r}")
143
+ raw = value.get("artifacts")
144
+ if not isinstance(raw, list):
145
+ raise CatalogError("catalog artifacts must be a list")
146
+ artifacts = tuple(CatalogArtifact.from_dict(item) for item in raw)
147
+ ids = [item.id for item in artifacts]
148
+ if len(set(ids)) != len(ids):
149
+ raise CatalogError("catalog artifact ids must be unique")
150
+ return cls(1, contract, artifacts)
151
+
152
+ def artifact(self, identifier: str) -> CatalogArtifact:
153
+ for artifact in self.artifacts:
154
+ if artifact.id == identifier:
155
+ return artifact
156
+ raise CatalogError(f"unknown catalog artifact: {identifier}")
157
+
158
+ def for_language(self, language: str) -> tuple[CatalogArtifact, ...]:
159
+ key = language.casefold().replace("_", "-")
160
+ return tuple(
161
+ item for item in self.artifacts if item.language.casefold().replace("_", "-") == key
162
+ )
163
+
164
+
165
+ def _read_location(location: str) -> bytes:
166
+ parsed = urlparse(location)
167
+ if parsed.scheme in {"http", "https", "file"}:
168
+ with urllib.request.urlopen(location, timeout=30) as response:
169
+ return response.read()
170
+ return Path(location).expanduser().read_bytes()
171
+
172
+
173
+ def load_catalog(location: str | None = None) -> Catalog:
174
+ source = location or os.environ.get("LEXPHON_CATALOG_URL") or DEFAULT_CATALOG_URL
175
+ try:
176
+ value = json.loads(_read_location(source).decode("utf-8"))
177
+ return Catalog.from_dict(value)
178
+ except CatalogError:
179
+ raise
180
+ except Exception as exc: # keep a stable public error at the catalog boundary
181
+ raise CatalogError(f"unable to load catalog from {source}: {exc}") from exc
lexphon/cli.py ADDED
@@ -0,0 +1,147 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+
7
+ from . import __version__
8
+ from .catalog import load_catalog
9
+ from .engine import Phonemizer
10
+ from .errors import LexphonError
11
+ from .profiles import ProfileRegistry
12
+ from .store import DataStore
13
+
14
+
15
+ def _data_main(argv: list[str]) -> int:
16
+ parser = argparse.ArgumentParser(
17
+ prog="lexphon data", description="Manage installed G2Lex pronunciation data"
18
+ )
19
+ parser.add_argument("--catalog", help="catalog URL/path (default: g2lex-data main catalog)")
20
+ parser.add_argument("--data-home")
21
+ sub = parser.add_subparsers(dest="command", required=True)
22
+ p_install = sub.add_parser("install")
23
+ p_install.add_argument("id", nargs="+")
24
+ sub.add_parser("list")
25
+ p_verify = sub.add_parser("verify")
26
+ p_verify.add_argument("id", nargs="*")
27
+ p_remove = sub.add_parser("remove")
28
+ p_remove.add_argument("id", nargs="+")
29
+ p_info = sub.add_parser("info")
30
+ p_info.add_argument("id", nargs="+")
31
+ p_available = sub.add_parser("available")
32
+ p_available.add_argument("language", nargs="?")
33
+ args = parser.parse_args(argv)
34
+ store = DataStore(args.data_home)
35
+
36
+ if args.command == "install":
37
+ catalog = load_catalog(args.catalog)
38
+ for identifier in args.id:
39
+ path = store.install(catalog.artifact(identifier))
40
+ print(f"installed {identifier}: {path}")
41
+ elif args.command == "list":
42
+ for item in store.installed():
43
+ print(f"{item['id']}\t{item['data_version']}\t{item['phoneme_encoding']}")
44
+ elif args.command == "info":
45
+ for identifier in args.id:
46
+ print(json.dumps(store.metadata(identifier), ensure_ascii=False, sort_keys=True))
47
+ elif args.command == "verify":
48
+ ids = args.id or [item["id"] for item in store.installed()]
49
+ failed = False
50
+ for identifier in ids:
51
+ ok = store.verify(identifier)
52
+ print(f"{'OK' if ok else 'FAIL'}\t{identifier}")
53
+ failed |= not ok
54
+ return int(failed)
55
+ elif args.command == "remove":
56
+ for identifier in args.id:
57
+ store.remove(identifier)
58
+ print(f"removed {identifier}")
59
+ elif args.command == "available":
60
+ catalog = load_catalog(args.catalog)
61
+ artifacts = catalog.for_language(args.language) if args.language else catalog.artifacts
62
+ for item in artifacts:
63
+ print(f"{item.id}\t{item.language}\t{item.phoneme_encoding}\t{item.data_version}")
64
+ return 0
65
+
66
+
67
+ def _voices_main() -> int:
68
+ for profile in ProfileRegistry().profiles:
69
+ print(profile.language)
70
+ return 0
71
+
72
+
73
+ def _phonemize_main(argv: list[str]) -> int:
74
+ parser = argparse.ArgumentParser(prog="lexphon", description="Lexicon-driven IPA phonemizer")
75
+ parser.add_argument("--version", action="version", version=__version__)
76
+ parser.add_argument("-v", "--voice", "--language", dest="language", required=True)
77
+ parser.add_argument(
78
+ "--lexicon", action="append", dest="lexicons", help="installed lexicon id; repeatable"
79
+ )
80
+ parser.add_argument("--tag", help="selector tag for tagged G2Lex values")
81
+ parser.add_argument("--fallback", choices=["none", "espeak"], default="none")
82
+ parser.add_argument("--unknown", choices=["error", "keep", "skip"], default="error")
83
+ parser.add_argument("--punctuation", choices=["keep", "drop"], default="keep")
84
+ parser.add_argument("--json", action="store_true")
85
+ parser.add_argument("--data-home")
86
+ parser.add_argument("text", nargs="*")
87
+ args = parser.parse_args(argv)
88
+ text = " ".join(args.text) if args.text else sys.stdin.read().strip()
89
+ if not text:
90
+ parser.error("text argument or stdin is required")
91
+ with Phonemizer(
92
+ args.language,
93
+ lexicons=args.lexicons,
94
+ store=DataStore(args.data_home),
95
+ fallback=None if args.fallback == "none" else args.fallback,
96
+ ) as engine:
97
+ result = engine.phonemize_tokens(text, tag=args.tag)
98
+ if args.json:
99
+ print(
100
+ json.dumps(
101
+ {
102
+ "text": result.text,
103
+ "language": result.language,
104
+ "phonemes": result.render(
105
+ unknown=args.unknown, punctuation=args.punctuation
106
+ ),
107
+ "tokens": [
108
+ {
109
+ "text": token.text,
110
+ "original_token": token.text,
111
+ "pronunciation": token.pronunciation,
112
+ "source": token.source,
113
+ "alphabet": token.alphabet,
114
+ "known": token.known,
115
+ "lexicon_id": token.lexicon_id,
116
+ "matched_key": token.matched_key,
117
+ "source_encoding": token.source_encoding,
118
+ "variants": list(token.variants),
119
+ "selector_tag": token.selector_tag,
120
+ "punctuation": token.punctuation,
121
+ }
122
+ for token in result.tokens
123
+ ],
124
+ },
125
+ ensure_ascii=False,
126
+ )
127
+ )
128
+ else:
129
+ print(result.render(unknown=args.unknown, punctuation=args.punctuation))
130
+ return 0
131
+
132
+
133
+ def main(argv: list[str] | None = None) -> int:
134
+ args = list(sys.argv[1:] if argv is None else argv)
135
+ try:
136
+ if args and args[0] == "data":
137
+ return _data_main(args[1:])
138
+ if args and args[0] == "voices":
139
+ return _voices_main()
140
+ return _phonemize_main(args)
141
+ except LexphonError as exc:
142
+ print(f"lexphon: {exc}", file=sys.stderr)
143
+ return 2
144
+
145
+
146
+ if __name__ == "__main__":
147
+ raise SystemExit(main())