cluesurf-talk 2.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,85 @@
1
+ Metadata-Version: 2.4
2
+ Name: cluesurf-talk
3
+ Version: 2.0.0
4
+ Summary: Tokenized IPA for AI and NLP, a keyboard-friendly ASCII phonetic encoding that converts to and from IPA with one token per effective sound.
5
+ Keywords: nlp,text-to-speech,transliteration,linguistics,tts,phonology,ipa,phonetics,computational-linguistics,romanization,writing-systems,international-phonetic-alphabet,phonetic-alphabet,phonetic-transcription
6
+ Author: Lance Pollard, ClueSurf
7
+ Author-email: Lance Pollard <lancejpollard@gmail.com>
8
+ License-Expression: MIT
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Programming Language :: Python :: 3.9
11
+ Classifier: Programming Language :: Python :: 3.10
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Topic :: Text Processing :: Linguistic
18
+ Requires-Python: >=3.9
19
+ Project-URL: Homepage, https://github.com/cluesurf/talk
20
+ Project-URL: Repository, https://github.com/cluesurf/talk
21
+ Project-URL: Issues, https://github.com/cluesurf/talk/issues
22
+ Description-Content-Type: text/markdown
23
+
24
+ # talk
25
+
26
+ A phonetic encoding. Convert between:
27
+
28
+ - **IPA** — the International Phonetic Alphabet (`tʰa`)
29
+ - **talk** — a plain-ASCII spelling of the same sounds (`th~a`)
30
+ - **simple** — a readable, human-facing rendering
31
+ - **token** — one Hangul code point per sound, for fixed-width machine use
32
+
33
+ Everything is derived from three data files and a double-array trie scan.
34
+ **Zero runtime dependencies.**
35
+
36
+ This is the Python port of [`@cluesurf/talk`](https://github.com/cluesurf/talk).
37
+ Both ports share one source of truth for the phonetic data.
38
+
39
+ ## Install
40
+
41
+ ```bash
42
+ pip install cluesurf-talk
43
+ # or
44
+ uv add cluesurf-talk
45
+ ```
46
+
47
+ ## Use
48
+
49
+ ```python
50
+ import talk
51
+
52
+ talk.ipa_to_talk("tʰa") # "th~a"
53
+ talk.talk_to_ipa("th~a") # "tʰa"
54
+ talk.readable("th~a") # "tʰa"
55
+ talk.machine("th~a") # one Hangul code point per sound
56
+
57
+ # Break a talk string into sounds with their features.
58
+ for sound in talk.segment("th~a"):
59
+ print(sound.talk, sound.kind, sound.base and sound.base.talk,
60
+ [m.feature for m in sound.modifiers])
61
+
62
+ # The full canonical inventory.
63
+ talk.enumerate_sounds()
64
+ ```
65
+
66
+ ## API
67
+
68
+ | Function | Description |
69
+ | --- | --- |
70
+ | `ipa_to_talk(text)` | IPA → talk |
71
+ | `talk_to_ipa(text)` | talk → IPA |
72
+ | `readable(text)` | talk → simple readable form |
73
+ | `machine(text)` | talk → one Hangul code point per sound |
74
+ | `machine_outputs(text)` | talk → list of per-sound code points |
75
+ | `segment(text)` / `tokenize(text)` | talk → list of `Sound` |
76
+ | `enumerate_sounds()` | the full canonical sound inventory |
77
+ | `combine(base_talk, mods)` | base + modifiers → canonical talk spelling |
78
+
79
+ ## License
80
+
81
+ MIT
82
+
83
+ ## ClueSurf
84
+
85
+ Part of the [ClueSurf](https://clue.surf) toolset.
@@ -0,0 +1,61 @@
1
+ """talk: a phonetic encoding. IPA <-> talk (ascii) <-> simple (readable)
2
+ <-> token (one Hangul code point per sound).
3
+
4
+ Everything is derived from three data files (in base/) and a double-array
5
+ trie scan. No runtime dependencies.
6
+
7
+ import talk
8
+
9
+ talk.ipa_to_talk("tʰa") # -> "th~a"
10
+ talk.talk_to_ipa("th~a") # -> "tʰa"
11
+ talk.readable("th~a") # -> "tʰa"
12
+ talk.machine("th~a") # -> one Hangul code point per sound
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from .string.combine import combine
18
+ from .string.convert import (
19
+ ipa_to_talk,
20
+ machine,
21
+ machine_outputs,
22
+ readable,
23
+ talk_to_ipa,
24
+ tokenize,
25
+ )
26
+ from .string.enumerate import enumerate_sounds
27
+ from .string.sound import segment
28
+ from .string.type import (
29
+ Attaches,
30
+ Kind,
31
+ Modifier,
32
+ Phone,
33
+ Sound,
34
+ SoundInfo,
35
+ SymbolEntry,
36
+ TokenEntry,
37
+ Unit,
38
+ )
39
+
40
+ __version__ = "2.0.0"
41
+
42
+ __all__ = [
43
+ "combine",
44
+ "ipa_to_talk",
45
+ "talk_to_ipa",
46
+ "tokenize",
47
+ "readable",
48
+ "machine",
49
+ "machine_outputs",
50
+ "segment",
51
+ "enumerate_sounds",
52
+ "Attaches",
53
+ "Kind",
54
+ "Modifier",
55
+ "Phone",
56
+ "Sound",
57
+ "SoundInfo",
58
+ "SymbolEntry",
59
+ "TokenEntry",
60
+ "Unit",
61
+ ]
@@ -0,0 +1,212 @@
1
+ [
2
+ {
3
+ "ipa": "̪",
4
+ "talk": "~",
5
+ "xsampa": "_d",
6
+ "simple": "",
7
+ "base": "consonant",
8
+ "feature": "dental",
9
+ "slot": "articulation",
10
+ "order": 10,
11
+ "attaches": {
12
+ "place": ["dental", "alveolar"]
13
+ }
14
+ },
15
+ {
16
+ "ipa": "ʲ",
17
+ "talk": "y~",
18
+ "xsampa": "_j",
19
+ "simple": "ʲ",
20
+ "base": "consonant",
21
+ "feature": "palatalized",
22
+ "slot": "tongue-body",
23
+ "order": 20,
24
+ "attaches": {
25
+ "notPlace": ["palatal", "pharyngeal-epiglottal", "glottal"]
26
+ }
27
+ },
28
+ {
29
+ "ipa": "ˠ",
30
+ "talk": "G~",
31
+ "xsampa": "_G",
32
+ "simple": "ˠ",
33
+ "base": "consonant",
34
+ "feature": "velarized",
35
+ "slot": "tongue-body",
36
+ "order": 20,
37
+ "attaches": {
38
+ "place": ["dental", "alveolar"],
39
+ "notPlace": ["pharyngeal-epiglottal", "glottal"]
40
+ }
41
+ },
42
+ {
43
+ "ipa": "ˤ",
44
+ "talk": "Q~",
45
+ "xsampa": "_?\\",
46
+ "simple": "ˤ",
47
+ "base": "consonant",
48
+ "feature": "pharyngealized",
49
+ "slot": "tongue-body",
50
+ "order": 20,
51
+ "attaches": {
52
+ "place": ["dental", "alveolar"],
53
+ "notPlace": ["pharyngeal-epiglottal", "glottal"]
54
+ }
55
+ },
56
+ {
57
+ "ipa": "ʷ",
58
+ "talk": "w~",
59
+ "xsampa": "_w",
60
+ "simple": "ʷ",
61
+ "base": "consonant",
62
+ "feature": "labialized",
63
+ "slot": "labial",
64
+ "order": 30,
65
+ "attaches": {
66
+ "notPlace": [
67
+ "bilabial",
68
+ "labiodental",
69
+ "pharyngeal-epiglottal",
70
+ "glottal"
71
+ ]
72
+ }
73
+ },
74
+ {
75
+ "ipa": "ʰ",
76
+ "talk": "h~",
77
+ "xsampa": "_h",
78
+ "simple": "ʰ",
79
+ "base": "consonant",
80
+ "feature": "aspirated",
81
+ "slot": "laryngeal",
82
+ "order": 40,
83
+ "attaches": {
84
+ "manner": ["plosive"],
85
+ "notPlace": ["glottal"]
86
+ }
87
+ },
88
+ {
89
+ "ipa": "ʼ",
90
+ "talk": "!",
91
+ "xsampa": "_>",
92
+ "simple": "ʼ",
93
+ "base": "consonant",
94
+ "feature": "ejective",
95
+ "slot": "laryngeal",
96
+ "order": 40,
97
+ "attaches": {
98
+ "manner": [
99
+ "plosive",
100
+ "fricative",
101
+ "sibilant-fricative",
102
+ "non-sibilant-fricative",
103
+ "lateral-fricative"
104
+ ],
105
+ "voicing": ["voiceless"],
106
+ "notPlace": ["glottal"]
107
+ }
108
+ },
109
+ {
110
+ "ipa": "̥",
111
+ "talk": "h!",
112
+ "xsampa": "_0",
113
+ "simple": "̥",
114
+ "base": "consonant",
115
+ "feature": "voiceless",
116
+ "slot": "phonation",
117
+ "order": 50,
118
+ "attaches": {
119
+ "manner": [
120
+ "nasal",
121
+ "approximant",
122
+ "fricative-approximant",
123
+ "trill",
124
+ "tap-flap",
125
+ "lateral-approximant",
126
+ "lateral-tap-flap"
127
+ ],
128
+ "voicing": ["voiced"]
129
+ }
130
+ },
131
+ {
132
+ "ipa": "̃",
133
+ "talk": "&",
134
+ "xsampa": "_~",
135
+ "simple": "̃",
136
+ "base": "vowel",
137
+ "feature": "nasalized",
138
+ "slot": "nasal",
139
+ "order": 12
140
+ },
141
+ {
142
+ "ipa": "̯",
143
+ "talk": "@",
144
+ "xsampa": "_^",
145
+ "simple": "̯",
146
+ "base": "vowel",
147
+ "feature": "non-syllabic",
148
+ "slot": "syllabicity",
149
+ "order": 14
150
+ },
151
+ {
152
+ "ipa": "˩",
153
+ "talk": "--",
154
+ "xsampa": "_B",
155
+ "simple": "˩",
156
+ "base": "vowel",
157
+ "feature": "extra-low-tone",
158
+ "slot": "tone",
159
+ "order": 16
160
+ },
161
+ {
162
+ "ipa": "˨",
163
+ "talk": "-",
164
+ "xsampa": "_L",
165
+ "simple": "˨",
166
+ "base": "vowel",
167
+ "feature": "low-tone",
168
+ "slot": "tone",
169
+ "order": 16
170
+ },
171
+ {
172
+ "ipa": "˦",
173
+ "talk": "+",
174
+ "xsampa": "_H",
175
+ "simple": "˦",
176
+ "base": "vowel",
177
+ "feature": "high-tone",
178
+ "slot": "tone",
179
+ "order": 16
180
+ },
181
+ {
182
+ "ipa": "˥",
183
+ "talk": "++",
184
+ "xsampa": "_T",
185
+ "simple": "˥",
186
+ "base": "vowel",
187
+ "feature": "extra-high-tone",
188
+ "slot": "tone",
189
+ "order": 16
190
+ },
191
+ {
192
+ "ipa": "ː",
193
+ "talk": "_",
194
+ "xsampa": ":",
195
+ "simple": "ː",
196
+ "base": "vowel",
197
+ "feature": "long",
198
+ "slot": "duration",
199
+ "order": 18
200
+ },
201
+ {
202
+ "ipa": "ˈ",
203
+ "talk": "^",
204
+ "xsampa": "\"",
205
+ "simple": "ˈ",
206
+ "base": "vowel",
207
+ "feature": "stress",
208
+ "slot": "stress",
209
+ "order": 22,
210
+ "prefix": true
211
+ }
212
+ ]