cluesurf-talk 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cluesurf_talk-2.0.0/PKG-INFO +85 -0
- cluesurf_talk-2.0.0/code/talk/__init__.py +61 -0
- cluesurf_talk-2.0.0/code/talk/base/modifiers.json +212 -0
- cluesurf_talk-2.0.0/code/talk/base/phones.json +1988 -0
- cluesurf_talk-2.0.0/code/talk/base/tokens.json +8218 -0
- cluesurf_talk-2.0.0/code/talk/py.typed +0 -0
- cluesurf_talk-2.0.0/code/talk/string/__init__.py +0 -0
- cluesurf_talk-2.0.0/code/talk/string/combine.py +11 -0
- cluesurf_talk-2.0.0/code/talk/string/convert.py +84 -0
- cluesurf_talk-2.0.0/code/talk/string/data.py +65 -0
- cluesurf_talk-2.0.0/code/talk/string/enumerate.py +102 -0
- cluesurf_talk-2.0.0/code/talk/string/runtime.py +104 -0
- cluesurf_talk-2.0.0/code/talk/string/sound.py +93 -0
- cluesurf_talk-2.0.0/code/talk/string/symbol.py +22 -0
- cluesurf_talk-2.0.0/code/talk/string/type.py +95 -0
- cluesurf_talk-2.0.0/code/talk/trie.py +238 -0
- cluesurf_talk-2.0.0/pyproject.toml +58 -0
- cluesurf_talk-2.0.0/readme.md +62 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cluesurf-talk
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Tokenized IPA for AI and NLP, a keyboard-friendly ASCII phonetic encoding that converts to and from IPA with one token per effective sound.
|
|
5
|
+
Keywords: nlp,text-to-speech,transliteration,linguistics,tts,phonology,ipa,phonetics,computational-linguistics,romanization,writing-systems,international-phonetic-alphabet,phonetic-alphabet,phonetic-transcription
|
|
6
|
+
Author: Lance Pollard, ClueSurf
|
|
7
|
+
Author-email: Lance Pollard <lancejpollard@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Project-URL: Homepage, https://github.com/cluesurf/talk
|
|
20
|
+
Project-URL: Repository, https://github.com/cluesurf/talk
|
|
21
|
+
Project-URL: Issues, https://github.com/cluesurf/talk/issues
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# talk
|
|
25
|
+
|
|
26
|
+
A phonetic encoding. Convert between:
|
|
27
|
+
|
|
28
|
+
- **IPA** — the International Phonetic Alphabet (`tʰa`)
|
|
29
|
+
- **talk** — a plain-ASCII spelling of the same sounds (`th~a`)
|
|
30
|
+
- **simple** — a readable, human-facing rendering
|
|
31
|
+
- **token** — one Hangul code point per sound, for fixed-width machine use
|
|
32
|
+
|
|
33
|
+
Everything is derived from three data files and a double-array trie scan.
|
|
34
|
+
**Zero runtime dependencies.**
|
|
35
|
+
|
|
36
|
+
This is the Python port of [`@cluesurf/talk`](https://github.com/cluesurf/talk).
|
|
37
|
+
Both ports share one source of truth for the phonetic data.
|
|
38
|
+
|
|
39
|
+
## Install
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
pip install cluesurf-talk
|
|
43
|
+
# or
|
|
44
|
+
uv add cluesurf-talk
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Use
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
import talk
|
|
51
|
+
|
|
52
|
+
talk.ipa_to_talk("tʰa") # "th~a"
|
|
53
|
+
talk.talk_to_ipa("th~a") # "tʰa"
|
|
54
|
+
talk.readable("th~a") # "tʰa"
|
|
55
|
+
talk.machine("th~a") # one Hangul code point per sound
|
|
56
|
+
|
|
57
|
+
# Break a talk string into sounds with their features.
|
|
58
|
+
for sound in talk.segment("th~a"):
|
|
59
|
+
print(sound.talk, sound.kind, sound.base and sound.base.talk,
|
|
60
|
+
[m.feature for m in sound.modifiers])
|
|
61
|
+
|
|
62
|
+
# The full canonical inventory.
|
|
63
|
+
talk.enumerate_sounds()
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## API
|
|
67
|
+
|
|
68
|
+
| Function | Description |
|
|
69
|
+
| --- | --- |
|
|
70
|
+
| `ipa_to_talk(text)` | IPA → talk |
|
|
71
|
+
| `talk_to_ipa(text)` | talk → IPA |
|
|
72
|
+
| `readable(text)` | talk → simple readable form |
|
|
73
|
+
| `machine(text)` | talk → one Hangul code point per sound |
|
|
74
|
+
| `machine_outputs(text)` | talk → list of per-sound code points |
|
|
75
|
+
| `segment(text)` / `tokenize(text)` | talk → list of `Sound` |
|
|
76
|
+
| `enumerate_sounds()` | the full canonical sound inventory |
|
|
77
|
+
| `combine(base_talk, mods)` | base + modifiers → canonical talk spelling |
|
|
78
|
+
|
|
79
|
+
## License
|
|
80
|
+
|
|
81
|
+
MIT
|
|
82
|
+
|
|
83
|
+
## ClueSurf
|
|
84
|
+
|
|
85
|
+
Part of the [ClueSurf](https://clue.surf) toolset.
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""talk: a phonetic encoding. IPA <-> talk (ascii) <-> simple (readable)
|
|
2
|
+
<-> token (one Hangul code point per sound).
|
|
3
|
+
|
|
4
|
+
Everything is derived from three data files (in base/) and a double-array
|
|
5
|
+
trie scan. No runtime dependencies.
|
|
6
|
+
|
|
7
|
+
import talk
|
|
8
|
+
|
|
9
|
+
talk.ipa_to_talk("tʰa") # -> "th~a"
|
|
10
|
+
talk.talk_to_ipa("th~a") # -> "tʰa"
|
|
11
|
+
talk.readable("th~a") # -> "tʰa"
|
|
12
|
+
talk.machine("th~a") # -> one Hangul code point per sound
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from .string.combine import combine
|
|
18
|
+
from .string.convert import (
|
|
19
|
+
ipa_to_talk,
|
|
20
|
+
machine,
|
|
21
|
+
machine_outputs,
|
|
22
|
+
readable,
|
|
23
|
+
talk_to_ipa,
|
|
24
|
+
tokenize,
|
|
25
|
+
)
|
|
26
|
+
from .string.enumerate import enumerate_sounds
|
|
27
|
+
from .string.sound import segment
|
|
28
|
+
from .string.type import (
|
|
29
|
+
Attaches,
|
|
30
|
+
Kind,
|
|
31
|
+
Modifier,
|
|
32
|
+
Phone,
|
|
33
|
+
Sound,
|
|
34
|
+
SoundInfo,
|
|
35
|
+
SymbolEntry,
|
|
36
|
+
TokenEntry,
|
|
37
|
+
Unit,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
__version__ = "2.0.0"
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"combine",
|
|
44
|
+
"ipa_to_talk",
|
|
45
|
+
"talk_to_ipa",
|
|
46
|
+
"tokenize",
|
|
47
|
+
"readable",
|
|
48
|
+
"machine",
|
|
49
|
+
"machine_outputs",
|
|
50
|
+
"segment",
|
|
51
|
+
"enumerate_sounds",
|
|
52
|
+
"Attaches",
|
|
53
|
+
"Kind",
|
|
54
|
+
"Modifier",
|
|
55
|
+
"Phone",
|
|
56
|
+
"Sound",
|
|
57
|
+
"SoundInfo",
|
|
58
|
+
"SymbolEntry",
|
|
59
|
+
"TokenEntry",
|
|
60
|
+
"Unit",
|
|
61
|
+
]
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"ipa": "̪",
|
|
4
|
+
"talk": "~",
|
|
5
|
+
"xsampa": "_d",
|
|
6
|
+
"simple": "",
|
|
7
|
+
"base": "consonant",
|
|
8
|
+
"feature": "dental",
|
|
9
|
+
"slot": "articulation",
|
|
10
|
+
"order": 10,
|
|
11
|
+
"attaches": {
|
|
12
|
+
"place": ["dental", "alveolar"]
|
|
13
|
+
}
|
|
14
|
+
},
|
|
15
|
+
{
|
|
16
|
+
"ipa": "ʲ",
|
|
17
|
+
"talk": "y~",
|
|
18
|
+
"xsampa": "_j",
|
|
19
|
+
"simple": "ʲ",
|
|
20
|
+
"base": "consonant",
|
|
21
|
+
"feature": "palatalized",
|
|
22
|
+
"slot": "tongue-body",
|
|
23
|
+
"order": 20,
|
|
24
|
+
"attaches": {
|
|
25
|
+
"notPlace": ["palatal", "pharyngeal-epiglottal", "glottal"]
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"ipa": "ˠ",
|
|
30
|
+
"talk": "G~",
|
|
31
|
+
"xsampa": "_G",
|
|
32
|
+
"simple": "ˠ",
|
|
33
|
+
"base": "consonant",
|
|
34
|
+
"feature": "velarized",
|
|
35
|
+
"slot": "tongue-body",
|
|
36
|
+
"order": 20,
|
|
37
|
+
"attaches": {
|
|
38
|
+
"place": ["dental", "alveolar"],
|
|
39
|
+
"notPlace": ["pharyngeal-epiglottal", "glottal"]
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"ipa": "ˤ",
|
|
44
|
+
"talk": "Q~",
|
|
45
|
+
"xsampa": "_?\\",
|
|
46
|
+
"simple": "ˤ",
|
|
47
|
+
"base": "consonant",
|
|
48
|
+
"feature": "pharyngealized",
|
|
49
|
+
"slot": "tongue-body",
|
|
50
|
+
"order": 20,
|
|
51
|
+
"attaches": {
|
|
52
|
+
"place": ["dental", "alveolar"],
|
|
53
|
+
"notPlace": ["pharyngeal-epiglottal", "glottal"]
|
|
54
|
+
}
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"ipa": "ʷ",
|
|
58
|
+
"talk": "w~",
|
|
59
|
+
"xsampa": "_w",
|
|
60
|
+
"simple": "ʷ",
|
|
61
|
+
"base": "consonant",
|
|
62
|
+
"feature": "labialized",
|
|
63
|
+
"slot": "labial",
|
|
64
|
+
"order": 30,
|
|
65
|
+
"attaches": {
|
|
66
|
+
"notPlace": [
|
|
67
|
+
"bilabial",
|
|
68
|
+
"labiodental",
|
|
69
|
+
"pharyngeal-epiglottal",
|
|
70
|
+
"glottal"
|
|
71
|
+
]
|
|
72
|
+
}
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"ipa": "ʰ",
|
|
76
|
+
"talk": "h~",
|
|
77
|
+
"xsampa": "_h",
|
|
78
|
+
"simple": "ʰ",
|
|
79
|
+
"base": "consonant",
|
|
80
|
+
"feature": "aspirated",
|
|
81
|
+
"slot": "laryngeal",
|
|
82
|
+
"order": 40,
|
|
83
|
+
"attaches": {
|
|
84
|
+
"manner": ["plosive"],
|
|
85
|
+
"notPlace": ["glottal"]
|
|
86
|
+
}
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"ipa": "ʼ",
|
|
90
|
+
"talk": "!",
|
|
91
|
+
"xsampa": "_>",
|
|
92
|
+
"simple": "ʼ",
|
|
93
|
+
"base": "consonant",
|
|
94
|
+
"feature": "ejective",
|
|
95
|
+
"slot": "laryngeal",
|
|
96
|
+
"order": 40,
|
|
97
|
+
"attaches": {
|
|
98
|
+
"manner": [
|
|
99
|
+
"plosive",
|
|
100
|
+
"fricative",
|
|
101
|
+
"sibilant-fricative",
|
|
102
|
+
"non-sibilant-fricative",
|
|
103
|
+
"lateral-fricative"
|
|
104
|
+
],
|
|
105
|
+
"voicing": ["voiceless"],
|
|
106
|
+
"notPlace": ["glottal"]
|
|
107
|
+
}
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"ipa": "̥",
|
|
111
|
+
"talk": "h!",
|
|
112
|
+
"xsampa": "_0",
|
|
113
|
+
"simple": "̥",
|
|
114
|
+
"base": "consonant",
|
|
115
|
+
"feature": "voiceless",
|
|
116
|
+
"slot": "phonation",
|
|
117
|
+
"order": 50,
|
|
118
|
+
"attaches": {
|
|
119
|
+
"manner": [
|
|
120
|
+
"nasal",
|
|
121
|
+
"approximant",
|
|
122
|
+
"fricative-approximant",
|
|
123
|
+
"trill",
|
|
124
|
+
"tap-flap",
|
|
125
|
+
"lateral-approximant",
|
|
126
|
+
"lateral-tap-flap"
|
|
127
|
+
],
|
|
128
|
+
"voicing": ["voiced"]
|
|
129
|
+
}
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
"ipa": "̃",
|
|
133
|
+
"talk": "&",
|
|
134
|
+
"xsampa": "_~",
|
|
135
|
+
"simple": "̃",
|
|
136
|
+
"base": "vowel",
|
|
137
|
+
"feature": "nasalized",
|
|
138
|
+
"slot": "nasal",
|
|
139
|
+
"order": 12
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
"ipa": "̯",
|
|
143
|
+
"talk": "@",
|
|
144
|
+
"xsampa": "_^",
|
|
145
|
+
"simple": "̯",
|
|
146
|
+
"base": "vowel",
|
|
147
|
+
"feature": "non-syllabic",
|
|
148
|
+
"slot": "syllabicity",
|
|
149
|
+
"order": 14
|
|
150
|
+
},
|
|
151
|
+
{
|
|
152
|
+
"ipa": "˩",
|
|
153
|
+
"talk": "--",
|
|
154
|
+
"xsampa": "_B",
|
|
155
|
+
"simple": "˩",
|
|
156
|
+
"base": "vowel",
|
|
157
|
+
"feature": "extra-low-tone",
|
|
158
|
+
"slot": "tone",
|
|
159
|
+
"order": 16
|
|
160
|
+
},
|
|
161
|
+
{
|
|
162
|
+
"ipa": "˨",
|
|
163
|
+
"talk": "-",
|
|
164
|
+
"xsampa": "_L",
|
|
165
|
+
"simple": "˨",
|
|
166
|
+
"base": "vowel",
|
|
167
|
+
"feature": "low-tone",
|
|
168
|
+
"slot": "tone",
|
|
169
|
+
"order": 16
|
|
170
|
+
},
|
|
171
|
+
{
|
|
172
|
+
"ipa": "˦",
|
|
173
|
+
"talk": "+",
|
|
174
|
+
"xsampa": "_H",
|
|
175
|
+
"simple": "˦",
|
|
176
|
+
"base": "vowel",
|
|
177
|
+
"feature": "high-tone",
|
|
178
|
+
"slot": "tone",
|
|
179
|
+
"order": 16
|
|
180
|
+
},
|
|
181
|
+
{
|
|
182
|
+
"ipa": "˥",
|
|
183
|
+
"talk": "++",
|
|
184
|
+
"xsampa": "_T",
|
|
185
|
+
"simple": "˥",
|
|
186
|
+
"base": "vowel",
|
|
187
|
+
"feature": "extra-high-tone",
|
|
188
|
+
"slot": "tone",
|
|
189
|
+
"order": 16
|
|
190
|
+
},
|
|
191
|
+
{
|
|
192
|
+
"ipa": "ː",
|
|
193
|
+
"talk": "_",
|
|
194
|
+
"xsampa": ":",
|
|
195
|
+
"simple": "ː",
|
|
196
|
+
"base": "vowel",
|
|
197
|
+
"feature": "long",
|
|
198
|
+
"slot": "duration",
|
|
199
|
+
"order": 18
|
|
200
|
+
},
|
|
201
|
+
{
|
|
202
|
+
"ipa": "ˈ",
|
|
203
|
+
"talk": "^",
|
|
204
|
+
"xsampa": "\"",
|
|
205
|
+
"simple": "ˈ",
|
|
206
|
+
"base": "vowel",
|
|
207
|
+
"feature": "stress",
|
|
208
|
+
"slot": "stress",
|
|
209
|
+
"order": 22,
|
|
210
|
+
"prefix": true
|
|
211
|
+
}
|
|
212
|
+
]
|