CPVI 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- CPVI/CPVI.py +165 -0
- CPVI/__init__.py +5 -0
- CPVI/data/conjugations.json +212 -0
- CPVI/data/irregulars.json +8239 -0
- CPVI/errors.py +30 -0
- CPVI/inflection.py +118 -0
- CPVI/loader.py +14 -0
- CPVI/utils.py +139 -0
- cpvi-0.2.0.dist-info/METADATA +392 -0
- cpvi-0.2.0.dist-info/RECORD +13 -0
- cpvi-0.2.0.dist-info/WHEEL +5 -0
- cpvi-0.2.0.dist-info/licenses/LICENSE +17 -0
- cpvi-0.2.0.dist-info/top_level.txt +1 -0
CPVI/CPVI.py
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from functools import lru_cache
|
|
6
|
+
|
|
7
|
+
from .loader import load
|
|
8
|
+
from .errors import IPAError, normalize_ipa, validate_input
|
|
9
|
+
from .inflection import inflector
|
|
10
|
+
|
|
11
|
+
FP_PAST, FP_PRES = 'formal Persian past stem', 'formal Persian present stem'
|
|
12
|
+
IP_PAST, IP_PRES = 'informal Persian past stem', 'informal Persian present stem'
|
|
13
|
+
FA_PAST, FA_PRES = 'formal IPA past stem', 'formal IPA present stem'
|
|
14
|
+
IA_PAST, IA_PRES = 'informal IPA past stem', 'informal IPA present stem'
|
|
15
|
+
_STEM_FIELDS = (FP_PRES, FP_PAST, IP_PRES, IP_PAST)
|
|
16
|
+
|
|
17
|
+
# Alternative verbs (e.g. رساندن / رسان+د|ید): present stem ends in "ان".
|
|
18
|
+
_ALT_FA = re.compile(r'(\w{2,}ان)(?:ی?د)?(?:ن)?')
|
|
19
|
+
_ALT_IPA = re.compile(r'(\w{2,}ɒn)(?:i?d)?(?:æn)?')
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@lru_cache(maxsize=None)
|
|
23
|
+
def _irregular_index():
|
|
24
|
+
"""Map every infinitive / stem string to its entry key (first entry wins,
|
|
25
|
+
same precedence as the old linear scan)."""
|
|
26
|
+
index = {}
|
|
27
|
+
for key, entry in load('irregulars').items():
|
|
28
|
+
index.setdefault(key, key)
|
|
29
|
+
for field in _STEM_FIELDS:
|
|
30
|
+
value = entry[field]
|
|
31
|
+
for stem in (value if isinstance(value, list) else [value]):
|
|
32
|
+
if stem: # '' means "no informal form"
|
|
33
|
+
index.setdefault(stem, key)
|
|
34
|
+
return index
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def save_json(profile, path, indent=2, ensure_ascii=False):
|
|
38
|
+
"""
|
|
39
|
+
Write a profile (as returned by ``CPVI.profiling``) to a JSON file.
|
|
40
|
+
|
|
41
|
+
Parameters
|
|
42
|
+
----------
|
|
43
|
+
profile : dict
|
|
44
|
+
The profile to save.
|
|
45
|
+
path : str or pathlib.Path
|
|
46
|
+
Destination file. Missing parent directories are created and an
|
|
47
|
+
existing file is overwritten. A ``.json`` suffix is added if the path
|
|
48
|
+
has no suffix.
|
|
49
|
+
indent : int or None, optional
|
|
50
|
+
JSON indentation (``None`` gives a compact single line).
|
|
51
|
+
ensure_ascii : bool, optional
|
|
52
|
+
If False (default) Persian/IPA letters are written as-is (UTF-8);
|
|
53
|
+
if True they are written as ``\\uXXXX`` escapes.
|
|
54
|
+
|
|
55
|
+
Returns
|
|
56
|
+
-------
|
|
57
|
+
pathlib.Path
|
|
58
|
+
The path that was written.
|
|
59
|
+
"""
|
|
60
|
+
path = Path(path)
|
|
61
|
+
if not path.suffix:
|
|
62
|
+
path = path.with_suffix('.json')
|
|
63
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
64
|
+
with open(path, 'w', encoding='utf-8') as file:
|
|
65
|
+
json.dump(profile, file, ensure_ascii=ensure_ascii, indent=indent)
|
|
66
|
+
return path
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _strip_regular(word, past_suffix, gerund_suffix):
|
|
70
|
+
"""Regular verbs: gerund = stem + id + n, past = stem + id."""
|
|
71
|
+
if word.endswith(past_suffix + gerund_suffix):
|
|
72
|
+
return word[:-len(past_suffix + gerund_suffix)]
|
|
73
|
+
if word.endswith(past_suffix):
|
|
74
|
+
return word[:-len(past_suffix)]
|
|
75
|
+
return word
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class CPVI():
|
|
79
|
+
"""
|
|
80
|
+
Identify and inflect a Persian verb given its present stem, past stem or
|
|
81
|
+
gerund.
|
|
82
|
+
"""
|
|
83
|
+
IPA = {'b': 'ب', 'p': 'پ', 'f': 'ف', 'v': 'و', 't': ['ت', 'ط'], 'd': 'د',
|
|
84
|
+
's': ['س', 'ص', 'ث'], 'z': ['ز', 'ض', 'ظ', 'ذ'], 'ʃ': 'ش', 'ʒ': 'ژ',
|
|
85
|
+
'ʤ': 'ج', 'ʧ': 'چ', 'c': 'ک', 'ɟ': 'گ', 'x': 'خ', 'G': ['ق', 'غ'],
|
|
86
|
+
'h': ['ه', 'ح'], 'ʔ': ['ع', 'همزه'], 'm': 'م', 'n': 'ن', 'r': 'ر',
|
|
87
|
+
'l': 'ل', 'j': 'ی', 'ɒ': ['آ', 'ا'], 'u': 'او', 'i': 'ی',
|
|
88
|
+
'æ': 'فتحه', 'e': 'کسره', 'o': 'ضمه'}
|
|
89
|
+
|
|
90
|
+
@staticmethod
|
|
91
|
+
def profiling(word, API_form='', space='\u200c', save_to=None):
|
|
92
|
+
"""
|
|
93
|
+
Return the profile (properties + full paradigm) of a verb.
|
|
94
|
+
|
|
95
|
+
Parameters
|
|
96
|
+
----------
|
|
97
|
+
word : str
|
|
98
|
+
Persian present stem, past stem or gerund.
|
|
99
|
+
API_form : str, optional
|
|
100
|
+
The same word in Persian IPA ('Ɉ' is accepted for 'ɟ').
|
|
101
|
+
Ignored for irregular verbs (their IPA comes from the data file).
|
|
102
|
+
space : str, optional
|
|
103
|
+
'', ' ' or ZWNJ. Has no effect on the few hand-curated irregular
|
|
104
|
+
paradigms (see README).
|
|
105
|
+
save_to : str or pathlib.Path, optional
|
|
106
|
+
If given, the profile is also written to this JSON file
|
|
107
|
+
(see ``CPVI.save``). The profile is still returned.
|
|
108
|
+
"""
|
|
109
|
+
validate_input(word, API_form, space)
|
|
110
|
+
API_form = normalize_ipa(API_form)
|
|
111
|
+
|
|
112
|
+
key = _irregular_index().get(word)
|
|
113
|
+
if key is not None:
|
|
114
|
+
return CPVI._finish(inflector(load('irregulars')[key], space), save_to)
|
|
115
|
+
|
|
116
|
+
profile = {k: '' for k in load('irregulars')['دانستن']}
|
|
117
|
+
alternative = bool(_ALT_FA.fullmatch(word))
|
|
118
|
+
|
|
119
|
+
if alternative:
|
|
120
|
+
m = _ALT_FA.fullmatch(word)
|
|
121
|
+
pres = m.group(1)
|
|
122
|
+
profile.update({'regularity': 'Alternative', 'transitivity': 'transitive',
|
|
123
|
+
'lexical aspect': 'action', 'present dual': False,
|
|
124
|
+
'past dual': True})
|
|
125
|
+
profile[FP_PRES] = pres
|
|
126
|
+
profile[FP_PAST] = [f'{pres}د', f'{pres}ید']
|
|
127
|
+
profile[IP_PRES] = f'{pres[:-2]}ون'
|
|
128
|
+
profile[IP_PAST] = [f'{profile[IP_PRES]}د', f'{profile[IP_PRES]}ید']
|
|
129
|
+
else:
|
|
130
|
+
pres = _strip_regular(word, 'ید', 'ن')
|
|
131
|
+
profile.update({'regularity': 'Regular', 'transitivity': 'unknown',
|
|
132
|
+
'lexical aspect': 'unknown', 'present dual': False,
|
|
133
|
+
'past dual': False})
|
|
134
|
+
profile[FP_PRES] = pres
|
|
135
|
+
profile[FP_PAST] = pres + ('ئید' if pres[-1] in 'اوی' else 'ید')
|
|
136
|
+
profile[IP_PRES], profile[IP_PAST] = profile[FP_PRES], profile[FP_PAST]
|
|
137
|
+
|
|
138
|
+
if API_form:
|
|
139
|
+
if alternative:
|
|
140
|
+
m = _ALT_IPA.fullmatch(API_form)
|
|
141
|
+
if not m:
|
|
142
|
+
raise IPAError('"word" is an alternative verb but "API_form" '
|
|
143
|
+
'does not look like one (expected ...ɒn[i]d[æn])')
|
|
144
|
+
pres = m.group(1)
|
|
145
|
+
profile[FA_PRES] = pres
|
|
146
|
+
profile[FA_PAST] = [f'{pres}d', f'{pres}id']
|
|
147
|
+
profile[IA_PRES] = f'{pres[:-2]}un'
|
|
148
|
+
profile[IA_PAST] = [f'{profile[IA_PRES]}d', f'{profile[IA_PRES]}id']
|
|
149
|
+
else:
|
|
150
|
+
pres = _strip_regular(API_form, 'id', 'æn')
|
|
151
|
+
profile[FA_PRES] = pres
|
|
152
|
+
profile[FA_PAST] = pres + ('ʔid' if pres[-1] in 'æɒouie' else 'id')
|
|
153
|
+
profile[IA_PRES], profile[IA_PAST] = profile[FA_PRES], profile[FA_PAST]
|
|
154
|
+
return CPVI._finish(inflector(profile, space), save_to)
|
|
155
|
+
|
|
156
|
+
@staticmethod
|
|
157
|
+
def _finish(profile, save_to):
|
|
158
|
+
if save_to is not None:
|
|
159
|
+
save_json(profile, save_to)
|
|
160
|
+
return profile
|
|
161
|
+
|
|
162
|
+
@staticmethod
|
|
163
|
+
def save(profile, path, indent=2, ensure_ascii=False):
|
|
164
|
+
"""Save an existing profile to a JSON file; see ``save_json``."""
|
|
165
|
+
return save_json(profile, path, indent=indent, ensure_ascii=ensure_ascii)
|
CPVI/__init__.py
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
{
|
|
2
|
+
"subjective": {
|
|
3
|
+
"formal": {
|
|
4
|
+
"IPA": {
|
|
5
|
+
"present": {
|
|
6
|
+
"s1": "æm",
|
|
7
|
+
"s2": "i",
|
|
8
|
+
"s3": "æd",
|
|
9
|
+
"p1": "im",
|
|
10
|
+
"p2": "id",
|
|
11
|
+
"p3": "ænd"
|
|
12
|
+
},
|
|
13
|
+
"past": {
|
|
14
|
+
"s1": "æm",
|
|
15
|
+
"s2": "i",
|
|
16
|
+
"s3": "",
|
|
17
|
+
"p1": "im",
|
|
18
|
+
"p2": "id",
|
|
19
|
+
"p3": "ænd"
|
|
20
|
+
},
|
|
21
|
+
"imperative": {
|
|
22
|
+
"s1": "",
|
|
23
|
+
"s2": "",
|
|
24
|
+
"s3": "",
|
|
25
|
+
"p1": "",
|
|
26
|
+
"p2": "id",
|
|
27
|
+
"p3": ""
|
|
28
|
+
},
|
|
29
|
+
"perfect": {
|
|
30
|
+
"s1": "ʔæm",
|
|
31
|
+
"s2": "ʔi",
|
|
32
|
+
"s3": " ʔæst",
|
|
33
|
+
"p1": "ʔim",
|
|
34
|
+
"p2": "ʔid",
|
|
35
|
+
"p3": "ʔænd"
|
|
36
|
+
}
|
|
37
|
+
},
|
|
38
|
+
"Persian": {
|
|
39
|
+
"present": {
|
|
40
|
+
"s1": "م",
|
|
41
|
+
"s2": "ی",
|
|
42
|
+
"s3": "د",
|
|
43
|
+
"p1": "یم",
|
|
44
|
+
"p2": "ید",
|
|
45
|
+
"p3": "ند"
|
|
46
|
+
},
|
|
47
|
+
"past": {
|
|
48
|
+
"s1": "م",
|
|
49
|
+
"s2": "ی",
|
|
50
|
+
"s3": "",
|
|
51
|
+
"p1": "یم",
|
|
52
|
+
"p2": "ید",
|
|
53
|
+
"p3": "ند"
|
|
54
|
+
},
|
|
55
|
+
"imperative": {
|
|
56
|
+
"s1": "",
|
|
57
|
+
"s2": "",
|
|
58
|
+
"s3": "",
|
|
59
|
+
"p1": "",
|
|
60
|
+
"p2": "ید",
|
|
61
|
+
"p3": ""
|
|
62
|
+
},
|
|
63
|
+
"perfect": {
|
|
64
|
+
"s1": "ام",
|
|
65
|
+
"s2": "ای",
|
|
66
|
+
"s3": "است",
|
|
67
|
+
"p1": "ایم",
|
|
68
|
+
"p2": "اید",
|
|
69
|
+
"p3": "اند"
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
},
|
|
73
|
+
"informal": {
|
|
74
|
+
"IPA": {
|
|
75
|
+
"present": {
|
|
76
|
+
"s1": "æm",
|
|
77
|
+
"s2": "i",
|
|
78
|
+
"s3": "e",
|
|
79
|
+
"p1": "im",
|
|
80
|
+
"p2": [
|
|
81
|
+
"in",
|
|
82
|
+
"id"
|
|
83
|
+
],
|
|
84
|
+
"p3": "æn"
|
|
85
|
+
},
|
|
86
|
+
"past": {
|
|
87
|
+
"s1": "æm",
|
|
88
|
+
"s2": "i",
|
|
89
|
+
"s3": [
|
|
90
|
+
"",
|
|
91
|
+
"eʃ"
|
|
92
|
+
],
|
|
93
|
+
"p1": "im",
|
|
94
|
+
"p2": [
|
|
95
|
+
"in",
|
|
96
|
+
"id"
|
|
97
|
+
],
|
|
98
|
+
"p3": "æn"
|
|
99
|
+
},
|
|
100
|
+
"imperative": {
|
|
101
|
+
"s1": "",
|
|
102
|
+
"s2": "",
|
|
103
|
+
"s3": "",
|
|
104
|
+
"p1": "",
|
|
105
|
+
"p2": [
|
|
106
|
+
"in",
|
|
107
|
+
"id"
|
|
108
|
+
],
|
|
109
|
+
"p3": ""
|
|
110
|
+
},
|
|
111
|
+
"perfect": {
|
|
112
|
+
"s1": "æ:m",
|
|
113
|
+
"s2": "i:",
|
|
114
|
+
"s3": "e:",
|
|
115
|
+
"p1": "i:m",
|
|
116
|
+
"p2": [
|
|
117
|
+
"i:n",
|
|
118
|
+
"i:d"
|
|
119
|
+
],
|
|
120
|
+
"p3": "æ:n"
|
|
121
|
+
}
|
|
122
|
+
},
|
|
123
|
+
"Persian": {
|
|
124
|
+
"present": {
|
|
125
|
+
"s1": "م",
|
|
126
|
+
"s2": "ی",
|
|
127
|
+
"s3": "ه",
|
|
128
|
+
"p1": "یم",
|
|
129
|
+
"p2": [
|
|
130
|
+
"ین",
|
|
131
|
+
"ید"
|
|
132
|
+
],
|
|
133
|
+
"p3": "ن"
|
|
134
|
+
},
|
|
135
|
+
"past": {
|
|
136
|
+
"s1": "م",
|
|
137
|
+
"s2": "ی",
|
|
138
|
+
"s3": [
|
|
139
|
+
"",
|
|
140
|
+
"ش"
|
|
141
|
+
],
|
|
142
|
+
"p1": "یم",
|
|
143
|
+
"p2": [
|
|
144
|
+
"ین",
|
|
145
|
+
"ید"
|
|
146
|
+
],
|
|
147
|
+
"p3": "ن"
|
|
148
|
+
},
|
|
149
|
+
"imperative": {
|
|
150
|
+
"s1": "",
|
|
151
|
+
"s2": "",
|
|
152
|
+
"s3": "",
|
|
153
|
+
"p1": "",
|
|
154
|
+
"p2": [
|
|
155
|
+
"ین",
|
|
156
|
+
"ید"
|
|
157
|
+
],
|
|
158
|
+
"p3": ""
|
|
159
|
+
},
|
|
160
|
+
"perfect": {
|
|
161
|
+
"s1": "م",
|
|
162
|
+
"s2": "ی",
|
|
163
|
+
"s3": "ه",
|
|
164
|
+
"p1": "یم",
|
|
165
|
+
"p2": [
|
|
166
|
+
"ین",
|
|
167
|
+
"ید"
|
|
168
|
+
],
|
|
169
|
+
"p3": "ن"
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
},
|
|
174
|
+
"objective": {
|
|
175
|
+
"formal": {
|
|
176
|
+
"IPA": {
|
|
177
|
+
"s1": "æm",
|
|
178
|
+
"s2": "æt",
|
|
179
|
+
"s3": "æʃ",
|
|
180
|
+
"p1": "emɒn",
|
|
181
|
+
"p2": "etɒn",
|
|
182
|
+
"p3": "eʃɒn"
|
|
183
|
+
},
|
|
184
|
+
"Persian": {
|
|
185
|
+
"s1": "م",
|
|
186
|
+
"s2": "ت",
|
|
187
|
+
"s3": "ش",
|
|
188
|
+
"p1": "مان",
|
|
189
|
+
"p2": "تان",
|
|
190
|
+
"p3": "شان"
|
|
191
|
+
}
|
|
192
|
+
},
|
|
193
|
+
"informal": {
|
|
194
|
+
"IPA": {
|
|
195
|
+
"s1": "æm",
|
|
196
|
+
"s2": "et",
|
|
197
|
+
"s3": "eʃ",
|
|
198
|
+
"p1": "emun",
|
|
199
|
+
"p2": "etun",
|
|
200
|
+
"p3": "eʃun"
|
|
201
|
+
},
|
|
202
|
+
"Persian": {
|
|
203
|
+
"s1": "م",
|
|
204
|
+
"s2": "ت",
|
|
205
|
+
"s3": "ش",
|
|
206
|
+
"p1": "مون",
|
|
207
|
+
"p2": "تون",
|
|
208
|
+
"p3": "شون"
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
}
|