wordninja-enhanced 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wordninja_enhanced-3.0.0/PKG-INFO +11 -0
- wordninja_enhanced-3.0.0/pyproject.toml +25 -0
- wordninja_enhanced-3.0.0/setup.cfg +4 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/__init__.py +13 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/de_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/en_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/es_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/fr_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/it_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/resources/pt_dict.txt.gz +0 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced/wordninja.py +304 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced.egg-info/PKG-INFO +11 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced.egg-info/SOURCES.txt +13 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced.egg-info/dependency_links.txt +1 -0
- wordninja_enhanced-3.0.0/wordninja_enhanced.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wordninja-enhanced
|
|
3
|
+
Version: 3.0.0
|
|
4
|
+
Summary: Probabilistically split concatenated words. Now with more functionality and languages!
|
|
5
|
+
Author: Tim Lodemann
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/timminator/wordninja-enhanced
|
|
8
|
+
Project-URL: Repository, https://github.com/timminator/wordninja-enhanced
|
|
9
|
+
Project-URL: Issues, https://github.com/timminator/wordninja-enhanced/issues
|
|
10
|
+
Keywords: text-processing,nlp,language,segmentation
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "wordninja-enhanced"
|
|
7
|
+
version = "3.0.0"
|
|
8
|
+
description = "Probabilistically split concatenated words. Now with more functionality and languages!"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text="MIT"}
|
|
11
|
+
keywords = ["text-processing", "nlp", "language", "segmentation"]
|
|
12
|
+
authors = [
|
|
13
|
+
{name = "Tim Lodemann"}
|
|
14
|
+
]
|
|
15
|
+
|
|
16
|
+
[project.urls]
|
|
17
|
+
Homepage = "https://github.com/timminator/wordninja-enhanced"
|
|
18
|
+
Repository = "https://github.com/timminator/wordninja-enhanced"
|
|
19
|
+
Issues = "https://github.com/timminator/wordninja-enhanced/issues"
|
|
20
|
+
|
|
21
|
+
[tool.setuptools]
|
|
22
|
+
packages = ["wordninja_enhanced"]
|
|
23
|
+
|
|
24
|
+
[tool.setuptools.package-data]
|
|
25
|
+
wordninja_enhanced = ["resources/*.gz"]
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
import gzip, os, re
|
|
2
|
+
from math import log
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
__version__ = '3.0.0'
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
# Original idea on how to split strings is from:
|
|
9
|
+
# http://stackoverflow.com/a/11642687/2449774
|
|
10
|
+
# Thanks Generic Human!
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
NO_SPACE_BEFORE_BASE = {'.', ',', ';', ':', '!', '?', ')', ']', '}', '%', "'", "’", "s", '»', '›', '-'}
|
|
14
|
+
NO_SPACE_AFTER_BASE = {'(', '[', '{', '«', '‹', '¡', '¿', '-', '$', '€', '£'}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LanguageModel(object):
|
|
18
|
+
"""
|
|
19
|
+
Splits, analyzes, and rejoins text based on real-world word frequencies
|
|
20
|
+
for a specified pre-defined language or a custom dictionary file.
|
|
21
|
+
"""
|
|
22
|
+
def __init__(self, language='en', word_file=None, add_words=None, blacklist=None, add_to_top=False):
|
|
23
|
+
"""
|
|
24
|
+
Initializes a LanguageModel.
|
|
25
|
+
|
|
26
|
+
Args:
|
|
27
|
+
language (str): The language code ('en', 'de', etc.) OR 'custom'. Defaults to 'en'.
|
|
28
|
+
word_file (str, optional): Path to a custom gzipped word frequency file.
|
|
29
|
+
**Required if language is 'custom'**.
|
|
30
|
+
add_words (list, optional): Words to add to the dictionary.
|
|
31
|
+
blacklist (list, optional): Words to remove from the dictionary.
|
|
32
|
+
add_to_top (bool, optional): If True, add_wordsed words are made common.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
if language == 'custom':
|
|
36
|
+
if not word_file or not os.path.exists(word_file):
|
|
37
|
+
raise ValueError("If language is 'custom', a valid 'word_file' path must be provided.")
|
|
38
|
+
else:
|
|
39
|
+
module_dir = os.path.dirname(os.path.abspath(__file__))
|
|
40
|
+
|
|
41
|
+
LANGUAGE_FILES = {
|
|
42
|
+
'en': 'en_dict.txt.gz',
|
|
43
|
+
'de': 'de_dict.txt.gz',
|
|
44
|
+
'fr': 'fr_dict.txt.gz',
|
|
45
|
+
'es': 'es_dict.txt.gz',
|
|
46
|
+
'it': 'it_dict.txt.gz',
|
|
47
|
+
'pt': 'pt_dict.txt.gz'
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
if language not in LANGUAGE_FILES:
|
|
51
|
+
raise ValueError(f"Language '{language}' not supported. Use 'custom' and provide a word_file.")
|
|
52
|
+
|
|
53
|
+
# For pre-defined languages, construct the path
|
|
54
|
+
word_file = os.path.join(module_dir, 'resources', LANGUAGE_FILES[language])
|
|
55
|
+
|
|
56
|
+
with gzip.open(word_file) as f:
|
|
57
|
+
words = f.read().decode('utf-8').split()
|
|
58
|
+
|
|
59
|
+
if blacklist:
|
|
60
|
+
blacklist_set = set(blacklist)
|
|
61
|
+
words = [word for word in words if word not in blacklist_set]
|
|
62
|
+
|
|
63
|
+
if add_words:
|
|
64
|
+
existing_words = set(words)
|
|
65
|
+
new_words = [w.lower() for w in add_words if w.lower() not in existing_words]
|
|
66
|
+
if add_to_top:
|
|
67
|
+
words = new_words + words
|
|
68
|
+
else:
|
|
69
|
+
words = words + new_words
|
|
70
|
+
|
|
71
|
+
self._wordcost = dict((k, log((i+1)*log(len(words)))) for i,k in enumerate(words))
|
|
72
|
+
self._maxword = max(len(x) for x in words)
|
|
73
|
+
|
|
74
|
+
self._no_space_before = NO_SPACE_BEFORE_BASE.copy()
|
|
75
|
+
self._no_space_after = NO_SPACE_AFTER_BASE.copy()
|
|
76
|
+
|
|
77
|
+
# Language-specific overrides and additions
|
|
78
|
+
if language == 'de':
|
|
79
|
+
for char in ['%', '-']:
|
|
80
|
+
self._no_space_before.discard(char)
|
|
81
|
+
for char in ['-', '$', '€', '£']:
|
|
82
|
+
self._no_space_after.discard(char)
|
|
83
|
+
elif language == 'fr':
|
|
84
|
+
for char in [':', ';', '!', '?', '»', '%']:
|
|
85
|
+
self._no_space_before.discard(char)
|
|
86
|
+
self._no_space_after.discard('«')
|
|
87
|
+
elif language == 'es':
|
|
88
|
+
self._no_space_before.discard('%')
|
|
89
|
+
|
|
90
|
+
self._SPLIT_RE = re.compile(r"\s+")
|
|
91
|
+
self._SPLIT_RE_FOR_CANDIDATES = re.compile(r"(\s+)")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def split(self, s):
|
|
95
|
+
"""
|
|
96
|
+
Uses dynamic programming to infer the location of spaces in a string without spaces.
|
|
97
|
+
"""
|
|
98
|
+
delimiters = self._SPLIT_RE.findall(s)
|
|
99
|
+
texts = self._SPLIT_RE.split(s)
|
|
100
|
+
new_texts = [self._split(x) for x in texts]
|
|
101
|
+
|
|
102
|
+
for i, delimiter in reversed(list(enumerate(delimiters))):
|
|
103
|
+
if delimiter:
|
|
104
|
+
new_texts.insert(i + 1, [delimiter])
|
|
105
|
+
|
|
106
|
+
return [item for sublist in new_texts for item in sublist if sublist]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _split(self, s):
|
|
110
|
+
# Find the best match for the i first characters, assuming cost has
|
|
111
|
+
# been built for the i-1 first characters.
|
|
112
|
+
# Returns a pair (match_cost, match_length).
|
|
113
|
+
def best_match(i):
|
|
114
|
+
candidates = enumerate(reversed(cost[max(0, i-self._maxword):i]))
|
|
115
|
+
min_cost = float('inf')
|
|
116
|
+
best_k = 0
|
|
117
|
+
|
|
118
|
+
for k, c in candidates:
|
|
119
|
+
word = s[i-k-1:i].lower()
|
|
120
|
+
word_cost = self._wordcost.get(word)
|
|
121
|
+
|
|
122
|
+
if word_cost is None:
|
|
123
|
+
if len(word) == 1:
|
|
124
|
+
# Use a high (but not infinite) penalty for unknown single characters to allow continuation of the algorithm.
|
|
125
|
+
word_cost = 25
|
|
126
|
+
else:
|
|
127
|
+
# Use a a very high penalty for unknown longer words to force splitting into known words.
|
|
128
|
+
word_cost = 9e999
|
|
129
|
+
|
|
130
|
+
current_total_cost = c + word_cost
|
|
131
|
+
|
|
132
|
+
if current_total_cost < min_cost:
|
|
133
|
+
min_cost = current_total_cost
|
|
134
|
+
best_k = k + 1
|
|
135
|
+
|
|
136
|
+
return min_cost, best_k
|
|
137
|
+
|
|
138
|
+
# Build the cost array.
|
|
139
|
+
cost = [0]
|
|
140
|
+
for i in range(1,len(s)+1):
|
|
141
|
+
c,k = best_match(i)
|
|
142
|
+
cost.append(c)
|
|
143
|
+
|
|
144
|
+
# Backtrack to recover the minimal-cost string.
|
|
145
|
+
out = []
|
|
146
|
+
i = len(s)
|
|
147
|
+
while i>0:
|
|
148
|
+
c,k = best_match(i)
|
|
149
|
+
assert c == cost[i]
|
|
150
|
+
# Apostrophe and digit handling
|
|
151
|
+
newToken = True
|
|
152
|
+
if not s[i-k:i] == "'": # ignore a lone apostrophe
|
|
153
|
+
if len(out) > 0:
|
|
154
|
+
# re-attach split 's and split digits
|
|
155
|
+
if out[-1] == "'s" or (s[i-1].isdigit() and out[-1][0].isdigit()): # digit followed by digit
|
|
156
|
+
out[-1] = s[i-k:i] + out[-1] # combine current token with previous token
|
|
157
|
+
newToken = False
|
|
158
|
+
|
|
159
|
+
if newToken:
|
|
160
|
+
out.append(s[i-k:i])
|
|
161
|
+
|
|
162
|
+
i -= k
|
|
163
|
+
|
|
164
|
+
return reversed(out)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _post_process_candidate(self, split: list) -> list:
|
|
168
|
+
if not split: return []
|
|
169
|
+
processed_split = [split[0]]
|
|
170
|
+
for i in range(1, len(split)):
|
|
171
|
+
token, prev_token = split[i], processed_split[-1]
|
|
172
|
+
should_merge = (token == "'s" and not prev_token.endswith("'")) or \
|
|
173
|
+
(token and prev_token and token[0].isdigit() and prev_token[-1].isdigit())
|
|
174
|
+
if should_merge:
|
|
175
|
+
processed_split[-1] += token
|
|
176
|
+
else:
|
|
177
|
+
processed_split.append(token)
|
|
178
|
+
return processed_split
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _beam_search_on_chunk(self, s_chunk: str, beam_width: int) -> list:
|
|
182
|
+
"""
|
|
183
|
+
Runs beam search on a single contiguous string of letters/numbers.
|
|
184
|
+
"""
|
|
185
|
+
dp = [[] for _ in range(len(s_chunk) + 1)]
|
|
186
|
+
dp[0] = [([], 0)]
|
|
187
|
+
|
|
188
|
+
for i in range(1, len(s_chunk) + 1):
|
|
189
|
+
candidates_for_i = []
|
|
190
|
+
for j in range(max(0, i - self._maxword), i):
|
|
191
|
+
word = s_chunk[j:i]
|
|
192
|
+
word_cost = self._wordcost.get(word)
|
|
193
|
+
|
|
194
|
+
if word_cost is None:
|
|
195
|
+
if len(word) == 1:
|
|
196
|
+
word_cost = 25 # High but manageable penalty for single unknown chars
|
|
197
|
+
else:
|
|
198
|
+
word_cost = 9e999 # Massive penalty for longer unknown words
|
|
199
|
+
|
|
200
|
+
if word_cost < 1e100:
|
|
201
|
+
for prev_split, prev_cost in dp[j]:
|
|
202
|
+
new_cost = prev_cost + word_cost
|
|
203
|
+
candidates_for_i.append((prev_split + [word], new_cost))
|
|
204
|
+
|
|
205
|
+
dp[i] = sorted(candidates_for_i, key=lambda x: x[1])[:beam_width]
|
|
206
|
+
|
|
207
|
+
return dp[len(s_chunk)]
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def candidates(self, s: str, top_n=10) -> list:
|
|
211
|
+
"""
|
|
212
|
+
The main public function. It orchestrates the splitting of a complex string
|
|
213
|
+
containing letters, numbers, spaces, and punctuation.
|
|
214
|
+
"""
|
|
215
|
+
s = s.lower()
|
|
216
|
+
final_result_count = top_n
|
|
217
|
+
beam_width = max(top_n, 10)
|
|
218
|
+
beam = [([], 0)]
|
|
219
|
+
|
|
220
|
+
chunks = [c for c in self._SPLIT_RE_FOR_CANDIDATES.split(s) if c]
|
|
221
|
+
|
|
222
|
+
for chunk in chunks:
|
|
223
|
+
new_beam = []
|
|
224
|
+
if self._SPLIT_RE_FOR_CANDIDATES.fullmatch(chunk):
|
|
225
|
+
for prev_split, prev_cost in beam:
|
|
226
|
+
new_beam.append((prev_split + [chunk], prev_cost))
|
|
227
|
+
else:
|
|
228
|
+
chunk_candidates = self._beam_search_on_chunk(chunk, beam_width)
|
|
229
|
+
if not chunk_candidates:
|
|
230
|
+
chunk_candidates = [([chunk], 9e999)]
|
|
231
|
+
|
|
232
|
+
for prev_split, prev_cost in beam:
|
|
233
|
+
for chunk_split, chunk_cost in chunk_candidates:
|
|
234
|
+
new_beam.append((prev_split + chunk_split, prev_cost + chunk_cost))
|
|
235
|
+
|
|
236
|
+
beam = sorted(new_beam, key=lambda x: x[1])[:beam_width]
|
|
237
|
+
|
|
238
|
+
raw_candidates = [split for split, _ in beam]
|
|
239
|
+
processed_candidates = [self._post_process_candidate(s) for s in raw_candidates]
|
|
240
|
+
|
|
241
|
+
return processed_candidates[:final_result_count]
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def rejoin(self, text_string: str) -> str:
|
|
245
|
+
"""
|
|
246
|
+
Takes a string, splits it into words using the split() method, and rejoins it
|
|
247
|
+
with typographically correct, multi-language spacing.
|
|
248
|
+
"""
|
|
249
|
+
tokens = self.split(text_string)
|
|
250
|
+
|
|
251
|
+
if not tokens:
|
|
252
|
+
return ""
|
|
253
|
+
|
|
254
|
+
result_parts = []
|
|
255
|
+
in_quotes = False
|
|
256
|
+
|
|
257
|
+
# TODO: Expand qoute rules also to single qoutes while keeping german words like "Elias' Haus" intact.
|
|
258
|
+
# Do not apply special spacing to a double qoute if a second one is never following
|
|
259
|
+
for i, token in enumerate(tokens):
|
|
260
|
+
is_opening_quote = token == '"' and not in_quotes
|
|
261
|
+
|
|
262
|
+
result_parts.append(token)
|
|
263
|
+
|
|
264
|
+
if token == '"':
|
|
265
|
+
in_quotes = not in_quotes
|
|
266
|
+
|
|
267
|
+
# Decide if a space is needed AFTER the current token by looking ahead
|
|
268
|
+
if i < len(tokens) - 1:
|
|
269
|
+
next_token = tokens[i+1]
|
|
270
|
+
|
|
271
|
+
add_space = True
|
|
272
|
+
|
|
273
|
+
# --- Apply Spacing Rules ---
|
|
274
|
+
# Rule 1: No space if it's an opening quote
|
|
275
|
+
if is_opening_quote:
|
|
276
|
+
add_space = False
|
|
277
|
+
# Rule 2: No space if the next token is a closing quote
|
|
278
|
+
elif next_token == '"' and in_quotes:
|
|
279
|
+
add_space = False
|
|
280
|
+
# Rule 3: Standard rules for punctuation and existing spaces
|
|
281
|
+
elif token in self._no_space_after or next_token in self._no_space_before:
|
|
282
|
+
add_space = False
|
|
283
|
+
elif token.isspace() or next_token.isspace():
|
|
284
|
+
add_space = False
|
|
285
|
+
|
|
286
|
+
if add_space:
|
|
287
|
+
result_parts.append(" ")
|
|
288
|
+
|
|
289
|
+
return "".join(result_parts)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
DEFAULT_LANGUAGE_MODEL = LanguageModel(language='en')
|
|
293
|
+
|
|
294
|
+
def split(s):
|
|
295
|
+
"""Splits a string using the default English model."""
|
|
296
|
+
return DEFAULT_LANGUAGE_MODEL.split(s)
|
|
297
|
+
|
|
298
|
+
def candidates(s, top_n=10):
|
|
299
|
+
"""Finds candidates for a string using the default English model."""
|
|
300
|
+
return DEFAULT_LANGUAGE_MODEL.candidates(s, top_n=top_n)
|
|
301
|
+
|
|
302
|
+
def rejoin(s):
|
|
303
|
+
"""Rejoins a string using the default English model's spacing rules."""
|
|
304
|
+
return DEFAULT_LANGUAGE_MODEL.rejoin(s)
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wordninja-enhanced
|
|
3
|
+
Version: 3.0.0
|
|
4
|
+
Summary: Probabilistically split concatenated words. Now with more functionality and languages!
|
|
5
|
+
Author: Tim Lodemann
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/timminator/wordninja-enhanced
|
|
8
|
+
Project-URL: Repository, https://github.com/timminator/wordninja-enhanced
|
|
9
|
+
Project-URL: Issues, https://github.com/timminator/wordninja-enhanced/issues
|
|
10
|
+
Keywords: text-processing,nlp,language,segmentation
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
pyproject.toml
|
|
2
|
+
wordninja_enhanced/__init__.py
|
|
3
|
+
wordninja_enhanced/wordninja.py
|
|
4
|
+
wordninja_enhanced.egg-info/PKG-INFO
|
|
5
|
+
wordninja_enhanced.egg-info/SOURCES.txt
|
|
6
|
+
wordninja_enhanced.egg-info/dependency_links.txt
|
|
7
|
+
wordninja_enhanced.egg-info/top_level.txt
|
|
8
|
+
wordninja_enhanced/resources/de_dict.txt.gz
|
|
9
|
+
wordninja_enhanced/resources/en_dict.txt.gz
|
|
10
|
+
wordninja_enhanced/resources/es_dict.txt.gz
|
|
11
|
+
wordninja_enhanced/resources/fr_dict.txt.gz
|
|
12
|
+
wordninja_enhanced/resources/it_dict.txt.gz
|
|
13
|
+
wordninja_enhanced/resources/pt_dict.txt.gz
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
wordninja_enhanced
|