wordninja-enhanced 3.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,11 @@
1
+ Metadata-Version: 2.4
2
+ Name: wordninja-enhanced
3
+ Version: 3.0.0
4
+ Summary: Probabilistically split concatenated words. Now with more functionality and languages!
5
+ Author: Tim Lodemann
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/timminator/wordninja-enhanced
8
+ Project-URL: Repository, https://github.com/timminator/wordninja-enhanced
9
+ Project-URL: Issues, https://github.com/timminator/wordninja-enhanced/issues
10
+ Keywords: text-processing,nlp,language,segmentation
11
+ Description-Content-Type: text/markdown
@@ -0,0 +1,25 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "wordninja-enhanced"
7
+ version = "3.0.0"
8
+ description = "Probabilistically split concatenated words. Now with more functionality and languages!"
9
+ readme = "README.md"
10
+ license = {text="MIT"}
11
+ keywords = ["text-processing", "nlp", "language", "segmentation"]
12
+ authors = [
13
+ {name = "Tim Lodemann"}
14
+ ]
15
+
16
+ [project.urls]
17
+ Homepage = "https://github.com/timminator/wordninja-enhanced"
18
+ Repository = "https://github.com/timminator/wordninja-enhanced"
19
+ Issues = "https://github.com/timminator/wordninja-enhanced/issues"
20
+
21
+ [tool.setuptools]
22
+ packages = ["wordninja_enhanced"]
23
+
24
+ [tool.setuptools.package-data]
25
+ wordninja_enhanced = ["resources/*.gz"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,13 @@
1
+ from .wordninja import (
2
+ LanguageModel,
3
+ split,
4
+ candidates,
5
+ rejoin
6
+ )
7
+
8
+ __all__ = [
9
+ 'LanguageModel',
10
+ 'split',
11
+ 'candidates',
12
+ 'rejoin'
13
+ ]
@@ -0,0 +1,304 @@
1
+ import gzip, os, re
2
+ from math import log
3
+
4
+
5
+ __version__ = '3.0.0'
6
+
7
+
8
+ # Original idea on how to split strings is from:
9
+ # http://stackoverflow.com/a/11642687/2449774
10
+ # Thanks Generic Human!
11
+
12
+
13
+ NO_SPACE_BEFORE_BASE = {'.', ',', ';', ':', '!', '?', ')', ']', '}', '%', "'", "’", "s", '»', '›', '-'}
14
+ NO_SPACE_AFTER_BASE = {'(', '[', '{', '«', '‹', '¡', '¿', '-', '$', '€', '£'}
15
+
16
+
17
+ class LanguageModel(object):
18
+ """
19
+ Splits, analyzes, and rejoins text based on real-world word frequencies
20
+ for a specified pre-defined language or a custom dictionary file.
21
+ """
22
+ def __init__(self, language='en', word_file=None, add_words=None, blacklist=None, add_to_top=False):
23
+ """
24
+ Initializes a LanguageModel.
25
+
26
+ Args:
27
+ language (str): The language code ('en', 'de', etc.) OR 'custom'. Defaults to 'en'.
28
+ word_file (str, optional): Path to a custom gzipped word frequency file.
29
+ **Required if language is 'custom'**.
30
+ add_words (list, optional): Words to add to the dictionary.
31
+ blacklist (list, optional): Words to remove from the dictionary.
32
+ add_to_top (bool, optional): If True, add_wordsed words are made common.
33
+ """
34
+
35
+ if language == 'custom':
36
+ if not word_file or not os.path.exists(word_file):
37
+ raise ValueError("If language is 'custom', a valid 'word_file' path must be provided.")
38
+ else:
39
+ module_dir = os.path.dirname(os.path.abspath(__file__))
40
+
41
+ LANGUAGE_FILES = {
42
+ 'en': 'en_dict.txt.gz',
43
+ 'de': 'de_dict.txt.gz',
44
+ 'fr': 'fr_dict.txt.gz',
45
+ 'es': 'es_dict.txt.gz',
46
+ 'it': 'it_dict.txt.gz',
47
+ 'pt': 'pt_dict.txt.gz'
48
+ }
49
+
50
+ if language not in LANGUAGE_FILES:
51
+ raise ValueError(f"Language '{language}' not supported. Use 'custom' and provide a word_file.")
52
+
53
+ # For pre-defined languages, construct the path
54
+ word_file = os.path.join(module_dir, 'resources', LANGUAGE_FILES[language])
55
+
56
+ with gzip.open(word_file) as f:
57
+ words = f.read().decode('utf-8').split()
58
+
59
+ if blacklist:
60
+ blacklist_set = set(blacklist)
61
+ words = [word for word in words if word not in blacklist_set]
62
+
63
+ if add_words:
64
+ existing_words = set(words)
65
+ new_words = [w.lower() for w in add_words if w.lower() not in existing_words]
66
+ if add_to_top:
67
+ words = new_words + words
68
+ else:
69
+ words = words + new_words
70
+
71
+ self._wordcost = dict((k, log((i+1)*log(len(words)))) for i,k in enumerate(words))
72
+ self._maxword = max(len(x) for x in words)
73
+
74
+ self._no_space_before = NO_SPACE_BEFORE_BASE.copy()
75
+ self._no_space_after = NO_SPACE_AFTER_BASE.copy()
76
+
77
+ # Language-specific overrides and additions
78
+ if language == 'de':
79
+ for char in ['%', '-']:
80
+ self._no_space_before.discard(char)
81
+ for char in ['-', '$', '€', '£']:
82
+ self._no_space_after.discard(char)
83
+ elif language == 'fr':
84
+ for char in [':', ';', '!', '?', '»', '%']:
85
+ self._no_space_before.discard(char)
86
+ self._no_space_after.discard('«')
87
+ elif language == 'es':
88
+ self._no_space_before.discard('%')
89
+
90
+ self._SPLIT_RE = re.compile(r"\s+")
91
+ self._SPLIT_RE_FOR_CANDIDATES = re.compile(r"(\s+)")
92
+
93
+
94
+ def split(self, s):
95
+ """
96
+ Uses dynamic programming to infer the location of spaces in a string without spaces.
97
+ """
98
+ delimiters = self._SPLIT_RE.findall(s)
99
+ texts = self._SPLIT_RE.split(s)
100
+ new_texts = [self._split(x) for x in texts]
101
+
102
+ for i, delimiter in reversed(list(enumerate(delimiters))):
103
+ if delimiter:
104
+ new_texts.insert(i + 1, [delimiter])
105
+
106
+ return [item for sublist in new_texts for item in sublist if sublist]
107
+
108
+
109
+ def _split(self, s):
110
+ # Find the best match for the i first characters, assuming cost has
111
+ # been built for the i-1 first characters.
112
+ # Returns a pair (match_cost, match_length).
113
+ def best_match(i):
114
+ candidates = enumerate(reversed(cost[max(0, i-self._maxword):i]))
115
+ min_cost = float('inf')
116
+ best_k = 0
117
+
118
+ for k, c in candidates:
119
+ word = s[i-k-1:i].lower()
120
+ word_cost = self._wordcost.get(word)
121
+
122
+ if word_cost is None:
123
+ if len(word) == 1:
124
+ # Use a high (but not infinite) penalty for unknown single characters to allow continuation of the algorithm.
125
+ word_cost = 25
126
+ else:
127
+ # Use a a very high penalty for unknown longer words to force splitting into known words.
128
+ word_cost = 9e999
129
+
130
+ current_total_cost = c + word_cost
131
+
132
+ if current_total_cost < min_cost:
133
+ min_cost = current_total_cost
134
+ best_k = k + 1
135
+
136
+ return min_cost, best_k
137
+
138
+ # Build the cost array.
139
+ cost = [0]
140
+ for i in range(1,len(s)+1):
141
+ c,k = best_match(i)
142
+ cost.append(c)
143
+
144
+ # Backtrack to recover the minimal-cost string.
145
+ out = []
146
+ i = len(s)
147
+ while i>0:
148
+ c,k = best_match(i)
149
+ assert c == cost[i]
150
+ # Apostrophe and digit handling
151
+ newToken = True
152
+ if not s[i-k:i] == "'": # ignore a lone apostrophe
153
+ if len(out) > 0:
154
+ # re-attach split 's and split digits
155
+ if out[-1] == "'s" or (s[i-1].isdigit() and out[-1][0].isdigit()): # digit followed by digit
156
+ out[-1] = s[i-k:i] + out[-1] # combine current token with previous token
157
+ newToken = False
158
+
159
+ if newToken:
160
+ out.append(s[i-k:i])
161
+
162
+ i -= k
163
+
164
+ return reversed(out)
165
+
166
+
167
+ def _post_process_candidate(self, split: list) -> list:
168
+ if not split: return []
169
+ processed_split = [split[0]]
170
+ for i in range(1, len(split)):
171
+ token, prev_token = split[i], processed_split[-1]
172
+ should_merge = (token == "'s" and not prev_token.endswith("'")) or \
173
+ (token and prev_token and token[0].isdigit() and prev_token[-1].isdigit())
174
+ if should_merge:
175
+ processed_split[-1] += token
176
+ else:
177
+ processed_split.append(token)
178
+ return processed_split
179
+
180
+
181
+ def _beam_search_on_chunk(self, s_chunk: str, beam_width: int) -> list:
182
+ """
183
+ Runs beam search on a single contiguous string of letters/numbers.
184
+ """
185
+ dp = [[] for _ in range(len(s_chunk) + 1)]
186
+ dp[0] = [([], 0)]
187
+
188
+ for i in range(1, len(s_chunk) + 1):
189
+ candidates_for_i = []
190
+ for j in range(max(0, i - self._maxword), i):
191
+ word = s_chunk[j:i]
192
+ word_cost = self._wordcost.get(word)
193
+
194
+ if word_cost is None:
195
+ if len(word) == 1:
196
+ word_cost = 25 # High but manageable penalty for single unknown chars
197
+ else:
198
+ word_cost = 9e999 # Massive penalty for longer unknown words
199
+
200
+ if word_cost < 1e100:
201
+ for prev_split, prev_cost in dp[j]:
202
+ new_cost = prev_cost + word_cost
203
+ candidates_for_i.append((prev_split + [word], new_cost))
204
+
205
+ dp[i] = sorted(candidates_for_i, key=lambda x: x[1])[:beam_width]
206
+
207
+ return dp[len(s_chunk)]
208
+
209
+
210
+ def candidates(self, s: str, top_n=10) -> list:
211
+ """
212
+ The main public function. It orchestrates the splitting of a complex string
213
+ containing letters, numbers, spaces, and punctuation.
214
+ """
215
+ s = s.lower()
216
+ final_result_count = top_n
217
+ beam_width = max(top_n, 10)
218
+ beam = [([], 0)]
219
+
220
+ chunks = [c for c in self._SPLIT_RE_FOR_CANDIDATES.split(s) if c]
221
+
222
+ for chunk in chunks:
223
+ new_beam = []
224
+ if self._SPLIT_RE_FOR_CANDIDATES.fullmatch(chunk):
225
+ for prev_split, prev_cost in beam:
226
+ new_beam.append((prev_split + [chunk], prev_cost))
227
+ else:
228
+ chunk_candidates = self._beam_search_on_chunk(chunk, beam_width)
229
+ if not chunk_candidates:
230
+ chunk_candidates = [([chunk], 9e999)]
231
+
232
+ for prev_split, prev_cost in beam:
233
+ for chunk_split, chunk_cost in chunk_candidates:
234
+ new_beam.append((prev_split + chunk_split, prev_cost + chunk_cost))
235
+
236
+ beam = sorted(new_beam, key=lambda x: x[1])[:beam_width]
237
+
238
+ raw_candidates = [split for split, _ in beam]
239
+ processed_candidates = [self._post_process_candidate(s) for s in raw_candidates]
240
+
241
+ return processed_candidates[:final_result_count]
242
+
243
+
244
+ def rejoin(self, text_string: str) -> str:
245
+ """
246
+ Takes a string, splits it into words using the split() method, and rejoins it
247
+ with typographically correct, multi-language spacing.
248
+ """
249
+ tokens = self.split(text_string)
250
+
251
+ if not tokens:
252
+ return ""
253
+
254
+ result_parts = []
255
+ in_quotes = False
256
+
257
+ # TODO: Expand qoute rules also to single qoutes while keeping german words like "Elias' Haus" intact.
258
+ # Do not apply special spacing to a double qoute if a second one is never following
259
+ for i, token in enumerate(tokens):
260
+ is_opening_quote = token == '"' and not in_quotes
261
+
262
+ result_parts.append(token)
263
+
264
+ if token == '"':
265
+ in_quotes = not in_quotes
266
+
267
+ # Decide if a space is needed AFTER the current token by looking ahead
268
+ if i < len(tokens) - 1:
269
+ next_token = tokens[i+1]
270
+
271
+ add_space = True
272
+
273
+ # --- Apply Spacing Rules ---
274
+ # Rule 1: No space if it's an opening quote
275
+ if is_opening_quote:
276
+ add_space = False
277
+ # Rule 2: No space if the next token is a closing quote
278
+ elif next_token == '"' and in_quotes:
279
+ add_space = False
280
+ # Rule 3: Standard rules for punctuation and existing spaces
281
+ elif token in self._no_space_after or next_token in self._no_space_before:
282
+ add_space = False
283
+ elif token.isspace() or next_token.isspace():
284
+ add_space = False
285
+
286
+ if add_space:
287
+ result_parts.append(" ")
288
+
289
+ return "".join(result_parts)
290
+
291
+
292
+ DEFAULT_LANGUAGE_MODEL = LanguageModel(language='en')
293
+
294
+ def split(s):
295
+ """Splits a string using the default English model."""
296
+ return DEFAULT_LANGUAGE_MODEL.split(s)
297
+
298
+ def candidates(s, top_n=10):
299
+ """Finds candidates for a string using the default English model."""
300
+ return DEFAULT_LANGUAGE_MODEL.candidates(s, top_n=top_n)
301
+
302
+ def rejoin(s):
303
+ """Rejoins a string using the default English model's spacing rules."""
304
+ return DEFAULT_LANGUAGE_MODEL.rejoin(s)
@@ -0,0 +1,11 @@
1
+ Metadata-Version: 2.4
2
+ Name: wordninja-enhanced
3
+ Version: 3.0.0
4
+ Summary: Probabilistically split concatenated words. Now with more functionality and languages!
5
+ Author: Tim Lodemann
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/timminator/wordninja-enhanced
8
+ Project-URL: Repository, https://github.com/timminator/wordninja-enhanced
9
+ Project-URL: Issues, https://github.com/timminator/wordninja-enhanced/issues
10
+ Keywords: text-processing,nlp,language,segmentation
11
+ Description-Content-Type: text/markdown
@@ -0,0 +1,13 @@
1
+ pyproject.toml
2
+ wordninja_enhanced/__init__.py
3
+ wordninja_enhanced/wordninja.py
4
+ wordninja_enhanced.egg-info/PKG-INFO
5
+ wordninja_enhanced.egg-info/SOURCES.txt
6
+ wordninja_enhanced.egg-info/dependency_links.txt
7
+ wordninja_enhanced.egg-info/top_level.txt
8
+ wordninja_enhanced/resources/de_dict.txt.gz
9
+ wordninja_enhanced/resources/en_dict.txt.gz
10
+ wordninja_enhanced/resources/es_dict.txt.gz
11
+ wordninja_enhanced/resources/fr_dict.txt.gz
12
+ wordninja_enhanced/resources/it_dict.txt.gz
13
+ wordninja_enhanced/resources/pt_dict.txt.gz
@@ -0,0 +1 @@
1
+ wordninja_enhanced