KL-Py 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. KL_Py/KL_Py.py +4400 -0
  2. KL_Py/__init__.py +1 -0
  3. KL_Py/hindGui/__init__.py +16529 -0
  4. KL_Py/hindGui/_utils.py +140 -0
  5. KL_Py/hindGui/elements/base.py +1055 -0
  6. KL_Py/hindGui/elements/button.py +947 -0
  7. KL_Py/hindGui/elements/calendar.py +222 -0
  8. KL_Py/hindGui/elements/canvas.py +122 -0
  9. KL_Py/hindGui/elements/center.py +44 -0
  10. KL_Py/hindGui/elements/checkbox.py +240 -0
  11. KL_Py/hindGui/elements/column.py +443 -0
  12. KL_Py/hindGui/elements/combo.py +329 -0
  13. KL_Py/hindGui/elements/error.py +47 -0
  14. KL_Py/hindGui/elements/frame.py +273 -0
  15. KL_Py/hindGui/elements/graph.py +817 -0
  16. KL_Py/hindGui/elements/helpers.py +202 -0
  17. KL_Py/hindGui/elements/image.py +324 -0
  18. KL_Py/hindGui/elements/input.py +310 -0
  19. KL_Py/hindGui/elements/list_box.py +392 -0
  20. KL_Py/hindGui/elements/menu.py +192 -0
  21. KL_Py/hindGui/elements/multiline.py +685 -0
  22. KL_Py/hindGui/elements/option_menu.py +168 -0
  23. KL_Py/hindGui/elements/pane.py +127 -0
  24. KL_Py/hindGui/elements/progress_bar.py +304 -0
  25. KL_Py/hindGui/elements/radio.py +252 -0
  26. KL_Py/hindGui/elements/separator.py +64 -0
  27. KL_Py/hindGui/elements/sizegrip.py +34 -0
  28. KL_Py/hindGui/elements/slider.py +199 -0
  29. KL_Py/hindGui/elements/spin.py +252 -0
  30. KL_Py/hindGui/elements/status_bar.py +165 -0
  31. KL_Py/hindGui/elements/stretch.py +26 -0
  32. KL_Py/hindGui/elements/tab.py +697 -0
  33. KL_Py/hindGui/elements/table.py +463 -0
  34. KL_Py/hindGui/elements/text.py +413 -0
  35. KL_Py/hindGui/elements/tree.py +493 -0
  36. KL_Py/hindGui/replace.py +45 -0
  37. KL_Py/hindGui/tray.py +358 -0
  38. KL_Py/hindGui/window.py +2906 -0
  39. KL_Py/requests/__init__.py +184 -0
  40. KL_Py/requests/__version__.py +14 -0
  41. KL_Py/requests/_internal_utils.py +50 -0
  42. KL_Py/requests/adapters.py +719 -0
  43. KL_Py/requests/api.py +157 -0
  44. KL_Py/requests/auth.py +314 -0
  45. KL_Py/requests/certifi/__init__.py +4 -0
  46. KL_Py/requests/certifi/__main__.py +12 -0
  47. KL_Py/requests/certifi/core.py +83 -0
  48. KL_Py/requests/certifi/py.typed +0 -0
  49. KL_Py/requests/certs.py +17 -0
  50. KL_Py/requests/charset_normalizer/__init__.py +48 -0
  51. KL_Py/requests/charset_normalizer/__main__.py +6 -0
  52. KL_Py/requests/charset_normalizer/api.py +671 -0
  53. KL_Py/requests/charset_normalizer/cd.py +421 -0
  54. KL_Py/requests/charset_normalizer/cli/__init__.py +8 -0
  55. KL_Py/requests/charset_normalizer/cli/__main__.py +363 -0
  56. KL_Py/requests/charset_normalizer/constant.py +2031 -0
  57. KL_Py/requests/charset_normalizer/legacy.py +80 -0
  58. KL_Py/requests/charset_normalizer/md.py +744 -0
  59. KL_Py/requests/charset_normalizer/models.py +359 -0
  60. KL_Py/requests/charset_normalizer/py.typed +0 -0
  61. KL_Py/requests/charset_normalizer/utils.py +420 -0
  62. KL_Py/requests/charset_normalizer/version.py +8 -0
  63. KL_Py/requests/compat.py +106 -0
  64. KL_Py/requests/cookies.py +561 -0
  65. KL_Py/requests/exceptions.py +151 -0
  66. KL_Py/requests/help.py +134 -0
  67. KL_Py/requests/hooks.py +33 -0
  68. KL_Py/requests/idna/__init__.py +45 -0
  69. KL_Py/requests/idna/codec.py +122 -0
  70. KL_Py/requests/idna/compat.py +15 -0
  71. KL_Py/requests/idna/core.py +437 -0
  72. KL_Py/requests/idna/idnadata.py +4309 -0
  73. KL_Py/requests/idna/intranges.py +57 -0
  74. KL_Py/requests/idna/package_data.py +1 -0
  75. KL_Py/requests/idna/py.typed +0 -0
  76. KL_Py/requests/idna/uts46data.py +8841 -0
  77. KL_Py/requests/models.py +1039 -0
  78. KL_Py/requests/packages.py +23 -0
  79. KL_Py/requests/sessions.py +831 -0
  80. KL_Py/requests/status_codes.py +128 -0
  81. KL_Py/requests/structures.py +99 -0
  82. KL_Py/requests/urllib3/__init__.py +211 -0
  83. KL_Py/requests/urllib3/_base_connection.py +165 -0
  84. KL_Py/requests/urllib3/_collections.py +487 -0
  85. KL_Py/requests/urllib3/_request_methods.py +278 -0
  86. KL_Py/requests/urllib3/_version.py +34 -0
  87. KL_Py/requests/urllib3/connection.py +1099 -0
  88. KL_Py/requests/urllib3/connectionpool.py +1178 -0
  89. KL_Py/requests/urllib3/contrib/__init__.py +0 -0
  90. KL_Py/requests/urllib3/contrib/emscripten/__init__.py +17 -0
  91. KL_Py/requests/urllib3/contrib/emscripten/connection.py +260 -0
  92. KL_Py/requests/urllib3/contrib/emscripten/fetch.py +726 -0
  93. KL_Py/requests/urllib3/contrib/emscripten/request.py +22 -0
  94. KL_Py/requests/urllib3/contrib/emscripten/response.py +277 -0
  95. KL_Py/requests/urllib3/contrib/pyopenssl.py +564 -0
  96. KL_Py/requests/urllib3/contrib/socks.py +228 -0
  97. KL_Py/requests/urllib3/exceptions.py +335 -0
  98. KL_Py/requests/urllib3/fields.py +341 -0
  99. KL_Py/requests/urllib3/filepost.py +89 -0
  100. KL_Py/requests/urllib3/http2/__init__.py +53 -0
  101. KL_Py/requests/urllib3/http2/connection.py +356 -0
  102. KL_Py/requests/urllib3/http2/probe.py +87 -0
  103. KL_Py/requests/urllib3/poolmanager.py +651 -0
  104. KL_Py/requests/urllib3/py.typed +2 -0
  105. KL_Py/requests/urllib3/response.py +1480 -0
  106. KL_Py/requests/urllib3/util/__init__.py +42 -0
  107. KL_Py/requests/urllib3/util/connection.py +137 -0
  108. KL_Py/requests/urllib3/util/proxy.py +43 -0
  109. KL_Py/requests/urllib3/util/request.py +263 -0
  110. KL_Py/requests/urllib3/util/response.py +101 -0
  111. KL_Py/requests/urllib3/util/retry.py +549 -0
  112. KL_Py/requests/urllib3/util/ssl_.py +527 -0
  113. KL_Py/requests/urllib3/util/ssl_match_hostname.py +159 -0
  114. KL_Py/requests/urllib3/util/ssltransport.py +271 -0
  115. KL_Py/requests/urllib3/util/timeout.py +275 -0
  116. KL_Py/requests/urllib3/util/url.py +469 -0
  117. KL_Py/requests/urllib3/util/util.py +42 -0
  118. KL_Py/requests/urllib3/util/wait.py +124 -0
  119. KL_Py/requests/utils.py +1086 -0
  120. KL_Py/when.py +144 -0
  121. kl_py-0.0.1.dist-info/METADATA +15 -0
  122. kl_py-0.0.1.dist-info/RECORD +125 -0
  123. kl_py-0.0.1.dist-info/WHEEL +5 -0
  124. kl_py-0.0.1.dist-info/licenses/LICENSE +21 -0
  125. kl_py-0.0.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,744 @@
1
+ from __future__ import annotations
2
+
3
+ import sys
4
+ from functools import lru_cache
5
+ from logging import getLogger
6
+
7
+ if sys.version_info >= (3, 8):
8
+ from typing import final
9
+ else:
10
+ try:
11
+ from typing_extensions import final
12
+ except ImportError:
13
+
14
+ def final(cls): # type: ignore[misc,no-untyped-def]
15
+ return cls
16
+
17
+
18
+ from .constant import (
19
+ COMMON_SAFE_ASCII_CHARACTERS,
20
+ TRACE,
21
+ UNICODE_SECONDARY_RANGE_KEYWORD,
22
+ _ACCENTUATED,
23
+ _CJK,
24
+ _HANGUL,
25
+ _HIRAGANA,
26
+ _KATAKANA,
27
+ _LATIN,
28
+ _THAI,
29
+ )
30
+ from .utils import (
31
+ _character_flags,
32
+ is_accentuated,
33
+ is_arabic,
34
+ is_arabic_isolated_form,
35
+ is_case_variable,
36
+ is_cjk,
37
+ is_emoticon,
38
+ is_latin,
39
+ is_punctuation,
40
+ is_separator,
41
+ is_symbol,
42
+ is_unprintable,
43
+ remove_accent,
44
+ unicode_range,
45
+ is_cjk_uncommon,
46
+ )
47
+
48
+ # Combined bitmask for CJK/Hangul/Katakana/Hiragana/Thai glyph detection.
49
+ _GLYPH_MASK: int = _CJK | _HANGUL | _KATAKANA | _HIRAGANA | _THAI
50
+
51
+
52
+ class MessDetectorPlugin:
53
+ """
54
+ Base abstract class used for mess detection plugins.
55
+ All detectors MUST extend and implement given methods.
56
+ """
57
+
58
+ __slots__ = ()
59
+
60
+ def eligible(self, character: str) -> bool:
61
+ """
62
+ Determine if given character should be fed in.
63
+ """
64
+ raise NotImplementedError # pragma: nocover
65
+
66
+ def feed(self, character: str) -> None:
67
+ """
68
+ The main routine to be executed upon character.
69
+ Insert the logic in witch the text would be considered chaotic.
70
+ """
71
+ raise NotImplementedError # pragma: nocover
72
+
73
+ def reset(self) -> None: # pragma: no cover
74
+ """
75
+ Permit to reset the plugin to the initial state.
76
+ """
77
+ raise NotImplementedError
78
+
79
+ @property
80
+ def ratio(self) -> float:
81
+ """
82
+ Compute the chaos ratio based on what your feed() has seen.
83
+ Must NOT be lower than 0.; No restriction gt 0.
84
+ """
85
+ raise NotImplementedError # pragma: nocover
86
+
87
+
88
+ @final
89
+ class TooManySymbolOrPunctuationPlugin(MessDetectorPlugin):
90
+ __slots__ = (
91
+ "_punctuation_count",
92
+ "_symbol_count",
93
+ "_character_count",
94
+ "_last_printable_char",
95
+ "_frenzy_symbol_in_word",
96
+ )
97
+
98
+ def __init__(self) -> None:
99
+ self._punctuation_count: int = 0
100
+ self._symbol_count: int = 0
101
+ self._character_count: int = 0
102
+
103
+ self._last_printable_char: str | None = None
104
+ self._frenzy_symbol_in_word: bool = False
105
+
106
+ def eligible(self, character: str) -> bool:
107
+ return character.isprintable()
108
+
109
+ def feed(self, character: str) -> None:
110
+ self._character_count += 1
111
+
112
+ if (
113
+ character != self._last_printable_char
114
+ and character not in COMMON_SAFE_ASCII_CHARACTERS
115
+ ):
116
+ if is_punctuation(character):
117
+ self._punctuation_count += 1
118
+ elif (
119
+ not character.isdigit()
120
+ and is_symbol(character)
121
+ and not is_emoticon(character)
122
+ ):
123
+ self._symbol_count += 2
124
+
125
+ self._last_printable_char = character
126
+
127
+ def reset(self) -> None: # Abstract
128
+ self._punctuation_count = 0
129
+ self._character_count = 0
130
+ self._symbol_count = 0
131
+
132
+ @property
133
+ def ratio(self) -> float:
134
+ if self._character_count == 0:
135
+ return 0.0
136
+
137
+ ratio_of_punctuation: float = (
138
+ self._punctuation_count + self._symbol_count
139
+ ) / self._character_count
140
+
141
+ return ratio_of_punctuation if ratio_of_punctuation >= 0.3 else 0.0
142
+
143
+
144
+ @final
145
+ class TooManyAccentuatedPlugin(MessDetectorPlugin):
146
+ __slots__ = ("_character_count", "_accentuated_count")
147
+
148
+ def __init__(self) -> None:
149
+ self._character_count: int = 0
150
+ self._accentuated_count: int = 0
151
+
152
+ def eligible(self, character: str) -> bool:
153
+ return character.isalpha()
154
+
155
+ def feed(self, character: str) -> None:
156
+ self._character_count += 1
157
+
158
+ if is_accentuated(character):
159
+ self._accentuated_count += 1
160
+
161
+ def reset(self) -> None: # Abstract
162
+ self._character_count = 0
163
+ self._accentuated_count = 0
164
+
165
+ @property
166
+ def ratio(self) -> float:
167
+ if self._character_count < 8:
168
+ return 0.0
169
+
170
+ ratio_of_accentuation: float = self._accentuated_count / self._character_count
171
+ return ratio_of_accentuation if ratio_of_accentuation >= 0.35 else 0.0
172
+
173
+
174
+ @final
175
+ class UnprintablePlugin(MessDetectorPlugin):
176
+ __slots__ = ("_unprintable_count", "_character_count")
177
+
178
+ def __init__(self) -> None:
179
+ self._unprintable_count: int = 0
180
+ self._character_count: int = 0
181
+
182
+ def eligible(self, character: str) -> bool:
183
+ return True
184
+
185
+ def feed(self, character: str) -> None:
186
+ if is_unprintable(character):
187
+ self._unprintable_count += 1
188
+ self._character_count += 1
189
+
190
+ def reset(self) -> None: # Abstract
191
+ self._unprintable_count = 0
192
+
193
+ @property
194
+ def ratio(self) -> float:
195
+ if self._character_count == 0:
196
+ return 0.0
197
+
198
+ return (self._unprintable_count * 8) / self._character_count
199
+
200
+
201
+ @final
202
+ class SuspiciousDuplicateAccentPlugin(MessDetectorPlugin):
203
+ __slots__ = (
204
+ "_successive_count",
205
+ "_character_count",
206
+ "_last_latin_character",
207
+ "_last_was_accentuated",
208
+ )
209
+
210
+ def __init__(self) -> None:
211
+ self._successive_count: int = 0
212
+ self._character_count: int = 0
213
+
214
+ self._last_latin_character: str | None = None
215
+ self._last_was_accentuated: bool = False
216
+
217
+ def eligible(self, character: str) -> bool:
218
+ return character.isalpha() and is_latin(character)
219
+
220
+ def feed(self, character: str) -> None:
221
+ self._character_count += 1
222
+ current_accentuated: bool = is_accentuated(character)
223
+ if (
224
+ self._last_latin_character is not None
225
+ and current_accentuated
226
+ and self._last_was_accentuated
227
+ ):
228
+ if character.isupper() and self._last_latin_character.isupper():
229
+ self._successive_count += 1
230
+ # Worse if its the same char duplicated with different accent.
231
+ if remove_accent(character) == remove_accent(self._last_latin_character):
232
+ self._successive_count += 1
233
+ self._last_latin_character = character
234
+ self._last_was_accentuated = current_accentuated
235
+
236
+ def reset(self) -> None: # Abstract
237
+ self._successive_count = 0
238
+ self._character_count = 0
239
+ self._last_latin_character = None
240
+ self._last_was_accentuated = False
241
+
242
+ @property
243
+ def ratio(self) -> float:
244
+ if self._character_count == 0:
245
+ return 0.0
246
+
247
+ return (self._successive_count * 2) / self._character_count
248
+
249
+
250
+ @final
251
+ class SuspiciousRange(MessDetectorPlugin):
252
+ __slots__ = (
253
+ "_suspicious_successive_range_count",
254
+ "_character_count",
255
+ "_last_printable_seen",
256
+ "_last_printable_range",
257
+ )
258
+
259
+ def __init__(self) -> None:
260
+ self._suspicious_successive_range_count: int = 0
261
+ self._character_count: int = 0
262
+ self._last_printable_seen: str | None = None
263
+ self._last_printable_range: str | None = None
264
+
265
+ def eligible(self, character: str) -> bool:
266
+ return character.isprintable()
267
+
268
+ def feed(self, character: str) -> None:
269
+ self._character_count += 1
270
+
271
+ if (
272
+ character.isspace()
273
+ or is_punctuation(character)
274
+ or character in COMMON_SAFE_ASCII_CHARACTERS
275
+ ):
276
+ self._last_printable_seen = None
277
+ self._last_printable_range = None
278
+ return
279
+
280
+ if self._last_printable_seen is None:
281
+ self._last_printable_seen = character
282
+ self._last_printable_range = unicode_range(character)
283
+ return
284
+
285
+ unicode_range_a: str | None = self._last_printable_range
286
+ unicode_range_b: str | None = unicode_range(character)
287
+
288
+ if is_suspiciously_successive_range(unicode_range_a, unicode_range_b):
289
+ self._suspicious_successive_range_count += 1
290
+
291
+ self._last_printable_seen = character
292
+ self._last_printable_range = unicode_range_b
293
+
294
+ def reset(self) -> None: # Abstract
295
+ self._character_count = 0
296
+ self._suspicious_successive_range_count = 0
297
+ self._last_printable_seen = None
298
+ self._last_printable_range = None
299
+
300
+ @property
301
+ def ratio(self) -> float:
302
+ if self._character_count <= 13:
303
+ return 0.0
304
+
305
+ ratio_of_suspicious_range_usage: float = (
306
+ self._suspicious_successive_range_count * 2
307
+ ) / self._character_count
308
+
309
+ return ratio_of_suspicious_range_usage
310
+
311
+
312
+ @final
313
+ class SuperWeirdWordPlugin(MessDetectorPlugin):
314
+ __slots__ = (
315
+ "_word_count",
316
+ "_bad_word_count",
317
+ "_foreign_long_count",
318
+ "_is_current_word_bad",
319
+ "_foreign_long_watch",
320
+ "_character_count",
321
+ "_bad_character_count",
322
+ "_buffer_length",
323
+ "_buffer_last_char",
324
+ "_buffer_last_char_accentuated",
325
+ "_buffer_accent_count",
326
+ "_buffer_glyph_count",
327
+ "_buffer_upper_count",
328
+ )
329
+
330
+ def __init__(self) -> None:
331
+ self._word_count: int = 0
332
+ self._bad_word_count: int = 0
333
+ self._foreign_long_count: int = 0
334
+
335
+ self._is_current_word_bad: bool = False
336
+ self._foreign_long_watch: bool = False
337
+
338
+ self._character_count: int = 0
339
+ self._bad_character_count: int = 0
340
+
341
+ self._buffer_length: int = 0
342
+ self._buffer_last_char: str | None = None
343
+ self._buffer_last_char_accentuated: bool = False
344
+ self._buffer_accent_count: int = 0
345
+ self._buffer_glyph_count: int = 0
346
+ self._buffer_upper_count: int = 0
347
+
348
+ def eligible(self, character: str) -> bool:
349
+ return True
350
+
351
+ def feed(self, character: str) -> None:
352
+ if character.isalpha():
353
+ self._buffer_length += 1
354
+ self._buffer_last_char = character
355
+
356
+ if character.isupper():
357
+ self._buffer_upper_count += 1
358
+
359
+ flags: int = _character_flags(character)
360
+ char_accentuated: bool = bool(flags & _ACCENTUATED)
361
+ self._buffer_last_char_accentuated = char_accentuated
362
+
363
+ if char_accentuated:
364
+ self._buffer_accent_count += 1
365
+ if (
366
+ not self._foreign_long_watch
367
+ and (not (flags & _LATIN) or char_accentuated)
368
+ and not (flags & _GLYPH_MASK)
369
+ ):
370
+ self._foreign_long_watch = True
371
+ if flags & _GLYPH_MASK:
372
+ self._buffer_glyph_count += 1
373
+ return
374
+ if not self._buffer_length:
375
+ return
376
+ if (
377
+ character.isspace() or is_punctuation(character) or is_separator(character)
378
+ ) and self._buffer_length:
379
+ self._word_count += 1
380
+ buffer_length: int = self._buffer_length
381
+
382
+ self._character_count += buffer_length
383
+
384
+ if buffer_length >= 4:
385
+ if self._buffer_accent_count / buffer_length >= 0.5:
386
+ self._is_current_word_bad = True
387
+ # Word/Buffer ending with an upper case accentuated letter are so rare,
388
+ # that we will consider them all as suspicious. Same weight as foreign_long suspicious.
389
+ elif (
390
+ self._buffer_last_char_accentuated
391
+ and self._buffer_last_char.isupper() # type: ignore[union-attr]
392
+ and self._buffer_upper_count != buffer_length
393
+ ):
394
+ self._foreign_long_count += 1
395
+ self._is_current_word_bad = True
396
+ elif self._buffer_glyph_count == 1:
397
+ self._is_current_word_bad = True
398
+ self._foreign_long_count += 1
399
+ if buffer_length >= 24 and self._foreign_long_watch:
400
+ probable_camel_cased: bool = (
401
+ self._buffer_upper_count > 0
402
+ and self._buffer_upper_count / buffer_length <= 0.3
403
+ )
404
+
405
+ if not probable_camel_cased:
406
+ self._foreign_long_count += 1
407
+ self._is_current_word_bad = True
408
+
409
+ if self._is_current_word_bad:
410
+ self._bad_word_count += 1
411
+ self._bad_character_count += buffer_length
412
+ self._is_current_word_bad = False
413
+
414
+ self._foreign_long_watch = False
415
+ self._buffer_length = 0
416
+ self._buffer_last_char = None
417
+ self._buffer_last_char_accentuated = False
418
+ self._buffer_accent_count = 0
419
+ self._buffer_glyph_count = 0
420
+ self._buffer_upper_count = 0
421
+ elif (
422
+ character not in {"<", ">", "-", "=", "~", "|", "_"}
423
+ and not character.isdigit()
424
+ and is_symbol(character)
425
+ ):
426
+ self._is_current_word_bad = True
427
+ self._buffer_length += 1
428
+ self._buffer_last_char = character
429
+ self._buffer_last_char_accentuated = False
430
+
431
+ def reset(self) -> None: # Abstract
432
+ self._buffer_length = 0
433
+ self._buffer_last_char = None
434
+ self._buffer_last_char_accentuated = False
435
+ self._is_current_word_bad = False
436
+ self._foreign_long_watch = False
437
+ self._bad_word_count = 0
438
+ self._word_count = 0
439
+ self._character_count = 0
440
+ self._bad_character_count = 0
441
+ self._foreign_long_count = 0
442
+ self._buffer_accent_count = 0
443
+ self._buffer_glyph_count = 0
444
+ self._buffer_upper_count = 0
445
+
446
+ @property
447
+ def ratio(self) -> float:
448
+ if self._word_count <= 10 and self._foreign_long_count == 0:
449
+ return 0.0
450
+
451
+ return self._bad_character_count / self._character_count
452
+
453
+
454
+ @final
455
+ class CjkUncommonPlugin(MessDetectorPlugin):
456
+ """
457
+ Detect messy CJK text that probably means nothing.
458
+ """
459
+
460
+ __slots__ = ("_character_count", "_uncommon_count")
461
+
462
+ def __init__(self) -> None:
463
+ self._character_count: int = 0
464
+ self._uncommon_count: int = 0
465
+
466
+ def eligible(self, character: str) -> bool:
467
+ return is_cjk(character)
468
+
469
+ def feed(self, character: str) -> None:
470
+ self._character_count += 1
471
+
472
+ if is_cjk_uncommon(character):
473
+ self._uncommon_count += 1
474
+ return
475
+
476
+ def reset(self) -> None: # Abstract
477
+ self._character_count = 0
478
+ self._uncommon_count = 0
479
+
480
+ @property
481
+ def ratio(self) -> float:
482
+ if self._character_count < 8:
483
+ return 0.0
484
+
485
+ uncommon_form_usage: float = self._uncommon_count / self._character_count
486
+
487
+ # we can be pretty sure it's garbage when uncommon characters are widely
488
+ # used. otherwise it could just be traditional chinese for example.
489
+ return uncommon_form_usage / 10 if uncommon_form_usage > 0.5 else 0.0
490
+
491
+
492
+ @final
493
+ class ArchaicUpperLowerPlugin(MessDetectorPlugin):
494
+ __slots__ = (
495
+ "_buf",
496
+ "_character_count_since_last_sep",
497
+ "_successive_upper_lower_count",
498
+ "_successive_upper_lower_count_final",
499
+ "_character_count",
500
+ "_last_alpha_seen",
501
+ "_current_ascii_only",
502
+ )
503
+
504
+ def __init__(self) -> None:
505
+ self._buf: bool = False
506
+
507
+ self._character_count_since_last_sep: int = 0
508
+
509
+ self._successive_upper_lower_count: int = 0
510
+ self._successive_upper_lower_count_final: int = 0
511
+
512
+ self._character_count: int = 0
513
+
514
+ self._last_alpha_seen: str | None = None
515
+ self._current_ascii_only: bool = True
516
+
517
+ def eligible(self, character: str) -> bool:
518
+ return True
519
+
520
+ def feed(self, character: str) -> None:
521
+ is_concerned: bool = character.isalpha() and is_case_variable(character)
522
+ chunk_sep: bool = not is_concerned
523
+
524
+ if chunk_sep and self._character_count_since_last_sep > 0:
525
+ if (
526
+ self._character_count_since_last_sep <= 64
527
+ and not character.isdigit()
528
+ and not self._current_ascii_only
529
+ ):
530
+ self._successive_upper_lower_count_final += (
531
+ self._successive_upper_lower_count
532
+ )
533
+
534
+ self._successive_upper_lower_count = 0
535
+ self._character_count_since_last_sep = 0
536
+ self._last_alpha_seen = None
537
+ self._buf = False
538
+ self._character_count += 1
539
+ self._current_ascii_only = True
540
+
541
+ return
542
+
543
+ if self._current_ascii_only and not character.isascii():
544
+ self._current_ascii_only = False
545
+
546
+ if self._last_alpha_seen is not None:
547
+ if (character.isupper() and self._last_alpha_seen.islower()) or (
548
+ character.islower() and self._last_alpha_seen.isupper()
549
+ ):
550
+ if self._buf:
551
+ self._successive_upper_lower_count += 2
552
+ self._buf = False
553
+ else:
554
+ self._buf = True
555
+ else:
556
+ self._buf = False
557
+
558
+ self._character_count += 1
559
+ self._character_count_since_last_sep += 1
560
+ self._last_alpha_seen = character
561
+
562
+ def reset(self) -> None: # Abstract
563
+ self._character_count = 0
564
+ self._character_count_since_last_sep = 0
565
+ self._successive_upper_lower_count = 0
566
+ self._successive_upper_lower_count_final = 0
567
+ self._last_alpha_seen = None
568
+ self._buf = False
569
+ self._current_ascii_only = True
570
+
571
+ @property
572
+ def ratio(self) -> float:
573
+ if self._character_count == 0:
574
+ return 0.0
575
+
576
+ return self._successive_upper_lower_count_final / self._character_count
577
+
578
+
579
+ @final
580
+ class ArabicIsolatedFormPlugin(MessDetectorPlugin):
581
+ __slots__ = ("_character_count", "_isolated_form_count")
582
+
583
+ def __init__(self) -> None:
584
+ self._character_count: int = 0
585
+ self._isolated_form_count: int = 0
586
+
587
+ def reset(self) -> None: # Abstract
588
+ self._character_count = 0
589
+ self._isolated_form_count = 0
590
+
591
+ def eligible(self, character: str) -> bool:
592
+ return is_arabic(character)
593
+
594
+ def feed(self, character: str) -> None:
595
+ self._character_count += 1
596
+
597
+ if is_arabic_isolated_form(character):
598
+ self._isolated_form_count += 1
599
+
600
+ @property
601
+ def ratio(self) -> float:
602
+ if self._character_count < 8:
603
+ return 0.0
604
+
605
+ isolated_form_usage: float = self._isolated_form_count / self._character_count
606
+
607
+ return isolated_form_usage
608
+
609
+
610
+ @lru_cache(maxsize=1024)
611
+ def is_suspiciously_successive_range(
612
+ unicode_range_a: str | None, unicode_range_b: str | None
613
+ ) -> bool:
614
+ """
615
+ Determine if two Unicode range seen next to each other can be considered as suspicious.
616
+ """
617
+ if unicode_range_a is None or unicode_range_b is None:
618
+ return True
619
+
620
+ if unicode_range_a == unicode_range_b:
621
+ return False
622
+
623
+ if "Latin" in unicode_range_a and "Latin" in unicode_range_b:
624
+ return False
625
+
626
+ if "Emoticons" in unicode_range_a or "Emoticons" in unicode_range_b:
627
+ return False
628
+
629
+ # Latin characters can be accompanied with a combining diacritical mark
630
+ # eg. Vietnamese.
631
+ if ("Latin" in unicode_range_a or "Latin" in unicode_range_b) and (
632
+ "Combining" in unicode_range_a or "Combining" in unicode_range_b
633
+ ):
634
+ return False
635
+
636
+ keywords_range_a, keywords_range_b = (
637
+ unicode_range_a.split(" "),
638
+ unicode_range_b.split(" "),
639
+ )
640
+
641
+ for el in keywords_range_a:
642
+ if el in UNICODE_SECONDARY_RANGE_KEYWORD:
643
+ continue
644
+ if el in keywords_range_b:
645
+ return False
646
+
647
+ # Japanese Exception
648
+ range_a_jp_chars, range_b_jp_chars = (
649
+ unicode_range_a
650
+ in (
651
+ "Hiragana",
652
+ "Katakana",
653
+ ),
654
+ unicode_range_b in ("Hiragana", "Katakana"),
655
+ )
656
+ if (range_a_jp_chars or range_b_jp_chars) and (
657
+ "CJK" in unicode_range_a or "CJK" in unicode_range_b
658
+ ):
659
+ return False
660
+ if range_a_jp_chars and range_b_jp_chars:
661
+ return False
662
+
663
+ if "Hangul" in unicode_range_a or "Hangul" in unicode_range_b:
664
+ if "CJK" in unicode_range_a or "CJK" in unicode_range_b:
665
+ return False
666
+ if unicode_range_a == "Basic Latin" or unicode_range_b == "Basic Latin":
667
+ return False
668
+
669
+ # Chinese/Japanese use dedicated range for punctuation and/or separators.
670
+ if ("CJK" in unicode_range_a or "CJK" in unicode_range_b) or (
671
+ unicode_range_a in ["Katakana", "Hiragana"]
672
+ and unicode_range_b in ["Katakana", "Hiragana"]
673
+ ):
674
+ if "Punctuation" in unicode_range_a or "Punctuation" in unicode_range_b:
675
+ return False
676
+ if "Forms" in unicode_range_a or "Forms" in unicode_range_b:
677
+ return False
678
+ if unicode_range_a == "Basic Latin" or unicode_range_b == "Basic Latin":
679
+ return False
680
+
681
+ return True
682
+
683
+
684
+ # import time messdetector plugins detection(...)
685
+ _DETECTOR_CLASSES: tuple[type[MessDetectorPlugin], ...] = tuple(
686
+ md_class for md_class in MessDetectorPlugin.__subclasses__()
687
+ )
688
+
689
+
690
+ @lru_cache(maxsize=2048)
691
+ def mess_ratio(
692
+ decoded_sequence: str, maximum_threshold: float = 0.2, debug: bool = False
693
+ ) -> float:
694
+ """
695
+ Compute a mess ratio given a decoded bytes sequence. The maximum threshold does stop the computation earlier.
696
+ """
697
+
698
+ detectors: list[MessDetectorPlugin] = [md_class() for md_class in _DETECTOR_CLASSES]
699
+
700
+ mean_mess_ratio: float
701
+ seq_len: int = len(decoded_sequence)
702
+
703
+ if seq_len < 511:
704
+ step: int = 32
705
+ elif seq_len < 1024:
706
+ step = 64
707
+ else:
708
+ step = 128
709
+
710
+ for block_start in range(0, seq_len, step):
711
+ for character in decoded_sequence[block_start : block_start + step]:
712
+ for detector in detectors:
713
+ if detector.eligible(character):
714
+ detector.feed(character)
715
+
716
+ mean_mess_ratio = sum(dt.ratio for dt in detectors)
717
+
718
+ if mean_mess_ratio >= maximum_threshold:
719
+ break
720
+ else:
721
+ # Flush last word buffer in SuperWeirdWordPlugin via trailing newline.
722
+ for detector in detectors:
723
+ if detector.eligible("\n"):
724
+ detector.feed("\n")
725
+ mean_mess_ratio = sum(dt.ratio for dt in detectors)
726
+
727
+ if debug:
728
+ logger = getLogger("charset_normalizer")
729
+
730
+ logger.log(
731
+ TRACE,
732
+ "Mess-detector extended-analysis start. "
733
+ f"intermediary_mean_mess_ratio_calc={step} mean_mess_ratio={mean_mess_ratio} "
734
+ f"maximum_threshold={maximum_threshold}",
735
+ )
736
+
737
+ if seq_len > 16:
738
+ logger.log(TRACE, f"Starting with: {decoded_sequence[:16]}")
739
+ logger.log(TRACE, f"Ending with: {decoded_sequence[-16::]}")
740
+
741
+ for dt in detectors:
742
+ logger.log(TRACE, f"{dt.__class__}: {dt.ratio}")
743
+
744
+ return round(mean_mess_ratio, 3)