bytesense 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bytesense/repair.py ADDED
@@ -0,0 +1,280 @@
1
+ """
2
+ Mojibake repair engine.
3
+
4
+ Mojibake = text that was decoded with the wrong encoding and then
5
+ re-encoded or displayed as-is, producing garbled characters.
6
+
7
+ Classic pattern: UTF-8 bytes → decoded as Latin-1 → re-encoded → "été" instead of "été"
8
+
9
+ Strategy:
10
+ 1. Re-encode the garbled string back to bytes using the assumed wrong encoding.
11
+ 2. Try to decode those bytes with the correct encoding.
12
+ 3. Accept the result if mess_ratio improves significantly.
13
+
14
+ Supports:
15
+ - Single-step mojibake (utf-8 → latin-1 → utf-8)
16
+ - Double-step mojibake (utf-8 → latin-1 → utf-8 → latin-1 → utf-8)
17
+ - Windows-1252 variant (utf-8 → cp1252 → utf-8)
18
+ - Detection-guided repair (auto-detect which transformation was applied)
19
+
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from dataclasses import dataclass
25
+ from typing import List, Optional
26
+
27
+ from .mess import mess_ratio
28
+
29
+ # Ordered list of (re-encode-as, then-decode-as) transformation chains to try.
30
+ # Most common patterns first.
31
+ _REPAIR_CHAINS: list[tuple[str, str]] = [
32
+ ("latin_1", "utf_8"), # UTF-8 read as Latin-1 (most common)
33
+ ("cp1252", "utf_8"), # UTF-8 read as Windows-1252
34
+ ("latin_1", "cp1252"), # cp1252 read as Latin-1
35
+ ("latin_1", "cp1251"), # Cyrillic UTF-8 read as Latin-1
36
+ ("latin_1", "cp1253"), # Greek UTF-8 read as Latin-1
37
+ ("latin_1", "cp1256"), # Arabic UTF-8 read as Latin-1
38
+ ("latin_1", "iso8859_2"), # Central European UTF-8 read as Latin-1
39
+ ("utf_8", "latin_1"), # Latin-1 re-encoded as UTF-8 (less common)
40
+ ("cp1252", "latin_1"), # Latin-1 mistaken for cp1252
41
+ ]
42
+
43
+ # Maximum improvement in mess ratio to accept a repair
44
+ _MIN_IMPROVEMENT: float = 0.10
45
+
46
+
47
+ def _likely_utf8_mojibake(text: str) -> bool:
48
+ """Heuristic markers when UTF-8 was misread as Latin-1 / cp1252 (mess may stay low)."""
49
+ return any(m in text for m in ("Ã", "Â", "â", "Ä", "Å", "Æ", "Ç", "Ð", "Ñ", "Ø", "Ù"))
50
+
51
+
52
+ # UTF-8 read as Latin-1 / cp1252: only these chains are used for equal-mess tie-break
53
+ # (broader chains are still tried for strict mess improvement).
54
+ _TIE_BREAK_CHAINS: frozenset[tuple[str, str]] = frozenset(
55
+ {
56
+ ("latin_1", "utf_8"),
57
+ ("cp1252", "utf_8"),
58
+ }
59
+ )
60
+
61
+
62
+ @dataclass
63
+ class RepairResult:
64
+ """Result of a mojibake repair attempt."""
65
+
66
+ original: str
67
+ repaired: str
68
+ improved: bool
69
+ chain: Optional[tuple[str, str]] # (re_encode, decode) or None
70
+ original_mess: float
71
+ repaired_mess: float
72
+ iterations: int # Number of repair steps applied (1 = single, 2 = double)
73
+
74
+ @property
75
+ def improvement(self) -> float:
76
+ return self.original_mess - self.repaired_mess
77
+
78
+ def __str__(self) -> str:
79
+ if not self.improved:
80
+ return self.original
81
+ return self.repaired
82
+
83
+ def __repr__(self) -> str:
84
+ return (
85
+ f"RepairResult(improved={self.improved}, "
86
+ f"chain={self.chain!r}, "
87
+ f"improvement={self.improvement:.3f})"
88
+ )
89
+
90
+
91
+ def _try_chain(text: str, re_encode: str, decode_as: str) -> Optional[str]:
92
+ """
93
+ Apply one transformation: re-encode text with `re_encode`, then decode as `decode_as`.
94
+ Returns repaired string or None if transformation fails or produces worse text.
95
+ """
96
+ try:
97
+ raw = text.encode(re_encode, errors="strict")
98
+ return raw.decode(decode_as, errors="strict")
99
+ except (UnicodeEncodeError, UnicodeDecodeError, LookupError):
100
+ return None
101
+
102
+
103
+ def repair(
104
+ text: str,
105
+ max_iterations: int = 2,
106
+ chains: Optional[List[tuple[str, str]]] = None,
107
+ ) -> RepairResult:
108
+ """
109
+ Attempt to repair mojibake in ``text``.
110
+
111
+ Args:
112
+ text: Potentially garbled string to repair.
113
+ max_iterations: Maximum repair cycles (1 = single chain, 2 = double).
114
+ chains: Override the default transformation chain list.
115
+
116
+ Returns:
117
+ :class:`RepairResult` — always contains the best available result.
118
+ Check ``.improved`` to know if repair was applied.
119
+
120
+ Example::
121
+
122
+ from bytesense.repair import repair
123
+
124
+ garbled = "été"
125
+ result = repair(garbled)
126
+ if result.improved:
127
+ print(result.repaired) # "été"
128
+ else:
129
+ print(result.original) # unchanged
130
+ """
131
+ if not isinstance(text, str):
132
+ raise TypeError("text must be str")
133
+ if (
134
+ isinstance(max_iterations, bool)
135
+ or not isinstance(max_iterations, int)
136
+ or not 1 <= max_iterations <= 2
137
+ ):
138
+ raise ValueError("max_iterations must be 1 or 2")
139
+ if not text:
140
+ return RepairResult(
141
+ original=text,
142
+ repaired=text,
143
+ improved=False,
144
+ chain=None,
145
+ original_mess=0.0,
146
+ repaired_mess=0.0,
147
+ iterations=0,
148
+ )
149
+
150
+ original_mess = mess_ratio(text)
151
+ best_text = text
152
+ best_mess = original_mess
153
+ best_chain: Optional[tuple[str, str]] = None
154
+ best_iters = 0
155
+
156
+ effective_chains = chains if chains is not None else _REPAIR_CHAINS
157
+
158
+ # Single-step: collect candidates, then pick lowest mess then preferred chain order.
159
+ options: list[tuple[tuple[str, str], str, float, int]] = []
160
+ for i, (re_encode, decode_as) in enumerate(effective_chains):
161
+ candidate = _try_chain(text, re_encode, decode_as)
162
+ if candidate is None or candidate == text:
163
+ continue
164
+ candidate_mess = mess_ratio(candidate)
165
+ strong_improvement = (original_mess - candidate_mess) >= _MIN_IMPROVEMENT
166
+ tie_utf8_mojibake = (
167
+ _likely_utf8_mojibake(text)
168
+ and (re_encode, decode_as) in _TIE_BREAK_CHAINS
169
+ and candidate_mess <= original_mess + 1e-9
170
+ )
171
+ if strong_improvement or tie_utf8_mojibake:
172
+ options.append(((re_encode, decode_as), candidate, candidate_mess, i))
173
+
174
+ if options:
175
+ options.sort(key=lambda x: (x[2], x[3]))
176
+ best_chain, best_text, best_mess, _ = (
177
+ options[0][0],
178
+ options[0][1],
179
+ options[0][2],
180
+ options[0][3],
181
+ )
182
+ best_iters = 1
183
+
184
+ # Double-step repair (only if single-step improved things)
185
+ if (
186
+ max_iterations >= 2
187
+ and best_iters == 1
188
+ and (best_mess > 0.05 or _likely_utf8_mojibake(best_text))
189
+ ):
190
+ opts2: list[tuple[tuple[str, str], str, float, int]] = []
191
+ for i, (re_encode2, decode_as2) in enumerate(effective_chains):
192
+ candidate2 = _try_chain(best_text, re_encode2, decode_as2)
193
+ if candidate2 is None or candidate2 == best_text:
194
+ continue
195
+ candidate2_mess = mess_ratio(candidate2)
196
+ strong_improvement = (best_mess - candidate2_mess) >= _MIN_IMPROVEMENT
197
+ tie_utf8_mojibake = (
198
+ _likely_utf8_mojibake(best_text)
199
+ and (re_encode2, decode_as2) in _TIE_BREAK_CHAINS
200
+ and candidate2_mess <= best_mess + 1e-9
201
+ )
202
+ if strong_improvement or tie_utf8_mojibake:
203
+ opts2.append(((re_encode2, decode_as2), candidate2, candidate2_mess, i))
204
+ if opts2:
205
+ opts2.sort(key=lambda x: (x[2], x[3]))
206
+ best_chain = opts2[0][0]
207
+ best_text = opts2[0][1]
208
+ best_mess = opts2[0][2]
209
+ best_iters = 2
210
+
211
+ improved = best_iters > 0 and (
212
+ (original_mess - best_mess) >= _MIN_IMPROVEMENT
213
+ or (best_text != text and _likely_utf8_mojibake(text) and best_mess <= original_mess + 1e-9)
214
+ )
215
+
216
+ return RepairResult(
217
+ original=text,
218
+ repaired=best_text if improved else text,
219
+ improved=improved,
220
+ chain=best_chain if improved else None,
221
+ original_mess=original_mess,
222
+ repaired_mess=best_mess if improved else original_mess,
223
+ iterations=best_iters if improved else 0,
224
+ )
225
+
226
+
227
+ def repair_bytes(
228
+ data: bytes,
229
+ encoding: Optional[str] = None,
230
+ *,
231
+ max_iterations: int = 2,
232
+ chains: Optional[List[tuple[str, str]]] = None,
233
+ ) -> RepairResult:
234
+ """
235
+ Repair mojibake in a byte sequence.
236
+
237
+ First decodes ``data`` with ``encoding`` (auto-detected if None),
238
+ then applies text-level repair.
239
+
240
+ Args:
241
+ data: Byte sequence to repair.
242
+ encoding: Known encoding. If None, auto-detected via ``from_bytes``.
243
+ max_iterations: Forwarded to :func:`repair`.
244
+ chains: Forwarded to :func:`repair`.
245
+
246
+ Returns:
247
+ :class:`RepairResult`
248
+ """
249
+ if encoding is None:
250
+ from .api import from_bytes as _fb
251
+
252
+ r = _fb(data)
253
+ if r.encoding is None:
254
+ raise UnicodeError("encoding is unknown; supply an explicit encoding")
255
+ enc = r.encoding
256
+ else:
257
+ enc = encoding
258
+ text = data.decode(enc, errors="strict")
259
+
260
+ return repair(text, max_iterations=max_iterations, chains=chains)
261
+
262
+
263
+ def is_mojibake(text: str, threshold: float = 0.15) -> bool:
264
+ """
265
+ Quick heuristic to determine if text looks like mojibake.
266
+
267
+ Returns True if the text has high mess ratio AND attempting at least one
268
+ repair chain produces a significantly cleaner result.
269
+
270
+ Args:
271
+ text: String to check.
272
+ threshold: Mess ratio above which we attempt repair and check.
273
+
274
+ Returns:
275
+ bool
276
+ """
277
+ if mess_ratio(text) < threshold and not _likely_utf8_mojibake(text):
278
+ return False
279
+ result = repair(text)
280
+ return result.improved
bytesense/scoring.py ADDED
@@ -0,0 +1,100 @@
1
+ """Compact character-pair evidence; optional native scoring has identical semantics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import gzip
6
+ import json
7
+ import unicodedata
8
+ from collections import Counter
9
+ from functools import lru_cache
10
+ from importlib.resources import files
11
+ from typing import Any
12
+
13
+ from ._rust import is_rust_available
14
+
15
+
16
+ @lru_cache(maxsize=1)
17
+ def _model() -> tuple[frozenset[str], dict[str, float], Any]:
18
+ payload = json.loads(
19
+ gzip.decompress(files("bytesense.data").joinpath("language.json.gz").read_bytes())
20
+ )
21
+ letters, pairs = frozenset(payload["letters"]), payload["pairs"]
22
+ native = None
23
+ if is_rust_available():
24
+ from ._rust_core import NgramModel # type: ignore[import-not-found]
25
+
26
+ native = NgramModel(payload["letters"], list(payload["pairs"].items()))
27
+ return letters, pairs, native
28
+
29
+
30
+ def _prepare(text: str) -> tuple[str, float, float]:
31
+ if not text:
32
+ return "", 0.0, 0.0
33
+ counts = Counter(text)
34
+ bad = sum(n for c, n in counts.items() if not c.isprintable() and c not in "\n\r\t") / len(text)
35
+ symbols = sum(
36
+ n
37
+ for c, n in counts.items()
38
+ if c in "\\^`{|}~#"
39
+ or (
40
+ ord(c) >= 128
41
+ and not c.isalpha()
42
+ and not c.isspace()
43
+ and unicodedata.category(c)[0] not in "PN"
44
+ )
45
+ ) / len(text)
46
+ lower = text.lower()
47
+ table = {ord(c): c if c.isalpha() else " " for c in set(lower)}
48
+ return lower.translate(table), bad, symbols
49
+
50
+
51
+ def _score_pure(
52
+ text: str, bad: float, symbols: float, letters: frozenset[str], pairs: dict[str, float]
53
+ ) -> float:
54
+ counts = Counter(text)
55
+ total_letters = sum(n for c, n in counts.items() if c != " ")
56
+ known_letters = sum(n for c, n in counts.items() if c != " " and c in letters)
57
+ total = 0
58
+ known = 0.0
59
+ for (a, b), n in Counter(zip(text, text[1:])).items():
60
+ if a == b == " ":
61
+ continue
62
+ weight = n * (1 if (a + b).isascii() else 4)
63
+ total += weight
64
+ if a + b in pairs:
65
+ known += weight * pairs[a + b]
66
+ return (
67
+ 0.25 * known_letters / max(1, total_letters)
68
+ + 0.75 * known / max(1, total)
69
+ - bad * 5
70
+ - symbols * 3
71
+ )
72
+
73
+
74
+ def _classify(chars: str) -> list[tuple[bool, bool, bool]]:
75
+ """Unicode scalar properties from this Python version; no document state."""
76
+ return [
77
+ (
78
+ c.isalpha(),
79
+ not c.isprintable() and c not in "\n\r\t",
80
+ c in "\\^`{|}~#"
81
+ or (
82
+ ord(c) >= 128
83
+ and not c.isalpha()
84
+ and not c.isspace()
85
+ and unicodedata.category(c)[0] not in "PN"
86
+ ),
87
+ )
88
+ for c in chars
89
+ ]
90
+
91
+
92
+ def text_quality(text: str) -> tuple[float, float]:
93
+ """Return linguistic support and control-character ratio (neither is a probability)."""
94
+ letters, pairs, native = _model()
95
+ if native:
96
+ score, bad = native.quality(text, text.lower(), _classify)
97
+ else:
98
+ normalized, bad, symbols = _prepare(text)
99
+ score = _score_pure(normalized, bad, symbols, letters, pairs)
100
+ return round(score, 10), bad