bytesense 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bytesense/__init__.py +47 -0
- bytesense/_rust.py +58 -0
- bytesense/api.py +469 -0
- bytesense/candidate.py +199 -0
- bytesense/cli.py +73 -0
- bytesense/coherence.py +68 -0
- bytesense/constant.py +1313 -0
- bytesense/data/__init__.py +0 -0
- bytesense/data/fingerprints.py +94 -0
- bytesense/data/language.json.gz +0 -0
- bytesense/fingerprint.py +246 -0
- bytesense/heuristics.py +161 -0
- bytesense/hints.py +104 -0
- bytesense/legacy.py +38 -0
- bytesense/mess.py +179 -0
- bytesense/models.py +98 -0
- bytesense/multi.py +209 -0
- bytesense/py.typed +0 -0
- bytesense/repair.py +280 -0
- bytesense/scoring.py +100 -0
- bytesense/streaming.py +416 -0
- bytesense/version.py +4 -0
- bytesense-1.0.0.dist-info/METADATA +160 -0
- bytesense-1.0.0.dist-info/RECORD +28 -0
- bytesense-1.0.0.dist-info/WHEEL +5 -0
- bytesense-1.0.0.dist-info/entry_points.txt +2 -0
- bytesense-1.0.0.dist-info/licenses/LICENSE +21 -0
- bytesense-1.0.0.dist-info/top_level.txt +1 -0
bytesense/repair.py
ADDED
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mojibake repair engine.
|
|
3
|
+
|
|
4
|
+
Mojibake = text that was decoded with the wrong encoding and then
|
|
5
|
+
re-encoded or displayed as-is, producing garbled characters.
|
|
6
|
+
|
|
7
|
+
Classic pattern: UTF-8 bytes → decoded as Latin-1 → re-encoded → "été" instead of "été"
|
|
8
|
+
|
|
9
|
+
Strategy:
|
|
10
|
+
1. Re-encode the garbled string back to bytes using the assumed wrong encoding.
|
|
11
|
+
2. Try to decode those bytes with the correct encoding.
|
|
12
|
+
3. Accept the result if mess_ratio improves significantly.
|
|
13
|
+
|
|
14
|
+
Supports:
|
|
15
|
+
- Single-step mojibake (utf-8 → latin-1 → utf-8)
|
|
16
|
+
- Double-step mojibake (utf-8 → latin-1 → utf-8 → latin-1 → utf-8)
|
|
17
|
+
- Windows-1252 variant (utf-8 → cp1252 → utf-8)
|
|
18
|
+
- Detection-guided repair (auto-detect which transformation was applied)
|
|
19
|
+
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from typing import List, Optional
|
|
26
|
+
|
|
27
|
+
from .mess import mess_ratio
|
|
28
|
+
|
|
29
|
+
# Ordered list of (re-encode-as, then-decode-as) transformation chains to try.
|
|
30
|
+
# Most common patterns first.
|
|
31
|
+
_REPAIR_CHAINS: list[tuple[str, str]] = [
|
|
32
|
+
("latin_1", "utf_8"), # UTF-8 read as Latin-1 (most common)
|
|
33
|
+
("cp1252", "utf_8"), # UTF-8 read as Windows-1252
|
|
34
|
+
("latin_1", "cp1252"), # cp1252 read as Latin-1
|
|
35
|
+
("latin_1", "cp1251"), # Cyrillic UTF-8 read as Latin-1
|
|
36
|
+
("latin_1", "cp1253"), # Greek UTF-8 read as Latin-1
|
|
37
|
+
("latin_1", "cp1256"), # Arabic UTF-8 read as Latin-1
|
|
38
|
+
("latin_1", "iso8859_2"), # Central European UTF-8 read as Latin-1
|
|
39
|
+
("utf_8", "latin_1"), # Latin-1 re-encoded as UTF-8 (less common)
|
|
40
|
+
("cp1252", "latin_1"), # Latin-1 mistaken for cp1252
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
# Maximum improvement in mess ratio to accept a repair
|
|
44
|
+
_MIN_IMPROVEMENT: float = 0.10
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _likely_utf8_mojibake(text: str) -> bool:
|
|
48
|
+
"""Heuristic markers when UTF-8 was misread as Latin-1 / cp1252 (mess may stay low)."""
|
|
49
|
+
return any(m in text for m in ("Ã", "Â", "â", "Ä", "Å", "Æ", "Ç", "Ð", "Ñ", "Ø", "Ù"))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# UTF-8 read as Latin-1 / cp1252: only these chains are used for equal-mess tie-break
|
|
53
|
+
# (broader chains are still tried for strict mess improvement).
|
|
54
|
+
_TIE_BREAK_CHAINS: frozenset[tuple[str, str]] = frozenset(
|
|
55
|
+
{
|
|
56
|
+
("latin_1", "utf_8"),
|
|
57
|
+
("cp1252", "utf_8"),
|
|
58
|
+
}
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class RepairResult:
|
|
64
|
+
"""Result of a mojibake repair attempt."""
|
|
65
|
+
|
|
66
|
+
original: str
|
|
67
|
+
repaired: str
|
|
68
|
+
improved: bool
|
|
69
|
+
chain: Optional[tuple[str, str]] # (re_encode, decode) or None
|
|
70
|
+
original_mess: float
|
|
71
|
+
repaired_mess: float
|
|
72
|
+
iterations: int # Number of repair steps applied (1 = single, 2 = double)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def improvement(self) -> float:
|
|
76
|
+
return self.original_mess - self.repaired_mess
|
|
77
|
+
|
|
78
|
+
def __str__(self) -> str:
|
|
79
|
+
if not self.improved:
|
|
80
|
+
return self.original
|
|
81
|
+
return self.repaired
|
|
82
|
+
|
|
83
|
+
def __repr__(self) -> str:
|
|
84
|
+
return (
|
|
85
|
+
f"RepairResult(improved={self.improved}, "
|
|
86
|
+
f"chain={self.chain!r}, "
|
|
87
|
+
f"improvement={self.improvement:.3f})"
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _try_chain(text: str, re_encode: str, decode_as: str) -> Optional[str]:
|
|
92
|
+
"""
|
|
93
|
+
Apply one transformation: re-encode text with `re_encode`, then decode as `decode_as`.
|
|
94
|
+
Returns repaired string or None if transformation fails or produces worse text.
|
|
95
|
+
"""
|
|
96
|
+
try:
|
|
97
|
+
raw = text.encode(re_encode, errors="strict")
|
|
98
|
+
return raw.decode(decode_as, errors="strict")
|
|
99
|
+
except (UnicodeEncodeError, UnicodeDecodeError, LookupError):
|
|
100
|
+
return None
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def repair(
|
|
104
|
+
text: str,
|
|
105
|
+
max_iterations: int = 2,
|
|
106
|
+
chains: Optional[List[tuple[str, str]]] = None,
|
|
107
|
+
) -> RepairResult:
|
|
108
|
+
"""
|
|
109
|
+
Attempt to repair mojibake in ``text``.
|
|
110
|
+
|
|
111
|
+
Args:
|
|
112
|
+
text: Potentially garbled string to repair.
|
|
113
|
+
max_iterations: Maximum repair cycles (1 = single chain, 2 = double).
|
|
114
|
+
chains: Override the default transformation chain list.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
:class:`RepairResult` — always contains the best available result.
|
|
118
|
+
Check ``.improved`` to know if repair was applied.
|
|
119
|
+
|
|
120
|
+
Example::
|
|
121
|
+
|
|
122
|
+
from bytesense.repair import repair
|
|
123
|
+
|
|
124
|
+
garbled = "été"
|
|
125
|
+
result = repair(garbled)
|
|
126
|
+
if result.improved:
|
|
127
|
+
print(result.repaired) # "été"
|
|
128
|
+
else:
|
|
129
|
+
print(result.original) # unchanged
|
|
130
|
+
"""
|
|
131
|
+
if not isinstance(text, str):
|
|
132
|
+
raise TypeError("text must be str")
|
|
133
|
+
if (
|
|
134
|
+
isinstance(max_iterations, bool)
|
|
135
|
+
or not isinstance(max_iterations, int)
|
|
136
|
+
or not 1 <= max_iterations <= 2
|
|
137
|
+
):
|
|
138
|
+
raise ValueError("max_iterations must be 1 or 2")
|
|
139
|
+
if not text:
|
|
140
|
+
return RepairResult(
|
|
141
|
+
original=text,
|
|
142
|
+
repaired=text,
|
|
143
|
+
improved=False,
|
|
144
|
+
chain=None,
|
|
145
|
+
original_mess=0.0,
|
|
146
|
+
repaired_mess=0.0,
|
|
147
|
+
iterations=0,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
original_mess = mess_ratio(text)
|
|
151
|
+
best_text = text
|
|
152
|
+
best_mess = original_mess
|
|
153
|
+
best_chain: Optional[tuple[str, str]] = None
|
|
154
|
+
best_iters = 0
|
|
155
|
+
|
|
156
|
+
effective_chains = chains if chains is not None else _REPAIR_CHAINS
|
|
157
|
+
|
|
158
|
+
# Single-step: collect candidates, then pick lowest mess then preferred chain order.
|
|
159
|
+
options: list[tuple[tuple[str, str], str, float, int]] = []
|
|
160
|
+
for i, (re_encode, decode_as) in enumerate(effective_chains):
|
|
161
|
+
candidate = _try_chain(text, re_encode, decode_as)
|
|
162
|
+
if candidate is None or candidate == text:
|
|
163
|
+
continue
|
|
164
|
+
candidate_mess = mess_ratio(candidate)
|
|
165
|
+
strong_improvement = (original_mess - candidate_mess) >= _MIN_IMPROVEMENT
|
|
166
|
+
tie_utf8_mojibake = (
|
|
167
|
+
_likely_utf8_mojibake(text)
|
|
168
|
+
and (re_encode, decode_as) in _TIE_BREAK_CHAINS
|
|
169
|
+
and candidate_mess <= original_mess + 1e-9
|
|
170
|
+
)
|
|
171
|
+
if strong_improvement or tie_utf8_mojibake:
|
|
172
|
+
options.append(((re_encode, decode_as), candidate, candidate_mess, i))
|
|
173
|
+
|
|
174
|
+
if options:
|
|
175
|
+
options.sort(key=lambda x: (x[2], x[3]))
|
|
176
|
+
best_chain, best_text, best_mess, _ = (
|
|
177
|
+
options[0][0],
|
|
178
|
+
options[0][1],
|
|
179
|
+
options[0][2],
|
|
180
|
+
options[0][3],
|
|
181
|
+
)
|
|
182
|
+
best_iters = 1
|
|
183
|
+
|
|
184
|
+
# Double-step repair (only if single-step improved things)
|
|
185
|
+
if (
|
|
186
|
+
max_iterations >= 2
|
|
187
|
+
and best_iters == 1
|
|
188
|
+
and (best_mess > 0.05 or _likely_utf8_mojibake(best_text))
|
|
189
|
+
):
|
|
190
|
+
opts2: list[tuple[tuple[str, str], str, float, int]] = []
|
|
191
|
+
for i, (re_encode2, decode_as2) in enumerate(effective_chains):
|
|
192
|
+
candidate2 = _try_chain(best_text, re_encode2, decode_as2)
|
|
193
|
+
if candidate2 is None or candidate2 == best_text:
|
|
194
|
+
continue
|
|
195
|
+
candidate2_mess = mess_ratio(candidate2)
|
|
196
|
+
strong_improvement = (best_mess - candidate2_mess) >= _MIN_IMPROVEMENT
|
|
197
|
+
tie_utf8_mojibake = (
|
|
198
|
+
_likely_utf8_mojibake(best_text)
|
|
199
|
+
and (re_encode2, decode_as2) in _TIE_BREAK_CHAINS
|
|
200
|
+
and candidate2_mess <= best_mess + 1e-9
|
|
201
|
+
)
|
|
202
|
+
if strong_improvement or tie_utf8_mojibake:
|
|
203
|
+
opts2.append(((re_encode2, decode_as2), candidate2, candidate2_mess, i))
|
|
204
|
+
if opts2:
|
|
205
|
+
opts2.sort(key=lambda x: (x[2], x[3]))
|
|
206
|
+
best_chain = opts2[0][0]
|
|
207
|
+
best_text = opts2[0][1]
|
|
208
|
+
best_mess = opts2[0][2]
|
|
209
|
+
best_iters = 2
|
|
210
|
+
|
|
211
|
+
improved = best_iters > 0 and (
|
|
212
|
+
(original_mess - best_mess) >= _MIN_IMPROVEMENT
|
|
213
|
+
or (best_text != text and _likely_utf8_mojibake(text) and best_mess <= original_mess + 1e-9)
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
return RepairResult(
|
|
217
|
+
original=text,
|
|
218
|
+
repaired=best_text if improved else text,
|
|
219
|
+
improved=improved,
|
|
220
|
+
chain=best_chain if improved else None,
|
|
221
|
+
original_mess=original_mess,
|
|
222
|
+
repaired_mess=best_mess if improved else original_mess,
|
|
223
|
+
iterations=best_iters if improved else 0,
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def repair_bytes(
|
|
228
|
+
data: bytes,
|
|
229
|
+
encoding: Optional[str] = None,
|
|
230
|
+
*,
|
|
231
|
+
max_iterations: int = 2,
|
|
232
|
+
chains: Optional[List[tuple[str, str]]] = None,
|
|
233
|
+
) -> RepairResult:
|
|
234
|
+
"""
|
|
235
|
+
Repair mojibake in a byte sequence.
|
|
236
|
+
|
|
237
|
+
First decodes ``data`` with ``encoding`` (auto-detected if None),
|
|
238
|
+
then applies text-level repair.
|
|
239
|
+
|
|
240
|
+
Args:
|
|
241
|
+
data: Byte sequence to repair.
|
|
242
|
+
encoding: Known encoding. If None, auto-detected via ``from_bytes``.
|
|
243
|
+
max_iterations: Forwarded to :func:`repair`.
|
|
244
|
+
chains: Forwarded to :func:`repair`.
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
:class:`RepairResult`
|
|
248
|
+
"""
|
|
249
|
+
if encoding is None:
|
|
250
|
+
from .api import from_bytes as _fb
|
|
251
|
+
|
|
252
|
+
r = _fb(data)
|
|
253
|
+
if r.encoding is None:
|
|
254
|
+
raise UnicodeError("encoding is unknown; supply an explicit encoding")
|
|
255
|
+
enc = r.encoding
|
|
256
|
+
else:
|
|
257
|
+
enc = encoding
|
|
258
|
+
text = data.decode(enc, errors="strict")
|
|
259
|
+
|
|
260
|
+
return repair(text, max_iterations=max_iterations, chains=chains)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def is_mojibake(text: str, threshold: float = 0.15) -> bool:
|
|
264
|
+
"""
|
|
265
|
+
Quick heuristic to determine if text looks like mojibake.
|
|
266
|
+
|
|
267
|
+
Returns True if the text has high mess ratio AND attempting at least one
|
|
268
|
+
repair chain produces a significantly cleaner result.
|
|
269
|
+
|
|
270
|
+
Args:
|
|
271
|
+
text: String to check.
|
|
272
|
+
threshold: Mess ratio above which we attempt repair and check.
|
|
273
|
+
|
|
274
|
+
Returns:
|
|
275
|
+
bool
|
|
276
|
+
"""
|
|
277
|
+
if mess_ratio(text) < threshold and not _likely_utf8_mojibake(text):
|
|
278
|
+
return False
|
|
279
|
+
result = repair(text)
|
|
280
|
+
return result.improved
|
bytesense/scoring.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Compact character-pair evidence; optional native scoring has identical semantics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import gzip
|
|
6
|
+
import json
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections import Counter
|
|
9
|
+
from functools import lru_cache
|
|
10
|
+
from importlib.resources import files
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from ._rust import is_rust_available
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@lru_cache(maxsize=1)
|
|
17
|
+
def _model() -> tuple[frozenset[str], dict[str, float], Any]:
|
|
18
|
+
payload = json.loads(
|
|
19
|
+
gzip.decompress(files("bytesense.data").joinpath("language.json.gz").read_bytes())
|
|
20
|
+
)
|
|
21
|
+
letters, pairs = frozenset(payload["letters"]), payload["pairs"]
|
|
22
|
+
native = None
|
|
23
|
+
if is_rust_available():
|
|
24
|
+
from ._rust_core import NgramModel # type: ignore[import-not-found]
|
|
25
|
+
|
|
26
|
+
native = NgramModel(payload["letters"], list(payload["pairs"].items()))
|
|
27
|
+
return letters, pairs, native
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _prepare(text: str) -> tuple[str, float, float]:
|
|
31
|
+
if not text:
|
|
32
|
+
return "", 0.0, 0.0
|
|
33
|
+
counts = Counter(text)
|
|
34
|
+
bad = sum(n for c, n in counts.items() if not c.isprintable() and c not in "\n\r\t") / len(text)
|
|
35
|
+
symbols = sum(
|
|
36
|
+
n
|
|
37
|
+
for c, n in counts.items()
|
|
38
|
+
if c in "\\^`{|}~#"
|
|
39
|
+
or (
|
|
40
|
+
ord(c) >= 128
|
|
41
|
+
and not c.isalpha()
|
|
42
|
+
and not c.isspace()
|
|
43
|
+
and unicodedata.category(c)[0] not in "PN"
|
|
44
|
+
)
|
|
45
|
+
) / len(text)
|
|
46
|
+
lower = text.lower()
|
|
47
|
+
table = {ord(c): c if c.isalpha() else " " for c in set(lower)}
|
|
48
|
+
return lower.translate(table), bad, symbols
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _score_pure(
|
|
52
|
+
text: str, bad: float, symbols: float, letters: frozenset[str], pairs: dict[str, float]
|
|
53
|
+
) -> float:
|
|
54
|
+
counts = Counter(text)
|
|
55
|
+
total_letters = sum(n for c, n in counts.items() if c != " ")
|
|
56
|
+
known_letters = sum(n for c, n in counts.items() if c != " " and c in letters)
|
|
57
|
+
total = 0
|
|
58
|
+
known = 0.0
|
|
59
|
+
for (a, b), n in Counter(zip(text, text[1:])).items():
|
|
60
|
+
if a == b == " ":
|
|
61
|
+
continue
|
|
62
|
+
weight = n * (1 if (a + b).isascii() else 4)
|
|
63
|
+
total += weight
|
|
64
|
+
if a + b in pairs:
|
|
65
|
+
known += weight * pairs[a + b]
|
|
66
|
+
return (
|
|
67
|
+
0.25 * known_letters / max(1, total_letters)
|
|
68
|
+
+ 0.75 * known / max(1, total)
|
|
69
|
+
- bad * 5
|
|
70
|
+
- symbols * 3
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _classify(chars: str) -> list[tuple[bool, bool, bool]]:
|
|
75
|
+
"""Unicode scalar properties from this Python version; no document state."""
|
|
76
|
+
return [
|
|
77
|
+
(
|
|
78
|
+
c.isalpha(),
|
|
79
|
+
not c.isprintable() and c not in "\n\r\t",
|
|
80
|
+
c in "\\^`{|}~#"
|
|
81
|
+
or (
|
|
82
|
+
ord(c) >= 128
|
|
83
|
+
and not c.isalpha()
|
|
84
|
+
and not c.isspace()
|
|
85
|
+
and unicodedata.category(c)[0] not in "PN"
|
|
86
|
+
),
|
|
87
|
+
)
|
|
88
|
+
for c in chars
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def text_quality(text: str) -> tuple[float, float]:
|
|
93
|
+
"""Return linguistic support and control-character ratio (neither is a probability)."""
|
|
94
|
+
letters, pairs, native = _model()
|
|
95
|
+
if native:
|
|
96
|
+
score, bad = native.quality(text, text.lower(), _classify)
|
|
97
|
+
else:
|
|
98
|
+
normalized, bad, symbols = _prepare(text)
|
|
99
|
+
score = _score_pure(normalized, bad, symbols, letters, pairs)
|
|
100
|
+
return round(score, 10), bad
|