bytesense 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bytesense/__init__.py +47 -0
- bytesense/_rust.py +58 -0
- bytesense/api.py +469 -0
- bytesense/candidate.py +199 -0
- bytesense/cli.py +73 -0
- bytesense/coherence.py +68 -0
- bytesense/constant.py +1313 -0
- bytesense/data/__init__.py +0 -0
- bytesense/data/fingerprints.py +94 -0
- bytesense/data/language.json.gz +0 -0
- bytesense/fingerprint.py +246 -0
- bytesense/heuristics.py +161 -0
- bytesense/hints.py +104 -0
- bytesense/legacy.py +38 -0
- bytesense/mess.py +179 -0
- bytesense/models.py +98 -0
- bytesense/multi.py +209 -0
- bytesense/py.typed +0 -0
- bytesense/repair.py +280 -0
- bytesense/scoring.py +100 -0
- bytesense/streaming.py +416 -0
- bytesense/version.py +4 -0
- bytesense-1.0.0.dist-info/METADATA +160 -0
- bytesense-1.0.0.dist-info/RECORD +28 -0
- bytesense-1.0.0.dist-info/WHEEL +5 -0
- bytesense-1.0.0.dist-info/entry_points.txt +2 -0
- bytesense-1.0.0.dist-info/licenses/LICENSE +21 -0
- bytesense-1.0.0.dist-info/top_level.txt +1 -0
bytesense/fingerprint.py
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Byte-distribution fingerprinting engine.
|
|
3
|
+
|
|
4
|
+
The central insight: every encoding has a characteristic byte-frequency
|
|
5
|
+
"signature". By computing the cosine similarity between the observed
|
|
6
|
+
byte histogram and pre-computed encoding fingerprints, we can shortlist
|
|
7
|
+
likely encodings in O(n) time without any decoding.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import array
|
|
13
|
+
from collections import Counter
|
|
14
|
+
from functools import lru_cache
|
|
15
|
+
from typing import List, Tuple
|
|
16
|
+
|
|
17
|
+
# ---------------------------------------------------------------------------
|
|
18
|
+
# Core histogram (may be replaced by Rust at module load time — see _rust.py)
|
|
19
|
+
# ---------------------------------------------------------------------------
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _byte_histogram_pure(data: bytes) -> array.array:
|
|
23
|
+
"""Pure-Python byte histogram (also used when the Rust extension is absent)."""
|
|
24
|
+
hist: array.array = array.array("Q", [0] * 256)
|
|
25
|
+
for value, count in Counter(data).items():
|
|
26
|
+
hist[value] = count
|
|
27
|
+
return hist
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def histogram_to_ratios(hist: array.array, total: int) -> List[float]:
|
|
31
|
+
"""Convert raw counts to frequency ratios."""
|
|
32
|
+
if total == 0:
|
|
33
|
+
return [0.0] * 256
|
|
34
|
+
inv = 1.0 / total
|
|
35
|
+
return [c * inv for c in hist]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def high_byte_ratio(hist: array.array, total: int) -> float:
|
|
39
|
+
"""Fraction of bytes with value >= 0x80."""
|
|
40
|
+
if total == 0:
|
|
41
|
+
return 0.0
|
|
42
|
+
return sum(hist[0x80:]) / total
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def null_byte_ratio(hist: array.array, total: int) -> float:
|
|
46
|
+
"""Fraction of 0x00 bytes."""
|
|
47
|
+
if total == 0:
|
|
48
|
+
return 0.0
|
|
49
|
+
return hist[0] / total
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def cp1252_zone_ratio(hist: array.array, total: int) -> float:
|
|
53
|
+
"""
|
|
54
|
+
Fraction of bytes in 0x80–0x9F.
|
|
55
|
+
Non-zero → almost certainly cp1252 family, NOT latin_1
|
|
56
|
+
(latin_1 treats 0x80-0x9F as C1 control codes; cp1252 maps them to printable chars).
|
|
57
|
+
"""
|
|
58
|
+
if total == 0:
|
|
59
|
+
return 0.0
|
|
60
|
+
return sum(hist[0x80:0xA0]) / total
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _utf8_continuation_score_pure(data: bytes) -> float:
|
|
64
|
+
"""
|
|
65
|
+
Pure-Python UTF-8 continuation scoring (also used when Rust is absent).
|
|
66
|
+
Score how well `data` fits UTF-8 multibyte sequence structure.
|
|
67
|
+
Returns 0.0 (no evidence) to 1.0 (strong UTF-8 multibyte pattern).
|
|
68
|
+
Works even when the data is not fully valid UTF-8.
|
|
69
|
+
"""
|
|
70
|
+
if not data:
|
|
71
|
+
return 0.0
|
|
72
|
+
|
|
73
|
+
valid = 0
|
|
74
|
+
invalid = 0
|
|
75
|
+
i = 0
|
|
76
|
+
n = len(data)
|
|
77
|
+
|
|
78
|
+
while i < n:
|
|
79
|
+
b = data[i]
|
|
80
|
+
if b < 0x80:
|
|
81
|
+
i += 1
|
|
82
|
+
continue
|
|
83
|
+
elif 0xC2 <= b <= 0xDF:
|
|
84
|
+
seq_len = 2
|
|
85
|
+
elif 0xE0 <= b <= 0xEF:
|
|
86
|
+
seq_len = 3
|
|
87
|
+
elif 0xF0 <= b <= 0xF4:
|
|
88
|
+
seq_len = 4
|
|
89
|
+
else:
|
|
90
|
+
invalid += 1
|
|
91
|
+
i += 1
|
|
92
|
+
continue
|
|
93
|
+
|
|
94
|
+
if i + seq_len > n:
|
|
95
|
+
invalid += 1
|
|
96
|
+
i += 1
|
|
97
|
+
continue
|
|
98
|
+
|
|
99
|
+
ok = all(0x80 <= data[i + j] <= 0xBF for j in range(1, seq_len))
|
|
100
|
+
if ok:
|
|
101
|
+
valid += 1
|
|
102
|
+
i += seq_len
|
|
103
|
+
else:
|
|
104
|
+
invalid += 1
|
|
105
|
+
i += 1
|
|
106
|
+
|
|
107
|
+
total = valid + invalid
|
|
108
|
+
return valid / total if total > 0 else 0.0
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def detect_null_pattern(data: bytes) -> str | None:
|
|
112
|
+
"""
|
|
113
|
+
Detect UTF-16/32 from null-byte distribution pattern.
|
|
114
|
+
Returns encoding name or None.
|
|
115
|
+
"""
|
|
116
|
+
if len(data) < 8:
|
|
117
|
+
return None
|
|
118
|
+
|
|
119
|
+
sample = data[:512]
|
|
120
|
+
# Distinguish byte lanes, including non-Latin UTF-16 and supplementary UTF-32.
|
|
121
|
+
lanes = [sample[i::4] for i in range(4)]
|
|
122
|
+
zero = [lane.count(0) / len(lane) for lane in lanes]
|
|
123
|
+
if zero[0] > 0.8 and zero[1] > 0.8 and zero[3] < 0.5:
|
|
124
|
+
return "utf_32_be"
|
|
125
|
+
if zero[2] > 0.8 and zero[3] > 0.8 and zero[0] < 0.5:
|
|
126
|
+
return "utf_32_le"
|
|
127
|
+
even, odd = sample[0::2], sample[1::2]
|
|
128
|
+
ze, zo = even.count(0) / len(even), odd.count(0) / len(odd)
|
|
129
|
+
if ze >= 0.08 and zo < ze * 0.1:
|
|
130
|
+
return "utf_16_be"
|
|
131
|
+
if zo >= 0.08 and ze < zo * 0.1:
|
|
132
|
+
return "utf_16_le"
|
|
133
|
+
return None
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _cosine_similarity(a: List[float], b: List[float]) -> float:
|
|
137
|
+
"""Cosine similarity between two equal-length vectors."""
|
|
138
|
+
dot = sum(x * y for x, y in zip(a, b))
|
|
139
|
+
mag_a = sum(x * x for x in a) ** 0.5
|
|
140
|
+
mag_b = sum(y * y for y in b) ** 0.5
|
|
141
|
+
if mag_a == 0.0 or mag_b == 0.0:
|
|
142
|
+
return 0.0
|
|
143
|
+
return dot / (mag_a * mag_b)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def shortlist_encodings(
|
|
147
|
+
hist: array.array,
|
|
148
|
+
total: int,
|
|
149
|
+
top_n: int = 12,
|
|
150
|
+
) -> List[Tuple[str, float]]:
|
|
151
|
+
"""
|
|
152
|
+
Use the pre-computed fingerprint table to rank all encodings by
|
|
153
|
+
byte-distribution similarity, returning the top_n most likely candidates.
|
|
154
|
+
O(k) where k = number of fingerprints (~99). No decoding performed.
|
|
155
|
+
"""
|
|
156
|
+
scores = list(_scores_for_hist(hist).items())
|
|
157
|
+
|
|
158
|
+
scores.sort(key=lambda x: x[1], reverse=True)
|
|
159
|
+
return scores[:top_n]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def fingerprint_cosine_for_encoding(data: bytes, encoding: str) -> float:
|
|
163
|
+
"""
|
|
164
|
+
Cosine similarity (0..1) between `data`'s byte histogram and the
|
|
165
|
+
precomputed fingerprint for `encoding`. Used to break ties when several
|
|
166
|
+
decodings look linguistically plausible (e.g. Big5 bytes mis-read as cp949).
|
|
167
|
+
"""
|
|
168
|
+
try:
|
|
169
|
+
from .data.fingerprints import ENCODING_FINGERPRINTS
|
|
170
|
+
except ImportError:
|
|
171
|
+
return 0.0
|
|
172
|
+
fp = ENCODING_FINGERPRINTS.get(encoding)
|
|
173
|
+
if fp is None:
|
|
174
|
+
return 0.0
|
|
175
|
+
n = len(data)
|
|
176
|
+
if n == 0:
|
|
177
|
+
return 0.0
|
|
178
|
+
hist = byte_histogram(data)
|
|
179
|
+
ratios = histogram_to_ratios(hist, n)
|
|
180
|
+
return _cosine_similarity(ratios, fp)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
@lru_cache(maxsize=1)
|
|
184
|
+
def _fingerprint_groups() -> list[tuple[list[str], tuple[float, ...]]]:
|
|
185
|
+
from .data.fingerprints import ENCODING_FINGERPRINTS
|
|
186
|
+
|
|
187
|
+
groups: dict[tuple[float, ...], list[str]] = {}
|
|
188
|
+
for encoding, raw_vector in ENCODING_FINGERPRINTS.items():
|
|
189
|
+
groups.setdefault(tuple(raw_vector), []).append(encoding)
|
|
190
|
+
result = []
|
|
191
|
+
for vector, names in groups.items():
|
|
192
|
+
norm = sum(v * v for v in vector) ** 0.5
|
|
193
|
+
result.append((names, tuple(v / norm if norm else 0.0 for v in vector)))
|
|
194
|
+
return result
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# Static profiles only: identical vectors share one dot product.
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _scores_for_hist(hist: array.array) -> dict[str, float]:
|
|
201
|
+
norm = sum(n * n for n in hist) ** 0.5
|
|
202
|
+
if not norm:
|
|
203
|
+
return {name: 0.0 for names, _ in _fingerprint_groups() for name in names}
|
|
204
|
+
observed = [(i, n / norm) for i, n in enumerate(hist) if n]
|
|
205
|
+
scores: dict[str, float] = {}
|
|
206
|
+
for names, vector in _fingerprint_groups():
|
|
207
|
+
score = sum(n * vector[i] for i, n in observed)
|
|
208
|
+
scores.update((name, score) for name in names)
|
|
209
|
+
# Restore table order to preserve stable ties.
|
|
210
|
+
from .data.fingerprints import ENCODING_FINGERPRINTS
|
|
211
|
+
|
|
212
|
+
return {name: scores[name] for name in ENCODING_FINGERPRINTS}
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def fingerprint_scores(data: bytes) -> dict[str, float]:
|
|
216
|
+
"""Score distinct static profiles from one sparse byte histogram."""
|
|
217
|
+
return _scores_for_hist(byte_histogram(data))
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
from ._rust import ( # noqa: E402 — intentional late import
|
|
221
|
+
_RUST_AVAILABLE,
|
|
222
|
+
rust_byte_histogram,
|
|
223
|
+
rust_utf8_continuation_score,
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def byte_histogram(data: bytes) -> array.array:
|
|
228
|
+
"""
|
|
229
|
+
Compute byte frequency histogram.
|
|
230
|
+
Returns a 256-element ``array.array("Q", ...)`` of occurrence counts.
|
|
231
|
+
O(n), single pass. Uses Rust when available.
|
|
232
|
+
"""
|
|
233
|
+
if _RUST_AVAILABLE:
|
|
234
|
+
return rust_byte_histogram(data)
|
|
235
|
+
return _byte_histogram_pure(data)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def utf8_continuation_score(data: bytes) -> float:
|
|
239
|
+
"""
|
|
240
|
+
Score how well `data` fits UTF-8 multibyte sequence structure.
|
|
241
|
+
Returns 0.0 (no evidence) to 1.0 (strong UTF-8 multibyte pattern).
|
|
242
|
+
Works even when the data is not fully valid UTF-8. Uses Rust when available.
|
|
243
|
+
"""
|
|
244
|
+
if _RUST_AVAILABLE:
|
|
245
|
+
return rust_utf8_continuation_score(data)
|
|
246
|
+
return _utf8_continuation_score_pure(data)
|
bytesense/heuristics.py
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Raw-byte priors for candidate ordering (no external detectors).
|
|
3
|
+
|
|
4
|
+
Heuristic ranges are approximate; they only reorder the existing shortlist so
|
|
5
|
+
likely encodings are tried earlier before decode+mess ranking.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import List, Optional
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _bump_to_front(candidates: List[str], *want: str) -> List[str]:
|
|
13
|
+
seen: set[str] = set()
|
|
14
|
+
front: List[str] = []
|
|
15
|
+
for e in want:
|
|
16
|
+
if e in candidates and e not in seen:
|
|
17
|
+
front.append(e)
|
|
18
|
+
seen.add(e)
|
|
19
|
+
rest = [c for c in candidates if c not in seen]
|
|
20
|
+
return front + rest
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def hebrew_sbcs_likelihood(data: bytes) -> float:
|
|
24
|
+
"""Share of high bytes in typical Windows-1255 Hebrew letter range (0xE0–0xFA)."""
|
|
25
|
+
if len(data) < 8:
|
|
26
|
+
return 0.0
|
|
27
|
+
hi = sum(1 for b in data if b >= 0x80)
|
|
28
|
+
if hi < 8:
|
|
29
|
+
return 0.0
|
|
30
|
+
heb = sum(1 for b in data if 0xE0 <= b <= 0xFA)
|
|
31
|
+
return heb / hi
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def thai_tis620_likelihood(data: bytes) -> float:
|
|
35
|
+
"""
|
|
36
|
+
Share of high bytes in the lower TIS-620 Thai band (0xA1–0xDF).
|
|
37
|
+
|
|
38
|
+
Excludes 0xE0–0xFA, which overlaps Windows-1255 Hebrew and inflates false Thai scores.
|
|
39
|
+
"""
|
|
40
|
+
if len(data) < 8:
|
|
41
|
+
return 0.0
|
|
42
|
+
hi = sum(1 for b in data if b >= 0x80)
|
|
43
|
+
if hi < 8:
|
|
44
|
+
return 0.0
|
|
45
|
+
th = sum(1 for b in data if 0xA1 <= b <= 0xDF)
|
|
46
|
+
return th / hi
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def koi8_byte_hint(data: bytes) -> bool:
|
|
50
|
+
"""KOI8-R Cyrillic uses many high bytes in 0xC0–0xFF — avoid false Hebrew reorder."""
|
|
51
|
+
hi = [b for b in data if b >= 0x80]
|
|
52
|
+
if len(hi) < 24:
|
|
53
|
+
return False
|
|
54
|
+
block = sum(1 for b in hi if 0xC0 <= b <= 0xFF)
|
|
55
|
+
return block / len(hi) >= 0.58
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def cp866_vs_cp1251_hint(data: bytes) -> Optional[str]:
|
|
59
|
+
"""DOS (cp866) block 0x80–0xAF vs Windows (cp1251) 0xC0–0xFF — rough split."""
|
|
60
|
+
hi = [b for b in data if b >= 0x80]
|
|
61
|
+
if len(hi) < 20:
|
|
62
|
+
return None
|
|
63
|
+
b866 = sum(1 for b in hi if 0x80 <= b <= 0xAF)
|
|
64
|
+
b1251 = sum(1 for b in hi if 0xC0 <= b <= 0xFF)
|
|
65
|
+
t = len(hi)
|
|
66
|
+
if b866 / t >= 0.40 and b1251 / t <= 0.42:
|
|
67
|
+
return "cp866"
|
|
68
|
+
if b1251 / t >= 0.40 and b866 / t <= 0.38:
|
|
69
|
+
return "cp1251"
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def japanese_mbcs_bias(data: bytes) -> Optional[str]:
|
|
74
|
+
"""Rough Shift_JIS vs EUC-JP from multi-byte pair patterns."""
|
|
75
|
+
if len(data) < 24:
|
|
76
|
+
return None
|
|
77
|
+
sj = 0
|
|
78
|
+
ej = 0
|
|
79
|
+
for i in range(len(data) - 1):
|
|
80
|
+
b1, b2 = data[i], data[i + 1]
|
|
81
|
+
if 0x81 <= b1 <= 0x9F or 0xE0 <= b1 <= 0xFC:
|
|
82
|
+
if 0x40 <= b2 <= 0xFC and b2 != 0x7F:
|
|
83
|
+
sj += 1
|
|
84
|
+
if 0xA1 <= b1 <= 0xFE and 0xA1 <= b2 <= 0xFE:
|
|
85
|
+
ej += 1
|
|
86
|
+
if b1 == 0x8E and 0xA1 <= b2 <= 0xDF:
|
|
87
|
+
ej += 2
|
|
88
|
+
if sj + ej < 14:
|
|
89
|
+
return None
|
|
90
|
+
# CP866 Cyrillic: moderate EUC-like pair counts without true JIS dominance — not Japanese.
|
|
91
|
+
hint = cp866_vs_cp1251_hint(data)
|
|
92
|
+
if hint == "cp866":
|
|
93
|
+
if ej > sj * 5.0:
|
|
94
|
+
return "euc_jp"
|
|
95
|
+
if sj > ej * 1.4 and sj >= 40:
|
|
96
|
+
return "shift_jis"
|
|
97
|
+
return None
|
|
98
|
+
if sj > ej * 1.2:
|
|
99
|
+
return "shift_jis"
|
|
100
|
+
if ej > sj * 1.2:
|
|
101
|
+
return "euc_jp"
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def chinese_big5_vs_gb_hint(data: bytes) -> Optional[str]:
|
|
106
|
+
"""
|
|
107
|
+
GB18030 can use 4-byte sequences; Big5 is overwhelmingly two-byte pairs.
|
|
108
|
+
Very rough — ranking still validates via decode+mess.
|
|
109
|
+
"""
|
|
110
|
+
if len(data) < 32:
|
|
111
|
+
return None
|
|
112
|
+
quadish = 0
|
|
113
|
+
for i in range(len(data) - 3):
|
|
114
|
+
if 0x81 <= data[i] <= 0xFE and 0x30 <= data[i + 1] <= 0x39:
|
|
115
|
+
quadish += 1
|
|
116
|
+
if quadish >= max(4, len(data) // 400):
|
|
117
|
+
return "gb18030"
|
|
118
|
+
pairs = 0
|
|
119
|
+
for i in range(len(data) - 1):
|
|
120
|
+
b1, b2 = data[i], data[i + 1]
|
|
121
|
+
if 0xA1 <= b1 <= 0xFE and 0x40 <= b2 <= 0xFE:
|
|
122
|
+
pairs += 1
|
|
123
|
+
if pairs >= len(data) // 6:
|
|
124
|
+
return "big5"
|
|
125
|
+
return None
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def reorder_candidates(data: bytes, candidates: List[str]) -> List[str]:
|
|
129
|
+
"""Apply raw-byte hints by moving likely encodings earlier."""
|
|
130
|
+
out = list(candidates)
|
|
131
|
+
|
|
132
|
+
if (
|
|
133
|
+
hebrew_sbcs_likelihood(data) >= 0.50
|
|
134
|
+
and cp866_vs_cp1251_hint(data) is None
|
|
135
|
+
and not koi8_byte_hint(data)
|
|
136
|
+
):
|
|
137
|
+
out = _bump_to_front(out, "cp1255", "iso8859_8")
|
|
138
|
+
|
|
139
|
+
if thai_tis620_likelihood(data) >= 0.52 and hebrew_sbcs_likelihood(data) < 0.55:
|
|
140
|
+
out = _bump_to_front(out, "tis_620", "iso8859_11")
|
|
141
|
+
|
|
142
|
+
hint = cp866_vs_cp1251_hint(data)
|
|
143
|
+
if hint == "cp866":
|
|
144
|
+
out = _bump_to_front(out, "cp866", "koi8_r", "cp1251")
|
|
145
|
+
elif hint == "cp1251":
|
|
146
|
+
out = _bump_to_front(out, "cp1251", "koi8_r", "cp866")
|
|
147
|
+
|
|
148
|
+
jb = japanese_mbcs_bias(data)
|
|
149
|
+
if jb == "shift_jis":
|
|
150
|
+
out = _bump_to_front(out, "shift_jis", "cp932", "euc_jp", "iso2022_jp")
|
|
151
|
+
elif jb == "euc_jp":
|
|
152
|
+
out = _bump_to_front(out, "euc_jp", "iso2022_jp", "shift_jis", "cp932")
|
|
153
|
+
|
|
154
|
+
if jb is None:
|
|
155
|
+
ch = chinese_big5_vs_gb_hint(data)
|
|
156
|
+
if ch == "gb18030":
|
|
157
|
+
out = _bump_to_front(out, "gb18030", "gbk", "gb2312", "big5", "big5hkscs")
|
|
158
|
+
elif ch == "big5":
|
|
159
|
+
out = _bump_to_front(out, "big5", "big5hkscs", "gb18030", "gbk", "gb2312")
|
|
160
|
+
|
|
161
|
+
return out
|
bytesense/hints.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Encoding hint extraction from HTTP headers and HTML/XML documents.
|
|
3
|
+
|
|
4
|
+
Used internally by StreamDetector, but also available as a public API
|
|
5
|
+
for callers who already have the raw content and headers.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import codecs
|
|
11
|
+
import re
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
_XML_DECL_RE = re.compile(rb'<\?xml[^>]*encoding=["\']([^"\']+)["\']', re.IGNORECASE)
|
|
15
|
+
_HTML_META_RE = re.compile(rb'<meta[^>]+charset\s*=\s*["\']?\s*([a-zA-Z0-9_\-]+)', re.IGNORECASE)
|
|
16
|
+
_HTTP_EQUIV_RE = re.compile(
|
|
17
|
+
rb'<meta[^>]+http-equiv\s*=\s*["\']?content-type["\']?[^>]*'
|
|
18
|
+
rb'content\s*=\s*["\']?[^"\']*charset=([a-zA-Z0-9_\-]+)',
|
|
19
|
+
re.IGNORECASE,
|
|
20
|
+
)
|
|
21
|
+
_HTTP_HEADER_RE = re.compile(r'charset\s*=\s*["\']?\s*([a-zA-Z0-9_\-]+)', re.IGNORECASE)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _normalise(enc_str: str) -> Optional[str]:
|
|
25
|
+
try:
|
|
26
|
+
return codecs.lookup(enc_str.strip()).name.replace("-", "_")
|
|
27
|
+
except LookupError:
|
|
28
|
+
return None
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def hint_from_http_headers(headers: dict[str, str]) -> Optional[str]:
|
|
32
|
+
"""
|
|
33
|
+
Extract encoding hint from HTTP response headers.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
headers: Dict of header name → value (case-insensitive matching).
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
Normalised IANA encoding name or None.
|
|
40
|
+
|
|
41
|
+
Example::
|
|
42
|
+
|
|
43
|
+
enc = hint_from_http_headers({"Content-Type": "text/html; charset=utf-8"})
|
|
44
|
+
# → "utf_8"
|
|
45
|
+
"""
|
|
46
|
+
ct = next((value for name, value in headers.items() if name.lower() == "content-type"), "")
|
|
47
|
+
m = _HTTP_HEADER_RE.search(ct)
|
|
48
|
+
if m:
|
|
49
|
+
return _normalise(m.group(1))
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def hint_from_content(data: bytes, max_scan_bytes: int = 4096) -> Optional[str]:
|
|
54
|
+
"""
|
|
55
|
+
Extract encoding hint from the first ``max_scan_bytes`` of an HTML or XML document.
|
|
56
|
+
|
|
57
|
+
Looks for:
|
|
58
|
+
- XML declaration: ``<?xml version="1.0" encoding="UTF-8"?>``
|
|
59
|
+
- HTML meta charset: ``<meta charset="utf-8">``
|
|
60
|
+
- HTML http-equiv: ``<meta http-equiv="Content-Type" content="text/html; charset=utf-8">``
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
data: Raw bytes (need not be fully decoded).
|
|
64
|
+
max_scan_bytes: How far into the document to scan.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
Normalised IANA encoding name or None.
|
|
68
|
+
"""
|
|
69
|
+
probe = data[:max_scan_bytes]
|
|
70
|
+
for pattern in (_XML_DECL_RE, _HTML_META_RE, _HTTP_EQUIV_RE):
|
|
71
|
+
m = pattern.search(probe)
|
|
72
|
+
if m:
|
|
73
|
+
try:
|
|
74
|
+
enc_str = m.group(1).decode("ascii", errors="ignore")
|
|
75
|
+
result = _normalise(enc_str)
|
|
76
|
+
if result:
|
|
77
|
+
return result
|
|
78
|
+
except Exception:
|
|
79
|
+
pass
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def best_hint(
|
|
84
|
+
data: bytes,
|
|
85
|
+
headers: Optional[dict[str, str]] = None,
|
|
86
|
+
max_scan_bytes: int = 4096,
|
|
87
|
+
) -> Optional[str]:
|
|
88
|
+
"""
|
|
89
|
+
Return the best encoding hint from HTTP headers and/or document content.
|
|
90
|
+
|
|
91
|
+
HTTP headers take priority (more reliable than meta tags).
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
data: Raw document bytes.
|
|
95
|
+
headers: HTTP response headers dict (optional).
|
|
96
|
+
|
|
97
|
+
Returns:
|
|
98
|
+
Normalised IANA encoding name or None.
|
|
99
|
+
"""
|
|
100
|
+
if headers:
|
|
101
|
+
h = hint_from_http_headers(headers)
|
|
102
|
+
if h:
|
|
103
|
+
return h
|
|
104
|
+
return hint_from_content(data, max_scan_bytes)
|
bytesense/legacy.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""
|
|
2
|
+
chardet / charset-normalizer drop-in compatibility layer.
|
|
3
|
+
|
|
4
|
+
Migration:
|
|
5
|
+
# Before
|
|
6
|
+
from chardet import detect
|
|
7
|
+
# or
|
|
8
|
+
from charset_normalizer import detect
|
|
9
|
+
|
|
10
|
+
# After
|
|
11
|
+
from bytesense import detect # identical interface
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Dict, Optional, Union
|
|
16
|
+
|
|
17
|
+
from .api import from_bytes
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def detect(byte_str: Union[bytes, bytearray]) -> Dict[str, Optional[object]]:
|
|
21
|
+
"""
|
|
22
|
+
Drop-in replacement for ``chardet.detect()`` and
|
|
23
|
+
``charset_normalizer.detect()``.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
byte_str: The byte sequence to examine.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
``{"encoding": str|None, "confidence": float|None, "language": str}``
|
|
30
|
+
"""
|
|
31
|
+
if not isinstance(byte_str, (bytes, bytearray)):
|
|
32
|
+
raise TypeError(f"Expected bytes or bytearray, got {type(byte_str).__name__!r}")
|
|
33
|
+
result = from_bytes(bytes(byte_str))
|
|
34
|
+
return {
|
|
35
|
+
"encoding": result.encoding,
|
|
36
|
+
"confidence": result.confidence if result.encoding else None,
|
|
37
|
+
"language": result.language,
|
|
38
|
+
}
|