bytesense 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,246 @@
1
+ """
2
+ Byte-distribution fingerprinting engine.
3
+
4
+ The central insight: every encoding has a characteristic byte-frequency
5
+ "signature". By computing the cosine similarity between the observed
6
+ byte histogram and pre-computed encoding fingerprints, we can shortlist
7
+ likely encodings in O(n) time without any decoding.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import array
13
+ from collections import Counter
14
+ from functools import lru_cache
15
+ from typing import List, Tuple
16
+
17
+ # ---------------------------------------------------------------------------
18
+ # Core histogram (may be replaced by Rust at module load time — see _rust.py)
19
+ # ---------------------------------------------------------------------------
20
+
21
+
22
+ def _byte_histogram_pure(data: bytes) -> array.array:
23
+ """Pure-Python byte histogram (also used when the Rust extension is absent)."""
24
+ hist: array.array = array.array("Q", [0] * 256)
25
+ for value, count in Counter(data).items():
26
+ hist[value] = count
27
+ return hist
28
+
29
+
30
+ def histogram_to_ratios(hist: array.array, total: int) -> List[float]:
31
+ """Convert raw counts to frequency ratios."""
32
+ if total == 0:
33
+ return [0.0] * 256
34
+ inv = 1.0 / total
35
+ return [c * inv for c in hist]
36
+
37
+
38
+ def high_byte_ratio(hist: array.array, total: int) -> float:
39
+ """Fraction of bytes with value >= 0x80."""
40
+ if total == 0:
41
+ return 0.0
42
+ return sum(hist[0x80:]) / total
43
+
44
+
45
+ def null_byte_ratio(hist: array.array, total: int) -> float:
46
+ """Fraction of 0x00 bytes."""
47
+ if total == 0:
48
+ return 0.0
49
+ return hist[0] / total
50
+
51
+
52
+ def cp1252_zone_ratio(hist: array.array, total: int) -> float:
53
+ """
54
+ Fraction of bytes in 0x80–0x9F.
55
+ Non-zero → almost certainly cp1252 family, NOT latin_1
56
+ (latin_1 treats 0x80-0x9F as C1 control codes; cp1252 maps them to printable chars).
57
+ """
58
+ if total == 0:
59
+ return 0.0
60
+ return sum(hist[0x80:0xA0]) / total
61
+
62
+
63
+ def _utf8_continuation_score_pure(data: bytes) -> float:
64
+ """
65
+ Pure-Python UTF-8 continuation scoring (also used when Rust is absent).
66
+ Score how well `data` fits UTF-8 multibyte sequence structure.
67
+ Returns 0.0 (no evidence) to 1.0 (strong UTF-8 multibyte pattern).
68
+ Works even when the data is not fully valid UTF-8.
69
+ """
70
+ if not data:
71
+ return 0.0
72
+
73
+ valid = 0
74
+ invalid = 0
75
+ i = 0
76
+ n = len(data)
77
+
78
+ while i < n:
79
+ b = data[i]
80
+ if b < 0x80:
81
+ i += 1
82
+ continue
83
+ elif 0xC2 <= b <= 0xDF:
84
+ seq_len = 2
85
+ elif 0xE0 <= b <= 0xEF:
86
+ seq_len = 3
87
+ elif 0xF0 <= b <= 0xF4:
88
+ seq_len = 4
89
+ else:
90
+ invalid += 1
91
+ i += 1
92
+ continue
93
+
94
+ if i + seq_len > n:
95
+ invalid += 1
96
+ i += 1
97
+ continue
98
+
99
+ ok = all(0x80 <= data[i + j] <= 0xBF for j in range(1, seq_len))
100
+ if ok:
101
+ valid += 1
102
+ i += seq_len
103
+ else:
104
+ invalid += 1
105
+ i += 1
106
+
107
+ total = valid + invalid
108
+ return valid / total if total > 0 else 0.0
109
+
110
+
111
+ def detect_null_pattern(data: bytes) -> str | None:
112
+ """
113
+ Detect UTF-16/32 from null-byte distribution pattern.
114
+ Returns encoding name or None.
115
+ """
116
+ if len(data) < 8:
117
+ return None
118
+
119
+ sample = data[:512]
120
+ # Distinguish byte lanes, including non-Latin UTF-16 and supplementary UTF-32.
121
+ lanes = [sample[i::4] for i in range(4)]
122
+ zero = [lane.count(0) / len(lane) for lane in lanes]
123
+ if zero[0] > 0.8 and zero[1] > 0.8 and zero[3] < 0.5:
124
+ return "utf_32_be"
125
+ if zero[2] > 0.8 and zero[3] > 0.8 and zero[0] < 0.5:
126
+ return "utf_32_le"
127
+ even, odd = sample[0::2], sample[1::2]
128
+ ze, zo = even.count(0) / len(even), odd.count(0) / len(odd)
129
+ if ze >= 0.08 and zo < ze * 0.1:
130
+ return "utf_16_be"
131
+ if zo >= 0.08 and ze < zo * 0.1:
132
+ return "utf_16_le"
133
+ return None
134
+
135
+
136
+ def _cosine_similarity(a: List[float], b: List[float]) -> float:
137
+ """Cosine similarity between two equal-length vectors."""
138
+ dot = sum(x * y for x, y in zip(a, b))
139
+ mag_a = sum(x * x for x in a) ** 0.5
140
+ mag_b = sum(y * y for y in b) ** 0.5
141
+ if mag_a == 0.0 or mag_b == 0.0:
142
+ return 0.0
143
+ return dot / (mag_a * mag_b)
144
+
145
+
146
+ def shortlist_encodings(
147
+ hist: array.array,
148
+ total: int,
149
+ top_n: int = 12,
150
+ ) -> List[Tuple[str, float]]:
151
+ """
152
+ Use the pre-computed fingerprint table to rank all encodings by
153
+ byte-distribution similarity, returning the top_n most likely candidates.
154
+ O(k) where k = number of fingerprints (~99). No decoding performed.
155
+ """
156
+ scores = list(_scores_for_hist(hist).items())
157
+
158
+ scores.sort(key=lambda x: x[1], reverse=True)
159
+ return scores[:top_n]
160
+
161
+
162
+ def fingerprint_cosine_for_encoding(data: bytes, encoding: str) -> float:
163
+ """
164
+ Cosine similarity (0..1) between `data`'s byte histogram and the
165
+ precomputed fingerprint for `encoding`. Used to break ties when several
166
+ decodings look linguistically plausible (e.g. Big5 bytes mis-read as cp949).
167
+ """
168
+ try:
169
+ from .data.fingerprints import ENCODING_FINGERPRINTS
170
+ except ImportError:
171
+ return 0.0
172
+ fp = ENCODING_FINGERPRINTS.get(encoding)
173
+ if fp is None:
174
+ return 0.0
175
+ n = len(data)
176
+ if n == 0:
177
+ return 0.0
178
+ hist = byte_histogram(data)
179
+ ratios = histogram_to_ratios(hist, n)
180
+ return _cosine_similarity(ratios, fp)
181
+
182
+
183
+ @lru_cache(maxsize=1)
184
+ def _fingerprint_groups() -> list[tuple[list[str], tuple[float, ...]]]:
185
+ from .data.fingerprints import ENCODING_FINGERPRINTS
186
+
187
+ groups: dict[tuple[float, ...], list[str]] = {}
188
+ for encoding, raw_vector in ENCODING_FINGERPRINTS.items():
189
+ groups.setdefault(tuple(raw_vector), []).append(encoding)
190
+ result = []
191
+ for vector, names in groups.items():
192
+ norm = sum(v * v for v in vector) ** 0.5
193
+ result.append((names, tuple(v / norm if norm else 0.0 for v in vector)))
194
+ return result
195
+
196
+
197
+ # Static profiles only: identical vectors share one dot product.
198
+
199
+
200
+ def _scores_for_hist(hist: array.array) -> dict[str, float]:
201
+ norm = sum(n * n for n in hist) ** 0.5
202
+ if not norm:
203
+ return {name: 0.0 for names, _ in _fingerprint_groups() for name in names}
204
+ observed = [(i, n / norm) for i, n in enumerate(hist) if n]
205
+ scores: dict[str, float] = {}
206
+ for names, vector in _fingerprint_groups():
207
+ score = sum(n * vector[i] for i, n in observed)
208
+ scores.update((name, score) for name in names)
209
+ # Restore table order to preserve stable ties.
210
+ from .data.fingerprints import ENCODING_FINGERPRINTS
211
+
212
+ return {name: scores[name] for name in ENCODING_FINGERPRINTS}
213
+
214
+
215
+ def fingerprint_scores(data: bytes) -> dict[str, float]:
216
+ """Score distinct static profiles from one sparse byte histogram."""
217
+ return _scores_for_hist(byte_histogram(data))
218
+
219
+
220
+ from ._rust import ( # noqa: E402 — intentional late import
221
+ _RUST_AVAILABLE,
222
+ rust_byte_histogram,
223
+ rust_utf8_continuation_score,
224
+ )
225
+
226
+
227
+ def byte_histogram(data: bytes) -> array.array:
228
+ """
229
+ Compute byte frequency histogram.
230
+ Returns a 256-element ``array.array("Q", ...)`` of occurrence counts.
231
+ O(n), single pass. Uses Rust when available.
232
+ """
233
+ if _RUST_AVAILABLE:
234
+ return rust_byte_histogram(data)
235
+ return _byte_histogram_pure(data)
236
+
237
+
238
+ def utf8_continuation_score(data: bytes) -> float:
239
+ """
240
+ Score how well `data` fits UTF-8 multibyte sequence structure.
241
+ Returns 0.0 (no evidence) to 1.0 (strong UTF-8 multibyte pattern).
242
+ Works even when the data is not fully valid UTF-8. Uses Rust when available.
243
+ """
244
+ if _RUST_AVAILABLE:
245
+ return rust_utf8_continuation_score(data)
246
+ return _utf8_continuation_score_pure(data)
@@ -0,0 +1,161 @@
1
+ """
2
+ Raw-byte priors for candidate ordering (no external detectors).
3
+
4
+ Heuristic ranges are approximate; they only reorder the existing shortlist so
5
+ likely encodings are tried earlier before decode+mess ranking.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from typing import List, Optional
10
+
11
+
12
+ def _bump_to_front(candidates: List[str], *want: str) -> List[str]:
13
+ seen: set[str] = set()
14
+ front: List[str] = []
15
+ for e in want:
16
+ if e in candidates and e not in seen:
17
+ front.append(e)
18
+ seen.add(e)
19
+ rest = [c for c in candidates if c not in seen]
20
+ return front + rest
21
+
22
+
23
+ def hebrew_sbcs_likelihood(data: bytes) -> float:
24
+ """Share of high bytes in typical Windows-1255 Hebrew letter range (0xE0–0xFA)."""
25
+ if len(data) < 8:
26
+ return 0.0
27
+ hi = sum(1 for b in data if b >= 0x80)
28
+ if hi < 8:
29
+ return 0.0
30
+ heb = sum(1 for b in data if 0xE0 <= b <= 0xFA)
31
+ return heb / hi
32
+
33
+
34
+ def thai_tis620_likelihood(data: bytes) -> float:
35
+ """
36
+ Share of high bytes in the lower TIS-620 Thai band (0xA1–0xDF).
37
+
38
+ Excludes 0xE0–0xFA, which overlaps Windows-1255 Hebrew and inflates false Thai scores.
39
+ """
40
+ if len(data) < 8:
41
+ return 0.0
42
+ hi = sum(1 for b in data if b >= 0x80)
43
+ if hi < 8:
44
+ return 0.0
45
+ th = sum(1 for b in data if 0xA1 <= b <= 0xDF)
46
+ return th / hi
47
+
48
+
49
+ def koi8_byte_hint(data: bytes) -> bool:
50
+ """KOI8-R Cyrillic uses many high bytes in 0xC0–0xFF — avoid false Hebrew reorder."""
51
+ hi = [b for b in data if b >= 0x80]
52
+ if len(hi) < 24:
53
+ return False
54
+ block = sum(1 for b in hi if 0xC0 <= b <= 0xFF)
55
+ return block / len(hi) >= 0.58
56
+
57
+
58
+ def cp866_vs_cp1251_hint(data: bytes) -> Optional[str]:
59
+ """DOS (cp866) block 0x80–0xAF vs Windows (cp1251) 0xC0–0xFF — rough split."""
60
+ hi = [b for b in data if b >= 0x80]
61
+ if len(hi) < 20:
62
+ return None
63
+ b866 = sum(1 for b in hi if 0x80 <= b <= 0xAF)
64
+ b1251 = sum(1 for b in hi if 0xC0 <= b <= 0xFF)
65
+ t = len(hi)
66
+ if b866 / t >= 0.40 and b1251 / t <= 0.42:
67
+ return "cp866"
68
+ if b1251 / t >= 0.40 and b866 / t <= 0.38:
69
+ return "cp1251"
70
+ return None
71
+
72
+
73
+ def japanese_mbcs_bias(data: bytes) -> Optional[str]:
74
+ """Rough Shift_JIS vs EUC-JP from multi-byte pair patterns."""
75
+ if len(data) < 24:
76
+ return None
77
+ sj = 0
78
+ ej = 0
79
+ for i in range(len(data) - 1):
80
+ b1, b2 = data[i], data[i + 1]
81
+ if 0x81 <= b1 <= 0x9F or 0xE0 <= b1 <= 0xFC:
82
+ if 0x40 <= b2 <= 0xFC and b2 != 0x7F:
83
+ sj += 1
84
+ if 0xA1 <= b1 <= 0xFE and 0xA1 <= b2 <= 0xFE:
85
+ ej += 1
86
+ if b1 == 0x8E and 0xA1 <= b2 <= 0xDF:
87
+ ej += 2
88
+ if sj + ej < 14:
89
+ return None
90
+ # CP866 Cyrillic: moderate EUC-like pair counts without true JIS dominance — not Japanese.
91
+ hint = cp866_vs_cp1251_hint(data)
92
+ if hint == "cp866":
93
+ if ej > sj * 5.0:
94
+ return "euc_jp"
95
+ if sj > ej * 1.4 and sj >= 40:
96
+ return "shift_jis"
97
+ return None
98
+ if sj > ej * 1.2:
99
+ return "shift_jis"
100
+ if ej > sj * 1.2:
101
+ return "euc_jp"
102
+ return None
103
+
104
+
105
+ def chinese_big5_vs_gb_hint(data: bytes) -> Optional[str]:
106
+ """
107
+ GB18030 can use 4-byte sequences; Big5 is overwhelmingly two-byte pairs.
108
+ Very rough — ranking still validates via decode+mess.
109
+ """
110
+ if len(data) < 32:
111
+ return None
112
+ quadish = 0
113
+ for i in range(len(data) - 3):
114
+ if 0x81 <= data[i] <= 0xFE and 0x30 <= data[i + 1] <= 0x39:
115
+ quadish += 1
116
+ if quadish >= max(4, len(data) // 400):
117
+ return "gb18030"
118
+ pairs = 0
119
+ for i in range(len(data) - 1):
120
+ b1, b2 = data[i], data[i + 1]
121
+ if 0xA1 <= b1 <= 0xFE and 0x40 <= b2 <= 0xFE:
122
+ pairs += 1
123
+ if pairs >= len(data) // 6:
124
+ return "big5"
125
+ return None
126
+
127
+
128
+ def reorder_candidates(data: bytes, candidates: List[str]) -> List[str]:
129
+ """Apply raw-byte hints by moving likely encodings earlier."""
130
+ out = list(candidates)
131
+
132
+ if (
133
+ hebrew_sbcs_likelihood(data) >= 0.50
134
+ and cp866_vs_cp1251_hint(data) is None
135
+ and not koi8_byte_hint(data)
136
+ ):
137
+ out = _bump_to_front(out, "cp1255", "iso8859_8")
138
+
139
+ if thai_tis620_likelihood(data) >= 0.52 and hebrew_sbcs_likelihood(data) < 0.55:
140
+ out = _bump_to_front(out, "tis_620", "iso8859_11")
141
+
142
+ hint = cp866_vs_cp1251_hint(data)
143
+ if hint == "cp866":
144
+ out = _bump_to_front(out, "cp866", "koi8_r", "cp1251")
145
+ elif hint == "cp1251":
146
+ out = _bump_to_front(out, "cp1251", "koi8_r", "cp866")
147
+
148
+ jb = japanese_mbcs_bias(data)
149
+ if jb == "shift_jis":
150
+ out = _bump_to_front(out, "shift_jis", "cp932", "euc_jp", "iso2022_jp")
151
+ elif jb == "euc_jp":
152
+ out = _bump_to_front(out, "euc_jp", "iso2022_jp", "shift_jis", "cp932")
153
+
154
+ if jb is None:
155
+ ch = chinese_big5_vs_gb_hint(data)
156
+ if ch == "gb18030":
157
+ out = _bump_to_front(out, "gb18030", "gbk", "gb2312", "big5", "big5hkscs")
158
+ elif ch == "big5":
159
+ out = _bump_to_front(out, "big5", "big5hkscs", "gb18030", "gbk", "gb2312")
160
+
161
+ return out
bytesense/hints.py ADDED
@@ -0,0 +1,104 @@
1
+ """
2
+ Encoding hint extraction from HTTP headers and HTML/XML documents.
3
+
4
+ Used internally by StreamDetector, but also available as a public API
5
+ for callers who already have the raw content and headers.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import codecs
11
+ import re
12
+ from typing import Optional
13
+
14
+ _XML_DECL_RE = re.compile(rb'<\?xml[^>]*encoding=["\']([^"\']+)["\']', re.IGNORECASE)
15
+ _HTML_META_RE = re.compile(rb'<meta[^>]+charset\s*=\s*["\']?\s*([a-zA-Z0-9_\-]+)', re.IGNORECASE)
16
+ _HTTP_EQUIV_RE = re.compile(
17
+ rb'<meta[^>]+http-equiv\s*=\s*["\']?content-type["\']?[^>]*'
18
+ rb'content\s*=\s*["\']?[^"\']*charset=([a-zA-Z0-9_\-]+)',
19
+ re.IGNORECASE,
20
+ )
21
+ _HTTP_HEADER_RE = re.compile(r'charset\s*=\s*["\']?\s*([a-zA-Z0-9_\-]+)', re.IGNORECASE)
22
+
23
+
24
+ def _normalise(enc_str: str) -> Optional[str]:
25
+ try:
26
+ return codecs.lookup(enc_str.strip()).name.replace("-", "_")
27
+ except LookupError:
28
+ return None
29
+
30
+
31
+ def hint_from_http_headers(headers: dict[str, str]) -> Optional[str]:
32
+ """
33
+ Extract encoding hint from HTTP response headers.
34
+
35
+ Args:
36
+ headers: Dict of header name → value (case-insensitive matching).
37
+
38
+ Returns:
39
+ Normalised IANA encoding name or None.
40
+
41
+ Example::
42
+
43
+ enc = hint_from_http_headers({"Content-Type": "text/html; charset=utf-8"})
44
+ # → "utf_8"
45
+ """
46
+ ct = next((value for name, value in headers.items() if name.lower() == "content-type"), "")
47
+ m = _HTTP_HEADER_RE.search(ct)
48
+ if m:
49
+ return _normalise(m.group(1))
50
+ return None
51
+
52
+
53
+ def hint_from_content(data: bytes, max_scan_bytes: int = 4096) -> Optional[str]:
54
+ """
55
+ Extract encoding hint from the first ``max_scan_bytes`` of an HTML or XML document.
56
+
57
+ Looks for:
58
+ - XML declaration: ``<?xml version="1.0" encoding="UTF-8"?>``
59
+ - HTML meta charset: ``<meta charset="utf-8">``
60
+ - HTML http-equiv: ``<meta http-equiv="Content-Type" content="text/html; charset=utf-8">``
61
+
62
+ Args:
63
+ data: Raw bytes (need not be fully decoded).
64
+ max_scan_bytes: How far into the document to scan.
65
+
66
+ Returns:
67
+ Normalised IANA encoding name or None.
68
+ """
69
+ probe = data[:max_scan_bytes]
70
+ for pattern in (_XML_DECL_RE, _HTML_META_RE, _HTTP_EQUIV_RE):
71
+ m = pattern.search(probe)
72
+ if m:
73
+ try:
74
+ enc_str = m.group(1).decode("ascii", errors="ignore")
75
+ result = _normalise(enc_str)
76
+ if result:
77
+ return result
78
+ except Exception:
79
+ pass
80
+ return None
81
+
82
+
83
+ def best_hint(
84
+ data: bytes,
85
+ headers: Optional[dict[str, str]] = None,
86
+ max_scan_bytes: int = 4096,
87
+ ) -> Optional[str]:
88
+ """
89
+ Return the best encoding hint from HTTP headers and/or document content.
90
+
91
+ HTTP headers take priority (more reliable than meta tags).
92
+
93
+ Args:
94
+ data: Raw document bytes.
95
+ headers: HTTP response headers dict (optional).
96
+
97
+ Returns:
98
+ Normalised IANA encoding name or None.
99
+ """
100
+ if headers:
101
+ h = hint_from_http_headers(headers)
102
+ if h:
103
+ return h
104
+ return hint_from_content(data, max_scan_bytes)
bytesense/legacy.py ADDED
@@ -0,0 +1,38 @@
1
+ """
2
+ chardet / charset-normalizer drop-in compatibility layer.
3
+
4
+ Migration:
5
+ # Before
6
+ from chardet import detect
7
+ # or
8
+ from charset_normalizer import detect
9
+
10
+ # After
11
+ from bytesense import detect # identical interface
12
+ """
13
+ from __future__ import annotations
14
+
15
+ from typing import Dict, Optional, Union
16
+
17
+ from .api import from_bytes
18
+
19
+
20
+ def detect(byte_str: Union[bytes, bytearray]) -> Dict[str, Optional[object]]:
21
+ """
22
+ Drop-in replacement for ``chardet.detect()`` and
23
+ ``charset_normalizer.detect()``.
24
+
25
+ Args:
26
+ byte_str: The byte sequence to examine.
27
+
28
+ Returns:
29
+ ``{"encoding": str|None, "confidence": float|None, "language": str}``
30
+ """
31
+ if not isinstance(byte_str, (bytes, bytearray)):
32
+ raise TypeError(f"Expected bytes or bytearray, got {type(byte_str).__name__!r}")
33
+ result = from_bytes(bytes(byte_str))
34
+ return {
35
+ "encoding": result.encoding,
36
+ "confidence": result.confidence if result.encoding else None,
37
+ "language": result.language,
38
+ }