bytesense 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bytesense/mess.py ADDED
@@ -0,0 +1,179 @@
1
+ """
2
+ Mess / chaos detector.
3
+
4
+ Scores how "garbled" a decoded string is.
5
+ 0.0 = clean human-readable text.
6
+ 1.0 = completely garbled (wrong encoding).
7
+
8
+ Improvements over charset-normalizer:
9
+ - Weighted multi-component score (not a single heuristic)
10
+ - Bigram validity check for Latin scripts
11
+ - Word-length plausibility check
12
+ - Sliding-window approach for large texts
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from collections import Counter
18
+ from functools import lru_cache
19
+ from typing import Optional, Tuple
20
+
21
+ from .constant import BIGRAM_FREQUENCIES
22
+
23
+ _VALID_BIGRAMS = {lang: frozenset(values) for lang, values in BIGRAM_FREQUENCIES.items()}
24
+ _ALL_BIGRAMS = frozenset(bg for values in _VALID_BIGRAMS.values() for bg in values)
25
+
26
+
27
+ def _cjk_ratio(text: str) -> float:
28
+ if not text:
29
+ return 0.0
30
+ return sum(n for c, n in Counter(text).items() if "\u4e00" <= c <= "\u9fff") / len(text)
31
+
32
+
33
+ def _hangul_ratio(text: str) -> float:
34
+ if not text:
35
+ return 0.0
36
+ return sum(n for c, n in Counter(text).items() if "\uac00" <= c <= "\ud7a3") / len(text)
37
+
38
+
39
+ def _skip_latin_mess_heuristics(text: str) -> bool:
40
+ """Ideographic / Hangul text: Latin bigram/word-length heuristics are misleading."""
41
+ if not text:
42
+ return False
43
+ return _cjk_ratio(text) > 0.18 or _hangul_ratio(text) > 0.18
44
+
45
+
46
+ @lru_cache(maxsize=16_384)
47
+ def _is_printable(char: str) -> bool:
48
+ return char.isprintable()
49
+
50
+
51
+ def _latin_extended_fraction(text: str) -> float:
52
+ """Share of Latin letters that use Latin Extended-A/B (Polish, Baltic, etc.)."""
53
+ counts = Counter(text)
54
+ latin = sum(n for c, n in counts.items() if c.isalpha() and ord(c) < 0x300)
55
+ if latin < 4:
56
+ return 0.0
57
+ ext = sum(n for c, n in counts.items() if c.isalpha() and 0x100 <= ord(c) <= 0x024F)
58
+ return ext / latin
59
+
60
+
61
+ def _bigram_mess(text: str, language_hint: Optional[str] = None) -> float:
62
+ """
63
+ Fraction of Latin bigrams that are NOT in the valid set for any language.
64
+ Lower = cleaner text.
65
+ """
66
+ valid = _VALID_BIGRAMS.get(language_hint or "", _ALL_BIGRAMS)
67
+
68
+ if not valid:
69
+ return 0.0
70
+
71
+ if _skip_latin_mess_heuristics(text):
72
+ return 0.0
73
+
74
+ # Latin Extended: English-centric bigrams falsely flag Polish/Czech/Baltic as noise.
75
+ if _latin_extended_fraction(text) > 0.22:
76
+ return 0.0
77
+
78
+ table = {ord(c): c.lower() if c.isalpha() and ord(c) < 0x250 else None for c in set(text)}
79
+ latin = text.translate(table)
80
+ if len(latin) < 8:
81
+ return 0.0
82
+
83
+ total = len(latin) - 1
84
+ invalid = sum(1 for i in range(total) if latin[i] + latin[i + 1] not in valid)
85
+ return invalid / total if total > 0 else 0.0
86
+
87
+
88
+ def _unprintable_ratio(text: str) -> float:
89
+ n = len(text)
90
+ if n == 0:
91
+ return 0.0
92
+ bad = sum(
93
+ count
94
+ for c, count in Counter(text).items()
95
+ if not _is_printable(c) and c not in ("\n", "\r", "\t")
96
+ )
97
+ return bad / n
98
+
99
+
100
+ def _suspicious_ratio(text: str) -> float:
101
+ n = len(text)
102
+ if n == 0:
103
+ return 0.0
104
+ suspicious = sum(
105
+ count for c, count in Counter(text).items() if c == "\ufffd" or "\ue000" <= c <= "\uf8ff"
106
+ )
107
+
108
+ return min(suspicious / n, 1.0)
109
+
110
+
111
+ def _word_length_mess(text: str) -> float:
112
+ if _skip_latin_mess_heuristics(text):
113
+ return 0.0
114
+ words = text.split()
115
+ if not words:
116
+ return 0.0
117
+ avg = sum(len(w) for w in words) / len(words)
118
+ # Penalise: average word length > 25 characters suggests garbled text
119
+ return min(max(avg - 25.0, 0.0) / 25.0, 1.0)
120
+
121
+
122
+ def mess_ratio(
123
+ decoded: str,
124
+ threshold: float = 0.2,
125
+ language_hint: Optional[str] = None,
126
+ ) -> float:
127
+ """
128
+ Compute chaos ratio for a decoded string chunk.
129
+
130
+ Components (weighted sum):
131
+ 0.45 — unprintable character ratio
132
+ 0.25 — suspicious Unicode range ratio
133
+ 0.20 — bigram invalidity ratio (Latin text only)
134
+ 0.10 — word length plausibility
135
+ """
136
+ if not decoded:
137
+ return 0.0
138
+
139
+ ratio = (
140
+ _unprintable_ratio(decoded) * 0.45
141
+ + _suspicious_ratio(decoded) * 0.25
142
+ + _bigram_mess(decoded, language_hint) * 0.20
143
+ + _word_length_mess(decoded) * 0.10
144
+ )
145
+ return min(ratio, 1.0)
146
+
147
+
148
+ def sliding_window_mess(
149
+ decoded: str,
150
+ window_size: int = 512,
151
+ threshold: float = 0.2,
152
+ language_hint: Optional[str] = None,
153
+ ) -> Tuple[float, bool]:
154
+ """
155
+ Compute mess ratio with a sliding window.
156
+ More accurate than a single pass on large or mixed-content strings.
157
+
158
+ Returns:
159
+ (mean_mess_ratio, exceeded_threshold_early)
160
+ """
161
+ if len(decoded) <= window_size:
162
+ r = mess_ratio(decoded, threshold, language_hint)
163
+ return r, r >= threshold
164
+
165
+ step = window_size // 2
166
+ ratios = []
167
+ exceeded = 0
168
+
169
+ for i in range(0, len(decoded) - window_size + 1, step):
170
+ r = mess_ratio(decoded[i : i + window_size], threshold, language_hint)
171
+ ratios.append(r)
172
+ if r >= threshold:
173
+ exceeded += 1
174
+ # Early exit: more than half the windows already exceed threshold
175
+ if exceeded > max(2, len(ratios) // 2):
176
+ return sum(ratios) / len(ratios), True
177
+
178
+ mean = sum(ratios) / len(ratios) if ratios else 0.0
179
+ return mean, mean >= threshold
bytesense/models.py ADDED
@@ -0,0 +1,98 @@
1
+ from __future__ import annotations
2
+
3
+ import sys
4
+ from dataclasses import dataclass
5
+ from typing import Dict, List, Optional, Tuple
6
+
7
+ _SLOTS_KW = {"slots": True} if sys.version_info >= (3, 10) else {}
8
+
9
+
10
+ @dataclass(frozen=True, **_SLOTS_KW)
11
+ class EncodingAlternative:
12
+ """A plausible alternative encoding that did not win."""
13
+
14
+ encoding: str
15
+ confidence: float
16
+ language: str
17
+
18
+ def to_dict(self) -> Dict[str, object]:
19
+ return {
20
+ "encoding": self.encoding,
21
+ "confidence": self.confidence,
22
+ "language": self.language,
23
+ }
24
+
25
+
26
+ @dataclass(**_SLOTS_KW)
27
+ class DetectionResult:
28
+ """
29
+ Full result object returned by all bytesense detection functions.
30
+
31
+ Attributes:
32
+ encoding: Python codec name, e.g. ``"utf_8"``, ``"cp1252"``.
33
+ ``None`` if detection failed completely.
34
+ confidence: Heuristic evidence score in 0.0–1.0, not a probability.
35
+ confidence_interval: Always None; heuristic scores have no statistical CI.
36
+ language: Human-readable language name, e.g. ``"French"``.
37
+ Empty string if not determined.
38
+ alternatives: Other plausible encodings, sorted by confidence descending.
39
+ bom_detected: ``True`` if a BOM/SIG was found.
40
+ chaos: Control-character ratio in the scoring sample.
41
+ coherence: Character-pair support (legacy) or optional language support (UTF-8).
42
+ why: Human-readable explanation of the detection decision.
43
+ byte_count: Number of bytes accepted by this call or stream.
44
+ bytes_examined: Bytes used for scoring or a direct fast-path decision.
45
+ bytes_validated: Bytes strictly validated under the returned codec.
46
+ complete: The input ended; False for previews/explicit budgets.
47
+ status: matched, ambiguous, unknown, binary, or invalid.
48
+ """
49
+
50
+ encoding: Optional[str]
51
+ confidence: float
52
+ confidence_interval: Optional[Tuple[float, float]]
53
+ language: str
54
+ alternatives: List[EncodingAlternative]
55
+ bom_detected: bool
56
+ chaos: float
57
+ coherence: float
58
+ why: str
59
+ byte_count: int
60
+ bytes_examined: int = 0
61
+ bytes_validated: int = 0
62
+ complete: bool = False
63
+ status: str = "unknown"
64
+
65
+ # ------------------------------------------------------------------
66
+ # chardet / charset-normalizer compatibility helpers
67
+ # ------------------------------------------------------------------
68
+
69
+ def __str__(self) -> str:
70
+ return (
71
+ f"DetectionResult(encoding={self.encoding!r}, "
72
+ f"confidence={self.confidence:.3f}, "
73
+ f"language={self.language!r})"
74
+ )
75
+
76
+ def __repr__(self) -> str:
77
+ return self.__str__()
78
+
79
+ def __bool__(self) -> bool:
80
+ return self.encoding is not None
81
+
82
+ def to_dict(self) -> Dict[str, object]:
83
+ return {
84
+ "encoding": self.encoding,
85
+ "confidence": self.confidence,
86
+ "confidence_interval": None,
87
+ "language": self.language,
88
+ "alternatives": [a.to_dict() for a in self.alternatives],
89
+ "bom_detected": self.bom_detected,
90
+ "chaos": self.chaos,
91
+ "coherence": self.coherence,
92
+ "why": self.why,
93
+ "byte_count": self.byte_count,
94
+ "bytes_examined": self.bytes_examined,
95
+ "bytes_validated": self.bytes_validated,
96
+ "complete": self.complete,
97
+ "status": self.status,
98
+ }
bytesense/multi.py ADDED
@@ -0,0 +1,209 @@
1
+ """
2
+ Multi-encoding document detector.
3
+
4
+ Splits a byte sequence into segments and detects encoding per-segment.
5
+ Useful for legacy email with mixed encodings or multi-part documents.
6
+
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import codecs
12
+ import sys
13
+ from dataclasses import dataclass, replace
14
+ from typing import Dict, List, Optional
15
+
16
+ from .api import from_bytes
17
+ from .models import DetectionResult
18
+
19
+ _SLOTS_KW = {"slots": True} if sys.version_info >= (3, 10) else {}
20
+
21
+
22
+ @dataclass(**_SLOTS_KW)
23
+ class DocumentSegment:
24
+ """A contiguous segment of bytes with a detected encoding."""
25
+
26
+ start: int
27
+ end: int
28
+ data: bytes
29
+ detection: DetectionResult
30
+
31
+ @property
32
+ def encoding(self) -> Optional[str]:
33
+ return self.detection.encoding
34
+
35
+ @property
36
+ def text(self) -> str:
37
+ if self.encoding is None:
38
+ raise UnicodeError("segment encoding is unknown; choose an explicit decoding policy")
39
+ return self.data.decode(self.encoding, errors="strict")
40
+
41
+ def to_dict(self) -> Dict[str, object]:
42
+ return {
43
+ "start": self.start,
44
+ "end": self.end,
45
+ "length": self.end - self.start,
46
+ "encoding": self.encoding,
47
+ "confidence": self.detection.confidence,
48
+ "language": self.detection.language,
49
+ }
50
+
51
+
52
+ @dataclass(**_SLOTS_KW)
53
+ class MultiEncodingResult:
54
+ """Result of multi-encoding document analysis."""
55
+
56
+ segments: List[DocumentSegment]
57
+ is_uniform: bool # True if all segments have the same encoding
58
+ dominant: Optional[str] # Most common encoding by byte weight
59
+
60
+ @property
61
+ def full_text(self) -> str:
62
+ """Concatenate all segments decoded with their respective encodings."""
63
+ return "".join(seg.text for seg in self.segments)
64
+
65
+ def to_dict(self) -> Dict[str, object]:
66
+ return {
67
+ "is_uniform": self.is_uniform,
68
+ "dominant": self.dominant,
69
+ "segment_count": len(self.segments),
70
+ "segments": [s.to_dict() for s in self.segments],
71
+ }
72
+
73
+
74
+ def detect_multi(
75
+ data: bytes,
76
+ segment_size: int = 4096,
77
+ min_segment_bytes: int = 128,
78
+ merge_threshold: float = 0.85,
79
+ ) -> MultiEncodingResult:
80
+ """
81
+ Detect encoding(s) in a potentially mixed-encoding document.
82
+
83
+ Algorithm:
84
+ 1. Split `data` into contiguous segments near `segment_size` bytes.
85
+ 2. Detect encoding for each segment independently.
86
+ 3. Merge adjacent segments with the same encoding.
87
+ 4. Return the segment list.
88
+
89
+ Args:
90
+ data: Byte sequence to analyse.
91
+ segment_size: Initial segment size in bytes.
92
+ min_segment_bytes: Minimum bytes per segment (smaller segments are merged).
93
+ merge_threshold: Confidence threshold to merge adjacent same-encoding segments.
94
+
95
+ Returns:
96
+ :class:`MultiEncodingResult`
97
+ """
98
+ if not isinstance(data, bytes):
99
+ raise TypeError("data must be bytes")
100
+ if not isinstance(segment_size, int) or isinstance(segment_size, bool) or segment_size <= 0:
101
+ raise ValueError("segment_size must be a positive integer")
102
+ if (
103
+ not isinstance(min_segment_bytes, int)
104
+ or isinstance(min_segment_bytes, bool)
105
+ or min_segment_bytes <= 0
106
+ ):
107
+ raise ValueError("min_segment_bytes must be a positive integer")
108
+ if not 0.0 <= merge_threshold <= 1.0:
109
+ raise ValueError("merge_threshold must be between 0 and 1")
110
+ whole = from_bytes(data)
111
+ # A fully validated Unicode/stateful stream must not be cut into arbitrary
112
+ # byte windows; those windows can start inside a code unit or shift state.
113
+ uniform = whole.encoding is not None and (
114
+ whole.encoding.startswith("utf_")
115
+ or (whole.encoding.startswith("iso2022_") or whole.encoding == "hz")
116
+ )
117
+ if len(data) <= segment_size or uniform:
118
+ # Single-segment case — fast path
119
+ result = whole
120
+ seg = DocumentSegment(start=0, end=len(data), data=data, detection=result)
121
+ return MultiEncodingResult(
122
+ segments=[seg],
123
+ is_uniform=True,
124
+ dominant=result.encoding,
125
+ )
126
+
127
+ # Detect per-segment
128
+ raw_segments: list[tuple[int, int, DetectionResult]] = []
129
+ pos = 0
130
+ while pos < len(data):
131
+ end = min(pos + segment_size, len(data))
132
+ if 0 < len(data) - end < min_segment_bytes:
133
+ end = len(data)
134
+ # When a full-file candidate is available, avoid cutting a complete
135
+ # multibyte character at the right edge. Internal decode failures are
136
+ # left to segment detection; no bytes are discarded.
137
+ if whole.encoding and end < len(data):
138
+ try:
139
+ decoder = codecs.getincrementaldecoder(whole.encoding)()
140
+ decoder.decode(data[pos:end], final=False)
141
+ for _ in range(8):
142
+ pending = decoder.getstate()[0]
143
+ if not pending or end == len(data):
144
+ break
145
+ decoder.decode(data[end : end + 1], final=False)
146
+ end += 1
147
+ except (UnicodeError, LookupError, TypeError):
148
+ pass
149
+ chunk = data[pos:end]
150
+ r = from_bytes(chunk)
151
+ raw_segments.append((pos, end, r))
152
+ pos = end
153
+
154
+ # Merge adjacent segments with same encoding
155
+ merged: list[DocumentSegment] = []
156
+ if raw_segments:
157
+ cur_start, cur_end, cur_result = raw_segments[0]
158
+ for start, end, result in raw_segments[1:]:
159
+ if (
160
+ result.encoding == cur_result.encoding
161
+ and result.confidence >= merge_threshold
162
+ and cur_result.confidence >= merge_threshold
163
+ ):
164
+ cur_end = end
165
+ merged_len = cur_end - cur_start
166
+ cur_result = replace(
167
+ cur_result,
168
+ byte_count=merged_len,
169
+ bytes_validated=merged_len if cur_result.encoding else 0,
170
+ bytes_examined=cur_result.bytes_examined + result.bytes_examined,
171
+ confidence=min(cur_result.confidence, result.confidence),
172
+ )
173
+ else:
174
+ merged.append(
175
+ DocumentSegment(
176
+ start=cur_start,
177
+ end=cur_end,
178
+ data=data[cur_start:cur_end],
179
+ detection=cur_result,
180
+ )
181
+ )
182
+ cur_start, cur_end, cur_result = start, end, result
183
+ merged.append(
184
+ DocumentSegment(
185
+ start=cur_start,
186
+ end=cur_end,
187
+ data=data[cur_start:cur_end],
188
+ detection=cur_result,
189
+ )
190
+ )
191
+
192
+ if not merged and data:
193
+ r = from_bytes(data)
194
+ merged = [DocumentSegment(start=0, end=len(data), data=data, detection=r)]
195
+
196
+ # Determine dominant encoding by byte weight
197
+ enc_weights: dict[str, int] = {}
198
+ for seg in merged:
199
+ enc = seg.encoding or "unknown"
200
+ enc_weights[enc] = enc_weights.get(enc, 0) + (seg.end - seg.start)
201
+ dominant = max(enc_weights, key=lambda k: enc_weights[k]) if enc_weights else None
202
+
203
+ is_uniform = len({s.encoding for s in merged}) <= 1
204
+
205
+ return MultiEncodingResult(
206
+ segments=merged,
207
+ is_uniform=is_uniform,
208
+ dominant=dominant,
209
+ )
bytesense/py.typed ADDED
File without changes