bytesense 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bytesense/__init__.py +47 -0
- bytesense/_rust.py +58 -0
- bytesense/api.py +469 -0
- bytesense/candidate.py +199 -0
- bytesense/cli.py +73 -0
- bytesense/coherence.py +68 -0
- bytesense/constant.py +1313 -0
- bytesense/data/__init__.py +0 -0
- bytesense/data/fingerprints.py +94 -0
- bytesense/data/language.json.gz +0 -0
- bytesense/fingerprint.py +246 -0
- bytesense/heuristics.py +161 -0
- bytesense/hints.py +104 -0
- bytesense/legacy.py +38 -0
- bytesense/mess.py +179 -0
- bytesense/models.py +98 -0
- bytesense/multi.py +209 -0
- bytesense/py.typed +0 -0
- bytesense/repair.py +280 -0
- bytesense/scoring.py +100 -0
- bytesense/streaming.py +416 -0
- bytesense/version.py +4 -0
- bytesense-1.0.0.dist-info/METADATA +160 -0
- bytesense-1.0.0.dist-info/RECORD +28 -0
- bytesense-1.0.0.dist-info/WHEEL +5 -0
- bytesense-1.0.0.dist-info/entry_points.txt +2 -0
- bytesense-1.0.0.dist-info/licenses/LICENSE +21 -0
- bytesense-1.0.0.dist-info/top_level.txt +1 -0
bytesense/mess.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Mess / chaos detector.
|
|
3
|
+
|
|
4
|
+
Scores how "garbled" a decoded string is.
|
|
5
|
+
0.0 = clean human-readable text.
|
|
6
|
+
1.0 = completely garbled (wrong encoding).
|
|
7
|
+
|
|
8
|
+
Improvements over charset-normalizer:
|
|
9
|
+
- Weighted multi-component score (not a single heuristic)
|
|
10
|
+
- Bigram validity check for Latin scripts
|
|
11
|
+
- Word-length plausibility check
|
|
12
|
+
- Sliding-window approach for large texts
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections import Counter
|
|
18
|
+
from functools import lru_cache
|
|
19
|
+
from typing import Optional, Tuple
|
|
20
|
+
|
|
21
|
+
from .constant import BIGRAM_FREQUENCIES
|
|
22
|
+
|
|
23
|
+
_VALID_BIGRAMS = {lang: frozenset(values) for lang, values in BIGRAM_FREQUENCIES.items()}
|
|
24
|
+
_ALL_BIGRAMS = frozenset(bg for values in _VALID_BIGRAMS.values() for bg in values)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _cjk_ratio(text: str) -> float:
|
|
28
|
+
if not text:
|
|
29
|
+
return 0.0
|
|
30
|
+
return sum(n for c, n in Counter(text).items() if "\u4e00" <= c <= "\u9fff") / len(text)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _hangul_ratio(text: str) -> float:
|
|
34
|
+
if not text:
|
|
35
|
+
return 0.0
|
|
36
|
+
return sum(n for c, n in Counter(text).items() if "\uac00" <= c <= "\ud7a3") / len(text)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _skip_latin_mess_heuristics(text: str) -> bool:
|
|
40
|
+
"""Ideographic / Hangul text: Latin bigram/word-length heuristics are misleading."""
|
|
41
|
+
if not text:
|
|
42
|
+
return False
|
|
43
|
+
return _cjk_ratio(text) > 0.18 or _hangul_ratio(text) > 0.18
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@lru_cache(maxsize=16_384)
|
|
47
|
+
def _is_printable(char: str) -> bool:
|
|
48
|
+
return char.isprintable()
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _latin_extended_fraction(text: str) -> float:
|
|
52
|
+
"""Share of Latin letters that use Latin Extended-A/B (Polish, Baltic, etc.)."""
|
|
53
|
+
counts = Counter(text)
|
|
54
|
+
latin = sum(n for c, n in counts.items() if c.isalpha() and ord(c) < 0x300)
|
|
55
|
+
if latin < 4:
|
|
56
|
+
return 0.0
|
|
57
|
+
ext = sum(n for c, n in counts.items() if c.isalpha() and 0x100 <= ord(c) <= 0x024F)
|
|
58
|
+
return ext / latin
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _bigram_mess(text: str, language_hint: Optional[str] = None) -> float:
|
|
62
|
+
"""
|
|
63
|
+
Fraction of Latin bigrams that are NOT in the valid set for any language.
|
|
64
|
+
Lower = cleaner text.
|
|
65
|
+
"""
|
|
66
|
+
valid = _VALID_BIGRAMS.get(language_hint or "", _ALL_BIGRAMS)
|
|
67
|
+
|
|
68
|
+
if not valid:
|
|
69
|
+
return 0.0
|
|
70
|
+
|
|
71
|
+
if _skip_latin_mess_heuristics(text):
|
|
72
|
+
return 0.0
|
|
73
|
+
|
|
74
|
+
# Latin Extended: English-centric bigrams falsely flag Polish/Czech/Baltic as noise.
|
|
75
|
+
if _latin_extended_fraction(text) > 0.22:
|
|
76
|
+
return 0.0
|
|
77
|
+
|
|
78
|
+
table = {ord(c): c.lower() if c.isalpha() and ord(c) < 0x250 else None for c in set(text)}
|
|
79
|
+
latin = text.translate(table)
|
|
80
|
+
if len(latin) < 8:
|
|
81
|
+
return 0.0
|
|
82
|
+
|
|
83
|
+
total = len(latin) - 1
|
|
84
|
+
invalid = sum(1 for i in range(total) if latin[i] + latin[i + 1] not in valid)
|
|
85
|
+
return invalid / total if total > 0 else 0.0
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _unprintable_ratio(text: str) -> float:
|
|
89
|
+
n = len(text)
|
|
90
|
+
if n == 0:
|
|
91
|
+
return 0.0
|
|
92
|
+
bad = sum(
|
|
93
|
+
count
|
|
94
|
+
for c, count in Counter(text).items()
|
|
95
|
+
if not _is_printable(c) and c not in ("\n", "\r", "\t")
|
|
96
|
+
)
|
|
97
|
+
return bad / n
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _suspicious_ratio(text: str) -> float:
|
|
101
|
+
n = len(text)
|
|
102
|
+
if n == 0:
|
|
103
|
+
return 0.0
|
|
104
|
+
suspicious = sum(
|
|
105
|
+
count for c, count in Counter(text).items() if c == "\ufffd" or "\ue000" <= c <= "\uf8ff"
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
return min(suspicious / n, 1.0)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _word_length_mess(text: str) -> float:
|
|
112
|
+
if _skip_latin_mess_heuristics(text):
|
|
113
|
+
return 0.0
|
|
114
|
+
words = text.split()
|
|
115
|
+
if not words:
|
|
116
|
+
return 0.0
|
|
117
|
+
avg = sum(len(w) for w in words) / len(words)
|
|
118
|
+
# Penalise: average word length > 25 characters suggests garbled text
|
|
119
|
+
return min(max(avg - 25.0, 0.0) / 25.0, 1.0)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def mess_ratio(
|
|
123
|
+
decoded: str,
|
|
124
|
+
threshold: float = 0.2,
|
|
125
|
+
language_hint: Optional[str] = None,
|
|
126
|
+
) -> float:
|
|
127
|
+
"""
|
|
128
|
+
Compute chaos ratio for a decoded string chunk.
|
|
129
|
+
|
|
130
|
+
Components (weighted sum):
|
|
131
|
+
0.45 — unprintable character ratio
|
|
132
|
+
0.25 — suspicious Unicode range ratio
|
|
133
|
+
0.20 — bigram invalidity ratio (Latin text only)
|
|
134
|
+
0.10 — word length plausibility
|
|
135
|
+
"""
|
|
136
|
+
if not decoded:
|
|
137
|
+
return 0.0
|
|
138
|
+
|
|
139
|
+
ratio = (
|
|
140
|
+
_unprintable_ratio(decoded) * 0.45
|
|
141
|
+
+ _suspicious_ratio(decoded) * 0.25
|
|
142
|
+
+ _bigram_mess(decoded, language_hint) * 0.20
|
|
143
|
+
+ _word_length_mess(decoded) * 0.10
|
|
144
|
+
)
|
|
145
|
+
return min(ratio, 1.0)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def sliding_window_mess(
|
|
149
|
+
decoded: str,
|
|
150
|
+
window_size: int = 512,
|
|
151
|
+
threshold: float = 0.2,
|
|
152
|
+
language_hint: Optional[str] = None,
|
|
153
|
+
) -> Tuple[float, bool]:
|
|
154
|
+
"""
|
|
155
|
+
Compute mess ratio with a sliding window.
|
|
156
|
+
More accurate than a single pass on large or mixed-content strings.
|
|
157
|
+
|
|
158
|
+
Returns:
|
|
159
|
+
(mean_mess_ratio, exceeded_threshold_early)
|
|
160
|
+
"""
|
|
161
|
+
if len(decoded) <= window_size:
|
|
162
|
+
r = mess_ratio(decoded, threshold, language_hint)
|
|
163
|
+
return r, r >= threshold
|
|
164
|
+
|
|
165
|
+
step = window_size // 2
|
|
166
|
+
ratios = []
|
|
167
|
+
exceeded = 0
|
|
168
|
+
|
|
169
|
+
for i in range(0, len(decoded) - window_size + 1, step):
|
|
170
|
+
r = mess_ratio(decoded[i : i + window_size], threshold, language_hint)
|
|
171
|
+
ratios.append(r)
|
|
172
|
+
if r >= threshold:
|
|
173
|
+
exceeded += 1
|
|
174
|
+
# Early exit: more than half the windows already exceed threshold
|
|
175
|
+
if exceeded > max(2, len(ratios) // 2):
|
|
176
|
+
return sum(ratios) / len(ratios), True
|
|
177
|
+
|
|
178
|
+
mean = sum(ratios) / len(ratios) if ratios else 0.0
|
|
179
|
+
return mean, mean >= threshold
|
bytesense/models.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Dict, List, Optional, Tuple
|
|
6
|
+
|
|
7
|
+
_SLOTS_KW = {"slots": True} if sys.version_info >= (3, 10) else {}
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True, **_SLOTS_KW)
|
|
11
|
+
class EncodingAlternative:
|
|
12
|
+
"""A plausible alternative encoding that did not win."""
|
|
13
|
+
|
|
14
|
+
encoding: str
|
|
15
|
+
confidence: float
|
|
16
|
+
language: str
|
|
17
|
+
|
|
18
|
+
def to_dict(self) -> Dict[str, object]:
|
|
19
|
+
return {
|
|
20
|
+
"encoding": self.encoding,
|
|
21
|
+
"confidence": self.confidence,
|
|
22
|
+
"language": self.language,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(**_SLOTS_KW)
|
|
27
|
+
class DetectionResult:
|
|
28
|
+
"""
|
|
29
|
+
Full result object returned by all bytesense detection functions.
|
|
30
|
+
|
|
31
|
+
Attributes:
|
|
32
|
+
encoding: Python codec name, e.g. ``"utf_8"``, ``"cp1252"``.
|
|
33
|
+
``None`` if detection failed completely.
|
|
34
|
+
confidence: Heuristic evidence score in 0.0–1.0, not a probability.
|
|
35
|
+
confidence_interval: Always None; heuristic scores have no statistical CI.
|
|
36
|
+
language: Human-readable language name, e.g. ``"French"``.
|
|
37
|
+
Empty string if not determined.
|
|
38
|
+
alternatives: Other plausible encodings, sorted by confidence descending.
|
|
39
|
+
bom_detected: ``True`` if a BOM/SIG was found.
|
|
40
|
+
chaos: Control-character ratio in the scoring sample.
|
|
41
|
+
coherence: Character-pair support (legacy) or optional language support (UTF-8).
|
|
42
|
+
why: Human-readable explanation of the detection decision.
|
|
43
|
+
byte_count: Number of bytes accepted by this call or stream.
|
|
44
|
+
bytes_examined: Bytes used for scoring or a direct fast-path decision.
|
|
45
|
+
bytes_validated: Bytes strictly validated under the returned codec.
|
|
46
|
+
complete: The input ended; False for previews/explicit budgets.
|
|
47
|
+
status: matched, ambiguous, unknown, binary, or invalid.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
encoding: Optional[str]
|
|
51
|
+
confidence: float
|
|
52
|
+
confidence_interval: Optional[Tuple[float, float]]
|
|
53
|
+
language: str
|
|
54
|
+
alternatives: List[EncodingAlternative]
|
|
55
|
+
bom_detected: bool
|
|
56
|
+
chaos: float
|
|
57
|
+
coherence: float
|
|
58
|
+
why: str
|
|
59
|
+
byte_count: int
|
|
60
|
+
bytes_examined: int = 0
|
|
61
|
+
bytes_validated: int = 0
|
|
62
|
+
complete: bool = False
|
|
63
|
+
status: str = "unknown"
|
|
64
|
+
|
|
65
|
+
# ------------------------------------------------------------------
|
|
66
|
+
# chardet / charset-normalizer compatibility helpers
|
|
67
|
+
# ------------------------------------------------------------------
|
|
68
|
+
|
|
69
|
+
def __str__(self) -> str:
|
|
70
|
+
return (
|
|
71
|
+
f"DetectionResult(encoding={self.encoding!r}, "
|
|
72
|
+
f"confidence={self.confidence:.3f}, "
|
|
73
|
+
f"language={self.language!r})"
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
def __repr__(self) -> str:
|
|
77
|
+
return self.__str__()
|
|
78
|
+
|
|
79
|
+
def __bool__(self) -> bool:
|
|
80
|
+
return self.encoding is not None
|
|
81
|
+
|
|
82
|
+
def to_dict(self) -> Dict[str, object]:
|
|
83
|
+
return {
|
|
84
|
+
"encoding": self.encoding,
|
|
85
|
+
"confidence": self.confidence,
|
|
86
|
+
"confidence_interval": None,
|
|
87
|
+
"language": self.language,
|
|
88
|
+
"alternatives": [a.to_dict() for a in self.alternatives],
|
|
89
|
+
"bom_detected": self.bom_detected,
|
|
90
|
+
"chaos": self.chaos,
|
|
91
|
+
"coherence": self.coherence,
|
|
92
|
+
"why": self.why,
|
|
93
|
+
"byte_count": self.byte_count,
|
|
94
|
+
"bytes_examined": self.bytes_examined,
|
|
95
|
+
"bytes_validated": self.bytes_validated,
|
|
96
|
+
"complete": self.complete,
|
|
97
|
+
"status": self.status,
|
|
98
|
+
}
|
bytesense/multi.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Multi-encoding document detector.
|
|
3
|
+
|
|
4
|
+
Splits a byte sequence into segments and detects encoding per-segment.
|
|
5
|
+
Useful for legacy email with mixed encodings or multi-part documents.
|
|
6
|
+
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import codecs
|
|
12
|
+
import sys
|
|
13
|
+
from dataclasses import dataclass, replace
|
|
14
|
+
from typing import Dict, List, Optional
|
|
15
|
+
|
|
16
|
+
from .api import from_bytes
|
|
17
|
+
from .models import DetectionResult
|
|
18
|
+
|
|
19
|
+
_SLOTS_KW = {"slots": True} if sys.version_info >= (3, 10) else {}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(**_SLOTS_KW)
|
|
23
|
+
class DocumentSegment:
|
|
24
|
+
"""A contiguous segment of bytes with a detected encoding."""
|
|
25
|
+
|
|
26
|
+
start: int
|
|
27
|
+
end: int
|
|
28
|
+
data: bytes
|
|
29
|
+
detection: DetectionResult
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def encoding(self) -> Optional[str]:
|
|
33
|
+
return self.detection.encoding
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def text(self) -> str:
|
|
37
|
+
if self.encoding is None:
|
|
38
|
+
raise UnicodeError("segment encoding is unknown; choose an explicit decoding policy")
|
|
39
|
+
return self.data.decode(self.encoding, errors="strict")
|
|
40
|
+
|
|
41
|
+
def to_dict(self) -> Dict[str, object]:
|
|
42
|
+
return {
|
|
43
|
+
"start": self.start,
|
|
44
|
+
"end": self.end,
|
|
45
|
+
"length": self.end - self.start,
|
|
46
|
+
"encoding": self.encoding,
|
|
47
|
+
"confidence": self.detection.confidence,
|
|
48
|
+
"language": self.detection.language,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(**_SLOTS_KW)
|
|
53
|
+
class MultiEncodingResult:
|
|
54
|
+
"""Result of multi-encoding document analysis."""
|
|
55
|
+
|
|
56
|
+
segments: List[DocumentSegment]
|
|
57
|
+
is_uniform: bool # True if all segments have the same encoding
|
|
58
|
+
dominant: Optional[str] # Most common encoding by byte weight
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def full_text(self) -> str:
|
|
62
|
+
"""Concatenate all segments decoded with their respective encodings."""
|
|
63
|
+
return "".join(seg.text for seg in self.segments)
|
|
64
|
+
|
|
65
|
+
def to_dict(self) -> Dict[str, object]:
|
|
66
|
+
return {
|
|
67
|
+
"is_uniform": self.is_uniform,
|
|
68
|
+
"dominant": self.dominant,
|
|
69
|
+
"segment_count": len(self.segments),
|
|
70
|
+
"segments": [s.to_dict() for s in self.segments],
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def detect_multi(
|
|
75
|
+
data: bytes,
|
|
76
|
+
segment_size: int = 4096,
|
|
77
|
+
min_segment_bytes: int = 128,
|
|
78
|
+
merge_threshold: float = 0.85,
|
|
79
|
+
) -> MultiEncodingResult:
|
|
80
|
+
"""
|
|
81
|
+
Detect encoding(s) in a potentially mixed-encoding document.
|
|
82
|
+
|
|
83
|
+
Algorithm:
|
|
84
|
+
1. Split `data` into contiguous segments near `segment_size` bytes.
|
|
85
|
+
2. Detect encoding for each segment independently.
|
|
86
|
+
3. Merge adjacent segments with the same encoding.
|
|
87
|
+
4. Return the segment list.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
data: Byte sequence to analyse.
|
|
91
|
+
segment_size: Initial segment size in bytes.
|
|
92
|
+
min_segment_bytes: Minimum bytes per segment (smaller segments are merged).
|
|
93
|
+
merge_threshold: Confidence threshold to merge adjacent same-encoding segments.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
:class:`MultiEncodingResult`
|
|
97
|
+
"""
|
|
98
|
+
if not isinstance(data, bytes):
|
|
99
|
+
raise TypeError("data must be bytes")
|
|
100
|
+
if not isinstance(segment_size, int) or isinstance(segment_size, bool) or segment_size <= 0:
|
|
101
|
+
raise ValueError("segment_size must be a positive integer")
|
|
102
|
+
if (
|
|
103
|
+
not isinstance(min_segment_bytes, int)
|
|
104
|
+
or isinstance(min_segment_bytes, bool)
|
|
105
|
+
or min_segment_bytes <= 0
|
|
106
|
+
):
|
|
107
|
+
raise ValueError("min_segment_bytes must be a positive integer")
|
|
108
|
+
if not 0.0 <= merge_threshold <= 1.0:
|
|
109
|
+
raise ValueError("merge_threshold must be between 0 and 1")
|
|
110
|
+
whole = from_bytes(data)
|
|
111
|
+
# A fully validated Unicode/stateful stream must not be cut into arbitrary
|
|
112
|
+
# byte windows; those windows can start inside a code unit or shift state.
|
|
113
|
+
uniform = whole.encoding is not None and (
|
|
114
|
+
whole.encoding.startswith("utf_")
|
|
115
|
+
or (whole.encoding.startswith("iso2022_") or whole.encoding == "hz")
|
|
116
|
+
)
|
|
117
|
+
if len(data) <= segment_size or uniform:
|
|
118
|
+
# Single-segment case — fast path
|
|
119
|
+
result = whole
|
|
120
|
+
seg = DocumentSegment(start=0, end=len(data), data=data, detection=result)
|
|
121
|
+
return MultiEncodingResult(
|
|
122
|
+
segments=[seg],
|
|
123
|
+
is_uniform=True,
|
|
124
|
+
dominant=result.encoding,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
# Detect per-segment
|
|
128
|
+
raw_segments: list[tuple[int, int, DetectionResult]] = []
|
|
129
|
+
pos = 0
|
|
130
|
+
while pos < len(data):
|
|
131
|
+
end = min(pos + segment_size, len(data))
|
|
132
|
+
if 0 < len(data) - end < min_segment_bytes:
|
|
133
|
+
end = len(data)
|
|
134
|
+
# When a full-file candidate is available, avoid cutting a complete
|
|
135
|
+
# multibyte character at the right edge. Internal decode failures are
|
|
136
|
+
# left to segment detection; no bytes are discarded.
|
|
137
|
+
if whole.encoding and end < len(data):
|
|
138
|
+
try:
|
|
139
|
+
decoder = codecs.getincrementaldecoder(whole.encoding)()
|
|
140
|
+
decoder.decode(data[pos:end], final=False)
|
|
141
|
+
for _ in range(8):
|
|
142
|
+
pending = decoder.getstate()[0]
|
|
143
|
+
if not pending or end == len(data):
|
|
144
|
+
break
|
|
145
|
+
decoder.decode(data[end : end + 1], final=False)
|
|
146
|
+
end += 1
|
|
147
|
+
except (UnicodeError, LookupError, TypeError):
|
|
148
|
+
pass
|
|
149
|
+
chunk = data[pos:end]
|
|
150
|
+
r = from_bytes(chunk)
|
|
151
|
+
raw_segments.append((pos, end, r))
|
|
152
|
+
pos = end
|
|
153
|
+
|
|
154
|
+
# Merge adjacent segments with same encoding
|
|
155
|
+
merged: list[DocumentSegment] = []
|
|
156
|
+
if raw_segments:
|
|
157
|
+
cur_start, cur_end, cur_result = raw_segments[0]
|
|
158
|
+
for start, end, result in raw_segments[1:]:
|
|
159
|
+
if (
|
|
160
|
+
result.encoding == cur_result.encoding
|
|
161
|
+
and result.confidence >= merge_threshold
|
|
162
|
+
and cur_result.confidence >= merge_threshold
|
|
163
|
+
):
|
|
164
|
+
cur_end = end
|
|
165
|
+
merged_len = cur_end - cur_start
|
|
166
|
+
cur_result = replace(
|
|
167
|
+
cur_result,
|
|
168
|
+
byte_count=merged_len,
|
|
169
|
+
bytes_validated=merged_len if cur_result.encoding else 0,
|
|
170
|
+
bytes_examined=cur_result.bytes_examined + result.bytes_examined,
|
|
171
|
+
confidence=min(cur_result.confidence, result.confidence),
|
|
172
|
+
)
|
|
173
|
+
else:
|
|
174
|
+
merged.append(
|
|
175
|
+
DocumentSegment(
|
|
176
|
+
start=cur_start,
|
|
177
|
+
end=cur_end,
|
|
178
|
+
data=data[cur_start:cur_end],
|
|
179
|
+
detection=cur_result,
|
|
180
|
+
)
|
|
181
|
+
)
|
|
182
|
+
cur_start, cur_end, cur_result = start, end, result
|
|
183
|
+
merged.append(
|
|
184
|
+
DocumentSegment(
|
|
185
|
+
start=cur_start,
|
|
186
|
+
end=cur_end,
|
|
187
|
+
data=data[cur_start:cur_end],
|
|
188
|
+
detection=cur_result,
|
|
189
|
+
)
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
if not merged and data:
|
|
193
|
+
r = from_bytes(data)
|
|
194
|
+
merged = [DocumentSegment(start=0, end=len(data), data=data, detection=r)]
|
|
195
|
+
|
|
196
|
+
# Determine dominant encoding by byte weight
|
|
197
|
+
enc_weights: dict[str, int] = {}
|
|
198
|
+
for seg in merged:
|
|
199
|
+
enc = seg.encoding or "unknown"
|
|
200
|
+
enc_weights[enc] = enc_weights.get(enc, 0) + (seg.end - seg.start)
|
|
201
|
+
dominant = max(enc_weights, key=lambda k: enc_weights[k]) if enc_weights else None
|
|
202
|
+
|
|
203
|
+
is_uniform = len({s.encoding for s in merged}) <= 1
|
|
204
|
+
|
|
205
|
+
return MultiEncodingResult(
|
|
206
|
+
segments=merged,
|
|
207
|
+
is_uniform=is_uniform,
|
|
208
|
+
dominant=dominant,
|
|
209
|
+
)
|
bytesense/py.typed
ADDED
|
File without changes
|