bytesense 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bytesense/__init__.py +47 -0
- bytesense/_rust.py +58 -0
- bytesense/api.py +469 -0
- bytesense/candidate.py +199 -0
- bytesense/cli.py +73 -0
- bytesense/coherence.py +68 -0
- bytesense/constant.py +1313 -0
- bytesense/data/__init__.py +0 -0
- bytesense/data/fingerprints.py +94 -0
- bytesense/data/language.json.gz +0 -0
- bytesense/fingerprint.py +246 -0
- bytesense/heuristics.py +161 -0
- bytesense/hints.py +104 -0
- bytesense/legacy.py +38 -0
- bytesense/mess.py +179 -0
- bytesense/models.py +98 -0
- bytesense/multi.py +209 -0
- bytesense/py.typed +0 -0
- bytesense/repair.py +280 -0
- bytesense/scoring.py +100 -0
- bytesense/streaming.py +416 -0
- bytesense/version.py +4 -0
- bytesense-1.0.0.dist-info/METADATA +160 -0
- bytesense-1.0.0.dist-info/RECORD +28 -0
- bytesense-1.0.0.dist-info/WHEEL +5 -0
- bytesense-1.0.0.dist-info/entry_points.txt +2 -0
- bytesense-1.0.0.dist-info/licenses/LICENSE +21 -0
- bytesense-1.0.0.dist-info/top_level.txt +1 -0
bytesense/streaming.py
ADDED
|
@@ -0,0 +1,416 @@
|
|
|
1
|
+
"""Bounded-memory streaming with explicit finalization and full decode validation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import codecs
|
|
6
|
+
import tempfile
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
from typing import Any, Iterator, Optional
|
|
9
|
+
|
|
10
|
+
from .api import (
|
|
11
|
+
_HIGH_BYTE,
|
|
12
|
+
_allowed,
|
|
13
|
+
_binary,
|
|
14
|
+
_bom_codec,
|
|
15
|
+
_filters,
|
|
16
|
+
_from_sample,
|
|
17
|
+
_result,
|
|
18
|
+
_transport_encoding,
|
|
19
|
+
from_bytes,
|
|
20
|
+
)
|
|
21
|
+
from .fingerprint import detect_null_pattern
|
|
22
|
+
from .hints import hint_from_content, hint_from_http_headers
|
|
23
|
+
from .models import DetectionResult
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class StreamDetector:
|
|
27
|
+
"""Consume every fed byte; spool beyond ``memory_limit`` to a local temp file.
|
|
28
|
+
|
|
29
|
+
Preview scoring runs at exponentially increasing checkpoints, then stops
|
|
30
|
+
after the sample fills. ``is_stable`` describes a guess, never end-of-input.
|
|
31
|
+
``finalize`` strictly validates the chosen codec over the complete stream.
|
|
32
|
+
Feeding after finalization raises instead of silently dropping data.
|
|
33
|
+
Use a context manager or ``close()`` to release an unfinished stream.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
MIN_BYTES = 64
|
|
37
|
+
SATURATION = 8192
|
|
38
|
+
STABILITY_ROUNDS = 2
|
|
39
|
+
|
|
40
|
+
def __init__(
|
|
41
|
+
self,
|
|
42
|
+
threshold: float = 0.2,
|
|
43
|
+
language_threshold: float = 0.1,
|
|
44
|
+
auto_stop_confidence: float = 0.97,
|
|
45
|
+
*,
|
|
46
|
+
memory_limit: int = 1_048_576,
|
|
47
|
+
**kwargs: Any,
|
|
48
|
+
) -> None:
|
|
49
|
+
if not isinstance(memory_limit, int) or isinstance(memory_limit, bool) or memory_limit < 64:
|
|
50
|
+
raise ValueError("memory_limit must be an integer of at least 64 bytes")
|
|
51
|
+
if not 0.0 <= auto_stop_confidence <= 1.0:
|
|
52
|
+
raise ValueError("auto_stop_confidence must be between 0 and 1")
|
|
53
|
+
self._options = dict(kwargs, threshold=threshold, language_threshold=language_threshold)
|
|
54
|
+
from_bytes(b"", **self._options) # Validate configuration before accepting any data.
|
|
55
|
+
self._memory_limit = memory_limit
|
|
56
|
+
self._auto_stop_confidence = auto_stop_confidence
|
|
57
|
+
self._sample_size = kwargs.get("sample_size", 4096)
|
|
58
|
+
self._spool = tempfile.SpooledTemporaryFile(max_size=memory_limit, mode="w+b")
|
|
59
|
+
self._buf = bytearray()
|
|
60
|
+
self._late_sample = bytearray()
|
|
61
|
+
self._tail = b""
|
|
62
|
+
self._bytes_fed = 0
|
|
63
|
+
self._ascii = True
|
|
64
|
+
self._escape = False
|
|
65
|
+
self._next_probe = self.MIN_BYTES
|
|
66
|
+
self._result: Optional[DetectionResult] = None
|
|
67
|
+
self._finalized = False
|
|
68
|
+
self._stable_rounds = 0
|
|
69
|
+
self._prev_encoding: Optional[str] = None
|
|
70
|
+
self._declared_hint: Optional[str] = None
|
|
71
|
+
|
|
72
|
+
def feed(self, chunk: bytes | bytearray) -> None:
|
|
73
|
+
if self._finalized or self._spool.closed:
|
|
74
|
+
raise RuntimeError("stream is closed; call reset() before feeding more data")
|
|
75
|
+
if not isinstance(chunk, (bytes, bytearray)):
|
|
76
|
+
raise TypeError("stream chunks must be bytes or bytearray")
|
|
77
|
+
if not chunk:
|
|
78
|
+
return
|
|
79
|
+
if self._bytes_fed + len(chunk) > self._memory_limit:
|
|
80
|
+
self._spool.rollover()
|
|
81
|
+
self._spool.write(chunk)
|
|
82
|
+
self._bytes_fed += len(chunk)
|
|
83
|
+
ascii_chunk = chunk.isascii()
|
|
84
|
+
if self._late_sample and len(self._late_sample) < self._sample_size:
|
|
85
|
+
self._late_sample.extend(chunk[: self._sample_size - len(self._late_sample)])
|
|
86
|
+
# Preserve a continuous informative window, independent of chunk sizes.
|
|
87
|
+
if (self._ascii and not ascii_chunk) or (not self._escape and b"\x1b" in chunk):
|
|
88
|
+
signal = _HIGH_BYTE.search(chunk)
|
|
89
|
+
if (
|
|
90
|
+
signal is not None
|
|
91
|
+
and self._bytes_fed - len(chunk) + signal.start() > self._sample_size // 4
|
|
92
|
+
):
|
|
93
|
+
start = signal.start() - 64
|
|
94
|
+
probe = (
|
|
95
|
+
self._tail[start:] + bytes(chunk[: self._sample_size])
|
|
96
|
+
if start < 0
|
|
97
|
+
else bytes(chunk[start : start + self._sample_size])
|
|
98
|
+
)
|
|
99
|
+
self._late_sample = bytearray(probe[: self._sample_size])
|
|
100
|
+
self._tail = bytes(chunk[-64:]) if len(chunk) >= 64 else (self._tail + bytes(chunk))[-64:]
|
|
101
|
+
self._ascii = self._ascii and ascii_chunk
|
|
102
|
+
self._escape = self._escape or b"\x1b" in chunk
|
|
103
|
+
if len(self._buf) < self._sample_size:
|
|
104
|
+
self._buf.extend(chunk[: self._sample_size - len(self._buf)])
|
|
105
|
+
if not self._declared_hint and self._options.get("use_hints", True):
|
|
106
|
+
self._declared_hint = hint_from_content(bytes(self._buf))
|
|
107
|
+
if self._bytes_fed >= self._next_probe and self._next_probe <= self._sample_size:
|
|
108
|
+
self._run()
|
|
109
|
+
while self._next_probe <= self._bytes_fed:
|
|
110
|
+
self._next_probe *= 4
|
|
111
|
+
|
|
112
|
+
def _run(self) -> None:
|
|
113
|
+
probe = bytes(self._late_sample or self._buf)
|
|
114
|
+
r = from_bytes(probe, **self._options)
|
|
115
|
+
self._result = replace(r, byte_count=self._bytes_fed, complete=False, bytes_validated=0)
|
|
116
|
+
if r.encoding == self._prev_encoding:
|
|
117
|
+
self._stable_rounds += 1
|
|
118
|
+
else:
|
|
119
|
+
self._stable_rounds = 0
|
|
120
|
+
self._prev_encoding = r.encoding
|
|
121
|
+
|
|
122
|
+
def hint_from_headers(self, headers: dict[str, str]) -> None:
|
|
123
|
+
if self._finalized:
|
|
124
|
+
raise RuntimeError("cannot change hints on a finalized stream")
|
|
125
|
+
hint = hint_from_http_headers(headers)
|
|
126
|
+
if hint:
|
|
127
|
+
self._declared_hint = hint
|
|
128
|
+
|
|
129
|
+
def _validate(self, encoding: str) -> tuple[bool, bytes]:
|
|
130
|
+
self._spool.seek(0)
|
|
131
|
+
try:
|
|
132
|
+
decoder = codecs.getincrementaldecoder(encoding)(errors="strict")
|
|
133
|
+
except LookupError:
|
|
134
|
+
return False, b""
|
|
135
|
+
chunk = b""
|
|
136
|
+
try:
|
|
137
|
+
while True:
|
|
138
|
+
chunk = self._spool.read(65536)
|
|
139
|
+
if not chunk:
|
|
140
|
+
break
|
|
141
|
+
decoder.decode(chunk, final=False)
|
|
142
|
+
decoder.decode(b"", final=True)
|
|
143
|
+
return True, b""
|
|
144
|
+
except UnicodeDecodeError as exc:
|
|
145
|
+
start = max(0, exc.start - 64)
|
|
146
|
+
return False, chunk[start : start + self._sample_size]
|
|
147
|
+
except UnicodeError:
|
|
148
|
+
return False, chunk[: self._sample_size]
|
|
149
|
+
|
|
150
|
+
def finalize(self) -> DetectionResult:
|
|
151
|
+
if self._finalized:
|
|
152
|
+
assert self._result is not None
|
|
153
|
+
return self._result
|
|
154
|
+
if self._spool.closed:
|
|
155
|
+
raise RuntimeError("stream is closed")
|
|
156
|
+
try:
|
|
157
|
+
self._result = self._finish()
|
|
158
|
+
self._finalized = True
|
|
159
|
+
return self._result
|
|
160
|
+
finally:
|
|
161
|
+
self._spool.close()
|
|
162
|
+
|
|
163
|
+
def _finish(self) -> DetectionResult:
|
|
164
|
+
prefix = bytes(self._buf)
|
|
165
|
+
n = self._bytes_fed
|
|
166
|
+
if n <= self._sample_size:
|
|
167
|
+
options = dict(self._options)
|
|
168
|
+
if self._declared_hint and not options.get("encoding_hint"):
|
|
169
|
+
options["encoding_hint"] = self._declared_hint
|
|
170
|
+
return from_bytes(prefix, **options)
|
|
171
|
+
include, exclude = _filters(
|
|
172
|
+
self._options.get("cp_isolation"), self._options.get("cp_exclusion")
|
|
173
|
+
)
|
|
174
|
+
minimum = self._options.get("min_confidence", 0.0)
|
|
175
|
+
if include == []:
|
|
176
|
+
return _result(None, n, "No encodings allowed.", examined=len(prefix))
|
|
177
|
+
bom = _bom_codec(prefix, include, exclude)
|
|
178
|
+
if bom:
|
|
179
|
+
valid, _ = self._validate(bom)
|
|
180
|
+
if valid:
|
|
181
|
+
return replace(
|
|
182
|
+
_result(bom, n, "BOM; complete stream validated.", 1.0, examined=len(prefix)),
|
|
183
|
+
bom_detected=True,
|
|
184
|
+
)
|
|
185
|
+
return _result(
|
|
186
|
+
None,
|
|
187
|
+
n,
|
|
188
|
+
"Invalid complete stream under its declared BOM.",
|
|
189
|
+
status="invalid",
|
|
190
|
+
examined=len(prefix),
|
|
191
|
+
)
|
|
192
|
+
if _binary(prefix):
|
|
193
|
+
return _result(
|
|
194
|
+
None,
|
|
195
|
+
n,
|
|
196
|
+
"Binary signature or excessive control bytes.",
|
|
197
|
+
status="binary",
|
|
198
|
+
examined=len(prefix),
|
|
199
|
+
)
|
|
200
|
+
transport = _transport_encoding(prefix) if self._ascii else None
|
|
201
|
+
if transport and _allowed(transport, include, exclude):
|
|
202
|
+
valid, _ = self._validate(transport)
|
|
203
|
+
if valid and minimum <= 0.85:
|
|
204
|
+
return _result(
|
|
205
|
+
transport,
|
|
206
|
+
n,
|
|
207
|
+
"7-bit shift syntax; complete stream validated.",
|
|
208
|
+
0.85,
|
|
209
|
+
"ambiguous",
|
|
210
|
+
len(prefix),
|
|
211
|
+
)
|
|
212
|
+
shape = detect_null_pattern(prefix)
|
|
213
|
+
if shape and _allowed(shape, include, exclude):
|
|
214
|
+
valid, _ = self._validate(shape)
|
|
215
|
+
if valid:
|
|
216
|
+
return (
|
|
217
|
+
_result(
|
|
218
|
+
shape,
|
|
219
|
+
n,
|
|
220
|
+
"Unicode byte-lane pattern; complete stream validated.",
|
|
221
|
+
0.9,
|
|
222
|
+
examined=len(prefix),
|
|
223
|
+
)
|
|
224
|
+
if minimum <= 0.9
|
|
225
|
+
else _result(None, n, "Below minimum confidence.", examined=len(prefix))
|
|
226
|
+
)
|
|
227
|
+
if not shape and not self._escape:
|
|
228
|
+
if self._ascii and _allowed("ascii", include, exclude):
|
|
229
|
+
return _result("ascii", n, "All stream bytes are ASCII text.", 1.0, examined=n)
|
|
230
|
+
if _allowed("utf_8", include, exclude):
|
|
231
|
+
valid, _ = self._validate("utf_8")
|
|
232
|
+
if valid:
|
|
233
|
+
r = _result("utf_8", n, "Complete stream validated as UTF-8.", 0.99, examined=n)
|
|
234
|
+
if self._options.get("include_language"):
|
|
235
|
+
from .coherence import detect_language
|
|
236
|
+
|
|
237
|
+
decoded = codecs.getincrementaldecoder("utf_8")().decode(
|
|
238
|
+
prefix, final=False
|
|
239
|
+
)
|
|
240
|
+
langs = detect_language(
|
|
241
|
+
decoded, threshold=self._options["language_threshold"]
|
|
242
|
+
)
|
|
243
|
+
if langs:
|
|
244
|
+
r = replace(r, language=langs[0][0], coherence=round(langs[0][1], 4))
|
|
245
|
+
return (
|
|
246
|
+
r
|
|
247
|
+
if r.confidence >= minimum
|
|
248
|
+
else _result(None, n, "Below minimum confidence.", examined=len(prefix))
|
|
249
|
+
)
|
|
250
|
+
probe = bytes(self._late_sample) or prefix
|
|
251
|
+
options = {key: self._options[key] for key in ("threshold", "language_threshold")}
|
|
252
|
+
r = _from_sample(
|
|
253
|
+
probe, cp_isolation=include, cp_exclusion=exclude, enable_fallback=False, **options
|
|
254
|
+
)
|
|
255
|
+
candidates = ([r.encoding] if r.encoding else []) + [alt.encoding for alt in r.alternatives]
|
|
256
|
+
hint = self._options.get("encoding_hint") or self._declared_hint
|
|
257
|
+
if hint:
|
|
258
|
+
from .api import _codec
|
|
259
|
+
|
|
260
|
+
try:
|
|
261
|
+
hint = _codec(hint)
|
|
262
|
+
except (LookupError, TypeError):
|
|
263
|
+
hint = None
|
|
264
|
+
if hint and _allowed(hint, include, exclude):
|
|
265
|
+
candidates.insert(0, hint)
|
|
266
|
+
if self._options.get("enable_fallback", True) and include:
|
|
267
|
+
candidates.extend(include)
|
|
268
|
+
tried: set[str] = set()
|
|
269
|
+
examined = len(probe)
|
|
270
|
+
for enc in candidates:
|
|
271
|
+
if enc in tried or not _allowed(enc, include, exclude):
|
|
272
|
+
continue
|
|
273
|
+
tried.add(enc)
|
|
274
|
+
valid, failure = self._validate(enc)
|
|
275
|
+
if valid:
|
|
276
|
+
confidence = 0.95 if enc == hint else r.confidence if enc == r.encoding else 0.1
|
|
277
|
+
if confidence < minimum:
|
|
278
|
+
continue
|
|
279
|
+
if self._options.get("include_language"):
|
|
280
|
+
from .coherence import detect_language
|
|
281
|
+
|
|
282
|
+
decoded = codecs.getincrementaldecoder(enc)().decode(probe, final=False)
|
|
283
|
+
langs = detect_language(decoded, threshold=self._options["language_threshold"])
|
|
284
|
+
if langs:
|
|
285
|
+
r = replace(r, language=langs[0][0])
|
|
286
|
+
return replace(
|
|
287
|
+
r,
|
|
288
|
+
encoding=enc,
|
|
289
|
+
confidence=confidence,
|
|
290
|
+
byte_count=n,
|
|
291
|
+
bytes_examined=min(n, examined),
|
|
292
|
+
bytes_validated=n,
|
|
293
|
+
complete=True,
|
|
294
|
+
status="ambiguous" if r.alternatives else "matched",
|
|
295
|
+
chaos=r.chaos if enc == r.encoding else 0.0,
|
|
296
|
+
coherence=r.coherence if enc == r.encoding else 0.0,
|
|
297
|
+
alternatives=[a for a in r.alternatives if a.encoding != enc],
|
|
298
|
+
language=r.language
|
|
299
|
+
if self._options.get("include_language") and enc == r.encoding
|
|
300
|
+
else "",
|
|
301
|
+
why=(r.why if enc == r.encoding else f"Selected validated alternative {enc}.")
|
|
302
|
+
+ " Complete stream validated."
|
|
303
|
+
+ (f" Hint: {hint}." if hint else ""),
|
|
304
|
+
)
|
|
305
|
+
if failure and len(tried) <= 4:
|
|
306
|
+
alternate = _from_sample(
|
|
307
|
+
failure,
|
|
308
|
+
cp_isolation=include,
|
|
309
|
+
cp_exclusion=list(set(exclude) | tried),
|
|
310
|
+
enable_fallback=False,
|
|
311
|
+
**options,
|
|
312
|
+
)
|
|
313
|
+
examined += len(failure)
|
|
314
|
+
if alternate.encoding:
|
|
315
|
+
candidates.append(alternate.encoding)
|
|
316
|
+
return _result(
|
|
317
|
+
None,
|
|
318
|
+
n,
|
|
319
|
+
"No permitted encoding validates the complete stream.",
|
|
320
|
+
examined=min(n, examined),
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
def close(self) -> None:
|
|
324
|
+
self._spool.close()
|
|
325
|
+
|
|
326
|
+
def reset(self) -> None:
|
|
327
|
+
self.close()
|
|
328
|
+
StreamDetector.__init__(
|
|
329
|
+
self,
|
|
330
|
+
auto_stop_confidence=self._auto_stop_confidence,
|
|
331
|
+
memory_limit=self._memory_limit,
|
|
332
|
+
**self._options,
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
def __enter__(self) -> StreamDetector:
|
|
336
|
+
return self
|
|
337
|
+
|
|
338
|
+
def __exit__(self, *args: object) -> None:
|
|
339
|
+
self.close()
|
|
340
|
+
|
|
341
|
+
def snapshot(self) -> dict[str, object]:
|
|
342
|
+
return {
|
|
343
|
+
"bytes_fed": self.bytes_fed,
|
|
344
|
+
"buffered_bytes": len(self._buf),
|
|
345
|
+
"encoding": self.encoding,
|
|
346
|
+
"confidence": self.confidence,
|
|
347
|
+
"language": self.language,
|
|
348
|
+
"stable_rounds": self._stable_rounds,
|
|
349
|
+
"finalized": self._finalized,
|
|
350
|
+
"declared_hint": self._declared_hint,
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
@property
|
|
354
|
+
def result(self) -> Optional[DetectionResult]:
|
|
355
|
+
return self._result
|
|
356
|
+
|
|
357
|
+
@property
|
|
358
|
+
def encoding(self) -> Optional[str]:
|
|
359
|
+
return self._result.encoding if self._result is not None else None
|
|
360
|
+
|
|
361
|
+
@property
|
|
362
|
+
def confidence(self) -> float:
|
|
363
|
+
return self._result.confidence if self._result is not None else 0.0
|
|
364
|
+
|
|
365
|
+
@property
|
|
366
|
+
def language(self) -> str:
|
|
367
|
+
return self._result.language if self._result is not None else ""
|
|
368
|
+
|
|
369
|
+
@property
|
|
370
|
+
def bytes_fed(self) -> int:
|
|
371
|
+
return self._bytes_fed
|
|
372
|
+
|
|
373
|
+
@property
|
|
374
|
+
def is_stable(self) -> bool:
|
|
375
|
+
return self._finalized or (
|
|
376
|
+
self._stable_rounds >= self.STABILITY_ROUNDS
|
|
377
|
+
and self.confidence >= self._auto_stop_confidence
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def detect_stream(
|
|
382
|
+
chunks: Iterator[bytes],
|
|
383
|
+
*,
|
|
384
|
+
stop_confidence: float = 0.97,
|
|
385
|
+
max_bytes: Optional[int] = None,
|
|
386
|
+
early_stop: bool = False,
|
|
387
|
+
**kwargs: Any,
|
|
388
|
+
) -> DetectionResult:
|
|
389
|
+
"""Consume to EOF by default. Explicit budgets/early stopping mark incomplete results.
|
|
390
|
+
|
|
391
|
+
An oversized yielded chunk is sliced before analysis. Its unused remainder
|
|
392
|
+
has already been yielded by the source and cannot be restored by this API.
|
|
393
|
+
"""
|
|
394
|
+
if max_bytes is not None and (
|
|
395
|
+
not isinstance(max_bytes, int) or isinstance(max_bytes, bool) or max_bytes <= 0
|
|
396
|
+
):
|
|
397
|
+
raise ValueError("max_bytes must be a positive integer or None")
|
|
398
|
+
with StreamDetector(auto_stop_confidence=stop_confidence, **kwargs) as detector:
|
|
399
|
+
limited = False
|
|
400
|
+
for chunk in chunks:
|
|
401
|
+
if not isinstance(chunk, (bytes, bytearray)):
|
|
402
|
+
raise TypeError("stream chunks must be bytes or bytearray")
|
|
403
|
+
if max_bytes is not None:
|
|
404
|
+
chunk = chunk[: max_bytes - detector.bytes_fed]
|
|
405
|
+
detector.feed(chunk)
|
|
406
|
+
if (max_bytes is not None and detector.bytes_fed >= max_bytes) or (
|
|
407
|
+
early_stop and detector.is_stable
|
|
408
|
+
):
|
|
409
|
+
limited = True
|
|
410
|
+
break
|
|
411
|
+
r = detector.finalize()
|
|
412
|
+
return (
|
|
413
|
+
replace(r, complete=False, why=r.why + " Input stopped before confirmed EOF.")
|
|
414
|
+
if limited
|
|
415
|
+
else r
|
|
416
|
+
)
|
bytesense/version.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: bytesense
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
|
|
5
|
+
Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
6
|
+
Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
7
|
+
License: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/oguzhankir/bytesense
|
|
9
|
+
Project-URL: Repository, https://github.com/oguzhankir/bytesense
|
|
10
|
+
Project-URL: Issues, https://github.com/oguzhankir/bytesense/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md
|
|
12
|
+
Project-URL: Documentation, https://oguzhankir.github.io/bytesense/
|
|
13
|
+
Keywords: encoding,charset,detection,unicode,bytes,chardet,charset-normalizer
|
|
14
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
25
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
26
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
27
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
28
|
+
Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
29
|
+
Classifier: Programming Language :: Rust
|
|
30
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
31
|
+
Classifier: Topic :: Utilities
|
|
32
|
+
Classifier: Typing :: Typed
|
|
33
|
+
Requires-Python: >=3.9
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
License-File: LICENSE
|
|
36
|
+
Provides-Extra: fast
|
|
37
|
+
Provides-Extra: docs
|
|
38
|
+
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
39
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "docs"
|
|
40
|
+
Requires-Dist: pymdown-extensions>=10.0; extra == "docs"
|
|
41
|
+
Provides-Extra: dev
|
|
42
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
43
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-benchmark>=4.0; extra == "dev"
|
|
45
|
+
Requires-Dist: mypy>=1.8; extra == "dev"
|
|
46
|
+
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
47
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
48
|
+
Requires-Dist: hypothesis>=6.100; (platform_python_implementation != "PyPy" or python_version >= "3.11") and extra == "dev"
|
|
49
|
+
Requires-Dist: hypothesis==6.100.0; (platform_python_implementation == "PyPy" and python_version < "3.11") and extra == "dev"
|
|
50
|
+
Requires-Dist: setuptools-rust>=1.12; extra == "dev"
|
|
51
|
+
Requires-Dist: charset-normalizer>=3.3; extra == "dev"
|
|
52
|
+
Requires-Dist: chardet>=5.0; extra == "dev"
|
|
53
|
+
Requires-Dist: requests>=2.31; extra == "dev"
|
|
54
|
+
Requires-Dist: certifi>=2023.0; extra == "dev"
|
|
55
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
56
|
+
Dynamic: license-file
|
|
57
|
+
|
|
58
|
+
<p align="center">
|
|
59
|
+
<img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
|
|
60
|
+
</p>
|
|
61
|
+
<p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
|
|
62
|
+
<p align="center">
|
|
63
|
+
<a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
|
|
64
|
+
<a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
65
|
+
<a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
|
|
66
|
+
</p>
|
|
67
|
+
|
|
68
|
+
**bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
|
|
69
|
+
|
|
70
|
+
Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from bytesense import from_bytes
|
|
74
|
+
|
|
75
|
+
data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
|
|
76
|
+
result = from_bytes(data)
|
|
77
|
+
if result.encoding:
|
|
78
|
+
text = data.decode(result.encoding) # strict decoding; no replacement characters
|
|
79
|
+
print(result.encoding, result.bytes_validated, result.complete)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Install
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install bytesense
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
|
|
92
|
+
BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
|
|
96
|
+
|
|
97
|
+
## Built for dependable imports
|
|
98
|
+
|
|
99
|
+
- **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
|
|
100
|
+
- **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
|
|
101
|
+
- **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
|
|
102
|
+
- **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
|
|
103
|
+
- **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
|
|
104
|
+
- **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
|
|
105
|
+
|
|
106
|
+
Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
|
|
107
|
+
|
|
108
|
+
## Files, streams and existing integrations
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from bytesense import detect, detect_stream, from_path
|
|
112
|
+
|
|
113
|
+
result = from_path("export.csv", include_language=True)
|
|
114
|
+
print(result.to_dict())
|
|
115
|
+
|
|
116
|
+
with open("archive.txt", "rb") as source:
|
|
117
|
+
result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
|
|
118
|
+
assert result.complete
|
|
119
|
+
|
|
120
|
+
# Familiar dictionary interface for code that uses chardet.detect().
|
|
121
|
+
metadata = detect(b"hello world")
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
bytesense report.csv
|
|
128
|
+
bytesense --language --verbose report.csv
|
|
129
|
+
cat report.csv | bytesense --minimal -
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
133
|
+
|
|
134
|
+
## Measured accuracy and performance
|
|
135
|
+
|
|
136
|
+
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
137
|
+
|
|
138
|
+
| Detector | Exact Unicode matches | Accuracy |
|
|
139
|
+
|---|---:|---:|
|
|
140
|
+
| **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
|
|
141
|
+
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
142
|
+
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
143
|
+
| charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
|
|
144
|
+
| chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
|
|
145
|
+
|
|
146
|
+
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
147
|
+
|
|
148
|
+
## Upgrading from 0.x
|
|
149
|
+
|
|
150
|
+
- Language reporting is opt-in: `include_language=True`.
|
|
151
|
+
- `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
|
|
152
|
+
- An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
|
|
153
|
+
- BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
|
|
154
|
+
- No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
|
|
155
|
+
- `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
|
|
156
|
+
- Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
|
|
157
|
+
|
|
158
|
+
[Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
|
|
159
|
+
|
|
160
|
+
Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
bytesense/__init__.py,sha256=4G1g_QKN6tga0OfyGyZq8eGxSQikwBKykYvkOCOB2hs,1201
|
|
2
|
+
bytesense/_rust.py,sha256=eA_IywAQI-wq478mT8LlSjMAH3vIz1Gh98_AIdUZPQc,1765
|
|
3
|
+
bytesense/api.py,sha256=AtBzk8LbKH1O6YQGcVh3_r80GW_C4n_Y5vgs_atQKO4,17336
|
|
4
|
+
bytesense/candidate.py,sha256=Pb6HxGettU0M8kA646jqIJVYYqIbSXX_RbuXZbn2-QA,6836
|
|
5
|
+
bytesense/cli.py,sha256=D7tAocEeHjpFpDywqfUF0fLf-Ku-W5s0kfoTIj4dVJE,2603
|
|
6
|
+
bytesense/coherence.py,sha256=GybGV0haTxRgnVZ3wr-IpHCvX5Htn9AEryGAlHEdxw4,2282
|
|
7
|
+
bytesense/constant.py,sha256=d6Fb4ldzEo8sDgC9iOMw69bKc6XPX6H-fc4ThuH_JXM,22841
|
|
8
|
+
bytesense/fingerprint.py,sha256=Se42g2-AHyGatrNxRKVQLMHqOmhw6TZqWIFEZB4qe7U,7703
|
|
9
|
+
bytesense/heuristics.py,sha256=6HAqRQ7jBQ6ivnTB0Nsw0347gUvbRBrDaytJ67enOfs,5135
|
|
10
|
+
bytesense/hints.py,sha256=dCuEcw3HlyikMYXyfaaNkBoyMEX4MFjrPf12sOHBzbU,3139
|
|
11
|
+
bytesense/legacy.py,sha256=pTz4FbHFSso_2rbDbJozDirirzE_5pDPr5WKgx9Jk8I,1032
|
|
12
|
+
bytesense/mess.py,sha256=z2Gi-UFy34fo0t1wGPqWxgGWnUu6KhlI5WqkRpizUHM,5219
|
|
13
|
+
bytesense/models.py,sha256=5neGZYQHnd9MrUI2QbPbsS-hvu96TsB4VMD0BS4mZoU,3598
|
|
14
|
+
bytesense/multi.py,sha256=q9QGxmI7StyeazafgCod4J319zObW2EgVpIwEY33Lm4,7370
|
|
15
|
+
bytesense/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
16
|
+
bytesense/repair.py,sha256=9Vr5byaqnC5TyqwEK731MIddxlOoUNpAK7zI5IPsTvw,9421
|
|
17
|
+
bytesense/scoring.py,sha256=hudsXocAwT7H_Vp7RY3CaOhmYXbVZgfuWO1-TiuJSAw,3163
|
|
18
|
+
bytesense/streaming.py,sha256=aBEooT6y3aGuURwZNhculQGSpiwOJGJdGgHHJG--oig,16644
|
|
19
|
+
bytesense/version.py,sha256=FumfuTF6gd-I--l6_x0cb828NjsK2PsBpghXnklN-uI,105
|
|
20
|
+
bytesense/data/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
21
|
+
bytesense/data/fingerprints.py,sha256=LkcpUsBrobaCVMdYYlQ30fJu2dG-78-iOMRSQ0zxWwo,221882
|
|
22
|
+
bytesense/data/language.json.gz,sha256=IhH1xsFnr-nOBg4oovwIZw_5t4OfOnr00ghdOg68Qgo,224991
|
|
23
|
+
bytesense-1.0.0.dist-info/licenses/LICENSE,sha256=M8CBGXubxHI3vsjlww0lhntaHbNteNGCZDNbh2-6sJs,1070
|
|
24
|
+
bytesense-1.0.0.dist-info/METADATA,sha256=FsRRoLnOwXKr_kdK-rjcrX9ly6zS_RvvGO6BI3l0VNM,9368
|
|
25
|
+
bytesense-1.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
26
|
+
bytesense-1.0.0.dist-info/entry_points.txt,sha256=iAlLc4aTbSGKTaplaCRIOUS9OfpyVsem0Hxipn_dv-I,49
|
|
27
|
+
bytesense-1.0.0.dist-info/top_level.txt,sha256=LjKoSh0s5BHlWtC22zfvIEa7sMD4pvJ8mtrAH8FRUhs,10
|
|
28
|
+
bytesense-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Oğuzhan Kır
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
bytesense
|