bytesense 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bytesense/streaming.py ADDED
@@ -0,0 +1,416 @@
1
+ """Bounded-memory streaming with explicit finalization and full decode validation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import codecs
6
+ import tempfile
7
+ from dataclasses import replace
8
+ from typing import Any, Iterator, Optional
9
+
10
+ from .api import (
11
+ _HIGH_BYTE,
12
+ _allowed,
13
+ _binary,
14
+ _bom_codec,
15
+ _filters,
16
+ _from_sample,
17
+ _result,
18
+ _transport_encoding,
19
+ from_bytes,
20
+ )
21
+ from .fingerprint import detect_null_pattern
22
+ from .hints import hint_from_content, hint_from_http_headers
23
+ from .models import DetectionResult
24
+
25
+
26
+ class StreamDetector:
27
+ """Consume every fed byte; spool beyond ``memory_limit`` to a local temp file.
28
+
29
+ Preview scoring runs at exponentially increasing checkpoints, then stops
30
+ after the sample fills. ``is_stable`` describes a guess, never end-of-input.
31
+ ``finalize`` strictly validates the chosen codec over the complete stream.
32
+ Feeding after finalization raises instead of silently dropping data.
33
+ Use a context manager or ``close()`` to release an unfinished stream.
34
+ """
35
+
36
+ MIN_BYTES = 64
37
+ SATURATION = 8192
38
+ STABILITY_ROUNDS = 2
39
+
40
+ def __init__(
41
+ self,
42
+ threshold: float = 0.2,
43
+ language_threshold: float = 0.1,
44
+ auto_stop_confidence: float = 0.97,
45
+ *,
46
+ memory_limit: int = 1_048_576,
47
+ **kwargs: Any,
48
+ ) -> None:
49
+ if not isinstance(memory_limit, int) or isinstance(memory_limit, bool) or memory_limit < 64:
50
+ raise ValueError("memory_limit must be an integer of at least 64 bytes")
51
+ if not 0.0 <= auto_stop_confidence <= 1.0:
52
+ raise ValueError("auto_stop_confidence must be between 0 and 1")
53
+ self._options = dict(kwargs, threshold=threshold, language_threshold=language_threshold)
54
+ from_bytes(b"", **self._options) # Validate configuration before accepting any data.
55
+ self._memory_limit = memory_limit
56
+ self._auto_stop_confidence = auto_stop_confidence
57
+ self._sample_size = kwargs.get("sample_size", 4096)
58
+ self._spool = tempfile.SpooledTemporaryFile(max_size=memory_limit, mode="w+b")
59
+ self._buf = bytearray()
60
+ self._late_sample = bytearray()
61
+ self._tail = b""
62
+ self._bytes_fed = 0
63
+ self._ascii = True
64
+ self._escape = False
65
+ self._next_probe = self.MIN_BYTES
66
+ self._result: Optional[DetectionResult] = None
67
+ self._finalized = False
68
+ self._stable_rounds = 0
69
+ self._prev_encoding: Optional[str] = None
70
+ self._declared_hint: Optional[str] = None
71
+
72
+ def feed(self, chunk: bytes | bytearray) -> None:
73
+ if self._finalized or self._spool.closed:
74
+ raise RuntimeError("stream is closed; call reset() before feeding more data")
75
+ if not isinstance(chunk, (bytes, bytearray)):
76
+ raise TypeError("stream chunks must be bytes or bytearray")
77
+ if not chunk:
78
+ return
79
+ if self._bytes_fed + len(chunk) > self._memory_limit:
80
+ self._spool.rollover()
81
+ self._spool.write(chunk)
82
+ self._bytes_fed += len(chunk)
83
+ ascii_chunk = chunk.isascii()
84
+ if self._late_sample and len(self._late_sample) < self._sample_size:
85
+ self._late_sample.extend(chunk[: self._sample_size - len(self._late_sample)])
86
+ # Preserve a continuous informative window, independent of chunk sizes.
87
+ if (self._ascii and not ascii_chunk) or (not self._escape and b"\x1b" in chunk):
88
+ signal = _HIGH_BYTE.search(chunk)
89
+ if (
90
+ signal is not None
91
+ and self._bytes_fed - len(chunk) + signal.start() > self._sample_size // 4
92
+ ):
93
+ start = signal.start() - 64
94
+ probe = (
95
+ self._tail[start:] + bytes(chunk[: self._sample_size])
96
+ if start < 0
97
+ else bytes(chunk[start : start + self._sample_size])
98
+ )
99
+ self._late_sample = bytearray(probe[: self._sample_size])
100
+ self._tail = bytes(chunk[-64:]) if len(chunk) >= 64 else (self._tail + bytes(chunk))[-64:]
101
+ self._ascii = self._ascii and ascii_chunk
102
+ self._escape = self._escape or b"\x1b" in chunk
103
+ if len(self._buf) < self._sample_size:
104
+ self._buf.extend(chunk[: self._sample_size - len(self._buf)])
105
+ if not self._declared_hint and self._options.get("use_hints", True):
106
+ self._declared_hint = hint_from_content(bytes(self._buf))
107
+ if self._bytes_fed >= self._next_probe and self._next_probe <= self._sample_size:
108
+ self._run()
109
+ while self._next_probe <= self._bytes_fed:
110
+ self._next_probe *= 4
111
+
112
+ def _run(self) -> None:
113
+ probe = bytes(self._late_sample or self._buf)
114
+ r = from_bytes(probe, **self._options)
115
+ self._result = replace(r, byte_count=self._bytes_fed, complete=False, bytes_validated=0)
116
+ if r.encoding == self._prev_encoding:
117
+ self._stable_rounds += 1
118
+ else:
119
+ self._stable_rounds = 0
120
+ self._prev_encoding = r.encoding
121
+
122
+ def hint_from_headers(self, headers: dict[str, str]) -> None:
123
+ if self._finalized:
124
+ raise RuntimeError("cannot change hints on a finalized stream")
125
+ hint = hint_from_http_headers(headers)
126
+ if hint:
127
+ self._declared_hint = hint
128
+
129
+ def _validate(self, encoding: str) -> tuple[bool, bytes]:
130
+ self._spool.seek(0)
131
+ try:
132
+ decoder = codecs.getincrementaldecoder(encoding)(errors="strict")
133
+ except LookupError:
134
+ return False, b""
135
+ chunk = b""
136
+ try:
137
+ while True:
138
+ chunk = self._spool.read(65536)
139
+ if not chunk:
140
+ break
141
+ decoder.decode(chunk, final=False)
142
+ decoder.decode(b"", final=True)
143
+ return True, b""
144
+ except UnicodeDecodeError as exc:
145
+ start = max(0, exc.start - 64)
146
+ return False, chunk[start : start + self._sample_size]
147
+ except UnicodeError:
148
+ return False, chunk[: self._sample_size]
149
+
150
+ def finalize(self) -> DetectionResult:
151
+ if self._finalized:
152
+ assert self._result is not None
153
+ return self._result
154
+ if self._spool.closed:
155
+ raise RuntimeError("stream is closed")
156
+ try:
157
+ self._result = self._finish()
158
+ self._finalized = True
159
+ return self._result
160
+ finally:
161
+ self._spool.close()
162
+
163
+ def _finish(self) -> DetectionResult:
164
+ prefix = bytes(self._buf)
165
+ n = self._bytes_fed
166
+ if n <= self._sample_size:
167
+ options = dict(self._options)
168
+ if self._declared_hint and not options.get("encoding_hint"):
169
+ options["encoding_hint"] = self._declared_hint
170
+ return from_bytes(prefix, **options)
171
+ include, exclude = _filters(
172
+ self._options.get("cp_isolation"), self._options.get("cp_exclusion")
173
+ )
174
+ minimum = self._options.get("min_confidence", 0.0)
175
+ if include == []:
176
+ return _result(None, n, "No encodings allowed.", examined=len(prefix))
177
+ bom = _bom_codec(prefix, include, exclude)
178
+ if bom:
179
+ valid, _ = self._validate(bom)
180
+ if valid:
181
+ return replace(
182
+ _result(bom, n, "BOM; complete stream validated.", 1.0, examined=len(prefix)),
183
+ bom_detected=True,
184
+ )
185
+ return _result(
186
+ None,
187
+ n,
188
+ "Invalid complete stream under its declared BOM.",
189
+ status="invalid",
190
+ examined=len(prefix),
191
+ )
192
+ if _binary(prefix):
193
+ return _result(
194
+ None,
195
+ n,
196
+ "Binary signature or excessive control bytes.",
197
+ status="binary",
198
+ examined=len(prefix),
199
+ )
200
+ transport = _transport_encoding(prefix) if self._ascii else None
201
+ if transport and _allowed(transport, include, exclude):
202
+ valid, _ = self._validate(transport)
203
+ if valid and minimum <= 0.85:
204
+ return _result(
205
+ transport,
206
+ n,
207
+ "7-bit shift syntax; complete stream validated.",
208
+ 0.85,
209
+ "ambiguous",
210
+ len(prefix),
211
+ )
212
+ shape = detect_null_pattern(prefix)
213
+ if shape and _allowed(shape, include, exclude):
214
+ valid, _ = self._validate(shape)
215
+ if valid:
216
+ return (
217
+ _result(
218
+ shape,
219
+ n,
220
+ "Unicode byte-lane pattern; complete stream validated.",
221
+ 0.9,
222
+ examined=len(prefix),
223
+ )
224
+ if minimum <= 0.9
225
+ else _result(None, n, "Below minimum confidence.", examined=len(prefix))
226
+ )
227
+ if not shape and not self._escape:
228
+ if self._ascii and _allowed("ascii", include, exclude):
229
+ return _result("ascii", n, "All stream bytes are ASCII text.", 1.0, examined=n)
230
+ if _allowed("utf_8", include, exclude):
231
+ valid, _ = self._validate("utf_8")
232
+ if valid:
233
+ r = _result("utf_8", n, "Complete stream validated as UTF-8.", 0.99, examined=n)
234
+ if self._options.get("include_language"):
235
+ from .coherence import detect_language
236
+
237
+ decoded = codecs.getincrementaldecoder("utf_8")().decode(
238
+ prefix, final=False
239
+ )
240
+ langs = detect_language(
241
+ decoded, threshold=self._options["language_threshold"]
242
+ )
243
+ if langs:
244
+ r = replace(r, language=langs[0][0], coherence=round(langs[0][1], 4))
245
+ return (
246
+ r
247
+ if r.confidence >= minimum
248
+ else _result(None, n, "Below minimum confidence.", examined=len(prefix))
249
+ )
250
+ probe = bytes(self._late_sample) or prefix
251
+ options = {key: self._options[key] for key in ("threshold", "language_threshold")}
252
+ r = _from_sample(
253
+ probe, cp_isolation=include, cp_exclusion=exclude, enable_fallback=False, **options
254
+ )
255
+ candidates = ([r.encoding] if r.encoding else []) + [alt.encoding for alt in r.alternatives]
256
+ hint = self._options.get("encoding_hint") or self._declared_hint
257
+ if hint:
258
+ from .api import _codec
259
+
260
+ try:
261
+ hint = _codec(hint)
262
+ except (LookupError, TypeError):
263
+ hint = None
264
+ if hint and _allowed(hint, include, exclude):
265
+ candidates.insert(0, hint)
266
+ if self._options.get("enable_fallback", True) and include:
267
+ candidates.extend(include)
268
+ tried: set[str] = set()
269
+ examined = len(probe)
270
+ for enc in candidates:
271
+ if enc in tried or not _allowed(enc, include, exclude):
272
+ continue
273
+ tried.add(enc)
274
+ valid, failure = self._validate(enc)
275
+ if valid:
276
+ confidence = 0.95 if enc == hint else r.confidence if enc == r.encoding else 0.1
277
+ if confidence < minimum:
278
+ continue
279
+ if self._options.get("include_language"):
280
+ from .coherence import detect_language
281
+
282
+ decoded = codecs.getincrementaldecoder(enc)().decode(probe, final=False)
283
+ langs = detect_language(decoded, threshold=self._options["language_threshold"])
284
+ if langs:
285
+ r = replace(r, language=langs[0][0])
286
+ return replace(
287
+ r,
288
+ encoding=enc,
289
+ confidence=confidence,
290
+ byte_count=n,
291
+ bytes_examined=min(n, examined),
292
+ bytes_validated=n,
293
+ complete=True,
294
+ status="ambiguous" if r.alternatives else "matched",
295
+ chaos=r.chaos if enc == r.encoding else 0.0,
296
+ coherence=r.coherence if enc == r.encoding else 0.0,
297
+ alternatives=[a for a in r.alternatives if a.encoding != enc],
298
+ language=r.language
299
+ if self._options.get("include_language") and enc == r.encoding
300
+ else "",
301
+ why=(r.why if enc == r.encoding else f"Selected validated alternative {enc}.")
302
+ + " Complete stream validated."
303
+ + (f" Hint: {hint}." if hint else ""),
304
+ )
305
+ if failure and len(tried) <= 4:
306
+ alternate = _from_sample(
307
+ failure,
308
+ cp_isolation=include,
309
+ cp_exclusion=list(set(exclude) | tried),
310
+ enable_fallback=False,
311
+ **options,
312
+ )
313
+ examined += len(failure)
314
+ if alternate.encoding:
315
+ candidates.append(alternate.encoding)
316
+ return _result(
317
+ None,
318
+ n,
319
+ "No permitted encoding validates the complete stream.",
320
+ examined=min(n, examined),
321
+ )
322
+
323
+ def close(self) -> None:
324
+ self._spool.close()
325
+
326
+ def reset(self) -> None:
327
+ self.close()
328
+ StreamDetector.__init__(
329
+ self,
330
+ auto_stop_confidence=self._auto_stop_confidence,
331
+ memory_limit=self._memory_limit,
332
+ **self._options,
333
+ )
334
+
335
+ def __enter__(self) -> StreamDetector:
336
+ return self
337
+
338
+ def __exit__(self, *args: object) -> None:
339
+ self.close()
340
+
341
+ def snapshot(self) -> dict[str, object]:
342
+ return {
343
+ "bytes_fed": self.bytes_fed,
344
+ "buffered_bytes": len(self._buf),
345
+ "encoding": self.encoding,
346
+ "confidence": self.confidence,
347
+ "language": self.language,
348
+ "stable_rounds": self._stable_rounds,
349
+ "finalized": self._finalized,
350
+ "declared_hint": self._declared_hint,
351
+ }
352
+
353
+ @property
354
+ def result(self) -> Optional[DetectionResult]:
355
+ return self._result
356
+
357
+ @property
358
+ def encoding(self) -> Optional[str]:
359
+ return self._result.encoding if self._result is not None else None
360
+
361
+ @property
362
+ def confidence(self) -> float:
363
+ return self._result.confidence if self._result is not None else 0.0
364
+
365
+ @property
366
+ def language(self) -> str:
367
+ return self._result.language if self._result is not None else ""
368
+
369
+ @property
370
+ def bytes_fed(self) -> int:
371
+ return self._bytes_fed
372
+
373
+ @property
374
+ def is_stable(self) -> bool:
375
+ return self._finalized or (
376
+ self._stable_rounds >= self.STABILITY_ROUNDS
377
+ and self.confidence >= self._auto_stop_confidence
378
+ )
379
+
380
+
381
+ def detect_stream(
382
+ chunks: Iterator[bytes],
383
+ *,
384
+ stop_confidence: float = 0.97,
385
+ max_bytes: Optional[int] = None,
386
+ early_stop: bool = False,
387
+ **kwargs: Any,
388
+ ) -> DetectionResult:
389
+ """Consume to EOF by default. Explicit budgets/early stopping mark incomplete results.
390
+
391
+ An oversized yielded chunk is sliced before analysis. Its unused remainder
392
+ has already been yielded by the source and cannot be restored by this API.
393
+ """
394
+ if max_bytes is not None and (
395
+ not isinstance(max_bytes, int) or isinstance(max_bytes, bool) or max_bytes <= 0
396
+ ):
397
+ raise ValueError("max_bytes must be a positive integer or None")
398
+ with StreamDetector(auto_stop_confidence=stop_confidence, **kwargs) as detector:
399
+ limited = False
400
+ for chunk in chunks:
401
+ if not isinstance(chunk, (bytes, bytearray)):
402
+ raise TypeError("stream chunks must be bytes or bytearray")
403
+ if max_bytes is not None:
404
+ chunk = chunk[: max_bytes - detector.bytes_fed]
405
+ detector.feed(chunk)
406
+ if (max_bytes is not None and detector.bytes_fed >= max_bytes) or (
407
+ early_stop and detector.is_stable
408
+ ):
409
+ limited = True
410
+ break
411
+ r = detector.finalize()
412
+ return (
413
+ replace(r, complete=False, why=r.why + " Input stopped before confirmed EOF.")
414
+ if limited
415
+ else r
416
+ )
bytesense/version.py ADDED
@@ -0,0 +1,4 @@
1
+ from __future__ import annotations
2
+
3
+ __version__: str = "1.0.0"
4
+ VERSION: tuple[int, int, int] = (1, 0, 0)
@@ -0,0 +1,160 @@
1
+ Metadata-Version: 2.4
2
+ Name: bytesense
3
+ Version: 1.0.0
4
+ Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
5
+ Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
6
+ Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
7
+ License: MIT
8
+ Project-URL: Homepage, https://github.com/oguzhankir/bytesense
9
+ Project-URL: Repository, https://github.com/oguzhankir/bytesense
10
+ Project-URL: Issues, https://github.com/oguzhankir/bytesense/issues
11
+ Project-URL: Changelog, https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md
12
+ Project-URL: Documentation, https://oguzhankir.github.io/bytesense/
13
+ Keywords: encoding,charset,detection,unicode,bytes,chardet,charset-normalizer
14
+ Classifier: Development Status :: 5 - Production/Stable
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3.9
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Programming Language :: Python :: 3.12
24
+ Classifier: Programming Language :: Python :: 3.13
25
+ Classifier: Programming Language :: Python :: 3.14
26
+ Classifier: Programming Language :: Python :: 3 :: Only
27
+ Classifier: Programming Language :: Python :: Implementation :: CPython
28
+ Classifier: Programming Language :: Python :: Implementation :: PyPy
29
+ Classifier: Programming Language :: Rust
30
+ Classifier: Topic :: Text Processing :: Linguistic
31
+ Classifier: Topic :: Utilities
32
+ Classifier: Typing :: Typed
33
+ Requires-Python: >=3.9
34
+ Description-Content-Type: text/markdown
35
+ License-File: LICENSE
36
+ Provides-Extra: fast
37
+ Provides-Extra: docs
38
+ Requires-Dist: mkdocs>=1.6; extra == "docs"
39
+ Requires-Dist: mkdocs-material>=9.5; extra == "docs"
40
+ Requires-Dist: pymdown-extensions>=10.0; extra == "docs"
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest>=8.0; extra == "dev"
43
+ Requires-Dist: pytest-cov>=5.0; extra == "dev"
44
+ Requires-Dist: pytest-benchmark>=4.0; extra == "dev"
45
+ Requires-Dist: mypy>=1.8; extra == "dev"
46
+ Requires-Dist: ruff>=0.4; extra == "dev"
47
+ Requires-Dist: build>=1.2; extra == "dev"
48
+ Requires-Dist: hypothesis>=6.100; (platform_python_implementation != "PyPy" or python_version >= "3.11") and extra == "dev"
49
+ Requires-Dist: hypothesis==6.100.0; (platform_python_implementation == "PyPy" and python_version < "3.11") and extra == "dev"
50
+ Requires-Dist: setuptools-rust>=1.12; extra == "dev"
51
+ Requires-Dist: charset-normalizer>=3.3; extra == "dev"
52
+ Requires-Dist: chardet>=5.0; extra == "dev"
53
+ Requires-Dist: requests>=2.31; extra == "dev"
54
+ Requires-Dist: certifi>=2023.0; extra == "dev"
55
+ Requires-Dist: twine>=5.0; extra == "dev"
56
+ Dynamic: license-file
57
+
58
+ <p align="center">
59
+ <img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
60
+ </p>
61
+ <p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
62
+ <p align="center">
63
+ <a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
64
+ <a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
65
+ <a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
66
+ </p>
67
+
68
+ **bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
69
+
70
+ Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
71
+
72
+ ```python
73
+ from bytesense import from_bytes
74
+
75
+ data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
76
+ result = from_bytes(data)
77
+ if result.encoding:
78
+ text = data.decode(result.encoding) # strict decoding; no replacement characters
79
+ print(result.encoding, result.bytes_validated, result.complete)
80
+ ```
81
+
82
+ ## Install
83
+
84
+ ```bash
85
+ pip install bytesense
86
+ ```
87
+
88
+ Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
89
+
90
+ ```bash
91
+ BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
92
+ BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
93
+ ```
94
+
95
+ The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
96
+
97
+ ## Built for dependable imports
98
+
99
+ - **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
100
+ - **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
101
+ - **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
102
+ - **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
103
+ - **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
104
+ - **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
105
+
106
+ Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
107
+
108
+ ## Files, streams and existing integrations
109
+
110
+ ```python
111
+ from bytesense import detect, detect_stream, from_path
112
+
113
+ result = from_path("export.csv", include_language=True)
114
+ print(result.to_dict())
115
+
116
+ with open("archive.txt", "rb") as source:
117
+ result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
118
+ assert result.complete
119
+
120
+ # Familiar dictionary interface for code that uses chardet.detect().
121
+ metadata = detect(b"hello world")
122
+ ```
123
+
124
+ `detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
125
+
126
+ ```bash
127
+ bytesense report.csv
128
+ bytesense --language --verbose report.csv
129
+ cat report.csv | bytesense --minimal -
130
+ ```
131
+
132
+ The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
133
+
134
+ ## Measured accuracy and performance
135
+
136
+ The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
137
+
138
+ | Detector | Exact Unicode matches | Accuracy |
139
+ |---|---:|---:|
140
+ | **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
141
+ | bytesense 0.1.2 | 883 / 2078 | 42.49% |
142
+ | chardet 7.6.0 | 2060 / 2078 | 99.13% |
143
+ | charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
144
+ | chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
145
+
146
+ See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
147
+
148
+ ## Upgrading from 0.x
149
+
150
+ - Language reporting is opt-in: `include_language=True`.
151
+ - `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
152
+ - An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
153
+ - BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
154
+ - No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
155
+ - `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
156
+ - Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
157
+
158
+ [Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
159
+
160
+ Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
@@ -0,0 +1,28 @@
1
+ bytesense/__init__.py,sha256=4G1g_QKN6tga0OfyGyZq8eGxSQikwBKykYvkOCOB2hs,1201
2
+ bytesense/_rust.py,sha256=eA_IywAQI-wq478mT8LlSjMAH3vIz1Gh98_AIdUZPQc,1765
3
+ bytesense/api.py,sha256=AtBzk8LbKH1O6YQGcVh3_r80GW_C4n_Y5vgs_atQKO4,17336
4
+ bytesense/candidate.py,sha256=Pb6HxGettU0M8kA646jqIJVYYqIbSXX_RbuXZbn2-QA,6836
5
+ bytesense/cli.py,sha256=D7tAocEeHjpFpDywqfUF0fLf-Ku-W5s0kfoTIj4dVJE,2603
6
+ bytesense/coherence.py,sha256=GybGV0haTxRgnVZ3wr-IpHCvX5Htn9AEryGAlHEdxw4,2282
7
+ bytesense/constant.py,sha256=d6Fb4ldzEo8sDgC9iOMw69bKc6XPX6H-fc4ThuH_JXM,22841
8
+ bytesense/fingerprint.py,sha256=Se42g2-AHyGatrNxRKVQLMHqOmhw6TZqWIFEZB4qe7U,7703
9
+ bytesense/heuristics.py,sha256=6HAqRQ7jBQ6ivnTB0Nsw0347gUvbRBrDaytJ67enOfs,5135
10
+ bytesense/hints.py,sha256=dCuEcw3HlyikMYXyfaaNkBoyMEX4MFjrPf12sOHBzbU,3139
11
+ bytesense/legacy.py,sha256=pTz4FbHFSso_2rbDbJozDirirzE_5pDPr5WKgx9Jk8I,1032
12
+ bytesense/mess.py,sha256=z2Gi-UFy34fo0t1wGPqWxgGWnUu6KhlI5WqkRpizUHM,5219
13
+ bytesense/models.py,sha256=5neGZYQHnd9MrUI2QbPbsS-hvu96TsB4VMD0BS4mZoU,3598
14
+ bytesense/multi.py,sha256=q9QGxmI7StyeazafgCod4J319zObW2EgVpIwEY33Lm4,7370
15
+ bytesense/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
16
+ bytesense/repair.py,sha256=9Vr5byaqnC5TyqwEK731MIddxlOoUNpAK7zI5IPsTvw,9421
17
+ bytesense/scoring.py,sha256=hudsXocAwT7H_Vp7RY3CaOhmYXbVZgfuWO1-TiuJSAw,3163
18
+ bytesense/streaming.py,sha256=aBEooT6y3aGuURwZNhculQGSpiwOJGJdGgHHJG--oig,16644
19
+ bytesense/version.py,sha256=FumfuTF6gd-I--l6_x0cb828NjsK2PsBpghXnklN-uI,105
20
+ bytesense/data/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
21
+ bytesense/data/fingerprints.py,sha256=LkcpUsBrobaCVMdYYlQ30fJu2dG-78-iOMRSQ0zxWwo,221882
22
+ bytesense/data/language.json.gz,sha256=IhH1xsFnr-nOBg4oovwIZw_5t4OfOnr00ghdOg68Qgo,224991
23
+ bytesense-1.0.0.dist-info/licenses/LICENSE,sha256=M8CBGXubxHI3vsjlww0lhntaHbNteNGCZDNbh2-6sJs,1070
24
+ bytesense-1.0.0.dist-info/METADATA,sha256=FsRRoLnOwXKr_kdK-rjcrX9ly6zS_RvvGO6BI3l0VNM,9368
25
+ bytesense-1.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
26
+ bytesense-1.0.0.dist-info/entry_points.txt,sha256=iAlLc4aTbSGKTaplaCRIOUS9OfpyVsem0Hxipn_dv-I,49
27
+ bytesense-1.0.0.dist-info/top_level.txt,sha256=LjKoSh0s5BHlWtC22zfvIEa7sMD4pvJ8mtrAH8FRUhs,10
28
+ bytesense-1.0.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ bytesense = bytesense.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Oğuzhan Kır
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ bytesense