bytesense 1.1.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. {bytesense-1.1.0 → bytesense-1.2.0}/CHANGELOG.md +15 -0
  2. {bytesense-1.1.0 → bytesense-1.2.0}/MANIFEST.in +1 -1
  3. {bytesense-1.1.0/src/bytesense.egg-info → bytesense-1.2.0}/PKG-INFO +13 -4
  4. {bytesense-1.1.0 → bytesense-1.2.0}/README.md +11 -3
  5. bytesense-1.2.0/THIRD_PARTY_LICENSES +23 -0
  6. {bytesense-1.1.0 → bytesense-1.2.0}/pyproject.toml +4 -1
  7. {bytesense-1.1.0 → bytesense-1.2.0}/rust/Cargo.lock +8 -1
  8. {bytesense-1.1.0 → bytesense-1.2.0}/rust/Cargo.toml +2 -1
  9. {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/utf8.rs +3 -1
  10. bytesense-1.2.0/scripts/benchmark_compare.py +234 -0
  11. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/benchmark_v1.py +4 -0
  12. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_dist.py +2 -0
  13. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/api.py +10 -3
  14. bytesense-1.2.0/src/bytesense/version.py +4 -0
  15. {bytesense-1.1.0 → bytesense-1.2.0/src/bytesense.egg-info}/PKG-INFO +13 -4
  16. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/SOURCES.txt +3 -0
  17. bytesense-1.2.0/tests/test_utf8_validation.py +109 -0
  18. bytesense-1.1.0/src/bytesense/version.py +0 -4
  19. {bytesense-1.1.0 → bytesense-1.2.0}/CODE_OF_CONDUCT.md +0 -0
  20. {bytesense-1.1.0 → bytesense-1.2.0}/CONTRIBUTING.md +0 -0
  21. {bytesense-1.1.0 → bytesense-1.2.0}/LICENSE +0 -0
  22. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/__init__.py +0 -0
  23. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/cn_official_manifest.json +0 -0
  24. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/conftest.py +0 -0
  25. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/hard_scenarios.py +0 -0
  26. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/test_bench_detection.py +0 -0
  27. {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/test_hard_scenarios.py +0 -0
  28. {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/histogram.rs +0 -0
  29. {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/language.rs +0 -0
  30. {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/lib.rs +0 -0
  31. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/build_fingerprints.py +0 -0
  32. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_evaluation.py +0 -0
  33. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_version.py +0 -0
  34. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/compare_libraries.sh +0 -0
  35. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/evaluate_corpus.py +0 -0
  36. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/evaluate_udhr.py +0 -0
  37. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/fetch_cn_benchmark_samples.py +0 -0
  38. {bytesense-1.1.0 → bytesense-1.2.0}/scripts/run_all_benchmarks.sh +0 -0
  39. {bytesense-1.1.0 → bytesense-1.2.0}/setup.cfg +0 -0
  40. {bytesense-1.1.0 → bytesense-1.2.0}/setup.py +0 -0
  41. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/__init__.py +0 -0
  42. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/_rust.py +0 -0
  43. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/candidate.py +0 -0
  44. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/cli.py +0 -0
  45. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/coherence.py +0 -0
  46. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/constant.py +0 -0
  47. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/__init__.py +0 -0
  48. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/fingerprints.py +0 -0
  49. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/language.json.gz +0 -0
  50. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/fingerprint.py +0 -0
  51. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/heuristics.py +0 -0
  52. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/hints.py +0 -0
  53. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/legacy.py +0 -0
  54. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/mess.py +0 -0
  55. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/models.py +0 -0
  56. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/multi.py +0 -0
  57. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/py.typed +0 -0
  58. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/repair.py +0 -0
  59. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/scoring.py +0 -0
  60. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/streaming.py +0 -0
  61. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/dependency_links.txt +0 -0
  62. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/entry_points.txt +0 -0
  63. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/not-zip-safe +0 -0
  64. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/requires.txt +0 -0
  65. {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/top_level.txt +0 -0
  66. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_api.py +0 -0
  67. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_api_extra.py +0 -0
  68. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_candidate.py +0 -0
  69. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_cli.py +0 -0
  70. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_coherence.py +0 -0
  71. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_fingerprint.py +0 -0
  72. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_heuristics_extra.py +0 -0
  73. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_hints.py +0 -0
  74. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_hints_extra.py +0 -0
  75. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_legacy.py +0 -0
  76. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_mess.py +0 -0
  77. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_mess_extra.py +0 -0
  78. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_multi.py +0 -0
  79. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_repair.py +0 -0
  80. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_rust.py +0 -0
  81. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_rust_layer.py +0 -0
  82. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_streaming.py +0 -0
  83. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_streaming_v2.py +0 -0
  84. {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_v1_contracts.py +0 -0
@@ -1,5 +1,20 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.2.0
4
+
5
+ ### UTF-8 validation
6
+ - Validate complete UTF-8 and UTF-8-SIG inputs in `from_bytes()` with SIMD in native builds, without allocating decoded Unicode. Pin `simdutf8` 0.1.5 and preserve strict rejection and first-error offsets with its compatibility validator.
7
+ - Retain runtime x86 CPU dispatch and scalar fallbacks; no target-specific CPU flags or Python runtime dependencies are required.
8
+ - Decode pure-Python UTF-8 inputs larger than 64 KiB in bounded blocks. Preserve the existing threshold for other codecs, including custom decoders.
9
+ - Reuse the ASCII check within each detection call. Public results, scoring and stream behavior are unchanged.
10
+
11
+ ### Verification
12
+ - Add strict-decoder parity, malformed-scalar, SIMD/block boundary, allocation and custom-codec regressions.
13
+ - Extend rotating-input benchmarks with 64 KiB and 8 MiB UTF-8 workloads and isolated before/after comparisons.
14
+ - Preserve existing corpus accuracy and native/Python predictions; include the native dependency's license in distributions.
15
+
16
+ See the benchmark documentation for measured gains and their scope.
17
+
3
18
  ## 1.1.0
4
19
 
5
20
  ### Faster legacy detection
@@ -1,4 +1,4 @@
1
- include LICENSE README.md CHANGELOG.md
1
+ include LICENSE THIRD_PARTY_LICENSES README.md CHANGELOG.md
2
2
  recursive-include rust *.toml *.lock *.rs
3
3
  recursive-include src/bytesense *.py py.typed
4
4
  prune rust/target
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bytesense
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
5
5
  Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
6
6
  Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
33
33
  Requires-Python: >=3.9
34
34
  Description-Content-Type: text/markdown
35
35
  License-File: LICENSE
36
+ License-File: THIRD_PARTY_LICENSES
36
37
  Provides-Extra: fast
37
38
  Provides-Extra: docs
38
39
  Requires-Dist: mkdocs>=1.6; extra == "docs"
@@ -131,13 +132,21 @@ cat report.csv | bytesense --minimal -
131
132
 
132
133
  The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
133
134
 
135
+ ## Used by
136
+
137
+ [ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
138
+ detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
139
+ safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
140
+ Built by the same maintainer; an independent project, not affiliated with Sentry.
141
+
134
142
  ## Measured accuracy and performance
135
143
 
136
144
  The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
137
145
 
138
146
  | Detector | Exact Unicode matches | Accuracy |
139
147
  |---|---:|---:|
140
- | **bytesense 1.1.0** | **1907 / 2078** | **91.77%** |
148
+ | **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
149
+ | bytesense 1.1.0 | 1907 / 2078 | 91.77% |
141
150
  | bytesense 1.0.0 | 1906 / 2078 | 91.72% |
142
151
  | bytesense 0.1.2 | 883 / 2078 | 42.49% |
143
152
  | chardet 7.6.0 | 2060 / 2078 | 99.13% |
@@ -146,9 +155,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
146
155
 
147
156
  See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
148
157
 
149
- **New in 1.1:** faster legacy scoring, canonical-Unicode evidence, and deeper full-input validation. On a new 432-case transfer check built from 28 UDHR translations, exact recovery increased from **374 to 394 cases**; charset-normalizer recovered 381 and chardet 422. These are correlated variants of one translated document, not a universal accuracy ranking. Chardet remains more accurate on both reported corpora.
158
+ **New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
150
159
 
151
- The measured Turkish CP1254 and Japanese EUC-JP workloads run approximately **2.8× faster than bytesense 1.0** with the native backend. See the benchmark table for absolute timings and where competitors remain faster.
160
+ The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
152
161
 
153
162
  ## Upgrading from 0.x
154
163
 
@@ -74,13 +74,21 @@ cat report.csv | bytesense --minimal -
74
74
 
75
75
  The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
76
76
 
77
+ ## Used by
78
+
79
+ [ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
80
+ detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
81
+ safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
82
+ Built by the same maintainer; an independent project, not affiliated with Sentry.
83
+
77
84
  ## Measured accuracy and performance
78
85
 
79
86
  The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
80
87
 
81
88
  | Detector | Exact Unicode matches | Accuracy |
82
89
  |---|---:|---:|
83
- | **bytesense 1.1.0** | **1907 / 2078** | **91.77%** |
90
+ | **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
91
+ | bytesense 1.1.0 | 1907 / 2078 | 91.77% |
84
92
  | bytesense 1.0.0 | 1906 / 2078 | 91.72% |
85
93
  | bytesense 0.1.2 | 883 / 2078 | 42.49% |
86
94
  | chardet 7.6.0 | 2060 / 2078 | 99.13% |
@@ -89,9 +97,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
89
97
 
90
98
  See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
91
99
 
92
- **New in 1.1:** faster legacy scoring, canonical-Unicode evidence, and deeper full-input validation. On a new 432-case transfer check built from 28 UDHR translations, exact recovery increased from **374 to 394 cases**; charset-normalizer recovered 381 and chardet 422. These are correlated variants of one translated document, not a universal accuracy ranking. Chardet remains more accurate on both reported corpora.
100
+ **New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
93
101
 
94
- The measured Turkish CP1254 and Japanese EUC-JP workloads run approximately **2.8× faster than bytesense 1.0** with the native backend. See the benchmark table for absolute timings and where competitors remain faster.
102
+ The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
95
103
 
96
104
  ## Upgrading from 0.x
97
105
 
@@ -0,0 +1,23 @@
1
+ The optional native extension includes simdutf8 0.1.5.
2
+ Source: https://github.com/rusticstuff/simdutf8/tree/87ee8d9d20b849eae1974a82b248ee6fe7491bcc
3
+ Upstream license: MIT OR Apache-2.0; distributed under the MIT option below.
4
+
5
+ MIT License
6
+
7
+ Permission is hereby granted, free of charge, to any person obtaining a copy
8
+ of this software and associated documentation files (the "Software"), to deal
9
+ in the Software without restriction, including without limitation the rights
10
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11
+ copies of the Software, and to permit persons to whom the Software is
12
+ furnished to do so, subject to the following conditions:
13
+
14
+ The above copyright notice and this permission notice shall be included in all
15
+ copies or substantial portions of the Software.
16
+
17
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
20
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
23
+ SOFTWARE.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "bytesense"
7
- version = "1.1.0"
7
+ version = "1.2.0"
8
8
  description = "Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -82,6 +82,9 @@ Issues = "https://github.com/oguzhankir/bytesense/issues"
82
82
  Changelog = "https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md"
83
83
  Documentation = "https://oguzhankir.github.io/bytesense/"
84
84
 
85
+ [tool.setuptools]
86
+ license-files = ["LICENSE", "THIRD_PARTY_LICENSES"]
87
+
85
88
  [tool.setuptools.packages.find]
86
89
  where = ["src"]
87
90
 
@@ -4,9 +4,10 @@ version = 4
4
4
 
5
5
  [[package]]
6
6
  name = "bytesense-core"
7
- version = "1.1.0"
7
+ version = "1.2.0"
8
8
  dependencies = [
9
9
  "pyo3",
10
+ "simdutf8",
10
11
  ]
11
12
 
12
13
  [[package]]
@@ -109,6 +110,12 @@ dependencies = [
109
110
  "proc-macro2",
110
111
  ]
111
112
 
113
+ [[package]]
114
+ name = "simdutf8"
115
+ version = "0.1.5"
116
+ source = "registry+https://github.com/rust-lang/crates.io-index"
117
+ checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
118
+
112
119
  [[package]]
113
120
  name = "syn"
114
121
  version = "2.0.117"
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "bytesense-core"
3
- version = "1.1.0"
3
+ version = "1.2.0"
4
4
  edition = "2021"
5
5
  rust-version = "1.88"
6
6
 
@@ -10,6 +10,7 @@ crate-type = ["cdylib"]
10
10
 
11
11
  [dependencies]
12
12
  pyo3 = { version = "0.28.2", features = ["abi3-py39"] }
13
+ simdutf8 = "=0.1.5"
13
14
 
14
15
  [profile.release]
15
16
  opt-level = 3
@@ -1,6 +1,8 @@
1
1
  /// Returns (is_valid, confidence_0_to_1).
2
2
  pub fn utf8_check(data: &[u8]) -> (bool, f64) {
3
- match std::str::from_utf8(data) {
3
+ // The compat validator preserves the first-error offset and early exit.
4
+ // x86 CPU features are detected at runtime; unsupported targets use scalar validation.
5
+ match simdutf8::compat::from_utf8(data) {
4
6
  Ok(_) => (true, 1.0),
5
7
  Err(e) => {
6
8
  let conf = if data.is_empty() {
@@ -0,0 +1,234 @@
1
+ """Compare installed baseline/candidate packages in paired, isolated processes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import contextlib
7
+ import hashlib
8
+ import importlib.metadata
9
+ import json
10
+ import os
11
+ import platform
12
+ import random
13
+ import statistics
14
+ import subprocess
15
+ import sys
16
+ import time
17
+ import tracemalloc
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ from benchmark_v1 import samples
22
+ from evaluate_corpus import engines
23
+
24
+ EXPECTED = {
25
+ "ascii_1mib": "ascii",
26
+ "utf8_1mib": "utf_8",
27
+ "utf8_8kib": "utf_8",
28
+ "utf8_64kib": "utf_8",
29
+ "utf8_8mib": "utf_8",
30
+ "turkish_cp1254_2100b": "cp1254",
31
+ "japanese_eucjp_4kib": "euc_jp",
32
+ }
33
+ COMPETITORS = ["chardet", "charset-normalizer", "chardetng-py"]
34
+
35
+
36
+ def worker(variant: str, competitors: list[str]) -> None:
37
+ import bytesense.api
38
+ from bytesense._rust import is_rust_available
39
+
40
+ names = ["bytesense"] + (competitors if variant == "candidate" else [])
41
+ detectors = engines(names)
42
+ payloads = samples()
43
+ for inputs in payloads.values():
44
+ for data in inputs[:4]:
45
+ for detector in detectors.values():
46
+ detector(data)
47
+ metadata = {
48
+ "ready": True,
49
+ "native": is_rust_available(),
50
+ "python": platform.python_version(),
51
+ "versions": {name: importlib.metadata.version(name) for name in names},
52
+ "api_sha256": hashlib.sha256(Path(bytesense.api.__file__).read_bytes()).hexdigest(),
53
+ }
54
+ if is_rust_available():
55
+ import bytesense._rust_core
56
+
57
+ metadata["native_extension_sha256"] = hashlib.sha256(
58
+ Path(bytesense._rust_core.__file__).read_bytes()
59
+ ).hexdigest()
60
+ print(json.dumps(metadata), flush=True)
61
+ for line in sys.stdin:
62
+ request = json.loads(line)
63
+ case, engine = request["case"], request["engine"]
64
+ data = payloads[case][request["index"]]
65
+
66
+ def call(
67
+ payload: bytes = data,
68
+ detect=detectors[engine],
69
+ validate: bool = request["mode"] == "full_validation",
70
+ ) -> str | None:
71
+ encoding = detect(payload)
72
+ if validate and encoding:
73
+ payload.decode(encoding, errors="strict")
74
+ return encoding
75
+
76
+ if request.get("allocation"):
77
+ tracemalloc.start()
78
+ try:
79
+ call()
80
+ _, peak = tracemalloc.get_traced_memory()
81
+ finally:
82
+ tracemalloc.stop()
83
+ result = {"peak_bytes": peak}
84
+ else:
85
+ start = time.perf_counter_ns()
86
+ encoding = call()
87
+ elapsed = (time.perf_counter_ns() - start) / 1e6
88
+ try:
89
+ correct = bool(encoding and data.decode(encoding) == data.decode(EXPECTED[case]))
90
+ except UnicodeError:
91
+ correct = False
92
+ result = {"ms": elapsed, "correct": correct}
93
+ print(json.dumps(result), flush=True)
94
+
95
+
96
+ class Worker:
97
+ def __init__(self, python: Path, variant: str, competitors: list[str]):
98
+ self.variant = variant
99
+ environment = dict(os.environ)
100
+ # Each supplied interpreter must resolve its own installed package.
101
+ environment.pop("PYTHONPATH", None)
102
+ # Keep virtualenv interpreter symlinks intact: resolving one loses its
103
+ # environment and can import a different installed package.
104
+ arguments = [str(python.absolute()), "-u", str(Path(__file__).resolve()), "--worker", variant]
105
+ for competitor in competitors:
106
+ arguments.extend(["--engine", competitor])
107
+ self.process = subprocess.Popen(
108
+ arguments,
109
+ stdin=subprocess.PIPE,
110
+ stdout=subprocess.PIPE,
111
+ text=True,
112
+ env=environment,
113
+ )
114
+
115
+ def read(self) -> dict[str, Any]:
116
+ assert self.process.stdout is not None
117
+ line = self.process.stdout.readline()
118
+ if not line:
119
+ raise RuntimeError(
120
+ f"{self.variant} worker closed its output; inspect its stderr above "
121
+ f"(exit status: {self.process.poll()})."
122
+ )
123
+ return json.loads(line)
124
+
125
+ def request(self, **payload: Any) -> dict[str, Any]:
126
+ assert self.process.stdin is not None
127
+ self.process.stdin.write(json.dumps(payload) + "\n")
128
+ self.process.stdin.flush()
129
+ return self.read()
130
+
131
+ def close(self) -> None:
132
+ if self.process.stdin is not None:
133
+ with contextlib.suppress(BrokenPipeError):
134
+ self.process.stdin.close()
135
+ try:
136
+ self.process.wait(timeout=5)
137
+ except subprocess.TimeoutExpired:
138
+ self.process.terminate()
139
+ try:
140
+ self.process.wait(timeout=5)
141
+ except subprocess.TimeoutExpired:
142
+ self.process.kill()
143
+ self.process.wait()
144
+ if self.process.stdout is not None:
145
+ self.process.stdout.close()
146
+
147
+
148
+ def main() -> None:
149
+ parser = argparse.ArgumentParser(description=__doc__)
150
+ parser.add_argument("--baseline-python", type=Path)
151
+ parser.add_argument("--candidate-python", type=Path)
152
+ parser.add_argument("--output", type=Path)
153
+ parser.add_argument("--rounds", type=int, default=128)
154
+ parser.add_argument("--engine", action="append", choices=COMPETITORS)
155
+ parser.add_argument("--worker", choices=["baseline", "candidate"], help=argparse.SUPPRESS)
156
+ args = parser.parse_args()
157
+ competitors = list(dict.fromkeys(args.engine or COMPETITORS))
158
+ if args.worker:
159
+ worker(args.worker, competitors)
160
+ return
161
+ if not args.baseline_python or not args.candidate_python or not args.output:
162
+ parser.error("--baseline-python, --candidate-python and --output are required")
163
+ if args.rounds < 1:
164
+ parser.error("--rounds must be positive")
165
+ processes: dict[str, Worker] = {}
166
+ metadata = {}
167
+ try:
168
+ for variant, python in [
169
+ ("baseline", args.baseline_python), ("candidate", args.candidate_python)
170
+ ]:
171
+ processes[variant] = Worker(python, variant, competitors)
172
+ metadata[variant] = processes[variant].read()
173
+ if not metadata[variant].get("ready"):
174
+ raise RuntimeError(f"{variant} worker did not initialize")
175
+ print(json.dumps(metadata), flush=True)
176
+ detectors = [("baseline", "bytesense"), ("candidate", "bytesense")]
177
+ detectors += [("candidate", name) for name in competitors]
178
+ rng = random.Random(20260919)
179
+ results = []
180
+ for case, payloads in samples().items():
181
+ for mode in ["defaults", "full_validation"]:
182
+ times: dict[tuple[str, str], list[float]] = {key: [] for key in detectors}
183
+ correct = dict.fromkeys(detectors, 0)
184
+ for i in range(args.rounds):
185
+ order = list(detectors)
186
+ rng.shuffle(order)
187
+ for variant, engine in order:
188
+ result = processes[variant].request(
189
+ case=case, engine=engine, index=i % len(payloads), mode=mode
190
+ )
191
+ times[(variant, engine)].append(result["ms"])
192
+ correct[(variant, engine)] += result["correct"]
193
+ for variant, engine in detectors:
194
+ durations = times[(variant, engine)]
195
+ allocation = processes[variant].request(
196
+ case=case, engine=engine, index=0, mode=mode, allocation=True
197
+ )
198
+ result = {
199
+ "case": case,
200
+ "mode": mode,
201
+ "variant": variant,
202
+ "engine": engine,
203
+ "bytes": len(payloads[0]),
204
+ "sha256": hashlib.sha256(b"".join(payloads)).hexdigest(),
205
+ "median_ms": statistics.median(durations),
206
+ "p95_ms": sorted(durations)[int(0.95 * (args.rounds - 1))],
207
+ "peak_bytes": allocation["peak_bytes"],
208
+ "correct": correct[(variant, engine)],
209
+ "rounds": args.rounds,
210
+ }
211
+ results.append(result)
212
+ print(json.dumps(result), flush=True)
213
+ args.output.write_text(
214
+ json.dumps(
215
+ {
216
+ "python": platform.python_version(),
217
+ "platform": platform.platform(),
218
+ "metadata": metadata,
219
+ "notes": "32 rotating payloads; four warmups per case; persistent isolated "
220
+ "baseline/candidate processes; seed 20260919 randomized engine order each "
221
+ "round; IPC excluded; tracemalloc excludes native allocations.",
222
+ "results": results,
223
+ },
224
+ indent=2,
225
+ ),
226
+ encoding="utf-8",
227
+ )
228
+ finally:
229
+ for process in processes.values():
230
+ process.close()
231
+
232
+
233
+ if __name__ == "__main__":
234
+ main()
@@ -21,6 +21,8 @@ def samples() -> dict[str, list[bytes]]:
21
21
  "ascii_1mib": (b"The quick brown fox jumps over the lazy dog. ", 1_048_576, "ascii"),
22
22
  "utf8_1mib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 1_048_576, "utf_8"),
23
23
  "utf8_8kib": ("Café dünyası: 中文测试 🎉 ".encode(), 8192, "utf_8"),
24
+ "utf8_64kib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 65536, "utf_8"),
25
+ "utf8_8mib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 8_388_608, "utf_8"),
24
26
  "turkish_cp1254_2100b": (
25
27
  ("İstanbul'da çalışan mühendisler için doğru metin çözümleme. " * 35).encode("cp1254"),
26
28
  2100,
@@ -65,6 +67,8 @@ def main() -> None:
65
67
  "ascii_1mib": "ascii",
66
68
  "utf8_1mib": "utf_8",
67
69
  "utf8_8kib": "utf_8",
70
+ "utf8_64kib": "utf_8",
71
+ "utf8_8mib": "utf_8",
68
72
  "turkish_cp1254_2100b": "cp1254",
69
73
  "japanese_eucjp_4kib": "euc_jp",
70
74
  }
@@ -13,12 +13,14 @@ for path in sorted(Path(sys.argv[1]).iterdir()):
13
13
  assert not any(n.endswith((".so", ".pyd", ".dll")) for n in names)
14
14
  assert "bytesense/py.typed" in names
15
15
  assert "bytesense/data/language.json.gz" in names
16
+ assert any(n.endswith("/THIRD_PARTY_LICENSES") for n in names)
16
17
  assert not any("/target/" in n for n in names)
17
18
  elif path.name.endswith(".tar.gz"):
18
19
  with tarfile.open(path) as archive:
19
20
  names = archive.getnames()
20
21
  for required in (
21
22
  "setup.py",
23
+ "THIRD_PARTY_LICENSES",
22
24
  "scripts/evaluate_corpus.py",
23
25
  "scripts/evaluate_udhr.py",
24
26
  "scripts/check_evaluation.py",
@@ -10,6 +10,7 @@ from functools import lru_cache
10
10
  from os import PathLike
11
11
  from typing import Any, BinaryIO, List, Optional
12
12
 
13
+ from ._rust import is_rust_available, rust_utf8_check
13
14
  from .candidate import CandidateSelector
14
15
  from .coherence import detect_language
15
16
  from .constant import ALL_ENCODINGS
@@ -236,8 +237,13 @@ def _unicode_fallbacks(
236
237
 
237
238
 
238
239
  def _valid(data: bytes, encoding: str) -> bool:
240
+ utf8 = encoding in ("utf_8", "utf_8_sig")
241
+ if utf8 and is_rust_available():
242
+ # A UTF-8 BOM changes decoded text, not strict byte validity. Native
243
+ # validation checks every byte without allocating decoded Unicode.
244
+ return rust_utf8_check(data)[0]
239
245
  try:
240
- if len(data) <= 1_048_576:
246
+ if len(data) <= (65536 if utf8 else 1_048_576):
241
247
  data.decode(encoding, errors="strict")
242
248
  else:
243
249
  decoder = codecs.getincrementaldecoder(encoding)(errors="strict")
@@ -375,7 +381,8 @@ def from_bytes(
375
381
  )
376
382
  r = _result(bom, size, f"BOM declares {bom}; all bytes validated.", 1.0)
377
383
  return replace(r, bom_detected=True)
378
- transport = _transport_encoding(data) if data.isascii() else None
384
+ ascii_data = data.isascii()
385
+ transport = _transport_encoding(data) if ascii_data else None
379
386
  if transport and _allowed(transport, include, exclude):
380
387
  if min_confidence <= 0.85:
381
388
  return _result(
@@ -412,7 +419,7 @@ def from_bytes(
412
419
  )
413
420
  shape = detect_null_pattern(data)
414
421
  if not shape and not _looks_like_iso2022(data):
415
- if data.isascii() and _allowed("ascii", include, exclude):
422
+ if ascii_data and _allowed("ascii", include, exclude):
416
423
  return _result("ascii", size, "All bytes are valid ASCII text.", 1.0)
417
424
  if _allowed("utf_8", include, exclude) and _valid(data, "utf_8"):
418
425
  r = _result("utf_8", size, "All bytes validated as UTF-8.", 0.99)
@@ -0,0 +1,4 @@
1
+ from __future__ import annotations
2
+
3
+ __version__: str = "1.2.0"
4
+ VERSION: tuple[int, int, int] = (1, 2, 0)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bytesense
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
5
5
  Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
6
6
  Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
33
33
  Requires-Python: >=3.9
34
34
  Description-Content-Type: text/markdown
35
35
  License-File: LICENSE
36
+ License-File: THIRD_PARTY_LICENSES
36
37
  Provides-Extra: fast
37
38
  Provides-Extra: docs
38
39
  Requires-Dist: mkdocs>=1.6; extra == "docs"
@@ -131,13 +132,21 @@ cat report.csv | bytesense --minimal -
131
132
 
132
133
  The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
133
134
 
135
+ ## Used by
136
+
137
+ [ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
138
+ detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
139
+ safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
140
+ Built by the same maintainer; an independent project, not affiliated with Sentry.
141
+
134
142
  ## Measured accuracy and performance
135
143
 
136
144
  The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
137
145
 
138
146
  | Detector | Exact Unicode matches | Accuracy |
139
147
  |---|---:|---:|
140
- | **bytesense 1.1.0** | **1907 / 2078** | **91.77%** |
148
+ | **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
149
+ | bytesense 1.1.0 | 1907 / 2078 | 91.77% |
141
150
  | bytesense 1.0.0 | 1906 / 2078 | 91.72% |
142
151
  | bytesense 0.1.2 | 883 / 2078 | 42.49% |
143
152
  | chardet 7.6.0 | 2060 / 2078 | 99.13% |
@@ -146,9 +155,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
146
155
 
147
156
  See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
148
157
 
149
- **New in 1.1:** faster legacy scoring, canonical-Unicode evidence, and deeper full-input validation. On a new 432-case transfer check built from 28 UDHR translations, exact recovery increased from **374 to 394 cases**; charset-normalizer recovered 381 and chardet 422. These are correlated variants of one translated document, not a universal accuracy ranking. Chardet remains more accurate on both reported corpora.
158
+ **New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
150
159
 
151
- The measured Turkish CP1254 and Japanese EUC-JP workloads run approximately **2.8× faster than bytesense 1.0** with the native backend. See the benchmark table for absolute timings and where competitors remain faster.
160
+ The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
152
161
 
153
162
  ## Upgrading from 0.x
154
163
 
@@ -4,6 +4,7 @@ CONTRIBUTING.md
4
4
  LICENSE
5
5
  MANIFEST.in
6
6
  README.md
7
+ THIRD_PARTY_LICENSES
7
8
  pyproject.toml
8
9
  setup.py
9
10
  benchmarks/__init__.py
@@ -18,6 +19,7 @@ rust/src/histogram.rs
18
19
  rust/src/language.rs
19
20
  rust/src/lib.rs
20
21
  rust/src/utf8.rs
22
+ scripts/benchmark_compare.py
21
23
  scripts/benchmark_v1.py
22
24
  scripts/build_fingerprints.py
23
25
  scripts/check_dist.py
@@ -75,4 +77,5 @@ tests/test_rust.py
75
77
  tests/test_rust_layer.py
76
78
  tests/test_streaming.py
77
79
  tests/test_streaming_v2.py
80
+ tests/test_utf8_validation.py
78
81
  tests/test_v1_contracts.py
@@ -0,0 +1,109 @@
1
+ """Strict UTF-8 validation across SIMD, BOM, and incremental decoder boundaries."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import codecs
6
+ import platform
7
+
8
+ import pytest
9
+ from hypothesis import given, settings
10
+ from hypothesis import strategies as st
11
+
12
+ from bytesense import api, from_bytes
13
+ from bytesense._rust import is_rust_available, rust_utf8_check
14
+
15
+
16
+ def _strictly_valid(data: bytes, encoding: str) -> bool:
17
+ try:
18
+ data.decode(encoding, errors="strict")
19
+ except UnicodeError:
20
+ return False
21
+ return True
22
+
23
+
24
+ @pytest.mark.parametrize("encoding", ["utf_8", "utf_8_sig"])
25
+ @given(data=st.one_of(st.binary(max_size=4096), st.text(max_size=4096).map(str.encode)))
26
+ @settings(max_examples=300, deadline=None)
27
+ def test_utf8_validation_matches_strict_python(data: bytes, encoding: str) -> None:
28
+ assert api._valid(data, encoding) == _strictly_valid(data, encoding)
29
+
30
+
31
+ @pytest.mark.parametrize("encoding", ["utf_8", "utf_8_sig"])
32
+ @pytest.mark.parametrize("offset", [31, 32, 63, 64, 65, 65535, 65536, 65537, 1_048_575])
33
+ @pytest.mark.parametrize(
34
+ "suffix",
35
+ [
36
+ "é中🙂".encode(),
37
+ b"\xc0\xaf", # Overlong sequence.
38
+ b"\xed\xa0\x80", # Surrogate scalar.
39
+ b"\xf4\x90\x80\x80", # Above U+10FFFF.
40
+ b"\xe2\x82", # Truncated code point.
41
+ b"\x80", # Unpaired continuation byte.
42
+ ],
43
+ )
44
+ def test_utf8_boundaries_preserve_full_input_validation(
45
+ encoding: str, offset: int, suffix: bytes
46
+ ) -> None:
47
+ prefix = codecs.BOM_UTF8 if encoding == "utf_8_sig" else b""
48
+ data = prefix + b"a" * offset + suffix
49
+ expected = _strictly_valid(data, encoding)
50
+ assert api._valid(data, encoding) == expected
51
+ result = from_bytes(data, cp_isolation=[encoding], enable_fallback=False)
52
+ assert (result.encoding is not None) == expected
53
+ if expected:
54
+ assert result.bytes_validated == len(data)
55
+
56
+
57
+ @pytest.mark.skipif(not is_rust_available(), reason="Native extension not compiled")
58
+ @given(data=st.binary(max_size=8192))
59
+ @settings(max_examples=500, deadline=None)
60
+ def test_native_error_offset_matches_python(data: bytes) -> None:
61
+ try:
62
+ data.decode("utf_8", errors="strict")
63
+ expected = (True, 1.0)
64
+ except UnicodeDecodeError as error:
65
+ expected = (False, error.start / len(data))
66
+ assert rust_utf8_check(data) == expected
67
+
68
+
69
+ @pytest.mark.skipif(platform.python_implementation() != "CPython", reason="CPython allocation probe")
70
+ @pytest.mark.parametrize("native", [False, True])
71
+ def test_utf8_validation_does_not_allocate_full_decoded_input(monkeypatch, native: bool) -> None:
72
+ import tracemalloc
73
+
74
+ if native and not is_rust_available():
75
+ pytest.skip("Native extension not compiled")
76
+ if not native:
77
+ monkeypatch.setattr(api, "is_rust_available", lambda: False)
78
+ data = ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode() * 30_000)[:1_048_576]
79
+ data = data.decode("utf_8", errors="ignore").encode("utf_8")
80
+ # Warm the decoder and feature dispatch before measuring call allocations.
81
+ assert api._valid(data, "utf_8")
82
+ tracemalloc.start()
83
+ try:
84
+ assert api._valid(data, "utf_8")
85
+ _, peak = tracemalloc.get_traced_memory()
86
+ finally:
87
+ tracemalloc.stop()
88
+ assert peak < 600_000
89
+
90
+
91
+ def test_non_utf8_custom_codec_keeps_existing_one_shot_boundary() -> None:
92
+ class BytesOnlyDecoder(codecs.IncrementalDecoder):
93
+ def decode(self, data: bytes, final: bool = False) -> str:
94
+ if not isinstance(data, bytes):
95
+ raise TypeError("decoder requires bytes")
96
+ return data.decode("ascii")
97
+
98
+ def search(name: str) -> codecs.CodecInfo | None:
99
+ if name != "bytesense_test_bytes_only":
100
+ return None
101
+ return codecs.CodecInfo(
102
+ name=name,
103
+ encode=codecs.ascii_encode,
104
+ decode=codecs.ascii_decode,
105
+ incrementaldecoder=BytesOnlyDecoder,
106
+ )
107
+
108
+ codecs.register(search)
109
+ assert api._valid(b"a" * 65537, "bytesense_test_bytes_only")
@@ -1,4 +0,0 @@
1
- from __future__ import annotations
2
-
3
- __version__: str = "1.1.0"
4
- VERSION: tuple[int, int, int] = (1, 1, 0)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes