bytesense 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bytesense-1.1.0 → bytesense-1.2.0}/CHANGELOG.md +15 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/MANIFEST.in +1 -1
- {bytesense-1.1.0/src/bytesense.egg-info → bytesense-1.2.0}/PKG-INFO +13 -4
- {bytesense-1.1.0 → bytesense-1.2.0}/README.md +11 -3
- bytesense-1.2.0/THIRD_PARTY_LICENSES +23 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/pyproject.toml +4 -1
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/Cargo.lock +8 -1
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/Cargo.toml +2 -1
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/utf8.rs +3 -1
- bytesense-1.2.0/scripts/benchmark_compare.py +234 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/benchmark_v1.py +4 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_dist.py +2 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/api.py +10 -3
- bytesense-1.2.0/src/bytesense/version.py +4 -0
- {bytesense-1.1.0 → bytesense-1.2.0/src/bytesense.egg-info}/PKG-INFO +13 -4
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/SOURCES.txt +3 -0
- bytesense-1.2.0/tests/test_utf8_validation.py +109 -0
- bytesense-1.1.0/src/bytesense/version.py +0 -4
- {bytesense-1.1.0 → bytesense-1.2.0}/CODE_OF_CONDUCT.md +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/CONTRIBUTING.md +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/LICENSE +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/__init__.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/cn_official_manifest.json +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/conftest.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/hard_scenarios.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/test_bench_detection.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/benchmarks/test_hard_scenarios.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/histogram.rs +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/language.rs +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/rust/src/lib.rs +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/build_fingerprints.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_evaluation.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/check_version.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/compare_libraries.sh +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/evaluate_corpus.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/evaluate_udhr.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/fetch_cn_benchmark_samples.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/scripts/run_all_benchmarks.sh +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/setup.cfg +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/setup.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/__init__.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/_rust.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/candidate.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/cli.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/coherence.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/constant.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/__init__.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/fingerprints.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/data/language.json.gz +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/fingerprint.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/heuristics.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/hints.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/legacy.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/mess.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/models.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/multi.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/py.typed +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/repair.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/scoring.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense/streaming.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/dependency_links.txt +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/entry_points.txt +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/not-zip-safe +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/requires.txt +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/src/bytesense.egg-info/top_level.txt +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_api.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_api_extra.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_candidate.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_cli.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_coherence.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_fingerprint.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_heuristics_extra.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_hints.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_hints_extra.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_legacy.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_mess.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_mess_extra.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_multi.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_repair.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_rust.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_rust_layer.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_streaming.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_streaming_v2.py +0 -0
- {bytesense-1.1.0 → bytesense-1.2.0}/tests/test_v1_contracts.py +0 -0
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.2.0
|
|
4
|
+
|
|
5
|
+
### UTF-8 validation
|
|
6
|
+
- Validate complete UTF-8 and UTF-8-SIG inputs in `from_bytes()` with SIMD in native builds, without allocating decoded Unicode. Pin `simdutf8` 0.1.5 and preserve strict rejection and first-error offsets with its compatibility validator.
|
|
7
|
+
- Retain runtime x86 CPU dispatch and scalar fallbacks; no target-specific CPU flags or Python runtime dependencies are required.
|
|
8
|
+
- Decode pure-Python UTF-8 inputs larger than 64 KiB in bounded blocks. Preserve the existing threshold for other codecs, including custom decoders.
|
|
9
|
+
- Reuse the ASCII check within each detection call. Public results, scoring and stream behavior are unchanged.
|
|
10
|
+
|
|
11
|
+
### Verification
|
|
12
|
+
- Add strict-decoder parity, malformed-scalar, SIMD/block boundary, allocation and custom-codec regressions.
|
|
13
|
+
- Extend rotating-input benchmarks with 64 KiB and 8 MiB UTF-8 workloads and isolated before/after comparisons.
|
|
14
|
+
- Preserve existing corpus accuracy and native/Python predictions; include the native dependency's license in distributions.
|
|
15
|
+
|
|
16
|
+
See the benchmark documentation for measured gains and their scope.
|
|
17
|
+
|
|
3
18
|
## 1.1.0
|
|
4
19
|
|
|
5
20
|
### Faster legacy detection
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bytesense
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
|
|
5
5
|
Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
6
6
|
Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
|
|
|
33
33
|
Requires-Python: >=3.9
|
|
34
34
|
Description-Content-Type: text/markdown
|
|
35
35
|
License-File: LICENSE
|
|
36
|
+
License-File: THIRD_PARTY_LICENSES
|
|
36
37
|
Provides-Extra: fast
|
|
37
38
|
Provides-Extra: docs
|
|
38
39
|
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
@@ -131,13 +132,21 @@ cat report.csv | bytesense --minimal -
|
|
|
131
132
|
|
|
132
133
|
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
133
134
|
|
|
135
|
+
## Used by
|
|
136
|
+
|
|
137
|
+
[ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
|
|
138
|
+
detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
|
|
139
|
+
safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
|
|
140
|
+
Built by the same maintainer; an independent project, not affiliated with Sentry.
|
|
141
|
+
|
|
134
142
|
## Measured accuracy and performance
|
|
135
143
|
|
|
136
144
|
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
137
145
|
|
|
138
146
|
| Detector | Exact Unicode matches | Accuracy |
|
|
139
147
|
|---|---:|---:|
|
|
140
|
-
| **bytesense 1.
|
|
148
|
+
| **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
|
|
149
|
+
| bytesense 1.1.0 | 1907 / 2078 | 91.77% |
|
|
141
150
|
| bytesense 1.0.0 | 1906 / 2078 | 91.72% |
|
|
142
151
|
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
143
152
|
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
@@ -146,9 +155,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
|
|
|
146
155
|
|
|
147
156
|
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
148
157
|
|
|
149
|
-
**New in 1.
|
|
158
|
+
**New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
|
|
150
159
|
|
|
151
|
-
The
|
|
160
|
+
The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
|
|
152
161
|
|
|
153
162
|
## Upgrading from 0.x
|
|
154
163
|
|
|
@@ -74,13 +74,21 @@ cat report.csv | bytesense --minimal -
|
|
|
74
74
|
|
|
75
75
|
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
76
76
|
|
|
77
|
+
## Used by
|
|
78
|
+
|
|
79
|
+
[ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
|
|
80
|
+
detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
|
|
81
|
+
safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
|
|
82
|
+
Built by the same maintainer; an independent project, not affiliated with Sentry.
|
|
83
|
+
|
|
77
84
|
## Measured accuracy and performance
|
|
78
85
|
|
|
79
86
|
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
80
87
|
|
|
81
88
|
| Detector | Exact Unicode matches | Accuracy |
|
|
82
89
|
|---|---:|---:|
|
|
83
|
-
| **bytesense 1.
|
|
90
|
+
| **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
|
|
91
|
+
| bytesense 1.1.0 | 1907 / 2078 | 91.77% |
|
|
84
92
|
| bytesense 1.0.0 | 1906 / 2078 | 91.72% |
|
|
85
93
|
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
86
94
|
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
@@ -89,9 +97,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
|
|
|
89
97
|
|
|
90
98
|
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
91
99
|
|
|
92
|
-
**New in 1.
|
|
100
|
+
**New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
|
|
93
101
|
|
|
94
|
-
The
|
|
102
|
+
The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
|
|
95
103
|
|
|
96
104
|
## Upgrading from 0.x
|
|
97
105
|
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
The optional native extension includes simdutf8 0.1.5.
|
|
2
|
+
Source: https://github.com/rusticstuff/simdutf8/tree/87ee8d9d20b849eae1974a82b248ee6fe7491bcc
|
|
3
|
+
Upstream license: MIT OR Apache-2.0; distributed under the MIT option below.
|
|
4
|
+
|
|
5
|
+
MIT License
|
|
6
|
+
|
|
7
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
8
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
9
|
+
in the Software without restriction, including without limitation the rights
|
|
10
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
11
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
12
|
+
furnished to do so, subject to the following conditions:
|
|
13
|
+
|
|
14
|
+
The above copyright notice and this permission notice shall be included in all
|
|
15
|
+
copies or substantial portions of the Software.
|
|
16
|
+
|
|
17
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
18
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
19
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
20
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
21
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
22
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
23
|
+
SOFTWARE.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "bytesense"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.2.0"
|
|
8
8
|
description = "Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -82,6 +82,9 @@ Issues = "https://github.com/oguzhankir/bytesense/issues"
|
|
|
82
82
|
Changelog = "https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md"
|
|
83
83
|
Documentation = "https://oguzhankir.github.io/bytesense/"
|
|
84
84
|
|
|
85
|
+
[tool.setuptools]
|
|
86
|
+
license-files = ["LICENSE", "THIRD_PARTY_LICENSES"]
|
|
87
|
+
|
|
85
88
|
[tool.setuptools.packages.find]
|
|
86
89
|
where = ["src"]
|
|
87
90
|
|
|
@@ -4,9 +4,10 @@ version = 4
|
|
|
4
4
|
|
|
5
5
|
[[package]]
|
|
6
6
|
name = "bytesense-core"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.2.0"
|
|
8
8
|
dependencies = [
|
|
9
9
|
"pyo3",
|
|
10
|
+
"simdutf8",
|
|
10
11
|
]
|
|
11
12
|
|
|
12
13
|
[[package]]
|
|
@@ -109,6 +110,12 @@ dependencies = [
|
|
|
109
110
|
"proc-macro2",
|
|
110
111
|
]
|
|
111
112
|
|
|
113
|
+
[[package]]
|
|
114
|
+
name = "simdutf8"
|
|
115
|
+
version = "0.1.5"
|
|
116
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
117
|
+
checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
|
|
118
|
+
|
|
112
119
|
[[package]]
|
|
113
120
|
name = "syn"
|
|
114
121
|
version = "2.0.117"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "bytesense-core"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.2.0"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
rust-version = "1.88"
|
|
6
6
|
|
|
@@ -10,6 +10,7 @@ crate-type = ["cdylib"]
|
|
|
10
10
|
|
|
11
11
|
[dependencies]
|
|
12
12
|
pyo3 = { version = "0.28.2", features = ["abi3-py39"] }
|
|
13
|
+
simdutf8 = "=0.1.5"
|
|
13
14
|
|
|
14
15
|
[profile.release]
|
|
15
16
|
opt-level = 3
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
/// Returns (is_valid, confidence_0_to_1).
|
|
2
2
|
pub fn utf8_check(data: &[u8]) -> (bool, f64) {
|
|
3
|
-
|
|
3
|
+
// The compat validator preserves the first-error offset and early exit.
|
|
4
|
+
// x86 CPU features are detected at runtime; unsupported targets use scalar validation.
|
|
5
|
+
match simdutf8::compat::from_utf8(data) {
|
|
4
6
|
Ok(_) => (true, 1.0),
|
|
5
7
|
Err(e) => {
|
|
6
8
|
let conf = if data.is_empty() {
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""Compare installed baseline/candidate packages in paired, isolated processes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import contextlib
|
|
7
|
+
import hashlib
|
|
8
|
+
import importlib.metadata
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import platform
|
|
12
|
+
import random
|
|
13
|
+
import statistics
|
|
14
|
+
import subprocess
|
|
15
|
+
import sys
|
|
16
|
+
import time
|
|
17
|
+
import tracemalloc
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
from benchmark_v1 import samples
|
|
22
|
+
from evaluate_corpus import engines
|
|
23
|
+
|
|
24
|
+
EXPECTED = {
|
|
25
|
+
"ascii_1mib": "ascii",
|
|
26
|
+
"utf8_1mib": "utf_8",
|
|
27
|
+
"utf8_8kib": "utf_8",
|
|
28
|
+
"utf8_64kib": "utf_8",
|
|
29
|
+
"utf8_8mib": "utf_8",
|
|
30
|
+
"turkish_cp1254_2100b": "cp1254",
|
|
31
|
+
"japanese_eucjp_4kib": "euc_jp",
|
|
32
|
+
}
|
|
33
|
+
COMPETITORS = ["chardet", "charset-normalizer", "chardetng-py"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def worker(variant: str, competitors: list[str]) -> None:
|
|
37
|
+
import bytesense.api
|
|
38
|
+
from bytesense._rust import is_rust_available
|
|
39
|
+
|
|
40
|
+
names = ["bytesense"] + (competitors if variant == "candidate" else [])
|
|
41
|
+
detectors = engines(names)
|
|
42
|
+
payloads = samples()
|
|
43
|
+
for inputs in payloads.values():
|
|
44
|
+
for data in inputs[:4]:
|
|
45
|
+
for detector in detectors.values():
|
|
46
|
+
detector(data)
|
|
47
|
+
metadata = {
|
|
48
|
+
"ready": True,
|
|
49
|
+
"native": is_rust_available(),
|
|
50
|
+
"python": platform.python_version(),
|
|
51
|
+
"versions": {name: importlib.metadata.version(name) for name in names},
|
|
52
|
+
"api_sha256": hashlib.sha256(Path(bytesense.api.__file__).read_bytes()).hexdigest(),
|
|
53
|
+
}
|
|
54
|
+
if is_rust_available():
|
|
55
|
+
import bytesense._rust_core
|
|
56
|
+
|
|
57
|
+
metadata["native_extension_sha256"] = hashlib.sha256(
|
|
58
|
+
Path(bytesense._rust_core.__file__).read_bytes()
|
|
59
|
+
).hexdigest()
|
|
60
|
+
print(json.dumps(metadata), flush=True)
|
|
61
|
+
for line in sys.stdin:
|
|
62
|
+
request = json.loads(line)
|
|
63
|
+
case, engine = request["case"], request["engine"]
|
|
64
|
+
data = payloads[case][request["index"]]
|
|
65
|
+
|
|
66
|
+
def call(
|
|
67
|
+
payload: bytes = data,
|
|
68
|
+
detect=detectors[engine],
|
|
69
|
+
validate: bool = request["mode"] == "full_validation",
|
|
70
|
+
) -> str | None:
|
|
71
|
+
encoding = detect(payload)
|
|
72
|
+
if validate and encoding:
|
|
73
|
+
payload.decode(encoding, errors="strict")
|
|
74
|
+
return encoding
|
|
75
|
+
|
|
76
|
+
if request.get("allocation"):
|
|
77
|
+
tracemalloc.start()
|
|
78
|
+
try:
|
|
79
|
+
call()
|
|
80
|
+
_, peak = tracemalloc.get_traced_memory()
|
|
81
|
+
finally:
|
|
82
|
+
tracemalloc.stop()
|
|
83
|
+
result = {"peak_bytes": peak}
|
|
84
|
+
else:
|
|
85
|
+
start = time.perf_counter_ns()
|
|
86
|
+
encoding = call()
|
|
87
|
+
elapsed = (time.perf_counter_ns() - start) / 1e6
|
|
88
|
+
try:
|
|
89
|
+
correct = bool(encoding and data.decode(encoding) == data.decode(EXPECTED[case]))
|
|
90
|
+
except UnicodeError:
|
|
91
|
+
correct = False
|
|
92
|
+
result = {"ms": elapsed, "correct": correct}
|
|
93
|
+
print(json.dumps(result), flush=True)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class Worker:
|
|
97
|
+
def __init__(self, python: Path, variant: str, competitors: list[str]):
|
|
98
|
+
self.variant = variant
|
|
99
|
+
environment = dict(os.environ)
|
|
100
|
+
# Each supplied interpreter must resolve its own installed package.
|
|
101
|
+
environment.pop("PYTHONPATH", None)
|
|
102
|
+
# Keep virtualenv interpreter symlinks intact: resolving one loses its
|
|
103
|
+
# environment and can import a different installed package.
|
|
104
|
+
arguments = [str(python.absolute()), "-u", str(Path(__file__).resolve()), "--worker", variant]
|
|
105
|
+
for competitor in competitors:
|
|
106
|
+
arguments.extend(["--engine", competitor])
|
|
107
|
+
self.process = subprocess.Popen(
|
|
108
|
+
arguments,
|
|
109
|
+
stdin=subprocess.PIPE,
|
|
110
|
+
stdout=subprocess.PIPE,
|
|
111
|
+
text=True,
|
|
112
|
+
env=environment,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
def read(self) -> dict[str, Any]:
|
|
116
|
+
assert self.process.stdout is not None
|
|
117
|
+
line = self.process.stdout.readline()
|
|
118
|
+
if not line:
|
|
119
|
+
raise RuntimeError(
|
|
120
|
+
f"{self.variant} worker closed its output; inspect its stderr above "
|
|
121
|
+
f"(exit status: {self.process.poll()})."
|
|
122
|
+
)
|
|
123
|
+
return json.loads(line)
|
|
124
|
+
|
|
125
|
+
def request(self, **payload: Any) -> dict[str, Any]:
|
|
126
|
+
assert self.process.stdin is not None
|
|
127
|
+
self.process.stdin.write(json.dumps(payload) + "\n")
|
|
128
|
+
self.process.stdin.flush()
|
|
129
|
+
return self.read()
|
|
130
|
+
|
|
131
|
+
def close(self) -> None:
|
|
132
|
+
if self.process.stdin is not None:
|
|
133
|
+
with contextlib.suppress(BrokenPipeError):
|
|
134
|
+
self.process.stdin.close()
|
|
135
|
+
try:
|
|
136
|
+
self.process.wait(timeout=5)
|
|
137
|
+
except subprocess.TimeoutExpired:
|
|
138
|
+
self.process.terminate()
|
|
139
|
+
try:
|
|
140
|
+
self.process.wait(timeout=5)
|
|
141
|
+
except subprocess.TimeoutExpired:
|
|
142
|
+
self.process.kill()
|
|
143
|
+
self.process.wait()
|
|
144
|
+
if self.process.stdout is not None:
|
|
145
|
+
self.process.stdout.close()
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def main() -> None:
|
|
149
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
150
|
+
parser.add_argument("--baseline-python", type=Path)
|
|
151
|
+
parser.add_argument("--candidate-python", type=Path)
|
|
152
|
+
parser.add_argument("--output", type=Path)
|
|
153
|
+
parser.add_argument("--rounds", type=int, default=128)
|
|
154
|
+
parser.add_argument("--engine", action="append", choices=COMPETITORS)
|
|
155
|
+
parser.add_argument("--worker", choices=["baseline", "candidate"], help=argparse.SUPPRESS)
|
|
156
|
+
args = parser.parse_args()
|
|
157
|
+
competitors = list(dict.fromkeys(args.engine or COMPETITORS))
|
|
158
|
+
if args.worker:
|
|
159
|
+
worker(args.worker, competitors)
|
|
160
|
+
return
|
|
161
|
+
if not args.baseline_python or not args.candidate_python or not args.output:
|
|
162
|
+
parser.error("--baseline-python, --candidate-python and --output are required")
|
|
163
|
+
if args.rounds < 1:
|
|
164
|
+
parser.error("--rounds must be positive")
|
|
165
|
+
processes: dict[str, Worker] = {}
|
|
166
|
+
metadata = {}
|
|
167
|
+
try:
|
|
168
|
+
for variant, python in [
|
|
169
|
+
("baseline", args.baseline_python), ("candidate", args.candidate_python)
|
|
170
|
+
]:
|
|
171
|
+
processes[variant] = Worker(python, variant, competitors)
|
|
172
|
+
metadata[variant] = processes[variant].read()
|
|
173
|
+
if not metadata[variant].get("ready"):
|
|
174
|
+
raise RuntimeError(f"{variant} worker did not initialize")
|
|
175
|
+
print(json.dumps(metadata), flush=True)
|
|
176
|
+
detectors = [("baseline", "bytesense"), ("candidate", "bytesense")]
|
|
177
|
+
detectors += [("candidate", name) for name in competitors]
|
|
178
|
+
rng = random.Random(20260919)
|
|
179
|
+
results = []
|
|
180
|
+
for case, payloads in samples().items():
|
|
181
|
+
for mode in ["defaults", "full_validation"]:
|
|
182
|
+
times: dict[tuple[str, str], list[float]] = {key: [] for key in detectors}
|
|
183
|
+
correct = dict.fromkeys(detectors, 0)
|
|
184
|
+
for i in range(args.rounds):
|
|
185
|
+
order = list(detectors)
|
|
186
|
+
rng.shuffle(order)
|
|
187
|
+
for variant, engine in order:
|
|
188
|
+
result = processes[variant].request(
|
|
189
|
+
case=case, engine=engine, index=i % len(payloads), mode=mode
|
|
190
|
+
)
|
|
191
|
+
times[(variant, engine)].append(result["ms"])
|
|
192
|
+
correct[(variant, engine)] += result["correct"]
|
|
193
|
+
for variant, engine in detectors:
|
|
194
|
+
durations = times[(variant, engine)]
|
|
195
|
+
allocation = processes[variant].request(
|
|
196
|
+
case=case, engine=engine, index=0, mode=mode, allocation=True
|
|
197
|
+
)
|
|
198
|
+
result = {
|
|
199
|
+
"case": case,
|
|
200
|
+
"mode": mode,
|
|
201
|
+
"variant": variant,
|
|
202
|
+
"engine": engine,
|
|
203
|
+
"bytes": len(payloads[0]),
|
|
204
|
+
"sha256": hashlib.sha256(b"".join(payloads)).hexdigest(),
|
|
205
|
+
"median_ms": statistics.median(durations),
|
|
206
|
+
"p95_ms": sorted(durations)[int(0.95 * (args.rounds - 1))],
|
|
207
|
+
"peak_bytes": allocation["peak_bytes"],
|
|
208
|
+
"correct": correct[(variant, engine)],
|
|
209
|
+
"rounds": args.rounds,
|
|
210
|
+
}
|
|
211
|
+
results.append(result)
|
|
212
|
+
print(json.dumps(result), flush=True)
|
|
213
|
+
args.output.write_text(
|
|
214
|
+
json.dumps(
|
|
215
|
+
{
|
|
216
|
+
"python": platform.python_version(),
|
|
217
|
+
"platform": platform.platform(),
|
|
218
|
+
"metadata": metadata,
|
|
219
|
+
"notes": "32 rotating payloads; four warmups per case; persistent isolated "
|
|
220
|
+
"baseline/candidate processes; seed 20260919 randomized engine order each "
|
|
221
|
+
"round; IPC excluded; tracemalloc excludes native allocations.",
|
|
222
|
+
"results": results,
|
|
223
|
+
},
|
|
224
|
+
indent=2,
|
|
225
|
+
),
|
|
226
|
+
encoding="utf-8",
|
|
227
|
+
)
|
|
228
|
+
finally:
|
|
229
|
+
for process in processes.values():
|
|
230
|
+
process.close()
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
if __name__ == "__main__":
|
|
234
|
+
main()
|
|
@@ -21,6 +21,8 @@ def samples() -> dict[str, list[bytes]]:
|
|
|
21
21
|
"ascii_1mib": (b"The quick brown fox jumps over the lazy dog. ", 1_048_576, "ascii"),
|
|
22
22
|
"utf8_1mib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 1_048_576, "utf_8"),
|
|
23
23
|
"utf8_8kib": ("Café dünyası: 中文测试 🎉 ".encode(), 8192, "utf_8"),
|
|
24
|
+
"utf8_64kib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 65536, "utf_8"),
|
|
25
|
+
"utf8_8mib": ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode(), 8_388_608, "utf_8"),
|
|
24
26
|
"turkish_cp1254_2100b": (
|
|
25
27
|
("İstanbul'da çalışan mühendisler için doğru metin çözümleme. " * 35).encode("cp1254"),
|
|
26
28
|
2100,
|
|
@@ -65,6 +67,8 @@ def main() -> None:
|
|
|
65
67
|
"ascii_1mib": "ascii",
|
|
66
68
|
"utf8_1mib": "utf_8",
|
|
67
69
|
"utf8_8kib": "utf_8",
|
|
70
|
+
"utf8_64kib": "utf_8",
|
|
71
|
+
"utf8_8mib": "utf_8",
|
|
68
72
|
"turkish_cp1254_2100b": "cp1254",
|
|
69
73
|
"japanese_eucjp_4kib": "euc_jp",
|
|
70
74
|
}
|
|
@@ -13,12 +13,14 @@ for path in sorted(Path(sys.argv[1]).iterdir()):
|
|
|
13
13
|
assert not any(n.endswith((".so", ".pyd", ".dll")) for n in names)
|
|
14
14
|
assert "bytesense/py.typed" in names
|
|
15
15
|
assert "bytesense/data/language.json.gz" in names
|
|
16
|
+
assert any(n.endswith("/THIRD_PARTY_LICENSES") for n in names)
|
|
16
17
|
assert not any("/target/" in n for n in names)
|
|
17
18
|
elif path.name.endswith(".tar.gz"):
|
|
18
19
|
with tarfile.open(path) as archive:
|
|
19
20
|
names = archive.getnames()
|
|
20
21
|
for required in (
|
|
21
22
|
"setup.py",
|
|
23
|
+
"THIRD_PARTY_LICENSES",
|
|
22
24
|
"scripts/evaluate_corpus.py",
|
|
23
25
|
"scripts/evaluate_udhr.py",
|
|
24
26
|
"scripts/check_evaluation.py",
|
|
@@ -10,6 +10,7 @@ from functools import lru_cache
|
|
|
10
10
|
from os import PathLike
|
|
11
11
|
from typing import Any, BinaryIO, List, Optional
|
|
12
12
|
|
|
13
|
+
from ._rust import is_rust_available, rust_utf8_check
|
|
13
14
|
from .candidate import CandidateSelector
|
|
14
15
|
from .coherence import detect_language
|
|
15
16
|
from .constant import ALL_ENCODINGS
|
|
@@ -236,8 +237,13 @@ def _unicode_fallbacks(
|
|
|
236
237
|
|
|
237
238
|
|
|
238
239
|
def _valid(data: bytes, encoding: str) -> bool:
|
|
240
|
+
utf8 = encoding in ("utf_8", "utf_8_sig")
|
|
241
|
+
if utf8 and is_rust_available():
|
|
242
|
+
# A UTF-8 BOM changes decoded text, not strict byte validity. Native
|
|
243
|
+
# validation checks every byte without allocating decoded Unicode.
|
|
244
|
+
return rust_utf8_check(data)[0]
|
|
239
245
|
try:
|
|
240
|
-
if len(data) <= 1_048_576:
|
|
246
|
+
if len(data) <= (65536 if utf8 else 1_048_576):
|
|
241
247
|
data.decode(encoding, errors="strict")
|
|
242
248
|
else:
|
|
243
249
|
decoder = codecs.getincrementaldecoder(encoding)(errors="strict")
|
|
@@ -375,7 +381,8 @@ def from_bytes(
|
|
|
375
381
|
)
|
|
376
382
|
r = _result(bom, size, f"BOM declares {bom}; all bytes validated.", 1.0)
|
|
377
383
|
return replace(r, bom_detected=True)
|
|
378
|
-
|
|
384
|
+
ascii_data = data.isascii()
|
|
385
|
+
transport = _transport_encoding(data) if ascii_data else None
|
|
379
386
|
if transport and _allowed(transport, include, exclude):
|
|
380
387
|
if min_confidence <= 0.85:
|
|
381
388
|
return _result(
|
|
@@ -412,7 +419,7 @@ def from_bytes(
|
|
|
412
419
|
)
|
|
413
420
|
shape = detect_null_pattern(data)
|
|
414
421
|
if not shape and not _looks_like_iso2022(data):
|
|
415
|
-
if
|
|
422
|
+
if ascii_data and _allowed("ascii", include, exclude):
|
|
416
423
|
return _result("ascii", size, "All bytes are valid ASCII text.", 1.0)
|
|
417
424
|
if _allowed("utf_8", include, exclude) and _valid(data, "utf_8"):
|
|
418
425
|
r = _result("utf_8", size, "All bytes validated as UTF-8.", 0.99)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bytesense
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
|
|
5
5
|
Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
6
6
|
Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
|
|
|
33
33
|
Requires-Python: >=3.9
|
|
34
34
|
Description-Content-Type: text/markdown
|
|
35
35
|
License-File: LICENSE
|
|
36
|
+
License-File: THIRD_PARTY_LICENSES
|
|
36
37
|
Provides-Extra: fast
|
|
37
38
|
Provides-Extra: docs
|
|
38
39
|
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
@@ -131,13 +132,21 @@ cat report.csv | bytesense --minimal -
|
|
|
131
132
|
|
|
132
133
|
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
133
134
|
|
|
135
|
+
## Used by
|
|
136
|
+
|
|
137
|
+
[ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
|
|
138
|
+
detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
|
|
139
|
+
safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
|
|
140
|
+
Built by the same maintainer; an independent project, not affiliated with Sentry.
|
|
141
|
+
|
|
134
142
|
## Measured accuracy and performance
|
|
135
143
|
|
|
136
144
|
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
137
145
|
|
|
138
146
|
| Detector | Exact Unicode matches | Accuracy |
|
|
139
147
|
|---|---:|---:|
|
|
140
|
-
| **bytesense 1.
|
|
148
|
+
| **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
|
|
149
|
+
| bytesense 1.1.0 | 1907 / 2078 | 91.77% |
|
|
141
150
|
| bytesense 1.0.0 | 1906 / 2078 | 91.72% |
|
|
142
151
|
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
143
152
|
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
@@ -146,9 +155,9 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
|
|
|
146
155
|
|
|
147
156
|
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
148
157
|
|
|
149
|
-
**New in 1.
|
|
158
|
+
**New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
|
|
150
159
|
|
|
151
|
-
The
|
|
160
|
+
The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
|
|
152
161
|
|
|
153
162
|
## Upgrading from 0.x
|
|
154
163
|
|
|
@@ -4,6 +4,7 @@ CONTRIBUTING.md
|
|
|
4
4
|
LICENSE
|
|
5
5
|
MANIFEST.in
|
|
6
6
|
README.md
|
|
7
|
+
THIRD_PARTY_LICENSES
|
|
7
8
|
pyproject.toml
|
|
8
9
|
setup.py
|
|
9
10
|
benchmarks/__init__.py
|
|
@@ -18,6 +19,7 @@ rust/src/histogram.rs
|
|
|
18
19
|
rust/src/language.rs
|
|
19
20
|
rust/src/lib.rs
|
|
20
21
|
rust/src/utf8.rs
|
|
22
|
+
scripts/benchmark_compare.py
|
|
21
23
|
scripts/benchmark_v1.py
|
|
22
24
|
scripts/build_fingerprints.py
|
|
23
25
|
scripts/check_dist.py
|
|
@@ -75,4 +77,5 @@ tests/test_rust.py
|
|
|
75
77
|
tests/test_rust_layer.py
|
|
76
78
|
tests/test_streaming.py
|
|
77
79
|
tests/test_streaming_v2.py
|
|
80
|
+
tests/test_utf8_validation.py
|
|
78
81
|
tests/test_v1_contracts.py
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Strict UTF-8 validation across SIMD, BOM, and incremental decoder boundaries."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import codecs
|
|
6
|
+
import platform
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
from hypothesis import given, settings
|
|
10
|
+
from hypothesis import strategies as st
|
|
11
|
+
|
|
12
|
+
from bytesense import api, from_bytes
|
|
13
|
+
from bytesense._rust import is_rust_available, rust_utf8_check
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _strictly_valid(data: bytes, encoding: str) -> bool:
|
|
17
|
+
try:
|
|
18
|
+
data.decode(encoding, errors="strict")
|
|
19
|
+
except UnicodeError:
|
|
20
|
+
return False
|
|
21
|
+
return True
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@pytest.mark.parametrize("encoding", ["utf_8", "utf_8_sig"])
|
|
25
|
+
@given(data=st.one_of(st.binary(max_size=4096), st.text(max_size=4096).map(str.encode)))
|
|
26
|
+
@settings(max_examples=300, deadline=None)
|
|
27
|
+
def test_utf8_validation_matches_strict_python(data: bytes, encoding: str) -> None:
|
|
28
|
+
assert api._valid(data, encoding) == _strictly_valid(data, encoding)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@pytest.mark.parametrize("encoding", ["utf_8", "utf_8_sig"])
|
|
32
|
+
@pytest.mark.parametrize("offset", [31, 32, 63, 64, 65, 65535, 65536, 65537, 1_048_575])
|
|
33
|
+
@pytest.mark.parametrize(
|
|
34
|
+
"suffix",
|
|
35
|
+
[
|
|
36
|
+
"é中🙂".encode(),
|
|
37
|
+
b"\xc0\xaf", # Overlong sequence.
|
|
38
|
+
b"\xed\xa0\x80", # Surrogate scalar.
|
|
39
|
+
b"\xf4\x90\x80\x80", # Above U+10FFFF.
|
|
40
|
+
b"\xe2\x82", # Truncated code point.
|
|
41
|
+
b"\x80", # Unpaired continuation byte.
|
|
42
|
+
],
|
|
43
|
+
)
|
|
44
|
+
def test_utf8_boundaries_preserve_full_input_validation(
|
|
45
|
+
encoding: str, offset: int, suffix: bytes
|
|
46
|
+
) -> None:
|
|
47
|
+
prefix = codecs.BOM_UTF8 if encoding == "utf_8_sig" else b""
|
|
48
|
+
data = prefix + b"a" * offset + suffix
|
|
49
|
+
expected = _strictly_valid(data, encoding)
|
|
50
|
+
assert api._valid(data, encoding) == expected
|
|
51
|
+
result = from_bytes(data, cp_isolation=[encoding], enable_fallback=False)
|
|
52
|
+
assert (result.encoding is not None) == expected
|
|
53
|
+
if expected:
|
|
54
|
+
assert result.bytes_validated == len(data)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@pytest.mark.skipif(not is_rust_available(), reason="Native extension not compiled")
|
|
58
|
+
@given(data=st.binary(max_size=8192))
|
|
59
|
+
@settings(max_examples=500, deadline=None)
|
|
60
|
+
def test_native_error_offset_matches_python(data: bytes) -> None:
|
|
61
|
+
try:
|
|
62
|
+
data.decode("utf_8", errors="strict")
|
|
63
|
+
expected = (True, 1.0)
|
|
64
|
+
except UnicodeDecodeError as error:
|
|
65
|
+
expected = (False, error.start / len(data))
|
|
66
|
+
assert rust_utf8_check(data) == expected
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@pytest.mark.skipif(platform.python_implementation() != "CPython", reason="CPython allocation probe")
|
|
70
|
+
@pytest.mark.parametrize("native", [False, True])
|
|
71
|
+
def test_utf8_validation_does_not_allocate_full_decoded_input(monkeypatch, native: bool) -> None:
|
|
72
|
+
import tracemalloc
|
|
73
|
+
|
|
74
|
+
if native and not is_rust_available():
|
|
75
|
+
pytest.skip("Native extension not compiled")
|
|
76
|
+
if not native:
|
|
77
|
+
monkeypatch.setattr(api, "is_rust_available", lambda: False)
|
|
78
|
+
data = ("Merhaba dünya! 日本語 Ελληνικά 🙂 ".encode() * 30_000)[:1_048_576]
|
|
79
|
+
data = data.decode("utf_8", errors="ignore").encode("utf_8")
|
|
80
|
+
# Warm the decoder and feature dispatch before measuring call allocations.
|
|
81
|
+
assert api._valid(data, "utf_8")
|
|
82
|
+
tracemalloc.start()
|
|
83
|
+
try:
|
|
84
|
+
assert api._valid(data, "utf_8")
|
|
85
|
+
_, peak = tracemalloc.get_traced_memory()
|
|
86
|
+
finally:
|
|
87
|
+
tracemalloc.stop()
|
|
88
|
+
assert peak < 600_000
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def test_non_utf8_custom_codec_keeps_existing_one_shot_boundary() -> None:
|
|
92
|
+
class BytesOnlyDecoder(codecs.IncrementalDecoder):
|
|
93
|
+
def decode(self, data: bytes, final: bool = False) -> str:
|
|
94
|
+
if not isinstance(data, bytes):
|
|
95
|
+
raise TypeError("decoder requires bytes")
|
|
96
|
+
return data.decode("ascii")
|
|
97
|
+
|
|
98
|
+
def search(name: str) -> codecs.CodecInfo | None:
|
|
99
|
+
if name != "bytesense_test_bytes_only":
|
|
100
|
+
return None
|
|
101
|
+
return codecs.CodecInfo(
|
|
102
|
+
name=name,
|
|
103
|
+
encode=codecs.ascii_encode,
|
|
104
|
+
decode=codecs.ascii_decode,
|
|
105
|
+
incrementaldecoder=BytesOnlyDecoder,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
codecs.register(search)
|
|
109
|
+
assert api._valid(b"a" * 65537, "bytesense_test_bytes_only")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|