bytesense 1.0.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bytesense-1.0.0 → bytesense-1.2.0}/CHANGELOG.md +36 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/MANIFEST.in +1 -1
- {bytesense-1.0.0 → bytesense-1.2.0}/PKG-INFO +17 -3
- {bytesense-1.0.0 → bytesense-1.2.0}/README.md +15 -2
- bytesense-1.2.0/THIRD_PARTY_LICENSES +23 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/pyproject.toml +4 -1
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/Cargo.lock +8 -1
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/Cargo.toml +2 -1
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/language.rs +95 -34
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/utf8.rs +3 -1
- bytesense-1.2.0/scripts/benchmark_compare.py +234 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/benchmark_v1.py +12 -1
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/check_dist.py +4 -0
- bytesense-1.2.0/scripts/check_evaluation.py +41 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/check_version.py +2 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/evaluate_corpus.py +1 -0
- bytesense-1.2.0/scripts/evaluate_udhr.py +172 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/api.py +85 -13
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/scoring.py +29 -5
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/streaming.py +24 -2
- bytesense-1.2.0/src/bytesense/version.py +4 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/PKG-INFO +17 -3
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/SOURCES.txt +5 -0
- bytesense-1.2.0/tests/test_rust.py +129 -0
- bytesense-1.2.0/tests/test_utf8_validation.py +109 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_v1_contracts.py +75 -0
- bytesense-1.0.0/src/bytesense/version.py +0 -4
- bytesense-1.0.0/tests/test_rust.py +0 -54
- {bytesense-1.0.0 → bytesense-1.2.0}/CODE_OF_CONDUCT.md +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/CONTRIBUTING.md +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/LICENSE +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/__init__.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/cn_official_manifest.json +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/conftest.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/hard_scenarios.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/test_bench_detection.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/test_hard_scenarios.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/histogram.rs +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/lib.rs +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/build_fingerprints.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/compare_libraries.sh +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/fetch_cn_benchmark_samples.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/scripts/run_all_benchmarks.sh +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/setup.cfg +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/setup.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/__init__.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/_rust.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/candidate.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/cli.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/coherence.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/constant.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/__init__.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/fingerprints.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/language.json.gz +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/fingerprint.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/heuristics.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/hints.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/legacy.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/mess.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/models.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/multi.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/py.typed +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/repair.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/dependency_links.txt +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/entry_points.txt +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/not-zip-safe +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/requires.txt +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/top_level.txt +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_api.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_api_extra.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_candidate.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_cli.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_coherence.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_fingerprint.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_heuristics_extra.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_hints.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_hints_extra.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_legacy.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_mess.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_mess_extra.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_multi.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_repair.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_rust_layer.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_streaming.py +0 -0
- {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_streaming_v2.py +0 -0
|
@@ -1,5 +1,41 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.2.0
|
|
4
|
+
|
|
5
|
+
### UTF-8 validation
|
|
6
|
+
- Validate complete UTF-8 and UTF-8-SIG inputs in `from_bytes()` with SIMD in native builds, without allocating decoded Unicode. Pin `simdutf8` 0.1.5 and preserve strict rejection and first-error offsets with its compatibility validator.
|
|
7
|
+
- Retain runtime x86 CPU dispatch and scalar fallbacks; no target-specific CPU flags or Python runtime dependencies are required.
|
|
8
|
+
- Decode pure-Python UTF-8 inputs larger than 64 KiB in bounded blocks. Preserve the existing threshold for other codecs, including custom decoders.
|
|
9
|
+
- Reuse the ASCII check within each detection call. Public results, scoring and stream behavior are unchanged.
|
|
10
|
+
|
|
11
|
+
### Verification
|
|
12
|
+
- Add strict-decoder parity, malformed-scalar, SIMD/block boundary, allocation and custom-codec regressions.
|
|
13
|
+
- Extend rotating-input benchmarks with 64 KiB and 8 MiB UTF-8 workloads and isolated before/after comparisons.
|
|
14
|
+
- Preserve existing corpus accuracy and native/Python predictions; include the native dependency's license in distributions.
|
|
15
|
+
|
|
16
|
+
See the benchmark documentation for measured gains and their scope.
|
|
17
|
+
|
|
18
|
+
## 1.1.0
|
|
19
|
+
|
|
20
|
+
### Faster legacy detection
|
|
21
|
+
- Cache Unicode scalar properties in a fixed 64 KiB atomic table; remove per-character property hash lookups and read locks.
|
|
22
|
+
- Use a bounded 512 KiB native lookup table for Latin/ASCII model pairs, with sparse lookup for other scripts.
|
|
23
|
+
- Prune pair scoring only when an optimistic evidence bound proves that a candidate cannot qualify. Retain matching Python/native decisions.
|
|
24
|
+
- Normalize and deduplicate the built-in codec catalog lazily once; caller-supplied codecs are still validated on each call.
|
|
25
|
+
|
|
26
|
+
### Correctness
|
|
27
|
+
- Normalize scoring input to NFC and stop penalizing combining marks as symbol noise. Original bytes and decoded text are never normalized or rewritten.
|
|
28
|
+
- Reach all statistically eligible candidates during full-input validation instead of stopping at the six display hypotheses. Public alternatives remain limited to five.
|
|
29
|
+
- Recover plausible BOM-less UTF-16 with NULs in both lanes before declaring binary content, subject to strict full-input validation and caller filters.
|
|
30
|
+
- Recognize ISO-2022 shift controls and UTF-7 Unicode spacing/format characters as text syntax.
|
|
31
|
+
|
|
32
|
+
### Verification
|
|
33
|
+
- Add cache-boundary, concurrent cold-cache, dense/sparse pair parity, pruning-bound, canonical-equivalence and validation-depth regressions.
|
|
34
|
+
- Add a pinned 28-document UDHR transfer evaluation, with exact Unicode comparisons, recorded exclusions and separate legacy/Unicode counts.
|
|
35
|
+
- Add single-engine benchmark runs for before/after comparisons. Model weights and public API signatures are unchanged.
|
|
36
|
+
|
|
37
|
+
See the benchmark documentation for measured gains and remaining limitations.
|
|
38
|
+
|
|
3
39
|
|
|
4
40
|
## 1.0.0
|
|
5
41
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bytesense
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
|
|
5
5
|
Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
6
6
|
Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
|
|
|
33
33
|
Requires-Python: >=3.9
|
|
34
34
|
Description-Content-Type: text/markdown
|
|
35
35
|
License-File: LICENSE
|
|
36
|
+
License-File: THIRD_PARTY_LICENSES
|
|
36
37
|
Provides-Extra: fast
|
|
37
38
|
Provides-Extra: docs
|
|
38
39
|
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
@@ -67,7 +68,7 @@ Dynamic: license-file
|
|
|
67
68
|
|
|
68
69
|
**bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
|
|
69
70
|
|
|
70
|
-
|
|
71
|
+
Final detection results strictly validate the selected encoding against the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
|
|
71
72
|
|
|
72
73
|
```python
|
|
73
74
|
from bytesense import from_bytes
|
|
@@ -131,13 +132,22 @@ cat report.csv | bytesense --minimal -
|
|
|
131
132
|
|
|
132
133
|
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
133
134
|
|
|
135
|
+
## Used by
|
|
136
|
+
|
|
137
|
+
[ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
|
|
138
|
+
detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
|
|
139
|
+
safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
|
|
140
|
+
Built by the same maintainer; an independent project, not affiliated with Sentry.
|
|
141
|
+
|
|
134
142
|
## Measured accuracy and performance
|
|
135
143
|
|
|
136
144
|
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
137
145
|
|
|
138
146
|
| Detector | Exact Unicode matches | Accuracy |
|
|
139
147
|
|---|---:|---:|
|
|
140
|
-
| **bytesense 1.
|
|
148
|
+
| **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
|
|
149
|
+
| bytesense 1.1.0 | 1907 / 2078 | 91.77% |
|
|
150
|
+
| bytesense 1.0.0 | 1906 / 2078 | 91.72% |
|
|
141
151
|
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
142
152
|
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
143
153
|
| charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
|
|
@@ -145,6 +155,10 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
|
|
|
145
155
|
|
|
146
156
|
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
147
157
|
|
|
158
|
+
**New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
|
|
159
|
+
|
|
160
|
+
The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
|
|
161
|
+
|
|
148
162
|
## Upgrading from 0.x
|
|
149
163
|
|
|
150
164
|
- Language reporting is opt-in: `include_language=True`.
|
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
|
|
11
11
|
**bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
|
|
12
12
|
|
|
13
|
-
|
|
13
|
+
Final detection results strictly validate the selected encoding against the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
|
|
14
14
|
|
|
15
15
|
```python
|
|
16
16
|
from bytesense import from_bytes
|
|
@@ -74,13 +74,22 @@ cat report.csv | bytesense --minimal -
|
|
|
74
74
|
|
|
75
75
|
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
76
76
|
|
|
77
|
+
## Used by
|
|
78
|
+
|
|
79
|
+
[ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
|
|
80
|
+
detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
|
|
81
|
+
safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
|
|
82
|
+
Built by the same maintainer; an independent project, not affiliated with Sentry.
|
|
83
|
+
|
|
77
84
|
## Measured accuracy and performance
|
|
78
85
|
|
|
79
86
|
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
80
87
|
|
|
81
88
|
| Detector | Exact Unicode matches | Accuracy |
|
|
82
89
|
|---|---:|---:|
|
|
83
|
-
| **bytesense 1.
|
|
90
|
+
| **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
|
|
91
|
+
| bytesense 1.1.0 | 1907 / 2078 | 91.77% |
|
|
92
|
+
| bytesense 1.0.0 | 1906 / 2078 | 91.72% |
|
|
84
93
|
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
85
94
|
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
86
95
|
| charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
|
|
@@ -88,6 +97,10 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
|
|
|
88
97
|
|
|
89
98
|
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
90
99
|
|
|
100
|
+
**New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
|
|
101
|
+
|
|
102
|
+
The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
|
|
103
|
+
|
|
91
104
|
## Upgrading from 0.x
|
|
92
105
|
|
|
93
106
|
- Language reporting is opt-in: `include_language=True`.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
The optional native extension includes simdutf8 0.1.5.
|
|
2
|
+
Source: https://github.com/rusticstuff/simdutf8/tree/87ee8d9d20b849eae1974a82b248ee6fe7491bcc
|
|
3
|
+
Upstream license: MIT OR Apache-2.0; distributed under the MIT option below.
|
|
4
|
+
|
|
5
|
+
MIT License
|
|
6
|
+
|
|
7
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
8
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
9
|
+
in the Software without restriction, including without limitation the rights
|
|
10
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
11
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
12
|
+
furnished to do so, subject to the following conditions:
|
|
13
|
+
|
|
14
|
+
The above copyright notice and this permission notice shall be included in all
|
|
15
|
+
copies or substantial portions of the Software.
|
|
16
|
+
|
|
17
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
18
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
19
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
20
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
21
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
22
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
23
|
+
SOFTWARE.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "bytesense"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.2.0"
|
|
8
8
|
description = "Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -82,6 +82,9 @@ Issues = "https://github.com/oguzhankir/bytesense/issues"
|
|
|
82
82
|
Changelog = "https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md"
|
|
83
83
|
Documentation = "https://oguzhankir.github.io/bytesense/"
|
|
84
84
|
|
|
85
|
+
[tool.setuptools]
|
|
86
|
+
license-files = ["LICENSE", "THIRD_PARTY_LICENSES"]
|
|
87
|
+
|
|
85
88
|
[tool.setuptools.packages.find]
|
|
86
89
|
where = ["src"]
|
|
87
90
|
|
|
@@ -4,9 +4,10 @@ version = 4
|
|
|
4
4
|
|
|
5
5
|
[[package]]
|
|
6
6
|
name = "bytesense-core"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.2.0"
|
|
8
8
|
dependencies = [
|
|
9
9
|
"pyo3",
|
|
10
|
+
"simdutf8",
|
|
10
11
|
]
|
|
11
12
|
|
|
12
13
|
[[package]]
|
|
@@ -109,6 +110,12 @@ dependencies = [
|
|
|
109
110
|
"proc-macro2",
|
|
110
111
|
]
|
|
111
112
|
|
|
113
|
+
[[package]]
|
|
114
|
+
name = "simdutf8"
|
|
115
|
+
version = "0.1.5"
|
|
116
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
117
|
+
checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
|
|
118
|
+
|
|
112
119
|
[[package]]
|
|
113
120
|
name = "syn"
|
|
114
121
|
version = "2.0.117"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "bytesense-core"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.2.0"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
rust-version = "1.88"
|
|
6
6
|
|
|
@@ -10,6 +10,7 @@ crate-type = ["cdylib"]
|
|
|
10
10
|
|
|
11
11
|
[dependencies]
|
|
12
12
|
pyo3 = { version = "0.28.2", features = ["abi3-py39"] }
|
|
13
|
+
simdutf8 = "=0.1.5"
|
|
13
14
|
|
|
14
15
|
[profile.release]
|
|
15
16
|
opt-level = 3
|
|
@@ -1,47 +1,93 @@
|
|
|
1
1
|
use pyo3::prelude::*;
|
|
2
2
|
use std::collections::{HashMap, HashSet};
|
|
3
|
-
use std::sync::
|
|
3
|
+
use std::sync::atomic::{AtomicU8, Ordering};
|
|
4
|
+
|
|
5
|
+
const BMP_LEN: usize = 1 << 16;
|
|
6
|
+
const INITIALIZED: u8 = 1;
|
|
7
|
+
const ALPHABETIC: u8 = 1 << 1;
|
|
8
|
+
const UNPRINTABLE: u8 = 1 << 2;
|
|
9
|
+
const SYMBOL: u8 = 1 << 3;
|
|
10
|
+
const KNOWN_LETTER: u8 = 1 << 4;
|
|
11
|
+
|
|
12
|
+
fn pair_key(a: char, b: char) -> u64 {
|
|
13
|
+
((a as u64) << 32) | b as u64
|
|
14
|
+
}
|
|
4
15
|
|
|
5
16
|
/// Immutable corpus statistics only. No input text is retained between calls.
|
|
6
17
|
#[pyclass(frozen)]
|
|
7
18
|
pub struct NgramModel {
|
|
8
19
|
letters: HashSet<char>,
|
|
9
|
-
pairs: HashMap<
|
|
10
|
-
|
|
20
|
+
pairs: HashMap<u64, f64>,
|
|
21
|
+
latin_pairs: Box<[f64]>,
|
|
22
|
+
max_pair_support: f64,
|
|
23
|
+
// One byte per BMP scalar. Each atomic contains the complete property
|
|
24
|
+
// record, so relaxed loads/stores need not publish any additional state.
|
|
25
|
+
properties: Box<[AtomicU8]>,
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
impl NgramModel {
|
|
29
|
+
#[inline]
|
|
30
|
+
fn pair_support(&self, before: char, current: char) -> f64 {
|
|
31
|
+
if before <= '\u{ff}' && current <= '\u{ff}' {
|
|
32
|
+
self.latin_pairs[((before as usize) << 8) | current as usize]
|
|
33
|
+
} else {
|
|
34
|
+
self.pairs
|
|
35
|
+
.get(&pair_key(before, current))
|
|
36
|
+
.copied()
|
|
37
|
+
.unwrap_or(0.0)
|
|
38
|
+
}
|
|
39
|
+
}
|
|
11
40
|
}
|
|
12
41
|
|
|
13
42
|
#[pymethods]
|
|
14
43
|
impl NgramModel {
|
|
15
44
|
#[new]
|
|
16
45
|
fn new(letters: &str, pairs: Vec<(String, f64)>) -> Self {
|
|
46
|
+
// A bounded 512 KiB table avoids hashing common Latin/ASCII pairs.
|
|
47
|
+
// It contains model weights only, independent of observed documents.
|
|
48
|
+
let mut latin_pairs = vec![0.0; 256 * 256].into_boxed_slice();
|
|
49
|
+
for (pair, weight) in &pairs {
|
|
50
|
+
let mut chars = pair.chars();
|
|
51
|
+
if let (Some(a), Some(b)) = (chars.next(), chars.next()) {
|
|
52
|
+
if a <= '\u{ff}' && b <= '\u{ff}' {
|
|
53
|
+
latin_pairs[((a as usize) << 8) | b as usize] = *weight;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
17
57
|
Self {
|
|
18
|
-
|
|
58
|
+
latin_pairs,
|
|
59
|
+
max_pair_support: pairs.iter().fold(1.0_f64, |m, (_, weight)| m.max(*weight)),
|
|
60
|
+
properties: (0..BMP_LEN).map(|_| AtomicU8::new(0)).collect(),
|
|
19
61
|
letters: letters.chars().collect(),
|
|
20
62
|
pairs: pairs
|
|
21
63
|
.iter()
|
|
22
64
|
.filter_map(|(s, weight)| {
|
|
23
65
|
let mut chars = s.chars();
|
|
24
|
-
Some(((chars.next()?, chars.next()?), *weight))
|
|
66
|
+
Some((pair_key(chars.next()?, chars.next()?), *weight))
|
|
25
67
|
})
|
|
26
68
|
.collect(),
|
|
27
69
|
}
|
|
28
70
|
}
|
|
29
71
|
|
|
30
72
|
/// Cache bounded Unicode scalar metadata, never input strings or sequences.
|
|
73
|
+
#[pyo3(signature = (raw, lower, classify, minimum=None))]
|
|
31
74
|
fn quality(
|
|
32
75
|
&self,
|
|
33
76
|
py: Python<'_>,
|
|
34
77
|
raw: &str,
|
|
35
78
|
lower: &str,
|
|
36
79
|
classify: &Bound<'_, PyAny>,
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
.
|
|
44
|
-
|
|
80
|
+
minimum: Option<f64>,
|
|
81
|
+
) -> PyResult<Option<(f64, f64)>> {
|
|
82
|
+
let unknown: HashSet<char> = raw
|
|
83
|
+
.chars()
|
|
84
|
+
.chain(lower.chars())
|
|
85
|
+
.filter(|&c| {
|
|
86
|
+
self.properties
|
|
87
|
+
.get(c as usize)
|
|
88
|
+
.is_none_or(|flags| flags.load(Ordering::Relaxed) == 0)
|
|
89
|
+
})
|
|
90
|
+
.collect();
|
|
45
91
|
let chars: Vec<char> = unknown.into_iter().collect();
|
|
46
92
|
let mut local = HashMap::new();
|
|
47
93
|
if !chars.is_empty() {
|
|
@@ -53,27 +99,46 @@ impl NgramModel {
|
|
|
53
99
|
"one property record per scalar is required",
|
|
54
100
|
));
|
|
55
101
|
}
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
102
|
+
for (c, (alpha, bad, symbol)) in chars.into_iter().zip(values) {
|
|
103
|
+
let flags = INITIALIZED
|
|
104
|
+
| if alpha { ALPHABETIC } else { 0 }
|
|
105
|
+
| if bad { UNPRINTABLE } else { 0 }
|
|
106
|
+
| if symbol { SYMBOL } else { 0 }
|
|
107
|
+
| if self.letters.contains(&c) {
|
|
108
|
+
KNOWN_LETTER
|
|
109
|
+
} else {
|
|
110
|
+
0
|
|
111
|
+
};
|
|
112
|
+
if let Some(entry) = self.properties.get(c as usize) {
|
|
113
|
+
entry.store(flags, Ordering::Relaxed);
|
|
61
114
|
} else {
|
|
62
115
|
local.insert(c, flags);
|
|
63
116
|
}
|
|
64
117
|
}
|
|
65
118
|
}
|
|
66
119
|
Ok(py.detach(|| {
|
|
67
|
-
let
|
|
68
|
-
|
|
120
|
+
let flags = |c: char| {
|
|
121
|
+
self.properties
|
|
122
|
+
.get(c as usize)
|
|
123
|
+
.map_or_else(|| local[&c], |entry| entry.load(Ordering::Relaxed))
|
|
124
|
+
};
|
|
69
125
|
let mut n = 0usize;
|
|
70
126
|
let mut bad = 0usize;
|
|
71
127
|
let mut symbols = 0usize;
|
|
72
128
|
for c in raw.chars() {
|
|
73
129
|
n += 1;
|
|
74
|
-
let
|
|
75
|
-
bad += usize::from(
|
|
76
|
-
symbols += usize::from(
|
|
130
|
+
let properties = flags(c);
|
|
131
|
+
bad += usize::from(properties & UNPRINTABLE != 0);
|
|
132
|
+
symbols += usize::from(properties & SYMBOL != 0);
|
|
133
|
+
}
|
|
134
|
+
let bad_ratio = bad as f64 / n.max(1) as f64;
|
|
135
|
+
let penalty = 5.0 * bad_ratio + 3.0 * symbols as f64 / n.max(1) as f64;
|
|
136
|
+
// An optimistic bound: even perfect letter/pair coverage cannot
|
|
137
|
+
// recover this candidate. Keep slack for Python's score rounding.
|
|
138
|
+
if minimum
|
|
139
|
+
.is_some_and(|floor| 0.25 + 0.75 * self.max_pair_support - penalty + 1e-9 < floor)
|
|
140
|
+
{
|
|
141
|
+
return None;
|
|
77
142
|
}
|
|
78
143
|
let mut letter_count = 0usize;
|
|
79
144
|
let mut known_letters = 0usize;
|
|
@@ -81,10 +146,11 @@ impl NgramModel {
|
|
|
81
146
|
let mut known_pair_weight = 0.0f64;
|
|
82
147
|
let mut previous = None;
|
|
83
148
|
for c in lower.chars() {
|
|
84
|
-
let
|
|
149
|
+
let properties = flags(c);
|
|
150
|
+
let current = if properties & ALPHABETIC != 0 { c } else { ' ' };
|
|
85
151
|
if current != ' ' {
|
|
86
152
|
letter_count += 1;
|
|
87
|
-
known_letters += usize::from(
|
|
153
|
+
known_letters += usize::from(properties & KNOWN_LETTER != 0);
|
|
88
154
|
}
|
|
89
155
|
if let Some(before) = previous {
|
|
90
156
|
if before != ' ' || current != ' ' {
|
|
@@ -94,21 +160,18 @@ impl NgramModel {
|
|
|
94
160
|
1
|
|
95
161
|
};
|
|
96
162
|
pair_weight += weight;
|
|
97
|
-
|
|
98
|
-
known_pair_weight += weight as f64 * support;
|
|
99
|
-
}
|
|
163
|
+
known_pair_weight += weight as f64 * self.pair_support(before, current);
|
|
100
164
|
}
|
|
101
165
|
}
|
|
102
166
|
previous = Some(current);
|
|
103
167
|
}
|
|
104
|
-
|
|
105
|
-
(
|
|
168
|
+
Some((
|
|
106
169
|
0.25 * known_letters as f64 / letter_count.max(1) as f64
|
|
107
170
|
+ 0.75 * known_pair_weight / pair_weight.max(1) as f64
|
|
108
171
|
- 5.0 * bad_ratio
|
|
109
172
|
- 3.0 * symbols as f64 / n.max(1) as f64,
|
|
110
173
|
bad_ratio,
|
|
111
|
-
)
|
|
174
|
+
))
|
|
112
175
|
}))
|
|
113
176
|
}
|
|
114
177
|
|
|
@@ -135,9 +198,7 @@ impl NgramModel {
|
|
|
135
198
|
1
|
|
136
199
|
};
|
|
137
200
|
pair_weight += weight;
|
|
138
|
-
|
|
139
|
-
known_pair_weight += weight as f64 * support;
|
|
140
|
-
}
|
|
201
|
+
known_pair_weight += weight as f64 * self.pair_support(before, current);
|
|
141
202
|
}
|
|
142
203
|
}
|
|
143
204
|
previous = Some(current);
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
/// Returns (is_valid, confidence_0_to_1).
|
|
2
2
|
pub fn utf8_check(data: &[u8]) -> (bool, f64) {
|
|
3
|
-
|
|
3
|
+
// The compat validator preserves the first-error offset and early exit.
|
|
4
|
+
// x86 CPU features are detected at runtime; unsupported targets use scalar validation.
|
|
5
|
+
match simdutf8::compat::from_utf8(data) {
|
|
4
6
|
Ok(_) => (true, 1.0),
|
|
5
7
|
Err(e) => {
|
|
6
8
|
let conf = if data.is_empty() {
|