bytesense 1.0.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. {bytesense-1.0.0 → bytesense-1.2.0}/CHANGELOG.md +36 -0
  2. {bytesense-1.0.0 → bytesense-1.2.0}/MANIFEST.in +1 -1
  3. {bytesense-1.0.0 → bytesense-1.2.0}/PKG-INFO +17 -3
  4. {bytesense-1.0.0 → bytesense-1.2.0}/README.md +15 -2
  5. bytesense-1.2.0/THIRD_PARTY_LICENSES +23 -0
  6. {bytesense-1.0.0 → bytesense-1.2.0}/pyproject.toml +4 -1
  7. {bytesense-1.0.0 → bytesense-1.2.0}/rust/Cargo.lock +8 -1
  8. {bytesense-1.0.0 → bytesense-1.2.0}/rust/Cargo.toml +2 -1
  9. {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/language.rs +95 -34
  10. {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/utf8.rs +3 -1
  11. bytesense-1.2.0/scripts/benchmark_compare.py +234 -0
  12. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/benchmark_v1.py +12 -1
  13. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/check_dist.py +4 -0
  14. bytesense-1.2.0/scripts/check_evaluation.py +41 -0
  15. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/check_version.py +2 -0
  16. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/evaluate_corpus.py +1 -0
  17. bytesense-1.2.0/scripts/evaluate_udhr.py +172 -0
  18. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/api.py +85 -13
  19. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/scoring.py +29 -5
  20. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/streaming.py +24 -2
  21. bytesense-1.2.0/src/bytesense/version.py +4 -0
  22. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/PKG-INFO +17 -3
  23. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/SOURCES.txt +5 -0
  24. bytesense-1.2.0/tests/test_rust.py +129 -0
  25. bytesense-1.2.0/tests/test_utf8_validation.py +109 -0
  26. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_v1_contracts.py +75 -0
  27. bytesense-1.0.0/src/bytesense/version.py +0 -4
  28. bytesense-1.0.0/tests/test_rust.py +0 -54
  29. {bytesense-1.0.0 → bytesense-1.2.0}/CODE_OF_CONDUCT.md +0 -0
  30. {bytesense-1.0.0 → bytesense-1.2.0}/CONTRIBUTING.md +0 -0
  31. {bytesense-1.0.0 → bytesense-1.2.0}/LICENSE +0 -0
  32. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/__init__.py +0 -0
  33. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/cn_official_manifest.json +0 -0
  34. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/conftest.py +0 -0
  35. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/hard_scenarios.py +0 -0
  36. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/test_bench_detection.py +0 -0
  37. {bytesense-1.0.0 → bytesense-1.2.0}/benchmarks/test_hard_scenarios.py +0 -0
  38. {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/histogram.rs +0 -0
  39. {bytesense-1.0.0 → bytesense-1.2.0}/rust/src/lib.rs +0 -0
  40. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/build_fingerprints.py +0 -0
  41. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/compare_libraries.sh +0 -0
  42. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/fetch_cn_benchmark_samples.py +0 -0
  43. {bytesense-1.0.0 → bytesense-1.2.0}/scripts/run_all_benchmarks.sh +0 -0
  44. {bytesense-1.0.0 → bytesense-1.2.0}/setup.cfg +0 -0
  45. {bytesense-1.0.0 → bytesense-1.2.0}/setup.py +0 -0
  46. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/__init__.py +0 -0
  47. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/_rust.py +0 -0
  48. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/candidate.py +0 -0
  49. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/cli.py +0 -0
  50. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/coherence.py +0 -0
  51. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/constant.py +0 -0
  52. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/__init__.py +0 -0
  53. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/fingerprints.py +0 -0
  54. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/data/language.json.gz +0 -0
  55. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/fingerprint.py +0 -0
  56. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/heuristics.py +0 -0
  57. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/hints.py +0 -0
  58. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/legacy.py +0 -0
  59. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/mess.py +0 -0
  60. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/models.py +0 -0
  61. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/multi.py +0 -0
  62. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/py.typed +0 -0
  63. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense/repair.py +0 -0
  64. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/dependency_links.txt +0 -0
  65. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/entry_points.txt +0 -0
  66. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/not-zip-safe +0 -0
  67. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/requires.txt +0 -0
  68. {bytesense-1.0.0 → bytesense-1.2.0}/src/bytesense.egg-info/top_level.txt +0 -0
  69. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_api.py +0 -0
  70. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_api_extra.py +0 -0
  71. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_candidate.py +0 -0
  72. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_cli.py +0 -0
  73. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_coherence.py +0 -0
  74. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_fingerprint.py +0 -0
  75. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_heuristics_extra.py +0 -0
  76. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_hints.py +0 -0
  77. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_hints_extra.py +0 -0
  78. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_legacy.py +0 -0
  79. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_mess.py +0 -0
  80. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_mess_extra.py +0 -0
  81. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_multi.py +0 -0
  82. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_repair.py +0 -0
  83. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_rust_layer.py +0 -0
  84. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_streaming.py +0 -0
  85. {bytesense-1.0.0 → bytesense-1.2.0}/tests/test_streaming_v2.py +0 -0
@@ -1,5 +1,41 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.2.0
4
+
5
+ ### UTF-8 validation
6
+ - Validate complete UTF-8 and UTF-8-SIG inputs in `from_bytes()` with SIMD in native builds, without allocating decoded Unicode. Pin `simdutf8` 0.1.5 and preserve strict rejection and first-error offsets with its compatibility validator.
7
+ - Retain runtime x86 CPU dispatch and scalar fallbacks; no target-specific CPU flags or Python runtime dependencies are required.
8
+ - Decode pure-Python UTF-8 inputs larger than 64 KiB in bounded blocks. Preserve the existing threshold for other codecs, including custom decoders.
9
+ - Reuse the ASCII check within each detection call. Public results, scoring and stream behavior are unchanged.
10
+
11
+ ### Verification
12
+ - Add strict-decoder parity, malformed-scalar, SIMD/block boundary, allocation and custom-codec regressions.
13
+ - Extend rotating-input benchmarks with 64 KiB and 8 MiB UTF-8 workloads and isolated before/after comparisons.
14
+ - Preserve existing corpus accuracy and native/Python predictions; include the native dependency's license in distributions.
15
+
16
+ See the benchmark documentation for measured gains and their scope.
17
+
18
+ ## 1.1.0
19
+
20
+ ### Faster legacy detection
21
+ - Cache Unicode scalar properties in a fixed 64 KiB atomic table; remove per-character property hash lookups and read locks.
22
+ - Use a bounded 512 KiB native lookup table for Latin/ASCII model pairs, with sparse lookup for other scripts.
23
+ - Prune pair scoring only when an optimistic evidence bound proves that a candidate cannot qualify. Retain matching Python/native decisions.
24
+ - Normalize and deduplicate the built-in codec catalog lazily once; caller-supplied codecs are still validated on each call.
25
+
26
+ ### Correctness
27
+ - Normalize scoring input to NFC and stop penalizing combining marks as symbol noise. Original bytes and decoded text are never normalized or rewritten.
28
+ - Reach all statistically eligible candidates during full-input validation instead of stopping at the six display hypotheses. Public alternatives remain limited to five.
29
+ - Recover plausible BOM-less UTF-16 with NULs in both lanes before declaring binary content, subject to strict full-input validation and caller filters.
30
+ - Recognize ISO-2022 shift controls and UTF-7 Unicode spacing/format characters as text syntax.
31
+
32
+ ### Verification
33
+ - Add cache-boundary, concurrent cold-cache, dense/sparse pair parity, pruning-bound, canonical-equivalence and validation-depth regressions.
34
+ - Add a pinned 28-document UDHR transfer evaluation, with exact Unicode comparisons, recorded exclusions and separate legacy/Unicode counts.
35
+ - Add single-engine benchmark runs for before/after comparisons. Model weights and public API signatures are unchanged.
36
+
37
+ See the benchmark documentation for measured gains and remaining limitations.
38
+
3
39
 
4
40
  ## 1.0.0
5
41
 
@@ -1,4 +1,4 @@
1
- include LICENSE README.md CHANGELOG.md
1
+ include LICENSE THIRD_PARTY_LICENSES README.md CHANGELOG.md
2
2
  recursive-include rust *.toml *.lock *.rs
3
3
  recursive-include src/bytesense *.py py.typed
4
4
  prune rust/target
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bytesense
3
- Version: 1.0.0
3
+ Version: 1.2.0
4
4
  Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
5
5
  Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
6
6
  Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
@@ -33,6 +33,7 @@ Classifier: Typing :: Typed
33
33
  Requires-Python: >=3.9
34
34
  Description-Content-Type: text/markdown
35
35
  License-File: LICENSE
36
+ License-File: THIRD_PARTY_LICENSES
36
37
  Provides-Extra: fast
37
38
  Provides-Extra: docs
38
39
  Requires-Dist: mkdocs>=1.6; extra == "docs"
@@ -67,7 +68,7 @@ Dynamic: license-file
67
68
 
68
69
  **bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
69
70
 
70
- Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
71
+ Final detection results strictly validate the selected encoding against the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
71
72
 
72
73
  ```python
73
74
  from bytesense import from_bytes
@@ -131,13 +132,22 @@ cat report.csv | bytesense --minimal -
131
132
 
132
133
  The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
133
134
 
135
+ ## Used by
136
+
137
+ [ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
138
+ detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
139
+ safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
140
+ Built by the same maintainer; an independent project, not affiliated with Sentry.
141
+
134
142
  ## Measured accuracy and performance
135
143
 
136
144
  The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
137
145
 
138
146
  | Detector | Exact Unicode matches | Accuracy |
139
147
  |---|---:|---:|
140
- | **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
148
+ | **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
149
+ | bytesense 1.1.0 | 1907 / 2078 | 91.77% |
150
+ | bytesense 1.0.0 | 1906 / 2078 | 91.72% |
141
151
  | bytesense 0.1.2 | 883 / 2078 | 42.49% |
142
152
  | chardet 7.6.0 | 2060 / 2078 | 99.13% |
143
153
  | charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
@@ -145,6 +155,10 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
145
155
 
146
156
  See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
147
157
 
158
+ **New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
159
+
160
+ The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
161
+
148
162
  ## Upgrading from 0.x
149
163
 
150
164
  - Language reporting is opt-in: `include_language=True`.
@@ -10,7 +10,7 @@
10
10
 
11
11
  **bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
12
12
 
13
- Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
13
+ Final detection results strictly validate the selected encoding against the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
14
14
 
15
15
  ```python
16
16
  from bytesense import from_bytes
@@ -74,13 +74,22 @@ cat report.csv | bytesense --minimal -
74
74
 
75
75
  The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
76
76
 
77
+ ## Used by
78
+
79
+ [ingest-sentry](https://github.com/oguzhankir/ingest-sentry) uses bytesense for encoding
80
+ detection and full-snapshot validation before CSV/TSV imports. It adds import contracts,
81
+ safe UTF-8 normalization, string-preserving pandas/Polars adapters, and a GitHub Action.
82
+ Built by the same maintainer; an independent project, not affiliated with Sentry.
83
+
77
84
  ## Measured accuracy and performance
78
85
 
79
86
  The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
80
87
 
81
88
  | Detector | Exact Unicode matches | Accuracy |
82
89
  |---|---:|---:|
83
- | **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
90
+ | **bytesense 1.2.0** | **1907 / 2078** | **91.77%** |
91
+ | bytesense 1.1.0 | 1907 / 2078 | 91.77% |
92
+ | bytesense 1.0.0 | 1906 / 2078 | 91.72% |
84
93
  | bytesense 0.1.2 | 883 / 2078 | 42.49% |
85
94
  | chardet 7.6.0 | 2060 / 2078 | 99.13% |
86
95
  | charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
@@ -88,6 +97,10 @@ The v1 engine replaces the previous collection of special-case ranking rules wit
88
97
 
89
98
  See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
90
99
 
100
+ **New in 1.2:** `from_bytes()` validates UTF-8 using native SIMD without allocating decoded Unicode; its pure-Python path uses 64 KiB blocks above 64 KiB. On the measured 1 MiB UTF-8 workload, native `from_bytes()` median latency decreased from **1.564 ms to 0.294 ms (5.3×)** versus 1.1, with every byte still validated. Peak temporary Python allocations decreased from about **6 MiB to 6 KiB**; this is not total process memory.
101
+
102
+ The existing 2,078-file and 432-case transfer results are unchanged. This release improves UTF-8 validation cost, not general encoding accuracy; chardet remains more accurate on both corpora. See the [benchmark methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for paired measurements, competitors, backend scope and reproduction commands.
103
+
91
104
  ## Upgrading from 0.x
92
105
 
93
106
  - Language reporting is opt-in: `include_language=True`.
@@ -0,0 +1,23 @@
1
+ The optional native extension includes simdutf8 0.1.5.
2
+ Source: https://github.com/rusticstuff/simdutf8/tree/87ee8d9d20b849eae1974a82b248ee6fe7491bcc
3
+ Upstream license: MIT OR Apache-2.0; distributed under the MIT option below.
4
+
5
+ MIT License
6
+
7
+ Permission is hereby granted, free of charge, to any person obtaining a copy
8
+ of this software and associated documentation files (the "Software"), to deal
9
+ in the Software without restriction, including without limitation the rights
10
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11
+ copies of the Software, and to permit persons to whom the Software is
12
+ furnished to do so, subject to the following conditions:
13
+
14
+ The above copyright notice and this permission notice shall be included in all
15
+ copies or substantial portions of the Software.
16
+
17
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
20
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
23
+ SOFTWARE.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "bytesense"
7
- version = "1.0.0"
7
+ version = "1.2.0"
8
8
  description = "Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -82,6 +82,9 @@ Issues = "https://github.com/oguzhankir/bytesense/issues"
82
82
  Changelog = "https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md"
83
83
  Documentation = "https://oguzhankir.github.io/bytesense/"
84
84
 
85
+ [tool.setuptools]
86
+ license-files = ["LICENSE", "THIRD_PARTY_LICENSES"]
87
+
85
88
  [tool.setuptools.packages.find]
86
89
  where = ["src"]
87
90
 
@@ -4,9 +4,10 @@ version = 4
4
4
 
5
5
  [[package]]
6
6
  name = "bytesense-core"
7
- version = "1.0.0"
7
+ version = "1.2.0"
8
8
  dependencies = [
9
9
  "pyo3",
10
+ "simdutf8",
10
11
  ]
11
12
 
12
13
  [[package]]
@@ -109,6 +110,12 @@ dependencies = [
109
110
  "proc-macro2",
110
111
  ]
111
112
 
113
+ [[package]]
114
+ name = "simdutf8"
115
+ version = "0.1.5"
116
+ source = "registry+https://github.com/rust-lang/crates.io-index"
117
+ checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e"
118
+
112
119
  [[package]]
113
120
  name = "syn"
114
121
  version = "2.0.117"
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "bytesense-core"
3
- version = "1.0.0"
3
+ version = "1.2.0"
4
4
  edition = "2021"
5
5
  rust-version = "1.88"
6
6
 
@@ -10,6 +10,7 @@ crate-type = ["cdylib"]
10
10
 
11
11
  [dependencies]
12
12
  pyo3 = { version = "0.28.2", features = ["abi3-py39"] }
13
+ simdutf8 = "=0.1.5"
13
14
 
14
15
  [profile.release]
15
16
  opt-level = 3
@@ -1,47 +1,93 @@
1
1
  use pyo3::prelude::*;
2
2
  use std::collections::{HashMap, HashSet};
3
- use std::sync::RwLock;
3
+ use std::sync::atomic::{AtomicU8, Ordering};
4
+
5
+ const BMP_LEN: usize = 1 << 16;
6
+ const INITIALIZED: u8 = 1;
7
+ const ALPHABETIC: u8 = 1 << 1;
8
+ const UNPRINTABLE: u8 = 1 << 2;
9
+ const SYMBOL: u8 = 1 << 3;
10
+ const KNOWN_LETTER: u8 = 1 << 4;
11
+
12
+ fn pair_key(a: char, b: char) -> u64 {
13
+ ((a as u64) << 32) | b as u64
14
+ }
4
15
 
5
16
  /// Immutable corpus statistics only. No input text is retained between calls.
6
17
  #[pyclass(frozen)]
7
18
  pub struct NgramModel {
8
19
  letters: HashSet<char>,
9
- pairs: HashMap<(char, char), f64>,
10
- properties: RwLock<HashMap<char, (bool, bool, bool)>>,
20
+ pairs: HashMap<u64, f64>,
21
+ latin_pairs: Box<[f64]>,
22
+ max_pair_support: f64,
23
+ // One byte per BMP scalar. Each atomic contains the complete property
24
+ // record, so relaxed loads/stores need not publish any additional state.
25
+ properties: Box<[AtomicU8]>,
26
+ }
27
+
28
+ impl NgramModel {
29
+ #[inline]
30
+ fn pair_support(&self, before: char, current: char) -> f64 {
31
+ if before <= '\u{ff}' && current <= '\u{ff}' {
32
+ self.latin_pairs[((before as usize) << 8) | current as usize]
33
+ } else {
34
+ self.pairs
35
+ .get(&pair_key(before, current))
36
+ .copied()
37
+ .unwrap_or(0.0)
38
+ }
39
+ }
11
40
  }
12
41
 
13
42
  #[pymethods]
14
43
  impl NgramModel {
15
44
  #[new]
16
45
  fn new(letters: &str, pairs: Vec<(String, f64)>) -> Self {
46
+ // A bounded 512 KiB table avoids hashing common Latin/ASCII pairs.
47
+ // It contains model weights only, independent of observed documents.
48
+ let mut latin_pairs = vec![0.0; 256 * 256].into_boxed_slice();
49
+ for (pair, weight) in &pairs {
50
+ let mut chars = pair.chars();
51
+ if let (Some(a), Some(b)) = (chars.next(), chars.next()) {
52
+ if a <= '\u{ff}' && b <= '\u{ff}' {
53
+ latin_pairs[((a as usize) << 8) | b as usize] = *weight;
54
+ }
55
+ }
56
+ }
17
57
  Self {
18
- properties: RwLock::new(HashMap::new()),
58
+ latin_pairs,
59
+ max_pair_support: pairs.iter().fold(1.0_f64, |m, (_, weight)| m.max(*weight)),
60
+ properties: (0..BMP_LEN).map(|_| AtomicU8::new(0)).collect(),
19
61
  letters: letters.chars().collect(),
20
62
  pairs: pairs
21
63
  .iter()
22
64
  .filter_map(|(s, weight)| {
23
65
  let mut chars = s.chars();
24
- Some(((chars.next()?, chars.next()?), *weight))
66
+ Some((pair_key(chars.next()?, chars.next()?), *weight))
25
67
  })
26
68
  .collect(),
27
69
  }
28
70
  }
29
71
 
30
72
  /// Cache bounded Unicode scalar metadata, never input strings or sequences.
73
+ #[pyo3(signature = (raw, lower, classify, minimum=None))]
31
74
  fn quality(
32
75
  &self,
33
76
  py: Python<'_>,
34
77
  raw: &str,
35
78
  lower: &str,
36
79
  classify: &Bound<'_, PyAny>,
37
- ) -> PyResult<(f64, f64)> {
38
- let unknown: HashSet<char> = {
39
- let properties = self.properties.read().unwrap();
40
- raw.chars()
41
- .chain(lower.chars())
42
- .filter(|c| !properties.contains_key(c))
43
- .collect()
44
- };
80
+ minimum: Option<f64>,
81
+ ) -> PyResult<Option<(f64, f64)>> {
82
+ let unknown: HashSet<char> = raw
83
+ .chars()
84
+ .chain(lower.chars())
85
+ .filter(|&c| {
86
+ self.properties
87
+ .get(c as usize)
88
+ .is_none_or(|flags| flags.load(Ordering::Relaxed) == 0)
89
+ })
90
+ .collect();
45
91
  let chars: Vec<char> = unknown.into_iter().collect();
46
92
  let mut local = HashMap::new();
47
93
  if !chars.is_empty() {
@@ -53,27 +99,46 @@ impl NgramModel {
53
99
  "one property record per scalar is required",
54
100
  ));
55
101
  }
56
- let mut properties = self.properties.write().unwrap();
57
- for (c, flags) in chars.into_iter().zip(values) {
58
- // BMP cache has at most 65,536 entries. Other scalars are local.
59
- if c <= '\u{ffff}' {
60
- properties.insert(c, flags);
102
+ for (c, (alpha, bad, symbol)) in chars.into_iter().zip(values) {
103
+ let flags = INITIALIZED
104
+ | if alpha { ALPHABETIC } else { 0 }
105
+ | if bad { UNPRINTABLE } else { 0 }
106
+ | if symbol { SYMBOL } else { 0 }
107
+ | if self.letters.contains(&c) {
108
+ KNOWN_LETTER
109
+ } else {
110
+ 0
111
+ };
112
+ if let Some(entry) = self.properties.get(c as usize) {
113
+ entry.store(flags, Ordering::Relaxed);
61
114
  } else {
62
115
  local.insert(c, flags);
63
116
  }
64
117
  }
65
118
  }
66
119
  Ok(py.detach(|| {
67
- let properties = self.properties.read().unwrap();
68
- let flags = |c: &char| properties.get(c).or_else(|| local.get(c)).unwrap();
120
+ let flags = |c: char| {
121
+ self.properties
122
+ .get(c as usize)
123
+ .map_or_else(|| local[&c], |entry| entry.load(Ordering::Relaxed))
124
+ };
69
125
  let mut n = 0usize;
70
126
  let mut bad = 0usize;
71
127
  let mut symbols = 0usize;
72
128
  for c in raw.chars() {
73
129
  n += 1;
74
- let (_, unprintable, symbol) = flags(&c);
75
- bad += usize::from(*unprintable);
76
- symbols += usize::from(*symbol);
130
+ let properties = flags(c);
131
+ bad += usize::from(properties & UNPRINTABLE != 0);
132
+ symbols += usize::from(properties & SYMBOL != 0);
133
+ }
134
+ let bad_ratio = bad as f64 / n.max(1) as f64;
135
+ let penalty = 5.0 * bad_ratio + 3.0 * symbols as f64 / n.max(1) as f64;
136
+ // An optimistic bound: even perfect letter/pair coverage cannot
137
+ // recover this candidate. Keep slack for Python's score rounding.
138
+ if minimum
139
+ .is_some_and(|floor| 0.25 + 0.75 * self.max_pair_support - penalty + 1e-9 < floor)
140
+ {
141
+ return None;
77
142
  }
78
143
  let mut letter_count = 0usize;
79
144
  let mut known_letters = 0usize;
@@ -81,10 +146,11 @@ impl NgramModel {
81
146
  let mut known_pair_weight = 0.0f64;
82
147
  let mut previous = None;
83
148
  for c in lower.chars() {
84
- let current = if flags(&c).0 { c } else { ' ' };
149
+ let properties = flags(c);
150
+ let current = if properties & ALPHABETIC != 0 { c } else { ' ' };
85
151
  if current != ' ' {
86
152
  letter_count += 1;
87
- known_letters += usize::from(self.letters.contains(&current));
153
+ known_letters += usize::from(properties & KNOWN_LETTER != 0);
88
154
  }
89
155
  if let Some(before) = previous {
90
156
  if before != ' ' || current != ' ' {
@@ -94,21 +160,18 @@ impl NgramModel {
94
160
  1
95
161
  };
96
162
  pair_weight += weight;
97
- if let Some(support) = self.pairs.get(&(before, current)) {
98
- known_pair_weight += weight as f64 * support;
99
- }
163
+ known_pair_weight += weight as f64 * self.pair_support(before, current);
100
164
  }
101
165
  }
102
166
  previous = Some(current);
103
167
  }
104
- let bad_ratio = bad as f64 / n.max(1) as f64;
105
- (
168
+ Some((
106
169
  0.25 * known_letters as f64 / letter_count.max(1) as f64
107
170
  + 0.75 * known_pair_weight / pair_weight.max(1) as f64
108
171
  - 5.0 * bad_ratio
109
172
  - 3.0 * symbols as f64 / n.max(1) as f64,
110
173
  bad_ratio,
111
- )
174
+ ))
112
175
  }))
113
176
  }
114
177
 
@@ -135,9 +198,7 @@ impl NgramModel {
135
198
  1
136
199
  };
137
200
  pair_weight += weight;
138
- if let Some(support) = self.pairs.get(&(before, current)) {
139
- known_pair_weight += weight as f64 * support;
140
- }
201
+ known_pair_weight += weight as f64 * self.pair_support(before, current);
141
202
  }
142
203
  }
143
204
  previous = Some(current);
@@ -1,6 +1,8 @@
1
1
  /// Returns (is_valid, confidence_0_to_1).
2
2
  pub fn utf8_check(data: &[u8]) -> (bool, f64) {
3
- match std::str::from_utf8(data) {
3
+ // The compat validator preserves the first-error offset and early exit.
4
+ // x86 CPU features are detected at runtime; unsupported targets use scalar validation.
5
+ match simdutf8::compat::from_utf8(data) {
4
6
  Ok(_) => (true, 1.0),
5
7
  Err(e) => {
6
8
  let conf = if data.is_empty() {