bytesense 0.1.2__tar.gz → 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. bytesense-1.0.0/CHANGELOG.md +83 -0
  2. bytesense-1.0.0/CODE_OF_CONDUCT.md +83 -0
  3. bytesense-1.0.0/CONTRIBUTING.md +47 -0
  4. bytesense-1.0.0/MANIFEST.in +8 -0
  5. bytesense-1.0.0/PKG-INFO +160 -0
  6. bytesense-1.0.0/README.md +103 -0
  7. bytesense-1.0.0/benchmarks/cn_official_manifest.json +118 -0
  8. bytesense-1.0.0/benchmarks/conftest.py +217 -0
  9. bytesense-1.0.0/benchmarks/hard_scenarios.py +183 -0
  10. bytesense-1.0.0/benchmarks/test_bench_detection.py +352 -0
  11. bytesense-1.0.0/benchmarks/test_hard_scenarios.py +118 -0
  12. {bytesense-0.1.2 → bytesense-1.0.0}/pyproject.toml +21 -19
  13. {bytesense-0.1.2 → bytesense-1.0.0}/rust/Cargo.lock +13 -60
  14. {bytesense-0.1.2 → bytesense-1.0.0}/rust/Cargo.toml +3 -2
  15. {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/histogram.rs +24 -8
  16. bytesense-1.0.0/rust/src/language.rs +151 -0
  17. {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/lib.rs +3 -1
  18. {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/utf8.rs +22 -1
  19. bytesense-1.0.0/scripts/benchmark_v1.py +128 -0
  20. bytesense-1.0.0/scripts/build_fingerprints.py +105 -0
  21. bytesense-1.0.0/scripts/check_dist.py +34 -0
  22. bytesense-1.0.0/scripts/check_version.py +16 -0
  23. bytesense-1.0.0/scripts/compare_libraries.sh +80 -0
  24. bytesense-1.0.0/scripts/evaluate_corpus.py +224 -0
  25. bytesense-1.0.0/scripts/fetch_cn_benchmark_samples.py +34 -0
  26. bytesense-1.0.0/scripts/run_all_benchmarks.sh +71 -0
  27. bytesense-1.0.0/setup.cfg +4 -0
  28. bytesense-1.0.0/setup.py +37 -0
  29. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/__init__.py +2 -1
  30. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/_rust.py +5 -1
  31. bytesense-1.0.0/src/bytesense/api.py +469 -0
  32. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/candidate.py +3 -10
  33. bytesense-1.0.0/src/bytesense/cli.py +73 -0
  34. bytesense-1.0.0/src/bytesense/coherence.py +68 -0
  35. bytesense-1.0.0/src/bytesense/data/__init__.py +0 -0
  36. bytesense-1.0.0/src/bytesense/data/language.json.gz +0 -0
  37. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/fingerprint.py +55 -56
  38. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/hints.py +4 -7
  39. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/mess.py +20 -19
  40. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/models.py +20 -8
  41. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/multi.py +52 -9
  42. bytesense-1.0.0/src/bytesense/py.typed +0 -0
  43. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/repair.py +19 -8
  44. bytesense-1.0.0/src/bytesense/scoring.py +100 -0
  45. bytesense-1.0.0/src/bytesense/streaming.py +416 -0
  46. bytesense-1.0.0/src/bytesense/version.py +4 -0
  47. bytesense-1.0.0/src/bytesense.egg-info/PKG-INFO +160 -0
  48. bytesense-1.0.0/src/bytesense.egg-info/SOURCES.txt +76 -0
  49. bytesense-1.0.0/src/bytesense.egg-info/dependency_links.txt +1 -0
  50. bytesense-1.0.0/src/bytesense.egg-info/entry_points.txt +2 -0
  51. bytesense-1.0.0/src/bytesense.egg-info/not-zip-safe +1 -0
  52. bytesense-1.0.0/src/bytesense.egg-info/requires.txt +27 -0
  53. bytesense-1.0.0/src/bytesense.egg-info/top_level.txt +1 -0
  54. bytesense-1.0.0/tests/test_api.py +100 -0
  55. bytesense-1.0.0/tests/test_api_extra.py +35 -0
  56. bytesense-1.0.0/tests/test_candidate.py +13 -0
  57. bytesense-1.0.0/tests/test_cli.py +144 -0
  58. bytesense-1.0.0/tests/test_coherence.py +21 -0
  59. bytesense-1.0.0/tests/test_fingerprint.py +171 -0
  60. bytesense-1.0.0/tests/test_heuristics_extra.py +59 -0
  61. bytesense-1.0.0/tests/test_hints.py +60 -0
  62. bytesense-1.0.0/tests/test_hints_extra.py +18 -0
  63. bytesense-1.0.0/tests/test_legacy.py +41 -0
  64. bytesense-1.0.0/tests/test_mess.py +13 -0
  65. bytesense-1.0.0/tests/test_mess_extra.py +14 -0
  66. bytesense-1.0.0/tests/test_multi.py +46 -0
  67. bytesense-1.0.0/tests/test_repair.py +82 -0
  68. bytesense-1.0.0/tests/test_rust.py +54 -0
  69. bytesense-1.0.0/tests/test_rust_layer.py +29 -0
  70. bytesense-1.0.0/tests/test_streaming.py +54 -0
  71. bytesense-1.0.0/tests/test_streaming_v2.py +73 -0
  72. bytesense-1.0.0/tests/test_v1_contracts.py +276 -0
  73. bytesense-0.1.2/PKG-INFO +0 -341
  74. bytesense-0.1.2/README.md +0 -287
  75. bytesense-0.1.2/src/bytesense/api.py +0 -797
  76. bytesense-0.1.2/src/bytesense/cli.py +0 -50
  77. bytesense-0.1.2/src/bytesense/coherence.py +0 -93
  78. bytesense-0.1.2/src/bytesense/streaming.py +0 -288
  79. bytesense-0.1.2/src/bytesense/version.py +0 -4
  80. {bytesense-0.1.2 → bytesense-1.0.0}/LICENSE +0 -0
  81. {bytesense-0.1.2/src/bytesense/data → bytesense-1.0.0/benchmarks}/__init__.py +0 -0
  82. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/constant.py +0 -0
  83. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/data/fingerprints.py +0 -0
  84. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/heuristics.py +0 -0
  85. {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/legacy.py +0 -0
@@ -0,0 +1,83 @@
1
+ # Changelog
2
+
3
+
4
+ ## 1.0.0
5
+
6
+ ### Detection and correctness
7
+ - Replace the bespoke ranking pipeline with a reproducibly generated character-pair model and optional native scoring.
8
+ - Strictly validate every returned codec against the complete input, with explicit sample/validation counts and completion status.
9
+ - Normalize filters on all paths; an empty isolation list permits no codecs.
10
+ - Use BOM-consuming UTF-16/32 codecs, improve non-Latin Unicode lane detection, and recognize 7-bit shift syntax.
11
+ - Remove fabricated confidence intervals; scores are explicitly uncalibrated. Language reporting is opt-in.
12
+
13
+ ### Streams and data handling
14
+ - Consume streams to EOF by default with bounded memory and temporary-file spooling.
15
+ - Preserve late informative samples across chunk boundaries; reject feeding after finalization.
16
+ - Retain every mixed-document byte, including short tails, and preserve Unicode boundaries.
17
+ - Use strict decoding for repair and segments; do not silently replace undecodable bytes.
18
+ - Add CLI stdin support, configuration flags and meaningful exit codes.
19
+
20
+ ### Packaging and verification
21
+ - Python 3.9+, portable compiler-free wheel/sdist installs, typed package marker, and CPython abi3 native wheels.
22
+ - Distinct pure/native CI, installed-wheel tests, version checks, pinned/hash-verified corpus inputs and reproducible comparison tools.
23
+ - Consolidate documentation; retain historical details in this changelog.
24
+
25
+ See the README migration notes for intentional 0.x behavior changes.
26
+
27
+ ## [0.1.2] — 2025-03-26
28
+
29
+ ### Changed
30
+
31
+ - **BOM fast path:** UTF-8 with BOM now reports `encoding="utf_8_sig"` (aligned with `codecs` / `CandidateSelector`), not `utf_8`.
32
+ - **Streaming:** In-band HTML/XML hints are probed on every `feed()` until a hint is found (scan limited to the first 4KB inside `_probe_inband_hint`); shared regex patterns imported from `hints.py`; `codecs` imported at module level.
33
+ - **`detect_multi`:** Adjacent same-encoding merges no longer re-run `from_bytes` on the full merged span; `byte_count` on the shared `DetectionResult` is updated with `dataclasses.replace`.
34
+ - **`repair_bytes`:** Explicit `max_iterations` and `chains` parameters instead of `**kwargs: object`.
35
+ - **Coverage:** `fail_under` raised from 50 to **75** (current suite ~75%; 85% remains a stretch goal with more `api` / fingerprint tests).
36
+
37
+ ### Fixed
38
+
39
+ - `DocumentSegment` / `MultiEncodingResult` now use `slots` on Python 3.10+ like other dataclasses in the package.
40
+
41
+ ## [0.1.1] — 2025-03-26
42
+
43
+ ### Added
44
+
45
+ - `examples/` scripts (basic detection, `detect()`, streaming, repair, HTTP hints, multi-encoding)
46
+ - MkDocs documentation site and GitHub Actions workflow **Docs Pages** → [GitHub Pages](https://oguzhankir.github.io/bytesense/)
47
+ - `[project.optional-dependencies]` group `docs` (`mkdocs`, `mkdocs-material`, `pymdown-extensions`)
48
+ - `project.urls.Documentation` in `pyproject.toml`
49
+
50
+ ### Fixed
51
+
52
+ - README logo on PyPI: use `raw.githubusercontent.com` URL (sdist has no `assets/` for the image)
53
+ - Source distribution: include `LICENSE` in maturin `include` so PyPI accepts `License-File` metadata
54
+
55
+ ## [0.1.0] — 2025-03-26
56
+
57
+ First published release.
58
+
59
+ ### Added
60
+
61
+ - `from_bytes()`, `from_path()`, `from_fp()`, `is_binary()` API
62
+ - `detect()` chardet / charset-normalizer drop-in compatibility
63
+ - `detect_stream()` for iterator-based streaming detection
64
+ - `StreamDetector` for incremental detection; `snapshot()`, `is_stable`, auto-stop when encoding is stable (configurable)
65
+ - In-band hints from HTML meta tags and XML declarations in `StreamDetector`
66
+ - Byte-distribution fingerprinting with pre-computed lookup table
67
+ - `repair()` and `repair_bytes()` for mojibake detection and repair; `is_mojibake()`; `RepairResult`
68
+ - `hint_from_http_headers()`, `hint_from_content()`, `best_hint()` for standalone hint extraction
69
+ - `detect_multi()` for multi-encoding documents; `MultiEncodingResult` and `DocumentSegment`
70
+ - Optional Rust core extension (`pip install "bytesense[fast]"`)
71
+ - CLI: `bytesense <file>`
72
+ - `py.typed` marker for PEP 561 compliance
73
+ - Full type annotations, mypy strict mode
74
+ - GitHub Actions CI across 3 OS × multiple Python versions
75
+
76
+ ### Changed
77
+
78
+ - `LANGUAGE_ENCODINGS` aligned with languages in `CHAR_FREQUENCIES`
79
+ - Candidate shortlist tuned for speed while preserving accuracy targets
80
+
81
+ ### Fixed
82
+
83
+ - Benchmark and packaging fixes (e.g. UTF-8 BOM test fixture, build backend configuration)
@@ -0,0 +1,83 @@
1
+ # Contributor Covenant Code of Conduct
2
+
3
+ ## Our pledge
4
+
5
+ We as members, contributors, and leaders pledge to make participation in our community a harassment-free experience for everyone, regardless of age, body size, visible or invisible disability, ethnicity, sex characteristics, gender identity and expression, level of experience, education, socio-economic status, nationality, personal appearance, race, caste, color, religion, or sexual identity and orientation.
6
+
7
+ We pledge to act and interact in ways that contribute to an open, welcoming, diverse, inclusive, and healthy community.
8
+
9
+ ## Our standards
10
+
11
+ Examples of behavior that contributes to a positive environment for our community include:
12
+
13
+ - Demonstrating empathy and kindness toward other people
14
+ - Being respectful of differing opinions, viewpoints, and experiences
15
+ - Giving and gracefully accepting constructive feedback
16
+ - Accepting responsibility and apologizing to those affected by our mistakes, and learning from the experience
17
+ - Focusing on what is best not just for us as individuals, but for the overall community
18
+
19
+ Examples of unacceptable behavior include:
20
+
21
+ - The use of sexualized language or imagery, and sexual attention or advances of any kind
22
+ - Trolling, insulting or derogatory comments, and personal or political attacks
23
+ - Public or private harassment
24
+ - Publishing others’ private information, such as a physical or email address, without their explicit permission
25
+ - Other conduct which could reasonably be considered inappropriate in a professional setting
26
+
27
+ ## Enforcement responsibilities
28
+
29
+ Community leaders are responsible for clarifying and enforcing our standards of acceptable behavior and will take appropriate and fair corrective action in response to any behavior that they deem inappropriate, threatening, offensive, or harmful.
30
+
31
+ Community leaders have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, and will communicate reasons for moderation decisions when appropriate.
32
+
33
+ ## Scope
34
+
35
+ This Code of Conduct applies within all community spaces, and also applies when an individual is officially representing the community in public spaces. Examples of representing our community include using an official email address, posting via an official social media account, or acting as an appointed representative at an online or offline event.
36
+
37
+ ## Enforcement
38
+
39
+ Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to the community leaders responsible for enforcement at **oguzhankir17@gmail.com**. All complaints will be reviewed and investigated promptly and fairly.
40
+
41
+ All community leaders are obligated to respect the privacy and security of the reporter of any incident.
42
+
43
+ ## Enforcement guidelines
44
+
45
+ Community leaders will follow these Community Impact Guidelines in determining the consequences for any action they deem in violation of this Code of Conduct:
46
+
47
+ ### 1. Correction
48
+
49
+ **Community Impact**: Use of inappropriate language or other behavior deemed unprofessional or unwelcome in the community.
50
+
51
+ **Consequence**: A private, written warning from community leaders, providing clarity around the nature of the violation and an explanation of why the behavior was inappropriate. A public apology may be requested.
52
+
53
+ ### 2. Warning
54
+
55
+ **Community Impact**: A violation through a single incident or series of actions.
56
+
57
+ **Consequence**: A warning with consequences for continued behavior. No interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, for a specified period of time. This includes avoiding interactions in community spaces as well as external channels like social media. Violating these terms may lead to a temporary or permanent ban.
58
+
59
+ ### 3. Temporary ban
60
+
61
+ **Community Impact**: A serious violation of community standards, including sustained inappropriate behavior.
62
+
63
+ **Consequence**: A temporary ban from any sort of interaction or public communication with the community for a specified period of time. No public or private interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, is allowed during this period. Violating these terms may lead to a permanent ban.
64
+
65
+ ### 4. Permanent ban
66
+
67
+ **Community Impact**: Demonstrating a pattern of violation of community standards, including sustained inappropriate behavior, harassment of an individual, or aggression toward or disparagement of classes of individuals.
68
+
69
+ **Consequence**: A permanent ban from any sort of public interaction within the community.
70
+
71
+ ## Attribution
72
+
73
+ This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 2.1, available at [https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
74
+
75
+ Community Impact Guidelines were inspired by [Mozilla’s code of conduct enforcement ladder][Mozilla CoC].
76
+
77
+ For answers to common questions about this code of conduct, see the FAQ at [https://www.contributor-covenant.org/faq][FAQ]. Translations are available at [https://www.contributor-covenant.org/translations][translations].
78
+
79
+ [homepage]: https://www.contributor-covenant.org
80
+ [v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
81
+ [Mozilla CoC]: https://github.com/mozilla/diversity
82
+ [FAQ]: https://www.contributor-covenant.org/faq
83
+ [translations]: https://www.contributor-covenant.org/translations
@@ -0,0 +1,47 @@
1
+ # Contributing
2
+
3
+ Follow the [Code of Conduct](CODE_OF_CONDUCT.md). Use a focused `oguzhankir/<topic>` branch and sign commits with `git commit -s` under the project's DCO.
4
+
5
+ ## Development
6
+
7
+ ```bash
8
+ python -m venv .venv
9
+ source .venv/bin/activate
10
+ BYTESENSE_BUILD_RUST=0 pip install -e '.[dev,docs]'
11
+ python scripts/fetch_cn_benchmark_samples.py
12
+ ruff check src tests benchmarks scripts setup.py
13
+ mypy src/bytesense
14
+ BYTESENSE_PURE_PYTHON=1 pytest tests --cov=bytesense --cov-branch
15
+ pytest benchmarks/test_bench_detection.py benchmarks/test_hard_scenarios.py --benchmark-disable
16
+ mkdocs build --strict
17
+ ```
18
+
19
+ On Windows, set environment variables using your shell's syntax. Native development requires Rust 1.88+:
20
+
21
+ ```bash
22
+ BYTESENSE_BUILD_RUST=1 pip install -e . --no-build-isolation
23
+ BYTESENSE_EXPECT_RUST=1 pytest tests
24
+ cargo test --manifest-path rust/Cargo.toml --locked
25
+ ```
26
+
27
+ CI tests pure and native installations separately on Linux, macOS and Windows. Packaging jobs install and test actual wheels, check abi3 compatibility, and test the sdist without compiling Rust. The coverage floor remains 75%; timing measurements are reports, not noisy speed assertions.
28
+
29
+ ## Model and benchmarks
30
+
31
+ The [benchmark documentation](docs/benchmarks.md) describes corpus pins, Unicode-based grouping and reproduction commands. Keep development and holdout documents separate. Do not tune the model on the holdout, silently omit missing inputs, compare differently sized payloads, or advertise sample-only latency as full-input validation latency.
32
+
33
+ `language.json.gz` contains aggregate character-pair statistics; no test documents are bundled. The core retains static model metadata, never document text in a global cache. Native and Python paths must agree within numerical tolerance and produce the same rounded detection scores.
34
+
35
+ ## Release
36
+
37
+ Keep `pyproject.toml`, `rust/Cargo.toml`, and `src/bytesense/version.py` aligned; `python scripts/check_version.py` checks them. Update the existing changelog and API/migration documentation when behavior changes.
38
+
39
+ ```bash
40
+ BYTESENSE_BUILD_RUST=0 python -m build
41
+ python scripts/check_dist.py dist
42
+ python -m twine check --strict dist/*
43
+ ```
44
+
45
+ Publishing a GitHub Release triggers the release workflow. It checks the tag, builds and tests portable/native distributions, then uses the configured PyPI trusted publisher environment. A PR does not publish a release.
46
+
47
+ Keep documentation focused: README, quick start, API reference, benchmarks, and the shared changelog. Put rationale and validation detail in the PR instead of creating additional planning documents.
@@ -0,0 +1,8 @@
1
+ include LICENSE README.md CHANGELOG.md
2
+ recursive-include rust *.toml *.lock *.rs
3
+ recursive-include src/bytesense *.py py.typed
4
+ prune rust/target
5
+
6
+ include CONTRIBUTING.md CODE_OF_CONDUCT.md
7
+ recursive-include scripts *.py *.sh
8
+ recursive-include benchmarks *.py cn_official_manifest.json
@@ -0,0 +1,160 @@
1
+ Metadata-Version: 2.4
2
+ Name: bytesense
3
+ Version: 1.0.0
4
+ Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
5
+ Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
6
+ Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
7
+ License: MIT
8
+ Project-URL: Homepage, https://github.com/oguzhankir/bytesense
9
+ Project-URL: Repository, https://github.com/oguzhankir/bytesense
10
+ Project-URL: Issues, https://github.com/oguzhankir/bytesense/issues
11
+ Project-URL: Changelog, https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md
12
+ Project-URL: Documentation, https://oguzhankir.github.io/bytesense/
13
+ Keywords: encoding,charset,detection,unicode,bytes,chardet,charset-normalizer
14
+ Classifier: Development Status :: 5 - Production/Stable
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3.9
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Programming Language :: Python :: 3.12
24
+ Classifier: Programming Language :: Python :: 3.13
25
+ Classifier: Programming Language :: Python :: 3.14
26
+ Classifier: Programming Language :: Python :: 3 :: Only
27
+ Classifier: Programming Language :: Python :: Implementation :: CPython
28
+ Classifier: Programming Language :: Python :: Implementation :: PyPy
29
+ Classifier: Programming Language :: Rust
30
+ Classifier: Topic :: Text Processing :: Linguistic
31
+ Classifier: Topic :: Utilities
32
+ Classifier: Typing :: Typed
33
+ Requires-Python: >=3.9
34
+ Description-Content-Type: text/markdown
35
+ License-File: LICENSE
36
+ Provides-Extra: fast
37
+ Provides-Extra: docs
38
+ Requires-Dist: mkdocs>=1.6; extra == "docs"
39
+ Requires-Dist: mkdocs-material>=9.5; extra == "docs"
40
+ Requires-Dist: pymdown-extensions>=10.0; extra == "docs"
41
+ Provides-Extra: dev
42
+ Requires-Dist: pytest>=8.0; extra == "dev"
43
+ Requires-Dist: pytest-cov>=5.0; extra == "dev"
44
+ Requires-Dist: pytest-benchmark>=4.0; extra == "dev"
45
+ Requires-Dist: mypy>=1.8; extra == "dev"
46
+ Requires-Dist: ruff>=0.4; extra == "dev"
47
+ Requires-Dist: build>=1.2; extra == "dev"
48
+ Requires-Dist: hypothesis>=6.100; (platform_python_implementation != "PyPy" or python_version >= "3.11") and extra == "dev"
49
+ Requires-Dist: hypothesis==6.100.0; (platform_python_implementation == "PyPy" and python_version < "3.11") and extra == "dev"
50
+ Requires-Dist: setuptools-rust>=1.12; extra == "dev"
51
+ Requires-Dist: charset-normalizer>=3.3; extra == "dev"
52
+ Requires-Dist: chardet>=5.0; extra == "dev"
53
+ Requires-Dist: requests>=2.31; extra == "dev"
54
+ Requires-Dist: certifi>=2023.0; extra == "dev"
55
+ Requires-Dist: twine>=5.0; extra == "dev"
56
+ Dynamic: license-file
57
+
58
+ <p align="center">
59
+ <img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
60
+ </p>
61
+ <p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
62
+ <p align="center">
63
+ <a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
64
+ <a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
65
+ <a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
66
+ </p>
67
+
68
+ **bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
69
+
70
+ Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
71
+
72
+ ```python
73
+ from bytesense import from_bytes
74
+
75
+ data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
76
+ result = from_bytes(data)
77
+ if result.encoding:
78
+ text = data.decode(result.encoding) # strict decoding; no replacement characters
79
+ print(result.encoding, result.bytes_validated, result.complete)
80
+ ```
81
+
82
+ ## Install
83
+
84
+ ```bash
85
+ pip install bytesense
86
+ ```
87
+
88
+ Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
89
+
90
+ ```bash
91
+ BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
92
+ BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
93
+ ```
94
+
95
+ The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
96
+
97
+ ## Built for dependable imports
98
+
99
+ - **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
100
+ - **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
101
+ - **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
102
+ - **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
103
+ - **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
104
+ - **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
105
+
106
+ Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
107
+
108
+ ## Files, streams and existing integrations
109
+
110
+ ```python
111
+ from bytesense import detect, detect_stream, from_path
112
+
113
+ result = from_path("export.csv", include_language=True)
114
+ print(result.to_dict())
115
+
116
+ with open("archive.txt", "rb") as source:
117
+ result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
118
+ assert result.complete
119
+
120
+ # Familiar dictionary interface for code that uses chardet.detect().
121
+ metadata = detect(b"hello world")
122
+ ```
123
+
124
+ `detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
125
+
126
+ ```bash
127
+ bytesense report.csv
128
+ bytesense --language --verbose report.csv
129
+ cat report.csv | bytesense --minimal -
130
+ ```
131
+
132
+ The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
133
+
134
+ ## Measured accuracy and performance
135
+
136
+ The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
137
+
138
+ | Detector | Exact Unicode matches | Accuracy |
139
+ |---|---:|---:|
140
+ | **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
141
+ | bytesense 0.1.2 | 883 / 2078 | 42.49% |
142
+ | chardet 7.6.0 | 2060 / 2078 | 99.13% |
143
+ | charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
144
+ | chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
145
+
146
+ See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
147
+
148
+ ## Upgrading from 0.x
149
+
150
+ - Language reporting is opt-in: `include_language=True`.
151
+ - `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
152
+ - An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
153
+ - BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
154
+ - No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
155
+ - `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
156
+ - Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
157
+
158
+ [Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
159
+
160
+ Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
@@ -0,0 +1,103 @@
1
+ <p align="center">
2
+ <img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
3
+ </p>
4
+ <p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
5
+ <p align="center">
6
+ <a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
7
+ <a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
8
+ <a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
9
+ </p>
10
+
11
+ **bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
12
+
13
+ Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
14
+
15
+ ```python
16
+ from bytesense import from_bytes
17
+
18
+ data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
19
+ result = from_bytes(data)
20
+ if result.encoding:
21
+ text = data.decode(result.encoding) # strict decoding; no replacement characters
22
+ print(result.encoding, result.bytes_validated, result.complete)
23
+ ```
24
+
25
+ ## Install
26
+
27
+ ```bash
28
+ pip install bytesense
29
+ ```
30
+
31
+ Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
32
+
33
+ ```bash
34
+ BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
35
+ BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
36
+ ```
37
+
38
+ The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
39
+
40
+ ## Built for dependable imports
41
+
42
+ - **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
43
+ - **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
44
+ - **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
45
+ - **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
46
+ - **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
47
+ - **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
48
+
49
+ Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
50
+
51
+ ## Files, streams and existing integrations
52
+
53
+ ```python
54
+ from bytesense import detect, detect_stream, from_path
55
+
56
+ result = from_path("export.csv", include_language=True)
57
+ print(result.to_dict())
58
+
59
+ with open("archive.txt", "rb") as source:
60
+ result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
61
+ assert result.complete
62
+
63
+ # Familiar dictionary interface for code that uses chardet.detect().
64
+ metadata = detect(b"hello world")
65
+ ```
66
+
67
+ `detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
68
+
69
+ ```bash
70
+ bytesense report.csv
71
+ bytesense --language --verbose report.csv
72
+ cat report.csv | bytesense --minimal -
73
+ ```
74
+
75
+ The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
76
+
77
+ ## Measured accuracy and performance
78
+
79
+ The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
80
+
81
+ | Detector | Exact Unicode matches | Accuracy |
82
+ |---|---:|---:|
83
+ | **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
84
+ | bytesense 0.1.2 | 883 / 2078 | 42.49% |
85
+ | chardet 7.6.0 | 2060 / 2078 | 99.13% |
86
+ | charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
87
+ | chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
88
+
89
+ See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
90
+
91
+ ## Upgrading from 0.x
92
+
93
+ - Language reporting is opt-in: `include_language=True`.
94
+ - `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
95
+ - An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
96
+ - BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
97
+ - No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
98
+ - `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
99
+ - Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
100
+
101
+ [Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
102
+
103
+ Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
@@ -0,0 +1,118 @@
1
+ {
2
+ "source_repo": "https://github.com/jawah/charset_normalizer",
3
+ "source_path": "data/",
4
+ "reference_test": "tests/test_full_detection.py",
5
+ "notice": "Sample files originate from charset-normalizer data/ (see upstream data/NOTICE.md). Used only for benchmark comparison.",
6
+ "base_url": "https://raw.githubusercontent.com/jawah/charset_normalizer/b130b7dae36658c7ba67381fe06a62c7c7c03894/data/",
7
+ "files": [
8
+ {
9
+ "file": "sample-arabic-1.txt",
10
+ "encoding": "cp1256",
11
+ "sha256": "00c83c71aaf34f6c17ee5ed4e6ad868fe7f281c56d96d5f88f21fc5d39203648",
12
+ "size": 906
13
+ },
14
+ {
15
+ "file": "sample-french-1.txt",
16
+ "encoding": "cp1252",
17
+ "sha256": "6b88988aa8cfd689df08f91a25ae0ea8032cc28b4557092432849a2712fce716",
18
+ "size": 3251
19
+ },
20
+ {
21
+ "file": "sample-arabic.txt",
22
+ "encoding": "utf_8",
23
+ "sha256": "3606d50f8dbd20d80aa4470bd89e74696ceda8ec9138f9524330869eb9aafa46",
24
+ "size": 1637
25
+ },
26
+ {
27
+ "file": "sample-russian-3.txt",
28
+ "encoding": "utf_8",
29
+ "sha256": "81c49a881b7175f68201edc41966016b5c0f16ddb4ac4a30407af0e4ac8b84e6",
30
+ "size": 3070
31
+ },
32
+ {
33
+ "file": "sample-french.txt",
34
+ "encoding": "utf_8",
35
+ "sha256": "ab1b0ebf22b7bd85d2a45600844c0a2c89ba6217b862a6d96b9fa46ce1e132bb",
36
+ "size": 3375
37
+ },
38
+ {
39
+ "file": "sample-chinese.txt",
40
+ "encoding": "big5",
41
+ "sha256": "5f4aa09479ff9366936b242697680f4f0a50dcb3829fb10d855644bc035a7b7a",
42
+ "size": 743
43
+ },
44
+ {
45
+ "file": "sample-greek.txt",
46
+ "encoding": "cp1253",
47
+ "sha256": "cf9afae3a58f28a422fa8edecf255744a8cffb559262d3b3f7dfa3e1d6f59cb4",
48
+ "size": 570
49
+ },
50
+ {
51
+ "file": "sample-greek-2.txt",
52
+ "encoding": "cp1253",
53
+ "sha256": "cf9afae3a58f28a422fa8edecf255744a8cffb559262d3b3f7dfa3e1d6f59cb4",
54
+ "size": 570
55
+ },
56
+ {
57
+ "file": "sample-hebrew-2.txt",
58
+ "encoding": "cp1255",
59
+ "sha256": "eb3a0aa5646487ee6b06c7ef8a9223a89cc2c7b920c787f982ba712d8fedf5f4",
60
+ "size": 340
61
+ },
62
+ {
63
+ "file": "sample-hebrew-3.txt",
64
+ "encoding": "cp1255",
65
+ "sha256": "eb3a0aa5646487ee6b06c7ef8a9223a89cc2c7b920c787f982ba712d8fedf5f4",
66
+ "size": 340
67
+ },
68
+ {
69
+ "file": "sample-bulgarian.txt",
70
+ "encoding": "utf_8",
71
+ "sha256": "d6f347a930569baa7150ab528cfd64420fd9335d89d6617cbed4addec7c4ad30",
72
+ "size": 2198
73
+ },
74
+ {
75
+ "file": "sample-english.bom.txt",
76
+ "encoding": "utf_8",
77
+ "sha256": "4a5850a424c075e25e86fbee489561d5869efdb42297ed08ae074238f312e818",
78
+ "size": 859
79
+ },
80
+ {
81
+ "file": "sample-spanish.txt",
82
+ "encoding": "utf_8",
83
+ "sha256": "b43d16b5398e26565153b1c1118a7f040fa2350493eac5cea51721ffe5367edd",
84
+ "size": 7281
85
+ },
86
+ {
87
+ "file": "sample-korean.txt",
88
+ "encoding": "cp949",
89
+ "sha256": "281246aaaefceb47f6bc0120b2646ebe1832d57101cc0d7bfaeedfbfed979329",
90
+ "size": 387
91
+ },
92
+ {
93
+ "file": "sample-turkish.txt",
94
+ "encoding": "cp1254",
95
+ "sha256": "df265e595aac51e92faafd41afefadbd21b820d1be3eec2e35dab4213fb16fba",
96
+ "size": 1840
97
+ },
98
+ {
99
+ "file": "sample-russian-2.txt",
100
+ "encoding": "utf_8",
101
+ "sha256": "2492ff4b9b15c174a998457ff02233cd1367bdfa5d7c066145f15616aaaa941a",
102
+ "size": 2209
103
+ },
104
+ {
105
+ "file": "sample-russian.txt",
106
+ "encoding": "mac_cyrillic",
107
+ "sha256": "becee0937e700266497ef4cbe84352be0449476a9299a7fc7c18647344e3ff8c",
108
+ "size": 1211
109
+ },
110
+ {
111
+ "file": "sample-polish.txt",
112
+ "encoding": "utf_8",
113
+ "sha256": "fe130e75df06b484e1a00cfa6c7679f2ab2b2c44f9a69780b89e729c651e5fcf",
114
+ "size": 5815
115
+ }
116
+ ],
117
+ "source_commit": "b130b7dae36658c7ba67381fe06a62c7c7c03894"
118
+ }