bytesense 0.1.2__tar.gz → 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bytesense-1.0.0/CHANGELOG.md +83 -0
- bytesense-1.0.0/CODE_OF_CONDUCT.md +83 -0
- bytesense-1.0.0/CONTRIBUTING.md +47 -0
- bytesense-1.0.0/MANIFEST.in +8 -0
- bytesense-1.0.0/PKG-INFO +160 -0
- bytesense-1.0.0/README.md +103 -0
- bytesense-1.0.0/benchmarks/cn_official_manifest.json +118 -0
- bytesense-1.0.0/benchmarks/conftest.py +217 -0
- bytesense-1.0.0/benchmarks/hard_scenarios.py +183 -0
- bytesense-1.0.0/benchmarks/test_bench_detection.py +352 -0
- bytesense-1.0.0/benchmarks/test_hard_scenarios.py +118 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/pyproject.toml +21 -19
- {bytesense-0.1.2 → bytesense-1.0.0}/rust/Cargo.lock +13 -60
- {bytesense-0.1.2 → bytesense-1.0.0}/rust/Cargo.toml +3 -2
- {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/histogram.rs +24 -8
- bytesense-1.0.0/rust/src/language.rs +151 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/lib.rs +3 -1
- {bytesense-0.1.2 → bytesense-1.0.0}/rust/src/utf8.rs +22 -1
- bytesense-1.0.0/scripts/benchmark_v1.py +128 -0
- bytesense-1.0.0/scripts/build_fingerprints.py +105 -0
- bytesense-1.0.0/scripts/check_dist.py +34 -0
- bytesense-1.0.0/scripts/check_version.py +16 -0
- bytesense-1.0.0/scripts/compare_libraries.sh +80 -0
- bytesense-1.0.0/scripts/evaluate_corpus.py +224 -0
- bytesense-1.0.0/scripts/fetch_cn_benchmark_samples.py +34 -0
- bytesense-1.0.0/scripts/run_all_benchmarks.sh +71 -0
- bytesense-1.0.0/setup.cfg +4 -0
- bytesense-1.0.0/setup.py +37 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/__init__.py +2 -1
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/_rust.py +5 -1
- bytesense-1.0.0/src/bytesense/api.py +469 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/candidate.py +3 -10
- bytesense-1.0.0/src/bytesense/cli.py +73 -0
- bytesense-1.0.0/src/bytesense/coherence.py +68 -0
- bytesense-1.0.0/src/bytesense/data/__init__.py +0 -0
- bytesense-1.0.0/src/bytesense/data/language.json.gz +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/fingerprint.py +55 -56
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/hints.py +4 -7
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/mess.py +20 -19
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/models.py +20 -8
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/multi.py +52 -9
- bytesense-1.0.0/src/bytesense/py.typed +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/repair.py +19 -8
- bytesense-1.0.0/src/bytesense/scoring.py +100 -0
- bytesense-1.0.0/src/bytesense/streaming.py +416 -0
- bytesense-1.0.0/src/bytesense/version.py +4 -0
- bytesense-1.0.0/src/bytesense.egg-info/PKG-INFO +160 -0
- bytesense-1.0.0/src/bytesense.egg-info/SOURCES.txt +76 -0
- bytesense-1.0.0/src/bytesense.egg-info/dependency_links.txt +1 -0
- bytesense-1.0.0/src/bytesense.egg-info/entry_points.txt +2 -0
- bytesense-1.0.0/src/bytesense.egg-info/not-zip-safe +1 -0
- bytesense-1.0.0/src/bytesense.egg-info/requires.txt +27 -0
- bytesense-1.0.0/src/bytesense.egg-info/top_level.txt +1 -0
- bytesense-1.0.0/tests/test_api.py +100 -0
- bytesense-1.0.0/tests/test_api_extra.py +35 -0
- bytesense-1.0.0/tests/test_candidate.py +13 -0
- bytesense-1.0.0/tests/test_cli.py +144 -0
- bytesense-1.0.0/tests/test_coherence.py +21 -0
- bytesense-1.0.0/tests/test_fingerprint.py +171 -0
- bytesense-1.0.0/tests/test_heuristics_extra.py +59 -0
- bytesense-1.0.0/tests/test_hints.py +60 -0
- bytesense-1.0.0/tests/test_hints_extra.py +18 -0
- bytesense-1.0.0/tests/test_legacy.py +41 -0
- bytesense-1.0.0/tests/test_mess.py +13 -0
- bytesense-1.0.0/tests/test_mess_extra.py +14 -0
- bytesense-1.0.0/tests/test_multi.py +46 -0
- bytesense-1.0.0/tests/test_repair.py +82 -0
- bytesense-1.0.0/tests/test_rust.py +54 -0
- bytesense-1.0.0/tests/test_rust_layer.py +29 -0
- bytesense-1.0.0/tests/test_streaming.py +54 -0
- bytesense-1.0.0/tests/test_streaming_v2.py +73 -0
- bytesense-1.0.0/tests/test_v1_contracts.py +276 -0
- bytesense-0.1.2/PKG-INFO +0 -341
- bytesense-0.1.2/README.md +0 -287
- bytesense-0.1.2/src/bytesense/api.py +0 -797
- bytesense-0.1.2/src/bytesense/cli.py +0 -50
- bytesense-0.1.2/src/bytesense/coherence.py +0 -93
- bytesense-0.1.2/src/bytesense/streaming.py +0 -288
- bytesense-0.1.2/src/bytesense/version.py +0 -4
- {bytesense-0.1.2 → bytesense-1.0.0}/LICENSE +0 -0
- {bytesense-0.1.2/src/bytesense/data → bytesense-1.0.0/benchmarks}/__init__.py +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/constant.py +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/data/fingerprints.py +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/heuristics.py +0 -0
- {bytesense-0.1.2 → bytesense-1.0.0}/src/bytesense/legacy.py +0 -0
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
## 1.0.0
|
|
5
|
+
|
|
6
|
+
### Detection and correctness
|
|
7
|
+
- Replace the bespoke ranking pipeline with a reproducibly generated character-pair model and optional native scoring.
|
|
8
|
+
- Strictly validate every returned codec against the complete input, with explicit sample/validation counts and completion status.
|
|
9
|
+
- Normalize filters on all paths; an empty isolation list permits no codecs.
|
|
10
|
+
- Use BOM-consuming UTF-16/32 codecs, improve non-Latin Unicode lane detection, and recognize 7-bit shift syntax.
|
|
11
|
+
- Remove fabricated confidence intervals; scores are explicitly uncalibrated. Language reporting is opt-in.
|
|
12
|
+
|
|
13
|
+
### Streams and data handling
|
|
14
|
+
- Consume streams to EOF by default with bounded memory and temporary-file spooling.
|
|
15
|
+
- Preserve late informative samples across chunk boundaries; reject feeding after finalization.
|
|
16
|
+
- Retain every mixed-document byte, including short tails, and preserve Unicode boundaries.
|
|
17
|
+
- Use strict decoding for repair and segments; do not silently replace undecodable bytes.
|
|
18
|
+
- Add CLI stdin support, configuration flags and meaningful exit codes.
|
|
19
|
+
|
|
20
|
+
### Packaging and verification
|
|
21
|
+
- Python 3.9+, portable compiler-free wheel/sdist installs, typed package marker, and CPython abi3 native wheels.
|
|
22
|
+
- Distinct pure/native CI, installed-wheel tests, version checks, pinned/hash-verified corpus inputs and reproducible comparison tools.
|
|
23
|
+
- Consolidate documentation; retain historical details in this changelog.
|
|
24
|
+
|
|
25
|
+
See the README migration notes for intentional 0.x behavior changes.
|
|
26
|
+
|
|
27
|
+
## [0.1.2] — 2025-03-26
|
|
28
|
+
|
|
29
|
+
### Changed
|
|
30
|
+
|
|
31
|
+
- **BOM fast path:** UTF-8 with BOM now reports `encoding="utf_8_sig"` (aligned with `codecs` / `CandidateSelector`), not `utf_8`.
|
|
32
|
+
- **Streaming:** In-band HTML/XML hints are probed on every `feed()` until a hint is found (scan limited to the first 4KB inside `_probe_inband_hint`); shared regex patterns imported from `hints.py`; `codecs` imported at module level.
|
|
33
|
+
- **`detect_multi`:** Adjacent same-encoding merges no longer re-run `from_bytes` on the full merged span; `byte_count` on the shared `DetectionResult` is updated with `dataclasses.replace`.
|
|
34
|
+
- **`repair_bytes`:** Explicit `max_iterations` and `chains` parameters instead of `**kwargs: object`.
|
|
35
|
+
- **Coverage:** `fail_under` raised from 50 to **75** (current suite ~75%; 85% remains a stretch goal with more `api` / fingerprint tests).
|
|
36
|
+
|
|
37
|
+
### Fixed
|
|
38
|
+
|
|
39
|
+
- `DocumentSegment` / `MultiEncodingResult` now use `slots` on Python 3.10+ like other dataclasses in the package.
|
|
40
|
+
|
|
41
|
+
## [0.1.1] — 2025-03-26
|
|
42
|
+
|
|
43
|
+
### Added
|
|
44
|
+
|
|
45
|
+
- `examples/` scripts (basic detection, `detect()`, streaming, repair, HTTP hints, multi-encoding)
|
|
46
|
+
- MkDocs documentation site and GitHub Actions workflow **Docs Pages** → [GitHub Pages](https://oguzhankir.github.io/bytesense/)
|
|
47
|
+
- `[project.optional-dependencies]` group `docs` (`mkdocs`, `mkdocs-material`, `pymdown-extensions`)
|
|
48
|
+
- `project.urls.Documentation` in `pyproject.toml`
|
|
49
|
+
|
|
50
|
+
### Fixed
|
|
51
|
+
|
|
52
|
+
- README logo on PyPI: use `raw.githubusercontent.com` URL (sdist has no `assets/` for the image)
|
|
53
|
+
- Source distribution: include `LICENSE` in maturin `include` so PyPI accepts `License-File` metadata
|
|
54
|
+
|
|
55
|
+
## [0.1.0] — 2025-03-26
|
|
56
|
+
|
|
57
|
+
First published release.
|
|
58
|
+
|
|
59
|
+
### Added
|
|
60
|
+
|
|
61
|
+
- `from_bytes()`, `from_path()`, `from_fp()`, `is_binary()` API
|
|
62
|
+
- `detect()` chardet / charset-normalizer drop-in compatibility
|
|
63
|
+
- `detect_stream()` for iterator-based streaming detection
|
|
64
|
+
- `StreamDetector` for incremental detection; `snapshot()`, `is_stable`, auto-stop when encoding is stable (configurable)
|
|
65
|
+
- In-band hints from HTML meta tags and XML declarations in `StreamDetector`
|
|
66
|
+
- Byte-distribution fingerprinting with pre-computed lookup table
|
|
67
|
+
- `repair()` and `repair_bytes()` for mojibake detection and repair; `is_mojibake()`; `RepairResult`
|
|
68
|
+
- `hint_from_http_headers()`, `hint_from_content()`, `best_hint()` for standalone hint extraction
|
|
69
|
+
- `detect_multi()` for multi-encoding documents; `MultiEncodingResult` and `DocumentSegment`
|
|
70
|
+
- Optional Rust core extension (`pip install "bytesense[fast]"`)
|
|
71
|
+
- CLI: `bytesense <file>`
|
|
72
|
+
- `py.typed` marker for PEP 561 compliance
|
|
73
|
+
- Full type annotations, mypy strict mode
|
|
74
|
+
- GitHub Actions CI across 3 OS × multiple Python versions
|
|
75
|
+
|
|
76
|
+
### Changed
|
|
77
|
+
|
|
78
|
+
- `LANGUAGE_ENCODINGS` aligned with languages in `CHAR_FREQUENCIES`
|
|
79
|
+
- Candidate shortlist tuned for speed while preserving accuracy targets
|
|
80
|
+
|
|
81
|
+
### Fixed
|
|
82
|
+
|
|
83
|
+
- Benchmark and packaging fixes (e.g. UTF-8 BOM test fixture, build backend configuration)
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# Contributor Covenant Code of Conduct
|
|
2
|
+
|
|
3
|
+
## Our pledge
|
|
4
|
+
|
|
5
|
+
We as members, contributors, and leaders pledge to make participation in our community a harassment-free experience for everyone, regardless of age, body size, visible or invisible disability, ethnicity, sex characteristics, gender identity and expression, level of experience, education, socio-economic status, nationality, personal appearance, race, caste, color, religion, or sexual identity and orientation.
|
|
6
|
+
|
|
7
|
+
We pledge to act and interact in ways that contribute to an open, welcoming, diverse, inclusive, and healthy community.
|
|
8
|
+
|
|
9
|
+
## Our standards
|
|
10
|
+
|
|
11
|
+
Examples of behavior that contributes to a positive environment for our community include:
|
|
12
|
+
|
|
13
|
+
- Demonstrating empathy and kindness toward other people
|
|
14
|
+
- Being respectful of differing opinions, viewpoints, and experiences
|
|
15
|
+
- Giving and gracefully accepting constructive feedback
|
|
16
|
+
- Accepting responsibility and apologizing to those affected by our mistakes, and learning from the experience
|
|
17
|
+
- Focusing on what is best not just for us as individuals, but for the overall community
|
|
18
|
+
|
|
19
|
+
Examples of unacceptable behavior include:
|
|
20
|
+
|
|
21
|
+
- The use of sexualized language or imagery, and sexual attention or advances of any kind
|
|
22
|
+
- Trolling, insulting or derogatory comments, and personal or political attacks
|
|
23
|
+
- Public or private harassment
|
|
24
|
+
- Publishing others’ private information, such as a physical or email address, without their explicit permission
|
|
25
|
+
- Other conduct which could reasonably be considered inappropriate in a professional setting
|
|
26
|
+
|
|
27
|
+
## Enforcement responsibilities
|
|
28
|
+
|
|
29
|
+
Community leaders are responsible for clarifying and enforcing our standards of acceptable behavior and will take appropriate and fair corrective action in response to any behavior that they deem inappropriate, threatening, offensive, or harmful.
|
|
30
|
+
|
|
31
|
+
Community leaders have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, and will communicate reasons for moderation decisions when appropriate.
|
|
32
|
+
|
|
33
|
+
## Scope
|
|
34
|
+
|
|
35
|
+
This Code of Conduct applies within all community spaces, and also applies when an individual is officially representing the community in public spaces. Examples of representing our community include using an official email address, posting via an official social media account, or acting as an appointed representative at an online or offline event.
|
|
36
|
+
|
|
37
|
+
## Enforcement
|
|
38
|
+
|
|
39
|
+
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to the community leaders responsible for enforcement at **oguzhankir17@gmail.com**. All complaints will be reviewed and investigated promptly and fairly.
|
|
40
|
+
|
|
41
|
+
All community leaders are obligated to respect the privacy and security of the reporter of any incident.
|
|
42
|
+
|
|
43
|
+
## Enforcement guidelines
|
|
44
|
+
|
|
45
|
+
Community leaders will follow these Community Impact Guidelines in determining the consequences for any action they deem in violation of this Code of Conduct:
|
|
46
|
+
|
|
47
|
+
### 1. Correction
|
|
48
|
+
|
|
49
|
+
**Community Impact**: Use of inappropriate language or other behavior deemed unprofessional or unwelcome in the community.
|
|
50
|
+
|
|
51
|
+
**Consequence**: A private, written warning from community leaders, providing clarity around the nature of the violation and an explanation of why the behavior was inappropriate. A public apology may be requested.
|
|
52
|
+
|
|
53
|
+
### 2. Warning
|
|
54
|
+
|
|
55
|
+
**Community Impact**: A violation through a single incident or series of actions.
|
|
56
|
+
|
|
57
|
+
**Consequence**: A warning with consequences for continued behavior. No interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, for a specified period of time. This includes avoiding interactions in community spaces as well as external channels like social media. Violating these terms may lead to a temporary or permanent ban.
|
|
58
|
+
|
|
59
|
+
### 3. Temporary ban
|
|
60
|
+
|
|
61
|
+
**Community Impact**: A serious violation of community standards, including sustained inappropriate behavior.
|
|
62
|
+
|
|
63
|
+
**Consequence**: A temporary ban from any sort of interaction or public communication with the community for a specified period of time. No public or private interaction with the people involved, including unsolicited interaction with those enforcing the Code of Conduct, is allowed during this period. Violating these terms may lead to a permanent ban.
|
|
64
|
+
|
|
65
|
+
### 4. Permanent ban
|
|
66
|
+
|
|
67
|
+
**Community Impact**: Demonstrating a pattern of violation of community standards, including sustained inappropriate behavior, harassment of an individual, or aggression toward or disparagement of classes of individuals.
|
|
68
|
+
|
|
69
|
+
**Consequence**: A permanent ban from any sort of public interaction within the community.
|
|
70
|
+
|
|
71
|
+
## Attribution
|
|
72
|
+
|
|
73
|
+
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 2.1, available at [https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
|
|
74
|
+
|
|
75
|
+
Community Impact Guidelines were inspired by [Mozilla’s code of conduct enforcement ladder][Mozilla CoC].
|
|
76
|
+
|
|
77
|
+
For answers to common questions about this code of conduct, see the FAQ at [https://www.contributor-covenant.org/faq][FAQ]. Translations are available at [https://www.contributor-covenant.org/translations][translations].
|
|
78
|
+
|
|
79
|
+
[homepage]: https://www.contributor-covenant.org
|
|
80
|
+
[v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
|
|
81
|
+
[Mozilla CoC]: https://github.com/mozilla/diversity
|
|
82
|
+
[FAQ]: https://www.contributor-covenant.org/faq
|
|
83
|
+
[translations]: https://www.contributor-covenant.org/translations
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Follow the [Code of Conduct](CODE_OF_CONDUCT.md). Use a focused `oguzhankir/<topic>` branch and sign commits with `git commit -s` under the project's DCO.
|
|
4
|
+
|
|
5
|
+
## Development
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
python -m venv .venv
|
|
9
|
+
source .venv/bin/activate
|
|
10
|
+
BYTESENSE_BUILD_RUST=0 pip install -e '.[dev,docs]'
|
|
11
|
+
python scripts/fetch_cn_benchmark_samples.py
|
|
12
|
+
ruff check src tests benchmarks scripts setup.py
|
|
13
|
+
mypy src/bytesense
|
|
14
|
+
BYTESENSE_PURE_PYTHON=1 pytest tests --cov=bytesense --cov-branch
|
|
15
|
+
pytest benchmarks/test_bench_detection.py benchmarks/test_hard_scenarios.py --benchmark-disable
|
|
16
|
+
mkdocs build --strict
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
On Windows, set environment variables using your shell's syntax. Native development requires Rust 1.88+:
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
BYTESENSE_BUILD_RUST=1 pip install -e . --no-build-isolation
|
|
23
|
+
BYTESENSE_EXPECT_RUST=1 pytest tests
|
|
24
|
+
cargo test --manifest-path rust/Cargo.toml --locked
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
CI tests pure and native installations separately on Linux, macOS and Windows. Packaging jobs install and test actual wheels, check abi3 compatibility, and test the sdist without compiling Rust. The coverage floor remains 75%; timing measurements are reports, not noisy speed assertions.
|
|
28
|
+
|
|
29
|
+
## Model and benchmarks
|
|
30
|
+
|
|
31
|
+
The [benchmark documentation](docs/benchmarks.md) describes corpus pins, Unicode-based grouping and reproduction commands. Keep development and holdout documents separate. Do not tune the model on the holdout, silently omit missing inputs, compare differently sized payloads, or advertise sample-only latency as full-input validation latency.
|
|
32
|
+
|
|
33
|
+
`language.json.gz` contains aggregate character-pair statistics; no test documents are bundled. The core retains static model metadata, never document text in a global cache. Native and Python paths must agree within numerical tolerance and produce the same rounded detection scores.
|
|
34
|
+
|
|
35
|
+
## Release
|
|
36
|
+
|
|
37
|
+
Keep `pyproject.toml`, `rust/Cargo.toml`, and `src/bytesense/version.py` aligned; `python scripts/check_version.py` checks them. Update the existing changelog and API/migration documentation when behavior changes.
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
BYTESENSE_BUILD_RUST=0 python -m build
|
|
41
|
+
python scripts/check_dist.py dist
|
|
42
|
+
python -m twine check --strict dist/*
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Publishing a GitHub Release triggers the release workflow. It checks the tag, builds and tests portable/native distributions, then uses the configured PyPI trusted publisher environment. A PR does not publish a release.
|
|
46
|
+
|
|
47
|
+
Keep documentation focused: README, quick start, API reference, benchmarks, and the shared changelog. Put rationale and validation detail in the PR instead of creating additional planning documents.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
include LICENSE README.md CHANGELOG.md
|
|
2
|
+
recursive-include rust *.toml *.lock *.rs
|
|
3
|
+
recursive-include src/bytesense *.py py.typed
|
|
4
|
+
prune rust/target
|
|
5
|
+
|
|
6
|
+
include CONTRIBUTING.md CODE_OF_CONDUCT.md
|
|
7
|
+
recursive-include scripts *.py *.sh
|
|
8
|
+
recursive-include benchmarks *.py cn_official_manifest.json
|
bytesense-1.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: bytesense
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Encoding detection with full-input validation, bounded-memory streams, and optional Rust acceleration.
|
|
5
|
+
Author-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
6
|
+
Maintainer-email: Oğuzhan Kır <oguzhankir17@gmail.com>
|
|
7
|
+
License: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/oguzhankir/bytesense
|
|
9
|
+
Project-URL: Repository, https://github.com/oguzhankir/bytesense
|
|
10
|
+
Project-URL: Issues, https://github.com/oguzhankir/bytesense/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md
|
|
12
|
+
Project-URL: Documentation, https://oguzhankir.github.io/bytesense/
|
|
13
|
+
Keywords: encoding,charset,detection,unicode,bytes,chardet,charset-normalizer
|
|
14
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
25
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
26
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
27
|
+
Classifier: Programming Language :: Python :: Implementation :: CPython
|
|
28
|
+
Classifier: Programming Language :: Python :: Implementation :: PyPy
|
|
29
|
+
Classifier: Programming Language :: Rust
|
|
30
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
31
|
+
Classifier: Topic :: Utilities
|
|
32
|
+
Classifier: Typing :: Typed
|
|
33
|
+
Requires-Python: >=3.9
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
License-File: LICENSE
|
|
36
|
+
Provides-Extra: fast
|
|
37
|
+
Provides-Extra: docs
|
|
38
|
+
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
39
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "docs"
|
|
40
|
+
Requires-Dist: pymdown-extensions>=10.0; extra == "docs"
|
|
41
|
+
Provides-Extra: dev
|
|
42
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
43
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest-benchmark>=4.0; extra == "dev"
|
|
45
|
+
Requires-Dist: mypy>=1.8; extra == "dev"
|
|
46
|
+
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
47
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
48
|
+
Requires-Dist: hypothesis>=6.100; (platform_python_implementation != "PyPy" or python_version >= "3.11") and extra == "dev"
|
|
49
|
+
Requires-Dist: hypothesis==6.100.0; (platform_python_implementation == "PyPy" and python_version < "3.11") and extra == "dev"
|
|
50
|
+
Requires-Dist: setuptools-rust>=1.12; extra == "dev"
|
|
51
|
+
Requires-Dist: charset-normalizer>=3.3; extra == "dev"
|
|
52
|
+
Requires-Dist: chardet>=5.0; extra == "dev"
|
|
53
|
+
Requires-Dist: requests>=2.31; extra == "dev"
|
|
54
|
+
Requires-Dist: certifi>=2023.0; extra == "dev"
|
|
55
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
56
|
+
Dynamic: license-file
|
|
57
|
+
|
|
58
|
+
<p align="center">
|
|
59
|
+
<img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
|
|
60
|
+
</p>
|
|
61
|
+
<p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
|
|
62
|
+
<p align="center">
|
|
63
|
+
<a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
|
|
64
|
+
<a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
65
|
+
<a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
|
|
66
|
+
</p>
|
|
67
|
+
|
|
68
|
+
**bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
|
|
69
|
+
|
|
70
|
+
Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from bytesense import from_bytes
|
|
74
|
+
|
|
75
|
+
data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
|
|
76
|
+
result = from_bytes(data)
|
|
77
|
+
if result.encoding:
|
|
78
|
+
text = data.decode(result.encoding) # strict decoding; no replacement characters
|
|
79
|
+
print(result.encoding, result.bytes_validated, result.complete)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Install
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install bytesense
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
|
|
92
|
+
BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
|
|
96
|
+
|
|
97
|
+
## Built for dependable imports
|
|
98
|
+
|
|
99
|
+
- **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
|
|
100
|
+
- **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
|
|
101
|
+
- **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
|
|
102
|
+
- **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
|
|
103
|
+
- **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
|
|
104
|
+
- **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
|
|
105
|
+
|
|
106
|
+
Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
|
|
107
|
+
|
|
108
|
+
## Files, streams and existing integrations
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from bytesense import detect, detect_stream, from_path
|
|
112
|
+
|
|
113
|
+
result = from_path("export.csv", include_language=True)
|
|
114
|
+
print(result.to_dict())
|
|
115
|
+
|
|
116
|
+
with open("archive.txt", "rb") as source:
|
|
117
|
+
result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
|
|
118
|
+
assert result.complete
|
|
119
|
+
|
|
120
|
+
# Familiar dictionary interface for code that uses chardet.detect().
|
|
121
|
+
metadata = detect(b"hello world")
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
bytesense report.csv
|
|
128
|
+
bytesense --language --verbose report.csv
|
|
129
|
+
cat report.csv | bytesense --minimal -
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
133
|
+
|
|
134
|
+
## Measured accuracy and performance
|
|
135
|
+
|
|
136
|
+
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
137
|
+
|
|
138
|
+
| Detector | Exact Unicode matches | Accuracy |
|
|
139
|
+
|---|---:|---:|
|
|
140
|
+
| **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
|
|
141
|
+
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
142
|
+
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
143
|
+
| charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
|
|
144
|
+
| chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
|
|
145
|
+
|
|
146
|
+
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
147
|
+
|
|
148
|
+
## Upgrading from 0.x
|
|
149
|
+
|
|
150
|
+
- Language reporting is opt-in: `include_language=True`.
|
|
151
|
+
- `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
|
|
152
|
+
- An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
|
|
153
|
+
- BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
|
|
154
|
+
- No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
|
|
155
|
+
- `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
|
|
156
|
+
- Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
|
|
157
|
+
|
|
158
|
+
[Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
|
|
159
|
+
|
|
160
|
+
Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="https://raw.githubusercontent.com/oguzhankir/bytesense/main/assets/bytesense_logo.svg" alt="bytesense" width="480" />
|
|
3
|
+
</p>
|
|
4
|
+
<p align="center"><strong>Detect the encoding. Validate every byte. Keep your data intact.</strong></p>
|
|
5
|
+
<p align="center">
|
|
6
|
+
<a href="https://pypi.org/project/bytesense/"><img src="https://img.shields.io/pypi/v/bytesense.svg" alt="PyPI" /></a>
|
|
7
|
+
<a href="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml"><img src="https://github.com/oguzhankir/bytesense/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
8
|
+
<a href="https://github.com/oguzhankir/bytesense/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT license" /></a>
|
|
9
|
+
</p>
|
|
10
|
+
|
|
11
|
+
**bytesense** is a Python encoding detector for file imports, multilingual text pipelines and legacy data. It combines fast Unicode checks, compact character-pair statistics and optional Rust acceleration—with **zero runtime dependencies**.
|
|
12
|
+
|
|
13
|
+
Every returned encoding strictly decodes the complete supplied input. Linguistic scoring uses a bounded sample; stream inputs spill to a temporary file after a configurable memory limit. Results tell you what was examined, what was validated, and whether the input ended.
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from bytesense import from_bytes
|
|
17
|
+
|
|
18
|
+
data = "İstanbul'da çalışan mühendisler için doğru metin çözümleme. ".encode("cp1254")
|
|
19
|
+
result = from_bytes(data)
|
|
20
|
+
if result.encoding:
|
|
21
|
+
text = data.decode(result.encoding) # strict decoding; no replacement characters
|
|
22
|
+
print(result.encoding, result.bytes_validated, result.complete)
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Install
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install bytesense
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Python **3.9+**. Platform wheels include Rust acceleration; the portable wheel works without a compiler. Source installations automatically use Rust when a toolchain is available. To explicitly choose a source build:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
BYTESENSE_BUILD_RUST=0 pip install --no-binary=bytesense bytesense # pure Python
|
|
35
|
+
BYTESENSE_BUILD_RUST=1 pip install --no-binary=bytesense bytesense # require Rust
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
The old `[fast]` extra is a compatibility alias. It does not install an accelerator separately. `BYTESENSE_PURE_PYTHON=1` selects the Python implementation at runtime.
|
|
39
|
+
|
|
40
|
+
## Built for dependable imports
|
|
41
|
+
|
|
42
|
+
- **Complete validation:** a valid prefix cannot hide an undecodable suffix. Unknown and binary inputs have explicit outcomes.
|
|
43
|
+
- **Bounded streams:** feed small chunks or large ones; finalization accounts for every accepted byte. Preview stability never silently ends a stream.
|
|
44
|
+
- **Useful controls:** codec aliases, inclusion/exclusion filters, encoding hints, optional language estimates and a minimum evidence score.
|
|
45
|
+
- **Transparent results:** `why`, alternatives, sample/validation counts and completion status. Confidence is a heuristic score; no fabricated statistical intervals.
|
|
46
|
+
- **Portable acceleration:** native character-pair scoring and byte histograms, with matching Python behavior and typed public APIs.
|
|
47
|
+
- **Data repair tools:** opt-in mojibake repair and byte-preserving mixed-document segmentation, with strict decoding.
|
|
48
|
+
|
|
49
|
+
Decodability alone cannot prove the original encoding. Several legacy codecs can decode identical bytes into different, plausible text. Use a known encoding or an explicit hint when you have one, and evaluate representative data before switching detectors.
|
|
50
|
+
|
|
51
|
+
## Files, streams and existing integrations
|
|
52
|
+
|
|
53
|
+
```python
|
|
54
|
+
from bytesense import detect, detect_stream, from_path
|
|
55
|
+
|
|
56
|
+
result = from_path("export.csv", include_language=True)
|
|
57
|
+
print(result.to_dict())
|
|
58
|
+
|
|
59
|
+
with open("archive.txt", "rb") as source:
|
|
60
|
+
result = detect_stream(iter(lambda: source.read(64 * 1024), b""))
|
|
61
|
+
assert result.complete
|
|
62
|
+
|
|
63
|
+
# Familiar dictionary interface for code that uses chardet.detect().
|
|
64
|
+
metadata = detect(b"hello world")
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`detect_stream()` consumes to EOF by default. `max_bytes=...` or `early_stop=True` is an explicit sampling choice and returns `complete=False` when it stops early. Streaming uses temporary disk space beyond `memory_limit` (1 MiB by default); close an unfinished `StreamDetector` or use it as a context manager.
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
bytesense report.csv
|
|
71
|
+
bytesense --language --verbose report.csv
|
|
72
|
+
cat report.csv | bytesense --minimal -
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
The CLI exits **0** when every input has an encoding, **1** for read failures or unknown/binary inputs, and **2** for invalid arguments. JSON goes to stdout; errors go to stderr.
|
|
76
|
+
|
|
77
|
+
## Measured accuracy and performance
|
|
78
|
+
|
|
79
|
+
The v1 engine replaces the previous collection of special-case ranking rules with a small, reproducibly generated statistical model. The evaluation separates development documents from held-out documents, grouping identical decoded text across encodings. Accuracy means **exact Unicode equality**, not just a compatible codec name.
|
|
80
|
+
|
|
81
|
+
| Detector | Exact Unicode matches | Accuracy |
|
|
82
|
+
|---|---:|---:|
|
|
83
|
+
| **bytesense 1.0.0** | **1906 / 2078** | **91.72%** |
|
|
84
|
+
| bytesense 0.1.2 | 883 / 2078 | 42.49% |
|
|
85
|
+
| chardet 7.6.0 | 2060 / 2078 | 99.13% |
|
|
86
|
+
| charset-normalizer 3.5.1 | 1679 / 2078 | 80.80% |
|
|
87
|
+
| chardetng-py 0.3.5 | 776 / 2078 | 37.34% |
|
|
88
|
+
|
|
89
|
+
See [benchmarks and methodology](https://oguzhankir.github.io/bytesense/benchmarks/) for measured comparisons with chardet, charset-normalizer and chardetng, including the cases where another detector wins. Timing reports distinguish default behavior from an additional full-decode validation step.
|
|
90
|
+
|
|
91
|
+
## Upgrading from 0.x
|
|
92
|
+
|
|
93
|
+
- Language reporting is opt-in: `include_language=True`.
|
|
94
|
+
- `confidence_interval` remains available and is `None`; scores are not calibrated probabilities.
|
|
95
|
+
- An empty `cp_isolation=[]` allows no codecs. Filters apply to every detection path.
|
|
96
|
+
- BOM-bearing UTF-16/32 selects `utf_16`/`utf_32`, which consume the signature when decoding.
|
|
97
|
+
- No unvalidated UTF-8 fallback; stream finalization and repair decoding reject silent data loss.
|
|
98
|
+
- `StreamDetector.feed()` after finalization raises. `reset()` starts a new stream.
|
|
99
|
+
- Python 3.8 is no longer supported. `steps` and `chunk_size` remain accepted compatibility arguments; use `sample_size` to control statistical work.
|
|
100
|
+
|
|
101
|
+
[Quick start](https://oguzhankir.github.io/bytesense/quickstart/) · [API](https://oguzhankir.github.io/bytesense/api/) · [Changes](https://github.com/oguzhankir/bytesense/blob/main/CHANGELOG.md) · [Contributing](https://github.com/oguzhankir/bytesense/blob/main/CONTRIBUTING.md)
|
|
102
|
+
|
|
103
|
+
Created by [Oğuzhan Kır](https://github.com/oguzhankir). MIT-licensed code. The aggregate model's data provenance and reproduction steps are documented with the benchmarks; upstream test documents remain copyrighted by their respective publishers.
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
{
|
|
2
|
+
"source_repo": "https://github.com/jawah/charset_normalizer",
|
|
3
|
+
"source_path": "data/",
|
|
4
|
+
"reference_test": "tests/test_full_detection.py",
|
|
5
|
+
"notice": "Sample files originate from charset-normalizer data/ (see upstream data/NOTICE.md). Used only for benchmark comparison.",
|
|
6
|
+
"base_url": "https://raw.githubusercontent.com/jawah/charset_normalizer/b130b7dae36658c7ba67381fe06a62c7c7c03894/data/",
|
|
7
|
+
"files": [
|
|
8
|
+
{
|
|
9
|
+
"file": "sample-arabic-1.txt",
|
|
10
|
+
"encoding": "cp1256",
|
|
11
|
+
"sha256": "00c83c71aaf34f6c17ee5ed4e6ad868fe7f281c56d96d5f88f21fc5d39203648",
|
|
12
|
+
"size": 906
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
"file": "sample-french-1.txt",
|
|
16
|
+
"encoding": "cp1252",
|
|
17
|
+
"sha256": "6b88988aa8cfd689df08f91a25ae0ea8032cc28b4557092432849a2712fce716",
|
|
18
|
+
"size": 3251
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"file": "sample-arabic.txt",
|
|
22
|
+
"encoding": "utf_8",
|
|
23
|
+
"sha256": "3606d50f8dbd20d80aa4470bd89e74696ceda8ec9138f9524330869eb9aafa46",
|
|
24
|
+
"size": 1637
|
|
25
|
+
},
|
|
26
|
+
{
|
|
27
|
+
"file": "sample-russian-3.txt",
|
|
28
|
+
"encoding": "utf_8",
|
|
29
|
+
"sha256": "81c49a881b7175f68201edc41966016b5c0f16ddb4ac4a30407af0e4ac8b84e6",
|
|
30
|
+
"size": 3070
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"file": "sample-french.txt",
|
|
34
|
+
"encoding": "utf_8",
|
|
35
|
+
"sha256": "ab1b0ebf22b7bd85d2a45600844c0a2c89ba6217b862a6d96b9fa46ce1e132bb",
|
|
36
|
+
"size": 3375
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"file": "sample-chinese.txt",
|
|
40
|
+
"encoding": "big5",
|
|
41
|
+
"sha256": "5f4aa09479ff9366936b242697680f4f0a50dcb3829fb10d855644bc035a7b7a",
|
|
42
|
+
"size": 743
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"file": "sample-greek.txt",
|
|
46
|
+
"encoding": "cp1253",
|
|
47
|
+
"sha256": "cf9afae3a58f28a422fa8edecf255744a8cffb559262d3b3f7dfa3e1d6f59cb4",
|
|
48
|
+
"size": 570
|
|
49
|
+
},
|
|
50
|
+
{
|
|
51
|
+
"file": "sample-greek-2.txt",
|
|
52
|
+
"encoding": "cp1253",
|
|
53
|
+
"sha256": "cf9afae3a58f28a422fa8edecf255744a8cffb559262d3b3f7dfa3e1d6f59cb4",
|
|
54
|
+
"size": 570
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"file": "sample-hebrew-2.txt",
|
|
58
|
+
"encoding": "cp1255",
|
|
59
|
+
"sha256": "eb3a0aa5646487ee6b06c7ef8a9223a89cc2c7b920c787f982ba712d8fedf5f4",
|
|
60
|
+
"size": 340
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
"file": "sample-hebrew-3.txt",
|
|
64
|
+
"encoding": "cp1255",
|
|
65
|
+
"sha256": "eb3a0aa5646487ee6b06c7ef8a9223a89cc2c7b920c787f982ba712d8fedf5f4",
|
|
66
|
+
"size": 340
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"file": "sample-bulgarian.txt",
|
|
70
|
+
"encoding": "utf_8",
|
|
71
|
+
"sha256": "d6f347a930569baa7150ab528cfd64420fd9335d89d6617cbed4addec7c4ad30",
|
|
72
|
+
"size": 2198
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
"file": "sample-english.bom.txt",
|
|
76
|
+
"encoding": "utf_8",
|
|
77
|
+
"sha256": "4a5850a424c075e25e86fbee489561d5869efdb42297ed08ae074238f312e818",
|
|
78
|
+
"size": 859
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"file": "sample-spanish.txt",
|
|
82
|
+
"encoding": "utf_8",
|
|
83
|
+
"sha256": "b43d16b5398e26565153b1c1118a7f040fa2350493eac5cea51721ffe5367edd",
|
|
84
|
+
"size": 7281
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"file": "sample-korean.txt",
|
|
88
|
+
"encoding": "cp949",
|
|
89
|
+
"sha256": "281246aaaefceb47f6bc0120b2646ebe1832d57101cc0d7bfaeedfbfed979329",
|
|
90
|
+
"size": 387
|
|
91
|
+
},
|
|
92
|
+
{
|
|
93
|
+
"file": "sample-turkish.txt",
|
|
94
|
+
"encoding": "cp1254",
|
|
95
|
+
"sha256": "df265e595aac51e92faafd41afefadbd21b820d1be3eec2e35dab4213fb16fba",
|
|
96
|
+
"size": 1840
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
"file": "sample-russian-2.txt",
|
|
100
|
+
"encoding": "utf_8",
|
|
101
|
+
"sha256": "2492ff4b9b15c174a998457ff02233cd1367bdfa5d7c066145f15616aaaa941a",
|
|
102
|
+
"size": 2209
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"file": "sample-russian.txt",
|
|
106
|
+
"encoding": "mac_cyrillic",
|
|
107
|
+
"sha256": "becee0937e700266497ef4cbe84352be0449476a9299a7fc7c18647344e3ff8c",
|
|
108
|
+
"size": 1211
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
"file": "sample-polish.txt",
|
|
112
|
+
"encoding": "utf_8",
|
|
113
|
+
"sha256": "fe130e75df06b484e1a00cfa6c7679f2ab2b2c44f9a69780b89e729c651e5fcf",
|
|
114
|
+
"size": 5815
|
|
115
|
+
}
|
|
116
|
+
],
|
|
117
|
+
"source_commit": "b130b7dae36658c7ba67381fe06a62c7c7c03894"
|
|
118
|
+
}
|