deepdiff-rs 0.10.0__tar.gz → 0.11.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/Cargo.lock +7 -4
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/Cargo.toml +1 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/PKG-INFO +4 -4
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/README.md +3 -3
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/Cargo.toml +5 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/mod.rs +1 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/scalar.rs +5 -16
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/tests.rs +1 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/distance.rs +3 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/fxhash.rs +11 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/hash.rs +31 -35
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/memo.rs +14 -3
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/tests.rs +34 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/lcs.rs +30 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/lcs_tests.rs +45 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/path.rs +5 -2
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/value.rs +139 -47
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/value_tests.rs +130 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/tests/golden.rs +105 -3
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/Cargo.toml +3 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/convert.rs +74 -16
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/guard.rs +36 -24
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/conftest.py +7 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_conversions.py +165 -9
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_datetimes.py +4 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_depth_guard.py +2 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_differential_fuzz.py +68 -23
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_golden_parity.py +59 -2
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_non_finite.py +5 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_sets.py +13 -2
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_signed_zero.py +4 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_stub_signatures.py +7 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_table_diff.py +2 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_timedeltas.py +4 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_times.py +4 -1
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_tuples.py +4 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/Cargo.toml +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/examples/row_diff_rss.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/error.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/json_rows.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/lib.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/options.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/row_diff.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/schema.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-arrow/src/table_diff.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/datetime.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/datetime_tests.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/array.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/dispatch.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/object.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/options.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/diff/set.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/error.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/mod.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/ignore_order/pairing.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/lib.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/report.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/report_tests.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/test_support.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/unified_diff.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/src/unified_diff_tests.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/tests/memory_footprint.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/tests/proptest_diff.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/.python-version +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/benchmarks/bench_bindings.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/deepdiff_rs.pyi +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/arrow.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/deepdiff.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/errors.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/fast_path.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/src/lib.rs +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_bindings_memory.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_smoke.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_stub_mypy.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_suite_hygiene.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_table_row_diff.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/crates/onix-py/tests/test_wheel_contents.py +0 -0
- {deepdiff_rs-0.10.0 → deepdiff_rs-0.11.0}/pyproject.toml +0 -0
|
@@ -629,7 +629,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
|
629
629
|
|
|
630
630
|
[[package]]
|
|
631
631
|
name = "onix-arrow"
|
|
632
|
-
version = "0.
|
|
632
|
+
version = "0.11.0"
|
|
633
633
|
dependencies = [
|
|
634
634
|
"arrow-array",
|
|
635
635
|
"arrow-buffer",
|
|
@@ -646,7 +646,7 @@ dependencies = [
|
|
|
646
646
|
|
|
647
647
|
[[package]]
|
|
648
648
|
name = "onix-cli"
|
|
649
|
-
version = "0.
|
|
649
|
+
version = "0.11.0"
|
|
650
650
|
dependencies = [
|
|
651
651
|
"onix-core",
|
|
652
652
|
"serde_json",
|
|
@@ -654,8 +654,10 @@ dependencies = [
|
|
|
654
654
|
|
|
655
655
|
[[package]]
|
|
656
656
|
name = "onix-core"
|
|
657
|
-
version = "0.
|
|
657
|
+
version = "0.11.0"
|
|
658
658
|
dependencies = [
|
|
659
|
+
"num-bigint",
|
|
660
|
+
"num-traits",
|
|
659
661
|
"proptest",
|
|
660
662
|
"serde",
|
|
661
663
|
"serde_json",
|
|
@@ -665,11 +667,12 @@ dependencies = [
|
|
|
665
667
|
|
|
666
668
|
[[package]]
|
|
667
669
|
name = "onix-py"
|
|
668
|
-
version = "0.
|
|
670
|
+
version = "0.11.0"
|
|
669
671
|
dependencies = [
|
|
670
672
|
"arrow-array",
|
|
671
673
|
"arrow-ipc",
|
|
672
674
|
"arrow-schema",
|
|
675
|
+
"num-bigint",
|
|
673
676
|
"onix-arrow",
|
|
674
677
|
"onix-core",
|
|
675
678
|
"pyo3",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepdiff-rs
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.11.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Rust
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -195,7 +195,7 @@ The engine's own diff-only time and peak resident memory against pinned `deepdif
|
|
|
195
195
|
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
196
196
|
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
197
197
|
|
|
198
|
-
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on
|
|
198
|
+
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared, still trails by a much larger multiple, see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
199
199
|
|
|
200
200
|
## Reference
|
|
201
201
|
|
|
@@ -236,7 +236,7 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
236
236
|
## Known limitations
|
|
237
237
|
|
|
238
238
|
- Only the core diff is implemented: `exclude_paths`, `significant_digits`, custom operators, `verbose_level != 2`, and delta/patch are not (yet) supported.
|
|
239
|
-
- Supported value types are `None`, `bool`, `int`, `float` (`NaN`/`Infinity`/`-Infinity` included), `str`, `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, and `datetime.timedelta`; a `set`/`frozenset` member may be any of these except a `list`, `dict` or `set`, matching Python's own hashability rule, transitively through whatever the member nests, and a `dict` key may be `str`, `None`, `bool`, `int`, `float`, `datetime.datetime`, `datetime.date`, or a `tuple` of those (never nested), or a `tuple`/`datetime`/`date` subclass key (`namedtuple` included), accepted and matched against its base-type value the same way — see [Known limitations](#known-limitations)'s subclass bullet. `int`
|
|
239
|
+
- Supported value types are `None`, `bool`, `int`, `float` (`NaN`/`Infinity`/`-Infinity` included), `str`, `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, and `datetime.timedelta`; a `set`/`frozenset` member may be any of these except a `list`, `dict` or `set`, matching Python's own hashability rule, transitively through whatever the member nests, and a `dict` key may be `str`, `None`, `bool`, `int`, `float`, `datetime.datetime`, `datetime.date`, or a `tuple` of those (never nested), or a `tuple`/`datetime`/`date` subclass key (`namedtuple` included), accepted and matched against its base-type value the same way — see [Known limitations](#known-limitations)'s subclass bullet. An `int` of any magnitude converts and keeps its exact value (an arbitrary-precision integer beyond `i64`/`u64` is compared, ordered, hashed and rendered by its full digits), while a custom object raises `TypeError` naming the exact path it was found at. Under `ignore_order`, comparing an integer beyond `f64::MAX` (about `2**1024`) reports the change deterministically where real DeepDiff raises `OverflowError` (its distance function calls `float()` on it unguarded), a documented crash-class divergence like the datetime one below (see [`tests/golden/README.md`](tests/golden/README.md)). Reading JSON *text* (`diff_json` and the CLI, not the Python-object path), an integer beyond `i64`/`u64` parses as the nearest `float`, so two documents whose integers differ only past that range compare **equal** and diff to `{}` — a limitation of the JSON number reader, not the value model, tracked in [#92](https://github.com/ksco92/onix/issues/92). A non-`str` key's path renders via Python's own `repr()`, except a `tuple` key, which splits into one bracket group per element (`root[1][2]` for `(1, 2)`, never `root[(1, 2)]`, with one deliberate exception for a real DeepDiff bug on the empty tuple, and another on a non-finite-float key — see [`tests/golden/README.md`](tests/golden/README.md)). A nested dict value's own `bool`/`None`/`int`/`float` key stringifies the way `json.dumps` does, but its `datetime`/`date`/`tuple` key renders that same `repr()` text where DeepDiff's own `to_json()` raises `TypeError` on one — a superset, not a difference in the findings (see [`tests/golden/README.md`](tests/golden/README.md)). `1`/`1.0`/`True` match as the same key between two dicts (Python `dict`/`set` equality). The **Datetimes**, **Sets** and **Non-finite floats** bullets below cover the deliberate divergences for those types. See [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs) and [`tests/golden/README.md`](tests/golden/README.md).
|
|
240
240
|
- A subclass of a supported `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, or `datetime.timedelta` (including `namedtuple` and pandas' `Timestamp`) converts and compares as its base type but carries its own class name into a `type_changes` entry, matching DeepDiff's `type(obj).__name__` reporting, with three exceptions: a calendar-type (`datetime`/`date`/`time`/`timedelta`) subclass held as a `set`/`frozenset` member compares as its base with no `type_changes`, matching DeepDiff, since set membership has no pairwise comparison to report one against; a `tuple`/`frozenset` subclass, including a `namedtuple`, is not accepted as a `set`/`frozenset` member at all and raises `TypeError` where DeepDiff accepts and compares it by value (a documented divergence); and `namedtuple` diffs positionally rather than by field (also documented). A `type_changes` entry's `old_type`/`new_type` are type *names* in `to_dict()`, where DeepDiff returns the type objects. See [`tests/golden/README.md`](tests/golden/README.md).
|
|
241
241
|
- **Datetimes** compare by instant, with a naive value read as UTC, matching DeepDiff. A changed pair is reported normalized to UTC (`to_json()` renders `...+00:00`, `to_dict()` returns UTC-aware `datetime`s); everywhere else a datetime keeps its raw value. Three deliberate departures: `to_json()` renders a `date` as `YYYY-MM-DD` where DeepDiff's own `to_json()` raises `TypeError` (a documented superset); a `zoneinfo`/`pytz` tzinfo comes back from `to_dict()` as a fixed-offset `datetime.timezone` carrying the offset it was in force at, not the original zone object; and a set holding both a naive and an aware value at one instant reports both as members, where DeepDiff's own digest cache can report only one (see [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs)). Comparing two datetimes whose UTC form would leave year 1..=9999 raises `ValueError` naming the path, where DeepDiff raises `OverflowError`; under `ignore_order` DeepDiff's hasher normalizes every datetime and so raises for such a value even when it is only added, removed, or shuffled, where onix hashes by instant and reports it normally (see [`tests/golden/README.md`](tests/golden/README.md)). `truncate_datetime` is not supported; the normalized-versus-raw split is documented there too. A `time`/`timedelta`, unlike a datetime, is never normalized for report (DeepDiff compares `time`/`date`/`timedelta` with a plain `!=`), and a naive `time` is never equal to an aware one; `to_json()` renders a `time` as `time.isoformat()`'s bytes and a `timedelta` as `str(timedelta)`'s, both supersets. Under `ignore_order`, `DeepHash` hashes a `time` by whole seconds-of-day only — dropping the microsecond and any offset, a confirmed upstream quirk — while a `timedelta` hashes exactly; see [`tests/golden/README.md`](tests/golden/README.md)'s "Known DeepDiff quirks" section.
|
|
242
242
|
- **Sets** are diffed deterministically, where DeepDiff's own answers depend on the order the running process happens to iterate a set in (hash order, and `PYTHONHASHSEED`-dependent for `str` members) or on how its digest cache/computation handles a tuple, frozenset, or calendar member independently of Python's own `==`. Each consequence — entry order, which member of an equality class is reported, set-versus-sequence coercion, and a tuple/frozenset member's own (positional, not order-/repetition-insensitive) matching rule — is shown with both tools' output in [`tests/golden/README.md`](tests/golden/README.md)'s "Set iteration order" section. A report holding a `frozenset` value also serializes to JSON here, where DeepDiff's own `to_json()` raises `TypeError` — a superset, not a difference in the findings.
|
|
@@ -250,7 +250,7 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
250
250
|
- `diff_tables` inherits some Arrow-interchange quirks: a column name with an embedded NUL byte (`\0`) arrives truncated at the NUL through the C Data Interface (the report shows the truncated name; rare in practice); a list of structs named exactly `key`/`value` with a nullable key is not distinguished from a real map, so a migration between the two is not reported as a type change (polars exports both as the same Arrow type — see `map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs)); and a polars all-null (`Null`-typed) column fails at Arrow C import with a `ValueError` (`the datatype "Null" doesn't expect buffer at index 0`), so give such a column a concrete type first (a pyarrow all-`None` column, inferred as `null`, works and compares as all-null).
|
|
251
251
|
- In `diff_tables`, DuckDB labels a `TIMESTAMP WITH TIME ZONE` column with the connection's *session* time zone when it exports to Arrow (a UTC session as `Timestamp(µs, "UTC")`, an `America/New_York` session as `Timestamp(µs, "America/New_York")`), so on a non-UTC machine such a column can be reported as a type change against a UTC column from another library. Run `SET TimeZone='UTC'` on the DuckDB connection first for a deterministic, machine-independent result.
|
|
252
252
|
- `diff_tables` refuses a column whose Arrow type is nested deeper than `MAX_NESTING_DEPTH` (128) with a `MaxDepthError`, because comparing arbitrarily deep nesting would overflow the native stack; 128 is far beyond any real schema. Importing a schema nested many thousands of levels deep is also slow regardless, a cost of the Arrow C Data Interface itself.
|
|
253
|
-
- `diff_tables`'s row diff costs, per call: RAM for a 32-byte hash per row per side (sorted), so peak memory is linear in row count (about 75 MB at 1M rows/side, 660 MB at 10M); the diff hashes and classifies across all cores by default, running single-threaded only when both sides stay under 50,000 rows and under 64 MB of decoded data (either bound reached first runs the parallel path), and appends the per-row hashes into shared per-partition buffers so there is no separate combined copy; the parallel path adds only the in-flight batches (worker count times batch size) plus the shared buffers' reallocation slack, measured at the linear shape as a peak of about 629 MB single-threaded and 857 MB at 18 threads at 8M rows/side (+228 MB) and about 2.9 GB and 3.6 GB at 37M rows/side (+685 MB) — roughly 10-15 bytes per row per side, under one extra copy of the 32-byte hash vectors, and within the run-to-run slack the single-threaded peak itself shows (2.9-4.2 GB at 37M); `threads=1` runs the sequential path; the size gate peeks each side only up to 50,000 rows or 64 MB of decoded data before choosing, reading the left side first; the peek holds at most 64 MB plus one producer batch per side — a streamed 49,999-row/side pair of 8 KB cells in 100-row batches peaks at about 78 MB at 18 threads and 14 MB single-threaded, rising to about 1.6 GB when the whole side arrives as one batch (see the size-gate peek table in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)); wall time is `N·log N` work spread across the workers, plus a second term for the duplicate-key report, which holds the key values of every *distinct duplicated* key — an all-duplicate 200k-rows/side table (100k distinct duplicated keys) peaks at about 37 MB with 16-byte keys and 1.05 GB with 1 KB keys, 1M rows/side (500k distinct) at about 165 MB with 16-byte keys; a third term for the per-cell diff, which re-reads both inputs, holds the changed rows of both sides, and renders every changed cell to a string held once in the output — so its cost is the number of changed cells times the cell width, not the changed-row count alone (every value is rendered in full). Measured, same method: with two narrow `int64` columns, 85 MB at 1M rows/side with 2% changed and 629 MB with every row changed; with a 1 KB `string` cell changed on every row, 1.18 GB at 100k rows/side and 2.34 GB at 200k (about 11.6 KB per changed row, linear, so on the order of 11.6 GB at 1M). Then temp disk, because each input is re-read several times and so is spooled to an anonymous file (`tempfile`: unlinked at once, mode 0600, no predictable name — nothing is left on disk even on abnormal exit), both spools resident at once, peaking at the decoded size of both inputs (about 315 MB — 161 + 154 MB uncompressed Arrow IPC — for the 1M-row fixture pair; on the order of 10 GB for the full 5 GB-per-side fixture pair, which on Linux may be a RAM-backed `tmpfs`), a full temp filesystem raising `ValueError` naming `TMPDIR`. None of these has a built-in cap: for untrusted input, bound the row count, the changed fraction, the column widths (key columns for duplicate-heavy data, changed cells — rendered in full — for wide value columns, and any column's width for the size-gate peek, which buffers decoded cells whether or not they change), and the producer's batch size, since the peek holds up to one whole batch per side beyond the 64 MB bound. Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per generated batch from `ROW_DIFF_BATCH`), single run, macOS on an Apple M-series laptop, 2026-09-06, same method as [Performance](#performance). See [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
253
|
+
- `diff_tables`'s row diff costs, per call: RAM for a 32-byte hash per row per side (sorted), so peak memory is linear in row count (about 75 MB at 1M rows/side, 660 MB at 10M); the diff hashes and classifies across all cores by default, running single-threaded only when both sides stay under 50,000 rows and under 64 MB of decoded data (either bound reached first runs the parallel path), and appends the per-row hashes into shared per-partition buffers so there is no separate combined copy; the parallel path adds only the in-flight batches (worker count times batch size) plus the shared buffers' reallocation slack, measured at the linear shape as a peak of about 629 MB single-threaded and 857 MB at 18 threads at 8M rows/side (+228 MB) and about 2.9 GB and 3.6 GB at 37M rows/side (+685 MB) — roughly 10-15 bytes per row per side, under one extra copy of the 32-byte hash vectors, and within the run-to-run slack the single-threaded peak itself shows (2.9-4.2 GB at 37M); `threads=1` runs the sequential path; the size gate peeks each side only up to 50,000 rows or 64 MB of decoded data before choosing, reading the left side first; the peek holds at most 64 MB plus one producer batch per side — a streamed 49,999-row/side pair of 8 KB cells in 100-row batches peaks at about 78 MB at 18 threads and 14 MB single-threaded, rising to about 1.6 GB when the whole side arrives as one batch (see the size-gate peek table in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)); wall time is `N·log N` work spread across the workers, plus a second term for the duplicate-key report, which holds the key values of every *distinct duplicated* key — an all-duplicate 200k-rows/side table (100k distinct duplicated keys) peaks at about 37 MB with 16-byte keys and 1.05 GB with 1 KB keys, 1M rows/side (500k distinct) at about 165 MB with 16-byte keys; a third term for the per-cell diff, which re-reads both inputs, holds the changed rows of both sides, and renders every changed cell to a string held once in the output — so its cost is the number of changed cells times the cell width, not the changed-row count alone (every value is rendered in full). Measured, same method: with two narrow `int64` columns, 85 MB at 1M rows/side with 2% changed and 629 MB with every row changed; with a 1 KB `string` cell changed on every row, 1.18 GB at 100k rows/side and 2.34 GB at 200k (about 11.6 KB per changed row, linear, so on the order of 11.6 GB at 1M); the full wide benchmark fixture (34 columns, nearly every cell changed) measures about 67 GB at 16.875M rows/side, see [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md). Then temp disk, because each input is re-read several times and so is spooled to an anonymous file (`tempfile`: unlinked at once, mode 0600, no predictable name — nothing is left on disk even on abnormal exit), both spools resident at once, peaking at the decoded size of both inputs (about 315 MB — 161 + 154 MB uncompressed Arrow IPC — for the 1M-row fixture pair; on the order of 10 GB for the full 5 GB-per-side fixture pair, which on Linux may be a RAM-backed `tmpfs`), a full temp filesystem raising `ValueError` naming `TMPDIR`. None of these has a built-in cap: for untrusted input, bound the row count, the changed fraction, the column widths (key columns for duplicate-heavy data, changed cells — rendered in full — for wide value columns, and any column's width for the size-gate peek, which buffers decoded cells whether or not they change), and the producer's batch size, since the peek holds up to one whole batch per side beyond the 64 MB bound. Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per generated batch from `ROW_DIFF_BATCH`), single run, macOS on an Apple M-series laptop, 2026-09-06, same method as [Performance](#performance). See [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
254
254
|
- `TableDiff.to_json()` is the one member with a built-in cap: it embeds `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` in full, one JSON object per row, so it refuses with `ValueError` — naming the row count and the cap — once those four together hold more than 10,000 rows. The cap bounds row count only, the same way the row-diff cost bullet above states its per-cell term as changed cells times cell width, not column count or cell width — so a table under the row cap but with wide or large cells can still be large; use the Arrow-returning accessors instead. See [`crates/onix-arrow/src/json_rows.rs`](crates/onix-arrow/src/json_rows.rs).
|
|
255
255
|
- `diff_tables` compares scalar columns by value (hashed: null, booleans, every integer, float and decimal width, strings and binary in every encoding, timestamps, dates, times, durations, intervals, and dictionaries of these; refused with `ValueError`: run-end encoded columns and any type-and-unit combination Arrow itself cannot build; nested non-key columns skipped, nested key columns refused), with the exact enumeration in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). It also refuses a key column whose type differs across the two inputs after encoding normalization (the conservative choice: a primary key that changed type is refused rather than guessed, not coerced). A row whose only difference is a lossless type change — an `Int32` widened to `Int64`, a timestamp unit change at the same instant — hashes equal on both sides, so it is in neither `rows_changed` nor `cells_changed`; the column's type change is still reported in `schema`.
|
|
256
256
|
- Output is byte-identical to DeepDiff except for the cases listed above and two path-rendering quirks; [`tests/golden/README.md`](tests/golden/README.md) enumerates every accepted exception, including integers past `2^53` (the limit of exact `f64` representation) inside ordered scalar lists and `ignore_order` pairing among naive datetimes, which DeepDiff ranks using the *process's local timezone* while onix reads a naive value as UTC everywhere.
|
|
@@ -177,7 +177,7 @@ The engine's own diff-only time and peak resident memory against pinned `deepdif
|
|
|
177
177
|
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
178
178
|
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
179
179
|
|
|
180
|
-
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on
|
|
180
|
+
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared, still trails by a much larger multiple, see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
181
181
|
|
|
182
182
|
## Reference
|
|
183
183
|
|
|
@@ -218,7 +218,7 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
218
218
|
## Known limitations
|
|
219
219
|
|
|
220
220
|
- Only the core diff is implemented: `exclude_paths`, `significant_digits`, custom operators, `verbose_level != 2`, and delta/patch are not (yet) supported.
|
|
221
|
-
- Supported value types are `None`, `bool`, `int`, `float` (`NaN`/`Infinity`/`-Infinity` included), `str`, `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, and `datetime.timedelta`; a `set`/`frozenset` member may be any of these except a `list`, `dict` or `set`, matching Python's own hashability rule, transitively through whatever the member nests, and a `dict` key may be `str`, `None`, `bool`, `int`, `float`, `datetime.datetime`, `datetime.date`, or a `tuple` of those (never nested), or a `tuple`/`datetime`/`date` subclass key (`namedtuple` included), accepted and matched against its base-type value the same way — see [Known limitations](#known-limitations)'s subclass bullet. `int`
|
|
221
|
+
- Supported value types are `None`, `bool`, `int`, `float` (`NaN`/`Infinity`/`-Infinity` included), `str`, `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, and `datetime.timedelta`; a `set`/`frozenset` member may be any of these except a `list`, `dict` or `set`, matching Python's own hashability rule, transitively through whatever the member nests, and a `dict` key may be `str`, `None`, `bool`, `int`, `float`, `datetime.datetime`, `datetime.date`, or a `tuple` of those (never nested), or a `tuple`/`datetime`/`date` subclass key (`namedtuple` included), accepted and matched against its base-type value the same way — see [Known limitations](#known-limitations)'s subclass bullet. An `int` of any magnitude converts and keeps its exact value (an arbitrary-precision integer beyond `i64`/`u64` is compared, ordered, hashed and rendered by its full digits), while a custom object raises `TypeError` naming the exact path it was found at. Under `ignore_order`, comparing an integer beyond `f64::MAX` (about `2**1024`) reports the change deterministically where real DeepDiff raises `OverflowError` (its distance function calls `float()` on it unguarded), a documented crash-class divergence like the datetime one below (see [`tests/golden/README.md`](tests/golden/README.md)). Reading JSON *text* (`diff_json` and the CLI, not the Python-object path), an integer beyond `i64`/`u64` parses as the nearest `float`, so two documents whose integers differ only past that range compare **equal** and diff to `{}` — a limitation of the JSON number reader, not the value model, tracked in [#92](https://github.com/ksco92/onix/issues/92). A non-`str` key's path renders via Python's own `repr()`, except a `tuple` key, which splits into one bracket group per element (`root[1][2]` for `(1, 2)`, never `root[(1, 2)]`, with one deliberate exception for a real DeepDiff bug on the empty tuple, and another on a non-finite-float key — see [`tests/golden/README.md`](tests/golden/README.md)). A nested dict value's own `bool`/`None`/`int`/`float` key stringifies the way `json.dumps` does, but its `datetime`/`date`/`tuple` key renders that same `repr()` text where DeepDiff's own `to_json()` raises `TypeError` on one — a superset, not a difference in the findings (see [`tests/golden/README.md`](tests/golden/README.md)). `1`/`1.0`/`True` match as the same key between two dicts (Python `dict`/`set` equality). The **Datetimes**, **Sets** and **Non-finite floats** bullets below cover the deliberate divergences for those types. See [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs) and [`tests/golden/README.md`](tests/golden/README.md).
|
|
222
222
|
- A subclass of a supported `dict`, `list`, `tuple`, `set`, `frozenset`, `datetime.datetime`, `datetime.date`, `datetime.time`, or `datetime.timedelta` (including `namedtuple` and pandas' `Timestamp`) converts and compares as its base type but carries its own class name into a `type_changes` entry, matching DeepDiff's `type(obj).__name__` reporting, with three exceptions: a calendar-type (`datetime`/`date`/`time`/`timedelta`) subclass held as a `set`/`frozenset` member compares as its base with no `type_changes`, matching DeepDiff, since set membership has no pairwise comparison to report one against; a `tuple`/`frozenset` subclass, including a `namedtuple`, is not accepted as a `set`/`frozenset` member at all and raises `TypeError` where DeepDiff accepts and compares it by value (a documented divergence); and `namedtuple` diffs positionally rather than by field (also documented). A `type_changes` entry's `old_type`/`new_type` are type *names* in `to_dict()`, where DeepDiff returns the type objects. See [`tests/golden/README.md`](tests/golden/README.md).
|
|
223
223
|
- **Datetimes** compare by instant, with a naive value read as UTC, matching DeepDiff. A changed pair is reported normalized to UTC (`to_json()` renders `...+00:00`, `to_dict()` returns UTC-aware `datetime`s); everywhere else a datetime keeps its raw value. Three deliberate departures: `to_json()` renders a `date` as `YYYY-MM-DD` where DeepDiff's own `to_json()` raises `TypeError` (a documented superset); a `zoneinfo`/`pytz` tzinfo comes back from `to_dict()` as a fixed-offset `datetime.timezone` carrying the offset it was in force at, not the original zone object; and a set holding both a naive and an aware value at one instant reports both as members, where DeepDiff's own digest cache can report only one (see [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs)). Comparing two datetimes whose UTC form would leave year 1..=9999 raises `ValueError` naming the path, where DeepDiff raises `OverflowError`; under `ignore_order` DeepDiff's hasher normalizes every datetime and so raises for such a value even when it is only added, removed, or shuffled, where onix hashes by instant and reports it normally (see [`tests/golden/README.md`](tests/golden/README.md)). `truncate_datetime` is not supported; the normalized-versus-raw split is documented there too. A `time`/`timedelta`, unlike a datetime, is never normalized for report (DeepDiff compares `time`/`date`/`timedelta` with a plain `!=`), and a naive `time` is never equal to an aware one; `to_json()` renders a `time` as `time.isoformat()`'s bytes and a `timedelta` as `str(timedelta)`'s, both supersets. Under `ignore_order`, `DeepHash` hashes a `time` by whole seconds-of-day only — dropping the microsecond and any offset, a confirmed upstream quirk — while a `timedelta` hashes exactly; see [`tests/golden/README.md`](tests/golden/README.md)'s "Known DeepDiff quirks" section.
|
|
224
224
|
- **Sets** are diffed deterministically, where DeepDiff's own answers depend on the order the running process happens to iterate a set in (hash order, and `PYTHONHASHSEED`-dependent for `str` members) or on how its digest cache/computation handles a tuple, frozenset, or calendar member independently of Python's own `==`. Each consequence — entry order, which member of an equality class is reported, set-versus-sequence coercion, and a tuple/frozenset member's own (positional, not order-/repetition-insensitive) matching rule — is shown with both tools' output in [`tests/golden/README.md`](tests/golden/README.md)'s "Set iteration order" section. A report holding a `frozenset` value also serializes to JSON here, where DeepDiff's own `to_json()` raises `TypeError` — a superset, not a difference in the findings.
|
|
@@ -232,7 +232,7 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
232
232
|
- `diff_tables` inherits some Arrow-interchange quirks: a column name with an embedded NUL byte (`\0`) arrives truncated at the NUL through the C Data Interface (the report shows the truncated name; rare in practice); a list of structs named exactly `key`/`value` with a nullable key is not distinguished from a real map, so a migration between the two is not reported as a type change (polars exports both as the same Arrow type — see `map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs)); and a polars all-null (`Null`-typed) column fails at Arrow C import with a `ValueError` (`the datatype "Null" doesn't expect buffer at index 0`), so give such a column a concrete type first (a pyarrow all-`None` column, inferred as `null`, works and compares as all-null).
|
|
233
233
|
- In `diff_tables`, DuckDB labels a `TIMESTAMP WITH TIME ZONE` column with the connection's *session* time zone when it exports to Arrow (a UTC session as `Timestamp(µs, "UTC")`, an `America/New_York` session as `Timestamp(µs, "America/New_York")`), so on a non-UTC machine such a column can be reported as a type change against a UTC column from another library. Run `SET TimeZone='UTC'` on the DuckDB connection first for a deterministic, machine-independent result.
|
|
234
234
|
- `diff_tables` refuses a column whose Arrow type is nested deeper than `MAX_NESTING_DEPTH` (128) with a `MaxDepthError`, because comparing arbitrarily deep nesting would overflow the native stack; 128 is far beyond any real schema. Importing a schema nested many thousands of levels deep is also slow regardless, a cost of the Arrow C Data Interface itself.
|
|
235
|
-
- `diff_tables`'s row diff costs, per call: RAM for a 32-byte hash per row per side (sorted), so peak memory is linear in row count (about 75 MB at 1M rows/side, 660 MB at 10M); the diff hashes and classifies across all cores by default, running single-threaded only when both sides stay under 50,000 rows and under 64 MB of decoded data (either bound reached first runs the parallel path), and appends the per-row hashes into shared per-partition buffers so there is no separate combined copy; the parallel path adds only the in-flight batches (worker count times batch size) plus the shared buffers' reallocation slack, measured at the linear shape as a peak of about 629 MB single-threaded and 857 MB at 18 threads at 8M rows/side (+228 MB) and about 2.9 GB and 3.6 GB at 37M rows/side (+685 MB) — roughly 10-15 bytes per row per side, under one extra copy of the 32-byte hash vectors, and within the run-to-run slack the single-threaded peak itself shows (2.9-4.2 GB at 37M); `threads=1` runs the sequential path; the size gate peeks each side only up to 50,000 rows or 64 MB of decoded data before choosing, reading the left side first; the peek holds at most 64 MB plus one producer batch per side — a streamed 49,999-row/side pair of 8 KB cells in 100-row batches peaks at about 78 MB at 18 threads and 14 MB single-threaded, rising to about 1.6 GB when the whole side arrives as one batch (see the size-gate peek table in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)); wall time is `N·log N` work spread across the workers, plus a second term for the duplicate-key report, which holds the key values of every *distinct duplicated* key — an all-duplicate 200k-rows/side table (100k distinct duplicated keys) peaks at about 37 MB with 16-byte keys and 1.05 GB with 1 KB keys, 1M rows/side (500k distinct) at about 165 MB with 16-byte keys; a third term for the per-cell diff, which re-reads both inputs, holds the changed rows of both sides, and renders every changed cell to a string held once in the output — so its cost is the number of changed cells times the cell width, not the changed-row count alone (every value is rendered in full). Measured, same method: with two narrow `int64` columns, 85 MB at 1M rows/side with 2% changed and 629 MB with every row changed; with a 1 KB `string` cell changed on every row, 1.18 GB at 100k rows/side and 2.34 GB at 200k (about 11.6 KB per changed row, linear, so on the order of 11.6 GB at 1M). Then temp disk, because each input is re-read several times and so is spooled to an anonymous file (`tempfile`: unlinked at once, mode 0600, no predictable name — nothing is left on disk even on abnormal exit), both spools resident at once, peaking at the decoded size of both inputs (about 315 MB — 161 + 154 MB uncompressed Arrow IPC — for the 1M-row fixture pair; on the order of 10 GB for the full 5 GB-per-side fixture pair, which on Linux may be a RAM-backed `tmpfs`), a full temp filesystem raising `ValueError` naming `TMPDIR`. None of these has a built-in cap: for untrusted input, bound the row count, the changed fraction, the column widths (key columns for duplicate-heavy data, changed cells — rendered in full — for wide value columns, and any column's width for the size-gate peek, which buffers decoded cells whether or not they change), and the producer's batch size, since the peek holds up to one whole batch per side beyond the 64 MB bound. Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per generated batch from `ROW_DIFF_BATCH`), single run, macOS on an Apple M-series laptop, 2026-09-06, same method as [Performance](#performance). See [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
235
|
+
- `diff_tables`'s row diff costs, per call: RAM for a 32-byte hash per row per side (sorted), so peak memory is linear in row count (about 75 MB at 1M rows/side, 660 MB at 10M); the diff hashes and classifies across all cores by default, running single-threaded only when both sides stay under 50,000 rows and under 64 MB of decoded data (either bound reached first runs the parallel path), and appends the per-row hashes into shared per-partition buffers so there is no separate combined copy; the parallel path adds only the in-flight batches (worker count times batch size) plus the shared buffers' reallocation slack, measured at the linear shape as a peak of about 629 MB single-threaded and 857 MB at 18 threads at 8M rows/side (+228 MB) and about 2.9 GB and 3.6 GB at 37M rows/side (+685 MB) — roughly 10-15 bytes per row per side, under one extra copy of the 32-byte hash vectors, and within the run-to-run slack the single-threaded peak itself shows (2.9-4.2 GB at 37M); `threads=1` runs the sequential path; the size gate peeks each side only up to 50,000 rows or 64 MB of decoded data before choosing, reading the left side first; the peek holds at most 64 MB plus one producer batch per side — a streamed 49,999-row/side pair of 8 KB cells in 100-row batches peaks at about 78 MB at 18 threads and 14 MB single-threaded, rising to about 1.6 GB when the whole side arrives as one batch (see the size-gate peek table in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)); wall time is `N·log N` work spread across the workers, plus a second term for the duplicate-key report, which holds the key values of every *distinct duplicated* key — an all-duplicate 200k-rows/side table (100k distinct duplicated keys) peaks at about 37 MB with 16-byte keys and 1.05 GB with 1 KB keys, 1M rows/side (500k distinct) at about 165 MB with 16-byte keys; a third term for the per-cell diff, which re-reads both inputs, holds the changed rows of both sides, and renders every changed cell to a string held once in the output — so its cost is the number of changed cells times the cell width, not the changed-row count alone (every value is rendered in full). Measured, same method: with two narrow `int64` columns, 85 MB at 1M rows/side with 2% changed and 629 MB with every row changed; with a 1 KB `string` cell changed on every row, 1.18 GB at 100k rows/side and 2.34 GB at 200k (about 11.6 KB per changed row, linear, so on the order of 11.6 GB at 1M); the full wide benchmark fixture (34 columns, nearly every cell changed) measures about 67 GB at 16.875M rows/side, see [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md). Then temp disk, because each input is re-read several times and so is spooled to an anonymous file (`tempfile`: unlinked at once, mode 0600, no predictable name — nothing is left on disk even on abnormal exit), both spools resident at once, peaking at the decoded size of both inputs (about 315 MB — 161 + 154 MB uncompressed Arrow IPC — for the 1M-row fixture pair; on the order of 10 GB for the full 5 GB-per-side fixture pair, which on Linux may be a RAM-backed `tmpfs`), a full temp filesystem raising `ValueError` naming `TMPDIR`. None of these has a built-in cap: for untrusted input, bound the row count, the changed fraction, the column widths (key columns for duplicate-heavy data, changed cells — rendered in full — for wide value columns, and any column's width for the size-gate peek, which buffers decoded cells whether or not they change), and the producer's batch size, since the peek holds up to one whole batch per side beyond the 64 MB bound. Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per generated batch from `ROW_DIFF_BATCH`), single run, macOS on an Apple M-series laptop, 2026-09-06, same method as [Performance](#performance). See [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
236
236
|
- `TableDiff.to_json()` is the one member with a built-in cap: it embeds `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` in full, one JSON object per row, so it refuses with `ValueError` — naming the row count and the cap — once those four together hold more than 10,000 rows. The cap bounds row count only, the same way the row-diff cost bullet above states its per-cell term as changed cells times cell width, not column count or cell width — so a table under the row cap but with wide or large cells can still be large; use the Arrow-returning accessors instead. See [`crates/onix-arrow/src/json_rows.rs`](crates/onix-arrow/src/json_rows.rs).
|
|
237
237
|
- `diff_tables` compares scalar columns by value (hashed: null, booleans, every integer, float and decimal width, strings and binary in every encoding, timestamps, dates, times, durations, intervals, and dictionaries of these; refused with `ValueError`: run-end encoded columns and any type-and-unit combination Arrow itself cannot build; nested non-key columns skipped, nested key columns refused), with the exact enumeration in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). It also refuses a key column whose type differs across the two inputs after encoding normalization (the conservative choice: a primary key that changed type is refused rather than guessed, not coerced). A row whose only difference is a lossless type change — an `Int32` widened to `Int64`, a timestamp unit change at the same instant — hashes equal on both sides, so it is in neither `rows_changed` nor `cells_changed`; the column's type change is still reported in `schema`.
|
|
238
238
|
- Output is byte-identical to DeepDiff except for the cases listed above and two path-rendering quirks; [`tests/golden/README.md`](tests/golden/README.md) enumerates every accepted exception, including integers past `2^53` (the limit of exact `f64` representation) inside ordered scalar lists and `ignore_order` pairing among naive datetimes, which DeepDiff ranks using the *process's local timezone* while onix reads a naive value as UTC everywhere.
|
|
@@ -13,6 +13,11 @@ workspace = true
|
|
|
13
13
|
serde = "1"
|
|
14
14
|
serde_json = "1"
|
|
15
15
|
unicode-general-category = "1.1"
|
|
16
|
+
# Arbitrary-precision integers (a Python `int` outside i64/u64 range). Pinned
|
|
17
|
+
# to the same 0.5 line arrow-buffer already resolves in this workspace, so no
|
|
18
|
+
# second copy enters the tree.
|
|
19
|
+
num-bigint = "0.5"
|
|
20
|
+
num-traits = "0.2"
|
|
16
21
|
|
|
17
22
|
[dev-dependencies]
|
|
18
23
|
proptest = "1"
|
|
@@ -189,7 +189,7 @@
|
|
|
189
189
|
//! risking the very overflow they guard against, and `scoped`, the shared
|
|
190
190
|
//! push/pop path-buffer helper every container loop below uses.
|
|
191
191
|
//! - `scalar` — leaf-level comparison: scalar/numeric equality
|
|
192
|
-
//! (`numbers_equal`, `floats_equal
|
|
192
|
+
//! (`numbers_equal`, `floats_equal`) and the
|
|
193
193
|
//! `type_changes`/`values_changed` finding builders (`type_change_report`,
|
|
194
194
|
//! `scalar_diff`, `numeric_diff`) `diff_at` dispatches to for a
|
|
195
195
|
//! non-container pair.
|
|
@@ -191,9 +191,10 @@ pub(crate) fn numeric_diff(
|
|
|
191
191
|
/// An int and a float are never equal (mirroring `DeepDiff` always reporting
|
|
192
192
|
/// that pairing as a `type_changes`, never a numeric comparison). Within the
|
|
193
193
|
/// same kind: floats compare by exact IEEE-754 `==` (see [`floats_equal`]);
|
|
194
|
-
/// ints compare by value across
|
|
195
|
-
/// [`
|
|
196
|
-
/// counterpart
|
|
194
|
+
/// ints compare by value across every representation via
|
|
195
|
+
/// [`Number::integer_cmp`], so `9_000_000_000_000_000_000u64` and its `i64`
|
|
196
|
+
/// counterpart — or an arbitrary-precision integer and its equal — compare
|
|
197
|
+
/// equal even though they use different representations.
|
|
197
198
|
pub(crate) fn numbers_equal(old: &Number, new: &Number) -> bool {
|
|
198
199
|
if old.is_f64() != new.is_f64() {
|
|
199
200
|
return false;
|
|
@@ -208,11 +209,7 @@ pub(crate) fn numbers_equal(old: &Number, new: &Number) -> bool {
|
|
|
208
209
|
.expect("Number::is_f64 guarantees as_f64 succeeds");
|
|
209
210
|
floats_equal(old_f, new_f)
|
|
210
211
|
} else {
|
|
211
|
-
|
|
212
|
-
number_as_i128(old).expect("non-float Number always has an i64 or u64 representation");
|
|
213
|
-
let new_int =
|
|
214
|
-
number_as_i128(new).expect("non-float Number always has an i64 or u64 representation");
|
|
215
|
-
old_int == new_int
|
|
212
|
+
old.integer_cmp(new).is_eq()
|
|
216
213
|
}
|
|
217
214
|
}
|
|
218
215
|
/// Compares two floats for exact equality.
|
|
@@ -233,11 +230,3 @@ fn floats_equal(a: f64, b: f64) -> bool {
|
|
|
233
230
|
a == b
|
|
234
231
|
}
|
|
235
232
|
}
|
|
236
|
-
/// Converts a non-float [`Number`] to `i128`, so ints stored as `u64` and
|
|
237
|
-
/// `i64` compare equal by value regardless of which representation
|
|
238
|
-
/// the compact `Number` preserves from `serde_json`'s original parse.
|
|
239
|
-
pub(crate) fn number_as_i128(n: &Number) -> Option<i128> {
|
|
240
|
-
n.as_i64()
|
|
241
|
-
.map(i128::from)
|
|
242
|
-
.or_else(|| n.as_u64().map(i128::from))
|
|
243
|
-
}
|
|
@@ -37,7 +37,7 @@ fn map_deeper_than(map: &Map<String, Value>, limit: usize) -> bool {
|
|
|
37
37
|
super::dispatch::map_deeper_than(&cobj(map), limit)
|
|
38
38
|
}
|
|
39
39
|
fn number_as_i128(n: &Number) -> Option<i128> {
|
|
40
|
-
|
|
40
|
+
cnum(n).as_i128()
|
|
41
41
|
}
|
|
42
42
|
|
|
43
43
|
/// Wraps `leaf` in `depth` single-key (`"k"`) nested dicts, so `leaf`
|
|
@@ -695,7 +695,9 @@ fn coerce_to_python_str(value: &Value) -> Option<String> {
|
|
|
695
695
|
} else if let Some(i) = n.as_i64() {
|
|
696
696
|
Some(i.to_string())
|
|
697
697
|
} else {
|
|
698
|
-
n.as_u64()
|
|
698
|
+
n.as_u64()
|
|
699
|
+
.map(|u| u.to_string())
|
|
700
|
+
.or_else(|| n.as_big().map(ToString::to_string))
|
|
699
701
|
}
|
|
700
702
|
}
|
|
701
703
|
// `str(x)` is the identity for a value already a `str`, including
|
|
@@ -68,6 +68,17 @@ pub(crate) type HashSet<T> = std::collections::HashSet<T, BuildHasherDefault<FxH
|
|
|
68
68
|
/// changes which values these tables treat as the same item, not their
|
|
69
69
|
/// per-lookup cost.
|
|
70
70
|
///
|
|
71
|
+
/// An integer beyond `i128` is carried as `ItemKey::BigInt` (hence a
|
|
72
|
+
/// `DistKey` leaf, via `number_key`) and, inside a hashable tuple, as
|
|
73
|
+
/// [`ScalarKey::Big`](crate::lcs::ScalarKey) reaching the `tuple_ids` memo key
|
|
74
|
+
/// through `PyHashPart::Scalar`; both hash and compare by its magnitude digits,
|
|
75
|
+
/// which carry ample entropy and so need no `mix_float_bits` treatment, but at
|
|
76
|
+
/// `O(digits)` rather than `O(1)` per lookup — a cost proportional to that one
|
|
77
|
+
/// operand's *digit length* (the number of machine words in its magnitude),
|
|
78
|
+
/// which the `ignore_order` element-count cap does not bound: a single
|
|
79
|
+
/// astronomically large integer costs its own digit length per hash and per
|
|
80
|
+
/// comparison regardless of how few elements the diff holds.
|
|
81
|
+
///
|
|
71
82
|
/// For those remaining `ignore_order`-only tables the trade is deliberate and
|
|
72
83
|
/// measured. Re-keying them to `SipHash` (`RandomState`) added a material,
|
|
73
84
|
/// measured double-digit-percentage per-call cost on the pairing-heavy
|
|
@@ -8,6 +8,8 @@ use std::collections::{BTreeMap, BTreeSet};
|
|
|
8
8
|
use std::hash::{Hash, Hasher};
|
|
9
9
|
use std::rc::Rc;
|
|
10
10
|
|
|
11
|
+
use num_bigint::BigInt;
|
|
12
|
+
|
|
11
13
|
use crate::lcs::{ScalarKey, mix_float_bits, python_scalar_key};
|
|
12
14
|
use crate::value::Value;
|
|
13
15
|
|
|
@@ -84,8 +86,9 @@ impl Hash for DistKey {
|
|
|
84
86
|
|
|
85
87
|
/// Hashes a value consistently with its structural `PartialEq` (equal values
|
|
86
88
|
/// hash equal): a per-variant discriminant, then the fields that equality
|
|
87
|
-
/// compares — numbers through [`number_key`] (so `-0.0`/`0.0` agree
|
|
88
|
-
/// int and an equal-valued float stay distinct
|
|
89
|
+
/// compares — numbers through [`number_key`] (so `-0.0`/`0.0` agree, an
|
|
90
|
+
/// int and an equal-valued float stay distinct, and an integer beyond `i128`
|
|
91
|
+
/// hashes by its digits at `O(digits)`), a datetime by its instant, a
|
|
89
92
|
/// date by its ordinal, a list/tuple by its length and elements (order and
|
|
90
93
|
/// repetition preserving), a set/frozenset by its canonical members, a dict by
|
|
91
94
|
/// its sorted keys and their values. The exact byte sequence is unspecified;
|
|
@@ -183,12 +186,19 @@ pub(crate) enum ItemKey {
|
|
|
183
186
|
/// `true`/`false` — tagged distinctly from any integer of the same
|
|
184
187
|
/// value (never collides with `Int`).
|
|
185
188
|
Bool(bool),
|
|
186
|
-
///
|
|
187
|
-
///
|
|
188
|
-
///
|
|
189
|
-
///
|
|
190
|
-
///
|
|
189
|
+
/// An integer whose value fits `i128` — see [`mod@crate::diff`]'s
|
|
190
|
+
/// `python_type_name` for the same int/float split used throughout this
|
|
191
|
+
/// crate. Covers every `i64`/`u64` value (both fit losslessly) and any
|
|
192
|
+
/// arbitrary-precision integer that happens to fit `i128`; a value equal
|
|
193
|
+
/// across those representations shares one key.
|
|
191
194
|
Int(i128),
|
|
195
|
+
/// An arbitrary-precision integer whose magnitude exceeds `i128`, in its
|
|
196
|
+
/// own bucket — never collides with [`ItemKey::Int`] because that arm
|
|
197
|
+
/// only ever holds a value `i128` can represent (see [`number_key`],
|
|
198
|
+
/// which sends every fitting value there first). Boxed so this rare arm
|
|
199
|
+
/// does not widen the enum (a `BigInt` dwarfs the other arms' payloads),
|
|
200
|
+
/// which would grow every key in the `ignore_order` hash tables.
|
|
201
|
+
BigInt(Box<BigInt>),
|
|
192
202
|
/// A float, keyed by [`deephash_float_bits`] — its exact bit pattern for
|
|
193
203
|
/// any finite value (kept as its own bucket even when whole-numbered:
|
|
194
204
|
/// `5.0` never collides with `Int(5)`; see this type's own doc), but one
|
|
@@ -279,6 +289,7 @@ impl std::hash::Hash for ItemKey {
|
|
|
279
289
|
Self::Null => {}
|
|
280
290
|
Self::Bool(b) => b.hash(state),
|
|
281
291
|
Self::Int(i) => i.hash(state),
|
|
292
|
+
Self::BigInt(b) => b.hash(state),
|
|
282
293
|
Self::Float(bits) => mix_float_bits(*bits).hash(state),
|
|
283
294
|
Self::Str(s) => s.hash(state),
|
|
284
295
|
Self::DateTime(instant) => instant.hash(state),
|
|
@@ -723,12 +734,13 @@ fn number_key(n: &crate::value::Number) -> ItemKey {
|
|
|
723
734
|
.expect("Number::is_f64 guarantees as_f64 succeeds");
|
|
724
735
|
return ItemKey::Float(deephash_float_bits(f));
|
|
725
736
|
}
|
|
726
|
-
if let Some(i) = n.
|
|
727
|
-
return ItemKey::Int(
|
|
737
|
+
if let Some(i) = n.as_i128() {
|
|
738
|
+
return ItemKey::Int(i);
|
|
728
739
|
}
|
|
729
|
-
ItemKey::
|
|
730
|
-
n.
|
|
731
|
-
.expect("a non-
|
|
740
|
+
ItemKey::BigInt(Box::new(
|
|
741
|
+
n.as_big()
|
|
742
|
+
.expect("a non-float Number that overflows i128 is an arbitrary-precision integer")
|
|
743
|
+
.clone(),
|
|
732
744
|
))
|
|
733
745
|
}
|
|
734
746
|
|
|
@@ -772,29 +784,13 @@ fn keyed(value: &Value, memo: &IgnoreOrderMemo, want_part: bool) -> (ItemKey, Op
|
|
|
772
784
|
Value::Date(value) => (ItemKey::Date(value.ordinal()), part()),
|
|
773
785
|
Value::Time(value) => (ItemKey::Time(value.hash_seconds_of_day()), part()),
|
|
774
786
|
Value::TimeDelta(value) => (ItemKey::TimeDelta(value.value()), part()),
|
|
775
|
-
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
// keeps a distinct `Float` key from the integer `2` (this
|
|
783
|
-
// deliberately does NOT take the ordered path's `ScalarKey`
|
|
784
|
-
// integral-to-`Int` canonicalization — the two paths have
|
|
785
|
-
// genuinely different number semantics), and it collapses
|
|
786
|
-
// every `NaN` onto one shared key regardless of bits.
|
|
787
|
-
ItemKey::Float(deephash_float_bits(f))
|
|
788
|
-
} else if let Some(i) = n.as_i64() {
|
|
789
|
-
ItemKey::Int(i128::from(i))
|
|
790
|
-
} else {
|
|
791
|
-
let u = n
|
|
792
|
-
.as_u64()
|
|
793
|
-
.expect("a non-f64 serde_json::Number always has an i64 or u64 repr");
|
|
794
|
-
ItemKey::Int(i128::from(u))
|
|
795
|
-
};
|
|
796
|
-
(number, part())
|
|
797
|
-
}
|
|
787
|
+
// See `number_key` (and `deephash_float_bits`): an integral float
|
|
788
|
+
// like `2.0` keeps a distinct `Float` key from the integer `2` (this
|
|
789
|
+
// deliberately does NOT take the ordered path's `ScalarKey`
|
|
790
|
+
// integral-to-`Int` canonicalization — the two paths have genuinely
|
|
791
|
+
// different number semantics), every `NaN` collapses onto one shared
|
|
792
|
+
// key, and an integer keys by value regardless of magnitude.
|
|
793
|
+
Value::Number(n) => (number_key(n), part()),
|
|
798
794
|
Value::Array(items) => (
|
|
799
795
|
ItemKey::List(items.iter().map(|i| item_key(i, memo)).collect()),
|
|
800
796
|
None,
|
|
@@ -142,7 +142,10 @@ use super::hash::{
|
|
|
142
142
|
/// and the `A * R` entries one pairing records just share those [`Rc`](std::rc::Rc)s. The
|
|
143
143
|
/// *lookup* cost is not constant, though — every probe hashes both keys (a
|
|
144
144
|
/// full walk of each value) and, on a bucket match, compares them structurally
|
|
145
|
-
/// — so per-pair work is proportional to record size (see [`super::hash::DistKey`])
|
|
145
|
+
/// — so per-pair work is proportional to record size (see [`super::hash::DistKey`]),
|
|
146
|
+
/// and a leaf that is an integer beyond `i128` (an [`ItemKey::BigInt`], via
|
|
147
|
+
/// `number_key`) additionally costs its own *digit length* per hash and per
|
|
148
|
+
/// comparison, which the element-count cap does not bound.
|
|
146
149
|
type DistanceKey = (DistKey, DistKey);
|
|
147
150
|
|
|
148
151
|
/// The per-top-level-diff caches described in this module's doc: container-pair
|
|
@@ -154,14 +157,22 @@ type DistanceKey = (DistKey, DistKey);
|
|
|
154
157
|
/// recursive diff, and dropped when it returns. No eviction and no tuning
|
|
155
158
|
/// knobs: each is bounded by the number of distinct queries one diff makes.
|
|
156
159
|
/// `member_content`'s own key can itself carry a nested dict's keys (see its
|
|
157
|
-
/// own field doc), so a lookup there is not always a flat comparison.
|
|
160
|
+
/// own field doc), so a lookup there is not always a flat comparison. A leaf
|
|
161
|
+
/// that is an integer beyond `i128` is keyed by an [`ItemKey::BigInt`] (in the
|
|
162
|
+
/// `cache`'s `DistKey`s and, when a set member, in `member_content`), hashed
|
|
163
|
+
/// and compared by its magnitude at `O(digit length)` — a per-lookup cost the
|
|
164
|
+
/// element-count bounds above do not cap (see `super::fxhash`'s doc).
|
|
158
165
|
///
|
|
159
166
|
/// [`rough_distance`]: super::distance::rough_distance
|
|
160
167
|
pub(crate) struct IgnoreOrderMemo {
|
|
161
168
|
cache: RefCell<HashMap<DistanceKey, f64>>,
|
|
162
169
|
/// Interns each distinct hashable-tuple identity to its place in
|
|
163
170
|
/// `tuple_digests`, so a nested tuple can be named by one [`TupleId`]
|
|
164
|
-
/// inside its parent's identity instead of by a copy of its own.
|
|
171
|
+
/// inside its parent's identity instead of by a copy of its own. A tuple
|
|
172
|
+
/// member that is an integer beyond `i128` reaches this `FxHash` key as a
|
|
173
|
+
/// [`ScalarKey::Big`](crate::lcs::ScalarKey) (via `PyHashPart::Scalar`),
|
|
174
|
+
/// hashed and compared by its magnitude digits at `O(digit length)` — the
|
|
175
|
+
/// same per-lookup cost `super::fxhash`'s doc notes for `ItemKey::BigInt`.
|
|
165
176
|
tuple_ids: RefCell<HashMap<PyHashKey, TupleId>>,
|
|
166
177
|
/// The digest assigned to each interned identity, indexed by
|
|
167
178
|
/// [`TupleId::index`].
|
|
@@ -336,6 +336,40 @@ fn item_key_handles_a_u64_beyond_i64_range() {
|
|
|
336
336
|
assert_eq!(ignore_order_diff(&a, &b), json!({}));
|
|
337
337
|
}
|
|
338
338
|
|
|
339
|
+
#[test]
|
|
340
|
+
fn item_key_distinguishes_arbitrary_precision_integers() {
|
|
341
|
+
use crate::value::Number;
|
|
342
|
+
|
|
343
|
+
let big = |s: &str| CValue::Number(Number::from_bigint(s.parse().expect("integer")));
|
|
344
|
+
let memo = IgnoreOrderMemo::new();
|
|
345
|
+
let k = |v: &CValue| super::hash::item_key(v, &memo);
|
|
346
|
+
|
|
347
|
+
// A big int fitting i128 keys as `Int`; one beyond i128 as `BigInt`. Two
|
|
348
|
+
// equal big ints share a key; different ones do not; and neither collides
|
|
349
|
+
// with a small int or an equal-valued float (numbers stay type-distinct
|
|
350
|
+
// under DeepHash — see `hash::ItemKey`'s doc).
|
|
351
|
+
let two_pow_100 = "1267650600228229401496703205376";
|
|
352
|
+
let beyond_i128 = "1".to_string() + &"0".repeat(40);
|
|
353
|
+
assert_eq!(k(&big(two_pow_100)), k(&big(two_pow_100)));
|
|
354
|
+
assert_ne!(
|
|
355
|
+
k(&big(two_pow_100)),
|
|
356
|
+
k(&big("1267650600228229401496703205377"))
|
|
357
|
+
);
|
|
358
|
+
assert_ne!(k(&big(two_pow_100)), k(&big(&beyond_i128)));
|
|
359
|
+
assert_ne!(k(&big(two_pow_100)), k(&cv(&json!(1))));
|
|
360
|
+
assert_ne!(
|
|
361
|
+
k(&big(two_pow_100)),
|
|
362
|
+
k(&CValue::Number(Number::from_f64(2f64.powi(100)))),
|
|
363
|
+
);
|
|
364
|
+
|
|
365
|
+
// Hashing a beyond-i128 `ItemKey::BigInt` agrees with its equality: the
|
|
366
|
+
// same value re-inserts as a duplicate, a different one as a new entry.
|
|
367
|
+
let mut seen = std::collections::HashSet::new();
|
|
368
|
+
assert!(seen.insert(k(&big(&beyond_i128))));
|
|
369
|
+
assert!(!seen.insert(k(&big(&beyond_i128))));
|
|
370
|
+
assert!(seen.insert(k(&big(two_pow_100))));
|
|
371
|
+
}
|
|
372
|
+
|
|
339
373
|
#[test]
|
|
340
374
|
fn numeric_pairing_at_two_distances_reuses_the_used_check_across_buckets() {
|
|
341
375
|
// "5" (added) has candidates at two distinct distances: "4"
|
|
@@ -58,6 +58,9 @@
|
|
|
58
58
|
|
|
59
59
|
use std::collections::HashMap;
|
|
60
60
|
|
|
61
|
+
use num_bigint::BigInt;
|
|
62
|
+
use num_traits::{FromPrimitive, ToPrimitive};
|
|
63
|
+
|
|
61
64
|
use crate::value::Value;
|
|
62
65
|
|
|
63
66
|
/// A bucket key for grouping list elements that compare equal the way
|
|
@@ -91,6 +94,15 @@ pub(crate) enum ScalarKey {
|
|
|
91
94
|
/// compares correctly instead of failing to compile or colliding.
|
|
92
95
|
Str(Vec<u8>),
|
|
93
96
|
Int(i128),
|
|
97
|
+
/// An integer-valued scalar whose magnitude exceeds `i128` and is not
|
|
98
|
+
/// exactly representable as an `f64` — a big int that shares Python `==`
|
|
99
|
+
/// with no float, so it needs its own bucket. An exactly-representable
|
|
100
|
+
/// big int takes [`ScalarKey::Float`] instead (so it collapses with the
|
|
101
|
+
/// equal float, matching Python's `10**20 == 1e20`); see
|
|
102
|
+
/// [`python_scalar_key`]. Boxed so this rare arm does not widen the enum
|
|
103
|
+
/// (a `BigInt` is far larger than the `i128`/`Vec` the other arms hold),
|
|
104
|
+
/// which would grow every key in the scalar-list matcher's hash table.
|
|
105
|
+
Big(Box<BigInt>),
|
|
94
106
|
/// Bit pattern of a non-integral (or too-large-to-be-exact) float —
|
|
95
107
|
/// hashed through [`mix_float_bits`]; see this type's hand-written `Hash`.
|
|
96
108
|
Float(u64),
|
|
@@ -154,6 +166,7 @@ impl std::hash::Hash for ScalarKey {
|
|
|
154
166
|
Self::Null => {}
|
|
155
167
|
Self::Str(s) => s.hash(state),
|
|
156
168
|
Self::Int(i) => i.hash(state),
|
|
169
|
+
Self::Big(b) => b.hash(state),
|
|
157
170
|
Self::Float(bits) => mix_float_bits(*bits).hash(state),
|
|
158
171
|
// A `Value` node's address is 8/16-byte-aligned like any other
|
|
159
172
|
// pointer, so its low bits carry no entropy; avalanche it the
|
|
@@ -264,7 +277,23 @@ pub(crate) fn python_scalar_key(value: &Value) -> Option<ScalarKey> {
|
|
|
264
277
|
{
|
|
265
278
|
return Some(ScalarKey::Int(i));
|
|
266
279
|
}
|
|
267
|
-
let
|
|
280
|
+
if let Some(big) = n.as_big() {
|
|
281
|
+
// An arbitrary-precision integer collapses with a float that
|
|
282
|
+
// is exactly equal to it (Python's own `10**20 == 1e20`): if
|
|
283
|
+
// it round-trips through `f64` losslessly it shares that
|
|
284
|
+
// float's `Float` key (the same rule the `>2^53` integral
|
|
285
|
+
// float below already uses), otherwise it keeps its own `Big`
|
|
286
|
+
// key. So `10**20`/`1e20` pair but `10**23`/`1e23` do not —
|
|
287
|
+
// see `tests/golden/README.md`.
|
|
288
|
+
let as_float = big.to_f64().unwrap_or(f64::INFINITY);
|
|
289
|
+
if as_float.is_finite()
|
|
290
|
+
&& BigInt::from_f64(as_float).is_some_and(|rounded| &rounded == big)
|
|
291
|
+
{
|
|
292
|
+
return Some(ScalarKey::Float(as_float.to_bits()));
|
|
293
|
+
}
|
|
294
|
+
return Some(ScalarKey::Big(Box::new(big.clone())));
|
|
295
|
+
}
|
|
296
|
+
let f = n.as_f64().expect("a non-integer Number is a float");
|
|
268
297
|
if f.is_nan() {
|
|
269
298
|
return Some(ScalarKey::Nan(std::ptr::from_ref(value) as usize));
|
|
270
299
|
}
|
|
@@ -817,3 +817,48 @@ fn build_b2j_purges_only_above_the_autojunk_threshold() {
|
|
|
817
817
|
let unpurged = super::build_b2j(&b_keys, false);
|
|
818
818
|
assert!(unpurged.contains_key(&scalar_key(&json!("pop"))));
|
|
819
819
|
}
|
|
820
|
+
|
|
821
|
+
// --- arbitrary-precision integer scalar keys ----------------------------
|
|
822
|
+
|
|
823
|
+
/// A big integer collapses onto the same `ScalarKey` as a float exactly equal
|
|
824
|
+
/// to it (Python's `10**20 == 1e20`), and stays distinct from a float it does
|
|
825
|
+
/// not equal — the ordered-list matcher's Python-`==` rule. Floats are built
|
|
826
|
+
/// directly from their exact `f64` bits here, not parsed from text, so this
|
|
827
|
+
/// pins the collapse logic independently of the JSON reader's own float
|
|
828
|
+
/// rounding.
|
|
829
|
+
#[test]
|
|
830
|
+
fn big_integer_scalar_key_collapses_with_its_equal_float_only() {
|
|
831
|
+
use crate::value::{Number, Value};
|
|
832
|
+
|
|
833
|
+
let bigint = |s: &str| Value::Number(Number::from_bigint(s.parse().expect("integer")));
|
|
834
|
+
let float = |f: f64| Value::Number(Number::from_f64(f));
|
|
835
|
+
|
|
836
|
+
// 2^70 is exactly representable, so the big int and the float share a key.
|
|
837
|
+
let two_pow_70 = "1180591620717411303424";
|
|
838
|
+
assert_eq!(
|
|
839
|
+
super::python_scalar_key(&bigint(two_pow_70)),
|
|
840
|
+
super::python_scalar_key(&float(2f64.powi(70))),
|
|
841
|
+
);
|
|
842
|
+
|
|
843
|
+
// 10^23 is NOT exactly representable (`10**23 != 1e23` in f64), so the big
|
|
844
|
+
// int keeps its own `Big` key and never matches the float `1e23`.
|
|
845
|
+
let ten_pow_23 = "100000000000000000000000";
|
|
846
|
+
assert_ne!(
|
|
847
|
+
super::python_scalar_key(&bigint(ten_pow_23)),
|
|
848
|
+
super::python_scalar_key(&float(1e23)),
|
|
849
|
+
);
|
|
850
|
+
assert!(matches!(
|
|
851
|
+
super::python_scalar_key(&bigint(ten_pow_23)),
|
|
852
|
+
Some(super::ScalarKey::Big(_))
|
|
853
|
+
));
|
|
854
|
+
|
|
855
|
+
// Two equal big ints share a key; two different ones do not.
|
|
856
|
+
assert_eq!(
|
|
857
|
+
super::python_scalar_key(&bigint(ten_pow_23)),
|
|
858
|
+
super::python_scalar_key(&bigint(ten_pow_23)),
|
|
859
|
+
);
|
|
860
|
+
assert_ne!(
|
|
861
|
+
super::python_scalar_key(&bigint(ten_pow_23)),
|
|
862
|
+
super::python_scalar_key(&bigint("100000000000000000000001")),
|
|
863
|
+
);
|
|
864
|
+
}
|
|
@@ -609,8 +609,11 @@ fn number_repr(n: &Number) -> String {
|
|
|
609
609
|
if let Some(i) = n.as_i64() {
|
|
610
610
|
return i.to_string();
|
|
611
611
|
}
|
|
612
|
-
n.as_u64()
|
|
613
|
-
.
|
|
612
|
+
if let Some(u) = n.as_u64() {
|
|
613
|
+
return u.to_string();
|
|
614
|
+
}
|
|
615
|
+
n.as_big()
|
|
616
|
+
.expect("a non-float Number is an i64, a u64, or an arbitrary-precision integer")
|
|
614
617
|
.to_string()
|
|
615
618
|
}
|
|
616
619
|
|