deepdiff-rs 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/Cargo.lock +3 -3
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/Cargo.toml +1 -1
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/PKG-INFO +36 -30
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/README.md +34 -28
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/fxhash.rs +7 -4
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/hash.rs +129 -6
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/memo.rs +48 -21
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/mod.rs +9 -6
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/pairing.rs +63 -43
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/tests.rs +500 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/value.rs +7 -7
- deepdiff_rs-0.5.0/crates/onix-core/tests/ignore_order_memory.rs +171 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/benchmarks/bench_bindings.py +121 -4
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/convert.rs +36 -4
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_conversions.py +64 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/pyproject.toml +1 -1
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/Cargo.toml +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/datetime.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/datetime_tests.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/array.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/dispatch.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/mod.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/object.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/options.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/scalar.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/set.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/diff/tests.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/error.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/ignore_order/distance.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/lcs.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/lcs_tests.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/lib.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/path.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/report.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/report_tests.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/test_support.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/src/value_tests.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/tests/golden.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/tests/memory_footprint.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/tests/proptest_diff.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/Cargo.toml +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/deepdiff.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/errors.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/fast_path.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/guard.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/src/lib.rs +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_bindings_memory.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_datetimes.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_depth_guard.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_differential_fuzz.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_golden_parity.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_sets.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_signed_zero.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_smoke.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_suite_hygiene.py +0 -0
- {deepdiff_rs-0.4.0 → deepdiff_rs-0.5.0}/crates/onix-py/tests/test_tuples.py +0 -0
|
@@ -127,7 +127,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
|
127
127
|
|
|
128
128
|
[[package]]
|
|
129
129
|
name = "onix-cli"
|
|
130
|
-
version = "0.
|
|
130
|
+
version = "0.5.0"
|
|
131
131
|
dependencies = [
|
|
132
132
|
"onix-core",
|
|
133
133
|
"serde_json",
|
|
@@ -135,7 +135,7 @@ dependencies = [
|
|
|
135
135
|
|
|
136
136
|
[[package]]
|
|
137
137
|
name = "onix-core"
|
|
138
|
-
version = "0.
|
|
138
|
+
version = "0.5.0"
|
|
139
139
|
dependencies = [
|
|
140
140
|
"proptest",
|
|
141
141
|
"serde",
|
|
@@ -145,7 +145,7 @@ dependencies = [
|
|
|
145
145
|
|
|
146
146
|
[[package]]
|
|
147
147
|
name = "onix-py"
|
|
148
|
-
version = "0.
|
|
148
|
+
version = "0.5.0"
|
|
149
149
|
dependencies = [
|
|
150
150
|
"onix-core",
|
|
151
151
|
"pyo3",
|
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepdiff-rs
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Rust
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
7
7
|
Classifier: Operating System :: OS Independent
|
|
8
|
-
Summary: onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-
|
|
8
|
+
Summary: onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-4245x faster, with ignore_order support included.
|
|
9
9
|
Keywords: diff,deepdiff,json,rust,performance
|
|
10
10
|
Author: Rodrigo Carvajal
|
|
11
11
|
License: MIT
|
|
@@ -23,7 +23,7 @@ Project-URL: Repository, https://github.com/ksco92/onix
|
|
|
23
23
|
[](LICENSE)
|
|
24
24
|
[](https://github.com/ksco92/onix/commits/main)
|
|
25
25
|
|
|
26
|
-
**onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-
|
|
26
|
+
**onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-4245x faster, with `ignore_order` support included.** Install it as `deepdiff-rs`, a drop-in `DeepDiff` class for Python, or run the diff engine as the `onix` command-line tool.
|
|
27
27
|
|
|
28
28
|
`deepdiff-rs` reads live Python objects (or JSON) and produces the exact same report [DeepDiff](https://github.com/seperman/deepdiff) does at `verbose_level=2`, so it slots into code that already parses DeepDiff output while running dramatically faster on large or deeply nested inputs.
|
|
29
29
|
|
|
@@ -118,7 +118,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
118
118
|
- A subclass of a supported type (a `tuple`, `set` or `frozenset` subclass including `namedtuple`, a `datetime`/`date` subclass such as pandas' `Timestamp`) raises `TypeError` rather than being diffed as its base type, because DeepDiff reports each value's own type name. A `type_changes` entry's `old_type`/`new_type` are type *names* in `to_dict()`, where DeepDiff returns the type objects. Both are described in [`tests/golden/README.md`](tests/golden/README.md).
|
|
119
119
|
- **Datetimes** compare by instant, with a naive value read as UTC, matching DeepDiff. A changed pair is reported normalized to UTC (`to_json()` renders `...+00:00`, `to_dict()` returns UTC-aware `datetime`s); everywhere else a datetime keeps its raw value. Three deliberate departures: `to_json()` renders a `date` as `YYYY-MM-DD` where DeepDiff's own `to_json()` raises `TypeError` (a documented superset); a `zoneinfo`/`pytz` tzinfo comes back from `to_dict()` as a fixed-offset `datetime.timezone` carrying the offset it was in force at, not the original zone object; and a set holding both a naive and an aware value at one instant reports both as members, where DeepDiff's own digest cache can report only one (see [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs)). Comparing two datetimes whose UTC form would leave year 1..=9999 raises `ValueError` naming the path, where DeepDiff raises `OverflowError`; under `ignore_order` DeepDiff's hasher normalizes every datetime and so raises for such a value even when it is only added, removed, or shuffled, where onix hashes by instant and reports it normally (see [`tests/golden/README.md`](tests/golden/README.md)). `truncate_datetime`, `time` and `timedelta` are not supported. The normalized-versus-raw split is documented in [`tests/golden/README.md`](tests/golden/README.md).
|
|
120
120
|
- **Sets** are diffed deterministically, where DeepDiff's own answers depend on the order the running process happens to iterate a set in (hash order, and `PYTHONHASHSEED`-dependent for `str` members) or on how its digest cache/computation handles a tuple, frozenset, or calendar member independently of Python's own `==`. Each consequence — entry order, which member of an equality class is reported, set-versus-sequence coercion, and a tuple/frozenset member's own (positional, not order-/repetition-insensitive) matching rule — is shown with both tools' output in [`tests/golden/README.md`](tests/golden/README.md)'s "Set iteration order" section. A report holding a `frozenset` value also serializes to JSON here, where DeepDiff's own `to_json()` raises `TypeError` — a superset, not a difference in the findings.
|
|
121
|
-
-
|
|
121
|
+
- A `str` containing a lone (unpaired) surrogate code point (e.g. `'\udc80'`, legal in Python but not encodable as UTF-8) raises `ValueError` naming the exact path on either side, before the two values are ever compared — including a pair DeepDiff would call equal and report as no change, since DeepDiff's scalar equality is plain Python `==` and never hits the encoding problem; DeepDiff does report a plain change for a *differing* pair, and crashes with an unhandled `UnicodeEncodeError` if such a string is ever hashed (a `set`/`frozenset` member). See [`tests/golden/README.md`](tests/golden/README.md)'s "Known DeepDiff quirks" section.
|
|
122
122
|
- A `str` inside a `tuple` or `frozenset` set item is rendered with Python's `repr()`, which escapes every non-printable character; onix escapes those below `U+0100` (the complete set in that range) and passes higher non-printable code points through literally, since escaping them would mean carrying a Unicode category table. Exact for all of ASCII and all printable text. See [`crates/onix-core/src/path.rs`](crates/onix-core/src/path.rs).
|
|
123
123
|
- Adversarially deep input raises `MaxDepthError` instead of crashing: the default `max_depth` is 512 and the hard ceiling is `MAX_DEPTH_CEILING` (20,000). See [`crates/onix-py/src/guard.rs`](crates/onix-py/src/guard.rs).
|
|
124
124
|
- `ignore_order` pairing is `O(N^2)` in unpaired elements per side and carries a polynomial cost in both time and memory with input depth; it has no `max_passes`/`max_diffs` cutoff, so bound the size and depth of untrusted input yourself. See [`crates/onix-core/src/ignore_order/mod.rs`](crates/onix-core/src/ignore_order/mod.rs).
|
|
@@ -128,40 +128,46 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
128
128
|
|
|
129
129
|
Two committed, regenerable reports back the numbers below; every figure here is copied verbatim from them.
|
|
130
130
|
|
|
131
|
-
The Python bindings against real `deepdiff` on **live Python objects**, the number a real caller pays (source: [`crates/onix-py/benchmarks/bench_bindings.py`](crates/onix-py/benchmarks/bench_bindings.py), macOS 26.5.1, Apple M5 Max, median of 11 isolated subprocess runs per side):
|
|
131
|
+
The Python bindings against real `deepdiff` on **live Python objects**, the number a real caller pays (source: [`crates/onix-py/benchmarks/bench_bindings.py`](crates/onix-py/benchmarks/bench_bindings.py), macOS 26.5.1, Apple M5 Max, median of 11 isolated subprocess runs per side, run on 2026-09-04):
|
|
132
132
|
|
|
133
133
|
| Shape | deepdiff | deepdiff_rs | Speedup |
|
|
134
134
|
| --- | --- | --- | --- |
|
|
135
|
-
| `ignore_order`, 10k shuffled ints, ~5% mutated (live objects) |
|
|
136
|
-
| peak RSS | 228.
|
|
137
|
-
| CPU seconds | 3.
|
|
138
|
-
| Heterogeneous API-payload records, n=20,000 (live objects) |
|
|
139
|
-
| peak RSS |
|
|
140
|
-
| CPU seconds | 3.
|
|
141
|
-
|
|
|
142
|
-
| peak RSS |
|
|
143
|
-
| CPU seconds |
|
|
144
|
-
| Same
|
|
145
|
-
| peak RSS |
|
|
146
|
-
| CPU seconds |
|
|
147
|
-
| Same
|
|
148
|
-
| peak RSS |
|
|
149
|
-
| CPU seconds |
|
|
135
|
+
| `ignore_order`, 10k shuffled ints, ~5% mutated (live objects) | 3111.56ms | 71.68ms | **43.41x** |
|
|
136
|
+
| peak RSS | 228.5 MB | 93.2 MB | **2.45x** |
|
|
137
|
+
| CPU seconds | 3.110 s | 0.072 s | **43.42x** |
|
|
138
|
+
| Heterogeneous API-payload records, n=20,000 (live objects) | 3439.96ms | 153.83ms | **22.36x** |
|
|
139
|
+
| peak RSS | 118.1 MB | 147.8 MB | **0.80x** |
|
|
140
|
+
| CPU seconds | 3.439 s | 0.154 s | **22.36x** |
|
|
141
|
+
| Typed records (datetime/tuple/set fields), n=10,000 (live objects) | 795.17ms | 48.48ms | **16.40x** |
|
|
142
|
+
| peak RSS | 60.2 MB | 62.2 MB | **0.97x** |
|
|
143
|
+
| CPU seconds | 0.795 s | 0.048 s | **16.40x** |
|
|
144
|
+
| Same typed-records shape, `ignore_order` (live objects) | 60506.65ms | 775.38ms | **78.03x** |
|
|
145
|
+
| peak RSS | 110.3 MB | 121.6 MB | **0.91x** |
|
|
146
|
+
| CPU seconds | 60.471 s | 0.774 s | **78.10x** |
|
|
147
|
+
| Same `ignore_order` shape, via `diff_json` (JSON-string path) | 3116.22ms | 73.85ms | **42.20x** |
|
|
148
|
+
| peak RSS | 228.8 MB | 93.8 MB | **2.44x** |
|
|
149
|
+
| CPU seconds | 3.114 s | 0.074 s | **42.20x** |
|
|
150
|
+
| Same API-payload shape, via `diff_json` (JSON-string path) | 4559.90ms | 87.02ms | **52.40x** |
|
|
151
|
+
| peak RSS | 139.5 MB | 140.9 MB | **0.99x** |
|
|
152
|
+
| CPU seconds | 4.558 s | 0.087 s | **52.58x** |
|
|
153
|
+
| Same API-payload shape, both tools reading two JSON files from disk | 4555.37ms | 85.62ms | **53.20x** |
|
|
154
|
+
| peak RSS | 139.5 MB | 141.0 MB | **0.99x** |
|
|
155
|
+
| CPU seconds | 4.553 s | 0.086 s | **53.21x** |
|
|
150
156
|
|
|
151
157
|
The engine's own diff-only time and peak resident memory against pinned `deepdiff` 9.1.0 (source: [`perf/RESULTS.md`](perf/RESULTS.md), same machine, median over tier-appropriate runs, diff time excluding process startup and JSON parsing on both sides):
|
|
152
158
|
|
|
153
159
|
| Fixture | onix diff-only (median, min-max) | deepdiff diff-only (median, min-max) | Speedup | onix peak RSS | deepdiff peak RSS | Memory ratio | ≥5x threshold |
|
|
154
160
|
|---|---|---|---|---|---|---|---|
|
|
155
|
-
| `flat_dict_10k` | 3.
|
|
156
|
-
| `flat_dict_100k` | 38.
|
|
157
|
-
| `flat_dict_1m` |
|
|
158
|
-
| `flat_list_100k` |
|
|
159
|
-
| `nested_uniform_d6_b10` |
|
|
160
|
-
| `api_payloads` |
|
|
161
|
-
| `deep_narrow_d120` | 0.
|
|
162
|
-
| `startup_trivial` | 0.001 ms (0.001 ms-0.001 ms) | 0.
|
|
163
|
-
| `ignore_order_10k` |
|
|
164
|
-
| `identical_1m` |
|
|
161
|
+
| `flat_dict_10k` | 3.154 ms (3.058 ms-3.220 ms) | 141.155 ms (140.606 ms-142.781 ms) | 44.75x | 5.78 MB | 39.29 MB | 6.79x | ✅ |
|
|
162
|
+
| `flat_dict_100k` | 38.440 ms (38.000 ms-38.806 ms) | 1.594 s (1.581 s-1.602 s) | 41.47x | 40.57 MB | 110.82 MB | 2.73x | ✅ |
|
|
163
|
+
| `flat_dict_1m` | 460.082 ms (454.820 ms-465.133 ms) | 17.061 s (16.926 s-17.162 s) | 37.08x | 478.15 MB | 753.65 MB | 1.58x | ✅ |
|
|
164
|
+
| `flat_list_100k` | 82.907 ms (81.131 ms-84.654 ms) | 4.751 s (4.715 s-4.820 s) | 57.31x | 38.17 MB | 154.95 MB | 4.06x | ✅ |
|
|
165
|
+
| `nested_uniform_d6_b10` | 207.242 ms (205.249 ms-217.027 ms) | 71.458 s (70.959 s-71.599 s) | 344.80x | 227.41 MB | 868.32 MB | 3.82x | ✅ |
|
|
166
|
+
| `api_payloads` | 162.489 ms (159.446 ms-178.736 ms) | 93.764 s (93.687 s-94.474 s) | 577.05x | 270.09 MB | 609.93 MB | 2.26x | ✅ |
|
|
167
|
+
| `deep_narrow_d120` | 0.029 ms (0.028 ms-0.030 ms) | 123.650 ms (123.230 ms-125.018 ms) | 4245.55x | 2.15 MB | 41.27 MB | 19.23x | ✅ |
|
|
168
|
+
| `startup_trivial` | 0.001 ms (0.001 ms-0.001 ms) | 0.175 ms (0.169 ms-0.183 ms) | 147.75x | 2.15 MB | 32.67 MB | 15.22x | ✅ |
|
|
169
|
+
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
170
|
+
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
165
171
|
|
|
166
172
|
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)).
|
|
167
173
|
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
[](LICENSE)
|
|
8
8
|
[](https://github.com/ksco92/onix/commits/main)
|
|
9
9
|
|
|
10
|
-
**onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-
|
|
10
|
+
**onix is a Rust rewrite of Python DeepDiff's core: byte-compatible output, 37-4245x faster, with `ignore_order` support included.** Install it as `deepdiff-rs`, a drop-in `DeepDiff` class for Python, or run the diff engine as the `onix` command-line tool.
|
|
11
11
|
|
|
12
12
|
`deepdiff-rs` reads live Python objects (or JSON) and produces the exact same report [DeepDiff](https://github.com/seperman/deepdiff) does at `verbose_level=2`, so it slots into code that already parses DeepDiff output while running dramatically faster on large or deeply nested inputs.
|
|
13
13
|
|
|
@@ -102,7 +102,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
102
102
|
- A subclass of a supported type (a `tuple`, `set` or `frozenset` subclass including `namedtuple`, a `datetime`/`date` subclass such as pandas' `Timestamp`) raises `TypeError` rather than being diffed as its base type, because DeepDiff reports each value's own type name. A `type_changes` entry's `old_type`/`new_type` are type *names* in `to_dict()`, where DeepDiff returns the type objects. Both are described in [`tests/golden/README.md`](tests/golden/README.md).
|
|
103
103
|
- **Datetimes** compare by instant, with a naive value read as UTC, matching DeepDiff. A changed pair is reported normalized to UTC (`to_json()` renders `...+00:00`, `to_dict()` returns UTC-aware `datetime`s); everywhere else a datetime keeps its raw value. Three deliberate departures: `to_json()` renders a `date` as `YYYY-MM-DD` where DeepDiff's own `to_json()` raises `TypeError` (a documented superset); a `zoneinfo`/`pytz` tzinfo comes back from `to_dict()` as a fixed-offset `datetime.timezone` carrying the offset it was in force at, not the original zone object; and a set holding both a naive and an aware value at one instant reports both as members, where DeepDiff's own digest cache can report only one (see [`crates/onix-py/src/convert.rs`](crates/onix-py/src/convert.rs)). Comparing two datetimes whose UTC form would leave year 1..=9999 raises `ValueError` naming the path, where DeepDiff raises `OverflowError`; under `ignore_order` DeepDiff's hasher normalizes every datetime and so raises for such a value even when it is only added, removed, or shuffled, where onix hashes by instant and reports it normally (see [`tests/golden/README.md`](tests/golden/README.md)). `truncate_datetime`, `time` and `timedelta` are not supported. The normalized-versus-raw split is documented in [`tests/golden/README.md`](tests/golden/README.md).
|
|
104
104
|
- **Sets** are diffed deterministically, where DeepDiff's own answers depend on the order the running process happens to iterate a set in (hash order, and `PYTHONHASHSEED`-dependent for `str` members) or on how its digest cache/computation handles a tuple, frozenset, or calendar member independently of Python's own `==`. Each consequence — entry order, which member of an equality class is reported, set-versus-sequence coercion, and a tuple/frozenset member's own (positional, not order-/repetition-insensitive) matching rule — is shown with both tools' output in [`tests/golden/README.md`](tests/golden/README.md)'s "Set iteration order" section. A report holding a `frozenset` value also serializes to JSON here, where DeepDiff's own `to_json()` raises `TypeError` — a superset, not a difference in the findings.
|
|
105
|
-
-
|
|
105
|
+
- A `str` containing a lone (unpaired) surrogate code point (e.g. `'\udc80'`, legal in Python but not encodable as UTF-8) raises `ValueError` naming the exact path on either side, before the two values are ever compared — including a pair DeepDiff would call equal and report as no change, since DeepDiff's scalar equality is plain Python `==` and never hits the encoding problem; DeepDiff does report a plain change for a *differing* pair, and crashes with an unhandled `UnicodeEncodeError` if such a string is ever hashed (a `set`/`frozenset` member). See [`tests/golden/README.md`](tests/golden/README.md)'s "Known DeepDiff quirks" section.
|
|
106
106
|
- A `str` inside a `tuple` or `frozenset` set item is rendered with Python's `repr()`, which escapes every non-printable character; onix escapes those below `U+0100` (the complete set in that range) and passes higher non-printable code points through literally, since escaping them would mean carrying a Unicode category table. Exact for all of ASCII and all printable text. See [`crates/onix-core/src/path.rs`](crates/onix-core/src/path.rs).
|
|
107
107
|
- Adversarially deep input raises `MaxDepthError` instead of crashing: the default `max_depth` is 512 and the hard ceiling is `MAX_DEPTH_CEILING` (20,000). See [`crates/onix-py/src/guard.rs`](crates/onix-py/src/guard.rs).
|
|
108
108
|
- `ignore_order` pairing is `O(N^2)` in unpaired elements per side and carries a polynomial cost in both time and memory with input depth; it has no `max_passes`/`max_diffs` cutoff, so bound the size and depth of untrusted input yourself. See [`crates/onix-core/src/ignore_order/mod.rs`](crates/onix-core/src/ignore_order/mod.rs).
|
|
@@ -112,40 +112,46 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
112
112
|
|
|
113
113
|
Two committed, regenerable reports back the numbers below; every figure here is copied verbatim from them.
|
|
114
114
|
|
|
115
|
-
The Python bindings against real `deepdiff` on **live Python objects**, the number a real caller pays (source: [`crates/onix-py/benchmarks/bench_bindings.py`](crates/onix-py/benchmarks/bench_bindings.py), macOS 26.5.1, Apple M5 Max, median of 11 isolated subprocess runs per side):
|
|
115
|
+
The Python bindings against real `deepdiff` on **live Python objects**, the number a real caller pays (source: [`crates/onix-py/benchmarks/bench_bindings.py`](crates/onix-py/benchmarks/bench_bindings.py), macOS 26.5.1, Apple M5 Max, median of 11 isolated subprocess runs per side, run on 2026-09-04):
|
|
116
116
|
|
|
117
117
|
| Shape | deepdiff | deepdiff_rs | Speedup |
|
|
118
118
|
| --- | --- | --- | --- |
|
|
119
|
-
| `ignore_order`, 10k shuffled ints, ~5% mutated (live objects) |
|
|
120
|
-
| peak RSS | 228.
|
|
121
|
-
| CPU seconds | 3.
|
|
122
|
-
| Heterogeneous API-payload records, n=20,000 (live objects) |
|
|
123
|
-
| peak RSS |
|
|
124
|
-
| CPU seconds | 3.
|
|
125
|
-
|
|
|
126
|
-
| peak RSS |
|
|
127
|
-
| CPU seconds |
|
|
128
|
-
| Same
|
|
129
|
-
| peak RSS |
|
|
130
|
-
| CPU seconds |
|
|
131
|
-
| Same
|
|
132
|
-
| peak RSS |
|
|
133
|
-
| CPU seconds |
|
|
119
|
+
| `ignore_order`, 10k shuffled ints, ~5% mutated (live objects) | 3111.56ms | 71.68ms | **43.41x** |
|
|
120
|
+
| peak RSS | 228.5 MB | 93.2 MB | **2.45x** |
|
|
121
|
+
| CPU seconds | 3.110 s | 0.072 s | **43.42x** |
|
|
122
|
+
| Heterogeneous API-payload records, n=20,000 (live objects) | 3439.96ms | 153.83ms | **22.36x** |
|
|
123
|
+
| peak RSS | 118.1 MB | 147.8 MB | **0.80x** |
|
|
124
|
+
| CPU seconds | 3.439 s | 0.154 s | **22.36x** |
|
|
125
|
+
| Typed records (datetime/tuple/set fields), n=10,000 (live objects) | 795.17ms | 48.48ms | **16.40x** |
|
|
126
|
+
| peak RSS | 60.2 MB | 62.2 MB | **0.97x** |
|
|
127
|
+
| CPU seconds | 0.795 s | 0.048 s | **16.40x** |
|
|
128
|
+
| Same typed-records shape, `ignore_order` (live objects) | 60506.65ms | 775.38ms | **78.03x** |
|
|
129
|
+
| peak RSS | 110.3 MB | 121.6 MB | **0.91x** |
|
|
130
|
+
| CPU seconds | 60.471 s | 0.774 s | **78.10x** |
|
|
131
|
+
| Same `ignore_order` shape, via `diff_json` (JSON-string path) | 3116.22ms | 73.85ms | **42.20x** |
|
|
132
|
+
| peak RSS | 228.8 MB | 93.8 MB | **2.44x** |
|
|
133
|
+
| CPU seconds | 3.114 s | 0.074 s | **42.20x** |
|
|
134
|
+
| Same API-payload shape, via `diff_json` (JSON-string path) | 4559.90ms | 87.02ms | **52.40x** |
|
|
135
|
+
| peak RSS | 139.5 MB | 140.9 MB | **0.99x** |
|
|
136
|
+
| CPU seconds | 4.558 s | 0.087 s | **52.58x** |
|
|
137
|
+
| Same API-payload shape, both tools reading two JSON files from disk | 4555.37ms | 85.62ms | **53.20x** |
|
|
138
|
+
| peak RSS | 139.5 MB | 141.0 MB | **0.99x** |
|
|
139
|
+
| CPU seconds | 4.553 s | 0.086 s | **53.21x** |
|
|
134
140
|
|
|
135
141
|
The engine's own diff-only time and peak resident memory against pinned `deepdiff` 9.1.0 (source: [`perf/RESULTS.md`](perf/RESULTS.md), same machine, median over tier-appropriate runs, diff time excluding process startup and JSON parsing on both sides):
|
|
136
142
|
|
|
137
143
|
| Fixture | onix diff-only (median, min-max) | deepdiff diff-only (median, min-max) | Speedup | onix peak RSS | deepdiff peak RSS | Memory ratio | ≥5x threshold |
|
|
138
144
|
|---|---|---|---|---|---|---|---|
|
|
139
|
-
| `flat_dict_10k` | 3.
|
|
140
|
-
| `flat_dict_100k` | 38.
|
|
141
|
-
| `flat_dict_1m` |
|
|
142
|
-
| `flat_list_100k` |
|
|
143
|
-
| `nested_uniform_d6_b10` |
|
|
144
|
-
| `api_payloads` |
|
|
145
|
-
| `deep_narrow_d120` | 0.
|
|
146
|
-
| `startup_trivial` | 0.001 ms (0.001 ms-0.001 ms) | 0.
|
|
147
|
-
| `ignore_order_10k` |
|
|
148
|
-
| `identical_1m` |
|
|
145
|
+
| `flat_dict_10k` | 3.154 ms (3.058 ms-3.220 ms) | 141.155 ms (140.606 ms-142.781 ms) | 44.75x | 5.78 MB | 39.29 MB | 6.79x | ✅ |
|
|
146
|
+
| `flat_dict_100k` | 38.440 ms (38.000 ms-38.806 ms) | 1.594 s (1.581 s-1.602 s) | 41.47x | 40.57 MB | 110.82 MB | 2.73x | ✅ |
|
|
147
|
+
| `flat_dict_1m` | 460.082 ms (454.820 ms-465.133 ms) | 17.061 s (16.926 s-17.162 s) | 37.08x | 478.15 MB | 753.65 MB | 1.58x | ✅ |
|
|
148
|
+
| `flat_list_100k` | 82.907 ms (81.131 ms-84.654 ms) | 4.751 s (4.715 s-4.820 s) | 57.31x | 38.17 MB | 154.95 MB | 4.06x | ✅ |
|
|
149
|
+
| `nested_uniform_d6_b10` | 207.242 ms (205.249 ms-217.027 ms) | 71.458 s (70.959 s-71.599 s) | 344.80x | 227.41 MB | 868.32 MB | 3.82x | ✅ |
|
|
150
|
+
| `api_payloads` | 162.489 ms (159.446 ms-178.736 ms) | 93.764 s (93.687 s-94.474 s) | 577.05x | 270.09 MB | 609.93 MB | 2.26x | ✅ |
|
|
151
|
+
| `deep_narrow_d120` | 0.029 ms (0.028 ms-0.030 ms) | 123.650 ms (123.230 ms-125.018 ms) | 4245.55x | 2.15 MB | 41.27 MB | 19.23x | ✅ |
|
|
152
|
+
| `startup_trivial` | 0.001 ms (0.001 ms-0.001 ms) | 0.175 ms (0.169 ms-0.183 ms) | 147.75x | 2.15 MB | 32.67 MB | 15.22x | ✅ |
|
|
153
|
+
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
154
|
+
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
149
155
|
|
|
150
156
|
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)).
|
|
151
157
|
|
|
@@ -48,10 +48,13 @@ pub(crate) type HashSet<T> = std::collections::HashSet<T, BuildHasherDefault<FxH
|
|
|
48
48
|
/// the pairing/`used` sets, and the distance memo — is reached **only** under
|
|
49
49
|
/// `ignore_order=true`, the pairing hot path already bounded by the `O(N²)`
|
|
50
50
|
/// caveat below; those keep `FxHash`, and their float-carrying keys
|
|
51
|
-
/// ([`ItemKey`](super::hash::ItemKey)
|
|
52
|
-
///
|
|
53
|
-
///
|
|
54
|
-
///
|
|
51
|
+
/// ([`ItemKey`](super::hash::ItemKey), [`ScalarKey`](super::hash::ScalarKey),
|
|
52
|
+
/// and the distance memo's [`DistKey`](super::hash::DistKey)) mix their bits
|
|
53
|
+
/// first ([`mix_float_bits`](crate::lcs::mix_float_bits)) — `DistKey`'s
|
|
54
|
+
/// encoding routes every float through `number_key`, hence `ItemKey::Float`,
|
|
55
|
+
/// hence `mix_float_bits`, so the hazard does not reach the new key either — a
|
|
56
|
+
/// *non-adversarial* run of integral/half-integer floats, whose raw bit
|
|
57
|
+
/// patterns share ~50 trailing zeros, does not accidentally collide.
|
|
55
58
|
///
|
|
56
59
|
/// For those remaining `ignore_order`-only tables the trade is deliberate and
|
|
57
60
|
/// measured. Re-keying them to `SipHash` (`RandomState`) added a material,
|
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
//! how this fits into the algorithm end to end.
|
|
6
6
|
|
|
7
7
|
use std::collections::{BTreeMap, BTreeSet};
|
|
8
|
+
use std::hash::{Hash, Hasher};
|
|
8
9
|
use std::rc::Rc;
|
|
9
10
|
|
|
10
11
|
use crate::lcs::{ScalarKey, mix_float_bits, python_scalar_key};
|
|
@@ -13,6 +14,126 @@ use crate::value::Value;
|
|
|
13
14
|
use super::IgnoreOrderMemo;
|
|
14
15
|
use super::fxhash::HashMap;
|
|
15
16
|
|
|
17
|
+
// ---------------------------------------------------------------------
|
|
18
|
+
// Distance-memo cache key
|
|
19
|
+
// ---------------------------------------------------------------------
|
|
20
|
+
|
|
21
|
+
/// The distance memo's cache key for one side of a candidate pair: a value's
|
|
22
|
+
/// **exact** structural identity.
|
|
23
|
+
///
|
|
24
|
+
/// [`ItemKey`] cannot serve here. It is deliberately order- and
|
|
25
|
+
/// repetition-*insensitive* for a list/tuple (its `List`/`Tuple` payload is a
|
|
26
|
+
/// [`BTreeSet`], matching `DeepHash`'s item-matching rules), but the distance a
|
|
27
|
+
/// candidate pair is ranked by reads multiplicity — [`super::distance::rough_length`]
|
|
28
|
+
/// counts every repeated element, and the trial diff's leaf count depends on
|
|
29
|
+
/// each list's first-occurrence representative — so two values that share an
|
|
30
|
+
/// `ItemKey` can have genuinely different distances. Keying the memo by
|
|
31
|
+
/// `ItemKey` handed one such value's cached distance to the other (issue #31).
|
|
32
|
+
///
|
|
33
|
+
/// This keys by [`Value`]'s own `PartialEq` instead, which is exact:
|
|
34
|
+
/// order- and repetition-preserving for lists and tuples, variant-sensitive
|
|
35
|
+
/// for numbers, by-instant for datetimes. Two entries therefore share a cache
|
|
36
|
+
/// slot only when their values are structurally identical and so their
|
|
37
|
+
/// distance is provably equal — the memo is decision-neutral by construction.
|
|
38
|
+
/// The [`Rc`] lets the shared value outlive the per-level [`HashedList`] that
|
|
39
|
+
/// produced it and keeps a cache entry two pointers wide.
|
|
40
|
+
#[derive(Clone)]
|
|
41
|
+
pub(crate) struct DistKey(Rc<Value>);
|
|
42
|
+
|
|
43
|
+
impl DistKey {
|
|
44
|
+
/// Interns a copy of `value` as a cache key: the value is cloned once per
|
|
45
|
+
/// distinct candidate (added/removed) entry, not once per pair, so the
|
|
46
|
+
/// `A * R` cache entries one pairing records cost only a refcount bump of
|
|
47
|
+
/// memory each. Per-*lookup* work is not constant, however — every probe
|
|
48
|
+
/// hashes the key ([`hash_value`] walks the whole value) and, on a bucket
|
|
49
|
+
/// match, compares two keys structurally — so a pairing's distance work is
|
|
50
|
+
/// proportional to record size, a constant-factor cost the `ignore_order`
|
|
51
|
+
/// input-size cap already covers (measured 2.3x-5.6x slower than a
|
|
52
|
+
/// deduplicating `ItemKey` key only on a crafted worst case: large values
|
|
53
|
+
/// whose `ItemKey` collapses to a tiny set; every realistic shape is
|
|
54
|
+
/// faster). The clone recurses natively like the report's own value clones
|
|
55
|
+
/// and is bounded by the same [`crate::diff::check_value_depth`] pre-pass;
|
|
56
|
+
/// the hashing the key is looked up by is iterative (see [`hash_value`]).
|
|
57
|
+
pub(crate) fn new(value: &Value) -> Self {
|
|
58
|
+
Self(Rc::new(value.clone()))
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/// Wraps an already-owned value with no clone — the test hook that lets the
|
|
62
|
+
/// stack-safety probe hash a value deeper than a native clone could build.
|
|
63
|
+
#[cfg(test)]
|
|
64
|
+
pub(crate) fn from_rc(value: Rc<Value>) -> Self {
|
|
65
|
+
Self(value)
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
impl PartialEq for DistKey {
|
|
70
|
+
fn eq(&self, other: &Self) -> bool {
|
|
71
|
+
// `Value`'s own iterative structural equality — exact, and stack-safe
|
|
72
|
+
// on the deep values `Rc::ptr_eq` would miss across nesting levels.
|
|
73
|
+
self.0 == other.0
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
impl Eq for DistKey {}
|
|
78
|
+
|
|
79
|
+
impl Hash for DistKey {
|
|
80
|
+
fn hash<H: Hasher>(&self, state: &mut H) {
|
|
81
|
+
hash_value(&self.0, state);
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/// Hashes a value consistently with its structural `PartialEq` (equal values
|
|
86
|
+
/// hash equal): a per-variant discriminant, then the fields that equality
|
|
87
|
+
/// compares — numbers through [`number_key`] (so `-0.0`/`0.0` agree and an
|
|
88
|
+
/// int and an equal-valued float stay distinct), a datetime by its instant, a
|
|
89
|
+
/// date by its ordinal, a list/tuple by its length and elements (order and
|
|
90
|
+
/// repetition preserving), a set/frozenset by its canonical members, a dict by
|
|
91
|
+
/// its sorted keys and their values. The exact byte sequence is unspecified;
|
|
92
|
+
/// only its determinism and its agreement with [`Value`]'s equality matter.
|
|
93
|
+
///
|
|
94
|
+
/// **Iterative** (an explicit work-stack, never native recursion), like the
|
|
95
|
+
/// engine's own [`Value`] `Drop`/`PartialEq`: a value's nesting is
|
|
96
|
+
/// user-controlled and can reach the caller-raised `max_depth`, so a recursive
|
|
97
|
+
/// hasher would reopen the uncatchable native-stack-overflow class those
|
|
98
|
+
/// iterative primitives exist to close — the default budget being safe is
|
|
99
|
+
/// coincidence, not a property. The walk hashes each node before pushing its
|
|
100
|
+
/// children, so two structurally equal values drive identical stack operations
|
|
101
|
+
/// and hash identically; children are pushed so a container's length prefix and
|
|
102
|
+
/// per-variant discriminant keep distinct shapes apart.
|
|
103
|
+
fn hash_value<H: Hasher>(root: &Value, state: &mut H) {
|
|
104
|
+
let mut stack: Vec<&Value> = vec![root];
|
|
105
|
+
while let Some(value) = stack.pop() {
|
|
106
|
+
core::mem::discriminant(value).hash(state);
|
|
107
|
+
match value {
|
|
108
|
+
Value::Null => {}
|
|
109
|
+
Value::Bool(b) => b.hash(state),
|
|
110
|
+
Value::Number(n) => number_key(n).hash(state),
|
|
111
|
+
Value::Str(s) => s.hash(state),
|
|
112
|
+
Value::DateTime(dt) => dt.instant().hash(state),
|
|
113
|
+
Value::Date(date) => date.ordinal().hash(state),
|
|
114
|
+
Value::Array(items) | Value::Tuple(items) => {
|
|
115
|
+
items.len().hash(state);
|
|
116
|
+
stack.extend(items.iter());
|
|
117
|
+
}
|
|
118
|
+
Value::Set(items) | Value::FrozenSet(items) => {
|
|
119
|
+
items.len().hash(state);
|
|
120
|
+
stack.extend(items.iter());
|
|
121
|
+
}
|
|
122
|
+
Value::Object(map) => {
|
|
123
|
+
map.len().hash(state);
|
|
124
|
+
// Keys carry the association and are hashed here in sorted
|
|
125
|
+
// order; the values are pushed and hashed as they pop. The
|
|
126
|
+
// order values come back in is deterministic, which is all
|
|
127
|
+
// equality-consistency needs.
|
|
128
|
+
for (key, _) in map {
|
|
129
|
+
key.hash(state);
|
|
130
|
+
}
|
|
131
|
+
stack.extend(map.values());
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
16
137
|
/// A canonical hash-equivalence key for one JSON value, matching
|
|
17
138
|
/// `DeepHash`'s default semantics for **item matching** under
|
|
18
139
|
/// `ignore_order=True` — deliberately **not** the same
|
|
@@ -667,9 +788,11 @@ fn tuple_keyed(items: &[Value], memo: &IgnoreOrderMemo) -> (ItemKey, Option<PyHa
|
|
|
667
788
|
pub(crate) struct HashedList<'a> {
|
|
668
789
|
/// Distinct keys, in first-occurrence (ascending original index) order
|
|
669
790
|
/// — this is `SetOrdered(full_t{1,2}_hashtable.keys())`'s own iteration
|
|
670
|
-
/// order (a Python dict's insertion order).
|
|
671
|
-
|
|
672
|
-
|
|
791
|
+
/// order (a Python dict's insertion order). Each key sits behind an [`Rc`]
|
|
792
|
+
/// shared with [`Self::info`] and the distance-memo cache keys (see
|
|
793
|
+
/// [`super::memo::DistanceKey`]).
|
|
794
|
+
pub(crate) distinct_order: Vec<Rc<ItemKey>>,
|
|
795
|
+
info: HashMap<Rc<ItemKey>, (usize, &'a Value)>,
|
|
673
796
|
}
|
|
674
797
|
|
|
675
798
|
impl<'a> HashedList<'a> {
|
|
@@ -680,12 +803,12 @@ impl<'a> HashedList<'a> {
|
|
|
680
803
|
/// builds its two hashtables against one shared `hashes` dict.
|
|
681
804
|
pub(crate) fn build(items: &'a [Value], memo: &IgnoreOrderMemo) -> Self {
|
|
682
805
|
let mut distinct_order = Vec::new();
|
|
683
|
-
let mut info: HashMap<ItemKey
|
|
806
|
+
let mut info: HashMap<Rc<ItemKey>, (usize, &'a Value)> = HashMap::default();
|
|
684
807
|
|
|
685
808
|
for (idx, item) in items.iter().enumerate() {
|
|
686
|
-
let key = item_key(item, memo);
|
|
809
|
+
let key = Rc::new(item_key(item, memo));
|
|
687
810
|
|
|
688
|
-
if let std::collections::hash_map::Entry::Vacant(entry) = info.entry(
|
|
811
|
+
if let std::collections::hash_map::Entry::Vacant(entry) = info.entry(Rc::clone(&key)) {
|
|
689
812
|
distinct_order.push(key);
|
|
690
813
|
entry.insert((idx, item));
|
|
691
814
|
}
|
|
@@ -40,12 +40,14 @@
|
|
|
40
40
|
//! `max_depth - depth - 1` — the trial always completes. So the structural
|
|
41
41
|
//! result depends only on content too.
|
|
42
42
|
//!
|
|
43
|
-
//!
|
|
44
|
-
//!
|
|
45
|
-
//! `
|
|
46
|
-
//!
|
|
47
|
-
//!
|
|
48
|
-
//!
|
|
43
|
+
//! Caching is therefore decision-neutral **as long as the key distinguishes
|
|
44
|
+
//! every value pair whose distance differs** — which is exactly why the cache
|
|
45
|
+
//! is keyed by [`super::hash::DistKey`] (a value's exact structural identity),
|
|
46
|
+
//! not by the order- and repetition-insensitive `ItemKey` that conflated
|
|
47
|
+
//! repetition-differing values into one slot (issue #31); see `DistKey`'s own
|
|
48
|
+
//! doc for that rationale. Verified empirically by the with/without
|
|
49
|
+
//! differential test in `super::tests` (including a repetition-only sibling
|
|
50
|
+
//! divergence).
|
|
49
51
|
//!
|
|
50
52
|
//! # Tuple digests
|
|
51
53
|
//!
|
|
@@ -129,11 +131,24 @@ use std::cell::RefCell;
|
|
|
129
131
|
use std::collections::BTreeMap;
|
|
130
132
|
|
|
131
133
|
use super::fxhash::HashMap;
|
|
132
|
-
use super::hash::{
|
|
134
|
+
use super::hash::{
|
|
135
|
+
DistKey, ItemKey, MemberContent, MemberHashKey, NodeId, PyHashKey, RepId, TupleId,
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
/// A `(removed, added)` container-pair distance-cache key — see
|
|
139
|
+
/// [`super::hash::DistKey`] for why each side is a value's exact structural
|
|
140
|
+
/// identity (not its order/repetition-insensitive `ItemKey`). Memory per
|
|
141
|
+
/// entry is a refcount bump: each side's value is interned once per candidate
|
|
142
|
+
/// and the `A * R` entries one pairing records just share those [`Rc`]s. The
|
|
143
|
+
/// *lookup* cost is not constant, though — every probe hashes both keys (a
|
|
144
|
+
/// full walk of each value) and, on a bucket match, compares them structurally
|
|
145
|
+
/// — so per-pair work is proportional to record size (see [`super::hash::DistKey`]).
|
|
146
|
+
type DistanceKey = (DistKey, DistKey);
|
|
133
147
|
|
|
134
148
|
/// The per-top-level-diff caches described in this module's doc: container-pair
|
|
135
|
-
/// [`rough_distance`] results keyed by the `(removed, added)` [`
|
|
136
|
-
/// pair
|
|
149
|
+
/// [`rough_distance`] results keyed by the `(removed, added)` [`DistKey`]
|
|
150
|
+
/// pair (each side a value's exact structural identity), the tuple-digest
|
|
151
|
+
/// interning table for list-item matching, and the two
|
|
137
152
|
/// set-member interning tables. Created in `crate::diff::diff_with_options`,
|
|
138
153
|
/// threaded (by shared reference, interior mutability) through the whole
|
|
139
154
|
/// recursive diff, and dropped when it returns. No eviction and no tuning
|
|
@@ -141,7 +156,7 @@ use super::hash::{ItemKey, MemberContent, MemberHashKey, NodeId, PyHashKey, RepI
|
|
|
141
156
|
///
|
|
142
157
|
/// [`rough_distance`]: super::distance::rough_distance
|
|
143
158
|
pub(crate) struct IgnoreOrderMemo {
|
|
144
|
-
cache: RefCell<HashMap<
|
|
159
|
+
cache: RefCell<HashMap<DistanceKey, f64>>,
|
|
145
160
|
/// Interns each distinct hashable-tuple identity to its place in
|
|
146
161
|
/// `tuple_digests`, so a nested tuple can be named by one [`TupleId`]
|
|
147
162
|
/// inside its parent's identity instead of by a copy of its own.
|
|
@@ -194,22 +209,33 @@ impl IgnoreOrderMemo {
|
|
|
194
209
|
}
|
|
195
210
|
}
|
|
196
211
|
|
|
197
|
-
/// Whether
|
|
198
|
-
///
|
|
199
|
-
///
|
|
200
|
-
///
|
|
201
|
-
/// (a list of numbers, say) free of any memoization overhead.
|
|
202
|
-
|
|
203
|
-
|
|
212
|
+
/// Whether distance memoization is live for this run. A candidate pair is
|
|
213
|
+
/// additionally only cached when both sides are containers (see
|
|
214
|
+
/// [`is_container`]): scalar-involving pairs never recurse, so they never
|
|
215
|
+
/// re-compute and skip the cache entirely, keeping flat `ignore_order`
|
|
216
|
+
/// shapes (a list of numbers, say) free of any memoization overhead. The
|
|
217
|
+
/// `disabled()` cache reports `false` here so the with/without differential
|
|
218
|
+
/// test runs the identical code path with the cache inert.
|
|
219
|
+
pub(crate) fn caching_enabled(&self) -> bool {
|
|
220
|
+
self.enabled
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/// The number of distinct container-pair distances currently memoized.
|
|
224
|
+
/// Test-only: lets the gate tests assert that scalar pairs are never
|
|
225
|
+
/// cached and that a `disabled()` memo caches nothing, pinning the two
|
|
226
|
+
/// conditions `caching_enabled`/`is_container` guard.
|
|
227
|
+
#[cfg(test)]
|
|
228
|
+
pub(crate) fn cache_len(&self) -> usize {
|
|
229
|
+
self.cache.borrow().len()
|
|
204
230
|
}
|
|
205
231
|
|
|
206
232
|
/// The cached distance for `key`, if present.
|
|
207
|
-
pub(crate) fn get(&self, key: &
|
|
233
|
+
pub(crate) fn get(&self, key: &DistanceKey) -> Option<f64> {
|
|
208
234
|
self.cache.borrow().get(key).copied()
|
|
209
235
|
}
|
|
210
236
|
|
|
211
237
|
/// Records `value` for `key` (moving the already-cloned key in).
|
|
212
|
-
pub(crate) fn put(&self, key:
|
|
238
|
+
pub(crate) fn put(&self, key: DistanceKey, value: f64) {
|
|
213
239
|
self.cache.borrow_mut().insert(key, value);
|
|
214
240
|
}
|
|
215
241
|
|
|
@@ -286,7 +312,8 @@ impl IgnoreOrderMemo {
|
|
|
286
312
|
}
|
|
287
313
|
|
|
288
314
|
/// Whether `key` is a container (list/tuple/dict) rather than a scalar — the
|
|
289
|
-
/// variants whose distance is computed by a recursive trial diff
|
|
290
|
-
|
|
315
|
+
/// variants whose distance is computed by a recursive trial diff, and so the
|
|
316
|
+
/// only ones worth memoizing.
|
|
317
|
+
pub(crate) fn is_container(key: &ItemKey) -> bool {
|
|
291
318
|
matches!(key, ItemKey::List(_) | ItemKey::Tuple(_) | ItemKey::Dict(_))
|
|
292
319
|
}
|