deepdiff-rs 0.13.0__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/Cargo.lock +5 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/Cargo.toml +1 -1
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/PKG-INFO +6 -6
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/README.md +5 -5
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/Cargo.toml +3 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/examples/row_diff_profile.rs +2 -3
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/lib.rs +6 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/row_diff.rs +537 -187
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/array.rs +9 -17
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/object.rs +6 -50
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/options.rs +4 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/set.rs +9 -24
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/distance.rs +18 -22
- deepdiff_rs-0.14.0/crates/onix-core/src/ignore_order/fxhash.rs +115 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/hash.rs +3 -3
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/memo.rs +8 -5
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/pairing.rs +12 -44
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/tests.rs +180 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/report.rs +1 -1
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/value_tests.rs +2 -1
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/tests/proptest_ignore_order.rs +1 -1
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/convert.rs +1 -1
- deepdiff_rs-0.14.0/crates/onix-py/tests/test_default_path_hashing.py +80 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_differential_fuzz.py +2 -1
- deepdiff_rs-0.13.0/crates/onix-core/src/ignore_order/fxhash.rs +0 -165
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/examples/row_diff_rss.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/examples/shared/gen_shapes.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/error.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/json_rows.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/options.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/profile.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/schema.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/spool.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/src/table_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-arrow/tests/profile_passes.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/Cargo.toml +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/datetime.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/datetime_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/dispatch.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/mod.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/scalar.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/diff/tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/error.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/ignore_order/mod.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/lcs.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/lcs_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/lib.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/path.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/report_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/test_support.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/unified_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/unified_diff_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/src/value.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/tests/golden.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/tests/memory_footprint.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-core/tests/proptest_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/.python-version +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/Cargo.toml +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/benchmarks/bench_bindings.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/deepdiff_rs.pyi +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/arrow.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/deepdiff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/errors.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/fast_path.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/guard.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/src/lib.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/conftest.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_bindings_memory.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_conversions.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_datetimes.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_depth_guard.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_golden_parity.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_non_finite.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_sets.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_signed_zero.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_smoke.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_stub_mypy.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_stub_signatures.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_suite_hygiene.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_table_diff.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_table_row_diff.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_timedeltas.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_times.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_tuples.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/crates/onix-py/tests/test_wheel_contents.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.14.0}/pyproject.toml +0 -0
|
@@ -629,12 +629,13 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
|
629
629
|
|
|
630
630
|
[[package]]
|
|
631
631
|
name = "onix-arrow"
|
|
632
|
-
version = "0.
|
|
632
|
+
version = "0.14.0"
|
|
633
633
|
dependencies = [
|
|
634
634
|
"arrow-array",
|
|
635
635
|
"arrow-buffer",
|
|
636
636
|
"arrow-cast",
|
|
637
637
|
"arrow-ipc",
|
|
638
|
+
"arrow-ord",
|
|
638
639
|
"arrow-schema",
|
|
639
640
|
"arrow-select",
|
|
640
641
|
"getrandom 0.3.4",
|
|
@@ -648,7 +649,7 @@ dependencies = [
|
|
|
648
649
|
|
|
649
650
|
[[package]]
|
|
650
651
|
name = "onix-cli"
|
|
651
|
-
version = "0.
|
|
652
|
+
version = "0.14.0"
|
|
652
653
|
dependencies = [
|
|
653
654
|
"onix-core",
|
|
654
655
|
"serde_json",
|
|
@@ -656,7 +657,7 @@ dependencies = [
|
|
|
656
657
|
|
|
657
658
|
[[package]]
|
|
658
659
|
name = "onix-core"
|
|
659
|
-
version = "0.
|
|
660
|
+
version = "0.14.0"
|
|
660
661
|
dependencies = [
|
|
661
662
|
"num-bigint",
|
|
662
663
|
"num-traits",
|
|
@@ -669,7 +670,7 @@ dependencies = [
|
|
|
669
670
|
|
|
670
671
|
[[package]]
|
|
671
672
|
name = "onix-py"
|
|
672
|
-
version = "0.
|
|
673
|
+
version = "0.14.0"
|
|
673
674
|
dependencies = [
|
|
674
675
|
"arrow-array",
|
|
675
676
|
"arrow-schema",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepdiff-rs
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Rust
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -136,7 +136,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
136
136
|
|
|
137
137
|
`diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
|
|
138
138
|
|
|
139
|
-
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in
|
|
139
|
+
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
|
|
140
140
|
|
|
141
141
|
```python
|
|
142
142
|
import pyarrow as pa
|
|
@@ -210,7 +210,7 @@ The engine's own diff-only time and peak resident memory against pinned `deepdif
|
|
|
210
210
|
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
211
211
|
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
212
212
|
|
|
213
|
-
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared and rendered, spills each side's changed rows by key-hash partition and renders them across every core, which took its 5 GB-per-side wall from 149.674 to 24.375 s and its peak RSS from about 67 GB to 33.1 GB; and since 0.13.0 the parallel diff decodes the right input once and the left twice instead of each three times, taking the 5 GB pairs from 5.727 to 3.879 s narrow (1.71x DuckDB, 1.93x polars) and from 21.294 to 14.968 s wide (2.90x DuckDB, 5.47x polars) in a rotated, interleaved run, with the wide pair's peak RSS down from 33.2 GB to 28.8 GB; see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
213
|
+
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared and rendered, spills each side's changed rows by key-hash partition and renders them across every core, which took its 5 GB-per-side wall from 149.674 to 24.375 s and its peak RSS from about 67 GB to 33.1 GB; and since 0.13.0 the parallel diff decodes the right input once and the left twice instead of each three times, taking the 5 GB pairs from 5.727 to 3.879 s narrow (1.71x DuckDB, 1.93x polars) and from 21.294 to 14.968 s wide (2.90x DuckDB, 5.47x polars) in a rotated, interleaved run, with the wide pair's peak RSS down from 33.2 GB to 28.8 GB; and since 0.14.0 the cell pass compares each same-typed column with one vectorized kernel call instead of hashing every cell, cutting the wide 5 GB pair's wall by 8% and its CPU time by 23% against 0.13.0 measured in the same interleaved run (absolute figures and load in the results' column-wise cell compare section); see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
214
214
|
|
|
215
215
|
## Reference
|
|
216
216
|
|
|
@@ -243,7 +243,7 @@ crates/onix-core # the diff engine (library, no I/O)
|
|
|
243
243
|
crates/onix-cli # the `onix` binary (thin CLI over the core)
|
|
244
244
|
crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
|
|
245
245
|
crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
|
|
246
|
-
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
|
|
246
|
+
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
|
|
247
247
|
scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
|
|
248
248
|
tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
|
|
249
249
|
perf/ # cross-language benchmark harness and RESULTS.md
|
|
@@ -267,10 +267,10 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
267
267
|
- `diff_tables` inherits some Arrow-interchange quirks: a column name with an embedded NUL byte (`\0`) arrives truncated at the NUL through the C Data Interface (the report shows the truncated name; rare in practice); a list of structs named exactly `key`/`value` with a nullable key is not distinguished from a real map, so a migration between the two is not reported as a type change (polars exports both as the same Arrow type — see `map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs)); and a polars all-null (`Null`-typed) column fails at Arrow C import with a `ValueError` (`the datatype "Null" doesn't expect buffer at index 0`), so give such a column a concrete type first (a pyarrow all-`None` column, inferred as `null`, works and compares as all-null).
|
|
268
268
|
- In `diff_tables`, DuckDB labels a `TIMESTAMP WITH TIME ZONE` column with the connection's *session* time zone when it exports to Arrow (a UTC session as `Timestamp(µs, "UTC")`, an `America/New_York` session as `Timestamp(µs, "America/New_York")`), so on a non-UTC machine such a column can be reported as a type change against a UTC column from another library. Run `SET TimeZone='UTC'` on the DuckDB connection first for a deterministic, machine-independent result.
|
|
269
269
|
- `diff_tables` refuses a column whose Arrow type is nested deeper than `MAX_NESTING_DEPTH` (128) with a `MaxDepthError`, because comparing arbitrarily deep nesting would overflow the native stack; 128 is far beyond any real schema. Importing a schema nested many thousands of levels deep is also slow regardless, a cost of the Arrow C Data Interface itself.
|
|
270
|
-
- `diff_tables` row diff: 32 B hash per row per side (left only in parallel): 75 MB at 1M, 660 MB at 10M (single-threaded); 600/618 MB at 8M, 2.7/2.5 GB at 37M (1/18 threads); wall time `N·log N` across workers. Duplicate-key report, 18 threads: 39 MB (16 B keys) to 0.83 GB (1 KB keys) at 200k. Cell pass (both sides' spilled rows × column width, plus 2x `cells_changed`), 18 threads: 0.96
|
|
270
|
+
- `diff_tables` row diff: 32 B hash per row per side (left only in parallel): 75 MB at 1M, 660 MB at 10M (single-threaded); 600/618 MB at 8M, 2.7/2.5 GB at 37M (1/18 threads); wall time `N·log N` across workers. Duplicate-key report, 18 threads: 39 MB (16 B keys) to 0.83 GB (1 KB keys) at 200k. Cell pass (both sides' spilled rows × column width, plus 2x `cells_changed`), 18 threads: 0.96 GB at 200k, about 4.8 GB at 1M (4.77 to 4.91 GB over five runs each of 0.13.0 and 0.14.0), 1.56 GB spill-dominated (150k rows, eight 512 B columns, one differing). The cell pass also copies each column whose two sides share a type once per partition, aligned for comparison (changed rows ÷ partitions × that column's width; byte-view columns are compared as their spilled large-offset type, so they copy in full): +0.3 GB (5.64 → 5.95 GB) for 500k changed rows/side, three equal 1 KB columns (`Utf8View`, `BinaryView`, `Utf8`), only an `Int64` differing, 2 threads, medians of 8 runs of 0.13.1 (identical to 0.13.0 on this path) and 0.14.0. Right-side duplicate keys, 18 threads: +0.95 GB per 500k repeated keys (2 KB of value columns, 1M rows/side); +85 MB per 1M keys the left lacks (8 B `int64` keys; the entry is 32 B at any key width). Kept rows pin their input batch, per side: the whole batch when over half is kept, its view data otherwise (~2.0 GB, 1 and 18 threads, for 100 removed rows over a 1M-row side with two 1 KB `Utf8View` columns). [RESULTS.md, Fused reads](perf/arrow/RESULTS.md#fused-reads-issue-90-1).
|
|
271
271
|
- Temp disk, same call: each input spools to an anonymous file (`tempfile`: unlinked, mode 0600, no name). Changed rows spill to anonymous per-partition files, byte-view columns cast to a large-offset type (i32 caps at ~2 GB) and dictionaries decoded, independent of the partition count (~97 GB at 64 threads undecoded vs ~32 GB, wide pair); in parallel also one row per right key the left holds once that a later batch repeats (1.03 GB per 500k such keys, 2 KB of value columns, 1M rows/side, 2 and 64 threads). All resident until the cell pass ends: ~23.8 GB, wide pair (2/18/64 threads; Linux spill may be a RAM-backed `tmpfs`). Whole-process peaks, wide pair: ~29 GB at 18 threads, 40 GB at 2, 31 GB at 64 (33/46/34 GB on 0.11.2). A full temp filesystem raises `ValueError` naming `TMPDIR`.
|
|
272
272
|
- The size gate peeks up to 50,000 rows or 64 MB decoded per side, holding one whole producer batch per side beyond it: a 49,999-row/side, 8 KB-cell, 100-row-batch pair peaks ~65 MB at 18 threads, 12 MB single-threaded, ~1.6 GB single-threaded (1.2 GB at 18 threads) for one whole-side batch.
|
|
273
|
-
- None of the row-diff memory, temp-disk or size-gate-peek costs above has a built-in cap. Bound: row count; changed fraction; right-side duplicate keys (on the parallel path, one full-width row per repeated key, spilled or held); column widths (key-column width, for duplicate-heavy data; for the cell pass, the total width of all common value columns, since every one spills in full per changed row whether it changed or not, plus the rendered changed cells; any column's width for the size-gate peek, which buffers decoded cells whether or not they change); the producer's batch size; thread count, lowest at 18 threads (measured default), higher at 2 threads (two partitions each holding half the rows), slightly higher at 64: spill-dominated 2.39 GB at 2 threads, 1.96 GB at 64; output-dominated at 200k rows, 1.45 GB at 2, 1.25 GB at 64.
|
|
273
|
+
- None of the row-diff memory, temp-disk or size-gate-peek costs above has a built-in cap. Bound: row count; changed fraction; right-side duplicate keys (on the parallel path, one full-width row per repeated key, spilled or held); column widths (key-column width, for duplicate-heavy data; for the cell pass, the total width of all common value columns, since every one spills in full per changed row whether it changed or not, plus the rendered changed cells; any column's width for the size-gate peek, which buffers decoded cells whether or not they change); the producer's batch size; thread count, lowest at 18 threads (measured default), higher at 2 threads (two partitions each holding half the rows), slightly higher at 64, and fewer partitions also enlarge the cell pass's aligned copy of each same-typed compared column (changed rows per partition × the widest such column): spill-dominated 2.39 GB at 2 threads, 1.96 GB at 64; output-dominated at 200k rows, 1.45 GB at 2, 1.25 GB at 64.
|
|
274
274
|
- Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per batch from `ROW_DIFF_BATCH`), macOS on an Apple M-series laptop, 2026-09-24, same method as [Performance](#performance); the single-threaded hash-term figures carry run-to-run slack up to 2.7-3.8 GB at 37M rows/side, the other cell-pass and thread-count figures are medians over 3 runs, and the two 150k spill/output shapes over 5. See [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md) for the size-gate peek table and the decode-vs-undecoded partition figures, and [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs) for the implementation.
|
|
275
275
|
- `TableDiff.to_json()` is the one member with a built-in cap: it embeds `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` in full, one JSON object per row, so it refuses with `ValueError` — naming the row count and the cap — once those four together hold more than 10,000 rows. The cap bounds row count only, the same way the first row-diff bullet above states its per-cell term as changed cells times cell width, not column count or cell width — so a table under the row cap but with wide or large cells can still be large; use the Arrow-returning accessors instead. See [`crates/onix-arrow/src/json_rows.rs`](crates/onix-arrow/src/json_rows.rs).
|
|
276
276
|
- `diff_tables` compares scalar columns by value (hashed: null, booleans, every integer, float and decimal width, strings and binary in every encoding, timestamps, dates, times, durations, intervals, and dictionaries of these; refused with `ValueError`: run-end encoded columns and any type-and-unit combination Arrow itself cannot build; nested non-key columns skipped, nested key columns refused), with the exact enumeration in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). It also refuses a key column whose type differs across the two inputs after encoding normalization (the conservative choice: a primary key that changed type is refused rather than guessed, not coerced). A row whose only difference is a lossless type change — an `Int32` widened to `Int64`, a timestamp unit change at the same instant — hashes equal on both sides, so it is in neither `rows_changed` nor `cells_changed`; the column's type change is still reported in `schema`.
|
|
@@ -118,7 +118,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
118
118
|
|
|
119
119
|
`diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
|
|
120
120
|
|
|
121
|
-
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in
|
|
121
|
+
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
|
|
122
122
|
|
|
123
123
|
```python
|
|
124
124
|
import pyarrow as pa
|
|
@@ -192,7 +192,7 @@ The engine's own diff-only time and peak resident memory against pinned `deepdif
|
|
|
192
192
|
| `ignore_order_10k` | 73.475 ms (71.672 ms-73.940 ms) | 12.976 s (12.900 s-13.005 s) | 176.60x | 60.11 MB | 345.19 MB | 5.74x | ✅ |
|
|
193
193
|
| `identical_1m` | 9.254 ms (6.726 ms-9.875 ms) | 15.790 s (15.660 s-15.989 s) | 1706.35x | 315.41 MB | 503.19 MB | 1.60x | ✅ |
|
|
194
194
|
|
|
195
|
-
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared and rendered, spills each side's changed rows by key-hash partition and renders them across every core, which took its 5 GB-per-side wall from 149.674 to 24.375 s and its peak RSS from about 67 GB to 33.1 GB; and since 0.13.0 the parallel diff decodes the right input once and the left twice instead of each three times, taking the 5 GB pairs from 5.727 to 3.879 s narrow (1.71x DuckDB, 1.93x polars) and from 21.294 to 14.968 s wide (2.90x DuckDB, 5.47x polars) in a rotated, interleaved run, with the wide pair's peak RSS down from 33.2 GB to 28.8 GB; see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
195
|
+
Both reports carry their full methodology, fairness rules, and the reproduce command. `perf/RESULTS.md` is an upper bound (JSON parsed straight into the engine, no Python-object conversion); the bindings table is the product-surface number. Regenerate them with `perf/run_bench.sh` and `crates/onix-py/benchmarks/bench_bindings.py` (see [CONTRIBUTING.md](CONTRIBUTING.md)). The Arrow table diff's cost against two hand-rolled baselines — a DuckDB SQL join and a polars anti-join/inner-join diff, on two seeded ~5 GB parquet pairs and their own 1M-row subsets (a five-column narrow fixture, and a wide fixture covering every scalar Arrow type) — is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md): the row diff now hashes and classifies across every core, so at the 5 GB size its wall time is a small multiple of the two parallelized baselines' for the narrow fixture (down from the several-fold single-threaded gap); the wide fixture, where nearly every row's 34 columns are compared and rendered, spills each side's changed rows by key-hash partition and renders them across every core, which took its 5 GB-per-side wall from 149.674 to 24.375 s and its peak RSS from about 67 GB to 33.1 GB; and since 0.13.0 the parallel diff decodes the right input once and the left twice instead of each three times, taking the 5 GB pairs from 5.727 to 3.879 s narrow (1.71x DuckDB, 1.93x polars) and from 21.294 to 14.968 s wide (2.90x DuckDB, 5.47x polars) in a rotated, interleaved run, with the wide pair's peak RSS down from 33.2 GB to 28.8 GB; and since 0.14.0 the cell pass compares each same-typed column with one vectorized kernel call instead of hashing every cell, cutting the wide 5 GB pair's wall by 8% and its CPU time by 23% against 0.13.0 measured in the same interleaved run (absolute figures and load in the results' column-wise cell compare section); see the linked results for both; regenerated with `perf/arrow/bench_tables.py`.
|
|
196
196
|
|
|
197
197
|
## Reference
|
|
198
198
|
|
|
@@ -225,7 +225,7 @@ crates/onix-core # the diff engine (library, no I/O)
|
|
|
225
225
|
crates/onix-cli # the `onix` binary (thin CLI over the core)
|
|
226
226
|
crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
|
|
227
227
|
crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
|
|
228
|
-
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
|
|
228
|
+
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
|
|
229
229
|
scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
|
|
230
230
|
tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
|
|
231
231
|
perf/ # cross-language benchmark harness and RESULTS.md
|
|
@@ -249,10 +249,10 @@ perf/ # cross-language benchmark harness and RESULTS.md
|
|
|
249
249
|
- `diff_tables` inherits some Arrow-interchange quirks: a column name with an embedded NUL byte (`\0`) arrives truncated at the NUL through the C Data Interface (the report shows the truncated name; rare in practice); a list of structs named exactly `key`/`value` with a nullable key is not distinguished from a real map, so a migration between the two is not reported as a type change (polars exports both as the same Arrow type — see `map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs)); and a polars all-null (`Null`-typed) column fails at Arrow C import with a `ValueError` (`the datatype "Null" doesn't expect buffer at index 0`), so give such a column a concrete type first (a pyarrow all-`None` column, inferred as `null`, works and compares as all-null).
|
|
250
250
|
- In `diff_tables`, DuckDB labels a `TIMESTAMP WITH TIME ZONE` column with the connection's *session* time zone when it exports to Arrow (a UTC session as `Timestamp(µs, "UTC")`, an `America/New_York` session as `Timestamp(µs, "America/New_York")`), so on a non-UTC machine such a column can be reported as a type change against a UTC column from another library. Run `SET TimeZone='UTC'` on the DuckDB connection first for a deterministic, machine-independent result.
|
|
251
251
|
- `diff_tables` refuses a column whose Arrow type is nested deeper than `MAX_NESTING_DEPTH` (128) with a `MaxDepthError`, because comparing arbitrarily deep nesting would overflow the native stack; 128 is far beyond any real schema. Importing a schema nested many thousands of levels deep is also slow regardless, a cost of the Arrow C Data Interface itself.
|
|
252
|
-
- `diff_tables` row diff: 32 B hash per row per side (left only in parallel): 75 MB at 1M, 660 MB at 10M (single-threaded); 600/618 MB at 8M, 2.7/2.5 GB at 37M (1/18 threads); wall time `N·log N` across workers. Duplicate-key report, 18 threads: 39 MB (16 B keys) to 0.83 GB (1 KB keys) at 200k. Cell pass (both sides' spilled rows × column width, plus 2x `cells_changed`), 18 threads: 0.96
|
|
252
|
+
- `diff_tables` row diff: 32 B hash per row per side (left only in parallel): 75 MB at 1M, 660 MB at 10M (single-threaded); 600/618 MB at 8M, 2.7/2.5 GB at 37M (1/18 threads); wall time `N·log N` across workers. Duplicate-key report, 18 threads: 39 MB (16 B keys) to 0.83 GB (1 KB keys) at 200k. Cell pass (both sides' spilled rows × column width, plus 2x `cells_changed`), 18 threads: 0.96 GB at 200k, about 4.8 GB at 1M (4.77 to 4.91 GB over five runs each of 0.13.0 and 0.14.0), 1.56 GB spill-dominated (150k rows, eight 512 B columns, one differing). The cell pass also copies each column whose two sides share a type once per partition, aligned for comparison (changed rows ÷ partitions × that column's width; byte-view columns are compared as their spilled large-offset type, so they copy in full): +0.3 GB (5.64 → 5.95 GB) for 500k changed rows/side, three equal 1 KB columns (`Utf8View`, `BinaryView`, `Utf8`), only an `Int64` differing, 2 threads, medians of 8 runs of 0.13.1 (identical to 0.13.0 on this path) and 0.14.0. Right-side duplicate keys, 18 threads: +0.95 GB per 500k repeated keys (2 KB of value columns, 1M rows/side); +85 MB per 1M keys the left lacks (8 B `int64` keys; the entry is 32 B at any key width). Kept rows pin their input batch, per side: the whole batch when over half is kept, its view data otherwise (~2.0 GB, 1 and 18 threads, for 100 removed rows over a 1M-row side with two 1 KB `Utf8View` columns). [RESULTS.md, Fused reads](perf/arrow/RESULTS.md#fused-reads-issue-90-1).
|
|
253
253
|
- Temp disk, same call: each input spools to an anonymous file (`tempfile`: unlinked, mode 0600, no name). Changed rows spill to anonymous per-partition files, byte-view columns cast to a large-offset type (i32 caps at ~2 GB) and dictionaries decoded, independent of the partition count (~97 GB at 64 threads undecoded vs ~32 GB, wide pair); in parallel also one row per right key the left holds once that a later batch repeats (1.03 GB per 500k such keys, 2 KB of value columns, 1M rows/side, 2 and 64 threads). All resident until the cell pass ends: ~23.8 GB, wide pair (2/18/64 threads; Linux spill may be a RAM-backed `tmpfs`). Whole-process peaks, wide pair: ~29 GB at 18 threads, 40 GB at 2, 31 GB at 64 (33/46/34 GB on 0.11.2). A full temp filesystem raises `ValueError` naming `TMPDIR`.
|
|
254
254
|
- The size gate peeks up to 50,000 rows or 64 MB decoded per side, holding one whole producer batch per side beyond it: a 49,999-row/side, 8 KB-cell, 100-row-batch pair peaks ~65 MB at 18 threads, 12 MB single-threaded, ~1.6 GB single-threaded (1.2 GB at 18 threads) for one whole-side batch.
|
|
255
|
-
- None of the row-diff memory, temp-disk or size-gate-peek costs above has a built-in cap. Bound: row count; changed fraction; right-side duplicate keys (on the parallel path, one full-width row per repeated key, spilled or held); column widths (key-column width, for duplicate-heavy data; for the cell pass, the total width of all common value columns, since every one spills in full per changed row whether it changed or not, plus the rendered changed cells; any column's width for the size-gate peek, which buffers decoded cells whether or not they change); the producer's batch size; thread count, lowest at 18 threads (measured default), higher at 2 threads (two partitions each holding half the rows), slightly higher at 64: spill-dominated 2.39 GB at 2 threads, 1.96 GB at 64; output-dominated at 200k rows, 1.45 GB at 2, 1.25 GB at 64.
|
|
255
|
+
- None of the row-diff memory, temp-disk or size-gate-peek costs above has a built-in cap. Bound: row count; changed fraction; right-side duplicate keys (on the parallel path, one full-width row per repeated key, spilled or held); column widths (key-column width, for duplicate-heavy data; for the cell pass, the total width of all common value columns, since every one spills in full per changed row whether it changed or not, plus the rendered changed cells; any column's width for the size-gate peek, which buffers decoded cells whether or not they change); the producer's batch size; thread count, lowest at 18 threads (measured default), higher at 2 threads (two partitions each holding half the rows), slightly higher at 64, and fewer partitions also enlarge the cell pass's aligned copy of each same-typed compared column (changed rows per partition × the widest such column): spill-dominated 2.39 GB at 2 threads, 1.96 GB at 64; output-dominated at 200k rows, 1.45 GB at 2, 1.25 GB at 64.
|
|
256
256
|
- Figures are the peak resident set of `cargo run -p onix-arrow --release --example row_diff_rss` (worker count from `ROW_DIFF_THREADS`, rows per batch from `ROW_DIFF_BATCH`), macOS on an Apple M-series laptop, 2026-09-24, same method as [Performance](#performance); the single-threaded hash-term figures carry run-to-run slack up to 2.7-3.8 GB at 37M rows/side, the other cell-pass and thread-count figures are medians over 3 runs, and the two 150k spill/output shapes over 5. See [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md) for the size-gate peek table and the decode-vs-undecoded partition figures, and [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs) for the implementation.
|
|
257
257
|
- `TableDiff.to_json()` is the one member with a built-in cap: it embeds `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` in full, one JSON object per row, so it refuses with `ValueError` — naming the row count and the cap — once those four together hold more than 10,000 rows. The cap bounds row count only, the same way the first row-diff bullet above states its per-cell term as changed cells times cell width, not column count or cell width — so a table under the row cap but with wide or large cells can still be large; use the Arrow-returning accessors instead. See [`crates/onix-arrow/src/json_rows.rs`](crates/onix-arrow/src/json_rows.rs).
|
|
258
258
|
- `diff_tables` compares scalar columns by value (hashed: null, booleans, every integer, float and decimal width, strings and binary in every encoding, timestamps, dates, times, durations, intervals, and dictionaries of these; refused with `ValueError`: run-end encoded columns and any type-and-unit combination Arrow itself cannot build; nested non-key columns skipped, nested key columns refused), with the exact enumeration in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). It also refuses a key column whose type differs across the two inputs after encoding normalization (the conservative choice: a primary key that changed type is refused rather than guessed, not coerced). A row whose only difference is a lossless type change — an `Int32` widened to `Int64`, a timestamp unit change at the same instant — hashes equal on both sides, so it is in neither `rows_changed` nor `cells_changed`; the column's type change is still reported in `schema`.
|
|
@@ -45,6 +45,9 @@ arrow-select = "=59.3.0"
|
|
|
45
45
|
# file so only one is resident at once. Pinned to the same 59.3.0 as the other
|
|
46
46
|
# arrow crates.
|
|
47
47
|
arrow-ipc = "=59.3.0"
|
|
48
|
+
# `arrow-ord`'s null-aware `distinct` kernel finds a partition's changed cells one
|
|
49
|
+
# column at a time in the parallel cell pass.
|
|
50
|
+
arrow-ord = "=59.3.0"
|
|
48
51
|
tempfile = "3"
|
|
49
52
|
# Row identity is a keyed 128-bit SipHash-1-3 (`siphasher`), keyed from 16
|
|
50
53
|
# bytes of OS randomness per diff (`getrandom`), so the row-matching table on
|
|
@@ -1,6 +1,5 @@
|
|
|
1
|
-
//! Per-pass wall-time and peak-RSS profile of one keyed row diff
|
|
2
|
-
//!
|
|
3
|
-
//! Builds only with the `profile` feature, which the release wheel never enables.
|
|
1
|
+
//! Per-pass wall-time and peak-RSS profile of one keyed row diff. Builds only
|
|
2
|
+
//! with the `profile` feature, which the release wheel never enables.
|
|
4
3
|
//!
|
|
5
4
|
//! Each invocation runs a discarded warm-up diff, a timed uninstrumented diff
|
|
6
5
|
//! (the `uninstrumented wall` line), and an instrumented diff whose passes make
|
|
@@ -9,8 +9,9 @@
|
|
|
9
9
|
//! tables use [`MemoryInput`]; a one-shot stream spools to a temporary file
|
|
10
10
|
//! and implements [`TableInput`] over it, as the Python bindings do.
|
|
11
11
|
//!
|
|
12
|
-
//! See `src/row_diff.rs` for the row-matching
|
|
13
|
-
//!
|
|
12
|
+
//! See `src/row_diff.rs` for the row-matching passes, `docs/design/row-diff.md`
|
|
13
|
+
//! for the algorithm, hashing, and value-comparison rules, and `src/schema.rs`
|
|
14
|
+
//! for the column type-normalization rules.
|
|
14
15
|
//!
|
|
15
16
|
//! # Example
|
|
16
17
|
//!
|
|
@@ -77,8 +78,9 @@ pub use schema::{ChangeKind, SchemaChange, diff_schemas};
|
|
|
77
78
|
pub use table_diff::{TableDiff, TableDiffSummary};
|
|
78
79
|
|
|
79
80
|
/// The maximum column-type nesting depth [`diff_tables`] will compare; deeper is refused
|
|
80
|
-
/// with [`TableDiffError::MaxDepthExceeded`],
|
|
81
|
-
/// comparison, `Display`, `Clone
|
|
81
|
+
/// with [`TableDiffError::MaxDepthExceeded`], bounding the native-stack recursion in
|
|
82
|
+
/// comparison, `Display`, `Clone`, and the drop of values onix builds from accepted
|
|
83
|
+
/// input — not a caller's own drop of a `DataType` it built past this depth.
|
|
82
84
|
/// Per-level cost is measured by `crates/onix-arrow/examples/type_stack_cost.rs`.
|
|
83
85
|
pub const MAX_NESTING_DEPTH: usize = 128;
|
|
84
86
|
|