deepdiff-rs 0.13.0__tar.gz → 0.13.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/Cargo.lock +4 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/Cargo.toml +1 -1
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/PKG-INFO +3 -3
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/README.md +2 -2
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/row_diff_profile.rs +2 -3
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/lib.rs +6 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/row_diff.rs +18 -161
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/options.rs +4 -4
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/distance.rs +18 -22
- deepdiff_rs-0.13.1/crates/onix-core/src/ignore_order/fxhash.rs +115 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/hash.rs +3 -3
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/memo.rs +8 -5
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/tests.rs +180 -0
- deepdiff_rs-0.13.1/crates/onix-py/tests/test_default_path_hashing.py +80 -0
- deepdiff_rs-0.13.0/crates/onix-core/src/ignore_order/fxhash.rs +0 -165
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/Cargo.toml +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/row_diff_rss.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/shared/gen_shapes.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/error.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/json_rows.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/options.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/profile.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/schema.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/spool.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/table_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/tests/profile_passes.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/Cargo.toml +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/datetime.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/datetime_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/array.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/dispatch.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/mod.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/object.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/scalar.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/set.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/error.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/mod.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/pairing.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lcs.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lcs_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lib.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/path.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/report.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/report_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/test_support.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/unified_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/unified_diff_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/value.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/value_tests.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/golden.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/memory_footprint.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/proptest_diff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/.python-version +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/Cargo.toml +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/benchmarks/bench_bindings.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/deepdiff_rs.pyi +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/arrow.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/convert.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/deepdiff.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/errors.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/fast_path.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/guard.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/lib.rs +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/conftest.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_bindings_memory.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_conversions.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_datetimes.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_depth_guard.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_differential_fuzz.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_golden_parity.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_non_finite.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_sets.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_signed_zero.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_smoke.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_stub_mypy.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_stub_signatures.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_suite_hygiene.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_table_diff.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_table_row_diff.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_timedeltas.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_times.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_tuples.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_wheel_contents.py +0 -0
- {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/pyproject.toml +0 -0
|
@@ -629,7 +629,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
|
629
629
|
|
|
630
630
|
[[package]]
|
|
631
631
|
name = "onix-arrow"
|
|
632
|
-
version = "0.13.
|
|
632
|
+
version = "0.13.1"
|
|
633
633
|
dependencies = [
|
|
634
634
|
"arrow-array",
|
|
635
635
|
"arrow-buffer",
|
|
@@ -648,7 +648,7 @@ dependencies = [
|
|
|
648
648
|
|
|
649
649
|
[[package]]
|
|
650
650
|
name = "onix-cli"
|
|
651
|
-
version = "0.13.
|
|
651
|
+
version = "0.13.1"
|
|
652
652
|
dependencies = [
|
|
653
653
|
"onix-core",
|
|
654
654
|
"serde_json",
|
|
@@ -656,7 +656,7 @@ dependencies = [
|
|
|
656
656
|
|
|
657
657
|
[[package]]
|
|
658
658
|
name = "onix-core"
|
|
659
|
-
version = "0.13.
|
|
659
|
+
version = "0.13.1"
|
|
660
660
|
dependencies = [
|
|
661
661
|
"num-bigint",
|
|
662
662
|
"num-traits",
|
|
@@ -669,7 +669,7 @@ dependencies = [
|
|
|
669
669
|
|
|
670
670
|
[[package]]
|
|
671
671
|
name = "onix-py"
|
|
672
|
-
version = "0.13.
|
|
672
|
+
version = "0.13.1"
|
|
673
673
|
dependencies = [
|
|
674
674
|
"arrow-array",
|
|
675
675
|
"arrow-schema",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepdiff-rs
|
|
3
|
-
Version: 0.13.
|
|
3
|
+
Version: 0.13.1
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Rust
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -136,7 +136,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
136
136
|
|
|
137
137
|
`diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
|
|
138
138
|
|
|
139
|
-
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in
|
|
139
|
+
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
|
|
140
140
|
|
|
141
141
|
```python
|
|
142
142
|
import pyarrow as pa
|
|
@@ -243,7 +243,7 @@ crates/onix-core # the diff engine (library, no I/O)
|
|
|
243
243
|
crates/onix-cli # the `onix` binary (thin CLI over the core)
|
|
244
244
|
crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
|
|
245
245
|
crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
|
|
246
|
-
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
|
|
246
|
+
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
|
|
247
247
|
scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
|
|
248
248
|
tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
|
|
249
249
|
perf/ # cross-language benchmark harness and RESULTS.md
|
|
@@ -118,7 +118,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
|
|
|
118
118
|
|
|
119
119
|
`diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
|
|
120
120
|
|
|
121
|
-
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in
|
|
121
|
+
It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
|
|
122
122
|
|
|
123
123
|
```python
|
|
124
124
|
import pyarrow as pa
|
|
@@ -225,7 +225,7 @@ crates/onix-core # the diff engine (library, no I/O)
|
|
|
225
225
|
crates/onix-cli # the `onix` binary (thin CLI over the core)
|
|
226
226
|
crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
|
|
227
227
|
crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
|
|
228
|
-
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
|
|
228
|
+
docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
|
|
229
229
|
scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
|
|
230
230
|
tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
|
|
231
231
|
perf/ # cross-language benchmark harness and RESULTS.md
|
|
@@ -1,6 +1,5 @@
|
|
|
1
|
-
//! Per-pass wall-time and peak-RSS profile of one keyed row diff
|
|
2
|
-
//!
|
|
3
|
-
//! Builds only with the `profile` feature, which the release wheel never enables.
|
|
1
|
+
//! Per-pass wall-time and peak-RSS profile of one keyed row diff. Builds only
|
|
2
|
+
//! with the `profile` feature, which the release wheel never enables.
|
|
4
3
|
//!
|
|
5
4
|
//! Each invocation runs a discarded warm-up diff, a timed uninstrumented diff
|
|
6
5
|
//! (the `uninstrumented wall` line), and an instrumented diff whose passes make
|
|
@@ -9,8 +9,9 @@
|
|
|
9
9
|
//! tables use [`MemoryInput`]; a one-shot stream spools to a temporary file
|
|
10
10
|
//! and implements [`TableInput`] over it, as the Python bindings do.
|
|
11
11
|
//!
|
|
12
|
-
//! See `src/row_diff.rs` for the row-matching
|
|
13
|
-
//!
|
|
12
|
+
//! See `src/row_diff.rs` for the row-matching passes, `docs/design/row-diff.md`
|
|
13
|
+
//! for the algorithm, hashing, and value-comparison rules, and `src/schema.rs`
|
|
14
|
+
//! for the column type-normalization rules.
|
|
14
15
|
//!
|
|
15
16
|
//! # Example
|
|
16
17
|
//!
|
|
@@ -77,8 +78,9 @@ pub use schema::{ChangeKind, SchemaChange, diff_schemas};
|
|
|
77
78
|
pub use table_diff::{TableDiff, TableDiffSummary};
|
|
78
79
|
|
|
79
80
|
/// The maximum column-type nesting depth [`diff_tables`] will compare; deeper is refused
|
|
80
|
-
/// with [`TableDiffError::MaxDepthExceeded`],
|
|
81
|
-
/// comparison, `Display`, `Clone
|
|
81
|
+
/// with [`TableDiffError::MaxDepthExceeded`], bounding the native-stack recursion in
|
|
82
|
+
/// comparison, `Display`, `Clone`, and the drop of values onix builds from accepted
|
|
83
|
+
/// input — not a caller's own drop of a `DataType` it built past this depth.
|
|
82
84
|
/// Per-level cost is measured by `crates/onix-arrow/examples/type_stack_cost.rs`.
|
|
83
85
|
pub const MAX_NESTING_DEPTH: usize = 128;
|
|
84
86
|
|
|
@@ -1,152 +1,12 @@
|
|
|
1
|
-
//! Keyed row diff: which rows were added, removed, or changed between two
|
|
2
|
-
//!
|
|
3
|
-
//!
|
|
1
|
+
//! Keyed row diff: which rows were added, removed, or changed between two tables, matched
|
|
2
|
+
//! by a required primary key, in memory proportional to the row count rather than the data
|
|
3
|
+
//! size. See `docs/design/row-diff.md` for the algorithm, hashing, and value-semantics detail.
|
|
4
4
|
//!
|
|
5
|
-
//!
|
|
6
|
-
//!
|
|
7
|
-
//!
|
|
8
|
-
//!
|
|
9
|
-
//!
|
|
10
|
-
//! yields a keyed 128-bit hash of its key columns and a keyed 128-bit hash of
|
|
11
|
-
//! its non-key columns (the value semantics are in [`hash_cell`]). The pairs
|
|
12
|
-
//! are collected into one `(key_hash, row_hash)` vector per side — 32 bytes
|
|
13
|
-
//! per row, the only per-row state that grows with the input — while the
|
|
14
|
-
//! batches themselves are dropped as they are consumed. Set arithmetic on the
|
|
15
|
-
//! two sorted vectors then classifies every key: only on the left (removed),
|
|
16
|
-
//! only on the right (added), on both with different row hashes (changed), on
|
|
17
|
-
//! both with equal row hashes (unchanged, never materialized), or appearing
|
|
18
|
-
//! more than once on either side (a duplicate key, excluded from the other
|
|
19
|
-
//! three and reported with its per-side counts).
|
|
20
|
-
//! 2. **Materialize pass.** Each side is read again (a [`TableInput`] is
|
|
21
|
-
//! re-openable) and filtered to the rows whose keys landed in the added /
|
|
22
|
-
//! removed sets, plus one row per duplicate key for the duplicate-key report.
|
|
23
|
-
//! Only the differing rows are ever built into an output batch, and a kept
|
|
24
|
-
//! selection copies any buffer it shares with its decoded input batch (see
|
|
25
|
-
//! [`unshared`]).
|
|
26
|
-
//! 3. **Cell pass.** The rows whose key is in the *changed* set are paired by
|
|
27
|
-
//! key hash and every common non-key column is compared cell by cell, one
|
|
28
|
-
//! output record per differing cell (see [`diff_cells`]).
|
|
29
|
-
//!
|
|
30
|
-
//! Single-threaded, every pass re-reads both sides and the cell pass holds both
|
|
31
|
-
//! sides' changed rows at once. The parallel path reads the right side once: the
|
|
32
|
-
//! left is hashed and indexed first ([`KeyIndex`]), so the right's hash pass
|
|
33
|
-
//! tallies each row against the left key it matches instead of keeping its
|
|
34
|
-
//! hashes ([`classify_indexed`]), keeps its added candidates, and spills its
|
|
35
|
-
//! changed value rows by key-hash partition to anonymous temporary IPC files
|
|
36
|
-
//! ([`RightFuse`]); one re-read of the left then materializes it and spills its
|
|
37
|
-
//! changed rows ([`reread_left`]), and the cell pass compares and renders one
|
|
38
|
-
//! partition at a time across the workers ([`diff_cells_streaming`]), holding one
|
|
39
|
-
//! partition plus the output.
|
|
40
|
-
//!
|
|
41
|
-
//! # Parallelism
|
|
42
|
-
//!
|
|
43
|
-
//! By default the diff runs across `TableDiffOptions::threads` workers (the
|
|
44
|
-
//! machine's available parallelism, capped at [`crate::MAX_THREADS`]): the hash
|
|
45
|
-
//! passes hash each batch on a worker and append its rows straight into shared
|
|
46
|
-
//! per-key-hash partition buffers, the classify step merge-joins each partition
|
|
47
|
-
//! on its own worker, the left re-read classifies and routes each batch on a
|
|
48
|
-
//! worker while the order-dependent filtering runs in batch order, and the cell
|
|
49
|
-
//! pass renders each partition across the workers. The partition count is
|
|
50
|
-
//! capped independently of the worker count (see [`partition_count`] and
|
|
51
|
-
//! [`MAX_PARTITIONS`]). Every partitioning is by key hash and every reduction is
|
|
52
|
-
//! order-independent or reordered back to batch order, so the output is
|
|
53
|
-
//! byte-identical at any thread count; `threads == 1` runs the single-threaded
|
|
54
|
-
//! path. The choice is made by peeking up to [`MIN_PARALLEL_ROWS`] rows or
|
|
55
|
-
//! [`MAX_PEEK_BYTES`] of each side (whichever comes first) before spawning: a
|
|
56
|
-
//! diff whose sides both fit under that bound runs single-threaded, and the peek
|
|
57
|
-
//! reads the left side first so a large left never also buffers the right.
|
|
58
|
-
//!
|
|
59
|
-
//! Per-row state: 32 bytes a row per side single-threaded; in parallel, 32 on
|
|
60
|
-
//! the left plus an 8-byte tally and under a byte of bucket directory, and a
|
|
61
|
-
//! 32-byte map entry (first-row position and count) per right key absent from
|
|
62
|
-
//! the left. Beyond that: in-flight batches (workers times batch size), buffer
|
|
63
|
-
//! slack, the size gate's peek (at most [`MAX_PEEK_BYTES`] plus one producer
|
|
64
|
-
//! batch per side), and every distinct duplicated key's values. In parallel, a
|
|
65
|
-
//! key the left lacks keeps its first right row at full width until the key
|
|
66
|
-
//! repeats, and each right batch holding such a row keeps a candidate record of
|
|
67
|
-
//! those rows' key columns (shared with the rows until compaction copies them).
|
|
68
|
-
//! The first right row of a key the left holds once with another row hash is
|
|
69
|
-
//! spilled unless its batch repeats the key. A selection kept past its scan
|
|
70
|
-
//! keeps its whole input batch resident, per side, if it keeps over half of it;
|
|
71
|
-
//! a smaller one copies its buffers out, but byte-view data buffers reach the
|
|
72
|
-
//! output whole, so its view data stays. The duplicate-key report always copies
|
|
73
|
-
//! its key columns out. The cell pass holds a spill of every common value
|
|
74
|
-
//! column of every changed row, both sides (resident where written temp pages
|
|
75
|
-
//! count), and about twice the `cells_changed` output, not bounded by the
|
|
76
|
-
//! changed *cell* count. The README's Known-limitations bullet has the figures.
|
|
77
|
-
//!
|
|
78
|
-
//! # Hashing
|
|
79
|
-
//!
|
|
80
|
-
//! Row identity is a single keyed 128-bit SipHash-1-3 ([`siphasher`]), keyed
|
|
81
|
-
//! from 16 bytes of OS randomness ([`getrandom`]) drawn once per diff. Both
|
|
82
|
-
//! sides of one diff share the key, so their hashes are comparable; a different
|
|
83
|
-
//! diff draws a fresh key. Because the key is secret and random per run, the
|
|
84
|
-
//! row-matching table cannot be forced into collisions by chosen input, and no
|
|
85
|
-
//! unkeyed content hash table is used on this default (no-flag) path. Two
|
|
86
|
-
//! distinct keys colliding to the same 128-bit hash — the only way this can
|
|
87
|
-
//! misclassify — has probability on the order of `n² / 2¹²⁸`, negligible for
|
|
88
|
-
//! any real table.
|
|
89
|
-
//!
|
|
90
|
-
//! # Value semantics
|
|
91
|
-
//!
|
|
92
|
-
//! Cell hashing largely matches how onix's core compares scalars: integers and
|
|
93
|
-
//! integral floats within `±2⁵³` fold to one integer form (so `1`, `1.0`,
|
|
94
|
-
//! `-0.0`, and a dictionary-encoded `1` all hash equal), other floats hash by
|
|
95
|
-
//! their bit pattern, decimals (128- and 256-bit) hash by their exact value with
|
|
96
|
-
//! trailing zeros removed (so `1.00` equals `1.0000`), timestamps hash by their
|
|
97
|
-
//! UTC instant in nanoseconds (so the same instant at microsecond and
|
|
98
|
-
//! millisecond precision hashes equal), times and durations likewise normalize
|
|
99
|
-
//! to nanoseconds (so the same clock time or elapsed span at different units
|
|
100
|
-
//! hashes equal), and a null is a distinct value that equals only another null —
|
|
101
|
-
//! the `IS DISTINCT FROM` semantics the `DuckDB` oracle uses. One rule is this
|
|
102
|
-
//! crate's own, not `onix-core`'s (which refuses NaN at conversion): every NaN
|
|
103
|
-
//! folds to one canonical NaN, so no NaN-payload difference is a change, because
|
|
104
|
-
//! the renderer cannot show two NaN payloads apart.
|
|
105
|
-
//!
|
|
106
|
-
//! # Per-cell changes
|
|
107
|
-
//!
|
|
108
|
-
//! [`diff_cells`] reports, for every changed row, which cells differ, as one
|
|
109
|
-
//! output row per differing cell: the key columns, then `column`, `old_value`,
|
|
110
|
-
//! `new_value`, and `change` (see [`diff_cells`] for the exact output order).
|
|
111
|
-
//! A cell is reported changed **if and only if its [`hash_cell`] contribution
|
|
112
|
-
//! differs** between the two matched rows — the same helper the row hash is
|
|
113
|
-
//! built from, so the cell list and the row-changed decision can never drift.
|
|
114
|
-
//! Each reported cell is labelled:
|
|
115
|
-
//!
|
|
116
|
-
//! - `became_null`/`became_non_null` when exactly one side is null;
|
|
117
|
-
//! - `type_changed` when both are non-null and the two sides' types are not
|
|
118
|
-
//! losslessly comparable — their [`value_domain`]s differ (a number becoming a
|
|
119
|
-
//! string, a timestamp becoming a date), both are timestamps but one is
|
|
120
|
-
//! zone-aware and the other naive (different meaning at the same instant), or
|
|
121
|
-
//! both are intervals of different variants (which are not one span);
|
|
122
|
-
//! - `value_changed` otherwise: the same value domain, differing in value over
|
|
123
|
-
//! the hash's lossless normalization. This covers a lossless type change —
|
|
124
|
-
//! `Int32`→`Int64`, a float width change, a time/duration unit change, a
|
|
125
|
-
//! decimal scale change — whose equal values hash equal and so are *not*
|
|
126
|
-
//! reported at all, and are `value_changed` only when the value genuinely
|
|
127
|
-
//! differs.
|
|
128
|
-
//!
|
|
129
|
-
//! A column present on only one side is a schema change and never a cell change.
|
|
130
|
-
//! `old_value`/`new_value` are a canonical string rendering
|
|
131
|
-
//! ([`arrow_cast::display`]), null for a null cell, produced so that a
|
|
132
|
-
//! `value_changed` record can never carry two equal renderings: numbers of
|
|
133
|
-
//! differing width render at the wider type (an `f32` `0.1` shows as
|
|
134
|
-
//! `0.10000000149011612` against an `f64` `0.1`), a timestamp renders as its UTC
|
|
135
|
-
//! instant with its zone appended when aware (so an aware and a naive timestamp
|
|
136
|
-
//! of the same instant differ), a decimal renders at its native scale, a string
|
|
137
|
-
//! verbatim (decimals and strings match the `DuckDB` oracle), a duration
|
|
138
|
-
//! renders as an ISO 8601 `PT<seconds>S` string computed from its value — never
|
|
139
|
-
//! through the Arrow formatter, whose second/millisecond duration formatter can
|
|
140
|
-
//! emit a `<invalid>` sentinel while still succeeding — and a cross-variant
|
|
141
|
-
//! interval renders with its variant appended, so two variants whose human
|
|
142
|
-
//! form would otherwise coincide stay distinct (see [`prepare_render`]). As a
|
|
143
|
-
//! construction guard, a `value_changed` record whose two renderings are
|
|
144
|
-
//! nonetheless equal is a
|
|
145
|
-
//! [`TableDiffError::EqualRenderings`], not a silent row. There is no typed
|
|
146
|
-
//! old/new
|
|
147
|
-
//! column: a long-format table mixes every compared column's type in one column,
|
|
148
|
-
//! so a single typed column cannot represent them and the string rendering is
|
|
149
|
-
//! the uniform form.
|
|
5
|
+
//! - **Hash pass** — hashes every row's key and non-key columns; classifies keys by presence
|
|
6
|
+
//! and hash equality across the two sides.
|
|
7
|
+
//! - **Materialize pass** — re-reads each side, keeping only added/removed rows and one row
|
|
8
|
+
//! per duplicate key.
|
|
9
|
+
//! - **Cell pass** — pairs changed rows by key hash and reports each differing cell.
|
|
150
10
|
//!
|
|
151
11
|
//! # Which column types are hashed, refused, or skipped
|
|
152
12
|
//!
|
|
@@ -172,11 +32,6 @@
|
|
|
172
32
|
//! non-key column (`List` and its variants, `FixedSizeList`, `Struct`, `Map`,
|
|
173
33
|
//! `Union`), which is out of scope for the row diff. A nested *key* column is
|
|
174
34
|
//! refused.
|
|
175
|
-
//!
|
|
176
|
-
//! [`hash_cell`] itself is non-recursive; the two recursive walks here —
|
|
177
|
-
//! [`is_hashable`] and [`value_domain`], each over a dictionary value type — are
|
|
178
|
-
//! bounded by the [`crate::MAX_NESTING_DEPTH`] depth check
|
|
179
|
-
//! [`crate::diff_schemas`] runs before any row is read.
|
|
180
35
|
|
|
181
36
|
use std::collections::{BTreeMap, HashMap, HashSet};
|
|
182
37
|
use std::fs::File;
|
|
@@ -664,7 +519,8 @@ fn decimal_value(array: &ArrayRef, row: usize) -> Option<(i256, i8)> {
|
|
|
664
519
|
|
|
665
520
|
/// Hashes one cell into `hasher`, tagged by kind so unlike kinds never collide.
|
|
666
521
|
/// A null writes only [`TAG_NULL`]; every other kind writes its tag then a
|
|
667
|
-
/// canonical form of the value (see
|
|
522
|
+
/// canonical form of the value (see `docs/design/row-diff.md`'s "Value
|
|
523
|
+
/// semantics" section).
|
|
668
524
|
fn hash_cell(
|
|
669
525
|
hasher: &mut CellHasher,
|
|
670
526
|
array: &ArrayRef,
|
|
@@ -2629,11 +2485,12 @@ struct ColumnInputs<'a> {
|
|
|
2629
2485
|
}
|
|
2630
2486
|
|
|
2631
2487
|
/// Compares one column across the paired changed rows and appends a record for
|
|
2632
|
-
/// every differing cell. The change kind and rendering follow
|
|
2633
|
-
///
|
|
2634
|
-
/// change, a differing value over the
|
|
2635
|
-
/// change, and each side renders in
|
|
2636
|
-
/// `value_changed` record never carries equal
|
|
2488
|
+
/// every differing cell. The change kind and rendering follow
|
|
2489
|
+
/// `docs/design/row-diff.md`'s "Per-cell changes" section: a value domain or
|
|
2490
|
+
/// timestamp-awareness mismatch is a type change, a differing value over the
|
|
2491
|
+
/// hash's lossless normalization is a value change, and each side renders in
|
|
2492
|
+
/// the common comparison form so a `value_changed` record never carries equal
|
|
2493
|
+
/// renderings.
|
|
2637
2494
|
fn emit_column_records(
|
|
2638
2495
|
column: &ColumnInputs<'_>,
|
|
2639
2496
|
records: &mut Vec<CellRecord>,
|
|
@@ -4203,8 +4060,8 @@ impl CellColumns {
|
|
|
4203
4060
|
}
|
|
4204
4061
|
}
|
|
4205
4062
|
|
|
4206
|
-
/// Diffs the rows of two tables matched by `key`. See
|
|
4207
|
-
/// algorithm and value semantics.
|
|
4063
|
+
/// Diffs the rows of two tables matched by `key`. See `docs/design/row-diff.md`
|
|
4064
|
+
/// for the algorithm and value semantics.
|
|
4208
4065
|
pub(crate) fn diff_rows(
|
|
4209
4066
|
left: &impl TableInput,
|
|
4210
4067
|
right: &impl TableInput,
|
|
@@ -103,10 +103,10 @@ pub fn diff(a: &Value, b: &Value) -> Result<Report, Error> {
|
|
|
103
103
|
/// assert!(report.is_empty());
|
|
104
104
|
/// ```
|
|
105
105
|
pub fn diff_with_options(a: &Value, b: &Value, opts: &DiffOptions) -> Result<Report, Error> {
|
|
106
|
-
// The
|
|
107
|
-
//
|
|
108
|
-
//
|
|
109
|
-
//
|
|
106
|
+
// The memo is created here, per diff invocation, and dropped when this
|
|
107
|
+
// returns — no cross-call state. An ordered diff consults it only through
|
|
108
|
+
// set comparison (set-member and tuple digests); `ignore_order` also
|
|
109
|
+
// caches container-pair distances in it.
|
|
110
110
|
diff_with_options_memo(a, b, opts, &crate::ignore_order::IgnoreOrderMemo::new())
|
|
111
111
|
}
|
|
112
112
|
|
|
@@ -11,8 +11,6 @@ use crate::diff::DiffOptions;
|
|
|
11
11
|
|
|
12
12
|
use super::IgnoreOrderMemo;
|
|
13
13
|
|
|
14
|
-
use super::fxhash::HashSet;
|
|
15
|
-
|
|
16
14
|
/// A total-ordering wrapper for the non-negative, always-finite distances
|
|
17
15
|
/// [`rough_distance`] computes, so they can key a [`BTreeMap`](std::collections::BTreeMap) (ascending
|
|
18
16
|
/// iteration, for [`compute_pairs`](super::pairing::compute_pairs)'s greedy loop) and group candidates by
|
|
@@ -872,7 +870,9 @@ pub(crate) fn match_dict_keys<'a>(a: &'a Object, b: &'a Object) -> DictKeyMatch<
|
|
|
872
870
|
/// The shared `threshold_to_diff_deeper` ratio check backing both
|
|
873
871
|
/// [`count_object_diff_leaves`] (the count-only distance mirror) and
|
|
874
872
|
/// `crate::diff::object_diff`'s own unconditional collapse — see
|
|
875
|
-
/// [`THRESHOLD_TO_DIFF_DEEPER`]'s own doc for why both exist.
|
|
873
|
+
/// [`THRESHOLD_TO_DIFF_DEEPER`]'s own doc for why both exist. All-`str` keys
|
|
874
|
+
/// are counted by a merge of the two ascending key sequences: one key
|
|
875
|
+
/// comparison per step, no hashing.
|
|
876
876
|
pub(crate) fn is_below_threshold_to_diff_deeper(a: &Object, b: &Object) -> bool {
|
|
877
877
|
let (union_len, intersect_len) = if a.has_non_str_keys() || b.has_non_str_keys() {
|
|
878
878
|
let matched = match_dict_keys(a, b);
|
|
@@ -881,28 +881,24 @@ pub(crate) fn is_below_threshold_to_diff_deeper(a: &Object, b: &Object) -> bool
|
|
|
881
881
|
matched.shared.len(),
|
|
882
882
|
)
|
|
883
883
|
} else {
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
884
|
+
let (mut a_keys, mut b_keys) = (a.keys().peekable(), b.keys().peekable());
|
|
885
|
+
let mut intersect_len = 0;
|
|
886
|
+
while let (Some(a_key), Some(b_key)) = (a_keys.peek(), b_keys.peek()) {
|
|
887
|
+
match a_key.cmp(b_key) {
|
|
888
|
+
std::cmp::Ordering::Less => {
|
|
889
|
+
a_keys.next();
|
|
890
|
+
}
|
|
891
|
+
std::cmp::Ordering::Greater => {
|
|
892
|
+
b_keys.next();
|
|
893
|
+
}
|
|
894
|
+
std::cmp::Ordering::Equal => {
|
|
895
|
+
intersect_len += 1;
|
|
896
|
+
a_keys.next();
|
|
897
|
+
b_keys.next();
|
|
895
898
|
}
|
|
896
899
|
}
|
|
897
900
|
}
|
|
898
|
-
|
|
899
|
-
.keys()
|
|
900
|
-
.map(key_bytes)
|
|
901
|
-
.chain(b.keys().map(key_bytes))
|
|
902
|
-
.collect::<HashSet<_>>()
|
|
903
|
-
.len();
|
|
904
|
-
let intersect_len = a.keys().filter(|key| b.contains_key(key)).count();
|
|
905
|
-
(union_len, intersect_len)
|
|
901
|
+
(a.len() + b.len() - intersect_len, intersect_len)
|
|
906
902
|
};
|
|
907
903
|
#[allow(
|
|
908
904
|
clippy::cast_precision_loss,
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
//! `FxHash`: a small, fast, non-cryptographic hasher for this module's `HashMap`/`HashSet`s.
|
|
2
|
+
//! See [`FxHasher`]'s doc for the accepted `DoS` trade-off and which tables carry it.
|
|
3
|
+
|
|
4
|
+
use std::hash::BuildHasherDefault;
|
|
5
|
+
|
|
6
|
+
/// This module's [`HashMap`](std::collections::HashMap), keyed with [`FxHasher`].
|
|
7
|
+
pub(crate) type HashMap<K, V> = std::collections::HashMap<K, V, BuildHasherDefault<FxHasher>>;
|
|
8
|
+
/// The [`HashMap`] equivalent for [`std::collections::HashSet`].
|
|
9
|
+
pub(crate) type HashSet<T> = std::collections::HashSet<T, BuildHasherDefault<FxHasher>>;
|
|
10
|
+
|
|
11
|
+
/// The `FxHash` algorithm (the one `rustc` itself uses for compiler-hot-path hash maps),
|
|
12
|
+
/// implemented from scratch.
|
|
13
|
+
///
|
|
14
|
+
/// # `DoS` trade-off (this hasher is *not* collision-resistant)
|
|
15
|
+
///
|
|
16
|
+
/// `FxHash` uses a fixed, public seed ([`FX_SEED`]) and an invertible step: a crafted
|
|
17
|
+
/// collision degrades an `FxHash` map or set from `O(1)` to `O(n)` per operation, pushing
|
|
18
|
+
/// [`HashedList::build`](super::hash::HashedList::build) from `O(n)` to `O(n²)` on an
|
|
19
|
+
/// all-colliding list, on top of the module's `O(N²)` pairing.
|
|
20
|
+
/// [`HashedList`](super::hash::HashedList), `AddedCandidates`, the pairing/`used` sets and
|
|
21
|
+
/// its result maps (`most_in_common_pairs`, `pairs`), `consumed_removed`, and the distance
|
|
22
|
+
/// memo are `FxHash`-keyed and reached only under `ignore_order=true`: an accepted,
|
|
23
|
+
/// documented `DoS` trade-off — `SipHash` there cost a measured per-call penalty on the
|
|
24
|
+
/// pairing hot path (PR #4). No default-path table is `FxHash`-keyed:
|
|
25
|
+
/// [`IgnoreOrderMemo`](super::memo::IgnoreOrderMemo)'s `tuple_ids`, which set comparison
|
|
26
|
+
/// reaches through `set_member_digest` -> `tuple_keyed` -> `tuple_digest` for a tuple dict
|
|
27
|
+
/// key inside a set member, is a `BTreeMap` (below), and
|
|
28
|
+
/// [`is_below_threshold_to_diff_deeper`](super::distance::is_below_threshold_to_diff_deeper),
|
|
29
|
+
/// called by `object_diff` for every unequal dict pair, counts dict-key overlap by a sorted
|
|
30
|
+
/// merge, with no table. Bound untrusted input against the module's `O(N²)` pairing
|
|
31
|
+
/// regardless of hasher.
|
|
32
|
+
///
|
|
33
|
+
/// Every `FxHash`-keyed type carrying a float ([`ItemKey`](super::hash::ItemKey),
|
|
34
|
+
/// [`ScalarKey`](crate::lcs::ScalarKey), and the distance memo's
|
|
35
|
+
/// [`DistKey`](super::hash::DistKey) via `number_key`) mixes its bits first
|
|
36
|
+
/// ([`mix_float_bits`](crate::lcs::mix_float_bits)), so integral and half-integer floats —
|
|
37
|
+
/// whose raw bit patterns share many trailing zeros — do not collide by accident. Every `NaN`
|
|
38
|
+
/// additionally folds onto one fixed key, matching `DeepHash`'s `str()`-based digest: this
|
|
39
|
+
/// changes which values these tables treat as the same item, not their per-lookup cost.
|
|
40
|
+
///
|
|
41
|
+
/// `node_table`, `member_content` and `tuple_ids` in `IgnoreOrderMemo` are `BTreeMap`s instead,
|
|
42
|
+
/// since they are keyed by attacker-controlled member content and reached on the default path
|
|
43
|
+
/// too, with no `Hash` derive: `O(log n)` worst case, always, though each comparison still
|
|
44
|
+
/// walks the whole probed key — `member_content`'s `MemberContent::UnhashableDict` key is
|
|
45
|
+
/// itself keyed by each dict key's own `ItemKey` tree, not a flat string.
|
|
46
|
+
///
|
|
47
|
+
/// An integer beyond `i128` (`ItemKey::BigInt`) hashes and compares by its magnitude digits,
|
|
48
|
+
/// `O(digits)` not `O(1)` per lookup — inside a hashable tuple it reaches the same cost via
|
|
49
|
+
/// `ScalarKey::Big` and `tuple_ids`, through `PyHashPart::Scalar` — a cost the `ignore_order`
|
|
50
|
+
/// element-count cap does not bound: one huge integer costs its own digit length regardless
|
|
51
|
+
/// of element count. A custom object's `ItemKey::Object` costs `ItemKey::Dict`'s full walk
|
|
52
|
+
/// plus one class-name comparison, and two distinct classes sharing a `__name__` share its
|
|
53
|
+
/// bucket; `ItemKey::Opaque` costs one identity string comparison.
|
|
54
|
+
#[derive(Default)]
|
|
55
|
+
pub(crate) struct FxHasher {
|
|
56
|
+
pub(crate) hash: u64,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/// `FxHash`'s seed constant (golden-ratio-derived odd constant).
|
|
60
|
+
pub(crate) const FX_SEED: u64 = 0x51_7c_c1_b7_27_22_0a_95;
|
|
61
|
+
|
|
62
|
+
impl FxHasher {
|
|
63
|
+
/// Folds one word into the hash: rotate, xor, multiply — `FxHash`'s mixing step.
|
|
64
|
+
pub(crate) fn add_to_hash(&mut self, word: u64) {
|
|
65
|
+
self.hash = (self.hash.rotate_left(5) ^ word).wrapping_mul(FX_SEED);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
impl std::hash::Hasher for FxHasher {
|
|
70
|
+
fn write(&mut self, mut bytes: &[u8]) {
|
|
71
|
+
while let Some(chunk) = bytes.get(..8) {
|
|
72
|
+
let word = u64::from_ne_bytes(chunk.try_into().expect("chunk is exactly 8 bytes"));
|
|
73
|
+
self.add_to_hash(word);
|
|
74
|
+
bytes = &bytes[8..];
|
|
75
|
+
}
|
|
76
|
+
for &byte in bytes {
|
|
77
|
+
self.add_to_hash(u64::from(byte));
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
fn write_u8(&mut self, i: u8) {
|
|
82
|
+
self.add_to_hash(u64::from(i));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
fn write_u16(&mut self, i: u16) {
|
|
86
|
+
self.add_to_hash(u64::from(i));
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
fn write_u32(&mut self, i: u32) {
|
|
90
|
+
self.add_to_hash(u64::from(i));
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
fn write_u64(&mut self, i: u64) {
|
|
94
|
+
self.add_to_hash(i);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
fn write_u128(&mut self, i: u128) {
|
|
98
|
+
#[allow(
|
|
99
|
+
clippy::cast_possible_truncation,
|
|
100
|
+
reason = "hash mixing only, truncation does not affect correctness, only distribution"
|
|
101
|
+
)]
|
|
102
|
+
{
|
|
103
|
+
self.add_to_hash(i as u64);
|
|
104
|
+
self.add_to_hash((i >> 64) as u64);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
fn write_usize(&mut self, i: usize) {
|
|
109
|
+
self.add_to_hash(i as u64);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
fn finish(&self) -> u64 {
|
|
113
|
+
self.hash
|
|
114
|
+
}
|
|
115
|
+
}
|
|
@@ -329,7 +329,7 @@ impl std::hash::Hash for ItemKey {
|
|
|
329
329
|
/// Referring to a nested tuple by id rather than by its whole identity is
|
|
330
330
|
/// what keeps this `O(arity)` per node: `((((1,),),),)` is four one-element
|
|
331
331
|
/// identities, not four identities of size 1, 2, 3 and 4.
|
|
332
|
-
#[derive(Debug, Clone, PartialEq, Eq,
|
|
332
|
+
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
|
|
333
333
|
pub(crate) enum PyHashPart {
|
|
334
334
|
Scalar(ScalarKey),
|
|
335
335
|
Tuple(TupleId),
|
|
@@ -345,12 +345,12 @@ pub(crate) enum PyHashPart {
|
|
|
345
345
|
/// ([`python_scalar_key`], which makes `1`, `1.0` and `True` one key), and a
|
|
346
346
|
/// list or dict is unhashable, which also makes any tuple containing one
|
|
347
347
|
/// unhashable.
|
|
348
|
-
#[derive(Debug, Clone, PartialEq, Eq,
|
|
348
|
+
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
|
|
349
349
|
pub(crate) struct PyHashKey(Box<[PyHashPart]>);
|
|
350
350
|
|
|
351
351
|
/// A hashable tuple identity's place in the run's interning table — see
|
|
352
352
|
/// [`super::IgnoreOrderMemo::tuple_digest`].
|
|
353
|
-
#[derive(Debug, Clone, Copy, PartialEq, Eq,
|
|
353
|
+
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
|
354
354
|
pub(crate) struct TupleId(usize);
|
|
355
355
|
|
|
356
356
|
impl TupleId {
|
|
@@ -24,14 +24,17 @@ type DistanceKey = (DistKey, DistKey);
|
|
|
24
24
|
/// "Distance memo" section: pairwise container distances, tuple
|
|
25
25
|
/// digests, and set-member digests, scoped to one diff run. `DistKey`
|
|
26
26
|
/// keys hash and compare by a full-tree walk; `MemberContent` keys
|
|
27
|
-
/// compare by one (it has no `Hash`)
|
|
27
|
+
/// compare by one (it has no `Hash`); a `PyHashKey` lookup makes `O(log n)`
|
|
28
|
+
/// comparisons, each `O(the probed key's element bytes)`, a nested tuple
|
|
29
|
+
/// comparing as one id. A big-integer leaf costs
|
|
28
30
|
/// `O(digits)` per lookup; `super::fxhash`'s doc enumerates each key's cost.
|
|
29
31
|
pub(crate) struct IgnoreOrderMemo<'r> {
|
|
30
32
|
cache: RefCell<HashMap<DistanceKey, f64>>,
|
|
31
33
|
/// Interns each hashable-tuple identity to its digest, shared
|
|
32
34
|
/// across the run — see `docs/design/ignore-order.md`'s "Distance
|
|
33
|
-
/// memo" section.
|
|
34
|
-
|
|
35
|
+
/// memo" section. A `BTreeMap`, reached on the default path through a
|
|
36
|
+
/// set member's tuple dict keys; its key type carries no `Hash` derive.
|
|
37
|
+
tuple_ids: RefCell<BTreeMap<PyHashKey, TupleId>>,
|
|
35
38
|
/// The digest assigned to each interned identity, indexed by
|
|
36
39
|
/// [`TupleId::index`].
|
|
37
40
|
tuple_digests: RefCell<Vec<ItemKey>>,
|
|
@@ -60,7 +63,7 @@ impl<'r> IgnoreOrderMemo<'r> {
|
|
|
60
63
|
pub(crate) fn new() -> Self {
|
|
61
64
|
Self {
|
|
62
65
|
cache: RefCell::new(HashMap::default()),
|
|
63
|
-
tuple_ids: RefCell::new(
|
|
66
|
+
tuple_ids: RefCell::new(BTreeMap::new()),
|
|
64
67
|
tuple_digests: RefCell::new(Vec::new()),
|
|
65
68
|
node_table: RefCell::new(BTreeMap::new()),
|
|
66
69
|
member_content: RefCell::new(BTreeMap::new()),
|
|
@@ -78,7 +81,7 @@ impl<'r> IgnoreOrderMemo<'r> {
|
|
|
78
81
|
pub(crate) fn disabled() -> Self {
|
|
79
82
|
Self {
|
|
80
83
|
cache: RefCell::new(HashMap::default()),
|
|
81
|
-
tuple_ids: RefCell::new(
|
|
84
|
+
tuple_ids: RefCell::new(BTreeMap::new()),
|
|
82
85
|
tuple_digests: RefCell::new(Vec::new()),
|
|
83
86
|
node_table: RefCell::new(BTreeMap::new()),
|
|
84
87
|
member_content: RefCell::new(BTreeMap::new()),
|