deepdiff-rs 0.13.0__tar.gz → 0.13.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/Cargo.lock +4 -4
  2. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/Cargo.toml +1 -1
  3. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/PKG-INFO +3 -3
  4. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/README.md +2 -2
  5. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/row_diff_profile.rs +2 -3
  6. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/lib.rs +6 -4
  7. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/row_diff.rs +18 -161
  8. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/options.rs +4 -4
  9. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/distance.rs +18 -22
  10. deepdiff_rs-0.13.1/crates/onix-core/src/ignore_order/fxhash.rs +115 -0
  11. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/hash.rs +3 -3
  12. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/memo.rs +8 -5
  13. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/tests.rs +180 -0
  14. deepdiff_rs-0.13.1/crates/onix-py/tests/test_default_path_hashing.py +80 -0
  15. deepdiff_rs-0.13.0/crates/onix-core/src/ignore_order/fxhash.rs +0 -165
  16. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/Cargo.toml +0 -0
  17. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/row_diff_rss.rs +0 -0
  18. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/shared/gen_shapes.rs +0 -0
  19. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
  20. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/error.rs +0 -0
  21. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/json_rows.rs +0 -0
  22. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/options.rs +0 -0
  23. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/profile.rs +0 -0
  24. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/schema.rs +0 -0
  25. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/spool.rs +0 -0
  26. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/src/table_diff.rs +0 -0
  27. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-arrow/tests/profile_passes.rs +0 -0
  28. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/Cargo.toml +0 -0
  29. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
  30. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/datetime.rs +0 -0
  31. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/datetime_tests.rs +0 -0
  32. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/array.rs +0 -0
  33. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/dispatch.rs +0 -0
  34. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/mod.rs +0 -0
  35. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/object.rs +0 -0
  36. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/scalar.rs +0 -0
  37. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/set.rs +0 -0
  38. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/diff/tests.rs +0 -0
  39. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/error.rs +0 -0
  40. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/mod.rs +0 -0
  41. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/ignore_order/pairing.rs +0 -0
  42. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lcs.rs +0 -0
  43. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lcs_tests.rs +0 -0
  44. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/lib.rs +0 -0
  45. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/path.rs +0 -0
  46. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/report.rs +0 -0
  47. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/report_tests.rs +0 -0
  48. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/test_support.rs +0 -0
  49. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/unified_diff.rs +0 -0
  50. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/unified_diff_tests.rs +0 -0
  51. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/value.rs +0 -0
  52. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/src/value_tests.rs +0 -0
  53. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/golden.rs +0 -0
  54. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
  55. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/memory_footprint.rs +0 -0
  56. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/proptest_diff.rs +0 -0
  57. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
  58. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/.python-version +0 -0
  59. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/Cargo.toml +0 -0
  60. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/benchmarks/bench_bindings.py +0 -0
  61. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/deepdiff_rs.pyi +0 -0
  62. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/arrow.rs +0 -0
  63. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/convert.rs +0 -0
  64. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/deepdiff.rs +0 -0
  65. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/errors.rs +0 -0
  66. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/fast_path.rs +0 -0
  67. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/guard.rs +0 -0
  68. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/src/lib.rs +0 -0
  69. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/conftest.py +0 -0
  70. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_bindings_memory.py +0 -0
  71. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_conversions.py +0 -0
  72. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_datetimes.py +0 -0
  73. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_depth_guard.py +0 -0
  74. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_differential_fuzz.py +0 -0
  75. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_golden_parity.py +0 -0
  76. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_non_finite.py +0 -0
  77. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_sets.py +0 -0
  78. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_signed_zero.py +0 -0
  79. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_smoke.py +0 -0
  80. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_stub_mypy.py +0 -0
  81. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_stub_signatures.py +0 -0
  82. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_suite_hygiene.py +0 -0
  83. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_table_diff.py +0 -0
  84. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_table_row_diff.py +0 -0
  85. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_timedeltas.py +0 -0
  86. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_times.py +0 -0
  87. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_tuples.py +0 -0
  88. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/crates/onix-py/tests/test_wheel_contents.py +0 -0
  89. {deepdiff_rs-0.13.0 → deepdiff_rs-0.13.1}/pyproject.toml +0 -0
@@ -629,7 +629,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
629
629
 
630
630
  [[package]]
631
631
  name = "onix-arrow"
632
- version = "0.13.0"
632
+ version = "0.13.1"
633
633
  dependencies = [
634
634
  "arrow-array",
635
635
  "arrow-buffer",
@@ -648,7 +648,7 @@ dependencies = [
648
648
 
649
649
  [[package]]
650
650
  name = "onix-cli"
651
- version = "0.13.0"
651
+ version = "0.13.1"
652
652
  dependencies = [
653
653
  "onix-core",
654
654
  "serde_json",
@@ -656,7 +656,7 @@ dependencies = [
656
656
 
657
657
  [[package]]
658
658
  name = "onix-core"
659
- version = "0.13.0"
659
+ version = "0.13.1"
660
660
  dependencies = [
661
661
  "num-bigint",
662
662
  "num-traits",
@@ -669,7 +669,7 @@ dependencies = [
669
669
 
670
670
  [[package]]
671
671
  name = "onix-py"
672
- version = "0.13.0"
672
+ version = "0.13.1"
673
673
  dependencies = [
674
674
  "arrow-array",
675
675
  "arrow-schema",
@@ -3,7 +3,7 @@ resolver = "3"
3
3
  members = ["crates/onix-core", "crates/onix-py", "crates/onix-arrow"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.13.0"
6
+ version = "0.13.1"
7
7
  edition = "2024"
8
8
  license = "MIT"
9
9
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deepdiff-rs
3
- Version: 0.13.0
3
+ Version: 0.13.1
4
4
  Classifier: Programming Language :: Python :: 3
5
5
  Classifier: Programming Language :: Rust
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -136,7 +136,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
136
136
 
137
137
  `diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
138
138
 
139
- It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
139
+ It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
140
140
 
141
141
  ```python
142
142
  import pyarrow as pa
@@ -243,7 +243,7 @@ crates/onix-core # the diff engine (library, no I/O)
243
243
  crates/onix-cli # the `onix` binary (thin CLI over the core)
244
244
  crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
245
245
  crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
246
- docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
246
+ docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
247
247
  scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
248
248
  tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
249
249
  perf/ # cross-language benchmark harness and RESULTS.md
@@ -118,7 +118,7 @@ Pass `--ignore-order` to compare every list by value instead of by position, mir
118
118
 
119
119
  `diff_tables` compares two tables the way `DeepDiff` compares two objects. It takes any object implementing the [Arrow PyCapsule interface](https://arrow.apache.org/docs/format/CDataInterface/PyCapsuleInterface.html) — a pyarrow `Table` or `RecordBatch`, a polars `DataFrame`, a DuckDB relation — and imports it with no Python round trip. The two tables are matched on a required, non-empty set of key columns (the table's primary key).
120
120
 
121
- It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in the module doc of [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs). Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
121
+ It reports the **schema** diff (which columns were added, removed, or changed type), the keyed **row** diff (which rows were added, removed, or changed, and which keys are duplicated), and the per-cell diff (`cells_changed`): one row per changed cell, carrying the key columns, `column`, `old_value`/`new_value`, and `change` (`became_null`/`became_non_null`, `type_changed`, or `value_changed`), ordered by the canonical string rendering of the key columns (nulls first), then left-schema column order — the exact rendering and change-classification rules are in [`docs/design/row-diff.md`](docs/design/row-diff.md)'s "Per-cell changes" section. Rows are matched by the key columns; `rows_added`, `rows_removed`, `cells_changed`, and `duplicate_keys` return Arrow tables, and `summary()` counts each outcome.
122
122
 
123
123
  ```python
124
124
  import pyarrow as pa
@@ -225,7 +225,7 @@ crates/onix-core # the diff engine (library, no I/O)
225
225
  crates/onix-cli # the `onix` binary (thin CLI over the core)
226
226
  crates/onix-arrow # Arrow table diffing (schema diff and keyed row diff)
227
227
  crates/onix-py # PyO3 bindings, published as `deepdiff-rs`
228
- docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget)
228
+ docs/design/ # algorithm/invariant reference pages (list-diff, ignore-order, value-model, depth-budget, row-diff)
229
229
  scripts/ # gen_goldens.py: regenerates tests/golden/ from real DeepDiff
230
230
  tests/golden # DeepDiff-generated expected outputs (the compatibility corpus)
231
231
  perf/ # cross-language benchmark harness and RESULTS.md
@@ -1,6 +1,5 @@
1
- //! Per-pass wall-time and peak-RSS profile of one keyed row diff, the committed
2
- //! harness every row-diff performance change posts a before/after table from.
3
- //! Builds only with the `profile` feature, which the release wheel never enables.
1
+ //! Per-pass wall-time and peak-RSS profile of one keyed row diff. Builds only
2
+ //! with the `profile` feature, which the release wheel never enables.
4
3
  //!
5
4
  //! Each invocation runs a discarded warm-up diff, a timed uninstrumented diff
6
5
  //! (the `uninstrumented wall` line), and an instrumented diff whose passes make
@@ -9,8 +9,9 @@
9
9
  //! tables use [`MemoryInput`]; a one-shot stream spools to a temporary file
10
10
  //! and implements [`TableInput`] over it, as the Python bindings do.
11
11
  //!
12
- //! See `src/row_diff.rs` for the row-matching and value-comparison rules, and
13
- //! `src/schema.rs` for the column type-normalization rules.
12
+ //! See `src/row_diff.rs` for the row-matching passes, `docs/design/row-diff.md`
13
+ //! for the algorithm, hashing, and value-comparison rules, and `src/schema.rs`
14
+ //! for the column type-normalization rules.
14
15
  //!
15
16
  //! # Example
16
17
  //!
@@ -77,8 +78,9 @@ pub use schema::{ChangeKind, SchemaChange, diff_schemas};
77
78
  pub use table_diff::{TableDiff, TableDiffSummary};
78
79
 
79
80
  /// The maximum column-type nesting depth [`diff_tables`] will compare; deeper is refused
80
- /// with [`TableDiffError::MaxDepthExceeded`], turning a native-stack overflow (recursive
81
- /// comparison, `Display`, `Clone`/`Drop`) into an error before it can run.
81
+ /// with [`TableDiffError::MaxDepthExceeded`], bounding the native-stack recursion in
82
+ /// comparison, `Display`, `Clone`, and the drop of values onix builds from accepted
83
+ /// input — not a caller's own drop of a `DataType` it built past this depth.
82
84
  /// Per-level cost is measured by `crates/onix-arrow/examples/type_stack_cost.rs`.
83
85
  pub const MAX_NESTING_DEPTH: usize = 128;
84
86
 
@@ -1,152 +1,12 @@
1
- //! Keyed row diff: which rows were added, removed, or changed between two
2
- //! tables, matched by a required primary key, in memory proportional to the
3
- //! row count rather than the data size.
1
+ //! Keyed row diff: which rows were added, removed, or changed between two tables, matched
2
+ //! by a required primary key, in memory proportional to the row count rather than the data
3
+ //! size. See `docs/design/row-diff.md` for the algorithm, hashing, and value-semantics detail.
4
4
  //!
5
- //! # Algorithm
6
- //!
7
- //! Three passes, so the full decoded tables never sit in memory at once:
8
- //!
9
- //! 1. **Hash pass.** Each side is opened and streamed batch by batch. Every row
10
- //! yields a keyed 128-bit hash of its key columns and a keyed 128-bit hash of
11
- //! its non-key columns (the value semantics are in [`hash_cell`]). The pairs
12
- //! are collected into one `(key_hash, row_hash)` vector per side — 32 bytes
13
- //! per row, the only per-row state that grows with the input — while the
14
- //! batches themselves are dropped as they are consumed. Set arithmetic on the
15
- //! two sorted vectors then classifies every key: only on the left (removed),
16
- //! only on the right (added), on both with different row hashes (changed), on
17
- //! both with equal row hashes (unchanged, never materialized), or appearing
18
- //! more than once on either side (a duplicate key, excluded from the other
19
- //! three and reported with its per-side counts).
20
- //! 2. **Materialize pass.** Each side is read again (a [`TableInput`] is
21
- //! re-openable) and filtered to the rows whose keys landed in the added /
22
- //! removed sets, plus one row per duplicate key for the duplicate-key report.
23
- //! Only the differing rows are ever built into an output batch, and a kept
24
- //! selection copies any buffer it shares with its decoded input batch (see
25
- //! [`unshared`]).
26
- //! 3. **Cell pass.** The rows whose key is in the *changed* set are paired by
27
- //! key hash and every common non-key column is compared cell by cell, one
28
- //! output record per differing cell (see [`diff_cells`]).
29
- //!
30
- //! Single-threaded, every pass re-reads both sides and the cell pass holds both
31
- //! sides' changed rows at once. The parallel path reads the right side once: the
32
- //! left is hashed and indexed first ([`KeyIndex`]), so the right's hash pass
33
- //! tallies each row against the left key it matches instead of keeping its
34
- //! hashes ([`classify_indexed`]), keeps its added candidates, and spills its
35
- //! changed value rows by key-hash partition to anonymous temporary IPC files
36
- //! ([`RightFuse`]); one re-read of the left then materializes it and spills its
37
- //! changed rows ([`reread_left`]), and the cell pass compares and renders one
38
- //! partition at a time across the workers ([`diff_cells_streaming`]), holding one
39
- //! partition plus the output.
40
- //!
41
- //! # Parallelism
42
- //!
43
- //! By default the diff runs across `TableDiffOptions::threads` workers (the
44
- //! machine's available parallelism, capped at [`crate::MAX_THREADS`]): the hash
45
- //! passes hash each batch on a worker and append its rows straight into shared
46
- //! per-key-hash partition buffers, the classify step merge-joins each partition
47
- //! on its own worker, the left re-read classifies and routes each batch on a
48
- //! worker while the order-dependent filtering runs in batch order, and the cell
49
- //! pass renders each partition across the workers. The partition count is
50
- //! capped independently of the worker count (see [`partition_count`] and
51
- //! [`MAX_PARTITIONS`]). Every partitioning is by key hash and every reduction is
52
- //! order-independent or reordered back to batch order, so the output is
53
- //! byte-identical at any thread count; `threads == 1` runs the single-threaded
54
- //! path. The choice is made by peeking up to [`MIN_PARALLEL_ROWS`] rows or
55
- //! [`MAX_PEEK_BYTES`] of each side (whichever comes first) before spawning: a
56
- //! diff whose sides both fit under that bound runs single-threaded, and the peek
57
- //! reads the left side first so a large left never also buffers the right.
58
- //!
59
- //! Per-row state: 32 bytes a row per side single-threaded; in parallel, 32 on
60
- //! the left plus an 8-byte tally and under a byte of bucket directory, and a
61
- //! 32-byte map entry (first-row position and count) per right key absent from
62
- //! the left. Beyond that: in-flight batches (workers times batch size), buffer
63
- //! slack, the size gate's peek (at most [`MAX_PEEK_BYTES`] plus one producer
64
- //! batch per side), and every distinct duplicated key's values. In parallel, a
65
- //! key the left lacks keeps its first right row at full width until the key
66
- //! repeats, and each right batch holding such a row keeps a candidate record of
67
- //! those rows' key columns (shared with the rows until compaction copies them).
68
- //! The first right row of a key the left holds once with another row hash is
69
- //! spilled unless its batch repeats the key. A selection kept past its scan
70
- //! keeps its whole input batch resident, per side, if it keeps over half of it;
71
- //! a smaller one copies its buffers out, but byte-view data buffers reach the
72
- //! output whole, so its view data stays. The duplicate-key report always copies
73
- //! its key columns out. The cell pass holds a spill of every common value
74
- //! column of every changed row, both sides (resident where written temp pages
75
- //! count), and about twice the `cells_changed` output, not bounded by the
76
- //! changed *cell* count. The README's Known-limitations bullet has the figures.
77
- //!
78
- //! # Hashing
79
- //!
80
- //! Row identity is a single keyed 128-bit SipHash-1-3 ([`siphasher`]), keyed
81
- //! from 16 bytes of OS randomness ([`getrandom`]) drawn once per diff. Both
82
- //! sides of one diff share the key, so their hashes are comparable; a different
83
- //! diff draws a fresh key. Because the key is secret and random per run, the
84
- //! row-matching table cannot be forced into collisions by chosen input, and no
85
- //! unkeyed content hash table is used on this default (no-flag) path. Two
86
- //! distinct keys colliding to the same 128-bit hash — the only way this can
87
- //! misclassify — has probability on the order of `n² / 2¹²⁸`, negligible for
88
- //! any real table.
89
- //!
90
- //! # Value semantics
91
- //!
92
- //! Cell hashing largely matches how onix's core compares scalars: integers and
93
- //! integral floats within `±2⁵³` fold to one integer form (so `1`, `1.0`,
94
- //! `-0.0`, and a dictionary-encoded `1` all hash equal), other floats hash by
95
- //! their bit pattern, decimals (128- and 256-bit) hash by their exact value with
96
- //! trailing zeros removed (so `1.00` equals `1.0000`), timestamps hash by their
97
- //! UTC instant in nanoseconds (so the same instant at microsecond and
98
- //! millisecond precision hashes equal), times and durations likewise normalize
99
- //! to nanoseconds (so the same clock time or elapsed span at different units
100
- //! hashes equal), and a null is a distinct value that equals only another null —
101
- //! the `IS DISTINCT FROM` semantics the `DuckDB` oracle uses. One rule is this
102
- //! crate's own, not `onix-core`'s (which refuses NaN at conversion): every NaN
103
- //! folds to one canonical NaN, so no NaN-payload difference is a change, because
104
- //! the renderer cannot show two NaN payloads apart.
105
- //!
106
- //! # Per-cell changes
107
- //!
108
- //! [`diff_cells`] reports, for every changed row, which cells differ, as one
109
- //! output row per differing cell: the key columns, then `column`, `old_value`,
110
- //! `new_value`, and `change` (see [`diff_cells`] for the exact output order).
111
- //! A cell is reported changed **if and only if its [`hash_cell`] contribution
112
- //! differs** between the two matched rows — the same helper the row hash is
113
- //! built from, so the cell list and the row-changed decision can never drift.
114
- //! Each reported cell is labelled:
115
- //!
116
- //! - `became_null`/`became_non_null` when exactly one side is null;
117
- //! - `type_changed` when both are non-null and the two sides' types are not
118
- //! losslessly comparable — their [`value_domain`]s differ (a number becoming a
119
- //! string, a timestamp becoming a date), both are timestamps but one is
120
- //! zone-aware and the other naive (different meaning at the same instant), or
121
- //! both are intervals of different variants (which are not one span);
122
- //! - `value_changed` otherwise: the same value domain, differing in value over
123
- //! the hash's lossless normalization. This covers a lossless type change —
124
- //! `Int32`→`Int64`, a float width change, a time/duration unit change, a
125
- //! decimal scale change — whose equal values hash equal and so are *not*
126
- //! reported at all, and are `value_changed` only when the value genuinely
127
- //! differs.
128
- //!
129
- //! A column present on only one side is a schema change and never a cell change.
130
- //! `old_value`/`new_value` are a canonical string rendering
131
- //! ([`arrow_cast::display`]), null for a null cell, produced so that a
132
- //! `value_changed` record can never carry two equal renderings: numbers of
133
- //! differing width render at the wider type (an `f32` `0.1` shows as
134
- //! `0.10000000149011612` against an `f64` `0.1`), a timestamp renders as its UTC
135
- //! instant with its zone appended when aware (so an aware and a naive timestamp
136
- //! of the same instant differ), a decimal renders at its native scale, a string
137
- //! verbatim (decimals and strings match the `DuckDB` oracle), a duration
138
- //! renders as an ISO 8601 `PT<seconds>S` string computed from its value — never
139
- //! through the Arrow formatter, whose second/millisecond duration formatter can
140
- //! emit a `<invalid>` sentinel while still succeeding — and a cross-variant
141
- //! interval renders with its variant appended, so two variants whose human
142
- //! form would otherwise coincide stay distinct (see [`prepare_render`]). As a
143
- //! construction guard, a `value_changed` record whose two renderings are
144
- //! nonetheless equal is a
145
- //! [`TableDiffError::EqualRenderings`], not a silent row. There is no typed
146
- //! old/new
147
- //! column: a long-format table mixes every compared column's type in one column,
148
- //! so a single typed column cannot represent them and the string rendering is
149
- //! the uniform form.
5
+ //! - **Hash pass** — hashes every row's key and non-key columns; classifies keys by presence
6
+ //! and hash equality across the two sides.
7
+ //! - **Materialize pass** — re-reads each side, keeping only added/removed rows and one row
8
+ //! per duplicate key.
9
+ //! - **Cell pass** — pairs changed rows by key hash and reports each differing cell.
150
10
  //!
151
11
  //! # Which column types are hashed, refused, or skipped
152
12
  //!
@@ -172,11 +32,6 @@
172
32
  //! non-key column (`List` and its variants, `FixedSizeList`, `Struct`, `Map`,
173
33
  //! `Union`), which is out of scope for the row diff. A nested *key* column is
174
34
  //! refused.
175
- //!
176
- //! [`hash_cell`] itself is non-recursive; the two recursive walks here —
177
- //! [`is_hashable`] and [`value_domain`], each over a dictionary value type — are
178
- //! bounded by the [`crate::MAX_NESTING_DEPTH`] depth check
179
- //! [`crate::diff_schemas`] runs before any row is read.
180
35
 
181
36
  use std::collections::{BTreeMap, HashMap, HashSet};
182
37
  use std::fs::File;
@@ -664,7 +519,8 @@ fn decimal_value(array: &ArrayRef, row: usize) -> Option<(i256, i8)> {
664
519
 
665
520
  /// Hashes one cell into `hasher`, tagged by kind so unlike kinds never collide.
666
521
  /// A null writes only [`TAG_NULL`]; every other kind writes its tag then a
667
- /// canonical form of the value (see the module docs' value semantics).
522
+ /// canonical form of the value (see `docs/design/row-diff.md`'s "Value
523
+ /// semantics" section).
668
524
  fn hash_cell(
669
525
  hasher: &mut CellHasher,
670
526
  array: &ArrayRef,
@@ -2629,11 +2485,12 @@ struct ColumnInputs<'a> {
2629
2485
  }
2630
2486
 
2631
2487
  /// Compares one column across the paired changed rows and appends a record for
2632
- /// every differing cell. The change kind and rendering follow the module docs'
2633
- /// per-cell rules: a value domain or timestamp-awareness mismatch is a type
2634
- /// change, a differing value over the hash's lossless normalization is a value
2635
- /// change, and each side renders in the common comparison form so a
2636
- /// `value_changed` record never carries equal renderings.
2488
+ /// every differing cell. The change kind and rendering follow
2489
+ /// `docs/design/row-diff.md`'s "Per-cell changes" section: a value domain or
2490
+ /// timestamp-awareness mismatch is a type change, a differing value over the
2491
+ /// hash's lossless normalization is a value change, and each side renders in
2492
+ /// the common comparison form so a `value_changed` record never carries equal
2493
+ /// renderings.
2637
2494
  fn emit_column_records(
2638
2495
  column: &ColumnInputs<'_>,
2639
2496
  records: &mut Vec<CellRecord>,
@@ -4203,8 +4060,8 @@ impl CellColumns {
4203
4060
  }
4204
4061
  }
4205
4062
 
4206
- /// Diffs the rows of two tables matched by `key`. See the module docs for the
4207
- /// algorithm and value semantics.
4063
+ /// Diffs the rows of two tables matched by `key`. See `docs/design/row-diff.md`
4064
+ /// for the algorithm and value semantics.
4208
4065
  pub(crate) fn diff_rows(
4209
4066
  left: &impl TableInput,
4210
4067
  right: &impl TableInput,
@@ -103,10 +103,10 @@ pub fn diff(a: &Value, b: &Value) -> Result<Report, Error> {
103
103
  /// assert!(report.is_empty());
104
104
  /// ```
105
105
  pub fn diff_with_options(a: &Value, b: &Value, opts: &DiffOptions) -> Result<Report, Error> {
106
- // The distance memo is created here, per diff invocation, and dropped
107
- // when this returns — no cross-call state. It only ever caches
108
- // `ignore_order` container-pair distances (see `crate::ignore_order`'s
109
- // `memo` module); for an ordered diff it is threaded but never consulted.
106
+ // The memo is created here, per diff invocation, and dropped when this
107
+ // returns — no cross-call state. An ordered diff consults it only through
108
+ // set comparison (set-member and tuple digests); `ignore_order` also
109
+ // caches container-pair distances in it.
110
110
  diff_with_options_memo(a, b, opts, &crate::ignore_order::IgnoreOrderMemo::new())
111
111
  }
112
112
 
@@ -11,8 +11,6 @@ use crate::diff::DiffOptions;
11
11
 
12
12
  use super::IgnoreOrderMemo;
13
13
 
14
- use super::fxhash::HashSet;
15
-
16
14
  /// A total-ordering wrapper for the non-negative, always-finite distances
17
15
  /// [`rough_distance`] computes, so they can key a [`BTreeMap`](std::collections::BTreeMap) (ascending
18
16
  /// iteration, for [`compute_pairs`](super::pairing::compute_pairs)'s greedy loop) and group candidates by
@@ -872,7 +870,9 @@ pub(crate) fn match_dict_keys<'a>(a: &'a Object, b: &'a Object) -> DictKeyMatch<
872
870
  /// The shared `threshold_to_diff_deeper` ratio check backing both
873
871
  /// [`count_object_diff_leaves`] (the count-only distance mirror) and
874
872
  /// `crate::diff::object_diff`'s own unconditional collapse — see
875
- /// [`THRESHOLD_TO_DIFF_DEEPER`]'s own doc for why both exist.
873
+ /// [`THRESHOLD_TO_DIFF_DEEPER`]'s own doc for why both exist. All-`str` keys
874
+ /// are counted by a merge of the two ascending key sequences: one key
875
+ /// comparison per step, no hashing.
876
876
  pub(crate) fn is_below_threshold_to_diff_deeper(a: &Object, b: &Object) -> bool {
877
877
  let (union_len, intersect_len) = if a.has_non_str_keys() || b.has_non_str_keys() {
878
878
  let matched = match_dict_keys(a, b);
@@ -881,28 +881,24 @@ pub(crate) fn is_below_threshold_to_diff_deeper(a: &Object, b: &Object) -> bool
881
881
  matched.shared.len(),
882
882
  )
883
883
  } else {
884
- // Every key here is an `ObjectKey::Str`, so structural and
885
- // python-equality matching coincide. Counted by WTF-8 bytes, not
886
- // `ObjectKey::as_str` (which is `None` for a lone-surrogate key —
887
- // `ObjectKey` has no `Hash` impl at all, see its own doc, so this
888
- // is also the only way to put one in a `HashSet` here), so a
889
- // surrogate key is counted correctly instead of silently dropped.
890
- fn key_bytes(key: &ObjectKey) -> &[u8] {
891
- match key {
892
- ObjectKey::Str(s) => s.as_bytes(),
893
- ObjectKey::Other(_) => {
894
- unreachable!("has_non_str_keys() is false on both sides in this branch")
884
+ let (mut a_keys, mut b_keys) = (a.keys().peekable(), b.keys().peekable());
885
+ let mut intersect_len = 0;
886
+ while let (Some(a_key), Some(b_key)) = (a_keys.peek(), b_keys.peek()) {
887
+ match a_key.cmp(b_key) {
888
+ std::cmp::Ordering::Less => {
889
+ a_keys.next();
890
+ }
891
+ std::cmp::Ordering::Greater => {
892
+ b_keys.next();
893
+ }
894
+ std::cmp::Ordering::Equal => {
895
+ intersect_len += 1;
896
+ a_keys.next();
897
+ b_keys.next();
895
898
  }
896
899
  }
897
900
  }
898
- let union_len = a
899
- .keys()
900
- .map(key_bytes)
901
- .chain(b.keys().map(key_bytes))
902
- .collect::<HashSet<_>>()
903
- .len();
904
- let intersect_len = a.keys().filter(|key| b.contains_key(key)).count();
905
- (union_len, intersect_len)
901
+ (a.len() + b.len() - intersect_len, intersect_len)
906
902
  };
907
903
  #[allow(
908
904
  clippy::cast_precision_loss,
@@ -0,0 +1,115 @@
1
+ //! `FxHash`: a small, fast, non-cryptographic hasher for this module's `HashMap`/`HashSet`s.
2
+ //! See [`FxHasher`]'s doc for the accepted `DoS` trade-off and which tables carry it.
3
+
4
+ use std::hash::BuildHasherDefault;
5
+
6
+ /// This module's [`HashMap`](std::collections::HashMap), keyed with [`FxHasher`].
7
+ pub(crate) type HashMap<K, V> = std::collections::HashMap<K, V, BuildHasherDefault<FxHasher>>;
8
+ /// The [`HashMap`] equivalent for [`std::collections::HashSet`].
9
+ pub(crate) type HashSet<T> = std::collections::HashSet<T, BuildHasherDefault<FxHasher>>;
10
+
11
+ /// The `FxHash` algorithm (the one `rustc` itself uses for compiler-hot-path hash maps),
12
+ /// implemented from scratch.
13
+ ///
14
+ /// # `DoS` trade-off (this hasher is *not* collision-resistant)
15
+ ///
16
+ /// `FxHash` uses a fixed, public seed ([`FX_SEED`]) and an invertible step: a crafted
17
+ /// collision degrades an `FxHash` map or set from `O(1)` to `O(n)` per operation, pushing
18
+ /// [`HashedList::build`](super::hash::HashedList::build) from `O(n)` to `O(n²)` on an
19
+ /// all-colliding list, on top of the module's `O(N²)` pairing.
20
+ /// [`HashedList`](super::hash::HashedList), `AddedCandidates`, the pairing/`used` sets and
21
+ /// its result maps (`most_in_common_pairs`, `pairs`), `consumed_removed`, and the distance
22
+ /// memo are `FxHash`-keyed and reached only under `ignore_order=true`: an accepted,
23
+ /// documented `DoS` trade-off — `SipHash` there cost a measured per-call penalty on the
24
+ /// pairing hot path (PR #4). No default-path table is `FxHash`-keyed:
25
+ /// [`IgnoreOrderMemo`](super::memo::IgnoreOrderMemo)'s `tuple_ids`, which set comparison
26
+ /// reaches through `set_member_digest` -> `tuple_keyed` -> `tuple_digest` for a tuple dict
27
+ /// key inside a set member, is a `BTreeMap` (below), and
28
+ /// [`is_below_threshold_to_diff_deeper`](super::distance::is_below_threshold_to_diff_deeper),
29
+ /// called by `object_diff` for every unequal dict pair, counts dict-key overlap by a sorted
30
+ /// merge, with no table. Bound untrusted input against the module's `O(N²)` pairing
31
+ /// regardless of hasher.
32
+ ///
33
+ /// Every `FxHash`-keyed type carrying a float ([`ItemKey`](super::hash::ItemKey),
34
+ /// [`ScalarKey`](crate::lcs::ScalarKey), and the distance memo's
35
+ /// [`DistKey`](super::hash::DistKey) via `number_key`) mixes its bits first
36
+ /// ([`mix_float_bits`](crate::lcs::mix_float_bits)), so integral and half-integer floats —
37
+ /// whose raw bit patterns share many trailing zeros — do not collide by accident. Every `NaN`
38
+ /// additionally folds onto one fixed key, matching `DeepHash`'s `str()`-based digest: this
39
+ /// changes which values these tables treat as the same item, not their per-lookup cost.
40
+ ///
41
+ /// `node_table`, `member_content` and `tuple_ids` in `IgnoreOrderMemo` are `BTreeMap`s instead,
42
+ /// since they are keyed by attacker-controlled member content and reached on the default path
43
+ /// too, with no `Hash` derive: `O(log n)` worst case, always, though each comparison still
44
+ /// walks the whole probed key — `member_content`'s `MemberContent::UnhashableDict` key is
45
+ /// itself keyed by each dict key's own `ItemKey` tree, not a flat string.
46
+ ///
47
+ /// An integer beyond `i128` (`ItemKey::BigInt`) hashes and compares by its magnitude digits,
48
+ /// `O(digits)` not `O(1)` per lookup — inside a hashable tuple it reaches the same cost via
49
+ /// `ScalarKey::Big` and `tuple_ids`, through `PyHashPart::Scalar` — a cost the `ignore_order`
50
+ /// element-count cap does not bound: one huge integer costs its own digit length regardless
51
+ /// of element count. A custom object's `ItemKey::Object` costs `ItemKey::Dict`'s full walk
52
+ /// plus one class-name comparison, and two distinct classes sharing a `__name__` share its
53
+ /// bucket; `ItemKey::Opaque` costs one identity string comparison.
54
+ #[derive(Default)]
55
+ pub(crate) struct FxHasher {
56
+ pub(crate) hash: u64,
57
+ }
58
+
59
+ /// `FxHash`'s seed constant (golden-ratio-derived odd constant).
60
+ pub(crate) const FX_SEED: u64 = 0x51_7c_c1_b7_27_22_0a_95;
61
+
62
+ impl FxHasher {
63
+ /// Folds one word into the hash: rotate, xor, multiply — `FxHash`'s mixing step.
64
+ pub(crate) fn add_to_hash(&mut self, word: u64) {
65
+ self.hash = (self.hash.rotate_left(5) ^ word).wrapping_mul(FX_SEED);
66
+ }
67
+ }
68
+
69
+ impl std::hash::Hasher for FxHasher {
70
+ fn write(&mut self, mut bytes: &[u8]) {
71
+ while let Some(chunk) = bytes.get(..8) {
72
+ let word = u64::from_ne_bytes(chunk.try_into().expect("chunk is exactly 8 bytes"));
73
+ self.add_to_hash(word);
74
+ bytes = &bytes[8..];
75
+ }
76
+ for &byte in bytes {
77
+ self.add_to_hash(u64::from(byte));
78
+ }
79
+ }
80
+
81
+ fn write_u8(&mut self, i: u8) {
82
+ self.add_to_hash(u64::from(i));
83
+ }
84
+
85
+ fn write_u16(&mut self, i: u16) {
86
+ self.add_to_hash(u64::from(i));
87
+ }
88
+
89
+ fn write_u32(&mut self, i: u32) {
90
+ self.add_to_hash(u64::from(i));
91
+ }
92
+
93
+ fn write_u64(&mut self, i: u64) {
94
+ self.add_to_hash(i);
95
+ }
96
+
97
+ fn write_u128(&mut self, i: u128) {
98
+ #[allow(
99
+ clippy::cast_possible_truncation,
100
+ reason = "hash mixing only, truncation does not affect correctness, only distribution"
101
+ )]
102
+ {
103
+ self.add_to_hash(i as u64);
104
+ self.add_to_hash((i >> 64) as u64);
105
+ }
106
+ }
107
+
108
+ fn write_usize(&mut self, i: usize) {
109
+ self.add_to_hash(i as u64);
110
+ }
111
+
112
+ fn finish(&self) -> u64 {
113
+ self.hash
114
+ }
115
+ }
@@ -329,7 +329,7 @@ impl std::hash::Hash for ItemKey {
329
329
  /// Referring to a nested tuple by id rather than by its whole identity is
330
330
  /// what keeps this `O(arity)` per node: `((((1,),),),)` is four one-element
331
331
  /// identities, not four identities of size 1, 2, 3 and 4.
332
- #[derive(Debug, Clone, PartialEq, Eq, Hash)]
332
+ #[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
333
333
  pub(crate) enum PyHashPart {
334
334
  Scalar(ScalarKey),
335
335
  Tuple(TupleId),
@@ -345,12 +345,12 @@ pub(crate) enum PyHashPart {
345
345
  /// ([`python_scalar_key`], which makes `1`, `1.0` and `True` one key), and a
346
346
  /// list or dict is unhashable, which also makes any tuple containing one
347
347
  /// unhashable.
348
- #[derive(Debug, Clone, PartialEq, Eq, Hash)]
348
+ #[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord)]
349
349
  pub(crate) struct PyHashKey(Box<[PyHashPart]>);
350
350
 
351
351
  /// A hashable tuple identity's place in the run's interning table — see
352
352
  /// [`super::IgnoreOrderMemo::tuple_digest`].
353
- #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
353
+ #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
354
354
  pub(crate) struct TupleId(usize);
355
355
 
356
356
  impl TupleId {
@@ -24,14 +24,17 @@ type DistanceKey = (DistKey, DistKey);
24
24
  /// "Distance memo" section: pairwise container distances, tuple
25
25
  /// digests, and set-member digests, scoped to one diff run. `DistKey`
26
26
  /// keys hash and compare by a full-tree walk; `MemberContent` keys
27
- /// compare by one (it has no `Hash`). A big-integer leaf costs
27
+ /// compare by one (it has no `Hash`); a `PyHashKey` lookup makes `O(log n)`
28
+ /// comparisons, each `O(the probed key's element bytes)`, a nested tuple
29
+ /// comparing as one id. A big-integer leaf costs
28
30
  /// `O(digits)` per lookup; `super::fxhash`'s doc enumerates each key's cost.
29
31
  pub(crate) struct IgnoreOrderMemo<'r> {
30
32
  cache: RefCell<HashMap<DistanceKey, f64>>,
31
33
  /// Interns each hashable-tuple identity to its digest, shared
32
34
  /// across the run — see `docs/design/ignore-order.md`'s "Distance
33
- /// memo" section.
34
- tuple_ids: RefCell<HashMap<PyHashKey, TupleId>>,
35
+ /// memo" section. A `BTreeMap`, reached on the default path through a
36
+ /// set member's tuple dict keys; its key type carries no `Hash` derive.
37
+ tuple_ids: RefCell<BTreeMap<PyHashKey, TupleId>>,
35
38
  /// The digest assigned to each interned identity, indexed by
36
39
  /// [`TupleId::index`].
37
40
  tuple_digests: RefCell<Vec<ItemKey>>,
@@ -60,7 +63,7 @@ impl<'r> IgnoreOrderMemo<'r> {
60
63
  pub(crate) fn new() -> Self {
61
64
  Self {
62
65
  cache: RefCell::new(HashMap::default()),
63
- tuple_ids: RefCell::new(HashMap::default()),
66
+ tuple_ids: RefCell::new(BTreeMap::new()),
64
67
  tuple_digests: RefCell::new(Vec::new()),
65
68
  node_table: RefCell::new(BTreeMap::new()),
66
69
  member_content: RefCell::new(BTreeMap::new()),
@@ -78,7 +81,7 @@ impl<'r> IgnoreOrderMemo<'r> {
78
81
  pub(crate) fn disabled() -> Self {
79
82
  Self {
80
83
  cache: RefCell::new(HashMap::default()),
81
- tuple_ids: RefCell::new(HashMap::default()),
84
+ tuple_ids: RefCell::new(BTreeMap::new()),
82
85
  tuple_digests: RefCell::new(Vec::new()),
83
86
  node_table: RefCell::new(BTreeMap::new()),
84
87
  member_content: RefCell::new(BTreeMap::new()),