deepdiff-rs 0.11.1__tar.gz → 0.11.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/Cargo.lock +4 -4
  2. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/Cargo.toml +1 -1
  3. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/PKG-INFO +10 -13
  4. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/README.md +9 -12
  5. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/Cargo.toml +16 -0
  6. deepdiff_rs-0.11.2/crates/onix-arrow/examples/row_diff_profile.rs +316 -0
  7. deepdiff_rs-0.11.2/crates/onix-arrow/examples/row_diff_rss.rs +141 -0
  8. deepdiff_rs-0.11.2/crates/onix-arrow/examples/shared/gen_shapes.rs +245 -0
  9. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/lib.rs +2 -0
  10. deepdiff_rs-0.11.2/crates/onix-arrow/src/profile.rs +418 -0
  11. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/row_diff.rs +237 -78
  12. deepdiff_rs-0.11.2/crates/onix-arrow/tests/profile_passes.rs +116 -0
  13. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/benchmarks/bench_bindings.py +26 -210
  14. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/errors.rs +1 -10
  15. deepdiff_rs-0.11.2/crates/onix-py/src/fast_path.rs +42 -0
  16. deepdiff_rs-0.11.2/crates/onix-py/src/guard.rs +296 -0
  17. deepdiff_rs-0.11.2/crates/onix-py/src/lib.rs +26 -0
  18. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/conftest.py +3 -12
  19. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_bindings_memory.py +7 -19
  20. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_differential_fuzz.py +78 -687
  21. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_non_finite.py +1 -7
  22. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_signed_zero.py +10 -43
  23. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_stub_signatures.py +7 -24
  24. deepdiff_rs-0.11.2/crates/onix-py/tests/test_suite_hygiene.py +32 -0
  25. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_table_row_diff.py +5 -9
  26. deepdiff_rs-0.11.1/crates/onix-arrow/examples/row_diff_rss.rs +0 -359
  27. deepdiff_rs-0.11.1/crates/onix-py/src/fast_path.rs +0 -64
  28. deepdiff_rs-0.11.1/crates/onix-py/src/guard.rs +0 -452
  29. deepdiff_rs-0.11.1/crates/onix-py/src/lib.rs +0 -52
  30. deepdiff_rs-0.11.1/crates/onix-py/tests/test_suite_hygiene.py +0 -39
  31. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
  32. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/error.rs +0 -0
  33. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/json_rows.rs +0 -0
  34. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/options.rs +0 -0
  35. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/schema.rs +0 -0
  36. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/spool.rs +0 -0
  37. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/table_diff.rs +0 -0
  38. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/Cargo.toml +0 -0
  39. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
  40. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/datetime.rs +0 -0
  41. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/datetime_tests.rs +0 -0
  42. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/array.rs +0 -0
  43. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/dispatch.rs +0 -0
  44. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/mod.rs +0 -0
  45. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/object.rs +0 -0
  46. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/options.rs +0 -0
  47. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/scalar.rs +0 -0
  48. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/set.rs +0 -0
  49. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/tests.rs +0 -0
  50. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/error.rs +0 -0
  51. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/distance.rs +0 -0
  52. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/fxhash.rs +0 -0
  53. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/hash.rs +0 -0
  54. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/memo.rs +0 -0
  55. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/mod.rs +0 -0
  56. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/pairing.rs +0 -0
  57. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/tests.rs +0 -0
  58. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lcs.rs +0 -0
  59. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lcs_tests.rs +0 -0
  60. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lib.rs +0 -0
  61. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/path.rs +0 -0
  62. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/report.rs +0 -0
  63. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/report_tests.rs +0 -0
  64. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/test_support.rs +0 -0
  65. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/unified_diff.rs +0 -0
  66. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/unified_diff_tests.rs +0 -0
  67. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/value.rs +0 -0
  68. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/value_tests.rs +0 -0
  69. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/golden.rs +0 -0
  70. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
  71. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/memory_footprint.rs +0 -0
  72. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/proptest_diff.rs +0 -0
  73. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
  74. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/.python-version +0 -0
  75. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/Cargo.toml +0 -0
  76. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/deepdiff_rs.pyi +0 -0
  77. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/arrow.rs +0 -0
  78. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/convert.rs +0 -0
  79. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/deepdiff.rs +0 -0
  80. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_conversions.py +0 -0
  81. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_datetimes.py +0 -0
  82. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_depth_guard.py +0 -0
  83. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_golden_parity.py +0 -0
  84. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_sets.py +0 -0
  85. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_smoke.py +0 -0
  86. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_stub_mypy.py +0 -0
  87. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_table_diff.py +0 -0
  88. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_timedeltas.py +0 -0
  89. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_times.py +0 -0
  90. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_tuples.py +0 -0
  91. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_wheel_contents.py +0 -0
  92. {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/pyproject.toml +0 -0
@@ -629,7 +629,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
629
629
 
630
630
  [[package]]
631
631
  name = "onix-arrow"
632
- version = "0.11.1"
632
+ version = "0.11.2"
633
633
  dependencies = [
634
634
  "arrow-array",
635
635
  "arrow-buffer",
@@ -648,7 +648,7 @@ dependencies = [
648
648
 
649
649
  [[package]]
650
650
  name = "onix-cli"
651
- version = "0.11.1"
651
+ version = "0.11.2"
652
652
  dependencies = [
653
653
  "onix-core",
654
654
  "serde_json",
@@ -656,7 +656,7 @@ dependencies = [
656
656
 
657
657
  [[package]]
658
658
  name = "onix-core"
659
- version = "0.11.1"
659
+ version = "0.11.2"
660
660
  dependencies = [
661
661
  "num-bigint",
662
662
  "num-traits",
@@ -669,7 +669,7 @@ dependencies = [
669
669
 
670
670
  [[package]]
671
671
  name = "onix-py"
672
- version = "0.11.1"
672
+ version = "0.11.2"
673
673
  dependencies = [
674
674
  "arrow-array",
675
675
  "arrow-schema",
@@ -3,7 +3,7 @@ resolver = "3"
3
3
  members = ["crates/onix-core", "crates/onix-py", "crates/onix-arrow"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.11.1"
6
+ version = "0.11.2"
7
7
  edition = "2024"
8
8
  license = "MIT"
9
9
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: deepdiff-rs
3
- Version: 0.11.1
3
+ Version: 0.11.2
4
4
  Classifier: Programming Language :: Python :: 3
5
5
  Classifier: Programming Language :: Rust
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -124,15 +124,10 @@ It reports the **schema** diff (which columns were added, removed, or changed ty
124
124
  import pyarrow as pa
125
125
  from deepdiff_rs import diff_tables
126
126
 
127
- left = pa.table({
128
- "id": pa.array([1, 2, 3, 9, 9], pa.int64()), # 9 is a duplicate key
129
- "amount": pa.array([10, 20, 30, 90, 91], pa.int32()),
130
- })
131
- right = pa.table({
132
- "id": pa.array([2, 3, 4], pa.int64()),
133
- "amount": pa.array([20, 31, 40], pa.int64()),
134
- "note": pa.array(["a", "b", "c"], pa.string()),
135
- })
127
+ # 9 is a duplicate key
128
+ left = pa.table({"id": pa.array([1, 2, 3, 9, 9], pa.int64()), "amount": pa.array([10, 20, 30, 90, 91], pa.int32())})
129
+ right = pa.table({"id": pa.array([2, 3, 4], pa.int64()), "amount": pa.array([20, 31, 40], pa.int64()),
130
+ "note": pa.array(["a", "b", "c"], pa.string())})
136
131
 
137
132
  diff = diff_tables(left, right, key=["id"])
138
133
  print(diff.summary(), "added ids:", pa.table(diff.rows_added()).column("id").to_pylist(), "removed ids:", pa.table(diff.rows_removed()).column("id").to_pylist())
@@ -144,11 +139,13 @@ print("cells changed:", pa.table(diff.cells_changed()).to_pylist(), "duplicate k
144
139
  cells changed: [{'id': 3, 'column': 'amount', 'old_value': '30', 'new_value': '31', 'change': 'value_changed'}] duplicate keys: [{'id': 9, 'left_count': 2, 'right_count': 0}]
145
140
  ```
146
141
 
147
- A key appearing more than once on either side is reported in `duplicate_keys` (with `left_count` and `right_count`) and excluded from the added/removed/changed sets; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is out of scope and is skipped rather than compared. The exact value-comparison rules are documented on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
142
+ A key appearing more than once on either side is reported in `duplicate_keys` (`left_count`/`right_count`) and excluded from added/removed/changed; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is skipped rather than compared. The exact rules are on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
148
143
 
149
- Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings that carry the same logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes. The full normalization rules are documented on `normalized_type` (and `map_entries`) in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs); nullability is ignored but reported in each record. Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` consumes it with no pyarrow needed, and `diff.schema_arrow.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but pandas' own implementation of that protocol needs pyarrow installed regardless of which path you take.
144
+ Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings sharing a logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes; nullability is ignored but reported in each record. The full rules are on `normalized_type`/`map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs). Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` needs no pyarrow, and `.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but needs pyarrow installed regardless.
150
145
 
151
- `pyarrow` is optional: install it with `pip install deepdiff-rs[arrow]`. It is needed only for `to_pyarrow()` and for passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. Passing an object that implements neither Arrow protocol raises `TypeError`; calling `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra; `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed at all; see [Known limitations](#known-limitations) for its row cap.
146
+ `pyarrow` is optional: `pip install deepdiff-rs[arrow]`. It's needed only for `to_pyarrow()` and passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. An object implementing neither Arrow protocol raises `TypeError`; `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra. `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed; see [Known limitations](#known-limitations) for its row cap.
147
+
148
+ Beyond a join-based diff, `diff_tables` reports duplicate and null keys rather than silently multiplying or dropping them, distinguishes `type_changed` from `value_changed` per cell, renders every value by one documented rule set (Python's, with `Duration` the one exception -- an ISO 8601 string, not Python's own `str()`), and produces byte-identical output at any thread count with memory proportional to row count from a streamed input. What that costs over a hand-rolled join is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)'s polars-backed ceiling section.
152
149
 
153
150
  ## Performance
154
151
 
@@ -106,15 +106,10 @@ It reports the **schema** diff (which columns were added, removed, or changed ty
106
106
  import pyarrow as pa
107
107
  from deepdiff_rs import diff_tables
108
108
 
109
- left = pa.table({
110
- "id": pa.array([1, 2, 3, 9, 9], pa.int64()), # 9 is a duplicate key
111
- "amount": pa.array([10, 20, 30, 90, 91], pa.int32()),
112
- })
113
- right = pa.table({
114
- "id": pa.array([2, 3, 4], pa.int64()),
115
- "amount": pa.array([20, 31, 40], pa.int64()),
116
- "note": pa.array(["a", "b", "c"], pa.string()),
117
- })
109
+ # 9 is a duplicate key
110
+ left = pa.table({"id": pa.array([1, 2, 3, 9, 9], pa.int64()), "amount": pa.array([10, 20, 30, 90, 91], pa.int32())})
111
+ right = pa.table({"id": pa.array([2, 3, 4], pa.int64()), "amount": pa.array([20, 31, 40], pa.int64()),
112
+ "note": pa.array(["a", "b", "c"], pa.string())})
118
113
 
119
114
  diff = diff_tables(left, right, key=["id"])
120
115
  print(diff.summary(), "added ids:", pa.table(diff.rows_added()).column("id").to_pylist(), "removed ids:", pa.table(diff.rows_removed()).column("id").to_pylist())
@@ -126,11 +121,13 @@ print("cells changed:", pa.table(diff.cells_changed()).to_pylist(), "duplicate k
126
121
  cells changed: [{'id': 3, 'column': 'amount', 'old_value': '30', 'new_value': '31', 'change': 'value_changed'}] duplicate keys: [{'id': 9, 'left_count': 2, 'right_count': 0}]
127
122
  ```
128
123
 
129
- A key appearing more than once on either side is reported in `duplicate_keys` (with `left_count` and `right_count`) and excluded from the added/removed/changed sets; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is out of scope and is skipped rather than compared. The exact value-comparison rules are documented on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
124
+ A key appearing more than once on either side is reported in `duplicate_keys` (`left_count`/`right_count`) and excluded from added/removed/changed; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is skipped rather than compared. The exact rules are on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
130
125
 
131
- Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings that carry the same logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes. The full normalization rules are documented on `normalized_type` (and `map_entries`) in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs); nullability is ignored but reported in each record. Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` consumes it with no pyarrow needed, and `diff.schema_arrow.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but pandas' own implementation of that protocol needs pyarrow installed regardless of which path you take.
126
+ Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings sharing a logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes; nullability is ignored but reported in each record. The full rules are on `normalized_type`/`map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs). Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` needs no pyarrow, and `.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but needs pyarrow installed regardless.
132
127
 
133
- `pyarrow` is optional: install it with `pip install deepdiff-rs[arrow]`. It is needed only for `to_pyarrow()` and for passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. Passing an object that implements neither Arrow protocol raises `TypeError`; calling `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra; `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed at all; see [Known limitations](#known-limitations) for its row cap.
128
+ `pyarrow` is optional: `pip install deepdiff-rs[arrow]`. It's needed only for `to_pyarrow()` and passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. An object implementing neither Arrow protocol raises `TypeError`; `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra. `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed; see [Known limitations](#known-limitations) for its row cap.
129
+
130
+ Beyond a join-based diff, `diff_tables` reports duplicate and null keys rather than silently multiplying or dropping them, distinguishes `type_changed` from `value_changed` per cell, renders every value by one documented rule set (Python's, with `Duration` the one exception -- an ISO 8601 string, not Python's own `str()`), and produces byte-identical output at any thread count with memory proportional to row count from a streamed input. What that costs over a hand-rolled join is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)'s polars-backed ceiling section.
134
131
 
135
132
  ## Performance
136
133
 
@@ -9,6 +9,12 @@ description = "Arrow table diffing (schema and keyed rows) built on the onix dif
9
9
  [lints]
10
10
  workspace = true
11
11
 
12
+ [features]
13
+ # Compiles in the per-pass wall-time and peak-RSS instrumentation the
14
+ # `row_diff_profile` example reads. Off by default and never enabled by the
15
+ # release wheel, so the instrumentation is absent from the shipped `.so`.
16
+ profile = []
17
+
12
18
  [dependencies]
13
19
  # Pinned to an exact Arrow version so this crate and the `pyo3-arrow` bridge
14
20
  # in `onix-py` (which requires `arrow` ^59) resolve to the identical Arrow
@@ -52,3 +58,13 @@ serde_json = "1"
52
58
  proptest = "1"
53
59
  # `half::f16` builds a Float16 test column; the same version arrow-array uses.
54
60
  half = "2"
61
+
62
+ # The profiling harness reads the per-pass instrumentation, so it only builds
63
+ # with the `profile` feature (and never as part of the default build or wheel).
64
+ [[example]]
65
+ name = "row_diff_profile"
66
+ required-features = ["profile"]
67
+
68
+ [[test]]
69
+ name = "profile_passes"
70
+ required-features = ["profile"]
@@ -0,0 +1,316 @@
1
+ //! Per-pass wall-time and peak-RSS profile of one keyed row diff, the committed
2
+ //! harness every row-diff performance change posts a before/after table from.
3
+ //! Builds only with the `profile` feature, which the release wheel never enables.
4
+ //!
5
+ //! Each invocation runs a discarded warm-up diff, a timed uninstrumented diff
6
+ //! (the `uninstrumented wall` line), and an instrumented diff whose passes make
7
+ //! up the table. The closure line compares the passes' sum with the instrumented
8
+ //! wall minus the profiler's own boundary `ps` reads; `share` is each row's wall
9
+ //! over that net wall. Peak RSS is the process's, third diff in the process.
10
+ //!
11
+ //! - **file**: reads each side from an uncompressed Arrow IPC file, re-opened
12
+ //! and re-decoded by every pass. Convert a parquet fixture once with
13
+ //! `python -c "import pyarrow.parquet as p, pyarrow.feather as f;
14
+ //! f.write_feather(p.read_table('a.parquet'), 'a.arrow', compression='uncompressed')"`.
15
+ //! - **generated**: spools both sides of a deterministic proxy shape to anonymous
16
+ //! Arrow IPC files (the `spool write` line times the IPC writer alone).
17
+ //!
18
+ //! ```sh
19
+ //! cargo build -p onix-arrow --release --features profile --example row_diff_profile
20
+ //! target/release/examples/row_diff_profile file a.arrow b.arrow --key id --threads 18
21
+ //! target/release/examples/row_diff_profile 1000000 linear 18
22
+ //! target/release/examples/row_diff_profile 1000000 manycols 18 34 64
23
+ //! ```
24
+ //!
25
+ //! Generated args: `[rows [shape [threads [shape params...]]]]`, key `id`,
26
+ //! `threads` defaulting to `ROW_DIFF_THREADS` or available parallelism. Shapes:
27
+ //!
28
+ //! - `linear`: `id`/`value` int64; 1% of keys added, 1% removed, ~2% changed.
29
+ //! - `allchange`: `id`/`value` int64, every row changed.
30
+ //! - `wide [width=512]`: `id` plus one `width`-byte string, every row changed.
31
+ //! - `manycols [ncols=34 [width=64]]`: `id` plus `ncols` `width`-byte strings,
32
+ //! only the first differing, so the spill carries every value column.
33
+
34
+ use std::fs::File;
35
+ use std::io::BufReader;
36
+ use std::num::NonZeroUsize;
37
+ use std::path::PathBuf;
38
+ use std::time::{Duration, Instant};
39
+
40
+ use arrow_array::RecordBatchReader;
41
+ use arrow_ipc::reader::FileReader;
42
+ use arrow_schema::SchemaRef;
43
+ use onix_arrow::{TableDiffError, TableDiffOptions, TableInput, diff_tables, profile, spool};
44
+
45
+ #[path = "shared/gen_shapes.rs"]
46
+ mod gen_shapes;
47
+ use gen_shapes::{Case, Generated, batch_rows};
48
+
49
+ /// A table read from an Arrow IPC file, re-opened on every `open`.
50
+ struct FileInput {
51
+ path: PathBuf,
52
+ schema: SchemaRef,
53
+ }
54
+
55
+ impl FileInput {
56
+ fn load(path: &str) -> FileInput {
57
+ let reader = open_ipc(path).unwrap_or_else(|e| panic!("open {path}: {e}"));
58
+ FileInput {
59
+ path: PathBuf::from(path),
60
+ schema: reader.schema(),
61
+ }
62
+ }
63
+ }
64
+
65
+ fn open_ipc(path: &str) -> Result<FileReader<BufReader<File>>, TableDiffError> {
66
+ let file = File::open(path).map_err(|e| TableDiffError::Read {
67
+ message: format!("open {path}: {e}"),
68
+ })?;
69
+ FileReader::try_new(BufReader::new(file), None).map_err(|e| TableDiffError::Read {
70
+ message: format!("read Arrow IPC {path}: {e}"),
71
+ })
72
+ }
73
+
74
+ impl TableInput for FileInput {
75
+ fn schema(&self) -> SchemaRef {
76
+ self.schema.clone()
77
+ }
78
+ fn open(&self) -> Result<Box<dyn RecordBatchReader + Send>, TableDiffError> {
79
+ let path = self.path.to_string_lossy();
80
+ Ok(Box::new(open_ipc(&path)?))
81
+ }
82
+ }
83
+
84
+ /// A side spooled to an anonymous Arrow IPC file, re-read on every `open` — the
85
+ /// same re-openable spool the Python bindings hand the row diff.
86
+ struct Spooled {
87
+ file: File,
88
+ schema: SchemaRef,
89
+ }
90
+
91
+ impl TableInput for Spooled {
92
+ fn schema(&self) -> SchemaRef {
93
+ self.schema.clone()
94
+ }
95
+ fn open(&self) -> Result<Box<dyn RecordBatchReader + Send>, TableDiffError> {
96
+ Ok(Box::new(spool::reopen(&self.file)?))
97
+ }
98
+ }
99
+
100
+ /// Spools one generated side, returning it and the time spent in the IPC
101
+ /// writer (generation excluded).
102
+ fn spool_side(side: &Generated) -> (Spooled, Duration) {
103
+ let (file, mut writer) = spool::open(&side.schema).expect("open spool");
104
+ let mut writing = Duration::ZERO;
105
+ for batch in side.open().expect("open generator") {
106
+ let batch = batch.expect("generate batch");
107
+ let start = Instant::now();
108
+ writer.write(&batch).expect("spool write");
109
+ writing += start.elapsed();
110
+ }
111
+ let start = Instant::now();
112
+ writer.finish().expect("spool finish");
113
+ writing += start.elapsed();
114
+ let spooled = Spooled {
115
+ file,
116
+ schema: side.schema.clone(),
117
+ };
118
+ (spooled, writing)
119
+ }
120
+
121
+ /// Runs one discarded warm-up diff, one timed uninstrumented diff, and one
122
+ /// instrumented diff, then prints both walls, the closure of the instrumented
123
+ /// wall over its passes, and the per-pass table.
124
+ fn run(left: &impl TableInput, right: &impl TableInput, options: &TableDiffOptions, label: &str) {
125
+ let _ = diff_tables(left, right, options).expect("diff succeeds");
126
+ let start = Instant::now();
127
+ let diff = diff_tables(left, right, options).expect("diff succeeds");
128
+ let uninstrumented = start.elapsed().as_secs_f64();
129
+ let summary = diff.summary();
130
+
131
+ let session = profile::begin();
132
+ let start = Instant::now();
133
+ let _ = diff_tables(left, right, options).expect("diff succeeds");
134
+ let instrumented = start.elapsed().as_secs_f64();
135
+ let report = session.finish();
136
+
137
+ let boundary: f64 = report
138
+ .iter()
139
+ .filter(|row| row.label == profile::BOUNDARY_LABEL)
140
+ .map(|row| row.wall_secs)
141
+ .sum();
142
+ let passes: f64 = report
143
+ .iter()
144
+ .filter(|row| row.peak_rss_mib.is_some())
145
+ .map(|row| row.wall_secs)
146
+ .sum();
147
+ let closed = instrumented - boundary;
148
+
149
+ println!("{label}");
150
+ println!("threads: {}", options.threads());
151
+ println!("uninstrumented wall: {uninstrumented:.3} s");
152
+ println!("instrumented wall: {instrumented:.3} s");
153
+ println!("boundary ps reads: {boundary:.3} s");
154
+ println!(
155
+ "passes sum: {passes:.3} s of {closed:.3} s (instrumented minus boundary); residual {:.3} s ({:.1}%)",
156
+ closed - passes,
157
+ 100.0 * (closed - passes) / closed
158
+ );
159
+ println!(
160
+ "rows_added={} rows_removed={} rows_changed={} duplicate_keys={} cells_changed={}",
161
+ summary.rows_added,
162
+ summary.rows_removed,
163
+ summary.rows_changed,
164
+ summary.duplicate_keys,
165
+ summary.cells_changed
166
+ );
167
+ println!();
168
+ println!(
169
+ "{:<44} {:>10} {:>8} {:>14}",
170
+ "pass", "wall (s)", "share", "peak RSS (MB)"
171
+ );
172
+ for row in report
173
+ .iter()
174
+ .filter(|row| row.label != profile::BOUNDARY_LABEL)
175
+ {
176
+ let share = 100.0 * row.wall_secs / closed;
177
+ match row.peak_rss_mib {
178
+ Some(rss) => println!(
179
+ "{:<44} {:>10.3} {:>7.1}% {:>14.1}",
180
+ row.label, row.wall_secs, share, rss
181
+ ),
182
+ None => println!(
183
+ "{:<44} {:>10.3} {:>7.1}% {:>14}",
184
+ format!(" {}", row.label),
185
+ row.wall_secs,
186
+ share,
187
+ "-"
188
+ ),
189
+ }
190
+ }
191
+ }
192
+
193
+ const USAGE: &str =
194
+ "usage: row_diff_profile file <left.arrow> <right.arrow> [--key col[,col...]] [--threads N]
195
+ row_diff_profile [rows [linear|allchange|wide|manycols [threads [shape params...]]]]";
196
+
197
+ fn usage_error(message: &str) -> ! {
198
+ eprintln!("{message}\n{USAGE}");
199
+ std::process::exit(2);
200
+ }
201
+
202
+ /// Parses a positive integer argument, exiting with a usage error otherwise.
203
+ fn positive(what: &str, value: &str) -> NonZeroUsize {
204
+ value.parse().unwrap_or_else(|_| {
205
+ usage_error(&format!("{what} must be a positive integer, got {value:?}"))
206
+ })
207
+ }
208
+
209
+ fn options_for(keys: Vec<String>, threads: Option<NonZeroUsize>) -> TableDiffOptions {
210
+ let threads = threads.or_else(|| {
211
+ std::env::var("ROW_DIFF_THREADS")
212
+ .ok()
213
+ .map(|v| positive("ROW_DIFF_THREADS", &v))
214
+ });
215
+ let options = TableDiffOptions::new(keys);
216
+ match threads {
217
+ Some(t) => options
218
+ .with_threads(t)
219
+ .unwrap_or_else(|e| usage_error(&e.to_string())),
220
+ None => options,
221
+ }
222
+ }
223
+
224
+ fn file_mode(args: &[String]) {
225
+ let [left_path, right_path, flags @ ..] = args else {
226
+ usage_error("file mode needs <left.arrow> <right.arrow>");
227
+ };
228
+ let mut keys = Vec::new();
229
+ let mut threads = None;
230
+ let mut flags = flags.iter();
231
+ while let Some(flag) = flags.next() {
232
+ let Some(value) = flags.next() else {
233
+ usage_error(&format!("{flag} needs a value"));
234
+ };
235
+ match flag.as_str() {
236
+ "--key" => keys.extend(value.split(',').map(str::to_string)),
237
+ "--threads" => threads = Some(positive("--threads", value)),
238
+ _ => usage_error(&format!("unknown flag {flag:?}")),
239
+ }
240
+ }
241
+ if keys.is_empty() {
242
+ keys.push("id".to_string());
243
+ }
244
+ let left = FileInput::load(left_path);
245
+ let right = FileInput::load(right_path);
246
+ run(
247
+ &left,
248
+ &right,
249
+ &options_for(keys, threads),
250
+ &format!("file: {left_path} vs {right_path}"),
251
+ );
252
+ }
253
+
254
+ fn generated_mode(args: &[String]) {
255
+ let rows = args.first().map_or(1_000_000, |a| {
256
+ i64::try_from(positive("rows", a).get()).unwrap_or_else(|_| usage_error("rows too large"))
257
+ });
258
+ let shape = args.get(1).map_or("linear", String::as_str);
259
+ let threads = args.get(2).map(|a| positive("threads", a));
260
+ let params: Vec<usize> = args
261
+ .iter()
262
+ .skip(3)
263
+ .map(|a| positive("shape parameter", a).get())
264
+ .collect();
265
+ let param = |i: usize, default: usize| params.get(i).copied().unwrap_or(default);
266
+ let (case, arity) = match shape {
267
+ "linear" => (Case::Linear, 0),
268
+ "allchange" => (Case::AllChange, 0),
269
+ "wide" => (Case::Wide(param(0, 512)), 1),
270
+ "manycols" => (
271
+ Case::ManyCols {
272
+ ncols: param(0, 34),
273
+ width: param(1, 64),
274
+ },
275
+ 2,
276
+ ),
277
+ _ => usage_error(&format!("unknown shape {shape:?}")),
278
+ };
279
+ if params.len() > arity {
280
+ usage_error(&format!("shape {shape} takes at most {arity} parameter(s)"));
281
+ }
282
+
283
+ let (schema, left_shape, right_shape, key) = case.build(rows);
284
+ let options = options_for(vec![key.to_string()], threads);
285
+ let batch = batch_rows();
286
+ let (left, left_write) = spool_side(&Generated {
287
+ schema: schema.clone(),
288
+ rows,
289
+ shape: left_shape,
290
+ batch,
291
+ });
292
+ let (right, right_write) = spool_side(&Generated {
293
+ schema,
294
+ rows,
295
+ shape: right_shape,
296
+ batch,
297
+ });
298
+ println!(
299
+ "spool write (both sides, before the diff): {:.3} s",
300
+ (left_write + right_write).as_secs_f64()
301
+ );
302
+ run(
303
+ &left,
304
+ &right,
305
+ &options,
306
+ &format!("rows per side: {rows} (shape={shape})"),
307
+ );
308
+ }
309
+
310
+ fn main() {
311
+ let args: Vec<String> = std::env::args().skip(1).collect();
312
+ match args.split_first() {
313
+ Some((mode, rest)) if mode == "file" => file_mode(rest),
314
+ _ => generated_mode(&args),
315
+ }
316
+ }
@@ -0,0 +1,141 @@
1
+ //! Measures the keyed row diff's peak memory and wall time, to check the memory
2
+ //! bounds the README states.
3
+ //!
4
+ //! Run under the OS's max-RSS reporter:
5
+ //!
6
+ //! ```sh
7
+ //! cargo build -p onix-arrow --release --example row_diff_rss
8
+ //! # linear shape (default): mostly-matching rows, 1% added/removed, ~2% changed
9
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 1000000
10
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 10000000
11
+ //! # same shape with no changed rows: the cell pass materializes nothing, the
12
+ //! # pass-one baseline the ~2%-changed run is measured against
13
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 nochange
14
+ //! # every row changed (narrow int cells): the cell pass at full width
15
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 allchange
16
+ //! # every row changed with a wide (1 KB) string cell: the rendering worst case
17
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 100000 wide 1024
18
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 200000 wide 1024
19
+ //! # wide rows, few changed cells: id + 8 512-byte columns, only one differing
20
+ //! ROW_DIFF_THREADS=18 /usr/bin/time -l target/release/examples/row_diff_rss 150000 manycols 8 512
21
+ //! # duplicate-heavy shape: every key duplicated, wide string key
22
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 dup 16
23
+ //! /usr/bin/time -l target/release/examples/row_diff_rss 200000 dup 1024
24
+ //! # size-gate peek: identical wide-cell sides (zero changes); ROW_DIFF_BATCH
25
+ //! # sets the producer's batch size, ROW_DIFF_THREADS the worker count
26
+ //! ROW_DIFF_BATCH=100 ROW_DIFF_THREADS=18 /usr/bin/time -l target/release/examples/row_diff_rss 49999 widesame 8192
27
+ //! ```
28
+ //!
29
+ //! Each side is generated on the fly, batch by batch, and nothing is retained
30
+ //! between batches, so the process's peak RSS is the diff's own state, not the
31
+ //! table data. The **linear** shape (`id`, `value` int64 columns) exercises the
32
+ //! per-row hash vectors: the left is ids `0..n`, the right `step..n + step` with
33
+ //! `step = n / 100`, so 1% removed, 1% added, ~2% changed — and the cell pass
34
+ //! materializes those ~2% changed rows on both sides. The **nochange** variant
35
+ //! keeps the 1% added/removed but makes every shared row equal, so the cell pass
36
+ //! materializes nothing: the difference in peak RSS between it and the default
37
+ //! run is the cell pass's cost. The **allchange** variant drops the offset and
38
+ //! changes every shared row (no added/removed), so the cell pass materializes
39
+ //! and renders every row. The **wide** shape (`id` int64, `value` a
40
+ //! `value_width`-byte Utf8 that differs between the sides) changes every row too
41
+ //! and renders `value_width` bytes per changed cell — the rendering worst case,
42
+ //! whose peak RSS scales with changed cells times cell width. The **dup** shape
43
+ //! (`key` Utf8 of the given width, `value` int64) makes every key appear twice
44
+ //! on each side, so every distinct
45
+ //! key is a duplicate and the whole `duplicate_keys` report is materialized —
46
+ //! the term that scales with distinct duplicated keys times the key width.
47
+
48
+ use onix_arrow::{TableDiffOptions, diff_tables};
49
+
50
+ #[path = "shared/gen_shapes.rs"]
51
+ mod gen_shapes;
52
+ use gen_shapes::{Case, Generated, batch_rows};
53
+
54
+ /// Options for the diff, honoring a `ROW_DIFF_THREADS` override so the parallel
55
+ /// path's peak RSS can be compared against the single-threaded baseline; unset
56
+ /// uses the default (available parallelism).
57
+ fn options_from_env(key: &str) -> TableDiffOptions {
58
+ let mut options = TableDiffOptions::new(vec![key.to_string()]);
59
+ if let Some(threads) = std::env::var("ROW_DIFF_THREADS")
60
+ .ok()
61
+ .and_then(|v| v.parse().ok())
62
+ .and_then(std::num::NonZeroUsize::new)
63
+ {
64
+ options = options
65
+ .with_threads(threads)
66
+ .expect("ROW_DIFF_THREADS within MAX_THREADS");
67
+ }
68
+ options
69
+ }
70
+
71
+ fn main() {
72
+ let args: Vec<String> = std::env::args().collect();
73
+ let rows: i64 = args
74
+ .get(1)
75
+ .and_then(|a| a.parse().ok())
76
+ .unwrap_or(1_000_000);
77
+ let mode = args.get(2).map_or("", String::as_str);
78
+ let width: usize = args
79
+ .get(3)
80
+ .and_then(|a| a.parse().ok())
81
+ .unwrap_or(if mode == "wide" { 1024 } else { 16 });
82
+
83
+ let (case, label) = match mode {
84
+ "" | "linear" => (Case::Linear, String::new()),
85
+ "nochange" => (Case::NoChange, " (nochange baseline)".to_string()),
86
+ "allchange" => (Case::AllChange, " (all changed)".to_string()),
87
+ "wide" => (Case::Wide(width), format!(" (wide, value_width={width})")),
88
+ "widesame" => (
89
+ Case::WideSame(width),
90
+ format!(" (widesame, value_width={width})"),
91
+ ),
92
+ "manycols" => {
93
+ let ncols: usize = args.get(3).and_then(|a| a.parse().ok()).unwrap_or(8);
94
+ let width: usize = args.get(4).and_then(|a| a.parse().ok()).unwrap_or(512);
95
+ (
96
+ Case::ManyCols { ncols, width },
97
+ format!(" (manycols, ncols={ncols}, width={width})"),
98
+ )
99
+ }
100
+ "dup" => (Case::Dup(width), format!(" (dup, key_width={width})")),
101
+ other => {
102
+ eprintln!(
103
+ "unknown mode {other:?}; expected linear (the default), nochange, allchange, wide, widesame, manycols or dup"
104
+ );
105
+ std::process::exit(2);
106
+ }
107
+ };
108
+ let (schema, left_shape, right_shape, key) = case.build(rows);
109
+
110
+ let batch = batch_rows();
111
+ let left = Generated {
112
+ schema: schema.clone(),
113
+ rows,
114
+ shape: left_shape,
115
+ batch,
116
+ };
117
+ let right = Generated {
118
+ schema,
119
+ rows,
120
+ shape: right_shape,
121
+ batch,
122
+ };
123
+
124
+ let options = options_from_env(key);
125
+ let start = std::time::Instant::now();
126
+ let diff = diff_tables(&left, &right, &options).expect("diff succeeds");
127
+ let elapsed = start.elapsed();
128
+ let summary = diff.summary();
129
+
130
+ println!("rows per side: {rows}{label}");
131
+ println!("threads: {}", options.threads());
132
+ println!("wall: {:.2}s", elapsed.as_secs_f64());
133
+ println!(
134
+ "rows_added={} rows_removed={} rows_changed={} duplicate_keys={} cells_changed={}",
135
+ summary.rows_added,
136
+ summary.rows_removed,
137
+ summary.rows_changed,
138
+ summary.duplicate_keys,
139
+ summary.cells_changed
140
+ );
141
+ }