deepdiff-rs 0.11.1__tar.gz → 0.11.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/Cargo.lock +4 -4
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/Cargo.toml +1 -1
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/PKG-INFO +10 -13
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/README.md +9 -12
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/Cargo.toml +16 -0
- deepdiff_rs-0.11.2/crates/onix-arrow/examples/row_diff_profile.rs +316 -0
- deepdiff_rs-0.11.2/crates/onix-arrow/examples/row_diff_rss.rs +141 -0
- deepdiff_rs-0.11.2/crates/onix-arrow/examples/shared/gen_shapes.rs +245 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/lib.rs +2 -0
- deepdiff_rs-0.11.2/crates/onix-arrow/src/profile.rs +418 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/row_diff.rs +237 -78
- deepdiff_rs-0.11.2/crates/onix-arrow/tests/profile_passes.rs +116 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/benchmarks/bench_bindings.py +26 -210
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/errors.rs +1 -10
- deepdiff_rs-0.11.2/crates/onix-py/src/fast_path.rs +42 -0
- deepdiff_rs-0.11.2/crates/onix-py/src/guard.rs +296 -0
- deepdiff_rs-0.11.2/crates/onix-py/src/lib.rs +26 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/conftest.py +3 -12
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_bindings_memory.py +7 -19
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_differential_fuzz.py +78 -687
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_non_finite.py +1 -7
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_signed_zero.py +10 -43
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_stub_signatures.py +7 -24
- deepdiff_rs-0.11.2/crates/onix-py/tests/test_suite_hygiene.py +32 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_table_row_diff.py +5 -9
- deepdiff_rs-0.11.1/crates/onix-arrow/examples/row_diff_rss.rs +0 -359
- deepdiff_rs-0.11.1/crates/onix-py/src/fast_path.rs +0 -64
- deepdiff_rs-0.11.1/crates/onix-py/src/guard.rs +0 -452
- deepdiff_rs-0.11.1/crates/onix-py/src/lib.rs +0 -52
- deepdiff_rs-0.11.1/crates/onix-py/tests/test_suite_hygiene.py +0 -39
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/examples/type_stack_cost.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/error.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/json_rows.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/options.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/schema.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/spool.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-arrow/src/table_diff.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/Cargo.toml +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/examples/stack_frame_cost.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/datetime.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/datetime_tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/array.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/dispatch.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/mod.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/object.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/options.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/scalar.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/set.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/diff/tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/error.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/distance.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/fxhash.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/hash.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/memo.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/mod.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/pairing.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/ignore_order/tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lcs.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lcs_tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/lib.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/path.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/report.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/report_tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/test_support.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/unified_diff.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/unified_diff_tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/value.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/src/value_tests.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/golden.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/ignore_order_memory.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/memory_footprint.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/proptest_diff.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-core/tests/proptest_ignore_order.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/.python-version +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/Cargo.toml +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/deepdiff_rs.pyi +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/arrow.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/convert.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/src/deepdiff.rs +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_conversions.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_datetimes.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_depth_guard.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_golden_parity.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_sets.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_smoke.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_stub_mypy.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_table_diff.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_timedeltas.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_times.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_tuples.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/crates/onix-py/tests/test_wheel_contents.py +0 -0
- {deepdiff_rs-0.11.1 → deepdiff_rs-0.11.2}/pyproject.toml +0 -0
|
@@ -629,7 +629,7 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
|
|
629
629
|
|
|
630
630
|
[[package]]
|
|
631
631
|
name = "onix-arrow"
|
|
632
|
-
version = "0.11.
|
|
632
|
+
version = "0.11.2"
|
|
633
633
|
dependencies = [
|
|
634
634
|
"arrow-array",
|
|
635
635
|
"arrow-buffer",
|
|
@@ -648,7 +648,7 @@ dependencies = [
|
|
|
648
648
|
|
|
649
649
|
[[package]]
|
|
650
650
|
name = "onix-cli"
|
|
651
|
-
version = "0.11.
|
|
651
|
+
version = "0.11.2"
|
|
652
652
|
dependencies = [
|
|
653
653
|
"onix-core",
|
|
654
654
|
"serde_json",
|
|
@@ -656,7 +656,7 @@ dependencies = [
|
|
|
656
656
|
|
|
657
657
|
[[package]]
|
|
658
658
|
name = "onix-core"
|
|
659
|
-
version = "0.11.
|
|
659
|
+
version = "0.11.2"
|
|
660
660
|
dependencies = [
|
|
661
661
|
"num-bigint",
|
|
662
662
|
"num-traits",
|
|
@@ -669,7 +669,7 @@ dependencies = [
|
|
|
669
669
|
|
|
670
670
|
[[package]]
|
|
671
671
|
name = "onix-py"
|
|
672
|
-
version = "0.11.
|
|
672
|
+
version = "0.11.2"
|
|
673
673
|
dependencies = [
|
|
674
674
|
"arrow-array",
|
|
675
675
|
"arrow-schema",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: deepdiff-rs
|
|
3
|
-
Version: 0.11.
|
|
3
|
+
Version: 0.11.2
|
|
4
4
|
Classifier: Programming Language :: Python :: 3
|
|
5
5
|
Classifier: Programming Language :: Rust
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -124,15 +124,10 @@ It reports the **schema** diff (which columns were added, removed, or changed ty
|
|
|
124
124
|
import pyarrow as pa
|
|
125
125
|
from deepdiff_rs import diff_tables
|
|
126
126
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
})
|
|
131
|
-
right = pa.table({
|
|
132
|
-
"id": pa.array([2, 3, 4], pa.int64()),
|
|
133
|
-
"amount": pa.array([20, 31, 40], pa.int64()),
|
|
134
|
-
"note": pa.array(["a", "b", "c"], pa.string()),
|
|
135
|
-
})
|
|
127
|
+
# 9 is a duplicate key
|
|
128
|
+
left = pa.table({"id": pa.array([1, 2, 3, 9, 9], pa.int64()), "amount": pa.array([10, 20, 30, 90, 91], pa.int32())})
|
|
129
|
+
right = pa.table({"id": pa.array([2, 3, 4], pa.int64()), "amount": pa.array([20, 31, 40], pa.int64()),
|
|
130
|
+
"note": pa.array(["a", "b", "c"], pa.string())})
|
|
136
131
|
|
|
137
132
|
diff = diff_tables(left, right, key=["id"])
|
|
138
133
|
print(diff.summary(), "added ids:", pa.table(diff.rows_added()).column("id").to_pylist(), "removed ids:", pa.table(diff.rows_removed()).column("id").to_pylist())
|
|
@@ -144,11 +139,13 @@ print("cells changed:", pa.table(diff.cells_changed()).to_pylist(), "duplicate k
|
|
|
144
139
|
cells changed: [{'id': 3, 'column': 'amount', 'old_value': '30', 'new_value': '31', 'change': 'value_changed'}] duplicate keys: [{'id': 9, 'left_count': 2, 'right_count': 0}]
|
|
145
140
|
```
|
|
146
141
|
|
|
147
|
-
A key appearing more than once on either side is reported in `duplicate_keys` (
|
|
142
|
+
A key appearing more than once on either side is reported in `duplicate_keys` (`left_count`/`right_count`) and excluded from added/removed/changed; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is skipped rather than compared. The exact rules are on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
148
143
|
|
|
149
|
-
Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings
|
|
144
|
+
Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings sharing a logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes; nullability is ignored but reported in each record. The full rules are on `normalized_type`/`map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs). Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` needs no pyarrow, and `.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but needs pyarrow installed regardless.
|
|
150
145
|
|
|
151
|
-
`pyarrow` is optional:
|
|
146
|
+
`pyarrow` is optional: `pip install deepdiff-rs[arrow]`. It's needed only for `to_pyarrow()` and passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. An object implementing neither Arrow protocol raises `TypeError`; `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra. `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed; see [Known limitations](#known-limitations) for its row cap.
|
|
147
|
+
|
|
148
|
+
Beyond a join-based diff, `diff_tables` reports duplicate and null keys rather than silently multiplying or dropping them, distinguishes `type_changed` from `value_changed` per cell, renders every value by one documented rule set (Python's, with `Duration` the one exception -- an ISO 8601 string, not Python's own `str()`), and produces byte-identical output at any thread count with memory proportional to row count from a streamed input. What that costs over a hand-rolled join is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)'s polars-backed ceiling section.
|
|
152
149
|
|
|
153
150
|
## Performance
|
|
154
151
|
|
|
@@ -106,15 +106,10 @@ It reports the **schema** diff (which columns were added, removed, or changed ty
|
|
|
106
106
|
import pyarrow as pa
|
|
107
107
|
from deepdiff_rs import diff_tables
|
|
108
108
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
})
|
|
113
|
-
right = pa.table({
|
|
114
|
-
"id": pa.array([2, 3, 4], pa.int64()),
|
|
115
|
-
"amount": pa.array([20, 31, 40], pa.int64()),
|
|
116
|
-
"note": pa.array(["a", "b", "c"], pa.string()),
|
|
117
|
-
})
|
|
109
|
+
# 9 is a duplicate key
|
|
110
|
+
left = pa.table({"id": pa.array([1, 2, 3, 9, 9], pa.int64()), "amount": pa.array([10, 20, 30, 90, 91], pa.int32())})
|
|
111
|
+
right = pa.table({"id": pa.array([2, 3, 4], pa.int64()), "amount": pa.array([20, 31, 40], pa.int64()),
|
|
112
|
+
"note": pa.array(["a", "b", "c"], pa.string())})
|
|
118
113
|
|
|
119
114
|
diff = diff_tables(left, right, key=["id"])
|
|
120
115
|
print(diff.summary(), "added ids:", pa.table(diff.rows_added()).column("id").to_pylist(), "removed ids:", pa.table(diff.rows_removed()).column("id").to_pylist())
|
|
@@ -126,11 +121,13 @@ print("cells changed:", pa.table(diff.cells_changed()).to_pylist(), "duplicate k
|
|
|
126
121
|
cells changed: [{'id': 3, 'column': 'amount', 'old_value': '30', 'new_value': '31', 'change': 'value_changed'}] duplicate keys: [{'id': 9, 'left_count': 2, 'right_count': 0}]
|
|
127
122
|
```
|
|
128
123
|
|
|
129
|
-
A key appearing more than once on either side is reported in `duplicate_keys` (
|
|
124
|
+
A key appearing more than once on either side is reported in `duplicate_keys` (`left_count`/`right_count`) and excluded from added/removed/changed; a null key matches its counterpart and is counted in `null_keys`. Rows are compared by the non-key columns present on *both* sides, with onix's value semantics (integers and integral floats fold together, all NaNs compare equal, `1.00` equals `1.0000`, a timestamp compares by its instant and a time or duration by its value across units, dictionary-encoded values equal their plain form, and null equals null); a nested non-key column is skipped rather than compared. The exact rules are on the hashing functions in [`crates/onix-arrow/src/row_diff.rs`](crates/onix-arrow/src/row_diff.rs).
|
|
130
125
|
|
|
131
|
-
Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings
|
|
126
|
+
Type comparison uses the full logical Arrow type (timestamp unit and timezone, decimal precision and scale, and so on), but physical encodings sharing a logical type compare equal — a dictionary-encoded string equals a plain string, polars' `Utf8View` equals pyarrow's `Utf8`, the list variants normalize together, and a map compares equal however a library spells it — so the same table read through pyarrow, polars, or DuckDB reports no spurious type changes; nullability is ignored but reported in each record. The full rules are on `normalized_type`/`map_entries` in [`crates/onix-arrow/src/schema.rs`](crates/onix-arrow/src/schema.rs). Column names must be unique on each side; a repeated name raises `ValueError`. `diff.schema_arrow` is the same result as an Arrow table: it implements `__arrow_c_stream__`, so `polars.DataFrame(diff.schema_arrow)` needs no pyarrow, and `.to_pyarrow()` returns a `pyarrow.Table`; `pandas.api.interchange.from_dataframe(diff.schema_arrow)` also works, but needs pyarrow installed regardless.
|
|
132
127
|
|
|
133
|
-
`pyarrow` is optional:
|
|
128
|
+
`pyarrow` is optional: `pip install deepdiff-rs[arrow]`. It's needed only for `to_pyarrow()` and passing pyarrow objects in — importing `deepdiff_rs` and diffing polars or DuckDB tables need it not at all. An object implementing neither Arrow protocol raises `TypeError`; `to_pyarrow()` without pyarrow installed raises `ImportError` naming the extra. `diff.to_json()` gives the whole diff — schema, summary, and `rows_added`/`rows_removed`/`cells_changed`/`duplicate_keys` in full, one JSON object per row — as a single string with no pyarrow, polars, or pandas needed; see [Known limitations](#known-limitations) for its row cap.
|
|
129
|
+
|
|
130
|
+
Beyond a join-based diff, `diff_tables` reports duplicate and null keys rather than silently multiplying or dropping them, distinguishes `type_changed` from `value_changed` per cell, renders every value by one documented rule set (Python's, with `Duration` the one exception -- an ISO 8601 string, not Python's own `str()`), and produces byte-identical output at any thread count with memory proportional to row count from a streamed input. What that costs over a hand-rolled join is in [`perf/arrow/RESULTS.md`](perf/arrow/RESULTS.md)'s polars-backed ceiling section.
|
|
134
131
|
|
|
135
132
|
## Performance
|
|
136
133
|
|
|
@@ -9,6 +9,12 @@ description = "Arrow table diffing (schema and keyed rows) built on the onix dif
|
|
|
9
9
|
[lints]
|
|
10
10
|
workspace = true
|
|
11
11
|
|
|
12
|
+
[features]
|
|
13
|
+
# Compiles in the per-pass wall-time and peak-RSS instrumentation the
|
|
14
|
+
# `row_diff_profile` example reads. Off by default and never enabled by the
|
|
15
|
+
# release wheel, so the instrumentation is absent from the shipped `.so`.
|
|
16
|
+
profile = []
|
|
17
|
+
|
|
12
18
|
[dependencies]
|
|
13
19
|
# Pinned to an exact Arrow version so this crate and the `pyo3-arrow` bridge
|
|
14
20
|
# in `onix-py` (which requires `arrow` ^59) resolve to the identical Arrow
|
|
@@ -52,3 +58,13 @@ serde_json = "1"
|
|
|
52
58
|
proptest = "1"
|
|
53
59
|
# `half::f16` builds a Float16 test column; the same version arrow-array uses.
|
|
54
60
|
half = "2"
|
|
61
|
+
|
|
62
|
+
# The profiling harness reads the per-pass instrumentation, so it only builds
|
|
63
|
+
# with the `profile` feature (and never as part of the default build or wheel).
|
|
64
|
+
[[example]]
|
|
65
|
+
name = "row_diff_profile"
|
|
66
|
+
required-features = ["profile"]
|
|
67
|
+
|
|
68
|
+
[[test]]
|
|
69
|
+
name = "profile_passes"
|
|
70
|
+
required-features = ["profile"]
|
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
//! Per-pass wall-time and peak-RSS profile of one keyed row diff, the committed
|
|
2
|
+
//! harness every row-diff performance change posts a before/after table from.
|
|
3
|
+
//! Builds only with the `profile` feature, which the release wheel never enables.
|
|
4
|
+
//!
|
|
5
|
+
//! Each invocation runs a discarded warm-up diff, a timed uninstrumented diff
|
|
6
|
+
//! (the `uninstrumented wall` line), and an instrumented diff whose passes make
|
|
7
|
+
//! up the table. The closure line compares the passes' sum with the instrumented
|
|
8
|
+
//! wall minus the profiler's own boundary `ps` reads; `share` is each row's wall
|
|
9
|
+
//! over that net wall. Peak RSS is the process's, third diff in the process.
|
|
10
|
+
//!
|
|
11
|
+
//! - **file**: reads each side from an uncompressed Arrow IPC file, re-opened
|
|
12
|
+
//! and re-decoded by every pass. Convert a parquet fixture once with
|
|
13
|
+
//! `python -c "import pyarrow.parquet as p, pyarrow.feather as f;
|
|
14
|
+
//! f.write_feather(p.read_table('a.parquet'), 'a.arrow', compression='uncompressed')"`.
|
|
15
|
+
//! - **generated**: spools both sides of a deterministic proxy shape to anonymous
|
|
16
|
+
//! Arrow IPC files (the `spool write` line times the IPC writer alone).
|
|
17
|
+
//!
|
|
18
|
+
//! ```sh
|
|
19
|
+
//! cargo build -p onix-arrow --release --features profile --example row_diff_profile
|
|
20
|
+
//! target/release/examples/row_diff_profile file a.arrow b.arrow --key id --threads 18
|
|
21
|
+
//! target/release/examples/row_diff_profile 1000000 linear 18
|
|
22
|
+
//! target/release/examples/row_diff_profile 1000000 manycols 18 34 64
|
|
23
|
+
//! ```
|
|
24
|
+
//!
|
|
25
|
+
//! Generated args: `[rows [shape [threads [shape params...]]]]`, key `id`,
|
|
26
|
+
//! `threads` defaulting to `ROW_DIFF_THREADS` or available parallelism. Shapes:
|
|
27
|
+
//!
|
|
28
|
+
//! - `linear`: `id`/`value` int64; 1% of keys added, 1% removed, ~2% changed.
|
|
29
|
+
//! - `allchange`: `id`/`value` int64, every row changed.
|
|
30
|
+
//! - `wide [width=512]`: `id` plus one `width`-byte string, every row changed.
|
|
31
|
+
//! - `manycols [ncols=34 [width=64]]`: `id` plus `ncols` `width`-byte strings,
|
|
32
|
+
//! only the first differing, so the spill carries every value column.
|
|
33
|
+
|
|
34
|
+
use std::fs::File;
|
|
35
|
+
use std::io::BufReader;
|
|
36
|
+
use std::num::NonZeroUsize;
|
|
37
|
+
use std::path::PathBuf;
|
|
38
|
+
use std::time::{Duration, Instant};
|
|
39
|
+
|
|
40
|
+
use arrow_array::RecordBatchReader;
|
|
41
|
+
use arrow_ipc::reader::FileReader;
|
|
42
|
+
use arrow_schema::SchemaRef;
|
|
43
|
+
use onix_arrow::{TableDiffError, TableDiffOptions, TableInput, diff_tables, profile, spool};
|
|
44
|
+
|
|
45
|
+
#[path = "shared/gen_shapes.rs"]
|
|
46
|
+
mod gen_shapes;
|
|
47
|
+
use gen_shapes::{Case, Generated, batch_rows};
|
|
48
|
+
|
|
49
|
+
/// A table read from an Arrow IPC file, re-opened on every `open`.
|
|
50
|
+
struct FileInput {
|
|
51
|
+
path: PathBuf,
|
|
52
|
+
schema: SchemaRef,
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
impl FileInput {
|
|
56
|
+
fn load(path: &str) -> FileInput {
|
|
57
|
+
let reader = open_ipc(path).unwrap_or_else(|e| panic!("open {path}: {e}"));
|
|
58
|
+
FileInput {
|
|
59
|
+
path: PathBuf::from(path),
|
|
60
|
+
schema: reader.schema(),
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
fn open_ipc(path: &str) -> Result<FileReader<BufReader<File>>, TableDiffError> {
|
|
66
|
+
let file = File::open(path).map_err(|e| TableDiffError::Read {
|
|
67
|
+
message: format!("open {path}: {e}"),
|
|
68
|
+
})?;
|
|
69
|
+
FileReader::try_new(BufReader::new(file), None).map_err(|e| TableDiffError::Read {
|
|
70
|
+
message: format!("read Arrow IPC {path}: {e}"),
|
|
71
|
+
})
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
impl TableInput for FileInput {
|
|
75
|
+
fn schema(&self) -> SchemaRef {
|
|
76
|
+
self.schema.clone()
|
|
77
|
+
}
|
|
78
|
+
fn open(&self) -> Result<Box<dyn RecordBatchReader + Send>, TableDiffError> {
|
|
79
|
+
let path = self.path.to_string_lossy();
|
|
80
|
+
Ok(Box::new(open_ipc(&path)?))
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/// A side spooled to an anonymous Arrow IPC file, re-read on every `open` — the
|
|
85
|
+
/// same re-openable spool the Python bindings hand the row diff.
|
|
86
|
+
struct Spooled {
|
|
87
|
+
file: File,
|
|
88
|
+
schema: SchemaRef,
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
impl TableInput for Spooled {
|
|
92
|
+
fn schema(&self) -> SchemaRef {
|
|
93
|
+
self.schema.clone()
|
|
94
|
+
}
|
|
95
|
+
fn open(&self) -> Result<Box<dyn RecordBatchReader + Send>, TableDiffError> {
|
|
96
|
+
Ok(Box::new(spool::reopen(&self.file)?))
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/// Spools one generated side, returning it and the time spent in the IPC
|
|
101
|
+
/// writer (generation excluded).
|
|
102
|
+
fn spool_side(side: &Generated) -> (Spooled, Duration) {
|
|
103
|
+
let (file, mut writer) = spool::open(&side.schema).expect("open spool");
|
|
104
|
+
let mut writing = Duration::ZERO;
|
|
105
|
+
for batch in side.open().expect("open generator") {
|
|
106
|
+
let batch = batch.expect("generate batch");
|
|
107
|
+
let start = Instant::now();
|
|
108
|
+
writer.write(&batch).expect("spool write");
|
|
109
|
+
writing += start.elapsed();
|
|
110
|
+
}
|
|
111
|
+
let start = Instant::now();
|
|
112
|
+
writer.finish().expect("spool finish");
|
|
113
|
+
writing += start.elapsed();
|
|
114
|
+
let spooled = Spooled {
|
|
115
|
+
file,
|
|
116
|
+
schema: side.schema.clone(),
|
|
117
|
+
};
|
|
118
|
+
(spooled, writing)
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/// Runs one discarded warm-up diff, one timed uninstrumented diff, and one
|
|
122
|
+
/// instrumented diff, then prints both walls, the closure of the instrumented
|
|
123
|
+
/// wall over its passes, and the per-pass table.
|
|
124
|
+
fn run(left: &impl TableInput, right: &impl TableInput, options: &TableDiffOptions, label: &str) {
|
|
125
|
+
let _ = diff_tables(left, right, options).expect("diff succeeds");
|
|
126
|
+
let start = Instant::now();
|
|
127
|
+
let diff = diff_tables(left, right, options).expect("diff succeeds");
|
|
128
|
+
let uninstrumented = start.elapsed().as_secs_f64();
|
|
129
|
+
let summary = diff.summary();
|
|
130
|
+
|
|
131
|
+
let session = profile::begin();
|
|
132
|
+
let start = Instant::now();
|
|
133
|
+
let _ = diff_tables(left, right, options).expect("diff succeeds");
|
|
134
|
+
let instrumented = start.elapsed().as_secs_f64();
|
|
135
|
+
let report = session.finish();
|
|
136
|
+
|
|
137
|
+
let boundary: f64 = report
|
|
138
|
+
.iter()
|
|
139
|
+
.filter(|row| row.label == profile::BOUNDARY_LABEL)
|
|
140
|
+
.map(|row| row.wall_secs)
|
|
141
|
+
.sum();
|
|
142
|
+
let passes: f64 = report
|
|
143
|
+
.iter()
|
|
144
|
+
.filter(|row| row.peak_rss_mib.is_some())
|
|
145
|
+
.map(|row| row.wall_secs)
|
|
146
|
+
.sum();
|
|
147
|
+
let closed = instrumented - boundary;
|
|
148
|
+
|
|
149
|
+
println!("{label}");
|
|
150
|
+
println!("threads: {}", options.threads());
|
|
151
|
+
println!("uninstrumented wall: {uninstrumented:.3} s");
|
|
152
|
+
println!("instrumented wall: {instrumented:.3} s");
|
|
153
|
+
println!("boundary ps reads: {boundary:.3} s");
|
|
154
|
+
println!(
|
|
155
|
+
"passes sum: {passes:.3} s of {closed:.3} s (instrumented minus boundary); residual {:.3} s ({:.1}%)",
|
|
156
|
+
closed - passes,
|
|
157
|
+
100.0 * (closed - passes) / closed
|
|
158
|
+
);
|
|
159
|
+
println!(
|
|
160
|
+
"rows_added={} rows_removed={} rows_changed={} duplicate_keys={} cells_changed={}",
|
|
161
|
+
summary.rows_added,
|
|
162
|
+
summary.rows_removed,
|
|
163
|
+
summary.rows_changed,
|
|
164
|
+
summary.duplicate_keys,
|
|
165
|
+
summary.cells_changed
|
|
166
|
+
);
|
|
167
|
+
println!();
|
|
168
|
+
println!(
|
|
169
|
+
"{:<44} {:>10} {:>8} {:>14}",
|
|
170
|
+
"pass", "wall (s)", "share", "peak RSS (MB)"
|
|
171
|
+
);
|
|
172
|
+
for row in report
|
|
173
|
+
.iter()
|
|
174
|
+
.filter(|row| row.label != profile::BOUNDARY_LABEL)
|
|
175
|
+
{
|
|
176
|
+
let share = 100.0 * row.wall_secs / closed;
|
|
177
|
+
match row.peak_rss_mib {
|
|
178
|
+
Some(rss) => println!(
|
|
179
|
+
"{:<44} {:>10.3} {:>7.1}% {:>14.1}",
|
|
180
|
+
row.label, row.wall_secs, share, rss
|
|
181
|
+
),
|
|
182
|
+
None => println!(
|
|
183
|
+
"{:<44} {:>10.3} {:>7.1}% {:>14}",
|
|
184
|
+
format!(" {}", row.label),
|
|
185
|
+
row.wall_secs,
|
|
186
|
+
share,
|
|
187
|
+
"-"
|
|
188
|
+
),
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const USAGE: &str =
|
|
194
|
+
"usage: row_diff_profile file <left.arrow> <right.arrow> [--key col[,col...]] [--threads N]
|
|
195
|
+
row_diff_profile [rows [linear|allchange|wide|manycols [threads [shape params...]]]]";
|
|
196
|
+
|
|
197
|
+
fn usage_error(message: &str) -> ! {
|
|
198
|
+
eprintln!("{message}\n{USAGE}");
|
|
199
|
+
std::process::exit(2);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/// Parses a positive integer argument, exiting with a usage error otherwise.
|
|
203
|
+
fn positive(what: &str, value: &str) -> NonZeroUsize {
|
|
204
|
+
value.parse().unwrap_or_else(|_| {
|
|
205
|
+
usage_error(&format!("{what} must be a positive integer, got {value:?}"))
|
|
206
|
+
})
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
fn options_for(keys: Vec<String>, threads: Option<NonZeroUsize>) -> TableDiffOptions {
|
|
210
|
+
let threads = threads.or_else(|| {
|
|
211
|
+
std::env::var("ROW_DIFF_THREADS")
|
|
212
|
+
.ok()
|
|
213
|
+
.map(|v| positive("ROW_DIFF_THREADS", &v))
|
|
214
|
+
});
|
|
215
|
+
let options = TableDiffOptions::new(keys);
|
|
216
|
+
match threads {
|
|
217
|
+
Some(t) => options
|
|
218
|
+
.with_threads(t)
|
|
219
|
+
.unwrap_or_else(|e| usage_error(&e.to_string())),
|
|
220
|
+
None => options,
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
fn file_mode(args: &[String]) {
|
|
225
|
+
let [left_path, right_path, flags @ ..] = args else {
|
|
226
|
+
usage_error("file mode needs <left.arrow> <right.arrow>");
|
|
227
|
+
};
|
|
228
|
+
let mut keys = Vec::new();
|
|
229
|
+
let mut threads = None;
|
|
230
|
+
let mut flags = flags.iter();
|
|
231
|
+
while let Some(flag) = flags.next() {
|
|
232
|
+
let Some(value) = flags.next() else {
|
|
233
|
+
usage_error(&format!("{flag} needs a value"));
|
|
234
|
+
};
|
|
235
|
+
match flag.as_str() {
|
|
236
|
+
"--key" => keys.extend(value.split(',').map(str::to_string)),
|
|
237
|
+
"--threads" => threads = Some(positive("--threads", value)),
|
|
238
|
+
_ => usage_error(&format!("unknown flag {flag:?}")),
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
if keys.is_empty() {
|
|
242
|
+
keys.push("id".to_string());
|
|
243
|
+
}
|
|
244
|
+
let left = FileInput::load(left_path);
|
|
245
|
+
let right = FileInput::load(right_path);
|
|
246
|
+
run(
|
|
247
|
+
&left,
|
|
248
|
+
&right,
|
|
249
|
+
&options_for(keys, threads),
|
|
250
|
+
&format!("file: {left_path} vs {right_path}"),
|
|
251
|
+
);
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
fn generated_mode(args: &[String]) {
|
|
255
|
+
let rows = args.first().map_or(1_000_000, |a| {
|
|
256
|
+
i64::try_from(positive("rows", a).get()).unwrap_or_else(|_| usage_error("rows too large"))
|
|
257
|
+
});
|
|
258
|
+
let shape = args.get(1).map_or("linear", String::as_str);
|
|
259
|
+
let threads = args.get(2).map(|a| positive("threads", a));
|
|
260
|
+
let params: Vec<usize> = args
|
|
261
|
+
.iter()
|
|
262
|
+
.skip(3)
|
|
263
|
+
.map(|a| positive("shape parameter", a).get())
|
|
264
|
+
.collect();
|
|
265
|
+
let param = |i: usize, default: usize| params.get(i).copied().unwrap_or(default);
|
|
266
|
+
let (case, arity) = match shape {
|
|
267
|
+
"linear" => (Case::Linear, 0),
|
|
268
|
+
"allchange" => (Case::AllChange, 0),
|
|
269
|
+
"wide" => (Case::Wide(param(0, 512)), 1),
|
|
270
|
+
"manycols" => (
|
|
271
|
+
Case::ManyCols {
|
|
272
|
+
ncols: param(0, 34),
|
|
273
|
+
width: param(1, 64),
|
|
274
|
+
},
|
|
275
|
+
2,
|
|
276
|
+
),
|
|
277
|
+
_ => usage_error(&format!("unknown shape {shape:?}")),
|
|
278
|
+
};
|
|
279
|
+
if params.len() > arity {
|
|
280
|
+
usage_error(&format!("shape {shape} takes at most {arity} parameter(s)"));
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
let (schema, left_shape, right_shape, key) = case.build(rows);
|
|
284
|
+
let options = options_for(vec![key.to_string()], threads);
|
|
285
|
+
let batch = batch_rows();
|
|
286
|
+
let (left, left_write) = spool_side(&Generated {
|
|
287
|
+
schema: schema.clone(),
|
|
288
|
+
rows,
|
|
289
|
+
shape: left_shape,
|
|
290
|
+
batch,
|
|
291
|
+
});
|
|
292
|
+
let (right, right_write) = spool_side(&Generated {
|
|
293
|
+
schema,
|
|
294
|
+
rows,
|
|
295
|
+
shape: right_shape,
|
|
296
|
+
batch,
|
|
297
|
+
});
|
|
298
|
+
println!(
|
|
299
|
+
"spool write (both sides, before the diff): {:.3} s",
|
|
300
|
+
(left_write + right_write).as_secs_f64()
|
|
301
|
+
);
|
|
302
|
+
run(
|
|
303
|
+
&left,
|
|
304
|
+
&right,
|
|
305
|
+
&options,
|
|
306
|
+
&format!("rows per side: {rows} (shape={shape})"),
|
|
307
|
+
);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
fn main() {
|
|
311
|
+
let args: Vec<String> = std::env::args().skip(1).collect();
|
|
312
|
+
match args.split_first() {
|
|
313
|
+
Some((mode, rest)) if mode == "file" => file_mode(rest),
|
|
314
|
+
_ => generated_mode(&args),
|
|
315
|
+
}
|
|
316
|
+
}
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
//! Measures the keyed row diff's peak memory and wall time, to check the memory
|
|
2
|
+
//! bounds the README states.
|
|
3
|
+
//!
|
|
4
|
+
//! Run under the OS's max-RSS reporter:
|
|
5
|
+
//!
|
|
6
|
+
//! ```sh
|
|
7
|
+
//! cargo build -p onix-arrow --release --example row_diff_rss
|
|
8
|
+
//! # linear shape (default): mostly-matching rows, 1% added/removed, ~2% changed
|
|
9
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 1000000
|
|
10
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 10000000
|
|
11
|
+
//! # same shape with no changed rows: the cell pass materializes nothing, the
|
|
12
|
+
//! # pass-one baseline the ~2%-changed run is measured against
|
|
13
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 nochange
|
|
14
|
+
//! # every row changed (narrow int cells): the cell pass at full width
|
|
15
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 allchange
|
|
16
|
+
//! # every row changed with a wide (1 KB) string cell: the rendering worst case
|
|
17
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 100000 wide 1024
|
|
18
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 200000 wide 1024
|
|
19
|
+
//! # wide rows, few changed cells: id + 8 512-byte columns, only one differing
|
|
20
|
+
//! ROW_DIFF_THREADS=18 /usr/bin/time -l target/release/examples/row_diff_rss 150000 manycols 8 512
|
|
21
|
+
//! # duplicate-heavy shape: every key duplicated, wide string key
|
|
22
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 1000000 dup 16
|
|
23
|
+
//! /usr/bin/time -l target/release/examples/row_diff_rss 200000 dup 1024
|
|
24
|
+
//! # size-gate peek: identical wide-cell sides (zero changes); ROW_DIFF_BATCH
|
|
25
|
+
//! # sets the producer's batch size, ROW_DIFF_THREADS the worker count
|
|
26
|
+
//! ROW_DIFF_BATCH=100 ROW_DIFF_THREADS=18 /usr/bin/time -l target/release/examples/row_diff_rss 49999 widesame 8192
|
|
27
|
+
//! ```
|
|
28
|
+
//!
|
|
29
|
+
//! Each side is generated on the fly, batch by batch, and nothing is retained
|
|
30
|
+
//! between batches, so the process's peak RSS is the diff's own state, not the
|
|
31
|
+
//! table data. The **linear** shape (`id`, `value` int64 columns) exercises the
|
|
32
|
+
//! per-row hash vectors: the left is ids `0..n`, the right `step..n + step` with
|
|
33
|
+
//! `step = n / 100`, so 1% removed, 1% added, ~2% changed — and the cell pass
|
|
34
|
+
//! materializes those ~2% changed rows on both sides. The **nochange** variant
|
|
35
|
+
//! keeps the 1% added/removed but makes every shared row equal, so the cell pass
|
|
36
|
+
//! materializes nothing: the difference in peak RSS between it and the default
|
|
37
|
+
//! run is the cell pass's cost. The **allchange** variant drops the offset and
|
|
38
|
+
//! changes every shared row (no added/removed), so the cell pass materializes
|
|
39
|
+
//! and renders every row. The **wide** shape (`id` int64, `value` a
|
|
40
|
+
//! `value_width`-byte Utf8 that differs between the sides) changes every row too
|
|
41
|
+
//! and renders `value_width` bytes per changed cell — the rendering worst case,
|
|
42
|
+
//! whose peak RSS scales with changed cells times cell width. The **dup** shape
|
|
43
|
+
//! (`key` Utf8 of the given width, `value` int64) makes every key appear twice
|
|
44
|
+
//! on each side, so every distinct
|
|
45
|
+
//! key is a duplicate and the whole `duplicate_keys` report is materialized —
|
|
46
|
+
//! the term that scales with distinct duplicated keys times the key width.
|
|
47
|
+
|
|
48
|
+
use onix_arrow::{TableDiffOptions, diff_tables};
|
|
49
|
+
|
|
50
|
+
#[path = "shared/gen_shapes.rs"]
|
|
51
|
+
mod gen_shapes;
|
|
52
|
+
use gen_shapes::{Case, Generated, batch_rows};
|
|
53
|
+
|
|
54
|
+
/// Options for the diff, honoring a `ROW_DIFF_THREADS` override so the parallel
|
|
55
|
+
/// path's peak RSS can be compared against the single-threaded baseline; unset
|
|
56
|
+
/// uses the default (available parallelism).
|
|
57
|
+
fn options_from_env(key: &str) -> TableDiffOptions {
|
|
58
|
+
let mut options = TableDiffOptions::new(vec![key.to_string()]);
|
|
59
|
+
if let Some(threads) = std::env::var("ROW_DIFF_THREADS")
|
|
60
|
+
.ok()
|
|
61
|
+
.and_then(|v| v.parse().ok())
|
|
62
|
+
.and_then(std::num::NonZeroUsize::new)
|
|
63
|
+
{
|
|
64
|
+
options = options
|
|
65
|
+
.with_threads(threads)
|
|
66
|
+
.expect("ROW_DIFF_THREADS within MAX_THREADS");
|
|
67
|
+
}
|
|
68
|
+
options
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
fn main() {
|
|
72
|
+
let args: Vec<String> = std::env::args().collect();
|
|
73
|
+
let rows: i64 = args
|
|
74
|
+
.get(1)
|
|
75
|
+
.and_then(|a| a.parse().ok())
|
|
76
|
+
.unwrap_or(1_000_000);
|
|
77
|
+
let mode = args.get(2).map_or("", String::as_str);
|
|
78
|
+
let width: usize = args
|
|
79
|
+
.get(3)
|
|
80
|
+
.and_then(|a| a.parse().ok())
|
|
81
|
+
.unwrap_or(if mode == "wide" { 1024 } else { 16 });
|
|
82
|
+
|
|
83
|
+
let (case, label) = match mode {
|
|
84
|
+
"" | "linear" => (Case::Linear, String::new()),
|
|
85
|
+
"nochange" => (Case::NoChange, " (nochange baseline)".to_string()),
|
|
86
|
+
"allchange" => (Case::AllChange, " (all changed)".to_string()),
|
|
87
|
+
"wide" => (Case::Wide(width), format!(" (wide, value_width={width})")),
|
|
88
|
+
"widesame" => (
|
|
89
|
+
Case::WideSame(width),
|
|
90
|
+
format!(" (widesame, value_width={width})"),
|
|
91
|
+
),
|
|
92
|
+
"manycols" => {
|
|
93
|
+
let ncols: usize = args.get(3).and_then(|a| a.parse().ok()).unwrap_or(8);
|
|
94
|
+
let width: usize = args.get(4).and_then(|a| a.parse().ok()).unwrap_or(512);
|
|
95
|
+
(
|
|
96
|
+
Case::ManyCols { ncols, width },
|
|
97
|
+
format!(" (manycols, ncols={ncols}, width={width})"),
|
|
98
|
+
)
|
|
99
|
+
}
|
|
100
|
+
"dup" => (Case::Dup(width), format!(" (dup, key_width={width})")),
|
|
101
|
+
other => {
|
|
102
|
+
eprintln!(
|
|
103
|
+
"unknown mode {other:?}; expected linear (the default), nochange, allchange, wide, widesame, manycols or dup"
|
|
104
|
+
);
|
|
105
|
+
std::process::exit(2);
|
|
106
|
+
}
|
|
107
|
+
};
|
|
108
|
+
let (schema, left_shape, right_shape, key) = case.build(rows);
|
|
109
|
+
|
|
110
|
+
let batch = batch_rows();
|
|
111
|
+
let left = Generated {
|
|
112
|
+
schema: schema.clone(),
|
|
113
|
+
rows,
|
|
114
|
+
shape: left_shape,
|
|
115
|
+
batch,
|
|
116
|
+
};
|
|
117
|
+
let right = Generated {
|
|
118
|
+
schema,
|
|
119
|
+
rows,
|
|
120
|
+
shape: right_shape,
|
|
121
|
+
batch,
|
|
122
|
+
};
|
|
123
|
+
|
|
124
|
+
let options = options_from_env(key);
|
|
125
|
+
let start = std::time::Instant::now();
|
|
126
|
+
let diff = diff_tables(&left, &right, &options).expect("diff succeeds");
|
|
127
|
+
let elapsed = start.elapsed();
|
|
128
|
+
let summary = diff.summary();
|
|
129
|
+
|
|
130
|
+
println!("rows per side: {rows}{label}");
|
|
131
|
+
println!("threads: {}", options.threads());
|
|
132
|
+
println!("wall: {:.2}s", elapsed.as_secs_f64());
|
|
133
|
+
println!(
|
|
134
|
+
"rows_added={} rows_removed={} rows_changed={} duplicate_keys={} cells_changed={}",
|
|
135
|
+
summary.rows_added,
|
|
136
|
+
summary.rows_removed,
|
|
137
|
+
summary.rows_changed,
|
|
138
|
+
summary.duplicate_keys,
|
|
139
|
+
summary.cells_changed
|
|
140
|
+
);
|
|
141
|
+
}
|