fpstreams 2.0.0__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fpstreams-2.2.0/CHANGELOG.md +323 -0
- fpstreams-2.2.0/PKG-INFO +467 -0
- fpstreams-2.2.0/README.md +411 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/pyproject.toml +50 -9
- {fpstreams-2.0.0 → fpstreams-2.2.0}/rust/Cargo.lock +12 -12
- {fpstreams-2.0.0 → fpstreams-2.2.0}/rust/Cargo.toml +8 -2
- fpstreams-2.2.0/rust/build.rs +3 -0
- fpstreams-2.2.0/rust/src/common.rs +706 -0
- fpstreams-2.2.0/rust/src/float/affine_pair.rs +192 -0
- fpstreams-2.2.0/rust/src/float/endpoints.rs +594 -0
- fpstreams-2.2.0/rust/src/float.rs +972 -0
- fpstreams-2.2.0/rust/src/integer/endpoints.rs +972 -0
- fpstreams-2.2.0/rust/src/integer.rs +892 -0
- fpstreams-2.2.0/rust/src/lib.rs +245 -0
- fpstreams-2.2.0/rust/src/numeric_mean.rs +653 -0
- fpstreams-2.2.0/rust/src/numpy_export.rs +41 -0
- fpstreams-2.2.0/rust/src/numpy_group/buffer.rs +519 -0
- fpstreams-2.2.0/rust/src/numpy_group.rs +849 -0
- fpstreams-2.2.0/rust/src/pair/expr.rs +137 -0
- fpstreams-2.2.0/rust/src/pair/prefix.rs +131 -0
- fpstreams-2.2.0/rust/src/pair/row_filter.rs +82 -0
- fpstreams-2.2.0/rust/src/pair/unique.rs +212 -0
- fpstreams-2.2.0/rust/src/pair/value_filter.rs +149 -0
- fpstreams-2.2.0/rust/src/pair/value_map.rs +184 -0
- fpstreams-2.2.0/rust/src/pair.rs +21 -0
- fpstreams-2.2.0/rust/src/pivot.rs +710 -0
- fpstreams-2.2.0/rust/src/records.rs +191 -0
- fpstreams-2.2.0/rust/src/relational/adapters/namedtuple.rs +1075 -0
- fpstreams-2.2.0/rust/src/relational/adapters.rs +314 -0
- fpstreams-2.2.0/rust/src/relational/global_numeric.rs +133 -0
- fpstreams-2.2.0/rust/src/relational/group_numeric.rs +1067 -0
- fpstreams-2.2.0/rust/src/relational/group_pair_expr.rs +316 -0
- fpstreams-2.2.0/rust/src/relational/join_callable/many.rs +428 -0
- fpstreams-2.2.0/rust/src/relational/join_callable/unique.rs +352 -0
- fpstreams-2.2.0/rust/src/relational/join_callable.rs +1041 -0
- fpstreams-2.2.0/rust/src/relational/join_exact.rs +696 -0
- fpstreams-2.2.0/rust/src/relational/join_i64.rs +585 -0
- fpstreams-2.2.0/rust/src/relational.rs +222 -0
- fpstreams-2.2.0/rust/src/relational_fixed/composite.rs +186 -0
- fpstreams-2.2.0/rust/src/relational_fixed/global_multi/same_field.rs +469 -0
- fpstreams-2.2.0/rust/src/relational_fixed/global_multi.rs +1296 -0
- fpstreams-2.2.0/rust/src/relational_fixed/group_multi.rs +1026 -0
- fpstreams-2.2.0/rust/src/relational_fixed/single.rs +565 -0
- fpstreams-2.2.0/rust/src/relational_fixed.rs +675 -0
- fpstreams-2.2.0/rust/src/scalar_sort.rs +443 -0
- fpstreams-2.2.0/rust/src/scalar_unique.rs +422 -0
- fpstreams-2.2.0/rust/src/select.rs +643 -0
- fpstreams-2.2.0/rust/src/tests/adapters.rs +484 -0
- fpstreams-2.2.0/rust/src/tests/group_exact/pairs.rs +574 -0
- fpstreams-2.2.0/rust/src/tests/group_exact/rows.rs +708 -0
- fpstreams-2.2.0/rust/src/tests/group_exact.rs +101 -0
- fpstreams-2.2.0/rust/src/tests/group_fixed.rs +808 -0
- fpstreams-2.2.0/rust/src/tests/join_callable/behavior.rs +836 -0
- fpstreams-2.2.0/rust/src/tests/join_callable/direct.rs +236 -0
- fpstreams-2.2.0/rust/src/tests/join_callable.rs +986 -0
- fpstreams-2.2.0/rust/src/tests/join_exact.rs +916 -0
- fpstreams-2.2.0/rust/src/tests/numeric/buffers.rs +691 -0
- fpstreams-2.2.0/rust/src/tests/numeric/kernels.rs +1036 -0
- fpstreams-2.2.0/rust/src/tests/numeric/reductions.rs +436 -0
- fpstreams-2.2.0/rust/src/tests/numeric.rs +485 -0
- fpstreams-2.2.0/rust/src/tests/numpy_group.rs +718 -0
- fpstreams-2.2.0/rust/src/tests.rs +459 -0
- fpstreams-2.2.0/rust/src/unnest.rs +236 -0
- fpstreams-2.2.0/rust/src/unpivot.rs +421 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/__init__.py +30 -4
- fpstreams-2.2.0/src/fpstreams/_native.pyi +707 -0
- fpstreams-2.2.0/src/fpstreams/_provenance.py +200 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/aggregate.py +1 -1
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/async_flow.py +1 -1
- fpstreams-2.2.0/src/fpstreams/collecting/__init__.py +31 -0
- fpstreams-2.2.0/src/fpstreams/collecting/_collector_base.py +128 -0
- fpstreams-2.2.0/src/fpstreams/collecting/aggregate_program.py +156 -0
- fpstreams-2.2.0/src/fpstreams/collecting/aggregation.py +1201 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/collecting/collector.py +241 -149
- fpstreams-2.2.0/src/fpstreams/collecting/program.py +288 -0
- fpstreams-2.2.0/src/fpstreams/collecting/reducer.py +245 -0
- fpstreams-2.2.0/src/fpstreams/collecting/statistics.py +194 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/collectors.py +1 -1
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/column.py +1 -1
- fpstreams-2.2.0/src/fpstreams/errors.py +33 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/exceptions.py +4 -0
- fpstreams-2.2.0/src/fpstreams/execution/__init__.py +320 -0
- fpstreams-2.2.0/src/fpstreams/execution/_pair_aggregate.py +68 -0
- fpstreams-2.2.0/src/fpstreams/execution/_pair_dict.py +1363 -0
- fpstreams-2.2.0/src/fpstreams/execution/_pair_row_filter.py +321 -0
- fpstreams-2.2.0/src/fpstreams/execution/_rows_fusion.py +975 -0
- fpstreams-2.2.0/src/fpstreams/execution/_scalar_fusion.py +251 -0
- fpstreams-2.2.0/src/fpstreams/execution/arrow.py +1778 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/execution/async_iterators.py +165 -143
- fpstreams-2.2.0/src/fpstreams/execution/async_map.py +92 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_merge.py +386 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_ops.py +93 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_prefetch.py +105 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_queue.py +94 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_scheduler.py +186 -0
- fpstreams-2.2.0/src/fpstreams/execution/async_timers.py +287 -0
- fpstreams-2.2.0/src/fpstreams/execution/native.py +559 -0
- fpstreams-2.2.0/src/fpstreams/execution/numpy_group.py +1359 -0
- fpstreams-2.2.0/src/fpstreams/execution/numpy_prefix.py +703 -0
- fpstreams-2.2.0/src/fpstreams/execution/physical.py +346 -0
- fpstreams-2.2.0/src/fpstreams/execution/relational/__init__.py +2110 -0
- fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_global.py +982 -0
- fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_group.py +1001 -0
- fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_group_rows.py +183 -0
- fpstreams-2.2.0/src/fpstreams/execution/relational/join.py +550 -0
- fpstreams-2.2.0/src/fpstreams/execution/sorted_streams.py +407 -0
- fpstreams-2.2.0/src/fpstreams/execution/sorting.py +760 -0
- fpstreams-2.2.0/src/fpstreams/execution/sync.py +417 -0
- fpstreams-2.2.0/src/fpstreams/execution/sync_ops.py +577 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/expr.py +1 -1
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/expressions/__init__.py +1 -1
- fpstreams-2.2.0/src/fpstreams/expressions/_codegen.py +29 -0
- fpstreams-2.2.0/src/fpstreams/expressions/_row_codegen.py +246 -0
- fpstreams-2.2.0/src/fpstreams/expressions/program.py +194 -0
- fpstreams-2.2.0/src/fpstreams/expressions/row.py +363 -0
- fpstreams-2.2.0/src/fpstreams/expressions/row_eval.py +441 -0
- fpstreams-2.2.0/src/fpstreams/expressions/row_ir.py +269 -0
- fpstreams-2.2.0/src/fpstreams/expressions/scalar.py +787 -0
- fpstreams-2.2.0/src/fpstreams/expressions/selectors.py +124 -0
- fpstreams-2.2.0/src/fpstreams/expressions/typed_ir.py +111 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/functional.py +32 -19
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/gather.py +1 -1
- fpstreams-2.2.0/src/fpstreams/io_safety.py +97 -0
- fpstreams-2.2.0/src/fpstreams/option.py +5 -0
- fpstreams-2.2.0/src/fpstreams/physical/__init__.py +13 -0
- fpstreams-2.2.0/src/fpstreams/physical/async_plan.py +170 -0
- fpstreams-2.2.0/src/fpstreams/physical/compiled.py +201 -0
- fpstreams-2.2.0/src/fpstreams/physical/kernel_cache.py +45 -0
- fpstreams-2.2.0/src/fpstreams/physical/plan.py +94 -0
- fpstreams-2.2.0/src/fpstreams/physical/relational.py +377 -0
- fpstreams-2.2.0/src/fpstreams/planning/__init__.py +1 -0
- fpstreams-2.2.0/src/fpstreams/planning/_pair_stages.py +55 -0
- fpstreams-2.2.0/src/fpstreams/planning/arrow.py +424 -0
- fpstreams-2.2.0/src/fpstreams/planning/arrow_source.py +70 -0
- fpstreams-2.2.0/src/fpstreams/planning/async_.py +501 -0
- fpstreams-2.2.0/src/fpstreams/planning/async_utils.py +79 -0
- fpstreams-2.2.0/src/fpstreams/planning/compiler.py +1684 -0
- fpstreams-2.2.0/src/fpstreams/planning/explain.py +454 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/planning/gather.py +52 -48
- fpstreams-2.2.0/src/fpstreams/planning/logical.py +234 -0
- fpstreams-2.2.0/src/fpstreams/planning/native.py +1047 -0
- fpstreams-2.2.0/src/fpstreams/planning/numpy.py +490 -0
- fpstreams-2.2.0/src/fpstreams/planning/pair_i64_expression.py +561 -0
- fpstreams-2.2.0/src/fpstreams/planning/plan_cache.py +64 -0
- fpstreams-2.2.0/src/fpstreams/planning/semantic_analyzer.py +168 -0
- fpstreams-2.2.0/src/fpstreams/planning/semantic_rules.py +744 -0
- fpstreams-2.2.0/src/fpstreams/planning/semantics.py +317 -0
- fpstreams-2.2.0/src/fpstreams/planning/source.py +378 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/planning/sync.py +69 -23
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/primitives/__init__.py +1 -1
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/primitives/option.py +48 -35
- fpstreams-2.2.0/src/fpstreams/primitives/result.py +327 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/result.py +1 -1
- fpstreams-2.2.0/src/fpstreams/runtime/__init__.py +25 -0
- fpstreams-2.2.0/src/fpstreams/runtime/_distinct.py +35 -0
- fpstreams-2.2.0/src/fpstreams/runtime/failpoints.py +42 -0
- fpstreams-2.2.0/src/fpstreams/runtime/files.py +147 -0
- fpstreams-2.2.0/src/fpstreams/runtime/iterators.py +58 -0
- fpstreams-2.2.0/src/fpstreams/runtime/limits.py +19 -0
- fpstreams-2.2.0/src/fpstreams/runtime/metrics.py +16 -0
- fpstreams-2.2.0/src/fpstreams/runtime/query.py +98 -0
- fpstreams-2.2.0/src/fpstreams/runtime/report.py +262 -0
- fpstreams-2.2.0/src/fpstreams/runtime/resources.py +253 -0
- fpstreams-2.2.0/src/fpstreams/runtime/spill.py +47 -0
- fpstreams-2.2.0/src/fpstreams/runtime/tasks.py +338 -0
- fpstreams-2.2.0/src/fpstreams/storage/__init__.py +14 -0
- fpstreams-2.2.0/src/fpstreams/storage/codec.py +112 -0
- fpstreams-2.2.0/src/fpstreams/storage/spill_store.py +295 -0
- fpstreams-2.2.0/src/fpstreams/streams/_async_identity.py +102 -0
- fpstreams-2.2.0/src/fpstreams/streams/_flow_structural_list.py +1176 -0
- fpstreams-2.2.0/src/fpstreams/streams/_flow_unique_list.py +551 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/async_flow.py +421 -169
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/async_terminals.py +312 -103
- fpstreams-2.2.0/src/fpstreams/streams/flow.py +1619 -0
- fpstreams-2.2.0/src/fpstreams/streams/flow_terminals.py +2917 -0
- fpstreams-2.2.0/src/fpstreams/streams/pairs.py +507 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/__init__.py +2 -1
- fpstreams-2.2.0/src/fpstreams/tabular/_text_sources.py +447 -0
- fpstreams-2.2.0/src/fpstreams/tabular/arrow.py +1035 -0
- fpstreams-2.2.0/src/fpstreams/tabular/dataframe.py +109 -0
- fpstreams-2.2.0/src/fpstreams/tabular/factory.py +333 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/grouped.py +37 -44
- fpstreams-2.2.0/src/fpstreams/tabular/io.py +759 -0
- fpstreams-2.2.0/src/fpstreams/tabular/join.py +944 -0
- fpstreams-2.2.0/src/fpstreams/tabular/numpy.py +687 -0
- fpstreams-2.2.0/src/fpstreams/tabular/polars.py +134 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/records.py +25 -1
- fpstreams-2.2.0/src/fpstreams/tabular/rows.py +2883 -0
- fpstreams-2.2.0/src/fpstreams/tabular/spill.py +1099 -0
- fpstreams-2.2.0/src/fpstreams/tabular/spill_io.py +469 -0
- fpstreams-2.2.0/src/fpstreams/tabular/spill_limits.py +121 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/sql.py +64 -25
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/sqlite_sink.py +30 -13
- fpstreams-2.2.0/src/fpstreams/testing.py +110 -0
- fpstreams-2.0.0/PKG-INFO +0 -294
- fpstreams-2.0.0/README.md +0 -249
- fpstreams-2.0.0/rust/src/common.rs +0 -64
- fpstreams-2.0.0/rust/src/float.rs +0 -519
- fpstreams-2.0.0/rust/src/integer.rs +0 -511
- fpstreams-2.0.0/rust/src/lib.rs +0 -47
- fpstreams-2.0.0/rust/src/tests.rs +0 -225
- fpstreams-2.0.0/src/fpstreams/_native.pyi +0 -124
- fpstreams-2.0.0/src/fpstreams/collecting/__init__.py +0 -13
- fpstreams-2.0.0/src/fpstreams/collecting/aggregation.py +0 -418
- fpstreams-2.0.0/src/fpstreams/collecting/statistics.py +0 -68
- fpstreams-2.0.0/src/fpstreams/errors.py +0 -29
- fpstreams-2.0.0/src/fpstreams/execution/__init__.py +0 -132
- fpstreams-2.0.0/src/fpstreams/execution/async_.py +0 -64
- fpstreams-2.0.0/src/fpstreams/execution/async_concurrency.py +0 -459
- fpstreams-2.0.0/src/fpstreams/execution/async_ops.py +0 -188
- fpstreams-2.0.0/src/fpstreams/execution/native.py +0 -93
- fpstreams-2.0.0/src/fpstreams/execution/sorting.py +0 -121
- fpstreams-2.0.0/src/fpstreams/execution/sync.py +0 -75
- fpstreams-2.0.0/src/fpstreams/execution/sync_ops.py +0 -467
- fpstreams-2.0.0/src/fpstreams/expressions/row.py +0 -337
- fpstreams-2.0.0/src/fpstreams/expressions/scalar.py +0 -377
- fpstreams-2.0.0/src/fpstreams/expressions/selectors.py +0 -42
- fpstreams-2.0.0/src/fpstreams/option.py +0 -5
- fpstreams-2.0.0/src/fpstreams/planning/__init__.py +0 -1
- fpstreams-2.0.0/src/fpstreams/planning/async_.py +0 -316
- fpstreams-2.0.0/src/fpstreams/planning/async_utils.py +0 -43
- fpstreams-2.0.0/src/fpstreams/planning/explain.py +0 -84
- fpstreams-2.0.0/src/fpstreams/planning/native.py +0 -334
- fpstreams-2.0.0/src/fpstreams/planning/source.py +0 -77
- fpstreams-2.0.0/src/fpstreams/primitives/result.py +0 -314
- fpstreams-2.0.0/src/fpstreams/streams/flow.py +0 -1112
- fpstreams-2.0.0/src/fpstreams/streams/flow_terminals.py +0 -864
- fpstreams-2.0.0/src/fpstreams/streams/pairs.py +0 -343
- fpstreams-2.0.0/src/fpstreams/tabular/arrow.py +0 -316
- fpstreams-2.0.0/src/fpstreams/tabular/dataframe.py +0 -54
- fpstreams-2.0.0/src/fpstreams/tabular/factory.py +0 -228
- fpstreams-2.0.0/src/fpstreams/tabular/io.py +0 -332
- fpstreams-2.0.0/src/fpstreams/tabular/join.py +0 -479
- fpstreams-2.0.0/src/fpstreams/tabular/polars.py +0 -71
- fpstreams-2.0.0/src/fpstreams/tabular/rows.py +0 -940
- fpstreams-2.0.0/src/fpstreams/tabular/spill.py +0 -370
- {fpstreams-2.0.0 → fpstreams-2.2.0}/LICENSE +0 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/py.typed +0 -0
- {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/__init__.py +0 -0
|
@@ -0,0 +1,323 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
This file records user-visible and compatibility-relevant changes in fpstreams 2,
|
|
4
|
+
including changed defaults.
|
|
5
|
+
|
|
6
|
+
## Unreleased
|
|
7
|
+
|
|
8
|
+
## 2.2.0 - 2026-10-06
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Add bounded `Rows.join_sorted()` for explicitly ordered records, with consumed-prefix
|
|
13
|
+
validation and first-right-record schema rules.
|
|
14
|
+
- Report explicit sorted group, merge, and join execution through existing report fields.
|
|
15
|
+
- Add optional atomic path output to Flow CSV/JSON/JSONL and Rows CSV/JSONL sinks,
|
|
16
|
+
including no-overwrite publication. Existing direct output remains the default.
|
|
17
|
+
- `Flow.to_jsonl()` streams arbitrary JSON values as one value per line.
|
|
18
|
+
`Rows.to_jsonl()` now accepts `default` for custom serialization.
|
|
19
|
+
|
|
20
|
+
- `Flow.merge_sorted()` stably merges two ascending inputs without sorting or
|
|
21
|
+
materializing them. Ties prefer the left input and outputs retain their identity.
|
|
22
|
+
|
|
23
|
+
- `Rows.group_by_sorted()` aggregates adjacent ascending keys in Python with
|
|
24
|
+
current group state and one lookahead row. Consumed keys are checked for exact
|
|
25
|
+
builtin types and order; this explicit mode does not sort or support spill.
|
|
26
|
+
|
|
27
|
+
- `Pairs.run_with_report()` executes a pair terminal once and returns its value
|
|
28
|
+
with an execution report. It supports `to_dict`, `group_values`,
|
|
29
|
+
`collect_values`, and `aggregate_values`.
|
|
30
|
+
|
|
31
|
+
### Fixed
|
|
32
|
+
|
|
33
|
+
- Rust source builds support the declared 1.85 minimum. Native kernels no longer
|
|
34
|
+
require Rust 1.88 through let-chain syntax; CI checks the minimum compiler.
|
|
35
|
+
- SQLite sinks match existing table, view, and column names using SQLite's ASCII
|
|
36
|
+
case rules. Duplicate column aliases raise `DuplicateKeyError` before inserts.
|
|
37
|
+
Fail mode and view rejection check the destination before reading rows.
|
|
38
|
+
- Sync and async flows accept built-in ranges longer than `sys.maxsize`. Bounded
|
|
39
|
+
reads and exact counts no longer fail during source size detection.
|
|
40
|
+
|
|
41
|
+
- `spreadsheet_safe=True` also neutralizes formula-like CSV headers, including
|
|
42
|
+
inferred record field names and explicitly named empty output. Record lookup
|
|
43
|
+
still uses the original names; the raw-output default is unchanged.
|
|
44
|
+
- Relative Parquet output paths stay anchored to the initial working directory
|
|
45
|
+
if a source or conversion callback changes directories during the write.
|
|
46
|
+
|
|
47
|
+
- Scalar keys in explicit sorted operations avoid temporary shape tuples while
|
|
48
|
+
retaining per-row type and ordering checks.
|
|
49
|
+
|
|
50
|
+
- Atomic CSV and JSON output keeps its original destination when a source or
|
|
51
|
+
serializer changes the working directory during execution.
|
|
52
|
+
|
|
53
|
+
- CSV and SQLite record sinks propagate `StopIteration` from first-record
|
|
54
|
+
conversion instead of treating it as empty input. SQLite replacement leaves
|
|
55
|
+
the existing table intact when that conversion fails.
|
|
56
|
+
|
|
57
|
+
- Sync and async `chunk`, `window`, and `batch_by_size` validate integer bounds
|
|
58
|
+
before execution, including their aliases. Floats, NaN, and infinity now raise
|
|
59
|
+
`TypeError` instead of failing after source consumption or leaving async batches
|
|
60
|
+
unbounded. Objects implementing `__index__` are normalized once per bound.
|
|
61
|
+
|
|
62
|
+
- `unique()`, `unique_by()`, and `Pairs.unique_keys()` propagate equality errors
|
|
63
|
+
instead of treating them as unhashable keys. Hash callbacks keep their existing
|
|
64
|
+
lookup and insertion counts.
|
|
65
|
+
- Async uniqueness, `agg.count_distinct()`, and native pair-uniqueness continuation
|
|
66
|
+
also propagate equality errors without extra hash calls. Iterator cleanup retains
|
|
67
|
+
nested exception notes when several owned resources fail to close.
|
|
68
|
+
- Distinct operations avoid key-wrapper allocations for exact built-in strings.
|
|
69
|
+
String subclasses and custom keys retain guarded hash and equality handling.
|
|
70
|
+
- Grouped reductions and frequency counts propagate `KeyError` raised by a key's
|
|
71
|
+
hash or equality method instead of treating it as a missing group. This includes
|
|
72
|
+
async reductions, grouping collectors, Pairs, and spilled grouping.
|
|
73
|
+
- Parquet `if_exists="error"` publishes with an atomic no-overwrite hard link.
|
|
74
|
+
A concurrent creator or dangling symlink cannot be overwritten. Filesystems
|
|
75
|
+
without hard-link support report an error; `replace` still uses atomic rename.
|
|
76
|
+
- Owned Arrow resources report close failures on successful queries and attach
|
|
77
|
+
cleanup diagnostics to an existing query error. Cleanup attempts every owned
|
|
78
|
+
resource. This also changes Arrow `first()`, which previously ignored close errors.
|
|
79
|
+
|
|
80
|
+
- Engine benchmarks calibrate warmed timing blocks for short tasks and first-row
|
|
81
|
+
latency, retaining elapsed time and call counts in their reports. Rebuild
|
|
82
|
+
single-call baselines before comparing. Timing and resource regression limits
|
|
83
|
+
are unchanged; scheduled CI now uploads the three raw reference reports.
|
|
84
|
+
|
|
85
|
+
- Benchmark report schema 6 records Python and glibc allocator environment
|
|
86
|
+
settings. Baseline creation and comparison reject missing or mismatched
|
|
87
|
+
settings; regenerate older reports. The runners do not change allocator defaults.
|
|
88
|
+
|
|
89
|
+
- NumPy frequency benchmarks check key types, integer counts, first-key order,
|
|
90
|
+
and floating-point bit patterns. Separate NaN entries, their signs, and their
|
|
91
|
+
payloads remain distinct when comparing independently computed outputs.
|
|
92
|
+
- Generated Python `with_columns` loops call the live enrichment function.
|
|
93
|
+
Changes to captured accessors, expression evaluators, or the selector list
|
|
94
|
+
remain visible after plan caching and while reading the source. The transform
|
|
95
|
+
still copies the row before invoking selectors against the original input.
|
|
96
|
+
|
|
97
|
+
- Arrow file projections keep selector changes made by custom scan openers
|
|
98
|
+
visible. Parquet rechecks the projection after creating its dataset, including
|
|
99
|
+
with an explicit source filter. CSV keeps full fields when its public reader
|
|
100
|
+
hooks differ from the extension entrypoints. Normal CSV projection retains
|
|
101
|
+
the bounded schema probe and the default data reader.
|
|
102
|
+
- `Rows.select()` rechecks captured field accessors before using Rust, NumPy,
|
|
103
|
+
or Arrow projection metadata. The generated Python loop calls the live
|
|
104
|
+
projection, so changes made during input reads remain visible. Custom Arrow
|
|
105
|
+
batch sources keep fields available for fallback; changed projections can
|
|
106
|
+
infer their output dtype instead of forcing the former field's schema.
|
|
107
|
+
NumPy fallback finishes the already-opened source without reopening it.
|
|
108
|
+
- Pivot calls the live index, column, and value selectors on its Python path.
|
|
109
|
+
Direct dict lookups had ignored changes to selector code, closure bindings,
|
|
110
|
+
and globals. Native admission now checks those bindings before execution;
|
|
111
|
+
its fallback shares the general selector loop.
|
|
112
|
+
- Native pivot falls back when Rows/Flow iteration or source hooks change,
|
|
113
|
+
including changes to function code. Replacement results, exceptions, and
|
|
114
|
+
iterator cleanup remain observable.
|
|
115
|
+
- `frequencies(key=...)` uses the Python pipeline for sequential `auto` plans,
|
|
116
|
+
so key callbacks can affect subsequent input reads. Native materialization
|
|
117
|
+
could hide those changes. Reports identify this route as `python_frequency`.
|
|
118
|
+
- Benchmark speedup requirements for composite count/sum groups and NamedTuple
|
|
119
|
+
callable joins now apply from 1,000 rows. Smaller inputs still emit timing and
|
|
120
|
+
allocation data for cross-run checks. The original ratios and CI workloads
|
|
121
|
+
are unchanged; small runs no longer inherit speedup requirements validated
|
|
122
|
+
for larger inputs.
|
|
123
|
+
- The engine benchmark CLI lists scenarios without running their tasks, timing,
|
|
124
|
+
allocation tracking, or execution observation. Listing and measurement share
|
|
125
|
+
filtering and fixture cleanup, including when a later scenario builder fails.
|
|
126
|
+
- Single-collector grouping preserves the general collector program's state
|
|
127
|
+
release order, including when output is closed early. Unused input keys are
|
|
128
|
+
released before reading the current finisher, so changes made by their release
|
|
129
|
+
callbacks take effect.
|
|
130
|
+
- Single-collector grouping uses the current step after truth-testing a custom
|
|
131
|
+
completion result. It also releases replaced completion values before pulling
|
|
132
|
+
another row, matching the general collector program's callback behavior.
|
|
133
|
+
- Python grouping calls the live key selector for each row. A field shortcut
|
|
134
|
+
could keep using an old field after a source or callback changed the selector's
|
|
135
|
+
code or closure, merging distinct groups. Selector globals and error types
|
|
136
|
+
now follow the same function calls as general grouped aggregation.
|
|
137
|
+
- Scalar caches no longer confuse numerically equal constants of different
|
|
138
|
+
types. Directly constructed expressions retain their Python result types,
|
|
139
|
+
wide integer values, and custom constant identities. Expressions with
|
|
140
|
+
nonstandard operand types use Python in `auto` mode; forcing `native` raises
|
|
141
|
+
`NativeUnsupportedError` instead of changing their numeric representation.
|
|
142
|
+
- Float expressions retain the sign of `-0.0` in their instructions and evaluator
|
|
143
|
+
cache. Compiling a positive-zero expression first no longer changes a later
|
|
144
|
+
negative-zero result. Scalar and Pairs execution keep their existing routes.
|
|
145
|
+
- Python grouped sums call the live collector lifecycle throughout traversal
|
|
146
|
+
and output. Changes to function code or selector closure cells made by a source,
|
|
147
|
+
key callback, or output consumer are no longer hidden by an inlined sum loop.
|
|
148
|
+
- Two-key count/sum groups also use the live collector program. Their separate
|
|
149
|
+
loop skipped changes to initializer and step code or the sum selector closure
|
|
150
|
+
made while reading the source, even though the same collectors worked correctly
|
|
151
|
+
in other group layouts.
|
|
152
|
+
- Grouped aggregation uses the general collector program for custom lifecycle
|
|
153
|
+
getters. A dynamic `step` property or `__getattribute__` override is read for
|
|
154
|
+
each step instead of being cached as a fixed function.
|
|
155
|
+
- Single-collector grouping no longer caches a temporary lifecycle hook between
|
|
156
|
+
key selection and lookup. If hashing replaces that hook, its release callback
|
|
157
|
+
runs before the next hash call, matching the general collector program.
|
|
158
|
+
- Exact builtin checks no longer use custom metaclass equality to infer source
|
|
159
|
+
size or select grouping, spill, range lookup, and float-expression shortcuts.
|
|
160
|
+
Custom iterables are counted by traversal; map/filter fusion does not request
|
|
161
|
+
their length hints. Grouping retains custom key and serialization calls.
|
|
162
|
+
- Execution reports distinguish successful top-level Rust and Arrow record
|
|
163
|
+
joins from Python joins. Native pair aggregation also records its direct route.
|
|
164
|
+
- Benchmark comparisons reject missing provenance and mismatched workloads.
|
|
165
|
+
Both suites record dependencies, Git state, code and workload fingerprints,
|
|
166
|
+
and an untimed observation of the task. Median baselines retain each run's
|
|
167
|
+
provenance. Older reports need to be regenerated for report schema 6.
|
|
168
|
+
- Reports include CPU affinity, NumPy CPU dispatch, and an allowlist of runtime
|
|
169
|
+
settings. Comparisons reject mismatched configurations; both runners also
|
|
170
|
+
reject configuration changes during measurement.
|
|
171
|
+
- Join benchmarks honor the requested engine. Twelve Python controls separate
|
|
172
|
+
dict and Mapping records, field and callable keys, and inner, left, and repeated
|
|
173
|
+
matches. References preserve callable-key columns and snapshots taken before
|
|
174
|
+
key selection, including unmatched left rows.
|
|
175
|
+
- NumPy frequency benchmarks cover integer and float arrays, with and without
|
|
176
|
+
a callable key, at low and high cardinality. All references preserve first-key
|
|
177
|
+
order; Python array-to-list conversion is included in its timed task.
|
|
178
|
+
- Competitive samples warm each task for at least 1ms after GC and record the
|
|
179
|
+
number of warmup calls. One warmup call left inconsistent timings for short
|
|
180
|
+
NumPy tasks in local measurements.
|
|
181
|
+
- Competitive benchmarks now record peak Python allocation in a separate,
|
|
182
|
+
untimed call for every implementation, using the engine suite's shared helper.
|
|
183
|
+
Schema 6 rejects missing or invalid resource measurements, and baseline
|
|
184
|
+
creation rejects inconsistent resource sets instead of substituting zero.
|
|
185
|
+
Earlier competitive reports contain no allocation evidence.
|
|
186
|
+
- Regression checks accept zero allocated bytes when both runs report zero.
|
|
187
|
+
- `frequencies()` on retained lists and tuples falls back to Python when the
|
|
188
|
+
optional Rust extension is unavailable, including in the browser wheel.
|
|
189
|
+
- Frequency benchmarks use bulk conversion for NumPy dictionaries and pandas'
|
|
190
|
+
`to_dict()`, avoiding per-key normalization overhead in the reference tasks.
|
|
191
|
+
|
|
192
|
+
### Changed
|
|
193
|
+
|
|
194
|
+
- Automatic NumPy `frequencies()` without a key can count a selected native
|
|
195
|
+
stream in Rust through its existing iterator. Source fallback and query cleanup
|
|
196
|
+
stay in the physical executor. Custom keys return to Python before hashing;
|
|
197
|
+
old wheels and free-threaded CPython retain the prior counting path.
|
|
198
|
+
- Python `Rows.select()` keeps its per-row selector snapshot as a list, avoiding
|
|
199
|
+
an intermediate tuple conversion. Selector edits still affect subsequent rows,
|
|
200
|
+
and replaced selectors stay alive until the current row finishes.
|
|
201
|
+
|
|
202
|
+
- Python scalar iteration over an exact NumPy ndarray checks its live size
|
|
203
|
+
without creating a shape tuple for every value. Dimension and length checks,
|
|
204
|
+
lazy reads, and the fallback for custom array objects remain in place.
|
|
205
|
+
- Python joins reduce layout-checking overhead for left records with more than
|
|
206
|
+
four fields. Repeated private dict snapshots need no temporary field-name tuple
|
|
207
|
+
or Python comparison generator. Field identity, short-circuiting, suffix keys,
|
|
208
|
+
snapshot lifetimes, and the bounded layout cache retain their existing behavior.
|
|
209
|
+
- Two-key count/sum groups can run in Rust with `engine="auto"` when a retained
|
|
210
|
+
list or tuple contains exact tuple rows and both keys and the selected values
|
|
211
|
+
are plain signed 64-bit integers. The path preserves first-key identities,
|
|
212
|
+
encounter order, and wider sums. Changed functions, one-shot sources, and
|
|
213
|
+
unsupported types use the Python collector program; older extensions can
|
|
214
|
+
decline the optional entry point.
|
|
215
|
+
- `frequencies()` can continue counting in Rust after its bounded integer
|
|
216
|
+
prefix. The continuation supports exact builtin integer, boolean, string,
|
|
217
|
+
bytes, float, and `None` keys in retained lists and tuples on GIL-enabled
|
|
218
|
+
CPython. It updates the output dictionary directly and hands custom keys
|
|
219
|
+
back to Python before calling their hash or equality methods.
|
|
220
|
+
- Frequency execution reports record a completed native count as `rust_direct`
|
|
221
|
+
and a count completed by Python after a native prefix as `python_frequency`.
|
|
222
|
+
|
|
223
|
+
### Internal
|
|
224
|
+
|
|
225
|
+
- Browser wheels record checkout identity and working-tree state. Release labels
|
|
226
|
+
require a clean checkout matching the version tag; dirty or unknown builds
|
|
227
|
+
remain development builds. The playground displays this provenance.
|
|
228
|
+
- Release smoke checks exercise paid-order aggregation and bounded async mapping
|
|
229
|
+
in addition to Python/native integer sums.
|
|
230
|
+
|
|
231
|
+
- Single-collector groups write step results directly to their stored entries
|
|
232
|
+
and avoid reading stored state for completed groups. Group benchmarks now
|
|
233
|
+
include `agg.first()` with repeated and distinct keys.
|
|
234
|
+
- Group benchmarks now cover custom completion predicates with repeated and
|
|
235
|
+
distinct keys, alongside collectors that cannot finish early.
|
|
236
|
+
- Generated scalar expressions, row expressions, and fused loops share a smaller
|
|
237
|
+
AST location pass. It preserves traversal order and the existing synthetic
|
|
238
|
+
source positions used in tracebacks.
|
|
239
|
+
- Scalar program fingerprints use fewer intermediate records while retaining
|
|
240
|
+
the binary framing format, arbitrary-size integers, and float payload bits.
|
|
241
|
+
- Scalar fingerprint encoding reuses fixed instruction headers instead of
|
|
242
|
+
rebuilding their bytes for every query. Expression reads and type checks are
|
|
243
|
+
unchanged; the table retains no expressions or source data.
|
|
244
|
+
- Planning benchmarks separate compilation from execution for integer and float
|
|
245
|
+
expressions and a callable control, using the same pipeline for each pair.
|
|
246
|
+
- Python grouped output uses one fewer forwarding generator. The collector
|
|
247
|
+
iterator still starts lazily and retains its existing cleanup path.
|
|
248
|
+
- Collector lifecycle properties read their original slots directly, removing
|
|
249
|
+
an extra Python wrapper. Frozen assignment, explicit replacement, and deletion
|
|
250
|
+
errors are preserved. The API manifest now classifies those existing fields
|
|
251
|
+
as properties; their names and constructor signatures are unchanged.
|
|
252
|
+
- Add Python dict-group benchmarks for field and callable selectors with repeated
|
|
253
|
+
and distinct keys. Execution observations record the case's requested engine.
|
|
254
|
+
- Include nominal Mapping and mappingproxy field-group cases with repeated and
|
|
255
|
+
distinct keys when measuring Python selector changes.
|
|
256
|
+
- Extend the two-key count/sum benchmarks to repeated keys. Keep the original
|
|
257
|
+
direct-versus-callable timing threshold and report any regression against it.
|
|
258
|
+
- Move the narrow integer-key record-join ABI adapter into the existing join
|
|
259
|
+
executor module. Kernel order, shape guards, and fallback behavior are unchanged.
|
|
260
|
+
|
|
261
|
+
### Documentation
|
|
262
|
+
|
|
263
|
+
- Repair split return descriptions and unresolved API references in generated
|
|
264
|
+
pages. The two `partition_results()` methods now describe their success and
|
|
265
|
+
exception lists separately.
|
|
266
|
+
- Correct the DB-API example's `batch_size` argument and the distinction between
|
|
267
|
+
`Rows.skip()` and `Rows.drop()` in the API index.
|
|
268
|
+
- Explain the difference between a compiled outer plan and a recorded direct
|
|
269
|
+
terminal route, including the limits of reports for compound queries.
|
|
270
|
+
- Revise the website's introductory and performance text and document the next
|
|
271
|
+
benchmark and refactoring steps in the roadmap.
|
|
272
|
+
|
|
273
|
+
## 2.1.0 - 2026-09-01
|
|
274
|
+
|
|
275
|
+
### Added
|
|
276
|
+
|
|
277
|
+
- `flow()` can enter record operations directly, while `Rows` remains available
|
|
278
|
+
as an explicit relational view. New column and NumPy factories make it possible
|
|
279
|
+
to keep columnar inputs columnar until execution.
|
|
280
|
+
- `ExecutionReport` and `run_with_report()` expose the strategy used by a terminal
|
|
281
|
+
without changing the terminal result.
|
|
282
|
+
- Standard `__arrow_c_stream__` and `__dataframe__` providers can be routed through
|
|
283
|
+
`flow()`.
|
|
284
|
+
- Arrow-backed CSV scanning supports typed incremental reads and query projection
|
|
285
|
+
under PyArrow's parsing and error contract.
|
|
286
|
+
- `AsyncFlow` now includes queue sources, bounded prefetch, session windows, numeric
|
|
287
|
+
terminals, and execution reports.
|
|
288
|
+
- `Pairs` accepts explicit engine selection and row expressions for pair filtering.
|
|
289
|
+
|
|
290
|
+
### Changed
|
|
291
|
+
|
|
292
|
+
- Retained NumPy matrices can execute guarded identity, projection, filter,
|
|
293
|
+
computed-column, aggregate, and grouped-aggregate paths without first building
|
|
294
|
+
one Python dictionary per input row.
|
|
295
|
+
- Native Rust execution now handles additional scalar, pair, reshape, join,
|
|
296
|
+
group, and global aggregation plans. Unsupported or data-semantics-sensitive
|
|
297
|
+
cases still use the Python path.
|
|
298
|
+
- Record joins and grouped aggregation use narrower shape checks and bounded native
|
|
299
|
+
kernels where they preserve Python ordering, identity, errors, and cleanup.
|
|
300
|
+
- `rows.from_csv()` and `rows.from_jsonl()` now accept caller-owned open handles
|
|
301
|
+
and replayable zero-argument opener functions in addition to paths.
|
|
302
|
+
- The benchmark runner now compares fpstreams with Python, NumPy, and pandas and
|
|
303
|
+
reports the percentage difference for each comparable case.
|
|
304
|
+
|
|
305
|
+
### Fixed
|
|
306
|
+
|
|
307
|
+
- Fast paths now revalidate cached row expressions, collector programs, NumPy
|
|
308
|
+
adapters, and implementation primitives before bypassing Python execution.
|
|
309
|
+
- One-shot sources, iterators, async tasks, database resources, spill files, and
|
|
310
|
+
retained tabular readers keep their cleanup behavior on early return and errors.
|
|
311
|
+
- Cleanup attempts every owned resource, preserves the operation error as primary,
|
|
312
|
+
and reports independent close failures without inheriting an unrelated outer
|
|
313
|
+
exception handler.
|
|
314
|
+
- Path and opener CSV/JSONL sources open on execution. Handles returned by an
|
|
315
|
+
opener are closed by fpstreams; caller-owned handles remain open. Arrow C
|
|
316
|
+
streams, explicit column mappings, and NumPy inputs keep their documented
|
|
317
|
+
construction-time import or conversion behavior.
|
|
318
|
+
|
|
319
|
+
## 2.0.0
|
|
320
|
+
|
|
321
|
+
fpstreams 2 replaced the v1 implementation with typed lazy plans, a primary
|
|
322
|
+
`Flow` API, explicit `Rows`, `AsyncFlow`, and `Pairs` views, and optional Rust and
|
|
323
|
+
Arrow execution.
|