fpstreams 2.1.0__tar.gz → 2.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fpstreams-2.2.1/CHANGELOG.md +339 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/PKG-INFO +109 -38
- {fpstreams-2.1.0 → fpstreams-2.2.1}/README.md +107 -36
- {fpstreams-2.1.0 → fpstreams-2.2.1}/pyproject.toml +3 -3
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/Cargo.lock +1 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/Cargo.toml +1 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float.rs +13 -12
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/integer/endpoints.rs +103 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/integer.rs +2 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/lib.rs +4 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_group/buffer.rs +8 -8
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/group_numeric.rs +21 -21
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/group_pair_expr.rs +24 -22
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_exact.rs +31 -31
- fpstreams-2.2.1/rust/src/relational_fixed/composite.rs +186 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/group_multi.rs +4 -2
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/single.rs +13 -9
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed.rs +23 -17
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_fixed.rs +158 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/kernels.rs +166 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests.rs +3 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/__init__.py +3 -2
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/_native.pyi +15 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/_collector_base.py +6 -6
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/aggregation.py +5 -2
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/collector.py +7 -7
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_dict.py +12 -4
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_rows_fusion.py +15 -24
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_scalar_fusion.py +2 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/arrow.py +65 -47
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_iterators.py +5 -3
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/numpy_prefix.py +32 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/__init__.py +416 -742
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/join.py +56 -1
- fpstreams-2.2.1/src/fpstreams/execution/sorted_streams.py +407 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sync.py +6 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sync_ops.py +24 -6
- fpstreams-2.2.1/src/fpstreams/expressions/_codegen.py +29 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/_row_codegen.py +2 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/scalar.py +81 -7
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/functional.py +4 -3
- fpstreams-2.2.1/src/fpstreams/io_safety.py +97 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/compiled.py +42 -14
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/relational.py +31 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/arrow.py +15 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/arrow_source.py +5 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/async_.py +21 -4
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/compiler.py +67 -2
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/explain.py +54 -18
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/logical.py +34 -2
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/native.py +83 -36
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/source.py +37 -6
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/option.py +6 -5
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/result.py +10 -10
- fpstreams-2.2.1/src/fpstreams/runtime/_distinct.py +35 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/iterators.py +7 -4
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_async_identity.py +4 -3
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_flow_structural_list.py +7 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_flow_unique_list.py +6 -4
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/async_flow.py +38 -24
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/async_terminals.py +6 -5
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/flow.py +64 -22
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/flow_terminals.py +185 -26
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/pairs.py +51 -6
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/__init__.py +3 -1
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/arrow.py +102 -59
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/factory.py +5 -13
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/grouped.py +7 -3
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/io.py +71 -38
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/join.py +20 -4
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/numpy.py +8 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/rows.py +198 -128
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill.py +10 -5
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/sqlite_sink.py +20 -7
- fpstreams-2.1.0/CHANGELOG.md +0 -56
- fpstreams-2.1.0/src/fpstreams/io_safety.py +0 -38
- {fpstreams-2.1.0 → fpstreams-2.2.1}/LICENSE +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/build.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/common.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float/affine_pair.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float/endpoints.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numeric_mean.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_export.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_group.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/expr.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/prefix.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/row_filter.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/unique.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/value_filter.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/value_map.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pivot.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/records.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/adapters/namedtuple.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/adapters.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/global_numeric.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable/many.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable/unique.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_i64.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/global_multi/same_field.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/global_multi.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/scalar_sort.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/scalar_unique.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/select.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/adapters.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact/pairs.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact/rows.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable/behavior.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable/direct.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_exact.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/buffers.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/reductions.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numpy_group.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/unnest.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/unpivot.rs +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/_provenance.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/aggregate.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/async_flow.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/aggregate_program.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/program.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/reducer.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/statistics.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collectors.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/column.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/errors.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/exceptions.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_aggregate.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_row_filter.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_map.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_merge.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_ops.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_prefetch.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_queue.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_scheduler.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_timers.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/native.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/numpy_group.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/physical.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_global.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_group.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_group_rows.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sorting.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expr.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/program.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row_eval.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row_ir.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/selectors.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/typed_ir.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/gather.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/option.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/async_plan.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/kernel_cache.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/plan.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/_pair_stages.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/async_utils.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/gather.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/numpy.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/pair_i64_expression.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/plan_cache.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantic_analyzer.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantic_rules.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantics.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/sync.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/py.typed +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/result.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/failpoints.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/files.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/limits.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/metrics.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/query.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/report.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/resources.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/spill.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/tasks.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/codec.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/spill_store.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/__init__.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/_text_sources.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/dataframe.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/polars.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/records.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill_io.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill_limits.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/sql.py +0 -0
- {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/testing.py +0 -0
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
This file records user-visible and compatibility-relevant changes in fpstreams 2,
|
|
4
|
+
including changed defaults.
|
|
5
|
+
|
|
6
|
+
## Unreleased
|
|
7
|
+
|
|
8
|
+
## 2.2.1 - 2026-10-07
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
|
|
12
|
+
- Editors resolve the public `rows` factory directly, preserving method completion,
|
|
13
|
+
parameter hints, and docstrings instead of treating it as the record module.
|
|
14
|
+
- Core `flow()` and `rows()` factories retain iterable element types without optional
|
|
15
|
+
data adapters installed. A shared series protocol preserves one-dimensional series
|
|
16
|
+
typing without importing Polars into the core type declarations.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
|
|
20
|
+
- Automatic planning skips NumPy buffer probes for exact list, tuple, and range
|
|
21
|
+
sources. Python identity terminal plans on exact containers also avoid a temporary
|
|
22
|
+
decision object, reducing the fixed cost of small workloads.
|
|
23
|
+
|
|
24
|
+
## 2.2.0 - 2026-10-06
|
|
25
|
+
|
|
26
|
+
### Added
|
|
27
|
+
|
|
28
|
+
- Add bounded `Rows.join_sorted()` for explicitly ordered records, with consumed-prefix
|
|
29
|
+
validation and first-right-record schema rules.
|
|
30
|
+
- Report explicit sorted group, merge, and join execution through existing report fields.
|
|
31
|
+
- Add optional atomic path output to Flow CSV/JSON/JSONL and Rows CSV/JSONL sinks,
|
|
32
|
+
including no-overwrite publication. Existing direct output remains the default.
|
|
33
|
+
- `Flow.to_jsonl()` streams arbitrary JSON values as one value per line.
|
|
34
|
+
`Rows.to_jsonl()` now accepts `default` for custom serialization.
|
|
35
|
+
|
|
36
|
+
- `Flow.merge_sorted()` stably merges two ascending inputs without sorting or
|
|
37
|
+
materializing them. Ties prefer the left input and outputs retain their identity.
|
|
38
|
+
|
|
39
|
+
- `Rows.group_by_sorted()` aggregates adjacent ascending keys in Python with
|
|
40
|
+
current group state and one lookahead row. Consumed keys are checked for exact
|
|
41
|
+
builtin types and order; this explicit mode does not sort or support spill.
|
|
42
|
+
|
|
43
|
+
- `Pairs.run_with_report()` executes a pair terminal once and returns its value
|
|
44
|
+
with an execution report. It supports `to_dict`, `group_values`,
|
|
45
|
+
`collect_values`, and `aggregate_values`.
|
|
46
|
+
|
|
47
|
+
### Fixed
|
|
48
|
+
|
|
49
|
+
- Rust source builds support the declared 1.85 minimum. Native kernels no longer
|
|
50
|
+
require Rust 1.88 through let-chain syntax; CI checks the minimum compiler.
|
|
51
|
+
- SQLite sinks match existing table, view, and column names using SQLite's ASCII
|
|
52
|
+
case rules. Duplicate column aliases raise `DuplicateKeyError` before inserts.
|
|
53
|
+
Fail mode and view rejection check the destination before reading rows.
|
|
54
|
+
- Sync and async flows accept built-in ranges longer than `sys.maxsize`. Bounded
|
|
55
|
+
reads and exact counts no longer fail during source size detection.
|
|
56
|
+
|
|
57
|
+
- `spreadsheet_safe=True` also neutralizes formula-like CSV headers, including
|
|
58
|
+
inferred record field names and explicitly named empty output. Record lookup
|
|
59
|
+
still uses the original names; the raw-output default is unchanged.
|
|
60
|
+
- Relative Parquet output paths stay anchored to the initial working directory
|
|
61
|
+
if a source or conversion callback changes directories during the write.
|
|
62
|
+
|
|
63
|
+
- Scalar keys in explicit sorted operations avoid temporary shape tuples while
|
|
64
|
+
retaining per-row type and ordering checks.
|
|
65
|
+
|
|
66
|
+
- Atomic CSV and JSON output keeps its original destination when a source or
|
|
67
|
+
serializer changes the working directory during execution.
|
|
68
|
+
|
|
69
|
+
- CSV and SQLite record sinks propagate `StopIteration` from first-record
|
|
70
|
+
conversion instead of treating it as empty input. SQLite replacement leaves
|
|
71
|
+
the existing table intact when that conversion fails.
|
|
72
|
+
|
|
73
|
+
- Sync and async `chunk`, `window`, and `batch_by_size` validate integer bounds
|
|
74
|
+
before execution, including their aliases. Floats, NaN, and infinity now raise
|
|
75
|
+
`TypeError` instead of failing after source consumption or leaving async batches
|
|
76
|
+
unbounded. Objects implementing `__index__` are normalized once per bound.
|
|
77
|
+
|
|
78
|
+
- `unique()`, `unique_by()`, and `Pairs.unique_keys()` propagate equality errors
|
|
79
|
+
instead of treating them as unhashable keys. Hash callbacks keep their existing
|
|
80
|
+
lookup and insertion counts.
|
|
81
|
+
- Async uniqueness, `agg.count_distinct()`, and native pair-uniqueness continuation
|
|
82
|
+
also propagate equality errors without extra hash calls. Iterator cleanup retains
|
|
83
|
+
nested exception notes when several owned resources fail to close.
|
|
84
|
+
- Distinct operations avoid key-wrapper allocations for exact built-in strings.
|
|
85
|
+
String subclasses and custom keys retain guarded hash and equality handling.
|
|
86
|
+
- Grouped reductions and frequency counts propagate `KeyError` raised by a key's
|
|
87
|
+
hash or equality method instead of treating it as a missing group. This includes
|
|
88
|
+
async reductions, grouping collectors, Pairs, and spilled grouping.
|
|
89
|
+
- Parquet `if_exists="error"` publishes with an atomic no-overwrite hard link.
|
|
90
|
+
A concurrent creator or dangling symlink cannot be overwritten. Filesystems
|
|
91
|
+
without hard-link support report an error; `replace` still uses atomic rename.
|
|
92
|
+
- Owned Arrow resources report close failures on successful queries and attach
|
|
93
|
+
cleanup diagnostics to an existing query error. Cleanup attempts every owned
|
|
94
|
+
resource. This also changes Arrow `first()`, which previously ignored close errors.
|
|
95
|
+
|
|
96
|
+
- Engine benchmarks calibrate warmed timing blocks for short tasks and first-row
|
|
97
|
+
latency, retaining elapsed time and call counts in their reports. Rebuild
|
|
98
|
+
single-call baselines before comparing. Timing and resource regression limits
|
|
99
|
+
are unchanged; scheduled CI now uploads the three raw reference reports.
|
|
100
|
+
|
|
101
|
+
- Benchmark report schema 6 records Python and glibc allocator environment
|
|
102
|
+
settings. Baseline creation and comparison reject missing or mismatched
|
|
103
|
+
settings; regenerate older reports. The runners do not change allocator defaults.
|
|
104
|
+
|
|
105
|
+
- NumPy frequency benchmarks check key types, integer counts, first-key order,
|
|
106
|
+
and floating-point bit patterns. Separate NaN entries, their signs, and their
|
|
107
|
+
payloads remain distinct when comparing independently computed outputs.
|
|
108
|
+
- Generated Python `with_columns` loops call the live enrichment function.
|
|
109
|
+
Changes to captured accessors, expression evaluators, or the selector list
|
|
110
|
+
remain visible after plan caching and while reading the source. The transform
|
|
111
|
+
still copies the row before invoking selectors against the original input.
|
|
112
|
+
|
|
113
|
+
- Arrow file projections keep selector changes made by custom scan openers
|
|
114
|
+
visible. Parquet rechecks the projection after creating its dataset, including
|
|
115
|
+
with an explicit source filter. CSV keeps full fields when its public reader
|
|
116
|
+
hooks differ from the extension entrypoints. Normal CSV projection retains
|
|
117
|
+
the bounded schema probe and the default data reader.
|
|
118
|
+
- `Rows.select()` rechecks captured field accessors before using Rust, NumPy,
|
|
119
|
+
or Arrow projection metadata. The generated Python loop calls the live
|
|
120
|
+
projection, so changes made during input reads remain visible. Custom Arrow
|
|
121
|
+
batch sources keep fields available for fallback; changed projections can
|
|
122
|
+
infer their output dtype instead of forcing the former field's schema.
|
|
123
|
+
NumPy fallback finishes the already-opened source without reopening it.
|
|
124
|
+
- Pivot calls the live index, column, and value selectors on its Python path.
|
|
125
|
+
Direct dict lookups had ignored changes to selector code, closure bindings,
|
|
126
|
+
and globals. Native admission now checks those bindings before execution;
|
|
127
|
+
its fallback shares the general selector loop.
|
|
128
|
+
- Native pivot falls back when Rows/Flow iteration or source hooks change,
|
|
129
|
+
including changes to function code. Replacement results, exceptions, and
|
|
130
|
+
iterator cleanup remain observable.
|
|
131
|
+
- `frequencies(key=...)` uses the Python pipeline for sequential `auto` plans,
|
|
132
|
+
so key callbacks can affect subsequent input reads. Native materialization
|
|
133
|
+
could hide those changes. Reports identify this route as `python_frequency`.
|
|
134
|
+
- Benchmark speedup requirements for composite count/sum groups and NamedTuple
|
|
135
|
+
callable joins now apply from 1,000 rows. Smaller inputs still emit timing and
|
|
136
|
+
allocation data for cross-run checks. The original ratios and CI workloads
|
|
137
|
+
are unchanged; small runs no longer inherit speedup requirements validated
|
|
138
|
+
for larger inputs.
|
|
139
|
+
- The engine benchmark CLI lists scenarios without running their tasks, timing,
|
|
140
|
+
allocation tracking, or execution observation. Listing and measurement share
|
|
141
|
+
filtering and fixture cleanup, including when a later scenario builder fails.
|
|
142
|
+
- Single-collector grouping preserves the general collector program's state
|
|
143
|
+
release order, including when output is closed early. Unused input keys are
|
|
144
|
+
released before reading the current finisher, so changes made by their release
|
|
145
|
+
callbacks take effect.
|
|
146
|
+
- Single-collector grouping uses the current step after truth-testing a custom
|
|
147
|
+
completion result. It also releases replaced completion values before pulling
|
|
148
|
+
another row, matching the general collector program's callback behavior.
|
|
149
|
+
- Python grouping calls the live key selector for each row. A field shortcut
|
|
150
|
+
could keep using an old field after a source or callback changed the selector's
|
|
151
|
+
code or closure, merging distinct groups. Selector globals and error types
|
|
152
|
+
now follow the same function calls as general grouped aggregation.
|
|
153
|
+
- Scalar caches no longer confuse numerically equal constants of different
|
|
154
|
+
types. Directly constructed expressions retain their Python result types,
|
|
155
|
+
wide integer values, and custom constant identities. Expressions with
|
|
156
|
+
nonstandard operand types use Python in `auto` mode; forcing `native` raises
|
|
157
|
+
`NativeUnsupportedError` instead of changing their numeric representation.
|
|
158
|
+
- Float expressions retain the sign of `-0.0` in their instructions and evaluator
|
|
159
|
+
cache. Compiling a positive-zero expression first no longer changes a later
|
|
160
|
+
negative-zero result. Scalar and Pairs execution keep their existing routes.
|
|
161
|
+
- Python grouped sums call the live collector lifecycle throughout traversal
|
|
162
|
+
and output. Changes to function code or selector closure cells made by a source,
|
|
163
|
+
key callback, or output consumer are no longer hidden by an inlined sum loop.
|
|
164
|
+
- Two-key count/sum groups also use the live collector program. Their separate
|
|
165
|
+
loop skipped changes to initializer and step code or the sum selector closure
|
|
166
|
+
made while reading the source, even though the same collectors worked correctly
|
|
167
|
+
in other group layouts.
|
|
168
|
+
- Grouped aggregation uses the general collector program for custom lifecycle
|
|
169
|
+
getters. A dynamic `step` property or `__getattribute__` override is read for
|
|
170
|
+
each step instead of being cached as a fixed function.
|
|
171
|
+
- Single-collector grouping no longer caches a temporary lifecycle hook between
|
|
172
|
+
key selection and lookup. If hashing replaces that hook, its release callback
|
|
173
|
+
runs before the next hash call, matching the general collector program.
|
|
174
|
+
- Exact builtin checks no longer use custom metaclass equality to infer source
|
|
175
|
+
size or select grouping, spill, range lookup, and float-expression shortcuts.
|
|
176
|
+
Custom iterables are counted by traversal; map/filter fusion does not request
|
|
177
|
+
their length hints. Grouping retains custom key and serialization calls.
|
|
178
|
+
- Execution reports distinguish successful top-level Rust and Arrow record
|
|
179
|
+
joins from Python joins. Native pair aggregation also records its direct route.
|
|
180
|
+
- Benchmark comparisons reject missing provenance and mismatched workloads.
|
|
181
|
+
Both suites record dependencies, Git state, code and workload fingerprints,
|
|
182
|
+
and an untimed observation of the task. Median baselines retain each run's
|
|
183
|
+
provenance. Older reports need to be regenerated for report schema 6.
|
|
184
|
+
- Reports include CPU affinity, NumPy CPU dispatch, and an allowlist of runtime
|
|
185
|
+
settings. Comparisons reject mismatched configurations; both runners also
|
|
186
|
+
reject configuration changes during measurement.
|
|
187
|
+
- Join benchmarks honor the requested engine. Twelve Python controls separate
|
|
188
|
+
dict and Mapping records, field and callable keys, and inner, left, and repeated
|
|
189
|
+
matches. References preserve callable-key columns and snapshots taken before
|
|
190
|
+
key selection, including unmatched left rows.
|
|
191
|
+
- NumPy frequency benchmarks cover integer and float arrays, with and without
|
|
192
|
+
a callable key, at low and high cardinality. All references preserve first-key
|
|
193
|
+
order; Python array-to-list conversion is included in its timed task.
|
|
194
|
+
- Competitive samples warm each task for at least 1ms after GC and record the
|
|
195
|
+
number of warmup calls. One warmup call left inconsistent timings for short
|
|
196
|
+
NumPy tasks in local measurements.
|
|
197
|
+
- Competitive benchmarks now record peak Python allocation in a separate,
|
|
198
|
+
untimed call for every implementation, using the engine suite's shared helper.
|
|
199
|
+
Schema 6 rejects missing or invalid resource measurements, and baseline
|
|
200
|
+
creation rejects inconsistent resource sets instead of substituting zero.
|
|
201
|
+
Earlier competitive reports contain no allocation evidence.
|
|
202
|
+
- Regression checks accept zero allocated bytes when both runs report zero.
|
|
203
|
+
- `frequencies()` on retained lists and tuples falls back to Python when the
|
|
204
|
+
optional Rust extension is unavailable, including in the browser wheel.
|
|
205
|
+
- Frequency benchmarks use bulk conversion for NumPy dictionaries and pandas'
|
|
206
|
+
`to_dict()`, avoiding per-key normalization overhead in the reference tasks.
|
|
207
|
+
|
|
208
|
+
### Changed
|
|
209
|
+
|
|
210
|
+
- Automatic NumPy `frequencies()` without a key can count a selected native
|
|
211
|
+
stream in Rust through its existing iterator. Source fallback and query cleanup
|
|
212
|
+
stay in the physical executor. Custom keys return to Python before hashing;
|
|
213
|
+
old wheels and free-threaded CPython retain the prior counting path.
|
|
214
|
+
- Python `Rows.select()` keeps its per-row selector snapshot as a list, avoiding
|
|
215
|
+
an intermediate tuple conversion. Selector edits still affect subsequent rows,
|
|
216
|
+
and replaced selectors stay alive until the current row finishes.
|
|
217
|
+
|
|
218
|
+
- Python scalar iteration over an exact NumPy ndarray checks its live size
|
|
219
|
+
without creating a shape tuple for every value. Dimension and length checks,
|
|
220
|
+
lazy reads, and the fallback for custom array objects remain in place.
|
|
221
|
+
- Python joins reduce layout-checking overhead for left records with more than
|
|
222
|
+
four fields. Repeated private dict snapshots need no temporary field-name tuple
|
|
223
|
+
or Python comparison generator. Field identity, short-circuiting, suffix keys,
|
|
224
|
+
snapshot lifetimes, and the bounded layout cache retain their existing behavior.
|
|
225
|
+
- Two-key count/sum groups can run in Rust with `engine="auto"` when a retained
|
|
226
|
+
list or tuple contains exact tuple rows and both keys and the selected values
|
|
227
|
+
are plain signed 64-bit integers. The path preserves first-key identities,
|
|
228
|
+
encounter order, and wider sums. Changed functions, one-shot sources, and
|
|
229
|
+
unsupported types use the Python collector program; older extensions can
|
|
230
|
+
decline the optional entry point.
|
|
231
|
+
- `frequencies()` can continue counting in Rust after its bounded integer
|
|
232
|
+
prefix. The continuation supports exact builtin integer, boolean, string,
|
|
233
|
+
bytes, float, and `None` keys in retained lists and tuples on GIL-enabled
|
|
234
|
+
CPython. It updates the output dictionary directly and hands custom keys
|
|
235
|
+
back to Python before calling their hash or equality methods.
|
|
236
|
+
- Frequency execution reports record a completed native count as `rust_direct`
|
|
237
|
+
and a count completed by Python after a native prefix as `python_frequency`.
|
|
238
|
+
|
|
239
|
+
### Internal
|
|
240
|
+
|
|
241
|
+
- Browser wheels record checkout identity and working-tree state. Release labels
|
|
242
|
+
require a clean checkout matching the version tag; dirty or unknown builds
|
|
243
|
+
remain development builds. The playground displays this provenance.
|
|
244
|
+
- Release smoke checks exercise paid-order aggregation and bounded async mapping
|
|
245
|
+
in addition to Python/native integer sums.
|
|
246
|
+
|
|
247
|
+
- Single-collector groups write step results directly to their stored entries
|
|
248
|
+
and avoid reading stored state for completed groups. Group benchmarks now
|
|
249
|
+
include `agg.first()` with repeated and distinct keys.
|
|
250
|
+
- Group benchmarks now cover custom completion predicates with repeated and
|
|
251
|
+
distinct keys, alongside collectors that cannot finish early.
|
|
252
|
+
- Generated scalar expressions, row expressions, and fused loops share a smaller
|
|
253
|
+
AST location pass. It preserves traversal order and the existing synthetic
|
|
254
|
+
source positions used in tracebacks.
|
|
255
|
+
- Scalar program fingerprints use fewer intermediate records while retaining
|
|
256
|
+
the binary framing format, arbitrary-size integers, and float payload bits.
|
|
257
|
+
- Scalar fingerprint encoding reuses fixed instruction headers instead of
|
|
258
|
+
rebuilding their bytes for every query. Expression reads and type checks are
|
|
259
|
+
unchanged; the table retains no expressions or source data.
|
|
260
|
+
- Planning benchmarks separate compilation from execution for integer and float
|
|
261
|
+
expressions and a callable control, using the same pipeline for each pair.
|
|
262
|
+
- Python grouped output uses one fewer forwarding generator. The collector
|
|
263
|
+
iterator still starts lazily and retains its existing cleanup path.
|
|
264
|
+
- Collector lifecycle properties read their original slots directly, removing
|
|
265
|
+
an extra Python wrapper. Frozen assignment, explicit replacement, and deletion
|
|
266
|
+
errors are preserved. The API manifest now classifies those existing fields
|
|
267
|
+
as properties; their names and constructor signatures are unchanged.
|
|
268
|
+
- Add Python dict-group benchmarks for field and callable selectors with repeated
|
|
269
|
+
and distinct keys. Execution observations record the case's requested engine.
|
|
270
|
+
- Include nominal Mapping and mappingproxy field-group cases with repeated and
|
|
271
|
+
distinct keys when measuring Python selector changes.
|
|
272
|
+
- Extend the two-key count/sum benchmarks to repeated keys. Keep the original
|
|
273
|
+
direct-versus-callable timing threshold and report any regression against it.
|
|
274
|
+
- Move the narrow integer-key record-join ABI adapter into the existing join
|
|
275
|
+
executor module. Kernel order, shape guards, and fallback behavior are unchanged.
|
|
276
|
+
|
|
277
|
+
### Documentation
|
|
278
|
+
|
|
279
|
+
- Repair split return descriptions and unresolved API references in generated
|
|
280
|
+
pages. The two `partition_results()` methods now describe their success and
|
|
281
|
+
exception lists separately.
|
|
282
|
+
- Correct the DB-API example's `batch_size` argument and the distinction between
|
|
283
|
+
`Rows.skip()` and `Rows.drop()` in the API index.
|
|
284
|
+
- Explain the difference between a compiled outer plan and a recorded direct
|
|
285
|
+
terminal route, including the limits of reports for compound queries.
|
|
286
|
+
- Revise the website's introductory and performance text and document the next
|
|
287
|
+
benchmark and refactoring steps in the roadmap.
|
|
288
|
+
|
|
289
|
+
## 2.1.0 - 2026-09-01
|
|
290
|
+
|
|
291
|
+
### Added
|
|
292
|
+
|
|
293
|
+
- `flow()` can enter record operations directly, while `Rows` remains available
|
|
294
|
+
as an explicit relational view. New column and NumPy factories make it possible
|
|
295
|
+
to keep columnar inputs columnar until execution.
|
|
296
|
+
- `ExecutionReport` and `run_with_report()` expose the strategy used by a terminal
|
|
297
|
+
without changing the terminal result.
|
|
298
|
+
- Standard `__arrow_c_stream__` and `__dataframe__` providers can be routed through
|
|
299
|
+
`flow()`.
|
|
300
|
+
- Arrow-backed CSV scanning supports typed incremental reads and query projection
|
|
301
|
+
under PyArrow's parsing and error contract.
|
|
302
|
+
- `AsyncFlow` now includes queue sources, bounded prefetch, session windows, numeric
|
|
303
|
+
terminals, and execution reports.
|
|
304
|
+
- `Pairs` accepts explicit engine selection and row expressions for pair filtering.
|
|
305
|
+
|
|
306
|
+
### Changed
|
|
307
|
+
|
|
308
|
+
- Retained NumPy matrices can execute guarded identity, projection, filter,
|
|
309
|
+
computed-column, aggregate, and grouped-aggregate paths without first building
|
|
310
|
+
one Python dictionary per input row.
|
|
311
|
+
- Native Rust execution now handles additional scalar, pair, reshape, join,
|
|
312
|
+
group, and global aggregation plans. Unsupported or data-semantics-sensitive
|
|
313
|
+
cases still use the Python path.
|
|
314
|
+
- Record joins and grouped aggregation use narrower shape checks and bounded native
|
|
315
|
+
kernels where they preserve Python ordering, identity, errors, and cleanup.
|
|
316
|
+
- `rows.from_csv()` and `rows.from_jsonl()` now accept caller-owned open handles
|
|
317
|
+
and replayable zero-argument opener functions in addition to paths.
|
|
318
|
+
- The benchmark runner now compares fpstreams with Python, NumPy, and pandas and
|
|
319
|
+
reports the percentage difference for each comparable case.
|
|
320
|
+
|
|
321
|
+
### Fixed
|
|
322
|
+
|
|
323
|
+
- Fast paths now revalidate cached row expressions, collector programs, NumPy
|
|
324
|
+
adapters, and implementation primitives before bypassing Python execution.
|
|
325
|
+
- One-shot sources, iterators, async tasks, database resources, spill files, and
|
|
326
|
+
retained tabular readers keep their cleanup behavior on early return and errors.
|
|
327
|
+
- Cleanup attempts every owned resource, preserves the operation error as primary,
|
|
328
|
+
and reports independent close failures without inheriting an unrelated outer
|
|
329
|
+
exception handler.
|
|
330
|
+
- Path and opener CSV/JSONL sources open on execution. Handles returned by an
|
|
331
|
+
opener are closed by fpstreams; caller-owned handles remain open. Arrow C
|
|
332
|
+
streams, explicit column mappings, and NumPy inputs keep their documented
|
|
333
|
+
construction-time import or conversion behavior.
|
|
334
|
+
|
|
335
|
+
## 2.0.0
|
|
336
|
+
|
|
337
|
+
fpstreams 2 replaced the v1 implementation with typed lazy plans, a primary
|
|
338
|
+
`Flow` API, explicit `Rows`, `AsyncFlow`, and `Pairs` views, and optional Rust and
|
|
339
|
+
Arrow execution.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fpstreams
|
|
3
|
-
Version: 2.1
|
|
3
|
+
Version: 2.2.1
|
|
4
4
|
Classifier: Development Status :: 4 - Beta
|
|
5
5
|
Classifier: Programming Language :: Python :: 3
|
|
6
6
|
Classifier: Programming Language :: Python :: 3.11
|
|
@@ -42,7 +42,7 @@ Provides-Extra: data
|
|
|
42
42
|
Provides-Extra: polars
|
|
43
43
|
Provides-Extra: test
|
|
44
44
|
License-File: LICENSE
|
|
45
|
-
Summary:
|
|
45
|
+
Summary: Lazy Python pipelines for iterables and records, with bounded async concurrency.
|
|
46
46
|
Keywords: apache-arrow,asyncio,data-pipelines,data-processing,functional-programming,lazy-evaluation,numpy,rust,stream-processing,typed-python
|
|
47
47
|
Author-email: Steven Yang <stevenyang0316@gmail.com>
|
|
48
48
|
License-Expression: MIT
|
|
@@ -62,8 +62,83 @@ Project-URL: Homepage, https://github.com/steventimes/fpstreams
|
|
|
62
62
|
|
|
63
63
|
[Documentation](https://steventimes.github.io/fpstreams/) · [Browser playground](https://steventimes.github.io/fpstreams/playground/) · [Changelog](https://github.com/steventimes/fpstreams/blob/master/CHANGELOG.md) · [Contributing](https://github.com/steventimes/fpstreams/blob/master/CONTRIBUTING.md)
|
|
64
64
|
|
|
65
|
-
|
|
66
|
-
|
|
65
|
+
Lazy Python pipelines for iterables and records, with bounded async concurrency.
|
|
66
|
+
|
|
67
|
+
Python 3.11 or newer is required.
|
|
68
|
+
|
|
69
|
+
## Install
|
|
70
|
+
|
|
71
|
+
Install the latest stable release:
|
|
72
|
+
|
|
73
|
+
~~~bash
|
|
74
|
+
python -m pip install fpstreams
|
|
75
|
+
~~~
|
|
76
|
+
|
|
77
|
+
Prebuilt wheels include the native extension. Building from source requires
|
|
78
|
+
Rust 1.85 or newer.
|
|
79
|
+
|
|
80
|
+
The `async` extra installs `aiofiles` for async file adapters. Core async
|
|
81
|
+
pipelines do not require an extra. Install other adapters only when needed:
|
|
82
|
+
|
|
83
|
+
~~~bash
|
|
84
|
+
python -m pip install "fpstreams[async]" # aiofiles for async file adapters
|
|
85
|
+
python -m pip install "fpstreams[arrow]" # PyArrow and Parquet
|
|
86
|
+
python -m pip install "fpstreams[data]" # NumPy, pandas, and PyArrow
|
|
87
|
+
python -m pip install "fpstreams[polars]" # Polars and PyArrow
|
|
88
|
+
~~~
|
|
89
|
+
|
|
90
|
+
## Editor support
|
|
91
|
+
|
|
92
|
+
fpstreams includes inline type annotations, docstrings, and the `py.typed` marker.
|
|
93
|
+
Select the Python interpreter where fpstreams is installed in your editor to get
|
|
94
|
+
method completions, parameter hints, and hover documentation.
|
|
95
|
+
|
|
96
|
+
Version 2.2.1 fixes `rows` factory completion and keeps core
|
|
97
|
+
iterable element types when optional data adapters are absent. For example,
|
|
98
|
+
`flow([1, 2, 3])` is inferred as `Flow[int]`.
|
|
99
|
+
|
|
100
|
+
## Quick start
|
|
101
|
+
|
|
102
|
+
This example filters paid orders, groups them by region, and reports both the
|
|
103
|
+
number of orders and their revenue:
|
|
104
|
+
|
|
105
|
+
~~~python
|
|
106
|
+
from fpstreams import agg, col, flow
|
|
107
|
+
|
|
108
|
+
orders = [
|
|
109
|
+
{"region": "eu", "status": "paid", "amount": 24},
|
|
110
|
+
{"region": "us", "status": "paid", "amount": 20},
|
|
111
|
+
{"region": "eu", "status": "cancelled", "amount": 99},
|
|
112
|
+
{"region": "eu", "status": "paid", "amount": 24},
|
|
113
|
+
]
|
|
114
|
+
|
|
115
|
+
result = (
|
|
116
|
+
flow(orders)
|
|
117
|
+
.filter(col("status") == "paid")
|
|
118
|
+
.group_by("region")
|
|
119
|
+
.aggregate(
|
|
120
|
+
orders=agg.count(),
|
|
121
|
+
revenue=agg.sum("amount"),
|
|
122
|
+
)
|
|
123
|
+
.sort_by("region")
|
|
124
|
+
.to_list()
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
print(result)
|
|
128
|
+
# [{'region': 'eu', 'orders': 2, 'revenue': 48},
|
|
129
|
+
# {'region': 'us', 'orders': 1, 'revenue': 20}]
|
|
130
|
+
~~~
|
|
131
|
+
|
|
132
|
+
## When to use fpstreams
|
|
133
|
+
|
|
134
|
+
For a short, local transformation, a comprehension is often enough. fpstreams
|
|
135
|
+
is useful when a pipeline needs grouped aggregations, early termination,
|
|
136
|
+
resource cleanup, or async work with a concurrency limit. Small pipelines may
|
|
137
|
+
be slower because planning and dispatch add overhead.
|
|
138
|
+
|
|
139
|
+
Lazy execution does not mean every operation uses constant memory. Sorting,
|
|
140
|
+
grouping, and joins need additional state; use the documented limits and spill
|
|
141
|
+
options when the input may not fit in memory.
|
|
67
142
|
|
|
68
143
|
> fpstreams 2 replaces the v1 implementation and retains the compatibility
|
|
69
144
|
> aliases listed below.
|
|
@@ -74,8 +149,7 @@ asynchronous concurrency, record-oriented transforms, and optional Rust executio
|
|
|
74
149
|
pipelines, including retained tabular sources.
|
|
75
150
|
- `AsyncFlow[T]`: asynchronous transforms with bounded concurrency, ordering,
|
|
76
151
|
timeouts, merging, debouncing, and cleanup of tasks created by the pipeline.
|
|
77
|
-
- `Rows[T]`:
|
|
78
|
-
joins, grouping, reshape operations, and record-oriented data I/O.
|
|
152
|
+
- `Rows[T]`: a record view for expressions, joins, grouping, reshaping, and I/O.
|
|
79
153
|
- `Pairs[K, V]`: key/value transforms and per-key collection or aggregation.
|
|
80
154
|
- `Collector` and `Aggregator`: single-pass reductions, including named
|
|
81
155
|
multi-aggregation.
|
|
@@ -90,30 +164,12 @@ standard Arrow/dataframe protocol routing, retained NumPy execution, and new
|
|
|
90
164
|
async queue, prefetch, window, and numeric terminal APIs. See the
|
|
91
165
|
[changelog](https://github.com/steventimes/fpstreams/blob/master/CHANGELOG.md) for the release summary.
|
|
92
166
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
conservatively.
|
|
98
|
-
|
|
99
|
-
## Installation
|
|
167
|
+
Release testing covers standard CPython 3.11 through 3.14. Free-threaded
|
|
168
|
+
CPython 3.14t is exercised by an experimental, non-blocking job that builds the
|
|
169
|
+
native extension; it is not currently a release-wheel target, and unsupported
|
|
170
|
+
fast paths fall back conservatively.
|
|
100
171
|
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
~~~bash
|
|
104
|
-
pip install fpstreams
|
|
105
|
-
~~~
|
|
106
|
-
|
|
107
|
-
Install optional integrations only when needed:
|
|
108
|
-
|
|
109
|
-
~~~bash
|
|
110
|
-
pip install "fpstreams[async]" # aiofiles
|
|
111
|
-
pip install "fpstreams[arrow]" # PyArrow and Parquet
|
|
112
|
-
pip install "fpstreams[data]" # NumPy, pandas, and PyArrow
|
|
113
|
-
pip install "fpstreams[polars]" # Polars and PyArrow
|
|
114
|
-
~~~
|
|
115
|
-
|
|
116
|
-
## Quick start
|
|
172
|
+
### Value pipelines
|
|
117
173
|
|
|
118
174
|
Pipelines are lazy. Transformations build a plan; terminal operations such as
|
|
119
175
|
`to_list()`, `aggregate()`, `first()`, and `count()` execute it.
|
|
@@ -228,6 +284,9 @@ active.to_parquet("active-accounts.parquet")
|
|
|
228
284
|
`map_async` accepts synchronous or asynchronous callables. `concurrency` bounds
|
|
229
285
|
in-flight tasks, while `ordered=True` preserves input order.
|
|
230
286
|
|
|
287
|
+
The core async API needs no optional extra. Install `fpstreams[async]` when you
|
|
288
|
+
also need the `aiofiles`-based async file adapters.
|
|
289
|
+
|
|
231
290
|
~~~python
|
|
232
291
|
import asyncio
|
|
233
292
|
|
|
@@ -246,7 +305,8 @@ async def main() -> None:
|
|
|
246
305
|
.filter(lambda value: value >= 20)
|
|
247
306
|
.to_list()
|
|
248
307
|
)
|
|
249
|
-
|
|
308
|
+
print(result)
|
|
309
|
+
# [20, 30, 40]
|
|
250
310
|
|
|
251
311
|
|
|
252
312
|
asyncio.run(main())
|
|
@@ -273,10 +333,10 @@ assert totals == {
|
|
|
273
333
|
|
|
274
334
|
## Execution engines
|
|
275
335
|
|
|
276
|
-
The default `auto` engine chooses
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
336
|
+
The default `auto` engine chooses Python, Rust, Arrow, NumPy, or a combination
|
|
337
|
+
for each supported plan. Relational plans may use native shortcuts while keeping
|
|
338
|
+
a Python fallback. Pass the terminal you intend to call to `explain()` so it
|
|
339
|
+
can include that terminal in its planning decisions:
|
|
280
340
|
|
|
281
341
|
~~~python
|
|
282
342
|
from fpstreams import flow, item
|
|
@@ -300,16 +360,27 @@ python_result = pipeline.with_engine("python").to_list()
|
|
|
300
360
|
native_result = pipeline.with_engine("native").to_list()
|
|
301
361
|
~~~
|
|
302
362
|
|
|
363
|
+
Use `run_with_report()` to execute a terminal and inspect its recorded route.
|
|
364
|
+
Version 2.2.0 adds Pairs terminals such as
|
|
365
|
+
`flow([("a", 1)]).pairs().run_with_report("group_values")`. See the
|
|
366
|
+
[execution report guide](https://steventimes.github.io/fpstreams/user-guide/execution-reports/)
|
|
367
|
+
for supported terminals and the limits of route reporting.
|
|
368
|
+
|
|
303
369
|
A forced native plan raises `NativeUnsupportedError` if its complete types or
|
|
304
370
|
operations cannot run natively. In particular, the presence of an internal
|
|
305
371
|
native relational specialization does not make the complete relation eligible
|
|
306
372
|
for `with_engine("native")`. An unsupported forced relational plan fails before
|
|
307
373
|
claiming its one-shot sources; `auto` selects a legal fallback path.
|
|
308
374
|
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
375
|
+
[`run_with_report()`](https://steventimes.github.io/fpstreams/user-guide/execution-reports/)
|
|
376
|
+
returns a terminal's value, recorded route, and query-owned resource counts in
|
|
377
|
+
one execution. It records the outer plan and some direct paths; it does not
|
|
378
|
+
identify every internal kernel or runtime fallback.
|
|
379
|
+
|
|
380
|
+
Identity list and tuple materialization stays in Python. Large, supported integer
|
|
381
|
+
sums and numeric range reductions may use Rust in `auto` mode. `count()` uses a
|
|
382
|
+
known exact size in O(1) when no operation changes cardinality and the source is
|
|
383
|
+
safely reiterable. Use `explain(terminal=...)` to inspect the planned route.
|
|
313
384
|
|
|
314
385
|
## Resource and file-safety controls
|
|
315
386
|
|