fpstreams 2.1.0__tar.gz → 2.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (199) hide show
  1. fpstreams-2.2.1/CHANGELOG.md +339 -0
  2. {fpstreams-2.1.0 → fpstreams-2.2.1}/PKG-INFO +109 -38
  3. {fpstreams-2.1.0 → fpstreams-2.2.1}/README.md +107 -36
  4. {fpstreams-2.1.0 → fpstreams-2.2.1}/pyproject.toml +3 -3
  5. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/Cargo.lock +1 -1
  6. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/Cargo.toml +1 -1
  7. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float.rs +13 -12
  8. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/integer/endpoints.rs +103 -1
  9. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/integer.rs +2 -1
  10. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/lib.rs +4 -1
  11. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_group/buffer.rs +8 -8
  12. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/group_numeric.rs +21 -21
  13. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/group_pair_expr.rs +24 -22
  14. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_exact.rs +31 -31
  15. fpstreams-2.2.1/rust/src/relational_fixed/composite.rs +186 -0
  16. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/group_multi.rs +4 -2
  17. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/single.rs +13 -9
  18. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed.rs +23 -17
  19. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_fixed.rs +158 -0
  20. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/kernels.rs +166 -0
  21. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests.rs +3 -1
  22. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/__init__.py +3 -2
  23. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/_native.pyi +15 -0
  24. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/_collector_base.py +6 -6
  25. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/aggregation.py +5 -2
  26. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/collector.py +7 -7
  27. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_dict.py +12 -4
  28. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_rows_fusion.py +15 -24
  29. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_scalar_fusion.py +2 -1
  30. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/arrow.py +65 -47
  31. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_iterators.py +5 -3
  32. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/numpy_prefix.py +32 -1
  33. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/__init__.py +416 -742
  34. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/join.py +56 -1
  35. fpstreams-2.2.1/src/fpstreams/execution/sorted_streams.py +407 -0
  36. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sync.py +6 -1
  37. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sync_ops.py +24 -6
  38. fpstreams-2.2.1/src/fpstreams/expressions/_codegen.py +29 -0
  39. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/_row_codegen.py +2 -1
  40. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/scalar.py +81 -7
  41. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/functional.py +4 -3
  42. fpstreams-2.2.1/src/fpstreams/io_safety.py +97 -0
  43. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/compiled.py +42 -14
  44. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/relational.py +31 -0
  45. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/arrow.py +15 -0
  46. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/arrow_source.py +5 -0
  47. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/async_.py +21 -4
  48. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/compiler.py +67 -2
  49. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/explain.py +54 -18
  50. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/logical.py +34 -2
  51. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/native.py +83 -36
  52. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/source.py +37 -6
  53. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/option.py +6 -5
  54. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/result.py +10 -10
  55. fpstreams-2.2.1/src/fpstreams/runtime/_distinct.py +35 -0
  56. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/iterators.py +7 -4
  57. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_async_identity.py +4 -3
  58. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_flow_structural_list.py +7 -1
  59. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/_flow_unique_list.py +6 -4
  60. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/async_flow.py +38 -24
  61. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/async_terminals.py +6 -5
  62. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/flow.py +64 -22
  63. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/flow_terminals.py +185 -26
  64. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/pairs.py +51 -6
  65. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/__init__.py +3 -1
  66. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/arrow.py +102 -59
  67. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/factory.py +5 -13
  68. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/grouped.py +7 -3
  69. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/io.py +71 -38
  70. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/join.py +20 -4
  71. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/numpy.py +8 -0
  72. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/rows.py +198 -128
  73. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill.py +10 -5
  74. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/sqlite_sink.py +20 -7
  75. fpstreams-2.1.0/CHANGELOG.md +0 -56
  76. fpstreams-2.1.0/src/fpstreams/io_safety.py +0 -38
  77. {fpstreams-2.1.0 → fpstreams-2.2.1}/LICENSE +0 -0
  78. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/build.rs +0 -0
  79. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/common.rs +0 -0
  80. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float/affine_pair.rs +0 -0
  81. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/float/endpoints.rs +0 -0
  82. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numeric_mean.rs +0 -0
  83. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_export.rs +0 -0
  84. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/numpy_group.rs +0 -0
  85. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/expr.rs +0 -0
  86. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/prefix.rs +0 -0
  87. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/row_filter.rs +0 -0
  88. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/unique.rs +0 -0
  89. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/value_filter.rs +0 -0
  90. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair/value_map.rs +0 -0
  91. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pair.rs +0 -0
  92. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/pivot.rs +0 -0
  93. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/records.rs +0 -0
  94. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/adapters/namedtuple.rs +0 -0
  95. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/adapters.rs +0 -0
  96. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/global_numeric.rs +0 -0
  97. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable/many.rs +0 -0
  98. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable/unique.rs +0 -0
  99. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_callable.rs +0 -0
  100. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational/join_i64.rs +0 -0
  101. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational.rs +0 -0
  102. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/global_multi/same_field.rs +0 -0
  103. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/relational_fixed/global_multi.rs +0 -0
  104. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/scalar_sort.rs +0 -0
  105. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/scalar_unique.rs +0 -0
  106. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/select.rs +0 -0
  107. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/adapters.rs +0 -0
  108. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact/pairs.rs +0 -0
  109. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact/rows.rs +0 -0
  110. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/group_exact.rs +0 -0
  111. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable/behavior.rs +0 -0
  112. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable/direct.rs +0 -0
  113. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_callable.rs +0 -0
  114. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/join_exact.rs +0 -0
  115. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/buffers.rs +0 -0
  116. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric/reductions.rs +0 -0
  117. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numeric.rs +0 -0
  118. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/tests/numpy_group.rs +0 -0
  119. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/unnest.rs +0 -0
  120. {fpstreams-2.1.0 → fpstreams-2.2.1}/rust/src/unpivot.rs +0 -0
  121. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/_provenance.py +0 -0
  122. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/aggregate.py +0 -0
  123. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/async_flow.py +0 -0
  124. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/__init__.py +0 -0
  125. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/aggregate_program.py +0 -0
  126. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/program.py +0 -0
  127. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/reducer.py +0 -0
  128. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collecting/statistics.py +0 -0
  129. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/collectors.py +0 -0
  130. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/column.py +0 -0
  131. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/errors.py +0 -0
  132. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/exceptions.py +0 -0
  133. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/__init__.py +0 -0
  134. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_aggregate.py +0 -0
  135. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/_pair_row_filter.py +0 -0
  136. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_map.py +0 -0
  137. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_merge.py +0 -0
  138. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_ops.py +0 -0
  139. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_prefetch.py +0 -0
  140. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_queue.py +0 -0
  141. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_scheduler.py +0 -0
  142. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/async_timers.py +0 -0
  143. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/native.py +0 -0
  144. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/numpy_group.py +0 -0
  145. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/physical.py +0 -0
  146. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_global.py +0 -0
  147. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_group.py +0 -0
  148. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/relational/arrow_group_rows.py +0 -0
  149. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/execution/sorting.py +0 -0
  150. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expr.py +0 -0
  151. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/__init__.py +0 -0
  152. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/program.py +0 -0
  153. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row.py +0 -0
  154. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row_eval.py +0 -0
  155. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/row_ir.py +0 -0
  156. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/selectors.py +0 -0
  157. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/expressions/typed_ir.py +0 -0
  158. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/gather.py +0 -0
  159. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/option.py +0 -0
  160. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/__init__.py +0 -0
  161. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/async_plan.py +0 -0
  162. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/kernel_cache.py +0 -0
  163. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/physical/plan.py +0 -0
  164. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/__init__.py +0 -0
  165. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/_pair_stages.py +0 -0
  166. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/async_utils.py +0 -0
  167. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/gather.py +0 -0
  168. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/numpy.py +0 -0
  169. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/pair_i64_expression.py +0 -0
  170. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/plan_cache.py +0 -0
  171. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantic_analyzer.py +0 -0
  172. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantic_rules.py +0 -0
  173. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/semantics.py +0 -0
  174. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/planning/sync.py +0 -0
  175. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/primitives/__init__.py +0 -0
  176. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/py.typed +0 -0
  177. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/result.py +0 -0
  178. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/__init__.py +0 -0
  179. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/failpoints.py +0 -0
  180. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/files.py +0 -0
  181. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/limits.py +0 -0
  182. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/metrics.py +0 -0
  183. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/query.py +0 -0
  184. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/report.py +0 -0
  185. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/resources.py +0 -0
  186. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/spill.py +0 -0
  187. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/runtime/tasks.py +0 -0
  188. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/__init__.py +0 -0
  189. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/codec.py +0 -0
  190. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/storage/spill_store.py +0 -0
  191. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/streams/__init__.py +0 -0
  192. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/_text_sources.py +0 -0
  193. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/dataframe.py +0 -0
  194. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/polars.py +0 -0
  195. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/records.py +0 -0
  196. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill_io.py +0 -0
  197. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/spill_limits.py +0 -0
  198. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/tabular/sql.py +0 -0
  199. {fpstreams-2.1.0 → fpstreams-2.2.1}/src/fpstreams/testing.py +0 -0
@@ -0,0 +1,339 @@
1
+ # Changelog
2
+
3
+ This file records user-visible and compatibility-relevant changes in fpstreams 2,
4
+ including changed defaults.
5
+
6
+ ## Unreleased
7
+
8
+ ## 2.2.1 - 2026-10-07
9
+
10
+ ### Fixed
11
+
12
+ - Editors resolve the public `rows` factory directly, preserving method completion,
13
+ parameter hints, and docstrings instead of treating it as the record module.
14
+ - Core `flow()` and `rows()` factories retain iterable element types without optional
15
+ data adapters installed. A shared series protocol preserves one-dimensional series
16
+ typing without importing Polars into the core type declarations.
17
+
18
+ ### Changed
19
+
20
+ - Automatic planning skips NumPy buffer probes for exact list, tuple, and range
21
+ sources. Python identity terminal plans on exact containers also avoid a temporary
22
+ decision object, reducing the fixed cost of small workloads.
23
+
24
+ ## 2.2.0 - 2026-10-06
25
+
26
+ ### Added
27
+
28
+ - Add bounded `Rows.join_sorted()` for explicitly ordered records, with consumed-prefix
29
+ validation and first-right-record schema rules.
30
+ - Report explicit sorted group, merge, and join execution through existing report fields.
31
+ - Add optional atomic path output to Flow CSV/JSON/JSONL and Rows CSV/JSONL sinks,
32
+ including no-overwrite publication. Existing direct output remains the default.
33
+ - `Flow.to_jsonl()` streams arbitrary JSON values as one value per line.
34
+ `Rows.to_jsonl()` now accepts `default` for custom serialization.
35
+
36
+ - `Flow.merge_sorted()` stably merges two ascending inputs without sorting or
37
+ materializing them. Ties prefer the left input and outputs retain their identity.
38
+
39
+ - `Rows.group_by_sorted()` aggregates adjacent ascending keys in Python with
40
+ current group state and one lookahead row. Consumed keys are checked for exact
41
+ builtin types and order; this explicit mode does not sort or support spill.
42
+
43
+ - `Pairs.run_with_report()` executes a pair terminal once and returns its value
44
+ with an execution report. It supports `to_dict`, `group_values`,
45
+ `collect_values`, and `aggregate_values`.
46
+
47
+ ### Fixed
48
+
49
+ - Rust source builds support the declared 1.85 minimum. Native kernels no longer
50
+ require Rust 1.88 through let-chain syntax; CI checks the minimum compiler.
51
+ - SQLite sinks match existing table, view, and column names using SQLite's ASCII
52
+ case rules. Duplicate column aliases raise `DuplicateKeyError` before inserts.
53
+ Fail mode and view rejection check the destination before reading rows.
54
+ - Sync and async flows accept built-in ranges longer than `sys.maxsize`. Bounded
55
+ reads and exact counts no longer fail during source size detection.
56
+
57
+ - `spreadsheet_safe=True` also neutralizes formula-like CSV headers, including
58
+ inferred record field names and explicitly named empty output. Record lookup
59
+ still uses the original names; the raw-output default is unchanged.
60
+ - Relative Parquet output paths stay anchored to the initial working directory
61
+ if a source or conversion callback changes directories during the write.
62
+
63
+ - Scalar keys in explicit sorted operations avoid temporary shape tuples while
64
+ retaining per-row type and ordering checks.
65
+
66
+ - Atomic CSV and JSON output keeps its original destination when a source or
67
+ serializer changes the working directory during execution.
68
+
69
+ - CSV and SQLite record sinks propagate `StopIteration` from first-record
70
+ conversion instead of treating it as empty input. SQLite replacement leaves
71
+ the existing table intact when that conversion fails.
72
+
73
+ - Sync and async `chunk`, `window`, and `batch_by_size` validate integer bounds
74
+ before execution, including their aliases. Floats, NaN, and infinity now raise
75
+ `TypeError` instead of failing after source consumption or leaving async batches
76
+ unbounded. Objects implementing `__index__` are normalized once per bound.
77
+
78
+ - `unique()`, `unique_by()`, and `Pairs.unique_keys()` propagate equality errors
79
+ instead of treating them as unhashable keys. Hash callbacks keep their existing
80
+ lookup and insertion counts.
81
+ - Async uniqueness, `agg.count_distinct()`, and native pair-uniqueness continuation
82
+ also propagate equality errors without extra hash calls. Iterator cleanup retains
83
+ nested exception notes when several owned resources fail to close.
84
+ - Distinct operations avoid key-wrapper allocations for exact built-in strings.
85
+ String subclasses and custom keys retain guarded hash and equality handling.
86
+ - Grouped reductions and frequency counts propagate `KeyError` raised by a key's
87
+ hash or equality method instead of treating it as a missing group. This includes
88
+ async reductions, grouping collectors, Pairs, and spilled grouping.
89
+ - Parquet `if_exists="error"` publishes with an atomic no-overwrite hard link.
90
+ A concurrent creator or dangling symlink cannot be overwritten. Filesystems
91
+ without hard-link support report an error; `replace` still uses atomic rename.
92
+ - Owned Arrow resources report close failures on successful queries and attach
93
+ cleanup diagnostics to an existing query error. Cleanup attempts every owned
94
+ resource. This also changes Arrow `first()`, which previously ignored close errors.
95
+
96
+ - Engine benchmarks calibrate warmed timing blocks for short tasks and first-row
97
+ latency, retaining elapsed time and call counts in their reports. Rebuild
98
+ single-call baselines before comparing. Timing and resource regression limits
99
+ are unchanged; scheduled CI now uploads the three raw reference reports.
100
+
101
+ - Benchmark report schema 6 records Python and glibc allocator environment
102
+ settings. Baseline creation and comparison reject missing or mismatched
103
+ settings; regenerate older reports. The runners do not change allocator defaults.
104
+
105
+ - NumPy frequency benchmarks check key types, integer counts, first-key order,
106
+ and floating-point bit patterns. Separate NaN entries, their signs, and their
107
+ payloads remain distinct when comparing independently computed outputs.
108
+ - Generated Python `with_columns` loops call the live enrichment function.
109
+ Changes to captured accessors, expression evaluators, or the selector list
110
+ remain visible after plan caching and while reading the source. The transform
111
+ still copies the row before invoking selectors against the original input.
112
+
113
+ - Arrow file projections keep selector changes made by custom scan openers
114
+ visible. Parquet rechecks the projection after creating its dataset, including
115
+ with an explicit source filter. CSV keeps full fields when its public reader
116
+ hooks differ from the extension entrypoints. Normal CSV projection retains
117
+ the bounded schema probe and the default data reader.
118
+ - `Rows.select()` rechecks captured field accessors before using Rust, NumPy,
119
+ or Arrow projection metadata. The generated Python loop calls the live
120
+ projection, so changes made during input reads remain visible. Custom Arrow
121
+ batch sources keep fields available for fallback; changed projections can
122
+ infer their output dtype instead of forcing the former field's schema.
123
+ NumPy fallback finishes the already-opened source without reopening it.
124
+ - Pivot calls the live index, column, and value selectors on its Python path.
125
+ Direct dict lookups had ignored changes to selector code, closure bindings,
126
+ and globals. Native admission now checks those bindings before execution;
127
+ its fallback shares the general selector loop.
128
+ - Native pivot falls back when Rows/Flow iteration or source hooks change,
129
+ including changes to function code. Replacement results, exceptions, and
130
+ iterator cleanup remain observable.
131
+ - `frequencies(key=...)` uses the Python pipeline for sequential `auto` plans,
132
+ so key callbacks can affect subsequent input reads. Native materialization
133
+ could hide those changes. Reports identify this route as `python_frequency`.
134
+ - Benchmark speedup requirements for composite count/sum groups and NamedTuple
135
+ callable joins now apply from 1,000 rows. Smaller inputs still emit timing and
136
+ allocation data for cross-run checks. The original ratios and CI workloads
137
+ are unchanged; small runs no longer inherit speedup requirements validated
138
+ for larger inputs.
139
+ - The engine benchmark CLI lists scenarios without running their tasks, timing,
140
+ allocation tracking, or execution observation. Listing and measurement share
141
+ filtering and fixture cleanup, including when a later scenario builder fails.
142
+ - Single-collector grouping preserves the general collector program's state
143
+ release order, including when output is closed early. Unused input keys are
144
+ released before reading the current finisher, so changes made by their release
145
+ callbacks take effect.
146
+ - Single-collector grouping uses the current step after truth-testing a custom
147
+ completion result. It also releases replaced completion values before pulling
148
+ another row, matching the general collector program's callback behavior.
149
+ - Python grouping calls the live key selector for each row. A field shortcut
150
+ could keep using an old field after a source or callback changed the selector's
151
+ code or closure, merging distinct groups. Selector globals and error types
152
+ now follow the same function calls as general grouped aggregation.
153
+ - Scalar caches no longer confuse numerically equal constants of different
154
+ types. Directly constructed expressions retain their Python result types,
155
+ wide integer values, and custom constant identities. Expressions with
156
+ nonstandard operand types use Python in `auto` mode; forcing `native` raises
157
+ `NativeUnsupportedError` instead of changing their numeric representation.
158
+ - Float expressions retain the sign of `-0.0` in their instructions and evaluator
159
+ cache. Compiling a positive-zero expression first no longer changes a later
160
+ negative-zero result. Scalar and Pairs execution keep their existing routes.
161
+ - Python grouped sums call the live collector lifecycle throughout traversal
162
+ and output. Changes to function code or selector closure cells made by a source,
163
+ key callback, or output consumer are no longer hidden by an inlined sum loop.
164
+ - Two-key count/sum groups also use the live collector program. Their separate
165
+ loop skipped changes to initializer and step code or the sum selector closure
166
+ made while reading the source, even though the same collectors worked correctly
167
+ in other group layouts.
168
+ - Grouped aggregation uses the general collector program for custom lifecycle
169
+ getters. A dynamic `step` property or `__getattribute__` override is read for
170
+ each step instead of being cached as a fixed function.
171
+ - Single-collector grouping no longer caches a temporary lifecycle hook between
172
+ key selection and lookup. If hashing replaces that hook, its release callback
173
+ runs before the next hash call, matching the general collector program.
174
+ - Exact builtin checks no longer use custom metaclass equality to infer source
175
+ size or select grouping, spill, range lookup, and float-expression shortcuts.
176
+ Custom iterables are counted by traversal; map/filter fusion does not request
177
+ their length hints. Grouping retains custom key and serialization calls.
178
+ - Execution reports distinguish successful top-level Rust and Arrow record
179
+ joins from Python joins. Native pair aggregation also records its direct route.
180
+ - Benchmark comparisons reject missing provenance and mismatched workloads.
181
+ Both suites record dependencies, Git state, code and workload fingerprints,
182
+ and an untimed observation of the task. Median baselines retain each run's
183
+ provenance. Older reports need to be regenerated for report schema 6.
184
+ - Reports include CPU affinity, NumPy CPU dispatch, and an allowlist of runtime
185
+ settings. Comparisons reject mismatched configurations; both runners also
186
+ reject configuration changes during measurement.
187
+ - Join benchmarks honor the requested engine. Twelve Python controls separate
188
+ dict and Mapping records, field and callable keys, and inner, left, and repeated
189
+ matches. References preserve callable-key columns and snapshots taken before
190
+ key selection, including unmatched left rows.
191
+ - NumPy frequency benchmarks cover integer and float arrays, with and without
192
+ a callable key, at low and high cardinality. All references preserve first-key
193
+ order; Python array-to-list conversion is included in its timed task.
194
+ - Competitive samples warm each task for at least 1ms after GC and record the
195
+ number of warmup calls. One warmup call left inconsistent timings for short
196
+ NumPy tasks in local measurements.
197
+ - Competitive benchmarks now record peak Python allocation in a separate,
198
+ untimed call for every implementation, using the engine suite's shared helper.
199
+ Schema 6 rejects missing or invalid resource measurements, and baseline
200
+ creation rejects inconsistent resource sets instead of substituting zero.
201
+ Earlier competitive reports contain no allocation evidence.
202
+ - Regression checks accept zero allocated bytes when both runs report zero.
203
+ - `frequencies()` on retained lists and tuples falls back to Python when the
204
+ optional Rust extension is unavailable, including in the browser wheel.
205
+ - Frequency benchmarks use bulk conversion for NumPy dictionaries and pandas'
206
+ `to_dict()`, avoiding per-key normalization overhead in the reference tasks.
207
+
208
+ ### Changed
209
+
210
+ - Automatic NumPy `frequencies()` without a key can count a selected native
211
+ stream in Rust through its existing iterator. Source fallback and query cleanup
212
+ stay in the physical executor. Custom keys return to Python before hashing;
213
+ old wheels and free-threaded CPython retain the prior counting path.
214
+ - Python `Rows.select()` keeps its per-row selector snapshot as a list, avoiding
215
+ an intermediate tuple conversion. Selector edits still affect subsequent rows,
216
+ and replaced selectors stay alive until the current row finishes.
217
+
218
+ - Python scalar iteration over an exact NumPy ndarray checks its live size
219
+ without creating a shape tuple for every value. Dimension and length checks,
220
+ lazy reads, and the fallback for custom array objects remain in place.
221
+ - Python joins reduce layout-checking overhead for left records with more than
222
+ four fields. Repeated private dict snapshots need no temporary field-name tuple
223
+ or Python comparison generator. Field identity, short-circuiting, suffix keys,
224
+ snapshot lifetimes, and the bounded layout cache retain their existing behavior.
225
+ - Two-key count/sum groups can run in Rust with `engine="auto"` when a retained
226
+ list or tuple contains exact tuple rows and both keys and the selected values
227
+ are plain signed 64-bit integers. The path preserves first-key identities,
228
+ encounter order, and wider sums. Changed functions, one-shot sources, and
229
+ unsupported types use the Python collector program; older extensions can
230
+ decline the optional entry point.
231
+ - `frequencies()` can continue counting in Rust after its bounded integer
232
+ prefix. The continuation supports exact builtin integer, boolean, string,
233
+ bytes, float, and `None` keys in retained lists and tuples on GIL-enabled
234
+ CPython. It updates the output dictionary directly and hands custom keys
235
+ back to Python before calling their hash or equality methods.
236
+ - Frequency execution reports record a completed native count as `rust_direct`
237
+ and a count completed by Python after a native prefix as `python_frequency`.
238
+
239
+ ### Internal
240
+
241
+ - Browser wheels record checkout identity and working-tree state. Release labels
242
+ require a clean checkout matching the version tag; dirty or unknown builds
243
+ remain development builds. The playground displays this provenance.
244
+ - Release smoke checks exercise paid-order aggregation and bounded async mapping
245
+ in addition to Python/native integer sums.
246
+
247
+ - Single-collector groups write step results directly to their stored entries
248
+ and avoid reading stored state for completed groups. Group benchmarks now
249
+ include `agg.first()` with repeated and distinct keys.
250
+ - Group benchmarks now cover custom completion predicates with repeated and
251
+ distinct keys, alongside collectors that cannot finish early.
252
+ - Generated scalar expressions, row expressions, and fused loops share a smaller
253
+ AST location pass. It preserves traversal order and the existing synthetic
254
+ source positions used in tracebacks.
255
+ - Scalar program fingerprints use fewer intermediate records while retaining
256
+ the binary framing format, arbitrary-size integers, and float payload bits.
257
+ - Scalar fingerprint encoding reuses fixed instruction headers instead of
258
+ rebuilding their bytes for every query. Expression reads and type checks are
259
+ unchanged; the table retains no expressions or source data.
260
+ - Planning benchmarks separate compilation from execution for integer and float
261
+ expressions and a callable control, using the same pipeline for each pair.
262
+ - Python grouped output uses one fewer forwarding generator. The collector
263
+ iterator still starts lazily and retains its existing cleanup path.
264
+ - Collector lifecycle properties read their original slots directly, removing
265
+ an extra Python wrapper. Frozen assignment, explicit replacement, and deletion
266
+ errors are preserved. The API manifest now classifies those existing fields
267
+ as properties; their names and constructor signatures are unchanged.
268
+ - Add Python dict-group benchmarks for field and callable selectors with repeated
269
+ and distinct keys. Execution observations record the case's requested engine.
270
+ - Include nominal Mapping and mappingproxy field-group cases with repeated and
271
+ distinct keys when measuring Python selector changes.
272
+ - Extend the two-key count/sum benchmarks to repeated keys. Keep the original
273
+ direct-versus-callable timing threshold and report any regression against it.
274
+ - Move the narrow integer-key record-join ABI adapter into the existing join
275
+ executor module. Kernel order, shape guards, and fallback behavior are unchanged.
276
+
277
+ ### Documentation
278
+
279
+ - Repair split return descriptions and unresolved API references in generated
280
+ pages. The two `partition_results()` methods now describe their success and
281
+ exception lists separately.
282
+ - Correct the DB-API example's `batch_size` argument and the distinction between
283
+ `Rows.skip()` and `Rows.drop()` in the API index.
284
+ - Explain the difference between a compiled outer plan and a recorded direct
285
+ terminal route, including the limits of reports for compound queries.
286
+ - Revise the website's introductory and performance text and document the next
287
+ benchmark and refactoring steps in the roadmap.
288
+
289
+ ## 2.1.0 - 2026-09-01
290
+
291
+ ### Added
292
+
293
+ - `flow()` can enter record operations directly, while `Rows` remains available
294
+ as an explicit relational view. New column and NumPy factories make it possible
295
+ to keep columnar inputs columnar until execution.
296
+ - `ExecutionReport` and `run_with_report()` expose the strategy used by a terminal
297
+ without changing the terminal result.
298
+ - Standard `__arrow_c_stream__` and `__dataframe__` providers can be routed through
299
+ `flow()`.
300
+ - Arrow-backed CSV scanning supports typed incremental reads and query projection
301
+ under PyArrow's parsing and error contract.
302
+ - `AsyncFlow` now includes queue sources, bounded prefetch, session windows, numeric
303
+ terminals, and execution reports.
304
+ - `Pairs` accepts explicit engine selection and row expressions for pair filtering.
305
+
306
+ ### Changed
307
+
308
+ - Retained NumPy matrices can execute guarded identity, projection, filter,
309
+ computed-column, aggregate, and grouped-aggregate paths without first building
310
+ one Python dictionary per input row.
311
+ - Native Rust execution now handles additional scalar, pair, reshape, join,
312
+ group, and global aggregation plans. Unsupported or data-semantics-sensitive
313
+ cases still use the Python path.
314
+ - Record joins and grouped aggregation use narrower shape checks and bounded native
315
+ kernels where they preserve Python ordering, identity, errors, and cleanup.
316
+ - `rows.from_csv()` and `rows.from_jsonl()` now accept caller-owned open handles
317
+ and replayable zero-argument opener functions in addition to paths.
318
+ - The benchmark runner now compares fpstreams with Python, NumPy, and pandas and
319
+ reports the percentage difference for each comparable case.
320
+
321
+ ### Fixed
322
+
323
+ - Fast paths now revalidate cached row expressions, collector programs, NumPy
324
+ adapters, and implementation primitives before bypassing Python execution.
325
+ - One-shot sources, iterators, async tasks, database resources, spill files, and
326
+ retained tabular readers keep their cleanup behavior on early return and errors.
327
+ - Cleanup attempts every owned resource, preserves the operation error as primary,
328
+ and reports independent close failures without inheriting an unrelated outer
329
+ exception handler.
330
+ - Path and opener CSV/JSONL sources open on execution. Handles returned by an
331
+ opener are closed by fpstreams; caller-owned handles remain open. Arrow C
332
+ streams, explicit column mappings, and NumPy inputs keep their documented
333
+ construction-time import or conversion behavior.
334
+
335
+ ## 2.0.0
336
+
337
+ fpstreams 2 replaced the v1 implementation with typed lazy plans, a primary
338
+ `Flow` API, explicit `Rows`, `AsyncFlow`, and `Pairs` views, and optional Rust and
339
+ Arrow execution.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fpstreams
3
- Version: 2.1.0
3
+ Version: 2.2.1
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Programming Language :: Python :: 3
6
6
  Classifier: Programming Language :: Python :: 3.11
@@ -42,7 +42,7 @@ Provides-Extra: data
42
42
  Provides-Extra: polars
43
43
  Provides-Extra: test
44
44
  License-File: LICENSE
45
- Summary: Typed, lazy Python data pipelines with bounded async concurrency and optional Rust, Arrow, and NumPy acceleration.
45
+ Summary: Lazy Python pipelines for iterables and records, with bounded async concurrency.
46
46
  Keywords: apache-arrow,asyncio,data-pipelines,data-processing,functional-programming,lazy-evaluation,numpy,rust,stream-processing,typed-python
47
47
  Author-email: Steven Yang <stevenyang0316@gmail.com>
48
48
  License-Expression: MIT
@@ -62,8 +62,83 @@ Project-URL: Homepage, https://github.com/steventimes/fpstreams
62
62
 
63
63
  [Documentation](https://steventimes.github.io/fpstreams/) · [Browser playground](https://steventimes.github.io/fpstreams/playground/) · [Changelog](https://github.com/steventimes/fpstreams/blob/master/CHANGELOG.md) · [Contributing](https://github.com/steventimes/fpstreams/blob/master/CONTRIBUTING.md)
64
64
 
65
- Typed, lazy data pipelines for Python, with synchronous streams, structured
66
- asynchronous concurrency, record-oriented transforms, and optional Rust execution.
65
+ Lazy Python pipelines for iterables and records, with bounded async concurrency.
66
+
67
+ Python 3.11 or newer is required.
68
+
69
+ ## Install
70
+
71
+ Install the latest stable release:
72
+
73
+ ~~~bash
74
+ python -m pip install fpstreams
75
+ ~~~
76
+
77
+ Prebuilt wheels include the native extension. Building from source requires
78
+ Rust 1.85 or newer.
79
+
80
+ The `async` extra installs `aiofiles` for async file adapters. Core async
81
+ pipelines do not require an extra. Install other adapters only when needed:
82
+
83
+ ~~~bash
84
+ python -m pip install "fpstreams[async]" # aiofiles for async file adapters
85
+ python -m pip install "fpstreams[arrow]" # PyArrow and Parquet
86
+ python -m pip install "fpstreams[data]" # NumPy, pandas, and PyArrow
87
+ python -m pip install "fpstreams[polars]" # Polars and PyArrow
88
+ ~~~
89
+
90
+ ## Editor support
91
+
92
+ fpstreams includes inline type annotations, docstrings, and the `py.typed` marker.
93
+ Select the Python interpreter where fpstreams is installed in your editor to get
94
+ method completions, parameter hints, and hover documentation.
95
+
96
+ Version 2.2.1 fixes `rows` factory completion and keeps core
97
+ iterable element types when optional data adapters are absent. For example,
98
+ `flow([1, 2, 3])` is inferred as `Flow[int]`.
99
+
100
+ ## Quick start
101
+
102
+ This example filters paid orders, groups them by region, and reports both the
103
+ number of orders and their revenue:
104
+
105
+ ~~~python
106
+ from fpstreams import agg, col, flow
107
+
108
+ orders = [
109
+ {"region": "eu", "status": "paid", "amount": 24},
110
+ {"region": "us", "status": "paid", "amount": 20},
111
+ {"region": "eu", "status": "cancelled", "amount": 99},
112
+ {"region": "eu", "status": "paid", "amount": 24},
113
+ ]
114
+
115
+ result = (
116
+ flow(orders)
117
+ .filter(col("status") == "paid")
118
+ .group_by("region")
119
+ .aggregate(
120
+ orders=agg.count(),
121
+ revenue=agg.sum("amount"),
122
+ )
123
+ .sort_by("region")
124
+ .to_list()
125
+ )
126
+
127
+ print(result)
128
+ # [{'region': 'eu', 'orders': 2, 'revenue': 48},
129
+ # {'region': 'us', 'orders': 1, 'revenue': 20}]
130
+ ~~~
131
+
132
+ ## When to use fpstreams
133
+
134
+ For a short, local transformation, a comprehension is often enough. fpstreams
135
+ is useful when a pipeline needs grouped aggregations, early termination,
136
+ resource cleanup, or async work with a concurrency limit. Small pipelines may
137
+ be slower because planning and dispatch add overhead.
138
+
139
+ Lazy execution does not mean every operation uses constant memory. Sorting,
140
+ grouping, and joins need additional state; use the documented limits and spill
141
+ options when the input may not fit in memory.
67
142
 
68
143
  > fpstreams 2 replaces the v1 implementation and retains the compatibility
69
144
  > aliases listed below.
@@ -74,8 +149,7 @@ asynchronous concurrency, record-oriented transforms, and optional Rust executio
74
149
  pipelines, including retained tabular sources.
75
150
  - `AsyncFlow[T]`: asynchronous transforms with bounded concurrency, ordering,
76
151
  timeouts, merging, debouncing, and cleanup of tasks created by the pipeline.
77
- - `Rows[T]`: an explicit relational and compatibility view for expressions,
78
- joins, grouping, reshape operations, and record-oriented data I/O.
152
+ - `Rows[T]`: a record view for expressions, joins, grouping, reshaping, and I/O.
79
153
  - `Pairs[K, V]`: key/value transforms and per-key collection or aggregation.
80
154
  - `Collector` and `Aggregator`: single-pass reductions, including named
81
155
  multi-aggregation.
@@ -90,30 +164,12 @@ standard Arrow/dataframe protocol routing, retained NumPy execution, and new
90
164
  async queue, prefetch, window, and numeric terminal APIs. See the
91
165
  [changelog](https://github.com/steventimes/fpstreams/blob/master/CHANGELOG.md) for the release summary.
92
166
 
93
- Python 3.11 or newer is required. Release testing covers standard CPython 3.11
94
- through 3.14. Free-threaded CPython 3.14t is exercised by an experimental,
95
- non-blocking job that builds the native extension on a 3.14t interpreter; it is
96
- not currently a release-wheel target, and unsupported fast paths fall back
97
- conservatively.
98
-
99
- ## Installation
167
+ Release testing covers standard CPython 3.11 through 3.14. Free-threaded
168
+ CPython 3.14t is exercised by an experimental, non-blocking job that builds the
169
+ native extension; it is not currently a release-wheel target, and unsupported
170
+ fast paths fall back conservatively.
100
171
 
101
- Install the latest stable release:
102
-
103
- ~~~bash
104
- pip install fpstreams
105
- ~~~
106
-
107
- Install optional integrations only when needed:
108
-
109
- ~~~bash
110
- pip install "fpstreams[async]" # aiofiles
111
- pip install "fpstreams[arrow]" # PyArrow and Parquet
112
- pip install "fpstreams[data]" # NumPy, pandas, and PyArrow
113
- pip install "fpstreams[polars]" # Polars and PyArrow
114
- ~~~
115
-
116
- ## Quick start
172
+ ### Value pipelines
117
173
 
118
174
  Pipelines are lazy. Transformations build a plan; terminal operations such as
119
175
  `to_list()`, `aggregate()`, `first()`, and `count()` execute it.
@@ -228,6 +284,9 @@ active.to_parquet("active-accounts.parquet")
228
284
  `map_async` accepts synchronous or asynchronous callables. `concurrency` bounds
229
285
  in-flight tasks, while `ordered=True` preserves input order.
230
286
 
287
+ The core async API needs no optional extra. Install `fpstreams[async]` when you
288
+ also need the `aiofiles`-based async file adapters.
289
+
231
290
  ~~~python
232
291
  import asyncio
233
292
 
@@ -246,7 +305,8 @@ async def main() -> None:
246
305
  .filter(lambda value: value >= 20)
247
306
  .to_list()
248
307
  )
249
- assert result == [20, 30, 40]
308
+ print(result)
309
+ # [20, 30, 40]
250
310
 
251
311
 
252
312
  asyncio.run(main())
@@ -273,10 +333,10 @@ assert totals == {
273
333
 
274
334
  ## Execution engines
275
335
 
276
- The default `auto` engine chooses among Python, native Rust, Arrow-native
277
- prefixes, and hybrid execution. Relational plans may also use guarded native
278
- subpaths while retaining their canonical Python fallback. Pass the terminal you
279
- intend to call to `explain()` so its answer matches execution:
336
+ The default `auto` engine chooses Python, Rust, Arrow, NumPy, or a combination
337
+ for each supported plan. Relational plans may use native shortcuts while keeping
338
+ a Python fallback. Pass the terminal you intend to call to `explain()` so it
339
+ can include that terminal in its planning decisions:
280
340
 
281
341
  ~~~python
282
342
  from fpstreams import flow, item
@@ -300,16 +360,27 @@ python_result = pipeline.with_engine("python").to_list()
300
360
  native_result = pipeline.with_engine("native").to_list()
301
361
  ~~~
302
362
 
363
+ Use `run_with_report()` to execute a terminal and inspect its recorded route.
364
+ Version 2.2.0 adds Pairs terminals such as
365
+ `flow([("a", 1)]).pairs().run_with_report("group_values")`. See the
366
+ [execution report guide](https://steventimes.github.io/fpstreams/user-guide/execution-reports/)
367
+ for supported terminals and the limits of route reporting.
368
+
303
369
  A forced native plan raises `NativeUnsupportedError` if its complete types or
304
370
  operations cannot run natively. In particular, the presence of an internal
305
371
  native relational specialization does not make the complete relation eligible
306
372
  for `with_engine("native")`. An unsupported forced relational plan fails before
307
373
  claiming its one-shot sources; `auto` selects a legal fallback path.
308
374
 
309
- For an unchanged list or tuple, automatic `list`, `sum`, and `count` terminals
310
- stay in Python instead of scanning and copying the container into Rust. Numeric
311
- range reductions can still use Rust. `count()` uses a known exact size in O(1)
312
- when no operation changes cardinality and the source is safely reiterable.
375
+ [`run_with_report()`](https://steventimes.github.io/fpstreams/user-guide/execution-reports/)
376
+ returns a terminal's value, recorded route, and query-owned resource counts in
377
+ one execution. It records the outer plan and some direct paths; it does not
378
+ identify every internal kernel or runtime fallback.
379
+
380
+ Identity list and tuple materialization stays in Python. Large, supported integer
381
+ sums and numeric range reductions may use Rust in `auto` mode. `count()` uses a
382
+ known exact size in O(1) when no operation changes cardinality and the source is
383
+ safely reiterable. Use `explain(terminal=...)` to inspect the planned route.
313
384
 
314
385
  ## Resource and file-safety controls
315
386