fpstreams 2.0.0__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (239) hide show
  1. fpstreams-2.2.0/CHANGELOG.md +323 -0
  2. fpstreams-2.2.0/PKG-INFO +467 -0
  3. fpstreams-2.2.0/README.md +411 -0
  4. {fpstreams-2.0.0 → fpstreams-2.2.0}/pyproject.toml +50 -9
  5. {fpstreams-2.0.0 → fpstreams-2.2.0}/rust/Cargo.lock +12 -12
  6. {fpstreams-2.0.0 → fpstreams-2.2.0}/rust/Cargo.toml +8 -2
  7. fpstreams-2.2.0/rust/build.rs +3 -0
  8. fpstreams-2.2.0/rust/src/common.rs +706 -0
  9. fpstreams-2.2.0/rust/src/float/affine_pair.rs +192 -0
  10. fpstreams-2.2.0/rust/src/float/endpoints.rs +594 -0
  11. fpstreams-2.2.0/rust/src/float.rs +972 -0
  12. fpstreams-2.2.0/rust/src/integer/endpoints.rs +972 -0
  13. fpstreams-2.2.0/rust/src/integer.rs +892 -0
  14. fpstreams-2.2.0/rust/src/lib.rs +245 -0
  15. fpstreams-2.2.0/rust/src/numeric_mean.rs +653 -0
  16. fpstreams-2.2.0/rust/src/numpy_export.rs +41 -0
  17. fpstreams-2.2.0/rust/src/numpy_group/buffer.rs +519 -0
  18. fpstreams-2.2.0/rust/src/numpy_group.rs +849 -0
  19. fpstreams-2.2.0/rust/src/pair/expr.rs +137 -0
  20. fpstreams-2.2.0/rust/src/pair/prefix.rs +131 -0
  21. fpstreams-2.2.0/rust/src/pair/row_filter.rs +82 -0
  22. fpstreams-2.2.0/rust/src/pair/unique.rs +212 -0
  23. fpstreams-2.2.0/rust/src/pair/value_filter.rs +149 -0
  24. fpstreams-2.2.0/rust/src/pair/value_map.rs +184 -0
  25. fpstreams-2.2.0/rust/src/pair.rs +21 -0
  26. fpstreams-2.2.0/rust/src/pivot.rs +710 -0
  27. fpstreams-2.2.0/rust/src/records.rs +191 -0
  28. fpstreams-2.2.0/rust/src/relational/adapters/namedtuple.rs +1075 -0
  29. fpstreams-2.2.0/rust/src/relational/adapters.rs +314 -0
  30. fpstreams-2.2.0/rust/src/relational/global_numeric.rs +133 -0
  31. fpstreams-2.2.0/rust/src/relational/group_numeric.rs +1067 -0
  32. fpstreams-2.2.0/rust/src/relational/group_pair_expr.rs +316 -0
  33. fpstreams-2.2.0/rust/src/relational/join_callable/many.rs +428 -0
  34. fpstreams-2.2.0/rust/src/relational/join_callable/unique.rs +352 -0
  35. fpstreams-2.2.0/rust/src/relational/join_callable.rs +1041 -0
  36. fpstreams-2.2.0/rust/src/relational/join_exact.rs +696 -0
  37. fpstreams-2.2.0/rust/src/relational/join_i64.rs +585 -0
  38. fpstreams-2.2.0/rust/src/relational.rs +222 -0
  39. fpstreams-2.2.0/rust/src/relational_fixed/composite.rs +186 -0
  40. fpstreams-2.2.0/rust/src/relational_fixed/global_multi/same_field.rs +469 -0
  41. fpstreams-2.2.0/rust/src/relational_fixed/global_multi.rs +1296 -0
  42. fpstreams-2.2.0/rust/src/relational_fixed/group_multi.rs +1026 -0
  43. fpstreams-2.2.0/rust/src/relational_fixed/single.rs +565 -0
  44. fpstreams-2.2.0/rust/src/relational_fixed.rs +675 -0
  45. fpstreams-2.2.0/rust/src/scalar_sort.rs +443 -0
  46. fpstreams-2.2.0/rust/src/scalar_unique.rs +422 -0
  47. fpstreams-2.2.0/rust/src/select.rs +643 -0
  48. fpstreams-2.2.0/rust/src/tests/adapters.rs +484 -0
  49. fpstreams-2.2.0/rust/src/tests/group_exact/pairs.rs +574 -0
  50. fpstreams-2.2.0/rust/src/tests/group_exact/rows.rs +708 -0
  51. fpstreams-2.2.0/rust/src/tests/group_exact.rs +101 -0
  52. fpstreams-2.2.0/rust/src/tests/group_fixed.rs +808 -0
  53. fpstreams-2.2.0/rust/src/tests/join_callable/behavior.rs +836 -0
  54. fpstreams-2.2.0/rust/src/tests/join_callable/direct.rs +236 -0
  55. fpstreams-2.2.0/rust/src/tests/join_callable.rs +986 -0
  56. fpstreams-2.2.0/rust/src/tests/join_exact.rs +916 -0
  57. fpstreams-2.2.0/rust/src/tests/numeric/buffers.rs +691 -0
  58. fpstreams-2.2.0/rust/src/tests/numeric/kernels.rs +1036 -0
  59. fpstreams-2.2.0/rust/src/tests/numeric/reductions.rs +436 -0
  60. fpstreams-2.2.0/rust/src/tests/numeric.rs +485 -0
  61. fpstreams-2.2.0/rust/src/tests/numpy_group.rs +718 -0
  62. fpstreams-2.2.0/rust/src/tests.rs +459 -0
  63. fpstreams-2.2.0/rust/src/unnest.rs +236 -0
  64. fpstreams-2.2.0/rust/src/unpivot.rs +421 -0
  65. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/__init__.py +30 -4
  66. fpstreams-2.2.0/src/fpstreams/_native.pyi +707 -0
  67. fpstreams-2.2.0/src/fpstreams/_provenance.py +200 -0
  68. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/aggregate.py +1 -1
  69. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/async_flow.py +1 -1
  70. fpstreams-2.2.0/src/fpstreams/collecting/__init__.py +31 -0
  71. fpstreams-2.2.0/src/fpstreams/collecting/_collector_base.py +128 -0
  72. fpstreams-2.2.0/src/fpstreams/collecting/aggregate_program.py +156 -0
  73. fpstreams-2.2.0/src/fpstreams/collecting/aggregation.py +1201 -0
  74. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/collecting/collector.py +241 -149
  75. fpstreams-2.2.0/src/fpstreams/collecting/program.py +288 -0
  76. fpstreams-2.2.0/src/fpstreams/collecting/reducer.py +245 -0
  77. fpstreams-2.2.0/src/fpstreams/collecting/statistics.py +194 -0
  78. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/collectors.py +1 -1
  79. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/column.py +1 -1
  80. fpstreams-2.2.0/src/fpstreams/errors.py +33 -0
  81. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/exceptions.py +4 -0
  82. fpstreams-2.2.0/src/fpstreams/execution/__init__.py +320 -0
  83. fpstreams-2.2.0/src/fpstreams/execution/_pair_aggregate.py +68 -0
  84. fpstreams-2.2.0/src/fpstreams/execution/_pair_dict.py +1363 -0
  85. fpstreams-2.2.0/src/fpstreams/execution/_pair_row_filter.py +321 -0
  86. fpstreams-2.2.0/src/fpstreams/execution/_rows_fusion.py +975 -0
  87. fpstreams-2.2.0/src/fpstreams/execution/_scalar_fusion.py +251 -0
  88. fpstreams-2.2.0/src/fpstreams/execution/arrow.py +1778 -0
  89. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/execution/async_iterators.py +165 -143
  90. fpstreams-2.2.0/src/fpstreams/execution/async_map.py +92 -0
  91. fpstreams-2.2.0/src/fpstreams/execution/async_merge.py +386 -0
  92. fpstreams-2.2.0/src/fpstreams/execution/async_ops.py +93 -0
  93. fpstreams-2.2.0/src/fpstreams/execution/async_prefetch.py +105 -0
  94. fpstreams-2.2.0/src/fpstreams/execution/async_queue.py +94 -0
  95. fpstreams-2.2.0/src/fpstreams/execution/async_scheduler.py +186 -0
  96. fpstreams-2.2.0/src/fpstreams/execution/async_timers.py +287 -0
  97. fpstreams-2.2.0/src/fpstreams/execution/native.py +559 -0
  98. fpstreams-2.2.0/src/fpstreams/execution/numpy_group.py +1359 -0
  99. fpstreams-2.2.0/src/fpstreams/execution/numpy_prefix.py +703 -0
  100. fpstreams-2.2.0/src/fpstreams/execution/physical.py +346 -0
  101. fpstreams-2.2.0/src/fpstreams/execution/relational/__init__.py +2110 -0
  102. fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_global.py +982 -0
  103. fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_group.py +1001 -0
  104. fpstreams-2.2.0/src/fpstreams/execution/relational/arrow_group_rows.py +183 -0
  105. fpstreams-2.2.0/src/fpstreams/execution/relational/join.py +550 -0
  106. fpstreams-2.2.0/src/fpstreams/execution/sorted_streams.py +407 -0
  107. fpstreams-2.2.0/src/fpstreams/execution/sorting.py +760 -0
  108. fpstreams-2.2.0/src/fpstreams/execution/sync.py +417 -0
  109. fpstreams-2.2.0/src/fpstreams/execution/sync_ops.py +577 -0
  110. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/expr.py +1 -1
  111. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/expressions/__init__.py +1 -1
  112. fpstreams-2.2.0/src/fpstreams/expressions/_codegen.py +29 -0
  113. fpstreams-2.2.0/src/fpstreams/expressions/_row_codegen.py +246 -0
  114. fpstreams-2.2.0/src/fpstreams/expressions/program.py +194 -0
  115. fpstreams-2.2.0/src/fpstreams/expressions/row.py +363 -0
  116. fpstreams-2.2.0/src/fpstreams/expressions/row_eval.py +441 -0
  117. fpstreams-2.2.0/src/fpstreams/expressions/row_ir.py +269 -0
  118. fpstreams-2.2.0/src/fpstreams/expressions/scalar.py +787 -0
  119. fpstreams-2.2.0/src/fpstreams/expressions/selectors.py +124 -0
  120. fpstreams-2.2.0/src/fpstreams/expressions/typed_ir.py +111 -0
  121. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/functional.py +32 -19
  122. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/gather.py +1 -1
  123. fpstreams-2.2.0/src/fpstreams/io_safety.py +97 -0
  124. fpstreams-2.2.0/src/fpstreams/option.py +5 -0
  125. fpstreams-2.2.0/src/fpstreams/physical/__init__.py +13 -0
  126. fpstreams-2.2.0/src/fpstreams/physical/async_plan.py +170 -0
  127. fpstreams-2.2.0/src/fpstreams/physical/compiled.py +201 -0
  128. fpstreams-2.2.0/src/fpstreams/physical/kernel_cache.py +45 -0
  129. fpstreams-2.2.0/src/fpstreams/physical/plan.py +94 -0
  130. fpstreams-2.2.0/src/fpstreams/physical/relational.py +377 -0
  131. fpstreams-2.2.0/src/fpstreams/planning/__init__.py +1 -0
  132. fpstreams-2.2.0/src/fpstreams/planning/_pair_stages.py +55 -0
  133. fpstreams-2.2.0/src/fpstreams/planning/arrow.py +424 -0
  134. fpstreams-2.2.0/src/fpstreams/planning/arrow_source.py +70 -0
  135. fpstreams-2.2.0/src/fpstreams/planning/async_.py +501 -0
  136. fpstreams-2.2.0/src/fpstreams/planning/async_utils.py +79 -0
  137. fpstreams-2.2.0/src/fpstreams/planning/compiler.py +1684 -0
  138. fpstreams-2.2.0/src/fpstreams/planning/explain.py +454 -0
  139. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/planning/gather.py +52 -48
  140. fpstreams-2.2.0/src/fpstreams/planning/logical.py +234 -0
  141. fpstreams-2.2.0/src/fpstreams/planning/native.py +1047 -0
  142. fpstreams-2.2.0/src/fpstreams/planning/numpy.py +490 -0
  143. fpstreams-2.2.0/src/fpstreams/planning/pair_i64_expression.py +561 -0
  144. fpstreams-2.2.0/src/fpstreams/planning/plan_cache.py +64 -0
  145. fpstreams-2.2.0/src/fpstreams/planning/semantic_analyzer.py +168 -0
  146. fpstreams-2.2.0/src/fpstreams/planning/semantic_rules.py +744 -0
  147. fpstreams-2.2.0/src/fpstreams/planning/semantics.py +317 -0
  148. fpstreams-2.2.0/src/fpstreams/planning/source.py +378 -0
  149. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/planning/sync.py +69 -23
  150. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/primitives/__init__.py +1 -1
  151. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/primitives/option.py +48 -35
  152. fpstreams-2.2.0/src/fpstreams/primitives/result.py +327 -0
  153. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/result.py +1 -1
  154. fpstreams-2.2.0/src/fpstreams/runtime/__init__.py +25 -0
  155. fpstreams-2.2.0/src/fpstreams/runtime/_distinct.py +35 -0
  156. fpstreams-2.2.0/src/fpstreams/runtime/failpoints.py +42 -0
  157. fpstreams-2.2.0/src/fpstreams/runtime/files.py +147 -0
  158. fpstreams-2.2.0/src/fpstreams/runtime/iterators.py +58 -0
  159. fpstreams-2.2.0/src/fpstreams/runtime/limits.py +19 -0
  160. fpstreams-2.2.0/src/fpstreams/runtime/metrics.py +16 -0
  161. fpstreams-2.2.0/src/fpstreams/runtime/query.py +98 -0
  162. fpstreams-2.2.0/src/fpstreams/runtime/report.py +262 -0
  163. fpstreams-2.2.0/src/fpstreams/runtime/resources.py +253 -0
  164. fpstreams-2.2.0/src/fpstreams/runtime/spill.py +47 -0
  165. fpstreams-2.2.0/src/fpstreams/runtime/tasks.py +338 -0
  166. fpstreams-2.2.0/src/fpstreams/storage/__init__.py +14 -0
  167. fpstreams-2.2.0/src/fpstreams/storage/codec.py +112 -0
  168. fpstreams-2.2.0/src/fpstreams/storage/spill_store.py +295 -0
  169. fpstreams-2.2.0/src/fpstreams/streams/_async_identity.py +102 -0
  170. fpstreams-2.2.0/src/fpstreams/streams/_flow_structural_list.py +1176 -0
  171. fpstreams-2.2.0/src/fpstreams/streams/_flow_unique_list.py +551 -0
  172. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/async_flow.py +421 -169
  173. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/async_terminals.py +312 -103
  174. fpstreams-2.2.0/src/fpstreams/streams/flow.py +1619 -0
  175. fpstreams-2.2.0/src/fpstreams/streams/flow_terminals.py +2917 -0
  176. fpstreams-2.2.0/src/fpstreams/streams/pairs.py +507 -0
  177. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/__init__.py +2 -1
  178. fpstreams-2.2.0/src/fpstreams/tabular/_text_sources.py +447 -0
  179. fpstreams-2.2.0/src/fpstreams/tabular/arrow.py +1035 -0
  180. fpstreams-2.2.0/src/fpstreams/tabular/dataframe.py +109 -0
  181. fpstreams-2.2.0/src/fpstreams/tabular/factory.py +333 -0
  182. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/grouped.py +37 -44
  183. fpstreams-2.2.0/src/fpstreams/tabular/io.py +759 -0
  184. fpstreams-2.2.0/src/fpstreams/tabular/join.py +944 -0
  185. fpstreams-2.2.0/src/fpstreams/tabular/numpy.py +687 -0
  186. fpstreams-2.2.0/src/fpstreams/tabular/polars.py +134 -0
  187. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/records.py +25 -1
  188. fpstreams-2.2.0/src/fpstreams/tabular/rows.py +2883 -0
  189. fpstreams-2.2.0/src/fpstreams/tabular/spill.py +1099 -0
  190. fpstreams-2.2.0/src/fpstreams/tabular/spill_io.py +469 -0
  191. fpstreams-2.2.0/src/fpstreams/tabular/spill_limits.py +121 -0
  192. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/sql.py +64 -25
  193. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/tabular/sqlite_sink.py +30 -13
  194. fpstreams-2.2.0/src/fpstreams/testing.py +110 -0
  195. fpstreams-2.0.0/PKG-INFO +0 -294
  196. fpstreams-2.0.0/README.md +0 -249
  197. fpstreams-2.0.0/rust/src/common.rs +0 -64
  198. fpstreams-2.0.0/rust/src/float.rs +0 -519
  199. fpstreams-2.0.0/rust/src/integer.rs +0 -511
  200. fpstreams-2.0.0/rust/src/lib.rs +0 -47
  201. fpstreams-2.0.0/rust/src/tests.rs +0 -225
  202. fpstreams-2.0.0/src/fpstreams/_native.pyi +0 -124
  203. fpstreams-2.0.0/src/fpstreams/collecting/__init__.py +0 -13
  204. fpstreams-2.0.0/src/fpstreams/collecting/aggregation.py +0 -418
  205. fpstreams-2.0.0/src/fpstreams/collecting/statistics.py +0 -68
  206. fpstreams-2.0.0/src/fpstreams/errors.py +0 -29
  207. fpstreams-2.0.0/src/fpstreams/execution/__init__.py +0 -132
  208. fpstreams-2.0.0/src/fpstreams/execution/async_.py +0 -64
  209. fpstreams-2.0.0/src/fpstreams/execution/async_concurrency.py +0 -459
  210. fpstreams-2.0.0/src/fpstreams/execution/async_ops.py +0 -188
  211. fpstreams-2.0.0/src/fpstreams/execution/native.py +0 -93
  212. fpstreams-2.0.0/src/fpstreams/execution/sorting.py +0 -121
  213. fpstreams-2.0.0/src/fpstreams/execution/sync.py +0 -75
  214. fpstreams-2.0.0/src/fpstreams/execution/sync_ops.py +0 -467
  215. fpstreams-2.0.0/src/fpstreams/expressions/row.py +0 -337
  216. fpstreams-2.0.0/src/fpstreams/expressions/scalar.py +0 -377
  217. fpstreams-2.0.0/src/fpstreams/expressions/selectors.py +0 -42
  218. fpstreams-2.0.0/src/fpstreams/option.py +0 -5
  219. fpstreams-2.0.0/src/fpstreams/planning/__init__.py +0 -1
  220. fpstreams-2.0.0/src/fpstreams/planning/async_.py +0 -316
  221. fpstreams-2.0.0/src/fpstreams/planning/async_utils.py +0 -43
  222. fpstreams-2.0.0/src/fpstreams/planning/explain.py +0 -84
  223. fpstreams-2.0.0/src/fpstreams/planning/native.py +0 -334
  224. fpstreams-2.0.0/src/fpstreams/planning/source.py +0 -77
  225. fpstreams-2.0.0/src/fpstreams/primitives/result.py +0 -314
  226. fpstreams-2.0.0/src/fpstreams/streams/flow.py +0 -1112
  227. fpstreams-2.0.0/src/fpstreams/streams/flow_terminals.py +0 -864
  228. fpstreams-2.0.0/src/fpstreams/streams/pairs.py +0 -343
  229. fpstreams-2.0.0/src/fpstreams/tabular/arrow.py +0 -316
  230. fpstreams-2.0.0/src/fpstreams/tabular/dataframe.py +0 -54
  231. fpstreams-2.0.0/src/fpstreams/tabular/factory.py +0 -228
  232. fpstreams-2.0.0/src/fpstreams/tabular/io.py +0 -332
  233. fpstreams-2.0.0/src/fpstreams/tabular/join.py +0 -479
  234. fpstreams-2.0.0/src/fpstreams/tabular/polars.py +0 -71
  235. fpstreams-2.0.0/src/fpstreams/tabular/rows.py +0 -940
  236. fpstreams-2.0.0/src/fpstreams/tabular/spill.py +0 -370
  237. {fpstreams-2.0.0 → fpstreams-2.2.0}/LICENSE +0 -0
  238. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/py.typed +0 -0
  239. {fpstreams-2.0.0 → fpstreams-2.2.0}/src/fpstreams/streams/__init__.py +0 -0
@@ -0,0 +1,323 @@
1
+ # Changelog
2
+
3
+ This file records user-visible and compatibility-relevant changes in fpstreams 2,
4
+ including changed defaults.
5
+
6
+ ## Unreleased
7
+
8
+ ## 2.2.0 - 2026-10-06
9
+
10
+ ### Added
11
+
12
+ - Add bounded `Rows.join_sorted()` for explicitly ordered records, with consumed-prefix
13
+ validation and first-right-record schema rules.
14
+ - Report explicit sorted group, merge, and join execution through existing report fields.
15
+ - Add optional atomic path output to Flow CSV/JSON/JSONL and Rows CSV/JSONL sinks,
16
+ including no-overwrite publication. Existing direct output remains the default.
17
+ - `Flow.to_jsonl()` streams arbitrary JSON values as one value per line.
18
+ `Rows.to_jsonl()` now accepts `default` for custom serialization.
19
+
20
+ - `Flow.merge_sorted()` stably merges two ascending inputs without sorting or
21
+ materializing them. Ties prefer the left input and outputs retain their identity.
22
+
23
+ - `Rows.group_by_sorted()` aggregates adjacent ascending keys in Python with
24
+ current group state and one lookahead row. Consumed keys are checked for exact
25
+ builtin types and order; this explicit mode does not sort or support spill.
26
+
27
+ - `Pairs.run_with_report()` executes a pair terminal once and returns its value
28
+ with an execution report. It supports `to_dict`, `group_values`,
29
+ `collect_values`, and `aggregate_values`.
30
+
31
+ ### Fixed
32
+
33
+ - Rust source builds support the declared 1.85 minimum. Native kernels no longer
34
+ require Rust 1.88 through let-chain syntax; CI checks the minimum compiler.
35
+ - SQLite sinks match existing table, view, and column names using SQLite's ASCII
36
+ case rules. Duplicate column aliases raise `DuplicateKeyError` before inserts.
37
+ Fail mode and view rejection check the destination before reading rows.
38
+ - Sync and async flows accept built-in ranges longer than `sys.maxsize`. Bounded
39
+ reads and exact counts no longer fail during source size detection.
40
+
41
+ - `spreadsheet_safe=True` also neutralizes formula-like CSV headers, including
42
+ inferred record field names and explicitly named empty output. Record lookup
43
+ still uses the original names; the raw-output default is unchanged.
44
+ - Relative Parquet output paths stay anchored to the initial working directory
45
+ if a source or conversion callback changes directories during the write.
46
+
47
+ - Scalar keys in explicit sorted operations avoid temporary shape tuples while
48
+ retaining per-row type and ordering checks.
49
+
50
+ - Atomic CSV and JSON output keeps its original destination when a source or
51
+ serializer changes the working directory during execution.
52
+
53
+ - CSV and SQLite record sinks propagate `StopIteration` from first-record
54
+ conversion instead of treating it as empty input. SQLite replacement leaves
55
+ the existing table intact when that conversion fails.
56
+
57
+ - Sync and async `chunk`, `window`, and `batch_by_size` validate integer bounds
58
+ before execution, including their aliases. Floats, NaN, and infinity now raise
59
+ `TypeError` instead of failing after source consumption or leaving async batches
60
+ unbounded. Objects implementing `__index__` are normalized once per bound.
61
+
62
+ - `unique()`, `unique_by()`, and `Pairs.unique_keys()` propagate equality errors
63
+ instead of treating them as unhashable keys. Hash callbacks keep their existing
64
+ lookup and insertion counts.
65
+ - Async uniqueness, `agg.count_distinct()`, and native pair-uniqueness continuation
66
+ also propagate equality errors without extra hash calls. Iterator cleanup retains
67
+ nested exception notes when several owned resources fail to close.
68
+ - Distinct operations avoid key-wrapper allocations for exact built-in strings.
69
+ String subclasses and custom keys retain guarded hash and equality handling.
70
+ - Grouped reductions and frequency counts propagate `KeyError` raised by a key's
71
+ hash or equality method instead of treating it as a missing group. This includes
72
+ async reductions, grouping collectors, Pairs, and spilled grouping.
73
+ - Parquet `if_exists="error"` publishes with an atomic no-overwrite hard link.
74
+ A concurrent creator or dangling symlink cannot be overwritten. Filesystems
75
+ without hard-link support report an error; `replace` still uses atomic rename.
76
+ - Owned Arrow resources report close failures on successful queries and attach
77
+ cleanup diagnostics to an existing query error. Cleanup attempts every owned
78
+ resource. This also changes Arrow `first()`, which previously ignored close errors.
79
+
80
+ - Engine benchmarks calibrate warmed timing blocks for short tasks and first-row
81
+ latency, retaining elapsed time and call counts in their reports. Rebuild
82
+ single-call baselines before comparing. Timing and resource regression limits
83
+ are unchanged; scheduled CI now uploads the three raw reference reports.
84
+
85
+ - Benchmark report schema 6 records Python and glibc allocator environment
86
+ settings. Baseline creation and comparison reject missing or mismatched
87
+ settings; regenerate older reports. The runners do not change allocator defaults.
88
+
89
+ - NumPy frequency benchmarks check key types, integer counts, first-key order,
90
+ and floating-point bit patterns. Separate NaN entries, their signs, and their
91
+ payloads remain distinct when comparing independently computed outputs.
92
+ - Generated Python `with_columns` loops call the live enrichment function.
93
+ Changes to captured accessors, expression evaluators, or the selector list
94
+ remain visible after plan caching and while reading the source. The transform
95
+ still copies the row before invoking selectors against the original input.
96
+
97
+ - Arrow file projections keep selector changes made by custom scan openers
98
+ visible. Parquet rechecks the projection after creating its dataset, including
99
+ with an explicit source filter. CSV keeps full fields when its public reader
100
+ hooks differ from the extension entrypoints. Normal CSV projection retains
101
+ the bounded schema probe and the default data reader.
102
+ - `Rows.select()` rechecks captured field accessors before using Rust, NumPy,
103
+ or Arrow projection metadata. The generated Python loop calls the live
104
+ projection, so changes made during input reads remain visible. Custom Arrow
105
+ batch sources keep fields available for fallback; changed projections can
106
+ infer their output dtype instead of forcing the former field's schema.
107
+ NumPy fallback finishes the already-opened source without reopening it.
108
+ - Pivot calls the live index, column, and value selectors on its Python path.
109
+ Direct dict lookups had ignored changes to selector code, closure bindings,
110
+ and globals. Native admission now checks those bindings before execution;
111
+ its fallback shares the general selector loop.
112
+ - Native pivot falls back when Rows/Flow iteration or source hooks change,
113
+ including changes to function code. Replacement results, exceptions, and
114
+ iterator cleanup remain observable.
115
+ - `frequencies(key=...)` uses the Python pipeline for sequential `auto` plans,
116
+ so key callbacks can affect subsequent input reads. Native materialization
117
+ could hide those changes. Reports identify this route as `python_frequency`.
118
+ - Benchmark speedup requirements for composite count/sum groups and NamedTuple
119
+ callable joins now apply from 1,000 rows. Smaller inputs still emit timing and
120
+ allocation data for cross-run checks. The original ratios and CI workloads
121
+ are unchanged; small runs no longer inherit speedup requirements validated
122
+ for larger inputs.
123
+ - The engine benchmark CLI lists scenarios without running their tasks, timing,
124
+ allocation tracking, or execution observation. Listing and measurement share
125
+ filtering and fixture cleanup, including when a later scenario builder fails.
126
+ - Single-collector grouping preserves the general collector program's state
127
+ release order, including when output is closed early. Unused input keys are
128
+ released before reading the current finisher, so changes made by their release
129
+ callbacks take effect.
130
+ - Single-collector grouping uses the current step after truth-testing a custom
131
+ completion result. It also releases replaced completion values before pulling
132
+ another row, matching the general collector program's callback behavior.
133
+ - Python grouping calls the live key selector for each row. A field shortcut
134
+ could keep using an old field after a source or callback changed the selector's
135
+ code or closure, merging distinct groups. Selector globals and error types
136
+ now follow the same function calls as general grouped aggregation.
137
+ - Scalar caches no longer confuse numerically equal constants of different
138
+ types. Directly constructed expressions retain their Python result types,
139
+ wide integer values, and custom constant identities. Expressions with
140
+ nonstandard operand types use Python in `auto` mode; forcing `native` raises
141
+ `NativeUnsupportedError` instead of changing their numeric representation.
142
+ - Float expressions retain the sign of `-0.0` in their instructions and evaluator
143
+ cache. Compiling a positive-zero expression first no longer changes a later
144
+ negative-zero result. Scalar and Pairs execution keep their existing routes.
145
+ - Python grouped sums call the live collector lifecycle throughout traversal
146
+ and output. Changes to function code or selector closure cells made by a source,
147
+ key callback, or output consumer are no longer hidden by an inlined sum loop.
148
+ - Two-key count/sum groups also use the live collector program. Their separate
149
+ loop skipped changes to initializer and step code or the sum selector closure
150
+ made while reading the source, even though the same collectors worked correctly
151
+ in other group layouts.
152
+ - Grouped aggregation uses the general collector program for custom lifecycle
153
+ getters. A dynamic `step` property or `__getattribute__` override is read for
154
+ each step instead of being cached as a fixed function.
155
+ - Single-collector grouping no longer caches a temporary lifecycle hook between
156
+ key selection and lookup. If hashing replaces that hook, its release callback
157
+ runs before the next hash call, matching the general collector program.
158
+ - Exact builtin checks no longer use custom metaclass equality to infer source
159
+ size or select grouping, spill, range lookup, and float-expression shortcuts.
160
+ Custom iterables are counted by traversal; map/filter fusion does not request
161
+ their length hints. Grouping retains custom key and serialization calls.
162
+ - Execution reports distinguish successful top-level Rust and Arrow record
163
+ joins from Python joins. Native pair aggregation also records its direct route.
164
+ - Benchmark comparisons reject missing provenance and mismatched workloads.
165
+ Both suites record dependencies, Git state, code and workload fingerprints,
166
+ and an untimed observation of the task. Median baselines retain each run's
167
+ provenance. Older reports need to be regenerated for report schema 6.
168
+ - Reports include CPU affinity, NumPy CPU dispatch, and an allowlist of runtime
169
+ settings. Comparisons reject mismatched configurations; both runners also
170
+ reject configuration changes during measurement.
171
+ - Join benchmarks honor the requested engine. Twelve Python controls separate
172
+ dict and Mapping records, field and callable keys, and inner, left, and repeated
173
+ matches. References preserve callable-key columns and snapshots taken before
174
+ key selection, including unmatched left rows.
175
+ - NumPy frequency benchmarks cover integer and float arrays, with and without
176
+ a callable key, at low and high cardinality. All references preserve first-key
177
+ order; Python array-to-list conversion is included in its timed task.
178
+ - Competitive samples warm each task for at least 1ms after GC and record the
179
+ number of warmup calls. One warmup call left inconsistent timings for short
180
+ NumPy tasks in local measurements.
181
+ - Competitive benchmarks now record peak Python allocation in a separate,
182
+ untimed call for every implementation, using the engine suite's shared helper.
183
+ Schema 6 rejects missing or invalid resource measurements, and baseline
184
+ creation rejects inconsistent resource sets instead of substituting zero.
185
+ Earlier competitive reports contain no allocation evidence.
186
+ - Regression checks accept zero allocated bytes when both runs report zero.
187
+ - `frequencies()` on retained lists and tuples falls back to Python when the
188
+ optional Rust extension is unavailable, including in the browser wheel.
189
+ - Frequency benchmarks use bulk conversion for NumPy dictionaries and pandas'
190
+ `to_dict()`, avoiding per-key normalization overhead in the reference tasks.
191
+
192
+ ### Changed
193
+
194
+ - Automatic NumPy `frequencies()` without a key can count a selected native
195
+ stream in Rust through its existing iterator. Source fallback and query cleanup
196
+ stay in the physical executor. Custom keys return to Python before hashing;
197
+ old wheels and free-threaded CPython retain the prior counting path.
198
+ - Python `Rows.select()` keeps its per-row selector snapshot as a list, avoiding
199
+ an intermediate tuple conversion. Selector edits still affect subsequent rows,
200
+ and replaced selectors stay alive until the current row finishes.
201
+
202
+ - Python scalar iteration over an exact NumPy ndarray checks its live size
203
+ without creating a shape tuple for every value. Dimension and length checks,
204
+ lazy reads, and the fallback for custom array objects remain in place.
205
+ - Python joins reduce layout-checking overhead for left records with more than
206
+ four fields. Repeated private dict snapshots need no temporary field-name tuple
207
+ or Python comparison generator. Field identity, short-circuiting, suffix keys,
208
+ snapshot lifetimes, and the bounded layout cache retain their existing behavior.
209
+ - Two-key count/sum groups can run in Rust with `engine="auto"` when a retained
210
+ list or tuple contains exact tuple rows and both keys and the selected values
211
+ are plain signed 64-bit integers. The path preserves first-key identities,
212
+ encounter order, and wider sums. Changed functions, one-shot sources, and
213
+ unsupported types use the Python collector program; older extensions can
214
+ decline the optional entry point.
215
+ - `frequencies()` can continue counting in Rust after its bounded integer
216
+ prefix. The continuation supports exact builtin integer, boolean, string,
217
+ bytes, float, and `None` keys in retained lists and tuples on GIL-enabled
218
+ CPython. It updates the output dictionary directly and hands custom keys
219
+ back to Python before calling their hash or equality methods.
220
+ - Frequency execution reports record a completed native count as `rust_direct`
221
+ and a count completed by Python after a native prefix as `python_frequency`.
222
+
223
+ ### Internal
224
+
225
+ - Browser wheels record checkout identity and working-tree state. Release labels
226
+ require a clean checkout matching the version tag; dirty or unknown builds
227
+ remain development builds. The playground displays this provenance.
228
+ - Release smoke checks exercise paid-order aggregation and bounded async mapping
229
+ in addition to Python/native integer sums.
230
+
231
+ - Single-collector groups write step results directly to their stored entries
232
+ and avoid reading stored state for completed groups. Group benchmarks now
233
+ include `agg.first()` with repeated and distinct keys.
234
+ - Group benchmarks now cover custom completion predicates with repeated and
235
+ distinct keys, alongside collectors that cannot finish early.
236
+ - Generated scalar expressions, row expressions, and fused loops share a smaller
237
+ AST location pass. It preserves traversal order and the existing synthetic
238
+ source positions used in tracebacks.
239
+ - Scalar program fingerprints use fewer intermediate records while retaining
240
+ the binary framing format, arbitrary-size integers, and float payload bits.
241
+ - Scalar fingerprint encoding reuses fixed instruction headers instead of
242
+ rebuilding their bytes for every query. Expression reads and type checks are
243
+ unchanged; the table retains no expressions or source data.
244
+ - Planning benchmarks separate compilation from execution for integer and float
245
+ expressions and a callable control, using the same pipeline for each pair.
246
+ - Python grouped output uses one fewer forwarding generator. The collector
247
+ iterator still starts lazily and retains its existing cleanup path.
248
+ - Collector lifecycle properties read their original slots directly, removing
249
+ an extra Python wrapper. Frozen assignment, explicit replacement, and deletion
250
+ errors are preserved. The API manifest now classifies those existing fields
251
+ as properties; their names and constructor signatures are unchanged.
252
+ - Add Python dict-group benchmarks for field and callable selectors with repeated
253
+ and distinct keys. Execution observations record the case's requested engine.
254
+ - Include nominal Mapping and mappingproxy field-group cases with repeated and
255
+ distinct keys when measuring Python selector changes.
256
+ - Extend the two-key count/sum benchmarks to repeated keys. Keep the original
257
+ direct-versus-callable timing threshold and report any regression against it.
258
+ - Move the narrow integer-key record-join ABI adapter into the existing join
259
+ executor module. Kernel order, shape guards, and fallback behavior are unchanged.
260
+
261
+ ### Documentation
262
+
263
+ - Repair split return descriptions and unresolved API references in generated
264
+ pages. The two `partition_results()` methods now describe their success and
265
+ exception lists separately.
266
+ - Correct the DB-API example's `batch_size` argument and the distinction between
267
+ `Rows.skip()` and `Rows.drop()` in the API index.
268
+ - Explain the difference between a compiled outer plan and a recorded direct
269
+ terminal route, including the limits of reports for compound queries.
270
+ - Revise the website's introductory and performance text and document the next
271
+ benchmark and refactoring steps in the roadmap.
272
+
273
+ ## 2.1.0 - 2026-09-01
274
+
275
+ ### Added
276
+
277
+ - `flow()` can enter record operations directly, while `Rows` remains available
278
+ as an explicit relational view. New column and NumPy factories make it possible
279
+ to keep columnar inputs columnar until execution.
280
+ - `ExecutionReport` and `run_with_report()` expose the strategy used by a terminal
281
+ without changing the terminal result.
282
+ - Standard `__arrow_c_stream__` and `__dataframe__` providers can be routed through
283
+ `flow()`.
284
+ - Arrow-backed CSV scanning supports typed incremental reads and query projection
285
+ under PyArrow's parsing and error contract.
286
+ - `AsyncFlow` now includes queue sources, bounded prefetch, session windows, numeric
287
+ terminals, and execution reports.
288
+ - `Pairs` accepts explicit engine selection and row expressions for pair filtering.
289
+
290
+ ### Changed
291
+
292
+ - Retained NumPy matrices can execute guarded identity, projection, filter,
293
+ computed-column, aggregate, and grouped-aggregate paths without first building
294
+ one Python dictionary per input row.
295
+ - Native Rust execution now handles additional scalar, pair, reshape, join,
296
+ group, and global aggregation plans. Unsupported or data-semantics-sensitive
297
+ cases still use the Python path.
298
+ - Record joins and grouped aggregation use narrower shape checks and bounded native
299
+ kernels where they preserve Python ordering, identity, errors, and cleanup.
300
+ - `rows.from_csv()` and `rows.from_jsonl()` now accept caller-owned open handles
301
+ and replayable zero-argument opener functions in addition to paths.
302
+ - The benchmark runner now compares fpstreams with Python, NumPy, and pandas and
303
+ reports the percentage difference for each comparable case.
304
+
305
+ ### Fixed
306
+
307
+ - Fast paths now revalidate cached row expressions, collector programs, NumPy
308
+ adapters, and implementation primitives before bypassing Python execution.
309
+ - One-shot sources, iterators, async tasks, database resources, spill files, and
310
+ retained tabular readers keep their cleanup behavior on early return and errors.
311
+ - Cleanup attempts every owned resource, preserves the operation error as primary,
312
+ and reports independent close failures without inheriting an unrelated outer
313
+ exception handler.
314
+ - Path and opener CSV/JSONL sources open on execution. Handles returned by an
315
+ opener are closed by fpstreams; caller-owned handles remain open. Arrow C
316
+ streams, explicit column mappings, and NumPy inputs keep their documented
317
+ construction-time import or conversion behavior.
318
+
319
+ ## 2.0.0
320
+
321
+ fpstreams 2 replaced the v1 implementation with typed lazy plans, a primary
322
+ `Flow` API, explicit `Rows`, `AsyncFlow`, and `Pairs` views, and optional Rust and
323
+ Arrow execution.