duckdb 0.7.2-dev12.0 → 0.7.2-dev1238.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (631) hide show
  1. package/binding.gyp +12 -7
  2. package/lib/duckdb.d.ts +55 -2
  3. package/lib/duckdb.js +20 -1
  4. package/package.json +1 -1
  5. package/src/connection.cpp +1 -2
  6. package/src/database.cpp +1 -1
  7. package/src/duckdb/extension/icu/icu-extension.cpp +4 -0
  8. package/src/duckdb/extension/icu/icu-list-range.cpp +207 -0
  9. package/src/duckdb/extension/icu/icu-table-range.cpp +194 -0
  10. package/src/duckdb/extension/icu/include/icu-list-range.hpp +17 -0
  11. package/src/duckdb/extension/icu/include/icu-table-range.hpp +17 -0
  12. package/src/duckdb/extension/icu/third_party/icu/stubdata/stubdata.cpp +1 -1
  13. package/src/duckdb/extension/json/include/json_common.hpp +1 -0
  14. package/src/duckdb/extension/json/include/json_functions.hpp +2 -0
  15. package/src/duckdb/extension/json/include/json_serializer.hpp +77 -0
  16. package/src/duckdb/extension/json/json_functions/json_serialize_sql.cpp +147 -0
  17. package/src/duckdb/extension/json/json_functions/read_json.cpp +6 -5
  18. package/src/duckdb/extension/json/json_functions.cpp +12 -4
  19. package/src/duckdb/extension/json/json_scan.cpp +2 -2
  20. package/src/duckdb/extension/json/json_serializer.cpp +217 -0
  21. package/src/duckdb/extension/parquet/column_reader.cpp +94 -15
  22. package/src/duckdb/extension/parquet/column_writer.cpp +0 -1
  23. package/src/duckdb/extension/parquet/include/column_reader.hpp +1 -2
  24. package/src/duckdb/extension/parquet/include/decode_utils.hpp +5 -4
  25. package/src/duckdb/extension/parquet/include/generated_column_reader.hpp +1 -11
  26. package/src/duckdb/extension/parquet/include/parquet_timestamp.hpp +2 -1
  27. package/src/duckdb/extension/parquet/parquet-extension.cpp +14 -3
  28. package/src/duckdb/extension/parquet/parquet_reader.cpp +6 -1
  29. package/src/duckdb/extension/parquet/parquet_statistics.cpp +49 -36
  30. package/src/duckdb/extension/parquet/parquet_timestamp.cpp +16 -6
  31. package/src/duckdb/src/catalog/catalog.cpp +34 -5
  32. package/src/duckdb/src/catalog/catalog_entry/duck_schema_entry.cpp +4 -0
  33. package/src/duckdb/src/catalog/catalog_entry/duck_table_entry.cpp +2 -21
  34. package/src/duckdb/src/catalog/catalog_entry/scalar_function_catalog_entry.cpp +7 -6
  35. package/src/duckdb/src/catalog/catalog_entry/table_catalog_entry.cpp +3 -3
  36. package/src/duckdb/src/catalog/catalog_entry/table_function_catalog_entry.cpp +20 -1
  37. package/src/duckdb/src/catalog/catalog_entry/type_catalog_entry.cpp +8 -2
  38. package/src/duckdb/src/catalog/catalog_set.cpp +1 -0
  39. package/src/duckdb/src/catalog/default/default_functions.cpp +3 -0
  40. package/src/duckdb/src/catalog/dependency_list.cpp +12 -0
  41. package/src/duckdb/src/catalog/duck_catalog.cpp +34 -7
  42. package/src/duckdb/src/common/arrow/arrow_appender.cpp +48 -4
  43. package/src/duckdb/src/common/arrow/arrow_converter.cpp +1 -1
  44. package/src/duckdb/src/common/box_renderer.cpp +109 -23
  45. package/src/duckdb/src/common/enums/expression_type.cpp +8 -222
  46. package/src/duckdb/src/common/enums/join_type.cpp +3 -22
  47. package/src/duckdb/src/common/enums/logical_operator_type.cpp +2 -0
  48. package/src/duckdb/src/common/enums/statement_type.cpp +2 -0
  49. package/src/duckdb/src/common/exception.cpp +15 -1
  50. package/src/duckdb/src/common/field_writer.cpp +1 -0
  51. package/src/duckdb/src/common/hive_partitioning.cpp +3 -1
  52. package/src/duckdb/src/common/local_file_system.cpp +64 -7
  53. package/src/duckdb/src/common/operator/cast_operators.cpp +1 -1
  54. package/src/duckdb/src/common/preserved_error.cpp +7 -5
  55. package/src/duckdb/src/common/progress_bar/progress_bar.cpp +7 -0
  56. package/src/duckdb/src/common/serializer/buffered_deserializer.cpp +4 -0
  57. package/src/duckdb/src/common/serializer/buffered_file_reader.cpp +15 -2
  58. package/src/duckdb/src/common/serializer/enum_serializer.cpp +1176 -0
  59. package/src/duckdb/src/common/sort/comparators.cpp +14 -5
  60. package/src/duckdb/src/common/sort/sort_state.cpp +5 -7
  61. package/src/duckdb/src/common/sort/sorted_block.cpp +0 -1
  62. package/src/duckdb/src/common/string_util.cpp +18 -1
  63. package/src/duckdb/src/common/types/bit.cpp +166 -87
  64. package/src/duckdb/src/common/types/blob.cpp +1 -1
  65. package/src/duckdb/src/common/types/chunk_collection.cpp +2 -2
  66. package/src/duckdb/src/common/types/column_data_collection.cpp +39 -2
  67. package/src/duckdb/src/common/types/column_data_collection_segment.cpp +12 -10
  68. package/src/duckdb/src/common/types/data_chunk.cpp +1 -1
  69. package/src/duckdb/src/common/types/interval.cpp +0 -41
  70. package/src/duckdb/src/common/types/list_segment.cpp +658 -0
  71. package/src/duckdb/src/common/types/string_heap.cpp +1 -1
  72. package/src/duckdb/src/common/types/string_type.cpp +1 -1
  73. package/src/duckdb/src/common/types/time.cpp +13 -0
  74. package/src/duckdb/src/common/types/validity_mask.cpp +24 -7
  75. package/src/duckdb/src/common/types/value.cpp +320 -154
  76. package/src/duckdb/src/common/types/vector.cpp +158 -134
  77. package/src/duckdb/src/common/types.cpp +313 -153
  78. package/src/duckdb/src/common/value_operations/comparison_operations.cpp +14 -22
  79. package/src/duckdb/src/common/vector_operations/comparison_operators.cpp +10 -10
  80. package/src/duckdb/src/common/vector_operations/is_distinct_from.cpp +11 -10
  81. package/src/duckdb/src/common/vector_operations/vector_cast.cpp +2 -1
  82. package/src/duckdb/src/execution/aggregate_hashtable.cpp +98 -74
  83. package/src/duckdb/src/execution/column_binding_resolver.cpp +21 -5
  84. package/src/duckdb/src/execution/expression_executor/execute_cast.cpp +2 -1
  85. package/src/duckdb/src/execution/expression_executor/execute_comparison.cpp +2 -2
  86. package/src/duckdb/src/execution/index/art/art.cpp +19 -5
  87. package/src/duckdb/src/execution/join_hashtable.cpp +3 -1
  88. package/src/duckdb/src/execution/operator/aggregate/physical_hash_aggregate.cpp +1 -1
  89. package/src/duckdb/src/execution/operator/aggregate/physical_perfecthash_aggregate.cpp +4 -5
  90. package/src/duckdb/src/execution/operator/aggregate/physical_window.cpp +117 -26
  91. package/src/duckdb/src/execution/operator/helper/physical_limit.cpp +3 -0
  92. package/src/duckdb/src/execution/operator/helper/physical_vacuum.cpp +5 -3
  93. package/src/duckdb/src/execution/operator/join/physical_blockwise_nl_join.cpp +64 -17
  94. package/src/duckdb/src/execution/operator/join/physical_hash_join.cpp +2 -0
  95. package/src/duckdb/src/execution/operator/join/physical_iejoin.cpp +2 -2
  96. package/src/duckdb/src/execution/operator/join/physical_index_join.cpp +13 -4
  97. package/src/duckdb/src/execution/operator/join/physical_join.cpp +0 -3
  98. package/src/duckdb/src/execution/operator/join/physical_piecewise_merge_join.cpp +6 -11
  99. package/src/duckdb/src/execution/operator/join/physical_range_join.cpp +3 -1
  100. package/src/duckdb/src/execution/operator/persistent/base_csv_reader.cpp +11 -4
  101. package/src/duckdb/src/execution/operator/persistent/buffered_csv_reader.cpp +24 -19
  102. package/src/duckdb/src/execution/operator/persistent/csv_reader_options.cpp +3 -0
  103. package/src/duckdb/src/execution/operator/persistent/physical_batch_insert.cpp +2 -1
  104. package/src/duckdb/src/execution/operator/persistent/physical_copy_to_file.cpp +2 -2
  105. package/src/duckdb/src/execution/operator/persistent/physical_delete.cpp +1 -3
  106. package/src/duckdb/src/execution/operator/persistent/physical_insert.cpp +1 -0
  107. package/src/duckdb/src/execution/operator/projection/physical_projection.cpp +34 -0
  108. package/src/duckdb/src/execution/operator/scan/physical_positional_scan.cpp +20 -5
  109. package/src/duckdb/src/execution/operator/schema/physical_create_type.cpp +20 -40
  110. package/src/duckdb/src/execution/operator/set/physical_recursive_cte.cpp +2 -5
  111. package/src/duckdb/src/execution/partitionable_hashtable.cpp +20 -5
  112. package/src/duckdb/src/execution/physical_plan/plan_aggregate.cpp +22 -16
  113. package/src/duckdb/src/execution/physical_plan/plan_asof_join.cpp +97 -0
  114. package/src/duckdb/src/execution/physical_plan/plan_comparison_join.cpp +95 -47
  115. package/src/duckdb/src/execution/physical_plan/plan_create_index.cpp +2 -1
  116. package/src/duckdb/src/execution/physical_plan/plan_distinct.cpp +5 -8
  117. package/src/duckdb/src/execution/physical_plan/plan_positional_join.cpp +14 -5
  118. package/src/duckdb/src/execution/physical_plan_generator.cpp +3 -0
  119. package/src/duckdb/src/execution/radix_partitioned_hashtable.cpp +23 -15
  120. package/src/duckdb/src/execution/window_segment_tree.cpp +173 -1
  121. package/src/duckdb/src/function/aggregate/algebraic/avg.cpp +0 -6
  122. package/src/duckdb/src/function/aggregate/distributive/bitagg.cpp +99 -95
  123. package/src/duckdb/src/function/aggregate/distributive/bitstring_agg.cpp +269 -0
  124. package/src/duckdb/src/function/aggregate/distributive/bool.cpp +2 -0
  125. package/src/duckdb/src/function/aggregate/distributive/count.cpp +3 -4
  126. package/src/duckdb/src/function/aggregate/distributive/first.cpp +1 -0
  127. package/src/duckdb/src/function/aggregate/distributive/minmax.cpp +2 -0
  128. package/src/duckdb/src/function/aggregate/distributive/sum.cpp +19 -16
  129. package/src/duckdb/src/function/aggregate/distributive_functions.cpp +1 -0
  130. package/src/duckdb/src/function/aggregate/holistic/approximate_quantile.cpp +5 -2
  131. package/src/duckdb/src/function/aggregate/holistic/mode.cpp +1 -1
  132. package/src/duckdb/src/function/aggregate/holistic/quantile.cpp +16 -1
  133. package/src/duckdb/src/function/aggregate/nested/list.cpp +6 -712
  134. package/src/duckdb/src/function/aggregate/sorted_aggregate_function.cpp +138 -45
  135. package/src/duckdb/src/function/cast/bit_cast.cpp +0 -2
  136. package/src/duckdb/src/function/cast/blob_cast.cpp +0 -1
  137. package/src/duckdb/src/function/cast/cast_function_set.cpp +1 -1
  138. package/src/duckdb/src/function/cast/enum_casts.cpp +25 -3
  139. package/src/duckdb/src/function/cast/list_casts.cpp +17 -4
  140. package/src/duckdb/src/function/cast/map_cast.cpp +5 -2
  141. package/src/duckdb/src/function/cast/string_cast.cpp +36 -10
  142. package/src/duckdb/src/function/cast/struct_cast.cpp +24 -4
  143. package/src/duckdb/src/function/cast/time_casts.cpp +2 -2
  144. package/src/duckdb/src/function/cast/union_casts.cpp +33 -7
  145. package/src/duckdb/src/function/cast_rules.cpp +9 -4
  146. package/src/duckdb/src/function/function_binder.cpp +1 -8
  147. package/src/duckdb/src/function/pragma/pragma_queries.cpp +24 -1
  148. package/src/duckdb/src/function/scalar/bit/bitstring.cpp +100 -0
  149. package/src/duckdb/src/function/scalar/date/current.cpp +0 -2
  150. package/src/duckdb/src/function/scalar/date/date_diff.cpp +0 -1
  151. package/src/duckdb/src/function/scalar/date/date_part.cpp +18 -26
  152. package/src/duckdb/src/function/scalar/date/date_sub.cpp +0 -1
  153. package/src/duckdb/src/function/scalar/date/date_trunc.cpp +10 -14
  154. package/src/duckdb/src/function/scalar/generic/stats.cpp +2 -4
  155. package/src/duckdb/src/function/scalar/list/contains_or_position.cpp +4 -146
  156. package/src/duckdb/src/function/scalar/list/flatten.cpp +5 -12
  157. package/src/duckdb/src/function/scalar/list/list_aggregates.cpp +1 -1
  158. package/src/duckdb/src/function/scalar/list/list_concat.cpp +8 -12
  159. package/src/duckdb/src/function/scalar/list/list_extract.cpp +5 -12
  160. package/src/duckdb/src/function/scalar/list/list_lambdas.cpp +7 -3
  161. package/src/duckdb/src/function/scalar/list/list_sort.cpp +25 -18
  162. package/src/duckdb/src/function/scalar/list/list_value.cpp +6 -10
  163. package/src/duckdb/src/function/scalar/map/map.cpp +47 -1
  164. package/src/duckdb/src/function/scalar/map/map_entries.cpp +61 -0
  165. package/src/duckdb/src/function/scalar/map/map_extract.cpp +68 -26
  166. package/src/duckdb/src/function/scalar/map/map_keys_values.cpp +97 -0
  167. package/src/duckdb/src/function/scalar/math/numeric.cpp +101 -17
  168. package/src/duckdb/src/function/scalar/math_functions.cpp +3 -0
  169. package/src/duckdb/src/function/scalar/nested_functions.cpp +3 -0
  170. package/src/duckdb/src/function/scalar/operators/add.cpp +0 -9
  171. package/src/duckdb/src/function/scalar/operators/arithmetic.cpp +29 -48
  172. package/src/duckdb/src/function/scalar/operators/bitwise.cpp +0 -63
  173. package/src/duckdb/src/function/scalar/operators/multiply.cpp +5 -6
  174. package/src/duckdb/src/function/scalar/operators/subtract.cpp +0 -6
  175. package/src/duckdb/src/function/scalar/string/caseconvert.cpp +2 -6
  176. package/src/duckdb/src/function/scalar/string/hex.cpp +201 -0
  177. package/src/duckdb/src/function/scalar/string/instr.cpp +2 -6
  178. package/src/duckdb/src/function/scalar/string/length.cpp +2 -6
  179. package/src/duckdb/src/function/scalar/string/like.cpp +2 -6
  180. package/src/duckdb/src/function/scalar/string/regexp/regexp_extract_all.cpp +243 -0
  181. package/src/duckdb/src/function/scalar/string/regexp/regexp_util.cpp +79 -0
  182. package/src/duckdb/src/function/scalar/string/regexp.cpp +21 -80
  183. package/src/duckdb/src/function/scalar/string/substring.cpp +2 -6
  184. package/src/duckdb/src/function/scalar/string_functions.cpp +2 -0
  185. package/src/duckdb/src/function/scalar/struct/struct_extract.cpp +5 -10
  186. package/src/duckdb/src/function/scalar/struct/struct_insert.cpp +11 -14
  187. package/src/duckdb/src/function/scalar/struct/struct_pack.cpp +6 -7
  188. package/src/duckdb/src/function/table/arrow.cpp +5 -2
  189. package/src/duckdb/src/function/table/arrow_conversion.cpp +25 -1
  190. package/src/duckdb/src/function/table/checkpoint.cpp +5 -1
  191. package/src/duckdb/src/function/table/read_csv.cpp +60 -0
  192. package/src/duckdb/src/function/table/system/duckdb_constraints.cpp +2 -2
  193. package/src/duckdb/src/function/table/system/test_all_types.cpp +2 -2
  194. package/src/duckdb/src/function/table/table_scan.cpp +9 -12
  195. package/src/duckdb/src/function/table/version/pragma_version.cpp +2 -2
  196. package/src/duckdb/src/function/table_function.cpp +30 -11
  197. package/src/duckdb/src/include/duckdb/catalog/catalog.hpp +6 -0
  198. package/src/duckdb/src/include/duckdb/catalog/catalog_entry/duck_table_entry.hpp +1 -1
  199. package/src/duckdb/src/include/duckdb/catalog/catalog_entry/table_function_catalog_entry.hpp +6 -8
  200. package/src/duckdb/src/include/duckdb/catalog/dependency_list.hpp +3 -0
  201. package/src/duckdb/src/include/duckdb/catalog/duck_catalog.hpp +2 -1
  202. package/src/duckdb/src/include/duckdb/common/box_renderer.hpp +8 -2
  203. package/src/duckdb/src/include/duckdb/common/constants.hpp +0 -19
  204. package/src/duckdb/src/include/duckdb/common/enums/aggregate_handling.hpp +2 -0
  205. package/src/duckdb/src/include/duckdb/common/enums/expression_type.hpp +2 -3
  206. package/src/duckdb/src/include/duckdb/common/enums/joinref_type.hpp +7 -4
  207. package/src/duckdb/src/include/duckdb/common/enums/logical_operator_type.hpp +1 -0
  208. package/src/duckdb/src/include/duckdb/common/enums/order_type.hpp +2 -0
  209. package/src/duckdb/src/include/duckdb/common/enums/set_operation_type.hpp +2 -1
  210. package/src/duckdb/src/include/duckdb/common/enums/statement_type.hpp +2 -1
  211. package/src/duckdb/src/include/duckdb/common/enums/tableref_type.hpp +2 -1
  212. package/src/duckdb/src/include/duckdb/common/exception.hpp +69 -2
  213. package/src/duckdb/src/include/duckdb/common/field_writer.hpp +12 -4
  214. package/src/duckdb/src/include/duckdb/common/helper.hpp +1 -1
  215. package/src/duckdb/src/include/duckdb/common/{http_stats.hpp → http_state.hpp} +18 -4
  216. package/src/duckdb/src/include/duckdb/common/operator/comparison_operators.hpp +45 -149
  217. package/src/duckdb/src/include/duckdb/common/operator/multiply.hpp +2 -0
  218. package/src/duckdb/src/include/duckdb/common/optional_ptr.hpp +45 -0
  219. package/src/duckdb/src/include/duckdb/common/preserved_error.hpp +6 -1
  220. package/src/duckdb/src/include/duckdb/common/progress_bar/progress_bar.hpp +2 -0
  221. package/src/duckdb/src/include/duckdb/common/serializer/buffered_deserializer.hpp +4 -2
  222. package/src/duckdb/src/include/duckdb/common/serializer/buffered_file_reader.hpp +8 -2
  223. package/src/duckdb/src/include/duckdb/common/serializer/enum_serializer.hpp +113 -0
  224. package/src/duckdb/src/include/duckdb/common/serializer/format_deserializer.hpp +336 -0
  225. package/src/duckdb/src/include/duckdb/common/serializer/format_serializer.hpp +268 -0
  226. package/src/duckdb/src/include/duckdb/common/serializer/serialization_traits.hpp +126 -0
  227. package/src/duckdb/src/include/duckdb/common/serializer.hpp +13 -0
  228. package/src/duckdb/src/include/duckdb/common/string_util.hpp +27 -0
  229. package/src/duckdb/src/include/duckdb/common/types/bit.hpp +12 -7
  230. package/src/duckdb/src/include/duckdb/common/types/interval.hpp +39 -3
  231. package/src/duckdb/src/include/duckdb/common/types/list_segment.hpp +70 -0
  232. package/src/duckdb/src/include/duckdb/common/types/string_type.hpp +73 -3
  233. package/src/duckdb/src/include/duckdb/common/types/time.hpp +3 -0
  234. package/src/duckdb/src/include/duckdb/common/types/validity_mask.hpp +4 -1
  235. package/src/duckdb/src/include/duckdb/common/types/value.hpp +17 -48
  236. package/src/duckdb/src/include/duckdb/common/types/value_map.hpp +1 -1
  237. package/src/duckdb/src/include/duckdb/common/types/vector.hpp +3 -1
  238. package/src/duckdb/src/include/duckdb/common/types.hpp +45 -8
  239. package/src/duckdb/src/include/duckdb/common/vector_operations/unary_executor.hpp +2 -2
  240. package/src/duckdb/src/include/duckdb/execution/aggregate_hashtable.hpp +35 -20
  241. package/src/duckdb/src/include/duckdb/execution/index/art/art.hpp +3 -14
  242. package/src/duckdb/src/include/duckdb/execution/operator/aggregate/physical_perfecthash_aggregate.hpp +1 -1
  243. package/src/duckdb/src/include/duckdb/execution/operator/join/physical_cross_product.hpp +2 -0
  244. package/src/duckdb/src/include/duckdb/execution/operator/persistent/csv_file_handle.hpp +1 -0
  245. package/src/duckdb/src/include/duckdb/execution/operator/persistent/csv_reader_options.hpp +10 -0
  246. package/src/duckdb/src/include/duckdb/execution/operator/projection/physical_projection.hpp +5 -0
  247. package/src/duckdb/src/include/duckdb/execution/partitionable_hashtable.hpp +5 -1
  248. package/src/duckdb/src/include/duckdb/execution/physical_plan_generator.hpp +1 -3
  249. package/src/duckdb/src/include/duckdb/execution/window_segment_tree.hpp +54 -0
  250. package/src/duckdb/src/include/duckdb/function/aggregate/distributive_functions.hpp +5 -0
  251. package/src/duckdb/src/include/duckdb/function/aggregate_function.hpp +18 -6
  252. package/src/duckdb/src/include/duckdb/function/cast/bound_cast_data.hpp +84 -0
  253. package/src/duckdb/src/include/duckdb/function/cast/cast_function_set.hpp +2 -2
  254. package/src/duckdb/src/include/duckdb/function/cast/default_casts.hpp +28 -64
  255. package/src/duckdb/src/include/duckdb/function/function_binder.hpp +3 -6
  256. package/src/duckdb/src/include/duckdb/function/scalar/bit_functions.hpp +4 -0
  257. package/src/duckdb/src/include/duckdb/function/scalar/list/contains_or_position.hpp +138 -0
  258. package/src/duckdb/src/include/duckdb/function/scalar/math_functions.hpp +8 -0
  259. package/src/duckdb/src/include/duckdb/function/scalar/nested_functions.hpp +59 -0
  260. package/src/duckdb/src/include/duckdb/function/scalar/regexp.hpp +81 -1
  261. package/src/duckdb/src/include/duckdb/function/scalar/string_functions.hpp +4 -0
  262. package/src/duckdb/src/include/duckdb/function/scalar_function.hpp +2 -2
  263. package/src/duckdb/src/include/duckdb/function/table/arrow.hpp +12 -1
  264. package/src/duckdb/src/include/duckdb/function/table_function.hpp +10 -0
  265. package/src/duckdb/src/include/duckdb/main/capi/capi_internal.hpp +2 -0
  266. package/src/duckdb/src/include/duckdb/main/client_config.hpp +2 -0
  267. package/src/duckdb/src/include/duckdb/main/client_data.hpp +3 -3
  268. package/src/duckdb/src/include/duckdb/main/config.hpp +3 -0
  269. package/src/duckdb/src/include/duckdb/main/connection_manager.hpp +2 -0
  270. package/src/duckdb/src/include/duckdb/main/database.hpp +1 -0
  271. package/src/duckdb/src/include/duckdb/main/extension_entries.hpp +2 -0
  272. package/src/duckdb/src/include/duckdb/main/prepared_statement.hpp +2 -0
  273. package/src/duckdb/src/include/duckdb/main/relation/explain_relation.hpp +2 -1
  274. package/src/duckdb/src/include/duckdb/main/relation.hpp +2 -1
  275. package/src/duckdb/src/include/duckdb/optimizer/filter_pushdown.hpp +2 -0
  276. package/src/duckdb/src/include/duckdb/optimizer/join_order/cardinality_estimator.hpp +2 -2
  277. package/src/duckdb/src/include/duckdb/optimizer/rule/list.hpp +1 -0
  278. package/src/duckdb/src/include/duckdb/optimizer/rule/ordered_aggregate_optimizer.hpp +24 -0
  279. package/src/duckdb/src/include/duckdb/parser/common_table_expression_info.hpp +4 -0
  280. package/src/duckdb/src/include/duckdb/parser/expression/between_expression.hpp +3 -0
  281. package/src/duckdb/src/include/duckdb/parser/expression/bound_expression.hpp +2 -0
  282. package/src/duckdb/src/include/duckdb/parser/expression/case_expression.hpp +5 -0
  283. package/src/duckdb/src/include/duckdb/parser/expression/cast_expression.hpp +2 -0
  284. package/src/duckdb/src/include/duckdb/parser/expression/collate_expression.hpp +2 -0
  285. package/src/duckdb/src/include/duckdb/parser/expression/columnref_expression.hpp +2 -0
  286. package/src/duckdb/src/include/duckdb/parser/expression/comparison_expression.hpp +2 -0
  287. package/src/duckdb/src/include/duckdb/parser/expression/conjunction_expression.hpp +2 -0
  288. package/src/duckdb/src/include/duckdb/parser/expression/constant_expression.hpp +3 -0
  289. package/src/duckdb/src/include/duckdb/parser/expression/default_expression.hpp +1 -0
  290. package/src/duckdb/src/include/duckdb/parser/expression/function_expression.hpp +4 -2
  291. package/src/duckdb/src/include/duckdb/parser/expression/lambda_expression.hpp +2 -0
  292. package/src/duckdb/src/include/duckdb/parser/expression/operator_expression.hpp +2 -0
  293. package/src/duckdb/src/include/duckdb/parser/expression/parameter_expression.hpp +2 -0
  294. package/src/duckdb/src/include/duckdb/parser/expression/positional_reference_expression.hpp +2 -0
  295. package/src/duckdb/src/include/duckdb/parser/expression/star_expression.hpp +4 -2
  296. package/src/duckdb/src/include/duckdb/parser/expression/subquery_expression.hpp +2 -0
  297. package/src/duckdb/src/include/duckdb/parser/expression/window_expression.hpp +5 -0
  298. package/src/duckdb/src/include/duckdb/parser/parsed_data/alter_info.hpp +5 -1
  299. package/src/duckdb/src/include/duckdb/parser/parsed_data/{alter_function_info.hpp → alter_scalar_function_info.hpp} +13 -13
  300. package/src/duckdb/src/include/duckdb/parser/parsed_data/alter_table_function_info.hpp +47 -0
  301. package/src/duckdb/src/include/duckdb/parser/parsed_data/alter_table_info.hpp +6 -0
  302. package/src/duckdb/src/include/duckdb/parser/parsed_data/create_table_function_info.hpp +2 -1
  303. package/src/duckdb/src/include/duckdb/parser/parsed_data/sample_options.hpp +2 -0
  304. package/src/duckdb/src/include/duckdb/parser/parsed_expression.hpp +5 -0
  305. package/src/duckdb/src/include/duckdb/parser/query_node/recursive_cte_node.hpp +3 -0
  306. package/src/duckdb/src/include/duckdb/parser/query_node/select_node.hpp +5 -0
  307. package/src/duckdb/src/include/duckdb/parser/query_node/set_operation_node.hpp +3 -0
  308. package/src/duckdb/src/include/duckdb/parser/query_node.hpp +13 -2
  309. package/src/duckdb/src/include/duckdb/parser/result_modifier.hpp +24 -1
  310. package/src/duckdb/src/include/duckdb/parser/sql_statement.hpp +2 -1
  311. package/src/duckdb/src/include/duckdb/parser/statement/multi_statement.hpp +28 -0
  312. package/src/duckdb/src/include/duckdb/parser/statement/select_statement.hpp +6 -1
  313. package/src/duckdb/src/include/duckdb/parser/tableref/basetableref.hpp +4 -0
  314. package/src/duckdb/src/include/duckdb/parser/tableref/emptytableref.hpp +2 -0
  315. package/src/duckdb/src/include/duckdb/parser/tableref/expressionlistref.hpp +3 -0
  316. package/src/duckdb/src/include/duckdb/parser/tableref/joinref.hpp +3 -0
  317. package/src/duckdb/src/include/duckdb/parser/tableref/list.hpp +1 -0
  318. package/src/duckdb/src/include/duckdb/parser/tableref/pivotref.hpp +87 -0
  319. package/src/duckdb/src/include/duckdb/parser/tableref/subqueryref.hpp +3 -0
  320. package/src/duckdb/src/include/duckdb/parser/tableref/table_function_ref.hpp +3 -0
  321. package/src/duckdb/src/include/duckdb/parser/tableref.hpp +3 -1
  322. package/src/duckdb/src/include/duckdb/parser/tokens.hpp +2 -0
  323. package/src/duckdb/src/include/duckdb/parser/transformer.hpp +33 -0
  324. package/src/duckdb/src/include/duckdb/planner/bind_context.hpp +2 -0
  325. package/src/duckdb/src/include/duckdb/planner/binder.hpp +15 -4
  326. package/src/duckdb/src/include/duckdb/planner/bound_result_modifier.hpp +3 -0
  327. package/src/duckdb/src/include/duckdb/planner/expression/bound_aggregate_expression.hpp +3 -0
  328. package/src/duckdb/src/include/duckdb/planner/expression_binder/base_select_binder.hpp +64 -0
  329. package/src/duckdb/src/include/duckdb/planner/expression_binder/having_binder.hpp +2 -2
  330. package/src/duckdb/src/include/duckdb/planner/expression_binder/order_binder.hpp +4 -1
  331. package/src/duckdb/src/include/duckdb/planner/expression_binder/qualify_binder.hpp +2 -2
  332. package/src/duckdb/src/include/duckdb/planner/expression_binder/select_binder.hpp +9 -38
  333. package/src/duckdb/src/include/duckdb/planner/expression_binder.hpp +1 -1
  334. package/src/duckdb/src/include/duckdb/planner/logical_tokens.hpp +1 -0
  335. package/src/duckdb/src/include/duckdb/planner/operator/list.hpp +1 -0
  336. package/src/duckdb/src/include/duckdb/planner/operator/logical_asof_join.hpp +22 -0
  337. package/src/duckdb/src/include/duckdb/planner/operator/logical_comparison_join.hpp +5 -2
  338. package/src/duckdb/src/include/duckdb/planner/operator/logical_distinct.hpp +3 -0
  339. package/src/duckdb/src/include/duckdb/planner/query_node/bound_select_node.hpp +8 -2
  340. package/src/duckdb/src/include/duckdb/storage/buffer/block_handle.hpp +2 -0
  341. package/src/duckdb/src/include/duckdb/storage/buffer_manager.hpp +76 -44
  342. package/src/duckdb/src/include/duckdb/storage/checkpoint/table_data_writer.hpp +3 -2
  343. package/src/duckdb/src/include/duckdb/storage/checkpoint_manager.hpp +1 -1
  344. package/src/duckdb/src/include/duckdb/storage/compression/chimp/chimp_compress.hpp +2 -2
  345. package/src/duckdb/src/include/duckdb/storage/compression/chimp/chimp_fetch.hpp +1 -1
  346. package/src/duckdb/src/include/duckdb/storage/compression/chimp/chimp_scan.hpp +2 -1
  347. package/src/duckdb/src/include/duckdb/storage/compression/patas/patas_compress.hpp +2 -2
  348. package/src/duckdb/src/include/duckdb/storage/compression/patas/patas_fetch.hpp +1 -1
  349. package/src/duckdb/src/include/duckdb/storage/compression/patas/patas_scan.hpp +2 -1
  350. package/src/duckdb/src/include/duckdb/storage/data_pointer.hpp +4 -3
  351. package/src/duckdb/src/include/duckdb/storage/data_table.hpp +4 -3
  352. package/src/duckdb/src/include/duckdb/storage/index.hpp +5 -4
  353. package/src/duckdb/src/include/duckdb/storage/meta_block_reader.hpp +7 -0
  354. package/src/duckdb/src/include/duckdb/storage/statistics/base_statistics.hpp +93 -29
  355. package/src/duckdb/src/include/duckdb/storage/statistics/column_statistics.hpp +22 -3
  356. package/src/duckdb/src/include/duckdb/storage/statistics/distinct_statistics.hpp +8 -6
  357. package/src/duckdb/src/include/duckdb/storage/statistics/list_stats.hpp +41 -0
  358. package/src/duckdb/src/include/duckdb/storage/statistics/node_statistics.hpp +26 -0
  359. package/src/duckdb/src/include/duckdb/storage/statistics/numeric_stats.hpp +114 -0
  360. package/src/duckdb/src/include/duckdb/storage/statistics/numeric_stats_union.hpp +62 -0
  361. package/src/duckdb/src/include/duckdb/storage/statistics/segment_statistics.hpp +2 -7
  362. package/src/duckdb/src/include/duckdb/storage/statistics/string_stats.hpp +74 -0
  363. package/src/duckdb/src/include/duckdb/storage/statistics/struct_stats.hpp +42 -0
  364. package/src/duckdb/src/include/duckdb/storage/string_uncompressed.hpp +2 -3
  365. package/src/duckdb/src/include/duckdb/storage/table/column_checkpoint_state.hpp +2 -1
  366. package/src/duckdb/src/include/duckdb/storage/table/column_data.hpp +21 -7
  367. package/src/duckdb/src/include/duckdb/storage/table/column_data_checkpointer.hpp +3 -2
  368. package/src/duckdb/src/include/duckdb/storage/table/column_segment.hpp +5 -6
  369. package/src/duckdb/src/include/duckdb/storage/table/column_segment_tree.hpp +18 -0
  370. package/src/duckdb/src/include/duckdb/storage/table/list_column_data.hpp +1 -1
  371. package/src/duckdb/src/include/duckdb/storage/table/persistent_table_data.hpp +6 -3
  372. package/src/duckdb/src/include/duckdb/storage/table/row_group.hpp +41 -45
  373. package/src/duckdb/src/include/duckdb/storage/table/row_group_collection.hpp +23 -7
  374. package/src/duckdb/src/include/duckdb/storage/table/row_group_segment_tree.hpp +35 -0
  375. package/src/duckdb/src/include/duckdb/storage/table/scan_state.hpp +21 -29
  376. package/src/duckdb/src/include/duckdb/storage/table/segment_base.hpp +6 -6
  377. package/src/duckdb/src/include/duckdb/storage/table/segment_tree.hpp +281 -26
  378. package/src/duckdb/src/include/duckdb/storage/table/standard_column_data.hpp +0 -4
  379. package/src/duckdb/src/include/duckdb/storage/table/table_statistics.hpp +5 -0
  380. package/src/duckdb/src/include/duckdb/storage/table/update_segment.hpp +0 -1
  381. package/src/duckdb/src/include/duckdb/storage/write_ahead_log.hpp +1 -1
  382. package/src/duckdb/src/include/duckdb/transaction/local_storage.hpp +6 -3
  383. package/src/duckdb/src/include/duckdb.h +71 -2
  384. package/src/duckdb/src/include/duckdb.hpp +0 -1
  385. package/src/duckdb/src/main/capi/pending-c.cpp +16 -3
  386. package/src/duckdb/src/main/capi/result-c.cpp +27 -1
  387. package/src/duckdb/src/main/capi/stream-c.cpp +25 -0
  388. package/src/duckdb/src/main/capi/table_function-c.cpp +23 -0
  389. package/src/duckdb/src/main/client_context.cpp +38 -34
  390. package/src/duckdb/src/main/client_data.cpp +7 -6
  391. package/src/duckdb/src/main/config.cpp +70 -1
  392. package/src/duckdb/src/main/database.cpp +19 -2
  393. package/src/duckdb/src/main/extension/extension_install.cpp +7 -2
  394. package/src/duckdb/src/main/prepared_statement.cpp +4 -0
  395. package/src/duckdb/src/main/query_profiler.cpp +17 -15
  396. package/src/duckdb/src/main/relation/explain_relation.cpp +3 -3
  397. package/src/duckdb/src/main/relation.cpp +3 -2
  398. package/src/duckdb/src/main/settings/settings.cpp +20 -8
  399. package/src/duckdb/src/optimizer/column_lifetime_analyzer.cpp +1 -0
  400. package/src/duckdb/src/optimizer/deliminator.cpp +1 -1
  401. package/src/duckdb/src/optimizer/filter_combiner.cpp +3 -6
  402. package/src/duckdb/src/optimizer/filter_pullup.cpp +3 -1
  403. package/src/duckdb/src/optimizer/filter_pushdown.cpp +14 -8
  404. package/src/duckdb/src/optimizer/join_order/cardinality_estimator.cpp +107 -71
  405. package/src/duckdb/src/optimizer/join_order/join_order_optimizer.cpp +32 -12
  406. package/src/duckdb/src/optimizer/optimizer.cpp +1 -0
  407. package/src/duckdb/src/optimizer/pullup/pullup_from_left.cpp +2 -2
  408. package/src/duckdb/src/optimizer/pushdown/pushdown_aggregate.cpp +33 -5
  409. package/src/duckdb/src/optimizer/pushdown/pushdown_cross_product.cpp +1 -1
  410. package/src/duckdb/src/optimizer/pushdown/pushdown_inner_join.cpp +3 -0
  411. package/src/duckdb/src/optimizer/pushdown/pushdown_left_join.cpp +5 -12
  412. package/src/duckdb/src/optimizer/pushdown/pushdown_mark_join.cpp +2 -2
  413. package/src/duckdb/src/optimizer/pushdown/pushdown_single_join.cpp +1 -1
  414. package/src/duckdb/src/optimizer/remove_unused_columns.cpp +1 -0
  415. package/src/duckdb/src/optimizer/rule/move_constants.cpp +10 -4
  416. package/src/duckdb/src/optimizer/rule/ordered_aggregate_optimizer.cpp +30 -0
  417. package/src/duckdb/src/optimizer/rule/regex_optimizations.cpp +9 -2
  418. package/src/duckdb/src/optimizer/statistics/expression/propagate_aggregate.cpp +9 -3
  419. package/src/duckdb/src/optimizer/statistics/expression/propagate_and_compress.cpp +6 -7
  420. package/src/duckdb/src/optimizer/statistics/expression/propagate_cast.cpp +14 -11
  421. package/src/duckdb/src/optimizer/statistics/expression/propagate_columnref.cpp +1 -1
  422. package/src/duckdb/src/optimizer/statistics/expression/propagate_comparison.cpp +13 -15
  423. package/src/duckdb/src/optimizer/statistics/expression/propagate_conjunction.cpp +0 -1
  424. package/src/duckdb/src/optimizer/statistics/expression/propagate_constant.cpp +3 -75
  425. package/src/duckdb/src/optimizer/statistics/expression/propagate_function.cpp +7 -2
  426. package/src/duckdb/src/optimizer/statistics/expression/propagate_operator.cpp +10 -0
  427. package/src/duckdb/src/optimizer/statistics/operator/propagate_aggregate.cpp +2 -3
  428. package/src/duckdb/src/optimizer/statistics/operator/propagate_filter.cpp +29 -32
  429. package/src/duckdb/src/optimizer/statistics/operator/propagate_join.cpp +5 -5
  430. package/src/duckdb/src/optimizer/statistics/operator/propagate_set_operation.cpp +3 -3
  431. package/src/duckdb/src/optimizer/statistics_propagator.cpp +2 -1
  432. package/src/duckdb/src/optimizer/unnest_rewriter.cpp +2 -2
  433. package/src/duckdb/src/parallel/meta_pipeline.cpp +0 -7
  434. package/src/duckdb/src/parser/common_table_expression_info.cpp +19 -0
  435. package/src/duckdb/src/parser/expression/between_expression.cpp +17 -0
  436. package/src/duckdb/src/parser/expression/case_expression.cpp +28 -0
  437. package/src/duckdb/src/parser/expression/cast_expression.cpp +17 -0
  438. package/src/duckdb/src/parser/expression/collate_expression.cpp +16 -0
  439. package/src/duckdb/src/parser/expression/columnref_expression.cpp +15 -0
  440. package/src/duckdb/src/parser/expression/comparison_expression.cpp +16 -0
  441. package/src/duckdb/src/parser/expression/conjunction_expression.cpp +17 -0
  442. package/src/duckdb/src/parser/expression/constant_expression.cpp +14 -0
  443. package/src/duckdb/src/parser/expression/default_expression.cpp +7 -0
  444. package/src/duckdb/src/parser/expression/function_expression.cpp +35 -0
  445. package/src/duckdb/src/parser/expression/lambda_expression.cpp +16 -0
  446. package/src/duckdb/src/parser/expression/operator_expression.cpp +15 -0
  447. package/src/duckdb/src/parser/expression/parameter_expression.cpp +15 -0
  448. package/src/duckdb/src/parser/expression/positional_reference_expression.cpp +14 -0
  449. package/src/duckdb/src/parser/expression/star_expression.cpp +26 -6
  450. package/src/duckdb/src/parser/expression/subquery_expression.cpp +20 -0
  451. package/src/duckdb/src/parser/expression/window_expression.cpp +43 -0
  452. package/src/duckdb/src/parser/parsed_data/alter_info.cpp +7 -3
  453. package/src/duckdb/src/parser/parsed_data/alter_scalar_function_info.cpp +56 -0
  454. package/src/duckdb/src/parser/parsed_data/alter_table_function_info.cpp +51 -0
  455. package/src/duckdb/src/parser/parsed_data/create_scalar_function_info.cpp +3 -2
  456. package/src/duckdb/src/parser/parsed_data/create_table_function_info.cpp +6 -0
  457. package/src/duckdb/src/parser/parsed_data/sample_options.cpp +22 -10
  458. package/src/duckdb/src/parser/parsed_expression.cpp +72 -0
  459. package/src/duckdb/src/parser/parsed_expression_iterator.cpp +15 -1
  460. package/src/duckdb/src/parser/query_node/recursive_cte_node.cpp +21 -0
  461. package/src/duckdb/src/parser/query_node/select_node.cpp +31 -0
  462. package/src/duckdb/src/parser/query_node/set_operation_node.cpp +17 -0
  463. package/src/duckdb/src/parser/query_node.cpp +51 -1
  464. package/src/duckdb/src/parser/result_modifier.cpp +78 -0
  465. package/src/duckdb/src/parser/statement/multi_statement.cpp +18 -0
  466. package/src/duckdb/src/parser/statement/select_statement.cpp +12 -0
  467. package/src/duckdb/src/parser/tableref/basetableref.cpp +21 -0
  468. package/src/duckdb/src/parser/tableref/emptytableref.cpp +4 -0
  469. package/src/duckdb/src/parser/tableref/expressionlistref.cpp +17 -0
  470. package/src/duckdb/src/parser/tableref/joinref.cpp +29 -0
  471. package/src/duckdb/src/parser/tableref/pivotref.cpp +373 -0
  472. package/src/duckdb/src/parser/tableref/subqueryref.cpp +15 -0
  473. package/src/duckdb/src/parser/tableref/table_function.cpp +17 -0
  474. package/src/duckdb/src/parser/tableref.cpp +49 -0
  475. package/src/duckdb/src/parser/transform/expression/transform_array_access.cpp +11 -0
  476. package/src/duckdb/src/parser/transform/expression/transform_bool_expr.cpp +1 -1
  477. package/src/duckdb/src/parser/transform/expression/transform_columnref.cpp +17 -2
  478. package/src/duckdb/src/parser/transform/expression/transform_function.cpp +85 -42
  479. package/src/duckdb/src/parser/transform/expression/transform_operator.cpp +1 -1
  480. package/src/duckdb/src/parser/transform/expression/transform_subquery.cpp +1 -1
  481. package/src/duckdb/src/parser/transform/helpers/transform_alias.cpp +12 -6
  482. package/src/duckdb/src/parser/transform/helpers/transform_cte.cpp +24 -0
  483. package/src/duckdb/src/parser/transform/helpers/transform_groupby.cpp +7 -0
  484. package/src/duckdb/src/parser/transform/helpers/transform_orderby.cpp +0 -7
  485. package/src/duckdb/src/parser/transform/helpers/transform_typename.cpp +3 -2
  486. package/src/duckdb/src/parser/transform/statement/transform_create_function.cpp +4 -0
  487. package/src/duckdb/src/parser/transform/statement/transform_create_view.cpp +4 -0
  488. package/src/duckdb/src/parser/transform/statement/transform_pivot_stmt.cpp +179 -0
  489. package/src/duckdb/src/parser/transform/statement/transform_rename.cpp +3 -4
  490. package/src/duckdb/src/parser/transform/statement/transform_select.cpp +8 -0
  491. package/src/duckdb/src/parser/transform/statement/transform_select_node.cpp +2 -3
  492. package/src/duckdb/src/parser/transform/tableref/transform_join.cpp +12 -1
  493. package/src/duckdb/src/parser/transform/tableref/transform_pivot.cpp +121 -0
  494. package/src/duckdb/src/parser/transform/tableref/transform_tableref.cpp +2 -0
  495. package/src/duckdb/src/parser/transformer.cpp +15 -3
  496. package/src/duckdb/src/planner/bind_context.cpp +18 -25
  497. package/src/duckdb/src/planner/binder/expression/bind_aggregate_expression.cpp +9 -7
  498. package/src/duckdb/src/planner/binder/expression/bind_columnref_expression.cpp +4 -3
  499. package/src/duckdb/src/planner/binder/expression/bind_function_expression.cpp +23 -12
  500. package/src/duckdb/src/planner/binder/expression/bind_lambda.cpp +3 -2
  501. package/src/duckdb/src/planner/binder/expression/bind_star_expression.cpp +176 -0
  502. package/src/duckdb/src/planner/binder/expression/bind_subquery_expression.cpp +4 -0
  503. package/src/duckdb/src/planner/binder/expression/bind_unnest_expression.cpp +163 -24
  504. package/src/duckdb/src/planner/binder/expression/bind_window_expression.cpp +2 -2
  505. package/src/duckdb/src/planner/binder/query_node/bind_select_node.cpp +109 -94
  506. package/src/duckdb/src/planner/binder/query_node/plan_query_node.cpp +11 -0
  507. package/src/duckdb/src/planner/binder/query_node/plan_select_node.cpp +9 -4
  508. package/src/duckdb/src/planner/binder/statement/bind_copy.cpp +5 -3
  509. package/src/duckdb/src/planner/binder/statement/bind_create.cpp +3 -2
  510. package/src/duckdb/src/planner/binder/statement/bind_create_table.cpp +10 -1
  511. package/src/duckdb/src/planner/binder/statement/bind_delete.cpp +1 -1
  512. package/src/duckdb/src/planner/binder/statement/bind_insert.cpp +12 -8
  513. package/src/duckdb/src/planner/binder/statement/bind_logical_plan.cpp +17 -0
  514. package/src/duckdb/src/planner/binder/statement/bind_update.cpp +4 -2
  515. package/src/duckdb/src/planner/binder/tableref/bind_joinref.cpp +19 -3
  516. package/src/duckdb/src/planner/binder/tableref/bind_pivot.cpp +366 -0
  517. package/src/duckdb/src/planner/binder/tableref/bind_table_function.cpp +11 -1
  518. package/src/duckdb/src/planner/binder/tableref/plan_cteref.cpp +1 -0
  519. package/src/duckdb/src/planner/binder/tableref/plan_joinref.cpp +61 -13
  520. package/src/duckdb/src/planner/binder.cpp +19 -24
  521. package/src/duckdb/src/planner/bound_result_modifier.cpp +27 -1
  522. package/src/duckdb/src/planner/expression/bound_aggregate_expression.cpp +9 -2
  523. package/src/duckdb/src/planner/expression/bound_expression.cpp +4 -0
  524. package/src/duckdb/src/planner/expression/bound_window_expression.cpp +1 -1
  525. package/src/duckdb/src/planner/expression_binder/base_select_binder.cpp +146 -0
  526. package/src/duckdb/src/planner/expression_binder/having_binder.cpp +6 -3
  527. package/src/duckdb/src/planner/expression_binder/qualify_binder.cpp +3 -3
  528. package/src/duckdb/src/planner/expression_binder/select_binder.cpp +1 -132
  529. package/src/duckdb/src/planner/expression_binder.cpp +10 -3
  530. package/src/duckdb/src/planner/expression_iterator.cpp +17 -10
  531. package/src/duckdb/src/planner/filter/constant_filter.cpp +4 -6
  532. package/src/duckdb/src/planner/logical_operator.cpp +7 -2
  533. package/src/duckdb/src/planner/logical_operator_visitor.cpp +6 -0
  534. package/src/duckdb/src/planner/operator/logical_asof_join.cpp +8 -0
  535. package/src/duckdb/src/planner/operator/logical_distinct.cpp +3 -0
  536. package/src/duckdb/src/planner/planner.cpp +2 -1
  537. package/src/duckdb/src/planner/pragma_handler.cpp +10 -2
  538. package/src/duckdb/src/planner/subquery/flatten_dependent_join.cpp +3 -1
  539. package/src/duckdb/src/storage/buffer_manager.cpp +44 -46
  540. package/src/duckdb/src/storage/checkpoint/row_group_writer.cpp +1 -1
  541. package/src/duckdb/src/storage/checkpoint/table_data_reader.cpp +4 -15
  542. package/src/duckdb/src/storage/checkpoint/table_data_writer.cpp +10 -4
  543. package/src/duckdb/src/storage/checkpoint_manager.cpp +9 -3
  544. package/src/duckdb/src/storage/compression/bitpacking.cpp +29 -25
  545. package/src/duckdb/src/storage/compression/fixed_size_uncompressed.cpp +45 -46
  546. package/src/duckdb/src/storage/compression/numeric_constant.cpp +10 -11
  547. package/src/duckdb/src/storage/compression/patas.cpp +1 -1
  548. package/src/duckdb/src/storage/compression/rle.cpp +20 -15
  549. package/src/duckdb/src/storage/compression/validity_uncompressed.cpp +6 -6
  550. package/src/duckdb/src/storage/data_table.cpp +23 -23
  551. package/src/duckdb/src/storage/index.cpp +12 -1
  552. package/src/duckdb/src/storage/local_storage.cpp +27 -23
  553. package/src/duckdb/src/storage/meta_block_reader.cpp +22 -0
  554. package/src/duckdb/src/storage/statistics/base_statistics.cpp +373 -128
  555. package/src/duckdb/src/storage/statistics/column_statistics.cpp +57 -3
  556. package/src/duckdb/src/storage/statistics/distinct_statistics.cpp +8 -9
  557. package/src/duckdb/src/storage/statistics/list_stats.cpp +121 -0
  558. package/src/duckdb/src/storage/statistics/numeric_stats.cpp +591 -0
  559. package/src/duckdb/src/storage/statistics/numeric_stats_union.cpp +65 -0
  560. package/src/duckdb/src/storage/statistics/segment_statistics.cpp +2 -11
  561. package/src/duckdb/src/storage/statistics/string_stats.cpp +273 -0
  562. package/src/duckdb/src/storage/statistics/struct_stats.cpp +133 -0
  563. package/src/duckdb/src/storage/storage_info.cpp +2 -2
  564. package/src/duckdb/src/storage/table/column_checkpoint_state.cpp +4 -10
  565. package/src/duckdb/src/storage/table/column_data.cpp +118 -62
  566. package/src/duckdb/src/storage/table/column_data_checkpointer.cpp +10 -9
  567. package/src/duckdb/src/storage/table/column_segment.cpp +30 -45
  568. package/src/duckdb/src/storage/table/list_column_data.cpp +50 -71
  569. package/src/duckdb/src/storage/table/persistent_table_data.cpp +2 -1
  570. package/src/duckdb/src/storage/table/row_group.cpp +213 -143
  571. package/src/duckdb/src/storage/table/row_group_collection.cpp +151 -105
  572. package/src/duckdb/src/storage/table/scan_state.cpp +45 -33
  573. package/src/duckdb/src/storage/table/standard_column_data.cpp +11 -12
  574. package/src/duckdb/src/storage/table/struct_column_data.cpp +27 -34
  575. package/src/duckdb/src/storage/table/table_statistics.cpp +27 -7
  576. package/src/duckdb/src/storage/table/update_segment.cpp +23 -18
  577. package/src/duckdb/src/storage/wal_replay.cpp +8 -5
  578. package/src/duckdb/src/storage/write_ahead_log.cpp +2 -2
  579. package/src/duckdb/src/transaction/commit_state.cpp +11 -7
  580. package/src/duckdb/src/verification/deserialized_statement_verifier.cpp +0 -1
  581. package/src/duckdb/third_party/libpg_query/include/nodes/nodes.hpp +35 -0
  582. package/src/duckdb/third_party/libpg_query/include/nodes/parsenodes.hpp +36 -2
  583. package/src/duckdb/third_party/libpg_query/include/nodes/primnodes.hpp +3 -3
  584. package/src/duckdb/third_party/libpg_query/include/parser/gram.hpp +1022 -530
  585. package/src/duckdb/third_party/libpg_query/include/parser/kwlist.hpp +8 -0
  586. package/src/duckdb/third_party/libpg_query/src_backend_parser_gram.cpp +24462 -22828
  587. package/src/duckdb/third_party/re2/re2/re2.cc +9 -0
  588. package/src/duckdb/third_party/re2/re2/re2.h +2 -0
  589. package/src/duckdb/ub_extension_icu_third_party_icu_i18n.cpp +4 -4
  590. package/src/duckdb/ub_extension_json_json_functions.cpp +2 -0
  591. package/src/duckdb/ub_src_common_serializer.cpp +2 -0
  592. package/src/duckdb/ub_src_common_types.cpp +2 -0
  593. package/src/duckdb/ub_src_execution_physical_plan.cpp +2 -0
  594. package/src/duckdb/ub_src_function_aggregate_distributive.cpp +2 -0
  595. package/src/duckdb/ub_src_function_scalar_bit.cpp +2 -0
  596. package/src/duckdb/ub_src_function_scalar_map.cpp +4 -0
  597. package/src/duckdb/ub_src_function_scalar_string.cpp +2 -0
  598. package/src/duckdb/ub_src_function_scalar_string_regexp.cpp +4 -0
  599. package/src/duckdb/ub_src_main_capi.cpp +2 -0
  600. package/src/duckdb/ub_src_optimizer_rule.cpp +2 -0
  601. package/src/duckdb/ub_src_parser.cpp +2 -0
  602. package/src/duckdb/ub_src_parser_parsed_data.cpp +4 -2
  603. package/src/duckdb/ub_src_parser_statement.cpp +2 -0
  604. package/src/duckdb/ub_src_parser_tableref.cpp +2 -0
  605. package/src/duckdb/ub_src_parser_transform_statement.cpp +2 -0
  606. package/src/duckdb/ub_src_parser_transform_tableref.cpp +2 -0
  607. package/src/duckdb/ub_src_planner_binder_expression.cpp +2 -0
  608. package/src/duckdb/ub_src_planner_binder_tableref.cpp +2 -0
  609. package/src/duckdb/ub_src_planner_expression_binder.cpp +2 -0
  610. package/src/duckdb/ub_src_planner_operator.cpp +2 -0
  611. package/src/duckdb/ub_src_storage_statistics.cpp +6 -6
  612. package/src/duckdb/ub_src_storage_table.cpp +0 -2
  613. package/src/duckdb_node.hpp +2 -1
  614. package/src/statement.cpp +5 -5
  615. package/src/utils.cpp +27 -2
  616. package/test/extension.test.ts +44 -26
  617. package/test/syntax_error.test.ts +3 -1
  618. package/filelist.cache +0 -0
  619. package/src/duckdb/src/include/duckdb/main/loadable_extension.hpp +0 -59
  620. package/src/duckdb/src/include/duckdb/storage/statistics/list_statistics.hpp +0 -36
  621. package/src/duckdb/src/include/duckdb/storage/statistics/numeric_statistics.hpp +0 -75
  622. package/src/duckdb/src/include/duckdb/storage/statistics/string_statistics.hpp +0 -49
  623. package/src/duckdb/src/include/duckdb/storage/statistics/struct_statistics.hpp +0 -36
  624. package/src/duckdb/src/include/duckdb/storage/statistics/validity_statistics.hpp +0 -45
  625. package/src/duckdb/src/parser/parsed_data/alter_function_info.cpp +0 -55
  626. package/src/duckdb/src/storage/statistics/list_statistics.cpp +0 -94
  627. package/src/duckdb/src/storage/statistics/numeric_statistics.cpp +0 -307
  628. package/src/duckdb/src/storage/statistics/string_statistics.cpp +0 -220
  629. package/src/duckdb/src/storage/statistics/struct_statistics.cpp +0 -108
  630. package/src/duckdb/src/storage/statistics/validity_statistics.cpp +0 -91
  631. package/src/duckdb/src/storage/table/segment_tree.cpp +0 -179
@@ -3,17 +3,51 @@
3
3
  #include "duckdb/execution/expression_executor.hpp"
4
4
  #include "duckdb/main/client_context.hpp"
5
5
  #include "duckdb/storage/data_table.hpp"
6
- #include "duckdb/transaction/transaction.hpp"
7
6
  #include "duckdb/planner/constraints/bound_not_null_constraint.hpp"
8
7
  #include "duckdb/storage/checkpoint/table_data_writer.hpp"
8
+ #include "duckdb/storage/table/row_group_segment_tree.hpp"
9
+ #include "duckdb/storage/meta_block_reader.hpp"
10
+ #include "duckdb/storage/table/append_state.hpp"
11
+ #include "duckdb/storage/table/scan_state.hpp"
9
12
 
10
13
  namespace duckdb {
11
14
 
15
+ //===--------------------------------------------------------------------===//
16
+ // Row Group Segment Tree
17
+ //===--------------------------------------------------------------------===//
18
+ RowGroupSegmentTree::RowGroupSegmentTree(RowGroupCollection &collection)
19
+ : SegmentTree<RowGroup, true>(), collection(collection), current_row_group(0), max_row_group(0) {
20
+ }
21
+ RowGroupSegmentTree::~RowGroupSegmentTree() {
22
+ }
23
+
24
+ void RowGroupSegmentTree::Initialize(PersistentTableData &data) {
25
+ D_ASSERT(data.row_group_count > 0);
26
+ current_row_group = 0;
27
+ max_row_group = data.row_group_count;
28
+ finished_loading = false;
29
+ reader = make_unique<MetaBlockReader>(collection.GetBlockManager(), data.block_id);
30
+ reader->offset = data.offset;
31
+ }
32
+
33
+ unique_ptr<RowGroup> RowGroupSegmentTree::LoadSegment() {
34
+ if (current_row_group >= max_row_group) {
35
+ finished_loading = true;
36
+ return nullptr;
37
+ }
38
+ auto row_group_pointer = RowGroup::Deserialize(*reader, collection.GetTypes());
39
+ current_row_group++;
40
+ return make_unique<RowGroup>(collection, std::move(row_group_pointer));
41
+ }
42
+
43
+ //===--------------------------------------------------------------------===//
44
+ // Row Group Collection
45
+ //===--------------------------------------------------------------------===//
12
46
  RowGroupCollection::RowGroupCollection(shared_ptr<DataTableInfo> info_p, BlockManager &block_manager,
13
47
  vector<LogicalType> types_p, idx_t row_start_p, idx_t total_rows_p)
14
48
  : block_manager(block_manager), total_rows(total_rows_p), info(std::move(info_p)), types(std::move(types_p)),
15
49
  row_start(row_start_p) {
16
- row_groups = make_shared<SegmentTree>();
50
+ row_groups = make_shared<RowGroupSegmentTree>(*this);
17
51
  }
18
52
 
19
53
  idx_t RowGroupCollection::GetTotalRows() const {
@@ -28,20 +62,22 @@ Allocator &RowGroupCollection::GetAllocator() const {
28
62
  return Allocator::Get(info->db);
29
63
  }
30
64
 
65
+ AttachedDatabase &RowGroupCollection::GetAttached() {
66
+ return GetTableInfo().db;
67
+ }
68
+
69
+ DatabaseInstance &RowGroupCollection::GetDatabase() {
70
+ return GetAttached().GetDatabase();
71
+ }
72
+
31
73
  //===--------------------------------------------------------------------===//
32
74
  // Initialize
33
75
  //===--------------------------------------------------------------------===//
34
76
  void RowGroupCollection::Initialize(PersistentTableData &data) {
35
77
  D_ASSERT(this->row_start == 0);
36
78
  auto l = row_groups->Lock();
37
- for (auto &row_group_pointer : data.row_groups) {
38
- auto new_row_group = make_unique<RowGroup>(info->db, block_manager, *info, types, std::move(row_group_pointer));
39
- auto row_group_count = new_row_group->start + new_row_group->count;
40
- if (row_group_count > this->total_rows) {
41
- this->total_rows = row_group_count;
42
- }
43
- row_groups->AppendSegment(l, std::move(new_row_group));
44
- }
79
+ this->total_rows = data.total_rows;
80
+ row_groups->Initialize(data);
45
81
  stats.Initialize(types, data);
46
82
  }
47
83
 
@@ -51,7 +87,7 @@ void RowGroupCollection::InitializeEmpty() {
51
87
 
52
88
  void RowGroupCollection::AppendRowGroup(SegmentLock &l, idx_t start_row) {
53
89
  D_ASSERT(start_row >= row_start);
54
- auto new_row_group = make_unique<RowGroup>(info->db, block_manager, *info, start_row, 0);
90
+ auto new_row_group = make_unique<RowGroup>(*this, start_row, 0);
55
91
  new_row_group->InitializeEmpty(types);
56
92
  row_groups->AppendSegment(l, std::move(new_row_group));
57
93
  }
@@ -64,9 +100,9 @@ void RowGroupCollection::Verify() {
64
100
  #ifdef DEBUG
65
101
  idx_t current_total_rows = 0;
66
102
  row_groups->Verify();
67
- for (auto segment = row_groups->GetRootSegment(); segment; segment = segment->Next()) {
68
- auto &row_group = (RowGroup &)*segment;
103
+ for (auto &row_group : row_groups->Segments()) {
69
104
  row_group.Verify();
105
+ D_ASSERT(&row_group.GetCollection() == this);
70
106
  D_ASSERT(row_group.start == this->row_start + current_total_rows);
71
107
  current_total_rows += row_group.count;
72
108
  }
@@ -79,11 +115,13 @@ void RowGroupCollection::Verify() {
79
115
  //===--------------------------------------------------------------------===//
80
116
  void RowGroupCollection::InitializeScan(CollectionScanState &state, const vector<column_t> &column_ids,
81
117
  TableFilterSet *table_filters) {
82
- auto row_group = (RowGroup *)row_groups->GetRootSegment();
118
+ auto row_group = row_groups->GetRootSegment();
83
119
  D_ASSERT(row_group);
120
+ state.row_groups = row_groups.get();
84
121
  state.max_row = row_start + total_rows;
85
- while (row_group && !row_group->InitializeScan(state.row_group_state)) {
86
- row_group = (RowGroup *)row_group->Next();
122
+ state.Initialize(GetTypes());
123
+ while (row_group && !row_group->InitializeScan(state)) {
124
+ row_group = row_groups->GetNextSegment(row_group);
87
125
  }
88
126
  }
89
127
 
@@ -93,57 +131,80 @@ void RowGroupCollection::InitializeCreateIndexScan(CreateIndexScanState &state)
93
131
 
94
132
  void RowGroupCollection::InitializeScanWithOffset(CollectionScanState &state, const vector<column_t> &column_ids,
95
133
  idx_t start_row, idx_t end_row) {
96
- auto row_group = (RowGroup *)row_groups->GetSegment(start_row);
134
+ auto row_group = row_groups->GetSegment(start_row);
97
135
  D_ASSERT(row_group);
136
+ state.row_groups = row_groups.get();
98
137
  state.max_row = end_row;
138
+ state.Initialize(GetTypes());
99
139
  idx_t start_vector = (start_row - row_group->start) / STANDARD_VECTOR_SIZE;
100
- if (!row_group->InitializeScanWithOffset(state.row_group_state, start_vector)) {
140
+ if (!row_group->InitializeScanWithOffset(state, start_vector)) {
101
141
  throw InternalException("Failed to initialize row group scan with offset");
102
142
  }
103
143
  }
104
144
 
105
- bool RowGroupCollection::InitializeScanInRowGroup(CollectionScanState &state, RowGroup *row_group, idx_t vector_index,
106
- idx_t max_row) {
145
+ bool RowGroupCollection::InitializeScanInRowGroup(CollectionScanState &state, RowGroupCollection &collection,
146
+ RowGroup &row_group, idx_t vector_index, idx_t max_row) {
107
147
  state.max_row = max_row;
108
- return row_group->InitializeScanWithOffset(state.row_group_state, vector_index);
148
+ state.row_groups = collection.row_groups.get();
149
+ if (!state.column_scans) {
150
+ // initialize the scan state
151
+ state.Initialize(collection.GetTypes());
152
+ }
153
+ return row_group.InitializeScanWithOffset(state, vector_index);
109
154
  }
110
155
 
111
156
  void RowGroupCollection::InitializeParallelScan(ParallelCollectionScanState &state) {
112
- state.current_row_group = (RowGroup *)row_groups->GetRootSegment();
157
+ state.collection = this;
158
+ state.current_row_group = row_groups->GetRootSegment();
113
159
  state.vector_index = 0;
114
160
  state.max_row = row_start + total_rows;
115
161
  state.batch_index = 0;
162
+ state.processed_rows = 0;
116
163
  }
117
164
 
118
165
  bool RowGroupCollection::NextParallelScan(ClientContext &context, ParallelCollectionScanState &state,
119
166
  CollectionScanState &scan_state) {
120
- while (state.current_row_group && state.current_row_group->count > 0) {
167
+ while (true) {
121
168
  idx_t vector_index;
122
169
  idx_t max_row;
123
- if (ClientConfig::GetConfig(context).verify_parallelism) {
124
- vector_index = state.vector_index;
125
- max_row = state.current_row_group->start +
126
- MinValue<idx_t>(state.current_row_group->count,
127
- STANDARD_VECTOR_SIZE * state.vector_index + STANDARD_VECTOR_SIZE);
128
- D_ASSERT(vector_index * STANDARD_VECTOR_SIZE < state.current_row_group->count);
129
- } else {
130
- vector_index = 0;
131
- max_row = state.current_row_group->start + state.current_row_group->count;
132
- }
133
- max_row = MinValue<idx_t>(max_row, state.max_row);
134
- bool need_to_scan = InitializeScanInRowGroup(scan_state, state.current_row_group, vector_index, max_row);
135
- if (ClientConfig::GetConfig(context).verify_parallelism) {
136
- state.vector_index++;
137
- if (state.vector_index * STANDARD_VECTOR_SIZE >= state.current_row_group->count) {
138
- state.current_row_group = (RowGroup *)state.current_row_group->Next();
139
- state.vector_index = 0;
170
+ RowGroupCollection *collection;
171
+ RowGroup *row_group;
172
+ {
173
+ // select the next row group to scan from the parallel state
174
+ lock_guard<mutex> l(state.lock);
175
+ if (!state.current_row_group || state.current_row_group->count == 0) {
176
+ // no more data left to scan
177
+ break;
140
178
  }
141
- } else {
142
- state.current_row_group = (RowGroup *)state.current_row_group->Next();
179
+ collection = state.collection;
180
+ row_group = state.current_row_group;
181
+ if (ClientConfig::GetConfig(context).verify_parallelism) {
182
+ vector_index = state.vector_index;
183
+ max_row = state.current_row_group->start +
184
+ MinValue<idx_t>(state.current_row_group->count,
185
+ STANDARD_VECTOR_SIZE * state.vector_index + STANDARD_VECTOR_SIZE);
186
+ D_ASSERT(vector_index * STANDARD_VECTOR_SIZE < state.current_row_group->count);
187
+ state.vector_index++;
188
+ if (state.vector_index * STANDARD_VECTOR_SIZE >= state.current_row_group->count) {
189
+ state.current_row_group = row_groups->GetNextSegment(state.current_row_group);
190
+ state.vector_index = 0;
191
+ }
192
+ } else {
193
+ state.processed_rows += state.current_row_group->count;
194
+ vector_index = 0;
195
+ max_row = state.current_row_group->start + state.current_row_group->count;
196
+ state.current_row_group = row_groups->GetNextSegment(state.current_row_group);
197
+ }
198
+ max_row = MinValue<idx_t>(max_row, state.max_row);
199
+ scan_state.batch_index = ++state.batch_index;
143
200
  }
144
- scan_state.batch_index = ++state.batch_index;
201
+ D_ASSERT(collection);
202
+ D_ASSERT(row_group);
203
+
204
+ // initialize the scan for this row group
205
+ bool need_to_scan = InitializeScanInRowGroup(scan_state, *collection, *row_group, vector_index, max_row);
145
206
  if (!need_to_scan) {
146
- // filters allow us to skip this row group: move to the next row group
207
+ // skip this row group
147
208
  continue;
148
209
  }
149
210
  return true;
@@ -204,7 +265,7 @@ void RowGroupCollection::Fetch(TransactionData transaction, DataChunk &result, c
204
265
  // in parallel append scenarios it is possible for the row_id
205
266
  continue;
206
267
  }
207
- row_group = (RowGroup *)row_groups->GetSegmentByIndex(l, segment_index);
268
+ row_group = row_groups->GetSegmentByIndex(l, segment_index);
208
269
  }
209
270
  if (!row_group->Fetch(transaction, row_id - row_group->start)) {
210
271
  continue;
@@ -246,7 +307,7 @@ void RowGroupCollection::InitializeAppend(TransactionData transaction, TableAppe
246
307
  // empty row group collection: empty first row group
247
308
  AppendRowGroup(l, row_start);
248
309
  }
249
- state.start_row_group = (RowGroup *)row_groups->GetLastSegment(l);
310
+ state.start_row_group = row_groups->GetLastSegment(l);
250
311
  D_ASSERT(this->row_start + total_rows == state.start_row_group->start + state.start_row_group->count);
251
312
  state.start_row_group->InitializeAppend(state.row_group_append_state);
252
313
  state.remaining = append_count;
@@ -280,7 +341,7 @@ bool RowGroupCollection::Append(DataChunk &chunk, TableAppendState &state) {
280
341
  // merge the stats
281
342
  auto stats_lock = stats.GetLock();
282
343
  for (idx_t i = 0; i < types.size(); i++) {
283
- current_row_group->MergeIntoStatistics(i, *stats.GetStats(i).stats);
344
+ current_row_group->MergeIntoStatistics(i, stats.GetStats(i).Statistics());
284
345
  }
285
346
  }
286
347
  remaining -= append_count;
@@ -306,7 +367,7 @@ bool RowGroupCollection::Append(DataChunk &chunk, TableAppendState &state) {
306
367
  auto l = row_groups->Lock();
307
368
  AppendRowGroup(l, next_start);
308
369
  // set up the append state for this row_group
309
- auto last_row_group = (RowGroup *)row_groups->GetLastSegment(l);
370
+ auto last_row_group = row_groups->GetLastSegment(l);
310
371
  last_row_group->InitializeAppend(state.row_group_append_state);
311
372
  if (state.remaining > 0) {
312
373
  last_row_group->AppendVersionInfo(state.transaction, state.remaining);
@@ -319,11 +380,7 @@ bool RowGroupCollection::Append(DataChunk &chunk, TableAppendState &state) {
319
380
  state.current_row += append_count;
320
381
  auto stats_lock = stats.GetLock();
321
382
  for (idx_t col_idx = 0; col_idx < types.size(); col_idx++) {
322
- auto type = types[col_idx].InternalType();
323
- if (type == PhysicalType::LIST || type == PhysicalType::STRUCT) {
324
- continue;
325
- }
326
- stats.GetStats(col_idx).stats->UpdateDistinctStatistics(chunk.data[col_idx], chunk.size());
383
+ stats.GetStats(col_idx).UpdateDistinctStatistics(chunk.data[col_idx], chunk.size());
327
384
  }
328
385
  return new_row_group;
329
386
  }
@@ -335,7 +392,7 @@ void RowGroupCollection::FinalizeAppend(TransactionData transaction, TableAppend
335
392
  auto append_count = MinValue<idx_t>(remaining, RowGroup::ROW_GROUP_SIZE - row_group->count);
336
393
  row_group->AppendVersionInfo(transaction, append_count);
337
394
  remaining -= append_count;
338
- row_group = (RowGroup *)row_group->Next();
395
+ row_group = row_groups->GetNextSegment(row_group);
339
396
  }
340
397
  total_rows += state.total_append_count;
341
398
 
@@ -346,7 +403,7 @@ void RowGroupCollection::FinalizeAppend(TransactionData transaction, TableAppend
346
403
  }
347
404
 
348
405
  void RowGroupCollection::CommitAppend(transaction_t commit_id, idx_t row_start, idx_t count) {
349
- auto row_group = (RowGroup *)row_groups->GetSegment(row_start);
406
+ auto row_group = row_groups->GetSegment(row_start);
350
407
  D_ASSERT(row_group);
351
408
  idx_t current_row = row_start;
352
409
  idx_t remaining = count;
@@ -361,7 +418,7 @@ void RowGroupCollection::CommitAppend(transaction_t commit_id, idx_t row_start,
361
418
  if (remaining == 0) {
362
419
  break;
363
420
  }
364
- row_group = (RowGroup *)row_group->Next();
421
+ row_group = row_groups->GetNextSegment(row_group);
365
422
  }
366
423
  }
367
424
 
@@ -375,7 +432,7 @@ void RowGroupCollection::RevertAppendInternal(idx_t start_row, idx_t count) {
375
432
  // find the segment index that the current row belongs to
376
433
  idx_t segment_index = row_groups->GetSegmentIndex(l, start_row);
377
434
  auto segment = row_groups->GetSegmentByIndex(l, segment_index);
378
- auto &info = (RowGroup &)*segment;
435
+ auto &info = *segment;
379
436
 
380
437
  // remove any segments AFTER this segment: they should be deleted entirely
381
438
  row_groups->EraseSegments(l, segment_index);
@@ -387,9 +444,8 @@ void RowGroupCollection::RevertAppendInternal(idx_t start_row, idx_t count) {
387
444
  void RowGroupCollection::MergeStorage(RowGroupCollection &data) {
388
445
  D_ASSERT(data.types == types);
389
446
  auto index = row_start + total_rows.load();
390
- for (auto segment = data.row_groups->GetRootSegment(); segment; segment = segment->Next()) {
391
- auto &row_group = (RowGroup &)*segment;
392
- auto new_group = make_unique<RowGroup>(row_group, index);
447
+ for (auto &row_group : data.row_groups->Segments()) {
448
+ auto new_group = make_unique<RowGroup>(row_group, *this, index);
393
449
  index += new_group->count;
394
450
  row_groups->AppendSegment(std::move(new_group));
395
451
  }
@@ -409,7 +465,7 @@ idx_t RowGroupCollection::Delete(TransactionData transaction, DataTable *table,
409
465
  idx_t pos = 0;
410
466
  do {
411
467
  idx_t start = pos;
412
- auto row_group = (RowGroup *)row_groups->GetSegment(ids[start]);
468
+ auto row_group = row_groups->GetSegment(ids[start]);
413
469
  for (pos++; pos < count; pos++) {
414
470
  D_ASSERT(ids[pos] >= 0);
415
471
  // check if this id still belongs to this row group
@@ -435,7 +491,7 @@ void RowGroupCollection::Update(TransactionData transaction, row_t *ids, const v
435
491
  idx_t pos = 0;
436
492
  do {
437
493
  idx_t start = pos;
438
- auto row_group = (RowGroup *)row_groups->GetSegment(ids[pos]);
494
+ auto row_group = row_groups->GetSegment(ids[pos]);
439
495
  row_t base_id =
440
496
  row_group->start + ((ids[pos] - row_group->start) / STANDARD_VECTOR_SIZE * STANDARD_VECTOR_SIZE);
441
497
  row_t max_id = MinValue<row_t>(base_id + STANDARD_VECTOR_SIZE, row_group->start + row_group->count);
@@ -465,7 +521,7 @@ void RowGroupCollection::RemoveFromIndexes(TableIndexList &indexes, Vector &row_
465
521
  auto row_ids = FlatVector::GetData<row_t>(row_identifiers);
466
522
 
467
523
  // figure out which row_group to fetch from
468
- auto row_group = (RowGroup *)row_groups->GetSegment(row_ids[0]);
524
+ auto row_group = row_groups->GetSegment(row_ids[0]);
469
525
  auto row_group_vector_idx = (row_ids[0] - row_group->start) / STANDARD_VECTOR_SIZE;
470
526
  auto base_row_id = row_group_vector_idx * STANDARD_VECTOR_SIZE + row_group->start;
471
527
 
@@ -492,8 +548,9 @@ void RowGroupCollection::RemoveFromIndexes(TableIndexList &indexes, Vector &row_
492
548
  DataChunk result;
493
549
  result.Initialize(GetAllocator(), types);
494
550
 
495
- row_group->InitializeScanWithOffset(state.table_state.row_group_state, row_group_vector_idx);
496
- row_group->ScanCommitted(state.table_state.row_group_state, result, TableScanType::TABLE_SCAN_COMMITTED_ROWS);
551
+ state.table_state.Initialize(GetTypes());
552
+ row_group->InitializeScanWithOffset(state.table_state, row_group_vector_idx);
553
+ row_group->ScanCommitted(state.table_state, result, TableScanType::TABLE_SCAN_COMMITTED_ROWS);
497
554
  result.Slice(sel, count);
498
555
 
499
556
  indexes.Scan([&](Index &index) {
@@ -510,20 +567,19 @@ void RowGroupCollection::UpdateColumn(TransactionData transaction, Vector &row_i
510
567
  }
511
568
  // find the row_group this id belongs to
512
569
  auto primary_column_idx = column_path[0];
513
- auto row_group = (RowGroup *)row_groups->GetSegment(first_id);
570
+ auto row_group = row_groups->GetSegment(first_id);
514
571
  row_group->UpdateColumn(transaction, updates, row_ids, column_path);
515
572
 
516
- row_group->MergeIntoStatistics(primary_column_idx, *stats.GetStats(primary_column_idx).stats);
573
+ row_group->MergeIntoStatistics(primary_column_idx, stats.GetStats(primary_column_idx).Statistics());
517
574
  }
518
575
 
519
576
  //===--------------------------------------------------------------------===//
520
577
  // Checkpoint
521
578
  //===--------------------------------------------------------------------===//
522
- void RowGroupCollection::Checkpoint(TableDataWriter &writer, vector<unique_ptr<BaseStatistics>> &global_stats) {
523
- for (auto row_group = (RowGroup *)row_groups->GetRootSegment(); row_group;
524
- row_group = (RowGroup *)row_group->Next()) {
525
- auto rowg_writer = writer.GetRowGroupWriter(*row_group);
526
- auto pointer = row_group->Checkpoint(*rowg_writer, global_stats);
579
+ void RowGroupCollection::Checkpoint(TableDataWriter &writer, TableStatistics &global_stats) {
580
+ for (auto &row_group : row_groups->Segments()) {
581
+ auto rowg_writer = writer.GetRowGroupWriter(row_group);
582
+ auto pointer = row_group.Checkpoint(*rowg_writer, global_stats);
527
583
  writer.AddRowGroup(std::move(pointer), std::move(rowg_writer));
528
584
  }
529
585
  }
@@ -532,18 +588,14 @@ void RowGroupCollection::Checkpoint(TableDataWriter &writer, vector<unique_ptr<B
532
588
  // CommitDrop
533
589
  //===--------------------------------------------------------------------===//
534
590
  void RowGroupCollection::CommitDropColumn(idx_t index) {
535
- auto segment = (RowGroup *)row_groups->GetRootSegment();
536
- while (segment) {
537
- segment->CommitDropColumn(index);
538
- segment = (RowGroup *)segment->Next();
591
+ for (auto &row_group : row_groups->Segments()) {
592
+ row_group.CommitDropColumn(index);
539
593
  }
540
594
  }
541
595
 
542
596
  void RowGroupCollection::CommitDropTable() {
543
- auto segment = (RowGroup *)row_groups->GetRootSegment();
544
- while (segment) {
545
- segment->CommitDrop();
546
- segment = (RowGroup *)segment->Next();
597
+ for (auto &row_group : row_groups->Segments()) {
598
+ row_group.CommitDrop();
547
599
  }
548
600
  }
549
601
 
@@ -551,13 +603,8 @@ void RowGroupCollection::CommitDropTable() {
551
603
  // GetStorageInfo
552
604
  //===--------------------------------------------------------------------===//
553
605
  void RowGroupCollection::GetStorageInfo(TableStorageInfo &result) {
554
- auto row_group = (RowGroup *)row_groups->GetRootSegment();
555
- idx_t row_group_index = 0;
556
- while (row_group) {
557
- row_group->GetStorageInfo(row_group_index, result);
558
- row_group_index++;
559
-
560
- row_group = (RowGroup *)row_group->Next();
606
+ for (auto &row_group : row_groups->Segments()) {
607
+ row_group.GetStorageInfo(row_group.index, result);
561
608
  }
562
609
  }
563
610
 
@@ -586,14 +633,12 @@ shared_ptr<RowGroupCollection> RowGroupCollection::AddColumn(ClientContext &cont
586
633
 
587
634
  // fill the column with its DEFAULT value, or NULL if none is specified
588
635
  auto new_stats = make_unique<SegmentStatistics>(new_column.GetType());
589
- auto current_row_group = (RowGroup *)row_groups->GetRootSegment();
590
- while (current_row_group) {
591
- auto new_row_group = current_row_group->AddColumn(new_column, executor, default_value, default_vector);
636
+ for (auto &current_row_group : row_groups->Segments()) {
637
+ auto new_row_group = current_row_group.AddColumn(*result, new_column, executor, default_value, default_vector);
592
638
  // merge in the statistics
593
- new_row_group->MergeIntoStatistics(new_column_idx, *new_column_stats.stats);
639
+ new_row_group->MergeIntoStatistics(new_column_idx, new_column_stats.Statistics());
594
640
 
595
641
  result->row_groups->AppendSegment(std::move(new_row_group));
596
- current_row_group = (RowGroup *)current_row_group->Next();
597
642
  }
598
643
  return result;
599
644
  }
@@ -607,11 +652,9 @@ shared_ptr<RowGroupCollection> RowGroupCollection::RemoveColumn(idx_t col_idx) {
607
652
  make_shared<RowGroupCollection>(info, block_manager, std::move(new_types), row_start, total_rows.load());
608
653
  result->stats.InitializeRemoveColumn(stats, col_idx);
609
654
 
610
- auto current_row_group = (RowGroup *)row_groups->GetRootSegment();
611
- while (current_row_group) {
612
- auto new_row_group = current_row_group->RemoveColumn(col_idx);
655
+ for (auto &current_row_group : row_groups->Segments()) {
656
+ auto new_row_group = current_row_group.RemoveColumn(*result, col_idx);
613
657
  result->row_groups->AppendSegment(std::move(new_row_group));
614
- current_row_group = (RowGroup *)current_row_group->Next();
615
658
  }
616
659
  return result;
617
660
  }
@@ -646,14 +689,12 @@ shared_ptr<RowGroupCollection> RowGroupCollection::AlterType(ClientContext &cont
646
689
  scan_state.table_state.max_row = row_start + total_rows;
647
690
 
648
691
  // now alter the type of the column within all of the row_groups individually
649
- auto current_row_group = (RowGroup *)row_groups->GetRootSegment();
650
692
  auto &changed_stats = result->stats.GetStats(changed_idx);
651
- while (current_row_group) {
652
- auto new_row_group = current_row_group->AlterType(target_type, changed_idx, executor,
653
- scan_state.table_state.row_group_state, scan_chunk);
654
- new_row_group->MergeIntoStatistics(changed_idx, *changed_stats.stats);
693
+ for (auto &current_row_group : row_groups->Segments()) {
694
+ auto new_row_group = current_row_group.AlterType(*result, target_type, changed_idx, executor,
695
+ scan_state.table_state, scan_chunk);
696
+ new_row_group->MergeIntoStatistics(changed_idx, changed_stats.Statistics());
655
697
  result->row_groups->AppendSegment(std::move(new_row_group));
656
- current_row_group = (RowGroup *)current_row_group->Next();
657
698
  }
658
699
 
659
700
  return result;
@@ -681,7 +722,8 @@ void RowGroupCollection::VerifyNewConstraint(DataTable &parent, const BoundConst
681
722
  InitializeCreateIndexScan(state);
682
723
  while (true) {
683
724
  scan_chunk.Reset();
684
- state.table_state.ScanCommitted(scan_chunk, TableScanType::TABLE_SCAN_COMMITTED_ROWS_OMIT_PERMANENTLY_DELETED);
725
+ state.table_state.ScanCommitted(scan_chunk, state.segment_lock,
726
+ TableScanType::TABLE_SCAN_COMMITTED_ROWS_OMIT_PERMANENTLY_DELETED);
685
727
  if (scan_chunk.size() == 0) {
686
728
  break;
687
729
  }
@@ -696,14 +738,18 @@ void RowGroupCollection::VerifyNewConstraint(DataTable &parent, const BoundConst
696
738
  //===--------------------------------------------------------------------===//
697
739
  // Statistics
698
740
  //===--------------------------------------------------------------------===//
741
+ void RowGroupCollection::CopyStats(TableStatistics &other_stats) {
742
+ stats.CopyStats(other_stats);
743
+ }
744
+
699
745
  unique_ptr<BaseStatistics> RowGroupCollection::CopyStats(column_t column_id) {
700
746
  return stats.CopyStats(column_id);
701
747
  }
702
748
 
703
- void RowGroupCollection::SetStatistics(column_t column_id, const std::function<void(BaseStatistics &)> &set_fun) {
749
+ void RowGroupCollection::SetDistinct(column_t column_id, unique_ptr<DistinctStatistics> distinct_stats) {
704
750
  D_ASSERT(column_id != COLUMN_IDENTIFIER_ROW_ID);
705
751
  auto stats_guard = stats.GetLock();
706
- set_fun(*stats.GetStats(column_id).stats);
752
+ stats.GetStats(column_id).SetDistinct(std::move(distinct_stats));
707
753
  }
708
754
 
709
755
  } // namespace duckdb
@@ -2,6 +2,9 @@
2
2
  #include "duckdb/storage/table/row_group.hpp"
3
3
  #include "duckdb/storage/table/column_segment.hpp"
4
4
  #include "duckdb/transaction/duck_transaction.hpp"
5
+ #include "duckdb/storage/table/column_data.hpp"
6
+ #include "duckdb/storage/table/row_group_collection.hpp"
7
+ #include "duckdb/storage/table/row_group_segment_tree.hpp"
5
8
 
6
9
  namespace duckdb {
7
10
 
@@ -35,7 +38,7 @@ void ColumnScanState::NextInternal(idx_t count) {
35
38
  }
36
39
  row_index += count;
37
40
  while (row_index >= current->start + current->count) {
38
- current = (ColumnSegment *)current->Next();
41
+ current = segment_tree->GetNextSegment(current);
39
42
  initialized = false;
40
43
  segment_checked = false;
41
44
  if (!current) {
@@ -52,70 +55,79 @@ void ColumnScanState::Next(idx_t count) {
52
55
  }
53
56
  }
54
57
 
55
- void ColumnScanState::NextVector() {
56
- Next(STANDARD_VECTOR_SIZE);
57
- }
58
-
59
- const vector<column_t> &RowGroupScanState::GetColumnIds() {
58
+ const vector<column_t> &CollectionScanState::GetColumnIds() {
60
59
  return parent.GetColumnIds();
61
60
  }
62
61
 
63
- TableFilterSet *RowGroupScanState::GetFilters() {
62
+ TableFilterSet *CollectionScanState::GetFilters() {
64
63
  return parent.GetFilters();
65
64
  }
66
65
 
67
- AdaptiveFilter *RowGroupScanState::GetAdaptiveFilter() {
66
+ AdaptiveFilter *CollectionScanState::GetAdaptiveFilter() {
68
67
  return parent.GetAdaptiveFilter();
69
68
  }
70
69
 
71
- idx_t RowGroupScanState::GetParentMaxRow() {
72
- return parent.max_row;
73
- }
74
-
75
- const vector<column_t> &CollectionScanState::GetColumnIds() {
76
- return parent.GetColumnIds();
77
- }
78
-
79
- TableFilterSet *CollectionScanState::GetFilters() {
80
- return parent.GetFilters();
70
+ ParallelCollectionScanState::ParallelCollectionScanState()
71
+ : collection(nullptr), current_row_group(nullptr), processed_rows(0) {
81
72
  }
82
73
 
83
- AdaptiveFilter *CollectionScanState::GetAdaptiveFilter() {
84
- return parent.GetAdaptiveFilter();
74
+ CollectionScanState::CollectionScanState(TableScanState &parent_p)
75
+ : row_group(nullptr), vector_index(0), max_row_group_row(0), row_groups(nullptr), max_row(0), batch_index(0),
76
+ parent(parent_p) {
85
77
  }
86
78
 
87
79
  bool CollectionScanState::Scan(DuckTransaction &transaction, DataChunk &result) {
88
- auto current_row_group = row_group_state.row_group;
89
- while (current_row_group) {
90
- current_row_group->Scan(transaction, row_group_state, result);
80
+ while (row_group) {
81
+ row_group->Scan(transaction, *this, result);
91
82
  if (result.size() > 0) {
92
83
  return true;
84
+ } else if (max_row <= row_group->start + row_group->count) {
85
+ row_group = nullptr;
86
+ return false;
93
87
  } else {
94
88
  do {
95
- current_row_group = row_group_state.row_group = (RowGroup *)current_row_group->Next();
96
- if (current_row_group) {
97
- bool scan_row_group = current_row_group->InitializeScan(row_group_state);
89
+ row_group = row_groups->GetNextSegment(row_group);
90
+ if (row_group) {
91
+ if (row_group->start >= max_row) {
92
+ row_group = nullptr;
93
+ break;
94
+ }
95
+ bool scan_row_group = row_group->InitializeScan(*this);
98
96
  if (scan_row_group) {
99
97
  // scan this row group
100
98
  break;
101
99
  }
102
100
  }
103
- } while (current_row_group);
101
+ } while (row_group);
102
+ }
103
+ }
104
+ return false;
105
+ }
106
+
107
+ bool CollectionScanState::ScanCommitted(DataChunk &result, SegmentLock &l, TableScanType type) {
108
+ while (row_group) {
109
+ row_group->ScanCommitted(*this, result, type);
110
+ if (result.size() > 0) {
111
+ return true;
112
+ } else {
113
+ row_group = row_groups->GetNextSegment(l, row_group);
114
+ if (row_group) {
115
+ row_group->InitializeScan(*this);
116
+ }
104
117
  }
105
118
  }
106
119
  return false;
107
120
  }
108
121
 
109
122
  bool CollectionScanState::ScanCommitted(DataChunk &result, TableScanType type) {
110
- auto current_row_group = row_group_state.row_group;
111
- while (current_row_group) {
112
- current_row_group->ScanCommitted(row_group_state, result, type);
123
+ while (row_group) {
124
+ row_group->ScanCommitted(*this, result, type);
113
125
  if (result.size() > 0) {
114
126
  return true;
115
127
  } else {
116
- current_row_group = row_group_state.row_group = (RowGroup *)current_row_group->Next();
117
- if (current_row_group) {
118
- current_row_group->InitializeScan(row_group_state);
128
+ row_group = row_groups->GetNextSegment(row_group);
129
+ if (row_group) {
130
+ row_group->InitializeScan(*this);
119
131
  }
120
132
  }
121
133
  }