polars-df 0.26.1 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,159 +17,33 @@ module Polars
17
17
 
18
18
  table = scan.table
19
19
  snapshot = scan.snapshot
20
- schema = snapshot ? table.schema_by_id(snapshot.is_a?(Hash) ? snapshot[:schema_id] : snapshot.schema_id) : table.current_schema
21
-
22
- if files.empty? && !schema.respond_to?(:arrow_c_schema)
23
- # TODO remove in 0.27.0
24
- schema =
25
- schema.fields.to_h do |field|
26
- dtype =
27
- case field[:type]
28
- when "unknown"
29
- Unknown
30
- when "boolean"
31
- Boolean
32
- when "int"
33
- Int32
34
- when "long"
35
- Int64
36
- when "float"
37
- Float32
38
- when "double"
39
- Float64
40
- when "decimal"
41
- Decimal.new(field[:precision], field[:scale])
42
- when "string"
43
- String
44
- when "uuid"
45
- Binary
46
- when "fixed"
47
- Binary
48
- when "binary"
49
- Binary
50
- when "date"
51
- Date
52
- when "time"
53
- Time # ns instead of us since does not support time unit
54
- when "timestamp"
55
- Datetime.new("us")
56
- when "timestamp_ns"
57
- Datetime.new("ns")
58
- when "timestamptz"
59
- Datetime.new("us", "+00:00")
60
- when "timestamptz_ns"
61
- Datetime.new("ns", "+00:00")
62
- else
63
- raise Todo
64
- end
65
-
66
- [field[:name], dtype]
67
- end
68
-
69
- LazyFrame.new(schema: schema)
70
- else
71
- sources = files.map { |v| v.is_a?(Hash) ? v[:data_file_path] : v.file.file_path }
72
-
73
- column_mapping = [
74
- "iceberg-column-mapping",
75
- schema.respond_to?(:arrow_c_schema) ? schema : arrow_schema(schema)
76
- ]
77
-
78
- deletion_files = [
79
- "iceberg-position-delete",
80
- files.map.with_index
81
- .select { |v, i| (v.is_a?(Hash) ? v[:deletes] : v.delete_files).any? }
82
- .to_h { |v, i| [i, v.is_a?(Hash) ? v[:deletes].map { |d| d[:file_path] } : v.delete_files.map(&:file_path)] }
83
- ]
84
-
85
- scan_options = {
86
- schema: schema.respond_to?(:arrow_c_schema) ? Schema.new(schema) : nil,
87
- cast_options: Polars::ScanCastOptions._default_iceberg,
88
- missing_columns: "insert",
89
- extra_columns: "ignore",
90
- storage_options: @storage_options,
91
- _column_mapping: column_mapping,
92
- _deletion_files: deletion_files
93
- }
94
-
95
- Polars.scan_parquet(sources, **scan_options)
96
- end
97
- end
98
-
99
- private
100
-
101
- # TODO remove in 0.27.0
102
- def arrow_schema(schema)
103
- fields =
104
- schema.fields.map do |field|
105
- type =
106
- case field[:type]
107
- when "unknown"
108
- "unknown"
109
- when "boolean"
110
- "boolean"
111
- when "int"
112
- "int32"
113
- when "long"
114
- "int64"
115
- when "float"
116
- "float32"
117
- when "double"
118
- "float64"
119
- when "decimal"
120
- "decimal"
121
- when "string"
122
- "string"
123
- when "uuid"
124
- limit = 16
125
- "fixed_size_binary"
126
- when "fixed"
127
- limit = field[:limit]
128
- "fixed_size_binary"
129
- when "binary"
130
- "large_binary"
131
- when "date"
132
- "date32"
133
- when "time"
134
- time_unit = "us"
135
- "time64"
136
- when "timestamp"
137
- time_unit = "us"
138
- "timestamp"
139
- when "timestamp_ns"
140
- time_unit = "ns"
141
- "timestamp"
142
- when "timestamptz"
143
- time_unit = "us"
144
- time_zone = "+00:00"
145
- "timestamp"
146
- when "timestamptz_ns"
147
- time_unit = "ns"
148
- time_zone = "+00:00"
149
- "timestamp"
150
- else
151
- raise Todo
152
- end
153
-
154
- arrow_field = {
155
- name: field[:name],
156
- type: type,
157
- nullable: !field[:required],
158
- metadata: {
159
- "PARQUET:field_id" => field[:id].to_s
160
- }
161
- }
162
- arrow_field[:limit] = limit if limit
163
- if type == "decimal"
164
- arrow_field[:precision] = field[:precision]
165
- arrow_field[:scale] = field[:scale]
166
- end
167
- arrow_field[:time_unit] = time_unit if time_unit
168
- arrow_field[:time_zone] = time_zone if time_zone
169
- arrow_field
170
- end
171
-
172
- {fields: fields}
20
+ schema = snapshot ? table.schema_by_id(snapshot.schema_id) : table.current_schema
21
+
22
+ sources = files.map { |v| v.file.file_path }
23
+
24
+ column_mapping = [
25
+ "iceberg-column-mapping",
26
+ schema
27
+ ]
28
+
29
+ deletion_files = [
30
+ "iceberg-position-delete",
31
+ files.map.with_index
32
+ .select { |v, i| v.delete_files.any? }
33
+ .to_h { |v, i| [i, v.delete_files.map(&:file_path)] }
34
+ ]
35
+
36
+ scan_options = {
37
+ schema: Schema.new(schema),
38
+ cast_options: Polars::ScanCastOptions._default_iceberg,
39
+ missing_columns: "insert",
40
+ extra_columns: "ignore",
41
+ storage_options: @storage_options,
42
+ _column_mapping: column_mapping,
43
+ _deletion_files: deletion_files
44
+ }
45
+
46
+ Polars.scan_parquet(sources, **scan_options)
173
47
  end
174
48
  end
175
49
  end
@@ -4490,7 +4490,7 @@ module Polars
4490
4490
  # "numbers" => [[1], [2, 3], [4, 5], [6, 7, 8]],
4491
4491
  # }
4492
4492
  # ).lazy
4493
- # df.explode("numbers").collect
4493
+ # df.explode("numbers", empty_as_null: false).collect
4494
4494
  # # =>
4495
4495
  # # shape: (8, 2)
4496
4496
  # # ┌─────────┬─────────┐
@@ -4510,9 +4510,16 @@ module Polars
4510
4510
  def explode(
4511
4511
  columns,
4512
4512
  *more_columns,
4513
- empty_as_null: true,
4513
+ empty_as_null: OMITTED,
4514
4514
  keep_nulls: true
4515
4515
  )
4516
+ if empty_as_null == OMITTED
4517
+ Utils.issue_deprecation_warning(
4518
+ "The default behavior for `empty_as_null` will change to `false`. " +
4519
+ "To keep the current behavior, explicitly set `empty_as_null: true`."
4520
+ )
4521
+ empty_as_null = true
4522
+ end
4516
4523
  subset = Utils.parse_list_into_selector(columns) | Utils.parse_list_into_selector(
4517
4524
  more_columns
4518
4525
  )
@@ -5149,7 +5156,8 @@ module Polars
5149
5156
  # # │ elise ┆ 44 │
5150
5157
  # # └────────┴─────┘
5151
5158
  def merge_sorted(other, key, maintain_order: false)
5152
- _from_rbldf(_ldf.merge_sorted(other._ldf, key, maintain_order))
5159
+ keys = Array(key)
5160
+ _from_rbldf(_ldf.merge_sorted(other._ldf, keys, maintain_order))
5153
5161
  end
5154
5162
 
5155
5163
  # Flag a column as sorted.
@@ -907,7 +907,7 @@ module Polars
907
907
  #
908
908
  # @example
909
909
  # df = Polars::DataFrame.new({"a" => [[1, 2, 3], [4, 5, 6]]})
910
- # df.select(Polars.col("a").list.explode)
910
+ # df.select(Polars.col("a").list.explode(empty_as_null: false))
911
911
  # # =>
912
912
  # # shape: (6, 1)
913
913
  # # ┌─────┐
@@ -922,7 +922,15 @@ module Polars
922
922
  # # │ 5 │
923
923
  # # │ 6 │
924
924
  # # └─────┘
925
- def explode(empty_as_null: true, keep_nulls: true)
925
+ def explode(empty_as_null: OMITTED, keep_nulls: true)
926
+ if empty_as_null == OMITTED
927
+ Utils.issue_deprecation_warning(
928
+ "The default behavior for `empty_as_null` will change to `false`. " +
929
+ "To keep the current behavior, explicitly set `empty_as_null: true`."
930
+ )
931
+ empty_as_null = true
932
+ end
933
+
926
934
  Utils.wrap_expr(_rbexpr.explode(empty_as_null, keep_nulls))
927
935
  end
928
936
 
@@ -744,7 +744,7 @@ module Polars
744
744
  #
745
745
  # @example
746
746
  # s = Polars::Series.new("a", [[1, 2, 3], [4, 5, 6]])
747
- # s.list.explode
747
+ # s.list.explode(empty_as_null: false)
748
748
  # # =>
749
749
  # # shape: (6,)
750
750
  # # Series: 'a' [i64]
@@ -756,7 +756,7 @@ module Polars
756
756
  # # 5
757
757
  # # 6
758
758
  # # ]
759
- def explode(empty_as_null: true, keep_nulls: true)
759
+ def explode(empty_as_null: nil, keep_nulls: true)
760
760
  super
761
761
  end
762
762
 
data/lib/polars/series.rb CHANGED
@@ -2807,7 +2807,7 @@ module Polars
2807
2807
  #
2808
2808
  # @example
2809
2809
  # s = Polars::Series.new("a", [[1, 2], [3, 4], [9, 10]])
2810
- # s.explode
2810
+ # s.explode(empty_as_null: false)
2811
2811
  # # =>
2812
2812
  # # shape: (6,)
2813
2813
  # # Series: 'a' [i64]
@@ -6118,7 +6118,7 @@ module Polars
6118
6118
  # # shape: (3,)
6119
6119
  # # Series: 'a' [f64]
6120
6120
  # # [
6121
- # # 0.0
6121
+ # # null
6122
6122
  # # 0.707107
6123
6123
  # # 0.963624
6124
6124
  # # ]
@@ -6146,7 +6146,7 @@ module Polars
6146
6146
  # # shape: (3,)
6147
6147
  # # Series: 'a' [f64]
6148
6148
  # # [
6149
- # # 0.0
6149
+ # # null
6150
6150
  # # 0.5
6151
6151
  # # 0.928571
6152
6152
  # # ]
@@ -1,4 +1,4 @@
1
1
  module Polars
2
2
  # @private
3
- VERSION = "0.26.1"
3
+ VERSION = "0.27.0"
4
4
  end
data/lib/polars.rb CHANGED
@@ -124,18 +124,10 @@ module Polars
124
124
  NO_DEFAULT = Object.new
125
125
 
126
126
  # @private
127
- N_INFER_DEFAULT = 100
128
-
129
- # @private
130
- # TODO remove in 0.27.0
131
- ArrowArrayStream = Capsule
127
+ OMITTED = Object.new
132
128
 
133
129
  # @private
134
- class ArrowArrayStream
135
- def arrow_c_stream
136
- self
137
- end
138
- end
130
+ N_INFER_DEFAULT = 100
139
131
 
140
132
  # Return the number of threads in the Polars thread pool.
141
133
  #
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: polars-df
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.26.1
4
+ version: 0.27.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Andrew Kane