polars-df 0.26.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,92 +17,33 @@ module Polars
17
17
 
18
18
  table = scan.table
19
19
  snapshot = scan.snapshot
20
- schema = snapshot ? table.schema_by_id(snapshot[:schema_id]) : table.current_schema
21
-
22
- if files.empty?
23
- # TODO improve
24
- schema =
25
- schema.fields.to_h do |field|
26
- dtype =
27
- case field[:type]
28
- when "int"
29
- Polars::Int32
30
- when "long"
31
- Polars::Int64
32
- when "double"
33
- Polars::Float64
34
- when "string"
35
- Polars::String
36
- when "timestamp"
37
- Polars::Datetime
38
- else
39
- raise Todo
40
- end
41
-
42
- [field[:name], dtype]
43
- end
44
-
45
- LazyFrame.new(schema: schema)
46
- else
47
- sources = files.map { |v| v[:data_file_path] }
48
-
49
- column_mapping = [
50
- "iceberg-column-mapping",
51
- arrow_schema(schema)
52
- ]
53
-
54
- deletion_files = [
55
- "iceberg-position-delete",
56
- files.map.with_index
57
- .select { |v, i| v[:deletes].any? }
58
- .to_h { |v, i| [i, v[:deletes].map { |d| d[:file_path] }] }
59
- ]
60
-
61
- scan_options = {
62
- storage_options: @storage_options,
63
- cast_options: Polars::ScanCastOptions._default_iceberg,
64
- missing_columns: "insert",
65
- extra_columns: "ignore",
66
- _column_mapping: column_mapping,
67
- _deletion_files: deletion_files
68
- }
69
-
70
- Polars.scan_parquet(sources, **scan_options)
71
- end
72
- end
73
-
74
- private
75
-
76
- def arrow_schema(schema)
77
- fields =
78
- schema.fields.map do |field|
79
- type =
80
- case field[:type]
81
- when "boolean"
82
- "boolean"
83
- when "int"
84
- "int32"
85
- when "long"
86
- "int64"
87
- when "float"
88
- "float32"
89
- when "double"
90
- "float64"
91
- else
92
- raise Todo
93
- end
94
-
95
- {
96
- name: field[:name],
97
- type: type,
98
- nullable: !field[:required],
99
- metadata: {
100
- "PARQUET:field_id" => field[:id].to_s
101
- }
102
- }
103
- end
104
-
105
- {fields: fields}
20
+ schema = snapshot ? table.schema_by_id(snapshot.schema_id) : table.current_schema
21
+
22
+ sources = files.map { |v| v.file.file_path }
23
+
24
+ column_mapping = [
25
+ "iceberg-column-mapping",
26
+ schema
27
+ ]
28
+
29
+ deletion_files = [
30
+ "iceberg-position-delete",
31
+ files.map.with_index
32
+ .select { |v, i| v.delete_files.any? }
33
+ .to_h { |v, i| [i, v.delete_files.map(&:file_path)] }
34
+ ]
35
+
36
+ scan_options = {
37
+ schema: Schema.new(schema),
38
+ cast_options: Polars::ScanCastOptions._default_iceberg,
39
+ missing_columns: "insert",
40
+ extra_columns: "ignore",
41
+ storage_options: @storage_options,
42
+ _column_mapping: column_mapping,
43
+ _deletion_files: deletion_files
44
+ }
45
+
46
+ Polars.scan_parquet(sources, **scan_options)
106
47
  end
107
48
  end
108
49
  end
@@ -25,6 +25,8 @@ module Polars
25
25
  storage_options: nil,
26
26
  delta_table_options: nil
27
27
  )
28
+ require "deltalake-rb"
29
+
28
30
  df =
29
31
  scan_delta(
30
32
  source,
@@ -62,6 +64,8 @@ module Polars
62
64
  delta_table_options: nil,
63
65
  rechunk: nil
64
66
  )
67
+ require "deltalake-rb"
68
+
65
69
  dl_tbl =
66
70
  _get_delta_lake_table(
67
71
  source,
@@ -4490,7 +4490,7 @@ module Polars
4490
4490
  # "numbers" => [[1], [2, 3], [4, 5], [6, 7, 8]],
4491
4491
  # }
4492
4492
  # ).lazy
4493
- # df.explode("numbers").collect
4493
+ # df.explode("numbers", empty_as_null: false).collect
4494
4494
  # # =>
4495
4495
  # # shape: (8, 2)
4496
4496
  # # ┌─────────┬─────────┐
@@ -4510,9 +4510,16 @@ module Polars
4510
4510
  def explode(
4511
4511
  columns,
4512
4512
  *more_columns,
4513
- empty_as_null: true,
4513
+ empty_as_null: OMITTED,
4514
4514
  keep_nulls: true
4515
4515
  )
4516
+ if empty_as_null == OMITTED
4517
+ Utils.issue_deprecation_warning(
4518
+ "The default behavior for `empty_as_null` will change to `false`. " +
4519
+ "To keep the current behavior, explicitly set `empty_as_null: true`."
4520
+ )
4521
+ empty_as_null = true
4522
+ end
4516
4523
  subset = Utils.parse_list_into_selector(columns) | Utils.parse_list_into_selector(
4517
4524
  more_columns
4518
4525
  )
@@ -5149,7 +5156,8 @@ module Polars
5149
5156
  # # │ elise ┆ 44 │
5150
5157
  # # └────────┴─────┘
5151
5158
  def merge_sorted(other, key, maintain_order: false)
5152
- _from_rbldf(_ldf.merge_sorted(other._ldf, key, maintain_order))
5159
+ keys = Array(key)
5160
+ _from_rbldf(_ldf.merge_sorted(other._ldf, keys, maintain_order))
5153
5161
  end
5154
5162
 
5155
5163
  # Flag a column as sorted.
@@ -907,7 +907,7 @@ module Polars
907
907
  #
908
908
  # @example
909
909
  # df = Polars::DataFrame.new({"a" => [[1, 2, 3], [4, 5, 6]]})
910
- # df.select(Polars.col("a").list.explode)
910
+ # df.select(Polars.col("a").list.explode(empty_as_null: false))
911
911
  # # =>
912
912
  # # shape: (6, 1)
913
913
  # # ┌─────┐
@@ -922,7 +922,15 @@ module Polars
922
922
  # # │ 5 │
923
923
  # # │ 6 │
924
924
  # # └─────┘
925
- def explode(empty_as_null: true, keep_nulls: true)
925
+ def explode(empty_as_null: OMITTED, keep_nulls: true)
926
+ if empty_as_null == OMITTED
927
+ Utils.issue_deprecation_warning(
928
+ "The default behavior for `empty_as_null` will change to `false`. " +
929
+ "To keep the current behavior, explicitly set `empty_as_null: true`."
930
+ )
931
+ empty_as_null = true
932
+ end
933
+
926
934
  Utils.wrap_expr(_rbexpr.explode(empty_as_null, keep_nulls))
927
935
  end
928
936
 
@@ -744,7 +744,7 @@ module Polars
744
744
  #
745
745
  # @example
746
746
  # s = Polars::Series.new("a", [[1, 2, 3], [4, 5, 6]])
747
- # s.list.explode
747
+ # s.list.explode(empty_as_null: false)
748
748
  # # =>
749
749
  # # shape: (6,)
750
750
  # # Series: 'a' [i64]
@@ -756,7 +756,7 @@ module Polars
756
756
  # # 5
757
757
  # # 6
758
758
  # # ]
759
- def explode(empty_as_null: true, keep_nulls: true)
759
+ def explode(empty_as_null: nil, keep_nulls: true)
760
760
  super
761
761
  end
762
762
 
data/lib/polars/schema.rb CHANGED
@@ -71,6 +71,15 @@ module Polars
71
71
  @schema.values
72
72
  end
73
73
 
74
+ # Convert the schema to a nanoarrow schema.
75
+ #
76
+ # @return [Nanoarrow::Schema]
77
+ def to_arrow
78
+ require "nanoarrow"
79
+
80
+ Nanoarrow::Schema.new(self)
81
+ end
82
+
74
83
  # Get the number of schema entries.
75
84
  #
76
85
  # @return [Integer]
data/lib/polars/series.rb CHANGED
@@ -73,6 +73,8 @@ module Polars
73
73
  self._s = Utils.dataframe_to_rbseries(
74
74
  original_name, values, dtype: dtype, strict: strict
75
75
  )
76
+ elsif values.respond_to?(:arrow_c_stream)
77
+ self._s = RbSeries.from_arrow_c_stream(values)
76
78
  else
77
79
  raise TypeError, "Series constructor called with unsupported type; got #{values.class.name}"
78
80
  end
@@ -496,6 +498,11 @@ module Polars
496
498
  end
497
499
  end
498
500
 
501
+ # @private
502
+ def arrow_c_stream
503
+ _s.arrow_c_stream
504
+ end
505
+
499
506
  # Return the Series as a scalar, or return the element at the given index.
500
507
  #
501
508
  # If no index is provided, this is equivalent to `s[0]`, with a check
@@ -2800,7 +2807,7 @@ module Polars
2800
2807
  #
2801
2808
  # @example
2802
2809
  # s = Polars::Series.new("a", [[1, 2], [3, 4], [9, 10]])
2803
- # s.explode
2810
+ # s.explode(empty_as_null: false)
2804
2811
  # # =>
2805
2812
  # # shape: (6,)
2806
2813
  # # Series: 'a' [i64]
@@ -3114,6 +3121,8 @@ module Polars
3114
3121
  # # Numo::Int64#shape=[3]
3115
3122
  # # [1, 2, 3]
3116
3123
  def to_numo
3124
+ require "numo/narray"
3125
+
3117
3126
  if dtype.temporal?
3118
3127
  Numo::RObject.cast(to_a)
3119
3128
  else
@@ -3121,6 +3130,15 @@ module Polars
3121
3130
  end
3122
3131
  end
3123
3132
 
3133
+ # Return the underlying Arrow array.
3134
+ #
3135
+ # @return [Nanoarrow::Array]
3136
+ def to_arrow
3137
+ require "nanoarrow"
3138
+
3139
+ Nanoarrow::Array.new(self)
3140
+ end
3141
+
3124
3142
  # Set masked values.
3125
3143
  #
3126
3144
  # @param filter [Series]
@@ -6100,7 +6118,7 @@ module Polars
6100
6118
  # # shape: (3,)
6101
6119
  # # Series: 'a' [f64]
6102
6120
  # # [
6103
- # # 0.0
6121
+ # # null
6104
6122
  # # 0.707107
6105
6123
  # # 0.963624
6106
6124
  # # ]
@@ -6128,7 +6146,7 @@ module Polars
6128
6146
  # # shape: (3,)
6129
6147
  # # Series: 'a' [f64]
6130
6148
  # # [
6131
- # # 0.0
6149
+ # # null
6132
6150
  # # 0.5
6133
6151
  # # 0.928571
6134
6152
  # # ]
@@ -1,4 +1,4 @@
1
1
  module Polars
2
2
  # @private
3
- VERSION = "0.26.0"
3
+ VERSION = "0.27.0"
4
4
  end
data/lib/polars.rb CHANGED
@@ -124,14 +124,10 @@ module Polars
124
124
  NO_DEFAULT = Object.new
125
125
 
126
126
  # @private
127
- N_INFER_DEFAULT = 100
127
+ OMITTED = Object.new
128
128
 
129
129
  # @private
130
- class ArrowArrayStream
131
- def arrow_c_stream
132
- self
133
- end
134
- end
130
+ N_INFER_DEFAULT = 100
135
131
 
136
132
  # Return the number of threads in the Polars thread pool.
137
133
  #
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: polars-df
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.26.0
4
+ version: 0.27.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Andrew Kane
@@ -113,6 +113,7 @@ files:
113
113
  - ext/polars/src/on_startup.rs
114
114
  - ext/polars/src/prelude.rs
115
115
  - ext/polars/src/rb_modules.rs
116
+ - ext/polars/src/ruby/capsule.rs
116
117
  - ext/polars/src/ruby/exceptions.rs
117
118
  - ext/polars/src/ruby/gvl.rs
118
119
  - ext/polars/src/ruby/lazy.rs
@@ -262,7 +263,7 @@ required_rubygems_version: !ruby/object:Gem::Requirement
262
263
  - !ruby/object:Gem::Version
263
264
  version: '0'
264
265
  requirements: []
265
- rubygems_version: 4.0.13
266
+ rubygems_version: 4.0.14
266
267
  specification_version: 4
267
268
  summary: Blazingly fast DataFrames for Ruby
268
269
  test_files: []