polars-df 0.26.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/Cargo.toml CHANGED
@@ -4,6 +4,3 @@ resolver = "2"
4
4
 
5
5
  [profile.release]
6
6
  strip = true
7
-
8
- [profile.dev]
9
- strip = true
data/LICENSE.txt CHANGED
@@ -1,6 +1,6 @@
1
1
  Copyright (c) 2025 Ritchie Vink
2
2
  Copyright (c) 2022-2026 Andrew Kane
3
- Some portions Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
3
+ Copyright (c) 2024 (Some portions) NVIDIA CORPORATION & AFFILIATES. All rights reserved.
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
data/README.md CHANGED
@@ -381,6 +381,12 @@ Delta Lake (experimental)
381
381
  df.write_delta("./table")
382
382
  ```
383
383
 
384
+ Arrow array (experimental, requires [nanoarrow](https://github.com/ankane/nanoarrow-ruby))
385
+
386
+ ```ruby
387
+ df.to_arrow
388
+ ```
389
+
384
390
  Numo array
385
391
 
386
392
  ```ruby
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "polars-ruby"
3
- version = "0.26.0"
3
+ version = "0.27.0"
4
4
  license = "MIT"
5
5
  authors = ["Andrew Kane <andrew@ankane.org>"]
6
6
  edition = "2024"
@@ -13,7 +13,7 @@ crate-type = ["cdylib"]
13
13
 
14
14
  [dependencies]
15
15
  ahash = "0.8"
16
- arrow = { package = "polars-arrow", version = "=0.54.4" }
16
+ arrow = { package = "polars-arrow", version = "=0.55.1" }
17
17
  bytes = "1"
18
18
  chrono = "0.4"
19
19
  chrono-tz = "0.10"
@@ -21,26 +21,26 @@ either = "1.8"
21
21
  magnus = { version = "0.8", features = ["chrono"] }
22
22
  num-traits = "0.2"
23
23
  parking_lot = "0.12"
24
- polars-buffer = "=0.54.4"
25
- polars-compute = "=0.54.4"
26
- polars-config = "=0.54.4"
27
- polars-core = "=0.54.4"
28
- polars-dtype = "=0.54.4"
29
- polars-error = "=0.54.4"
30
- polars-io = "=0.54.4"
31
- polars-lazy = { version = "=0.54.4", features = ["catalog"] }
32
- polars-ops = "=0.54.4"
33
- polars-plan = "=0.54.4"
34
- polars-parquet = "=0.54.4"
35
- polars-testing = "=0.54.4"
36
- polars-utils = "=0.54.4"
24
+ polars-buffer = "=0.55.1"
25
+ polars-compute = "=0.55.1"
26
+ polars-config = "=0.55.1"
27
+ polars-core = "=0.55.1"
28
+ polars-dtype = "=0.55.1"
29
+ polars-error = "=0.55.1"
30
+ polars-io = "=0.55.1"
31
+ polars-lazy = { version = "=0.55.1", features = ["catalog"] }
32
+ polars-ops = "=0.55.1"
33
+ polars-plan = "=0.55.1"
34
+ polars-parquet = "=0.55.1"
35
+ polars-testing = "=0.55.1"
36
+ polars-utils = "=0.55.1"
37
37
  rayon = "1.9"
38
38
  rb-sys = "0.9"
39
39
  regex = "1"
40
40
  serde_json = "1"
41
41
 
42
42
  [dependencies.polars]
43
- version = "=0.54.4"
43
+ version = "=0.55.1"
44
44
  features = [
45
45
  "abs",
46
46
  "approx_unique",
@@ -3,7 +3,6 @@ mod categorical;
3
3
  mod chunked_array;
4
4
  mod datetime;
5
5
 
6
- use std::collections::BTreeMap;
7
6
  use std::fmt::{Debug, Display, Formatter};
8
7
  use std::fs::File;
9
8
  use std::hash::{Hash, Hasher};
@@ -19,9 +18,7 @@ use polars::datatypes::AnyValue;
19
18
  use polars::frame::PivotColumnNaming;
20
19
  use polars::frame::row::Row;
21
20
  use polars::io::avro::AvroCompression;
22
- use polars::prelude::default_values::{
23
- DefaultFieldValues, IcebergIdentityTransformedPartitionFields,
24
- };
21
+ use polars::prelude::default_values::DefaultFieldValues;
25
22
  use polars::prelude::deletion::DeletionFilesList;
26
23
  use polars::prelude::*;
27
24
  use polars::series::ops::NullBehavior;
@@ -31,6 +28,7 @@ use polars_core::schema::iceberg::IcebergSchema;
31
28
  use polars_core::utils::arrow::array::Array;
32
29
  use polars_core::utils::materialize_dyn_int;
33
30
  use polars_plan::dsl::ScanSources;
31
+ use polars_plan::dsl::default_values::IcebergDefaultFieldValues;
34
32
  use polars_utils::compression::{BrotliLevel, GzipLevel, ZstdLevel};
35
33
  use polars_utils::total_ord::{TotalEq, TotalHash};
36
34
 
@@ -39,6 +37,7 @@ use crate::object::OBJECT_NAME;
39
37
  use crate::rb_modules::pl_series;
40
38
  use crate::ruby::gvl::GvlExt;
41
39
  use crate::ruby::utils::TryIntoValue;
40
+ use crate::series::import_schema_rbcapsule;
42
41
  use crate::utils::to_rb_err;
43
42
  use crate::{
44
43
  RbDataFrame, RbExpr, RbLazyFrame, RbPolarsErr, RbResult, RbSeries, RbTypeError, RbValueError,
@@ -514,55 +513,25 @@ impl TryConvert for Wrap<Schema> {
514
513
 
515
514
  impl TryConvert for Wrap<ArrowSchema> {
516
515
  fn try_convert(ob: Value) -> RbResult<Self> {
517
- let ruby = Ruby::get_with(ob);
518
- // TODO improve
519
- let ob = RHash::try_convert(ob)?;
520
- let fields: RArray = ob.aref(ruby.to_symbol("fields"))?;
521
- let mut arrow_schema = ArrowSchema::with_capacity(fields.len());
522
- for f in fields {
523
- let f = RHash::try_convert(f)?;
524
- let name: String = f.aref(ruby.to_symbol("name"))?;
525
- let rb_dtype: String = f.aref(ruby.to_symbol("type"))?;
526
- let dtype = match rb_dtype.as_str() {
527
- "null" => ArrowDataType::Null,
528
- "boolean" => ArrowDataType::Boolean,
529
- "int8" => ArrowDataType::Int8,
530
- "int16" => ArrowDataType::Int16,
531
- "int32" => ArrowDataType::Int32,
532
- "int64" => ArrowDataType::Int64,
533
- "uint8" => ArrowDataType::UInt8,
534
- "uint16" => ArrowDataType::UInt16,
535
- "uint32" => ArrowDataType::UInt32,
536
- "uint64" => ArrowDataType::UInt64,
537
- "float16" => ArrowDataType::Float16,
538
- "float32" => ArrowDataType::Float32,
539
- "float64" => ArrowDataType::Float64,
540
- "date32" => ArrowDataType::Date32,
541
- "date64" => ArrowDataType::Date64,
542
- "binary" => ArrowDataType::Binary,
543
- "large_binary" => ArrowDataType::LargeBinary,
544
- "string" => ArrowDataType::Utf8,
545
- "large_string" => ArrowDataType::LargeUtf8,
546
- "binary_view" => ArrowDataType::BinaryView,
547
- "string_view" => ArrowDataType::Utf8View,
548
- "unknown" => ArrowDataType::Unknown,
549
- _ => todo!(),
550
- };
551
- let is_nullable = f.aref(ruby.to_symbol("nullable"))?;
552
- let rb_metadata: RHash = f.aref(ruby.to_symbol("metadata"))?;
553
- let mut metadata = BTreeMap::new();
554
- rb_metadata.foreach(|k: String, v: String| {
555
- metadata.insert(k.into(), v.into());
556
- Ok(ForEach::Continue)
557
- })?;
558
- arrow_schema
559
- .try_insert(
560
- name.clone().into(),
561
- ArrowField::new(name.into(), dtype, is_nullable).with_metadata(metadata),
562
- )
563
- .map_err(to_rb_err)?;
516
+ let schema_capsule: Value = ob.funcall("arrow_c_schema", ())?;
517
+
518
+ let field = import_schema_rbcapsule(schema_capsule)?;
519
+
520
+ let ArrowDataType::Struct(fields) = field.dtype else {
521
+ return Err(RbValueError::new_err(format!(
522
+ "arrow_c_schema of object did not return struct dtype: \
523
+ object: {:?}, dtype: {:?}",
524
+ ob, &field.dtype
525
+ )));
526
+ };
527
+
528
+ let mut schema = ArrowSchema::from_iter_check_duplicates(fields).unwrap();
529
+
530
+ if let Some(md) = field.metadata {
531
+ *schema.metadata_mut() = Arc::unwrap_or_clone(md);
564
532
  }
565
- Ok(Wrap(arrow_schema))
533
+
534
+ Ok(Wrap(schema))
566
535
  }
567
536
  }
568
537
 
@@ -1281,6 +1250,9 @@ impl TryConvert for Wrap<CastColumnsPolicy> {
1281
1250
  })?;
1282
1251
 
1283
1252
  let mut datetime_nanoseconds_downcast = false;
1253
+ let mut datetime_microseconds_downcast = false;
1254
+ let mut datetime_milliseconds_upcast = false;
1255
+ let mut datetime_microseconds_upcast = false;
1284
1256
  let mut datetime_convert_timezone = false;
1285
1257
 
1286
1258
  let datetime_cast_object: Value = ob.funcall("datetime_cast", ())?;
@@ -1289,6 +1261,9 @@ impl TryConvert for Wrap<CastColumnsPolicy> {
1289
1261
  match v {
1290
1262
  "forbid" => {}
1291
1263
  "nanosecond-downcast" => datetime_nanoseconds_downcast = true,
1264
+ "microsecond-downcast" => datetime_microseconds_downcast = true,
1265
+ "millisecond-upcast" => datetime_milliseconds_upcast = true,
1266
+ "microsecond-upcast" => datetime_microseconds_upcast = true,
1292
1267
  "convert-timezone" => datetime_convert_timezone = true,
1293
1268
  v => {
1294
1269
  return Err(RbValueError::new_err(format!(
@@ -1338,7 +1313,9 @@ impl TryConvert for Wrap<CastColumnsPolicy> {
1338
1313
  float_upcast,
1339
1314
  float_downcast,
1340
1315
  datetime_nanoseconds_downcast,
1341
- datetime_microseconds_downcast: false,
1316
+ datetime_microseconds_downcast,
1317
+ datetime_milliseconds_upcast,
1318
+ datetime_microseconds_upcast,
1342
1319
  datetime_convert_timezone,
1343
1320
  null_upcast: true,
1344
1321
  categorical_to_string,
@@ -1636,11 +1613,13 @@ impl TryConvert for Wrap<DefaultFieldValues> {
1636
1613
 
1637
1614
  Ok(Wrap(match &*default_values_type {
1638
1615
  "iceberg" => {
1639
- let dict = RHash::try_convert(ob)?;
1616
+ let (identity_transformed_partition_values, initial_defaults) =
1617
+ <(RHash, RHash)>::try_convert(ob)?;
1640
1618
 
1641
- let mut out = PlIndexMap::new();
1619
+ let mut converted_identity_transformed_partition_values = PlIndexMap::new();
1620
+ let mut converted_initial_defaults = PlIndexMap::new();
1642
1621
 
1643
- dict.foreach(|k: u32, v: Value| {
1622
+ identity_transformed_partition_values.foreach(|k: u32, v: Value| {
1644
1623
  let v: Result<Column, String> = if let Ok(s) = get_series(v) {
1645
1624
  Ok(s.into_column())
1646
1625
  } else {
@@ -1648,14 +1627,28 @@ impl TryConvert for Wrap<DefaultFieldValues> {
1648
1627
  Err(err_msg)
1649
1628
  };
1650
1629
 
1651
- out.insert(k, v);
1630
+ converted_identity_transformed_partition_values.insert(k, v);
1631
+
1632
+ Ok(ForEach::Continue)
1633
+ })?;
1634
+
1635
+ initial_defaults.foreach(|k: u32, v: Value| {
1636
+ let v = get_series(v)?;
1637
+ let v = Scalar::new(
1638
+ v.dtype().clone(),
1639
+ v.get(0).map_err(to_rb_err)?.into_static(),
1640
+ );
1641
+ converted_initial_defaults.insert(k, v);
1652
1642
 
1653
1643
  Ok(ForEach::Continue)
1654
1644
  })?;
1655
1645
 
1656
- DefaultFieldValues::Iceberg(Arc::new(IcebergIdentityTransformedPartitionFields(
1657
- out,
1658
- )))
1646
+ DefaultFieldValues::Iceberg(Arc::new(IcebergDefaultFieldValues {
1647
+ identity_transformed_partition_fields: PlIndexMapHashable(
1648
+ converted_identity_transformed_partition_values,
1649
+ ),
1650
+ initial_defaults: PlIndexMapHashable(converted_initial_defaults),
1651
+ }))
1659
1652
  }
1660
1653
 
1661
1654
  v => {
@@ -84,7 +84,7 @@ impl RbDataFrame {
84
84
  self_: &Self,
85
85
  n: &RbSeries,
86
86
  with_replacement: bool,
87
- shuffle: bool,
87
+ shuffle: Option<bool>,
88
88
  seed: Option<u64>,
89
89
  ) -> RbResult<Self> {
90
90
  rb.enter_polars_df(|| {
@@ -100,7 +100,7 @@ impl RbDataFrame {
100
100
  self_: &Self,
101
101
  frac: &RbSeries,
102
102
  with_replacement: bool,
103
- shuffle: bool,
103
+ shuffle: Option<bool>,
104
104
  seed: Option<u64>,
105
105
  ) -> RbResult<Self> {
106
106
  rb.enter_polars_df(|| {
@@ -310,6 +310,19 @@ impl RbDataFrame {
310
310
  rb.enter_polars_series(|| self_.df.read().is_duplicated())
311
311
  }
312
312
 
313
+ pub fn is_sorted(
314
+ rb: &Ruby,
315
+ self_: &Self,
316
+ by: Vec<String>,
317
+ descending: Vec<bool>,
318
+ nulls_last: Vec<bool>,
319
+ ) -> RbResult<bool> {
320
+ rb.enter_polars(|| {
321
+ let by = strings_to_pl_smallstr(by);
322
+ self_.df.read().is_sorted(&by, &descending, &nulls_last)
323
+ })
324
+ }
325
+
313
326
  pub fn equals(
314
327
  rb: &Ruby,
315
328
  self_: &Self,
@@ -1,4 +1,4 @@
1
- use magnus::{Value, prelude::*};
1
+ use magnus::{Ruby, Value, prelude::*};
2
2
  use polars::io::RowIndex;
3
3
  use polars::io::avro::AvroCompression;
4
4
  use polars::prelude::*;
@@ -6,14 +6,15 @@ use std::io::BufWriter;
6
6
  use std::num::NonZeroUsize;
7
7
 
8
8
  use super::*;
9
+ use crate::RbResult;
9
10
  use crate::conversion::*;
10
11
  use crate::file::{
11
12
  get_file_like, get_mmap_bytes_reader, get_mmap_bytes_reader_and_path, read_if_bytesio,
12
13
  };
13
- use crate::{RbPolarsErr, RbResult};
14
+ use crate::utils::EnterPolarsExt;
14
15
 
15
16
  impl RbDataFrame {
16
- pub fn read_csv(arguments: &[Value]) -> RbResult<Self> {
17
+ pub fn read_csv(rb: &Ruby, arguments: &[Value]) -> RbResult<Self> {
17
18
  // start arguments
18
19
  // this pattern is needed for more than 16
19
20
  let rb_f = arguments[0];
@@ -83,46 +84,47 @@ impl RbDataFrame {
83
84
 
84
85
  let rb_f = read_if_bytesio(rb_f);
85
86
  let mmap_bytes_r = get_mmap_bytes_reader(&rb_f)?;
86
- let df = CsvReadOptions::default()
87
- .with_path(path)
88
- .with_infer_schema_length(infer_schema_length)
89
- .with_has_header(has_header)
90
- .with_n_rows(n_rows)
91
- .with_skip_rows(skip_rows)
92
- .with_skip_lines(skip_lines)
93
- .with_ignore_errors(ignore_errors)
94
- .with_projection(projection.map(Arc::new))
95
- .with_rechunk(rechunk)
96
- .with_chunk_size(chunk_size)
97
- .with_columns(columns.map(|x| x.into_iter().map(|x| x.into()).collect()))
98
- .with_n_threads(n_threads)
99
- .with_schema_overwrite(overwrite_dtype.map(Arc::new))
100
- .with_dtype_overwrite(overwrite_dtype_slice.map(Arc::new))
101
- .with_schema(schema.map(|schema| Arc::new(schema.0)))
102
- .with_low_memory(low_memory)
103
- .with_skip_rows_after_header(skip_rows_after_header)
104
- .with_row_index(row_index)
105
- .with_raise_if_empty(raise_if_empty)
106
- .with_parse_options(
107
- CsvParseOptions::default()
108
- .with_separator(separator.as_bytes()[0])
109
- .with_encoding(encoding.0)
110
- .with_missing_is_null(!missing_utf8_is_empty_string)
111
- .with_comment_prefix(comment_prefix.as_deref())
112
- .with_null_values(null_values)
113
- .with_try_parse_dates(try_parse_dates)
114
- .with_quote_char(quote_char)
115
- .with_eol_char(eol_char)
116
- .with_truncate_ragged_lines(truncate_ragged_lines)
117
- .with_decimal_comma(decimal_comma),
118
- )
119
- .into_reader_with_file_handle(mmap_bytes_r)
120
- .finish()
121
- .map_err(RbPolarsErr::from)?;
122
- Ok(df.into())
87
+ rb.enter_polars_df(move || {
88
+ CsvReadOptions::default()
89
+ .with_path(path)
90
+ .with_infer_schema_length(infer_schema_length)
91
+ .with_has_header(has_header)
92
+ .with_n_rows(n_rows)
93
+ .with_skip_rows(skip_rows)
94
+ .with_skip_lines(skip_lines)
95
+ .with_ignore_errors(ignore_errors)
96
+ .with_projection(projection.map(Arc::new))
97
+ .with_rechunk(rechunk)
98
+ .with_chunk_size(chunk_size)
99
+ .with_columns(columns.map(|x| x.into_iter().map(|x| x.into()).collect()))
100
+ .with_n_threads(n_threads)
101
+ .with_schema_overwrite(overwrite_dtype.map(Arc::new))
102
+ .with_dtype_overwrite(overwrite_dtype_slice.map(Arc::new))
103
+ .with_schema(schema.map(|schema| Arc::new(schema.0)))
104
+ .with_low_memory(low_memory)
105
+ .with_skip_rows_after_header(skip_rows_after_header)
106
+ .with_row_index(row_index)
107
+ .with_raise_if_empty(raise_if_empty)
108
+ .with_parse_options(
109
+ CsvParseOptions::default()
110
+ .with_separator(separator.as_bytes()[0])
111
+ .with_encoding(encoding.0)
112
+ .with_missing_is_null(!missing_utf8_is_empty_string)
113
+ .with_comment_prefix(comment_prefix.as_deref())
114
+ .with_null_values(null_values)
115
+ .with_try_parse_dates(try_parse_dates)
116
+ .with_quote_char(quote_char)
117
+ .with_eol_char(eol_char)
118
+ .with_truncate_ragged_lines(truncate_ragged_lines)
119
+ .with_decimal_comma(decimal_comma),
120
+ )
121
+ .into_reader_with_file_handle(mmap_bytes_r)
122
+ .finish()
123
+ })
123
124
  }
124
125
 
125
126
  pub fn read_json(
127
+ rb: &Ruby,
126
128
  rb_f: Value,
127
129
  infer_schema_length: Option<usize>,
128
130
  schema: Option<Wrap<Schema>>,
@@ -131,23 +133,25 @@ impl RbDataFrame {
131
133
  let rb_f = read_if_bytesio(rb_f);
132
134
  let mmap_bytes_r = get_mmap_bytes_reader(&rb_f)?;
133
135
 
134
- let mut builder = JsonReader::new(mmap_bytes_r)
135
- .with_json_format(JsonFormat::Json)
136
- .infer_schema_len(infer_schema_length.and_then(NonZeroUsize::new));
136
+ rb.enter_polars_df(move || {
137
+ let mut reader = JsonReader::new(mmap_bytes_r)
138
+ .with_json_format(JsonFormat::Json)
139
+ .infer_schema_len(infer_schema_length.and_then(NonZeroUsize::new));
137
140
 
138
- if let Some(schema) = schema {
139
- builder = builder.with_schema(Arc::new(schema.0));
140
- }
141
+ if let Some(schema) = schema {
142
+ reader = reader.with_schema(Arc::new(schema.0));
143
+ }
141
144
 
142
- if let Some(schema) = schema_overrides.as_ref() {
143
- builder = builder.with_schema_overwrite(&schema.0);
144
- }
145
+ if let Some(schema) = schema_overrides.as_ref() {
146
+ reader = reader.with_schema_overwrite(&schema.0);
147
+ }
145
148
 
146
- let out = builder.finish().map_err(RbPolarsErr::from)?;
147
- Ok(out.into())
149
+ reader.finish()
150
+ })
148
151
  }
149
152
 
150
153
  pub fn read_ipc(
154
+ rb: &Ruby,
151
155
  rb_f: Value,
152
156
  columns: Option<Vec<String>>,
153
157
  projection: Option<Vec<usize>>,
@@ -163,18 +167,19 @@ impl RbDataFrame {
163
167
  let (mmap_bytes_r, mmap_path) = get_mmap_bytes_reader_and_path(&rb_f)?;
164
168
 
165
169
  let mmap_path = if memory_map { mmap_path } else { None };
166
- let df = IpcReader::new(mmap_bytes_r)
167
- .with_projection(projection)
168
- .with_columns(columns)
169
- .with_n_rows(n_rows)
170
- .with_row_index(row_index)
171
- .memory_mapped(mmap_path)
172
- .finish()
173
- .map_err(RbPolarsErr::from)?;
174
- Ok(RbDataFrame::new(df))
170
+ rb.enter_polars_df(move || unsafe {
171
+ IpcReader::new(mmap_bytes_r)
172
+ .with_projection(projection)
173
+ .with_columns(columns)
174
+ .with_n_rows(n_rows)
175
+ .with_row_index(row_index)
176
+ .memory_mapped(mmap_path)
177
+ .finish()
178
+ })
175
179
  }
176
180
 
177
181
  pub fn read_ipc_stream(
182
+ rb: &Ruby,
178
183
  rb_f: Value,
179
184
  columns: Option<Vec<String>>,
180
185
  projection: Option<Vec<usize>>,
@@ -188,18 +193,19 @@ impl RbDataFrame {
188
193
  });
189
194
  let rb_f = read_if_bytesio(rb_f);
190
195
  let mmap_bytes_r = get_mmap_bytes_reader(&rb_f)?;
191
- let df = IpcStreamReader::new(mmap_bytes_r)
192
- .with_projection(projection)
193
- .with_columns(columns)
194
- .with_n_rows(n_rows)
195
- .with_row_index(row_index)
196
- .set_rechunk(rechunk)
197
- .finish()
198
- .map_err(RbPolarsErr::from)?;
199
- Ok(RbDataFrame::new(df))
196
+ rb.enter_polars_df(move || {
197
+ IpcStreamReader::new(mmap_bytes_r)
198
+ .with_projection(projection)
199
+ .with_columns(columns)
200
+ .with_n_rows(n_rows)
201
+ .with_row_index(row_index)
202
+ .set_rechunk(rechunk)
203
+ .finish()
204
+ })
200
205
  }
201
206
 
202
207
  pub fn read_avro(
208
+ rb: &Ruby,
203
209
  rb_f: Value,
204
210
  columns: Option<Vec<String>>,
205
211
  projection: Option<Vec<usize>>,
@@ -208,53 +214,54 @@ impl RbDataFrame {
208
214
  use polars::io::avro::AvroReader;
209
215
 
210
216
  let file = get_file_like(rb_f, false)?;
211
- let df = AvroReader::new(file)
212
- .with_projection(projection)
213
- .with_columns(columns)
214
- .with_n_rows(n_rows)
215
- .finish()
216
- .map_err(RbPolarsErr::from)?;
217
- Ok(RbDataFrame::new(df))
217
+ rb.enter_polars_df(move || {
218
+ AvroReader::new(file)
219
+ .with_projection(projection)
220
+ .with_columns(columns)
221
+ .with_n_rows(n_rows)
222
+ .finish()
223
+ })
218
224
  }
219
225
 
220
- pub fn write_json(&self, rb_f: Value) -> RbResult<()> {
226
+ pub fn write_json(rb: &Ruby, self_: &Self, rb_f: Value) -> RbResult<()> {
221
227
  let file = BufWriter::new(get_file_like(rb_f, true)?);
222
-
223
- JsonWriter::new(file)
224
- .with_json_format(JsonFormat::Json)
225
- .finish(&mut self.df.write())
226
- .map_err(RbPolarsErr::from)?;
227
- Ok(())
228
+ rb.enter_polars(|| {
229
+ JsonWriter::new(file)
230
+ .with_json_format(JsonFormat::Json)
231
+ .finish(&mut self_.df.write())
232
+ })
228
233
  }
229
234
 
230
235
  pub fn write_ipc_stream(
231
- &self,
236
+ rb: &Ruby,
237
+ self_: &Self,
232
238
  rb_f: Value,
233
239
  compression: Wrap<Option<IpcCompression>>,
234
240
  compat_level: RbCompatLevel,
235
241
  ) -> RbResult<()> {
236
242
  let mut buf = get_file_like(rb_f, true)?;
237
- IpcStreamWriter::new(&mut buf)
238
- .with_compression(compression.0)
239
- .with_compat_level(compat_level.0)
240
- .finish(&mut self.df.write())
241
- .map_err(RbPolarsErr::from)?;
242
- Ok(())
243
+ rb.enter_polars(|| {
244
+ IpcStreamWriter::new(&mut buf)
245
+ .with_compression(compression.0)
246
+ .with_compat_level(compat_level.0)
247
+ .finish(&mut self_.df.write())
248
+ })
243
249
  }
244
250
 
245
251
  pub fn write_avro(
246
- &self,
252
+ rb: &Ruby,
253
+ self_: &Self,
247
254
  rb_f: Value,
248
255
  compression: Wrap<Option<AvroCompression>>,
249
256
  name: String,
250
257
  ) -> RbResult<()> {
251
258
  use polars::io::avro::AvroWriter;
252
259
  let mut buf = get_file_like(rb_f, true)?;
253
- AvroWriter::new(&mut buf)
254
- .with_compression(compression.0)
255
- .with_name(name)
256
- .finish(&mut self.df.write())
257
- .map_err(RbPolarsErr::from)?;
258
- Ok(())
260
+ rb.enter_polars(|| {
261
+ AvroWriter::new(&mut buf)
262
+ .with_compression(compression.0)
263
+ .with_name(name)
264
+ .finish(&mut self_.df.write())
265
+ })
259
266
  }
260
267
  }
@@ -473,6 +473,10 @@ impl RbExpr {
473
473
  .into()
474
474
  }
475
475
 
476
+ pub fn is_sorted(&self, descending: Option<bool>, nulls_last: Option<bool>) -> Self {
477
+ self.inner.clone().is_sorted(descending, nulls_last).into()
478
+ }
479
+
476
480
  pub fn approx_n_unique(&self) -> Self {
477
481
  self.inner.clone().approx_n_unique().into()
478
482
  }
@@ -815,7 +819,7 @@ impl RbExpr {
815
819
  &self,
816
820
  n: &Self,
817
821
  with_replacement: bool,
818
- shuffle: bool,
822
+ shuffle: Option<bool>,
819
823
  seed: Option<u64>,
820
824
  ) -> Self {
821
825
  self.inner
@@ -828,7 +832,7 @@ impl RbExpr {
828
832
  &self,
829
833
  frac: &Self,
830
834
  with_replacement: bool,
831
- shuffle: bool,
835
+ shuffle: Option<bool>,
832
836
  seed: Option<u64>,
833
837
  ) -> Self {
834
838
  self.inner
@@ -143,7 +143,7 @@ impl RbExpr {
143
143
  &self,
144
144
  n: &RbExpr,
145
145
  with_replacement: bool,
146
- shuffle: bool,
146
+ shuffle: Option<bool>,
147
147
  seed: Option<u64>,
148
148
  ) -> Self {
149
149
  self.inner
@@ -157,7 +157,7 @@ impl RbExpr {
157
157
  &self,
158
158
  fraction: &RbExpr,
159
159
  with_replacement: bool,
160
- shuffle: bool,
160
+ shuffle: Option<bool>,
161
161
  seed: Option<u64>,
162
162
  ) -> Self {
163
163
  self.inner