carray 2.0.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.yardopts +5 -25
- data/CHANGELOG.md +16 -0
- data/LICENSE +1 -1
- data/NEWS.md +3 -0
- data/README.md +128 -44
- data/carray.gemspec +22 -24
- data/ext/ca_array_pool.c +91 -0
- data/ext/ca_axis_descriptor.h +186 -0
- data/ext/ca_axis_dispatch.c +924 -0
- data/ext/ca_axis_group.c +1208 -0
- data/ext/ca_bincmp_dispatch.c +76 -0
- data/ext/ca_bincmp_dispatch.h +85 -0
- data/ext/ca_binop_dispatch.c +125 -0
- data/ext/ca_binop_dispatch.h +159 -0
- data/ext/ca_categorical_iterator.c +1375 -0
- data/ext/ca_compare.c +94 -0
- data/ext/ca_compare.h +26 -0
- data/ext/ca_composite_dispatch.c +414 -0
- data/ext/ca_composite_dispatch.h +116 -0
- data/ext/ca_for_buffer.h +96 -0
- data/ext/ca_for_each_element.h +241 -0
- data/ext/ca_group_iter.c +304 -0
- data/ext/ca_iter_substrate.h +325 -0
- data/ext/ca_kernel_iterator.c +4321 -0
- data/ext/ca_kernel_iterator.h +2603 -0
- data/ext/ca_moncmp_dispatch.c +37 -0
- data/ext/ca_moncmp_dispatch.h +62 -0
- data/ext/ca_monop_dispatch.c +200 -0
- data/ext/ca_monop_dispatch.h +235 -0
- data/ext/ca_obj_array.c +355 -359
- data/ext/ca_obj_bincmp.c +809 -0
- data/ext/ca_obj_binop.c +892 -0
- data/ext/ca_obj_bitarray.c +369 -164
- data/ext/ca_obj_bitfield.c +294 -234
- data/ext/ca_obj_block.c +189 -711
- data/ext/ca_obj_byte_swap.c +766 -0
- data/ext/ca_obj_const_string.c +965 -0
- data/ext/ca_obj_face.c +670 -0
- data/ext/ca_obj_face.h +247 -0
- data/ext/ca_obj_fake.c +228 -100
- data/ext/ca_obj_farray.c +54 -441
- data/ext/ca_obj_field.c +82 -529
- data/ext/ca_obj_fixlen_string.c +306 -0
- data/ext/ca_obj_grid.c +858 -440
- data/ext/ca_obj_meld.c +1034 -0
- data/ext/ca_obj_moncmp.c +569 -0
- data/ext/ca_obj_monop.c +1111 -0
- data/ext/ca_obj_object.c +774 -298
- data/ext/ca_obj_record.c +468 -0
- data/ext/ca_obj_reduce.c +97 -82
- data/ext/ca_obj_refer.c +569 -459
- data/ext/ca_obj_remap.c +475 -0
- data/ext/ca_obj_repeat.c +92 -477
- data/ext/ca_obj_roll.c +616 -0
- data/ext/ca_obj_select.c +344 -296
- data/ext/ca_obj_select_axis.c +1296 -0
- data/ext/ca_obj_shift.c +230 -792
- data/ext/ca_obj_source.c +78 -0
- data/ext/ca_obj_stack.c +1173 -0
- data/ext/ca_obj_stride.c +2501 -0
- data/ext/ca_obj_string.c +268 -0
- data/ext/ca_obj_tile.c +614 -0
- data/ext/ca_obj_time.c +546 -0
- data/ext/ca_obj_timedelta.c +435 -0
- data/ext/ca_obj_transpose.c +62 -516
- data/ext/ca_obj_triop.c +746 -0
- data/ext/ca_obj_unbound_repeat.c +208 -241
- data/ext/ca_obj_window.c +1131 -563
- data/ext/ca_op_byte_swap.c +175 -0
- data/ext/ca_op_ipower.c +319 -0
- data/ext/ca_op_powi.h +88 -0
- data/ext/ca_sort_kernels.h +132 -0
- data/ext/ca_sweep_engine.c +430 -0
- data/ext/ca_sweep_engine.h +157 -0
- data/ext/ca_transform_common.c +228 -0
- data/ext/ca_triop_dispatch.c +55 -0
- data/ext/ca_triop_dispatch.h +62 -0
- data/ext/carray.h +795 -402
- data/ext/carray_access.c +831 -711
- data/ext/carray_attribute.c +98 -330
- data/ext/carray_bincount.c +255 -0
- data/ext/carray_broadcast.c +283 -0
- data/ext/carray_call_cfunc.c +1360 -828
- data/ext/carray_call_cfunc.h +160 -0
- data/ext/carray_cast.c +1212 -301
- data/ext/carray_cast_func.rb +81 -40
- data/ext/carray_class.c +53 -63
- data/ext/carray_config.h +28 -0
- data/ext/carray_conversion.c +350 -346
- data/ext/carray_copy.c +156 -268
- data/ext/carray_core.c +1342 -199
- data/ext/carray_count.c +312 -0
- data/ext/carray_data_type.c +43 -19
- data/ext/carray_element.c +585 -213
- data/ext/carray_factorize.c +2542 -0
- data/ext/carray_generate.c +230 -559
- data/ext/carray_histogram.c +490 -0
- data/ext/carray_hold.c +228 -0
- data/ext/carray_index_classifier.c +1035 -0
- data/ext/carray_index_classifier.h +27 -0
- data/ext/carray_internal.h +120 -0
- data/ext/carray_kernels_bincmp.c +4445 -0
- data/ext/carray_kernels_binop.c +10979 -0
- data/ext/carray_kernels_init.c +36 -0
- data/ext/carray_kernels_map.c +3466 -0
- data/ext/carray_kernels_moncmp.c +2096 -0
- data/ext/carray_kernels_monop.c +18312 -0
- data/ext/carray_kernels_reduce_aggregate.c +25836 -0
- data/ext/carray_kernels_reduce_boolean.c +329 -0
- data/ext/carray_kernels_reduce_cumulative.c +14592 -0
- data/ext/carray_kernels_reduce_extreme.c +16947 -0
- data/ext/carray_kernels_reduce_variance.c +3909 -0
- data/ext/carray_kernels_scan.c +3692 -0
- data/ext/carray_kernels_search.c +32137 -0
- data/ext/carray_kernels_sort.c +10625 -0
- data/ext/carray_kernels_triop.c +1391 -0
- data/ext/carray_lazy.c +567 -0
- data/ext/carray_loop.c +88 -200
- data/ext/carray_mask.c +848 -154
- data/ext/carray_math_kernel.h +120 -0
- data/ext/carray_mathfunc.c +10 -241
- data/ext/carray_median_percentile.c +1257 -0
- data/ext/carray_memory_view.c +1625 -0
- data/ext/carray_operator.c +1526 -318
- data/ext/carray_order.c +664 -1394
- data/ext/carray_partition.c +416 -0
- data/ext/carray_random.c +518 -0
- data/ext/carray_scatter.c +357 -0
- data/ext/carray_slab.c +1219 -0
- data/ext/carray_slab.h +84 -0
- data/ext/carray_sort.c +829 -0
- data/ext/carray_sort_kernel.c +620 -0
- data/ext/carray_struct.c +695 -0
- data/ext/carray_test.c +343 -229
- data/ext/carray_undef.c +34 -17
- data/ext/carray_utils.c +175 -74
- data/ext/extconf.rb +216 -55
- data/ext/mk_call_cfunc.rb +480 -0
- data/ext/mkkernel.rb +8842 -0
- data/ext/ruby_carray.c +202 -101
- data/ext/version.h +4 -14
- data/ext/version.rb +5 -13
- data/lib/carray/arrow_tensor.rb +401 -0
- data/lib/carray/attribute.rb +166 -0
- data/lib/carray/autoload_carray.rb +220 -0
- data/lib/carray/autoload_method_extension.rb +44 -0
- data/lib/carray/axis_group.rb +711 -0
- data/lib/carray/basics.rb +481 -0
- data/lib/carray/bincount_nd.rb +358 -0
- data/lib/carray/block_iterator.rb +604 -0
- data/lib/carray/boolean_reduce.rb +109 -0
- data/lib/carray/categorical.rb +561 -0
- data/lib/carray/categorical_iterator.rb +1062 -0
- data/lib/carray/complex.rb +150 -0
- data/lib/carray/conditional.rb +216 -0
- data/lib/carray/const_string.rb +228 -0
- data/lib/carray/construct.rb +139 -328
- data/lib/carray/core_extensions.rb +240 -0
- data/lib/carray/data_type_extension.rb +233 -0
- data/lib/carray/fixlen_string.rb +95 -0
- data/lib/carray/frame/concat.rb +132 -0
- data/lib/carray/frame/convert.rb +95 -0
- data/lib/carray/frame/csv_parser.rb +211 -0
- data/lib/carray/frame/frame.rb +649 -0
- data/lib/carray/frame/group.rb +186 -0
- data/lib/carray/frame/io.rb +164 -0
- data/lib/carray/frame/join.rb +248 -0
- data/lib/carray/frame/records.rb +99 -0
- data/lib/carray/frame/sort.rb +113 -0
- data/lib/carray/frame/verbs.rb +299 -0
- data/lib/carray/frame.rb +16 -0
- data/lib/carray/histogram.rb +512 -0
- data/lib/carray/inspect.rb +37 -20
- data/lib/carray/iterator.rb +57 -349
- data/lib/carray/lazy.rb +889 -0
- data/lib/carray/mask_gap_fill.rb +200 -0
- data/lib/carray/math.rb +78 -342
- data/lib/carray/meld_reduce.rb +289 -0
- data/lib/carray/methods/align_addr.rb +116 -0
- data/lib/carray/methods/bin.rb +128 -0
- data/lib/carray/methods/bincount.rb +87 -0
- data/lib/carray/methods/bit_string.rb +92 -0
- data/lib/carray/methods/broadcast.rb +63 -0
- data/lib/carray/methods/choose.rb +39 -0
- data/lib/carray/methods/composition.rb +280 -0
- data/lib/carray/methods/gather_nd.rb +206 -0
- data/lib/carray/methods/index.rb +39 -0
- data/lib/carray/methods/insert_block.rb +99 -0
- data/lib/carray/methods/is_in.rb +141 -0
- data/lib/carray/methods/join.rb +90 -0
- data/lib/carray/methods/locate_addr.rb +47 -0
- data/lib/carray/methods/mask_duplicates.rb +41 -0
- data/lib/carray/methods/meshgrid.rb +91 -0
- data/lib/carray/methods/mode.rb +126 -0
- data/lib/carray/methods/nunique.rb +46 -0
- data/lib/carray/methods/resize.rb +56 -0
- data/lib/carray/methods/snap.rb +156 -0
- data/lib/carray/methods/string_format.rb +57 -0
- data/lib/carray/methods/unique.rb +47 -0
- data/lib/carray/methods/value_counts.rb +71 -0
- data/lib/carray/mkmf.rb +124 -101
- data/lib/carray/runtime.rb +108 -0
- data/lib/carray/serialize.rb +478 -167
- data/lib/carray/slab_iterator.rb +292 -0
- data/lib/carray/stack.rb +291 -0
- data/lib/carray/string.rb +56 -180
- data/lib/carray/string_operation_extension.rb +289 -0
- data/lib/carray/struct.rb +335 -323
- data/lib/carray/struct_builder.rb +697 -0
- data/lib/carray/table.rb +41 -2
- data/lib/carray/time.rb +2255 -38
- data/lib/carray/window_iterator.rb +655 -0
- data/lib/carray.rb +55 -57
- metadata +163 -130
- data/Rakefile +0 -51
- data/TODO.md +0 -18
- data/ext/ca_iter_block.c +0 -257
- data/ext/ca_iter_dimension.c +0 -299
- data/ext/ca_iter_window.c +0 -214
- data/ext/ca_obj_mapping.c +0 -644
- data/ext/carray_iterator.c +0 -641
- data/ext/carray_math.rb +0 -850
- data/ext/carray_numeric.c +0 -259
- data/ext/carray_sort_addr.c +0 -254
- data/ext/carray_stat.c +0 -2100
- data/ext/carray_stat_proc.rb +0 -1999
- data/ext/mkmath.rb +0 -741
- data/ext/ruby_ccomplex.c +0 -509
- data/ext/ruby_float_func.c +0 -86
- data/lib/carray/array.rb +0 -8
- data/lib/carray/autoload/autoload_base.rb +0 -19
- data/lib/carray/autoload/autoload_gem_cairo.rb +0 -9
- data/lib/carray/autoload/autoload_gem_ffi.rb +0 -9
- data/lib/carray/autoload/autoload_gem_gnuplot.rb +0 -2
- data/lib/carray/autoload/autoload_gem_io_csv.rb +0 -14
- data/lib/carray/autoload/autoload_gem_io_pg.rb +0 -6
- data/lib/carray/autoload/autoload_gem_io_sqlite3.rb +0 -12
- data/lib/carray/autoload/autoload_gem_narray.rb +0 -10
- data/lib/carray/autoload/autoload_gem_numo_narray.rb +0 -15
- data/lib/carray/autoload/autoload_gem_opencv.rb +0 -16
- data/lib/carray/autoload/autoload_gem_random.rb +0 -8
- data/lib/carray/autoload/autoload_gem_rmagick.rb +0 -23
- data/lib/carray/autoload/autoload_gem_zimg.rb +0 -3
- data/lib/carray/autoload/autoload_io_imagemagick.rb +0 -6
- data/lib/carray/autoload/autoload_math_histogram.rb +0 -5
- data/lib/carray/autoload/autoload_math_recurrence.rb +0 -6
- data/lib/carray/autoload/autoload_object_iterator.rb +0 -1
- data/lib/carray/autoload/autoload_object_link.rb +0 -1
- data/lib/carray/autoload/autoload_object_pack.rb +0 -2
- data/lib/carray/autoload.rb +0 -141
- data/lib/carray/basic.rb +0 -191
- data/lib/carray/broadcast.rb +0 -101
- data/lib/carray/compose.rb +0 -315
- data/lib/carray/convert.rb +0 -115
- data/lib/carray/info.rb +0 -110
- data/lib/carray/io/imagemagick.rb +0 -235
- data/lib/carray/mask.rb +0 -102
- data/lib/carray/math/histogram.rb +0 -177
- data/lib/carray/math/recurrence.rb +0 -93
- data/lib/carray/object/ca_obj_iterator.rb +0 -50
- data/lib/carray/object/ca_obj_link.rb +0 -50
- data/lib/carray/object/ca_obj_pack.rb +0 -99
- data/lib/carray/obsolete.rb +0 -256
- data/lib/carray/ordering.rb +0 -181
- data/lib/carray/testing.rb +0 -51
- data/lib/carray/transform.rb +0 -109
- data/misc/Methods.ja.md +0 -182
- data/misc/NOTE +0 -51
- data/spec/Classes/CABitfield_spec.rb +0 -58
- data/spec/Classes/CABlockIterator_spec.rb +0 -114
- data/spec/Classes/CABlock_spec.rb +0 -205
- data/spec/Classes/CAField_spec.rb +0 -39
- data/spec/Classes/CAGrid_spec.rb +0 -75
- data/spec/Classes/CAMap_spec.rb +0 -0
- data/spec/Classes/CAMapping_spec.rb +0 -105
- data/spec/Classes/CAObject_attribute_spec.rb +0 -33
- data/spec/Classes/CAObject_spec.rb +0 -33
- data/spec/Classes/CARefer_spec.rb +0 -93
- data/spec/Classes/CARepeat_spec.rb +0 -65
- data/spec/Classes/CASelect_spec.rb +0 -22
- data/spec/Classes/CAShift_spec.rb +0 -16
- data/spec/Classes/CAStruct_spec.rb +0 -71
- data/spec/Classes/CATranspose_spec.rb +0 -60
- data/spec/Classes/CAUnboudRepeat_spec.rb +0 -102
- data/spec/Classes/CAWindow_spec.rb +0 -54
- data/spec/Classes/CAWrap_spec.rb +0 -8
- data/spec/Classes/CArray_spec.rb +0 -184
- data/spec/Classes/CScalar_spec.rb +0 -55
- data/spec/Classes/ex1.rb +0 -46
- data/spec/Features/feature_130_spec.rb +0 -19
- data/spec/Features/feature_attributes_spec.rb +0 -280
- data/spec/Features/feature_boolean_spec.rb +0 -98
- data/spec/Features/feature_broadcast.rb +0 -116
- data/spec/Features/feature_cast_function.rb +0 -19
- data/spec/Features/feature_cast_spec.rb +0 -33
- data/spec/Features/feature_class_spec.rb +0 -84
- data/spec/Features/feature_complex_spec.rb +0 -42
- data/spec/Features/feature_composite_spec.rb +0 -124
- data/spec/Features/feature_convert_spec.rb +0 -46
- data/spec/Features/feature_copy_spec.rb +0 -123
- data/spec/Features/feature_creation_spec.rb +0 -84
- data/spec/Features/feature_element_spec.rb +0 -144
- data/spec/Features/feature_extream_spec.rb +0 -54
- data/spec/Features/feature_generate_spec.rb +0 -74
- data/spec/Features/feature_index_spec.rb +0 -69
- data/spec/Features/feature_mask_spec.rb +0 -580
- data/spec/Features/feature_math_spec.rb +0 -97
- data/spec/Features/feature_order_spec.rb +0 -146
- data/spec/Features/feature_ref_store_spec.rb +0 -209
- data/spec/Features/feature_serialization_spec.rb +0 -125
- data/spec/Features/feature_stat_spec.rb +0 -397
- data/spec/Features/feature_virtual_spec.rb +0 -48
- data/spec/Features/method_eq_spec.rb +0 -81
- data/spec/Features/method_is_nan_spec.rb +0 -12
- data/spec/Features/method_map_spec.rb +0 -54
- data/spec/Features/method_max_with.rb +0 -20
- data/spec/Features/method_min_with.rb +0 -19
- data/spec/Features/method_ne_spec.rb +0 -18
- data/spec/Features/method_project_spec.rb +0 -188
- data/spec/Features/method_ref_spec.rb +0 -27
- data/spec/Features/method_round_spec.rb +0 -11
- data/spec/Features/method_s_linspace_spec.rb +0 -48
- data/spec/Features/method_s_span_spec.rb +0 -14
- data/spec/Features/method_seq_spec.rb +0 -47
- data/spec/Features/method_sort_with.rb +0 -43
- data/spec/Features/method_sorted_with.rb +0 -29
- data/spec/Features/method_span_spec.rb +0 -42
- data/spec/Features/method_wrap_readonly_spec.rb +0 -43
- data/spec/UnitTest/test_CAVirtual.rb +0 -214
- data/spec/spec_all.rb +0 -10
- data/utils/ca_ase.rb +0 -21
- data/utils/ca_methods.rb +0 -15
- data/utils/cast_checker.rb +0 -30
- data/utils/convert_test.rb +0 -73
- data/utils/extract_yard.rb +0 -22
- data/utils/guess_shape.rb +0 -76
- data/utils/monkey_patch_methods.rb +0 -62
- data/utils/remove_resource_fork.sh +0 -5
|
@@ -0,0 +1,649 @@
|
|
|
1
|
+
# CAFrame — DataFrame built on CArray columns.
|
|
2
|
+
#
|
|
3
|
+
# Internal structure (memo §3): a Hash of named columns, an axis name for
|
|
4
|
+
# the row axis, and an optional index column. Every column agrees on its
|
|
5
|
+
# axis-0 length N; trailing shape is free per column (§3.2). The only
|
|
6
|
+
# substance the frame adds is names — the columns themselves are borrowed
|
|
7
|
+
# CArrays and are handed back raw so callers escape to CArray (§4.3).
|
|
8
|
+
#
|
|
9
|
+
# View semantics follow CArray (§3.6): +df["col"]+ is the stored column
|
|
10
|
+
# itself (an alias), row slices and filters return view-frames sharing
|
|
11
|
+
# storage, and +copy+ is the way to an independent frame.
|
|
12
|
+
|
|
13
|
+
class CAFrame
|
|
14
|
+
# Integer data_types that select rows positionally when used as a df[] key.
|
|
15
|
+
INTEGER_TYPES = [:int8, :int16, :int32, :int64,
|
|
16
|
+
:uint8, :uint16, :uint32, :uint64].freeze
|
|
17
|
+
private_constant :INTEGER_TYPES
|
|
18
|
+
|
|
19
|
+
# @!visibility private
|
|
20
|
+
DEFAULT_AXIS_NAME = "row"
|
|
21
|
+
|
|
22
|
+
# Build a frame from a Hash of +name => column+. Columns may be CArrays or
|
|
23
|
+
# anything that answers +to_ca+ (Ruby Array, lazy view). All columns must
|
|
24
|
+
# share axis-0 length N.
|
|
25
|
+
#
|
|
26
|
+
# The column hash may be passed either explicitly (+CAFrame.new(hash,
|
|
27
|
+
# axis_name: ...)+) or as bare inline pairs (+CAFrame.new("a" => x, "b" =>
|
|
28
|
+
# y)+). +:axis_name+ / +:index+ are control options; any remaining
|
|
29
|
+
# (string-keyed) options are treated as columns.
|
|
30
|
+
#
|
|
31
|
+
# Column names are normalized to Strings, so a Symbol key in an explicit
|
|
32
|
+
# column hash is stringified. The Symbol rejection below applies only to the
|
|
33
|
+
# keyword channel, which doubles as the +:axis_name+ / +:index+ control-option
|
|
34
|
+
# channel: a stray Symbol there is a mistyped control option, not a column.
|
|
35
|
+
def initialize(columns = {}, **opts)
|
|
36
|
+
axis_name = opts.delete(:axis_name)
|
|
37
|
+
index = opts.delete(:index)
|
|
38
|
+
stray = opts.keys.reject { |k| k.is_a?(String) }
|
|
39
|
+
unless stray.empty?
|
|
40
|
+
raise ArgumentError,
|
|
41
|
+
"column keys must be Strings (Symbols are reserved); got #{stray.inspect}"
|
|
42
|
+
end
|
|
43
|
+
columns = columns.merge(opts) unless opts.empty?
|
|
44
|
+
|
|
45
|
+
@columns = {}
|
|
46
|
+
@axis_name = axis_name || DEFAULT_AXIS_NAME
|
|
47
|
+
@index = nil
|
|
48
|
+
|
|
49
|
+
n = nil
|
|
50
|
+
columns.each do |name, col|
|
|
51
|
+
key = name.to_s
|
|
52
|
+
ca = coerce_column(col)
|
|
53
|
+
len = ca.shape[0]
|
|
54
|
+
if n.nil?
|
|
55
|
+
n = len
|
|
56
|
+
elsif len != n
|
|
57
|
+
raise ArgumentError,
|
|
58
|
+
"column #{key.inspect} has axis-0 length #{len}, expected #{n}"
|
|
59
|
+
end
|
|
60
|
+
@columns[key] = ca
|
|
61
|
+
end
|
|
62
|
+
@nrow = n || 0
|
|
63
|
+
|
|
64
|
+
if index
|
|
65
|
+
idx = coerce_column(index)
|
|
66
|
+
unless idx.ndim == 1
|
|
67
|
+
raise ArgumentError, "index must be a 1-D column (got ndim #{idx.ndim})"
|
|
68
|
+
end
|
|
69
|
+
if n && idx.shape[0] != n
|
|
70
|
+
raise ArgumentError,
|
|
71
|
+
"index length #{idx.shape[0]} does not match nrow #{n}"
|
|
72
|
+
end
|
|
73
|
+
@index = idx
|
|
74
|
+
@nrow = idx.shape[0] if n.nil?
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# With an index, the row axis is named after it; a column of the same name
|
|
78
|
+
# would shadow the index in +row+ and +reset_index+. Reject the collision at
|
|
79
|
+
# construction so those paths never silently drop one for the other.
|
|
80
|
+
if @index && @columns.key?(@axis_name)
|
|
81
|
+
raise ArgumentError,
|
|
82
|
+
"axis_name #{@axis_name.inspect} collides with a column of the same name"
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# --- df[...] : key type decides the axis (memo §13.2) ------------------
|
|
87
|
+
|
|
88
|
+
# String -> a column (raw CArray, the escape unit)
|
|
89
|
+
# String x2+ -> Array<CArray> (the escape unit, plural)
|
|
90
|
+
# Integer -> a row (Ruby Hash)
|
|
91
|
+
# Range(int)-> positional row slice (view-frame)
|
|
92
|
+
# boolean CA-> row filter (view-frame)
|
|
93
|
+
# integer CA-> row gather (view-frame)
|
|
94
|
+
#
|
|
95
|
+
# String keys escape: df[...] hands back the raw column(s), never a frame.
|
|
96
|
+
# One name collapses to a bare CArray; several give an Array of them (the
|
|
97
|
+
# "single collapses, plural is an array" rule of ca[i] vs ca[i..j]), so
|
|
98
|
+
# +t, rh = df["temp", "rh"]+ destructures. A column-subset *frame* comes
|
|
99
|
+
# from +select+ (memo §13.2).
|
|
100
|
+
def [](*keys)
|
|
101
|
+
if keys.size > 1
|
|
102
|
+
unless keys.all? { |k| k.is_a?(String) }
|
|
103
|
+
raise ArgumentError,
|
|
104
|
+
"multi-key df[...] escapes columns; every key must be a String " \
|
|
105
|
+
"(use df.select(...) for a subset frame)"
|
|
106
|
+
end
|
|
107
|
+
return keys.map { |k| @columns.fetch(k) { raise KeyError, "no column #{k.inspect}" } }
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
key = keys.first
|
|
111
|
+
case key
|
|
112
|
+
when String
|
|
113
|
+
@columns.fetch(key) { raise KeyError, "no column #{key.inspect}" }
|
|
114
|
+
when Integer
|
|
115
|
+
row(key)
|
|
116
|
+
when Range
|
|
117
|
+
unless positional_range?(key)
|
|
118
|
+
raise ArgumentError,
|
|
119
|
+
"df[range] takes positional (integer) ranges only; " \
|
|
120
|
+
"use filter { |f| f.index ... } for label ranges"
|
|
121
|
+
end
|
|
122
|
+
select_rows(key)
|
|
123
|
+
when CArray
|
|
124
|
+
case key.data_type
|
|
125
|
+
when :boolean
|
|
126
|
+
select_rows(key)
|
|
127
|
+
when *INTEGER_TYPES
|
|
128
|
+
select_rows(key)
|
|
129
|
+
else
|
|
130
|
+
raise ArgumentError,
|
|
131
|
+
"CArray df[] key must be boolean or integer (got #{key.data_type})"
|
|
132
|
+
end
|
|
133
|
+
when Symbol
|
|
134
|
+
raise NotImplementedError, "df[symbol] is reserved for predicate keys"
|
|
135
|
+
else
|
|
136
|
+
raise ArgumentError, "unsupported df[] key: #{key.class}"
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# --- df[...] = value : row-level assignment ---------------------------
|
|
141
|
+
#
|
|
142
|
+
# The RHS value decides the operation; the key selects the rows:
|
|
143
|
+
#
|
|
144
|
+
# df[sel] = UNDEF -> mask the selected rows across every column, in place.
|
|
145
|
+
# Shape is unchanged and the write goes through to the
|
|
146
|
+
# column storage; the index stays, so the rows remain
|
|
147
|
+
# identifiable.
|
|
148
|
+
# df[sel] = nil -> delete the selected rows. The frame shrinks and the
|
|
149
|
+
# survivors keep their order.
|
|
150
|
+
# df[sel] = other -> splice: replace the selected (contiguous) rows with
|
|
151
|
+
# +other+'s rows. +other+ may carry any number of rows,
|
|
152
|
+
# so the row count changes (Ruby Array#[]= splice
|
|
153
|
+
# semantics); its column set must match exactly.
|
|
154
|
+
#
|
|
155
|
+
# +sel+ is a row-axis indexer key, classified exactly as CArray classifies a
|
|
156
|
+
# 1-D column key (docs/topics/Indexer_decision_tree.md): a slice (BLOCK —
|
|
157
|
+
# Range / ArithmeticSequence / [start, count, step]), a boolean CArray
|
|
158
|
+
# (SELECT), an integer CArray (GRID), or an Integer (a single row). Mask and
|
|
159
|
+
# delete take any of them and are forwarded to the column indexer, so their
|
|
160
|
+
# errors match the rest of CArray. Splice reuses the same classifier
|
|
161
|
+
# (CArray.scan_index) but needs a contiguous slice, so it takes an Integer or
|
|
162
|
+
# a step-1 BLOCK (Range, step-1 ArithmeticSequence, or [start, count]); a
|
|
163
|
+
# strided or scattered selector raises.
|
|
164
|
+
#
|
|
165
|
+
# Delete and splice rebuild the column set, so an alias previously taken
|
|
166
|
+
# from this frame no longer tracks the frame's new column identity
|
|
167
|
+
# (@columns is rebound to a fresh Hash). Reads through the old alias see
|
|
168
|
+
# the ORIGINAL column's data (unchanged by splice construction), matching
|
|
169
|
+
# the pre-splice snapshot. Writes flow along the CAMeld chain:
|
|
170
|
+
#
|
|
171
|
+
# - Writes reaching the HEAD or TAIL segment (the parts of the original
|
|
172
|
+
# column that survived the splice) land back in that column's storage.
|
|
173
|
+
# External aliases into the original column see those writes — this is
|
|
174
|
+
# the chain composability that CAMeld deliberately preserves: a
|
|
175
|
+
# reference stays connected to what it references.
|
|
176
|
+
# - Writes reaching the spliced MIDDLE segment (the rows from +other+)
|
|
177
|
+
# do NOT reach +other+'s storage: +other+ is snapshotted via +.copy+
|
|
178
|
+
# at splice time so it behaves like a value that was handed over.
|
|
179
|
+
# +df1[...] = df2+ followed by writes to df1's spliced rows leaves
|
|
180
|
+
# df2 untouched, which matches the usual "assignment" intuition for
|
|
181
|
+
# the RHS of +[]=+.
|
|
182
|
+
#
|
|
183
|
+
# Callers who want write isolation on the head/tail segments too can
|
|
184
|
+
# materialise explicitly with +col_view.copy+ before the splice, or
|
|
185
|
+
# +df["c"] = df["c"].copy+ afterwards. The shape-preserving +UNDEF+
|
|
186
|
+
# write path always propagates to derived views (unchanged from
|
|
187
|
+
# previous behaviour).
|
|
188
|
+
def []=(*keys, value)
|
|
189
|
+
if keys.size != 1
|
|
190
|
+
raise ArgumentError,
|
|
191
|
+
"df[...] = takes one key (a column name or a row selector); " \
|
|
192
|
+
"got #{keys.size}"
|
|
193
|
+
end
|
|
194
|
+
selector = keys.first
|
|
195
|
+
|
|
196
|
+
# The key picks the axis, exactly as it does when reading (§13.2): a
|
|
197
|
+
# String names a column, anything else selects rows.
|
|
198
|
+
if selector.is_a?(String)
|
|
199
|
+
column_assign(selector, value)
|
|
200
|
+
elsif UNDEF.equal?(value)
|
|
201
|
+
mask_rows(selector)
|
|
202
|
+
elsif value.nil?
|
|
203
|
+
delete_rows(selector)
|
|
204
|
+
elsif value.is_a?(CAFrame)
|
|
205
|
+
splice_rows(selector, value)
|
|
206
|
+
else
|
|
207
|
+
raise ArgumentError,
|
|
208
|
+
"df[...] = expects UNDEF (mask rows), nil (delete rows), or a " \
|
|
209
|
+
"CAFrame (splice rows); got #{value.class}"
|
|
210
|
+
end
|
|
211
|
+
value
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
# df["name"] = ... : the column forms of []=. Unlike the verbs, +[]=+ can
|
|
215
|
+
# only ever mutate the receiver -- Ruby hands the right-hand side back as
|
|
216
|
+
# the value of an assignment, so there is no way to return a new frame.
|
|
217
|
+
# Membership is per-frame (§12-B), so a parent frame's column set is
|
|
218
|
+
# untouched.
|
|
219
|
+
#
|
|
220
|
+
# df["temp"] = col -> bind the name to that column (a new name is added)
|
|
221
|
+
# df["temp"] = nil -> remove the column
|
|
222
|
+
# df["temp"] = UNDEF -> mask every cell of the column, in place
|
|
223
|
+
private def column_assign(key, value)
|
|
224
|
+
if value.nil?
|
|
225
|
+
delete_column(key)
|
|
226
|
+
elsif UNDEF.equal?(value)
|
|
227
|
+
mask_column(key)
|
|
228
|
+
elsif value.is_a?(CAFrame)
|
|
229
|
+
raise ArgumentError,
|
|
230
|
+
"df[name] = takes a column, not a CAFrame; escape one with " \
|
|
231
|
+
"other[\"name\"], or splice rows with df[rows] = other"
|
|
232
|
+
else
|
|
233
|
+
rebind_column(key, value)
|
|
234
|
+
end
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# Bind +key+ to +column+, adding the name when it is new. This is the one
|
|
238
|
+
# place a column enters an existing frame, so it is where the axis-0 length
|
|
239
|
+
# invariant is enforced (§12-A). It is a replacement rather than an edit,
|
|
240
|
+
# which is what sets it apart from the rest: +fill+ / +mask_eq+ /
|
|
241
|
+
# +df[rows] = UNDEF+ write to the shared column and are therefore visible
|
|
242
|
+
# wherever it is held, while this binds the name to a different column and
|
|
243
|
+
# leaves the old one alone (the same rule +cast+ follows, for the physical
|
|
244
|
+
# reason that a type change cannot reuse the buffer).
|
|
245
|
+
private def rebind_column(key, column)
|
|
246
|
+
if @index && key == @axis_name
|
|
247
|
+
raise ArgumentError,
|
|
248
|
+
"#{key.inspect} names the index, not a column; " \
|
|
249
|
+
"use set_index / reset_index to change the index"
|
|
250
|
+
end
|
|
251
|
+
ca = coerce_column(column)
|
|
252
|
+
len = ca.shape[0]
|
|
253
|
+
if @columns.empty? && @index.nil? && @nrow.zero?
|
|
254
|
+
@nrow = len # the first column of an empty frame fixes N
|
|
255
|
+
elsif len != @nrow
|
|
256
|
+
raise ArgumentError,
|
|
257
|
+
"column #{key.inspect} has axis-0 length #{len}, expected #{@nrow}"
|
|
258
|
+
end
|
|
259
|
+
@columns[key] = ca
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
# Remove a column. +drop+ is the same edit as a new frame (§3.8); this is
|
|
263
|
+
# the in-place form, which is all an assignment can be.
|
|
264
|
+
private def delete_column(key)
|
|
265
|
+
raise KeyError, "no column #{key.inspect}" unless @columns.key?(key)
|
|
266
|
+
@columns.delete(key)
|
|
267
|
+
@nrow = 0 if @columns.empty? && @index.nil?
|
|
268
|
+
self
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
# Mask every cell of a column, in place. The row form (+df[rows] = UNDEF+)
|
|
272
|
+
# masks the selected rows across every column; this is its column-wise
|
|
273
|
+
# counterpart, and like it the write goes through to the stored column and
|
|
274
|
+
# is visible through every alias.
|
|
275
|
+
private def mask_column(key)
|
|
276
|
+
col = @columns.fetch(key) { raise KeyError, "no column #{key.inspect}" }
|
|
277
|
+
col[] = UNDEF
|
|
278
|
+
self
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
# Column projection (memo §13.2): a view-frame holding the named columns as
|
|
282
|
+
# aliases (zero-copy, sharing storage with the parent frame). Distinct from
|
|
283
|
+
# +df[...]+, which escapes columns to raw CArrays, and from +filter+, which
|
|
284
|
+
# selects rows. +select+ takes no block — row conditions go through +filter+
|
|
285
|
+
# (old carray-dataframe fused the two into +select(cols) { cond }+; here they
|
|
286
|
+
# are orthogonal and compose by chaining: +df.select(...).filter { ... }+).
|
|
287
|
+
def select(*names)
|
|
288
|
+
raise ArgumentError, "select requires at least one column name" if names.empty?
|
|
289
|
+
cols = {}
|
|
290
|
+
names.each do |name|
|
|
291
|
+
key = name.to_s
|
|
292
|
+
raise KeyError, "no column #{key.inspect}" unless @columns.key?(key)
|
|
293
|
+
cols[key] = @columns[key]
|
|
294
|
+
end
|
|
295
|
+
rebuild(cols)
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
# Filter rows with a block that receives the frame and returns a boolean
|
|
299
|
+
# column (memo §15). The block builds the mask from +f["col"]+ / +f.index+.
|
|
300
|
+
# Keep the rows where the block's boolean selector is true.
|
|
301
|
+
#
|
|
302
|
+
# A masked (UNDEF) selector cell means the row's membership is *genuinely
|
|
303
|
+
# undetermined* (e.g. the predicate read a masked input). By default such
|
|
304
|
+
# rows are silently dropped, exactly as a false cell would be. Pass
|
|
305
|
+
# +keep_masked: true+ to carry the UNDEF forward instead: the undetermined
|
|
306
|
+
# rows survive into the result with their data cells masked (and their index
|
|
307
|
+
# value kept, so the row stays identifiable), leaving a later, better-informed
|
|
308
|
+
# pass to re-judge them. Definitely-true rows carry their values through
|
|
309
|
+
# unchanged in both modes.
|
|
310
|
+
def filter(keep_masked: false)
|
|
311
|
+
mask = yield(self)
|
|
312
|
+
unless mask.is_a?(CArray) && mask.data_type == :boolean
|
|
313
|
+
raise ArgumentError,
|
|
314
|
+
"filter block must return a boolean CArray (got #{mask.class})"
|
|
315
|
+
end
|
|
316
|
+
unless mask.shape[0] == @nrow
|
|
317
|
+
raise ArgumentError,
|
|
318
|
+
"filter mask has axis-0 length #{mask.shape[0]}, expected nrow #{@nrow}"
|
|
319
|
+
end
|
|
320
|
+
select_rows(mask, keep_masked: keep_masked)
|
|
321
|
+
end
|
|
322
|
+
|
|
323
|
+
# Move a column to the index and name the row axis after it (memo §3).
|
|
324
|
+
# An index-role change: the data is unchanged, so this mutates self and
|
|
325
|
+
# returns it (memo §3.8). Any existing index is replaced.
|
|
326
|
+
def set_index(name)
|
|
327
|
+
key = name.to_s
|
|
328
|
+
raise KeyError, "no column #{key.inspect}" unless @columns.key?(key)
|
|
329
|
+
idx = @columns[key]
|
|
330
|
+
unless idx.ndim == 1
|
|
331
|
+
raise ArgumentError, "index must be a 1-D column (got ndim #{idx.ndim})"
|
|
332
|
+
end
|
|
333
|
+
@columns.delete(key)
|
|
334
|
+
@index = idx
|
|
335
|
+
@axis_name = key
|
|
336
|
+
@nrow = idx.shape[0]
|
|
337
|
+
self
|
|
338
|
+
end
|
|
339
|
+
|
|
340
|
+
# Drop the index back into an ordinary column (the inverse of set_index).
|
|
341
|
+
# Also an index-role change, so it mutates self (memo §3.8). The former
|
|
342
|
+
# index becomes the first column, named after the row axis.
|
|
343
|
+
def reset_index
|
|
344
|
+
return self unless @index
|
|
345
|
+
@columns = { @axis_name => @index }.merge(@columns)
|
|
346
|
+
@index = nil
|
|
347
|
+
@axis_name = DEFAULT_AXIS_NAME
|
|
348
|
+
self
|
|
349
|
+
end
|
|
350
|
+
|
|
351
|
+
# The first +n+ rows as a positional view-frame (memo §11.3). +n+ larger
|
|
352
|
+
# than +nrow+ yields the whole frame; +n+ of 0 yields an empty frame.
|
|
353
|
+
def head(n = 5)
|
|
354
|
+
raise ArgumentError, "head count must be non-negative (got #{n})" if n < 0
|
|
355
|
+
row_span(0, [n, @nrow].min)
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
# The last +n+ rows as a positional view-frame (memo §11.3). +n+ larger
|
|
359
|
+
# than +nrow+ yields the whole frame; +n+ of 0 yields an empty frame.
|
|
360
|
+
def tail(n = 5)
|
|
361
|
+
raise ArgumentError, "tail count must be non-negative (got #{n})" if n < 0
|
|
362
|
+
row_span(@nrow - [n, @nrow].min, @nrow)
|
|
363
|
+
end
|
|
364
|
+
|
|
365
|
+
# The single row whose index label equals +label+, as a Ruby Hash (memo
|
|
366
|
+
# §13.2b). +label+ is matched exactly against the index (+index.eq+), so any
|
|
367
|
+
# orderable / object / datetime / categorical index works. The return type is
|
|
368
|
+
# a row Hash and stays that way: zero matches raise KeyError, and duplicate
|
|
369
|
+
# labels raise (go through +filter { |f| f.index.eq(label) }+ for the
|
|
370
|
+
# multi-row, frame-returning path). Positional access is +df[i]+.
|
|
371
|
+
def at(label)
|
|
372
|
+
raise ArgumentError, "at requires an index (set one with set_index)" unless @index
|
|
373
|
+
pos = @index.eq(label).where
|
|
374
|
+
case pos.elements
|
|
375
|
+
when 0
|
|
376
|
+
raise KeyError, "no row with index label #{label.inspect}"
|
|
377
|
+
when 1
|
|
378
|
+
row(pos[0])
|
|
379
|
+
else
|
|
380
|
+
raise ArgumentError,
|
|
381
|
+
"index label #{label.inspect} matches #{pos.elements} rows; " \
|
|
382
|
+
"use filter { |f| f.index.eq(label) } for duplicate labels"
|
|
383
|
+
end
|
|
384
|
+
end
|
|
385
|
+
|
|
386
|
+
# An independent frame — every column and the index materialized (§3.6).
|
|
387
|
+
def copy
|
|
388
|
+
cols = {}
|
|
389
|
+
@columns.each { |k, v| cols[k] = v.copy }
|
|
390
|
+
CAFrame.new(cols, axis_name: @axis_name, index: @index && @index.copy)
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
# +dup+ / +clone+ share the column data -- that is the CArray dup contract
|
|
394
|
+
# (§3.6) -- but must not share the columns Hash itself: Ruby's shallow copy
|
|
395
|
+
# hands over the same Hash object, so a membership edit (+append+ / +drop+ /
|
|
396
|
+
# a +cast+ rebind) on the copy would show up in the original, which is the
|
|
397
|
+
# one thing per-frame membership (§12-B) says never happens.
|
|
398
|
+
private def initialize_copy(other)
|
|
399
|
+
super
|
|
400
|
+
@columns = @columns.dup
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
# --- metadata readers (memo §13.2) ------------------------------------
|
|
404
|
+
# Each returns a fresh object; the live columns Hash is never exposed.
|
|
405
|
+
|
|
406
|
+
# Variable (column) names in column order. Returns +Array<String>+.
|
|
407
|
+
def variable_names
|
|
408
|
+
@columns.keys
|
|
409
|
+
end
|
|
410
|
+
|
|
411
|
+
# Variables (raw column CArrays) in column order. Returns
|
|
412
|
+
# +Array<CArray>+. Equivalent to +df[*variable_names]+ but built
|
|
413
|
+
# directly from the columns Hash.
|
|
414
|
+
def variables
|
|
415
|
+
@columns.values
|
|
416
|
+
end
|
|
417
|
+
|
|
418
|
+
# Number of variables (columns).
|
|
419
|
+
def nvar
|
|
420
|
+
@columns.size
|
|
421
|
+
end
|
|
422
|
+
|
|
423
|
+
# Number of rows (axis-0 length N).
|
|
424
|
+
attr_reader :nrow
|
|
425
|
+
|
|
426
|
+
# Row-axis name.
|
|
427
|
+
attr_reader :axis_name
|
|
428
|
+
|
|
429
|
+
# Index column (CArray) or nil.
|
|
430
|
+
attr_reader :index
|
|
431
|
+
|
|
432
|
+
# name => data_type, freshly derived.
|
|
433
|
+
def data_types
|
|
434
|
+
h = {}
|
|
435
|
+
@columns.each { |k, v| h[k] = v.data_type }
|
|
436
|
+
h
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
# @return [String]
|
|
440
|
+
def inspect
|
|
441
|
+
parts = @columns.map { |k, v| "#{k}:#{v.data_type}#{v.ndim > 1 ? v.shape[1..].inspect : ''}" }
|
|
442
|
+
idx = @index ? " index=#{@axis_name.inspect}" : ""
|
|
443
|
+
"#<CAFrame nrow=#{@nrow} vars=[#{parts.join(', ')}]#{idx}>"
|
|
444
|
+
end
|
|
445
|
+
|
|
446
|
+
private def rebuild(cols)
|
|
447
|
+
CAFrame.new(cols, axis_name: @axis_name, index: @index)
|
|
448
|
+
end
|
|
449
|
+
|
|
450
|
+
private def coerce_column(col)
|
|
451
|
+
if col.is_a?(CAFrame)
|
|
452
|
+
raise ArgumentError,
|
|
453
|
+
"a CAFrame cannot be a column (it answers to_ca as a 2-D matrix); " \
|
|
454
|
+
"escape a column with df[\"name\"] or df.to_ca first"
|
|
455
|
+
end
|
|
456
|
+
unless col.respond_to?(:to_ca)
|
|
457
|
+
raise ArgumentError, "column must be a CArray or Array (got #{col.class})"
|
|
458
|
+
end
|
|
459
|
+
ca = col.to_ca
|
|
460
|
+
unless ca.is_a?(CArray)
|
|
461
|
+
raise ArgumentError, "column did not convert to a CArray (got #{ca.class})"
|
|
462
|
+
end
|
|
463
|
+
ca
|
|
464
|
+
end
|
|
465
|
+
|
|
466
|
+
private def row(i)
|
|
467
|
+
i += @nrow if i < 0
|
|
468
|
+
unless i >= 0 && i < @nrow
|
|
469
|
+
raise IndexError, "row index #{i} out of range (nrow=#{@nrow})"
|
|
470
|
+
end
|
|
471
|
+
h = {}
|
|
472
|
+
h[@axis_name] = elem_at(@index, i) if @index
|
|
473
|
+
@columns.each { |name, col| h[name] = elem_at(col, i) }
|
|
474
|
+
h
|
|
475
|
+
end
|
|
476
|
+
|
|
477
|
+
private def elem_at(col, i)
|
|
478
|
+
col[i, *([nil] * (col.ndim - 1))]
|
|
479
|
+
end
|
|
480
|
+
|
|
481
|
+
private def row_span(lo, hi)
|
|
482
|
+
if hi <= lo
|
|
483
|
+
select_rows(CArray.int32(0))
|
|
484
|
+
else
|
|
485
|
+
select_rows(lo...hi)
|
|
486
|
+
end
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
# df[sel] = UNDEF : mask the selected rows in place (write-through). Shape
|
|
490
|
+
# and index are untouched; the selected cells of every column go to UNDEF.
|
|
491
|
+
# The selector is forwarded to the column indexer, which classifies it.
|
|
492
|
+
private def mask_rows(selector)
|
|
493
|
+
@columns.each_value do |col|
|
|
494
|
+
col[selector, *([nil] * (col.ndim - 1))] = UNDEF
|
|
495
|
+
end
|
|
496
|
+
self
|
|
497
|
+
end
|
|
498
|
+
|
|
499
|
+
# df[sel] = nil : drop the selected rows. The surviving rows are gathered by
|
|
500
|
+
# the complement of the selection (order preserved), and the frame shrinks.
|
|
501
|
+
private def delete_rows(selector)
|
|
502
|
+
keep = selected_row_mask(selector).not
|
|
503
|
+
new_cols = {}
|
|
504
|
+
@columns.each do |name, col|
|
|
505
|
+
new_cols[name] = col[keep, *([nil] * (col.ndim - 1))]
|
|
506
|
+
end
|
|
507
|
+
@columns = new_cols
|
|
508
|
+
@index = @index[keep] if @index
|
|
509
|
+
@nrow = keep.count(true)
|
|
510
|
+
self
|
|
511
|
+
end
|
|
512
|
+
|
|
513
|
+
# df[sel] = other : replace the selected contiguous span with other's rows
|
|
514
|
+
# (any length). Columns are concatenated head + other + tail, so dtypes
|
|
515
|
+
# promote and the row count shifts by other.nrow - span.
|
|
516
|
+
private def splice_rows(selector, other)
|
|
517
|
+
lo, hi = contiguous_span(selector)
|
|
518
|
+
unless other.variable_names.sort == variable_names.sort
|
|
519
|
+
raise ArgumentError,
|
|
520
|
+
"splice frame has columns #{other.variable_names.inspect}, " \
|
|
521
|
+
"expected the same set as #{variable_names.inspect}"
|
|
522
|
+
end
|
|
523
|
+
new_cols = {}
|
|
524
|
+
@columns.each do |name, col|
|
|
525
|
+
tail = [nil] * (col.ndim - 1)
|
|
526
|
+
pieces = []
|
|
527
|
+
pieces << col[0...lo, *tail] if lo > 0
|
|
528
|
+
# Snapshot the RHS: df1[...] = df2 is a value assignment, so subsequent
|
|
529
|
+
# writes to df1's spliced rows must not leak into df2 via the CAMeld
|
|
530
|
+
# chain. Head / tail slices come from df1 itself and keep the intended
|
|
531
|
+
# chain composability (writes to df1["c"][k] still reach df1's own
|
|
532
|
+
# storage there — external aliases into df1's old column see them).
|
|
533
|
+
pieces << other[name].copy if other.nrow > 0
|
|
534
|
+
pieces << col[hi...@nrow, *tail] if hi < @nrow
|
|
535
|
+
new_cols[name] =
|
|
536
|
+
case pieces.size
|
|
537
|
+
when 0 then col[CArray.int32(0), *tail] # replaced every row with none
|
|
538
|
+
when 1 then pieces.first
|
|
539
|
+
else CArray.meld(pieces, axis: 0) # CAMeld view; dtype mismatch
|
|
540
|
+
end # across pieces raises.
|
|
541
|
+
# For dtype conversion, cast
|
|
542
|
+
# the incoming +other+'s
|
|
543
|
+
# column beforehand — silent
|
|
544
|
+
# promotion in a splice would
|
|
545
|
+
# hide schema drift.
|
|
546
|
+
end
|
|
547
|
+
new_index = splice_index(other, lo, hi)
|
|
548
|
+
@columns = new_cols
|
|
549
|
+
@index = new_index
|
|
550
|
+
@nrow = @nrow - (hi - lo) + other.nrow
|
|
551
|
+
self
|
|
552
|
+
end
|
|
553
|
+
|
|
554
|
+
# The index for a splice: head + other's index + tail. When this frame has
|
|
555
|
+
# an index, the spliced-in frame must have one too (its rows need labels).
|
|
556
|
+
private def splice_index(other, lo, hi)
|
|
557
|
+
return nil unless @index
|
|
558
|
+
if other.nrow > 0 && other.index.nil?
|
|
559
|
+
raise ArgumentError,
|
|
560
|
+
"splice frame needs an index because the target frame has one"
|
|
561
|
+
end
|
|
562
|
+
pieces = []
|
|
563
|
+
pieces << @index[0...lo] if lo > 0
|
|
564
|
+
pieces << other.index.copy if other.nrow > 0 # snapshot RHS (see splice_rows)
|
|
565
|
+
pieces << @index[hi...@nrow] if hi < @nrow
|
|
566
|
+
case pieces.size
|
|
567
|
+
when 0 then @index[CArray.int32(0)]
|
|
568
|
+
when 1 then pieces.first
|
|
569
|
+
else CArray.meld(pieces, axis: 0) # index dtype must match across frames
|
|
570
|
+
end
|
|
571
|
+
end
|
|
572
|
+
|
|
573
|
+
# A boolean mask over the rows, true where +selector+ selects. Built by
|
|
574
|
+
# marking a false vector through the same selector the reader accepts, so
|
|
575
|
+
# Range / Integer / boolean / integer-array all normalize the same way.
|
|
576
|
+
private def selected_row_mask(selector)
|
|
577
|
+
m = CArray.boolean(@nrow) { 0 }
|
|
578
|
+
m[selector] = 1
|
|
579
|
+
m
|
|
580
|
+
end
|
|
581
|
+
|
|
582
|
+
# Resolve a contiguous span [lo, hi) for splice, reusing the row-axis indexer
|
|
583
|
+
# classifier (CArray.scan_index) so negatives, exclusive ends,
|
|
584
|
+
# ArithmeticSequence and the [start, count(, step)] array forms parse exactly
|
|
585
|
+
# as they do for a column key. Only a POINT (single row) or a step-1 BLOCK
|
|
586
|
+
# (contiguous slice) has a span to replace; a strided or scattered selector
|
|
587
|
+
# raises.
|
|
588
|
+
#
|
|
589
|
+
# The one place splice diverges from element indexing is the tail insertion
|
|
590
|
+
# point: scan_index bound-checks the start to 0..nrow-1, but splice — like
|
|
591
|
+
# Ruby Array#[]= — also accepts nrow as an empty span to append to, written
|
|
592
|
+
# +nrow...nrow+.
|
|
593
|
+
private def contiguous_span(selector)
|
|
594
|
+
if (selector.is_a?(Range) || selector.is_a?(Enumerator::ArithmeticSequence)) &&
|
|
595
|
+
selector.exclude_end? && selector.begin == @nrow && selector.end == @nrow
|
|
596
|
+
return [@nrow, @nrow]
|
|
597
|
+
end
|
|
598
|
+
|
|
599
|
+
info = CArray.scan_index([@nrow], [selector])
|
|
600
|
+
case info.type
|
|
601
|
+
when CA_REG_POINT
|
|
602
|
+
i = info.index[0]
|
|
603
|
+
[i, i + 1]
|
|
604
|
+
when CA_REG_BLOCK
|
|
605
|
+
start, count, step = info.index[0]
|
|
606
|
+
unless step == 1
|
|
607
|
+
raise ArgumentError,
|
|
608
|
+
"splice needs a contiguous slice; a strided step #{step} has no span"
|
|
609
|
+
end
|
|
610
|
+
[start, start + count]
|
|
611
|
+
else
|
|
612
|
+
raise ArgumentError,
|
|
613
|
+
"splice (df[...] = frame) needs a contiguous slice (Range, Integer, " \
|
|
614
|
+
"or [start, count]); got #{selector.class}"
|
|
615
|
+
end
|
|
616
|
+
end
|
|
617
|
+
|
|
618
|
+
private def select_rows(selector, keep_masked: false)
|
|
619
|
+
if keep_masked && selector.is_a?(CArray) &&
|
|
620
|
+
selector.data_type == :boolean && selector.has_mask?
|
|
621
|
+
return select_rows_keep_masked(selector)
|
|
622
|
+
end
|
|
623
|
+
cols = {}
|
|
624
|
+
@columns.each do |name, col|
|
|
625
|
+
cols[name] = col[selector, *([nil] * (col.ndim - 1))]
|
|
626
|
+
end
|
|
627
|
+
new_index = @index ? @index[selector] : nil
|
|
628
|
+
CAFrame.new(cols, axis_name: @axis_name, index: new_index)
|
|
629
|
+
end
|
|
630
|
+
|
|
631
|
+
private def select_rows_keep_masked(selector)
|
|
632
|
+
keep = selector.strip_mask(1) # true where selected OR undetermined
|
|
633
|
+
undet_kept = selector.is_masked[keep] # among kept rows, which were undetermined
|
|
634
|
+
cols = {}
|
|
635
|
+
@columns.each do |name, col|
|
|
636
|
+
tail = [nil] * (col.ndim - 1)
|
|
637
|
+
g = col[keep, *tail].copy
|
|
638
|
+
g[undet_kept, *tail] = UNDEF
|
|
639
|
+
cols[name] = g
|
|
640
|
+
end
|
|
641
|
+
new_index = @index ? @index[keep] : nil
|
|
642
|
+
CAFrame.new(cols, axis_name: @axis_name, index: new_index)
|
|
643
|
+
end
|
|
644
|
+
|
|
645
|
+
private def positional_range?(range)
|
|
646
|
+
(range.begin.nil? || range.begin.is_a?(Integer)) &&
|
|
647
|
+
(range.end.nil? || range.end.is_a?(Integer))
|
|
648
|
+
end
|
|
649
|
+
end
|