hypertabular-wasm 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +179 -0
- data/ext/hypertabular_native/3.4/hypertabular_native.a.gz +0 -0
- data/ext/hypertabular_native/4.0/hypertabular_native.a.gz +0 -0
- data/ext/hypertabular_native/extconf.rb +108 -0
- data/lib/hypertabular/batch.rb +204 -0
- data/lib/hypertabular/column.rb +170 -0
- data/lib/hypertabular/delimited_reader.rb +309 -0
- data/lib/hypertabular/dialect.rb +42 -0
- data/lib/hypertabular/runtime/columns.rb +46 -0
- data/lib/hypertabular/runtime/delimited.rb +204 -0
- data/lib/hypertabular/runtime/workbook.rb +299 -0
- data/lib/hypertabular/runtime.rb +142 -0
- data/lib/hypertabular/tabular_error.rb +99 -0
- data/lib/hypertabular/workbook.rb +173 -0
- data/lib/hypertabular.rb +103 -0
- metadata +107 -0
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
module HyperTabular
|
|
2
|
+
# The doors a column can be cast through, by the code the core knows each as: HyperCast's
|
|
3
|
+
# twenty-one, named as its gem names them, plus :text for the bytes themselves.
|
|
4
|
+
DOORS = {
|
|
5
|
+
bool: 1, i8: 2, i16: 3, i32: 4, i64: 5, u8: 6, u16: 7, u32: 8, u64: 9, f32: 10, f64: 11,
|
|
6
|
+
uuid: 12, timestamp: 13, unix: 14, date: 15, time: 16, duration: 17, text: 18, decimal: 19,
|
|
7
|
+
date_ordered: 20, datetime: 21, excel_serial: 22
|
|
8
|
+
}.freeze
|
|
9
|
+
|
|
10
|
+
# One output column of a plan: which source column it reads, the door it casts through,
|
|
11
|
+
# what that door declares, and — for the numeric doors — the notation. A plan is a
|
|
12
|
+
# projection: a forty-column file can be read into five typed columns, in any order, and
|
|
13
|
+
# a source column can be read through more than one door.
|
|
14
|
+
#
|
|
15
|
+
# Built by the factory named after its door, never guessed from the data:
|
|
16
|
+
#
|
|
17
|
+
# plan = [
|
|
18
|
+
# HyperTabular::Column.i32(0),
|
|
19
|
+
# HyperTabular::Column.text(1),
|
|
20
|
+
# HyperTabular::Column.decimal(2, eurozone),
|
|
21
|
+
# HyperTabular::Column.date(3, :day_month_year),
|
|
22
|
+
# HyperTabular::Column.unix(4, :milliseconds)
|
|
23
|
+
# ]
|
|
24
|
+
#
|
|
25
|
+
# +ordinal+ is the zero-based source column; past a record's last cell it reads as empty.
|
|
26
|
+
# +door+ is a key of DOORS. +declared+ is the Symbol the door was declared with — a Unix
|
|
27
|
+
# precision, a date order, an Excel epoch, as HyperCast names them — or nil. +format+ is
|
|
28
|
+
# the HyperCast::NumFormat of a numeric door, or nil.
|
|
29
|
+
Column = Data.define(:ordinal, :door, :declared, :format) do
|
|
30
|
+
# Checks the column as the caller bug it would otherwise become at the first read: an
|
|
31
|
+
# ordinal that is not a non-negative Integer or a format that is not a
|
|
32
|
+
# HyperCast::NumFormat (ArgumentError), an unknown door or an undeclared option
|
|
33
|
+
# (KeyError, as HyperCast raises for the same).
|
|
34
|
+
def initialize(ordinal:, door:, declared: nil, format: nil)
|
|
35
|
+
raise ArgumentError, "ordinal must be an Integer in 0..#{Column::MAX_ORDINAL}; got #{ordinal.inspect}" unless
|
|
36
|
+
ordinal.is_a?(Integer) && ordinal.between?(0, Column::MAX_ORDINAL)
|
|
37
|
+
|
|
38
|
+
DOORS.fetch(door)
|
|
39
|
+
if (options = Column::DECLARES[door])
|
|
40
|
+
options.fetch(declared)
|
|
41
|
+
elsif !declared.nil?
|
|
42
|
+
raise ArgumentError, "the #{door} door declares nothing; got #{declared.inspect}"
|
|
43
|
+
end
|
|
44
|
+
if Column::NUMERIC.include?(door)
|
|
45
|
+
format = HyperCast::NumFormat::INVARIANT if format.nil?
|
|
46
|
+
raise ArgumentError, "format must be a HyperCast::NumFormat; got #{format.inspect}" unless
|
|
47
|
+
format.is_a?(HyperCast::NumFormat)
|
|
48
|
+
elsif !format.nil?
|
|
49
|
+
raise ArgumentError, "the #{door} door reads no numeric format"
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
super(ordinal: ordinal, door: door, declared: declared, format: format)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# A boolean column: HyperCast's boolean lexicon, as true or false.
|
|
56
|
+
def self.bool(ordinal) = new(ordinal: ordinal, door: :bool)
|
|
57
|
+
|
|
58
|
+
# A signed 8-bit column, as an Integer. Invariant notation unless one is declared.
|
|
59
|
+
def self.i8(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i8, format: format)
|
|
60
|
+
|
|
61
|
+
# A signed 16-bit column, as an Integer. Invariant notation unless one is declared.
|
|
62
|
+
def self.i16(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i16, format: format)
|
|
63
|
+
|
|
64
|
+
# A signed 32-bit column, as an Integer. Invariant notation unless one is declared.
|
|
65
|
+
def self.i32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i32, format: format)
|
|
66
|
+
|
|
67
|
+
# A signed 64-bit column, as an Integer. Invariant notation unless one is declared.
|
|
68
|
+
def self.i64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i64, format: format)
|
|
69
|
+
|
|
70
|
+
# An unsigned 8-bit column, as an Integer. Invariant notation unless one is declared.
|
|
71
|
+
def self.u8(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u8, format: format)
|
|
72
|
+
|
|
73
|
+
# An unsigned 16-bit column, as an Integer. Invariant notation unless one is declared.
|
|
74
|
+
def self.u16(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u16, format: format)
|
|
75
|
+
|
|
76
|
+
# An unsigned 32-bit column, as an Integer. Invariant notation unless one is declared.
|
|
77
|
+
def self.u32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u32, format: format)
|
|
78
|
+
|
|
79
|
+
# An unsigned 64-bit column, as the true unsigned Integer. Invariant notation unless
|
|
80
|
+
# one is declared.
|
|
81
|
+
def self.u64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u64, format: format)
|
|
82
|
+
|
|
83
|
+
# An IEEE single column, widened losslessly to a Float. Invariant notation unless one
|
|
84
|
+
# is declared.
|
|
85
|
+
def self.f32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :f32, format: format)
|
|
86
|
+
|
|
87
|
+
# An IEEE double column, as a Float. Invariant notation unless one is declared.
|
|
88
|
+
def self.f64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :f64, format: format)
|
|
89
|
+
|
|
90
|
+
# An exact decimal column, as a HyperCast::Decimal; no float is ever formed. Invariant
|
|
91
|
+
# notation unless one is declared.
|
|
92
|
+
def self.decimal(ordinal, format = HyperCast::NumFormat::INVARIANT)
|
|
93
|
+
new(ordinal: ordinal, door: :decimal, format: format)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# A UUID column, as the lowercase hyphenated String.
|
|
97
|
+
def self.uuid(ordinal) = new(ordinal: ordinal, door: :uuid)
|
|
98
|
+
|
|
99
|
+
# An RFC 3339 instant column, as a UTC Time at full nanosecond fidelity.
|
|
100
|
+
def self.timestamp(ordinal) = new(ordinal: ordinal, door: :timestamp)
|
|
101
|
+
|
|
102
|
+
# A Unix-epoch column at the declared precision (:seconds, :milliseconds,
|
|
103
|
+
# :microseconds, :nanoseconds) — never guessed from magnitude — as a UTC Time.
|
|
104
|
+
def self.unix(ordinal, precision) = new(ordinal: ordinal, door: :unix, declared: precision)
|
|
105
|
+
|
|
106
|
+
# An Excel date-serial column under the declared date system (:y1900, :y1904), as a
|
|
107
|
+
# UTC Time.
|
|
108
|
+
def self.excel_serial(ordinal, epoch) = new(ordinal: ordinal, door: :excel_serial, declared: epoch)
|
|
109
|
+
|
|
110
|
+
# A calendar date column, as a Date. With no order declared: the strict ISO 8601
|
|
111
|
+
# yyyy-MM-dd door. With one (:year_month_day, :month_day_year, :day_month_year): the
|
|
112
|
+
# date_ordered door, which also reads the separated forms — the same split
|
|
113
|
+
# HyperCast.date makes.
|
|
114
|
+
def self.date(ordinal, order = nil)
|
|
115
|
+
order.nil? ? new(ordinal: ordinal, door: :date) : date_ordered(ordinal, order)
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# A separated calendar date column under the declared field order, as a Date.
|
|
119
|
+
def self.date_ordered(ordinal, order) = new(ordinal: ordinal, door: :date_ordered, declared: order)
|
|
120
|
+
|
|
121
|
+
# A zone-less civil date-time column under the declared field order, as a DateTime
|
|
122
|
+
# whose +00:00 offset is a carrier artifact, not data — the text named no zone.
|
|
123
|
+
def self.datetime(ordinal, order) = new(ordinal: ordinal, door: :datetime, declared: order)
|
|
124
|
+
|
|
125
|
+
# A 24-hour time-of-day column, as an exact Integer of nanoseconds since midnight.
|
|
126
|
+
def self.time(ordinal) = new(ordinal: ordinal, door: :time)
|
|
127
|
+
|
|
128
|
+
# A duration column, as exact Rational seconds.
|
|
129
|
+
def self.duration(ordinal) = new(ordinal: ordinal, door: :duration)
|
|
130
|
+
|
|
131
|
+
# A text column: the cell's bytes themselves, untrimmed, quotes resolved, as a UTF-8
|
|
132
|
+
# String. A cell with no bytes at all is the one way text fails (an :empty Fault).
|
|
133
|
+
def self.text(ordinal) = new(ordinal: ordinal, door: :text)
|
|
134
|
+
|
|
135
|
+
# Bytes one value of this column's door takes in a column buffer.
|
|
136
|
+
def value_bytes
|
|
137
|
+
Column::VALUE_BYTES.fetch(door)
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# The 44 little-endian bytes of the core's ColumnSpec: ordinal, door code, what the
|
|
141
|
+
# door declares (numbered as HyperCast numbers it), then the notation in HyperCast's
|
|
142
|
+
# own packed form — all zeros, which the core reads as invariant, for a door that
|
|
143
|
+
# reads none.
|
|
144
|
+
def packed
|
|
145
|
+
param = declared.nil? ? 0 : Column::DECLARES.fetch(door).fetch(declared)
|
|
146
|
+
[ordinal, DOORS.fetch(door), param].pack("L<3") + (format.nil? ? Column::NO_FORMAT : format.packed)
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# The widest source ordinal a column can name.
|
|
151
|
+
Column::MAX_ORDINAL = (1 << 31) - 1
|
|
152
|
+
|
|
153
|
+
# The doors that read a HyperCast::NumFormat: the integers, the reals and the decimal.
|
|
154
|
+
Column::NUMERIC = %i[i8 i16 i32 i64 u8 u16 u32 u64 f32 f64 decimal].freeze
|
|
155
|
+
|
|
156
|
+
# What the doors that declare something declare, as HyperCast's own tables number it.
|
|
157
|
+
Column::DECLARES = {
|
|
158
|
+
unix: HyperCast::UNIX_PRECISIONS,
|
|
159
|
+
excel_serial: HyperCast::EXCEL_EPOCHS,
|
|
160
|
+
date_ordered: HyperCast::DATE_ORDERS,
|
|
161
|
+
datetime: HyperCast::DATE_ORDERS
|
|
162
|
+
}.freeze
|
|
163
|
+
|
|
164
|
+
# Bytes per value in a column buffer, by door (rust/src/kernel/abi.rs, ColumnBuffer):
|
|
165
|
+
# HyperCast's for its doors, and a text cell's span — two u32s — for this one's.
|
|
166
|
+
Column::VALUE_BYTES = HyperCast::Interop::VALUE_BYTES.merge(text: 8).freeze
|
|
167
|
+
|
|
168
|
+
# The notation of a door that reads none: thirty-two zero bytes.
|
|
169
|
+
Column::NO_FORMAT = ("\0" * 32).b.freeze
|
|
170
|
+
end
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
module HyperTabular
|
|
2
|
+
# Delimited text — CSV, TSV, any single-byte ASCII separator — read a batch at a time
|
|
3
|
+
# into typed columns, every cell a HyperCast verdict.
|
|
4
|
+
#
|
|
5
|
+
# plan = [HyperTabular::Column.i32(0), HyperTabular::Column.text(1), HyperTabular::Column.f64(2)]
|
|
6
|
+
# HyperTabular::DelimitedReader.open("orders.csv", HyperTabular::Dialect::CSV, plan) do |reader|
|
|
7
|
+
# while (batch = reader.read)
|
|
8
|
+
# ids = batch.values(0) # a column at a time: [1, 2, nil, 4, ...]
|
|
9
|
+
# batch.rows.times do |row|
|
|
10
|
+
# case batch.get(2, row) # or a cell at a time, as HyperCast's union
|
|
11
|
+
# in HyperCast::Success(value:) then total += value
|
|
12
|
+
# in HyperCast::Fault(reason:) then warn "#{reason}: #{batch.raw(2, row).inspect}"
|
|
13
|
+
# end
|
|
14
|
+
# end
|
|
15
|
+
# end
|
|
16
|
+
# end
|
|
17
|
+
#
|
|
18
|
+
# The native core (libhypertabular) owns no memory and reads no files. It is handed a
|
|
19
|
+
# chunk of input and the buffers to fill, casts each plan column in one native loop, and
|
|
20
|
+
# says how many rows it wrote and how many bytes it is finished with. Everything else is
|
|
21
|
+
# here: this class holds the input, has one value array and one verdict array per column
|
|
22
|
+
# allocated once and reused for every batch, and puts what the core did not consume back
|
|
23
|
+
# in front of it. The native boundary is crossed once per batch, not once per cell, and a
|
|
24
|
+
# column comes out of its buffer in one String#unpack.
|
|
25
|
+
#
|
|
26
|
+
# A value that does not cast is that cell's verdict, and the read goes on. Input that is
|
|
27
|
+
# not rows of cells at all — a record of the wrong width, a quote never closed — is a
|
|
28
|
+
# TabularError, raised after every intact row before it has been delivered.
|
|
29
|
+
#
|
|
30
|
+
# Each #read returns a Batch, which owns what it shows: it stays good after the next
|
|
31
|
+
# #read. Not thread-safe.
|
|
32
|
+
class DelimitedReader
|
|
33
|
+
# The row ceiling: a single record larger than this is a :row_too_long TabularError.
|
|
34
|
+
MAX_ROW_BYTES = 1 << 30
|
|
35
|
+
|
|
36
|
+
# How much is asked of an IO at a time, until a record does not fit.
|
|
37
|
+
DEFAULT_BUFFER_BYTES = 256 * 1024
|
|
38
|
+
|
|
39
|
+
# Rows per batch unless told otherwise.
|
|
40
|
+
DEFAULT_BATCH_ROWS = 4096
|
|
41
|
+
|
|
42
|
+
# Encodings whose bytes already are the UTF-8 (or byte-identical) form the core reads.
|
|
43
|
+
BYTE_COMPATIBLE = HyperCast::Interop::BYTE_COMPATIBLE
|
|
44
|
+
|
|
45
|
+
# Opens a file of UTF-8 delimited text. With a block, yields the reader, closes it —
|
|
46
|
+
# and the file — when the block ends, and returns the block's value; without one,
|
|
47
|
+
# returns the reader, whose #close closes the file.
|
|
48
|
+
def self.open(path, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES)
|
|
49
|
+
file = File.open(path, "rb")
|
|
50
|
+
begin
|
|
51
|
+
reader = new(file, dialect, plan, batch_rows: batch_rows, buffer_bytes: buffer_bytes, close_source: true)
|
|
52
|
+
ensure
|
|
53
|
+
# The file is ours to close when no reader came of it, whatever went wrong.
|
|
54
|
+
file.close if reader.nil?
|
|
55
|
+
end
|
|
56
|
+
return reader unless block_given?
|
|
57
|
+
|
|
58
|
+
begin
|
|
59
|
+
yield reader
|
|
60
|
+
ensure
|
|
61
|
+
reader.close
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# Reads UTF-8 delimited text from +source+: a String, read in place — nothing is copied
|
|
66
|
+
# — or an IO (anything with #readpartial or #read), read forward only through a buffer
|
|
67
|
+
# of +buffer_bytes+ that grows when a record does not fit it. An IO is read as bytes;
|
|
68
|
+
# one opened in text mode on Windows should be in binmode first.
|
|
69
|
+
#
|
|
70
|
+
# +dialect+ is the declared Dialect and +plan+ the output Columns, in output order.
|
|
71
|
+
# +batch_rows+ is the most rows one #read delivers. +close_source+ says whether #close
|
|
72
|
+
# also closes the IO.
|
|
73
|
+
#
|
|
74
|
+
# A dialect the scanner cannot honour, or a plan that is not Columns, is an
|
|
75
|
+
# ArgumentError. With a header declared the header record is read here, so a header
|
|
76
|
+
# that is structurally broken is a TabularError here.
|
|
77
|
+
def initialize(source, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES,
|
|
78
|
+
close_source: false)
|
|
79
|
+
raise ArgumentError, "dialect must be a HyperTabular::Dialect; got #{dialect.inspect}" unless
|
|
80
|
+
dialect.is_a?(Dialect)
|
|
81
|
+
raise ArgumentError, "batch_rows must be a positive Integer; got #{batch_rows.inspect}" unless
|
|
82
|
+
batch_rows.is_a?(Integer) && batch_rows.positive?
|
|
83
|
+
raise ArgumentError, "buffer_bytes must be a positive Integer; got #{buffer_bytes.inspect}" unless
|
|
84
|
+
buffer_bytes.is_a?(Integer) && buffer_bytes.positive?
|
|
85
|
+
|
|
86
|
+
@plan = Array(plan).dup.freeze
|
|
87
|
+
@plan.each_with_index do |column, index|
|
|
88
|
+
raise ArgumentError, "plan column #{index} must be a HyperTabular::Column; got #{column.inspect}" unless
|
|
89
|
+
column.is_a?(Column)
|
|
90
|
+
end
|
|
91
|
+
@dialect = dialect
|
|
92
|
+
@batch_rows = batch_rows
|
|
93
|
+
@buffer_bytes = buffer_bytes
|
|
94
|
+
@close_source = close_source
|
|
95
|
+
# Cell-table entries one row takes: the widest ordinal the plan reads, plus two.
|
|
96
|
+
per_row = (@plan.map(&:ordinal).max || -1) + 2
|
|
97
|
+
@kernel = Runtime::Delimited.start(dialect.packed, @plan.map(&:packed), @plan.map(&:value_bytes),
|
|
98
|
+
batch_rows, per_row)
|
|
99
|
+
raise ArgumentError, "separator #{dialect.separator.inspect} is not tab or printable ASCII other than '\"'" if
|
|
100
|
+
@kernel.nil?
|
|
101
|
+
|
|
102
|
+
@start = 0
|
|
103
|
+
@closed = false
|
|
104
|
+
@failure = nil
|
|
105
|
+
take(source)
|
|
106
|
+
@header = dialect.has_header ? read_header : nil
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# The header's names, when the dialect declares a header: frozen UTF-8 Strings, quotes
|
|
110
|
+
# resolved. Empty for an input with no record at all; nil when the dialect declares no
|
|
111
|
+
# header.
|
|
112
|
+
attr_reader :header
|
|
113
|
+
|
|
114
|
+
# The plan: the output Columns, in output order. Column +i+ of every batch is +plan[i]+.
|
|
115
|
+
attr_reader :plan
|
|
116
|
+
|
|
117
|
+
# The declared Dialect.
|
|
118
|
+
attr_reader :dialect
|
|
119
|
+
|
|
120
|
+
# The most rows one #read delivers.
|
|
121
|
+
attr_reader :batch_rows
|
|
122
|
+
|
|
123
|
+
# Records finished so far — the header and skipped blank lines included.
|
|
124
|
+
def records
|
|
125
|
+
@kernel.records
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
# Reads the next batch: a Batch of up to #batch_rows rows, or nil once the input is
|
|
129
|
+
# exhausted. A TabularError when the input is structurally broken — raised after every
|
|
130
|
+
# intact row before the break has been delivered, and the same error again on every
|
|
131
|
+
# later call. IOError on a closed reader.
|
|
132
|
+
def read
|
|
133
|
+
raise IOError, "closed reader" if @closed
|
|
134
|
+
raise @failure if @failure
|
|
135
|
+
|
|
136
|
+
loop do
|
|
137
|
+
length, last = window
|
|
138
|
+
raise structural unless @kernel.fill(@start, length, last) == Runtime::Delimited::OK
|
|
139
|
+
|
|
140
|
+
consumed = @kernel.consumed
|
|
141
|
+
if @kernel.rows.positive?
|
|
142
|
+
batch = made(@kernel.rows)
|
|
143
|
+
@start += consumed
|
|
144
|
+
return batch
|
|
145
|
+
end
|
|
146
|
+
@start += consumed
|
|
147
|
+
return nil if last && (consumed == length || consumed.zero?)
|
|
148
|
+
|
|
149
|
+
refill if consumed.zero?
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# Every remaining row, batch after batch: yields one frozen Array per row, a verdict
|
|
154
|
+
# per plan column. Without a block, an Enumerator. Raises what #read raises.
|
|
155
|
+
def each_row
|
|
156
|
+
return to_enum(:each_row) unless block_given?
|
|
157
|
+
|
|
158
|
+
while (batch = read)
|
|
159
|
+
columns = Array.new(@plan.size) { |column| batch.verdicts(column) }
|
|
160
|
+
batch.rows.times { |row| yield columns.map { |column| column[row] }.freeze }
|
|
161
|
+
end
|
|
162
|
+
self
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# Closes the IO if this reader was told to (+close_source+, or DelimitedReader.open).
|
|
166
|
+
# Batches already read stay good. Safe to call twice.
|
|
167
|
+
def close
|
|
168
|
+
return if @closed
|
|
169
|
+
|
|
170
|
+
@closed = true
|
|
171
|
+
@io.close if @close_source && @io
|
|
172
|
+
nil
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
# True once #close has been called.
|
|
176
|
+
def closed?
|
|
177
|
+
@closed
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# The reader in a line — not its buffers.
|
|
181
|
+
def inspect
|
|
182
|
+
"#<#{self.class.name} columns=#{@plan.size} records=#{records}#{' closed' if @closed}>"
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
private
|
|
186
|
+
|
|
187
|
+
# Takes the source: a String becomes the whole input, an IO the thing to refill from.
|
|
188
|
+
def take(source)
|
|
189
|
+
if source.is_a?(String)
|
|
190
|
+
@buffer = utf8(source)
|
|
191
|
+
@eof = true
|
|
192
|
+
elsif defined?(::Pathname) && source.is_a?(::Pathname)
|
|
193
|
+
# It has a #read, and one that starts the file over on every call.
|
|
194
|
+
raise ArgumentError, "source is a Pathname; DelimitedReader.open reads a path"
|
|
195
|
+
elsif source.respond_to?(:readpartial) || source.respond_to?(:read)
|
|
196
|
+
@io = source
|
|
197
|
+
@partial = source.respond_to?(:readpartial)
|
|
198
|
+
@buffer = String.new(encoding: Encoding::UTF_8)
|
|
199
|
+
@eof = false
|
|
200
|
+
else
|
|
201
|
+
raise ArgumentError, "source must be a String or an IO; got #{source.class}"
|
|
202
|
+
end
|
|
203
|
+
@kernel.attach(@buffer)
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
# The text as a frozen String of UTF-8 bytes that is this reader's alone: the caller's
|
|
207
|
+
# own when it is already that; otherwise a copy-on-write duplicate, which shares the
|
|
208
|
+
# bytes and is immune to what the caller does to theirs next. Only a foreign encoding
|
|
209
|
+
# pays a transcode.
|
|
210
|
+
def utf8(text)
|
|
211
|
+
return text if text.frozen? && text.encoding == Encoding::UTF_8
|
|
212
|
+
return text.dup.force_encoding(Encoding::UTF_8).freeze if BYTE_COMPATIBLE.include?(text.encoding)
|
|
213
|
+
|
|
214
|
+
text.encode(Encoding::UTF_8).freeze
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
# What the core is to read next — how many bytes from @start — and whether nothing
|
|
218
|
+
# follows them.
|
|
219
|
+
def window
|
|
220
|
+
length = @buffer.bytesize - @start
|
|
221
|
+
length > MAX_ROW_BYTES ? [MAX_ROW_BYTES, false] : [length, @eof]
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
# Puts the unfinished record at the front of the buffer and reads more behind it —
|
|
225
|
+
# at least as much again as is pending, so a record that does not fit doubles the
|
|
226
|
+
# buffer rather than creeping up on it.
|
|
227
|
+
def refill
|
|
228
|
+
pending = @buffer.bytesize - @start
|
|
229
|
+
raise too_long if @io.nil? || pending >= MAX_ROW_BYTES
|
|
230
|
+
|
|
231
|
+
chunk = more([@buffer_bytes, pending].max)
|
|
232
|
+
if chunk.nil? || chunk.empty?
|
|
233
|
+
@eof = true
|
|
234
|
+
return
|
|
235
|
+
end
|
|
236
|
+
chunk = chunk.dup if chunk.frozen?
|
|
237
|
+
@buffer = @buffer.byteslice(@start, pending) if @start.positive?
|
|
238
|
+
@buffer << chunk.force_encoding(Encoding::UTF_8)
|
|
239
|
+
@start = 0
|
|
240
|
+
@kernel.attach(@buffer)
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
# Up to +bytes+ more of the IO, or nil at its end. #readpartial where there is one, so
|
|
244
|
+
# a pipe or a socket hands over what it has rather than waiting to fill the request.
|
|
245
|
+
def more(bytes)
|
|
246
|
+
@partial ? @io.readpartial(bytes) : @io.read(bytes)
|
|
247
|
+
rescue EOFError
|
|
248
|
+
nil
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
# The batch the core just wrote, copied out of the kernel's arrays: its spans index the
|
|
252
|
+
# input from @start, where the core was handed it.
|
|
253
|
+
def made(rows)
|
|
254
|
+
columns = @kernel.columns
|
|
255
|
+
Batch.new(@plan, rows, Array.new(@plan.size) { |column| columns.values(column, rows) },
|
|
256
|
+
Array.new(@plan.size) { |column| columns.verdicts(column, rows) },
|
|
257
|
+
@kernel.cells, @kernel.per_row, @buffer, @start, @kernel.arena, workbook: false)
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
# Reads the header record: its names, or none for an input with no record at all.
|
|
261
|
+
def read_header
|
|
262
|
+
loop do
|
|
263
|
+
length, last = window
|
|
264
|
+
raise structural unless @kernel.header(@start, length, last) == Runtime::Delimited::OK
|
|
265
|
+
|
|
266
|
+
consumed = @kernel.consumed
|
|
267
|
+
if @kernel.rows.positive?
|
|
268
|
+
names = header_names
|
|
269
|
+
@start += consumed
|
|
270
|
+
return names
|
|
271
|
+
end
|
|
272
|
+
@start += consumed
|
|
273
|
+
# An empty input has no header and no rows; the width is unknown.
|
|
274
|
+
return [].freeze if last && (consumed == length || consumed.zero?)
|
|
275
|
+
|
|
276
|
+
refill if consumed.zero?
|
|
277
|
+
end
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
# The names the core just located: in the input as written, or — flagged — unescaped
|
|
281
|
+
# in the arena.
|
|
282
|
+
def header_names
|
|
283
|
+
spans = @kernel.names
|
|
284
|
+
arena = nil
|
|
285
|
+
Array.new(@kernel.rows) do |index|
|
|
286
|
+
offset = spans[index * 2]
|
|
287
|
+
length = spans[index * 2 + 1]
|
|
288
|
+
name =
|
|
289
|
+
if length < Runtime::Delimited::SPAN_FLAG
|
|
290
|
+
@buffer.byteslice(@start + offset, length)
|
|
291
|
+
else
|
|
292
|
+
arena ||= @kernel.arena.force_encoding(Encoding::UTF_8)
|
|
293
|
+
arena.byteslice(offset, length & Runtime::Delimited::SPAN_LENGTH)
|
|
294
|
+
end
|
|
295
|
+
name.freeze
|
|
296
|
+
end.freeze
|
|
297
|
+
end
|
|
298
|
+
|
|
299
|
+
# The structural failure the core just reported, kept: it is final.
|
|
300
|
+
def structural
|
|
301
|
+
@failure = TabularError.from(@kernel.failure)
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
# The failure of a record that outgrew the ceiling, kept: it is final too.
|
|
305
|
+
def too_long
|
|
306
|
+
@failure = TabularError.new(:row_too_long, @kernel.records, @kernel.line, @kernel.offset)
|
|
307
|
+
end
|
|
308
|
+
end
|
|
309
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
module HyperTabular
|
|
2
|
+
# How the text is delimited, declared by the caller. Nothing is sniffed: the separator is
|
|
3
|
+
# stated, quoting is stated, the header is stated — the same stance HyperCast's NumFormat
|
|
4
|
+
# takes for numeric notation.
|
|
5
|
+
#
|
|
6
|
+
# +separator+ is one byte: tab, or any printable ASCII character except the double quote.
|
|
7
|
+
# +quoting+ says whether " quotes cells (RFC 4180, "" for a literal quote); off, a quote
|
|
8
|
+
# is an ordinary byte. +has_header+ says whether the first record is a header, given by
|
|
9
|
+
# DelimitedReader#header and never delivered as a row. +skip_blank_lines+ says whether a
|
|
10
|
+
# completely empty line is skipped rather than read as a one-cell row.
|
|
11
|
+
#
|
|
12
|
+
# HyperTabular::Dialect::CSV
|
|
13
|
+
# HyperTabular::Dialect.new(separator: ";", has_header: false)
|
|
14
|
+
Dialect = Data.define(:separator, :quoting, :has_header, :skip_blank_lines) do
|
|
15
|
+
# Everything but the separator defaults to what CSV means: quoted, with a header, blank
|
|
16
|
+
# lines skipped. A separator that is not a one-byte String is a caller bug
|
|
17
|
+
# (ArgumentError); which bytes the scanner can honour is the core's to say, and it says
|
|
18
|
+
# so when a reader is built.
|
|
19
|
+
def initialize(separator:, quoting: true, has_header: true, skip_blank_lines: true)
|
|
20
|
+
raise ArgumentError, "separator must be a one-byte String; got #{separator.inspect}" unless
|
|
21
|
+
separator.is_a?(String) && separator.bytesize == 1
|
|
22
|
+
|
|
23
|
+
super(separator: -separator, quoting: quoting ? true : false,
|
|
24
|
+
has_header: has_header ? true : false, skip_blank_lines: skip_blank_lines ? true : false)
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# The four bytes of the core's RawDialect: separator, quoting, skip_blank_lines, and
|
|
28
|
+
# the scan engine left to the core's own choice.
|
|
29
|
+
def packed
|
|
30
|
+
[separator.getbyte(0), quoting ? 1 : 0, skip_blank_lines ? 1 : 0, 0].pack("C4")
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# Comma-separated, quoted, with a header, blank lines skipped.
|
|
35
|
+
Dialect::CSV = Dialect.new(separator: ",")
|
|
36
|
+
|
|
37
|
+
# Tab-separated, otherwise as CSV.
|
|
38
|
+
Dialect::TSV = Dialect.new(separator: "\t")
|
|
39
|
+
|
|
40
|
+
# Pipe-separated, otherwise as CSV.
|
|
41
|
+
Dialect::PSV = Dialect.new(separator: "|")
|
|
42
|
+
end
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
module HyperTabular
|
|
2
|
+
module Runtime
|
|
3
|
+
# A plan as the core takes it, and the arrays the core casts it into: one packed
|
|
4
|
+
# ColumnSpec per plan column, one value array and one verdict array per column sized
|
|
5
|
+
# for one batch, and the ColumnBuffer table that names them — allocated once and reused
|
|
6
|
+
# for every batch. A delimited read and a sheet each own one.
|
|
7
|
+
class Columns
|
|
8
|
+
# Bytes in one verdict (CellVerdict: offset, len, reason — three u32).
|
|
9
|
+
VERDICT_BYTES = 12
|
|
10
|
+
# ColumnSpec: ordinal, door, param, then HyperCast's 32-byte RawNumFormat.
|
|
11
|
+
SPEC_BYTES = 44
|
|
12
|
+
|
|
13
|
+
# The packed specs (nil for no columns), the ColumnBuffer table (likewise), how many
|
|
14
|
+
# columns there are, and the most rows a batch holds.
|
|
15
|
+
attr_reader :specs, :table, :count, :batch_rows
|
|
16
|
+
|
|
17
|
+
# +specs+ is one packed ColumnSpec per plan column and +sizes+ the bytes one value of
|
|
18
|
+
# each takes.
|
|
19
|
+
def initialize(specs, sizes, batch_rows)
|
|
20
|
+
@sizes = sizes
|
|
21
|
+
@count = specs.size
|
|
22
|
+
@batch_rows = batch_rows
|
|
23
|
+
return if specs.empty?
|
|
24
|
+
|
|
25
|
+
@specs = Runtime.buffer(SPEC_BYTES * @count)
|
|
26
|
+
@specs[0, SPEC_BYTES * @count] = specs.join
|
|
27
|
+
@values = sizes.map { |size| Runtime.buffer(size * batch_rows) }
|
|
28
|
+
@verdicts = sizes.map { Runtime.buffer(VERDICT_BYTES * batch_rows) }
|
|
29
|
+
addresses = @values.zip(@verdicts).flatten.map(&:to_i).pack("J*")
|
|
30
|
+
@table = Runtime.buffer(addresses.bytesize)
|
|
31
|
+
@table[0, addresses.bytesize] = addresses
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# A column's value array for the first +rows+ rows, copied out as the core wrote it.
|
|
35
|
+
def values(column, rows)
|
|
36
|
+
@values[column][0, @sizes[column] * rows]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# A column's verdict array for the first +rows+ rows — offset, len, reason per row —
|
|
40
|
+
# copied out as the core wrote it.
|
|
41
|
+
def verdicts(column, rows)
|
|
42
|
+
@verdicts[column][0, VERDICT_BYTES * rows]
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|