hypertabular-wasm 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,170 @@
1
+ module HyperTabular
2
+ # The doors a column can be cast through, by the code the core knows each as: HyperCast's
3
+ # twenty-one, named as its gem names them, plus :text for the bytes themselves.
4
+ DOORS = {
5
+ bool: 1, i8: 2, i16: 3, i32: 4, i64: 5, u8: 6, u16: 7, u32: 8, u64: 9, f32: 10, f64: 11,
6
+ uuid: 12, timestamp: 13, unix: 14, date: 15, time: 16, duration: 17, text: 18, decimal: 19,
7
+ date_ordered: 20, datetime: 21, excel_serial: 22
8
+ }.freeze
9
+
10
+ # One output column of a plan: which source column it reads, the door it casts through,
11
+ # what that door declares, and — for the numeric doors — the notation. A plan is a
12
+ # projection: a forty-column file can be read into five typed columns, in any order, and
13
+ # a source column can be read through more than one door.
14
+ #
15
+ # Built by the factory named after its door, never guessed from the data:
16
+ #
17
+ # plan = [
18
+ # HyperTabular::Column.i32(0),
19
+ # HyperTabular::Column.text(1),
20
+ # HyperTabular::Column.decimal(2, eurozone),
21
+ # HyperTabular::Column.date(3, :day_month_year),
22
+ # HyperTabular::Column.unix(4, :milliseconds)
23
+ # ]
24
+ #
25
+ # +ordinal+ is the zero-based source column; past a record's last cell it reads as empty.
26
+ # +door+ is a key of DOORS. +declared+ is the Symbol the door was declared with — a Unix
27
+ # precision, a date order, an Excel epoch, as HyperCast names them — or nil. +format+ is
28
+ # the HyperCast::NumFormat of a numeric door, or nil.
29
+ Column = Data.define(:ordinal, :door, :declared, :format) do
30
+ # Checks the column as the caller bug it would otherwise become at the first read: an
31
+ # ordinal that is not a non-negative Integer or a format that is not a
32
+ # HyperCast::NumFormat (ArgumentError), an unknown door or an undeclared option
33
+ # (KeyError, as HyperCast raises for the same).
34
+ def initialize(ordinal:, door:, declared: nil, format: nil)
35
+ raise ArgumentError, "ordinal must be an Integer in 0..#{Column::MAX_ORDINAL}; got #{ordinal.inspect}" unless
36
+ ordinal.is_a?(Integer) && ordinal.between?(0, Column::MAX_ORDINAL)
37
+
38
+ DOORS.fetch(door)
39
+ if (options = Column::DECLARES[door])
40
+ options.fetch(declared)
41
+ elsif !declared.nil?
42
+ raise ArgumentError, "the #{door} door declares nothing; got #{declared.inspect}"
43
+ end
44
+ if Column::NUMERIC.include?(door)
45
+ format = HyperCast::NumFormat::INVARIANT if format.nil?
46
+ raise ArgumentError, "format must be a HyperCast::NumFormat; got #{format.inspect}" unless
47
+ format.is_a?(HyperCast::NumFormat)
48
+ elsif !format.nil?
49
+ raise ArgumentError, "the #{door} door reads no numeric format"
50
+ end
51
+
52
+ super(ordinal: ordinal, door: door, declared: declared, format: format)
53
+ end
54
+
55
+ # A boolean column: HyperCast's boolean lexicon, as true or false.
56
+ def self.bool(ordinal) = new(ordinal: ordinal, door: :bool)
57
+
58
+ # A signed 8-bit column, as an Integer. Invariant notation unless one is declared.
59
+ def self.i8(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i8, format: format)
60
+
61
+ # A signed 16-bit column, as an Integer. Invariant notation unless one is declared.
62
+ def self.i16(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i16, format: format)
63
+
64
+ # A signed 32-bit column, as an Integer. Invariant notation unless one is declared.
65
+ def self.i32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i32, format: format)
66
+
67
+ # A signed 64-bit column, as an Integer. Invariant notation unless one is declared.
68
+ def self.i64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :i64, format: format)
69
+
70
+ # An unsigned 8-bit column, as an Integer. Invariant notation unless one is declared.
71
+ def self.u8(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u8, format: format)
72
+
73
+ # An unsigned 16-bit column, as an Integer. Invariant notation unless one is declared.
74
+ def self.u16(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u16, format: format)
75
+
76
+ # An unsigned 32-bit column, as an Integer. Invariant notation unless one is declared.
77
+ def self.u32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u32, format: format)
78
+
79
+ # An unsigned 64-bit column, as the true unsigned Integer. Invariant notation unless
80
+ # one is declared.
81
+ def self.u64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :u64, format: format)
82
+
83
+ # An IEEE single column, widened losslessly to a Float. Invariant notation unless one
84
+ # is declared.
85
+ def self.f32(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :f32, format: format)
86
+
87
+ # An IEEE double column, as a Float. Invariant notation unless one is declared.
88
+ def self.f64(ordinal, format = HyperCast::NumFormat::INVARIANT) = new(ordinal: ordinal, door: :f64, format: format)
89
+
90
+ # An exact decimal column, as a HyperCast::Decimal; no float is ever formed. Invariant
91
+ # notation unless one is declared.
92
+ def self.decimal(ordinal, format = HyperCast::NumFormat::INVARIANT)
93
+ new(ordinal: ordinal, door: :decimal, format: format)
94
+ end
95
+
96
+ # A UUID column, as the lowercase hyphenated String.
97
+ def self.uuid(ordinal) = new(ordinal: ordinal, door: :uuid)
98
+
99
+ # An RFC 3339 instant column, as a UTC Time at full nanosecond fidelity.
100
+ def self.timestamp(ordinal) = new(ordinal: ordinal, door: :timestamp)
101
+
102
+ # A Unix-epoch column at the declared precision (:seconds, :milliseconds,
103
+ # :microseconds, :nanoseconds) — never guessed from magnitude — as a UTC Time.
104
+ def self.unix(ordinal, precision) = new(ordinal: ordinal, door: :unix, declared: precision)
105
+
106
+ # An Excel date-serial column under the declared date system (:y1900, :y1904), as a
107
+ # UTC Time.
108
+ def self.excel_serial(ordinal, epoch) = new(ordinal: ordinal, door: :excel_serial, declared: epoch)
109
+
110
+ # A calendar date column, as a Date. With no order declared: the strict ISO 8601
111
+ # yyyy-MM-dd door. With one (:year_month_day, :month_day_year, :day_month_year): the
112
+ # date_ordered door, which also reads the separated forms — the same split
113
+ # HyperCast.date makes.
114
+ def self.date(ordinal, order = nil)
115
+ order.nil? ? new(ordinal: ordinal, door: :date) : date_ordered(ordinal, order)
116
+ end
117
+
118
+ # A separated calendar date column under the declared field order, as a Date.
119
+ def self.date_ordered(ordinal, order) = new(ordinal: ordinal, door: :date_ordered, declared: order)
120
+
121
+ # A zone-less civil date-time column under the declared field order, as a DateTime
122
+ # whose +00:00 offset is a carrier artifact, not data — the text named no zone.
123
+ def self.datetime(ordinal, order) = new(ordinal: ordinal, door: :datetime, declared: order)
124
+
125
+ # A 24-hour time-of-day column, as an exact Integer of nanoseconds since midnight.
126
+ def self.time(ordinal) = new(ordinal: ordinal, door: :time)
127
+
128
+ # A duration column, as exact Rational seconds.
129
+ def self.duration(ordinal) = new(ordinal: ordinal, door: :duration)
130
+
131
+ # A text column: the cell's bytes themselves, untrimmed, quotes resolved, as a UTF-8
132
+ # String. A cell with no bytes at all is the one way text fails (an :empty Fault).
133
+ def self.text(ordinal) = new(ordinal: ordinal, door: :text)
134
+
135
+ # Bytes one value of this column's door takes in a column buffer.
136
+ def value_bytes
137
+ Column::VALUE_BYTES.fetch(door)
138
+ end
139
+
140
+ # The 44 little-endian bytes of the core's ColumnSpec: ordinal, door code, what the
141
+ # door declares (numbered as HyperCast numbers it), then the notation in HyperCast's
142
+ # own packed form — all zeros, which the core reads as invariant, for a door that
143
+ # reads none.
144
+ def packed
145
+ param = declared.nil? ? 0 : Column::DECLARES.fetch(door).fetch(declared)
146
+ [ordinal, DOORS.fetch(door), param].pack("L<3") + (format.nil? ? Column::NO_FORMAT : format.packed)
147
+ end
148
+ end
149
+
150
+ # The widest source ordinal a column can name.
151
+ Column::MAX_ORDINAL = (1 << 31) - 1
152
+
153
+ # The doors that read a HyperCast::NumFormat: the integers, the reals and the decimal.
154
+ Column::NUMERIC = %i[i8 i16 i32 i64 u8 u16 u32 u64 f32 f64 decimal].freeze
155
+
156
+ # What the doors that declare something declare, as HyperCast's own tables number it.
157
+ Column::DECLARES = {
158
+ unix: HyperCast::UNIX_PRECISIONS,
159
+ excel_serial: HyperCast::EXCEL_EPOCHS,
160
+ date_ordered: HyperCast::DATE_ORDERS,
161
+ datetime: HyperCast::DATE_ORDERS
162
+ }.freeze
163
+
164
+ # Bytes per value in a column buffer, by door (rust/src/kernel/abi.rs, ColumnBuffer):
165
+ # HyperCast's for its doors, and a text cell's span — two u32s — for this one's.
166
+ Column::VALUE_BYTES = HyperCast::Interop::VALUE_BYTES.merge(text: 8).freeze
167
+
168
+ # The notation of a door that reads none: thirty-two zero bytes.
169
+ Column::NO_FORMAT = ("\0" * 32).b.freeze
170
+ end
@@ -0,0 +1,309 @@
1
+ module HyperTabular
2
+ # Delimited text — CSV, TSV, any single-byte ASCII separator — read a batch at a time
3
+ # into typed columns, every cell a HyperCast verdict.
4
+ #
5
+ # plan = [HyperTabular::Column.i32(0), HyperTabular::Column.text(1), HyperTabular::Column.f64(2)]
6
+ # HyperTabular::DelimitedReader.open("orders.csv", HyperTabular::Dialect::CSV, plan) do |reader|
7
+ # while (batch = reader.read)
8
+ # ids = batch.values(0) # a column at a time: [1, 2, nil, 4, ...]
9
+ # batch.rows.times do |row|
10
+ # case batch.get(2, row) # or a cell at a time, as HyperCast's union
11
+ # in HyperCast::Success(value:) then total += value
12
+ # in HyperCast::Fault(reason:) then warn "#{reason}: #{batch.raw(2, row).inspect}"
13
+ # end
14
+ # end
15
+ # end
16
+ # end
17
+ #
18
+ # The native core (libhypertabular) owns no memory and reads no files. It is handed a
19
+ # chunk of input and the buffers to fill, casts each plan column in one native loop, and
20
+ # says how many rows it wrote and how many bytes it is finished with. Everything else is
21
+ # here: this class holds the input, has one value array and one verdict array per column
22
+ # allocated once and reused for every batch, and puts what the core did not consume back
23
+ # in front of it. The native boundary is crossed once per batch, not once per cell, and a
24
+ # column comes out of its buffer in one String#unpack.
25
+ #
26
+ # A value that does not cast is that cell's verdict, and the read goes on. Input that is
27
+ # not rows of cells at all — a record of the wrong width, a quote never closed — is a
28
+ # TabularError, raised after every intact row before it has been delivered.
29
+ #
30
+ # Each #read returns a Batch, which owns what it shows: it stays good after the next
31
+ # #read. Not thread-safe.
32
+ class DelimitedReader
33
+ # The row ceiling: a single record larger than this is a :row_too_long TabularError.
34
+ MAX_ROW_BYTES = 1 << 30
35
+
36
+ # How much is asked of an IO at a time, until a record does not fit.
37
+ DEFAULT_BUFFER_BYTES = 256 * 1024
38
+
39
+ # Rows per batch unless told otherwise.
40
+ DEFAULT_BATCH_ROWS = 4096
41
+
42
+ # Encodings whose bytes already are the UTF-8 (or byte-identical) form the core reads.
43
+ BYTE_COMPATIBLE = HyperCast::Interop::BYTE_COMPATIBLE
44
+
45
+ # Opens a file of UTF-8 delimited text. With a block, yields the reader, closes it —
46
+ # and the file — when the block ends, and returns the block's value; without one,
47
+ # returns the reader, whose #close closes the file.
48
+ def self.open(path, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES)
49
+ file = File.open(path, "rb")
50
+ begin
51
+ reader = new(file, dialect, plan, batch_rows: batch_rows, buffer_bytes: buffer_bytes, close_source: true)
52
+ ensure
53
+ # The file is ours to close when no reader came of it, whatever went wrong.
54
+ file.close if reader.nil?
55
+ end
56
+ return reader unless block_given?
57
+
58
+ begin
59
+ yield reader
60
+ ensure
61
+ reader.close
62
+ end
63
+ end
64
+
65
+ # Reads UTF-8 delimited text from +source+: a String, read in place — nothing is copied
66
+ # — or an IO (anything with #readpartial or #read), read forward only through a buffer
67
+ # of +buffer_bytes+ that grows when a record does not fit it. An IO is read as bytes;
68
+ # one opened in text mode on Windows should be in binmode first.
69
+ #
70
+ # +dialect+ is the declared Dialect and +plan+ the output Columns, in output order.
71
+ # +batch_rows+ is the most rows one #read delivers. +close_source+ says whether #close
72
+ # also closes the IO.
73
+ #
74
+ # A dialect the scanner cannot honour, or a plan that is not Columns, is an
75
+ # ArgumentError. With a header declared the header record is read here, so a header
76
+ # that is structurally broken is a TabularError here.
77
+ def initialize(source, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES,
78
+ close_source: false)
79
+ raise ArgumentError, "dialect must be a HyperTabular::Dialect; got #{dialect.inspect}" unless
80
+ dialect.is_a?(Dialect)
81
+ raise ArgumentError, "batch_rows must be a positive Integer; got #{batch_rows.inspect}" unless
82
+ batch_rows.is_a?(Integer) && batch_rows.positive?
83
+ raise ArgumentError, "buffer_bytes must be a positive Integer; got #{buffer_bytes.inspect}" unless
84
+ buffer_bytes.is_a?(Integer) && buffer_bytes.positive?
85
+
86
+ @plan = Array(plan).dup.freeze
87
+ @plan.each_with_index do |column, index|
88
+ raise ArgumentError, "plan column #{index} must be a HyperTabular::Column; got #{column.inspect}" unless
89
+ column.is_a?(Column)
90
+ end
91
+ @dialect = dialect
92
+ @batch_rows = batch_rows
93
+ @buffer_bytes = buffer_bytes
94
+ @close_source = close_source
95
+ # Cell-table entries one row takes: the widest ordinal the plan reads, plus two.
96
+ per_row = (@plan.map(&:ordinal).max || -1) + 2
97
+ @kernel = Runtime::Delimited.start(dialect.packed, @plan.map(&:packed), @plan.map(&:value_bytes),
98
+ batch_rows, per_row)
99
+ raise ArgumentError, "separator #{dialect.separator.inspect} is not tab or printable ASCII other than '\"'" if
100
+ @kernel.nil?
101
+
102
+ @start = 0
103
+ @closed = false
104
+ @failure = nil
105
+ take(source)
106
+ @header = dialect.has_header ? read_header : nil
107
+ end
108
+
109
+ # The header's names, when the dialect declares a header: frozen UTF-8 Strings, quotes
110
+ # resolved. Empty for an input with no record at all; nil when the dialect declares no
111
+ # header.
112
+ attr_reader :header
113
+
114
+ # The plan: the output Columns, in output order. Column +i+ of every batch is +plan[i]+.
115
+ attr_reader :plan
116
+
117
+ # The declared Dialect.
118
+ attr_reader :dialect
119
+
120
+ # The most rows one #read delivers.
121
+ attr_reader :batch_rows
122
+
123
+ # Records finished so far — the header and skipped blank lines included.
124
+ def records
125
+ @kernel.records
126
+ end
127
+
128
+ # Reads the next batch: a Batch of up to #batch_rows rows, or nil once the input is
129
+ # exhausted. A TabularError when the input is structurally broken — raised after every
130
+ # intact row before the break has been delivered, and the same error again on every
131
+ # later call. IOError on a closed reader.
132
+ def read
133
+ raise IOError, "closed reader" if @closed
134
+ raise @failure if @failure
135
+
136
+ loop do
137
+ length, last = window
138
+ raise structural unless @kernel.fill(@start, length, last) == Runtime::Delimited::OK
139
+
140
+ consumed = @kernel.consumed
141
+ if @kernel.rows.positive?
142
+ batch = made(@kernel.rows)
143
+ @start += consumed
144
+ return batch
145
+ end
146
+ @start += consumed
147
+ return nil if last && (consumed == length || consumed.zero?)
148
+
149
+ refill if consumed.zero?
150
+ end
151
+ end
152
+
153
+ # Every remaining row, batch after batch: yields one frozen Array per row, a verdict
154
+ # per plan column. Without a block, an Enumerator. Raises what #read raises.
155
+ def each_row
156
+ return to_enum(:each_row) unless block_given?
157
+
158
+ while (batch = read)
159
+ columns = Array.new(@plan.size) { |column| batch.verdicts(column) }
160
+ batch.rows.times { |row| yield columns.map { |column| column[row] }.freeze }
161
+ end
162
+ self
163
+ end
164
+
165
+ # Closes the IO if this reader was told to (+close_source+, or DelimitedReader.open).
166
+ # Batches already read stay good. Safe to call twice.
167
+ def close
168
+ return if @closed
169
+
170
+ @closed = true
171
+ @io.close if @close_source && @io
172
+ nil
173
+ end
174
+
175
+ # True once #close has been called.
176
+ def closed?
177
+ @closed
178
+ end
179
+
180
+ # The reader in a line — not its buffers.
181
+ def inspect
182
+ "#<#{self.class.name} columns=#{@plan.size} records=#{records}#{' closed' if @closed}>"
183
+ end
184
+
185
+ private
186
+
187
+ # Takes the source: a String becomes the whole input, an IO the thing to refill from.
188
+ def take(source)
189
+ if source.is_a?(String)
190
+ @buffer = utf8(source)
191
+ @eof = true
192
+ elsif defined?(::Pathname) && source.is_a?(::Pathname)
193
+ # It has a #read, and one that starts the file over on every call.
194
+ raise ArgumentError, "source is a Pathname; DelimitedReader.open reads a path"
195
+ elsif source.respond_to?(:readpartial) || source.respond_to?(:read)
196
+ @io = source
197
+ @partial = source.respond_to?(:readpartial)
198
+ @buffer = String.new(encoding: Encoding::UTF_8)
199
+ @eof = false
200
+ else
201
+ raise ArgumentError, "source must be a String or an IO; got #{source.class}"
202
+ end
203
+ @kernel.attach(@buffer)
204
+ end
205
+
206
+ # The text as a frozen String of UTF-8 bytes that is this reader's alone: the caller's
207
+ # own when it is already that; otherwise a copy-on-write duplicate, which shares the
208
+ # bytes and is immune to what the caller does to theirs next. Only a foreign encoding
209
+ # pays a transcode.
210
+ def utf8(text)
211
+ return text if text.frozen? && text.encoding == Encoding::UTF_8
212
+ return text.dup.force_encoding(Encoding::UTF_8).freeze if BYTE_COMPATIBLE.include?(text.encoding)
213
+
214
+ text.encode(Encoding::UTF_8).freeze
215
+ end
216
+
217
+ # What the core is to read next — how many bytes from @start — and whether nothing
218
+ # follows them.
219
+ def window
220
+ length = @buffer.bytesize - @start
221
+ length > MAX_ROW_BYTES ? [MAX_ROW_BYTES, false] : [length, @eof]
222
+ end
223
+
224
+ # Puts the unfinished record at the front of the buffer and reads more behind it —
225
+ # at least as much again as is pending, so a record that does not fit doubles the
226
+ # buffer rather than creeping up on it.
227
+ def refill
228
+ pending = @buffer.bytesize - @start
229
+ raise too_long if @io.nil? || pending >= MAX_ROW_BYTES
230
+
231
+ chunk = more([@buffer_bytes, pending].max)
232
+ if chunk.nil? || chunk.empty?
233
+ @eof = true
234
+ return
235
+ end
236
+ chunk = chunk.dup if chunk.frozen?
237
+ @buffer = @buffer.byteslice(@start, pending) if @start.positive?
238
+ @buffer << chunk.force_encoding(Encoding::UTF_8)
239
+ @start = 0
240
+ @kernel.attach(@buffer)
241
+ end
242
+
243
+ # Up to +bytes+ more of the IO, or nil at its end. #readpartial where there is one, so
244
+ # a pipe or a socket hands over what it has rather than waiting to fill the request.
245
+ def more(bytes)
246
+ @partial ? @io.readpartial(bytes) : @io.read(bytes)
247
+ rescue EOFError
248
+ nil
249
+ end
250
+
251
+ # The batch the core just wrote, copied out of the kernel's arrays: its spans index the
252
+ # input from @start, where the core was handed it.
253
+ def made(rows)
254
+ columns = @kernel.columns
255
+ Batch.new(@plan, rows, Array.new(@plan.size) { |column| columns.values(column, rows) },
256
+ Array.new(@plan.size) { |column| columns.verdicts(column, rows) },
257
+ @kernel.cells, @kernel.per_row, @buffer, @start, @kernel.arena, workbook: false)
258
+ end
259
+
260
+ # Reads the header record: its names, or none for an input with no record at all.
261
+ def read_header
262
+ loop do
263
+ length, last = window
264
+ raise structural unless @kernel.header(@start, length, last) == Runtime::Delimited::OK
265
+
266
+ consumed = @kernel.consumed
267
+ if @kernel.rows.positive?
268
+ names = header_names
269
+ @start += consumed
270
+ return names
271
+ end
272
+ @start += consumed
273
+ # An empty input has no header and no rows; the width is unknown.
274
+ return [].freeze if last && (consumed == length || consumed.zero?)
275
+
276
+ refill if consumed.zero?
277
+ end
278
+ end
279
+
280
+ # The names the core just located: in the input as written, or — flagged — unescaped
281
+ # in the arena.
282
+ def header_names
283
+ spans = @kernel.names
284
+ arena = nil
285
+ Array.new(@kernel.rows) do |index|
286
+ offset = spans[index * 2]
287
+ length = spans[index * 2 + 1]
288
+ name =
289
+ if length < Runtime::Delimited::SPAN_FLAG
290
+ @buffer.byteslice(@start + offset, length)
291
+ else
292
+ arena ||= @kernel.arena.force_encoding(Encoding::UTF_8)
293
+ arena.byteslice(offset, length & Runtime::Delimited::SPAN_LENGTH)
294
+ end
295
+ name.freeze
296
+ end.freeze
297
+ end
298
+
299
+ # The structural failure the core just reported, kept: it is final.
300
+ def structural
301
+ @failure = TabularError.from(@kernel.failure)
302
+ end
303
+
304
+ # The failure of a record that outgrew the ceiling, kept: it is final too.
305
+ def too_long
306
+ @failure = TabularError.new(:row_too_long, @kernel.records, @kernel.line, @kernel.offset)
307
+ end
308
+ end
309
+ end
@@ -0,0 +1,42 @@
1
+ module HyperTabular
2
+ # How the text is delimited, declared by the caller. Nothing is sniffed: the separator is
3
+ # stated, quoting is stated, the header is stated — the same stance HyperCast's NumFormat
4
+ # takes for numeric notation.
5
+ #
6
+ # +separator+ is one byte: tab, or any printable ASCII character except the double quote.
7
+ # +quoting+ says whether " quotes cells (RFC 4180, "" for a literal quote); off, a quote
8
+ # is an ordinary byte. +has_header+ says whether the first record is a header, given by
9
+ # DelimitedReader#header and never delivered as a row. +skip_blank_lines+ says whether a
10
+ # completely empty line is skipped rather than read as a one-cell row.
11
+ #
12
+ # HyperTabular::Dialect::CSV
13
+ # HyperTabular::Dialect.new(separator: ";", has_header: false)
14
+ Dialect = Data.define(:separator, :quoting, :has_header, :skip_blank_lines) do
15
+ # Everything but the separator defaults to what CSV means: quoted, with a header, blank
16
+ # lines skipped. A separator that is not a one-byte String is a caller bug
17
+ # (ArgumentError); which bytes the scanner can honour is the core's to say, and it says
18
+ # so when a reader is built.
19
+ def initialize(separator:, quoting: true, has_header: true, skip_blank_lines: true)
20
+ raise ArgumentError, "separator must be a one-byte String; got #{separator.inspect}" unless
21
+ separator.is_a?(String) && separator.bytesize == 1
22
+
23
+ super(separator: -separator, quoting: quoting ? true : false,
24
+ has_header: has_header ? true : false, skip_blank_lines: skip_blank_lines ? true : false)
25
+ end
26
+
27
+ # The four bytes of the core's RawDialect: separator, quoting, skip_blank_lines, and
28
+ # the scan engine left to the core's own choice.
29
+ def packed
30
+ [separator.getbyte(0), quoting ? 1 : 0, skip_blank_lines ? 1 : 0, 0].pack("C4")
31
+ end
32
+ end
33
+
34
+ # Comma-separated, quoted, with a header, blank lines skipped.
35
+ Dialect::CSV = Dialect.new(separator: ",")
36
+
37
+ # Tab-separated, otherwise as CSV.
38
+ Dialect::TSV = Dialect.new(separator: "\t")
39
+
40
+ # Pipe-separated, otherwise as CSV.
41
+ Dialect::PSV = Dialect.new(separator: "|")
42
+ end
@@ -0,0 +1,46 @@
1
+ module HyperTabular
2
+ module Runtime
3
+ # A plan as the core takes it, and the arrays the core casts it into: one packed
4
+ # ColumnSpec per plan column, one value array and one verdict array per column sized
5
+ # for one batch, and the ColumnBuffer table that names them — allocated once and reused
6
+ # for every batch. A delimited read and a sheet each own one.
7
+ class Columns
8
+ # Bytes in one verdict (CellVerdict: offset, len, reason — three u32).
9
+ VERDICT_BYTES = 12
10
+ # ColumnSpec: ordinal, door, param, then HyperCast's 32-byte RawNumFormat.
11
+ SPEC_BYTES = 44
12
+
13
+ # The packed specs (nil for no columns), the ColumnBuffer table (likewise), how many
14
+ # columns there are, and the most rows a batch holds.
15
+ attr_reader :specs, :table, :count, :batch_rows
16
+
17
+ # +specs+ is one packed ColumnSpec per plan column and +sizes+ the bytes one value of
18
+ # each takes.
19
+ def initialize(specs, sizes, batch_rows)
20
+ @sizes = sizes
21
+ @count = specs.size
22
+ @batch_rows = batch_rows
23
+ return if specs.empty?
24
+
25
+ @specs = Runtime.buffer(SPEC_BYTES * @count)
26
+ @specs[0, SPEC_BYTES * @count] = specs.join
27
+ @values = sizes.map { |size| Runtime.buffer(size * batch_rows) }
28
+ @verdicts = sizes.map { Runtime.buffer(VERDICT_BYTES * batch_rows) }
29
+ addresses = @values.zip(@verdicts).flatten.map(&:to_i).pack("J*")
30
+ @table = Runtime.buffer(addresses.bytesize)
31
+ @table[0, addresses.bytesize] = addresses
32
+ end
33
+
34
+ # A column's value array for the first +rows+ rows, copied out as the core wrote it.
35
+ def values(column, rows)
36
+ @values[column][0, @sizes[column] * rows]
37
+ end
38
+
39
+ # A column's verdict array for the first +rows+ rows — offset, len, reason per row —
40
+ # copied out as the core wrote it.
41
+ def verdicts(column, rows)
42
+ @verdicts[column][0, VERDICT_BYTES * rows]
43
+ end
44
+ end
45
+ end
46
+ end