hypertabular 0.7.0-aarch64-mingw-ucrt

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,309 @@
1
+ module HyperTabular
2
+ # Delimited text — CSV, TSV, any single-byte ASCII separator — read a batch at a time
3
+ # into typed columns, every cell a HyperCast verdict.
4
+ #
5
+ # plan = [HyperTabular::Column.i32(0), HyperTabular::Column.text(1), HyperTabular::Column.f64(2)]
6
+ # HyperTabular::DelimitedReader.open("orders.csv", HyperTabular::Dialect::CSV, plan) do |reader|
7
+ # while (batch = reader.read)
8
+ # ids = batch.values(0) # a column at a time: [1, 2, nil, 4, ...]
9
+ # batch.rows.times do |row|
10
+ # case batch.get(2, row) # or a cell at a time, as HyperCast's union
11
+ # in HyperCast::Success(value:) then total += value
12
+ # in HyperCast::Fault(reason:) then warn "#{reason}: #{batch.raw(2, row).inspect}"
13
+ # end
14
+ # end
15
+ # end
16
+ # end
17
+ #
18
+ # The native core (libhypertabular) owns no memory and reads no files. It is handed a
19
+ # chunk of input and the buffers to fill, casts each plan column in one native loop, and
20
+ # says how many rows it wrote and how many bytes it is finished with. Everything else is
21
+ # here: this class holds the input, has one value array and one verdict array per column
22
+ # allocated once and reused for every batch, and puts what the core did not consume back
23
+ # in front of it. The native boundary is crossed once per batch, not once per cell, and a
24
+ # column comes out of its buffer in one String#unpack.
25
+ #
26
+ # A value that does not cast is that cell's verdict, and the read goes on. Input that is
27
+ # not rows of cells at all — a record of the wrong width, a quote never closed — is a
28
+ # TabularError, raised after every intact row before it has been delivered.
29
+ #
30
+ # Each #read returns a Batch, which owns what it shows: it stays good after the next
31
+ # #read. Not thread-safe.
32
+ class DelimitedReader
33
+ # The row ceiling: a single record larger than this is a :row_too_long TabularError.
34
+ MAX_ROW_BYTES = 1 << 30
35
+
36
+ # How much is asked of an IO at a time, until a record does not fit.
37
+ DEFAULT_BUFFER_BYTES = 256 * 1024
38
+
39
+ # Rows per batch unless told otherwise.
40
+ DEFAULT_BATCH_ROWS = 4096
41
+
42
+ # Encodings whose bytes already are the UTF-8 (or byte-identical) form the core reads.
43
+ BYTE_COMPATIBLE = HyperCast::Interop::BYTE_COMPATIBLE
44
+
45
+ # Opens a file of UTF-8 delimited text. With a block, yields the reader, closes it —
46
+ # and the file — when the block ends, and returns the block's value; without one,
47
+ # returns the reader, whose #close closes the file.
48
+ def self.open(path, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES)
49
+ file = File.open(path, "rb")
50
+ begin
51
+ reader = new(file, dialect, plan, batch_rows: batch_rows, buffer_bytes: buffer_bytes, close_source: true)
52
+ ensure
53
+ # The file is ours to close when no reader came of it, whatever went wrong.
54
+ file.close if reader.nil?
55
+ end
56
+ return reader unless block_given?
57
+
58
+ begin
59
+ yield reader
60
+ ensure
61
+ reader.close
62
+ end
63
+ end
64
+
65
+ # Reads UTF-8 delimited text from +source+: a String, read in place — nothing is copied
66
+ # — or an IO (anything with #readpartial or #read), read forward only through a buffer
67
+ # of +buffer_bytes+ that grows when a record does not fit it. An IO is read as bytes;
68
+ # one opened in text mode on Windows should be in binmode first.
69
+ #
70
+ # +dialect+ is the declared Dialect and +plan+ the output Columns, in output order.
71
+ # +batch_rows+ is the most rows one #read delivers. +close_source+ says whether #close
72
+ # also closes the IO.
73
+ #
74
+ # A dialect the scanner cannot honour, or a plan that is not Columns, is an
75
+ # ArgumentError. With a header declared the header record is read here, so a header
76
+ # that is structurally broken is a TabularError here.
77
+ def initialize(source, dialect, plan, batch_rows: DEFAULT_BATCH_ROWS, buffer_bytes: DEFAULT_BUFFER_BYTES,
78
+ close_source: false)
79
+ raise ArgumentError, "dialect must be a HyperTabular::Dialect; got #{dialect.inspect}" unless
80
+ dialect.is_a?(Dialect)
81
+ raise ArgumentError, "batch_rows must be a positive Integer; got #{batch_rows.inspect}" unless
82
+ batch_rows.is_a?(Integer) && batch_rows.positive?
83
+ raise ArgumentError, "buffer_bytes must be a positive Integer; got #{buffer_bytes.inspect}" unless
84
+ buffer_bytes.is_a?(Integer) && buffer_bytes.positive?
85
+
86
+ @plan = Array(plan).dup.freeze
87
+ @plan.each_with_index do |column, index|
88
+ raise ArgumentError, "plan column #{index} must be a HyperTabular::Column; got #{column.inspect}" unless
89
+ column.is_a?(Column)
90
+ end
91
+ @dialect = dialect
92
+ @batch_rows = batch_rows
93
+ @buffer_bytes = buffer_bytes
94
+ @close_source = close_source
95
+ # Cell-table entries one row takes: the widest ordinal the plan reads, plus two.
96
+ per_row = (@plan.map(&:ordinal).max || -1) + 2
97
+ @kernel = Runtime::Delimited.start(dialect.packed, @plan.map(&:packed), @plan.map(&:value_bytes),
98
+ batch_rows, per_row)
99
+ raise ArgumentError, "separator #{dialect.separator.inspect} is not tab or printable ASCII other than '\"'" if
100
+ @kernel.nil?
101
+
102
+ @start = 0
103
+ @closed = false
104
+ @failure = nil
105
+ take(source)
106
+ @header = dialect.has_header ? read_header : nil
107
+ end
108
+
109
+ # The header's names, when the dialect declares a header: frozen UTF-8 Strings, quotes
110
+ # resolved. Empty for an input with no record at all; nil when the dialect declares no
111
+ # header.
112
+ attr_reader :header
113
+
114
+ # The plan: the output Columns, in output order. Column +i+ of every batch is +plan[i]+.
115
+ attr_reader :plan
116
+
117
+ # The declared Dialect.
118
+ attr_reader :dialect
119
+
120
+ # The most rows one #read delivers.
121
+ attr_reader :batch_rows
122
+
123
+ # Records finished so far — the header and skipped blank lines included.
124
+ def records
125
+ @kernel.records
126
+ end
127
+
128
+ # Reads the next batch: a Batch of up to #batch_rows rows, or nil once the input is
129
+ # exhausted. A TabularError when the input is structurally broken — raised after every
130
+ # intact row before the break has been delivered, and the same error again on every
131
+ # later call. IOError on a closed reader.
132
+ def read
133
+ raise IOError, "closed reader" if @closed
134
+ raise @failure if @failure
135
+
136
+ loop do
137
+ length, last = window
138
+ raise structural unless @kernel.fill(@start, length, last) == Runtime::Delimited::OK
139
+
140
+ consumed = @kernel.consumed
141
+ if @kernel.rows.positive?
142
+ batch = made(@kernel.rows)
143
+ @start += consumed
144
+ return batch
145
+ end
146
+ @start += consumed
147
+ return nil if last && (consumed == length || consumed.zero?)
148
+
149
+ refill if consumed.zero?
150
+ end
151
+ end
152
+
153
+ # Every remaining row, batch after batch: yields one frozen Array per row, a verdict
154
+ # per plan column. Without a block, an Enumerator. Raises what #read raises.
155
+ def each_row
156
+ return to_enum(:each_row) unless block_given?
157
+
158
+ while (batch = read)
159
+ columns = Array.new(@plan.size) { |column| batch.verdicts(column) }
160
+ batch.rows.times { |row| yield columns.map { |column| column[row] }.freeze }
161
+ end
162
+ self
163
+ end
164
+
165
+ # Closes the IO if this reader was told to (+close_source+, or DelimitedReader.open).
166
+ # Batches already read stay good. Safe to call twice.
167
+ def close
168
+ return if @closed
169
+
170
+ @closed = true
171
+ @io.close if @close_source && @io
172
+ nil
173
+ end
174
+
175
+ # True once #close has been called.
176
+ def closed?
177
+ @closed
178
+ end
179
+
180
+ # The reader in a line — not its buffers.
181
+ def inspect
182
+ "#<#{self.class.name} columns=#{@plan.size} records=#{records}#{' closed' if @closed}>"
183
+ end
184
+
185
+ private
186
+
187
+ # Takes the source: a String becomes the whole input, an IO the thing to refill from.
188
+ def take(source)
189
+ if source.is_a?(String)
190
+ @buffer = utf8(source)
191
+ @eof = true
192
+ elsif defined?(::Pathname) && source.is_a?(::Pathname)
193
+ # It has a #read, and one that starts the file over on every call.
194
+ raise ArgumentError, "source is a Pathname; DelimitedReader.open reads a path"
195
+ elsif source.respond_to?(:readpartial) || source.respond_to?(:read)
196
+ @io = source
197
+ @partial = source.respond_to?(:readpartial)
198
+ @buffer = String.new(encoding: Encoding::UTF_8)
199
+ @eof = false
200
+ else
201
+ raise ArgumentError, "source must be a String or an IO; got #{source.class}"
202
+ end
203
+ @kernel.attach(@buffer)
204
+ end
205
+
206
+ # The text as a frozen String of UTF-8 bytes that is this reader's alone: the caller's
207
+ # own when it is already that; otherwise a copy-on-write duplicate, which shares the
208
+ # bytes and is immune to what the caller does to theirs next. Only a foreign encoding
209
+ # pays a transcode.
210
+ def utf8(text)
211
+ return text if text.frozen? && text.encoding == Encoding::UTF_8
212
+ return text.dup.force_encoding(Encoding::UTF_8).freeze if BYTE_COMPATIBLE.include?(text.encoding)
213
+
214
+ text.encode(Encoding::UTF_8).freeze
215
+ end
216
+
217
+ # What the core is to read next — how many bytes from @start — and whether nothing
218
+ # follows them.
219
+ def window
220
+ length = @buffer.bytesize - @start
221
+ length > MAX_ROW_BYTES ? [MAX_ROW_BYTES, false] : [length, @eof]
222
+ end
223
+
224
+ # Puts the unfinished record at the front of the buffer and reads more behind it —
225
+ # at least as much again as is pending, so a record that does not fit doubles the
226
+ # buffer rather than creeping up on it.
227
+ def refill
228
+ pending = @buffer.bytesize - @start
229
+ raise too_long if @io.nil? || pending >= MAX_ROW_BYTES
230
+
231
+ chunk = more([@buffer_bytes, pending].max)
232
+ if chunk.nil? || chunk.empty?
233
+ @eof = true
234
+ return
235
+ end
236
+ chunk = chunk.dup if chunk.frozen?
237
+ @buffer = @buffer.byteslice(@start, pending) if @start.positive?
238
+ @buffer << chunk.force_encoding(Encoding::UTF_8)
239
+ @start = 0
240
+ @kernel.attach(@buffer)
241
+ end
242
+
243
+ # Up to +bytes+ more of the IO, or nil at its end. #readpartial where there is one, so
244
+ # a pipe or a socket hands over what it has rather than waiting to fill the request.
245
+ def more(bytes)
246
+ @partial ? @io.readpartial(bytes) : @io.read(bytes)
247
+ rescue EOFError
248
+ nil
249
+ end
250
+
251
+ # The batch the core just wrote, copied out of the kernel's arrays: its spans index the
252
+ # input from @start, where the core was handed it.
253
+ def made(rows)
254
+ columns = @kernel.columns
255
+ Batch.new(@plan, rows, Array.new(@plan.size) { |column| columns.values(column, rows) },
256
+ Array.new(@plan.size) { |column| columns.verdicts(column, rows) },
257
+ @kernel.cells, @kernel.per_row, @buffer, @start, @kernel.arena, workbook: false)
258
+ end
259
+
260
+ # Reads the header record: its names, or none for an input with no record at all.
261
+ def read_header
262
+ loop do
263
+ length, last = window
264
+ raise structural unless @kernel.header(@start, length, last) == Runtime::Delimited::OK
265
+
266
+ consumed = @kernel.consumed
267
+ if @kernel.rows.positive?
268
+ names = header_names
269
+ @start += consumed
270
+ return names
271
+ end
272
+ @start += consumed
273
+ # An empty input has no header and no rows; the width is unknown.
274
+ return [].freeze if last && (consumed == length || consumed.zero?)
275
+
276
+ refill if consumed.zero?
277
+ end
278
+ end
279
+
280
+ # The names the core just located: in the input as written, or — flagged — unescaped
281
+ # in the arena.
282
+ def header_names
283
+ spans = @kernel.names
284
+ arena = nil
285
+ Array.new(@kernel.rows) do |index|
286
+ offset = spans[index * 2]
287
+ length = spans[index * 2 + 1]
288
+ name =
289
+ if length < Runtime::Delimited::SPAN_FLAG
290
+ @buffer.byteslice(@start + offset, length)
291
+ else
292
+ arena ||= @kernel.arena.force_encoding(Encoding::UTF_8)
293
+ arena.byteslice(offset, length & Runtime::Delimited::SPAN_LENGTH)
294
+ end
295
+ name.freeze
296
+ end.freeze
297
+ end
298
+
299
+ # The structural failure the core just reported, kept: it is final.
300
+ def structural
301
+ @failure = TabularError.from(@kernel.failure)
302
+ end
303
+
304
+ # The failure of a record that outgrew the ceiling, kept: it is final too.
305
+ def too_long
306
+ @failure = TabularError.new(:row_too_long, @kernel.records, @kernel.line, @kernel.offset)
307
+ end
308
+ end
309
+ end
@@ -0,0 +1,42 @@
1
+ module HyperTabular
2
+ # How the text is delimited, declared by the caller. Nothing is sniffed: the separator is
3
+ # stated, quoting is stated, the header is stated — the same stance HyperCast's NumFormat
4
+ # takes for numeric notation.
5
+ #
6
+ # +separator+ is one byte: tab, or any printable ASCII character except the double quote.
7
+ # +quoting+ says whether " quotes cells (RFC 4180, "" for a literal quote); off, a quote
8
+ # is an ordinary byte. +has_header+ says whether the first record is a header, given by
9
+ # DelimitedReader#header and never delivered as a row. +skip_blank_lines+ says whether a
10
+ # completely empty line is skipped rather than read as a one-cell row.
11
+ #
12
+ # HyperTabular::Dialect::CSV
13
+ # HyperTabular::Dialect.new(separator: ";", has_header: false)
14
+ Dialect = Data.define(:separator, :quoting, :has_header, :skip_blank_lines) do
15
+ # Everything but the separator defaults to what CSV means: quoted, with a header, blank
16
+ # lines skipped. A separator that is not a one-byte String is a caller bug
17
+ # (ArgumentError); which bytes the scanner can honour is the core's to say, and it says
18
+ # so when a reader is built.
19
+ def initialize(separator:, quoting: true, has_header: true, skip_blank_lines: true)
20
+ raise ArgumentError, "separator must be a one-byte String; got #{separator.inspect}" unless
21
+ separator.is_a?(String) && separator.bytesize == 1
22
+
23
+ super(separator: -separator, quoting: quoting ? true : false,
24
+ has_header: has_header ? true : false, skip_blank_lines: skip_blank_lines ? true : false)
25
+ end
26
+
27
+ # The four bytes of the core's RawDialect: separator, quoting, skip_blank_lines, and
28
+ # the scan engine left to the core's own choice.
29
+ def packed
30
+ [separator.getbyte(0), quoting ? 1 : 0, skip_blank_lines ? 1 : 0, 0].pack("C4")
31
+ end
32
+ end
33
+
34
+ # Comma-separated, quoted, with a header, blank lines skipped.
35
+ Dialect::CSV = Dialect.new(separator: ",")
36
+
37
+ # Tab-separated, otherwise as CSV.
38
+ Dialect::TSV = Dialect.new(separator: "\t")
39
+
40
+ # Pipe-separated, otherwise as CSV.
41
+ Dialect::PSV = Dialect.new(separator: "|")
42
+ end
@@ -0,0 +1,46 @@
1
+ module HyperTabular
2
+ module Runtime
3
+ # A plan as the core takes it, and the arrays the core casts it into: one packed
4
+ # ColumnSpec per plan column, one value array and one verdict array per column sized
5
+ # for one batch, and the ColumnBuffer table that names them — allocated once and reused
6
+ # for every batch. A delimited read and a sheet each own one.
7
+ class Columns
8
+ # Bytes in one verdict (CellVerdict: offset, len, reason — three u32).
9
+ VERDICT_BYTES = 12
10
+ # ColumnSpec: ordinal, door, param, then HyperCast's 32-byte RawNumFormat.
11
+ SPEC_BYTES = 44
12
+
13
+ # The packed specs (nil for no columns), the ColumnBuffer table (likewise), how many
14
+ # columns there are, and the most rows a batch holds.
15
+ attr_reader :specs, :table, :count, :batch_rows
16
+
17
+ # +specs+ is one packed ColumnSpec per plan column and +sizes+ the bytes one value of
18
+ # each takes.
19
+ def initialize(specs, sizes, batch_rows)
20
+ @sizes = sizes
21
+ @count = specs.size
22
+ @batch_rows = batch_rows
23
+ return if specs.empty?
24
+
25
+ @specs = Runtime.buffer(SPEC_BYTES * @count)
26
+ @specs[0, SPEC_BYTES * @count] = specs.join
27
+ @values = sizes.map { |size| Runtime.buffer(size * batch_rows) }
28
+ @verdicts = sizes.map { Runtime.buffer(VERDICT_BYTES * batch_rows) }
29
+ addresses = @values.zip(@verdicts).flatten.map(&:to_i).pack("J*")
30
+ @table = Runtime.buffer(addresses.bytesize)
31
+ @table[0, addresses.bytesize] = addresses
32
+ end
33
+
34
+ # A column's value array for the first +rows+ rows, copied out as the core wrote it.
35
+ def values(column, rows)
36
+ @values[column][0, @sizes[column] * rows]
37
+ end
38
+
39
+ # A column's verdict array for the first +rows+ rows — offset, len, reason per row —
40
+ # copied out as the core wrote it.
41
+ def verdicts(column, rows)
42
+ @verdicts[column][0, VERDICT_BYTES * rows]
43
+ end
44
+ end
45
+ end
46
+ end
@@ -0,0 +1,204 @@
1
+ module HyperTabular
2
+ module Runtime
3
+ # The core's delimited reader as one object: every byte of native memory a read needs
4
+ # and every native call it makes. This class is the whole Fiddle crossing — the reader
5
+ # above it (DelimitedReader) is plain Ruby that never sees a pointer — and so it is the
6
+ # one thing a compiled extension would replace: same methods, bytes in and bytes out.
7
+ #
8
+ # The memory is allocated once and reused for every batch: the state block, the plan,
9
+ # one value array and one verdict array per column, the cell table, the arena. The core
10
+ # keeps none of it between calls beyond what it writes into the state block.
11
+ class Delimited
12
+ # The call did what it could; the result says how far it got.
13
+ OK = 0
14
+ # A caller bug, never a data verdict.
15
+ ERR_CONTRACT = -1
16
+ # The data is structurally broken; the failure says where.
17
+ ERR_STRUCTURE = -2
18
+ # The arena cannot hold what one row needs.
19
+ ERR_ARENA = -3
20
+ # The cell table (or the header's name table) cannot hold one row.
21
+ ERR_CELLS = -4
22
+
23
+ # The flag in the top bit of a span's length. On a cell-table entry: the cell has ""
24
+ # inside and has to be unescaped to be read. On a text value or a header name: the
25
+ # bytes are in the arena rather than in the input.
26
+ SPAN_FLAG = 1 << 31
27
+ # A span's length without its flag.
28
+ SPAN_LENGTH = SPAN_FLAG - 1
29
+
30
+ # Bytes in one span.
31
+ SPAN_BYTES = 8
32
+ # Filled: rows, consumed, arena_used, needed, then Failure (code, line, record, byte,
33
+ # expected, found).
34
+ FILLED_BYTES = 64
35
+ FILLED = "Q<4L<2Q<2L<2".freeze
36
+ # Buffers: seven pointer and size pairs (window, arena, cells, row, strings, table,
37
+ # kinds), as every workbook call takes them. A delimited fill reads the arena and the
38
+ # cell table from it and nothing else.
39
+ BUFFERS = "Q<14".freeze
40
+ BUFFERS_BYTES = 112
41
+
42
+ ARENA_BYTES = 4096
43
+ NAMES = 64
44
+
45
+ CONTRACT = "hypertabular: libhypertabular reported a contract violation — a binding bug, " \
46
+ "please report it".freeze
47
+
48
+ # What the last call wrote: rows (or header names), input bytes finished with, arena
49
+ # bytes written, and — after ERR_STRUCTURE — the failure as
50
+ # [code, line, record, byte, expected, found].
51
+ attr_reader :rows, :consumed, :arena_used, :failure
52
+
53
+ # The plan's arrays (Columns), and the cell-table entries one row takes: the widest
54
+ # ordinal the plan reads, plus two — or more, if the core asked for more.
55
+ attr_reader :columns, :per_row
56
+
57
+ # The loaded core's version word, major << 16 | minor << 8 | patch.
58
+ def self.version
59
+ Runtime.function(:hypertabular_version).call
60
+ end
61
+
62
+ # +dialect+ is the four bytes of a RawDialect; +specs+ one packed ColumnSpec per plan
63
+ # column and +sizes+ the bytes one value of each takes; +per_row+ the cell-table
64
+ # entries one row takes. Nil when the core refuses the dialect.
65
+ def self.start(dialect, specs, sizes, batch_rows, per_row)
66
+ reader = new(specs, sizes, batch_rows, per_row)
67
+ reader.send(:init, dialect) ? reader : nil
68
+ end
69
+
70
+ def initialize(specs, sizes, batch_rows, per_row)
71
+ @header = Runtime.function(:hypertabular_delimited_header)
72
+ @fill = Runtime.function(:hypertabular_delimited_fill)
73
+ @state = Runtime.buffer(Runtime.function(:hypertabular_delimited_state_size).call)
74
+ @batch_rows = batch_rows
75
+ @columns = Columns.new(specs, sizes, batch_rows)
76
+ @per_row = per_row
77
+ @cramped = false
78
+ @cells_cap = per_row * batch_rows
79
+ @cells = Runtime.buffer(SPAN_BYTES * @cells_cap)
80
+ @arena_cap = ARENA_BYTES
81
+ @arena = Runtime.buffer(@arena_cap)
82
+ @out = Runtime.buffer(FILLED_BYTES)
83
+ @buffers = Runtime.buffer(BUFFERS_BYTES)
84
+ @rows = @consumed = @arena_used = 0
85
+ end
86
+
87
+ # Names the String the calls that follow read: +start+ and +offset+ below index its
88
+ # bytes. It is held where it is, not copied, and must not be modified until another
89
+ # one is attached.
90
+ def attach(input)
91
+ @pin = Runtime.pin(input)
92
+ @base = @pin.to_i
93
+ end
94
+
95
+ # Reads the next record of the attached input, +length+ bytes from +start+, as a
96
+ # header. OK or ERR_STRUCTURE; +rows+ is then the number of names.
97
+ def header(start, length, last)
98
+ @names_cap ||= NAMES
99
+ @names ||= Runtime.buffer(SPAN_BYTES * @names_cap)
100
+ loop do
101
+ code = @header.call(@state, @base + start, length, last ? 1 : 0,
102
+ @names, @names_cap, @arena, @arena_cap, @out)
103
+ needed = finished
104
+ case code
105
+ when OK, ERR_STRUCTURE then return code
106
+ when ERR_CELLS
107
+ @names_cap = needed
108
+ @names = Runtime.buffer(SPAN_BYTES * @names_cap)
109
+ when ERR_ARENA then grow_arena(needed)
110
+ else raise CONTRACT
111
+ end
112
+ end
113
+ end
114
+
115
+ # The header's names as the core located them: offset and flagged length, a pair per
116
+ # name, offsets relative to the +start+ the header was read at (or into the arena,
117
+ # for a flagged one).
118
+ def names
119
+ @names[0, SPAN_BYTES * @rows].unpack("L<*")
120
+ end
121
+
122
+ # Fills every column from the attached input, +length+ bytes from +start+ — the one
123
+ # native call a batch makes. OK or ERR_STRUCTURE.
124
+ #
125
+ # The core ends a batch early when the arena fills, so a batch that came back short
126
+ # with the arena half used or more has the next one start with it doubled: escaped
127
+ # text costs a few batches, not one per row.
128
+ def fill(start, length, last)
129
+ grow_arena(@arena_cap * 2) if @cramped
130
+ @cramped = false
131
+ loop do
132
+ code = @fill.call(@state, @base + start, length, last ? 1 : 0,
133
+ @columns.specs, @columns.table, @columns.count, @batch_rows,
134
+ buffers, @out)
135
+ needed = finished
136
+ case code
137
+ when OK
138
+ @cramped = @rows.positive? && @rows < @batch_rows && @arena_used * 2 >= @arena_cap
139
+ return code
140
+ when ERR_STRUCTURE then return code
141
+ when ERR_CELLS
142
+ @per_row = [@per_row, needed].max
143
+ @cells_cap = @per_row * @batch_rows
144
+ @cells = Runtime.buffer(SPAN_BYTES * @cells_cap)
145
+ when ERR_ARENA then grow_arena(needed)
146
+ else raise CONTRACT
147
+ end
148
+ end
149
+ end
150
+
151
+ # The cell table for the batch in hand — #per_row spans a row — copied out.
152
+ def cells
153
+ @cells[0, SPAN_BYTES * @per_row * @rows]
154
+ end
155
+
156
+ # The arena as the last call left it: the unescaped text of every flagged span.
157
+ def arena
158
+ @arena[0, @arena_used]
159
+ end
160
+
161
+ # Records finished so far — the header and skipped blank lines included.
162
+ def records
163
+ @state[8, 8].unpack1("Q<")
164
+ end
165
+
166
+ # One-based line number of the next unread byte.
167
+ def line
168
+ @state[4, 4].unpack1("L<")
169
+ end
170
+
171
+ # Absolute byte offset of the next unread byte.
172
+ def offset
173
+ @state[16, 8].unpack1("Q<")
174
+ end
175
+
176
+ private
177
+
178
+ def init(dialect)
179
+ raw = Runtime.buffer(4)
180
+ raw[0, 4] = dialect
181
+ Runtime.function(:hypertabular_delimited_init).call(@state, raw) == OK
182
+ end
183
+
184
+ # The arena and the cell table as the Buffers block a fill takes them — packed again
185
+ # for every call, since either may have been grown since the last.
186
+ def buffers
187
+ @buffers[0, BUFFERS_BYTES] = [0, 0, @arena.to_i, @arena_cap, @cells.to_i, @cells_cap, 0, 0, 0, 0, 0, 0, 0, 0]
188
+ .pack(BUFFERS)
189
+ @buffers
190
+ end
191
+
192
+ # Reads what the call wrote to its Filled block; returns `needed`.
193
+ def finished
194
+ @rows, @consumed, @arena_used, needed, *@failure = @out[0, FILLED_BYTES].unpack(FILLED)
195
+ needed
196
+ end
197
+
198
+ def grow_arena(needed)
199
+ @arena_cap = [needed, @arena_cap * 2].max
200
+ @arena = Runtime.buffer(@arena_cap)
201
+ end
202
+ end
203
+ end
204
+ end