herringbone 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +209 -0
- data/bin/herringbone +54 -0
- data/lib/herringbone/active_record.rb +131 -0
- data/lib/herringbone/byte_values.rb +144 -0
- data/lib/herringbone/codecs/lz4.rb +305 -0
- data/lib/herringbone/codecs/snappy.rb +375 -0
- data/lib/herringbone/compression.rb +75 -0
- data/lib/herringbone/encodings/delta.rb +153 -0
- data/lib/herringbone/encodings/plain.rb +87 -0
- data/lib/herringbone/encodings/rle.rb +203 -0
- data/lib/herringbone/format.rb +309 -0
- data/lib/herringbone/io_buffer_support.rb +25 -0
- data/lib/herringbone/reader.rb +469 -0
- data/lib/herringbone/schema.rb +551 -0
- data/lib/herringbone/thrift.rb +334 -0
- data/lib/herringbone/types.rb +454 -0
- data/lib/herringbone/version.rb +5 -0
- data/lib/herringbone/writer.rb +634 -0
- data/lib/herringbone.rb +45 -0
- metadata +104 -0
checksums.yaml
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
---
|
|
2
|
+
SHA256:
|
|
3
|
+
metadata.gz: 1148482072b7346372f0ce7d404c237359dbda86443c0b12dd8200937f6336cc
|
|
4
|
+
data.tar.gz: 9c1abff1ff39a715b0cb32f9bb9385d3642cdf6a0f2f5f6c45e741f25832ca21
|
|
5
|
+
SHA512:
|
|
6
|
+
metadata.gz: 84e2d8192f4a0e9e15430cf8df20d1dd070028dc3561973d639ca887ff82c8673e7864063419a63b05649526ccf6dd0f2a6eb9d291067fd5c674a2ba76ec9771
|
|
7
|
+
data.tar.gz: 532f1deff266b6f4867f47c6e80b200cfa9bcdb1a5d45442ef8e40b5f25247087fa2a85975822c3f572c6a64575ae6b4f87bbd72a13df1143db2a0c82fde856b
|
data/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Julik Tarkhanov
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
data/README.md
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
# Herringbone
|
|
2
|
+
|
|
3
|
+
A pure-Ruby reader and writer for [Apache Parquet](https://parquet.apache.org/) files.
|
|
4
|
+
|
|
5
|
+
- No native extensions, no Thrift gem: the Thrift compact protocol, all encodings and the
|
|
6
|
+
Snappy and LZ4 codecs are implemented in Ruby
|
|
7
|
+
- Full nesting support (structs, lists, maps, any depth) via Dremel record shredding/assembly
|
|
8
|
+
- Reads files from parquet-mr, Arrow, Spark, Impala, DuckDB, Rust writers etc.
|
|
9
|
+
- Ruby 3.0+
|
|
10
|
+
|
|
11
|
+
## Installation
|
|
12
|
+
|
|
13
|
+
```ruby
|
|
14
|
+
gem "herringbone"
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Snappy and LZ4 are implemented in Ruby, GZIP uses `zlib`, and ZSTD and Brotli come from the
|
|
18
|
+
`zstd-ruby` and `brotli` gems (runtime dependencies, along with `bigdecimal`).
|
|
19
|
+
|
|
20
|
+
## Reading
|
|
21
|
+
|
|
22
|
+
```ruby
|
|
23
|
+
require "herringbone"
|
|
24
|
+
|
|
25
|
+
Herringbone::Reader.open("data.parquet") do |reader|
|
|
26
|
+
reader.schema # => #<Herringbone::Schema ...>
|
|
27
|
+
reader.num_rows
|
|
28
|
+
|
|
29
|
+
reader.each_row do |row| # Hash with String keys, nested values as Hash/Array
|
|
30
|
+
p row
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
reader.each_row(columns: ["id", "name"]) { |row| ... } # projection
|
|
34
|
+
reader.column("name") # => all values of one top-level field
|
|
35
|
+
reader.read_row_group(0) # => { "id" => [...], "name" => [...] }
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
Herringbone.read("data.parquet") # => Array of row Hashes
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Writing
|
|
42
|
+
|
|
43
|
+
```ruby
|
|
44
|
+
schema = Herringbone::Schema.define do
|
|
45
|
+
int64 :id, null: false
|
|
46
|
+
string :name
|
|
47
|
+
enum :status, values: %w[pending paid shipped]
|
|
48
|
+
list :tags, :string
|
|
49
|
+
map :scores, :string, :double
|
|
50
|
+
struct :address do
|
|
51
|
+
string :city
|
|
52
|
+
string :zip
|
|
53
|
+
end
|
|
54
|
+
decimal :price, precision: 12, scale: 2
|
|
55
|
+
json :payload
|
|
56
|
+
timestamp :created_at
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
Herringbone::Writer.open("out.parquet", schema) do |w|
|
|
60
|
+
w << { "id" => 1, "name" => "Anna", "status" => "paid", "tags" => ["a", "b"],
|
|
61
|
+
"scores" => { "x" => 1.5 }, "address" => { "city" => "Amsterdam" },
|
|
62
|
+
"price" => BigDecimal("9.99"), "payload" => { "any" => ["json"] }, "created_at" => Time.now }
|
|
63
|
+
w << { id: 2, status: :pending } # Symbol keys and values work; missing keys are nulls
|
|
64
|
+
w << [3, "Bo", nil, nil, nil, nil, nil, nil, nil, nil] # Arrays in schema order
|
|
65
|
+
w << order # anything with #attributes (ActiveRecord) or #to_h (Struct, Data)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Or with an inferred schema, optionally overriding some columns
|
|
69
|
+
Herringbone.write("out.parquet", rows, schema: Herringbone::Schema.infer(rows, types: { payload: :json }))
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Schemas can also be given as a Hash, anywhere a schema is accepted:
|
|
73
|
+
|
|
74
|
+
```ruby
|
|
75
|
+
Herringbone::Schema.define(
|
|
76
|
+
id: { type: :int64, null: false },
|
|
77
|
+
name: :string,
|
|
78
|
+
tags: [:string], # list of strings
|
|
79
|
+
address: { city: :string, zip: :string }, # struct
|
|
80
|
+
price: { type: :decimal, precision: 12, scale: 2 },
|
|
81
|
+
scores: { type: :map, key: :string, value: :double }
|
|
82
|
+
)
|
|
83
|
+
Herringbone::Writer.open("out.parquet", { id: :int64, name: :string }) { |w| w << [1, "x"] }
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Column types in the DSL: `boolean int8 int16 int32 int64 uint8 uint16 uint32 uint64 float double
|
|
87
|
+
float16 string binary json bson enum uuid date int96 time timestamp decimal fixed`, plus
|
|
88
|
+
`struct`, `list` and `map`. Fields are nullable unless `null: false` is given.
|
|
89
|
+
List elements are nullable unless `element_null: false`; map values unless `value_null: false`.
|
|
90
|
+
Nested lists: `list :matrix do list :element, :double end`.
|
|
91
|
+
|
|
92
|
+
`enum` is a string column. `values:` restricts what may be written, and also takes a Rails-style
|
|
93
|
+
Hash (`values: Order.statuses`), in which case both labels and stored integers are accepted and
|
|
94
|
+
the label is written. `parquet_enum: true` adds the Parquet ENUM annotation (pyarrow and pandas
|
|
95
|
+
read such columns as binary, which is why it is off by default).
|
|
96
|
+
|
|
97
|
+
Columns accept the values Ruby and Rails code usually has at hand:
|
|
98
|
+
|
|
99
|
+
| column | accepts |
|
|
100
|
+
|---|---|
|
|
101
|
+
| `date` | `Date`, `Time`/`DateTime` (their date), `"2024-05-01"` |
|
|
102
|
+
| `timestamp` | `Time`, `DateTime`, `ActiveSupport::TimeWithZone`, `Date` (midnight UTC), ISO-8601 strings, Integers in the column's unit |
|
|
103
|
+
| `time` | `Time` (its time of day, as Rails returns for `time` columns), `"13:45:30.25"`, Integers |
|
|
104
|
+
| `json` | Strings as-is, anything else through `JSON.generate` |
|
|
105
|
+
| `string`, `enum` | Strings, Symbols, anything with `to_s` |
|
|
106
|
+
| `boolean` | `true`/`false`, `1`/`0`, `"t"`/`"f"`, `"true"`/`"false"`, `"yes"`/`"no"` |
|
|
107
|
+
| integers | Integers, whole-number Floats/BigDecimals/Rationals, numeric Strings; out-of-range values raise |
|
|
108
|
+
| `decimal` | `BigDecimal`, Integer, Rational, Float, numeric Strings |
|
|
109
|
+
| `uuid` | Strings with or without dashes, or 16 raw bytes |
|
|
110
|
+
|
|
111
|
+
Values that don't fit raise `Herringbone::EncodeError` naming the row number and column path; the
|
|
112
|
+
failed row is discarded and the writer can carry on.
|
|
113
|
+
|
|
114
|
+
When given a path, the writer writes to a temporary file next to it and renames it into place on
|
|
115
|
+
close. If the block raises (or `#abort` is called), the temporary file is removed and any
|
|
116
|
+
existing file at the path is left untouched. An IO (`File`, `StringIO`, a socket...) can be
|
|
117
|
+
given instead of a path.
|
|
118
|
+
|
|
119
|
+
Writer options:
|
|
120
|
+
|
|
121
|
+
| option | default | |
|
|
122
|
+
|---|---|---|
|
|
123
|
+
| `compression` | `:zstd` | `:none`, `:snappy`, `:gzip`, `:lz4` (LZ4_RAW), `:lz4_hadoop`, `:zstd`, `:brotli` |
|
|
124
|
+
| `row_group_bytes` | 16MB | flush a row group once the buffered values take about this much memory; bounds memory use. Low-cardinality string columns are dictionary-encoded as rows arrive and other strings are packed into byte buffers, so a 15-column table exported in 16MB groups peaks around 290 MB RSS (420 MB with 64MB groups) |
|
|
125
|
+
| `row_group_size` | none | also flush after this many rows |
|
|
126
|
+
| `page_size` | 1MB | approximate data page size |
|
|
127
|
+
| `data_page_version` | `1` | `1` or `2` |
|
|
128
|
+
| `dictionary` | `true` | `false`, or an Array of column paths to dictionary-encode |
|
|
129
|
+
| `encodings` | `{}` | e.g. `{ "id" => :delta_binary_packed, "x" => :byte_stream_split }` |
|
|
130
|
+
| `statistics` | `true` | write min/max/null_count |
|
|
131
|
+
| `metadata` | `{}` | footer key/value metadata |
|
|
132
|
+
|
|
133
|
+
## Exporting ActiveRecord models
|
|
134
|
+
|
|
135
|
+
`Herringbone::Schema.from_active_record` builds a schema from a model's columns, so that
|
|
136
|
+
`record.attributes` can be written as-is. Rails is not a dependency: it only calls
|
|
137
|
+
`columns`, `primary_key` and `defined_enums` on the model.
|
|
138
|
+
|
|
139
|
+
```ruby
|
|
140
|
+
schema = Herringbone::Schema.from_active_record(Order)
|
|
141
|
+
Herringbone::Writer.open("orders.parquet", schema) do |w|
|
|
142
|
+
Order.find_each { |order| w << order.attributes }
|
|
143
|
+
end
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Options: `only:` and `except:` take attribute names; `enums: :enum` writes enum attributes with
|
|
147
|
+
the Parquet ENUM annotation instead of as plain strings (the default, `enums: :string`). Enum
|
|
148
|
+
attributes are written as their labels, and the writer rejects values outside the enum.
|
|
149
|
+
|
|
150
|
+
| Column | Parquet |
|
|
151
|
+
|---|---|
|
|
152
|
+
| `integer` | `int16`/`int32`/`int64` by SQL type (`smallint`, `integer`, `bigint`...) or limit; `uintN` for `unsigned`; primary keys are always `int64` |
|
|
153
|
+
| `float` | `double` (`float` for Postgres `float4`) |
|
|
154
|
+
| `decimal` | `decimal(precision, scale)`; `decimal(38, 9)` when the column has no precision |
|
|
155
|
+
| `boolean`, `date`, `binary`, `uuid` | same |
|
|
156
|
+
| `json`, `jsonb` | `json` |
|
|
157
|
+
| `datetime`, `timestamp`, `timestamptz` | `timestamp` (microseconds, UTC) |
|
|
158
|
+
| `time` | `time` (microseconds) |
|
|
159
|
+
| `hstore` | `map` of `string` to `string` |
|
|
160
|
+
| enum attributes | `string` (or `enum`) |
|
|
161
|
+
| `string`, `text`, `citext`, anything else | `string` |
|
|
162
|
+
| Postgres array columns | `list` of the element type |
|
|
163
|
+
|
|
164
|
+
Columns declared `NOT NULL` (and primary keys) are required, all others nullable.
|
|
165
|
+
Column order follows `Model.columns`.
|
|
166
|
+
|
|
167
|
+
## Type mapping
|
|
168
|
+
|
|
169
|
+
| Parquet | Ruby |
|
|
170
|
+
|---|---|
|
|
171
|
+
| BOOLEAN | `true`/`false` |
|
|
172
|
+
| INT32/INT64 (incl. signed/unsigned INTEGER) | `Integer` |
|
|
173
|
+
| FLOAT, DOUBLE, FLOAT16 | `Float` |
|
|
174
|
+
| STRING, ENUM, JSON | `String` (UTF-8) |
|
|
175
|
+
| BYTE_ARRAY, FIXED_LEN_BYTE_ARRAY, BSON | `String` (binary) |
|
|
176
|
+
| DATE | `Date` (proleptic Gregorian) |
|
|
177
|
+
| TIMESTAMP, INT96 | `Time` (UTC) |
|
|
178
|
+
| TIME | `Integer` in the column's unit since midnight |
|
|
179
|
+
| DECIMAL | `BigDecimal` |
|
|
180
|
+
| UUID | `String` like `"0f1e2d3c-..."` |
|
|
181
|
+
| struct / list / map | `Hash` / `Array` / `Hash` |
|
|
182
|
+
|
|
183
|
+
## Supported format features
|
|
184
|
+
|
|
185
|
+
- Encodings (read and write): PLAIN, PLAIN_DICTIONARY/RLE_DICTIONARY, RLE, DELTA_BINARY_PACKED,
|
|
186
|
+
DELTA_LENGTH_BYTE_ARRAY, DELTA_BYTE_ARRAY, BYTE_STREAM_SPLIT; legacy BIT_PACKED levels (read)
|
|
187
|
+
- Data page v1 and v2, dictionary pages, page CRCs (written)
|
|
188
|
+
- Legacy list and map layouts per the Parquet backward-compatibility rules
|
|
189
|
+
- Not supported: encryption, column chunks in external files, bloom filters and page indexes
|
|
190
|
+
(ignored when reading, not written)
|
|
191
|
+
|
|
192
|
+
## Command line
|
|
193
|
+
|
|
194
|
+
```
|
|
195
|
+
bin/herringbone schema FILE
|
|
196
|
+
bin/herringbone meta FILE
|
|
197
|
+
bin/herringbone cat FILE [N]
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
## Development
|
|
201
|
+
|
|
202
|
+
```
|
|
203
|
+
bundle install
|
|
204
|
+
bundle exec rake test
|
|
205
|
+
HERRINGBONE_PYTHON=/path/to/python-with-pyarrow bundle exec rake test # also run pyarrow interop tests
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
`test/fixtures/parquet-testing` holds files from [apache/parquet-testing](https://github.com/apache/parquet-testing)
|
|
209
|
+
(Apache-2.0); expectations for them were generated with pyarrow, see `test/fixtures/generate_expectations.py`.
|
data/bin/herringbone
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
# Usage:
|
|
5
|
+
# herringbone schema FILE print the schema
|
|
6
|
+
# herringbone meta FILE print row groups, codecs and encodings
|
|
7
|
+
# herringbone cat FILE [N] print rows as JSON lines (optionally only the first N)
|
|
8
|
+
$LOAD_PATH.unshift File.expand_path("../lib", __dir__)
|
|
9
|
+
require "herringbone"
|
|
10
|
+
require "json"
|
|
11
|
+
|
|
12
|
+
def jsonable(v)
|
|
13
|
+
case v
|
|
14
|
+
when Hash then v.to_h { |k, x| [k.to_s, jsonable(x)] }
|
|
15
|
+
when Array then v.map { |x| jsonable(x) }
|
|
16
|
+
when String then v.encoding == Encoding::BINARY && !v.ascii_only? ? v.unpack1("H*") : v
|
|
17
|
+
when Float then v.finite? ? v : v.to_s
|
|
18
|
+
when Time then v.iso8601(9)
|
|
19
|
+
when Date then v.iso8601
|
|
20
|
+
when Integer, true, false, nil then v
|
|
21
|
+
else v.to_s
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
command, path, limit = ARGV
|
|
26
|
+
unless %w[schema meta cat].include?(command) && path
|
|
27
|
+
warn "usage: herringbone (schema|meta|cat) FILE [N]"
|
|
28
|
+
exit 1
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
Herringbone::Reader.open(path) do |r|
|
|
32
|
+
case command
|
|
33
|
+
when "schema"
|
|
34
|
+
puts r.schema.inspect
|
|
35
|
+
when "meta"
|
|
36
|
+
puts "rows: #{r.num_rows}, row groups: #{r.num_row_groups}, created by: #{r.created_by}"
|
|
37
|
+
r.key_value_metadata.each { |k, v| puts " #{k} = #{v.to_s[0, 80]}" }
|
|
38
|
+
r.row_groups.each_with_index do |rg, i|
|
|
39
|
+
puts "row group #{i}: #{rg.num_rows} rows"
|
|
40
|
+
rg.columns.each do |cc|
|
|
41
|
+
m = cc.meta_data
|
|
42
|
+
encodings = m.encodings.map { |e| Herringbone::Format::Encoding::NAMES[e] }.join(",")
|
|
43
|
+
puts " #{m.path_in_schema.join(".")}: #{Herringbone::Format::Codec::NAMES[m.codec]} #{encodings} " \
|
|
44
|
+
"#{m.total_compressed_size}/#{m.total_uncompressed_size} bytes"
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
when "cat"
|
|
48
|
+
n = limit&.to_i
|
|
49
|
+
r.each_row.with_index do |row, i|
|
|
50
|
+
break if n && i >= n
|
|
51
|
+
puts JSON.generate(jsonable(row))
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Schema
|
|
5
|
+
# Builds a schema from an ActiveRecord model, so that the Hashes returned by
|
|
6
|
+
# +record.attributes+ can be written directly:
|
|
7
|
+
#
|
|
8
|
+
# schema = Herringbone::Schema.from_active_record(Order)
|
|
9
|
+
# Herringbone::Writer.open("orders.parquet", schema) do |w|
|
|
10
|
+
# Order.find_each { |order| w << order.attributes }
|
|
11
|
+
# end
|
|
12
|
+
#
|
|
13
|
+
# ActiveRecord is not required: this only uses what a model class exposes
|
|
14
|
+
# (+columns+, +primary_key+ and, when present, +defined_enums+).
|
|
15
|
+
#
|
|
16
|
+
# only: attribute names to include (Strings or Symbols)
|
|
17
|
+
# except: attribute names to leave out
|
|
18
|
+
# enums: :string (default) writes enum attributes as STRING columns, :enum uses the
|
|
19
|
+
# Parquet ENUM annotation (which pyarrow/pandas read as binary). Either way the
|
|
20
|
+
# enum labels are written, and the writer rejects values outside the enum mapping
|
|
21
|
+
# (stored values such as 0/1 are accepted and written as their labels).
|
|
22
|
+
def self.from_active_record(model, only: nil, except: nil, enums: :string)
|
|
23
|
+
raise ArgumentError, "enums: must be :string or :enum" unless %i[string enum].include?(enums)
|
|
24
|
+
only = only && Array(only).map(&:to_s)
|
|
25
|
+
except = Array(except).map(&:to_s)
|
|
26
|
+
defined_enums = model.respond_to?(:defined_enums) ? model.defined_enums.to_h { |k, v| [k.to_s, v] } : {}
|
|
27
|
+
primary_keys = Array(model.respond_to?(:primary_key) ? model.primary_key : nil).map(&:to_s)
|
|
28
|
+
|
|
29
|
+
builder = Builder.new
|
|
30
|
+
model.columns.each do |column|
|
|
31
|
+
name = column.name.to_s
|
|
32
|
+
next if only && !only.include?(name)
|
|
33
|
+
next if except.include?(name)
|
|
34
|
+
primary = primary_keys.include?(name)
|
|
35
|
+
nullable = column.null != false && !primary
|
|
36
|
+
if (mapping = defined_enums[name])
|
|
37
|
+
builder.enum(name, values: mapping.to_h, parquet_enum: enums == :enum, null: nullable)
|
|
38
|
+
else
|
|
39
|
+
ActiveRecordMapping.add_column(builder, column, nullable, primary)
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
raise ArgumentError, "No columns selected from #{model}" if builder.nodes.empty?
|
|
43
|
+
new(Node.new(name: "schema", repetition: :required, children: builder.nodes))
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# Maps ActiveRecord column metadata to schema DSL types.
|
|
47
|
+
#
|
|
48
|
+
# integer by sql_type (smallint -> int16, bigint -> int64, ...) or by limit
|
|
49
|
+
# (1 -> int8, 2 -> int16, 8 -> int64, otherwise int32); "unsigned" -> uintN.
|
|
50
|
+
# Integer primary keys are always int64.
|
|
51
|
+
# float double; float (32-bit) only for sql_type "float4"
|
|
52
|
+
# decimal decimal(precision, scale); decimal(38, 9) when the column has no precision
|
|
53
|
+
# boolean, date, binary, uuid, json map to the same-named types (jsonb -> json)
|
|
54
|
+
# datetime, timestamp, timestamptz -> timestamp(micros, UTC); time -> time(micros)
|
|
55
|
+
# hstore map<string, string>
|
|
56
|
+
# anything else (string, text, citext, inet, cidr, macaddr, ...) -> string
|
|
57
|
+
# Postgres arrays (column.array, or a sql_type ending in "[]") -> list of the element type
|
|
58
|
+
module ActiveRecordMapping
|
|
59
|
+
module_function
|
|
60
|
+
|
|
61
|
+
DEFAULT_DECIMAL_PRECISION = 38
|
|
62
|
+
DEFAULT_DECIMAL_SCALE = 9
|
|
63
|
+
|
|
64
|
+
INTEGER_SQL_TYPES = {
|
|
65
|
+
"tinyint" => 8, "int1" => 8,
|
|
66
|
+
"smallint" => 16, "int2" => 16, "smallserial" => 16, "serial2" => 16,
|
|
67
|
+
"mediumint" => 32, "int4" => 32, "serial" => 32, "serial4" => 32,
|
|
68
|
+
"bigint" => 64, "int8" => 64, "bigserial" => 64, "serial8" => 64
|
|
69
|
+
}.freeze
|
|
70
|
+
|
|
71
|
+
def add_column(builder, column, nullable, primary)
|
|
72
|
+
sql_type = column.respond_to?(:sql_type) ? column.sql_type.to_s.downcase : ""
|
|
73
|
+
array = (column.respond_to?(:array) && column.array) || sql_type.end_with?("[]")
|
|
74
|
+
sql_type = sql_type.sub(/(\[\d*\])+\z/, "")
|
|
75
|
+
type = column.type&.to_sym
|
|
76
|
+
|
|
77
|
+
if type == :hstore
|
|
78
|
+
return builder.map(column.name, :string, :string, null: nullable) unless array
|
|
79
|
+
return builder.list(column.name, null: nullable) { map :element, :string, :string }
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
dsl_type, opts = scalar_type(column, type, sql_type, primary)
|
|
83
|
+
if array
|
|
84
|
+
builder.list(column.name, dsl_type, null: nullable, **opts)
|
|
85
|
+
else
|
|
86
|
+
builder.column(column.name, dsl_type, null: nullable, **opts)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def scalar_type(column, type, sql_type, primary)
|
|
91
|
+
case type
|
|
92
|
+
when :integer, :bigint then [integer_type(column, sql_type, primary), {}]
|
|
93
|
+
when :float then [sql_type == "float4" ? :float : :double, {}]
|
|
94
|
+
when :decimal, :money
|
|
95
|
+
precision = column.respond_to?(:precision) ? column.precision : nil
|
|
96
|
+
scale = column.respond_to?(:scale) ? column.scale : nil
|
|
97
|
+
if precision.nil?
|
|
98
|
+
precision = DEFAULT_DECIMAL_PRECISION
|
|
99
|
+
scale ||= DEFAULT_DECIMAL_SCALE
|
|
100
|
+
end
|
|
101
|
+
[:decimal, { precision: precision, scale: scale || 0 }]
|
|
102
|
+
when :boolean then [:boolean, {}]
|
|
103
|
+
when :binary then [:binary, {}]
|
|
104
|
+
when :date then [:date, {}]
|
|
105
|
+
when :datetime, :timestamp, :timestamptz then [:timestamp, { unit: :micros, utc: true }]
|
|
106
|
+
when :time then [:time, { unit: :micros }]
|
|
107
|
+
when :json, :jsonb then [:json, {}]
|
|
108
|
+
when :uuid then [:uuid, {}]
|
|
109
|
+
else [:string, {}]
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# Named SQL types (smallint, bigint, ...) decide the width; a generic "integer"/"int"
|
|
114
|
+
# uses the column limit in bytes, since e.g. SQLite reports `t.integer limit: 2` as "integer(2)".
|
|
115
|
+
def integer_type(column, sql_type, primary)
|
|
116
|
+
bits = INTEGER_SQL_TYPES[sql_type[/\A[a-z0-9]+/]]
|
|
117
|
+
bits ||= case (column.respond_to?(:limit) ? column.limit : nil)
|
|
118
|
+
when 1 then 8
|
|
119
|
+
when 2 then 16
|
|
120
|
+
when 5..8 then 64
|
|
121
|
+
else 32
|
|
122
|
+
end
|
|
123
|
+
# Row ids can exceed 32 bits even where the column is declared "integer" (SQLite rowids)
|
|
124
|
+
bits = 64 if primary
|
|
125
|
+
unsigned = sql_type.include?("unsigned")
|
|
126
|
+
return :"uint#{bits}" if unsigned
|
|
127
|
+
{ 8 => :int8, 16 => :int16, 32 => :int32, 64 => :int64 }.fetch(bits)
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
end
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
# Compact buffer for the values of a BYTE_ARRAY or FIXED_LEN_BYTE_ARRAY column while a row
|
|
5
|
+
# group is being collected. Instead of holding on to one Ruby String per value, it either
|
|
6
|
+
#
|
|
7
|
+
# * dictionary-encodes values as they arrive (keeping each distinct value once, plus an Integer
|
|
8
|
+
# index per value), which suits low-cardinality columns such as enums and statuses, or
|
|
9
|
+
# * appends the raw bytes to one binary String (plus a length per value for BYTE_ARRAY).
|
|
10
|
+
#
|
|
11
|
+
# It starts out in dictionary mode (when allowed) and switches to raw bytes once the dictionary
|
|
12
|
+
# grows too large or too many values turn out to be distinct. Strings are only rebuilt when the
|
|
13
|
+
# row group is flushed, one column at a time.
|
|
14
|
+
class ByteValues
|
|
15
|
+
# Values in the dictionary may take this many bytes before switching to raw bytes
|
|
16
|
+
MAX_DICTIONARY_BYTES = 1024 * 1024
|
|
17
|
+
# After this many values, give up on the dictionary if more than half of them are distinct
|
|
18
|
+
CARDINALITY_CHECK_AT = 4096
|
|
19
|
+
|
|
20
|
+
APPEND_AS_BYTES = "".respond_to?(:append_as_bytes)
|
|
21
|
+
|
|
22
|
+
def initialize(width: nil, dictionary: true)
|
|
23
|
+
@width = width
|
|
24
|
+
@bytes = String.new(encoding: Encoding::BINARY)
|
|
25
|
+
@lengths = width ? nil : []
|
|
26
|
+
@count = 0
|
|
27
|
+
if dictionary
|
|
28
|
+
@dictionary = {}
|
|
29
|
+
@dictionary_bytes = 0
|
|
30
|
+
@indices = []
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def size
|
|
35
|
+
@indices ? @indices.size : @count
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def empty? = size.zero?
|
|
39
|
+
|
|
40
|
+
def dictionary? = !@indices.nil?
|
|
41
|
+
|
|
42
|
+
def <<(value)
|
|
43
|
+
if @indices
|
|
44
|
+
index = @dictionary[value]
|
|
45
|
+
if index.nil?
|
|
46
|
+
if @dictionary_bytes + value.bytesize > MAX_DICTIONARY_BYTES
|
|
47
|
+
switch_to_bytes
|
|
48
|
+
return append_bytes(value)
|
|
49
|
+
end
|
|
50
|
+
# Hash#[]= stores a frozen copy of unfrozen String keys
|
|
51
|
+
index = @dictionary.size
|
|
52
|
+
@dictionary[value] = index
|
|
53
|
+
@dictionary_bytes += value.bytesize + 4
|
|
54
|
+
end
|
|
55
|
+
@indices << index
|
|
56
|
+
switch_to_bytes if @indices.size == CARDINALITY_CHECK_AT && @dictionary.size > CARDINALITY_CHECK_AT / 2
|
|
57
|
+
self
|
|
58
|
+
else
|
|
59
|
+
append_bytes(value)
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Removes the last value
|
|
64
|
+
def pop
|
|
65
|
+
truncate(size - 1) unless size.zero?
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Supports the `slice!(n..)` form used to roll back a failed row
|
|
69
|
+
def slice!(range)
|
|
70
|
+
truncate(range.begin)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Approximate memory held, used to size row groups
|
|
74
|
+
def memory_bytes
|
|
75
|
+
if @indices
|
|
76
|
+
# Each distinct value is a String object plus a Hash entry
|
|
77
|
+
@indices.size * 8 + @dictionary_bytes + @dictionary.size * 48
|
|
78
|
+
else
|
|
79
|
+
@bytes.bytesize + (@lengths ? @lengths.size * 8 : 0)
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Returns [:dictionary, values, indices] when the column is worth dictionary-encoding,
|
|
84
|
+
# otherwise [:plain, values]
|
|
85
|
+
def materialize
|
|
86
|
+
if @indices
|
|
87
|
+
keys = @dictionary.keys
|
|
88
|
+
return [:dictionary, keys, @indices] unless keys.size > @indices.size / 2 + 1 && @indices.size > 16
|
|
89
|
+
return [:plain, @indices.map { |i| keys[i] }]
|
|
90
|
+
end
|
|
91
|
+
[:plain, strings]
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
private
|
|
95
|
+
|
|
96
|
+
def append_bytes(value)
|
|
97
|
+
if value.encoding == Encoding::BINARY || value.ascii_only?
|
|
98
|
+
@bytes << value
|
|
99
|
+
elsif APPEND_AS_BYTES
|
|
100
|
+
@bytes.append_as_bytes(value)
|
|
101
|
+
else
|
|
102
|
+
@bytes << value.b
|
|
103
|
+
end
|
|
104
|
+
@lengths << value.bytesize if @lengths
|
|
105
|
+
@count += 1
|
|
106
|
+
self
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def switch_to_bytes
|
|
110
|
+
keys = @dictionary.keys
|
|
111
|
+
indices = @indices
|
|
112
|
+
@indices = @dictionary = nil
|
|
113
|
+
indices.each { |i| append_bytes(keys[i]) }
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def truncate(n)
|
|
117
|
+
if @indices
|
|
118
|
+
@indices.slice!(n..)
|
|
119
|
+
else
|
|
120
|
+
keep = if @lengths
|
|
121
|
+
@lengths.slice!(n..)
|
|
122
|
+
@lengths.sum
|
|
123
|
+
else
|
|
124
|
+
n * @width
|
|
125
|
+
end
|
|
126
|
+
@bytes.slice!(keep..)
|
|
127
|
+
@count = n
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def strings
|
|
132
|
+
if @lengths
|
|
133
|
+
pos = 0
|
|
134
|
+
@lengths.map do |len|
|
|
135
|
+
s = @bytes.byteslice(pos, len)
|
|
136
|
+
pos += len
|
|
137
|
+
s
|
|
138
|
+
end
|
|
139
|
+
else
|
|
140
|
+
Array.new(@count) { |i| @bytes.byteslice(i * @width, @width) }
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|