herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: '08770d40f03a2320d9aa60e6fe63fbfe40ea5ddeddc627bff3be2bf46f431795'
|
|
4
|
+
data.tar.gz: b18fef0b86fd290576d254f338da456ad80dc2600c68eb0c903e6085a024ebe2
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 5c941e4b5f31f278ff819d25d464dab518c46e50d10435c360faf2da5994a9cff891754d00a73d8834aa08f2cd00be0387d9f6c8536459389797ba7daa1a82d0
|
|
7
|
+
data.tar.gz: 4532d47095f0832bf4cebb86ef370882dd7287625d80b9b94f6bc49fffb1ad3317075ecf8c45a1fef3fdb2fa19c3a641c0a6c2c4664bf24f5aa664e5b686e03e
|
data/README.md
CHANGED
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
A pure-Ruby reader and writer for [Apache Parquet](https://parquet.apache.org/) files.
|
|
4
4
|
|
|
5
|
-
- No native extensions
|
|
6
|
-
Snappy and LZ4 codecs are implemented in Ruby
|
|
5
|
+
- No Thrift gem and no native extensions of its own: the Thrift compact protocol, all encodings
|
|
6
|
+
and the Snappy and LZ4 codecs are implemented in Ruby (ZSTD and Brotli use optional gems)
|
|
7
7
|
- Full nesting support (structs, lists, maps, any depth) via Dremel record shredding/assembly
|
|
8
8
|
- Reads files from parquet-mr, Arrow, Spark, Impala, DuckDB, Rust writers etc.
|
|
9
9
|
- Ruby 3.0+
|
|
@@ -12,32 +12,102 @@ A pure-Ruby reader and writer for [Apache Parquet](https://parquet.apache.org/)
|
|
|
12
12
|
|
|
13
13
|
```ruby
|
|
14
14
|
gem "herringbone"
|
|
15
|
+
gem "snappy" # optional: native Snappy, 2-3x faster reads and writes of typical files
|
|
16
|
+
gem "zstd-ruby" # optional: ZSTD (faster writes and smaller files than the default Snappy)
|
|
17
|
+
gem "brotli" # optional: Brotli
|
|
18
|
+
gem "xxhash" # optional: faster bloom filters
|
|
19
|
+
gem "numo-narray-alt" # optional: read(as: :numo)
|
|
15
20
|
```
|
|
16
21
|
|
|
17
|
-
|
|
18
|
-
|
|
22
|
+
The only dependency is `bigdecimal`. Snappy, LZ4 and GZIP always work. Snappy, Parquet's most
|
|
23
|
+
common codec, is pure Ruby unless the `snappy` gem is installed (it needs libsnappy or cmake to
|
|
24
|
+
build); Herringbone then uses it automatically. `Herringbone.codecs` lists
|
|
25
|
+
the codecs this process can use, e.g. `[:none, :snappy, :gzip, :lz4, :lz4_hadoop, :zstd]`. Using
|
|
26
|
+
a missing one raises `Herringbone::MissingCodecError` naming the gem to add: a writer raises it
|
|
27
|
+
before writing anything, a reader when it reaches the first such page (the schema and metadata
|
|
28
|
+
are still readable). LZO is not supported.
|
|
19
29
|
|
|
20
30
|
## Reading
|
|
21
31
|
|
|
32
|
+
Herringbone never opens files by path: readers take a random-access IO (a `File` opened with
|
|
33
|
+
`"rb"`, `StringIO`, `Tempfile`...), which stays open and belongs to the caller.
|
|
34
|
+
|
|
22
35
|
```ruby
|
|
23
36
|
require "herringbone"
|
|
24
37
|
|
|
25
|
-
|
|
26
|
-
reader
|
|
38
|
+
File.open("data.parquet", "rb") do |file|
|
|
39
|
+
reader = Herringbone::Reader.new(file)
|
|
40
|
+
reader.schema
|
|
27
41
|
reader.num_rows
|
|
28
42
|
|
|
29
|
-
reader.each_row
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
reader.each_row(columns: ["id", "name"]) { |row| ... } # projection
|
|
34
|
-
reader.column("name") # => all values of one top-level field
|
|
35
|
-
reader.read_row_group(0) # => { "id" => [...], "name" => [...] }
|
|
43
|
+
reader.each_row { |row| p row } # Hashes with String keys, nested values as Hash/Array
|
|
44
|
+
reader.each_batch(1000) { |rows| ... } # Arrays of up to 1000 rows
|
|
45
|
+
reader.each_batch(10_000, as: :columns) { |batch| batch["amount"].sum } # { "id" => [...], ... }
|
|
46
|
+
reader.read # all rows at once; read(as: :columns) for column Arrays
|
|
36
47
|
end
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Reads stream: pages are read and decoded one at a time per column and rows are assembled in
|
|
51
|
+
batches, so memory depends on the batch and page sizes rather than on the row group size (1M
|
|
52
|
+
rows in a single row group peak at about 60–160 MB RSS growth instead of 660 MB). Column-order
|
|
53
|
+
batches (`as: :columns`) skip building a Hash per row and are 25–30% faster.
|
|
54
|
+
|
|
55
|
+
`Reader.new` takes `keys: :symbol` for Symbol keys (in rows and in structs; map keys stay as
|
|
56
|
+
stored) and `time_zone:` to return timestamps in a zone instead of UTC: a UTC offset (`"+02:00"`,
|
|
57
|
+
or seconds), `"UTC"`, a `TZInfo::Timezone`, or anything responding to `#at` such as `Time.zone`
|
|
58
|
+
in Rails (which yields `ActiveSupport::TimeWithZone`). Zone names like `"Europe/Amsterdam"` work
|
|
59
|
+
when ActiveSupport or TZInfo is loaded. Timestamps stored with `isAdjustedToUTC=false` are
|
|
60
|
+
wall-clock values and stay as they are.
|
|
61
|
+
|
|
62
|
+
### Selecting rows
|
|
37
63
|
|
|
38
|
-
|
|
64
|
+
`each_row`, `each_batch` and `read` take `columns:`, `where:`, `from:` and `limit:`:
|
|
65
|
+
|
|
66
|
+
```ruby
|
|
67
|
+
reader.read(columns: %w[id name]) # projection
|
|
68
|
+
reader.read(where: { user_id: 42 })
|
|
69
|
+
reader.read(where: { status: %w[paid shipped], created_at: 1.week.ago.. }) # IN, Ranges
|
|
70
|
+
reader.read(where: { "address.city" => "Amsterdam", deleted_at: nil }) # struct members, IS NULL
|
|
71
|
+
reader.read(where: { amount: ->(v) { v && v > 100 } }) # any callable
|
|
72
|
+
reader.read(from: 1_000_000, limit: 100) # rows 1,000,000..1,000,099
|
|
39
73
|
```
|
|
40
74
|
|
|
75
|
+
All conditions must hold; filtered columns need not be among `columns:`, and any column not inside
|
|
76
|
+
a list or map can be filtered on. Row groups are skipped using min/max statistics, null counts and
|
|
77
|
+
bloom filters, pages using the page index (written by Herringbone, parquet-mr, Arrow and others),
|
|
78
|
+
and `from:` jumps to its row through the offset index; every row read is then checked, so results
|
|
79
|
+
are exact. On a 1M-row file with 20k-row pages, looking up one `id` takes 0.1 s instead of 6.3 s.
|
|
80
|
+
Filters help most on columns the data is sorted or clustered by. `reader.scan_plan(where: ...)`
|
|
81
|
+
shows which row groups and row ranges a read would touch, without reading them.
|
|
82
|
+
|
|
83
|
+
### Numo arrays
|
|
84
|
+
|
|
85
|
+
`as: :numo` returns (or yields, with `each_batch`) a Hash of column name => Numo array, and takes
|
|
86
|
+
the same `columns:`, `where:`, `from:` and `limit:`. Add `gem "numo-narray-alt"` (or
|
|
87
|
+
`numo-narray`) to your Gemfile: Herringbone requires it on first use and raises
|
|
88
|
+
`Herringbone::UnsupportedError` naming the gem when it is missing.
|
|
89
|
+
|
|
90
|
+
```ruby
|
|
91
|
+
cols = reader.read(as: :numo, columns: %w[id amount]) # { "id" => Numo::Int64, "amount" => Numo::DFloat }
|
|
92
|
+
df = Rover::DataFrame.new(reader.read(as: :numo)) # no conversion needed
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Flat numeric and boolean columns are decoded from the page bytes without a Ruby object per value:
|
|
96
|
+
two numeric columns of 1M uncompressed rows read in about 25 ms, against 50 ms with `as: :columns`.
|
|
97
|
+
|
|
98
|
+
| Parquet | Numo |
|
|
99
|
+
|---|---|
|
|
100
|
+
| INT32, INT64 (also TIME) | `Int32`, `Int64` |
|
|
101
|
+
| INT(8/16) signed; INT(8/16/32/64) unsigned | `Int8`, `Int16`; `UInt8` … `UInt64` |
|
|
102
|
+
| FLOAT, FLOAT16, DOUBLE | `SFloat`, `SFloat`, `DFloat` (nulls are NaN) |
|
|
103
|
+
| integers with nulls | `DFloat` with NaN (exact up to 2**53) |
|
|
104
|
+
| BOOLEAN | `Bit`; with nulls `RObject` of true/false/nil |
|
|
105
|
+
| list of numbers, every row the same length and no nulls | 2-D `[rows, length]` (e.g. embeddings) |
|
|
106
|
+
| strings, binary, decimals, dates, timestamps, UUIDs, structs, maps, other lists | `RObject` of the values `read` returns |
|
|
107
|
+
|
|
108
|
+
Whether a column has nulls (or a list column is rectangular) is decided from the rows read, so
|
|
109
|
+
with `each_batch` it can differ between batches.
|
|
110
|
+
|
|
41
111
|
## Writing
|
|
42
112
|
|
|
43
113
|
```ruby
|
|
@@ -56,38 +126,34 @@ schema = Herringbone::Schema.define do
|
|
|
56
126
|
timestamp :created_at
|
|
57
127
|
end
|
|
58
128
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
129
|
+
File.open("out.parquet", "wb") do |file|
|
|
130
|
+
Herringbone::Writer.open(file, schema) do |w|
|
|
131
|
+
w << { "id" => 1, "name" => "Anna", "status" => "paid", "tags" => ["a", "b"],
|
|
132
|
+
"scores" => { "x" => 1.5 }, "address" => { "city" => "Amsterdam" },
|
|
133
|
+
"price" => BigDecimal("9.99"), "payload" => { "any" => ["json"] }, "created_at" => Time.now }
|
|
134
|
+
w << { id: 2, status: :pending } # Symbol keys and values work; missing keys are nulls
|
|
135
|
+
w << [3, "Bo", nil, nil, nil, nil, nil, nil, nil, nil] # Arrays in schema order
|
|
136
|
+
w << order # anything with #attributes (ActiveRecord) or #to_h (Struct, Data)
|
|
137
|
+
end
|
|
66
138
|
end
|
|
67
|
-
|
|
68
|
-
# Or with an inferred schema, optionally overriding some columns
|
|
69
|
-
Herringbone.write("out.parquet", rows, schema: Herringbone::Schema.infer(rows, types: { payload: :json }))
|
|
70
139
|
```
|
|
71
140
|
|
|
72
|
-
|
|
141
|
+
`Herringbone.write(io, rows)` writes an Enumerable of rows in one go, inferring the schema from the
|
|
142
|
+
first 1000 rows unless `schema:` is given; fields declared in a block replace inferred ones:
|
|
143
|
+
`Herringbone.write(io, rows, schema: Herringbone::Schema.infer(rows) { json :payload })`.
|
|
73
144
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
address: { city: :string, zip: :string }, # struct
|
|
80
|
-
price: { type: :decimal, precision: 12, scale: 2 },
|
|
81
|
-
scores: { type: :map, key: :string, value: :double }
|
|
82
|
-
)
|
|
83
|
-
Herringbone::Writer.open("out.parquet", { id: :int64, name: :string }) { |w| w << [1, "x"] }
|
|
84
|
-
```
|
|
145
|
+
The writer writes to any IO that responds to `#write` (a `File`, `StringIO`, `Tempfile`, socket or
|
|
146
|
+
pipe), sequentially, and never seeks, rewinds or closes it (it does switch it to binary mode). If
|
|
147
|
+
the `Writer.open` block raises (or `#abort` is called), no footer is written and what was written
|
|
148
|
+
so far is left for you to discard. To replace a file only once it is complete, write to a temporary
|
|
149
|
+
file and rename it.
|
|
85
150
|
|
|
86
|
-
Column types
|
|
87
|
-
|
|
88
|
-
`
|
|
89
|
-
|
|
90
|
-
Nested lists: `list :matrix do list :element, :double end`.
|
|
151
|
+
Column types: `boolean int8 int16 int32 int64 uint8 uint16 uint32 uint64 float double float16
|
|
152
|
+
string binary json bson enum uuid date int96 time timestamp decimal fixed`, plus `struct`, `list`
|
|
153
|
+
and `map`; `column :name, :int32` declares one by name. Fields are nullable unless `null: false`
|
|
154
|
+
is given; list elements unless `element_null: false`, map values unless `value_null: false`.
|
|
155
|
+
Nested lists: `list :matrix do list :element, :double end`. `time` and `timestamp` take `unit:`
|
|
156
|
+
(`:millis`, `:micros`, `:nanos`) and `utc:`.
|
|
91
157
|
|
|
92
158
|
`enum` is a string column. `values:` restricts what may be written, and also takes a Rails-style
|
|
93
159
|
Hash (`values: Order.statuses`), in which case both labels and stored integers are accepted and
|
|
@@ -111,41 +177,72 @@ Columns accept the values Ruby and Rails code usually has at hand:
|
|
|
111
177
|
Values that don't fit raise `Herringbone::EncodeError` naming the row number and column path; the
|
|
112
178
|
failed row is discarded and the writer can carry on.
|
|
113
179
|
|
|
114
|
-
When given a path, the writer writes to a temporary file next to it and renames it into place on
|
|
115
|
-
close. If the block raises (or `#abort` is called), the temporary file is removed and any
|
|
116
|
-
existing file at the path is left untouched. An IO (`File`, `StringIO`, a socket...) can be
|
|
117
|
-
given instead of a path.
|
|
118
|
-
|
|
119
180
|
Writer options:
|
|
120
181
|
|
|
121
182
|
| option | default | |
|
|
122
183
|
|---|---|---|
|
|
123
|
-
| `compression` | `:
|
|
124
|
-
| `row_group_bytes` | 16MB | flush a row group once the buffered values take about this much memory
|
|
125
|
-
| `
|
|
126
|
-
| `
|
|
184
|
+
| `compression` | `:snappy` | `:none`, `:snappy`, `:gzip`, `:lz4` (LZ4_RAW), `:lz4_hadoop`, `:zstd`, `:brotli` |
|
|
185
|
+
| `row_group_bytes` | 16MB | flush a row group once the buffered values take about this much memory, which bounds memory use (a 15-column table peaks around 290 MB RSS) |
|
|
186
|
+
| `row_group_rows` | none | also flush after this many rows |
|
|
187
|
+
| `page_bytes` | 1MB | approximate data page size |
|
|
188
|
+
| `page_rows` | `20_000` | at most this many rows per data page |
|
|
127
189
|
| `data_page_version` | `1` | `1` or `2` |
|
|
128
190
|
| `dictionary` | `true` | `false`, or an Array of column paths to dictionary-encode |
|
|
129
191
|
| `encodings` | `{}` | e.g. `{ "id" => :delta_binary_packed, "x" => :byte_stream_split }` |
|
|
130
|
-
| `
|
|
131
|
-
| `
|
|
192
|
+
| `metadata` | `{}` | footer key/value metadata, read back with `reader.metadata` |
|
|
193
|
+
| `bloom_filters` | none | `true`, an Array of column paths, or `{ "path" => { ndv:, fpp:, max_bytes: } }` |
|
|
194
|
+
|
|
195
|
+
### Statistics, page indexes and bloom filters
|
|
196
|
+
|
|
197
|
+
Every column chunk gets min/max/null-count statistics and the Parquet page index (per-page min/max
|
|
198
|
+
and row offsets), which Herringbone, DuckDB, Spark, Trino, Arrow and DataFusion use to skip pages.
|
|
199
|
+
With 20,000 rows per page, `WHERE id BETWEEN ...` on a sorted column reads a handful of pages
|
|
200
|
+
instead of the whole row group, so sort rows by the columns you filter on.
|
|
201
|
+
|
|
202
|
+
Min/max cannot rule out a value inside a row group's range, the usual case for unsorted IDs,
|
|
203
|
+
emails or UUIDs; a bloom filter can. They are off by default. `bloom_filters: ["user_id", "email"]`
|
|
204
|
+
(or `true` for every non-boolean column) writes Parquet's split block bloom filters, sized from
|
|
205
|
+
each row group's distinct values at a false positive probability of 1% (`fpp:`, up to `max_bytes:`,
|
|
206
|
+
default 1MB) unless `ndv:` gives the number of distinct values. Nested leaves are named by their
|
|
207
|
+
dotted path (`"tags.list.element"`). Hashing is pure Ruby unless the `xxhash` gem is installed:
|
|
208
|
+
a million rows take about 1.2 s longer to write with a filter on an INT64 column and 2.8 s longer
|
|
209
|
+
with one on a ~22-byte string column, and about 0.5 s longer with `xxhash`.
|
|
132
210
|
|
|
133
|
-
|
|
211
|
+
### Writing to S3
|
|
134
212
|
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
`columns`, `primary_key` and `defined_enums` on the model.
|
|
213
|
+
Parquet keeps its metadata in a footer, so a file can be streamed into an S3 multipart upload
|
|
214
|
+
without a local copy, using `upload_stream` from `aws-sdk-s3`:
|
|
138
215
|
|
|
139
216
|
```ruby
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
217
|
+
s3 = Aws::S3::TransferManager.new # aws-sdk-s3 1.197+; before that, Aws::S3::Object#upload_stream
|
|
218
|
+
s3.upload_stream(bucket: "exports", key: "events.parquet", part_size: 16 * 1024 * 1024) do |io|
|
|
219
|
+
Herringbone::Writer.open(io, schema) do |w|
|
|
220
|
+
events.each { |event| w << event }
|
|
221
|
+
end
|
|
222
|
+
end
|
|
223
|
+
s3.upload_stream(bucket: "exports", key: "orders.parquet") { |io| Herringbone.write(io, Order.all) }
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
If the block raises, the SDK aborts the multipart upload and raises `Aws::S3::MultipartUploadError`,
|
|
227
|
+
so no partial object is left. S3 allows at most 10,000 parts, which with the default 5MB parts caps
|
|
228
|
+
the file at about 48GB; raise `part_size:` for bigger files.
|
|
229
|
+
|
|
230
|
+
## ActiveRecord
|
|
231
|
+
|
|
232
|
+
`Herringbone.write` also takes a model or relation, which it reads with `find_each`, using a schema
|
|
233
|
+
built from the model's columns. It returns the number of rows written:
|
|
234
|
+
|
|
235
|
+
```ruby
|
|
236
|
+
File.open("orders.parquet", "wb") do |file|
|
|
237
|
+
Herringbone.write(file, Order.where(created_at: 1.year.ago..), compression: :zstd)
|
|
143
238
|
end
|
|
239
|
+
schema = Herringbone::Schema.from_active_record(Order, only: %w[id status total])
|
|
240
|
+
Herringbone.write(io, Order, schema: schema)
|
|
144
241
|
```
|
|
145
242
|
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
243
|
+
`from_active_record` takes `only:`, `except:` and `parquet_enum: true`. Rails is not a dependency:
|
|
244
|
+
it only calls `columns`, `primary_key` and `defined_enums` on the model. Enum attributes are
|
|
245
|
+
written as their labels, and the writer rejects values outside the enum.
|
|
149
246
|
|
|
150
247
|
| Column | Parquet |
|
|
151
248
|
|---|---|
|
|
@@ -161,8 +258,8 @@ attributes are written as their labels, and the writer rejects values outside th
|
|
|
161
258
|
| `string`, `text`, `citext`, anything else | `string` |
|
|
162
259
|
| Postgres array columns | `list` of the element type |
|
|
163
260
|
|
|
164
|
-
Columns declared `NOT NULL` (and primary keys) are required, all others nullable.
|
|
165
|
-
|
|
261
|
+
Columns declared `NOT NULL` (and primary keys) are required, all others nullable. Column order
|
|
262
|
+
follows `Model.columns`.
|
|
166
263
|
|
|
167
264
|
## Type mapping
|
|
168
265
|
|
|
@@ -180,23 +277,50 @@ Column order follows `Model.columns`.
|
|
|
180
277
|
| UUID | `String` like `"0f1e2d3c-..."` |
|
|
181
278
|
| struct / list / map | `Hash` / `Array` / `Hash` |
|
|
182
279
|
|
|
183
|
-
##
|
|
280
|
+
## Inspecting files
|
|
184
281
|
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
282
|
+
> The inspector and its HTML view are modelled on
|
|
283
|
+
> **[Parquet X-ray](https://huggingface.co/spaces/cfahlgren1/parquet-xray) by cfahlgren1** —
|
|
284
|
+
> the design and the idea are theirs. Go check it out.
|
|
285
|
+
|
|
286
|
+
`Herringbone::Inspector` examines a file using only its footer, page headers, page indexes and
|
|
287
|
+
bloom filter headers. Nothing is decompressed, so it is fast on big files and works for ZSTD and
|
|
288
|
+
Brotli files without the codec gems.
|
|
289
|
+
|
|
290
|
+
```ruby
|
|
291
|
+
File.open("data.parquet", "rb") do |file|
|
|
292
|
+
inspector = Herringbone::Inspector.new(file)
|
|
293
|
+
inspector.summary # size, rows, row groups, codecs, page index / bloom filter presence...
|
|
294
|
+
inspector.row_groups[0].column("name").pages # page headers; also statistics, column_index...
|
|
295
|
+
puts inspector.report # the schema, every column chunk, key/value metadata (Arrow schema decoded)
|
|
296
|
+
inspector.to_h # all of it, JSON-serializable
|
|
297
|
+
inspector.to_html # one self-contained HTML page with a to-scale byte map of the file
|
|
298
|
+
end
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
`inspector.verify_checksums` reads every page body (still without decompressing) to check the page
|
|
302
|
+
CRCs; the results then appear in `summary`, `report`, `to_h` and `to_html`.
|
|
191
303
|
|
|
192
304
|
## Command line
|
|
193
305
|
|
|
194
306
|
```
|
|
195
|
-
bin/herringbone
|
|
196
|
-
bin/herringbone
|
|
197
|
-
bin/herringbone
|
|
307
|
+
bin/herringbone cat FILE [N] # rows as JSON lines
|
|
308
|
+
bin/herringbone inspect FILE [--pages] # text report (--pages lists every page header)
|
|
309
|
+
bin/herringbone inspect FILE --json # everything as JSON
|
|
310
|
+
bin/herringbone inspect FILE --html > out.html # the HTML page
|
|
198
311
|
```
|
|
199
312
|
|
|
313
|
+
Add `--verify-checksums` to any `inspect` form to check page CRCs.
|
|
314
|
+
|
|
315
|
+
## Supported format features
|
|
316
|
+
|
|
317
|
+
- Encodings (read and write): PLAIN, PLAIN_DICTIONARY/RLE_DICTIONARY, RLE, DELTA_BINARY_PACKED,
|
|
318
|
+
DELTA_LENGTH_BYTE_ARRAY, DELTA_BYTE_ARRAY, BYTE_STREAM_SPLIT; legacy BIT_PACKED levels (read)
|
|
319
|
+
- Data page v1 and v2, dictionary pages, page CRCs (written)
|
|
320
|
+
- Page indexes and split block bloom filters (read and written)
|
|
321
|
+
- Legacy list and map layouts per the Parquet backward-compatibility rules
|
|
322
|
+
- Not supported: encryption, column chunks in external files
|
|
323
|
+
|
|
200
324
|
## Development
|
|
201
325
|
|
|
202
326
|
```
|
data/bin/herringbone
CHANGED
|
@@ -2,9 +2,12 @@
|
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
4
|
# Usage:
|
|
5
|
-
# herringbone
|
|
6
|
-
# herringbone
|
|
7
|
-
# herringbone
|
|
5
|
+
# herringbone cat FILE [N] print rows as JSON lines (optionally only the first N)
|
|
6
|
+
# herringbone inspect FILE [--pages] print the schema, row groups, column chunks (and every page header)
|
|
7
|
+
# herringbone inspect FILE --json print all of it as JSON
|
|
8
|
+
# herringbone inspect FILE --html print a self-contained HTML page showing the file's layout
|
|
9
|
+
# add --verify-checksums to any inspect form to read page bodies and check their CRCs
|
|
10
|
+
# `inspect` reads only the footer, page headers and page indexes: no values are decompressed.
|
|
8
11
|
$LOAD_PATH.unshift File.expand_path("../lib", __dir__)
|
|
9
12
|
require "herringbone"
|
|
10
13
|
require "json"
|
|
@@ -22,33 +25,42 @@ def jsonable(v)
|
|
|
22
25
|
end
|
|
23
26
|
end
|
|
24
27
|
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
+
USAGE = "usage: herringbone cat FILE [N]\n" \
|
|
29
|
+
" herringbone inspect FILE [--pages | --json | --html] [--verify-checksums]"
|
|
30
|
+
|
|
31
|
+
command, *rest = ARGV
|
|
32
|
+
flags, args = rest.partition { |a| a.start_with?("--") }
|
|
33
|
+
path, limit = args
|
|
34
|
+
valid = case command
|
|
35
|
+
when "cat" then flags.empty? && path && (limit.nil? || limit.match?(/\A\d+\z/))
|
|
36
|
+
when "inspect"
|
|
37
|
+
formats = flags & %w[--pages --json --html]
|
|
38
|
+
path && limit.nil? && (flags - %w[--pages --json --html --verify-checksums]).empty? && formats.size <= 1
|
|
39
|
+
end
|
|
40
|
+
unless valid
|
|
41
|
+
warn USAGE
|
|
28
42
|
exit 1
|
|
29
43
|
end
|
|
30
44
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
+
# Parquet and file errors are reported in one line, without a backtrace
|
|
46
|
+
begin
|
|
47
|
+
File.open(path, "rb") do |io|
|
|
48
|
+
if command == "cat"
|
|
49
|
+
n = limit&.to_i
|
|
50
|
+
Herringbone::Reader.new(io).each_row(limit: n) { |row| puts JSON.generate(jsonable(row)) }
|
|
51
|
+
else
|
|
52
|
+
inspector = Herringbone::Inspector.new(io)
|
|
53
|
+
inspector.verify_checksums if flags.include?("--verify-checksums")
|
|
54
|
+
if flags.include?("--json")
|
|
55
|
+
puts JSON.pretty_generate(inspector.to_h)
|
|
56
|
+
elsif flags.include?("--html")
|
|
57
|
+
$stdout.write(inspector.to_html)
|
|
58
|
+
else
|
|
59
|
+
puts inspector.report(pages: flags.include?("--pages"))
|
|
45
60
|
end
|
|
46
61
|
end
|
|
47
|
-
when "cat"
|
|
48
|
-
n = limit&.to_i
|
|
49
|
-
r.each_row.with_index do |row, i|
|
|
50
|
-
break if n && i >= n
|
|
51
|
-
puts JSON.generate(jsonable(row))
|
|
52
|
-
end
|
|
53
62
|
end
|
|
63
|
+
rescue Herringbone::Error, SystemCallError => e
|
|
64
|
+
warn "herringbone: #{e.message}"
|
|
65
|
+
exit 1
|
|
54
66
|
end
|
|
@@ -5,23 +5,24 @@ module Herringbone
|
|
|
5
5
|
# Builds a schema from an ActiveRecord model, so that the Hashes returned by
|
|
6
6
|
# +record.attributes+ can be written directly:
|
|
7
7
|
#
|
|
8
|
-
# schema = Herringbone::Schema.from_active_record(Order)
|
|
9
|
-
#
|
|
10
|
-
# Order.find_each { |order| w << order.attributes }
|
|
11
|
-
# end
|
|
8
|
+
# schema = Herringbone::Schema.from_active_record(Order, except: %w[notes])
|
|
9
|
+
# File.open("orders.parquet", "wb") { |f| Herringbone.write(f, Order, schema: schema) }
|
|
12
10
|
#
|
|
13
11
|
# ActiveRecord is not required: this only uses what a model class exposes
|
|
14
12
|
# (+columns+, +primary_key+ and, when present, +defined_enums+).
|
|
15
13
|
#
|
|
16
|
-
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
19
|
-
#
|
|
20
|
-
#
|
|
21
|
-
#
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
14
|
+
# Primary key columns are never nullable. Rails enum attributes are string columns holding the
|
|
15
|
+
# labels (the writer rejects values outside the enum; stored values such as 0/1 are written as
|
|
16
|
+
# their labels).
|
|
17
|
+
#
|
|
18
|
+
# @param model [Class] ActiveRecord model class, or anything exposing +columns+ the same way
|
|
19
|
+
# @param only [Array<String, Symbol>, String, Symbol, nil] attribute names to include
|
|
20
|
+
# @param except [Array<String, Symbol>, String, Symbol, nil] attribute names to leave out
|
|
21
|
+
# @param parquet_enum [Boolean] true adds the Parquet ENUM annotation to enum columns, see Builder#enum
|
|
22
|
+
# @return [Schema] schema with one top-level field per selected column, in model column order
|
|
23
|
+
# @raise [ArgumentError] when no columns are left after +only+ / +except+
|
|
24
|
+
def self.from_active_record(model, only: nil, except: nil, parquet_enum: false)
|
|
25
|
+
only &&= Array(only).map(&:to_s)
|
|
25
26
|
except = Array(except).map(&:to_s)
|
|
26
27
|
defined_enums = model.respond_to?(:defined_enums) ? model.defined_enums.to_h { |k, v| [k.to_s, v] } : {}
|
|
27
28
|
primary_keys = Array(model.respond_to?(:primary_key) ? model.primary_key : nil).map(&:to_s)
|
|
@@ -34,7 +35,7 @@ module Herringbone
|
|
|
34
35
|
primary = primary_keys.include?(name)
|
|
35
36
|
nullable = column.null != false && !primary
|
|
36
37
|
if (mapping = defined_enums[name])
|
|
37
|
-
builder.enum(name, values: mapping.to_h, parquet_enum:
|
|
38
|
+
builder.enum(name, values: mapping.to_h, parquet_enum: parquet_enum, null: nullable)
|
|
38
39
|
else
|
|
39
40
|
ActiveRecordMapping.add_column(builder, column, nullable, primary)
|
|
40
41
|
end
|
|
@@ -58,9 +59,14 @@ module Herringbone
|
|
|
58
59
|
module ActiveRecordMapping
|
|
59
60
|
module_function
|
|
60
61
|
|
|
62
|
+
# Precision for decimal columns that declare none (Postgres +numeric+ without arguments);
|
|
63
|
+
# 38 digits is the most a 16-byte FIXED_LEN_BYTE_ARRAY decimal holds
|
|
61
64
|
DEFAULT_DECIMAL_PRECISION = 38
|
|
65
|
+
# Scale used together with DEFAULT_DECIMAL_PRECISION when the column declares neither
|
|
62
66
|
DEFAULT_DECIMAL_SCALE = 9
|
|
63
67
|
|
|
68
|
+
# Bit width of named SQL integer types (MySQL, Postgres and SQLite spellings), keyed by the
|
|
69
|
+
# leading word of the lowercased +sql_type+
|
|
64
70
|
INTEGER_SQL_TYPES = {
|
|
65
71
|
"tinyint" => 8, "int1" => 8,
|
|
66
72
|
"smallint" => 16, "int2" => 16, "smallserial" => 16, "serial2" => 16,
|
|
@@ -68,6 +74,16 @@ module Herringbone
|
|
|
68
74
|
"bigint" => 64, "int8" => 64, "bigserial" => 64, "serial8" => 64
|
|
69
75
|
}.freeze
|
|
70
76
|
|
|
77
|
+
# Declares +column+ on +builder+: hstore as map<string, string>, array columns as a list of
|
|
78
|
+
# the element type, everything else as the scalar type picked by #scalar_type.
|
|
79
|
+
#
|
|
80
|
+
# @param builder [Builder] builder to add the field to
|
|
81
|
+
# @param column [ActiveRecord::ConnectionAdapters::Column] column metadata (+name+, +type+, and
|
|
82
|
+
# when available +sql_type+, +array+, +limit+, +precision+, +scale+)
|
|
83
|
+
# @param nullable [Boolean] whether the field may be null
|
|
84
|
+
# @param primary [Boolean] whether the column is (part of) the primary key
|
|
85
|
+
# @return [Node] the added schema node
|
|
86
|
+
# @raise [ArgumentError] when the builder rejects the field (e.g. a duplicate name)
|
|
71
87
|
def add_column(builder, column, nullable, primary)
|
|
72
88
|
sql_type = column.respond_to?(:sql_type) ? column.sql_type.to_s.downcase : ""
|
|
73
89
|
array = (column.respond_to?(:array) && column.array) || sql_type.end_with?("[]")
|
|
@@ -87,10 +103,18 @@ module Herringbone
|
|
|
87
103
|
end
|
|
88
104
|
end
|
|
89
105
|
|
|
106
|
+
# Picks the DSL type for a non-array, non-hstore column, following the table above.
|
|
107
|
+
#
|
|
108
|
+
# @param column [ActiveRecord::ConnectionAdapters::Column] column metadata, read for
|
|
109
|
+
# +limit+, +precision+ and +scale+
|
|
110
|
+
# @param type [Symbol, nil] ActiveRecord's abstract type (+column.type+)
|
|
111
|
+
# @param sql_type [String] lowercased database type with any array suffix removed
|
|
112
|
+
# @param primary [Boolean] whether the column is (part of) the primary key
|
|
113
|
+
# @return [Array(Symbol, Hash{Symbol => Object})] DSL type and its options for Builder#column
|
|
90
114
|
def scalar_type(column, type, sql_type, primary)
|
|
91
115
|
case type
|
|
92
116
|
when :integer, :bigint then [integer_type(column, sql_type, primary), {}]
|
|
93
|
-
when :float then [sql_type == "float4" ? :float : :double, {}]
|
|
117
|
+
when :float then [(sql_type == "float4") ? :float : :double, {}]
|
|
94
118
|
when :decimal, :money
|
|
95
119
|
precision = column.respond_to?(:precision) ? column.precision : nil
|
|
96
120
|
scale = column.respond_to?(:scale) ? column.scale : nil
|
|
@@ -98,12 +122,12 @@ module Herringbone
|
|
|
98
122
|
precision = DEFAULT_DECIMAL_PRECISION
|
|
99
123
|
scale ||= DEFAULT_DECIMAL_SCALE
|
|
100
124
|
end
|
|
101
|
-
[:decimal, {
|
|
125
|
+
[:decimal, {precision: precision, scale: scale || 0}]
|
|
102
126
|
when :boolean then [:boolean, {}]
|
|
103
127
|
when :binary then [:binary, {}]
|
|
104
128
|
when :date then [:date, {}]
|
|
105
|
-
when :datetime, :timestamp, :timestamptz then [:timestamp, {
|
|
106
|
-
when :time then [:time, {
|
|
129
|
+
when :datetime, :timestamp, :timestamptz then [:timestamp, {unit: :micros, utc: true}]
|
|
130
|
+
when :time then [:time, {unit: :micros}]
|
|
107
131
|
when :json, :jsonb then [:json, {}]
|
|
108
132
|
when :uuid then [:uuid, {}]
|
|
109
133
|
else [:string, {}]
|
|
@@ -112,6 +136,11 @@ module Herringbone
|
|
|
112
136
|
|
|
113
137
|
# Named SQL types (smallint, bigint, ...) decide the width; a generic "integer"/"int"
|
|
114
138
|
# uses the column limit in bytes, since e.g. SQLite reports `t.integer limit: 2` as "integer(2)".
|
|
139
|
+
#
|
|
140
|
+
# @param column [ActiveRecord::ConnectionAdapters::Column] column metadata, read for +limit+
|
|
141
|
+
# @param sql_type [String] lowercased database type with any array suffix removed
|
|
142
|
+
# @param primary [Boolean] true forces int64, since row ids can outgrow the declared width
|
|
143
|
+
# @return [Symbol] one of +:int8+ .. +:int64+ or +:uint8+ .. +:uint64+
|
|
115
144
|
def integer_type(column, sql_type, primary)
|
|
116
145
|
bits = INTEGER_SQL_TYPES[sql_type[/\A[a-z0-9]+/]]
|
|
117
146
|
bits ||= case (column.respond_to?(:limit) ? column.limit : nil)
|
|
@@ -124,7 +153,7 @@ module Herringbone
|
|
|
124
153
|
bits = 64 if primary
|
|
125
154
|
unsigned = sql_type.include?("unsigned")
|
|
126
155
|
return :"uint#{bits}" if unsigned
|
|
127
|
-
{
|
|
156
|
+
{8 => :int8, 16 => :int16, 32 => :int32, 64 => :int64}.fetch(bits)
|
|
128
157
|
end
|
|
129
158
|
end
|
|
130
159
|
end
|