@gscdump/lakehouse 2.0.6 → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/_virtual/node_modules/.pnpm/db0@0.3.4_drizzle-orm@1.0.0-rc.4_@cloudflare+workers-types@5.20260728.1_valibot@1.4.2_typescript@6.0.3__zod@4.4.3_/node_modules/db0/dist/index.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/denque@2.1.0/node_modules/denque/index.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/fzstd@0.1.1/node_modules/fzstd/esm/index.mjs +6 -4
- package/dist/_virtual/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.bitreader.mjs +1 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.blocks.mjs +2 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.huffman.mjs +0 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.mjs +23 -20
- package/dist/_virtual/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.transform.mjs +1 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash=7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/bloom.mjs +3 -5
- package/dist/_virtual/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash=7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/column.mjs +4 -2
- package/dist/_virtual/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash=7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/datapage.mjs +11 -4
- package/dist/_virtual/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash=7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/types/wkb.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/src/types.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/types/snappy.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/column.mjs +7 -5
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/metadata.mjs +3 -2
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/plain.mjs +2 -1
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/rowgroup.mjs +8 -5
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/snappy.mjs +0 -2
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/variant.mjs +7 -4
- package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/types/snappy.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/avro/avro.read.mjs +25 -24
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/avro/avro.write.mjs +18 -17
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/catalog/rest.mjs +13 -8
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/fetch.mjs +36 -33
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/metadata.mjs +11 -6
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/read.mjs +2 -1
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/utils.mjs +4 -3
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/write/snapshot.mjs +4 -2
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/write/stage-position-delete.mjs +14 -13
- package/dist/_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/write/write.mjs +1 -1
- package/dist/_virtual/node_modules/.pnpm/ioredis@5.10.1_supports-color@10.2.2/node_modules/ioredis/built/ScanStream.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/ioredis@5.10.1_supports-color@10.2.2/node_modules/ioredis/built/cluster/util.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/ioredis@5.10.1_supports-color@10.2.2/node_modules/ioredis/built/types.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/lru-cache@11.5.0/node_modules/lru-cache/dist/esm/perf.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/readdirp@5.0.0/node_modules/readdirp/index.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/squirreling@0.15.0/node_modules/squirreling/src/ast.d.mts +1 -1
- package/dist/_virtual/node_modules/.pnpm/unstorage@1.17.5_aws4fetch@1.0.20_db0@0.3.4_ioredis@5.10.1_supports-color@10.2.2_/node_modules/unstorage/dist/shared/unstorage2.DqlWKU2I.d.mts +1 -1
- package/dist/catalog.mjs +5 -5
- package/dist/orphan-sweep.mjs +1 -1
- package/package.json +2 -2
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -145,7 +145,8 @@ var rfse = function(dat, bt, mal) {
|
|
|
145
145
|
if (sympos) err(0);
|
|
146
146
|
for (i = 0; i < sz; ++i) {
|
|
147
147
|
var ns = dstate[syms[i]]++;
|
|
148
|
-
|
|
148
|
+
var nb = nbits[i] = al - msb(ns);
|
|
149
|
+
nstate[i] = (ns << nb) - sz;
|
|
149
150
|
}
|
|
150
151
|
return [tpos + 7 >> 3, {
|
|
151
152
|
b: al,
|
|
@@ -382,9 +383,10 @@ var rzb = function(dat, st, out) {
|
|
|
382
383
|
if (btype == 2) {
|
|
383
384
|
var b3 = dat[bt], lbt = b3 & 3, sf = b3 >> 2 & 3;
|
|
384
385
|
var lss = b3 >> 4, lcs = 0, s4 = 0;
|
|
385
|
-
if (lbt < 2)
|
|
386
|
-
|
|
387
|
-
|
|
386
|
+
if (lbt < 2) {
|
|
387
|
+
if (sf & 1) lss |= dat[++bt] << 4 | (sf & 2 && dat[++bt] << 12);
|
|
388
|
+
else lss = b3 >> 3;
|
|
389
|
+
} else {
|
|
388
390
|
s4 = sf;
|
|
389
391
|
if (sf < 2) lss |= (dat[++bt] & 63) << 4, lcs = dat[bt] >> 6 | dat[++bt] << 2;
|
|
390
392
|
else if (sf == 2) lss |= dat[++bt] << 4 | (dat[++bt] & 3) << 12, lcs = dat[bt] >> 2 | dat[++bt] << 6;
|
|
@@ -64,7 +64,7 @@ BrotliBitReader.prototype.readMoreInput = function() {
|
|
|
64
64
|
for (let p = 0; p < 32; p++) this.buf_[dst + bytes_read + p] = 0;
|
|
65
65
|
}
|
|
66
66
|
if (dst === 0) {
|
|
67
|
-
for (let p = 0; p < 32; p++) this.buf_[
|
|
67
|
+
for (let p = 0; p < 32; p++) this.buf_[8192 + p] = this.buf_[p];
|
|
68
68
|
this.buf_ptr_ = BROTLI_READ_SIZE;
|
|
69
69
|
} else this.buf_ptr_ = 0;
|
|
70
70
|
this.bit_end_pos_ += bytes_read << 3;
|
|
@@ -114,7 +114,8 @@ function jumpToByteBoundary(br) {
|
|
|
114
114
|
return !br.readBits(new_bit_pos - br.bit_pos_);
|
|
115
115
|
}
|
|
116
116
|
function readBlockLength(table, index, br) {
|
|
117
|
-
const
|
|
117
|
+
const code = readSymbol(table, index, br);
|
|
118
|
+
const { offset, nbits } = kBlockLengthPrefixCode[code];
|
|
118
119
|
return offset + br.readBits(nbits);
|
|
119
120
|
}
|
|
120
121
|
export { copyUncompressedBlockToOutput, decodeBlockType, decodeMetaBlockLength, decodeVarLenUint8, decodeWindowBits, jumpToByteBoundary, readBlockLength };
|
|
@@ -147,7 +147,6 @@ function readHuffmanCode(alphabet_size, tables, table, br) {
|
|
|
147
147
|
code_lengths[symbols[2]] = 3;
|
|
148
148
|
code_lengths[symbols[3]] = 3;
|
|
149
149
|
} else code_lengths[symbols[0]] = 2;
|
|
150
|
-
break;
|
|
151
150
|
}
|
|
152
151
|
} else {
|
|
153
152
|
const code_length_code_lengths = new Uint8Array(CODE_LENGTH_CODES);
|
|
@@ -240,9 +240,11 @@ function brotli(input, output) {
|
|
|
240
240
|
range_idx -= 2;
|
|
241
241
|
distance_code = -1;
|
|
242
242
|
} else distance_code = 0;
|
|
243
|
-
const
|
|
243
|
+
const insertIndex = kInsertRangeLut[range_idx] + (cmd_code >> 3 & 7);
|
|
244
|
+
const insertPrefix = kInsertLengthPrefixCode[insertIndex];
|
|
244
245
|
const insertLength = insertPrefix.offset + br.readBits(insertPrefix.nbits);
|
|
245
|
-
const
|
|
246
|
+
const copyIndex = kCopyRangeLut[range_idx] + (cmd_code & 7);
|
|
247
|
+
const copyCode = kCopyLengthPrefixCode[copyIndex];
|
|
246
248
|
const copyLength = copyCode.offset + br.readBits(copyCode.nbits);
|
|
247
249
|
prev_byte1 = ringbuffer[pos - 1 & ringbuffer_mask];
|
|
248
250
|
prev_byte2 = ringbuffer[pos - 2 & ringbuffer_mask];
|
|
@@ -292,25 +294,26 @@ function brotli(input, output) {
|
|
|
292
294
|
if (pos < max_backward_distance && max_distance !== max_backward_distance) max_distance = pos;
|
|
293
295
|
else max_distance = max_backward_distance;
|
|
294
296
|
let copy_dst = pos & ringbuffer_mask;
|
|
295
|
-
if (distance > max_distance)
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
297
|
+
if (distance > max_distance) {
|
|
298
|
+
if (copyLength >= minDictionaryWordLength && copyLength <= maxDictionaryWordLength) {
|
|
299
|
+
let offset = offsetsByLength[copyLength];
|
|
300
|
+
const word_id = distance - max_distance - 1;
|
|
301
|
+
const shift = sizeBitsByLength[copyLength];
|
|
302
|
+
const word_idx = word_id & (1 << shift) - 1;
|
|
303
|
+
const transform_idx = word_id >> shift;
|
|
304
|
+
offset += word_idx * copyLength;
|
|
305
|
+
if (transform_idx < kNumTransforms) {
|
|
306
|
+
const len = transformDictionaryWord(ringbuffer, copy_dst, offset, copyLength, transform_idx);
|
|
307
|
+
copy_dst += len;
|
|
308
|
+
pos += len;
|
|
309
|
+
meta_block_remaining_len -= len;
|
|
310
|
+
if (copy_dst >= ringbuffer_end) {
|
|
311
|
+
output.write(ringbuffer, ringbuffer_size);
|
|
312
|
+
for (let _x = 0; _x < copy_dst - ringbuffer_end; _x++) ringbuffer[_x] = ringbuffer[ringbuffer_end + _x];
|
|
313
|
+
}
|
|
314
|
+
} else throw new Error("Invalid backward reference");
|
|
311
315
|
} else throw new Error("Invalid backward reference");
|
|
312
|
-
} else
|
|
313
|
-
else {
|
|
316
|
+
} else {
|
|
314
317
|
if (distance_code > 0) {
|
|
315
318
|
dist_rb[dist_rb_idx & 3] = distance;
|
|
316
319
|
dist_rb_idx++;
|
|
@@ -167,7 +167,7 @@ function transformDictionaryWord(dst, idx, word, len, transform) {
|
|
|
167
167
|
const { prefix } = kTransforms[transform];
|
|
168
168
|
const { suffix } = kTransforms[transform];
|
|
169
169
|
const t = kTransforms[transform].transform;
|
|
170
|
-
let skip = t < kOmitFirst1 ? 0 : t -
|
|
170
|
+
let skip = t < kOmitFirst1 ? 0 : t - 11;
|
|
171
171
|
const start_idx = idx;
|
|
172
172
|
if (skip > len) skip = len;
|
|
173
173
|
let prefix_pos = 0;
|
|
@@ -10,9 +10,7 @@ const SALT = new Uint32Array([
|
|
|
10
10
|
2667333959,
|
|
11
11
|
1550580529
|
|
12
12
|
]);
|
|
13
|
-
const BYTES_PER_BLOCK = 32;
|
|
14
13
|
const MIN_BYTES = 32;
|
|
15
|
-
const MAX_BYTES = 128 * 1024 * 1024;
|
|
16
14
|
function blockIndex(hash, numBlocks) {
|
|
17
15
|
return Number((hash >> 32n) * BigInt(numBlocks) >> 32n);
|
|
18
16
|
}
|
|
@@ -37,8 +35,8 @@ function optimalNumBytes(ndv, fpp) {
|
|
|
37
35
|
if (!(ndv >= 0)) throw new Error(`bloom filter ndv must be >= 0, got ${ndv}`);
|
|
38
36
|
const m = -8 * ndv / Math.log(1 - fpp ** (1 / 8));
|
|
39
37
|
let numBits = Math.ceil(m);
|
|
40
|
-
if (!isFinite(numBits) || numBits >
|
|
41
|
-
const blockBits =
|
|
38
|
+
if (!isFinite(numBits) || numBits > 1073741824) numBits = 1073741824;
|
|
39
|
+
const blockBits = 256;
|
|
42
40
|
numBits = Math.ceil(numBits / blockBits) * blockBits;
|
|
43
41
|
let numBytes = numBits >> 3;
|
|
44
42
|
if (numBytes < MIN_BYTES) numBytes = MIN_BYTES;
|
|
@@ -46,7 +44,7 @@ function optimalNumBytes(ndv, fpp) {
|
|
|
46
44
|
return numBytes;
|
|
47
45
|
}
|
|
48
46
|
var BloomBuilder = class {
|
|
49
|
-
constructor(element, { fpp = .01, maxBytes =
|
|
47
|
+
constructor(element, { fpp = .01, maxBytes = 1048576 } = {}) {
|
|
50
48
|
this.element = element;
|
|
51
49
|
this.fpp = fpp;
|
|
52
50
|
this.maxBytes = maxBytes;
|
|
@@ -15,7 +15,8 @@ function writeColumn({ writer, column, pageData }) {
|
|
|
15
15
|
const geospatial_statistics = stats && isGeospatial ? geospatialStatistics(values) : void 0;
|
|
16
16
|
let bloomFilter;
|
|
17
17
|
if (column.bloomFilter) {
|
|
18
|
-
const
|
|
18
|
+
const opts = typeof column.bloomFilter === "object" ? column.bloomFilter : void 0;
|
|
19
|
+
const builder = new BloomBuilder(element, opts);
|
|
19
20
|
for (const v of values) builder.insert(v);
|
|
20
21
|
bloomFilter = builder.finalize();
|
|
21
22
|
}
|
|
@@ -29,7 +30,8 @@ function writeColumn({ writer, column, pageData }) {
|
|
|
29
30
|
writeType = "INT32";
|
|
30
31
|
encoding = "RLE_DICTIONARY";
|
|
31
32
|
dictionary_page_offset = BigInt(writer.offset);
|
|
32
|
-
|
|
33
|
+
const unconverted = unconvert(element, dictionary);
|
|
34
|
+
writeDictionaryPage(writer, column, unconverted);
|
|
33
35
|
} else {
|
|
34
36
|
writeValues = unconvert(element, values);
|
|
35
37
|
encoding = userEncoding ?? (type === "BOOLEAN" && values.length > 16 ? "RLE" : "PLAIN");
|
|
@@ -59,7 +59,7 @@ function writeDataPageV2({ writer, column, encoding, pageData }) {
|
|
|
59
59
|
writer.appendBytes(compressedBytes);
|
|
60
60
|
}
|
|
61
61
|
function writePageHeader(writer, header) {
|
|
62
|
-
|
|
62
|
+
const compact = {
|
|
63
63
|
field_1: PageTypes.indexOf(header.type),
|
|
64
64
|
field_2: header.uncompressed_page_size,
|
|
65
65
|
field_3: header.compressed_page_size,
|
|
@@ -83,7 +83,8 @@ function writePageHeader(writer, header) {
|
|
|
83
83
|
field_6: header.data_page_header_v2.repetition_levels_byte_length,
|
|
84
84
|
field_7: header.data_page_header_v2.is_compressed ? void 0 : false
|
|
85
85
|
}
|
|
86
|
-
}
|
|
86
|
+
};
|
|
87
|
+
serializeTCompactProtocol(writer, compact);
|
|
87
88
|
}
|
|
88
89
|
function writeLevels(writer, column, dataPage) {
|
|
89
90
|
const { schemaPath } = column;
|
|
@@ -99,9 +100,15 @@ function writeLevels(writer, column, dataPage) {
|
|
|
99
100
|
}
|
|
100
101
|
const maxRepetitionLevel = getMaxRepetitionLevel(schemaPath);
|
|
101
102
|
let repetition_levels_byte_length = 0;
|
|
102
|
-
if (maxRepetitionLevel)
|
|
103
|
+
if (maxRepetitionLevel) {
|
|
104
|
+
const bitWidth = Math.ceil(Math.log2(maxRepetitionLevel + 1));
|
|
105
|
+
repetition_levels_byte_length = writeRleBitPackedHybrid(writer, repetitionLevels, bitWidth);
|
|
106
|
+
}
|
|
103
107
|
let definition_levels_byte_length = 0;
|
|
104
|
-
if (maxDefinitionLevel)
|
|
108
|
+
if (maxDefinitionLevel) {
|
|
109
|
+
const bitWidth = Math.ceil(Math.log2(maxDefinitionLevel + 1));
|
|
110
|
+
definition_levels_byte_length = writeRleBitPackedHybrid(writer, definitionLevels, bitWidth);
|
|
111
|
+
}
|
|
105
112
|
return {
|
|
106
113
|
definition_levels_byte_length,
|
|
107
114
|
repetition_levels_byte_length,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/src/types.d.mts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/types/snappy.d.mts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/column.mjs
CHANGED
|
@@ -56,7 +56,8 @@ function readPage(reader, header, columnDecoder, dictionary, previousChunk, page
|
|
|
56
56
|
const daph = header.data_page_header;
|
|
57
57
|
if (!daph) throw new Error("parquet data page header is undefined");
|
|
58
58
|
if (pageStart > daph.num_values && isFlatColumn(schemaPath)) return { skipped: daph.num_values };
|
|
59
|
-
const
|
|
59
|
+
const page = decompressPage(compressedBytes, Number(header.uncompressed_page_size), codec, compressors);
|
|
60
|
+
const { definitionLevels, repetitionLevels, dataPage } = readDataPage(page, daph, columnDecoder);
|
|
60
61
|
const values = convertWithDictionary(dataPage, dictionary, daph.encoding, columnDecoder);
|
|
61
62
|
return {
|
|
62
63
|
skipped: 0,
|
|
@@ -76,12 +77,13 @@ function readPage(reader, header, columnDecoder, dictionary, previousChunk, page
|
|
|
76
77
|
const diph = header.dictionary_page_header;
|
|
77
78
|
if (!diph) throw new Error("parquet dictionary page header is undefined");
|
|
78
79
|
const page = decompressPage(compressedBytes, Number(header.uncompressed_page_size), codec, compressors);
|
|
80
|
+
const reader = {
|
|
81
|
+
view: new DataView(page.buffer, page.byteOffset, page.byteLength),
|
|
82
|
+
offset: 0
|
|
83
|
+
};
|
|
79
84
|
return {
|
|
80
85
|
skipped: 0,
|
|
81
|
-
data: readPlain(
|
|
82
|
-
view: new DataView(page.buffer, page.byteOffset, page.byteLength),
|
|
83
|
-
offset: 0
|
|
84
|
-
}, type, diph.num_values, element.type_length)
|
|
86
|
+
data: readPlain(reader, type, diph.num_values, element.type_length)
|
|
85
87
|
};
|
|
86
88
|
} else throw new Error(`parquet unsupported page type: ${header.type}`);
|
|
87
89
|
}
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/metadata.mjs
CHANGED
|
@@ -44,10 +44,11 @@ function parquetMetadata(arrayBuffer, { parsers, geoparquet = true } = {}) {
|
|
|
44
44
|
const metadataLengthOffset = view.byteLength - 8;
|
|
45
45
|
const metadataLength = view.getUint32(metadataLengthOffset, true);
|
|
46
46
|
if (metadataLength > view.byteLength - 8) throw new Error(`parquet metadata length ${metadataLength} exceeds available buffer ${view.byteLength - 8}`);
|
|
47
|
-
const
|
|
47
|
+
const reader = {
|
|
48
48
|
view,
|
|
49
49
|
offset: metadataLengthOffset - metadataLength
|
|
50
|
-
}
|
|
50
|
+
};
|
|
51
|
+
const metadata = deserializeTCompactProtocol(reader);
|
|
51
52
|
const version = metadata.field_1;
|
|
52
53
|
const schema = metadata.field_2.map((field) => ({
|
|
53
54
|
type: ParquetTypes[field.field_1],
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/plain.mjs
CHANGED
|
@@ -17,7 +17,8 @@ function readPlainBoolean(reader, count) {
|
|
|
17
17
|
for (let i = 0; i < count; i++) {
|
|
18
18
|
const byteOffset = reader.offset + (i / 8 | 0);
|
|
19
19
|
const bitOffset = i % 8;
|
|
20
|
-
|
|
20
|
+
const byte = reader.view.getUint8(byteOffset);
|
|
21
|
+
values[i] = (byte & 1 << bitOffset) !== 0;
|
|
21
22
|
}
|
|
22
23
|
reader.offset += Math.ceil(count / 8);
|
|
23
24
|
return values;
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/rowgroup.mjs
CHANGED
|
@@ -25,10 +25,11 @@ function readRowGroup(options, { metadata }, groupPlan) {
|
|
|
25
25
|
asyncColumns.push({
|
|
26
26
|
pathInSchema,
|
|
27
27
|
data: Promise.resolve(options.file.slice(startByte, endByte)).then((buffer) => {
|
|
28
|
-
|
|
28
|
+
const reader = {
|
|
29
29
|
view: new DataView(buffer),
|
|
30
30
|
offset: 0
|
|
31
|
-
}
|
|
31
|
+
};
|
|
32
|
+
return readColumn(reader, groupPlan, columnDecoder, options.onPage);
|
|
32
33
|
})
|
|
33
34
|
});
|
|
34
35
|
continue;
|
|
@@ -55,15 +56,17 @@ function readRowGroup(options, { metadata }, groupPlan) {
|
|
|
55
56
|
}
|
|
56
57
|
if (skipped < 0) skipped = 0;
|
|
57
58
|
const buffer = await options.file.slice(startByte, endByte);
|
|
58
|
-
const
|
|
59
|
+
const reader = {
|
|
59
60
|
view: new DataView(buffer),
|
|
60
61
|
offset: 0
|
|
61
|
-
}
|
|
62
|
+
};
|
|
63
|
+
const adjustedGroupPlan = skipped ? {
|
|
62
64
|
...groupPlan,
|
|
63
65
|
groupStart: groupPlan.groupStart + skipped,
|
|
64
66
|
selectStart: groupPlan.selectStart - skipped,
|
|
65
67
|
selectEnd: groupPlan.selectEnd - skipped
|
|
66
|
-
} : groupPlan
|
|
68
|
+
} : groupPlan;
|
|
69
|
+
const { data, skipped: columnSkipped } = readColumn(reader, adjustedGroupPlan, columnDecoder, options.onPage);
|
|
67
70
|
return {
|
|
68
71
|
data,
|
|
69
72
|
skipped: skipped + columnSkipped
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/snappy.mjs
CHANGED
|
@@ -56,8 +56,6 @@ function snappyUncompress(input, output) {
|
|
|
56
56
|
len = (c >>> 2) + 1;
|
|
57
57
|
offset = input[pos] + (input[pos + 1] << 8) + (input[pos + 2] << 16) + (input[pos + 3] << 24);
|
|
58
58
|
pos += 4;
|
|
59
|
-
break;
|
|
60
|
-
default: break;
|
|
61
59
|
}
|
|
62
60
|
if (offset === 0 || isNaN(offset)) throw new Error(`invalid offset ${offset} pos ${pos} inputLength ${inputLength}`);
|
|
63
61
|
if (offset > outPos) throw new Error("cannot copy from before start of buffer");
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/variant.mjs
CHANGED
|
@@ -191,10 +191,13 @@ function readVariantArray(reader, header, metadata, parsers) {
|
|
|
191
191
|
for (let i = 0; i < offsets.length; i++) offsets[i] = readUnsigned(reader, offsetWidth);
|
|
192
192
|
const valuesStart = reader.offset;
|
|
193
193
|
const result = new Array(numElements);
|
|
194
|
-
for (let i = 0; i < numElements; i++)
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
194
|
+
for (let i = 0; i < numElements; i++) {
|
|
195
|
+
const valueReader = {
|
|
196
|
+
view: reader.view,
|
|
197
|
+
offset: valuesStart + offsets[i]
|
|
198
|
+
};
|
|
199
|
+
result[i] = readVariant(valueReader, metadata, parsers);
|
|
200
|
+
}
|
|
198
201
|
reader.offset = valuesStart + offsets[offsets.length - 1];
|
|
199
202
|
return result;
|
|
200
203
|
}
|
package/dist/_virtual/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/types/snappy.d.mts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -68,30 +68,31 @@ function readType(reader, type) {
|
|
|
68
68
|
}
|
|
69
69
|
return map;
|
|
70
70
|
} else if (typeof type === "object" && type.type === "enum") return type.symbols[readZigZag(reader)];
|
|
71
|
-
else if (typeof type === "object" && "logicalType" in type && type.logicalType)
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
71
|
+
else if (typeof type === "object" && "logicalType" in type && type.logicalType) {
|
|
72
|
+
if (type.logicalType === "date" && type.type === "int") {
|
|
73
|
+
const value = readZigZag(reader);
|
|
74
|
+
return /* @__PURE__ */ new Date(value * 864e5);
|
|
75
|
+
} else if (type.logicalType === "time-millis" && type.type === "int") return readZigZag(reader);
|
|
76
|
+
else if (type.logicalType === "time-micros" && type.type === "long") return readZigZagBigInt(reader);
|
|
77
|
+
else if (type.logicalType === "timestamp-millis" && type.type === "long") {
|
|
78
|
+
const value = readZigZagBigInt(reader);
|
|
79
|
+
return new Date(Number(value));
|
|
80
|
+
} else if (type.logicalType === "timestamp-micros" && type.type === "long") {
|
|
81
|
+
const value = readZigZagBigInt(reader);
|
|
82
|
+
return new Date(Number(value / 1000n));
|
|
83
|
+
} else if (type.logicalType === "timestamp-nanos" && type.type === "long") {
|
|
84
|
+
const value = readZigZagBigInt(reader);
|
|
85
|
+
return new Date(Number(value / 1000000n));
|
|
86
|
+
} else if (type.logicalType === "decimal" && "precision" in type) {
|
|
87
|
+
const bytes = type.type === "fixed" ? readFixed(reader, type.size) : readType(reader, type.type);
|
|
88
|
+
const factor = 10 ** -(type.scale || 0);
|
|
89
|
+
return parseDecimal(bytes) * factor;
|
|
90
|
+
} else if (type.logicalType === "uuid" && type.type === "fixed" && type.size === 16) return bytesToUuid(readFixed(reader, 16));
|
|
91
|
+
else {
|
|
92
|
+
console.warn(`unknown logical type: ${type.logicalType}`);
|
|
93
|
+
return type.type === "fixed" ? readFixed(reader, type.size) : readType(reader, type.type);
|
|
94
|
+
}
|
|
95
|
+
} else if (typeof type === "object" && type.type === "fixed") return readFixed(reader, type.size);
|
|
95
96
|
else if (type === "boolean") {
|
|
96
97
|
const value = reader.view.getUint8(reader.offset) === 1;
|
|
97
98
|
reader.offset++;
|
|
@@ -95,23 +95,24 @@ function writeType(writer, schema, value) {
|
|
|
95
95
|
if (!(bytes instanceof Uint8Array)) throw new Error("expected Uint8Array value");
|
|
96
96
|
if (bytes.length !== schema.size) throw new Error(`expected fixed[${schema.size}] value`);
|
|
97
97
|
writer.appendBytes(bytes);
|
|
98
|
-
} else if ("logicalType" in schema)
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
98
|
+
} else if ("logicalType" in schema) {
|
|
99
|
+
if (schema.logicalType === "date") appendZigZag(writer, value instanceof Date ? Math.floor(value.getTime() / 864e5) : value);
|
|
100
|
+
else if (schema.logicalType === "time-millis") appendZigZag(writer, value);
|
|
101
|
+
else if (schema.logicalType === "time-micros") appendZigZag64(writer, BigInt(value));
|
|
102
|
+
else if (schema.logicalType === "timestamp-millis") appendZigZag64(writer, value instanceof Date ? BigInt(value.getTime()) : BigInt(value));
|
|
103
|
+
else if (schema.logicalType === "timestamp-micros") appendZigZag64(writer, value instanceof Date ? BigInt(value.getTime()) * 1000n : BigInt(value));
|
|
104
|
+
else if (schema.logicalType === "timestamp-nanos") appendZigZag64(writer, value instanceof Date ? BigInt(value.getTime()) * 1000000n : BigInt(value));
|
|
105
|
+
else if (schema.logicalType === "decimal") {
|
|
106
|
+
const scale = "scale" in schema ? schema.scale ?? 0 : 0;
|
|
107
|
+
let u;
|
|
108
|
+
if (typeof value === "bigint") u = value;
|
|
109
|
+
else if (typeof value === "number") u = BigInt(Math.round(value * 10 ** scale));
|
|
110
|
+
else throw new Error("decimal value must be bigint or number");
|
|
111
|
+
const b = bigIntToBytes(u);
|
|
112
|
+
appendZigZag(writer, b.length);
|
|
113
|
+
writer.appendBytes(b);
|
|
114
|
+
} else throw new Error(`unknown logical type ${schema.logicalType}`);
|
|
115
|
+
} else throw new Error(`unknown schema type ${JSON.stringify(schema)}`);
|
|
115
116
|
}
|
|
116
117
|
function appendZigZag(writer, v) {
|
|
117
118
|
writer.appendVarInt(v << 1 ^ v >> 31);
|
|
@@ -24,7 +24,8 @@ async function restCatalogConnect({ url, warehouse, requestInit, signRequest })
|
|
|
24
24
|
function restCatalogListTables(ctx, { namespace }) {
|
|
25
25
|
const ns = encodeNamespace(namespace);
|
|
26
26
|
return paginate({}, async (query) => {
|
|
27
|
-
const
|
|
27
|
+
const res = await restFetch(ctx, `namespaces/${ns}/tables${query}`);
|
|
28
|
+
const body = parseIcebergJson(await res.text());
|
|
28
29
|
return {
|
|
29
30
|
items: body.identifiers ?? [],
|
|
30
31
|
nextPageToken: body["next-page-token"]
|
|
@@ -32,7 +33,8 @@ function restCatalogListTables(ctx, { namespace }) {
|
|
|
32
33
|
});
|
|
33
34
|
}
|
|
34
35
|
async function restCatalogLoadTable(ctx, { namespace, table }) {
|
|
35
|
-
const
|
|
36
|
+
const res = await restFetch(ctx, `namespaces/${encodeNamespace(namespace)}/tables/${encodeURIComponent(table)}`);
|
|
37
|
+
const body = parseIcebergJson(await res.text());
|
|
36
38
|
return {
|
|
37
39
|
metadataLocation: body["metadata-location"],
|
|
38
40
|
metadata: body.metadata,
|
|
@@ -50,11 +52,12 @@ async function restCatalogCreateTable(ctx, { namespace, table, schema, location,
|
|
|
50
52
|
if (writeOrder !== void 0) body["write-order"] = writeOrder;
|
|
51
53
|
if (stageCreate !== void 0) body["stage-create"] = stageCreate;
|
|
52
54
|
if (properties !== void 0) body.properties = properties;
|
|
53
|
-
const
|
|
55
|
+
const res = await restFetch(ctx, `namespaces/${ns}/tables`, {
|
|
54
56
|
method: "POST",
|
|
55
57
|
headers: { "content-type": "application/json" },
|
|
56
58
|
body: stringifyIcebergJson(body)
|
|
57
|
-
})
|
|
59
|
+
});
|
|
60
|
+
const responseBody = parseIcebergJson(await res.text());
|
|
58
61
|
return {
|
|
59
62
|
metadataLocation: responseBody["metadata-location"],
|
|
60
63
|
metadata: responseBody.metadata,
|
|
@@ -62,14 +65,15 @@ async function restCatalogCreateTable(ctx, { namespace, table, schema, location,
|
|
|
62
65
|
};
|
|
63
66
|
}
|
|
64
67
|
async function restCatalogUpdateTable(ctx, { namespace, table, requirements, updates }) {
|
|
65
|
-
const
|
|
68
|
+
const res = await restFetch(ctx, `namespaces/${encodeNamespace(namespace)}/tables/${encodeURIComponent(table)}`, {
|
|
66
69
|
method: "POST",
|
|
67
70
|
headers: { "content-type": "application/json" },
|
|
68
71
|
body: stringifyIcebergJson({
|
|
69
72
|
requirements,
|
|
70
73
|
updates
|
|
71
74
|
})
|
|
72
|
-
})
|
|
75
|
+
});
|
|
76
|
+
const responseBody = parseIcebergJson(await res.text());
|
|
73
77
|
return {
|
|
74
78
|
metadataLocation: responseBody["metadata-location"],
|
|
75
79
|
metadata: responseBody.metadata,
|
|
@@ -81,14 +85,15 @@ async function restCatalogDropTable(ctx, { namespace, table, purgeRequested }) {
|
|
|
81
85
|
}
|
|
82
86
|
async function restCatalogCreateNamespace(ctx, { namespace, properties }) {
|
|
83
87
|
const ns = Array.isArray(namespace) ? namespace : namespace.split(".");
|
|
84
|
-
const
|
|
88
|
+
const res = await restFetch(ctx, "namespaces", {
|
|
85
89
|
method: "POST",
|
|
86
90
|
headers: { "content-type": "application/json" },
|
|
87
91
|
body: stringifyIcebergJson({
|
|
88
92
|
namespace: ns,
|
|
89
93
|
properties: properties ?? {}
|
|
90
94
|
})
|
|
91
|
-
})
|
|
95
|
+
});
|
|
96
|
+
const body = parseIcebergJson(await res.text());
|
|
92
97
|
return {
|
|
93
98
|
namespace: body.namespace ?? ns,
|
|
94
99
|
properties: body.properties ?? {}
|
|
@@ -136,50 +136,53 @@ async function fetchDeleteMaps(deleteEntries, resolver) {
|
|
|
136
136
|
const equalityDeleteGroups = [];
|
|
137
137
|
await Promise.all(deleteEntries.map(async (deleteEntry) => {
|
|
138
138
|
const { content, file_path, file_size_in_bytes } = deleteEntry.data_file;
|
|
139
|
-
const
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
if (
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
const
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
set =
|
|
166
|
-
|
|
139
|
+
const asyncBuffer = await resolver.reader(file_path, Number(file_size_in_bytes));
|
|
140
|
+
const file = cachedAsyncBuffer(asyncBuffer);
|
|
141
|
+
if (content === 1) {
|
|
142
|
+
if (isDeletionVector(deleteEntry)) {
|
|
143
|
+
const { referenced_data_file, content_offset, content_size_in_bytes } = deleteEntry.data_file;
|
|
144
|
+
if (!referenced_data_file) throw new Error("deletion vector missing referenced_data_file");
|
|
145
|
+
if (content_offset == null) throw new Error("deletion vector missing content_offset");
|
|
146
|
+
if (content_size_in_bytes == null) throw new Error("deletion vector missing content_size_in_bytes");
|
|
147
|
+
const positions = await puffinReadDeletionVector(file, {
|
|
148
|
+
offset: content_offset,
|
|
149
|
+
length: content_size_in_bytes,
|
|
150
|
+
referencedDataFile: referenced_data_file
|
|
151
|
+
});
|
|
152
|
+
const set = /* @__PURE__ */ new Set();
|
|
153
|
+
for (const pos of positions) set.add(pos);
|
|
154
|
+
addPositionDeleteGroup(positionDeletesMap, referenced_data_file, deleteEntry, set);
|
|
155
|
+
} else {
|
|
156
|
+
const deleteRows = await parquetReadObjects({
|
|
157
|
+
file,
|
|
158
|
+
compressors
|
|
159
|
+
});
|
|
160
|
+
const positionsByPath = /* @__PURE__ */ new Map();
|
|
161
|
+
for (const deleteRow of deleteRows) {
|
|
162
|
+
const { file_path, pos } = deleteRow;
|
|
163
|
+
if (!file_path) throw new Error("position delete missing target file path");
|
|
164
|
+
if (pos === void 0) throw new Error("position delete missing pos");
|
|
165
|
+
let set = positionsByPath.get(file_path);
|
|
166
|
+
if (!set) {
|
|
167
|
+
set = /* @__PURE__ */ new Set();
|
|
168
|
+
positionsByPath.set(file_path, set);
|
|
169
|
+
}
|
|
170
|
+
set.add(pos);
|
|
167
171
|
}
|
|
168
|
-
|
|
172
|
+
for (const [targetPath, positions] of positionsByPath) addPositionDeleteGroup(positionDeletesMap, targetPath, deleteEntry, positions);
|
|
169
173
|
}
|
|
170
|
-
|
|
171
|
-
}
|
|
172
|
-
else if (content === 2) {
|
|
174
|
+
} else if (content === 2) {
|
|
173
175
|
const { sequence_number } = deleteEntry;
|
|
174
176
|
const equalityIds = deleteEntry.data_file.equality_ids;
|
|
175
177
|
if (sequence_number === void 0) throw new Error("equality delete missing sequence number");
|
|
176
178
|
if (!equalityIds?.length) throw new Error("equality delete missing equality_ids");
|
|
177
179
|
const metadata = await parquetMetadataAsync(file);
|
|
178
180
|
const columnNamesById = equalityColumnNamesById(metadata, equalityIds);
|
|
181
|
+
const columns = equalityIds.map((id) => columnNamesById[id]);
|
|
179
182
|
const deleteRows = await parquetReadObjects({
|
|
180
183
|
file,
|
|
181
184
|
metadata,
|
|
182
|
-
columns
|
|
185
|
+
columns,
|
|
183
186
|
compressors
|
|
184
187
|
});
|
|
185
188
|
const rows = [];
|
|
@@ -58,18 +58,22 @@ async function resolveMetadata({ tableUrl, metadataFileName, resolver, lister })
|
|
|
58
58
|
})}.metadata.json`;
|
|
59
59
|
const url = `${tableUrl}/metadata/${metadataFileName}`;
|
|
60
60
|
try {
|
|
61
|
+
const text = await resolveText(resolver, url);
|
|
61
62
|
return {
|
|
62
|
-
metadata: parseIcebergJson(
|
|
63
|
+
metadata: parseIcebergJson(text),
|
|
63
64
|
metadataFileName
|
|
64
65
|
};
|
|
65
66
|
} catch (err) {
|
|
66
67
|
try {
|
|
67
68
|
const metadataDir = `${tableUrl}/metadata`;
|
|
68
69
|
const match = findMetadataFile(await lister(metadataDir), metadataFileName);
|
|
69
|
-
if (match)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
70
|
+
if (match) {
|
|
71
|
+
const text = await resolveText(resolver, `${metadataDir}/${match}`);
|
|
72
|
+
return {
|
|
73
|
+
metadata: parseIcebergJson(text),
|
|
74
|
+
metadataFileName: match
|
|
75
|
+
};
|
|
76
|
+
}
|
|
73
77
|
} catch {}
|
|
74
78
|
throw new Error(`failed to get iceberg metadata: ${err.message}`);
|
|
75
79
|
}
|
|
@@ -137,9 +141,10 @@ async function tryReadVersion(resolver, tableUrl, version) {
|
|
|
137
141
|
const fileName = `v${version}.metadata.json`;
|
|
138
142
|
const metadataLocation = `${tableUrl}/metadata/${fileName}`;
|
|
139
143
|
try {
|
|
144
|
+
const text = await resolveText(resolver, metadataLocation);
|
|
140
145
|
return {
|
|
141
146
|
version,
|
|
142
|
-
metadata: parseIcebergJson(
|
|
147
|
+
metadata: parseIcebergJson(text),
|
|
143
148
|
metadataFileName: fileName,
|
|
144
149
|
metadataLocation
|
|
145
150
|
};
|
|
@@ -83,7 +83,8 @@ async function* readDataFile({ dataEntry, fileRowStart, fileRowEnd, schema, meta
|
|
|
83
83
|
if (sequence_number === void 0) throw new Error("sequence number not found, check v2 inheritance logic");
|
|
84
84
|
const sequenceNumber = sequence_number;
|
|
85
85
|
const partitionSpec = metadata["partition-specs"].find((s) => s["spec-id"] === partition_spec_id);
|
|
86
|
-
const
|
|
86
|
+
const resolved = await resolver.reader(data_file.file_path, Number(data_file.file_size_in_bytes));
|
|
87
|
+
const asyncBuffer = cachedAsyncBuffer(resolved);
|
|
87
88
|
const parquetMetadata = await parquetMetadataAsync(asyncBuffer);
|
|
88
89
|
const kv = parquetMetadata.key_value_metadata?.find((k) => k.key === "iceberg.schema");
|
|
89
90
|
let parquetIcebergSchema;
|
|
@@ -4,9 +4,10 @@ function sanitize(name) {
|
|
|
4
4
|
const ch = name.charAt(i);
|
|
5
5
|
const isLetter = /^[A-Za-z]$/.test(ch);
|
|
6
6
|
const isDigit = /^[0-9]$/.test(ch);
|
|
7
|
-
if (i === 0)
|
|
8
|
-
|
|
9
|
-
|
|
7
|
+
if (i === 0) {
|
|
8
|
+
if (isLetter || ch === "_") result += ch;
|
|
9
|
+
else result += isDigit ? "_" + ch : "_x" + ch.charCodeAt(0).toString(16).toUpperCase();
|
|
10
|
+
} else if (isLetter || isDigit || ch === "_") result += ch;
|
|
10
11
|
else result += "_x" + ch.charCodeAt(0).toString(16).toUpperCase();
|
|
11
12
|
}
|
|
12
13
|
return result;
|
|
@@ -22,8 +22,9 @@ async function buildSnapshotUpdate({ tableUrl, metadata, resolver, snapshotId, s
|
|
|
22
22
|
const allManifests = [...priorManifests, ...newManifests];
|
|
23
23
|
const addedRows = rowLineage ? assignFirstRowIds(allManifests, firstRowId) : 0n;
|
|
24
24
|
const manifestListPath = `${tableUrl}/metadata/snap-${snapshotId}-1-${manifestUuid}.avro`;
|
|
25
|
+
const listWriter = writerFn(manifestListPath);
|
|
25
26
|
await writeManifestList({
|
|
26
|
-
writer:
|
|
27
|
+
writer: listWriter,
|
|
27
28
|
snapshotId,
|
|
28
29
|
sequenceNumber,
|
|
29
30
|
manifests: allManifests,
|
|
@@ -76,7 +77,8 @@ function buildPartitionSummaries(partitions, schema, partitionSpec) {
|
|
|
76
77
|
const sourceField = schema.fields.find((f) => f.id === pf["source-id"]);
|
|
77
78
|
if (!sourceField) throw new Error(`partition source field id ${pf["source-id"]} not found`);
|
|
78
79
|
const resultType = transformResultType(pf.transform, sourceField.type);
|
|
79
|
-
|
|
80
|
+
const values = partitions.map((p) => p[pf.name]);
|
|
81
|
+
return computeFieldSummary(values, resultType);
|
|
80
82
|
});
|
|
81
83
|
}
|
|
82
84
|
function assignFirstRowIds(manifests, firstRowId) {
|
|
@@ -133,6 +133,19 @@ async function icebergStagePositionDelete({ tableUrl, metadata, deletes, resolve
|
|
|
133
133
|
};
|
|
134
134
|
const addedSize = writtenDeleteFiles.reduce((sum, f) => sum + f.deleteFile.file_size_in_bytes, 0n);
|
|
135
135
|
const addedPosDeletes = writtenDeleteFiles.reduce((sum, f) => sum + f.deleteFile.record_count, 0n);
|
|
136
|
+
const summary = {
|
|
137
|
+
operation: "delete",
|
|
138
|
+
"added-delete-files": String(writtenDeleteFiles.length),
|
|
139
|
+
"added-position-deletes": String(addedPosDeletes),
|
|
140
|
+
"added-files-size": String(addedSize),
|
|
141
|
+
"changed-partition-count": String(groups.size),
|
|
142
|
+
"total-records": String(prevTotals.records),
|
|
143
|
+
"total-files-size": String(prevTotals.size + addedSize),
|
|
144
|
+
"total-data-files": String(prevTotals.dataFiles),
|
|
145
|
+
"total-delete-files": String(prevTotals.deleteFiles + BigInt(writtenDeleteFiles.length)),
|
|
146
|
+
"total-position-deletes": String(prevTotals.posDeletes + addedPosDeletes),
|
|
147
|
+
"total-equality-deletes": String(prevTotals.eqDeletes)
|
|
148
|
+
};
|
|
136
149
|
return await buildSnapshotUpdate({
|
|
137
150
|
tableUrl,
|
|
138
151
|
metadata,
|
|
@@ -143,19 +156,7 @@ async function icebergStagePositionDelete({ tableUrl, metadata, deletes, resolve
|
|
|
143
156
|
timestampMs,
|
|
144
157
|
formatVersion,
|
|
145
158
|
newManifests,
|
|
146
|
-
summary
|
|
147
|
-
operation: "delete",
|
|
148
|
-
"added-delete-files": String(writtenDeleteFiles.length),
|
|
149
|
-
"added-position-deletes": String(addedPosDeletes),
|
|
150
|
-
"added-files-size": String(addedSize),
|
|
151
|
-
"changed-partition-count": String(groups.size),
|
|
152
|
-
"total-records": String(prevTotals.records),
|
|
153
|
-
"total-files-size": String(prevTotals.size + addedSize),
|
|
154
|
-
"total-data-files": String(prevTotals.dataFiles),
|
|
155
|
-
"total-delete-files": String(prevTotals.deleteFiles + BigInt(writtenDeleteFiles.length)),
|
|
156
|
-
"total-position-deletes": String(prevTotals.posDeletes + addedPosDeletes),
|
|
157
|
-
"total-equality-deletes": String(prevTotals.eqDeletes)
|
|
158
|
-
},
|
|
159
|
+
summary,
|
|
159
160
|
writtenFiles: [...writtenDeleteFiles.map((f) => f.path), ...writtenManifestPaths]
|
|
160
161
|
});
|
|
161
162
|
}
|
|
@@ -12,7 +12,7 @@ const DEFAULT_RETRY = Object.freeze({
|
|
|
12
12
|
initialMs: 50,
|
|
13
13
|
maxMs: 3e3,
|
|
14
14
|
factor: 2,
|
|
15
|
-
totalTimeoutMs:
|
|
15
|
+
totalTimeoutMs: 18e5
|
|
16
16
|
});
|
|
17
17
|
async function icebergAppend({ catalog, namespace, table, tableUrl, resolver, records, sortOrderId, snapshotProperties }) {
|
|
18
18
|
const ctx = await loadTable({
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/_virtual/node_modules/.pnpm/lru-cache@11.5.0/node_modules/lru-cache/dist/esm/perf.d.mts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/_virtual/node_modules/.pnpm/squirreling@0.15.0/node_modules/squirreling/src/ast.d.mts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/catalog.mjs
CHANGED
|
@@ -10,7 +10,7 @@ function useIcebirdWriteModule() {
|
|
|
10
10
|
icebirdWriteModule ??= import("./_virtual/node_modules/.pnpm/icebird@0.8.15_patch_hash=d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/write/write.mjs");
|
|
11
11
|
return icebirdWriteModule;
|
|
12
12
|
}
|
|
13
|
-
const CATALOG_CONFIG_TTL_MS =
|
|
13
|
+
const CATALOG_CONFIG_TTL_MS = 36e5;
|
|
14
14
|
function catalogConfigKey(config) {
|
|
15
15
|
return `lakehouse-catalog-cfg\0${config.catalogUri}\0${config.warehouse}`;
|
|
16
16
|
}
|
|
@@ -225,10 +225,10 @@ async function appendAlreadyLanded(args, appendId) {
|
|
|
225
225
|
for (let i = snapshots.length - 1; i >= from; i--) if (snapshots[i]?.summary?.[APPEND_ID_SUMMARY_KEY] === appendId) return true;
|
|
226
226
|
return false;
|
|
227
227
|
}
|
|
228
|
-
const SNAPSHOT_REF_TTL_MS =
|
|
229
|
-
const RESOLVED_FILES_TTL_MS =
|
|
230
|
-
const METADATA_TTL_MS =
|
|
231
|
-
const MAX_CACHED_METADATA_BYTES =
|
|
228
|
+
const SNAPSHOT_REF_TTL_MS = 18e5;
|
|
229
|
+
const RESOLVED_FILES_TTL_MS = 864e5;
|
|
230
|
+
const METADATA_TTL_MS = 864e5;
|
|
231
|
+
const MAX_CACHED_METADATA_BYTES = 2097152;
|
|
232
232
|
function snapshotRefKey(scope, namespace, table) {
|
|
233
233
|
return `lh-snapref\0${scope}\0${namespace}\0${table}`;
|
|
234
234
|
}
|
package/dist/orphan-sweep.mjs
CHANGED
|
@@ -5,7 +5,7 @@ const DEFAULT_GRACE_HOURS = 48;
|
|
|
5
5
|
const DEFAULT_MAX_DELETES = 500;
|
|
6
6
|
const DEFAULT_IO_CONCURRENCY = 4;
|
|
7
7
|
const MAX_IO_CONCURRENCY = 32;
|
|
8
|
-
const HOUR_MS =
|
|
8
|
+
const HOUR_MS = 36e5;
|
|
9
9
|
function createIoLimiter(requested) {
|
|
10
10
|
const concurrency = typeof requested === "number" && Number.isFinite(requested) ? Math.max(1, Math.min(MAX_IO_CONCURRENCY, Math.floor(requested))) : DEFAULT_IO_CONCURRENCY;
|
|
11
11
|
let active = 0;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@gscdump/lakehouse",
|
|
3
3
|
"type": "module",
|
|
4
|
-
"version": "2.
|
|
4
|
+
"version": "2.1.1",
|
|
5
5
|
"description": "Dataset-agnostic Iceberg lakehouse layer + dataset registry for R2 Data Catalog producers (ADR-0021).",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Harlan Wilton",
|
|
@@ -71,7 +71,7 @@
|
|
|
71
71
|
"node": ">=22"
|
|
72
72
|
},
|
|
73
73
|
"devDependencies": {
|
|
74
|
-
"@types/node": "^26.
|
|
74
|
+
"@types/node": "^26.2.0",
|
|
75
75
|
"hyparquet": "^1.26.2",
|
|
76
76
|
"icebird": "^0.8.15",
|
|
77
77
|
"unstorage": "^1.17.5",
|