@gscdump/engine 2.0.6 → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/duckdb-node.mjs +6 -2
- package/dist/adapters/filesystem.mjs +4 -2
- package/dist/adapters/hyparquet.mjs +5 -7
- package/dist/analysis-range.mjs +2 -1
- package/dist/duckdb.mjs +2 -1
- package/dist/engine.mjs +7 -5
- package/dist/entities/inspection.mjs +4 -3
- package/dist/entities/query-dim.mjs +4 -2
- package/dist/gc.mjs +1 -1
- package/dist/iceberg/overwrite-writer.mjs +4 -2
- package/dist/iceberg/pyiceberg-runtime.mjs +1 -1
- package/dist/node_modules/.pnpm/fzstd@0.1.1/node_modules/fzstd/esm/index.mjs +6 -4
- package/dist/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.bitreader.mjs +1 -1
- package/dist/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.blocks.mjs +2 -1
- package/dist/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.huffman.mjs +0 -1
- package/dist/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.mjs +23 -20
- package/dist/node_modules/.pnpm/hyparquet-compressors@1.1.1/node_modules/hyparquet-compressors/src/brotli.transform.mjs +1 -1
- package/dist/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash_7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/bloom.mjs +3 -5
- package/dist/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash_7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/column.mjs +4 -2
- package/dist/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash_7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/src/datapage.mjs +11 -4
- package/dist/node_modules/.pnpm/hyparquet-writer@0.16.1_patch_hash_7a5e451ac7634d7546ac56bcc5eadacf7715a6241640c2b5a5bc0cfc9763b8ec/node_modules/hyparquet-writer/types/wkb.d.mts +1 -1
- package/dist/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/src/types.d.mts +1 -1
- package/dist/node_modules/.pnpm/hyparquet@1.26.1/node_modules/hyparquet/types/snappy.d.mts +1 -1
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/column.mjs +7 -5
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/metadata.mjs +3 -2
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/plain.mjs +2 -1
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/rowgroup.mjs +8 -5
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/snappy.mjs +0 -2
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/src/variant.mjs +7 -4
- package/dist/node_modules/.pnpm/hyparquet@1.26.2/node_modules/hyparquet/types/snappy.d.mts +1 -1
- package/dist/node_modules/.pnpm/icebird@0.8.15_patch_hash_d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/catalog/rest.mjs +3 -2
- package/dist/node_modules/.pnpm/icebird@0.8.15_patch_hash_d1902372376af18885e7c7dce807f1ecab86c2f86a9d1ff65d51e3b63b64b057/node_modules/icebird/src/write/write.mjs +1 -1
- package/dist/node_modules/.pnpm/squirreling@0.15.0/node_modules/squirreling/src/ast.d.mts +1 -1
- package/dist/parquet-plan.mjs +0 -2
- package/dist/period/index.mjs +0 -1
- package/dist/resolver/compile.mjs +1 -4
- package/dist/resolver/fragments.mjs +4 -4
- package/dist/resolver/run-query.mjs +2 -1
- package/dist/resolver/schema-drift.mjs +2 -1
- package/dist/rollups/canonical.mjs +3 -2
- package/dist/rollups/core.mjs +6 -4
- package/dist/rollups/indexing.mjs +3 -2
- package/dist/rollups/traffic.mjs +12 -8
- package/dist/rollups/windows.mjs +1 -1
- package/dist/schedule.mjs +1 -1
- package/dist/source/attached-table.mjs +2 -1
- package/dist/source/create-sql-query-source.mjs +4 -2
- package/dist/source/index.mjs +3 -2
- package/dist/sync-config.mjs +1 -1
- package/package.json +5 -5
|
@@ -43,10 +43,14 @@ function createNodeDuckDBHandle(opts = {}) {
|
|
|
43
43
|
return {
|
|
44
44
|
async query(sql, params) {
|
|
45
45
|
const { conn } = await getSingleton(opts);
|
|
46
|
-
if (!params || params.length === 0)
|
|
46
|
+
if (!params || params.length === 0) {
|
|
47
|
+
const result = conn.query(sql);
|
|
48
|
+
return arrowToRows(result);
|
|
49
|
+
}
|
|
47
50
|
const stmt = conn.prepare(sql);
|
|
48
51
|
try {
|
|
49
|
-
|
|
52
|
+
const result = stmt.query(...params);
|
|
53
|
+
return arrowToRows(result);
|
|
50
54
|
} finally {
|
|
51
55
|
stmt.close();
|
|
52
56
|
}
|
|
@@ -79,7 +79,8 @@ function createFilesystemDataSource(opts) {
|
|
|
79
79
|
for await (const p of walkStream(full)) yield p.slice(root.length + 1);
|
|
80
80
|
},
|
|
81
81
|
async head(key) {
|
|
82
|
-
|
|
82
|
+
const path = pathFor(key);
|
|
83
|
+
return stat(path).then((s) => ({ bytes: s.size }), (err) => {
|
|
83
84
|
if (err.code === "ENOENT") return void 0;
|
|
84
85
|
throw err;
|
|
85
86
|
});
|
|
@@ -121,7 +122,8 @@ async function walk(dir, out) {
|
|
|
121
122
|
}
|
|
122
123
|
}
|
|
123
124
|
function lockFileFor(locksDir, scope) {
|
|
124
|
-
|
|
125
|
+
const safe = `${scope.userId}|${scope.siteId ?? ""}|${scope.table}|${scope.partition}`.replace(/[^\w.-]/g, "_");
|
|
126
|
+
return join(locksDir, `${safe}.lock`);
|
|
125
127
|
}
|
|
126
128
|
function createFilesystemManifestStore(opts) {
|
|
127
129
|
const manifestPath = resolve(opts.path);
|
|
@@ -73,13 +73,11 @@ function buildWriteSchema(columns) {
|
|
|
73
73
|
repetition_type
|
|
74
74
|
});
|
|
75
75
|
break;
|
|
76
|
-
case "DOUBLE":
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
});
|
|
82
|
-
break;
|
|
76
|
+
case "DOUBLE": schema.push({
|
|
77
|
+
name: col.name,
|
|
78
|
+
type: "DOUBLE",
|
|
79
|
+
repetition_type
|
|
80
|
+
});
|
|
83
81
|
}
|
|
84
82
|
}
|
|
85
83
|
return schema;
|
package/dist/analysis-range.mjs
CHANGED
|
@@ -17,7 +17,8 @@ function analysisRequestedRange(params, onInvalidFilter) {
|
|
|
17
17
|
const direct = state.filter;
|
|
18
18
|
addRange(ranges, direct?.startDate, direct?.endDate);
|
|
19
19
|
try {
|
|
20
|
-
const
|
|
20
|
+
const normalized = normalizeFilter(state.filter) ?? state.filter;
|
|
21
|
+
const extracted = extractDateRange(normalized);
|
|
21
22
|
addRange(ranges, extracted.startDate, extracted.endDate);
|
|
22
23
|
} catch (error) {
|
|
23
24
|
if (!onInvalidFilter) throw error;
|
package/dist/duckdb.mjs
CHANGED
|
@@ -168,7 +168,8 @@ function createDuckDBExecutor(factory, options = {}) {
|
|
|
168
168
|
await registerBufferedFiles(db, [...bufferedByName.values()], dataSource, registered, signal, options.bufferReadConcurrency);
|
|
169
169
|
endRegister?.({ buffered: registered.length });
|
|
170
170
|
signal?.throwIfAborted();
|
|
171
|
-
const
|
|
171
|
+
const rewritten = rewriteEmptyFileSets(sql, placeholders, table, placeholderTables);
|
|
172
|
+
const finalSql = substituteNamedFiles(rewritten, placeholders);
|
|
172
173
|
const endQuery = profiler?.start("query.run");
|
|
173
174
|
const rows = await db.query(finalSql, params);
|
|
174
175
|
endQuery?.({ rows: rows.length });
|
package/dist/engine.mjs
CHANGED
|
@@ -7,7 +7,7 @@ import { extractParquetPushdown } from "./parquet-pushdown.mjs";
|
|
|
7
7
|
import { buildLogicalPlan } from "gscdump/query/plan";
|
|
8
8
|
import { normalizeUrl } from "gscdump/normalize";
|
|
9
9
|
const URL_PURGE_TABLES = ["pages", "page_queries"];
|
|
10
|
-
const MAX_DAY_BYTES =
|
|
10
|
+
const MAX_DAY_BYTES = 104857600;
|
|
11
11
|
const URL_COLUMNS = /* @__PURE__ */ new Set();
|
|
12
12
|
for (const t of Object.keys(SCHEMAS)) for (const col of SCHEMAS[t].columns) if (col.name === "url") URL_COLUMNS.add(`${t}:url`);
|
|
13
13
|
function normalizeRow(table, row) {
|
|
@@ -179,13 +179,14 @@ function createStorageEngine(opts) {
|
|
|
179
179
|
const endList = profiler?.start("manifest.list", { fileSets: entries.length });
|
|
180
180
|
const perSet = await Promise.all(entries.map(async ([name, ref]) => {
|
|
181
181
|
if (ref.keys !== void 0) return [name, ref.keys];
|
|
182
|
-
|
|
182
|
+
const list = await manifestStore.listLive({
|
|
183
183
|
userId: opts.ctx.userId,
|
|
184
184
|
siteId: opts.ctx.siteId,
|
|
185
185
|
table: ref.table,
|
|
186
186
|
partitions: ref.partitions,
|
|
187
187
|
...opts.searchType !== void 0 ? { searchType: opts.searchType } : {}
|
|
188
|
-
})
|
|
188
|
+
});
|
|
189
|
+
return [name, dedupeOverlappingTiers(list, queryRangeOf(ref.partitions)).map((e) => e.objectKey)];
|
|
189
190
|
}));
|
|
190
191
|
opts.signal?.throwIfAborted();
|
|
191
192
|
const fileKeys = {};
|
|
@@ -250,11 +251,12 @@ function createStorageEngine(opts) {
|
|
|
250
251
|
}, ctx, (ctx.now ?? defaultNow)(), thresholds);
|
|
251
252
|
}
|
|
252
253
|
async function reconcileSubsumed(ctx) {
|
|
253
|
-
const
|
|
254
|
+
const live = await manifestStore.listLive({
|
|
254
255
|
userId: ctx.userId,
|
|
255
256
|
siteId: ctx.siteId,
|
|
256
257
|
table: ctx.table
|
|
257
|
-
})
|
|
258
|
+
});
|
|
259
|
+
const { subsumed } = splitOverlappingTiers(live);
|
|
258
260
|
if (subsumed.length === 0) return {
|
|
259
261
|
retired: 0,
|
|
260
262
|
partitions: []
|
|
@@ -7,7 +7,7 @@ import { GSCDUMP_INDEXING_TRANSITION_FIELDS } from "@gscdump/contracts";
|
|
|
7
7
|
import { classifyCanonicalDifference } from "gscdump";
|
|
8
8
|
const YEAR_MONTH_RE = /^(\d{4})-(\d{2})-/;
|
|
9
9
|
const INSPECTION_EVENT_KEY_RE = /\/inspections\/events\/\d{4}-\d{2}\/[^/]+\.parquet$/;
|
|
10
|
-
const INSPECTION_HISTORY_MAX_BYTES =
|
|
10
|
+
const INSPECTION_HISTORY_MAX_BYTES = 5242880;
|
|
11
11
|
const INSPECTION_PARQUET_COLUMNS = [
|
|
12
12
|
{
|
|
13
13
|
name: "urlHash",
|
|
@@ -265,7 +265,7 @@ function createInspectionStore(opts) {
|
|
|
265
265
|
if (!byMonth.has(month)) byMonth.set(month, []);
|
|
266
266
|
byMonth.get(month).push(r);
|
|
267
267
|
}
|
|
268
|
-
|
|
268
|
+
const shards = [...byMonth].map(([yearMonth, batch]) => {
|
|
269
269
|
const bytes = encodeJsonBigintSafe({
|
|
270
270
|
version: 1,
|
|
271
271
|
records: batch
|
|
@@ -275,7 +275,8 @@ function createInspectionStore(opts) {
|
|
|
275
275
|
key: inspectionHistoryShardKey(ctx, yearMonth, batchId),
|
|
276
276
|
bytes
|
|
277
277
|
};
|
|
278
|
-
})
|
|
278
|
+
});
|
|
279
|
+
await mapEntityIo(shards, (shard) => ds.write(shard.key, shard.bytes));
|
|
279
280
|
},
|
|
280
281
|
async loadHistory(ctx, yearMonth) {
|
|
281
282
|
const keys = await ds.list(inspectionHistoryPrefix(ctx, yearMonth));
|
|
@@ -71,12 +71,14 @@ function createQueryDimStore({ dataSource }) {
|
|
|
71
71
|
};
|
|
72
72
|
},
|
|
73
73
|
async loadMeta(ctx) {
|
|
74
|
-
const
|
|
74
|
+
const key = queryDimMetaKey(ctx);
|
|
75
|
+
const bytes = await readOptional(dataSource, key);
|
|
75
76
|
if (!bytes) return null;
|
|
76
77
|
return JSON.parse(new TextDecoder().decode(bytes));
|
|
77
78
|
},
|
|
78
79
|
async loadRecords(ctx) {
|
|
79
|
-
const
|
|
80
|
+
const key = queryDimParquetKey(ctx);
|
|
81
|
+
const bytes = await readOptional(dataSource, key);
|
|
80
82
|
if (!bytes) return [];
|
|
81
83
|
return (await decodeParquetToRows(bytes)).map((r) => ({
|
|
82
84
|
query: String(r.query),
|
package/dist/gc.mjs
CHANGED
|
@@ -37,9 +37,11 @@ function subprocessBackend(opts = {}) {
|
|
|
37
37
|
return async (job) => {
|
|
38
38
|
const { dirname, join } = await import("node:path");
|
|
39
39
|
const { fileURLToPath } = await import("node:url");
|
|
40
|
+
const python = resolvePyIcebergPython(opts.python);
|
|
41
|
+
const script = opts.writerScript ?? join(dirname(fileURLToPath(import.meta.url)), "..", "scripts", "iceberg-writer.py");
|
|
40
42
|
return runPyIcebergWriter({
|
|
41
|
-
python
|
|
42
|
-
script
|
|
43
|
+
python,
|
|
44
|
+
script,
|
|
43
45
|
job,
|
|
44
46
|
label: "iceberg overwrite subprocess",
|
|
45
47
|
processErrorAsParseFailure: true
|
|
@@ -7,7 +7,7 @@ function resolvePyIcebergPython(override) {
|
|
|
7
7
|
async function runPyIcebergWriter(options) {
|
|
8
8
|
const { execFile } = await import("node:child_process");
|
|
9
9
|
return new Promise((resolve, reject) => {
|
|
10
|
-
execFile(options.python, [options.script], { maxBuffer:
|
|
10
|
+
execFile(options.python, [options.script], { maxBuffer: 67108864 }, (err, stdout, stderr) => {
|
|
11
11
|
let parsed;
|
|
12
12
|
let parseError;
|
|
13
13
|
if (stdout.trim()) try {
|
|
@@ -145,7 +145,8 @@ var rfse = function(dat, bt, mal) {
|
|
|
145
145
|
if (sympos) err(0);
|
|
146
146
|
for (i = 0; i < sz; ++i) {
|
|
147
147
|
var ns = dstate[syms[i]]++;
|
|
148
|
-
|
|
148
|
+
var nb = nbits[i] = al - msb(ns);
|
|
149
|
+
nstate[i] = (ns << nb) - sz;
|
|
149
150
|
}
|
|
150
151
|
return [tpos + 7 >> 3, {
|
|
151
152
|
b: al,
|
|
@@ -382,9 +383,10 @@ var rzb = function(dat, st, out) {
|
|
|
382
383
|
if (btype == 2) {
|
|
383
384
|
var b3 = dat[bt], lbt = b3 & 3, sf = b3 >> 2 & 3;
|
|
384
385
|
var lss = b3 >> 4, lcs = 0, s4 = 0;
|
|
385
|
-
if (lbt < 2)
|
|
386
|
-
|
|
387
|
-
|
|
386
|
+
if (lbt < 2) {
|
|
387
|
+
if (sf & 1) lss |= dat[++bt] << 4 | (sf & 2 && dat[++bt] << 12);
|
|
388
|
+
else lss = b3 >> 3;
|
|
389
|
+
} else {
|
|
388
390
|
s4 = sf;
|
|
389
391
|
if (sf < 2) lss |= (dat[++bt] & 63) << 4, lcs = dat[bt] >> 6 | dat[++bt] << 2;
|
|
390
392
|
else if (sf == 2) lss |= dat[++bt] << 4 | (dat[++bt] & 3) << 12, lcs = dat[bt] >> 2 | dat[++bt] << 6;
|
|
@@ -64,7 +64,7 @@ BrotliBitReader.prototype.readMoreInput = function() {
|
|
|
64
64
|
for (let p = 0; p < 32; p++) this.buf_[dst + bytes_read + p] = 0;
|
|
65
65
|
}
|
|
66
66
|
if (dst === 0) {
|
|
67
|
-
for (let p = 0; p < 32; p++) this.buf_[
|
|
67
|
+
for (let p = 0; p < 32; p++) this.buf_[8192 + p] = this.buf_[p];
|
|
68
68
|
this.buf_ptr_ = BROTLI_READ_SIZE;
|
|
69
69
|
} else this.buf_ptr_ = 0;
|
|
70
70
|
this.bit_end_pos_ += bytes_read << 3;
|
|
@@ -114,7 +114,8 @@ function jumpToByteBoundary(br) {
|
|
|
114
114
|
return !br.readBits(new_bit_pos - br.bit_pos_);
|
|
115
115
|
}
|
|
116
116
|
function readBlockLength(table, index, br) {
|
|
117
|
-
const
|
|
117
|
+
const code = readSymbol(table, index, br);
|
|
118
|
+
const { offset, nbits } = kBlockLengthPrefixCode[code];
|
|
118
119
|
return offset + br.readBits(nbits);
|
|
119
120
|
}
|
|
120
121
|
export { copyUncompressedBlockToOutput, decodeBlockType, decodeMetaBlockLength, decodeVarLenUint8, decodeWindowBits, jumpToByteBoundary, readBlockLength };
|
|
@@ -147,7 +147,6 @@ function readHuffmanCode(alphabet_size, tables, table, br) {
|
|
|
147
147
|
code_lengths[symbols[2]] = 3;
|
|
148
148
|
code_lengths[symbols[3]] = 3;
|
|
149
149
|
} else code_lengths[symbols[0]] = 2;
|
|
150
|
-
break;
|
|
151
150
|
}
|
|
152
151
|
} else {
|
|
153
152
|
const code_length_code_lengths = new Uint8Array(CODE_LENGTH_CODES);
|
|
@@ -240,9 +240,11 @@ function brotli(input, output) {
|
|
|
240
240
|
range_idx -= 2;
|
|
241
241
|
distance_code = -1;
|
|
242
242
|
} else distance_code = 0;
|
|
243
|
-
const
|
|
243
|
+
const insertIndex = kInsertRangeLut[range_idx] + (cmd_code >> 3 & 7);
|
|
244
|
+
const insertPrefix = kInsertLengthPrefixCode[insertIndex];
|
|
244
245
|
const insertLength = insertPrefix.offset + br.readBits(insertPrefix.nbits);
|
|
245
|
-
const
|
|
246
|
+
const copyIndex = kCopyRangeLut[range_idx] + (cmd_code & 7);
|
|
247
|
+
const copyCode = kCopyLengthPrefixCode[copyIndex];
|
|
246
248
|
const copyLength = copyCode.offset + br.readBits(copyCode.nbits);
|
|
247
249
|
prev_byte1 = ringbuffer[pos - 1 & ringbuffer_mask];
|
|
248
250
|
prev_byte2 = ringbuffer[pos - 2 & ringbuffer_mask];
|
|
@@ -292,25 +294,26 @@ function brotli(input, output) {
|
|
|
292
294
|
if (pos < max_backward_distance && max_distance !== max_backward_distance) max_distance = pos;
|
|
293
295
|
else max_distance = max_backward_distance;
|
|
294
296
|
let copy_dst = pos & ringbuffer_mask;
|
|
295
|
-
if (distance > max_distance)
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
297
|
+
if (distance > max_distance) {
|
|
298
|
+
if (copyLength >= minDictionaryWordLength && copyLength <= maxDictionaryWordLength) {
|
|
299
|
+
let offset = offsetsByLength[copyLength];
|
|
300
|
+
const word_id = distance - max_distance - 1;
|
|
301
|
+
const shift = sizeBitsByLength[copyLength];
|
|
302
|
+
const word_idx = word_id & (1 << shift) - 1;
|
|
303
|
+
const transform_idx = word_id >> shift;
|
|
304
|
+
offset += word_idx * copyLength;
|
|
305
|
+
if (transform_idx < kNumTransforms) {
|
|
306
|
+
const len = transformDictionaryWord(ringbuffer, copy_dst, offset, copyLength, transform_idx);
|
|
307
|
+
copy_dst += len;
|
|
308
|
+
pos += len;
|
|
309
|
+
meta_block_remaining_len -= len;
|
|
310
|
+
if (copy_dst >= ringbuffer_end) {
|
|
311
|
+
output.write(ringbuffer, ringbuffer_size);
|
|
312
|
+
for (let _x = 0; _x < copy_dst - ringbuffer_end; _x++) ringbuffer[_x] = ringbuffer[ringbuffer_end + _x];
|
|
313
|
+
}
|
|
314
|
+
} else throw new Error("Invalid backward reference");
|
|
311
315
|
} else throw new Error("Invalid backward reference");
|
|
312
|
-
} else
|
|
313
|
-
else {
|
|
316
|
+
} else {
|
|
314
317
|
if (distance_code > 0) {
|
|
315
318
|
dist_rb[dist_rb_idx & 3] = distance;
|
|
316
319
|
dist_rb_idx++;
|
|
@@ -167,7 +167,7 @@ function transformDictionaryWord(dst, idx, word, len, transform) {
|
|
|
167
167
|
const { prefix } = kTransforms[transform];
|
|
168
168
|
const { suffix } = kTransforms[transform];
|
|
169
169
|
const t = kTransforms[transform].transform;
|
|
170
|
-
let skip = t < kOmitFirst1 ? 0 : t -
|
|
170
|
+
let skip = t < kOmitFirst1 ? 0 : t - 11;
|
|
171
171
|
const start_idx = idx;
|
|
172
172
|
if (skip > len) skip = len;
|
|
173
173
|
let prefix_pos = 0;
|
|
@@ -10,9 +10,7 @@ const SALT = new Uint32Array([
|
|
|
10
10
|
2667333959,
|
|
11
11
|
1550580529
|
|
12
12
|
]);
|
|
13
|
-
const BYTES_PER_BLOCK = 32;
|
|
14
13
|
const MIN_BYTES = 32;
|
|
15
|
-
const MAX_BYTES = 128 * 1024 * 1024;
|
|
16
14
|
function blockIndex(hash, numBlocks) {
|
|
17
15
|
return Number((hash >> 32n) * BigInt(numBlocks) >> 32n);
|
|
18
16
|
}
|
|
@@ -37,8 +35,8 @@ function optimalNumBytes(ndv, fpp) {
|
|
|
37
35
|
if (!(ndv >= 0)) throw new Error(`bloom filter ndv must be >= 0, got ${ndv}`);
|
|
38
36
|
const m = -8 * ndv / Math.log(1 - fpp ** (1 / 8));
|
|
39
37
|
let numBits = Math.ceil(m);
|
|
40
|
-
if (!isFinite(numBits) || numBits >
|
|
41
|
-
const blockBits =
|
|
38
|
+
if (!isFinite(numBits) || numBits > 1073741824) numBits = 1073741824;
|
|
39
|
+
const blockBits = 256;
|
|
42
40
|
numBits = Math.ceil(numBits / blockBits) * blockBits;
|
|
43
41
|
let numBytes = numBits >> 3;
|
|
44
42
|
if (numBytes < MIN_BYTES) numBytes = MIN_BYTES;
|
|
@@ -46,7 +44,7 @@ function optimalNumBytes(ndv, fpp) {
|
|
|
46
44
|
return numBytes;
|
|
47
45
|
}
|
|
48
46
|
var BloomBuilder = class {
|
|
49
|
-
constructor(element, { fpp = .01, maxBytes =
|
|
47
|
+
constructor(element, { fpp = .01, maxBytes = 1048576 } = {}) {
|
|
50
48
|
this.element = element;
|
|
51
49
|
this.fpp = fpp;
|
|
52
50
|
this.maxBytes = maxBytes;
|
|
@@ -15,7 +15,8 @@ function writeColumn({ writer, column, pageData }) {
|
|
|
15
15
|
const geospatial_statistics = stats && isGeospatial ? geospatialStatistics(values) : void 0;
|
|
16
16
|
let bloomFilter;
|
|
17
17
|
if (column.bloomFilter) {
|
|
18
|
-
const
|
|
18
|
+
const opts = typeof column.bloomFilter === "object" ? column.bloomFilter : void 0;
|
|
19
|
+
const builder = new BloomBuilder(element, opts);
|
|
19
20
|
for (const v of values) builder.insert(v);
|
|
20
21
|
bloomFilter = builder.finalize();
|
|
21
22
|
}
|
|
@@ -29,7 +30,8 @@ function writeColumn({ writer, column, pageData }) {
|
|
|
29
30
|
writeType = "INT32";
|
|
30
31
|
encoding = "RLE_DICTIONARY";
|
|
31
32
|
dictionary_page_offset = BigInt(writer.offset);
|
|
32
|
-
|
|
33
|
+
const unconverted = unconvert(element, dictionary);
|
|
34
|
+
writeDictionaryPage(writer, column, unconverted);
|
|
33
35
|
} else {
|
|
34
36
|
writeValues = unconvert(element, values);
|
|
35
37
|
encoding = userEncoding ?? (type === "BOOLEAN" && values.length > 16 ? "RLE" : "PLAIN");
|
|
@@ -59,7 +59,7 @@ function writeDataPageV2({ writer, column, encoding, pageData }) {
|
|
|
59
59
|
writer.appendBytes(compressedBytes);
|
|
60
60
|
}
|
|
61
61
|
function writePageHeader(writer, header) {
|
|
62
|
-
|
|
62
|
+
const compact = {
|
|
63
63
|
field_1: PageTypes.indexOf(header.type),
|
|
64
64
|
field_2: header.uncompressed_page_size,
|
|
65
65
|
field_3: header.compressed_page_size,
|
|
@@ -83,7 +83,8 @@ function writePageHeader(writer, header) {
|
|
|
83
83
|
field_6: header.data_page_header_v2.repetition_levels_byte_length,
|
|
84
84
|
field_7: header.data_page_header_v2.is_compressed ? void 0 : false
|
|
85
85
|
}
|
|
86
|
-
}
|
|
86
|
+
};
|
|
87
|
+
serializeTCompactProtocol(writer, compact);
|
|
87
88
|
}
|
|
88
89
|
function writeLevels(writer, column, dataPage) {
|
|
89
90
|
const { schemaPath } = column;
|
|
@@ -99,9 +100,15 @@ function writeLevels(writer, column, dataPage) {
|
|
|
99
100
|
}
|
|
100
101
|
const maxRepetitionLevel = getMaxRepetitionLevel(schemaPath);
|
|
101
102
|
let repetition_levels_byte_length = 0;
|
|
102
|
-
if (maxRepetitionLevel)
|
|
103
|
+
if (maxRepetitionLevel) {
|
|
104
|
+
const bitWidth = Math.ceil(Math.log2(maxRepetitionLevel + 1));
|
|
105
|
+
repetition_levels_byte_length = writeRleBitPackedHybrid(writer, repetitionLevels, bitWidth);
|
|
106
|
+
}
|
|
103
107
|
let definition_levels_byte_length = 0;
|
|
104
|
-
if (maxDefinitionLevel)
|
|
108
|
+
if (maxDefinitionLevel) {
|
|
109
|
+
const bitWidth = Math.ceil(Math.log2(maxDefinitionLevel + 1));
|
|
110
|
+
definition_levels_byte_length = writeRleBitPackedHybrid(writer, definitionLevels, bitWidth);
|
|
111
|
+
}
|
|
105
112
|
return {
|
|
106
113
|
definition_levels_byte_length,
|
|
107
114
|
repetition_levels_byte_length,
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -56,7 +56,8 @@ function readPage(reader, header, columnDecoder, dictionary, previousChunk, page
|
|
|
56
56
|
const daph = header.data_page_header;
|
|
57
57
|
if (!daph) throw new Error("parquet data page header is undefined");
|
|
58
58
|
if (pageStart > daph.num_values && isFlatColumn(schemaPath)) return { skipped: daph.num_values };
|
|
59
|
-
const
|
|
59
|
+
const page = decompressPage(compressedBytes, Number(header.uncompressed_page_size), codec, compressors);
|
|
60
|
+
const { definitionLevels, repetitionLevels, dataPage } = readDataPage(page, daph, columnDecoder);
|
|
60
61
|
const values = convertWithDictionary(dataPage, dictionary, daph.encoding, columnDecoder);
|
|
61
62
|
return {
|
|
62
63
|
skipped: 0,
|
|
@@ -76,12 +77,13 @@ function readPage(reader, header, columnDecoder, dictionary, previousChunk, page
|
|
|
76
77
|
const diph = header.dictionary_page_header;
|
|
77
78
|
if (!diph) throw new Error("parquet dictionary page header is undefined");
|
|
78
79
|
const page = decompressPage(compressedBytes, Number(header.uncompressed_page_size), codec, compressors);
|
|
80
|
+
const reader = {
|
|
81
|
+
view: new DataView(page.buffer, page.byteOffset, page.byteLength),
|
|
82
|
+
offset: 0
|
|
83
|
+
};
|
|
79
84
|
return {
|
|
80
85
|
skipped: 0,
|
|
81
|
-
data: readPlain(
|
|
82
|
-
view: new DataView(page.buffer, page.byteOffset, page.byteLength),
|
|
83
|
-
offset: 0
|
|
84
|
-
}, type, diph.num_values, element.type_length)
|
|
86
|
+
data: readPlain(reader, type, diph.num_values, element.type_length)
|
|
85
87
|
};
|
|
86
88
|
} else throw new Error(`parquet unsupported page type: ${header.type}`);
|
|
87
89
|
}
|
|
@@ -44,10 +44,11 @@ function parquetMetadata(arrayBuffer, { parsers, geoparquet = true } = {}) {
|
|
|
44
44
|
const metadataLengthOffset = view.byteLength - 8;
|
|
45
45
|
const metadataLength = view.getUint32(metadataLengthOffset, true);
|
|
46
46
|
if (metadataLength > view.byteLength - 8) throw new Error(`parquet metadata length ${metadataLength} exceeds available buffer ${view.byteLength - 8}`);
|
|
47
|
-
const
|
|
47
|
+
const reader = {
|
|
48
48
|
view,
|
|
49
49
|
offset: metadataLengthOffset - metadataLength
|
|
50
|
-
}
|
|
50
|
+
};
|
|
51
|
+
const metadata = deserializeTCompactProtocol(reader);
|
|
51
52
|
const version = metadata.field_1;
|
|
52
53
|
const schema = metadata.field_2.map((field) => ({
|
|
53
54
|
type: ParquetTypes[field.field_1],
|
|
@@ -17,7 +17,8 @@ function readPlainBoolean(reader, count) {
|
|
|
17
17
|
for (let i = 0; i < count; i++) {
|
|
18
18
|
const byteOffset = reader.offset + (i / 8 | 0);
|
|
19
19
|
const bitOffset = i % 8;
|
|
20
|
-
|
|
20
|
+
const byte = reader.view.getUint8(byteOffset);
|
|
21
|
+
values[i] = (byte & 1 << bitOffset) !== 0;
|
|
21
22
|
}
|
|
22
23
|
reader.offset += Math.ceil(count / 8);
|
|
23
24
|
return values;
|
|
@@ -25,10 +25,11 @@ function readRowGroup(options, { metadata }, groupPlan) {
|
|
|
25
25
|
asyncColumns.push({
|
|
26
26
|
pathInSchema,
|
|
27
27
|
data: Promise.resolve(options.file.slice(startByte, endByte)).then((buffer) => {
|
|
28
|
-
|
|
28
|
+
const reader = {
|
|
29
29
|
view: new DataView(buffer),
|
|
30
30
|
offset: 0
|
|
31
|
-
}
|
|
31
|
+
};
|
|
32
|
+
return readColumn(reader, groupPlan, columnDecoder, options.onPage);
|
|
32
33
|
})
|
|
33
34
|
});
|
|
34
35
|
continue;
|
|
@@ -55,15 +56,17 @@ function readRowGroup(options, { metadata }, groupPlan) {
|
|
|
55
56
|
}
|
|
56
57
|
if (skipped < 0) skipped = 0;
|
|
57
58
|
const buffer = await options.file.slice(startByte, endByte);
|
|
58
|
-
const
|
|
59
|
+
const reader = {
|
|
59
60
|
view: new DataView(buffer),
|
|
60
61
|
offset: 0
|
|
61
|
-
}
|
|
62
|
+
};
|
|
63
|
+
const adjustedGroupPlan = skipped ? {
|
|
62
64
|
...groupPlan,
|
|
63
65
|
groupStart: groupPlan.groupStart + skipped,
|
|
64
66
|
selectStart: groupPlan.selectStart - skipped,
|
|
65
67
|
selectEnd: groupPlan.selectEnd - skipped
|
|
66
|
-
} : groupPlan
|
|
68
|
+
} : groupPlan;
|
|
69
|
+
const { data, skipped: columnSkipped } = readColumn(reader, adjustedGroupPlan, columnDecoder, options.onPage);
|
|
67
70
|
return {
|
|
68
71
|
data,
|
|
69
72
|
skipped: skipped + columnSkipped
|
|
@@ -56,8 +56,6 @@ function snappyUncompress(input, output) {
|
|
|
56
56
|
len = (c >>> 2) + 1;
|
|
57
57
|
offset = input[pos] + (input[pos + 1] << 8) + (input[pos + 2] << 16) + (input[pos + 3] << 24);
|
|
58
58
|
pos += 4;
|
|
59
|
-
break;
|
|
60
|
-
default: break;
|
|
61
59
|
}
|
|
62
60
|
if (offset === 0 || isNaN(offset)) throw new Error(`invalid offset ${offset} pos ${pos} inputLength ${inputLength}`);
|
|
63
61
|
if (offset > outPos) throw new Error("cannot copy from before start of buffer");
|
|
@@ -191,10 +191,13 @@ function readVariantArray(reader, header, metadata, parsers) {
|
|
|
191
191
|
for (let i = 0; i < offsets.length; i++) offsets[i] = readUnsigned(reader, offsetWidth);
|
|
192
192
|
const valuesStart = reader.offset;
|
|
193
193
|
const result = new Array(numElements);
|
|
194
|
-
for (let i = 0; i < numElements; i++)
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
194
|
+
for (let i = 0; i < numElements; i++) {
|
|
195
|
+
const valueReader = {
|
|
196
|
+
view: reader.view,
|
|
197
|
+
offset: valuesStart + offsets[i]
|
|
198
|
+
};
|
|
199
|
+
result[i] = readVariant(valueReader, metadata, parsers);
|
|
200
|
+
}
|
|
198
201
|
reader.offset = valuesStart + offsets[offsets.length - 1];
|
|
199
202
|
return result;
|
|
200
203
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
|
@@ -10,11 +10,12 @@ async function restCatalogCreateTable(ctx, { namespace, table, schema, location,
|
|
|
10
10
|
if (writeOrder !== void 0) body["write-order"] = writeOrder;
|
|
11
11
|
if (stageCreate !== void 0) body["stage-create"] = stageCreate;
|
|
12
12
|
if (properties !== void 0) body.properties = properties;
|
|
13
|
-
const
|
|
13
|
+
const res = await restFetch(ctx, `namespaces/${ns}/tables`, {
|
|
14
14
|
method: "POST",
|
|
15
15
|
headers: { "content-type": "application/json" },
|
|
16
16
|
body: stringifyIcebergJson(body)
|
|
17
|
-
})
|
|
17
|
+
});
|
|
18
|
+
const responseBody = parseIcebergJson(await res.text());
|
|
18
19
|
return {
|
|
19
20
|
metadataLocation: responseBody["metadata-location"],
|
|
20
21
|
metadata: responseBody.metadata,
|
|
@@ -12,7 +12,7 @@ Object.freeze({
|
|
|
12
12
|
initialMs: 50,
|
|
13
13
|
maxMs: 3e3,
|
|
14
14
|
factor: 2,
|
|
15
|
-
totalTimeoutMs:
|
|
15
|
+
totalTimeoutMs: 18e5
|
|
16
16
|
});
|
|
17
17
|
async function icebergCreateTable({ catalog, namespace, table, tableUrl, schema, partitionSpec, sortOrder, properties, formatVersion, stageCreate }) {
|
|
18
18
|
if (catalog.type === "rest") {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export {}
|
package/dist/parquet-plan.mjs
CHANGED
|
@@ -32,7 +32,6 @@ function buildDimensionWhere(filters, table) {
|
|
|
32
32
|
case "excludingRegex":
|
|
33
33
|
clauses.push(`NOT regexp_matches(${column}, ?)`);
|
|
34
34
|
params.push(filter.expression);
|
|
35
|
-
break;
|
|
36
35
|
}
|
|
37
36
|
}
|
|
38
37
|
return {
|
|
@@ -73,7 +72,6 @@ function buildHaving(filters) {
|
|
|
73
72
|
case "metricBetween":
|
|
74
73
|
clauses.push(`${expr} >= ? AND ${expr} <= ?`);
|
|
75
74
|
params.push(filter.expression, filter.expression2 ?? filter.expression);
|
|
76
|
-
break;
|
|
77
75
|
}
|
|
78
76
|
}
|
|
79
77
|
return {
|
package/dist/period/index.mjs
CHANGED
|
@@ -171,7 +171,6 @@ function resolveToSQLOptimized(state, options) {
|
|
|
171
171
|
case "position":
|
|
172
172
|
outerSelect.push(sql.raw("sum_position / NULLIF(impressions, 0) + 1 as position"));
|
|
173
173
|
outerTotals.push(sql.raw("SUM(sum_position) OVER() / NULLIF(SUM(impressions) OVER(), 0) + 1 as totalPosition"));
|
|
174
|
-
break;
|
|
175
174
|
}
|
|
176
175
|
let orderByColumnOverride;
|
|
177
176
|
const orderColumn = state.orderBy?.column;
|
|
@@ -187,9 +186,7 @@ function resolveToSQLOptimized(state, options) {
|
|
|
187
186
|
case "ctr":
|
|
188
187
|
outerSelect.push(sql.raw(`CAST(clicks AS REAL) / NULLIF(impressions, 0) as "${orderByColumnOverride}"`));
|
|
189
188
|
break;
|
|
190
|
-
case "position":
|
|
191
|
-
outerSelect.push(sql.raw(`sum_position / NULLIF(impressions, 0) + 1 as "${orderByColumnOverride}"`));
|
|
192
|
-
break;
|
|
189
|
+
case "position": outerSelect.push(sql.raw(`sum_position / NULLIF(impressions, 0) + 1 as "${orderByColumnOverride}"`));
|
|
193
190
|
}
|
|
194
191
|
}
|
|
195
192
|
outerSelect.push(sql.raw("COUNT(*) OVER() as totalCount"));
|
|
@@ -53,7 +53,9 @@ function createSqlFragments(config) {
|
|
|
53
53
|
function fromSql(tableKey, options = {}) {
|
|
54
54
|
const base = tableRef(tableKey);
|
|
55
55
|
if (!options.queryCanonical || queryCanonicalSource !== "queryDim") return base;
|
|
56
|
-
|
|
56
|
+
const dimTable = queryDimTableRef?.() ?? sql.raw(`${quoteIdent(QUERY_DIM_ALIAS)}`);
|
|
57
|
+
const joinOn = queryDimSiteScoped ? sql`${colRef(tableKey, "query")} = ${qualifiedRaw(QUERY_DIM_ALIAS, "query")} AND ${colRef(tableKey, "site_id")} = ${qualifiedRaw(QUERY_DIM_ALIAS, "site_id")}` : sql`${colRef(tableKey, "query")} = ${qualifiedRaw(QUERY_DIM_ALIAS, "query")}`;
|
|
58
|
+
return sql`${base} LEFT JOIN ${dimTable} ON ${joinOn}`;
|
|
57
59
|
}
|
|
58
60
|
function dateColRef(tableKey) {
|
|
59
61
|
return colRef(tableKey, "date");
|
|
@@ -169,9 +171,7 @@ function createSqlFragments(config) {
|
|
|
169
171
|
case "includingRegex":
|
|
170
172
|
preds.push(regexPredicate(patternExpr, f.expression, false));
|
|
171
173
|
break;
|
|
172
|
-
case "excludingRegex":
|
|
173
|
-
preds.push(regexPredicate(patternExpr, f.expression, true));
|
|
174
|
-
break;
|
|
174
|
+
case "excludingRegex": preds.push(regexPredicate(patternExpr, f.expression, true));
|
|
175
175
|
}
|
|
176
176
|
}
|
|
177
177
|
return preds;
|
|
@@ -126,8 +126,9 @@ async function runOptimizedQuery(runSQL, ctx, state, dateRange, options = {}) {
|
|
|
126
126
|
const canonicalDecision = decideCanonicalSource(state, probe.capabilities, options, dateRange.endDate);
|
|
127
127
|
const useCanonicalSource = canonicalDecision.kind === "canonical-rollup";
|
|
128
128
|
const source = useCanonicalSource ? canonicalDecision : decidePrimarySource(options, dateRange, sourceFallbacks(canonicalDecision));
|
|
129
|
+
const adapter = useCanonicalSource ? createParquetResolverAdapter({ queryCanonicalSource: "column" }) : probe;
|
|
129
130
|
const optimized = resolveToSQLOptimized(state, {
|
|
130
|
-
adapter
|
|
131
|
+
adapter,
|
|
131
132
|
siteId: void 0
|
|
132
133
|
});
|
|
133
134
|
const extras = buildExtrasQueries(state, {
|
|
@@ -2,7 +2,8 @@ import { SCHEMAS } from "../schema.mjs";
|
|
|
2
2
|
function assertSchemaInSync(options) {
|
|
3
3
|
const { label, schema, tableKeyToName, mode } = options;
|
|
4
4
|
for (const [key, table] of Object.entries(schema)) {
|
|
5
|
-
const
|
|
5
|
+
const tableName = tableKeyToName(key);
|
|
6
|
+
const sourceCols = SCHEMAS[tableName].columns.map((c) => c.name).sort();
|
|
6
7
|
const drizzleCols = Object.keys(table[Symbol.for("drizzle:Columns")] ?? {}).sort();
|
|
7
8
|
const missing = sourceCols.filter((c) => !drizzleCols.includes(c));
|
|
8
9
|
const extra = mode === "exact" ? drizzleCols.filter((c) => !sourceCols.includes(c)) : [];
|
|
@@ -221,11 +221,12 @@ async function rebuildCanonicalDailyResumable(opts) {
|
|
|
221
221
|
const pageRows = opts.pageRows ?? 7e4;
|
|
222
222
|
const maxWindowDays = opts.maxWindowDays ?? 7;
|
|
223
223
|
const shardCount = Math.max(1, Math.floor(opts.shardCount ?? 1));
|
|
224
|
-
const
|
|
224
|
+
const parts = await engine.listPartitions({
|
|
225
225
|
ctx,
|
|
226
226
|
table: "queries",
|
|
227
227
|
...sType
|
|
228
|
-
})
|
|
228
|
+
});
|
|
229
|
+
const windows = planRollupWindows(parts.map((p) => ({
|
|
229
230
|
partition: p.partition,
|
|
230
231
|
bytes: p.bytes
|
|
231
232
|
})), void 0, maxWindowDays);
|
package/dist/rollups/core.mjs
CHANGED
|
@@ -67,13 +67,14 @@ async function rebuildRollups(opts) {
|
|
|
67
67
|
parquetKey,
|
|
68
68
|
rowCount: rows.length
|
|
69
69
|
};
|
|
70
|
-
const
|
|
70
|
+
const envelope = {
|
|
71
71
|
version: 1,
|
|
72
72
|
id: def.id,
|
|
73
73
|
builtAt,
|
|
74
74
|
windowDays: def.windowDays,
|
|
75
75
|
payload: pointer
|
|
76
|
-
}
|
|
76
|
+
};
|
|
77
|
+
const envelopeBytes = encodeJsonBigintSafe(envelope);
|
|
77
78
|
const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
|
|
78
79
|
await opts.dataSource.write(key, envelopeBytes);
|
|
79
80
|
results.push({
|
|
@@ -86,13 +87,14 @@ async function rebuildRollups(opts) {
|
|
|
86
87
|
});
|
|
87
88
|
continue;
|
|
88
89
|
}
|
|
89
|
-
const
|
|
90
|
+
const envelope = {
|
|
90
91
|
version: 1,
|
|
91
92
|
id: def.id,
|
|
92
93
|
builtAt,
|
|
93
94
|
windowDays: def.windowDays,
|
|
94
95
|
payload
|
|
95
|
-
}
|
|
96
|
+
};
|
|
97
|
+
const bytes = encodeJsonBigintSafe(envelope);
|
|
96
98
|
const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
|
|
97
99
|
await opts.dataSource.write(key, bytes);
|
|
98
100
|
results.push({
|
|
@@ -116,11 +116,12 @@ const indexPercentRollup = {
|
|
|
116
116
|
const currentMembership = "read_parquet({{SITEMAP_BASES}}, union_by_name = true)";
|
|
117
117
|
const cutoff = utcDateMinusDays(windowAnchorMs, 90);
|
|
118
118
|
const factSearchType = searchType ?? "web";
|
|
119
|
-
const
|
|
119
|
+
const pagesParts = await engine.listPartitions({
|
|
120
120
|
ctx,
|
|
121
121
|
table: "pages",
|
|
122
122
|
searchType: factSearchType
|
|
123
|
-
})
|
|
123
|
+
});
|
|
124
|
+
const pagesPartitions = partitionsInRange(pagesParts, cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
124
125
|
const numerator = await engine.runSQL({
|
|
125
126
|
ctx,
|
|
126
127
|
table: "pages",
|
package/dist/rollups/traffic.mjs
CHANGED
|
@@ -111,11 +111,12 @@ const topPages28dRollup = {
|
|
|
111
111
|
windowDays: 28,
|
|
112
112
|
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
113
113
|
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
114
|
-
const
|
|
114
|
+
const parts = await engine.listPartitions({
|
|
115
115
|
ctx,
|
|
116
116
|
table: "pages",
|
|
117
117
|
...searchType !== void 0 ? { searchType } : {}
|
|
118
|
-
})
|
|
118
|
+
});
|
|
119
|
+
const partitions = partitionsInRange(parts, cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
119
120
|
if (partitions.length === 0) return [];
|
|
120
121
|
return (await engine.runSQL({
|
|
121
122
|
ctx,
|
|
@@ -150,11 +151,12 @@ const topCountries28dRollup = {
|
|
|
150
151
|
windowDays: 28,
|
|
151
152
|
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
152
153
|
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
153
|
-
const
|
|
154
|
+
const parts = await engine.listPartitions({
|
|
154
155
|
ctx,
|
|
155
156
|
table: "countries",
|
|
156
157
|
...searchType !== void 0 ? { searchType } : {}
|
|
157
|
-
})
|
|
158
|
+
});
|
|
159
|
+
const partitions = partitionsInRange(parts, cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
158
160
|
if (partitions.length === 0) return [];
|
|
159
161
|
return (await engine.runSQL({
|
|
160
162
|
ctx,
|
|
@@ -189,11 +191,12 @@ const topKeywords28dRollup = {
|
|
|
189
191
|
windowDays: 28,
|
|
190
192
|
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
191
193
|
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
192
|
-
const
|
|
194
|
+
const parts = await engine.listPartitions({
|
|
193
195
|
ctx,
|
|
194
196
|
table: "queries",
|
|
195
197
|
...searchType !== void 0 ? { searchType } : {}
|
|
196
|
-
})
|
|
198
|
+
});
|
|
199
|
+
const partitions = partitionsInRange(parts, cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
197
200
|
if (partitions.length === 0) return [];
|
|
198
201
|
return (await engine.runSQL({
|
|
199
202
|
ctx,
|
|
@@ -252,11 +255,12 @@ const topKeywords28dParquetRollup = {
|
|
|
252
255
|
parquetSortKey: ["clicks"],
|
|
253
256
|
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
254
257
|
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
255
|
-
const
|
|
258
|
+
const parts = await engine.listPartitions({
|
|
256
259
|
ctx,
|
|
257
260
|
table: "queries",
|
|
258
261
|
...searchType !== void 0 ? { searchType } : {}
|
|
259
|
-
})
|
|
262
|
+
});
|
|
263
|
+
const partitions = partitionsInRange(parts, cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
260
264
|
if (partitions.length === 0) return [];
|
|
261
265
|
return (await engine.runSQL({
|
|
262
266
|
ctx,
|
package/dist/rollups/windows.mjs
CHANGED
package/dist/schedule.mjs
CHANGED
|
@@ -48,7 +48,8 @@ function createAttachedTableSource(runner, options) {
|
|
|
48
48
|
if (missing.length > 0) throw new AttachedTableMissingError(missing);
|
|
49
49
|
}
|
|
50
50
|
const rewritten = rewriteForTableSource(sql, schema, fileSets);
|
|
51
|
-
|
|
51
|
+
const rows = await runner.query(rewritten, params ?? [], signal);
|
|
52
|
+
return coerceRows(rows);
|
|
52
53
|
}
|
|
53
54
|
};
|
|
54
55
|
}
|
|
@@ -18,10 +18,12 @@ function createSqlQuerySource(options) {
|
|
|
18
18
|
siteId,
|
|
19
19
|
searchType
|
|
20
20
|
});
|
|
21
|
-
|
|
21
|
+
const rows = await execute(resolved.sql, resolved.params);
|
|
22
|
+
return coerceRows(rows);
|
|
22
23
|
},
|
|
23
24
|
async executeSql(sql, params) {
|
|
24
|
-
|
|
25
|
+
const rows = await execute(sql, params ?? []);
|
|
26
|
+
return coerceRows(rows);
|
|
25
27
|
}
|
|
26
28
|
};
|
|
27
29
|
}
|
package/dist/source/index.mjs
CHANGED
|
@@ -37,10 +37,11 @@ function createEngineQuerySource(options) {
|
|
|
37
37
|
const filterDims = getFilterDimensions(state.filter, isMetricDimension);
|
|
38
38
|
assertDimensionsSupported([...state.dimensions, ...filterDims], "stored", "engine query source");
|
|
39
39
|
if (state.dimensions.includes("queryCanonical") || filterDims.includes("queryCanonical")) throw new Error("engine query source does not support queryCanonical; use browser/sqlite query sources for derived dimensions");
|
|
40
|
-
|
|
40
|
+
const result = await engine.query({
|
|
41
41
|
...ctx,
|
|
42
42
|
...searchType !== void 0 ? { searchType } : {}
|
|
43
|
-
}, state)
|
|
43
|
+
}, state);
|
|
44
|
+
return coerceRows(result.rows);
|
|
44
45
|
},
|
|
45
46
|
async executeSql(sql, params, opts) {
|
|
46
47
|
const fileSets = opts?.fileSets;
|
package/dist/sync-config.mjs
CHANGED
|
@@ -74,7 +74,7 @@ function getTablesForTier(tier) {
|
|
|
74
74
|
}
|
|
75
75
|
function getDateWeight(date, now = /* @__PURE__ */ new Date()) {
|
|
76
76
|
const target = new Date(date);
|
|
77
|
-
const daysAgo = Math.floor((now.getTime() - target.getTime()) /
|
|
77
|
+
const daysAgo = Math.floor((now.getTime() - target.getTime()) / 864e5);
|
|
78
78
|
if (daysAgo <= 3) return "fresh";
|
|
79
79
|
if (daysAgo <= 60) return "recent";
|
|
80
80
|
return "historical";
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@gscdump/engine",
|
|
3
3
|
"type": "module",
|
|
4
|
-
"version": "2.
|
|
4
|
+
"version": "2.1.1",
|
|
5
5
|
"description": "Append-only Parquet/DuckDB storage engine + planner + adapters for the gscdump pipeline. Node + edge runtimes; opt-in heavy peers.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Harlan Wilton",
|
|
@@ -186,13 +186,13 @@
|
|
|
186
186
|
"dependencies": {
|
|
187
187
|
"drizzle-orm": "1.0.0-rc.4",
|
|
188
188
|
"proper-lockfile": "^4.1.2",
|
|
189
|
-
"@gscdump/
|
|
190
|
-
"
|
|
191
|
-
"gscdump": "^2.
|
|
189
|
+
"@gscdump/lakehouse": "^2.1.1",
|
|
190
|
+
"gscdump": "^2.1.1",
|
|
191
|
+
"@gscdump/contracts": "^2.1.1"
|
|
192
192
|
},
|
|
193
193
|
"devDependencies": {
|
|
194
194
|
"@duckdb/duckdb-wasm": "1.33.1-dev57.0",
|
|
195
|
-
"@types/node": "^26.
|
|
195
|
+
"@types/node": "^26.2.0",
|
|
196
196
|
"@types/proper-lockfile": "^4.1.4",
|
|
197
197
|
"aws4fetch": "^1.0.20",
|
|
198
198
|
"hyparquet": "^1.26.2",
|