@gscdump/engine 1.4.11 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/entities/empty-types.d.mts +22 -0
- package/dist/entities/empty-types.mjs +58 -0
- package/dist/entities/indexing-metadata.d.mts +26 -0
- package/dist/entities/indexing-metadata.mjs +31 -0
- package/dist/entities/inspection.d.mts +240 -0
- package/dist/entities/inspection.mjs +443 -0
- package/dist/entities/io.mjs +16 -0
- package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
- package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
- package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
- package/dist/entities/sitemap-shared.d.mts +199 -0
- package/dist/entities/sitemap-shared.mjs +243 -0
- package/dist/entities/sitemap-write.d.mts +3 -0
- package/dist/entities/sitemap-write.mjs +528 -0
- package/dist/entities/sitemap.d.mts +3 -0
- package/dist/entities/sitemap.mjs +91 -0
- package/dist/entities.d.mts +9 -481
- package/dist/entities.mjs +7 -1380
- package/dist/rollups/canonical.d.mts +71 -0
- package/dist/rollups/canonical.mjs +335 -0
- package/dist/rollups/core.d.mts +201 -0
- package/dist/rollups/core.mjs +116 -0
- package/dist/rollups/dates.mjs +11 -0
- package/dist/rollups/defaults.d.mts +11 -0
- package/dist/rollups/defaults.mjs +17 -0
- package/dist/rollups/hourly.d.mts +38 -0
- package/dist/rollups/hourly.mjs +38 -0
- package/dist/rollups/indexing.d.mts +46 -0
- package/dist/rollups/indexing.mjs +357 -0
- package/dist/rollups/traffic.d.mts +38 -0
- package/dist/rollups/traffic.mjs +289 -0
- package/dist/rollups/windows.d.mts +90 -0
- package/dist/rollups/windows.mjs +176 -0
- package/dist/rollups.d.mts +8 -471
- package/dist/rollups.mjs +7 -1316
- package/package.json +4 -4
- /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import { DataSource } from "../storage.mjs";
|
|
2
|
+
import "../contracts.mjs";
|
|
3
|
+
import { RollupDef, RollupEngine } from "./core.mjs";
|
|
4
|
+
import { TenantCtx } from "@gscdump/contracts";
|
|
5
|
+
import { SearchType } from "gscdump/query";
|
|
6
|
+
declare const queryCanonicalVariantsRollup: RollupDef;
|
|
7
|
+
/**
|
|
8
|
+
* Canonical-grained fact aggregate (ADR-0018 Gap 2): pre-sums the raw
|
|
9
|
+
* `(query × date)` query rows to `(query_canonical × date)`, so canonical-
|
|
10
|
+
* primary top/gaining/losing reads a small pre-aggregated table instead of
|
|
11
|
+
* re-collapsing variants on every request. Metrics are additive, so summing
|
|
12
|
+
* these per-date sums over a window is exact — identical to aggregating the raw
|
|
13
|
+
* rows.
|
|
14
|
+
*
|
|
15
|
+
* Null-free by construction: groups by the versioned query dimension when it
|
|
16
|
+
* exists, with raw query as the fallback, so the rollup never carries a NULL/''
|
|
17
|
+
* canonical bucket and the read path can treat the rollup's `query_canonical`
|
|
18
|
+
* column as already-derived.
|
|
19
|
+
*
|
|
20
|
+
* Date-grained full history (`windowDays: null`): one rollup serves every date
|
|
21
|
+
* range (reads filter by `date`) and both windows of a comparison. Opt-in (not
|
|
22
|
+
* in `DEFAULT_ROLLUPS`); the host points the main query's file set at it for
|
|
23
|
+
* queries the rollup covers (see `canonicalRollupCovers` /
|
|
24
|
+
* `RunOptimizedQueryOptions.canonicalSource`).
|
|
25
|
+
*/
|
|
26
|
+
declare const queryCanonicalDailyRollup: RollupDef;
|
|
27
|
+
/**
|
|
28
|
+
* Resumable, cross-invocation build of `query_canonical_daily` for a high-
|
|
29
|
+
* cardinality site whose full windowed build exceeds one job reservation (300s).
|
|
30
|
+
*
|
|
31
|
+
* Each call builds from `(windowOffset, pageOffset)` until `deadlineMs`, writes
|
|
32
|
+
* that batch's rows to a PART parquet, and returns `{ done:false, nextWindowOffset,
|
|
33
|
+
* nextPageOffset }` for the caller to re-enqueue. When the last window is fully
|
|
34
|
+
* paged it publishes a multi-file envelope listing every part (parts are disjoint
|
|
35
|
+
* by `(query_canonical, date)`, so the read path just unions them — no merge),
|
|
36
|
+
* returning `{ done:true }`. `builtAt` MUST be stable across the continuation chain
|
|
37
|
+
* (it versions both the part keys and the final rollup key).
|
|
38
|
+
*
|
|
39
|
+
* INTRA-WINDOW resumability: the deadline is checked between raw-query hash shards,
|
|
40
|
+
* not just between date windows. A single high-cardinality day can spend a full
|
|
41
|
+
* reservation inside one grouped/sorted aggregate before the deadline check gets
|
|
42
|
+
* control back. `pageOffset` is the next shard index for the current window, so a
|
|
43
|
+
* continuation resumes the SAME day at the next shard. Parts are keyed by
|
|
44
|
+
* `(windowOffset, pageOffset)`; multiple parts may contain the same canonical/date
|
|
45
|
+
* from different raw-query shards, and rollup reads sum over the union.
|
|
46
|
+
*/
|
|
47
|
+
declare function rebuildCanonicalDailyResumable(opts: {
|
|
48
|
+
engine: RollupEngine;
|
|
49
|
+
ctx: TenantCtx;
|
|
50
|
+
dataSource: DataSource;
|
|
51
|
+
searchType?: SearchType;
|
|
52
|
+
builtAt: number;
|
|
53
|
+
windowOffset: number;
|
|
54
|
+
/** Resume the `windowOffset` window at this shard offset (0 = window start). */
|
|
55
|
+
pageOffset?: number;
|
|
56
|
+
/** Output rows per page (default `ROLLUP_PAGE_ROWS_DAILY`). Injectable for tests. */
|
|
57
|
+
pageRows?: number;
|
|
58
|
+
/** Cap each input window's day span (default `DAILY_MAX_WINDOW_DAYS`). */
|
|
59
|
+
maxWindowDays?: number;
|
|
60
|
+
/** Split each date window by raw-query hash before grouping (1 = no sharding). */
|
|
61
|
+
shardCount?: number;
|
|
62
|
+
deadlineMs: number;
|
|
63
|
+
}): Promise<{
|
|
64
|
+
done: boolean;
|
|
65
|
+
nextWindowOffset: number;
|
|
66
|
+
nextPageOffset: number;
|
|
67
|
+
windowsTotal: number;
|
|
68
|
+
windowsBuilt: number;
|
|
69
|
+
rowsWritten: number;
|
|
70
|
+
}>;
|
|
71
|
+
export { queryCanonicalDailyRollup, queryCanonicalVariantsRollup, rebuildCanonicalDailyResumable };
|
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
import { encodeRowsToParquetFlex } from "../adapters/hyparquet.mjs";
|
|
2
|
+
import { createQueryDimStore } from "../entities/query-dim.mjs";
|
|
3
|
+
import "../entities.mjs";
|
|
4
|
+
import { rollupKey, rollupParquetKey } from "./core.mjs";
|
|
5
|
+
import { ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, planRollupWindows, runWindowed } from "./windows.mjs";
|
|
6
|
+
import { encodeJsonBigintSafe } from "@gscdump/lakehouse/bigint";
|
|
7
|
+
const CANONICAL_VARIANT_LIMIT = 10;
|
|
8
|
+
function retainCanonicalVariant(bucket, query, clicks, impressions, sumPos) {
|
|
9
|
+
bucket.count++;
|
|
10
|
+
let insertAt = 0;
|
|
11
|
+
while (insertAt < bucket.top.length) {
|
|
12
|
+
const existing = bucket.top[insertAt];
|
|
13
|
+
if (clicks > existing.clicks || clicks === existing.clicks && query.localeCompare(existing.query) < 0) break;
|
|
14
|
+
insertAt++;
|
|
15
|
+
}
|
|
16
|
+
if (insertAt >= CANONICAL_VARIANT_LIMIT) return;
|
|
17
|
+
bucket.top.splice(insertAt, 0, {
|
|
18
|
+
query,
|
|
19
|
+
clicks,
|
|
20
|
+
impressions,
|
|
21
|
+
sumPos
|
|
22
|
+
});
|
|
23
|
+
if (bucket.top.length > CANONICAL_VARIANT_LIMIT) bucket.top.pop();
|
|
24
|
+
}
|
|
25
|
+
const queryCanonicalVariantsRollup = {
|
|
26
|
+
id: "query_canonical_variants",
|
|
27
|
+
windowDays: null,
|
|
28
|
+
format: "parquet",
|
|
29
|
+
parquetColumns: [
|
|
30
|
+
{
|
|
31
|
+
name: "joinKey",
|
|
32
|
+
type: "VARCHAR",
|
|
33
|
+
nullable: false
|
|
34
|
+
},
|
|
35
|
+
{
|
|
36
|
+
name: "variantCount",
|
|
37
|
+
type: "BIGINT",
|
|
38
|
+
nullable: false
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
name: "canonicalName",
|
|
42
|
+
type: "VARCHAR",
|
|
43
|
+
nullable: true
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
name: "variants",
|
|
47
|
+
type: "VARCHAR",
|
|
48
|
+
nullable: true
|
|
49
|
+
}
|
|
50
|
+
],
|
|
51
|
+
parquetSortKey: ["joinKey"],
|
|
52
|
+
async build({ engine, ctx, dataSource, searchType }) {
|
|
53
|
+
const parts = await engine.listPartitions({
|
|
54
|
+
ctx,
|
|
55
|
+
table: "queries",
|
|
56
|
+
...searchType !== void 0 ? { searchType } : {}
|
|
57
|
+
});
|
|
58
|
+
if (parts.length === 0) return [];
|
|
59
|
+
const partitions = parts.map((p) => p.partition);
|
|
60
|
+
const byCanonical = /* @__PURE__ */ new Map();
|
|
61
|
+
const dimStore = createQueryDimStore({ dataSource });
|
|
62
|
+
const useDim = await dimStore.loadMeta(ctx) !== null;
|
|
63
|
+
const canonExpr = useDim ? "COALESCE(qd.query_canonical, q.query)" : "q.query";
|
|
64
|
+
let cursor = null;
|
|
65
|
+
for (;;) {
|
|
66
|
+
const after = cursor === null ? "" : `AND q.query > '${cursor.replace(/'/g, "''")}'`;
|
|
67
|
+
const fileSets = { FILES: {
|
|
68
|
+
table: "queries",
|
|
69
|
+
partitions
|
|
70
|
+
} };
|
|
71
|
+
if (useDim) fileSets.QUERY_DIM = {
|
|
72
|
+
table: "queries",
|
|
73
|
+
keys: [dimStore.parquetKey(ctx)]
|
|
74
|
+
};
|
|
75
|
+
const { rows } = await engine.runSQL({
|
|
76
|
+
ctx,
|
|
77
|
+
table: "queries",
|
|
78
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
79
|
+
fileSets,
|
|
80
|
+
sql: `
|
|
81
|
+
SELECT
|
|
82
|
+
${canonExpr} AS joinKey,
|
|
83
|
+
q.query AS query,
|
|
84
|
+
SUM(q.clicks) AS clicks,
|
|
85
|
+
SUM(q.impressions) AS impressions,
|
|
86
|
+
SUM(q.sum_position) AS sum_pos
|
|
87
|
+
FROM read_parquet({{FILES}}, union_by_name = true) q
|
|
88
|
+
${useDim ? "LEFT JOIN read_parquet({{QUERY_DIM}}, union_by_name = true) qd ON q.query = qd.query" : ""}
|
|
89
|
+
WHERE q.query IS NOT NULL ${after}
|
|
90
|
+
GROUP BY ${canonExpr}, q.query
|
|
91
|
+
ORDER BY q.query
|
|
92
|
+
LIMIT ${ROLLUP_PAGE_ROWS_WIDE}
|
|
93
|
+
`
|
|
94
|
+
});
|
|
95
|
+
for (const r of rows) {
|
|
96
|
+
const joinKey = String(r.joinKey);
|
|
97
|
+
let bucket = byCanonical.get(joinKey);
|
|
98
|
+
if (!bucket) {
|
|
99
|
+
bucket = {
|
|
100
|
+
count: 0,
|
|
101
|
+
top: []
|
|
102
|
+
};
|
|
103
|
+
byCanonical.set(joinKey, bucket);
|
|
104
|
+
}
|
|
105
|
+
retainCanonicalVariant(bucket, String(r.query), Number(r.clicks), Number(r.impressions), Number(r.sum_pos));
|
|
106
|
+
}
|
|
107
|
+
if (rows.length < 2e4) break;
|
|
108
|
+
cursor = String(rows[rows.length - 1].query);
|
|
109
|
+
}
|
|
110
|
+
const out = [];
|
|
111
|
+
for (const [joinKey, bucket] of byCanonical) {
|
|
112
|
+
const canonicalName = bucket.top[0]?.query ?? null;
|
|
113
|
+
const top = bucket.top.filter((v) => v.impressions > 0);
|
|
114
|
+
const variantsStr = top.length === 0 ? null : top.map((v) => `${v.query}:::${v.clicks}:::${v.impressions}:::${(v.sumPos / v.impressions + 1).toFixed(1)}`).join("||");
|
|
115
|
+
out.push({
|
|
116
|
+
joinKey,
|
|
117
|
+
variantCount: BigInt(bucket.count),
|
|
118
|
+
canonicalName,
|
|
119
|
+
variants: variantsStr
|
|
120
|
+
});
|
|
121
|
+
}
|
|
122
|
+
return out;
|
|
123
|
+
}
|
|
124
|
+
};
|
|
125
|
+
const CANONICAL_DAILY_ROLLUP_FINAL_ID = "query_canonical_daily";
|
|
126
|
+
const CANONICAL_DAILY_PART_STEM = "query_canonical_daily__part";
|
|
127
|
+
const queryCanonicalDailyRollup = {
|
|
128
|
+
id: "query_canonical_daily",
|
|
129
|
+
windowDays: null,
|
|
130
|
+
format: "parquet",
|
|
131
|
+
parquetColumns: [
|
|
132
|
+
{
|
|
133
|
+
name: "query_canonical",
|
|
134
|
+
type: "VARCHAR",
|
|
135
|
+
nullable: false
|
|
136
|
+
},
|
|
137
|
+
{
|
|
138
|
+
name: "date",
|
|
139
|
+
type: "DATE",
|
|
140
|
+
nullable: false
|
|
141
|
+
},
|
|
142
|
+
{
|
|
143
|
+
name: "clicks",
|
|
144
|
+
type: "BIGINT",
|
|
145
|
+
nullable: false
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
name: "impressions",
|
|
149
|
+
type: "BIGINT",
|
|
150
|
+
nullable: false
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
name: "sum_position",
|
|
154
|
+
type: "DOUBLE",
|
|
155
|
+
nullable: false
|
|
156
|
+
}
|
|
157
|
+
],
|
|
158
|
+
parquetSortKey: ["date", "query_canonical"],
|
|
159
|
+
async build({ engine, ctx, dataSource, searchType }) {
|
|
160
|
+
const dimStore = createQueryDimStore({ dataSource });
|
|
161
|
+
const useDim = await dimStore.loadMeta(ctx) !== null;
|
|
162
|
+
const canonExpr = useDim ? `COALESCE(qd.query_canonical, q.query)` : `query`;
|
|
163
|
+
return (await runWindowed({
|
|
164
|
+
engine,
|
|
165
|
+
ctx,
|
|
166
|
+
table: "queries",
|
|
167
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
168
|
+
...useDim ? { extraFileSets: { QUERY_DIM: {
|
|
169
|
+
table: "queries",
|
|
170
|
+
keys: [dimStore.parquetKey(ctx)]
|
|
171
|
+
} } } : {},
|
|
172
|
+
paginate: {
|
|
173
|
+
orderBy: "date, query_canonical",
|
|
174
|
+
pageRows: ROLLUP_PAGE_ROWS_DAILY
|
|
175
|
+
},
|
|
176
|
+
maxWindowDays: 7,
|
|
177
|
+
sqlFor: dailyWindowSqlFor(useDim, canonExpr)
|
|
178
|
+
})).map(mapDailyRow);
|
|
179
|
+
}
|
|
180
|
+
};
|
|
181
|
+
function canonicalDailyShardPredicate(queryExpr, shardIndex, shardCount) {
|
|
182
|
+
return shardCount <= 1 ? "" : ` AND (hash(${queryExpr}) % ${shardCount}) = ${shardIndex}`;
|
|
183
|
+
}
|
|
184
|
+
function dailyWindowSqlFor(useDim, canonExpr) {
|
|
185
|
+
return useDim ? (w, shard) => `
|
|
186
|
+
SELECT
|
|
187
|
+
${canonExpr} AS query_canonical,
|
|
188
|
+
CAST(q.date AS VARCHAR) AS date,
|
|
189
|
+
SUM(q.clicks)::BIGINT AS clicks,
|
|
190
|
+
SUM(q.impressions)::BIGINT AS impressions,
|
|
191
|
+
SUM(q.sum_position)::DOUBLE AS sum_position
|
|
192
|
+
FROM read_parquet({{FILES}}, union_by_name = true) q
|
|
193
|
+
LEFT JOIN read_parquet({{QUERY_DIM}}, union_by_name = true) qd ON q.query = qd.query
|
|
194
|
+
WHERE q.date >= '${w.start}' AND q.date <= '${w.end}'${canonicalDailyShardPredicate("q.query", shard?.index ?? 0, shard?.count ?? 1)}
|
|
195
|
+
GROUP BY ${canonExpr}, q.date
|
|
196
|
+
` : (w, shard) => `
|
|
197
|
+
SELECT
|
|
198
|
+
${canonExpr} AS query_canonical,
|
|
199
|
+
CAST(q.date AS VARCHAR) AS date,
|
|
200
|
+
SUM(q.clicks)::BIGINT AS clicks,
|
|
201
|
+
SUM(q.impressions)::BIGINT AS impressions,
|
|
202
|
+
SUM(q.sum_position)::DOUBLE AS sum_position
|
|
203
|
+
FROM read_parquet({{FILES}}, union_by_name = true) q
|
|
204
|
+
WHERE q.date >= '${w.start}' AND q.date <= '${w.end}'${canonicalDailyShardPredicate("q.query", shard?.index ?? 0, shard?.count ?? 1)}
|
|
205
|
+
GROUP BY ${canonExpr}, q.date
|
|
206
|
+
`;
|
|
207
|
+
}
|
|
208
|
+
function mapDailyRow(r) {
|
|
209
|
+
return {
|
|
210
|
+
query_canonical: String(r.query_canonical),
|
|
211
|
+
date: String(r.date),
|
|
212
|
+
clicks: BigInt(r.clicks),
|
|
213
|
+
impressions: BigInt(r.impressions),
|
|
214
|
+
sum_position: Number(r.sum_position)
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
async function rebuildCanonicalDailyResumable(opts) {
|
|
218
|
+
const { engine, ctx, dataSource, searchType, builtAt, windowOffset, deadlineMs } = opts;
|
|
219
|
+
const sType = searchType !== void 0 ? { searchType } : {};
|
|
220
|
+
const startPageOffset = opts.pageOffset ?? 0;
|
|
221
|
+
const pageRows = opts.pageRows ?? 7e4;
|
|
222
|
+
const maxWindowDays = opts.maxWindowDays ?? 7;
|
|
223
|
+
const shardCount = Math.max(1, Math.floor(opts.shardCount ?? 1));
|
|
224
|
+
const windows = planRollupWindows((await engine.listPartitions({
|
|
225
|
+
ctx,
|
|
226
|
+
table: "queries",
|
|
227
|
+
...sType
|
|
228
|
+
})).map((p) => ({
|
|
229
|
+
partition: p.partition,
|
|
230
|
+
bytes: p.bytes
|
|
231
|
+
})), void 0, maxWindowDays);
|
|
232
|
+
const windowsTotal = windows.length;
|
|
233
|
+
const dimStore = createQueryDimStore({ dataSource });
|
|
234
|
+
const useDim = await dimStore.loadMeta(ctx) !== null;
|
|
235
|
+
const canonExpr = useDim ? `COALESCE(qd.query_canonical, q.query)` : `query`;
|
|
236
|
+
const extraFileSets = useDim ? { QUERY_DIM: {
|
|
237
|
+
table: "queries",
|
|
238
|
+
keys: [dimStore.parquetKey(ctx)]
|
|
239
|
+
} } : void 0;
|
|
240
|
+
const sqlFor = dailyWindowSqlFor(useDim, canonExpr);
|
|
241
|
+
const cols = queryCanonicalDailyRollup.parquetColumns;
|
|
242
|
+
const sortKey = queryCanonicalDailyRollup.parquetSortKey;
|
|
243
|
+
let i = windowOffset;
|
|
244
|
+
let page = startPageOffset;
|
|
245
|
+
let nextWindowOffset = windowsTotal;
|
|
246
|
+
let nextPageOffset = 0;
|
|
247
|
+
let rowsWritten = 0;
|
|
248
|
+
let pausedMidWindow = false;
|
|
249
|
+
const flushPage = async (windowIdx, pageOffset, rows) => {
|
|
250
|
+
if (rows.length === 0) return;
|
|
251
|
+
const key = rollupParquetKey(ctx, `${CANONICAL_DAILY_PART_STEM}__w${windowIdx}_p${pageOffset}`, builtAt, searchType);
|
|
252
|
+
await dataSource.write(key, encodeRowsToParquetFlex(rows, {
|
|
253
|
+
columns: cols,
|
|
254
|
+
sortKey
|
|
255
|
+
}));
|
|
256
|
+
rowsWritten += rows.length;
|
|
257
|
+
};
|
|
258
|
+
for (; i < windowsTotal; i++) {
|
|
259
|
+
const w = windows[i];
|
|
260
|
+
for (; page < shardCount;) {
|
|
261
|
+
const coreSql = sqlFor(w, {
|
|
262
|
+
index: page,
|
|
263
|
+
count: shardCount
|
|
264
|
+
});
|
|
265
|
+
const result = await engine.runSQL({
|
|
266
|
+
ctx,
|
|
267
|
+
table: "queries",
|
|
268
|
+
...sType,
|
|
269
|
+
fileSets: {
|
|
270
|
+
FILES: {
|
|
271
|
+
table: "queries",
|
|
272
|
+
partitions: w.partitions
|
|
273
|
+
},
|
|
274
|
+
...extraFileSets
|
|
275
|
+
},
|
|
276
|
+
sql: `${coreSql}\nORDER BY date, query_canonical\nLIMIT ${pageRows}`
|
|
277
|
+
});
|
|
278
|
+
if (result.rows.length >= pageRows) throw new Error(`query_canonical_daily shard overflow: window=${i} shard=${page}/${shardCount} returned >= ${pageRows} rows; increase shardCount or pageRows`);
|
|
279
|
+
await flushPage(i, page, result.rows.map(mapDailyRow));
|
|
280
|
+
page += 1;
|
|
281
|
+
if (Date.now() > deadlineMs) {
|
|
282
|
+
nextWindowOffset = i;
|
|
283
|
+
nextPageOffset = page;
|
|
284
|
+
pausedMidWindow = true;
|
|
285
|
+
break;
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
if (pausedMidWindow) break;
|
|
289
|
+
page = 0;
|
|
290
|
+
if (Date.now() > deadlineMs) {
|
|
291
|
+
nextWindowOffset = i + 1;
|
|
292
|
+
nextPageOffset = 0;
|
|
293
|
+
break;
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
if (nextWindowOffset < windowsTotal || nextPageOffset > 0) return {
|
|
297
|
+
done: false,
|
|
298
|
+
nextWindowOffset,
|
|
299
|
+
nextPageOffset,
|
|
300
|
+
windowsTotal,
|
|
301
|
+
windowsBuilt: nextWindowOffset - windowOffset,
|
|
302
|
+
rowsWritten
|
|
303
|
+
};
|
|
304
|
+
const partPrefixDir = rollupParquetKey(ctx, CANONICAL_DAILY_PART_STEM, builtAt, searchType).replace(/[^/]*$/, "");
|
|
305
|
+
const partKeys = (await dataSource.list(partPrefixDir)).filter((k) => k.includes(`${CANONICAL_DAILY_PART_STEM}__w`) && k.endsWith(`__v${builtAt}.parquet`)).sort();
|
|
306
|
+
if (partKeys.length === 0) {
|
|
307
|
+
const emptyKey = rollupParquetKey(ctx, `${CANONICAL_DAILY_PART_STEM}__w0_p0`, builtAt, searchType);
|
|
308
|
+
await dataSource.write(emptyKey, encodeRowsToParquetFlex([], {
|
|
309
|
+
columns: cols,
|
|
310
|
+
sortKey
|
|
311
|
+
}));
|
|
312
|
+
partKeys.push(emptyKey);
|
|
313
|
+
}
|
|
314
|
+
const envelope = {
|
|
315
|
+
version: 1,
|
|
316
|
+
id: CANONICAL_DAILY_ROLLUP_FINAL_ID,
|
|
317
|
+
builtAt,
|
|
318
|
+
windowDays: queryCanonicalDailyRollup.windowDays,
|
|
319
|
+
payload: {
|
|
320
|
+
parquetKey: partKeys[0],
|
|
321
|
+
parquetKeys: partKeys,
|
|
322
|
+
rowCount: 0
|
|
323
|
+
}
|
|
324
|
+
};
|
|
325
|
+
await dataSource.write(rollupKey(ctx, CANONICAL_DAILY_ROLLUP_FINAL_ID, builtAt, searchType), encodeJsonBigintSafe(envelope));
|
|
326
|
+
return {
|
|
327
|
+
done: true,
|
|
328
|
+
nextWindowOffset,
|
|
329
|
+
nextPageOffset,
|
|
330
|
+
windowsTotal,
|
|
331
|
+
windowsBuilt: nextWindowOffset - windowOffset,
|
|
332
|
+
rowsWritten
|
|
333
|
+
};
|
|
334
|
+
}
|
|
335
|
+
export { queryCanonicalDailyRollup, queryCanonicalVariantsRollup, rebuildCanonicalDailyResumable };
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
import { DataSource, FileSetRef, Row as Row$1, TableName as TableName$1 } from "../storage.mjs";
|
|
2
|
+
import { ColumnDef } from "../schema.mjs";
|
|
3
|
+
import { EngineError } from "../errors.mjs";
|
|
4
|
+
import "../contracts.mjs";
|
|
5
|
+
import { TenantCtx } from "@gscdump/contracts";
|
|
6
|
+
import { SearchType } from "gscdump/query";
|
|
7
|
+
interface RollupCtx extends TenantCtx {
|
|
8
|
+
/** When the rollup was built. Stamped into payload + filename. */
|
|
9
|
+
builtAt: number;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Tenant-scoped engine surface a rollup builder needs. Subset of
|
|
13
|
+
* `StorageEngine.runSQL` so rollups stay testable without a full engine.
|
|
14
|
+
*/
|
|
15
|
+
interface RollupEngine {
|
|
16
|
+
runSQL: (opts: {
|
|
17
|
+
ctx: TenantCtx;
|
|
18
|
+
fileSets: Record<string, FileSetRef>;
|
|
19
|
+
table?: TableName$1;
|
|
20
|
+
sql: string;
|
|
21
|
+
params?: unknown[];
|
|
22
|
+
/**
|
|
23
|
+
* Restrict every manifest lookup to a single GSC search-type slice. The
|
|
24
|
+
* rollup runner forwards `RebuildRollupsOptions.searchType` so the
|
|
25
|
+
* aggregated facts never mix web + non-web rows. Undefined preserves
|
|
26
|
+
* the legacy cross-type union (web-only tenants).
|
|
27
|
+
*/
|
|
28
|
+
searchType?: SearchType;
|
|
29
|
+
}) => Promise<{
|
|
30
|
+
rows: Row$1[];
|
|
31
|
+
}>;
|
|
32
|
+
/**
|
|
33
|
+
* Read the live manifest for a (tenant, table[, searchType]) cohort —
|
|
34
|
+
* cheap, no parquet decode. Builders use this to chunk a full-history scan
|
|
35
|
+
* into byte-bounded windows (see `WINDOW_BYTE_BUDGET`) so a single `runSQL`
|
|
36
|
+
* call never ships an oversized Arrow IPC payload across the Workers
|
|
37
|
+
* service-binding RPC (32MiB hard cap).
|
|
38
|
+
*/
|
|
39
|
+
listPartitions: (opts: {
|
|
40
|
+
ctx: TenantCtx;
|
|
41
|
+
table: TableName$1;
|
|
42
|
+
searchType?: SearchType;
|
|
43
|
+
}) => Promise<Array<{
|
|
44
|
+
partition: string;
|
|
45
|
+
bytes: number;
|
|
46
|
+
}>>;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* One rollup definition. Build runs SQL over the tenant's facts and/or reads
|
|
50
|
+
* from entity stores via `dataSource`, returning a JSON-serializable payload
|
|
51
|
+
* that the runner timestamps + writes.
|
|
52
|
+
*/
|
|
53
|
+
interface RollupDef {
|
|
54
|
+
id: string;
|
|
55
|
+
/**
|
|
56
|
+
* Window in days the rollup covers. `null` means full history. Used by
|
|
57
|
+
* the runner to populate `windowDays` in the payload metadata so readers
|
|
58
|
+
* can validate freshness.
|
|
59
|
+
*/
|
|
60
|
+
windowDays: number | null;
|
|
61
|
+
/**
|
|
62
|
+
* Storage format. `'json'` (default) wraps the build payload in a
|
|
63
|
+
* `RollupEnvelope` and writes as a JSON blob. `'parquet'` expects `build`
|
|
64
|
+
* to return rows matching `parquetColumns` and writes a parquet file plus
|
|
65
|
+
* a tiny JSON sidecar envelope that points at it, so metadata
|
|
66
|
+
* (`builtAt` / `windowDays`) stays readable without decoding parquet.
|
|
67
|
+
*/
|
|
68
|
+
format?: 'json' | 'parquet';
|
|
69
|
+
/**
|
|
70
|
+
* Column schema for parquet output. Required when `format === 'parquet'`.
|
|
71
|
+
* Types map the same way as the fact-table encoder: VARCHAR / DATE go
|
|
72
|
+
* through BYTE_ARRAY/UTF8; BIGINT → INT64; INTEGER → INT32; DOUBLE → DOUBLE.
|
|
73
|
+
*/
|
|
74
|
+
parquetColumns?: readonly ColumnDef[];
|
|
75
|
+
/** Sort-key column names for parquet row-group stats. Optional. */
|
|
76
|
+
parquetSortKey?: readonly string[];
|
|
77
|
+
/**
|
|
78
|
+
* When true, this rollup's payload is independent of GSC slice (e.g. entity
|
|
79
|
+
* rollups sourced from sitemap / indexing snapshots, not slice-partitioned
|
|
80
|
+
* fact tables). The runner rejects calls that pass `searchType` alongside
|
|
81
|
+
* a slice-orthogonal def so the output never lands under a per-slice prefix
|
|
82
|
+
* that the read path won't look at.
|
|
83
|
+
*/
|
|
84
|
+
sliceOrthogonal?: boolean;
|
|
85
|
+
build: (deps: {
|
|
86
|
+
engine: RollupEngine;
|
|
87
|
+
ctx: TenantCtx;
|
|
88
|
+
/**
|
|
89
|
+
* Tenant-scoped object store. Rollups that aggregate over entity
|
|
90
|
+
* snapshots (e.g. indexing metadata) read JSON docs through this.
|
|
91
|
+
* Pure-SQL rollups can ignore it.
|
|
92
|
+
*/
|
|
93
|
+
dataSource: DataSource;
|
|
94
|
+
/**
|
|
95
|
+
* UTC millis the trailing window anchors to — its inclusive END. Equals
|
|
96
|
+
* the newest synced/finalized data date when the runner is given
|
|
97
|
+
* `dataEndDate`, otherwise wall-clock build time. Builders derive window
|
|
98
|
+
* cutoffs from this (e.g. the trailing-28d boundary) and inline a date
|
|
99
|
+
* literal so the SQL stays portable across DuckDB builds without the ICU
|
|
100
|
+
* extension (Workers DuckDB — `CURRENT_DATE` lives in ICU).
|
|
101
|
+
*/
|
|
102
|
+
windowAnchorMs: number;
|
|
103
|
+
/**
|
|
104
|
+
* GSC search-type slice the runner was invoked for. Builders forward
|
|
105
|
+
* this to every `engine.runSQL` call so the aggregated facts come
|
|
106
|
+
* from one cohort. Undefined preserves the legacy cross-type union
|
|
107
|
+
* (used by web-only tenants and admin paths).
|
|
108
|
+
*/
|
|
109
|
+
searchType?: SearchType;
|
|
110
|
+
}) => Promise<unknown>;
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* Wire shape persisted to R2/disk. Readers can rely on the `version` + `builtAt`.
|
|
114
|
+
* Parquet rollups write this envelope as a sidecar whose `payload` points at
|
|
115
|
+
* the co-located `.parquet` object via `{ parquetKey, rowCount }`.
|
|
116
|
+
*/
|
|
117
|
+
interface RollupEnvelope<T = unknown> {
|
|
118
|
+
version: 1;
|
|
119
|
+
id: string;
|
|
120
|
+
builtAt: number;
|
|
121
|
+
windowDays: number | null;
|
|
122
|
+
payload: T;
|
|
123
|
+
}
|
|
124
|
+
interface ParquetRollupPointer {
|
|
125
|
+
parquetKey: string;
|
|
126
|
+
rowCount: number;
|
|
127
|
+
/**
|
|
128
|
+
* MULTI-FILE rollup: when set, the rollup is the UNION of these parquet keys
|
|
129
|
+
* (disjoint by the grain's partition column, e.g. `date` for the resumable
|
|
130
|
+
* `query_canonical_daily` build). Readers MUST union all keys; `parquetKey`
|
|
131
|
+
* stays populated (the first part) for single-file readers. Avoids a JS
|
|
132
|
+
* merge/re-encode of the whole rollup — the scaling bottleneck for a
|
|
133
|
+
* cross-invocation resumable build.
|
|
134
|
+
*/
|
|
135
|
+
parquetKeys?: string[];
|
|
136
|
+
}
|
|
137
|
+
declare function rollupKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
|
|
138
|
+
declare function rollupParquetKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
|
|
139
|
+
interface RollupBucket {
|
|
140
|
+
list: (opts: {
|
|
141
|
+
prefix: string;
|
|
142
|
+
cursor?: string;
|
|
143
|
+
}) => Promise<{
|
|
144
|
+
objects: Array<{
|
|
145
|
+
key: string;
|
|
146
|
+
}>;
|
|
147
|
+
truncated?: boolean;
|
|
148
|
+
cursor?: string;
|
|
149
|
+
}>;
|
|
150
|
+
get: (key: string) => Promise<{
|
|
151
|
+
text: () => Promise<string>;
|
|
152
|
+
} | null>;
|
|
153
|
+
}
|
|
154
|
+
declare function readLatestRollup<T = unknown>(bucket: RollupBucket, ctx: TenantCtx, id: string, searchType?: SearchType): Promise<RollupEnvelope<T> | null>;
|
|
155
|
+
interface RebuildRollupsOptions {
|
|
156
|
+
engine: RollupEngine;
|
|
157
|
+
dataSource: DataSource;
|
|
158
|
+
ctx: TenantCtx;
|
|
159
|
+
defs: readonly RollupDef[];
|
|
160
|
+
now?: () => number;
|
|
161
|
+
/**
|
|
162
|
+
* Build rollups for a single GSC search-type slice. Threads into every
|
|
163
|
+
* builder's `engine.runSQL` call so the aggregated facts come from one
|
|
164
|
+
* cohort, and namespaces the output object keys under a `<searchType>/`
|
|
165
|
+
* segment so per-slice rollups coexist without overwriting each other.
|
|
166
|
+
* Undefined preserves the legacy cross-type behaviour (one rollup over
|
|
167
|
+
* the union of all slices, written to the legacy path) — fine for web-
|
|
168
|
+
* only tenants and explicit cross-type admin views.
|
|
169
|
+
*/
|
|
170
|
+
searchType?: SearchType;
|
|
171
|
+
/**
|
|
172
|
+
* ISO date (`YYYY-MM-DD`) of the newest synced/finalized day. Trailing-
|
|
173
|
+
* window rollups (28d/90d) anchor their window END here instead of
|
|
174
|
+
* wall-clock build time, so a "last 28 days" rollup covers the 28 days of
|
|
175
|
+
* data that actually exist — not 28 days back from whenever the job ran,
|
|
176
|
+
* which would include GSC's 2-3 day empty tail. Omit for the legacy
|
|
177
|
+
* wall-clock behaviour.
|
|
178
|
+
*/
|
|
179
|
+
dataEndDate?: string;
|
|
180
|
+
}
|
|
181
|
+
interface RebuildRollupResult {
|
|
182
|
+
id: string;
|
|
183
|
+
/** JSON envelope key. For parquet rollups this is the sidecar pointer. */
|
|
184
|
+
objectKey: string;
|
|
185
|
+
/** Parquet payload key. Present only when `format === 'parquet'`. */
|
|
186
|
+
parquetKey?: string;
|
|
187
|
+
/** Envelope byte size; for parquet rollups does NOT include parquet bytes. */
|
|
188
|
+
bytes: number;
|
|
189
|
+
/** Parquet payload byte size when `format === 'parquet'`. */
|
|
190
|
+
parquetBytes?: number;
|
|
191
|
+
builtAt: number;
|
|
192
|
+
/**
|
|
193
|
+
* Set when this def's build/encode/write failed. The runner records the
|
|
194
|
+
* failure and continues with the remaining defs so one bad rollup never
|
|
195
|
+
* aborts the rest. Successful defs have no `error`. The human-readable
|
|
196
|
+
* message (including the stack when available) lives on `error.message`.
|
|
197
|
+
*/
|
|
198
|
+
error?: EngineError;
|
|
199
|
+
}
|
|
200
|
+
declare function rebuildRollups(opts: RebuildRollupsOptions): Promise<RebuildRollupResult[]>;
|
|
201
|
+
export { ParquetRollupPointer, RebuildRollupResult, RebuildRollupsOptions, RollupBucket, RollupCtx, RollupDef, RollupEngine, RollupEnvelope, readLatestRollup, rebuildRollups, rollupKey, rollupParquetKey };
|