@gscdump/engine 1.4.11 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/entities/empty-types.d.mts +22 -0
  2. package/dist/entities/empty-types.mjs +58 -0
  3. package/dist/entities/indexing-metadata.d.mts +26 -0
  4. package/dist/entities/indexing-metadata.mjs +31 -0
  5. package/dist/entities/inspection.d.mts +240 -0
  6. package/dist/entities/inspection.mjs +443 -0
  7. package/dist/entities/io.mjs +16 -0
  8. package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
  9. package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
  10. package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
  11. package/dist/entities/sitemap-shared.d.mts +199 -0
  12. package/dist/entities/sitemap-shared.mjs +243 -0
  13. package/dist/entities/sitemap-write.d.mts +3 -0
  14. package/dist/entities/sitemap-write.mjs +528 -0
  15. package/dist/entities/sitemap.d.mts +3 -0
  16. package/dist/entities/sitemap.mjs +91 -0
  17. package/dist/entities.d.mts +9 -481
  18. package/dist/entities.mjs +7 -1380
  19. package/dist/rollups/canonical.d.mts +71 -0
  20. package/dist/rollups/canonical.mjs +335 -0
  21. package/dist/rollups/core.d.mts +201 -0
  22. package/dist/rollups/core.mjs +116 -0
  23. package/dist/rollups/dates.mjs +11 -0
  24. package/dist/rollups/defaults.d.mts +11 -0
  25. package/dist/rollups/defaults.mjs +17 -0
  26. package/dist/rollups/hourly.d.mts +38 -0
  27. package/dist/rollups/hourly.mjs +38 -0
  28. package/dist/rollups/indexing.d.mts +46 -0
  29. package/dist/rollups/indexing.mjs +357 -0
  30. package/dist/rollups/traffic.d.mts +38 -0
  31. package/dist/rollups/traffic.mjs +289 -0
  32. package/dist/rollups/windows.d.mts +90 -0
  33. package/dist/rollups/windows.mjs +176 -0
  34. package/dist/rollups.d.mts +8 -471
  35. package/dist/rollups.mjs +7 -1316
  36. package/package.json +4 -4
  37. /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
@@ -0,0 +1,71 @@
1
+ import { DataSource } from "../storage.mjs";
2
+ import "../contracts.mjs";
3
+ import { RollupDef, RollupEngine } from "./core.mjs";
4
+ import { TenantCtx } from "@gscdump/contracts";
5
+ import { SearchType } from "gscdump/query";
6
+ declare const queryCanonicalVariantsRollup: RollupDef;
7
+ /**
8
+ * Canonical-grained fact aggregate (ADR-0018 Gap 2): pre-sums the raw
9
+ * `(query × date)` query rows to `(query_canonical × date)`, so canonical-
10
+ * primary top/gaining/losing reads a small pre-aggregated table instead of
11
+ * re-collapsing variants on every request. Metrics are additive, so summing
12
+ * these per-date sums over a window is exact — identical to aggregating the raw
13
+ * rows.
14
+ *
15
+ * Null-free by construction: groups by the versioned query dimension when it
16
+ * exists, with raw query as the fallback, so the rollup never carries a NULL/''
17
+ * canonical bucket and the read path can treat the rollup's `query_canonical`
18
+ * column as already-derived.
19
+ *
20
+ * Date-grained full history (`windowDays: null`): one rollup serves every date
21
+ * range (reads filter by `date`) and both windows of a comparison. Opt-in (not
22
+ * in `DEFAULT_ROLLUPS`); the host points the main query's file set at it for
23
+ * queries the rollup covers (see `canonicalRollupCovers` /
24
+ * `RunOptimizedQueryOptions.canonicalSource`).
25
+ */
26
+ declare const queryCanonicalDailyRollup: RollupDef;
27
+ /**
28
+ * Resumable, cross-invocation build of `query_canonical_daily` for a high-
29
+ * cardinality site whose full windowed build exceeds one job reservation (300s).
30
+ *
31
+ * Each call builds from `(windowOffset, pageOffset)` until `deadlineMs`, writes
32
+ * that batch's rows to a PART parquet, and returns `{ done:false, nextWindowOffset,
33
+ * nextPageOffset }` for the caller to re-enqueue. When the last window is fully
34
+ * paged it publishes a multi-file envelope listing every part (parts are disjoint
35
+ * by `(query_canonical, date)`, so the read path just unions them — no merge),
36
+ * returning `{ done:true }`. `builtAt` MUST be stable across the continuation chain
37
+ * (it versions both the part keys and the final rollup key).
38
+ *
39
+ * INTRA-WINDOW resumability: the deadline is checked between raw-query hash shards,
40
+ * not just between date windows. A single high-cardinality day can spend a full
41
+ * reservation inside one grouped/sorted aggregate before the deadline check gets
42
+ * control back. `pageOffset` is the next shard index for the current window, so a
43
+ * continuation resumes the SAME day at the next shard. Parts are keyed by
44
+ * `(windowOffset, pageOffset)`; multiple parts may contain the same canonical/date
45
+ * from different raw-query shards, and rollup reads sum over the union.
46
+ */
47
+ declare function rebuildCanonicalDailyResumable(opts: {
48
+ engine: RollupEngine;
49
+ ctx: TenantCtx;
50
+ dataSource: DataSource;
51
+ searchType?: SearchType;
52
+ builtAt: number;
53
+ windowOffset: number;
54
+ /** Resume the `windowOffset` window at this shard offset (0 = window start). */
55
+ pageOffset?: number;
56
+ /** Output rows per page (default `ROLLUP_PAGE_ROWS_DAILY`). Injectable for tests. */
57
+ pageRows?: number;
58
+ /** Cap each input window's day span (default `DAILY_MAX_WINDOW_DAYS`). */
59
+ maxWindowDays?: number;
60
+ /** Split each date window by raw-query hash before grouping (1 = no sharding). */
61
+ shardCount?: number;
62
+ deadlineMs: number;
63
+ }): Promise<{
64
+ done: boolean;
65
+ nextWindowOffset: number;
66
+ nextPageOffset: number;
67
+ windowsTotal: number;
68
+ windowsBuilt: number;
69
+ rowsWritten: number;
70
+ }>;
71
+ export { queryCanonicalDailyRollup, queryCanonicalVariantsRollup, rebuildCanonicalDailyResumable };
@@ -0,0 +1,335 @@
1
+ import { encodeRowsToParquetFlex } from "../adapters/hyparquet.mjs";
2
+ import { createQueryDimStore } from "../entities/query-dim.mjs";
3
+ import "../entities.mjs";
4
+ import { rollupKey, rollupParquetKey } from "./core.mjs";
5
+ import { ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, planRollupWindows, runWindowed } from "./windows.mjs";
6
+ import { encodeJsonBigintSafe } from "@gscdump/lakehouse/bigint";
7
+ const CANONICAL_VARIANT_LIMIT = 10;
8
+ function retainCanonicalVariant(bucket, query, clicks, impressions, sumPos) {
9
+ bucket.count++;
10
+ let insertAt = 0;
11
+ while (insertAt < bucket.top.length) {
12
+ const existing = bucket.top[insertAt];
13
+ if (clicks > existing.clicks || clicks === existing.clicks && query.localeCompare(existing.query) < 0) break;
14
+ insertAt++;
15
+ }
16
+ if (insertAt >= CANONICAL_VARIANT_LIMIT) return;
17
+ bucket.top.splice(insertAt, 0, {
18
+ query,
19
+ clicks,
20
+ impressions,
21
+ sumPos
22
+ });
23
+ if (bucket.top.length > CANONICAL_VARIANT_LIMIT) bucket.top.pop();
24
+ }
25
+ const queryCanonicalVariantsRollup = {
26
+ id: "query_canonical_variants",
27
+ windowDays: null,
28
+ format: "parquet",
29
+ parquetColumns: [
30
+ {
31
+ name: "joinKey",
32
+ type: "VARCHAR",
33
+ nullable: false
34
+ },
35
+ {
36
+ name: "variantCount",
37
+ type: "BIGINT",
38
+ nullable: false
39
+ },
40
+ {
41
+ name: "canonicalName",
42
+ type: "VARCHAR",
43
+ nullable: true
44
+ },
45
+ {
46
+ name: "variants",
47
+ type: "VARCHAR",
48
+ nullable: true
49
+ }
50
+ ],
51
+ parquetSortKey: ["joinKey"],
52
+ async build({ engine, ctx, dataSource, searchType }) {
53
+ const parts = await engine.listPartitions({
54
+ ctx,
55
+ table: "queries",
56
+ ...searchType !== void 0 ? { searchType } : {}
57
+ });
58
+ if (parts.length === 0) return [];
59
+ const partitions = parts.map((p) => p.partition);
60
+ const byCanonical = /* @__PURE__ */ new Map();
61
+ const dimStore = createQueryDimStore({ dataSource });
62
+ const useDim = await dimStore.loadMeta(ctx) !== null;
63
+ const canonExpr = useDim ? "COALESCE(qd.query_canonical, q.query)" : "q.query";
64
+ let cursor = null;
65
+ for (;;) {
66
+ const after = cursor === null ? "" : `AND q.query > '${cursor.replace(/'/g, "''")}'`;
67
+ const fileSets = { FILES: {
68
+ table: "queries",
69
+ partitions
70
+ } };
71
+ if (useDim) fileSets.QUERY_DIM = {
72
+ table: "queries",
73
+ keys: [dimStore.parquetKey(ctx)]
74
+ };
75
+ const { rows } = await engine.runSQL({
76
+ ctx,
77
+ table: "queries",
78
+ ...searchType !== void 0 ? { searchType } : {},
79
+ fileSets,
80
+ sql: `
81
+ SELECT
82
+ ${canonExpr} AS joinKey,
83
+ q.query AS query,
84
+ SUM(q.clicks) AS clicks,
85
+ SUM(q.impressions) AS impressions,
86
+ SUM(q.sum_position) AS sum_pos
87
+ FROM read_parquet({{FILES}}, union_by_name = true) q
88
+ ${useDim ? "LEFT JOIN read_parquet({{QUERY_DIM}}, union_by_name = true) qd ON q.query = qd.query" : ""}
89
+ WHERE q.query IS NOT NULL ${after}
90
+ GROUP BY ${canonExpr}, q.query
91
+ ORDER BY q.query
92
+ LIMIT ${ROLLUP_PAGE_ROWS_WIDE}
93
+ `
94
+ });
95
+ for (const r of rows) {
96
+ const joinKey = String(r.joinKey);
97
+ let bucket = byCanonical.get(joinKey);
98
+ if (!bucket) {
99
+ bucket = {
100
+ count: 0,
101
+ top: []
102
+ };
103
+ byCanonical.set(joinKey, bucket);
104
+ }
105
+ retainCanonicalVariant(bucket, String(r.query), Number(r.clicks), Number(r.impressions), Number(r.sum_pos));
106
+ }
107
+ if (rows.length < 2e4) break;
108
+ cursor = String(rows[rows.length - 1].query);
109
+ }
110
+ const out = [];
111
+ for (const [joinKey, bucket] of byCanonical) {
112
+ const canonicalName = bucket.top[0]?.query ?? null;
113
+ const top = bucket.top.filter((v) => v.impressions > 0);
114
+ const variantsStr = top.length === 0 ? null : top.map((v) => `${v.query}:::${v.clicks}:::${v.impressions}:::${(v.sumPos / v.impressions + 1).toFixed(1)}`).join("||");
115
+ out.push({
116
+ joinKey,
117
+ variantCount: BigInt(bucket.count),
118
+ canonicalName,
119
+ variants: variantsStr
120
+ });
121
+ }
122
+ return out;
123
+ }
124
+ };
125
+ const CANONICAL_DAILY_ROLLUP_FINAL_ID = "query_canonical_daily";
126
+ const CANONICAL_DAILY_PART_STEM = "query_canonical_daily__part";
127
+ const queryCanonicalDailyRollup = {
128
+ id: "query_canonical_daily",
129
+ windowDays: null,
130
+ format: "parquet",
131
+ parquetColumns: [
132
+ {
133
+ name: "query_canonical",
134
+ type: "VARCHAR",
135
+ nullable: false
136
+ },
137
+ {
138
+ name: "date",
139
+ type: "DATE",
140
+ nullable: false
141
+ },
142
+ {
143
+ name: "clicks",
144
+ type: "BIGINT",
145
+ nullable: false
146
+ },
147
+ {
148
+ name: "impressions",
149
+ type: "BIGINT",
150
+ nullable: false
151
+ },
152
+ {
153
+ name: "sum_position",
154
+ type: "DOUBLE",
155
+ nullable: false
156
+ }
157
+ ],
158
+ parquetSortKey: ["date", "query_canonical"],
159
+ async build({ engine, ctx, dataSource, searchType }) {
160
+ const dimStore = createQueryDimStore({ dataSource });
161
+ const useDim = await dimStore.loadMeta(ctx) !== null;
162
+ const canonExpr = useDim ? `COALESCE(qd.query_canonical, q.query)` : `query`;
163
+ return (await runWindowed({
164
+ engine,
165
+ ctx,
166
+ table: "queries",
167
+ ...searchType !== void 0 ? { searchType } : {},
168
+ ...useDim ? { extraFileSets: { QUERY_DIM: {
169
+ table: "queries",
170
+ keys: [dimStore.parquetKey(ctx)]
171
+ } } } : {},
172
+ paginate: {
173
+ orderBy: "date, query_canonical",
174
+ pageRows: ROLLUP_PAGE_ROWS_DAILY
175
+ },
176
+ maxWindowDays: 7,
177
+ sqlFor: dailyWindowSqlFor(useDim, canonExpr)
178
+ })).map(mapDailyRow);
179
+ }
180
+ };
181
+ function canonicalDailyShardPredicate(queryExpr, shardIndex, shardCount) {
182
+ return shardCount <= 1 ? "" : ` AND (hash(${queryExpr}) % ${shardCount}) = ${shardIndex}`;
183
+ }
184
+ function dailyWindowSqlFor(useDim, canonExpr) {
185
+ return useDim ? (w, shard) => `
186
+ SELECT
187
+ ${canonExpr} AS query_canonical,
188
+ CAST(q.date AS VARCHAR) AS date,
189
+ SUM(q.clicks)::BIGINT AS clicks,
190
+ SUM(q.impressions)::BIGINT AS impressions,
191
+ SUM(q.sum_position)::DOUBLE AS sum_position
192
+ FROM read_parquet({{FILES}}, union_by_name = true) q
193
+ LEFT JOIN read_parquet({{QUERY_DIM}}, union_by_name = true) qd ON q.query = qd.query
194
+ WHERE q.date >= '${w.start}' AND q.date <= '${w.end}'${canonicalDailyShardPredicate("q.query", shard?.index ?? 0, shard?.count ?? 1)}
195
+ GROUP BY ${canonExpr}, q.date
196
+ ` : (w, shard) => `
197
+ SELECT
198
+ ${canonExpr} AS query_canonical,
199
+ CAST(q.date AS VARCHAR) AS date,
200
+ SUM(q.clicks)::BIGINT AS clicks,
201
+ SUM(q.impressions)::BIGINT AS impressions,
202
+ SUM(q.sum_position)::DOUBLE AS sum_position
203
+ FROM read_parquet({{FILES}}, union_by_name = true) q
204
+ WHERE q.date >= '${w.start}' AND q.date <= '${w.end}'${canonicalDailyShardPredicate("q.query", shard?.index ?? 0, shard?.count ?? 1)}
205
+ GROUP BY ${canonExpr}, q.date
206
+ `;
207
+ }
208
+ function mapDailyRow(r) {
209
+ return {
210
+ query_canonical: String(r.query_canonical),
211
+ date: String(r.date),
212
+ clicks: BigInt(r.clicks),
213
+ impressions: BigInt(r.impressions),
214
+ sum_position: Number(r.sum_position)
215
+ };
216
+ }
217
+ async function rebuildCanonicalDailyResumable(opts) {
218
+ const { engine, ctx, dataSource, searchType, builtAt, windowOffset, deadlineMs } = opts;
219
+ const sType = searchType !== void 0 ? { searchType } : {};
220
+ const startPageOffset = opts.pageOffset ?? 0;
221
+ const pageRows = opts.pageRows ?? 7e4;
222
+ const maxWindowDays = opts.maxWindowDays ?? 7;
223
+ const shardCount = Math.max(1, Math.floor(opts.shardCount ?? 1));
224
+ const windows = planRollupWindows((await engine.listPartitions({
225
+ ctx,
226
+ table: "queries",
227
+ ...sType
228
+ })).map((p) => ({
229
+ partition: p.partition,
230
+ bytes: p.bytes
231
+ })), void 0, maxWindowDays);
232
+ const windowsTotal = windows.length;
233
+ const dimStore = createQueryDimStore({ dataSource });
234
+ const useDim = await dimStore.loadMeta(ctx) !== null;
235
+ const canonExpr = useDim ? `COALESCE(qd.query_canonical, q.query)` : `query`;
236
+ const extraFileSets = useDim ? { QUERY_DIM: {
237
+ table: "queries",
238
+ keys: [dimStore.parquetKey(ctx)]
239
+ } } : void 0;
240
+ const sqlFor = dailyWindowSqlFor(useDim, canonExpr);
241
+ const cols = queryCanonicalDailyRollup.parquetColumns;
242
+ const sortKey = queryCanonicalDailyRollup.parquetSortKey;
243
+ let i = windowOffset;
244
+ let page = startPageOffset;
245
+ let nextWindowOffset = windowsTotal;
246
+ let nextPageOffset = 0;
247
+ let rowsWritten = 0;
248
+ let pausedMidWindow = false;
249
+ const flushPage = async (windowIdx, pageOffset, rows) => {
250
+ if (rows.length === 0) return;
251
+ const key = rollupParquetKey(ctx, `${CANONICAL_DAILY_PART_STEM}__w${windowIdx}_p${pageOffset}`, builtAt, searchType);
252
+ await dataSource.write(key, encodeRowsToParquetFlex(rows, {
253
+ columns: cols,
254
+ sortKey
255
+ }));
256
+ rowsWritten += rows.length;
257
+ };
258
+ for (; i < windowsTotal; i++) {
259
+ const w = windows[i];
260
+ for (; page < shardCount;) {
261
+ const coreSql = sqlFor(w, {
262
+ index: page,
263
+ count: shardCount
264
+ });
265
+ const result = await engine.runSQL({
266
+ ctx,
267
+ table: "queries",
268
+ ...sType,
269
+ fileSets: {
270
+ FILES: {
271
+ table: "queries",
272
+ partitions: w.partitions
273
+ },
274
+ ...extraFileSets
275
+ },
276
+ sql: `${coreSql}\nORDER BY date, query_canonical\nLIMIT ${pageRows}`
277
+ });
278
+ if (result.rows.length >= pageRows) throw new Error(`query_canonical_daily shard overflow: window=${i} shard=${page}/${shardCount} returned >= ${pageRows} rows; increase shardCount or pageRows`);
279
+ await flushPage(i, page, result.rows.map(mapDailyRow));
280
+ page += 1;
281
+ if (Date.now() > deadlineMs) {
282
+ nextWindowOffset = i;
283
+ nextPageOffset = page;
284
+ pausedMidWindow = true;
285
+ break;
286
+ }
287
+ }
288
+ if (pausedMidWindow) break;
289
+ page = 0;
290
+ if (Date.now() > deadlineMs) {
291
+ nextWindowOffset = i + 1;
292
+ nextPageOffset = 0;
293
+ break;
294
+ }
295
+ }
296
+ if (nextWindowOffset < windowsTotal || nextPageOffset > 0) return {
297
+ done: false,
298
+ nextWindowOffset,
299
+ nextPageOffset,
300
+ windowsTotal,
301
+ windowsBuilt: nextWindowOffset - windowOffset,
302
+ rowsWritten
303
+ };
304
+ const partPrefixDir = rollupParquetKey(ctx, CANONICAL_DAILY_PART_STEM, builtAt, searchType).replace(/[^/]*$/, "");
305
+ const partKeys = (await dataSource.list(partPrefixDir)).filter((k) => k.includes(`${CANONICAL_DAILY_PART_STEM}__w`) && k.endsWith(`__v${builtAt}.parquet`)).sort();
306
+ if (partKeys.length === 0) {
307
+ const emptyKey = rollupParquetKey(ctx, `${CANONICAL_DAILY_PART_STEM}__w0_p0`, builtAt, searchType);
308
+ await dataSource.write(emptyKey, encodeRowsToParquetFlex([], {
309
+ columns: cols,
310
+ sortKey
311
+ }));
312
+ partKeys.push(emptyKey);
313
+ }
314
+ const envelope = {
315
+ version: 1,
316
+ id: CANONICAL_DAILY_ROLLUP_FINAL_ID,
317
+ builtAt,
318
+ windowDays: queryCanonicalDailyRollup.windowDays,
319
+ payload: {
320
+ parquetKey: partKeys[0],
321
+ parquetKeys: partKeys,
322
+ rowCount: 0
323
+ }
324
+ };
325
+ await dataSource.write(rollupKey(ctx, CANONICAL_DAILY_ROLLUP_FINAL_ID, builtAt, searchType), encodeJsonBigintSafe(envelope));
326
+ return {
327
+ done: true,
328
+ nextWindowOffset,
329
+ nextPageOffset,
330
+ windowsTotal,
331
+ windowsBuilt: nextWindowOffset - windowOffset,
332
+ rowsWritten
333
+ };
334
+ }
335
+ export { queryCanonicalDailyRollup, queryCanonicalVariantsRollup, rebuildCanonicalDailyResumable };
@@ -0,0 +1,201 @@
1
+ import { DataSource, FileSetRef, Row as Row$1, TableName as TableName$1 } from "../storage.mjs";
2
+ import { ColumnDef } from "../schema.mjs";
3
+ import { EngineError } from "../errors.mjs";
4
+ import "../contracts.mjs";
5
+ import { TenantCtx } from "@gscdump/contracts";
6
+ import { SearchType } from "gscdump/query";
7
+ interface RollupCtx extends TenantCtx {
8
+ /** When the rollup was built. Stamped into payload + filename. */
9
+ builtAt: number;
10
+ }
11
+ /**
12
+ * Tenant-scoped engine surface a rollup builder needs. Subset of
13
+ * `StorageEngine.runSQL` so rollups stay testable without a full engine.
14
+ */
15
+ interface RollupEngine {
16
+ runSQL: (opts: {
17
+ ctx: TenantCtx;
18
+ fileSets: Record<string, FileSetRef>;
19
+ table?: TableName$1;
20
+ sql: string;
21
+ params?: unknown[];
22
+ /**
23
+ * Restrict every manifest lookup to a single GSC search-type slice. The
24
+ * rollup runner forwards `RebuildRollupsOptions.searchType` so the
25
+ * aggregated facts never mix web + non-web rows. Undefined preserves
26
+ * the legacy cross-type union (web-only tenants).
27
+ */
28
+ searchType?: SearchType;
29
+ }) => Promise<{
30
+ rows: Row$1[];
31
+ }>;
32
+ /**
33
+ * Read the live manifest for a (tenant, table[, searchType]) cohort —
34
+ * cheap, no parquet decode. Builders use this to chunk a full-history scan
35
+ * into byte-bounded windows (see `WINDOW_BYTE_BUDGET`) so a single `runSQL`
36
+ * call never ships an oversized Arrow IPC payload across the Workers
37
+ * service-binding RPC (32MiB hard cap).
38
+ */
39
+ listPartitions: (opts: {
40
+ ctx: TenantCtx;
41
+ table: TableName$1;
42
+ searchType?: SearchType;
43
+ }) => Promise<Array<{
44
+ partition: string;
45
+ bytes: number;
46
+ }>>;
47
+ }
48
+ /**
49
+ * One rollup definition. Build runs SQL over the tenant's facts and/or reads
50
+ * from entity stores via `dataSource`, returning a JSON-serializable payload
51
+ * that the runner timestamps + writes.
52
+ */
53
+ interface RollupDef {
54
+ id: string;
55
+ /**
56
+ * Window in days the rollup covers. `null` means full history. Used by
57
+ * the runner to populate `windowDays` in the payload metadata so readers
58
+ * can validate freshness.
59
+ */
60
+ windowDays: number | null;
61
+ /**
62
+ * Storage format. `'json'` (default) wraps the build payload in a
63
+ * `RollupEnvelope` and writes as a JSON blob. `'parquet'` expects `build`
64
+ * to return rows matching `parquetColumns` and writes a parquet file plus
65
+ * a tiny JSON sidecar envelope that points at it, so metadata
66
+ * (`builtAt` / `windowDays`) stays readable without decoding parquet.
67
+ */
68
+ format?: 'json' | 'parquet';
69
+ /**
70
+ * Column schema for parquet output. Required when `format === 'parquet'`.
71
+ * Types map the same way as the fact-table encoder: VARCHAR / DATE go
72
+ * through BYTE_ARRAY/UTF8; BIGINT → INT64; INTEGER → INT32; DOUBLE → DOUBLE.
73
+ */
74
+ parquetColumns?: readonly ColumnDef[];
75
+ /** Sort-key column names for parquet row-group stats. Optional. */
76
+ parquetSortKey?: readonly string[];
77
+ /**
78
+ * When true, this rollup's payload is independent of GSC slice (e.g. entity
79
+ * rollups sourced from sitemap / indexing snapshots, not slice-partitioned
80
+ * fact tables). The runner rejects calls that pass `searchType` alongside
81
+ * a slice-orthogonal def so the output never lands under a per-slice prefix
82
+ * that the read path won't look at.
83
+ */
84
+ sliceOrthogonal?: boolean;
85
+ build: (deps: {
86
+ engine: RollupEngine;
87
+ ctx: TenantCtx;
88
+ /**
89
+ * Tenant-scoped object store. Rollups that aggregate over entity
90
+ * snapshots (e.g. indexing metadata) read JSON docs through this.
91
+ * Pure-SQL rollups can ignore it.
92
+ */
93
+ dataSource: DataSource;
94
+ /**
95
+ * UTC millis the trailing window anchors to — its inclusive END. Equals
96
+ * the newest synced/finalized data date when the runner is given
97
+ * `dataEndDate`, otherwise wall-clock build time. Builders derive window
98
+ * cutoffs from this (e.g. the trailing-28d boundary) and inline a date
99
+ * literal so the SQL stays portable across DuckDB builds without the ICU
100
+ * extension (Workers DuckDB — `CURRENT_DATE` lives in ICU).
101
+ */
102
+ windowAnchorMs: number;
103
+ /**
104
+ * GSC search-type slice the runner was invoked for. Builders forward
105
+ * this to every `engine.runSQL` call so the aggregated facts come
106
+ * from one cohort. Undefined preserves the legacy cross-type union
107
+ * (used by web-only tenants and admin paths).
108
+ */
109
+ searchType?: SearchType;
110
+ }) => Promise<unknown>;
111
+ }
112
+ /**
113
+ * Wire shape persisted to R2/disk. Readers can rely on the `version` + `builtAt`.
114
+ * Parquet rollups write this envelope as a sidecar whose `payload` points at
115
+ * the co-located `.parquet` object via `{ parquetKey, rowCount }`.
116
+ */
117
+ interface RollupEnvelope<T = unknown> {
118
+ version: 1;
119
+ id: string;
120
+ builtAt: number;
121
+ windowDays: number | null;
122
+ payload: T;
123
+ }
124
+ interface ParquetRollupPointer {
125
+ parquetKey: string;
126
+ rowCount: number;
127
+ /**
128
+ * MULTI-FILE rollup: when set, the rollup is the UNION of these parquet keys
129
+ * (disjoint by the grain's partition column, e.g. `date` for the resumable
130
+ * `query_canonical_daily` build). Readers MUST union all keys; `parquetKey`
131
+ * stays populated (the first part) for single-file readers. Avoids a JS
132
+ * merge/re-encode of the whole rollup — the scaling bottleneck for a
133
+ * cross-invocation resumable build.
134
+ */
135
+ parquetKeys?: string[];
136
+ }
137
+ declare function rollupKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
138
+ declare function rollupParquetKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
139
+ interface RollupBucket {
140
+ list: (opts: {
141
+ prefix: string;
142
+ cursor?: string;
143
+ }) => Promise<{
144
+ objects: Array<{
145
+ key: string;
146
+ }>;
147
+ truncated?: boolean;
148
+ cursor?: string;
149
+ }>;
150
+ get: (key: string) => Promise<{
151
+ text: () => Promise<string>;
152
+ } | null>;
153
+ }
154
+ declare function readLatestRollup<T = unknown>(bucket: RollupBucket, ctx: TenantCtx, id: string, searchType?: SearchType): Promise<RollupEnvelope<T> | null>;
155
+ interface RebuildRollupsOptions {
156
+ engine: RollupEngine;
157
+ dataSource: DataSource;
158
+ ctx: TenantCtx;
159
+ defs: readonly RollupDef[];
160
+ now?: () => number;
161
+ /**
162
+ * Build rollups for a single GSC search-type slice. Threads into every
163
+ * builder's `engine.runSQL` call so the aggregated facts come from one
164
+ * cohort, and namespaces the output object keys under a `<searchType>/`
165
+ * segment so per-slice rollups coexist without overwriting each other.
166
+ * Undefined preserves the legacy cross-type behaviour (one rollup over
167
+ * the union of all slices, written to the legacy path) — fine for web-
168
+ * only tenants and explicit cross-type admin views.
169
+ */
170
+ searchType?: SearchType;
171
+ /**
172
+ * ISO date (`YYYY-MM-DD`) of the newest synced/finalized day. Trailing-
173
+ * window rollups (28d/90d) anchor their window END here instead of
174
+ * wall-clock build time, so a "last 28 days" rollup covers the 28 days of
175
+ * data that actually exist — not 28 days back from whenever the job ran,
176
+ * which would include GSC's 2-3 day empty tail. Omit for the legacy
177
+ * wall-clock behaviour.
178
+ */
179
+ dataEndDate?: string;
180
+ }
181
+ interface RebuildRollupResult {
182
+ id: string;
183
+ /** JSON envelope key. For parquet rollups this is the sidecar pointer. */
184
+ objectKey: string;
185
+ /** Parquet payload key. Present only when `format === 'parquet'`. */
186
+ parquetKey?: string;
187
+ /** Envelope byte size; for parquet rollups does NOT include parquet bytes. */
188
+ bytes: number;
189
+ /** Parquet payload byte size when `format === 'parquet'`. */
190
+ parquetBytes?: number;
191
+ builtAt: number;
192
+ /**
193
+ * Set when this def's build/encode/write failed. The runner records the
194
+ * failure and continues with the remaining defs so one bad rollup never
195
+ * aborts the rest. Successful defs have no `error`. The human-readable
196
+ * message (including the stack when available) lives on `error.message`.
197
+ */
198
+ error?: EngineError;
199
+ }
200
+ declare function rebuildRollups(opts: RebuildRollupsOptions): Promise<RebuildRollupResult[]>;
201
+ export { ParquetRollupPointer, RebuildRollupResult, RebuildRollupsOptions, RollupBucket, RollupCtx, RollupDef, RollupEngine, RollupEnvelope, readLatestRollup, rebuildRollups, rollupKey, rollupParquetKey };