@gscdump/engine 1.4.10 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/entities/empty-types.d.mts +22 -0
- package/dist/entities/empty-types.mjs +58 -0
- package/dist/entities/indexing-metadata.d.mts +26 -0
- package/dist/entities/indexing-metadata.mjs +31 -0
- package/dist/entities/inspection.d.mts +240 -0
- package/dist/entities/inspection.mjs +443 -0
- package/dist/entities/io.mjs +16 -0
- package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
- package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
- package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
- package/dist/entities/sitemap-shared.d.mts +199 -0
- package/dist/entities/sitemap-shared.mjs +243 -0
- package/dist/entities/sitemap-write.d.mts +3 -0
- package/dist/entities/sitemap-write.mjs +528 -0
- package/dist/entities/sitemap.d.mts +3 -0
- package/dist/entities/sitemap.mjs +91 -0
- package/dist/entities.d.mts +9 -481
- package/dist/entities.mjs +7 -1380
- package/dist/rollups/canonical.d.mts +71 -0
- package/dist/rollups/canonical.mjs +335 -0
- package/dist/rollups/core.d.mts +201 -0
- package/dist/rollups/core.mjs +116 -0
- package/dist/rollups/dates.mjs +11 -0
- package/dist/rollups/defaults.d.mts +11 -0
- package/dist/rollups/defaults.mjs +17 -0
- package/dist/rollups/hourly.d.mts +38 -0
- package/dist/rollups/hourly.mjs +38 -0
- package/dist/rollups/indexing.d.mts +46 -0
- package/dist/rollups/indexing.mjs +357 -0
- package/dist/rollups/traffic.d.mts +38 -0
- package/dist/rollups/traffic.mjs +289 -0
- package/dist/rollups/windows.d.mts +90 -0
- package/dist/rollups/windows.mjs +176 -0
- package/dist/rollups.d.mts +8 -471
- package/dist/rollups.mjs +7 -1316
- package/package.json +4 -4
- /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
package/dist/rollups.d.mts
CHANGED
|
@@ -1,471 +1,8 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import "./
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
builtAt: number;
|
|
10
|
-
}
|
|
11
|
-
/**
|
|
12
|
-
* Tenant-scoped engine surface a rollup builder needs. Subset of
|
|
13
|
-
* `StorageEngine.runSQL` so rollups stay testable without a full engine.
|
|
14
|
-
*/
|
|
15
|
-
interface RollupEngine {
|
|
16
|
-
runSQL: (opts: {
|
|
17
|
-
ctx: TenantCtx;
|
|
18
|
-
fileSets: Record<string, FileSetRef>;
|
|
19
|
-
table?: TableName$1;
|
|
20
|
-
sql: string;
|
|
21
|
-
params?: unknown[];
|
|
22
|
-
/**
|
|
23
|
-
* Restrict every manifest lookup to a single GSC search-type slice. The
|
|
24
|
-
* rollup runner forwards `RebuildRollupsOptions.searchType` so the
|
|
25
|
-
* aggregated facts never mix web + non-web rows. Undefined preserves
|
|
26
|
-
* the legacy cross-type union (web-only tenants).
|
|
27
|
-
*/
|
|
28
|
-
searchType?: SearchType;
|
|
29
|
-
}) => Promise<{
|
|
30
|
-
rows: Row$1[];
|
|
31
|
-
}>;
|
|
32
|
-
/**
|
|
33
|
-
* Read the live manifest for a (tenant, table[, searchType]) cohort —
|
|
34
|
-
* cheap, no parquet decode. Builders use this to chunk a full-history scan
|
|
35
|
-
* into byte-bounded windows (see `WINDOW_BYTE_BUDGET`) so a single `runSQL`
|
|
36
|
-
* call never ships an oversized Arrow IPC payload across the Workers
|
|
37
|
-
* service-binding RPC (32MiB hard cap).
|
|
38
|
-
*/
|
|
39
|
-
listPartitions: (opts: {
|
|
40
|
-
ctx: TenantCtx;
|
|
41
|
-
table: TableName$1;
|
|
42
|
-
searchType?: SearchType;
|
|
43
|
-
}) => Promise<Array<{
|
|
44
|
-
partition: string;
|
|
45
|
-
bytes: number;
|
|
46
|
-
}>>;
|
|
47
|
-
}
|
|
48
|
-
/**
|
|
49
|
-
* One rollup definition. Build runs SQL over the tenant's facts and/or reads
|
|
50
|
-
* from entity stores via `dataSource`, returning a JSON-serializable payload
|
|
51
|
-
* that the runner timestamps + writes.
|
|
52
|
-
*/
|
|
53
|
-
interface RollupDef {
|
|
54
|
-
id: string;
|
|
55
|
-
/**
|
|
56
|
-
* Window in days the rollup covers. `null` means full history. Used by
|
|
57
|
-
* the runner to populate `windowDays` in the payload metadata so readers
|
|
58
|
-
* can validate freshness.
|
|
59
|
-
*/
|
|
60
|
-
windowDays: number | null;
|
|
61
|
-
/**
|
|
62
|
-
* Storage format. `'json'` (default) wraps the build payload in a
|
|
63
|
-
* `RollupEnvelope` and writes as a JSON blob. `'parquet'` expects `build`
|
|
64
|
-
* to return rows matching `parquetColumns` and writes a parquet file plus
|
|
65
|
-
* a tiny JSON sidecar envelope that points at it, so metadata
|
|
66
|
-
* (`builtAt` / `windowDays`) stays readable without decoding parquet.
|
|
67
|
-
*/
|
|
68
|
-
format?: 'json' | 'parquet';
|
|
69
|
-
/**
|
|
70
|
-
* Column schema for parquet output. Required when `format === 'parquet'`.
|
|
71
|
-
* Types map the same way as the fact-table encoder: VARCHAR / DATE go
|
|
72
|
-
* through BYTE_ARRAY/UTF8; BIGINT → INT64; INTEGER → INT32; DOUBLE → DOUBLE.
|
|
73
|
-
*/
|
|
74
|
-
parquetColumns?: readonly ColumnDef[];
|
|
75
|
-
/** Sort-key column names for parquet row-group stats. Optional. */
|
|
76
|
-
parquetSortKey?: readonly string[];
|
|
77
|
-
/**
|
|
78
|
-
* When true, this rollup's payload is independent of GSC slice (e.g. entity
|
|
79
|
-
* rollups sourced from sitemap / indexing snapshots, not slice-partitioned
|
|
80
|
-
* fact tables). The runner rejects calls that pass `searchType` alongside
|
|
81
|
-
* a slice-orthogonal def so the output never lands under a per-slice prefix
|
|
82
|
-
* that the read path won't look at.
|
|
83
|
-
*/
|
|
84
|
-
sliceOrthogonal?: boolean;
|
|
85
|
-
build: (deps: {
|
|
86
|
-
engine: RollupEngine;
|
|
87
|
-
ctx: TenantCtx;
|
|
88
|
-
/**
|
|
89
|
-
* Tenant-scoped object store. Rollups that aggregate over entity
|
|
90
|
-
* snapshots (e.g. indexing metadata) read JSON docs through this.
|
|
91
|
-
* Pure-SQL rollups can ignore it.
|
|
92
|
-
*/
|
|
93
|
-
dataSource: DataSource;
|
|
94
|
-
/**
|
|
95
|
-
* UTC millis the trailing window anchors to — its inclusive END. Equals
|
|
96
|
-
* the newest synced/finalized data date when the runner is given
|
|
97
|
-
* `dataEndDate`, otherwise wall-clock build time. Builders derive window
|
|
98
|
-
* cutoffs from this (e.g. the trailing-28d boundary) and inline a date
|
|
99
|
-
* literal so the SQL stays portable across DuckDB builds without the ICU
|
|
100
|
-
* extension (Workers DuckDB — `CURRENT_DATE` lives in ICU).
|
|
101
|
-
*/
|
|
102
|
-
windowAnchorMs: number;
|
|
103
|
-
/**
|
|
104
|
-
* GSC search-type slice the runner was invoked for. Builders forward
|
|
105
|
-
* this to every `engine.runSQL` call so the aggregated facts come
|
|
106
|
-
* from one cohort. Undefined preserves the legacy cross-type union
|
|
107
|
-
* (used by web-only tenants and admin paths).
|
|
108
|
-
*/
|
|
109
|
-
searchType?: SearchType;
|
|
110
|
-
}) => Promise<unknown>;
|
|
111
|
-
}
|
|
112
|
-
/**
|
|
113
|
-
* Wire shape persisted to R2/disk. Readers can rely on the `version` + `builtAt`.
|
|
114
|
-
* Parquet rollups write this envelope as a sidecar whose `payload` points at
|
|
115
|
-
* the co-located `.parquet` object via `{ parquetKey, rowCount }`.
|
|
116
|
-
*/
|
|
117
|
-
interface RollupEnvelope<T = unknown> {
|
|
118
|
-
version: 1;
|
|
119
|
-
id: string;
|
|
120
|
-
builtAt: number;
|
|
121
|
-
windowDays: number | null;
|
|
122
|
-
payload: T;
|
|
123
|
-
}
|
|
124
|
-
interface ParquetRollupPointer {
|
|
125
|
-
parquetKey: string;
|
|
126
|
-
rowCount: number;
|
|
127
|
-
/**
|
|
128
|
-
* MULTI-FILE rollup: when set, the rollup is the UNION of these parquet keys
|
|
129
|
-
* (disjoint by the grain's partition column, e.g. `date` for the resumable
|
|
130
|
-
* `query_canonical_daily` build). Readers MUST union all keys; `parquetKey`
|
|
131
|
-
* stays populated (the first part) for single-file readers. Avoids a JS
|
|
132
|
-
* merge/re-encode of the whole rollup — the scaling bottleneck for a
|
|
133
|
-
* cross-invocation resumable build.
|
|
134
|
-
*/
|
|
135
|
-
parquetKeys?: string[];
|
|
136
|
-
}
|
|
137
|
-
declare function rollupKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
|
|
138
|
-
declare function rollupParquetKey(ctx: TenantCtx, id: string, builtAt: number, searchType?: SearchType): string;
|
|
139
|
-
interface RollupBucket {
|
|
140
|
-
list: (opts: {
|
|
141
|
-
prefix: string;
|
|
142
|
-
cursor?: string;
|
|
143
|
-
}) => Promise<{
|
|
144
|
-
objects: Array<{
|
|
145
|
-
key: string;
|
|
146
|
-
}>;
|
|
147
|
-
truncated?: boolean;
|
|
148
|
-
cursor?: string;
|
|
149
|
-
}>;
|
|
150
|
-
get: (key: string) => Promise<{
|
|
151
|
-
text: () => Promise<string>;
|
|
152
|
-
} | null>;
|
|
153
|
-
}
|
|
154
|
-
declare function readLatestRollup<T = unknown>(bucket: RollupBucket, ctx: TenantCtx, id: string, searchType?: SearchType): Promise<RollupEnvelope<T> | null>;
|
|
155
|
-
interface RebuildRollupsOptions {
|
|
156
|
-
engine: RollupEngine;
|
|
157
|
-
dataSource: DataSource;
|
|
158
|
-
ctx: TenantCtx;
|
|
159
|
-
defs: readonly RollupDef[];
|
|
160
|
-
now?: () => number;
|
|
161
|
-
/**
|
|
162
|
-
* Build rollups for a single GSC search-type slice. Threads into every
|
|
163
|
-
* builder's `engine.runSQL` call so the aggregated facts come from one
|
|
164
|
-
* cohort, and namespaces the output object keys under a `<searchType>/`
|
|
165
|
-
* segment so per-slice rollups coexist without overwriting each other.
|
|
166
|
-
* Undefined preserves the legacy cross-type behaviour (one rollup over
|
|
167
|
-
* the union of all slices, written to the legacy path) — fine for web-
|
|
168
|
-
* only tenants and explicit cross-type admin views.
|
|
169
|
-
*/
|
|
170
|
-
searchType?: SearchType;
|
|
171
|
-
/**
|
|
172
|
-
* ISO date (`YYYY-MM-DD`) of the newest synced/finalized day. Trailing-
|
|
173
|
-
* window rollups (28d/90d) anchor their window END here instead of
|
|
174
|
-
* wall-clock build time, so a "last 28 days" rollup covers the 28 days of
|
|
175
|
-
* data that actually exist — not 28 days back from whenever the job ran,
|
|
176
|
-
* which would include GSC's 2-3 day empty tail. Omit for the legacy
|
|
177
|
-
* wall-clock behaviour.
|
|
178
|
-
*/
|
|
179
|
-
dataEndDate?: string;
|
|
180
|
-
}
|
|
181
|
-
interface RebuildRollupResult {
|
|
182
|
-
id: string;
|
|
183
|
-
/** JSON envelope key. For parquet rollups this is the sidecar pointer. */
|
|
184
|
-
objectKey: string;
|
|
185
|
-
/** Parquet payload key. Present only when `format === 'parquet'`. */
|
|
186
|
-
parquetKey?: string;
|
|
187
|
-
/** Envelope byte size; for parquet rollups does NOT include parquet bytes. */
|
|
188
|
-
bytes: number;
|
|
189
|
-
/** Parquet payload byte size when `format === 'parquet'`. */
|
|
190
|
-
parquetBytes?: number;
|
|
191
|
-
builtAt: number;
|
|
192
|
-
/**
|
|
193
|
-
* Set when this def's build/encode/write failed. The runner records the
|
|
194
|
-
* failure and continues with the remaining defs so one bad rollup never
|
|
195
|
-
* aborts the rest. Successful defs have no `error`. The human-readable
|
|
196
|
-
* message (including the stack when available) lives on `error.message`.
|
|
197
|
-
*/
|
|
198
|
-
error?: EngineError;
|
|
199
|
-
}
|
|
200
|
-
declare function rebuildRollups(opts: RebuildRollupsOptions): Promise<RebuildRollupResult[]>;
|
|
201
|
-
/**
|
|
202
|
-
* Per-window budget, measured in *parquet* bytes (manifest `bytes`), used by
|
|
203
|
-
* `planRollupWindows` to chunk a full-history scan.
|
|
204
|
-
*
|
|
205
|
-
* The executor decodes a window's parquet and ships it as an Arrow IPC stream
|
|
206
|
-
* over the service binding; that IPC is hard-guarded at 28MiB
|
|
207
|
-
* (`IPC_PLACEHOLDER_BUDGET` in @gscdump/cloudflare). Parquet is compressed and
|
|
208
|
-
* the IPC stream is not, so a window inflates on the wire — keep this
|
|
209
|
-
* conservatively below the guard. Re-measure the parquet→IPC ratio against
|
|
210
|
-
* production and raise if headroom allows.
|
|
211
|
-
*/
|
|
212
|
-
declare const WINDOW_BYTE_BUDGET: number;
|
|
213
|
-
/**
|
|
214
|
-
* Per-page OUTPUT row cap for key-paginated rollups (`runWindowed({ paginate })`
|
|
215
|
-
* and `runPagedQuery`). `planRollupWindows` bounds the *input* parquet bytes a
|
|
216
|
-
* window scans, which is a fine proxy for output size on fact aggregations whose
|
|
217
|
-
* grain matches the input (one output row per input date). It is NOT a proxy for
|
|
218
|
-
* aggregations that COLLAPSE to a smaller-cardinality grain whose row count is
|
|
219
|
-
* driven by a high-cardinality GROUP key — `(query_canonical × date)` and
|
|
220
|
-
* `(query_canonical)` — where output rows scale with distinct canonicals, not
|
|
221
|
-
* input bytes. For those, each `runSQL` result (shipped as an Arrow IPC stream
|
|
222
|
-
* over the Workers service-binding RPC; 28MiB guard in `@gscdump/cloudflare`,
|
|
223
|
-
* duckdb-worker `assertResultBudget` at 24MiB / 100k rows) must be bounded by
|
|
224
|
-
* paging the OUTPUT, independent of how the input is windowed.
|
|
225
|
-
*
|
|
226
|
-
* Narrow rows — `(canonical, date, 3 metrics)` — page at 50k (≈16MiB at the
|
|
227
|
-
* worker's `cols×64` heuristic, well under both guards). WIDE rows carry a
|
|
228
|
-
* `GROUP_CONCAT` variants string (up to ~10 variants × ~60 chars) the heuristic
|
|
229
|
-
* under-counts, so they page smaller to keep the real IPC payload bounded.
|
|
230
|
-
*/
|
|
231
|
-
declare const ROLLUP_PAGE_ROWS = 50000;
|
|
232
|
-
declare const ROLLUP_PAGE_ROWS_WIDE = 20000;
|
|
233
|
-
declare const ROLLUP_PAGE_ROWS_DAILY = 70000;
|
|
234
|
-
/**
|
|
235
|
-
* Plan byte-bounded windows over a partition set. Each window names the
|
|
236
|
-
* partitions whose span intersects it; a coarse tier file can land in two
|
|
237
|
-
* windows, so every windowed SQL MUST also date-filter to the window bounds.
|
|
238
|
-
*/
|
|
239
|
-
declare function planRollupWindows(parts: Array<{
|
|
240
|
-
partition: string;
|
|
241
|
-
bytes: number;
|
|
242
|
-
}>, clampRange?: {
|
|
243
|
-
start: string;
|
|
244
|
-
end: string;
|
|
245
|
-
}, maxWindowDays?: number): Array<{
|
|
246
|
-
start: string;
|
|
247
|
-
end: string;
|
|
248
|
-
partitions: string[];
|
|
249
|
-
}>;
|
|
250
|
-
/**
|
|
251
|
-
* Run a full-history aggregation in byte-bounded windows and concat the rows.
|
|
252
|
-
* Each window's SQL MUST date-filter to `[w.start, w.end]` (see `sqlFor`) so a
|
|
253
|
-
* tier file spanning a window boundary doesn't double-count calendar dates.
|
|
254
|
-
*
|
|
255
|
-
* `paginate` additionally pages each window's OUTPUT (see `runPagedQuery`) so a
|
|
256
|
-
* window whose GROUP cardinality is high — `(query_canonical × date)` on a large
|
|
257
|
-
* site — can't ship an oversized result even though its input bytes fit a window.
|
|
258
|
-
* Date-windowing bounds the per-query scan; output paging bounds the IPC payload.
|
|
259
|
-
* The two are orthogonal and compose. When `paginate` is set, `sqlFor` MUST emit
|
|
260
|
-
* no trailing `ORDER BY`/`LIMIT` and `paginate.orderBy` MUST be a total order.
|
|
261
|
-
*/
|
|
262
|
-
declare function runWindowed(opts: {
|
|
263
|
-
engine: RollupEngine;
|
|
264
|
-
ctx: TenantCtx;
|
|
265
|
-
table: TableName$1;
|
|
266
|
-
searchType?: SearchType;
|
|
267
|
-
sqlFor: (w: {
|
|
268
|
-
start: string;
|
|
269
|
-
end: string;
|
|
270
|
-
}) => string;
|
|
271
|
-
/**
|
|
272
|
-
* Extra named file sets merged into every window's `runSQL` (alongside the
|
|
273
|
-
* windowed `FILES`). Use to JOIN a non-windowed sidecar (e.g. the query
|
|
274
|
-
* dimension parquet via `{ QUERY_DIM: { keys: [...] } }`) inside `sqlFor`.
|
|
275
|
-
*/
|
|
276
|
-
extraFileSets?: Record<string, FileSetRef>;
|
|
277
|
-
/** Page each window's output by a total-order key. See `runPagedQuery`. */
|
|
278
|
-
paginate?: {
|
|
279
|
-
orderBy: string;
|
|
280
|
-
pageRows: number;
|
|
281
|
-
};
|
|
282
|
-
/** Cap each window's day span (output-cardinality bound). See `planRollupWindows`. */
|
|
283
|
-
maxWindowDays?: number;
|
|
284
|
-
}): Promise<Row$1[]>;
|
|
285
|
-
/**
|
|
286
|
-
* Daily totals across the full history. One row per (date, table) with
|
|
287
|
-
* clicks + impressions + position. Powers sparklines and headline totals.
|
|
288
|
-
*
|
|
289
|
-
* Includes `anonymizedImpressionsPct` per day computed as
|
|
290
|
-
* 1 - sum(query_grained_impressions) / sum(page_grained_impressions)
|
|
291
|
-
* — surfaces GSC's anonymous-query gap so the dashboard can warn users not
|
|
292
|
-
* to trust query-grained breakdowns as comprehensive.
|
|
293
|
-
*/
|
|
294
|
-
declare const dailyTotalsRollup: RollupDef;
|
|
295
|
-
/** Weekly totals, ISO week aligned. Cheap and stable for trend widgets. */
|
|
296
|
-
declare const weeklyTotalsRollup: RollupDef;
|
|
297
|
-
/**
|
|
298
|
-
* Top 1000 pages by clicks over the trailing 28-day window. JSON for v1;
|
|
299
|
-
* promote to parquet (`top_pages_28d.parquet`) when the dashboard needs
|
|
300
|
-
* server-side WHERE filtering on this rollup.
|
|
301
|
-
*/
|
|
302
|
-
declare const topPages28dRollup: RollupDef;
|
|
303
|
-
/**
|
|
304
|
-
* Top 250 countries by clicks over the trailing 28-day window. Countries
|
|
305
|
-
* cardinality is bounded (~250 ISO codes), so the list fits in a tiny JSON
|
|
306
|
-
* payload regardless of traffic shape. Powers a geo-overview widget without
|
|
307
|
-
* spinning up DuckDB-WASM.
|
|
308
|
-
*/
|
|
309
|
-
declare const topCountries28dRollup: RollupDef;
|
|
310
|
-
/**
|
|
311
|
-
* Parquet-format companion to `topKeywords28dRollup`. Same shape, but persists
|
|
312
|
-
* as a parquet object plus JSON sidecar pointer so widgets that need
|
|
313
|
-
* server-side WHERE (filter by prefix, by clicks threshold, paginate) can scan
|
|
314
|
-
* it directly with DuckDB-WASM instead of loading all 1000 rows into JS.
|
|
315
|
-
*
|
|
316
|
-
* Opt-in: include in the caller's rollup def list alongside (or instead of)
|
|
317
|
-
* the JSON variant; the runner treats the two as independent ids so they can
|
|
318
|
-
* coexist during a migration.
|
|
319
|
-
*/
|
|
320
|
-
declare const topKeywords28dParquetRollup: RollupDef;
|
|
321
|
-
declare const queryCanonicalVariantsRollup: RollupDef;
|
|
322
|
-
/**
|
|
323
|
-
* Canonical-grained fact aggregate (ADR-0018 Gap 2): pre-sums the raw
|
|
324
|
-
* `(query × date)` query rows to `(query_canonical × date)`, so canonical-
|
|
325
|
-
* primary top/gaining/losing reads a small pre-aggregated table instead of
|
|
326
|
-
* re-collapsing variants on every request. Metrics are additive, so summing
|
|
327
|
-
* these per-date sums over a window is exact — identical to aggregating the raw
|
|
328
|
-
* rows.
|
|
329
|
-
*
|
|
330
|
-
* Null-free by construction: groups by the versioned query dimension when it
|
|
331
|
-
* exists, with raw query as the fallback, so the rollup never carries a NULL/''
|
|
332
|
-
* canonical bucket and the read path can treat the rollup's `query_canonical`
|
|
333
|
-
* column as already-derived.
|
|
334
|
-
*
|
|
335
|
-
* Date-grained full history (`windowDays: null`): one rollup serves every date
|
|
336
|
-
* range (reads filter by `date`) and both windows of a comparison. Opt-in (not
|
|
337
|
-
* in `DEFAULT_ROLLUPS`); the host points the main query's file set at it for
|
|
338
|
-
* queries the rollup covers (see `canonicalRollupCovers` /
|
|
339
|
-
* `RunOptimizedQueryOptions.canonicalSource`).
|
|
340
|
-
*/
|
|
341
|
-
declare const queryCanonicalDailyRollup: RollupDef;
|
|
342
|
-
/**
|
|
343
|
-
* Resumable, cross-invocation build of `query_canonical_daily` for a high-
|
|
344
|
-
* cardinality site whose full windowed build exceeds one job reservation (300s).
|
|
345
|
-
*
|
|
346
|
-
* Each call builds from `(windowOffset, pageOffset)` until `deadlineMs`, writes
|
|
347
|
-
* that batch's rows to a PART parquet, and returns `{ done:false, nextWindowOffset,
|
|
348
|
-
* nextPageOffset }` for the caller to re-enqueue. When the last window is fully
|
|
349
|
-
* paged it publishes a multi-file envelope listing every part (parts are disjoint
|
|
350
|
-
* by `(query_canonical, date)`, so the read path just unions them — no merge),
|
|
351
|
-
* returning `{ done:true }`. `builtAt` MUST be stable across the continuation chain
|
|
352
|
-
* (it versions both the part keys and the final rollup key).
|
|
353
|
-
*
|
|
354
|
-
* INTRA-WINDOW resumability: the deadline is checked between raw-query hash shards,
|
|
355
|
-
* not just between date windows. A single high-cardinality day can spend a full
|
|
356
|
-
* reservation inside one grouped/sorted aggregate before the deadline check gets
|
|
357
|
-
* control back. `pageOffset` is the next shard index for the current window, so a
|
|
358
|
-
* continuation resumes the SAME day at the next shard. Parts are keyed by
|
|
359
|
-
* `(windowOffset, pageOffset)`; multiple parts may contain the same canonical/date
|
|
360
|
-
* from different raw-query shards, and rollup reads sum over the union.
|
|
361
|
-
*/
|
|
362
|
-
declare function rebuildCanonicalDailyResumable(opts: {
|
|
363
|
-
engine: RollupEngine;
|
|
364
|
-
ctx: TenantCtx;
|
|
365
|
-
dataSource: DataSource;
|
|
366
|
-
searchType?: SearchType;
|
|
367
|
-
builtAt: number;
|
|
368
|
-
windowOffset: number;
|
|
369
|
-
/** Resume the `windowOffset` window at this shard offset (0 = window start). */
|
|
370
|
-
pageOffset?: number;
|
|
371
|
-
/** Output rows per page (default `ROLLUP_PAGE_ROWS_DAILY`). Injectable for tests. */
|
|
372
|
-
pageRows?: number;
|
|
373
|
-
/** Cap each input window's day span (default `DAILY_MAX_WINDOW_DAYS`). */
|
|
374
|
-
maxWindowDays?: number;
|
|
375
|
-
/** Split each date window by raw-query hash before grouping (1 = no sharding). */
|
|
376
|
-
shardCount?: number;
|
|
377
|
-
deadlineMs: number;
|
|
378
|
-
}): Promise<{
|
|
379
|
-
done: boolean;
|
|
380
|
-
nextWindowOffset: number;
|
|
381
|
-
nextPageOffset: number;
|
|
382
|
-
windowsTotal: number;
|
|
383
|
-
windowsBuilt: number;
|
|
384
|
-
rowsWritten: number;
|
|
385
|
-
}>;
|
|
386
|
-
/**
|
|
387
|
-
* Aggregates the per-URL Indexing API metadata entity store (populated by
|
|
388
|
-
* `gscdump entities indexing snapshot`) into daily counts of `URL_UPDATED`
|
|
389
|
-
* and `URL_REMOVED` notifications. Covers the third entity-snapshot shape
|
|
390
|
-
* without needing its own parquet family — publish events are sparse and
|
|
391
|
-
* aggregate cleanly into a small JSON rollup.
|
|
392
|
-
*
|
|
393
|
-
* Safe no-op when the entity store is empty: returns `{ totals: {...}, days: [] }`
|
|
394
|
-
* so downstream readers don't have to special-case first-run sites.
|
|
395
|
-
*/
|
|
396
|
-
declare const indexingMetadataRollup: RollupDef;
|
|
397
|
-
/**
|
|
398
|
-
* Indexing-API health by day: per `inspectedAt` date, counts of indexed,
|
|
399
|
-
* soft-404, redirect, not-found, mobile passes, rich-results passes, and
|
|
400
|
-
* canonical mismatches. Sourced from the inspections parquet sidecar
|
|
401
|
-
* (`InspectionStore.parquetUri`), which holds the latest record per URL.
|
|
402
|
-
*
|
|
403
|
-
* Empty-payload no-op when the sidecar URI is unavailable (in-memory
|
|
404
|
-
* `DataSource`, or before `materialize` has run).
|
|
405
|
-
*/
|
|
406
|
-
declare const indexingHealthRollup: RollupDef;
|
|
407
|
-
/**
|
|
408
|
-
* Per-day index-percent: ratio of (sitemap URLs that received GSC clicks on
|
|
409
|
-
* that date) / (total live sitemap URLs). Uses a DuckDB JOIN between the
|
|
410
|
-
* sitemap urls parquet (`SitemapStore.urlsParquetUri`) and the `pages` fact
|
|
411
|
-
* parquet. Total denominator is the count of live URLs in the urls index;
|
|
412
|
-
* numerator is per-day distinct loc count where pages.clicks > 0.
|
|
413
|
-
*/
|
|
414
|
-
declare const indexPercentRollup: RollupDef;
|
|
415
|
-
/**
|
|
416
|
-
* Sitemap-health per-day series materialized from the sitemap-store JSON
|
|
417
|
-
* index. Each `SitemapRecord` carries `urlCount`, `errors`, `warnings`,
|
|
418
|
-
* `contentHash`, and `lastDownloaded`. We bucket records by the day of their
|
|
419
|
-
* `capturedAt` (or `lastDownloaded` fallback) and emit per-day aggregates plus
|
|
420
|
-
* a snapshot of per-feed stats at the most recent capture.
|
|
421
|
-
*/
|
|
422
|
-
declare const sitemapHealthRollup: RollupDef;
|
|
423
|
-
/**
|
|
424
|
-
* Trailing-28-day sitemap URL changes: per-day per-feedpath {added, removed}
|
|
425
|
-
* counts plus rolling top-200 added and removed URLs. Streams from
|
|
426
|
-
* retained `SitemapReadStore.loadEvents()` history. State compaction cannot
|
|
427
|
-
* erase analytics input; memory scales independently of total site state.
|
|
428
|
-
*/
|
|
429
|
-
declare const sitemapChanges28dRollup: RollupDef;
|
|
430
|
-
/**
|
|
431
|
-
* Aggregate one day's `hourly_pages` partition into the daily `pages` shape
|
|
432
|
-
* and write it to the daily Discover partition. After this runs for date D,
|
|
433
|
-
* the daily query path serves D from `pages/.../daily/D` and the `hourly/D`
|
|
434
|
-
* partition becomes read-only / GC-only.
|
|
435
|
-
*
|
|
436
|
-
* `(position - 1)` weighting matches the storage convention encoded by
|
|
437
|
-
* `toSumPosition`: `sum_position = SUM((position - 1) * impressions)`, so a
|
|
438
|
-
* downstream `SUM(sum_position) / SUM(impressions) + 1` recovers the mean.
|
|
439
|
-
*
|
|
440
|
-
* searchType-scoped: only call with `searchType: 'discover'`. The hourly
|
|
441
|
-
* partition lives under `hourly_pages` and the output lands under `pages` so
|
|
442
|
-
* existing dashboard queries (which read `pages`) see the rolled-up day
|
|
443
|
-
* transparently.
|
|
444
|
-
*/
|
|
445
|
-
interface RebuildDailyFromHourlyOptions {
|
|
446
|
-
engine: RollupEngine & {
|
|
447
|
-
writeDay: (scope: TenantCtx & {
|
|
448
|
-
table: TableTypeName;
|
|
449
|
-
date: string;
|
|
450
|
-
searchType?: SearchType;
|
|
451
|
-
}, rows: Row$1[]) => Promise<void>;
|
|
452
|
-
};
|
|
453
|
-
ctx: TenantCtx;
|
|
454
|
-
/** PT calendar day to roll up. */
|
|
455
|
-
date: string;
|
|
456
|
-
searchType: 'discover';
|
|
457
|
-
}
|
|
458
|
-
type TableTypeName = import('@gscdump/contracts').TableName;
|
|
459
|
-
declare function rebuildDailyFromHourly(opts: RebuildDailyFromHourlyOptions): Promise<{
|
|
460
|
-
rowsWritten: number;
|
|
461
|
-
}>;
|
|
462
|
-
declare const DEFAULT_ROLLUPS: readonly RollupDef[];
|
|
463
|
-
/**
|
|
464
|
-
* Canonical-primary rollups (ADR-0017 / ADR-0018). Opt-in — kept out of
|
|
465
|
-
* `DEFAULT_ROLLUPS` because they only pay off once the consumer queries by
|
|
466
|
-
* `queryCanonical` and wires the read seams (`resolveExtra` /
|
|
467
|
-
* `canonicalSource`). Hosts opt in by concatenating these onto their def list
|
|
468
|
-
* (CLI: `gscdump rollups --with-canonical`).
|
|
469
|
-
*/
|
|
470
|
-
declare const CANONICAL_ROLLUPS: readonly RollupDef[];
|
|
471
|
-
export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS, ParquetRollupPointer, ROLLUP_PAGE_ROWS, ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, RebuildDailyFromHourlyOptions, RebuildRollupResult, RebuildRollupsOptions, RollupBucket, RollupCtx, RollupDef, RollupEngine, RollupEnvelope, WINDOW_BYTE_BUDGET, dailyTotalsRollup, indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, planRollupWindows, queryCanonicalDailyRollup, queryCanonicalVariantsRollup, readLatestRollup, rebuildCanonicalDailyResumable, rebuildDailyFromHourly, rebuildRollups, rollupKey, rollupParquetKey, runWindowed, sitemapChanges28dRollup, sitemapHealthRollup, topCountries28dRollup, topKeywords28dParquetRollup, topPages28dRollup, weeklyTotalsRollup };
|
|
1
|
+
import { ParquetRollupPointer, RebuildRollupResult, RebuildRollupsOptions, RollupBucket, RollupCtx, RollupDef, RollupEngine, RollupEnvelope, readLatestRollup, rebuildRollups, rollupKey, rollupParquetKey } from "./rollups/core.mjs";
|
|
2
|
+
import { queryCanonicalDailyRollup, queryCanonicalVariantsRollup, rebuildCanonicalDailyResumable } from "./rollups/canonical.mjs";
|
|
3
|
+
import { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS } from "./rollups/defaults.mjs";
|
|
4
|
+
import { RebuildDailyFromHourlyOptions, rebuildDailyFromHourly } from "./rollups/hourly.mjs";
|
|
5
|
+
import { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup } from "./rollups/indexing.mjs";
|
|
6
|
+
import { dailyTotalsRollup, topCountries28dRollup, topKeywords28dParquetRollup, topPages28dRollup, weeklyTotalsRollup } from "./rollups/traffic.mjs";
|
|
7
|
+
import { ROLLUP_PAGE_ROWS, ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, WINDOW_BYTE_BUDGET, planRollupWindows, runWindowed } from "./rollups/windows.mjs";
|
|
8
|
+
export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS, type ParquetRollupPointer, ROLLUP_PAGE_ROWS, ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, type RebuildDailyFromHourlyOptions, type RebuildRollupResult, type RebuildRollupsOptions, type RollupBucket, type RollupCtx, type RollupDef, type RollupEngine, type RollupEnvelope, WINDOW_BYTE_BUDGET, dailyTotalsRollup, indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, planRollupWindows, queryCanonicalDailyRollup, queryCanonicalVariantsRollup, readLatestRollup, rebuildCanonicalDailyResumable, rebuildDailyFromHourly, rebuildRollups, rollupKey, rollupParquetKey, runWindowed, sitemapChanges28dRollup, sitemapHealthRollup, topCountries28dRollup, topKeywords28dParquetRollup, topPages28dRollup, weeklyTotalsRollup };
|