@gscdump/engine 1.4.10 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/entities/empty-types.d.mts +22 -0
- package/dist/entities/empty-types.mjs +58 -0
- package/dist/entities/indexing-metadata.d.mts +26 -0
- package/dist/entities/indexing-metadata.mjs +31 -0
- package/dist/entities/inspection.d.mts +240 -0
- package/dist/entities/inspection.mjs +443 -0
- package/dist/entities/io.mjs +16 -0
- package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
- package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
- package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
- package/dist/entities/sitemap-shared.d.mts +199 -0
- package/dist/entities/sitemap-shared.mjs +243 -0
- package/dist/entities/sitemap-write.d.mts +3 -0
- package/dist/entities/sitemap-write.mjs +528 -0
- package/dist/entities/sitemap.d.mts +3 -0
- package/dist/entities/sitemap.mjs +91 -0
- package/dist/entities.d.mts +9 -481
- package/dist/entities.mjs +7 -1380
- package/dist/rollups/canonical.d.mts +71 -0
- package/dist/rollups/canonical.mjs +335 -0
- package/dist/rollups/core.d.mts +201 -0
- package/dist/rollups/core.mjs +116 -0
- package/dist/rollups/dates.mjs +11 -0
- package/dist/rollups/defaults.d.mts +11 -0
- package/dist/rollups/defaults.mjs +17 -0
- package/dist/rollups/hourly.d.mts +38 -0
- package/dist/rollups/hourly.mjs +38 -0
- package/dist/rollups/indexing.d.mts +46 -0
- package/dist/rollups/indexing.mjs +357 -0
- package/dist/rollups/traffic.d.mts +38 -0
- package/dist/rollups/traffic.mjs +289 -0
- package/dist/rollups/windows.d.mts +90 -0
- package/dist/rollups/windows.mjs +176 -0
- package/dist/rollups.d.mts +8 -471
- package/dist/rollups.mjs +7 -1316
- package/package.json +4 -4
- /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import { utcDateMinusDays } from "./dates.mjs";
|
|
2
|
+
import { partitionsInRange, runWindowed } from "./windows.mjs";
|
|
3
|
+
const dailyTotalsRollup = {
|
|
4
|
+
id: "daily_totals",
|
|
5
|
+
windowDays: null,
|
|
6
|
+
async build({ engine, ctx, searchType }) {
|
|
7
|
+
const pageRows = await runWindowed({
|
|
8
|
+
engine,
|
|
9
|
+
ctx,
|
|
10
|
+
table: "pages",
|
|
11
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
12
|
+
sqlFor: (w) => `
|
|
13
|
+
SELECT
|
|
14
|
+
date,
|
|
15
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
16
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
17
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
18
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
19
|
+
WHERE date >= '${w.start}' AND date <= '${w.end}'
|
|
20
|
+
GROUP BY date
|
|
21
|
+
ORDER BY date
|
|
22
|
+
`
|
|
23
|
+
});
|
|
24
|
+
const queryRows = await runWindowed({
|
|
25
|
+
engine,
|
|
26
|
+
ctx,
|
|
27
|
+
table: "queries",
|
|
28
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
29
|
+
sqlFor: (w) => `
|
|
30
|
+
SELECT
|
|
31
|
+
date,
|
|
32
|
+
SUM(impressions)::BIGINT AS impressions
|
|
33
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
34
|
+
WHERE date >= '${w.start}' AND date <= '${w.end}'
|
|
35
|
+
GROUP BY date
|
|
36
|
+
`
|
|
37
|
+
});
|
|
38
|
+
const pagesByDate = /* @__PURE__ */ new Map();
|
|
39
|
+
for (const r of pageRows) {
|
|
40
|
+
const date = String(r.date);
|
|
41
|
+
const cur = pagesByDate.get(date) ?? {
|
|
42
|
+
date,
|
|
43
|
+
clicks: BigInt(0),
|
|
44
|
+
impressions: BigInt(0),
|
|
45
|
+
sum_position: 0
|
|
46
|
+
};
|
|
47
|
+
cur.clicks += BigInt(r.clicks);
|
|
48
|
+
cur.impressions += BigInt(r.impressions);
|
|
49
|
+
cur.sum_position += Number(r.sum_position);
|
|
50
|
+
pagesByDate.set(date, cur);
|
|
51
|
+
}
|
|
52
|
+
const queryImpressionsByDate = /* @__PURE__ */ new Map();
|
|
53
|
+
for (const r of queryRows) {
|
|
54
|
+
const date = String(r.date);
|
|
55
|
+
queryImpressionsByDate.set(date, (queryImpressionsByDate.get(date) ?? BigInt(0)) + BigInt(r.impressions));
|
|
56
|
+
}
|
|
57
|
+
return Array.from(pagesByDate.values()).sort((a, b) => a.date < b.date ? -1 : 1).map((r) => {
|
|
58
|
+
const totalImpressions = BigInt(r.impressions);
|
|
59
|
+
const queryImpressions = queryImpressionsByDate.get(String(r.date)) ?? BigInt(0);
|
|
60
|
+
const anonymized = totalImpressions === BigInt(0) ? 0 : 1 - Number(queryImpressions) / Number(totalImpressions);
|
|
61
|
+
return {
|
|
62
|
+
date: r.date,
|
|
63
|
+
clicks: Number(r.clicks),
|
|
64
|
+
impressions: Number(r.impressions),
|
|
65
|
+
sum_position: Number(r.sum_position),
|
|
66
|
+
anonymizedImpressionsPct: Math.max(0, Math.min(1, anonymized))
|
|
67
|
+
};
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
};
|
|
71
|
+
const weeklyTotalsRollup = {
|
|
72
|
+
id: "weekly_totals",
|
|
73
|
+
windowDays: null,
|
|
74
|
+
async build({ engine, ctx, searchType }) {
|
|
75
|
+
const rows = await runWindowed({
|
|
76
|
+
engine,
|
|
77
|
+
ctx,
|
|
78
|
+
table: "pages",
|
|
79
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
80
|
+
sqlFor: (w) => `
|
|
81
|
+
SELECT
|
|
82
|
+
strftime(date_trunc('week', date::DATE), '%Y-%m-%d') AS week,
|
|
83
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
84
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
85
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
86
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
87
|
+
WHERE date >= '${w.start}' AND date <= '${w.end}'
|
|
88
|
+
GROUP BY 1
|
|
89
|
+
ORDER BY 1
|
|
90
|
+
`
|
|
91
|
+
});
|
|
92
|
+
const byWeek = /* @__PURE__ */ new Map();
|
|
93
|
+
for (const r of rows) {
|
|
94
|
+
const week = String(r.week);
|
|
95
|
+
const cur = byWeek.get(week) ?? {
|
|
96
|
+
week,
|
|
97
|
+
clicks: 0,
|
|
98
|
+
impressions: 0,
|
|
99
|
+
sum_position: 0
|
|
100
|
+
};
|
|
101
|
+
cur.clicks += Number(r.clicks);
|
|
102
|
+
cur.impressions += Number(r.impressions);
|
|
103
|
+
cur.sum_position += Number(r.sum_position);
|
|
104
|
+
byWeek.set(week, cur);
|
|
105
|
+
}
|
|
106
|
+
return Array.from(byWeek.values()).sort((a, b) => a.week < b.week ? -1 : 1);
|
|
107
|
+
}
|
|
108
|
+
};
|
|
109
|
+
const topPages28dRollup = {
|
|
110
|
+
id: "top_pages_28d",
|
|
111
|
+
windowDays: 28,
|
|
112
|
+
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
113
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
114
|
+
const partitions = partitionsInRange(await engine.listPartitions({
|
|
115
|
+
ctx,
|
|
116
|
+
table: "pages",
|
|
117
|
+
...searchType !== void 0 ? { searchType } : {}
|
|
118
|
+
}), cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
119
|
+
if (partitions.length === 0) return [];
|
|
120
|
+
return (await engine.runSQL({
|
|
121
|
+
ctx,
|
|
122
|
+
table: "pages",
|
|
123
|
+
fileSets: { FILES: {
|
|
124
|
+
table: "pages",
|
|
125
|
+
partitions
|
|
126
|
+
} },
|
|
127
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
128
|
+
sql: `
|
|
129
|
+
SELECT
|
|
130
|
+
url,
|
|
131
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
132
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
133
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
134
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
135
|
+
WHERE date >= '${cutoff}'
|
|
136
|
+
GROUP BY url
|
|
137
|
+
ORDER BY clicks DESC
|
|
138
|
+
LIMIT 1000
|
|
139
|
+
`
|
|
140
|
+
})).rows.map((r) => ({
|
|
141
|
+
url: r.url,
|
|
142
|
+
clicks: Number(r.clicks),
|
|
143
|
+
impressions: Number(r.impressions),
|
|
144
|
+
sum_position: Number(r.sum_position)
|
|
145
|
+
}));
|
|
146
|
+
}
|
|
147
|
+
};
|
|
148
|
+
const topCountries28dRollup = {
|
|
149
|
+
id: "top_countries_28d",
|
|
150
|
+
windowDays: 28,
|
|
151
|
+
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
152
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
153
|
+
const partitions = partitionsInRange(await engine.listPartitions({
|
|
154
|
+
ctx,
|
|
155
|
+
table: "countries",
|
|
156
|
+
...searchType !== void 0 ? { searchType } : {}
|
|
157
|
+
}), cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
158
|
+
if (partitions.length === 0) return [];
|
|
159
|
+
return (await engine.runSQL({
|
|
160
|
+
ctx,
|
|
161
|
+
table: "countries",
|
|
162
|
+
fileSets: { FILES: {
|
|
163
|
+
table: "countries",
|
|
164
|
+
partitions
|
|
165
|
+
} },
|
|
166
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
167
|
+
sql: `
|
|
168
|
+
SELECT
|
|
169
|
+
country,
|
|
170
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
171
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
172
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
173
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
174
|
+
WHERE date >= '${cutoff}'
|
|
175
|
+
GROUP BY country
|
|
176
|
+
ORDER BY clicks DESC
|
|
177
|
+
LIMIT 250
|
|
178
|
+
`
|
|
179
|
+
})).rows.map((r) => ({
|
|
180
|
+
country: r.country,
|
|
181
|
+
clicks: Number(r.clicks),
|
|
182
|
+
impressions: Number(r.impressions),
|
|
183
|
+
sum_position: Number(r.sum_position)
|
|
184
|
+
}));
|
|
185
|
+
}
|
|
186
|
+
};
|
|
187
|
+
const topKeywords28dRollup = {
|
|
188
|
+
id: "top_keywords_28d",
|
|
189
|
+
windowDays: 28,
|
|
190
|
+
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
191
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
192
|
+
const partitions = partitionsInRange(await engine.listPartitions({
|
|
193
|
+
ctx,
|
|
194
|
+
table: "queries",
|
|
195
|
+
...searchType !== void 0 ? { searchType } : {}
|
|
196
|
+
}), cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
197
|
+
if (partitions.length === 0) return [];
|
|
198
|
+
return (await engine.runSQL({
|
|
199
|
+
ctx,
|
|
200
|
+
table: "queries",
|
|
201
|
+
fileSets: { FILES: {
|
|
202
|
+
table: "queries",
|
|
203
|
+
partitions
|
|
204
|
+
} },
|
|
205
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
206
|
+
sql: `
|
|
207
|
+
SELECT
|
|
208
|
+
query,
|
|
209
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
210
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
211
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
212
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
213
|
+
WHERE date >= '${cutoff}'
|
|
214
|
+
GROUP BY query
|
|
215
|
+
ORDER BY clicks DESC
|
|
216
|
+
LIMIT 1000
|
|
217
|
+
`
|
|
218
|
+
})).rows.map((r) => ({
|
|
219
|
+
query: r.query,
|
|
220
|
+
clicks: Number(r.clicks),
|
|
221
|
+
impressions: Number(r.impressions),
|
|
222
|
+
sum_position: Number(r.sum_position)
|
|
223
|
+
}));
|
|
224
|
+
}
|
|
225
|
+
};
|
|
226
|
+
const topKeywords28dParquetRollup = {
|
|
227
|
+
id: "top_keywords_28d_parquet",
|
|
228
|
+
windowDays: 28,
|
|
229
|
+
format: "parquet",
|
|
230
|
+
parquetColumns: [
|
|
231
|
+
{
|
|
232
|
+
name: "query",
|
|
233
|
+
type: "VARCHAR",
|
|
234
|
+
nullable: false
|
|
235
|
+
},
|
|
236
|
+
{
|
|
237
|
+
name: "clicks",
|
|
238
|
+
type: "BIGINT",
|
|
239
|
+
nullable: false
|
|
240
|
+
},
|
|
241
|
+
{
|
|
242
|
+
name: "impressions",
|
|
243
|
+
type: "BIGINT",
|
|
244
|
+
nullable: false
|
|
245
|
+
},
|
|
246
|
+
{
|
|
247
|
+
name: "sum_position",
|
|
248
|
+
type: "DOUBLE",
|
|
249
|
+
nullable: false
|
|
250
|
+
}
|
|
251
|
+
],
|
|
252
|
+
parquetSortKey: ["clicks"],
|
|
253
|
+
async build({ engine, ctx, windowAnchorMs, searchType }) {
|
|
254
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 28);
|
|
255
|
+
const partitions = partitionsInRange(await engine.listPartitions({
|
|
256
|
+
ctx,
|
|
257
|
+
table: "queries",
|
|
258
|
+
...searchType !== void 0 ? { searchType } : {}
|
|
259
|
+
}), cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
260
|
+
if (partitions.length === 0) return [];
|
|
261
|
+
return (await engine.runSQL({
|
|
262
|
+
ctx,
|
|
263
|
+
table: "queries",
|
|
264
|
+
fileSets: { FILES: {
|
|
265
|
+
table: "queries",
|
|
266
|
+
partitions
|
|
267
|
+
} },
|
|
268
|
+
...searchType !== void 0 ? { searchType } : {},
|
|
269
|
+
sql: `
|
|
270
|
+
SELECT
|
|
271
|
+
query,
|
|
272
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
273
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
274
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
275
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
276
|
+
WHERE date >= '${cutoff}'
|
|
277
|
+
GROUP BY query
|
|
278
|
+
ORDER BY clicks DESC
|
|
279
|
+
LIMIT 1000
|
|
280
|
+
`
|
|
281
|
+
})).rows.map((r) => ({
|
|
282
|
+
query: String(r.query),
|
|
283
|
+
clicks: BigInt(r.clicks),
|
|
284
|
+
impressions: BigInt(r.impressions),
|
|
285
|
+
sum_position: Number(r.sum_position)
|
|
286
|
+
}));
|
|
287
|
+
}
|
|
288
|
+
};
|
|
289
|
+
export { dailyTotalsRollup, topCountries28dRollup, topKeywords28dParquetRollup, topKeywords28dRollup, topPages28dRollup, weeklyTotalsRollup };
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import { FileSetRef, Row as Row$1, TableName as TableName$1 } from "../storage.mjs";
|
|
2
|
+
import "../contracts.mjs";
|
|
3
|
+
import { RollupEngine } from "./core.mjs";
|
|
4
|
+
import { TenantCtx } from "@gscdump/contracts";
|
|
5
|
+
import { SearchType } from "gscdump/query";
|
|
6
|
+
/**
|
|
7
|
+
* Per-window budget, measured in *parquet* bytes (manifest `bytes`), used by
|
|
8
|
+
* `planRollupWindows` to chunk a full-history scan.
|
|
9
|
+
*
|
|
10
|
+
* The executor decodes a window's parquet and ships it as an Arrow IPC stream
|
|
11
|
+
* over the service binding; that IPC is hard-guarded at 28MiB
|
|
12
|
+
* (`IPC_PLACEHOLDER_BUDGET` in @gscdump/cloudflare). Parquet is compressed and
|
|
13
|
+
* the IPC stream is not, so a window inflates on the wire — keep this
|
|
14
|
+
* conservatively below the guard. Re-measure the parquet→IPC ratio against
|
|
15
|
+
* production and raise if headroom allows.
|
|
16
|
+
*/
|
|
17
|
+
declare const WINDOW_BYTE_BUDGET: number;
|
|
18
|
+
/**
|
|
19
|
+
* Per-page OUTPUT row cap for key-paginated rollups (`runWindowed({ paginate })`
|
|
20
|
+
* and `runPagedQuery`). `planRollupWindows` bounds the *input* parquet bytes a
|
|
21
|
+
* window scans, which is a fine proxy for output size on fact aggregations whose
|
|
22
|
+
* grain matches the input (one output row per input date). It is NOT a proxy for
|
|
23
|
+
* aggregations that COLLAPSE to a smaller-cardinality grain whose row count is
|
|
24
|
+
* driven by a high-cardinality GROUP key — `(query_canonical × date)` and
|
|
25
|
+
* `(query_canonical)` — where output rows scale with distinct canonicals, not
|
|
26
|
+
* input bytes. For those, each `runSQL` result (shipped as an Arrow IPC stream
|
|
27
|
+
* over the Workers service-binding RPC; 28MiB guard in `@gscdump/cloudflare`,
|
|
28
|
+
* duckdb-worker `assertResultBudget` at 24MiB / 100k rows) must be bounded by
|
|
29
|
+
* paging the OUTPUT, independent of how the input is windowed.
|
|
30
|
+
*
|
|
31
|
+
* Narrow rows — `(canonical, date, 3 metrics)` — page at 50k (≈16MiB at the
|
|
32
|
+
* worker's `cols×64` heuristic, well under both guards). WIDE rows carry a
|
|
33
|
+
* `GROUP_CONCAT` variants string (up to ~10 variants × ~60 chars) the heuristic
|
|
34
|
+
* under-counts, so they page smaller to keep the real IPC payload bounded.
|
|
35
|
+
*/
|
|
36
|
+
declare const ROLLUP_PAGE_ROWS = 50000;
|
|
37
|
+
declare const ROLLUP_PAGE_ROWS_WIDE = 20000;
|
|
38
|
+
declare const ROLLUP_PAGE_ROWS_DAILY = 70000;
|
|
39
|
+
/**
|
|
40
|
+
* Plan byte-bounded windows over a partition set. Each window names the
|
|
41
|
+
* partitions whose span intersects it; a coarse tier file can land in two
|
|
42
|
+
* windows, so every windowed SQL MUST also date-filter to the window bounds.
|
|
43
|
+
*/
|
|
44
|
+
declare function planRollupWindows(parts: Array<{
|
|
45
|
+
partition: string;
|
|
46
|
+
bytes: number;
|
|
47
|
+
}>, clampRange?: {
|
|
48
|
+
start: string;
|
|
49
|
+
end: string;
|
|
50
|
+
}, maxWindowDays?: number): Array<{
|
|
51
|
+
start: string;
|
|
52
|
+
end: string;
|
|
53
|
+
partitions: string[];
|
|
54
|
+
}>;
|
|
55
|
+
/**
|
|
56
|
+
* Run a full-history aggregation in byte-bounded windows and concat the rows.
|
|
57
|
+
* Each window's SQL MUST date-filter to `[w.start, w.end]` (see `sqlFor`) so a
|
|
58
|
+
* tier file spanning a window boundary doesn't double-count calendar dates.
|
|
59
|
+
*
|
|
60
|
+
* `paginate` additionally pages each window's OUTPUT (see `runPagedQuery`) so a
|
|
61
|
+
* window whose GROUP cardinality is high — `(query_canonical × date)` on a large
|
|
62
|
+
* site — can't ship an oversized result even though its input bytes fit a window.
|
|
63
|
+
* Date-windowing bounds the per-query scan; output paging bounds the IPC payload.
|
|
64
|
+
* The two are orthogonal and compose. When `paginate` is set, `sqlFor` MUST emit
|
|
65
|
+
* no trailing `ORDER BY`/`LIMIT` and `paginate.orderBy` MUST be a total order.
|
|
66
|
+
*/
|
|
67
|
+
declare function runWindowed(opts: {
|
|
68
|
+
engine: RollupEngine;
|
|
69
|
+
ctx: TenantCtx;
|
|
70
|
+
table: TableName$1;
|
|
71
|
+
searchType?: SearchType;
|
|
72
|
+
sqlFor: (w: {
|
|
73
|
+
start: string;
|
|
74
|
+
end: string;
|
|
75
|
+
}) => string;
|
|
76
|
+
/**
|
|
77
|
+
* Extra named file sets merged into every window's `runSQL` (alongside the
|
|
78
|
+
* windowed `FILES`). Use to JOIN a non-windowed sidecar (e.g. the query
|
|
79
|
+
* dimension parquet via `{ QUERY_DIM: { keys: [...] } }`) inside `sqlFor`.
|
|
80
|
+
*/
|
|
81
|
+
extraFileSets?: Record<string, FileSetRef>;
|
|
82
|
+
/** Page each window's output by a total-order key. See `runPagedQuery`. */
|
|
83
|
+
paginate?: {
|
|
84
|
+
orderBy: string;
|
|
85
|
+
pageRows: number;
|
|
86
|
+
};
|
|
87
|
+
/** Cap each window's day span (output-cardinality bound). See `planRollupWindows`. */
|
|
88
|
+
maxWindowDays?: number;
|
|
89
|
+
}): Promise<Row$1[]>;
|
|
90
|
+
export { ROLLUP_PAGE_ROWS, ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, WINDOW_BYTE_BUDGET, planRollupWindows, runWindowed };
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
import { MS_PER_DAY } from "gscdump/dates";
|
|
2
|
+
const WINDOW_BYTE_BUDGET = 10 * 1024 * 1024;
|
|
3
|
+
const ROLLUP_PAGE_ROWS = 5e4;
|
|
4
|
+
const ROLLUP_PAGE_ROWS_WIDE = 2e4;
|
|
5
|
+
const ROLLUP_PAGE_ROWS_DAILY = 7e4;
|
|
6
|
+
const DAY_RE = /^daily\/(\d{4})-(\d{2})-(\d{2})$/;
|
|
7
|
+
const WEEK_RE = /^weekly\/(\d{4})-(\d{2})-(\d{2})$/;
|
|
8
|
+
const MONTH_RE = /^monthly\/(\d{4})-(\d{2})$/;
|
|
9
|
+
const QUARTER_RE = /^quarterly\/(\d{4})-Q([1-4])$/;
|
|
10
|
+
function isoDate(ms) {
|
|
11
|
+
const d = new Date(ms);
|
|
12
|
+
return `${d.getUTCFullYear()}-${String(d.getUTCMonth() + 1).padStart(2, "0")}-${String(d.getUTCDate()).padStart(2, "0")}`;
|
|
13
|
+
}
|
|
14
|
+
function partitionDaySpan(partition) {
|
|
15
|
+
const day = DAY_RE.exec(partition);
|
|
16
|
+
if (day) {
|
|
17
|
+
const ms = Date.UTC(Number(day[1]), Number(day[2]) - 1, Number(day[3]));
|
|
18
|
+
return {
|
|
19
|
+
startMs: ms,
|
|
20
|
+
endMs: ms
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
const week = WEEK_RE.exec(partition);
|
|
24
|
+
if (week) {
|
|
25
|
+
const ms = Date.UTC(Number(week[1]), Number(week[2]) - 1, Number(week[3]));
|
|
26
|
+
return {
|
|
27
|
+
startMs: ms,
|
|
28
|
+
endMs: ms + 6 * MS_PER_DAY
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
const month = MONTH_RE.exec(partition);
|
|
32
|
+
if (month) {
|
|
33
|
+
const y = Number(month[1]);
|
|
34
|
+
const m = Number(month[2]) - 1;
|
|
35
|
+
return {
|
|
36
|
+
startMs: Date.UTC(y, m, 1),
|
|
37
|
+
endMs: Date.UTC(y, m + 1, 1) - MS_PER_DAY
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
const quarter = QUARTER_RE.exec(partition);
|
|
41
|
+
if (quarter) {
|
|
42
|
+
const y = Number(quarter[1]);
|
|
43
|
+
const startMonth = (Number(quarter[2]) - 1) * 3;
|
|
44
|
+
return {
|
|
45
|
+
startMs: Date.UTC(y, startMonth, 1),
|
|
46
|
+
endMs: Date.UTC(y, startMonth + 3, 1) - MS_PER_DAY
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
function clamp(n, lo, hi) {
|
|
52
|
+
return Math.max(lo, Math.min(hi, n));
|
|
53
|
+
}
|
|
54
|
+
function planRollupWindows(parts, clampRange, maxWindowDays) {
|
|
55
|
+
const clampStartMs = clampRange ? Date.parse(`${clampRange.start}T00:00:00Z`) : void 0;
|
|
56
|
+
const clampEndMs = clampRange ? Date.parse(`${clampRange.end}T00:00:00Z`) : void 0;
|
|
57
|
+
const spans = [];
|
|
58
|
+
for (const p of parts) {
|
|
59
|
+
const span = partitionDaySpan(p.partition);
|
|
60
|
+
if (!span) continue;
|
|
61
|
+
if (clampStartMs !== void 0 && clampEndMs !== void 0) {
|
|
62
|
+
if (span.endMs < clampStartMs || span.startMs > clampEndMs) continue;
|
|
63
|
+
}
|
|
64
|
+
spans.push({
|
|
65
|
+
partition: p.partition,
|
|
66
|
+
bytes: p.bytes,
|
|
67
|
+
startMs: span.startMs,
|
|
68
|
+
endMs: span.endMs
|
|
69
|
+
});
|
|
70
|
+
}
|
|
71
|
+
if (spans.length === 0) return [];
|
|
72
|
+
let rangeStartMs = Number.POSITIVE_INFINITY;
|
|
73
|
+
let rangeEndMs = Number.NEGATIVE_INFINITY;
|
|
74
|
+
let totalBytes = 0;
|
|
75
|
+
for (const span of spans) {
|
|
76
|
+
if (span.startMs < rangeStartMs) rangeStartMs = span.startMs;
|
|
77
|
+
if (span.endMs > rangeEndMs) rangeEndMs = span.endMs;
|
|
78
|
+
totalBytes += span.bytes;
|
|
79
|
+
}
|
|
80
|
+
if (clampStartMs !== void 0) rangeStartMs = Math.max(rangeStartMs, clampStartMs);
|
|
81
|
+
if (clampEndMs !== void 0) rangeEndMs = Math.min(rangeEndMs, clampEndMs);
|
|
82
|
+
const spanDays = Math.floor((rangeEndMs - rangeStartMs) / MS_PER_DAY) + 1;
|
|
83
|
+
const bytesPerDay = Math.max(1, totalBytes / spanDays);
|
|
84
|
+
const byteWindowDays = clamp(Math.floor(WINDOW_BYTE_BUDGET / bytesPerDay), 7, 400);
|
|
85
|
+
const windowDays = maxWindowDays != null ? Math.max(1, Math.min(byteWindowDays, maxWindowDays)) : byteWindowDays;
|
|
86
|
+
if (!Number.isFinite(windowDays) || !Number.isFinite(rangeStartMs) || !Number.isFinite(rangeEndMs) || rangeEndMs < rangeStartMs) return [];
|
|
87
|
+
const windowWidthMs = windowDays * MS_PER_DAY;
|
|
88
|
+
const windowCount = Math.floor((rangeEndMs - rangeStartMs) / windowWidthMs) + 1;
|
|
89
|
+
const partitionsByWindow = Array.from({ length: windowCount }, () => []);
|
|
90
|
+
for (const span of spans) {
|
|
91
|
+
const overlapStartMs = Math.max(span.startMs, rangeStartMs);
|
|
92
|
+
const overlapEndMs = Math.min(span.endMs, rangeEndMs);
|
|
93
|
+
if (overlapStartMs > overlapEndMs) continue;
|
|
94
|
+
const firstWindow = Math.max(0, Math.ceil((overlapStartMs - rangeStartMs - (windowDays - 1) * MS_PER_DAY) / windowWidthMs));
|
|
95
|
+
const lastWindow = Math.min(windowCount - 1, Math.floor((overlapEndMs - rangeStartMs) / windowWidthMs));
|
|
96
|
+
for (let index = firstWindow; index <= lastWindow; index++) partitionsByWindow[index].push(span.partition);
|
|
97
|
+
}
|
|
98
|
+
const windows = [];
|
|
99
|
+
for (let index = 0; index < windowCount; index++) {
|
|
100
|
+
const cursorMs = rangeStartMs + index * windowWidthMs;
|
|
101
|
+
const windowEndMs = Math.min(cursorMs + (windowDays - 1) * MS_PER_DAY, rangeEndMs);
|
|
102
|
+
const partitions = partitionsByWindow[index];
|
|
103
|
+
if (partitions.length > 0) windows.push({
|
|
104
|
+
start: isoDate(cursorMs),
|
|
105
|
+
end: isoDate(windowEndMs),
|
|
106
|
+
partitions
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
return windows;
|
|
110
|
+
}
|
|
111
|
+
function partitionsInRange(parts, start, end) {
|
|
112
|
+
const startMs = Date.parse(`${start}T00:00:00Z`);
|
|
113
|
+
const endMs = Date.parse(`${end}T00:00:00Z`);
|
|
114
|
+
const out = [];
|
|
115
|
+
for (const p of parts) {
|
|
116
|
+
const span = partitionDaySpan(p.partition);
|
|
117
|
+
if (!span) continue;
|
|
118
|
+
if (span.endMs >= startMs && span.startMs <= endMs) out.push(p.partition);
|
|
119
|
+
}
|
|
120
|
+
return out;
|
|
121
|
+
}
|
|
122
|
+
async function runPagedQuery(opts) {
|
|
123
|
+
const out = [];
|
|
124
|
+
for (let offset = 0;; offset += opts.pageRows) {
|
|
125
|
+
const result = await opts.engine.runSQL({
|
|
126
|
+
ctx: opts.ctx,
|
|
127
|
+
table: opts.table,
|
|
128
|
+
fileSets: opts.fileSets,
|
|
129
|
+
sql: `${opts.coreSql}\nORDER BY ${opts.orderBy}\nLIMIT ${opts.pageRows} OFFSET ${offset}`,
|
|
130
|
+
...opts.searchType !== void 0 ? { searchType: opts.searchType } : {}
|
|
131
|
+
});
|
|
132
|
+
out.push(...result.rows);
|
|
133
|
+
if (result.rows.length < opts.pageRows) break;
|
|
134
|
+
}
|
|
135
|
+
return out;
|
|
136
|
+
}
|
|
137
|
+
async function runWindowed(opts) {
|
|
138
|
+
const windows = planRollupWindows(await opts.engine.listPartitions({
|
|
139
|
+
ctx: opts.ctx,
|
|
140
|
+
table: opts.table,
|
|
141
|
+
...opts.searchType !== void 0 ? { searchType: opts.searchType } : {}
|
|
142
|
+
}), void 0, opts.maxWindowDays);
|
|
143
|
+
const rows = [];
|
|
144
|
+
for (const w of windows) {
|
|
145
|
+
const fileSets = {
|
|
146
|
+
FILES: {
|
|
147
|
+
table: opts.table,
|
|
148
|
+
partitions: w.partitions
|
|
149
|
+
},
|
|
150
|
+
...opts.extraFileSets
|
|
151
|
+
};
|
|
152
|
+
if (opts.paginate) {
|
|
153
|
+
rows.push(...await runPagedQuery({
|
|
154
|
+
engine: opts.engine,
|
|
155
|
+
ctx: opts.ctx,
|
|
156
|
+
table: opts.table,
|
|
157
|
+
...opts.searchType !== void 0 ? { searchType: opts.searchType } : {},
|
|
158
|
+
fileSets,
|
|
159
|
+
coreSql: opts.sqlFor(w),
|
|
160
|
+
orderBy: opts.paginate.orderBy,
|
|
161
|
+
pageRows: opts.paginate.pageRows
|
|
162
|
+
}));
|
|
163
|
+
continue;
|
|
164
|
+
}
|
|
165
|
+
const result = await opts.engine.runSQL({
|
|
166
|
+
ctx: opts.ctx,
|
|
167
|
+
table: opts.table,
|
|
168
|
+
fileSets,
|
|
169
|
+
sql: opts.sqlFor(w),
|
|
170
|
+
...opts.searchType !== void 0 ? { searchType: opts.searchType } : {}
|
|
171
|
+
});
|
|
172
|
+
rows.push(...result.rows);
|
|
173
|
+
}
|
|
174
|
+
return rows;
|
|
175
|
+
}
|
|
176
|
+
export { ROLLUP_PAGE_ROWS, ROLLUP_PAGE_ROWS_DAILY, ROLLUP_PAGE_ROWS_WIDE, WINDOW_BYTE_BUDGET, partitionsInRange, planRollupWindows, runWindowed };
|