@gscdump/engine 1.4.11 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/entities/empty-types.d.mts +22 -0
  2. package/dist/entities/empty-types.mjs +58 -0
  3. package/dist/entities/indexing-metadata.d.mts +26 -0
  4. package/dist/entities/indexing-metadata.mjs +31 -0
  5. package/dist/entities/inspection.d.mts +240 -0
  6. package/dist/entities/inspection.mjs +443 -0
  7. package/dist/entities/io.mjs +16 -0
  8. package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
  9. package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
  10. package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
  11. package/dist/entities/sitemap-shared.d.mts +199 -0
  12. package/dist/entities/sitemap-shared.mjs +243 -0
  13. package/dist/entities/sitemap-write.d.mts +3 -0
  14. package/dist/entities/sitemap-write.mjs +528 -0
  15. package/dist/entities/sitemap.d.mts +3 -0
  16. package/dist/entities/sitemap.mjs +91 -0
  17. package/dist/entities.d.mts +9 -481
  18. package/dist/entities.mjs +7 -1380
  19. package/dist/rollups/canonical.d.mts +71 -0
  20. package/dist/rollups/canonical.mjs +335 -0
  21. package/dist/rollups/core.d.mts +201 -0
  22. package/dist/rollups/core.mjs +116 -0
  23. package/dist/rollups/dates.mjs +11 -0
  24. package/dist/rollups/defaults.d.mts +11 -0
  25. package/dist/rollups/defaults.mjs +17 -0
  26. package/dist/rollups/hourly.d.mts +38 -0
  27. package/dist/rollups/hourly.mjs +38 -0
  28. package/dist/rollups/indexing.d.mts +46 -0
  29. package/dist/rollups/indexing.mjs +357 -0
  30. package/dist/rollups/traffic.d.mts +38 -0
  31. package/dist/rollups/traffic.mjs +289 -0
  32. package/dist/rollups/windows.d.mts +90 -0
  33. package/dist/rollups/windows.mjs +176 -0
  34. package/dist/rollups.d.mts +8 -471
  35. package/dist/rollups.mjs +7 -1316
  36. package/package.json +4 -4
  37. /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
@@ -0,0 +1,116 @@
1
+ import { engineErrors } from "../errors.mjs";
2
+ import { encodeRowsToParquetFlex } from "../adapters/hyparquet.mjs";
3
+ import { isoDateToUtcMs } from "./dates.mjs";
4
+ import { encodeJsonBigintSafe } from "@gscdump/lakehouse/bigint";
5
+ function rollupPrefix(ctx, searchType) {
6
+ const base = ctx.siteId ? `u_${ctx.userId}/${ctx.siteId}/rollups` : `u_${ctx.userId}/rollups`;
7
+ return searchType !== void 0 && searchType !== "web" ? `${base}/${searchType}` : base;
8
+ }
9
+ function rollupKey(ctx, id, builtAt, searchType) {
10
+ return `${rollupPrefix(ctx, searchType)}/${id}__v${builtAt}.json`;
11
+ }
12
+ function rollupParquetKey(ctx, id, builtAt, searchType) {
13
+ return `${rollupPrefix(ctx, searchType)}/${id}__v${builtAt}.parquet`;
14
+ }
15
+ const ROLLUP_FILE_RE = /^(?<id>[a-z0-9_]+)__v(?<ts>\d+)\.json$/;
16
+ async function readLatestRollup(bucket, ctx, id, searchType) {
17
+ const prefix = `${rollupPrefix(ctx, searchType)}/`;
18
+ let newest = null;
19
+ let cursor;
20
+ do {
21
+ const listing = await bucket.list({
22
+ prefix,
23
+ cursor
24
+ });
25
+ for (const obj of listing.objects) {
26
+ const m = ROLLUP_FILE_RE.exec(obj.key.slice(prefix.length));
27
+ if (!m?.groups || m.groups.id !== id) continue;
28
+ const ts = Number(m.groups.ts);
29
+ if (!newest || ts > newest.ts) newest = {
30
+ ts,
31
+ key: obj.key
32
+ };
33
+ }
34
+ cursor = listing.truncated ? listing.cursor : void 0;
35
+ } while (cursor !== void 0);
36
+ if (!newest) return null;
37
+ const obj = await bucket.get(newest.key);
38
+ if (!obj) return null;
39
+ return JSON.parse(await obj.text());
40
+ }
41
+ async function rebuildRollups(opts) {
42
+ const now = opts.now ?? (() => Date.now());
43
+ const dataEndMs = opts.dataEndDate !== void 0 ? isoDateToUtcMs(opts.dataEndDate) : null;
44
+ const results = [];
45
+ for (const def of opts.defs) {
46
+ const builtAt = now();
47
+ const windowAnchorMs = dataEndMs ?? builtAt;
48
+ const defSearchType = def.sliceOrthogonal === true ? void 0 : opts.searchType;
49
+ try {
50
+ const payload = await def.build({
51
+ engine: opts.engine,
52
+ ctx: opts.ctx,
53
+ dataSource: opts.dataSource,
54
+ windowAnchorMs,
55
+ ...defSearchType !== void 0 ? { searchType: defSearchType } : {}
56
+ });
57
+ if (def.format === "parquet") {
58
+ if (!def.parquetColumns || def.parquetColumns.length === 0) throw new Error(`rollup '${def.id}' declared format='parquet' without parquetColumns`);
59
+ const rows = payload ?? [];
60
+ const parquetBytes = encodeRowsToParquetFlex(rows, {
61
+ columns: def.parquetColumns,
62
+ sortKey: def.parquetSortKey
63
+ });
64
+ const parquetKey = rollupParquetKey(opts.ctx, def.id, builtAt, defSearchType);
65
+ await opts.dataSource.write(parquetKey, parquetBytes);
66
+ const pointer = {
67
+ parquetKey,
68
+ rowCount: rows.length
69
+ };
70
+ const envelopeBytes = encodeJsonBigintSafe({
71
+ version: 1,
72
+ id: def.id,
73
+ builtAt,
74
+ windowDays: def.windowDays,
75
+ payload: pointer
76
+ });
77
+ const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
78
+ await opts.dataSource.write(key, envelopeBytes);
79
+ results.push({
80
+ id: def.id,
81
+ objectKey: key,
82
+ parquetKey,
83
+ bytes: envelopeBytes.byteLength,
84
+ parquetBytes: parquetBytes.byteLength,
85
+ builtAt
86
+ });
87
+ continue;
88
+ }
89
+ const bytes = encodeJsonBigintSafe({
90
+ version: 1,
91
+ id: def.id,
92
+ builtAt,
93
+ windowDays: def.windowDays,
94
+ payload
95
+ });
96
+ const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
97
+ await opts.dataSource.write(key, bytes);
98
+ results.push({
99
+ id: def.id,
100
+ objectKey: key,
101
+ bytes: bytes.byteLength,
102
+ builtAt
103
+ });
104
+ } catch (err) {
105
+ results.push({
106
+ id: def.id,
107
+ objectKey: "",
108
+ bytes: 0,
109
+ builtAt,
110
+ error: engineErrors.rollupBuildFailed(def.id, err)
111
+ });
112
+ }
113
+ }
114
+ return results;
115
+ }
116
+ export { readLatestRollup, rebuildRollups, rollupKey, rollupParquetKey };
@@ -0,0 +1,11 @@
1
+ import { MS_PER_DAY } from "gscdump/dates";
2
+ function isoDateToUtcMs(iso) {
3
+ const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(iso);
4
+ if (!m) throw new Error(`dataEndDate must be ISO YYYY-MM-DD, got: ${iso}`);
5
+ return Date.UTC(Number(m[1]), Number(m[2]) - 1, Number(m[3]));
6
+ }
7
+ function utcDateMinusDays(at, days) {
8
+ const d = new Date(at - days * MS_PER_DAY);
9
+ return `${d.getUTCFullYear()}-${String(d.getUTCMonth() + 1).padStart(2, "0")}-${String(d.getUTCDate()).padStart(2, "0")}`;
10
+ }
11
+ export { isoDateToUtcMs, utcDateMinusDays };
@@ -0,0 +1,11 @@
1
+ import { RollupDef } from "./core.mjs";
2
+ declare const DEFAULT_ROLLUPS: readonly RollupDef[];
3
+ /**
4
+ * Canonical-primary rollups (ADR-0017 / ADR-0018). Opt-in — kept out of
5
+ * `DEFAULT_ROLLUPS` because they only pay off once the consumer queries by
6
+ * `queryCanonical` and wires the read seams (`resolveExtra` /
7
+ * `canonicalSource`). Hosts opt in by concatenating these onto their def list
8
+ * (CLI: `gscdump rollups --with-canonical`).
9
+ */
10
+ declare const CANONICAL_ROLLUPS: readonly RollupDef[];
11
+ export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS };
@@ -0,0 +1,17 @@
1
+ import { queryCanonicalDailyRollup, queryCanonicalVariantsRollup } from "./canonical.mjs";
2
+ import { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup } from "./indexing.mjs";
3
+ import { dailyTotalsRollup, topCountries28dRollup, topKeywords28dRollup, topPages28dRollup, weeklyTotalsRollup } from "./traffic.mjs";
4
+ const DEFAULT_ROLLUPS = [
5
+ dailyTotalsRollup,
6
+ weeklyTotalsRollup,
7
+ topPages28dRollup,
8
+ topKeywords28dRollup,
9
+ topCountries28dRollup,
10
+ indexingMetadataRollup,
11
+ indexingHealthRollup,
12
+ indexPercentRollup,
13
+ sitemapHealthRollup,
14
+ sitemapChanges28dRollup
15
+ ];
16
+ const CANONICAL_ROLLUPS = [queryCanonicalVariantsRollup, queryCanonicalDailyRollup];
17
+ export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS };
@@ -0,0 +1,38 @@
1
+ import { Row as Row$1 } from "../storage.mjs";
2
+ import "../contracts.mjs";
3
+ import { RollupEngine } from "./core.mjs";
4
+ import { TenantCtx } from "@gscdump/contracts";
5
+ import { SearchType } from "gscdump/query";
6
+ /**
7
+ * Aggregate one day's `hourly_pages` partition into the daily `pages` shape
8
+ * and write it to the daily Discover partition. After this runs for date D,
9
+ * the daily query path serves D from `pages/.../daily/D` and the `hourly/D`
10
+ * partition becomes read-only / GC-only.
11
+ *
12
+ * `(position - 1)` weighting matches the storage convention encoded by
13
+ * `toSumPosition`: `sum_position = SUM((position - 1) * impressions)`, so a
14
+ * downstream `SUM(sum_position) / SUM(impressions) + 1` recovers the mean.
15
+ *
16
+ * searchType-scoped: only call with `searchType: 'discover'`. The hourly
17
+ * partition lives under `hourly_pages` and the output lands under `pages` so
18
+ * existing dashboard queries (which read `pages`) see the rolled-up day
19
+ * transparently.
20
+ */
21
+ interface RebuildDailyFromHourlyOptions {
22
+ engine: RollupEngine & {
23
+ writeDay: (scope: TenantCtx & {
24
+ table: TableTypeName;
25
+ date: string;
26
+ searchType?: SearchType;
27
+ }, rows: Row$1[]) => Promise<void>;
28
+ };
29
+ ctx: TenantCtx;
30
+ /** PT calendar day to roll up. */
31
+ date: string;
32
+ searchType: 'discover';
33
+ }
34
+ type TableTypeName = import('@gscdump/contracts').TableName;
35
+ declare function rebuildDailyFromHourly(opts: RebuildDailyFromHourlyOptions): Promise<{
36
+ rowsWritten: number;
37
+ }>;
38
+ export { RebuildDailyFromHourlyOptions, rebuildDailyFromHourly };
@@ -0,0 +1,38 @@
1
+ async function rebuildDailyFromHourly(opts) {
2
+ const { engine, ctx, date, searchType } = opts;
3
+ const rows = (await engine.runSQL({
4
+ ctx,
5
+ table: "hourly_pages",
6
+ fileSets: { FILES: {
7
+ table: "hourly_pages",
8
+ partitions: [`hourly/${date}`]
9
+ } },
10
+ searchType,
11
+ sql: `
12
+ SELECT
13
+ url,
14
+ DATE '${date}' AS date,
15
+ SUM(clicks)::BIGINT AS clicks,
16
+ SUM(impressions)::BIGINT AS impressions,
17
+ SUM(sum_position)::DOUBLE AS sum_position
18
+ FROM read_parquet({{FILES}}, union_by_name = true)
19
+ WHERE date = '${date}'
20
+ GROUP BY url
21
+ `
22
+ })).rows.map((r) => ({
23
+ url: r.url,
24
+ date,
25
+ clicks: Number(r.clicks),
26
+ impressions: Number(r.impressions),
27
+ sum_position: Number(r.sum_position)
28
+ }));
29
+ await engine.writeDay({
30
+ userId: ctx.userId,
31
+ siteId: ctx.siteId,
32
+ table: "pages",
33
+ date,
34
+ searchType
35
+ }, rows);
36
+ return { rowsWritten: rows.length };
37
+ }
38
+ export { rebuildDailyFromHourly };
@@ -0,0 +1,46 @@
1
+ import { RollupDef } from "./core.mjs";
2
+ /**
3
+ * Aggregates the per-URL Indexing API metadata entity store (populated by
4
+ * `gscdump entities indexing snapshot`) into daily counts of `URL_UPDATED`
5
+ * and `URL_REMOVED` notifications. Covers the third entity-snapshot shape
6
+ * without needing its own parquet family — publish events are sparse and
7
+ * aggregate cleanly into a small JSON rollup.
8
+ *
9
+ * Safe no-op when the entity store is empty: returns `{ totals: {...}, days: [] }`
10
+ * so downstream readers don't have to special-case first-run sites.
11
+ */
12
+ declare const indexingMetadataRollup: RollupDef;
13
+ /**
14
+ * Indexing-API health by day: per `inspectedAt` date, counts of indexed,
15
+ * soft-404, redirect, not-found, mobile passes, rich-results passes, and
16
+ * canonical mismatches. Sourced from the inspections parquet sidecar
17
+ * (`InspectionStore.parquetUri`), which holds the latest record per URL.
18
+ *
19
+ * Empty-payload no-op when the sidecar URI is unavailable (in-memory
20
+ * `DataSource`, or before `materialize` has run).
21
+ */
22
+ declare const indexingHealthRollup: RollupDef;
23
+ /**
24
+ * Per-day index-percent: ratio of (sitemap URLs that received GSC clicks on
25
+ * that date) / (total live sitemap URLs). Uses a DuckDB JOIN between the
26
+ * sitemap urls parquet (`SitemapStore.urlsParquetUri`) and the `pages` fact
27
+ * parquet. Total denominator is the count of live URLs in the urls index;
28
+ * numerator is per-day distinct loc count where pages.clicks > 0.
29
+ */
30
+ declare const indexPercentRollup: RollupDef;
31
+ /**
32
+ * Sitemap-health per-day series materialized from the sitemap-store JSON
33
+ * index. Each `SitemapRecord` carries `urlCount`, `errors`, `warnings`,
34
+ * `contentHash`, and `lastDownloaded`. We bucket records by the day of their
35
+ * `capturedAt` (or `lastDownloaded` fallback) and emit per-day aggregates plus
36
+ * a snapshot of per-feed stats at the most recent capture.
37
+ */
38
+ declare const sitemapHealthRollup: RollupDef;
39
+ /**
40
+ * Trailing-28-day sitemap URL changes: per-day per-feedpath {added, removed}
41
+ * counts plus rolling top-200 added and removed URLs. Streams from
42
+ * retained `SitemapReadStore.loadEvents()` history. State compaction cannot
43
+ * erase analytics input; memory scales independently of total site state.
44
+ */
45
+ declare const sitemapChanges28dRollup: RollupDef;
46
+ export { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup };
@@ -0,0 +1,357 @@
1
+ import "../layout.mjs";
2
+ import { inspectionParquetKey, sitemapUrlsIndexPrefix, sitemapUrlsPrefix, sitemapUrlsProjectionManifestKey } from "../entity-keys.mjs";
3
+ import { readOptional } from "../adapters/read-optional.mjs";
4
+ import { createIndexingMetadataStore } from "../entities/indexing-metadata.mjs";
5
+ import { decodeSitemapProjectionManifest, selectSitemapProjectionFiles } from "../entities/sitemap-projection.mjs";
6
+ import { createSitemapReadStore } from "../entities/sitemap.mjs";
7
+ import "../entities.mjs";
8
+ import { utcDateMinusDays } from "./dates.mjs";
9
+ import { partitionsInRange } from "./windows.mjs";
10
+ const indexingMetadataRollup = {
11
+ id: "indexing_metadata",
12
+ windowDays: null,
13
+ async build({ dataSource, ctx }) {
14
+ const index = await createIndexingMetadataStore({ dataSource }).loadIndex(ctx);
15
+ const records = Object.values(index.records);
16
+ const updatesByDay = /* @__PURE__ */ new Map();
17
+ const removesByDay = /* @__PURE__ */ new Map();
18
+ let totalUpdates = 0;
19
+ let totalRemoves = 0;
20
+ let latestUpdate;
21
+ let latestRemove;
22
+ for (const r of records) {
23
+ if (r.latestUpdateAt) {
24
+ totalUpdates++;
25
+ const day = r.latestUpdateAt.slice(0, 10);
26
+ updatesByDay.set(day, (updatesByDay.get(day) ?? 0) + 1);
27
+ if (!latestUpdate || r.latestUpdateAt > latestUpdate) latestUpdate = r.latestUpdateAt;
28
+ }
29
+ if (r.latestRemoveAt) {
30
+ totalRemoves++;
31
+ const day = r.latestRemoveAt.slice(0, 10);
32
+ removesByDay.set(day, (removesByDay.get(day) ?? 0) + 1);
33
+ if (!latestRemove || r.latestRemoveAt > latestRemove) latestRemove = r.latestRemoveAt;
34
+ }
35
+ }
36
+ const days = /* @__PURE__ */ new Set([...updatesByDay.keys(), ...removesByDay.keys()]);
37
+ const perDay = Array.from(days).sort().map((day) => ({
38
+ day,
39
+ updates: updatesByDay.get(day) ?? 0,
40
+ removes: removesByDay.get(day) ?? 0
41
+ }));
42
+ return {
43
+ totals: {
44
+ urls: records.length,
45
+ updates: totalUpdates,
46
+ removes: totalRemoves,
47
+ latestUpdateAt: latestUpdate ?? null,
48
+ latestRemoveAt: latestRemove ?? null
49
+ },
50
+ days: perDay
51
+ };
52
+ }
53
+ };
54
+ const indexingHealthRollup = {
55
+ id: "indexing_health",
56
+ windowDays: 90,
57
+ sliceOrthogonal: true,
58
+ async build({ engine, ctx, dataSource, windowAnchorMs }) {
59
+ const key = inspectionParquetKey(ctx);
60
+ if (!await dataSource.head?.(key)) return { days: [] };
61
+ const sql = `
62
+ SELECT
63
+ substr(CAST(inspectedAt AS VARCHAR), 1, 10) AS date,
64
+ COUNT(*)::BIGINT AS total_urls,
65
+ SUM(CASE WHEN CAST(indexStatus AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS indexed_count,
66
+ SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'SOFT_404' THEN 1 ELSE 0 END)::BIGINT AS soft_404,
67
+ SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'REDIRECT_ERROR' THEN 1 ELSE 0 END)::BIGINT AS redirect,
68
+ SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'NOT_FOUND' THEN 1 ELSE 0 END)::BIGINT AS not_found,
69
+ SUM(CASE WHEN CAST(mobileUsabilityVerdict AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS mobile_passes,
70
+ SUM(CASE WHEN CAST(richResultsVerdict AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS rich_results_passes,
71
+ SUM(CASE WHEN canonicalMismatchKind IN ('path', 'cross_domain') THEN 1 ELSE 0 END)::BIGINT AS canonical_mismatches
72
+ FROM read_parquet({{INSPECTIONS}}, union_by_name = true)
73
+ WHERE substr(CAST(inspectedAt AS VARCHAR), 1, 10) >= '${utcDateMinusDays(windowAnchorMs, 90)}'
74
+ GROUP BY 1
75
+ ORDER BY 1
76
+ `;
77
+ return { days: (await engine.runSQL({
78
+ ctx,
79
+ table: "pages",
80
+ fileSets: { INSPECTIONS: {
81
+ table: "pages",
82
+ keys: [key]
83
+ } },
84
+ sql
85
+ })).rows.map((r) => ({
86
+ date: String(r.date),
87
+ total_urls: Number(r.total_urls),
88
+ indexed_count: Number(r.indexed_count),
89
+ soft_404: Number(r.soft_404),
90
+ redirect: Number(r.redirect),
91
+ not_found: Number(r.not_found),
92
+ mobile_passes: Number(r.mobile_passes),
93
+ rich_results_passes: Number(r.rich_results_passes),
94
+ canonical_mismatches: Number(r.canonical_mismatches)
95
+ })) };
96
+ }
97
+ };
98
+ function currentSitemapProjectionRelation(hasIndexes, hasDeltas) {
99
+ const sources = [];
100
+ if (hasIndexes) sources.push(`
101
+ SELECT
102
+ feedpath_hash,
103
+ url_hash,
104
+ loc,
105
+ removed_at,
106
+ greatest(
107
+ coalesce(removed_at, 0),
108
+ coalesce(last_seen_at, 0),
109
+ coalesce(first_seen_at, 0)
110
+ )::BIGINT AS observed_at,
111
+ 0::INTEGER AS source_order
112
+ FROM read_parquet({{URLS_INDEX}}, union_by_name = true)
113
+ `);
114
+ if (hasDeltas) sources.push(`
115
+ SELECT
116
+ feedpath_hash,
117
+ url_hash,
118
+ loc,
119
+ CASE WHEN op = 'removed' THEN "at" ELSE NULL END AS removed_at,
120
+ "at"::BIGINT AS observed_at,
121
+ 1::INTEGER AS source_order
122
+ FROM read_parquet({{URLS_DELTA}}, union_by_name = true)
123
+ WHERE op IN ('added', 'removed')
124
+ `);
125
+ return `(
126
+ WITH membership_events AS (
127
+ ${sources.join("\nUNION ALL\n")}
128
+ )
129
+ SELECT feedpath_hash, url_hash, loc, removed_at
130
+ FROM membership_events
131
+ QUALIFY row_number() OVER (
132
+ PARTITION BY feedpath_hash, url_hash
133
+ ORDER BY observed_at DESC, source_order DESC
134
+ ) = 1
135
+ )`;
136
+ }
137
+ const indexPercentRollup = {
138
+ id: "index_percent",
139
+ windowDays: 90,
140
+ sliceOrthogonal: true,
141
+ async build({ engine, ctx, dataSource, windowAnchorMs, searchType }) {
142
+ const [listedIndexKeys, listedDeltaKeys, manifestBytes] = await Promise.all([
143
+ dataSource.list(sitemapUrlsIndexPrefix(ctx)),
144
+ dataSource.list(`${sitemapUrlsPrefix(ctx)}/deltas/`),
145
+ readOptional(dataSource, sitemapUrlsProjectionManifestKey(ctx))
146
+ ]);
147
+ const projection = selectSitemapProjectionFiles(listedIndexKeys, listedDeltaKeys, manifestBytes ? decodeSitemapProjectionManifest(new TextDecoder().decode(manifestBytes)) : void 0);
148
+ if (projection.indexKeys.length === 0 && projection.deltaKeys.length === 0) return {
149
+ totalSitemapUrls: 0,
150
+ days: []
151
+ };
152
+ const sitemapFileSets = {
153
+ ...projection.indexKeys.length > 0 ? { URLS_INDEX: {
154
+ table: "pages",
155
+ keys: projection.indexKeys
156
+ } } : {},
157
+ ...projection.deltaKeys.length > 0 ? { URLS_DELTA: {
158
+ table: "pages",
159
+ keys: projection.deltaKeys
160
+ } } : {}
161
+ };
162
+ const currentMembership = currentSitemapProjectionRelation(projection.indexKeys.length > 0, projection.deltaKeys.length > 0);
163
+ const cutoff = utcDateMinusDays(windowAnchorMs, 90);
164
+ const factSearchType = searchType ?? "web";
165
+ const pagesPartitions = partitionsInRange(await engine.listPartitions({
166
+ ctx,
167
+ table: "pages",
168
+ searchType: factSearchType
169
+ }), cutoff, utcDateMinusDays(windowAnchorMs, 0));
170
+ const numerator = await engine.runSQL({
171
+ ctx,
172
+ table: "pages",
173
+ fileSets: {
174
+ PAGES: {
175
+ table: "pages",
176
+ partitions: pagesPartitions
177
+ },
178
+ ...sitemapFileSets
179
+ },
180
+ searchType: factSearchType,
181
+ sql: `
182
+ SELECT
183
+ p.date AS date,
184
+ COUNT(DISTINCT p.url)::BIGINT AS clicked_urls
185
+ FROM read_parquet({{PAGES}}, union_by_name = true) p
186
+ INNER JOIN ${currentMembership} s
187
+ ON s.loc = p.url AND s.removed_at IS NULL
188
+ WHERE p.clicks > 0 AND p.date >= '${cutoff}'
189
+ GROUP BY p.date
190
+ ORDER BY p.date
191
+ `
192
+ });
193
+ const denom = await engine.runSQL({
194
+ ctx,
195
+ table: "pages",
196
+ fileSets: sitemapFileSets,
197
+ sql: `
198
+ SELECT COUNT(DISTINCT loc)::BIGINT AS total
199
+ FROM ${currentMembership}
200
+ WHERE removed_at IS NULL
201
+ `
202
+ });
203
+ const total = Number(denom.rows[0]?.total ?? 0);
204
+ return {
205
+ totalSitemapUrls: total,
206
+ days: numerator.rows.map((r) => {
207
+ const clicked = Number(r.clicked_urls);
208
+ return {
209
+ date: String(r.date),
210
+ clicked_urls: clicked,
211
+ total_sitemap_urls: total,
212
+ ratio: total === 0 ? 0 : clicked / total
213
+ };
214
+ })
215
+ };
216
+ }
217
+ };
218
+ const sitemapHealthRollup = {
219
+ id: "sitemap_health",
220
+ windowDays: 90,
221
+ sliceOrthogonal: true,
222
+ async build({ dataSource, ctx, windowAnchorMs }) {
223
+ const index = await createSitemapReadStore({ dataSource }).loadIndex(ctx);
224
+ const records = Object.values(index.records);
225
+ const cutoff = utcDateMinusDays(windowAnchorMs, 90);
226
+ const byDay = /* @__PURE__ */ new Map();
227
+ const feeds = [];
228
+ for (const r of records) {
229
+ const day = (r.capturedAt ?? r.lastDownloaded ?? "").slice(0, 10);
230
+ if (!day || day < cutoff) continue;
231
+ const errors = Number(r.errors ?? 0);
232
+ const warnings = Number(r.warnings ?? 0);
233
+ const urlCount = Number(r.urlCount ?? 0);
234
+ const bucket = byDay.get(day) ?? {
235
+ day,
236
+ feeds: 0,
237
+ total_urls: 0,
238
+ errors: 0,
239
+ warnings: 0
240
+ };
241
+ bucket.feeds += 1;
242
+ bucket.total_urls += urlCount;
243
+ bucket.errors += errors;
244
+ bucket.warnings += warnings;
245
+ byDay.set(day, bucket);
246
+ feeds.push({
247
+ path: r.path,
248
+ urlCount,
249
+ errors,
250
+ warnings,
251
+ contentHash: r.contentHash ?? null,
252
+ lastDownloaded: r.lastDownloaded ?? null,
253
+ capturedAt: r.capturedAt
254
+ });
255
+ }
256
+ return {
257
+ days: Array.from(byDay.values()).sort((a, b) => a.day < b.day ? -1 : 1),
258
+ feeds
259
+ };
260
+ }
261
+ };
262
+ const RECENT_SITEMAP_CHANGE_LIMIT = 200;
263
+ function isLessRecent(a, b) {
264
+ return a.at < b.at || a.at === b.at && a.sequence > b.sequence;
265
+ }
266
+ function retainRecentSitemapChange(heap, change) {
267
+ if (heap.length < RECENT_SITEMAP_CHANGE_LIMIT) {
268
+ heap.push(change);
269
+ let index = heap.length - 1;
270
+ while (index > 0) {
271
+ const parent = index - 1 >> 1;
272
+ if (!isLessRecent(heap[index], heap[parent])) break;
273
+ const parentValue = heap[parent];
274
+ heap[parent] = heap[index];
275
+ heap[index] = parentValue;
276
+ index = parent;
277
+ }
278
+ return;
279
+ }
280
+ if (!isLessRecent(heap[0], change)) return;
281
+ heap[0] = change;
282
+ let index = 0;
283
+ for (;;) {
284
+ const left = index * 2 + 1;
285
+ const right = left + 1;
286
+ let leastRecent = index;
287
+ if (left < heap.length && isLessRecent(heap[left], heap[leastRecent])) leastRecent = left;
288
+ if (right < heap.length && isLessRecent(heap[right], heap[leastRecent])) leastRecent = right;
289
+ if (leastRecent === index) return;
290
+ const childValue = heap[leastRecent];
291
+ heap[leastRecent] = heap[index];
292
+ heap[index] = childValue;
293
+ index = leastRecent;
294
+ }
295
+ }
296
+ const sitemapChanges28dRollup = {
297
+ id: "sitemap_changes_28d",
298
+ windowDays: 28,
299
+ sliceOrthogonal: true,
300
+ async build({ dataSource, ctx, windowAnchorMs }) {
301
+ const store = createSitemapReadStore({ dataSource });
302
+ const from = utcDateMinusDays(windowAnchorMs, 28);
303
+ const to = utcDateMinusDays(windowAnchorMs, 0);
304
+ const counts = /* @__PURE__ */ new Map();
305
+ const addedTop = [];
306
+ const removedTop = [];
307
+ let sequence = 0;
308
+ function key(k) {
309
+ return `${k.day}\x00${k.feedpath}`;
310
+ }
311
+ for await (const d of store.loadEvents(ctx, {
312
+ from,
313
+ to
314
+ })) {
315
+ const day = new Date(d.observedAt).toISOString().slice(0, 10);
316
+ const k = key({
317
+ day,
318
+ feedpath: d.feedpath
319
+ });
320
+ const cur = counts.get(k) ?? {
321
+ day,
322
+ feedpath: d.feedpath,
323
+ added: 0,
324
+ removed: 0
325
+ };
326
+ const change = {
327
+ loc: d.loc,
328
+ feedpath: d.feedpath,
329
+ at: d.observedAt,
330
+ sequence: sequence++
331
+ };
332
+ if (d.op === "added") {
333
+ cur.added += 1;
334
+ retainRecentSitemapChange(addedTop, change);
335
+ } else {
336
+ cur.removed += 1;
337
+ retainRecentSitemapChange(removedTop, change);
338
+ }
339
+ counts.set(k, cur);
340
+ }
341
+ const days = Array.from(counts.values()).sort((a, b) => {
342
+ if (a.day !== b.day) return a.day < b.day ? -1 : 1;
343
+ return a.feedpath < b.feedpath ? -1 : 1;
344
+ });
345
+ const toRecentList = (heap) => heap.sort((a, b) => b.at - a.at || a.sequence - b.sequence).map(({ loc, feedpath, at }) => ({
346
+ loc,
347
+ feedpath,
348
+ at
349
+ }));
350
+ return {
351
+ days,
352
+ topAdded: toRecentList(addedTop),
353
+ topRemoved: toRecentList(removedTop)
354
+ };
355
+ }
356
+ };
357
+ export { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup };
@@ -0,0 +1,38 @@
1
+ import { RollupDef } from "./core.mjs";
2
+ /**
3
+ * Daily totals across the full history. One row per (date, table) with
4
+ * clicks + impressions + position. Powers sparklines and headline totals.
5
+ *
6
+ * Includes `anonymizedImpressionsPct` per day computed as
7
+ * 1 - sum(query_grained_impressions) / sum(page_grained_impressions)
8
+ * — surfaces GSC's anonymous-query gap so the dashboard can warn users not
9
+ * to trust query-grained breakdowns as comprehensive.
10
+ */
11
+ declare const dailyTotalsRollup: RollupDef;
12
+ /** Weekly totals, ISO week aligned. Cheap and stable for trend widgets. */
13
+ declare const weeklyTotalsRollup: RollupDef;
14
+ /**
15
+ * Top 1000 pages by clicks over the trailing 28-day window. JSON for v1;
16
+ * promote to parquet (`top_pages_28d.parquet`) when the dashboard needs
17
+ * server-side WHERE filtering on this rollup.
18
+ */
19
+ declare const topPages28dRollup: RollupDef;
20
+ /**
21
+ * Top 250 countries by clicks over the trailing 28-day window. Countries
22
+ * cardinality is bounded (~250 ISO codes), so the list fits in a tiny JSON
23
+ * payload regardless of traffic shape. Powers a geo-overview widget without
24
+ * spinning up DuckDB-WASM.
25
+ */
26
+ declare const topCountries28dRollup: RollupDef;
27
+ /**
28
+ * Parquet-format companion to `topKeywords28dRollup`. Same shape, but persists
29
+ * as a parquet object plus JSON sidecar pointer so widgets that need
30
+ * server-side WHERE (filter by prefix, by clicks threshold, paginate) can scan
31
+ * it directly with DuckDB-WASM instead of loading all 1000 rows into JS.
32
+ *
33
+ * Opt-in: include in the caller's rollup def list alongside (or instead of)
34
+ * the JSON variant; the runner treats the two as independent ids so they can
35
+ * coexist during a migration.
36
+ */
37
+ declare const topKeywords28dParquetRollup: RollupDef;
38
+ export { dailyTotalsRollup, topCountries28dRollup, topKeywords28dParquetRollup, topPages28dRollup, weeklyTotalsRollup };