@gscdump/engine 1.4.10 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/entities/empty-types.d.mts +22 -0
- package/dist/entities/empty-types.mjs +58 -0
- package/dist/entities/indexing-metadata.d.mts +26 -0
- package/dist/entities/indexing-metadata.mjs +31 -0
- package/dist/entities/inspection.d.mts +240 -0
- package/dist/entities/inspection.mjs +443 -0
- package/dist/entities/io.mjs +16 -0
- package/dist/{query-dim.d.mts → entities/query-dim.d.mts} +3 -3
- package/dist/{query-dim.mjs → entities/query-dim.mjs} +4 -4
- package/dist/{sitemap-projection.mjs → entities/sitemap-projection.mjs} +1 -1
- package/dist/entities/sitemap-shared.d.mts +199 -0
- package/dist/entities/sitemap-shared.mjs +243 -0
- package/dist/entities/sitemap-write.d.mts +3 -0
- package/dist/entities/sitemap-write.mjs +528 -0
- package/dist/entities/sitemap.d.mts +3 -0
- package/dist/entities/sitemap.mjs +91 -0
- package/dist/entities.d.mts +9 -481
- package/dist/entities.mjs +7 -1380
- package/dist/rollups/canonical.d.mts +71 -0
- package/dist/rollups/canonical.mjs +335 -0
- package/dist/rollups/core.d.mts +201 -0
- package/dist/rollups/core.mjs +116 -0
- package/dist/rollups/dates.mjs +11 -0
- package/dist/rollups/defaults.d.mts +11 -0
- package/dist/rollups/defaults.mjs +17 -0
- package/dist/rollups/hourly.d.mts +38 -0
- package/dist/rollups/hourly.mjs +38 -0
- package/dist/rollups/indexing.d.mts +46 -0
- package/dist/rollups/indexing.mjs +357 -0
- package/dist/rollups/traffic.d.mts +38 -0
- package/dist/rollups/traffic.mjs +289 -0
- package/dist/rollups/windows.d.mts +90 -0
- package/dist/rollups/windows.mjs +176 -0
- package/dist/rollups.d.mts +8 -471
- package/dist/rollups.mjs +7 -1316
- package/package.json +4 -4
- /package/dist/{sitemap-projection.d.mts → entities/sitemap-projection.d.mts} +0 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import { engineErrors } from "../errors.mjs";
|
|
2
|
+
import { encodeRowsToParquetFlex } from "../adapters/hyparquet.mjs";
|
|
3
|
+
import { isoDateToUtcMs } from "./dates.mjs";
|
|
4
|
+
import { encodeJsonBigintSafe } from "@gscdump/lakehouse/bigint";
|
|
5
|
+
function rollupPrefix(ctx, searchType) {
|
|
6
|
+
const base = ctx.siteId ? `u_${ctx.userId}/${ctx.siteId}/rollups` : `u_${ctx.userId}/rollups`;
|
|
7
|
+
return searchType !== void 0 && searchType !== "web" ? `${base}/${searchType}` : base;
|
|
8
|
+
}
|
|
9
|
+
function rollupKey(ctx, id, builtAt, searchType) {
|
|
10
|
+
return `${rollupPrefix(ctx, searchType)}/${id}__v${builtAt}.json`;
|
|
11
|
+
}
|
|
12
|
+
function rollupParquetKey(ctx, id, builtAt, searchType) {
|
|
13
|
+
return `${rollupPrefix(ctx, searchType)}/${id}__v${builtAt}.parquet`;
|
|
14
|
+
}
|
|
15
|
+
const ROLLUP_FILE_RE = /^(?<id>[a-z0-9_]+)__v(?<ts>\d+)\.json$/;
|
|
16
|
+
async function readLatestRollup(bucket, ctx, id, searchType) {
|
|
17
|
+
const prefix = `${rollupPrefix(ctx, searchType)}/`;
|
|
18
|
+
let newest = null;
|
|
19
|
+
let cursor;
|
|
20
|
+
do {
|
|
21
|
+
const listing = await bucket.list({
|
|
22
|
+
prefix,
|
|
23
|
+
cursor
|
|
24
|
+
});
|
|
25
|
+
for (const obj of listing.objects) {
|
|
26
|
+
const m = ROLLUP_FILE_RE.exec(obj.key.slice(prefix.length));
|
|
27
|
+
if (!m?.groups || m.groups.id !== id) continue;
|
|
28
|
+
const ts = Number(m.groups.ts);
|
|
29
|
+
if (!newest || ts > newest.ts) newest = {
|
|
30
|
+
ts,
|
|
31
|
+
key: obj.key
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
cursor = listing.truncated ? listing.cursor : void 0;
|
|
35
|
+
} while (cursor !== void 0);
|
|
36
|
+
if (!newest) return null;
|
|
37
|
+
const obj = await bucket.get(newest.key);
|
|
38
|
+
if (!obj) return null;
|
|
39
|
+
return JSON.parse(await obj.text());
|
|
40
|
+
}
|
|
41
|
+
async function rebuildRollups(opts) {
|
|
42
|
+
const now = opts.now ?? (() => Date.now());
|
|
43
|
+
const dataEndMs = opts.dataEndDate !== void 0 ? isoDateToUtcMs(opts.dataEndDate) : null;
|
|
44
|
+
const results = [];
|
|
45
|
+
for (const def of opts.defs) {
|
|
46
|
+
const builtAt = now();
|
|
47
|
+
const windowAnchorMs = dataEndMs ?? builtAt;
|
|
48
|
+
const defSearchType = def.sliceOrthogonal === true ? void 0 : opts.searchType;
|
|
49
|
+
try {
|
|
50
|
+
const payload = await def.build({
|
|
51
|
+
engine: opts.engine,
|
|
52
|
+
ctx: opts.ctx,
|
|
53
|
+
dataSource: opts.dataSource,
|
|
54
|
+
windowAnchorMs,
|
|
55
|
+
...defSearchType !== void 0 ? { searchType: defSearchType } : {}
|
|
56
|
+
});
|
|
57
|
+
if (def.format === "parquet") {
|
|
58
|
+
if (!def.parquetColumns || def.parquetColumns.length === 0) throw new Error(`rollup '${def.id}' declared format='parquet' without parquetColumns`);
|
|
59
|
+
const rows = payload ?? [];
|
|
60
|
+
const parquetBytes = encodeRowsToParquetFlex(rows, {
|
|
61
|
+
columns: def.parquetColumns,
|
|
62
|
+
sortKey: def.parquetSortKey
|
|
63
|
+
});
|
|
64
|
+
const parquetKey = rollupParquetKey(opts.ctx, def.id, builtAt, defSearchType);
|
|
65
|
+
await opts.dataSource.write(parquetKey, parquetBytes);
|
|
66
|
+
const pointer = {
|
|
67
|
+
parquetKey,
|
|
68
|
+
rowCount: rows.length
|
|
69
|
+
};
|
|
70
|
+
const envelopeBytes = encodeJsonBigintSafe({
|
|
71
|
+
version: 1,
|
|
72
|
+
id: def.id,
|
|
73
|
+
builtAt,
|
|
74
|
+
windowDays: def.windowDays,
|
|
75
|
+
payload: pointer
|
|
76
|
+
});
|
|
77
|
+
const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
|
|
78
|
+
await opts.dataSource.write(key, envelopeBytes);
|
|
79
|
+
results.push({
|
|
80
|
+
id: def.id,
|
|
81
|
+
objectKey: key,
|
|
82
|
+
parquetKey,
|
|
83
|
+
bytes: envelopeBytes.byteLength,
|
|
84
|
+
parquetBytes: parquetBytes.byteLength,
|
|
85
|
+
builtAt
|
|
86
|
+
});
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
const bytes = encodeJsonBigintSafe({
|
|
90
|
+
version: 1,
|
|
91
|
+
id: def.id,
|
|
92
|
+
builtAt,
|
|
93
|
+
windowDays: def.windowDays,
|
|
94
|
+
payload
|
|
95
|
+
});
|
|
96
|
+
const key = rollupKey(opts.ctx, def.id, builtAt, defSearchType);
|
|
97
|
+
await opts.dataSource.write(key, bytes);
|
|
98
|
+
results.push({
|
|
99
|
+
id: def.id,
|
|
100
|
+
objectKey: key,
|
|
101
|
+
bytes: bytes.byteLength,
|
|
102
|
+
builtAt
|
|
103
|
+
});
|
|
104
|
+
} catch (err) {
|
|
105
|
+
results.push({
|
|
106
|
+
id: def.id,
|
|
107
|
+
objectKey: "",
|
|
108
|
+
bytes: 0,
|
|
109
|
+
builtAt,
|
|
110
|
+
error: engineErrors.rollupBuildFailed(def.id, err)
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
return results;
|
|
115
|
+
}
|
|
116
|
+
export { readLatestRollup, rebuildRollups, rollupKey, rollupParquetKey };
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { MS_PER_DAY } from "gscdump/dates";
|
|
2
|
+
function isoDateToUtcMs(iso) {
|
|
3
|
+
const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(iso);
|
|
4
|
+
if (!m) throw new Error(`dataEndDate must be ISO YYYY-MM-DD, got: ${iso}`);
|
|
5
|
+
return Date.UTC(Number(m[1]), Number(m[2]) - 1, Number(m[3]));
|
|
6
|
+
}
|
|
7
|
+
function utcDateMinusDays(at, days) {
|
|
8
|
+
const d = new Date(at - days * MS_PER_DAY);
|
|
9
|
+
return `${d.getUTCFullYear()}-${String(d.getUTCMonth() + 1).padStart(2, "0")}-${String(d.getUTCDate()).padStart(2, "0")}`;
|
|
10
|
+
}
|
|
11
|
+
export { isoDateToUtcMs, utcDateMinusDays };
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { RollupDef } from "./core.mjs";
|
|
2
|
+
declare const DEFAULT_ROLLUPS: readonly RollupDef[];
|
|
3
|
+
/**
|
|
4
|
+
* Canonical-primary rollups (ADR-0017 / ADR-0018). Opt-in — kept out of
|
|
5
|
+
* `DEFAULT_ROLLUPS` because they only pay off once the consumer queries by
|
|
6
|
+
* `queryCanonical` and wires the read seams (`resolveExtra` /
|
|
7
|
+
* `canonicalSource`). Hosts opt in by concatenating these onto their def list
|
|
8
|
+
* (CLI: `gscdump rollups --with-canonical`).
|
|
9
|
+
*/
|
|
10
|
+
declare const CANONICAL_ROLLUPS: readonly RollupDef[];
|
|
11
|
+
export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS };
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { queryCanonicalDailyRollup, queryCanonicalVariantsRollup } from "./canonical.mjs";
|
|
2
|
+
import { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup } from "./indexing.mjs";
|
|
3
|
+
import { dailyTotalsRollup, topCountries28dRollup, topKeywords28dRollup, topPages28dRollup, weeklyTotalsRollup } from "./traffic.mjs";
|
|
4
|
+
const DEFAULT_ROLLUPS = [
|
|
5
|
+
dailyTotalsRollup,
|
|
6
|
+
weeklyTotalsRollup,
|
|
7
|
+
topPages28dRollup,
|
|
8
|
+
topKeywords28dRollup,
|
|
9
|
+
topCountries28dRollup,
|
|
10
|
+
indexingMetadataRollup,
|
|
11
|
+
indexingHealthRollup,
|
|
12
|
+
indexPercentRollup,
|
|
13
|
+
sitemapHealthRollup,
|
|
14
|
+
sitemapChanges28dRollup
|
|
15
|
+
];
|
|
16
|
+
const CANONICAL_ROLLUPS = [queryCanonicalVariantsRollup, queryCanonicalDailyRollup];
|
|
17
|
+
export { CANONICAL_ROLLUPS, DEFAULT_ROLLUPS };
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import { Row as Row$1 } from "../storage.mjs";
|
|
2
|
+
import "../contracts.mjs";
|
|
3
|
+
import { RollupEngine } from "./core.mjs";
|
|
4
|
+
import { TenantCtx } from "@gscdump/contracts";
|
|
5
|
+
import { SearchType } from "gscdump/query";
|
|
6
|
+
/**
|
|
7
|
+
* Aggregate one day's `hourly_pages` partition into the daily `pages` shape
|
|
8
|
+
* and write it to the daily Discover partition. After this runs for date D,
|
|
9
|
+
* the daily query path serves D from `pages/.../daily/D` and the `hourly/D`
|
|
10
|
+
* partition becomes read-only / GC-only.
|
|
11
|
+
*
|
|
12
|
+
* `(position - 1)` weighting matches the storage convention encoded by
|
|
13
|
+
* `toSumPosition`: `sum_position = SUM((position - 1) * impressions)`, so a
|
|
14
|
+
* downstream `SUM(sum_position) / SUM(impressions) + 1` recovers the mean.
|
|
15
|
+
*
|
|
16
|
+
* searchType-scoped: only call with `searchType: 'discover'`. The hourly
|
|
17
|
+
* partition lives under `hourly_pages` and the output lands under `pages` so
|
|
18
|
+
* existing dashboard queries (which read `pages`) see the rolled-up day
|
|
19
|
+
* transparently.
|
|
20
|
+
*/
|
|
21
|
+
interface RebuildDailyFromHourlyOptions {
|
|
22
|
+
engine: RollupEngine & {
|
|
23
|
+
writeDay: (scope: TenantCtx & {
|
|
24
|
+
table: TableTypeName;
|
|
25
|
+
date: string;
|
|
26
|
+
searchType?: SearchType;
|
|
27
|
+
}, rows: Row$1[]) => Promise<void>;
|
|
28
|
+
};
|
|
29
|
+
ctx: TenantCtx;
|
|
30
|
+
/** PT calendar day to roll up. */
|
|
31
|
+
date: string;
|
|
32
|
+
searchType: 'discover';
|
|
33
|
+
}
|
|
34
|
+
type TableTypeName = import('@gscdump/contracts').TableName;
|
|
35
|
+
declare function rebuildDailyFromHourly(opts: RebuildDailyFromHourlyOptions): Promise<{
|
|
36
|
+
rowsWritten: number;
|
|
37
|
+
}>;
|
|
38
|
+
export { RebuildDailyFromHourlyOptions, rebuildDailyFromHourly };
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
async function rebuildDailyFromHourly(opts) {
|
|
2
|
+
const { engine, ctx, date, searchType } = opts;
|
|
3
|
+
const rows = (await engine.runSQL({
|
|
4
|
+
ctx,
|
|
5
|
+
table: "hourly_pages",
|
|
6
|
+
fileSets: { FILES: {
|
|
7
|
+
table: "hourly_pages",
|
|
8
|
+
partitions: [`hourly/${date}`]
|
|
9
|
+
} },
|
|
10
|
+
searchType,
|
|
11
|
+
sql: `
|
|
12
|
+
SELECT
|
|
13
|
+
url,
|
|
14
|
+
DATE '${date}' AS date,
|
|
15
|
+
SUM(clicks)::BIGINT AS clicks,
|
|
16
|
+
SUM(impressions)::BIGINT AS impressions,
|
|
17
|
+
SUM(sum_position)::DOUBLE AS sum_position
|
|
18
|
+
FROM read_parquet({{FILES}}, union_by_name = true)
|
|
19
|
+
WHERE date = '${date}'
|
|
20
|
+
GROUP BY url
|
|
21
|
+
`
|
|
22
|
+
})).rows.map((r) => ({
|
|
23
|
+
url: r.url,
|
|
24
|
+
date,
|
|
25
|
+
clicks: Number(r.clicks),
|
|
26
|
+
impressions: Number(r.impressions),
|
|
27
|
+
sum_position: Number(r.sum_position)
|
|
28
|
+
}));
|
|
29
|
+
await engine.writeDay({
|
|
30
|
+
userId: ctx.userId,
|
|
31
|
+
siteId: ctx.siteId,
|
|
32
|
+
table: "pages",
|
|
33
|
+
date,
|
|
34
|
+
searchType
|
|
35
|
+
}, rows);
|
|
36
|
+
return { rowsWritten: rows.length };
|
|
37
|
+
}
|
|
38
|
+
export { rebuildDailyFromHourly };
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { RollupDef } from "./core.mjs";
|
|
2
|
+
/**
|
|
3
|
+
* Aggregates the per-URL Indexing API metadata entity store (populated by
|
|
4
|
+
* `gscdump entities indexing snapshot`) into daily counts of `URL_UPDATED`
|
|
5
|
+
* and `URL_REMOVED` notifications. Covers the third entity-snapshot shape
|
|
6
|
+
* without needing its own parquet family — publish events are sparse and
|
|
7
|
+
* aggregate cleanly into a small JSON rollup.
|
|
8
|
+
*
|
|
9
|
+
* Safe no-op when the entity store is empty: returns `{ totals: {...}, days: [] }`
|
|
10
|
+
* so downstream readers don't have to special-case first-run sites.
|
|
11
|
+
*/
|
|
12
|
+
declare const indexingMetadataRollup: RollupDef;
|
|
13
|
+
/**
|
|
14
|
+
* Indexing-API health by day: per `inspectedAt` date, counts of indexed,
|
|
15
|
+
* soft-404, redirect, not-found, mobile passes, rich-results passes, and
|
|
16
|
+
* canonical mismatches. Sourced from the inspections parquet sidecar
|
|
17
|
+
* (`InspectionStore.parquetUri`), which holds the latest record per URL.
|
|
18
|
+
*
|
|
19
|
+
* Empty-payload no-op when the sidecar URI is unavailable (in-memory
|
|
20
|
+
* `DataSource`, or before `materialize` has run).
|
|
21
|
+
*/
|
|
22
|
+
declare const indexingHealthRollup: RollupDef;
|
|
23
|
+
/**
|
|
24
|
+
* Per-day index-percent: ratio of (sitemap URLs that received GSC clicks on
|
|
25
|
+
* that date) / (total live sitemap URLs). Uses a DuckDB JOIN between the
|
|
26
|
+
* sitemap urls parquet (`SitemapStore.urlsParquetUri`) and the `pages` fact
|
|
27
|
+
* parquet. Total denominator is the count of live URLs in the urls index;
|
|
28
|
+
* numerator is per-day distinct loc count where pages.clicks > 0.
|
|
29
|
+
*/
|
|
30
|
+
declare const indexPercentRollup: RollupDef;
|
|
31
|
+
/**
|
|
32
|
+
* Sitemap-health per-day series materialized from the sitemap-store JSON
|
|
33
|
+
* index. Each `SitemapRecord` carries `urlCount`, `errors`, `warnings`,
|
|
34
|
+
* `contentHash`, and `lastDownloaded`. We bucket records by the day of their
|
|
35
|
+
* `capturedAt` (or `lastDownloaded` fallback) and emit per-day aggregates plus
|
|
36
|
+
* a snapshot of per-feed stats at the most recent capture.
|
|
37
|
+
*/
|
|
38
|
+
declare const sitemapHealthRollup: RollupDef;
|
|
39
|
+
/**
|
|
40
|
+
* Trailing-28-day sitemap URL changes: per-day per-feedpath {added, removed}
|
|
41
|
+
* counts plus rolling top-200 added and removed URLs. Streams from
|
|
42
|
+
* retained `SitemapReadStore.loadEvents()` history. State compaction cannot
|
|
43
|
+
* erase analytics input; memory scales independently of total site state.
|
|
44
|
+
*/
|
|
45
|
+
declare const sitemapChanges28dRollup: RollupDef;
|
|
46
|
+
export { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup };
|
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
import "../layout.mjs";
|
|
2
|
+
import { inspectionParquetKey, sitemapUrlsIndexPrefix, sitemapUrlsPrefix, sitemapUrlsProjectionManifestKey } from "../entity-keys.mjs";
|
|
3
|
+
import { readOptional } from "../adapters/read-optional.mjs";
|
|
4
|
+
import { createIndexingMetadataStore } from "../entities/indexing-metadata.mjs";
|
|
5
|
+
import { decodeSitemapProjectionManifest, selectSitemapProjectionFiles } from "../entities/sitemap-projection.mjs";
|
|
6
|
+
import { createSitemapReadStore } from "../entities/sitemap.mjs";
|
|
7
|
+
import "../entities.mjs";
|
|
8
|
+
import { utcDateMinusDays } from "./dates.mjs";
|
|
9
|
+
import { partitionsInRange } from "./windows.mjs";
|
|
10
|
+
const indexingMetadataRollup = {
|
|
11
|
+
id: "indexing_metadata",
|
|
12
|
+
windowDays: null,
|
|
13
|
+
async build({ dataSource, ctx }) {
|
|
14
|
+
const index = await createIndexingMetadataStore({ dataSource }).loadIndex(ctx);
|
|
15
|
+
const records = Object.values(index.records);
|
|
16
|
+
const updatesByDay = /* @__PURE__ */ new Map();
|
|
17
|
+
const removesByDay = /* @__PURE__ */ new Map();
|
|
18
|
+
let totalUpdates = 0;
|
|
19
|
+
let totalRemoves = 0;
|
|
20
|
+
let latestUpdate;
|
|
21
|
+
let latestRemove;
|
|
22
|
+
for (const r of records) {
|
|
23
|
+
if (r.latestUpdateAt) {
|
|
24
|
+
totalUpdates++;
|
|
25
|
+
const day = r.latestUpdateAt.slice(0, 10);
|
|
26
|
+
updatesByDay.set(day, (updatesByDay.get(day) ?? 0) + 1);
|
|
27
|
+
if (!latestUpdate || r.latestUpdateAt > latestUpdate) latestUpdate = r.latestUpdateAt;
|
|
28
|
+
}
|
|
29
|
+
if (r.latestRemoveAt) {
|
|
30
|
+
totalRemoves++;
|
|
31
|
+
const day = r.latestRemoveAt.slice(0, 10);
|
|
32
|
+
removesByDay.set(day, (removesByDay.get(day) ?? 0) + 1);
|
|
33
|
+
if (!latestRemove || r.latestRemoveAt > latestRemove) latestRemove = r.latestRemoveAt;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
const days = /* @__PURE__ */ new Set([...updatesByDay.keys(), ...removesByDay.keys()]);
|
|
37
|
+
const perDay = Array.from(days).sort().map((day) => ({
|
|
38
|
+
day,
|
|
39
|
+
updates: updatesByDay.get(day) ?? 0,
|
|
40
|
+
removes: removesByDay.get(day) ?? 0
|
|
41
|
+
}));
|
|
42
|
+
return {
|
|
43
|
+
totals: {
|
|
44
|
+
urls: records.length,
|
|
45
|
+
updates: totalUpdates,
|
|
46
|
+
removes: totalRemoves,
|
|
47
|
+
latestUpdateAt: latestUpdate ?? null,
|
|
48
|
+
latestRemoveAt: latestRemove ?? null
|
|
49
|
+
},
|
|
50
|
+
days: perDay
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
};
|
|
54
|
+
const indexingHealthRollup = {
|
|
55
|
+
id: "indexing_health",
|
|
56
|
+
windowDays: 90,
|
|
57
|
+
sliceOrthogonal: true,
|
|
58
|
+
async build({ engine, ctx, dataSource, windowAnchorMs }) {
|
|
59
|
+
const key = inspectionParquetKey(ctx);
|
|
60
|
+
if (!await dataSource.head?.(key)) return { days: [] };
|
|
61
|
+
const sql = `
|
|
62
|
+
SELECT
|
|
63
|
+
substr(CAST(inspectedAt AS VARCHAR), 1, 10) AS date,
|
|
64
|
+
COUNT(*)::BIGINT AS total_urls,
|
|
65
|
+
SUM(CASE WHEN CAST(indexStatus AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS indexed_count,
|
|
66
|
+
SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'SOFT_404' THEN 1 ELSE 0 END)::BIGINT AS soft_404,
|
|
67
|
+
SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'REDIRECT_ERROR' THEN 1 ELSE 0 END)::BIGINT AS redirect,
|
|
68
|
+
SUM(CASE WHEN CAST(pageFetchState AS VARCHAR) = 'NOT_FOUND' THEN 1 ELSE 0 END)::BIGINT AS not_found,
|
|
69
|
+
SUM(CASE WHEN CAST(mobileUsabilityVerdict AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS mobile_passes,
|
|
70
|
+
SUM(CASE WHEN CAST(richResultsVerdict AS VARCHAR) = 'PASS' THEN 1 ELSE 0 END)::BIGINT AS rich_results_passes,
|
|
71
|
+
SUM(CASE WHEN canonicalMismatchKind IN ('path', 'cross_domain') THEN 1 ELSE 0 END)::BIGINT AS canonical_mismatches
|
|
72
|
+
FROM read_parquet({{INSPECTIONS}}, union_by_name = true)
|
|
73
|
+
WHERE substr(CAST(inspectedAt AS VARCHAR), 1, 10) >= '${utcDateMinusDays(windowAnchorMs, 90)}'
|
|
74
|
+
GROUP BY 1
|
|
75
|
+
ORDER BY 1
|
|
76
|
+
`;
|
|
77
|
+
return { days: (await engine.runSQL({
|
|
78
|
+
ctx,
|
|
79
|
+
table: "pages",
|
|
80
|
+
fileSets: { INSPECTIONS: {
|
|
81
|
+
table: "pages",
|
|
82
|
+
keys: [key]
|
|
83
|
+
} },
|
|
84
|
+
sql
|
|
85
|
+
})).rows.map((r) => ({
|
|
86
|
+
date: String(r.date),
|
|
87
|
+
total_urls: Number(r.total_urls),
|
|
88
|
+
indexed_count: Number(r.indexed_count),
|
|
89
|
+
soft_404: Number(r.soft_404),
|
|
90
|
+
redirect: Number(r.redirect),
|
|
91
|
+
not_found: Number(r.not_found),
|
|
92
|
+
mobile_passes: Number(r.mobile_passes),
|
|
93
|
+
rich_results_passes: Number(r.rich_results_passes),
|
|
94
|
+
canonical_mismatches: Number(r.canonical_mismatches)
|
|
95
|
+
})) };
|
|
96
|
+
}
|
|
97
|
+
};
|
|
98
|
+
function currentSitemapProjectionRelation(hasIndexes, hasDeltas) {
|
|
99
|
+
const sources = [];
|
|
100
|
+
if (hasIndexes) sources.push(`
|
|
101
|
+
SELECT
|
|
102
|
+
feedpath_hash,
|
|
103
|
+
url_hash,
|
|
104
|
+
loc,
|
|
105
|
+
removed_at,
|
|
106
|
+
greatest(
|
|
107
|
+
coalesce(removed_at, 0),
|
|
108
|
+
coalesce(last_seen_at, 0),
|
|
109
|
+
coalesce(first_seen_at, 0)
|
|
110
|
+
)::BIGINT AS observed_at,
|
|
111
|
+
0::INTEGER AS source_order
|
|
112
|
+
FROM read_parquet({{URLS_INDEX}}, union_by_name = true)
|
|
113
|
+
`);
|
|
114
|
+
if (hasDeltas) sources.push(`
|
|
115
|
+
SELECT
|
|
116
|
+
feedpath_hash,
|
|
117
|
+
url_hash,
|
|
118
|
+
loc,
|
|
119
|
+
CASE WHEN op = 'removed' THEN "at" ELSE NULL END AS removed_at,
|
|
120
|
+
"at"::BIGINT AS observed_at,
|
|
121
|
+
1::INTEGER AS source_order
|
|
122
|
+
FROM read_parquet({{URLS_DELTA}}, union_by_name = true)
|
|
123
|
+
WHERE op IN ('added', 'removed')
|
|
124
|
+
`);
|
|
125
|
+
return `(
|
|
126
|
+
WITH membership_events AS (
|
|
127
|
+
${sources.join("\nUNION ALL\n")}
|
|
128
|
+
)
|
|
129
|
+
SELECT feedpath_hash, url_hash, loc, removed_at
|
|
130
|
+
FROM membership_events
|
|
131
|
+
QUALIFY row_number() OVER (
|
|
132
|
+
PARTITION BY feedpath_hash, url_hash
|
|
133
|
+
ORDER BY observed_at DESC, source_order DESC
|
|
134
|
+
) = 1
|
|
135
|
+
)`;
|
|
136
|
+
}
|
|
137
|
+
const indexPercentRollup = {
|
|
138
|
+
id: "index_percent",
|
|
139
|
+
windowDays: 90,
|
|
140
|
+
sliceOrthogonal: true,
|
|
141
|
+
async build({ engine, ctx, dataSource, windowAnchorMs, searchType }) {
|
|
142
|
+
const [listedIndexKeys, listedDeltaKeys, manifestBytes] = await Promise.all([
|
|
143
|
+
dataSource.list(sitemapUrlsIndexPrefix(ctx)),
|
|
144
|
+
dataSource.list(`${sitemapUrlsPrefix(ctx)}/deltas/`),
|
|
145
|
+
readOptional(dataSource, sitemapUrlsProjectionManifestKey(ctx))
|
|
146
|
+
]);
|
|
147
|
+
const projection = selectSitemapProjectionFiles(listedIndexKeys, listedDeltaKeys, manifestBytes ? decodeSitemapProjectionManifest(new TextDecoder().decode(manifestBytes)) : void 0);
|
|
148
|
+
if (projection.indexKeys.length === 0 && projection.deltaKeys.length === 0) return {
|
|
149
|
+
totalSitemapUrls: 0,
|
|
150
|
+
days: []
|
|
151
|
+
};
|
|
152
|
+
const sitemapFileSets = {
|
|
153
|
+
...projection.indexKeys.length > 0 ? { URLS_INDEX: {
|
|
154
|
+
table: "pages",
|
|
155
|
+
keys: projection.indexKeys
|
|
156
|
+
} } : {},
|
|
157
|
+
...projection.deltaKeys.length > 0 ? { URLS_DELTA: {
|
|
158
|
+
table: "pages",
|
|
159
|
+
keys: projection.deltaKeys
|
|
160
|
+
} } : {}
|
|
161
|
+
};
|
|
162
|
+
const currentMembership = currentSitemapProjectionRelation(projection.indexKeys.length > 0, projection.deltaKeys.length > 0);
|
|
163
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 90);
|
|
164
|
+
const factSearchType = searchType ?? "web";
|
|
165
|
+
const pagesPartitions = partitionsInRange(await engine.listPartitions({
|
|
166
|
+
ctx,
|
|
167
|
+
table: "pages",
|
|
168
|
+
searchType: factSearchType
|
|
169
|
+
}), cutoff, utcDateMinusDays(windowAnchorMs, 0));
|
|
170
|
+
const numerator = await engine.runSQL({
|
|
171
|
+
ctx,
|
|
172
|
+
table: "pages",
|
|
173
|
+
fileSets: {
|
|
174
|
+
PAGES: {
|
|
175
|
+
table: "pages",
|
|
176
|
+
partitions: pagesPartitions
|
|
177
|
+
},
|
|
178
|
+
...sitemapFileSets
|
|
179
|
+
},
|
|
180
|
+
searchType: factSearchType,
|
|
181
|
+
sql: `
|
|
182
|
+
SELECT
|
|
183
|
+
p.date AS date,
|
|
184
|
+
COUNT(DISTINCT p.url)::BIGINT AS clicked_urls
|
|
185
|
+
FROM read_parquet({{PAGES}}, union_by_name = true) p
|
|
186
|
+
INNER JOIN ${currentMembership} s
|
|
187
|
+
ON s.loc = p.url AND s.removed_at IS NULL
|
|
188
|
+
WHERE p.clicks > 0 AND p.date >= '${cutoff}'
|
|
189
|
+
GROUP BY p.date
|
|
190
|
+
ORDER BY p.date
|
|
191
|
+
`
|
|
192
|
+
});
|
|
193
|
+
const denom = await engine.runSQL({
|
|
194
|
+
ctx,
|
|
195
|
+
table: "pages",
|
|
196
|
+
fileSets: sitemapFileSets,
|
|
197
|
+
sql: `
|
|
198
|
+
SELECT COUNT(DISTINCT loc)::BIGINT AS total
|
|
199
|
+
FROM ${currentMembership}
|
|
200
|
+
WHERE removed_at IS NULL
|
|
201
|
+
`
|
|
202
|
+
});
|
|
203
|
+
const total = Number(denom.rows[0]?.total ?? 0);
|
|
204
|
+
return {
|
|
205
|
+
totalSitemapUrls: total,
|
|
206
|
+
days: numerator.rows.map((r) => {
|
|
207
|
+
const clicked = Number(r.clicked_urls);
|
|
208
|
+
return {
|
|
209
|
+
date: String(r.date),
|
|
210
|
+
clicked_urls: clicked,
|
|
211
|
+
total_sitemap_urls: total,
|
|
212
|
+
ratio: total === 0 ? 0 : clicked / total
|
|
213
|
+
};
|
|
214
|
+
})
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
};
|
|
218
|
+
const sitemapHealthRollup = {
|
|
219
|
+
id: "sitemap_health",
|
|
220
|
+
windowDays: 90,
|
|
221
|
+
sliceOrthogonal: true,
|
|
222
|
+
async build({ dataSource, ctx, windowAnchorMs }) {
|
|
223
|
+
const index = await createSitemapReadStore({ dataSource }).loadIndex(ctx);
|
|
224
|
+
const records = Object.values(index.records);
|
|
225
|
+
const cutoff = utcDateMinusDays(windowAnchorMs, 90);
|
|
226
|
+
const byDay = /* @__PURE__ */ new Map();
|
|
227
|
+
const feeds = [];
|
|
228
|
+
for (const r of records) {
|
|
229
|
+
const day = (r.capturedAt ?? r.lastDownloaded ?? "").slice(0, 10);
|
|
230
|
+
if (!day || day < cutoff) continue;
|
|
231
|
+
const errors = Number(r.errors ?? 0);
|
|
232
|
+
const warnings = Number(r.warnings ?? 0);
|
|
233
|
+
const urlCount = Number(r.urlCount ?? 0);
|
|
234
|
+
const bucket = byDay.get(day) ?? {
|
|
235
|
+
day,
|
|
236
|
+
feeds: 0,
|
|
237
|
+
total_urls: 0,
|
|
238
|
+
errors: 0,
|
|
239
|
+
warnings: 0
|
|
240
|
+
};
|
|
241
|
+
bucket.feeds += 1;
|
|
242
|
+
bucket.total_urls += urlCount;
|
|
243
|
+
bucket.errors += errors;
|
|
244
|
+
bucket.warnings += warnings;
|
|
245
|
+
byDay.set(day, bucket);
|
|
246
|
+
feeds.push({
|
|
247
|
+
path: r.path,
|
|
248
|
+
urlCount,
|
|
249
|
+
errors,
|
|
250
|
+
warnings,
|
|
251
|
+
contentHash: r.contentHash ?? null,
|
|
252
|
+
lastDownloaded: r.lastDownloaded ?? null,
|
|
253
|
+
capturedAt: r.capturedAt
|
|
254
|
+
});
|
|
255
|
+
}
|
|
256
|
+
return {
|
|
257
|
+
days: Array.from(byDay.values()).sort((a, b) => a.day < b.day ? -1 : 1),
|
|
258
|
+
feeds
|
|
259
|
+
};
|
|
260
|
+
}
|
|
261
|
+
};
|
|
262
|
+
const RECENT_SITEMAP_CHANGE_LIMIT = 200;
|
|
263
|
+
function isLessRecent(a, b) {
|
|
264
|
+
return a.at < b.at || a.at === b.at && a.sequence > b.sequence;
|
|
265
|
+
}
|
|
266
|
+
function retainRecentSitemapChange(heap, change) {
|
|
267
|
+
if (heap.length < RECENT_SITEMAP_CHANGE_LIMIT) {
|
|
268
|
+
heap.push(change);
|
|
269
|
+
let index = heap.length - 1;
|
|
270
|
+
while (index > 0) {
|
|
271
|
+
const parent = index - 1 >> 1;
|
|
272
|
+
if (!isLessRecent(heap[index], heap[parent])) break;
|
|
273
|
+
const parentValue = heap[parent];
|
|
274
|
+
heap[parent] = heap[index];
|
|
275
|
+
heap[index] = parentValue;
|
|
276
|
+
index = parent;
|
|
277
|
+
}
|
|
278
|
+
return;
|
|
279
|
+
}
|
|
280
|
+
if (!isLessRecent(heap[0], change)) return;
|
|
281
|
+
heap[0] = change;
|
|
282
|
+
let index = 0;
|
|
283
|
+
for (;;) {
|
|
284
|
+
const left = index * 2 + 1;
|
|
285
|
+
const right = left + 1;
|
|
286
|
+
let leastRecent = index;
|
|
287
|
+
if (left < heap.length && isLessRecent(heap[left], heap[leastRecent])) leastRecent = left;
|
|
288
|
+
if (right < heap.length && isLessRecent(heap[right], heap[leastRecent])) leastRecent = right;
|
|
289
|
+
if (leastRecent === index) return;
|
|
290
|
+
const childValue = heap[leastRecent];
|
|
291
|
+
heap[leastRecent] = heap[index];
|
|
292
|
+
heap[index] = childValue;
|
|
293
|
+
index = leastRecent;
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
const sitemapChanges28dRollup = {
|
|
297
|
+
id: "sitemap_changes_28d",
|
|
298
|
+
windowDays: 28,
|
|
299
|
+
sliceOrthogonal: true,
|
|
300
|
+
async build({ dataSource, ctx, windowAnchorMs }) {
|
|
301
|
+
const store = createSitemapReadStore({ dataSource });
|
|
302
|
+
const from = utcDateMinusDays(windowAnchorMs, 28);
|
|
303
|
+
const to = utcDateMinusDays(windowAnchorMs, 0);
|
|
304
|
+
const counts = /* @__PURE__ */ new Map();
|
|
305
|
+
const addedTop = [];
|
|
306
|
+
const removedTop = [];
|
|
307
|
+
let sequence = 0;
|
|
308
|
+
function key(k) {
|
|
309
|
+
return `${k.day}\x00${k.feedpath}`;
|
|
310
|
+
}
|
|
311
|
+
for await (const d of store.loadEvents(ctx, {
|
|
312
|
+
from,
|
|
313
|
+
to
|
|
314
|
+
})) {
|
|
315
|
+
const day = new Date(d.observedAt).toISOString().slice(0, 10);
|
|
316
|
+
const k = key({
|
|
317
|
+
day,
|
|
318
|
+
feedpath: d.feedpath
|
|
319
|
+
});
|
|
320
|
+
const cur = counts.get(k) ?? {
|
|
321
|
+
day,
|
|
322
|
+
feedpath: d.feedpath,
|
|
323
|
+
added: 0,
|
|
324
|
+
removed: 0
|
|
325
|
+
};
|
|
326
|
+
const change = {
|
|
327
|
+
loc: d.loc,
|
|
328
|
+
feedpath: d.feedpath,
|
|
329
|
+
at: d.observedAt,
|
|
330
|
+
sequence: sequence++
|
|
331
|
+
};
|
|
332
|
+
if (d.op === "added") {
|
|
333
|
+
cur.added += 1;
|
|
334
|
+
retainRecentSitemapChange(addedTop, change);
|
|
335
|
+
} else {
|
|
336
|
+
cur.removed += 1;
|
|
337
|
+
retainRecentSitemapChange(removedTop, change);
|
|
338
|
+
}
|
|
339
|
+
counts.set(k, cur);
|
|
340
|
+
}
|
|
341
|
+
const days = Array.from(counts.values()).sort((a, b) => {
|
|
342
|
+
if (a.day !== b.day) return a.day < b.day ? -1 : 1;
|
|
343
|
+
return a.feedpath < b.feedpath ? -1 : 1;
|
|
344
|
+
});
|
|
345
|
+
const toRecentList = (heap) => heap.sort((a, b) => b.at - a.at || a.sequence - b.sequence).map(({ loc, feedpath, at }) => ({
|
|
346
|
+
loc,
|
|
347
|
+
feedpath,
|
|
348
|
+
at
|
|
349
|
+
}));
|
|
350
|
+
return {
|
|
351
|
+
days,
|
|
352
|
+
topAdded: toRecentList(addedTop),
|
|
353
|
+
topRemoved: toRecentList(removedTop)
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
};
|
|
357
|
+
export { indexPercentRollup, indexingHealthRollup, indexingMetadataRollup, sitemapChanges28dRollup, sitemapHealthRollup };
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import { RollupDef } from "./core.mjs";
|
|
2
|
+
/**
|
|
3
|
+
* Daily totals across the full history. One row per (date, table) with
|
|
4
|
+
* clicks + impressions + position. Powers sparklines and headline totals.
|
|
5
|
+
*
|
|
6
|
+
* Includes `anonymizedImpressionsPct` per day computed as
|
|
7
|
+
* 1 - sum(query_grained_impressions) / sum(page_grained_impressions)
|
|
8
|
+
* — surfaces GSC's anonymous-query gap so the dashboard can warn users not
|
|
9
|
+
* to trust query-grained breakdowns as comprehensive.
|
|
10
|
+
*/
|
|
11
|
+
declare const dailyTotalsRollup: RollupDef;
|
|
12
|
+
/** Weekly totals, ISO week aligned. Cheap and stable for trend widgets. */
|
|
13
|
+
declare const weeklyTotalsRollup: RollupDef;
|
|
14
|
+
/**
|
|
15
|
+
* Top 1000 pages by clicks over the trailing 28-day window. JSON for v1;
|
|
16
|
+
* promote to parquet (`top_pages_28d.parquet`) when the dashboard needs
|
|
17
|
+
* server-side WHERE filtering on this rollup.
|
|
18
|
+
*/
|
|
19
|
+
declare const topPages28dRollup: RollupDef;
|
|
20
|
+
/**
|
|
21
|
+
* Top 250 countries by clicks over the trailing 28-day window. Countries
|
|
22
|
+
* cardinality is bounded (~250 ISO codes), so the list fits in a tiny JSON
|
|
23
|
+
* payload regardless of traffic shape. Powers a geo-overview widget without
|
|
24
|
+
* spinning up DuckDB-WASM.
|
|
25
|
+
*/
|
|
26
|
+
declare const topCountries28dRollup: RollupDef;
|
|
27
|
+
/**
|
|
28
|
+
* Parquet-format companion to `topKeywords28dRollup`. Same shape, but persists
|
|
29
|
+
* as a parquet object plus JSON sidecar pointer so widgets that need
|
|
30
|
+
* server-side WHERE (filter by prefix, by clicks threshold, paginate) can scan
|
|
31
|
+
* it directly with DuckDB-WASM instead of loading all 1000 rows into JS.
|
|
32
|
+
*
|
|
33
|
+
* Opt-in: include in the caller's rollup def list alongside (or instead of)
|
|
34
|
+
* the JSON variant; the runner treats the two as independent ids so they can
|
|
35
|
+
* coexist during a migration.
|
|
36
|
+
*/
|
|
37
|
+
declare const topKeywords28dParquetRollup: RollupDef;
|
|
38
|
+
export { dailyTotalsRollup, topCountries28dRollup, topKeywords28dParquetRollup, topPages28dRollup, weeklyTotalsRollup };
|