@gscdump/engine 3.4.3 → 3.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -312,7 +312,7 @@ function buildExtrasQueries(state, options) {
|
|
|
312
312
|
const whereExpr = whereParts.length > 0 ? sql`WHERE ${joinAnd(whereParts)}` : sql``;
|
|
313
313
|
const outerQueryCol = sql.raw("query");
|
|
314
314
|
const canonKey = adapter.dimExprSql("queryCanonical", queriesKey);
|
|
315
|
-
const compiled = compileCollapsed(adapter, sql`WITH per_variant AS (SELECT ${canonKey} as joinKey, ${t.query} as query, SUM(${t.clicks}) as clicks, SUM(${t.impressions}) as impressions, SUM(${t.sum_position}) as sum_pos, ROW_NUMBER() OVER (PARTITION BY ${canonKey} ORDER BY SUM(${t.clicks}) DESC) as rn, COUNT(*) OVER (PARTITION BY ${canonKey}) as variantCount FROM ${table} ${whereExpr} GROUP BY ${canonKey}, ${t.query}) SELECT joinKey, MAX(variantCount) as variantCount, MAX(CASE WHEN rn = 1 THEN ${outerQueryCol} END) as canonicalName,
|
|
315
|
+
const compiled = compileCollapsed(adapter, sql`WITH per_variant AS (SELECT ${canonKey} as joinKey, ${t.query} as query, SUM(${t.clicks}) as clicks, SUM(${t.impressions}) as impressions, SUM(${t.sum_position}) as sum_pos, ROW_NUMBER() OVER (PARTITION BY ${canonKey} ORDER BY SUM(${t.clicks}) DESC) as rn, COUNT(*) OVER (PARTITION BY ${canonKey}) as variantCount FROM ${table} ${whereExpr} GROUP BY ${canonKey}, ${t.query}) SELECT joinKey, MAX(variantCount) as variantCount, MAX(CASE WHEN rn = 1 THEN ${outerQueryCol} END) as canonicalName, STRING_AGG(CASE WHEN rn <= 10 THEN ${outerQueryCol} || ':::' || clicks || ':::' || impressions || ':::' || CAST(ROUND(CAST(sum_pos AS REAL) / NULLIF(impressions, 0) + 1, 1) AS TEXT) END, '||') as variants FROM per_variant GROUP BY joinKey`);
|
|
316
316
|
extras.push({
|
|
317
317
|
key: "canonicalExtras",
|
|
318
318
|
sql: compiled.sql,
|
|
@@ -32,6 +32,7 @@ interface R2SqlResolverAdapterOptions extends ResolverAdapterOptions {
|
|
|
32
32
|
declare function createParquetResolverAdapter(options?: ResolverAdapterOptions): ResolverAdapter<PgTableKey>;
|
|
33
33
|
/**
|
|
34
34
|
* Multi-tenant pg-flavored adapter for the Iceberg / R2 SQL read path.
|
|
35
|
+
* Set `dialect: 'r2sql'` to emit R2 SQL regex predicates. The default targets DuckDB.
|
|
35
36
|
* Identical SQL output to `pgResolverAdapter` except WHERE clauses inject
|
|
36
37
|
* `site_id = ?` AND `search_type = ?` automatically when those scopes are
|
|
37
38
|
* passed to `resolveToSQL`. Required for the Iceberg fact tables which are
|
|
@@ -39,8 +40,15 @@ declare function createParquetResolverAdapter(options?: ResolverAdapterOptions):
|
|
|
39
40
|
* cross-tenant data. Single-use: the adapter has no `tableRef` override,
|
|
40
41
|
* so callers must rewrite bare table names to their qualified form (e.g.
|
|
41
42
|
* `${namespace}.pages`) before sending to R2 SQL.
|
|
43
|
+
*
|
|
44
|
+
* The bare `site_id = ?` / `search_type = ?` partition predicates are only
|
|
45
|
+
* safe on int-partition catalogs. Legacy string-partition catalogs must pass
|
|
46
|
+
* `partitionKeyEncoding: 'string'` to emit `CONCAT(col, '') = ?` instead —
|
|
47
|
+
* bare equality undercounts (silently returns fewer rows) on those catalogs.
|
|
42
48
|
*/
|
|
43
|
-
declare function createIcebergResolverAdapter(options?:
|
|
49
|
+
declare function createIcebergResolverAdapter(options?: R2SqlResolverAdapterOptions & {
|
|
50
|
+
dialect?: 'duckdb' | 'r2sql';
|
|
51
|
+
}): ResolverAdapter<PgTableKey>;
|
|
44
52
|
/**
|
|
45
53
|
* R2 SQL adapter for the Iceberg fact tables.
|
|
46
54
|
*
|
|
@@ -65,6 +65,14 @@ const pgResolverAdapter = createResolverAdapter({
|
|
|
65
65
|
...PG_BASE_CONFIG,
|
|
66
66
|
tableLabel: "pg-resolver-adapter"
|
|
67
67
|
});
|
|
68
|
+
function withPartitionKeyEncoding(adapter, encoding) {
|
|
69
|
+
if ((encoding ?? DEFAULT_PARTITION_KEY_ENCODING) === "int") return adapter;
|
|
70
|
+
return {
|
|
71
|
+
...adapter,
|
|
72
|
+
siteIdColRef: (tk) => sql`CONCAT(${adapter.siteIdColRef(tk)}, '')`,
|
|
73
|
+
searchTypeColRef: (tk) => sql`CONCAT(${adapter.searchTypeColRef(tk)}, '')`
|
|
74
|
+
};
|
|
75
|
+
}
|
|
68
76
|
function createParquetResolverAdapter(options = {}) {
|
|
69
77
|
return createResolverAdapter({
|
|
70
78
|
...PG_BASE_CONFIG,
|
|
@@ -75,19 +83,20 @@ function createParquetResolverAdapter(options = {}) {
|
|
|
75
83
|
});
|
|
76
84
|
}
|
|
77
85
|
function createIcebergResolverAdapter(options = {}) {
|
|
78
|
-
return createResolverAdapter({
|
|
86
|
+
return withPartitionKeyEncoding(createResolverAdapter({
|
|
79
87
|
...PG_BASE_CONFIG,
|
|
80
88
|
schema: icebergSchema,
|
|
81
89
|
includeSiteId: true,
|
|
82
90
|
includeSearchType: true,
|
|
83
91
|
tableLabel: "iceberg-resolver-adapter",
|
|
92
|
+
regexPredicate: options.dialect === "r2sql" ? (expr, pattern, negate) => negate ? sql`NOT regexp_like(${expr}, ${pattern})` : sql`regexp_like(${expr}, ${pattern})` : PG_BASE_CONFIG.regexPredicate,
|
|
84
93
|
queryCanonicalSource: options.queryCanonicalSource ?? "queryDim",
|
|
85
94
|
tableRef: (tk) => sql.raw(`"${tk}"`),
|
|
86
95
|
queryDimTableRef: () => sql.raw("\"query_dim\"")
|
|
87
|
-
});
|
|
96
|
+
}), options.partitionKeyEncoding);
|
|
88
97
|
}
|
|
89
98
|
function createR2SqlResolverAdapter(options = {}) {
|
|
90
|
-
|
|
99
|
+
return withPartitionKeyEncoding(createResolverAdapter({
|
|
91
100
|
...PG_BASE_CONFIG,
|
|
92
101
|
schema: icebergSchema,
|
|
93
102
|
includeSiteId: true,
|
|
@@ -101,12 +110,6 @@ function createR2SqlResolverAdapter(options = {}) {
|
|
|
101
110
|
},
|
|
102
111
|
tableRef: (tk) => sql.raw(`"${tk}"`),
|
|
103
112
|
queryDimTableRef: () => sql.raw("\"query_dim\"")
|
|
104
|
-
});
|
|
105
|
-
if ((options.partitionKeyEncoding ?? DEFAULT_PARTITION_KEY_ENCODING) === "int") return adapter;
|
|
106
|
-
return {
|
|
107
|
-
...adapter,
|
|
108
|
-
siteIdColRef: (tk) => sql`CONCAT(${adapter.siteIdColRef(tk)}, '')`,
|
|
109
|
-
searchTypeColRef: (tk) => sql`CONCAT(${adapter.searchTypeColRef(tk)}, '')`
|
|
110
|
-
};
|
|
113
|
+
}), options.partitionKeyEncoding);
|
|
111
114
|
}
|
|
112
115
|
export { createIcebergResolverAdapter, createParquetResolverAdapter, createR2SqlResolverAdapter, pgResolverAdapter };
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@gscdump/engine",
|
|
3
3
|
"type": "module",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.5.0",
|
|
5
5
|
"description": "Append-only Parquet/DuckDB storage engine + planner + adapters for the gscdump pipeline. Node + edge runtimes; opt-in heavy peers.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Harlan Wilton",
|
|
@@ -189,10 +189,10 @@
|
|
|
189
189
|
}
|
|
190
190
|
},
|
|
191
191
|
"dependencies": {
|
|
192
|
-
"@gscdump/contracts": "^3.
|
|
193
|
-
"@gscdump/lakehouse": "^3.
|
|
192
|
+
"@gscdump/contracts": "^3.5.0",
|
|
193
|
+
"@gscdump/lakehouse": "^3.5.0",
|
|
194
194
|
"drizzle-orm": "1.0.0-rc.4",
|
|
195
|
-
"gscdump": "^3.
|
|
195
|
+
"gscdump": "^3.5.0",
|
|
196
196
|
"proper-lockfile": "^4.1.2"
|
|
197
197
|
},
|
|
198
198
|
"devDependencies": {
|