@haystackeditor/cli 0.17.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,140 @@
1
+ /**
2
+ * Replaces every literal in a piece of Postgres SQL text with `?`, for index
3
+ * definitions: a partial-index predicate or an index expression can hold a
4
+ * value (`WHERE email <> 'ceo@example.com'`, `WHERE tenant_id = 4711`).
5
+ *
6
+ * Approach: a lexer, not a pattern. The text is read left to right with the
7
+ * same token rules as the Postgres lexer, so every character is known to be
8
+ * inside a quoted identifier, a string, a number or plain syntax before
9
+ * anything is replaced:
10
+ * - quoted identifiers ("Mixed Case", with "" escapes) are copied untouched,
11
+ * so a name that looks like a string or holds digits is never mangled;
12
+ * - unquoted identifiers and keywords (col1, idx_2024, text_pattern_ops) are
13
+ * read whole, so the digits inside them are never taken for numbers;
14
+ * - strings become `?`: '...' with '' escapes, E'...' with backslash escapes,
15
+ * B'...' / X'...' bit strings, N'...', U&'...' and $tag$...$tag$;
16
+ * - numbers (42, 27182.8182, 1e300, 0x1F, 1_000) become `?`.
17
+ * Everything else (operators, parentheses, `::` and the type names after it)
18
+ * is copied. So `'2024-01-01'::date` becomes `?::date`: the value goes, the
19
+ * type (schema, not data) stays; `ARRAY['a'::text, 'b'::text]` becomes
20
+ * `ARRAY[?::text, ?::text]`; `'{1,2}'::integer[]` becomes `?::integer[]`.
21
+ * Numbers in type modifiers (`numeric(10,2)`) and storage options
22
+ * (`fillfactor='70'`) are replaced too: over-redacting schema is harmless.
23
+ * The keywords TRUE, FALSE and NULL are kept; they cannot carry a value.
24
+ *
25
+ * Postgres never prints a comment in a definition, and a comment could hide
26
+ * text this lexer does not read, so a comment (or an unterminated string or
27
+ * quoted identifier) is refused instead of guessed at.
28
+ */
29
+ export class SqlLiteralError extends Error {
30
+ constructor(message) {
31
+ super(message);
32
+ this.name = 'SqlLiteralError';
33
+ }
34
+ }
35
+ const IDENT_START = /[A-Za-z_\u0080-￿]/u;
36
+ const IDENT_PART = /[A-Za-z0-9_$\u0080-￿]/u;
37
+ const DIGIT = /[0-9]/u;
38
+ const NUMBER = /^(?:0[xX][0-9A-Fa-f_]+|0[oO][0-7_]+|0[bB][01_]+|(?:[0-9][0-9_]*(?:\.[0-9_]*)?|\.[0-9][0-9_]*)(?:[eE][+-]?[0-9]+)?)/u;
39
+ const DOLLAR_TAG = /^\$(?:[A-Za-z_\u0080-￿][A-Za-z0-9_\u0080-￿]*)?\$/u;
40
+ /** End (exclusive) of a quote-delimited token starting at `start` (the
41
+ * opening quote). A doubled quote is an escaped quote; with `backslashes`, a
42
+ * backslash escapes the next character (E'...' strings). */
43
+ function quotedEnd(text, start, quote, backslashes, what) {
44
+ let index = start + 1;
45
+ while (index < text.length) {
46
+ const char = text[index];
47
+ if (backslashes && char === '\\') {
48
+ index += 2;
49
+ continue;
50
+ }
51
+ if (char === quote) {
52
+ if (text[index + 1] === quote) {
53
+ index += 2;
54
+ continue;
55
+ }
56
+ return index + 1;
57
+ }
58
+ index += 1;
59
+ }
60
+ throw new SqlLiteralError(`it has an unterminated ${what}`);
61
+ }
62
+ /** `text` with every literal replaced by `?`. Throws SqlLiteralError when the
63
+ * text holds something the lexer refuses to guess about. */
64
+ export function redactSqlLiterals(text) {
65
+ let out = '';
66
+ let index = 0;
67
+ while (index < text.length) {
68
+ const char = text[index];
69
+ const next = text[index + 1] ?? '';
70
+ if ((char === '-' && next === '-') || (char === '/' && next === '*')) {
71
+ throw new SqlLiteralError('it contains a comment, which Postgres never prints in a definition');
72
+ }
73
+ if (char === '"') {
74
+ const end = quotedEnd(text, index, '"', false, 'quoted identifier');
75
+ out += text.slice(index, end);
76
+ index = end;
77
+ continue;
78
+ }
79
+ if (char === "'") {
80
+ index = quotedEnd(text, index, "'", false, 'string');
81
+ out += '?';
82
+ continue;
83
+ }
84
+ if (char === '$') {
85
+ const tag = text.slice(index).match(DOLLAR_TAG)?.[0];
86
+ if (tag !== undefined) {
87
+ const close = text.indexOf(tag, index + tag.length);
88
+ if (close < 0)
89
+ throw new SqlLiteralError('it has an unterminated dollar-quoted string');
90
+ index = close + tag.length;
91
+ out += '?';
92
+ continue;
93
+ }
94
+ // A parameter ($1) never appears in a definition; copy it whole.
95
+ const parameter = text.slice(index).match(/^\$[0-9]+/u)?.[0] ?? '$';
96
+ out += parameter;
97
+ index += parameter.length;
98
+ continue;
99
+ }
100
+ if (IDENT_START.test(char)) {
101
+ let end = index + 1;
102
+ while (end < text.length && IDENT_PART.test(text[end]))
103
+ end += 1;
104
+ const word = text.slice(index, end);
105
+ const upper = word.toUpperCase();
106
+ if (text[end] === "'" && (upper === 'E' || upper === 'B' || upper === 'X' || upper === 'N')) {
107
+ index = quotedEnd(text, end, "'", upper === 'E', 'string');
108
+ out += '?';
109
+ continue;
110
+ }
111
+ if (upper === 'U' && text[end] === '&' && (text[end + 1] === "'" || text[end + 1] === '"')) {
112
+ const quote = text[end + 1];
113
+ const close = quotedEnd(text, end + 1, quote, false, quote === '"' ? 'quoted identifier' : 'string');
114
+ out += quote === '"' ? text.slice(index, close) : '?';
115
+ index = close;
116
+ continue;
117
+ }
118
+ out += word;
119
+ index = end;
120
+ continue;
121
+ }
122
+ const previous = index === 0 ? '' : text[index - 1];
123
+ const startsNumber = DIGIT.test(char)
124
+ || (char === '.' && DIGIT.test(next) && !(IDENT_PART.test(previous) || previous === '"' || previous === ')' || previous === ']'));
125
+ if (startsNumber) {
126
+ const number = text.slice(index).match(NUMBER)?.[0];
127
+ if (number === undefined)
128
+ throw new SqlLiteralError(`it has a malformed number at character ${index + 1}`);
129
+ const after = text[index + number.length] ?? '';
130
+ if (IDENT_PART.test(after))
131
+ throw new SqlLiteralError(`it has a number followed by letters at character ${index + 1}`);
132
+ out += '?';
133
+ index += number.length;
134
+ continue;
135
+ }
136
+ out += char;
137
+ index += 1;
138
+ }
139
+ return out;
140
+ }
@@ -0,0 +1,362 @@
1
+ /**
2
+ * SQL for `haystack db profile`. Every statement here is a single SELECT (or
3
+ * WITH ... SELECT) that returns aggregates: counts, fractions, lengths,
4
+ * quantiles. The only strings a statement may return are ones that already
5
+ * passed an owner threshold in the same statement (a `HAVING` on the number of
6
+ * distinct owners), or values the schema itself declares (enum labels and
7
+ * CHECK value lists, filtered with `= ANY($1)`).
8
+ *
9
+ * Identifiers come from the catalog and are always quoted; numbers embedded in
10
+ * SQL are produced here from validated inputs. Thresholds travel as parameters.
11
+ */
12
+ import { averageWordsSql, fourWordsPredicate, KEPT_VALUE_PERSONAL_SHAPES, shapeSql, VALUE_SHAPE_TESTS, } from './db-profile-privacy.js';
13
+ export function quoteIdent(name) {
14
+ if (name.length === 0 || name.includes('\u0000'))
15
+ throw new Error(`Refusing to quote an empty or NUL-containing identifier.`);
16
+ return `"${name.replace(/"/gu, '""')}"`;
17
+ }
18
+ export function qualified(ref) {
19
+ return `${quoteIdent(ref.schema)}.${quoteIdent(ref.table)}`;
20
+ }
21
+ function sqlNumber(value, label) {
22
+ if (!Number.isFinite(value))
23
+ throw new Error(`${label} must be finite to embed in SQL.`);
24
+ return String(value);
25
+ }
26
+ /** Seed for TABLESAMPLE ... REPEATABLE, so every query in a run (and a re-run
27
+ * on unchanged data) reads the same sampled pages. */
28
+ export const SAMPLE_SEED = 20260924;
29
+ export const QUANTILE_POINTS = [0.05, 0.25, 0.5, 0.75, 0.95];
30
+ const QUANTILE_ARRAY = `ARRAY[${QUANTILE_POINTS.join(', ')}]::float8[]`;
31
+ function tableSample(samplePercent) {
32
+ if (samplePercent === null)
33
+ return '';
34
+ if (!(samplePercent > 0 && samplePercent < 100))
35
+ throw new Error(`Sample percentage ${samplePercent} is outside (0, 100).`);
36
+ return ` TABLESAMPLE SYSTEM (${sqlNumber(Number(samplePercent.toFixed(6)), 'sample percentage')}) REPEATABLE (${SAMPLE_SEED})`;
37
+ }
38
+ function columnTuple(alias, columns) {
39
+ if (columns.length === 1)
40
+ return `${alias}.${quoteIdent(columns[0])}`;
41
+ // A multi-column key with any null part references nothing, so it owns nothing.
42
+ const nullCheck = columns.map(column => `${alias}.${quoteIdent(column)} IS NULL`).join(' OR ');
43
+ return `CASE WHEN ${nullCheck} THEN NULL ELSE ROW(${columns.map(column => `${alias}.${quoteIdent(column)}`).join(', ')}) END`;
44
+ }
45
+ function joinOn(childAlias, parentAlias, edge) {
46
+ return edge.columns
47
+ .map((column, index) => `${parentAlias}.${quoteIdent(edge.referencedColumns[index])} = ${childAlias}.${quoteIdent(column)}`)
48
+ .join(' AND ');
49
+ }
50
+ /** The table alone as `t0` (sampled or not). ONLY: a table with
51
+ * legacy-inheritance children is measured on its own rows. */
52
+ function tableFrom(table, samplePercent) {
53
+ return `ONLY ${qualified(table)} AS t0${tableSample(samplePercent)}`;
54
+ }
55
+ /** `FROM` clause of a table (sampled or not) joined along each owner path. */
56
+ export function scanSql(source) {
57
+ let from = tableFrom(source.table, source.samplePercent);
58
+ const owners = [];
59
+ if (source.reach.kind === 'paths') {
60
+ source.reach.paths.forEach((path, pathIndex) => {
61
+ let childAlias = 't0';
62
+ // Join every table on the path, the owner table included, and count the
63
+ // owner table's own key: a key no owner row holds (an orphan under a
64
+ // NOT VALID foreign key) is null, so it is no owner. Each join is on a
65
+ // primary or unique key, so it never repeats a row.
66
+ path.forEach((edge, hopIndex) => {
67
+ const alias = `p${pathIndex}h${hopIndex}`;
68
+ from += ` LEFT JOIN ${qualified(edge.to)} AS ${alias} ON ${joinOn(childAlias, alias, edge)}`;
69
+ childAlias = alias;
70
+ });
71
+ owners.push(columnTuple(childAlias, path[path.length - 1].referencedColumns));
72
+ });
73
+ }
74
+ return { from, owners };
75
+ }
76
+ /** Distinct owners among the rows selected by `filter` (all rows when null).
77
+ * With no owner expressions (rows are owners, or this is the owner table) it
78
+ * counts rows. A table the owner table is unreachable from also has none, and
79
+ * the profiler withholds every owner-gated number for it. */
80
+ function ownerCount(scan, filter, ownerRefs = scan.owners) {
81
+ const filterSql = filter === null ? '' : ` FILTER (WHERE ${filter})`;
82
+ if (ownerRefs.length === 0)
83
+ return `count(*)${filterSql}`;
84
+ const counts = ownerRefs.map(owner => `count(DISTINCT ${owner})${filterSql}`);
85
+ return counts.length === 1 ? counts[0] : `LEAST(${counts.join(', ')})`;
86
+ }
87
+ export function columnRef(name) {
88
+ return `t0.${quoteIdent(name)}`;
89
+ }
90
+ export function hasLengthQuantiles(family) {
91
+ return family === 'text' || family === 'json' || family === 'binary';
92
+ }
93
+ export function hasNumberQuantiles(family) {
94
+ return family === 'number' || family === 'time';
95
+ }
96
+ function byteLengthSql(column) {
97
+ const ref = columnRef(column.name);
98
+ return column.family === 'binary' ? `octet_length(${ref})` : `octet_length(${ref}::text)`;
99
+ }
100
+ function numberSql(column) {
101
+ const ref = columnRef(column.name);
102
+ if (column.family === 'time')
103
+ return `extract(epoch FROM ${ref})::float8`;
104
+ if (column.typname === 'money')
105
+ return `${ref}::numeric::float8`;
106
+ return `${ref}::float8`;
107
+ }
108
+ const NON_FINITE = `('Infinity'::float8, '-Infinity'::float8, 'NaN'::float8)`;
109
+ const NUMERIC_TEXT = `'^[[:space:]]*[-+]?[0-9]+(\\.[0-9]+)?[[:space:]]*$'`;
110
+ const ISO_DATE_TEXT = `'^[0-9]{4}-[0-9]{2}-[0-9]{2}'`;
111
+ const BAD_CHARACTERS = `'[\\x01-\\x08\\x0b\\x0c\\x0e-\\x1f\\x7f]'`;
112
+ export const JSON_ROOT_TYPES = ['object', 'array', 'string', 'number', 'boolean', 'null'];
113
+ /** What `count(DISTINCT ...)` counts for a column read in full. A type with
114
+ * no default btree operator class (varchar has none of its own) is counted by
115
+ * its text, which is exact for text and never fewer than the true count for
116
+ * numbers, since the count decides category columns. Other such types (JSON,
117
+ * geometry) are counted by a 64-bit hash of their text. */
118
+ function exactDistinctValue(column) {
119
+ const ref = columnRef(column.name);
120
+ if (column.sortable)
121
+ return ref;
122
+ if (column.family === 'text' || column.family === 'number')
123
+ return `${ref}::text`;
124
+ return `hashtextextended(${ref}::text, 0)`;
125
+ }
126
+ /**
127
+ * One pass over the table (or its sample) computing, for every column, the
128
+ * numbers the classification and the profile need. Result keys are
129
+ * `c<index>_<measure>`.
130
+ */
131
+ export function statsQuery(scan, columns, exactDistinct) {
132
+ const parts = ['count(*)::float8 AS total'];
133
+ columns.forEach(({ index, column }) => {
134
+ const ref = columnRef(column.name);
135
+ const key = `c${index}`;
136
+ parts.push(`count(${ref})::float8 AS ${key}_nn`);
137
+ if (exactDistinct)
138
+ parts.push(`count(DISTINCT ${exactDistinctValue(column)})::float8 AS ${key}_nd`);
139
+ parts.push(`avg(pg_column_size(${ref}))::float8 AS ${key}_aw`);
140
+ parts.push(`(${ownerCount(scan, `${ref} IS NOT NULL`)})::float8 AS ${key}_own`);
141
+ if (hasLengthQuantiles(column.family)) {
142
+ parts.push(`percentile_cont(${QUANTILE_ARRAY}) WITHIN GROUP (ORDER BY ${byteLengthSql(column)}) AS ${key}_lq`);
143
+ }
144
+ if (hasNumberQuantiles(column.family)) {
145
+ const value = numberSql(column);
146
+ parts.push(`percentile_cont(${QUANTILE_ARRAY}) WITHIN GROUP (ORDER BY ${value}) FILTER (WHERE ${value} NOT IN ${NON_FINITE}) AS ${key}_nq`);
147
+ parts.push(`count(*) FILTER (WHERE ${value} IN ${NON_FINITE})::float8 AS ${key}_nonfinite`);
148
+ }
149
+ if (column.family === 'text') {
150
+ const value = `${ref}::text`;
151
+ for (const test of VALUE_SHAPE_TESTS) {
152
+ parts.push(`count(*) FILTER (WHERE ${test.predicate(value)})::float8 AS ${key}_t_${test.id.replace('-', '_')}`);
153
+ }
154
+ parts.push(`${averageWordsSql(value)}::float8 AS ${key}_words`);
155
+ parts.push(`count(*) FILTER (WHERE ${fourWordsPredicate(value)})::float8 AS ${key}_w4`);
156
+ parts.push(`count(*) FILTER (WHERE ${value} ~ ${NUMERIC_TEXT})::float8 AS ${key}_numeric`);
157
+ parts.push(`count(*) FILTER (WHERE ${value} ~ ${ISO_DATE_TEXT})::float8 AS ${key}_date`);
158
+ parts.push(`count(*) FILTER (WHERE strpos(${value}, chr(65533)) > 0 OR ${value} ~ ${BAD_CHARACTERS})::float8 AS ${key}_bad`);
159
+ }
160
+ if (column.family === 'json') {
161
+ for (const type of JSON_ROOT_TYPES) {
162
+ parts.push(`count(*) FILTER (WHERE jsonb_typeof(${ref}::jsonb) = '${type}')::float8 AS ${key}_j_${type}`);
163
+ }
164
+ }
165
+ if (column.family === 'boolean') {
166
+ parts.push(`count(*) FILTER (WHERE ${ref}::boolean)::float8 AS ${key}_true`);
167
+ }
168
+ });
169
+ return `SELECT ${parts.join(',\n ')}\nFROM ${scan.from}`;
170
+ }
171
+ /** Select-list items `statsQuery` emits for a column. */
172
+ export function statsItemCount(column, exactDistinct) {
173
+ return 3 + (exactDistinct ? 1 : 0)
174
+ + (hasLengthQuantiles(column.family) ? 1 : 0)
175
+ + (hasNumberQuantiles(column.family) ? 2 : 0)
176
+ + (column.family === 'text' ? VALUE_SHAPE_TESTS.length + 5 : 0)
177
+ + (column.family === 'json' ? JSON_ROOT_TYPES.length : 0)
178
+ + (column.family === 'boolean' ? 1 : 0);
179
+ }
180
+ /** Postgres allows 1,664 select-list entries; wide tables are measured in
181
+ * several passes of at most this many (the same sample pages each time). */
182
+ export const STATS_ITEMS_PER_QUERY = 1_000;
183
+ export function statsBatches(columns, exactDistinct) {
184
+ const batches = [[]];
185
+ let items = 1;
186
+ columns.forEach((column, index) => {
187
+ const needed = statsItemCount(column, exactDistinct);
188
+ if (items + needed > STATS_ITEMS_PER_QUERY && batches[batches.length - 1].length > 0) {
189
+ batches.push([]);
190
+ items = 1;
191
+ }
192
+ batches[batches.length - 1].push({ index, column });
193
+ items += needed;
194
+ });
195
+ return batches;
196
+ }
197
+ /**
198
+ * For one column of a sampled table: how many distinct values the sample
199
+ * holds on exactly `pages` sampled pages, one row per page count that occurs
200
+ * (see `db-profile-distinct.ts`). Only counts leave the database. A value is
201
+ * compared as itself, or by a 64-bit hash of its text when its type has no
202
+ * default btree operator class; a hash collision can only lower the counts.
203
+ */
204
+ export function sampleDistinctQuery(table, samplePercent, column) {
205
+ const ref = columnRef(column.name);
206
+ const value = column.sortable ? ref : `hashtextextended(${ref}::text, 0)`;
207
+ return `SELECT f.pages::float8 AS pages, count(*)::float8 AS value_count
208
+ FROM (SELECT p.v, count(*) AS pages
209
+ FROM (SELECT DISTINCT ${value} AS v, (t0.ctid::text::point)[0] AS page
210
+ FROM ${tableFrom(table, samplePercent)}
211
+ WHERE ${ref} IS NOT NULL) p
212
+ GROUP BY p.v) f
213
+ GROUP BY f.pages
214
+ ORDER BY f.pages`;
215
+ }
216
+ /**
217
+ * Exact distinct non-null counts of several columns over the WHOLE table, in
218
+ * one read: each grouping set holds one column's distinct values, and in the
219
+ * other sets' rows that column is null, so `count(x.cN)` counts exactly its
220
+ * own distinct non-null values. Result keys are `c<index>_d`.
221
+ */
222
+ export function categoryDistinctQuery(table, columns) {
223
+ if (columns.length === 0)
224
+ throw new Error('categoryDistinctQuery needs at least one column.');
225
+ const counts = columns.map(({ index }) => `count(x.c${index})::float8 AS c${index}_d`).join(',\n ');
226
+ const select = columns.map(({ index, column }) => `${columnRef(column.name)} AS c${index}`).join(', ');
227
+ const sets = columns.map(({ column }) => `(${columnRef(column.name)})`).join(', ');
228
+ return `SELECT ${counts}
229
+ FROM (SELECT ${select}
230
+ FROM ${tableFrom(table, null)}
231
+ GROUP BY GROUPING SETS (${sets})) x`;
232
+ }
233
+ /** Rows whose byte length exceeds a per-column threshold; one pass per table. */
234
+ export function oversizeQuery(scan, checks) {
235
+ const parts = checks.map(({ index, column, thresholdBytes }) => `count(*) FILTER (WHERE ${byteLengthSql(column)} > ${sqlNumber(Math.round(thresholdBytes), 'oversize threshold')})::float8 AS c${index}_big`);
236
+ return `SELECT ${parts.join(',\n ')}\nFROM ${scan.from}`;
237
+ }
238
+ /* ------------------------------------------------------------ values */
239
+ /** Data values of a category column, each returned only when it is shared by
240
+ * at least $1 owners. */
241
+ export function categoryValuesQuery(scan, column) {
242
+ const ref = columnRef(column);
243
+ return `SELECT ${ref}::text AS value, count(*)::float8 AS rows
244
+ FROM ${scan.from}
245
+ WHERE ${ref} IS NOT NULL
246
+ GROUP BY ${ref}
247
+ HAVING ${ownerCount(scan, null)} >= $1
248
+ ORDER BY rows DESC, value`;
249
+ }
250
+ /** Row counts for values the schema declares ($1, a text array). */
251
+ export function schemaValuesQuery(scan, column) {
252
+ const ref = columnRef(column);
253
+ return `SELECT ${ref}::text AS value, count(*)::float8 AS rows
254
+ FROM ${scan.from}
255
+ WHERE ${ref}::text = ANY($1::text[])
256
+ GROUP BY 1`;
257
+ }
258
+ function ownerColumns(scan) {
259
+ const select = scan.owners.map((owner, index) => `, ${owner} AS o${index}`).join('');
260
+ return { select, refs: scan.owners.map((_, index) => `x.o${index}`) };
261
+ }
262
+ /** Character-class shapes shared by at least $1 owners. */
263
+ export function shapesQuery(scan, column) {
264
+ const ref = columnRef(column);
265
+ const owners = ownerColumns(scan);
266
+ return `SELECT x.shape, count(*)::float8 AS rows
267
+ FROM (SELECT ${shapeSql(`${ref}::text`)} AS shape${owners.select}
268
+ FROM ${scan.from}
269
+ WHERE ${ref} IS NOT NULL) x
270
+ GROUP BY x.shape
271
+ HAVING ${ownerCount(scan, null, owners.refs)} >= $1
272
+ ORDER BY rows DESC, x.shape`;
273
+ }
274
+ /** Deepest JSON key path followed; deeper structure is summarised by its parent. */
275
+ export const JSON_KEY_MAX_DEPTH = 6;
276
+ /**
277
+ * JSON key paths (objects as `a.b`, array elements as `a[]`) shared by at
278
+ * least $1 owners, with the JSON types seen at each path and how many rows
279
+ * carry each type. A path is never returned, whatever its owner count, when
280
+ * any key on it (tested one key at a time, since a key may itself hold a
281
+ * dot) or the whole path has a personal value shape (KEPT_VALUE_PERSONAL_SHAPES:
282
+ * an email, phone number, IP address, government id, card number or secret):
283
+ * `15551234567.verified` is withheld with `15551234567`.
284
+ */
285
+ export function jsonKeysQuery(scan, column) {
286
+ const ref = columnRef(column);
287
+ const owners = ownerColumns(scan);
288
+ const ownerNames = scan.owners.map((_, index) => `o${index}`);
289
+ const carry = ownerNames.map(name => `, w.${name}`).join('');
290
+ const baseOwners = ownerNames.map(name => `, b.${name}`).join('');
291
+ const walkOwners = ownerNames.map(name => `, ${name}`).join('');
292
+ const pathOwnerCount = ownerNames.length === 0
293
+ ? 'count(DISTINCT w.rid)'
294
+ : ownerNames.length === 1
295
+ ? `count(DISTINCT w.${ownerNames[0]})`
296
+ : `LEAST(${ownerNames.map(name => `count(DISTINCT w.${name})`).join(', ')})`;
297
+ const personal = (value) => VALUE_SHAPE_TESTS
298
+ .filter(test => KEPT_VALUE_PERSONAL_SHAPES.some(shape => shape.id === test.id))
299
+ .map(test => test.predicate(value))
300
+ .join(' OR ');
301
+ return `WITH RECURSIVE base AS (
302
+ SELECT t0.ctid AS rid${owners.select}, ${ref}::jsonb AS doc
303
+ FROM ${scan.from}
304
+ WHERE ${ref} IS NOT NULL
305
+ ), walk(rid${walkOwners}, path, value, depth, personal) AS (
306
+ SELECT b.rid${baseOwners}, ''::text, b.doc, 0, false FROM base b
307
+ UNION ALL
308
+ SELECT w.rid${carry}, child.path, child.value, w.depth + 1, w.personal OR child.personal
309
+ FROM walk w CROSS JOIN LATERAL (
310
+ SELECT CASE WHEN w.path = '' THEN e.key ELSE w.path || '.' || e.key END AS path, e.value, (${personal('e.key')}) AS personal
311
+ FROM jsonb_each(CASE WHEN jsonb_typeof(w.value) = 'object' THEN w.value ELSE '{}'::jsonb END) e
312
+ UNION ALL
313
+ SELECT w.path || '[]', a.value, false
314
+ FROM jsonb_array_elements(CASE WHEN jsonb_typeof(w.value) = 'array' THEN w.value ELSE '[]'::jsonb END) a
315
+ ) child
316
+ WHERE w.depth < ${JSON_KEY_MAX_DEPTH}
317
+ ), clean AS (
318
+ SELECT w.* FROM walk w
319
+ WHERE w.depth > 0 AND NOT w.personal AND NOT (${personal('w.path')})
320
+ ), kept AS (
321
+ SELECT w.path, count(DISTINCT w.rid)::float8 AS present
322
+ FROM clean w
323
+ GROUP BY w.path
324
+ HAVING ${pathOwnerCount} >= $1
325
+ )
326
+ SELECT k.path, k.present, jsonb_typeof(w.value) AS type, count(DISTINCT w.rid)::float8 AS rows
327
+ FROM clean w JOIN kept k ON k.path = w.path
328
+ GROUP BY k.path, k.present, jsonb_typeof(w.value)
329
+ ORDER BY k.path, type`;
330
+ }
331
+ /* ------------------------------------------------------------ foreign keys */
332
+ /** Children per parent over the (sampled) parent table, counting parents
333
+ * with no children, as quantiles; and how many of those parents have at least
334
+ * one child, which for a validated single-column key is the number of
335
+ * distinct values the child's column holds (exactly, when the parent is read
336
+ * in full). */
337
+ export function childrenPerParentQuery(edge, parentSamplePercent) {
338
+ const parentKeyNotNull = edge.referencedColumns.map(column => `p.${quoteIdent(column)} IS NOT NULL`).join(' AND ');
339
+ const groupBy = edge.referencedColumns.map(column => `p.${quoteIdent(column)}`).join(', ');
340
+ const on = edge.columns
341
+ .map((column, index) => `c.${quoteIdent(column)} = p.${quoteIdent(edge.referencedColumns[index])}`)
342
+ .join(' AND ');
343
+ return `SELECT percentile_cont(${QUANTILE_ARRAY}) WITHIN GROUP (ORDER BY x.children) AS q, count(*)::float8 AS parents,
344
+ count(*) FILTER (WHERE x.children > 0)::float8 AS referenced
345
+ FROM (SELECT count(c.ctid)::float8 AS children
346
+ FROM ${qualified(edge.to)} AS p${tableSample(parentSamplePercent)}
347
+ LEFT JOIN ONLY ${qualified(edge.from)} AS c ON ${on}
348
+ WHERE ${parentKeyNotNull}
349
+ GROUP BY ${groupBy}) x`;
350
+ }
351
+ /** Share of (sampled) child rows whose non-null key has no parent. Only run
352
+ * for NOT VALID constraints; a validated constraint guarantees zero. */
353
+ export function orphanQuery(edge, childSamplePercent) {
354
+ const childKeyNotNull = edge.columns.map(column => `c.${quoteIdent(column)} IS NOT NULL`).join(' AND ');
355
+ const match = edge.columns
356
+ .map((column, index) => `p.${quoteIdent(edge.referencedColumns[index])} = c.${quoteIdent(column)}`)
357
+ .join(' AND ');
358
+ return `SELECT count(*)::float8 AS children,
359
+ count(*) FILTER (WHERE NOT EXISTS (SELECT 1 FROM ${qualified(edge.to)} AS p WHERE ${match}))::float8 AS orphans
360
+ FROM ONLY ${qualified(edge.from)} AS c${tableSample(childSamplePercent)}
361
+ WHERE ${childKeyNotNull}`;
362
+ }
@@ -0,0 +1,88 @@
1
+ /**
2
+ * `haystack db profile upload <file> --repo owner/repo [--dependency <id>]` —
3
+ * send a reviewed profile to `POST /api/agent/cloud-verifier/database-profiles`
4
+ * as a `DatabaseProfileUploadRequestV1`, authenticated with the caller's
5
+ * Haystack login like every other CLI call, and print the stored
6
+ * `DatabaseProfileUploadV1`. Without --dependency the profile is the
7
+ * repository's: onboarding builds the run map's Postgres stand-in from it when
8
+ * the run map declares exactly one, whatever id it gave that dependency.
9
+ *
10
+ * The file is parsed with the same strict parser `show` uses, so what is sent
11
+ * is exactly what `show` printed.
12
+ */
13
+ import chalk from 'chalk';
14
+ import { DATABASE_PROFILE_UPLOAD_REQUEST_VERSION, parseDatabaseProfileUpload, } from './db-profile-contract.js';
15
+ import { readDatabaseProfileFile } from './db-profile-show.js';
16
+ import { resolveAuthContext } from '../utils/auth.js';
17
+ import { classifyHttpError, haystackApiUrl } from '../utils/haystack-api.js';
18
+ export const DATABASE_PROFILES_PATH = '/api/agent/cloud-verifier/database-profiles';
19
+ /** GitHub's owner and repository name alphabets. */
20
+ const REPOSITORY = /^([A-Za-z0-9](?:[A-Za-z0-9-]{0,38}))\/([A-Za-z0-9_.-]{1,100})$/u;
21
+ /** Run-map dependency ids are stable kebab-case ids (the same check as
22
+ * `requireId` in verifier/front-half/src/derivations/surface-map.ts). */
23
+ const DEPENDENCY_ID = /^[a-z0-9]+(?:-[a-z0-9]+)*$/u;
24
+ export function buildDatabaseProfileUploadRequest(file, options) {
25
+ if (options.repo === undefined)
26
+ throw new Error('--repo is required: the GitHub repository this database belongs to, as owner/repo.');
27
+ const repository = options.repo.match(REPOSITORY);
28
+ if (!repository || repository[2] === '.' || repository[2] === '..') {
29
+ throw new Error(`--repo must be a GitHub owner/repo (got "${options.repo}").`);
30
+ }
31
+ if (options.dependency !== undefined && !DEPENDENCY_ID.test(options.dependency)) {
32
+ throw new Error(`--dependency must be a run-map dependency id in kebab-case, e.g. app-postgres (got "${options.dependency}").`);
33
+ }
34
+ const profile = readDatabaseProfileFile(file);
35
+ return {
36
+ owner: repository[1],
37
+ name: repository[2],
38
+ request: {
39
+ version: DATABASE_PROFILE_UPLOAD_REQUEST_VERSION,
40
+ repository: options.repo,
41
+ ...(options.dependency === undefined ? {} : { dependencyId: options.dependency }),
42
+ profile,
43
+ },
44
+ };
45
+ }
46
+ export async function uploadDatabaseProfile(request, token) {
47
+ const response = await fetch(haystackApiUrl(DATABASE_PROFILES_PATH), {
48
+ method: 'POST',
49
+ headers: {
50
+ Authorization: `Bearer ${token}`,
51
+ Accept: 'application/json',
52
+ 'User-Agent': 'Haystack-CLI',
53
+ 'Content-Type': 'application/json',
54
+ },
55
+ body: JSON.stringify(request),
56
+ signal: AbortSignal.timeout(120_000),
57
+ });
58
+ if (!response.ok)
59
+ throw await classifyHttpError(response, `Haystack API ${DATABASE_PROFILES_PATH}`);
60
+ let body;
61
+ try {
62
+ body = await response.json();
63
+ }
64
+ catch (error) {
65
+ throw new Error(`Haystack API ${DATABASE_PROFILES_PATH} returned a body that is not JSON: ${error instanceof Error ? error.message : String(error)}`);
66
+ }
67
+ const upload = parseDatabaseProfileUpload(body);
68
+ const expected = request.dependencyId ?? null;
69
+ if (upload.dependencyId !== expected) {
70
+ const describe = (id) => (id === null ? 'the repository' : `dependency ${id}`);
71
+ throw new Error(`Haystack API ${DATABASE_PROFILES_PATH} stored the profile for ${describe(upload.dependencyId)}, not ${describe(expected)}.`);
72
+ }
73
+ return upload;
74
+ }
75
+ export async function databaseProfileUploadCommand(file, options) {
76
+ const { owner, name, request } = buildDatabaseProfileUploadRequest(file, options);
77
+ const auth = await resolveAuthContext({ preferredLogin: options.account, owner, repo: name });
78
+ const upload = await uploadDatabaseProfile(request, auth.token);
79
+ if (upload.dependencyId === null) {
80
+ console.log(chalk.green(`Uploaded ${file} for ${request.repository}.`));
81
+ console.log(' Onboarding builds the stand-in database from it when the run map declares exactly one Postgres dependency;');
82
+ console.log(' otherwise haystack verify hosted start says which ids it declares, and you upload again with --dependency <id>.');
83
+ }
84
+ else {
85
+ console.log(chalk.green(`Uploaded ${file} for ${request.repository}, dependency ${upload.dependencyId}.`));
86
+ }
87
+ console.log(` stored as ${upload.profile.artifactId} (sha256 ${upload.profile.sha256}) at ${upload.uploadedAt}`);
88
+ }