@haystackeditor/cli 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +208 -26
- package/dist/commands/case-batch-contract.js +2 -2
- package/dist/commands/case-batch.js +7 -17
- package/dist/commands/crawl-contract.js +3 -0
- package/dist/commands/db-profile-contract.js +419 -0
- package/dist/commands/db-profile-database.js +289 -0
- package/dist/commands/db-profile-distinct.js +97 -0
- package/dist/commands/db-profile-owners.js +94 -0
- package/dist/commands/db-profile-privacy.js +450 -0
- package/dist/commands/db-profile-show.js +218 -0
- package/dist/commands/db-profile-sql-literals.js +140 -0
- package/dist/commands/db-profile-sql.js +362 -0
- package/dist/commands/db-profile-upload.js +88 -0
- package/dist/commands/db-profile.js +1020 -0
- package/dist/commands/fleet-policy-contract.js +28 -0
- package/dist/commands/fleet-policy.js +305 -0
- package/dist/commands/precompute-delivery-worker.js +17 -3
- package/dist/commands/precompute-delivery.js +16 -6
- package/dist/commands/verify-hosted.js +20 -0
- package/dist/commands/verify-precompute.js +140 -95
- package/dist/commands/verify.js +523 -52
- package/dist/index.js +173 -21
- package/dist/schema.js +1 -0
- package/dist/triage/astra.js +5 -2
- package/dist/triage/runner.js +1 -1
- package/dist/utils/verify-base.js +31 -0
- package/package.json +4 -1
- package/schemas/verify.v1.json +217 -0
- package/dist/commands/combination-search-hook-contract.js +0 -1
- package/dist/commands/combination-search-hook.js +0 -139
|
@@ -0,0 +1,1020 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `haystack db profile` — a read-only, PII-free profile of a customer's
|
|
3
|
+
* Postgres database, run by the customer inside their own environment.
|
|
4
|
+
*
|
|
5
|
+
* The database computes every number; the profiler never selects a raw row.
|
|
6
|
+
* The only value strings that reach this process are enum labels and CHECK
|
|
7
|
+
* value lists (schema, not data) and category values, shapes and JSON key
|
|
8
|
+
* paths that a `HAVING` clause has already shown to be shared by at least
|
|
9
|
+
* `categoryMinOwners` owners. Personal columns are decided before any value
|
|
10
|
+
* query runs, and a personal column never has one. Index definitions also
|
|
11
|
+
* arrive as written, and every literal in them (a partial-index predicate can
|
|
12
|
+
* name a value) is replaced with `?` before the definition is kept.
|
|
13
|
+
*
|
|
14
|
+
* It runs on a primary or on a read replica (hot standby). A replica keeps no
|
|
15
|
+
* dead-row counts, so there each table's dead rows are reported unavailable.
|
|
16
|
+
*
|
|
17
|
+
* Distinct counts are exact for a table read in full. A table larger than
|
|
18
|
+
* --scan-max-mb is sampled, but a column whose kind only its distinct count
|
|
19
|
+
* can settle (category or not) is still counted exactly, over the whole
|
|
20
|
+
* table, unless the sample alone already holds too many values; a column that
|
|
21
|
+
* is the whole of a validated foreign key is counted through the parents it
|
|
22
|
+
* references; every other sampled column is estimated from the sample (see
|
|
23
|
+
* `db-profile-distinct.ts`). The planner's statistics are never used for them.
|
|
24
|
+
*
|
|
25
|
+
* See `db-profile-privacy.ts` for the personal rules, `db-profile-sql.ts` for
|
|
26
|
+
* every statement, and `db-profile-database.ts` for the read-only guarantees.
|
|
27
|
+
*/
|
|
28
|
+
import { readFileSync, renameSync, writeFileSync } from 'node:fs';
|
|
29
|
+
import { dirname, resolve } from 'node:path';
|
|
30
|
+
import { fileURLToPath } from 'node:url';
|
|
31
|
+
import chalk from 'chalk';
|
|
32
|
+
import { DATABASE_PROFILE_VERSION, parseDatabaseProfile, } from './db-profile-contract.js';
|
|
33
|
+
import { ReadOnlyDatabase, runBounded } from './db-profile-database.js';
|
|
34
|
+
import { redactSqlLiterals, SqlLiteralError } from './db-profile-sql-literals.js';
|
|
35
|
+
import { describeReach, ownerReach, tableKey, } from './db-profile-owners.js';
|
|
36
|
+
import { checkConstraintValues, classifyColumnFacts, decideByDistinct, formatShare as pct, jsonPathPersonalShape, keptValueProblem, typeFamily, VALUE_SHAPE_TESTS, } from './db-profile-privacy.js';
|
|
37
|
+
import { estimateDistinct } from './db-profile-distinct.js';
|
|
38
|
+
import { categoryDistinctQuery, categoryValuesQuery, childrenPerParentQuery, hasLengthQuantiles, hasNumberQuantiles, JSON_ROOT_TYPES, jsonKeysQuery, orphanQuery, oversizeQuery, QUANTILE_POINTS, SAMPLE_SEED, sampleDistinctQuery, scanSql, schemaValuesQuery, shapesQuery, statsBatches, statsQuery, } from './db-profile-sql.js';
|
|
39
|
+
/* ------------------------------------------------------------ knobs */
|
|
40
|
+
/** Every knob, its default and its bounds. The README documents what turning
|
|
41
|
+
* each one up or down does; keep the two in step. */
|
|
42
|
+
export const DB_PROFILE_KNOBS = {
|
|
43
|
+
categoryMaxDistinct: { flag: '--category-max-distinct', default: 50, min: 1, max: 1_000 },
|
|
44
|
+
categoryMinOwners: { flag: '--category-min-owners', default: 50, min: 5, max: 1_000_000 },
|
|
45
|
+
concurrency: { flag: '--concurrency', default: 2, min: 1, max: 8 },
|
|
46
|
+
statementTimeoutSeconds: { flag: '--statement-timeout', default: 120, min: 1, max: 3_600 },
|
|
47
|
+
scanMaxMb: { flag: '--scan-max-mb', default: 256, min: 0.01, max: 1_048_576 },
|
|
48
|
+
};
|
|
49
|
+
/** Oldest server this profiler supports (hashtextextended, row_security_active). */
|
|
50
|
+
export const MIN_SERVER_VERSION_NUM = 120_000;
|
|
51
|
+
/** Planner and storage settings only. Never connection, authentication, SSL,
|
|
52
|
+
* logging or file-location settings. `autovacuum*` and `enable_*` (planner
|
|
53
|
+
* method switches) are included by prefix. */
|
|
54
|
+
export const SETTINGS_ALLOWLIST = [
|
|
55
|
+
'shared_buffers', 'work_mem', 'maintenance_work_mem', 'effective_cache_size', 'temp_buffers', 'hash_mem_multiplier',
|
|
56
|
+
'random_page_cost', 'seq_page_cost', 'cpu_tuple_cost', 'cpu_index_tuple_cost', 'cpu_operator_cost',
|
|
57
|
+
'parallel_setup_cost', 'parallel_tuple_cost', 'min_parallel_table_scan_size', 'min_parallel_index_scan_size',
|
|
58
|
+
'effective_io_concurrency', 'maintenance_io_concurrency',
|
|
59
|
+
'default_statistics_target', 'constraint_exclusion', 'cursor_tuple_fraction', 'from_collapse_limit',
|
|
60
|
+
'join_collapse_limit', 'geqo', 'geqo_threshold', 'jit', 'jit_above_cost', 'jit_inline_above_cost',
|
|
61
|
+
'jit_optimize_above_cost', 'plan_cache_mode', 'recursive_worktable_factor',
|
|
62
|
+
'max_parallel_workers_per_gather', 'max_parallel_workers', 'max_parallel_maintenance_workers', 'max_worker_processes',
|
|
63
|
+
'vacuum_cost_delay', 'vacuum_cost_limit', 'vacuum_cost_page_hit', 'vacuum_cost_page_miss', 'vacuum_cost_page_dirty',
|
|
64
|
+
'vacuum_freeze_min_age', 'vacuum_freeze_table_age', 'vacuum_multixact_freeze_min_age',
|
|
65
|
+
'vacuum_multixact_freeze_table_age', 'default_toast_compression', 'block_size', 'server_encoding',
|
|
66
|
+
'checkpoint_timeout', 'checkpoint_completion_target', 'max_wal_size', 'min_wal_size', 'wal_buffers',
|
|
67
|
+
];
|
|
68
|
+
export const SETTINGS_PREFIXES = ['autovacuum', 'enable_'];
|
|
69
|
+
const ENV_NAME = /^[A-Za-z_][A-Za-z0-9_]*$/u;
|
|
70
|
+
function knob(value, spec, integer) {
|
|
71
|
+
if (value === undefined)
|
|
72
|
+
return spec.default;
|
|
73
|
+
const parsed = Number(value);
|
|
74
|
+
if (!Number.isFinite(parsed) || (integer && !Number.isInteger(parsed)) || parsed < spec.min || parsed > spec.max) {
|
|
75
|
+
throw new Error(`${spec.flag} must be ${integer ? 'an integer' : 'a number'} from ${spec.min} to ${spec.max} (got "${value}").`);
|
|
76
|
+
}
|
|
77
|
+
return parsed;
|
|
78
|
+
}
|
|
79
|
+
export function parseTableRef(value, flag) {
|
|
80
|
+
const match = value.match(/^([^.\s]+)\.([^.\s]+)$/u);
|
|
81
|
+
if (!match)
|
|
82
|
+
throw new Error(`${flag} must be schema.table, e.g. public.users (got "${value}").`);
|
|
83
|
+
return { schema: match[1], table: match[2] };
|
|
84
|
+
}
|
|
85
|
+
export function parseDatabaseProfileSettings(options) {
|
|
86
|
+
if (options.urlEnv === undefined || options.urlEnv.length === 0) {
|
|
87
|
+
throw new Error('--url-env is required: name the environment variable that holds the connection string, e.g. --url-env DATABASE_URL.');
|
|
88
|
+
}
|
|
89
|
+
if (!ENV_NAME.test(options.urlEnv)) {
|
|
90
|
+
// Never echo the value: it may be a connection string with a password.
|
|
91
|
+
throw new Error('--url-env takes the NAME of an environment variable (letters, digits, underscore), not a connection string.');
|
|
92
|
+
}
|
|
93
|
+
if (options.out === undefined || options.out.length === 0) {
|
|
94
|
+
throw new Error('--out is required: the file to write the profile to, e.g. --out profile.json.');
|
|
95
|
+
}
|
|
96
|
+
const schemas = options.schema === undefined || options.schema.length === 0 ? null : [...new Set(options.schema)];
|
|
97
|
+
for (const schema of schemas ?? []) {
|
|
98
|
+
if (schema.trim().length === 0)
|
|
99
|
+
throw new Error('--schema must not be empty.');
|
|
100
|
+
}
|
|
101
|
+
return {
|
|
102
|
+
urlEnv: options.urlEnv,
|
|
103
|
+
out: options.out,
|
|
104
|
+
ownerTable: options.ownerTable === undefined ? null : parseTableRef(options.ownerTable, '--owner-table'),
|
|
105
|
+
schemas,
|
|
106
|
+
categoryMaxDistinct: knob(options.categoryMaxDistinct, DB_PROFILE_KNOBS.categoryMaxDistinct, true),
|
|
107
|
+
categoryMinOwners: knob(options.categoryMinOwners, DB_PROFILE_KNOBS.categoryMinOwners, true),
|
|
108
|
+
concurrency: knob(options.concurrency, DB_PROFILE_KNOBS.concurrency, true),
|
|
109
|
+
statementTimeoutMs: knob(options.statementTimeout, DB_PROFILE_KNOBS.statementTimeoutSeconds, true) * 1000,
|
|
110
|
+
scanMaxBytes: Math.round(knob(options.scanMaxMb, DB_PROFILE_KNOBS.scanMaxMb, false) * 1024 * 1024),
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
export function readConnectionString(name, env = process.env) {
|
|
114
|
+
const value = env[name];
|
|
115
|
+
if (value === undefined || value.trim().length === 0) {
|
|
116
|
+
throw new Error(`Environment variable ${name} is not set. Export the connection string in it (the profiler never takes it as a flag, so it stays out of shell history and process lists).`);
|
|
117
|
+
}
|
|
118
|
+
if (!/^postgres(ql)?:\/\//u.test(value.trim())) {
|
|
119
|
+
throw new Error(`Environment variable ${name} must hold a postgres:// or postgresql:// connection URL.`);
|
|
120
|
+
}
|
|
121
|
+
return value.trim();
|
|
122
|
+
}
|
|
123
|
+
/* ------------------------------------------------------------ version */
|
|
124
|
+
export function profilerVersion() {
|
|
125
|
+
const packagePath = resolve(dirname(fileURLToPath(import.meta.url)), '../../package.json');
|
|
126
|
+
const manifest = JSON.parse(readFileSync(packagePath, 'utf8'));
|
|
127
|
+
if (typeof manifest.version !== 'string' || manifest.version.length === 0) {
|
|
128
|
+
throw new Error(`Could not read the CLI version from ${packagePath}.`);
|
|
129
|
+
}
|
|
130
|
+
return `haystack-cli ${manifest.version}`;
|
|
131
|
+
}
|
|
132
|
+
function num(row, key, label) {
|
|
133
|
+
const value = row[key];
|
|
134
|
+
const parsed = typeof value === 'number' ? value : typeof value === 'string' ? Number(value) : Number.NaN;
|
|
135
|
+
if (!Number.isFinite(parsed))
|
|
136
|
+
throw new Error(`${label}: the database returned no number for ${key}.`);
|
|
137
|
+
return parsed;
|
|
138
|
+
}
|
|
139
|
+
function numOrNull(row, key, label) {
|
|
140
|
+
return row[key] === null || row[key] === undefined ? null : num(row, key, label);
|
|
141
|
+
}
|
|
142
|
+
function str(row, key, label) {
|
|
143
|
+
const value = row[key];
|
|
144
|
+
if (typeof value !== 'string')
|
|
145
|
+
throw new Error(`${label}: the database returned no text for ${key}.`);
|
|
146
|
+
return value;
|
|
147
|
+
}
|
|
148
|
+
function strList(row, key, label) {
|
|
149
|
+
const value = row[key];
|
|
150
|
+
if (!Array.isArray(value) || value.some(item => typeof item !== 'string')) {
|
|
151
|
+
throw new Error(`${label}: the database returned no text list for ${key}.`);
|
|
152
|
+
}
|
|
153
|
+
return value;
|
|
154
|
+
}
|
|
155
|
+
function quantiles(row, key, label) {
|
|
156
|
+
const value = row[key];
|
|
157
|
+
if (value === null || value === undefined)
|
|
158
|
+
return null;
|
|
159
|
+
if (!Array.isArray(value) || value.length !== QUANTILE_POINTS.length) {
|
|
160
|
+
throw new Error(`${label}: the database returned malformed quantiles for ${key}.`);
|
|
161
|
+
}
|
|
162
|
+
const numbers = value.map(item => (typeof item === 'number' ? item : Number(item)));
|
|
163
|
+
if (numbers.some(item => !Number.isFinite(item)))
|
|
164
|
+
throw new Error(`${label}: the database returned non-finite quantiles for ${key}.`);
|
|
165
|
+
const [p5, p25, p50, p75, p95] = numbers.map(item => round(item, 4));
|
|
166
|
+
return { p5, p25, p50, p75, p95 };
|
|
167
|
+
}
|
|
168
|
+
function round(value, digits) {
|
|
169
|
+
const factor = 10 ** digits;
|
|
170
|
+
return Math.round(value * factor) / factor;
|
|
171
|
+
}
|
|
172
|
+
function share(part, whole) {
|
|
173
|
+
return whole === 0 ? 0 : round(part / whole, 6);
|
|
174
|
+
}
|
|
175
|
+
function rowsWord(count) {
|
|
176
|
+
return count === 1 ? '1 row' : `${count} rows`;
|
|
177
|
+
}
|
|
178
|
+
function rowsDo(count, where, one, many) {
|
|
179
|
+
return `${rowsWord(count)}${where} ${count === 1 ? one : many}`;
|
|
180
|
+
}
|
|
181
|
+
function formatBytes(bytes) {
|
|
182
|
+
const units = ['B', 'kB', 'MB', 'GB', 'TB'];
|
|
183
|
+
let value = bytes;
|
|
184
|
+
let unit = 0;
|
|
185
|
+
while (value >= 1024 && unit < units.length - 1) {
|
|
186
|
+
value /= 1024;
|
|
187
|
+
unit += 1;
|
|
188
|
+
}
|
|
189
|
+
return `${unit === 0 ? value.toFixed(0) : value.toFixed(1)} ${units[unit]}`;
|
|
190
|
+
}
|
|
191
|
+
async function loadServer(db) {
|
|
192
|
+
const [server] = await db.query('server version', `SELECT current_setting('server_version') AS version, current_setting('server_version_num') AS num, pg_is_in_recovery() AS in_recovery`);
|
|
193
|
+
const versionNum = Number(str(server, 'num', 'server version'));
|
|
194
|
+
if (!Number.isInteger(versionNum) || versionNum < MIN_SERVER_VERSION_NUM) {
|
|
195
|
+
throw new Error(`Postgres ${str(server, 'version', 'server version')} is older than the oldest supported server (12).`);
|
|
196
|
+
}
|
|
197
|
+
if (typeof server.in_recovery !== 'boolean')
|
|
198
|
+
throw new Error('server version: the database returned no true or false for pg_is_in_recovery().');
|
|
199
|
+
return { version: str(server, 'version', 'server version'), inRecovery: server.in_recovery };
|
|
200
|
+
}
|
|
201
|
+
async function loadSettings(db) {
|
|
202
|
+
const prefixes = SETTINGS_PREFIXES.map(prefix => `name LIKE '${prefix.replace('_', '\\_')}%'`).join(' OR ');
|
|
203
|
+
const rows = await db.query('allowlisted settings', `SELECT name, current_setting(name) AS value FROM pg_settings WHERE name = ANY($1::text[]) OR ${prefixes} ORDER BY name`, [SETTINGS_ALLOWLIST]);
|
|
204
|
+
return rows.map(row => ({ name: str(row, 'name', 'settings'), value: str(row, 'value', 'settings') }));
|
|
205
|
+
}
|
|
206
|
+
async function loadSchemas(db, requested) {
|
|
207
|
+
if (requested === null) {
|
|
208
|
+
const rows = await db.query('schemas', `SELECT nspname::text AS name FROM pg_namespace WHERE nspname <> 'information_schema' AND nspname !~ '^pg_' ORDER BY 1`);
|
|
209
|
+
return rows.map(row => str(row, 'name', 'schemas'));
|
|
210
|
+
}
|
|
211
|
+
const rows = await db.query('schemas', `SELECT nspname::text AS name FROM pg_namespace WHERE nspname = ANY($1::text[])`, [requested]);
|
|
212
|
+
const found = new Set(rows.map(row => str(row, 'name', 'schemas')));
|
|
213
|
+
const missing = requested.filter(schema => !found.has(schema));
|
|
214
|
+
if (missing.length > 0)
|
|
215
|
+
throw new Error(`--schema names schemas that do not exist: ${missing.join(', ')}.`);
|
|
216
|
+
return requested;
|
|
217
|
+
}
|
|
218
|
+
async function loadTables(db, schemas) {
|
|
219
|
+
const rows = await db.query('tables', `SELECT c.oid::text AS oid, n.nspname::text AS schema, c.relname::text AS name,
|
|
220
|
+
c.relkind::text AS relkind, c.relispartition AS partition, c.reltuples::float8 AS reltuples,
|
|
221
|
+
pg_relation_size(c.oid)::float8 AS heap_bytes, pg_table_size(c.oid)::float8 AS table_size,
|
|
222
|
+
pg_indexes_size(c.oid)::float8 AS index_bytes,
|
|
223
|
+
(CASE WHEN c.reltoastrelid = 0 THEN 0 ELSE pg_total_relation_size(c.reltoastrelid) END)::float8 AS toast_bytes,
|
|
224
|
+
has_table_privilege(c.oid, 'SELECT') AS can_select, row_security_active(c.oid) AS rls_active,
|
|
225
|
+
s.n_live_tup::float8 AS live, s.n_dead_tup::float8 AS dead,
|
|
226
|
+
EXISTS (SELECT 1 FROM pg_depend d WHERE d.classid = 'pg_class'::regclass AND d.objid = c.oid AND d.deptype = 'e') AS extension_member
|
|
227
|
+
FROM pg_class c
|
|
228
|
+
JOIN pg_namespace n ON n.oid = c.relnamespace
|
|
229
|
+
LEFT JOIN pg_stat_user_tables s ON s.relid = c.oid
|
|
230
|
+
WHERE n.nspname = ANY($1::text[]) AND c.relkind IN ('r', 'p')
|
|
231
|
+
ORDER BY n.nspname, c.relname`, [schemas]);
|
|
232
|
+
return rows.map(row => ({
|
|
233
|
+
oid: str(row, 'oid', 'tables'),
|
|
234
|
+
ref: { schema: str(row, 'schema', 'tables'), table: str(row, 'name', 'tables') },
|
|
235
|
+
relkind: str(row, 'relkind', 'tables'),
|
|
236
|
+
partition: row.partition === true,
|
|
237
|
+
reltuples: num(row, 'reltuples', 'tables'),
|
|
238
|
+
heapBytes: num(row, 'heap_bytes', 'tables'),
|
|
239
|
+
tableSize: num(row, 'table_size', 'tables'),
|
|
240
|
+
indexBytes: num(row, 'index_bytes', 'tables'),
|
|
241
|
+
toastBytes: num(row, 'toast_bytes', 'tables'),
|
|
242
|
+
canSelect: row.can_select === true,
|
|
243
|
+
rlsActive: row.rls_active === true,
|
|
244
|
+
live: numOrNull(row, 'live', 'tables'),
|
|
245
|
+
dead: numOrNull(row, 'dead', 'tables'),
|
|
246
|
+
extensionMember: row.extension_member === true,
|
|
247
|
+
}));
|
|
248
|
+
}
|
|
249
|
+
async function loadColumns(db, oids) {
|
|
250
|
+
const rows = await db.query('columns', `SELECT a.attrelid::text AS table_oid, a.attname::text AS name,
|
|
251
|
+
format_type(a.atttypid, a.atttypmod) AS data_type, a.attnotnull AS not_null,
|
|
252
|
+
b.oid::text AS base_oid, b.typname::text AS typname, b.typcategory::text AS typcategory, b.typtype::text AS typtype,
|
|
253
|
+
EXISTS (SELECT 1 FROM pg_opclass oc JOIN pg_am am ON am.oid = oc.opcmethod
|
|
254
|
+
WHERE am.amname = 'btree' AND oc.opcdefault AND oc.opcintype = b.oid) AS sortable
|
|
255
|
+
FROM pg_attribute a
|
|
256
|
+
JOIN pg_type t ON t.oid = a.atttypid
|
|
257
|
+
JOIN pg_type b ON b.oid = (CASE WHEN t.typtype = 'd' THEN t.typbasetype ELSE t.oid END)
|
|
258
|
+
WHERE a.attrelid = ANY($1::text[]::oid[]) AND a.attnum > 0 AND NOT a.attisdropped
|
|
259
|
+
ORDER BY a.attrelid, a.attnum`, [oids]);
|
|
260
|
+
return rows.map(row => ({
|
|
261
|
+
tableOid: str(row, 'table_oid', 'columns'),
|
|
262
|
+
name: str(row, 'name', 'columns'),
|
|
263
|
+
dataType: str(row, 'data_type', 'columns'),
|
|
264
|
+
notNull: row.not_null === true,
|
|
265
|
+
baseOid: str(row, 'base_oid', 'columns'),
|
|
266
|
+
typname: str(row, 'typname', 'columns'),
|
|
267
|
+
family: typeFamily({
|
|
268
|
+
typname: str(row, 'typname', 'columns'),
|
|
269
|
+
typcategory: str(row, 'typcategory', 'columns'),
|
|
270
|
+
typtype: str(row, 'typtype', 'columns'),
|
|
271
|
+
}),
|
|
272
|
+
sortable: row.sortable === true,
|
|
273
|
+
}));
|
|
274
|
+
}
|
|
275
|
+
async function loadConstraints(db, oids) {
|
|
276
|
+
const rows = await db.query('constraints', `SELECT c.conrelid::text AS table_oid, c.conname::text AS name, c.contype::text AS type,
|
|
277
|
+
c.convalidated AS validated,
|
|
278
|
+
ARRAY(SELECT a.attname::text FROM unnest(c.conkey) WITH ORDINALITY k(attnum, ord)
|
|
279
|
+
JOIN pg_attribute a ON a.attrelid = c.conrelid AND a.attnum = k.attnum ORDER BY k.ord) AS columns,
|
|
280
|
+
rn.nspname::text AS ref_schema, rc.relname::text AS ref_table,
|
|
281
|
+
ARRAY(SELECT a.attname::text FROM unnest(c.confkey) WITH ORDINALITY k(attnum, ord)
|
|
282
|
+
JOIN pg_attribute a ON a.attrelid = c.confrelid AND a.attnum = k.attnum ORDER BY k.ord) AS ref_columns,
|
|
283
|
+
CASE WHEN c.contype = 'c' AND array_length(c.conkey, 1) = 1 THEN pg_get_constraintdef(c.oid) END AS check_definition
|
|
284
|
+
FROM pg_constraint c
|
|
285
|
+
LEFT JOIN pg_class rc ON rc.oid = c.confrelid
|
|
286
|
+
LEFT JOIN pg_namespace rn ON rn.oid = rc.relnamespace
|
|
287
|
+
WHERE c.conrelid = ANY($1::text[]::oid[]) AND c.contype IN ('p', 'u', 'f', 'c')
|
|
288
|
+
ORDER BY c.conrelid, c.conname`, [oids]);
|
|
289
|
+
return rows.map(row => {
|
|
290
|
+
const type = str(row, 'type', 'constraints');
|
|
291
|
+
return {
|
|
292
|
+
tableOid: str(row, 'table_oid', 'constraints'),
|
|
293
|
+
name: str(row, 'name', 'constraints'),
|
|
294
|
+
type,
|
|
295
|
+
validated: row.validated === true,
|
|
296
|
+
columns: strList(row, 'columns', 'constraints'),
|
|
297
|
+
referenced: type === 'f'
|
|
298
|
+
? { schema: str(row, 'ref_schema', 'constraints'), table: str(row, 'ref_table', 'constraints') }
|
|
299
|
+
: null,
|
|
300
|
+
referencedColumns: type === 'f' ? strList(row, 'ref_columns', 'constraints') : [],
|
|
301
|
+
checkDefinition: typeof row.check_definition === 'string' ? row.check_definition : null,
|
|
302
|
+
};
|
|
303
|
+
});
|
|
304
|
+
}
|
|
305
|
+
async function loadEnumLabels(db, typeOids) {
|
|
306
|
+
if (typeOids.length === 0)
|
|
307
|
+
return new Map();
|
|
308
|
+
const rows = await db.query('enum labels', `SELECT enumtypid::text AS type_oid, array_agg(enumlabel::text ORDER BY enumsortorder) AS labels
|
|
309
|
+
FROM pg_enum WHERE enumtypid = ANY($1::text[]::oid[]) GROUP BY enumtypid`, [typeOids]);
|
|
310
|
+
return new Map(rows.map(row => [str(row, 'type_oid', 'enum labels'), strList(row, 'labels', 'enum labels')]));
|
|
311
|
+
}
|
|
312
|
+
async function loadIndexes(db, schemas) {
|
|
313
|
+
const rows = await db.query('indexes', `SELECT schemaname::text AS schema, tablename::text AS table_name, indexname::text AS name, indexdef AS definition
|
|
314
|
+
FROM pg_indexes WHERE schemaname = ANY($1::text[]) ORDER BY schemaname, tablename, indexname`, [schemas]);
|
|
315
|
+
const byTable = new Map();
|
|
316
|
+
for (const row of rows) {
|
|
317
|
+
const key = tableKey({ schema: str(row, 'schema', 'indexes'), table: str(row, 'table_name', 'indexes') });
|
|
318
|
+
const name = str(row, 'name', 'indexes');
|
|
319
|
+
let definition;
|
|
320
|
+
try {
|
|
321
|
+
definition = redactSqlLiterals(str(row, 'definition', 'indexes'));
|
|
322
|
+
}
|
|
323
|
+
catch (error) {
|
|
324
|
+
// Never quote the definition: it is exactly what may hold a value.
|
|
325
|
+
if (error instanceof SqlLiteralError) {
|
|
326
|
+
throw new Error(`Index ${str(row, 'schema', 'indexes')}.${name}: its definition cannot be checked for literal values (${error.message}), so the profile cannot include it; report this to Haystack.`);
|
|
327
|
+
}
|
|
328
|
+
throw error;
|
|
329
|
+
}
|
|
330
|
+
byTable.set(key, [...(byTable.get(key) ?? []), { name, definition }]);
|
|
331
|
+
}
|
|
332
|
+
return byTable;
|
|
333
|
+
}
|
|
334
|
+
/** Rounded to the six decimals the TABLESAMPLE clause and the profile carry. */
|
|
335
|
+
function samplePercentFor(heapBytes, scanMaxBytes) {
|
|
336
|
+
if (heapBytes <= scanMaxBytes)
|
|
337
|
+
return null;
|
|
338
|
+
return round(Math.max(0.000001, Math.min(99.999999, (100 * scanMaxBytes) / heapBytes)), 6);
|
|
339
|
+
}
|
|
340
|
+
/* ------------------------------------------------------------ profile */
|
|
341
|
+
export async function profileDatabase(db, run, log, now = () => new Date()) {
|
|
342
|
+
const server = await loadServer(db);
|
|
343
|
+
if (server.inRecovery) {
|
|
344
|
+
log('Connected to a read replica (hot standby): dead-row shares are reported as unavailable, because Postgres counts dead rows only on the primary.');
|
|
345
|
+
}
|
|
346
|
+
const settings = await loadSettings(db);
|
|
347
|
+
const schemas = await loadSchemas(db, run.schemas);
|
|
348
|
+
const allTables = await loadTables(db, schemas);
|
|
349
|
+
const partitioned = allTables.filter(table => table.relkind === 'p');
|
|
350
|
+
if (partitioned.length > 0) {
|
|
351
|
+
log(`Partitioned tables are profiled through their partitions, not as parents: ${partitioned.map(table => tableKey(table.ref)).join(', ')}.`);
|
|
352
|
+
}
|
|
353
|
+
const extensionTables = allTables.filter(table => table.relkind === 'r' && table.extensionMember);
|
|
354
|
+
if (extensionTables.length > 0) {
|
|
355
|
+
log(`Skipping tables that belong to installed extensions, not the application: ${extensionTables.map(table => tableKey(table.ref)).join(', ')}.`);
|
|
356
|
+
}
|
|
357
|
+
const tables = allTables.filter(table => table.relkind === 'r' && !table.extensionMember);
|
|
358
|
+
if (tables.length === 0)
|
|
359
|
+
throw new Error(`No application tables found in schema${schemas.length === 1 ? '' : 's'} ${schemas.join(', ')}.`);
|
|
360
|
+
const unreadable = tables.filter(table => !table.canSelect);
|
|
361
|
+
if (unreadable.length > 0) {
|
|
362
|
+
throw new Error(`The profiler's role cannot SELECT from ${unreadable.map(table => tableKey(table.ref)).join(', ')}; grant SELECT (read only) or narrow --schema.`);
|
|
363
|
+
}
|
|
364
|
+
const rowSecured = tables.filter(table => table.rlsActive);
|
|
365
|
+
if (rowSecured.length > 0) {
|
|
366
|
+
throw new Error(`Row-level security hides rows of ${rowSecured.map(table => tableKey(table.ref)).join(', ')} from the profiler's role, so the profile would be partial; use a role with BYPASSRLS or narrow --schema.`);
|
|
367
|
+
}
|
|
368
|
+
if (run.ownerTable !== null && !tables.some(table => tableKey(table.ref) === tableKey(run.ownerTable))) {
|
|
369
|
+
throw new Error(`--owner-table ${tableKey(run.ownerTable)} is not an application table in the profiled schemas (${schemas.join(', ')}).`);
|
|
370
|
+
}
|
|
371
|
+
const oids = tables.map(table => table.oid);
|
|
372
|
+
const columns = await loadColumns(db, oids);
|
|
373
|
+
const constraints = await loadConstraints(db, oids);
|
|
374
|
+
const enumLabels = await loadEnumLabels(db, [...new Set(columns.filter(column => column.family === 'enum').map(column => column.baseOid))]);
|
|
375
|
+
const indexes = await loadIndexes(db, schemas);
|
|
376
|
+
const tableByOid = new Map(tables.map(table => [table.oid, table]));
|
|
377
|
+
const edges = [];
|
|
378
|
+
const foreignKeysByTable = new Map();
|
|
379
|
+
for (const constraint of constraints) {
|
|
380
|
+
if (constraint.type !== 'f' || constraint.referenced === null)
|
|
381
|
+
continue;
|
|
382
|
+
const table = tableByOid.get(constraint.tableOid);
|
|
383
|
+
if (table === undefined)
|
|
384
|
+
continue;
|
|
385
|
+
const edge = {
|
|
386
|
+
name: constraint.name,
|
|
387
|
+
from: table.ref,
|
|
388
|
+
columns: constraint.columns,
|
|
389
|
+
to: constraint.referenced,
|
|
390
|
+
referencedColumns: constraint.referencedColumns,
|
|
391
|
+
};
|
|
392
|
+
edges.push(edge);
|
|
393
|
+
foreignKeysByTable.set(table.oid, [...(foreignKeysByTable.get(table.oid) ?? []), { edge, validated: constraint.validated }]);
|
|
394
|
+
}
|
|
395
|
+
// Owner paths may only run through profiled tables: `profile show` rebuilds
|
|
396
|
+
// them from the file, which holds only profiled tables' foreign keys.
|
|
397
|
+
const profiled = new Set(tables.map(table => tableKey(table.ref)));
|
|
398
|
+
const ownerEdges = edges.filter(edge => profiled.has(tableKey(edge.to)));
|
|
399
|
+
log(`Profiling ${tables.length} table${tables.length === 1 ? '' : 's'} in ${schemas.join(', ')} with ${db.concurrency} connection${db.concurrency === 1 ? '' : 's'}.`);
|
|
400
|
+
const plans = tables.map(table => {
|
|
401
|
+
const label = tableKey(table.ref);
|
|
402
|
+
const reach = ownerReach(table.ref, run.ownerTable, ownerEdges);
|
|
403
|
+
const samplePercent = samplePercentFor(table.heapBytes, run.scanMaxBytes);
|
|
404
|
+
const tableConstraints = constraints.filter(constraint => constraint.tableOid === table.oid);
|
|
405
|
+
const keyColumns = new Set(tableConstraints
|
|
406
|
+
.filter(constraint => constraint.type === 'p' || constraint.type === 'f' || (constraint.type === 'u' && constraint.columns.length === 1))
|
|
407
|
+
.flatMap(constraint => constraint.columns));
|
|
408
|
+
const planColumns = columns.filter(column => column.tableOid === table.oid).map(column => {
|
|
409
|
+
let schemaValues = null;
|
|
410
|
+
if (column.family === 'enum') {
|
|
411
|
+
const labels = enumLabels.get(column.baseOid);
|
|
412
|
+
if (labels === undefined)
|
|
413
|
+
throw new Error(`${label}.${column.name}: enum type ${column.dataType} has no labels in pg_enum.`);
|
|
414
|
+
schemaValues = labels;
|
|
415
|
+
}
|
|
416
|
+
else {
|
|
417
|
+
for (const constraint of tableConstraints) {
|
|
418
|
+
if (constraint.type !== 'c' || constraint.checkDefinition === null || constraint.columns[0] !== column.name)
|
|
419
|
+
continue;
|
|
420
|
+
const values = checkConstraintValues(constraint.checkDefinition, column.name);
|
|
421
|
+
if (values !== null) {
|
|
422
|
+
schemaValues = values;
|
|
423
|
+
break;
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
return {
|
|
428
|
+
name: column.name,
|
|
429
|
+
family: column.family,
|
|
430
|
+
typname: column.typname,
|
|
431
|
+
sortable: column.sortable,
|
|
432
|
+
catalog: column,
|
|
433
|
+
schemaValues,
|
|
434
|
+
isKey: keyColumns.has(column.name),
|
|
435
|
+
};
|
|
436
|
+
});
|
|
437
|
+
log(` ${label}: ${samplePercent === null
|
|
438
|
+
? `reading all ${formatBytes(table.heapBytes)}`
|
|
439
|
+
: `reading a ${pct(samplePercent / 100)} TABLESAMPLE SYSTEM sample (about ${formatBytes(run.scanMaxBytes)} of ${formatBytes(table.heapBytes)})`}; ${describeReach(reach, run.ownerTable)}.`);
|
|
440
|
+
const foreignKeys = foreignKeysByTable.get(table.oid) ?? [];
|
|
441
|
+
const uniqueColumns = new Set();
|
|
442
|
+
const keyCoverage = new Map();
|
|
443
|
+
if (samplePercent !== null) {
|
|
444
|
+
for (const constraint of tableConstraints) {
|
|
445
|
+
if ((constraint.type === 'p' || constraint.type === 'u') && constraint.columns.length === 1)
|
|
446
|
+
uniqueColumns.add(constraint.columns[0]);
|
|
447
|
+
}
|
|
448
|
+
for (const { edge, validated } of foreignKeys) {
|
|
449
|
+
const [only] = edge.columns;
|
|
450
|
+
if (validated && edge.columns.length === 1 && !uniqueColumns.has(only) && !keyCoverage.has(only))
|
|
451
|
+
keyCoverage.set(only, edge);
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
return {
|
|
455
|
+
table,
|
|
456
|
+
label,
|
|
457
|
+
columns: planColumns,
|
|
458
|
+
reach,
|
|
459
|
+
samplePercent,
|
|
460
|
+
scan: scanSql({ table: table.ref, samplePercent, reach }),
|
|
461
|
+
foreignKeys,
|
|
462
|
+
uniqueColumns,
|
|
463
|
+
keyCoverage,
|
|
464
|
+
indexes: indexes.get(label) ?? [],
|
|
465
|
+
};
|
|
466
|
+
});
|
|
467
|
+
// Pass 1: one aggregate scan per table (more for very wide tables).
|
|
468
|
+
const stats = new Map();
|
|
469
|
+
const statsWork = plans.flatMap(plan => {
|
|
470
|
+
const batches = statsBatches(plan.columns, plan.samplePercent === null);
|
|
471
|
+
return batches.map((batch, batchIndex) => ({ plan, batch, part: batches.length === 1 ? '' : ` (part ${batchIndex + 1} of ${batches.length})` }));
|
|
472
|
+
});
|
|
473
|
+
await runBounded(statsWork, db.concurrency, async ({ plan, batch, part }) => {
|
|
474
|
+
const [row] = await db.query(`${plan.label}: column statistics${part}`, statsQuery(plan.scan, batch, plan.samplePercent === null));
|
|
475
|
+
stats.set(plan, { ...(stats.get(plan) ?? {}), ...row });
|
|
476
|
+
});
|
|
477
|
+
// Classify every column before any value query exists. A table read in
|
|
478
|
+
// full is settled here from its exact distinct counts; a sampled table's
|
|
479
|
+
// columns that only a distinct count can settle wait for the passes below.
|
|
480
|
+
const columnStates = new Map();
|
|
481
|
+
for (const plan of plans) {
|
|
482
|
+
columnStates.set(plan, plan.columns.map((column, index) => columnState(plan, column, index, stats.get(plan), run)));
|
|
483
|
+
}
|
|
484
|
+
// Pass 1b: each sampled column's distinct values and the sampled pages each
|
|
485
|
+
// is found on, for its estimate, except key columns counted another way.
|
|
486
|
+
const sampleWork = plans.filter(plan => plan.samplePercent !== null).flatMap(plan => columnStates.get(plan)
|
|
487
|
+
.filter(state => !plan.uniqueColumns.has(state.plan.name) && !plan.keyCoverage.has(state.plan.name))
|
|
488
|
+
.map(state => ({ plan, state })));
|
|
489
|
+
await runBounded(sampleWork, db.concurrency, async ({ plan, state }) => {
|
|
490
|
+
const at = `${plan.label}.${state.plan.name}`;
|
|
491
|
+
const rows = await db.query(`${at}: distinct values in the sample`, sampleDistinctQuery(plan.table.ref, plan.samplePercent, state.plan));
|
|
492
|
+
state.sample = rows.map(row => ({ pages: num(row, 'pages', at), values: num(row, 'value_count', at) }));
|
|
493
|
+
});
|
|
494
|
+
// Pass 1c: a category column rests on an exact count over the whole table,
|
|
495
|
+
// never on an estimate. A sample never holds more distinct values than its
|
|
496
|
+
// table, so one with more than categoryMaxDistinct already rules the column
|
|
497
|
+
// out; the rest are counted exactly, all of a table's in one read of it.
|
|
498
|
+
const categoryWork = [];
|
|
499
|
+
for (const plan of plans) {
|
|
500
|
+
if (plan.samplePercent === null)
|
|
501
|
+
continue;
|
|
502
|
+
const candidates = [];
|
|
503
|
+
for (const state of columnStates.get(plan)) {
|
|
504
|
+
if (state.verdict.kind !== 'distinct-decides')
|
|
505
|
+
continue;
|
|
506
|
+
const inSample = sampleOf(state, plan).reduce((sum, entry) => sum + entry.values, 0);
|
|
507
|
+
if (inSample > run.categoryMaxDistinct)
|
|
508
|
+
settle(state, decideByDistinct(state.verdict, inSample, run.categoryMaxDistinct));
|
|
509
|
+
else
|
|
510
|
+
candidates.push(state);
|
|
511
|
+
}
|
|
512
|
+
if (candidates.length === 0)
|
|
513
|
+
continue;
|
|
514
|
+
const names = candidates.map(state => state.plan.name);
|
|
515
|
+
log(` ${plan.label}: reading the whole table once to count the distinct values of ${names.join(', ')} exactly, since ${names.length === 1 ? 'that decides whether it is a category column' : 'those decide whether each is a category column'}.`);
|
|
516
|
+
categoryWork.push({ plan, candidates });
|
|
517
|
+
}
|
|
518
|
+
await runBounded(categoryWork, db.concurrency, async ({ plan, candidates }) => {
|
|
519
|
+
const [row] = await db.query(`${plan.label}: exact distinct counts of possible category columns (reads the whole table)`, categoryDistinctQuery(plan.table.ref, candidates.map(state => ({ index: state.index, column: state.plan }))));
|
|
520
|
+
for (const state of candidates) {
|
|
521
|
+
const exact = num(row, `c${state.index}_d`, `${plan.label}.${state.plan.name}`);
|
|
522
|
+
state.distinctEstimate = exact;
|
|
523
|
+
settle(state, decideByDistinct(state.verdict, exact, run.categoryMaxDistinct));
|
|
524
|
+
}
|
|
525
|
+
});
|
|
526
|
+
// Pass 2: value, shape, JSON key, oversize and foreign-key queries.
|
|
527
|
+
const tasks = [];
|
|
528
|
+
const foreignKeyProfiles = new Map();
|
|
529
|
+
const tableByKey = new Map(tables.map(table => [tableKey(table.ref), table]));
|
|
530
|
+
const planByKey = new Map(plans.map(plan => [plan.label, plan]));
|
|
531
|
+
for (const plan of plans) {
|
|
532
|
+
const states = columnStates.get(plan);
|
|
533
|
+
const ownersCountable = plan.reach.kind !== 'unreachable';
|
|
534
|
+
for (const state of states) {
|
|
535
|
+
const column = state.plan;
|
|
536
|
+
if (state.nonNull === 0) {
|
|
537
|
+
// Schema-declared values need no data: an empty column still lists them.
|
|
538
|
+
if (settled(state).kind === 'category' && column.schemaValues !== null) {
|
|
539
|
+
state.categories = column.schemaValues.map(value => ({ value, fraction: 0, source: 'schema' }));
|
|
540
|
+
}
|
|
541
|
+
continue;
|
|
542
|
+
}
|
|
543
|
+
const at = `${plan.label}.${column.name}`;
|
|
544
|
+
const kind = settled(state).kind;
|
|
545
|
+
if (kind === 'category' && column.schemaValues !== null) {
|
|
546
|
+
tasks.push(async () => {
|
|
547
|
+
const rows = await db.query(`${at}: schema category counts`, schemaValuesQuery(plan.scan, column.name), [column.schemaValues]);
|
|
548
|
+
const counts = new Map(rows.map(row => [str(row, 'value', at), num(row, 'rows', at)]));
|
|
549
|
+
state.categories = column.schemaValues.map(value => ({
|
|
550
|
+
value,
|
|
551
|
+
fraction: share(counts.get(value) ?? 0, state.nonNull),
|
|
552
|
+
source: 'schema',
|
|
553
|
+
}));
|
|
554
|
+
});
|
|
555
|
+
}
|
|
556
|
+
else if (kind === 'category' && ownersCountable) {
|
|
557
|
+
tasks.push(async () => {
|
|
558
|
+
const rows = await db.query(`${at}: category values`, categoryValuesQuery(plan.scan, column.name), [run.categoryMinOwners]);
|
|
559
|
+
const values = rows.map(row => ({ value: str(row, 'value', at), rows: num(row, 'rows', at) }));
|
|
560
|
+
// A value of a personal shape, or one that does not fit the
|
|
561
|
+
// column's declared type, never leaves however many owners share it
|
|
562
|
+
// (keptValueProblem, the rule the upload server applies too): it is
|
|
563
|
+
// dropped exactly like a value that fails the owner rule (left out,
|
|
564
|
+
// its rows counted as withheld, the other values' shares unchanged).
|
|
565
|
+
// The value itself is never logged.
|
|
566
|
+
const problems = values.map(entry => keptValueProblem(column.catalog.dataType, entry.value));
|
|
567
|
+
const dropped = problems.filter((problem) => problem !== null);
|
|
568
|
+
if (dropped.length > 0) {
|
|
569
|
+
log(` ${at}: not keeping ${dropped.length === 1 ? 'a value that' : `${dropped.length} values that each`} ${[...new Set(dropped)].join('; or ')}, though shared by enough owners.`);
|
|
570
|
+
}
|
|
571
|
+
state.categories = values
|
|
572
|
+
.filter((_, index) => problems[index] === null)
|
|
573
|
+
.map(entry => ({ value: entry.value, fraction: share(entry.rows, state.nonNull), source: 'data' }));
|
|
574
|
+
});
|
|
575
|
+
}
|
|
576
|
+
if (column.family === 'text' && kind !== 'free-text' && ownersCountable) {
|
|
577
|
+
tasks.push(async () => {
|
|
578
|
+
const rows = await db.query(`${at}: shapes`, shapesQuery(plan.scan, column.name), [run.categoryMinOwners]);
|
|
579
|
+
state.shapeRows = rows.map(row => ({ shape: str(row, 'shape', at), rows: num(row, 'rows', at) }));
|
|
580
|
+
});
|
|
581
|
+
}
|
|
582
|
+
if (column.family === 'json' && ownersCountable) {
|
|
583
|
+
tasks.push(async () => {
|
|
584
|
+
const rows = await db.query(`${at}: JSON key paths`, jsonKeysQuery(plan.scan, column.name), [run.categoryMinOwners]);
|
|
585
|
+
const entries = rows.map(row => ({
|
|
586
|
+
path: str(row, 'path', at),
|
|
587
|
+
present: num(row, 'present', at),
|
|
588
|
+
type: str(row, 'type', at),
|
|
589
|
+
rows: num(row, 'rows', at),
|
|
590
|
+
}));
|
|
591
|
+
// The query already left out paths with a personal key or too few
|
|
592
|
+
// owners. A path whose keys only read as personal together (run by
|
|
593
|
+
// run, as the upload check tests it) is withheld here, entirely: no
|
|
594
|
+
// key entry and no oddity is built for it, since both come from
|
|
595
|
+
// these rows alone. The path itself is never logged.
|
|
596
|
+
const shaped = [...new Set(entries.map(entry => entry.path))]
|
|
597
|
+
.map(path => jsonPathPersonalShape(path))
|
|
598
|
+
.filter((what) => what !== null);
|
|
599
|
+
if (shaped.length > 0) {
|
|
600
|
+
log(` ${at}: not keeping ${shaped.length === 1 ? 'a JSON key path' : `${shaped.length} JSON key paths`} shaped like ${[...new Set(shaped)].join(', ')}, though shared by enough owners.`);
|
|
601
|
+
}
|
|
602
|
+
state.jsonRows = entries.filter(entry => jsonPathPersonalShape(entry.path) === null);
|
|
603
|
+
});
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
const oversizeChecks = states
|
|
607
|
+
.map((state, index) => ({ state, index }))
|
|
608
|
+
.filter(({ state }) => hasLengthQuantiles(state.plan.family) && state.lengthP95 !== null)
|
|
609
|
+
.map(({ state, index }) => ({
|
|
610
|
+
index,
|
|
611
|
+
column: state.plan,
|
|
612
|
+
thresholdBytes: Math.max(10 * state.lengthP95, 1024),
|
|
613
|
+
state,
|
|
614
|
+
}));
|
|
615
|
+
if (oversizeChecks.length > 0) {
|
|
616
|
+
tasks.push(async () => {
|
|
617
|
+
const [row] = await db.query(`${plan.label}: oversize values`, oversizeQuery(plan.scan, oversizeChecks));
|
|
618
|
+
for (const check of oversizeChecks)
|
|
619
|
+
check.state.oversize = num(row, `c${check.index}_big`, plan.label);
|
|
620
|
+
});
|
|
621
|
+
}
|
|
622
|
+
const keys = [];
|
|
623
|
+
foreignKeyProfiles.set(plan, keys);
|
|
624
|
+
for (const { edge, validated } of plan.foreignKeys) {
|
|
625
|
+
const at = `${plan.label} foreign key ${edge.name}`;
|
|
626
|
+
const profileEntry = {
|
|
627
|
+
name: edge.name,
|
|
628
|
+
columns: edge.columns,
|
|
629
|
+
references: { schema: edge.to.schema, table: edge.to.table, columns: edge.referencedColumns },
|
|
630
|
+
childrenPerParent: { p5: 0, p25: 0, p50: 0, p75: 0, p95: 0 },
|
|
631
|
+
orphanFraction: 0,
|
|
632
|
+
};
|
|
633
|
+
keys.push(profileEntry);
|
|
634
|
+
const parent = tableByKey.get(tableKey(edge.to));
|
|
635
|
+
const parentSample = parent === undefined ? null : samplePercentFor(parent.heapBytes, run.scanMaxBytes);
|
|
636
|
+
const covered = plan.keyCoverage.get(edge.columns[0]) === edge
|
|
637
|
+
? states.find(state => state.plan.name === edge.columns[0])
|
|
638
|
+
: undefined;
|
|
639
|
+
tasks.push(async () => {
|
|
640
|
+
const [row] = await db.query(`${at}: children per parent`, childrenPerParentQuery(edge, parentSample));
|
|
641
|
+
const measured = quantiles(row, 'q', at);
|
|
642
|
+
// An empty parent table has no children-per-parent distribution; the
|
|
643
|
+
// profile records zeros, and the parent's row count of 0 says why.
|
|
644
|
+
if (measured !== null)
|
|
645
|
+
profileEntry.childrenPerParent = measured;
|
|
646
|
+
if (covered !== undefined) {
|
|
647
|
+
// Every child value matches exactly one parent key, so the parents
|
|
648
|
+
// with a child are the column's distinct values.
|
|
649
|
+
const referenced = num(row, 'referenced', at);
|
|
650
|
+
covered.distinctEstimate = parentSample === null
|
|
651
|
+
? referenced
|
|
652
|
+
: referencedInTable(referenced, planByKey.get(tableKey(edge.to)), stats, `${plan.label}.${edge.columns[0]}`);
|
|
653
|
+
}
|
|
654
|
+
});
|
|
655
|
+
if (!validated) {
|
|
656
|
+
tasks.push(async () => {
|
|
657
|
+
const [row] = await db.query(`${at}: orphans`, orphanQuery(edge, plan.samplePercent));
|
|
658
|
+
profileEntry.orphanFraction = share(num(row, 'orphans', at), num(row, 'children', at));
|
|
659
|
+
});
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
await runBounded(tasks, db.concurrency, task => task());
|
|
664
|
+
const tableProfiles = plans.map(plan => {
|
|
665
|
+
const row = stats.get(plan);
|
|
666
|
+
const total = num(row, 'total', plan.label);
|
|
667
|
+
const rowCount = plan.samplePercent === null ? total : plannerRowCount(plan, total);
|
|
668
|
+
for (const state of columnStates.get(plan)) {
|
|
669
|
+
if (state.distinctEstimate !== null)
|
|
670
|
+
continue;
|
|
671
|
+
// Scaled like the row count: the table's rows times the sample's non-null share.
|
|
672
|
+
const nonNullRows = total === 0 ? 0 : (rowCount * state.nonNull) / total;
|
|
673
|
+
state.distinctEstimate = plan.uniqueColumns.has(state.plan.name)
|
|
674
|
+
? nonNullRows
|
|
675
|
+
: estimateDistinct(sampleOf(state, plan), total / rowCount, nonNullRows);
|
|
676
|
+
}
|
|
677
|
+
return {
|
|
678
|
+
schema: plan.table.ref.schema,
|
|
679
|
+
name: plan.table.ref.table,
|
|
680
|
+
rowCount: Math.round(rowCount),
|
|
681
|
+
rowCountSource: plan.samplePercent === null ? 'exact' : 'estimate',
|
|
682
|
+
sampling: plan.samplePercent === null
|
|
683
|
+
? { status: 'full-scan' }
|
|
684
|
+
: { status: 'sampled', percent: plan.samplePercent, seed: SAMPLE_SEED },
|
|
685
|
+
tableBytes: Math.round(plan.table.tableSize - plan.table.toastBytes),
|
|
686
|
+
indexBytes: Math.round(plan.table.indexBytes),
|
|
687
|
+
toastBytes: Math.round(plan.table.toastBytes),
|
|
688
|
+
deadRows: deadRows(plan, rowCount, server.inRecovery),
|
|
689
|
+
columns: columnStates.get(plan).map(state => finishColumn(state, plan, run)),
|
|
690
|
+
foreignKeys: foreignKeyProfiles.get(plan).sort((left, right) => left.name.localeCompare(right.name)),
|
|
691
|
+
indexes: plan.indexes,
|
|
692
|
+
};
|
|
693
|
+
});
|
|
694
|
+
const rules = {
|
|
695
|
+
categoryMaxDistinct: run.categoryMaxDistinct,
|
|
696
|
+
categoryMinOwners: run.categoryMinOwners,
|
|
697
|
+
ownerCounting: run.ownerTable === null
|
|
698
|
+
? {
|
|
699
|
+
kind: 'rows',
|
|
700
|
+
reason: `No --owner-table was named, so each row counts as one owner: a value is kept only when at least ${run.categoryMinOwners} rows share it, even if they all belong to one person.`,
|
|
701
|
+
}
|
|
702
|
+
: { kind: 'owner-table', schema: run.ownerTable.schema, table: run.ownerTable.table },
|
|
703
|
+
};
|
|
704
|
+
const profile = {
|
|
705
|
+
version: DATABASE_PROFILE_VERSION,
|
|
706
|
+
profilerVersion: profilerVersion(),
|
|
707
|
+
createdAt: now().toISOString(),
|
|
708
|
+
engine: 'postgres',
|
|
709
|
+
engineVersion: server.version,
|
|
710
|
+
settings,
|
|
711
|
+
rules,
|
|
712
|
+
tables: tableProfiles,
|
|
713
|
+
};
|
|
714
|
+
// The file must satisfy the same strict parser `show` and `upload` use.
|
|
715
|
+
return parseDatabaseProfile(profile);
|
|
716
|
+
}
|
|
717
|
+
/** A sampled parent's referenced rows scaled to its whole table, like its row count. */
|
|
718
|
+
function referencedInTable(referenced, parent, stats, at) {
|
|
719
|
+
const sampledRows = num(stats.get(parent), 'total', parent.label);
|
|
720
|
+
if (sampledRows === 0) {
|
|
721
|
+
throw new Error(`${at}: the ${pct(parent.samplePercent / 100)} sample of ${parent.label} holds no rows, so how many of its rows ${at} references cannot be estimated; VACUUM ${parent.label} (its sampled pages are empty) or raise --scan-max-mb to read it in full.`);
|
|
722
|
+
}
|
|
723
|
+
return Math.round((referenced * plannerRowCount(parent, sampledRows)) / sampledRows);
|
|
724
|
+
}
|
|
725
|
+
/**
|
|
726
|
+
* The planner's row count of a sampled table (`sampledRows` is its sample's
|
|
727
|
+
* count). Postgres 14 and later mark a never-analyzed table -1; 12 and 13
|
|
728
|
+
* leave it 0, which is also what an analyzed empty table has. So 0 counts as
|
|
729
|
+
* unknown when the sample itself holds rows (checking only that the heap has
|
|
730
|
+
* pages would refuse a table emptied and analyzed however often ANALYZE ran).
|
|
731
|
+
*/
|
|
732
|
+
function plannerRowCount(plan, sampledRows) {
|
|
733
|
+
const reltuples = plan.table.reltuples;
|
|
734
|
+
if (reltuples < 0 || (reltuples === 0 && sampledRows > 0)) {
|
|
735
|
+
throw new Error(`${plan.label} is larger than --scan-max-mb and has never been analyzed${reltuples === 0 ? ' (or not since its rows were written)' : ''}, so its row count is unknown without a full count; run ANALYZE ${plan.label} (or let autovacuum do so), or raise --scan-max-mb to count it in full.`);
|
|
736
|
+
}
|
|
737
|
+
return reltuples;
|
|
738
|
+
}
|
|
739
|
+
export const REPLICA_DEAD_ROWS_REASON = 'Profiled on a read replica (hot standby): Postgres counts dead rows only on the primary, so their share cannot be measured here.';
|
|
740
|
+
function deadRows(plan, rowCount, inRecovery) {
|
|
741
|
+
// A standby's pg_stat_user_tables shows 0 live and 0 dead rows for every
|
|
742
|
+
// table: the counters are kept by the server that writes, not replayed.
|
|
743
|
+
if (inRecovery)
|
|
744
|
+
return { status: 'unavailable', reason: REPLICA_DEAD_ROWS_REASON };
|
|
745
|
+
const live = plan.table.live ?? 0;
|
|
746
|
+
const dead = plan.table.dead ?? 0;
|
|
747
|
+
if (live + dead > 0)
|
|
748
|
+
return { status: 'measured', fraction: round(dead / (live + dead), 6) };
|
|
749
|
+
if (rowCount === 0)
|
|
750
|
+
return { status: 'measured', fraction: 0 };
|
|
751
|
+
throw new Error(`${plan.label} has rows but pg_stat_user_tables has no live or dead row counts for it (its statistics were reset, or it was never vacuumed or analyzed), so its dead-row share cannot be measured; run ANALYZE ${plan.label} and profile again.`);
|
|
752
|
+
}
|
|
753
|
+
function settle(state, classification) {
|
|
754
|
+
state.verdict = classification;
|
|
755
|
+
state.categories = classification.kind === 'category' ? [] : null;
|
|
756
|
+
}
|
|
757
|
+
function settled(state) {
|
|
758
|
+
if (state.verdict.kind === 'distinct-decides')
|
|
759
|
+
throw new Error(`${state.plan.name}: classified before its distinct count was known.`);
|
|
760
|
+
return state.verdict;
|
|
761
|
+
}
|
|
762
|
+
function sampleOf(state, plan) {
|
|
763
|
+
if (state.sample === null)
|
|
764
|
+
throw new Error(`${plan.label}.${state.plan.name}: its sample's distinct values were never counted.`);
|
|
765
|
+
return state.sample;
|
|
766
|
+
}
|
|
767
|
+
function columnState(plan, column, index, row, run) {
|
|
768
|
+
const key = `c${index}`;
|
|
769
|
+
const at = `${plan.label}.${column.name}`;
|
|
770
|
+
const total = num(row, 'total', at);
|
|
771
|
+
const nonNull = num(row, `${key}_nn`, at);
|
|
772
|
+
const exactDistinct = plan.samplePercent === null ? num(row, `${key}_nd`, at) : null;
|
|
773
|
+
const lengthQuantiles = hasLengthQuantiles(column.family) ? quantiles(row, `${key}_lq`, at) : null;
|
|
774
|
+
const valueShapes = column.family === 'text'
|
|
775
|
+
? {
|
|
776
|
+
nonNull,
|
|
777
|
+
matches: Object.fromEntries(VALUE_SHAPE_TESTS.map(test => [test.id, num(row, `${key}_t_${test.id.replace('-', '_')}`, at)])),
|
|
778
|
+
averageWords: numOrNull(row, `${key}_words`, at),
|
|
779
|
+
fourWordValues: num(row, `${key}_w4`, at),
|
|
780
|
+
}
|
|
781
|
+
: null;
|
|
782
|
+
const facts = classifyColumnFacts({
|
|
783
|
+
name: column.name,
|
|
784
|
+
dataType: column.catalog.dataType,
|
|
785
|
+
family: column.family,
|
|
786
|
+
isKey: column.isKey,
|
|
787
|
+
schemaValues: column.schemaValues,
|
|
788
|
+
valueShapes,
|
|
789
|
+
});
|
|
790
|
+
const verdict = facts.kind === 'distinct-decides' && exactDistinct !== null
|
|
791
|
+
? decideByDistinct(facts, exactDistinct, run.categoryMaxDistinct)
|
|
792
|
+
: facts;
|
|
793
|
+
return {
|
|
794
|
+
plan: column,
|
|
795
|
+
index,
|
|
796
|
+
total,
|
|
797
|
+
nonNull,
|
|
798
|
+
verdict,
|
|
799
|
+
sample: null,
|
|
800
|
+
distinctEstimate: exactDistinct,
|
|
801
|
+
avgWidth: nonNull === 0 ? 0 : round(num(row, `${key}_aw`, at), 2),
|
|
802
|
+
owners: num(row, `${key}_own`, at),
|
|
803
|
+
lengthQuantiles,
|
|
804
|
+
lengthP95: lengthQuantiles?.p95 ?? null,
|
|
805
|
+
numberQuantiles: hasNumberQuantiles(column.family) ? quantiles(row, `${key}_nq`, at) : null,
|
|
806
|
+
nonFinite: hasNumberQuantiles(column.family) ? num(row, `${key}_nonfinite`, at) : 0,
|
|
807
|
+
numericText: column.family === 'text' ? num(row, `${key}_numeric`, at) : 0,
|
|
808
|
+
dateText: column.family === 'text' ? num(row, `${key}_date`, at) : 0,
|
|
809
|
+
badCharacters: column.family === 'text' ? num(row, `${key}_bad`, at) : 0,
|
|
810
|
+
jsonRoots: column.family === 'json'
|
|
811
|
+
? Object.fromEntries(JSON_ROOT_TYPES.map(type => [type, num(row, `${key}_j_${type}`, at)]))
|
|
812
|
+
: null,
|
|
813
|
+
trueRows: column.family === 'boolean' ? num(row, `${key}_true`, at) : null,
|
|
814
|
+
categories: verdict.kind === 'category' ? [] : null,
|
|
815
|
+
shapeRows: null,
|
|
816
|
+
jsonRows: null,
|
|
817
|
+
oversize: 0,
|
|
818
|
+
};
|
|
819
|
+
}
|
|
820
|
+
function finishColumn(state, plan, run) {
|
|
821
|
+
const { kind, personalReason } = settled(state);
|
|
822
|
+
if (state.distinctEstimate === null)
|
|
823
|
+
throw new Error(`${plan.label}.${state.plan.name}: its distinct count was never measured.`);
|
|
824
|
+
// Quantiles over fewer owners than the threshold sit next to a real
|
|
825
|
+
// person's value (p5 of three rows is nearly the minimum), so they are kept
|
|
826
|
+
// only when at least `categoryMinOwners` owners have a value here. A table
|
|
827
|
+
// the owner table is unreachable from has no owner count (its `owners` is a
|
|
828
|
+
// row count, and 50 rows can be one person's), so nothing owner-gated
|
|
829
|
+
// leaves it: no percentiles, no share of true, as no value, shape or key.
|
|
830
|
+
const quantilesAllowed = plan.reach.kind !== 'unreachable' && state.owners >= run.categoryMinOwners;
|
|
831
|
+
// A percentile of a personal number is itself a personal value (the median
|
|
832
|
+
// birth date is somebody's birth date; the median of a numeric phone column
|
|
833
|
+
// is a phone number), so personal columns keep length quantiles only.
|
|
834
|
+
const personal = kind === 'personal' || kind === 'free-text';
|
|
835
|
+
// A true share over fewer owners describes those few owners (one user's
|
|
836
|
+
// flag in a one-row table), so it is withheld on the same rule.
|
|
837
|
+
const booleanTrueFraction = kind === 'boolean' && state.trueRows !== null && state.nonNull > 0 && quantilesAllowed
|
|
838
|
+
? share(state.trueRows, state.nonNull)
|
|
839
|
+
: null;
|
|
840
|
+
const shapes = (state.shapeRows ?? []).map(entry => ({ shape: entry.shape, fraction: share(entry.rows, state.nonNull) }));
|
|
841
|
+
const jsonKeys = jsonKeyProfiles(state);
|
|
842
|
+
return {
|
|
843
|
+
name: state.plan.name,
|
|
844
|
+
dataType: state.plan.catalog.dataType,
|
|
845
|
+
notNull: state.plan.catalog.notNull,
|
|
846
|
+
nullFraction: share(state.total - state.nonNull, state.total),
|
|
847
|
+
distinctEstimate: Math.round(state.distinctEstimate),
|
|
848
|
+
avgWidthBytes: state.avgWidth,
|
|
849
|
+
kind,
|
|
850
|
+
personalReason,
|
|
851
|
+
categories: state.categories,
|
|
852
|
+
lengthQuantiles: quantilesAllowed ? state.lengthQuantiles : null,
|
|
853
|
+
numberQuantiles: quantilesAllowed && !personal ? state.numberQuantiles : null,
|
|
854
|
+
booleanTrueFraction,
|
|
855
|
+
shapes,
|
|
856
|
+
jsonKeys,
|
|
857
|
+
oddities: oddities(state, plan, run),
|
|
858
|
+
};
|
|
859
|
+
}
|
|
860
|
+
function jsonKeyProfiles(state) {
|
|
861
|
+
const byPath = new Map();
|
|
862
|
+
for (const entry of state.jsonRows ?? []) {
|
|
863
|
+
const known = byPath.get(entry.path) ?? { present: entry.present, types: new Set() };
|
|
864
|
+
known.types.add(entry.type);
|
|
865
|
+
byPath.set(entry.path, known);
|
|
866
|
+
}
|
|
867
|
+
return [...byPath.entries()].map(([path, entry]) => ({
|
|
868
|
+
path,
|
|
869
|
+
types: [...entry.types].sort(),
|
|
870
|
+
presentFraction: share(entry.present, state.nonNull),
|
|
871
|
+
}));
|
|
872
|
+
}
|
|
873
|
+
function oddities(state, plan, run) {
|
|
874
|
+
const found = [];
|
|
875
|
+
const where = plan.samplePercent === null ? '' : ` in a ${pct(plan.samplePercent / 100)} sample`;
|
|
876
|
+
const nulls = state.total - state.nonNull;
|
|
877
|
+
if (state.badCharacters > 0) {
|
|
878
|
+
found.push({
|
|
879
|
+
kind: 'invalid-encoding',
|
|
880
|
+
rowCount: state.badCharacters,
|
|
881
|
+
jsonPath: null,
|
|
882
|
+
types: null,
|
|
883
|
+
description: `${rowsDo(state.badCharacters, where, 'contains', 'contain')} control characters or the Unicode replacement character, a sign of text decoded with the wrong encoding.`,
|
|
884
|
+
});
|
|
885
|
+
}
|
|
886
|
+
if (!state.plan.catalog.notNull && nulls > 0 && state.total > 0 && nulls / state.total <= 0.01) {
|
|
887
|
+
found.push({
|
|
888
|
+
kind: 'unexpected-null',
|
|
889
|
+
rowCount: nulls,
|
|
890
|
+
jsonPath: null,
|
|
891
|
+
types: null,
|
|
892
|
+
description: `${rowsDo(nulls, where, 'is', 'are')} null while ${pct(state.nonNull / state.total)} of rows have a value.`,
|
|
893
|
+
});
|
|
894
|
+
}
|
|
895
|
+
if (state.nonNull > 0 && state.plan.family === 'text') {
|
|
896
|
+
const numericShare = state.numericText / state.nonNull;
|
|
897
|
+
const dateShare = state.dateText / state.nonNull;
|
|
898
|
+
if (numericShare >= 0.9 && state.numericText < state.nonNull) {
|
|
899
|
+
const odd = state.nonNull - state.numericText;
|
|
900
|
+
found.push({ kind: 'type-mismatch', rowCount: odd, jsonPath: null, types: { expected: 'number', found: ['text'] },
|
|
901
|
+
description: `${rowsDo(odd, where, 'is', 'are')} not a number while ${pct(numericShare)} of values are.` });
|
|
902
|
+
}
|
|
903
|
+
else if (dateShare >= 0.9 && state.dateText < state.nonNull) {
|
|
904
|
+
const odd = state.nonNull - state.dateText;
|
|
905
|
+
found.push({ kind: 'type-mismatch', rowCount: odd, jsonPath: null, types: { expected: 'iso-date', found: ['text'] },
|
|
906
|
+
description: `${rowsDo(odd, where, 'is', 'are')} not an ISO date while ${pct(dateShare)} of values are.` });
|
|
907
|
+
}
|
|
908
|
+
}
|
|
909
|
+
if (state.nonFinite > 0) {
|
|
910
|
+
found.push({ kind: 'type-mismatch', rowCount: state.nonFinite, jsonPath: null, types: { expected: 'finite', found: ['nan-or-infinity'] },
|
|
911
|
+
description: `${rowsDo(state.nonFinite, where, 'holds', 'hold')} NaN or infinity instead of a finite value.` });
|
|
912
|
+
}
|
|
913
|
+
if (state.jsonRoots !== null && state.nonNull > 0) {
|
|
914
|
+
const ranked = Object.entries(state.jsonRoots).sort((left, right) => right[1] - left[1]);
|
|
915
|
+
const [dominant, dominantRows] = ranked[0];
|
|
916
|
+
const others = ranked.slice(1).filter(([, rows]) => rows > 0);
|
|
917
|
+
const otherRows = others.reduce((sum, [, rows]) => sum + rows, 0);
|
|
918
|
+
if (otherRows > 0 && dominantRows / state.nonNull >= 0.9) {
|
|
919
|
+
found.push({
|
|
920
|
+
kind: 'type-mismatch',
|
|
921
|
+
rowCount: otherRows,
|
|
922
|
+
jsonPath: null,
|
|
923
|
+
types: { expected: dominant, found: others.map(([type]) => type) },
|
|
924
|
+
description: `${rowsDo(otherRows, where, 'holds', 'hold')} a JSON ${others.map(([type]) => type).join(' or ')} where ${pct(dominantRows / state.nonNull)} of rows hold a JSON ${dominant}.`,
|
|
925
|
+
});
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
if (state.oversize > 0) {
|
|
929
|
+
found.push({
|
|
930
|
+
kind: 'oversize',
|
|
931
|
+
rowCount: state.oversize,
|
|
932
|
+
jsonPath: null,
|
|
933
|
+
types: null,
|
|
934
|
+
description: `${rowsDo(state.oversize, where, 'is', 'are')} over 1 KiB and more than ten times the column's 95th-percentile length.`,
|
|
935
|
+
});
|
|
936
|
+
}
|
|
937
|
+
if (state.shapeRows !== null && state.nonNull > 0) {
|
|
938
|
+
const covered = state.shapeRows.reduce((sum, entry) => sum + entry.rows, 0);
|
|
939
|
+
if (covered < state.nonNull && covered / state.nonNull >= 0.9) {
|
|
940
|
+
const odd = state.nonNull - covered;
|
|
941
|
+
found.push({
|
|
942
|
+
kind: 'shape-outlier',
|
|
943
|
+
rowCount: odd,
|
|
944
|
+
jsonPath: null,
|
|
945
|
+
types: null,
|
|
946
|
+
description: `${rowsDo(odd, where, 'has', 'have')} a shape shared by fewer than ${run.categoryMinOwners} owners, while ${pct(covered / state.nonNull)} of values have one of the column's common shapes.`,
|
|
947
|
+
});
|
|
948
|
+
}
|
|
949
|
+
}
|
|
950
|
+
if (state.jsonRows !== null && state.jsonRoots !== null) {
|
|
951
|
+
const objects = state.jsonRoots.object;
|
|
952
|
+
const byPath = new Map();
|
|
953
|
+
for (const entry of state.jsonRows) {
|
|
954
|
+
const known = byPath.get(entry.path) ?? { present: entry.present, types: new Map() };
|
|
955
|
+
known.types.set(entry.type, entry.rows);
|
|
956
|
+
byPath.set(entry.path, known);
|
|
957
|
+
}
|
|
958
|
+
for (const [path, entry] of byPath) {
|
|
959
|
+
const topLevel = !path.includes('.') && !path.includes('[]');
|
|
960
|
+
if (topLevel && objects > 0 && entry.present < objects && entry.present / objects >= 0.9) {
|
|
961
|
+
const missing = objects - entry.present;
|
|
962
|
+
found.push({
|
|
963
|
+
kind: 'missing-json-key',
|
|
964
|
+
rowCount: missing,
|
|
965
|
+
jsonPath: path,
|
|
966
|
+
types: null,
|
|
967
|
+
description: `${rowsDo(missing, where, 'is', 'are')} missing the key "${path}" that ${pct(entry.present / objects)} of JSON objects have.`,
|
|
968
|
+
});
|
|
969
|
+
}
|
|
970
|
+
const typed = [...entry.types.entries()].filter(([type]) => type !== 'null').sort((left, right) => right[1] - left[1]);
|
|
971
|
+
if (typed.length > 1 && typed[0][1] / entry.present >= 0.9) {
|
|
972
|
+
const odd = typed.slice(1).reduce((sum, [, rows]) => sum + rows, 0);
|
|
973
|
+
found.push({
|
|
974
|
+
kind: 'type-mismatch',
|
|
975
|
+
rowCount: odd,
|
|
976
|
+
jsonPath: path,
|
|
977
|
+
types: { expected: typed[0][0], found: typed.slice(1).map(([type]) => type) },
|
|
978
|
+
description: `${rowsDo(odd, where, 'holds', 'hold')} "${path}" as a JSON ${typed.slice(1).map(([type]) => type).join(' or ')} where ${pct(typed[0][1] / entry.present)} hold it as a ${typed[0][0]}.`,
|
|
979
|
+
});
|
|
980
|
+
}
|
|
981
|
+
const nullRows = entry.types.get('null') ?? 0;
|
|
982
|
+
if (nullRows > 0 && nullRows / entry.present <= 0.01) {
|
|
983
|
+
found.push({
|
|
984
|
+
kind: 'unexpected-null',
|
|
985
|
+
rowCount: nullRows,
|
|
986
|
+
jsonPath: path,
|
|
987
|
+
types: null,
|
|
988
|
+
description: `${rowsDo(nullRows, where, 'has', 'have')} "${path}" set to JSON null while ${pct(1 - nullRows / entry.present)} of rows with the key have a value.`,
|
|
989
|
+
});
|
|
990
|
+
}
|
|
991
|
+
}
|
|
992
|
+
}
|
|
993
|
+
return found;
|
|
994
|
+
}
|
|
995
|
+
/* ------------------------------------------------------------ command */
|
|
996
|
+
function writeProfileFile(path, profile) {
|
|
997
|
+
const target = resolve(path);
|
|
998
|
+
const temporary = `${target}.tmp-${process.pid}`;
|
|
999
|
+
writeFileSync(temporary, `${JSON.stringify(profile, null, 2)}\n`, { encoding: 'utf8', mode: 0o600 });
|
|
1000
|
+
renameSync(temporary, target);
|
|
1001
|
+
}
|
|
1002
|
+
export async function databaseProfileCommand(options) {
|
|
1003
|
+
const settings = parseDatabaseProfileSettings(options);
|
|
1004
|
+
const connectionString = readConnectionString(settings.urlEnv);
|
|
1005
|
+
const db = new ReadOnlyDatabase(connectionString, settings.concurrency, settings.statementTimeoutMs);
|
|
1006
|
+
const log = (line) => { process.stderr.write(`${line}\n`); };
|
|
1007
|
+
let profile;
|
|
1008
|
+
try {
|
|
1009
|
+
profile = await profileDatabase(db, settings, log);
|
|
1010
|
+
}
|
|
1011
|
+
finally {
|
|
1012
|
+
await db.close();
|
|
1013
|
+
}
|
|
1014
|
+
writeProfileFile(settings.out, profile);
|
|
1015
|
+
const columns = profile.tables.flatMap(table => table.columns);
|
|
1016
|
+
const kept = columns.filter(column => (column.categories ?? []).length > 0).length;
|
|
1017
|
+
const personal = columns.filter(column => column.kind === 'personal' || column.kind === 'free-text').length;
|
|
1018
|
+
console.log(chalk.green(`Wrote ${settings.out}: ${profile.tables.length} tables, ${columns.length} columns; ${kept} keep category values, ${personal} are personal (values never leave).`));
|
|
1019
|
+
console.log(`Review exactly what would be sent: haystack db profile show ${settings.out}`);
|
|
1020
|
+
}
|