@haystackeditor/cli 0.17.1 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +210 -24
- package/dist/commands/case-batch-contract.js +2 -2
- package/dist/commands/case-batch.js +10 -18
- package/dist/commands/crawl-contract.js +9 -0
- package/dist/commands/db-profile-contract.js +419 -0
- package/dist/commands/db-profile-database.js +289 -0
- package/dist/commands/db-profile-distinct.js +97 -0
- package/dist/commands/db-profile-owners.js +94 -0
- package/dist/commands/db-profile-privacy.js +450 -0
- package/dist/commands/db-profile-show.js +218 -0
- package/dist/commands/db-profile-sql-literals.js +140 -0
- package/dist/commands/db-profile-sql.js +362 -0
- package/dist/commands/db-profile-upload.js +88 -0
- package/dist/commands/db-profile.js +1020 -0
- package/dist/commands/fleet-policy-contract.js +28 -0
- package/dist/commands/fleet-policy.js +305 -0
- package/dist/commands/precompute-delivery-worker.js +17 -3
- package/dist/commands/precompute-delivery.js +19 -6
- package/dist/commands/verify-hosted.js +20 -0
- package/dist/commands/verify-precompute.js +161 -97
- package/dist/commands/verify.js +608 -52
- package/dist/index.js +178 -20
- package/dist/schema.js +1 -0
- package/dist/types.js +3 -0
- package/dist/utils/verify-base.js +31 -0
- package/package.json +4 -1
- package/schemas/verify.v1.json +237 -0
- package/dist/commands/combination-search-hook-contract.js +0 -1
- package/dist/commands/combination-search-hook.js +0 -139
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The profiler's only door into the customer's database.
|
|
3
|
+
*
|
|
4
|
+
* Every statement runs in its own READ ONLY transaction with a statement
|
|
5
|
+
* timeout and a lock timeout, and every one of those transactions reads the
|
|
6
|
+
* same REPEATABLE READ snapshot (exported by one more transaction that stays
|
|
7
|
+
* open for the run, `pg_export_snapshot`), on a connection that starts with
|
|
8
|
+
* `default_transaction_read_only = on` (a startup parameter, so it applies to
|
|
9
|
+
* the profiler's own server session and never to a pooled connection another
|
|
10
|
+
* client reuses), and both read-only settings are checked before the
|
|
11
|
+
* statement runs. The transaction is always rolled back: nothing the profiler
|
|
12
|
+
* does is ever meant to be kept. Only single SELECT / WITH statements are
|
|
13
|
+
* accepted, so a bug that built anything else fails here instead of running.
|
|
14
|
+
*
|
|
15
|
+
* It works on a read replica (hot standby) too. Nothing here writes, locks
|
|
16
|
+
* for write or creates a temporary table, and the isolation level is named
|
|
17
|
+
* (REPEATABLE READ) rather than inherited, because a standby refuses every
|
|
18
|
+
* statement when its default_transaction_isolation is serializable.
|
|
19
|
+
*/
|
|
20
|
+
import pg from 'pg';
|
|
21
|
+
import { parse as parseConnectionString } from 'pg-connection-string';
|
|
22
|
+
/** A table lock the profiler cannot get within this time fails the run rather
|
|
23
|
+
* than queueing behind (and in front of) a migration's exclusive lock. */
|
|
24
|
+
export const LOCK_TIMEOUT_MS = 5_000;
|
|
25
|
+
const READ_STATEMENT = /^\s*(SELECT|WITH)\b/iu;
|
|
26
|
+
export function assertReadStatement(sql, label) {
|
|
27
|
+
if (!READ_STATEMENT.test(sql))
|
|
28
|
+
throw new Error(`${label}: refusing to run a statement that is not a SELECT.`);
|
|
29
|
+
if (sql.includes(';'))
|
|
30
|
+
throw new Error(`${label}: refusing to run more than one statement at once.`);
|
|
31
|
+
}
|
|
32
|
+
function errorText(error) {
|
|
33
|
+
if (error instanceof Error) {
|
|
34
|
+
const code = error.code;
|
|
35
|
+
return typeof code === 'string' && /^[0-9A-Z]{5}$/u.test(code) ? `${error.message} (SQLSTATE ${code})` : error.message;
|
|
36
|
+
}
|
|
37
|
+
return String(error);
|
|
38
|
+
}
|
|
39
|
+
/** What to change when a read replica cancels a query because replaying the
|
|
40
|
+
* primary's changes (a vacuum removing rows the query still reads) could not
|
|
41
|
+
* wait for it any longer. */
|
|
42
|
+
export const RECOVERY_CONFLICT_HINT = 'The read replica canceled the query to keep replaying the primary; turn on hot_standby_feedback on the replica, or raise its max_standby_streaming_delay (on RDS and Cloud SQL, in its parameter group or flags) to longer than --statement-timeout, then run again.';
|
|
43
|
+
function isRecoveryConflict(error) {
|
|
44
|
+
if (!(error instanceof Error))
|
|
45
|
+
return false;
|
|
46
|
+
// A READ COMMITTED read-only statement meets serialization_failure only as
|
|
47
|
+
// a recovery conflict; the message covers servers with English messages
|
|
48
|
+
// that report it under another code (a buffer-pin deadlock is 40P01).
|
|
49
|
+
return error.code === '40001' || /conflict with recovery/iu.test(error.message);
|
|
50
|
+
}
|
|
51
|
+
const READ_ONLY_STARTUP = '-c default_transaction_read_only=on';
|
|
52
|
+
/**
|
|
53
|
+
* The connection settings of a connection string, parsed by pg-connection-string
|
|
54
|
+
* (the parser pg itself runs on a `connectionString`), so every URL pg accepts
|
|
55
|
+
* works, a Unix socket too (`postgresql://user@/db?host=/var/run/postgresql`,
|
|
56
|
+
* which WHATWG `URL` refuses), with `-c default_transaction_read_only=on`
|
|
57
|
+
* added to the `options` startup parameter (after any options it already
|
|
58
|
+
* has). The pool is given these settings, never the string: pg would let the
|
|
59
|
+
* string override them. A `connectionString` inside the string would do the
|
|
60
|
+
* same, so it is refused. The string is never echoed: it may hold a password.
|
|
61
|
+
*/
|
|
62
|
+
export function readOnlyConnectionSettings(connectionString) {
|
|
63
|
+
let parsed;
|
|
64
|
+
try {
|
|
65
|
+
parsed = parseConnectionString(connectionString);
|
|
66
|
+
}
|
|
67
|
+
catch (error) {
|
|
68
|
+
throw new Error(`The connection string is not a valid postgres:// URL (${error instanceof Error ? error.name : 'parse error'}).`);
|
|
69
|
+
}
|
|
70
|
+
if (parsed.connectionString !== undefined) {
|
|
71
|
+
throw new Error('The connection string sets connectionString, which would replace the read-only startup option; remove it.');
|
|
72
|
+
}
|
|
73
|
+
const existing = parsed.options;
|
|
74
|
+
return {
|
|
75
|
+
...parsed,
|
|
76
|
+
options: existing === undefined || existing.trim() === '' ? READ_ONLY_STARTUP : `${existing} ${READ_ONLY_STARTUP}`,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
/** The first server version (17) with the transaction_timeout setting. */
|
|
80
|
+
const TRANSACTION_TIMEOUT_VERSION_NUM = 170_000;
|
|
81
|
+
/** A `pg_export_snapshot()` identifier, checked before it is embedded in SQL. */
|
|
82
|
+
const SNAPSHOT_ID = /^[0-9A-Fa-f]+(-[0-9A-Fa-f]+)+$/u;
|
|
83
|
+
export class ReadOnlyDatabase {
|
|
84
|
+
concurrency;
|
|
85
|
+
pool;
|
|
86
|
+
statementTimeoutMs;
|
|
87
|
+
poolError = null;
|
|
88
|
+
/** The transaction whose snapshot every query reads; opened by the first query. */
|
|
89
|
+
snapshot = null;
|
|
90
|
+
snapshotError = null;
|
|
91
|
+
/** Pooled sessions whose transaction timeouts are already off (see `connect`). */
|
|
92
|
+
untimed = new WeakSet();
|
|
93
|
+
constructor(connectionString, concurrency, statementTimeoutMs) {
|
|
94
|
+
this.concurrency = concurrency;
|
|
95
|
+
this.statementTimeoutMs = statementTimeoutMs;
|
|
96
|
+
this.pool = new pg.Pool({
|
|
97
|
+
// The string's own application_name wins, as it did when pg parsed it.
|
|
98
|
+
application_name: 'haystack-db-profiler',
|
|
99
|
+
...readOnlyConnectionSettings(connectionString),
|
|
100
|
+
// Pool and timeout settings come last: nothing in the string changes them.
|
|
101
|
+
// One more than the queries in flight: the connection holding the snapshot.
|
|
102
|
+
max: concurrency + 1,
|
|
103
|
+
// Never keep a customer connection open longer than the run needs.
|
|
104
|
+
idleTimeoutMillis: 10_000,
|
|
105
|
+
connectionTimeoutMillis: 30_000,
|
|
106
|
+
});
|
|
107
|
+
this.pool.on('error', error => {
|
|
108
|
+
// An idle connection died (server restart, network). Remember it; the
|
|
109
|
+
// next query reports it instead of the process crashing.
|
|
110
|
+
this.poolError = error;
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
async connect(label) {
|
|
114
|
+
if (this.poolError)
|
|
115
|
+
throw new Error(`${label}: the database connection failed earlier: ${errorText(this.poolError)}`);
|
|
116
|
+
let client;
|
|
117
|
+
try {
|
|
118
|
+
client = await this.pool.connect();
|
|
119
|
+
}
|
|
120
|
+
catch (error) {
|
|
121
|
+
const text = errorText(error);
|
|
122
|
+
const pooler = /unsupported startup parameter/iu.test(text)
|
|
123
|
+
? ' The connection goes through a pooler that refuses startup options (PgBouncer does by default); point the profiler at Postgres directly.'
|
|
124
|
+
: '';
|
|
125
|
+
throw new Error(`${label}: could not connect to the database: ${text}.${pooler}`);
|
|
126
|
+
}
|
|
127
|
+
if (this.untimed.has(client))
|
|
128
|
+
return client;
|
|
129
|
+
try {
|
|
130
|
+
// The snapshot's transaction sits idle for the whole run, and a query
|
|
131
|
+
// may run longer than a hosted default allows a transaction: a role or
|
|
132
|
+
// database default for idle_in_transaction_session_timeout (or, from
|
|
133
|
+
// Postgres 17, transaction_timeout) would end it and the run with it.
|
|
134
|
+
// These are the profiler's own sessions, and any role may set both.
|
|
135
|
+
// SHOW and SET take no snapshot, so they run outside a transaction even
|
|
136
|
+
// on a standby whose default isolation is serializable.
|
|
137
|
+
const [{ server_version_num: num }] = (await client.query('SHOW server_version_num')).rows;
|
|
138
|
+
await client.query('SET idle_in_transaction_session_timeout = 0');
|
|
139
|
+
if (Number(num) >= TRANSACTION_TIMEOUT_VERSION_NUM)
|
|
140
|
+
await client.query('SET transaction_timeout = 0');
|
|
141
|
+
}
|
|
142
|
+
catch (error) {
|
|
143
|
+
client.release(error instanceof Error ? error : new Error(String(error)));
|
|
144
|
+
throw new Error(`${label}: could not turn off the session's transaction timeouts: ${errorText(error)}`);
|
|
145
|
+
}
|
|
146
|
+
this.untimed.add(client);
|
|
147
|
+
return client;
|
|
148
|
+
}
|
|
149
|
+
/** Settings and read-only checks every transaction starts with, after any
|
|
150
|
+
* SET TRANSACTION SNAPSHOT (which must come first). */
|
|
151
|
+
async prepare(client) {
|
|
152
|
+
// SET LOCAL lasts until the rollback, so a pooled connection (including
|
|
153
|
+
// behind PgBouncer in transaction mode) is left unchanged.
|
|
154
|
+
await client.query(`SET LOCAL statement_timeout = ${Math.round(this.statementTimeoutMs)}`);
|
|
155
|
+
await client.query(`SET LOCAL lock_timeout = ${LOCK_TIMEOUT_MS}`);
|
|
156
|
+
await client.query('SET LOCAL standard_conforming_strings = on');
|
|
157
|
+
await client.query('SET LOCAL search_path = pg_catalog, pg_temp');
|
|
158
|
+
const state = await client.query(`SELECT current_setting('transaction_read_only') AS read_only, current_setting('default_transaction_read_only') AS default_read_only`);
|
|
159
|
+
if (state.rows[0]?.read_only !== 'on') {
|
|
160
|
+
throw new Error('the transaction is not read only; refusing to run');
|
|
161
|
+
}
|
|
162
|
+
if (state.rows[0]?.default_read_only !== 'on') {
|
|
163
|
+
throw new Error('the session did not start with default_transaction_read_only = on (a pooler may have dropped the startup option); refusing to run');
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* Every query reads the database as of one moment. The first query opens a
|
|
168
|
+
* REPEATABLE READ, READ ONLY transaction on a connection of its own,
|
|
169
|
+
* exports its snapshot and keeps it open until `close`; every query's own
|
|
170
|
+
* transaction imports that snapshot. Otherwise a row written between two
|
|
171
|
+
* queries could make one count exceed another (a share above 1).
|
|
172
|
+
*/
|
|
173
|
+
exportedSnapshot() {
|
|
174
|
+
this.snapshot ??= (async () => {
|
|
175
|
+
const label = 'read-only snapshot';
|
|
176
|
+
const client = await this.connect(label);
|
|
177
|
+
// A checked-out connection that dies emits here, not on the pool.
|
|
178
|
+
client.on('error', error => { this.snapshotError = error; });
|
|
179
|
+
try {
|
|
180
|
+
await client.query('BEGIN TRANSACTION ISOLATION LEVEL REPEATABLE READ, READ ONLY');
|
|
181
|
+
await this.prepare(client);
|
|
182
|
+
const { rows } = await client.query('SELECT pg_export_snapshot() AS id');
|
|
183
|
+
const id = rows[0]?.id;
|
|
184
|
+
if (typeof id !== 'string' || !SNAPSHOT_ID.test(id))
|
|
185
|
+
throw new Error('the server returned no snapshot identifier');
|
|
186
|
+
return { client, id };
|
|
187
|
+
}
|
|
188
|
+
catch (error) {
|
|
189
|
+
client.release(error instanceof Error ? error : new Error(String(error)));
|
|
190
|
+
throw new Error(`${label}: ${errorText(error)}${isRecoveryConflict(error) ? ` ${RECOVERY_CONFLICT_HINT}` : ''}`);
|
|
191
|
+
}
|
|
192
|
+
})();
|
|
193
|
+
return this.snapshot;
|
|
194
|
+
}
|
|
195
|
+
async query(label, sql, params = []) {
|
|
196
|
+
assertReadStatement(sql, label);
|
|
197
|
+
const snapshot = await this.exportedSnapshot();
|
|
198
|
+
if (this.snapshotError) {
|
|
199
|
+
throw new Error(`${label}: the connection holding the profiler's snapshot failed: ${errorText(this.snapshotError)}`);
|
|
200
|
+
}
|
|
201
|
+
const client = await this.connect(label);
|
|
202
|
+
let failure = null;
|
|
203
|
+
let inTransaction = false;
|
|
204
|
+
try {
|
|
205
|
+
// The isolation level is named, never inherited: a standby refuses
|
|
206
|
+
// every transaction whose default is serializable.
|
|
207
|
+
await client.query('BEGIN TRANSACTION ISOLATION LEVEL REPEATABLE READ, READ ONLY');
|
|
208
|
+
inTransaction = true;
|
|
209
|
+
await client.query(`SET TRANSACTION SNAPSHOT '${snapshot.id}'`);
|
|
210
|
+
await this.prepare(client);
|
|
211
|
+
const result = await client.query(sql, params);
|
|
212
|
+
return result.rows;
|
|
213
|
+
}
|
|
214
|
+
catch (error) {
|
|
215
|
+
failure = error;
|
|
216
|
+
throw new Error(`${label}: ${errorText(error)}${isRecoveryConflict(error) ? ` ${RECOVERY_CONFLICT_HINT}` : ''}`);
|
|
217
|
+
}
|
|
218
|
+
finally {
|
|
219
|
+
let rollbackFailure = null;
|
|
220
|
+
if (inTransaction) {
|
|
221
|
+
try {
|
|
222
|
+
await client.query('ROLLBACK');
|
|
223
|
+
}
|
|
224
|
+
catch (error) {
|
|
225
|
+
rollbackFailure = error;
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
// A connection whose rollback failed is in an unknown state: destroy it.
|
|
229
|
+
const broken = rollbackFailure ?? (inTransaction ? null : failure);
|
|
230
|
+
client.release(broken instanceof Error ? broken : broken === null ? undefined : new Error(String(broken)));
|
|
231
|
+
if (rollbackFailure !== null && failure === null) {
|
|
232
|
+
// eslint-disable-next-line no-unsafe-finally -- the rollback failing is itself the failure to report
|
|
233
|
+
throw new Error(`${label}: rolling back the read-only transaction failed: ${errorText(rollbackFailure)}`);
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
async close() {
|
|
238
|
+
let holder = null;
|
|
239
|
+
if (this.snapshot !== null) {
|
|
240
|
+
// A snapshot that failed to open was already reported by the query that
|
|
241
|
+
// opened it; there is nothing to roll back then.
|
|
242
|
+
holder = await this.snapshot.then(value => value, () => null);
|
|
243
|
+
}
|
|
244
|
+
if (holder !== null && this.snapshotError !== null) {
|
|
245
|
+
// The server already ended that session (and its transaction); the
|
|
246
|
+
// query that found out reported it, so there is nothing to roll back
|
|
247
|
+
// and no second error to put in front of the first.
|
|
248
|
+
holder.client.release(this.snapshotError);
|
|
249
|
+
}
|
|
250
|
+
else if (holder !== null) {
|
|
251
|
+
try {
|
|
252
|
+
await holder.client.query('ROLLBACK');
|
|
253
|
+
holder.client.release();
|
|
254
|
+
}
|
|
255
|
+
catch (error) {
|
|
256
|
+
holder.client.release(error instanceof Error ? error : new Error(String(error)));
|
|
257
|
+
await this.pool.end();
|
|
258
|
+
throw new Error(`read-only snapshot: rolling back failed: ${errorText(error)}`);
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
await this.pool.end();
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
/** Run `work` over `items` with at most `limit` in flight. After the first
|
|
265
|
+
* failure no new item starts; in-flight items finish, then the first failure
|
|
266
|
+
* is thrown. */
|
|
267
|
+
export async function runBounded(items, limit, work) {
|
|
268
|
+
let next = 0;
|
|
269
|
+
const failures = [];
|
|
270
|
+
const worker = async () => {
|
|
271
|
+
while (failures.length === 0 && next < items.length) {
|
|
272
|
+
const item = items[next];
|
|
273
|
+
next += 1;
|
|
274
|
+
try {
|
|
275
|
+
await work(item);
|
|
276
|
+
}
|
|
277
|
+
catch (error) {
|
|
278
|
+
failures.push(error);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
};
|
|
282
|
+
await Promise.all(Array.from({ length: Math.max(1, Math.min(limit, items.length)) }, () => worker()));
|
|
283
|
+
if (failures.length === 1)
|
|
284
|
+
throw failures[0];
|
|
285
|
+
if (failures.length > 1) {
|
|
286
|
+
const first = failures[0] instanceof Error ? failures[0].message : String(failures[0]);
|
|
287
|
+
throw new Error(`${first} (and ${failures.length - 1} other in-flight ${failures.length === 2 ? 'query' : 'queries'} also failed)`);
|
|
288
|
+
}
|
|
289
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Distinct-value estimates for sampled tables, kept free of database access
|
|
3
|
+
* so the arithmetic can be run on its own.
|
|
4
|
+
*
|
|
5
|
+
* A `TABLESAMPLE SYSTEM` sample reads whole pages, so rows that sit together
|
|
6
|
+
* (an order's line items, a user's rows written in one burst) arrive
|
|
7
|
+
* together. The estimate therefore counts, for each distinct value, the
|
|
8
|
+
* sampled PAGES holding it rather than its rows: a value clustered on one page
|
|
9
|
+
* is then seen once, which is what it is to a page sample. Each page is in the
|
|
10
|
+
* sample with probability q, so these page counts are a Bernoulli(q) sample of
|
|
11
|
+
* each value's pages, and the classical estimators apply to them. Only the
|
|
12
|
+
* summary (how many values were found on exactly j pages, for each j) leaves
|
|
13
|
+
* the database.
|
|
14
|
+
*
|
|
15
|
+
* The unseen values are the geometric mean of two estimates that err in
|
|
16
|
+
* opposite directions on skewed data: the adaptive estimator (AE) of Charikar,
|
|
17
|
+
* Chaudhuri, Motwani and Narasayya ("Towards estimation error guarantees for
|
|
18
|
+
* distinct values", PODS 2000), which reads low on heavy tails, and Shlosser's
|
|
19
|
+
* estimator, which Haas et al. recommend for skew and which reads high there.
|
|
20
|
+
* The result is held between two bounds of Bernoulli page sampling:
|
|
21
|
+
* - at most f1 (1 - q) / q unseen values, reached when every value lives on
|
|
22
|
+
* exactly one page;
|
|
23
|
+
* - at least the Chao1 bound for sampling without replacement (Chao and
|
|
24
|
+
* Lin, 2012), f1^2 / (2 f2 n / (n - 1) + f1 q / (1 - q)).
|
|
25
|
+
* Measured against exact counts on a skewed (Zipf) 2 GB dataset, at page
|
|
26
|
+
* samples from 0.24% to 66%, it was the closest of the estimators tried (Duj1,
|
|
27
|
+
* which the planner uses; Duj2; GEE; Chao1; Chao-Lee; Shlosser; AE; a
|
|
28
|
+
* unique-value mixture), and counting pages beat counting rows for each. No
|
|
29
|
+
* sample estimate is exact. There, heavy-tailed columns read 1.1x to 1.6x
|
|
30
|
+
* under a 2.4% sample and up to 2.3x under 1%, and the worst column read
|
|
31
|
+
* 0.17x (a JSON column of mostly unique values under a sample below 1%);
|
|
32
|
+
* the planner's estimate read 0.08x and 0.02x on those columns. Raising
|
|
33
|
+
* --scan-max-mb narrows it; a table read in full is counted exactly.
|
|
34
|
+
*/
|
|
35
|
+
function tally(fingerprint, term) {
|
|
36
|
+
return fingerprint.reduce((sum, entry) => sum + entry.values * term(entry.pages), 0);
|
|
37
|
+
}
|
|
38
|
+
function valuesOn(fingerprint, pages) {
|
|
39
|
+
return fingerprint.find(entry => entry.pages === pages)?.values ?? 0;
|
|
40
|
+
}
|
|
41
|
+
/** AE's unseen values (m - f1 - f2), at most `most`. */
|
|
42
|
+
function adaptiveUnseen(fingerprint, most) {
|
|
43
|
+
const f1 = valuesOn(fingerprint, 1);
|
|
44
|
+
const f2 = valuesOn(fingerprint, 2);
|
|
45
|
+
const rate = f1 + 2 * f2;
|
|
46
|
+
const tailWeight = tally(fingerprint, pages => (pages >= 3 ? Math.exp(-pages) : 0));
|
|
47
|
+
const tailPages = tally(fingerprint, pages => (pages >= 3 ? pages * Math.exp(-pages) : 0));
|
|
48
|
+
// m is AE's count of low-frequency values in the table. It solves g(m) = 0,
|
|
49
|
+
// and g(f1 + f2) < 0 whenever f1 > 0.
|
|
50
|
+
const g = (m) => {
|
|
51
|
+
const seen = Math.exp(-rate / m);
|
|
52
|
+
return m - f1 - f2 - (f1 * (tailWeight + m * seen)) / (tailPages + rate * seen);
|
|
53
|
+
};
|
|
54
|
+
let low = f1 + f2;
|
|
55
|
+
let high = f1 + f2 + most;
|
|
56
|
+
if (g(high) <= 0)
|
|
57
|
+
return most;
|
|
58
|
+
for (let step = 0; step < 200 && high - low > 1e-9 * high; step += 1) {
|
|
59
|
+
const middle = (low + high) / 2;
|
|
60
|
+
if (g(middle) > 0)
|
|
61
|
+
high = middle;
|
|
62
|
+
else
|
|
63
|
+
low = middle;
|
|
64
|
+
}
|
|
65
|
+
return (low + high) / 2 - f1 - f2;
|
|
66
|
+
}
|
|
67
|
+
/** Shlosser's unseen values. */
|
|
68
|
+
function shlosserUnseen(fingerprint, q) {
|
|
69
|
+
const f1 = valuesOn(fingerprint, 1);
|
|
70
|
+
return (f1 * tally(fingerprint, pages => (1 - q) ** pages))
|
|
71
|
+
/ tally(fingerprint, pages => pages * q * (1 - q) ** (pages - 1));
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Estimated distinct non-null values in the whole table from a page sample
|
|
75
|
+
* holding a fraction `q` of its rows (the sampled rows over the table's row
|
|
76
|
+
* count, so the estimate scales like the row count), never above
|
|
77
|
+
* `nonNullRows` (the table's estimated non-null rows) or below the values the
|
|
78
|
+
* sample holds. A `q` of 1 or more (the planner counts no more rows than the
|
|
79
|
+
* sample holds) leaves no value unseen.
|
|
80
|
+
*/
|
|
81
|
+
export function estimateDistinct(fingerprint, q, nonNullRows) {
|
|
82
|
+
const d = tally(fingerprint, () => 1);
|
|
83
|
+
if (d === 0)
|
|
84
|
+
return 0;
|
|
85
|
+
if (!(q > 0))
|
|
86
|
+
throw new Error(`A sample holding values must hold some share of the table's rows (got ${q}).`);
|
|
87
|
+
const f1 = valuesOn(fingerprint, 1);
|
|
88
|
+
if (f1 === 0 || q >= 1)
|
|
89
|
+
return Math.round(d);
|
|
90
|
+
const f2 = valuesOn(fingerprint, 2);
|
|
91
|
+
const valuePages = tally(fingerprint, pages => pages);
|
|
92
|
+
const most = (f1 * (1 - q)) / q;
|
|
93
|
+
// With no value on exactly two pages the lower bound meets the upper one.
|
|
94
|
+
const least = f2 === 0 ? most : (f1 * f1) / ((2 * f2 * valuePages) / (valuePages - 1) + (f1 * q) / (1 - q));
|
|
95
|
+
const unseen = Math.sqrt(adaptiveUnseen(fingerprint, most) * shlosserUnseen(fingerprint, q));
|
|
96
|
+
return Math.round(Math.max(d, Math.min(nonNullRows, d + Math.min(most, Math.max(least, unseen)))));
|
|
97
|
+
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Who owns a row. With `--owner-table`, a row's owners are the owner-table
|
|
3
|
+
* rows reached by following foreign keys from the row (child to parent), e.g.
|
|
4
|
+
* `order_items -> orders -> users`. Every shortest path is used; when a table
|
|
5
|
+
* reaches the owner table by more than one path (a `buyer_id` and a
|
|
6
|
+
* `seller_id`), a value must meet the owner threshold along every path.
|
|
7
|
+
*
|
|
8
|
+
* Shared by the profiler (which builds SQL from the paths) and `profile show`
|
|
9
|
+
* (which rebuilds the same paths from the foreign keys in the file, so the
|
|
10
|
+
* review explains how owners were counted).
|
|
11
|
+
*/
|
|
12
|
+
/** Longer chains are unusual and each hop is a join in every value query. */
|
|
13
|
+
export const OWNER_PATH_MAX_HOPS = 4;
|
|
14
|
+
/** More shortest paths than this points at a modelling problem worth a human look. */
|
|
15
|
+
export const OWNER_PATH_MAX_PATHS = 8;
|
|
16
|
+
export function sameTable(left, right) {
|
|
17
|
+
return left.schema === right.schema && left.table === right.table;
|
|
18
|
+
}
|
|
19
|
+
export function tableKey(ref) {
|
|
20
|
+
return `${ref.schema}.${ref.table}`;
|
|
21
|
+
}
|
|
22
|
+
export function ownerReach(table, owner, edges) {
|
|
23
|
+
if (owner === null)
|
|
24
|
+
return { kind: 'rows' };
|
|
25
|
+
if (sameTable(table, owner))
|
|
26
|
+
return { kind: 'self' };
|
|
27
|
+
// Breadth-first over child -> parent edges, remembering every edge that
|
|
28
|
+
// reaches a table at its shortest distance, then enumerate the paths.
|
|
29
|
+
const outgoing = new Map();
|
|
30
|
+
for (const edge of edges) {
|
|
31
|
+
if (sameTable(edge.from, edge.to))
|
|
32
|
+
continue;
|
|
33
|
+
const key = tableKey(edge.from);
|
|
34
|
+
outgoing.set(key, [...(outgoing.get(key) ?? []), edge]);
|
|
35
|
+
}
|
|
36
|
+
const distance = new Map([[tableKey(table), 0]]);
|
|
37
|
+
const arrivals = new Map();
|
|
38
|
+
let frontier = [table];
|
|
39
|
+
for (let hop = 1; hop <= OWNER_PATH_MAX_HOPS && frontier.length > 0 && !distance.has(tableKey(owner)); hop += 1) {
|
|
40
|
+
const next = [];
|
|
41
|
+
for (const current of frontier) {
|
|
42
|
+
for (const edge of outgoing.get(tableKey(current)) ?? []) {
|
|
43
|
+
const target = tableKey(edge.to);
|
|
44
|
+
const known = distance.get(target);
|
|
45
|
+
if (known === undefined) {
|
|
46
|
+
distance.set(target, hop);
|
|
47
|
+
arrivals.set(target, [edge]);
|
|
48
|
+
next.push(edge.to);
|
|
49
|
+
}
|
|
50
|
+
else if (known === hop) {
|
|
51
|
+
arrivals.get(target)?.push(edge);
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
frontier = next;
|
|
56
|
+
}
|
|
57
|
+
if (!distance.has(tableKey(owner)))
|
|
58
|
+
return { kind: 'unreachable' };
|
|
59
|
+
const paths = [];
|
|
60
|
+
const walk = (at, suffix) => {
|
|
61
|
+
if (sameTable(at, table)) {
|
|
62
|
+
paths.push(suffix);
|
|
63
|
+
return;
|
|
64
|
+
}
|
|
65
|
+
for (const edge of arrivals.get(tableKey(at)) ?? [])
|
|
66
|
+
walk(edge.from, [edge, ...suffix]);
|
|
67
|
+
};
|
|
68
|
+
walk(owner, []);
|
|
69
|
+
if (paths.length > OWNER_PATH_MAX_PATHS) {
|
|
70
|
+
throw new Error(`${tableKey(table)} reaches the owner table ${tableKey(owner)} by ${paths.length} different foreign-key paths `
|
|
71
|
+
+ `(more than ${OWNER_PATH_MAX_PATHS}); profile it separately with --schema, or pick an owner table it reaches more directly.`);
|
|
72
|
+
}
|
|
73
|
+
paths.sort((left, right) => describePath(left).localeCompare(describePath(right)));
|
|
74
|
+
return { kind: 'paths', paths };
|
|
75
|
+
}
|
|
76
|
+
export function describePath(path) {
|
|
77
|
+
return path.map(edge => `${edge.from.table}.${edge.columns.join('+')} -> ${tableKey(edge.to)}`).join(', then ');
|
|
78
|
+
}
|
|
79
|
+
export function describeReach(reach, owner) {
|
|
80
|
+
switch (reach.kind) {
|
|
81
|
+
case 'rows':
|
|
82
|
+
return 'each row counts as one owner (no owner table was named)';
|
|
83
|
+
case 'self':
|
|
84
|
+
return 'this is the owner table: each row is one owner';
|
|
85
|
+
case 'unreachable':
|
|
86
|
+
return `no foreign-key path (at most ${OWNER_PATH_MAX_HOPS} hops) reaches ${owner === null ? 'the owner table' : tableKey(owner)}, `
|
|
87
|
+
+ 'so owners cannot be counted and no value, shape, JSON key, percentile or share of true from this table is kept';
|
|
88
|
+
case 'paths':
|
|
89
|
+
return reach.paths.length === 1
|
|
90
|
+
? `owners reached through ${describePath(reach.paths[0])}`
|
|
91
|
+
: `owners reached through ${reach.paths.length} paths, and a value must pass on every one: `
|
|
92
|
+
+ reach.paths.map(path => `(${describePath(path)})`).join('; ');
|
|
93
|
+
}
|
|
94
|
+
}
|