squirreling 0.16.2 → 0.16.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/execute/aggregates.js +32 -36
- package/src/execute/fold.js +74 -0
- package/src/execute/streamingAggregate.js +81 -62
- package/src/execute/window.js +12 -17
- package/src/expression/regexp.js +6 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "squirreling",
|
|
3
|
-
"version": "0.16.
|
|
3
|
+
"version": "0.16.3",
|
|
4
4
|
"description": "Squirreling Async SQL Engine",
|
|
5
5
|
"author": "Hyperparam",
|
|
6
6
|
"homepage": "https://hyperparam.app",
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
"test": "vitest run"
|
|
40
40
|
},
|
|
41
41
|
"devDependencies": {
|
|
42
|
-
"@types/node": "26.
|
|
42
|
+
"@types/node": "26.4.0",
|
|
43
43
|
"@vitest/coverage-v8": "4.1.11",
|
|
44
44
|
"eslint": "9.39.4",
|
|
45
45
|
"eslint-plugin-jsdoc": "64.2.1",
|
|
@@ -3,6 +3,7 @@ import { derivedAlias } from '../expression/alias.js'
|
|
|
3
3
|
import { evaluateExpr } from '../expression/evaluate.js'
|
|
4
4
|
import { finalizeAccumulator, newAccumulator, updateAccumulator } from './accumulator.js'
|
|
5
5
|
import { executePlan, executeScan, selectColumnNames } from './execute.js'
|
|
6
|
+
import { foldEvaluatedRows } from './fold.js'
|
|
6
7
|
import { normalizeScanColumnResult } from './scanColumn.js'
|
|
7
8
|
import { sortEntriesByTerms } from './sort.js'
|
|
8
9
|
import { planStreamingAggregates, streamingHashAggregateRows, streamingScalarAggregateRows } from './streamingAggregate.js'
|
|
@@ -109,47 +110,42 @@ export function executeHashAggregate(plan, context) {
|
|
|
109
110
|
}
|
|
110
111
|
context.signal?.throwIfAborted()
|
|
111
112
|
|
|
112
|
-
// Group rows by GROUP BY keys.
|
|
113
|
-
//
|
|
114
|
-
//
|
|
115
|
-
//
|
|
116
|
-
// inner Promise.all wrapper when there's a single GROUP BY expression.
|
|
113
|
+
// Group rows by GROUP BY keys. Keys are evaluated in adaptive chunks
|
|
114
|
+
// so async cells (e.g. lazy parquet decode) overlap while evaluated
|
|
115
|
+
// key values stay byte-bounded. The single-key branch skips the inner
|
|
116
|
+
// Promise.all wrapper so synchronous cells stay cheap.
|
|
117
117
|
/** @type {Map<any, AsyncRow[]>} */
|
|
118
118
|
const groups = new Map()
|
|
119
119
|
const { groupBy } = plan
|
|
120
|
-
const
|
|
121
|
-
const singleExpr = singleKey ? groupBy[0] : null
|
|
120
|
+
const singleExpr = groupBy.length === 1 ? groupBy[0] : null
|
|
122
121
|
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
if (singleKey) {
|
|
133
|
-
for (let j = 0; j < chunkLen; j++) {
|
|
134
|
-
pending[j] = evaluateExpr({ node: singleExpr, row: allRows[chunkStart + j], context })
|
|
135
|
-
}
|
|
136
|
-
} else {
|
|
137
|
-
for (let j = 0; j < chunkLen; j++) {
|
|
138
|
-
const row = allRows[chunkStart + j]
|
|
139
|
-
pending[j] = Promise.all(groupBy.map(expr => evaluateExpr({ node: expr, row, context })))
|
|
140
|
-
}
|
|
141
|
-
}
|
|
142
|
-
const chunkKeys = await Promise.all(pending)
|
|
143
|
-
for (let j = 0; j < chunkLen; j++) {
|
|
144
|
-
const key = singleKey ? keyify(chunkKeys[j]) : keyify(...chunkKeys[j])
|
|
145
|
-
const row = allRows[chunkStart + j]
|
|
146
|
-
let group = groups.get(key)
|
|
147
|
-
if (!group) {
|
|
148
|
-
group = []
|
|
149
|
-
groups.set(key, group)
|
|
150
|
-
}
|
|
151
|
-
group.push(row)
|
|
122
|
+
/**
|
|
123
|
+
* @param {any} key
|
|
124
|
+
* @param {number} index
|
|
125
|
+
*/
|
|
126
|
+
function addToGroup(key, index) {
|
|
127
|
+
let group = groups.get(key)
|
|
128
|
+
if (!group) {
|
|
129
|
+
group = []
|
|
130
|
+
groups.set(key, group)
|
|
152
131
|
}
|
|
132
|
+
group.push(allRows[index])
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
if (singleExpr) {
|
|
136
|
+
await foldEvaluatedRows({
|
|
137
|
+
rows: allRows,
|
|
138
|
+
signal: context.signal,
|
|
139
|
+
evaluate: row => evaluateExpr({ node: singleExpr, row, context }),
|
|
140
|
+
fold: (value, index) => addToGroup(keyify(value), index),
|
|
141
|
+
})
|
|
142
|
+
} else {
|
|
143
|
+
await foldEvaluatedRows({
|
|
144
|
+
rows: allRows,
|
|
145
|
+
signal: context.signal,
|
|
146
|
+
evaluate: row => Promise.all(groupBy.map(expr => evaluateExpr({ node: expr, row, context }))),
|
|
147
|
+
fold: (values, index) => addToGroup(keyify(...values), index),
|
|
148
|
+
})
|
|
153
149
|
}
|
|
154
150
|
|
|
155
151
|
/** @type {{ row: AsyncRow, rows: AsyncRow[], outputRow: AsyncRow }[]} */
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import { yieldToEventLoop } from './yield.js'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* @import { AsyncRow } from '../types.js'
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
// Chunk sizing for foldEvaluatedRows. Dispatch starts small so one chunk of
|
|
8
|
+
// unexpectedly fat values cannot overshoot far, then adapts to the observed
|
|
9
|
+
// result sizes: small values grow the chunk toward MAX_CHUNK_ROWS so async
|
|
10
|
+
// cells still overlap, while large values (e.g. long string group keys)
|
|
11
|
+
// shrink it so in-flight results stay near CHUNK_BYTE_BUDGET instead of
|
|
12
|
+
// scaling with row count.
|
|
13
|
+
const INITIAL_CHUNK_ROWS = 64
|
|
14
|
+
const MAX_CHUNK_ROWS = 4000
|
|
15
|
+
const CHUNK_BYTE_BUDGET = 16 * 1024 * 1024
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Approximate retained bytes of an evaluated value. Strings dominate the
|
|
19
|
+
* workloads where size matters; everything else counts as a small constant.
|
|
20
|
+
*
|
|
21
|
+
* @param {unknown} value
|
|
22
|
+
* @returns {number}
|
|
23
|
+
*/
|
|
24
|
+
function valueBytes(value) {
|
|
25
|
+
if (typeof value === 'string') return value.length * 2
|
|
26
|
+
if (Array.isArray(value)) {
|
|
27
|
+
let bytes = 0
|
|
28
|
+
for (const item of value) bytes += valueBytes(item)
|
|
29
|
+
return bytes
|
|
30
|
+
}
|
|
31
|
+
return 16
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Evaluates a value for every row and folds each result in row order, holding
|
|
36
|
+
* at most one adaptively sized chunk of evaluated values. Rows in a chunk are
|
|
37
|
+
* dispatched together so async cells overlap, and the chunk boundary yields
|
|
38
|
+
* to the event loop so aborts can fire. Chunks are bounded by bytes, not row
|
|
39
|
+
* count: a fixed 4000-row chunk of ~90KB string keys would hold hundreds of
|
|
40
|
+
* megabytes of results at once.
|
|
41
|
+
*
|
|
42
|
+
* @template T
|
|
43
|
+
* @param {Object} options
|
|
44
|
+
* @param {AsyncRow[]} options.rows
|
|
45
|
+
* @param {(row: AsyncRow, index: number) => Promise<T>} options.evaluate
|
|
46
|
+
* @param {(value: T, index: number) => void} options.fold
|
|
47
|
+
* @param {AbortSignal} [options.signal]
|
|
48
|
+
* @returns {Promise<void>}
|
|
49
|
+
*/
|
|
50
|
+
export async function foldEvaluatedRows({ rows, evaluate, fold, signal }) {
|
|
51
|
+
let chunkSize = INITIAL_CHUNK_ROWS
|
|
52
|
+
/** @type {Promise<T>[]} */
|
|
53
|
+
const pending = []
|
|
54
|
+
for (let start = 0; start < rows.length;) {
|
|
55
|
+
if (start > 0) {
|
|
56
|
+
await yieldToEventLoop()
|
|
57
|
+
signal?.throwIfAborted()
|
|
58
|
+
}
|
|
59
|
+
const end = Math.min(start + chunkSize, rows.length)
|
|
60
|
+
pending.length = end - start
|
|
61
|
+
for (let i = start; i < end; i++) {
|
|
62
|
+
pending[i - start] = evaluate(rows[i], i)
|
|
63
|
+
}
|
|
64
|
+
const values = await Promise.all(pending)
|
|
65
|
+
let bytes = 0
|
|
66
|
+
for (let j = 0; j < values.length; j++) {
|
|
67
|
+
bytes += valueBytes(values[j])
|
|
68
|
+
fold(values[j], start + j)
|
|
69
|
+
}
|
|
70
|
+
start = end
|
|
71
|
+
const bytesPerRow = Math.max(1, bytes / values.length)
|
|
72
|
+
chunkSize = Math.min(MAX_CHUNK_ROWS, Math.max(1, Math.floor(CHUNK_BYTE_BUDGET / bytesPerRow)))
|
|
73
|
+
}
|
|
74
|
+
}
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { selectedRowCount, valueAt } from '../backend/batch.js'
|
|
2
2
|
import { derivedAlias } from '../expression/alias.js'
|
|
3
3
|
import { compileBatchExpression } from '../expression/batch.js'
|
|
4
|
-
import {
|
|
4
|
+
import { evaluateExpr } from '../expression/evaluate.js'
|
|
5
5
|
import { collectColumnsFromExpr } from '../plan/columns.js'
|
|
6
6
|
import { isAggregateFunc } from '../validation/functions.js'
|
|
7
7
|
import { finalizeAccumulator, newAccumulator, updateAccumulator } from './accumulator.js'
|
|
8
|
+
import { foldEvaluatedRows } from './fold.js'
|
|
8
9
|
import { referencesRowScope } from './rowScope.js'
|
|
9
10
|
import { sortEntriesByTerms } from './sort.js'
|
|
10
11
|
import { keyify } from './utils.js'
|
|
@@ -342,9 +343,11 @@ function substituteValues(node, values) {
|
|
|
342
343
|
}
|
|
343
344
|
|
|
344
345
|
/**
|
|
345
|
-
* Folds one chunk of rows into the group accumulators.
|
|
346
|
-
* conditions, and aggregate arguments are
|
|
347
|
-
*
|
|
346
|
+
* Folds one chunk of rows into the group accumulators. Each row's group keys,
|
|
347
|
+
* FILTER conditions, and aggregate arguments are dispatched together so async
|
|
348
|
+
* cells overlap, and folded in row order as they resolve, so evaluated values
|
|
349
|
+
* (which can be large strings) are bounded by bytes instead of being held for
|
|
350
|
+
* the whole chunk.
|
|
348
351
|
*
|
|
349
352
|
* @param {object} options
|
|
350
353
|
* @param {AsyncRow[]} options.chunk
|
|
@@ -355,72 +358,88 @@ function substituteValues(node, values) {
|
|
|
355
358
|
* @param {ExecuteContext} options.context
|
|
356
359
|
* @returns {Promise<void>}
|
|
357
360
|
*/
|
|
358
|
-
|
|
359
|
-
/**
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
361
|
+
function accumulateChunk({ chunk, groupBy, specs, groups, needsRow, context }) {
|
|
362
|
+
/**
|
|
363
|
+
* @param {SqlPrimitive[]} keyValues
|
|
364
|
+
* @param {number} index
|
|
365
|
+
* @returns {StreamingGroup}
|
|
366
|
+
*/
|
|
367
|
+
function newGroup(keyValues, index) {
|
|
368
|
+
return {
|
|
369
|
+
firstRow: needsRow ? chunk[index] : undefined,
|
|
370
|
+
keyValues,
|
|
371
|
+
accumulators: specs.map(spec => newAccumulator(spec.funcName, spec.node.distinct)),
|
|
372
|
+
}
|
|
363
373
|
}
|
|
364
374
|
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
const
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
for (let j = 0; j < chunk.length; j++) {
|
|
382
|
-
if (passes[j]) {
|
|
383
|
-
passingRows.push(chunk[j])
|
|
384
|
-
passingIndices.push(j)
|
|
385
|
-
}
|
|
375
|
+
// Fast path: a single group key and only bare star aggregates (the common
|
|
376
|
+
// COUNT(*) GROUP BY x shape) skip the per-row tuple wrapper so synchronous
|
|
377
|
+
// cells stay cheap.
|
|
378
|
+
const singleExpr = groupBy.length === 1 && specs.every(spec => spec.star && !spec.node.filter)
|
|
379
|
+
? groupBy[0] : null
|
|
380
|
+
if (singleExpr) {
|
|
381
|
+
return foldEvaluatedRows({
|
|
382
|
+
rows: chunk,
|
|
383
|
+
signal: context.signal,
|
|
384
|
+
evaluate: row => evaluateExpr({ node: singleExpr, row, context }),
|
|
385
|
+
fold(value, index) {
|
|
386
|
+
const key = keyify(value)
|
|
387
|
+
let group = groups.get(key)
|
|
388
|
+
if (!group) {
|
|
389
|
+
group = newGroup([value], index)
|
|
390
|
+
groups.set(key, group)
|
|
386
391
|
}
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
spread[passingIndices[k]] = values[k]
|
|
392
|
+
for (let s = 0; s < specs.length; s++) {
|
|
393
|
+
if (specs[s].funcName === 'COUNT') group.accumulators[s].count++
|
|
394
|
+
else updateAccumulator(specs[s].funcName, group.accumulators[s], null)
|
|
391
395
|
}
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
} else {
|
|
395
|
-
args[s] = star ? undefined : await evaluateAll(node.args[0], chunk, context)
|
|
396
|
-
}
|
|
396
|
+
},
|
|
397
|
+
})
|
|
397
398
|
}
|
|
398
399
|
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
400
|
+
return foldEvaluatedRows({
|
|
401
|
+
rows: chunk,
|
|
402
|
+
signal: context.signal,
|
|
403
|
+
// One flat tuple per row: group key values, then per spec its FILTER
|
|
404
|
+
// result and its argument. The argument is only evaluated for rows that
|
|
405
|
+
// pass the FILTER, matching the buffered path.
|
|
406
|
+
evaluate(row) {
|
|
407
|
+
const pending = groupBy.map(expr => evaluateExpr({ node: expr, row, context }))
|
|
408
|
+
for (const { node, star } of specs) {
|
|
409
|
+
if (node.filter) {
|
|
410
|
+
const passes = evaluateExpr({ node: node.filter, row, context })
|
|
411
|
+
pending.push(passes)
|
|
412
|
+
if (!star) {
|
|
413
|
+
pending.push(passes.then(pass => pass ? evaluateExpr({ node: node.args[0], row, context }) : null))
|
|
414
|
+
}
|
|
415
|
+
} else if (!star) {
|
|
416
|
+
pending.push(evaluateExpr({ node: node.args[0], row, context }))
|
|
417
|
+
}
|
|
409
418
|
}
|
|
410
|
-
|
|
411
|
-
}
|
|
412
|
-
|
|
413
|
-
const
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
if (
|
|
417
|
-
group.
|
|
418
|
-
|
|
419
|
-
const arg = args[s]
|
|
420
|
-
updateAccumulator(spec.funcName, group.accumulators[s], arg ? arg[j] : null)
|
|
419
|
+
return Promise.all(pending)
|
|
420
|
+
},
|
|
421
|
+
fold(values, index) {
|
|
422
|
+
const key = groupBy.length === 0 ? true
|
|
423
|
+
: groupBy.length === 1 ? keyify(values[0]) : keyify(...values.slice(0, groupBy.length))
|
|
424
|
+
let group = groups.get(key)
|
|
425
|
+
if (!group) {
|
|
426
|
+
group = newGroup(values.slice(0, groupBy.length), index)
|
|
427
|
+
groups.set(key, group)
|
|
421
428
|
}
|
|
422
|
-
|
|
423
|
-
|
|
429
|
+
let slot = groupBy.length
|
|
430
|
+
for (let s = 0; s < specs.length; s++) {
|
|
431
|
+
const spec = specs[s]
|
|
432
|
+
const passes = spec.node.filter ? values[slot++] : true
|
|
433
|
+
const arg = spec.star ? null : values[slot++]
|
|
434
|
+
if (!passes) continue
|
|
435
|
+
if (spec.star && spec.funcName === 'COUNT') {
|
|
436
|
+
group.accumulators[s].count++
|
|
437
|
+
} else {
|
|
438
|
+
updateAccumulator(spec.funcName, group.accumulators[s], arg)
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
},
|
|
442
|
+
})
|
|
424
443
|
}
|
|
425
444
|
|
|
426
445
|
/**
|
package/src/execute/window.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { evaluateExpr } from '../expression/evaluate.js'
|
|
2
2
|
import { executePlan } from './execute.js'
|
|
3
|
+
import { foldEvaluatedRows } from './fold.js'
|
|
3
4
|
import { compareForTerm, keyify } from './utils.js'
|
|
4
5
|
import { yieldToEventLoop } from './yield.js'
|
|
5
6
|
|
|
@@ -116,30 +117,24 @@ export function executeWindow(plan, context) {
|
|
|
116
117
|
* @param {ExecuteContext} context
|
|
117
118
|
*/
|
|
118
119
|
async function computeWindow(spec, rows, output, context) {
|
|
119
|
-
// Bucket row indices by partition key.
|
|
120
|
+
// Bucket row indices by partition key. Keys are evaluated in adaptive
|
|
121
|
+
// chunks so async cells overlap while evaluated key values stay byte-bounded.
|
|
120
122
|
/** @type {Map<string | number | bigint | boolean, number[]>} */
|
|
121
123
|
const partitions = new Map()
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
const chunkKeys = await Promise.all(
|
|
129
|
-
rows.slice(chunkStart, chunkEnd).map(row =>
|
|
130
|
-
Promise.all(spec.partitionBy.map(expr => evaluateExpr({ node: expr, row, context })))
|
|
131
|
-
)
|
|
132
|
-
)
|
|
133
|
-
for (let j = 0; j < chunkKeys.length; j++) {
|
|
134
|
-
const key = keyify(...chunkKeys[j])
|
|
124
|
+
await foldEvaluatedRows({
|
|
125
|
+
rows,
|
|
126
|
+
signal: context.signal,
|
|
127
|
+
evaluate: row => Promise.all(spec.partitionBy.map(expr => evaluateExpr({ node: expr, row, context }))),
|
|
128
|
+
fold(keyValues, index) {
|
|
129
|
+
const key = keyify(...keyValues)
|
|
135
130
|
let bucket = partitions.get(key)
|
|
136
131
|
if (!bucket) {
|
|
137
132
|
bucket = []
|
|
138
133
|
partitions.set(key, bucket)
|
|
139
134
|
}
|
|
140
|
-
bucket.push(
|
|
141
|
-
}
|
|
142
|
-
}
|
|
135
|
+
bucket.push(index)
|
|
136
|
+
},
|
|
137
|
+
})
|
|
143
138
|
|
|
144
139
|
for (const bucket of partitions.values()) {
|
|
145
140
|
context.signal?.throwIfAborted()
|
package/src/expression/regexp.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { stringify } from '../execute/utils.js'
|
|
1
2
|
import { ArgValueError } from '../validation/executionErrors.js'
|
|
2
3
|
|
|
3
4
|
/**
|
|
@@ -187,5 +188,9 @@ export function evaluateRegexpLike({ node, args, rowIndex, cache }) {
|
|
|
187
188
|
cache.regex = regex
|
|
188
189
|
}
|
|
189
190
|
}
|
|
190
|
-
|
|
191
|
+
// Objects, arrays, and Dates test against their JSON text, the same
|
|
192
|
+
// coercion LIKE and CAST(x AS VARCHAR) use, so the three agree on
|
|
193
|
+
// JSON-typed columns. String() would collapse every object to
|
|
194
|
+
// '[object Object]', making REGEXP_LIKE a silent never-match on them.
|
|
195
|
+
return regex.test(typeof string === 'object' ? stringify(string) : String(string))
|
|
191
196
|
}
|