squirreling 0.16.2 → 0.16.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "squirreling",
3
- "version": "0.16.2",
3
+ "version": "0.16.3",
4
4
  "description": "Squirreling Async SQL Engine",
5
5
  "author": "Hyperparam",
6
6
  "homepage": "https://hyperparam.app",
@@ -39,7 +39,7 @@
39
39
  "test": "vitest run"
40
40
  },
41
41
  "devDependencies": {
42
- "@types/node": "26.2.0",
42
+ "@types/node": "26.4.0",
43
43
  "@vitest/coverage-v8": "4.1.11",
44
44
  "eslint": "9.39.4",
45
45
  "eslint-plugin-jsdoc": "64.2.1",
@@ -3,6 +3,7 @@ import { derivedAlias } from '../expression/alias.js'
3
3
  import { evaluateExpr } from '../expression/evaluate.js'
4
4
  import { finalizeAccumulator, newAccumulator, updateAccumulator } from './accumulator.js'
5
5
  import { executePlan, executeScan, selectColumnNames } from './execute.js'
6
+ import { foldEvaluatedRows } from './fold.js'
6
7
  import { normalizeScanColumnResult } from './scanColumn.js'
7
8
  import { sortEntriesByTerms } from './sort.js'
8
9
  import { planStreamingAggregates, streamingHashAggregateRows, streamingScalarAggregateRows } from './streamingAggregate.js'
@@ -109,47 +110,42 @@ export function executeHashAggregate(plan, context) {
109
110
  }
110
111
  context.signal?.throwIfAborted()
111
112
 
112
- // Group rows by GROUP BY keys.
113
- // Each chunk dispatches all per-row key evaluations in parallel so
114
- // async cells (e.g. lazy parquet decode) overlap; the await is at the
115
- // chunk boundary. Synchronous cells stay cheap because we skip the
116
- // inner Promise.all wrapper when there's a single GROUP BY expression.
113
+ // Group rows by GROUP BY keys. Keys are evaluated in adaptive chunks
114
+ // so async cells (e.g. lazy parquet decode) overlap while evaluated
115
+ // key values stay byte-bounded. The single-key branch skips the inner
116
+ // Promise.all wrapper so synchronous cells stay cheap.
117
117
  /** @type {Map<any, AsyncRow[]>} */
118
118
  const groups = new Map()
119
119
  const { groupBy } = plan
120
- const singleKey = groupBy.length === 1
121
- const singleExpr = singleKey ? groupBy[0] : null
120
+ const singleExpr = groupBy.length === 1 ? groupBy[0] : null
122
121
 
123
- for (let chunkStart = 0; chunkStart < allRows.length; chunkStart += YIELD_INTERVAL) {
124
- if (chunkStart > 0) {
125
- await yieldToEventLoop()
126
- context.signal?.throwIfAborted()
127
- }
128
- const chunkEnd = Math.min(chunkStart + YIELD_INTERVAL, allRows.length)
129
- const chunkLen = chunkEnd - chunkStart
130
- /** @type {Promise<any>[]} */
131
- const pending = new Array(chunkLen)
132
- if (singleKey) {
133
- for (let j = 0; j < chunkLen; j++) {
134
- pending[j] = evaluateExpr({ node: singleExpr, row: allRows[chunkStart + j], context })
135
- }
136
- } else {
137
- for (let j = 0; j < chunkLen; j++) {
138
- const row = allRows[chunkStart + j]
139
- pending[j] = Promise.all(groupBy.map(expr => evaluateExpr({ node: expr, row, context })))
140
- }
141
- }
142
- const chunkKeys = await Promise.all(pending)
143
- for (let j = 0; j < chunkLen; j++) {
144
- const key = singleKey ? keyify(chunkKeys[j]) : keyify(...chunkKeys[j])
145
- const row = allRows[chunkStart + j]
146
- let group = groups.get(key)
147
- if (!group) {
148
- group = []
149
- groups.set(key, group)
150
- }
151
- group.push(row)
122
+ /**
123
+ * @param {any} key
124
+ * @param {number} index
125
+ */
126
+ function addToGroup(key, index) {
127
+ let group = groups.get(key)
128
+ if (!group) {
129
+ group = []
130
+ groups.set(key, group)
152
131
  }
132
+ group.push(allRows[index])
133
+ }
134
+
135
+ if (singleExpr) {
136
+ await foldEvaluatedRows({
137
+ rows: allRows,
138
+ signal: context.signal,
139
+ evaluate: row => evaluateExpr({ node: singleExpr, row, context }),
140
+ fold: (value, index) => addToGroup(keyify(value), index),
141
+ })
142
+ } else {
143
+ await foldEvaluatedRows({
144
+ rows: allRows,
145
+ signal: context.signal,
146
+ evaluate: row => Promise.all(groupBy.map(expr => evaluateExpr({ node: expr, row, context }))),
147
+ fold: (values, index) => addToGroup(keyify(...values), index),
148
+ })
153
149
  }
154
150
 
155
151
  /** @type {{ row: AsyncRow, rows: AsyncRow[], outputRow: AsyncRow }[]} */
@@ -0,0 +1,74 @@
1
+ import { yieldToEventLoop } from './yield.js'
2
+
3
+ /**
4
+ * @import { AsyncRow } from '../types.js'
5
+ */
6
+
7
+ // Chunk sizing for foldEvaluatedRows. Dispatch starts small so one chunk of
8
+ // unexpectedly fat values cannot overshoot far, then adapts to the observed
9
+ // result sizes: small values grow the chunk toward MAX_CHUNK_ROWS so async
10
+ // cells still overlap, while large values (e.g. long string group keys)
11
+ // shrink it so in-flight results stay near CHUNK_BYTE_BUDGET instead of
12
+ // scaling with row count.
13
+ const INITIAL_CHUNK_ROWS = 64
14
+ const MAX_CHUNK_ROWS = 4000
15
+ const CHUNK_BYTE_BUDGET = 16 * 1024 * 1024
16
+
17
+ /**
18
+ * Approximate retained bytes of an evaluated value. Strings dominate the
19
+ * workloads where size matters; everything else counts as a small constant.
20
+ *
21
+ * @param {unknown} value
22
+ * @returns {number}
23
+ */
24
+ function valueBytes(value) {
25
+ if (typeof value === 'string') return value.length * 2
26
+ if (Array.isArray(value)) {
27
+ let bytes = 0
28
+ for (const item of value) bytes += valueBytes(item)
29
+ return bytes
30
+ }
31
+ return 16
32
+ }
33
+
34
+ /**
35
+ * Evaluates a value for every row and folds each result in row order, holding
36
+ * at most one adaptively sized chunk of evaluated values. Rows in a chunk are
37
+ * dispatched together so async cells overlap, and the chunk boundary yields
38
+ * to the event loop so aborts can fire. Chunks are bounded by bytes, not row
39
+ * count: a fixed 4000-row chunk of ~90KB string keys would hold hundreds of
40
+ * megabytes of results at once.
41
+ *
42
+ * @template T
43
+ * @param {Object} options
44
+ * @param {AsyncRow[]} options.rows
45
+ * @param {(row: AsyncRow, index: number) => Promise<T>} options.evaluate
46
+ * @param {(value: T, index: number) => void} options.fold
47
+ * @param {AbortSignal} [options.signal]
48
+ * @returns {Promise<void>}
49
+ */
50
+ export async function foldEvaluatedRows({ rows, evaluate, fold, signal }) {
51
+ let chunkSize = INITIAL_CHUNK_ROWS
52
+ /** @type {Promise<T>[]} */
53
+ const pending = []
54
+ for (let start = 0; start < rows.length;) {
55
+ if (start > 0) {
56
+ await yieldToEventLoop()
57
+ signal?.throwIfAborted()
58
+ }
59
+ const end = Math.min(start + chunkSize, rows.length)
60
+ pending.length = end - start
61
+ for (let i = start; i < end; i++) {
62
+ pending[i - start] = evaluate(rows[i], i)
63
+ }
64
+ const values = await Promise.all(pending)
65
+ let bytes = 0
66
+ for (let j = 0; j < values.length; j++) {
67
+ bytes += valueBytes(values[j])
68
+ fold(values[j], start + j)
69
+ }
70
+ start = end
71
+ const bytesPerRow = Math.max(1, bytes / values.length)
72
+ chunkSize = Math.min(MAX_CHUNK_ROWS, Math.max(1, Math.floor(CHUNK_BYTE_BUDGET / bytesPerRow)))
73
+ }
74
+ }
@@ -1,10 +1,11 @@
1
1
  import { selectedRowCount, valueAt } from '../backend/batch.js'
2
2
  import { derivedAlias } from '../expression/alias.js'
3
3
  import { compileBatchExpression } from '../expression/batch.js'
4
- import { evaluateAll, evaluateExpr } from '../expression/evaluate.js'
4
+ import { evaluateExpr } from '../expression/evaluate.js'
5
5
  import { collectColumnsFromExpr } from '../plan/columns.js'
6
6
  import { isAggregateFunc } from '../validation/functions.js'
7
7
  import { finalizeAccumulator, newAccumulator, updateAccumulator } from './accumulator.js'
8
+ import { foldEvaluatedRows } from './fold.js'
8
9
  import { referencesRowScope } from './rowScope.js'
9
10
  import { sortEntriesByTerms } from './sort.js'
10
11
  import { keyify } from './utils.js'
@@ -342,9 +343,11 @@ function substituteValues(node, values) {
342
343
  }
343
344
 
344
345
  /**
345
- * Folds one chunk of rows into the group accumulators. Group keys, FILTER
346
- * conditions, and aggregate arguments are each evaluated across the whole
347
- * chunk so async cells overlap; the chunk is released afterwards.
346
+ * Folds one chunk of rows into the group accumulators. Each row's group keys,
347
+ * FILTER conditions, and aggregate arguments are dispatched together so async
348
+ * cells overlap, and folded in row order as they resolve, so evaluated values
349
+ * (which can be large strings) are bounded by bytes instead of being held for
350
+ * the whole chunk.
348
351
  *
349
352
  * @param {object} options
350
353
  * @param {AsyncRow[]} options.chunk
@@ -355,72 +358,88 @@ function substituteValues(node, values) {
355
358
  * @param {ExecuteContext} options.context
356
359
  * @returns {Promise<void>}
357
360
  */
358
- async function accumulateChunk({ chunk, groupBy, specs, groups, needsRow, context }) {
359
- /** @type {SqlPrimitive[][] | undefined} */
360
- let keyColumns
361
- if (groupBy.length) {
362
- keyColumns = await Promise.all(groupBy.map(expr => evaluateAll(expr, chunk, context)))
361
+ function accumulateChunk({ chunk, groupBy, specs, groups, needsRow, context }) {
362
+ /**
363
+ * @param {SqlPrimitive[]} keyValues
364
+ * @param {number} index
365
+ * @returns {StreamingGroup}
366
+ */
367
+ function newGroup(keyValues, index) {
368
+ return {
369
+ firstRow: needsRow ? chunk[index] : undefined,
370
+ keyValues,
371
+ accumulators: specs.map(spec => newAccumulator(spec.funcName, spec.node.distinct)),
372
+ }
363
373
  }
364
374
 
365
- /** @type {(SqlPrimitive[] | undefined)[]} */
366
- const filters = new Array(specs.length)
367
- /** @type {(SqlPrimitive[] | undefined)[]} */
368
- const args = new Array(specs.length)
369
- for (let s = 0; s < specs.length; s++) {
370
- const { node, star } = specs[s]
371
- if (node.filter) {
372
- const passes = await evaluateAll(node.filter, chunk, context)
373
- filters[s] = passes
374
- if (!star) {
375
- // The buffered path filters the group before evaluating arguments,
376
- // so only evaluate the argument for rows that pass the FILTER
377
- /** @type {AsyncRow[]} */
378
- const passingRows = []
379
- /** @type {number[]} */
380
- const passingIndices = []
381
- for (let j = 0; j < chunk.length; j++) {
382
- if (passes[j]) {
383
- passingRows.push(chunk[j])
384
- passingIndices.push(j)
385
- }
375
+ // Fast path: a single group key and only bare star aggregates (the common
376
+ // COUNT(*) GROUP BY x shape) skip the per-row tuple wrapper so synchronous
377
+ // cells stay cheap.
378
+ const singleExpr = groupBy.length === 1 && specs.every(spec => spec.star && !spec.node.filter)
379
+ ? groupBy[0] : null
380
+ if (singleExpr) {
381
+ return foldEvaluatedRows({
382
+ rows: chunk,
383
+ signal: context.signal,
384
+ evaluate: row => evaluateExpr({ node: singleExpr, row, context }),
385
+ fold(value, index) {
386
+ const key = keyify(value)
387
+ let group = groups.get(key)
388
+ if (!group) {
389
+ group = newGroup([value], index)
390
+ groups.set(key, group)
386
391
  }
387
- const values = await evaluateAll(node.args[0], passingRows, context)
388
- const spread = new Array(chunk.length).fill(null)
389
- for (let k = 0; k < passingIndices.length; k++) {
390
- spread[passingIndices[k]] = values[k]
392
+ for (let s = 0; s < specs.length; s++) {
393
+ if (specs[s].funcName === 'COUNT') group.accumulators[s].count++
394
+ else updateAccumulator(specs[s].funcName, group.accumulators[s], null)
391
395
  }
392
- args[s] = spread
393
- }
394
- } else {
395
- args[s] = star ? undefined : await evaluateAll(node.args[0], chunk, context)
396
- }
396
+ },
397
+ })
397
398
  }
398
399
 
399
- for (let j = 0; j < chunk.length; j++) {
400
- const key = keyColumns
401
- ? keyColumns.length === 1 ? keyify(keyColumns[0][j]) : keyify(...keyColumns.map(c => c[j]))
402
- : true
403
- let group = groups.get(key)
404
- if (!group) {
405
- group = {
406
- firstRow: needsRow ? chunk[j] : undefined,
407
- keyValues: keyColumns ? keyColumns.map(c => c[j]) : [],
408
- accumulators: specs.map(spec => newAccumulator(spec.funcName, spec.node.distinct)),
400
+ return foldEvaluatedRows({
401
+ rows: chunk,
402
+ signal: context.signal,
403
+ // One flat tuple per row: group key values, then per spec its FILTER
404
+ // result and its argument. The argument is only evaluated for rows that
405
+ // pass the FILTER, matching the buffered path.
406
+ evaluate(row) {
407
+ const pending = groupBy.map(expr => evaluateExpr({ node: expr, row, context }))
408
+ for (const { node, star } of specs) {
409
+ if (node.filter) {
410
+ const passes = evaluateExpr({ node: node.filter, row, context })
411
+ pending.push(passes)
412
+ if (!star) {
413
+ pending.push(passes.then(pass => pass ? evaluateExpr({ node: node.args[0], row, context }) : null))
414
+ }
415
+ } else if (!star) {
416
+ pending.push(evaluateExpr({ node: node.args[0], row, context }))
417
+ }
409
418
  }
410
- groups.set(key, group)
411
- }
412
- for (let s = 0; s < specs.length; s++) {
413
- const filter = filters[s]
414
- if (filter && !filter[j]) continue
415
- const spec = specs[s]
416
- if (spec.star && spec.funcName === 'COUNT') {
417
- group.accumulators[s].count++
418
- } else {
419
- const arg = args[s]
420
- updateAccumulator(spec.funcName, group.accumulators[s], arg ? arg[j] : null)
419
+ return Promise.all(pending)
420
+ },
421
+ fold(values, index) {
422
+ const key = groupBy.length === 0 ? true
423
+ : groupBy.length === 1 ? keyify(values[0]) : keyify(...values.slice(0, groupBy.length))
424
+ let group = groups.get(key)
425
+ if (!group) {
426
+ group = newGroup(values.slice(0, groupBy.length), index)
427
+ groups.set(key, group)
421
428
  }
422
- }
423
- }
429
+ let slot = groupBy.length
430
+ for (let s = 0; s < specs.length; s++) {
431
+ const spec = specs[s]
432
+ const passes = spec.node.filter ? values[slot++] : true
433
+ const arg = spec.star ? null : values[slot++]
434
+ if (!passes) continue
435
+ if (spec.star && spec.funcName === 'COUNT') {
436
+ group.accumulators[s].count++
437
+ } else {
438
+ updateAccumulator(spec.funcName, group.accumulators[s], arg)
439
+ }
440
+ }
441
+ },
442
+ })
424
443
  }
425
444
 
426
445
  /**
@@ -1,5 +1,6 @@
1
1
  import { evaluateExpr } from '../expression/evaluate.js'
2
2
  import { executePlan } from './execute.js'
3
+ import { foldEvaluatedRows } from './fold.js'
3
4
  import { compareForTerm, keyify } from './utils.js'
4
5
  import { yieldToEventLoop } from './yield.js'
5
6
 
@@ -116,30 +117,24 @@ export function executeWindow(plan, context) {
116
117
  * @param {ExecuteContext} context
117
118
  */
118
119
  async function computeWindow(spec, rows, output, context) {
119
- // Bucket row indices by partition key.
120
+ // Bucket row indices by partition key. Keys are evaluated in adaptive
121
+ // chunks so async cells overlap while evaluated key values stay byte-bounded.
120
122
  /** @type {Map<string | number | bigint | boolean, number[]>} */
121
123
  const partitions = new Map()
122
- for (let chunkStart = 0; chunkStart < rows.length; chunkStart += YIELD_INTERVAL) {
123
- if (chunkStart > 0) {
124
- await yieldToEventLoop()
125
- context.signal?.throwIfAborted()
126
- }
127
- const chunkEnd = Math.min(chunkStart + YIELD_INTERVAL, rows.length)
128
- const chunkKeys = await Promise.all(
129
- rows.slice(chunkStart, chunkEnd).map(row =>
130
- Promise.all(spec.partitionBy.map(expr => evaluateExpr({ node: expr, row, context })))
131
- )
132
- )
133
- for (let j = 0; j < chunkKeys.length; j++) {
134
- const key = keyify(...chunkKeys[j])
124
+ await foldEvaluatedRows({
125
+ rows,
126
+ signal: context.signal,
127
+ evaluate: row => Promise.all(spec.partitionBy.map(expr => evaluateExpr({ node: expr, row, context }))),
128
+ fold(keyValues, index) {
129
+ const key = keyify(...keyValues)
135
130
  let bucket = partitions.get(key)
136
131
  if (!bucket) {
137
132
  bucket = []
138
133
  partitions.set(key, bucket)
139
134
  }
140
- bucket.push(chunkStart + j)
141
- }
142
- }
135
+ bucket.push(index)
136
+ },
137
+ })
143
138
 
144
139
  for (const bucket of partitions.values()) {
145
140
  context.signal?.throwIfAborted()
@@ -1,3 +1,4 @@
1
+ import { stringify } from '../execute/utils.js'
1
2
  import { ArgValueError } from '../validation/executionErrors.js'
2
3
 
3
4
  /**
@@ -187,5 +188,9 @@ export function evaluateRegexpLike({ node, args, rowIndex, cache }) {
187
188
  cache.regex = regex
188
189
  }
189
190
  }
190
- return regex.test(String(string))
191
+ // Objects, arrays, and Dates test against their JSON text, the same
192
+ // coercion LIKE and CAST(x AS VARCHAR) use, so the three agree on
193
+ // JSON-typed columns. String() would collapse every object to
194
+ // '[object Object]', making REGEXP_LIKE a silent never-match on them.
195
+ return regex.test(typeof string === 'object' ? stringify(string) : String(string))
191
196
  }