quiverdb 0.9.7 → 0.9.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Binary file
Binary file
Binary file
Binary file
Binary file
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "quiverdb",
3
- "version": "0.9.7",
3
+ "version": "0.9.8",
4
4
  "license": "MIT",
5
5
  "repository": {
6
6
  "type": "git",
@@ -141,14 +141,17 @@ export function allocNativeString(str: string): Allocation {
141
141
  return { ptr: ptr(buf), buf };
142
142
  }
143
143
 
144
- /** Build a native string pointer table (const char* const*) from an array of strings. */
145
- export function allocNativeStringArray(strings: string[]): {
144
+ /**
145
+ * Build a native string pointer table (const char* const*) from an array of strings.
146
+ * A `null` entry becomes a NULL pointer in the table (used for SQL-NULL string cells).
147
+ */
148
+ export function allocNativeStringArray(strings: (string | null)[]): {
146
149
  table: Allocation;
147
150
  keepalive: Allocation[];
148
151
  } {
149
- const strAllocs = strings.map((s) => allocNativeString(s));
150
- const table = allocNativePtrTable(strAllocs.map((a) => a.ptr));
151
- return { table, keepalive: strAllocs };
152
+ const strAllocs = strings.map((s) => (s === null ? null : allocNativeString(s)));
153
+ const table = allocNativePtrTable(strAllocs.map((a) => (a ? a.ptr : null)));
154
+ return { table, keepalive: strAllocs.filter((a): a is Allocation => a !== null) };
152
155
  }
153
156
 
154
157
  /** Get the native address of a pointer as BigInt. */
package/src/loader.ts CHANGED
@@ -118,14 +118,17 @@ const describeSymbols = {
118
118
  } as const;
119
119
 
120
120
  const timeSeriesSymbols = {
121
- quiver_database_read_time_series_group: { args: [P, BUF, BUF, I64, P, P, P, P, P], returns: I32 },
121
+ quiver_database_read_time_series_group: {
122
+ args: [P, BUF, BUF, I64, P, P, P, P, P, P],
123
+ returns: I32,
124
+ },
122
125
  quiver_database_read_time_series_row: { args: [P, BUF, BUF, BUF, BUF, P, P, P], returns: I32 },
123
126
  quiver_database_add_time_series_row: { args: [P, BUF, BUF, I64, P, P, P, USIZE], returns: I32 },
124
127
  quiver_database_update_time_series_group: {
125
- args: [P, BUF, BUF, I64, P, P, P, USIZE, USIZE],
128
+ args: [P, BUF, BUF, I64, P, P, P, P, USIZE, USIZE],
126
129
  returns: I32,
127
130
  },
128
- quiver_database_free_time_series_data: { args: [P, P, P, USIZE, USIZE], returns: I32 },
131
+ quiver_database_free_time_series_data: { args: [P, P, P, P, USIZE, USIZE], returns: I32 },
129
132
  quiver_database_has_time_series_files: { args: [P, BUF, P], returns: I32 },
130
133
  quiver_database_list_time_series_files_columns: { args: [P, BUF, P, P], returns: I32 },
131
134
  quiver_database_read_time_series_files: { args: [P, BUF, P, P, P], returns: I32 },
package/src/lua-api.ts CHANGED
@@ -5,9 +5,9 @@
5
5
  // binding changes, re-diff every signature below against `bind_database()` (names, arg order, arg
6
6
  // types, return shapes).
7
7
  //
8
- // NOTE: the binary/expression subsystems (quiver.* globals + file:/expr: methods) are bound in the
9
- // native binding and documented below. They are quiver.*/file:/expr: calls, not db:<name> tokens,
10
- // and they read/write files on the host filesystem.
8
+ // NOTE: the binary/expression subsystems are bound in the native binding and documented below.
9
+ // File-touching operations (db:open_file, db:bin_to_csv, db:csv_to_bin, expr:save) are sandboxed
10
+ // to the database file's directory; the pure-metadata builders stay under the quiver.* global.
11
11
  //
12
12
  // FORMAT CONVENTION: every db: method should appear at least once as the literal token
13
13
  // `db:<snake_case_name>` so coverage is greppable.
@@ -66,7 +66,14 @@ datetime surface — there are no DateTime wrapper helpers, unlike Julia/Dart/Py
66
66
  \`Failed to run Lua script: <message>\`. Validation failures roll back whatever the current
67
67
  transaction covered.
68
68
  - **Standard library.** Only the \`base\`, \`string\`, and \`table\` libraries are loaded — there is NO
69
- \`math\` (use \`//\` for integer division), and no \`os\`, \`io\`, \`require\`, \`load\`, or \`dofile\`.
69
+ \`math\` (use \`//\` for integer division), and no \`os\`, \`io\`, or \`require\`; \`dofile\` and
70
+ \`loadfile\` are removed (string-form \`load\` is available).
71
+ - **Filesystem sandbox.** Every file-touching operation (\`db:export_csv\`, \`db:import_csv\`,
72
+ \`db:open_file\`, \`db:bin_to_csv\`, \`db:csv_to_bin\`, \`expr:save\`) resolves relative paths against
73
+ the directory containing the database file and rejects anything outside it (subdirectories are
74
+ fine; \`..\` escapes and outside absolute paths throw \`Cannot <op>: path '...' escapes the
75
+ database directory ...\`). On an in-memory database these operations throw
76
+ \`Cannot <op>: database is in-memory, file operations are unavailable\`.
70
77
  - **Output.** Scripts cannot return values to the tool — use \`print()\` for anything you need to
71
78
  see (it is captured). Arrays are 1-indexed (iterate with \`ipairs\`); reading a NULL yields \`nil\`,
72
79
  writing \`nil\` stores NULL where NULL is accepted (query params, ts rows, file columns — but NOT
@@ -218,6 +225,11 @@ local ts = db:read_time_series_group(collection, group, id)
218
225
  -- returns an empty table {} if the element has no rows
219
226
  \`\`\`
220
227
 
228
+ A \`NULL\` value cell comes back as a \`nil\` hole (the index is simply absent), and an all-\`NULL\`
229
+ value column is an empty table \`{}\` (its key is still present). Because \`nil\` cannot occupy an
230
+ array slot, **take the row count from the dimension column** (\`#ts.date_time\`), never from a value
231
+ column.
232
+
221
233
  ### Read one value per element at a date (\`read_time_series_row\`)
222
234
 
223
235
  \`\`\`lua
@@ -258,14 +270,18 @@ column names). Each value of the top-level table must be an **array**, not a sca
258
270
 
259
271
  **Rules** (validation throws, rolling the script back):
260
272
  - Every column value must be an array — a bare scalar throws \`column '...' must be an array of values\`.
261
- - All columns must have the **same length** — a mismatch throws \`column '...' has length N but expected M\`.
262
- - Named-but-empty columns throw (\`contain no rows; pass an empty table {} to clear\`) rather than
263
- silently clearing — only a bare \`{}\` clears.
264
- - Write REAL-column values as float literals (\`30.0\`, not \`30\`); STRICT validation rejects an
265
- integer for a REAL column. Other Lua types throw \`column '...' has unsupported Lua type\`.
266
- - \`nil\` is **not** accepted inside a column here (it throws \`unsupported Lua type\`); columns must
267
- be fully populated. To write a NULL into a single row, use \`add_time_series_row\` instead, which
268
- does accept \`nil\`.
273
+ - The **dimension column(s) set the row count** and must be present and fully populated: the
274
+ \`date_*\` ordering column, plus any extra primary-key columns in a multi-dimensional group (e.g.
275
+ \`block\`). A missing one throws \`missing dimension column '...'\`; a \`nil\` inside one throws
276
+ \`dimension column '...' has nil at index N\` (they are primary-key columns and cannot be NULL).
277
+ - **Value columns may be shorter, sparser, or absent** relative to the dimension column — every
278
+ missing cell is written as \`NULL\`. So \`value = { 10.0, nil, 30.0 }\` or a too-short \`value = { 10.0 }\`
279
+ both write NULLs for the gaps; this is how you round-trip the \`nil\` holes a read produces. A value
280
+ column **longer** than the dimension column throws \`column '...' has length N but expected M\`.
281
+ - Named columns whose dimension transposes to zero rows throw (\`contain no rows; pass an empty
282
+ table {} to clear the group\`) — only a bare \`{}\` clears.
283
+ - Integer values are accepted for REAL columns (converted on insert). Booleans, functions, and
284
+ other unsupported Lua types throw \`column '...' has unsupported Lua type\`.
269
285
 
270
286
  ### Append/upsert a single row (\`add_time_series_row\` — ROW-oriented, the one exception)
271
287
 
@@ -373,8 +389,9 @@ local count = db:query_integer("SELECT COUNT(*) FROM Collection")
373
389
 
374
390
  ## CSV import / export
375
391
 
376
- Export a time-series group to a CSV file, or import one from a CSV file. \`path\` is resolved by the
377
- host filesystem; \`options\` is optional.
392
+ Export a time-series group to a CSV file, or import one from a CSV file. \`path\` is sandboxed:
393
+ relative paths resolve against the database file's directory and must stay inside it (see
394
+ Critical rules); \`options\` is optional.
378
395
 
379
396
  \`\`\`lua
380
397
  db:export_csv(collection, group, path, options)
@@ -433,10 +450,12 @@ db:delete_element("Collection", item2)
433
450
 
434
451
  ## Binary & expression subsystems
435
452
 
436
- Dense N-dimensional \`float64\` arrays (\`.qvr\` + \`.toml\` sidecar) plus lazy arithmetic over them,
437
- under a global \`quiver\` table (not \`db\`). Mirrors the Julia surface; aggregation ops are strings
438
- (Lua has no enums); operators are \`+ - * /\` and unary \`-\`, with scalars allowed on either side.
439
- These read/write files on the host filesystem (\`path\` is host-resolved).
453
+ Dense N-dimensional \`float64\` arrays (\`.qvr\` + \`.toml\` sidecar) plus lazy arithmetic over them.
454
+ File I/O is db-scoped (\`db:open_file\` / \`db:bin_to_csv\` / \`db:csv_to_bin\`); paths are extensionless
455
+ base paths, sandboxed to the database directory (see Critical rules), and \`get_file_path()\` returns
456
+ the resolved absolute path. The pure-metadata builders and expression constructors live under the
457
+ global \`quiver\` table. Mirrors the Julia surface; aggregation ops are strings (Lua has no enums);
458
+ operators are \`+ - * /\` and unary \`-\`, with scalars allowed on either side.
440
459
 
441
460
  \`\`\`lua
442
461
  local md = quiver.metadata{
@@ -444,17 +463,17 @@ local md = quiver.metadata{
444
463
  labels = {"v1", "v2"}, dimensions = {"stage", "block"}, dimension_sizes = {4, 31},
445
464
  time_dimensions = {"stage", "block"}, frequencies = {"monthly", "daily"},
446
465
  }
447
- local f = quiver.open_file(path, "w", md) -- mode "r"/"w"; md required for "w"
466
+ local f = db:open_file(path, "w", md) -- mode "r"/"w"; md required for "w"
448
467
  f:write({1.0, 2.0}, {stage = 1, block = 1}) -- data table, dims table
449
468
  f:close()
450
- local r = quiver.open_file(path, "r")
469
+ local r = db:open_file(path, "r")
451
470
  local cell = r:read({stage = 1, block = 1}) -- { v1, v2 }; pass true as 2nd arg to allow NaN
452
471
  r:get_metadata(); r:get_file_path(); r:is_open()
453
472
  md:get_unit(); md:get_version(); md:get_initial_datetime()
454
473
  md:get_labels(); md:get_dimensions(); md:get_number_of_time_dimensions(); md:to_toml()
455
474
  quiver.metadata_from_toml(text); quiver.metadata_from_element(tbl)
456
- quiver.bin_to_csv(path) -- aggregate=true by default; pass false to keep time dims as columns
457
- quiver.csv_to_bin(path)
475
+ db:bin_to_csv(path) -- aggregate=true by default; pass false to keep time dims as columns
476
+ db:csv_to_bin(path)
458
477
 
459
478
  local e = (quiver.expression(r) + 10.0) * 2.0 -- files auto-wrap; scalars either side
460
479
  e = quiver.abs(e); e = quiver.sqrt(e) -- also quiver.log / quiver.exp
@@ -463,7 +482,7 @@ e = e:aggregate("stage", "sum") -- sum/mean/min/max/percent
463
482
  e = e:aggregate("stage", "percentile", 0.9) -- percentile needs the fraction
464
483
  e = e:aggregate_agents("mean") -- collapse the label axis
465
484
  e = e:select_agents({"v2"}); e = e:rename_agents({v1 = "alpha"})
466
- e:save(out_path); e:metadata()
485
+ e:save(out_path); e:metadata() -- save path is sandboxed like db:open_file
467
486
  \`\`\`
468
487
 
469
488
  **\`quiver.metadata{...}\` kwargs and defaults:** \`version\` defaults to \`"1"\`; \`initial_datetime\`
@@ -26,7 +26,7 @@ import {
26
26
  DATA_TYPE_STRING,
27
27
  } from "./types.ts";
28
28
 
29
- export type TimeSeriesData = Record<string, (number | string)[]>;
29
+ export type TimeSeriesData = Record<string, (number | string | null)[]>;
30
30
 
31
31
  Database.prototype.readTimeSeriesGroup = function (
32
32
  this: Database,
@@ -40,6 +40,7 @@ Database.prototype.readTimeSeriesGroup = function (
40
40
  const outNames = allocPtrOut();
41
41
  const outTypes = allocPtrOut();
42
42
  const outData = allocPtrOut();
43
+ const outHasValue = allocPtrOut();
43
44
  const outColCount = allocUint64Out();
44
45
  const outRowCount = allocUint64Out();
45
46
 
@@ -52,6 +53,7 @@ Database.prototype.readTimeSeriesGroup = function (
52
53
  outNames.buf,
53
54
  outTypes.buf,
54
55
  outData.buf,
56
+ outHasValue.buf,
55
57
  outColCount.buf,
56
58
  outRowCount.buf,
57
59
  ),
@@ -64,26 +66,49 @@ Database.prototype.readTimeSeriesGroup = function (
64
66
  const namesPtr = readPtrOut(outNames);
65
67
  const typesPtr = readPtrOut(outTypes);
66
68
  const dataPtr = readPtrOut(outData);
69
+ const hasValuePtr = readPtrOut(outHasValue);
67
70
 
68
71
  const colNames = decodeStringArray(namesPtr, colCount);
69
72
  const typesAb = toArrayBuffer(typesPtr as Pointer, 0, colCount * 4);
70
73
  const types = Array.from(new Int32Array(typesAb));
71
74
  const dataPtrs = decodePtrArray(dataPtr, colCount);
75
+ const maskPtrs = decodePtrArray(hasValuePtr, colCount);
72
76
 
77
+ // Per-cell NULL mask: mask[r] === 0 means SQL NULL, surfaced as JS null. The
78
+ // dimension column's mask is always all 1, so it stays dense.
73
79
  const result: TimeSeriesData = {};
74
80
  for (let c = 0; c < colCount; c++) {
75
81
  const colName = colNames[c];
82
+ const maskPtr = maskPtrs[c];
83
+ const mask = maskPtr ? new Uint8Array(toArrayBuffer(maskPtr as Pointer, 0, rowCount)) : null;
76
84
  switch (types[c]) {
77
- case DATA_TYPE_INTEGER:
78
- result[colName] = decodeInt64Array(dataPtrs[c], rowCount);
85
+ case DATA_TYPE_INTEGER: {
86
+ const vals = decodeInt64Array(dataPtrs[c], rowCount);
87
+ result[colName] = mask ? vals.map((v, r) => (mask[r] ? v : null)) : vals;
79
88
  break;
80
- case DATA_TYPE_FLOAT:
81
- result[colName] = decodeFloat64Array(dataPtrs[c], rowCount);
89
+ }
90
+ case DATA_TYPE_FLOAT: {
91
+ const vals = decodeFloat64Array(dataPtrs[c], rowCount);
92
+ result[colName] = mask ? vals.map((v, r) => (mask[r] ? v : null)) : vals;
82
93
  break;
94
+ }
83
95
  case DATA_TYPE_STRING:
84
- case DATA_TYPE_DATE_TIME:
85
- result[colName] = decodeStringArray(dataPtrs[c], rowCount);
96
+ case DATA_TYPE_DATE_TIME: {
97
+ // Read pointer-by-pointer (not decodeStringArray): a masked-out cell is a
98
+ // NULL char* that CString cannot construct from.
99
+ const base = dataPtrs[c];
100
+ const col: (string | null)[] = new Array(rowCount);
101
+ for (let r = 0; r < rowCount; r++) {
102
+ if (mask && !mask[r]) {
103
+ col[r] = null;
104
+ continue;
105
+ }
106
+ const strPtr = base ? read.ptr(base as Pointer, r * 8) : 0;
107
+ col[r] = strPtr === 0 ? null : new CString(strPtr as Pointer).toString();
108
+ }
109
+ result[colName] = col;
86
110
  break;
111
+ }
87
112
  }
88
113
  }
89
114
 
@@ -91,6 +116,7 @@ Database.prototype.readTimeSeriesGroup = function (
91
116
  namesPtr,
92
117
  typesPtr,
93
118
  dataPtr,
119
+ hasValuePtr,
94
120
  BigInt(colCount),
95
121
  BigInt(rowCount),
96
122
  );
@@ -177,6 +203,7 @@ Database.prototype.updateTimeSeriesGroup = function (
177
203
  null,
178
204
  null,
179
205
  null,
206
+ null,
180
207
  0n,
181
208
  0n,
182
209
  ),
@@ -213,32 +240,58 @@ Database.prototype.updateTimeSeriesGroup = function (
213
240
  const { table: namesTable, keepalive: namesPtrs } = allocNativeStringArray(colNames);
214
241
  keepalive.push(namesTable, ...namesPtrs);
215
242
 
216
- // Build column types and data
243
+ // Build column types, data, and per-cell NULL masks. A null cell becomes
244
+ // mask 0 + a placeholder in the data array (the C API never reads it). An
245
+ // all-null column is tagged FLOAT with zeroed data — the type tag is ignored
246
+ // for masked-out cells.
217
247
  const typesBuf = new Uint8Array(columnCount * 4);
218
248
  const typesDv = new DataView(typesBuf.buffer);
219
249
  const dataPtrs: (Pointer | null)[] = [];
250
+ const maskPtrs: (Pointer | null)[] = [];
220
251
 
221
252
  for (let c = 0; c < columnCount; c++) {
222
- const values = entries[c][1];
223
-
224
- if (typeof values[0] === "string") {
253
+ const [colName, values] = entries[c];
254
+ const first = values.find((v) => v !== null);
255
+
256
+ // Mask via direct indexing — never a DataView, to avoid the documented
257
+ // .buffer-materialization pitfall between ptr() and the FFI call.
258
+ const maskBuf = new Uint8Array(rowCount);
259
+ for (let r = 0; r < rowCount; r++) maskBuf[r] = values[r] === null ? 0 : 1;
260
+ const maskAlloc: Allocation = { ptr: ptr(maskBuf), buf: maskBuf };
261
+ keepalive.push(maskAlloc);
262
+ maskPtrs.push(maskAlloc.ptr);
263
+
264
+ if (first === undefined) {
265
+ // All-null column
266
+ typesDv.setInt32(c * 4, DATA_TYPE_FLOAT, true);
267
+ const p = allocNativeFloat64(new Array(rowCount).fill(0));
268
+ keepalive.push(p);
269
+ dataPtrs.push(p.ptr);
270
+ } else if (typeof first === "string") {
225
271
  typesDv.setInt32(c * 4, DATA_TYPE_STRING, true);
226
- const { table, keepalive: strPtrs } = allocNativeStringArray(values as string[]);
272
+ const { table, keepalive: strPtrs } = allocNativeStringArray(
273
+ values.map((v) => (v === null ? null : (v as string))),
274
+ );
227
275
  keepalive.push(table, ...strPtrs);
228
276
  dataPtrs.push(table.ptr);
229
- } else if (typeof values[0] === "number") {
230
- const allIntegers = (values as number[]).every((v) => Number.isInteger(v));
231
- if (allIntegers) {
277
+ } else if (typeof first === "number") {
278
+ const nonNull = values.filter((v) => v !== null) as number[];
279
+ const sanitized = values.map((v) => (v === null ? 0 : (v as number)));
280
+ if (nonNull.every((v) => Number.isInteger(v))) {
232
281
  typesDv.setInt32(c * 4, DATA_TYPE_INTEGER, true);
233
- const p = allocNativeInt64(values as number[]);
282
+ const p = allocNativeInt64(sanitized);
234
283
  keepalive.push(p);
235
284
  dataPtrs.push(p.ptr);
236
285
  } else {
237
286
  typesDv.setInt32(c * 4, DATA_TYPE_FLOAT, true);
238
- const p = allocNativeFloat64(values as number[]);
287
+ const p = allocNativeFloat64(sanitized);
239
288
  keepalive.push(p);
240
289
  dataPtrs.push(p.ptr);
241
290
  }
291
+ } else {
292
+ throw new QuiverError(
293
+ `Cannot updateTimeSeriesGroup: column '${colName}' has unsupported value type ${typeof first}`,
294
+ );
242
295
  }
243
296
  }
244
297
 
@@ -246,6 +299,8 @@ Database.prototype.updateTimeSeriesGroup = function (
246
299
  keepalive.push(typesAlloc);
247
300
  const dataTable = allocNativePtrTable(dataPtrs);
248
301
  keepalive.push(dataTable);
302
+ const maskTable = allocNativePtrTable(maskPtrs);
303
+ keepalive.push(maskTable);
249
304
 
250
305
  check(
251
306
  lib.quiver_database_update_time_series_group(
@@ -256,6 +311,7 @@ Database.prototype.updateTimeSeriesGroup = function (
256
311
  namesTable.buf,
257
312
  typesAlloc.buf,
258
313
  dataTable.buf,
314
+ maskTable.buf,
259
315
  BigInt(columnCount),
260
316
  BigInt(rowCount),
261
317
  ),