@lancedb/lancedb 0.40.0-beta.11 → 0.40.0-beta.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/NODEJS_THIRD_PARTY_LICENSES.md +8 -0
- package/dist/arrow.d.ts +18 -1
- package/dist/arrow.js +182 -13
- package/dist/blob.d.ts +95 -7
- package/dist/blob.js +167 -13
- package/dist/catalog.d.ts +40 -6
- package/dist/catalog.js +58 -9
- package/dist/connection.d.ts +18 -4
- package/dist/connection.js +8 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.js +2 -1
- package/dist/materialized_view.d.ts +3 -9
- package/dist/materialized_view.js +4 -9
- package/dist/native.d.ts +40 -13
- package/dist/native.js +53 -52
- package/dist/query.d.ts +7 -1
- package/dist/query.js +12 -15
- package/dist/sanitize.d.ts +19 -18
- package/dist/sanitize.js +78 -11
- package/dist/schema.js +31 -3
- package/dist/table.d.ts +21 -7
- package/dist/table.js +7 -7
- package/dist/util.js +9 -0
- package/package.json +13 -9
|
@@ -275,6 +275,7 @@
|
|
|
275
275
|
[@types/node@20.16.10](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
276
276
|
[@types/node@20.17.9](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
277
277
|
[@types/node@22.7.4](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
278
|
+
[@types/node@25.9.8](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
278
279
|
[@types/semver@7.5.6](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
279
280
|
[@types/stack-utils@2.0.3](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
280
281
|
[@types/tmp@0.2.6](https://github.com/DefinitelyTyped/DefinitelyTyped) - MIT
|
|
@@ -306,6 +307,9 @@
|
|
|
306
307
|
[apache-arrow@16.0.0](https://github.com/apache/arrow) - Apache-2.0
|
|
307
308
|
[apache-arrow@17.0.0](https://github.com/apache/arrow) - Apache-2.0
|
|
308
309
|
[apache-arrow@18.0.0](https://github.com/apache/arrow) - Apache-2.0
|
|
310
|
+
[apache-arrow@19.0.1](https://github.com/apache/arrow) - Apache-2.0
|
|
311
|
+
[apache-arrow@20.0.0](https://github.com/apache/arrow) - Apache-2.0
|
|
312
|
+
[apache-arrow@21.2.0](https://github.com/apache/arrow-js) - Apache-2.0
|
|
309
313
|
[argparse@1.0.10](https://github.com/nodeca/argparse) - MIT
|
|
310
314
|
[argparse@2.0.1](https://github.com/nodeca/argparse) - Python-2.0
|
|
311
315
|
[array-back@3.1.0](https://github.com/75lb/array-back) - MIT
|
|
@@ -348,6 +352,7 @@
|
|
|
348
352
|
[color@4.2.3](https://github.com/Qix-/color) - MIT
|
|
349
353
|
[combined-stream@1.0.8](https://github.com/felixge/node-combined-stream) - MIT
|
|
350
354
|
[command-line-args@5.2.1](https://github.com/75lb/command-line-args) - MIT
|
|
355
|
+
[command-line-args@6.0.2](https://github.com/75lb/command-line-args) - MIT
|
|
351
356
|
[command-line-usage@7.0.1](https://github.com/75lb/command-line-usage) - MIT
|
|
352
357
|
[concat-map@0.0.1](https://github.com/substack/node-concat-map) - MIT
|
|
353
358
|
[convert-source-map@2.0.0](https://github.com/thlorenz/convert-source-map) - MIT
|
|
@@ -398,12 +403,14 @@
|
|
|
398
403
|
[file-entry-cache@6.0.1](https://github.com/royriojas/file-entry-cache) - MIT
|
|
399
404
|
[fill-range@7.1.1](https://github.com/jonschlinkert/fill-range) - MIT
|
|
400
405
|
[find-replace@3.0.0](https://github.com/75lb/find-replace) - MIT
|
|
406
|
+
[find-replace@5.0.2](https://github.com/75lb/find-replace) - MIT
|
|
401
407
|
[find-up@4.1.0](https://github.com/sindresorhus/find-up) - MIT
|
|
402
408
|
[find-up@5.0.0](https://github.com/sindresorhus/find-up) - MIT
|
|
403
409
|
[flat-cache@3.2.0](https://github.com/jaredwray/flat-cache) - MIT
|
|
404
410
|
[flatbuffers@1.12.0](https://github.com/google/flatbuffers) - Apache*
|
|
405
411
|
[flatbuffers@23.5.26](https://github.com/google/flatbuffers) - Apache*
|
|
406
412
|
[flatbuffers@24.3.25](https://github.com/google/flatbuffers) - Apache-2.0
|
|
413
|
+
[flatbuffers@25.9.23](https://github.com/google/flatbuffers) - Apache-2.0
|
|
407
414
|
[flatted@3.2.9](https://github.com/WebReflection/flatted) - ISC
|
|
408
415
|
[follow-redirects@1.15.6](https://github.com/follow-redirects/follow-redirects) - MIT
|
|
409
416
|
[foreground-child@3.3.0](https://github.com/tapjs/foreground-child) - ISC
|
|
@@ -644,6 +651,7 @@
|
|
|
644
651
|
[uc.micro@2.1.0](https://github.com/markdown-it/uc.micro) - MIT
|
|
645
652
|
[undici-types@5.26.5](https://github.com/nodejs/undici) - MIT
|
|
646
653
|
[undici-types@6.19.8](https://github.com/nodejs/undici) - MIT
|
|
654
|
+
[undici-types@7.24.6](https://github.com/nodejs/undici) - MIT
|
|
647
655
|
[update-browserslist-db@1.0.13](https://github.com/browserslist/update-db) - MIT
|
|
648
656
|
[uri-js@4.4.1](https://github.com/garycourt/uri-js) - BSD-2-Clause
|
|
649
657
|
[uuid@9.0.1](https://github.com/uuidjs/uuid) - MIT
|
package/dist/arrow.d.ts
CHANGED
|
@@ -9,11 +9,22 @@ export type SchemaLike = Schema | {
|
|
|
9
9
|
get names(): unknown[];
|
|
10
10
|
};
|
|
11
11
|
export type FieldLike = Field | {
|
|
12
|
-
type: string;
|
|
12
|
+
type: string | DataTypeLike;
|
|
13
13
|
name: string;
|
|
14
14
|
nullable: boolean;
|
|
15
15
|
metadata?: Map<string, string>;
|
|
16
16
|
};
|
|
17
|
+
/**
|
|
18
|
+
* A `DataType` from any copy or version of apache-arrow.
|
|
19
|
+
*
|
|
20
|
+
* Arrow 21 brands its classes with `unique symbol` properties, so a type
|
|
21
|
+
* object from a second copy of the library no longer satisfies the `DataType`
|
|
22
|
+
* type of this one even though it is structurally identical. Inputs that only
|
|
23
|
+
* need to be sanitized accept this looser shape instead.
|
|
24
|
+
*/
|
|
25
|
+
export type DataTypeLike = DataType | {
|
|
26
|
+
readonly typeId: number;
|
|
27
|
+
};
|
|
17
28
|
/**
|
|
18
29
|
* Create an Arrow field backed by LanceDB's JSON extension type.
|
|
19
30
|
*
|
|
@@ -205,6 +216,12 @@ export declare class MakeArrowTableOptions {
|
|
|
205
216
|
* ```
|
|
206
217
|
*/
|
|
207
218
|
export declare function makeArrowTable(data: Array<Record<string, unknown>>, options?: Partial<MakeArrowTableOptions>, metadata?: Map<string, string>): ArrowTable;
|
|
219
|
+
/**
|
|
220
|
+
* Reads `Blob` / `File` values in the blob columns of `schema` into bytes so
|
|
221
|
+
* the synchronous conversion in {@link makeArrowTable} can accept them.
|
|
222
|
+
* Returns `data` itself when there is nothing to read.
|
|
223
|
+
*/
|
|
224
|
+
export declare function resolveBlobInputs(data: Array<Record<string, unknown>>, schema?: SchemaLike): Promise<Array<Record<string, unknown>>>;
|
|
208
225
|
/**
|
|
209
226
|
* Create an empty Arrow table with the provided schema
|
|
210
227
|
*/
|
package/dist/arrow.js
CHANGED
|
@@ -42,6 +42,7 @@ exports.isUnion = isUnion;
|
|
|
42
42
|
exports.isFixedSizeBinary = isFixedSizeBinary;
|
|
43
43
|
exports.isFixedSizeList = isFixedSizeList;
|
|
44
44
|
exports.makeArrowTable = makeArrowTable;
|
|
45
|
+
exports.resolveBlobInputs = resolveBlobInputs;
|
|
45
46
|
exports.makeEmptyTable = makeEmptyTable;
|
|
46
47
|
exports.convertToTable = convertToTable;
|
|
47
48
|
exports.newVectorType = newVectorType;
|
|
@@ -354,6 +355,22 @@ function makeArrowTable(data, options, metadata) {
|
|
|
354
355
|
schema = (0, sanitize_1.sanitizeSchema)(opt.schema);
|
|
355
356
|
schema = validateSchemaEmbeddings(schema, data, options?.embeddingFunction);
|
|
356
357
|
}
|
|
358
|
+
if (schema !== undefined) {
|
|
359
|
+
// Validate blob values up front and give every one the full
|
|
360
|
+
// `{ data, uri }` shape, so inference sees the same struct whether a row
|
|
361
|
+
// passed bytes, a URI, or a partial struct.
|
|
362
|
+
data = mapBlobInputs(data, schema, (value, field, row) => {
|
|
363
|
+
if (value === undefined) {
|
|
364
|
+
return value;
|
|
365
|
+
}
|
|
366
|
+
try {
|
|
367
|
+
return (0, blob_1.coerceBlobValue)(value);
|
|
368
|
+
}
|
|
369
|
+
catch (e) {
|
|
370
|
+
throw new Error(`Invalid value for blob field ${field} at row ${row}: ${e.message}`);
|
|
371
|
+
}
|
|
372
|
+
});
|
|
373
|
+
}
|
|
357
374
|
let schemaMetadata = schema?.metadata || new Map();
|
|
358
375
|
if (metadata !== undefined) {
|
|
359
376
|
schemaMetadata = new Map([...schemaMetadata, ...metadata]);
|
|
@@ -401,6 +418,87 @@ function containsBlobField(field) {
|
|
|
401
418
|
}
|
|
402
419
|
return (field.type.children ?? []).some((child) => containsBlobField(child));
|
|
403
420
|
}
|
|
421
|
+
/**
|
|
422
|
+
* Calls `visit` on every value that lands in a blob column of `schema`,
|
|
423
|
+
* including blobs nested in structs and lists, and replaces it with the
|
|
424
|
+
* result. Records and arrays are copied only where a value changed, so the
|
|
425
|
+
* caller's data is never mutated.
|
|
426
|
+
*/
|
|
427
|
+
function mapBlobInputs(data, schema, visit) {
|
|
428
|
+
if (!schema.fields.some(containsBlobField)) {
|
|
429
|
+
return data;
|
|
430
|
+
}
|
|
431
|
+
return data.map((record, row) => isObject(record)
|
|
432
|
+
? mapBlobFields(record, schema.fields, "", row, visit)
|
|
433
|
+
: record);
|
|
434
|
+
}
|
|
435
|
+
function mapBlobFields(record, fields, prefix, row, visit) {
|
|
436
|
+
let out;
|
|
437
|
+
for (const field of fields) {
|
|
438
|
+
if (!containsBlobField(field) || !Object.hasOwn(record, field.name)) {
|
|
439
|
+
continue;
|
|
440
|
+
}
|
|
441
|
+
const value = record[field.name];
|
|
442
|
+
const mapped = mapBlobField(value, field, `${prefix}${field.name}`, row, visit);
|
|
443
|
+
if (mapped !== value) {
|
|
444
|
+
out ??= { ...record };
|
|
445
|
+
out[field.name] = mapped;
|
|
446
|
+
}
|
|
447
|
+
}
|
|
448
|
+
return out ?? record;
|
|
449
|
+
}
|
|
450
|
+
function mapBlobField(value, field, label, row, visit) {
|
|
451
|
+
if ((0, blob_1.isBlobField)(field)) {
|
|
452
|
+
return visit(value, label, row);
|
|
453
|
+
}
|
|
454
|
+
if (field.type instanceof apache_arrow_1.Struct && isObject(value)) {
|
|
455
|
+
return mapBlobFields(value, field.type.children, `${label}.`, row, visit);
|
|
456
|
+
}
|
|
457
|
+
if (isList(field.type) && Array.isArray(value)) {
|
|
458
|
+
const child = field.type.children[0];
|
|
459
|
+
let out;
|
|
460
|
+
for (const [index, element] of value.entries()) {
|
|
461
|
+
const mapped = mapBlobField(element, child, `${label}[${index}]`, row, visit);
|
|
462
|
+
if (mapped !== element) {
|
|
463
|
+
out ??= [...value];
|
|
464
|
+
out[index] = mapped;
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
return out ?? value;
|
|
468
|
+
}
|
|
469
|
+
return value;
|
|
470
|
+
}
|
|
471
|
+
/**
|
|
472
|
+
* Reads `Blob` / `File` values in the blob columns of `schema` into bytes so
|
|
473
|
+
* the synchronous conversion in {@link makeArrowTable} can accept them.
|
|
474
|
+
* Returns `data` itself when there is nothing to read.
|
|
475
|
+
*/
|
|
476
|
+
async function resolveBlobInputs(data, schema) {
|
|
477
|
+
if (schema === undefined || schema === null) {
|
|
478
|
+
return data;
|
|
479
|
+
}
|
|
480
|
+
const sanitized = (0, sanitize_1.sanitizeSchema)(schema);
|
|
481
|
+
// Keyed by the Blob itself, so rows sharing one Blob (bare or as
|
|
482
|
+
// `{ data }`) read it once and share the bytes.
|
|
483
|
+
const reads = new Map();
|
|
484
|
+
mapBlobInputs(data, sanitized, (value) => {
|
|
485
|
+
const source = (0, blob_1.blobToRead)(value);
|
|
486
|
+
if (source !== undefined && !reads.has(source)) {
|
|
487
|
+
reads.set(source, source.arrayBuffer().then((buffer) => new Uint8Array(buffer)));
|
|
488
|
+
}
|
|
489
|
+
return value;
|
|
490
|
+
});
|
|
491
|
+
if (reads.size === 0) {
|
|
492
|
+
return data;
|
|
493
|
+
}
|
|
494
|
+
const bytes = new Map(await Promise.all([...reads].map(async ([source, read]) => [source, await read])));
|
|
495
|
+
return mapBlobInputs(data, sanitized, (value) => {
|
|
496
|
+
const source = (0, blob_1.blobToRead)(value);
|
|
497
|
+
return source === undefined
|
|
498
|
+
? value
|
|
499
|
+
: (0, blob_1.withBlobBytes)(value, bytes.get(source));
|
|
500
|
+
});
|
|
501
|
+
}
|
|
404
502
|
function isObject(value) {
|
|
405
503
|
return (typeof value === "object" &&
|
|
406
504
|
value !== null &&
|
|
@@ -795,6 +893,7 @@ async function convertToTable(data, embeddings, makeTableOptions) {
|
|
|
795
893
|
makeTableOptions.schema.metadata?.has("embedding_functions")) {
|
|
796
894
|
processedData = ensureNestedFieldsExist(data, makeTableOptions.schema);
|
|
797
895
|
}
|
|
896
|
+
processedData = await resolveBlobInputs(processedData, makeTableOptions?.schema);
|
|
798
897
|
const table = makeArrowTable(processedData, makeTableOptions);
|
|
799
898
|
return await applyEmbeddings(table, embeddings, makeTableOptions?.schema);
|
|
800
899
|
}
|
|
@@ -835,6 +934,69 @@ async function fromRecordsToStreamBuffer(data, embeddings, schema) {
|
|
|
835
934
|
const writer = apache_arrow_1.RecordBatchStreamWriter.writeAll(table);
|
|
836
935
|
return Buffer.from(await writer.toUint8Array());
|
|
837
936
|
}
|
|
937
|
+
// `Type.Utf8View` / `Type.BinaryView` as numbers: the enum members only exist
|
|
938
|
+
// in Arrow 21+, and this module compiles against every supported release.
|
|
939
|
+
const UTF8_VIEW_TYPE_ID = 24;
|
|
940
|
+
const BINARY_VIEW_TYPE_ID = 23;
|
|
941
|
+
/**
|
|
942
|
+
* Copy a Utf8View / BinaryView `Data` into a single `Data` of `type`.
|
|
943
|
+
*/
|
|
944
|
+
function materializeViewData(child, type) {
|
|
945
|
+
const builder = (0, apache_arrow_1.makeBuilder)({ type, nullValues: [null] });
|
|
946
|
+
for (const value of new apache_arrow_1.Vector([child])) {
|
|
947
|
+
builder.append(value);
|
|
948
|
+
}
|
|
949
|
+
return builder.finish().flush();
|
|
950
|
+
}
|
|
951
|
+
/**
|
|
952
|
+
* Rebuild any top-level Utf8View / BinaryView column as Utf8 / Binary.
|
|
953
|
+
*
|
|
954
|
+
* Lance stores the view types as their offset-based equivalents anyway, so
|
|
955
|
+
* nothing is lost. Doing it here also sidesteps an Arrow JS 21 bug: its IPC
|
|
956
|
+
* writer emits a truncated views buffer for a *sliced* view array, which the
|
|
957
|
+
* Rust reader rejects with "Need at least N bytes in buffers[0]".
|
|
958
|
+
*
|
|
959
|
+
* The record batches are rebuilt positionally rather than through a
|
|
960
|
+
* `Record<string, Vector>`: JavaScript enumerates integer-like keys first, so
|
|
961
|
+
* a field named e.g. `"1"` would otherwise be paired with the wrong column.
|
|
962
|
+
*
|
|
963
|
+
* Tables without view columns are returned as-is.
|
|
964
|
+
*/
|
|
965
|
+
function materializeViewColumns(table) {
|
|
966
|
+
const replacements = new Map();
|
|
967
|
+
table.schema.fields.forEach((field, i) => {
|
|
968
|
+
if (field.type.typeId === UTF8_VIEW_TYPE_ID) {
|
|
969
|
+
replacements.set(i, new apache_arrow_1.Utf8());
|
|
970
|
+
}
|
|
971
|
+
else if (field.type.typeId === BINARY_VIEW_TYPE_ID) {
|
|
972
|
+
replacements.set(i, new apache_arrow_1.Binary());
|
|
973
|
+
}
|
|
974
|
+
});
|
|
975
|
+
if (replacements.size === 0) {
|
|
976
|
+
return table;
|
|
977
|
+
}
|
|
978
|
+
const fields = table.schema.fields.map((field, i) => {
|
|
979
|
+
const type = replacements.get(i);
|
|
980
|
+
return type === undefined
|
|
981
|
+
? field
|
|
982
|
+
: new apache_arrow_1.Field(field.name, type, field.nullable, field.metadata);
|
|
983
|
+
});
|
|
984
|
+
const schema = new apache_arrow_1.Schema(fields, table.schema.metadata);
|
|
985
|
+
const batches = table.batches.map((batch) => {
|
|
986
|
+
const children = batch.data.children.map((child, i) => {
|
|
987
|
+
const type = replacements.get(i);
|
|
988
|
+
return type === undefined ? child : materializeViewData(child, type);
|
|
989
|
+
});
|
|
990
|
+
const data = (0, apache_arrow_1.makeData)({
|
|
991
|
+
type: new apache_arrow_1.Struct(fields),
|
|
992
|
+
length: batch.numRows,
|
|
993
|
+
nullCount: 0,
|
|
994
|
+
children,
|
|
995
|
+
});
|
|
996
|
+
return new apache_arrow_1.RecordBatch(schema, data);
|
|
997
|
+
});
|
|
998
|
+
return new apache_arrow_1.Table(schema, batches);
|
|
999
|
+
}
|
|
838
1000
|
/**
|
|
839
1001
|
* Serialize an Arrow Table into a buffer using the Arrow IPC File serialization
|
|
840
1002
|
*
|
|
@@ -847,7 +1009,7 @@ async function fromTableToBuffer(table, embeddings, schema) {
|
|
|
847
1009
|
if (schema !== undefined && schema !== null) {
|
|
848
1010
|
schema = (0, sanitize_1.sanitizeSchema)(schema);
|
|
849
1011
|
}
|
|
850
|
-
const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema);
|
|
1012
|
+
const tableWithEmbeddings = materializeViewColumns(await applyEmbeddings(table, embeddings, schema));
|
|
851
1013
|
validateBlobSchema(tableWithEmbeddings.schema);
|
|
852
1014
|
const writer = apache_arrow_1.RecordBatchFileWriter.writeAll(tableWithEmbeddings);
|
|
853
1015
|
return Buffer.from(await writer.toUint8Array());
|
|
@@ -906,7 +1068,8 @@ async function fromRecordBatchToBuffer(batch) {
|
|
|
906
1068
|
* batch + EOS) suitable for incremental decode by `arrow_ipc::reader::StreamReader`.
|
|
907
1069
|
*/
|
|
908
1070
|
async function fromRecordBatchToStreamBuffer(batch) {
|
|
909
|
-
const
|
|
1071
|
+
const table = materializeViewColumns(new apache_arrow_1.Table([batch]));
|
|
1072
|
+
const writer = apache_arrow_1.RecordBatchStreamWriter.writeAll(table);
|
|
910
1073
|
return Buffer.from(await writer.toUint8Array());
|
|
911
1074
|
}
|
|
912
1075
|
/**
|
|
@@ -918,7 +1081,7 @@ async function fromRecordBatchToStreamBuffer(batch) {
|
|
|
918
1081
|
* `schema` is required if the table is empty
|
|
919
1082
|
*/
|
|
920
1083
|
async function fromTableToStreamBuffer(table, embeddings, schema) {
|
|
921
|
-
const tableWithEmbeddings = await applyEmbeddings(table, embeddings, schema);
|
|
1084
|
+
const tableWithEmbeddings = materializeViewColumns(await applyEmbeddings(table, embeddings, schema));
|
|
922
1085
|
const writer = apache_arrow_1.RecordBatchStreamWriter.writeAll(tableWithEmbeddings);
|
|
923
1086
|
return Buffer.from(await writer.toUint8Array());
|
|
924
1087
|
}
|
|
@@ -1019,7 +1182,7 @@ function ensureNestedFieldsExist(data, schema) {
|
|
|
1019
1182
|
const completeRow = {};
|
|
1020
1183
|
for (const field of schema.fields) {
|
|
1021
1184
|
if (field.name in row) {
|
|
1022
|
-
if (field
|
|
1185
|
+
if (isPlainStructField(field) &&
|
|
1023
1186
|
row[field.name] !== null &&
|
|
1024
1187
|
row[field.name] !== undefined) {
|
|
1025
1188
|
// Handle nested struct
|
|
@@ -1034,15 +1197,22 @@ function ensureNestedFieldsExist(data, schema) {
|
|
|
1034
1197
|
else {
|
|
1035
1198
|
// Keep a missing struct valid while filling each of its children with
|
|
1036
1199
|
// null. This is distinct from an explicitly null struct value.
|
|
1037
|
-
completeRow[field.name] =
|
|
1038
|
-
field.type
|
|
1039
|
-
|
|
1040
|
-
: null;
|
|
1200
|
+
completeRow[field.name] = isPlainStructField(field)
|
|
1201
|
+
? ensureStructFieldsExist({}, field.type)
|
|
1202
|
+
: null;
|
|
1041
1203
|
}
|
|
1042
1204
|
}
|
|
1043
1205
|
return completeRow;
|
|
1044
1206
|
});
|
|
1045
1207
|
}
|
|
1208
|
+
/**
|
|
1209
|
+
* Blob fields are Arrow structs, but their values are bytes, URIs, or
|
|
1210
|
+
* `{ data } | { uri }` inputs that the blob coercion in `makeArrowTable`
|
|
1211
|
+
* handles, so they must not be filled in like ordinary structs.
|
|
1212
|
+
*/
|
|
1213
|
+
function isPlainStructField(field) {
|
|
1214
|
+
return field.type.constructor.name === "Struct" && !(0, blob_1.isBlobField)(field);
|
|
1215
|
+
}
|
|
1046
1216
|
/**
|
|
1047
1217
|
* Recursively ensures that all fields in a struct type exist in the data,
|
|
1048
1218
|
* filling missing fields with null values.
|
|
@@ -1051,7 +1221,7 @@ function ensureStructFieldsExist(data, structType) {
|
|
|
1051
1221
|
const completeStruct = {};
|
|
1052
1222
|
for (const childField of structType.children) {
|
|
1053
1223
|
if (childField.name in data) {
|
|
1054
|
-
if (childField
|
|
1224
|
+
if (isPlainStructField(childField) &&
|
|
1055
1225
|
data[childField.name] !== null &&
|
|
1056
1226
|
data[childField.name] !== undefined) {
|
|
1057
1227
|
// Recursively handle nested struct
|
|
@@ -1065,10 +1235,9 @@ function ensureStructFieldsExist(data, structType) {
|
|
|
1065
1235
|
else {
|
|
1066
1236
|
// Keep a missing struct valid while filling each of its children with
|
|
1067
1237
|
// null. This is distinct from an explicitly null struct value.
|
|
1068
|
-
completeStruct[childField.name] =
|
|
1069
|
-
childField.type
|
|
1070
|
-
|
|
1071
|
-
: null;
|
|
1238
|
+
completeStruct[childField.name] = isPlainStructField(childField)
|
|
1239
|
+
? ensureStructFieldsExist({}, childField.type)
|
|
1240
|
+
: null;
|
|
1072
1241
|
}
|
|
1073
1242
|
}
|
|
1074
1243
|
return completeStruct;
|
package/dist/blob.d.ts
CHANGED
|
@@ -1,6 +1,42 @@
|
|
|
1
1
|
import { Field } from "apache-arrow";
|
|
2
2
|
import { BlobFile as NativeBlobFile } from "./native";
|
|
3
|
-
|
|
3
|
+
/**
|
|
4
|
+
* Bytes accepted for a blob value. `ArrayBuffer` is wrapped without copying.
|
|
5
|
+
* `Blob` and `File` are read with `arrayBuffer()`, which only the async write
|
|
6
|
+
* paths ({@link Connection.createTable}, {@link Table.add},
|
|
7
|
+
* {@link Table.mergeInsert}) can do.
|
|
8
|
+
*/
|
|
9
|
+
export type BlobData = Buffer | Uint8Array | ArrayBuffer | Blob;
|
|
10
|
+
/** A URI accepted for a blob value. A `URL` is stored as its `href`. */
|
|
11
|
+
export type BlobUri = string | URL;
|
|
12
|
+
/**
|
|
13
|
+
* A value accepted for a `lance.blob.v2` column (see {@link blob}).
|
|
14
|
+
*
|
|
15
|
+
* Either inline bytes, a URI pointing at external bytes, or a struct that sets
|
|
16
|
+
* exactly one of `data` or `uri`. `null` and `undefined` write a null blob.
|
|
17
|
+
*
|
|
18
|
+
* @example
|
|
19
|
+
* ```ts
|
|
20
|
+
* import { pathToFileURL } from "node:url";
|
|
21
|
+
* import type { BlobInput } from "@lancedb/lancedb";
|
|
22
|
+
*
|
|
23
|
+
* const rows: { id: bigint; image: BlobInput }[] = [
|
|
24
|
+
* { id: 1n, image: await (await fetch(url)).arrayBuffer() },
|
|
25
|
+
* { id: 2n, image: pathToFileURL("/data/cat.png") },
|
|
26
|
+
* { id: 3n, image: { data: new Blob(["hello"]) } },
|
|
27
|
+
* ];
|
|
28
|
+
* await table.add(rows);
|
|
29
|
+
* ```
|
|
30
|
+
*/
|
|
31
|
+
export type BlobInput = BlobData | BlobUri | {
|
|
32
|
+
data: BlobData;
|
|
33
|
+
uri?: null;
|
|
34
|
+
} | {
|
|
35
|
+
data?: null;
|
|
36
|
+
uri: BlobUri;
|
|
37
|
+
} | null | undefined;
|
|
38
|
+
/** A blob value normalized to the columns of the blob struct. */
|
|
39
|
+
export type BlobValue = {
|
|
4
40
|
data: Buffer | Uint8Array | null;
|
|
5
41
|
uri: string | null;
|
|
6
42
|
};
|
|
@@ -76,17 +112,69 @@ export declare class BlobFile {
|
|
|
76
112
|
/** Returns the blob size in bytes. */
|
|
77
113
|
size(): bigint;
|
|
78
114
|
/**
|
|
79
|
-
* Reads from the cursor
|
|
115
|
+
* Reads from the cursor and advances the cursor.
|
|
80
116
|
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
117
|
+
* Reads to the end when `maxBytes` is omitted, or at most `maxBytes` bytes
|
|
118
|
+
* otherwise. Returns an empty buffer at the end of the blob.
|
|
119
|
+
* {@link BlobFile.readRange} does not move the cursor.
|
|
83
120
|
*/
|
|
84
|
-
read(): Promise<Buffer>;
|
|
121
|
+
read(maxBytes?: bigint, options?: BlobReadOptions): Promise<Buffer>;
|
|
85
122
|
/**
|
|
86
123
|
* Reads the half-open byte range `[start, end)`.
|
|
87
124
|
*
|
|
88
125
|
* Fails when `end` is past the blob size. Does not move the cursor.
|
|
89
126
|
*/
|
|
90
|
-
readRange(start: bigint, end: bigint): Promise<Buffer>;
|
|
127
|
+
readRange(start: bigint, end: bigint, options?: BlobReadOptions): Promise<Buffer>;
|
|
128
|
+
/**
|
|
129
|
+
* Reads several half-open byte ranges. Returns one buffer per range, in
|
|
130
|
+
* the order given.
|
|
131
|
+
*
|
|
132
|
+
* Fails when any `end` is past the blob size. Does not move the cursor.
|
|
133
|
+
*/
|
|
134
|
+
readRanges(ranges: BlobRange[], options?: BlobReadOptions): Promise<Buffer[]>;
|
|
135
|
+
/** Moves the cursor to `position`, in bytes from the start of the blob. */
|
|
136
|
+
seek(position: bigint): Promise<void>;
|
|
137
|
+
/** Returns the cursor position in bytes. */
|
|
138
|
+
tell(): Promise<bigint>;
|
|
139
|
+
/**
|
|
140
|
+
* Releases the handle. Reads after `close()` fail. Calling it again does
|
|
141
|
+
* nothing.
|
|
142
|
+
*/
|
|
143
|
+
close(): Promise<void>;
|
|
144
|
+
/** Returns true after {@link BlobFile.close}. */
|
|
145
|
+
isClosed(): boolean;
|
|
91
146
|
}
|
|
92
|
-
|
|
147
|
+
/** Options for blob reads. */
|
|
148
|
+
export type BlobReadOptions = {
|
|
149
|
+
/**
|
|
150
|
+
* Cancels the read. The call rejects with `signal.reason`, and the native
|
|
151
|
+
* read stops, including in-flight requests to a remote table.
|
|
152
|
+
*/
|
|
153
|
+
signal?: AbortSignal;
|
|
154
|
+
};
|
|
155
|
+
/** @ignore */
|
|
156
|
+
export declare function runWithSignal<T>(signal: AbortSignal | undefined, run: (signal: AbortSignal | undefined) => Promise<T>): Promise<T>;
|
|
157
|
+
/** A half-open byte range `[start, end)` for {@link BlobFile.readRanges}. */
|
|
158
|
+
export type BlobRange = {
|
|
159
|
+
start: bigint;
|
|
160
|
+
end: bigint;
|
|
161
|
+
};
|
|
162
|
+
/**
|
|
163
|
+
* Rewrites the synchronous-only widenings of {@link BlobInput} (`ArrayBuffer`,
|
|
164
|
+
* `URL`) into `Uint8Array` and URI strings, keeping the value's shape. Other
|
|
165
|
+
* values are returned unchanged and validated later by {@link coerceBlobValue}.
|
|
166
|
+
*
|
|
167
|
+
* Throws on `Blob` / `File`, which must be read into bytes first.
|
|
168
|
+
*/
|
|
169
|
+
export declare function normalizeBlobInput(value: unknown): unknown;
|
|
170
|
+
/**
|
|
171
|
+
* Returns the `Blob` / `File` a blob value needs read, either the value itself
|
|
172
|
+
* or its `data` field, or `undefined` when there is nothing to read.
|
|
173
|
+
*/
|
|
174
|
+
export declare function blobToRead(value: unknown): Blob | undefined;
|
|
175
|
+
/**
|
|
176
|
+
* Replaces the `Blob` that {@link blobToRead} found in `value` with `bytes`,
|
|
177
|
+
* keeping the value's shape.
|
|
178
|
+
*/
|
|
179
|
+
export declare function withBlobBytes(value: unknown, bytes: Uint8Array): unknown;
|
|
180
|
+
export declare function coerceBlobValue(input: unknown): BlobValue | null;
|