qvdjs 0.10.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +286 -56
- package/dist/index.cjs +690 -210
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +690 -210
- package/dist/index.js.map +1 -1
- package/img/logo/qvdjs_logo-512.png +0 -0
- package/package.json +5 -2
package/dist/index.js
CHANGED
|
@@ -250,6 +250,128 @@ var init_QvdSymbol = __esm({
|
|
|
250
250
|
};
|
|
251
251
|
}
|
|
252
252
|
});
|
|
253
|
+
|
|
254
|
+
// src/util/readOptions.js
|
|
255
|
+
function requireRowCount(value, name, filePath) {
|
|
256
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
|
|
257
|
+
throw new QvdValidationError(`${name} must be a non-negative integer`, {
|
|
258
|
+
option: name,
|
|
259
|
+
provided: value,
|
|
260
|
+
type: typeof value,
|
|
261
|
+
file: filePath
|
|
262
|
+
});
|
|
263
|
+
}
|
|
264
|
+
return value;
|
|
265
|
+
}
|
|
266
|
+
function normaliseWindow(window, filePath) {
|
|
267
|
+
if (window === null || window === void 0) {
|
|
268
|
+
return { offset: 0, limit: null };
|
|
269
|
+
}
|
|
270
|
+
if (typeof window === "number") {
|
|
271
|
+
return { offset: 0, limit: requireRowCount(window, "maxRows", filePath) };
|
|
272
|
+
}
|
|
273
|
+
if (typeof window !== "object" || Array.isArray(window)) {
|
|
274
|
+
throw new QvdValidationError("The row window must be a number, null, or an {offset, limit} object", {
|
|
275
|
+
provided: window,
|
|
276
|
+
type: typeof window,
|
|
277
|
+
file: filePath
|
|
278
|
+
});
|
|
279
|
+
}
|
|
280
|
+
const { offset, limit, maxRows } = window;
|
|
281
|
+
const limitGiven = limit !== void 0 && limit !== null;
|
|
282
|
+
const maxRowsGiven = maxRows !== void 0 && maxRows !== null;
|
|
283
|
+
if (limitGiven && maxRowsGiven) {
|
|
284
|
+
throw new QvdValidationError("maxRows and limit are two names for the same option; pass one of them, not both", {
|
|
285
|
+
maxRows,
|
|
286
|
+
limit,
|
|
287
|
+
file: filePath
|
|
288
|
+
});
|
|
289
|
+
}
|
|
290
|
+
return {
|
|
291
|
+
offset: offset === void 0 || offset === null ? 0 : requireRowCount(offset, "offset", filePath),
|
|
292
|
+
limit: limitGiven ? requireRowCount(limit, "limit", filePath) : maxRowsGiven ? requireRowCount(maxRows, "maxRows", filePath) : null
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
function resolveWindow(window, totalRows) {
|
|
296
|
+
const rows = Number.isSafeInteger(totalRows) && totalRows > 0 ? totalRows : 0;
|
|
297
|
+
const offset = Math.min(window.offset, rows);
|
|
298
|
+
return {
|
|
299
|
+
offset,
|
|
300
|
+
limit: Math.max(0, Math.min(window.limit === null ? Infinity : window.limit, rows - offset))
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
function selectFields(fields, requested, filePath) {
|
|
304
|
+
if (requested === null || requested === void 0) {
|
|
305
|
+
return fields;
|
|
306
|
+
}
|
|
307
|
+
if (!Array.isArray(requested)) {
|
|
308
|
+
throw new QvdValidationError("fields must be an array of field names", {
|
|
309
|
+
provided: requested,
|
|
310
|
+
type: typeof requested,
|
|
311
|
+
file: filePath
|
|
312
|
+
});
|
|
313
|
+
}
|
|
314
|
+
const available = fields.map((field) => field["FieldName"]);
|
|
315
|
+
if (requested.length === 0) {
|
|
316
|
+
throw new QvdValidationError("fields must name at least one field", {
|
|
317
|
+
availableColumns: available,
|
|
318
|
+
file: filePath
|
|
319
|
+
});
|
|
320
|
+
}
|
|
321
|
+
const seen = /* @__PURE__ */ new Set();
|
|
322
|
+
return requested.map((name) => {
|
|
323
|
+
if (typeof name !== "string") {
|
|
324
|
+
throw new QvdValidationError("Field names must be strings", {
|
|
325
|
+
provided: name,
|
|
326
|
+
type: typeof name,
|
|
327
|
+
availableColumns: available,
|
|
328
|
+
file: filePath
|
|
329
|
+
});
|
|
330
|
+
}
|
|
331
|
+
if (seen.has(name)) {
|
|
332
|
+
throw new QvdValidationError(`Field '${name}' is listed twice`, {
|
|
333
|
+
column: name,
|
|
334
|
+
fields: requested,
|
|
335
|
+
file: filePath
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
seen.add(name);
|
|
339
|
+
const index = available.indexOf(name);
|
|
340
|
+
if (index === -1) {
|
|
341
|
+
throw new QvdValidationError(`Column '${name}' does not exist`, {
|
|
342
|
+
column: name,
|
|
343
|
+
availableColumns: available,
|
|
344
|
+
file: filePath
|
|
345
|
+
});
|
|
346
|
+
}
|
|
347
|
+
return fields[index];
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
function readerOptionsFrom(options) {
|
|
351
|
+
return {
|
|
352
|
+
allowedDir: options.allowedDir,
|
|
353
|
+
memorySafetyFactor: options.memorySafetyFactor,
|
|
354
|
+
symbolFilteringThreshold: options.symbolFilteringThreshold,
|
|
355
|
+
fields: options.fields === void 0 ? null : options.fields,
|
|
356
|
+
onProgress: options.onProgress,
|
|
357
|
+
signal: options.signal
|
|
358
|
+
};
|
|
359
|
+
}
|
|
360
|
+
function metadataOptionsFrom(options) {
|
|
361
|
+
return {
|
|
362
|
+
allowedDir: options.allowedDir,
|
|
363
|
+
onProgress: options.onProgress,
|
|
364
|
+
signal: options.signal
|
|
365
|
+
};
|
|
366
|
+
}
|
|
367
|
+
function windowFrom(options) {
|
|
368
|
+
return { offset: options.offset, limit: options.limit, maxRows: options.maxRows };
|
|
369
|
+
}
|
|
370
|
+
var init_readOptions = __esm({
|
|
371
|
+
"src/util/readOptions.js"() {
|
|
372
|
+
init_QvdErrors();
|
|
373
|
+
}
|
|
374
|
+
});
|
|
253
375
|
function isWithinDirectoryLexically(resolvedBaseDir, resolvedPath) {
|
|
254
376
|
const isCaseInsensitiveFS = process.platform === "win32";
|
|
255
377
|
const base = isCaseInsensitiveFS ? resolvedBaseDir.toLowerCase() : resolvedBaseDir;
|
|
@@ -807,11 +929,12 @@ function estimateRowMemory(rows, columnCount) {
|
|
|
807
929
|
}
|
|
808
930
|
return BASE_BYTES + rows * (ROW_BASE_BYTES + PER_CELL_BYTES * columnCount);
|
|
809
931
|
}
|
|
810
|
-
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
|
|
932
|
+
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null) {
|
|
811
933
|
const FULL_PARSE_OVERHEAD = 6;
|
|
812
934
|
const MINIMAL_OVERHEAD = 0.01;
|
|
813
935
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
814
|
-
const
|
|
936
|
+
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
937
|
+
const rowMemory = materialisesRows ? estimateRowMemory(liveRows, columnCount) : BASE_BYTES;
|
|
815
938
|
if (maxRows === null || maxRows >= totalRows) {
|
|
816
939
|
return symbolTableSize * FULL_PARSE_OVERHEAD + rowMemory;
|
|
817
940
|
}
|
|
@@ -821,18 +944,44 @@ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount =
|
|
|
821
944
|
const skippedSymbolsMemory = symbolTableSize * (1 - symbolPercentage) * MINIMAL_OVERHEAD;
|
|
822
945
|
return keptSymbolsMemory + skippedSymbolsMemory + rowMemory;
|
|
823
946
|
}
|
|
824
|
-
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true) {
|
|
825
|
-
|
|
947
|
+
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false) {
|
|
948
|
+
const costOf = (rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0);
|
|
949
|
+
if (costOf(totalRows) <= budget) {
|
|
826
950
|
return totalRows;
|
|
827
951
|
}
|
|
828
|
-
if (
|
|
952
|
+
if (costOf(0) > budget) {
|
|
829
953
|
return 0;
|
|
830
954
|
}
|
|
831
955
|
let low = 0;
|
|
832
956
|
let high = totalRows;
|
|
833
957
|
while (high - low > 1) {
|
|
834
958
|
const mid = Math.floor((low + high) / 2);
|
|
835
|
-
if (
|
|
959
|
+
if (costOf(mid) <= budget) {
|
|
960
|
+
low = mid;
|
|
961
|
+
} else {
|
|
962
|
+
high = mid;
|
|
963
|
+
}
|
|
964
|
+
}
|
|
965
|
+
return low;
|
|
966
|
+
}
|
|
967
|
+
function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false) {
|
|
968
|
+
const covered = windowRows === null || windowRows >= totalRows ? totalRows : windowRows;
|
|
969
|
+
const fits = (chunk) => {
|
|
970
|
+
const live = Math.min(chunk * liveRowsPerChunk, covered);
|
|
971
|
+
const cost = estimateMemoryUsage(symbolTableSize, windowRows, totalRows, columnCount, true, chunk * liveRowsPerChunk) + (includeExternal ? estimateExternalMemory(live, columnCount) : 0);
|
|
972
|
+
return cost <= budget;
|
|
973
|
+
};
|
|
974
|
+
if (fits(covered)) {
|
|
975
|
+
return covered;
|
|
976
|
+
}
|
|
977
|
+
if (!fits(1)) {
|
|
978
|
+
return 0;
|
|
979
|
+
}
|
|
980
|
+
let low = 1;
|
|
981
|
+
let high = covered;
|
|
982
|
+
while (high - low > 1) {
|
|
983
|
+
const mid = Math.floor((low + high) / 2);
|
|
984
|
+
if (fits(mid)) {
|
|
836
985
|
low = mid;
|
|
837
986
|
} else {
|
|
838
987
|
high = mid;
|
|
@@ -840,7 +989,7 @@ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, mat
|
|
|
840
989
|
}
|
|
841
990
|
return low;
|
|
842
991
|
}
|
|
843
|
-
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true) {
|
|
992
|
+
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null) {
|
|
844
993
|
if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
|
|
845
994
|
throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", { safetyFactor });
|
|
846
995
|
}
|
|
@@ -849,12 +998,16 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
849
998
|
}
|
|
850
999
|
const budget = getMemoryBudget();
|
|
851
1000
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
852
|
-
const
|
|
853
|
-
const
|
|
1001
|
+
const rowsLive = live === null ? null : live.rows;
|
|
1002
|
+
const liveRowsPerChunk = live === null ? 1 : live.perChunk;
|
|
1003
|
+
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
1004
|
+
const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows, rowsLive);
|
|
1005
|
+
const externalMemory = estimateExternalMemory(liveRows, columnCount);
|
|
854
1006
|
const bounded = budget.candidates.map((candidate) => {
|
|
855
1007
|
const heapOnly = candidate.source === "V8 heap limit";
|
|
856
1008
|
return {
|
|
857
1009
|
...candidate,
|
|
1010
|
+
heapOnly,
|
|
858
1011
|
needs: heapOnly ? heapMemory : heapMemory + externalMemory,
|
|
859
1012
|
allowed: candidate.bytes * safetyFactor,
|
|
860
1013
|
bounds: heapOnly ? "the V8 heap" : "the whole process"
|
|
@@ -870,12 +1023,14 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
870
1023
|
const estimatedMemory = binding ? binding.needs : heapMemory;
|
|
871
1024
|
const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
|
|
872
1025
|
if (binding) {
|
|
1026
|
+
const includeExternal = !binding.heapOnly;
|
|
873
1027
|
const recommendedMaxRows = recommendedRowsFor(
|
|
874
1028
|
maxAllowedMemory,
|
|
875
1029
|
symbolTableSize,
|
|
876
1030
|
totalRows,
|
|
877
1031
|
columnCount,
|
|
878
|
-
materialisesRows
|
|
1032
|
+
materialisesRows,
|
|
1033
|
+
includeExternal
|
|
879
1034
|
);
|
|
880
1035
|
const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
|
|
881
1036
|
const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
|
|
@@ -887,15 +1042,27 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
887
1042
|
const limitingScope = binding.bounds;
|
|
888
1043
|
const budgetBreakdown = budget.candidates.map((candidate) => `${candidate.source} ${Math.round(candidate.bytes / 1024 / 1024)}MB`).join(", ");
|
|
889
1044
|
const observedBreakdown = budget.observed.map((entry) => `${entry.source} ${Math.round(entry.bytes / 1024 / 1024)}MB`).join(", ");
|
|
890
|
-
const nothingFits = recommendedMaxRows === 0;
|
|
891
1045
|
const containerBound = binding.source === "container memory limit";
|
|
1046
|
+
const chunked = rowsLive !== null;
|
|
1047
|
+
const recommendedChunk = chunked ? recommendedChunkFor(
|
|
1048
|
+
maxAllowedMemory,
|
|
1049
|
+
symbolTableSize,
|
|
1050
|
+
maxRows,
|
|
1051
|
+
totalRows,
|
|
1052
|
+
columnCount,
|
|
1053
|
+
liveRowsPerChunk,
|
|
1054
|
+
includeExternal
|
|
1055
|
+
) : 0;
|
|
1056
|
+
const knob = chunked ? "chunkSize" : "limit";
|
|
1057
|
+
const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
|
|
1058
|
+
const nothingFits = recommendedValue === 0;
|
|
892
1059
|
let advice;
|
|
893
1060
|
if (nothingFits) {
|
|
894
|
-
advice = `No row count fits this budget - the symbol table alone exceeds it, so
|
|
1061
|
+
advice = `No row count fits this budget - the symbol table alone exceeds it, so ${knob} cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
|
|
895
1062
|
} else if (containerBound) {
|
|
896
|
-
advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or
|
|
1063
|
+
advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${recommendedValue.toLocaleString()} rows or less).`;
|
|
897
1064
|
} else {
|
|
898
|
-
advice = `Try
|
|
1065
|
+
advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${recommendedValue.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
|
|
899
1066
|
}
|
|
900
1067
|
throw new QvdValidationError(
|
|
901
1068
|
`Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
|
|
@@ -915,22 +1082,30 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
915
1082
|
columnCount,
|
|
916
1083
|
totalRows,
|
|
917
1084
|
maxRows,
|
|
918
|
-
recommendedMaxRows
|
|
1085
|
+
recommendedMaxRows,
|
|
1086
|
+
// Only present when a chunk size is what overflowed, so a caller cannot mistake one
|
|
1087
|
+
// recommendation for the other.
|
|
1088
|
+
...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
|
|
919
1089
|
}
|
|
920
1090
|
);
|
|
921
1091
|
}
|
|
922
1092
|
}
|
|
923
1093
|
function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
|
|
924
1094
|
const LARGE_SYMBOL_TABLE_WARNING = usableOldSpaceLimit() * 0.125;
|
|
925
|
-
if (symbolTableSize
|
|
926
|
-
|
|
927
|
-
const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
|
|
928
|
-
const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
|
|
929
|
-
const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
|
|
930
|
-
console.warn(
|
|
931
|
-
`\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). Loading all ${totalRows.toLocaleString()} rows will use ~${estimatedMB}MB RAM. Consider using the maxRows parameter for better performance and lower memory usage.`
|
|
932
|
-
);
|
|
1095
|
+
if (symbolTableSize <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
1096
|
+
return;
|
|
933
1097
|
}
|
|
1098
|
+
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
1099
|
+
const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
|
|
1100
|
+
if (estimatedMemory <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
1101
|
+
return;
|
|
1102
|
+
}
|
|
1103
|
+
const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
|
|
1104
|
+
const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
|
|
1105
|
+
const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
|
|
1106
|
+
console.warn(
|
|
1107
|
+
`\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). This read materialises ${rowsToLoad.toLocaleString()} of ${totalRows.toLocaleString()} rows and will use ~${estimatedMB}MB RAM. Reading fewer rows - with limit, maxRows, or a narrower offset window - lowers the row cost, though the symbol table is read in full either way.`
|
|
1108
|
+
);
|
|
934
1109
|
}
|
|
935
1110
|
var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES;
|
|
936
1111
|
var init_memoryUtils = __esm({
|
|
@@ -971,7 +1146,7 @@ function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
|
|
|
971
1146
|
const maxMB = Math.round(MAX_SYMBOL_TABLE_SIZE / 1024 / 1024);
|
|
972
1147
|
const heapMB = Math.round(heapLimit / 1024 / 1024);
|
|
973
1148
|
throw new QvdValidationError(
|
|
974
|
-
`Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without maxRows, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
|
|
1149
|
+
`Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without a row window - maxRows, limit or offset - since the symbol table is read in full either way, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
|
|
975
1150
|
{
|
|
976
1151
|
file: filePath,
|
|
977
1152
|
symbolTableSize: symbolTableLength,
|
|
@@ -1045,7 +1220,7 @@ function validateRecordCount(totalRows, filePath, stage = "parseIndexTable") {
|
|
|
1045
1220
|
});
|
|
1046
1221
|
}
|
|
1047
1222
|
}
|
|
1048
|
-
function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null) {
|
|
1223
|
+
function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null, windowFirstRow = 0, bufferFirstRow = 0) {
|
|
1049
1224
|
if (isNaN(recordSize) || !Number.isSafeInteger(recordSize) || recordSize < 0) {
|
|
1050
1225
|
throw new QvdCorruptedError("Invalid record byte size", {
|
|
1051
1226
|
recordSize,
|
|
@@ -1107,23 +1282,28 @@ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, ind
|
|
|
1107
1282
|
}
|
|
1108
1283
|
}
|
|
1109
1284
|
const requiredIndexBytes = rowsToLoad * recordSize;
|
|
1110
|
-
|
|
1285
|
+
const bufferRecordStart = (windowFirstRow - bufferFirstRow) * recordSize;
|
|
1286
|
+
if (indexTableOffset + bufferRecordStart + requiredIndexBytes > bufferLength) {
|
|
1111
1287
|
throw new QvdCorruptedError("Index table truncated", {
|
|
1112
1288
|
indexTableOffset,
|
|
1113
1289
|
requiredBytes: requiredIndexBytes,
|
|
1114
|
-
availableBytes: Math.max(0, bufferLength - indexTableOffset),
|
|
1290
|
+
availableBytes: Math.max(0, bufferLength - indexTableOffset - bufferRecordStart),
|
|
1115
1291
|
rowsToLoad,
|
|
1292
|
+
windowFirstRow,
|
|
1293
|
+
bufferFirstRow,
|
|
1116
1294
|
recordSize,
|
|
1117
1295
|
bufferSize: bufferLength,
|
|
1118
1296
|
file: filePath,
|
|
1119
1297
|
stage: "parseIndexTable"
|
|
1120
1298
|
});
|
|
1121
1299
|
}
|
|
1122
|
-
|
|
1300
|
+
const requiredTableBytes = (windowFirstRow + rowsToLoad) * recordSize;
|
|
1301
|
+
if (indexTableLength < requiredTableBytes) {
|
|
1123
1302
|
throw new QvdCorruptedError("Index table length smaller than required", {
|
|
1124
1303
|
indexTableLength,
|
|
1125
|
-
requiredBytes:
|
|
1304
|
+
requiredBytes: requiredTableBytes,
|
|
1126
1305
|
rowsToLoad,
|
|
1306
|
+
windowFirstRow,
|
|
1127
1307
|
recordSize,
|
|
1128
1308
|
file: filePath,
|
|
1129
1309
|
stage: "parseIndexTable"
|
|
@@ -1453,6 +1633,7 @@ var QvdColumn, QvdColumnTable;
|
|
|
1453
1633
|
var init_QvdColumnTable = __esm({
|
|
1454
1634
|
"src/QvdColumnTable.js"() {
|
|
1455
1635
|
init_QvdErrors();
|
|
1636
|
+
init_readOptions();
|
|
1456
1637
|
QvdColumn = class {
|
|
1457
1638
|
/**
|
|
1458
1639
|
* @param {string} name The field name.
|
|
@@ -1647,9 +1828,21 @@ var init_QvdColumnTable = __esm({
|
|
|
1647
1828
|
/**
|
|
1648
1829
|
* Reads a QVD file as columns.
|
|
1649
1830
|
*
|
|
1831
|
+
* Takes the same options as `QvdDataFrame.fromQvd`, with the same meanings - one option
|
|
1832
|
+
* vocabulary for both read paths, because they are two answers about the same file rather than
|
|
1833
|
+
* two features. `{offset, limit}` is how a caller pages through a file columnwise; there is no
|
|
1834
|
+
* columnar `iterate()` because there is nothing for it to bound - a columnar read materialises
|
|
1835
|
+
* no rows, which is the memory chunking exists to cap.
|
|
1836
|
+
*
|
|
1650
1837
|
* @param {string} path The path to the QVD file.
|
|
1651
1838
|
* @param {Object} [options] Loading options, with the same meanings they have on `fromQvd`.
|
|
1652
|
-
* @param {number|null} [options.maxRows]
|
|
1839
|
+
* @param {number|null} [options.maxRows] Rows to decode. The older name for `limit`.
|
|
1840
|
+
* @param {number|null} [options.limit] Rows to decode, counting from `offset`.
|
|
1841
|
+
* @param {number} [options.offset] File row to start at.
|
|
1842
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
1843
|
+
* appear. Unselected fields have their symbols skipped entirely.
|
|
1844
|
+
* @param {Function} [options.onProgress] Progress callback, `{stage, current, total, percent}`.
|
|
1845
|
+
* @param {AbortSignal} [options.signal] Cancels the read.
|
|
1653
1846
|
* @param {string} [options.allowedDir] Directory the path must resolve inside.
|
|
1654
1847
|
* @param {number} [options.memorySafetyFactor] Fraction of the memory budget a load may use.
|
|
1655
1848
|
* @param {number} [options.symbolFilteringThreshold] Symbol table size above which a limited
|
|
@@ -1659,15 +1852,13 @@ var init_QvdColumnTable = __esm({
|
|
|
1659
1852
|
static async fromQvd(path3, options = {}) {
|
|
1660
1853
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
1661
1854
|
const reader = new QvdFileReader2(path3, {
|
|
1662
|
-
|
|
1663
|
-
memorySafetyFactor: options.memorySafetyFactor,
|
|
1664
|
-
symbolFilteringThreshold: options.symbolFilteringThreshold,
|
|
1855
|
+
...readerOptionsFrom(options),
|
|
1665
1856
|
// This read builds no rows, so the memory guard must not charge it for them. A columnar
|
|
1666
1857
|
// read of the 38MB taxi fixture completes in a 15MB heap; charged the row cost it was
|
|
1667
1858
|
// refused below a 2GB one.
|
|
1668
1859
|
materialisesRows: false
|
|
1669
1860
|
});
|
|
1670
|
-
return await reader.loadColumnar(options
|
|
1861
|
+
return await reader.loadColumnar(windowFrom(options));
|
|
1671
1862
|
}
|
|
1672
1863
|
/** @return {Array<string>} Field names, in file order. */
|
|
1673
1864
|
get columns() {
|
|
@@ -1715,7 +1906,7 @@ var QvdFileReader_exports = {};
|
|
|
1715
1906
|
__export(QvdFileReader_exports, {
|
|
1716
1907
|
QvdFileReader: () => QvdFileReader
|
|
1717
1908
|
});
|
|
1718
|
-
var MAX_HEADER_SIZE, READ_CHUNK_SIZE, QvdFileReader;
|
|
1909
|
+
var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, QvdFileReader;
|
|
1719
1910
|
var init_QvdFileReader = __esm({
|
|
1720
1911
|
"src/QvdFileReader.js"() {
|
|
1721
1912
|
init_QvdDataFrame();
|
|
@@ -1725,8 +1916,10 @@ var init_QvdFileReader = __esm({
|
|
|
1725
1916
|
init_memoryUtils();
|
|
1726
1917
|
init_validationUtils();
|
|
1727
1918
|
init_symbolParser();
|
|
1919
|
+
init_readOptions();
|
|
1728
1920
|
MAX_HEADER_SIZE = 16 * 1024 * 1024;
|
|
1729
1921
|
READ_CHUNK_SIZE = 512 * 1024 * 1024;
|
|
1922
|
+
ANALYSIS_SLICE_ROWS = 65536;
|
|
1730
1923
|
QvdFileReader = class {
|
|
1731
1924
|
/**
|
|
1732
1925
|
* Constructs a new QVD file parser.
|
|
@@ -1738,9 +1931,10 @@ var init_QvdFileReader = __esm({
|
|
|
1738
1931
|
* points outside it is rejected. Defaults to the current working directory. To permit
|
|
1739
1932
|
* an entire volume, pass its root explicitly ('/' on POSIX, 'C:\\' on Windows); a null or
|
|
1740
1933
|
* empty value falls back to the working directory rather than removing the restriction.
|
|
1741
|
-
* @param {number} [options.memorySafetyFactor=0.
|
|
1742
|
-
* load may use. The budget is the
|
|
1743
|
-
*
|
|
1934
|
+
* @param {number} [options.memorySafetyFactor=0.8] Fraction (0.0-1.0) of the memory budget a
|
|
1935
|
+
* load may use. The budget is the smaller of the V8 heap limit and any container memory limit;
|
|
1936
|
+
* what the OS reports as available is recorded for diagnostics and deliberately not allowed to
|
|
1937
|
+
* bind - see `getMemoryBudget`. Default is 0.8. **Zero disables the memory
|
|
1744
1938
|
* check entirely**, which is the escape hatch for runtimes whose limits cannot be measured -
|
|
1745
1939
|
* Bun reports its current heap as its heap limit - and for callers who would rather manage
|
|
1746
1940
|
* memory themselves than trust the estimate.
|
|
@@ -1751,29 +1945,93 @@ var init_QvdFileReader = __esm({
|
|
|
1751
1945
|
* above which a lazy load switches to the two-pass filtering path. The default of 50MB is
|
|
1752
1946
|
* the point where the extra analysis pass pays for itself; lower it to use filtering on
|
|
1753
1947
|
* smaller files, raise it to keep the simpler single-pass read for longer.
|
|
1948
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
1949
|
+
* appear. Null reads every field, in file order. An unknown or repeated name is refused.
|
|
1950
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
|
|
1951
|
+
* read proceeds - the same shape `QvdFileWriter` emits.
|
|
1952
|
+
* @param {AbortSignal} [options.signal] Cancels the read. When it is aborted the read throws
|
|
1953
|
+
* `signal.reason`, exactly as `signal.throwIfAborted()` does.
|
|
1754
1954
|
*/
|
|
1755
1955
|
constructor(filePath, options = {}) {
|
|
1756
1956
|
const {
|
|
1757
1957
|
allowedDir,
|
|
1758
1958
|
memorySafetyFactor = 0.8,
|
|
1759
1959
|
symbolFilteringThreshold = 50 * 1024 * 1024,
|
|
1760
|
-
materialisesRows = true
|
|
1960
|
+
materialisesRows = true,
|
|
1961
|
+
fields = null,
|
|
1962
|
+
onProgress,
|
|
1963
|
+
signal
|
|
1761
1964
|
} = options;
|
|
1762
1965
|
this._materialisesRows = materialisesRows;
|
|
1763
1966
|
this._path = validatePath(filePath, allowedDir);
|
|
1764
1967
|
this._memorySafetyFactor = memorySafetyFactor;
|
|
1765
1968
|
this._symbolFilteringThreshold = symbolFilteringThreshold;
|
|
1969
|
+
if (onProgress !== void 0 && typeof onProgress !== "function") {
|
|
1970
|
+
throw new QvdValidationError("onProgress must be a function", {
|
|
1971
|
+
provided: onProgress,
|
|
1972
|
+
type: typeof onProgress,
|
|
1973
|
+
file: this._path
|
|
1974
|
+
});
|
|
1975
|
+
}
|
|
1976
|
+
if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
|
|
1977
|
+
throw new QvdValidationError("signal must be an AbortSignal", {
|
|
1978
|
+
provided: signal,
|
|
1979
|
+
type: typeof signal,
|
|
1980
|
+
file: this._path
|
|
1981
|
+
});
|
|
1982
|
+
}
|
|
1983
|
+
this._requestedFields = fields === void 0 ? null : fields;
|
|
1984
|
+
this._onProgress = onProgress;
|
|
1985
|
+
this._signal = signal;
|
|
1766
1986
|
this._buffer = null;
|
|
1767
1987
|
this._headerOffset = null;
|
|
1768
1988
|
this._symbolTableOffset = null;
|
|
1769
1989
|
this._indexTableOffset = null;
|
|
1770
1990
|
this._header = null;
|
|
1991
|
+
this._allFields = null;
|
|
1992
|
+
this._selectedFields = null;
|
|
1993
|
+
this._fieldBitMetadataValidated = false;
|
|
1771
1994
|
this._symbolTable = null;
|
|
1772
1995
|
this._indexColumns = null;
|
|
1773
1996
|
this._rowsDecoded = 0;
|
|
1997
|
+
this._bufferFirstRow = 0;
|
|
1774
1998
|
this._fileSize = null;
|
|
1775
1999
|
this._headerMatchesFile = false;
|
|
1776
2000
|
}
|
|
2001
|
+
/**
|
|
2002
|
+
* Emits a progress event if a callback is registered.
|
|
2003
|
+
*
|
|
2004
|
+
* The same shape `QvdFileWriter._emitProgress` emits, deliberately: a caller who has written a
|
|
2005
|
+
* progress bar for a write should not have to write a second one for a read. The stage names
|
|
2006
|
+
* differ because the stages differ, but `symbol-table` and `index-table` mean the same thing on
|
|
2007
|
+
* both sides.
|
|
2008
|
+
*
|
|
2009
|
+
* @param {string} stage The current stage of the read.
|
|
2010
|
+
* @param {number} current The current progress value.
|
|
2011
|
+
* @param {number} total The total progress value.
|
|
2012
|
+
* @private
|
|
2013
|
+
*/
|
|
2014
|
+
_emitProgress(stage, current, total) {
|
|
2015
|
+
if (this._onProgress) {
|
|
2016
|
+
const percent = total > 0 ? Math.round(current / total * 100) : 100;
|
|
2017
|
+
this._onProgress({ stage, current, total, percent });
|
|
2018
|
+
}
|
|
2019
|
+
}
|
|
2020
|
+
/**
|
|
2021
|
+
* Throws if the caller has cancelled the read.
|
|
2022
|
+
*
|
|
2023
|
+
* Throws `signal.reason` - a `DOMException` named `AbortError` unless the caller aborted with a
|
|
2024
|
+
* reason of their own. That is what `AbortSignal` means everywhere else in Node, and inventing
|
|
2025
|
+
* a `QvdAbortError` here would make this library's cancellation the one a caller has to special
|
|
2026
|
+
* case.
|
|
2027
|
+
*
|
|
2028
|
+
* @private
|
|
2029
|
+
*/
|
|
2030
|
+
_throwIfAborted() {
|
|
2031
|
+
if (this._signal) {
|
|
2032
|
+
this._signal.throwIfAborted();
|
|
2033
|
+
}
|
|
2034
|
+
}
|
|
1777
2035
|
/**
|
|
1778
2036
|
* Reads the binary data of the QVD file.
|
|
1779
2037
|
*
|
|
@@ -1798,14 +2056,23 @@ var init_QvdFileReader = __esm({
|
|
|
1798
2056
|
* - Streaming for header finding is efficient for unknown header sizes
|
|
1799
2057
|
* - Direct byte-range reading for remaining data is fastest
|
|
1800
2058
|
*
|
|
1801
|
-
*
|
|
2059
|
+
* A window with a non-zero `offset` reads two ranges rather than one: the header and symbol
|
|
2060
|
+
* table from the front of the file, and the window's records from wherever they sit. The bytes
|
|
2061
|
+
* between are never read, which is what makes `{offset: 1_700_000, limit: 100}` on the taxi
|
|
2062
|
+
* fixture a 0.4MB read rather than a 38MB one.
|
|
2063
|
+
*
|
|
2064
|
+
* @param {QvdRowWindow} window The rows to read.
|
|
1802
2065
|
* @param {boolean} [headerOnly=false] Stop once the XML header has been read, leaving the
|
|
1803
2066
|
* symbol and index tables on disk. This is the metadata-only path: the header is a few
|
|
1804
2067
|
* kilobytes whatever the file's size, so reading a schema costs the same for a 40MB file as
|
|
1805
2068
|
* for a 40GB one.
|
|
2069
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
|
|
2070
|
+
* that is fewer than the window covers - see `_prepare`.
|
|
1806
2071
|
* @private
|
|
1807
2072
|
*/
|
|
1808
|
-
async _readData(
|
|
2073
|
+
async _readData(window = { offset: 0, limit: null }, headerOnly = false, liveRows = null) {
|
|
2074
|
+
this._throwIfAborted();
|
|
2075
|
+
this._emitProgress("read", 0, 1);
|
|
1809
2076
|
const HEADER_DELIMITER = "\r\n\0";
|
|
1810
2077
|
const CHUNK_SIZE = 64 * 1024;
|
|
1811
2078
|
const stream = fs.createReadStream(this._path, {
|
|
@@ -1875,13 +2142,14 @@ var init_QvdFileReader = __esm({
|
|
|
1875
2142
|
const totalRows = parseInt(headerObj["QvdTableHeader"]["NoOfRecords"], 10);
|
|
1876
2143
|
if (headerOnly) {
|
|
1877
2144
|
this._buffer = headerBuffer.subarray(0, headerEndIndex);
|
|
2145
|
+
this._emitProgress("read", 1, 1);
|
|
1878
2146
|
return;
|
|
1879
2147
|
}
|
|
1880
2148
|
let headerFields = headerObj["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
|
|
1881
2149
|
if (headerFields && !Array.isArray(headerFields)) {
|
|
1882
2150
|
headerFields = [headerFields];
|
|
1883
2151
|
}
|
|
1884
|
-
const columnCount = Array.isArray(headerFields) ? headerFields.length : 0;
|
|
2152
|
+
const columnCount = Array.isArray(headerFields) ? selectFields(headerFields, this._requestedFields, this._path).length : 0;
|
|
1885
2153
|
const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
|
|
1886
2154
|
(value) => Number.isSafeInteger(value) && value >= 0
|
|
1887
2155
|
);
|
|
@@ -1890,23 +2158,28 @@ var init_QvdFileReader = __esm({
|
|
|
1890
2158
|
this._fileSize = fileSize;
|
|
1891
2159
|
this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize;
|
|
1892
2160
|
}
|
|
2161
|
+
const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
|
|
2162
|
+
const windowRows = resolved.limit;
|
|
1893
2163
|
if (headerNumbersUsable && this._headerMatchesFile) {
|
|
1894
2164
|
validateMemoryAvailability(
|
|
1895
2165
|
symbolTableLength,
|
|
1896
|
-
|
|
2166
|
+
windowRows,
|
|
1897
2167
|
totalRows,
|
|
1898
2168
|
this._path,
|
|
1899
2169
|
this._memorySafetyFactor,
|
|
1900
2170
|
columnCount,
|
|
1901
|
-
this._materialisesRows
|
|
2171
|
+
this._materialisesRows,
|
|
2172
|
+
liveRows
|
|
1902
2173
|
);
|
|
1903
2174
|
}
|
|
1904
|
-
if (
|
|
2175
|
+
if (window.offset === 0 && window.limit === null) {
|
|
1905
2176
|
this._buffer = await fs.promises.readFile(this._path);
|
|
1906
2177
|
this._fileSize = this._buffer.length;
|
|
2178
|
+
this._bufferFirstRow = 0;
|
|
2179
|
+
this._emitProgress("read", 1, 1);
|
|
1907
2180
|
return;
|
|
1908
2181
|
}
|
|
1909
|
-
const rowsToLoad =
|
|
2182
|
+
const rowsToLoad = windowRows;
|
|
1910
2183
|
validateSymbolTableSizeEarly(symbolTableLength, this._path);
|
|
1911
2184
|
for (const [name, value] of [
|
|
1912
2185
|
["Offset", symbolTableLength],
|
|
@@ -1922,39 +2195,78 @@ var init_QvdFileReader = __esm({
|
|
|
1922
2195
|
});
|
|
1923
2196
|
}
|
|
1924
2197
|
}
|
|
2198
|
+
const skippedIndexBytes = resolved.offset * recordSize;
|
|
1925
2199
|
const indexTableBytesToRead = rowsToLoad * recordSize;
|
|
1926
2200
|
const totalBytesToRead = indexTableOffset + indexTableBytesToRead;
|
|
2201
|
+
const fileBytesRequired = indexTableOffset + skippedIndexBytes + indexTableBytesToRead;
|
|
1927
2202
|
const fd = await fs.promises.open(this._path, "r");
|
|
1928
2203
|
try {
|
|
1929
2204
|
const { size: fileSize } = await fd.stat();
|
|
1930
2205
|
this._fileSize = fileSize;
|
|
1931
|
-
if (
|
|
2206
|
+
if (fileBytesRequired > fileSize) {
|
|
1932
2207
|
throw new QvdCorruptedError("The file is shorter than its header claims.", {
|
|
1933
2208
|
file: this._path,
|
|
1934
2209
|
fileSize,
|
|
1935
|
-
requiredBytes:
|
|
2210
|
+
requiredBytes: fileBytesRequired,
|
|
1936
2211
|
stage: "readData"
|
|
1937
2212
|
});
|
|
1938
2213
|
}
|
|
1939
2214
|
this._buffer = Buffer.alloc(totalBytesToRead);
|
|
1940
|
-
|
|
1941
|
-
|
|
1942
|
-
|
|
1943
|
-
|
|
1944
|
-
|
|
1945
|
-
|
|
1946
|
-
|
|
1947
|
-
|
|
1948
|
-
|
|
1949
|
-
|
|
1950
|
-
stage: "readData"
|
|
1951
|
-
});
|
|
1952
|
-
}
|
|
1953
|
-
position += bytesRead;
|
|
2215
|
+
await this._readRange(fd, 0, indexTableOffset, 0, fileSize, totalBytesToRead);
|
|
2216
|
+
if (indexTableBytesToRead > 0) {
|
|
2217
|
+
await this._readRange(
|
|
2218
|
+
fd,
|
|
2219
|
+
indexTableOffset,
|
|
2220
|
+
indexTableBytesToRead,
|
|
2221
|
+
indexTableOffset + skippedIndexBytes,
|
|
2222
|
+
fileSize,
|
|
2223
|
+
fileBytesRequired
|
|
2224
|
+
);
|
|
1954
2225
|
}
|
|
2226
|
+
this._bufferFirstRow = resolved.offset;
|
|
1955
2227
|
} finally {
|
|
1956
2228
|
await fd.close();
|
|
1957
2229
|
}
|
|
2230
|
+
this._emitProgress("read", 1, 1);
|
|
2231
|
+
}
|
|
2232
|
+
/**
|
|
2233
|
+
* Reads one byte range of the file into the buffer.
|
|
2234
|
+
*
|
|
2235
|
+
* Read in bounded chunks, checking bytesRead each time. A single fs.read call with a length of
|
|
2236
|
+
* 2^31 or more does not throw - it trips a C++ assertion and aborts the whole process, which no
|
|
2237
|
+
* try/catch can intercept.
|
|
2238
|
+
*
|
|
2239
|
+
* @param {import('fs/promises').FileHandle} fd The open file.
|
|
2240
|
+
* @param {number} bufferOffset Where in the buffer to write.
|
|
2241
|
+
* @param {number} byteCount How many bytes to read.
|
|
2242
|
+
* @param {number} filePosition Where in the file to read from.
|
|
2243
|
+
* @param {number} fileSize The file's size, for the error.
|
|
2244
|
+
* @param {number} requiredBytes Bytes the whole read needs, for the error.
|
|
2245
|
+
* @private
|
|
2246
|
+
*/
|
|
2247
|
+
async _readRange(fd, bufferOffset, byteCount, filePosition, fileSize, requiredBytes) {
|
|
2248
|
+
assert2(this._buffer, "The read buffer has not been allocated.");
|
|
2249
|
+
let done = 0;
|
|
2250
|
+
while (done < byteCount) {
|
|
2251
|
+
const length = Math.min(READ_CHUNK_SIZE, byteCount - done);
|
|
2252
|
+
const { bytesRead } = await fd.read(this._buffer, bufferOffset + done, length, filePosition + done);
|
|
2253
|
+
if (bytesRead === 0) {
|
|
2254
|
+
throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
|
|
2255
|
+
file: this._path,
|
|
2256
|
+
fileSize,
|
|
2257
|
+
// Two numbers, because they stopped being the same one when a window began reading two
|
|
2258
|
+
// ranges: `bytesRead` is how much of this range arrived, `filePosition` is where in the
|
|
2259
|
+
// file it gave up. Reporting the position under the name of the count made a windowed
|
|
2260
|
+
// read of a truncated file claim tens of megabytes had been read when a few hundred
|
|
2261
|
+
// bytes had.
|
|
2262
|
+
bytesRead: done,
|
|
2263
|
+
filePosition: filePosition + done,
|
|
2264
|
+
requiredBytes,
|
|
2265
|
+
stage: "readData"
|
|
2266
|
+
});
|
|
2267
|
+
}
|
|
2268
|
+
done += bytesRead;
|
|
2269
|
+
}
|
|
1958
2270
|
}
|
|
1959
2271
|
/**
|
|
1960
2272
|
* Parses the XML header of the QVD file. This method is part of the parsing process
|
|
@@ -2014,6 +2326,8 @@ var init_QvdFileReader = __esm({
|
|
|
2014
2326
|
this._headerOffset = headerBeginIndex;
|
|
2015
2327
|
this._symbolTableOffset = headerEndIndex;
|
|
2016
2328
|
this._indexTableOffset = this._symbolTableOffset + parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2329
|
+
this._allFields = fieldList;
|
|
2330
|
+
this._selectedFields = selectFields(this._allFields, this._requestedFields, this._path);
|
|
2017
2331
|
}
|
|
2018
2332
|
/**
|
|
2019
2333
|
* Establishes the geometry of the index table, and validates it.
|
|
@@ -2024,14 +2338,15 @@ var init_QvdFileReader = __esm({
|
|
|
2024
2338
|
* about keeping the sign in step with the other one: the two could drift, and #113 is what
|
|
2025
2339
|
* that looks like when they do. There is one copy now.
|
|
2026
2340
|
*
|
|
2027
|
-
* @param {
|
|
2341
|
+
* @param {QvdRowWindow} window The rows of interest, as file row indices.
|
|
2028
2342
|
* @param {string} stage Stage name for any error raised here.
|
|
2029
2343
|
* @return {{fields: Array<any>, recordSize: number, totalRows: number, rowsToLoad: number,
|
|
2030
|
-
* indexBuffer: Buffer}} The record geometry.
|
|
2344
|
+
* indexBuffer: Buffer}} The record geometry. `indexBuffer` starts at the window's first
|
|
2345
|
+
* record, so the decoder always counts from zero.
|
|
2031
2346
|
* @private
|
|
2032
2347
|
*/
|
|
2033
|
-
_planIndexTable(
|
|
2034
|
-
if (!this._buffer || !this._header || !this._indexTableOffset) {
|
|
2348
|
+
_planIndexTable(window, stage) {
|
|
2349
|
+
if (!this._buffer || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
|
|
2035
2350
|
throw new QvdCorruptedError(
|
|
2036
2351
|
"The QVD file has not been loaded in the proper order or has not been loaded at all.",
|
|
2037
2352
|
{
|
|
@@ -2040,14 +2355,12 @@ var init_QvdFileReader = __esm({
|
|
|
2040
2355
|
}
|
|
2041
2356
|
);
|
|
2042
2357
|
}
|
|
2043
|
-
|
|
2044
|
-
|
|
2045
|
-
fields = [fields];
|
|
2046
|
-
}
|
|
2358
|
+
const allFields = this._allFields;
|
|
2359
|
+
const fields = this._selectedFields;
|
|
2047
2360
|
const recordSize = parseInt(this._header["QvdTableHeader"]["RecordByteSize"], 10);
|
|
2048
2361
|
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
2049
|
-
const rowsToLoad = rowLimit !== null ? Math.min(rowLimit, totalRows) : totalRows;
|
|
2050
2362
|
const indexTableLength = parseInt(this._header["QvdTableHeader"]["Length"], 10);
|
|
2363
|
+
const { offset: firstRow, limit: rowsToLoad } = resolveWindow(window, totalRows);
|
|
2051
2364
|
validateIndexTableMetadata(
|
|
2052
2365
|
recordSize,
|
|
2053
2366
|
totalRows,
|
|
@@ -2056,11 +2369,20 @@ var init_QvdFileReader = __esm({
|
|
|
2056
2369
|
this._buffer.length,
|
|
2057
2370
|
rowsToLoad,
|
|
2058
2371
|
this._path,
|
|
2059
|
-
this._fileSize
|
|
2372
|
+
this._fileSize,
|
|
2373
|
+
firstRow,
|
|
2374
|
+
this._bufferFirstRow
|
|
2375
|
+
);
|
|
2376
|
+
const bufferRecordStart = (firstRow - this._bufferFirstRow) * recordSize;
|
|
2377
|
+
const indexBuffer = this._buffer.subarray(
|
|
2378
|
+
this._indexTableOffset + bufferRecordStart,
|
|
2379
|
+
this._indexTableOffset + bufferRecordStart + rowsToLoad * recordSize
|
|
2060
2380
|
);
|
|
2061
|
-
|
|
2062
|
-
|
|
2063
|
-
|
|
2381
|
+
if (!this._fieldBitMetadataValidated) {
|
|
2382
|
+
for (const field of allFields) {
|
|
2383
|
+
validateFieldBitMetadata(field, recordSize, this._path);
|
|
2384
|
+
}
|
|
2385
|
+
this._fieldBitMetadataValidated = true;
|
|
2064
2386
|
}
|
|
2065
2387
|
assert2(
|
|
2066
2388
|
rowsToLoad === 0 || recordSize === 0 || Math.floor(indexBuffer.length / recordSize) >= rowsToLoad,
|
|
@@ -2072,31 +2394,45 @@ var init_QvdFileReader = __esm({
|
|
|
2072
2394
|
* Analyzes the index table to determine which symbols are actually needed.
|
|
2073
2395
|
* This is used for two-pass symbol filtering optimization.
|
|
2074
2396
|
*
|
|
2075
|
-
*
|
|
2076
|
-
*
|
|
2397
|
+
* Only the selected fields are analysed. An unselected field's symbols are never parsed, so
|
|
2398
|
+
* there is nothing for a usage set to filter and decoding its column would be a pass over the
|
|
2399
|
+
* whole window for an answer nobody reads.
|
|
2400
|
+
*
|
|
2401
|
+
* @param {QvdRowWindow} window The rows to analyse.
|
|
2402
|
+
* @return {Promise<Array<Set<number>>>} One set of needed symbol indices per selected field, in
|
|
2403
|
+
* the same order `_parseSymbolTable` walks them.
|
|
2077
2404
|
* @private
|
|
2078
2405
|
*/
|
|
2079
|
-
async _analyzeIndexTableSymbolUsage(
|
|
2080
|
-
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(
|
|
2081
|
-
const symbolUsage =
|
|
2082
|
-
const
|
|
2083
|
-
|
|
2406
|
+
async _analyzeIndexTableSymbolUsage(window) {
|
|
2407
|
+
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
|
|
2408
|
+
const symbolUsage = [];
|
|
2409
|
+
const sliceRows = Math.min(rowsToLoad, ANALYSIS_SLICE_ROWS);
|
|
2410
|
+
const column = new Int32Array(sliceRows);
|
|
2411
|
+
fields.forEach((field, position) => {
|
|
2412
|
+
this._throwIfAborted();
|
|
2084
2413
|
const needed = /* @__PURE__ */ new Set();
|
|
2085
|
-
symbolUsage
|
|
2086
|
-
|
|
2087
|
-
|
|
2088
|
-
|
|
2089
|
-
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
|
|
2093
|
-
|
|
2094
|
-
|
|
2095
|
-
|
|
2096
|
-
|
|
2097
|
-
|
|
2414
|
+
symbolUsage[position] = needed;
|
|
2415
|
+
const bitOffset = parseInt(field["BitOffset"], 10);
|
|
2416
|
+
const bitWidth = parseInt(field["BitWidth"], 10);
|
|
2417
|
+
const bias = parseInt(field["Bias"], 10);
|
|
2418
|
+
for (let first = 0; first < rowsToLoad; first += sliceRows) {
|
|
2419
|
+
const count = Math.min(sliceRows, rowsToLoad - first);
|
|
2420
|
+
decodeIndexColumn(
|
|
2421
|
+
first === 0 ? indexBuffer : indexBuffer.subarray(first * recordSize),
|
|
2422
|
+
recordSize,
|
|
2423
|
+
count,
|
|
2424
|
+
bitOffset,
|
|
2425
|
+
bitWidth,
|
|
2426
|
+
bias,
|
|
2427
|
+
column
|
|
2428
|
+
);
|
|
2429
|
+
for (let row = 0; row < count; row++) {
|
|
2430
|
+
if (column[row] >= 0) {
|
|
2431
|
+
needed.add(column[row]);
|
|
2432
|
+
}
|
|
2098
2433
|
}
|
|
2099
2434
|
}
|
|
2435
|
+
this._emitProgress("symbol-analysis", position + 1, fields.length);
|
|
2100
2436
|
});
|
|
2101
2437
|
return symbolUsage;
|
|
2102
2438
|
}
|
|
@@ -2104,12 +2440,20 @@ var init_QvdFileReader = __esm({
|
|
|
2104
2440
|
* Parses the symbol table of the QVD file. This method is part of the parsing process
|
|
2105
2441
|
* and should not be called directly.
|
|
2106
2442
|
*
|
|
2107
|
-
*
|
|
2108
|
-
*
|
|
2109
|
-
*
|
|
2443
|
+
* A field the caller did not select is skipped whole. Its symbol area is neither scanned nor
|
|
2444
|
+
* parsed - the per-field `Offset` and `Length` say exactly where it is, so there is nothing to
|
|
2445
|
+
* walk past - and that is where field selection earns its keep. The index decode is cheap by
|
|
2446
|
+
* comparison; parsing symbols is not.
|
|
2447
|
+
*
|
|
2448
|
+
* @param {Array<Set<number>>|null} symbolsToKeep Optional set of symbol indices to keep per
|
|
2449
|
+
* selected field, indexed by position. If provided, only these symbols will be parsed
|
|
2450
|
+
* (two-pass filtering optimization).
|
|
2451
|
+
* @param {number} rowsToLoad Rows the read covers, for memory estimation.
|
|
2452
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
|
|
2453
|
+
* that is fewer than the window covers - see `_prepare`.
|
|
2110
2454
|
*/
|
|
2111
|
-
async _parseSymbolTable(symbolsToKeep = null,
|
|
2112
|
-
if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset) {
|
|
2455
|
+
async _parseSymbolTable(symbolsToKeep = null, rowsToLoad = 0, liveRows = null) {
|
|
2456
|
+
if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
|
|
2113
2457
|
throw new QvdCorruptedError(
|
|
2114
2458
|
"The QVD file has not been loaded in the proper order or has not been loaded at all.",
|
|
2115
2459
|
{
|
|
@@ -2118,7 +2462,8 @@ var init_QvdFileReader = __esm({
|
|
|
2118
2462
|
}
|
|
2119
2463
|
);
|
|
2120
2464
|
}
|
|
2121
|
-
|
|
2465
|
+
const allFields = this._allFields;
|
|
2466
|
+
const fields = this._selectedFields;
|
|
2122
2467
|
const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
|
|
2123
2468
|
const symbolTableSize = symbolBuffer.length;
|
|
2124
2469
|
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
@@ -2126,32 +2471,25 @@ var init_QvdFileReader = __esm({
|
|
|
2126
2471
|
if (this._headerMatchesFile) {
|
|
2127
2472
|
validateMemoryAvailability(
|
|
2128
2473
|
symbolTableSize,
|
|
2129
|
-
|
|
2474
|
+
rowsToLoad,
|
|
2130
2475
|
totalRows,
|
|
2131
2476
|
this._path,
|
|
2132
2477
|
this._memorySafetyFactor,
|
|
2133
|
-
|
|
2134
|
-
this._materialisesRows
|
|
2478
|
+
fields.length,
|
|
2479
|
+
this._materialisesRows,
|
|
2480
|
+
liveRows
|
|
2135
2481
|
);
|
|
2136
2482
|
}
|
|
2137
|
-
warnLargeSymbolTable(
|
|
2138
|
-
|
|
2139
|
-
maxRows,
|
|
2140
|
-
totalRows,
|
|
2141
|
-
Array.isArray(fields) ? fields.length : 1,
|
|
2142
|
-
this._materialisesRows
|
|
2143
|
-
);
|
|
2144
|
-
if (!Array.isArray(fields)) {
|
|
2145
|
-
fields = [fields];
|
|
2146
|
-
}
|
|
2147
|
-
for (const field of fields) {
|
|
2483
|
+
warnLargeSymbolTable(symbolTableSize, rowsToLoad, totalRows, fields.length, this._materialisesRows);
|
|
2484
|
+
for (const field of allFields) {
|
|
2148
2485
|
validateFieldMetadata(field, symbolBuffer.length, this._path);
|
|
2149
2486
|
}
|
|
2150
|
-
this._symbolTable = fields.map((field) => {
|
|
2487
|
+
this._symbolTable = fields.map((field, position) => {
|
|
2488
|
+
this._throwIfAborted();
|
|
2151
2489
|
const symbolsOffset = parseInt(field["Offset"], 10);
|
|
2152
2490
|
const symbolsLength = parseInt(field["Length"], 10);
|
|
2153
2491
|
const fieldName = field["FieldName"];
|
|
2154
|
-
const neededSymbols = symbolsToKeep ? symbolsToKeep
|
|
2492
|
+
const neededSymbols = symbolsToKeep ? symbolsToKeep[position] : null;
|
|
2155
2493
|
const filteringEnabled = neededSymbols !== null;
|
|
2156
2494
|
const symbols = [];
|
|
2157
2495
|
let symbolIndex = 0;
|
|
@@ -2171,6 +2509,7 @@ var init_QvdFileReader = __esm({
|
|
|
2171
2509
|
pointer += bytesRead - 1;
|
|
2172
2510
|
symbolIndex++;
|
|
2173
2511
|
}
|
|
2512
|
+
this._emitProgress("symbol-table", position + 1, fields.length);
|
|
2174
2513
|
return symbols;
|
|
2175
2514
|
});
|
|
2176
2515
|
}
|
|
@@ -2189,13 +2528,18 @@ var init_QvdFileReader = __esm({
|
|
|
2189
2528
|
* same for every row, so they are hoisted out of the loop and the inner loop does arithmetic
|
|
2190
2529
|
* into a typed array and nothing else. Rows are assembled later, once, in `load()`.
|
|
2191
2530
|
*
|
|
2192
|
-
*
|
|
2531
|
+
* The window is what makes chunked iteration cheap: `decodeIndexColumn` walks records by
|
|
2532
|
+
* `base += recordSize`, so decoding rows k to k+n is a question of where the buffer slice starts
|
|
2533
|
+
* and how many iterations run. Nothing about the decoder changed to support it.
|
|
2534
|
+
*
|
|
2535
|
+
* @param {QvdRowWindow} window The rows to decode.
|
|
2193
2536
|
*/
|
|
2194
|
-
async _parseIndexTable(
|
|
2195
|
-
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(
|
|
2537
|
+
async _parseIndexTable(window) {
|
|
2538
|
+
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "parseIndexTable");
|
|
2196
2539
|
this._rowsDecoded = rowsToLoad;
|
|
2197
|
-
this._indexColumns = fields.map(
|
|
2198
|
-
(
|
|
2540
|
+
this._indexColumns = fields.map((field, position) => {
|
|
2541
|
+
this._throwIfAborted();
|
|
2542
|
+
const column = decodeIndexColumn(
|
|
2199
2543
|
indexBuffer,
|
|
2200
2544
|
recordSize,
|
|
2201
2545
|
rowsToLoad,
|
|
@@ -2203,8 +2547,10 @@ var init_QvdFileReader = __esm({
|
|
|
2203
2547
|
parseInt(field["BitWidth"], 10),
|
|
2204
2548
|
parseInt(field["Bias"], 10),
|
|
2205
2549
|
new Int32Array(rowsToLoad)
|
|
2206
|
-
)
|
|
2207
|
-
|
|
2550
|
+
);
|
|
2551
|
+
this._emitProgress("index-table", position + 1, fields.length);
|
|
2552
|
+
return column;
|
|
2553
|
+
});
|
|
2208
2554
|
}
|
|
2209
2555
|
/**
|
|
2210
2556
|
* Reads the file's schema and header metadata, without touching the symbol or index tables.
|
|
@@ -2221,8 +2567,11 @@ var init_QvdFileReader = __esm({
|
|
|
2221
2567
|
* @return {Promise<import('./QvdDataFrame.js').QvdFileMetadata>} The file's schema and header.
|
|
2222
2568
|
*/
|
|
2223
2569
|
async loadMetadata() {
|
|
2224
|
-
await this._readData(null, true);
|
|
2570
|
+
await this._readData({ offset: 0, limit: null }, true);
|
|
2571
|
+
this._emitProgress("header", 0, 1);
|
|
2225
2572
|
await this._parseHeader();
|
|
2573
|
+
this._emitProgress("header", 1, 1);
|
|
2574
|
+
this._throwIfAborted();
|
|
2226
2575
|
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2227
2576
|
const header = this._header["QvdTableHeader"];
|
|
2228
2577
|
let fields = header["Fields"]?.["QvdFieldHeader"] ?? [];
|
|
@@ -2257,114 +2606,200 @@ var init_QvdFileReader = __esm({
|
|
|
2257
2606
|
/**
|
|
2258
2607
|
* Loads the QVD file into memory and parses it.
|
|
2259
2608
|
*
|
|
2260
|
-
* @param {number|null
|
|
2261
|
-
*
|
|
2262
|
-
*
|
|
2609
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
|
|
2610
|
+
* The rows to load. A number or null means what it always meant - the first N rows, or all of
|
|
2611
|
+
* them - and `{offset, limit}` is the same thing said more precisely, so `5` and
|
|
2612
|
+
* `{offset: 0, limit: 5}` are one read. `maxRows` is accepted as a second name for `limit`.
|
|
2613
|
+
* @throws {QvdValidationError} If the window is not a non-negative integer, null, or a valid
|
|
2614
|
+
* `{offset, limit}` object.
|
|
2263
2615
|
* @return {Promise<QvdDataFrame>} The loaded QVD file.
|
|
2264
2616
|
*/
|
|
2265
|
-
async load(
|
|
2266
|
-
const
|
|
2617
|
+
async load(window = null) {
|
|
2618
|
+
const rows = normaliseWindow(window, this._path);
|
|
2619
|
+
const prepared = await this._prepare(rows);
|
|
2620
|
+
await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
|
|
2621
|
+
const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
|
|
2622
|
+
return new QvdDataFrame(data, prepared.columns, prepared.metadata, {
|
|
2623
|
+
...prepared.loadStats,
|
|
2624
|
+
rowsLoaded: data.length
|
|
2625
|
+
});
|
|
2626
|
+
}
|
|
2627
|
+
/**
|
|
2628
|
+
* Reads the file as columns, without ever materialising rows.
|
|
2629
|
+
*
|
|
2630
|
+
* Shares every step with `load()` up to the point where rows would be built - see `_prepare`.
|
|
2631
|
+
* What it keeps instead is what the decoder already produced: one `Int32Array` of stored
|
|
2632
|
+
* indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
|
|
2633
|
+
* fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
|
|
2634
|
+
* bytes per row rather than a boxed value per cell, and the symbols are a few thousand
|
|
2635
|
+
* entries shared across every row that uses them.
|
|
2636
|
+
*
|
|
2637
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
|
|
2638
|
+
* The rows to decode, in the same spellings `load()` accepts.
|
|
2639
|
+
* @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
|
|
2640
|
+
*/
|
|
2641
|
+
async loadColumnar(window = null) {
|
|
2642
|
+
const rows = normaliseWindow(window, this._path);
|
|
2643
|
+
const prepared = await this._prepare(rows);
|
|
2644
|
+
await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
|
|
2645
|
+
const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
|
|
2267
2646
|
assert2(this._indexColumns, "The QVD file index table has not been parsed.");
|
|
2268
|
-
|
|
2269
|
-
|
|
2270
|
-
|
|
2271
|
-
|
|
2272
|
-
|
|
2273
|
-
|
|
2274
|
-
|
|
2275
|
-
|
|
2276
|
-
}
|
|
2277
|
-
data[row] = values;
|
|
2278
|
-
}
|
|
2279
|
-
loadStats.rowsLoaded = data.length;
|
|
2280
|
-
return new QvdDataFrame(data, columns, metadata, loadStats);
|
|
2647
|
+
return new QvdColumnTable2({
|
|
2648
|
+
columns: prepared.columns,
|
|
2649
|
+
codesByField: this._indexColumns,
|
|
2650
|
+
symbolsByField: prepared.resolvedByField,
|
|
2651
|
+
rowCount: this._rowsDecoded,
|
|
2652
|
+
metadata: prepared.metadata,
|
|
2653
|
+
loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
|
|
2654
|
+
});
|
|
2281
2655
|
}
|
|
2282
2656
|
/**
|
|
2283
|
-
*
|
|
2657
|
+
* Yields the window as data frames of at most `chunkSize` rows.
|
|
2284
2658
|
*
|
|
2285
|
-
*
|
|
2286
|
-
*
|
|
2287
|
-
*
|
|
2288
|
-
*
|
|
2289
|
-
*
|
|
2659
|
+
* The file is opened, read and parsed **once**; only the index decode and the row building
|
|
2660
|
+
* happen per chunk. That is the whole reason this exists as a method rather than as a loop of
|
|
2661
|
+
* `load({offset, limit})` calls at the call site: the symbol table has to be parsed in full
|
|
2662
|
+
* whatever the chunk size - a stored index in the last chunk can address the first symbol -
|
|
2663
|
+
* and re-parsing it per chunk is what makes the obvious implementation cost more than a plain
|
|
2664
|
+
* load rather than less. PyQvd's chunked read does re-read it, and the comment on #140 records
|
|
2665
|
+
* that as a limitation rather than a design.
|
|
2290
2666
|
*
|
|
2291
|
-
*
|
|
2292
|
-
*
|
|
2293
|
-
*
|
|
2294
|
-
*
|
|
2667
|
+
* What it bounds is row materialisation, which is what actually dominates a large read's heap.
|
|
2668
|
+
* Two chunks of rows are alive at a time, not one - `for await` keeps the yielded frame
|
|
2669
|
+
* reachable while this generator builds the next - which is why `liveRows` below is
|
|
2670
|
+
* `chunkSize * 2`, and why the heap it needs is twice what one chunk suggests.
|
|
2671
|
+
*
|
|
2672
|
+
* A window covering no rows yields nothing at all, rather than one empty frame - so
|
|
2673
|
+
* `for await` over an exhausted offset does nothing, which is what a paging loop wants.
|
|
2674
|
+
*
|
|
2675
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} window
|
|
2676
|
+
* The rows to cover, in the same spellings `load()` accepts.
|
|
2677
|
+
* @param {number} chunkSize Rows per frame. Must be a positive integer.
|
|
2678
|
+
* @return {AsyncGenerator<QvdDataFrame>} The chunks, in order.
|
|
2295
2679
|
*/
|
|
2296
|
-
async
|
|
2297
|
-
if (
|
|
2298
|
-
throw new QvdValidationError("
|
|
2299
|
-
provided:
|
|
2300
|
-
type: typeof
|
|
2680
|
+
async *iterateRows(window, chunkSize) {
|
|
2681
|
+
if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
|
|
2682
|
+
throw new QvdValidationError("chunkSize must be a positive integer", {
|
|
2683
|
+
provided: chunkSize,
|
|
2684
|
+
type: typeof chunkSize,
|
|
2301
2685
|
file: this._path
|
|
2302
2686
|
});
|
|
2303
2687
|
}
|
|
2304
|
-
|
|
2688
|
+
const liveRows = { rows: chunkSize * 2, perChunk: 2 };
|
|
2689
|
+
const rows = normaliseWindow(window, this._path);
|
|
2690
|
+
const prepared = await this._prepare(rows, liveRows);
|
|
2691
|
+
for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
|
|
2692
|
+
this._throwIfAborted();
|
|
2693
|
+
const count = Math.min(chunkSize, prepared.rowsAvailable - done);
|
|
2694
|
+
const offset = prepared.offset + done;
|
|
2695
|
+
await this._parseIndexTable({ offset, limit: count });
|
|
2696
|
+
const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
|
|
2697
|
+
yield new QvdDataFrame(data, prepared.columns, prepared.metadata, {
|
|
2698
|
+
...prepared.loadStats,
|
|
2699
|
+
offset,
|
|
2700
|
+
rowsLoaded: data.length
|
|
2701
|
+
});
|
|
2702
|
+
}
|
|
2703
|
+
}
|
|
2704
|
+
/**
|
|
2705
|
+
* Reads the file and resolves its symbols, stopping short of decoding any rows.
|
|
2706
|
+
*
|
|
2707
|
+
* Everything `load()`, `loadColumnar()` and `iterateRows()` have in common, which is everything
|
|
2708
|
+
* that depends on the file rather than on the window. Two read paths for one binary format is
|
|
2709
|
+
* the drift risk #113 is the standing example of - a stored index resolved one way here and
|
|
2710
|
+
* another way there returns plausible wrong values and throws nothing - so there is one path,
|
|
2711
|
+
* and the entry points differ only in what they do with what it returns and how many rows they
|
|
2712
|
+
* ask for at a time.
|
|
2713
|
+
*
|
|
2714
|
+
* @param {QvdRowWindow} window The rows the read covers.
|
|
2715
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows] Rows held at one instant when that
|
|
2716
|
+
* is fewer than the window covers, and how many of them one row of the caller's chunk size
|
|
2717
|
+
* accounts for. Only `iterateRows` passes it; every other read holds what it covers.
|
|
2718
|
+
* @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
|
|
2719
|
+
* resolvedByField: Array<Array<any>>, rowsAvailable: number, offset: number}>} The parsed
|
|
2720
|
+
* file, with the window as it resolved against it.
|
|
2721
|
+
* @private
|
|
2722
|
+
*/
|
|
2723
|
+
async _prepare(window, liveRows = null) {
|
|
2724
|
+
this._throwIfAborted();
|
|
2725
|
+
await this._readData(window, false, liveRows);
|
|
2726
|
+
this._emitProgress("header", 0, 1);
|
|
2305
2727
|
await this._parseHeader();
|
|
2728
|
+
this._emitProgress("header", 1, 1);
|
|
2729
|
+
this._throwIfAborted();
|
|
2730
|
+
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2731
|
+
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
2732
|
+
const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2733
|
+
const resolved = resolveWindow(window, totalRows);
|
|
2734
|
+
const rowsAvailable = resolved.limit;
|
|
2306
2735
|
let symbolsToKeep = null;
|
|
2307
2736
|
let symbolsKept = null;
|
|
2308
|
-
if (
|
|
2309
|
-
const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2737
|
+
if (window.limit !== null || window.offset > 0) {
|
|
2310
2738
|
if (symbolTableLength > this._symbolFilteringThreshold) {
|
|
2311
|
-
symbolsToKeep = await this._analyzeIndexTableSymbolUsage(
|
|
2312
|
-
symbolsKept =
|
|
2739
|
+
symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
|
|
2740
|
+
symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
|
|
2313
2741
|
}
|
|
2314
2742
|
}
|
|
2315
|
-
await this._parseSymbolTable(symbolsToKeep,
|
|
2316
|
-
await this._parseIndexTable(maxRows);
|
|
2317
|
-
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2743
|
+
await this._parseSymbolTable(symbolsToKeep, rowsAvailable, liveRows);
|
|
2318
2744
|
assert2(this._symbolTable, "The QVD file symbol table has not been parsed.");
|
|
2319
|
-
|
|
2745
|
+
this._throwIfAborted();
|
|
2320
2746
|
const resolvedByField = this._symbolTable.map((symbols) => {
|
|
2321
|
-
const
|
|
2747
|
+
const resolved2 = new Array(symbols.length);
|
|
2322
2748
|
for (let index = 0; index < symbols.length; index++) {
|
|
2323
2749
|
const value = symbols[index]?.toPrimaryValue();
|
|
2324
|
-
|
|
2750
|
+
resolved2[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
|
|
2325
2751
|
}
|
|
2326
|
-
return
|
|
2752
|
+
return resolved2;
|
|
2327
2753
|
});
|
|
2328
|
-
|
|
2329
|
-
|
|
2330
|
-
fields = [fields];
|
|
2331
|
-
}
|
|
2332
|
-
const columns = fields.map((field) => field["FieldName"]);
|
|
2754
|
+
assert2(this._selectedFields, "The QVD file fields have not been resolved.");
|
|
2755
|
+
const columns = this._selectedFields.map((field) => field["FieldName"]);
|
|
2333
2756
|
const metadata = this._header["QvdTableHeader"];
|
|
2334
2757
|
const loadStats = {
|
|
2335
|
-
symbolTableBytes:
|
|
2336
|
-
totalRows
|
|
2337
|
-
rowsLoaded:
|
|
2758
|
+
symbolTableBytes: symbolTableLength,
|
|
2759
|
+
totalRows,
|
|
2760
|
+
rowsLoaded: 0,
|
|
2761
|
+
offset: resolved.offset,
|
|
2338
2762
|
symbolFiltering: symbolsToKeep !== null,
|
|
2339
2763
|
symbolsKept
|
|
2340
2764
|
};
|
|
2341
|
-
return { columns, metadata, loadStats, resolvedByField };
|
|
2765
|
+
return { columns, metadata, loadStats, resolvedByField, rowsAvailable, offset: resolved.offset };
|
|
2342
2766
|
}
|
|
2343
2767
|
/**
|
|
2344
|
-
*
|
|
2768
|
+
* Builds rows from the columns currently decoded.
|
|
2345
2769
|
*
|
|
2346
|
-
*
|
|
2347
|
-
*
|
|
2348
|
-
*
|
|
2349
|
-
*
|
|
2350
|
-
*
|
|
2351
|
-
* entries shared across every row that uses them.
|
|
2770
|
+
* `data` stays eager: of the four ways this library is used - a full read, a preview already
|
|
2771
|
+
* bounded by a limit, writing an array out, and reading metadata - not one is helped by
|
|
2772
|
+
* materialising a row only when it is touched, and a lazy accessor would cost a proxy, a cache
|
|
2773
|
+
* and mutation semantics to serve none of them. A caller who wants columns without paying for
|
|
2774
|
+
* rows uses `QvdColumnTable`, which stops before this loop.
|
|
2352
2775
|
*
|
|
2353
|
-
* @param {
|
|
2354
|
-
* @
|
|
2776
|
+
* @param {Array<Array<any>>} resolvedByField One resolved value per distinct symbol, per field.
|
|
2777
|
+
* @param {number} progressBase Rows already delivered before this call, so that progress over a
|
|
2778
|
+
* chunked iteration counts the whole window rather than restarting at every chunk.
|
|
2779
|
+
* @param {number} progressTotal Rows the whole window covers.
|
|
2780
|
+
* @return {Array<Array<any>>} The rows.
|
|
2781
|
+
* @private
|
|
2355
2782
|
*/
|
|
2356
|
-
|
|
2357
|
-
const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
|
|
2358
|
-
const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
|
|
2783
|
+
_buildRows(resolvedByField, progressBase, progressTotal) {
|
|
2359
2784
|
assert2(this._indexColumns, "The QVD file index table has not been parsed.");
|
|
2360
|
-
|
|
2361
|
-
|
|
2362
|
-
|
|
2363
|
-
|
|
2364
|
-
|
|
2365
|
-
|
|
2366
|
-
|
|
2367
|
-
|
|
2785
|
+
const indexColumns = this._indexColumns;
|
|
2786
|
+
const fieldCount = indexColumns.length;
|
|
2787
|
+
const rowCount = this._rowsDecoded;
|
|
2788
|
+
const data = new Array(rowCount);
|
|
2789
|
+
const reportInterval = Math.max(1, Math.floor(progressTotal / 100));
|
|
2790
|
+
for (let row = 0; row < rowCount; row++) {
|
|
2791
|
+
const values = new Array(fieldCount);
|
|
2792
|
+
for (let field = 0; field < fieldCount; field++) {
|
|
2793
|
+
const symbolIndex = indexColumns[field][row];
|
|
2794
|
+
values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
|
|
2795
|
+
}
|
|
2796
|
+
data[row] = values;
|
|
2797
|
+
if ((progressBase + row + 1) % reportInterval === 0 || row + 1 === rowCount) {
|
|
2798
|
+
this._throwIfAborted();
|
|
2799
|
+
this._emitProgress("rows", progressBase + row + 1, progressTotal);
|
|
2800
|
+
}
|
|
2801
|
+
}
|
|
2802
|
+
return data;
|
|
2368
2803
|
}
|
|
2369
2804
|
};
|
|
2370
2805
|
}
|
|
@@ -2375,6 +2810,7 @@ var QvdDataFrame;
|
|
|
2375
2810
|
var init_QvdDataFrame = __esm({
|
|
2376
2811
|
"src/QvdDataFrame.js"() {
|
|
2377
2812
|
init_QvdErrors();
|
|
2813
|
+
init_readOptions();
|
|
2378
2814
|
QvdDataFrame = class _QvdDataFrame {
|
|
2379
2815
|
/**
|
|
2380
2816
|
* Represents the data frame stored inside a QVD file.
|
|
@@ -2417,12 +2853,15 @@ var init_QvdDataFrame = __esm({
|
|
|
2417
2853
|
/**
|
|
2418
2854
|
* Returns statistics about the read that produced this data frame.
|
|
2419
2855
|
*
|
|
2420
|
-
*
|
|
2421
|
-
*
|
|
2856
|
+
* Carried by every frame that came from a file - `fromQvd()`, and each chunk `iterate()` yields,
|
|
2857
|
+
* which is how a chunk reports its `offset`. `fromDict()`, `head()`, `tail()`, `rows()` and
|
|
2858
|
+
* `select()` describe no particular read and report null rather than a stale figure.
|
|
2422
2859
|
*
|
|
2423
2860
|
* The main use is confirming that a lazy load actually filtered the symbol table:
|
|
2424
2861
|
* `symbolFiltering` says whether the two-pass path ran, and `symbolsKept` how many symbols
|
|
2425
|
-
* survived it
|
|
2862
|
+
* survived it. Note that a bounded read does not filter on its own - the two-pass path engages
|
|
2863
|
+
* only above `symbolFilteringThreshold`, so on a file below it this reports false and every
|
|
2864
|
+
* symbol was parsed however few rows were asked for.
|
|
2426
2865
|
*
|
|
2427
2866
|
* @return {QvdLoadStats|null} Load statistics, or null if this frame did not come from a file.
|
|
2428
2867
|
*/
|
|
@@ -2799,6 +3238,18 @@ var init_QvdDataFrame = __esm({
|
|
|
2799
3238
|
* @param {Object} [options] Optional loading options.
|
|
2800
3239
|
* @param {number|null} [options.maxRows] The maximum number of rows to load. Must be a non-negative
|
|
2801
3240
|
* integer; if not specified or null, all rows are loaded. Anything else throws a QvdValidationError.
|
|
3241
|
+
* This is the older name for `limit`; the two are the same option and passing both throws.
|
|
3242
|
+
* @param {number|null} [options.limit] Rows to read, counting from `offset`. The same number as
|
|
3243
|
+
* `maxRows`, spelled so that it reads correctly beside an offset.
|
|
3244
|
+
* @param {number} [options.offset=0] File row to start at. An offset past the end of the file
|
|
3245
|
+
* returns no rows rather than throwing, so a paging loop terminates on its own.
|
|
3246
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
3247
|
+
* appear in the result. Unselected fields have their symbols skipped entirely rather than
|
|
3248
|
+
* parsed and discarded. An unknown or repeated name throws.
|
|
3249
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
|
|
3250
|
+
* read proceeds - the same shape `toQvd`'s callback receives.
|
|
3251
|
+
* @param {AbortSignal} [options.signal] Cancels the read. The rejection is `signal.reason`,
|
|
3252
|
+
* which is a `DOMException` named `AbortError` unless you aborted with a reason of your own.
|
|
2802
3253
|
* @param {string} [options.allowedDir] Optional allowed directory path. If provided, the file path
|
|
2803
3254
|
* must be within this directory, with symlinks resolved first, so a link inside it that points
|
|
2804
3255
|
* outside it is rejected. Defaults to the current working directory. To permit an entire
|
|
@@ -2811,17 +3262,42 @@ var init_QvdDataFrame = __esm({
|
|
|
2811
3262
|
* **Zero disables the memory check entirely.**
|
|
2812
3263
|
* @param {number} [options.symbolFilteringThreshold=52428800] Symbol table size, in bytes, above which
|
|
2813
3264
|
* a lazy load switches to the two-pass filtering path. Defaults to 50MB.
|
|
2814
|
-
* @throws {QvdValidationError} If
|
|
3265
|
+
* @throws {QvdValidationError} If a window option is not a non-negative integer, if both
|
|
3266
|
+
* `maxRows` and `limit` are given, or if `fields` names a column the file does not have.
|
|
2815
3267
|
* @return {Promise<QvdDataFrame>} The data frame of the QVD file.
|
|
2816
3268
|
*/
|
|
2817
3269
|
static async fromQvd(path3, options = {}) {
|
|
2818
3270
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
2819
|
-
|
|
2820
|
-
|
|
2821
|
-
|
|
2822
|
-
|
|
2823
|
-
|
|
2824
|
-
|
|
3271
|
+
return await new QvdFileReader2(path3, readerOptionsFrom(options)).load(windowFrom(options));
|
|
3272
|
+
}
|
|
3273
|
+
/**
|
|
3274
|
+
* Reads a QVD file in chunks, as an async generator of data frames.
|
|
3275
|
+
*
|
|
3276
|
+
* The file is opened, read and parsed once; only the index decode and the row building happen
|
|
3277
|
+
* per chunk, so what this bounds is row materialisation - the part that actually dominates a
|
|
3278
|
+
* large read's heap. It is **not** constant-memory reading of an arbitrarily large file: the
|
|
3279
|
+
* symbol table is parsed in full whatever the chunk size, because a stored index in the last
|
|
3280
|
+
* chunk can address the first symbol. On a high-cardinality file that table is the bulk of the
|
|
3281
|
+
* cost, and `readMetadata` is the only read that avoids it.
|
|
3282
|
+
*
|
|
3283
|
+
* ```js
|
|
3284
|
+
* for await (const chunk of QvdDataFrame.iterate('big.qvd', {chunkSize: 50_000})) {
|
|
3285
|
+
* process(chunk.data);
|
|
3286
|
+
* }
|
|
3287
|
+
* ```
|
|
3288
|
+
*
|
|
3289
|
+
* A window covering no rows yields nothing, so a loop over an exhausted offset simply does not
|
|
3290
|
+
* run its body.
|
|
3291
|
+
*
|
|
3292
|
+
* @param {string} path The path to the QVD file.
|
|
3293
|
+
* @param {Object} [options] The same options `fromQvd` takes, plus:
|
|
3294
|
+
* @param {number} [options.chunkSize=100000] Rows per frame. Must be a positive integer.
|
|
3295
|
+
* @return {AsyncGenerator<QvdDataFrame>} The chunks, in file order.
|
|
3296
|
+
*/
|
|
3297
|
+
static async *iterate(path3, options = {}) {
|
|
3298
|
+
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
3299
|
+
const reader = new QvdFileReader2(path3, readerOptionsFrom(options));
|
|
3300
|
+
yield* reader.iterateRows(windowFrom(options), options.chunkSize === void 0 ? 1e5 : options.chunkSize);
|
|
2825
3301
|
}
|
|
2826
3302
|
/**
|
|
2827
3303
|
* Reads a QVD file's schema and header metadata, without reading its data.
|
|
@@ -2845,11 +3321,15 @@ var init_QvdDataFrame = __esm({
|
|
|
2845
3321
|
* @param {Object} [options] Optional reading options.
|
|
2846
3322
|
* @param {string} [options.allowedDir] Optional allowed directory path, applied exactly as it
|
|
2847
3323
|
* is for `fromQvd`.
|
|
3324
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}`, as on
|
|
3325
|
+
* the reads that return data. Only the `read` and `header` stages occur here; there are no
|
|
3326
|
+
* symbols to parse and no rows to build.
|
|
3327
|
+
* @param {AbortSignal} [options.signal] Cancels the read, rejecting with `signal.reason`.
|
|
2848
3328
|
* @return {Promise<QvdFileMetadata>} The file's schema and header metadata.
|
|
2849
3329
|
*/
|
|
2850
3330
|
static async readMetadata(path3, options = {}) {
|
|
2851
3331
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
2852
|
-
return await new QvdFileReader2(path3,
|
|
3332
|
+
return await new QvdFileReader2(path3, metadataOptionsFrom(options)).loadMetadata();
|
|
2853
3333
|
}
|
|
2854
3334
|
/**
|
|
2855
3335
|
* Constructs a data frame from a dictionary.
|