qvdjs 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +310 -43
- package/dist/index.cjs +723 -212
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +723 -212
- package/dist/index.js.map +1 -1
- package/img/logo/qvdjs_logo-512.png +0 -0
- package/package.json +5 -2
package/dist/index.js
CHANGED
|
@@ -250,6 +250,128 @@ var init_QvdSymbol = __esm({
|
|
|
250
250
|
};
|
|
251
251
|
}
|
|
252
252
|
});
|
|
253
|
+
|
|
254
|
+
// src/util/readOptions.js
|
|
255
|
+
function requireRowCount(value, name, filePath) {
|
|
256
|
+
if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
|
|
257
|
+
throw new QvdValidationError(`${name} must be a non-negative integer`, {
|
|
258
|
+
option: name,
|
|
259
|
+
provided: value,
|
|
260
|
+
type: typeof value,
|
|
261
|
+
file: filePath
|
|
262
|
+
});
|
|
263
|
+
}
|
|
264
|
+
return value;
|
|
265
|
+
}
|
|
266
|
+
function normaliseWindow(window, filePath) {
|
|
267
|
+
if (window === null || window === void 0) {
|
|
268
|
+
return { offset: 0, limit: null };
|
|
269
|
+
}
|
|
270
|
+
if (typeof window === "number") {
|
|
271
|
+
return { offset: 0, limit: requireRowCount(window, "maxRows", filePath) };
|
|
272
|
+
}
|
|
273
|
+
if (typeof window !== "object" || Array.isArray(window)) {
|
|
274
|
+
throw new QvdValidationError("The row window must be a number, null, or an {offset, limit} object", {
|
|
275
|
+
provided: window,
|
|
276
|
+
type: typeof window,
|
|
277
|
+
file: filePath
|
|
278
|
+
});
|
|
279
|
+
}
|
|
280
|
+
const { offset, limit, maxRows } = window;
|
|
281
|
+
const limitGiven = limit !== void 0 && limit !== null;
|
|
282
|
+
const maxRowsGiven = maxRows !== void 0 && maxRows !== null;
|
|
283
|
+
if (limitGiven && maxRowsGiven) {
|
|
284
|
+
throw new QvdValidationError("maxRows and limit are two names for the same option; pass one of them, not both", {
|
|
285
|
+
maxRows,
|
|
286
|
+
limit,
|
|
287
|
+
file: filePath
|
|
288
|
+
});
|
|
289
|
+
}
|
|
290
|
+
return {
|
|
291
|
+
offset: offset === void 0 || offset === null ? 0 : requireRowCount(offset, "offset", filePath),
|
|
292
|
+
limit: limitGiven ? requireRowCount(limit, "limit", filePath) : maxRowsGiven ? requireRowCount(maxRows, "maxRows", filePath) : null
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
function resolveWindow(window, totalRows) {
|
|
296
|
+
const rows = Number.isSafeInteger(totalRows) && totalRows > 0 ? totalRows : 0;
|
|
297
|
+
const offset = Math.min(window.offset, rows);
|
|
298
|
+
return {
|
|
299
|
+
offset,
|
|
300
|
+
limit: Math.max(0, Math.min(window.limit === null ? Infinity : window.limit, rows - offset))
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
function selectFields(fields, requested, filePath) {
|
|
304
|
+
if (requested === null || requested === void 0) {
|
|
305
|
+
return fields;
|
|
306
|
+
}
|
|
307
|
+
if (!Array.isArray(requested)) {
|
|
308
|
+
throw new QvdValidationError("fields must be an array of field names", {
|
|
309
|
+
provided: requested,
|
|
310
|
+
type: typeof requested,
|
|
311
|
+
file: filePath
|
|
312
|
+
});
|
|
313
|
+
}
|
|
314
|
+
const available = fields.map((field) => field["FieldName"]);
|
|
315
|
+
if (requested.length === 0) {
|
|
316
|
+
throw new QvdValidationError("fields must name at least one field", {
|
|
317
|
+
availableColumns: available,
|
|
318
|
+
file: filePath
|
|
319
|
+
});
|
|
320
|
+
}
|
|
321
|
+
const seen = /* @__PURE__ */ new Set();
|
|
322
|
+
return requested.map((name) => {
|
|
323
|
+
if (typeof name !== "string") {
|
|
324
|
+
throw new QvdValidationError("Field names must be strings", {
|
|
325
|
+
provided: name,
|
|
326
|
+
type: typeof name,
|
|
327
|
+
availableColumns: available,
|
|
328
|
+
file: filePath
|
|
329
|
+
});
|
|
330
|
+
}
|
|
331
|
+
if (seen.has(name)) {
|
|
332
|
+
throw new QvdValidationError(`Field '${name}' is listed twice`, {
|
|
333
|
+
column: name,
|
|
334
|
+
fields: requested,
|
|
335
|
+
file: filePath
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
seen.add(name);
|
|
339
|
+
const index = available.indexOf(name);
|
|
340
|
+
if (index === -1) {
|
|
341
|
+
throw new QvdValidationError(`Column '${name}' does not exist`, {
|
|
342
|
+
column: name,
|
|
343
|
+
availableColumns: available,
|
|
344
|
+
file: filePath
|
|
345
|
+
});
|
|
346
|
+
}
|
|
347
|
+
return fields[index];
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
function readerOptionsFrom(options) {
|
|
351
|
+
return {
|
|
352
|
+
allowedDir: options.allowedDir,
|
|
353
|
+
memorySafetyFactor: options.memorySafetyFactor,
|
|
354
|
+
symbolFilteringThreshold: options.symbolFilteringThreshold,
|
|
355
|
+
fields: options.fields === void 0 ? null : options.fields,
|
|
356
|
+
onProgress: options.onProgress,
|
|
357
|
+
signal: options.signal
|
|
358
|
+
};
|
|
359
|
+
}
|
|
360
|
+
function metadataOptionsFrom(options) {
|
|
361
|
+
return {
|
|
362
|
+
allowedDir: options.allowedDir,
|
|
363
|
+
onProgress: options.onProgress,
|
|
364
|
+
signal: options.signal
|
|
365
|
+
};
|
|
366
|
+
}
|
|
367
|
+
function windowFrom(options) {
|
|
368
|
+
return { offset: options.offset, limit: options.limit, maxRows: options.maxRows };
|
|
369
|
+
}
|
|
370
|
+
var init_readOptions = __esm({
|
|
371
|
+
"src/util/readOptions.js"() {
|
|
372
|
+
init_QvdErrors();
|
|
373
|
+
}
|
|
374
|
+
});
|
|
253
375
|
function isWithinDirectoryLexically(resolvedBaseDir, resolvedPath) {
|
|
254
376
|
const isCaseInsensitiveFS = process.platform === "win32";
|
|
255
377
|
const base = isCaseInsensitiveFS ? resolvedBaseDir.toLowerCase() : resolvedBaseDir;
|
|
@@ -807,11 +929,12 @@ function estimateRowMemory(rows, columnCount) {
|
|
|
807
929
|
}
|
|
808
930
|
return BASE_BYTES + rows * (ROW_BASE_BYTES + PER_CELL_BYTES * columnCount);
|
|
809
931
|
}
|
|
810
|
-
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
|
|
932
|
+
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null) {
|
|
811
933
|
const FULL_PARSE_OVERHEAD = 6;
|
|
812
934
|
const MINIMAL_OVERHEAD = 0.01;
|
|
813
935
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
814
|
-
const
|
|
936
|
+
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
937
|
+
const rowMemory = materialisesRows ? estimateRowMemory(liveRows, columnCount) : BASE_BYTES;
|
|
815
938
|
if (maxRows === null || maxRows >= totalRows) {
|
|
816
939
|
return symbolTableSize * FULL_PARSE_OVERHEAD + rowMemory;
|
|
817
940
|
}
|
|
@@ -821,18 +944,44 @@ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount =
|
|
|
821
944
|
const skippedSymbolsMemory = symbolTableSize * (1 - symbolPercentage) * MINIMAL_OVERHEAD;
|
|
822
945
|
return keptSymbolsMemory + skippedSymbolsMemory + rowMemory;
|
|
823
946
|
}
|
|
824
|
-
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true) {
|
|
825
|
-
|
|
947
|
+
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false) {
|
|
948
|
+
const costOf = (rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0);
|
|
949
|
+
if (costOf(totalRows) <= budget) {
|
|
826
950
|
return totalRows;
|
|
827
951
|
}
|
|
828
|
-
if (
|
|
952
|
+
if (costOf(0) > budget) {
|
|
829
953
|
return 0;
|
|
830
954
|
}
|
|
831
955
|
let low = 0;
|
|
832
956
|
let high = totalRows;
|
|
833
957
|
while (high - low > 1) {
|
|
834
958
|
const mid = Math.floor((low + high) / 2);
|
|
835
|
-
if (
|
|
959
|
+
if (costOf(mid) <= budget) {
|
|
960
|
+
low = mid;
|
|
961
|
+
} else {
|
|
962
|
+
high = mid;
|
|
963
|
+
}
|
|
964
|
+
}
|
|
965
|
+
return low;
|
|
966
|
+
}
|
|
967
|
+
function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false) {
|
|
968
|
+
const covered = windowRows === null || windowRows >= totalRows ? totalRows : windowRows;
|
|
969
|
+
const fits = (chunk) => {
|
|
970
|
+
const live = Math.min(chunk * liveRowsPerChunk, covered);
|
|
971
|
+
const cost = estimateMemoryUsage(symbolTableSize, windowRows, totalRows, columnCount, true, chunk * liveRowsPerChunk) + (includeExternal ? estimateExternalMemory(live, columnCount) : 0);
|
|
972
|
+
return cost <= budget;
|
|
973
|
+
};
|
|
974
|
+
if (fits(covered)) {
|
|
975
|
+
return covered;
|
|
976
|
+
}
|
|
977
|
+
if (!fits(1)) {
|
|
978
|
+
return 0;
|
|
979
|
+
}
|
|
980
|
+
let low = 1;
|
|
981
|
+
let high = covered;
|
|
982
|
+
while (high - low > 1) {
|
|
983
|
+
const mid = Math.floor((low + high) / 2);
|
|
984
|
+
if (fits(mid)) {
|
|
836
985
|
low = mid;
|
|
837
986
|
} else {
|
|
838
987
|
high = mid;
|
|
@@ -840,7 +989,7 @@ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, mat
|
|
|
840
989
|
}
|
|
841
990
|
return low;
|
|
842
991
|
}
|
|
843
|
-
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true) {
|
|
992
|
+
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null) {
|
|
844
993
|
if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
|
|
845
994
|
throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", { safetyFactor });
|
|
846
995
|
}
|
|
@@ -849,12 +998,16 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
849
998
|
}
|
|
850
999
|
const budget = getMemoryBudget();
|
|
851
1000
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
852
|
-
const
|
|
853
|
-
const
|
|
1001
|
+
const rowsLive = live === null ? null : live.rows;
|
|
1002
|
+
const liveRowsPerChunk = live === null ? 1 : live.perChunk;
|
|
1003
|
+
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
1004
|
+
const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows, rowsLive);
|
|
1005
|
+
const externalMemory = estimateExternalMemory(liveRows, columnCount);
|
|
854
1006
|
const bounded = budget.candidates.map((candidate) => {
|
|
855
1007
|
const heapOnly = candidate.source === "V8 heap limit";
|
|
856
1008
|
return {
|
|
857
1009
|
...candidate,
|
|
1010
|
+
heapOnly,
|
|
858
1011
|
needs: heapOnly ? heapMemory : heapMemory + externalMemory,
|
|
859
1012
|
allowed: candidate.bytes * safetyFactor,
|
|
860
1013
|
bounds: heapOnly ? "the V8 heap" : "the whole process"
|
|
@@ -870,12 +1023,14 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
870
1023
|
const estimatedMemory = binding ? binding.needs : heapMemory;
|
|
871
1024
|
const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
|
|
872
1025
|
if (binding) {
|
|
1026
|
+
const includeExternal = !binding.heapOnly;
|
|
873
1027
|
const recommendedMaxRows = recommendedRowsFor(
|
|
874
1028
|
maxAllowedMemory,
|
|
875
1029
|
symbolTableSize,
|
|
876
1030
|
totalRows,
|
|
877
1031
|
columnCount,
|
|
878
|
-
materialisesRows
|
|
1032
|
+
materialisesRows,
|
|
1033
|
+
includeExternal
|
|
879
1034
|
);
|
|
880
1035
|
const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
|
|
881
1036
|
const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
|
|
@@ -887,15 +1042,27 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
887
1042
|
const limitingScope = binding.bounds;
|
|
888
1043
|
const budgetBreakdown = budget.candidates.map((candidate) => `${candidate.source} ${Math.round(candidate.bytes / 1024 / 1024)}MB`).join(", ");
|
|
889
1044
|
const observedBreakdown = budget.observed.map((entry) => `${entry.source} ${Math.round(entry.bytes / 1024 / 1024)}MB`).join(", ");
|
|
890
|
-
const nothingFits = recommendedMaxRows === 0;
|
|
891
1045
|
const containerBound = binding.source === "container memory limit";
|
|
1046
|
+
const chunked = rowsLive !== null;
|
|
1047
|
+
const recommendedChunk = chunked ? recommendedChunkFor(
|
|
1048
|
+
maxAllowedMemory,
|
|
1049
|
+
symbolTableSize,
|
|
1050
|
+
maxRows,
|
|
1051
|
+
totalRows,
|
|
1052
|
+
columnCount,
|
|
1053
|
+
liveRowsPerChunk,
|
|
1054
|
+
includeExternal
|
|
1055
|
+
) : 0;
|
|
1056
|
+
const knob = chunked ? "chunkSize" : "limit";
|
|
1057
|
+
const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
|
|
1058
|
+
const nothingFits = recommendedValue === 0;
|
|
892
1059
|
let advice;
|
|
893
1060
|
if (nothingFits) {
|
|
894
|
-
advice = `No row count fits this budget - the symbol table alone exceeds it, so
|
|
1061
|
+
advice = `No row count fits this budget - the symbol table alone exceeds it, so ${knob} cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
|
|
895
1062
|
} else if (containerBound) {
|
|
896
|
-
advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or
|
|
1063
|
+
advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${recommendedValue.toLocaleString()} rows or less).`;
|
|
897
1064
|
} else {
|
|
898
|
-
advice = `Try
|
|
1065
|
+
advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${recommendedValue.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
|
|
899
1066
|
}
|
|
900
1067
|
throw new QvdValidationError(
|
|
901
1068
|
`Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
|
|
@@ -915,22 +1082,30 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
|
|
|
915
1082
|
columnCount,
|
|
916
1083
|
totalRows,
|
|
917
1084
|
maxRows,
|
|
918
|
-
recommendedMaxRows
|
|
1085
|
+
recommendedMaxRows,
|
|
1086
|
+
// Only present when a chunk size is what overflowed, so a caller cannot mistake one
|
|
1087
|
+
// recommendation for the other.
|
|
1088
|
+
...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
|
|
919
1089
|
}
|
|
920
1090
|
);
|
|
921
1091
|
}
|
|
922
1092
|
}
|
|
923
1093
|
function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
|
|
924
1094
|
const LARGE_SYMBOL_TABLE_WARNING = usableOldSpaceLimit() * 0.125;
|
|
925
|
-
if (symbolTableSize
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
);
|
|
1095
|
+
if (symbolTableSize <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
1096
|
+
return;
|
|
1097
|
+
}
|
|
1098
|
+
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
1099
|
+
const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
|
|
1100
|
+
if (estimatedMemory <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
1101
|
+
return;
|
|
933
1102
|
}
|
|
1103
|
+
const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
|
|
1104
|
+
const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
|
|
1105
|
+
const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
|
|
1106
|
+
console.warn(
|
|
1107
|
+
`\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). This read materialises ${rowsToLoad.toLocaleString()} of ${totalRows.toLocaleString()} rows and will use ~${estimatedMB}MB RAM. Reading fewer rows - with limit, maxRows, or a narrower offset window - lowers the row cost, though the symbol table is read in full either way.`
|
|
1108
|
+
);
|
|
934
1109
|
}
|
|
935
1110
|
var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES;
|
|
936
1111
|
var init_memoryUtils = __esm({
|
|
@@ -945,6 +1120,24 @@ var init_memoryUtils = __esm({
|
|
|
945
1120
|
});
|
|
946
1121
|
|
|
947
1122
|
// src/util/validationUtils.js
|
|
1123
|
+
function validateHeaderStructure(headerObj, filePath, stage) {
|
|
1124
|
+
const tableHeader = headerObj?.["QvdTableHeader"];
|
|
1125
|
+
if (tableHeader === null || typeof tableHeader !== "object" || Array.isArray(tableHeader)) {
|
|
1126
|
+
throw new QvdCorruptedError("The XML header contains no usable QvdTableHeader element", {
|
|
1127
|
+
rootElements: headerObj && typeof headerObj === "object" ? Object.keys(headerObj) : [],
|
|
1128
|
+
file: filePath,
|
|
1129
|
+
stage
|
|
1130
|
+
});
|
|
1131
|
+
}
|
|
1132
|
+
const symbolTableLength = parseInt(tableHeader["Offset"], 10);
|
|
1133
|
+
if (isNaN(symbolTableLength) || !Number.isSafeInteger(symbolTableLength) || symbolTableLength < 0) {
|
|
1134
|
+
throw new QvdCorruptedError("Invalid symbol table offset", {
|
|
1135
|
+
offset: tableHeader["Offset"],
|
|
1136
|
+
file: filePath,
|
|
1137
|
+
stage
|
|
1138
|
+
});
|
|
1139
|
+
}
|
|
1140
|
+
}
|
|
948
1141
|
function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
|
|
949
1142
|
const heapLimit = getHeapLimit();
|
|
950
1143
|
const MAX_SYMBOL_TABLE_SIZE = heapLimit * 0.125;
|
|
@@ -953,7 +1146,7 @@ function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
|
|
|
953
1146
|
const maxMB = Math.round(MAX_SYMBOL_TABLE_SIZE / 1024 / 1024);
|
|
954
1147
|
const heapMB = Math.round(heapLimit / 1024 / 1024);
|
|
955
1148
|
throw new QvdValidationError(
|
|
956
|
-
`Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without maxRows, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
|
|
1149
|
+
`Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without a row window - maxRows, limit or offset - since the symbol table is read in full either way, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
|
|
957
1150
|
{
|
|
958
1151
|
file: filePath,
|
|
959
1152
|
symbolTableSize: symbolTableLength,
|
|
@@ -1027,7 +1220,7 @@ function validateRecordCount(totalRows, filePath, stage = "parseIndexTable") {
|
|
|
1027
1220
|
});
|
|
1028
1221
|
}
|
|
1029
1222
|
}
|
|
1030
|
-
function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null) {
|
|
1223
|
+
function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null, windowFirstRow = 0, bufferFirstRow = 0) {
|
|
1031
1224
|
if (isNaN(recordSize) || !Number.isSafeInteger(recordSize) || recordSize < 0) {
|
|
1032
1225
|
throw new QvdCorruptedError("Invalid record byte size", {
|
|
1033
1226
|
recordSize,
|
|
@@ -1089,23 +1282,28 @@ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, ind
|
|
|
1089
1282
|
}
|
|
1090
1283
|
}
|
|
1091
1284
|
const requiredIndexBytes = rowsToLoad * recordSize;
|
|
1092
|
-
|
|
1285
|
+
const bufferRecordStart = (windowFirstRow - bufferFirstRow) * recordSize;
|
|
1286
|
+
if (indexTableOffset + bufferRecordStart + requiredIndexBytes > bufferLength) {
|
|
1093
1287
|
throw new QvdCorruptedError("Index table truncated", {
|
|
1094
1288
|
indexTableOffset,
|
|
1095
1289
|
requiredBytes: requiredIndexBytes,
|
|
1096
|
-
availableBytes: Math.max(0, bufferLength - indexTableOffset),
|
|
1290
|
+
availableBytes: Math.max(0, bufferLength - indexTableOffset - bufferRecordStart),
|
|
1097
1291
|
rowsToLoad,
|
|
1292
|
+
windowFirstRow,
|
|
1293
|
+
bufferFirstRow,
|
|
1098
1294
|
recordSize,
|
|
1099
1295
|
bufferSize: bufferLength,
|
|
1100
1296
|
file: filePath,
|
|
1101
1297
|
stage: "parseIndexTable"
|
|
1102
1298
|
});
|
|
1103
1299
|
}
|
|
1104
|
-
|
|
1300
|
+
const requiredTableBytes = (windowFirstRow + rowsToLoad) * recordSize;
|
|
1301
|
+
if (indexTableLength < requiredTableBytes) {
|
|
1105
1302
|
throw new QvdCorruptedError("Index table length smaller than required", {
|
|
1106
1303
|
indexTableLength,
|
|
1107
|
-
requiredBytes:
|
|
1304
|
+
requiredBytes: requiredTableBytes,
|
|
1108
1305
|
rowsToLoad,
|
|
1306
|
+
windowFirstRow,
|
|
1109
1307
|
recordSize,
|
|
1110
1308
|
file: filePath,
|
|
1111
1309
|
stage: "parseIndexTable"
|
|
@@ -1435,6 +1633,7 @@ var QvdColumn, QvdColumnTable;
|
|
|
1435
1633
|
var init_QvdColumnTable = __esm({
|
|
1436
1634
|
"src/QvdColumnTable.js"() {
|
|
1437
1635
|
init_QvdErrors();
|
|
1636
|
+
init_readOptions();
|
|
1438
1637
|
QvdColumn = class {
|
|
1439
1638
|
/**
|
|
1440
1639
|
* @param {string} name The field name.
|
|
@@ -1629,9 +1828,21 @@ var init_QvdColumnTable = __esm({
|
|
|
1629
1828
|
/**
|
|
1630
1829
|
* Reads a QVD file as columns.
|
|
1631
1830
|
*
|
|
1831
|
+
* Takes the same options as `QvdDataFrame.fromQvd`, with the same meanings - one option
|
|
1832
|
+
* vocabulary for both read paths, because they are two answers about the same file rather than
|
|
1833
|
+
* two features. `{offset, limit}` is how a caller pages through a file columnwise; there is no
|
|
1834
|
+
* columnar `iterate()` because there is nothing for it to bound - a columnar read materialises
|
|
1835
|
+
* no rows, which is the memory chunking exists to cap.
|
|
1836
|
+
*
|
|
1632
1837
|
* @param {string} path The path to the QVD file.
|
|
1633
1838
|
* @param {Object} [options] Loading options, with the same meanings they have on `fromQvd`.
|
|
1634
|
-
* @param {number|null} [options.maxRows]
|
|
1839
|
+
* @param {number|null} [options.maxRows] Rows to decode. The older name for `limit`.
|
|
1840
|
+
* @param {number|null} [options.limit] Rows to decode, counting from `offset`.
|
|
1841
|
+
* @param {number} [options.offset] File row to start at.
|
|
1842
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
1843
|
+
* appear. Unselected fields have their symbols skipped entirely.
|
|
1844
|
+
* @param {Function} [options.onProgress] Progress callback, `{stage, current, total, percent}`.
|
|
1845
|
+
* @param {AbortSignal} [options.signal] Cancels the read.
|
|
1635
1846
|
* @param {string} [options.allowedDir] Directory the path must resolve inside.
|
|
1636
1847
|
* @param {number} [options.memorySafetyFactor] Fraction of the memory budget a load may use.
|
|
1637
1848
|
* @param {number} [options.symbolFilteringThreshold] Symbol table size above which a limited
|
|
@@ -1641,15 +1852,13 @@ var init_QvdColumnTable = __esm({
|
|
|
1641
1852
|
static async fromQvd(path3, options = {}) {
|
|
1642
1853
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
1643
1854
|
const reader = new QvdFileReader2(path3, {
|
|
1644
|
-
|
|
1645
|
-
memorySafetyFactor: options.memorySafetyFactor,
|
|
1646
|
-
symbolFilteringThreshold: options.symbolFilteringThreshold,
|
|
1855
|
+
...readerOptionsFrom(options),
|
|
1647
1856
|
// This read builds no rows, so the memory guard must not charge it for them. A columnar
|
|
1648
1857
|
// read of the 38MB taxi fixture completes in a 15MB heap; charged the row cost it was
|
|
1649
1858
|
// refused below a 2GB one.
|
|
1650
1859
|
materialisesRows: false
|
|
1651
1860
|
});
|
|
1652
|
-
return await reader.loadColumnar(options
|
|
1861
|
+
return await reader.loadColumnar(windowFrom(options));
|
|
1653
1862
|
}
|
|
1654
1863
|
/** @return {Array<string>} Field names, in file order. */
|
|
1655
1864
|
get columns() {
|
|
@@ -1697,7 +1906,7 @@ var QvdFileReader_exports = {};
|
|
|
1697
1906
|
__export(QvdFileReader_exports, {
|
|
1698
1907
|
QvdFileReader: () => QvdFileReader
|
|
1699
1908
|
});
|
|
1700
|
-
var MAX_HEADER_SIZE, READ_CHUNK_SIZE, QvdFileReader;
|
|
1909
|
+
var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, QvdFileReader;
|
|
1701
1910
|
var init_QvdFileReader = __esm({
|
|
1702
1911
|
"src/QvdFileReader.js"() {
|
|
1703
1912
|
init_QvdDataFrame();
|
|
@@ -1707,8 +1916,10 @@ var init_QvdFileReader = __esm({
|
|
|
1707
1916
|
init_memoryUtils();
|
|
1708
1917
|
init_validationUtils();
|
|
1709
1918
|
init_symbolParser();
|
|
1919
|
+
init_readOptions();
|
|
1710
1920
|
MAX_HEADER_SIZE = 16 * 1024 * 1024;
|
|
1711
1921
|
READ_CHUNK_SIZE = 512 * 1024 * 1024;
|
|
1922
|
+
ANALYSIS_SLICE_ROWS = 65536;
|
|
1712
1923
|
QvdFileReader = class {
|
|
1713
1924
|
/**
|
|
1714
1925
|
* Constructs a new QVD file parser.
|
|
@@ -1720,9 +1931,10 @@ var init_QvdFileReader = __esm({
|
|
|
1720
1931
|
* points outside it is rejected. Defaults to the current working directory. To permit
|
|
1721
1932
|
* an entire volume, pass its root explicitly ('/' on POSIX, 'C:\\' on Windows); a null or
|
|
1722
1933
|
* empty value falls back to the working directory rather than removing the restriction.
|
|
1723
|
-
* @param {number} [options.memorySafetyFactor=0.
|
|
1724
|
-
* load may use. The budget is the
|
|
1725
|
-
*
|
|
1934
|
+
* @param {number} [options.memorySafetyFactor=0.8] Fraction (0.0-1.0) of the memory budget a
|
|
1935
|
+
* load may use. The budget is the smaller of the V8 heap limit and any container memory limit;
|
|
1936
|
+
* what the OS reports as available is recorded for diagnostics and deliberately not allowed to
|
|
1937
|
+
* bind - see `getMemoryBudget`. Default is 0.8. **Zero disables the memory
|
|
1726
1938
|
* check entirely**, which is the escape hatch for runtimes whose limits cannot be measured -
|
|
1727
1939
|
* Bun reports its current heap as its heap limit - and for callers who would rather manage
|
|
1728
1940
|
* memory themselves than trust the estimate.
|
|
@@ -1733,29 +1945,93 @@ var init_QvdFileReader = __esm({
|
|
|
1733
1945
|
* above which a lazy load switches to the two-pass filtering path. The default of 50MB is
|
|
1734
1946
|
* the point where the extra analysis pass pays for itself; lower it to use filtering on
|
|
1735
1947
|
* smaller files, raise it to keep the simpler single-pass read for longer.
|
|
1948
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
1949
|
+
* appear. Null reads every field, in file order. An unknown or repeated name is refused.
|
|
1950
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
|
|
1951
|
+
* read proceeds - the same shape `QvdFileWriter` emits.
|
|
1952
|
+
* @param {AbortSignal} [options.signal] Cancels the read. When it is aborted the read throws
|
|
1953
|
+
* `signal.reason`, exactly as `signal.throwIfAborted()` does.
|
|
1736
1954
|
*/
|
|
1737
1955
|
constructor(filePath, options = {}) {
|
|
1738
1956
|
const {
|
|
1739
1957
|
allowedDir,
|
|
1740
1958
|
memorySafetyFactor = 0.8,
|
|
1741
1959
|
symbolFilteringThreshold = 50 * 1024 * 1024,
|
|
1742
|
-
materialisesRows = true
|
|
1960
|
+
materialisesRows = true,
|
|
1961
|
+
fields = null,
|
|
1962
|
+
onProgress,
|
|
1963
|
+
signal
|
|
1743
1964
|
} = options;
|
|
1744
1965
|
this._materialisesRows = materialisesRows;
|
|
1745
1966
|
this._path = validatePath(filePath, allowedDir);
|
|
1746
1967
|
this._memorySafetyFactor = memorySafetyFactor;
|
|
1747
1968
|
this._symbolFilteringThreshold = symbolFilteringThreshold;
|
|
1969
|
+
if (onProgress !== void 0 && typeof onProgress !== "function") {
|
|
1970
|
+
throw new QvdValidationError("onProgress must be a function", {
|
|
1971
|
+
provided: onProgress,
|
|
1972
|
+
type: typeof onProgress,
|
|
1973
|
+
file: this._path
|
|
1974
|
+
});
|
|
1975
|
+
}
|
|
1976
|
+
if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
|
|
1977
|
+
throw new QvdValidationError("signal must be an AbortSignal", {
|
|
1978
|
+
provided: signal,
|
|
1979
|
+
type: typeof signal,
|
|
1980
|
+
file: this._path
|
|
1981
|
+
});
|
|
1982
|
+
}
|
|
1983
|
+
this._requestedFields = fields === void 0 ? null : fields;
|
|
1984
|
+
this._onProgress = onProgress;
|
|
1985
|
+
this._signal = signal;
|
|
1748
1986
|
this._buffer = null;
|
|
1749
1987
|
this._headerOffset = null;
|
|
1750
1988
|
this._symbolTableOffset = null;
|
|
1751
1989
|
this._indexTableOffset = null;
|
|
1752
1990
|
this._header = null;
|
|
1991
|
+
this._allFields = null;
|
|
1992
|
+
this._selectedFields = null;
|
|
1993
|
+
this._fieldBitMetadataValidated = false;
|
|
1753
1994
|
this._symbolTable = null;
|
|
1754
1995
|
this._indexColumns = null;
|
|
1755
1996
|
this._rowsDecoded = 0;
|
|
1997
|
+
this._bufferFirstRow = 0;
|
|
1756
1998
|
this._fileSize = null;
|
|
1757
1999
|
this._headerMatchesFile = false;
|
|
1758
2000
|
}
|
|
2001
|
+
/**
|
|
2002
|
+
* Emits a progress event if a callback is registered.
|
|
2003
|
+
*
|
|
2004
|
+
* The same shape `QvdFileWriter._emitProgress` emits, deliberately: a caller who has written a
|
|
2005
|
+
* progress bar for a write should not have to write a second one for a read. The stage names
|
|
2006
|
+
* differ because the stages differ, but `symbol-table` and `index-table` mean the same thing on
|
|
2007
|
+
* both sides.
|
|
2008
|
+
*
|
|
2009
|
+
* @param {string} stage The current stage of the read.
|
|
2010
|
+
* @param {number} current The current progress value.
|
|
2011
|
+
* @param {number} total The total progress value.
|
|
2012
|
+
* @private
|
|
2013
|
+
*/
|
|
2014
|
+
_emitProgress(stage, current, total) {
|
|
2015
|
+
if (this._onProgress) {
|
|
2016
|
+
const percent = total > 0 ? Math.round(current / total * 100) : 100;
|
|
2017
|
+
this._onProgress({ stage, current, total, percent });
|
|
2018
|
+
}
|
|
2019
|
+
}
|
|
2020
|
+
/**
|
|
2021
|
+
* Throws if the caller has cancelled the read.
|
|
2022
|
+
*
|
|
2023
|
+
* Throws `signal.reason` - a `DOMException` named `AbortError` unless the caller aborted with a
|
|
2024
|
+
* reason of their own. That is what `AbortSignal` means everywhere else in Node, and inventing
|
|
2025
|
+
* a `QvdAbortError` here would make this library's cancellation the one a caller has to special
|
|
2026
|
+
* case.
|
|
2027
|
+
*
|
|
2028
|
+
* @private
|
|
2029
|
+
*/
|
|
2030
|
+
_throwIfAborted() {
|
|
2031
|
+
if (this._signal) {
|
|
2032
|
+
this._signal.throwIfAborted();
|
|
2033
|
+
}
|
|
2034
|
+
}
|
|
1759
2035
|
/**
|
|
1760
2036
|
* Reads the binary data of the QVD file.
|
|
1761
2037
|
*
|
|
@@ -1780,14 +2056,23 @@ var init_QvdFileReader = __esm({
|
|
|
1780
2056
|
* - Streaming for header finding is efficient for unknown header sizes
|
|
1781
2057
|
* - Direct byte-range reading for remaining data is fastest
|
|
1782
2058
|
*
|
|
1783
|
-
*
|
|
2059
|
+
* A window with a non-zero `offset` reads two ranges rather than one: the header and symbol
|
|
2060
|
+
* table from the front of the file, and the window's records from wherever they sit. The bytes
|
|
2061
|
+
* between are never read, which is what makes `{offset: 1_700_000, limit: 100}` on the taxi
|
|
2062
|
+
* fixture a 0.4MB read rather than a 38MB one.
|
|
2063
|
+
*
|
|
2064
|
+
* @param {QvdRowWindow} window The rows to read.
|
|
1784
2065
|
* @param {boolean} [headerOnly=false] Stop once the XML header has been read, leaving the
|
|
1785
2066
|
* symbol and index tables on disk. This is the metadata-only path: the header is a few
|
|
1786
2067
|
* kilobytes whatever the file's size, so reading a schema costs the same for a 40MB file as
|
|
1787
2068
|
* for a 40GB one.
|
|
2069
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
|
|
2070
|
+
* that is fewer than the window covers - see `_prepare`.
|
|
1788
2071
|
* @private
|
|
1789
2072
|
*/
|
|
1790
|
-
async _readData(
|
|
2073
|
+
async _readData(window = { offset: 0, limit: null }, headerOnly = false, liveRows = null) {
|
|
2074
|
+
this._throwIfAborted();
|
|
2075
|
+
this._emitProgress("read", 0, 1);
|
|
1791
2076
|
const HEADER_DELIMITER = "\r\n\0";
|
|
1792
2077
|
const CHUNK_SIZE = 64 * 1024;
|
|
1793
2078
|
const stream = fs.createReadStream(this._path, {
|
|
@@ -1849,6 +2134,7 @@ var init_QvdFileReader = __esm({
|
|
|
1849
2134
|
stage: "readData"
|
|
1850
2135
|
});
|
|
1851
2136
|
}
|
|
2137
|
+
validateHeaderStructure(headerObj, this._path, "readData");
|
|
1852
2138
|
const symbolTableOffset = headerEndIndex;
|
|
1853
2139
|
const symbolTableLength = parseInt(headerObj["QvdTableHeader"]["Offset"], 10);
|
|
1854
2140
|
const indexTableOffset = symbolTableOffset + symbolTableLength;
|
|
@@ -1856,13 +2142,14 @@ var init_QvdFileReader = __esm({
|
|
|
1856
2142
|
const totalRows = parseInt(headerObj["QvdTableHeader"]["NoOfRecords"], 10);
|
|
1857
2143
|
if (headerOnly) {
|
|
1858
2144
|
this._buffer = headerBuffer.subarray(0, headerEndIndex);
|
|
2145
|
+
this._emitProgress("read", 1, 1);
|
|
1859
2146
|
return;
|
|
1860
2147
|
}
|
|
1861
2148
|
let headerFields = headerObj["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
|
|
1862
2149
|
if (headerFields && !Array.isArray(headerFields)) {
|
|
1863
2150
|
headerFields = [headerFields];
|
|
1864
2151
|
}
|
|
1865
|
-
const columnCount = Array.isArray(headerFields) ? headerFields.length : 0;
|
|
2152
|
+
const columnCount = Array.isArray(headerFields) ? selectFields(headerFields, this._requestedFields, this._path).length : 0;
|
|
1866
2153
|
const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
|
|
1867
2154
|
(value) => Number.isSafeInteger(value) && value >= 0
|
|
1868
2155
|
);
|
|
@@ -1871,23 +2158,28 @@ var init_QvdFileReader = __esm({
|
|
|
1871
2158
|
this._fileSize = fileSize;
|
|
1872
2159
|
this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize;
|
|
1873
2160
|
}
|
|
2161
|
+
const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
|
|
2162
|
+
const windowRows = resolved.limit;
|
|
1874
2163
|
if (headerNumbersUsable && this._headerMatchesFile) {
|
|
1875
2164
|
validateMemoryAvailability(
|
|
1876
2165
|
symbolTableLength,
|
|
1877
|
-
|
|
2166
|
+
windowRows,
|
|
1878
2167
|
totalRows,
|
|
1879
2168
|
this._path,
|
|
1880
2169
|
this._memorySafetyFactor,
|
|
1881
2170
|
columnCount,
|
|
1882
|
-
this._materialisesRows
|
|
2171
|
+
this._materialisesRows,
|
|
2172
|
+
liveRows
|
|
1883
2173
|
);
|
|
1884
2174
|
}
|
|
1885
|
-
if (
|
|
2175
|
+
if (window.offset === 0 && window.limit === null) {
|
|
1886
2176
|
this._buffer = await fs.promises.readFile(this._path);
|
|
1887
2177
|
this._fileSize = this._buffer.length;
|
|
2178
|
+
this._bufferFirstRow = 0;
|
|
2179
|
+
this._emitProgress("read", 1, 1);
|
|
1888
2180
|
return;
|
|
1889
2181
|
}
|
|
1890
|
-
const rowsToLoad =
|
|
2182
|
+
const rowsToLoad = windowRows;
|
|
1891
2183
|
validateSymbolTableSizeEarly(symbolTableLength, this._path);
|
|
1892
2184
|
for (const [name, value] of [
|
|
1893
2185
|
["Offset", symbolTableLength],
|
|
@@ -1903,39 +2195,78 @@ var init_QvdFileReader = __esm({
|
|
|
1903
2195
|
});
|
|
1904
2196
|
}
|
|
1905
2197
|
}
|
|
2198
|
+
const skippedIndexBytes = resolved.offset * recordSize;
|
|
1906
2199
|
const indexTableBytesToRead = rowsToLoad * recordSize;
|
|
1907
2200
|
const totalBytesToRead = indexTableOffset + indexTableBytesToRead;
|
|
2201
|
+
const fileBytesRequired = indexTableOffset + skippedIndexBytes + indexTableBytesToRead;
|
|
1908
2202
|
const fd = await fs.promises.open(this._path, "r");
|
|
1909
2203
|
try {
|
|
1910
2204
|
const { size: fileSize } = await fd.stat();
|
|
1911
2205
|
this._fileSize = fileSize;
|
|
1912
|
-
if (
|
|
2206
|
+
if (fileBytesRequired > fileSize) {
|
|
1913
2207
|
throw new QvdCorruptedError("The file is shorter than its header claims.", {
|
|
1914
2208
|
file: this._path,
|
|
1915
2209
|
fileSize,
|
|
1916
|
-
requiredBytes:
|
|
2210
|
+
requiredBytes: fileBytesRequired,
|
|
1917
2211
|
stage: "readData"
|
|
1918
2212
|
});
|
|
1919
2213
|
}
|
|
1920
2214
|
this._buffer = Buffer.alloc(totalBytesToRead);
|
|
1921
|
-
|
|
1922
|
-
|
|
1923
|
-
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
1927
|
-
|
|
1928
|
-
|
|
1929
|
-
|
|
1930
|
-
|
|
1931
|
-
stage: "readData"
|
|
1932
|
-
});
|
|
1933
|
-
}
|
|
1934
|
-
position += bytesRead;
|
|
2215
|
+
await this._readRange(fd, 0, indexTableOffset, 0, fileSize, totalBytesToRead);
|
|
2216
|
+
if (indexTableBytesToRead > 0) {
|
|
2217
|
+
await this._readRange(
|
|
2218
|
+
fd,
|
|
2219
|
+
indexTableOffset,
|
|
2220
|
+
indexTableBytesToRead,
|
|
2221
|
+
indexTableOffset + skippedIndexBytes,
|
|
2222
|
+
fileSize,
|
|
2223
|
+
fileBytesRequired
|
|
2224
|
+
);
|
|
1935
2225
|
}
|
|
2226
|
+
this._bufferFirstRow = resolved.offset;
|
|
1936
2227
|
} finally {
|
|
1937
2228
|
await fd.close();
|
|
1938
2229
|
}
|
|
2230
|
+
this._emitProgress("read", 1, 1);
|
|
2231
|
+
}
|
|
2232
|
+
/**
|
|
2233
|
+
* Reads one byte range of the file into the buffer.
|
|
2234
|
+
*
|
|
2235
|
+
* Read in bounded chunks, checking bytesRead each time. A single fs.read call with a length of
|
|
2236
|
+
* 2^31 or more does not throw - it trips a C++ assertion and aborts the whole process, which no
|
|
2237
|
+
* try/catch can intercept.
|
|
2238
|
+
*
|
|
2239
|
+
* @param {import('fs/promises').FileHandle} fd The open file.
|
|
2240
|
+
* @param {number} bufferOffset Where in the buffer to write.
|
|
2241
|
+
* @param {number} byteCount How many bytes to read.
|
|
2242
|
+
* @param {number} filePosition Where in the file to read from.
|
|
2243
|
+
* @param {number} fileSize The file's size, for the error.
|
|
2244
|
+
* @param {number} requiredBytes Bytes the whole read needs, for the error.
|
|
2245
|
+
* @private
|
|
2246
|
+
*/
|
|
2247
|
+
async _readRange(fd, bufferOffset, byteCount, filePosition, fileSize, requiredBytes) {
|
|
2248
|
+
assert2(this._buffer, "The read buffer has not been allocated.");
|
|
2249
|
+
let done = 0;
|
|
2250
|
+
while (done < byteCount) {
|
|
2251
|
+
const length = Math.min(READ_CHUNK_SIZE, byteCount - done);
|
|
2252
|
+
const { bytesRead } = await fd.read(this._buffer, bufferOffset + done, length, filePosition + done);
|
|
2253
|
+
if (bytesRead === 0) {
|
|
2254
|
+
throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
|
|
2255
|
+
file: this._path,
|
|
2256
|
+
fileSize,
|
|
2257
|
+
// Two numbers, because they stopped being the same one when a window began reading two
|
|
2258
|
+
// ranges: `bytesRead` is how much of this range arrived, `filePosition` is where in the
|
|
2259
|
+
// file it gave up. Reporting the position under the name of the count made a windowed
|
|
2260
|
+
// read of a truncated file claim tens of megabytes had been read when a few hundred
|
|
2261
|
+
// bytes had.
|
|
2262
|
+
bytesRead: done,
|
|
2263
|
+
filePosition: filePosition + done,
|
|
2264
|
+
requiredBytes,
|
|
2265
|
+
stage: "readData"
|
|
2266
|
+
});
|
|
2267
|
+
}
|
|
2268
|
+
done += bytesRead;
|
|
2269
|
+
}
|
|
1939
2270
|
}
|
|
1940
2271
|
/**
|
|
1941
2272
|
* Parses the XML header of the QVD file. This method is part of the parsing process
|
|
@@ -1972,17 +2303,31 @@ var init_QvdFileReader = __esm({
|
|
|
1972
2303
|
stage: "parseHeader"
|
|
1973
2304
|
});
|
|
1974
2305
|
}
|
|
2306
|
+
validateHeaderStructure(this._header, this._path, "parseHeader");
|
|
1975
2307
|
const fields = this._header["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
|
|
1976
|
-
const
|
|
1977
|
-
if (
|
|
2308
|
+
const fieldList = fields === void 0 || fields === null ? [] : Array.isArray(fields) ? fields : [fields];
|
|
2309
|
+
if (fieldList.length === 0) {
|
|
1978
2310
|
throw new QvdCorruptedError("The QVD file header declares no fields", {
|
|
1979
2311
|
file: this._path,
|
|
1980
2312
|
stage: "parseHeader"
|
|
1981
2313
|
});
|
|
1982
2314
|
}
|
|
2315
|
+
const malformedIndex = fieldList.findIndex(
|
|
2316
|
+
(field) => field === null || typeof field !== "object" || Array.isArray(field)
|
|
2317
|
+
);
|
|
2318
|
+
if (malformedIndex !== -1) {
|
|
2319
|
+
throw new QvdCorruptedError("The QVD file header declares a field with no properties", {
|
|
2320
|
+
fieldIndex: malformedIndex,
|
|
2321
|
+
fieldCount: fieldList.length,
|
|
2322
|
+
file: this._path,
|
|
2323
|
+
stage: "parseHeader"
|
|
2324
|
+
});
|
|
2325
|
+
}
|
|
1983
2326
|
this._headerOffset = headerBeginIndex;
|
|
1984
2327
|
this._symbolTableOffset = headerEndIndex;
|
|
1985
2328
|
this._indexTableOffset = this._symbolTableOffset + parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2329
|
+
this._allFields = fieldList;
|
|
2330
|
+
this._selectedFields = selectFields(this._allFields, this._requestedFields, this._path);
|
|
1986
2331
|
}
|
|
1987
2332
|
/**
|
|
1988
2333
|
* Establishes the geometry of the index table, and validates it.
|
|
@@ -1993,14 +2338,15 @@ var init_QvdFileReader = __esm({
|
|
|
1993
2338
|
* about keeping the sign in step with the other one: the two could drift, and #113 is what
|
|
1994
2339
|
* that looks like when they do. There is one copy now.
|
|
1995
2340
|
*
|
|
1996
|
-
* @param {
|
|
2341
|
+
* @param {QvdRowWindow} window The rows of interest, as file row indices.
|
|
1997
2342
|
* @param {string} stage Stage name for any error raised here.
|
|
1998
2343
|
* @return {{fields: Array<any>, recordSize: number, totalRows: number, rowsToLoad: number,
|
|
1999
|
-
* indexBuffer: Buffer}} The record geometry.
|
|
2344
|
+
* indexBuffer: Buffer}} The record geometry. `indexBuffer` starts at the window's first
|
|
2345
|
+
* record, so the decoder always counts from zero.
|
|
2000
2346
|
* @private
|
|
2001
2347
|
*/
|
|
2002
|
-
_planIndexTable(
|
|
2003
|
-
if (!this._buffer || !this._header || !this._indexTableOffset) {
|
|
2348
|
+
_planIndexTable(window, stage) {
|
|
2349
|
+
if (!this._buffer || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
|
|
2004
2350
|
throw new QvdCorruptedError(
|
|
2005
2351
|
"The QVD file has not been loaded in the proper order or has not been loaded at all.",
|
|
2006
2352
|
{
|
|
@@ -2009,14 +2355,12 @@ var init_QvdFileReader = __esm({
|
|
|
2009
2355
|
}
|
|
2010
2356
|
);
|
|
2011
2357
|
}
|
|
2012
|
-
|
|
2013
|
-
|
|
2014
|
-
fields = [fields];
|
|
2015
|
-
}
|
|
2358
|
+
const allFields = this._allFields;
|
|
2359
|
+
const fields = this._selectedFields;
|
|
2016
2360
|
const recordSize = parseInt(this._header["QvdTableHeader"]["RecordByteSize"], 10);
|
|
2017
2361
|
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
2018
|
-
const rowsToLoad = rowLimit !== null ? Math.min(rowLimit, totalRows) : totalRows;
|
|
2019
2362
|
const indexTableLength = parseInt(this._header["QvdTableHeader"]["Length"], 10);
|
|
2363
|
+
const { offset: firstRow, limit: rowsToLoad } = resolveWindow(window, totalRows);
|
|
2020
2364
|
validateIndexTableMetadata(
|
|
2021
2365
|
recordSize,
|
|
2022
2366
|
totalRows,
|
|
@@ -2025,11 +2369,20 @@ var init_QvdFileReader = __esm({
|
|
|
2025
2369
|
this._buffer.length,
|
|
2026
2370
|
rowsToLoad,
|
|
2027
2371
|
this._path,
|
|
2028
|
-
this._fileSize
|
|
2372
|
+
this._fileSize,
|
|
2373
|
+
firstRow,
|
|
2374
|
+
this._bufferFirstRow
|
|
2029
2375
|
);
|
|
2030
|
-
const
|
|
2031
|
-
|
|
2032
|
-
|
|
2376
|
+
const bufferRecordStart = (firstRow - this._bufferFirstRow) * recordSize;
|
|
2377
|
+
const indexBuffer = this._buffer.subarray(
|
|
2378
|
+
this._indexTableOffset + bufferRecordStart,
|
|
2379
|
+
this._indexTableOffset + bufferRecordStart + rowsToLoad * recordSize
|
|
2380
|
+
);
|
|
2381
|
+
if (!this._fieldBitMetadataValidated) {
|
|
2382
|
+
for (const field of allFields) {
|
|
2383
|
+
validateFieldBitMetadata(field, recordSize, this._path);
|
|
2384
|
+
}
|
|
2385
|
+
this._fieldBitMetadataValidated = true;
|
|
2033
2386
|
}
|
|
2034
2387
|
assert2(
|
|
2035
2388
|
rowsToLoad === 0 || recordSize === 0 || Math.floor(indexBuffer.length / recordSize) >= rowsToLoad,
|
|
@@ -2041,31 +2394,45 @@ var init_QvdFileReader = __esm({
|
|
|
2041
2394
|
* Analyzes the index table to determine which symbols are actually needed.
|
|
2042
2395
|
* This is used for two-pass symbol filtering optimization.
|
|
2043
2396
|
*
|
|
2044
|
-
*
|
|
2045
|
-
*
|
|
2397
|
+
* Only the selected fields are analysed. An unselected field's symbols are never parsed, so
|
|
2398
|
+
* there is nothing for a usage set to filter and decoding its column would be a pass over the
|
|
2399
|
+
* whole window for an answer nobody reads.
|
|
2400
|
+
*
|
|
2401
|
+
* @param {QvdRowWindow} window The rows to analyse.
|
|
2402
|
+
* @return {Promise<Array<Set<number>>>} One set of needed symbol indices per selected field, in
|
|
2403
|
+
* the same order `_parseSymbolTable` walks them.
|
|
2046
2404
|
* @private
|
|
2047
2405
|
*/
|
|
2048
|
-
async _analyzeIndexTableSymbolUsage(
|
|
2049
|
-
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(
|
|
2050
|
-
const symbolUsage =
|
|
2051
|
-
const
|
|
2052
|
-
|
|
2406
|
+
async _analyzeIndexTableSymbolUsage(window) {
|
|
2407
|
+
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
|
|
2408
|
+
const symbolUsage = [];
|
|
2409
|
+
const sliceRows = Math.min(rowsToLoad, ANALYSIS_SLICE_ROWS);
|
|
2410
|
+
const column = new Int32Array(sliceRows);
|
|
2411
|
+
fields.forEach((field, position) => {
|
|
2412
|
+
this._throwIfAborted();
|
|
2053
2413
|
const needed = /* @__PURE__ */ new Set();
|
|
2054
|
-
symbolUsage
|
|
2055
|
-
|
|
2056
|
-
|
|
2057
|
-
|
|
2058
|
-
|
|
2059
|
-
|
|
2060
|
-
|
|
2061
|
-
|
|
2062
|
-
|
|
2063
|
-
|
|
2064
|
-
|
|
2065
|
-
|
|
2066
|
-
|
|
2414
|
+
symbolUsage[position] = needed;
|
|
2415
|
+
const bitOffset = parseInt(field["BitOffset"], 10);
|
|
2416
|
+
const bitWidth = parseInt(field["BitWidth"], 10);
|
|
2417
|
+
const bias = parseInt(field["Bias"], 10);
|
|
2418
|
+
for (let first = 0; first < rowsToLoad; first += sliceRows) {
|
|
2419
|
+
const count = Math.min(sliceRows, rowsToLoad - first);
|
|
2420
|
+
decodeIndexColumn(
|
|
2421
|
+
first === 0 ? indexBuffer : indexBuffer.subarray(first * recordSize),
|
|
2422
|
+
recordSize,
|
|
2423
|
+
count,
|
|
2424
|
+
bitOffset,
|
|
2425
|
+
bitWidth,
|
|
2426
|
+
bias,
|
|
2427
|
+
column
|
|
2428
|
+
);
|
|
2429
|
+
for (let row = 0; row < count; row++) {
|
|
2430
|
+
if (column[row] >= 0) {
|
|
2431
|
+
needed.add(column[row]);
|
|
2432
|
+
}
|
|
2067
2433
|
}
|
|
2068
2434
|
}
|
|
2435
|
+
this._emitProgress("symbol-analysis", position + 1, fields.length);
|
|
2069
2436
|
});
|
|
2070
2437
|
return symbolUsage;
|
|
2071
2438
|
}
|
|
@@ -2073,12 +2440,20 @@ var init_QvdFileReader = __esm({
|
|
|
2073
2440
|
* Parses the symbol table of the QVD file. This method is part of the parsing process
|
|
2074
2441
|
* and should not be called directly.
|
|
2075
2442
|
*
|
|
2076
|
-
*
|
|
2077
|
-
*
|
|
2078
|
-
*
|
|
2443
|
+
* A field the caller did not select is skipped whole. Its symbol area is neither scanned nor
|
|
2444
|
+
* parsed - the per-field `Offset` and `Length` say exactly where it is, so there is nothing to
|
|
2445
|
+
* walk past - and that is where field selection earns its keep. The index decode is cheap by
|
|
2446
|
+
* comparison; parsing symbols is not.
|
|
2447
|
+
*
|
|
2448
|
+
* @param {Array<Set<number>>|null} symbolsToKeep Optional set of symbol indices to keep per
|
|
2449
|
+
* selected field, indexed by position. If provided, only these symbols will be parsed
|
|
2450
|
+
* (two-pass filtering optimization).
|
|
2451
|
+
* @param {number} rowsToLoad Rows the read covers, for memory estimation.
|
|
2452
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
|
|
2453
|
+
* that is fewer than the window covers - see `_prepare`.
|
|
2079
2454
|
*/
|
|
2080
|
-
async _parseSymbolTable(symbolsToKeep = null,
|
|
2081
|
-
if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset) {
|
|
2455
|
+
async _parseSymbolTable(symbolsToKeep = null, rowsToLoad = 0, liveRows = null) {
|
|
2456
|
+
if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
|
|
2082
2457
|
throw new QvdCorruptedError(
|
|
2083
2458
|
"The QVD file has not been loaded in the proper order or has not been loaded at all.",
|
|
2084
2459
|
{
|
|
@@ -2087,7 +2462,8 @@ var init_QvdFileReader = __esm({
|
|
|
2087
2462
|
}
|
|
2088
2463
|
);
|
|
2089
2464
|
}
|
|
2090
|
-
|
|
2465
|
+
const allFields = this._allFields;
|
|
2466
|
+
const fields = this._selectedFields;
|
|
2091
2467
|
const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
|
|
2092
2468
|
const symbolTableSize = symbolBuffer.length;
|
|
2093
2469
|
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
@@ -2095,32 +2471,25 @@ var init_QvdFileReader = __esm({
|
|
|
2095
2471
|
if (this._headerMatchesFile) {
|
|
2096
2472
|
validateMemoryAvailability(
|
|
2097
2473
|
symbolTableSize,
|
|
2098
|
-
|
|
2474
|
+
rowsToLoad,
|
|
2099
2475
|
totalRows,
|
|
2100
2476
|
this._path,
|
|
2101
2477
|
this._memorySafetyFactor,
|
|
2102
|
-
|
|
2103
|
-
this._materialisesRows
|
|
2478
|
+
fields.length,
|
|
2479
|
+
this._materialisesRows,
|
|
2480
|
+
liveRows
|
|
2104
2481
|
);
|
|
2105
2482
|
}
|
|
2106
|
-
warnLargeSymbolTable(
|
|
2107
|
-
|
|
2108
|
-
maxRows,
|
|
2109
|
-
totalRows,
|
|
2110
|
-
Array.isArray(fields) ? fields.length : 1,
|
|
2111
|
-
this._materialisesRows
|
|
2112
|
-
);
|
|
2113
|
-
if (!Array.isArray(fields)) {
|
|
2114
|
-
fields = [fields];
|
|
2115
|
-
}
|
|
2116
|
-
for (const field of fields) {
|
|
2483
|
+
warnLargeSymbolTable(symbolTableSize, rowsToLoad, totalRows, fields.length, this._materialisesRows);
|
|
2484
|
+
for (const field of allFields) {
|
|
2117
2485
|
validateFieldMetadata(field, symbolBuffer.length, this._path);
|
|
2118
2486
|
}
|
|
2119
|
-
this._symbolTable = fields.map((field) => {
|
|
2487
|
+
this._symbolTable = fields.map((field, position) => {
|
|
2488
|
+
this._throwIfAborted();
|
|
2120
2489
|
const symbolsOffset = parseInt(field["Offset"], 10);
|
|
2121
2490
|
const symbolsLength = parseInt(field["Length"], 10);
|
|
2122
2491
|
const fieldName = field["FieldName"];
|
|
2123
|
-
const neededSymbols = symbolsToKeep ? symbolsToKeep
|
|
2492
|
+
const neededSymbols = symbolsToKeep ? symbolsToKeep[position] : null;
|
|
2124
2493
|
const filteringEnabled = neededSymbols !== null;
|
|
2125
2494
|
const symbols = [];
|
|
2126
2495
|
let symbolIndex = 0;
|
|
@@ -2140,6 +2509,7 @@ var init_QvdFileReader = __esm({
|
|
|
2140
2509
|
pointer += bytesRead - 1;
|
|
2141
2510
|
symbolIndex++;
|
|
2142
2511
|
}
|
|
2512
|
+
this._emitProgress("symbol-table", position + 1, fields.length);
|
|
2143
2513
|
return symbols;
|
|
2144
2514
|
});
|
|
2145
2515
|
}
|
|
@@ -2158,13 +2528,18 @@ var init_QvdFileReader = __esm({
|
|
|
2158
2528
|
* same for every row, so they are hoisted out of the loop and the inner loop does arithmetic
|
|
2159
2529
|
* into a typed array and nothing else. Rows are assembled later, once, in `load()`.
|
|
2160
2530
|
*
|
|
2161
|
-
*
|
|
2531
|
+
* The window is what makes chunked iteration cheap: `decodeIndexColumn` walks records by
|
|
2532
|
+
* `base += recordSize`, so decoding rows k to k+n is a question of where the buffer slice starts
|
|
2533
|
+
* and how many iterations run. Nothing about the decoder changed to support it.
|
|
2534
|
+
*
|
|
2535
|
+
* @param {QvdRowWindow} window The rows to decode.
|
|
2162
2536
|
*/
|
|
2163
|
-
async _parseIndexTable(
|
|
2164
|
-
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(
|
|
2537
|
+
async _parseIndexTable(window) {
|
|
2538
|
+
const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "parseIndexTable");
|
|
2165
2539
|
this._rowsDecoded = rowsToLoad;
|
|
2166
|
-
this._indexColumns = fields.map(
|
|
2167
|
-
(
|
|
2540
|
+
this._indexColumns = fields.map((field, position) => {
|
|
2541
|
+
this._throwIfAborted();
|
|
2542
|
+
const column = decodeIndexColumn(
|
|
2168
2543
|
indexBuffer,
|
|
2169
2544
|
recordSize,
|
|
2170
2545
|
rowsToLoad,
|
|
@@ -2172,8 +2547,10 @@ var init_QvdFileReader = __esm({
|
|
|
2172
2547
|
parseInt(field["BitWidth"], 10),
|
|
2173
2548
|
parseInt(field["Bias"], 10),
|
|
2174
2549
|
new Int32Array(rowsToLoad)
|
|
2175
|
-
)
|
|
2176
|
-
|
|
2550
|
+
);
|
|
2551
|
+
this._emitProgress("index-table", position + 1, fields.length);
|
|
2552
|
+
return column;
|
|
2553
|
+
});
|
|
2177
2554
|
}
|
|
2178
2555
|
/**
|
|
2179
2556
|
* Reads the file's schema and header metadata, without touching the symbol or index tables.
|
|
@@ -2190,8 +2567,11 @@ var init_QvdFileReader = __esm({
|
|
|
2190
2567
|
* @return {Promise<import('./QvdDataFrame.js').QvdFileMetadata>} The file's schema and header.
|
|
2191
2568
|
*/
|
|
2192
2569
|
async loadMetadata() {
|
|
2193
|
-
await this._readData(null, true);
|
|
2570
|
+
await this._readData({ offset: 0, limit: null }, true);
|
|
2571
|
+
this._emitProgress("header", 0, 1);
|
|
2194
2572
|
await this._parseHeader();
|
|
2573
|
+
this._emitProgress("header", 1, 1);
|
|
2574
|
+
this._throwIfAborted();
|
|
2195
2575
|
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2196
2576
|
const header = this._header["QvdTableHeader"];
|
|
2197
2577
|
let fields = header["Fields"]?.["QvdFieldHeader"] ?? [];
|
|
@@ -2226,114 +2606,200 @@ var init_QvdFileReader = __esm({
|
|
|
2226
2606
|
/**
|
|
2227
2607
|
* Loads the QVD file into memory and parses it.
|
|
2228
2608
|
*
|
|
2229
|
-
* @param {number|null
|
|
2230
|
-
*
|
|
2231
|
-
*
|
|
2609
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
|
|
2610
|
+
* The rows to load. A number or null means what it always meant - the first N rows, or all of
|
|
2611
|
+
* them - and `{offset, limit}` is the same thing said more precisely, so `5` and
|
|
2612
|
+
* `{offset: 0, limit: 5}` are one read. `maxRows` is accepted as a second name for `limit`.
|
|
2613
|
+
* @throws {QvdValidationError} If the window is not a non-negative integer, null, or a valid
|
|
2614
|
+
* `{offset, limit}` object.
|
|
2232
2615
|
* @return {Promise<QvdDataFrame>} The loaded QVD file.
|
|
2233
2616
|
*/
|
|
2234
|
-
async load(
|
|
2235
|
-
const
|
|
2617
|
+
async load(window = null) {
|
|
2618
|
+
const rows = normaliseWindow(window, this._path);
|
|
2619
|
+
const prepared = await this._prepare(rows);
|
|
2620
|
+
await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
|
|
2621
|
+
const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
|
|
2622
|
+
return new QvdDataFrame(data, prepared.columns, prepared.metadata, {
|
|
2623
|
+
...prepared.loadStats,
|
|
2624
|
+
rowsLoaded: data.length
|
|
2625
|
+
});
|
|
2626
|
+
}
|
|
2627
|
+
/**
|
|
2628
|
+
* Reads the file as columns, without ever materialising rows.
|
|
2629
|
+
*
|
|
2630
|
+
* Shares every step with `load()` up to the point where rows would be built - see `_prepare`.
|
|
2631
|
+
* What it keeps instead is what the decoder already produced: one `Int32Array` of stored
|
|
2632
|
+
* indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
|
|
2633
|
+
* fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
|
|
2634
|
+
* bytes per row rather than a boxed value per cell, and the symbols are a few thousand
|
|
2635
|
+
* entries shared across every row that uses them.
|
|
2636
|
+
*
|
|
2637
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
|
|
2638
|
+
* The rows to decode, in the same spellings `load()` accepts.
|
|
2639
|
+
* @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
|
|
2640
|
+
*/
|
|
2641
|
+
async loadColumnar(window = null) {
|
|
2642
|
+
const rows = normaliseWindow(window, this._path);
|
|
2643
|
+
const prepared = await this._prepare(rows);
|
|
2644
|
+
await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
|
|
2645
|
+
const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
|
|
2236
2646
|
assert2(this._indexColumns, "The QVD file index table has not been parsed.");
|
|
2237
|
-
|
|
2238
|
-
|
|
2239
|
-
|
|
2240
|
-
|
|
2241
|
-
|
|
2242
|
-
|
|
2243
|
-
|
|
2244
|
-
|
|
2245
|
-
}
|
|
2246
|
-
data[row] = values;
|
|
2247
|
-
}
|
|
2248
|
-
loadStats.rowsLoaded = data.length;
|
|
2249
|
-
return new QvdDataFrame(data, columns, metadata, loadStats);
|
|
2647
|
+
return new QvdColumnTable2({
|
|
2648
|
+
columns: prepared.columns,
|
|
2649
|
+
codesByField: this._indexColumns,
|
|
2650
|
+
symbolsByField: prepared.resolvedByField,
|
|
2651
|
+
rowCount: this._rowsDecoded,
|
|
2652
|
+
metadata: prepared.metadata,
|
|
2653
|
+
loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
|
|
2654
|
+
});
|
|
2250
2655
|
}
|
|
2251
2656
|
/**
|
|
2252
|
-
*
|
|
2657
|
+
* Yields the window as data frames of at most `chunkSize` rows.
|
|
2253
2658
|
*
|
|
2254
|
-
*
|
|
2255
|
-
*
|
|
2256
|
-
*
|
|
2257
|
-
*
|
|
2258
|
-
*
|
|
2659
|
+
* The file is opened, read and parsed **once**; only the index decode and the row building
|
|
2660
|
+
* happen per chunk. That is the whole reason this exists as a method rather than as a loop of
|
|
2661
|
+
* `load({offset, limit})` calls at the call site: the symbol table has to be parsed in full
|
|
2662
|
+
* whatever the chunk size - a stored index in the last chunk can address the first symbol -
|
|
2663
|
+
* and re-parsing it per chunk is what makes the obvious implementation cost more than a plain
|
|
2664
|
+
* load rather than less. PyQvd's chunked read does re-read it, and the comment on #140 records
|
|
2665
|
+
* that as a limitation rather than a design.
|
|
2259
2666
|
*
|
|
2260
|
-
*
|
|
2261
|
-
*
|
|
2262
|
-
*
|
|
2263
|
-
*
|
|
2667
|
+
* What it bounds is row materialisation, which is what actually dominates a large read's heap.
|
|
2668
|
+
* Two chunks of rows are alive at a time, not one - `for await` keeps the yielded frame
|
|
2669
|
+
* reachable while this generator builds the next - which is why `liveRows` below is
|
|
2670
|
+
* `chunkSize * 2`, and why the heap it needs is twice what one chunk suggests.
|
|
2671
|
+
*
|
|
2672
|
+
* A window covering no rows yields nothing at all, rather than one empty frame - so
|
|
2673
|
+
* `for await` over an exhausted offset does nothing, which is what a paging loop wants.
|
|
2674
|
+
*
|
|
2675
|
+
* @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} window
|
|
2676
|
+
* The rows to cover, in the same spellings `load()` accepts.
|
|
2677
|
+
* @param {number} chunkSize Rows per frame. Must be a positive integer.
|
|
2678
|
+
* @return {AsyncGenerator<QvdDataFrame>} The chunks, in order.
|
|
2264
2679
|
*/
|
|
2265
|
-
async
|
|
2266
|
-
if (
|
|
2267
|
-
throw new QvdValidationError("
|
|
2268
|
-
provided:
|
|
2269
|
-
type: typeof
|
|
2680
|
+
async *iterateRows(window, chunkSize) {
|
|
2681
|
+
if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
|
|
2682
|
+
throw new QvdValidationError("chunkSize must be a positive integer", {
|
|
2683
|
+
provided: chunkSize,
|
|
2684
|
+
type: typeof chunkSize,
|
|
2270
2685
|
file: this._path
|
|
2271
2686
|
});
|
|
2272
2687
|
}
|
|
2273
|
-
|
|
2688
|
+
const liveRows = { rows: chunkSize * 2, perChunk: 2 };
|
|
2689
|
+
const rows = normaliseWindow(window, this._path);
|
|
2690
|
+
const prepared = await this._prepare(rows, liveRows);
|
|
2691
|
+
for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
|
|
2692
|
+
this._throwIfAborted();
|
|
2693
|
+
const count = Math.min(chunkSize, prepared.rowsAvailable - done);
|
|
2694
|
+
const offset = prepared.offset + done;
|
|
2695
|
+
await this._parseIndexTable({ offset, limit: count });
|
|
2696
|
+
const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
|
|
2697
|
+
yield new QvdDataFrame(data, prepared.columns, prepared.metadata, {
|
|
2698
|
+
...prepared.loadStats,
|
|
2699
|
+
offset,
|
|
2700
|
+
rowsLoaded: data.length
|
|
2701
|
+
});
|
|
2702
|
+
}
|
|
2703
|
+
}
|
|
2704
|
+
/**
|
|
2705
|
+
* Reads the file and resolves its symbols, stopping short of decoding any rows.
|
|
2706
|
+
*
|
|
2707
|
+
* Everything `load()`, `loadColumnar()` and `iterateRows()` have in common, which is everything
|
|
2708
|
+
* that depends on the file rather than on the window. Two read paths for one binary format is
|
|
2709
|
+
* the drift risk #113 is the standing example of - a stored index resolved one way here and
|
|
2710
|
+
* another way there returns plausible wrong values and throws nothing - so there is one path,
|
|
2711
|
+
* and the entry points differ only in what they do with what it returns and how many rows they
|
|
2712
|
+
* ask for at a time.
|
|
2713
|
+
*
|
|
2714
|
+
* @param {QvdRowWindow} window The rows the read covers.
|
|
2715
|
+
* @param {{rows: number, perChunk: number}|null} [liveRows] Rows held at one instant when that
|
|
2716
|
+
* is fewer than the window covers, and how many of them one row of the caller's chunk size
|
|
2717
|
+
* accounts for. Only `iterateRows` passes it; every other read holds what it covers.
|
|
2718
|
+
* @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
|
|
2719
|
+
* resolvedByField: Array<Array<any>>, rowsAvailable: number, offset: number}>} The parsed
|
|
2720
|
+
* file, with the window as it resolved against it.
|
|
2721
|
+
* @private
|
|
2722
|
+
*/
|
|
2723
|
+
async _prepare(window, liveRows = null) {
|
|
2724
|
+
this._throwIfAborted();
|
|
2725
|
+
await this._readData(window, false, liveRows);
|
|
2726
|
+
this._emitProgress("header", 0, 1);
|
|
2274
2727
|
await this._parseHeader();
|
|
2728
|
+
this._emitProgress("header", 1, 1);
|
|
2729
|
+
this._throwIfAborted();
|
|
2730
|
+
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2731
|
+
const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
|
|
2732
|
+
const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2733
|
+
const resolved = resolveWindow(window, totalRows);
|
|
2734
|
+
const rowsAvailable = resolved.limit;
|
|
2275
2735
|
let symbolsToKeep = null;
|
|
2276
2736
|
let symbolsKept = null;
|
|
2277
|
-
if (
|
|
2278
|
-
const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
|
|
2737
|
+
if (window.limit !== null || window.offset > 0) {
|
|
2279
2738
|
if (symbolTableLength > this._symbolFilteringThreshold) {
|
|
2280
|
-
symbolsToKeep = await this._analyzeIndexTableSymbolUsage(
|
|
2281
|
-
symbolsKept =
|
|
2739
|
+
symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
|
|
2740
|
+
symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
|
|
2282
2741
|
}
|
|
2283
2742
|
}
|
|
2284
|
-
await this._parseSymbolTable(symbolsToKeep,
|
|
2285
|
-
await this._parseIndexTable(maxRows);
|
|
2286
|
-
assert2(this._header, "The QVD file header has not been parsed.");
|
|
2743
|
+
await this._parseSymbolTable(symbolsToKeep, rowsAvailable, liveRows);
|
|
2287
2744
|
assert2(this._symbolTable, "The QVD file symbol table has not been parsed.");
|
|
2288
|
-
|
|
2745
|
+
this._throwIfAborted();
|
|
2289
2746
|
const resolvedByField = this._symbolTable.map((symbols) => {
|
|
2290
|
-
const
|
|
2747
|
+
const resolved2 = new Array(symbols.length);
|
|
2291
2748
|
for (let index = 0; index < symbols.length; index++) {
|
|
2292
2749
|
const value = symbols[index]?.toPrimaryValue();
|
|
2293
|
-
|
|
2750
|
+
resolved2[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
|
|
2294
2751
|
}
|
|
2295
|
-
return
|
|
2752
|
+
return resolved2;
|
|
2296
2753
|
});
|
|
2297
|
-
|
|
2298
|
-
|
|
2299
|
-
fields = [fields];
|
|
2300
|
-
}
|
|
2301
|
-
const columns = fields.map((field) => field["FieldName"]);
|
|
2754
|
+
assert2(this._selectedFields, "The QVD file fields have not been resolved.");
|
|
2755
|
+
const columns = this._selectedFields.map((field) => field["FieldName"]);
|
|
2302
2756
|
const metadata = this._header["QvdTableHeader"];
|
|
2303
2757
|
const loadStats = {
|
|
2304
|
-
symbolTableBytes:
|
|
2305
|
-
totalRows
|
|
2306
|
-
rowsLoaded:
|
|
2758
|
+
symbolTableBytes: symbolTableLength,
|
|
2759
|
+
totalRows,
|
|
2760
|
+
rowsLoaded: 0,
|
|
2761
|
+
offset: resolved.offset,
|
|
2307
2762
|
symbolFiltering: symbolsToKeep !== null,
|
|
2308
2763
|
symbolsKept
|
|
2309
2764
|
};
|
|
2310
|
-
return { columns, metadata, loadStats, resolvedByField };
|
|
2765
|
+
return { columns, metadata, loadStats, resolvedByField, rowsAvailable, offset: resolved.offset };
|
|
2311
2766
|
}
|
|
2312
2767
|
/**
|
|
2313
|
-
*
|
|
2768
|
+
* Builds rows from the columns currently decoded.
|
|
2314
2769
|
*
|
|
2315
|
-
*
|
|
2316
|
-
*
|
|
2317
|
-
*
|
|
2318
|
-
*
|
|
2319
|
-
*
|
|
2320
|
-
* entries shared across every row that uses them.
|
|
2770
|
+
* `data` stays eager: of the four ways this library is used - a full read, a preview already
|
|
2771
|
+
* bounded by a limit, writing an array out, and reading metadata - not one is helped by
|
|
2772
|
+
* materialising a row only when it is touched, and a lazy accessor would cost a proxy, a cache
|
|
2773
|
+
* and mutation semantics to serve none of them. A caller who wants columns without paying for
|
|
2774
|
+
* rows uses `QvdColumnTable`, which stops before this loop.
|
|
2321
2775
|
*
|
|
2322
|
-
* @param {
|
|
2323
|
-
* @
|
|
2776
|
+
* @param {Array<Array<any>>} resolvedByField One resolved value per distinct symbol, per field.
|
|
2777
|
+
* @param {number} progressBase Rows already delivered before this call, so that progress over a
|
|
2778
|
+
* chunked iteration counts the whole window rather than restarting at every chunk.
|
|
2779
|
+
* @param {number} progressTotal Rows the whole window covers.
|
|
2780
|
+
* @return {Array<Array<any>>} The rows.
|
|
2781
|
+
* @private
|
|
2324
2782
|
*/
|
|
2325
|
-
|
|
2326
|
-
const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
|
|
2327
|
-
const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
|
|
2783
|
+
_buildRows(resolvedByField, progressBase, progressTotal) {
|
|
2328
2784
|
assert2(this._indexColumns, "The QVD file index table has not been parsed.");
|
|
2329
|
-
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
|
|
2333
|
-
|
|
2334
|
-
|
|
2335
|
-
|
|
2336
|
-
|
|
2785
|
+
const indexColumns = this._indexColumns;
|
|
2786
|
+
const fieldCount = indexColumns.length;
|
|
2787
|
+
const rowCount = this._rowsDecoded;
|
|
2788
|
+
const data = new Array(rowCount);
|
|
2789
|
+
const reportInterval = Math.max(1, Math.floor(progressTotal / 100));
|
|
2790
|
+
for (let row = 0; row < rowCount; row++) {
|
|
2791
|
+
const values = new Array(fieldCount);
|
|
2792
|
+
for (let field = 0; field < fieldCount; field++) {
|
|
2793
|
+
const symbolIndex = indexColumns[field][row];
|
|
2794
|
+
values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
|
|
2795
|
+
}
|
|
2796
|
+
data[row] = values;
|
|
2797
|
+
if ((progressBase + row + 1) % reportInterval === 0 || row + 1 === rowCount) {
|
|
2798
|
+
this._throwIfAborted();
|
|
2799
|
+
this._emitProgress("rows", progressBase + row + 1, progressTotal);
|
|
2800
|
+
}
|
|
2801
|
+
}
|
|
2802
|
+
return data;
|
|
2337
2803
|
}
|
|
2338
2804
|
};
|
|
2339
2805
|
}
|
|
@@ -2344,6 +2810,7 @@ var QvdDataFrame;
|
|
|
2344
2810
|
var init_QvdDataFrame = __esm({
|
|
2345
2811
|
"src/QvdDataFrame.js"() {
|
|
2346
2812
|
init_QvdErrors();
|
|
2813
|
+
init_readOptions();
|
|
2347
2814
|
QvdDataFrame = class _QvdDataFrame {
|
|
2348
2815
|
/**
|
|
2349
2816
|
* Represents the data frame stored inside a QVD file.
|
|
@@ -2386,12 +2853,15 @@ var init_QvdDataFrame = __esm({
|
|
|
2386
2853
|
/**
|
|
2387
2854
|
* Returns statistics about the read that produced this data frame.
|
|
2388
2855
|
*
|
|
2389
|
-
*
|
|
2390
|
-
*
|
|
2856
|
+
* Carried by every frame that came from a file - `fromQvd()`, and each chunk `iterate()` yields,
|
|
2857
|
+
* which is how a chunk reports its `offset`. `fromDict()`, `head()`, `tail()`, `rows()` and
|
|
2858
|
+
* `select()` describe no particular read and report null rather than a stale figure.
|
|
2391
2859
|
*
|
|
2392
2860
|
* The main use is confirming that a lazy load actually filtered the symbol table:
|
|
2393
2861
|
* `symbolFiltering` says whether the two-pass path ran, and `symbolsKept` how many symbols
|
|
2394
|
-
* survived it
|
|
2862
|
+
* survived it. Note that a bounded read does not filter on its own - the two-pass path engages
|
|
2863
|
+
* only above `symbolFilteringThreshold`, so on a file below it this reports false and every
|
|
2864
|
+
* symbol was parsed however few rows were asked for.
|
|
2395
2865
|
*
|
|
2396
2866
|
* @return {QvdLoadStats|null} Load statistics, or null if this frame did not come from a file.
|
|
2397
2867
|
*/
|
|
@@ -2768,6 +3238,18 @@ var init_QvdDataFrame = __esm({
|
|
|
2768
3238
|
* @param {Object} [options] Optional loading options.
|
|
2769
3239
|
* @param {number|null} [options.maxRows] The maximum number of rows to load. Must be a non-negative
|
|
2770
3240
|
* integer; if not specified or null, all rows are loaded. Anything else throws a QvdValidationError.
|
|
3241
|
+
* This is the older name for `limit`; the two are the same option and passing both throws.
|
|
3242
|
+
* @param {number|null} [options.limit] Rows to read, counting from `offset`. The same number as
|
|
3243
|
+
* `maxRows`, spelled so that it reads correctly beside an offset.
|
|
3244
|
+
* @param {number} [options.offset=0] File row to start at. An offset past the end of the file
|
|
3245
|
+
* returns no rows rather than throwing, so a paging loop terminates on its own.
|
|
3246
|
+
* @param {Array<string>|null} [options.fields] Field names to read, in the order they should
|
|
3247
|
+
* appear in the result. Unselected fields have their symbols skipped entirely rather than
|
|
3248
|
+
* parsed and discarded. An unknown or repeated name throws.
|
|
3249
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
|
|
3250
|
+
* read proceeds - the same shape `toQvd`'s callback receives.
|
|
3251
|
+
* @param {AbortSignal} [options.signal] Cancels the read. The rejection is `signal.reason`,
|
|
3252
|
+
* which is a `DOMException` named `AbortError` unless you aborted with a reason of your own.
|
|
2771
3253
|
* @param {string} [options.allowedDir] Optional allowed directory path. If provided, the file path
|
|
2772
3254
|
* must be within this directory, with symlinks resolved first, so a link inside it that points
|
|
2773
3255
|
* outside it is rejected. Defaults to the current working directory. To permit an entire
|
|
@@ -2780,17 +3262,42 @@ var init_QvdDataFrame = __esm({
|
|
|
2780
3262
|
* **Zero disables the memory check entirely.**
|
|
2781
3263
|
* @param {number} [options.symbolFilteringThreshold=52428800] Symbol table size, in bytes, above which
|
|
2782
3264
|
* a lazy load switches to the two-pass filtering path. Defaults to 50MB.
|
|
2783
|
-
* @throws {QvdValidationError} If
|
|
3265
|
+
* @throws {QvdValidationError} If a window option is not a non-negative integer, if both
|
|
3266
|
+
* `maxRows` and `limit` are given, or if `fields` names a column the file does not have.
|
|
2784
3267
|
* @return {Promise<QvdDataFrame>} The data frame of the QVD file.
|
|
2785
3268
|
*/
|
|
2786
3269
|
static async fromQvd(path3, options = {}) {
|
|
2787
3270
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
2788
|
-
|
|
2789
|
-
|
|
2790
|
-
|
|
2791
|
-
|
|
2792
|
-
|
|
2793
|
-
|
|
3271
|
+
return await new QvdFileReader2(path3, readerOptionsFrom(options)).load(windowFrom(options));
|
|
3272
|
+
}
|
|
3273
|
+
/**
|
|
3274
|
+
* Reads a QVD file in chunks, as an async generator of data frames.
|
|
3275
|
+
*
|
|
3276
|
+
* The file is opened, read and parsed once; only the index decode and the row building happen
|
|
3277
|
+
* per chunk, so what this bounds is row materialisation - the part that actually dominates a
|
|
3278
|
+
* large read's heap. It is **not** constant-memory reading of an arbitrarily large file: the
|
|
3279
|
+
* symbol table is parsed in full whatever the chunk size, because a stored index in the last
|
|
3280
|
+
* chunk can address the first symbol. On a high-cardinality file that table is the bulk of the
|
|
3281
|
+
* cost, and `readMetadata` is the only read that avoids it.
|
|
3282
|
+
*
|
|
3283
|
+
* ```js
|
|
3284
|
+
* for await (const chunk of QvdDataFrame.iterate('big.qvd', {chunkSize: 50_000})) {
|
|
3285
|
+
* process(chunk.data);
|
|
3286
|
+
* }
|
|
3287
|
+
* ```
|
|
3288
|
+
*
|
|
3289
|
+
* A window covering no rows yields nothing, so a loop over an exhausted offset simply does not
|
|
3290
|
+
* run its body.
|
|
3291
|
+
*
|
|
3292
|
+
* @param {string} path The path to the QVD file.
|
|
3293
|
+
* @param {Object} [options] The same options `fromQvd` takes, plus:
|
|
3294
|
+
* @param {number} [options.chunkSize=100000] Rows per frame. Must be a positive integer.
|
|
3295
|
+
* @return {AsyncGenerator<QvdDataFrame>} The chunks, in file order.
|
|
3296
|
+
*/
|
|
3297
|
+
static async *iterate(path3, options = {}) {
|
|
3298
|
+
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
3299
|
+
const reader = new QvdFileReader2(path3, readerOptionsFrom(options));
|
|
3300
|
+
yield* reader.iterateRows(windowFrom(options), options.chunkSize === void 0 ? 1e5 : options.chunkSize);
|
|
2794
3301
|
}
|
|
2795
3302
|
/**
|
|
2796
3303
|
* Reads a QVD file's schema and header metadata, without reading its data.
|
|
@@ -2814,11 +3321,15 @@ var init_QvdDataFrame = __esm({
|
|
|
2814
3321
|
* @param {Object} [options] Optional reading options.
|
|
2815
3322
|
* @param {string} [options.allowedDir] Optional allowed directory path, applied exactly as it
|
|
2816
3323
|
* is for `fromQvd`.
|
|
3324
|
+
* @param {Function} [options.onProgress] Called with `{stage, current, total, percent}`, as on
|
|
3325
|
+
* the reads that return data. Only the `read` and `header` stages occur here; there are no
|
|
3326
|
+
* symbols to parse and no rows to build.
|
|
3327
|
+
* @param {AbortSignal} [options.signal] Cancels the read, rejecting with `signal.reason`.
|
|
2817
3328
|
* @return {Promise<QvdFileMetadata>} The file's schema and header metadata.
|
|
2818
3329
|
*/
|
|
2819
3330
|
static async readMetadata(path3, options = {}) {
|
|
2820
3331
|
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
2821
|
-
return await new QvdFileReader2(path3,
|
|
3332
|
+
return await new QvdFileReader2(path3, metadataOptionsFrom(options)).loadMetadata();
|
|
2822
3333
|
}
|
|
2823
3334
|
/**
|
|
2824
3335
|
* Constructs a data frame from a dictionary.
|