qvdjs 2.1.0 → 2.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -3
- package/dist/index.cjs +804 -151
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +805 -152
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -2197,13 +2197,13 @@ function estimateRowMemory(rows, columnCount) {
|
|
|
2197
2197
|
}
|
|
2198
2198
|
return BASE_BYTES + rows * (ROW_BASE_BYTES + PER_CELL_BYTES * columnCount);
|
|
2199
2199
|
}
|
|
2200
|
-
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null) {
|
|
2200
|
+
function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null, wholeSymbols = false) {
|
|
2201
2201
|
const FULL_PARSE_OVERHEAD = 6;
|
|
2202
2202
|
const MINIMAL_OVERHEAD = 0.01;
|
|
2203
2203
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
2204
2204
|
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
2205
2205
|
const rowMemory = materialisesRows ? estimateRowMemory(liveRows, columnCount) : BASE_BYTES;
|
|
2206
|
-
if (maxRows === null || maxRows >= totalRows) {
|
|
2206
|
+
if (maxRows === null || maxRows >= totalRows || wholeSymbols) {
|
|
2207
2207
|
return symbolTableSize * FULL_PARSE_OVERHEAD + rowMemory;
|
|
2208
2208
|
}
|
|
2209
2209
|
const rowPercentage = maxRows / totalRows;
|
|
@@ -2212,8 +2212,8 @@ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount =
|
|
|
2212
2212
|
const skippedSymbolsMemory = symbolTableSize * (1 - symbolPercentage) * MINIMAL_OVERHEAD;
|
|
2213
2213
|
return keptSymbolsMemory + skippedSymbolsMemory + rowMemory;
|
|
2214
2214
|
}
|
|
2215
|
-
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false) {
|
|
2216
|
-
const costOf = /* @__PURE__ */ __name((rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0), "costOf");
|
|
2215
|
+
function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false, wholeSymbols = false) {
|
|
2216
|
+
const costOf = /* @__PURE__ */ __name((rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows, null, wholeSymbols) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0), "costOf");
|
|
2217
2217
|
if (costOf(totalRows) <= budget) {
|
|
2218
2218
|
return totalRows;
|
|
2219
2219
|
}
|
|
@@ -2232,11 +2232,19 @@ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, mat
|
|
|
2232
2232
|
}
|
|
2233
2233
|
return low;
|
|
2234
2234
|
}
|
|
2235
|
-
function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false) {
|
|
2235
|
+
function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false, wholeSymbols = false) {
|
|
2236
2236
|
const covered = windowRows === null || windowRows >= totalRows ? totalRows : windowRows;
|
|
2237
2237
|
const fits = /* @__PURE__ */ __name((chunk) => {
|
|
2238
2238
|
const live = Math.min(chunk * liveRowsPerChunk, covered);
|
|
2239
|
-
const cost = estimateMemoryUsage(
|
|
2239
|
+
const cost = estimateMemoryUsage(
|
|
2240
|
+
symbolTableSize,
|
|
2241
|
+
windowRows,
|
|
2242
|
+
totalRows,
|
|
2243
|
+
columnCount,
|
|
2244
|
+
true,
|
|
2245
|
+
chunk * liveRowsPerChunk,
|
|
2246
|
+
wholeSymbols
|
|
2247
|
+
) + (includeExternal ? estimateExternalMemory(live, columnCount) : 0);
|
|
2240
2248
|
return cost <= budget;
|
|
2241
2249
|
}, "fits");
|
|
2242
2250
|
if (fits(covered)) {
|
|
@@ -2257,8 +2265,10 @@ function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, col
|
|
|
2257
2265
|
}
|
|
2258
2266
|
return low;
|
|
2259
2267
|
}
|
|
2260
|
-
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null, bytesHeld = null, readBytes = null) {
|
|
2268
|
+
function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null, bytesHeld = null, readBytes = null, wholeSymbols = false, retainedBytes = 0) {
|
|
2261
2269
|
const answer = checkMemory({
|
|
2270
|
+
wholeSymbols,
|
|
2271
|
+
retainedBytes,
|
|
2262
2272
|
symbolTableSize,
|
|
2263
2273
|
maxRows,
|
|
2264
2274
|
totalRows,
|
|
@@ -2290,7 +2300,9 @@ function checkMemory({
|
|
|
2290
2300
|
live = null,
|
|
2291
2301
|
bytesHeld = null,
|
|
2292
2302
|
readBytes = null,
|
|
2293
|
-
measured = null
|
|
2303
|
+
measured = null,
|
|
2304
|
+
wholeSymbols = false,
|
|
2305
|
+
retainedBytes = 0
|
|
2294
2306
|
}) {
|
|
2295
2307
|
if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
|
|
2296
2308
|
throw new exports.QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", {
|
|
@@ -2305,8 +2317,21 @@ function checkMemory({
|
|
|
2305
2317
|
const rowsLive = live === null ? null : live.rows;
|
|
2306
2318
|
const liveRowsPerChunk = live === null ? 1 : live.perChunk;
|
|
2307
2319
|
const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
|
|
2308
|
-
const heapMemory = estimateMemoryUsage(
|
|
2309
|
-
|
|
2320
|
+
const heapMemory = estimateMemoryUsage(
|
|
2321
|
+
symbolTableSize,
|
|
2322
|
+
maxRows,
|
|
2323
|
+
totalRows,
|
|
2324
|
+
columnCount,
|
|
2325
|
+
materialisesRows,
|
|
2326
|
+
rowsLive,
|
|
2327
|
+
wholeSymbols
|
|
2328
|
+
);
|
|
2329
|
+
const {
|
|
2330
|
+
held,
|
|
2331
|
+
afterRelease: heldAfterRelease = null,
|
|
2332
|
+
forRows: heldForRows,
|
|
2333
|
+
forChunk: heldForChunk
|
|
2334
|
+
} = bytesHeld ?? noBytesHeld;
|
|
2310
2335
|
const externalMemory = estimateExternalMemory(liveRows, columnCount) + held;
|
|
2311
2336
|
const bounded = budget.candidates.map((candidate) => {
|
|
2312
2337
|
const heapOnly = candidate.source === "V8 heap limit";
|
|
@@ -2342,6 +2367,25 @@ function checkMemory({
|
|
|
2342
2367
|
const availableMemory = binding ? binding.bytes : budget.bytes;
|
|
2343
2368
|
const estimatedMemory = binding ? binding.needs : heapMemory;
|
|
2344
2369
|
const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
|
|
2370
|
+
const retainedIsTheReason = (() => {
|
|
2371
|
+
if (binding === null || retainedBytes <= 0) {
|
|
2372
|
+
return false;
|
|
2373
|
+
}
|
|
2374
|
+
const heapAfter = estimateMemoryUsage(
|
|
2375
|
+
Math.max(0, symbolTableSize - retainedBytes),
|
|
2376
|
+
maxRows,
|
|
2377
|
+
totalRows,
|
|
2378
|
+
columnCount,
|
|
2379
|
+
materialisesRows,
|
|
2380
|
+
rowsLive,
|
|
2381
|
+
wholeSymbols
|
|
2382
|
+
);
|
|
2383
|
+
const externalAfter = estimateExternalMemory(liveRows, columnCount) + (heldAfterRelease ?? held);
|
|
2384
|
+
return budget.candidates.every((candidate) => {
|
|
2385
|
+
const heapOnly = candidate.source === "V8 heap limit";
|
|
2386
|
+
return (heapOnly ? heapAfter : heapAfter + externalAfter) <= candidate.bytes * safetyFactor;
|
|
2387
|
+
});
|
|
2388
|
+
})();
|
|
2345
2389
|
if (binding) {
|
|
2346
2390
|
const includeExternal = !binding.heapOnly;
|
|
2347
2391
|
const rowsHeldBudget = /* @__PURE__ */ __name((rows) => maxAllowedMemory - (includeExternal ? heldForRows(rows) : 0), "rowsHeldBudget");
|
|
@@ -2351,7 +2395,8 @@ function checkMemory({
|
|
|
2351
2395
|
totalRows,
|
|
2352
2396
|
columnCount,
|
|
2353
2397
|
materialisesRows,
|
|
2354
|
-
includeExternal
|
|
2398
|
+
includeExternal,
|
|
2399
|
+
wholeSymbols
|
|
2355
2400
|
), "fitting");
|
|
2356
2401
|
const firstGuess = fitting(liveRows);
|
|
2357
2402
|
const over = fitting(firstGuess);
|
|
@@ -2377,7 +2422,8 @@ function checkMemory({
|
|
|
2377
2422
|
totalRows,
|
|
2378
2423
|
columnCount,
|
|
2379
2424
|
liveRowsPerChunk,
|
|
2380
|
-
includeExternal
|
|
2425
|
+
includeExternal,
|
|
2426
|
+
wholeSymbols
|
|
2381
2427
|
), "chunkFitting");
|
|
2382
2428
|
const callersChunk = chunked ? Math.max(1, Math.floor(rowsLive / Math.max(1, liveRowsPerChunk))) : 0;
|
|
2383
2429
|
const firstChunk = chunked ? chunkFitting(callersChunk) : 0;
|
|
@@ -2386,13 +2432,17 @@ function checkMemory({
|
|
|
2386
2432
|
const knob = chunked ? "chunkSize" : "limit";
|
|
2387
2433
|
const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
|
|
2388
2434
|
const nothingFits = recommendedValue === 0;
|
|
2435
|
+
const held2 = retainedIsTheReason;
|
|
2436
|
+
const fixedCost = held2 ? "the columns this file is holding exceed" : "the symbol table alone exceeds";
|
|
2437
|
+
const release = held2 ? `Close the file and open it again to release them, or raise ` : `Raise `;
|
|
2438
|
+
const releaseFirst = held2 ? `Close the file and open it again to release the columns it is holding, which is the direct remedy. ` : "";
|
|
2389
2439
|
let advice;
|
|
2390
2440
|
if (nothingFits) {
|
|
2391
|
-
advice = `No row count fits this budget -
|
|
2441
|
+
advice = `No row count fits this budget - ${fixedCost} it, so ${knob} cannot help. ` + (containerBound ? `${release}the container's memory limit.` : `${release}the heap with --max-old-space-size, or raise memorySafetyFactor.`);
|
|
2392
2442
|
} else if (containerBound) {
|
|
2393
|
-
advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${formatCount(recommendedValue)} rows or less).`;
|
|
2443
|
+
advice = releaseFirst + `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${formatCount(recommendedValue)} rows or less).`;
|
|
2394
2444
|
} else {
|
|
2395
|
-
advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${formatCount(recommendedValue)} rows or less), or raise the heap with --max-old-space-size.`;
|
|
2445
|
+
advice = releaseFirst + `Try holding fewer rows using the ${knob} parameter (recommended: ${formatCount(recommendedValue)} rows or less), or raise the heap with --max-old-space-size.`;
|
|
2396
2446
|
}
|
|
2397
2447
|
const suggestions = [];
|
|
2398
2448
|
if (!nothingFits) {
|
|
@@ -2416,10 +2466,16 @@ function checkMemory({
|
|
|
2416
2466
|
exact: symbolTableSize === 0,
|
|
2417
2467
|
suggestions,
|
|
2418
2468
|
refusal: {
|
|
2419
|
-
message: `Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
|
|
2469
|
+
message: `Insufficient memory to load file safely. ${retainedIsTheReason ? "Columns held" : "Symbol table"}: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
|
|
2420
2470
|
context: {
|
|
2421
2471
|
symbolTableSize,
|
|
2422
2472
|
symbolTableSizeMB: sizeMB,
|
|
2473
|
+
// Whether that figure is the file's symbol table or what an open file is still holding. A
|
|
2474
|
+
// paging read is charged for every column any of its pages decoded and kept, so a caller
|
|
2475
|
+
// branching on the refusal needs to know which of the two it is looking at - the remedies
|
|
2476
|
+
// differ, and for this one releasing is a remedy where raising the limit is only a workaround.
|
|
2477
|
+
holdsDecodedColumns: retainedIsTheReason,
|
|
2478
|
+
retainedSymbolBytes: retainedBytes,
|
|
2423
2479
|
estimatedMemoryMB: estimatedMB,
|
|
2424
2480
|
availableMemoryMB: availableMB,
|
|
2425
2481
|
heapLimitMB,
|
|
@@ -2467,13 +2523,21 @@ function budgetOf(budget, tightest, safetyFactor) {
|
|
|
2467
2523
|
function formatCount(value) {
|
|
2468
2524
|
return value.toLocaleString("en-US");
|
|
2469
2525
|
}
|
|
2470
|
-
function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
|
|
2526
|
+
function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, wholeSymbols = false) {
|
|
2471
2527
|
const LARGE_SYMBOL_TABLE_WARNING = usableOldSpaceLimit() * 0.125;
|
|
2472
2528
|
if (symbolTableSize <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
2473
2529
|
return;
|
|
2474
2530
|
}
|
|
2475
2531
|
const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
|
|
2476
|
-
const estimatedMemory = estimateMemoryUsage(
|
|
2532
|
+
const estimatedMemory = estimateMemoryUsage(
|
|
2533
|
+
symbolTableSize,
|
|
2534
|
+
maxRows,
|
|
2535
|
+
totalRows,
|
|
2536
|
+
columnCount,
|
|
2537
|
+
materialisesRows,
|
|
2538
|
+
null,
|
|
2539
|
+
wholeSymbols
|
|
2540
|
+
);
|
|
2477
2541
|
if (estimatedMemory <= LARGE_SYMBOL_TABLE_WARNING) {
|
|
2478
2542
|
return;
|
|
2479
2543
|
}
|
|
@@ -2579,19 +2643,39 @@ function validateHeaderStructure(headerObj, filePath, stage) {
|
|
|
2579
2643
|
});
|
|
2580
2644
|
return fieldList;
|
|
2581
2645
|
}
|
|
2582
|
-
function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
|
|
2646
|
+
function validateSymbolTableSizeEarly(symbolTableLength, filePath, retainedBytes = 0) {
|
|
2583
2647
|
const heapLimit = getHeapLimit();
|
|
2584
2648
|
const MAX_SYMBOL_TABLE_SIZE = heapLimit * 0.125;
|
|
2585
2649
|
if (symbolTableLength > MAX_SYMBOL_TABLE_SIZE) {
|
|
2586
2650
|
const sizeMB = Math.round(symbolTableLength / 1024 / 1024);
|
|
2587
2651
|
const maxMB = Math.round(MAX_SYMBOL_TABLE_SIZE / 1024 / 1024);
|
|
2588
2652
|
const heapMB = Math.round(heapLimit / 1024 / 1024);
|
|
2653
|
+
if (retainedBytes > 0 && symbolTableLength - retainedBytes <= MAX_SYMBOL_TABLE_SIZE) {
|
|
2654
|
+
throw new exports.QvdValidationError(
|
|
2655
|
+
`Columns held too large (${sizeMB}MB exceeds ${maxMB}MB limit). This open file is holding the columns its pages have decoded, and they have grown past the ceiling rather than the file's own symbol table being large. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) closing the file and opening it again, which releases what the pages decoded, (2) paging over fewer columns with fields, or (3) increasing heap size with --max-old-space-size.`,
|
|
2656
|
+
{
|
|
2657
|
+
file: filePath,
|
|
2658
|
+
symbolTableSize: symbolTableLength,
|
|
2659
|
+
symbolTableSizeMB: sizeMB,
|
|
2660
|
+
holdsDecodedColumns: true,
|
|
2661
|
+
retainedSymbolBytes: retainedBytes,
|
|
2662
|
+
maxAllowed: MAX_SYMBOL_TABLE_SIZE,
|
|
2663
|
+
maxAllowedMB: maxMB,
|
|
2664
|
+
heapLimitMB: heapMB,
|
|
2665
|
+
reason: "memory"
|
|
2666
|
+
}
|
|
2667
|
+
);
|
|
2668
|
+
}
|
|
2589
2669
|
throw new exports.QvdValidationError(
|
|
2590
2670
|
`Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without a row window - maxRows, limit or offset - since the symbol table is read in full either way, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
|
|
2591
2671
|
{
|
|
2592
2672
|
file: filePath,
|
|
2593
2673
|
symbolTableSize: symbolTableLength,
|
|
2594
2674
|
symbolTableSizeMB: sizeMB,
|
|
2675
|
+
// Present and false, not absent. A caller told it can branch on this has to find it on both
|
|
2676
|
+
// refusals, or the branch reads `undefined` for the commoner of the two.
|
|
2677
|
+
holdsDecodedColumns: false,
|
|
2678
|
+
retainedSymbolBytes: retainedBytes,
|
|
2595
2679
|
maxAllowed: MAX_SYMBOL_TABLE_SIZE,
|
|
2596
2680
|
maxAllowedMB: maxMB,
|
|
2597
2681
|
heapLimitMB: heapMB,
|
|
@@ -3601,6 +3685,26 @@ function symbolBytesOf(selected, symbolTableLength) {
|
|
|
3601
3685
|
function readPasses(analysisAhead) {
|
|
3602
3686
|
return analysisAhead ? 2 : 1;
|
|
3603
3687
|
}
|
|
3688
|
+
function validateWatchers(onProgress, signal, path5) {
|
|
3689
|
+
if (onProgress !== void 0 && typeof onProgress !== "function") {
|
|
3690
|
+
throw new exports.QvdValidationError("onProgress must be a function", {
|
|
3691
|
+
provided: onProgress,
|
|
3692
|
+
type: typeof onProgress,
|
|
3693
|
+
reason: "option",
|
|
3694
|
+
option: "onProgress",
|
|
3695
|
+
file: path5
|
|
3696
|
+
});
|
|
3697
|
+
}
|
|
3698
|
+
if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
|
|
3699
|
+
throw new exports.QvdValidationError("signal must be an AbortSignal", {
|
|
3700
|
+
provided: signal,
|
|
3701
|
+
type: typeof signal,
|
|
3702
|
+
reason: "option",
|
|
3703
|
+
option: "signal",
|
|
3704
|
+
file: path5
|
|
3705
|
+
});
|
|
3706
|
+
}
|
|
3707
|
+
}
|
|
3604
3708
|
var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, SLICE_BYTES, COUNT_SYMBOLS_PAST; exports.QvdFileReader = void 0;
|
|
3605
3709
|
var init_QvdFileReader = __esm({
|
|
3606
3710
|
"src/QvdFileReader.js"() {
|
|
@@ -3625,6 +3729,7 @@ var init_QvdFileReader = __esm({
|
|
|
3625
3729
|
__name(parseHeaderXml, "parseHeaderXml");
|
|
3626
3730
|
__name(symbolBytesOf, "symbolBytesOf");
|
|
3627
3731
|
__name(readPasses, "readPasses");
|
|
3732
|
+
__name(validateWatchers, "validateWatchers");
|
|
3628
3733
|
exports.QvdFileReader = class {
|
|
3629
3734
|
static {
|
|
3630
3735
|
__name(this, "QvdFileReader");
|
|
@@ -3705,20 +3810,7 @@ var init_QvdFileReader = __esm({
|
|
|
3705
3810
|
});
|
|
3706
3811
|
}
|
|
3707
3812
|
this._sliceBytes = sliceBytes;
|
|
3708
|
-
|
|
3709
|
-
throw new exports.QvdValidationError("onProgress must be a function", {
|
|
3710
|
-
provided: onProgress,
|
|
3711
|
-
type: typeof onProgress,
|
|
3712
|
-
file: this._path
|
|
3713
|
-
});
|
|
3714
|
-
}
|
|
3715
|
-
if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
|
|
3716
|
-
throw new exports.QvdValidationError("signal must be an AbortSignal", {
|
|
3717
|
-
provided: signal,
|
|
3718
|
-
type: typeof signal,
|
|
3719
|
-
file: this._path
|
|
3720
|
-
});
|
|
3721
|
-
}
|
|
3813
|
+
validateWatchers(onProgress, signal, this._path);
|
|
3722
3814
|
this._requestedFields = fields === void 0 ? null : fields;
|
|
3723
3815
|
this._onProgress = onProgress;
|
|
3724
3816
|
this._signal = signal;
|
|
@@ -3727,6 +3819,8 @@ var init_QvdFileReader = __esm({
|
|
|
3727
3819
|
this._failed = null;
|
|
3728
3820
|
this._reading = false;
|
|
3729
3821
|
this._symbolAreas = null;
|
|
3822
|
+
this._symbolCache = null;
|
|
3823
|
+
this._cachedFor = null;
|
|
3730
3824
|
this._headerOffset = null;
|
|
3731
3825
|
this._symbolTableOffset = null;
|
|
3732
3826
|
this._indexTableOffset = null;
|
|
@@ -3738,6 +3832,7 @@ var init_QvdFileReader = __esm({
|
|
|
3738
3832
|
this._indexColumns = null;
|
|
3739
3833
|
this._rowsDecoded = 0;
|
|
3740
3834
|
this._fileSize = null;
|
|
3835
|
+
this._fileIdentity = null;
|
|
3741
3836
|
this._headerMatchesFile = false;
|
|
3742
3837
|
}
|
|
3743
3838
|
/**
|
|
@@ -3966,8 +4061,9 @@ var init_QvdFileReader = __esm({
|
|
|
3966
4061
|
const indexTableOffset = symbolTableOffset + symbolTableLength;
|
|
3967
4062
|
const recordSize = headerInteger(headerObj["QvdTableHeader"]["RecordByteSize"]);
|
|
3968
4063
|
const totalRows = headerInteger(headerObj["QvdTableHeader"]["NoOfRecords"]);
|
|
3969
|
-
const { size: fileSize } = await handle.stat().catch(failed);
|
|
4064
|
+
const { size: fileSize, ino, dev, mtimeMs } = await handle.stat().catch(failed);
|
|
3970
4065
|
this._fileSize = fileSize;
|
|
4066
|
+
this._fileIdentity = `${dev}:${ino}:${mtimeMs}:${fileSize}`;
|
|
3971
4067
|
this._headerMatchesFile = false;
|
|
3972
4068
|
const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
|
|
3973
4069
|
(value) => Number.isSafeInteger(value) && value >= 0
|
|
@@ -3983,7 +4079,12 @@ var init_QvdFileReader = __esm({
|
|
|
3983
4079
|
this._headerBuffer = headerBuffer.subarray(0, headerEndIndex);
|
|
3984
4080
|
const selected = selectFields(headerFields, this._requestedFields, this._path);
|
|
3985
4081
|
const columnCount = selected.length;
|
|
3986
|
-
const symbolBytes = symbolBytesOf(selected, symbolTableLength);
|
|
4082
|
+
const symbolBytes = symbolBytesOf(this._fieldsHeldAfter(selected, headerFields), symbolTableLength);
|
|
4083
|
+
const readSymbolBytes = symbolBytesOf(this._fieldsReadBy(selected), symbolTableLength);
|
|
4084
|
+
const retainedBytes = symbolBytesOf(
|
|
4085
|
+
this._fieldsHeldAfter(selected, headerFields).slice(selected.length),
|
|
4086
|
+
symbolTableLength
|
|
4087
|
+
);
|
|
3987
4088
|
const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
|
|
3988
4089
|
const windowRows = resolved.limit;
|
|
3989
4090
|
if (headerNumbersUsable && this._headerMatchesFile) {
|
|
@@ -3997,11 +4098,12 @@ var init_QvdFileReader = __esm({
|
|
|
3997
4098
|
this._materialisesRows,
|
|
3998
4099
|
liveRows,
|
|
3999
4100
|
this._bytesHeld(
|
|
4000
|
-
|
|
4101
|
+
readSymbolBytes,
|
|
4001
4102
|
windowRows,
|
|
4002
4103
|
recordSize,
|
|
4003
4104
|
liveRows,
|
|
4004
|
-
this.
|
|
4105
|
+
this._analysisAhead(window, resolved, totalRows, symbolTableLength),
|
|
4106
|
+
symbolBytes - retainedBytes
|
|
4005
4107
|
),
|
|
4006
4108
|
// What it reads, which is not what it holds - the records go through one buffer and are not
|
|
4007
4109
|
// kept. Carried so that a refusal's `check` says everything the pre-flight would have said.
|
|
@@ -4009,7 +4111,12 @@ var init_QvdFileReader = __esm({
|
|
|
4009
4111
|
// Twice over where the symbol-usage pass is still ahead of it: that pass reads the window's
|
|
4010
4112
|
// records to find which symbols the rows use, and the decode then reads them again. Counted
|
|
4011
4113
|
// once, the figure understated the I/O of exactly the reads that do the most of it.
|
|
4012
|
-
|
|
4114
|
+
readSymbolBytes + readPasses(this._analysisAhead(window, resolved, totalRows, symbolTableLength)) * windowRows * recordSize,
|
|
4115
|
+
// A paging read keeps whole columns, so the estimate must not discount its symbols as a window's
|
|
4116
|
+
// sample of them - see `estimateMemoryUsage`. Under-charging is the direction that ends in a
|
|
4117
|
+
// heap-limit abort rather than an error.
|
|
4118
|
+
this._symbolCache !== null,
|
|
4119
|
+
retainedBytes
|
|
4013
4120
|
);
|
|
4014
4121
|
}
|
|
4015
4122
|
if (window.offset === 0 && window.limit === null) {
|
|
@@ -4017,7 +4124,7 @@ var init_QvdFileReader = __esm({
|
|
|
4017
4124
|
return;
|
|
4018
4125
|
}
|
|
4019
4126
|
const rowsToLoad = windowRows;
|
|
4020
|
-
validateSymbolTableSizeEarly(symbolBytes, this._path);
|
|
4127
|
+
validateSymbolTableSizeEarly(symbolBytes, this._path, retainedBytes);
|
|
4021
4128
|
validateRecordSize(recordSize, this._path, "readData");
|
|
4022
4129
|
validateRecordCount(totalRows, this._path, "readData");
|
|
4023
4130
|
const fileBytesRequired = indexTableOffset + (resolved.offset + rowsToLoad) * recordSize;
|
|
@@ -4129,9 +4236,14 @@ var init_QvdFileReader = __esm({
|
|
|
4129
4236
|
* would hold.
|
|
4130
4237
|
* @private
|
|
4131
4238
|
*/
|
|
4132
|
-
_bytesHeld(symbolBytes, windowRows, recordSize, liveRows, analysisAhead) {
|
|
4239
|
+
_bytesHeld(symbolBytes, windowRows, recordSize, liveRows, analysisAhead, freshSymbolBytes = null) {
|
|
4133
4240
|
return {
|
|
4134
4241
|
held: this._bytesHeldBy(symbolBytes, this._recordsAtOnce(windowRows, liveRows, analysisAhead), recordSize),
|
|
4242
|
+
// What it would hold having closed the file and opened it again: every selected column read fresh,
|
|
4243
|
+
// because nothing is cached any more. Higher than `held`, not lower - a cached column this read
|
|
4244
|
+
// selects costs nothing to read now and would cost its area then. Without it the counterfactual
|
|
4245
|
+
// that decides whether releasing helps was answered against the warm figure and said yes too often.
|
|
4246
|
+
afterRelease: freshSymbolBytes === null ? null : this._bytesHeldBy(freshSymbolBytes, this._recordsAtOnce(windowRows, liveRows, analysisAhead), recordSize),
|
|
4135
4247
|
// A window of so many rows reads so many records at a time, and the pass that reads it ahead of the
|
|
4136
4248
|
// decode reads the same rows, so the buffer is sized from the rows either way.
|
|
4137
4249
|
forRows: /* @__PURE__ */ __name((rows) => this._bytesHeldBy(symbolBytes, rows, recordSize), "forRows"),
|
|
@@ -4194,6 +4306,191 @@ var init_QvdFileReader = __esm({
|
|
|
4194
4306
|
const chunkRows = liveRows === null ? windowRows : Math.max(1, Math.floor(liveRows.rows / Math.max(1, liveRows.perChunk)));
|
|
4195
4307
|
return analysisAhead ? Math.max(windowRows, chunkRows) : chunkRows;
|
|
4196
4308
|
}
|
|
4309
|
+
/**
|
|
4310
|
+
* The fields this reader will be holding the symbols of once this read has finished.
|
|
4311
|
+
*
|
|
4312
|
+
* The ones it selects, and - while paging - the ones it decoded for an earlier page and kept. That
|
|
4313
|
+
* union is what the memory checks have to be sized by, because it is what is live: a reader four
|
|
4314
|
+
* pages into a wide file holds four columns' values whether or not this page asks about them, and a
|
|
4315
|
+
* check sized by this page alone would approve a fifth column that does not fit beside them.
|
|
4316
|
+
*
|
|
4317
|
+
* The same set for every check, so the ceilings, the guard and the pre-flight cannot disagree about
|
|
4318
|
+
* what a paging read costs. Without a cache it is just the selection, which is what every one-shot
|
|
4319
|
+
* read has always been sized by.
|
|
4320
|
+
*
|
|
4321
|
+
* @param {Array<any>} selected The fields this read selects.
|
|
4322
|
+
* @param {Array<any>} all Every field in the header, to find a cached one by name.
|
|
4323
|
+
* @return {Array<any>} The fields whose symbols will be live.
|
|
4324
|
+
* @private
|
|
4325
|
+
*/
|
|
4326
|
+
_fieldsHeldAfter(selected, all) {
|
|
4327
|
+
if (this._symbolCache === null || this._symbolCache.size === 0) {
|
|
4328
|
+
return selected;
|
|
4329
|
+
}
|
|
4330
|
+
const names = new Set(selected.map((field) => field["FieldName"]));
|
|
4331
|
+
const cached = all.filter(
|
|
4332
|
+
(field) => !names.has(field["FieldName"]) && this._symbolCache !== null && this._symbolCache.has(field["FieldName"])
|
|
4333
|
+
);
|
|
4334
|
+
return [...selected, ...cached];
|
|
4335
|
+
}
|
|
4336
|
+
/**
|
|
4337
|
+
* The fields whose symbol areas this read will actually read.
|
|
4338
|
+
*
|
|
4339
|
+
* The selection, less anything already decoded and kept. `_fieldsHeldAfter` answers what the read will
|
|
4340
|
+
* be *holding*, which is the right figure for the heap; this is the right one for the bytes it buffers
|
|
4341
|
+
* while parsing and for the I/O it reports, because a cached column's area is left out of the plan
|
|
4342
|
+
* entirely and never read.
|
|
4343
|
+
*
|
|
4344
|
+
* Sized by the wrong one of the two, a warm page was charged external bytes for buffers it never
|
|
4345
|
+
* allocates - and external bytes bind against a container limit, so a page that fits could be refused -
|
|
4346
|
+
* and `estimate.readBytes` claimed I/O it does not perform: on four columns with three cached it
|
|
4347
|
+
* reported 3,155,600 bytes for a read of 788,930.
|
|
4348
|
+
*
|
|
4349
|
+
* @param {Array<any>} selected The fields this read selects.
|
|
4350
|
+
* @return {Array<any>} The fields whose areas will be read.
|
|
4351
|
+
* @private
|
|
4352
|
+
*/
|
|
4353
|
+
_fieldsReadBy(selected) {
|
|
4354
|
+
if (this._symbolCache === null || this._symbolCache.size === 0) {
|
|
4355
|
+
return selected;
|
|
4356
|
+
}
|
|
4357
|
+
return selected.filter(
|
|
4358
|
+
(field) => this._symbolCache !== null && !this._symbolCache.has(field["FieldName"])
|
|
4359
|
+
);
|
|
4360
|
+
}
|
|
4361
|
+
/**
|
|
4362
|
+
* Keeps what this reader decodes, so that a later read of the same file does not decode it again.
|
|
4363
|
+
*
|
|
4364
|
+
* For a caller reading one file many times over - a `QvdFile` and its pages - and off by default,
|
|
4365
|
+
* because every other entry point is one read and would only be holding values nobody will ask for
|
|
4366
|
+
* again. Decoding the symbols is 72% of a page of a hundred rows from a 300,000-row file; the rest is
|
|
4367
|
+
* the open, the header, the records and the rows.
|
|
4368
|
+
*
|
|
4369
|
+
* It turns the two-pass symbol path off with it. That path decodes only the symbols a window's rows
|
|
4370
|
+
* use, which is right for one read and wrong for a cache: a later page asking for a row that uses a
|
|
4371
|
+
* skipped symbol would read `undefined` where the value is. So a cached field is always a whole
|
|
4372
|
+
* field, walked and checked against its `NoOfSymbols` like any other.
|
|
4373
|
+
*
|
|
4374
|
+
* @return {void}
|
|
4375
|
+
*/
|
|
4376
|
+
beginPaging() {
|
|
4377
|
+
this._symbolCache = /* @__PURE__ */ new Map();
|
|
4378
|
+
}
|
|
4379
|
+
/**
|
|
4380
|
+
* Reads with the fields the caller names for this read alone, rather than the reader's own.
|
|
4381
|
+
*
|
|
4382
|
+
* A `QvdFile` is opened once and its pages may each name a projection, so the selection cannot be
|
|
4383
|
+
* fixed at construction as it is for every other entry point.
|
|
4384
|
+
*
|
|
4385
|
+
* It holds until the next call replaces it rather than being cleared by the read, so **every caller
|
|
4386
|
+
* sets it before every read**, passing the file's own fields where the page named none. A caller that
|
|
4387
|
+
* relied on it being empty would instead get the projection of whatever ran last: that is what made a
|
|
4388
|
+
* `check()` naming no fields answer for the previous page's columns.
|
|
4389
|
+
*
|
|
4390
|
+
* @param {Array<string>|null|undefined} fields The fields, or undefined to use the reader's own.
|
|
4391
|
+
* @return {void}
|
|
4392
|
+
*/
|
|
4393
|
+
selectForNextRead(fields) {
|
|
4394
|
+
if (fields !== void 0) {
|
|
4395
|
+
this._requestedFields = fields;
|
|
4396
|
+
}
|
|
4397
|
+
}
|
|
4398
|
+
/**
|
|
4399
|
+
* Watches the next read with the caller's `onProgress` and `signal`, rather than the reader's own.
|
|
4400
|
+
*
|
|
4401
|
+
* Both belong to one call, and a reader is told them when it is built - so a `QvdFile` page that named
|
|
4402
|
+
* either used to get a reader of its own. That made passing a progress callback change what the read
|
|
4403
|
+
* did rather than only observing it: a fresh reader is not paging, so it took the two-pass symbol
|
|
4404
|
+
* path, reported a different `loadStats.symbolFiltering`, and cached nothing. An observer must not
|
|
4405
|
+
* change what it observes, and a caller must not have to choose between cancelling a page and paging
|
|
4406
|
+
* cheaply.
|
|
4407
|
+
*
|
|
4408
|
+
* Like `selectForNextRead`, it holds until the next call replaces it rather than being cleared by the
|
|
4409
|
+
* read, so a caller that sets it for one page and not the next is still watched on the next - pass the
|
|
4410
|
+
* file's own watchers explicitly, as `QvdFile._page` does, rather than leaving them out.
|
|
4411
|
+
*
|
|
4412
|
+
* @param {{onProgress?: Function, signal?: AbortSignal}} [watchers] What this read is watched with.
|
|
4413
|
+
* @return {void}
|
|
4414
|
+
*/
|
|
4415
|
+
observeNextRead({ onProgress, signal } = {}) {
|
|
4416
|
+
validateWatchers(onProgress, signal, this._path);
|
|
4417
|
+
this._onProgress = onProgress;
|
|
4418
|
+
this._signal = signal;
|
|
4419
|
+
}
|
|
4420
|
+
/**
|
|
4421
|
+
* Whether the symbol-usage pass runs for this read, cache and all.
|
|
4422
|
+
*
|
|
4423
|
+
* `_analysisWouldRun` answers whether the window wants the pass; a paging read never takes it, because
|
|
4424
|
+
* a column decoded in part cannot be kept. Asked in one place because it was asked in two and they
|
|
4425
|
+
* disagreed: the pass was gated on the cache while the memory charge and `estimate.readBytes` were
|
|
4426
|
+
* not, so every page of a file above the threshold was charged a slice of records it never buffered
|
|
4427
|
+
* and reported twice the bytes it read.
|
|
4428
|
+
*
|
|
4429
|
+
* @param {QvdRowWindow} window The window as the caller spelled it.
|
|
4430
|
+
* @param {{offset: number, limit: number}} resolved Where it lands in this file.
|
|
4431
|
+
* @param {number} totalRows Rows the file declares.
|
|
4432
|
+
* @param {number} symbolTableLength The symbol table's declared length.
|
|
4433
|
+
* @return {boolean} Whether the pass will run.
|
|
4434
|
+
* @private
|
|
4435
|
+
*/
|
|
4436
|
+
_analysisAhead(window, resolved, totalRows, symbolTableLength) {
|
|
4437
|
+
return this._symbolCache === null && this._analysisWouldRun(window, resolved, totalRows, symbolTableLength);
|
|
4438
|
+
}
|
|
4439
|
+
/**
|
|
4440
|
+
* Empties the cache when the header in front of us is not the one it was decoded from.
|
|
4441
|
+
*
|
|
4442
|
+
* The fingerprint is what a rewrite moves: the file's identity on disk - device, inode, modification
|
|
4443
|
+
* time and size - and then the header numbers, down to each cached field's own offset, length and
|
|
4444
|
+
* symbol count.
|
|
4445
|
+
*
|
|
4446
|
+
* The header numbers alone were not enough, and the gap is not exotic. `QvdFileWriter` carries
|
|
4447
|
+
* `CreateUtcTime` over from the metadata it is handed, so reading a QVD, changing one text to another
|
|
4448
|
+
* of the same byte length and writing it back leaves `CreateUtcTime`, `NoOfRecords`, `Offset` and
|
|
4449
|
+
* every field's `Offset`, `Length` and `NoOfSymbols` exactly as they were - a different file the
|
|
4450
|
+
* fingerprint could not tell from the first. The filesystem sees it either way: an atomic write
|
|
4451
|
+
* renames a new file into place, which changes the inode, and an in-place one moves `mtimeMs`.
|
|
4452
|
+
*
|
|
4453
|
+
* @return {void}
|
|
4454
|
+
* @private
|
|
4455
|
+
*/
|
|
4456
|
+
_forgetCacheIfFileChanged() {
|
|
4457
|
+
if (this._symbolCache === null || this._symbolCache.size === 0) {
|
|
4458
|
+
return;
|
|
4459
|
+
}
|
|
4460
|
+
if (this._cachedFor !== this._fileFingerprint()) {
|
|
4461
|
+
this._symbolCache = /* @__PURE__ */ new Map();
|
|
4462
|
+
this._cachedFor = null;
|
|
4463
|
+
}
|
|
4464
|
+
}
|
|
4465
|
+
/**
|
|
4466
|
+
* What identifies the file this reader's cache was decoded from.
|
|
4467
|
+
*
|
|
4468
|
+
* @return {string} The fingerprint.
|
|
4469
|
+
* @private
|
|
4470
|
+
*/
|
|
4471
|
+
_fileFingerprint() {
|
|
4472
|
+
assert4__default.default(this._header && this._allFields, "The QVD file header has not been parsed.");
|
|
4473
|
+
const header = this._header["QvdTableHeader"];
|
|
4474
|
+
return [
|
|
4475
|
+
// First, because it is the only part that moves when a rewrite preserves the header's numbers.
|
|
4476
|
+
this._fileIdentity,
|
|
4477
|
+
header["CreateUtcTime"],
|
|
4478
|
+
header["NoOfRecords"],
|
|
4479
|
+
header["Offset"],
|
|
4480
|
+
...this._allFields.map(
|
|
4481
|
+
(field) => `${field["FieldName"]}:${field["Offset"]}:${field["Length"]}:${field["NoOfSymbols"]}`
|
|
4482
|
+
)
|
|
4483
|
+
].join("|");
|
|
4484
|
+
}
|
|
4485
|
+
/**
|
|
4486
|
+
* Drops everything this reader has decoded, so that nothing outlives the caller that wanted it.
|
|
4487
|
+
*
|
|
4488
|
+
* @return {void}
|
|
4489
|
+
*/
|
|
4490
|
+
endPaging() {
|
|
4491
|
+
this._symbolCache = null;
|
|
4492
|
+
this._cachedFor = null;
|
|
4493
|
+
}
|
|
4197
4494
|
/**
|
|
4198
4495
|
* The symbol table's length, as much of it as the file holds: what the header declares, cut short where
|
|
4199
4496
|
* the file ends. Known before a byte of the table is read, so everything that can refuse the table is
|
|
@@ -4222,7 +4519,13 @@ var init_QvdFileReader = __esm({
|
|
|
4222
4519
|
* order, so ranges that touch are merged: a read of every field is one range, and so is a read of fields
|
|
4223
4520
|
* that happen to be neighbours. A read of one field of twenty reads that field's area alone.
|
|
4224
4521
|
*
|
|
4225
|
-
*
|
|
4522
|
+
* A field whose symbols this reader already holds is left out, because its bytes are not wanted: the
|
|
4523
|
+
* ranges are what gets read, and including a cached field's span had a page read every byte of every
|
|
4524
|
+
* column it named, cached or not. Measured on four columns of 20,000 distinct texts, a page naming all
|
|
4525
|
+
* four with three of them cached read all four columns' bytes - 1,155,600 of them, where 288,930 were
|
|
4526
|
+
* needed. The decode was saved and the I/O was not, which on the files #122 is about is the whole cost.
|
|
4527
|
+
*
|
|
4528
|
+
* Built once per read, from the fields the read must read, and each field's metadata is checked as it is
|
|
4226
4529
|
* added - a range is arithmetic on `Offset` and `Length`, and those have to be inside the table first.
|
|
4227
4530
|
* `_parseSymbolTable` checks every field of the file, selected or not, before it parses any.
|
|
4228
4531
|
*
|
|
@@ -4237,7 +4540,10 @@ var init_QvdFileReader = __esm({
|
|
|
4237
4540
|
}
|
|
4238
4541
|
assert4__default.default(this._selectedFields, "The QVD file fields have not been resolved before their symbols were read.");
|
|
4239
4542
|
const tableLength = this._symbolTableLength();
|
|
4240
|
-
const
|
|
4543
|
+
const toRead = this._selectedFields.filter(
|
|
4544
|
+
(field) => this._symbolCache === null || !this._symbolCache.has(field["FieldName"])
|
|
4545
|
+
);
|
|
4546
|
+
const areas = toRead.map((field) => {
|
|
4241
4547
|
validateFieldMetadata(field, tableLength, this._path);
|
|
4242
4548
|
const start = headerInteger(field["Offset"]);
|
|
4243
4549
|
return { field, start, end: start + headerInteger(field["Length"]) };
|
|
@@ -4559,9 +4865,12 @@ var init_QvdFileReader = __esm({
|
|
|
4559
4865
|
}
|
|
4560
4866
|
const allFields = this._allFields;
|
|
4561
4867
|
const fields = this._selectedFields;
|
|
4868
|
+
this._forgetCacheIfFileChanged();
|
|
4562
4869
|
const symbolTableSize = this._symbolTableLength();
|
|
4563
4870
|
const plan = this._symbolAreaPlan();
|
|
4564
|
-
const
|
|
4871
|
+
const readSymbolBytes = plan.ranges.reduce((sum, range) => sum + (range.end - range.start), 0);
|
|
4872
|
+
const retainedBytes = symbolBytesOf(this._fieldsHeldAfter(fields, allFields).slice(fields.length), symbolTableSize);
|
|
4873
|
+
const symbolBytes = readSymbolBytes + retainedBytes;
|
|
4565
4874
|
const totalRows = headerInteger(this._header["QvdTableHeader"]["NoOfRecords"]);
|
|
4566
4875
|
const recordSize = headerInteger(this._header["QvdTableHeader"]["RecordByteSize"]);
|
|
4567
4876
|
validateSymbolTableSize(symbolBytes, this._path, totalRows);
|
|
@@ -4575,13 +4884,22 @@ var init_QvdFileReader = __esm({
|
|
|
4575
4884
|
fields.length,
|
|
4576
4885
|
this._materialisesRows,
|
|
4577
4886
|
liveRows,
|
|
4578
|
-
this._bytesHeld(
|
|
4887
|
+
this._bytesHeld(readSymbolBytes, rowsToLoad, recordSize, liveRows, false, symbolBytes - retainedBytes),
|
|
4579
4888
|
// `symbolsToKeep` is non-null exactly when the symbol-usage pass has run, and a pass that has
|
|
4580
4889
|
// run has read the window's records once already - so the read's total is two passes over them.
|
|
4581
|
-
|
|
4890
|
+
readSymbolBytes + readPasses(symbolsToKeep !== null) * rowsToLoad * recordSize,
|
|
4891
|
+
this._symbolCache !== null,
|
|
4892
|
+
retainedBytes
|
|
4582
4893
|
);
|
|
4583
4894
|
}
|
|
4584
|
-
warnLargeSymbolTable(
|
|
4895
|
+
warnLargeSymbolTable(
|
|
4896
|
+
symbolBytes,
|
|
4897
|
+
rowsToLoad,
|
|
4898
|
+
totalRows,
|
|
4899
|
+
fields.length,
|
|
4900
|
+
this._materialisesRows,
|
|
4901
|
+
this._symbolCache !== null
|
|
4902
|
+
);
|
|
4585
4903
|
for (const field of allFields) {
|
|
4586
4904
|
validateFieldMetadata(field, symbolTableSize, this._path);
|
|
4587
4905
|
}
|
|
@@ -4589,24 +4907,36 @@ var init_QvdFileReader = __esm({
|
|
|
4589
4907
|
const symbolTable = [];
|
|
4590
4908
|
for (const [position, field] of fields.entries()) {
|
|
4591
4909
|
this._throwIfAborted();
|
|
4910
|
+
const cached = this._symbolCache?.get(field["FieldName"]);
|
|
4911
|
+
if (cached) {
|
|
4912
|
+
symbolTable.push(cached);
|
|
4913
|
+
this._emitProgress("symbol-table", position + 1, fields.length);
|
|
4914
|
+
continue;
|
|
4915
|
+
}
|
|
4592
4916
|
const area = await this._symbolAreaOf(field);
|
|
4593
|
-
|
|
4594
|
-
|
|
4595
|
-
|
|
4596
|
-
|
|
4597
|
-
|
|
4598
|
-
|
|
4599
|
-
|
|
4600
|
-
|
|
4601
|
-
|
|
4602
|
-
|
|
4603
|
-
|
|
4604
|
-
|
|
4605
|
-
|
|
4606
|
-
|
|
4607
|
-
|
|
4608
|
-
)
|
|
4917
|
+
const parsed = parseFieldSymbols(
|
|
4918
|
+
area.buffer,
|
|
4919
|
+
area.start,
|
|
4920
|
+
area.end,
|
|
4921
|
+
// Checked against the symbols the area holds, which is the one check that sees a terminator
|
|
4922
|
+
// damaged in the middle of it (#124). A cached field was checked when it was decoded, which is
|
|
4923
|
+
// why the cache may only hold a field a full walk produced.
|
|
4924
|
+
headerInteger(field["NoOfSymbols"]),
|
|
4925
|
+
// By position, matching how `_analyzeIndexTableSymbolUsage` built it. Both walk
|
|
4926
|
+
// `this._selectedFields`, so position is the one key that cannot collide.
|
|
4927
|
+
symbolsToKeep ? symbolsToKeep[position] : null,
|
|
4928
|
+
field["FieldName"],
|
|
4929
|
+
this._path,
|
|
4930
|
+
void 0,
|
|
4931
|
+
area.base
|
|
4609
4932
|
);
|
|
4933
|
+
symbolTable.push(parsed);
|
|
4934
|
+
if (this._symbolCache && symbolsToKeep === null) {
|
|
4935
|
+
if (this._symbolCache.size === 0) {
|
|
4936
|
+
this._cachedFor = this._fileFingerprint();
|
|
4937
|
+
}
|
|
4938
|
+
this._symbolCache.set(field["FieldName"], parsed);
|
|
4939
|
+
}
|
|
4610
4940
|
this._releaseSymbolArea(field);
|
|
4611
4941
|
this._emitProgress("symbol-table", position + 1, fields.length);
|
|
4612
4942
|
}
|
|
@@ -4701,28 +5031,39 @@ var init_QvdFileReader = __esm({
|
|
|
4701
5031
|
await this._parseHeader();
|
|
4702
5032
|
this._emitProgress("header", 1, 1);
|
|
4703
5033
|
this._throwIfAborted();
|
|
4704
|
-
|
|
4705
|
-
const header = this._header["QvdTableHeader"];
|
|
4706
|
-
const columns = this._allFields.map((field) => field["FieldName"]);
|
|
4707
|
-
const rowCount = headerInteger(header["NoOfRecords"]);
|
|
4708
|
-
validateRecordCount(rowCount, this._path, "readMetadata");
|
|
4709
|
-
const shape = new exports.QvdDataFrame([], columns, header, {
|
|
4710
|
-
symbolTableBytes: headerInteger(header["Offset"]),
|
|
4711
|
-
totalRows: rowCount,
|
|
4712
|
-
rowsLoaded: 0,
|
|
4713
|
-
symbolFiltering: false,
|
|
4714
|
-
symbolsKept: null
|
|
4715
|
-
});
|
|
4716
|
-
return {
|
|
4717
|
-
columns,
|
|
4718
|
-
rowCount,
|
|
4719
|
-
columnCount: columns.length,
|
|
4720
|
-
fields: columns.map((name) => shape.getFieldMetadata(name)),
|
|
4721
|
-
fileMetadata: shape.fileMetadata,
|
|
4722
|
-
metadata: header
|
|
4723
|
-
};
|
|
5034
|
+
return this.describeParsed();
|
|
4724
5035
|
});
|
|
4725
5036
|
}
|
|
5037
|
+
/**
|
|
5038
|
+
* The schema and header of the file this reader has parsed, as `readMetadata()` reports them.
|
|
5039
|
+
*
|
|
5040
|
+
* Built from the parsed header and nothing else, so a caller holding a header - a `QvdFile` - can have
|
|
5041
|
+
* it without reading the file a second time.
|
|
5042
|
+
*
|
|
5043
|
+
* @return {any} The metadata.
|
|
5044
|
+
*/
|
|
5045
|
+
describeParsed() {
|
|
5046
|
+
assert4__default.default(this._header && this._allFields, "The QVD file header has not been parsed.");
|
|
5047
|
+
const header = this._header["QvdTableHeader"];
|
|
5048
|
+
const columns = this._allFields.map((field) => field["FieldName"]);
|
|
5049
|
+
const rowCount = headerInteger(header["NoOfRecords"]);
|
|
5050
|
+
validateRecordCount(rowCount, this._path, "readMetadata");
|
|
5051
|
+
const shape = new exports.QvdDataFrame([], columns, header, {
|
|
5052
|
+
symbolTableBytes: headerInteger(header["Offset"]),
|
|
5053
|
+
totalRows: rowCount,
|
|
5054
|
+
rowsLoaded: 0,
|
|
5055
|
+
symbolFiltering: false,
|
|
5056
|
+
symbolsKept: null
|
|
5057
|
+
});
|
|
5058
|
+
return {
|
|
5059
|
+
columns,
|
|
5060
|
+
rowCount,
|
|
5061
|
+
columnCount: columns.length,
|
|
5062
|
+
fields: columns.map((name) => shape.getFieldMetadata(name)),
|
|
5063
|
+
fileMetadata: shape.fileMetadata,
|
|
5064
|
+
metadata: header
|
|
5065
|
+
};
|
|
5066
|
+
}
|
|
4726
5067
|
/**
|
|
4727
5068
|
* What a read of this file would cost, and whether it fits, without reading it.
|
|
4728
5069
|
*
|
|
@@ -4740,77 +5081,127 @@ var init_QvdFileReader = __esm({
|
|
|
4740
5081
|
async checkRead(rawWindow, { chunkSize = null } = {}) {
|
|
4741
5082
|
const window = normaliseWindow(rawWindow, this._path);
|
|
4742
5083
|
return await this._closingAfter(async () => {
|
|
4743
|
-
await this.
|
|
4744
|
-
this.
|
|
4745
|
-
|
|
4746
|
-
|
|
4747
|
-
|
|
4748
|
-
|
|
4749
|
-
|
|
4750
|
-
|
|
5084
|
+
await this._parseHeaderChecked();
|
|
5085
|
+
return this.checkParsed(window, { chunkSize });
|
|
5086
|
+
});
|
|
5087
|
+
}
|
|
5088
|
+
/**
|
|
5089
|
+
* Reads this file's header, and nothing else, leaving it parsed on the reader.
|
|
5090
|
+
*
|
|
5091
|
+
* What `checkRead` and `QvdFile` both start with: the second asks many questions of one header, so the
|
|
5092
|
+
* read that produces it is separate from the questions. Every check a read makes before it trusts the
|
|
5093
|
+
* header's numbers is made here, so that nothing downstream has to wonder whether they hold.
|
|
5094
|
+
*
|
|
5095
|
+
* @return {Promise<void>} When the header is parsed and checked.
|
|
5096
|
+
*/
|
|
5097
|
+
async parseHeaderOnly() {
|
|
5098
|
+
return await this._closingAfter(async () => await this._parseHeaderChecked());
|
|
5099
|
+
}
|
|
5100
|
+
/**
|
|
5101
|
+
* `parseHeaderOnly`'s body, for a caller already inside a read session - `checkRead` is one.
|
|
5102
|
+
*
|
|
5103
|
+
* @return {Promise<void>} When the header is parsed and checked.
|
|
5104
|
+
* @private
|
|
5105
|
+
*/
|
|
5106
|
+
async _parseHeaderChecked() {
|
|
5107
|
+
await this._readData({ offset: 0, limit: null }, true);
|
|
5108
|
+
this._emitProgress("header", 0, 1);
|
|
5109
|
+
await this._parseHeader();
|
|
5110
|
+
this._emitProgress("header", 1, 1);
|
|
5111
|
+
this._throwIfAborted();
|
|
5112
|
+
assert4__default.default(
|
|
5113
|
+
this._header && this._selectedFields && this._allFields && this._symbolTableOffset !== null,
|
|
5114
|
+
"The QVD file header has not been parsed."
|
|
5115
|
+
);
|
|
5116
|
+
const header = this._header["QvdTableHeader"];
|
|
5117
|
+
const totalRows = headerInteger(header["NoOfRecords"]);
|
|
5118
|
+
const recordSize = headerInteger(header["RecordByteSize"]);
|
|
5119
|
+
const symbolTableLength = headerInteger(header["Offset"]);
|
|
5120
|
+
validateRecordSize(recordSize, this._path, "checkRead");
|
|
5121
|
+
validateRecordCount(totalRows, this._path, "checkRead");
|
|
5122
|
+
if (!this._headerMatchesFile) {
|
|
5123
|
+
throw new exports.QvdCorruptedError("The file is shorter than its header claims.", {
|
|
5124
|
+
file: this._path,
|
|
5125
|
+
fileSize: this._fileSize,
|
|
5126
|
+
requiredBytes: this._symbolTableOffset + symbolTableLength + totalRows * recordSize,
|
|
5127
|
+
stage: "checkRead"
|
|
5128
|
+
});
|
|
5129
|
+
}
|
|
5130
|
+
const tableLength = this._symbolTableLength();
|
|
5131
|
+
for (const field of this._allFields) {
|
|
5132
|
+
validateFieldMetadata(field, tableLength, this._path);
|
|
5133
|
+
validateFieldBitMetadata(field, recordSize, this._path);
|
|
5134
|
+
}
|
|
5135
|
+
validateSymbolAreas(this._allFields, this._path);
|
|
5136
|
+
}
|
|
5137
|
+
/**
|
|
5138
|
+
* What a read of the parsed header's file would cost, with no I/O at all.
|
|
5139
|
+
*
|
|
5140
|
+
* Separate from `checkRead` because a `QvdFile` asks this of one header many times - once per page a
|
|
5141
|
+
* viewer scrolls to - and the header is already in hand. `parseHeaderOnly` has to have run.
|
|
5142
|
+
*
|
|
5143
|
+
* @param {QvdRowWindow} window The rows the read would cover, normalised.
|
|
5144
|
+
* @param {{chunkSize?: number|null, fields?: Array<string>|null, materialisesRows?: boolean}} [options]
|
|
5145
|
+
* `chunkSize` for an `iterate()`; `fields` and `materialisesRows` to ask about a read other than the
|
|
5146
|
+
* one this reader was built for, which is what a `QvdFile` does per call.
|
|
5147
|
+
* @return {any} The answer - see `checkMemory`.
|
|
5148
|
+
*/
|
|
5149
|
+
checkParsed(window, { chunkSize = null, fields = void 0, materialisesRows = void 0 } = {}) {
|
|
5150
|
+
assert4__default.default(this._header && this._allFields, "The QVD file header has not been parsed.");
|
|
5151
|
+
const header = this._header["QvdTableHeader"];
|
|
5152
|
+
const totalRows = headerInteger(header["NoOfRecords"]);
|
|
5153
|
+
const recordSize = headerInteger(header["RecordByteSize"]);
|
|
5154
|
+
const symbolTableLength = headerInteger(header["Offset"]);
|
|
5155
|
+
const builds = materialisesRows === void 0 ? this._materialisesRows : materialisesRows;
|
|
5156
|
+
const selected = selectFields(this._allFields, fields === void 0 ? this._requestedFields : fields, this._path);
|
|
5157
|
+
const resolved = resolveWindow(window, totalRows);
|
|
5158
|
+
const windowRows = resolved.limit;
|
|
5159
|
+
const liveRows = chunkSize === null ? null : { rows: chunkSize * 2, perChunk: 2 };
|
|
5160
|
+
const analysisAhead = this._analysisAhead(window, resolved, totalRows, symbolTableLength);
|
|
5161
|
+
const measured = getMemoryBudget();
|
|
5162
|
+
const ask = /* @__PURE__ */ __name((asked, rows) => {
|
|
5163
|
+
const bytes = symbolBytesOf(this._fieldsHeldAfter(asked, this._allFields), symbolTableLength);
|
|
5164
|
+
const read = symbolBytesOf(this._fieldsReadBy(asked), symbolTableLength);
|
|
5165
|
+
const retained = symbolBytesOf(
|
|
5166
|
+
this._fieldsHeldAfter(asked, this._allFields).slice(asked.length),
|
|
5167
|
+
symbolTableLength
|
|
4751
5168
|
);
|
|
4752
|
-
|
|
4753
|
-
|
|
4754
|
-
|
|
4755
|
-
|
|
4756
|
-
|
|
4757
|
-
|
|
4758
|
-
|
|
4759
|
-
|
|
4760
|
-
|
|
4761
|
-
|
|
4762
|
-
|
|
4763
|
-
|
|
4764
|
-
|
|
4765
|
-
|
|
4766
|
-
|
|
4767
|
-
|
|
4768
|
-
|
|
4769
|
-
|
|
4770
|
-
|
|
4771
|
-
|
|
4772
|
-
|
|
4773
|
-
const
|
|
4774
|
-
|
|
4775
|
-
|
|
4776
|
-
|
|
4777
|
-
|
|
4778
|
-
|
|
4779
|
-
|
|
4780
|
-
|
|
4781
|
-
|
|
4782
|
-
|
|
4783
|
-
|
|
4784
|
-
totalRows,
|
|
4785
|
-
safetyFactor: this._memorySafetyFactor,
|
|
4786
|
-
columnCount: fields.length,
|
|
4787
|
-
materialisesRows: this._materialisesRows,
|
|
4788
|
-
live: liveRows,
|
|
4789
|
-
bytesHeld: this._bytesHeld(bytes, rows, recordSize, liveRows, analysisAhead),
|
|
4790
|
-
// What it reads from the file, which is not what it holds: the symbol areas, and every record
|
|
4791
|
-
// the window covers, read a slice at a time and not kept - twice over where the symbol-usage
|
|
4792
|
-
// pass will run, since it reads them before the decode reads them again.
|
|
4793
|
-
readBytes: bytes + readPasses(analysisAhead) * windowRows * recordSize
|
|
4794
|
-
});
|
|
4795
|
-
}, "ask");
|
|
4796
|
-
const answer = ask(selected, windowRows);
|
|
4797
|
-
if (!answer.fits && selected.length > 1) {
|
|
4798
|
-
const bySize = [...selected].sort(
|
|
4799
|
-
(a, b) => headerInteger(a["Length"]) - headerInteger(b["Length"])
|
|
4800
|
-
);
|
|
4801
|
-
for (let take = selected.length - 1; take >= 1; take -= 1) {
|
|
4802
|
-
const fewer = bySize.slice(0, take);
|
|
4803
|
-
if (ask(fewer, windowRows).fits) {
|
|
4804
|
-
answer.suggestions.push({
|
|
4805
|
-
option: "fields",
|
|
4806
|
-
value: fewer.map((field) => field["FieldName"])
|
|
4807
|
-
});
|
|
4808
|
-
break;
|
|
4809
|
-
}
|
|
5169
|
+
return checkMemory({
|
|
5170
|
+
measured,
|
|
5171
|
+
symbolTableSize: bytes,
|
|
5172
|
+
maxRows: rows,
|
|
5173
|
+
totalRows,
|
|
5174
|
+
safetyFactor: this._memorySafetyFactor,
|
|
5175
|
+
columnCount: asked.length,
|
|
5176
|
+
materialisesRows: builds,
|
|
5177
|
+
live: liveRows,
|
|
5178
|
+
// A paging read keeps whole columns, so it is charged for whole columns - see `estimateMemoryUsage`.
|
|
5179
|
+
wholeSymbols: this._symbolCache !== null,
|
|
5180
|
+
retainedBytes: retained,
|
|
5181
|
+
bytesHeld: this._bytesHeld(read, rows, recordSize, liveRows, analysisAhead, bytes - retained),
|
|
5182
|
+
// What it reads from the file, which is not what it holds: the symbol areas it has still to read,
|
|
5183
|
+
// and every record the window covers, read a slice at a time and not kept - twice over where the
|
|
5184
|
+
// symbol-usage pass will run, since it reads them before the decode reads them again.
|
|
5185
|
+
readBytes: read + readPasses(analysisAhead) * windowRows * recordSize
|
|
5186
|
+
});
|
|
5187
|
+
}, "ask");
|
|
5188
|
+
const answer = ask(selected, windowRows);
|
|
5189
|
+
if (!answer.fits && selected.length > 1) {
|
|
5190
|
+
const bySize = [...selected].sort(
|
|
5191
|
+
(a, b) => headerInteger(a["Length"]) - headerInteger(b["Length"])
|
|
5192
|
+
);
|
|
5193
|
+
for (let take = selected.length - 1; take >= 1; take -= 1) {
|
|
5194
|
+
const fewer = bySize.slice(0, take);
|
|
5195
|
+
if (ask(fewer, windowRows).fits) {
|
|
5196
|
+
answer.suggestions.push({
|
|
5197
|
+
option: "fields",
|
|
5198
|
+
value: fewer.map((field) => field["FieldName"])
|
|
5199
|
+
});
|
|
5200
|
+
break;
|
|
4810
5201
|
}
|
|
4811
5202
|
}
|
|
4812
|
-
|
|
4813
|
-
|
|
5203
|
+
}
|
|
5204
|
+
return answer;
|
|
4814
5205
|
}
|
|
4815
5206
|
/**
|
|
4816
5207
|
* Loads the QVD file into memory and parses it.
|
|
@@ -4976,7 +5367,7 @@ var init_QvdFileReader = __esm({
|
|
|
4976
5367
|
const rowsAvailable = resolved.limit;
|
|
4977
5368
|
let symbolsToKeep = null;
|
|
4978
5369
|
let symbolsKept = null;
|
|
4979
|
-
if (this.
|
|
5370
|
+
if (this._analysisAhead(window, resolved, totalRows, symbolTableLength)) {
|
|
4980
5371
|
symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
|
|
4981
5372
|
symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
|
|
4982
5373
|
}
|
|
@@ -5068,6 +5459,224 @@ var init_QvdFileReader = __esm({
|
|
|
5068
5459
|
}
|
|
5069
5460
|
});
|
|
5070
5461
|
|
|
5462
|
+
// src/QvdFile.js
|
|
5463
|
+
var QvdFile_exports = {};
|
|
5464
|
+
__export(QvdFile_exports, {
|
|
5465
|
+
QvdFile: () => exports.QvdFile
|
|
5466
|
+
});
|
|
5467
|
+
exports.QvdFile = void 0;
|
|
5468
|
+
var init_QvdFile = __esm({
|
|
5469
|
+
"src/QvdFile.js"() {
|
|
5470
|
+
init_QvdErrors();
|
|
5471
|
+
init_readOptions();
|
|
5472
|
+
exports.QvdFile = class {
|
|
5473
|
+
static {
|
|
5474
|
+
__name(this, "QvdFile");
|
|
5475
|
+
}
|
|
5476
|
+
/**
|
|
5477
|
+
* Not called directly - `QvdDataFrame.open()` is the way in, because a `QvdFile` is only ever a file
|
|
5478
|
+
* whose header has been read, and a constructor cannot wait for that.
|
|
5479
|
+
*
|
|
5480
|
+
* @param {any} reader The reader holding the parsed header.
|
|
5481
|
+
* @param {any} metadata What `readMetadata()` returns for this file.
|
|
5482
|
+
* @param {any} options The options the file was opened with.
|
|
5483
|
+
* @private
|
|
5484
|
+
*/
|
|
5485
|
+
constructor(reader, metadata, options) {
|
|
5486
|
+
this._reader = reader;
|
|
5487
|
+
this._metadata = metadata;
|
|
5488
|
+
this._options = options;
|
|
5489
|
+
this._closed = false;
|
|
5490
|
+
this._tail = Promise.resolve();
|
|
5491
|
+
this._readers = { rows: null, columns: null };
|
|
5492
|
+
}
|
|
5493
|
+
/**
|
|
5494
|
+
* The file's header and schema, as `QvdDataFrame.readMetadata()` returns them.
|
|
5495
|
+
*
|
|
5496
|
+
* Read when the file was opened, so this costs nothing and cannot fail.
|
|
5497
|
+
*
|
|
5498
|
+
* @return {any} The metadata.
|
|
5499
|
+
*/
|
|
5500
|
+
get metadata() {
|
|
5501
|
+
return this._metadata;
|
|
5502
|
+
}
|
|
5503
|
+
/**
|
|
5504
|
+
* Whether `close()` has been called.
|
|
5505
|
+
*
|
|
5506
|
+
* @return {boolean} True once it has.
|
|
5507
|
+
*/
|
|
5508
|
+
get closed() {
|
|
5509
|
+
return this._closed;
|
|
5510
|
+
}
|
|
5511
|
+
/**
|
|
5512
|
+
* What a read of this file would cost, and whether it fits - with no I/O at all.
|
|
5513
|
+
*
|
|
5514
|
+
* The same answer `QvdDataFrame.checkRead()` gives, from the header this file already holds, so a
|
|
5515
|
+
* viewer can size a page before asking for it without touching the disk.
|
|
5516
|
+
*
|
|
5517
|
+
* @param {{offset?: number, limit?: number|null, maxRows?: number|null, fields?: Array<string>|null,
|
|
5518
|
+
* as?: 'rows'|'columns', chunkSize?: number|null}} [options] The read being asked about - the same
|
|
5519
|
+
* bag `rows()` takes, plus `as` and `chunkSize` to say which shape of read it is.
|
|
5520
|
+
* @return {any} The answer - `fits`, `reason`, `estimate`, `budget`, `exact`, `suggestions`.
|
|
5521
|
+
* @throws {QvdValidationError} If the file is closed, or an option's value is not valid.
|
|
5522
|
+
*/
|
|
5523
|
+
check(options = {}) {
|
|
5524
|
+
this._refuseWhenClosed("check");
|
|
5525
|
+
const { as = "rows", chunkSize = null } = options;
|
|
5526
|
+
if (as !== "rows" && as !== "columns") {
|
|
5527
|
+
throw new exports.QvdValidationError("as must be 'rows' or 'columns'", {
|
|
5528
|
+
provided: as,
|
|
5529
|
+
reason: "option",
|
|
5530
|
+
option: "as",
|
|
5531
|
+
value: as,
|
|
5532
|
+
file: this._options.path
|
|
5533
|
+
});
|
|
5534
|
+
}
|
|
5535
|
+
if (chunkSize !== null) {
|
|
5536
|
+
requireChunkSize(chunkSize, this._options.path);
|
|
5537
|
+
}
|
|
5538
|
+
const reader = this._readers[as] ?? this._reader;
|
|
5539
|
+
return reader.checkParsed(normaliseWindow(windowFrom(options), this._options.path), {
|
|
5540
|
+
chunkSize,
|
|
5541
|
+
// Resolved here rather than left to the reader, exactly as `_page` resolves it. A warm reader is
|
|
5542
|
+
// still holding the last page's selection, and `checkParsed` falls back to it - so a `check()`
|
|
5543
|
+
// naming no fields answered for whatever the previous page happened to name. On a four-column
|
|
5544
|
+
// file after a page naming one of them, it reported 27,490 bytes for a read that costs 108,160:
|
|
5545
|
+
// understating, which is the direction that approves a read the read then refuses.
|
|
5546
|
+
fields: options.fields === void 0 ? this._options.fields ?? null : options.fields,
|
|
5547
|
+
materialisesRows: as === "rows"
|
|
5548
|
+
});
|
|
5549
|
+
}
|
|
5550
|
+
/**
|
|
5551
|
+
* Reads a page of rows.
|
|
5552
|
+
*
|
|
5553
|
+
* @param {{offset?: number, limit?: number|null, maxRows?: number|null, fields?: Array<string>|null,
|
|
5554
|
+
* onProgress?: Function, signal?: AbortSignal}} [options] The page, and how to read it. One bag,
|
|
5555
|
+
* as every other entry point takes: `offset` and `limit` say which rows, `fields` names a projection
|
|
5556
|
+
* for this page alone, and anything left out falls back to what the file was opened with.
|
|
5557
|
+
* @return {Promise<any>} The page, as a `QvdDataFrame`.
|
|
5558
|
+
* @throws {QvdValidationError} If the file is closed.
|
|
5559
|
+
*/
|
|
5560
|
+
async rows(options = {}) {
|
|
5561
|
+
this._refuseWhenClosed("rows");
|
|
5562
|
+
return await this._serialised(async () => await this._page(options, true, (reader, window) => reader.load(window)));
|
|
5563
|
+
}
|
|
5564
|
+
/**
|
|
5565
|
+
* Reads a page as columns, building no rows.
|
|
5566
|
+
*
|
|
5567
|
+
* @param {{offset?: number, limit?: number|null, maxRows?: number|null, fields?: Array<string>|null,
|
|
5568
|
+
* onProgress?: Function, signal?: AbortSignal}} [options] The page, as `rows()` takes it.
|
|
5569
|
+
* @return {Promise<any>} The page, as a `QvdColumnTable`.
|
|
5570
|
+
* @throws {QvdValidationError} If the file is closed.
|
|
5571
|
+
*/
|
|
5572
|
+
async columns(options = {}) {
|
|
5573
|
+
this._refuseWhenClosed("columns");
|
|
5574
|
+
return await this._serialised(
|
|
5575
|
+
async () => await this._page(options, false, (reader, window) => reader.loadColumnar(window))
|
|
5576
|
+
);
|
|
5577
|
+
}
|
|
5578
|
+
/**
|
|
5579
|
+
* Closes the file.
|
|
5580
|
+
*
|
|
5581
|
+
* Every call after it is refused with `reason: 'closed'`. Calling it twice is not an error: a
|
|
5582
|
+
* `finally` that closes and an `await using` that closes are both right, and both may run.
|
|
5583
|
+
*
|
|
5584
|
+
* @return {Promise<void>} When the pages already in flight have finished.
|
|
5585
|
+
*/
|
|
5586
|
+
async close() {
|
|
5587
|
+
if (this._closed) {
|
|
5588
|
+
return;
|
|
5589
|
+
}
|
|
5590
|
+
this._closed = true;
|
|
5591
|
+
await this._tail;
|
|
5592
|
+
for (const reader of Object.values(this._readers)) {
|
|
5593
|
+
reader?.endPaging();
|
|
5594
|
+
}
|
|
5595
|
+
this._readers = { rows: null, columns: null };
|
|
5596
|
+
this._reader.endPaging();
|
|
5597
|
+
this._reader = null;
|
|
5598
|
+
}
|
|
5599
|
+
/**
|
|
5600
|
+
* `await using` support, where the runtime has it.
|
|
5601
|
+
*
|
|
5602
|
+
* @return {Promise<void>} When closed.
|
|
5603
|
+
*/
|
|
5604
|
+
async [Symbol.asyncDispose]() {
|
|
5605
|
+
await this.close();
|
|
5606
|
+
}
|
|
5607
|
+
/**
|
|
5608
|
+
* Refuses a call on a closed file, in the vocabulary the rest of the API uses.
|
|
5609
|
+
*
|
|
5610
|
+
* @param {string} call The method the caller reached for, for the error.
|
|
5611
|
+
* @private
|
|
5612
|
+
*/
|
|
5613
|
+
_refuseWhenClosed(call) {
|
|
5614
|
+
if (this._closed) {
|
|
5615
|
+
throw new exports.QvdValidationError("The file is closed: open it again to read from it", {
|
|
5616
|
+
reason: "closed",
|
|
5617
|
+
call,
|
|
5618
|
+
file: this._options.path
|
|
5619
|
+
});
|
|
5620
|
+
}
|
|
5621
|
+
}
|
|
5622
|
+
/**
|
|
5623
|
+
* Runs `work` after everything asked for before it, and before everything asked for after.
|
|
5624
|
+
*
|
|
5625
|
+
* @param {() => Promise<any>} work The page to read.
|
|
5626
|
+
* @return {Promise<any>} Its result.
|
|
5627
|
+
* @private
|
|
5628
|
+
*/
|
|
5629
|
+
async _serialised(work) {
|
|
5630
|
+
const run = this._tail.then(work, work);
|
|
5631
|
+
this._tail = run.then(
|
|
5632
|
+
() => void 0,
|
|
5633
|
+
() => void 0
|
|
5634
|
+
);
|
|
5635
|
+
return await run;
|
|
5636
|
+
}
|
|
5637
|
+
/**
|
|
5638
|
+
* Reads one page, through the reader that keeps what the pages before it decoded.
|
|
5639
|
+
*
|
|
5640
|
+
* One reader for every page rather than one per page, which is what makes the symbol cache possible:
|
|
5641
|
+
* decoding the symbols is 72% of a page of a hundred rows from a 300,000-row file, and a reader built
|
|
5642
|
+
* fresh each time did all of it again. Two readers, because a columnar page builds no rows and a row
|
|
5643
|
+
* page does, and `materialisesRows` is fixed when a reader is constructed - so each shape keeps its
|
|
5644
|
+
* own, and its own cache.
|
|
5645
|
+
*
|
|
5646
|
+
* Options that belong to one call rather than to the file - the fields this page alone wants, and the
|
|
5647
|
+
* `onProgress` and `signal` watching it - are told to that shared reader for the next read and no
|
|
5648
|
+
* further. A page naming one of them used to build a reader of its own instead, which quietly turned
|
|
5649
|
+
* the cache off and the two-pass symbol path on: watching a page changed what the page did.
|
|
5650
|
+
*
|
|
5651
|
+
* @param {any} options What the call passed.
|
|
5652
|
+
* @param {boolean} builds Whether the page materialises rows.
|
|
5653
|
+
* @param {(reader: any, window: any) => Promise<any>} read The read to make.
|
|
5654
|
+
* @return {Promise<any>} The page.
|
|
5655
|
+
* @private
|
|
5656
|
+
*/
|
|
5657
|
+
async _page(options, builds, read) {
|
|
5658
|
+
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
5659
|
+
const kept = builds ? "rows" : "columns";
|
|
5660
|
+
if (!this._readers[kept]) {
|
|
5661
|
+
const reader2 = new QvdFileReader2(this._options.path, {
|
|
5662
|
+
...readerOptionsFrom(this._options),
|
|
5663
|
+
materialisesRows: builds
|
|
5664
|
+
});
|
|
5665
|
+
reader2.beginPaging();
|
|
5666
|
+
this._readers[kept] = reader2;
|
|
5667
|
+
}
|
|
5668
|
+
const reader = this._readers[kept];
|
|
5669
|
+
reader.selectForNextRead(options.fields === void 0 ? this._options.fields ?? null : options.fields);
|
|
5670
|
+
reader.observeNextRead({
|
|
5671
|
+
onProgress: options.onProgress ?? this._options.onProgress,
|
|
5672
|
+
signal: options.signal ?? this._options.signal
|
|
5673
|
+
});
|
|
5674
|
+
return await read(reader, windowFrom(options));
|
|
5675
|
+
}
|
|
5676
|
+
};
|
|
5677
|
+
}
|
|
5678
|
+
});
|
|
5679
|
+
|
|
5071
5680
|
// src/QvdDataFrame.js
|
|
5072
5681
|
function defaultFieldHeader(fieldName) {
|
|
5073
5682
|
return {
|
|
@@ -5878,6 +6487,49 @@ var init_QvdDataFrame = __esm({
|
|
|
5878
6487
|
const reader = new QvdFileReader2(path5, { ...readerOptionsFrom(options), materialisesRows: as === "rows" });
|
|
5879
6488
|
return await reader.checkRead(windowFrom(options), { chunkSize });
|
|
5880
6489
|
}
|
|
6490
|
+
/**
|
|
6491
|
+
* Opens a QVD file for paging, reading its header and nothing else.
|
|
6492
|
+
*
|
|
6493
|
+
* Every other entry point is one read from start to finish. A viewer showing a hundred rows at a time
|
|
6494
|
+
* pays the header again on every page, and `iterate()` goes forwards only - it cannot jump to row five
|
|
6495
|
+
* million and it cannot go back. This holds the header so that `check()` costs nothing and a page can
|
|
6496
|
+
* be asked for by position.
|
|
6497
|
+
*
|
|
6498
|
+
* ```js
|
|
6499
|
+
* const qvd = await QvdDataFrame.open('sales.qvd', {allowedDir: '/data'});
|
|
6500
|
+
*
|
|
6501
|
+
* qvd.metadata; // read once, when it opened
|
|
6502
|
+
* const answer = qvd.check({offset: 0, limit: 100}); // no I/O at all
|
|
6503
|
+
* const page = await qvd.rows({offset: 5_000_000, limit: 100});
|
|
6504
|
+
* const cols = await qvd.columns({offset: 0, limit: 100, fields: ['Amount']});
|
|
6505
|
+
*
|
|
6506
|
+
* await qvd.close();
|
|
6507
|
+
* ```
|
|
6508
|
+
*
|
|
6509
|
+
* The header is read once, and so is each column: a column decoded for one page is kept for the pages
|
|
6510
|
+
* after it, so the first page costs about what a single read costs and the ones after it are cheap.
|
|
6511
|
+
* What a file has decoded is charged to the memory check, so a page is refused rather than the process
|
|
6512
|
+
* aborting, and `close()` releases it - close a file you have finished with.
|
|
6513
|
+
*
|
|
6514
|
+
* Each page still opens the file, and a first touch still decodes a whole column rather than only as
|
|
6515
|
+
* far as the page needs.
|
|
6516
|
+
*
|
|
6517
|
+
* @param {string} path The QVD file.
|
|
6518
|
+
* @param {object} [options] What `fromQvd()` takes - `allowedDir`, `fields`, `duals`,
|
|
6519
|
+
* `coerceNumericStrings`, `memorySafetyFactor` - describing the file and how its values read. A
|
|
6520
|
+
* window means nothing here: pages carry their own.
|
|
6521
|
+
* @return {Promise<import('./QvdFile.js').QvdFile>} The open file.
|
|
6522
|
+
* @throws {QvdValidationError} If an option's value is not valid, with `context.reason` of `option`.
|
|
6523
|
+
* @throws {QvdCorruptedError} If the header cannot be read, or describes a file this is not.
|
|
6524
|
+
*/
|
|
6525
|
+
static async open(path5, options = {}) {
|
|
6526
|
+
const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
|
|
6527
|
+
const { QvdFile: QvdFile2 } = await Promise.resolve().then(() => (init_QvdFile(), QvdFile_exports));
|
|
6528
|
+
const reader = new QvdFileReader2(path5, readerOptionsFrom(options));
|
|
6529
|
+
reader.beginPaging();
|
|
6530
|
+
await reader.parseHeaderOnly();
|
|
6531
|
+
return new QvdFile2(reader, reader.describeParsed(), { ...options, path: path5 });
|
|
6532
|
+
}
|
|
5881
6533
|
/**
|
|
5882
6534
|
* Constructs a data frame from a dictionary.
|
|
5883
6535
|
*
|
|
@@ -6182,6 +6834,7 @@ __name(dateToQlikSerial, "dateToQlikSerial");
|
|
|
6182
6834
|
// src/index.js
|
|
6183
6835
|
init_QvdDataFrame();
|
|
6184
6836
|
init_QvdColumnTable();
|
|
6837
|
+
init_QvdFile();
|
|
6185
6838
|
init_QvdFileReader();
|
|
6186
6839
|
init_QvdFileWriter();
|
|
6187
6840
|
init_QvdErrors();
|