qvdjs 2.0.5 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import fs2 from 'fs';
2
2
  import path2 from 'path';
3
- import assert3 from 'assert';
3
+ import assert4 from 'assert';
4
4
  import crypto2 from 'crypto';
5
5
  import { setTimeout } from 'timers/promises';
6
6
  import xml from 'xml2js';
@@ -441,7 +441,9 @@ var init_optionTypes = __esm({
441
441
  function requireRowCount(value, name, filePath) {
442
442
  if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
443
443
  throw new QvdValidationError(`${name} must be a non-negative integer`, {
444
+ reason: "option",
444
445
  option: name,
446
+ value,
445
447
  provided: value,
446
448
  type: typeof value,
447
449
  file: filePath
@@ -458,6 +460,9 @@ function normaliseWindow(window, filePath) {
458
460
  }
459
461
  if (typeof window !== "object" || Array.isArray(window)) {
460
462
  throw new QvdValidationError("The row window must be a number, null, or an {offset, limit} object", {
463
+ reason: "option",
464
+ option: "limit",
465
+ value: window,
461
466
  provided: window,
462
467
  type: typeof window,
463
468
  file: filePath
@@ -468,6 +473,9 @@ function normaliseWindow(window, filePath) {
468
473
  const maxRowsGiven = maxRows !== void 0 && maxRows !== null;
469
474
  if (limitGiven && maxRowsGiven) {
470
475
  throw new QvdValidationError("maxRows and limit are two names for the same option; pass one of them, not both", {
476
+ reason: "option",
477
+ option: "limit",
478
+ value: limit,
471
479
  maxRows,
472
480
  limit,
473
481
  file: filePath
@@ -478,6 +486,19 @@ function normaliseWindow(window, filePath) {
478
486
  limit: limitGiven ? requireRowCount(limit, "limit", filePath) : maxRowsGiven ? requireRowCount(maxRows, "maxRows", filePath) : null
479
487
  };
480
488
  }
489
+ function requireChunkSize(chunkSize, filePath) {
490
+ if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
491
+ throw new QvdValidationError("chunkSize must be a positive integer", {
492
+ reason: "option",
493
+ option: "chunkSize",
494
+ value: chunkSize,
495
+ provided: chunkSize,
496
+ type: typeof chunkSize,
497
+ file: filePath
498
+ });
499
+ }
500
+ return chunkSize;
501
+ }
481
502
  function resolveWindow(window, totalRows) {
482
503
  const rows = Number.isSafeInteger(totalRows) && totalRows > 0 ? totalRows : 0;
483
504
  const offset = Math.min(window.offset, rows);
@@ -492,6 +513,9 @@ function selectFields(fields, requested, filePath) {
492
513
  }
493
514
  if (!Array.isArray(requested)) {
494
515
  throw new QvdValidationError("fields must be an array of field names", {
516
+ reason: "option",
517
+ option: "fields",
518
+ value: requested,
495
519
  provided: requested,
496
520
  type: typeof requested,
497
521
  file: filePath
@@ -500,6 +524,9 @@ function selectFields(fields, requested, filePath) {
500
524
  const available = fields.map((field) => field["FieldName"]);
501
525
  if (requested.length === 0) {
502
526
  throw new QvdValidationError("fields must name at least one field", {
527
+ reason: "option",
528
+ option: "fields",
529
+ value: requested,
503
530
  availableColumns: available,
504
531
  file: filePath
505
532
  });
@@ -508,6 +535,9 @@ function selectFields(fields, requested, filePath) {
508
535
  return requested.map((name) => {
509
536
  if (typeof name !== "string") {
510
537
  throw new QvdValidationError("Field names must be strings", {
538
+ reason: "option",
539
+ option: "fields",
540
+ value: name,
511
541
  provided: name,
512
542
  type: typeof name,
513
543
  availableColumns: available,
@@ -516,6 +546,9 @@ function selectFields(fields, requested, filePath) {
516
546
  }
517
547
  if (seen.has(name)) {
518
548
  throw new QvdValidationError(`Field '${name}' is listed twice`, {
549
+ reason: "option",
550
+ option: "fields",
551
+ value: name,
519
552
  column: name,
520
553
  fields: requested,
521
554
  file: filePath
@@ -525,6 +558,9 @@ function selectFields(fields, requested, filePath) {
525
558
  const index = available.indexOf(name);
526
559
  if (index === -1) {
527
560
  throw new QvdValidationError(`Column '${name}' does not exist`, {
561
+ reason: "option",
562
+ option: "fields",
563
+ value: name,
528
564
  column: name,
529
565
  availableColumns: available,
530
566
  file: filePath
@@ -539,7 +575,9 @@ function normaliseDuals(value, filePath) {
539
575
  }
540
576
  if (!DUAL_MODES.includes(value)) {
541
577
  throw new QvdValidationError(`duals must be one of ${DUAL_MODES.map((mode) => `'${mode}'`).join(", ")}`, {
578
+ reason: "option",
542
579
  option: "duals",
580
+ value,
543
581
  provided: value,
544
582
  file: filePath
545
583
  });
@@ -578,6 +616,7 @@ var init_readOptions = __esm({
578
616
  init_optionTypes();
579
617
  __name(requireRowCount, "requireRowCount");
580
618
  __name(normaliseWindow, "normaliseWindow");
619
+ __name(requireChunkSize, "requireChunkSize");
581
620
  __name(resolveWindow, "resolveWindow");
582
621
  __name(selectFields, "selectFields");
583
622
  DUAL_MODES = Object.freeze(["number", "text", "both"]);
@@ -959,7 +998,7 @@ function changedAfterCheck(checked, change) {
959
998
  });
960
999
  }
961
1000
  async function openChecked(checked, purpose, failed, { nofollow = NOFOLLOW } = {}) {
962
- assert3(purpose === "read" || checked.stats !== null, "A rewrite in place is of a file that exists.");
1001
+ assert4(purpose === "read" || checked.stats !== null, "A rewrite in place is of a file that exists.");
963
1002
  const noFollow = checked.onDisk ? nofollow : 0;
964
1003
  const flags = (purpose === "rewrite" ? O_WRONLY : O_RDONLY) | noFollow;
965
1004
  let handle;
@@ -1535,9 +1574,9 @@ var init_QvdFileWriter = __esm({
1535
1574
  * Writes the data to the QVD file.
1536
1575
  */
1537
1576
  async _writeData() {
1538
- assert3(this._header, "The QVD file header has not been parsed.");
1539
- assert3(this._symbolBuffer, "The QVD file symbol table has not been parsed.");
1540
- assert3(this._indexBuffer, "The QVD file index table has not been parsed.");
1577
+ assert4(this._header, "The QVD file header has not been parsed.");
1578
+ assert4(this._symbolBuffer, "The QVD file symbol table has not been parsed.");
1579
+ assert4(this._indexBuffer, "The QVD file index table has not been parsed.");
1541
1580
  this._emitProgress("write", 0, 1);
1542
1581
  const headerBuffer = Buffer.concat([Buffer.from(this._header, "utf-8"), Buffer.from([0])]);
1543
1582
  const failed = rethrowAsIoError(this._path, "write");
@@ -1985,7 +2024,7 @@ var init_QvdFileWriter = __esm({
1985
2024
  const key = keys[slot];
1986
2025
  offset = typeof key === "number" ? writeSymbol(columnBuffer, offset, kinds[slot], key, texts[slot]) : writeSymbol(columnBuffer, offset, kinds[slot], null, key);
1987
2026
  }
1988
- assert3(offset === byteLength, "A column was encoded into a different number of bytes than it was sized for.");
2027
+ assert4(offset === byteLength, "A column was encoded into a different number of bytes than it was sized for.");
1989
2028
  columnBuffers.push(columnBuffer);
1990
2029
  this._symbolTableMetadata?.push([symbolsOffset, byteLength, containsNull[column]]);
1991
2030
  this._symbolCounts?.push(keys.length);
@@ -2024,9 +2063,9 @@ var init_QvdFileWriter = __esm({
2024
2063
  * @private
2025
2064
  */
2026
2065
  _buildIndexTable() {
2027
- assert3(this._symbolCounts, "The QVD file symbol table has not been built.");
2028
- assert3(this._symbolTableMetadata, "The QVD file symbol table metadata has not been built.");
2029
- assert3(this._symbolIndexByValue, "The QVD file symbol index has not been built.");
2066
+ assert4(this._symbolCounts, "The QVD file symbol table has not been built.");
2067
+ assert4(this._symbolTableMetadata, "The QVD file symbol table metadata has not been built.");
2068
+ assert4(this._symbolIndexByValue, "The QVD file symbol index has not been built.");
2030
2069
  this._indexTableMetadata = [];
2031
2070
  const columns = this._df.columns;
2032
2071
  const data = this._df.data;
@@ -2206,20 +2245,57 @@ function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, col
2206
2245
  }
2207
2246
  return low;
2208
2247
  }
2209
- function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null) {
2210
- if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
2211
- throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", { safetyFactor });
2212
- }
2213
- if (safetyFactor === 0) {
2248
+ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null, bytesHeld = null, readBytes = null) {
2249
+ const answer = checkMemory({
2250
+ symbolTableSize,
2251
+ maxRows,
2252
+ totalRows,
2253
+ safetyFactor,
2254
+ columnCount,
2255
+ materialisesRows,
2256
+ live,
2257
+ bytesHeld,
2258
+ readBytes
2259
+ });
2260
+ if (answer.fits) {
2214
2261
  return;
2215
2262
  }
2216
- const budget = getMemoryBudget();
2263
+ const { message, context } = answer.refusal;
2264
+ throw new QvdValidationError(message, {
2265
+ file: filePath,
2266
+ ...context,
2267
+ reason: "memory",
2268
+ check: answerOf(answer)
2269
+ });
2270
+ }
2271
+ function checkMemory({
2272
+ symbolTableSize,
2273
+ maxRows,
2274
+ totalRows,
2275
+ safetyFactor = 0.8,
2276
+ columnCount = 0,
2277
+ materialisesRows = true,
2278
+ live = null,
2279
+ bytesHeld = null,
2280
+ readBytes = null,
2281
+ measured = null
2282
+ }) {
2283
+ if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
2284
+ throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", {
2285
+ safetyFactor,
2286
+ reason: "option",
2287
+ option: "memorySafetyFactor",
2288
+ value: safetyFactor
2289
+ });
2290
+ }
2291
+ const budget = measured ?? getMemoryBudget();
2217
2292
  const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
2218
2293
  const rowsLive = live === null ? null : live.rows;
2219
2294
  const liveRowsPerChunk = live === null ? 1 : live.perChunk;
2220
2295
  const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
2221
2296
  const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows, rowsLive);
2222
- const externalMemory = estimateExternalMemory(liveRows, columnCount);
2297
+ const { held, forRows: heldForRows, forChunk: heldForChunk } = bytesHeld ?? noBytesHeld;
2298
+ const externalMemory = estimateExternalMemory(liveRows, columnCount) + held;
2223
2299
  const bounded = budget.candidates.map((candidate) => {
2224
2300
  const heapOnly = candidate.source === "V8 heap limit";
2225
2301
  return {
@@ -2230,25 +2306,45 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
2230
2306
  bounds: heapOnly ? "the V8 heap" : "the whole process"
2231
2307
  };
2232
2308
  });
2233
- const exceeded = bounded.filter((candidate) => candidate.needs > candidate.allowed);
2234
- const binding = exceeded.reduce(
2235
- (worst, candidate) => candidate.needs / candidate.allowed > worst.needs / worst.allowed ? candidate : worst,
2236
- exceeded[0]
2309
+ if (safetyFactor === 0) {
2310
+ const lowest = budget.candidates.reduce((least, candidate) => candidate.bytes < least.bytes ? candidate : least);
2311
+ return {
2312
+ fits: true,
2313
+ estimate: { heapBytes: heapMemory, externalBytes: externalMemory, readBytes },
2314
+ // The ceilings are still there and still named - what is missing is any measurement against them.
2315
+ // An earlier version reported `bound: 'none'` here, a third value in a two-value vocabulary that a
2316
+ // caller switching on the documented two would fall straight through.
2317
+ budget: {
2318
+ ...budgetOf(budget, { ...lowest, heapOnly: lowest.source === "V8 heap limit" }, 0),
2319
+ allowedBytes: Infinity
2320
+ },
2321
+ exact: symbolTableSize === 0,
2322
+ suggestions: []
2323
+ };
2324
+ }
2325
+ const tightest = bounded.reduce(
2326
+ (worst, candidate) => candidate.needs / candidate.allowed > worst.needs / worst.allowed ? candidate : worst
2237
2327
  );
2328
+ const binding = tightest.needs > tightest.allowed ? tightest : null;
2238
2329
  const heapLimit = getHeapLimit();
2239
2330
  const availableMemory = binding ? binding.bytes : budget.bytes;
2240
2331
  const estimatedMemory = binding ? binding.needs : heapMemory;
2241
2332
  const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
2242
2333
  if (binding) {
2243
2334
  const includeExternal = !binding.heapOnly;
2244
- const recommendedMaxRows = recommendedRowsFor(
2245
- maxAllowedMemory,
2335
+ const rowsHeldBudget = /* @__PURE__ */ __name((rows) => maxAllowedMemory - (includeExternal ? heldForRows(rows) : 0), "rowsHeldBudget");
2336
+ const fitting = /* @__PURE__ */ __name((rows) => recommendedRowsFor(
2337
+ rowsHeldBudget(rows),
2246
2338
  symbolTableSize,
2247
2339
  totalRows,
2248
2340
  columnCount,
2249
2341
  materialisesRows,
2250
2342
  includeExternal
2251
- );
2343
+ ), "fitting");
2344
+ const firstGuess = fitting(liveRows);
2345
+ const over = fitting(firstGuess);
2346
+ const under = fitting(over);
2347
+ const recommendedMaxRows = Math.max(firstGuess, under);
2252
2348
  const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
2253
2349
  const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
2254
2350
  const availableMB = Math.round(maxAllowedMemory / 1024 / 1024);
@@ -2261,15 +2357,20 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
2261
2357
  const observedBreakdown = budget.observed.map((entry) => `${entry.source} ${Math.round(entry.bytes / 1024 / 1024)}MB`).join(", ");
2262
2358
  const containerBound = binding.source === "container memory limit";
2263
2359
  const chunked = rowsLive !== null;
2264
- const recommendedChunk = chunked ? recommendedChunkFor(
2265
- maxAllowedMemory,
2360
+ const chunkHeldBudget = /* @__PURE__ */ __name((rows) => maxAllowedMemory - (includeExternal ? heldForChunk(rows) : 0), "chunkHeldBudget");
2361
+ const chunkFitting = /* @__PURE__ */ __name((rows) => recommendedChunkFor(
2362
+ chunkHeldBudget(rows),
2266
2363
  symbolTableSize,
2267
2364
  maxRows,
2268
2365
  totalRows,
2269
2366
  columnCount,
2270
2367
  liveRowsPerChunk,
2271
2368
  includeExternal
2272
- ) : 0;
2369
+ ), "chunkFitting");
2370
+ const callersChunk = chunked ? Math.max(1, Math.floor(rowsLive / Math.max(1, liveRowsPerChunk))) : 0;
2371
+ const firstChunk = chunked ? chunkFitting(callersChunk) : 0;
2372
+ const overChunk = chunked ? chunkFitting(firstChunk) : 0;
2373
+ const recommendedChunk = chunked ? Math.max(firstChunk, chunkFitting(overChunk)) : 0;
2273
2374
  const knob = chunked ? "chunkSize" : "limit";
2274
2375
  const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
2275
2376
  const nothingFits = recommendedValue === 0;
@@ -2281,31 +2382,75 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
2281
2382
  } else {
2282
2383
  advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${formatCount(recommendedValue)} rows or less), or raise the heap with --max-old-space-size.`;
2283
2384
  }
2284
- throw new QvdValidationError(
2285
- `Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
2286
- {
2287
- file: filePath,
2288
- symbolTableSize,
2289
- symbolTableSizeMB: sizeMB,
2290
- estimatedMemoryMB: estimatedMB,
2291
- availableMemoryMB: availableMB,
2292
- heapLimitMB,
2293
- reportedHeapLimitMB,
2294
- availableRamMB,
2295
- limitingFactor,
2296
- limitingScope,
2297
- memoryBudget: budget.candidates,
2298
- memoryObserved: budget.observed,
2299
- columnCount,
2300
- totalRows,
2301
- maxRows,
2302
- recommendedMaxRows,
2303
- // Only present when a chunk size is what overflowed, so a caller cannot mistake one
2304
- // recommendation for the other.
2305
- ...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
2385
+ const suggestions = [];
2386
+ if (!nothingFits) {
2387
+ suggestions.push({ option: knob, value: recommendedValue });
2388
+ }
2389
+ if (!containerBound) {
2390
+ let needed = Math.ceil(estimatedMemory / safetyFactor / (1024 * 1024));
2391
+ while (Math.max(needed * 1024 * 1024, MINIMUM_BUDGET_BYTES) * safetyFactor < estimatedMemory) {
2392
+ needed += 1;
2306
2393
  }
2307
- );
2394
+ suggestions.push({ nodeOption: "--max-old-space-size", value: needed });
2395
+ }
2396
+ return {
2397
+ fits: false,
2398
+ reason: "memory",
2399
+ estimate: { heapBytes: heapMemory, externalBytes: externalMemory, readBytes },
2400
+ budget: budgetOf(budget, tightest, safetyFactor),
2401
+ // The symbol term is six times the bytes on disk, an overhead measured across files rather than
2402
+ // derived, so any read with symbols in it is an estimate and says so. Only a read that decodes
2403
+ // nothing can be exact.
2404
+ exact: symbolTableSize === 0,
2405
+ suggestions,
2406
+ refusal: {
2407
+ message: `Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
2408
+ context: {
2409
+ symbolTableSize,
2410
+ symbolTableSizeMB: sizeMB,
2411
+ estimatedMemoryMB: estimatedMB,
2412
+ availableMemoryMB: availableMB,
2413
+ heapLimitMB,
2414
+ reportedHeapLimitMB,
2415
+ availableRamMB,
2416
+ limitingFactor,
2417
+ limitingScope,
2418
+ memoryBudget: budget.candidates,
2419
+ memoryObserved: budget.observed,
2420
+ columnCount,
2421
+ totalRows,
2422
+ maxRows,
2423
+ recommendedMaxRows,
2424
+ // Only present when a chunk size is what overflowed, so a caller cannot mistake one
2425
+ // recommendation for the other.
2426
+ ...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
2427
+ }
2428
+ }
2429
+ };
2308
2430
  }
2431
+ return {
2432
+ fits: true,
2433
+ estimate: { heapBytes: heapMemory, externalBytes: externalMemory, readBytes },
2434
+ budget: budgetOf(budget, tightest, safetyFactor),
2435
+ exact: symbolTableSize === 0,
2436
+ suggestions: []
2437
+ };
2438
+ }
2439
+ function answerOf(answer) {
2440
+ const { refusal, ...rest } = answer;
2441
+ return rest;
2442
+ }
2443
+ function budgetOf(budget, tightest, safetyFactor) {
2444
+ const processLimit = budget.candidates.find((candidate) => candidate.source !== "V8 heap limit");
2445
+ return {
2446
+ heapBytes: usableOldSpaceLimit(),
2447
+ processBytes: processLimit ? processLimit.bytes : null,
2448
+ bound: tightest.heapOnly ? "heap" : "process",
2449
+ safetyFactor,
2450
+ allowedBytes: tightest.allowed ?? tightest.bytes * safetyFactor,
2451
+ candidates: budget.candidates,
2452
+ observed: budget.observed
2453
+ };
2309
2454
  }
2310
2455
  function formatCount(value) {
2311
2456
  return value.toLocaleString("en-US");
@@ -2327,7 +2472,7 @@ function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount =
2327
2472
  `\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). This read materialises ${formatCount(rowsToLoad)} of ${formatCount(totalRows)} rows and will use ~${estimatedMB}MB RAM. Reading fewer rows - with limit, maxRows, or a narrower offset window - lowers the row cost, though the symbol table is read in full either way.`
2328
2473
  );
2329
2474
  }
2330
- var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES;
2475
+ var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES, noBytesHeld;
2331
2476
  var init_memoryUtils = __esm({
2332
2477
  "src/util/memoryUtils.js"() {
2333
2478
  init_QvdErrors();
@@ -2345,7 +2490,11 @@ var init_memoryUtils = __esm({
2345
2490
  __name(estimateMemoryUsage, "estimateMemoryUsage");
2346
2491
  __name(recommendedRowsFor, "recommendedRowsFor");
2347
2492
  __name(recommendedChunkFor, "recommendedChunkFor");
2493
+ noBytesHeld = Object.freeze({ held: 0, forRows: /* @__PURE__ */ __name(() => 0, "forRows"), forChunk: /* @__PURE__ */ __name(() => 0, "forChunk") });
2348
2494
  __name(validateMemoryAvailability, "validateMemoryAvailability");
2495
+ __name(checkMemory, "checkMemory");
2496
+ __name(answerOf, "answerOf");
2497
+ __name(budgetOf, "budgetOf");
2349
2498
  __name(formatCount, "formatCount");
2350
2499
  __name(warnLargeSymbolTable, "warnLargeSymbolTable");
2351
2500
  }
@@ -2464,6 +2613,7 @@ function validateSymbolTableSize(symbolTableLength, filePath, totalRows) {
2464
2613
  function validateFieldMetadata(field, symbolBufferLength, filePath) {
2465
2614
  const symbolsOffset = headerInteger(field["Offset"]);
2466
2615
  const symbolsLength = headerInteger(field["Length"]);
2616
+ const symbolCount = headerInteger(field["NoOfSymbols"]);
2467
2617
  if (isNaN(symbolsOffset) || !Number.isSafeInteger(symbolsOffset) || symbolsOffset < 0) {
2468
2618
  throw new QvdCorruptedError("Invalid symbol offset", {
2469
2619
  field: field["FieldName"],
@@ -2480,6 +2630,14 @@ function validateFieldMetadata(field, symbolBufferLength, filePath) {
2480
2630
  stage: "parseSymbolTable"
2481
2631
  });
2482
2632
  }
2633
+ if (isNaN(symbolCount) || !Number.isSafeInteger(symbolCount) || symbolCount < 0) {
2634
+ throw new QvdCorruptedError("Invalid symbol count", {
2635
+ field: field["FieldName"],
2636
+ noOfSymbols: symbolCount,
2637
+ file: filePath,
2638
+ stage: "parseSymbolTable"
2639
+ });
2640
+ }
2483
2641
  if (symbolsOffset + symbolsLength > symbolBufferLength) {
2484
2642
  throw new QvdCorruptedError("Symbol data extends beyond buffer", {
2485
2643
  field: field["FieldName"],
@@ -2735,12 +2893,25 @@ var init_validationUtils = __esm({
2735
2893
  __name(validateFieldBitMetadata, "validateFieldBitMetadata");
2736
2894
  }
2737
2895
  });
2738
-
2739
- // src/util/symbolParser.js
2740
- function textEnd(symbolBuffer, from, kind, fieldName, filePath) {
2741
- const bufferLength = symbolBuffer.length;
2742
- const found = symbolBuffer.indexOf(0, from);
2743
- if ((found === -1 ? bufferLength : found) - from > MAX_TEXT_BYTES) {
2896
+ function nulFinder(area, { reach, rebaseAfter }) {
2897
+ assert4(reach - rebaseAfter > MAX_TEXT_BYTES, "A text search view must hold the longest text a symbol may have.");
2898
+ if (area.length <= reach) {
2899
+ return (from) => area.indexOf(0, from);
2900
+ }
2901
+ let base = 0;
2902
+ let view = area.subarray(0, reach);
2903
+ return (from) => {
2904
+ if (from - base > rebaseAfter) {
2905
+ base = from;
2906
+ view = area.subarray(base, Math.min(area.length, base + reach));
2907
+ }
2908
+ const found = view.indexOf(0, from - base);
2909
+ return found === -1 ? -1 : base + found;
2910
+ };
2911
+ }
2912
+ function textEnd(findNul, areaEnd, from, kind, fieldName, filePath, base) {
2913
+ const found = findNul(from);
2914
+ if ((found === -1 ? areaEnd : found) - from > MAX_TEXT_BYTES) {
2744
2915
  throw new QvdCorruptedError(`${kind} exceeds maximum length`, {
2745
2916
  field: fieldName,
2746
2917
  maxLength: MAX_TEXT_BYTES,
@@ -2751,58 +2922,59 @@ function textEnd(symbolBuffer, from, kind, fieldName, filePath) {
2751
2922
  if (found === -1) {
2752
2923
  throw new QvdCorruptedError(`${kind} not null-terminated`, {
2753
2924
  field: fieldName,
2754
- pointer: bufferLength,
2755
- bufferSize: bufferLength,
2925
+ pointer: base + from,
2926
+ areaEnd: base + areaEnd,
2756
2927
  file: filePath,
2757
2928
  stage: "parseSymbolTable"
2758
2929
  });
2759
2930
  }
2760
2931
  return found;
2761
2932
  }
2762
- function overflow(message, pointer, bufferLength, fieldName, filePath) {
2933
+ function overflow(message, pointer, areaEnd, fieldName, filePath, base) {
2763
2934
  throw new QvdCorruptedError(message, {
2764
2935
  field: fieldName,
2765
- pointer,
2766
- bufferSize: bufferLength,
2936
+ pointer: base + pointer,
2937
+ areaEnd: base + areaEnd,
2767
2938
  file: filePath,
2768
2939
  stage: "parseSymbolTable"
2769
2940
  });
2770
2941
  }
2771
- function parseFieldSymbols(symbolBuffer, start, end, keep, fieldName, filePath) {
2772
- const bufferLength = symbolBuffer.length;
2942
+ function parseFieldSymbols(symbolBuffer, start, end, symbolCount, keep, fieldName, filePath, search = TEXT_SEARCH, base = 0) {
2943
+ const area = symbolBuffer.subarray(0, end);
2944
+ const findNul = nulFinder(area, search);
2773
2945
  const numbers = [];
2774
2946
  const texts = [];
2775
2947
  let pointer = start;
2776
2948
  while (pointer < end) {
2777
- const typeByte = symbolBuffer[pointer++];
2949
+ const typeByte = area[pointer++];
2778
2950
  const decode = keep === null || keep.has(numbers.length);
2779
2951
  let number = null;
2780
2952
  let text = null;
2781
2953
  switch (typeByte) {
2782
2954
  case 1: {
2955
+ if (pointer + 4 > end) {
2956
+ overflow("Buffer overflow reading integer symbol", pointer, end, fieldName, filePath, base);
2957
+ }
2783
2958
  if (decode) {
2784
- if (pointer + 4 > bufferLength) {
2785
- overflow("Buffer overflow reading integer symbol", pointer, bufferLength, fieldName, filePath);
2786
- }
2787
- number = symbolBuffer.readInt32LE(pointer);
2959
+ number = area.readInt32LE(pointer);
2788
2960
  }
2789
2961
  pointer += 4;
2790
2962
  break;
2791
2963
  }
2792
2964
  case 2: {
2965
+ if (pointer + 8 > end) {
2966
+ overflow("Buffer overflow reading double symbol", pointer, end, fieldName, filePath, base);
2967
+ }
2793
2968
  if (decode) {
2794
- if (pointer + 8 > bufferLength) {
2795
- overflow("Buffer overflow reading double symbol", pointer, bufferLength, fieldName, filePath);
2796
- }
2797
- number = symbolBuffer.readDoubleLE(pointer);
2969
+ number = area.readDoubleLE(pointer);
2798
2970
  }
2799
2971
  pointer += 8;
2800
2972
  break;
2801
2973
  }
2802
2974
  case 4: {
2803
- const terminator = textEnd(symbolBuffer, pointer, "String symbol", fieldName, filePath);
2975
+ const terminator = textEnd(findNul, end, pointer, "String symbol", fieldName, filePath, base);
2804
2976
  if (decode) {
2805
- text = symbolBuffer.toString("utf8", pointer, terminator);
2977
+ text = area.toString("utf8", pointer, terminator);
2806
2978
  }
2807
2979
  pointer = terminator + 1;
2808
2980
  break;
@@ -2810,14 +2982,22 @@ function parseFieldSymbols(symbolBuffer, start, end, keep, fieldName, filePath)
2810
2982
  case 5:
2811
2983
  case 6: {
2812
2984
  const numberBytes = typeByte === 5 ? 4 : 8;
2813
- if (pointer + numberBytes > bufferLength) {
2814
- const read = !decode ? "dual symbol" : typeByte === 5 ? "dual integer symbol" : "dual double symbol";
2815
- overflow(`Buffer overflow reading ${read}`, pointer, bufferLength, fieldName, filePath);
2985
+ if (pointer + numberBytes > end) {
2986
+ const read = typeByte === 5 ? "dual integer symbol" : "dual double symbol";
2987
+ overflow(`Buffer overflow reading ${read}`, pointer, end, fieldName, filePath, base);
2816
2988
  }
2817
- const terminator = textEnd(symbolBuffer, pointer + numberBytes, "Dual string symbol", fieldName, filePath);
2989
+ const terminator = textEnd(
2990
+ findNul,
2991
+ end,
2992
+ pointer + numberBytes,
2993
+ "Dual string symbol",
2994
+ fieldName,
2995
+ filePath,
2996
+ base
2997
+ );
2818
2998
  if (decode) {
2819
- number = typeByte === 5 ? symbolBuffer.readInt32LE(pointer) : symbolBuffer.readDoubleLE(pointer);
2820
- text = symbolBuffer.toString("utf8", pointer + numberBytes, terminator);
2999
+ number = typeByte === 5 ? area.readInt32LE(pointer) : area.readDoubleLE(pointer);
3000
+ text = area.toString("utf8", pointer + numberBytes, terminator);
2821
3001
  }
2822
3002
  pointer = terminator + 1;
2823
3003
  break;
@@ -2825,7 +3005,7 @@ function parseFieldSymbols(symbolBuffer, start, end, keep, fieldName, filePath)
2825
3005
  default: {
2826
3006
  throw new QvdParseError("Unknown symbol type byte", {
2827
3007
  typeByte: typeByte.toString(16),
2828
- offset: pointer - 1,
3008
+ offset: base + pointer - 1,
2829
3009
  file: filePath,
2830
3010
  stage: "parseSymbolTable"
2831
3011
  });
@@ -2834,16 +3014,28 @@ function parseFieldSymbols(symbolBuffer, start, end, keep, fieldName, filePath)
2834
3014
  numbers.push(number);
2835
3015
  texts.push(text);
2836
3016
  }
3017
+ assert4(pointer === end, `The symbols of ${fieldName} were walked to byte ${pointer} of an area ending at ${end}.`);
3018
+ if (numbers.length !== symbolCount) {
3019
+ throw new QvdCorruptedError("Symbol count mismatch", {
3020
+ field: fieldName,
3021
+ symbolCount: numbers.length,
3022
+ noOfSymbols: symbolCount,
3023
+ file: filePath,
3024
+ stage: "parseSymbolTable"
3025
+ });
3026
+ }
2837
3027
  return { numbers, texts };
2838
3028
  }
2839
- function countFieldSymbols(symbolBuffer, start, end, fieldName, filePath) {
2840
- return parseFieldSymbols(symbolBuffer, start, end, DECODE_NOTHING, fieldName, filePath).numbers.length;
3029
+ function countFieldSymbols(symbolBuffer, start, end, symbolCount, fieldName, filePath, base = 0) {
3030
+ return parseFieldSymbols(symbolBuffer, start, end, symbolCount, DECODE_NOTHING, fieldName, filePath, void 0, base).numbers.length;
2841
3031
  }
2842
- var MAX_TEXT_BYTES, DECODE_NOTHING;
3032
+ var MAX_TEXT_BYTES, TEXT_SEARCH, DECODE_NOTHING;
2843
3033
  var init_symbolParser = __esm({
2844
3034
  "src/util/symbolParser.js"() {
2845
3035
  init_QvdErrors();
2846
3036
  MAX_TEXT_BYTES = 1048576;
3037
+ TEXT_SEARCH = Object.freeze({ reach: 2 ** 31 - 1, rebaseAfter: 2 ** 30 });
3038
+ __name(nulFinder, "nulFinder");
2847
3039
  __name(textEnd, "textEnd");
2848
3040
  __name(overflow, "overflow");
2849
3041
  __name(parseFieldSymbols, "parseFieldSymbols");
@@ -3387,7 +3579,17 @@ async function parseHeaderXml(text, file, stage) {
3387
3579
  }
3388
3580
  return parsed;
3389
3581
  }
3390
- var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, COUNT_SYMBOLS_PAST, QvdFileReader;
3582
+ function symbolBytesOf(selected, symbolTableLength) {
3583
+ const areaBytes = selected.map((field) => headerInteger(field["Length"]));
3584
+ return areaBytes.every((bytes) => Number.isSafeInteger(bytes) && bytes >= 0) ? Math.min(
3585
+ symbolTableLength,
3586
+ areaBytes.reduce((sum, bytes) => sum + bytes, 0)
3587
+ ) : symbolTableLength;
3588
+ }
3589
+ function readPasses(analysisAhead) {
3590
+ return analysisAhead ? 2 : 1;
3591
+ }
3592
+ var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, SLICE_BYTES, COUNT_SYMBOLS_PAST, QvdFileReader;
3391
3593
  var init_QvdFileReader = __esm({
3392
3594
  "src/QvdFileReader.js"() {
3393
3595
  init_QvdDataFrame();
@@ -3405,9 +3607,12 @@ var init_QvdFileReader = __esm({
3405
3607
  MAX_HEADER_SIZE = 16 * 1024 * 1024;
3406
3608
  READ_CHUNK_SIZE = 512 * 1024 * 1024;
3407
3609
  ANALYSIS_SLICE_ROWS = 65536;
3610
+ SLICE_BYTES = 16 * 1024 * 1024;
3408
3611
  COUNT_SYMBOLS_PAST = 65536;
3409
3612
  __name(chunksFrom, "chunksFrom");
3410
3613
  __name(parseHeaderXml, "parseHeaderXml");
3614
+ __name(symbolBytesOf, "symbolBytesOf");
3615
+ __name(readPasses, "readPasses");
3411
3616
  QvdFileReader = class {
3412
3617
  static {
3413
3618
  __name(this, "QvdFileReader");
@@ -3436,6 +3641,10 @@ var init_QvdFileReader = __esm({
3436
3641
  * above which a lazy load switches to the two-pass filtering path. The default of 50MB is
3437
3642
  * the point where the extra analysis pass pays for itself; lower it to use filtering on
3438
3643
  * smaller files, raise it to keep the simpler single-pass read for longer.
3644
+ * @param {number} [options.sliceBytes=16777216] The most bytes of records a read holds at a time. An
3645
+ * option rather than a constant for the reason `symbolFilteringThreshold` is one: so that a test can
3646
+ * cross the boundaries between slices in a small file. There is no other reason to change it, and it
3647
+ * is not one of the options a read through `QvdDataFrame` or `QvdColumnTable` passes on.
3439
3648
  * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
3440
3649
  * appear. Null reads every field, in file order. An unknown or repeated name is refused.
3441
3650
  * @param {'number'|'text'|'both'} [options.duals='number'] What a dual symbol - a number with the
@@ -3460,6 +3669,7 @@ var init_QvdFileReader = __esm({
3460
3669
  allowedDir,
3461
3670
  memorySafetyFactor = 0.8,
3462
3671
  symbolFilteringThreshold = 50 * 1024 * 1024,
3672
+ sliceBytes = SLICE_BYTES,
3463
3673
  materialisesRows = true,
3464
3674
  fields = null,
3465
3675
  duals,
@@ -3475,6 +3685,14 @@ var init_QvdFileReader = __esm({
3475
3685
  this._coerceNumericStrings = normaliseCoerceNumericStrings(coerceNumericStrings, this._path);
3476
3686
  this._memorySafetyFactor = memorySafetyFactor;
3477
3687
  this._symbolFilteringThreshold = symbolFilteringThreshold;
3688
+ if (!Number.isSafeInteger(sliceBytes) || sliceBytes <= 0) {
3689
+ throw new QvdValidationError("sliceBytes must be a positive integer", {
3690
+ provided: sliceBytes,
3691
+ type: typeof sliceBytes,
3692
+ file: this._path
3693
+ });
3694
+ }
3695
+ this._sliceBytes = sliceBytes;
3478
3696
  if (onProgress !== void 0 && typeof onProgress !== "function") {
3479
3697
  throw new QvdValidationError("onProgress must be a function", {
3480
3698
  provided: onProgress,
@@ -3492,7 +3710,11 @@ var init_QvdFileReader = __esm({
3492
3710
  this._requestedFields = fields === void 0 ? null : fields;
3493
3711
  this._onProgress = onProgress;
3494
3712
  this._signal = signal;
3495
- this._buffer = null;
3713
+ this._headerBuffer = null;
3714
+ this._handle = null;
3715
+ this._failed = null;
3716
+ this._reading = false;
3717
+ this._symbolAreas = null;
3496
3718
  this._headerOffset = null;
3497
3719
  this._symbolTableOffset = null;
3498
3720
  this._indexTableOffset = null;
@@ -3503,7 +3725,6 @@ var init_QvdFileReader = __esm({
3503
3725
  this._symbolTable = null;
3504
3726
  this._indexColumns = null;
3505
3727
  this._rowsDecoded = 0;
3506
- this._bufferFirstRow = 0;
3507
3728
  this._fileSize = null;
3508
3729
  this._headerMatchesFile = false;
3509
3730
  }
@@ -3542,54 +3763,137 @@ var init_QvdFileReader = __esm({
3542
3763
  }
3543
3764
  }
3544
3765
  /**
3545
- * Reads the binary data of the QVD file.
3546
- *
3547
- * A windowed read - anything with `offset`, `limit` or `maxRows` - reads only the bytes it
3548
- * needs, rather than the file. Measured on `chicago_taxi_rides_2016_01.qvd`, 1,705,805 rows
3549
- * over 20 fields: the last thousand rows take 19 ms against 636 ms for the whole file.
3550
- *
3551
- * The saving is in the index table and the rows, not in the symbol table, which is read in
3552
- * full whatever the window because a stored index in any row can address any symbol. So the
3553
- * gain scales with how much of the file is rows: on a file whose bytes are mostly distinct
3554
- * values there is very little to save, which is what `symbolFilteringThreshold` and the
3555
- * two-pass path exist for.
3556
- *
3557
- * Algorithm for a windowed read:
3558
- * 1. Read the file a chunk at a time until the XML header delimiter is found
3559
- * 2. Parse header to determine symbol table and index table locations
3560
- * 3. Calculate bytes needed: header + full symbol table + partial index table
3561
- * 4. Read only those calculated bytes, by position
3562
- * 5. Rest of parsing proceeds normally with limited data
3563
- *
3564
- * WHY THIS APPROACH:
3565
- * - Symbol table must be fully loaded (contains all unique values)
3566
- * - Index table can be partially loaded (only rows we need)
3567
- * - Reading chunks to find the header is efficient for unknown header sizes
3568
- * - Direct byte-range reading for remaining data is fastest
3766
+ * Opens the file and reads its header, and for a read of rows checks that the read can be made.
3767
+ *
3768
+ * Nothing past the header is read here. The selected fields' symbols and the records are read by position
3769
+ * as they are parsed - see `_symbolAreaOf` and `_forEachSlice` - so no read holds the file in one buffer
3770
+ * (#122). A read of rows therefore leaves the file open, and whatever started the read closes it with
3771
+ * `_closeFile` once the last record it needs is decoded; a header-only read closes it here.
3772
+ *
3773
+ * A windowed read - anything with `offset`, `limit` or `maxRows` - reads the header, the symbols of the
3774
+ * fields it selects, and the window's records, and no byte between. Measured on
3775
+ * `chicago_taxi_rides_2016_01.qvd`, 1,705,805 rows over 20 fields: the last thousand rows take 19 ms
3776
+ * against 636 ms for the whole file. A selected field's area is read in full whatever the window,
3777
+ * because a stored index in any row can address any of that field's symbols; the areas of the fields
3778
+ * `fields` leaves out are not read at all. So a window's gain scales with how much of the file is
3779
+ * rows: on a file whose bytes are mostly distinct values of the fields it reads there is very little
3780
+ * to save, which is what `symbolFilteringThreshold` and the two-pass path exist for.
3569
3781
  *
3570
3782
  * All of it goes through one handle, opened once, on the file the containment check approved. See
3571
3783
  * `chunksFrom` and `openChecked` for why a read no longer opens the path more than once.
3572
3784
  *
3573
- * A window with a non-zero `offset` reads two ranges rather than one: the header and symbol
3574
- * table from the front of the file, and the window's records from wherever they sit. The bytes
3575
- * between are never read, which is what makes `{offset: 1_700_000, limit: 100}` on the taxi
3576
- * fixture a 0.4MB read rather than a 38MB one.
3577
- *
3578
3785
  * @param {QvdRowWindow} window The rows to read.
3579
- * @param {boolean} [headerOnly=false] Stop once the XML header has been read, leaving the
3580
- * symbol and index tables on disk. This is the metadata-only path: the header is a few
3581
- * kilobytes whatever the file's size, so reading a schema costs the same for a 40MB file as
3582
- * for a 40GB one.
3786
+ * @param {boolean} [headerOnly=false] Stop once the XML header has been read, and close the file. This
3787
+ * is the metadata-only path: the header is a few kilobytes whatever the file's size, so reading a
3788
+ * schema costs the same for a 40MB file as for a 40GB one.
3583
3789
  * @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
3584
3790
  * that is fewer than the window covers - see `_prepare`.
3585
3791
  * @private
3586
3792
  */
3587
3793
  async _readData(window = { offset: 0, limit: null }, headerOnly = false, liveRows = null) {
3794
+ assert4(this._reading, "A read opens the QVD file only once it has started, through _startRead.");
3795
+ this._symbolTable = null;
3796
+ this._indexColumns = null;
3797
+ this._rowsDecoded = 0;
3588
3798
  this._throwIfAborted();
3589
3799
  this._emitProgress("read", 0, 1);
3590
3800
  const failed = rethrowAsIoError(this._path, "read");
3591
3801
  const handle = await openChecked(checkPath(this._path, this._allowedDir), "read", failed);
3592
- await closeAfter(handle, failed, () => this._readFrom(handle, window, headerOnly, liveRows, failed));
3802
+ if (headerOnly) {
3803
+ await closeAfter(handle, failed, () => this._readFrom(handle, window, true, liveRows, failed));
3804
+ return;
3805
+ }
3806
+ this._handle = handle;
3807
+ this._failed = failed;
3808
+ try {
3809
+ await this._readFrom(handle, window, false, liveRows, failed);
3810
+ } catch (error) {
3811
+ await this._closeFile(true);
3812
+ throw error;
3813
+ }
3814
+ }
3815
+ /**
3816
+ * Starts a read on this reader, refusing it while another is under way.
3817
+ *
3818
+ * A read of rows holds the file, and what it has read of it, on the reader until it ends. A second read
3819
+ * started meanwhile - `load()` while an iteration is suspended, say - would take over that state and
3820
+ * leave the first read's file open. Reads one after another are fine. Called before a read takes charge
3821
+ * of closing the file, so that refusing the second read cannot close the first one's.
3822
+ *
3823
+ * The first thing every read does, and synchronous: the flag is set before the read's first `await`, so
3824
+ * two reads started together - `Promise.all([reader.load(), reader.load()])` - cannot both pass. The check
3825
+ * used to be of the handle, which is set only once the file has opened, and both did: the second read's
3826
+ * handle replaced the first's, which was never closed, and the first read to finish closed the file the
3827
+ * other was still reading.
3828
+ *
3829
+ * @throws {QvdValidationError} If a read is under way.
3830
+ * @private
3831
+ */
3832
+ _startRead() {
3833
+ if (this._reading) {
3834
+ throw new QvdValidationError("The reader is already reading this file: finish that read first", {
3835
+ file: this._path
3836
+ });
3837
+ }
3838
+ this._reading = true;
3839
+ }
3840
+ /**
3841
+ * Ends the read `_startRead` began: closes its file, if it still holds one, and lets the next read start.
3842
+ *
3843
+ * @param {boolean} failing Whether the read is already throwing - see `_closeFile`.
3844
+ * @private
3845
+ */
3846
+ async _endRead(failing) {
3847
+ try {
3848
+ await this._closeFile(failing);
3849
+ } finally {
3850
+ this._reading = false;
3851
+ }
3852
+ }
3853
+ /**
3854
+ * Closes the file a read of rows opened, and drops what it had read of it.
3855
+ *
3856
+ * The rule `closeAfter` follows: a close that fails is reported only when the read succeeded, so it can
3857
+ * never replace the error that says what went wrong. After a successful read it is the only news.
3858
+ *
3859
+ * @param {boolean} failing Whether the read is already throwing.
3860
+ * @private
3861
+ */
3862
+ async _closeFile(failing) {
3863
+ const handle = this._handle;
3864
+ const failed = this._failed;
3865
+ this._handle = null;
3866
+ this._failed = null;
3867
+ this._symbolAreas = null;
3868
+ if (handle === null || failed === null) {
3869
+ return;
3870
+ }
3871
+ if (failing) {
3872
+ await handle.close().catch(() => {
3873
+ });
3874
+ return;
3875
+ }
3876
+ await handle.close().catch(failed);
3877
+ }
3878
+ /**
3879
+ * Runs a read, the only one under way on this reader, and closes its file when it ends, however it ends.
3880
+ *
3881
+ * @template T
3882
+ * @param {() => Promise<T>} read The read, from opening the file to its last record.
3883
+ * @return {Promise<T>} What it returned.
3884
+ * @private
3885
+ */
3886
+ async _closingAfter(read) {
3887
+ this._startRead();
3888
+ let result;
3889
+ try {
3890
+ result = await read();
3891
+ } catch (error) {
3892
+ await this._endRead(true);
3893
+ throw error;
3894
+ }
3895
+ await this._endRead(false);
3896
+ return result;
3593
3897
  }
3594
3898
  /**
3595
3899
  * Reads what `_readData` was asked for, through the handle it opened.
@@ -3650,51 +3954,61 @@ var init_QvdFileReader = __esm({
3650
3954
  const indexTableOffset = symbolTableOffset + symbolTableLength;
3651
3955
  const recordSize = headerInteger(headerObj["QvdTableHeader"]["RecordByteSize"]);
3652
3956
  const totalRows = headerInteger(headerObj["QvdTableHeader"]["NoOfRecords"]);
3653
- if (headerOnly) {
3654
- this._buffer = headerBuffer.subarray(0, headerEndIndex);
3655
- this._emitProgress("read", 1, 1);
3656
- return;
3657
- }
3658
- const columnCount = selectFields(headerFields, this._requestedFields, this._path).length;
3957
+ const { size: fileSize } = await handle.stat().catch(failed);
3958
+ this._fileSize = fileSize;
3959
+ this._headerMatchesFile = false;
3659
3960
  const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
3660
3961
  (value) => Number.isSafeInteger(value) && value >= 0
3661
3962
  );
3662
3963
  if (headerNumbersUsable) {
3663
- const { size: fileSize2 } = await handle.stat().catch(failed);
3664
- this._fileSize = fileSize2;
3665
- this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize2;
3964
+ this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize;
3965
+ }
3966
+ if (headerOnly) {
3967
+ this._headerBuffer = headerBuffer.subarray(0, headerEndIndex);
3968
+ this._emitProgress("read", 1, 1);
3969
+ return;
3666
3970
  }
3971
+ this._headerBuffer = headerBuffer.subarray(0, headerEndIndex);
3972
+ const selected = selectFields(headerFields, this._requestedFields, this._path);
3973
+ const columnCount = selected.length;
3974
+ const symbolBytes = symbolBytesOf(selected, symbolTableLength);
3667
3975
  const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
3668
3976
  const windowRows = resolved.limit;
3669
3977
  if (headerNumbersUsable && this._headerMatchesFile) {
3670
3978
  validateMemoryAvailability(
3671
- symbolTableLength,
3979
+ symbolBytes,
3672
3980
  windowRows,
3673
3981
  totalRows,
3674
3982
  this._path,
3675
3983
  this._memorySafetyFactor,
3676
3984
  columnCount,
3677
3985
  this._materialisesRows,
3678
- liveRows
3986
+ liveRows,
3987
+ this._bytesHeld(
3988
+ symbolBytes,
3989
+ windowRows,
3990
+ recordSize,
3991
+ liveRows,
3992
+ this._analysisWouldRun(window, resolved, totalRows, symbolTableLength)
3993
+ ),
3994
+ // What it reads, which is not what it holds - the records go through one buffer and are not
3995
+ // kept. Carried so that a refusal's `check` says everything the pre-flight would have said.
3996
+ //
3997
+ // Twice over where the symbol-usage pass is still ahead of it: that pass reads the window's
3998
+ // records to find which symbols the rows use, and the decode then reads them again. Counted
3999
+ // once, the figure understated the I/O of exactly the reads that do the most of it.
4000
+ symbolBytes + readPasses(this._analysisWouldRun(window, resolved, totalRows, symbolTableLength)) * windowRows * recordSize
3679
4001
  );
3680
4002
  }
3681
4003
  if (window.offset === 0 && window.limit === null) {
3682
- this._buffer = await handle.readFile().catch(failed);
3683
- this._fileSize = this._buffer.length;
3684
- this._bufferFirstRow = 0;
3685
4004
  this._emitProgress("read", 1, 1);
3686
4005
  return;
3687
4006
  }
3688
4007
  const rowsToLoad = windowRows;
3689
- validateSymbolTableSizeEarly(symbolTableLength, this._path);
4008
+ validateSymbolTableSizeEarly(symbolBytes, this._path);
3690
4009
  validateRecordSize(recordSize, this._path, "readData");
3691
4010
  validateRecordCount(totalRows, this._path, "readData");
3692
- const skippedIndexBytes = resolved.offset * recordSize;
3693
- const indexTableBytesToRead = rowsToLoad * recordSize;
3694
- const totalBytesToRead = indexTableOffset + indexTableBytesToRead;
3695
- const fileBytesRequired = indexTableOffset + skippedIndexBytes + indexTableBytesToRead;
3696
- const { size: fileSize } = await handle.stat().catch(failed);
3697
- this._fileSize = fileSize;
4011
+ const fileBytesRequired = indexTableOffset + (resolved.offset + rowsToLoad) * recordSize;
3698
4012
  if (fileBytesRequired > fileSize) {
3699
4013
  throw new QvdCorruptedError("The file is shorter than its header claims.", {
3700
4014
  file: this._path,
@@ -3703,47 +4017,35 @@ var init_QvdFileReader = __esm({
3703
4017
  stage: "readData"
3704
4018
  });
3705
4019
  }
3706
- this._buffer = Buffer.alloc(totalBytesToRead);
3707
- await this._readRange(handle, 0, indexTableOffset, 0, fileSize, totalBytesToRead);
3708
- if (indexTableBytesToRead > 0) {
3709
- await this._readRange(
3710
- handle,
3711
- indexTableOffset,
3712
- indexTableBytesToRead,
3713
- indexTableOffset + skippedIndexBytes,
3714
- fileSize,
3715
- fileBytesRequired
3716
- );
3717
- }
3718
- this._bufferFirstRow = resolved.offset;
3719
4020
  this._emitProgress("read", 1, 1);
3720
4021
  }
3721
4022
  /**
3722
- * Reads one byte range of the file into the buffer.
4023
+ * Reads one byte range of the open file into a buffer.
3723
4024
  *
3724
4025
  * Read in bounded chunks, checking bytesRead each time. A single fs.read call with a length of
3725
4026
  * 2^31 or more does not throw - it trips a C++ assertion and aborts the whole process, which no
3726
4027
  * try/catch can intercept.
3727
4028
  *
3728
- * @param {import('fs/promises').FileHandle} fd The open file.
3729
- * @param {number} bufferOffset Where in the buffer to write.
4029
+ * @param {Buffer} target The buffer to read into.
4030
+ * @param {number} targetOffset Where in it to write.
3730
4031
  * @param {number} byteCount How many bytes to read.
3731
4032
  * @param {number} filePosition Where in the file to read from.
3732
- * @param {number} fileSize The file's size, for the error.
3733
- * @param {number} requiredBytes Bytes the whole read needs, for the error.
4033
+ * @param {number} requiredBytes How far into the file the read has to reach, for the error.
4034
+ * @throws {QvdCorruptedError} If the file ends before the range does.
3734
4035
  * @private
3735
4036
  */
3736
- async _readRange(fd, bufferOffset, byteCount, filePosition, fileSize, requiredBytes) {
3737
- assert3(this._buffer, "The read buffer has not been allocated.");
3738
- const failed = rethrowAsIoError(this._path, "read");
4037
+ async _readAt(target, targetOffset, byteCount, filePosition, requiredBytes) {
4038
+ assert4(this._handle && this._failed, "The QVD file is not open.");
4039
+ const handle = this._handle;
4040
+ const failed = this._failed;
3739
4041
  let done = 0;
3740
4042
  while (done < byteCount) {
3741
4043
  const length = Math.min(READ_CHUNK_SIZE, byteCount - done);
3742
- const { bytesRead } = await fd.read(this._buffer, bufferOffset + done, length, filePosition + done).catch(failed);
4044
+ const { bytesRead } = await handle.read(target, targetOffset + done, length, filePosition + done).catch(failed);
3743
4045
  if (bytesRead === 0) {
3744
4046
  throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
3745
4047
  file: this._path,
3746
- fileSize,
4048
+ fileSize: this._fileSize,
3747
4049
  // Two numbers, because they stopped being the same one when a window began reading two
3748
4050
  // ranges: `bytesRead` is how much of this range arrived, `filePosition` is where in the
3749
4051
  // file it gave up. Reporting the position under the name of the count made a windowed
@@ -3758,12 +4060,279 @@ var init_QvdFileReader = __esm({
3758
4060
  done += bytesRead;
3759
4061
  }
3760
4062
  }
4063
+ /**
4064
+ * Bytes of the file a read holds outside the heap while it works, beside the codes the guard counts for
4065
+ * itself: the symbol areas it reads, and the one buffer its records come through.
4066
+ *
4067
+ * The areas are counted whole, although the read lets each range go once its fields are parsed, because
4068
+ * two ranges are both live when a field of one is parsed between two fields of the other - the order the
4069
+ * caller asked for the fields decides it, so the sum is what holds in every order. The slice is the
4070
+ * buffer `_forEachSlice` will allocate, sized by `_sliceRowsFor` so that the charge and the allocation
4071
+ * are one expression rather than two that agree today.
4072
+ *
4073
+ * @param {number} symbolBytes Bytes of symbols the read will read.
4074
+ * @param {number} rows Records it will read.
4075
+ * @param {number} recordSize Bytes per record.
4076
+ * @return {number} Bytes.
4077
+ * @private
4078
+ */
4079
+ _bytesHeldBy(symbolBytes, rows, recordSize) {
4080
+ const usable = Number.isSafeInteger(rows) && Number.isSafeInteger(recordSize) && rows > 0 && recordSize > 0;
4081
+ return symbolBytes + (usable ? this._sliceRowsFor(rows, recordSize) * recordSize : 0);
4082
+ }
4083
+ /**
4084
+ * Records the one buffer holds while a read of `rowCount` records goes through it.
4085
+ *
4086
+ * A slice is `sliceBytes` of records, rounded down to a whole record, or every record the read has left
4087
+ * when that is fewer - and at least one, since a read of a record wider than `sliceBytes` still has to
4088
+ * hold that record. The single definition: `_forEachSlice` allocates from it and the memory guard is
4089
+ * charged from it, so a change to how a read slices cannot leave the guard pricing the old rule.
4090
+ *
4091
+ * @param {number} rowCount Records the read will read.
4092
+ * @param {number} recordSize Bytes per record.
4093
+ * @return {number} Records in one slice.
4094
+ * @private
4095
+ */
4096
+ _sliceRowsFor(rowCount, recordSize) {
4097
+ return Math.max(1, Math.min(rowCount, Math.floor(this._sliceBytes / Math.max(1, recordSize))));
4098
+ }
4099
+ /**
4100
+ * What a read holds in bytes of the file, for the memory guard: what it holds now, and what a read
4101
+ * following either piece of advice a refusal can carry would hold instead.
4102
+ *
4103
+ * The two knobs are not the same knob, which is why there are two functions rather than one. A smaller
4104
+ * `limit` is a smaller window, so every record the read touches is one of fewer - the symbol-usage pass
4105
+ * included, since it reads the window. A smaller `chunkSize` leaves the window exactly where it is and
4106
+ * only changes how much of it is decoded at a time, so a read with that pass still ahead of it holds the
4107
+ * window's slice however small the chunk. Priced with `forRows`, such a chunk was charged for the records
4108
+ * of one chunk and then held sixteen megabytes more than that.
4109
+ *
4110
+ * @param {number} symbolBytes Bytes of symbols the read will read.
4111
+ * @param {number} windowRows Rows the read covers.
4112
+ * @param {number} recordSize Bytes per record.
4113
+ * @param {{rows: number, perChunk: number}|null} liveRows Rows held at one instant - see `_prepare`.
4114
+ * @param {boolean} analysisAhead Whether the symbol-usage pass has still to run.
4115
+ * @return {{held: number, forRows: (rows: number) => number, forChunk: (rows: number) => number}} What
4116
+ * this read holds, what a read of so many rows would hold, and what one reading so many rows a chunk
4117
+ * would hold.
4118
+ * @private
4119
+ */
4120
+ _bytesHeld(symbolBytes, windowRows, recordSize, liveRows, analysisAhead) {
4121
+ return {
4122
+ held: this._bytesHeldBy(symbolBytes, this._recordsAtOnce(windowRows, liveRows, analysisAhead), recordSize),
4123
+ // A window of so many rows reads so many records at a time, and the pass that reads it ahead of the
4124
+ // decode reads the same rows, so the buffer is sized from the rows either way.
4125
+ forRows: /* @__PURE__ */ __name((rows) => this._bytesHeldBy(symbolBytes, rows, recordSize), "forRows"),
4126
+ // A chunk of so many rows, over this read's window - which is what `chunkSize` changes and what it
4127
+ // leaves alone. `_recordsAtOnce` is what answers that, given a chunk size as the rows held at once.
4128
+ forChunk: /* @__PURE__ */ __name((rows) => this._bytesHeldBy(symbolBytes, this._recordsAtOnce(windowRows, { rows, perChunk: 1 }, analysisAhead), recordSize), "forChunk")
4129
+ };
4130
+ }
4131
+ /**
4132
+ * Whether a read takes the symbol-usage pass, which reads the window's records before the symbol table
4133
+ * is parsed and so before the first chunk is built.
4134
+ *
4135
+ * The one statement of the condition. `_prepare` asks it to decide, and `_readData` asks it before the
4136
+ * header has been parsed, to know what to charge the memory guard: a read with the pass ahead of it
4137
+ * holds a whole slice of records, and a read without it holds only what it reads at a time. Said in two
4138
+ * places, the two would drift and a read would be charged for one path and take the other - which fails
4139
+ * open for a chunked read, and that is the direction the guard exists to prevent.
4140
+ *
4141
+ * Any window that does not cover the whole file is a candidate, which includes one bounded by its offset
4142
+ * rather than by its limit. Covering every row rules it out, however the window was spelled -
4143
+ * `{offset: 0, limit: n}` over all n rows, or an `iterate` of them. Such a read needs every symbol any
4144
+ * row uses, which is what `estimateMemoryUsage` assumes for a full read as well, so the pass has nothing
4145
+ * to filter. It is not free: since the records are read as they are decoded rather than held in one
4146
+ * buffer, the pass reads the window's records and the decode then reads them again. A window that really
4147
+ * is a window pays that for the symbols it saves parsing; a window that is a full read in disguise paid
4148
+ * it for nothing.
4149
+ *
4150
+ * The threshold is an option rather than a constant so this path can be exercised with a small fixture:
4151
+ * it is the most intricate code in the reader, and the only files large enough to reach the 50MB default
4152
+ * are ones no repository should be carrying around. It measures the whole table rather than the areas a
4153
+ * read selects, because what the pass saves is parsing work across the table.
4154
+ *
4155
+ * @param {QvdRowWindow} window The window as the caller spelled it.
4156
+ * @param {{offset: number, limit: number}} resolved Where it lands in this file.
4157
+ * @param {number} totalRows Rows the file declares.
4158
+ * @param {number} symbolTableLength The symbol table's declared length.
4159
+ * @return {boolean} Whether the pass will run.
4160
+ * @private
4161
+ */
4162
+ _analysisWouldRun(window, resolved, totalRows, symbolTableLength) {
4163
+ return resolved.limit < totalRows && (window.limit !== null || window.offset > 0) && symbolTableLength > this._symbolFilteringThreshold;
4164
+ }
4165
+ /**
4166
+ * Records a read holds at one time, which is what its record buffer is sized from.
4167
+ *
4168
+ * A slice holds `sliceBytes` of records, or every record the read has left to read when that is fewer -
4169
+ * so what it costs depends on how many a read asks for at a time, not on how many it covers. An
4170
+ * iteration asks for a chunk: `iterate({limit: 20_000_000, chunkSize: 1000})` reads a thousand records at
4171
+ * a time however many its window covers, and charging it a full slice would refuse it for 16 MiB it never
4172
+ * allocates. The symbol-usage pass is the exception, because it reads the whole window in slices of its
4173
+ * own before the first chunk is built, so a read that still has that pass ahead of it is charged for it.
4174
+ *
4175
+ * @param {number} windowRows Rows the read covers.
4176
+ * @param {{rows: number, perChunk: number}|null} liveRows Rows held at one instant - see `_prepare`.
4177
+ * @param {boolean} analysisAhead Whether the symbol-usage pass has still to run.
4178
+ * @return {number} Records read at one time.
4179
+ * @private
4180
+ */
4181
+ _recordsAtOnce(windowRows, liveRows, analysisAhead) {
4182
+ const chunkRows = liveRows === null ? windowRows : Math.max(1, Math.floor(liveRows.rows / Math.max(1, liveRows.perChunk)));
4183
+ return analysisAhead ? Math.max(windowRows, chunkRows) : chunkRows;
4184
+ }
4185
+ /**
4186
+ * The symbol table's length, as much of it as the file holds: what the header declares, cut short where
4187
+ * the file ends. Known before a byte of the table is read, so everything that can refuse the table is
4188
+ * checked on this, before the table is allocated.
4189
+ *
4190
+ * A file that ends inside its symbol table is measured to where it ends, and the fields whose areas it cut
4191
+ * short are refused as `Symbol data extends beyond buffer` when their metadata is checked - what a
4192
+ * whole-file read has always said of such a file. A window has refused it already, before reading anything.
4193
+ *
4194
+ * @return {number} Bytes.
4195
+ * @private
4196
+ */
4197
+ _symbolTableLength() {
4198
+ assert4(
4199
+ this._symbolTableOffset !== null && this._indexTableOffset !== null && this._fileSize !== null,
4200
+ "The QVD file header has not been parsed before its symbol table was measured."
4201
+ );
4202
+ const declared = this._indexTableOffset - this._symbolTableOffset;
4203
+ return Math.max(0, Math.min(declared, this._fileSize - this._symbolTableOffset));
4204
+ }
4205
+ /**
4206
+ * Where each selected field's symbols are, as ranges of the symbol table this read will read.
4207
+ *
4208
+ * A field's `Offset` and `Length` say exactly where its symbols are, so a read of some of a file's fields
4209
+ * has no reason to read the areas of the rest (#122). Qlik writes the areas one after another in field
4210
+ * order, so ranges that touch are merged: a read of every field is one range, and so is a read of fields
4211
+ * that happen to be neighbours. A read of one field of twenty reads that field's area alone.
4212
+ *
4213
+ * Built once per read, from the fields the read selected, and each field's metadata is checked as it is
4214
+ * added - a range is arithmetic on `Offset` and `Length`, and those have to be inside the table first.
4215
+ * `_parseSymbolTable` checks every field of the file, selected or not, before it parses any.
4216
+ *
4217
+ * @return {{ranges: Array<{start: number, end: number, fields: number, buffer: Buffer|null}>,
4218
+ * byField: Map<any, {range: {start: number, end: number, fields: number, buffer: Buffer|null},
4219
+ * start: number, end: number}>}} The ranges, and where in its range each field's area sits.
4220
+ * @private
4221
+ */
4222
+ _symbolAreaPlan() {
4223
+ if (this._symbolAreas !== null) {
4224
+ return this._symbolAreas;
4225
+ }
4226
+ assert4(this._selectedFields, "The QVD file fields have not been resolved before their symbols were read.");
4227
+ const tableLength = this._symbolTableLength();
4228
+ const areas = this._selectedFields.map((field) => {
4229
+ validateFieldMetadata(field, tableLength, this._path);
4230
+ const start = headerInteger(field["Offset"]);
4231
+ return { field, start, end: start + headerInteger(field["Length"]) };
4232
+ });
4233
+ const ranges = [];
4234
+ const byField = /* @__PURE__ */ new Map();
4235
+ for (const area of [...areas].sort((a, b) => a.start - b.start)) {
4236
+ const last = ranges.at(-1);
4237
+ const range = last !== void 0 && area.start <= last.end ? last : { start: area.start, end: area.end, fields: 0, buffer: null };
4238
+ if (range !== last) {
4239
+ ranges.push(range);
4240
+ }
4241
+ range.end = Math.max(range.end, area.end);
4242
+ range.fields += 1;
4243
+ byField.set(area.field, { range, start: area.start, end: area.end });
4244
+ }
4245
+ this._symbolAreas = { ranges, byField };
4246
+ return this._symbolAreas;
4247
+ }
4248
+ /**
4249
+ * One field's symbols, as bytes: the range that holds them, read from the open file the first time a field
4250
+ * of that range needs it.
4251
+ *
4252
+ * @param {any} field The field, one this read selected.
4253
+ * @return {Promise<{buffer: Buffer, start: number, end: number, base: number}>} Its area, as a range of
4254
+ * `buffer`, with where that buffer starts in the symbol table - what an error adds back to say where a
4255
+ * damaged symbol is in the file, rather than where it is in the bytes this read happened to read.
4256
+ * @throws {QvdValidationError} If the range is larger than half the heap.
4257
+ * @private
4258
+ */
4259
+ async _symbolAreaOf(field) {
4260
+ assert4(
4261
+ this._header && this._symbolTableOffset !== null,
4262
+ "The QVD file header has not been parsed before its symbols were read."
4263
+ );
4264
+ const area = this._symbolAreaPlan().byField.get(field);
4265
+ assert4(area, "A field this read did not select has no symbol area.");
4266
+ const { range } = area;
4267
+ if (range.buffer === null) {
4268
+ const length = range.end - range.start;
4269
+ validateSymbolTableSize(length, this._path, headerInteger(this._header["QvdTableHeader"]["NoOfRecords"]));
4270
+ const buffer = Buffer.alloc(length);
4271
+ const from = this._symbolTableOffset + range.start;
4272
+ await this._readAt(buffer, 0, length, from, from + length);
4273
+ range.buffer = buffer;
4274
+ }
4275
+ return { buffer: range.buffer, start: area.start - range.start, end: area.end - range.start, base: range.start };
4276
+ }
4277
+ /**
4278
+ * Lets go of a field's symbols once they are parsed, and of the bytes of its range once every field in it
4279
+ * has been.
4280
+ *
4281
+ * Every text is copied out of the bytes as it is decoded, so what the read keeps is the values. A read of
4282
+ * one field of a file whose other fields are large therefore holds that field's bytes and no others.
4283
+ *
4284
+ * @param {any} field The field whose symbols are parsed.
4285
+ * @private
4286
+ */
4287
+ _releaseSymbolArea(field) {
4288
+ const area = this._symbolAreaPlan().byField.get(field);
4289
+ assert4(area, "A field this read did not select has no symbol area.");
4290
+ area.range.fields -= 1;
4291
+ if (area.range.fields === 0) {
4292
+ area.range.buffer = null;
4293
+ }
4294
+ }
4295
+ /**
4296
+ * Reads records from the open file a slice at a time, and hands each slice to `visit`.
4297
+ *
4298
+ * One buffer of at most `sliceBytes` holds a slice, and is reused for the next one, so a read of any
4299
+ * number of records holds that much of them and no more. Cancellation is checked before each slice.
4300
+ *
4301
+ * @param {number} firstRow The file row of the first record.
4302
+ * @param {number} rowCount How many records.
4303
+ * @param {number} recordSize Bytes per record.
4304
+ * @param {(slice: Buffer, done: number, count: number) => void|Promise<void>} visit Called with each
4305
+ * slice's records, how many records came before it, and how many it holds.
4306
+ * @private
4307
+ */
4308
+ async _forEachSlice(firstRow, rowCount, recordSize, visit) {
4309
+ if (rowCount === 0) {
4310
+ return;
4311
+ }
4312
+ assert4(this._indexTableOffset !== null, "The QVD file header has not been parsed before its records were read.");
4313
+ const sliceRows = this._sliceRowsFor(rowCount, recordSize);
4314
+ const slice = Buffer.alloc(sliceRows * recordSize);
4315
+ const requiredBytes = this._indexTableOffset + (firstRow + rowCount) * recordSize;
4316
+ for (let done = 0; done < rowCount; done += sliceRows) {
4317
+ this._throwIfAborted();
4318
+ const count = Math.min(sliceRows, rowCount - done);
4319
+ const records = slice.subarray(0, count * recordSize);
4320
+ await this._readAt(
4321
+ records,
4322
+ 0,
4323
+ records.length,
4324
+ this._indexTableOffset + (firstRow + done) * recordSize,
4325
+ requiredBytes
4326
+ );
4327
+ await visit(records, done, count);
4328
+ }
4329
+ }
3761
4330
  /**
3762
4331
  * Parses the XML header of the QVD file. This method is part of the parsing process
3763
4332
  * and should not be called directly.
3764
4333
  */
3765
4334
  async _parseHeader() {
3766
- if (!this._buffer) {
4335
+ if (!this._headerBuffer) {
3767
4336
  throw new QvdCorruptedError(
3768
4337
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
3769
4338
  {
@@ -3774,7 +4343,7 @@ var init_QvdFileReader = __esm({
3774
4343
  }
3775
4344
  const HEADER_DELIMITER = "\r\n\0";
3776
4345
  const headerBeginIndex = 0;
3777
- const headerDelimiterIndex = this._buffer.indexOf(HEADER_DELIMITER, headerBeginIndex);
4346
+ const headerDelimiterIndex = this._headerBuffer.indexOf(HEADER_DELIMITER, headerBeginIndex);
3778
4347
  if (headerDelimiterIndex === -1) {
3779
4348
  throw new QvdCorruptedError(
3780
4349
  "The XML header section does not exist or is not properly delimited from the binary data.",
@@ -3785,7 +4354,7 @@ var init_QvdFileReader = __esm({
3785
4354
  );
3786
4355
  }
3787
4356
  const headerEndIndex = headerDelimiterIndex + HEADER_DELIMITER.length;
3788
- const headerBuffer = this._buffer.subarray(headerBeginIndex, headerEndIndex);
4357
+ const headerBuffer = this._headerBuffer.subarray(headerBeginIndex, headerEndIndex);
3789
4358
  this._fieldBitMetadataValidated = false;
3790
4359
  this._header = await parseHeaderXml(headerBuffer.toString(), this._path, "parseHeader");
3791
4360
  const fieldList = validateHeaderStructure(this._header, this._path, "parseHeader");
@@ -3807,12 +4376,12 @@ var init_QvdFileReader = __esm({
3807
4376
  * @param {QvdRowWindow} window The rows of interest, as file row indices.
3808
4377
  * @param {string} stage Stage name for any error raised here.
3809
4378
  * @return {{fields: Array<any>, recordSize: number, totalRows: number, rowsToLoad: number,
3810
- * indexBuffer: Buffer, firstRow: number}} The record geometry. `indexBuffer` starts at the
3811
- * window's first record, file row `firstRow`, so the decoder always counts from zero.
4379
+ * firstRow: number}} The record geometry: the window's `rowsToLoad` records start at file row
4380
+ * `firstRow`, and `_forEachSlice` reads them.
3812
4381
  * @private
3813
4382
  */
3814
4383
  _planIndexTable(window, stage) {
3815
- if (!this._buffer || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
4384
+ if (!this._handle || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
3816
4385
  throw new QvdCorruptedError(
3817
4386
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
3818
4387
  {
@@ -3827,22 +4396,18 @@ var init_QvdFileReader = __esm({
3827
4396
  const totalRows = headerInteger(this._header["QvdTableHeader"]["NoOfRecords"]);
3828
4397
  const indexTableLength = headerInteger(this._header["QvdTableHeader"]["Length"]);
3829
4398
  const { offset: firstRow, limit: rowsToLoad } = resolveWindow(window, totalRows);
4399
+ assert4(this._fileSize !== null, "The QVD file has not been measured before its records were planned.");
3830
4400
  validateIndexTableMetadata(
3831
4401
  recordSize,
3832
4402
  totalRows,
3833
4403
  indexTableLength,
3834
4404
  this._indexTableOffset,
3835
- this._buffer.length,
4405
+ this._fileSize,
3836
4406
  rowsToLoad,
3837
4407
  this._path,
3838
4408
  this._fileSize,
3839
4409
  firstRow,
3840
- this._bufferFirstRow
3841
- );
3842
- const bufferRecordStart = (firstRow - this._bufferFirstRow) * recordSize;
3843
- const indexBuffer = this._buffer.subarray(
3844
- this._indexTableOffset + bufferRecordStart,
3845
- this._indexTableOffset + bufferRecordStart + rowsToLoad * recordSize
4410
+ 0
3846
4411
  );
3847
4412
  if (!this._fieldBitMetadataValidated) {
3848
4413
  for (const field of allFields) {
@@ -3851,11 +4416,12 @@ var init_QvdFileReader = __esm({
3851
4416
  validateBitFields(allFields, this._path);
3852
4417
  this._fieldBitMetadataValidated = true;
3853
4418
  }
3854
- assert3(
3855
- rowsToLoad === 0 || recordSize === 0 || Math.floor(indexBuffer.length / recordSize) >= rowsToLoad,
3856
- `The index table holds ${Math.floor(indexBuffer.length / (recordSize || 1))} whole records but ${rowsToLoad} were validated as present.`
4419
+ const windowEnd = this._indexTableOffset + (firstRow + rowsToLoad) * recordSize;
4420
+ assert4(
4421
+ rowsToLoad === 0 || recordSize === 0 || windowEnd <= this._fileSize,
4422
+ `The window's records end at byte ${windowEnd} of a file of ${this._fileSize}, but ${rowsToLoad} were validated as present.`
3857
4423
  );
3858
- return { fields, recordSize, totalRows, rowsToLoad, indexBuffer, firstRow };
4424
+ return { fields, recordSize, totalRows, rowsToLoad, firstRow };
3859
4425
  }
3860
4426
  /**
3861
4427
  * Analyzes the index table to determine which symbols are actually needed.
@@ -3871,48 +4437,58 @@ var init_QvdFileReader = __esm({
3871
4437
  * @private
3872
4438
  */
3873
4439
  async _analyzeIndexTableSymbolUsage(window) {
3874
- const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
4440
+ const { fields, recordSize, rowsToLoad, firstRow } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
3875
4441
  const symbolUsage = [];
3876
4442
  const sliceRows = Math.min(rowsToLoad, ANALYSIS_SLICE_ROWS);
3877
4443
  const column = new Int32Array(sliceRows);
3878
- fields.forEach((field, position) => {
3879
- this._throwIfAborted();
4444
+ const state = fields.map((field) => {
3880
4445
  const needed = /* @__PURE__ */ new Set();
3881
- symbolUsage[position] = needed;
3882
- const bitOffset = headerInteger(field["BitOffset"]);
3883
- const bitWidth = headerInteger(field["BitWidth"]);
3884
- const bias = headerInteger(field["Bias"]);
4446
+ symbolUsage.push(needed);
3885
4447
  const length = headerInteger(field["Length"]);
3886
- let indexLimit = Number.isSafeInteger(length) && length >= 0 ? Math.ceil(length / 2) : Infinity;
3887
- let counted = false;
3888
- for (let first = 0; first < rowsToLoad; first += sliceRows) {
3889
- const count = Math.min(sliceRows, rowsToLoad - first);
3890
- decodeIndexColumn(
3891
- first === 0 ? indexBuffer : indexBuffer.subarray(first * recordSize),
3892
- recordSize,
3893
- count,
3894
- bitOffset,
3895
- bitWidth,
3896
- bias,
3897
- column
3898
- );
3899
- for (let row = 0; row < count; row++) {
3900
- if (column[row] >= 0 && column[row] < indexLimit) {
3901
- needed.add(column[row]);
4448
+ return {
4449
+ field,
4450
+ needed,
4451
+ bitOffset: headerInteger(field["BitOffset"]),
4452
+ bitWidth: headerInteger(field["BitWidth"]),
4453
+ bias: headerInteger(field["Bias"]),
4454
+ indexLimit: Number.isSafeInteger(length) && length >= 0 ? Math.ceil(length / 2) : Infinity,
4455
+ counted: false
4456
+ };
4457
+ });
4458
+ await this._forEachSlice(firstRow, rowsToLoad, recordSize, async (records, done, recordCount) => {
4459
+ for (let first = 0; first < recordCount; first += sliceRows) {
4460
+ const count = Math.min(sliceRows, recordCount - first);
4461
+ for (const field of state) {
4462
+ decodeIndexColumn(
4463
+ first === 0 ? records : records.subarray(first * recordSize),
4464
+ recordSize,
4465
+ count,
4466
+ field.bitOffset,
4467
+ field.bitWidth,
4468
+ field.bias,
4469
+ column
4470
+ );
4471
+ for (let row = 0; row < count; row++) {
4472
+ if (column[row] >= 0 && column[row] < field.indexLimit) {
4473
+ field.needed.add(column[row]);
4474
+ }
3902
4475
  }
3903
- }
3904
- if (!counted && needed.size > COUNT_SYMBOLS_PAST) {
3905
- counted = true;
3906
- indexLimit = Math.min(indexLimit, this._countFieldSymbols(field));
3907
- for (const index of needed) {
3908
- if (index >= indexLimit) {
3909
- needed.delete(index);
4476
+ if (!field.counted && field.needed.size > COUNT_SYMBOLS_PAST) {
4477
+ field.counted = true;
4478
+ field.indexLimit = Math.min(field.indexLimit, await this._countFieldSymbols(field.field));
4479
+ for (const index of field.needed) {
4480
+ if (index >= field.indexLimit) {
4481
+ field.needed.delete(index);
4482
+ }
3910
4483
  }
3911
4484
  }
3912
4485
  }
3913
4486
  }
3914
- this._emitProgress("symbol-analysis", position + 1, fields.length);
4487
+ this._emitProgress("symbol-analysis", done + recordCount, rowsToLoad);
3915
4488
  });
4489
+ if (rowsToLoad === 0) {
4490
+ this._emitProgress("symbol-analysis", 0, 0);
4491
+ }
3916
4492
  return symbolUsage;
3917
4493
  }
3918
4494
  /**
@@ -3920,28 +4496,27 @@ var init_QvdFileReader = __esm({
3920
4496
  *
3921
4497
  * The count is `countFieldSymbols`, the parse itself told to decode nothing, so it is the count
3922
4498
  * `_parseSymbolTable` will produce and `_parseIndexTable` will check against. The field's area is
3923
- * validated first, as `_parseSymbolTable` would, so a damaged `Offset` or `Length` is reported the
3924
- * same way wherever it is met.
4499
+ * validated before it is read, by `_symbolAreaPlan`, so a damaged `Offset`, `Length` or `NoOfSymbols` is
4500
+ * reported the same way wherever it is met. Its bytes are kept for the parse that follows. The walk checks the count against `NoOfSymbols` as the
4501
+ * parse does, so a field whose count is wrong is refused here, before the pass keeps anything on the
4502
+ * strength of it.
3925
4503
  *
3926
4504
  * @param {any} field The field's header.
3927
- * @return {number} Its symbols.
3928
- * @throws {QvdCorruptedError} If the area is not inside the symbol table, or a symbol runs past it.
4505
+ * @return {Promise<number>} Its symbols.
4506
+ * @throws {QvdCorruptedError} If the area is not inside the symbol table, a symbol runs past it, or it
4507
+ * holds a different number of symbols from its `NoOfSymbols`.
3929
4508
  * @private
3930
4509
  */
3931
- _countFieldSymbols(field) {
3932
- assert3(
3933
- this._buffer && this._symbolTableOffset && this._indexTableOffset,
3934
- "The QVD file has not been read before its symbols were counted."
3935
- );
3936
- const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
3937
- validateFieldMetadata(field, symbolBuffer.length, this._path);
3938
- const offset = headerInteger(field["Offset"]);
4510
+ async _countFieldSymbols(field) {
4511
+ const area = await this._symbolAreaOf(field);
3939
4512
  return countFieldSymbols(
3940
- symbolBuffer,
3941
- offset,
3942
- offset + headerInteger(field["Length"]),
4513
+ area.buffer,
4514
+ area.start,
4515
+ area.end,
4516
+ headerInteger(field["NoOfSymbols"]),
3943
4517
  field["FieldName"],
3944
- this._path
4518
+ this._path,
4519
+ area.base
3945
4520
  );
3946
4521
  }
3947
4522
  /**
@@ -3961,7 +4536,7 @@ var init_QvdFileReader = __esm({
3961
4536
  * that is fewer than the window covers - see `_prepare`.
3962
4537
  */
3963
4538
  async _parseSymbolTable(symbolsToKeep = null, rowsToLoad = 0, liveRows = null) {
3964
- if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
4539
+ if (!this._handle || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
3965
4540
  throw new QvdCorruptedError(
3966
4541
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
3967
4542
  {
@@ -3972,44 +4547,58 @@ var init_QvdFileReader = __esm({
3972
4547
  }
3973
4548
  const allFields = this._allFields;
3974
4549
  const fields = this._selectedFields;
3975
- const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
3976
- const symbolTableSize = symbolBuffer.length;
4550
+ const symbolTableSize = this._symbolTableLength();
4551
+ const plan = this._symbolAreaPlan();
4552
+ const symbolBytes = plan.ranges.reduce((sum, range) => sum + (range.end - range.start), 0);
3977
4553
  const totalRows = headerInteger(this._header["QvdTableHeader"]["NoOfRecords"]);
3978
- validateSymbolTableSize(symbolTableSize, this._path, totalRows);
4554
+ const recordSize = headerInteger(this._header["QvdTableHeader"]["RecordByteSize"]);
4555
+ validateSymbolTableSize(symbolBytes, this._path, totalRows);
3979
4556
  if (this._headerMatchesFile) {
3980
4557
  validateMemoryAvailability(
3981
- symbolTableSize,
4558
+ symbolBytes,
3982
4559
  rowsToLoad,
3983
4560
  totalRows,
3984
4561
  this._path,
3985
4562
  this._memorySafetyFactor,
3986
4563
  fields.length,
3987
4564
  this._materialisesRows,
3988
- liveRows
4565
+ liveRows,
4566
+ this._bytesHeld(symbolBytes, rowsToLoad, recordSize, liveRows, false),
4567
+ // `symbolsToKeep` is non-null exactly when the symbol-usage pass has run, and a pass that has
4568
+ // run has read the window's records once already - so the read's total is two passes over them.
4569
+ symbolBytes + readPasses(symbolsToKeep !== null) * rowsToLoad * recordSize
3989
4570
  );
3990
4571
  }
3991
- warnLargeSymbolTable(symbolTableSize, rowsToLoad, totalRows, fields.length, this._materialisesRows);
4572
+ warnLargeSymbolTable(symbolBytes, rowsToLoad, totalRows, fields.length, this._materialisesRows);
3992
4573
  for (const field of allFields) {
3993
- validateFieldMetadata(field, symbolBuffer.length, this._path);
4574
+ validateFieldMetadata(field, symbolTableSize, this._path);
3994
4575
  }
3995
4576
  validateSymbolAreas(allFields, this._path);
3996
- this._symbolTable = fields.map((field, position) => {
4577
+ const symbolTable = [];
4578
+ for (const [position, field] of fields.entries()) {
3997
4579
  this._throwIfAborted();
3998
- const symbolsOffset = headerInteger(field["Offset"]);
3999
- const symbolsLength = headerInteger(field["Length"]);
4000
- const symbols = parseFieldSymbols(
4001
- symbolBuffer,
4002
- symbolsOffset,
4003
- symbolsOffset + symbolsLength,
4004
- // By position, matching how `_analyzeIndexTableSymbolUsage` built it. Both walk
4005
- // `this._selectedFields`, so position is the one key that cannot collide.
4006
- symbolsToKeep ? symbolsToKeep[position] : null,
4007
- field["FieldName"],
4008
- this._path
4580
+ const area = await this._symbolAreaOf(field);
4581
+ symbolTable.push(
4582
+ parseFieldSymbols(
4583
+ area.buffer,
4584
+ area.start,
4585
+ area.end,
4586
+ // Checked against the symbols the area holds, which is the one check that sees a terminator
4587
+ // damaged in the middle of it (#124).
4588
+ headerInteger(field["NoOfSymbols"]),
4589
+ // By position, matching how `_analyzeIndexTableSymbolUsage` built it. Both walk
4590
+ // `this._selectedFields`, so position is the one key that cannot collide.
4591
+ symbolsToKeep ? symbolsToKeep[position] : null,
4592
+ field["FieldName"],
4593
+ this._path,
4594
+ void 0,
4595
+ area.base
4596
+ )
4009
4597
  );
4598
+ this._releaseSymbolArea(field);
4010
4599
  this._emitProgress("symbol-table", position + 1, fields.length);
4011
- return symbols;
4012
- });
4600
+ }
4601
+ this._symbolTable = symbolTable;
4013
4602
  }
4014
4603
  /**
4015
4604
  * Parses the bit stuffed index table of the QVD file. This method is part of the parsing process
@@ -4036,27 +4625,46 @@ var init_QvdFileReader = __esm({
4036
4625
  * straight to the caller. Rows outside the window are not decoded, so they are not checked.
4037
4626
  *
4038
4627
  * @param {QvdRowWindow} window The rows to decode.
4628
+ * @param {number} [progressBase=0] Rows decoded before this call, so that progress over a chunked
4629
+ * iteration counts the whole window rather than restarting at every chunk - what `_buildRows` takes
4630
+ * for the same reason.
4631
+ * @param {number|null} [progressTotal=null] Rows the whole window covers, or null for this call's own.
4632
+ * @param {Array<Int32Array>|null} [into=null] Arrays to decode into, one per selected field and at least
4633
+ * `limit` long, for a caller that decodes chunk after chunk and keeps none of them. Null allocates.
4039
4634
  * @throws {QvdCorruptedError} If an index in the window addresses neither a symbol nor NULL.
4040
4635
  */
4041
- async _parseIndexTable(window) {
4042
- const { fields, recordSize, rowsToLoad, indexBuffer, firstRow } = this._planIndexTable(window, "parseIndexTable");
4043
- assert3(this._symbolTable, "The QVD file symbol table has not been parsed.");
4636
+ async _parseIndexTable(window, progressBase = 0, progressTotal = null, into = null) {
4637
+ const { fields, recordSize, rowsToLoad, firstRow } = this._planIndexTable(window, "parseIndexTable");
4638
+ const decodedBefore = progressBase;
4639
+ const decodedTotal = progressTotal === null ? rowsToLoad : progressTotal;
4640
+ assert4(this._symbolTable, "The QVD file symbol table has not been parsed.");
4044
4641
  const symbolTable = this._symbolTable;
4045
- const columns = fields.map((field, position) => {
4046
- this._throwIfAborted();
4047
- const column = decodeIndexColumn(
4048
- indexBuffer,
4049
- recordSize,
4050
- rowsToLoad,
4051
- headerInteger(field["BitOffset"]),
4052
- headerInteger(field["BitWidth"]),
4053
- headerInteger(field["Bias"]),
4054
- new Int32Array(rowsToLoad),
4055
- { symbolCount: symbolTable[position].numbers.length, field: field["FieldName"], file: this._path, firstRow }
4056
- );
4057
- this._emitProgress("index-table", position + 1, fields.length);
4058
- return column;
4642
+ const decoders = fields.map((field, position) => ({
4643
+ bitOffset: headerInteger(field["BitOffset"]),
4644
+ bitWidth: headerInteger(field["BitWidth"]),
4645
+ bias: headerInteger(field["Bias"]),
4646
+ symbolCount: symbolTable[position].numbers.length,
4647
+ name: field["FieldName"]
4648
+ }));
4649
+ const columns = into === null ? fields.map(() => new Int32Array(rowsToLoad)) : into.map((codes) => codes.subarray(0, rowsToLoad));
4650
+ await this._forEachSlice(firstRow, rowsToLoad, recordSize, (records, done, count) => {
4651
+ decoders.forEach((decoder, position) => {
4652
+ decodeIndexColumn(
4653
+ records,
4654
+ recordSize,
4655
+ count,
4656
+ decoder.bitOffset,
4657
+ decoder.bitWidth,
4658
+ decoder.bias,
4659
+ columns[position].subarray(done, done + count),
4660
+ { symbolCount: decoder.symbolCount, field: decoder.name, file: this._path, firstRow: firstRow + done }
4661
+ );
4662
+ });
4663
+ this._emitProgress("index-table", decodedBefore + done + count, decodedTotal);
4059
4664
  });
4665
+ if (rowsToLoad === 0) {
4666
+ this._emitProgress("index-table", decodedBefore, decodedTotal);
4667
+ }
4060
4668
  this._indexColumns = columns;
4061
4669
  this._rowsDecoded = rowsToLoad;
4062
4670
  }
@@ -4075,31 +4683,122 @@ var init_QvdFileReader = __esm({
4075
4683
  * @return {Promise<import('./QvdDataFrame.js').QvdFileMetadata>} The file's schema and header.
4076
4684
  */
4077
4685
  async loadMetadata() {
4078
- await this._readData({ offset: 0, limit: null }, true);
4079
- this._emitProgress("header", 0, 1);
4080
- await this._parseHeader();
4081
- this._emitProgress("header", 1, 1);
4082
- this._throwIfAborted();
4083
- assert3(this._header && this._allFields, "The QVD file header has not been parsed.");
4084
- const header = this._header["QvdTableHeader"];
4085
- const columns = this._allFields.map((field) => field["FieldName"]);
4086
- const rowCount = headerInteger(header["NoOfRecords"]);
4087
- validateRecordCount(rowCount, this._path, "readMetadata");
4088
- const shape = new QvdDataFrame([], columns, header, {
4089
- symbolTableBytes: headerInteger(header["Offset"]),
4090
- totalRows: rowCount,
4091
- rowsLoaded: 0,
4092
- symbolFiltering: false,
4093
- symbolsKept: null
4686
+ return await this._closingAfter(async () => {
4687
+ await this._readData({ offset: 0, limit: null }, true);
4688
+ this._emitProgress("header", 0, 1);
4689
+ await this._parseHeader();
4690
+ this._emitProgress("header", 1, 1);
4691
+ this._throwIfAborted();
4692
+ assert4(this._header && this._allFields, "The QVD file header has not been parsed.");
4693
+ const header = this._header["QvdTableHeader"];
4694
+ const columns = this._allFields.map((field) => field["FieldName"]);
4695
+ const rowCount = headerInteger(header["NoOfRecords"]);
4696
+ validateRecordCount(rowCount, this._path, "readMetadata");
4697
+ const shape = new QvdDataFrame([], columns, header, {
4698
+ symbolTableBytes: headerInteger(header["Offset"]),
4699
+ totalRows: rowCount,
4700
+ rowsLoaded: 0,
4701
+ symbolFiltering: false,
4702
+ symbolsKept: null
4703
+ });
4704
+ return {
4705
+ columns,
4706
+ rowCount,
4707
+ columnCount: columns.length,
4708
+ fields: columns.map((name) => shape.getFieldMetadata(name)),
4709
+ fileMetadata: shape.fileMetadata,
4710
+ metadata: header
4711
+ };
4712
+ });
4713
+ }
4714
+ /**
4715
+ * What a read of this file would cost, and whether it fits, without reading it.
4716
+ *
4717
+ * Reads the header and the file's size and nothing else, at the constant cost of `loadMetadata()`,
4718
+ * then asks the same question a read asks before it allocates anything - through the same function,
4719
+ * from the same numbers. That is the whole point: an answer computed a second way would be a second
4720
+ * opinion, and a read this approves would still be refused.
4721
+ *
4722
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [rawWindow]
4723
+ * The rows the read would cover, spelled any of the ways a read accepts.
4724
+ * @param {{chunkSize?: number|null}} [options] `chunkSize` when the read would be an `iterate()`,
4725
+ * which holds two chunks of rows rather than the window.
4726
+ * @return {Promise<any>} The answer - see `checkMemory`.
4727
+ */
4728
+ async checkRead(rawWindow, { chunkSize = null } = {}) {
4729
+ const window = normaliseWindow(rawWindow, this._path);
4730
+ return await this._closingAfter(async () => {
4731
+ await this._readData({ offset: 0, limit: null }, true);
4732
+ this._emitProgress("header", 0, 1);
4733
+ await this._parseHeader();
4734
+ this._emitProgress("header", 1, 1);
4735
+ this._throwIfAborted();
4736
+ assert4(
4737
+ this._header && this._selectedFields && this._allFields && this._symbolTableOffset !== null,
4738
+ "The QVD file header has not been parsed."
4739
+ );
4740
+ const header = this._header["QvdTableHeader"];
4741
+ const totalRows = headerInteger(header["NoOfRecords"]);
4742
+ const recordSize = headerInteger(header["RecordByteSize"]);
4743
+ const symbolTableLength = headerInteger(header["Offset"]);
4744
+ const selected = this._selectedFields;
4745
+ validateRecordSize(recordSize, this._path, "checkRead");
4746
+ validateRecordCount(totalRows, this._path, "checkRead");
4747
+ if (!this._headerMatchesFile) {
4748
+ throw new QvdCorruptedError("The file is shorter than its header claims.", {
4749
+ file: this._path,
4750
+ fileSize: this._fileSize,
4751
+ requiredBytes: this._symbolTableOffset + symbolTableLength + totalRows * recordSize,
4752
+ stage: "checkRead"
4753
+ });
4754
+ }
4755
+ const tableLength = this._symbolTableLength();
4756
+ for (const field of this._allFields) {
4757
+ validateFieldMetadata(field, tableLength, this._path);
4758
+ validateFieldBitMetadata(field, recordSize, this._path);
4759
+ }
4760
+ validateSymbolAreas(this._allFields, this._path);
4761
+ const resolved = resolveWindow(window, totalRows);
4762
+ const windowRows = resolved.limit;
4763
+ const liveRows = chunkSize === null ? null : { rows: chunkSize * 2, perChunk: 2 };
4764
+ const analysisAhead = this._analysisWouldRun(window, resolved, totalRows, symbolTableLength);
4765
+ const measured = getMemoryBudget();
4766
+ const ask = /* @__PURE__ */ __name((fields, rows) => {
4767
+ const bytes = symbolBytesOf(fields, symbolTableLength);
4768
+ return checkMemory({
4769
+ measured,
4770
+ symbolTableSize: bytes,
4771
+ maxRows: rows,
4772
+ totalRows,
4773
+ safetyFactor: this._memorySafetyFactor,
4774
+ columnCount: fields.length,
4775
+ materialisesRows: this._materialisesRows,
4776
+ live: liveRows,
4777
+ bytesHeld: this._bytesHeld(bytes, rows, recordSize, liveRows, analysisAhead),
4778
+ // What it reads from the file, which is not what it holds: the symbol areas, and every record
4779
+ // the window covers, read a slice at a time and not kept - twice over where the symbol-usage
4780
+ // pass will run, since it reads them before the decode reads them again.
4781
+ readBytes: bytes + readPasses(analysisAhead) * windowRows * recordSize
4782
+ });
4783
+ }, "ask");
4784
+ const answer = ask(selected, windowRows);
4785
+ if (!answer.fits && selected.length > 1) {
4786
+ const bySize = [...selected].sort(
4787
+ (a, b) => headerInteger(a["Length"]) - headerInteger(b["Length"])
4788
+ );
4789
+ for (let take = selected.length - 1; take >= 1; take -= 1) {
4790
+ const fewer = bySize.slice(0, take);
4791
+ if (ask(fewer, windowRows).fits) {
4792
+ answer.suggestions.push({
4793
+ option: "fields",
4794
+ value: fewer.map((field) => field["FieldName"])
4795
+ });
4796
+ break;
4797
+ }
4798
+ }
4799
+ }
4800
+ return answer;
4094
4801
  });
4095
- return {
4096
- columns,
4097
- rowCount,
4098
- columnCount: columns.length,
4099
- fields: columns.map((name) => shape.getFieldMetadata(name)),
4100
- fileMetadata: shape.fileMetadata,
4101
- metadata: header
4102
- };
4103
4802
  }
4104
4803
  /**
4105
4804
  * Loads the QVD file into memory and parses it.
@@ -4114,19 +4813,21 @@ var init_QvdFileReader = __esm({
4114
4813
  */
4115
4814
  async load(window = null) {
4116
4815
  const rows = normaliseWindow(window, this._path);
4117
- const prepared = await this._prepare(rows);
4118
- await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
4119
- const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
4120
- return new QvdDataFrame(
4121
- data,
4122
- prepared.columns,
4123
- prepared.metadata,
4124
- {
4125
- ...prepared.loadStats,
4126
- rowsLoaded: data.length
4127
- },
4128
- prepared.storedSymbols
4129
- );
4816
+ return await this._closingAfter(async () => {
4817
+ const prepared = await this._prepare(rows);
4818
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
4819
+ const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
4820
+ return new QvdDataFrame(
4821
+ data,
4822
+ prepared.columns,
4823
+ prepared.metadata,
4824
+ {
4825
+ ...prepared.loadStats,
4826
+ rowsLoaded: data.length
4827
+ },
4828
+ prepared.storedSymbols
4829
+ );
4830
+ });
4130
4831
  }
4131
4832
  /**
4132
4833
  * Reads the file as columns, without ever materialising rows.
@@ -4145,19 +4846,21 @@ var init_QvdFileReader = __esm({
4145
4846
  */
4146
4847
  async loadColumnar(window = null) {
4147
4848
  const rows = normaliseWindow(window, this._path);
4148
- const prepared = await this._prepare(rows, null, true);
4149
- await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
4150
- const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
4151
- assert3(this._indexColumns, "The QVD file index table has not been parsed.");
4152
- return new QvdColumnTable2({
4153
- columns: prepared.columns,
4154
- codesByField: this._indexColumns,
4155
- symbolsByField: prepared.resolvedByField,
4156
- halvesByField: prepared.halvesByField,
4157
- rowCount: this._rowsDecoded,
4158
- metadata: prepared.metadata,
4159
- storedSymbols: prepared.storedSymbols,
4160
- loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
4849
+ return await this._closingAfter(async () => {
4850
+ const prepared = await this._prepare(rows, null, true);
4851
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
4852
+ const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
4853
+ assert4(this._indexColumns, "The QVD file index table has not been parsed.");
4854
+ return new QvdColumnTable2({
4855
+ columns: prepared.columns,
4856
+ codesByField: this._indexColumns,
4857
+ symbolsByField: prepared.resolvedByField,
4858
+ halvesByField: prepared.halvesByField,
4859
+ rowCount: this._rowsDecoded,
4860
+ metadata: prepared.metadata,
4861
+ storedSymbols: prepared.storedSymbols,
4862
+ loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
4863
+ });
4161
4864
  });
4162
4865
  }
4163
4866
  /**
@@ -4186,37 +4889,41 @@ var init_QvdFileReader = __esm({
4186
4889
  * @return {AsyncGenerator<QvdDataFrame>} The chunks, in order.
4187
4890
  */
4188
4891
  async *iterateRows(window, chunkSize) {
4189
- if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
4190
- throw new QvdValidationError("chunkSize must be a positive integer", {
4191
- provided: chunkSize,
4192
- type: typeof chunkSize,
4193
- file: this._path
4194
- });
4195
- }
4892
+ requireChunkSize(chunkSize, this._path);
4196
4893
  const liveRows = { rows: chunkSize * 2, perChunk: 2 };
4197
4894
  const rows = normaliseWindow(window, this._path);
4198
- const prepared = await this._prepare(rows, liveRows);
4199
- if (prepared.rowsAvailable === 0) {
4200
- this._planIndexTable({ offset: prepared.offset, limit: 0 }, "parseIndexTable");
4201
- return;
4202
- }
4203
- for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
4204
- this._throwIfAborted();
4205
- const count = Math.min(chunkSize, prepared.rowsAvailable - done);
4206
- const offset = prepared.offset + done;
4207
- await this._parseIndexTable({ offset, limit: count });
4208
- const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
4209
- yield new QvdDataFrame(
4210
- data,
4211
- prepared.columns,
4212
- prepared.metadata,
4213
- {
4214
- ...prepared.loadStats,
4215
- offset,
4216
- rowsLoaded: data.length
4217
- },
4218
- prepared.storedSymbols
4219
- );
4895
+ this._startRead();
4896
+ let failing = false;
4897
+ try {
4898
+ const prepared = await this._prepare(rows, liveRows);
4899
+ if (prepared.rowsAvailable === 0) {
4900
+ this._planIndexTable({ offset: prepared.offset, limit: 0 }, "parseIndexTable");
4901
+ return;
4902
+ }
4903
+ const codes = prepared.columns.map(() => new Int32Array(Math.min(chunkSize, prepared.rowsAvailable)));
4904
+ for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
4905
+ this._throwIfAborted();
4906
+ const count = Math.min(chunkSize, prepared.rowsAvailable - done);
4907
+ const offset = prepared.offset + done;
4908
+ await this._parseIndexTable({ offset, limit: count }, done, prepared.rowsAvailable, codes);
4909
+ const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
4910
+ yield new QvdDataFrame(
4911
+ data,
4912
+ prepared.columns,
4913
+ prepared.metadata,
4914
+ {
4915
+ ...prepared.loadStats,
4916
+ offset,
4917
+ rowsLoaded: data.length
4918
+ },
4919
+ prepared.storedSymbols
4920
+ );
4921
+ }
4922
+ } catch (error) {
4923
+ failing = true;
4924
+ throw error;
4925
+ } finally {
4926
+ await this._endRead(failing);
4220
4927
  }
4221
4928
  }
4222
4929
  /**
@@ -4250,23 +4957,21 @@ var init_QvdFileReader = __esm({
4250
4957
  await this._parseHeader();
4251
4958
  this._emitProgress("header", 1, 1);
4252
4959
  this._throwIfAborted();
4253
- assert3(this._header, "The QVD file header has not been parsed.");
4960
+ assert4(this._header, "The QVD file header has not been parsed.");
4254
4961
  const totalRows = headerInteger(this._header["QvdTableHeader"]["NoOfRecords"]);
4255
4962
  const symbolTableLength = headerInteger(this._header["QvdTableHeader"]["Offset"]);
4256
4963
  const resolved = resolveWindow(window, totalRows);
4257
4964
  const rowsAvailable = resolved.limit;
4258
4965
  let symbolsToKeep = null;
4259
4966
  let symbolsKept = null;
4260
- if (window.limit !== null || window.offset > 0) {
4261
- if (symbolTableLength > this._symbolFilteringThreshold) {
4262
- symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
4263
- symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
4264
- }
4967
+ if (this._analysisWouldRun(window, resolved, totalRows, symbolTableLength)) {
4968
+ symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
4969
+ symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
4265
4970
  }
4266
4971
  await this._parseSymbolTable(symbolsToKeep, rowsAvailable, liveRows);
4267
- assert3(this._symbolTable, "The QVD file symbol table has not been parsed.");
4972
+ assert4(this._symbolTable, "The QVD file symbol table has not been parsed.");
4268
4973
  this._throwIfAborted();
4269
- assert3(this._selectedFields, "The QVD file fields have not been resolved.");
4974
+ assert4(this._selectedFields, "The QVD file fields have not been resolved.");
4270
4975
  const resolvedByField = [];
4271
4976
  const halvesByField = [];
4272
4977
  const entries = [];
@@ -4327,7 +5032,7 @@ var init_QvdFileReader = __esm({
4327
5032
  * @private
4328
5033
  */
4329
5034
  _buildRows(resolvedByField, progressBase, progressTotal) {
4330
- assert3(this._indexColumns, "The QVD file index table has not been parsed.");
5035
+ assert4(this._indexColumns, "The QVD file index table has not been parsed.");
4331
5036
  const indexColumns = this._indexColumns;
4332
5037
  const fieldCount = indexColumns.length;
4333
5038
  const rowCount = this._rowsDecoded;
@@ -5016,12 +5721,16 @@ var init_QvdDataFrame = __esm({
5016
5721
  /**
5017
5722
  * Reads a QVD file in chunks, as an async generator of data frames.
5018
5723
  *
5019
- * The file is opened, read and parsed once; only the index decode and the row building happen
5020
- * per chunk, so what this bounds is row materialisation - the part that actually dominates a
5021
- * large read's heap. It is **not** constant-memory reading of an arbitrarily large file: the
5022
- * symbol table is parsed in full whatever the chunk size, because a stored index in the last
5023
- * chunk can address the first symbol. On a high-cardinality file that table is the bulk of the
5024
- * cost, and `readMetadata` is the only read that avoids it.
5724
+ * The file is opened once and its symbol table parsed once. Each chunk's records are read from the
5725
+ * file when that chunk is built, so what this holds is the symbol table and two chunks of rows,
5726
+ * whatever the size of the file: a 20 GB QVD iterates in the memory its symbol table needs. That
5727
+ * table is still parsed in full whatever the chunk size, because a stored index in the last chunk
5728
+ * can address the first symbol. On a high-cardinality file that table is the bulk of the cost, and
5729
+ * `readMetadata` is the only read that avoids it.
5730
+ *
5731
+ * The file stays open until the iteration ends. Running it to the end closes it, and so do
5732
+ * `break` or a throw inside `for await` and a call to `return()` on the iterator; an iterator
5733
+ * abandoned part-way without any of those holds the file until it is garbage-collected.
5025
5734
  *
5026
5735
  * ```js
5027
5736
  * for await (const chunk of QvdDataFrame.iterate('big.qvd', {chunkSize: 50_000})) {
@@ -5095,6 +5804,68 @@ var init_QvdDataFrame = __esm({
5095
5804
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
5096
5805
  return await new QvdFileReader2(path5, metadataOptionsFrom(options)).loadMetadata();
5097
5806
  }
5807
+ /**
5808
+ * Answers what a read would cost, and whether it fits, without doing it.
5809
+ *
5810
+ * Takes the options `fromQvd()` takes, plus `as` and `chunkSize` to say which read is being asked
5811
+ * about. Reads the header and the file's size and nothing else, at a cost that does not grow with
5812
+ * the file.
5813
+ *
5814
+ * The answer comes from the same function a read consults before it allocates anything, from the
5815
+ * same numbers, so **a read this approves is not refused later for memory** - and every suggestion
5816
+ * it carries has been read back through the check, so following one gives a read that fits.
5817
+ *
5818
+ * It answers about resources, so it answers only for a header it can trust. A header whose numbers
5819
+ * are not usable, or that claims more than the file holds, is refused as a `QvdCorruptedError` rather
5820
+ * than answered: sizing a read from numbers the file contradicts produced a memory verdict about a
5821
+ * file whose real problem was structural, and it was wrong in both directions - approving a read the
5822
+ * library then refused, and refusing another with advice that was refused too.
5823
+ *
5824
+ * That covers everything a reader can tell from the header: a field area past the end of the symbol
5825
+ * table, two fields claiming one area, a `Bias` that is neither 0 nor -2, a `BitWidth` past 31. Damage
5826
+ * that is not in the header - a value or an index the file has spoiled - is still found only by
5827
+ * reading, and still refused as a `QvdCorruptedError` after this has said the read fits.
5828
+ *
5829
+ * ```js
5830
+ * const answer = await QvdDataFrame.checkRead('huge.qvd', {as: 'columns', fields: ['Amount']});
5831
+ *
5832
+ * if (!answer.fits) {
5833
+ * console.log(answer.reason); // 'memory'
5834
+ * console.log(answer.suggestions); // [{option: 'limit', value: 1250000}, ...]
5835
+ * }
5836
+ * ```
5837
+ *
5838
+ * @param {string} path The QVD file.
5839
+ * @param {object} [options] What `fromQvd()` takes, plus the two below.
5840
+ * @param {'rows'|'columns'} [options.as='rows'] Which read is being asked about: `rows` builds row
5841
+ * arrays and `columns` does not, which is most of what a read costs.
5842
+ * @param {number|null} [options.chunkSize=null] The chunk an `iterate()` would use, which holds two
5843
+ * chunks of rows rather than the whole window.
5844
+ * @return {Promise<any>} The answer: `fits`, `reason` when it does not, `estimate`, `budget`,
5845
+ * `exact` and `suggestions`.
5846
+ * @throws {QvdValidationError} If an option's value is not valid, with `context.reason` of `option`.
5847
+ * @throws {QvdCorruptedError} If the header cannot be read, its numbers are not usable, or it claims
5848
+ * more than the file holds. The read refuses such a file too, though it may name the fault
5849
+ * differently - it gets there by planning the index table, where this gets there from the size.
5850
+ */
5851
+ static async checkRead(path5, options = {}) {
5852
+ const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
5853
+ const { as = "rows", chunkSize = null } = options;
5854
+ if (as !== "rows" && as !== "columns") {
5855
+ throw new QvdValidationError("as must be 'rows' or 'columns'", {
5856
+ provided: as,
5857
+ reason: "option",
5858
+ option: "as",
5859
+ value: as,
5860
+ file: path5
5861
+ });
5862
+ }
5863
+ if (chunkSize !== null) {
5864
+ requireChunkSize(chunkSize, path5);
5865
+ }
5866
+ const reader = new QvdFileReader2(path5, { ...readerOptionsFrom(options), materialisesRows: as === "rows" });
5867
+ return await reader.checkRead(windowFrom(options), { chunkSize });
5868
+ }
5098
5869
  /**
5099
5870
  * Constructs a data frame from a dictionary.
5100
5871
  *