qvdjs 0.10.1 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -250,6 +250,128 @@ var init_QvdSymbol = __esm({
250
250
  };
251
251
  }
252
252
  });
253
+
254
+ // src/util/readOptions.js
255
+ function requireRowCount(value, name, filePath) {
256
+ if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
257
+ throw new QvdValidationError(`${name} must be a non-negative integer`, {
258
+ option: name,
259
+ provided: value,
260
+ type: typeof value,
261
+ file: filePath
262
+ });
263
+ }
264
+ return value;
265
+ }
266
+ function normaliseWindow(window, filePath) {
267
+ if (window === null || window === void 0) {
268
+ return { offset: 0, limit: null };
269
+ }
270
+ if (typeof window === "number") {
271
+ return { offset: 0, limit: requireRowCount(window, "maxRows", filePath) };
272
+ }
273
+ if (typeof window !== "object" || Array.isArray(window)) {
274
+ throw new QvdValidationError("The row window must be a number, null, or an {offset, limit} object", {
275
+ provided: window,
276
+ type: typeof window,
277
+ file: filePath
278
+ });
279
+ }
280
+ const { offset, limit, maxRows } = window;
281
+ const limitGiven = limit !== void 0 && limit !== null;
282
+ const maxRowsGiven = maxRows !== void 0 && maxRows !== null;
283
+ if (limitGiven && maxRowsGiven) {
284
+ throw new QvdValidationError("maxRows and limit are two names for the same option; pass one of them, not both", {
285
+ maxRows,
286
+ limit,
287
+ file: filePath
288
+ });
289
+ }
290
+ return {
291
+ offset: offset === void 0 || offset === null ? 0 : requireRowCount(offset, "offset", filePath),
292
+ limit: limitGiven ? requireRowCount(limit, "limit", filePath) : maxRowsGiven ? requireRowCount(maxRows, "maxRows", filePath) : null
293
+ };
294
+ }
295
+ function resolveWindow(window, totalRows) {
296
+ const rows = Number.isSafeInteger(totalRows) && totalRows > 0 ? totalRows : 0;
297
+ const offset = Math.min(window.offset, rows);
298
+ return {
299
+ offset,
300
+ limit: Math.max(0, Math.min(window.limit === null ? Infinity : window.limit, rows - offset))
301
+ };
302
+ }
303
+ function selectFields(fields, requested, filePath) {
304
+ if (requested === null || requested === void 0) {
305
+ return fields;
306
+ }
307
+ if (!Array.isArray(requested)) {
308
+ throw new QvdValidationError("fields must be an array of field names", {
309
+ provided: requested,
310
+ type: typeof requested,
311
+ file: filePath
312
+ });
313
+ }
314
+ const available = fields.map((field) => field["FieldName"]);
315
+ if (requested.length === 0) {
316
+ throw new QvdValidationError("fields must name at least one field", {
317
+ availableColumns: available,
318
+ file: filePath
319
+ });
320
+ }
321
+ const seen = /* @__PURE__ */ new Set();
322
+ return requested.map((name) => {
323
+ if (typeof name !== "string") {
324
+ throw new QvdValidationError("Field names must be strings", {
325
+ provided: name,
326
+ type: typeof name,
327
+ availableColumns: available,
328
+ file: filePath
329
+ });
330
+ }
331
+ if (seen.has(name)) {
332
+ throw new QvdValidationError(`Field '${name}' is listed twice`, {
333
+ column: name,
334
+ fields: requested,
335
+ file: filePath
336
+ });
337
+ }
338
+ seen.add(name);
339
+ const index = available.indexOf(name);
340
+ if (index === -1) {
341
+ throw new QvdValidationError(`Column '${name}' does not exist`, {
342
+ column: name,
343
+ availableColumns: available,
344
+ file: filePath
345
+ });
346
+ }
347
+ return fields[index];
348
+ });
349
+ }
350
+ function readerOptionsFrom(options) {
351
+ return {
352
+ allowedDir: options.allowedDir,
353
+ memorySafetyFactor: options.memorySafetyFactor,
354
+ symbolFilteringThreshold: options.symbolFilteringThreshold,
355
+ fields: options.fields === void 0 ? null : options.fields,
356
+ onProgress: options.onProgress,
357
+ signal: options.signal
358
+ };
359
+ }
360
+ function metadataOptionsFrom(options) {
361
+ return {
362
+ allowedDir: options.allowedDir,
363
+ onProgress: options.onProgress,
364
+ signal: options.signal
365
+ };
366
+ }
367
+ function windowFrom(options) {
368
+ return { offset: options.offset, limit: options.limit, maxRows: options.maxRows };
369
+ }
370
+ var init_readOptions = __esm({
371
+ "src/util/readOptions.js"() {
372
+ init_QvdErrors();
373
+ }
374
+ });
253
375
  function isWithinDirectoryLexically(resolvedBaseDir, resolvedPath) {
254
376
  const isCaseInsensitiveFS = process.platform === "win32";
255
377
  const base = isCaseInsensitiveFS ? resolvedBaseDir.toLowerCase() : resolvedBaseDir;
@@ -807,11 +929,12 @@ function estimateRowMemory(rows, columnCount) {
807
929
  }
808
930
  return BASE_BYTES + rows * (ROW_BASE_BYTES + PER_CELL_BYTES * columnCount);
809
931
  }
810
- function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
932
+ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null) {
811
933
  const FULL_PARSE_OVERHEAD = 6;
812
934
  const MINIMAL_OVERHEAD = 0.01;
813
935
  const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
814
- const rowMemory = materialisesRows ? estimateRowMemory(rowsToLoad, columnCount) : BASE_BYTES;
936
+ const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
937
+ const rowMemory = materialisesRows ? estimateRowMemory(liveRows, columnCount) : BASE_BYTES;
815
938
  if (maxRows === null || maxRows >= totalRows) {
816
939
  return symbolTableSize * FULL_PARSE_OVERHEAD + rowMemory;
817
940
  }
@@ -821,18 +944,44 @@ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount =
821
944
  const skippedSymbolsMemory = symbolTableSize * (1 - symbolPercentage) * MINIMAL_OVERHEAD;
822
945
  return keptSymbolsMemory + skippedSymbolsMemory + rowMemory;
823
946
  }
824
- function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true) {
825
- if (estimateMemoryUsage(symbolTableSize, totalRows, totalRows, columnCount, materialisesRows) <= budget) {
947
+ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false) {
948
+ const costOf = (rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0);
949
+ if (costOf(totalRows) <= budget) {
826
950
  return totalRows;
827
951
  }
828
- if (estimateMemoryUsage(symbolTableSize, 0, totalRows, columnCount, materialisesRows) > budget) {
952
+ if (costOf(0) > budget) {
829
953
  return 0;
830
954
  }
831
955
  let low = 0;
832
956
  let high = totalRows;
833
957
  while (high - low > 1) {
834
958
  const mid = Math.floor((low + high) / 2);
835
- if (estimateMemoryUsage(symbolTableSize, mid, totalRows, columnCount, materialisesRows) <= budget) {
959
+ if (costOf(mid) <= budget) {
960
+ low = mid;
961
+ } else {
962
+ high = mid;
963
+ }
964
+ }
965
+ return low;
966
+ }
967
+ function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false) {
968
+ const covered = windowRows === null || windowRows >= totalRows ? totalRows : windowRows;
969
+ const fits = (chunk) => {
970
+ const live = Math.min(chunk * liveRowsPerChunk, covered);
971
+ const cost = estimateMemoryUsage(symbolTableSize, windowRows, totalRows, columnCount, true, chunk * liveRowsPerChunk) + (includeExternal ? estimateExternalMemory(live, columnCount) : 0);
972
+ return cost <= budget;
973
+ };
974
+ if (fits(covered)) {
975
+ return covered;
976
+ }
977
+ if (!fits(1)) {
978
+ return 0;
979
+ }
980
+ let low = 1;
981
+ let high = covered;
982
+ while (high - low > 1) {
983
+ const mid = Math.floor((low + high) / 2);
984
+ if (fits(mid)) {
836
985
  low = mid;
837
986
  } else {
838
987
  high = mid;
@@ -840,7 +989,7 @@ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, mat
840
989
  }
841
990
  return low;
842
991
  }
843
- function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true) {
992
+ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null) {
844
993
  if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
845
994
  throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", { safetyFactor });
846
995
  }
@@ -849,12 +998,16 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
849
998
  }
850
999
  const budget = getMemoryBudget();
851
1000
  const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
852
- const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
853
- const externalMemory = estimateExternalMemory(rowsToLoad, columnCount);
1001
+ const rowsLive = live === null ? null : live.rows;
1002
+ const liveRowsPerChunk = live === null ? 1 : live.perChunk;
1003
+ const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
1004
+ const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows, rowsLive);
1005
+ const externalMemory = estimateExternalMemory(liveRows, columnCount);
854
1006
  const bounded = budget.candidates.map((candidate) => {
855
1007
  const heapOnly = candidate.source === "V8 heap limit";
856
1008
  return {
857
1009
  ...candidate,
1010
+ heapOnly,
858
1011
  needs: heapOnly ? heapMemory : heapMemory + externalMemory,
859
1012
  allowed: candidate.bytes * safetyFactor,
860
1013
  bounds: heapOnly ? "the V8 heap" : "the whole process"
@@ -870,12 +1023,14 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
870
1023
  const estimatedMemory = binding ? binding.needs : heapMemory;
871
1024
  const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
872
1025
  if (binding) {
1026
+ const includeExternal = !binding.heapOnly;
873
1027
  const recommendedMaxRows = recommendedRowsFor(
874
1028
  maxAllowedMemory,
875
1029
  symbolTableSize,
876
1030
  totalRows,
877
1031
  columnCount,
878
- materialisesRows
1032
+ materialisesRows,
1033
+ includeExternal
879
1034
  );
880
1035
  const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
881
1036
  const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
@@ -887,15 +1042,27 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
887
1042
  const limitingScope = binding.bounds;
888
1043
  const budgetBreakdown = budget.candidates.map((candidate) => `${candidate.source} ${Math.round(candidate.bytes / 1024 / 1024)}MB`).join(", ");
889
1044
  const observedBreakdown = budget.observed.map((entry) => `${entry.source} ${Math.round(entry.bytes / 1024 / 1024)}MB`).join(", ");
890
- const nothingFits = recommendedMaxRows === 0;
891
1045
  const containerBound = binding.source === "container memory limit";
1046
+ const chunked = rowsLive !== null;
1047
+ const recommendedChunk = chunked ? recommendedChunkFor(
1048
+ maxAllowedMemory,
1049
+ symbolTableSize,
1050
+ maxRows,
1051
+ totalRows,
1052
+ columnCount,
1053
+ liveRowsPerChunk,
1054
+ includeExternal
1055
+ ) : 0;
1056
+ const knob = chunked ? "chunkSize" : "limit";
1057
+ const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
1058
+ const nothingFits = recommendedValue === 0;
892
1059
  let advice;
893
1060
  if (nothingFits) {
894
- advice = `No row count fits this budget - the symbol table alone exceeds it, so maxRows cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
1061
+ advice = `No row count fits this budget - the symbol table alone exceeds it, so ${knob} cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
895
1062
  } else if (containerBound) {
896
- advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or load fewer rows with maxRows (recommended: ${recommendedMaxRows.toLocaleString()} rows or less).`;
1063
+ advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${recommendedValue.toLocaleString()} rows or less).`;
897
1064
  } else {
898
- advice = `Try loading fewer rows using the maxRows parameter (recommended: ${recommendedMaxRows.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
1065
+ advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${recommendedValue.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
899
1066
  }
900
1067
  throw new QvdValidationError(
901
1068
  `Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
@@ -915,22 +1082,30 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
915
1082
  columnCount,
916
1083
  totalRows,
917
1084
  maxRows,
918
- recommendedMaxRows
1085
+ recommendedMaxRows,
1086
+ // Only present when a chunk size is what overflowed, so a caller cannot mistake one
1087
+ // recommendation for the other.
1088
+ ...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
919
1089
  }
920
1090
  );
921
1091
  }
922
1092
  }
923
1093
  function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
924
1094
  const LARGE_SYMBOL_TABLE_WARNING = usableOldSpaceLimit() * 0.125;
925
- if (symbolTableSize > LARGE_SYMBOL_TABLE_WARNING && (maxRows === null || maxRows >= totalRows)) {
926
- const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
927
- const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
928
- const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
929
- const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
930
- console.warn(
931
- `\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). Loading all ${totalRows.toLocaleString()} rows will use ~${estimatedMB}MB RAM. Consider using the maxRows parameter for better performance and lower memory usage.`
932
- );
1095
+ if (symbolTableSize <= LARGE_SYMBOL_TABLE_WARNING) {
1096
+ return;
933
1097
  }
1098
+ const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
1099
+ const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
1100
+ if (estimatedMemory <= LARGE_SYMBOL_TABLE_WARNING) {
1101
+ return;
1102
+ }
1103
+ const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
1104
+ const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
1105
+ const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
1106
+ console.warn(
1107
+ `\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). This read materialises ${rowsToLoad.toLocaleString()} of ${totalRows.toLocaleString()} rows and will use ~${estimatedMB}MB RAM. Reading fewer rows - with limit, maxRows, or a narrower offset window - lowers the row cost, though the symbol table is read in full either way.`
1108
+ );
934
1109
  }
935
1110
  var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES;
936
1111
  var init_memoryUtils = __esm({
@@ -971,7 +1146,7 @@ function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
971
1146
  const maxMB = Math.round(MAX_SYMBOL_TABLE_SIZE / 1024 / 1024);
972
1147
  const heapMB = Math.round(heapLimit / 1024 / 1024);
973
1148
  throw new QvdValidationError(
974
- `Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without maxRows, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
1149
+ `Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without a row window - maxRows, limit or offset - since the symbol table is read in full either way, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
975
1150
  {
976
1151
  file: filePath,
977
1152
  symbolTableSize: symbolTableLength,
@@ -1045,7 +1220,7 @@ function validateRecordCount(totalRows, filePath, stage = "parseIndexTable") {
1045
1220
  });
1046
1221
  }
1047
1222
  }
1048
- function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null) {
1223
+ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null, windowFirstRow = 0, bufferFirstRow = 0) {
1049
1224
  if (isNaN(recordSize) || !Number.isSafeInteger(recordSize) || recordSize < 0) {
1050
1225
  throw new QvdCorruptedError("Invalid record byte size", {
1051
1226
  recordSize,
@@ -1107,23 +1282,28 @@ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, ind
1107
1282
  }
1108
1283
  }
1109
1284
  const requiredIndexBytes = rowsToLoad * recordSize;
1110
- if (indexTableOffset + requiredIndexBytes > bufferLength) {
1285
+ const bufferRecordStart = (windowFirstRow - bufferFirstRow) * recordSize;
1286
+ if (indexTableOffset + bufferRecordStart + requiredIndexBytes > bufferLength) {
1111
1287
  throw new QvdCorruptedError("Index table truncated", {
1112
1288
  indexTableOffset,
1113
1289
  requiredBytes: requiredIndexBytes,
1114
- availableBytes: Math.max(0, bufferLength - indexTableOffset),
1290
+ availableBytes: Math.max(0, bufferLength - indexTableOffset - bufferRecordStart),
1115
1291
  rowsToLoad,
1292
+ windowFirstRow,
1293
+ bufferFirstRow,
1116
1294
  recordSize,
1117
1295
  bufferSize: bufferLength,
1118
1296
  file: filePath,
1119
1297
  stage: "parseIndexTable"
1120
1298
  });
1121
1299
  }
1122
- if (indexTableLength < requiredIndexBytes) {
1300
+ const requiredTableBytes = (windowFirstRow + rowsToLoad) * recordSize;
1301
+ if (indexTableLength < requiredTableBytes) {
1123
1302
  throw new QvdCorruptedError("Index table length smaller than required", {
1124
1303
  indexTableLength,
1125
- requiredBytes: requiredIndexBytes,
1304
+ requiredBytes: requiredTableBytes,
1126
1305
  rowsToLoad,
1306
+ windowFirstRow,
1127
1307
  recordSize,
1128
1308
  file: filePath,
1129
1309
  stage: "parseIndexTable"
@@ -1453,6 +1633,7 @@ var QvdColumn, QvdColumnTable;
1453
1633
  var init_QvdColumnTable = __esm({
1454
1634
  "src/QvdColumnTable.js"() {
1455
1635
  init_QvdErrors();
1636
+ init_readOptions();
1456
1637
  QvdColumn = class {
1457
1638
  /**
1458
1639
  * @param {string} name The field name.
@@ -1647,9 +1828,21 @@ var init_QvdColumnTable = __esm({
1647
1828
  /**
1648
1829
  * Reads a QVD file as columns.
1649
1830
  *
1831
+ * Takes the same options as `QvdDataFrame.fromQvd`, with the same meanings - one option
1832
+ * vocabulary for both read paths, because they are two answers about the same file rather than
1833
+ * two features. `{offset, limit}` is how a caller pages through a file columnwise; there is no
1834
+ * columnar `iterate()` because there is nothing for it to bound - a columnar read materialises
1835
+ * no rows, which is the memory chunking exists to cap.
1836
+ *
1650
1837
  * @param {string} path The path to the QVD file.
1651
1838
  * @param {Object} [options] Loading options, with the same meanings they have on `fromQvd`.
1652
- * @param {number|null} [options.maxRows] Maximum rows to decode.
1839
+ * @param {number|null} [options.maxRows] Rows to decode. The older name for `limit`.
1840
+ * @param {number|null} [options.limit] Rows to decode, counting from `offset`.
1841
+ * @param {number} [options.offset] File row to start at.
1842
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
1843
+ * appear. Unselected fields have their symbols skipped entirely.
1844
+ * @param {Function} [options.onProgress] Progress callback, `{stage, current, total, percent}`.
1845
+ * @param {AbortSignal} [options.signal] Cancels the read.
1653
1846
  * @param {string} [options.allowedDir] Directory the path must resolve inside.
1654
1847
  * @param {number} [options.memorySafetyFactor] Fraction of the memory budget a load may use.
1655
1848
  * @param {number} [options.symbolFilteringThreshold] Symbol table size above which a limited
@@ -1659,15 +1852,13 @@ var init_QvdColumnTable = __esm({
1659
1852
  static async fromQvd(path3, options = {}) {
1660
1853
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
1661
1854
  const reader = new QvdFileReader2(path3, {
1662
- allowedDir: options.allowedDir,
1663
- memorySafetyFactor: options.memorySafetyFactor,
1664
- symbolFilteringThreshold: options.symbolFilteringThreshold,
1855
+ ...readerOptionsFrom(options),
1665
1856
  // This read builds no rows, so the memory guard must not charge it for them. A columnar
1666
1857
  // read of the 38MB taxi fixture completes in a 15MB heap; charged the row cost it was
1667
1858
  // refused below a 2GB one.
1668
1859
  materialisesRows: false
1669
1860
  });
1670
- return await reader.loadColumnar(options.maxRows !== void 0 ? options.maxRows : null);
1861
+ return await reader.loadColumnar(windowFrom(options));
1671
1862
  }
1672
1863
  /** @return {Array<string>} Field names, in file order. */
1673
1864
  get columns() {
@@ -1715,7 +1906,7 @@ var QvdFileReader_exports = {};
1715
1906
  __export(QvdFileReader_exports, {
1716
1907
  QvdFileReader: () => QvdFileReader
1717
1908
  });
1718
- var MAX_HEADER_SIZE, READ_CHUNK_SIZE, QvdFileReader;
1909
+ var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, QvdFileReader;
1719
1910
  var init_QvdFileReader = __esm({
1720
1911
  "src/QvdFileReader.js"() {
1721
1912
  init_QvdDataFrame();
@@ -1725,8 +1916,10 @@ var init_QvdFileReader = __esm({
1725
1916
  init_memoryUtils();
1726
1917
  init_validationUtils();
1727
1918
  init_symbolParser();
1919
+ init_readOptions();
1728
1920
  MAX_HEADER_SIZE = 16 * 1024 * 1024;
1729
1921
  READ_CHUNK_SIZE = 512 * 1024 * 1024;
1922
+ ANALYSIS_SLICE_ROWS = 65536;
1730
1923
  QvdFileReader = class {
1731
1924
  /**
1732
1925
  * Constructs a new QVD file parser.
@@ -1738,9 +1931,10 @@ var init_QvdFileReader = __esm({
1738
1931
  * points outside it is rejected. Defaults to the current working directory. To permit
1739
1932
  * an entire volume, pass its root explicitly ('/' on POSIX, 'C:\\' on Windows); a null or
1740
1933
  * empty value falls back to the working directory rather than removing the restriction.
1741
- * @param {number} [options.memorySafetyFactor=0.3] Fraction (0.0-1.0) of the memory budget a
1742
- * load may use. The budget is the smallest of the V8 heap limit, any container memory limit,
1743
- * and the memory the OS reports as available. Default is 0.3. **Zero disables the memory
1934
+ * @param {number} [options.memorySafetyFactor=0.8] Fraction (0.0-1.0) of the memory budget a
1935
+ * load may use. The budget is the smaller of the V8 heap limit and any container memory limit;
1936
+ * what the OS reports as available is recorded for diagnostics and deliberately not allowed to
1937
+ * bind - see `getMemoryBudget`. Default is 0.8. **Zero disables the memory
1744
1938
  * check entirely**, which is the escape hatch for runtimes whose limits cannot be measured -
1745
1939
  * Bun reports its current heap as its heap limit - and for callers who would rather manage
1746
1940
  * memory themselves than trust the estimate.
@@ -1751,29 +1945,93 @@ var init_QvdFileReader = __esm({
1751
1945
  * above which a lazy load switches to the two-pass filtering path. The default of 50MB is
1752
1946
  * the point where the extra analysis pass pays for itself; lower it to use filtering on
1753
1947
  * smaller files, raise it to keep the simpler single-pass read for longer.
1948
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
1949
+ * appear. Null reads every field, in file order. An unknown or repeated name is refused.
1950
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
1951
+ * read proceeds - the same shape `QvdFileWriter` emits.
1952
+ * @param {AbortSignal} [options.signal] Cancels the read. When it is aborted the read throws
1953
+ * `signal.reason`, exactly as `signal.throwIfAborted()` does.
1754
1954
  */
1755
1955
  constructor(filePath, options = {}) {
1756
1956
  const {
1757
1957
  allowedDir,
1758
1958
  memorySafetyFactor = 0.8,
1759
1959
  symbolFilteringThreshold = 50 * 1024 * 1024,
1760
- materialisesRows = true
1960
+ materialisesRows = true,
1961
+ fields = null,
1962
+ onProgress,
1963
+ signal
1761
1964
  } = options;
1762
1965
  this._materialisesRows = materialisesRows;
1763
1966
  this._path = validatePath(filePath, allowedDir);
1764
1967
  this._memorySafetyFactor = memorySafetyFactor;
1765
1968
  this._symbolFilteringThreshold = symbolFilteringThreshold;
1969
+ if (onProgress !== void 0 && typeof onProgress !== "function") {
1970
+ throw new QvdValidationError("onProgress must be a function", {
1971
+ provided: onProgress,
1972
+ type: typeof onProgress,
1973
+ file: this._path
1974
+ });
1975
+ }
1976
+ if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
1977
+ throw new QvdValidationError("signal must be an AbortSignal", {
1978
+ provided: signal,
1979
+ type: typeof signal,
1980
+ file: this._path
1981
+ });
1982
+ }
1983
+ this._requestedFields = fields === void 0 ? null : fields;
1984
+ this._onProgress = onProgress;
1985
+ this._signal = signal;
1766
1986
  this._buffer = null;
1767
1987
  this._headerOffset = null;
1768
1988
  this._symbolTableOffset = null;
1769
1989
  this._indexTableOffset = null;
1770
1990
  this._header = null;
1991
+ this._allFields = null;
1992
+ this._selectedFields = null;
1993
+ this._fieldBitMetadataValidated = false;
1771
1994
  this._symbolTable = null;
1772
1995
  this._indexColumns = null;
1773
1996
  this._rowsDecoded = 0;
1997
+ this._bufferFirstRow = 0;
1774
1998
  this._fileSize = null;
1775
1999
  this._headerMatchesFile = false;
1776
2000
  }
2001
+ /**
2002
+ * Emits a progress event if a callback is registered.
2003
+ *
2004
+ * The same shape `QvdFileWriter._emitProgress` emits, deliberately: a caller who has written a
2005
+ * progress bar for a write should not have to write a second one for a read. The stage names
2006
+ * differ because the stages differ, but `symbol-table` and `index-table` mean the same thing on
2007
+ * both sides.
2008
+ *
2009
+ * @param {string} stage The current stage of the read.
2010
+ * @param {number} current The current progress value.
2011
+ * @param {number} total The total progress value.
2012
+ * @private
2013
+ */
2014
+ _emitProgress(stage, current, total) {
2015
+ if (this._onProgress) {
2016
+ const percent = total > 0 ? Math.round(current / total * 100) : 100;
2017
+ this._onProgress({ stage, current, total, percent });
2018
+ }
2019
+ }
2020
+ /**
2021
+ * Throws if the caller has cancelled the read.
2022
+ *
2023
+ * Throws `signal.reason` - a `DOMException` named `AbortError` unless the caller aborted with a
2024
+ * reason of their own. That is what `AbortSignal` means everywhere else in Node, and inventing
2025
+ * a `QvdAbortError` here would make this library's cancellation the one a caller has to special
2026
+ * case.
2027
+ *
2028
+ * @private
2029
+ */
2030
+ _throwIfAborted() {
2031
+ if (this._signal) {
2032
+ this._signal.throwIfAborted();
2033
+ }
2034
+ }
1777
2035
  /**
1778
2036
  * Reads the binary data of the QVD file.
1779
2037
  *
@@ -1798,14 +2056,23 @@ var init_QvdFileReader = __esm({
1798
2056
  * - Streaming for header finding is efficient for unknown header sizes
1799
2057
  * - Direct byte-range reading for remaining data is fastest
1800
2058
  *
1801
- * @param {number|null} maxRows The maximum number of rows to load. If null, all data is loaded.
2059
+ * A window with a non-zero `offset` reads two ranges rather than one: the header and symbol
2060
+ * table from the front of the file, and the window's records from wherever they sit. The bytes
2061
+ * between are never read, which is what makes `{offset: 1_700_000, limit: 100}` on the taxi
2062
+ * fixture a 0.4MB read rather than a 38MB one.
2063
+ *
2064
+ * @param {QvdRowWindow} window The rows to read.
1802
2065
  * @param {boolean} [headerOnly=false] Stop once the XML header has been read, leaving the
1803
2066
  * symbol and index tables on disk. This is the metadata-only path: the header is a few
1804
2067
  * kilobytes whatever the file's size, so reading a schema costs the same for a 40MB file as
1805
2068
  * for a 40GB one.
2069
+ * @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
2070
+ * that is fewer than the window covers - see `_prepare`.
1806
2071
  * @private
1807
2072
  */
1808
- async _readData(maxRows = null, headerOnly = false) {
2073
+ async _readData(window = { offset: 0, limit: null }, headerOnly = false, liveRows = null) {
2074
+ this._throwIfAborted();
2075
+ this._emitProgress("read", 0, 1);
1809
2076
  const HEADER_DELIMITER = "\r\n\0";
1810
2077
  const CHUNK_SIZE = 64 * 1024;
1811
2078
  const stream = fs.createReadStream(this._path, {
@@ -1875,13 +2142,14 @@ var init_QvdFileReader = __esm({
1875
2142
  const totalRows = parseInt(headerObj["QvdTableHeader"]["NoOfRecords"], 10);
1876
2143
  if (headerOnly) {
1877
2144
  this._buffer = headerBuffer.subarray(0, headerEndIndex);
2145
+ this._emitProgress("read", 1, 1);
1878
2146
  return;
1879
2147
  }
1880
2148
  let headerFields = headerObj["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
1881
2149
  if (headerFields && !Array.isArray(headerFields)) {
1882
2150
  headerFields = [headerFields];
1883
2151
  }
1884
- const columnCount = Array.isArray(headerFields) ? headerFields.length : 0;
2152
+ const columnCount = Array.isArray(headerFields) ? selectFields(headerFields, this._requestedFields, this._path).length : 0;
1885
2153
  const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
1886
2154
  (value) => Number.isSafeInteger(value) && value >= 0
1887
2155
  );
@@ -1890,23 +2158,28 @@ var init_QvdFileReader = __esm({
1890
2158
  this._fileSize = fileSize;
1891
2159
  this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize;
1892
2160
  }
2161
+ const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
2162
+ const windowRows = resolved.limit;
1893
2163
  if (headerNumbersUsable && this._headerMatchesFile) {
1894
2164
  validateMemoryAvailability(
1895
2165
  symbolTableLength,
1896
- maxRows,
2166
+ windowRows,
1897
2167
  totalRows,
1898
2168
  this._path,
1899
2169
  this._memorySafetyFactor,
1900
2170
  columnCount,
1901
- this._materialisesRows
2171
+ this._materialisesRows,
2172
+ liveRows
1902
2173
  );
1903
2174
  }
1904
- if (maxRows === null) {
2175
+ if (window.offset === 0 && window.limit === null) {
1905
2176
  this._buffer = await fs.promises.readFile(this._path);
1906
2177
  this._fileSize = this._buffer.length;
2178
+ this._bufferFirstRow = 0;
2179
+ this._emitProgress("read", 1, 1);
1907
2180
  return;
1908
2181
  }
1909
- const rowsToLoad = Math.min(maxRows, totalRows);
2182
+ const rowsToLoad = windowRows;
1910
2183
  validateSymbolTableSizeEarly(symbolTableLength, this._path);
1911
2184
  for (const [name, value] of [
1912
2185
  ["Offset", symbolTableLength],
@@ -1922,39 +2195,78 @@ var init_QvdFileReader = __esm({
1922
2195
  });
1923
2196
  }
1924
2197
  }
2198
+ const skippedIndexBytes = resolved.offset * recordSize;
1925
2199
  const indexTableBytesToRead = rowsToLoad * recordSize;
1926
2200
  const totalBytesToRead = indexTableOffset + indexTableBytesToRead;
2201
+ const fileBytesRequired = indexTableOffset + skippedIndexBytes + indexTableBytesToRead;
1927
2202
  const fd = await fs.promises.open(this._path, "r");
1928
2203
  try {
1929
2204
  const { size: fileSize } = await fd.stat();
1930
2205
  this._fileSize = fileSize;
1931
- if (totalBytesToRead > fileSize) {
2206
+ if (fileBytesRequired > fileSize) {
1932
2207
  throw new QvdCorruptedError("The file is shorter than its header claims.", {
1933
2208
  file: this._path,
1934
2209
  fileSize,
1935
- requiredBytes: totalBytesToRead,
2210
+ requiredBytes: fileBytesRequired,
1936
2211
  stage: "readData"
1937
2212
  });
1938
2213
  }
1939
2214
  this._buffer = Buffer.alloc(totalBytesToRead);
1940
- let position = 0;
1941
- while (position < totalBytesToRead) {
1942
- const length = Math.min(READ_CHUNK_SIZE, totalBytesToRead - position);
1943
- const { bytesRead } = await fd.read(this._buffer, position, length, position);
1944
- if (bytesRead === 0) {
1945
- throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
1946
- file: this._path,
1947
- fileSize,
1948
- bytesRead: position,
1949
- requiredBytes: totalBytesToRead,
1950
- stage: "readData"
1951
- });
1952
- }
1953
- position += bytesRead;
2215
+ await this._readRange(fd, 0, indexTableOffset, 0, fileSize, totalBytesToRead);
2216
+ if (indexTableBytesToRead > 0) {
2217
+ await this._readRange(
2218
+ fd,
2219
+ indexTableOffset,
2220
+ indexTableBytesToRead,
2221
+ indexTableOffset + skippedIndexBytes,
2222
+ fileSize,
2223
+ fileBytesRequired
2224
+ );
1954
2225
  }
2226
+ this._bufferFirstRow = resolved.offset;
1955
2227
  } finally {
1956
2228
  await fd.close();
1957
2229
  }
2230
+ this._emitProgress("read", 1, 1);
2231
+ }
2232
+ /**
2233
+ * Reads one byte range of the file into the buffer.
2234
+ *
2235
+ * Read in bounded chunks, checking bytesRead each time. A single fs.read call with a length of
2236
+ * 2^31 or more does not throw - it trips a C++ assertion and aborts the whole process, which no
2237
+ * try/catch can intercept.
2238
+ *
2239
+ * @param {import('fs/promises').FileHandle} fd The open file.
2240
+ * @param {number} bufferOffset Where in the buffer to write.
2241
+ * @param {number} byteCount How many bytes to read.
2242
+ * @param {number} filePosition Where in the file to read from.
2243
+ * @param {number} fileSize The file's size, for the error.
2244
+ * @param {number} requiredBytes Bytes the whole read needs, for the error.
2245
+ * @private
2246
+ */
2247
+ async _readRange(fd, bufferOffset, byteCount, filePosition, fileSize, requiredBytes) {
2248
+ assert2(this._buffer, "The read buffer has not been allocated.");
2249
+ let done = 0;
2250
+ while (done < byteCount) {
2251
+ const length = Math.min(READ_CHUNK_SIZE, byteCount - done);
2252
+ const { bytesRead } = await fd.read(this._buffer, bufferOffset + done, length, filePosition + done);
2253
+ if (bytesRead === 0) {
2254
+ throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
2255
+ file: this._path,
2256
+ fileSize,
2257
+ // Two numbers, because they stopped being the same one when a window began reading two
2258
+ // ranges: `bytesRead` is how much of this range arrived, `filePosition` is where in the
2259
+ // file it gave up. Reporting the position under the name of the count made a windowed
2260
+ // read of a truncated file claim tens of megabytes had been read when a few hundred
2261
+ // bytes had.
2262
+ bytesRead: done,
2263
+ filePosition: filePosition + done,
2264
+ requiredBytes,
2265
+ stage: "readData"
2266
+ });
2267
+ }
2268
+ done += bytesRead;
2269
+ }
1958
2270
  }
1959
2271
  /**
1960
2272
  * Parses the XML header of the QVD file. This method is part of the parsing process
@@ -2014,6 +2326,8 @@ var init_QvdFileReader = __esm({
2014
2326
  this._headerOffset = headerBeginIndex;
2015
2327
  this._symbolTableOffset = headerEndIndex;
2016
2328
  this._indexTableOffset = this._symbolTableOffset + parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2329
+ this._allFields = fieldList;
2330
+ this._selectedFields = selectFields(this._allFields, this._requestedFields, this._path);
2017
2331
  }
2018
2332
  /**
2019
2333
  * Establishes the geometry of the index table, and validates it.
@@ -2024,14 +2338,15 @@ var init_QvdFileReader = __esm({
2024
2338
  * about keeping the sign in step with the other one: the two could drift, and #113 is what
2025
2339
  * that looks like when they do. There is one copy now.
2026
2340
  *
2027
- * @param {number|null} rowLimit Maximum rows of interest, or null for all of them.
2341
+ * @param {QvdRowWindow} window The rows of interest, as file row indices.
2028
2342
  * @param {string} stage Stage name for any error raised here.
2029
2343
  * @return {{fields: Array<any>, recordSize: number, totalRows: number, rowsToLoad: number,
2030
- * indexBuffer: Buffer}} The record geometry.
2344
+ * indexBuffer: Buffer}} The record geometry. `indexBuffer` starts at the window's first
2345
+ * record, so the decoder always counts from zero.
2031
2346
  * @private
2032
2347
  */
2033
- _planIndexTable(rowLimit, stage) {
2034
- if (!this._buffer || !this._header || !this._indexTableOffset) {
2348
+ _planIndexTable(window, stage) {
2349
+ if (!this._buffer || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
2035
2350
  throw new QvdCorruptedError(
2036
2351
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
2037
2352
  {
@@ -2040,14 +2355,12 @@ var init_QvdFileReader = __esm({
2040
2355
  }
2041
2356
  );
2042
2357
  }
2043
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2044
- if (!Array.isArray(fields)) {
2045
- fields = [fields];
2046
- }
2358
+ const allFields = this._allFields;
2359
+ const fields = this._selectedFields;
2047
2360
  const recordSize = parseInt(this._header["QvdTableHeader"]["RecordByteSize"], 10);
2048
2361
  const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
2049
- const rowsToLoad = rowLimit !== null ? Math.min(rowLimit, totalRows) : totalRows;
2050
2362
  const indexTableLength = parseInt(this._header["QvdTableHeader"]["Length"], 10);
2363
+ const { offset: firstRow, limit: rowsToLoad } = resolveWindow(window, totalRows);
2051
2364
  validateIndexTableMetadata(
2052
2365
  recordSize,
2053
2366
  totalRows,
@@ -2056,11 +2369,20 @@ var init_QvdFileReader = __esm({
2056
2369
  this._buffer.length,
2057
2370
  rowsToLoad,
2058
2371
  this._path,
2059
- this._fileSize
2372
+ this._fileSize,
2373
+ firstRow,
2374
+ this._bufferFirstRow
2375
+ );
2376
+ const bufferRecordStart = (firstRow - this._bufferFirstRow) * recordSize;
2377
+ const indexBuffer = this._buffer.subarray(
2378
+ this._indexTableOffset + bufferRecordStart,
2379
+ this._indexTableOffset + bufferRecordStart + rowsToLoad * recordSize
2060
2380
  );
2061
- const indexBuffer = this._buffer.subarray(this._indexTableOffset, this._indexTableOffset + indexTableLength + 1);
2062
- for (const field of fields) {
2063
- validateFieldBitMetadata(field, recordSize, this._path);
2381
+ if (!this._fieldBitMetadataValidated) {
2382
+ for (const field of allFields) {
2383
+ validateFieldBitMetadata(field, recordSize, this._path);
2384
+ }
2385
+ this._fieldBitMetadataValidated = true;
2064
2386
  }
2065
2387
  assert2(
2066
2388
  rowsToLoad === 0 || recordSize === 0 || Math.floor(indexBuffer.length / recordSize) >= rowsToLoad,
@@ -2072,31 +2394,45 @@ var init_QvdFileReader = __esm({
2072
2394
  * Analyzes the index table to determine which symbols are actually needed.
2073
2395
  * This is used for two-pass symbol filtering optimization.
2074
2396
  *
2075
- * @param {number} maxRows The maximum number of rows to analyze.
2076
- * @return {Promise<Map<string, Set<number>>>} Map of field names to Set of needed symbol indices.
2397
+ * Only the selected fields are analysed. An unselected field's symbols are never parsed, so
2398
+ * there is nothing for a usage set to filter and decoding its column would be a pass over the
2399
+ * whole window for an answer nobody reads.
2400
+ *
2401
+ * @param {QvdRowWindow} window The rows to analyse.
2402
+ * @return {Promise<Array<Set<number>>>} One set of needed symbol indices per selected field, in
2403
+ * the same order `_parseSymbolTable` walks them.
2077
2404
  * @private
2078
2405
  */
2079
- async _analyzeIndexTableSymbolUsage(maxRows) {
2080
- const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(maxRows, "analyzeIndexTableSymbolUsage");
2081
- const symbolUsage = /* @__PURE__ */ new Map();
2082
- const column = new Int32Array(rowsToLoad);
2083
- fields.forEach((field) => {
2406
+ async _analyzeIndexTableSymbolUsage(window) {
2407
+ const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
2408
+ const symbolUsage = [];
2409
+ const sliceRows = Math.min(rowsToLoad, ANALYSIS_SLICE_ROWS);
2410
+ const column = new Int32Array(sliceRows);
2411
+ fields.forEach((field, position) => {
2412
+ this._throwIfAborted();
2084
2413
  const needed = /* @__PURE__ */ new Set();
2085
- symbolUsage.set(field["FieldName"], needed);
2086
- decodeIndexColumn(
2087
- indexBuffer,
2088
- recordSize,
2089
- rowsToLoad,
2090
- parseInt(field["BitOffset"], 10),
2091
- parseInt(field["BitWidth"], 10),
2092
- parseInt(field["Bias"], 10),
2093
- column
2094
- );
2095
- for (let row = 0; row < rowsToLoad; row++) {
2096
- if (column[row] >= 0) {
2097
- needed.add(column[row]);
2414
+ symbolUsage[position] = needed;
2415
+ const bitOffset = parseInt(field["BitOffset"], 10);
2416
+ const bitWidth = parseInt(field["BitWidth"], 10);
2417
+ const bias = parseInt(field["Bias"], 10);
2418
+ for (let first = 0; first < rowsToLoad; first += sliceRows) {
2419
+ const count = Math.min(sliceRows, rowsToLoad - first);
2420
+ decodeIndexColumn(
2421
+ first === 0 ? indexBuffer : indexBuffer.subarray(first * recordSize),
2422
+ recordSize,
2423
+ count,
2424
+ bitOffset,
2425
+ bitWidth,
2426
+ bias,
2427
+ column
2428
+ );
2429
+ for (let row = 0; row < count; row++) {
2430
+ if (column[row] >= 0) {
2431
+ needed.add(column[row]);
2432
+ }
2098
2433
  }
2099
2434
  }
2435
+ this._emitProgress("symbol-analysis", position + 1, fields.length);
2100
2436
  });
2101
2437
  return symbolUsage;
2102
2438
  }
@@ -2104,12 +2440,20 @@ var init_QvdFileReader = __esm({
2104
2440
  * Parses the symbol table of the QVD file. This method is part of the parsing process
2105
2441
  * and should not be called directly.
2106
2442
  *
2107
- * @param {Map<string, Set<number>>|null} symbolsToKeep Optional map of field names to symbol indices to keep.
2108
- * If provided, only these symbols will be parsed (two-pass filtering optimization).
2109
- * @param {number|null} maxRows Optional maximum number of rows being loaded (for memory estimation).
2443
+ * A field the caller did not select is skipped whole. Its symbol area is neither scanned nor
2444
+ * parsed - the per-field `Offset` and `Length` say exactly where it is, so there is nothing to
2445
+ * walk past - and that is where field selection earns its keep. The index decode is cheap by
2446
+ * comparison; parsing symbols is not.
2447
+ *
2448
+ * @param {Array<Set<number>>|null} symbolsToKeep Optional set of symbol indices to keep per
2449
+ * selected field, indexed by position. If provided, only these symbols will be parsed
2450
+ * (two-pass filtering optimization).
2451
+ * @param {number} rowsToLoad Rows the read covers, for memory estimation.
2452
+ * @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
2453
+ * that is fewer than the window covers - see `_prepare`.
2110
2454
  */
2111
- async _parseSymbolTable(symbolsToKeep = null, maxRows = null) {
2112
- if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset) {
2455
+ async _parseSymbolTable(symbolsToKeep = null, rowsToLoad = 0, liveRows = null) {
2456
+ if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
2113
2457
  throw new QvdCorruptedError(
2114
2458
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
2115
2459
  {
@@ -2118,7 +2462,8 @@ var init_QvdFileReader = __esm({
2118
2462
  }
2119
2463
  );
2120
2464
  }
2121
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2465
+ const allFields = this._allFields;
2466
+ const fields = this._selectedFields;
2122
2467
  const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
2123
2468
  const symbolTableSize = symbolBuffer.length;
2124
2469
  const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
@@ -2126,32 +2471,25 @@ var init_QvdFileReader = __esm({
2126
2471
  if (this._headerMatchesFile) {
2127
2472
  validateMemoryAvailability(
2128
2473
  symbolTableSize,
2129
- maxRows,
2474
+ rowsToLoad,
2130
2475
  totalRows,
2131
2476
  this._path,
2132
2477
  this._memorySafetyFactor,
2133
- Array.isArray(fields) ? fields.length : 1,
2134
- this._materialisesRows
2478
+ fields.length,
2479
+ this._materialisesRows,
2480
+ liveRows
2135
2481
  );
2136
2482
  }
2137
- warnLargeSymbolTable(
2138
- symbolTableSize,
2139
- maxRows,
2140
- totalRows,
2141
- Array.isArray(fields) ? fields.length : 1,
2142
- this._materialisesRows
2143
- );
2144
- if (!Array.isArray(fields)) {
2145
- fields = [fields];
2146
- }
2147
- for (const field of fields) {
2483
+ warnLargeSymbolTable(symbolTableSize, rowsToLoad, totalRows, fields.length, this._materialisesRows);
2484
+ for (const field of allFields) {
2148
2485
  validateFieldMetadata(field, symbolBuffer.length, this._path);
2149
2486
  }
2150
- this._symbolTable = fields.map((field) => {
2487
+ this._symbolTable = fields.map((field, position) => {
2488
+ this._throwIfAborted();
2151
2489
  const symbolsOffset = parseInt(field["Offset"], 10);
2152
2490
  const symbolsLength = parseInt(field["Length"], 10);
2153
2491
  const fieldName = field["FieldName"];
2154
- const neededSymbols = symbolsToKeep ? symbolsToKeep.get(fieldName) : null;
2492
+ const neededSymbols = symbolsToKeep ? symbolsToKeep[position] : null;
2155
2493
  const filteringEnabled = neededSymbols !== null;
2156
2494
  const symbols = [];
2157
2495
  let symbolIndex = 0;
@@ -2171,6 +2509,7 @@ var init_QvdFileReader = __esm({
2171
2509
  pointer += bytesRead - 1;
2172
2510
  symbolIndex++;
2173
2511
  }
2512
+ this._emitProgress("symbol-table", position + 1, fields.length);
2174
2513
  return symbols;
2175
2514
  });
2176
2515
  }
@@ -2189,13 +2528,18 @@ var init_QvdFileReader = __esm({
2189
2528
  * same for every row, so they are hoisted out of the loop and the inner loop does arithmetic
2190
2529
  * into a typed array and nothing else. Rows are assembled later, once, in `load()`.
2191
2530
  *
2192
- * @param {number|null} maxRows The maximum number of rows to parse. If null, all rows are parsed.
2531
+ * The window is what makes chunked iteration cheap: `decodeIndexColumn` walks records by
2532
+ * `base += recordSize`, so decoding rows k to k+n is a question of where the buffer slice starts
2533
+ * and how many iterations run. Nothing about the decoder changed to support it.
2534
+ *
2535
+ * @param {QvdRowWindow} window The rows to decode.
2193
2536
  */
2194
- async _parseIndexTable(maxRows = null) {
2195
- const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(maxRows, "parseIndexTable");
2537
+ async _parseIndexTable(window) {
2538
+ const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "parseIndexTable");
2196
2539
  this._rowsDecoded = rowsToLoad;
2197
- this._indexColumns = fields.map(
2198
- (field) => decodeIndexColumn(
2540
+ this._indexColumns = fields.map((field, position) => {
2541
+ this._throwIfAborted();
2542
+ const column = decodeIndexColumn(
2199
2543
  indexBuffer,
2200
2544
  recordSize,
2201
2545
  rowsToLoad,
@@ -2203,8 +2547,10 @@ var init_QvdFileReader = __esm({
2203
2547
  parseInt(field["BitWidth"], 10),
2204
2548
  parseInt(field["Bias"], 10),
2205
2549
  new Int32Array(rowsToLoad)
2206
- )
2207
- );
2550
+ );
2551
+ this._emitProgress("index-table", position + 1, fields.length);
2552
+ return column;
2553
+ });
2208
2554
  }
2209
2555
  /**
2210
2556
  * Reads the file's schema and header metadata, without touching the symbol or index tables.
@@ -2221,8 +2567,11 @@ var init_QvdFileReader = __esm({
2221
2567
  * @return {Promise<import('./QvdDataFrame.js').QvdFileMetadata>} The file's schema and header.
2222
2568
  */
2223
2569
  async loadMetadata() {
2224
- await this._readData(null, true);
2570
+ await this._readData({ offset: 0, limit: null }, true);
2571
+ this._emitProgress("header", 0, 1);
2225
2572
  await this._parseHeader();
2573
+ this._emitProgress("header", 1, 1);
2574
+ this._throwIfAborted();
2226
2575
  assert2(this._header, "The QVD file header has not been parsed.");
2227
2576
  const header = this._header["QvdTableHeader"];
2228
2577
  let fields = header["Fields"]?.["QvdFieldHeader"] ?? [];
@@ -2257,114 +2606,200 @@ var init_QvdFileReader = __esm({
2257
2606
  /**
2258
2607
  * Loads the QVD file into memory and parses it.
2259
2608
  *
2260
- * @param {number|null} maxRows The maximum number of rows to load. If null, all rows are loaded.
2261
- * Must be a non-negative integer when given.
2262
- * @throws {QvdValidationError} If maxRows is neither null nor a non-negative integer.
2609
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
2610
+ * The rows to load. A number or null means what it always meant - the first N rows, or all of
2611
+ * them - and `{offset, limit}` is the same thing said more precisely, so `5` and
2612
+ * `{offset: 0, limit: 5}` are one read. `maxRows` is accepted as a second name for `limit`.
2613
+ * @throws {QvdValidationError} If the window is not a non-negative integer, null, or a valid
2614
+ * `{offset, limit}` object.
2263
2615
  * @return {Promise<QvdDataFrame>} The loaded QVD file.
2264
2616
  */
2265
- async load(maxRows = null) {
2266
- const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
2617
+ async load(window = null) {
2618
+ const rows = normaliseWindow(window, this._path);
2619
+ const prepared = await this._prepare(rows);
2620
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
2621
+ const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
2622
+ return new QvdDataFrame(data, prepared.columns, prepared.metadata, {
2623
+ ...prepared.loadStats,
2624
+ rowsLoaded: data.length
2625
+ });
2626
+ }
2627
+ /**
2628
+ * Reads the file as columns, without ever materialising rows.
2629
+ *
2630
+ * Shares every step with `load()` up to the point where rows would be built - see `_prepare`.
2631
+ * What it keeps instead is what the decoder already produced: one `Int32Array` of stored
2632
+ * indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
2633
+ * fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
2634
+ * bytes per row rather than a boxed value per cell, and the symbols are a few thousand
2635
+ * entries shared across every row that uses them.
2636
+ *
2637
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
2638
+ * The rows to decode, in the same spellings `load()` accepts.
2639
+ * @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
2640
+ */
2641
+ async loadColumnar(window = null) {
2642
+ const rows = normaliseWindow(window, this._path);
2643
+ const prepared = await this._prepare(rows);
2644
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
2645
+ const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
2267
2646
  assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2268
- const indexColumns = this._indexColumns;
2269
- const fieldCount = indexColumns.length;
2270
- const data = new Array(this._rowsDecoded);
2271
- for (let row = 0; row < this._rowsDecoded; row++) {
2272
- const values = new Array(fieldCount);
2273
- for (let field = 0; field < fieldCount; field++) {
2274
- const symbolIndex = indexColumns[field][row];
2275
- values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
2276
- }
2277
- data[row] = values;
2278
- }
2279
- loadStats.rowsLoaded = data.length;
2280
- return new QvdDataFrame(data, columns, metadata, loadStats);
2647
+ return new QvdColumnTable2({
2648
+ columns: prepared.columns,
2649
+ codesByField: this._indexColumns,
2650
+ symbolsByField: prepared.resolvedByField,
2651
+ rowCount: this._rowsDecoded,
2652
+ metadata: prepared.metadata,
2653
+ loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
2654
+ });
2281
2655
  }
2282
2656
  /**
2283
- * Reads and decodes the file, stopping short of building rows.
2657
+ * Yields the window as data frames of at most `chunkSize` rows.
2284
2658
  *
2285
- * Everything `load()` and `loadColumnar()` have in common, which is everything except the
2286
- * shape of the answer. Two read paths for one binary format is the drift risk #113 is the
2287
- * standing example of - a stored index resolved one way here and another way there returns
2288
- * plausible wrong values and throws nothing - so there is one path, and the two entry points
2289
- * differ only in what they do with what it returns.
2659
+ * The file is opened, read and parsed **once**; only the index decode and the row building
2660
+ * happen per chunk. That is the whole reason this exists as a method rather than as a loop of
2661
+ * `load({offset, limit})` calls at the call site: the symbol table has to be parsed in full
2662
+ * whatever the chunk size - a stored index in the last chunk can address the first symbol -
2663
+ * and re-parsing it per chunk is what makes the obvious implementation cost more than a plain
2664
+ * load rather than less. PyQvd's chunked read does re-read it, and the comment on #140 records
2665
+ * that as a limitation rather than a design.
2290
2666
  *
2291
- * @param {number|null} maxRows Maximum rows to decode, or null for all of them.
2292
- * @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
2293
- * resolvedByField: Array<Array<any>>}>} The decoded file.
2294
- * @private
2667
+ * What it bounds is row materialisation, which is what actually dominates a large read's heap.
2668
+ * Two chunks of rows are alive at a time, not one - `for await` keeps the yielded frame
2669
+ * reachable while this generator builds the next - which is why `liveRows` below is
2670
+ * `chunkSize * 2`, and why the heap it needs is twice what one chunk suggests.
2671
+ *
2672
+ * A window covering no rows yields nothing at all, rather than one empty frame - so
2673
+ * `for await` over an exhausted offset does nothing, which is what a paging loop wants.
2674
+ *
2675
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} window
2676
+ * The rows to cover, in the same spellings `load()` accepts.
2677
+ * @param {number} chunkSize Rows per frame. Must be a positive integer.
2678
+ * @return {AsyncGenerator<QvdDataFrame>} The chunks, in order.
2295
2679
  */
2296
- async _decode(maxRows = null) {
2297
- if (maxRows !== null && (typeof maxRows !== "number" || !Number.isInteger(maxRows) || maxRows < 0)) {
2298
- throw new QvdValidationError("maxRows must be a non-negative integer, or null to load all rows", {
2299
- provided: maxRows,
2300
- type: typeof maxRows,
2680
+ async *iterateRows(window, chunkSize) {
2681
+ if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
2682
+ throw new QvdValidationError("chunkSize must be a positive integer", {
2683
+ provided: chunkSize,
2684
+ type: typeof chunkSize,
2301
2685
  file: this._path
2302
2686
  });
2303
2687
  }
2304
- await this._readData(maxRows);
2688
+ const liveRows = { rows: chunkSize * 2, perChunk: 2 };
2689
+ const rows = normaliseWindow(window, this._path);
2690
+ const prepared = await this._prepare(rows, liveRows);
2691
+ for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
2692
+ this._throwIfAborted();
2693
+ const count = Math.min(chunkSize, prepared.rowsAvailable - done);
2694
+ const offset = prepared.offset + done;
2695
+ await this._parseIndexTable({ offset, limit: count });
2696
+ const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
2697
+ yield new QvdDataFrame(data, prepared.columns, prepared.metadata, {
2698
+ ...prepared.loadStats,
2699
+ offset,
2700
+ rowsLoaded: data.length
2701
+ });
2702
+ }
2703
+ }
2704
+ /**
2705
+ * Reads the file and resolves its symbols, stopping short of decoding any rows.
2706
+ *
2707
+ * Everything `load()`, `loadColumnar()` and `iterateRows()` have in common, which is everything
2708
+ * that depends on the file rather than on the window. Two read paths for one binary format is
2709
+ * the drift risk #113 is the standing example of - a stored index resolved one way here and
2710
+ * another way there returns plausible wrong values and throws nothing - so there is one path,
2711
+ * and the entry points differ only in what they do with what it returns and how many rows they
2712
+ * ask for at a time.
2713
+ *
2714
+ * @param {QvdRowWindow} window The rows the read covers.
2715
+ * @param {{rows: number, perChunk: number}|null} [liveRows] Rows held at one instant when that
2716
+ * is fewer than the window covers, and how many of them one row of the caller's chunk size
2717
+ * accounts for. Only `iterateRows` passes it; every other read holds what it covers.
2718
+ * @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
2719
+ * resolvedByField: Array<Array<any>>, rowsAvailable: number, offset: number}>} The parsed
2720
+ * file, with the window as it resolved against it.
2721
+ * @private
2722
+ */
2723
+ async _prepare(window, liveRows = null) {
2724
+ this._throwIfAborted();
2725
+ await this._readData(window, false, liveRows);
2726
+ this._emitProgress("header", 0, 1);
2305
2727
  await this._parseHeader();
2728
+ this._emitProgress("header", 1, 1);
2729
+ this._throwIfAborted();
2730
+ assert2(this._header, "The QVD file header has not been parsed.");
2731
+ const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
2732
+ const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2733
+ const resolved = resolveWindow(window, totalRows);
2734
+ const rowsAvailable = resolved.limit;
2306
2735
  let symbolsToKeep = null;
2307
2736
  let symbolsKept = null;
2308
- if (maxRows !== null && this._header) {
2309
- const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2737
+ if (window.limit !== null || window.offset > 0) {
2310
2738
  if (symbolTableLength > this._symbolFilteringThreshold) {
2311
- symbolsToKeep = await this._analyzeIndexTableSymbolUsage(maxRows);
2312
- symbolsKept = Array.from(symbolsToKeep.values()).reduce((sum, set) => sum + set.size, 0);
2739
+ symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
2740
+ symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
2313
2741
  }
2314
2742
  }
2315
- await this._parseSymbolTable(symbolsToKeep, maxRows);
2316
- await this._parseIndexTable(maxRows);
2317
- assert2(this._header, "The QVD file header has not been parsed.");
2743
+ await this._parseSymbolTable(symbolsToKeep, rowsAvailable, liveRows);
2318
2744
  assert2(this._symbolTable, "The QVD file symbol table has not been parsed.");
2319
- assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2745
+ this._throwIfAborted();
2320
2746
  const resolvedByField = this._symbolTable.map((symbols) => {
2321
- const resolved = new Array(symbols.length);
2747
+ const resolved2 = new Array(symbols.length);
2322
2748
  for (let index = 0; index < symbols.length; index++) {
2323
2749
  const value = symbols[index]?.toPrimaryValue();
2324
- resolved[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
2750
+ resolved2[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
2325
2751
  }
2326
- return resolved;
2752
+ return resolved2;
2327
2753
  });
2328
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2329
- if (!Array.isArray(fields)) {
2330
- fields = [fields];
2331
- }
2332
- const columns = fields.map((field) => field["FieldName"]);
2754
+ assert2(this._selectedFields, "The QVD file fields have not been resolved.");
2755
+ const columns = this._selectedFields.map((field) => field["FieldName"]);
2333
2756
  const metadata = this._header["QvdTableHeader"];
2334
2757
  const loadStats = {
2335
- symbolTableBytes: parseInt(this._header["QvdTableHeader"]["Offset"], 10),
2336
- totalRows: parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10),
2337
- rowsLoaded: this._rowsDecoded,
2758
+ symbolTableBytes: symbolTableLength,
2759
+ totalRows,
2760
+ rowsLoaded: 0,
2761
+ offset: resolved.offset,
2338
2762
  symbolFiltering: symbolsToKeep !== null,
2339
2763
  symbolsKept
2340
2764
  };
2341
- return { columns, metadata, loadStats, resolvedByField };
2765
+ return { columns, metadata, loadStats, resolvedByField, rowsAvailable, offset: resolved.offset };
2342
2766
  }
2343
2767
  /**
2344
- * Reads the file as columns, without ever materialising rows.
2768
+ * Builds rows from the columns currently decoded.
2345
2769
  *
2346
- * Shares every step with `load()` up to the point where rows would be built - see `_decode`.
2347
- * What it keeps instead is what the decoder already produced: one `Int32Array` of stored
2348
- * indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
2349
- * fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
2350
- * bytes per row rather than a boxed value per cell, and the symbols are a few thousand
2351
- * entries shared across every row that uses them.
2770
+ * `data` stays eager: of the four ways this library is used - a full read, a preview already
2771
+ * bounded by a limit, writing an array out, and reading metadata - not one is helped by
2772
+ * materialising a row only when it is touched, and a lazy accessor would cost a proxy, a cache
2773
+ * and mutation semantics to serve none of them. A caller who wants columns without paying for
2774
+ * rows uses `QvdColumnTable`, which stops before this loop.
2352
2775
  *
2353
- * @param {number|null} maxRows The maximum number of rows to decode.
2354
- * @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
2776
+ * @param {Array<Array<any>>} resolvedByField One resolved value per distinct symbol, per field.
2777
+ * @param {number} progressBase Rows already delivered before this call, so that progress over a
2778
+ * chunked iteration counts the whole window rather than restarting at every chunk.
2779
+ * @param {number} progressTotal Rows the whole window covers.
2780
+ * @return {Array<Array<any>>} The rows.
2781
+ * @private
2355
2782
  */
2356
- async loadColumnar(maxRows = null) {
2357
- const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
2358
- const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
2783
+ _buildRows(resolvedByField, progressBase, progressTotal) {
2359
2784
  assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2360
- return new QvdColumnTable2({
2361
- columns,
2362
- codesByField: this._indexColumns,
2363
- symbolsByField: resolvedByField,
2364
- rowCount: this._rowsDecoded,
2365
- metadata,
2366
- loadStats
2367
- });
2785
+ const indexColumns = this._indexColumns;
2786
+ const fieldCount = indexColumns.length;
2787
+ const rowCount = this._rowsDecoded;
2788
+ const data = new Array(rowCount);
2789
+ const reportInterval = Math.max(1, Math.floor(progressTotal / 100));
2790
+ for (let row = 0; row < rowCount; row++) {
2791
+ const values = new Array(fieldCount);
2792
+ for (let field = 0; field < fieldCount; field++) {
2793
+ const symbolIndex = indexColumns[field][row];
2794
+ values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
2795
+ }
2796
+ data[row] = values;
2797
+ if ((progressBase + row + 1) % reportInterval === 0 || row + 1 === rowCount) {
2798
+ this._throwIfAborted();
2799
+ this._emitProgress("rows", progressBase + row + 1, progressTotal);
2800
+ }
2801
+ }
2802
+ return data;
2368
2803
  }
2369
2804
  };
2370
2805
  }
@@ -2375,6 +2810,7 @@ var QvdDataFrame;
2375
2810
  var init_QvdDataFrame = __esm({
2376
2811
  "src/QvdDataFrame.js"() {
2377
2812
  init_QvdErrors();
2813
+ init_readOptions();
2378
2814
  QvdDataFrame = class _QvdDataFrame {
2379
2815
  /**
2380
2816
  * Represents the data frame stored inside a QVD file.
@@ -2417,12 +2853,15 @@ var init_QvdDataFrame = __esm({
2417
2853
  /**
2418
2854
  * Returns statistics about the read that produced this data frame.
2419
2855
  *
2420
- * Only a frame returned by fromQvd() carries these; fromDict(), head() and tail() produce
2421
- * frames that describe no particular read, and report null rather than a stale figure.
2856
+ * Carried by every frame that came from a file - `fromQvd()`, and each chunk `iterate()` yields,
2857
+ * which is how a chunk reports its `offset`. `fromDict()`, `head()`, `tail()`, `rows()` and
2858
+ * `select()` describe no particular read and report null rather than a stale figure.
2422
2859
  *
2423
2860
  * The main use is confirming that a lazy load actually filtered the symbol table:
2424
2861
  * `symbolFiltering` says whether the two-pass path ran, and `symbolsKept` how many symbols
2425
- * survived it, which for a small maxRows should be a tiny fraction of the file's total.
2862
+ * survived it. Note that a bounded read does not filter on its own - the two-pass path engages
2863
+ * only above `symbolFilteringThreshold`, so on a file below it this reports false and every
2864
+ * symbol was parsed however few rows were asked for.
2426
2865
  *
2427
2866
  * @return {QvdLoadStats|null} Load statistics, or null if this frame did not come from a file.
2428
2867
  */
@@ -2799,6 +3238,18 @@ var init_QvdDataFrame = __esm({
2799
3238
  * @param {Object} [options] Optional loading options.
2800
3239
  * @param {number|null} [options.maxRows] The maximum number of rows to load. Must be a non-negative
2801
3240
  * integer; if not specified or null, all rows are loaded. Anything else throws a QvdValidationError.
3241
+ * This is the older name for `limit`; the two are the same option and passing both throws.
3242
+ * @param {number|null} [options.limit] Rows to read, counting from `offset`. The same number as
3243
+ * `maxRows`, spelled so that it reads correctly beside an offset.
3244
+ * @param {number} [options.offset=0] File row to start at. An offset past the end of the file
3245
+ * returns no rows rather than throwing, so a paging loop terminates on its own.
3246
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
3247
+ * appear in the result. Unselected fields have their symbols skipped entirely rather than
3248
+ * parsed and discarded. An unknown or repeated name throws.
3249
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
3250
+ * read proceeds - the same shape `toQvd`'s callback receives.
3251
+ * @param {AbortSignal} [options.signal] Cancels the read. The rejection is `signal.reason`,
3252
+ * which is a `DOMException` named `AbortError` unless you aborted with a reason of your own.
2802
3253
  * @param {string} [options.allowedDir] Optional allowed directory path. If provided, the file path
2803
3254
  * must be within this directory, with symlinks resolved first, so a link inside it that points
2804
3255
  * outside it is rejected. Defaults to the current working directory. To permit an entire
@@ -2811,17 +3262,42 @@ var init_QvdDataFrame = __esm({
2811
3262
  * **Zero disables the memory check entirely.**
2812
3263
  * @param {number} [options.symbolFilteringThreshold=52428800] Symbol table size, in bytes, above which
2813
3264
  * a lazy load switches to the two-pass filtering path. Defaults to 50MB.
2814
- * @throws {QvdValidationError} If options.maxRows is neither null/undefined nor a non-negative integer.
3265
+ * @throws {QvdValidationError} If a window option is not a non-negative integer, if both
3266
+ * `maxRows` and `limit` are given, or if `fields` names a column the file does not have.
2815
3267
  * @return {Promise<QvdDataFrame>} The data frame of the QVD file.
2816
3268
  */
2817
3269
  static async fromQvd(path3, options = {}) {
2818
3270
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
2819
- const readerOptions = {
2820
- allowedDir: options.allowedDir,
2821
- memorySafetyFactor: options.memorySafetyFactor,
2822
- symbolFilteringThreshold: options.symbolFilteringThreshold
2823
- };
2824
- return await new QvdFileReader2(path3, readerOptions).load(options.maxRows !== void 0 ? options.maxRows : null);
3271
+ return await new QvdFileReader2(path3, readerOptionsFrom(options)).load(windowFrom(options));
3272
+ }
3273
+ /**
3274
+ * Reads a QVD file in chunks, as an async generator of data frames.
3275
+ *
3276
+ * The file is opened, read and parsed once; only the index decode and the row building happen
3277
+ * per chunk, so what this bounds is row materialisation - the part that actually dominates a
3278
+ * large read's heap. It is **not** constant-memory reading of an arbitrarily large file: the
3279
+ * symbol table is parsed in full whatever the chunk size, because a stored index in the last
3280
+ * chunk can address the first symbol. On a high-cardinality file that table is the bulk of the
3281
+ * cost, and `readMetadata` is the only read that avoids it.
3282
+ *
3283
+ * ```js
3284
+ * for await (const chunk of QvdDataFrame.iterate('big.qvd', {chunkSize: 50_000})) {
3285
+ * process(chunk.data);
3286
+ * }
3287
+ * ```
3288
+ *
3289
+ * A window covering no rows yields nothing, so a loop over an exhausted offset simply does not
3290
+ * run its body.
3291
+ *
3292
+ * @param {string} path The path to the QVD file.
3293
+ * @param {Object} [options] The same options `fromQvd` takes, plus:
3294
+ * @param {number} [options.chunkSize=100000] Rows per frame. Must be a positive integer.
3295
+ * @return {AsyncGenerator<QvdDataFrame>} The chunks, in file order.
3296
+ */
3297
+ static async *iterate(path3, options = {}) {
3298
+ const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
3299
+ const reader = new QvdFileReader2(path3, readerOptionsFrom(options));
3300
+ yield* reader.iterateRows(windowFrom(options), options.chunkSize === void 0 ? 1e5 : options.chunkSize);
2825
3301
  }
2826
3302
  /**
2827
3303
  * Reads a QVD file's schema and header metadata, without reading its data.
@@ -2845,11 +3321,15 @@ var init_QvdDataFrame = __esm({
2845
3321
  * @param {Object} [options] Optional reading options.
2846
3322
  * @param {string} [options.allowedDir] Optional allowed directory path, applied exactly as it
2847
3323
  * is for `fromQvd`.
3324
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}`, as on
3325
+ * the reads that return data. Only the `read` and `header` stages occur here; there are no
3326
+ * symbols to parse and no rows to build.
3327
+ * @param {AbortSignal} [options.signal] Cancels the read, rejecting with `signal.reason`.
2848
3328
  * @return {Promise<QvdFileMetadata>} The file's schema and header metadata.
2849
3329
  */
2850
3330
  static async readMetadata(path3, options = {}) {
2851
3331
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
2852
- return await new QvdFileReader2(path3, { allowedDir: options.allowedDir }).loadMetadata();
3332
+ return await new QvdFileReader2(path3, metadataOptionsFrom(options)).loadMetadata();
2853
3333
  }
2854
3334
  /**
2855
3335
  * Constructs a data frame from a dictionary.