qvdjs 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -250,6 +250,128 @@ var init_QvdSymbol = __esm({
250
250
  };
251
251
  }
252
252
  });
253
+
254
+ // src/util/readOptions.js
255
+ function requireRowCount(value, name, filePath) {
256
+ if (typeof value !== "number" || !Number.isInteger(value) || value < 0) {
257
+ throw new QvdValidationError(`${name} must be a non-negative integer`, {
258
+ option: name,
259
+ provided: value,
260
+ type: typeof value,
261
+ file: filePath
262
+ });
263
+ }
264
+ return value;
265
+ }
266
+ function normaliseWindow(window, filePath) {
267
+ if (window === null || window === void 0) {
268
+ return { offset: 0, limit: null };
269
+ }
270
+ if (typeof window === "number") {
271
+ return { offset: 0, limit: requireRowCount(window, "maxRows", filePath) };
272
+ }
273
+ if (typeof window !== "object" || Array.isArray(window)) {
274
+ throw new QvdValidationError("The row window must be a number, null, or an {offset, limit} object", {
275
+ provided: window,
276
+ type: typeof window,
277
+ file: filePath
278
+ });
279
+ }
280
+ const { offset, limit, maxRows } = window;
281
+ const limitGiven = limit !== void 0 && limit !== null;
282
+ const maxRowsGiven = maxRows !== void 0 && maxRows !== null;
283
+ if (limitGiven && maxRowsGiven) {
284
+ throw new QvdValidationError("maxRows and limit are two names for the same option; pass one of them, not both", {
285
+ maxRows,
286
+ limit,
287
+ file: filePath
288
+ });
289
+ }
290
+ return {
291
+ offset: offset === void 0 || offset === null ? 0 : requireRowCount(offset, "offset", filePath),
292
+ limit: limitGiven ? requireRowCount(limit, "limit", filePath) : maxRowsGiven ? requireRowCount(maxRows, "maxRows", filePath) : null
293
+ };
294
+ }
295
+ function resolveWindow(window, totalRows) {
296
+ const rows = Number.isSafeInteger(totalRows) && totalRows > 0 ? totalRows : 0;
297
+ const offset = Math.min(window.offset, rows);
298
+ return {
299
+ offset,
300
+ limit: Math.max(0, Math.min(window.limit === null ? Infinity : window.limit, rows - offset))
301
+ };
302
+ }
303
+ function selectFields(fields, requested, filePath) {
304
+ if (requested === null || requested === void 0) {
305
+ return fields;
306
+ }
307
+ if (!Array.isArray(requested)) {
308
+ throw new QvdValidationError("fields must be an array of field names", {
309
+ provided: requested,
310
+ type: typeof requested,
311
+ file: filePath
312
+ });
313
+ }
314
+ const available = fields.map((field) => field["FieldName"]);
315
+ if (requested.length === 0) {
316
+ throw new QvdValidationError("fields must name at least one field", {
317
+ availableColumns: available,
318
+ file: filePath
319
+ });
320
+ }
321
+ const seen = /* @__PURE__ */ new Set();
322
+ return requested.map((name) => {
323
+ if (typeof name !== "string") {
324
+ throw new QvdValidationError("Field names must be strings", {
325
+ provided: name,
326
+ type: typeof name,
327
+ availableColumns: available,
328
+ file: filePath
329
+ });
330
+ }
331
+ if (seen.has(name)) {
332
+ throw new QvdValidationError(`Field '${name}' is listed twice`, {
333
+ column: name,
334
+ fields: requested,
335
+ file: filePath
336
+ });
337
+ }
338
+ seen.add(name);
339
+ const index = available.indexOf(name);
340
+ if (index === -1) {
341
+ throw new QvdValidationError(`Column '${name}' does not exist`, {
342
+ column: name,
343
+ availableColumns: available,
344
+ file: filePath
345
+ });
346
+ }
347
+ return fields[index];
348
+ });
349
+ }
350
+ function readerOptionsFrom(options) {
351
+ return {
352
+ allowedDir: options.allowedDir,
353
+ memorySafetyFactor: options.memorySafetyFactor,
354
+ symbolFilteringThreshold: options.symbolFilteringThreshold,
355
+ fields: options.fields === void 0 ? null : options.fields,
356
+ onProgress: options.onProgress,
357
+ signal: options.signal
358
+ };
359
+ }
360
+ function metadataOptionsFrom(options) {
361
+ return {
362
+ allowedDir: options.allowedDir,
363
+ onProgress: options.onProgress,
364
+ signal: options.signal
365
+ };
366
+ }
367
+ function windowFrom(options) {
368
+ return { offset: options.offset, limit: options.limit, maxRows: options.maxRows };
369
+ }
370
+ var init_readOptions = __esm({
371
+ "src/util/readOptions.js"() {
372
+ init_QvdErrors();
373
+ }
374
+ });
253
375
  function isWithinDirectoryLexically(resolvedBaseDir, resolvedPath) {
254
376
  const isCaseInsensitiveFS = process.platform === "win32";
255
377
  const base = isCaseInsensitiveFS ? resolvedBaseDir.toLowerCase() : resolvedBaseDir;
@@ -807,11 +929,12 @@ function estimateRowMemory(rows, columnCount) {
807
929
  }
808
930
  return BASE_BYTES + rows * (ROW_BASE_BYTES + PER_CELL_BYTES * columnCount);
809
931
  }
810
- function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
932
+ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true, rowsLive = null) {
811
933
  const FULL_PARSE_OVERHEAD = 6;
812
934
  const MINIMAL_OVERHEAD = 0.01;
813
935
  const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
814
- const rowMemory = materialisesRows ? estimateRowMemory(rowsToLoad, columnCount) : BASE_BYTES;
936
+ const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
937
+ const rowMemory = materialisesRows ? estimateRowMemory(liveRows, columnCount) : BASE_BYTES;
815
938
  if (maxRows === null || maxRows >= totalRows) {
816
939
  return symbolTableSize * FULL_PARSE_OVERHEAD + rowMemory;
817
940
  }
@@ -821,18 +944,44 @@ function estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount =
821
944
  const skippedSymbolsMemory = symbolTableSize * (1 - symbolPercentage) * MINIMAL_OVERHEAD;
822
945
  return keptSymbolsMemory + skippedSymbolsMemory + rowMemory;
823
946
  }
824
- function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true) {
825
- if (estimateMemoryUsage(symbolTableSize, totalRows, totalRows, columnCount, materialisesRows) <= budget) {
947
+ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, materialisesRows = true, includeExternal = false) {
948
+ const costOf = (rows) => estimateMemoryUsage(symbolTableSize, rows, totalRows, columnCount, materialisesRows) + (includeExternal ? estimateExternalMemory(Math.min(rows, totalRows), columnCount) : 0);
949
+ if (costOf(totalRows) <= budget) {
826
950
  return totalRows;
827
951
  }
828
- if (estimateMemoryUsage(symbolTableSize, 0, totalRows, columnCount, materialisesRows) > budget) {
952
+ if (costOf(0) > budget) {
829
953
  return 0;
830
954
  }
831
955
  let low = 0;
832
956
  let high = totalRows;
833
957
  while (high - low > 1) {
834
958
  const mid = Math.floor((low + high) / 2);
835
- if (estimateMemoryUsage(symbolTableSize, mid, totalRows, columnCount, materialisesRows) <= budget) {
959
+ if (costOf(mid) <= budget) {
960
+ low = mid;
961
+ } else {
962
+ high = mid;
963
+ }
964
+ }
965
+ return low;
966
+ }
967
+ function recommendedChunkFor(budget, symbolTableSize, windowRows, totalRows, columnCount, liveRowsPerChunk = 1, includeExternal = false) {
968
+ const covered = windowRows === null || windowRows >= totalRows ? totalRows : windowRows;
969
+ const fits = (chunk) => {
970
+ const live = Math.min(chunk * liveRowsPerChunk, covered);
971
+ const cost = estimateMemoryUsage(symbolTableSize, windowRows, totalRows, columnCount, true, chunk * liveRowsPerChunk) + (includeExternal ? estimateExternalMemory(live, columnCount) : 0);
972
+ return cost <= budget;
973
+ };
974
+ if (fits(covered)) {
975
+ return covered;
976
+ }
977
+ if (!fits(1)) {
978
+ return 0;
979
+ }
980
+ let low = 1;
981
+ let high = covered;
982
+ while (high - low > 1) {
983
+ const mid = Math.floor((low + high) / 2);
984
+ if (fits(mid)) {
836
985
  low = mid;
837
986
  } else {
838
987
  high = mid;
@@ -840,7 +989,7 @@ function recommendedRowsFor(budget, symbolTableSize, totalRows, columnCount, mat
840
989
  }
841
990
  return low;
842
991
  }
843
- function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true) {
992
+ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePath, safetyFactor = 0.8, columnCount = 0, materialisesRows = true, live = null) {
844
993
  if (typeof safetyFactor !== "number" || safetyFactor < 0 || safetyFactor > 1) {
845
994
  throw new QvdValidationError("safetyFactor must be a number between 0.0 and 1.0", { safetyFactor });
846
995
  }
@@ -849,12 +998,16 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
849
998
  }
850
999
  const budget = getMemoryBudget();
851
1000
  const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
852
- const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
853
- const externalMemory = estimateExternalMemory(rowsToLoad, columnCount);
1001
+ const rowsLive = live === null ? null : live.rows;
1002
+ const liveRowsPerChunk = live === null ? 1 : live.perChunk;
1003
+ const liveRows = rowsLive === null ? rowsToLoad : Math.min(rowsLive, rowsToLoad);
1004
+ const heapMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows, rowsLive);
1005
+ const externalMemory = estimateExternalMemory(liveRows, columnCount);
854
1006
  const bounded = budget.candidates.map((candidate) => {
855
1007
  const heapOnly = candidate.source === "V8 heap limit";
856
1008
  return {
857
1009
  ...candidate,
1010
+ heapOnly,
858
1011
  needs: heapOnly ? heapMemory : heapMemory + externalMemory,
859
1012
  allowed: candidate.bytes * safetyFactor,
860
1013
  bounds: heapOnly ? "the V8 heap" : "the whole process"
@@ -870,12 +1023,14 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
870
1023
  const estimatedMemory = binding ? binding.needs : heapMemory;
871
1024
  const maxAllowedMemory = binding ? binding.allowed : budget.bytes * safetyFactor;
872
1025
  if (binding) {
1026
+ const includeExternal = !binding.heapOnly;
873
1027
  const recommendedMaxRows = recommendedRowsFor(
874
1028
  maxAllowedMemory,
875
1029
  symbolTableSize,
876
1030
  totalRows,
877
1031
  columnCount,
878
- materialisesRows
1032
+ materialisesRows,
1033
+ includeExternal
879
1034
  );
880
1035
  const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
881
1036
  const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
@@ -887,15 +1042,27 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
887
1042
  const limitingScope = binding.bounds;
888
1043
  const budgetBreakdown = budget.candidates.map((candidate) => `${candidate.source} ${Math.round(candidate.bytes / 1024 / 1024)}MB`).join(", ");
889
1044
  const observedBreakdown = budget.observed.map((entry) => `${entry.source} ${Math.round(entry.bytes / 1024 / 1024)}MB`).join(", ");
890
- const nothingFits = recommendedMaxRows === 0;
891
1045
  const containerBound = binding.source === "container memory limit";
1046
+ const chunked = rowsLive !== null;
1047
+ const recommendedChunk = chunked ? recommendedChunkFor(
1048
+ maxAllowedMemory,
1049
+ symbolTableSize,
1050
+ maxRows,
1051
+ totalRows,
1052
+ columnCount,
1053
+ liveRowsPerChunk,
1054
+ includeExternal
1055
+ ) : 0;
1056
+ const knob = chunked ? "chunkSize" : "limit";
1057
+ const recommendedValue = chunked ? recommendedChunk : recommendedMaxRows;
1058
+ const nothingFits = recommendedValue === 0;
892
1059
  let advice;
893
1060
  if (nothingFits) {
894
- advice = `No row count fits this budget - the symbol table alone exceeds it, so maxRows cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
1061
+ advice = `No row count fits this budget - the symbol table alone exceeds it, so ${knob} cannot help. ` + (containerBound ? `Raise the container's memory limit.` : `Raise the heap with --max-old-space-size, or raise memorySafetyFactor.`);
895
1062
  } else if (containerBound) {
896
- advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or load fewer rows with maxRows (recommended: ${recommendedMaxRows.toLocaleString()} rows or less).`;
1063
+ advice = `The binding limit is the container's, so raising --max-old-space-size would let V8 grow past it and be killed by the OOM killer instead. Set it below the container limit, raise the limit, or hold fewer rows with ${knob} (recommended: ${recommendedValue.toLocaleString()} rows or less).`;
897
1064
  } else {
898
- advice = `Try loading fewer rows using the maxRows parameter (recommended: ${recommendedMaxRows.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
1065
+ advice = `Try holding fewer rows using the ${knob} parameter (recommended: ${recommendedValue.toLocaleString()} rows or less), or raise the heap with --max-old-space-size.`;
899
1066
  }
900
1067
  throw new QvdValidationError(
901
1068
  `Insufficient memory to load file safely. Symbol table: ${sizeMB}MB, Estimated memory needed: ${estimatedMB}MB, Available: ${availableMB}MB (limited by ${limitingFactor}, which bounds ${limitingScope}; considered: ${budgetBreakdown}; observed but not used: ${observedBreakdown}). ` + advice,
@@ -915,22 +1082,30 @@ function validateMemoryAvailability(symbolTableSize, maxRows, totalRows, filePat
915
1082
  columnCount,
916
1083
  totalRows,
917
1084
  maxRows,
918
- recommendedMaxRows
1085
+ recommendedMaxRows,
1086
+ // Only present when a chunk size is what overflowed, so a caller cannot mistake one
1087
+ // recommendation for the other.
1088
+ ...chunked ? { rowsLive, recommendedChunkSize: recommendedChunk } : {}
919
1089
  }
920
1090
  );
921
1091
  }
922
1092
  }
923
1093
  function warnLargeSymbolTable(symbolTableSize, maxRows, totalRows, columnCount = 0, materialisesRows = true) {
924
1094
  const LARGE_SYMBOL_TABLE_WARNING = usableOldSpaceLimit() * 0.125;
925
- if (symbolTableSize > LARGE_SYMBOL_TABLE_WARNING && (maxRows === null || maxRows >= totalRows)) {
926
- const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
927
- const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
928
- const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
929
- const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
930
- console.warn(
931
- `\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). Loading all ${totalRows.toLocaleString()} rows will use ~${estimatedMB}MB RAM. Consider using the maxRows parameter for better performance and lower memory usage.`
932
- );
1095
+ if (symbolTableSize <= LARGE_SYMBOL_TABLE_WARNING) {
1096
+ return;
1097
+ }
1098
+ const rowsToLoad = maxRows === null || maxRows >= totalRows ? totalRows : maxRows;
1099
+ const estimatedMemory = estimateMemoryUsage(symbolTableSize, maxRows, totalRows, columnCount, materialisesRows);
1100
+ if (estimatedMemory <= LARGE_SYMBOL_TABLE_WARNING) {
1101
+ return;
933
1102
  }
1103
+ const sizeMB = Math.round(symbolTableSize / 1024 / 1024);
1104
+ const estimatedMB = Math.round(estimatedMemory / 1024 / 1024);
1105
+ const warnMB = Math.round(LARGE_SYMBOL_TABLE_WARNING / 1024 / 1024);
1106
+ console.warn(
1107
+ `\u26A0\uFE0F Large symbol table detected (${sizeMB}MB > ${warnMB}MB threshold). This read materialises ${rowsToLoad.toLocaleString()} of ${totalRows.toLocaleString()} rows and will use ~${estimatedMB}MB RAM. Reading fewer rows - with limit, maxRows, or a narrower offset window - lowers the row cost, though the symbol table is read in full either way.`
1108
+ );
934
1109
  }
935
1110
  var HEAP_LIMIT_OVERSTATEMENT_BYTES, MINIMUM_BUDGET_BYTES, BASE_BYTES, ROW_BASE_BYTES, PER_CELL_BYTES;
936
1111
  var init_memoryUtils = __esm({
@@ -945,6 +1120,24 @@ var init_memoryUtils = __esm({
945
1120
  });
946
1121
 
947
1122
  // src/util/validationUtils.js
1123
+ function validateHeaderStructure(headerObj, filePath, stage) {
1124
+ const tableHeader = headerObj?.["QvdTableHeader"];
1125
+ if (tableHeader === null || typeof tableHeader !== "object" || Array.isArray(tableHeader)) {
1126
+ throw new QvdCorruptedError("The XML header contains no usable QvdTableHeader element", {
1127
+ rootElements: headerObj && typeof headerObj === "object" ? Object.keys(headerObj) : [],
1128
+ file: filePath,
1129
+ stage
1130
+ });
1131
+ }
1132
+ const symbolTableLength = parseInt(tableHeader["Offset"], 10);
1133
+ if (isNaN(symbolTableLength) || !Number.isSafeInteger(symbolTableLength) || symbolTableLength < 0) {
1134
+ throw new QvdCorruptedError("Invalid symbol table offset", {
1135
+ offset: tableHeader["Offset"],
1136
+ file: filePath,
1137
+ stage
1138
+ });
1139
+ }
1140
+ }
948
1141
  function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
949
1142
  const heapLimit = getHeapLimit();
950
1143
  const MAX_SYMBOL_TABLE_SIZE = heapLimit * 0.125;
@@ -953,7 +1146,7 @@ function validateSymbolTableSizeEarly(symbolTableLength, filePath) {
953
1146
  const maxMB = Math.round(MAX_SYMBOL_TABLE_SIZE / 1024 / 1024);
954
1147
  const heapMB = Math.round(heapLimit / 1024 / 1024);
955
1148
  throw new QvdValidationError(
956
- `Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without maxRows, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
1149
+ `Symbol table too large (${sizeMB}MB exceeds ${maxMB}MB limit for lazy loading). This QVD file contains extremely high-cardinality fields. Limit scales with heap size (current: ${heapMB}MB, limit: 12.5% = ${maxMB}MB). Consider: (1) loading the full file without a row window - maxRows, limit or offset - since the symbol table is read in full either way, (2) increasing heap size with --max-old-space-size, or (3) aggregating high-cardinality fields.`,
957
1150
  {
958
1151
  file: filePath,
959
1152
  symbolTableSize: symbolTableLength,
@@ -1027,7 +1220,7 @@ function validateRecordCount(totalRows, filePath, stage = "parseIndexTable") {
1027
1220
  });
1028
1221
  }
1029
1222
  }
1030
- function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null) {
1223
+ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, indexTableOffset, bufferLength, rowsToLoad, filePath, fileSize = null, windowFirstRow = 0, bufferFirstRow = 0) {
1031
1224
  if (isNaN(recordSize) || !Number.isSafeInteger(recordSize) || recordSize < 0) {
1032
1225
  throw new QvdCorruptedError("Invalid record byte size", {
1033
1226
  recordSize,
@@ -1089,23 +1282,28 @@ function validateIndexTableMetadata(recordSize, totalRows, indexTableLength, ind
1089
1282
  }
1090
1283
  }
1091
1284
  const requiredIndexBytes = rowsToLoad * recordSize;
1092
- if (indexTableOffset + requiredIndexBytes > bufferLength) {
1285
+ const bufferRecordStart = (windowFirstRow - bufferFirstRow) * recordSize;
1286
+ if (indexTableOffset + bufferRecordStart + requiredIndexBytes > bufferLength) {
1093
1287
  throw new QvdCorruptedError("Index table truncated", {
1094
1288
  indexTableOffset,
1095
1289
  requiredBytes: requiredIndexBytes,
1096
- availableBytes: Math.max(0, bufferLength - indexTableOffset),
1290
+ availableBytes: Math.max(0, bufferLength - indexTableOffset - bufferRecordStart),
1097
1291
  rowsToLoad,
1292
+ windowFirstRow,
1293
+ bufferFirstRow,
1098
1294
  recordSize,
1099
1295
  bufferSize: bufferLength,
1100
1296
  file: filePath,
1101
1297
  stage: "parseIndexTable"
1102
1298
  });
1103
1299
  }
1104
- if (indexTableLength < requiredIndexBytes) {
1300
+ const requiredTableBytes = (windowFirstRow + rowsToLoad) * recordSize;
1301
+ if (indexTableLength < requiredTableBytes) {
1105
1302
  throw new QvdCorruptedError("Index table length smaller than required", {
1106
1303
  indexTableLength,
1107
- requiredBytes: requiredIndexBytes,
1304
+ requiredBytes: requiredTableBytes,
1108
1305
  rowsToLoad,
1306
+ windowFirstRow,
1109
1307
  recordSize,
1110
1308
  file: filePath,
1111
1309
  stage: "parseIndexTable"
@@ -1435,6 +1633,7 @@ var QvdColumn, QvdColumnTable;
1435
1633
  var init_QvdColumnTable = __esm({
1436
1634
  "src/QvdColumnTable.js"() {
1437
1635
  init_QvdErrors();
1636
+ init_readOptions();
1438
1637
  QvdColumn = class {
1439
1638
  /**
1440
1639
  * @param {string} name The field name.
@@ -1629,9 +1828,21 @@ var init_QvdColumnTable = __esm({
1629
1828
  /**
1630
1829
  * Reads a QVD file as columns.
1631
1830
  *
1831
+ * Takes the same options as `QvdDataFrame.fromQvd`, with the same meanings - one option
1832
+ * vocabulary for both read paths, because they are two answers about the same file rather than
1833
+ * two features. `{offset, limit}` is how a caller pages through a file columnwise; there is no
1834
+ * columnar `iterate()` because there is nothing for it to bound - a columnar read materialises
1835
+ * no rows, which is the memory chunking exists to cap.
1836
+ *
1632
1837
  * @param {string} path The path to the QVD file.
1633
1838
  * @param {Object} [options] Loading options, with the same meanings they have on `fromQvd`.
1634
- * @param {number|null} [options.maxRows] Maximum rows to decode.
1839
+ * @param {number|null} [options.maxRows] Rows to decode. The older name for `limit`.
1840
+ * @param {number|null} [options.limit] Rows to decode, counting from `offset`.
1841
+ * @param {number} [options.offset] File row to start at.
1842
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
1843
+ * appear. Unselected fields have their symbols skipped entirely.
1844
+ * @param {Function} [options.onProgress] Progress callback, `{stage, current, total, percent}`.
1845
+ * @param {AbortSignal} [options.signal] Cancels the read.
1635
1846
  * @param {string} [options.allowedDir] Directory the path must resolve inside.
1636
1847
  * @param {number} [options.memorySafetyFactor] Fraction of the memory budget a load may use.
1637
1848
  * @param {number} [options.symbolFilteringThreshold] Symbol table size above which a limited
@@ -1641,15 +1852,13 @@ var init_QvdColumnTable = __esm({
1641
1852
  static async fromQvd(path3, options = {}) {
1642
1853
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
1643
1854
  const reader = new QvdFileReader2(path3, {
1644
- allowedDir: options.allowedDir,
1645
- memorySafetyFactor: options.memorySafetyFactor,
1646
- symbolFilteringThreshold: options.symbolFilteringThreshold,
1855
+ ...readerOptionsFrom(options),
1647
1856
  // This read builds no rows, so the memory guard must not charge it for them. A columnar
1648
1857
  // read of the 38MB taxi fixture completes in a 15MB heap; charged the row cost it was
1649
1858
  // refused below a 2GB one.
1650
1859
  materialisesRows: false
1651
1860
  });
1652
- return await reader.loadColumnar(options.maxRows !== void 0 ? options.maxRows : null);
1861
+ return await reader.loadColumnar(windowFrom(options));
1653
1862
  }
1654
1863
  /** @return {Array<string>} Field names, in file order. */
1655
1864
  get columns() {
@@ -1697,7 +1906,7 @@ var QvdFileReader_exports = {};
1697
1906
  __export(QvdFileReader_exports, {
1698
1907
  QvdFileReader: () => QvdFileReader
1699
1908
  });
1700
- var MAX_HEADER_SIZE, READ_CHUNK_SIZE, QvdFileReader;
1909
+ var MAX_HEADER_SIZE, READ_CHUNK_SIZE, ANALYSIS_SLICE_ROWS, QvdFileReader;
1701
1910
  var init_QvdFileReader = __esm({
1702
1911
  "src/QvdFileReader.js"() {
1703
1912
  init_QvdDataFrame();
@@ -1707,8 +1916,10 @@ var init_QvdFileReader = __esm({
1707
1916
  init_memoryUtils();
1708
1917
  init_validationUtils();
1709
1918
  init_symbolParser();
1919
+ init_readOptions();
1710
1920
  MAX_HEADER_SIZE = 16 * 1024 * 1024;
1711
1921
  READ_CHUNK_SIZE = 512 * 1024 * 1024;
1922
+ ANALYSIS_SLICE_ROWS = 65536;
1712
1923
  QvdFileReader = class {
1713
1924
  /**
1714
1925
  * Constructs a new QVD file parser.
@@ -1720,9 +1931,10 @@ var init_QvdFileReader = __esm({
1720
1931
  * points outside it is rejected. Defaults to the current working directory. To permit
1721
1932
  * an entire volume, pass its root explicitly ('/' on POSIX, 'C:\\' on Windows); a null or
1722
1933
  * empty value falls back to the working directory rather than removing the restriction.
1723
- * @param {number} [options.memorySafetyFactor=0.3] Fraction (0.0-1.0) of the memory budget a
1724
- * load may use. The budget is the smallest of the V8 heap limit, any container memory limit,
1725
- * and the memory the OS reports as available. Default is 0.3. **Zero disables the memory
1934
+ * @param {number} [options.memorySafetyFactor=0.8] Fraction (0.0-1.0) of the memory budget a
1935
+ * load may use. The budget is the smaller of the V8 heap limit and any container memory limit;
1936
+ * what the OS reports as available is recorded for diagnostics and deliberately not allowed to
1937
+ * bind - see `getMemoryBudget`. Default is 0.8. **Zero disables the memory
1726
1938
  * check entirely**, which is the escape hatch for runtimes whose limits cannot be measured -
1727
1939
  * Bun reports its current heap as its heap limit - and for callers who would rather manage
1728
1940
  * memory themselves than trust the estimate.
@@ -1733,29 +1945,93 @@ var init_QvdFileReader = __esm({
1733
1945
  * above which a lazy load switches to the two-pass filtering path. The default of 50MB is
1734
1946
  * the point where the extra analysis pass pays for itself; lower it to use filtering on
1735
1947
  * smaller files, raise it to keep the simpler single-pass read for longer.
1948
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
1949
+ * appear. Null reads every field, in file order. An unknown or repeated name is refused.
1950
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
1951
+ * read proceeds - the same shape `QvdFileWriter` emits.
1952
+ * @param {AbortSignal} [options.signal] Cancels the read. When it is aborted the read throws
1953
+ * `signal.reason`, exactly as `signal.throwIfAborted()` does.
1736
1954
  */
1737
1955
  constructor(filePath, options = {}) {
1738
1956
  const {
1739
1957
  allowedDir,
1740
1958
  memorySafetyFactor = 0.8,
1741
1959
  symbolFilteringThreshold = 50 * 1024 * 1024,
1742
- materialisesRows = true
1960
+ materialisesRows = true,
1961
+ fields = null,
1962
+ onProgress,
1963
+ signal
1743
1964
  } = options;
1744
1965
  this._materialisesRows = materialisesRows;
1745
1966
  this._path = validatePath(filePath, allowedDir);
1746
1967
  this._memorySafetyFactor = memorySafetyFactor;
1747
1968
  this._symbolFilteringThreshold = symbolFilteringThreshold;
1969
+ if (onProgress !== void 0 && typeof onProgress !== "function") {
1970
+ throw new QvdValidationError("onProgress must be a function", {
1971
+ provided: onProgress,
1972
+ type: typeof onProgress,
1973
+ file: this._path
1974
+ });
1975
+ }
1976
+ if (signal !== void 0 && (typeof signal !== "object" || signal === null || typeof signal.aborted !== "boolean")) {
1977
+ throw new QvdValidationError("signal must be an AbortSignal", {
1978
+ provided: signal,
1979
+ type: typeof signal,
1980
+ file: this._path
1981
+ });
1982
+ }
1983
+ this._requestedFields = fields === void 0 ? null : fields;
1984
+ this._onProgress = onProgress;
1985
+ this._signal = signal;
1748
1986
  this._buffer = null;
1749
1987
  this._headerOffset = null;
1750
1988
  this._symbolTableOffset = null;
1751
1989
  this._indexTableOffset = null;
1752
1990
  this._header = null;
1991
+ this._allFields = null;
1992
+ this._selectedFields = null;
1993
+ this._fieldBitMetadataValidated = false;
1753
1994
  this._symbolTable = null;
1754
1995
  this._indexColumns = null;
1755
1996
  this._rowsDecoded = 0;
1997
+ this._bufferFirstRow = 0;
1756
1998
  this._fileSize = null;
1757
1999
  this._headerMatchesFile = false;
1758
2000
  }
2001
+ /**
2002
+ * Emits a progress event if a callback is registered.
2003
+ *
2004
+ * The same shape `QvdFileWriter._emitProgress` emits, deliberately: a caller who has written a
2005
+ * progress bar for a write should not have to write a second one for a read. The stage names
2006
+ * differ because the stages differ, but `symbol-table` and `index-table` mean the same thing on
2007
+ * both sides.
2008
+ *
2009
+ * @param {string} stage The current stage of the read.
2010
+ * @param {number} current The current progress value.
2011
+ * @param {number} total The total progress value.
2012
+ * @private
2013
+ */
2014
+ _emitProgress(stage, current, total) {
2015
+ if (this._onProgress) {
2016
+ const percent = total > 0 ? Math.round(current / total * 100) : 100;
2017
+ this._onProgress({ stage, current, total, percent });
2018
+ }
2019
+ }
2020
+ /**
2021
+ * Throws if the caller has cancelled the read.
2022
+ *
2023
+ * Throws `signal.reason` - a `DOMException` named `AbortError` unless the caller aborted with a
2024
+ * reason of their own. That is what `AbortSignal` means everywhere else in Node, and inventing
2025
+ * a `QvdAbortError` here would make this library's cancellation the one a caller has to special
2026
+ * case.
2027
+ *
2028
+ * @private
2029
+ */
2030
+ _throwIfAborted() {
2031
+ if (this._signal) {
2032
+ this._signal.throwIfAborted();
2033
+ }
2034
+ }
1759
2035
  /**
1760
2036
  * Reads the binary data of the QVD file.
1761
2037
  *
@@ -1780,14 +2056,23 @@ var init_QvdFileReader = __esm({
1780
2056
  * - Streaming for header finding is efficient for unknown header sizes
1781
2057
  * - Direct byte-range reading for remaining data is fastest
1782
2058
  *
1783
- * @param {number|null} maxRows The maximum number of rows to load. If null, all data is loaded.
2059
+ * A window with a non-zero `offset` reads two ranges rather than one: the header and symbol
2060
+ * table from the front of the file, and the window's records from wherever they sit. The bytes
2061
+ * between are never read, which is what makes `{offset: 1_700_000, limit: 100}` on the taxi
2062
+ * fixture a 0.4MB read rather than a 38MB one.
2063
+ *
2064
+ * @param {QvdRowWindow} window The rows to read.
1784
2065
  * @param {boolean} [headerOnly=false] Stop once the XML header has been read, leaving the
1785
2066
  * symbol and index tables on disk. This is the metadata-only path: the header is a few
1786
2067
  * kilobytes whatever the file's size, so reading a schema costs the same for a 40MB file as
1787
2068
  * for a 40GB one.
2069
+ * @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
2070
+ * that is fewer than the window covers - see `_prepare`.
1788
2071
  * @private
1789
2072
  */
1790
- async _readData(maxRows = null, headerOnly = false) {
2073
+ async _readData(window = { offset: 0, limit: null }, headerOnly = false, liveRows = null) {
2074
+ this._throwIfAborted();
2075
+ this._emitProgress("read", 0, 1);
1791
2076
  const HEADER_DELIMITER = "\r\n\0";
1792
2077
  const CHUNK_SIZE = 64 * 1024;
1793
2078
  const stream = fs.createReadStream(this._path, {
@@ -1849,6 +2134,7 @@ var init_QvdFileReader = __esm({
1849
2134
  stage: "readData"
1850
2135
  });
1851
2136
  }
2137
+ validateHeaderStructure(headerObj, this._path, "readData");
1852
2138
  const symbolTableOffset = headerEndIndex;
1853
2139
  const symbolTableLength = parseInt(headerObj["QvdTableHeader"]["Offset"], 10);
1854
2140
  const indexTableOffset = symbolTableOffset + symbolTableLength;
@@ -1856,13 +2142,14 @@ var init_QvdFileReader = __esm({
1856
2142
  const totalRows = parseInt(headerObj["QvdTableHeader"]["NoOfRecords"], 10);
1857
2143
  if (headerOnly) {
1858
2144
  this._buffer = headerBuffer.subarray(0, headerEndIndex);
2145
+ this._emitProgress("read", 1, 1);
1859
2146
  return;
1860
2147
  }
1861
2148
  let headerFields = headerObj["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
1862
2149
  if (headerFields && !Array.isArray(headerFields)) {
1863
2150
  headerFields = [headerFields];
1864
2151
  }
1865
- const columnCount = Array.isArray(headerFields) ? headerFields.length : 0;
2152
+ const columnCount = Array.isArray(headerFields) ? selectFields(headerFields, this._requestedFields, this._path).length : 0;
1866
2153
  const headerNumbersUsable = [symbolTableLength, recordSize, totalRows].every(
1867
2154
  (value) => Number.isSafeInteger(value) && value >= 0
1868
2155
  );
@@ -1871,23 +2158,28 @@ var init_QvdFileReader = __esm({
1871
2158
  this._fileSize = fileSize;
1872
2159
  this._headerMatchesFile = headerEndIndex + symbolTableLength + totalRows * recordSize <= fileSize;
1873
2160
  }
2161
+ const resolved = headerNumbersUsable ? resolveWindow(window, totalRows) : { offset: 0, limit: 0 };
2162
+ const windowRows = resolved.limit;
1874
2163
  if (headerNumbersUsable && this._headerMatchesFile) {
1875
2164
  validateMemoryAvailability(
1876
2165
  symbolTableLength,
1877
- maxRows,
2166
+ windowRows,
1878
2167
  totalRows,
1879
2168
  this._path,
1880
2169
  this._memorySafetyFactor,
1881
2170
  columnCount,
1882
- this._materialisesRows
2171
+ this._materialisesRows,
2172
+ liveRows
1883
2173
  );
1884
2174
  }
1885
- if (maxRows === null) {
2175
+ if (window.offset === 0 && window.limit === null) {
1886
2176
  this._buffer = await fs.promises.readFile(this._path);
1887
2177
  this._fileSize = this._buffer.length;
2178
+ this._bufferFirstRow = 0;
2179
+ this._emitProgress("read", 1, 1);
1888
2180
  return;
1889
2181
  }
1890
- const rowsToLoad = Math.min(maxRows, totalRows);
2182
+ const rowsToLoad = windowRows;
1891
2183
  validateSymbolTableSizeEarly(symbolTableLength, this._path);
1892
2184
  for (const [name, value] of [
1893
2185
  ["Offset", symbolTableLength],
@@ -1903,39 +2195,78 @@ var init_QvdFileReader = __esm({
1903
2195
  });
1904
2196
  }
1905
2197
  }
2198
+ const skippedIndexBytes = resolved.offset * recordSize;
1906
2199
  const indexTableBytesToRead = rowsToLoad * recordSize;
1907
2200
  const totalBytesToRead = indexTableOffset + indexTableBytesToRead;
2201
+ const fileBytesRequired = indexTableOffset + skippedIndexBytes + indexTableBytesToRead;
1908
2202
  const fd = await fs.promises.open(this._path, "r");
1909
2203
  try {
1910
2204
  const { size: fileSize } = await fd.stat();
1911
2205
  this._fileSize = fileSize;
1912
- if (totalBytesToRead > fileSize) {
2206
+ if (fileBytesRequired > fileSize) {
1913
2207
  throw new QvdCorruptedError("The file is shorter than its header claims.", {
1914
2208
  file: this._path,
1915
2209
  fileSize,
1916
- requiredBytes: totalBytesToRead,
2210
+ requiredBytes: fileBytesRequired,
1917
2211
  stage: "readData"
1918
2212
  });
1919
2213
  }
1920
2214
  this._buffer = Buffer.alloc(totalBytesToRead);
1921
- let position = 0;
1922
- while (position < totalBytesToRead) {
1923
- const length = Math.min(READ_CHUNK_SIZE, totalBytesToRead - position);
1924
- const { bytesRead } = await fd.read(this._buffer, position, length, position);
1925
- if (bytesRead === 0) {
1926
- throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
1927
- file: this._path,
1928
- fileSize,
1929
- bytesRead: position,
1930
- requiredBytes: totalBytesToRead,
1931
- stage: "readData"
1932
- });
1933
- }
1934
- position += bytesRead;
2215
+ await this._readRange(fd, 0, indexTableOffset, 0, fileSize, totalBytesToRead);
2216
+ if (indexTableBytesToRead > 0) {
2217
+ await this._readRange(
2218
+ fd,
2219
+ indexTableOffset,
2220
+ indexTableBytesToRead,
2221
+ indexTableOffset + skippedIndexBytes,
2222
+ fileSize,
2223
+ fileBytesRequired
2224
+ );
1935
2225
  }
2226
+ this._bufferFirstRow = resolved.offset;
1936
2227
  } finally {
1937
2228
  await fd.close();
1938
2229
  }
2230
+ this._emitProgress("read", 1, 1);
2231
+ }
2232
+ /**
2233
+ * Reads one byte range of the file into the buffer.
2234
+ *
2235
+ * Read in bounded chunks, checking bytesRead each time. A single fs.read call with a length of
2236
+ * 2^31 or more does not throw - it trips a C++ assertion and aborts the whole process, which no
2237
+ * try/catch can intercept.
2238
+ *
2239
+ * @param {import('fs/promises').FileHandle} fd The open file.
2240
+ * @param {number} bufferOffset Where in the buffer to write.
2241
+ * @param {number} byteCount How many bytes to read.
2242
+ * @param {number} filePosition Where in the file to read from.
2243
+ * @param {number} fileSize The file's size, for the error.
2244
+ * @param {number} requiredBytes Bytes the whole read needs, for the error.
2245
+ * @private
2246
+ */
2247
+ async _readRange(fd, bufferOffset, byteCount, filePosition, fileSize, requiredBytes) {
2248
+ assert2(this._buffer, "The read buffer has not been allocated.");
2249
+ let done = 0;
2250
+ while (done < byteCount) {
2251
+ const length = Math.min(READ_CHUNK_SIZE, byteCount - done);
2252
+ const { bytesRead } = await fd.read(this._buffer, bufferOffset + done, length, filePosition + done);
2253
+ if (bytesRead === 0) {
2254
+ throw new QvdCorruptedError("Unexpected end of file while reading QVD data.", {
2255
+ file: this._path,
2256
+ fileSize,
2257
+ // Two numbers, because they stopped being the same one when a window began reading two
2258
+ // ranges: `bytesRead` is how much of this range arrived, `filePosition` is where in the
2259
+ // file it gave up. Reporting the position under the name of the count made a windowed
2260
+ // read of a truncated file claim tens of megabytes had been read when a few hundred
2261
+ // bytes had.
2262
+ bytesRead: done,
2263
+ filePosition: filePosition + done,
2264
+ requiredBytes,
2265
+ stage: "readData"
2266
+ });
2267
+ }
2268
+ done += bytesRead;
2269
+ }
1939
2270
  }
1940
2271
  /**
1941
2272
  * Parses the XML header of the QVD file. This method is part of the parsing process
@@ -1972,17 +2303,31 @@ var init_QvdFileReader = __esm({
1972
2303
  stage: "parseHeader"
1973
2304
  });
1974
2305
  }
2306
+ validateHeaderStructure(this._header, this._path, "parseHeader");
1975
2307
  const fields = this._header["QvdTableHeader"]?.["Fields"]?.["QvdFieldHeader"];
1976
- const fieldCount = fields === void 0 || fields === null ? 0 : Array.isArray(fields) ? fields.length : 1;
1977
- if (fieldCount === 0) {
2308
+ const fieldList = fields === void 0 || fields === null ? [] : Array.isArray(fields) ? fields : [fields];
2309
+ if (fieldList.length === 0) {
1978
2310
  throw new QvdCorruptedError("The QVD file header declares no fields", {
1979
2311
  file: this._path,
1980
2312
  stage: "parseHeader"
1981
2313
  });
1982
2314
  }
2315
+ const malformedIndex = fieldList.findIndex(
2316
+ (field) => field === null || typeof field !== "object" || Array.isArray(field)
2317
+ );
2318
+ if (malformedIndex !== -1) {
2319
+ throw new QvdCorruptedError("The QVD file header declares a field with no properties", {
2320
+ fieldIndex: malformedIndex,
2321
+ fieldCount: fieldList.length,
2322
+ file: this._path,
2323
+ stage: "parseHeader"
2324
+ });
2325
+ }
1983
2326
  this._headerOffset = headerBeginIndex;
1984
2327
  this._symbolTableOffset = headerEndIndex;
1985
2328
  this._indexTableOffset = this._symbolTableOffset + parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2329
+ this._allFields = fieldList;
2330
+ this._selectedFields = selectFields(this._allFields, this._requestedFields, this._path);
1986
2331
  }
1987
2332
  /**
1988
2333
  * Establishes the geometry of the index table, and validates it.
@@ -1993,14 +2338,15 @@ var init_QvdFileReader = __esm({
1993
2338
  * about keeping the sign in step with the other one: the two could drift, and #113 is what
1994
2339
  * that looks like when they do. There is one copy now.
1995
2340
  *
1996
- * @param {number|null} rowLimit Maximum rows of interest, or null for all of them.
2341
+ * @param {QvdRowWindow} window The rows of interest, as file row indices.
1997
2342
  * @param {string} stage Stage name for any error raised here.
1998
2343
  * @return {{fields: Array<any>, recordSize: number, totalRows: number, rowsToLoad: number,
1999
- * indexBuffer: Buffer}} The record geometry.
2344
+ * indexBuffer: Buffer}} The record geometry. `indexBuffer` starts at the window's first
2345
+ * record, so the decoder always counts from zero.
2000
2346
  * @private
2001
2347
  */
2002
- _planIndexTable(rowLimit, stage) {
2003
- if (!this._buffer || !this._header || !this._indexTableOffset) {
2348
+ _planIndexTable(window, stage) {
2349
+ if (!this._buffer || !this._header || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
2004
2350
  throw new QvdCorruptedError(
2005
2351
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
2006
2352
  {
@@ -2009,14 +2355,12 @@ var init_QvdFileReader = __esm({
2009
2355
  }
2010
2356
  );
2011
2357
  }
2012
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2013
- if (!Array.isArray(fields)) {
2014
- fields = [fields];
2015
- }
2358
+ const allFields = this._allFields;
2359
+ const fields = this._selectedFields;
2016
2360
  const recordSize = parseInt(this._header["QvdTableHeader"]["RecordByteSize"], 10);
2017
2361
  const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
2018
- const rowsToLoad = rowLimit !== null ? Math.min(rowLimit, totalRows) : totalRows;
2019
2362
  const indexTableLength = parseInt(this._header["QvdTableHeader"]["Length"], 10);
2363
+ const { offset: firstRow, limit: rowsToLoad } = resolveWindow(window, totalRows);
2020
2364
  validateIndexTableMetadata(
2021
2365
  recordSize,
2022
2366
  totalRows,
@@ -2025,11 +2369,20 @@ var init_QvdFileReader = __esm({
2025
2369
  this._buffer.length,
2026
2370
  rowsToLoad,
2027
2371
  this._path,
2028
- this._fileSize
2372
+ this._fileSize,
2373
+ firstRow,
2374
+ this._bufferFirstRow
2029
2375
  );
2030
- const indexBuffer = this._buffer.subarray(this._indexTableOffset, this._indexTableOffset + indexTableLength + 1);
2031
- for (const field of fields) {
2032
- validateFieldBitMetadata(field, recordSize, this._path);
2376
+ const bufferRecordStart = (firstRow - this._bufferFirstRow) * recordSize;
2377
+ const indexBuffer = this._buffer.subarray(
2378
+ this._indexTableOffset + bufferRecordStart,
2379
+ this._indexTableOffset + bufferRecordStart + rowsToLoad * recordSize
2380
+ );
2381
+ if (!this._fieldBitMetadataValidated) {
2382
+ for (const field of allFields) {
2383
+ validateFieldBitMetadata(field, recordSize, this._path);
2384
+ }
2385
+ this._fieldBitMetadataValidated = true;
2033
2386
  }
2034
2387
  assert2(
2035
2388
  rowsToLoad === 0 || recordSize === 0 || Math.floor(indexBuffer.length / recordSize) >= rowsToLoad,
@@ -2041,31 +2394,45 @@ var init_QvdFileReader = __esm({
2041
2394
  * Analyzes the index table to determine which symbols are actually needed.
2042
2395
  * This is used for two-pass symbol filtering optimization.
2043
2396
  *
2044
- * @param {number} maxRows The maximum number of rows to analyze.
2045
- * @return {Promise<Map<string, Set<number>>>} Map of field names to Set of needed symbol indices.
2397
+ * Only the selected fields are analysed. An unselected field's symbols are never parsed, so
2398
+ * there is nothing for a usage set to filter and decoding its column would be a pass over the
2399
+ * whole window for an answer nobody reads.
2400
+ *
2401
+ * @param {QvdRowWindow} window The rows to analyse.
2402
+ * @return {Promise<Array<Set<number>>>} One set of needed symbol indices per selected field, in
2403
+ * the same order `_parseSymbolTable` walks them.
2046
2404
  * @private
2047
2405
  */
2048
- async _analyzeIndexTableSymbolUsage(maxRows) {
2049
- const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(maxRows, "analyzeIndexTableSymbolUsage");
2050
- const symbolUsage = /* @__PURE__ */ new Map();
2051
- const column = new Int32Array(rowsToLoad);
2052
- fields.forEach((field) => {
2406
+ async _analyzeIndexTableSymbolUsage(window) {
2407
+ const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "analyzeIndexTableSymbolUsage");
2408
+ const symbolUsage = [];
2409
+ const sliceRows = Math.min(rowsToLoad, ANALYSIS_SLICE_ROWS);
2410
+ const column = new Int32Array(sliceRows);
2411
+ fields.forEach((field, position) => {
2412
+ this._throwIfAborted();
2053
2413
  const needed = /* @__PURE__ */ new Set();
2054
- symbolUsage.set(field["FieldName"], needed);
2055
- decodeIndexColumn(
2056
- indexBuffer,
2057
- recordSize,
2058
- rowsToLoad,
2059
- parseInt(field["BitOffset"], 10),
2060
- parseInt(field["BitWidth"], 10),
2061
- parseInt(field["Bias"], 10),
2062
- column
2063
- );
2064
- for (let row = 0; row < rowsToLoad; row++) {
2065
- if (column[row] >= 0) {
2066
- needed.add(column[row]);
2414
+ symbolUsage[position] = needed;
2415
+ const bitOffset = parseInt(field["BitOffset"], 10);
2416
+ const bitWidth = parseInt(field["BitWidth"], 10);
2417
+ const bias = parseInt(field["Bias"], 10);
2418
+ for (let first = 0; first < rowsToLoad; first += sliceRows) {
2419
+ const count = Math.min(sliceRows, rowsToLoad - first);
2420
+ decodeIndexColumn(
2421
+ first === 0 ? indexBuffer : indexBuffer.subarray(first * recordSize),
2422
+ recordSize,
2423
+ count,
2424
+ bitOffset,
2425
+ bitWidth,
2426
+ bias,
2427
+ column
2428
+ );
2429
+ for (let row = 0; row < count; row++) {
2430
+ if (column[row] >= 0) {
2431
+ needed.add(column[row]);
2432
+ }
2067
2433
  }
2068
2434
  }
2435
+ this._emitProgress("symbol-analysis", position + 1, fields.length);
2069
2436
  });
2070
2437
  return symbolUsage;
2071
2438
  }
@@ -2073,12 +2440,20 @@ var init_QvdFileReader = __esm({
2073
2440
  * Parses the symbol table of the QVD file. This method is part of the parsing process
2074
2441
  * and should not be called directly.
2075
2442
  *
2076
- * @param {Map<string, Set<number>>|null} symbolsToKeep Optional map of field names to symbol indices to keep.
2077
- * If provided, only these symbols will be parsed (two-pass filtering optimization).
2078
- * @param {number|null} maxRows Optional maximum number of rows being loaded (for memory estimation).
2443
+ * A field the caller did not select is skipped whole. Its symbol area is neither scanned nor
2444
+ * parsed - the per-field `Offset` and `Length` say exactly where it is, so there is nothing to
2445
+ * walk past - and that is where field selection earns its keep. The index decode is cheap by
2446
+ * comparison; parsing symbols is not.
2447
+ *
2448
+ * @param {Array<Set<number>>|null} symbolsToKeep Optional set of symbol indices to keep per
2449
+ * selected field, indexed by position. If provided, only these symbols will be parsed
2450
+ * (two-pass filtering optimization).
2451
+ * @param {number} rowsToLoad Rows the read covers, for memory estimation.
2452
+ * @param {{rows: number, perChunk: number}|null} [liveRows=null] Rows held at one instant when
2453
+ * that is fewer than the window covers - see `_prepare`.
2079
2454
  */
2080
- async _parseSymbolTable(symbolsToKeep = null, maxRows = null) {
2081
- if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset) {
2455
+ async _parseSymbolTable(symbolsToKeep = null, rowsToLoad = 0, liveRows = null) {
2456
+ if (!this._buffer || !this._header || !this._symbolTableOffset || !this._indexTableOffset || !this._selectedFields || !this._allFields) {
2082
2457
  throw new QvdCorruptedError(
2083
2458
  "The QVD file has not been loaded in the proper order or has not been loaded at all.",
2084
2459
  {
@@ -2087,7 +2462,8 @@ var init_QvdFileReader = __esm({
2087
2462
  }
2088
2463
  );
2089
2464
  }
2090
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2465
+ const allFields = this._allFields;
2466
+ const fields = this._selectedFields;
2091
2467
  const symbolBuffer = this._buffer.subarray(this._symbolTableOffset, this._indexTableOffset);
2092
2468
  const symbolTableSize = symbolBuffer.length;
2093
2469
  const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
@@ -2095,32 +2471,25 @@ var init_QvdFileReader = __esm({
2095
2471
  if (this._headerMatchesFile) {
2096
2472
  validateMemoryAvailability(
2097
2473
  symbolTableSize,
2098
- maxRows,
2474
+ rowsToLoad,
2099
2475
  totalRows,
2100
2476
  this._path,
2101
2477
  this._memorySafetyFactor,
2102
- Array.isArray(fields) ? fields.length : 1,
2103
- this._materialisesRows
2478
+ fields.length,
2479
+ this._materialisesRows,
2480
+ liveRows
2104
2481
  );
2105
2482
  }
2106
- warnLargeSymbolTable(
2107
- symbolTableSize,
2108
- maxRows,
2109
- totalRows,
2110
- Array.isArray(fields) ? fields.length : 1,
2111
- this._materialisesRows
2112
- );
2113
- if (!Array.isArray(fields)) {
2114
- fields = [fields];
2115
- }
2116
- for (const field of fields) {
2483
+ warnLargeSymbolTable(symbolTableSize, rowsToLoad, totalRows, fields.length, this._materialisesRows);
2484
+ for (const field of allFields) {
2117
2485
  validateFieldMetadata(field, symbolBuffer.length, this._path);
2118
2486
  }
2119
- this._symbolTable = fields.map((field) => {
2487
+ this._symbolTable = fields.map((field, position) => {
2488
+ this._throwIfAborted();
2120
2489
  const symbolsOffset = parseInt(field["Offset"], 10);
2121
2490
  const symbolsLength = parseInt(field["Length"], 10);
2122
2491
  const fieldName = field["FieldName"];
2123
- const neededSymbols = symbolsToKeep ? symbolsToKeep.get(fieldName) : null;
2492
+ const neededSymbols = symbolsToKeep ? symbolsToKeep[position] : null;
2124
2493
  const filteringEnabled = neededSymbols !== null;
2125
2494
  const symbols = [];
2126
2495
  let symbolIndex = 0;
@@ -2140,6 +2509,7 @@ var init_QvdFileReader = __esm({
2140
2509
  pointer += bytesRead - 1;
2141
2510
  symbolIndex++;
2142
2511
  }
2512
+ this._emitProgress("symbol-table", position + 1, fields.length);
2143
2513
  return symbols;
2144
2514
  });
2145
2515
  }
@@ -2158,13 +2528,18 @@ var init_QvdFileReader = __esm({
2158
2528
  * same for every row, so they are hoisted out of the loop and the inner loop does arithmetic
2159
2529
  * into a typed array and nothing else. Rows are assembled later, once, in `load()`.
2160
2530
  *
2161
- * @param {number|null} maxRows The maximum number of rows to parse. If null, all rows are parsed.
2531
+ * The window is what makes chunked iteration cheap: `decodeIndexColumn` walks records by
2532
+ * `base += recordSize`, so decoding rows k to k+n is a question of where the buffer slice starts
2533
+ * and how many iterations run. Nothing about the decoder changed to support it.
2534
+ *
2535
+ * @param {QvdRowWindow} window The rows to decode.
2162
2536
  */
2163
- async _parseIndexTable(maxRows = null) {
2164
- const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(maxRows, "parseIndexTable");
2537
+ async _parseIndexTable(window) {
2538
+ const { fields, recordSize, rowsToLoad, indexBuffer } = this._planIndexTable(window, "parseIndexTable");
2165
2539
  this._rowsDecoded = rowsToLoad;
2166
- this._indexColumns = fields.map(
2167
- (field) => decodeIndexColumn(
2540
+ this._indexColumns = fields.map((field, position) => {
2541
+ this._throwIfAborted();
2542
+ const column = decodeIndexColumn(
2168
2543
  indexBuffer,
2169
2544
  recordSize,
2170
2545
  rowsToLoad,
@@ -2172,8 +2547,10 @@ var init_QvdFileReader = __esm({
2172
2547
  parseInt(field["BitWidth"], 10),
2173
2548
  parseInt(field["Bias"], 10),
2174
2549
  new Int32Array(rowsToLoad)
2175
- )
2176
- );
2550
+ );
2551
+ this._emitProgress("index-table", position + 1, fields.length);
2552
+ return column;
2553
+ });
2177
2554
  }
2178
2555
  /**
2179
2556
  * Reads the file's schema and header metadata, without touching the symbol or index tables.
@@ -2190,8 +2567,11 @@ var init_QvdFileReader = __esm({
2190
2567
  * @return {Promise<import('./QvdDataFrame.js').QvdFileMetadata>} The file's schema and header.
2191
2568
  */
2192
2569
  async loadMetadata() {
2193
- await this._readData(null, true);
2570
+ await this._readData({ offset: 0, limit: null }, true);
2571
+ this._emitProgress("header", 0, 1);
2194
2572
  await this._parseHeader();
2573
+ this._emitProgress("header", 1, 1);
2574
+ this._throwIfAborted();
2195
2575
  assert2(this._header, "The QVD file header has not been parsed.");
2196
2576
  const header = this._header["QvdTableHeader"];
2197
2577
  let fields = header["Fields"]?.["QvdFieldHeader"] ?? [];
@@ -2226,114 +2606,200 @@ var init_QvdFileReader = __esm({
2226
2606
  /**
2227
2607
  * Loads the QVD file into memory and parses it.
2228
2608
  *
2229
- * @param {number|null} maxRows The maximum number of rows to load. If null, all rows are loaded.
2230
- * Must be a non-negative integer when given.
2231
- * @throws {QvdValidationError} If maxRows is neither null nor a non-negative integer.
2609
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
2610
+ * The rows to load. A number or null means what it always meant - the first N rows, or all of
2611
+ * them - and `{offset, limit}` is the same thing said more precisely, so `5` and
2612
+ * `{offset: 0, limit: 5}` are one read. `maxRows` is accepted as a second name for `limit`.
2613
+ * @throws {QvdValidationError} If the window is not a non-negative integer, null, or a valid
2614
+ * `{offset, limit}` object.
2232
2615
  * @return {Promise<QvdDataFrame>} The loaded QVD file.
2233
2616
  */
2234
- async load(maxRows = null) {
2235
- const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
2617
+ async load(window = null) {
2618
+ const rows = normaliseWindow(window, this._path);
2619
+ const prepared = await this._prepare(rows);
2620
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
2621
+ const data = this._buildRows(prepared.resolvedByField, 0, prepared.rowsAvailable);
2622
+ return new QvdDataFrame(data, prepared.columns, prepared.metadata, {
2623
+ ...prepared.loadStats,
2624
+ rowsLoaded: data.length
2625
+ });
2626
+ }
2627
+ /**
2628
+ * Reads the file as columns, without ever materialising rows.
2629
+ *
2630
+ * Shares every step with `load()` up to the point where rows would be built - see `_prepare`.
2631
+ * What it keeps instead is what the decoder already produced: one `Int32Array` of stored
2632
+ * indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
2633
+ * fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
2634
+ * bytes per row rather than a boxed value per cell, and the symbols are a few thousand
2635
+ * entries shared across every row that uses them.
2636
+ *
2637
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} [window]
2638
+ * The rows to decode, in the same spellings `load()` accepts.
2639
+ * @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
2640
+ */
2641
+ async loadColumnar(window = null) {
2642
+ const rows = normaliseWindow(window, this._path);
2643
+ const prepared = await this._prepare(rows);
2644
+ await this._parseIndexTable({ offset: prepared.offset, limit: prepared.rowsAvailable });
2645
+ const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
2236
2646
  assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2237
- const indexColumns = this._indexColumns;
2238
- const fieldCount = indexColumns.length;
2239
- const data = new Array(this._rowsDecoded);
2240
- for (let row = 0; row < this._rowsDecoded; row++) {
2241
- const values = new Array(fieldCount);
2242
- for (let field = 0; field < fieldCount; field++) {
2243
- const symbolIndex = indexColumns[field][row];
2244
- values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
2245
- }
2246
- data[row] = values;
2247
- }
2248
- loadStats.rowsLoaded = data.length;
2249
- return new QvdDataFrame(data, columns, metadata, loadStats);
2647
+ return new QvdColumnTable2({
2648
+ columns: prepared.columns,
2649
+ codesByField: this._indexColumns,
2650
+ symbolsByField: prepared.resolvedByField,
2651
+ rowCount: this._rowsDecoded,
2652
+ metadata: prepared.metadata,
2653
+ loadStats: { ...prepared.loadStats, rowsLoaded: this._rowsDecoded }
2654
+ });
2250
2655
  }
2251
2656
  /**
2252
- * Reads and decodes the file, stopping short of building rows.
2657
+ * Yields the window as data frames of at most `chunkSize` rows.
2253
2658
  *
2254
- * Everything `load()` and `loadColumnar()` have in common, which is everything except the
2255
- * shape of the answer. Two read paths for one binary format is the drift risk #113 is the
2256
- * standing example of - a stored index resolved one way here and another way there returns
2257
- * plausible wrong values and throws nothing - so there is one path, and the two entry points
2258
- * differ only in what they do with what it returns.
2659
+ * The file is opened, read and parsed **once**; only the index decode and the row building
2660
+ * happen per chunk. That is the whole reason this exists as a method rather than as a loop of
2661
+ * `load({offset, limit})` calls at the call site: the symbol table has to be parsed in full
2662
+ * whatever the chunk size - a stored index in the last chunk can address the first symbol -
2663
+ * and re-parsing it per chunk is what makes the obvious implementation cost more than a plain
2664
+ * load rather than less. PyQvd's chunked read does re-read it, and the comment on #140 records
2665
+ * that as a limitation rather than a design.
2259
2666
  *
2260
- * @param {number|null} maxRows Maximum rows to decode, or null for all of them.
2261
- * @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
2262
- * resolvedByField: Array<Array<any>>}>} The decoded file.
2263
- * @private
2667
+ * What it bounds is row materialisation, which is what actually dominates a large read's heap.
2668
+ * Two chunks of rows are alive at a time, not one - `for await` keeps the yielded frame
2669
+ * reachable while this generator builds the next - which is why `liveRows` below is
2670
+ * `chunkSize * 2`, and why the heap it needs is twice what one chunk suggests.
2671
+ *
2672
+ * A window covering no rows yields nothing at all, rather than one empty frame - so
2673
+ * `for await` over an exhausted offset does nothing, which is what a paging loop wants.
2674
+ *
2675
+ * @param {number|null|{offset?: number, limit?: number|null, maxRows?: number|null}} window
2676
+ * The rows to cover, in the same spellings `load()` accepts.
2677
+ * @param {number} chunkSize Rows per frame. Must be a positive integer.
2678
+ * @return {AsyncGenerator<QvdDataFrame>} The chunks, in order.
2264
2679
  */
2265
- async _decode(maxRows = null) {
2266
- if (maxRows !== null && (typeof maxRows !== "number" || !Number.isInteger(maxRows) || maxRows < 0)) {
2267
- throw new QvdValidationError("maxRows must be a non-negative integer, or null to load all rows", {
2268
- provided: maxRows,
2269
- type: typeof maxRows,
2680
+ async *iterateRows(window, chunkSize) {
2681
+ if (typeof chunkSize !== "number" || !Number.isInteger(chunkSize) || chunkSize <= 0) {
2682
+ throw new QvdValidationError("chunkSize must be a positive integer", {
2683
+ provided: chunkSize,
2684
+ type: typeof chunkSize,
2270
2685
  file: this._path
2271
2686
  });
2272
2687
  }
2273
- await this._readData(maxRows);
2688
+ const liveRows = { rows: chunkSize * 2, perChunk: 2 };
2689
+ const rows = normaliseWindow(window, this._path);
2690
+ const prepared = await this._prepare(rows, liveRows);
2691
+ for (let done = 0; done < prepared.rowsAvailable; done += chunkSize) {
2692
+ this._throwIfAborted();
2693
+ const count = Math.min(chunkSize, prepared.rowsAvailable - done);
2694
+ const offset = prepared.offset + done;
2695
+ await this._parseIndexTable({ offset, limit: count });
2696
+ const data = this._buildRows(prepared.resolvedByField, done, prepared.rowsAvailable);
2697
+ yield new QvdDataFrame(data, prepared.columns, prepared.metadata, {
2698
+ ...prepared.loadStats,
2699
+ offset,
2700
+ rowsLoaded: data.length
2701
+ });
2702
+ }
2703
+ }
2704
+ /**
2705
+ * Reads the file and resolves its symbols, stopping short of decoding any rows.
2706
+ *
2707
+ * Everything `load()`, `loadColumnar()` and `iterateRows()` have in common, which is everything
2708
+ * that depends on the file rather than on the window. Two read paths for one binary format is
2709
+ * the drift risk #113 is the standing example of - a stored index resolved one way here and
2710
+ * another way there returns plausible wrong values and throws nothing - so there is one path,
2711
+ * and the entry points differ only in what they do with what it returns and how many rows they
2712
+ * ask for at a time.
2713
+ *
2714
+ * @param {QvdRowWindow} window The rows the read covers.
2715
+ * @param {{rows: number, perChunk: number}|null} [liveRows] Rows held at one instant when that
2716
+ * is fewer than the window covers, and how many of them one row of the caller's chunk size
2717
+ * accounts for. Only `iterateRows` passes it; every other read holds what it covers.
2718
+ * @return {Promise<{columns: Array<string>, metadata: any, loadStats: any,
2719
+ * resolvedByField: Array<Array<any>>, rowsAvailable: number, offset: number}>} The parsed
2720
+ * file, with the window as it resolved against it.
2721
+ * @private
2722
+ */
2723
+ async _prepare(window, liveRows = null) {
2724
+ this._throwIfAborted();
2725
+ await this._readData(window, false, liveRows);
2726
+ this._emitProgress("header", 0, 1);
2274
2727
  await this._parseHeader();
2728
+ this._emitProgress("header", 1, 1);
2729
+ this._throwIfAborted();
2730
+ assert2(this._header, "The QVD file header has not been parsed.");
2731
+ const totalRows = parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10);
2732
+ const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2733
+ const resolved = resolveWindow(window, totalRows);
2734
+ const rowsAvailable = resolved.limit;
2275
2735
  let symbolsToKeep = null;
2276
2736
  let symbolsKept = null;
2277
- if (maxRows !== null && this._header) {
2278
- const symbolTableLength = parseInt(this._header["QvdTableHeader"]["Offset"], 10);
2737
+ if (window.limit !== null || window.offset > 0) {
2279
2738
  if (symbolTableLength > this._symbolFilteringThreshold) {
2280
- symbolsToKeep = await this._analyzeIndexTableSymbolUsage(maxRows);
2281
- symbolsKept = Array.from(symbolsToKeep.values()).reduce((sum, set) => sum + set.size, 0);
2739
+ symbolsToKeep = await this._analyzeIndexTableSymbolUsage({ offset: resolved.offset, limit: rowsAvailable });
2740
+ symbolsKept = symbolsToKeep.reduce((sum, set) => sum + set.size, 0);
2282
2741
  }
2283
2742
  }
2284
- await this._parseSymbolTable(symbolsToKeep, maxRows);
2285
- await this._parseIndexTable(maxRows);
2286
- assert2(this._header, "The QVD file header has not been parsed.");
2743
+ await this._parseSymbolTable(symbolsToKeep, rowsAvailable, liveRows);
2287
2744
  assert2(this._symbolTable, "The QVD file symbol table has not been parsed.");
2288
- assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2745
+ this._throwIfAborted();
2289
2746
  const resolvedByField = this._symbolTable.map((symbols) => {
2290
- const resolved = new Array(symbols.length);
2747
+ const resolved2 = new Array(symbols.length);
2291
2748
  for (let index = 0; index < symbols.length; index++) {
2292
2749
  const value = symbols[index]?.toPrimaryValue();
2293
- resolved[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
2750
+ resolved2[index] = typeof value === "string" && value.trim() !== "" && !isNaN(Number(value)) ? Number(value) : value;
2294
2751
  }
2295
- return resolved;
2752
+ return resolved2;
2296
2753
  });
2297
- let fields = this._header["QvdTableHeader"]["Fields"]["QvdFieldHeader"];
2298
- if (!Array.isArray(fields)) {
2299
- fields = [fields];
2300
- }
2301
- const columns = fields.map((field) => field["FieldName"]);
2754
+ assert2(this._selectedFields, "The QVD file fields have not been resolved.");
2755
+ const columns = this._selectedFields.map((field) => field["FieldName"]);
2302
2756
  const metadata = this._header["QvdTableHeader"];
2303
2757
  const loadStats = {
2304
- symbolTableBytes: parseInt(this._header["QvdTableHeader"]["Offset"], 10),
2305
- totalRows: parseInt(this._header["QvdTableHeader"]["NoOfRecords"], 10),
2306
- rowsLoaded: this._rowsDecoded,
2758
+ symbolTableBytes: symbolTableLength,
2759
+ totalRows,
2760
+ rowsLoaded: 0,
2761
+ offset: resolved.offset,
2307
2762
  symbolFiltering: symbolsToKeep !== null,
2308
2763
  symbolsKept
2309
2764
  };
2310
- return { columns, metadata, loadStats, resolvedByField };
2765
+ return { columns, metadata, loadStats, resolvedByField, rowsAvailable, offset: resolved.offset };
2311
2766
  }
2312
2767
  /**
2313
- * Reads the file as columns, without ever materialising rows.
2768
+ * Builds rows from the columns currently decoded.
2314
2769
  *
2315
- * Shares every step with `load()` up to the point where rows would be built - see `_decode`.
2316
- * What it keeps instead is what the decoder already produced: one `Int32Array` of stored
2317
- * indices per field, and one resolved value per distinct symbol. On the 1.7M x 20 taxi
2318
- * fixture that is 38.6 MiB against the 352.8 MiB `data` retains, because a column costs four
2319
- * bytes per row rather than a boxed value per cell, and the symbols are a few thousand
2320
- * entries shared across every row that uses them.
2770
+ * `data` stays eager: of the four ways this library is used - a full read, a preview already
2771
+ * bounded by a limit, writing an array out, and reading metadata - not one is helped by
2772
+ * materialising a row only when it is touched, and a lazy accessor would cost a proxy, a cache
2773
+ * and mutation semantics to serve none of them. A caller who wants columns without paying for
2774
+ * rows uses `QvdColumnTable`, which stops before this loop.
2321
2775
  *
2322
- * @param {number|null} maxRows The maximum number of rows to decode.
2323
- * @return {Promise<import('./QvdColumnTable.js').QvdColumnTable>} The decoded columns.
2776
+ * @param {Array<Array<any>>} resolvedByField One resolved value per distinct symbol, per field.
2777
+ * @param {number} progressBase Rows already delivered before this call, so that progress over a
2778
+ * chunked iteration counts the whole window rather than restarting at every chunk.
2779
+ * @param {number} progressTotal Rows the whole window covers.
2780
+ * @return {Array<Array<any>>} The rows.
2781
+ * @private
2324
2782
  */
2325
- async loadColumnar(maxRows = null) {
2326
- const { columns, metadata, loadStats, resolvedByField } = await this._decode(maxRows);
2327
- const { QvdColumnTable: QvdColumnTable2 } = await Promise.resolve().then(() => (init_QvdColumnTable(), QvdColumnTable_exports));
2783
+ _buildRows(resolvedByField, progressBase, progressTotal) {
2328
2784
  assert2(this._indexColumns, "The QVD file index table has not been parsed.");
2329
- return new QvdColumnTable2({
2330
- columns,
2331
- codesByField: this._indexColumns,
2332
- symbolsByField: resolvedByField,
2333
- rowCount: this._rowsDecoded,
2334
- metadata,
2335
- loadStats
2336
- });
2785
+ const indexColumns = this._indexColumns;
2786
+ const fieldCount = indexColumns.length;
2787
+ const rowCount = this._rowsDecoded;
2788
+ const data = new Array(rowCount);
2789
+ const reportInterval = Math.max(1, Math.floor(progressTotal / 100));
2790
+ for (let row = 0; row < rowCount; row++) {
2791
+ const values = new Array(fieldCount);
2792
+ for (let field = 0; field < fieldCount; field++) {
2793
+ const symbolIndex = indexColumns[field][row];
2794
+ values[field] = symbolIndex < 0 ? null : resolvedByField[field][symbolIndex];
2795
+ }
2796
+ data[row] = values;
2797
+ if ((progressBase + row + 1) % reportInterval === 0 || row + 1 === rowCount) {
2798
+ this._throwIfAborted();
2799
+ this._emitProgress("rows", progressBase + row + 1, progressTotal);
2800
+ }
2801
+ }
2802
+ return data;
2337
2803
  }
2338
2804
  };
2339
2805
  }
@@ -2344,6 +2810,7 @@ var QvdDataFrame;
2344
2810
  var init_QvdDataFrame = __esm({
2345
2811
  "src/QvdDataFrame.js"() {
2346
2812
  init_QvdErrors();
2813
+ init_readOptions();
2347
2814
  QvdDataFrame = class _QvdDataFrame {
2348
2815
  /**
2349
2816
  * Represents the data frame stored inside a QVD file.
@@ -2386,12 +2853,15 @@ var init_QvdDataFrame = __esm({
2386
2853
  /**
2387
2854
  * Returns statistics about the read that produced this data frame.
2388
2855
  *
2389
- * Only a frame returned by fromQvd() carries these; fromDict(), head() and tail() produce
2390
- * frames that describe no particular read, and report null rather than a stale figure.
2856
+ * Carried by every frame that came from a file - `fromQvd()`, and each chunk `iterate()` yields,
2857
+ * which is how a chunk reports its `offset`. `fromDict()`, `head()`, `tail()`, `rows()` and
2858
+ * `select()` describe no particular read and report null rather than a stale figure.
2391
2859
  *
2392
2860
  * The main use is confirming that a lazy load actually filtered the symbol table:
2393
2861
  * `symbolFiltering` says whether the two-pass path ran, and `symbolsKept` how many symbols
2394
- * survived it, which for a small maxRows should be a tiny fraction of the file's total.
2862
+ * survived it. Note that a bounded read does not filter on its own - the two-pass path engages
2863
+ * only above `symbolFilteringThreshold`, so on a file below it this reports false and every
2864
+ * symbol was parsed however few rows were asked for.
2395
2865
  *
2396
2866
  * @return {QvdLoadStats|null} Load statistics, or null if this frame did not come from a file.
2397
2867
  */
@@ -2768,6 +3238,18 @@ var init_QvdDataFrame = __esm({
2768
3238
  * @param {Object} [options] Optional loading options.
2769
3239
  * @param {number|null} [options.maxRows] The maximum number of rows to load. Must be a non-negative
2770
3240
  * integer; if not specified or null, all rows are loaded. Anything else throws a QvdValidationError.
3241
+ * This is the older name for `limit`; the two are the same option and passing both throws.
3242
+ * @param {number|null} [options.limit] Rows to read, counting from `offset`. The same number as
3243
+ * `maxRows`, spelled so that it reads correctly beside an offset.
3244
+ * @param {number} [options.offset=0] File row to start at. An offset past the end of the file
3245
+ * returns no rows rather than throwing, so a paging loop terminates on its own.
3246
+ * @param {Array<string>|null} [options.fields] Field names to read, in the order they should
3247
+ * appear in the result. Unselected fields have their symbols skipped entirely rather than
3248
+ * parsed and discarded. An unknown or repeated name throws.
3249
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}` as the
3250
+ * read proceeds - the same shape `toQvd`'s callback receives.
3251
+ * @param {AbortSignal} [options.signal] Cancels the read. The rejection is `signal.reason`,
3252
+ * which is a `DOMException` named `AbortError` unless you aborted with a reason of your own.
2771
3253
  * @param {string} [options.allowedDir] Optional allowed directory path. If provided, the file path
2772
3254
  * must be within this directory, with symlinks resolved first, so a link inside it that points
2773
3255
  * outside it is rejected. Defaults to the current working directory. To permit an entire
@@ -2780,17 +3262,42 @@ var init_QvdDataFrame = __esm({
2780
3262
  * **Zero disables the memory check entirely.**
2781
3263
  * @param {number} [options.symbolFilteringThreshold=52428800] Symbol table size, in bytes, above which
2782
3264
  * a lazy load switches to the two-pass filtering path. Defaults to 50MB.
2783
- * @throws {QvdValidationError} If options.maxRows is neither null/undefined nor a non-negative integer.
3265
+ * @throws {QvdValidationError} If a window option is not a non-negative integer, if both
3266
+ * `maxRows` and `limit` are given, or if `fields` names a column the file does not have.
2784
3267
  * @return {Promise<QvdDataFrame>} The data frame of the QVD file.
2785
3268
  */
2786
3269
  static async fromQvd(path3, options = {}) {
2787
3270
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
2788
- const readerOptions = {
2789
- allowedDir: options.allowedDir,
2790
- memorySafetyFactor: options.memorySafetyFactor,
2791
- symbolFilteringThreshold: options.symbolFilteringThreshold
2792
- };
2793
- return await new QvdFileReader2(path3, readerOptions).load(options.maxRows !== void 0 ? options.maxRows : null);
3271
+ return await new QvdFileReader2(path3, readerOptionsFrom(options)).load(windowFrom(options));
3272
+ }
3273
+ /**
3274
+ * Reads a QVD file in chunks, as an async generator of data frames.
3275
+ *
3276
+ * The file is opened, read and parsed once; only the index decode and the row building happen
3277
+ * per chunk, so what this bounds is row materialisation - the part that actually dominates a
3278
+ * large read's heap. It is **not** constant-memory reading of an arbitrarily large file: the
3279
+ * symbol table is parsed in full whatever the chunk size, because a stored index in the last
3280
+ * chunk can address the first symbol. On a high-cardinality file that table is the bulk of the
3281
+ * cost, and `readMetadata` is the only read that avoids it.
3282
+ *
3283
+ * ```js
3284
+ * for await (const chunk of QvdDataFrame.iterate('big.qvd', {chunkSize: 50_000})) {
3285
+ * process(chunk.data);
3286
+ * }
3287
+ * ```
3288
+ *
3289
+ * A window covering no rows yields nothing, so a loop over an exhausted offset simply does not
3290
+ * run its body.
3291
+ *
3292
+ * @param {string} path The path to the QVD file.
3293
+ * @param {Object} [options] The same options `fromQvd` takes, plus:
3294
+ * @param {number} [options.chunkSize=100000] Rows per frame. Must be a positive integer.
3295
+ * @return {AsyncGenerator<QvdDataFrame>} The chunks, in file order.
3296
+ */
3297
+ static async *iterate(path3, options = {}) {
3298
+ const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
3299
+ const reader = new QvdFileReader2(path3, readerOptionsFrom(options));
3300
+ yield* reader.iterateRows(windowFrom(options), options.chunkSize === void 0 ? 1e5 : options.chunkSize);
2794
3301
  }
2795
3302
  /**
2796
3303
  * Reads a QVD file's schema and header metadata, without reading its data.
@@ -2814,11 +3321,15 @@ var init_QvdDataFrame = __esm({
2814
3321
  * @param {Object} [options] Optional reading options.
2815
3322
  * @param {string} [options.allowedDir] Optional allowed directory path, applied exactly as it
2816
3323
  * is for `fromQvd`.
3324
+ * @param {Function} [options.onProgress] Called with `{stage, current, total, percent}`, as on
3325
+ * the reads that return data. Only the `read` and `header` stages occur here; there are no
3326
+ * symbols to parse and no rows to build.
3327
+ * @param {AbortSignal} [options.signal] Cancels the read, rejecting with `signal.reason`.
2817
3328
  * @return {Promise<QvdFileMetadata>} The file's schema and header metadata.
2818
3329
  */
2819
3330
  static async readMetadata(path3, options = {}) {
2820
3331
  const { QvdFileReader: QvdFileReader2 } = await Promise.resolve().then(() => (init_QvdFileReader(), QvdFileReader_exports));
2821
- return await new QvdFileReader2(path3, { allowedDir: options.allowedDir }).loadMetadata();
3332
+ return await new QvdFileReader2(path3, metadataOptionsFrom(options)).loadMetadata();
2822
3333
  }
2823
3334
  /**
2824
3335
  * Constructs a data frame from a dictionary.