@gmod/tabix 3.5.0 → 3.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,5 @@
1
1
  import AbortablePromiseCache from '@gmod/abortable-promise-cache'
2
2
  import { unzip, unzipChunkSlice } from '@gmod/bgzf-filehandle'
3
- import LRU from '@jbrowse/quick-lru'
4
3
  import { LocalFile, RemoteFile } from 'generic-filehandle2'
5
4
 
6
5
  import CSI from './csi.ts'
@@ -17,6 +16,118 @@ const NEWLINE = 10
17
16
  const CARRIAGE_RETURN = 13
18
17
  const SEMICOLON = 59
19
18
 
19
+ // Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
20
+ // is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
21
+ // going much above it buys nothing on the transport that matters.
22
+ const MAX_READ_AHEAD_CHUNKS = 6
23
+
24
+ // SYNC: ~/src/gmod/bam-js/src/bamFile.ts DEFAULT_MAX_CACHE_BYTES
25
+ //
26
+ // We fetch compressed and cache decompressed, and an entry is a whole chunk, so
27
+ // entry count says nothing about memory. How little is easy to underestimate:
28
+ // a dense VCF (test/data/1kg.chr1.subset.vcf.gz — 213MB over 600kb of chr1)
29
+ // has single index bins of 17MB compressed, 120MB decompressed. Panning it
30
+ // under the old 80-entry cache peaked at 2GB RSS.
31
+ const DEFAULT_CHUNK_CACHE_BYTES = 100 * 2 ** 20
32
+
33
+ // The entry type AbortablePromiseCache stores in its backing cache. Not exported
34
+ // by the package, so it is recovered from the constructor signature rather than
35
+ // restated (and left to drift) here.
36
+ type CacheEntry = NonNullable<
37
+ ReturnType<
38
+ ConstructorParameters<
39
+ typeof AbortablePromiseCache<Chunk, ReadChunk>
40
+ >[0]['cache']['get']
41
+ >
42
+ >
43
+
44
+ interface CachedChunk {
45
+ entry: CacheEntry
46
+ /** 0 until the read settles and the decompressed size is known */
47
+ bytes: number
48
+ }
49
+
50
+ /**
51
+ * Backing store for the chunk cache, bounded by the decompressed size of the
52
+ * chunks it holds rather than by entry count.
53
+ *
54
+ * A chunk's size is only known once its read settles, so `set` records the
55
+ * entry immediately and charges it to the budget later. Unsettled entries are
56
+ * therefore free, which is what we want: they are reads a query is waiting on.
57
+ */
58
+ class ByteBoundedChunkCache {
59
+ private entries = new Map<string, CachedChunk>()
60
+ private bytes = 0
61
+ private maxBytes: number
62
+
63
+ constructor(maxBytes: number) {
64
+ this.maxBytes = maxBytes
65
+ }
66
+
67
+ get byteSize() {
68
+ return this.bytes
69
+ }
70
+
71
+ get size() {
72
+ return this.entries.size
73
+ }
74
+
75
+ has(key: string) {
76
+ return this.entries.has(key)
77
+ }
78
+
79
+ get(key: string) {
80
+ const cached = this.entries.get(key)
81
+ if (cached) {
82
+ // re-insert so Map iteration order stays least-recently-used first
83
+ this.entries.delete(key)
84
+ this.entries.set(key, cached)
85
+ }
86
+ return cached?.entry
87
+ }
88
+
89
+ set(key: string, entry: CacheEntry) {
90
+ this.delete(key)
91
+ const cached = { entry, bytes: 0 }
92
+ this.entries.set(key, cached)
93
+ void entry.promise
94
+ .then(chunk => {
95
+ // a later set() may have replaced this key while the read was in
96
+ // flight; charging these bytes to it would then double-count
97
+ if (this.entries.get(key) === cached) {
98
+ cached.bytes = chunk.buffer.byteLength
99
+ this.bytes += cached.bytes
100
+ this.evict()
101
+ }
102
+ })
103
+ .catch(() => {
104
+ // a failed or aborted read caches nothing and costs nothing
105
+ })
106
+ }
107
+
108
+ delete(key: string) {
109
+ const cached = this.entries.get(key)
110
+ if (cached) {
111
+ this.entries.delete(key)
112
+ this.bytes -= cached.bytes
113
+ }
114
+ }
115
+
116
+ keys() {
117
+ return this.entries.keys()
118
+ }
119
+
120
+ // Evict from the least-recently-used end. The size > 1 guard means a single
121
+ // chunk larger than the whole budget is still kept: the caller needs it for
122
+ // the query in flight, and dropping it would only force a re-decompress.
123
+ private evict() {
124
+ const lru = this.entries.keys()
125
+ while (this.bytes > this.maxBytes && this.entries.size > 1) {
126
+ this.delete(lru.next().value!)
127
+ }
128
+ }
129
+ }
130
+
20
131
  type GetLinesCallback = (
21
132
  line: string,
22
133
  fileOffset: number,
@@ -209,7 +320,7 @@ export default class TabixIndexedFile {
209
320
  csiPath,
210
321
  csiUrl,
211
322
  csiFilehandle,
212
- chunkCacheSize = 5 * 2 ** 20,
323
+ chunkCacheSize = DEFAULT_CHUNK_CACHE_BYTES,
213
324
  }: {
214
325
  path?: string
215
326
  filehandle?: GenericFilehandle
@@ -220,6 +331,7 @@ export default class TabixIndexedFile {
220
331
  csiPath?: string
221
332
  csiUrl?: string
222
333
  csiFilehandle?: GenericFilehandle
334
+ /** budget for the decompressed chunk cache, in bytes */
223
335
  chunkCacheSize?: number
224
336
  }) {
225
337
  this.filehandle = resolveFilehandle(filehandle, path, url)
@@ -235,7 +347,7 @@ export default class TabixIndexedFile {
235
347
  })
236
348
 
237
349
  this.chunkCache = new AbortablePromiseCache<Chunk, ReadChunk>({
238
- cache: new LRU({ maxSize: Math.floor(chunkCacheSize / (1 << 16)) }),
350
+ cache: new ByteBoundedChunkCache(chunkCacheSize),
239
351
  fill: (args: Chunk, signal?: AbortSignal) =>
240
352
  this.readChunk(args, { signal }),
241
353
  })
@@ -327,12 +439,39 @@ export default class TabixIndexedFile {
327
439
  let downloadedBytes = 0
328
440
  onProgress?.(0, totalBytes)
329
441
 
330
- for (const c of chunks) {
331
- const { buffer, cpositions, dpositions } = await this.chunkCache.get(
332
- c.toString(),
333
- c,
334
- signal,
335
- )
442
+ // Read ahead, but only as far as the scan has earned. Every chunk is its
443
+ // own range request, so a query spanning many of them pays a network round
444
+ // trip apiece, serially — for a remote file that dominates, well ahead of
445
+ // decompression (1kg.chr1 over a 1Mb window reads 22 chunks in a row).
446
+ //
447
+ // A fixed window would be wrong, though: blocksForRange offers a chunk per
448
+ // overlapping bin across every level, and on a sparse file the early return
449
+ // below stops the scan inside the first one, leaving the rest untouched
450
+ // (chr22_nanopore_subset offers 7 chunks and reads 1). Prefetching those
451
+ // would multiply the bytes such a query fetches for no gain.
452
+ //
453
+ // Finishing a chunk without hitting the early return proves the next chunk
454
+ // has to be examined, so the window starts at one and doubles per chunk
455
+ // consumed. A query that stops in its first chunk issues exactly the reads
456
+ // a sequential scan did; a long scan reaches full concurrency after three.
457
+ let readAhead = 1
458
+ const reads: Promise<ReadChunk>[] = []
459
+ const ensureReadsStarted = (count: number) => {
460
+ while (reads.length < Math.min(count, chunks.length)) {
461
+ const c = chunks[reads.length]!
462
+ const read = this.chunkCache.get(c.toString(), c, signal)
463
+ void read.catch(() => {
464
+ // a prefetch the early return skips is never awaited, so swallow its
465
+ // rejection here rather than let it surface unhandled
466
+ })
467
+ reads.push(read)
468
+ }
469
+ }
470
+ ensureReadsStarted(1)
471
+
472
+ for (let ci = 0, cl = chunks.length; ci < cl; ci++) {
473
+ const c = chunks[ci]!
474
+ const { buffer, cpositions, dpositions } = await reads[ci]!
336
475
  downloadedBytes += c.fetchedSize()
337
476
  onProgress?.(downloadedBytes, totalBytes)
338
477
  const minvDataPosition = c.minv.dataPosition
@@ -375,14 +514,14 @@ export default class TabixIndexedFile {
375
514
  blockStart = n + 1
376
515
  continue
377
516
  }
378
- let refMatch = true
517
+ let isRefMatch = true
379
518
  for (let i = 0; i < refLen; i++) {
380
519
  if (buffer[refStart + i] !== regionRefNameBytes[i]) {
381
- refMatch = false
520
+ isRefMatch = false
382
521
  break
383
522
  }
384
523
  }
385
- if (!refMatch) {
524
+ if (!isRefMatch) {
386
525
  blockStart = n + 1
387
526
  continue
388
527
  }
@@ -436,6 +575,11 @@ export default class TabixIndexedFile {
436
575
  }
437
576
  blockStart = n + 1
438
577
  }
578
+
579
+ // every line in this chunk was still inside the query, so the next chunk
580
+ // has to be examined too - widen the window and start the reads for it
581
+ readAhead = Math.min(readAhead * 2, MAX_READ_AHEAD_CHUNKS)
582
+ ensureReadsStarted(ci + 1 + readAhead)
439
583
  }
440
584
  }
441
585
 
package/src/tbi.ts CHANGED
@@ -4,8 +4,8 @@ import Chunk from './chunk.ts'
4
4
  import IndexFile from './indexFile.ts'
5
5
  import {
6
6
  clampChunkEnds,
7
- findFirstData,
8
7
  memoizeByRefId,
8
+ minVirtualOffset,
9
9
  optimizeChunks,
10
10
  parseAuxData,
11
11
  parsePseudoBin,
@@ -78,8 +78,15 @@ export default class TabixIndex extends IndexFile {
78
78
  // nameSectionLength is at TBI offset 32; re-read to find where bin data starts
79
79
  const nameSectionLength = dataView.getInt32(32, true)
80
80
 
81
- // SYNC: ~/src/gmod/bam-js/src/csi.ts _parse — two-pass structure
82
- // First pass: record per-refId byte offsets and find firstDataLine
81
+ // SYNC: ~/src/gmod/bam-js/src/bai.ts _parse — two-pass structure
82
+ // First pass: record per-refId byte offsets and find firstDataLine.
83
+ //
84
+ // Only the linear index is consulted. Its entry for a window is the
85
+ // smallest virtual offset of any record overlapping that window, so the
86
+ // minimum over the linear index is already the minimum over the bin chunks
87
+ // — walking the chunks too only re-derives it. Checked against every .tbi
88
+ // in test/data: same answer on all 23, and no ref has bins without a
89
+ // linear index.
83
90
  let curr = 36 + nameSectionLength
84
91
  let firstDataLine: VirtualOffset | undefined
85
92
  const offsets: number[] = []
@@ -97,21 +104,13 @@ export default class TabixIndex extends IndexFile {
97
104
  throw new Error(
98
105
  'tabix index contains too many bins, please use a CSI index',
99
106
  )
100
- } else if (bin === maxBinNumber + 1) {
101
- curr += 16 * chunkCount
102
- } else {
103
- for (let k = 0; k < chunkCount; k++) {
104
- firstDataLine = findFirstData(firstDataLine, fromBytes(bytes, curr))
105
- curr += 16
106
- }
107
107
  }
108
+ curr += 16 * chunkCount
108
109
  }
109
110
  const linearCount = dataView.getInt32(curr, true)
110
111
  curr += 4
111
- for (let k = 0; k < linearCount; k++) {
112
- firstDataLine = findFirstData(firstDataLine, fromBytes(bytes, curr))
113
- curr += 8
114
- }
112
+ firstDataLine = minVirtualOffset(bytes, curr, linearCount, firstDataLine)
113
+ curr += 8 * linearCount
115
114
  }
116
115
 
117
116
  function getIndices(refId: number): RefIndex | undefined {
package/src/util.ts CHANGED
@@ -2,8 +2,7 @@ import LRU from '@jbrowse/quick-lru'
2
2
 
3
3
  import Chunk from './chunk.ts'
4
4
  import { longFromBytesToUnsigned } from './long.ts'
5
-
6
- import type VirtualOffset from './virtualOffset.ts'
5
+ import VirtualOffset from './virtualOffset.ts'
7
6
 
8
7
  // SYNC: ~/src/gmod/bam-js/src/util.ts optimizeChunks
9
8
  export function optimizeChunks(chunks: Chunk[], lowest?: VirtualOffset) {
@@ -118,13 +117,41 @@ export function clampChunkEnds(
118
117
  }
119
118
  }
120
119
 
121
- export function findFirstData(
122
- currentFdl: VirtualOffset | undefined,
123
- virtualOffset: VirtualOffset,
120
+ /**
121
+ * The smallest of `current` and the `count` packed virtual offsets starting at
122
+ * `offset`, allocating at most one VirtualOffset rather than one per entry.
123
+ *
124
+ * The index first pass exists only to find this minimum, and it visits every
125
+ * linear-index entry in the file to do it — 301k of them on
126
+ * test/data/failing_tabix.vcf.gz.tbi against 19k bin chunks. Building a
127
+ * VirtualOffset per entry to compare and discard it is the bulk of that pass.
128
+ */
129
+ export function minVirtualOffset(
130
+ bytes: Uint8Array,
131
+ offset: number,
132
+ count: number,
133
+ current: VirtualOffset | undefined,
124
134
  ) {
125
- return !currentFdl || currentFdl.compareTo(virtualOffset) > 0
126
- ? virtualOffset
127
- : currentFdl
135
+ let minBlock = current ? current.blockPosition : Infinity
136
+ let minData = current ? current.dataPosition : 0
137
+ let found = false
138
+ for (let i = 0; i < count; i++) {
139
+ const p = offset + i * 8
140
+ const block =
141
+ bytes[p + 7]! * 0x1_00_00_00_00_00 +
142
+ bytes[p + 6]! * 0x1_00_00_00_00 +
143
+ bytes[p + 5]! * 0x1_00_00_00 +
144
+ bytes[p + 4]! * 0x1_00_00 +
145
+ bytes[p + 3]! * 0x1_00 +
146
+ bytes[p + 2]!
147
+ const data = (bytes[p + 1]! << 8) | bytes[p]!
148
+ if (block < minBlock || (block === minBlock && data < minData)) {
149
+ minBlock = block
150
+ minData = data
151
+ found = true
152
+ }
153
+ }
154
+ return found ? new VirtualOffset(minBlock, minData) : current
128
155
  }
129
156
 
130
157
  export function parseNameBytes(namesBytes: Uint8Array) {