@gmod/tabix 3.5.0 → 3.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -20
- package/dist/csi.js +1 -1
- package/dist/csi.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.d.ts +1 -0
- package/dist/tabixIndexedFile.js +124 -8
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/tbi.js +12 -15
- package/dist/tbi.js.map +1 -1
- package/dist/util.d.ts +11 -2
- package/dist/util.js +31 -5
- package/dist/util.js.map +1 -1
- package/esm/csi.js +2 -2
- package/esm/csi.js.map +1 -1
- package/esm/tabixIndexedFile.d.ts +1 -0
- package/esm/tabixIndexedFile.js +124 -8
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/tbi.js +13 -16
- package/esm/tbi.js.map +1 -1
- package/esm/util.d.ts +11 -2
- package/esm/util.js +30 -4
- package/esm/util.js.map +1 -1
- package/package.json +15 -14
- package/src/csi.ts +2 -2
- package/src/tabixIndexedFile.ts +156 -12
- package/src/tbi.ts +13 -14
- package/src/util.ts +35 -8
package/src/tabixIndexedFile.ts
CHANGED
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
import AbortablePromiseCache from '@gmod/abortable-promise-cache'
|
|
2
2
|
import { unzip, unzipChunkSlice } from '@gmod/bgzf-filehandle'
|
|
3
|
-
import LRU from '@jbrowse/quick-lru'
|
|
4
3
|
import { LocalFile, RemoteFile } from 'generic-filehandle2'
|
|
5
4
|
|
|
6
5
|
import CSI from './csi.ts'
|
|
@@ -17,6 +16,118 @@ const NEWLINE = 10
|
|
|
17
16
|
const CARRIAGE_RETURN = 13
|
|
18
17
|
const SEMICOLON = 59
|
|
19
18
|
|
|
19
|
+
// Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
|
|
20
|
+
// is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
|
|
21
|
+
// going much above it buys nothing on the transport that matters.
|
|
22
|
+
const MAX_READ_AHEAD_CHUNKS = 6
|
|
23
|
+
|
|
24
|
+
// SYNC: ~/src/gmod/bam-js/src/bamFile.ts DEFAULT_MAX_CACHE_BYTES
|
|
25
|
+
//
|
|
26
|
+
// We fetch compressed and cache decompressed, and an entry is a whole chunk, so
|
|
27
|
+
// entry count says nothing about memory. How little is easy to underestimate:
|
|
28
|
+
// a dense VCF (test/data/1kg.chr1.subset.vcf.gz — 213MB over 600kb of chr1)
|
|
29
|
+
// has single index bins of 17MB compressed, 120MB decompressed. Panning it
|
|
30
|
+
// under the old 80-entry cache peaked at 2GB RSS.
|
|
31
|
+
const DEFAULT_CHUNK_CACHE_BYTES = 100 * 2 ** 20
|
|
32
|
+
|
|
33
|
+
// The entry type AbortablePromiseCache stores in its backing cache. Not exported
|
|
34
|
+
// by the package, so it is recovered from the constructor signature rather than
|
|
35
|
+
// restated (and left to drift) here.
|
|
36
|
+
type CacheEntry = NonNullable<
|
|
37
|
+
ReturnType<
|
|
38
|
+
ConstructorParameters<
|
|
39
|
+
typeof AbortablePromiseCache<Chunk, ReadChunk>
|
|
40
|
+
>[0]['cache']['get']
|
|
41
|
+
>
|
|
42
|
+
>
|
|
43
|
+
|
|
44
|
+
interface CachedChunk {
|
|
45
|
+
entry: CacheEntry
|
|
46
|
+
/** 0 until the read settles and the decompressed size is known */
|
|
47
|
+
bytes: number
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Backing store for the chunk cache, bounded by the decompressed size of the
|
|
52
|
+
* chunks it holds rather than by entry count.
|
|
53
|
+
*
|
|
54
|
+
* A chunk's size is only known once its read settles, so `set` records the
|
|
55
|
+
* entry immediately and charges it to the budget later. Unsettled entries are
|
|
56
|
+
* therefore free, which is what we want: they are reads a query is waiting on.
|
|
57
|
+
*/
|
|
58
|
+
class ByteBoundedChunkCache {
|
|
59
|
+
private entries = new Map<string, CachedChunk>()
|
|
60
|
+
private bytes = 0
|
|
61
|
+
private maxBytes: number
|
|
62
|
+
|
|
63
|
+
constructor(maxBytes: number) {
|
|
64
|
+
this.maxBytes = maxBytes
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
get byteSize() {
|
|
68
|
+
return this.bytes
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
get size() {
|
|
72
|
+
return this.entries.size
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
has(key: string) {
|
|
76
|
+
return this.entries.has(key)
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
get(key: string) {
|
|
80
|
+
const cached = this.entries.get(key)
|
|
81
|
+
if (cached) {
|
|
82
|
+
// re-insert so Map iteration order stays least-recently-used first
|
|
83
|
+
this.entries.delete(key)
|
|
84
|
+
this.entries.set(key, cached)
|
|
85
|
+
}
|
|
86
|
+
return cached?.entry
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
set(key: string, entry: CacheEntry) {
|
|
90
|
+
this.delete(key)
|
|
91
|
+
const cached = { entry, bytes: 0 }
|
|
92
|
+
this.entries.set(key, cached)
|
|
93
|
+
void entry.promise
|
|
94
|
+
.then(chunk => {
|
|
95
|
+
// a later set() may have replaced this key while the read was in
|
|
96
|
+
// flight; charging these bytes to it would then double-count
|
|
97
|
+
if (this.entries.get(key) === cached) {
|
|
98
|
+
cached.bytes = chunk.buffer.byteLength
|
|
99
|
+
this.bytes += cached.bytes
|
|
100
|
+
this.evict()
|
|
101
|
+
}
|
|
102
|
+
})
|
|
103
|
+
.catch(() => {
|
|
104
|
+
// a failed or aborted read caches nothing and costs nothing
|
|
105
|
+
})
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
delete(key: string) {
|
|
109
|
+
const cached = this.entries.get(key)
|
|
110
|
+
if (cached) {
|
|
111
|
+
this.entries.delete(key)
|
|
112
|
+
this.bytes -= cached.bytes
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
keys() {
|
|
117
|
+
return this.entries.keys()
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// Evict from the least-recently-used end. The size > 1 guard means a single
|
|
121
|
+
// chunk larger than the whole budget is still kept: the caller needs it for
|
|
122
|
+
// the query in flight, and dropping it would only force a re-decompress.
|
|
123
|
+
private evict() {
|
|
124
|
+
const lru = this.entries.keys()
|
|
125
|
+
while (this.bytes > this.maxBytes && this.entries.size > 1) {
|
|
126
|
+
this.delete(lru.next().value!)
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
20
131
|
type GetLinesCallback = (
|
|
21
132
|
line: string,
|
|
22
133
|
fileOffset: number,
|
|
@@ -209,7 +320,7 @@ export default class TabixIndexedFile {
|
|
|
209
320
|
csiPath,
|
|
210
321
|
csiUrl,
|
|
211
322
|
csiFilehandle,
|
|
212
|
-
chunkCacheSize =
|
|
323
|
+
chunkCacheSize = DEFAULT_CHUNK_CACHE_BYTES,
|
|
213
324
|
}: {
|
|
214
325
|
path?: string
|
|
215
326
|
filehandle?: GenericFilehandle
|
|
@@ -220,6 +331,7 @@ export default class TabixIndexedFile {
|
|
|
220
331
|
csiPath?: string
|
|
221
332
|
csiUrl?: string
|
|
222
333
|
csiFilehandle?: GenericFilehandle
|
|
334
|
+
/** budget for the decompressed chunk cache, in bytes */
|
|
223
335
|
chunkCacheSize?: number
|
|
224
336
|
}) {
|
|
225
337
|
this.filehandle = resolveFilehandle(filehandle, path, url)
|
|
@@ -235,7 +347,7 @@ export default class TabixIndexedFile {
|
|
|
235
347
|
})
|
|
236
348
|
|
|
237
349
|
this.chunkCache = new AbortablePromiseCache<Chunk, ReadChunk>({
|
|
238
|
-
cache: new
|
|
350
|
+
cache: new ByteBoundedChunkCache(chunkCacheSize),
|
|
239
351
|
fill: (args: Chunk, signal?: AbortSignal) =>
|
|
240
352
|
this.readChunk(args, { signal }),
|
|
241
353
|
})
|
|
@@ -327,12 +439,39 @@ export default class TabixIndexedFile {
|
|
|
327
439
|
let downloadedBytes = 0
|
|
328
440
|
onProgress?.(0, totalBytes)
|
|
329
441
|
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
442
|
+
// Read ahead, but only as far as the scan has earned. Every chunk is its
|
|
443
|
+
// own range request, so a query spanning many of them pays a network round
|
|
444
|
+
// trip apiece, serially — for a remote file that dominates, well ahead of
|
|
445
|
+
// decompression (1kg.chr1 over a 1Mb window reads 22 chunks in a row).
|
|
446
|
+
//
|
|
447
|
+
// A fixed window would be wrong, though: blocksForRange offers a chunk per
|
|
448
|
+
// overlapping bin across every level, and on a sparse file the early return
|
|
449
|
+
// below stops the scan inside the first one, leaving the rest untouched
|
|
450
|
+
// (chr22_nanopore_subset offers 7 chunks and reads 1). Prefetching those
|
|
451
|
+
// would multiply the bytes such a query fetches for no gain.
|
|
452
|
+
//
|
|
453
|
+
// Finishing a chunk without hitting the early return proves the next chunk
|
|
454
|
+
// has to be examined, so the window starts at one and doubles per chunk
|
|
455
|
+
// consumed. A query that stops in its first chunk issues exactly the reads
|
|
456
|
+
// a sequential scan did; a long scan reaches full concurrency after three.
|
|
457
|
+
let readAhead = 1
|
|
458
|
+
const reads: Promise<ReadChunk>[] = []
|
|
459
|
+
const ensureReadsStarted = (count: number) => {
|
|
460
|
+
while (reads.length < Math.min(count, chunks.length)) {
|
|
461
|
+
const c = chunks[reads.length]!
|
|
462
|
+
const read = this.chunkCache.get(c.toString(), c, signal)
|
|
463
|
+
void read.catch(() => {
|
|
464
|
+
// a prefetch the early return skips is never awaited, so swallow its
|
|
465
|
+
// rejection here rather than let it surface unhandled
|
|
466
|
+
})
|
|
467
|
+
reads.push(read)
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
ensureReadsStarted(1)
|
|
471
|
+
|
|
472
|
+
for (let ci = 0, cl = chunks.length; ci < cl; ci++) {
|
|
473
|
+
const c = chunks[ci]!
|
|
474
|
+
const { buffer, cpositions, dpositions } = await reads[ci]!
|
|
336
475
|
downloadedBytes += c.fetchedSize()
|
|
337
476
|
onProgress?.(downloadedBytes, totalBytes)
|
|
338
477
|
const minvDataPosition = c.minv.dataPosition
|
|
@@ -375,14 +514,14 @@ export default class TabixIndexedFile {
|
|
|
375
514
|
blockStart = n + 1
|
|
376
515
|
continue
|
|
377
516
|
}
|
|
378
|
-
let
|
|
517
|
+
let isRefMatch = true
|
|
379
518
|
for (let i = 0; i < refLen; i++) {
|
|
380
519
|
if (buffer[refStart + i] !== regionRefNameBytes[i]) {
|
|
381
|
-
|
|
520
|
+
isRefMatch = false
|
|
382
521
|
break
|
|
383
522
|
}
|
|
384
523
|
}
|
|
385
|
-
if (!
|
|
524
|
+
if (!isRefMatch) {
|
|
386
525
|
blockStart = n + 1
|
|
387
526
|
continue
|
|
388
527
|
}
|
|
@@ -436,6 +575,11 @@ export default class TabixIndexedFile {
|
|
|
436
575
|
}
|
|
437
576
|
blockStart = n + 1
|
|
438
577
|
}
|
|
578
|
+
|
|
579
|
+
// every line in this chunk was still inside the query, so the next chunk
|
|
580
|
+
// has to be examined too - widen the window and start the reads for it
|
|
581
|
+
readAhead = Math.min(readAhead * 2, MAX_READ_AHEAD_CHUNKS)
|
|
582
|
+
ensureReadsStarted(ci + 1 + readAhead)
|
|
439
583
|
}
|
|
440
584
|
}
|
|
441
585
|
|
package/src/tbi.ts
CHANGED
|
@@ -4,8 +4,8 @@ import Chunk from './chunk.ts'
|
|
|
4
4
|
import IndexFile from './indexFile.ts'
|
|
5
5
|
import {
|
|
6
6
|
clampChunkEnds,
|
|
7
|
-
findFirstData,
|
|
8
7
|
memoizeByRefId,
|
|
8
|
+
minVirtualOffset,
|
|
9
9
|
optimizeChunks,
|
|
10
10
|
parseAuxData,
|
|
11
11
|
parsePseudoBin,
|
|
@@ -78,8 +78,15 @@ export default class TabixIndex extends IndexFile {
|
|
|
78
78
|
// nameSectionLength is at TBI offset 32; re-read to find where bin data starts
|
|
79
79
|
const nameSectionLength = dataView.getInt32(32, true)
|
|
80
80
|
|
|
81
|
-
// SYNC: ~/src/gmod/bam-js/src/
|
|
82
|
-
// First pass: record per-refId byte offsets and find firstDataLine
|
|
81
|
+
// SYNC: ~/src/gmod/bam-js/src/bai.ts _parse — two-pass structure
|
|
82
|
+
// First pass: record per-refId byte offsets and find firstDataLine.
|
|
83
|
+
//
|
|
84
|
+
// Only the linear index is consulted. Its entry for a window is the
|
|
85
|
+
// smallest virtual offset of any record overlapping that window, so the
|
|
86
|
+
// minimum over the linear index is already the minimum over the bin chunks
|
|
87
|
+
// — walking the chunks too only re-derives it. Checked against every .tbi
|
|
88
|
+
// in test/data: same answer on all 23, and no ref has bins without a
|
|
89
|
+
// linear index.
|
|
83
90
|
let curr = 36 + nameSectionLength
|
|
84
91
|
let firstDataLine: VirtualOffset | undefined
|
|
85
92
|
const offsets: number[] = []
|
|
@@ -97,21 +104,13 @@ export default class TabixIndex extends IndexFile {
|
|
|
97
104
|
throw new Error(
|
|
98
105
|
'tabix index contains too many bins, please use a CSI index',
|
|
99
106
|
)
|
|
100
|
-
} else if (bin === maxBinNumber + 1) {
|
|
101
|
-
curr += 16 * chunkCount
|
|
102
|
-
} else {
|
|
103
|
-
for (let k = 0; k < chunkCount; k++) {
|
|
104
|
-
firstDataLine = findFirstData(firstDataLine, fromBytes(bytes, curr))
|
|
105
|
-
curr += 16
|
|
106
|
-
}
|
|
107
107
|
}
|
|
108
|
+
curr += 16 * chunkCount
|
|
108
109
|
}
|
|
109
110
|
const linearCount = dataView.getInt32(curr, true)
|
|
110
111
|
curr += 4
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
curr += 8
|
|
114
|
-
}
|
|
112
|
+
firstDataLine = minVirtualOffset(bytes, curr, linearCount, firstDataLine)
|
|
113
|
+
curr += 8 * linearCount
|
|
115
114
|
}
|
|
116
115
|
|
|
117
116
|
function getIndices(refId: number): RefIndex | undefined {
|
package/src/util.ts
CHANGED
|
@@ -2,8 +2,7 @@ import LRU from '@jbrowse/quick-lru'
|
|
|
2
2
|
|
|
3
3
|
import Chunk from './chunk.ts'
|
|
4
4
|
import { longFromBytesToUnsigned } from './long.ts'
|
|
5
|
-
|
|
6
|
-
import type VirtualOffset from './virtualOffset.ts'
|
|
5
|
+
import VirtualOffset from './virtualOffset.ts'
|
|
7
6
|
|
|
8
7
|
// SYNC: ~/src/gmod/bam-js/src/util.ts optimizeChunks
|
|
9
8
|
export function optimizeChunks(chunks: Chunk[], lowest?: VirtualOffset) {
|
|
@@ -118,13 +117,41 @@ export function clampChunkEnds(
|
|
|
118
117
|
}
|
|
119
118
|
}
|
|
120
119
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
120
|
+
/**
|
|
121
|
+
* The smallest of `current` and the `count` packed virtual offsets starting at
|
|
122
|
+
* `offset`, allocating at most one VirtualOffset rather than one per entry.
|
|
123
|
+
*
|
|
124
|
+
* The index first pass exists only to find this minimum, and it visits every
|
|
125
|
+
* linear-index entry in the file to do it — 301k of them on
|
|
126
|
+
* test/data/failing_tabix.vcf.gz.tbi against 19k bin chunks. Building a
|
|
127
|
+
* VirtualOffset per entry to compare and discard it is the bulk of that pass.
|
|
128
|
+
*/
|
|
129
|
+
export function minVirtualOffset(
|
|
130
|
+
bytes: Uint8Array,
|
|
131
|
+
offset: number,
|
|
132
|
+
count: number,
|
|
133
|
+
current: VirtualOffset | undefined,
|
|
124
134
|
) {
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
135
|
+
let minBlock = current ? current.blockPosition : Infinity
|
|
136
|
+
let minData = current ? current.dataPosition : 0
|
|
137
|
+
let found = false
|
|
138
|
+
for (let i = 0; i < count; i++) {
|
|
139
|
+
const p = offset + i * 8
|
|
140
|
+
const block =
|
|
141
|
+
bytes[p + 7]! * 0x1_00_00_00_00_00 +
|
|
142
|
+
bytes[p + 6]! * 0x1_00_00_00_00 +
|
|
143
|
+
bytes[p + 5]! * 0x1_00_00_00 +
|
|
144
|
+
bytes[p + 4]! * 0x1_00_00 +
|
|
145
|
+
bytes[p + 3]! * 0x1_00 +
|
|
146
|
+
bytes[p + 2]!
|
|
147
|
+
const data = (bytes[p + 1]! << 8) | bytes[p]!
|
|
148
|
+
if (block < minBlock || (block === minBlock && data < minData)) {
|
|
149
|
+
minBlock = block
|
|
150
|
+
minData = data
|
|
151
|
+
found = true
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
return found ? new VirtualOffset(minBlock, minData) : current
|
|
128
155
|
}
|
|
129
156
|
|
|
130
157
|
export function parseNameBytes(namesBytes: Uint8Array) {
|