@gmod/bam 8.3.0 → 8.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -11
- package/dist/bai.d.ts +1 -0
- package/dist/bai.js +15 -0
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +58 -21
- package/dist/bamFile.js +25 -60
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.d.ts +1 -0
- package/dist/csi.js +5 -0
- package/dist/csi.js.map +1 -1
- package/dist/htsget.d.ts +11 -2
- package/dist/htsget.js.map +1 -1
- package/dist/indexFile.d.ts +44 -22
- package/dist/indexFile.js +55 -60
- package/dist/indexFile.js.map +1 -1
- package/dist/util.d.ts +40 -1
- package/dist/util.js +66 -6
- package/dist/util.js.map +1 -1
- package/esm/bai.d.ts +1 -0
- package/esm/bai.js +15 -0
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +58 -21
- package/esm/bamFile.js +23 -58
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.d.ts +1 -0
- package/esm/csi.js +5 -0
- package/esm/csi.js.map +1 -1
- package/esm/htsget.d.ts +11 -2
- package/esm/htsget.js.map +1 -1
- package/esm/indexFile.d.ts +44 -22
- package/esm/indexFile.js +56 -61
- package/esm/indexFile.js.map +1 -1
- package/esm/util.d.ts +40 -1
- package/esm/util.js +63 -4
- package/esm/util.js.map +1 -1
- package/package.json +2 -2
- package/src/bai.ts +16 -0
- package/src/bamFile.ts +65 -67
- package/src/csi.ts +6 -0
- package/src/htsget.ts +11 -2
- package/src/indexFile.ts +67 -64
- package/src/util.ts +73 -5
package/src/indexFile.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
+
import { SharedReadCache } from '@gmod/shared-read-cache'
|
|
1
2
|
import QuickLRU from '@jbrowse/quick-lru'
|
|
2
3
|
|
|
3
|
-
import {
|
|
4
|
+
import { chunksLikelyRead, optimizeChunks } from './util.ts'
|
|
4
5
|
|
|
5
6
|
import type Chunk from './chunk.ts'
|
|
6
7
|
import type { BaseOpts } from './util.ts'
|
|
@@ -53,14 +54,11 @@ export default abstract class IndexFile<
|
|
|
53
54
|
public filehandle: GenericFilehandle
|
|
54
55
|
public renameRefSeq: (s: string) => string
|
|
55
56
|
|
|
56
|
-
private parseP?: Promise<TParsed>
|
|
57
57
|
/**
|
|
58
|
-
* The
|
|
59
|
-
*
|
|
60
|
-
* this the first query to arrive would own a read all the others depend on —
|
|
61
|
-
* see {@link parse}.
|
|
58
|
+
* The parsed index, as a shared read — see {@link parse}. One entry, never
|
|
59
|
+
* evicted, which is what a memo is.
|
|
62
60
|
*/
|
|
63
|
-
private
|
|
61
|
+
private parseCache = new SharedReadCache<string, TParsed>({})
|
|
64
62
|
|
|
65
63
|
constructor({
|
|
66
64
|
filehandle,
|
|
@@ -95,6 +93,16 @@ export default abstract class IndexFile<
|
|
|
95
93
|
min: number,
|
|
96
94
|
): OffsetCoords | undefined
|
|
97
95
|
|
|
96
|
+
// Block position past which a chunk is EXPECTED to hold nothing overlapping
|
|
97
|
+
// [..., max] — the counterpart of getLowestChunk, and unlike it an estimate
|
|
98
|
+
// rather than a bound (see chunksLikelyRead). Only `estimatedBytesForRegions`
|
|
99
|
+
// may use it. CSI has no linear index and returns undefined, which reads as
|
|
100
|
+
// "no opinion" and leaves that estimate summing every chunk.
|
|
101
|
+
protected abstract getHighestChunk(
|
|
102
|
+
refIndex: RefIndex,
|
|
103
|
+
max: number,
|
|
104
|
+
): number | undefined
|
|
105
|
+
|
|
98
106
|
async blocksForRange(
|
|
99
107
|
refId: number,
|
|
100
108
|
min: number,
|
|
@@ -128,72 +136,34 @@ export default abstract class IndexFile<
|
|
|
128
136
|
return optimizeChunks(chunks, this.getLowestChunk(ba, min))
|
|
129
137
|
}
|
|
130
138
|
|
|
131
|
-
// SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same
|
|
132
|
-
//
|
|
139
|
+
// SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same shape and the same
|
|
140
|
+
// reasoning below.
|
|
133
141
|
/**
|
|
134
142
|
* Parse the index, or join the parse already running.
|
|
135
143
|
*
|
|
136
144
|
* The index is downloaded and parsed once for the life of this object, so it
|
|
137
145
|
* is the one read here that is shared between queries — and therefore the one
|
|
138
146
|
* place a cancellation can leak from the query that asked for it to a query
|
|
139
|
-
* that did not. `_parse` hands `opts` straight to `filehandle.readFile`, so
|
|
140
|
-
*
|
|
141
|
-
* depends on: when it pans away, every concurrent query
|
|
147
|
+
* that did not. `_parse` hands `opts` straight to `filehandle.readFile`, so a
|
|
148
|
+
* bare memoized promise makes the first query to arrive the owner of a read
|
|
149
|
+
* every other query depends on: when it pans away, every concurrent query
|
|
150
|
+
* fails with its abort.
|
|
142
151
|
*
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
*
|
|
147
|
-
*
|
|
148
|
-
* duplicate parse rather than a recursion whose depth depends on how the
|
|
149
|
-
* aborts interleave.
|
|
152
|
+
* The same cache the chunk reads use, for the same reason and with the same
|
|
153
|
+
* rule: the parse runs under a signal of its own and is cancelled only once
|
|
154
|
+
* every caller waiting on it has given up, so one query's abort is reported
|
|
155
|
+
* to that query alone. A rejection is dropped rather than cached, so a
|
|
156
|
+
* transient failure does not poison the index for the life of the file.
|
|
150
157
|
*
|
|
151
|
-
*
|
|
152
|
-
* the
|
|
153
|
-
*
|
|
154
|
-
*
|
|
158
|
+
* The fill is per call rather than on the cache so that the caller who starts
|
|
159
|
+
* the parse has its `onProgress` reach `filehandle.readFile` — the index is a
|
|
160
|
+
* whole-file read, and a determinate "downloading index" bar is what that
|
|
161
|
+
* callback exists for.
|
|
155
162
|
*/
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
if (!pending) {
|
|
160
|
-
return this.startParse(opts)
|
|
161
|
-
}
|
|
162
|
-
|
|
163
|
-
// read before awaiting: the owner is forgotten as soon as the parse settles
|
|
164
|
-
const ownerSignal = this.parseSignal
|
|
165
|
-
try {
|
|
166
|
-
return await pending
|
|
167
|
-
} catch (e) {
|
|
168
|
-
if (retried || !ownerSignal?.aborted || opts.signal?.aborted) {
|
|
169
|
-
throw e
|
|
170
|
-
}
|
|
171
|
-
return this.parse(opts, true)
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
|
|
175
|
-
private startParse(opts: BaseOpts) {
|
|
176
|
-
const pending = this._parse(opts)
|
|
177
|
-
this.parseP = pending
|
|
178
|
-
this.parseSignal = opts.signal
|
|
179
|
-
// Drop a rejection rather than keeping it, so one transient failure does not
|
|
180
|
-
// poison the index for the lifetime of the file. Both branches are
|
|
181
|
-
// identity-checked so a retry started after this settles is not cleared by
|
|
182
|
-
// the attempt it already replaced.
|
|
183
|
-
pending.then(
|
|
184
|
-
() => {
|
|
185
|
-
if (this.parseP === pending) {
|
|
186
|
-
this.parseSignal = undefined
|
|
187
|
-
}
|
|
188
|
-
},
|
|
189
|
-
() => {
|
|
190
|
-
if (this.parseP === pending) {
|
|
191
|
-
this.parseP = undefined
|
|
192
|
-
this.parseSignal = undefined
|
|
193
|
-
}
|
|
194
|
-
},
|
|
163
|
+
parse(opts: BaseOpts = {}): Promise<TParsed> {
|
|
164
|
+
return this.parseCache.get('index', opts.signal, signal =>
|
|
165
|
+
this._parse({ ...opts, signal }),
|
|
195
166
|
)
|
|
196
|
-
return pending
|
|
197
167
|
}
|
|
198
168
|
|
|
199
169
|
async lineCount(refId: number, opts?: BaseOpts) {
|
|
@@ -206,9 +176,42 @@ export default abstract class IndexFile<
|
|
|
206
176
|
return !!indexData.indices(seqId)
|
|
207
177
|
}
|
|
208
178
|
|
|
179
|
+
/**
|
|
180
|
+
* Compressed bytes a `getRecordsForRange` over these regions is expected to
|
|
181
|
+
* download, from the index alone.
|
|
182
|
+
*
|
|
183
|
+
* Per region this is `chunksLikelyRead`, not every chunk `blocksForRange`
|
|
184
|
+
* returns: the difference is the whole point on a long-read file, where a
|
|
185
|
+
* narrow window inherits every chunk of every overlapping bin and reads a
|
|
186
|
+
* handful of them. Measured on a 40x ONT BAM (COLO829BL, chr3), summing all
|
|
187
|
+
* chunks against the bytes the same query really pulls:
|
|
188
|
+
*
|
|
189
|
+
* | window | all chunks | actually read | this estimate |
|
|
190
|
+
* | ------ | ---------- | ------------- | ------------- |
|
|
191
|
+
* | 380bp | 43.5MB | 7.8MB | 7.8MB |
|
|
192
|
+
* | 3.4kb | 43.5MB | 7.8MB | 7.8MB |
|
|
193
|
+
* | 100kb | 46.6MB | 10.4MB | 10.4MB |
|
|
194
|
+
* | 2Mb | 155.9MB | 155.9MB | 123.8MB |
|
|
195
|
+
*
|
|
196
|
+
* A caller gating on this — jbrowse's "too much data" banner is the one that
|
|
197
|
+
* exists — was being told 5.6x the truth on exactly the windows a reader
|
|
198
|
+
* spends their time in, and cannot answer it by zooming: every window narrower
|
|
199
|
+
* than a linear-index interval resolves to the same chunks and so to the same
|
|
200
|
+
* number.
|
|
201
|
+
*
|
|
202
|
+
* Still summed over merged chunks rather than per region, so two regions
|
|
203
|
+
* sharing a chunk are charged for it once.
|
|
204
|
+
*/
|
|
209
205
|
async estimatedBytesForRegions(regions: Region[], opts?: BaseOpts) {
|
|
206
|
+
const indexData = await this.parse(opts)
|
|
210
207
|
const blockResults = await Promise.all(
|
|
211
|
-
regions.map(r =>
|
|
208
|
+
regions.map(async r => {
|
|
209
|
+
const chunks = await this.blocksForRange(r.refId, r.start, r.end, opts)
|
|
210
|
+
const refIndex = indexData.indices(r.refId)
|
|
211
|
+
return refIndex
|
|
212
|
+
? chunksLikelyRead(chunks, this.getHighestChunk(refIndex, r.end))
|
|
213
|
+
: chunks
|
|
214
|
+
}),
|
|
212
215
|
)
|
|
213
216
|
|
|
214
217
|
// Deduplicate and merge overlapping blocks across all regions
|
package/src/util.ts
CHANGED
|
@@ -29,6 +29,79 @@ export interface BaseOpts {
|
|
|
29
29
|
onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
// Re-exported so the internal import path stays './util.ts'. The
|
|
33
|
+
// implementation moved to @gmod/shared-read-cache, which needs it anyway --
|
|
34
|
+
// every consumer of that package was carrying an identical copy.
|
|
35
|
+
export { throwIfAborted } from '@gmod/shared-read-cache'
|
|
36
|
+
|
|
37
|
+
// How many of a query's chunks to read at once. Six is the HTTP/1.1 per-host
|
|
38
|
+
// connection cap browsers enforce, so going much above it buys nothing on the
|
|
39
|
+
// transport that matters and only widens peak memory.
|
|
40
|
+
//
|
|
41
|
+
// It is also the SMALLEST number of chunks any query reads, because
|
|
42
|
+
// `_fetchChunkFeatures` checks its early stop once, after this first batch
|
|
43
|
+
// (ADR 0010) — which is what `chunksLikelyRead` below is built on.
|
|
44
|
+
export const MAX_CONCURRENT_CHUNK_READS = 6
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* The prefix of `chunks` a query is expected to actually read, for callers that
|
|
48
|
+
* want to forecast a query's cost without running it.
|
|
49
|
+
*
|
|
50
|
+
* `blocksForRange` returns every chunk of every bin overlapping the query, at
|
|
51
|
+
* every level of the binning scheme, minus the ones the linear index puts
|
|
52
|
+
* entirely before it. On a long-read file that is wildly more than the query
|
|
53
|
+
* reads: a coarse bin's chunks run to the end of the bin's span, so a 380bp
|
|
54
|
+
* window on a deep ONT BAM resolves to 90 chunks / 43.5MB, of which
|
|
55
|
+
* `getRecordsForRange` reads 6 / 7.8MB before the early stop fires.
|
|
56
|
+
*
|
|
57
|
+
* Two bounds, and the answer is the larger:
|
|
58
|
+
*
|
|
59
|
+
* - `upperBoundBlockPosition` — the file offset past which a chunk holds no
|
|
60
|
+
* record the query wants, from the linear index one window beyond the query
|
|
61
|
+
* end. Chunks come back sorted by `minv`, so the ones under it are a prefix.
|
|
62
|
+
* - `MAX_CONCURRENT_CHUNK_READS` — the first batch is read unconditionally, so
|
|
63
|
+
* no query ever costs less than that however tight the upper bound is.
|
|
64
|
+
*
|
|
65
|
+
* **This is a forecast, not a bound the reader obeys**, and it is deliberately
|
|
66
|
+
* only used to describe a query rather than to answer one. The upper bound is
|
|
67
|
+
* an approximation: a record starting before the query end can sit at a higher
|
|
68
|
+
* offset than the linear-index entry past it (a long read reaching into the
|
|
69
|
+
* next window pins that entry low), so pruning a fetch this way would drop
|
|
70
|
+
* records. A forecast that is occasionally 20% under warns slightly early;
|
|
71
|
+
* a fetch that is occasionally short returns wrong data.
|
|
72
|
+
*
|
|
73
|
+
* An EMPTY prefix is that pin at its worst — the bound has landed at or before
|
|
74
|
+
* the query's own first chunk, so it orders nothing — and the answer there is
|
|
75
|
+
* every chunk rather than the batch floor. Without it this returns the floor on
|
|
76
|
+
* a file whose features are long against the linear-index interval, which is
|
|
77
|
+
* where the same forecast really does go wrong: ported to tabix-js and measured
|
|
78
|
+
* on the 1000 Genomes SV callset, whose 1.4Mb deletions pin the entry at the
|
|
79
|
+
* data start, it forecast 0.04MB against the 0.22MB the query read. On this
|
|
80
|
+
* corpus the fallback costs one fixture's win (chr22_nanopore, whose 10kb
|
|
81
|
+
* window keeps summing all 22 chunks) and no accuracy anywhere else.
|
|
82
|
+
*/
|
|
83
|
+
export function chunksLikelyRead(
|
|
84
|
+
chunks: Chunk[],
|
|
85
|
+
upperBoundBlockPosition?: number,
|
|
86
|
+
) {
|
|
87
|
+
if (
|
|
88
|
+
upperBoundBlockPosition === undefined ||
|
|
89
|
+
chunks.length <= MAX_CONCURRENT_CHUNK_READS
|
|
90
|
+
) {
|
|
91
|
+
return chunks
|
|
92
|
+
}
|
|
93
|
+
let n = 0
|
|
94
|
+
while (
|
|
95
|
+
n < chunks.length &&
|
|
96
|
+
chunks[n]!.minv.blockPosition < upperBoundBlockPosition
|
|
97
|
+
) {
|
|
98
|
+
n++
|
|
99
|
+
}
|
|
100
|
+
return n === 0 || n >= chunks.length
|
|
101
|
+
? chunks
|
|
102
|
+
: chunks.slice(0, Math.max(n, MAX_CONCURRENT_CHUNK_READS))
|
|
103
|
+
}
|
|
104
|
+
|
|
32
105
|
/**
|
|
33
106
|
* Merge and order the chunks a query resolved to.
|
|
34
107
|
*
|
|
@@ -39,11 +112,6 @@ export interface BaseOpts {
|
|
|
39
112
|
* underneath you. (The Chunk objects themselves are never mutated; a merged
|
|
40
113
|
* span produces a new instance.)
|
|
41
114
|
*/
|
|
42
|
-
// Re-exported so the internal import path stays './util.ts'. The
|
|
43
|
-
// implementation moved to @gmod/shared-read-cache, which needs it anyway --
|
|
44
|
-
// every consumer of that package was carrying an identical copy.
|
|
45
|
-
export { throwIfAborted } from '@gmod/shared-read-cache'
|
|
46
|
-
|
|
47
115
|
export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
|
|
48
116
|
const n = chunks.length
|
|
49
117
|
if (n === 0) {
|