@gmod/bam 8.9.0 → 8.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +92 -38
- package/dist/bamFile.d.ts +4 -4
- package/dist/bamFile.js +8 -14
- package/dist/bamFile.js.map +1 -1
- package/dist/htsget.js +2 -35
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +4 -0
- package/dist/index.js +6 -1
- package/dist/index.js.map +1 -1
- package/dist/streamBam.d.ts +101 -0
- package/dist/streamBam.js +227 -0
- package/dist/streamBam.js.map +1 -0
- package/dist/util.d.ts +4 -0
- package/dist/util.js +9 -1
- package/dist/util.js.map +1 -1
- package/esm/bamFile.d.ts +4 -4
- package/esm/bamFile.js +3 -9
- package/esm/bamFile.js.map +1 -1
- package/esm/htsget.js +2 -2
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +4 -0
- package/esm/index.js +4 -0
- package/esm/index.js.map +1 -1
- package/esm/streamBam.d.ts +101 -0
- package/esm/streamBam.js +221 -0
- package/esm/streamBam.js.map +1 -0
- package/esm/util.d.ts +4 -0
- package/esm/util.js +7 -0
- package/esm/util.js.map +1 -1
- package/package.json +1 -1
- package/src/bamFile.ts +8 -19
- package/src/htsget.ts +7 -2
- package/src/index.ts +11 -0
- package/src/streamBam.ts +344 -0
- package/src/util.ts +17 -0
package/src/streamBam.ts
ADDED
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
import {
|
|
2
|
+
MAX_BGZF_BLOCK_SIZE,
|
|
3
|
+
scanBgzfBlocks,
|
|
4
|
+
unzip,
|
|
5
|
+
} from '@gmod/bgzf-filehandle'
|
|
6
|
+
|
|
7
|
+
import BAMFeature from './record.ts'
|
|
8
|
+
import { parseHeaderText } from './sam.ts'
|
|
9
|
+
import {
|
|
10
|
+
BAM_MAGIC,
|
|
11
|
+
concatUint8Array,
|
|
12
|
+
parseRefSeqs,
|
|
13
|
+
resolveFilehandle,
|
|
14
|
+
throwIfAborted,
|
|
15
|
+
} from './util.ts'
|
|
16
|
+
|
|
17
|
+
import type { BamRecordClass, BamRecordLike } from './bamFile.ts'
|
|
18
|
+
import type { BgzfBlockInfo, BgzfWorkerPool } from '@gmod/bgzf-filehandle'
|
|
19
|
+
import type { GenericFilehandle } from 'generic-filehandle2'
|
|
20
|
+
|
|
21
|
+
/** compressed bytes per read; ~15 BGZF blocks, so ~8MB decompressed */
|
|
22
|
+
const DEFAULT_WINDOW_SIZE = 1 << 20
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Window reads in flight at once. Four, matching the browser's per-host
|
|
26
|
+
* connection cap of six with room left for whatever else the page is fetching,
|
|
27
|
+
* and holding 4MB of compressed bytes at the default window size.
|
|
28
|
+
*/
|
|
29
|
+
const DEFAULT_READ_AHEAD = 4
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* A window's blocks, inflated on the pool when there is one.
|
|
33
|
+
*
|
|
34
|
+
* The pool's own interface is what a window already has to hand: the raw
|
|
35
|
+
* compressed bytes and the {@link scanBgzfBlocks} listing the loop needed
|
|
36
|
+
* anyway to find the window's edge. It splits the blocks across its workers and
|
|
37
|
+
* hands them back individually, which concatenate to exactly what `unzip` of
|
|
38
|
+
* the same range returns — the blocks of a BGZF stream are independent, which
|
|
39
|
+
* is what makes any of this parallel.
|
|
40
|
+
*
|
|
41
|
+
* One block is not worth a round trip, the same threshold `unzipChunkSlice`
|
|
42
|
+
* uses. `pool` being undefined is the ordinary case, not a failure: node has no
|
|
43
|
+
* workers, and `getSharedWorkerPool` resolves to undefined there.
|
|
44
|
+
*/
|
|
45
|
+
async function inflate(
|
|
46
|
+
compressed: Uint8Array,
|
|
47
|
+
blocks: BgzfBlockInfo[],
|
|
48
|
+
blocksEnd: number,
|
|
49
|
+
pool: BgzfWorkerPool | undefined,
|
|
50
|
+
) {
|
|
51
|
+
if (pool && blocks.length > 1) {
|
|
52
|
+
const { blocks: inflated } = await pool.decompressBlocks(compressed, blocks)
|
|
53
|
+
return concatUint8Array(inflated)
|
|
54
|
+
}
|
|
55
|
+
return unzip(compressed.subarray(0, blocksEnd))
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export interface BamStreamHeader {
|
|
59
|
+
/** the raw SAM header text, i.e. what `samtools view -H` prints */
|
|
60
|
+
headerText: string
|
|
61
|
+
/** the same text parsed into `@HD`/`@SQ`/`@RG`… lines and their tags */
|
|
62
|
+
samHeader: ReturnType<typeof parseHeaderText>
|
|
63
|
+
/** ref name to the `refId` records carry, from the binary ref-seq table */
|
|
64
|
+
chrToIndex: Record<string, number>
|
|
65
|
+
/** the inverse, indexed by `refId` */
|
|
66
|
+
indexToChr: { refName: string; length: number }[]
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export interface StreamBamOptions<T extends BamRecordLike = BAMFeature> {
|
|
70
|
+
bamFilehandle?: GenericFilehandle
|
|
71
|
+
bamPath?: string
|
|
72
|
+
bamUrl?: string
|
|
73
|
+
recordClass?: BamRecordClass<T>
|
|
74
|
+
renameRefSeqs?: (a: string) => string
|
|
75
|
+
signal?: AbortSignal
|
|
76
|
+
/**
|
|
77
|
+
* Fires once, before the first batch, with everything the ref-seq table and
|
|
78
|
+
* header text hold. A callback rather than a separate call, because the
|
|
79
|
+
* stream has to read the header anyway to find where the records start —
|
|
80
|
+
* making it a method would mean either reading the front of the file twice
|
|
81
|
+
* or holding state between two calls.
|
|
82
|
+
*/
|
|
83
|
+
onHeader?: (header: BamStreamHeader) => void
|
|
84
|
+
/**
|
|
85
|
+
* Worker pool to inflate each window on, as in {@link BamFile}. Without one
|
|
86
|
+
* the whole walk inflates on the calling thread, which on a large file is
|
|
87
|
+
* long enough to be worth keeping off whichever thread draws — pass
|
|
88
|
+
* `getSharedWorkerPool()` in a browser, or run the stream in a worker.
|
|
89
|
+
*
|
|
90
|
+
* Takes the promise as readily as the pool, since `getSharedWorkerPool()`
|
|
91
|
+
* returns one and awaiting an already-settled promise costs a microtask.
|
|
92
|
+
*/
|
|
93
|
+
bgzfWorkerPool?: BgzfWorkerPool | Promise<BgzfWorkerPool | undefined>
|
|
94
|
+
/**
|
|
95
|
+
* How many window reads to keep in flight at once. Defaults to
|
|
96
|
+
* {@link DEFAULT_READ_AHEAD}; 1 issues the next read only once the previous
|
|
97
|
+
* window has been handed to the caller.
|
|
98
|
+
*
|
|
99
|
+
* Depth is the whole point of it. One outstanding read overlaps a round trip
|
|
100
|
+
* with only the CPU spent on the window before it — 8% end to end over a
|
|
101
|
+
* local server with 20ms of latency — because the wait itself is still
|
|
102
|
+
* serial. Several outstanding turn N waits into roughly N/depth.
|
|
103
|
+
*
|
|
104
|
+
* Costs `depth` windows of compressed bytes held at once, and up to
|
|
105
|
+
* `depth - 1` wasted requests at EOF: a read's length is the only thing that
|
|
106
|
+
* says the file has ended, so the reads queued behind the last one have
|
|
107
|
+
* already gone out by the time it lands. A file that fits in a single window
|
|
108
|
+
* pays neither — the depth only opens up once a read comes back full.
|
|
109
|
+
*/
|
|
110
|
+
readAhead?: number
|
|
111
|
+
/**
|
|
112
|
+
* Compressed bytes to read per request. The default reads ~1MB at a time,
|
|
113
|
+
* which is a reasonable HTTP request size and bounds how much sits
|
|
114
|
+
* decompressed at once (~8MB, since BGZF blocks are capped at 64KB and
|
|
115
|
+
* compress ~8x). Values below one maximum-size block are raised to it, since
|
|
116
|
+
* a window that cannot hold a whole block can never make progress.
|
|
117
|
+
*/
|
|
118
|
+
windowSize?: number
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Reads every record in a BAM, in file order, without an index.
|
|
123
|
+
*
|
|
124
|
+
* For BAMs that no index can address: unsorted, or name-sorted as they come off
|
|
125
|
+
* the sequencer. {@link BamFile} answers `chr:start-end` by seeking to the
|
|
126
|
+
* chunks an index names, which a file in neither order has none of.
|
|
127
|
+
*
|
|
128
|
+
* Yields records a batch at a time, one batch per window read, rather than one
|
|
129
|
+
* record per `yield`. A whole-file walk over a 1GB BAM is tens of millions of
|
|
130
|
+
* records, and an async generator pays a promise per yield — batching moves
|
|
131
|
+
* that cost to once per few thousand records and lets the caller's inner loop
|
|
132
|
+
* be synchronous:
|
|
133
|
+
*
|
|
134
|
+
* ```js
|
|
135
|
+
* for await (const records of streamBamRecords({ bamPath: 'reads.bam' })) {
|
|
136
|
+
* for (const record of records) {
|
|
137
|
+
* // ...
|
|
138
|
+
* }
|
|
139
|
+
* }
|
|
140
|
+
* ```
|
|
141
|
+
*
|
|
142
|
+
* Deliberately a standalone function and not a `BamFile` method: it shares the
|
|
143
|
+
* record parser and header parser but none of the index, chunk or cache
|
|
144
|
+
* machinery, so a consumer who only streams does not pay for `BAI`/`CSI` in
|
|
145
|
+
* their bundle.
|
|
146
|
+
*
|
|
147
|
+
* Records are views into the window they were decompressed from, as everywhere
|
|
148
|
+
* else in this library. Holding one record from a batch retains that whole
|
|
149
|
+
* window (see {@link StreamBamOptions.windowSize}), so copy out the fields you
|
|
150
|
+
* want rather than keeping a sparse selection of records from a large file.
|
|
151
|
+
*/
|
|
152
|
+
export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
|
|
153
|
+
bamFilehandle,
|
|
154
|
+
bamPath,
|
|
155
|
+
bamUrl,
|
|
156
|
+
recordClass,
|
|
157
|
+
renameRefSeqs = n => n,
|
|
158
|
+
signal,
|
|
159
|
+
onHeader,
|
|
160
|
+
bgzfWorkerPool,
|
|
161
|
+
windowSize = DEFAULT_WINDOW_SIZE,
|
|
162
|
+
readAhead = DEFAULT_READ_AHEAD,
|
|
163
|
+
}: StreamBamOptions<T>): AsyncGenerator<T[], void, undefined> {
|
|
164
|
+
const bam = resolveFilehandle(bamFilehandle, bamPath, bamUrl)
|
|
165
|
+
if (!bam) {
|
|
166
|
+
throw new Error('no bam source: pass bamFilehandle, bamPath, or bamUrl')
|
|
167
|
+
}
|
|
168
|
+
const RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
|
|
169
|
+
const readLen = Math.max(windowSize, MAX_BGZF_BLOCK_SIZE)
|
|
170
|
+
const depth = Math.max(1, Math.floor(readAhead))
|
|
171
|
+
|
|
172
|
+
let filePosition = 0
|
|
173
|
+
// trailing bytes of the last window that did not complete a BGZF block, and
|
|
174
|
+
// that did not complete a BAM record, respectively. Both boundaries fall
|
|
175
|
+
// wherever they fall, so each window starts by finishing the last one's
|
|
176
|
+
// remainder.
|
|
177
|
+
let blockCarry: Uint8Array | undefined
|
|
178
|
+
let recordCarry: Uint8Array | undefined
|
|
179
|
+
let sawHeader = false
|
|
180
|
+
// 1-based, as readBamFeatures' virtual-offset ids are
|
|
181
|
+
let recordIndex = 0
|
|
182
|
+
const pool = await bgzfWorkerPool
|
|
183
|
+
|
|
184
|
+
// Several windows are in flight at once, consumed in the order they were
|
|
185
|
+
// asked for. Depth is what makes this worth doing: with one read outstanding
|
|
186
|
+
// the wait for a window can only overlap the CPU spent on the window before
|
|
187
|
+
// it, which is a fraction of a round trip — measured at no gain at all, where
|
|
188
|
+
// depth 4 hides most of the latency instead. See {@link
|
|
189
|
+
// StreamBamOptions.readAhead}.
|
|
190
|
+
const queue: Promise<Uint8Array>[] = []
|
|
191
|
+
let nextReadPosition = 0
|
|
192
|
+
let exhausted = false
|
|
193
|
+
// Widened to `depth` only once a read has come back full-length, i.e. once
|
|
194
|
+
// the file is known to be longer than one window. Opening at full depth
|
|
195
|
+
// would fetch a BAM smaller than a window in four requests, three of them
|
|
196
|
+
// answered 416, which is the common case for a small file and looks like a
|
|
197
|
+
// bug from the network tab. The ramp costs one round trip of depth on a file
|
|
198
|
+
// big enough to want it, out of however many windows it takes.
|
|
199
|
+
let allowed = 1
|
|
200
|
+
const topUp = () => {
|
|
201
|
+
while (!exhausted && queue.length < allowed) {
|
|
202
|
+
queue.push(bam.read(readLen, nextReadPosition, { signal }))
|
|
203
|
+
nextReadPosition += readLen
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
// Reads nobody will consume — queued past EOF, or still out when the caller
|
|
207
|
+
// broke out of the loop or a window threw. Each needs its rejection taken,
|
|
208
|
+
// or a failing read that arrives after we have stopped looking surfaces as
|
|
209
|
+
// an unhandled rejection.
|
|
210
|
+
const abandon = () => {
|
|
211
|
+
for (const p of queue.splice(0)) {
|
|
212
|
+
p.catch(() => undefined)
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
try {
|
|
217
|
+
topUp()
|
|
218
|
+
while (queue.length > 0) {
|
|
219
|
+
throwIfAborted(signal)
|
|
220
|
+
const read = await queue.shift()!
|
|
221
|
+
filePosition += read.length
|
|
222
|
+
// A short read is the end of the file — the filehandles here return
|
|
223
|
+
// exactly what was asked for until then, and an empty buffer past it.
|
|
224
|
+
// The reads already queued behind it are past EOF; drop them rather than
|
|
225
|
+
// decompressing empty buffers.
|
|
226
|
+
if (read.length < readLen) {
|
|
227
|
+
exhausted = true
|
|
228
|
+
abandon()
|
|
229
|
+
} else {
|
|
230
|
+
allowed = depth
|
|
231
|
+
}
|
|
232
|
+
topUp()
|
|
233
|
+
const compressed = blockCarry
|
|
234
|
+
? concatUint8Array([blockCarry, read])
|
|
235
|
+
: read
|
|
236
|
+
if (compressed.length === 0) {
|
|
237
|
+
break
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
const blocks = scanBgzfBlocks(compressed, 0, Number.POSITIVE_INFINITY)
|
|
241
|
+
const last = blocks.at(-1)
|
|
242
|
+
if (!last) {
|
|
243
|
+
throw new Error(
|
|
244
|
+
`not a BGZF stream: no valid block at byte ${filePosition - compressed.length}`,
|
|
245
|
+
)
|
|
246
|
+
}
|
|
247
|
+
const blocksEnd = last.inputOffset + last.compressedSize
|
|
248
|
+
// copied, not subarray'd: the remainder is carried across an await, and a
|
|
249
|
+
// view would pin the whole window it came from
|
|
250
|
+
blockCarry =
|
|
251
|
+
blocksEnd < compressed.length ? compressed.slice(blocksEnd) : undefined
|
|
252
|
+
|
|
253
|
+
const decompressed = await inflate(compressed, blocks, blocksEnd, pool)
|
|
254
|
+
const bytes = recordCarry
|
|
255
|
+
? concatUint8Array([recordCarry, decompressed])
|
|
256
|
+
: decompressed
|
|
257
|
+
const dataView = new DataView(
|
|
258
|
+
bytes.buffer,
|
|
259
|
+
bytes.byteOffset,
|
|
260
|
+
bytes.byteLength,
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
let blockStart = 0
|
|
264
|
+
if (!sawHeader) {
|
|
265
|
+
if (bytes.length < 8 || dataView.getInt32(0, true) !== BAM_MAGIC) {
|
|
266
|
+
throw new Error('not a BAM file: bad magic')
|
|
267
|
+
}
|
|
268
|
+
const lText = dataView.getInt32(4, true)
|
|
269
|
+
const refs = parseRefSeqs(bytes, 8 + lText, renameRefSeqs)
|
|
270
|
+
if (!refs) {
|
|
271
|
+
// a header with enough contigs to span a window: keep the whole thing
|
|
272
|
+
// and try again with more of it
|
|
273
|
+
recordCarry = bytes
|
|
274
|
+
continue
|
|
275
|
+
}
|
|
276
|
+
const headerText = new TextDecoder('utf8').decode(
|
|
277
|
+
bytes.subarray(8, 8 + lText),
|
|
278
|
+
)
|
|
279
|
+
onHeader?.({
|
|
280
|
+
headerText,
|
|
281
|
+
samHeader: parseHeaderText(headerText),
|
|
282
|
+
chrToIndex: refs.chrToIndex,
|
|
283
|
+
indexToChr: refs.indexToChr,
|
|
284
|
+
})
|
|
285
|
+
sawHeader = true
|
|
286
|
+
blockStart = refs.end
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
const sink: T[] = []
|
|
290
|
+
while (blockStart + 4 <= bytes.length) {
|
|
291
|
+
const blockSize = dataView.getInt32(blockStart, true)
|
|
292
|
+
const blockEnd = blockStart + 4 + blockSize - 1
|
|
293
|
+
if (blockEnd >= bytes.length) {
|
|
294
|
+
break
|
|
295
|
+
}
|
|
296
|
+
sink.push(
|
|
297
|
+
new RecordClass(
|
|
298
|
+
bytes,
|
|
299
|
+
blockStart,
|
|
300
|
+
blockEnd,
|
|
301
|
+
// An indexed read derives fileOffset from the record's BGZF virtual
|
|
302
|
+
// offset, which nothing here has: there is no index to seek back
|
|
303
|
+
// with. Its ordinal in the file is the other thing that identifies a
|
|
304
|
+
// record by position, and it is the same for a given file whatever
|
|
305
|
+
// the window size, so it is as stable as a virtual offset without
|
|
306
|
+
// pretending to be one.
|
|
307
|
+
//
|
|
308
|
+
// NOT the crc32 of the record bytes that readBamFeatures falls back
|
|
309
|
+
// to when it has no positions. That is a content hash, so the two
|
|
310
|
+
// byte-identical records in exact_duplicate.bam collide on it, and
|
|
311
|
+
// hashing every record costs ~40% of the walk (135ms of 340ms over
|
|
312
|
+
// out.bam) where a counter costs nothing. The fallback only fires on
|
|
313
|
+
// an unusual path there; here it would fire on every record.
|
|
314
|
+
recordIndex++,
|
|
315
|
+
dataView,
|
|
316
|
+
),
|
|
317
|
+
)
|
|
318
|
+
blockStart = blockEnd + 1
|
|
319
|
+
}
|
|
320
|
+
recordCarry =
|
|
321
|
+
blockStart < bytes.length ? bytes.slice(blockStart) : undefined
|
|
322
|
+
|
|
323
|
+
if (sink.length > 0) {
|
|
324
|
+
yield sink
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
} finally {
|
|
328
|
+
abandon()
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
if (!sawHeader) {
|
|
332
|
+
throw new Error('Insufficient data for reference sequences')
|
|
333
|
+
}
|
|
334
|
+
if (blockCarry !== undefined) {
|
|
335
|
+
throw new Error(
|
|
336
|
+
`truncated BAM: ${blockCarry.length} trailing bytes are not a complete BGZF block`,
|
|
337
|
+
)
|
|
338
|
+
}
|
|
339
|
+
if (recordCarry !== undefined) {
|
|
340
|
+
throw new Error(
|
|
341
|
+
`truncated BAM: ${recordCarry.length} bytes left after the last complete record`,
|
|
342
|
+
)
|
|
343
|
+
}
|
|
344
|
+
}
|
package/src/util.ts
CHANGED
|
@@ -1,7 +1,24 @@
|
|
|
1
|
+
import { LocalFile, RemoteFile } from 'generic-filehandle2'
|
|
2
|
+
|
|
1
3
|
import Chunk from './chunk.ts'
|
|
2
4
|
import { VirtualOffset } from './virtualOffset.ts'
|
|
3
5
|
|
|
4
6
|
import type { OffsetCoords } from './virtualOffset.ts'
|
|
7
|
+
import type { GenericFilehandle } from 'generic-filehandle2'
|
|
8
|
+
|
|
9
|
+
/** 'BAM\1' read as a little-endian int32 */
|
|
10
|
+
export const BAM_MAGIC = 21840194
|
|
11
|
+
|
|
12
|
+
export function resolveFilehandle(
|
|
13
|
+
filehandle?: GenericFilehandle,
|
|
14
|
+
path?: string,
|
|
15
|
+
url?: string,
|
|
16
|
+
) {
|
|
17
|
+
return (
|
|
18
|
+
filehandle ??
|
|
19
|
+
(path ? new LocalFile(path) : url ? new RemoteFile(url) : undefined)
|
|
20
|
+
)
|
|
21
|
+
}
|
|
5
22
|
|
|
6
23
|
export interface BamOpts {
|
|
7
24
|
viewAsPairs?: boolean
|