@gmod/bam 8.10.0 → 8.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,344 @@
1
+ import {
2
+ MAX_BGZF_BLOCK_SIZE,
3
+ scanBgzfBlocks,
4
+ unzip,
5
+ } from '@gmod/bgzf-filehandle'
6
+
7
+ import BAMFeature from './record.ts'
8
+ import { parseHeaderText } from './sam.ts'
9
+ import {
10
+ BAM_MAGIC,
11
+ concatUint8Array,
12
+ parseRefSeqs,
13
+ resolveFilehandle,
14
+ throwIfAborted,
15
+ } from './util.ts'
16
+
17
+ import type { BamRecordClass, BamRecordLike } from './bamFile.ts'
18
+ import type { BgzfBlockInfo, BgzfWorkerPool } from '@gmod/bgzf-filehandle'
19
+ import type { GenericFilehandle } from 'generic-filehandle2'
20
+
21
+ /** compressed bytes per read; ~15 BGZF blocks, so ~8MB decompressed */
22
+ const DEFAULT_WINDOW_SIZE = 1 << 20
23
+
24
+ /**
25
+ * Window reads in flight at once. Four, matching the browser's per-host
26
+ * connection cap of six with room left for whatever else the page is fetching,
27
+ * and holding 4MB of compressed bytes at the default window size.
28
+ */
29
+ const DEFAULT_READ_AHEAD = 4
30
+
31
+ /**
32
+ * A window's blocks, inflated on the pool when there is one.
33
+ *
34
+ * The pool's own interface is what a window already has to hand: the raw
35
+ * compressed bytes and the {@link scanBgzfBlocks} listing the loop needed
36
+ * anyway to find the window's edge. It splits the blocks across its workers and
37
+ * hands them back individually, which concatenate to exactly what `unzip` of
38
+ * the same range returns — the blocks of a BGZF stream are independent, which
39
+ * is what makes any of this parallel.
40
+ *
41
+ * One block is not worth a round trip, the same threshold `unzipChunkSlice`
42
+ * uses. `pool` being undefined is the ordinary case, not a failure: node has no
43
+ * workers, and `getSharedWorkerPool` resolves to undefined there.
44
+ */
45
+ async function inflate(
46
+ compressed: Uint8Array,
47
+ blocks: BgzfBlockInfo[],
48
+ blocksEnd: number,
49
+ pool: BgzfWorkerPool | undefined,
50
+ ) {
51
+ if (pool && blocks.length > 1) {
52
+ const { blocks: inflated } = await pool.decompressBlocks(compressed, blocks)
53
+ return concatUint8Array(inflated)
54
+ }
55
+ return unzip(compressed.subarray(0, blocksEnd))
56
+ }
57
+
58
+ export interface BamStreamHeader {
59
+ /** the raw SAM header text, i.e. what `samtools view -H` prints */
60
+ headerText: string
61
+ /** the same text parsed into `@HD`/`@SQ`/`@RG`… lines and their tags */
62
+ samHeader: ReturnType<typeof parseHeaderText>
63
+ /** ref name to the `refId` records carry, from the binary ref-seq table */
64
+ chrToIndex: Record<string, number>
65
+ /** the inverse, indexed by `refId` */
66
+ indexToChr: { refName: string; length: number }[]
67
+ }
68
+
69
+ export interface StreamBamOptions<T extends BamRecordLike = BAMFeature> {
70
+ bamFilehandle?: GenericFilehandle
71
+ bamPath?: string
72
+ bamUrl?: string
73
+ recordClass?: BamRecordClass<T>
74
+ renameRefSeqs?: (a: string) => string
75
+ signal?: AbortSignal
76
+ /**
77
+ * Fires once, before the first batch, with everything the ref-seq table and
78
+ * header text hold. A callback rather than a separate call, because the
79
+ * stream has to read the header anyway to find where the records start —
80
+ * making it a method would mean either reading the front of the file twice
81
+ * or holding state between two calls.
82
+ */
83
+ onHeader?: (header: BamStreamHeader) => void
84
+ /**
85
+ * Worker pool to inflate each window on, as in {@link BamFile}. Without one
86
+ * the whole walk inflates on the calling thread, which on a large file is
87
+ * long enough to be worth keeping off whichever thread draws — pass
88
+ * `getSharedWorkerPool()` in a browser, or run the stream in a worker.
89
+ *
90
+ * Takes the promise as readily as the pool, since `getSharedWorkerPool()`
91
+ * returns one and awaiting an already-settled promise costs a microtask.
92
+ */
93
+ bgzfWorkerPool?: BgzfWorkerPool | Promise<BgzfWorkerPool | undefined>
94
+ /**
95
+ * How many window reads to keep in flight at once. Defaults to
96
+ * {@link DEFAULT_READ_AHEAD}; 1 issues the next read only once the previous
97
+ * window has been handed to the caller.
98
+ *
99
+ * Depth is the whole point of it. One outstanding read overlaps a round trip
100
+ * with only the CPU spent on the window before it — 8% end to end over a
101
+ * local server with 20ms of latency — because the wait itself is still
102
+ * serial. Several outstanding turn N waits into roughly N/depth.
103
+ *
104
+ * Costs `depth` windows of compressed bytes held at once, and up to
105
+ * `depth - 1` wasted requests at EOF: a read's length is the only thing that
106
+ * says the file has ended, so the reads queued behind the last one have
107
+ * already gone out by the time it lands. A file that fits in a single window
108
+ * pays neither — the depth only opens up once a read comes back full.
109
+ */
110
+ readAhead?: number
111
+ /**
112
+ * Compressed bytes to read per request. The default reads ~1MB at a time,
113
+ * which is a reasonable HTTP request size and bounds how much sits
114
+ * decompressed at once (~8MB, since BGZF blocks are capped at 64KB and
115
+ * compress ~8x). Values below one maximum-size block are raised to it, since
116
+ * a window that cannot hold a whole block can never make progress.
117
+ */
118
+ windowSize?: number
119
+ }
120
+
121
+ /**
122
+ * Reads every record in a BAM, in file order, without an index.
123
+ *
124
+ * For BAMs that no index can address: unsorted, or name-sorted as they come off
125
+ * the sequencer. {@link BamFile} answers `chr:start-end` by seeking to the
126
+ * chunks an index names, which a file in neither order has none of.
127
+ *
128
+ * Yields records a batch at a time, one batch per window read, rather than one
129
+ * record per `yield`. A whole-file walk over a 1GB BAM is tens of millions of
130
+ * records, and an async generator pays a promise per yield — batching moves
131
+ * that cost to once per few thousand records and lets the caller's inner loop
132
+ * be synchronous:
133
+ *
134
+ * ```js
135
+ * for await (const records of streamBamRecords({ bamPath: 'reads.bam' })) {
136
+ * for (const record of records) {
137
+ * // ...
138
+ * }
139
+ * }
140
+ * ```
141
+ *
142
+ * Deliberately a standalone function and not a `BamFile` method: it shares the
143
+ * record parser and header parser but none of the index, chunk or cache
144
+ * machinery, so a consumer who only streams does not pay for `BAI`/`CSI` in
145
+ * their bundle.
146
+ *
147
+ * Records are views into the window they were decompressed from, as everywhere
148
+ * else in this library. Holding one record from a batch retains that whole
149
+ * window (see {@link StreamBamOptions.windowSize}), so copy out the fields you
150
+ * want rather than keeping a sparse selection of records from a large file.
151
+ */
152
+ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
153
+ bamFilehandle,
154
+ bamPath,
155
+ bamUrl,
156
+ recordClass,
157
+ renameRefSeqs = n => n,
158
+ signal,
159
+ onHeader,
160
+ bgzfWorkerPool,
161
+ windowSize = DEFAULT_WINDOW_SIZE,
162
+ readAhead = DEFAULT_READ_AHEAD,
163
+ }: StreamBamOptions<T>): AsyncGenerator<T[], void, undefined> {
164
+ const bam = resolveFilehandle(bamFilehandle, bamPath, bamUrl)
165
+ if (!bam) {
166
+ throw new Error('no bam source: pass bamFilehandle, bamPath, or bamUrl')
167
+ }
168
+ const RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
169
+ const readLen = Math.max(windowSize, MAX_BGZF_BLOCK_SIZE)
170
+ const depth = Math.max(1, Math.floor(readAhead))
171
+
172
+ let filePosition = 0
173
+ // trailing bytes of the last window that did not complete a BGZF block, and
174
+ // that did not complete a BAM record, respectively. Both boundaries fall
175
+ // wherever they fall, so each window starts by finishing the last one's
176
+ // remainder.
177
+ let blockCarry: Uint8Array | undefined
178
+ let recordCarry: Uint8Array | undefined
179
+ let sawHeader = false
180
+ // 1-based, as readBamFeatures' virtual-offset ids are
181
+ let recordIndex = 0
182
+ const pool = await bgzfWorkerPool
183
+
184
+ // Several windows are in flight at once, consumed in the order they were
185
+ // asked for. Depth is what makes this worth doing: with one read outstanding
186
+ // the wait for a window can only overlap the CPU spent on the window before
187
+ // it, which is a fraction of a round trip — measured at no gain at all, where
188
+ // depth 4 hides most of the latency instead. See {@link
189
+ // StreamBamOptions.readAhead}.
190
+ const queue: Promise<Uint8Array>[] = []
191
+ let nextReadPosition = 0
192
+ let exhausted = false
193
+ // Widened to `depth` only once a read has come back full-length, i.e. once
194
+ // the file is known to be longer than one window. Opening at full depth
195
+ // would fetch a BAM smaller than a window in four requests, three of them
196
+ // answered 416, which is the common case for a small file and looks like a
197
+ // bug from the network tab. The ramp costs one round trip of depth on a file
198
+ // big enough to want it, out of however many windows it takes.
199
+ let allowed = 1
200
+ const topUp = () => {
201
+ while (!exhausted && queue.length < allowed) {
202
+ queue.push(bam.read(readLen, nextReadPosition, { signal }))
203
+ nextReadPosition += readLen
204
+ }
205
+ }
206
+ // Reads nobody will consume — queued past EOF, or still out when the caller
207
+ // broke out of the loop or a window threw. Each needs its rejection taken,
208
+ // or a failing read that arrives after we have stopped looking surfaces as
209
+ // an unhandled rejection.
210
+ const abandon = () => {
211
+ for (const p of queue.splice(0)) {
212
+ p.catch(() => undefined)
213
+ }
214
+ }
215
+
216
+ try {
217
+ topUp()
218
+ while (queue.length > 0) {
219
+ throwIfAborted(signal)
220
+ const read = await queue.shift()!
221
+ filePosition += read.length
222
+ // A short read is the end of the file — the filehandles here return
223
+ // exactly what was asked for until then, and an empty buffer past it.
224
+ // The reads already queued behind it are past EOF; drop them rather than
225
+ // decompressing empty buffers.
226
+ if (read.length < readLen) {
227
+ exhausted = true
228
+ abandon()
229
+ } else {
230
+ allowed = depth
231
+ }
232
+ topUp()
233
+ const compressed = blockCarry
234
+ ? concatUint8Array([blockCarry, read])
235
+ : read
236
+ if (compressed.length === 0) {
237
+ break
238
+ }
239
+
240
+ const blocks = scanBgzfBlocks(compressed, 0, Number.POSITIVE_INFINITY)
241
+ const last = blocks.at(-1)
242
+ if (!last) {
243
+ throw new Error(
244
+ `not a BGZF stream: no valid block at byte ${filePosition - compressed.length}`,
245
+ )
246
+ }
247
+ const blocksEnd = last.inputOffset + last.compressedSize
248
+ // copied, not subarray'd: the remainder is carried across an await, and a
249
+ // view would pin the whole window it came from
250
+ blockCarry =
251
+ blocksEnd < compressed.length ? compressed.slice(blocksEnd) : undefined
252
+
253
+ const decompressed = await inflate(compressed, blocks, blocksEnd, pool)
254
+ const bytes = recordCarry
255
+ ? concatUint8Array([recordCarry, decompressed])
256
+ : decompressed
257
+ const dataView = new DataView(
258
+ bytes.buffer,
259
+ bytes.byteOffset,
260
+ bytes.byteLength,
261
+ )
262
+
263
+ let blockStart = 0
264
+ if (!sawHeader) {
265
+ if (bytes.length < 8 || dataView.getInt32(0, true) !== BAM_MAGIC) {
266
+ throw new Error('not a BAM file: bad magic')
267
+ }
268
+ const lText = dataView.getInt32(4, true)
269
+ const refs = parseRefSeqs(bytes, 8 + lText, renameRefSeqs)
270
+ if (!refs) {
271
+ // a header with enough contigs to span a window: keep the whole thing
272
+ // and try again with more of it
273
+ recordCarry = bytes
274
+ continue
275
+ }
276
+ const headerText = new TextDecoder('utf8').decode(
277
+ bytes.subarray(8, 8 + lText),
278
+ )
279
+ onHeader?.({
280
+ headerText,
281
+ samHeader: parseHeaderText(headerText),
282
+ chrToIndex: refs.chrToIndex,
283
+ indexToChr: refs.indexToChr,
284
+ })
285
+ sawHeader = true
286
+ blockStart = refs.end
287
+ }
288
+
289
+ const sink: T[] = []
290
+ while (blockStart + 4 <= bytes.length) {
291
+ const blockSize = dataView.getInt32(blockStart, true)
292
+ const blockEnd = blockStart + 4 + blockSize - 1
293
+ if (blockEnd >= bytes.length) {
294
+ break
295
+ }
296
+ sink.push(
297
+ new RecordClass(
298
+ bytes,
299
+ blockStart,
300
+ blockEnd,
301
+ // An indexed read derives fileOffset from the record's BGZF virtual
302
+ // offset, which nothing here has: there is no index to seek back
303
+ // with. Its ordinal in the file is the other thing that identifies a
304
+ // record by position, and it is the same for a given file whatever
305
+ // the window size, so it is as stable as a virtual offset without
306
+ // pretending to be one.
307
+ //
308
+ // NOT the crc32 of the record bytes that readBamFeatures falls back
309
+ // to when it has no positions. That is a content hash, so the two
310
+ // byte-identical records in exact_duplicate.bam collide on it, and
311
+ // hashing every record costs ~40% of the walk (135ms of 340ms over
312
+ // out.bam) where a counter costs nothing. The fallback only fires on
313
+ // an unusual path there; here it would fire on every record.
314
+ recordIndex++,
315
+ dataView,
316
+ ),
317
+ )
318
+ blockStart = blockEnd + 1
319
+ }
320
+ recordCarry =
321
+ blockStart < bytes.length ? bytes.slice(blockStart) : undefined
322
+
323
+ if (sink.length > 0) {
324
+ yield sink
325
+ }
326
+ }
327
+ } finally {
328
+ abandon()
329
+ }
330
+
331
+ if (!sawHeader) {
332
+ throw new Error('Insufficient data for reference sequences')
333
+ }
334
+ if (blockCarry !== undefined) {
335
+ throw new Error(
336
+ `truncated BAM: ${blockCarry.length} trailing bytes are not a complete BGZF block`,
337
+ )
338
+ }
339
+ if (recordCarry !== undefined) {
340
+ throw new Error(
341
+ `truncated BAM: ${recordCarry.length} bytes left after the last complete record`,
342
+ )
343
+ }
344
+ }
package/src/util.ts CHANGED
@@ -1,7 +1,24 @@
1
+ import { LocalFile, RemoteFile } from 'generic-filehandle2'
2
+
1
3
  import Chunk from './chunk.ts'
2
4
  import { VirtualOffset } from './virtualOffset.ts'
3
5
 
4
6
  import type { OffsetCoords } from './virtualOffset.ts'
7
+ import type { GenericFilehandle } from 'generic-filehandle2'
8
+
9
+ /** 'BAM\1' read as a little-endian int32 */
10
+ export const BAM_MAGIC = 21840194
11
+
12
+ export function resolveFilehandle(
13
+ filehandle?: GenericFilehandle,
14
+ path?: string,
15
+ url?: string,
16
+ ) {
17
+ return (
18
+ filehandle ??
19
+ (path ? new LocalFile(path) : url ? new RemoteFile(url) : undefined)
20
+ )
21
+ }
5
22
 
6
23
  export interface BamOpts {
7
24
  viewAsPairs?: boolean