@gmod/bam 7.8.1 → 7.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/bai.js +42 -17
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +4 -1
- package/dist/bamFile.js +40 -15
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.js +20 -0
- package/dist/csi.js.map +1 -1
- package/dist/index.d.ts +3 -1
- package/dist/indexFile.js +5 -2
- package/dist/indexFile.js.map +1 -1
- package/dist/record.d.ts +16 -5
- package/dist/record.js +165 -16
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +10 -0
- package/dist/util.js +39 -14
- package/dist/util.js.map +1 -1
- package/esm/bai.js +42 -17
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +4 -1
- package/esm/bamFile.js +40 -15
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.js +20 -0
- package/esm/csi.js.map +1 -1
- package/esm/index.d.ts +3 -1
- package/esm/indexFile.js +5 -2
- package/esm/indexFile.js.map +1 -1
- package/esm/record.d.ts +16 -5
- package/esm/record.js +165 -16
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +10 -0
- package/esm/util.js +39 -14
- package/esm/util.js.map +1 -1
- package/package.json +4 -1
- package/src/bai.ts +43 -17
- package/src/bamFile.ts +47 -21
- package/src/csi.ts +20 -0
- package/src/index.ts +6 -1
- package/src/indexFile.ts +5 -2
- package/src/record.ts +166 -23
- package/src/util.ts +40 -14
package/src/bai.ts
CHANGED
|
@@ -33,12 +33,20 @@ const BAI_MAGIC = 21578050 // BAI\1
|
|
|
33
33
|
// https://github.com/samtools/hts-specs/blob/master/SAMv1.pdf
|
|
34
34
|
const BAI_LINEAR_SHIFT = 14
|
|
35
35
|
const BAI_LINEAR_INTERVAL = 1 << BAI_LINEAR_SHIFT // 16384
|
|
36
|
+
const BAI_DEPTH = 5
|
|
37
|
+
// Highest coordinate the scheme addresses: the deepest level's bins are
|
|
38
|
+
// BAI_LINEAR_INTERVAL wide and there are 8^BAI_DEPTH of them.
|
|
39
|
+
const BAI_MAX_POS = 2 ** (BAI_LINEAR_SHIFT + BAI_DEPTH * 3) // 2^29
|
|
36
40
|
|
|
37
41
|
function roundDown(n: number, multiple: number) {
|
|
38
42
|
return n - (n % multiple)
|
|
39
43
|
}
|
|
44
|
+
// Note the `rem === 0` case: without it a coordinate already on a window
|
|
45
|
+
// boundary rounds up a whole extra window, which is enough to push indexCov's
|
|
46
|
+
// range past the end of the linear index.
|
|
40
47
|
function roundUp(n: number, multiple: number) {
|
|
41
|
-
|
|
48
|
+
const rem = n % multiple
|
|
49
|
+
return rem === 0 ? n : n - rem + multiple
|
|
42
50
|
}
|
|
43
51
|
|
|
44
52
|
export interface IndexCovEntry {
|
|
@@ -50,6 +58,19 @@ export interface IndexCovEntry {
|
|
|
50
58
|
// Compute bin ranges that overlap [beg, end). Each level's first-bin offset
|
|
51
59
|
// is (8^L - 1) / 7. See SAMv1.pdf §5.1.1 for the binning derivation.
|
|
52
60
|
function reg2bins(beg: number, end: number) {
|
|
61
|
+
// Clamp to what the scheme can address, the way CSI's reg2bins clamps to
|
|
62
|
+
// its own. The shifts below are the `>>` operator, so a coordinate past
|
|
63
|
+
// 2^31 wraps to a negative bin number and every level yields an empty
|
|
64
|
+
// range: `getRecordsForRange(chr, 0, 2**32)` — a caller asking for a whole
|
|
65
|
+
// reference without knowing its length — came back with NO records at all
|
|
66
|
+
// rather than all of them. Clamping is also what keeps a merely-large end
|
|
67
|
+
// from walking ~130k absent bin numbers before finding the same chunks.
|
|
68
|
+
if (beg > BAI_MAX_POS) {
|
|
69
|
+
beg = BAI_MAX_POS
|
|
70
|
+
}
|
|
71
|
+
if (end > BAI_MAX_POS) {
|
|
72
|
+
end = BAI_MAX_POS
|
|
73
|
+
}
|
|
53
74
|
end -= 1
|
|
54
75
|
return [
|
|
55
76
|
[0, 0],
|
|
@@ -76,8 +97,7 @@ export default class BAI extends IndexFile<BaiParsed> {
|
|
|
76
97
|
}
|
|
77
98
|
|
|
78
99
|
const refCount = dataView.getInt32(4, true)
|
|
79
|
-
const
|
|
80
|
-
const binLimit = ((1 << ((depth + 1) * 3)) - 1) / 7
|
|
100
|
+
const binLimit = ((1 << ((BAI_DEPTH + 1) * 3)) - 1) / 7
|
|
81
101
|
|
|
82
102
|
// read the indexes for each reference sequence
|
|
83
103
|
let curr = 8
|
|
@@ -100,11 +120,9 @@ export default class BAI extends IndexFile<BaiParsed> {
|
|
|
100
120
|
throw new Error('bai index contains too many bins, please use CSI')
|
|
101
121
|
} else {
|
|
102
122
|
const chunkCount = dataView.getInt32(curr, true)
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
curr += 8
|
|
107
|
-
}
|
|
123
|
+
// 16 bytes per chunk (two virtual offsets); the first pass only
|
|
124
|
+
// needs to step over them. Same shape as csi.ts's first pass.
|
|
125
|
+
curr += 4 + 16 * chunkCount
|
|
108
126
|
}
|
|
109
127
|
}
|
|
110
128
|
|
|
@@ -199,7 +217,6 @@ export default class BAI extends IndexFile<BaiParsed> {
|
|
|
199
217
|
opts?: BaseOpts,
|
|
200
218
|
): Promise<IndexCovEntry[]> {
|
|
201
219
|
const v = BAI_LINEAR_INTERVAL
|
|
202
|
-
const range = start !== undefined
|
|
203
220
|
const indexData = await this.parse(opts)
|
|
204
221
|
const seqIdx = indexData.indices(seqId)
|
|
205
222
|
|
|
@@ -211,15 +228,24 @@ export default class BAI extends IndexFile<BaiParsed> {
|
|
|
211
228
|
if (nintv === 0) {
|
|
212
229
|
return []
|
|
213
230
|
}
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
231
|
+
// The linear index describes [0, indexEnd): each window's score is the gap
|
|
232
|
+
// to the NEXT entry, so the final entry is a boundary rather than a window
|
|
233
|
+
// of its own. Both ends are clamped to it instead of throwing, so a range
|
|
234
|
+
// query returns the part of the reference the index covers — asking for the
|
|
235
|
+
// whole reference by its length (`indexCov(ref, 0, ctgLength)`, the obvious
|
|
236
|
+
// call, and one where end lands in the last window) used to throw "query
|
|
237
|
+
// outside of range of linear index" while `indexCov(ref)` answered fine.
|
|
238
|
+
const indexEnd = (nintv - 1) * v
|
|
239
|
+
const s =
|
|
240
|
+
start === undefined
|
|
241
|
+
? 0
|
|
242
|
+
: Math.min(Math.max(roundDown(start, v), 0), indexEnd)
|
|
243
|
+
const e = end === undefined ? indexEnd : Math.min(roundUp(end, v), indexEnd)
|
|
244
|
+
if (e <= s) {
|
|
245
|
+
return []
|
|
222
246
|
}
|
|
247
|
+
const depths: IndexCovEntry[] = new Array((e - s) / v)
|
|
248
|
+
const totalSize = linearBlockPositions[nintv - 1]!
|
|
223
249
|
// Scale the block-delta into a read count as we go, rather than building the
|
|
224
250
|
// entries and then rebuilding every one of them to apply the scale. Keep the
|
|
225
251
|
// multiply-then-divide order: hoisting lineCount/totalSize into a factor
|
package/src/bamFile.ts
CHANGED
|
@@ -37,10 +37,13 @@ export const BAM_MAGIC = 21840194
|
|
|
37
37
|
|
|
38
38
|
const blockLen = 1 << 16
|
|
39
39
|
|
|
40
|
-
// Ceiling on the header read. A million contigs is roughly 10MB of
|
|
41
|
-
// @SQ lines and ref-seq table, so
|
|
42
|
-
// claiming a huge n_ref rather than a real one, and
|
|
43
|
-
// just downloads the file.
|
|
40
|
+
// Ceiling on GROWING the header read. A million contigs is roughly 10MB of
|
|
41
|
+
// compressed @SQ lines and ref-seq table, so a read that has doubled past this
|
|
42
|
+
// is chasing a corrupt header claiming a huge n_ref rather than a real one, and
|
|
43
|
+
// growing further just downloads the file. It does not cap the FIRST read: that
|
|
44
|
+
// length comes from the index's firstDataLine, which is a real offset rather
|
|
45
|
+
// than a guess, so a header that genuinely runs past this still gets its one
|
|
46
|
+
// exact read. See getHeaderPre.
|
|
44
47
|
const maxHeaderReadLen = 32 * 1024 * 1024
|
|
45
48
|
|
|
46
49
|
function resolveFilehandle(
|
|
@@ -118,12 +121,24 @@ export const DEFAULT_MAX_CACHE_BYTES = 100 * 1024 * 1024
|
|
|
118
121
|
const MAX_CONCURRENT_CHUNK_READS = 6
|
|
119
122
|
|
|
120
123
|
class ChunkFeatureCache<T> {
|
|
121
|
-
|
|
124
|
+
private _maxBytes: number
|
|
122
125
|
private entries = new Map<string, ChunkEntry<T>>()
|
|
123
126
|
private bytes = 0
|
|
124
127
|
|
|
125
128
|
constructor(maxBytes: number) {
|
|
126
|
-
this.
|
|
129
|
+
this._maxBytes = maxBytes
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
get maxBytes() {
|
|
133
|
+
return this._maxBytes
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// Accessor rather than a plain field so lowering the budget frees memory now.
|
|
137
|
+
// As a field, a caller trimming the cache under memory pressure got nothing
|
|
138
|
+
// back until the next chunk read happened to call set().
|
|
139
|
+
set maxBytes(maxBytes: number) {
|
|
140
|
+
this._maxBytes = maxBytes
|
|
141
|
+
this.evict()
|
|
127
142
|
}
|
|
128
143
|
|
|
129
144
|
get size() {
|
|
@@ -148,11 +163,15 @@ class ChunkFeatureCache<T> {
|
|
|
148
163
|
this.delete(key)
|
|
149
164
|
this.entries.set(key, entry)
|
|
150
165
|
this.bytes += entry.bytes
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
166
|
+
this.evict()
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// Evict from the least-recently-used end. The size > 1 guard means a single
|
|
170
|
+
// chunk larger than the whole budget is still kept: the caller needs it for
|
|
171
|
+
// the query in flight, and dropping it would only force a re-decompress.
|
|
172
|
+
private evict() {
|
|
154
173
|
const lru = this.entries.keys()
|
|
155
|
-
while (this.bytes > this.
|
|
174
|
+
while (this.bytes > this._maxBytes && this.entries.size > 1) {
|
|
156
175
|
this.delete(lru.next().value!)
|
|
157
176
|
}
|
|
158
177
|
}
|
|
@@ -285,16 +304,21 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
285
304
|
? blockLen
|
|
286
305
|
: indexData.firstDataLine.blockPosition + blockLen
|
|
287
306
|
|
|
307
|
+
// do/while, so the index-derived length is always read once and only the
|
|
308
|
+
// doubling is bounded. Testing readLen before the first read instead meant
|
|
309
|
+
// a BAM whose header genuinely exceeds maxHeaderReadLen — millions of
|
|
310
|
+
// contigs — was rejected without a single byte being fetched, and reported
|
|
311
|
+
// as 'Insufficient data for reference sequences' when the data was there.
|
|
288
312
|
let samHeader
|
|
289
|
-
let atEof
|
|
290
|
-
|
|
313
|
+
let atEof: boolean
|
|
314
|
+
do {
|
|
291
315
|
const buffer = await this.bam.read(readLen, 0, { signal: opts.signal })
|
|
292
316
|
// a short read means readLen ran past the end of the file, so there are
|
|
293
317
|
// no more bytes to grow into
|
|
294
318
|
atEof = buffer.length < readLen
|
|
295
319
|
samHeader = this.applyHeader(await unzip(buffer))
|
|
296
320
|
readLen *= 2
|
|
297
|
-
}
|
|
321
|
+
} while (samHeader === undefined && !atEof && readLen <= maxHeaderReadLen)
|
|
298
322
|
if (samHeader === undefined) {
|
|
299
323
|
throw new Error('Insufficient data for reference sequences')
|
|
300
324
|
}
|
|
@@ -499,9 +523,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
499
523
|
// unbarriered pool exactly as before. That caps the cost of being wrong
|
|
500
524
|
// at 0.92x-0.95x, against 0.82x-0.88x for barriering every wave.
|
|
501
525
|
const batch = Math.min(MAX_CONCURRENT_CHUNK_READS, chunks.length)
|
|
502
|
-
await Promise.all(
|
|
503
|
-
Array.from({ length: batch }, (_, ci) => readOne(ci)),
|
|
504
|
-
)
|
|
526
|
+
await Promise.all(Array.from({ length: batch }, (_, ci) => readOne(ci)))
|
|
505
527
|
let stopped = false
|
|
506
528
|
for (let ci = 0; ci < batch; ci++) {
|
|
507
529
|
if (isPastQuery(featureLists[ci], chrId, max)) {
|
|
@@ -615,9 +637,13 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
615
637
|
const mateRecs = [] as T[]
|
|
616
638
|
for (let i = 0, l = features.length; i < l; i++) {
|
|
617
639
|
const feature = features[i]!
|
|
640
|
+
// fileOffset first: it is a number already on the record, where
|
|
641
|
+
// `name` decodes a string per record. A mate chunk usually overlaps
|
|
642
|
+
// the query region, so this skips the decode for every record the
|
|
643
|
+
// caller is already holding.
|
|
618
644
|
if (
|
|
619
|
-
|
|
620
|
-
|
|
645
|
+
!readIds.has(feature.fileOffset) &&
|
|
646
|
+
readNameCounts.get(feature.name) === 1
|
|
621
647
|
) {
|
|
622
648
|
mateRecs.push(feature)
|
|
623
649
|
}
|
|
@@ -691,9 +717,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
691
717
|
blockEnd,
|
|
692
718
|
hasCpositions
|
|
693
719
|
? cpositions[pos]! * (1 << 8) +
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
720
|
+
(blockStart - dpositions[pos]!) +
|
|
721
|
+
chunk.minv.dataPosition +
|
|
722
|
+
1
|
|
697
723
|
: crc32(ba.subarray(blockStart, blockEnd)) >>> 0,
|
|
698
724
|
dataView,
|
|
699
725
|
)
|
package/src/csi.ts
CHANGED
|
@@ -96,6 +96,20 @@ export default class CSI extends IndexFile {
|
|
|
96
96
|
this.maxBinNumber = ((1 << ((this.depth + 1) * 3)) - 1) / 7
|
|
97
97
|
const maxBinNumber = this.maxBinNumber
|
|
98
98
|
const auxLength = dataView.getInt32(12, true)
|
|
99
|
+
// A tabix-only branch, which is why parseAuxData and the parseNameBytes it
|
|
100
|
+
// calls show as uncovered. CSI is shared between `samtools index -c` and
|
|
101
|
+
// `tabix -C`, and the aux block is how a tabix index carries what a reader
|
|
102
|
+
// of bgzipped TEXT needs: which columns hold ref/start/end, the comment
|
|
103
|
+
// character, lines to skip, and the reference names — which a BAM takes
|
|
104
|
+
// from its own header instead. So a BAM .csi sets l_aux to 0 and never
|
|
105
|
+
// reaches here (checked: 0 of the 19 .csi fixtures carry one).
|
|
106
|
+
//
|
|
107
|
+
// Kept rather than deleted for two reasons. CSI is exported from
|
|
108
|
+
// index.ts, so pointing it at a tabix index is reachable (@gmod/tabix is
|
|
109
|
+
// the right tool, but this would half-work and then silently not).
|
|
110
|
+
// More importantly parseNameBytes is marked SYNC: with its tabix-js
|
|
111
|
+
// counterpart, and dropping one side of a deliberately-paired pair of
|
|
112
|
+
// implementations is exactly what that marker exists to prevent.
|
|
99
113
|
const aux = auxLength >= 30 ? this.parseAuxData(bytes, 16) : undefined
|
|
100
114
|
const refCount = dataView.getInt32(16 + auxLength, true)
|
|
101
115
|
|
|
@@ -200,6 +214,12 @@ export default class CSI extends IndexFile {
|
|
|
200
214
|
for (; l <= this.depth; s -= 3, t += lshift(1, l * 3), l += 1) {
|
|
201
215
|
const b = t + rshift(beg, s)
|
|
202
216
|
const e = t + rshift(end, s)
|
|
217
|
+
// Unreachable as long as the clamp above stands, which is why it shows
|
|
218
|
+
// as uncovered: with end bounded to 2^(minShift + depth*3), level l
|
|
219
|
+
// spans at most 8^l - 1 bins, so the worst case is 8^depth - 1 + depth
|
|
220
|
+
// against a maxBinNumber of (8^(depth+1) - 1)/7, i.e. about 1.14 * 8^depth.
|
|
221
|
+
// Kept as a guard on the arithmetic rather than deleted — reg2bins is
|
|
222
|
+
// shared with tabix-js, whose callers reach it by other routes.
|
|
203
223
|
if (e - b + bins.length > this.maxBinNumber) {
|
|
204
224
|
throw new Error(
|
|
205
225
|
`query ${beg}-${end} is too large for current binning scheme (shift ${this.minShift}, depth ${this.depth}), try a smaller query or a coarser index binning scheme`,
|
package/src/index.ts
CHANGED
|
@@ -4,7 +4,12 @@ export { default as CSI } from './csi.ts'
|
|
|
4
4
|
export { default as BamRecord } from './record.ts'
|
|
5
5
|
export { default as HtsgetFile } from './htsget.ts'
|
|
6
6
|
|
|
7
|
-
export type {
|
|
7
|
+
export type { NumericCigar } from './record.ts'
|
|
8
8
|
export type { BamRecordClass, BamRecordLike } from './bamFile.ts'
|
|
9
|
+
// the options every query method takes, and the shapes they hand back. The
|
|
10
|
+
// package has no subpath exports, so a consumer typing a wrapper around
|
|
11
|
+
// getRecordsForRange/indexCov can only name these if they come out of here.
|
|
12
|
+
export type { BamOpts, BaseOpts } from './util.ts'
|
|
13
|
+
export type { IndexCovEntry } from './bai.ts'
|
|
9
14
|
// for typing the HtsgetFile `fetch` option
|
|
10
15
|
export type { Fetcher } from 'generic-filehandle2'
|
package/src/indexFile.ts
CHANGED
|
@@ -33,8 +33,11 @@ export function memoizeByRefId<T>(
|
|
|
33
33
|
) {
|
|
34
34
|
const cache = new QuickLRU<number, T>({ maxSize })
|
|
35
35
|
return (refId: number) => {
|
|
36
|
-
|
|
37
|
-
|
|
36
|
+
// one lookup, not has()+get(): only truthy results are ever cached, so a
|
|
37
|
+
// miss and a cached value are already distinguishable
|
|
38
|
+
const cached = cache.get(refId)
|
|
39
|
+
if (cached !== undefined) {
|
|
40
|
+
return cached
|
|
38
41
|
}
|
|
39
42
|
const result = getIndices(refId)
|
|
40
43
|
if (result) {
|
package/src/record.ts
CHANGED
|
@@ -31,6 +31,41 @@ for (let hi = 0; hi < 16; hi++) {
|
|
|
31
31
|
// read lengths.
|
|
32
32
|
const SEQ_DECODER_THRESHOLD = 300
|
|
33
33
|
|
|
34
|
+
// Four bases per entry, indexed by a PAIR of SEQ bytes, so the sub-threshold
|
|
35
|
+
// path halves its concat count: 1.5-1.6x on a short-read query end to end
|
|
36
|
+
// (shortreads_300x 46.5 -> 30.5 ms over 53.6k reads, volvox 4.8 -> 2.9 ms).
|
|
37
|
+
//
|
|
38
|
+
// Built lazily, and only once the short path has been taken enough times to pay
|
|
39
|
+
// for it. Filling 65536 entries costs ~6 ms and retains ~2 MB, so building on
|
|
40
|
+
// first use is a LOSS on a file that decodes only a handful of short reads:
|
|
41
|
+
// long-read fixtures with a short-read tail measured 1.4x slower that way
|
|
42
|
+
// (ecoli_nanopore has 27 sub-300bp reads out of 480, chm1 has 5 of 204). The
|
|
43
|
+
// counter is module-global on purpose — the table is shared, so what has to
|
|
44
|
+
// amortize is the total number of short decodes in the process, not per file.
|
|
45
|
+
const SEQRET_QUAD_WARMUP = 1024
|
|
46
|
+
let seqretShortCalls = 0
|
|
47
|
+
let SEQRET_QUAD_STRINGS: string[] | undefined
|
|
48
|
+
|
|
49
|
+
// The 4-base table, or undefined while still warming up (caller falls back to
|
|
50
|
+
// the 2-base table).
|
|
51
|
+
function seqretQuads() {
|
|
52
|
+
if (SEQRET_QUAD_STRINGS === undefined) {
|
|
53
|
+
if (++seqretShortCalls < SEQRET_QUAD_WARMUP) {
|
|
54
|
+
return undefined
|
|
55
|
+
}
|
|
56
|
+
const quads = new Array<string>(65536)
|
|
57
|
+
for (let a = 0; a < 256; a++) {
|
|
58
|
+
const sa = SEQRET_PAIR_STRINGS[a]!
|
|
59
|
+
const base = a << 8
|
|
60
|
+
for (let b = 0; b < 256; b++) {
|
|
61
|
+
quads[base | b] = sa + SEQRET_PAIR_STRINGS[b]!
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
SEQRET_QUAD_STRINGS = quads
|
|
65
|
+
}
|
|
66
|
+
return SEQRET_QUAD_STRINGS
|
|
67
|
+
}
|
|
68
|
+
|
|
34
69
|
// Precomputed pair orientation strings, indexed by
|
|
35
70
|
// ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
|
|
36
71
|
// bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
|
|
@@ -87,9 +122,20 @@ function decodeTagString(ba: Uint8Array, start: number, end: number) {
|
|
|
87
122
|
return textDecoder.decode(ba.subarray(start, end))
|
|
88
123
|
}
|
|
89
124
|
|
|
90
|
-
// Bitmask for ops that consume ref: M=0, D=2, N=3,
|
|
91
|
-
// Binary:
|
|
92
|
-
|
|
125
|
+
// Bitmask for ops that consume ref: M=0, D=2, N=3, ==7, X=8
|
|
126
|
+
// Binary: 0b110001101 = 0x18D
|
|
127
|
+
//
|
|
128
|
+
// P (=6) is NOT among them. Padding is a silent placeholder against a padded
|
|
129
|
+
// reference and advances neither query nor reference, which is what htslib's
|
|
130
|
+
// bam_cigar_type(P) == 0 says. Including it made every read carrying a P op
|
|
131
|
+
// report an end one base per padded base too far.
|
|
132
|
+
//
|
|
133
|
+
// That said: nobody cares about P. Padded alignments come out of a handful of
|
|
134
|
+
// assembly tools and are effectively absent from real BAMs — this was fixed
|
|
135
|
+
// because it disagreed with htslib, not because it was hurting anyone. It is
|
|
136
|
+
// covered incidentally by samspec.bam in the samtools agreement suite, and
|
|
137
|
+
// that is enough; don't spend a dedicated fixture or test on it.
|
|
138
|
+
const CIGAR_CONSUMES_REF_MASK = 0x18d
|
|
93
139
|
|
|
94
140
|
// A CIGAR as packed op words. Either a view over (or copy of) the record's own
|
|
95
141
|
// CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
|
|
@@ -104,12 +150,6 @@ function isNumericCigar(value: unknown): value is NumericCigar {
|
|
|
104
150
|
)
|
|
105
151
|
}
|
|
106
152
|
|
|
107
|
-
export interface Bytes {
|
|
108
|
-
start: number
|
|
109
|
-
end: number
|
|
110
|
-
byteArray: Uint8Array
|
|
111
|
-
}
|
|
112
|
-
|
|
113
153
|
type BArrayValue =
|
|
114
154
|
| Int8Array
|
|
115
155
|
| Uint8Array
|
|
@@ -202,13 +242,20 @@ function decodeBArrayTag(
|
|
|
202
242
|
}
|
|
203
243
|
|
|
204
244
|
// Byte span of a 'B' tag's element payload, for advancing the cursor past it.
|
|
245
|
+
// Returns -1 for a subtype outside the spec's cCsSiIf, whose element width is
|
|
246
|
+
// unknowable. Guessing one byte per element (which is what falling through to
|
|
247
|
+
// `limit` did) doesn't skip the tag, it lands the cursor mid-value and decodes
|
|
248
|
+
// the whole rest of the record's tag list out of garbage — where an unknown
|
|
249
|
+
// top-level type has always stopped the walk instead. Same choice here.
|
|
205
250
|
function bArrayByteLength(Btype: number, limit: number) {
|
|
206
251
|
if (Btype === 0x69 || Btype === 0x49 || Btype === 0x66) {
|
|
207
252
|
return limit << 2
|
|
208
253
|
} else if (Btype === 0x73 || Btype === 0x53) {
|
|
209
254
|
return limit << 1
|
|
210
|
-
} else {
|
|
255
|
+
} else if (Btype === 0x63 || Btype === 0x43) {
|
|
211
256
|
return limit
|
|
257
|
+
} else {
|
|
258
|
+
return -1
|
|
212
259
|
}
|
|
213
260
|
}
|
|
214
261
|
|
|
@@ -242,6 +289,11 @@ function tagValueEnd(
|
|
|
242
289
|
case 0x5a: // 'Z'
|
|
243
290
|
case 0x48: {
|
|
244
291
|
// 'H'
|
|
292
|
+
// Stays a plain byte loop. Swapping in Uint8Array.indexOf past a short
|
|
293
|
+
// inline probe is 2.7x faster on long-read MD (mean 9083 bytes on
|
|
294
|
+
// jb2bench's 200x.longread) but ~1.13x slower on short-read Z values,
|
|
295
|
+
// which are 4-13 bytes and are what the dominant case is made of. Measured
|
|
296
|
+
// both ways against the realistic corpus — see ADR 0012.
|
|
245
297
|
let q = p
|
|
246
298
|
while (q < blockEnd && ba[q] !== 0) {
|
|
247
299
|
q++
|
|
@@ -252,7 +304,12 @@ function tagValueEnd(
|
|
|
252
304
|
// 'B'
|
|
253
305
|
const Btype = ba[p]!
|
|
254
306
|
const limit = dataView.getInt32(p + 1, true)
|
|
255
|
-
|
|
307
|
+
const payload = bArrayByteLength(Btype, limit)
|
|
308
|
+
if (payload < 0) {
|
|
309
|
+
console.error('Unknown BAM B tag subtype', Btype)
|
|
310
|
+
return 0
|
|
311
|
+
}
|
|
312
|
+
return p + 5 + payload
|
|
256
313
|
}
|
|
257
314
|
default:
|
|
258
315
|
console.error('Unknown BAM tag type', type)
|
|
@@ -451,6 +508,60 @@ export default class BamRecord {
|
|
|
451
508
|
return this._findTag(tagName, true)
|
|
452
509
|
}
|
|
453
510
|
|
|
511
|
+
/**
|
|
512
|
+
* The value of `tagName`, or of `altName` if the record carries no `tagName` —
|
|
513
|
+
* resolved in ONE pass over the tag block instead of two.
|
|
514
|
+
*
|
|
515
|
+
* For the MM/Mm and ML/Ml alias pairs that modified-base callers emit, where
|
|
516
|
+
* `getTag(a) ?? getTag(b)` walks every tag on the record TWICE whenever
|
|
517
|
+
* neither is present, which is every read in a file without base
|
|
518
|
+
* modifications. jbrowse-components issues exactly that lookup per record on
|
|
519
|
+
* every render (extractModifications runs unconditionally), and on
|
|
520
|
+
* jb2bench's 1000x.shortread it was 12.9% of the whole query — more than the
|
|
521
|
+
* CIGAR, SEQ and MD reads the pileup actually uses, spent proving absence.
|
|
522
|
+
*
|
|
523
|
+
* `tagName` wins wherever it appears, so the result matches the two-lookup
|
|
524
|
+
* form even for a (malformed) record carrying both.
|
|
525
|
+
*/
|
|
526
|
+
getTagAlt(tagName: string, altName: string) {
|
|
527
|
+
if (this._cachedTags !== undefined) {
|
|
528
|
+
return this._cachedTags[tagName] ?? this._cachedTags[altName]
|
|
529
|
+
}
|
|
530
|
+
const a0 = tagName.charCodeAt(0)
|
|
531
|
+
const a1 = tagName.charCodeAt(1)
|
|
532
|
+
const b0 = altName.charCodeAt(0)
|
|
533
|
+
const b1 = altName.charCodeAt(1)
|
|
534
|
+
const blockEnd = this._end
|
|
535
|
+
const ba = this._byteArray
|
|
536
|
+
let p = this.tagsStart
|
|
537
|
+
// where the alternate landed, if it turns up before the primary does
|
|
538
|
+
let altType = -1
|
|
539
|
+
let altStart = 0
|
|
540
|
+
let altEnd = 0
|
|
541
|
+
while (p < blockEnd) {
|
|
542
|
+
const c0 = ba[p]
|
|
543
|
+
const c1 = ba[p + 1]
|
|
544
|
+
const type = ba[p + 2]!
|
|
545
|
+
const valueStart = p + 3
|
|
546
|
+
const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
|
|
547
|
+
if (end === 0) {
|
|
548
|
+
break // unknown type: can't compute how far to advance
|
|
549
|
+
}
|
|
550
|
+
if (c0 === a0 && c1 === a1) {
|
|
551
|
+
return decodeTagValue(ba, this._dataView, type, valueStart, end, false)
|
|
552
|
+
}
|
|
553
|
+
if (altType < 0 && c0 === b0 && c1 === b1) {
|
|
554
|
+
altType = type
|
|
555
|
+
altStart = valueStart
|
|
556
|
+
altEnd = end
|
|
557
|
+
}
|
|
558
|
+
p = end
|
|
559
|
+
}
|
|
560
|
+
return altType < 0
|
|
561
|
+
? undefined
|
|
562
|
+
: decodeTagValue(ba, this._dataView, altType, altStart, altEnd, false)
|
|
563
|
+
}
|
|
564
|
+
|
|
454
565
|
private _findTag(tagName: string, raw: boolean) {
|
|
455
566
|
const tag1 = tagName.charCodeAt(0)
|
|
456
567
|
const tag2 = tagName.charCodeAt(1)
|
|
@@ -622,12 +733,18 @@ export default class BamRecord {
|
|
|
622
733
|
return lref
|
|
623
734
|
}
|
|
624
735
|
|
|
736
|
+
// Unmapped records are not special-cased here, unlike in
|
|
737
|
+
// _computeLengthOnRef. The spec says no assumptions can be made about an
|
|
738
|
+
// unmapped read's CIGAR, but "no assumptions" is not "no CIGAR": aligners
|
|
739
|
+
// emit placed unmapped mates that carry a real one, and htslib prints
|
|
740
|
+
// whatever is stored. Dropping it lost data the file had — paired.bam's
|
|
741
|
+
// SRR062635.1831187 at 20:74230 is FLAG 133 with 35M65S.
|
|
742
|
+
//
|
|
743
|
+
// Reference span stays 0 for them regardless, since that is a claim about
|
|
744
|
+
// alignment rather than about the stored bytes, and the query filter reads
|
|
745
|
+
// it.
|
|
625
746
|
private _computeNumericCigar(): NumericCigar {
|
|
626
747
|
const flag_nc = this._dataView.getInt32(this._start + 16, true)
|
|
627
|
-
if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
|
|
628
|
-
return new Uint32Array(0)
|
|
629
|
-
}
|
|
630
|
-
|
|
631
748
|
const numCigarOps = flag_nc & 0xffff
|
|
632
749
|
const p = this.b0 + this.read_name_length
|
|
633
750
|
|
|
@@ -664,14 +781,26 @@ export default class BamRecord {
|
|
|
664
781
|
return this._cachedNumericCigar
|
|
665
782
|
}
|
|
666
783
|
|
|
784
|
+
// Two appends per op, NOT `result += length + String.fromCharCode(op)`. The
|
|
785
|
+
// one-append form builds an intermediate cons string per op just to append it
|
|
786
|
+
// and drop it; appending each piece straight onto the rope is 1.23-1.35x
|
|
787
|
+
// faster on long reads, which is where this accessor dominates
|
|
788
|
+
// (chr22_nanopore_subset 51.2 -> 38.6 ms for 757 reads averaging 2171 ops,
|
|
789
|
+
// ultra-long-ont 13.2 -> 9.9 ms). Short reads carry 1-3 ops and land inside
|
|
790
|
+
// this box's noise band either way.
|
|
791
|
+
//
|
|
792
|
+
// ADR 0003 rejected two other rewrites of this loop — a precomputed 16-entry
|
|
793
|
+
// op-char table, and digits into a Uint8Array with one TextDecoder.decode —
|
|
794
|
+
// and both are still losers. Re-measured here: the op-char table is 1.13x
|
|
795
|
+
// SLOWER than String.fromCharCode even on top of this change, because V8
|
|
796
|
+
// already hands back an interned single-character string.
|
|
667
797
|
get CIGAR() {
|
|
668
798
|
const numeric = this.NUMERIC_CIGAR
|
|
669
799
|
let result = ''
|
|
670
800
|
for (let i = 0, l = numeric.length; i < l; i++) {
|
|
671
801
|
const packed = numeric[i]!
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
result += length + String.fromCharCode(opCode)
|
|
802
|
+
result += packed >> 4
|
|
803
|
+
result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
|
|
675
804
|
}
|
|
676
805
|
return result
|
|
677
806
|
}
|
|
@@ -697,9 +826,10 @@ export default class BamRecord {
|
|
|
697
826
|
return this._byteArray.subarray(p, p + this.num_seq_bytes)
|
|
698
827
|
}
|
|
699
828
|
|
|
700
|
-
// Decode
|
|
701
|
-
//
|
|
702
|
-
//
|
|
829
|
+
// Decode four bases per iteration off the 65536-entry table (two off the
|
|
830
|
+
// 256-entry one until that table has warmed up). Building an array of 1-char
|
|
831
|
+
// strings and join()ing it — the obvious approach — is 3x slower at 100bp and
|
|
832
|
+
// 35x slower at 15kb.
|
|
703
833
|
get seq() {
|
|
704
834
|
const len = this.seq_length
|
|
705
835
|
const ba = this._byteArray
|
|
@@ -707,9 +837,22 @@ export default class BamRecord {
|
|
|
707
837
|
const nPairs = len >> 1
|
|
708
838
|
let seq: string
|
|
709
839
|
if (len < SEQ_DECODER_THRESHOLD) {
|
|
840
|
+
const quads = seqretQuads()
|
|
710
841
|
seq = ''
|
|
711
|
-
|
|
712
|
-
|
|
842
|
+
if (quads === undefined) {
|
|
843
|
+
for (let j = 0; j < nPairs; j++) {
|
|
844
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
|
|
845
|
+
}
|
|
846
|
+
} else {
|
|
847
|
+
const nQuads = nPairs >> 1
|
|
848
|
+
for (let j = 0; j < nQuads; j++) {
|
|
849
|
+
const q = p + 2 * j
|
|
850
|
+
seq += quads[(ba[q]! << 8) | ba[q + 1]!]!
|
|
851
|
+
}
|
|
852
|
+
// odd byte count: two more bases off the 2-base table
|
|
853
|
+
if (nPairs & 1) {
|
|
854
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + nPairs - 1]!]!
|
|
855
|
+
}
|
|
713
856
|
}
|
|
714
857
|
if (len & 1) {
|
|
715
858
|
seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
|