@gmod/bam 7.8.2 → 7.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/bai.js +42 -17
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +4 -1
- package/dist/bamFile.js +40 -15
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.js +20 -0
- package/dist/csi.js.map +1 -1
- package/dist/index.d.ts +3 -1
- package/dist/indexFile.js +5 -2
- package/dist/indexFile.js.map +1 -1
- package/dist/record.d.ts +16 -5
- package/dist/record.js +151 -13
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +10 -0
- package/dist/util.js +18 -13
- package/dist/util.js.map +1 -1
- package/esm/bai.js +42 -17
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +4 -1
- package/esm/bamFile.js +40 -15
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.js +20 -0
- package/esm/csi.js.map +1 -1
- package/esm/index.d.ts +3 -1
- package/esm/indexFile.js +5 -2
- package/esm/indexFile.js.map +1 -1
- package/esm/record.d.ts +16 -5
- package/esm/record.js +151 -13
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +10 -0
- package/esm/util.js +18 -13
- package/esm/util.js.map +1 -1
- package/package.json +4 -1
- package/src/bai.ts +43 -17
- package/src/bamFile.ts +47 -21
- package/src/csi.ts +20 -0
- package/src/index.ts +6 -1
- package/src/indexFile.ts +5 -2
- package/src/record.ts +152 -20
- package/src/util.ts +18 -13
package/src/record.ts
CHANGED
|
@@ -31,6 +31,41 @@ for (let hi = 0; hi < 16; hi++) {
|
|
|
31
31
|
// read lengths.
|
|
32
32
|
const SEQ_DECODER_THRESHOLD = 300
|
|
33
33
|
|
|
34
|
+
// Four bases per entry, indexed by a PAIR of SEQ bytes, so the sub-threshold
|
|
35
|
+
// path halves its concat count: 1.5-1.6x on a short-read query end to end
|
|
36
|
+
// (shortreads_300x 46.5 -> 30.5 ms over 53.6k reads, volvox 4.8 -> 2.9 ms).
|
|
37
|
+
//
|
|
38
|
+
// Built lazily, and only once the short path has been taken enough times to pay
|
|
39
|
+
// for it. Filling 65536 entries costs ~6 ms and retains ~2 MB, so building on
|
|
40
|
+
// first use is a LOSS on a file that decodes only a handful of short reads:
|
|
41
|
+
// long-read fixtures with a short-read tail measured 1.4x slower that way
|
|
42
|
+
// (ecoli_nanopore has 27 sub-300bp reads out of 480, chm1 has 5 of 204). The
|
|
43
|
+
// counter is module-global on purpose — the table is shared, so what has to
|
|
44
|
+
// amortize is the total number of short decodes in the process, not per file.
|
|
45
|
+
const SEQRET_QUAD_WARMUP = 1024
|
|
46
|
+
let seqretShortCalls = 0
|
|
47
|
+
let SEQRET_QUAD_STRINGS: string[] | undefined
|
|
48
|
+
|
|
49
|
+
// The 4-base table, or undefined while still warming up (caller falls back to
|
|
50
|
+
// the 2-base table).
|
|
51
|
+
function seqretQuads() {
|
|
52
|
+
if (SEQRET_QUAD_STRINGS === undefined) {
|
|
53
|
+
if (++seqretShortCalls < SEQRET_QUAD_WARMUP) {
|
|
54
|
+
return undefined
|
|
55
|
+
}
|
|
56
|
+
const quads = new Array<string>(65536)
|
|
57
|
+
for (let a = 0; a < 256; a++) {
|
|
58
|
+
const sa = SEQRET_PAIR_STRINGS[a]!
|
|
59
|
+
const base = a << 8
|
|
60
|
+
for (let b = 0; b < 256; b++) {
|
|
61
|
+
quads[base | b] = sa + SEQRET_PAIR_STRINGS[b]!
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
SEQRET_QUAD_STRINGS = quads
|
|
65
|
+
}
|
|
66
|
+
return SEQRET_QUAD_STRINGS
|
|
67
|
+
}
|
|
68
|
+
|
|
34
69
|
// Precomputed pair orientation strings, indexed by
|
|
35
70
|
// ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
|
|
36
71
|
// bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
|
|
@@ -115,12 +150,6 @@ function isNumericCigar(value: unknown): value is NumericCigar {
|
|
|
115
150
|
)
|
|
116
151
|
}
|
|
117
152
|
|
|
118
|
-
export interface Bytes {
|
|
119
|
-
start: number
|
|
120
|
-
end: number
|
|
121
|
-
byteArray: Uint8Array
|
|
122
|
-
}
|
|
123
|
-
|
|
124
153
|
type BArrayValue =
|
|
125
154
|
| Int8Array
|
|
126
155
|
| Uint8Array
|
|
@@ -213,13 +242,20 @@ function decodeBArrayTag(
|
|
|
213
242
|
}
|
|
214
243
|
|
|
215
244
|
// Byte span of a 'B' tag's element payload, for advancing the cursor past it.
|
|
245
|
+
// Returns -1 for a subtype outside the spec's cCsSiIf, whose element width is
|
|
246
|
+
// unknowable. Guessing one byte per element (which is what falling through to
|
|
247
|
+
// `limit` did) doesn't skip the tag, it lands the cursor mid-value and decodes
|
|
248
|
+
// the whole rest of the record's tag list out of garbage — where an unknown
|
|
249
|
+
// top-level type has always stopped the walk instead. Same choice here.
|
|
216
250
|
function bArrayByteLength(Btype: number, limit: number) {
|
|
217
251
|
if (Btype === 0x69 || Btype === 0x49 || Btype === 0x66) {
|
|
218
252
|
return limit << 2
|
|
219
253
|
} else if (Btype === 0x73 || Btype === 0x53) {
|
|
220
254
|
return limit << 1
|
|
221
|
-
} else {
|
|
255
|
+
} else if (Btype === 0x63 || Btype === 0x43) {
|
|
222
256
|
return limit
|
|
257
|
+
} else {
|
|
258
|
+
return -1
|
|
223
259
|
}
|
|
224
260
|
}
|
|
225
261
|
|
|
@@ -253,6 +289,11 @@ function tagValueEnd(
|
|
|
253
289
|
case 0x5a: // 'Z'
|
|
254
290
|
case 0x48: {
|
|
255
291
|
// 'H'
|
|
292
|
+
// Stays a plain byte loop. Swapping in Uint8Array.indexOf past a short
|
|
293
|
+
// inline probe is 2.7x faster on long-read MD (mean 9083 bytes on
|
|
294
|
+
// jb2bench's 200x.longread) but ~1.13x slower on short-read Z values,
|
|
295
|
+
// which are 4-13 bytes and are what the dominant case is made of. Measured
|
|
296
|
+
// both ways against the realistic corpus — see ADR 0012.
|
|
256
297
|
let q = p
|
|
257
298
|
while (q < blockEnd && ba[q] !== 0) {
|
|
258
299
|
q++
|
|
@@ -263,7 +304,12 @@ function tagValueEnd(
|
|
|
263
304
|
// 'B'
|
|
264
305
|
const Btype = ba[p]!
|
|
265
306
|
const limit = dataView.getInt32(p + 1, true)
|
|
266
|
-
|
|
307
|
+
const payload = bArrayByteLength(Btype, limit)
|
|
308
|
+
if (payload < 0) {
|
|
309
|
+
console.error('Unknown BAM B tag subtype', Btype)
|
|
310
|
+
return 0
|
|
311
|
+
}
|
|
312
|
+
return p + 5 + payload
|
|
267
313
|
}
|
|
268
314
|
default:
|
|
269
315
|
console.error('Unknown BAM tag type', type)
|
|
@@ -462,6 +508,60 @@ export default class BamRecord {
|
|
|
462
508
|
return this._findTag(tagName, true)
|
|
463
509
|
}
|
|
464
510
|
|
|
511
|
+
/**
|
|
512
|
+
* The value of `tagName`, or of `altName` if the record carries no `tagName` —
|
|
513
|
+
* resolved in ONE pass over the tag block instead of two.
|
|
514
|
+
*
|
|
515
|
+
* For the MM/Mm and ML/Ml alias pairs that modified-base callers emit, where
|
|
516
|
+
* `getTag(a) ?? getTag(b)` walks every tag on the record TWICE whenever
|
|
517
|
+
* neither is present, which is every read in a file without base
|
|
518
|
+
* modifications. jbrowse-components issues exactly that lookup per record on
|
|
519
|
+
* every render (extractModifications runs unconditionally), and on
|
|
520
|
+
* jb2bench's 1000x.shortread it was 12.9% of the whole query — more than the
|
|
521
|
+
* CIGAR, SEQ and MD reads the pileup actually uses, spent proving absence.
|
|
522
|
+
*
|
|
523
|
+
* `tagName` wins wherever it appears, so the result matches the two-lookup
|
|
524
|
+
* form even for a (malformed) record carrying both.
|
|
525
|
+
*/
|
|
526
|
+
getTagAlt(tagName: string, altName: string) {
|
|
527
|
+
if (this._cachedTags !== undefined) {
|
|
528
|
+
return this._cachedTags[tagName] ?? this._cachedTags[altName]
|
|
529
|
+
}
|
|
530
|
+
const a0 = tagName.charCodeAt(0)
|
|
531
|
+
const a1 = tagName.charCodeAt(1)
|
|
532
|
+
const b0 = altName.charCodeAt(0)
|
|
533
|
+
const b1 = altName.charCodeAt(1)
|
|
534
|
+
const blockEnd = this._end
|
|
535
|
+
const ba = this._byteArray
|
|
536
|
+
let p = this.tagsStart
|
|
537
|
+
// where the alternate landed, if it turns up before the primary does
|
|
538
|
+
let altType = -1
|
|
539
|
+
let altStart = 0
|
|
540
|
+
let altEnd = 0
|
|
541
|
+
while (p < blockEnd) {
|
|
542
|
+
const c0 = ba[p]
|
|
543
|
+
const c1 = ba[p + 1]
|
|
544
|
+
const type = ba[p + 2]!
|
|
545
|
+
const valueStart = p + 3
|
|
546
|
+
const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
|
|
547
|
+
if (end === 0) {
|
|
548
|
+
break // unknown type: can't compute how far to advance
|
|
549
|
+
}
|
|
550
|
+
if (c0 === a0 && c1 === a1) {
|
|
551
|
+
return decodeTagValue(ba, this._dataView, type, valueStart, end, false)
|
|
552
|
+
}
|
|
553
|
+
if (altType < 0 && c0 === b0 && c1 === b1) {
|
|
554
|
+
altType = type
|
|
555
|
+
altStart = valueStart
|
|
556
|
+
altEnd = end
|
|
557
|
+
}
|
|
558
|
+
p = end
|
|
559
|
+
}
|
|
560
|
+
return altType < 0
|
|
561
|
+
? undefined
|
|
562
|
+
: decodeTagValue(ba, this._dataView, altType, altStart, altEnd, false)
|
|
563
|
+
}
|
|
564
|
+
|
|
465
565
|
private _findTag(tagName: string, raw: boolean) {
|
|
466
566
|
const tag1 = tagName.charCodeAt(0)
|
|
467
567
|
const tag2 = tagName.charCodeAt(1)
|
|
@@ -633,12 +733,18 @@ export default class BamRecord {
|
|
|
633
733
|
return lref
|
|
634
734
|
}
|
|
635
735
|
|
|
736
|
+
// Unmapped records are not special-cased here, unlike in
|
|
737
|
+
// _computeLengthOnRef. The spec says no assumptions can be made about an
|
|
738
|
+
// unmapped read's CIGAR, but "no assumptions" is not "no CIGAR": aligners
|
|
739
|
+
// emit placed unmapped mates that carry a real one, and htslib prints
|
|
740
|
+
// whatever is stored. Dropping it lost data the file had — paired.bam's
|
|
741
|
+
// SRR062635.1831187 at 20:74230 is FLAG 133 with 35M65S.
|
|
742
|
+
//
|
|
743
|
+
// Reference span stays 0 for them regardless, since that is a claim about
|
|
744
|
+
// alignment rather than about the stored bytes, and the query filter reads
|
|
745
|
+
// it.
|
|
636
746
|
private _computeNumericCigar(): NumericCigar {
|
|
637
747
|
const flag_nc = this._dataView.getInt32(this._start + 16, true)
|
|
638
|
-
if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
|
|
639
|
-
return new Uint32Array(0)
|
|
640
|
-
}
|
|
641
|
-
|
|
642
748
|
const numCigarOps = flag_nc & 0xffff
|
|
643
749
|
const p = this.b0 + this.read_name_length
|
|
644
750
|
|
|
@@ -675,14 +781,26 @@ export default class BamRecord {
|
|
|
675
781
|
return this._cachedNumericCigar
|
|
676
782
|
}
|
|
677
783
|
|
|
784
|
+
// Two appends per op, NOT `result += length + String.fromCharCode(op)`. The
|
|
785
|
+
// one-append form builds an intermediate cons string per op just to append it
|
|
786
|
+
// and drop it; appending each piece straight onto the rope is 1.23-1.35x
|
|
787
|
+
// faster on long reads, which is where this accessor dominates
|
|
788
|
+
// (chr22_nanopore_subset 51.2 -> 38.6 ms for 757 reads averaging 2171 ops,
|
|
789
|
+
// ultra-long-ont 13.2 -> 9.9 ms). Short reads carry 1-3 ops and land inside
|
|
790
|
+
// this box's noise band either way.
|
|
791
|
+
//
|
|
792
|
+
// ADR 0003 rejected two other rewrites of this loop — a precomputed 16-entry
|
|
793
|
+
// op-char table, and digits into a Uint8Array with one TextDecoder.decode —
|
|
794
|
+
// and both are still losers. Re-measured here: the op-char table is 1.13x
|
|
795
|
+
// SLOWER than String.fromCharCode even on top of this change, because V8
|
|
796
|
+
// already hands back an interned single-character string.
|
|
678
797
|
get CIGAR() {
|
|
679
798
|
const numeric = this.NUMERIC_CIGAR
|
|
680
799
|
let result = ''
|
|
681
800
|
for (let i = 0, l = numeric.length; i < l; i++) {
|
|
682
801
|
const packed = numeric[i]!
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
result += length + String.fromCharCode(opCode)
|
|
802
|
+
result += packed >> 4
|
|
803
|
+
result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
|
|
686
804
|
}
|
|
687
805
|
return result
|
|
688
806
|
}
|
|
@@ -708,9 +826,10 @@ export default class BamRecord {
|
|
|
708
826
|
return this._byteArray.subarray(p, p + this.num_seq_bytes)
|
|
709
827
|
}
|
|
710
828
|
|
|
711
|
-
// Decode
|
|
712
|
-
//
|
|
713
|
-
//
|
|
829
|
+
// Decode four bases per iteration off the 65536-entry table (two off the
|
|
830
|
+
// 256-entry one until that table has warmed up). Building an array of 1-char
|
|
831
|
+
// strings and join()ing it — the obvious approach — is 3x slower at 100bp and
|
|
832
|
+
// 35x slower at 15kb.
|
|
714
833
|
get seq() {
|
|
715
834
|
const len = this.seq_length
|
|
716
835
|
const ba = this._byteArray
|
|
@@ -718,9 +837,22 @@ export default class BamRecord {
|
|
|
718
837
|
const nPairs = len >> 1
|
|
719
838
|
let seq: string
|
|
720
839
|
if (len < SEQ_DECODER_THRESHOLD) {
|
|
840
|
+
const quads = seqretQuads()
|
|
721
841
|
seq = ''
|
|
722
|
-
|
|
723
|
-
|
|
842
|
+
if (quads === undefined) {
|
|
843
|
+
for (let j = 0; j < nPairs; j++) {
|
|
844
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
|
|
845
|
+
}
|
|
846
|
+
} else {
|
|
847
|
+
const nQuads = nPairs >> 1
|
|
848
|
+
for (let j = 0; j < nQuads; j++) {
|
|
849
|
+
const q = p + 2 * j
|
|
850
|
+
seq += quads[(ba[q]! << 8) | ba[q + 1]!]!
|
|
851
|
+
}
|
|
852
|
+
// odd byte count: two more bases off the 2-base table
|
|
853
|
+
if (nPairs & 1) {
|
|
854
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + nPairs - 1]!]!
|
|
855
|
+
}
|
|
724
856
|
}
|
|
725
857
|
if (len & 1) {
|
|
726
858
|
seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
|
package/src/util.ts
CHANGED
|
@@ -29,6 +29,16 @@ export interface BaseOpts {
|
|
|
29
29
|
onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Merge and order the chunks a query resolved to.
|
|
34
|
+
*
|
|
35
|
+
* Takes ownership of `chunks`: with no `lowest` to pre-filter against it sorts
|
|
36
|
+
* the array IN PLACE rather than copying. Every caller builds a fresh array to
|
|
37
|
+
* hand over, which is what makes that safe — passing something you still hold,
|
|
38
|
+
* or anything reachable from the index's per-refId cache, would reorder it
|
|
39
|
+
* underneath you. (The Chunk objects themselves are never mutated; a merged
|
|
40
|
+
* span produces a new instance.)
|
|
41
|
+
*/
|
|
32
42
|
export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
|
|
33
43
|
const n = chunks.length
|
|
34
44
|
if (n === 0) {
|
|
@@ -185,19 +195,14 @@ export function clampChunkEnds(
|
|
|
185
195
|
if (chunks.length === 0) {
|
|
186
196
|
return
|
|
187
197
|
}
|
|
188
|
-
// A plain array, sized
|
|
189
|
-
//
|
|
190
|
-
//
|
|
191
|
-
//
|
|
192
|
-
//
|
|
193
|
-
//
|
|
194
|
-
//
|
|
195
|
-
//
|
|
196
|
-
// real .bai shapes (min of 21, interleaved, sign-stable over 3 runs):
|
|
197
|
-
// building with push instead is 5-8% slower here, and a Float64Array is
|
|
198
|
-
// faster to sort but allocates per reference, which costs 1.7x on an
|
|
199
|
-
// assembly with 28751 scaffolds. Filling by index also avoids the
|
|
200
|
-
// intermediate array that mapping the linear index used to build.
|
|
198
|
+
// A plain array, pre-sized and filled by index. Measured against both
|
|
199
|
+
// alternatives on five real .bai shapes (min of 21, interleaved, sign-stable
|
|
200
|
+
// over 3 runs): building with push instead is 5-8% slower, and a
|
|
201
|
+
// Float64Array is faster to sort but allocates per reference, which costs
|
|
202
|
+
// 1.7x on an assembly with tens of thousands of unplaced scaffolds
|
|
203
|
+
// (cho.bam.bai has 28751 references, each with a handful of boundaries).
|
|
204
|
+
// Filling by index also avoids the intermediate array that mapping the
|
|
205
|
+
// linear index to block positions used to build.
|
|
201
206
|
const boundaries = new Array<number>(
|
|
202
207
|
extraBoundaries.length + chunks.length * 2,
|
|
203
208
|
)
|