@gmod/bam 7.8.2 → 7.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/record.ts CHANGED
@@ -31,6 +31,41 @@ for (let hi = 0; hi < 16; hi++) {
31
31
  // read lengths.
32
32
  const SEQ_DECODER_THRESHOLD = 300
33
33
 
34
+ // Four bases per entry, indexed by a PAIR of SEQ bytes, so the sub-threshold
35
+ // path halves its concat count: 1.5-1.6x on a short-read query end to end
36
+ // (shortreads_300x 46.5 -> 30.5 ms over 53.6k reads, volvox 4.8 -> 2.9 ms).
37
+ //
38
+ // Built lazily, and only once the short path has been taken enough times to pay
39
+ // for it. Filling 65536 entries costs ~6 ms and retains ~2 MB, so building on
40
+ // first use is a LOSS on a file that decodes only a handful of short reads:
41
+ // long-read fixtures with a short-read tail measured 1.4x slower that way
42
+ // (ecoli_nanopore has 27 sub-300bp reads out of 480, chm1 has 5 of 204). The
43
+ // counter is module-global on purpose — the table is shared, so what has to
44
+ // amortize is the total number of short decodes in the process, not per file.
45
+ const SEQRET_QUAD_WARMUP = 1024
46
+ let seqretShortCalls = 0
47
+ let SEQRET_QUAD_STRINGS: string[] | undefined
48
+
49
+ // The 4-base table, or undefined while still warming up (caller falls back to
50
+ // the 2-base table).
51
+ function seqretQuads() {
52
+ if (SEQRET_QUAD_STRINGS === undefined) {
53
+ if (++seqretShortCalls < SEQRET_QUAD_WARMUP) {
54
+ return undefined
55
+ }
56
+ const quads = new Array<string>(65536)
57
+ for (let a = 0; a < 256; a++) {
58
+ const sa = SEQRET_PAIR_STRINGS[a]!
59
+ const base = a << 8
60
+ for (let b = 0; b < 256; b++) {
61
+ quads[base | b] = sa + SEQRET_PAIR_STRINGS[b]!
62
+ }
63
+ }
64
+ SEQRET_QUAD_STRINGS = quads
65
+ }
66
+ return SEQRET_QUAD_STRINGS
67
+ }
68
+
34
69
  // Precomputed pair orientation strings, indexed by
35
70
  // ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
36
71
  // bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
@@ -115,12 +150,6 @@ function isNumericCigar(value: unknown): value is NumericCigar {
115
150
  )
116
151
  }
117
152
 
118
- export interface Bytes {
119
- start: number
120
- end: number
121
- byteArray: Uint8Array
122
- }
123
-
124
153
  type BArrayValue =
125
154
  | Int8Array
126
155
  | Uint8Array
@@ -213,13 +242,20 @@ function decodeBArrayTag(
213
242
  }
214
243
 
215
244
  // Byte span of a 'B' tag's element payload, for advancing the cursor past it.
245
+ // Returns -1 for a subtype outside the spec's cCsSiIf, whose element width is
246
+ // unknowable. Guessing one byte per element (which is what falling through to
247
+ // `limit` did) doesn't skip the tag, it lands the cursor mid-value and decodes
248
+ // the whole rest of the record's tag list out of garbage — where an unknown
249
+ // top-level type has always stopped the walk instead. Same choice here.
216
250
  function bArrayByteLength(Btype: number, limit: number) {
217
251
  if (Btype === 0x69 || Btype === 0x49 || Btype === 0x66) {
218
252
  return limit << 2
219
253
  } else if (Btype === 0x73 || Btype === 0x53) {
220
254
  return limit << 1
221
- } else {
255
+ } else if (Btype === 0x63 || Btype === 0x43) {
222
256
  return limit
257
+ } else {
258
+ return -1
223
259
  }
224
260
  }
225
261
 
@@ -253,6 +289,11 @@ function tagValueEnd(
253
289
  case 0x5a: // 'Z'
254
290
  case 0x48: {
255
291
  // 'H'
292
+ // Stays a plain byte loop. Swapping in Uint8Array.indexOf past a short
293
+ // inline probe is 2.7x faster on long-read MD (mean 9083 bytes on
294
+ // jb2bench's 200x.longread) but ~1.13x slower on short-read Z values,
295
+ // which are 4-13 bytes and are what the dominant case is made of. Measured
296
+ // both ways against the realistic corpus — see ADR 0012.
256
297
  let q = p
257
298
  while (q < blockEnd && ba[q] !== 0) {
258
299
  q++
@@ -263,7 +304,12 @@ function tagValueEnd(
263
304
  // 'B'
264
305
  const Btype = ba[p]!
265
306
  const limit = dataView.getInt32(p + 1, true)
266
- return p + 5 + bArrayByteLength(Btype, limit)
307
+ const payload = bArrayByteLength(Btype, limit)
308
+ if (payload < 0) {
309
+ console.error('Unknown BAM B tag subtype', Btype)
310
+ return 0
311
+ }
312
+ return p + 5 + payload
267
313
  }
268
314
  default:
269
315
  console.error('Unknown BAM tag type', type)
@@ -462,6 +508,60 @@ export default class BamRecord {
462
508
  return this._findTag(tagName, true)
463
509
  }
464
510
 
511
+ /**
512
+ * The value of `tagName`, or of `altName` if the record carries no `tagName` —
513
+ * resolved in ONE pass over the tag block instead of two.
514
+ *
515
+ * For the MM/Mm and ML/Ml alias pairs that modified-base callers emit, where
516
+ * `getTag(a) ?? getTag(b)` walks every tag on the record TWICE whenever
517
+ * neither is present, which is every read in a file without base
518
+ * modifications. jbrowse-components issues exactly that lookup per record on
519
+ * every render (extractModifications runs unconditionally), and on
520
+ * jb2bench's 1000x.shortread it was 12.9% of the whole query — more than the
521
+ * CIGAR, SEQ and MD reads the pileup actually uses, spent proving absence.
522
+ *
523
+ * `tagName` wins wherever it appears, so the result matches the two-lookup
524
+ * form even for a (malformed) record carrying both.
525
+ */
526
+ getTagAlt(tagName: string, altName: string) {
527
+ if (this._cachedTags !== undefined) {
528
+ return this._cachedTags[tagName] ?? this._cachedTags[altName]
529
+ }
530
+ const a0 = tagName.charCodeAt(0)
531
+ const a1 = tagName.charCodeAt(1)
532
+ const b0 = altName.charCodeAt(0)
533
+ const b1 = altName.charCodeAt(1)
534
+ const blockEnd = this._end
535
+ const ba = this._byteArray
536
+ let p = this.tagsStart
537
+ // where the alternate landed, if it turns up before the primary does
538
+ let altType = -1
539
+ let altStart = 0
540
+ let altEnd = 0
541
+ while (p < blockEnd) {
542
+ const c0 = ba[p]
543
+ const c1 = ba[p + 1]
544
+ const type = ba[p + 2]!
545
+ const valueStart = p + 3
546
+ const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
547
+ if (end === 0) {
548
+ break // unknown type: can't compute how far to advance
549
+ }
550
+ if (c0 === a0 && c1 === a1) {
551
+ return decodeTagValue(ba, this._dataView, type, valueStart, end, false)
552
+ }
553
+ if (altType < 0 && c0 === b0 && c1 === b1) {
554
+ altType = type
555
+ altStart = valueStart
556
+ altEnd = end
557
+ }
558
+ p = end
559
+ }
560
+ return altType < 0
561
+ ? undefined
562
+ : decodeTagValue(ba, this._dataView, altType, altStart, altEnd, false)
563
+ }
564
+
465
565
  private _findTag(tagName: string, raw: boolean) {
466
566
  const tag1 = tagName.charCodeAt(0)
467
567
  const tag2 = tagName.charCodeAt(1)
@@ -633,12 +733,18 @@ export default class BamRecord {
633
733
  return lref
634
734
  }
635
735
 
736
+ // Unmapped records are not special-cased here, unlike in
737
+ // _computeLengthOnRef. The spec says no assumptions can be made about an
738
+ // unmapped read's CIGAR, but "no assumptions" is not "no CIGAR": aligners
739
+ // emit placed unmapped mates that carry a real one, and htslib prints
740
+ // whatever is stored. Dropping it lost data the file had — paired.bam's
741
+ // SRR062635.1831187 at 20:74230 is FLAG 133 with 35M65S.
742
+ //
743
+ // Reference span stays 0 for them regardless, since that is a claim about
744
+ // alignment rather than about the stored bytes, and the query filter reads
745
+ // it.
636
746
  private _computeNumericCigar(): NumericCigar {
637
747
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
638
- if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
639
- return new Uint32Array(0)
640
- }
641
-
642
748
  const numCigarOps = flag_nc & 0xffff
643
749
  const p = this.b0 + this.read_name_length
644
750
 
@@ -675,14 +781,26 @@ export default class BamRecord {
675
781
  return this._cachedNumericCigar
676
782
  }
677
783
 
784
+ // Two appends per op, NOT `result += length + String.fromCharCode(op)`. The
785
+ // one-append form builds an intermediate cons string per op just to append it
786
+ // and drop it; appending each piece straight onto the rope is 1.23-1.35x
787
+ // faster on long reads, which is where this accessor dominates
788
+ // (chr22_nanopore_subset 51.2 -> 38.6 ms for 757 reads averaging 2171 ops,
789
+ // ultra-long-ont 13.2 -> 9.9 ms). Short reads carry 1-3 ops and land inside
790
+ // this box's noise band either way.
791
+ //
792
+ // ADR 0003 rejected two other rewrites of this loop — a precomputed 16-entry
793
+ // op-char table, and digits into a Uint8Array with one TextDecoder.decode —
794
+ // and both are still losers. Re-measured here: the op-char table is 1.13x
795
+ // SLOWER than String.fromCharCode even on top of this change, because V8
796
+ // already hands back an interned single-character string.
678
797
  get CIGAR() {
679
798
  const numeric = this.NUMERIC_CIGAR
680
799
  let result = ''
681
800
  for (let i = 0, l = numeric.length; i < l; i++) {
682
801
  const packed = numeric[i]!
683
- const length = packed >> 4
684
- const opCode = ASCII_CIGAR_CODES[packed & 0xf]!
685
- result += length + String.fromCharCode(opCode)
802
+ result += packed >> 4
803
+ result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
686
804
  }
687
805
  return result
688
806
  }
@@ -708,9 +826,10 @@ export default class BamRecord {
708
826
  return this._byteArray.subarray(p, p + this.num_seq_bytes)
709
827
  }
710
828
 
711
- // Decode two bases per iteration off a 256-entry table. Building an array of
712
- // 1-char strings and join()ing it — the obvious approach — is 3x slower at
713
- // 100bp and 35x slower at 15kb.
829
+ // Decode four bases per iteration off the 65536-entry table (two off the
830
+ // 256-entry one until that table has warmed up). Building an array of 1-char
831
+ // strings and join()ing it — the obvious approach — is 3x slower at 100bp and
832
+ // 35x slower at 15kb.
714
833
  get seq() {
715
834
  const len = this.seq_length
716
835
  const ba = this._byteArray
@@ -718,9 +837,22 @@ export default class BamRecord {
718
837
  const nPairs = len >> 1
719
838
  let seq: string
720
839
  if (len < SEQ_DECODER_THRESHOLD) {
840
+ const quads = seqretQuads()
721
841
  seq = ''
722
- for (let j = 0; j < nPairs; j++) {
723
- seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
842
+ if (quads === undefined) {
843
+ for (let j = 0; j < nPairs; j++) {
844
+ seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
845
+ }
846
+ } else {
847
+ const nQuads = nPairs >> 1
848
+ for (let j = 0; j < nQuads; j++) {
849
+ const q = p + 2 * j
850
+ seq += quads[(ba[q]! << 8) | ba[q + 1]!]!
851
+ }
852
+ // odd byte count: two more bases off the 2-base table
853
+ if (nPairs & 1) {
854
+ seq += SEQRET_PAIR_STRINGS[ba[p + nPairs - 1]!]!
855
+ }
724
856
  }
725
857
  if (len & 1) {
726
858
  seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
package/src/util.ts CHANGED
@@ -29,6 +29,16 @@ export interface BaseOpts {
29
29
  onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
30
30
  }
31
31
 
32
+ /**
33
+ * Merge and order the chunks a query resolved to.
34
+ *
35
+ * Takes ownership of `chunks`: with no `lowest` to pre-filter against it sorts
36
+ * the array IN PLACE rather than copying. Every caller builds a fresh array to
37
+ * hand over, which is what makes that safe — passing something you still hold,
38
+ * or anything reachable from the index's per-refId cache, would reorder it
39
+ * underneath you. (The Chunk objects themselves are never mutated; a merged
40
+ * span produces a new instance.)
41
+ */
32
42
  export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
33
43
  const n = chunks.length
34
44
  if (n === 0) {
@@ -185,19 +195,14 @@ export function clampChunkEnds(
185
195
  if (chunks.length === 0) {
186
196
  return
187
197
  }
188
- // A plain array, sized once and filled by index. Not a Float64Array: this
189
- // runs once per reference, and an assembly with tens of thousands of unplaced
190
- // scaffolds (cho.bam.bai has 28751 references) pays the allocation that many
191
- // times over, where each individual boundary list is a handful of entries.
192
- // The small-allocation cost dominates the faster typed sort by a wide margin
193
- // at that shape. Filling by index still avoids the intermediate array that
194
- // mapping the linear index to block positions used to build.
195
- // Pre-sized and filled by index. Measured against both alternatives on five
196
- // real .bai shapes (min of 21, interleaved, sign-stable over 3 runs):
197
- // building with push instead is 5-8% slower here, and a Float64Array is
198
- // faster to sort but allocates per reference, which costs 1.7x on an
199
- // assembly with 28751 scaffolds. Filling by index also avoids the
200
- // intermediate array that mapping the linear index used to build.
198
+ // A plain array, pre-sized and filled by index. Measured against both
199
+ // alternatives on five real .bai shapes (min of 21, interleaved, sign-stable
200
+ // over 3 runs): building with push instead is 5-8% slower, and a
201
+ // Float64Array is faster to sort but allocates per reference, which costs
202
+ // 1.7x on an assembly with tens of thousands of unplaced scaffolds
203
+ // (cho.bam.bai has 28751 references, each with a handful of boundaries).
204
+ // Filling by index also avoids the intermediate array that mapping the
205
+ // linear index to block positions used to build.
201
206
  const boundaries = new Array<number>(
202
207
  extraBoundaries.length + chunks.length * 2,
203
208
  )