@gmod/bam 7.3.4 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/record.ts CHANGED
@@ -1,16 +1,45 @@
1
1
  import { CIGAR_REF_SKIP, CIGAR_SOFT_CLIP } from './cigar.ts'
2
2
  import Constants from './constants.ts'
3
3
 
4
- const SEQRET_DECODER = '=ACMGRSVTWYHKDBN'.split('')
4
+ const SEQRET = '=ACMGRSVTWYHKDBN'
5
+ const SEQRET_DECODER = SEQRET.split('')
6
+ const SEQRET_CODES = Uint8Array.from(SEQRET, c => c.charCodeAt(0))
7
+
8
+ // Both bases of a SEQ byte, precomputed for all 256 bytes so decoding advances a
9
+ // byte at a time. Two forms because `seq` has two strategies (see below): packed
10
+ // ASCII codes for a single Uint16Array store, and the 2-char string to append.
11
+ // Byte order within the u16 depends on host endianness.
12
+ const LITTLE_ENDIAN = new Uint16Array(Uint8Array.of(1, 0).buffer)[0] === 1
13
+ const SEQRET_PAIR_CODES = new Uint16Array(256)
14
+ const SEQRET_PAIR_STRINGS = new Array<string>(256)
15
+ for (let hi = 0; hi < 16; hi++) {
16
+ for (let lo = 0; lo < 16; lo++) {
17
+ const h = SEQRET_CODES[hi]!
18
+ const l = SEQRET_CODES[lo]!
19
+ SEQRET_PAIR_CODES[(hi << 4) | lo] = LITTLE_ENDIAN
20
+ ? h | (l << 8)
21
+ : (h << 8) | l
22
+ SEQRET_PAIR_STRINGS[(hi << 4) | lo] =
23
+ SEQRET_DECODER[hi]! + SEQRET_DECODER[lo]!
24
+ }
25
+ }
5
26
 
6
- // precomputed pair orientation strings indexed by ((flags >> 4) & 0xF) | (isize > 0 ? 16 : 0)
7
- // bits 0-3 encode flag bits 0x10(reverse),0x20(mate reverse),0x40(read1),0x80(read2)
8
- // bit 4 encodes whether isize > 0
27
+ // Read length at which building a byte buffer and calling TextDecoder once
28
+ // overtakes plain string concatenation. Below it concat wins by 2-4x (rope
29
+ // building is cheap and the decode has fixed overhead); above it the decoder
30
+ // wins by 5-8x. Measured crossover is ~300bp, i.e. just past typical Illumina
31
+ // read lengths.
32
+ const SEQ_DECODER_THRESHOLD = 300
33
+
34
+ // Precomputed pair orientation strings, indexed by
35
+ // ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
36
+ // bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
37
+ // bit 3 is whether this read is the leftmost of the pair. The read2 flag (0x80)
38
+ // is deliberately not consulted — "not read1" is what decides the numbering, so
39
+ // a record with neither flag set reads the same as read2.
9
40
  // prettier-ignore
10
41
  const PAIR_ORIENTATION_TABLE = [
11
- 'F F ','F R ','R F ','R R ','F2F1','F2R1','R2F1','R2R1',
12
42
  'F1F2','F1R2','R1F2','R1R2','F2F1','F2R1','R2F1','R2R1',
13
- 'F F ','R F ','F R ','R R ','F1F2','R1F2','F1R2','R1R2',
14
43
  'F2F1','R2F1','F2R1','R2R1','F1F2','R1F2','F1R2','R1R2',
15
44
  ]
16
45
  const ASCII_CIGAR_CODES = [
@@ -23,6 +52,19 @@ const textDecoder = new TextDecoder()
23
52
  // Binary: 0b111001101 = 0x1CD
24
53
  const CIGAR_CONSUMES_REF_MASK = 0x1cd
25
54
 
55
+ // A CIGAR as packed op words. Either a view over (or copy of) the record's own
56
+ // CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
57
+ // too, since a writer may encode CG as B:i rather than the usual B:I.
58
+ export type NumericCigar = Uint32Array | Int32Array | number[]
59
+
60
+ function isNumericCigar(value: unknown): value is NumericCigar {
61
+ return (
62
+ value instanceof Uint32Array ||
63
+ value instanceof Int32Array ||
64
+ Array.isArray(value)
65
+ )
66
+ }
67
+
26
68
  export interface Bytes {
27
69
  start: number
28
70
  end: number
@@ -42,7 +84,11 @@ type BArrayValue =
42
84
  // Decode a 'B' (array) tag value starting at `p` (the byte after type+subtype+
43
85
  // count). When the data is naturally aligned we return a typed-array view over
44
86
  // the underlying buffer (zero-copy); otherwise we copy element-by-element since
45
- // typed-array views require alignment. Shared by getTag and the full-tag parse.
87
+ // typed-array views require alignment. The copy target is a plain array, not a
88
+ // typed one: benchmarking the unaligned path found number[] fills faster below
89
+ // ~10k elements and reads at least as fast at every size, so the only thing a
90
+ // typed copy would buy is half the retained bytes. Shared by getTag and the
91
+ // full-tag parse.
46
92
  function decodeBArrayTag(
47
93
  ba: Uint8Array,
48
94
  dataView: DataView,
@@ -228,7 +274,7 @@ export default class BamRecord {
228
274
  private _cachedEnd?: number
229
275
  private _cachedTags?: Record<string, unknown>
230
276
  private _cachedLengthOnRef?: number
231
- private _cachedNumericCigar?: Uint32Array | number[]
277
+ private _cachedNumericCigar?: NumericCigar
232
278
  private _cachedNUMERIC_MD?: Uint8Array | null
233
279
  private _cachedSeqStart?: number
234
280
 
@@ -374,7 +420,9 @@ export default class BamRecord {
374
420
  private _computeTags() {
375
421
  const blockEnd = this._end
376
422
  const ba = this._byteArray
377
- const tags: Record<string, unknown> = {}
423
+ // null prototype: tag names come from the file, so a read carrying a
424
+ // "constructor" or "toString" tag must not resolve to Object.prototype's
425
+ const tags: Record<string, unknown> = Object.create(null)
378
426
  let p = this.tagsStart
379
427
  while (p < blockEnd) {
380
428
  const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
@@ -499,6 +547,8 @@ export default class BamRecord {
499
547
  absOffset,
500
548
  numCigarOps,
501
549
  )
550
+ // the view we need to sum is exactly what NUMERIC_CIGAR would build, so
551
+ // seed its cache rather than making it construct a second one
502
552
  this._cachedNumericCigar = cigarView
503
553
  let lref = 0
504
554
  for (let c = 0; c < numCigarOps; ++c) {
@@ -516,7 +566,7 @@ export default class BamRecord {
516
566
  return lref
517
567
  }
518
568
 
519
- private _computeNumericCigar(): Uint32Array | number[] {
569
+ private _computeNumericCigar(): NumericCigar {
520
570
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
521
571
  if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
522
572
  return new Uint32Array(0)
@@ -526,10 +576,10 @@ export default class BamRecord {
526
576
  const p = this.b0 + this.read_name_length
527
577
 
528
578
  if (this._isCGTagPattern(p, numCigarOps)) {
529
- return (
530
- (this.tags.CG as Uint32Array | number[] | undefined) ??
531
- new Uint32Array(0)
532
- )
579
+ // getTag, not this.tags: the real CIGAR lives in one tag, so there's no
580
+ // reason to decode every other tag on the record to reach it
581
+ const cg = this.getTag('CG')
582
+ return isNumericCigar(cg) ? cg : new Uint32Array(0)
533
583
  }
534
584
 
535
585
  const absOffset = this._byteArray.byteOffset + p
@@ -591,43 +641,66 @@ export default class BamRecord {
591
641
  return this._byteArray.subarray(p, p + this.num_seq_bytes)
592
642
  }
593
643
 
644
+ // Decode two bases per iteration off a 256-entry table. Building an array of
645
+ // 1-char strings and join()ing it — the obvious approach — is 3x slower at
646
+ // 100bp and 35x slower at 15kb.
594
647
  get seq() {
595
648
  const len = this.seq_length
596
- const seqStart = this.seqStart
597
- const numeric = this._byteArray
598
- const buf = new Array(len)
599
- let i = 0
600
- const fullBytes = len >> 1
601
-
602
- for (let j = 0; j < fullBytes; ++j) {
603
- const sb = numeric[seqStart + j]!
604
- buf[i++] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
605
- buf[i++] = SEQRET_DECODER[sb & 0x0f]!
606
- }
607
-
608
- if (i < len) {
609
- const sb = numeric[seqStart + fullBytes]!
610
- buf[i] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
649
+ const ba = this._byteArray
650
+ const p = this.seqStart
651
+ const nPairs = len >> 1
652
+ let seq: string
653
+ if (len < SEQ_DECODER_THRESHOLD) {
654
+ seq = ''
655
+ for (let j = 0; j < nPairs; j++) {
656
+ seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
657
+ }
658
+ if (len & 1) {
659
+ seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
660
+ }
661
+ } else {
662
+ // round up to an even length so the Uint16Array view spans every pair; a
663
+ // trailing odd base is written as a single byte and trimmed on decode
664
+ const out = new Uint8Array((len + 1) & ~1)
665
+ const pairs = new Uint16Array(out.buffer)
666
+ for (let j = 0; j < nPairs; j++) {
667
+ pairs[j] = SEQRET_PAIR_CODES[ba[p + j]!]!
668
+ }
669
+ if (len & 1) {
670
+ out[len - 1] = SEQRET_CODES[(ba[p + nPairs]! & 0xf0) >> 4]!
671
+ }
672
+ seq = textDecoder.decode(out.subarray(0, len))
611
673
  }
612
-
613
- return buf.join('')
674
+ return seq
614
675
  }
615
676
 
616
- // adapted from igv.js
617
- // uses precomputed lookup table indexed by flag bits + isize sign.
618
- // the BAM spec defines tlen as positive for the leftmost segment and
619
- // negative for the rightmost, so tlen > 0 reliably indicates which
620
- // read comes first without needing position-based correction
677
+ // Must come out identical from either mate, or the two halves of one normal
678
+ // pair render as different orientations. The leftmost mate is therefore picked
679
+ // by a total order on (refId, pos) that both mates evaluate the same way, with
680
+ // a read1-first tie-break for equal loci.
681
+ //
682
+ // Deriving "leftmost" from template_length looks tempting — the spec makes
683
+ // tlen positive for the leftmost segment and negative for the rightmost — but
684
+ // aligners leave tlen at 0 whenever the insert size is unavailable, which
685
+ // includes every cross-reference pair. Both mates then read as "not leftmost"
686
+ // and disagree with each other.
621
687
  // (see also: gmod/cram-js src/cramFile/record.ts getPairOrientation)
622
688
  get pair_orientation() {
623
689
  const f = this.flags
624
- // unmapped (0x4) or mate unmapped (0x8) -> undefined
625
- if (f & 0xc || this.ref_id !== this.next_refid) {
690
+ if (!(f & Constants.BAM_FPAIRED)) {
626
691
  return undefined
627
692
  }
628
- return PAIR_ORIENTATION_TABLE[
629
- ((f >> 4) & 0xf) | (this.template_length > 0 ? 16 : 0)
630
- ]
693
+ const refId = this.ref_id
694
+ const mateRefId = this.next_refid
695
+ const pos = this.start
696
+ const matePos = this.next_pos
697
+ const selfIsLeft =
698
+ refId !== mateRefId
699
+ ? refId < mateRefId
700
+ : pos !== matePos
701
+ ? pos < matePos
702
+ : !!(f & Constants.BAM_FREAD1)
703
+ return PAIR_ORIENTATION_TABLE[((f >> 4) & 0x7) | (selfIsLeft ? 8 : 0)]
631
704
  }
632
705
 
633
706
  get bin_mq_nl() {
@@ -658,7 +731,7 @@ export default class BamRecord {
658
731
  if (idx < this.seq_length) {
659
732
  const sb = this._byteArray[this.seqStart + (idx >> 1)]!
660
733
 
661
- return idx % 2 === 0
734
+ return (idx & 1) === 0
662
735
  ? SEQRET_DECODER[(sb & 0xf0) >> 4]!
663
736
  : SEQRET_DECODER[sb & 0x0f]!
664
737
  } else {
package/src/util.ts CHANGED
@@ -126,7 +126,10 @@ export function parsePseudoBin(bytes: Uint8Array, offset: number) {
126
126
  // maxv.blockPosition is an upper bound on where that final block ends — always at
127
127
  // least the true block end, so the clamped fetch still contains the whole block.
128
128
  // Shrinks both the byte estimate and the actual fetch with no extra I/O.
129
- export function clampChunkEnds(chunks: Chunk[], extraBoundaries: number[] = []) {
129
+ export function clampChunkEnds(
130
+ chunks: Chunk[],
131
+ extraBoundaries: number[] = [],
132
+ ) {
130
133
  const boundaries = [...extraBoundaries]
131
134
  for (const c of chunks) {
132
135
  boundaries.push(c.minv.blockPosition, c.maxv.blockPosition)
@@ -163,9 +166,15 @@ export function parseRefSeqs(
163
166
  if (start + 4 > uncba.length) {
164
167
  return undefined
165
168
  }
166
- const dataView = new DataView(uncba.buffer)
169
+ const dataView = new DataView(
170
+ uncba.buffer,
171
+ uncba.byteOffset,
172
+ uncba.byteLength,
173
+ )
167
174
  const nRef = dataView.getInt32(start, true)
168
- const chrToIndex: Record<string, number> = {}
175
+ // null prototype: ref names come from the file, so a contig named
176
+ // "constructor" must not resolve to Object.prototype's
177
+ const chrToIndex: Record<string, number> = Object.create(null)
169
178
  const indexToChr: { refName: string; length: number }[] = []
170
179
  const decoder = new TextDecoder('utf8')
171
180
 
@@ -209,7 +218,7 @@ export function parseNameBytes(
209
218
  let currRefId = 0
210
219
  let currNameStart = 0
211
220
  const refIdToName: string[] = []
212
- const refNameToId: Record<string, number> = {}
221
+ const refNameToId: Record<string, number> = Object.create(null)
213
222
  for (let i = 0; i < namesBytes.length; i++) {
214
223
  if (!namesBytes[i]) {
215
224
  if (currNameStart < i) {
@@ -254,18 +263,19 @@ export function filterTagValue(readVal: unknown, filterVal?: string) {
254
263
  : `${readVal}` !== `${filterVal}`
255
264
  }
256
265
 
257
- export function filterCacheKey(filterBy?: FilterBy) {
258
- if (!filterBy) {
259
- return ''
260
- }
261
- const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
262
- const tagPart = tagFilter ? `:${tagFilter.tag}=${tagFilter.value ?? '*'}` : ''
263
- return `:f${flagInclude}x${flagExclude}${tagPart}`
264
- }
265
-
266
266
  interface Filterable {
267
267
  flags: number
268
268
  tags: Record<string, unknown>
269
+ // BamRecord decodes one tag by walking the tag block, without building the
270
+ // whole tags object. Optional because a custom recordClass need not have it.
271
+ getTag?(tag: string): unknown
272
+ }
273
+
274
+ // Read a single tag, preferring the targeted accessor. Reaching for `tags`
275
+ // instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
276
+ // ~10 per read) just to test one, which measured 2.6x the cost of getTag.
277
+ function readTag(record: Filterable, tag: string) {
278
+ return record.getTag ? record.getTag(tag) : record.tags[tag]
269
279
  }
270
280
 
271
281
  // Apply flagInclude/flagExclude/tagFilter to a list of records.
@@ -279,7 +289,7 @@ export function applyFilters<T extends Filterable>(
279
289
  const r = records[i]!
280
290
  if (
281
291
  !filterReadFlag(r.flags, flagInclude, flagExclude) &&
282
- !(tagFilter && filterTagValue(r.tags[tagFilter.tag], tagFilter.value))
292
+ !(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
283
293
  ) {
284
294
  out.push(r)
285
295
  }