@gmod/bam 7.3.4 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/record.ts CHANGED
@@ -1,16 +1,45 @@
1
1
  import { CIGAR_REF_SKIP, CIGAR_SOFT_CLIP } from './cigar.ts'
2
2
  import Constants from './constants.ts'
3
3
 
4
- const SEQRET_DECODER = '=ACMGRSVTWYHKDBN'.split('')
4
+ const SEQRET = '=ACMGRSVTWYHKDBN'
5
+ const SEQRET_DECODER = SEQRET.split('')
6
+ const SEQRET_CODES = Uint8Array.from(SEQRET, c => c.charCodeAt(0))
7
+
8
+ // Both bases of a SEQ byte, precomputed for all 256 bytes so decoding advances a
9
+ // byte at a time. Two forms because `seq` has two strategies (see below): packed
10
+ // ASCII codes for a single Uint16Array store, and the 2-char string to append.
11
+ // Byte order within the u16 depends on host endianness.
12
+ const LITTLE_ENDIAN = new Uint16Array(Uint8Array.of(1, 0).buffer)[0] === 1
13
+ const SEQRET_PAIR_CODES = new Uint16Array(256)
14
+ const SEQRET_PAIR_STRINGS = new Array<string>(256)
15
+ for (let hi = 0; hi < 16; hi++) {
16
+ for (let lo = 0; lo < 16; lo++) {
17
+ const h = SEQRET_CODES[hi]!
18
+ const l = SEQRET_CODES[lo]!
19
+ SEQRET_PAIR_CODES[(hi << 4) | lo] = LITTLE_ENDIAN
20
+ ? h | (l << 8)
21
+ : (h << 8) | l
22
+ SEQRET_PAIR_STRINGS[(hi << 4) | lo] =
23
+ SEQRET_DECODER[hi]! + SEQRET_DECODER[lo]!
24
+ }
25
+ }
5
26
 
6
- // precomputed pair orientation strings indexed by ((flags >> 4) & 0xF) | (isize > 0 ? 16 : 0)
7
- // bits 0-3 encode flag bits 0x10(reverse),0x20(mate reverse),0x40(read1),0x80(read2)
8
- // bit 4 encodes whether isize > 0
27
+ // Read length at which building a byte buffer and calling TextDecoder once
28
+ // overtakes plain string concatenation. Below it concat wins by 2-4x (rope
29
+ // building is cheap and the decode has fixed overhead); above it the decoder
30
+ // wins by 5-8x. Measured crossover is ~300bp, i.e. just past typical Illumina
31
+ // read lengths.
32
+ const SEQ_DECODER_THRESHOLD = 300
33
+
34
+ // Precomputed pair orientation strings, indexed by
35
+ // ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
36
+ // bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
37
+ // bit 3 is whether this read is the leftmost of the pair. The read2 flag (0x80)
38
+ // is deliberately not consulted — "not read1" is what decides the numbering, so
39
+ // a record with neither flag set reads the same as read2.
9
40
  // prettier-ignore
10
41
  const PAIR_ORIENTATION_TABLE = [
11
- 'F F ','F R ','R F ','R R ','F2F1','F2R1','R2F1','R2R1',
12
42
  'F1F2','F1R2','R1F2','R1R2','F2F1','F2R1','R2F1','R2R1',
13
- 'F F ','R F ','F R ','R R ','F1F2','R1F2','F1R2','R1R2',
14
43
  'F2F1','R2F1','F2R1','R2R1','F1F2','R1F2','F1R2','R1R2',
15
44
  ]
16
45
  const ASCII_CIGAR_CODES = [
@@ -19,10 +48,62 @@ const ASCII_CIGAR_CODES = [
19
48
 
20
49
  const textDecoder = new TextDecoder()
21
50
 
51
+ // Interned two-char tag names keyed on the packed byte pair. Worth more than
52
+ // the saved String.fromCharCode: the tags object is null-prototype and so in
53
+ // dictionary mode, and handing it a string V8 has already hashed halves the
54
+ // per-tag insert cost.
55
+ const TAG_NAMES = new Map<number, string>()
56
+ function tagName(b0: number, b1: number) {
57
+ const key = (b0 << 8) | b1
58
+ let name = TAG_NAMES.get(key)
59
+ if (name === undefined) {
60
+ name = String.fromCharCode(b0, b1)
61
+ TAG_NAMES.set(key, name)
62
+ }
63
+ return name
64
+ }
65
+
66
+ // TextDecoder carries ~0.35us of fixed setup per call, which dominates for the
67
+ // short values Z/H tags usually hold (MD, RG, PG, ...): char codes are 4x
68
+ // faster at 8 bytes, 2x at 16, and the two cross over at 32.
69
+ const SHORT_STRING_THRESHOLD = 32
70
+
71
+ // Z/H tag values are spec'd as printable ASCII, but a non-conforming writer
72
+ // would make the char-code path mis-decode UTF-8, so bail to TextDecoder on any
73
+ // high byte.
74
+ function decodeTagString(ba: Uint8Array, start: number, end: number) {
75
+ const len = end - start
76
+ if (len < SHORT_STRING_THRESHOLD) {
77
+ const codes = new Array<number>(len)
78
+ for (let i = 0; i < len; i++) {
79
+ const byte = ba[start + i]!
80
+ if (byte > 0x7f) {
81
+ return textDecoder.decode(ba.subarray(start, end))
82
+ }
83
+ codes[i] = byte
84
+ }
85
+ return String.fromCharCode(...codes)
86
+ }
87
+ return textDecoder.decode(ba.subarray(start, end))
88
+ }
89
+
22
90
  // Bitmask for ops that consume ref: M=0, D=2, N=3, P=6, ==7, X=8
23
91
  // Binary: 0b111001101 = 0x1CD
24
92
  const CIGAR_CONSUMES_REF_MASK = 0x1cd
25
93
 
94
+ // A CIGAR as packed op words. Either a view over (or copy of) the record's own
95
+ // CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
96
+ // too, since a writer may encode CG as B:i rather than the usual B:I.
97
+ export type NumericCigar = Uint32Array | Int32Array | number[]
98
+
99
+ function isNumericCigar(value: unknown): value is NumericCigar {
100
+ return (
101
+ value instanceof Uint32Array ||
102
+ value instanceof Int32Array ||
103
+ Array.isArray(value)
104
+ )
105
+ }
106
+
26
107
  export interface Bytes {
27
108
  start: number
28
109
  end: number
@@ -42,7 +123,11 @@ type BArrayValue =
42
123
  // Decode a 'B' (array) tag value starting at `p` (the byte after type+subtype+
43
124
  // count). When the data is naturally aligned we return a typed-array view over
44
125
  // the underlying buffer (zero-copy); otherwise we copy element-by-element since
45
- // typed-array views require alignment. Shared by getTag and the full-tag parse.
126
+ // typed-array views require alignment. The copy target is a plain array, not a
127
+ // typed one: benchmarking the unaligned path found number[] fills faster below
128
+ // ~10k elements and reads at least as fast at every size, so the only thing a
129
+ // typed copy would buy is half the retained bytes. Shared by getTag and the
130
+ // full-tag parse.
46
131
  function decodeBArrayTag(
47
132
  ba: Uint8Array,
48
133
  dataView: DataView,
@@ -206,9 +291,7 @@ function decodeTagValue(
206
291
  return dataView.getFloat32(p, true)
207
292
  case 0x5a: // 'Z'
208
293
  case 0x48: // 'H'
209
- return raw
210
- ? ba.subarray(p, end - 1)
211
- : textDecoder.decode(ba.subarray(p, end - 1))
294
+ return raw ? ba.subarray(p, end - 1) : decodeTagString(ba, p, end - 1)
212
295
  default: {
213
296
  // 'B'
214
297
  const Btype = ba[p]!
@@ -228,7 +311,7 @@ export default class BamRecord {
228
311
  private _cachedEnd?: number
229
312
  private _cachedTags?: Record<string, unknown>
230
313
  private _cachedLengthOnRef?: number
231
- private _cachedNumericCigar?: Uint32Array | number[]
314
+ private _cachedNumericCigar?: NumericCigar
232
315
  private _cachedNUMERIC_MD?: Uint8Array | null
233
316
  private _cachedSeqStart?: number
234
317
 
@@ -312,6 +395,13 @@ export default class BamRecord {
312
395
  }
313
396
 
314
397
  // batch fromCharCode: fastest for typical name lengths (see benchmarks/string-building.bench.ts)
398
+ //
399
+ // Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
400
+ // read name about once — jbrowse-components' buildBaseFeatureData copies it
401
+ // straight into its own FeatureData — so a cache would pay a field slot on
402
+ // every record (+180KB per 22k-record chunk) to save zero decodes, and would
403
+ // pin every name string for as long as the chunk stays cached (+520KB) where
404
+ // today it dies with the consumer's copy.
315
405
  get name() {
316
406
  const len = this.read_name_length - 1
317
407
  const start = this.b0
@@ -374,10 +464,12 @@ export default class BamRecord {
374
464
  private _computeTags() {
375
465
  const blockEnd = this._end
376
466
  const ba = this._byteArray
377
- const tags: Record<string, unknown> = {}
467
+ // null prototype: tag names come from the file, so a read carrying a
468
+ // "constructor" or "toString" tag must not resolve to Object.prototype's
469
+ const tags: Record<string, unknown> = Object.create(null)
378
470
  let p = this.tagsStart
379
471
  while (p < blockEnd) {
380
- const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
472
+ const tag = tagName(ba[p]!, ba[p + 1]!)
381
473
  const type = ba[p + 2]!
382
474
  const valueStart = p + 3
383
475
  const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
@@ -499,6 +591,8 @@ export default class BamRecord {
499
591
  absOffset,
500
592
  numCigarOps,
501
593
  )
594
+ // the view we need to sum is exactly what NUMERIC_CIGAR would build, so
595
+ // seed its cache rather than making it construct a second one
502
596
  this._cachedNumericCigar = cigarView
503
597
  let lref = 0
504
598
  for (let c = 0; c < numCigarOps; ++c) {
@@ -516,7 +610,7 @@ export default class BamRecord {
516
610
  return lref
517
611
  }
518
612
 
519
- private _computeNumericCigar(): Uint32Array | number[] {
613
+ private _computeNumericCigar(): NumericCigar {
520
614
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
521
615
  if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
522
616
  return new Uint32Array(0)
@@ -526,10 +620,10 @@ export default class BamRecord {
526
620
  const p = this.b0 + this.read_name_length
527
621
 
528
622
  if (this._isCGTagPattern(p, numCigarOps)) {
529
- return (
530
- (this.tags.CG as Uint32Array | number[] | undefined) ??
531
- new Uint32Array(0)
532
- )
623
+ // getTag, not this.tags: the real CIGAR lives in one tag, so there's no
624
+ // reason to decode every other tag on the record to reach it
625
+ const cg = this.getTag('CG')
626
+ return isNumericCigar(cg) ? cg : new Uint32Array(0)
533
627
  }
534
628
 
535
629
  const absOffset = this._byteArray.byteOffset + p
@@ -591,43 +685,66 @@ export default class BamRecord {
591
685
  return this._byteArray.subarray(p, p + this.num_seq_bytes)
592
686
  }
593
687
 
688
+ // Decode two bases per iteration off a 256-entry table. Building an array of
689
+ // 1-char strings and join()ing it — the obvious approach — is 3x slower at
690
+ // 100bp and 35x slower at 15kb.
594
691
  get seq() {
595
692
  const len = this.seq_length
596
- const seqStart = this.seqStart
597
- const numeric = this._byteArray
598
- const buf = new Array(len)
599
- let i = 0
600
- const fullBytes = len >> 1
601
-
602
- for (let j = 0; j < fullBytes; ++j) {
603
- const sb = numeric[seqStart + j]!
604
- buf[i++] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
605
- buf[i++] = SEQRET_DECODER[sb & 0x0f]!
606
- }
607
-
608
- if (i < len) {
609
- const sb = numeric[seqStart + fullBytes]!
610
- buf[i] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
693
+ const ba = this._byteArray
694
+ const p = this.seqStart
695
+ const nPairs = len >> 1
696
+ let seq: string
697
+ if (len < SEQ_DECODER_THRESHOLD) {
698
+ seq = ''
699
+ for (let j = 0; j < nPairs; j++) {
700
+ seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
701
+ }
702
+ if (len & 1) {
703
+ seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
704
+ }
705
+ } else {
706
+ // round up to an even length so the Uint16Array view spans every pair; a
707
+ // trailing odd base is written as a single byte and trimmed on decode
708
+ const out = new Uint8Array((len + 1) & ~1)
709
+ const pairs = new Uint16Array(out.buffer)
710
+ for (let j = 0; j < nPairs; j++) {
711
+ pairs[j] = SEQRET_PAIR_CODES[ba[p + j]!]!
712
+ }
713
+ if (len & 1) {
714
+ out[len - 1] = SEQRET_CODES[(ba[p + nPairs]! & 0xf0) >> 4]!
715
+ }
716
+ seq = textDecoder.decode(out.subarray(0, len))
611
717
  }
612
-
613
- return buf.join('')
718
+ return seq
614
719
  }
615
720
 
616
- // adapted from igv.js
617
- // uses precomputed lookup table indexed by flag bits + isize sign.
618
- // the BAM spec defines tlen as positive for the leftmost segment and
619
- // negative for the rightmost, so tlen > 0 reliably indicates which
620
- // read comes first without needing position-based correction
721
+ // Must come out identical from either mate, or the two halves of one normal
722
+ // pair render as different orientations. The leftmost mate is therefore picked
723
+ // by a total order on (refId, pos) that both mates evaluate the same way, with
724
+ // a read1-first tie-break for equal loci.
725
+ //
726
+ // Deriving "leftmost" from template_length looks tempting — the spec makes
727
+ // tlen positive for the leftmost segment and negative for the rightmost — but
728
+ // aligners leave tlen at 0 whenever the insert size is unavailable, which
729
+ // includes every cross-reference pair. Both mates then read as "not leftmost"
730
+ // and disagree with each other.
621
731
  // (see also: gmod/cram-js src/cramFile/record.ts getPairOrientation)
622
732
  get pair_orientation() {
623
733
  const f = this.flags
624
- // unmapped (0x4) or mate unmapped (0x8) -> undefined
625
- if (f & 0xc || this.ref_id !== this.next_refid) {
734
+ if (!(f & Constants.BAM_FPAIRED)) {
626
735
  return undefined
627
736
  }
628
- return PAIR_ORIENTATION_TABLE[
629
- ((f >> 4) & 0xf) | (this.template_length > 0 ? 16 : 0)
630
- ]
737
+ const refId = this.ref_id
738
+ const mateRefId = this.next_refid
739
+ const pos = this.start
740
+ const matePos = this.next_pos
741
+ const selfIsLeft =
742
+ refId !== mateRefId
743
+ ? refId < mateRefId
744
+ : pos !== matePos
745
+ ? pos < matePos
746
+ : !!(f & Constants.BAM_FREAD1)
747
+ return PAIR_ORIENTATION_TABLE[((f >> 4) & 0x7) | (selfIsLeft ? 8 : 0)]
631
748
  }
632
749
 
633
750
  get bin_mq_nl() {
@@ -658,7 +775,7 @@ export default class BamRecord {
658
775
  if (idx < this.seq_length) {
659
776
  const sb = this._byteArray[this.seqStart + (idx >> 1)]!
660
777
 
661
- return idx % 2 === 0
778
+ return (idx & 1) === 0
662
779
  ? SEQRET_DECODER[(sb & 0xf0) >> 4]!
663
780
  : SEQRET_DECODER[sb & 0x0f]!
664
781
  } else {
package/src/util.ts CHANGED
@@ -3,23 +3,11 @@ import { longFromBytesToUnsigned } from './long.ts'
3
3
 
4
4
  import type { Offset, VirtualOffset } from './virtualOffset.ts'
5
5
 
6
- export interface TagFilter {
7
- tag: string
8
- value?: string
9
- }
10
-
11
- export interface FilterBy {
12
- flagInclude?: number
13
- flagExclude?: number
14
- tagFilter?: TagFilter
15
- }
16
-
17
6
  export interface BamOpts {
18
7
  viewAsPairs?: boolean
19
8
  pairAcrossChr?: boolean
20
9
  maxInsertSize?: number
21
10
  signal?: AbortSignal
22
- filterBy?: FilterBy
23
11
  /**
24
12
  * Called as the BGZF blocks covering the query are fetched, with cumulative
25
13
  * downloaded bytes and the total to fetch. Reported at block granularity (one
@@ -126,7 +114,10 @@ export function parsePseudoBin(bytes: Uint8Array, offset: number) {
126
114
  // maxv.blockPosition is an upper bound on where that final block ends — always at
127
115
  // least the true block end, so the clamped fetch still contains the whole block.
128
116
  // Shrinks both the byte estimate and the actual fetch with no extra I/O.
129
- export function clampChunkEnds(chunks: Chunk[], extraBoundaries: number[] = []) {
117
+ export function clampChunkEnds(
118
+ chunks: Chunk[],
119
+ extraBoundaries: number[] = [],
120
+ ) {
130
121
  const boundaries = [...extraBoundaries]
131
122
  for (const c of chunks) {
132
123
  boundaries.push(c.minv.blockPosition, c.maxv.blockPosition)
@@ -163,9 +154,15 @@ export function parseRefSeqs(
163
154
  if (start + 4 > uncba.length) {
164
155
  return undefined
165
156
  }
166
- const dataView = new DataView(uncba.buffer)
157
+ const dataView = new DataView(
158
+ uncba.buffer,
159
+ uncba.byteOffset,
160
+ uncba.byteLength,
161
+ )
167
162
  const nRef = dataView.getInt32(start, true)
168
- const chrToIndex: Record<string, number> = {}
163
+ // null prototype: ref names come from the file, so a contig named
164
+ // "constructor" must not resolve to Object.prototype's
165
+ const chrToIndex: Record<string, number> = Object.create(null)
169
166
  const indexToChr: { refName: string; length: number }[] = []
170
167
  const decoder = new TextDecoder('utf8')
171
168
 
@@ -209,7 +206,7 @@ export function parseNameBytes(
209
206
  let currRefId = 0
210
207
  let currNameStart = 0
211
208
  const refIdToName: string[] = []
212
- const refNameToId: Record<string, number> = {}
209
+ const refNameToId: Record<string, number> = Object.create(null)
213
210
  for (let i = 0; i < namesBytes.length; i++) {
214
211
  if (!namesBytes[i]) {
215
212
  if (currNameStart < i) {
@@ -240,53 +237,6 @@ export function concatUint8Array(args: Uint8Array[]) {
240
237
  return mergedArray
241
238
  }
242
239
 
243
- export function filterReadFlag(
244
- flags: number,
245
- flagInclude: number,
246
- flagExclude: number,
247
- ) {
248
- return (flags & flagInclude) !== flagInclude || (flags & flagExclude) !== 0
249
- }
250
-
251
- export function filterTagValue(readVal: unknown, filterVal?: string) {
252
- return filterVal === '*'
253
- ? readVal === undefined
254
- : `${readVal}` !== `${filterVal}`
255
- }
256
-
257
- export function filterCacheKey(filterBy?: FilterBy) {
258
- if (!filterBy) {
259
- return ''
260
- }
261
- const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
262
- const tagPart = tagFilter ? `:${tagFilter.tag}=${tagFilter.value ?? '*'}` : ''
263
- return `:f${flagInclude}x${flagExclude}${tagPart}`
264
- }
265
-
266
- interface Filterable {
267
- flags: number
268
- tags: Record<string, unknown>
269
- }
270
-
271
- // Apply flagInclude/flagExclude/tagFilter to a list of records.
272
- export function applyFilters<T extends Filterable>(
273
- records: T[],
274
- filterBy: FilterBy,
275
- ): T[] {
276
- const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
277
- const out: T[] = []
278
- for (let i = 0, l = records.length; i < l; i++) {
279
- const r = records[i]!
280
- if (
281
- !filterReadFlag(r.flags, flagInclude, flagExclude) &&
282
- !(tagFilter && filterTagValue(r.tags[tagFilter.tag], tagFilter.value))
283
- ) {
284
- out.push(r)
285
- }
286
- }
287
- return out
288
- }
289
-
290
240
  interface Positioned {
291
241
  ref_id: number
292
242
  start: number