@gmod/bam 7.3.3 → 7.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/dist/bai.js +10 -7
  2. package/dist/bai.js.map +1 -1
  3. package/dist/bamFile.d.ts +25 -6
  4. package/dist/bamFile.js +91 -33
  5. package/dist/bamFile.js.map +1 -1
  6. package/dist/csi.js +2 -2
  7. package/dist/csi.js.map +1 -1
  8. package/dist/htsget.js +2 -2
  9. package/dist/htsget.js.map +1 -1
  10. package/dist/index.d.ts +1 -1
  11. package/dist/index.js +2 -1
  12. package/dist/index.js.map +1 -1
  13. package/dist/long.d.ts +0 -2
  14. package/dist/long.js +2 -4
  15. package/dist/long.js.map +1 -1
  16. package/dist/record.d.ts +3 -2
  17. package/dist/record.js +198 -178
  18. package/dist/record.js.map +1 -1
  19. package/dist/util.d.ts +1 -1
  20. package/dist/util.js +11 -12
  21. package/dist/util.js.map +1 -1
  22. package/dist/virtualOffset.d.ts +1 -1
  23. package/dist/virtualOffset.js +1 -4
  24. package/dist/virtualOffset.js.map +1 -1
  25. package/esm/bai.js +10 -7
  26. package/esm/bai.js.map +1 -1
  27. package/esm/bamFile.d.ts +25 -6
  28. package/esm/bamFile.js +91 -33
  29. package/esm/bamFile.js.map +1 -1
  30. package/esm/csi.js +2 -2
  31. package/esm/csi.js.map +1 -1
  32. package/esm/htsget.js +2 -2
  33. package/esm/htsget.js.map +1 -1
  34. package/esm/index.d.ts +1 -1
  35. package/esm/index.js +1 -1
  36. package/esm/index.js.map +1 -1
  37. package/esm/long.d.ts +0 -2
  38. package/esm/long.js +1 -2
  39. package/esm/long.js.map +1 -1
  40. package/esm/record.d.ts +3 -2
  41. package/esm/record.js +198 -178
  42. package/esm/record.js.map +1 -1
  43. package/esm/util.d.ts +1 -1
  44. package/esm/util.js +11 -11
  45. package/esm/util.js.map +1 -1
  46. package/esm/virtualOffset.d.ts +1 -1
  47. package/esm/virtualOffset.js +1 -4
  48. package/esm/virtualOffset.js.map +1 -1
  49. package/package.json +2 -1
  50. package/src/bai.ts +15 -8
  51. package/src/bamFile.ts +117 -40
  52. package/src/csi.ts +10 -2
  53. package/src/htsget.ts +6 -2
  54. package/src/index.ts +1 -1
  55. package/src/long.ts +1 -2
  56. package/src/record.ts +231 -192
  57. package/src/util.ts +24 -14
  58. package/src/virtualOffset.ts +1 -5
package/src/record.ts CHANGED
@@ -1,16 +1,45 @@
1
1
  import { CIGAR_REF_SKIP, CIGAR_SOFT_CLIP } from './cigar.ts'
2
2
  import Constants from './constants.ts'
3
3
 
4
- const SEQRET_DECODER = '=ACMGRSVTWYHKDBN'.split('')
4
+ const SEQRET = '=ACMGRSVTWYHKDBN'
5
+ const SEQRET_DECODER = SEQRET.split('')
6
+ const SEQRET_CODES = Uint8Array.from(SEQRET, c => c.charCodeAt(0))
7
+
8
+ // Both bases of a SEQ byte, precomputed for all 256 bytes so decoding advances a
9
+ // byte at a time. Two forms because `seq` has two strategies (see below): packed
10
+ // ASCII codes for a single Uint16Array store, and the 2-char string to append.
11
+ // Byte order within the u16 depends on host endianness.
12
+ const LITTLE_ENDIAN = new Uint16Array(Uint8Array.of(1, 0).buffer)[0] === 1
13
+ const SEQRET_PAIR_CODES = new Uint16Array(256)
14
+ const SEQRET_PAIR_STRINGS = new Array<string>(256)
15
+ for (let hi = 0; hi < 16; hi++) {
16
+ for (let lo = 0; lo < 16; lo++) {
17
+ const h = SEQRET_CODES[hi]!
18
+ const l = SEQRET_CODES[lo]!
19
+ SEQRET_PAIR_CODES[(hi << 4) | lo] = LITTLE_ENDIAN
20
+ ? h | (l << 8)
21
+ : (h << 8) | l
22
+ SEQRET_PAIR_STRINGS[(hi << 4) | lo] =
23
+ SEQRET_DECODER[hi]! + SEQRET_DECODER[lo]!
24
+ }
25
+ }
5
26
 
6
- // precomputed pair orientation strings indexed by ((flags >> 4) & 0xF) | (isize > 0 ? 16 : 0)
7
- // bits 0-3 encode flag bits 0x10(reverse),0x20(mate reverse),0x40(read1),0x80(read2)
8
- // bit 4 encodes whether isize > 0
27
+ // Read length at which building a byte buffer and calling TextDecoder once
28
+ // overtakes plain string concatenation. Below it concat wins by 2-4x (rope
29
+ // building is cheap and the decode has fixed overhead); above it the decoder
30
+ // wins by 5-8x. Measured crossover is ~300bp, i.e. just past typical Illumina
31
+ // read lengths.
32
+ const SEQ_DECODER_THRESHOLD = 300
33
+
34
+ // Precomputed pair orientation strings, indexed by
35
+ // ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
36
+ // bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
37
+ // bit 3 is whether this read is the leftmost of the pair. The read2 flag (0x80)
38
+ // is deliberately not consulted — "not read1" is what decides the numbering, so
39
+ // a record with neither flag set reads the same as read2.
9
40
  // prettier-ignore
10
41
  const PAIR_ORIENTATION_TABLE = [
11
- 'F F ','F R ','R F ','R R ','F2F1','F2R1','R2F1','R2R1',
12
42
  'F1F2','F1R2','R1F2','R1R2','F2F1','F2R1','R2F1','R2R1',
13
- 'F F ','R F ','F R ','R R ','F1F2','R1F2','F1R2','R1R2',
14
43
  'F2F1','R2F1','F2R1','R2R1','F1F2','R1F2','F1R2','R1R2',
15
44
  ]
16
45
  const ASCII_CIGAR_CODES = [
@@ -23,6 +52,19 @@ const textDecoder = new TextDecoder()
23
52
  // Binary: 0b111001101 = 0x1CD
24
53
  const CIGAR_CONSUMES_REF_MASK = 0x1cd
25
54
 
55
+ // A CIGAR as packed op words. Either a view over (or copy of) the record's own
56
+ // CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
57
+ // too, since a writer may encode CG as B:i rather than the usual B:I.
58
+ export type NumericCigar = Uint32Array | Int32Array | number[]
59
+
60
+ function isNumericCigar(value: unknown): value is NumericCigar {
61
+ return (
62
+ value instanceof Uint32Array ||
63
+ value instanceof Int32Array ||
64
+ Array.isArray(value)
65
+ )
66
+ }
67
+
26
68
  export interface Bytes {
27
69
  start: number
28
70
  end: number
@@ -42,7 +84,11 @@ type BArrayValue =
42
84
  // Decode a 'B' (array) tag value starting at `p` (the byte after type+subtype+
43
85
  // count). When the data is naturally aligned we return a typed-array view over
44
86
  // the underlying buffer (zero-copy); otherwise we copy element-by-element since
45
- // typed-array views require alignment. Shared by getTag and the full-tag parse.
87
+ // typed-array views require alignment. The copy target is a plain array, not a
88
+ // typed one: benchmarking the unaligned path found number[] fills faster below
89
+ // ~10k elements and reads at least as fast at every size, so the only thing a
90
+ // typed copy would buy is half the retained bytes. Shared by getTag and the
91
+ // full-tag parse.
46
92
  function decodeBArrayTag(
47
93
  ba: Uint8Array,
48
94
  dataView: DataView,
@@ -127,6 +173,97 @@ function bArrayByteLength(Btype: number, limit: number) {
127
173
  }
128
174
  }
129
175
 
176
+ // Cursor position just past a tag's value — i.e. the start of the next tag —
177
+ // given the value starts at `p`. For null-terminated Z/H this steps over the
178
+ // terminator. Returns 0 for an unknown type, whose value width is unknowable so
179
+ // the caller must stop. Shared by _findTag and _computeTags so the two loops
180
+ // can't drift in how they walk the tag layout. (0 is an impossible real end:
181
+ // `p` is always past the fixed record prefix.)
182
+ function tagValueEnd(
183
+ ba: Uint8Array,
184
+ dataView: DataView,
185
+ type: number,
186
+ p: number,
187
+ blockEnd: number,
188
+ ) {
189
+ switch (type) {
190
+ case 0x41: // 'A'
191
+ case 0x63: // 'c'
192
+ case 0x43: // 'C'
193
+ return p + 1
194
+ case 0x73: // 's'
195
+ case 0x53: // 'S'
196
+ return p + 2
197
+ case 0x69: // 'i'
198
+ case 0x49: // 'I'
199
+ case 0x66: {
200
+ // 'f'
201
+ return p + 4
202
+ }
203
+ case 0x5a: // 'Z'
204
+ case 0x48: {
205
+ // 'H'
206
+ let q = p
207
+ while (q < blockEnd && ba[q] !== 0) {
208
+ q++
209
+ }
210
+ return q + 1 // step past the null terminator
211
+ }
212
+ case 0x42: {
213
+ // 'B'
214
+ const Btype = ba[p]!
215
+ const limit = dataView.getInt32(p + 1, true)
216
+ return p + 5 + bArrayByteLength(Btype, limit)
217
+ }
218
+ default:
219
+ console.error('Unknown BAM tag type', type)
220
+ return 0
221
+ }
222
+ }
223
+
224
+ // Decode the value of a tag of `type` whose bytes start at `p`; `end` is the
225
+ // next-tag cursor from tagValueEnd (used only to bound Z/H strings, whose bytes
226
+ // run up to the null terminator at end-1). When `raw`, Z/H return the undecoded
227
+ // byte subarray. Assumes `type` is a known scalar/string/B type.
228
+ function decodeTagValue(
229
+ ba: Uint8Array,
230
+ dataView: DataView,
231
+ type: number,
232
+ p: number,
233
+ end: number,
234
+ raw: boolean,
235
+ ): unknown {
236
+ switch (type) {
237
+ case 0x41: // 'A'
238
+ return String.fromCharCode(ba[p]!)
239
+ case 0x69: // 'i'
240
+ return dataView.getInt32(p, true)
241
+ case 0x49: // 'I'
242
+ return dataView.getUint32(p, true)
243
+ case 0x63: // 'c'
244
+ return dataView.getInt8(p)
245
+ case 0x43: // 'C'
246
+ return dataView.getUint8(p)
247
+ case 0x73: // 's'
248
+ return dataView.getInt16(p, true)
249
+ case 0x53: // 'S'
250
+ return dataView.getUint16(p, true)
251
+ case 0x66: // 'f'
252
+ return dataView.getFloat32(p, true)
253
+ case 0x5a: // 'Z'
254
+ case 0x48: // 'H'
255
+ return raw
256
+ ? ba.subarray(p, end - 1)
257
+ : textDecoder.decode(ba.subarray(p, end - 1))
258
+ default: {
259
+ // 'B'
260
+ const Btype = ba[p]!
261
+ const limit = dataView.getInt32(p + 1, true)
262
+ return decodeBArrayTag(ba, dataView, p + 5, Btype, limit)
263
+ }
264
+ }
265
+ }
266
+
130
267
  export default class BamRecord {
131
268
  public fileOffset: number
132
269
  private _byteArray: Uint8Array
@@ -137,7 +274,7 @@ export default class BamRecord {
137
274
  private _cachedEnd?: number
138
275
  private _cachedTags?: Record<string, unknown>
139
276
  private _cachedLengthOnRef?: number
140
- private _cachedNumericCigar?: Uint32Array | number[]
277
+ private _cachedNumericCigar?: NumericCigar
141
278
  private _cachedNUMERIC_MD?: Uint8Array | null
142
279
  private _cachedSeqStart?: number
143
280
 
@@ -261,172 +398,49 @@ export default class BamRecord {
261
398
  private _findTag(tagName: string, raw: boolean) {
262
399
  const tag1 = tagName.charCodeAt(0)
263
400
  const tag2 = tagName.charCodeAt(1)
264
-
265
- let p = this.tagsStart
266
-
267
401
  const blockEnd = this._end
268
402
  const ba = this._byteArray
403
+ let p = this.tagsStart
269
404
  while (p < blockEnd) {
270
- const currentTag1 = ba[p]
271
- const currentTag2 = ba[p + 1]
272
- const type = ba[p + 2]
273
- p += 3
274
-
275
- const isMatch = currentTag1 === tag1 && currentTag2 === tag2
276
-
277
- switch (type) {
278
- case 0x41: // 'A'
279
- if (isMatch) {
280
- return String.fromCharCode(ba[p]!)
281
- }
282
- p += 1
283
- break
284
- case 0x69: // 'i'
285
- if (isMatch) {
286
- return this._dataView.getInt32(p, true)
287
- }
288
- p += 4
289
- break
290
- case 0x49: // 'I'
291
- if (isMatch) {
292
- return this._dataView.getUint32(p, true)
293
- }
294
- p += 4
295
- break
296
- case 0x63: // 'c'
297
- if (isMatch) {
298
- return this._dataView.getInt8(p)
299
- }
300
- p += 1
301
- break
302
- case 0x43: // 'C'
303
- if (isMatch) {
304
- return this._dataView.getUint8(p)
305
- }
306
- p += 1
307
- break
308
- case 0x73: // 's'
309
- if (isMatch) {
310
- return this._dataView.getInt16(p, true)
311
- }
312
- p += 2
313
- break
314
- case 0x53: // 'S'
315
- if (isMatch) {
316
- return this._dataView.getUint16(p, true)
317
- }
318
- p += 2
319
- break
320
- case 0x66: // 'f'
321
- if (isMatch) {
322
- return this._dataView.getFloat32(p, true)
323
- }
324
- p += 4
325
- break
326
- case 0x5a: // 'Z'
327
- case 0x48: {
328
- // 'H'
329
- const start = p
330
- while (p < blockEnd && ba[p] !== 0) {
331
- p++
332
- }
333
- if (isMatch) {
334
- return raw
335
- ? ba.subarray(start, p)
336
- : textDecoder.decode(ba.subarray(start, p))
337
- }
338
- p++ // advance past null terminator
339
- break
340
- }
341
- case 0x42: {
342
- // 'B'
343
- const Btype = ba[p++]!
344
- const limit = this._dataView.getInt32(p, true)
345
- p += 4
346
- if (isMatch) {
347
- return decodeBArrayTag(ba, this._dataView, p, Btype, limit)
348
- }
349
- p += bArrayByteLength(Btype, limit)
350
- break
351
- }
352
- default:
353
- if (type !== undefined) {
354
- console.error('Unknown BAM tag type', type)
355
- }
356
- break
405
+ const isMatch = ba[p] === tag1 && ba[p + 1] === tag2
406
+ const type = ba[p + 2]!
407
+ const valueStart = p + 3
408
+ const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
409
+ if (end === 0) {
410
+ break // unknown type: can't compute how far to advance
411
+ }
412
+ if (isMatch) {
413
+ return decodeTagValue(ba, this._dataView, type, valueStart, end, raw)
357
414
  }
415
+ p = end
358
416
  }
359
417
  return undefined
360
418
  }
361
419
 
362
420
  private _computeTags() {
363
- let p = this.tagsStart
364
-
365
421
  const blockEnd = this._end
366
422
  const ba = this._byteArray
367
- const tags: Record<string, unknown> = {}
423
+ // null prototype: tag names come from the file, so a read carrying a
424
+ // "constructor" or "toString" tag must not resolve to Object.prototype's
425
+ const tags: Record<string, unknown> = Object.create(null)
426
+ let p = this.tagsStart
368
427
  while (p < blockEnd) {
369
428
  const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
370
429
  const type = ba[p + 2]!
371
- p += 3
372
-
373
- switch (type) {
374
- case 0x41: // 'A'
375
- tags[tag] = String.fromCharCode(ba[p]!)
376
- p += 1
377
- break
378
- case 0x69: // 'i'
379
- tags[tag] = this._dataView.getInt32(p, true)
380
- p += 4
381
- break
382
- case 0x49: // 'I'
383
- tags[tag] = this._dataView.getUint32(p, true)
384
- p += 4
385
- break
386
- case 0x63: // 'c'
387
- tags[tag] = this._dataView.getInt8(p)
388
- p += 1
389
- break
390
- case 0x43: // 'C'
391
- tags[tag] = this._dataView.getUint8(p)
392
- p += 1
393
- break
394
- case 0x73: // 's'
395
- tags[tag] = this._dataView.getInt16(p, true)
396
- p += 2
397
- break
398
- case 0x53: // 'S'
399
- tags[tag] = this._dataView.getUint16(p, true)
400
- p += 2
401
- break
402
- case 0x66: // 'f'
403
- tags[tag] = this._dataView.getFloat32(p, true)
404
- p += 4
405
- break
406
- case 0x5a: // 'Z'
407
- case 0x48: {
408
- // 'H'
409
- const start = p
410
- while (p < blockEnd && ba[p] !== 0) {
411
- p++
412
- }
413
- tags[tag] = textDecoder.decode(ba.subarray(start, p))
414
- p++ // advance past null terminator
415
- break
416
- }
417
- case 0x42: {
418
- // 'B'
419
- const Btype = ba[p++]!
420
- const limit = this._dataView.getInt32(p, true)
421
- p += 4
422
- tags[tag] = decodeBArrayTag(ba, this._dataView, p, Btype, limit)
423
- p += bArrayByteLength(Btype, limit)
424
- break
425
- }
426
- default:
427
- console.error('Unknown BAM tag type', type)
428
- break
430
+ const valueStart = p + 3
431
+ const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
432
+ if (end === 0) {
433
+ break // unknown type: can't compute how far to advance
429
434
  }
435
+ tags[tag] = decodeTagValue(
436
+ ba,
437
+ this._dataView,
438
+ type,
439
+ valueStart,
440
+ end,
441
+ false,
442
+ )
443
+ p = end
430
444
  }
431
445
  return tags
432
446
  }
@@ -533,6 +547,8 @@ export default class BamRecord {
533
547
  absOffset,
534
548
  numCigarOps,
535
549
  )
550
+ // the view we need to sum is exactly what NUMERIC_CIGAR would build, so
551
+ // seed its cache rather than making it construct a second one
536
552
  this._cachedNumericCigar = cigarView
537
553
  let lref = 0
538
554
  for (let c = 0; c < numCigarOps; ++c) {
@@ -550,7 +566,7 @@ export default class BamRecord {
550
566
  return lref
551
567
  }
552
568
 
553
- private _computeNumericCigar(): Uint32Array | number[] {
569
+ private _computeNumericCigar(): NumericCigar {
554
570
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
555
571
  if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
556
572
  return new Uint32Array(0)
@@ -560,10 +576,10 @@ export default class BamRecord {
560
576
  const p = this.b0 + this.read_name_length
561
577
 
562
578
  if (this._isCGTagPattern(p, numCigarOps)) {
563
- return (
564
- (this.tags.CG as Uint32Array | number[] | undefined) ??
565
- new Uint32Array(0)
566
- )
579
+ // getTag, not this.tags: the real CIGAR lives in one tag, so there's no
580
+ // reason to decode every other tag on the record to reach it
581
+ const cg = this.getTag('CG')
582
+ return isNumericCigar(cg) ? cg : new Uint32Array(0)
567
583
  }
568
584
 
569
585
  const absOffset = this._byteArray.byteOffset + p
@@ -625,43 +641,66 @@ export default class BamRecord {
625
641
  return this._byteArray.subarray(p, p + this.num_seq_bytes)
626
642
  }
627
643
 
644
+ // Decode two bases per iteration off a 256-entry table. Building an array of
645
+ // 1-char strings and join()ing it — the obvious approach — is 3x slower at
646
+ // 100bp and 35x slower at 15kb.
628
647
  get seq() {
629
648
  const len = this.seq_length
630
- const seqStart = this.seqStart
631
- const numeric = this._byteArray
632
- const buf = new Array(len)
633
- let i = 0
634
- const fullBytes = len >> 1
635
-
636
- for (let j = 0; j < fullBytes; ++j) {
637
- const sb = numeric[seqStart + j]!
638
- buf[i++] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
639
- buf[i++] = SEQRET_DECODER[sb & 0x0f]!
640
- }
641
-
642
- if (i < len) {
643
- const sb = numeric[seqStart + fullBytes]!
644
- buf[i] = SEQRET_DECODER[(sb & 0xf0) >> 4]!
649
+ const ba = this._byteArray
650
+ const p = this.seqStart
651
+ const nPairs = len >> 1
652
+ let seq: string
653
+ if (len < SEQ_DECODER_THRESHOLD) {
654
+ seq = ''
655
+ for (let j = 0; j < nPairs; j++) {
656
+ seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
657
+ }
658
+ if (len & 1) {
659
+ seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
660
+ }
661
+ } else {
662
+ // round up to an even length so the Uint16Array view spans every pair; a
663
+ // trailing odd base is written as a single byte and trimmed on decode
664
+ const out = new Uint8Array((len + 1) & ~1)
665
+ const pairs = new Uint16Array(out.buffer)
666
+ for (let j = 0; j < nPairs; j++) {
667
+ pairs[j] = SEQRET_PAIR_CODES[ba[p + j]!]!
668
+ }
669
+ if (len & 1) {
670
+ out[len - 1] = SEQRET_CODES[(ba[p + nPairs]! & 0xf0) >> 4]!
671
+ }
672
+ seq = textDecoder.decode(out.subarray(0, len))
645
673
  }
646
-
647
- return buf.join('')
674
+ return seq
648
675
  }
649
676
 
650
- // adapted from igv.js
651
- // uses precomputed lookup table indexed by flag bits + isize sign.
652
- // the BAM spec defines tlen as positive for the leftmost segment and
653
- // negative for the rightmost, so tlen > 0 reliably indicates which
654
- // read comes first without needing position-based correction
677
+ // Must come out identical from either mate, or the two halves of one normal
678
+ // pair render as different orientations. The leftmost mate is therefore picked
679
+ // by a total order on (refId, pos) that both mates evaluate the same way, with
680
+ // a read1-first tie-break for equal loci.
681
+ //
682
+ // Deriving "leftmost" from template_length looks tempting — the spec makes
683
+ // tlen positive for the leftmost segment and negative for the rightmost — but
684
+ // aligners leave tlen at 0 whenever the insert size is unavailable, which
685
+ // includes every cross-reference pair. Both mates then read as "not leftmost"
686
+ // and disagree with each other.
655
687
  // (see also: gmod/cram-js src/cramFile/record.ts getPairOrientation)
656
688
  get pair_orientation() {
657
689
  const f = this.flags
658
- // unmapped (0x4) or mate unmapped (0x8) -> undefined
659
- if (f & 0xc || this.ref_id !== this.next_refid) {
690
+ if (!(f & Constants.BAM_FPAIRED)) {
660
691
  return undefined
661
692
  }
662
- return PAIR_ORIENTATION_TABLE[
663
- ((f >> 4) & 0xf) | (this.template_length > 0 ? 16 : 0)
664
- ]
693
+ const refId = this.ref_id
694
+ const mateRefId = this.next_refid
695
+ const pos = this.start
696
+ const matePos = this.next_pos
697
+ const selfIsLeft =
698
+ refId !== mateRefId
699
+ ? refId < mateRefId
700
+ : pos !== matePos
701
+ ? pos < matePos
702
+ : !!(f & Constants.BAM_FREAD1)
703
+ return PAIR_ORIENTATION_TABLE[((f >> 4) & 0x7) | (selfIsLeft ? 8 : 0)]
665
704
  }
666
705
 
667
706
  get bin_mq_nl() {
@@ -692,7 +731,7 @@ export default class BamRecord {
692
731
  if (idx < this.seq_length) {
693
732
  const sb = this._byteArray[this.seqStart + (idx >> 1)]!
694
733
 
695
- return idx % 2 === 0
734
+ return (idx & 1) === 0
696
735
  ? SEQRET_DECODER[(sb & 0xf0) >> 4]!
697
736
  : SEQRET_DECODER[sb & 0x0f]!
698
737
  } else {
package/src/util.ts CHANGED
@@ -126,7 +126,10 @@ export function parsePseudoBin(bytes: Uint8Array, offset: number) {
126
126
  // maxv.blockPosition is an upper bound on where that final block ends — always at
127
127
  // least the true block end, so the clamped fetch still contains the whole block.
128
128
  // Shrinks both the byte estimate and the actual fetch with no extra I/O.
129
- export function clampChunkEnds(chunks: Chunk[], extraBoundaries: number[] = []) {
129
+ export function clampChunkEnds(
130
+ chunks: Chunk[],
131
+ extraBoundaries: number[] = [],
132
+ ) {
130
133
  const boundaries = [...extraBoundaries]
131
134
  for (const c of chunks) {
132
135
  boundaries.push(c.minv.blockPosition, c.maxv.blockPosition)
@@ -163,9 +166,15 @@ export function parseRefSeqs(
163
166
  if (start + 4 > uncba.length) {
164
167
  return undefined
165
168
  }
166
- const dataView = new DataView(uncba.buffer)
169
+ const dataView = new DataView(
170
+ uncba.buffer,
171
+ uncba.byteOffset,
172
+ uncba.byteLength,
173
+ )
167
174
  const nRef = dataView.getInt32(start, true)
168
- const chrToIndex: Record<string, number> = {}
175
+ // null prototype: ref names come from the file, so a contig named
176
+ // "constructor" must not resolve to Object.prototype's
177
+ const chrToIndex: Record<string, number> = Object.create(null)
169
178
  const indexToChr: { refName: string; length: number }[] = []
170
179
  const decoder = new TextDecoder('utf8')
171
180
 
@@ -209,7 +218,7 @@ export function parseNameBytes(
209
218
  let currRefId = 0
210
219
  let currNameStart = 0
211
220
  const refIdToName: string[] = []
212
- const refNameToId: Record<string, number> = {}
221
+ const refNameToId: Record<string, number> = Object.create(null)
213
222
  for (let i = 0; i < namesBytes.length; i++) {
214
223
  if (!namesBytes[i]) {
215
224
  if (currNameStart < i) {
@@ -254,18 +263,19 @@ export function filterTagValue(readVal: unknown, filterVal?: string) {
254
263
  : `${readVal}` !== `${filterVal}`
255
264
  }
256
265
 
257
- export function filterCacheKey(filterBy?: FilterBy) {
258
- if (!filterBy) {
259
- return ''
260
- }
261
- const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
262
- const tagPart = tagFilter ? `:${tagFilter.tag}=${tagFilter.value ?? '*'}` : ''
263
- return `:f${flagInclude}x${flagExclude}${tagPart}`
264
- }
265
-
266
266
  interface Filterable {
267
267
  flags: number
268
268
  tags: Record<string, unknown>
269
+ // BamRecord decodes one tag by walking the tag block, without building the
270
+ // whole tags object. Optional because a custom recordClass need not have it.
271
+ getTag?(tag: string): unknown
272
+ }
273
+
274
+ // Read a single tag, preferring the targeted accessor. Reaching for `tags`
275
+ // instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
276
+ // ~10 per read) just to test one, which measured 2.6x the cost of getTag.
277
+ function readTag(record: Filterable, tag: string) {
278
+ return record.getTag ? record.getTag(tag) : record.tags[tag]
269
279
  }
270
280
 
271
281
  // Apply flagInclude/flagExclude/tagFilter to a list of records.
@@ -279,7 +289,7 @@ export function applyFilters<T extends Filterable>(
279
289
  const r = records[i]!
280
290
  if (
281
291
  !filterReadFlag(r.flags, flagInclude, flagExclude) &&
282
- !(tagFilter && filterTagValue(r.tags[tagFilter.tag], tagFilter.value))
292
+ !(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
283
293
  ) {
284
294
  out.push(r)
285
295
  }
@@ -23,11 +23,7 @@ export class VirtualOffset {
23
23
  )
24
24
  }
25
25
  }
26
- export function fromBytes(bytes: Uint8Array, offset = 0, bigendian = false) {
27
- if (bigendian) {
28
- throw new Error('big-endian virtual file offsets not implemented')
29
- }
30
-
26
+ export function fromBytes(bytes: Uint8Array, offset = 0) {
31
27
  return new VirtualOffset(
32
28
  bytes[offset + 7]! * 0x10000000000 +
33
29
  bytes[offset + 6]! * 0x100000000 +