@gmod/bam 7.8.1 → 7.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/bai.ts CHANGED
@@ -33,12 +33,20 @@ const BAI_MAGIC = 21578050 // BAI\1
33
33
  // https://github.com/samtools/hts-specs/blob/master/SAMv1.pdf
34
34
  const BAI_LINEAR_SHIFT = 14
35
35
  const BAI_LINEAR_INTERVAL = 1 << BAI_LINEAR_SHIFT // 16384
36
+ const BAI_DEPTH = 5
37
+ // Highest coordinate the scheme addresses: the deepest level's bins are
38
+ // BAI_LINEAR_INTERVAL wide and there are 8^BAI_DEPTH of them.
39
+ const BAI_MAX_POS = 2 ** (BAI_LINEAR_SHIFT + BAI_DEPTH * 3) // 2^29
36
40
 
37
41
  function roundDown(n: number, multiple: number) {
38
42
  return n - (n % multiple)
39
43
  }
44
+ // Note the `rem === 0` case: without it a coordinate already on a window
45
+ // boundary rounds up a whole extra window, which is enough to push indexCov's
46
+ // range past the end of the linear index.
40
47
  function roundUp(n: number, multiple: number) {
41
- return n - (n % multiple) + multiple
48
+ const rem = n % multiple
49
+ return rem === 0 ? n : n - rem + multiple
42
50
  }
43
51
 
44
52
  export interface IndexCovEntry {
@@ -50,6 +58,19 @@ export interface IndexCovEntry {
50
58
  // Compute bin ranges that overlap [beg, end). Each level's first-bin offset
51
59
  // is (8^L - 1) / 7. See SAMv1.pdf §5.1.1 for the binning derivation.
52
60
  function reg2bins(beg: number, end: number) {
61
+ // Clamp to what the scheme can address, the way CSI's reg2bins clamps to
62
+ // its own. The shifts below are the `>>` operator, so a coordinate past
63
+ // 2^31 wraps to a negative bin number and every level yields an empty
64
+ // range: `getRecordsForRange(chr, 0, 2**32)` — a caller asking for a whole
65
+ // reference without knowing its length — came back with NO records at all
66
+ // rather than all of them. Clamping is also what keeps a merely-large end
67
+ // from walking ~130k absent bin numbers before finding the same chunks.
68
+ if (beg > BAI_MAX_POS) {
69
+ beg = BAI_MAX_POS
70
+ }
71
+ if (end > BAI_MAX_POS) {
72
+ end = BAI_MAX_POS
73
+ }
53
74
  end -= 1
54
75
  return [
55
76
  [0, 0],
@@ -76,8 +97,7 @@ export default class BAI extends IndexFile<BaiParsed> {
76
97
  }
77
98
 
78
99
  const refCount = dataView.getInt32(4, true)
79
- const depth = 5
80
- const binLimit = ((1 << ((depth + 1) * 3)) - 1) / 7
100
+ const binLimit = ((1 << ((BAI_DEPTH + 1) * 3)) - 1) / 7
81
101
 
82
102
  // read the indexes for each reference sequence
83
103
  let curr = 8
@@ -100,11 +120,9 @@ export default class BAI extends IndexFile<BaiParsed> {
100
120
  throw new Error('bai index contains too many bins, please use CSI')
101
121
  } else {
102
122
  const chunkCount = dataView.getInt32(curr, true)
103
- curr += 4
104
- for (let k = 0; k < chunkCount; k++) {
105
- curr += 8
106
- curr += 8
107
- }
123
+ // 16 bytes per chunk (two virtual offsets); the first pass only
124
+ // needs to step over them. Same shape as csi.ts's first pass.
125
+ curr += 4 + 16 * chunkCount
108
126
  }
109
127
  }
110
128
 
@@ -199,7 +217,6 @@ export default class BAI extends IndexFile<BaiParsed> {
199
217
  opts?: BaseOpts,
200
218
  ): Promise<IndexCovEntry[]> {
201
219
  const v = BAI_LINEAR_INTERVAL
202
- const range = start !== undefined
203
220
  const indexData = await this.parse(opts)
204
221
  const seqIdx = indexData.indices(seqId)
205
222
 
@@ -211,15 +228,24 @@ export default class BAI extends IndexFile<BaiParsed> {
211
228
  if (nintv === 0) {
212
229
  return []
213
230
  }
214
- const e = end === undefined ? (nintv - 1) * v : roundUp(end, v)
215
- const s = start === undefined ? 0 : roundDown(start, v)
216
- const depths: IndexCovEntry[] = range
217
- ? new Array((e - s) / v)
218
- : new Array(nintv - 1)
219
- const totalSize = linearBlockPositions[nintv - 1]!
220
- if (e > (nintv - 1) * v) {
221
- throw new Error('query outside of range of linear index')
231
+ // The linear index describes [0, indexEnd): each window's score is the gap
232
+ // to the NEXT entry, so the final entry is a boundary rather than a window
233
+ // of its own. Both ends are clamped to it instead of throwing, so a range
234
+ // query returns the part of the reference the index covers — asking for the
235
+ // whole reference by its length (`indexCov(ref, 0, ctgLength)`, the obvious
236
+ // call, and one where end lands in the last window) used to throw "query
237
+ // outside of range of linear index" while `indexCov(ref)` answered fine.
238
+ const indexEnd = (nintv - 1) * v
239
+ const s =
240
+ start === undefined
241
+ ? 0
242
+ : Math.min(Math.max(roundDown(start, v), 0), indexEnd)
243
+ const e = end === undefined ? indexEnd : Math.min(roundUp(end, v), indexEnd)
244
+ if (e <= s) {
245
+ return []
222
246
  }
247
+ const depths: IndexCovEntry[] = new Array((e - s) / v)
248
+ const totalSize = linearBlockPositions[nintv - 1]!
223
249
  // Scale the block-delta into a read count as we go, rather than building the
224
250
  // entries and then rebuilding every one of them to apply the scale. Keep the
225
251
  // multiply-then-divide order: hoisting lineCount/totalSize into a factor
package/src/bamFile.ts CHANGED
@@ -37,10 +37,13 @@ export const BAM_MAGIC = 21840194
37
37
 
38
38
  const blockLen = 1 << 16
39
39
 
40
- // Ceiling on the header read. A million contigs is roughly 10MB of compressed
41
- // @SQ lines and ref-seq table, so anything past this is a corrupt header
42
- // claiming a huge n_ref rather than a real one, and growing the read further
43
- // just downloads the file.
40
+ // Ceiling on GROWING the header read. A million contigs is roughly 10MB of
41
+ // compressed @SQ lines and ref-seq table, so a read that has doubled past this
42
+ // is chasing a corrupt header claiming a huge n_ref rather than a real one, and
43
+ // growing further just downloads the file. It does not cap the FIRST read: that
44
+ // length comes from the index's firstDataLine, which is a real offset rather
45
+ // than a guess, so a header that genuinely runs past this still gets its one
46
+ // exact read. See getHeaderPre.
44
47
  const maxHeaderReadLen = 32 * 1024 * 1024
45
48
 
46
49
  function resolveFilehandle(
@@ -118,12 +121,24 @@ export const DEFAULT_MAX_CACHE_BYTES = 100 * 1024 * 1024
118
121
  const MAX_CONCURRENT_CHUNK_READS = 6
119
122
 
120
123
  class ChunkFeatureCache<T> {
121
- public maxBytes: number
124
+ private _maxBytes: number
122
125
  private entries = new Map<string, ChunkEntry<T>>()
123
126
  private bytes = 0
124
127
 
125
128
  constructor(maxBytes: number) {
126
- this.maxBytes = maxBytes
129
+ this._maxBytes = maxBytes
130
+ }
131
+
132
+ get maxBytes() {
133
+ return this._maxBytes
134
+ }
135
+
136
+ // Accessor rather than a plain field so lowering the budget frees memory now.
137
+ // As a field, a caller trimming the cache under memory pressure got nothing
138
+ // back until the next chunk read happened to call set().
139
+ set maxBytes(maxBytes: number) {
140
+ this._maxBytes = maxBytes
141
+ this.evict()
127
142
  }
128
143
 
129
144
  get size() {
@@ -148,11 +163,15 @@ class ChunkFeatureCache<T> {
148
163
  this.delete(key)
149
164
  this.entries.set(key, entry)
150
165
  this.bytes += entry.bytes
151
- // Evict from the least-recently-used end. The size > 1 guard means a single
152
- // chunk larger than the whole budget is still kept: the caller needs it for
153
- // the query in flight, and dropping it would only force a re-decompress.
166
+ this.evict()
167
+ }
168
+
169
+ // Evict from the least-recently-used end. The size > 1 guard means a single
170
+ // chunk larger than the whole budget is still kept: the caller needs it for
171
+ // the query in flight, and dropping it would only force a re-decompress.
172
+ private evict() {
154
173
  const lru = this.entries.keys()
155
- while (this.bytes > this.maxBytes && this.entries.size > 1) {
174
+ while (this.bytes > this._maxBytes && this.entries.size > 1) {
156
175
  this.delete(lru.next().value!)
157
176
  }
158
177
  }
@@ -285,16 +304,21 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
285
304
  ? blockLen
286
305
  : indexData.firstDataLine.blockPosition + blockLen
287
306
 
307
+ // do/while, so the index-derived length is always read once and only the
308
+ // doubling is bounded. Testing readLen before the first read instead meant
309
+ // a BAM whose header genuinely exceeds maxHeaderReadLen — millions of
310
+ // contigs — was rejected without a single byte being fetched, and reported
311
+ // as 'Insufficient data for reference sequences' when the data was there.
288
312
  let samHeader
289
- let atEof = false
290
- while (samHeader === undefined && !atEof && readLen <= maxHeaderReadLen) {
313
+ let atEof: boolean
314
+ do {
291
315
  const buffer = await this.bam.read(readLen, 0, { signal: opts.signal })
292
316
  // a short read means readLen ran past the end of the file, so there are
293
317
  // no more bytes to grow into
294
318
  atEof = buffer.length < readLen
295
319
  samHeader = this.applyHeader(await unzip(buffer))
296
320
  readLen *= 2
297
- }
321
+ } while (samHeader === undefined && !atEof && readLen <= maxHeaderReadLen)
298
322
  if (samHeader === undefined) {
299
323
  throw new Error('Insufficient data for reference sequences')
300
324
  }
@@ -499,9 +523,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
499
523
  // unbarriered pool exactly as before. That caps the cost of being wrong
500
524
  // at 0.92x-0.95x, against 0.82x-0.88x for barriering every wave.
501
525
  const batch = Math.min(MAX_CONCURRENT_CHUNK_READS, chunks.length)
502
- await Promise.all(
503
- Array.from({ length: batch }, (_, ci) => readOne(ci)),
504
- )
526
+ await Promise.all(Array.from({ length: batch }, (_, ci) => readOne(ci)))
505
527
  let stopped = false
506
528
  for (let ci = 0; ci < batch; ci++) {
507
529
  if (isPastQuery(featureLists[ci], chrId, max)) {
@@ -615,9 +637,13 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
615
637
  const mateRecs = [] as T[]
616
638
  for (let i = 0, l = features.length; i < l; i++) {
617
639
  const feature = features[i]!
640
+ // fileOffset first: it is a number already on the record, where
641
+ // `name` decodes a string per record. A mate chunk usually overlaps
642
+ // the query region, so this skips the decode for every record the
643
+ // caller is already holding.
618
644
  if (
619
- readNameCounts.get(feature.name) === 1 &&
620
- !readIds.has(feature.fileOffset)
645
+ !readIds.has(feature.fileOffset) &&
646
+ readNameCounts.get(feature.name) === 1
621
647
  ) {
622
648
  mateRecs.push(feature)
623
649
  }
@@ -691,9 +717,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
691
717
  blockEnd,
692
718
  hasCpositions
693
719
  ? cpositions[pos]! * (1 << 8) +
694
- (blockStart - dpositions[pos]!) +
695
- chunk.minv.dataPosition +
696
- 1
720
+ (blockStart - dpositions[pos]!) +
721
+ chunk.minv.dataPosition +
722
+ 1
697
723
  : crc32(ba.subarray(blockStart, blockEnd)) >>> 0,
698
724
  dataView,
699
725
  )
package/src/csi.ts CHANGED
@@ -96,6 +96,20 @@ export default class CSI extends IndexFile {
96
96
  this.maxBinNumber = ((1 << ((this.depth + 1) * 3)) - 1) / 7
97
97
  const maxBinNumber = this.maxBinNumber
98
98
  const auxLength = dataView.getInt32(12, true)
99
+ // A tabix-only branch, which is why parseAuxData and the parseNameBytes it
100
+ // calls show as uncovered. CSI is shared between `samtools index -c` and
101
+ // `tabix -C`, and the aux block is how a tabix index carries what a reader
102
+ // of bgzipped TEXT needs: which columns hold ref/start/end, the comment
103
+ // character, lines to skip, and the reference names — which a BAM takes
104
+ // from its own header instead. So a BAM .csi sets l_aux to 0 and never
105
+ // reaches here (checked: 0 of the 19 .csi fixtures carry one).
106
+ //
107
+ // Kept rather than deleted for two reasons. CSI is exported from
108
+ // index.ts, so pointing it at a tabix index is reachable (@gmod/tabix is
109
+ // the right tool, but this would half-work and then silently not).
110
+ // More importantly parseNameBytes is marked SYNC: with its tabix-js
111
+ // counterpart, and dropping one side of a deliberately-paired pair of
112
+ // implementations is exactly what that marker exists to prevent.
99
113
  const aux = auxLength >= 30 ? this.parseAuxData(bytes, 16) : undefined
100
114
  const refCount = dataView.getInt32(16 + auxLength, true)
101
115
 
@@ -200,6 +214,12 @@ export default class CSI extends IndexFile {
200
214
  for (; l <= this.depth; s -= 3, t += lshift(1, l * 3), l += 1) {
201
215
  const b = t + rshift(beg, s)
202
216
  const e = t + rshift(end, s)
217
+ // Unreachable as long as the clamp above stands, which is why it shows
218
+ // as uncovered: with end bounded to 2^(minShift + depth*3), level l
219
+ // spans at most 8^l - 1 bins, so the worst case is 8^depth - 1 + depth
220
+ // against a maxBinNumber of (8^(depth+1) - 1)/7, i.e. about 1.14 * 8^depth.
221
+ // Kept as a guard on the arithmetic rather than deleted — reg2bins is
222
+ // shared with tabix-js, whose callers reach it by other routes.
203
223
  if (e - b + bins.length > this.maxBinNumber) {
204
224
  throw new Error(
205
225
  `query ${beg}-${end} is too large for current binning scheme (shift ${this.minShift}, depth ${this.depth}), try a smaller query or a coarser index binning scheme`,
package/src/index.ts CHANGED
@@ -4,7 +4,12 @@ export { default as CSI } from './csi.ts'
4
4
  export { default as BamRecord } from './record.ts'
5
5
  export { default as HtsgetFile } from './htsget.ts'
6
6
 
7
- export type { Bytes } from './record.ts'
7
+ export type { NumericCigar } from './record.ts'
8
8
  export type { BamRecordClass, BamRecordLike } from './bamFile.ts'
9
+ // the options every query method takes, and the shapes they hand back. The
10
+ // package has no subpath exports, so a consumer typing a wrapper around
11
+ // getRecordsForRange/indexCov can only name these if they come out of here.
12
+ export type { BamOpts, BaseOpts } from './util.ts'
13
+ export type { IndexCovEntry } from './bai.ts'
9
14
  // for typing the HtsgetFile `fetch` option
10
15
  export type { Fetcher } from 'generic-filehandle2'
package/src/indexFile.ts CHANGED
@@ -33,8 +33,11 @@ export function memoizeByRefId<T>(
33
33
  ) {
34
34
  const cache = new QuickLRU<number, T>({ maxSize })
35
35
  return (refId: number) => {
36
- if (cache.has(refId)) {
37
- return cache.get(refId)
36
+ // one lookup, not has()+get(): only truthy results are ever cached, so a
37
+ // miss and a cached value are already distinguishable
38
+ const cached = cache.get(refId)
39
+ if (cached !== undefined) {
40
+ return cached
38
41
  }
39
42
  const result = getIndices(refId)
40
43
  if (result) {
package/src/record.ts CHANGED
@@ -31,6 +31,41 @@ for (let hi = 0; hi < 16; hi++) {
31
31
  // read lengths.
32
32
  const SEQ_DECODER_THRESHOLD = 300
33
33
 
34
+ // Four bases per entry, indexed by a PAIR of SEQ bytes, so the sub-threshold
35
+ // path halves its concat count: 1.5-1.6x on a short-read query end to end
36
+ // (shortreads_300x 46.5 -> 30.5 ms over 53.6k reads, volvox 4.8 -> 2.9 ms).
37
+ //
38
+ // Built lazily, and only once the short path has been taken enough times to pay
39
+ // for it. Filling 65536 entries costs ~6 ms and retains ~2 MB, so building on
40
+ // first use is a LOSS on a file that decodes only a handful of short reads:
41
+ // long-read fixtures with a short-read tail measured 1.4x slower that way
42
+ // (ecoli_nanopore has 27 sub-300bp reads out of 480, chm1 has 5 of 204). The
43
+ // counter is module-global on purpose — the table is shared, so what has to
44
+ // amortize is the total number of short decodes in the process, not per file.
45
+ const SEQRET_QUAD_WARMUP = 1024
46
+ let seqretShortCalls = 0
47
+ let SEQRET_QUAD_STRINGS: string[] | undefined
48
+
49
+ // The 4-base table, or undefined while still warming up (caller falls back to
50
+ // the 2-base table).
51
+ function seqretQuads() {
52
+ if (SEQRET_QUAD_STRINGS === undefined) {
53
+ if (++seqretShortCalls < SEQRET_QUAD_WARMUP) {
54
+ return undefined
55
+ }
56
+ const quads = new Array<string>(65536)
57
+ for (let a = 0; a < 256; a++) {
58
+ const sa = SEQRET_PAIR_STRINGS[a]!
59
+ const base = a << 8
60
+ for (let b = 0; b < 256; b++) {
61
+ quads[base | b] = sa + SEQRET_PAIR_STRINGS[b]!
62
+ }
63
+ }
64
+ SEQRET_QUAD_STRINGS = quads
65
+ }
66
+ return SEQRET_QUAD_STRINGS
67
+ }
68
+
34
69
  // Precomputed pair orientation strings, indexed by
35
70
  // ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
36
71
  // bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
@@ -87,9 +122,20 @@ function decodeTagString(ba: Uint8Array, start: number, end: number) {
87
122
  return textDecoder.decode(ba.subarray(start, end))
88
123
  }
89
124
 
90
- // Bitmask for ops that consume ref: M=0, D=2, N=3, P=6, ==7, X=8
91
- // Binary: 0b111001101 = 0x1CD
92
- const CIGAR_CONSUMES_REF_MASK = 0x1cd
125
+ // Bitmask for ops that consume ref: M=0, D=2, N=3, ==7, X=8
126
+ // Binary: 0b110001101 = 0x18D
127
+ //
128
+ // P (=6) is NOT among them. Padding is a silent placeholder against a padded
129
+ // reference and advances neither query nor reference, which is what htslib's
130
+ // bam_cigar_type(P) == 0 says. Including it made every read carrying a P op
131
+ // report an end one base per padded base too far.
132
+ //
133
+ // That said: nobody cares about P. Padded alignments come out of a handful of
134
+ // assembly tools and are effectively absent from real BAMs — this was fixed
135
+ // because it disagreed with htslib, not because it was hurting anyone. It is
136
+ // covered incidentally by samspec.bam in the samtools agreement suite, and
137
+ // that is enough; don't spend a dedicated fixture or test on it.
138
+ const CIGAR_CONSUMES_REF_MASK = 0x18d
93
139
 
94
140
  // A CIGAR as packed op words. Either a view over (or copy of) the record's own
95
141
  // CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
@@ -104,12 +150,6 @@ function isNumericCigar(value: unknown): value is NumericCigar {
104
150
  )
105
151
  }
106
152
 
107
- export interface Bytes {
108
- start: number
109
- end: number
110
- byteArray: Uint8Array
111
- }
112
-
113
153
  type BArrayValue =
114
154
  | Int8Array
115
155
  | Uint8Array
@@ -202,13 +242,20 @@ function decodeBArrayTag(
202
242
  }
203
243
 
204
244
  // Byte span of a 'B' tag's element payload, for advancing the cursor past it.
245
+ // Returns -1 for a subtype outside the spec's cCsSiIf, whose element width is
246
+ // unknowable. Guessing one byte per element (which is what falling through to
247
+ // `limit` did) doesn't skip the tag, it lands the cursor mid-value and decodes
248
+ // the whole rest of the record's tag list out of garbage — where an unknown
249
+ // top-level type has always stopped the walk instead. Same choice here.
205
250
  function bArrayByteLength(Btype: number, limit: number) {
206
251
  if (Btype === 0x69 || Btype === 0x49 || Btype === 0x66) {
207
252
  return limit << 2
208
253
  } else if (Btype === 0x73 || Btype === 0x53) {
209
254
  return limit << 1
210
- } else {
255
+ } else if (Btype === 0x63 || Btype === 0x43) {
211
256
  return limit
257
+ } else {
258
+ return -1
212
259
  }
213
260
  }
214
261
 
@@ -242,6 +289,11 @@ function tagValueEnd(
242
289
  case 0x5a: // 'Z'
243
290
  case 0x48: {
244
291
  // 'H'
292
+ // Stays a plain byte loop. Swapping in Uint8Array.indexOf past a short
293
+ // inline probe is 2.7x faster on long-read MD (mean 9083 bytes on
294
+ // jb2bench's 200x.longread) but ~1.13x slower on short-read Z values,
295
+ // which are 4-13 bytes and are what the dominant case is made of. Measured
296
+ // both ways against the realistic corpus — see ADR 0012.
245
297
  let q = p
246
298
  while (q < blockEnd && ba[q] !== 0) {
247
299
  q++
@@ -252,7 +304,12 @@ function tagValueEnd(
252
304
  // 'B'
253
305
  const Btype = ba[p]!
254
306
  const limit = dataView.getInt32(p + 1, true)
255
- return p + 5 + bArrayByteLength(Btype, limit)
307
+ const payload = bArrayByteLength(Btype, limit)
308
+ if (payload < 0) {
309
+ console.error('Unknown BAM B tag subtype', Btype)
310
+ return 0
311
+ }
312
+ return p + 5 + payload
256
313
  }
257
314
  default:
258
315
  console.error('Unknown BAM tag type', type)
@@ -451,6 +508,60 @@ export default class BamRecord {
451
508
  return this._findTag(tagName, true)
452
509
  }
453
510
 
511
+ /**
512
+ * The value of `tagName`, or of `altName` if the record carries no `tagName` —
513
+ * resolved in ONE pass over the tag block instead of two.
514
+ *
515
+ * For the MM/Mm and ML/Ml alias pairs that modified-base callers emit, where
516
+ * `getTag(a) ?? getTag(b)` walks every tag on the record TWICE whenever
517
+ * neither is present, which is every read in a file without base
518
+ * modifications. jbrowse-components issues exactly that lookup per record on
519
+ * every render (extractModifications runs unconditionally), and on
520
+ * jb2bench's 1000x.shortread it was 12.9% of the whole query — more than the
521
+ * CIGAR, SEQ and MD reads the pileup actually uses, spent proving absence.
522
+ *
523
+ * `tagName` wins wherever it appears, so the result matches the two-lookup
524
+ * form even for a (malformed) record carrying both.
525
+ */
526
+ getTagAlt(tagName: string, altName: string) {
527
+ if (this._cachedTags !== undefined) {
528
+ return this._cachedTags[tagName] ?? this._cachedTags[altName]
529
+ }
530
+ const a0 = tagName.charCodeAt(0)
531
+ const a1 = tagName.charCodeAt(1)
532
+ const b0 = altName.charCodeAt(0)
533
+ const b1 = altName.charCodeAt(1)
534
+ const blockEnd = this._end
535
+ const ba = this._byteArray
536
+ let p = this.tagsStart
537
+ // where the alternate landed, if it turns up before the primary does
538
+ let altType = -1
539
+ let altStart = 0
540
+ let altEnd = 0
541
+ while (p < blockEnd) {
542
+ const c0 = ba[p]
543
+ const c1 = ba[p + 1]
544
+ const type = ba[p + 2]!
545
+ const valueStart = p + 3
546
+ const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
547
+ if (end === 0) {
548
+ break // unknown type: can't compute how far to advance
549
+ }
550
+ if (c0 === a0 && c1 === a1) {
551
+ return decodeTagValue(ba, this._dataView, type, valueStart, end, false)
552
+ }
553
+ if (altType < 0 && c0 === b0 && c1 === b1) {
554
+ altType = type
555
+ altStart = valueStart
556
+ altEnd = end
557
+ }
558
+ p = end
559
+ }
560
+ return altType < 0
561
+ ? undefined
562
+ : decodeTagValue(ba, this._dataView, altType, altStart, altEnd, false)
563
+ }
564
+
454
565
  private _findTag(tagName: string, raw: boolean) {
455
566
  const tag1 = tagName.charCodeAt(0)
456
567
  const tag2 = tagName.charCodeAt(1)
@@ -622,12 +733,18 @@ export default class BamRecord {
622
733
  return lref
623
734
  }
624
735
 
736
+ // Unmapped records are not special-cased here, unlike in
737
+ // _computeLengthOnRef. The spec says no assumptions can be made about an
738
+ // unmapped read's CIGAR, but "no assumptions" is not "no CIGAR": aligners
739
+ // emit placed unmapped mates that carry a real one, and htslib prints
740
+ // whatever is stored. Dropping it lost data the file had — paired.bam's
741
+ // SRR062635.1831187 at 20:74230 is FLAG 133 with 35M65S.
742
+ //
743
+ // Reference span stays 0 for them regardless, since that is a claim about
744
+ // alignment rather than about the stored bytes, and the query filter reads
745
+ // it.
625
746
  private _computeNumericCigar(): NumericCigar {
626
747
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
627
- if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
628
- return new Uint32Array(0)
629
- }
630
-
631
748
  const numCigarOps = flag_nc & 0xffff
632
749
  const p = this.b0 + this.read_name_length
633
750
 
@@ -664,14 +781,26 @@ export default class BamRecord {
664
781
  return this._cachedNumericCigar
665
782
  }
666
783
 
784
+ // Two appends per op, NOT `result += length + String.fromCharCode(op)`. The
785
+ // one-append form builds an intermediate cons string per op just to append it
786
+ // and drop it; appending each piece straight onto the rope is 1.23-1.35x
787
+ // faster on long reads, which is where this accessor dominates
788
+ // (chr22_nanopore_subset 51.2 -> 38.6 ms for 757 reads averaging 2171 ops,
789
+ // ultra-long-ont 13.2 -> 9.9 ms). Short reads carry 1-3 ops and land inside
790
+ // this box's noise band either way.
791
+ //
792
+ // ADR 0003 rejected two other rewrites of this loop — a precomputed 16-entry
793
+ // op-char table, and digits into a Uint8Array with one TextDecoder.decode —
794
+ // and both are still losers. Re-measured here: the op-char table is 1.13x
795
+ // SLOWER than String.fromCharCode even on top of this change, because V8
796
+ // already hands back an interned single-character string.
667
797
  get CIGAR() {
668
798
  const numeric = this.NUMERIC_CIGAR
669
799
  let result = ''
670
800
  for (let i = 0, l = numeric.length; i < l; i++) {
671
801
  const packed = numeric[i]!
672
- const length = packed >> 4
673
- const opCode = ASCII_CIGAR_CODES[packed & 0xf]!
674
- result += length + String.fromCharCode(opCode)
802
+ result += packed >> 4
803
+ result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
675
804
  }
676
805
  return result
677
806
  }
@@ -697,9 +826,10 @@ export default class BamRecord {
697
826
  return this._byteArray.subarray(p, p + this.num_seq_bytes)
698
827
  }
699
828
 
700
- // Decode two bases per iteration off a 256-entry table. Building an array of
701
- // 1-char strings and join()ing it — the obvious approach — is 3x slower at
702
- // 100bp and 35x slower at 15kb.
829
+ // Decode four bases per iteration off the 65536-entry table (two off the
830
+ // 256-entry one until that table has warmed up). Building an array of 1-char
831
+ // strings and join()ing it — the obvious approach — is 3x slower at 100bp and
832
+ // 35x slower at 15kb.
703
833
  get seq() {
704
834
  const len = this.seq_length
705
835
  const ba = this._byteArray
@@ -707,9 +837,22 @@ export default class BamRecord {
707
837
  const nPairs = len >> 1
708
838
  let seq: string
709
839
  if (len < SEQ_DECODER_THRESHOLD) {
840
+ const quads = seqretQuads()
710
841
  seq = ''
711
- for (let j = 0; j < nPairs; j++) {
712
- seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
842
+ if (quads === undefined) {
843
+ for (let j = 0; j < nPairs; j++) {
844
+ seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
845
+ }
846
+ } else {
847
+ const nQuads = nPairs >> 1
848
+ for (let j = 0; j < nQuads; j++) {
849
+ const q = p + 2 * j
850
+ seq += quads[(ba[q]! << 8) | ba[q + 1]!]!
851
+ }
852
+ // odd byte count: two more bases off the 2-base table
853
+ if (nPairs & 1) {
854
+ seq += SEQRET_PAIR_STRINGS[ba[p + nPairs - 1]!]!
855
+ }
713
856
  }
714
857
  if (len & 1) {
715
858
  seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!