@gmod/bam 9.0.1 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +18 -17
  2. package/dist/bai.d.ts +2 -0
  3. package/dist/bai.js +19 -0
  4. package/dist/bai.js.map +1 -1
  5. package/dist/bamFile.js +11 -19
  6. package/dist/bamFile.js.map +1 -1
  7. package/dist/csi.d.ts +20 -29
  8. package/dist/csi.js +27 -10
  9. package/dist/csi.js.map +1 -1
  10. package/dist/indexFile.d.ts +21 -1
  11. package/dist/indexFile.js +60 -2
  12. package/dist/indexFile.js.map +1 -1
  13. package/dist/record.d.ts +1 -0
  14. package/dist/record.js +33 -21
  15. package/dist/record.js.map +1 -1
  16. package/dist/streamBam.js +2 -2
  17. package/dist/streamBam.js.map +1 -1
  18. package/dist/util.d.ts +15 -4
  19. package/dist/util.js +29 -4
  20. package/dist/util.js.map +1 -1
  21. package/dist/virtualOffset.d.ts +1 -0
  22. package/dist/virtualOffset.js +4 -0
  23. package/dist/virtualOffset.js.map +1 -1
  24. package/esm/bai.d.ts +2 -0
  25. package/esm/bai.js +19 -0
  26. package/esm/bai.js.map +1 -1
  27. package/esm/bamFile.js +12 -20
  28. package/esm/bamFile.js.map +1 -1
  29. package/esm/csi.d.ts +20 -29
  30. package/esm/csi.js +27 -10
  31. package/esm/csi.js.map +1 -1
  32. package/esm/indexFile.d.ts +21 -1
  33. package/esm/indexFile.js +59 -2
  34. package/esm/indexFile.js.map +1 -1
  35. package/esm/record.d.ts +1 -0
  36. package/esm/record.js +33 -21
  37. package/esm/record.js.map +1 -1
  38. package/esm/streamBam.js +3 -3
  39. package/esm/streamBam.js.map +1 -1
  40. package/esm/util.d.ts +15 -4
  41. package/esm/util.js +27 -4
  42. package/esm/util.js.map +1 -1
  43. package/esm/virtualOffset.d.ts +1 -0
  44. package/esm/virtualOffset.js +3 -0
  45. package/esm/virtualOffset.js.map +1 -1
  46. package/package.json +7 -7
  47. package/src/bai.ts +21 -0
  48. package/src/bamFile.ts +14 -21
  49. package/src/csi.ts +41 -14
  50. package/src/indexFile.ts +72 -3
  51. package/src/record.ts +39 -20
  52. package/src/streamBam.ts +4 -4
  53. package/src/util.ts +33 -4
  54. package/src/virtualOffset.ts +4 -0
package/src/bamFile.ts CHANGED
@@ -12,7 +12,10 @@ import {
12
12
  BAM_MAGIC,
13
13
  MAX_CONCURRENT_CHUNK_READS,
14
14
  appendInRange,
15
+ decodeHeaderText,
16
+ optimizeChunks,
15
17
  parseRefSeqs,
18
+ readBlockSize,
16
19
  resolveFilehandle,
17
20
  throwIfAborted,
18
21
  } from './util.ts'
@@ -464,9 +467,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
464
467
  const parsed = parseRefSeqs(uncba, headLen + 8, this.renameRefSeq)
465
468
  let samHeader
466
469
  if (parsed) {
467
- const headerText = new TextDecoder('utf8').decode(
468
- uncba.subarray(8, 8 + headLen),
469
- )
470
+ const headerText = decodeHeaderText(uncba, headLen)
470
471
  this.header = headerText
471
472
  this.chrToIndex = parsed.chrToIndex
472
473
  this.indexToChr = parsed.indexToChr
@@ -539,7 +540,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
539
540
  if (chrId === undefined || !this.index) {
540
541
  return []
541
542
  }
542
- const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
543
+ const chunks = await this.index.blocksForRange(chrId, min, max, opts)
543
544
  return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts)
544
545
  }
545
546
 
@@ -818,10 +819,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
818
819
  // make the count NaN.
819
820
  const readNameCounts = new Map<string, number>()
820
821
  const readIds = new Set<number>()
822
+ const names = new Array<string>(records.length)
821
823
 
822
824
  for (let i = 0, l = records.length; i < l; i++) {
823
825
  const r = records[i]!
824
826
  const name = r.name
827
+ names[i] = name
825
828
  readNameCounts.set(name, (readNameCounts.get(name) ?? 0) + 1)
826
829
  readIds.add(r.fileOffset)
827
830
  }
@@ -829,10 +832,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
829
832
  const matePromises: Promise<Chunk[]>[] = []
830
833
  for (let i = 0, l = records.length; i < l; i++) {
831
834
  const f = records[i]!
832
- const name = f.name
833
835
  if (
834
836
  this.index &&
835
- readNameCounts.get(name) === 1 &&
837
+ readNameCounts.get(names[i]!) === 1 &&
836
838
  (pairAcrossChr ||
837
839
  (f.next_refid === chrId &&
838
840
  Math.abs(f.start - f.next_pos) < maxInsertSize))
@@ -848,20 +850,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
848
850
  }
849
851
  }
850
852
 
851
- const map = new Map<string, Chunk>()
852
- const res = await Promise.all(matePromises)
853
- for (let i = 0, l = res.length; i < l; i++) {
854
- const chunks = res[i]!
855
- for (let j = 0, jl = chunks.length; j < jl; j++) {
856
- const m = chunks[j]!
857
- // Key on the virtual-offset span — the same key _cachedChunkFeatures
858
- // uses. Chunk.toString() also folds in `bin` and fetchedSize(), which
859
- // keeps two chunks covering an identical span apart here even though
860
- // the cache below collapses them, so their records came back twice.
861
- map.set(chunkCacheKey(m), m)
862
- }
863
- }
864
- const mateChunks = [...map.values()]
853
+ // Each mate's lookup is merged on its own, so two of them can resolve to
854
+ // different spans over the same records. Merging the union again makes
855
+ // them disjoint, which is what keeps a mate in the overlap from coming
856
+ // back once per span.
857
+ const mateChunks = optimizeChunks((await Promise.all(matePromises)).flat())
865
858
 
866
859
  // Bounded for the reason ADR 0008 bounds the main query path: a viewAsPairs
867
860
  // query over a busy region resolves to many distinct mate chunks, and an
@@ -961,7 +954,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
961
954
  const hasCpositions = cpositions.length > 0
962
955
 
963
956
  while (blockStart + 4 < ba.length) {
964
- const blockSize = dataView.getInt32(blockStart, true)
957
+ const blockSize = readBlockSize(dataView, blockStart)
965
958
  const blockEnd = blockStart + 4 + blockSize - 1
966
959
 
967
960
  if (hasDpositions) {
package/src/csi.ts CHANGED
@@ -10,6 +10,7 @@ import {
10
10
  } from './util.ts'
11
11
  import { VirtualOffset, fromBytes } from './virtualOffset.ts'
12
12
 
13
+ import type { ParsedIndexBase, RefIndex } from './indexFile.ts'
13
14
  import type { BaseOpts } from './util.ts'
14
15
 
15
16
  const CSI1_MAGIC = 21582659 // CSI\1
@@ -24,10 +25,18 @@ function rshift(num: number, bits: number) {
24
25
  return Math.floor(num / 2 ** bits)
25
26
  }
26
27
 
27
- export default class CSI extends IndexFile {
28
+ interface CsiRefIndex extends RefIndex {
29
+ loffsets: Map<number, VirtualOffset>
30
+ }
31
+
32
+ interface CsiParsed extends ParsedIndexBase<CsiRefIndex> {
33
+ csi: true
34
+ }
35
+
36
+ export default class CSI extends IndexFile<CsiParsed> {
28
37
  private maxBinNumber = 0
29
- private depth = 0
30
- private minShift = 0
38
+ protected depth = 0
39
+ protected minShift = 0
31
40
 
32
41
  // CSI omits the linear index that BAI's indexCov derives coverage from
33
42
  // (CSIv1.tex §3, hts-specs), so there's no equivalent to return.
@@ -74,7 +83,7 @@ export default class CSI extends IndexFile {
74
83
  }
75
84
 
76
85
  // fetch and parse the index
77
- async _parse(opts: BaseOpts) {
86
+ async _parse(opts: BaseOpts): Promise<CsiParsed> {
78
87
  const buffer = await this.filehandle.readFile(opts)
79
88
  const bytes = await unzip(buffer)
80
89
 
@@ -93,7 +102,7 @@ export default class CSI extends IndexFile {
93
102
 
94
103
  this.minShift = dataView.getInt32(4, true)
95
104
  this.depth = dataView.getInt32(8, true)
96
- this.maxBinNumber = ((1 << ((this.depth + 1) * 3)) - 1) / 7
105
+ this.maxBinNumber = (8 ** (this.depth + 1) - 1) / 7
97
106
  const maxBinNumber = this.maxBinNumber
98
107
  const auxLength = dataView.getInt32(12, true)
99
108
  // A tabix-only branch, which is why parseAuxData and the parseNameBytes it
@@ -128,10 +137,10 @@ export default class CSI extends IndexFile {
128
137
  if (bin > this.maxBinNumber) {
129
138
  curr += 28 + 16
130
139
  } else {
131
- // A bin's loffset is the smallest virtual offset of any record in it,
132
- // so the minimum over loffsets is already the minimum over the bin's
133
- // chunks — one read per bin instead of one per chunk. Checked against
134
- // every .csi in test/data: same answer on all 19.
140
+ // A bin's loffset is the linear-index entry at its first window, so
141
+ // the smallest over every bin is the first record's offset — one
142
+ // read per bin instead of one per chunk. Checked against every .csi
143
+ // in test/data: same answer on all 19.
135
144
  firstDataLine = minVirtualOffset(bytes, curr, 1, firstDataLine)
136
145
  curr += 8 // loffset
137
146
  const chunkCount = dataView.getInt32(curr, true)
@@ -149,6 +158,7 @@ export default class CSI extends IndexFile {
149
158
  const binCount = dataView.getInt32(curr, true)
150
159
  curr += 4
151
160
  const binIndex: Record<number, Chunk[]> = {}
161
+ const loffsets = new Map<number, VirtualOffset>()
152
162
  let pseudoBinStats
153
163
  for (let j = 0; j < binCount; j++) {
154
164
  const bin = dataView.getUint32(curr, true)
@@ -157,7 +167,8 @@ export default class CSI extends IndexFile {
157
167
  pseudoBinStats = parsePseudoBin(bytes, curr + 28)
158
168
  curr += 28 + 16
159
169
  } else {
160
- curr += 8 // skip loffset; firstDataLine was computed in the first pass
170
+ loffsets.set(bin, fromBytes(bytes, curr))
171
+ curr += 8
161
172
  const chunkCount = dataView.getInt32(curr, true)
162
173
  curr += 4
163
174
  const chunks = new Array<Chunk>(chunkCount)
@@ -175,6 +186,7 @@ export default class CSI extends IndexFile {
175
186
  clampChunkEnds(Object.values(binIndex).flat())
176
187
  return {
177
188
  binIndex,
189
+ loffsets,
178
190
  stats: pseudoBinStats,
179
191
  }
180
192
  }
@@ -188,12 +200,27 @@ export default class CSI extends IndexFile {
188
200
  }
189
201
  }
190
202
 
191
- // CSI has no linear index — every refId starts from the beginning of file.
192
- protected getLowestChunk() {
193
- return ZERO_OFFSET
203
+ /**
204
+ * CSI has no linear index, but each bin's `loffset` is the linear-index entry
205
+ * at the bin's first window (htslib's `update_loff`), so the finest bin at or
206
+ * left of `min` bounds the query from below the way BAI's linear index does.
207
+ * Walks left through siblings and then up to the parent, as `hts_itr_query`
208
+ * does for CSI.
209
+ */
210
+ protected getLowestChunk(refIndex: CsiRefIndex, min: number) {
211
+ const { loffsets } = refIndex
212
+ const leaves = 8 ** this.depth
213
+ let bin =
214
+ (leaves - 1) / 7 +
215
+ Math.min(Math.floor(min / 2 ** this.minShift), leaves - 1)
216
+ while (bin > 0 && !loffsets.has(bin)) {
217
+ const parent = Math.floor((bin - 1) / 8)
218
+ bin = bin > parent * 8 + 1 ? bin - 1 : parent
219
+ }
220
+ return loffsets.get(bin) ?? ZERO_OFFSET
194
221
  }
195
222
 
196
- // ...and so there is nothing to bound the far end of a query with either, so
223
+ // No linear index means nothing to forecast the far end of a query with, so
197
224
  // estimatedBytesForRegions keeps summing every chunk on a CSI-indexed file.
198
225
  protected getHighestChunk() {
199
226
  return undefined
package/src/indexFile.ts CHANGED
@@ -2,6 +2,7 @@ import { SharedReadCache } from '@gmod/shared-read-cache'
2
2
  import QuickLRU from '@jbrowse/quick-lru'
3
3
 
4
4
  import { chunksLikelyRead, optimizeChunks } from './util.ts'
5
+ import { compareOffsets } from './virtualOffset.ts'
5
6
 
6
7
  import type Chunk from './chunk.ts'
7
8
  import type { BaseOpts } from './util.ts'
@@ -48,6 +49,58 @@ export function memoizeByRefId<T>(
48
49
  }
49
50
  }
50
51
 
52
+ /**
53
+ * The virtual offset past which a coordinate-sorted file holds nothing
54
+ * overlapping `[.., end)`, from the binning index alone: htslib's `max_off`
55
+ * (`hts_itr_query` in hts.c).
56
+ *
57
+ * Walk right from the finest bin after the one holding `end - 1`, stepping up
58
+ * to the parent at every first child, so each bin visited begins at or past
59
+ * `end` and never overlaps the query. Every record in such a bin starts at or
60
+ * past `end`, so the first chunk of the first bin that exists is a record past
61
+ * the query, and in a sorted file so is every record after it.
62
+ *
63
+ * A bound, unlike the linear-index forecast in `chunksLikelyRead`: it rests on
64
+ * the same sort order `appendInRange` and the early stop already assume, and
65
+ * cannot drop a record they would keep. See ADR 0023 for why the caller drops
66
+ * whole merged chunks with it rather than trimming them.
67
+ */
68
+ export function maxOffset(
69
+ binIndex: Record<number, Chunk[]>,
70
+ end: number,
71
+ minShift: number,
72
+ depth: number,
73
+ ) {
74
+ if (end > 2 ** (minShift + depth * 3)) {
75
+ return undefined
76
+ }
77
+ const binCount = (8 ** (depth + 1) - 1) / 7
78
+ let bin = (8 ** depth - 1) / 7 + Math.floor((end - 1) / 2 ** minShift) + 1
79
+ if (bin >= binCount) {
80
+ bin = 0
81
+ }
82
+ for (;;) {
83
+ while (bin % 8 === 1) {
84
+ bin = (bin - 1) / 8
85
+ }
86
+ if (bin === 0) {
87
+ return undefined
88
+ }
89
+ const chunks = binIndex[bin]
90
+ if (chunks?.length) {
91
+ let lowest = chunks[0]!.minv
92
+ for (let i = 1; i < chunks.length; i++) {
93
+ const minv = chunks[i]!.minv
94
+ if (compareOffsets(minv, lowest) < 0) {
95
+ lowest = minv
96
+ }
97
+ }
98
+ return lowest
99
+ }
100
+ bin++
101
+ }
102
+ }
103
+
51
104
  export default abstract class IndexFile<
52
105
  TParsed extends ParsedIndexBase = ParsedIndexBase,
53
106
  > {
@@ -79,6 +132,11 @@ export default abstract class IndexFile<
79
132
  end?: number,
80
133
  ): Promise<{ start: number; end: number; score: number }[]>
81
134
 
135
+ // The binning scheme: the finest bins are 2^minShift wide, and there are
136
+ // depth levels below bin 0. BAI is CSI with minShift 14 and depth 5.
137
+ protected abstract minShift: number
138
+ protected abstract depth: number
139
+
82
140
  // Bin numbers that overlap [min, max). Subclasses implement BAI's fixed
83
141
  // 5-level scheme or CSI's configurable scheme (SAMv1.pdf §5.1.1, CSIv1.tex §2).
84
142
  protected abstract reg2bins(
@@ -87,7 +145,8 @@ export default abstract class IndexFile<
87
145
  ): readonly (readonly [number, number])[]
88
146
 
89
147
  // Lower-bound virtual offset for chunks that could contain alignments in
90
- // [min, ...). BAI uses its linear index; CSI has none and returns 0:0.
148
+ // [min, ...). BAI uses its linear index, CSI the loffset of a bin at or left
149
+ // of min.
91
150
  protected abstract getLowestChunk(
92
151
  refIndex: RefIndex,
93
152
  min: number,
@@ -133,7 +192,16 @@ export default abstract class IndexFile<
133
192
  }
134
193
  }
135
194
  }
136
- return optimizeChunks(chunks, this.getLowestChunk(ba, min))
195
+ const merged = optimizeChunks(chunks, this.getLowestChunk(ba, min))
196
+ const past = maxOffset(binIndex, max, this.minShift, this.depth)
197
+ if (past) {
198
+ let n = merged.length
199
+ while (n > 0 && compareOffsets(merged[n - 1]!.minv, past) >= 0) {
200
+ n--
201
+ }
202
+ merged.length = n
203
+ }
204
+ return merged
137
205
  }
138
206
 
139
207
  // SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same shape and the same
@@ -197,7 +265,8 @@ export default abstract class IndexFile<
197
265
  * exists — was being told 5.6x the truth on exactly the windows a reader
198
266
  * spends their time in, and cannot answer it by zooming: every window narrower
199
267
  * than a linear-index interval resolves to the same chunks and so to the same
200
- * number.
268
+ * number. The table predates `max_off` (ADR 0023), which now drops most of
269
+ * the gap between the first two columns from `blocksForRange` itself.
201
270
  *
202
271
  * Still summed over merged chunks rather than per region, so two regions
203
272
  * sharing a chunk are charged for it once.
package/src/record.ts CHANGED
@@ -440,16 +440,18 @@ export default class BamRecord {
440
440
  }
441
441
 
442
442
  // QUAL is present whenever the record has bases — independent of the unmapped
443
- // flag (unmapped reads routinely carry SEQ/QUAL). A zero-length SEQ means
444
- // there is no quality to return.
443
+ // flag (unmapped reads routinely carry SEQ/QUAL). null for a zero-length SEQ,
444
+ // and for a QUAL of `*`, which BAM stores as 0xFF in every byte (SAMv1
445
+ // §4.2.3). htslib decides on the first byte alone, and so does this.
445
446
  get qual() {
446
447
  const seqLen = this.seq_length
447
448
  if (seqLen === 0) {
448
449
  return null
449
- } else {
450
- const p = this.seqStart + ((seqLen + 1) >> 1)
451
- return this._byteArray.subarray(p, p + seqLen)
452
450
  }
451
+ const p = this.seqStart + ((seqLen + 1) >> 1)
452
+ return this._byteArray[p] === 0xff
453
+ ? null
454
+ : this._byteArray.subarray(p, p + seqLen)
453
455
  }
454
456
 
455
457
  get strand() {
@@ -475,7 +477,7 @@ export default class BamRecord {
475
477
  return this.seqStart + ((seqLen + 1) >> 1) + seqLen
476
478
  }
477
479
 
478
- // batch fromCharCode: fastest for typical name lengths (see benchmarks/string-building.bench.ts)
480
+ // batch fromCharCode: fastest for typical name lengths (benchmarks/string-building.bench.ts, removed in 36a3968)
479
481
  //
480
482
  // Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
481
483
  // read name about once — jbrowse-components' buildBaseFeatureData copies it
@@ -510,8 +512,8 @@ export default class BamRecord {
510
512
  }
511
513
 
512
514
  getTag(tagName: string) {
513
- if (this._cachedTags !== undefined) {
514
- return this._cachedTags[tagName]
515
+ if (this._cachedTags !== undefined || tagName === 'CG') {
516
+ return this.tags[tagName]
515
517
  }
516
518
  return this._findTag(tagName, false)
517
519
  }
@@ -621,6 +623,9 @@ export default class BamRecord {
621
623
  )
622
624
  p = end
623
625
  }
626
+ if (this._hasCGPlaceholder() && isNumericCigar(tags.CG)) {
627
+ delete tags.CG
628
+ }
624
629
  return tags
625
630
  }
626
631
 
@@ -672,7 +677,7 @@ export default class BamRecord {
672
677
  return !!(this.flags & Constants.BAM_FSUPPLEMENTARY)
673
678
  }
674
679
 
675
- // Benchmark results for CIGAR parsing strategies (see benchmarks/cigar-lifecycle.bench.ts):
680
+ // Benchmark results for CIGAR parsing strategies (benchmarks/cigar-strategies.bench.ts, removed in 36a3968):
676
681
  //
677
682
  // Aligned data:
678
683
  // - Plain array is 1.6-1.8x faster than Uint32Array for small CIGARs (≤50 ops)
@@ -696,12 +701,23 @@ export default class BamRecord {
696
701
  // htslib stores the placeholder as exactly two ops: <seqlen>S<reflen>N.
697
702
  if (numCigarOps === 2) {
698
703
  const cigop = this._dataView.getInt32(p, true)
699
- return (cigop & 0xf) === CIGAR_SOFT_CLIP && cigop >> 4 === this.seq_length
704
+ return (
705
+ (cigop & 0xf) === CIGAR_SOFT_CLIP && cigop >>> 4 === this.seq_length
706
+ )
700
707
  } else {
701
708
  return false
702
709
  }
703
710
  }
704
711
 
712
+ // SAMv1 §4.2.2: with the placeholder and a CG tag, the tag is the CIGAR and
713
+ // a reader removes it from the tags, as htslib does
714
+ private _hasCGPlaceholder() {
715
+ return this._isCGTagPattern(
716
+ this.b0 + this.read_name_length,
717
+ this.flag_nc & 0xffff,
718
+ )
719
+ }
720
+
705
721
  private _computeLengthOnRef(): number {
706
722
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
707
723
  if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
@@ -716,7 +732,7 @@ export default class BamRecord {
716
732
  if ((cigop2 & 0xf) !== CIGAR_REF_SKIP) {
717
733
  console.warn('CG tag with no N tag')
718
734
  }
719
- return cigop2 >> 4
735
+ return cigop2 >>> 4
720
736
  }
721
737
 
722
738
  const absOffset = this._byteArray.byteOffset + p
@@ -732,7 +748,7 @@ export default class BamRecord {
732
748
  let lref = 0
733
749
  for (let c = 0; c < numCigarOps; ++c) {
734
750
  const co = cigarView[c]!
735
- lref += (co >> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
751
+ lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
736
752
  }
737
753
  return lref
738
754
  }
@@ -740,7 +756,7 @@ export default class BamRecord {
740
756
  let lref = 0
741
757
  for (let c = 0; c < numCigarOps; ++c) {
742
758
  const co = this._dataView.getInt32(p + c * 4, true)
743
- lref += (co >> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
759
+ lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
744
760
  }
745
761
  return lref
746
762
  }
@@ -761,10 +777,10 @@ export default class BamRecord {
761
777
  const p = this.b0 + this.read_name_length
762
778
 
763
779
  if (this._isCGTagPattern(p, numCigarOps)) {
764
- // getTag, not this.tags: the real CIGAR lives in one tag, so there's no
765
- // reason to decode every other tag on the record to reach it
766
- const cg = this.getTag('CG')
767
- return isNumericCigar(cg) ? cg : new Uint32Array(0)
780
+ const cg = this._findTag('CG', false)
781
+ if (isNumericCigar(cg)) {
782
+ return cg
783
+ }
768
784
  }
769
785
 
770
786
  const absOffset = this._byteArray.byteOffset + p
@@ -811,18 +827,21 @@ export default class BamRecord {
811
827
  let result = ''
812
828
  for (let i = 0, l = numeric.length; i < l; i++) {
813
829
  const packed = numeric[i]!
814
- result += packed >> 4
830
+ result += packed >>> 4
815
831
  result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
816
832
  }
817
833
  return result
818
834
  }
819
835
 
820
836
  get num_cigar_ops() {
821
- return this.flag_nc & 0xffff
837
+ return this._hasCGPlaceholder()
838
+ ? this.NUMERIC_CIGAR.length
839
+ : this.flag_nc & 0xffff
822
840
  }
823
841
 
842
+ // the stored CIGAR field, which for a long CIGAR is the two-op placeholder
824
843
  get num_cigar_bytes() {
825
- return this.num_cigar_ops << 2
844
+ return (this.flag_nc & 0xffff) << 2
826
845
  }
827
846
 
828
847
  get read_name_length() {
package/src/streamBam.ts CHANGED
@@ -9,7 +9,9 @@ import { parseHeaderText } from './sam.ts'
9
9
  import {
10
10
  BAM_MAGIC,
11
11
  concatUint8Array,
12
+ decodeHeaderText,
12
13
  parseRefSeqs,
14
+ readBlockSize,
13
15
  resolveFilehandle,
14
16
  throwIfAborted,
15
17
  } from './util.ts'
@@ -273,9 +275,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
273
275
  recordCarry = bytes
274
276
  continue
275
277
  }
276
- const headerText = new TextDecoder('utf8').decode(
277
- bytes.subarray(8, 8 + lText),
278
- )
278
+ const headerText = decodeHeaderText(bytes, lText)
279
279
  onHeader?.({
280
280
  headerText,
281
281
  samHeader: parseHeaderText(headerText),
@@ -288,7 +288,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
288
288
 
289
289
  const sink: T[] = []
290
290
  while (blockStart + 4 <= bytes.length) {
291
- const blockSize = dataView.getInt32(blockStart, true)
291
+ const blockSize = readBlockSize(dataView, blockStart)
292
292
  const blockEnd = blockStart + 4 + blockSize - 1
293
293
  if (blockEnd >= bytes.length) {
294
294
  break
package/src/util.ts CHANGED
@@ -9,6 +9,22 @@ import type { GenericFilehandle } from 'generic-filehandle2'
9
9
  /** 'BAM\1' read as a little-endian int32 */
10
10
  export const BAM_MAGIC = 21840194
11
11
 
12
+ /**
13
+ * The `block_size` of the record at `offset`, rejecting one too small to hold a
14
+ * record's 32 bytes of fixed fields, as htslib's `bam_read1` does. A negative
15
+ * one would otherwise leave the loops that advance by it standing still,
16
+ * pushing a record per turn until the process runs out of memory.
17
+ */
18
+ export function readBlockSize(dataView: DataView, offset: number) {
19
+ const blockSize = dataView.getInt32(offset, true)
20
+ if (blockSize < 32) {
21
+ throw new Error(
22
+ `corrupt BAM record: block_size ${blockSize} at byte ${offset}`,
23
+ )
24
+ }
25
+ return blockSize
26
+ }
27
+
12
28
  export function resolveFilehandle(
13
29
  filehandle?: GenericFilehandle,
14
30
  path?: string,
@@ -66,10 +82,13 @@ export const MAX_CONCURRENT_CHUNK_READS = 6
66
82
  *
67
83
  * `blocksForRange` returns every chunk of every bin overlapping the query, at
68
84
  * every level of the binning scheme, minus the ones the linear index puts
69
- * entirely before it. On a long-read file that is wildly more than the query
70
- * reads: a coarse bin's chunks run to the end of the bin's span, so a 380bp
71
- * window on a deep ONT BAM resolves to 90 chunks / 43.5MB, of which
72
- * `getRecordsForRange` reads 6 / 7.8MB before the early stop fires.
85
+ * entirely before it. Until `max_off` (ADR 0023) also dropped the ones past it,
86
+ * that was wildly more than a long-read query reads: a coarse bin's chunks run
87
+ * to the end of the bin's span, so a 380bp window on a deep ONT BAM resolved to
88
+ * 90 chunks / 43.5MB, of which `getRecordsForRange` read 6 / 7.8MB before the
89
+ * early stop fired. Since then this changes the forecast in 4 of 320 fixture
90
+ * windows, undershooting in all 4; ADR 0023 says what to check before
91
+ * removing it.
73
92
  *
74
93
  * Two bounds, and the answer is the larger:
75
94
  *
@@ -325,6 +344,16 @@ export function clampChunkEnds(
325
344
  }
326
345
  }
327
346
 
347
+ // SAMv1 §4.2 counts NUL padding in l_text, and htslib reads the text as a C
348
+ // string, so the header ends at the first NUL.
349
+ export function decodeHeaderText(bytes: Uint8Array, lText: number) {
350
+ const text = bytes.subarray(8, 8 + lText)
351
+ const nul = text.indexOf(0)
352
+ return new TextDecoder('utf8').decode(
353
+ nul === -1 ? text : text.subarray(0, nul),
354
+ )
355
+ }
356
+
328
357
  // Parse the BAM reference-sequence table (SAMv1.pdf §4.2). Returns undefined
329
358
  // if `uncba` doesn't yet contain the full table — caller fetches more bytes
330
359
  // and retries.
@@ -35,3 +35,7 @@ export function fromBytes(bytes: Uint8Array, offset = 0) {
35
35
  (bytes[offset + 1]! << 8) | bytes[offset]!,
36
36
  )
37
37
  }
38
+
39
+ export function compareOffsets(a: OffsetCoords, b: OffsetCoords) {
40
+ return a.blockPosition - b.blockPosition || a.dataPosition - b.dataPosition
41
+ }