@gmod/bam 9.0.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +24 -21
  2. package/dist/bai.d.ts +2 -0
  3. package/dist/bai.js +19 -0
  4. package/dist/bai.js.map +1 -1
  5. package/dist/bamFile.js +11 -19
  6. package/dist/bamFile.js.map +1 -1
  7. package/dist/csi.d.ts +20 -29
  8. package/dist/csi.js +27 -10
  9. package/dist/csi.js.map +1 -1
  10. package/dist/indexFile.d.ts +21 -1
  11. package/dist/indexFile.js +60 -2
  12. package/dist/indexFile.js.map +1 -1
  13. package/dist/record.d.ts +1 -0
  14. package/dist/record.js +39 -22
  15. package/dist/record.js.map +1 -1
  16. package/dist/streamBam.js +2 -2
  17. package/dist/streamBam.js.map +1 -1
  18. package/dist/util.d.ts +15 -4
  19. package/dist/util.js +29 -4
  20. package/dist/util.js.map +1 -1
  21. package/dist/virtualOffset.d.ts +1 -0
  22. package/dist/virtualOffset.js +4 -0
  23. package/dist/virtualOffset.js.map +1 -1
  24. package/esm/bai.d.ts +2 -0
  25. package/esm/bai.js +19 -0
  26. package/esm/bai.js.map +1 -1
  27. package/esm/bamFile.js +12 -20
  28. package/esm/bamFile.js.map +1 -1
  29. package/esm/csi.d.ts +20 -29
  30. package/esm/csi.js +27 -10
  31. package/esm/csi.js.map +1 -1
  32. package/esm/indexFile.d.ts +21 -1
  33. package/esm/indexFile.js +59 -2
  34. package/esm/indexFile.js.map +1 -1
  35. package/esm/record.d.ts +1 -0
  36. package/esm/record.js +39 -22
  37. package/esm/record.js.map +1 -1
  38. package/esm/streamBam.js +3 -3
  39. package/esm/streamBam.js.map +1 -1
  40. package/esm/util.d.ts +15 -4
  41. package/esm/util.js +27 -4
  42. package/esm/util.js.map +1 -1
  43. package/esm/virtualOffset.d.ts +1 -0
  44. package/esm/virtualOffset.js +3 -0
  45. package/esm/virtualOffset.js.map +1 -1
  46. package/package.json +7 -7
  47. package/src/bai.ts +21 -0
  48. package/src/bamFile.ts +14 -21
  49. package/src/csi.ts +41 -14
  50. package/src/indexFile.ts +72 -3
  51. package/src/record.ts +45 -21
  52. package/src/streamBam.ts +4 -4
  53. package/src/util.ts +33 -4
  54. package/src/virtualOffset.ts +4 -0
package/src/bai.ts CHANGED
@@ -49,6 +49,23 @@ function roundUp(n: number, multiple: number) {
49
49
  return rem === 0 ? n : n - rem + multiple
50
50
  }
51
51
 
52
+ /**
53
+ * Older samtools leaves 0:0 in linear-index windows no read overlaps. htslib
54
+ * fills interior ones from the next entry on load (`hts.c`, "fill missing
55
+ * values"); this also fills leading ones, which htslib leaves at 0. Either
56
+ * way the entry stays a lower bound, since no record overlaps that window.
57
+ * Left as 0, a leading entry scores its whole absolute file offset in
58
+ * `indexCov`, and `getLowestChunk` falls back to the start of the file.
59
+ */
60
+ function fillLinearGaps(blocks: Float64Array, data: Float64Array) {
61
+ for (let j = blocks.length - 2; j >= 0; j--) {
62
+ if (blocks[j] === 0 && data[j] === 0) {
63
+ blocks[j] = blocks[j + 1]!
64
+ data[j] = data[j + 1]!
65
+ }
66
+ }
67
+ }
68
+
52
69
  export interface IndexCovEntry {
53
70
  start: number
54
71
  end: number
@@ -83,6 +100,9 @@ function reg2bins(beg: number, end: number) {
83
100
  }
84
101
 
85
102
  export default class BAI extends IndexFile<BaiParsed> {
103
+ protected minShift = BAI_LINEAR_SHIFT
104
+ protected depth = BAI_DEPTH
105
+
86
106
  async _parse(opts: BaseOpts): Promise<BaiParsed> {
87
107
  const bytes = await this.filehandle.readFile(opts)
88
108
  const dataView = new DataView(
@@ -192,6 +212,7 @@ export default class BAI extends IndexFile<BaiParsed> {
192
212
  linearDataPositions[j] = (bytes[curr + 1]! << 8) | bytes[curr]!
193
213
  curr += 8
194
214
  }
215
+ fillLinearGaps(linearBlockPositions, linearDataPositions)
195
216
 
196
217
  clampChunkEnds(Object.values(binIndex).flat(), linearBlockPositions)
197
218
  return {
package/src/bamFile.ts CHANGED
@@ -12,7 +12,10 @@ import {
12
12
  BAM_MAGIC,
13
13
  MAX_CONCURRENT_CHUNK_READS,
14
14
  appendInRange,
15
+ decodeHeaderText,
16
+ optimizeChunks,
15
17
  parseRefSeqs,
18
+ readBlockSize,
16
19
  resolveFilehandle,
17
20
  throwIfAborted,
18
21
  } from './util.ts'
@@ -464,9 +467,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
464
467
  const parsed = parseRefSeqs(uncba, headLen + 8, this.renameRefSeq)
465
468
  let samHeader
466
469
  if (parsed) {
467
- const headerText = new TextDecoder('utf8').decode(
468
- uncba.subarray(8, 8 + headLen),
469
- )
470
+ const headerText = decodeHeaderText(uncba, headLen)
470
471
  this.header = headerText
471
472
  this.chrToIndex = parsed.chrToIndex
472
473
  this.indexToChr = parsed.indexToChr
@@ -539,7 +540,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
539
540
  if (chrId === undefined || !this.index) {
540
541
  return []
541
542
  }
542
- const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
543
+ const chunks = await this.index.blocksForRange(chrId, min, max, opts)
543
544
  return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts)
544
545
  }
545
546
 
@@ -818,10 +819,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
818
819
  // make the count NaN.
819
820
  const readNameCounts = new Map<string, number>()
820
821
  const readIds = new Set<number>()
822
+ const names = new Array<string>(records.length)
821
823
 
822
824
  for (let i = 0, l = records.length; i < l; i++) {
823
825
  const r = records[i]!
824
826
  const name = r.name
827
+ names[i] = name
825
828
  readNameCounts.set(name, (readNameCounts.get(name) ?? 0) + 1)
826
829
  readIds.add(r.fileOffset)
827
830
  }
@@ -829,10 +832,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
829
832
  const matePromises: Promise<Chunk[]>[] = []
830
833
  for (let i = 0, l = records.length; i < l; i++) {
831
834
  const f = records[i]!
832
- const name = f.name
833
835
  if (
834
836
  this.index &&
835
- readNameCounts.get(name) === 1 &&
837
+ readNameCounts.get(names[i]!) === 1 &&
836
838
  (pairAcrossChr ||
837
839
  (f.next_refid === chrId &&
838
840
  Math.abs(f.start - f.next_pos) < maxInsertSize))
@@ -848,20 +850,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
848
850
  }
849
851
  }
850
852
 
851
- const map = new Map<string, Chunk>()
852
- const res = await Promise.all(matePromises)
853
- for (let i = 0, l = res.length; i < l; i++) {
854
- const chunks = res[i]!
855
- for (let j = 0, jl = chunks.length; j < jl; j++) {
856
- const m = chunks[j]!
857
- // Key on the virtual-offset span — the same key _cachedChunkFeatures
858
- // uses. Chunk.toString() also folds in `bin` and fetchedSize(), which
859
- // keeps two chunks covering an identical span apart here even though
860
- // the cache below collapses them, so their records came back twice.
861
- map.set(chunkCacheKey(m), m)
862
- }
863
- }
864
- const mateChunks = [...map.values()]
853
+ // Each mate's lookup is merged on its own, so two of them can resolve to
854
+ // different spans over the same records. Merging the union again makes
855
+ // them disjoint, which is what keeps a mate in the overlap from coming
856
+ // back once per span.
857
+ const mateChunks = optimizeChunks((await Promise.all(matePromises)).flat())
865
858
 
866
859
  // Bounded for the reason ADR 0008 bounds the main query path: a viewAsPairs
867
860
  // query over a busy region resolves to many distinct mate chunks, and an
@@ -961,7 +954,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
961
954
  const hasCpositions = cpositions.length > 0
962
955
 
963
956
  while (blockStart + 4 < ba.length) {
964
- const blockSize = dataView.getInt32(blockStart, true)
957
+ const blockSize = readBlockSize(dataView, blockStart)
965
958
  const blockEnd = blockStart + 4 + blockSize - 1
966
959
 
967
960
  if (hasDpositions) {
package/src/csi.ts CHANGED
@@ -10,6 +10,7 @@ import {
10
10
  } from './util.ts'
11
11
  import { VirtualOffset, fromBytes } from './virtualOffset.ts'
12
12
 
13
+ import type { ParsedIndexBase, RefIndex } from './indexFile.ts'
13
14
  import type { BaseOpts } from './util.ts'
14
15
 
15
16
  const CSI1_MAGIC = 21582659 // CSI\1
@@ -24,10 +25,18 @@ function rshift(num: number, bits: number) {
24
25
  return Math.floor(num / 2 ** bits)
25
26
  }
26
27
 
27
- export default class CSI extends IndexFile {
28
+ interface CsiRefIndex extends RefIndex {
29
+ loffsets: Map<number, VirtualOffset>
30
+ }
31
+
32
+ interface CsiParsed extends ParsedIndexBase<CsiRefIndex> {
33
+ csi: true
34
+ }
35
+
36
+ export default class CSI extends IndexFile<CsiParsed> {
28
37
  private maxBinNumber = 0
29
- private depth = 0
30
- private minShift = 0
38
+ protected depth = 0
39
+ protected minShift = 0
31
40
 
32
41
  // CSI omits the linear index that BAI's indexCov derives coverage from
33
42
  // (CSIv1.tex §3, hts-specs), so there's no equivalent to return.
@@ -74,7 +83,7 @@ export default class CSI extends IndexFile {
74
83
  }
75
84
 
76
85
  // fetch and parse the index
77
- async _parse(opts: BaseOpts) {
86
+ async _parse(opts: BaseOpts): Promise<CsiParsed> {
78
87
  const buffer = await this.filehandle.readFile(opts)
79
88
  const bytes = await unzip(buffer)
80
89
 
@@ -93,7 +102,7 @@ export default class CSI extends IndexFile {
93
102
 
94
103
  this.minShift = dataView.getInt32(4, true)
95
104
  this.depth = dataView.getInt32(8, true)
96
- this.maxBinNumber = ((1 << ((this.depth + 1) * 3)) - 1) / 7
105
+ this.maxBinNumber = (8 ** (this.depth + 1) - 1) / 7
97
106
  const maxBinNumber = this.maxBinNumber
98
107
  const auxLength = dataView.getInt32(12, true)
99
108
  // A tabix-only branch, which is why parseAuxData and the parseNameBytes it
@@ -128,10 +137,10 @@ export default class CSI extends IndexFile {
128
137
  if (bin > this.maxBinNumber) {
129
138
  curr += 28 + 16
130
139
  } else {
131
- // A bin's loffset is the smallest virtual offset of any record in it,
132
- // so the minimum over loffsets is already the minimum over the bin's
133
- // chunks — one read per bin instead of one per chunk. Checked against
134
- // every .csi in test/data: same answer on all 19.
140
+ // A bin's loffset is the linear-index entry at its first window, so
141
+ // the smallest over every bin is the first record's offset — one
142
+ // read per bin instead of one per chunk. Checked against every .csi
143
+ // in test/data: same answer on all 19.
135
144
  firstDataLine = minVirtualOffset(bytes, curr, 1, firstDataLine)
136
145
  curr += 8 // loffset
137
146
  const chunkCount = dataView.getInt32(curr, true)
@@ -149,6 +158,7 @@ export default class CSI extends IndexFile {
149
158
  const binCount = dataView.getInt32(curr, true)
150
159
  curr += 4
151
160
  const binIndex: Record<number, Chunk[]> = {}
161
+ const loffsets = new Map<number, VirtualOffset>()
152
162
  let pseudoBinStats
153
163
  for (let j = 0; j < binCount; j++) {
154
164
  const bin = dataView.getUint32(curr, true)
@@ -157,7 +167,8 @@ export default class CSI extends IndexFile {
157
167
  pseudoBinStats = parsePseudoBin(bytes, curr + 28)
158
168
  curr += 28 + 16
159
169
  } else {
160
- curr += 8 // skip loffset; firstDataLine was computed in the first pass
170
+ loffsets.set(bin, fromBytes(bytes, curr))
171
+ curr += 8
161
172
  const chunkCount = dataView.getInt32(curr, true)
162
173
  curr += 4
163
174
  const chunks = new Array<Chunk>(chunkCount)
@@ -175,6 +186,7 @@ export default class CSI extends IndexFile {
175
186
  clampChunkEnds(Object.values(binIndex).flat())
176
187
  return {
177
188
  binIndex,
189
+ loffsets,
178
190
  stats: pseudoBinStats,
179
191
  }
180
192
  }
@@ -188,12 +200,27 @@ export default class CSI extends IndexFile {
188
200
  }
189
201
  }
190
202
 
191
- // CSI has no linear index — every refId starts from the beginning of file.
192
- protected getLowestChunk() {
193
- return ZERO_OFFSET
203
+ /**
204
+ * CSI has no linear index, but each bin's `loffset` is the linear-index entry
205
+ * at the bin's first window (htslib's `update_loff`), so the finest bin at or
206
+ * left of `min` bounds the query from below the way BAI's linear index does.
207
+ * Walks left through siblings and then up to the parent, as `hts_itr_query`
208
+ * does for CSI.
209
+ */
210
+ protected getLowestChunk(refIndex: CsiRefIndex, min: number) {
211
+ const { loffsets } = refIndex
212
+ const leaves = 8 ** this.depth
213
+ let bin =
214
+ (leaves - 1) / 7 +
215
+ Math.min(Math.floor(min / 2 ** this.minShift), leaves - 1)
216
+ while (bin > 0 && !loffsets.has(bin)) {
217
+ const parent = Math.floor((bin - 1) / 8)
218
+ bin = bin > parent * 8 + 1 ? bin - 1 : parent
219
+ }
220
+ return loffsets.get(bin) ?? ZERO_OFFSET
194
221
  }
195
222
 
196
- // ...and so there is nothing to bound the far end of a query with either, so
223
+ // No linear index means nothing to forecast the far end of a query with, so
197
224
  // estimatedBytesForRegions keeps summing every chunk on a CSI-indexed file.
198
225
  protected getHighestChunk() {
199
226
  return undefined
package/src/indexFile.ts CHANGED
@@ -2,6 +2,7 @@ import { SharedReadCache } from '@gmod/shared-read-cache'
2
2
  import QuickLRU from '@jbrowse/quick-lru'
3
3
 
4
4
  import { chunksLikelyRead, optimizeChunks } from './util.ts'
5
+ import { compareOffsets } from './virtualOffset.ts'
5
6
 
6
7
  import type Chunk from './chunk.ts'
7
8
  import type { BaseOpts } from './util.ts'
@@ -48,6 +49,58 @@ export function memoizeByRefId<T>(
48
49
  }
49
50
  }
50
51
 
52
+ /**
53
+ * The virtual offset past which a coordinate-sorted file holds nothing
54
+ * overlapping `[.., end)`, from the binning index alone: htslib's `max_off`
55
+ * (`hts_itr_query` in hts.c).
56
+ *
57
+ * Walk right from the finest bin after the one holding `end - 1`, stepping up
58
+ * to the parent at every first child, so each bin visited begins at or past
59
+ * `end` and never overlaps the query. Every record in such a bin starts at or
60
+ * past `end`, so the first chunk of the first bin that exists is a record past
61
+ * the query, and in a sorted file so is every record after it.
62
+ *
63
+ * A bound, unlike the linear-index forecast in `chunksLikelyRead`: it rests on
64
+ * the same sort order `appendInRange` and the early stop already assume, and
65
+ * cannot drop a record they would keep. See ADR 0023 for why the caller drops
66
+ * whole merged chunks with it rather than trimming them.
67
+ */
68
+ export function maxOffset(
69
+ binIndex: Record<number, Chunk[]>,
70
+ end: number,
71
+ minShift: number,
72
+ depth: number,
73
+ ) {
74
+ if (end > 2 ** (minShift + depth * 3)) {
75
+ return undefined
76
+ }
77
+ const binCount = (8 ** (depth + 1) - 1) / 7
78
+ let bin = (8 ** depth - 1) / 7 + Math.floor((end - 1) / 2 ** minShift) + 1
79
+ if (bin >= binCount) {
80
+ bin = 0
81
+ }
82
+ for (;;) {
83
+ while (bin % 8 === 1) {
84
+ bin = (bin - 1) / 8
85
+ }
86
+ if (bin === 0) {
87
+ return undefined
88
+ }
89
+ const chunks = binIndex[bin]
90
+ if (chunks?.length) {
91
+ let lowest = chunks[0]!.minv
92
+ for (let i = 1; i < chunks.length; i++) {
93
+ const minv = chunks[i]!.minv
94
+ if (compareOffsets(minv, lowest) < 0) {
95
+ lowest = minv
96
+ }
97
+ }
98
+ return lowest
99
+ }
100
+ bin++
101
+ }
102
+ }
103
+
51
104
  export default abstract class IndexFile<
52
105
  TParsed extends ParsedIndexBase = ParsedIndexBase,
53
106
  > {
@@ -79,6 +132,11 @@ export default abstract class IndexFile<
79
132
  end?: number,
80
133
  ): Promise<{ start: number; end: number; score: number }[]>
81
134
 
135
+ // The binning scheme: the finest bins are 2^minShift wide, and there are
136
+ // depth levels below bin 0. BAI is CSI with minShift 14 and depth 5.
137
+ protected abstract minShift: number
138
+ protected abstract depth: number
139
+
82
140
  // Bin numbers that overlap [min, max). Subclasses implement BAI's fixed
83
141
  // 5-level scheme or CSI's configurable scheme (SAMv1.pdf §5.1.1, CSIv1.tex §2).
84
142
  protected abstract reg2bins(
@@ -87,7 +145,8 @@ export default abstract class IndexFile<
87
145
  ): readonly (readonly [number, number])[]
88
146
 
89
147
  // Lower-bound virtual offset for chunks that could contain alignments in
90
- // [min, ...). BAI uses its linear index; CSI has none and returns 0:0.
148
+ // [min, ...). BAI uses its linear index, CSI the loffset of a bin at or left
149
+ // of min.
91
150
  protected abstract getLowestChunk(
92
151
  refIndex: RefIndex,
93
152
  min: number,
@@ -133,7 +192,16 @@ export default abstract class IndexFile<
133
192
  }
134
193
  }
135
194
  }
136
- return optimizeChunks(chunks, this.getLowestChunk(ba, min))
195
+ const merged = optimizeChunks(chunks, this.getLowestChunk(ba, min))
196
+ const past = maxOffset(binIndex, max, this.minShift, this.depth)
197
+ if (past) {
198
+ let n = merged.length
199
+ while (n > 0 && compareOffsets(merged[n - 1]!.minv, past) >= 0) {
200
+ n--
201
+ }
202
+ merged.length = n
203
+ }
204
+ return merged
137
205
  }
138
206
 
139
207
  // SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same shape and the same
@@ -197,7 +265,8 @@ export default abstract class IndexFile<
197
265
  * exists — was being told 5.6x the truth on exactly the windows a reader
198
266
  * spends their time in, and cannot answer it by zooming: every window narrower
199
267
  * than a linear-index interval resolves to the same chunks and so to the same
200
- * number.
268
+ * number. The table predates `max_off` (ADR 0023), which now drops most of
269
+ * the gap between the first two columns from `blocksForRange` itself.
201
270
  *
202
271
  * Still summed over merged chunks rather than per region, so two regions
203
272
  * sharing a chunk are charged for it once.
package/src/record.ts CHANGED
@@ -418,9 +418,14 @@ export default class BamRecord {
418
418
  return this._dataView.getInt32(this._start + 8, true)
419
419
  }
420
420
 
421
+ // The end htslib's bam_endpos() reports: a record consuming no reference —
422
+ // an unmapped mate placed at its mate's coordinate, or an empty CIGAR —
423
+ // still covers one base rather than none, so a consumer's interval search on
424
+ // the base it sits at can find it (endpos() in util.ts applies the same rule
425
+ // to this library's own range filter).
421
426
  get end() {
422
427
  if (this._cachedEnd === undefined) {
423
- this._cachedEnd = this.start + this.length_on_ref
428
+ this._cachedEnd = this.start + Math.max(this.length_on_ref, 1)
424
429
  }
425
430
  return this._cachedEnd
426
431
  }
@@ -435,16 +440,18 @@ export default class BamRecord {
435
440
  }
436
441
 
437
442
  // QUAL is present whenever the record has bases — independent of the unmapped
438
- // flag (unmapped reads routinely carry SEQ/QUAL). A zero-length SEQ means
439
- // there is no quality to return.
443
+ // flag (unmapped reads routinely carry SEQ/QUAL). null for a zero-length SEQ,
444
+ // and for a QUAL of `*`, which BAM stores as 0xFF in every byte (SAMv1
445
+ // §4.2.3). htslib decides on the first byte alone, and so does this.
440
446
  get qual() {
441
447
  const seqLen = this.seq_length
442
448
  if (seqLen === 0) {
443
449
  return null
444
- } else {
445
- const p = this.seqStart + ((seqLen + 1) >> 1)
446
- return this._byteArray.subarray(p, p + seqLen)
447
450
  }
451
+ const p = this.seqStart + ((seqLen + 1) >> 1)
452
+ return this._byteArray[p] === 0xff
453
+ ? null
454
+ : this._byteArray.subarray(p, p + seqLen)
448
455
  }
449
456
 
450
457
  get strand() {
@@ -470,7 +477,7 @@ export default class BamRecord {
470
477
  return this.seqStart + ((seqLen + 1) >> 1) + seqLen
471
478
  }
472
479
 
473
- // batch fromCharCode: fastest for typical name lengths (see benchmarks/string-building.bench.ts)
480
+ // batch fromCharCode: fastest for typical name lengths (benchmarks/string-building.bench.ts, removed in 36a3968)
474
481
  //
475
482
  // Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
476
483
  // read name about once — jbrowse-components' buildBaseFeatureData copies it
@@ -505,8 +512,8 @@ export default class BamRecord {
505
512
  }
506
513
 
507
514
  getTag(tagName: string) {
508
- if (this._cachedTags !== undefined) {
509
- return this._cachedTags[tagName]
515
+ if (this._cachedTags !== undefined || tagName === 'CG') {
516
+ return this.tags[tagName]
510
517
  }
511
518
  return this._findTag(tagName, false)
512
519
  }
@@ -616,6 +623,9 @@ export default class BamRecord {
616
623
  )
617
624
  p = end
618
625
  }
626
+ if (this._hasCGPlaceholder() && isNumericCigar(tags.CG)) {
627
+ delete tags.CG
628
+ }
619
629
  return tags
620
630
  }
621
631
 
@@ -667,7 +677,7 @@ export default class BamRecord {
667
677
  return !!(this.flags & Constants.BAM_FSUPPLEMENTARY)
668
678
  }
669
679
 
670
- // Benchmark results for CIGAR parsing strategies (see benchmarks/cigar-lifecycle.bench.ts):
680
+ // Benchmark results for CIGAR parsing strategies (benchmarks/cigar-strategies.bench.ts, removed in 36a3968):
671
681
  //
672
682
  // Aligned data:
673
683
  // - Plain array is 1.6-1.8x faster than Uint32Array for small CIGARs (≤50 ops)
@@ -691,12 +701,23 @@ export default class BamRecord {
691
701
  // htslib stores the placeholder as exactly two ops: <seqlen>S<reflen>N.
692
702
  if (numCigarOps === 2) {
693
703
  const cigop = this._dataView.getInt32(p, true)
694
- return (cigop & 0xf) === CIGAR_SOFT_CLIP && cigop >> 4 === this.seq_length
704
+ return (
705
+ (cigop & 0xf) === CIGAR_SOFT_CLIP && cigop >>> 4 === this.seq_length
706
+ )
695
707
  } else {
696
708
  return false
697
709
  }
698
710
  }
699
711
 
712
+ // SAMv1 §4.2.2: with the placeholder and a CG tag, the tag is the CIGAR and
713
+ // a reader removes it from the tags, as htslib does
714
+ private _hasCGPlaceholder() {
715
+ return this._isCGTagPattern(
716
+ this.b0 + this.read_name_length,
717
+ this.flag_nc & 0xffff,
718
+ )
719
+ }
720
+
700
721
  private _computeLengthOnRef(): number {
701
722
  const flag_nc = this._dataView.getInt32(this._start + 16, true)
702
723
  if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
@@ -711,7 +732,7 @@ export default class BamRecord {
711
732
  if ((cigop2 & 0xf) !== CIGAR_REF_SKIP) {
712
733
  console.warn('CG tag with no N tag')
713
734
  }
714
- return cigop2 >> 4
735
+ return cigop2 >>> 4
715
736
  }
716
737
 
717
738
  const absOffset = this._byteArray.byteOffset + p
@@ -727,7 +748,7 @@ export default class BamRecord {
727
748
  let lref = 0
728
749
  for (let c = 0; c < numCigarOps; ++c) {
729
750
  const co = cigarView[c]!
730
- lref += (co >> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
751
+ lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
731
752
  }
732
753
  return lref
733
754
  }
@@ -735,7 +756,7 @@ export default class BamRecord {
735
756
  let lref = 0
736
757
  for (let c = 0; c < numCigarOps; ++c) {
737
758
  const co = this._dataView.getInt32(p + c * 4, true)
738
- lref += (co >> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
759
+ lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
739
760
  }
740
761
  return lref
741
762
  }
@@ -756,10 +777,10 @@ export default class BamRecord {
756
777
  const p = this.b0 + this.read_name_length
757
778
 
758
779
  if (this._isCGTagPattern(p, numCigarOps)) {
759
- // getTag, not this.tags: the real CIGAR lives in one tag, so there's no
760
- // reason to decode every other tag on the record to reach it
761
- const cg = this.getTag('CG')
762
- return isNumericCigar(cg) ? cg : new Uint32Array(0)
780
+ const cg = this._findTag('CG', false)
781
+ if (isNumericCigar(cg)) {
782
+ return cg
783
+ }
763
784
  }
764
785
 
765
786
  const absOffset = this._byteArray.byteOffset + p
@@ -806,18 +827,21 @@ export default class BamRecord {
806
827
  let result = ''
807
828
  for (let i = 0, l = numeric.length; i < l; i++) {
808
829
  const packed = numeric[i]!
809
- result += packed >> 4
830
+ result += packed >>> 4
810
831
  result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
811
832
  }
812
833
  return result
813
834
  }
814
835
 
815
836
  get num_cigar_ops() {
816
- return this.flag_nc & 0xffff
837
+ return this._hasCGPlaceholder()
838
+ ? this.NUMERIC_CIGAR.length
839
+ : this.flag_nc & 0xffff
817
840
  }
818
841
 
842
+ // the stored CIGAR field, which for a long CIGAR is the two-op placeholder
819
843
  get num_cigar_bytes() {
820
- return this.num_cigar_ops << 2
844
+ return (this.flag_nc & 0xffff) << 2
821
845
  }
822
846
 
823
847
  get read_name_length() {
package/src/streamBam.ts CHANGED
@@ -9,7 +9,9 @@ import { parseHeaderText } from './sam.ts'
9
9
  import {
10
10
  BAM_MAGIC,
11
11
  concatUint8Array,
12
+ decodeHeaderText,
12
13
  parseRefSeqs,
14
+ readBlockSize,
13
15
  resolveFilehandle,
14
16
  throwIfAborted,
15
17
  } from './util.ts'
@@ -273,9 +275,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
273
275
  recordCarry = bytes
274
276
  continue
275
277
  }
276
- const headerText = new TextDecoder('utf8').decode(
277
- bytes.subarray(8, 8 + lText),
278
- )
278
+ const headerText = decodeHeaderText(bytes, lText)
279
279
  onHeader?.({
280
280
  headerText,
281
281
  samHeader: parseHeaderText(headerText),
@@ -288,7 +288,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
288
288
 
289
289
  const sink: T[] = []
290
290
  while (blockStart + 4 <= bytes.length) {
291
- const blockSize = dataView.getInt32(blockStart, true)
291
+ const blockSize = readBlockSize(dataView, blockStart)
292
292
  const blockEnd = blockStart + 4 + blockSize - 1
293
293
  if (blockEnd >= bytes.length) {
294
294
  break
package/src/util.ts CHANGED
@@ -9,6 +9,22 @@ import type { GenericFilehandle } from 'generic-filehandle2'
9
9
  /** 'BAM\1' read as a little-endian int32 */
10
10
  export const BAM_MAGIC = 21840194
11
11
 
12
+ /**
13
+ * The `block_size` of the record at `offset`, rejecting one too small to hold a
14
+ * record's 32 bytes of fixed fields, as htslib's `bam_read1` does. A negative
15
+ * one would otherwise leave the loops that advance by it standing still,
16
+ * pushing a record per turn until the process runs out of memory.
17
+ */
18
+ export function readBlockSize(dataView: DataView, offset: number) {
19
+ const blockSize = dataView.getInt32(offset, true)
20
+ if (blockSize < 32) {
21
+ throw new Error(
22
+ `corrupt BAM record: block_size ${blockSize} at byte ${offset}`,
23
+ )
24
+ }
25
+ return blockSize
26
+ }
27
+
12
28
  export function resolveFilehandle(
13
29
  filehandle?: GenericFilehandle,
14
30
  path?: string,
@@ -66,10 +82,13 @@ export const MAX_CONCURRENT_CHUNK_READS = 6
66
82
  *
67
83
  * `blocksForRange` returns every chunk of every bin overlapping the query, at
68
84
  * every level of the binning scheme, minus the ones the linear index puts
69
- * entirely before it. On a long-read file that is wildly more than the query
70
- * reads: a coarse bin's chunks run to the end of the bin's span, so a 380bp
71
- * window on a deep ONT BAM resolves to 90 chunks / 43.5MB, of which
72
- * `getRecordsForRange` reads 6 / 7.8MB before the early stop fires.
85
+ * entirely before it. Until `max_off` (ADR 0023) also dropped the ones past it,
86
+ * that was wildly more than a long-read query reads: a coarse bin's chunks run
87
+ * to the end of the bin's span, so a 380bp window on a deep ONT BAM resolved to
88
+ * 90 chunks / 43.5MB, of which `getRecordsForRange` read 6 / 7.8MB before the
89
+ * early stop fired. Since then this changes the forecast in 4 of 320 fixture
90
+ * windows, undershooting in all 4; ADR 0023 says what to check before
91
+ * removing it.
73
92
  *
74
93
  * Two bounds, and the answer is the larger:
75
94
  *
@@ -325,6 +344,16 @@ export function clampChunkEnds(
325
344
  }
326
345
  }
327
346
 
347
+ // SAMv1 §4.2 counts NUL padding in l_text, and htslib reads the text as a C
348
+ // string, so the header ends at the first NUL.
349
+ export function decodeHeaderText(bytes: Uint8Array, lText: number) {
350
+ const text = bytes.subarray(8, 8 + lText)
351
+ const nul = text.indexOf(0)
352
+ return new TextDecoder('utf8').decode(
353
+ nul === -1 ? text : text.subarray(0, nul),
354
+ )
355
+ }
356
+
328
357
  // Parse the BAM reference-sequence table (SAMv1.pdf §4.2). Returns undefined
329
358
  // if `uncba` doesn't yet contain the full table — caller fetches more bytes
330
359
  // and retries.
@@ -35,3 +35,7 @@ export function fromBytes(bytes: Uint8Array, offset = 0) {
35
35
  (bytes[offset + 1]! << 8) | bytes[offset]!,
36
36
  )
37
37
  }
38
+
39
+ export function compareOffsets(a: OffsetCoords, b: OffsetCoords) {
40
+ return a.blockPosition - b.blockPosition || a.dataPosition - b.dataPosition
41
+ }