@gmod/bam 7.8.1 → 7.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/util.ts CHANGED
@@ -29,6 +29,16 @@ export interface BaseOpts {
29
29
  onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
30
30
  }
31
31
 
32
+ /**
33
+ * Merge and order the chunks a query resolved to.
34
+ *
35
+ * Takes ownership of `chunks`: with no `lowest` to pre-filter against it sorts
36
+ * the array IN PLACE rather than copying. Every caller builds a fresh array to
37
+ * hand over, which is what makes that safe — passing something you still hold,
38
+ * or anything reachable from the index's per-refId cache, would reorder it
39
+ * underneath you. (The Chunk objects themselves are never mutated; a merged
40
+ * span produces a new instance.)
41
+ */
32
42
  export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
33
43
  const n = chunks.length
34
44
  if (n === 0) {
@@ -74,6 +84,14 @@ export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
74
84
  const chunkMaxBlock = chunk.maxv.blockPosition
75
85
  // Merge if chunks are close enough: small gap between them, and the
76
86
  // combined span is bounded so we don't grow a single chunk indefinitely.
87
+ //
88
+ // Both constants were swept before being left alone — see ADR 0011.
89
+ // Dropping merging entirely, on the theory that a caller with a coalescing
90
+ // range cache makes it redundant, is much worse: a bare consumer goes from
91
+ // 6 reads to 95-378 on the same queries AND downloads MORE, because every
92
+ // small chunk pays its own tail padding where a merged one amortizes it.
93
+ // Raising the gap is worse too, partly because it blunts the early stop in
94
+ // _fetchChunkFeatures. Records are identical either way; only the I/O moves.
77
95
  if (
78
96
  chunkMinBlock - lastMaxBlock < 65000 &&
79
97
  chunkMaxBlock - lastMinBlock < 5000000
@@ -177,19 +195,14 @@ export function clampChunkEnds(
177
195
  if (chunks.length === 0) {
178
196
  return
179
197
  }
180
- // A plain array, sized once and filled by index. Not a Float64Array: this
181
- // runs once per reference, and an assembly with tens of thousands of unplaced
182
- // scaffolds (cho.bam.bai has 28751 references) pays the allocation that many
183
- // times over, where each individual boundary list is a handful of entries.
184
- // The small-allocation cost dominates the faster typed sort by a wide margin
185
- // at that shape. Filling by index still avoids the intermediate array that
186
- // mapping the linear index to block positions used to build.
187
- // Pre-sized and filled by index. Measured against both alternatives on five
188
- // real .bai shapes (min of 21, interleaved, sign-stable over 3 runs):
189
- // building with push instead is 5-8% slower here, and a Float64Array is
190
- // faster to sort but allocates per reference, which costs 1.7x on an
191
- // assembly with 28751 scaffolds. Filling by index also avoids the
192
- // intermediate array that mapping the linear index used to build.
198
+ // A plain array, pre-sized and filled by index. Measured against both
199
+ // alternatives on five real .bai shapes (min of 21, interleaved, sign-stable
200
+ // over 3 runs): building with push instead is 5-8% slower, and a
201
+ // Float64Array is faster to sort but allocates per reference, which costs
202
+ // 1.7x on an assembly with tens of thousands of unplaced scaffolds
203
+ // (cho.bam.bai has 28751 references, each with a handful of boundaries).
204
+ // Filling by index also avoids the intermediate array that mapping the
205
+ // linear index to block positions used to build.
193
206
  const boundaries = new Array<number>(
194
207
  extraBoundaries.length + chunks.length * 2,
195
208
  )
@@ -361,10 +374,23 @@ interface Positioned {
361
374
  end: number
362
375
  }
363
376
 
377
+ // The end htslib's bam_endpos() reports: a record consuming no reference —
378
+ // an unmapped mate placed at its mate's coordinate, or an empty CIGAR — still
379
+ // covers one base rather than none, so it can be found by a query on the base
380
+ // it sits at.
381
+ function endpos(r: Positioned) {
382
+ return r.end > r.start ? r.end : r.start + 1
383
+ }
384
+
364
385
  // Append records overlapping [min, max) on `chrId` into `out` (or a fresh
365
386
  // array if omitted). Records are assumed coordinate-sorted (by ref_id, then
366
387
  // start), so we stop scanning once we pass `max` within `chrId` or move past
367
388
  // `chrId` entirely. Returns the populated array.
389
+ //
390
+ // `end` is exclusive, so overlap is `end > min`, not `end >= min`: a read
391
+ // finishing exactly where the query begins shares no base with it. samtools
392
+ // agrees — `samtools view f.bam chr:124001-124300` omits a 150M read at
393
+ // 1-based POS 123851, which ends at 124000.
368
394
  export function appendInRange<T extends Positioned>(
369
395
  records: T[],
370
396
  chrId: number,
@@ -377,7 +403,7 @@ export function appendInRange<T extends Positioned>(
377
403
  if (r.ref_id === chrId) {
378
404
  if (r.start >= max) {
379
405
  break
380
- } else if (r.end >= min) {
406
+ } else if (endpos(r) > min) {
381
407
  out.push(r)
382
408
  }
383
409
  } else if (r.ref_id > chrId) {