@gmod/bam 7.3.4 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/bamFile.ts CHANGED
@@ -1,5 +1,4 @@
1
1
  import { unzip, unzipChunkSlice } from '@gmod/bgzf-filehandle'
2
- import QuickLRU from '@jbrowse/quick-lru'
3
2
  import crc32 from 'crc/calculators/crc32'
4
3
  import { LocalFile, RemoteFile } from 'generic-filehandle2'
5
4
 
@@ -8,16 +7,11 @@ import CSI from './csi.ts'
8
7
  import NullFilehandle from './nullFilehandle.ts'
9
8
  import BAMFeature from './record.ts'
10
9
  import { parseHeaderText } from './sam.ts'
11
- import {
12
- appendInRange,
13
- applyFilters,
14
- filterCacheKey,
15
- parseRefSeqs,
16
- } from './util.ts'
10
+ import { appendInRange, parseRefSeqs } from './util.ts'
17
11
 
18
12
  import type Chunk from './chunk.ts'
19
13
  import type { Bytes } from './record.ts'
20
- import type { BamOpts, BaseOpts, FilterBy } from './util.ts'
14
+ import type { BamOpts, BaseOpts } from './util.ts'
21
15
  import type { GenericFilehandle } from 'generic-filehandle2'
22
16
 
23
17
  export interface BamRecordLike {
@@ -54,14 +48,81 @@ function resolveFilehandle(
54
48
  }
55
49
 
56
50
  interface ChunkEntry<T> {
57
- minBlock: number
58
- maxBlock: number
51
+ // decompressed size of the chunk these features are views into
52
+ bytes: number
59
53
  features: T[]
60
54
  }
61
55
 
62
- function chunkCacheKey(chunk: Chunk, filterBy?: FilterBy) {
56
+ function chunkCacheKey(chunk: Chunk) {
63
57
  const { minv, maxv } = chunk
64
- return `${minv.blockPosition}:${minv.dataPosition}-${maxv.blockPosition}:${maxv.dataPosition}${filterCacheKey(filterBy)}`
58
+ return `${minv.blockPosition}:${minv.dataPosition}-${maxv.blockPosition}:${maxv.dataPosition}`
59
+ }
60
+
61
+ // Every record in an entry is a view into its chunk's decompressed buffer, so
62
+ // caching one entry pins that whole buffer — 8MB apiece on the nanopore and
63
+ // 2kb-read test files, and optimizeChunks merges spans up to 5MB *compressed*,
64
+ // so tens of MB is possible. A count-based LRU therefore gives no bound on
65
+ // memory at all, which is why this budgets by decompressed bytes instead. It is
66
+ // the *only* bound: a query keeps every chunk it parsed, since a chunk dropped
67
+ // here costs a re-download and re-decompress the next time the view moves.
68
+ export const DEFAULT_MAX_CACHE_BYTES = 100 * 1024 * 1024
69
+
70
+ class ChunkFeatureCache<T> {
71
+ public maxBytes: number
72
+ private entries = new Map<string, ChunkEntry<T>>()
73
+ private bytes = 0
74
+
75
+ constructor(maxBytes: number) {
76
+ this.maxBytes = maxBytes
77
+ }
78
+
79
+ get size() {
80
+ return this.entries.size
81
+ }
82
+
83
+ get byteSize() {
84
+ return this.bytes
85
+ }
86
+
87
+ get(key: string) {
88
+ const entry = this.entries.get(key)
89
+ if (entry) {
90
+ // re-insert so Map iteration order stays least-recently-used first
91
+ this.entries.delete(key)
92
+ this.entries.set(key, entry)
93
+ }
94
+ return entry
95
+ }
96
+
97
+ set(key: string, entry: ChunkEntry<T>) {
98
+ this.delete(key)
99
+ this.entries.set(key, entry)
100
+ this.bytes += entry.bytes
101
+ // Evict from the least-recently-used end. The size > 1 guard means a single
102
+ // chunk larger than the whole budget is still kept: the caller needs it for
103
+ // the query in flight, and dropping it would only force a re-decompress.
104
+ const lru = this.entries.keys()
105
+ while (this.bytes > this.maxBytes && this.entries.size > 1) {
106
+ this.delete(lru.next().value!)
107
+ }
108
+ }
109
+
110
+ delete(key: string) {
111
+ const entry = this.entries.get(key)
112
+ if (entry) {
113
+ this.entries.delete(key)
114
+ this.bytes -= entry.bytes
115
+ }
116
+ }
117
+
118
+ clear() {
119
+ this.entries.clear()
120
+ this.bytes = 0
121
+ }
122
+
123
+ [Symbol.iterator]() {
124
+ return this.entries[Symbol.iterator]()
125
+ }
65
126
  }
66
127
 
67
128
  export default class BamFile<T extends BamRecordLike = BAMFeature> {
@@ -74,11 +135,8 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
74
135
  public htsget = false
75
136
  public headerP?: ReturnType<BamFile<T>['getHeaderPre']>
76
137
 
77
- // Cache for parsed features by chunk
78
- // When a new chunk overlaps a cached chunk, we evict the cached one
79
- public chunkFeatureCache = new QuickLRU<string, ChunkEntry<T>>({
80
- maxSize: 100,
81
- })
138
+ // Cache for parsed features by chunk, bounded by decompressed bytes
139
+ public chunkFeatureCache: ChunkFeatureCache<T>
82
140
 
83
141
  private RecordClass: BamRecordClass<T>
84
142
 
@@ -95,6 +153,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
95
153
  htsget,
96
154
  renameRefSeqs = n => n,
97
155
  recordClass,
156
+ maxCacheBytes = DEFAULT_MAX_CACHE_BYTES,
98
157
  }: {
99
158
  bamFilehandle?: GenericFilehandle
100
159
  bamPath?: string
@@ -108,9 +167,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
108
167
  renameRefSeqs?: (a: string) => string
109
168
  htsget?: boolean
110
169
  recordClass?: BamRecordClass<T>
170
+ /** budget for the parsed-chunk cache, in decompressed bytes */
171
+ maxCacheBytes?: number
111
172
  }) {
112
173
  this.renameRefSeq = renameRefSeqs
113
174
  this.RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
175
+ this.chunkFeatureCache = new ChunkFeatureCache<T>(maxCacheBytes)
114
176
 
115
177
  const bamFh = resolveFilehandle(bamFilehandle, bamPath, bamUrl)
116
178
  if (bamFh) {
@@ -162,7 +224,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
162
224
  ? await this.bam.readFile()
163
225
  : await this.bam.read(readLen, 0)
164
226
  let uncba = await unzip(buffer)
165
- const dataView = new DataView(uncba.buffer)
227
+ const dataView = new DataView(
228
+ uncba.buffer,
229
+ uncba.byteOffset,
230
+ uncba.byteLength,
231
+ )
166
232
 
167
233
  if (dataView.getInt32(0, true) !== BAM_MAGIC) {
168
234
  throw new Error('Not a BAM file')
@@ -221,26 +287,31 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
221
287
  return []
222
288
  }
223
289
  const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
224
- return this._fetchChunkFeaturesDirect(chunks, chrId, min, max, opts)
290
+ return this._fetchChunkFeatures(chunks, chrId, min, max, opts)
225
291
  }
226
292
 
227
- // Evict any cached chunks whose block range overlaps [minBlock, maxBlock]
228
- private evictOverlappingChunks(minBlock: number, maxBlock: number) {
229
- for (const [key, entry] of this.chunkFeatureCache) {
230
- if (minBlock <= entry.maxBlock && maxBlock >= entry.minBlock) {
231
- this.chunkFeatureCache.delete(key)
232
- }
293
+ // Parsed records for a chunk, reading and decompressing it only on a miss.
294
+ // Every path that wants a chunk's features goes through here — mate lookups
295
+ // included, since a viewAsPairs query revisits the same mate chunks each time
296
+ // the view moves.
297
+ private async _cachedChunkFeatures(chunk: Chunk, opts: BaseOpts) {
298
+ const cacheKey = chunkCacheKey(chunk)
299
+ let entry = this.chunkFeatureCache.get(cacheKey)
300
+ if (!entry) {
301
+ entry = await this._readChunkFeatures(chunk, opts)
302
+ this.chunkFeatureCache.set(cacheKey, entry)
233
303
  }
304
+ return entry.features
234
305
  }
235
306
 
236
- private async _fetchChunkFeaturesDirect(
307
+ private async _fetchChunkFeatures(
237
308
  chunks: Chunk[],
238
309
  chrId: number,
239
310
  min: number,
240
311
  max: number,
241
312
  opts: BamOpts = {},
242
313
  ) {
243
- const { viewAsPairs, filterBy, onProgress } = opts
314
+ const { viewAsPairs, onProgress } = opts
244
315
  const result: T[] = []
245
316
 
246
317
  let totalBytes = 0
@@ -252,28 +323,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
252
323
 
253
324
  for (let ci = 0, cl = chunks.length; ci < cl; ci++) {
254
325
  const chunk = chunks[ci]!
255
- const cacheKey = chunkCacheKey(chunk, filterBy)
256
- const minBlock = chunk.minv.blockPosition
257
- const maxBlock = chunk.maxv.blockPosition
258
-
259
- let records: T[]
260
- const cached = this.chunkFeatureCache.get(cacheKey)
261
- if (cached) {
262
- records = cached.features
263
- } else {
264
- this.evictOverlappingChunks(minBlock, maxBlock)
265
- const allRecords = await this._readChunkFeatures(chunk, opts)
266
- records = filterBy ? applyFilters(allRecords, filterBy) : allRecords
267
- this.chunkFeatureCache.set(cacheKey, {
268
- minBlock,
269
- maxBlock,
270
- features: records,
271
- })
272
- }
326
+ const features = await this._cachedChunkFeatures(chunk, opts)
273
327
 
274
328
  downloadedBytes += chunk.fetchedSize()
275
329
  onProgress?.(downloadedBytes, totalBytes)
276
- appendInRange(records, chrId, min, max, result)
330
+ appendInRange(features, chrId, min, max, result)
277
331
  }
278
332
 
279
333
  if (viewAsPairs) {
@@ -288,13 +342,16 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
288
342
 
289
343
  async fetchPairs(chrId: number, records: T[], opts: BamOpts) {
290
344
  const { pairAcrossChr, maxInsertSize = 200000 } = opts
291
- const readNameCounts: Record<string, number> = {}
345
+ // Map, not a plain object: read names come from the file, and on a plain
346
+ // object a read named "constructor" would read back Object.prototype's and
347
+ // make the count NaN.
348
+ const readNameCounts = new Map<string, number>()
292
349
  const readIds = new Set<number>()
293
350
 
294
351
  for (let i = 0, l = records.length; i < l; i++) {
295
352
  const r = records[i]!
296
353
  const name = r.name
297
- readNameCounts[name] = (readNameCounts[name] ?? 0) + 1
354
+ readNameCounts.set(name, (readNameCounts.get(name) ?? 0) + 1)
298
355
  readIds.add(r.fileOffset)
299
356
  }
300
357
 
@@ -304,7 +361,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
304
361
  const name = f.name
305
362
  if (
306
363
  this.index &&
307
- readNameCounts[name] === 1 &&
364
+ readNameCounts.get(name) === 1 &&
308
365
  (pairAcrossChr ||
309
366
  (f.next_refid === chrId &&
310
367
  Math.abs(f.start - f.next_pos) < maxInsertSize))
@@ -332,12 +389,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
332
389
 
333
390
  const mateFeatLists = await Promise.all(
334
391
  [...map.values()].map(async c => {
335
- const features = await this._readChunkFeatures(c, opts)
392
+ const features = await this._cachedChunkFeatures(c, opts)
336
393
  const mateRecs = [] as T[]
337
394
  for (let i = 0, l = features.length; i < l; i++) {
338
395
  const feature = features[i]!
339
396
  if (
340
- readNameCounts[feature.name] === 1 &&
397
+ readNameCounts.get(feature.name) === 1 &&
341
398
  !readIds.has(feature.fileOffset)
342
399
  ) {
343
400
  mateRecs.push(feature)
@@ -355,18 +412,25 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
355
412
  // filehandle's own streaming onProgress also fired this callback it would
356
413
  // report a different `total` (this chunk's size, not the whole query),
357
414
  // making the determinate bar jump around.
358
- const buf = await this.bam.read(chunk.fetchedSize(), chunk.minv.blockPosition, {
359
- signal: opts.signal,
360
- })
415
+ const buf = await this.bam.read(
416
+ chunk.fetchedSize(),
417
+ chunk.minv.blockPosition,
418
+ {
419
+ signal: opts.signal,
420
+ },
421
+ )
361
422
  const {
362
423
  buffer: data,
363
424
  cpositions,
364
425
  dpositions,
365
426
  } = await unzipChunkSlice(buf, chunk)
366
- return this.readBamFeatures(data, cpositions, dpositions, chunk)
427
+ return {
428
+ features: this.readBamFeatures(data, cpositions, dpositions, chunk),
429
+ bytes: data.byteLength,
430
+ }
367
431
  }
368
432
 
369
- async readBamFeatures(
433
+ readBamFeatures(
370
434
  ba: Uint8Array,
371
435
  cpositions: number[],
372
436
  dpositions: number[],
@@ -376,7 +440,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
376
440
  const sink = [] as T[]
377
441
  let pos = 0
378
442
 
379
- const dataView = new DataView(ba.buffer)
443
+ const dataView = new DataView(ba.buffer, ba.byteOffset, ba.byteLength)
380
444
  const hasDpositions = dpositions.length > 0
381
445
  const hasCpositions = cpositions.length > 0
382
446
 
package/src/csi.ts CHANGED
@@ -36,7 +36,11 @@ export default class CSI extends IndexFile {
36
36
  }
37
37
 
38
38
  parseAuxData(bytes: Uint8Array, offset: number) {
39
- const dataView = new DataView(bytes.buffer)
39
+ const dataView = new DataView(
40
+ bytes.buffer,
41
+ bytes.byteOffset,
42
+ bytes.byteLength,
43
+ )
40
44
  const formatFlags = dataView.getUint32(offset, true)
41
45
  const coordinateType =
42
46
  formatFlags & 0x10000 ? 'zero-based-half-open' : '1-based-closed'
@@ -74,7 +78,11 @@ export default class CSI extends IndexFile {
74
78
  const buffer = await this.filehandle.readFile(opts)
75
79
  const bytes = await unzip(buffer)
76
80
 
77
- const dataView = new DataView(bytes.buffer)
81
+ const dataView = new DataView(
82
+ bytes.buffer,
83
+ bytes.byteOffset,
84
+ bytes.byteLength,
85
+ )
78
86
  let csiVersion
79
87
  const magic = dataView.getUint32(0, true)
80
88
 
package/src/htsget.ts CHANGED
@@ -78,7 +78,7 @@ export default class HtsgetFile<
78
78
  })
79
79
 
80
80
  const zero = new VirtualOffset(0, 0)
81
- const allRecords = await this.readBamFeatures(
81
+ const allRecords = this.readBamFeatures(
82
82
  uncba,
83
83
  [],
84
84
  [],
@@ -95,7 +95,11 @@ export default class HtsgetFile<
95
95
  const uncba = await fetchAndConcat(data.htsget.urls, {
96
96
  signal: opts.signal,
97
97
  })
98
- const dataView = new DataView(uncba.buffer)
98
+ const dataView = new DataView(
99
+ uncba.buffer,
100
+ uncba.byteOffset,
101
+ uncba.byteLength,
102
+ )
99
103
 
100
104
  if (dataView.getInt32(0, true) !== BAM_MAGIC) {
101
105
  throw new Error('Not a BAM file')
package/src/index.ts CHANGED
@@ -1,9 +1,8 @@
1
1
  export { default as BAI } from './bai.ts'
2
- export { default as BamFile } from './bamFile.ts'
2
+ export { DEFAULT_MAX_CACHE_BYTES, default as BamFile } from './bamFile.ts'
3
3
  export { default as CSI } from './csi.ts'
4
4
  export { default as BamRecord } from './record.ts'
5
5
  export { default as HtsgetFile } from './htsget.ts'
6
6
 
7
7
  export type { Bytes } from './record.ts'
8
- export type { FilterBy, TagFilter } from './util.ts'
9
8
  export type { BamRecordClass, BamRecordLike } from './bamFile.ts'