@gmod/bam 7.3.4 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -29
- package/dist/bai.js +10 -7
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +26 -9
- package/dist/bamFile.js +95 -47
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.js +2 -2
- package/dist/csi.js.map +1 -1
- package/dist/htsget.js +2 -2
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +1 -2
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/record.d.ts +2 -1
- package/dist/record.js +150 -39
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +0 -18
- package/dist/util.js +5 -36
- package/dist/util.js.map +1 -1
- package/esm/bai.js +10 -7
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +26 -9
- package/esm/bamFile.js +95 -47
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.js +2 -2
- package/esm/csi.js.map +1 -1
- package/esm/htsget.js +2 -2
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +1 -2
- package/esm/index.js +1 -1
- package/esm/index.js.map +1 -1
- package/esm/record.d.ts +2 -1
- package/esm/record.js +150 -39
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +0 -18
- package/esm/util.js +5 -32
- package/esm/util.js.map +1 -1
- package/package.json +3 -1
- package/src/bai.ts +15 -8
- package/src/bamFile.ts +121 -57
- package/src/csi.ts +10 -2
- package/src/htsget.ts +6 -2
- package/src/index.ts +1 -2
- package/src/record.ts +163 -46
- package/src/util.ts +13 -63
package/src/bamFile.ts
CHANGED
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
import { unzip, unzipChunkSlice } from '@gmod/bgzf-filehandle'
|
|
2
|
-
import QuickLRU from '@jbrowse/quick-lru'
|
|
3
2
|
import crc32 from 'crc/calculators/crc32'
|
|
4
3
|
import { LocalFile, RemoteFile } from 'generic-filehandle2'
|
|
5
4
|
|
|
@@ -8,16 +7,11 @@ import CSI from './csi.ts'
|
|
|
8
7
|
import NullFilehandle from './nullFilehandle.ts'
|
|
9
8
|
import BAMFeature from './record.ts'
|
|
10
9
|
import { parseHeaderText } from './sam.ts'
|
|
11
|
-
import {
|
|
12
|
-
appendInRange,
|
|
13
|
-
applyFilters,
|
|
14
|
-
filterCacheKey,
|
|
15
|
-
parseRefSeqs,
|
|
16
|
-
} from './util.ts'
|
|
10
|
+
import { appendInRange, parseRefSeqs } from './util.ts'
|
|
17
11
|
|
|
18
12
|
import type Chunk from './chunk.ts'
|
|
19
13
|
import type { Bytes } from './record.ts'
|
|
20
|
-
import type { BamOpts, BaseOpts
|
|
14
|
+
import type { BamOpts, BaseOpts } from './util.ts'
|
|
21
15
|
import type { GenericFilehandle } from 'generic-filehandle2'
|
|
22
16
|
|
|
23
17
|
export interface BamRecordLike {
|
|
@@ -54,14 +48,81 @@ function resolveFilehandle(
|
|
|
54
48
|
}
|
|
55
49
|
|
|
56
50
|
interface ChunkEntry<T> {
|
|
57
|
-
|
|
58
|
-
|
|
51
|
+
// decompressed size of the chunk these features are views into
|
|
52
|
+
bytes: number
|
|
59
53
|
features: T[]
|
|
60
54
|
}
|
|
61
55
|
|
|
62
|
-
function chunkCacheKey(chunk: Chunk
|
|
56
|
+
function chunkCacheKey(chunk: Chunk) {
|
|
63
57
|
const { minv, maxv } = chunk
|
|
64
|
-
return `${minv.blockPosition}:${minv.dataPosition}-${maxv.blockPosition}:${maxv.dataPosition}
|
|
58
|
+
return `${minv.blockPosition}:${minv.dataPosition}-${maxv.blockPosition}:${maxv.dataPosition}`
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// Every record in an entry is a view into its chunk's decompressed buffer, so
|
|
62
|
+
// caching one entry pins that whole buffer — 8MB apiece on the nanopore and
|
|
63
|
+
// 2kb-read test files, and optimizeChunks merges spans up to 5MB *compressed*,
|
|
64
|
+
// so tens of MB is possible. A count-based LRU therefore gives no bound on
|
|
65
|
+
// memory at all, which is why this budgets by decompressed bytes instead. It is
|
|
66
|
+
// the *only* bound: a query keeps every chunk it parsed, since a chunk dropped
|
|
67
|
+
// here costs a re-download and re-decompress the next time the view moves.
|
|
68
|
+
export const DEFAULT_MAX_CACHE_BYTES = 100 * 1024 * 1024
|
|
69
|
+
|
|
70
|
+
class ChunkFeatureCache<T> {
|
|
71
|
+
public maxBytes: number
|
|
72
|
+
private entries = new Map<string, ChunkEntry<T>>()
|
|
73
|
+
private bytes = 0
|
|
74
|
+
|
|
75
|
+
constructor(maxBytes: number) {
|
|
76
|
+
this.maxBytes = maxBytes
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
get size() {
|
|
80
|
+
return this.entries.size
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
get byteSize() {
|
|
84
|
+
return this.bytes
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
get(key: string) {
|
|
88
|
+
const entry = this.entries.get(key)
|
|
89
|
+
if (entry) {
|
|
90
|
+
// re-insert so Map iteration order stays least-recently-used first
|
|
91
|
+
this.entries.delete(key)
|
|
92
|
+
this.entries.set(key, entry)
|
|
93
|
+
}
|
|
94
|
+
return entry
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
set(key: string, entry: ChunkEntry<T>) {
|
|
98
|
+
this.delete(key)
|
|
99
|
+
this.entries.set(key, entry)
|
|
100
|
+
this.bytes += entry.bytes
|
|
101
|
+
// Evict from the least-recently-used end. The size > 1 guard means a single
|
|
102
|
+
// chunk larger than the whole budget is still kept: the caller needs it for
|
|
103
|
+
// the query in flight, and dropping it would only force a re-decompress.
|
|
104
|
+
const lru = this.entries.keys()
|
|
105
|
+
while (this.bytes > this.maxBytes && this.entries.size > 1) {
|
|
106
|
+
this.delete(lru.next().value!)
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
delete(key: string) {
|
|
111
|
+
const entry = this.entries.get(key)
|
|
112
|
+
if (entry) {
|
|
113
|
+
this.entries.delete(key)
|
|
114
|
+
this.bytes -= entry.bytes
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
clear() {
|
|
119
|
+
this.entries.clear()
|
|
120
|
+
this.bytes = 0
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
[Symbol.iterator]() {
|
|
124
|
+
return this.entries[Symbol.iterator]()
|
|
125
|
+
}
|
|
65
126
|
}
|
|
66
127
|
|
|
67
128
|
export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
@@ -74,11 +135,8 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
74
135
|
public htsget = false
|
|
75
136
|
public headerP?: ReturnType<BamFile<T>['getHeaderPre']>
|
|
76
137
|
|
|
77
|
-
// Cache for parsed features by chunk
|
|
78
|
-
|
|
79
|
-
public chunkFeatureCache = new QuickLRU<string, ChunkEntry<T>>({
|
|
80
|
-
maxSize: 100,
|
|
81
|
-
})
|
|
138
|
+
// Cache for parsed features by chunk, bounded by decompressed bytes
|
|
139
|
+
public chunkFeatureCache: ChunkFeatureCache<T>
|
|
82
140
|
|
|
83
141
|
private RecordClass: BamRecordClass<T>
|
|
84
142
|
|
|
@@ -95,6 +153,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
95
153
|
htsget,
|
|
96
154
|
renameRefSeqs = n => n,
|
|
97
155
|
recordClass,
|
|
156
|
+
maxCacheBytes = DEFAULT_MAX_CACHE_BYTES,
|
|
98
157
|
}: {
|
|
99
158
|
bamFilehandle?: GenericFilehandle
|
|
100
159
|
bamPath?: string
|
|
@@ -108,9 +167,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
108
167
|
renameRefSeqs?: (a: string) => string
|
|
109
168
|
htsget?: boolean
|
|
110
169
|
recordClass?: BamRecordClass<T>
|
|
170
|
+
/** budget for the parsed-chunk cache, in decompressed bytes */
|
|
171
|
+
maxCacheBytes?: number
|
|
111
172
|
}) {
|
|
112
173
|
this.renameRefSeq = renameRefSeqs
|
|
113
174
|
this.RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
|
|
175
|
+
this.chunkFeatureCache = new ChunkFeatureCache<T>(maxCacheBytes)
|
|
114
176
|
|
|
115
177
|
const bamFh = resolveFilehandle(bamFilehandle, bamPath, bamUrl)
|
|
116
178
|
if (bamFh) {
|
|
@@ -162,7 +224,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
162
224
|
? await this.bam.readFile()
|
|
163
225
|
: await this.bam.read(readLen, 0)
|
|
164
226
|
let uncba = await unzip(buffer)
|
|
165
|
-
const dataView = new DataView(
|
|
227
|
+
const dataView = new DataView(
|
|
228
|
+
uncba.buffer,
|
|
229
|
+
uncba.byteOffset,
|
|
230
|
+
uncba.byteLength,
|
|
231
|
+
)
|
|
166
232
|
|
|
167
233
|
if (dataView.getInt32(0, true) !== BAM_MAGIC) {
|
|
168
234
|
throw new Error('Not a BAM file')
|
|
@@ -221,26 +287,31 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
221
287
|
return []
|
|
222
288
|
}
|
|
223
289
|
const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
|
|
224
|
-
return this.
|
|
290
|
+
return this._fetchChunkFeatures(chunks, chrId, min, max, opts)
|
|
225
291
|
}
|
|
226
292
|
|
|
227
|
-
//
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
293
|
+
// Parsed records for a chunk, reading and decompressing it only on a miss.
|
|
294
|
+
// Every path that wants a chunk's features goes through here — mate lookups
|
|
295
|
+
// included, since a viewAsPairs query revisits the same mate chunks each time
|
|
296
|
+
// the view moves.
|
|
297
|
+
private async _cachedChunkFeatures(chunk: Chunk, opts: BaseOpts) {
|
|
298
|
+
const cacheKey = chunkCacheKey(chunk)
|
|
299
|
+
let entry = this.chunkFeatureCache.get(cacheKey)
|
|
300
|
+
if (!entry) {
|
|
301
|
+
entry = await this._readChunkFeatures(chunk, opts)
|
|
302
|
+
this.chunkFeatureCache.set(cacheKey, entry)
|
|
233
303
|
}
|
|
304
|
+
return entry.features
|
|
234
305
|
}
|
|
235
306
|
|
|
236
|
-
private async
|
|
307
|
+
private async _fetchChunkFeatures(
|
|
237
308
|
chunks: Chunk[],
|
|
238
309
|
chrId: number,
|
|
239
310
|
min: number,
|
|
240
311
|
max: number,
|
|
241
312
|
opts: BamOpts = {},
|
|
242
313
|
) {
|
|
243
|
-
const { viewAsPairs,
|
|
314
|
+
const { viewAsPairs, onProgress } = opts
|
|
244
315
|
const result: T[] = []
|
|
245
316
|
|
|
246
317
|
let totalBytes = 0
|
|
@@ -252,28 +323,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
252
323
|
|
|
253
324
|
for (let ci = 0, cl = chunks.length; ci < cl; ci++) {
|
|
254
325
|
const chunk = chunks[ci]!
|
|
255
|
-
const
|
|
256
|
-
const minBlock = chunk.minv.blockPosition
|
|
257
|
-
const maxBlock = chunk.maxv.blockPosition
|
|
258
|
-
|
|
259
|
-
let records: T[]
|
|
260
|
-
const cached = this.chunkFeatureCache.get(cacheKey)
|
|
261
|
-
if (cached) {
|
|
262
|
-
records = cached.features
|
|
263
|
-
} else {
|
|
264
|
-
this.evictOverlappingChunks(minBlock, maxBlock)
|
|
265
|
-
const allRecords = await this._readChunkFeatures(chunk, opts)
|
|
266
|
-
records = filterBy ? applyFilters(allRecords, filterBy) : allRecords
|
|
267
|
-
this.chunkFeatureCache.set(cacheKey, {
|
|
268
|
-
minBlock,
|
|
269
|
-
maxBlock,
|
|
270
|
-
features: records,
|
|
271
|
-
})
|
|
272
|
-
}
|
|
326
|
+
const features = await this._cachedChunkFeatures(chunk, opts)
|
|
273
327
|
|
|
274
328
|
downloadedBytes += chunk.fetchedSize()
|
|
275
329
|
onProgress?.(downloadedBytes, totalBytes)
|
|
276
|
-
appendInRange(
|
|
330
|
+
appendInRange(features, chrId, min, max, result)
|
|
277
331
|
}
|
|
278
332
|
|
|
279
333
|
if (viewAsPairs) {
|
|
@@ -288,13 +342,16 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
288
342
|
|
|
289
343
|
async fetchPairs(chrId: number, records: T[], opts: BamOpts) {
|
|
290
344
|
const { pairAcrossChr, maxInsertSize = 200000 } = opts
|
|
291
|
-
|
|
345
|
+
// Map, not a plain object: read names come from the file, and on a plain
|
|
346
|
+
// object a read named "constructor" would read back Object.prototype's and
|
|
347
|
+
// make the count NaN.
|
|
348
|
+
const readNameCounts = new Map<string, number>()
|
|
292
349
|
const readIds = new Set<number>()
|
|
293
350
|
|
|
294
351
|
for (let i = 0, l = records.length; i < l; i++) {
|
|
295
352
|
const r = records[i]!
|
|
296
353
|
const name = r.name
|
|
297
|
-
readNameCounts
|
|
354
|
+
readNameCounts.set(name, (readNameCounts.get(name) ?? 0) + 1)
|
|
298
355
|
readIds.add(r.fileOffset)
|
|
299
356
|
}
|
|
300
357
|
|
|
@@ -304,7 +361,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
304
361
|
const name = f.name
|
|
305
362
|
if (
|
|
306
363
|
this.index &&
|
|
307
|
-
readNameCounts
|
|
364
|
+
readNameCounts.get(name) === 1 &&
|
|
308
365
|
(pairAcrossChr ||
|
|
309
366
|
(f.next_refid === chrId &&
|
|
310
367
|
Math.abs(f.start - f.next_pos) < maxInsertSize))
|
|
@@ -332,12 +389,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
332
389
|
|
|
333
390
|
const mateFeatLists = await Promise.all(
|
|
334
391
|
[...map.values()].map(async c => {
|
|
335
|
-
const features = await this.
|
|
392
|
+
const features = await this._cachedChunkFeatures(c, opts)
|
|
336
393
|
const mateRecs = [] as T[]
|
|
337
394
|
for (let i = 0, l = features.length; i < l; i++) {
|
|
338
395
|
const feature = features[i]!
|
|
339
396
|
if (
|
|
340
|
-
readNameCounts
|
|
397
|
+
readNameCounts.get(feature.name) === 1 &&
|
|
341
398
|
!readIds.has(feature.fileOffset)
|
|
342
399
|
) {
|
|
343
400
|
mateRecs.push(feature)
|
|
@@ -355,18 +412,25 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
355
412
|
// filehandle's own streaming onProgress also fired this callback it would
|
|
356
413
|
// report a different `total` (this chunk's size, not the whole query),
|
|
357
414
|
// making the determinate bar jump around.
|
|
358
|
-
const buf = await this.bam.read(
|
|
359
|
-
|
|
360
|
-
|
|
415
|
+
const buf = await this.bam.read(
|
|
416
|
+
chunk.fetchedSize(),
|
|
417
|
+
chunk.minv.blockPosition,
|
|
418
|
+
{
|
|
419
|
+
signal: opts.signal,
|
|
420
|
+
},
|
|
421
|
+
)
|
|
361
422
|
const {
|
|
362
423
|
buffer: data,
|
|
363
424
|
cpositions,
|
|
364
425
|
dpositions,
|
|
365
426
|
} = await unzipChunkSlice(buf, chunk)
|
|
366
|
-
return
|
|
427
|
+
return {
|
|
428
|
+
features: this.readBamFeatures(data, cpositions, dpositions, chunk),
|
|
429
|
+
bytes: data.byteLength,
|
|
430
|
+
}
|
|
367
431
|
}
|
|
368
432
|
|
|
369
|
-
|
|
433
|
+
readBamFeatures(
|
|
370
434
|
ba: Uint8Array,
|
|
371
435
|
cpositions: number[],
|
|
372
436
|
dpositions: number[],
|
|
@@ -376,7 +440,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
376
440
|
const sink = [] as T[]
|
|
377
441
|
let pos = 0
|
|
378
442
|
|
|
379
|
-
const dataView = new DataView(ba.buffer)
|
|
443
|
+
const dataView = new DataView(ba.buffer, ba.byteOffset, ba.byteLength)
|
|
380
444
|
const hasDpositions = dpositions.length > 0
|
|
381
445
|
const hasCpositions = cpositions.length > 0
|
|
382
446
|
|
package/src/csi.ts
CHANGED
|
@@ -36,7 +36,11 @@ export default class CSI extends IndexFile {
|
|
|
36
36
|
}
|
|
37
37
|
|
|
38
38
|
parseAuxData(bytes: Uint8Array, offset: number) {
|
|
39
|
-
const dataView = new DataView(
|
|
39
|
+
const dataView = new DataView(
|
|
40
|
+
bytes.buffer,
|
|
41
|
+
bytes.byteOffset,
|
|
42
|
+
bytes.byteLength,
|
|
43
|
+
)
|
|
40
44
|
const formatFlags = dataView.getUint32(offset, true)
|
|
41
45
|
const coordinateType =
|
|
42
46
|
formatFlags & 0x10000 ? 'zero-based-half-open' : '1-based-closed'
|
|
@@ -74,7 +78,11 @@ export default class CSI extends IndexFile {
|
|
|
74
78
|
const buffer = await this.filehandle.readFile(opts)
|
|
75
79
|
const bytes = await unzip(buffer)
|
|
76
80
|
|
|
77
|
-
const dataView = new DataView(
|
|
81
|
+
const dataView = new DataView(
|
|
82
|
+
bytes.buffer,
|
|
83
|
+
bytes.byteOffset,
|
|
84
|
+
bytes.byteLength,
|
|
85
|
+
)
|
|
78
86
|
let csiVersion
|
|
79
87
|
const magic = dataView.getUint32(0, true)
|
|
80
88
|
|
package/src/htsget.ts
CHANGED
|
@@ -78,7 +78,7 @@ export default class HtsgetFile<
|
|
|
78
78
|
})
|
|
79
79
|
|
|
80
80
|
const zero = new VirtualOffset(0, 0)
|
|
81
|
-
const allRecords =
|
|
81
|
+
const allRecords = this.readBamFeatures(
|
|
82
82
|
uncba,
|
|
83
83
|
[],
|
|
84
84
|
[],
|
|
@@ -95,7 +95,11 @@ export default class HtsgetFile<
|
|
|
95
95
|
const uncba = await fetchAndConcat(data.htsget.urls, {
|
|
96
96
|
signal: opts.signal,
|
|
97
97
|
})
|
|
98
|
-
const dataView = new DataView(
|
|
98
|
+
const dataView = new DataView(
|
|
99
|
+
uncba.buffer,
|
|
100
|
+
uncba.byteOffset,
|
|
101
|
+
uncba.byteLength,
|
|
102
|
+
)
|
|
99
103
|
|
|
100
104
|
if (dataView.getInt32(0, true) !== BAM_MAGIC) {
|
|
101
105
|
throw new Error('Not a BAM file')
|
package/src/index.ts
CHANGED
|
@@ -1,9 +1,8 @@
|
|
|
1
1
|
export { default as BAI } from './bai.ts'
|
|
2
|
-
export { default as BamFile } from './bamFile.ts'
|
|
2
|
+
export { DEFAULT_MAX_CACHE_BYTES, default as BamFile } from './bamFile.ts'
|
|
3
3
|
export { default as CSI } from './csi.ts'
|
|
4
4
|
export { default as BamRecord } from './record.ts'
|
|
5
5
|
export { default as HtsgetFile } from './htsget.ts'
|
|
6
6
|
|
|
7
7
|
export type { Bytes } from './record.ts'
|
|
8
|
-
export type { FilterBy, TagFilter } from './util.ts'
|
|
9
8
|
export type { BamRecordClass, BamRecordLike } from './bamFile.ts'
|