@gmod/bam 9.0.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +24 -21
- package/dist/bai.d.ts +2 -0
- package/dist/bai.js +19 -0
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.js +11 -19
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.d.ts +20 -29
- package/dist/csi.js +27 -10
- package/dist/csi.js.map +1 -1
- package/dist/indexFile.d.ts +21 -1
- package/dist/indexFile.js +60 -2
- package/dist/indexFile.js.map +1 -1
- package/dist/record.d.ts +1 -0
- package/dist/record.js +39 -22
- package/dist/record.js.map +1 -1
- package/dist/streamBam.js +2 -2
- package/dist/streamBam.js.map +1 -1
- package/dist/util.d.ts +15 -4
- package/dist/util.js +29 -4
- package/dist/util.js.map +1 -1
- package/dist/virtualOffset.d.ts +1 -0
- package/dist/virtualOffset.js +4 -0
- package/dist/virtualOffset.js.map +1 -1
- package/esm/bai.d.ts +2 -0
- package/esm/bai.js +19 -0
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.js +12 -20
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.d.ts +20 -29
- package/esm/csi.js +27 -10
- package/esm/csi.js.map +1 -1
- package/esm/indexFile.d.ts +21 -1
- package/esm/indexFile.js +59 -2
- package/esm/indexFile.js.map +1 -1
- package/esm/record.d.ts +1 -0
- package/esm/record.js +39 -22
- package/esm/record.js.map +1 -1
- package/esm/streamBam.js +3 -3
- package/esm/streamBam.js.map +1 -1
- package/esm/util.d.ts +15 -4
- package/esm/util.js +27 -4
- package/esm/util.js.map +1 -1
- package/esm/virtualOffset.d.ts +1 -0
- package/esm/virtualOffset.js +3 -0
- package/esm/virtualOffset.js.map +1 -1
- package/package.json +7 -7
- package/src/bai.ts +21 -0
- package/src/bamFile.ts +14 -21
- package/src/csi.ts +41 -14
- package/src/indexFile.ts +72 -3
- package/src/record.ts +45 -21
- package/src/streamBam.ts +4 -4
- package/src/util.ts +33 -4
- package/src/virtualOffset.ts +4 -0
package/src/bai.ts
CHANGED
|
@@ -49,6 +49,23 @@ function roundUp(n: number, multiple: number) {
|
|
|
49
49
|
return rem === 0 ? n : n - rem + multiple
|
|
50
50
|
}
|
|
51
51
|
|
|
52
|
+
/**
|
|
53
|
+
* Older samtools leaves 0:0 in linear-index windows no read overlaps. htslib
|
|
54
|
+
* fills interior ones from the next entry on load (`hts.c`, "fill missing
|
|
55
|
+
* values"); this also fills leading ones, which htslib leaves at 0. Either
|
|
56
|
+
* way the entry stays a lower bound, since no record overlaps that window.
|
|
57
|
+
* Left as 0, a leading entry scores its whole absolute file offset in
|
|
58
|
+
* `indexCov`, and `getLowestChunk` falls back to the start of the file.
|
|
59
|
+
*/
|
|
60
|
+
function fillLinearGaps(blocks: Float64Array, data: Float64Array) {
|
|
61
|
+
for (let j = blocks.length - 2; j >= 0; j--) {
|
|
62
|
+
if (blocks[j] === 0 && data[j] === 0) {
|
|
63
|
+
blocks[j] = blocks[j + 1]!
|
|
64
|
+
data[j] = data[j + 1]!
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
52
69
|
export interface IndexCovEntry {
|
|
53
70
|
start: number
|
|
54
71
|
end: number
|
|
@@ -83,6 +100,9 @@ function reg2bins(beg: number, end: number) {
|
|
|
83
100
|
}
|
|
84
101
|
|
|
85
102
|
export default class BAI extends IndexFile<BaiParsed> {
|
|
103
|
+
protected minShift = BAI_LINEAR_SHIFT
|
|
104
|
+
protected depth = BAI_DEPTH
|
|
105
|
+
|
|
86
106
|
async _parse(opts: BaseOpts): Promise<BaiParsed> {
|
|
87
107
|
const bytes = await this.filehandle.readFile(opts)
|
|
88
108
|
const dataView = new DataView(
|
|
@@ -192,6 +212,7 @@ export default class BAI extends IndexFile<BaiParsed> {
|
|
|
192
212
|
linearDataPositions[j] = (bytes[curr + 1]! << 8) | bytes[curr]!
|
|
193
213
|
curr += 8
|
|
194
214
|
}
|
|
215
|
+
fillLinearGaps(linearBlockPositions, linearDataPositions)
|
|
195
216
|
|
|
196
217
|
clampChunkEnds(Object.values(binIndex).flat(), linearBlockPositions)
|
|
197
218
|
return {
|
package/src/bamFile.ts
CHANGED
|
@@ -12,7 +12,10 @@ import {
|
|
|
12
12
|
BAM_MAGIC,
|
|
13
13
|
MAX_CONCURRENT_CHUNK_READS,
|
|
14
14
|
appendInRange,
|
|
15
|
+
decodeHeaderText,
|
|
16
|
+
optimizeChunks,
|
|
15
17
|
parseRefSeqs,
|
|
18
|
+
readBlockSize,
|
|
16
19
|
resolveFilehandle,
|
|
17
20
|
throwIfAborted,
|
|
18
21
|
} from './util.ts'
|
|
@@ -464,9 +467,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
464
467
|
const parsed = parseRefSeqs(uncba, headLen + 8, this.renameRefSeq)
|
|
465
468
|
let samHeader
|
|
466
469
|
if (parsed) {
|
|
467
|
-
const headerText =
|
|
468
|
-
uncba.subarray(8, 8 + headLen),
|
|
469
|
-
)
|
|
470
|
+
const headerText = decodeHeaderText(uncba, headLen)
|
|
470
471
|
this.header = headerText
|
|
471
472
|
this.chrToIndex = parsed.chrToIndex
|
|
472
473
|
this.indexToChr = parsed.indexToChr
|
|
@@ -539,7 +540,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
539
540
|
if (chrId === undefined || !this.index) {
|
|
540
541
|
return []
|
|
541
542
|
}
|
|
542
|
-
const chunks = await this.index.blocksForRange(chrId, min
|
|
543
|
+
const chunks = await this.index.blocksForRange(chrId, min, max, opts)
|
|
543
544
|
return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts)
|
|
544
545
|
}
|
|
545
546
|
|
|
@@ -818,10 +819,12 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
818
819
|
// make the count NaN.
|
|
819
820
|
const readNameCounts = new Map<string, number>()
|
|
820
821
|
const readIds = new Set<number>()
|
|
822
|
+
const names = new Array<string>(records.length)
|
|
821
823
|
|
|
822
824
|
for (let i = 0, l = records.length; i < l; i++) {
|
|
823
825
|
const r = records[i]!
|
|
824
826
|
const name = r.name
|
|
827
|
+
names[i] = name
|
|
825
828
|
readNameCounts.set(name, (readNameCounts.get(name) ?? 0) + 1)
|
|
826
829
|
readIds.add(r.fileOffset)
|
|
827
830
|
}
|
|
@@ -829,10 +832,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
829
832
|
const matePromises: Promise<Chunk[]>[] = []
|
|
830
833
|
for (let i = 0, l = records.length; i < l; i++) {
|
|
831
834
|
const f = records[i]!
|
|
832
|
-
const name = f.name
|
|
833
835
|
if (
|
|
834
836
|
this.index &&
|
|
835
|
-
readNameCounts.get(
|
|
837
|
+
readNameCounts.get(names[i]!) === 1 &&
|
|
836
838
|
(pairAcrossChr ||
|
|
837
839
|
(f.next_refid === chrId &&
|
|
838
840
|
Math.abs(f.start - f.next_pos) < maxInsertSize))
|
|
@@ -848,20 +850,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
848
850
|
}
|
|
849
851
|
}
|
|
850
852
|
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
const m = chunks[j]!
|
|
857
|
-
// Key on the virtual-offset span — the same key _cachedChunkFeatures
|
|
858
|
-
// uses. Chunk.toString() also folds in `bin` and fetchedSize(), which
|
|
859
|
-
// keeps two chunks covering an identical span apart here even though
|
|
860
|
-
// the cache below collapses them, so their records came back twice.
|
|
861
|
-
map.set(chunkCacheKey(m), m)
|
|
862
|
-
}
|
|
863
|
-
}
|
|
864
|
-
const mateChunks = [...map.values()]
|
|
853
|
+
// Each mate's lookup is merged on its own, so two of them can resolve to
|
|
854
|
+
// different spans over the same records. Merging the union again makes
|
|
855
|
+
// them disjoint, which is what keeps a mate in the overlap from coming
|
|
856
|
+
// back once per span.
|
|
857
|
+
const mateChunks = optimizeChunks((await Promise.all(matePromises)).flat())
|
|
865
858
|
|
|
866
859
|
// Bounded for the reason ADR 0008 bounds the main query path: a viewAsPairs
|
|
867
860
|
// query over a busy region resolves to many distinct mate chunks, and an
|
|
@@ -961,7 +954,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
961
954
|
const hasCpositions = cpositions.length > 0
|
|
962
955
|
|
|
963
956
|
while (blockStart + 4 < ba.length) {
|
|
964
|
-
const blockSize = dataView
|
|
957
|
+
const blockSize = readBlockSize(dataView, blockStart)
|
|
965
958
|
const blockEnd = blockStart + 4 + blockSize - 1
|
|
966
959
|
|
|
967
960
|
if (hasDpositions) {
|
package/src/csi.ts
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
} from './util.ts'
|
|
11
11
|
import { VirtualOffset, fromBytes } from './virtualOffset.ts'
|
|
12
12
|
|
|
13
|
+
import type { ParsedIndexBase, RefIndex } from './indexFile.ts'
|
|
13
14
|
import type { BaseOpts } from './util.ts'
|
|
14
15
|
|
|
15
16
|
const CSI1_MAGIC = 21582659 // CSI\1
|
|
@@ -24,10 +25,18 @@ function rshift(num: number, bits: number) {
|
|
|
24
25
|
return Math.floor(num / 2 ** bits)
|
|
25
26
|
}
|
|
26
27
|
|
|
27
|
-
|
|
28
|
+
interface CsiRefIndex extends RefIndex {
|
|
29
|
+
loffsets: Map<number, VirtualOffset>
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
interface CsiParsed extends ParsedIndexBase<CsiRefIndex> {
|
|
33
|
+
csi: true
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export default class CSI extends IndexFile<CsiParsed> {
|
|
28
37
|
private maxBinNumber = 0
|
|
29
|
-
|
|
30
|
-
|
|
38
|
+
protected depth = 0
|
|
39
|
+
protected minShift = 0
|
|
31
40
|
|
|
32
41
|
// CSI omits the linear index that BAI's indexCov derives coverage from
|
|
33
42
|
// (CSIv1.tex §3, hts-specs), so there's no equivalent to return.
|
|
@@ -74,7 +83,7 @@ export default class CSI extends IndexFile {
|
|
|
74
83
|
}
|
|
75
84
|
|
|
76
85
|
// fetch and parse the index
|
|
77
|
-
async _parse(opts: BaseOpts) {
|
|
86
|
+
async _parse(opts: BaseOpts): Promise<CsiParsed> {
|
|
78
87
|
const buffer = await this.filehandle.readFile(opts)
|
|
79
88
|
const bytes = await unzip(buffer)
|
|
80
89
|
|
|
@@ -93,7 +102,7 @@ export default class CSI extends IndexFile {
|
|
|
93
102
|
|
|
94
103
|
this.minShift = dataView.getInt32(4, true)
|
|
95
104
|
this.depth = dataView.getInt32(8, true)
|
|
96
|
-
this.maxBinNumber = (
|
|
105
|
+
this.maxBinNumber = (8 ** (this.depth + 1) - 1) / 7
|
|
97
106
|
const maxBinNumber = this.maxBinNumber
|
|
98
107
|
const auxLength = dataView.getInt32(12, true)
|
|
99
108
|
// A tabix-only branch, which is why parseAuxData and the parseNameBytes it
|
|
@@ -128,10 +137,10 @@ export default class CSI extends IndexFile {
|
|
|
128
137
|
if (bin > this.maxBinNumber) {
|
|
129
138
|
curr += 28 + 16
|
|
130
139
|
} else {
|
|
131
|
-
// A bin's loffset is the
|
|
132
|
-
//
|
|
133
|
-
//
|
|
134
|
-
//
|
|
140
|
+
// A bin's loffset is the linear-index entry at its first window, so
|
|
141
|
+
// the smallest over every bin is the first record's offset — one
|
|
142
|
+
// read per bin instead of one per chunk. Checked against every .csi
|
|
143
|
+
// in test/data: same answer on all 19.
|
|
135
144
|
firstDataLine = minVirtualOffset(bytes, curr, 1, firstDataLine)
|
|
136
145
|
curr += 8 // loffset
|
|
137
146
|
const chunkCount = dataView.getInt32(curr, true)
|
|
@@ -149,6 +158,7 @@ export default class CSI extends IndexFile {
|
|
|
149
158
|
const binCount = dataView.getInt32(curr, true)
|
|
150
159
|
curr += 4
|
|
151
160
|
const binIndex: Record<number, Chunk[]> = {}
|
|
161
|
+
const loffsets = new Map<number, VirtualOffset>()
|
|
152
162
|
let pseudoBinStats
|
|
153
163
|
for (let j = 0; j < binCount; j++) {
|
|
154
164
|
const bin = dataView.getUint32(curr, true)
|
|
@@ -157,7 +167,8 @@ export default class CSI extends IndexFile {
|
|
|
157
167
|
pseudoBinStats = parsePseudoBin(bytes, curr + 28)
|
|
158
168
|
curr += 28 + 16
|
|
159
169
|
} else {
|
|
160
|
-
|
|
170
|
+
loffsets.set(bin, fromBytes(bytes, curr))
|
|
171
|
+
curr += 8
|
|
161
172
|
const chunkCount = dataView.getInt32(curr, true)
|
|
162
173
|
curr += 4
|
|
163
174
|
const chunks = new Array<Chunk>(chunkCount)
|
|
@@ -175,6 +186,7 @@ export default class CSI extends IndexFile {
|
|
|
175
186
|
clampChunkEnds(Object.values(binIndex).flat())
|
|
176
187
|
return {
|
|
177
188
|
binIndex,
|
|
189
|
+
loffsets,
|
|
178
190
|
stats: pseudoBinStats,
|
|
179
191
|
}
|
|
180
192
|
}
|
|
@@ -188,12 +200,27 @@ export default class CSI extends IndexFile {
|
|
|
188
200
|
}
|
|
189
201
|
}
|
|
190
202
|
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
203
|
+
/**
|
|
204
|
+
* CSI has no linear index, but each bin's `loffset` is the linear-index entry
|
|
205
|
+
* at the bin's first window (htslib's `update_loff`), so the finest bin at or
|
|
206
|
+
* left of `min` bounds the query from below the way BAI's linear index does.
|
|
207
|
+
* Walks left through siblings and then up to the parent, as `hts_itr_query`
|
|
208
|
+
* does for CSI.
|
|
209
|
+
*/
|
|
210
|
+
protected getLowestChunk(refIndex: CsiRefIndex, min: number) {
|
|
211
|
+
const { loffsets } = refIndex
|
|
212
|
+
const leaves = 8 ** this.depth
|
|
213
|
+
let bin =
|
|
214
|
+
(leaves - 1) / 7 +
|
|
215
|
+
Math.min(Math.floor(min / 2 ** this.minShift), leaves - 1)
|
|
216
|
+
while (bin > 0 && !loffsets.has(bin)) {
|
|
217
|
+
const parent = Math.floor((bin - 1) / 8)
|
|
218
|
+
bin = bin > parent * 8 + 1 ? bin - 1 : parent
|
|
219
|
+
}
|
|
220
|
+
return loffsets.get(bin) ?? ZERO_OFFSET
|
|
194
221
|
}
|
|
195
222
|
|
|
196
|
-
//
|
|
223
|
+
// No linear index means nothing to forecast the far end of a query with, so
|
|
197
224
|
// estimatedBytesForRegions keeps summing every chunk on a CSI-indexed file.
|
|
198
225
|
protected getHighestChunk() {
|
|
199
226
|
return undefined
|
package/src/indexFile.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { SharedReadCache } from '@gmod/shared-read-cache'
|
|
|
2
2
|
import QuickLRU from '@jbrowse/quick-lru'
|
|
3
3
|
|
|
4
4
|
import { chunksLikelyRead, optimizeChunks } from './util.ts'
|
|
5
|
+
import { compareOffsets } from './virtualOffset.ts'
|
|
5
6
|
|
|
6
7
|
import type Chunk from './chunk.ts'
|
|
7
8
|
import type { BaseOpts } from './util.ts'
|
|
@@ -48,6 +49,58 @@ export function memoizeByRefId<T>(
|
|
|
48
49
|
}
|
|
49
50
|
}
|
|
50
51
|
|
|
52
|
+
/**
|
|
53
|
+
* The virtual offset past which a coordinate-sorted file holds nothing
|
|
54
|
+
* overlapping `[.., end)`, from the binning index alone: htslib's `max_off`
|
|
55
|
+
* (`hts_itr_query` in hts.c).
|
|
56
|
+
*
|
|
57
|
+
* Walk right from the finest bin after the one holding `end - 1`, stepping up
|
|
58
|
+
* to the parent at every first child, so each bin visited begins at or past
|
|
59
|
+
* `end` and never overlaps the query. Every record in such a bin starts at or
|
|
60
|
+
* past `end`, so the first chunk of the first bin that exists is a record past
|
|
61
|
+
* the query, and in a sorted file so is every record after it.
|
|
62
|
+
*
|
|
63
|
+
* A bound, unlike the linear-index forecast in `chunksLikelyRead`: it rests on
|
|
64
|
+
* the same sort order `appendInRange` and the early stop already assume, and
|
|
65
|
+
* cannot drop a record they would keep. See ADR 0023 for why the caller drops
|
|
66
|
+
* whole merged chunks with it rather than trimming them.
|
|
67
|
+
*/
|
|
68
|
+
export function maxOffset(
|
|
69
|
+
binIndex: Record<number, Chunk[]>,
|
|
70
|
+
end: number,
|
|
71
|
+
minShift: number,
|
|
72
|
+
depth: number,
|
|
73
|
+
) {
|
|
74
|
+
if (end > 2 ** (minShift + depth * 3)) {
|
|
75
|
+
return undefined
|
|
76
|
+
}
|
|
77
|
+
const binCount = (8 ** (depth + 1) - 1) / 7
|
|
78
|
+
let bin = (8 ** depth - 1) / 7 + Math.floor((end - 1) / 2 ** minShift) + 1
|
|
79
|
+
if (bin >= binCount) {
|
|
80
|
+
bin = 0
|
|
81
|
+
}
|
|
82
|
+
for (;;) {
|
|
83
|
+
while (bin % 8 === 1) {
|
|
84
|
+
bin = (bin - 1) / 8
|
|
85
|
+
}
|
|
86
|
+
if (bin === 0) {
|
|
87
|
+
return undefined
|
|
88
|
+
}
|
|
89
|
+
const chunks = binIndex[bin]
|
|
90
|
+
if (chunks?.length) {
|
|
91
|
+
let lowest = chunks[0]!.minv
|
|
92
|
+
for (let i = 1; i < chunks.length; i++) {
|
|
93
|
+
const minv = chunks[i]!.minv
|
|
94
|
+
if (compareOffsets(minv, lowest) < 0) {
|
|
95
|
+
lowest = minv
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
return lowest
|
|
99
|
+
}
|
|
100
|
+
bin++
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
51
104
|
export default abstract class IndexFile<
|
|
52
105
|
TParsed extends ParsedIndexBase = ParsedIndexBase,
|
|
53
106
|
> {
|
|
@@ -79,6 +132,11 @@ export default abstract class IndexFile<
|
|
|
79
132
|
end?: number,
|
|
80
133
|
): Promise<{ start: number; end: number; score: number }[]>
|
|
81
134
|
|
|
135
|
+
// The binning scheme: the finest bins are 2^minShift wide, and there are
|
|
136
|
+
// depth levels below bin 0. BAI is CSI with minShift 14 and depth 5.
|
|
137
|
+
protected abstract minShift: number
|
|
138
|
+
protected abstract depth: number
|
|
139
|
+
|
|
82
140
|
// Bin numbers that overlap [min, max). Subclasses implement BAI's fixed
|
|
83
141
|
// 5-level scheme or CSI's configurable scheme (SAMv1.pdf §5.1.1, CSIv1.tex §2).
|
|
84
142
|
protected abstract reg2bins(
|
|
@@ -87,7 +145,8 @@ export default abstract class IndexFile<
|
|
|
87
145
|
): readonly (readonly [number, number])[]
|
|
88
146
|
|
|
89
147
|
// Lower-bound virtual offset for chunks that could contain alignments in
|
|
90
|
-
// [min, ...). BAI uses its linear index
|
|
148
|
+
// [min, ...). BAI uses its linear index, CSI the loffset of a bin at or left
|
|
149
|
+
// of min.
|
|
91
150
|
protected abstract getLowestChunk(
|
|
92
151
|
refIndex: RefIndex,
|
|
93
152
|
min: number,
|
|
@@ -133,7 +192,16 @@ export default abstract class IndexFile<
|
|
|
133
192
|
}
|
|
134
193
|
}
|
|
135
194
|
}
|
|
136
|
-
|
|
195
|
+
const merged = optimizeChunks(chunks, this.getLowestChunk(ba, min))
|
|
196
|
+
const past = maxOffset(binIndex, max, this.minShift, this.depth)
|
|
197
|
+
if (past) {
|
|
198
|
+
let n = merged.length
|
|
199
|
+
while (n > 0 && compareOffsets(merged[n - 1]!.minv, past) >= 0) {
|
|
200
|
+
n--
|
|
201
|
+
}
|
|
202
|
+
merged.length = n
|
|
203
|
+
}
|
|
204
|
+
return merged
|
|
137
205
|
}
|
|
138
206
|
|
|
139
207
|
// SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same shape and the same
|
|
@@ -197,7 +265,8 @@ export default abstract class IndexFile<
|
|
|
197
265
|
* exists — was being told 5.6x the truth on exactly the windows a reader
|
|
198
266
|
* spends their time in, and cannot answer it by zooming: every window narrower
|
|
199
267
|
* than a linear-index interval resolves to the same chunks and so to the same
|
|
200
|
-
* number.
|
|
268
|
+
* number. The table predates `max_off` (ADR 0023), which now drops most of
|
|
269
|
+
* the gap between the first two columns from `blocksForRange` itself.
|
|
201
270
|
*
|
|
202
271
|
* Still summed over merged chunks rather than per region, so two regions
|
|
203
272
|
* sharing a chunk are charged for it once.
|
package/src/record.ts
CHANGED
|
@@ -418,9 +418,14 @@ export default class BamRecord {
|
|
|
418
418
|
return this._dataView.getInt32(this._start + 8, true)
|
|
419
419
|
}
|
|
420
420
|
|
|
421
|
+
// The end htslib's bam_endpos() reports: a record consuming no reference —
|
|
422
|
+
// an unmapped mate placed at its mate's coordinate, or an empty CIGAR —
|
|
423
|
+
// still covers one base rather than none, so a consumer's interval search on
|
|
424
|
+
// the base it sits at can find it (endpos() in util.ts applies the same rule
|
|
425
|
+
// to this library's own range filter).
|
|
421
426
|
get end() {
|
|
422
427
|
if (this._cachedEnd === undefined) {
|
|
423
|
-
this._cachedEnd = this.start + this.length_on_ref
|
|
428
|
+
this._cachedEnd = this.start + Math.max(this.length_on_ref, 1)
|
|
424
429
|
}
|
|
425
430
|
return this._cachedEnd
|
|
426
431
|
}
|
|
@@ -435,16 +440,18 @@ export default class BamRecord {
|
|
|
435
440
|
}
|
|
436
441
|
|
|
437
442
|
// QUAL is present whenever the record has bases — independent of the unmapped
|
|
438
|
-
// flag (unmapped reads routinely carry SEQ/QUAL).
|
|
439
|
-
//
|
|
443
|
+
// flag (unmapped reads routinely carry SEQ/QUAL). null for a zero-length SEQ,
|
|
444
|
+
// and for a QUAL of `*`, which BAM stores as 0xFF in every byte (SAMv1
|
|
445
|
+
// §4.2.3). htslib decides on the first byte alone, and so does this.
|
|
440
446
|
get qual() {
|
|
441
447
|
const seqLen = this.seq_length
|
|
442
448
|
if (seqLen === 0) {
|
|
443
449
|
return null
|
|
444
|
-
} else {
|
|
445
|
-
const p = this.seqStart + ((seqLen + 1) >> 1)
|
|
446
|
-
return this._byteArray.subarray(p, p + seqLen)
|
|
447
450
|
}
|
|
451
|
+
const p = this.seqStart + ((seqLen + 1) >> 1)
|
|
452
|
+
return this._byteArray[p] === 0xff
|
|
453
|
+
? null
|
|
454
|
+
: this._byteArray.subarray(p, p + seqLen)
|
|
448
455
|
}
|
|
449
456
|
|
|
450
457
|
get strand() {
|
|
@@ -470,7 +477,7 @@ export default class BamRecord {
|
|
|
470
477
|
return this.seqStart + ((seqLen + 1) >> 1) + seqLen
|
|
471
478
|
}
|
|
472
479
|
|
|
473
|
-
// batch fromCharCode: fastest for typical name lengths (
|
|
480
|
+
// batch fromCharCode: fastest for typical name lengths (benchmarks/string-building.bench.ts, removed in 36a3968)
|
|
474
481
|
//
|
|
475
482
|
// Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
|
|
476
483
|
// read name about once — jbrowse-components' buildBaseFeatureData copies it
|
|
@@ -505,8 +512,8 @@ export default class BamRecord {
|
|
|
505
512
|
}
|
|
506
513
|
|
|
507
514
|
getTag(tagName: string) {
|
|
508
|
-
if (this._cachedTags !== undefined) {
|
|
509
|
-
return this.
|
|
515
|
+
if (this._cachedTags !== undefined || tagName === 'CG') {
|
|
516
|
+
return this.tags[tagName]
|
|
510
517
|
}
|
|
511
518
|
return this._findTag(tagName, false)
|
|
512
519
|
}
|
|
@@ -616,6 +623,9 @@ export default class BamRecord {
|
|
|
616
623
|
)
|
|
617
624
|
p = end
|
|
618
625
|
}
|
|
626
|
+
if (this._hasCGPlaceholder() && isNumericCigar(tags.CG)) {
|
|
627
|
+
delete tags.CG
|
|
628
|
+
}
|
|
619
629
|
return tags
|
|
620
630
|
}
|
|
621
631
|
|
|
@@ -667,7 +677,7 @@ export default class BamRecord {
|
|
|
667
677
|
return !!(this.flags & Constants.BAM_FSUPPLEMENTARY)
|
|
668
678
|
}
|
|
669
679
|
|
|
670
|
-
// Benchmark results for CIGAR parsing strategies (
|
|
680
|
+
// Benchmark results for CIGAR parsing strategies (benchmarks/cigar-strategies.bench.ts, removed in 36a3968):
|
|
671
681
|
//
|
|
672
682
|
// Aligned data:
|
|
673
683
|
// - Plain array is 1.6-1.8x faster than Uint32Array for small CIGARs (≤50 ops)
|
|
@@ -691,12 +701,23 @@ export default class BamRecord {
|
|
|
691
701
|
// htslib stores the placeholder as exactly two ops: <seqlen>S<reflen>N.
|
|
692
702
|
if (numCigarOps === 2) {
|
|
693
703
|
const cigop = this._dataView.getInt32(p, true)
|
|
694
|
-
return (
|
|
704
|
+
return (
|
|
705
|
+
(cigop & 0xf) === CIGAR_SOFT_CLIP && cigop >>> 4 === this.seq_length
|
|
706
|
+
)
|
|
695
707
|
} else {
|
|
696
708
|
return false
|
|
697
709
|
}
|
|
698
710
|
}
|
|
699
711
|
|
|
712
|
+
// SAMv1 §4.2.2: with the placeholder and a CG tag, the tag is the CIGAR and
|
|
713
|
+
// a reader removes it from the tags, as htslib does
|
|
714
|
+
private _hasCGPlaceholder() {
|
|
715
|
+
return this._isCGTagPattern(
|
|
716
|
+
this.b0 + this.read_name_length,
|
|
717
|
+
this.flag_nc & 0xffff,
|
|
718
|
+
)
|
|
719
|
+
}
|
|
720
|
+
|
|
700
721
|
private _computeLengthOnRef(): number {
|
|
701
722
|
const flag_nc = this._dataView.getInt32(this._start + 16, true)
|
|
702
723
|
if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
|
|
@@ -711,7 +732,7 @@ export default class BamRecord {
|
|
|
711
732
|
if ((cigop2 & 0xf) !== CIGAR_REF_SKIP) {
|
|
712
733
|
console.warn('CG tag with no N tag')
|
|
713
734
|
}
|
|
714
|
-
return cigop2
|
|
735
|
+
return cigop2 >>> 4
|
|
715
736
|
}
|
|
716
737
|
|
|
717
738
|
const absOffset = this._byteArray.byteOffset + p
|
|
@@ -727,7 +748,7 @@ export default class BamRecord {
|
|
|
727
748
|
let lref = 0
|
|
728
749
|
for (let c = 0; c < numCigarOps; ++c) {
|
|
729
750
|
const co = cigarView[c]!
|
|
730
|
-
lref += (co
|
|
751
|
+
lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
|
|
731
752
|
}
|
|
732
753
|
return lref
|
|
733
754
|
}
|
|
@@ -735,7 +756,7 @@ export default class BamRecord {
|
|
|
735
756
|
let lref = 0
|
|
736
757
|
for (let c = 0; c < numCigarOps; ++c) {
|
|
737
758
|
const co = this._dataView.getInt32(p + c * 4, true)
|
|
738
|
-
lref += (co
|
|
759
|
+
lref += (co >>> 4) * ((CIGAR_CONSUMES_REF_MASK >> (co & 0xf)) & 1)
|
|
739
760
|
}
|
|
740
761
|
return lref
|
|
741
762
|
}
|
|
@@ -756,10 +777,10 @@ export default class BamRecord {
|
|
|
756
777
|
const p = this.b0 + this.read_name_length
|
|
757
778
|
|
|
758
779
|
if (this._isCGTagPattern(p, numCigarOps)) {
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
780
|
+
const cg = this._findTag('CG', false)
|
|
781
|
+
if (isNumericCigar(cg)) {
|
|
782
|
+
return cg
|
|
783
|
+
}
|
|
763
784
|
}
|
|
764
785
|
|
|
765
786
|
const absOffset = this._byteArray.byteOffset + p
|
|
@@ -806,18 +827,21 @@ export default class BamRecord {
|
|
|
806
827
|
let result = ''
|
|
807
828
|
for (let i = 0, l = numeric.length; i < l; i++) {
|
|
808
829
|
const packed = numeric[i]!
|
|
809
|
-
result += packed
|
|
830
|
+
result += packed >>> 4
|
|
810
831
|
result += String.fromCharCode(ASCII_CIGAR_CODES[packed & 0xf]!)
|
|
811
832
|
}
|
|
812
833
|
return result
|
|
813
834
|
}
|
|
814
835
|
|
|
815
836
|
get num_cigar_ops() {
|
|
816
|
-
return this.
|
|
837
|
+
return this._hasCGPlaceholder()
|
|
838
|
+
? this.NUMERIC_CIGAR.length
|
|
839
|
+
: this.flag_nc & 0xffff
|
|
817
840
|
}
|
|
818
841
|
|
|
842
|
+
// the stored CIGAR field, which for a long CIGAR is the two-op placeholder
|
|
819
843
|
get num_cigar_bytes() {
|
|
820
|
-
return this.
|
|
844
|
+
return (this.flag_nc & 0xffff) << 2
|
|
821
845
|
}
|
|
822
846
|
|
|
823
847
|
get read_name_length() {
|
package/src/streamBam.ts
CHANGED
|
@@ -9,7 +9,9 @@ import { parseHeaderText } from './sam.ts'
|
|
|
9
9
|
import {
|
|
10
10
|
BAM_MAGIC,
|
|
11
11
|
concatUint8Array,
|
|
12
|
+
decodeHeaderText,
|
|
12
13
|
parseRefSeqs,
|
|
14
|
+
readBlockSize,
|
|
13
15
|
resolveFilehandle,
|
|
14
16
|
throwIfAborted,
|
|
15
17
|
} from './util.ts'
|
|
@@ -273,9 +275,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
|
|
|
273
275
|
recordCarry = bytes
|
|
274
276
|
continue
|
|
275
277
|
}
|
|
276
|
-
const headerText =
|
|
277
|
-
bytes.subarray(8, 8 + lText),
|
|
278
|
-
)
|
|
278
|
+
const headerText = decodeHeaderText(bytes, lText)
|
|
279
279
|
onHeader?.({
|
|
280
280
|
headerText,
|
|
281
281
|
samHeader: parseHeaderText(headerText),
|
|
@@ -288,7 +288,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
|
|
|
288
288
|
|
|
289
289
|
const sink: T[] = []
|
|
290
290
|
while (blockStart + 4 <= bytes.length) {
|
|
291
|
-
const blockSize = dataView
|
|
291
|
+
const blockSize = readBlockSize(dataView, blockStart)
|
|
292
292
|
const blockEnd = blockStart + 4 + blockSize - 1
|
|
293
293
|
if (blockEnd >= bytes.length) {
|
|
294
294
|
break
|
package/src/util.ts
CHANGED
|
@@ -9,6 +9,22 @@ import type { GenericFilehandle } from 'generic-filehandle2'
|
|
|
9
9
|
/** 'BAM\1' read as a little-endian int32 */
|
|
10
10
|
export const BAM_MAGIC = 21840194
|
|
11
11
|
|
|
12
|
+
/**
|
|
13
|
+
* The `block_size` of the record at `offset`, rejecting one too small to hold a
|
|
14
|
+
* record's 32 bytes of fixed fields, as htslib's `bam_read1` does. A negative
|
|
15
|
+
* one would otherwise leave the loops that advance by it standing still,
|
|
16
|
+
* pushing a record per turn until the process runs out of memory.
|
|
17
|
+
*/
|
|
18
|
+
export function readBlockSize(dataView: DataView, offset: number) {
|
|
19
|
+
const blockSize = dataView.getInt32(offset, true)
|
|
20
|
+
if (blockSize < 32) {
|
|
21
|
+
throw new Error(
|
|
22
|
+
`corrupt BAM record: block_size ${blockSize} at byte ${offset}`,
|
|
23
|
+
)
|
|
24
|
+
}
|
|
25
|
+
return blockSize
|
|
26
|
+
}
|
|
27
|
+
|
|
12
28
|
export function resolveFilehandle(
|
|
13
29
|
filehandle?: GenericFilehandle,
|
|
14
30
|
path?: string,
|
|
@@ -66,10 +82,13 @@ export const MAX_CONCURRENT_CHUNK_READS = 6
|
|
|
66
82
|
*
|
|
67
83
|
* `blocksForRange` returns every chunk of every bin overlapping the query, at
|
|
68
84
|
* every level of the binning scheme, minus the ones the linear index puts
|
|
69
|
-
* entirely before it.
|
|
70
|
-
*
|
|
71
|
-
* window on a deep ONT BAM
|
|
72
|
-
* `getRecordsForRange`
|
|
85
|
+
* entirely before it. Until `max_off` (ADR 0023) also dropped the ones past it,
|
|
86
|
+
* that was wildly more than a long-read query reads: a coarse bin's chunks run
|
|
87
|
+
* to the end of the bin's span, so a 380bp window on a deep ONT BAM resolved to
|
|
88
|
+
* 90 chunks / 43.5MB, of which `getRecordsForRange` read 6 / 7.8MB before the
|
|
89
|
+
* early stop fired. Since then this changes the forecast in 4 of 320 fixture
|
|
90
|
+
* windows, undershooting in all 4; ADR 0023 says what to check before
|
|
91
|
+
* removing it.
|
|
73
92
|
*
|
|
74
93
|
* Two bounds, and the answer is the larger:
|
|
75
94
|
*
|
|
@@ -325,6 +344,16 @@ export function clampChunkEnds(
|
|
|
325
344
|
}
|
|
326
345
|
}
|
|
327
346
|
|
|
347
|
+
// SAMv1 §4.2 counts NUL padding in l_text, and htslib reads the text as a C
|
|
348
|
+
// string, so the header ends at the first NUL.
|
|
349
|
+
export function decodeHeaderText(bytes: Uint8Array, lText: number) {
|
|
350
|
+
const text = bytes.subarray(8, 8 + lText)
|
|
351
|
+
const nul = text.indexOf(0)
|
|
352
|
+
return new TextDecoder('utf8').decode(
|
|
353
|
+
nul === -1 ? text : text.subarray(0, nul),
|
|
354
|
+
)
|
|
355
|
+
}
|
|
356
|
+
|
|
328
357
|
// Parse the BAM reference-sequence table (SAMv1.pdf §4.2). Returns undefined
|
|
329
358
|
// if `uncba` doesn't yet contain the full table — caller fetches more bytes
|
|
330
359
|
// and retries.
|
package/src/virtualOffset.ts
CHANGED
|
@@ -35,3 +35,7 @@ export function fromBytes(bytes: Uint8Array, offset = 0) {
|
|
|
35
35
|
(bytes[offset + 1]! << 8) | bytes[offset]!,
|
|
36
36
|
)
|
|
37
37
|
}
|
|
38
|
+
|
|
39
|
+
export function compareOffsets(a: OffsetCoords, b: OffsetCoords) {
|
|
40
|
+
return a.blockPosition - b.blockPosition || a.dataPosition - b.dataPosition
|
|
41
|
+
}
|