@gmod/bam 7.3.4 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bai.js +10 -7
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +25 -6
- package/dist/bamFile.js +91 -33
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.js +2 -2
- package/dist/csi.js.map +1 -1
- package/dist/htsget.js +2 -2
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/record.d.ts +2 -1
- package/dist/record.js +105 -35
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +1 -1
- package/dist/util.js +11 -12
- package/dist/util.js.map +1 -1
- package/esm/bai.js +10 -7
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +25 -6
- package/esm/bamFile.js +91 -33
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.js +2 -2
- package/esm/csi.js.map +1 -1
- package/esm/htsget.js +2 -2
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +1 -1
- package/esm/index.js +1 -1
- package/esm/index.js.map +1 -1
- package/esm/record.d.ts +2 -1
- package/esm/record.js +105 -35
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +1 -1
- package/esm/util.js +11 -11
- package/esm/util.js.map +1 -1
- package/package.json +2 -1
- package/src/bai.ts +15 -8
- package/src/bamFile.ts +117 -40
- package/src/csi.ts +10 -2
- package/src/htsget.ts +6 -2
- package/src/index.ts +1 -1
- package/src/record.ts +115 -42
- package/src/util.ts +24 -14
package/src/record.ts
CHANGED
|
@@ -1,16 +1,45 @@
|
|
|
1
1
|
import { CIGAR_REF_SKIP, CIGAR_SOFT_CLIP } from './cigar.ts'
|
|
2
2
|
import Constants from './constants.ts'
|
|
3
3
|
|
|
4
|
-
const
|
|
4
|
+
const SEQRET = '=ACMGRSVTWYHKDBN'
|
|
5
|
+
const SEQRET_DECODER = SEQRET.split('')
|
|
6
|
+
const SEQRET_CODES = Uint8Array.from(SEQRET, c => c.charCodeAt(0))
|
|
7
|
+
|
|
8
|
+
// Both bases of a SEQ byte, precomputed for all 256 bytes so decoding advances a
|
|
9
|
+
// byte at a time. Two forms because `seq` has two strategies (see below): packed
|
|
10
|
+
// ASCII codes for a single Uint16Array store, and the 2-char string to append.
|
|
11
|
+
// Byte order within the u16 depends on host endianness.
|
|
12
|
+
const LITTLE_ENDIAN = new Uint16Array(Uint8Array.of(1, 0).buffer)[0] === 1
|
|
13
|
+
const SEQRET_PAIR_CODES = new Uint16Array(256)
|
|
14
|
+
const SEQRET_PAIR_STRINGS = new Array<string>(256)
|
|
15
|
+
for (let hi = 0; hi < 16; hi++) {
|
|
16
|
+
for (let lo = 0; lo < 16; lo++) {
|
|
17
|
+
const h = SEQRET_CODES[hi]!
|
|
18
|
+
const l = SEQRET_CODES[lo]!
|
|
19
|
+
SEQRET_PAIR_CODES[(hi << 4) | lo] = LITTLE_ENDIAN
|
|
20
|
+
? h | (l << 8)
|
|
21
|
+
: (h << 8) | l
|
|
22
|
+
SEQRET_PAIR_STRINGS[(hi << 4) | lo] =
|
|
23
|
+
SEQRET_DECODER[hi]! + SEQRET_DECODER[lo]!
|
|
24
|
+
}
|
|
25
|
+
}
|
|
5
26
|
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
27
|
+
// Read length at which building a byte buffer and calling TextDecoder once
|
|
28
|
+
// overtakes plain string concatenation. Below it concat wins by 2-4x (rope
|
|
29
|
+
// building is cheap and the decode has fixed overhead); above it the decoder
|
|
30
|
+
// wins by 5-8x. Measured crossover is ~300bp, i.e. just past typical Illumina
|
|
31
|
+
// read lengths.
|
|
32
|
+
const SEQ_DECODER_THRESHOLD = 300
|
|
33
|
+
|
|
34
|
+
// Precomputed pair orientation strings, indexed by
|
|
35
|
+
// ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
|
|
36
|
+
// bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
|
|
37
|
+
// bit 3 is whether this read is the leftmost of the pair. The read2 flag (0x80)
|
|
38
|
+
// is deliberately not consulted — "not read1" is what decides the numbering, so
|
|
39
|
+
// a record with neither flag set reads the same as read2.
|
|
9
40
|
// prettier-ignore
|
|
10
41
|
const PAIR_ORIENTATION_TABLE = [
|
|
11
|
-
'F F ','F R ','R F ','R R ','F2F1','F2R1','R2F1','R2R1',
|
|
12
42
|
'F1F2','F1R2','R1F2','R1R2','F2F1','F2R1','R2F1','R2R1',
|
|
13
|
-
'F F ','R F ','F R ','R R ','F1F2','R1F2','F1R2','R1R2',
|
|
14
43
|
'F2F1','R2F1','F2R1','R2R1','F1F2','R1F2','F1R2','R1R2',
|
|
15
44
|
]
|
|
16
45
|
const ASCII_CIGAR_CODES = [
|
|
@@ -23,6 +52,19 @@ const textDecoder = new TextDecoder()
|
|
|
23
52
|
// Binary: 0b111001101 = 0x1CD
|
|
24
53
|
const CIGAR_CONSUMES_REF_MASK = 0x1cd
|
|
25
54
|
|
|
55
|
+
// A CIGAR as packed op words. Either a view over (or copy of) the record's own
|
|
56
|
+
// CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
|
|
57
|
+
// too, since a writer may encode CG as B:i rather than the usual B:I.
|
|
58
|
+
export type NumericCigar = Uint32Array | Int32Array | number[]
|
|
59
|
+
|
|
60
|
+
function isNumericCigar(value: unknown): value is NumericCigar {
|
|
61
|
+
return (
|
|
62
|
+
value instanceof Uint32Array ||
|
|
63
|
+
value instanceof Int32Array ||
|
|
64
|
+
Array.isArray(value)
|
|
65
|
+
)
|
|
66
|
+
}
|
|
67
|
+
|
|
26
68
|
export interface Bytes {
|
|
27
69
|
start: number
|
|
28
70
|
end: number
|
|
@@ -42,7 +84,11 @@ type BArrayValue =
|
|
|
42
84
|
// Decode a 'B' (array) tag value starting at `p` (the byte after type+subtype+
|
|
43
85
|
// count). When the data is naturally aligned we return a typed-array view over
|
|
44
86
|
// the underlying buffer (zero-copy); otherwise we copy element-by-element since
|
|
45
|
-
// typed-array views require alignment.
|
|
87
|
+
// typed-array views require alignment. The copy target is a plain array, not a
|
|
88
|
+
// typed one: benchmarking the unaligned path found number[] fills faster below
|
|
89
|
+
// ~10k elements and reads at least as fast at every size, so the only thing a
|
|
90
|
+
// typed copy would buy is half the retained bytes. Shared by getTag and the
|
|
91
|
+
// full-tag parse.
|
|
46
92
|
function decodeBArrayTag(
|
|
47
93
|
ba: Uint8Array,
|
|
48
94
|
dataView: DataView,
|
|
@@ -228,7 +274,7 @@ export default class BamRecord {
|
|
|
228
274
|
private _cachedEnd?: number
|
|
229
275
|
private _cachedTags?: Record<string, unknown>
|
|
230
276
|
private _cachedLengthOnRef?: number
|
|
231
|
-
private _cachedNumericCigar?:
|
|
277
|
+
private _cachedNumericCigar?: NumericCigar
|
|
232
278
|
private _cachedNUMERIC_MD?: Uint8Array | null
|
|
233
279
|
private _cachedSeqStart?: number
|
|
234
280
|
|
|
@@ -374,7 +420,9 @@ export default class BamRecord {
|
|
|
374
420
|
private _computeTags() {
|
|
375
421
|
const blockEnd = this._end
|
|
376
422
|
const ba = this._byteArray
|
|
377
|
-
|
|
423
|
+
// null prototype: tag names come from the file, so a read carrying a
|
|
424
|
+
// "constructor" or "toString" tag must not resolve to Object.prototype's
|
|
425
|
+
const tags: Record<string, unknown> = Object.create(null)
|
|
378
426
|
let p = this.tagsStart
|
|
379
427
|
while (p < blockEnd) {
|
|
380
428
|
const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
|
|
@@ -499,6 +547,8 @@ export default class BamRecord {
|
|
|
499
547
|
absOffset,
|
|
500
548
|
numCigarOps,
|
|
501
549
|
)
|
|
550
|
+
// the view we need to sum is exactly what NUMERIC_CIGAR would build, so
|
|
551
|
+
// seed its cache rather than making it construct a second one
|
|
502
552
|
this._cachedNumericCigar = cigarView
|
|
503
553
|
let lref = 0
|
|
504
554
|
for (let c = 0; c < numCigarOps; ++c) {
|
|
@@ -516,7 +566,7 @@ export default class BamRecord {
|
|
|
516
566
|
return lref
|
|
517
567
|
}
|
|
518
568
|
|
|
519
|
-
private _computeNumericCigar():
|
|
569
|
+
private _computeNumericCigar(): NumericCigar {
|
|
520
570
|
const flag_nc = this._dataView.getInt32(this._start + 16, true)
|
|
521
571
|
if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
|
|
522
572
|
return new Uint32Array(0)
|
|
@@ -526,10 +576,10 @@ export default class BamRecord {
|
|
|
526
576
|
const p = this.b0 + this.read_name_length
|
|
527
577
|
|
|
528
578
|
if (this._isCGTagPattern(p, numCigarOps)) {
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
)
|
|
579
|
+
// getTag, not this.tags: the real CIGAR lives in one tag, so there's no
|
|
580
|
+
// reason to decode every other tag on the record to reach it
|
|
581
|
+
const cg = this.getTag('CG')
|
|
582
|
+
return isNumericCigar(cg) ? cg : new Uint32Array(0)
|
|
533
583
|
}
|
|
534
584
|
|
|
535
585
|
const absOffset = this._byteArray.byteOffset + p
|
|
@@ -591,43 +641,66 @@ export default class BamRecord {
|
|
|
591
641
|
return this._byteArray.subarray(p, p + this.num_seq_bytes)
|
|
592
642
|
}
|
|
593
643
|
|
|
644
|
+
// Decode two bases per iteration off a 256-entry table. Building an array of
|
|
645
|
+
// 1-char strings and join()ing it — the obvious approach — is 3x slower at
|
|
646
|
+
// 100bp and 35x slower at 15kb.
|
|
594
647
|
get seq() {
|
|
595
648
|
const len = this.seq_length
|
|
596
|
-
const
|
|
597
|
-
const
|
|
598
|
-
const
|
|
599
|
-
let
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
649
|
+
const ba = this._byteArray
|
|
650
|
+
const p = this.seqStart
|
|
651
|
+
const nPairs = len >> 1
|
|
652
|
+
let seq: string
|
|
653
|
+
if (len < SEQ_DECODER_THRESHOLD) {
|
|
654
|
+
seq = ''
|
|
655
|
+
for (let j = 0; j < nPairs; j++) {
|
|
656
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
|
|
657
|
+
}
|
|
658
|
+
if (len & 1) {
|
|
659
|
+
seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
|
|
660
|
+
}
|
|
661
|
+
} else {
|
|
662
|
+
// round up to an even length so the Uint16Array view spans every pair; a
|
|
663
|
+
// trailing odd base is written as a single byte and trimmed on decode
|
|
664
|
+
const out = new Uint8Array((len + 1) & ~1)
|
|
665
|
+
const pairs = new Uint16Array(out.buffer)
|
|
666
|
+
for (let j = 0; j < nPairs; j++) {
|
|
667
|
+
pairs[j] = SEQRET_PAIR_CODES[ba[p + j]!]!
|
|
668
|
+
}
|
|
669
|
+
if (len & 1) {
|
|
670
|
+
out[len - 1] = SEQRET_CODES[(ba[p + nPairs]! & 0xf0) >> 4]!
|
|
671
|
+
}
|
|
672
|
+
seq = textDecoder.decode(out.subarray(0, len))
|
|
611
673
|
}
|
|
612
|
-
|
|
613
|
-
return buf.join('')
|
|
674
|
+
return seq
|
|
614
675
|
}
|
|
615
676
|
|
|
616
|
-
//
|
|
617
|
-
//
|
|
618
|
-
//
|
|
619
|
-
//
|
|
620
|
-
//
|
|
677
|
+
// Must come out identical from either mate, or the two halves of one normal
|
|
678
|
+
// pair render as different orientations. The leftmost mate is therefore picked
|
|
679
|
+
// by a total order on (refId, pos) that both mates evaluate the same way, with
|
|
680
|
+
// a read1-first tie-break for equal loci.
|
|
681
|
+
//
|
|
682
|
+
// Deriving "leftmost" from template_length looks tempting — the spec makes
|
|
683
|
+
// tlen positive for the leftmost segment and negative for the rightmost — but
|
|
684
|
+
// aligners leave tlen at 0 whenever the insert size is unavailable, which
|
|
685
|
+
// includes every cross-reference pair. Both mates then read as "not leftmost"
|
|
686
|
+
// and disagree with each other.
|
|
621
687
|
// (see also: gmod/cram-js src/cramFile/record.ts getPairOrientation)
|
|
622
688
|
get pair_orientation() {
|
|
623
689
|
const f = this.flags
|
|
624
|
-
|
|
625
|
-
if (f & 0xc || this.ref_id !== this.next_refid) {
|
|
690
|
+
if (!(f & Constants.BAM_FPAIRED)) {
|
|
626
691
|
return undefined
|
|
627
692
|
}
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
693
|
+
const refId = this.ref_id
|
|
694
|
+
const mateRefId = this.next_refid
|
|
695
|
+
const pos = this.start
|
|
696
|
+
const matePos = this.next_pos
|
|
697
|
+
const selfIsLeft =
|
|
698
|
+
refId !== mateRefId
|
|
699
|
+
? refId < mateRefId
|
|
700
|
+
: pos !== matePos
|
|
701
|
+
? pos < matePos
|
|
702
|
+
: !!(f & Constants.BAM_FREAD1)
|
|
703
|
+
return PAIR_ORIENTATION_TABLE[((f >> 4) & 0x7) | (selfIsLeft ? 8 : 0)]
|
|
631
704
|
}
|
|
632
705
|
|
|
633
706
|
get bin_mq_nl() {
|
|
@@ -658,7 +731,7 @@ export default class BamRecord {
|
|
|
658
731
|
if (idx < this.seq_length) {
|
|
659
732
|
const sb = this._byteArray[this.seqStart + (idx >> 1)]!
|
|
660
733
|
|
|
661
|
-
return idx
|
|
734
|
+
return (idx & 1) === 0
|
|
662
735
|
? SEQRET_DECODER[(sb & 0xf0) >> 4]!
|
|
663
736
|
: SEQRET_DECODER[sb & 0x0f]!
|
|
664
737
|
} else {
|
package/src/util.ts
CHANGED
|
@@ -126,7 +126,10 @@ export function parsePseudoBin(bytes: Uint8Array, offset: number) {
|
|
|
126
126
|
// maxv.blockPosition is an upper bound on where that final block ends — always at
|
|
127
127
|
// least the true block end, so the clamped fetch still contains the whole block.
|
|
128
128
|
// Shrinks both the byte estimate and the actual fetch with no extra I/O.
|
|
129
|
-
export function clampChunkEnds(
|
|
129
|
+
export function clampChunkEnds(
|
|
130
|
+
chunks: Chunk[],
|
|
131
|
+
extraBoundaries: number[] = [],
|
|
132
|
+
) {
|
|
130
133
|
const boundaries = [...extraBoundaries]
|
|
131
134
|
for (const c of chunks) {
|
|
132
135
|
boundaries.push(c.minv.blockPosition, c.maxv.blockPosition)
|
|
@@ -163,9 +166,15 @@ export function parseRefSeqs(
|
|
|
163
166
|
if (start + 4 > uncba.length) {
|
|
164
167
|
return undefined
|
|
165
168
|
}
|
|
166
|
-
const dataView = new DataView(
|
|
169
|
+
const dataView = new DataView(
|
|
170
|
+
uncba.buffer,
|
|
171
|
+
uncba.byteOffset,
|
|
172
|
+
uncba.byteLength,
|
|
173
|
+
)
|
|
167
174
|
const nRef = dataView.getInt32(start, true)
|
|
168
|
-
|
|
175
|
+
// null prototype: ref names come from the file, so a contig named
|
|
176
|
+
// "constructor" must not resolve to Object.prototype's
|
|
177
|
+
const chrToIndex: Record<string, number> = Object.create(null)
|
|
169
178
|
const indexToChr: { refName: string; length: number }[] = []
|
|
170
179
|
const decoder = new TextDecoder('utf8')
|
|
171
180
|
|
|
@@ -209,7 +218,7 @@ export function parseNameBytes(
|
|
|
209
218
|
let currRefId = 0
|
|
210
219
|
let currNameStart = 0
|
|
211
220
|
const refIdToName: string[] = []
|
|
212
|
-
const refNameToId: Record<string, number> =
|
|
221
|
+
const refNameToId: Record<string, number> = Object.create(null)
|
|
213
222
|
for (let i = 0; i < namesBytes.length; i++) {
|
|
214
223
|
if (!namesBytes[i]) {
|
|
215
224
|
if (currNameStart < i) {
|
|
@@ -254,18 +263,19 @@ export function filterTagValue(readVal: unknown, filterVal?: string) {
|
|
|
254
263
|
: `${readVal}` !== `${filterVal}`
|
|
255
264
|
}
|
|
256
265
|
|
|
257
|
-
export function filterCacheKey(filterBy?: FilterBy) {
|
|
258
|
-
if (!filterBy) {
|
|
259
|
-
return ''
|
|
260
|
-
}
|
|
261
|
-
const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
|
|
262
|
-
const tagPart = tagFilter ? `:${tagFilter.tag}=${tagFilter.value ?? '*'}` : ''
|
|
263
|
-
return `:f${flagInclude}x${flagExclude}${tagPart}`
|
|
264
|
-
}
|
|
265
|
-
|
|
266
266
|
interface Filterable {
|
|
267
267
|
flags: number
|
|
268
268
|
tags: Record<string, unknown>
|
|
269
|
+
// BamRecord decodes one tag by walking the tag block, without building the
|
|
270
|
+
// whole tags object. Optional because a custom recordClass need not have it.
|
|
271
|
+
getTag?(tag: string): unknown
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// Read a single tag, preferring the targeted accessor. Reaching for `tags`
|
|
275
|
+
// instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
|
|
276
|
+
// ~10 per read) just to test one, which measured 2.6x the cost of getTag.
|
|
277
|
+
function readTag(record: Filterable, tag: string) {
|
|
278
|
+
return record.getTag ? record.getTag(tag) : record.tags[tag]
|
|
269
279
|
}
|
|
270
280
|
|
|
271
281
|
// Apply flagInclude/flagExclude/tagFilter to a list of records.
|
|
@@ -279,7 +289,7 @@ export function applyFilters<T extends Filterable>(
|
|
|
279
289
|
const r = records[i]!
|
|
280
290
|
if (
|
|
281
291
|
!filterReadFlag(r.flags, flagInclude, flagExclude) &&
|
|
282
|
-
!(tagFilter && filterTagValue(r
|
|
292
|
+
!(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
|
|
283
293
|
) {
|
|
284
294
|
out.push(r)
|
|
285
295
|
}
|