@gmod/bam 7.3.3 → 7.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/bai.js +10 -7
- package/dist/bai.js.map +1 -1
- package/dist/bamFile.d.ts +25 -6
- package/dist/bamFile.js +91 -33
- package/dist/bamFile.js.map +1 -1
- package/dist/csi.js +2 -2
- package/dist/csi.js.map +1 -1
- package/dist/htsget.js +2 -2
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/long.d.ts +0 -2
- package/dist/long.js +2 -4
- package/dist/long.js.map +1 -1
- package/dist/record.d.ts +3 -2
- package/dist/record.js +198 -178
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +1 -1
- package/dist/util.js +11 -12
- package/dist/util.js.map +1 -1
- package/dist/virtualOffset.d.ts +1 -1
- package/dist/virtualOffset.js +1 -4
- package/dist/virtualOffset.js.map +1 -1
- package/esm/bai.js +10 -7
- package/esm/bai.js.map +1 -1
- package/esm/bamFile.d.ts +25 -6
- package/esm/bamFile.js +91 -33
- package/esm/bamFile.js.map +1 -1
- package/esm/csi.js +2 -2
- package/esm/csi.js.map +1 -1
- package/esm/htsget.js +2 -2
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +1 -1
- package/esm/index.js +1 -1
- package/esm/index.js.map +1 -1
- package/esm/long.d.ts +0 -2
- package/esm/long.js +1 -2
- package/esm/long.js.map +1 -1
- package/esm/record.d.ts +3 -2
- package/esm/record.js +198 -178
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +1 -1
- package/esm/util.js +11 -11
- package/esm/util.js.map +1 -1
- package/esm/virtualOffset.d.ts +1 -1
- package/esm/virtualOffset.js +1 -4
- package/esm/virtualOffset.js.map +1 -1
- package/package.json +2 -1
- package/src/bai.ts +15 -8
- package/src/bamFile.ts +117 -40
- package/src/csi.ts +10 -2
- package/src/htsget.ts +6 -2
- package/src/index.ts +1 -1
- package/src/long.ts +1 -2
- package/src/record.ts +231 -192
- package/src/util.ts +24 -14
- package/src/virtualOffset.ts +1 -5
package/src/record.ts
CHANGED
|
@@ -1,16 +1,45 @@
|
|
|
1
1
|
import { CIGAR_REF_SKIP, CIGAR_SOFT_CLIP } from './cigar.ts'
|
|
2
2
|
import Constants from './constants.ts'
|
|
3
3
|
|
|
4
|
-
const
|
|
4
|
+
const SEQRET = '=ACMGRSVTWYHKDBN'
|
|
5
|
+
const SEQRET_DECODER = SEQRET.split('')
|
|
6
|
+
const SEQRET_CODES = Uint8Array.from(SEQRET, c => c.charCodeAt(0))
|
|
7
|
+
|
|
8
|
+
// Both bases of a SEQ byte, precomputed for all 256 bytes so decoding advances a
|
|
9
|
+
// byte at a time. Two forms because `seq` has two strategies (see below): packed
|
|
10
|
+
// ASCII codes for a single Uint16Array store, and the 2-char string to append.
|
|
11
|
+
// Byte order within the u16 depends on host endianness.
|
|
12
|
+
const LITTLE_ENDIAN = new Uint16Array(Uint8Array.of(1, 0).buffer)[0] === 1
|
|
13
|
+
const SEQRET_PAIR_CODES = new Uint16Array(256)
|
|
14
|
+
const SEQRET_PAIR_STRINGS = new Array<string>(256)
|
|
15
|
+
for (let hi = 0; hi < 16; hi++) {
|
|
16
|
+
for (let lo = 0; lo < 16; lo++) {
|
|
17
|
+
const h = SEQRET_CODES[hi]!
|
|
18
|
+
const l = SEQRET_CODES[lo]!
|
|
19
|
+
SEQRET_PAIR_CODES[(hi << 4) | lo] = LITTLE_ENDIAN
|
|
20
|
+
? h | (l << 8)
|
|
21
|
+
: (h << 8) | l
|
|
22
|
+
SEQRET_PAIR_STRINGS[(hi << 4) | lo] =
|
|
23
|
+
SEQRET_DECODER[hi]! + SEQRET_DECODER[lo]!
|
|
24
|
+
}
|
|
25
|
+
}
|
|
5
26
|
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
27
|
+
// Read length at which building a byte buffer and calling TextDecoder once
|
|
28
|
+
// overtakes plain string concatenation. Below it concat wins by 2-4x (rope
|
|
29
|
+
// building is cheap and the decode has fixed overhead); above it the decoder
|
|
30
|
+
// wins by 5-8x. Measured crossover is ~300bp, i.e. just past typical Illumina
|
|
31
|
+
// read lengths.
|
|
32
|
+
const SEQ_DECODER_THRESHOLD = 300
|
|
33
|
+
|
|
34
|
+
// Precomputed pair orientation strings, indexed by
|
|
35
|
+
// ((flags >> 4) & 0x7) | (selfIsLeft ? 8 : 0)
|
|
36
|
+
// bits 0-2 are flag bits 0x10 (self reverse), 0x20 (mate reverse), 0x40 (read1);
|
|
37
|
+
// bit 3 is whether this read is the leftmost of the pair. The read2 flag (0x80)
|
|
38
|
+
// is deliberately not consulted — "not read1" is what decides the numbering, so
|
|
39
|
+
// a record with neither flag set reads the same as read2.
|
|
9
40
|
// prettier-ignore
|
|
10
41
|
const PAIR_ORIENTATION_TABLE = [
|
|
11
|
-
'F F ','F R ','R F ','R R ','F2F1','F2R1','R2F1','R2R1',
|
|
12
42
|
'F1F2','F1R2','R1F2','R1R2','F2F1','F2R1','R2F1','R2R1',
|
|
13
|
-
'F F ','R F ','F R ','R R ','F1F2','R1F2','F1R2','R1R2',
|
|
14
43
|
'F2F1','R2F1','F2R1','R2R1','F1F2','R1F2','F1R2','R1R2',
|
|
15
44
|
]
|
|
16
45
|
const ASCII_CIGAR_CODES = [
|
|
@@ -23,6 +52,19 @@ const textDecoder = new TextDecoder()
|
|
|
23
52
|
// Binary: 0b111001101 = 0x1CD
|
|
24
53
|
const CIGAR_CONSUMES_REF_MASK = 0x1cd
|
|
25
54
|
|
|
55
|
+
// A CIGAR as packed op words. Either a view over (or copy of) the record's own
|
|
56
|
+
// CIGAR field, or the CG tag's array for long-CIGAR records — hence Int32Array
|
|
57
|
+
// too, since a writer may encode CG as B:i rather than the usual B:I.
|
|
58
|
+
export type NumericCigar = Uint32Array | Int32Array | number[]
|
|
59
|
+
|
|
60
|
+
function isNumericCigar(value: unknown): value is NumericCigar {
|
|
61
|
+
return (
|
|
62
|
+
value instanceof Uint32Array ||
|
|
63
|
+
value instanceof Int32Array ||
|
|
64
|
+
Array.isArray(value)
|
|
65
|
+
)
|
|
66
|
+
}
|
|
67
|
+
|
|
26
68
|
export interface Bytes {
|
|
27
69
|
start: number
|
|
28
70
|
end: number
|
|
@@ -42,7 +84,11 @@ type BArrayValue =
|
|
|
42
84
|
// Decode a 'B' (array) tag value starting at `p` (the byte after type+subtype+
|
|
43
85
|
// count). When the data is naturally aligned we return a typed-array view over
|
|
44
86
|
// the underlying buffer (zero-copy); otherwise we copy element-by-element since
|
|
45
|
-
// typed-array views require alignment.
|
|
87
|
+
// typed-array views require alignment. The copy target is a plain array, not a
|
|
88
|
+
// typed one: benchmarking the unaligned path found number[] fills faster below
|
|
89
|
+
// ~10k elements and reads at least as fast at every size, so the only thing a
|
|
90
|
+
// typed copy would buy is half the retained bytes. Shared by getTag and the
|
|
91
|
+
// full-tag parse.
|
|
46
92
|
function decodeBArrayTag(
|
|
47
93
|
ba: Uint8Array,
|
|
48
94
|
dataView: DataView,
|
|
@@ -127,6 +173,97 @@ function bArrayByteLength(Btype: number, limit: number) {
|
|
|
127
173
|
}
|
|
128
174
|
}
|
|
129
175
|
|
|
176
|
+
// Cursor position just past a tag's value — i.e. the start of the next tag —
|
|
177
|
+
// given the value starts at `p`. For null-terminated Z/H this steps over the
|
|
178
|
+
// terminator. Returns 0 for an unknown type, whose value width is unknowable so
|
|
179
|
+
// the caller must stop. Shared by _findTag and _computeTags so the two loops
|
|
180
|
+
// can't drift in how they walk the tag layout. (0 is an impossible real end:
|
|
181
|
+
// `p` is always past the fixed record prefix.)
|
|
182
|
+
function tagValueEnd(
|
|
183
|
+
ba: Uint8Array,
|
|
184
|
+
dataView: DataView,
|
|
185
|
+
type: number,
|
|
186
|
+
p: number,
|
|
187
|
+
blockEnd: number,
|
|
188
|
+
) {
|
|
189
|
+
switch (type) {
|
|
190
|
+
case 0x41: // 'A'
|
|
191
|
+
case 0x63: // 'c'
|
|
192
|
+
case 0x43: // 'C'
|
|
193
|
+
return p + 1
|
|
194
|
+
case 0x73: // 's'
|
|
195
|
+
case 0x53: // 'S'
|
|
196
|
+
return p + 2
|
|
197
|
+
case 0x69: // 'i'
|
|
198
|
+
case 0x49: // 'I'
|
|
199
|
+
case 0x66: {
|
|
200
|
+
// 'f'
|
|
201
|
+
return p + 4
|
|
202
|
+
}
|
|
203
|
+
case 0x5a: // 'Z'
|
|
204
|
+
case 0x48: {
|
|
205
|
+
// 'H'
|
|
206
|
+
let q = p
|
|
207
|
+
while (q < blockEnd && ba[q] !== 0) {
|
|
208
|
+
q++
|
|
209
|
+
}
|
|
210
|
+
return q + 1 // step past the null terminator
|
|
211
|
+
}
|
|
212
|
+
case 0x42: {
|
|
213
|
+
// 'B'
|
|
214
|
+
const Btype = ba[p]!
|
|
215
|
+
const limit = dataView.getInt32(p + 1, true)
|
|
216
|
+
return p + 5 + bArrayByteLength(Btype, limit)
|
|
217
|
+
}
|
|
218
|
+
default:
|
|
219
|
+
console.error('Unknown BAM tag type', type)
|
|
220
|
+
return 0
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// Decode the value of a tag of `type` whose bytes start at `p`; `end` is the
|
|
225
|
+
// next-tag cursor from tagValueEnd (used only to bound Z/H strings, whose bytes
|
|
226
|
+
// run up to the null terminator at end-1). When `raw`, Z/H return the undecoded
|
|
227
|
+
// byte subarray. Assumes `type` is a known scalar/string/B type.
|
|
228
|
+
function decodeTagValue(
|
|
229
|
+
ba: Uint8Array,
|
|
230
|
+
dataView: DataView,
|
|
231
|
+
type: number,
|
|
232
|
+
p: number,
|
|
233
|
+
end: number,
|
|
234
|
+
raw: boolean,
|
|
235
|
+
): unknown {
|
|
236
|
+
switch (type) {
|
|
237
|
+
case 0x41: // 'A'
|
|
238
|
+
return String.fromCharCode(ba[p]!)
|
|
239
|
+
case 0x69: // 'i'
|
|
240
|
+
return dataView.getInt32(p, true)
|
|
241
|
+
case 0x49: // 'I'
|
|
242
|
+
return dataView.getUint32(p, true)
|
|
243
|
+
case 0x63: // 'c'
|
|
244
|
+
return dataView.getInt8(p)
|
|
245
|
+
case 0x43: // 'C'
|
|
246
|
+
return dataView.getUint8(p)
|
|
247
|
+
case 0x73: // 's'
|
|
248
|
+
return dataView.getInt16(p, true)
|
|
249
|
+
case 0x53: // 'S'
|
|
250
|
+
return dataView.getUint16(p, true)
|
|
251
|
+
case 0x66: // 'f'
|
|
252
|
+
return dataView.getFloat32(p, true)
|
|
253
|
+
case 0x5a: // 'Z'
|
|
254
|
+
case 0x48: // 'H'
|
|
255
|
+
return raw
|
|
256
|
+
? ba.subarray(p, end - 1)
|
|
257
|
+
: textDecoder.decode(ba.subarray(p, end - 1))
|
|
258
|
+
default: {
|
|
259
|
+
// 'B'
|
|
260
|
+
const Btype = ba[p]!
|
|
261
|
+
const limit = dataView.getInt32(p + 1, true)
|
|
262
|
+
return decodeBArrayTag(ba, dataView, p + 5, Btype, limit)
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
130
267
|
export default class BamRecord {
|
|
131
268
|
public fileOffset: number
|
|
132
269
|
private _byteArray: Uint8Array
|
|
@@ -137,7 +274,7 @@ export default class BamRecord {
|
|
|
137
274
|
private _cachedEnd?: number
|
|
138
275
|
private _cachedTags?: Record<string, unknown>
|
|
139
276
|
private _cachedLengthOnRef?: number
|
|
140
|
-
private _cachedNumericCigar?:
|
|
277
|
+
private _cachedNumericCigar?: NumericCigar
|
|
141
278
|
private _cachedNUMERIC_MD?: Uint8Array | null
|
|
142
279
|
private _cachedSeqStart?: number
|
|
143
280
|
|
|
@@ -261,172 +398,49 @@ export default class BamRecord {
|
|
|
261
398
|
private _findTag(tagName: string, raw: boolean) {
|
|
262
399
|
const tag1 = tagName.charCodeAt(0)
|
|
263
400
|
const tag2 = tagName.charCodeAt(1)
|
|
264
|
-
|
|
265
|
-
let p = this.tagsStart
|
|
266
|
-
|
|
267
401
|
const blockEnd = this._end
|
|
268
402
|
const ba = this._byteArray
|
|
403
|
+
let p = this.tagsStart
|
|
269
404
|
while (p < blockEnd) {
|
|
270
|
-
const
|
|
271
|
-
const
|
|
272
|
-
const
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
if (isMatch) {
|
|
280
|
-
return String.fromCharCode(ba[p]!)
|
|
281
|
-
}
|
|
282
|
-
p += 1
|
|
283
|
-
break
|
|
284
|
-
case 0x69: // 'i'
|
|
285
|
-
if (isMatch) {
|
|
286
|
-
return this._dataView.getInt32(p, true)
|
|
287
|
-
}
|
|
288
|
-
p += 4
|
|
289
|
-
break
|
|
290
|
-
case 0x49: // 'I'
|
|
291
|
-
if (isMatch) {
|
|
292
|
-
return this._dataView.getUint32(p, true)
|
|
293
|
-
}
|
|
294
|
-
p += 4
|
|
295
|
-
break
|
|
296
|
-
case 0x63: // 'c'
|
|
297
|
-
if (isMatch) {
|
|
298
|
-
return this._dataView.getInt8(p)
|
|
299
|
-
}
|
|
300
|
-
p += 1
|
|
301
|
-
break
|
|
302
|
-
case 0x43: // 'C'
|
|
303
|
-
if (isMatch) {
|
|
304
|
-
return this._dataView.getUint8(p)
|
|
305
|
-
}
|
|
306
|
-
p += 1
|
|
307
|
-
break
|
|
308
|
-
case 0x73: // 's'
|
|
309
|
-
if (isMatch) {
|
|
310
|
-
return this._dataView.getInt16(p, true)
|
|
311
|
-
}
|
|
312
|
-
p += 2
|
|
313
|
-
break
|
|
314
|
-
case 0x53: // 'S'
|
|
315
|
-
if (isMatch) {
|
|
316
|
-
return this._dataView.getUint16(p, true)
|
|
317
|
-
}
|
|
318
|
-
p += 2
|
|
319
|
-
break
|
|
320
|
-
case 0x66: // 'f'
|
|
321
|
-
if (isMatch) {
|
|
322
|
-
return this._dataView.getFloat32(p, true)
|
|
323
|
-
}
|
|
324
|
-
p += 4
|
|
325
|
-
break
|
|
326
|
-
case 0x5a: // 'Z'
|
|
327
|
-
case 0x48: {
|
|
328
|
-
// 'H'
|
|
329
|
-
const start = p
|
|
330
|
-
while (p < blockEnd && ba[p] !== 0) {
|
|
331
|
-
p++
|
|
332
|
-
}
|
|
333
|
-
if (isMatch) {
|
|
334
|
-
return raw
|
|
335
|
-
? ba.subarray(start, p)
|
|
336
|
-
: textDecoder.decode(ba.subarray(start, p))
|
|
337
|
-
}
|
|
338
|
-
p++ // advance past null terminator
|
|
339
|
-
break
|
|
340
|
-
}
|
|
341
|
-
case 0x42: {
|
|
342
|
-
// 'B'
|
|
343
|
-
const Btype = ba[p++]!
|
|
344
|
-
const limit = this._dataView.getInt32(p, true)
|
|
345
|
-
p += 4
|
|
346
|
-
if (isMatch) {
|
|
347
|
-
return decodeBArrayTag(ba, this._dataView, p, Btype, limit)
|
|
348
|
-
}
|
|
349
|
-
p += bArrayByteLength(Btype, limit)
|
|
350
|
-
break
|
|
351
|
-
}
|
|
352
|
-
default:
|
|
353
|
-
if (type !== undefined) {
|
|
354
|
-
console.error('Unknown BAM tag type', type)
|
|
355
|
-
}
|
|
356
|
-
break
|
|
405
|
+
const isMatch = ba[p] === tag1 && ba[p + 1] === tag2
|
|
406
|
+
const type = ba[p + 2]!
|
|
407
|
+
const valueStart = p + 3
|
|
408
|
+
const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
|
|
409
|
+
if (end === 0) {
|
|
410
|
+
break // unknown type: can't compute how far to advance
|
|
411
|
+
}
|
|
412
|
+
if (isMatch) {
|
|
413
|
+
return decodeTagValue(ba, this._dataView, type, valueStart, end, raw)
|
|
357
414
|
}
|
|
415
|
+
p = end
|
|
358
416
|
}
|
|
359
417
|
return undefined
|
|
360
418
|
}
|
|
361
419
|
|
|
362
420
|
private _computeTags() {
|
|
363
|
-
let p = this.tagsStart
|
|
364
|
-
|
|
365
421
|
const blockEnd = this._end
|
|
366
422
|
const ba = this._byteArray
|
|
367
|
-
|
|
423
|
+
// null prototype: tag names come from the file, so a read carrying a
|
|
424
|
+
// "constructor" or "toString" tag must not resolve to Object.prototype's
|
|
425
|
+
const tags: Record<string, unknown> = Object.create(null)
|
|
426
|
+
let p = this.tagsStart
|
|
368
427
|
while (p < blockEnd) {
|
|
369
428
|
const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
|
|
370
429
|
const type = ba[p + 2]!
|
|
371
|
-
p
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
tags[tag] = String.fromCharCode(ba[p]!)
|
|
376
|
-
p += 1
|
|
377
|
-
break
|
|
378
|
-
case 0x69: // 'i'
|
|
379
|
-
tags[tag] = this._dataView.getInt32(p, true)
|
|
380
|
-
p += 4
|
|
381
|
-
break
|
|
382
|
-
case 0x49: // 'I'
|
|
383
|
-
tags[tag] = this._dataView.getUint32(p, true)
|
|
384
|
-
p += 4
|
|
385
|
-
break
|
|
386
|
-
case 0x63: // 'c'
|
|
387
|
-
tags[tag] = this._dataView.getInt8(p)
|
|
388
|
-
p += 1
|
|
389
|
-
break
|
|
390
|
-
case 0x43: // 'C'
|
|
391
|
-
tags[tag] = this._dataView.getUint8(p)
|
|
392
|
-
p += 1
|
|
393
|
-
break
|
|
394
|
-
case 0x73: // 's'
|
|
395
|
-
tags[tag] = this._dataView.getInt16(p, true)
|
|
396
|
-
p += 2
|
|
397
|
-
break
|
|
398
|
-
case 0x53: // 'S'
|
|
399
|
-
tags[tag] = this._dataView.getUint16(p, true)
|
|
400
|
-
p += 2
|
|
401
|
-
break
|
|
402
|
-
case 0x66: // 'f'
|
|
403
|
-
tags[tag] = this._dataView.getFloat32(p, true)
|
|
404
|
-
p += 4
|
|
405
|
-
break
|
|
406
|
-
case 0x5a: // 'Z'
|
|
407
|
-
case 0x48: {
|
|
408
|
-
// 'H'
|
|
409
|
-
const start = p
|
|
410
|
-
while (p < blockEnd && ba[p] !== 0) {
|
|
411
|
-
p++
|
|
412
|
-
}
|
|
413
|
-
tags[tag] = textDecoder.decode(ba.subarray(start, p))
|
|
414
|
-
p++ // advance past null terminator
|
|
415
|
-
break
|
|
416
|
-
}
|
|
417
|
-
case 0x42: {
|
|
418
|
-
// 'B'
|
|
419
|
-
const Btype = ba[p++]!
|
|
420
|
-
const limit = this._dataView.getInt32(p, true)
|
|
421
|
-
p += 4
|
|
422
|
-
tags[tag] = decodeBArrayTag(ba, this._dataView, p, Btype, limit)
|
|
423
|
-
p += bArrayByteLength(Btype, limit)
|
|
424
|
-
break
|
|
425
|
-
}
|
|
426
|
-
default:
|
|
427
|
-
console.error('Unknown BAM tag type', type)
|
|
428
|
-
break
|
|
430
|
+
const valueStart = p + 3
|
|
431
|
+
const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
|
|
432
|
+
if (end === 0) {
|
|
433
|
+
break // unknown type: can't compute how far to advance
|
|
429
434
|
}
|
|
435
|
+
tags[tag] = decodeTagValue(
|
|
436
|
+
ba,
|
|
437
|
+
this._dataView,
|
|
438
|
+
type,
|
|
439
|
+
valueStart,
|
|
440
|
+
end,
|
|
441
|
+
false,
|
|
442
|
+
)
|
|
443
|
+
p = end
|
|
430
444
|
}
|
|
431
445
|
return tags
|
|
432
446
|
}
|
|
@@ -533,6 +547,8 @@ export default class BamRecord {
|
|
|
533
547
|
absOffset,
|
|
534
548
|
numCigarOps,
|
|
535
549
|
)
|
|
550
|
+
// the view we need to sum is exactly what NUMERIC_CIGAR would build, so
|
|
551
|
+
// seed its cache rather than making it construct a second one
|
|
536
552
|
this._cachedNumericCigar = cigarView
|
|
537
553
|
let lref = 0
|
|
538
554
|
for (let c = 0; c < numCigarOps; ++c) {
|
|
@@ -550,7 +566,7 @@ export default class BamRecord {
|
|
|
550
566
|
return lref
|
|
551
567
|
}
|
|
552
568
|
|
|
553
|
-
private _computeNumericCigar():
|
|
569
|
+
private _computeNumericCigar(): NumericCigar {
|
|
554
570
|
const flag_nc = this._dataView.getInt32(this._start + 16, true)
|
|
555
571
|
if (flag_nc & (Constants.BAM_FUNMAP << 16)) {
|
|
556
572
|
return new Uint32Array(0)
|
|
@@ -560,10 +576,10 @@ export default class BamRecord {
|
|
|
560
576
|
const p = this.b0 + this.read_name_length
|
|
561
577
|
|
|
562
578
|
if (this._isCGTagPattern(p, numCigarOps)) {
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
)
|
|
579
|
+
// getTag, not this.tags: the real CIGAR lives in one tag, so there's no
|
|
580
|
+
// reason to decode every other tag on the record to reach it
|
|
581
|
+
const cg = this.getTag('CG')
|
|
582
|
+
return isNumericCigar(cg) ? cg : new Uint32Array(0)
|
|
567
583
|
}
|
|
568
584
|
|
|
569
585
|
const absOffset = this._byteArray.byteOffset + p
|
|
@@ -625,43 +641,66 @@ export default class BamRecord {
|
|
|
625
641
|
return this._byteArray.subarray(p, p + this.num_seq_bytes)
|
|
626
642
|
}
|
|
627
643
|
|
|
644
|
+
// Decode two bases per iteration off a 256-entry table. Building an array of
|
|
645
|
+
// 1-char strings and join()ing it — the obvious approach — is 3x slower at
|
|
646
|
+
// 100bp and 35x slower at 15kb.
|
|
628
647
|
get seq() {
|
|
629
648
|
const len = this.seq_length
|
|
630
|
-
const
|
|
631
|
-
const
|
|
632
|
-
const
|
|
633
|
-
let
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
649
|
+
const ba = this._byteArray
|
|
650
|
+
const p = this.seqStart
|
|
651
|
+
const nPairs = len >> 1
|
|
652
|
+
let seq: string
|
|
653
|
+
if (len < SEQ_DECODER_THRESHOLD) {
|
|
654
|
+
seq = ''
|
|
655
|
+
for (let j = 0; j < nPairs; j++) {
|
|
656
|
+
seq += SEQRET_PAIR_STRINGS[ba[p + j]!]!
|
|
657
|
+
}
|
|
658
|
+
if (len & 1) {
|
|
659
|
+
seq += SEQRET_DECODER[(ba[p + nPairs]! & 0xf0) >> 4]!
|
|
660
|
+
}
|
|
661
|
+
} else {
|
|
662
|
+
// round up to an even length so the Uint16Array view spans every pair; a
|
|
663
|
+
// trailing odd base is written as a single byte and trimmed on decode
|
|
664
|
+
const out = new Uint8Array((len + 1) & ~1)
|
|
665
|
+
const pairs = new Uint16Array(out.buffer)
|
|
666
|
+
for (let j = 0; j < nPairs; j++) {
|
|
667
|
+
pairs[j] = SEQRET_PAIR_CODES[ba[p + j]!]!
|
|
668
|
+
}
|
|
669
|
+
if (len & 1) {
|
|
670
|
+
out[len - 1] = SEQRET_CODES[(ba[p + nPairs]! & 0xf0) >> 4]!
|
|
671
|
+
}
|
|
672
|
+
seq = textDecoder.decode(out.subarray(0, len))
|
|
645
673
|
}
|
|
646
|
-
|
|
647
|
-
return buf.join('')
|
|
674
|
+
return seq
|
|
648
675
|
}
|
|
649
676
|
|
|
650
|
-
//
|
|
651
|
-
//
|
|
652
|
-
//
|
|
653
|
-
//
|
|
654
|
-
//
|
|
677
|
+
// Must come out identical from either mate, or the two halves of one normal
|
|
678
|
+
// pair render as different orientations. The leftmost mate is therefore picked
|
|
679
|
+
// by a total order on (refId, pos) that both mates evaluate the same way, with
|
|
680
|
+
// a read1-first tie-break for equal loci.
|
|
681
|
+
//
|
|
682
|
+
// Deriving "leftmost" from template_length looks tempting — the spec makes
|
|
683
|
+
// tlen positive for the leftmost segment and negative for the rightmost — but
|
|
684
|
+
// aligners leave tlen at 0 whenever the insert size is unavailable, which
|
|
685
|
+
// includes every cross-reference pair. Both mates then read as "not leftmost"
|
|
686
|
+
// and disagree with each other.
|
|
655
687
|
// (see also: gmod/cram-js src/cramFile/record.ts getPairOrientation)
|
|
656
688
|
get pair_orientation() {
|
|
657
689
|
const f = this.flags
|
|
658
|
-
|
|
659
|
-
if (f & 0xc || this.ref_id !== this.next_refid) {
|
|
690
|
+
if (!(f & Constants.BAM_FPAIRED)) {
|
|
660
691
|
return undefined
|
|
661
692
|
}
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
693
|
+
const refId = this.ref_id
|
|
694
|
+
const mateRefId = this.next_refid
|
|
695
|
+
const pos = this.start
|
|
696
|
+
const matePos = this.next_pos
|
|
697
|
+
const selfIsLeft =
|
|
698
|
+
refId !== mateRefId
|
|
699
|
+
? refId < mateRefId
|
|
700
|
+
: pos !== matePos
|
|
701
|
+
? pos < matePos
|
|
702
|
+
: !!(f & Constants.BAM_FREAD1)
|
|
703
|
+
return PAIR_ORIENTATION_TABLE[((f >> 4) & 0x7) | (selfIsLeft ? 8 : 0)]
|
|
665
704
|
}
|
|
666
705
|
|
|
667
706
|
get bin_mq_nl() {
|
|
@@ -692,7 +731,7 @@ export default class BamRecord {
|
|
|
692
731
|
if (idx < this.seq_length) {
|
|
693
732
|
const sb = this._byteArray[this.seqStart + (idx >> 1)]!
|
|
694
733
|
|
|
695
|
-
return idx
|
|
734
|
+
return (idx & 1) === 0
|
|
696
735
|
? SEQRET_DECODER[(sb & 0xf0) >> 4]!
|
|
697
736
|
: SEQRET_DECODER[sb & 0x0f]!
|
|
698
737
|
} else {
|
package/src/util.ts
CHANGED
|
@@ -126,7 +126,10 @@ export function parsePseudoBin(bytes: Uint8Array, offset: number) {
|
|
|
126
126
|
// maxv.blockPosition is an upper bound on where that final block ends — always at
|
|
127
127
|
// least the true block end, so the clamped fetch still contains the whole block.
|
|
128
128
|
// Shrinks both the byte estimate and the actual fetch with no extra I/O.
|
|
129
|
-
export function clampChunkEnds(
|
|
129
|
+
export function clampChunkEnds(
|
|
130
|
+
chunks: Chunk[],
|
|
131
|
+
extraBoundaries: number[] = [],
|
|
132
|
+
) {
|
|
130
133
|
const boundaries = [...extraBoundaries]
|
|
131
134
|
for (const c of chunks) {
|
|
132
135
|
boundaries.push(c.minv.blockPosition, c.maxv.blockPosition)
|
|
@@ -163,9 +166,15 @@ export function parseRefSeqs(
|
|
|
163
166
|
if (start + 4 > uncba.length) {
|
|
164
167
|
return undefined
|
|
165
168
|
}
|
|
166
|
-
const dataView = new DataView(
|
|
169
|
+
const dataView = new DataView(
|
|
170
|
+
uncba.buffer,
|
|
171
|
+
uncba.byteOffset,
|
|
172
|
+
uncba.byteLength,
|
|
173
|
+
)
|
|
167
174
|
const nRef = dataView.getInt32(start, true)
|
|
168
|
-
|
|
175
|
+
// null prototype: ref names come from the file, so a contig named
|
|
176
|
+
// "constructor" must not resolve to Object.prototype's
|
|
177
|
+
const chrToIndex: Record<string, number> = Object.create(null)
|
|
169
178
|
const indexToChr: { refName: string; length: number }[] = []
|
|
170
179
|
const decoder = new TextDecoder('utf8')
|
|
171
180
|
|
|
@@ -209,7 +218,7 @@ export function parseNameBytes(
|
|
|
209
218
|
let currRefId = 0
|
|
210
219
|
let currNameStart = 0
|
|
211
220
|
const refIdToName: string[] = []
|
|
212
|
-
const refNameToId: Record<string, number> =
|
|
221
|
+
const refNameToId: Record<string, number> = Object.create(null)
|
|
213
222
|
for (let i = 0; i < namesBytes.length; i++) {
|
|
214
223
|
if (!namesBytes[i]) {
|
|
215
224
|
if (currNameStart < i) {
|
|
@@ -254,18 +263,19 @@ export function filterTagValue(readVal: unknown, filterVal?: string) {
|
|
|
254
263
|
: `${readVal}` !== `${filterVal}`
|
|
255
264
|
}
|
|
256
265
|
|
|
257
|
-
export function filterCacheKey(filterBy?: FilterBy) {
|
|
258
|
-
if (!filterBy) {
|
|
259
|
-
return ''
|
|
260
|
-
}
|
|
261
|
-
const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
|
|
262
|
-
const tagPart = tagFilter ? `:${tagFilter.tag}=${tagFilter.value ?? '*'}` : ''
|
|
263
|
-
return `:f${flagInclude}x${flagExclude}${tagPart}`
|
|
264
|
-
}
|
|
265
|
-
|
|
266
266
|
interface Filterable {
|
|
267
267
|
flags: number
|
|
268
268
|
tags: Record<string, unknown>
|
|
269
|
+
// BamRecord decodes one tag by walking the tag block, without building the
|
|
270
|
+
// whole tags object. Optional because a custom recordClass need not have it.
|
|
271
|
+
getTag?(tag: string): unknown
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// Read a single tag, preferring the targeted accessor. Reaching for `tags`
|
|
275
|
+
// instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
|
|
276
|
+
// ~10 per read) just to test one, which measured 2.6x the cost of getTag.
|
|
277
|
+
function readTag(record: Filterable, tag: string) {
|
|
278
|
+
return record.getTag ? record.getTag(tag) : record.tags[tag]
|
|
269
279
|
}
|
|
270
280
|
|
|
271
281
|
// Apply flagInclude/flagExclude/tagFilter to a list of records.
|
|
@@ -279,7 +289,7 @@ export function applyFilters<T extends Filterable>(
|
|
|
279
289
|
const r = records[i]!
|
|
280
290
|
if (
|
|
281
291
|
!filterReadFlag(r.flags, flagInclude, flagExclude) &&
|
|
282
|
-
!(tagFilter && filterTagValue(r
|
|
292
|
+
!(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
|
|
283
293
|
) {
|
|
284
294
|
out.push(r)
|
|
285
295
|
}
|
package/src/virtualOffset.ts
CHANGED
|
@@ -23,11 +23,7 @@ export class VirtualOffset {
|
|
|
23
23
|
)
|
|
24
24
|
}
|
|
25
25
|
}
|
|
26
|
-
export function fromBytes(bytes: Uint8Array, offset = 0
|
|
27
|
-
if (bigendian) {
|
|
28
|
-
throw new Error('big-endian virtual file offsets not implemented')
|
|
29
|
-
}
|
|
30
|
-
|
|
26
|
+
export function fromBytes(bytes: Uint8Array, offset = 0) {
|
|
31
27
|
return new VirtualOffset(
|
|
32
28
|
bytes[offset + 7]! * 0x10000000000 +
|
|
33
29
|
bytes[offset + 6]! * 0x100000000 +
|