@gmod/bam 7.4.0 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -29
- package/dist/bamFile.d.ts +1 -3
- package/dist/bamFile.js +19 -29
- package/dist/bamFile.js.map +1 -1
- package/dist/index.d.ts +0 -1
- package/dist/record.js +45 -4
- package/dist/record.js.map +1 -1
- package/dist/util.d.ts +0 -18
- package/dist/util.js +0 -30
- package/dist/util.js.map +1 -1
- package/esm/bamFile.d.ts +1 -3
- package/esm/bamFile.js +20 -30
- package/esm/bamFile.js.map +1 -1
- package/esm/index.d.ts +0 -1
- package/esm/record.js +45 -4
- package/esm/record.js.map +1 -1
- package/esm/util.d.ts +0 -18
- package/esm/util.js +0 -27
- package/esm/util.js.map +1 -1
- package/package.json +2 -1
- package/src/bamFile.ts +20 -33
- package/src/index.ts +0 -1
- package/src/record.ts +48 -4
- package/src/util.ts +0 -60
package/src/record.ts
CHANGED
|
@@ -48,6 +48,45 @@ const ASCII_CIGAR_CODES = [
|
|
|
48
48
|
|
|
49
49
|
const textDecoder = new TextDecoder()
|
|
50
50
|
|
|
51
|
+
// Interned two-char tag names keyed on the packed byte pair. Worth more than
|
|
52
|
+
// the saved String.fromCharCode: the tags object is null-prototype and so in
|
|
53
|
+
// dictionary mode, and handing it a string V8 has already hashed halves the
|
|
54
|
+
// per-tag insert cost.
|
|
55
|
+
const TAG_NAMES = new Map<number, string>()
|
|
56
|
+
function tagName(b0: number, b1: number) {
|
|
57
|
+
const key = (b0 << 8) | b1
|
|
58
|
+
let name = TAG_NAMES.get(key)
|
|
59
|
+
if (name === undefined) {
|
|
60
|
+
name = String.fromCharCode(b0, b1)
|
|
61
|
+
TAG_NAMES.set(key, name)
|
|
62
|
+
}
|
|
63
|
+
return name
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
// TextDecoder carries ~0.35us of fixed setup per call, which dominates for the
|
|
67
|
+
// short values Z/H tags usually hold (MD, RG, PG, ...): char codes are 4x
|
|
68
|
+
// faster at 8 bytes, 2x at 16, and the two cross over at 32.
|
|
69
|
+
const SHORT_STRING_THRESHOLD = 32
|
|
70
|
+
|
|
71
|
+
// Z/H tag values are spec'd as printable ASCII, but a non-conforming writer
|
|
72
|
+
// would make the char-code path mis-decode UTF-8, so bail to TextDecoder on any
|
|
73
|
+
// high byte.
|
|
74
|
+
function decodeTagString(ba: Uint8Array, start: number, end: number) {
|
|
75
|
+
const len = end - start
|
|
76
|
+
if (len < SHORT_STRING_THRESHOLD) {
|
|
77
|
+
const codes = new Array<number>(len)
|
|
78
|
+
for (let i = 0; i < len; i++) {
|
|
79
|
+
const byte = ba[start + i]!
|
|
80
|
+
if (byte > 0x7f) {
|
|
81
|
+
return textDecoder.decode(ba.subarray(start, end))
|
|
82
|
+
}
|
|
83
|
+
codes[i] = byte
|
|
84
|
+
}
|
|
85
|
+
return String.fromCharCode(...codes)
|
|
86
|
+
}
|
|
87
|
+
return textDecoder.decode(ba.subarray(start, end))
|
|
88
|
+
}
|
|
89
|
+
|
|
51
90
|
// Bitmask for ops that consume ref: M=0, D=2, N=3, P=6, ==7, X=8
|
|
52
91
|
// Binary: 0b111001101 = 0x1CD
|
|
53
92
|
const CIGAR_CONSUMES_REF_MASK = 0x1cd
|
|
@@ -252,9 +291,7 @@ function decodeTagValue(
|
|
|
252
291
|
return dataView.getFloat32(p, true)
|
|
253
292
|
case 0x5a: // 'Z'
|
|
254
293
|
case 0x48: // 'H'
|
|
255
|
-
return raw
|
|
256
|
-
? ba.subarray(p, end - 1)
|
|
257
|
-
: textDecoder.decode(ba.subarray(p, end - 1))
|
|
294
|
+
return raw ? ba.subarray(p, end - 1) : decodeTagString(ba, p, end - 1)
|
|
258
295
|
default: {
|
|
259
296
|
// 'B'
|
|
260
297
|
const Btype = ba[p]!
|
|
@@ -358,6 +395,13 @@ export default class BamRecord {
|
|
|
358
395
|
}
|
|
359
396
|
|
|
360
397
|
// batch fromCharCode: fastest for typical name lengths (see benchmarks/string-building.bench.ts)
|
|
398
|
+
//
|
|
399
|
+
// Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
|
|
400
|
+
// read name about once — jbrowse-components' buildBaseFeatureData copies it
|
|
401
|
+
// straight into its own FeatureData — so a cache would pay a field slot on
|
|
402
|
+
// every record (+180KB per 22k-record chunk) to save zero decodes, and would
|
|
403
|
+
// pin every name string for as long as the chunk stays cached (+520KB) where
|
|
404
|
+
// today it dies with the consumer's copy.
|
|
361
405
|
get name() {
|
|
362
406
|
const len = this.read_name_length - 1
|
|
363
407
|
const start = this.b0
|
|
@@ -425,7 +469,7 @@ export default class BamRecord {
|
|
|
425
469
|
const tags: Record<string, unknown> = Object.create(null)
|
|
426
470
|
let p = this.tagsStart
|
|
427
471
|
while (p < blockEnd) {
|
|
428
|
-
const tag =
|
|
472
|
+
const tag = tagName(ba[p]!, ba[p + 1]!)
|
|
429
473
|
const type = ba[p + 2]!
|
|
430
474
|
const valueStart = p + 3
|
|
431
475
|
const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
|
package/src/util.ts
CHANGED
|
@@ -3,23 +3,11 @@ import { longFromBytesToUnsigned } from './long.ts'
|
|
|
3
3
|
|
|
4
4
|
import type { Offset, VirtualOffset } from './virtualOffset.ts'
|
|
5
5
|
|
|
6
|
-
export interface TagFilter {
|
|
7
|
-
tag: string
|
|
8
|
-
value?: string
|
|
9
|
-
}
|
|
10
|
-
|
|
11
|
-
export interface FilterBy {
|
|
12
|
-
flagInclude?: number
|
|
13
|
-
flagExclude?: number
|
|
14
|
-
tagFilter?: TagFilter
|
|
15
|
-
}
|
|
16
|
-
|
|
17
6
|
export interface BamOpts {
|
|
18
7
|
viewAsPairs?: boolean
|
|
19
8
|
pairAcrossChr?: boolean
|
|
20
9
|
maxInsertSize?: number
|
|
21
10
|
signal?: AbortSignal
|
|
22
|
-
filterBy?: FilterBy
|
|
23
11
|
/**
|
|
24
12
|
* Called as the BGZF blocks covering the query are fetched, with cumulative
|
|
25
13
|
* downloaded bytes and the total to fetch. Reported at block granularity (one
|
|
@@ -249,54 +237,6 @@ export function concatUint8Array(args: Uint8Array[]) {
|
|
|
249
237
|
return mergedArray
|
|
250
238
|
}
|
|
251
239
|
|
|
252
|
-
export function filterReadFlag(
|
|
253
|
-
flags: number,
|
|
254
|
-
flagInclude: number,
|
|
255
|
-
flagExclude: number,
|
|
256
|
-
) {
|
|
257
|
-
return (flags & flagInclude) !== flagInclude || (flags & flagExclude) !== 0
|
|
258
|
-
}
|
|
259
|
-
|
|
260
|
-
export function filterTagValue(readVal: unknown, filterVal?: string) {
|
|
261
|
-
return filterVal === '*'
|
|
262
|
-
? readVal === undefined
|
|
263
|
-
: `${readVal}` !== `${filterVal}`
|
|
264
|
-
}
|
|
265
|
-
|
|
266
|
-
interface Filterable {
|
|
267
|
-
flags: number
|
|
268
|
-
tags: Record<string, unknown>
|
|
269
|
-
// BamRecord decodes one tag by walking the tag block, without building the
|
|
270
|
-
// whole tags object. Optional because a custom recordClass need not have it.
|
|
271
|
-
getTag?(tag: string): unknown
|
|
272
|
-
}
|
|
273
|
-
|
|
274
|
-
// Read a single tag, preferring the targeted accessor. Reaching for `tags`
|
|
275
|
-
// instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
|
|
276
|
-
// ~10 per read) just to test one, which measured 2.6x the cost of getTag.
|
|
277
|
-
function readTag(record: Filterable, tag: string) {
|
|
278
|
-
return record.getTag ? record.getTag(tag) : record.tags[tag]
|
|
279
|
-
}
|
|
280
|
-
|
|
281
|
-
// Apply flagInclude/flagExclude/tagFilter to a list of records.
|
|
282
|
-
export function applyFilters<T extends Filterable>(
|
|
283
|
-
records: T[],
|
|
284
|
-
filterBy: FilterBy,
|
|
285
|
-
): T[] {
|
|
286
|
-
const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
|
|
287
|
-
const out: T[] = []
|
|
288
|
-
for (let i = 0, l = records.length; i < l; i++) {
|
|
289
|
-
const r = records[i]!
|
|
290
|
-
if (
|
|
291
|
-
!filterReadFlag(r.flags, flagInclude, flagExclude) &&
|
|
292
|
-
!(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
|
|
293
|
-
) {
|
|
294
|
-
out.push(r)
|
|
295
|
-
}
|
|
296
|
-
}
|
|
297
|
-
return out
|
|
298
|
-
}
|
|
299
|
-
|
|
300
240
|
interface Positioned {
|
|
301
241
|
ref_id: number
|
|
302
242
|
start: number
|