@gmod/bam 7.4.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/record.ts CHANGED
@@ -48,6 +48,45 @@ const ASCII_CIGAR_CODES = [
48
48
 
49
49
  const textDecoder = new TextDecoder()
50
50
 
51
+ // Interned two-char tag names keyed on the packed byte pair. Worth more than
52
+ // the saved String.fromCharCode: the tags object is null-prototype and so in
53
+ // dictionary mode, and handing it a string V8 has already hashed halves the
54
+ // per-tag insert cost.
55
+ const TAG_NAMES = new Map<number, string>()
56
+ function tagName(b0: number, b1: number) {
57
+ const key = (b0 << 8) | b1
58
+ let name = TAG_NAMES.get(key)
59
+ if (name === undefined) {
60
+ name = String.fromCharCode(b0, b1)
61
+ TAG_NAMES.set(key, name)
62
+ }
63
+ return name
64
+ }
65
+
66
+ // TextDecoder carries ~0.35us of fixed setup per call, which dominates for the
67
+ // short values Z/H tags usually hold (MD, RG, PG, ...): char codes are 4x
68
+ // faster at 8 bytes, 2x at 16, and the two cross over at 32.
69
+ const SHORT_STRING_THRESHOLD = 32
70
+
71
+ // Z/H tag values are spec'd as printable ASCII, but a non-conforming writer
72
+ // would make the char-code path mis-decode UTF-8, so bail to TextDecoder on any
73
+ // high byte.
74
+ function decodeTagString(ba: Uint8Array, start: number, end: number) {
75
+ const len = end - start
76
+ if (len < SHORT_STRING_THRESHOLD) {
77
+ const codes = new Array<number>(len)
78
+ for (let i = 0; i < len; i++) {
79
+ const byte = ba[start + i]!
80
+ if (byte > 0x7f) {
81
+ return textDecoder.decode(ba.subarray(start, end))
82
+ }
83
+ codes[i] = byte
84
+ }
85
+ return String.fromCharCode(...codes)
86
+ }
87
+ return textDecoder.decode(ba.subarray(start, end))
88
+ }
89
+
51
90
  // Bitmask for ops that consume ref: M=0, D=2, N=3, P=6, ==7, X=8
52
91
  // Binary: 0b111001101 = 0x1CD
53
92
  const CIGAR_CONSUMES_REF_MASK = 0x1cd
@@ -252,9 +291,7 @@ function decodeTagValue(
252
291
  return dataView.getFloat32(p, true)
253
292
  case 0x5a: // 'Z'
254
293
  case 0x48: // 'H'
255
- return raw
256
- ? ba.subarray(p, end - 1)
257
- : textDecoder.decode(ba.subarray(p, end - 1))
294
+ return raw ? ba.subarray(p, end - 1) : decodeTagString(ba, p, end - 1)
258
295
  default: {
259
296
  // 'B'
260
297
  const Btype = ba[p]!
@@ -358,6 +395,13 @@ export default class BamRecord {
358
395
  }
359
396
 
360
397
  // batch fromCharCode: fastest for typical name lengths (see benchmarks/string-building.bench.ts)
398
+ //
399
+ // Deliberately NOT memoized, unlike end/tags/length_on_ref. Consumers read a
400
+ // read name about once — jbrowse-components' buildBaseFeatureData copies it
401
+ // straight into its own FeatureData — so a cache would pay a field slot on
402
+ // every record (+180KB per 22k-record chunk) to save zero decodes, and would
403
+ // pin every name string for as long as the chunk stays cached (+520KB) where
404
+ // today it dies with the consumer's copy.
361
405
  get name() {
362
406
  const len = this.read_name_length - 1
363
407
  const start = this.b0
@@ -425,7 +469,7 @@ export default class BamRecord {
425
469
  const tags: Record<string, unknown> = Object.create(null)
426
470
  let p = this.tagsStart
427
471
  while (p < blockEnd) {
428
- const tag = String.fromCharCode(ba[p]!, ba[p + 1]!)
472
+ const tag = tagName(ba[p]!, ba[p + 1]!)
429
473
  const type = ba[p + 2]!
430
474
  const valueStart = p + 3
431
475
  const end = tagValueEnd(ba, this._dataView, type, valueStart, blockEnd)
package/src/util.ts CHANGED
@@ -3,23 +3,11 @@ import { longFromBytesToUnsigned } from './long.ts'
3
3
 
4
4
  import type { Offset, VirtualOffset } from './virtualOffset.ts'
5
5
 
6
- export interface TagFilter {
7
- tag: string
8
- value?: string
9
- }
10
-
11
- export interface FilterBy {
12
- flagInclude?: number
13
- flagExclude?: number
14
- tagFilter?: TagFilter
15
- }
16
-
17
6
  export interface BamOpts {
18
7
  viewAsPairs?: boolean
19
8
  pairAcrossChr?: boolean
20
9
  maxInsertSize?: number
21
10
  signal?: AbortSignal
22
- filterBy?: FilterBy
23
11
  /**
24
12
  * Called as the BGZF blocks covering the query are fetched, with cumulative
25
13
  * downloaded bytes and the total to fetch. Reported at block granularity (one
@@ -249,54 +237,6 @@ export function concatUint8Array(args: Uint8Array[]) {
249
237
  return mergedArray
250
238
  }
251
239
 
252
- export function filterReadFlag(
253
- flags: number,
254
- flagInclude: number,
255
- flagExclude: number,
256
- ) {
257
- return (flags & flagInclude) !== flagInclude || (flags & flagExclude) !== 0
258
- }
259
-
260
- export function filterTagValue(readVal: unknown, filterVal?: string) {
261
- return filterVal === '*'
262
- ? readVal === undefined
263
- : `${readVal}` !== `${filterVal}`
264
- }
265
-
266
- interface Filterable {
267
- flags: number
268
- tags: Record<string, unknown>
269
- // BamRecord decodes one tag by walking the tag block, without building the
270
- // whole tags object. Optional because a custom recordClass need not have it.
271
- getTag?(tag: string): unknown
272
- }
273
-
274
- // Read a single tag, preferring the targeted accessor. Reaching for `tags`
275
- // instead would decode every unrelated tag on the record (NM/AS/ms/de/… — often
276
- // ~10 per read) just to test one, which measured 2.6x the cost of getTag.
277
- function readTag(record: Filterable, tag: string) {
278
- return record.getTag ? record.getTag(tag) : record.tags[tag]
279
- }
280
-
281
- // Apply flagInclude/flagExclude/tagFilter to a list of records.
282
- export function applyFilters<T extends Filterable>(
283
- records: T[],
284
- filterBy: FilterBy,
285
- ): T[] {
286
- const { flagInclude = 0, flagExclude = 0, tagFilter } = filterBy
287
- const out: T[] = []
288
- for (let i = 0, l = records.length; i < l; i++) {
289
- const r = records[i]!
290
- if (
291
- !filterReadFlag(r.flags, flagInclude, flagExclude) &&
292
- !(tagFilter && filterTagValue(readTag(r, tagFilter.tag), tagFilter.value))
293
- ) {
294
- out.push(r)
295
- }
296
- }
297
- return out
298
- }
299
-
300
240
  interface Positioned {
301
241
  ref_id: number
302
242
  start: number