@gmod/tabix 3.8.2 → 3.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,6 +17,8 @@ const TAB = 9
17
17
  const NEWLINE = 10
18
18
  const CARRIAGE_RETURN = 13
19
19
  const SEMICOLON = 59
20
+ const LESS_THAN = 60
21
+ const GREATER_THAN = 62
20
22
 
21
23
  // Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
22
24
  // is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
@@ -272,6 +274,111 @@ function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
272
274
  return val
273
275
  }
274
276
 
277
+ interface GafQuery {
278
+ start: number
279
+ end: number
280
+ pathCol: number
281
+ metaCharCode: number | undefined
282
+ decoder: TextDecoder
283
+ callback: GetLinesCallback
284
+ }
285
+
286
+ /**
287
+ * The GAF scan of one chunk. A read spans the lowest to the highest node id on
288
+ * its path, which htslib's tbx_parse1 reads the same way. Returns true once a
289
+ * read starts at or past `end`: records are sorted by their lowest node.
290
+ */
291
+ function scanGafChunk(
292
+ buffer: Uint8Array,
293
+ cpositions: ArrayLike<number>,
294
+ dpositions: ArrayLike<number>,
295
+ minvDataPosition: number,
296
+ { start, end, pathCol, metaCharCode, decoder, callback }: GafQuery,
297
+ ) {
298
+ let blockStart = 0
299
+ let pos = 0
300
+ while (blockStart < buffer.length) {
301
+ const n = buffer.indexOf(NEWLINE, blockStart)
302
+ if (n === -1) {
303
+ break
304
+ }
305
+ const lineStart = blockStart
306
+ blockStart = n + 1
307
+
308
+ const target = lineStart + minvDataPosition
309
+ while (pos < dpositions.length && target >= dpositions[pos]!) {
310
+ pos++
311
+ }
312
+ if (buffer[lineStart] === metaCharCode) {
313
+ continue
314
+ }
315
+
316
+ let pathStart = lineStart
317
+ for (let i = 1; i < pathCol; i++) {
318
+ const t = buffer.indexOf(TAB, pathStart)
319
+ if (t === -1 || t >= n) {
320
+ pathStart = n
321
+ break
322
+ }
323
+ pathStart = t + 1
324
+ }
325
+ // A path without a leading step is a stable sequence name (or `*`), which
326
+ // carries no node ids. htslib parses one anyway and indexes garbage
327
+ // (`GRCh38#0#chr1` as nodes 0..38), so skip it rather than match that.
328
+ const first = buffer[pathStart]
329
+ if (first !== GREATER_THAN && first !== LESS_THAN) {
330
+ continue
331
+ }
332
+ let pathEnd = buffer.indexOf(TAB, pathStart)
333
+ if (pathEnd === -1 || pathEnd > n) {
334
+ pathEnd = n
335
+ }
336
+
337
+ // like htslib, skip one byte (the orientation) before each node id
338
+ let min = Infinity
339
+ let max = -1
340
+ let i = pathStart
341
+ while (i < pathEnd) {
342
+ let id = 0
343
+ let j = i + 1
344
+ for (; j < pathEnd; j++) {
345
+ const digit = buffer[j]! - 48
346
+ if (digit < 0 || digit > 9) {
347
+ break
348
+ }
349
+ id = id * 10 + digit
350
+ }
351
+ if (id < min) {
352
+ min = id
353
+ }
354
+ if (id > max) {
355
+ max = id
356
+ }
357
+ i = j
358
+ }
359
+
360
+ if (min >= end) {
361
+ return true
362
+ }
363
+ if (max >= start) {
364
+ const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
365
+ callback(
366
+ decoder.decode(buffer.subarray(lineStart, lineEnd)),
367
+ calculateFileOffset(
368
+ cpositions,
369
+ dpositions,
370
+ pos,
371
+ lineStart,
372
+ minvDataPosition,
373
+ ),
374
+ min,
375
+ max + 1,
376
+ )
377
+ }
378
+ }
379
+ return false
380
+ }
381
+
275
382
  /**
276
383
  * Reads Tabix-indexed files (bgzipped), supporting both .tbi and .csi index formats.
277
384
  */
@@ -506,6 +613,10 @@ export default class TabixIndexedFile {
506
613
  // tabs[N] holds the byte offset of the N-th tab on the current line; with
507
614
  // the sentinel tabs[0] = blockStart - 1, column N spans tabs[N-1]+1..tabs[N]
508
615
  const tabs = new Int32Array(maxColumn + 1)
616
+ const gaf: GafQuery | undefined =
617
+ metadata.format === 'GAF'
618
+ ? { start, end, pathCol: startCol, metaCharCode, decoder, callback }
619
+ : undefined
509
620
 
510
621
  let totalBytes = 0
511
622
  for (const c of chunks) {
@@ -521,8 +632,9 @@ export default class TabixIndexedFile {
521
632
  //
522
633
  // A fixed window would be wrong, though: blocksForRange offers a chunk per
523
634
  // overlapping bin across every level, and on a sparse file the early return
524
- // below stops the scan inside the first one, leaving the rest untouched
525
- // (chr22_nanopore_subset offers 7 chunks and reads 1). Prefetching those
635
+ // below can stop the scan inside the first one, leaving the rest untouched.
636
+ // max_off drops only the chunks it can prove lie past the query, so the
637
+ // scan can still end before the last one it is offered. Prefetching those
526
638
  // would multiply the bytes such a query fetches for no gain.
527
639
  //
528
640
  // Finishing a chunk without hitting the early return proves the next chunk
@@ -551,104 +663,118 @@ export default class TabixIndexedFile {
551
663
  onProgress?.(downloadedBytes, totalBytes)
552
664
  const minvDataPosition = c.minv.dataPosition
553
665
 
554
- let blockStart = 0
555
- let pos = 0
556
-
557
- while (blockStart < buffer.length) {
558
- const n = buffer.indexOf(NEWLINE, blockStart)
559
- if (n === -1) {
560
- break
666
+ if (gaf) {
667
+ if (
668
+ scanGafChunk(buffer, cpositions, dpositions, minvDataPosition, gaf)
669
+ ) {
670
+ return
561
671
  }
672
+ } else {
673
+ let blockStart = 0
674
+ let pos = 0
562
675
 
563
- const target = blockStart + minvDataPosition
564
- while (pos < dpositions.length && target >= dpositions[pos]!) {
565
- pos++
566
- }
676
+ while (blockStart < buffer.length) {
677
+ const n = buffer.indexOf(NEWLINE, blockStart)
678
+ if (n === -1) {
679
+ break
680
+ }
567
681
 
568
- // skip meta lines
569
- if (metaCharCode !== undefined && buffer[blockStart] === metaCharCode) {
570
- blockStart = n + 1
571
- continue
572
- }
682
+ const target = blockStart + minvDataPosition
683
+ while (pos < dpositions.length && target >= dpositions[pos]!) {
684
+ pos++
685
+ }
573
686
 
574
- // find tab positions. Columns past the end of the line all get `n`
575
- // rather than breaking out, which would leave stale offsets from the
576
- // previous line in the tail of the array.
577
- tabs[0] = blockStart - 1
578
- for (let i = 0; i < maxColumn; i++) {
579
- const prev = tabs[i]!
580
- const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
581
- tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
582
- }
687
+ // skip meta lines
688
+ if (
689
+ metaCharCode !== undefined &&
690
+ buffer[blockStart] === metaCharCode
691
+ ) {
692
+ blockStart = n + 1
693
+ continue
694
+ }
583
695
 
584
- // compare ref name bytes directly
585
- const refStart = tabs[refCol - 1]! + 1
586
- const refEnd = tabs[refCol]!
587
- const refLen = refEnd - refStart
588
- if (refLen !== regionRefNameBytes.length) {
589
- blockStart = n + 1
590
- continue
591
- }
592
- let isRefMatch = true
593
- for (let i = 0; i < refLen; i++) {
594
- if (buffer[refStart + i] !== regionRefNameBytes[i]) {
595
- isRefMatch = false
596
- break
696
+ // find tab positions. Columns past the end of the line all get `n`
697
+ // rather than breaking out, which would leave stale offsets from the
698
+ // previous line in the tail of the array.
699
+ tabs[0] = blockStart - 1
700
+ for (let i = 0; i < maxColumn; i++) {
701
+ const prev = tabs[i]!
702
+ const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
703
+ tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
597
704
  }
598
- }
599
- if (!isRefMatch) {
600
- blockStart = n + 1
601
- continue
602
- }
603
705
 
604
- // parse start coordinate
605
- const startCoordinate =
606
- parseIntFromBytes(buffer, tabs[startCol - 1]! + 1, tabs[startCol]!) +
607
- coordinateOffset
706
+ // compare ref name bytes directly
707
+ const refStart = tabs[refCol - 1]! + 1
708
+ const refEnd = tabs[refCol]!
709
+ const refLen = refEnd - refStart
710
+ if (refLen !== regionRefNameBytes.length) {
711
+ blockStart = n + 1
712
+ continue
713
+ }
714
+ let isRefMatch = true
715
+ for (let i = 0; i < refLen; i++) {
716
+ if (buffer[refStart + i] !== regionRefNameBytes[i]) {
717
+ isRefMatch = false
718
+ break
719
+ }
720
+ }
721
+ if (!isRefMatch) {
722
+ blockStart = n + 1
723
+ continue
724
+ }
608
725
 
609
- if (startCoordinate >= end) {
610
- return
611
- }
726
+ // parse start coordinate
727
+ const startCoordinate =
728
+ parseIntFromBytes(
729
+ buffer,
730
+ tabs[startCol - 1]! + 1,
731
+ tabs[startCol]!,
732
+ ) + coordinateOffset
612
733
 
613
- // parse end coordinate
614
- let endCoordinate: number
615
- if (endCol === 0 || endCol === startCol) {
616
- endCoordinate = startCoordinate + 1
617
- } else if (isVCF) {
618
- endCoordinate = getVcfEnd(
619
- buffer,
620
- startCoordinate,
621
- tabs[3]! + 1,
622
- tabs[4]!,
623
- tabs[endCol - 1]! + 1,
624
- tabs[endCol]!,
625
- )
626
- } else {
627
- endCoordinate = parseIntFromBytes(
628
- buffer,
629
- tabs[endCol - 1]! + 1,
630
- tabs[endCol]!,
631
- )
632
- }
734
+ if (startCoordinate >= end) {
735
+ return
736
+ }
633
737
 
634
- if (endCoordinate > start) {
635
- // trim a CRLF terminator, matching htslib's line reader
636
- const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
637
- const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
638
- callback(
639
- line,
640
- calculateFileOffset(
641
- cpositions,
642
- dpositions,
643
- pos,
644
- blockStart,
645
- minvDataPosition,
646
- ),
647
- startCoordinate,
648
- endCoordinate,
649
- )
738
+ // parse end coordinate
739
+ let endCoordinate: number
740
+ if (endCol === 0 || endCol === startCol) {
741
+ endCoordinate = startCoordinate + 1
742
+ } else if (isVCF) {
743
+ endCoordinate = getVcfEnd(
744
+ buffer,
745
+ startCoordinate,
746
+ tabs[3]! + 1,
747
+ tabs[4]!,
748
+ tabs[endCol - 1]! + 1,
749
+ tabs[endCol]!,
750
+ )
751
+ } else {
752
+ endCoordinate = parseIntFromBytes(
753
+ buffer,
754
+ tabs[endCol - 1]! + 1,
755
+ tabs[endCol]!,
756
+ )
757
+ }
758
+
759
+ if (endCoordinate > start) {
760
+ // trim a CRLF terminator, matching htslib's line reader
761
+ const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
762
+ const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
763
+ callback(
764
+ line,
765
+ calculateFileOffset(
766
+ cpositions,
767
+ dpositions,
768
+ pos,
769
+ blockStart,
770
+ minvDataPosition,
771
+ ),
772
+ startCoordinate,
773
+ endCoordinate,
774
+ )
775
+ }
776
+ blockStart = n + 1
650
777
  }
651
- blockStart = n + 1
652
778
  }
653
779
 
654
780
  // every line in this chunk was still inside the query, so the next chunk
package/src/util.ts CHANGED
@@ -227,6 +227,7 @@ const tabixFormats: Record<number, string> = {
227
227
  0: 'generic',
228
228
  1: 'SAM',
229
229
  2: 'VCF',
230
+ 3: 'GAF',
230
231
  }
231
232
 
232
233
  export function parseAuxData(bytes: Uint8Array, offset: number) {