@gmod/tabix 3.8.3 → 3.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,6 +17,8 @@ const TAB = 9
17
17
  const NEWLINE = 10
18
18
  const CARRIAGE_RETURN = 13
19
19
  const SEMICOLON = 59
20
+ const LESS_THAN = 60
21
+ const GREATER_THAN = 62
20
22
 
21
23
  // Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
22
24
  // is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
@@ -272,6 +274,111 @@ function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
272
274
  return val
273
275
  }
274
276
 
277
+ interface GafQuery {
278
+ start: number
279
+ end: number
280
+ pathCol: number
281
+ metaCharCode: number | undefined
282
+ decoder: TextDecoder
283
+ callback: GetLinesCallback
284
+ }
285
+
286
+ /**
287
+ * The GAF scan of one chunk. A read spans the lowest to the highest node id on
288
+ * its path, which htslib's tbx_parse1 reads the same way. Returns true once a
289
+ * read starts at or past `end`: records are sorted by their lowest node.
290
+ */
291
+ function scanGafChunk(
292
+ buffer: Uint8Array,
293
+ cpositions: ArrayLike<number>,
294
+ dpositions: ArrayLike<number>,
295
+ minvDataPosition: number,
296
+ { start, end, pathCol, metaCharCode, decoder, callback }: GafQuery,
297
+ ) {
298
+ let blockStart = 0
299
+ let pos = 0
300
+ while (blockStart < buffer.length) {
301
+ const n = buffer.indexOf(NEWLINE, blockStart)
302
+ if (n === -1) {
303
+ break
304
+ }
305
+ const lineStart = blockStart
306
+ blockStart = n + 1
307
+
308
+ const target = lineStart + minvDataPosition
309
+ while (pos < dpositions.length && target >= dpositions[pos]!) {
310
+ pos++
311
+ }
312
+ if (buffer[lineStart] === metaCharCode) {
313
+ continue
314
+ }
315
+
316
+ let pathStart = lineStart
317
+ for (let i = 1; i < pathCol; i++) {
318
+ const t = buffer.indexOf(TAB, pathStart)
319
+ if (t === -1 || t >= n) {
320
+ pathStart = n
321
+ break
322
+ }
323
+ pathStart = t + 1
324
+ }
325
+ // A path without a leading step is a stable sequence name (or `*`), which
326
+ // carries no node ids. htslib parses one anyway and indexes garbage
327
+ // (`GRCh38#0#chr1` as nodes 0..38), so skip it rather than match that.
328
+ const first = buffer[pathStart]
329
+ if (first !== GREATER_THAN && first !== LESS_THAN) {
330
+ continue
331
+ }
332
+ let pathEnd = buffer.indexOf(TAB, pathStart)
333
+ if (pathEnd === -1 || pathEnd > n) {
334
+ pathEnd = n
335
+ }
336
+
337
+ // like htslib, skip one byte (the orientation) before each node id
338
+ let min = Infinity
339
+ let max = -1
340
+ let i = pathStart
341
+ while (i < pathEnd) {
342
+ let id = 0
343
+ let j = i + 1
344
+ for (; j < pathEnd; j++) {
345
+ const digit = buffer[j]! - 48
346
+ if (digit < 0 || digit > 9) {
347
+ break
348
+ }
349
+ id = id * 10 + digit
350
+ }
351
+ if (id < min) {
352
+ min = id
353
+ }
354
+ if (id > max) {
355
+ max = id
356
+ }
357
+ i = j
358
+ }
359
+
360
+ if (min >= end) {
361
+ return true
362
+ }
363
+ if (max >= start) {
364
+ const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
365
+ callback(
366
+ decoder.decode(buffer.subarray(lineStart, lineEnd)),
367
+ calculateFileOffset(
368
+ cpositions,
369
+ dpositions,
370
+ pos,
371
+ lineStart,
372
+ minvDataPosition,
373
+ ),
374
+ min,
375
+ max + 1,
376
+ )
377
+ }
378
+ }
379
+ return false
380
+ }
381
+
275
382
  /**
276
383
  * Reads Tabix-indexed files (bgzipped), supporting both .tbi and .csi index formats.
277
384
  */
@@ -506,6 +613,10 @@ export default class TabixIndexedFile {
506
613
  // tabs[N] holds the byte offset of the N-th tab on the current line; with
507
614
  // the sentinel tabs[0] = blockStart - 1, column N spans tabs[N-1]+1..tabs[N]
508
615
  const tabs = new Int32Array(maxColumn + 1)
616
+ const gaf: GafQuery | undefined =
617
+ metadata.format === 'GAF'
618
+ ? { start, end, pathCol: startCol, metaCharCode, decoder, callback }
619
+ : undefined
509
620
 
510
621
  let totalBytes = 0
511
622
  for (const c of chunks) {
@@ -552,104 +663,118 @@ export default class TabixIndexedFile {
552
663
  onProgress?.(downloadedBytes, totalBytes)
553
664
  const minvDataPosition = c.minv.dataPosition
554
665
 
555
- let blockStart = 0
556
- let pos = 0
557
-
558
- while (blockStart < buffer.length) {
559
- const n = buffer.indexOf(NEWLINE, blockStart)
560
- if (n === -1) {
561
- break
666
+ if (gaf) {
667
+ if (
668
+ scanGafChunk(buffer, cpositions, dpositions, minvDataPosition, gaf)
669
+ ) {
670
+ return
562
671
  }
672
+ } else {
673
+ let blockStart = 0
674
+ let pos = 0
563
675
 
564
- const target = blockStart + minvDataPosition
565
- while (pos < dpositions.length && target >= dpositions[pos]!) {
566
- pos++
567
- }
676
+ while (blockStart < buffer.length) {
677
+ const n = buffer.indexOf(NEWLINE, blockStart)
678
+ if (n === -1) {
679
+ break
680
+ }
568
681
 
569
- // skip meta lines
570
- if (metaCharCode !== undefined && buffer[blockStart] === metaCharCode) {
571
- blockStart = n + 1
572
- continue
573
- }
682
+ const target = blockStart + minvDataPosition
683
+ while (pos < dpositions.length && target >= dpositions[pos]!) {
684
+ pos++
685
+ }
574
686
 
575
- // find tab positions. Columns past the end of the line all get `n`
576
- // rather than breaking out, which would leave stale offsets from the
577
- // previous line in the tail of the array.
578
- tabs[0] = blockStart - 1
579
- for (let i = 0; i < maxColumn; i++) {
580
- const prev = tabs[i]!
581
- const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
582
- tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
583
- }
687
+ // skip meta lines
688
+ if (
689
+ metaCharCode !== undefined &&
690
+ buffer[blockStart] === metaCharCode
691
+ ) {
692
+ blockStart = n + 1
693
+ continue
694
+ }
584
695
 
585
- // compare ref name bytes directly
586
- const refStart = tabs[refCol - 1]! + 1
587
- const refEnd = tabs[refCol]!
588
- const refLen = refEnd - refStart
589
- if (refLen !== regionRefNameBytes.length) {
590
- blockStart = n + 1
591
- continue
592
- }
593
- let isRefMatch = true
594
- for (let i = 0; i < refLen; i++) {
595
- if (buffer[refStart + i] !== regionRefNameBytes[i]) {
596
- isRefMatch = false
597
- break
696
+ // find tab positions. Columns past the end of the line all get `n`
697
+ // rather than breaking out, which would leave stale offsets from the
698
+ // previous line in the tail of the array.
699
+ tabs[0] = blockStart - 1
700
+ for (let i = 0; i < maxColumn; i++) {
701
+ const prev = tabs[i]!
702
+ const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
703
+ tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
598
704
  }
599
- }
600
- if (!isRefMatch) {
601
- blockStart = n + 1
602
- continue
603
- }
604
705
 
605
- // parse start coordinate
606
- const startCoordinate =
607
- parseIntFromBytes(buffer, tabs[startCol - 1]! + 1, tabs[startCol]!) +
608
- coordinateOffset
706
+ // compare ref name bytes directly
707
+ const refStart = tabs[refCol - 1]! + 1
708
+ const refEnd = tabs[refCol]!
709
+ const refLen = refEnd - refStart
710
+ if (refLen !== regionRefNameBytes.length) {
711
+ blockStart = n + 1
712
+ continue
713
+ }
714
+ let isRefMatch = true
715
+ for (let i = 0; i < refLen; i++) {
716
+ if (buffer[refStart + i] !== regionRefNameBytes[i]) {
717
+ isRefMatch = false
718
+ break
719
+ }
720
+ }
721
+ if (!isRefMatch) {
722
+ blockStart = n + 1
723
+ continue
724
+ }
609
725
 
610
- if (startCoordinate >= end) {
611
- return
612
- }
726
+ // parse start coordinate
727
+ const startCoordinate =
728
+ parseIntFromBytes(
729
+ buffer,
730
+ tabs[startCol - 1]! + 1,
731
+ tabs[startCol]!,
732
+ ) + coordinateOffset
613
733
 
614
- // parse end coordinate
615
- let endCoordinate: number
616
- if (endCol === 0 || endCol === startCol) {
617
- endCoordinate = startCoordinate + 1
618
- } else if (isVCF) {
619
- endCoordinate = getVcfEnd(
620
- buffer,
621
- startCoordinate,
622
- tabs[3]! + 1,
623
- tabs[4]!,
624
- tabs[endCol - 1]! + 1,
625
- tabs[endCol]!,
626
- )
627
- } else {
628
- endCoordinate = parseIntFromBytes(
629
- buffer,
630
- tabs[endCol - 1]! + 1,
631
- tabs[endCol]!,
632
- )
633
- }
734
+ if (startCoordinate >= end) {
735
+ return
736
+ }
634
737
 
635
- if (endCoordinate > start) {
636
- // trim a CRLF terminator, matching htslib's line reader
637
- const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
638
- const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
639
- callback(
640
- line,
641
- calculateFileOffset(
642
- cpositions,
643
- dpositions,
644
- pos,
645
- blockStart,
646
- minvDataPosition,
647
- ),
648
- startCoordinate,
649
- endCoordinate,
650
- )
738
+ // parse end coordinate
739
+ let endCoordinate: number
740
+ if (endCol === 0 || endCol === startCol) {
741
+ endCoordinate = startCoordinate + 1
742
+ } else if (isVCF) {
743
+ endCoordinate = getVcfEnd(
744
+ buffer,
745
+ startCoordinate,
746
+ tabs[3]! + 1,
747
+ tabs[4]!,
748
+ tabs[endCol - 1]! + 1,
749
+ tabs[endCol]!,
750
+ )
751
+ } else {
752
+ endCoordinate = parseIntFromBytes(
753
+ buffer,
754
+ tabs[endCol - 1]! + 1,
755
+ tabs[endCol]!,
756
+ )
757
+ }
758
+
759
+ if (endCoordinate > start) {
760
+ // trim a CRLF terminator, matching htslib's line reader
761
+ const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
762
+ const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
763
+ callback(
764
+ line,
765
+ calculateFileOffset(
766
+ cpositions,
767
+ dpositions,
768
+ pos,
769
+ blockStart,
770
+ minvDataPosition,
771
+ ),
772
+ startCoordinate,
773
+ endCoordinate,
774
+ )
775
+ }
776
+ blockStart = n + 1
651
777
  }
652
- blockStart = n + 1
653
778
  }
654
779
 
655
780
  // every line in this chunk was still inside the query, so the next chunk
package/src/util.ts CHANGED
@@ -227,6 +227,7 @@ const tabixFormats: Record<number, string> = {
227
227
  0: 'generic',
228
228
  1: 'SAM',
229
229
  2: 'VCF',
230
+ 3: 'GAF',
230
231
  }
231
232
 
232
233
  export function parseAuxData(bytes: Uint8Array, offset: number) {