@gmod/tabix 3.8.3 → 3.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -0
- package/dist/csi.js +4 -1
- package/dist/csi.js.map +1 -1
- package/dist/indexFile.js +12 -2
- package/dist/indexFile.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.js +149 -63
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/util.js +1 -0
- package/dist/util.js.map +1 -1
- package/esm/csi.js +4 -1
- package/esm/csi.js.map +1 -1
- package/esm/indexFile.js +12 -2
- package/esm/indexFile.js.map +1 -1
- package/esm/tabixIndexedFile.js +149 -63
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/util.js +1 -0
- package/esm/util.js.map +1 -1
- package/package.json +3 -3
- package/src/csi.ts +4 -1
- package/src/indexFile.ts +13 -2
- package/src/tabixIndexedFile.ts +212 -87
- package/src/util.ts +1 -0
package/src/tabixIndexedFile.ts
CHANGED
|
@@ -17,6 +17,8 @@ const TAB = 9
|
|
|
17
17
|
const NEWLINE = 10
|
|
18
18
|
const CARRIAGE_RETURN = 13
|
|
19
19
|
const SEMICOLON = 59
|
|
20
|
+
const LESS_THAN = 60
|
|
21
|
+
const GREATER_THAN = 62
|
|
20
22
|
|
|
21
23
|
// Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
|
|
22
24
|
// is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
|
|
@@ -272,6 +274,111 @@ function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
|
|
|
272
274
|
return val
|
|
273
275
|
}
|
|
274
276
|
|
|
277
|
+
interface GafQuery {
|
|
278
|
+
start: number
|
|
279
|
+
end: number
|
|
280
|
+
pathCol: number
|
|
281
|
+
metaCharCode: number | undefined
|
|
282
|
+
decoder: TextDecoder
|
|
283
|
+
callback: GetLinesCallback
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/**
|
|
287
|
+
* The GAF scan of one chunk. A read spans the lowest to the highest node id on
|
|
288
|
+
* its path, which htslib's tbx_parse1 reads the same way. Returns true once a
|
|
289
|
+
* read starts at or past `end`: records are sorted by their lowest node.
|
|
290
|
+
*/
|
|
291
|
+
function scanGafChunk(
|
|
292
|
+
buffer: Uint8Array,
|
|
293
|
+
cpositions: ArrayLike<number>,
|
|
294
|
+
dpositions: ArrayLike<number>,
|
|
295
|
+
minvDataPosition: number,
|
|
296
|
+
{ start, end, pathCol, metaCharCode, decoder, callback }: GafQuery,
|
|
297
|
+
) {
|
|
298
|
+
let blockStart = 0
|
|
299
|
+
let pos = 0
|
|
300
|
+
while (blockStart < buffer.length) {
|
|
301
|
+
const n = buffer.indexOf(NEWLINE, blockStart)
|
|
302
|
+
if (n === -1) {
|
|
303
|
+
break
|
|
304
|
+
}
|
|
305
|
+
const lineStart = blockStart
|
|
306
|
+
blockStart = n + 1
|
|
307
|
+
|
|
308
|
+
const target = lineStart + minvDataPosition
|
|
309
|
+
while (pos < dpositions.length && target >= dpositions[pos]!) {
|
|
310
|
+
pos++
|
|
311
|
+
}
|
|
312
|
+
if (buffer[lineStart] === metaCharCode) {
|
|
313
|
+
continue
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
let pathStart = lineStart
|
|
317
|
+
for (let i = 1; i < pathCol; i++) {
|
|
318
|
+
const t = buffer.indexOf(TAB, pathStart)
|
|
319
|
+
if (t === -1 || t >= n) {
|
|
320
|
+
pathStart = n
|
|
321
|
+
break
|
|
322
|
+
}
|
|
323
|
+
pathStart = t + 1
|
|
324
|
+
}
|
|
325
|
+
// A path without a leading step is a stable sequence name (or `*`), which
|
|
326
|
+
// carries no node ids. htslib parses one anyway and indexes garbage
|
|
327
|
+
// (`GRCh38#0#chr1` as nodes 0..38), so skip it rather than match that.
|
|
328
|
+
const first = buffer[pathStart]
|
|
329
|
+
if (first !== GREATER_THAN && first !== LESS_THAN) {
|
|
330
|
+
continue
|
|
331
|
+
}
|
|
332
|
+
let pathEnd = buffer.indexOf(TAB, pathStart)
|
|
333
|
+
if (pathEnd === -1 || pathEnd > n) {
|
|
334
|
+
pathEnd = n
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
// like htslib, skip one byte (the orientation) before each node id
|
|
338
|
+
let min = Infinity
|
|
339
|
+
let max = -1
|
|
340
|
+
let i = pathStart
|
|
341
|
+
while (i < pathEnd) {
|
|
342
|
+
let id = 0
|
|
343
|
+
let j = i + 1
|
|
344
|
+
for (; j < pathEnd; j++) {
|
|
345
|
+
const digit = buffer[j]! - 48
|
|
346
|
+
if (digit < 0 || digit > 9) {
|
|
347
|
+
break
|
|
348
|
+
}
|
|
349
|
+
id = id * 10 + digit
|
|
350
|
+
}
|
|
351
|
+
if (id < min) {
|
|
352
|
+
min = id
|
|
353
|
+
}
|
|
354
|
+
if (id > max) {
|
|
355
|
+
max = id
|
|
356
|
+
}
|
|
357
|
+
i = j
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
if (min >= end) {
|
|
361
|
+
return true
|
|
362
|
+
}
|
|
363
|
+
if (max >= start) {
|
|
364
|
+
const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
|
|
365
|
+
callback(
|
|
366
|
+
decoder.decode(buffer.subarray(lineStart, lineEnd)),
|
|
367
|
+
calculateFileOffset(
|
|
368
|
+
cpositions,
|
|
369
|
+
dpositions,
|
|
370
|
+
pos,
|
|
371
|
+
lineStart,
|
|
372
|
+
minvDataPosition,
|
|
373
|
+
),
|
|
374
|
+
min,
|
|
375
|
+
max + 1,
|
|
376
|
+
)
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
return false
|
|
380
|
+
}
|
|
381
|
+
|
|
275
382
|
/**
|
|
276
383
|
* Reads Tabix-indexed files (bgzipped), supporting both .tbi and .csi index formats.
|
|
277
384
|
*/
|
|
@@ -506,6 +613,10 @@ export default class TabixIndexedFile {
|
|
|
506
613
|
// tabs[N] holds the byte offset of the N-th tab on the current line; with
|
|
507
614
|
// the sentinel tabs[0] = blockStart - 1, column N spans tabs[N-1]+1..tabs[N]
|
|
508
615
|
const tabs = new Int32Array(maxColumn + 1)
|
|
616
|
+
const gaf: GafQuery | undefined =
|
|
617
|
+
metadata.format === 'GAF'
|
|
618
|
+
? { start, end, pathCol: startCol, metaCharCode, decoder, callback }
|
|
619
|
+
: undefined
|
|
509
620
|
|
|
510
621
|
let totalBytes = 0
|
|
511
622
|
for (const c of chunks) {
|
|
@@ -552,104 +663,118 @@ export default class TabixIndexedFile {
|
|
|
552
663
|
onProgress?.(downloadedBytes, totalBytes)
|
|
553
664
|
const minvDataPosition = c.minv.dataPosition
|
|
554
665
|
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
if (n === -1) {
|
|
561
|
-
break
|
|
666
|
+
if (gaf) {
|
|
667
|
+
if (
|
|
668
|
+
scanGafChunk(buffer, cpositions, dpositions, minvDataPosition, gaf)
|
|
669
|
+
) {
|
|
670
|
+
return
|
|
562
671
|
}
|
|
672
|
+
} else {
|
|
673
|
+
let blockStart = 0
|
|
674
|
+
let pos = 0
|
|
563
675
|
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
676
|
+
while (blockStart < buffer.length) {
|
|
677
|
+
const n = buffer.indexOf(NEWLINE, blockStart)
|
|
678
|
+
if (n === -1) {
|
|
679
|
+
break
|
|
680
|
+
}
|
|
568
681
|
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
}
|
|
682
|
+
const target = blockStart + minvDataPosition
|
|
683
|
+
while (pos < dpositions.length && target >= dpositions[pos]!) {
|
|
684
|
+
pos++
|
|
685
|
+
}
|
|
574
686
|
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
}
|
|
687
|
+
// skip meta lines
|
|
688
|
+
if (
|
|
689
|
+
metaCharCode !== undefined &&
|
|
690
|
+
buffer[blockStart] === metaCharCode
|
|
691
|
+
) {
|
|
692
|
+
blockStart = n + 1
|
|
693
|
+
continue
|
|
694
|
+
}
|
|
584
695
|
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
let isRefMatch = true
|
|
594
|
-
for (let i = 0; i < refLen; i++) {
|
|
595
|
-
if (buffer[refStart + i] !== regionRefNameBytes[i]) {
|
|
596
|
-
isRefMatch = false
|
|
597
|
-
break
|
|
696
|
+
// find tab positions. Columns past the end of the line all get `n`
|
|
697
|
+
// rather than breaking out, which would leave stale offsets from the
|
|
698
|
+
// previous line in the tail of the array.
|
|
699
|
+
tabs[0] = blockStart - 1
|
|
700
|
+
for (let i = 0; i < maxColumn; i++) {
|
|
701
|
+
const prev = tabs[i]!
|
|
702
|
+
const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
|
|
703
|
+
tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
|
|
598
704
|
}
|
|
599
|
-
}
|
|
600
|
-
if (!isRefMatch) {
|
|
601
|
-
blockStart = n + 1
|
|
602
|
-
continue
|
|
603
|
-
}
|
|
604
705
|
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
706
|
+
// compare ref name bytes directly
|
|
707
|
+
const refStart = tabs[refCol - 1]! + 1
|
|
708
|
+
const refEnd = tabs[refCol]!
|
|
709
|
+
const refLen = refEnd - refStart
|
|
710
|
+
if (refLen !== regionRefNameBytes.length) {
|
|
711
|
+
blockStart = n + 1
|
|
712
|
+
continue
|
|
713
|
+
}
|
|
714
|
+
let isRefMatch = true
|
|
715
|
+
for (let i = 0; i < refLen; i++) {
|
|
716
|
+
if (buffer[refStart + i] !== regionRefNameBytes[i]) {
|
|
717
|
+
isRefMatch = false
|
|
718
|
+
break
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
if (!isRefMatch) {
|
|
722
|
+
blockStart = n + 1
|
|
723
|
+
continue
|
|
724
|
+
}
|
|
609
725
|
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
726
|
+
// parse start coordinate
|
|
727
|
+
const startCoordinate =
|
|
728
|
+
parseIntFromBytes(
|
|
729
|
+
buffer,
|
|
730
|
+
tabs[startCol - 1]! + 1,
|
|
731
|
+
tabs[startCol]!,
|
|
732
|
+
) + coordinateOffset
|
|
613
733
|
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
endCoordinate = startCoordinate + 1
|
|
618
|
-
} else if (isVCF) {
|
|
619
|
-
endCoordinate = getVcfEnd(
|
|
620
|
-
buffer,
|
|
621
|
-
startCoordinate,
|
|
622
|
-
tabs[3]! + 1,
|
|
623
|
-
tabs[4]!,
|
|
624
|
-
tabs[endCol - 1]! + 1,
|
|
625
|
-
tabs[endCol]!,
|
|
626
|
-
)
|
|
627
|
-
} else {
|
|
628
|
-
endCoordinate = parseIntFromBytes(
|
|
629
|
-
buffer,
|
|
630
|
-
tabs[endCol - 1]! + 1,
|
|
631
|
-
tabs[endCol]!,
|
|
632
|
-
)
|
|
633
|
-
}
|
|
734
|
+
if (startCoordinate >= end) {
|
|
735
|
+
return
|
|
736
|
+
}
|
|
634
737
|
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
)
|
|
648
|
-
|
|
649
|
-
endCoordinate
|
|
650
|
-
|
|
738
|
+
// parse end coordinate
|
|
739
|
+
let endCoordinate: number
|
|
740
|
+
if (endCol === 0 || endCol === startCol) {
|
|
741
|
+
endCoordinate = startCoordinate + 1
|
|
742
|
+
} else if (isVCF) {
|
|
743
|
+
endCoordinate = getVcfEnd(
|
|
744
|
+
buffer,
|
|
745
|
+
startCoordinate,
|
|
746
|
+
tabs[3]! + 1,
|
|
747
|
+
tabs[4]!,
|
|
748
|
+
tabs[endCol - 1]! + 1,
|
|
749
|
+
tabs[endCol]!,
|
|
750
|
+
)
|
|
751
|
+
} else {
|
|
752
|
+
endCoordinate = parseIntFromBytes(
|
|
753
|
+
buffer,
|
|
754
|
+
tabs[endCol - 1]! + 1,
|
|
755
|
+
tabs[endCol]!,
|
|
756
|
+
)
|
|
757
|
+
}
|
|
758
|
+
|
|
759
|
+
if (endCoordinate > start) {
|
|
760
|
+
// trim a CRLF terminator, matching htslib's line reader
|
|
761
|
+
const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
|
|
762
|
+
const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
|
|
763
|
+
callback(
|
|
764
|
+
line,
|
|
765
|
+
calculateFileOffset(
|
|
766
|
+
cpositions,
|
|
767
|
+
dpositions,
|
|
768
|
+
pos,
|
|
769
|
+
blockStart,
|
|
770
|
+
minvDataPosition,
|
|
771
|
+
),
|
|
772
|
+
startCoordinate,
|
|
773
|
+
endCoordinate,
|
|
774
|
+
)
|
|
775
|
+
}
|
|
776
|
+
blockStart = n + 1
|
|
651
777
|
}
|
|
652
|
-
blockStart = n + 1
|
|
653
778
|
}
|
|
654
779
|
|
|
655
780
|
// every line in this chunk was still inside the query, so the next chunk
|