@gmod/tabix 3.8.2 → 3.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +24 -11
- package/dist/csi.js +5 -2
- package/dist/csi.js.map +1 -1
- package/dist/indexFile.d.ts +16 -0
- package/dist/indexFile.js +68 -3
- package/dist/indexFile.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.js +152 -65
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/util.js +1 -0
- package/dist/util.js.map +1 -1
- package/esm/csi.js +5 -2
- package/esm/csi.js.map +1 -1
- package/esm/indexFile.d.ts +16 -0
- package/esm/indexFile.js +67 -3
- package/esm/indexFile.js.map +1 -1
- package/esm/tabixIndexedFile.js +152 -65
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/util.js +1 -0
- package/esm/util.js.map +1 -1
- package/package.json +2 -2
- package/src/csi.ts +5 -2
- package/src/indexFile.ts +79 -3
- package/src/tabixIndexedFile.ts +215 -89
- package/src/util.ts +1 -0
package/src/tabixIndexedFile.ts
CHANGED
|
@@ -17,6 +17,8 @@ const TAB = 9
|
|
|
17
17
|
const NEWLINE = 10
|
|
18
18
|
const CARRIAGE_RETURN = 13
|
|
19
19
|
const SEMICOLON = 59
|
|
20
|
+
const LESS_THAN = 60
|
|
21
|
+
const GREATER_THAN = 62
|
|
20
22
|
|
|
21
23
|
// Ceiling on how many chunk reads getLines keeps in flight ahead of the one it
|
|
22
24
|
// is parsing. Six is the HTTP/1.1 per-host connection cap browsers enforce, so
|
|
@@ -272,6 +274,111 @@ function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
|
|
|
272
274
|
return val
|
|
273
275
|
}
|
|
274
276
|
|
|
277
|
+
interface GafQuery {
|
|
278
|
+
start: number
|
|
279
|
+
end: number
|
|
280
|
+
pathCol: number
|
|
281
|
+
metaCharCode: number | undefined
|
|
282
|
+
decoder: TextDecoder
|
|
283
|
+
callback: GetLinesCallback
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/**
|
|
287
|
+
* The GAF scan of one chunk. A read spans the lowest to the highest node id on
|
|
288
|
+
* its path, which htslib's tbx_parse1 reads the same way. Returns true once a
|
|
289
|
+
* read starts at or past `end`: records are sorted by their lowest node.
|
|
290
|
+
*/
|
|
291
|
+
function scanGafChunk(
|
|
292
|
+
buffer: Uint8Array,
|
|
293
|
+
cpositions: ArrayLike<number>,
|
|
294
|
+
dpositions: ArrayLike<number>,
|
|
295
|
+
minvDataPosition: number,
|
|
296
|
+
{ start, end, pathCol, metaCharCode, decoder, callback }: GafQuery,
|
|
297
|
+
) {
|
|
298
|
+
let blockStart = 0
|
|
299
|
+
let pos = 0
|
|
300
|
+
while (blockStart < buffer.length) {
|
|
301
|
+
const n = buffer.indexOf(NEWLINE, blockStart)
|
|
302
|
+
if (n === -1) {
|
|
303
|
+
break
|
|
304
|
+
}
|
|
305
|
+
const lineStart = blockStart
|
|
306
|
+
blockStart = n + 1
|
|
307
|
+
|
|
308
|
+
const target = lineStart + minvDataPosition
|
|
309
|
+
while (pos < dpositions.length && target >= dpositions[pos]!) {
|
|
310
|
+
pos++
|
|
311
|
+
}
|
|
312
|
+
if (buffer[lineStart] === metaCharCode) {
|
|
313
|
+
continue
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
let pathStart = lineStart
|
|
317
|
+
for (let i = 1; i < pathCol; i++) {
|
|
318
|
+
const t = buffer.indexOf(TAB, pathStart)
|
|
319
|
+
if (t === -1 || t >= n) {
|
|
320
|
+
pathStart = n
|
|
321
|
+
break
|
|
322
|
+
}
|
|
323
|
+
pathStart = t + 1
|
|
324
|
+
}
|
|
325
|
+
// A path without a leading step is a stable sequence name (or `*`), which
|
|
326
|
+
// carries no node ids. htslib parses one anyway and indexes garbage
|
|
327
|
+
// (`GRCh38#0#chr1` as nodes 0..38), so skip it rather than match that.
|
|
328
|
+
const first = buffer[pathStart]
|
|
329
|
+
if (first !== GREATER_THAN && first !== LESS_THAN) {
|
|
330
|
+
continue
|
|
331
|
+
}
|
|
332
|
+
let pathEnd = buffer.indexOf(TAB, pathStart)
|
|
333
|
+
if (pathEnd === -1 || pathEnd > n) {
|
|
334
|
+
pathEnd = n
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
// like htslib, skip one byte (the orientation) before each node id
|
|
338
|
+
let min = Infinity
|
|
339
|
+
let max = -1
|
|
340
|
+
let i = pathStart
|
|
341
|
+
while (i < pathEnd) {
|
|
342
|
+
let id = 0
|
|
343
|
+
let j = i + 1
|
|
344
|
+
for (; j < pathEnd; j++) {
|
|
345
|
+
const digit = buffer[j]! - 48
|
|
346
|
+
if (digit < 0 || digit > 9) {
|
|
347
|
+
break
|
|
348
|
+
}
|
|
349
|
+
id = id * 10 + digit
|
|
350
|
+
}
|
|
351
|
+
if (id < min) {
|
|
352
|
+
min = id
|
|
353
|
+
}
|
|
354
|
+
if (id > max) {
|
|
355
|
+
max = id
|
|
356
|
+
}
|
|
357
|
+
i = j
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
if (min >= end) {
|
|
361
|
+
return true
|
|
362
|
+
}
|
|
363
|
+
if (max >= start) {
|
|
364
|
+
const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
|
|
365
|
+
callback(
|
|
366
|
+
decoder.decode(buffer.subarray(lineStart, lineEnd)),
|
|
367
|
+
calculateFileOffset(
|
|
368
|
+
cpositions,
|
|
369
|
+
dpositions,
|
|
370
|
+
pos,
|
|
371
|
+
lineStart,
|
|
372
|
+
minvDataPosition,
|
|
373
|
+
),
|
|
374
|
+
min,
|
|
375
|
+
max + 1,
|
|
376
|
+
)
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
return false
|
|
380
|
+
}
|
|
381
|
+
|
|
275
382
|
/**
|
|
276
383
|
* Reads Tabix-indexed files (bgzipped), supporting both .tbi and .csi index formats.
|
|
277
384
|
*/
|
|
@@ -506,6 +613,10 @@ export default class TabixIndexedFile {
|
|
|
506
613
|
// tabs[N] holds the byte offset of the N-th tab on the current line; with
|
|
507
614
|
// the sentinel tabs[0] = blockStart - 1, column N spans tabs[N-1]+1..tabs[N]
|
|
508
615
|
const tabs = new Int32Array(maxColumn + 1)
|
|
616
|
+
const gaf: GafQuery | undefined =
|
|
617
|
+
metadata.format === 'GAF'
|
|
618
|
+
? { start, end, pathCol: startCol, metaCharCode, decoder, callback }
|
|
619
|
+
: undefined
|
|
509
620
|
|
|
510
621
|
let totalBytes = 0
|
|
511
622
|
for (const c of chunks) {
|
|
@@ -521,8 +632,9 @@ export default class TabixIndexedFile {
|
|
|
521
632
|
//
|
|
522
633
|
// A fixed window would be wrong, though: blocksForRange offers a chunk per
|
|
523
634
|
// overlapping bin across every level, and on a sparse file the early return
|
|
524
|
-
// below
|
|
525
|
-
//
|
|
635
|
+
// below can stop the scan inside the first one, leaving the rest untouched.
|
|
636
|
+
// max_off drops only the chunks it can prove lie past the query, so the
|
|
637
|
+
// scan can still end before the last one it is offered. Prefetching those
|
|
526
638
|
// would multiply the bytes such a query fetches for no gain.
|
|
527
639
|
//
|
|
528
640
|
// Finishing a chunk without hitting the early return proves the next chunk
|
|
@@ -551,104 +663,118 @@ export default class TabixIndexedFile {
|
|
|
551
663
|
onProgress?.(downloadedBytes, totalBytes)
|
|
552
664
|
const minvDataPosition = c.minv.dataPosition
|
|
553
665
|
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
if (n === -1) {
|
|
560
|
-
break
|
|
666
|
+
if (gaf) {
|
|
667
|
+
if (
|
|
668
|
+
scanGafChunk(buffer, cpositions, dpositions, minvDataPosition, gaf)
|
|
669
|
+
) {
|
|
670
|
+
return
|
|
561
671
|
}
|
|
672
|
+
} else {
|
|
673
|
+
let blockStart = 0
|
|
674
|
+
let pos = 0
|
|
562
675
|
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
676
|
+
while (blockStart < buffer.length) {
|
|
677
|
+
const n = buffer.indexOf(NEWLINE, blockStart)
|
|
678
|
+
if (n === -1) {
|
|
679
|
+
break
|
|
680
|
+
}
|
|
567
681
|
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
}
|
|
682
|
+
const target = blockStart + minvDataPosition
|
|
683
|
+
while (pos < dpositions.length && target >= dpositions[pos]!) {
|
|
684
|
+
pos++
|
|
685
|
+
}
|
|
573
686
|
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
}
|
|
687
|
+
// skip meta lines
|
|
688
|
+
if (
|
|
689
|
+
metaCharCode !== undefined &&
|
|
690
|
+
buffer[blockStart] === metaCharCode
|
|
691
|
+
) {
|
|
692
|
+
blockStart = n + 1
|
|
693
|
+
continue
|
|
694
|
+
}
|
|
583
695
|
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
let isRefMatch = true
|
|
593
|
-
for (let i = 0; i < refLen; i++) {
|
|
594
|
-
if (buffer[refStart + i] !== regionRefNameBytes[i]) {
|
|
595
|
-
isRefMatch = false
|
|
596
|
-
break
|
|
696
|
+
// find tab positions. Columns past the end of the line all get `n`
|
|
697
|
+
// rather than breaking out, which would leave stale offsets from the
|
|
698
|
+
// previous line in the tail of the array.
|
|
699
|
+
tabs[0] = blockStart - 1
|
|
700
|
+
for (let i = 0; i < maxColumn; i++) {
|
|
701
|
+
const prev = tabs[i]!
|
|
702
|
+
const tabPos = prev < n ? buffer.indexOf(TAB, prev + 1) : -1
|
|
703
|
+
tabs[i + 1] = tabPos === -1 || tabPos >= n ? n : tabPos
|
|
597
704
|
}
|
|
598
|
-
}
|
|
599
|
-
if (!isRefMatch) {
|
|
600
|
-
blockStart = n + 1
|
|
601
|
-
continue
|
|
602
|
-
}
|
|
603
705
|
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
706
|
+
// compare ref name bytes directly
|
|
707
|
+
const refStart = tabs[refCol - 1]! + 1
|
|
708
|
+
const refEnd = tabs[refCol]!
|
|
709
|
+
const refLen = refEnd - refStart
|
|
710
|
+
if (refLen !== regionRefNameBytes.length) {
|
|
711
|
+
blockStart = n + 1
|
|
712
|
+
continue
|
|
713
|
+
}
|
|
714
|
+
let isRefMatch = true
|
|
715
|
+
for (let i = 0; i < refLen; i++) {
|
|
716
|
+
if (buffer[refStart + i] !== regionRefNameBytes[i]) {
|
|
717
|
+
isRefMatch = false
|
|
718
|
+
break
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
if (!isRefMatch) {
|
|
722
|
+
blockStart = n + 1
|
|
723
|
+
continue
|
|
724
|
+
}
|
|
608
725
|
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
726
|
+
// parse start coordinate
|
|
727
|
+
const startCoordinate =
|
|
728
|
+
parseIntFromBytes(
|
|
729
|
+
buffer,
|
|
730
|
+
tabs[startCol - 1]! + 1,
|
|
731
|
+
tabs[startCol]!,
|
|
732
|
+
) + coordinateOffset
|
|
612
733
|
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
endCoordinate = startCoordinate + 1
|
|
617
|
-
} else if (isVCF) {
|
|
618
|
-
endCoordinate = getVcfEnd(
|
|
619
|
-
buffer,
|
|
620
|
-
startCoordinate,
|
|
621
|
-
tabs[3]! + 1,
|
|
622
|
-
tabs[4]!,
|
|
623
|
-
tabs[endCol - 1]! + 1,
|
|
624
|
-
tabs[endCol]!,
|
|
625
|
-
)
|
|
626
|
-
} else {
|
|
627
|
-
endCoordinate = parseIntFromBytes(
|
|
628
|
-
buffer,
|
|
629
|
-
tabs[endCol - 1]! + 1,
|
|
630
|
-
tabs[endCol]!,
|
|
631
|
-
)
|
|
632
|
-
}
|
|
734
|
+
if (startCoordinate >= end) {
|
|
735
|
+
return
|
|
736
|
+
}
|
|
633
737
|
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
)
|
|
647
|
-
|
|
648
|
-
endCoordinate
|
|
649
|
-
|
|
738
|
+
// parse end coordinate
|
|
739
|
+
let endCoordinate: number
|
|
740
|
+
if (endCol === 0 || endCol === startCol) {
|
|
741
|
+
endCoordinate = startCoordinate + 1
|
|
742
|
+
} else if (isVCF) {
|
|
743
|
+
endCoordinate = getVcfEnd(
|
|
744
|
+
buffer,
|
|
745
|
+
startCoordinate,
|
|
746
|
+
tabs[3]! + 1,
|
|
747
|
+
tabs[4]!,
|
|
748
|
+
tabs[endCol - 1]! + 1,
|
|
749
|
+
tabs[endCol]!,
|
|
750
|
+
)
|
|
751
|
+
} else {
|
|
752
|
+
endCoordinate = parseIntFromBytes(
|
|
753
|
+
buffer,
|
|
754
|
+
tabs[endCol - 1]! + 1,
|
|
755
|
+
tabs[endCol]!,
|
|
756
|
+
)
|
|
757
|
+
}
|
|
758
|
+
|
|
759
|
+
if (endCoordinate > start) {
|
|
760
|
+
// trim a CRLF terminator, matching htslib's line reader
|
|
761
|
+
const lineEnd = buffer[n - 1] === CARRIAGE_RETURN ? n - 1 : n
|
|
762
|
+
const line = decoder.decode(buffer.subarray(blockStart, lineEnd))
|
|
763
|
+
callback(
|
|
764
|
+
line,
|
|
765
|
+
calculateFileOffset(
|
|
766
|
+
cpositions,
|
|
767
|
+
dpositions,
|
|
768
|
+
pos,
|
|
769
|
+
blockStart,
|
|
770
|
+
minvDataPosition,
|
|
771
|
+
),
|
|
772
|
+
startCoordinate,
|
|
773
|
+
endCoordinate,
|
|
774
|
+
)
|
|
775
|
+
}
|
|
776
|
+
blockStart = n + 1
|
|
650
777
|
}
|
|
651
|
-
blockStart = n + 1
|
|
652
778
|
}
|
|
653
779
|
|
|
654
780
|
// every line in this chunk was still inside the query, so the next chunk
|