@gmod/tabix 3.5.3 → 3.5.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -0
- package/dist/csi.d.ts +5 -9
- package/dist/csi.js +17 -54
- package/dist/csi.js.map +1 -1
- package/dist/indexFile.d.ts +43 -6
- package/dist/indexFile.js +48 -0
- package/dist/indexFile.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.d.ts +42 -7
- package/dist/tabixIndexedFile.js +104 -27
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/tbi.d.ts +14 -2
- package/dist/tbi.js +36 -58
- package/dist/tbi.js.map +1 -1
- package/esm/csi.d.ts +5 -9
- package/esm/csi.js +18 -55
- package/esm/csi.js.map +1 -1
- package/esm/indexFile.d.ts +43 -6
- package/esm/indexFile.js +48 -0
- package/esm/indexFile.js.map +1 -1
- package/esm/tabixIndexedFile.d.ts +42 -7
- package/esm/tabixIndexedFile.js +104 -27
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/tbi.d.ts +14 -2
- package/esm/tbi.js +37 -59
- package/esm/tbi.js.map +1 -1
- package/package.json +2 -2
- package/src/csi.ts +26 -74
- package/src/indexFile.ts +86 -8
- package/src/tabixIndexedFile.ts +120 -44
- package/src/tbi.ts +48 -82
package/src/tabixIndexedFile.ts
CHANGED
|
@@ -9,6 +9,7 @@ import { optimizeChunks } from './util.ts'
|
|
|
9
9
|
import type Chunk from './chunk.ts'
|
|
10
10
|
import type IndexFile from './indexFile.ts'
|
|
11
11
|
import type { Options } from './indexFile.ts'
|
|
12
|
+
import type { ChunkSlice } from '@gmod/bgzf-filehandle'
|
|
12
13
|
import type { GenericFilehandle } from 'generic-filehandle2'
|
|
13
14
|
|
|
14
15
|
const TAB = 9
|
|
@@ -148,11 +149,11 @@ interface GetLinesOpts {
|
|
|
148
149
|
onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
|
|
149
150
|
}
|
|
150
151
|
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
152
|
+
// The decompressed chunk plus its block offsets, as @gmod/bgzf-filehandle
|
|
153
|
+
// returns them: cpositions/dpositions are Float64Arrays, which is what the
|
|
154
|
+
// wasm decompressor produces, and only indexed reads and `.length` are used
|
|
155
|
+
// here. Taken from the package rather than restated so the two can't drift.
|
|
156
|
+
type ReadChunk = ChunkSlice
|
|
156
157
|
|
|
157
158
|
function resolveFilehandle(
|
|
158
159
|
filehandle?: GenericFilehandle,
|
|
@@ -222,8 +223,8 @@ function resolveIndex({
|
|
|
222
223
|
}
|
|
223
224
|
|
|
224
225
|
function calculateFileOffset(
|
|
225
|
-
cpositions: number
|
|
226
|
-
dpositions: number
|
|
226
|
+
cpositions: ArrayLike<number>,
|
|
227
|
+
dpositions: ArrayLike<number>,
|
|
227
228
|
pos: number,
|
|
228
229
|
blockStart: number,
|
|
229
230
|
minvDataPosition: number,
|
|
@@ -289,6 +290,50 @@ function getVcfEnd(
|
|
|
289
290
|
return endCoordinate
|
|
290
291
|
}
|
|
291
292
|
|
|
293
|
+
const textDecoder = new TextDecoder()
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* The leading run of meta-character lines — what `tabix -H` prints. Everything
|
|
297
|
+
* from the first line that doesn't begin with the meta character is dropped.
|
|
298
|
+
*/
|
|
299
|
+
function trimToMetaLines(bytes: Uint8Array, metaChar: string) {
|
|
300
|
+
let lastNewline = -1
|
|
301
|
+
const metaByte = metaChar.charCodeAt(0)
|
|
302
|
+
|
|
303
|
+
for (let i = 0, l = bytes.length; i < l; i++) {
|
|
304
|
+
const byte = bytes[i]
|
|
305
|
+
if (i === lastNewline + 1 && byte !== metaByte) {
|
|
306
|
+
break
|
|
307
|
+
}
|
|
308
|
+
if (byte === NEWLINE) {
|
|
309
|
+
lastNewline = i
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
return bytes.subarray(0, lastNewline + 1)
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* The first `count` lines. Scans for the count-th newline and decodes only that
|
|
317
|
+
* far, rather than decoding and splitting the whole buffer to keep its first
|
|
318
|
+
* few lines — the buffer runs to the first data line, which for a file with a
|
|
319
|
+
* long commented preamble above its counted rows can be megabytes.
|
|
320
|
+
*/
|
|
321
|
+
function firstLines(bytes: Uint8Array, count: number) {
|
|
322
|
+
let end = 0
|
|
323
|
+
for (let i = 0; i < count; i++) {
|
|
324
|
+
const n = bytes.indexOf(NEWLINE, end)
|
|
325
|
+
if (n === -1) {
|
|
326
|
+
end = bytes.length
|
|
327
|
+
break
|
|
328
|
+
}
|
|
329
|
+
end = n + 1
|
|
330
|
+
}
|
|
331
|
+
return textDecoder
|
|
332
|
+
.decode(bytes.subarray(0, end))
|
|
333
|
+
.split(/\r?\n/)
|
|
334
|
+
.slice(0, count)
|
|
335
|
+
}
|
|
336
|
+
|
|
292
337
|
function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
|
|
293
338
|
let val = 0
|
|
294
339
|
for (let i = start; i < end; i++) {
|
|
@@ -309,6 +354,7 @@ export default class TabixIndexedFile {
|
|
|
309
354
|
private filehandle: GenericFilehandle
|
|
310
355
|
private index: IndexFile
|
|
311
356
|
private chunkCache: AbortablePromiseCache<Chunk, ReadChunk>
|
|
357
|
+
private headerP?: Promise<{ header: string; skippedLines: string[] }>
|
|
312
358
|
|
|
313
359
|
constructor({
|
|
314
360
|
path,
|
|
@@ -588,34 +634,59 @@ export default class TabixIndexedFile {
|
|
|
588
634
|
return this.index.getMetadata(opts)
|
|
589
635
|
}
|
|
590
636
|
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
637
|
+
/**
|
|
638
|
+
* The file's leading blocks, decompressed: everything from the start through
|
|
639
|
+
* the end of the block holding the first data line.
|
|
640
|
+
*/
|
|
641
|
+
private async readHeaderBytes(opts: Options) {
|
|
642
|
+
const { firstDataLine, maxBlockSize } = await this.getMetadata(opts)
|
|
594
643
|
|
|
595
644
|
const maxFetch = (firstDataLine?.blockPosition ?? 0) + maxBlockSize
|
|
596
645
|
// TODO: what if we don't have a firstDataLine, and the header actually
|
|
597
646
|
// takes up more than one block? this case is not covered here
|
|
598
647
|
|
|
599
648
|
const buf = await this.filehandle.read(maxFetch, 0, opts)
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
// trim off lines after the last meta line
|
|
603
|
-
if (metaChar) {
|
|
604
|
-
let lastNewline = -1
|
|
605
|
-
const metaByte = metaChar.charCodeAt(0)
|
|
649
|
+
return unzip(buf)
|
|
650
|
+
}
|
|
606
651
|
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
652
|
+
/**
|
|
653
|
+
* Both header forms, from one read of the same bytes: the commented block
|
|
654
|
+
* `tabix -H` prints, and the rows `tabix -S N` counted.
|
|
655
|
+
*
|
|
656
|
+
* Memoized, because asking for one and then the other is the normal way to
|
|
657
|
+
* find a header whichever way the file keeps it (see `getHeaderLines`), and
|
|
658
|
+
* that used to fetch and decompress the file's leading blocks twice. Only
|
|
659
|
+
* the parsed results are retained — the decompressed bytes are dropped,
|
|
660
|
+
* which matters for a VCF header that can run to megabytes.
|
|
661
|
+
*/
|
|
662
|
+
private async parseHeader(opts: Options) {
|
|
663
|
+
const { metaChar, skipLines = 0 } = await this.getMetadata(opts)
|
|
664
|
+
const bytes = await this.readHeaderBytes(opts)
|
|
665
|
+
return {
|
|
666
|
+
header: textDecoder.decode(
|
|
667
|
+
metaChar ? trimToMetaLines(bytes, metaChar) : bytes,
|
|
668
|
+
),
|
|
669
|
+
skippedLines: skipLines > 0 ? firstLines(bytes, skipLines) : [],
|
|
617
670
|
}
|
|
618
|
-
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
private async getParsedHeader(opts: Options = {}) {
|
|
674
|
+
this.headerP ??= this.parseHeader(opts).catch((error: unknown) => {
|
|
675
|
+
this.headerP = undefined
|
|
676
|
+
throw error
|
|
677
|
+
})
|
|
678
|
+
return this.headerP
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
/**
|
|
682
|
+
* The bytes of the commented header. Deliberately not memoized, unlike the
|
|
683
|
+
* string form: this hands back the buffer, and holding one for the lifetime
|
|
684
|
+
* of the file is the caller's decision to make.
|
|
685
|
+
*/
|
|
686
|
+
async getHeaderBuffer(opts: Options = {}) {
|
|
687
|
+
const { metaChar } = await this.getMetadata(opts)
|
|
688
|
+
const bytes = await this.readHeaderBytes(opts)
|
|
689
|
+
return metaChar ? trimToMetaLines(bytes, metaChar) : bytes
|
|
619
690
|
}
|
|
620
691
|
|
|
621
692
|
/**
|
|
@@ -634,29 +705,34 @@ export default class TabixIndexedFile {
|
|
|
634
705
|
* way.
|
|
635
706
|
*/
|
|
636
707
|
async getSkippedLines(opts: Options = {}) {
|
|
637
|
-
const {
|
|
638
|
-
|
|
639
|
-
skipLines = 0,
|
|
640
|
-
maxBlockSize,
|
|
641
|
-
} = await this.getMetadata(opts)
|
|
708
|
+
const { skipLines = 0 } = await this.getMetadata(opts)
|
|
709
|
+
// the index already answers this without reading the file at all
|
|
642
710
|
if (skipLines <= 0) {
|
|
643
711
|
return []
|
|
644
712
|
}
|
|
645
|
-
|
|
646
|
-
// same read getHeaderBuffer makes, and the same caveat: a header spanning
|
|
647
|
-
// more blocks than this is not covered
|
|
648
|
-
const buf = await this.filehandle.read(
|
|
649
|
-
(firstDataLine?.blockPosition ?? 0) + maxBlockSize,
|
|
650
|
-
0,
|
|
651
|
-
opts,
|
|
652
|
-
)
|
|
653
|
-
const bytes = (await unzip(buf)) as Uint8Array
|
|
654
|
-
return new TextDecoder().decode(bytes).split(/\r?\n/).slice(0, skipLines)
|
|
713
|
+
return (await this.getParsedHeader(opts)).skippedLines
|
|
655
714
|
}
|
|
656
715
|
|
|
657
716
|
async getHeader(opts: Options = {}) {
|
|
658
|
-
|
|
659
|
-
|
|
717
|
+
return (await this.getParsedHeader(opts)).header
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
/**
|
|
721
|
+
* The file's header lines, however that file keeps them: the meta-character
|
|
722
|
+
* block when there is one, and otherwise the rows the index counted. Empty
|
|
723
|
+
* lines are dropped.
|
|
724
|
+
*
|
|
725
|
+
* The two halves answer different questions (see `getSkippedLines`), but
|
|
726
|
+
* "what are this file's header lines" is nearly always the question a caller
|
|
727
|
+
* actually has, and answering it from `getHeader` alone is wrong in a way
|
|
728
|
+
* that doesn't announce itself: a bare header row comes back as the empty
|
|
729
|
+
* string, indistinguishable from a file that has no header, so callers fall
|
|
730
|
+
* back to an assumed column layout and quietly mis-name columns. Deciding it
|
|
731
|
+
* here also means one read of the leading blocks instead of two.
|
|
732
|
+
*/
|
|
733
|
+
async getHeaderLines(opts: Options = {}) {
|
|
734
|
+
const { header, skippedLines } = await this.getParsedHeader(opts)
|
|
735
|
+
return (header ? header.split(/\r?\n/) : skippedLines).filter(Boolean)
|
|
660
736
|
}
|
|
661
737
|
|
|
662
738
|
async getReferenceSequenceNames(opts: Options = {}) {
|
package/src/tbi.ts
CHANGED
|
@@ -1,12 +1,9 @@
|
|
|
1
|
-
import { unzip } from '@gmod/bgzf-filehandle'
|
|
2
|
-
|
|
3
1
|
import Chunk from './chunk.ts'
|
|
4
2
|
import IndexFile from './indexFile.ts'
|
|
5
3
|
import {
|
|
6
4
|
clampChunkEnds,
|
|
7
5
|
memoizeByRefId,
|
|
8
6
|
minVirtualOffset,
|
|
9
|
-
optimizeChunks,
|
|
10
7
|
parseAuxData,
|
|
11
8
|
parsePseudoBin,
|
|
12
9
|
} from './util.ts'
|
|
@@ -19,44 +16,57 @@ const TBI_MAGIC = 21_578_324 // TBI\1
|
|
|
19
16
|
|
|
20
17
|
// TBI is always the fixed depth-5, 14-bit-leaf binning scheme
|
|
21
18
|
const TBI_DEPTH = 5
|
|
22
|
-
const
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
function
|
|
29
|
-
|
|
30
|
-
// truncate to int32 and go negative, which would silently yield no bins.
|
|
31
|
-
const b = Math.min(Math.max(beg, 0), TBI_MAX_REF_LENGTH)
|
|
32
|
-
const e = Math.min(end, TBI_MAX_REF_LENGTH) - 1
|
|
33
|
-
const bins: (readonly [number, number])[] = []
|
|
34
|
-
if (b <= e) {
|
|
35
|
-
bins.push(
|
|
36
|
-
[0, 0],
|
|
37
|
-
[1 + (b >> 26), 1 + (e >> 26)],
|
|
38
|
-
[9 + (b >> 23), 9 + (e >> 23)],
|
|
39
|
-
[73 + (b >> 20), 73 + (e >> 20)],
|
|
40
|
-
[585 + (b >> 17), 585 + (e >> 17)],
|
|
41
|
-
[4681 + (b >> 14), 4681 + (e >> 14)],
|
|
42
|
-
)
|
|
43
|
-
}
|
|
44
|
-
return bins
|
|
19
|
+
const TBI_MIN_SHIFT = 14
|
|
20
|
+
const TBI_MAX_REF_LENGTH = 2 ** (TBI_MIN_SHIFT + TBI_DEPTH * 3)
|
|
21
|
+
|
|
22
|
+
// Clamp a query coordinate into the binning scheme's range. Past 2**31 the
|
|
23
|
+
// int32 shifts below truncate and go negative, which would silently yield no
|
|
24
|
+
// bins.
|
|
25
|
+
function clampToRefLength(pos: number) {
|
|
26
|
+
return Math.min(Math.max(pos, 0), TBI_MAX_REF_LENGTH)
|
|
45
27
|
}
|
|
46
28
|
|
|
47
29
|
export default class TabixIndex extends IndexFile {
|
|
30
|
+
/**
|
|
31
|
+
* The bins that may overlap [beg, end) (zero-based half-open). TBI's scheme
|
|
32
|
+
* is CSI's with minShift/depth fixed, so this is CSI's reg2bins with the
|
|
33
|
+
* levels unrolled and its float shifts replaced by int32 ones.
|
|
34
|
+
*/
|
|
35
|
+
protected reg2bins(beg: number, end: number) {
|
|
36
|
+
if (beg > TBI_MAX_REF_LENGTH) {
|
|
37
|
+
console.warn('querying outside of possible tabix range')
|
|
38
|
+
}
|
|
39
|
+
const b = clampToRefLength(beg)
|
|
40
|
+
const e = Math.min(end, TBI_MAX_REF_LENGTH) - 1
|
|
41
|
+
const bins: (readonly [number, number])[] = []
|
|
42
|
+
if (b <= e) {
|
|
43
|
+
bins.push(
|
|
44
|
+
[0, 0],
|
|
45
|
+
[1 + (b >> 26), 1 + (e >> 26)],
|
|
46
|
+
[9 + (b >> 23), 9 + (e >> 23)],
|
|
47
|
+
[73 + (b >> 20), 73 + (e >> 20)],
|
|
48
|
+
[585 + (b >> 17), 585 + (e >> 17)],
|
|
49
|
+
[4681 + (b >> 14), 4681 + (e >> 14)],
|
|
50
|
+
)
|
|
51
|
+
}
|
|
52
|
+
return bins
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* The linear index is monotonically non-decreasing, so the minimum virtual
|
|
57
|
+
* offset for chunks that could overlap [beg, ...) is its entry for beg.
|
|
58
|
+
* SYNC: ~/src/gmod/bam-js/src/bai.ts getLowestChunk
|
|
59
|
+
*/
|
|
60
|
+
protected lowestOffset(ref: RefIndex, beg: number) {
|
|
61
|
+
const linearIndex = ref.linearIndex
|
|
62
|
+
return linearIndex?.[
|
|
63
|
+
Math.min(clampToRefLength(beg) >> TBI_MIN_SHIFT, linearIndex.length - 1)
|
|
64
|
+
]
|
|
65
|
+
}
|
|
66
|
+
|
|
48
67
|
/** @internal */
|
|
49
68
|
async _parse(opts: Options = {}) {
|
|
50
|
-
const
|
|
51
|
-
signal: opts.signal,
|
|
52
|
-
onProgress: opts.onProgress,
|
|
53
|
-
})
|
|
54
|
-
const bytes = (await unzip(buf)) as Uint8Array
|
|
55
|
-
const dataView = new DataView(
|
|
56
|
-
bytes.buffer,
|
|
57
|
-
bytes.byteOffset,
|
|
58
|
-
bytes.byteLength,
|
|
59
|
-
)
|
|
69
|
+
const { bytes, dataView } = await this.readIndexBytes(opts)
|
|
60
70
|
|
|
61
71
|
if (dataView.getUint32(0, true) !== TBI_MAGIC) {
|
|
62
72
|
throw new Error('Not a TBI file')
|
|
@@ -169,6 +179,8 @@ export default class TabixIndex extends IndexFile {
|
|
|
169
179
|
return {
|
|
170
180
|
indices: memoizeByRefId(getIndices),
|
|
171
181
|
metaChar,
|
|
182
|
+
minShift: TBI_MIN_SHIFT,
|
|
183
|
+
depth: TBI_DEPTH,
|
|
172
184
|
maxBinNumber,
|
|
173
185
|
maxRefLength: TBI_MAX_REF_LENGTH,
|
|
174
186
|
skipLines,
|
|
@@ -181,50 +193,4 @@ export default class TabixIndex extends IndexFile {
|
|
|
181
193
|
maxBlockSize: 1 << 16,
|
|
182
194
|
}
|
|
183
195
|
}
|
|
184
|
-
|
|
185
|
-
async blocksForRange(
|
|
186
|
-
refName: string,
|
|
187
|
-
min: number,
|
|
188
|
-
max: number,
|
|
189
|
-
opts: Options = {},
|
|
190
|
-
) {
|
|
191
|
-
const indexData = await this.parse(opts)
|
|
192
|
-
const refId = indexData.refNameToId[refName]
|
|
193
|
-
if (refId === undefined) {
|
|
194
|
-
return []
|
|
195
|
-
}
|
|
196
|
-
const ba = indexData.indices(refId)
|
|
197
|
-
if (!ba) {
|
|
198
|
-
return []
|
|
199
|
-
}
|
|
200
|
-
|
|
201
|
-
if (min > TBI_MAX_REF_LENGTH) {
|
|
202
|
-
console.warn('querying outside of possible tabix range')
|
|
203
|
-
}
|
|
204
|
-
// clamp before the >> 14 below, which would truncate to int32 and go
|
|
205
|
-
// negative past 2**31
|
|
206
|
-
const beg = Math.min(Math.max(min, 0), TBI_MAX_REF_LENGTH)
|
|
207
|
-
const overlappingBins = reg2bins(min, max) // List of bin #s that overlap min, max
|
|
208
|
-
const chunks: Chunk[] = []
|
|
209
|
-
|
|
210
|
-
// Find chunks in overlapping bins. Leaf bins (< 4681) are not pruned
|
|
211
|
-
for (const [start, end] of overlappingBins) {
|
|
212
|
-
for (let bin = start; bin <= end; bin++) {
|
|
213
|
-
const binChunks = ba.binIndex[bin]
|
|
214
|
-
if (binChunks) {
|
|
215
|
-
for (const c of binChunks) {
|
|
216
|
-
chunks.push(c)
|
|
217
|
-
}
|
|
218
|
-
}
|
|
219
|
-
}
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
// The linear index is monotonically non-decreasing, so the minimum virtual
|
|
223
|
-
// offset for chunks that could overlap [min, ...) is at index minLin.
|
|
224
|
-
// SYNC: ~/src/gmod/bam-js/src/bai.ts getLowestChunk
|
|
225
|
-
const linearIndex = ba.linearIndex
|
|
226
|
-
const lowest = linearIndex?.[Math.min(beg >> 14, linearIndex.length - 1)]
|
|
227
|
-
|
|
228
|
-
return optimizeChunks(chunks, lowest)
|
|
229
|
-
}
|
|
230
196
|
}
|