@gmod/tabix 3.5.3 → 3.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -0
- package/dist/csi.d.ts +5 -9
- package/dist/csi.js +17 -54
- package/dist/csi.js.map +1 -1
- package/dist/indexFile.d.ts +43 -6
- package/dist/indexFile.js +48 -0
- package/dist/indexFile.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.d.ts +42 -7
- package/dist/tabixIndexedFile.js +104 -27
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/tbi.d.ts +14 -2
- package/dist/tbi.js +36 -58
- package/dist/tbi.js.map +1 -1
- package/dist/util.js +40 -1
- package/dist/util.js.map +1 -1
- package/esm/csi.d.ts +5 -9
- package/esm/csi.js +18 -55
- package/esm/csi.js.map +1 -1
- package/esm/indexFile.d.ts +43 -6
- package/esm/indexFile.js +48 -0
- package/esm/indexFile.js.map +1 -1
- package/esm/tabixIndexedFile.d.ts +42 -7
- package/esm/tabixIndexedFile.js +104 -27
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/tbi.d.ts +14 -2
- package/esm/tbi.js +37 -59
- package/esm/tbi.js.map +1 -1
- package/esm/util.js +40 -1
- package/esm/util.js.map +1 -1
- package/package.json +2 -2
- package/src/csi.ts +26 -74
- package/src/indexFile.ts +86 -8
- package/src/tabixIndexedFile.ts +120 -44
- package/src/tbi.ts +48 -82
- package/src/util.ts +43 -1
package/esm/csi.d.ts
CHANGED
|
@@ -1,11 +1,7 @@
|
|
|
1
|
-
import Chunk from './chunk.ts';
|
|
2
1
|
import IndexFile from './indexFile.ts';
|
|
3
|
-
import type { Options, RefIndex } from './indexFile.ts';
|
|
2
|
+
import type { IndexData, Options, RefIndex } from './indexFile.ts';
|
|
4
3
|
import type VirtualOffset from './virtualOffset.ts';
|
|
5
4
|
export default class CSI extends IndexFile {
|
|
6
|
-
private maxBinNumber;
|
|
7
|
-
private depth;
|
|
8
|
-
private minShift;
|
|
9
5
|
/** @internal */
|
|
10
6
|
_parse(opts?: Options): Promise<{
|
|
11
7
|
csi: boolean;
|
|
@@ -14,6 +10,7 @@ export default class CSI extends IndexFile {
|
|
|
14
10
|
firstDataLine: VirtualOffset | undefined;
|
|
15
11
|
csiVersion: number;
|
|
16
12
|
indices: (refId: number) => RefIndex | undefined;
|
|
13
|
+
minShift: number;
|
|
17
14
|
depth: number;
|
|
18
15
|
maxBinNumber: number;
|
|
19
16
|
maxRefLength: number;
|
|
@@ -35,6 +32,7 @@ export default class CSI extends IndexFile {
|
|
|
35
32
|
firstDataLine: VirtualOffset | undefined;
|
|
36
33
|
csiVersion: number;
|
|
37
34
|
indices: (refId: number) => RefIndex | undefined;
|
|
35
|
+
minShift: number;
|
|
38
36
|
depth: number;
|
|
39
37
|
maxBinNumber: number;
|
|
40
38
|
maxRefLength: number;
|
|
@@ -49,7 +47,6 @@ export default class CSI extends IndexFile {
|
|
|
49
47
|
coordinateType: string;
|
|
50
48
|
format: string;
|
|
51
49
|
}>;
|
|
52
|
-
blocksForRange(refName: string, min: number, max: number, opts?: Options): Promise<Chunk[]>;
|
|
53
50
|
/**
|
|
54
51
|
* CSI's equivalent of the BAI/TBI linear index: the loffset of the deepest
|
|
55
52
|
* indexed bin covering `beg` is the earliest virtual offset any record
|
|
@@ -57,7 +54,6 @@ export default class CSI extends IndexFile {
|
|
|
57
54
|
* Walks leaf -> previous sibling -> parent until an indexed bin is found.
|
|
58
55
|
* SYNC: htslib hts_itr_query min_off computation
|
|
59
56
|
*/
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
reg2bins(beg: number, end: number): (readonly [number, number])[];
|
|
57
|
+
protected lowestOffset(ref: RefIndex, beg: number, { minShift, depth }: IndexData): VirtualOffset | undefined;
|
|
58
|
+
protected reg2bins(beg: number, end: number, { minShift, depth, maxBinNumber }: IndexData): (readonly [number, number])[];
|
|
63
59
|
}
|
package/esm/csi.js
CHANGED
|
@@ -1,7 +1,6 @@
|
|
|
1
|
-
import { unzip } from '@gmod/bgzf-filehandle';
|
|
2
1
|
import Chunk from "./chunk.js";
|
|
3
2
|
import IndexFile from "./indexFile.js";
|
|
4
|
-
import { clampChunkEnds, memoizeByRefId, minVirtualOffset,
|
|
3
|
+
import { clampChunkEnds, memoizeByRefId, minVirtualOffset, parseAuxData, parsePseudoBin, } from "./util.js";
|
|
5
4
|
import { fromBytes } from "./virtualOffset.js";
|
|
6
5
|
const CSI1_MAGIC = 21_582_659; // CSI\1
|
|
7
6
|
const CSI2_MAGIC = 38_359_875; // CSI\2
|
|
@@ -14,17 +13,9 @@ function rshift(num, bits) {
|
|
|
14
13
|
return Math.floor(num / 2 ** bits);
|
|
15
14
|
}
|
|
16
15
|
export default class CSI extends IndexFile {
|
|
17
|
-
maxBinNumber = 0;
|
|
18
|
-
depth = 0;
|
|
19
|
-
minShift = 0;
|
|
20
16
|
/** @internal */
|
|
21
17
|
async _parse(opts = {}) {
|
|
22
|
-
const
|
|
23
|
-
signal: opts.signal,
|
|
24
|
-
onProgress: opts.onProgress,
|
|
25
|
-
});
|
|
26
|
-
const bytes = (await unzip(buf));
|
|
27
|
-
const dataView = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
18
|
+
const { bytes, dataView } = await this.readIndexBytes(opts);
|
|
28
19
|
const magic = dataView.getUint32(0, true);
|
|
29
20
|
let csiVersion;
|
|
30
21
|
if (magic === CSI1_MAGIC) {
|
|
@@ -36,11 +27,10 @@ export default class CSI extends IndexFile {
|
|
|
36
27
|
else {
|
|
37
28
|
throw new Error(`Not a CSI file (magic=${magic})`);
|
|
38
29
|
}
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
const
|
|
43
|
-
const maxRefLength = 2 ** (this.minShift + this.depth * 3);
|
|
30
|
+
const minShift = dataView.getInt32(4, true);
|
|
31
|
+
const depth = dataView.getInt32(8, true);
|
|
32
|
+
const maxBinNumber = ((1 << ((depth + 1) * 3)) - 1) / 7;
|
|
33
|
+
const maxRefLength = 2 ** (minShift + depth * 3);
|
|
44
34
|
const auxLength = dataView.getInt32(12, true);
|
|
45
35
|
const aux = auxLength >= 30
|
|
46
36
|
? parseAuxData(bytes, 16)
|
|
@@ -118,40 +108,12 @@ export default class CSI extends IndexFile {
|
|
|
118
108
|
firstDataLine,
|
|
119
109
|
csiVersion,
|
|
120
110
|
indices: memoizeByRefId(getIndices),
|
|
121
|
-
|
|
111
|
+
minShift,
|
|
112
|
+
depth,
|
|
122
113
|
maxBinNumber,
|
|
123
114
|
maxRefLength,
|
|
124
115
|
};
|
|
125
116
|
}
|
|
126
|
-
async blocksForRange(refName, min, max, opts = {}) {
|
|
127
|
-
if (min < 0) {
|
|
128
|
-
min = 0;
|
|
129
|
-
}
|
|
130
|
-
const indexData = await this.parse(opts);
|
|
131
|
-
const refId = indexData.refNameToId[refName];
|
|
132
|
-
if (refId === undefined) {
|
|
133
|
-
return [];
|
|
134
|
-
}
|
|
135
|
-
const ba = indexData.indices(refId);
|
|
136
|
-
if (!ba) {
|
|
137
|
-
return [];
|
|
138
|
-
}
|
|
139
|
-
// List of bin #s that overlap min, max
|
|
140
|
-
const overlappingBins = this.reg2bins(min, max);
|
|
141
|
-
const chunks = [];
|
|
142
|
-
// Find chunks in overlapping bins. Leaf bins (< 4681) are not pruned
|
|
143
|
-
for (const [start, end] of overlappingBins) {
|
|
144
|
-
for (let bin = start; bin <= end; bin++) {
|
|
145
|
-
const binChunks = ba.binIndex[bin];
|
|
146
|
-
if (binChunks) {
|
|
147
|
-
for (const c of binChunks) {
|
|
148
|
-
chunks.push(c);
|
|
149
|
-
}
|
|
150
|
-
}
|
|
151
|
-
}
|
|
152
|
-
}
|
|
153
|
-
return optimizeChunks(chunks, this.minOffset(ba.loffsets, min));
|
|
154
|
-
}
|
|
155
117
|
/**
|
|
156
118
|
* CSI's equivalent of the BAI/TBI linear index: the loffset of the deepest
|
|
157
119
|
* indexed bin covering `beg` is the earliest virtual offset any record
|
|
@@ -159,11 +121,12 @@ export default class CSI extends IndexFile {
|
|
|
159
121
|
* Walks leaf -> previous sibling -> parent until an indexed bin is found.
|
|
160
122
|
* SYNC: htslib hts_itr_query min_off computation
|
|
161
123
|
*/
|
|
162
|
-
|
|
124
|
+
lowestOffset(ref, beg, { minShift, depth }) {
|
|
125
|
+
const { loffsets } = ref;
|
|
163
126
|
let found;
|
|
164
127
|
if (loffsets) {
|
|
165
128
|
// first bin of the deepest level, i.e. (8**depth - 1) / 7
|
|
166
|
-
let bin = (lshift(1, 3 *
|
|
129
|
+
let bin = (lshift(1, 3 * depth) - 1) / 7 + rshift(Math.max(beg, 0), minShift);
|
|
167
130
|
while (found === undefined && bin > 0) {
|
|
168
131
|
found = loffsets[bin];
|
|
169
132
|
if (found === undefined) {
|
|
@@ -176,22 +139,22 @@ export default class CSI extends IndexFile {
|
|
|
176
139
|
}
|
|
177
140
|
return found;
|
|
178
141
|
}
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
const maxPos = 2 ** (
|
|
142
|
+
reg2bins(beg, end, { minShift, depth, maxBinNumber }) {
|
|
143
|
+
beg = Math.max(beg, 0);
|
|
144
|
+
const maxPos = 2 ** (minShift + depth * 3);
|
|
182
145
|
if (end > maxPos) {
|
|
183
146
|
end = maxPos;
|
|
184
147
|
}
|
|
185
148
|
end -= 1;
|
|
186
149
|
let l = 0;
|
|
187
150
|
let t = 0;
|
|
188
|
-
let s =
|
|
151
|
+
let s = minShift + depth * 3;
|
|
189
152
|
const bins = [];
|
|
190
|
-
for (; l <=
|
|
153
|
+
for (; l <= depth; s -= 3, t += lshift(1, l * 3), l += 1) {
|
|
191
154
|
const b = t + rshift(beg, s);
|
|
192
155
|
const e = t + rshift(end, s);
|
|
193
|
-
if (e - b + bins.length >
|
|
194
|
-
throw new Error(`query ${beg}-${end} is too large for current binning scheme (shift ${
|
|
156
|
+
if (e - b + bins.length > maxBinNumber) {
|
|
157
|
+
throw new Error(`query ${beg}-${end} is too large for current binning scheme (shift ${minShift}, depth ${depth}), try a smaller query or a coarser index binning scheme`);
|
|
195
158
|
}
|
|
196
159
|
bins.push([b, e]);
|
|
197
160
|
}
|
package/esm/csi.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"csi.js","sourceRoot":"","sources":["../src/csi.ts"],"names":[],"mappings":"AAAA,OAAO,
|
|
1
|
+
{"version":3,"file":"csi.js","sourceRoot":"","sources":["../src/csi.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,MAAM,YAAY,CAAA;AAC9B,OAAO,SAAS,MAAM,gBAAgB,CAAA;AACtC,OAAO,EACL,cAAc,EACd,cAAc,EACd,gBAAgB,EAChB,YAAY,EACZ,cAAc,GACf,MAAM,WAAW,CAAA;AAClB,OAAO,EAAE,SAAS,EAAE,MAAM,oBAAoB,CAAA;AAK9C,MAAM,UAAU,GAAG,UAAU,CAAA,CAAC,QAAQ;AACtC,MAAM,UAAU,GAAG,UAAU,CAAA,CAAC,QAAQ;AAEtC,gFAAgF;AAChF,wDAAwD;AACxD,SAAS,MAAM,CAAC,GAAW,EAAE,IAAY;IACvC,OAAO,GAAG,GAAG,CAAC,IAAI,IAAI,CAAA;AACxB,CAAC;AACD,SAAS,MAAM,CAAC,GAAW,EAAE,IAAY;IACvC,OAAO,IAAI,CAAC,KAAK,CAAC,GAAG,GAAG,CAAC,IAAI,IAAI,CAAC,CAAA;AACpC,CAAC;AAED,MAAM,CAAC,OAAO,OAAO,GAAI,SAAQ,SAAS;IACxC,gBAAgB;IAChB,KAAK,CAAC,MAAM,CAAC,OAAgB,EAAE;QAC7B,MAAM,EAAE,KAAK,EAAE,QAAQ,EAAE,GAAG,MAAM,IAAI,CAAC,cAAc,CAAC,IAAI,CAAC,CAAA;QAE3D,MAAM,KAAK,GAAG,QAAQ,CAAC,SAAS,CAAC,CAAC,EAAE,IAAI,CAAC,CAAA;QACzC,IAAI,UAAU,CAAA;QACd,IAAI,KAAK,KAAK,UAAU,EAAE,CAAC;YACzB,UAAU,GAAG,CAAC,CAAA;QAChB,CAAC;aAAM,IAAI,KAAK,KAAK,UAAU,EAAE,CAAC;YAChC,UAAU,GAAG,CAAC,CAAA;QAChB,CAAC;aAAM,CAAC;YACN,MAAM,IAAI,KAAK,CAAC,yBAAyB,KAAK,GAAG,CAAC,CAAA;QACpD,CAAC;QAED,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,CAAC,EAAE,IAAI,CAAC,CAAA;QAC3C,MAAM,KAAK,GAAG,QAAQ,CAAC,QAAQ,CAAC,CAAC,EAAE,IAAI,CAAC,CAAA;QACxC,MAAM,YAAY,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,GAAG,CAAC,CAAA;QACvD,MAAM,YAAY,GAAG,CAAC,IAAI,CAAC,QAAQ,GAAG,KAAK,GAAG,CAAC,CAAC,CAAA;QAChD,MAAM,SAAS,GAAG,QAAQ,CAAC,QAAQ,CAAC,EAAE,EAAE,IAAI,CAAC,CAAA;QAC7C,MAAM,GAAG,GACP,SAAS,IAAI,EAAE;YACb,CAAC,CAAC,YAAY,CAAC,KAAK,EAAE,EAAE,CAAC;YACzB,CAAC,CAAC;gBACE,WAAW,EAAE,EAAc;gBAC3B,WAAW,EAAE,EAA4B;gBACzC,QAAQ,EAAE,SAAS;gBACnB,aAAa,EAAE,EAAE,GAAG,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,GAAG,EAAE,CAAC,EAAE;gBAC3C,cAAc,EAAE,sBAAsB;gBACtC,MAAM,EAAE,SAAS;aAClB,CAAA;QACP,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,EAAE,GAAG,SAAS,EAAE,IAAI,CAAC,CAAA;QAExD,iEAAiE;QACjE,gFAAgF;QAChF,IAAI,IAAI,GAAG,EAAE,GAAG,SAAS,GAAG,CAAC,CAAA;QAC7B,IAAI,aAAwC,CAAA;QAC5C,MAAM,OAAO,GAAa,EAAE,CAAA;QAE5B,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YAClC,OAAO,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;YAClB,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,IAAI,EAAE,IAAI,CAAC,CAAA;YAC9C,IAAI,IAAI,CAAC,CAAA;YACT,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;gBAClC,MAAM,GAAG,GAAG,QAAQ,CAAC,SAAS,CAAC,IAAI,EAAE,IAAI,CAAC,CAAA;gBAC1C,IAAI,IAAI,CAAC,CAAA;gBACT,IAAI,GAAG,GAAG,YAAY,EAAE,CAAC;oBACvB,IAAI,IAAI,EAAE,GAAG,EAAE,CAAA,CAAC,gDAAgD;gBAClE,CAAC;qBAAM,CAAC;oBACN,aAAa,GAAG,gBAAgB,CAAC,KAAK,EAAE,IAAI,EAAE,CAAC,EAAE,aAAa,CAAC,CAAA;oBAC/D,IAAI,IAAI,CAAC,CAAA,CAAC,UAAU;oBACpB,MAAM,UAAU,GAAG,QAAQ,CAAC,QAAQ,CAAC,IAAI,EAAE,IAAI,CAAC,CAAA;oBAChD,IAAI,IAAI,CAAC,GAAG,EAAE,GAAG,UAAU,CAAA;gBAC7B,CAAC;YACH,CAAC;QACH,CAAC;QAED,SAAS,UAAU,CAAC,KAAa;YAC/B,MAAM,KAAK,GAAG,OAAO,CAAC,KAAK,CAAC,CAAA;YAC5B,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;gBACxB,OAAO,SAAS,CAAA;YAClB,CAAC;YACD,IAAI,GAAG,GAAG,KAAK,CAAA;YACf,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,GAAG,EAAE,IAAI,CAAC,CAAA;YAC7C,GAAG,IAAI,CAAC,CAAA;YACR,MAAM,QAAQ,GAA4B,EAAE,CAAA;YAC5C,MAAM,QAAQ,GAAkC,EAAE,CAAA;YAClD,IAAI,KAAK,CAAA;YACT,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;gBAClC,MAAM,GAAG,GAAG,QAAQ,CAAC,SAAS,CAAC,GAAG,EAAE,IAAI,CAAC,CAAA;gBACzC,GAAG,IAAI,CAAC,CAAA;gBACR,IAAI,GAAG,GAAG,YAAY,EAAE,CAAC;oBACvB,KAAK,GAAG,cAAc,CAAC,KAAK,EAAE,GAAG,GAAG,EAAE,CAAC,CAAA;oBACvC,GAAG,IAAI,EAAE,GAAG,EAAE,CAAA;gBAChB,CAAC;qBAAM,CAAC;oBACN,QAAQ,CAAC,GAAG,CAAC,GAAG,SAAS,CAAC,KAAK,EAAE,GAAG,CAAC,CAAA;oBACrC,GAAG,IAAI,CAAC,CAAA;oBACR,MAAM,UAAU,GAAG,QAAQ,CAAC,QAAQ,CAAC,GAAG,EAAE,IAAI,CAAC,CAAA;oBAC/C,GAAG,IAAI,CAAC,CAAA;oBACR,MAAM,MAAM,GAAG,KAAK,CAAC,IAAI,CAAQ,EAAE,MAAM,EAAE,UAAU,EAAE,CAAC,CAAA;oBACxD,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,UAAU,EAAE,CAAC,EAAE,EAAE,CAAC;wBACpC,MAAM,CAAC,CAAC,CAAC,GAAG,IAAI,KAAK,CACnB,SAAS,CAAC,KAAK,EAAE,GAAG,CAAC,EACrB,SAAS,CAAC,KAAK,EAAE,GAAG,GAAG,CAAC,CAAC,EACzB,GAAG,CACJ,CAAA;wBACD,GAAG,IAAI,EAAE,CAAA;oBACX,CAAC;oBACD,QAAQ,CAAC,GAAG,CAAC,GAAG,MAAM,CAAA;gBACxB,CAAC;YACH,CAAC;YACD,cAAc,CAAC,MAAM,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,IAAI,EAAE,CAAC,CAAA;YAC9C,OAAO,EAAE,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,CAAA;QACtC,CAAC;QAED,OAAO;YACL,GAAG,GAAG;YACN,GAAG,EAAE,IAAI;YACT,QAAQ;YACR,YAAY,EAAE,CAAC,IAAI,EAAE;YACrB,aAAa;YACb,UAAU;YACV,OAAO,EAAE,cAAc,CAAC,UAAU,CAAC;YACnC,QAAQ;YACR,KAAK;YACL,YAAY;YACZ,YAAY;SACb,CAAA;IACH,CAAC;IAED;;;;;;OAMG;IACO,YAAY,CACpB,GAAa,EACb,GAAW,EACX,EAAE,QAAQ,EAAE,KAAK,EAAa;QAE9B,MAAM,EAAE,QAAQ,EAAE,GAAG,GAAG,CAAA;QACxB,IAAI,KAAgC,CAAA;QACpC,IAAI,QAAQ,EAAE,CAAC;YACb,0DAA0D;YAC1D,IAAI,GAAG,GACL,CAAC,MAAM,CAAC,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,GAAG,CAAC,CAAC,GAAG,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,CAAC,EAAE,QAAQ,CAAC,CAAA;YACrE,OAAO,KAAK,KAAK,SAAS,IAAI,GAAG,GAAG,CAAC,EAAE,CAAC;gBACtC,KAAK,GAAG,QAAQ,CAAC,GAAG,CAAC,CAAA;gBACrB,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;oBACxB,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,GAAG,GAAG,CAAC,CAAC,GAAG,CAAC,CAAC,CAAA;oBACxC,MAAM,UAAU,GAAG,MAAM,GAAG,CAAC,GAAG,CAAC,CAAA;oBACjC,GAAG,GAAG,GAAG,GAAG,UAAU,CAAC,CAAC,CAAC,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC,MAAM,CAAA;gBAC3C,CAAC;YACH,CAAC;YACD,KAAK,KAAK,QAAQ,CAAC,CAAC,CAAC,CAAA;QACvB,CAAC;QACD,OAAO,KAAK,CAAA;IACd,CAAC;IAES,QAAQ,CAChB,GAAW,EACX,GAAW,EACX,EAAE,QAAQ,EAAE,KAAK,EAAE,YAAY,EAAa;QAE5C,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;QACtB,MAAM,MAAM,GAAG,CAAC,IAAI,CAAC,QAAQ,GAAG,KAAK,GAAG,CAAC,CAAC,CAAA;QAC1C,IAAI,GAAG,GAAG,MAAM,EAAE,CAAC;YACjB,GAAG,GAAG,MAAM,CAAA;QACd,CAAC;QACD,GAAG,IAAI,CAAC,CAAA;QACR,IAAI,CAAC,GAAG,CAAC,CAAA;QACT,IAAI,CAAC,GAAG,CAAC,CAAA;QACT,IAAI,CAAC,GAAG,QAAQ,GAAG,KAAK,GAAG,CAAC,CAAA;QAC5B,MAAM,IAAI,GAAkC,EAAE,CAAA;QAC9C,OAAO,CAAC,IAAI,KAAK,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC,IAAI,MAAM,CAAC,CAAC,EAAE,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC;YACzD,MAAM,CAAC,GAAG,CAAC,GAAG,MAAM,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;YAC5B,MAAM,CAAC,GAAG,CAAC,GAAG,MAAM,CAAC,GAAG,EAAE,CAAC,CAAC,CAAA;YAC5B,IAAI,CAAC,GAAG,CAAC,GAAG,IAAI,CAAC,MAAM,GAAG,YAAY,EAAE,CAAC;gBACvC,MAAM,IAAI,KAAK,CACb,SAAS,GAAG,IAAI,GAAG,mDAAmD,QAAQ,WAAW,KAAK,0DAA0D,CACzJ,CAAA;YACH,CAAC;YACD,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,CAAU,CAAC,CAAA;QAC5B,CAAC;QACD,OAAO,IAAI,CAAA;IACb,CAAC;CACF"}
|
package/esm/indexFile.d.ts
CHANGED
|
@@ -33,13 +33,19 @@ export interface IndexData {
|
|
|
33
33
|
indices: (refId: number) => RefIndex | undefined;
|
|
34
34
|
maxRefLength: number;
|
|
35
35
|
skipLines?: number;
|
|
36
|
-
maxBinNumber?: number;
|
|
37
36
|
maxBlockSize: number;
|
|
38
37
|
firstDataLine?: VirtualOffset;
|
|
39
38
|
refCount?: number;
|
|
40
39
|
csi?: boolean;
|
|
41
40
|
csiVersion?: number;
|
|
42
|
-
|
|
41
|
+
/**
|
|
42
|
+
* The binning scheme. TBI reports these too, though its own are fixed by the
|
|
43
|
+
* format (minShift 14, depth 5) rather than read from the file: they are the
|
|
44
|
+
* same scheme, and CSI exists to let a file choose other values for them.
|
|
45
|
+
*/
|
|
46
|
+
minShift: number;
|
|
47
|
+
depth: number;
|
|
48
|
+
maxBinNumber: number;
|
|
43
49
|
}
|
|
44
50
|
export default abstract class IndexFile {
|
|
45
51
|
filehandle: GenericFilehandle;
|
|
@@ -48,6 +54,25 @@ export default abstract class IndexFile {
|
|
|
48
54
|
filehandle: GenericFilehandle;
|
|
49
55
|
});
|
|
50
56
|
protected abstract _parse(opts: Options): Promise<IndexData>;
|
|
57
|
+
/**
|
|
58
|
+
* The bins that may overlap [beg, end), as one inclusive [first, last] range
|
|
59
|
+
* per level of the binning scheme.
|
|
60
|
+
*/
|
|
61
|
+
protected abstract reg2bins(beg: number, end: number, indexData: IndexData): (readonly [number, number])[];
|
|
62
|
+
/**
|
|
63
|
+
* The earliest virtual offset any record overlapping `beg` can have, so that
|
|
64
|
+
* chunks ending at or before it can be dropped. TBI reads it off the linear
|
|
65
|
+
* index; CSI, which has none, off the bins' loffsets.
|
|
66
|
+
*/
|
|
67
|
+
protected abstract lowestOffset(ref: RefIndex, beg: number, indexData: IndexData): VirtualOffset | undefined;
|
|
68
|
+
/**
|
|
69
|
+
* The whole index file, decompressed, with a DataView over it. Both .tbi and
|
|
70
|
+
* .csi are bgzf-compressed and read whole, so both parsers start here.
|
|
71
|
+
*/
|
|
72
|
+
protected readIndexBytes(opts: Options): Promise<{
|
|
73
|
+
bytes: Uint8Array<ArrayBufferLike>;
|
|
74
|
+
dataView: DataView<ArrayBufferLike>;
|
|
75
|
+
}>;
|
|
51
76
|
/** @internal */
|
|
52
77
|
lineCount(refName: string, opts?: Options): Promise<number>;
|
|
53
78
|
/** @internal */
|
|
@@ -64,16 +89,28 @@ export default abstract class IndexFile {
|
|
|
64
89
|
format: string;
|
|
65
90
|
maxRefLength: number;
|
|
66
91
|
skipLines?: number;
|
|
67
|
-
maxBinNumber?: number;
|
|
68
92
|
maxBlockSize: number;
|
|
69
93
|
firstDataLine?: VirtualOffset;
|
|
70
94
|
refCount?: number;
|
|
71
95
|
csi?: boolean;
|
|
72
96
|
csiVersion?: number;
|
|
73
|
-
|
|
97
|
+
/**
|
|
98
|
+
* The binning scheme. TBI reports these too, though its own are fixed by the
|
|
99
|
+
* format (minShift 14, depth 5) rather than read from the file: they are the
|
|
100
|
+
* same scheme, and CSI exists to let a file choose other values for them.
|
|
101
|
+
*/
|
|
102
|
+
minShift: number;
|
|
103
|
+
depth: number;
|
|
104
|
+
maxBinNumber: number;
|
|
74
105
|
}>;
|
|
75
|
-
/**
|
|
76
|
-
|
|
106
|
+
/**
|
|
107
|
+
* The chunks of the data file that may hold records overlapping the region.
|
|
108
|
+
* The two index formats differ only in their binning scheme and in where
|
|
109
|
+
* they keep the pruning floor, which is what the two hooks above supply.
|
|
110
|
+
*
|
|
111
|
+
* @internal
|
|
112
|
+
*/
|
|
113
|
+
blocksForRange(refName: string, min: number, max: number, opts?: Options): Promise<Chunk[]>;
|
|
77
114
|
/** @internal */
|
|
78
115
|
parse(opts?: Options): Promise<IndexData>;
|
|
79
116
|
/** @internal */
|
package/esm/indexFile.js
CHANGED
|
@@ -1,9 +1,26 @@
|
|
|
1
|
+
import { unzip } from '@gmod/bgzf-filehandle';
|
|
2
|
+
import { optimizeChunks } from "./util.js";
|
|
1
3
|
export default class IndexFile {
|
|
2
4
|
filehandle;
|
|
3
5
|
parseP;
|
|
4
6
|
constructor({ filehandle }) {
|
|
5
7
|
this.filehandle = filehandle;
|
|
6
8
|
}
|
|
9
|
+
/**
|
|
10
|
+
* The whole index file, decompressed, with a DataView over it. Both .tbi and
|
|
11
|
+
* .csi are bgzf-compressed and read whole, so both parsers start here.
|
|
12
|
+
*/
|
|
13
|
+
async readIndexBytes(opts) {
|
|
14
|
+
const buf = await this.filehandle.readFile({
|
|
15
|
+
signal: opts.signal,
|
|
16
|
+
onProgress: opts.onProgress,
|
|
17
|
+
});
|
|
18
|
+
const bytes = await unzip(buf);
|
|
19
|
+
return {
|
|
20
|
+
bytes,
|
|
21
|
+
dataView: new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength),
|
|
22
|
+
};
|
|
23
|
+
}
|
|
7
24
|
/** @internal */
|
|
8
25
|
async lineCount(refName, opts = {}) {
|
|
9
26
|
const indexData = await this.parse(opts);
|
|
@@ -18,6 +35,37 @@ export default class IndexFile {
|
|
|
18
35
|
const { indices: _indices, ...rest } = await this.parse(opts);
|
|
19
36
|
return rest;
|
|
20
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* The chunks of the data file that may hold records overlapping the region.
|
|
40
|
+
* The two index formats differ only in their binning scheme and in where
|
|
41
|
+
* they keep the pruning floor, which is what the two hooks above supply.
|
|
42
|
+
*
|
|
43
|
+
* @internal
|
|
44
|
+
*/
|
|
45
|
+
async blocksForRange(refName, min, max, opts = {}) {
|
|
46
|
+
const indexData = await this.parse(opts);
|
|
47
|
+
const refId = indexData.refNameToId[refName];
|
|
48
|
+
if (refId === undefined) {
|
|
49
|
+
return [];
|
|
50
|
+
}
|
|
51
|
+
const ba = indexData.indices(refId);
|
|
52
|
+
if (!ba) {
|
|
53
|
+
return [];
|
|
54
|
+
}
|
|
55
|
+
// Find chunks in overlapping bins. Leaf bins are not pruned.
|
|
56
|
+
const chunks = [];
|
|
57
|
+
for (const [start, end] of this.reg2bins(min, max, indexData)) {
|
|
58
|
+
for (let bin = start; bin <= end; bin++) {
|
|
59
|
+
const binChunks = ba.binIndex[bin];
|
|
60
|
+
if (binChunks) {
|
|
61
|
+
for (const c of binChunks) {
|
|
62
|
+
chunks.push(c);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return optimizeChunks(chunks, this.lowestOffset(ba, min, indexData));
|
|
68
|
+
}
|
|
21
69
|
/** @internal */
|
|
22
70
|
async parse(opts = {}) {
|
|
23
71
|
this.parseP ??= this._parse(opts).catch((error) => {
|
package/esm/indexFile.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"indexFile.js","sourceRoot":"","sources":["../src/indexFile.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"indexFile.js","sourceRoot":"","sources":["../src/indexFile.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,KAAK,EAAE,MAAM,uBAAuB,CAAA;AAE7C,OAAO,EAAE,cAAc,EAAE,MAAM,WAAW,CAAA;AAiD1C,MAAM,CAAC,OAAO,OAAgB,SAAS;IAC9B,UAAU,CAAmB;IAC5B,MAAM,CAAqB;IAEnC,YAAY,EAAE,UAAU,EAAqC;QAC3D,IAAI,CAAC,UAAU,GAAG,UAAU,CAAA;IAC9B,CAAC;IAyBD;;;OAGG;IACO,KAAK,CAAC,cAAc,CAAC,IAAa;QAC1C,MAAM,GAAG,GAAG,MAAM,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC;YACzC,MAAM,EAAE,IAAI,CAAC,MAAM;YACnB,UAAU,EAAE,IAAI,CAAC,UAAU;SAC5B,CAAC,CAAA;QACF,MAAM,KAAK,GAAG,MAAM,KAAK,CAAC,GAAG,CAAC,CAAA;QAC9B,OAAO;YACL,KAAK;YACL,QAAQ,EAAE,IAAI,QAAQ,CAAC,KAAK,CAAC,MAAM,EAAE,KAAK,CAAC,UAAU,EAAE,KAAK,CAAC,UAAU,CAAC;SACzE,CAAA;IACH,CAAC;IAED,gBAAgB;IACT,KAAK,CAAC,SAAS,CAAC,OAAe,EAAE,OAAgB,EAAE;QACxD,MAAM,SAAS,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;QACxC,MAAM,KAAK,GAAG,SAAS,CAAC,WAAW,CAAC,OAAO,CAAC,CAAA;QAC5C,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;YACxB,OAAO,CAAC,CAAC,CAAA;QACX,CAAC;QACD,OAAO,SAAS,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,KAAK,EAAE,SAAS,IAAI,CAAC,CAAC,CAAA;IACzD,CAAC;IAED,gBAAgB;IACT,KAAK,CAAC,WAAW,CAAC,OAAgB,EAAE;QACzC,MAAM,EAAE,OAAO,EAAE,QAAQ,EAAE,GAAG,IAAI,EAAE,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;QAC7D,OAAO,IAAI,CAAA;IACb,CAAC;IAED;;;;;;OAMG;IACI,KAAK,CAAC,cAAc,CACzB,OAAe,EACf,GAAW,EACX,GAAW,EACX,OAAgB,EAAE;QAElB,MAAM,SAAS,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;QACxC,MAAM,KAAK,GAAG,SAAS,CAAC,WAAW,CAAC,OAAO,CAAC,CAAA;QAC5C,IAAI,KAAK,KAAK,SAAS,EAAE,CAAC;YACxB,OAAO,EAAE,CAAA;QACX,CAAC;QACD,MAAM,EAAE,GAAG,SAAS,CAAC,OAAO,CAAC,KAAK,CAAC,CAAA;QACnC,IAAI,CAAC,EAAE,EAAE,CAAC;YACR,OAAO,EAAE,CAAA;QACX,CAAC;QAED,6DAA6D;QAC7D,MAAM,MAAM,GAAY,EAAE,CAAA;QAC1B,KAAK,MAAM,CAAC,KAAK,EAAE,GAAG,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,GAAG,EAAE,GAAG,EAAE,SAAS,CAAC,EAAE,CAAC;YAC9D,KAAK,IAAI,GAAG,GAAG,KAAK,EAAE,GAAG,IAAI,GAAG,EAAE,GAAG,EAAE,EAAE,CAAC;gBACxC,MAAM,SAAS,GAAG,EAAE,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAA;gBAClC,IAAI,SAAS,EAAE,CAAC;oBACd,KAAK,MAAM,CAAC,IAAI,SAAS,EAAE,CAAC;wBAC1B,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,CAAA;oBAChB,CAAC;gBACH,CAAC;YACH,CAAC;QACH,CAAC;QAED,OAAO,cAAc,CAAC,MAAM,EAAE,IAAI,CAAC,YAAY,CAAC,EAAE,EAAE,GAAG,EAAE,SAAS,CAAC,CAAC,CAAA;IACtE,CAAC;IAED,gBAAgB;IAChB,KAAK,CAAC,KAAK,CAAC,OAAgB,EAAE;QAC5B,IAAI,CAAC,MAAM,KAAK,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,KAAK,CAAC,CAAC,KAAc,EAAE,EAAE;YACzD,IAAI,CAAC,MAAM,GAAG,SAAS,CAAA;YACvB,MAAM,KAAK,CAAA;QACb,CAAC,CAAC,CAAA;QACF,OAAO,IAAI,CAAC,MAAM,CAAA;IACpB,CAAC;IAED,gBAAgB;IAChB,KAAK,CAAC,SAAS,CAAC,KAAa,EAAE,OAAgB,EAAE;QAC/C,MAAM,GAAG,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;QAClC,OAAO,CAAC,CAAC,GAAG,CAAC,OAAO,CAAC,KAAK,CAAC,EAAE,QAAQ,CAAA;IACvC,CAAC;CACF"}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type Chunk from './chunk.ts';
|
|
2
2
|
import type { Options } from './indexFile.ts';
|
|
3
|
+
import type { ChunkSlice } from '@gmod/bgzf-filehandle';
|
|
3
4
|
import type { GenericFilehandle } from 'generic-filehandle2';
|
|
4
5
|
type GetLinesCallback = (line: string, fileOffset: number, start: number, end: number) => void;
|
|
5
6
|
interface GetLinesOpts {
|
|
@@ -21,6 +22,7 @@ export default class TabixIndexedFile {
|
|
|
21
22
|
private filehandle;
|
|
22
23
|
private index;
|
|
23
24
|
private chunkCache;
|
|
25
|
+
private headerP?;
|
|
24
26
|
constructor({ path, filehandle, url, tbiPath, tbiUrl, tbiFilehandle, csiPath, csiUrl, csiFilehandle, chunkCacheSize, }: {
|
|
25
27
|
path?: string;
|
|
26
28
|
filehandle?: GenericFilehandle;
|
|
@@ -65,14 +67,37 @@ export default class TabixIndexedFile {
|
|
|
65
67
|
format: string;
|
|
66
68
|
maxRefLength: number;
|
|
67
69
|
skipLines?: number;
|
|
68
|
-
maxBinNumber?: number;
|
|
69
70
|
maxBlockSize: number;
|
|
70
71
|
firstDataLine?: import("./virtualOffset.ts").default;
|
|
71
72
|
refCount?: number;
|
|
72
73
|
csi?: boolean;
|
|
73
74
|
csiVersion?: number;
|
|
74
|
-
|
|
75
|
+
minShift: number;
|
|
76
|
+
depth: number;
|
|
77
|
+
maxBinNumber: number;
|
|
75
78
|
}>;
|
|
79
|
+
/**
|
|
80
|
+
* The file's leading blocks, decompressed: everything from the start through
|
|
81
|
+
* the end of the block holding the first data line.
|
|
82
|
+
*/
|
|
83
|
+
private readHeaderBytes;
|
|
84
|
+
/**
|
|
85
|
+
* Both header forms, from one read of the same bytes: the commented block
|
|
86
|
+
* `tabix -H` prints, and the rows `tabix -S N` counted.
|
|
87
|
+
*
|
|
88
|
+
* Memoized, because asking for one and then the other is the normal way to
|
|
89
|
+
* find a header whichever way the file keeps it (see `getHeaderLines`), and
|
|
90
|
+
* that used to fetch and decompress the file's leading blocks twice. Only
|
|
91
|
+
* the parsed results are retained — the decompressed bytes are dropped,
|
|
92
|
+
* which matters for a VCF header that can run to megabytes.
|
|
93
|
+
*/
|
|
94
|
+
private parseHeader;
|
|
95
|
+
private getParsedHeader;
|
|
96
|
+
/**
|
|
97
|
+
* The bytes of the commented header. Deliberately not memoized, unlike the
|
|
98
|
+
* string form: this hands back the buffer, and holding one for the lifetime
|
|
99
|
+
* of the file is the caller's decision to make.
|
|
100
|
+
*/
|
|
76
101
|
getHeaderBuffer(opts?: Options): Promise<Uint8Array<ArrayBufferLike>>;
|
|
77
102
|
/**
|
|
78
103
|
* The leading lines the index says to skip — `tabix -S N` — which is where a
|
|
@@ -91,14 +116,24 @@ export default class TabixIndexedFile {
|
|
|
91
116
|
*/
|
|
92
117
|
getSkippedLines(opts?: Options): Promise<string[]>;
|
|
93
118
|
getHeader(opts?: Options): Promise<string>;
|
|
119
|
+
/**
|
|
120
|
+
* The file's header lines, however that file keeps them: the meta-character
|
|
121
|
+
* block when there is one, and otherwise the rows the index counted. Empty
|
|
122
|
+
* lines are dropped.
|
|
123
|
+
*
|
|
124
|
+
* The two halves answer different questions (see `getSkippedLines`), but
|
|
125
|
+
* "what are this file's header lines" is nearly always the question a caller
|
|
126
|
+
* actually has, and answering it from `getHeader` alone is wrong in a way
|
|
127
|
+
* that doesn't announce itself: a bare header row comes back as the empty
|
|
128
|
+
* string, indistinguishable from a file that has no header, so callers fall
|
|
129
|
+
* back to an assumed column layout and quietly mis-name columns. Deciding it
|
|
130
|
+
* here also means one read of the leading blocks instead of two.
|
|
131
|
+
*/
|
|
132
|
+
getHeaderLines(opts?: Options): Promise<string[]>;
|
|
94
133
|
getReferenceSequenceNames(opts?: Options): Promise<string[]>;
|
|
95
134
|
/** @param refName reference sequence name */
|
|
96
135
|
lineCount(refName: string, opts?: Options): Promise<number>;
|
|
97
136
|
/** @internal */
|
|
98
|
-
readChunk(c: Chunk, opts?: Options): Promise<
|
|
99
|
-
buffer: Uint8Array<ArrayBufferLike>;
|
|
100
|
-
cpositions: number[];
|
|
101
|
-
dpositions: number[];
|
|
102
|
-
}>;
|
|
137
|
+
readChunk(c: Chunk, opts?: Options): Promise<ChunkSlice>;
|
|
103
138
|
}
|
|
104
139
|
export {};
|
package/esm/tabixIndexedFile.js
CHANGED
|
@@ -179,6 +179,46 @@ function getVcfEnd(buffer, startCoordinate, refStart, refEnd, infoStart, infoEnd
|
|
|
179
179
|
}
|
|
180
180
|
return endCoordinate;
|
|
181
181
|
}
|
|
182
|
+
const textDecoder = new TextDecoder();
|
|
183
|
+
/**
|
|
184
|
+
* The leading run of meta-character lines — what `tabix -H` prints. Everything
|
|
185
|
+
* from the first line that doesn't begin with the meta character is dropped.
|
|
186
|
+
*/
|
|
187
|
+
function trimToMetaLines(bytes, metaChar) {
|
|
188
|
+
let lastNewline = -1;
|
|
189
|
+
const metaByte = metaChar.charCodeAt(0);
|
|
190
|
+
for (let i = 0, l = bytes.length; i < l; i++) {
|
|
191
|
+
const byte = bytes[i];
|
|
192
|
+
if (i === lastNewline + 1 && byte !== metaByte) {
|
|
193
|
+
break;
|
|
194
|
+
}
|
|
195
|
+
if (byte === NEWLINE) {
|
|
196
|
+
lastNewline = i;
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
return bytes.subarray(0, lastNewline + 1);
|
|
200
|
+
}
|
|
201
|
+
/**
|
|
202
|
+
* The first `count` lines. Scans for the count-th newline and decodes only that
|
|
203
|
+
* far, rather than decoding and splitting the whole buffer to keep its first
|
|
204
|
+
* few lines — the buffer runs to the first data line, which for a file with a
|
|
205
|
+
* long commented preamble above its counted rows can be megabytes.
|
|
206
|
+
*/
|
|
207
|
+
function firstLines(bytes, count) {
|
|
208
|
+
let end = 0;
|
|
209
|
+
for (let i = 0; i < count; i++) {
|
|
210
|
+
const n = bytes.indexOf(NEWLINE, end);
|
|
211
|
+
if (n === -1) {
|
|
212
|
+
end = bytes.length;
|
|
213
|
+
break;
|
|
214
|
+
}
|
|
215
|
+
end = n + 1;
|
|
216
|
+
}
|
|
217
|
+
return textDecoder
|
|
218
|
+
.decode(bytes.subarray(0, end))
|
|
219
|
+
.split(/\r?\n/)
|
|
220
|
+
.slice(0, count);
|
|
221
|
+
}
|
|
182
222
|
function parseIntFromBytes(buffer, start, end) {
|
|
183
223
|
let val = 0;
|
|
184
224
|
for (let i = start; i < end; i++) {
|
|
@@ -199,6 +239,7 @@ export default class TabixIndexedFile {
|
|
|
199
239
|
filehandle;
|
|
200
240
|
index;
|
|
201
241
|
chunkCache;
|
|
242
|
+
headerP;
|
|
202
243
|
constructor({ path, filehandle, url, tbiPath, tbiUrl, tbiFilehandle, csiPath, csiUrl, csiFilehandle, chunkCacheSize = DEFAULT_CHUNK_CACHE_BYTES, }) {
|
|
203
244
|
this.filehandle = resolveFilehandle(filehandle, path, url);
|
|
204
245
|
this.index = resolveIndex({
|
|
@@ -398,29 +439,52 @@ export default class TabixIndexedFile {
|
|
|
398
439
|
async getMetadata(opts = {}) {
|
|
399
440
|
return this.index.getMetadata(opts);
|
|
400
441
|
}
|
|
401
|
-
|
|
402
|
-
|
|
442
|
+
/**
|
|
443
|
+
* The file's leading blocks, decompressed: everything from the start through
|
|
444
|
+
* the end of the block holding the first data line.
|
|
445
|
+
*/
|
|
446
|
+
async readHeaderBytes(opts) {
|
|
447
|
+
const { firstDataLine, maxBlockSize } = await this.getMetadata(opts);
|
|
403
448
|
const maxFetch = (firstDataLine?.blockPosition ?? 0) + maxBlockSize;
|
|
404
449
|
// TODO: what if we don't have a firstDataLine, and the header actually
|
|
405
450
|
// takes up more than one block? this case is not covered here
|
|
406
451
|
const buf = await this.filehandle.read(maxFetch, 0, opts);
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
452
|
+
return unzip(buf);
|
|
453
|
+
}
|
|
454
|
+
/**
|
|
455
|
+
* Both header forms, from one read of the same bytes: the commented block
|
|
456
|
+
* `tabix -H` prints, and the rows `tabix -S N` counted.
|
|
457
|
+
*
|
|
458
|
+
* Memoized, because asking for one and then the other is the normal way to
|
|
459
|
+
* find a header whichever way the file keeps it (see `getHeaderLines`), and
|
|
460
|
+
* that used to fetch and decompress the file's leading blocks twice. Only
|
|
461
|
+
* the parsed results are retained — the decompressed bytes are dropped,
|
|
462
|
+
* which matters for a VCF header that can run to megabytes.
|
|
463
|
+
*/
|
|
464
|
+
async parseHeader(opts) {
|
|
465
|
+
const { metaChar, skipLines = 0 } = await this.getMetadata(opts);
|
|
466
|
+
const bytes = await this.readHeaderBytes(opts);
|
|
467
|
+
return {
|
|
468
|
+
header: textDecoder.decode(metaChar ? trimToMetaLines(bytes, metaChar) : bytes),
|
|
469
|
+
skippedLines: skipLines > 0 ? firstLines(bytes, skipLines) : [],
|
|
470
|
+
};
|
|
471
|
+
}
|
|
472
|
+
async getParsedHeader(opts = {}) {
|
|
473
|
+
this.headerP ??= this.parseHeader(opts).catch((error) => {
|
|
474
|
+
this.headerP = undefined;
|
|
475
|
+
throw error;
|
|
476
|
+
});
|
|
477
|
+
return this.headerP;
|
|
478
|
+
}
|
|
479
|
+
/**
|
|
480
|
+
* The bytes of the commented header. Deliberately not memoized, unlike the
|
|
481
|
+
* string form: this hands back the buffer, and holding one for the lifetime
|
|
482
|
+
* of the file is the caller's decision to make.
|
|
483
|
+
*/
|
|
484
|
+
async getHeaderBuffer(opts = {}) {
|
|
485
|
+
const { metaChar } = await this.getMetadata(opts);
|
|
486
|
+
const bytes = await this.readHeaderBytes(opts);
|
|
487
|
+
return metaChar ? trimToMetaLines(bytes, metaChar) : bytes;
|
|
424
488
|
}
|
|
425
489
|
/**
|
|
426
490
|
* The leading lines the index says to skip — `tabix -S N` — which is where a
|
|
@@ -438,19 +502,32 @@ export default class TabixIndexedFile {
|
|
|
438
502
|
* way.
|
|
439
503
|
*/
|
|
440
504
|
async getSkippedLines(opts = {}) {
|
|
441
|
-
const {
|
|
505
|
+
const { skipLines = 0 } = await this.getMetadata(opts);
|
|
506
|
+
// the index already answers this without reading the file at all
|
|
442
507
|
if (skipLines <= 0) {
|
|
443
508
|
return [];
|
|
444
509
|
}
|
|
445
|
-
|
|
446
|
-
// more blocks than this is not covered
|
|
447
|
-
const buf = await this.filehandle.read((firstDataLine?.blockPosition ?? 0) + maxBlockSize, 0, opts);
|
|
448
|
-
const bytes = (await unzip(buf));
|
|
449
|
-
return new TextDecoder().decode(bytes).split(/\r?\n/).slice(0, skipLines);
|
|
510
|
+
return (await this.getParsedHeader(opts)).skippedLines;
|
|
450
511
|
}
|
|
451
512
|
async getHeader(opts = {}) {
|
|
452
|
-
|
|
453
|
-
|
|
513
|
+
return (await this.getParsedHeader(opts)).header;
|
|
514
|
+
}
|
|
515
|
+
/**
|
|
516
|
+
* The file's header lines, however that file keeps them: the meta-character
|
|
517
|
+
* block when there is one, and otherwise the rows the index counted. Empty
|
|
518
|
+
* lines are dropped.
|
|
519
|
+
*
|
|
520
|
+
* The two halves answer different questions (see `getSkippedLines`), but
|
|
521
|
+
* "what are this file's header lines" is nearly always the question a caller
|
|
522
|
+
* actually has, and answering it from `getHeader` alone is wrong in a way
|
|
523
|
+
* that doesn't announce itself: a bare header row comes back as the empty
|
|
524
|
+
* string, indistinguishable from a file that has no header, so callers fall
|
|
525
|
+
* back to an assumed column layout and quietly mis-name columns. Deciding it
|
|
526
|
+
* here also means one read of the leading blocks instead of two.
|
|
527
|
+
*/
|
|
528
|
+
async getHeaderLines(opts = {}) {
|
|
529
|
+
const { header, skippedLines } = await this.getParsedHeader(opts);
|
|
530
|
+
return (header ? header.split(/\r?\n/) : skippedLines).filter(Boolean);
|
|
454
531
|
}
|
|
455
532
|
async getReferenceSequenceNames(opts = {}) {
|
|
456
533
|
const metadata = await this.getMetadata(opts);
|