@gmod/tabix 3.5.1 → 3.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -19
- package/dist/csi.js +1 -1
- package/dist/csi.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.d.ts +17 -0
- package/dist/tabixIndexedFile.js +107 -3
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/tbi.js +12 -15
- package/dist/tbi.js.map +1 -1
- package/dist/util.d.ts +11 -2
- package/dist/util.js +31 -5
- package/dist/util.js.map +1 -1
- package/esm/csi.js +2 -2
- package/esm/csi.js.map +1 -1
- package/esm/tabixIndexedFile.d.ts +17 -0
- package/esm/tabixIndexedFile.js +107 -3
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/tbi.js +13 -16
- package/esm/tbi.js.map +1 -1
- package/esm/util.d.ts +11 -2
- package/esm/util.js +30 -4
- package/esm/util.js.map +1 -1
- package/package.json +4 -7
- package/src/csi.ts +2 -2
- package/src/tabixIndexedFile.ts +146 -3
- package/src/tbi.ts +13 -14
- package/src/util.ts +35 -8
package/src/tbi.ts
CHANGED
|
@@ -4,8 +4,8 @@ import Chunk from './chunk.ts'
|
|
|
4
4
|
import IndexFile from './indexFile.ts'
|
|
5
5
|
import {
|
|
6
6
|
clampChunkEnds,
|
|
7
|
-
findFirstData,
|
|
8
7
|
memoizeByRefId,
|
|
8
|
+
minVirtualOffset,
|
|
9
9
|
optimizeChunks,
|
|
10
10
|
parseAuxData,
|
|
11
11
|
parsePseudoBin,
|
|
@@ -78,8 +78,15 @@ export default class TabixIndex extends IndexFile {
|
|
|
78
78
|
// nameSectionLength is at TBI offset 32; re-read to find where bin data starts
|
|
79
79
|
const nameSectionLength = dataView.getInt32(32, true)
|
|
80
80
|
|
|
81
|
-
// SYNC: ~/src/gmod/bam-js/src/
|
|
82
|
-
// First pass: record per-refId byte offsets and find firstDataLine
|
|
81
|
+
// SYNC: ~/src/gmod/bam-js/src/bai.ts _parse — two-pass structure
|
|
82
|
+
// First pass: record per-refId byte offsets and find firstDataLine.
|
|
83
|
+
//
|
|
84
|
+
// Only the linear index is consulted. Its entry for a window is the
|
|
85
|
+
// smallest virtual offset of any record overlapping that window, so the
|
|
86
|
+
// minimum over the linear index is already the minimum over the bin chunks
|
|
87
|
+
// — walking the chunks too only re-derives it. Checked against every .tbi
|
|
88
|
+
// in test/data: same answer on all 23, and no ref has bins without a
|
|
89
|
+
// linear index.
|
|
83
90
|
let curr = 36 + nameSectionLength
|
|
84
91
|
let firstDataLine: VirtualOffset | undefined
|
|
85
92
|
const offsets: number[] = []
|
|
@@ -97,21 +104,13 @@ export default class TabixIndex extends IndexFile {
|
|
|
97
104
|
throw new Error(
|
|
98
105
|
'tabix index contains too many bins, please use a CSI index',
|
|
99
106
|
)
|
|
100
|
-
} else if (bin === maxBinNumber + 1) {
|
|
101
|
-
curr += 16 * chunkCount
|
|
102
|
-
} else {
|
|
103
|
-
for (let k = 0; k < chunkCount; k++) {
|
|
104
|
-
firstDataLine = findFirstData(firstDataLine, fromBytes(bytes, curr))
|
|
105
|
-
curr += 16
|
|
106
|
-
}
|
|
107
107
|
}
|
|
108
|
+
curr += 16 * chunkCount
|
|
108
109
|
}
|
|
109
110
|
const linearCount = dataView.getInt32(curr, true)
|
|
110
111
|
curr += 4
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
curr += 8
|
|
114
|
-
}
|
|
112
|
+
firstDataLine = minVirtualOffset(bytes, curr, linearCount, firstDataLine)
|
|
113
|
+
curr += 8 * linearCount
|
|
115
114
|
}
|
|
116
115
|
|
|
117
116
|
function getIndices(refId: number): RefIndex | undefined {
|
package/src/util.ts
CHANGED
|
@@ -2,8 +2,7 @@ import LRU from '@jbrowse/quick-lru'
|
|
|
2
2
|
|
|
3
3
|
import Chunk from './chunk.ts'
|
|
4
4
|
import { longFromBytesToUnsigned } from './long.ts'
|
|
5
|
-
|
|
6
|
-
import type VirtualOffset from './virtualOffset.ts'
|
|
5
|
+
import VirtualOffset from './virtualOffset.ts'
|
|
7
6
|
|
|
8
7
|
// SYNC: ~/src/gmod/bam-js/src/util.ts optimizeChunks
|
|
9
8
|
export function optimizeChunks(chunks: Chunk[], lowest?: VirtualOffset) {
|
|
@@ -118,13 +117,41 @@ export function clampChunkEnds(
|
|
|
118
117
|
}
|
|
119
118
|
}
|
|
120
119
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
120
|
+
/**
|
|
121
|
+
* The smallest of `current` and the `count` packed virtual offsets starting at
|
|
122
|
+
* `offset`, allocating at most one VirtualOffset rather than one per entry.
|
|
123
|
+
*
|
|
124
|
+
* The index first pass exists only to find this minimum, and it visits every
|
|
125
|
+
* linear-index entry in the file to do it — 301k of them on
|
|
126
|
+
* test/data/failing_tabix.vcf.gz.tbi against 19k bin chunks. Building a
|
|
127
|
+
* VirtualOffset per entry to compare and discard it is the bulk of that pass.
|
|
128
|
+
*/
|
|
129
|
+
export function minVirtualOffset(
|
|
130
|
+
bytes: Uint8Array,
|
|
131
|
+
offset: number,
|
|
132
|
+
count: number,
|
|
133
|
+
current: VirtualOffset | undefined,
|
|
124
134
|
) {
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
135
|
+
let minBlock = current ? current.blockPosition : Infinity
|
|
136
|
+
let minData = current ? current.dataPosition : 0
|
|
137
|
+
let found = false
|
|
138
|
+
for (let i = 0; i < count; i++) {
|
|
139
|
+
const p = offset + i * 8
|
|
140
|
+
const block =
|
|
141
|
+
bytes[p + 7]! * 0x1_00_00_00_00_00 +
|
|
142
|
+
bytes[p + 6]! * 0x1_00_00_00_00 +
|
|
143
|
+
bytes[p + 5]! * 0x1_00_00_00 +
|
|
144
|
+
bytes[p + 4]! * 0x1_00_00 +
|
|
145
|
+
bytes[p + 3]! * 0x1_00 +
|
|
146
|
+
bytes[p + 2]!
|
|
147
|
+
const data = (bytes[p + 1]! << 8) | bytes[p]!
|
|
148
|
+
if (block < minBlock || (block === minBlock && data < minData)) {
|
|
149
|
+
minBlock = block
|
|
150
|
+
minData = data
|
|
151
|
+
found = true
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
return found ? new VirtualOffset(minBlock, minData) : current
|
|
128
155
|
}
|
|
129
156
|
|
|
130
157
|
export function parseNameBytes(namesBytes: Uint8Array) {
|