@gmod/tabix 3.5.3 → 3.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,6 +9,7 @@ import { optimizeChunks } from './util.ts'
9
9
  import type Chunk from './chunk.ts'
10
10
  import type IndexFile from './indexFile.ts'
11
11
  import type { Options } from './indexFile.ts'
12
+ import type { ChunkSlice } from '@gmod/bgzf-filehandle'
12
13
  import type { GenericFilehandle } from 'generic-filehandle2'
13
14
 
14
15
  const TAB = 9
@@ -148,11 +149,11 @@ interface GetLinesOpts {
148
149
  onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
149
150
  }
150
151
 
151
- interface ReadChunk {
152
- buffer: Uint8Array
153
- cpositions: number[]
154
- dpositions: number[]
155
- }
152
+ // The decompressed chunk plus its block offsets, as @gmod/bgzf-filehandle
153
+ // returns them: cpositions/dpositions are Float64Arrays, which is what the
154
+ // wasm decompressor produces, and only indexed reads and `.length` are used
155
+ // here. Taken from the package rather than restated so the two can't drift.
156
+ type ReadChunk = ChunkSlice
156
157
 
157
158
  function resolveFilehandle(
158
159
  filehandle?: GenericFilehandle,
@@ -222,8 +223,8 @@ function resolveIndex({
222
223
  }
223
224
 
224
225
  function calculateFileOffset(
225
- cpositions: number[],
226
- dpositions: number[],
226
+ cpositions: ArrayLike<number>,
227
+ dpositions: ArrayLike<number>,
227
228
  pos: number,
228
229
  blockStart: number,
229
230
  minvDataPosition: number,
@@ -289,6 +290,50 @@ function getVcfEnd(
289
290
  return endCoordinate
290
291
  }
291
292
 
293
+ const textDecoder = new TextDecoder()
294
+
295
+ /**
296
+ * The leading run of meta-character lines — what `tabix -H` prints. Everything
297
+ * from the first line that doesn't begin with the meta character is dropped.
298
+ */
299
+ function trimToMetaLines(bytes: Uint8Array, metaChar: string) {
300
+ let lastNewline = -1
301
+ const metaByte = metaChar.charCodeAt(0)
302
+
303
+ for (let i = 0, l = bytes.length; i < l; i++) {
304
+ const byte = bytes[i]
305
+ if (i === lastNewline + 1 && byte !== metaByte) {
306
+ break
307
+ }
308
+ if (byte === NEWLINE) {
309
+ lastNewline = i
310
+ }
311
+ }
312
+ return bytes.subarray(0, lastNewline + 1)
313
+ }
314
+
315
+ /**
316
+ * The first `count` lines. Scans for the count-th newline and decodes only that
317
+ * far, rather than decoding and splitting the whole buffer to keep its first
318
+ * few lines — the buffer runs to the first data line, which for a file with a
319
+ * long commented preamble above its counted rows can be megabytes.
320
+ */
321
+ function firstLines(bytes: Uint8Array, count: number) {
322
+ let end = 0
323
+ for (let i = 0; i < count; i++) {
324
+ const n = bytes.indexOf(NEWLINE, end)
325
+ if (n === -1) {
326
+ end = bytes.length
327
+ break
328
+ }
329
+ end = n + 1
330
+ }
331
+ return textDecoder
332
+ .decode(bytes.subarray(0, end))
333
+ .split(/\r?\n/)
334
+ .slice(0, count)
335
+ }
336
+
292
337
  function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
293
338
  let val = 0
294
339
  for (let i = start; i < end; i++) {
@@ -309,6 +354,7 @@ export default class TabixIndexedFile {
309
354
  private filehandle: GenericFilehandle
310
355
  private index: IndexFile
311
356
  private chunkCache: AbortablePromiseCache<Chunk, ReadChunk>
357
+ private headerP?: Promise<{ header: string; skippedLines: string[] }>
312
358
 
313
359
  constructor({
314
360
  path,
@@ -588,34 +634,59 @@ export default class TabixIndexedFile {
588
634
  return this.index.getMetadata(opts)
589
635
  }
590
636
 
591
- async getHeaderBuffer(opts: Options = {}) {
592
- const { firstDataLine, metaChar, maxBlockSize } =
593
- await this.getMetadata(opts)
637
+ /**
638
+ * The file's leading blocks, decompressed: everything from the start through
639
+ * the end of the block holding the first data line.
640
+ */
641
+ private async readHeaderBytes(opts: Options) {
642
+ const { firstDataLine, maxBlockSize } = await this.getMetadata(opts)
594
643
 
595
644
  const maxFetch = (firstDataLine?.blockPosition ?? 0) + maxBlockSize
596
645
  // TODO: what if we don't have a firstDataLine, and the header actually
597
646
  // takes up more than one block? this case is not covered here
598
647
 
599
648
  const buf = await this.filehandle.read(maxFetch, 0, opts)
600
- const bytes = (await unzip(buf)) as Uint8Array
601
-
602
- // trim off lines after the last meta line
603
- if (metaChar) {
604
- let lastNewline = -1
605
- const metaByte = metaChar.charCodeAt(0)
649
+ return unzip(buf)
650
+ }
606
651
 
607
- for (let i = 0, l = bytes.length; i < l; i++) {
608
- const byte = bytes[i]
609
- if (i === lastNewline + 1 && byte !== metaByte) {
610
- break
611
- }
612
- if (byte === NEWLINE) {
613
- lastNewline = i
614
- }
615
- }
616
- return bytes.subarray(0, lastNewline + 1)
652
+ /**
653
+ * Both header forms, from one read of the same bytes: the commented block
654
+ * `tabix -H` prints, and the rows `tabix -S N` counted.
655
+ *
656
+ * Memoized, because asking for one and then the other is the normal way to
657
+ * find a header whichever way the file keeps it (see `getHeaderLines`), and
658
+ * that used to fetch and decompress the file's leading blocks twice. Only
659
+ * the parsed results are retained — the decompressed bytes are dropped,
660
+ * which matters for a VCF header that can run to megabytes.
661
+ */
662
+ private async parseHeader(opts: Options) {
663
+ const { metaChar, skipLines = 0 } = await this.getMetadata(opts)
664
+ const bytes = await this.readHeaderBytes(opts)
665
+ return {
666
+ header: textDecoder.decode(
667
+ metaChar ? trimToMetaLines(bytes, metaChar) : bytes,
668
+ ),
669
+ skippedLines: skipLines > 0 ? firstLines(bytes, skipLines) : [],
617
670
  }
618
- return bytes
671
+ }
672
+
673
+ private async getParsedHeader(opts: Options = {}) {
674
+ this.headerP ??= this.parseHeader(opts).catch((error: unknown) => {
675
+ this.headerP = undefined
676
+ throw error
677
+ })
678
+ return this.headerP
679
+ }
680
+
681
+ /**
682
+ * The bytes of the commented header. Deliberately not memoized, unlike the
683
+ * string form: this hands back the buffer, and holding one for the lifetime
684
+ * of the file is the caller's decision to make.
685
+ */
686
+ async getHeaderBuffer(opts: Options = {}) {
687
+ const { metaChar } = await this.getMetadata(opts)
688
+ const bytes = await this.readHeaderBytes(opts)
689
+ return metaChar ? trimToMetaLines(bytes, metaChar) : bytes
619
690
  }
620
691
 
621
692
  /**
@@ -634,29 +705,34 @@ export default class TabixIndexedFile {
634
705
  * way.
635
706
  */
636
707
  async getSkippedLines(opts: Options = {}) {
637
- const {
638
- firstDataLine,
639
- skipLines = 0,
640
- maxBlockSize,
641
- } = await this.getMetadata(opts)
708
+ const { skipLines = 0 } = await this.getMetadata(opts)
709
+ // the index already answers this without reading the file at all
642
710
  if (skipLines <= 0) {
643
711
  return []
644
712
  }
645
-
646
- // same read getHeaderBuffer makes, and the same caveat: a header spanning
647
- // more blocks than this is not covered
648
- const buf = await this.filehandle.read(
649
- (firstDataLine?.blockPosition ?? 0) + maxBlockSize,
650
- 0,
651
- opts,
652
- )
653
- const bytes = (await unzip(buf)) as Uint8Array
654
- return new TextDecoder().decode(bytes).split(/\r?\n/).slice(0, skipLines)
713
+ return (await this.getParsedHeader(opts)).skippedLines
655
714
  }
656
715
 
657
716
  async getHeader(opts: Options = {}) {
658
- const bytes = await this.getHeaderBuffer(opts)
659
- return new TextDecoder().decode(bytes)
717
+ return (await this.getParsedHeader(opts)).header
718
+ }
719
+
720
+ /**
721
+ * The file's header lines, however that file keeps them: the meta-character
722
+ * block when there is one, and otherwise the rows the index counted. Empty
723
+ * lines are dropped.
724
+ *
725
+ * The two halves answer different questions (see `getSkippedLines`), but
726
+ * "what are this file's header lines" is nearly always the question a caller
727
+ * actually has, and answering it from `getHeader` alone is wrong in a way
728
+ * that doesn't announce itself: a bare header row comes back as the empty
729
+ * string, indistinguishable from a file that has no header, so callers fall
730
+ * back to an assumed column layout and quietly mis-name columns. Deciding it
731
+ * here also means one read of the leading blocks instead of two.
732
+ */
733
+ async getHeaderLines(opts: Options = {}) {
734
+ const { header, skippedLines } = await this.getParsedHeader(opts)
735
+ return (header ? header.split(/\r?\n/) : skippedLines).filter(Boolean)
660
736
  }
661
737
 
662
738
  async getReferenceSequenceNames(opts: Options = {}) {
package/src/tbi.ts CHANGED
@@ -1,12 +1,9 @@
1
- import { unzip } from '@gmod/bgzf-filehandle'
2
-
3
1
  import Chunk from './chunk.ts'
4
2
  import IndexFile from './indexFile.ts'
5
3
  import {
6
4
  clampChunkEnds,
7
5
  memoizeByRefId,
8
6
  minVirtualOffset,
9
- optimizeChunks,
10
7
  parseAuxData,
11
8
  parsePseudoBin,
12
9
  } from './util.ts'
@@ -19,44 +16,57 @@ const TBI_MAGIC = 21_578_324 // TBI\1
19
16
 
20
17
  // TBI is always the fixed depth-5, 14-bit-leaf binning scheme
21
18
  const TBI_DEPTH = 5
22
- const TBI_MAX_REF_LENGTH = 2 ** (14 + TBI_DEPTH * 3)
23
-
24
- /**
25
- * calculate the list of bins that may overlap with region [beg,end)
26
- * (zero-based half-open)
27
- */
28
- function reg2bins(beg: number, end: number) {
29
- // Clamp into the binning scheme's range. Past 2**31 the shifts below
30
- // truncate to int32 and go negative, which would silently yield no bins.
31
- const b = Math.min(Math.max(beg, 0), TBI_MAX_REF_LENGTH)
32
- const e = Math.min(end, TBI_MAX_REF_LENGTH) - 1
33
- const bins: (readonly [number, number])[] = []
34
- if (b <= e) {
35
- bins.push(
36
- [0, 0],
37
- [1 + (b >> 26), 1 + (e >> 26)],
38
- [9 + (b >> 23), 9 + (e >> 23)],
39
- [73 + (b >> 20), 73 + (e >> 20)],
40
- [585 + (b >> 17), 585 + (e >> 17)],
41
- [4681 + (b >> 14), 4681 + (e >> 14)],
42
- )
43
- }
44
- return bins
19
+ const TBI_MIN_SHIFT = 14
20
+ const TBI_MAX_REF_LENGTH = 2 ** (TBI_MIN_SHIFT + TBI_DEPTH * 3)
21
+
22
+ // Clamp a query coordinate into the binning scheme's range. Past 2**31 the
23
+ // int32 shifts below truncate and go negative, which would silently yield no
24
+ // bins.
25
+ function clampToRefLength(pos: number) {
26
+ return Math.min(Math.max(pos, 0), TBI_MAX_REF_LENGTH)
45
27
  }
46
28
 
47
29
  export default class TabixIndex extends IndexFile {
30
+ /**
31
+ * The bins that may overlap [beg, end) (zero-based half-open). TBI's scheme
32
+ * is CSI's with minShift/depth fixed, so this is CSI's reg2bins with the
33
+ * levels unrolled and its float shifts replaced by int32 ones.
34
+ */
35
+ protected reg2bins(beg: number, end: number) {
36
+ if (beg > TBI_MAX_REF_LENGTH) {
37
+ console.warn('querying outside of possible tabix range')
38
+ }
39
+ const b = clampToRefLength(beg)
40
+ const e = Math.min(end, TBI_MAX_REF_LENGTH) - 1
41
+ const bins: (readonly [number, number])[] = []
42
+ if (b <= e) {
43
+ bins.push(
44
+ [0, 0],
45
+ [1 + (b >> 26), 1 + (e >> 26)],
46
+ [9 + (b >> 23), 9 + (e >> 23)],
47
+ [73 + (b >> 20), 73 + (e >> 20)],
48
+ [585 + (b >> 17), 585 + (e >> 17)],
49
+ [4681 + (b >> 14), 4681 + (e >> 14)],
50
+ )
51
+ }
52
+ return bins
53
+ }
54
+
55
+ /**
56
+ * The linear index is monotonically non-decreasing, so the minimum virtual
57
+ * offset for chunks that could overlap [beg, ...) is its entry for beg.
58
+ * SYNC: ~/src/gmod/bam-js/src/bai.ts getLowestChunk
59
+ */
60
+ protected lowestOffset(ref: RefIndex, beg: number) {
61
+ const linearIndex = ref.linearIndex
62
+ return linearIndex?.[
63
+ Math.min(clampToRefLength(beg) >> TBI_MIN_SHIFT, linearIndex.length - 1)
64
+ ]
65
+ }
66
+
48
67
  /** @internal */
49
68
  async _parse(opts: Options = {}) {
50
- const buf = await this.filehandle.readFile({
51
- signal: opts.signal,
52
- onProgress: opts.onProgress,
53
- })
54
- const bytes = (await unzip(buf)) as Uint8Array
55
- const dataView = new DataView(
56
- bytes.buffer,
57
- bytes.byteOffset,
58
- bytes.byteLength,
59
- )
69
+ const { bytes, dataView } = await this.readIndexBytes(opts)
60
70
 
61
71
  if (dataView.getUint32(0, true) !== TBI_MAGIC) {
62
72
  throw new Error('Not a TBI file')
@@ -169,6 +179,8 @@ export default class TabixIndex extends IndexFile {
169
179
  return {
170
180
  indices: memoizeByRefId(getIndices),
171
181
  metaChar,
182
+ minShift: TBI_MIN_SHIFT,
183
+ depth: TBI_DEPTH,
172
184
  maxBinNumber,
173
185
  maxRefLength: TBI_MAX_REF_LENGTH,
174
186
  skipLines,
@@ -181,50 +193,4 @@ export default class TabixIndex extends IndexFile {
181
193
  maxBlockSize: 1 << 16,
182
194
  }
183
195
  }
184
-
185
- async blocksForRange(
186
- refName: string,
187
- min: number,
188
- max: number,
189
- opts: Options = {},
190
- ) {
191
- const indexData = await this.parse(opts)
192
- const refId = indexData.refNameToId[refName]
193
- if (refId === undefined) {
194
- return []
195
- }
196
- const ba = indexData.indices(refId)
197
- if (!ba) {
198
- return []
199
- }
200
-
201
- if (min > TBI_MAX_REF_LENGTH) {
202
- console.warn('querying outside of possible tabix range')
203
- }
204
- // clamp before the >> 14 below, which would truncate to int32 and go
205
- // negative past 2**31
206
- const beg = Math.min(Math.max(min, 0), TBI_MAX_REF_LENGTH)
207
- const overlappingBins = reg2bins(min, max) // List of bin #s that overlap min, max
208
- const chunks: Chunk[] = []
209
-
210
- // Find chunks in overlapping bins. Leaf bins (< 4681) are not pruned
211
- for (const [start, end] of overlappingBins) {
212
- for (let bin = start; bin <= end; bin++) {
213
- const binChunks = ba.binIndex[bin]
214
- if (binChunks) {
215
- for (const c of binChunks) {
216
- chunks.push(c)
217
- }
218
- }
219
- }
220
- }
221
-
222
- // The linear index is monotonically non-decreasing, so the minimum virtual
223
- // offset for chunks that could overlap [min, ...) is at index minLin.
224
- // SYNC: ~/src/gmod/bam-js/src/bai.ts getLowestChunk
225
- const linearIndex = ba.linearIndex
226
- const lowest = linearIndex?.[Math.min(beg >> 14, linearIndex.length - 1)]
227
-
228
- return optimizeChunks(chunks, lowest)
229
- }
230
196
  }