@gmod/tabix 3.5.7 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -12
- package/dist/indexFile.d.ts +5 -5
- package/dist/indexFile.js +16 -23
- package/dist/indexFile.js.map +1 -1
- package/dist/tabix-bundle.js +1 -1
- package/dist/tabixIndexedFile.d.ts +54 -3
- package/dist/tabixIndexedFile.js +102 -83
- package/dist/tabixIndexedFile.js.map +1 -1
- package/dist/util.d.ts +1 -18
- package/dist/util.js +6 -31
- package/dist/util.js.map +1 -1
- package/esm/indexFile.d.ts +5 -5
- package/esm/indexFile.js +16 -23
- package/esm/indexFile.js.map +1 -1
- package/esm/tabixIndexedFile.d.ts +54 -3
- package/esm/tabixIndexedFile.js +103 -84
- package/esm/tabixIndexedFile.js.map +1 -1
- package/esm/util.d.ts +1 -18
- package/esm/util.js +4 -30
- package/esm/util.js.map +1 -1
- package/package.json +11 -9
- package/src/indexFile.ts +21 -24
- package/src/tabixIndexedFile.ts +130 -113
- package/src/util.ts +4 -30
package/src/tabixIndexedFile.ts
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import AbortablePromiseCache from '@gmod/abortable-promise-cache'
|
|
2
1
|
import { unzip, unzipChunkSlice } from '@gmod/bgzf-filehandle'
|
|
2
|
+
import { SharedReadCache } from '@gmod/shared-read-cache'
|
|
3
3
|
import { LocalFile, RemoteFile } from 'generic-filehandle2'
|
|
4
4
|
|
|
5
5
|
import CSI from './csi.ts'
|
|
6
6
|
import TBI from './tbi.ts'
|
|
7
|
-
import { optimizeChunks } from './util.ts'
|
|
7
|
+
import { optimizeChunks, throwIfAborted } from './util.ts'
|
|
8
8
|
|
|
9
9
|
import type Chunk from './chunk.ts'
|
|
10
10
|
import type IndexFile from './indexFile.ts'
|
|
@@ -29,105 +29,28 @@ const MAX_READ_AHEAD_CHUNKS = 6
|
|
|
29
29
|
// a dense VCF (test/data/1kg.chr1.subset.vcf.gz — 213MB over 600kb of chr1)
|
|
30
30
|
// has single index bins of 17MB compressed, 120MB decompressed. Panning it
|
|
31
31
|
// under the old 80-entry cache peaked at 2GB RSS.
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
//
|
|
35
|
-
//
|
|
36
|
-
//
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
interface CachedChunk {
|
|
46
|
-
entry: CacheEntry
|
|
47
|
-
/** 0 until the read settles and the decompressed size is known */
|
|
48
|
-
bytes: number
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
/**
|
|
52
|
-
* Backing store for the chunk cache, bounded by the decompressed size of the
|
|
53
|
-
* chunks it holds rather than by entry count.
|
|
54
|
-
*
|
|
55
|
-
* A chunk's size is only known once its read settles, so `set` records the
|
|
56
|
-
* entry immediately and charges it to the budget later. Unsettled entries are
|
|
57
|
-
* therefore free, which is what we want: they are reads a query is waiting on.
|
|
58
|
-
*/
|
|
59
|
-
class ByteBoundedChunkCache {
|
|
60
|
-
private entries = new Map<string, CachedChunk>()
|
|
61
|
-
private bytes = 0
|
|
62
|
-
private maxBytes: number
|
|
63
|
-
|
|
64
|
-
constructor(maxBytes: number) {
|
|
65
|
-
this.maxBytes = maxBytes
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
get byteSize() {
|
|
69
|
-
return this.bytes
|
|
70
|
-
}
|
|
71
|
-
|
|
72
|
-
get size() {
|
|
73
|
-
return this.entries.size
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
has(key: string) {
|
|
77
|
-
return this.entries.has(key)
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
get(key: string) {
|
|
81
|
-
const cached = this.entries.get(key)
|
|
82
|
-
if (cached) {
|
|
83
|
-
// re-insert so Map iteration order stays least-recently-used first
|
|
84
|
-
this.entries.delete(key)
|
|
85
|
-
this.entries.set(key, cached)
|
|
86
|
-
}
|
|
87
|
-
return cached?.entry
|
|
88
|
-
}
|
|
89
|
-
|
|
90
|
-
set(key: string, entry: CacheEntry) {
|
|
91
|
-
this.delete(key)
|
|
92
|
-
const cached = { entry, bytes: 0 }
|
|
93
|
-
this.entries.set(key, cached)
|
|
94
|
-
void entry.promise
|
|
95
|
-
.then(chunk => {
|
|
96
|
-
// a later set() may have replaced this key while the read was in
|
|
97
|
-
// flight; charging these bytes to it would then double-count
|
|
98
|
-
if (this.entries.get(key) === cached) {
|
|
99
|
-
cached.bytes = chunk.buffer.byteLength
|
|
100
|
-
this.bytes += cached.bytes
|
|
101
|
-
this.evict()
|
|
102
|
-
}
|
|
103
|
-
})
|
|
104
|
-
.catch(() => {
|
|
105
|
-
// a failed or aborted read caches nothing and costs nothing
|
|
106
|
-
})
|
|
107
|
-
}
|
|
108
|
-
|
|
109
|
-
delete(key: string) {
|
|
110
|
-
const cached = this.entries.get(key)
|
|
111
|
-
if (cached) {
|
|
112
|
-
this.entries.delete(key)
|
|
113
|
-
this.bytes -= cached.bytes
|
|
114
|
-
}
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
keys() {
|
|
118
|
-
return this.entries.keys()
|
|
119
|
-
}
|
|
32
|
+
//
|
|
33
|
+
// That 120MB figure is also why this is no longer 100MB. A budget below one
|
|
34
|
+
// query's working set does not cache less, it caches NOTHING: each entry is
|
|
35
|
+
// evicted before the next pan can reuse it, so the hit rate is zero and the
|
|
36
|
+
// decompress is paid again every time. On that same fixture, a six-window 50kb
|
|
37
|
+
// pan measured 17 refills out of 17 — a total miss — at 100MB, against 0 at
|
|
38
|
+
// 800MB, and 2596ms against 600ms. The working set plateaus at 497MB held, so
|
|
39
|
+
// 1GB clears it with headroom and nothing above 800MB buys anything.
|
|
40
|
+
//
|
|
41
|
+
// Affordable as a ceiling only because of the idle timeout below: it is a peak
|
|
42
|
+
// under panning, not a level a parked consumer holds. A small file is
|
|
43
|
+
// unaffected either way, this being a ceiling and not an allocation.
|
|
44
|
+
const DEFAULT_CHUNK_CACHE_BYTES = 1024 * 2 ** 20
|
|
120
45
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
}
|
|
130
|
-
}
|
|
46
|
+
// SYNC: ~/src/gmod/bam-js/src/bamFile.ts DEFAULT_CACHE_IDLE_TIMEOUT_MS
|
|
47
|
+
//
|
|
48
|
+
// Drop a chunk nothing has looked at for three minutes. The budget above is
|
|
49
|
+
// only applied when a read settles, so it does nothing at all for a consumer
|
|
50
|
+
// sitting still — and a genome browser holds one of these per track for as long
|
|
51
|
+
// as the track is open. Timed from the last read, not the fetch, so panning
|
|
52
|
+
// back and forth over one region never expires it.
|
|
53
|
+
const DEFAULT_CHUNK_CACHE_IDLE_TIMEOUT_MS = 3 * 60 * 1000
|
|
131
54
|
|
|
132
55
|
type GetLinesCallback = (
|
|
133
56
|
line: string,
|
|
@@ -353,8 +276,15 @@ function parseIntFromBytes(buffer: Uint8Array, start: number, end: number) {
|
|
|
353
276
|
export default class TabixIndexedFile {
|
|
354
277
|
private filehandle: GenericFilehandle
|
|
355
278
|
private index: IndexFile
|
|
356
|
-
|
|
279
|
+
public chunkCache: SharedReadCache<Chunk, ReadChunk>
|
|
357
280
|
private headerP?: Promise<{ header: string; skippedLines: string[] }>
|
|
281
|
+
/**
|
|
282
|
+
* The signal `headerP` was started under, while it is still in flight. The
|
|
283
|
+
* header is parsed once and shared by every caller, so without this the first
|
|
284
|
+
* one to arrive would own a read all the others depend on — see
|
|
285
|
+
* {@link getParsedHeader}.
|
|
286
|
+
*/
|
|
287
|
+
private headerSignal?: AbortSignal
|
|
358
288
|
|
|
359
289
|
constructor({
|
|
360
290
|
path,
|
|
@@ -367,6 +297,7 @@ export default class TabixIndexedFile {
|
|
|
367
297
|
csiUrl,
|
|
368
298
|
csiFilehandle,
|
|
369
299
|
chunkCacheSize = DEFAULT_CHUNK_CACHE_BYTES,
|
|
300
|
+
chunkCacheIdleTimeoutMs = DEFAULT_CHUNK_CACHE_IDLE_TIMEOUT_MS,
|
|
370
301
|
}: {
|
|
371
302
|
path?: string
|
|
372
303
|
filehandle?: GenericFilehandle
|
|
@@ -377,8 +308,24 @@ export default class TabixIndexedFile {
|
|
|
377
308
|
csiPath?: string
|
|
378
309
|
csiUrl?: string
|
|
379
310
|
csiFilehandle?: GenericFilehandle
|
|
380
|
-
/**
|
|
311
|
+
/**
|
|
312
|
+
* Budget for the decompressed chunk cache, in bytes. Default 1GB.
|
|
313
|
+
*
|
|
314
|
+
* A retention bound, not a bound on peak memory: reads in flight are never
|
|
315
|
+
* evicted and the last settled entry is kept whatever the budget. Size it
|
|
316
|
+
* to hold several queries — below one query's working set the hit rate
|
|
317
|
+
* drops to zero while the memory is retained anyway, so a number between
|
|
318
|
+
* the two is the worst available choice.
|
|
319
|
+
*/
|
|
381
320
|
chunkCacheSize?: number
|
|
321
|
+
/**
|
|
322
|
+
* Drop a cached chunk once nothing has read it for this many milliseconds.
|
|
323
|
+
* Default 3 minutes; `0` keeps chunks until `chunkCacheSize` evicts them.
|
|
324
|
+
*
|
|
325
|
+
* The only thing that lowers the cache while nothing is happening, and what
|
|
326
|
+
* makes the budget above a peak rather than a resting level.
|
|
327
|
+
*/
|
|
328
|
+
chunkCacheIdleTimeoutMs?: number
|
|
382
329
|
}) {
|
|
383
330
|
this.filehandle = resolveFilehandle(filehandle, path, url)
|
|
384
331
|
this.index = resolveIndex({
|
|
@@ -392,13 +339,29 @@ export default class TabixIndexedFile {
|
|
|
392
339
|
url,
|
|
393
340
|
})
|
|
394
341
|
|
|
395
|
-
this.chunkCache = new
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
342
|
+
this.chunkCache = new SharedReadCache<Chunk, ReadChunk>({
|
|
343
|
+
maxSize: chunkCacheSize,
|
|
344
|
+
// decompressed bytes, not entry count: we fetch compressed and cache
|
|
345
|
+
// decompressed, and an entry is a whole chunk
|
|
346
|
+
sizeOf: read => read.buffer.byteLength,
|
|
347
|
+
cacheKey: chunk => chunk.toString(),
|
|
348
|
+
idleTimeoutMs: chunkCacheIdleTimeoutMs,
|
|
349
|
+
fill: (chunk, signal) => this.readChunk(chunk, { signal }),
|
|
399
350
|
})
|
|
400
351
|
}
|
|
401
352
|
|
|
353
|
+
/**
|
|
354
|
+
* Drops every decompressed chunk held by the cache, and stops the idle sweep
|
|
355
|
+
* until something is cached again.
|
|
356
|
+
*
|
|
357
|
+
* `chunkCacheIdleTimeoutMs` reclaims a view the user has wandered away from;
|
|
358
|
+
* this is for a consumer that knows it is finished — a closed track, a
|
|
359
|
+
* changed assembly — and should not have to wait it out.
|
|
360
|
+
*/
|
|
361
|
+
clearChunkCache() {
|
|
362
|
+
this.chunkCache.clear()
|
|
363
|
+
}
|
|
364
|
+
|
|
402
365
|
/**
|
|
403
366
|
* Estimates the compressed byte size of the index chunks covering the given
|
|
404
367
|
* regions. Useful for byte budgeting before issuing a `getLines` call to
|
|
@@ -505,7 +468,7 @@ export default class TabixIndexedFile {
|
|
|
505
468
|
const ensureReadsStarted = (count: number) => {
|
|
506
469
|
while (reads.length < Math.min(count, chunks.length)) {
|
|
507
470
|
const c = chunks[reads.length]!
|
|
508
|
-
const read = this.chunkCache.get(c
|
|
471
|
+
const read = this.chunkCache.get(c, signal)
|
|
509
472
|
void read.catch(() => {
|
|
510
473
|
// a prefetch the early return skips is never awaited, so swallow its
|
|
511
474
|
// rejection here rather than let it surface unhandled
|
|
@@ -670,12 +633,66 @@ export default class TabixIndexedFile {
|
|
|
670
633
|
}
|
|
671
634
|
}
|
|
672
635
|
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
636
|
+
/**
|
|
637
|
+
* Parse the header, or join the parse already running.
|
|
638
|
+
*
|
|
639
|
+
* `parseHeader` threads `opts` into both `getMetadata` and the header read,
|
|
640
|
+
* so memoizing it on the first caller's opts put that caller's signal in
|
|
641
|
+
* charge of a read every later caller joins: when it aborted, they failed
|
|
642
|
+
* with its cancellation, their own signals untouched.
|
|
643
|
+
*
|
|
644
|
+
* A caller that joined someone else's parse and saw it fail because *they*
|
|
645
|
+
* aborted starts over rather than inheriting the failure — once, then
|
|
646
|
+
* propagates. Same bounded retry, and the same reasoning, as
|
|
647
|
+
* `IndexFile.parse`.
|
|
648
|
+
*
|
|
649
|
+
* SYNC: ~/src/gmod/bam-js/src/bamFile.ts getHeader — same shape for the same
|
|
650
|
+
* reason, on the header rather than the index.
|
|
651
|
+
*/
|
|
652
|
+
private async getParsedHeader(
|
|
653
|
+
opts: Options = {},
|
|
654
|
+
retried = false,
|
|
655
|
+
): Promise<{ header: string; skippedLines: string[] }> {
|
|
656
|
+
throwIfAborted(opts.signal)
|
|
657
|
+
const pending = this.headerP
|
|
658
|
+
if (!pending) {
|
|
659
|
+
return this.startHeaderParse(opts)
|
|
660
|
+
}
|
|
661
|
+
|
|
662
|
+
// read before awaiting: the owner is forgotten as soon as the parse settles
|
|
663
|
+
const ownerSignal = this.headerSignal
|
|
664
|
+
try {
|
|
665
|
+
return await pending
|
|
666
|
+
} catch (e) {
|
|
667
|
+
if (retried || !ownerSignal?.aborted || opts.signal?.aborted) {
|
|
668
|
+
throw e
|
|
669
|
+
}
|
|
670
|
+
return this.getParsedHeader(opts, true)
|
|
671
|
+
}
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
private startHeaderParse(opts: Options) {
|
|
675
|
+
const pending = this.parseHeader(opts)
|
|
676
|
+
this.headerP = pending
|
|
677
|
+
this.headerSignal = opts.signal
|
|
678
|
+
// Drop a rejection rather than keeping it, so one transient failure does not
|
|
679
|
+
// poison the header for the lifetime of the file. Both branches are
|
|
680
|
+
// identity-checked so a retry started after this settles is not cleared by
|
|
681
|
+
// the attempt it already replaced.
|
|
682
|
+
pending.then(
|
|
683
|
+
() => {
|
|
684
|
+
if (this.headerP === pending) {
|
|
685
|
+
this.headerSignal = undefined
|
|
686
|
+
}
|
|
687
|
+
},
|
|
688
|
+
() => {
|
|
689
|
+
if (this.headerP === pending) {
|
|
690
|
+
this.headerP = undefined
|
|
691
|
+
this.headerSignal = undefined
|
|
692
|
+
}
|
|
693
|
+
},
|
|
694
|
+
)
|
|
695
|
+
return pending
|
|
679
696
|
}
|
|
680
697
|
|
|
681
698
|
/**
|
package/src/util.ts
CHANGED
|
@@ -5,36 +5,10 @@ import { longFromBytesToUnsigned } from './long.ts'
|
|
|
5
5
|
import VirtualOffset from './virtualOffset.ts'
|
|
6
6
|
|
|
7
7
|
// SYNC: ~/src/gmod/bam-js/src/util.ts optimizeChunks
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
* `AbortSignal`, and callers pass duck-typed ones, where calling a missing
|
|
13
|
-
* method is a `TypeError` rather than the cancellation the caller asked for —
|
|
14
|
-
* a strictly worse failure.
|
|
15
|
-
*
|
|
16
|
-
* And it sets a browser floor. `AbortSignal.prototype.throwIfAborted` and
|
|
17
|
-
* `AbortSignal.reason` are Safari 15.4 / Chrome 100 / Firefox 97 (March 2022),
|
|
18
|
-
* higher than anything else here needs: this package otherwise touches only
|
|
19
|
-
* `.aborted`, and `generic-filehandle2` only forwards a signal to `fetch`.
|
|
20
|
-
*
|
|
21
|
-
* Faithful to the spec otherwise: an aborted signal throws its `reason`
|
|
22
|
-
* whatever that is, and only synthesizes an `AbortError` when there is none.
|
|
23
|
-
* Kept in sync with the copy in `@gmod/bam`'s `src/util.ts`.
|
|
24
|
-
*/
|
|
25
|
-
export function throwIfAborted(signal?: AbortSignal) {
|
|
26
|
-
if (signal?.aborted) {
|
|
27
|
-
const reason: unknown = signal.reason
|
|
28
|
-
// Spec-faithful: throwIfAborted throws `reason` verbatim, and `reason` is
|
|
29
|
-
// whatever the caller passed to abort() — `controller.abort('too slow')`
|
|
30
|
-
// makes it a string. Coercing it to an Error here would hide that from a
|
|
31
|
-
// consumer who set it deliberately.
|
|
32
|
-
// eslint-disable-next-line @typescript-eslint/only-throw-error
|
|
33
|
-
throw reason === undefined
|
|
34
|
-
? new DOMException('This operation was aborted', 'AbortError')
|
|
35
|
-
: reason
|
|
36
|
-
}
|
|
37
|
-
}
|
|
8
|
+
// Re-exported so the internal import path stays './util.ts'. The
|
|
9
|
+
// implementation moved to @gmod/shared-read-cache, which needs it anyway --
|
|
10
|
+
// every consumer of that package was carrying an identical copy.
|
|
11
|
+
export { throwIfAborted } from '@gmod/shared-read-cache'
|
|
38
12
|
|
|
39
13
|
export function optimizeChunks(chunks: Chunk[], lowest?: VirtualOffset) {
|
|
40
14
|
const n = chunks.length
|