@gmod/bam 8.2.0 → 8.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/indexFile.ts CHANGED
@@ -1,6 +1,7 @@
1
+ import { SharedReadCache } from '@gmod/shared-read-cache'
1
2
  import QuickLRU from '@jbrowse/quick-lru'
2
3
 
3
- import { optimizeChunks, throwIfAborted } from './util.ts'
4
+ import { chunksLikelyRead, optimizeChunks } from './util.ts'
4
5
 
5
6
  import type Chunk from './chunk.ts'
6
7
  import type { BaseOpts } from './util.ts'
@@ -53,14 +54,11 @@ export default abstract class IndexFile<
53
54
  public filehandle: GenericFilehandle
54
55
  public renameRefSeq: (s: string) => string
55
56
 
56
- private parseP?: Promise<TParsed>
57
57
  /**
58
- * The signal `parseP` was started under, while it is still in flight. The
59
- * index is parsed once and shared by every query against the file, so without
60
- * this the first query to arrive would own a read all the others depend on —
61
- * see {@link parse}.
58
+ * The parsed index, as a shared read — see {@link parse}. One entry, never
59
+ * evicted, which is what a memo is.
62
60
  */
63
- private parseSignal?: AbortSignal
61
+ private parseCache = new SharedReadCache<string, TParsed>({})
64
62
 
65
63
  constructor({
66
64
  filehandle,
@@ -95,6 +93,16 @@ export default abstract class IndexFile<
95
93
  min: number,
96
94
  ): OffsetCoords | undefined
97
95
 
96
+ // Block position past which a chunk is EXPECTED to hold nothing overlapping
97
+ // [..., max] — the counterpart of getLowestChunk, and unlike it an estimate
98
+ // rather than a bound (see chunksLikelyRead). Only `estimatedBytesForRegions`
99
+ // may use it. CSI has no linear index and returns undefined, which reads as
100
+ // "no opinion" and leaves that estimate summing every chunk.
101
+ protected abstract getHighestChunk(
102
+ refIndex: RefIndex,
103
+ max: number,
104
+ ): number | undefined
105
+
98
106
  async blocksForRange(
99
107
  refId: number,
100
108
  min: number,
@@ -128,72 +136,34 @@ export default abstract class IndexFile<
128
136
  return optimizeChunks(chunks, this.getLowestChunk(ba, min))
129
137
  }
130
138
 
131
- // SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same owner-signal
132
- // tracking and one-attempt retry, and the same reasoning below.
139
+ // SYNC: ~/src/gmod/tabix-js/src/indexFile.ts parse — same shape and the same
140
+ // reasoning below.
133
141
  /**
134
142
  * Parse the index, or join the parse already running.
135
143
  *
136
144
  * The index is downloaded and parsed once for the life of this object, so it
137
145
  * is the one read here that is shared between queries — and therefore the one
138
146
  * place a cancellation can leak from the query that asked for it to a query
139
- * that did not. `_parse` hands `opts` straight to `filehandle.readFile`, so
140
- * without this the first query to arrive owns a read every other query
141
- * depends on: when it pans away, every concurrent query fails with its abort.
147
+ * that did not. `_parse` hands `opts` straight to `filehandle.readFile`, so a
148
+ * bare memoized promise makes the first query to arrive the owner of a read
149
+ * every other query depends on: when it pans away, every concurrent query
150
+ * fails with its abort.
142
151
  *
143
- * A caller that joined someone else's parse and saw it fail because *they*
144
- * aborted starts over rather than inheriting the failure — once, then
145
- * propagates. Bounding it at one attempt is what jbrowse's
146
- * `RemoteFileWithRangeCache.joinChunk` does with the same retry one layer
147
- * down, and for the reason it gives: the pathological case becomes one
148
- * duplicate parse rather than a recursion whose depth depends on how the
149
- * aborts interleave.
152
+ * The same cache the chunk reads use, for the same reason and with the same
153
+ * rule: the parse runs under a signal of its own and is cancelled only once
154
+ * every caller waiting on it has given up, so one query's abort is reported
155
+ * to that query alone. A rejection is dropped rather than cached, so a
156
+ * transient failure does not poison the index for the life of the file.
150
157
  *
151
- * A retry rather than the reference count `_cachedChunkFeatures` uses, for
152
- * the reason `@gmod/cram` gives for the same split in `CraiIndex`: the index
153
- * is parsed once for the life of the object, so there is no repeated waste to
154
- * recover, and this is a dozen lines against restructuring the memo.
158
+ * The fill is per call rather than on the cache so that the caller who starts
159
+ * the parse has its `onProgress` reach `filehandle.readFile` — the index is a
160
+ * whole-file read, and a determinate "downloading index" bar is what that
161
+ * callback exists for.
155
162
  */
156
- async parse(opts: BaseOpts = {}, retried = false): Promise<TParsed> {
157
- throwIfAborted(opts.signal)
158
- const pending = this.parseP
159
- if (!pending) {
160
- return this.startParse(opts)
161
- }
162
-
163
- // read before awaiting: the owner is forgotten as soon as the parse settles
164
- const ownerSignal = this.parseSignal
165
- try {
166
- return await pending
167
- } catch (e) {
168
- if (retried || !ownerSignal?.aborted || opts.signal?.aborted) {
169
- throw e
170
- }
171
- return this.parse(opts, true)
172
- }
173
- }
174
-
175
- private startParse(opts: BaseOpts) {
176
- const pending = this._parse(opts)
177
- this.parseP = pending
178
- this.parseSignal = opts.signal
179
- // Drop a rejection rather than keeping it, so one transient failure does not
180
- // poison the index for the lifetime of the file. Both branches are
181
- // identity-checked so a retry started after this settles is not cleared by
182
- // the attempt it already replaced.
183
- pending.then(
184
- () => {
185
- if (this.parseP === pending) {
186
- this.parseSignal = undefined
187
- }
188
- },
189
- () => {
190
- if (this.parseP === pending) {
191
- this.parseP = undefined
192
- this.parseSignal = undefined
193
- }
194
- },
163
+ parse(opts: BaseOpts = {}): Promise<TParsed> {
164
+ return this.parseCache.get('index', opts.signal, signal =>
165
+ this._parse({ ...opts, signal }),
195
166
  )
196
- return pending
197
167
  }
198
168
 
199
169
  async lineCount(refId: number, opts?: BaseOpts) {
@@ -206,9 +176,42 @@ export default abstract class IndexFile<
206
176
  return !!indexData.indices(seqId)
207
177
  }
208
178
 
179
+ /**
180
+ * Compressed bytes a `getRecordsForRange` over these regions is expected to
181
+ * download, from the index alone.
182
+ *
183
+ * Per region this is `chunksLikelyRead`, not every chunk `blocksForRange`
184
+ * returns: the difference is the whole point on a long-read file, where a
185
+ * narrow window inherits every chunk of every overlapping bin and reads a
186
+ * handful of them. Measured on a 40x ONT BAM (COLO829BL, chr3), summing all
187
+ * chunks against the bytes the same query really pulls:
188
+ *
189
+ * | window | all chunks | actually read | this estimate |
190
+ * | ------ | ---------- | ------------- | ------------- |
191
+ * | 380bp | 43.5MB | 7.8MB | 7.8MB |
192
+ * | 3.4kb | 43.5MB | 7.8MB | 7.8MB |
193
+ * | 100kb | 46.6MB | 10.4MB | 10.4MB |
194
+ * | 2Mb | 155.9MB | 155.9MB | 123.8MB |
195
+ *
196
+ * A caller gating on this — jbrowse's "too much data" banner is the one that
197
+ * exists — was being told 5.6x the truth on exactly the windows a reader
198
+ * spends their time in, and cannot answer it by zooming: every window narrower
199
+ * than a linear-index interval resolves to the same chunks and so to the same
200
+ * number.
201
+ *
202
+ * Still summed over merged chunks rather than per region, so two regions
203
+ * sharing a chunk are charged for it once.
204
+ */
209
205
  async estimatedBytesForRegions(regions: Region[], opts?: BaseOpts) {
206
+ const indexData = await this.parse(opts)
210
207
  const blockResults = await Promise.all(
211
- regions.map(r => this.blocksForRange(r.refId, r.start, r.end, opts)),
208
+ regions.map(async r => {
209
+ const chunks = await this.blocksForRange(r.refId, r.start, r.end, opts)
210
+ const refIndex = indexData.indices(r.refId)
211
+ return refIndex
212
+ ? chunksLikelyRead(chunks, this.getHighestChunk(refIndex, r.end))
213
+ : chunks
214
+ }),
212
215
  )
213
216
 
214
217
  // Deduplicate and merge overlapping blocks across all regions
package/src/util.ts CHANGED
@@ -29,6 +29,79 @@ export interface BaseOpts {
29
29
  onProgress?: (bytesDownloaded: number, totalBytes?: number) => void
30
30
  }
31
31
 
32
+ // Re-exported so the internal import path stays './util.ts'. The
33
+ // implementation moved to @gmod/shared-read-cache, which needs it anyway --
34
+ // every consumer of that package was carrying an identical copy.
35
+ export { throwIfAborted } from '@gmod/shared-read-cache'
36
+
37
+ // How many of a query's chunks to read at once. Six is the HTTP/1.1 per-host
38
+ // connection cap browsers enforce, so going much above it buys nothing on the
39
+ // transport that matters and only widens peak memory.
40
+ //
41
+ // It is also the SMALLEST number of chunks any query reads, because
42
+ // `_fetchChunkFeatures` checks its early stop once, after this first batch
43
+ // (ADR 0010) — which is what `chunksLikelyRead` below is built on.
44
+ export const MAX_CONCURRENT_CHUNK_READS = 6
45
+
46
+ /**
47
+ * The prefix of `chunks` a query is expected to actually read, for callers that
48
+ * want to forecast a query's cost without running it.
49
+ *
50
+ * `blocksForRange` returns every chunk of every bin overlapping the query, at
51
+ * every level of the binning scheme, minus the ones the linear index puts
52
+ * entirely before it. On a long-read file that is wildly more than the query
53
+ * reads: a coarse bin's chunks run to the end of the bin's span, so a 380bp
54
+ * window on a deep ONT BAM resolves to 90 chunks / 43.5MB, of which
55
+ * `getRecordsForRange` reads 6 / 7.8MB before the early stop fires.
56
+ *
57
+ * Two bounds, and the answer is the larger:
58
+ *
59
+ * - `upperBoundBlockPosition` — the file offset past which a chunk holds no
60
+ * record the query wants, from the linear index one window beyond the query
61
+ * end. Chunks come back sorted by `minv`, so the ones under it are a prefix.
62
+ * - `MAX_CONCURRENT_CHUNK_READS` — the first batch is read unconditionally, so
63
+ * no query ever costs less than that however tight the upper bound is.
64
+ *
65
+ * **This is a forecast, not a bound the reader obeys**, and it is deliberately
66
+ * only used to describe a query rather than to answer one. The upper bound is
67
+ * an approximation: a record starting before the query end can sit at a higher
68
+ * offset than the linear-index entry past it (a long read reaching into the
69
+ * next window pins that entry low), so pruning a fetch this way would drop
70
+ * records. A forecast that is occasionally 20% under warns slightly early;
71
+ * a fetch that is occasionally short returns wrong data.
72
+ *
73
+ * An EMPTY prefix is that pin at its worst — the bound has landed at or before
74
+ * the query's own first chunk, so it orders nothing — and the answer there is
75
+ * every chunk rather than the batch floor. Without it this returns the floor on
76
+ * a file whose features are long against the linear-index interval, which is
77
+ * where the same forecast really does go wrong: ported to tabix-js and measured
78
+ * on the 1000 Genomes SV callset, whose 1.4Mb deletions pin the entry at the
79
+ * data start, it forecast 0.04MB against the 0.22MB the query read. On this
80
+ * corpus the fallback costs one fixture's win (chr22_nanopore, whose 10kb
81
+ * window keeps summing all 22 chunks) and no accuracy anywhere else.
82
+ */
83
+ export function chunksLikelyRead(
84
+ chunks: Chunk[],
85
+ upperBoundBlockPosition?: number,
86
+ ) {
87
+ if (
88
+ upperBoundBlockPosition === undefined ||
89
+ chunks.length <= MAX_CONCURRENT_CHUNK_READS
90
+ ) {
91
+ return chunks
92
+ }
93
+ let n = 0
94
+ while (
95
+ n < chunks.length &&
96
+ chunks[n]!.minv.blockPosition < upperBoundBlockPosition
97
+ ) {
98
+ n++
99
+ }
100
+ return n === 0 || n >= chunks.length
101
+ ? chunks
102
+ : chunks.slice(0, Math.max(n, MAX_CONCURRENT_CHUNK_READS))
103
+ }
104
+
32
105
  /**
33
106
  * Merge and order the chunks a query resolved to.
34
107
  *
@@ -39,11 +112,6 @@ export interface BaseOpts {
39
112
  * underneath you. (The Chunk objects themselves are never mutated; a merged
40
113
  * span produces a new instance.)
41
114
  */
42
- // Re-exported so the internal import path stays './util.ts'. The
43
- // implementation moved to @gmod/shared-read-cache, which needs it anyway --
44
- // every consumer of that package was carrying an identical copy.
45
- export { throwIfAborted } from '@gmod/shared-read-cache'
46
-
47
115
  export function optimizeChunks(chunks: Chunk[], lowest?: OffsetCoords) {
48
116
  const n = chunks.length
49
117
  if (n === 0) {