@gmod/bam 8.5.0 → 8.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +79 -226
  2. package/dist/bamFile.d.ts +94 -1
  3. package/dist/bamFile.js +94 -3
  4. package/dist/bamFile.js.map +1 -1
  5. package/dist/htsget.d.ts +8 -1
  6. package/dist/htsget.js +4 -1
  7. package/dist/htsget.js.map +1 -1
  8. package/dist/index.d.ts +5 -1
  9. package/dist/index.js +15 -1
  10. package/dist/index.js.map +1 -1
  11. package/dist/mismatches.d.ts +114 -0
  12. package/dist/mismatches.js +397 -0
  13. package/dist/mismatches.js.map +1 -0
  14. package/dist/record.d.ts +45 -0
  15. package/dist/record.js +76 -10
  16. package/dist/record.js.map +1 -1
  17. package/dist/reference.d.ts +45 -0
  18. package/dist/reference.js +61 -0
  19. package/dist/reference.js.map +1 -0
  20. package/dist/seqAlphabet.d.ts +3 -0
  21. package/dist/seqAlphabet.js +14 -0
  22. package/dist/seqAlphabet.js.map +1 -0
  23. package/esm/bamFile.d.ts +94 -1
  24. package/esm/bamFile.js +94 -3
  25. package/esm/bamFile.js.map +1 -1
  26. package/esm/htsget.d.ts +8 -1
  27. package/esm/htsget.js +4 -1
  28. package/esm/htsget.js.map +1 -1
  29. package/esm/index.d.ts +5 -1
  30. package/esm/index.js +6 -0
  31. package/esm/index.js.map +1 -1
  32. package/esm/mismatches.d.ts +114 -0
  33. package/esm/mismatches.js +393 -0
  34. package/esm/mismatches.js.map +1 -0
  35. package/esm/record.d.ts +45 -0
  36. package/esm/record.js +69 -3
  37. package/esm/record.js.map +1 -1
  38. package/esm/reference.d.ts +45 -0
  39. package/esm/reference.js +55 -0
  40. package/esm/reference.js.map +1 -0
  41. package/esm/seqAlphabet.d.ts +3 -0
  42. package/esm/seqAlphabet.js +11 -0
  43. package/esm/seqAlphabet.js.map +1 -0
  44. package/package.json +1 -1
  45. package/src/bamFile.ts +178 -1
  46. package/src/htsget.ts +16 -2
  47. package/src/index.ts +26 -1
  48. package/src/mismatches.ts +623 -0
  49. package/src/record.ts +95 -3
  50. package/src/reference.ts +87 -0
  51. package/src/seqAlphabet.ts +10 -0
package/README.md CHANGED
@@ -21,18 +21,81 @@ const records = await bam.getRecordsForRange('ctgA', 0, 50000)
21
21
  ```
22
22
 
23
23
  Coordinates are 0-based half-open (not the same as `samtools view` inputs).
24
- `bamPath` reads a local file, so it is node-only; in the browser pass a
25
- filehandle or URL instead:
24
+ `bamPath` reads a local file, so it is node-only; in the browser pass a URL or a
25
+ generic-filehandle2 filehandle instead:
26
26
 
27
27
  ```typescript
28
- import { BamFile } from '@gmod/bam'
29
-
30
28
  const bam = new BamFile({
31
29
  bamUrl: 'https://example.com/yourfile.bam',
32
30
  baiUrl: 'https://example.com/yourfile.bam.bai',
33
31
  })
34
32
  ```
35
33
 
34
+ Records come back unfiltered, and are shared between overlapping queries — treat
35
+ them as read-only. Filter them yourself with the flag helpers and `getTag`,
36
+ which decodes one tag instead of all of them:
37
+
38
+ ```typescript
39
+ const records = (await bam.getRecordsForRange('chr1', 0, 100000)).filter(
40
+ r => r.isProperlyPaired() && !r.isSecondary() && r.getTag('RG') === 'rg1',
41
+ )
42
+ ```
43
+
44
+ ## Mismatches
45
+
46
+ `record.getMismatches()` gives every difference between a read and the reference
47
+ — substitutions, insertions, deletions, reference skips and clips — without you
48
+ having to interpret `CIGAR` and `MD` yourself. There is a callback form,
49
+ `record.forEachMismatch(cb, opts?)`, which allocates nothing per difference and
50
+ takes a reference window to report within.
51
+
52
+ Substitutions need either an `MD` tag on the read or the reference bases, and
53
+ most aligners leave `MD` off. `fetchReferenceSequence` is how you supply them:
54
+
55
+ ```typescript
56
+ const bam = new BamFile({
57
+ bamPath: 'test.bam',
58
+ fetchReferenceSequence: async (refName, start, end) =>
59
+ myGenome.getSequence(refName, start, end),
60
+ })
61
+
62
+ // one sequence fetch for the whole query, and only if some read needs it
63
+ const records = await bam.getRecordsForRange('ctgA', 0, 50000)
64
+ records[0].getMismatches()
65
+ // [{ code: 88 /* 'X' */, refPos: 188, length: 1, bases: 'A', qual: 17,
66
+ // refBaseCode: 84 /* 'T' */, clipLength: 0 }, ...]
67
+ ```
68
+
69
+ Without it, a read lacking `MD` still reports its indels and clips, but no
70
+ substitutions — nothing in the record says where they are. See
71
+ [docs/api.md](docs/api.md#mismatches) for the field meanings and for reads
72
+ longer than the region you are looking at.
73
+
74
+ ## Decompressing on a worker pool
75
+
76
+ BGZF decompression is 70-90% of a cold query, and BGZF blocks are independently
77
+ inflatable. Hand `BamFile` a
78
+ [`@gmod/bgzf-filehandle`](https://github.com/GMOD/bgzf-filehandle) worker pool
79
+ and it inflates chunks there instead of on the calling thread — measured
80
+ 2.7-4.1x on the pool's own fixtures.
81
+
82
+ ```typescript
83
+ import { getSharedWorkerPool } from '@gmod/bgzf-filehandle'
84
+
85
+ const bam = new BamFile({
86
+ bamUrl: 'https://example.com/yourfile.bam',
87
+ // the pending promise is fine — it is awaited at the point of use
88
+ bgzfWorkerPool: getSharedWorkerPool(),
89
+ })
90
+ ```
91
+
92
+ No cross-origin isolation is needed. `getSharedWorkerPool()` gives back
93
+ `undefined` under node, or anywhere Workers cannot be created, which keeps the
94
+ in-process path — so this is safe to pass unconditionally. bam-js never creates
95
+ a pool on its own: the thread budget belongs to the consumer. For worker counts,
96
+ lifecycle and the pool's own benchmarks, see
97
+ [bgzf-filehandle's worker pool docs](https://github.com/GMOD/bgzf-filehandle/blob/main/docs/worker-pool.md).
98
+
36
99
  ## Usage with htsget
37
100
 
38
101
  ```typescript
@@ -46,10 +109,8 @@ const records = await bam.getRecordsForRange('1', 2000000, 2000001)
46
109
  ```
47
110
 
48
111
  htsget fetches the server's range as-is, so `viewAsPairs`, `pairAcrossChr` and
49
- `maxInsertSize` are ignored.
50
-
51
- For a server that requires authentication, pass a `fetch` that adds the bearer
52
- token the spec calls for:
112
+ `maxInsertSize` are ignored. For a server that requires authentication, pass a
113
+ `fetch` that adds the bearer token:
53
114
 
54
115
  ```typescript
55
116
  const bam = new HtsgetFile({
@@ -63,227 +124,19 @@ const bam = new HtsgetFile({
63
124
  })
64
125
  ```
65
126
 
66
- Your `fetch` is called for the ticket request and for the data-block urls the
67
- ticket points at, so only attach credentials to hosts you trust — data blocks
68
- may live on a third-party host, and the spec has servers put whatever those need
69
- in each url's own `headers` field, which is applied either way.
70
-
71
- ## Documentation
72
-
73
- ### BamFile constructor
74
-
75
- - `bamPath`/`bamUrl`/`bamFilehandle` - local path, remote URL, or a
76
- generic-filehandle2 object
77
- - `baiPath`/`baiUrl`/`baiFilehandle` - BAI index. Defaults to the `.bai` sibling
78
- of `bamPath`/`bamUrl`
79
- - `csiPath`/`csiUrl`/`csiFilehandle` - CSI index, required for chromosomes
80
- longer than 2^29
81
- - `renameRefSeqs` - `(refName: string) => string` applied to header ref names
82
- - `recordClass` - custom class extending `BamRecord` (see below)
83
- - `maxCacheBytes` - ceiling for the parsed-chunk cache, in decompressed bytes.
84
- default: 1GB, `Infinity` for none. See [Caching](#caching)
85
- - `cacheIdleTimeoutMs` - drop a cached chunk once nothing has read it for this
86
- long. default: 3 minutes; `0` disables it. See [Caching](#caching)
87
-
88
- The `path`/`url` forms are convenience wrappers for generic-filehandle2's
89
- `LocalFile` and `RemoteFile`.
90
-
91
- ### Caching
92
-
93
- Parsed chunks — the unit the BAM index hands out — are cached, so overlapping
94
- and adjacent queries reuse decompressed records instead of re-fetching them. Two
95
- options bound that cache, and they answer different questions.
96
-
97
- **`maxCacheBytes` is a ceiling under load, not a limit on what you can ask
98
- for.** Nothing is ever refused for being too large: a chunk bigger than the
99
- whole budget is still cached, reads in flight are never evicted, and eviction
100
- only drops a value that has already been returned once. The worst a budget can
101
- cost you is a re-read. It can make a query slower; it can never make one fail or
102
- come back short.
103
-
104
- **It binds less often than its size suggests.** On the deepest data we measure —
105
- 1000x coverage long reads, 240 windows over six laps — the cache settles at
106
- 573MB across 60 entries and eviction never runs at the 1GB default. Treat it as
107
- a backstop against a session that pans forever, not as an operating constraint.
108
-
109
- **Don't pick a number between one query and several.** Below one query's working
110
- set the cache inverts: each chunk is evicted before the next pan can reuse it,
111
- so the hit rate is zero, the full re-decompress is paid every time, and the
112
- unevictable entries retain the memory anyway. At a 200MB budget on that same
113
- file the cache holds exactly one entry, because one chunk there decompresses to
114
- 181MB. Either size it above the working set or pass `Infinity` and bound memory
115
- some other way.
116
-
117
- **`cacheIdleTimeoutMs` is the only thing that gives memory back.**
118
- `maxCacheBytes` is enforced when a read settles, so an idle cache sits at
119
- whatever it reached and never lowers — and for a page that holds a `BamFile` for
120
- the life of a track, that resting level is the number that actually matters. The
121
- idle sweep is what makes a generous ceiling affordable, by turning it into a
122
- peak under panning rather than a level a parked tab holds indefinitely. The
123
- clock runs from the last _read_ of a chunk, or from its parse landing if nothing
124
- has read it since, so panning back and forth over one region never expires it
125
- and a slow chunk still gets the full timeout to be reused in. Measured on a pan
126
- that held 331MB: 0MB once idle.
127
-
128
- **Neither bounds peak memory**, and on deep data the gap is not small:
129
-
130
- - Up to six chunk reads run at once and an in-flight read is never evicted. At
131
- 181MB a chunk, that is ~476MB in flight before retention holds anything.
132
- - A query holds every chunk it parsed until it returns, whether or not the cache
133
- still does.
134
- - An entry is weighed once, when its read settles, and records grow after that.
135
- `end`, `CIGAR` and `tags` each memoize onto the record the first time they are
136
- read, which is what a renderer does to every visible read — measured at +38%
137
- over the weighed size.
138
-
139
- So these are bounds on retained decompressed bytes, not on the heap. Size
140
- against what you want to keep, and bound total memory at a level that can see
141
- the whole process.
142
-
143
- ### HtsgetFile constructor
144
-
145
- - `baseUrl` - htsget reads endpoint, e.g. `https://htsget.example.com/reads`
146
- - `trackId` - id of the resource under `baseUrl`
147
- - `fetch` - `fetch` replacement for adding auth headers (see above)
148
- - `recordClass` - custom class extending `BamRecord` (see below)
149
-
150
- ### async getRecordsForRange(refName, start, end, opts?)
151
-
152
- - `refName` - chromosome to fetch from
153
- - `start`/`end` - 0-based half-open coordinates
154
- - `opts.signal` - `AbortSignal` to stop processing
155
- - `opts.viewAsPairs` - re-dispatch requests to find mate pairs. default: false
156
- - `opts.pairAcrossChr` - let `viewAsPairs` pair across chromosomes. default:
157
- false
158
- - `opts.maxInsertSize` - distance limit for `viewAsPairs` within a chromosome.
159
- default: 200kb
160
- - `opts.onProgress` - `(bytesDownloaded, totalBytes?) => void`, called per BGZF
161
- chunk for a determinate progress bar
162
-
163
- Returned records are cached and shared between overlapping queries, so treat
164
- them as read-only — attaching your own fields to a record mutates it for every
165
- other query holding it.
166
-
167
- Records come back unfiltered. Filter them yourself with the flag helpers and
168
- `getTag`, which decodes one tag instead of all of them:
127
+ Your `fetch` is called for the ticket request _and_ for the data-block urls the
128
+ ticket points at, which may live on a third-party host — so only attach
129
+ credentials to hosts you trust.
169
130
 
170
- ```typescript
171
- const records = (await bam.getRecordsForRange('chr1', 0, 100000)).filter(
172
- r => r.isProperlyPaired() && !r.isSecondary() && r.getTag('RG') === 'rg1',
173
- )
174
- ```
175
-
176
- ### async getHeader(opts?)
177
-
178
- Returns the parsed SAM header. Called automatically by the query methods and
179
- cached, so you only need it when you want the header itself.
180
- `getHeaderText(opts?)` returns the raw header string.
181
-
182
- ### async indexCov(refName, start?, end?)
183
-
184
- Returns `{start, end, score}` features estimating read density over 16kb
185
- windows, derived from the BAI linear index. CSI has no linear index, so a
186
- CSI-indexed file returns `[]`.
187
-
188
- ### async lineCount(refName)
189
-
190
- Number of records on `refName` from the index's pseudo-bin (bin 37450 in BAI,
191
- `n_mapped` in the SAM spec), or 0 if `refName` is absent.
131
+ ## Docs
192
132
 
193
- ### async hasRefSeq(refName)
194
-
195
- Whether `refName` is present in the file.
196
-
197
- ### async estimatedBytesForRegions(regions, opts?)
198
-
199
- Compressed bytes the given `{refName, start, end}[]` would fetch — useful for
200
- warning before a large query.
201
-
202
- ### clearFeatureCache()
203
-
204
- Drops the parsed-chunk cache immediately, rather than waiting for
205
- `maxCacheBytes` or `cacheIdleTimeoutMs` to reclaim it. See [Caching](#caching).
206
-
207
- ### BamRecord
208
-
209
- ```typescript
210
- // Core alignment fields
211
- record.fileOffset // "file offset" based id -- not a true file offset
212
- record.ref_id // numerical sequence id from SAM header
213
- record.start // 0-based start coordinate
214
- record.end // 0-based end coordinate
215
- record.name // QNAME
216
- record.seq // sequence string
217
- record.qual // Uint8Array of quality scores (null if SEQ is empty)
218
- record.CIGAR // CIGAR string e.g. "50M2I48M"
219
- record.flags // SAM flags integer
220
- record.mq // mapping quality (undefined if 255)
221
- record.strand // 1 or -1
222
- record.template_length // TLEN
223
-
224
- // Mate info
225
- record.next_refid
226
- record.next_pos
227
-
228
- // Auxiliary data
229
- record.tags // all aux tags e.g. {MD: "100", NM: 0}
230
- record.getTag('MD') // one tag, without decoding the rest
231
- record.getTagRaw('MD') // string tag as Uint8Array, skipping string conversion
232
-
233
- // Typed-array views, for rendering without allocating strings
234
- record.NUMERIC_MD // MD tag as Uint8Array
235
- record.NUMERIC_CIGAR // Uint32Array of packed CIGAR operations
236
- record.NUMERIC_SEQ // Uint8Array of 4-bit encoded sequence
237
-
238
- // Flag methods
239
- record.isPaired()
240
- record.isProperlyPaired()
241
- record.isSegmentUnmapped()
242
- record.isMateUnmapped()
243
- record.isReverseComplemented()
244
- record.isMateReverseComplemented()
245
- record.isRead1()
246
- record.isRead2()
247
- record.isSecondary()
248
- record.isFailedQc()
249
- record.isDuplicate()
250
- record.isSupplementary()
251
-
252
- // Utility
253
- record.seqAt(idx) // single base at position
254
- record.toJSON()
255
- ```
256
-
257
- ### Custom BamRecord class
258
-
259
- ```typescript
260
- import { BamFile, BamRecord } from '@gmod/bam'
261
-
262
- class CustomBamRecord extends BamRecord {
263
- get customProperty() {
264
- return `custom-${this.name}`
265
- }
266
- }
267
-
268
- const bam = new BamFile<CustomBamRecord>({
269
- bamPath: 'test.bam',
270
- recordClass: CustomBamRecord,
271
- })
272
-
273
- // records are typed as CustomBamRecord[]
274
- const records = await bam.getRecordsForRange('ctgA', 0, 50000)
275
- console.log(records[0].customProperty)
276
- ```
133
+ - [docs/api.md](docs/api.md) — every constructor option, method and `BamRecord`
134
+ field, plus custom record classes
135
+ - [docs/caching.md](docs/caching.md) — sizing the parsed-chunk cache
136
+ - [agent-docs/adr/](agent-docs/adr/) — the measurements behind the performance
137
+ and caching decisions
138
+ - [CONTRIBUTING.md](CONTRIBUTING.md) — development and release steps
277
139
 
278
140
  ## License
279
141
 
280
142
  MIT © [Colin Diesh](https://github.com/cmdcolin)
281
-
282
- ## Publishing
283
-
284
- [Trusted publishing](https://docs.npmjs.com/about-trusted-publishing) via GitHub
285
- Actions.
286
-
287
- ```bash
288
- pnpm version patch # or minor/major
289
- ```
package/dist/bamFile.d.ts CHANGED
@@ -3,6 +3,7 @@ import BAI from './bai.ts';
3
3
  import CSI from './csi.ts';
4
4
  import BAMFeature from './record.ts';
5
5
  import type Chunk from './chunk.ts';
6
+ import type { PackedReference } from './reference.ts';
6
7
  import type { BamOpts, BaseOpts } from './util.ts';
7
8
  import type { BgzfWorkerPool } from '@gmod/bgzf-filehandle';
8
9
  import type { SharedBudget } from '@gmod/shared-read-cache';
@@ -17,7 +18,35 @@ export interface BamRecordLike {
17
18
  next_refid: number;
18
19
  flags: number;
19
20
  tags: Record<string, unknown>;
21
+ /**
22
+ * Optional, and read only to decide whether a read needs reference bases at
23
+ * all — a read with an MD tag carries its own. A `recordClass` without it is
24
+ * treated as having no MD, i.e. as always wanting the reference.
25
+ */
26
+ NUMERIC_MD?: Uint8Array | undefined;
27
+ /**
28
+ * Optional; see {@link BamRecord.setReference}. A `recordClass` that does not
29
+ * implement it simply never gets a reference bound, whatever
30
+ * `fetchReferenceSequence` returns.
31
+ */
32
+ setReference?: (ref: PackedReference) => void;
20
33
  }
34
+ /**
35
+ * Supplies reference bases for a region, so reads with no MD tag can still
36
+ * report their substitutions. See {@link BamFile}'s option of the same name.
37
+ *
38
+ * `refName` is the name the query used, unchanged — `renameRefSeqs` maps the
39
+ * FILE's names into the caller's namespace, and this callback is on the
40
+ * caller's side of that. `start`/`end` are 0-based half-open.
41
+ *
42
+ * **The bases returned must begin at `start`.** Returning fewer than asked for
43
+ * is fine and is taken as `[start, start + seq.length)` — the end of a contig,
44
+ * or a source declining to hand over a huge span — and reads the shorter region
45
+ * does not cover are then left unresolved. Returning bases from somewhere else,
46
+ * e.g. clipping the LEFT of the requested range, cannot be detected and
47
+ * resolves every read against the wrong position.
48
+ */
49
+ export type ReferenceSequenceFetcher = (refName: string, start: number, end: number, opts?: BaseOpts) => Promise<string>;
21
50
  export type BamRecordClass<T extends BamRecordLike = BAMFeature> = new (byteArray: Uint8Array, start: number, end: number, fileOffset: number, dataView: DataView) => T;
22
51
  export declare const BAM_MAGIC = 21840194;
23
52
  interface ChunkEntry<T> {
@@ -46,7 +75,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
46
75
  private RecordClass;
47
76
  /** see the constructor option of the same name */
48
77
  private bgzfWorkerPool?;
49
- constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs, recordClass, maxCacheBytes, cacheIdleTimeoutMs, cacheBudget, bgzfWorkerPool, }: {
78
+ /** see the constructor option of the same name */
79
+ fetchReferenceSequence?: ReferenceSequenceFetcher;
80
+ constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs, recordClass, maxCacheBytes, cacheIdleTimeoutMs, cacheBudget, bgzfWorkerPool, fetchReferenceSequence, }: {
50
81
  bamFilehandle?: GenericFilehandle;
51
82
  bamPath?: string;
52
83
  bamUrl?: string;
@@ -135,6 +166,33 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
135
166
  * in-process path, so this is safe to pass unconditionally.
136
167
  */
137
168
  bgzfWorkerPool?: BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined;
169
+ /**
170
+ * Reference bases for a region, as `(refName, start, end, opts) =>
171
+ * Promise<string>`. Only reads that carry no `MD` tag need it, and it is
172
+ * only what {@link BamRecord.forEachMismatch} uses — nothing else in a
173
+ * query touches the reference.
174
+ *
175
+ * Without it, a read lacking MD still reports its indels and clips but no
176
+ * substitutions at all: neither the CIGAR nor SEQ says where they are.
177
+ * That is most aligners' output — minimap2 and bwa both leave MD off unless
178
+ * asked — so this is the difference between mismatches rendering and not.
179
+ *
180
+ * `getRecordsForRange` calls it at most ONCE per query, for the span of the
181
+ * reads that need it, and binds the result to each of them (see
182
+ * {@link BamRecord.setReference}). A query whose reads all carry MD does
183
+ * not call it at all.
184
+ *
185
+ * **The span asked for is the reads' union, not the query's**: the query's
186
+ * range plus however far its edge reads overhang it. Usually that is a few
187
+ * hundred bases more for Illumina and a couple of Mb for ultra-long ONT,
188
+ * but a BAM holding whole chromosomes as reads can make it a chromosome.
189
+ * Nothing here clamps that for you, because only you know what your
190
+ * sequence source can afford — clamp inside the callback and return the
191
+ * shorter region; reads it does not cover are then simply left unresolved,
192
+ * and {@link BamRecord.forEachMismatch}'s `opts.ref` is the windowed way to
193
+ * handle one of those reads.
194
+ */
195
+ fetchReferenceSequence?: ReferenceSequenceFetcher;
138
196
  });
139
197
  getHeaderPre(opts?: BaseOpts): Promise<{
140
198
  tag: string;
@@ -201,6 +259,41 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
201
259
  * it for every other query still holding that read. See ADR 0006.
202
260
  */
203
261
  getRecordsForRange(chr: string, min: number, max: number, opts?: BamOpts): Promise<T[]>;
262
+ /**
263
+ * Reference bases for a region, packed for
264
+ * {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
265
+ * `fetchReferenceSequence` was configured.
266
+ *
267
+ * The way to resolve a read that is longer than the region you are looking at
268
+ * — a contig or a whole chromosome stored as one BAM read. Those are never
269
+ * bound to a record automatically, since binding a partial region to a
270
+ * record shared between queries is the hazard ADR 0006 is about; walking one
271
+ * with a window and a region of your own choosing has no such problem:
272
+ *
273
+ * ```js
274
+ * const ref = await bam.getReferenceRegion('chr1', start, end)
275
+ * record.forEachMismatch(cb, { ref, start, end })
276
+ * ```
277
+ */
278
+ getReferenceRegion(refName: string, start: number, end: number, opts?: BaseOpts): Promise<PackedReference | undefined>;
279
+ /**
280
+ * Fetch reference bases for the reads in `records` that have no MD tag, and
281
+ * bind them to those reads so their substitutions resolve. A no-op unless
282
+ * `fetchReferenceSequence` was configured, and it issues at most one fetch.
283
+ *
284
+ * The span fetched is the union of the reads that need it, NOT the queried
285
+ * range: a read overhanging the range is only bound to a region covering all
286
+ * of it, because the records are shared between queries and a binding that
287
+ * varied by query would make one query's reads answer out of another's region
288
+ * (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
289
+ * `pairAcrossChr` — are left out for the same reason, and unbound.
290
+ *
291
+ * Nothing clamps that union, since only the callback knows what its sequence
292
+ * source can afford; a callback that returns a shorter region than it was
293
+ * asked for leaves the reads it does not cover unbound, which is the same
294
+ * outcome by a route the consumer controls.
295
+ */
296
+ protected _applyReferenceSequence(records: T[], chrId: number, refName: string, opts?: BaseOpts): Promise<void>;
204
297
  private _cachedChunkFeatures;
205
298
  private _fetchChunkFeatures;
206
299
  fetchPairs(chrId: number, records: T[], opts: BamOpts): Promise<T[]>;
package/dist/bamFile.js CHANGED
@@ -12,6 +12,7 @@ const bai_ts_1 = __importDefault(require("./bai.js"));
12
12
  const csi_ts_1 = __importDefault(require("./csi.js"));
13
13
  const nullFilehandle_ts_1 = __importDefault(require("./nullFilehandle.js"));
14
14
  const record_ts_1 = __importDefault(require("./record.js"));
15
+ const reference_ts_1 = require("./reference.js");
15
16
  const sam_ts_1 = require("./sam.js");
16
17
  const util_ts_1 = require("./util.js");
17
18
  exports.BAM_MAGIC = 21840194;
@@ -112,9 +113,12 @@ class BamFile {
112
113
  RecordClass;
113
114
  /** see the constructor option of the same name */
114
115
  bgzfWorkerPool;
115
- constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs = n => n, recordClass, maxCacheBytes = exports.DEFAULT_MAX_CACHE_BYTES, cacheIdleTimeoutMs = exports.DEFAULT_CACHE_IDLE_TIMEOUT_MS, cacheBudget, bgzfWorkerPool, }) {
116
+ /** see the constructor option of the same name */
117
+ fetchReferenceSequence;
118
+ constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs = n => n, recordClass, maxCacheBytes = exports.DEFAULT_MAX_CACHE_BYTES, cacheIdleTimeoutMs = exports.DEFAULT_CACHE_IDLE_TIMEOUT_MS, cacheBudget, bgzfWorkerPool, fetchReferenceSequence, }) {
116
119
  this.renameRefSeq = renameRefSeqs;
117
120
  this.bgzfWorkerPool = bgzfWorkerPool;
121
+ this.fetchReferenceSequence = fetchReferenceSequence;
118
122
  this.RecordClass = (recordClass ?? record_ts_1.default);
119
123
  this.chunkFeatureCache = new shared_read_cache_1.SharedReadCache({
120
124
  maxSize: maxCacheBytes,
@@ -269,7 +273,90 @@ class BamFile {
269
273
  return [];
270
274
  }
271
275
  const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts);
272
- return this._fetchChunkFeatures(chunks, chrId, min, max, opts);
276
+ return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts);
277
+ }
278
+ /**
279
+ * Reference bases for a region, packed for
280
+ * {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
281
+ * `fetchReferenceSequence` was configured.
282
+ *
283
+ * The way to resolve a read that is longer than the region you are looking at
284
+ * — a contig or a whole chromosome stored as one BAM read. Those are never
285
+ * bound to a record automatically, since binding a partial region to a
286
+ * record shared between queries is the hazard ADR 0006 is about; walking one
287
+ * with a window and a region of your own choosing has no such problem:
288
+ *
289
+ * ```js
290
+ * const ref = await bam.getReferenceRegion('chr1', start, end)
291
+ * record.forEachMismatch(cb, { ref, start, end })
292
+ * ```
293
+ */
294
+ async getReferenceRegion(refName, start, end, opts) {
295
+ const fetchReferenceSequence = this.fetchReferenceSequence;
296
+ if (!fetchReferenceSequence) {
297
+ return undefined;
298
+ }
299
+ // Packed once for the region rather than per read — the walk compares two
300
+ // bases per byte against the read's own packed SEQ, and this is the only
301
+ // per-base pass in it. Length comes from what came back rather than from
302
+ // what was asked for, so a callback that clips at the end of the contig
303
+ // still leaves the bases it did return usable.
304
+ return (0, reference_ts_1.packReference)(await fetchReferenceSequence(refName, start, end, opts), start);
305
+ }
306
+ /**
307
+ * Fetch reference bases for the reads in `records` that have no MD tag, and
308
+ * bind them to those reads so their substitutions resolve. A no-op unless
309
+ * `fetchReferenceSequence` was configured, and it issues at most one fetch.
310
+ *
311
+ * The span fetched is the union of the reads that need it, NOT the queried
312
+ * range: a read overhanging the range is only bound to a region covering all
313
+ * of it, because the records are shared between queries and a binding that
314
+ * varied by query would make one query's reads answer out of another's region
315
+ * (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
316
+ * `pairAcrossChr` — are left out for the same reason, and unbound.
317
+ *
318
+ * Nothing clamps that union, since only the callback knows what its sequence
319
+ * source can afford; a callback that returns a shorter region than it was
320
+ * asked for leaves the reads it does not cover unbound, which is the same
321
+ * outcome by a route the consumer controls.
322
+ */
323
+ async _applyReferenceSequence(records, chrId, refName, opts = {}) {
324
+ const fetchReferenceSequence = this.fetchReferenceSequence;
325
+ if (!fetchReferenceSequence) {
326
+ return;
327
+ }
328
+ let start = Infinity;
329
+ let end = 0;
330
+ for (let i = 0, l = records.length; i < l; i++) {
331
+ const record = records[i];
332
+ if (record.ref_id === chrId &&
333
+ !record.NUMERIC_MD &&
334
+ record.setReference) {
335
+ if (record.start < start) {
336
+ start = record.start;
337
+ }
338
+ if (record.end > end) {
339
+ end = record.end;
340
+ }
341
+ }
342
+ }
343
+ // every read carries MD, or there are no reads: nothing to fetch
344
+ if (start >= end) {
345
+ return;
346
+ }
347
+ const ref = await this.getReferenceRegion(refName, start, end, opts);
348
+ if (!ref) {
349
+ return;
350
+ }
351
+ for (let i = 0, l = records.length; i < l; i++) {
352
+ const record = records[i];
353
+ if (record.ref_id === chrId &&
354
+ !record.NUMERIC_MD &&
355
+ record.setReference &&
356
+ (0, reference_ts_1.referenceCovers)(ref, record.start, record.end)) {
357
+ record.setReference(ref);
358
+ }
359
+ }
273
360
  }
274
361
  // Parsed records for a chunk, reading and decompressing it only on a miss.
275
362
  // Every path that wants a chunk's features goes through here — mate lookups
@@ -288,7 +375,7 @@ class BamFile {
288
375
  const entry = await this.chunkFeatureCache.get(chunk, opts.signal);
289
376
  return entry.features;
290
377
  }
291
- async _fetchChunkFeatures(chunks, chrId, min, max, opts = {}) {
378
+ async _fetchChunkFeatures(chunks, chrId, chr, min, max, opts = {}) {
292
379
  const { viewAsPairs, onProgress } = opts;
293
380
  const result = [];
294
381
  let totalBytes = 0;
@@ -375,6 +462,10 @@ class BamFile {
375
462
  result.push(pairs[i]);
376
463
  }
377
464
  }
465
+ // After the pairs, so a mate fetched from another chunk is resolved on the
466
+ // same terms as the reads it was fetched for. A no-op without
467
+ // `fetchReferenceSequence`, which is the default.
468
+ await this._applyReferenceSequence(result, chrId, chr, opts);
378
469
  return result;
379
470
  }
380
471
  async fetchPairs(chrId, records, opts) {