@gmod/bam 8.5.0 → 8.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -226
- package/dist/bamFile.d.ts +94 -1
- package/dist/bamFile.js +94 -3
- package/dist/bamFile.js.map +1 -1
- package/dist/htsget.d.ts +8 -1
- package/dist/htsget.js +4 -1
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +5 -1
- package/dist/index.js +15 -1
- package/dist/index.js.map +1 -1
- package/dist/mismatches.d.ts +114 -0
- package/dist/mismatches.js +397 -0
- package/dist/mismatches.js.map +1 -0
- package/dist/record.d.ts +45 -0
- package/dist/record.js +76 -10
- package/dist/record.js.map +1 -1
- package/dist/reference.d.ts +45 -0
- package/dist/reference.js +61 -0
- package/dist/reference.js.map +1 -0
- package/dist/seqAlphabet.d.ts +3 -0
- package/dist/seqAlphabet.js +14 -0
- package/dist/seqAlphabet.js.map +1 -0
- package/esm/bamFile.d.ts +94 -1
- package/esm/bamFile.js +94 -3
- package/esm/bamFile.js.map +1 -1
- package/esm/htsget.d.ts +8 -1
- package/esm/htsget.js +4 -1
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +5 -1
- package/esm/index.js +6 -0
- package/esm/index.js.map +1 -1
- package/esm/mismatches.d.ts +114 -0
- package/esm/mismatches.js +393 -0
- package/esm/mismatches.js.map +1 -0
- package/esm/record.d.ts +45 -0
- package/esm/record.js +69 -3
- package/esm/record.js.map +1 -1
- package/esm/reference.d.ts +45 -0
- package/esm/reference.js +55 -0
- package/esm/reference.js.map +1 -0
- package/esm/seqAlphabet.d.ts +3 -0
- package/esm/seqAlphabet.js +11 -0
- package/esm/seqAlphabet.js.map +1 -0
- package/package.json +1 -1
- package/src/bamFile.ts +178 -1
- package/src/htsget.ts +16 -2
- package/src/index.ts +26 -1
- package/src/mismatches.ts +623 -0
- package/src/record.ts +95 -3
- package/src/reference.ts +87 -0
- package/src/seqAlphabet.ts +10 -0
package/README.md
CHANGED
|
@@ -21,18 +21,81 @@ const records = await bam.getRecordsForRange('ctgA', 0, 50000)
|
|
|
21
21
|
```
|
|
22
22
|
|
|
23
23
|
Coordinates are 0-based half-open (not the same as `samtools view` inputs).
|
|
24
|
-
`bamPath` reads a local file, so it is node-only; in the browser pass a
|
|
25
|
-
filehandle
|
|
24
|
+
`bamPath` reads a local file, so it is node-only; in the browser pass a URL or a
|
|
25
|
+
generic-filehandle2 filehandle instead:
|
|
26
26
|
|
|
27
27
|
```typescript
|
|
28
|
-
import { BamFile } from '@gmod/bam'
|
|
29
|
-
|
|
30
28
|
const bam = new BamFile({
|
|
31
29
|
bamUrl: 'https://example.com/yourfile.bam',
|
|
32
30
|
baiUrl: 'https://example.com/yourfile.bam.bai',
|
|
33
31
|
})
|
|
34
32
|
```
|
|
35
33
|
|
|
34
|
+
Records come back unfiltered, and are shared between overlapping queries — treat
|
|
35
|
+
them as read-only. Filter them yourself with the flag helpers and `getTag`,
|
|
36
|
+
which decodes one tag instead of all of them:
|
|
37
|
+
|
|
38
|
+
```typescript
|
|
39
|
+
const records = (await bam.getRecordsForRange('chr1', 0, 100000)).filter(
|
|
40
|
+
r => r.isProperlyPaired() && !r.isSecondary() && r.getTag('RG') === 'rg1',
|
|
41
|
+
)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Mismatches
|
|
45
|
+
|
|
46
|
+
`record.getMismatches()` gives every difference between a read and the reference
|
|
47
|
+
— substitutions, insertions, deletions, reference skips and clips — without you
|
|
48
|
+
having to interpret `CIGAR` and `MD` yourself. There is a callback form,
|
|
49
|
+
`record.forEachMismatch(cb, opts?)`, which allocates nothing per difference and
|
|
50
|
+
takes a reference window to report within.
|
|
51
|
+
|
|
52
|
+
Substitutions need either an `MD` tag on the read or the reference bases, and
|
|
53
|
+
most aligners leave `MD` off. `fetchReferenceSequence` is how you supply them:
|
|
54
|
+
|
|
55
|
+
```typescript
|
|
56
|
+
const bam = new BamFile({
|
|
57
|
+
bamPath: 'test.bam',
|
|
58
|
+
fetchReferenceSequence: async (refName, start, end) =>
|
|
59
|
+
myGenome.getSequence(refName, start, end),
|
|
60
|
+
})
|
|
61
|
+
|
|
62
|
+
// one sequence fetch for the whole query, and only if some read needs it
|
|
63
|
+
const records = await bam.getRecordsForRange('ctgA', 0, 50000)
|
|
64
|
+
records[0].getMismatches()
|
|
65
|
+
// [{ code: 88 /* 'X' */, refPos: 188, length: 1, bases: 'A', qual: 17,
|
|
66
|
+
// refBaseCode: 84 /* 'T' */, clipLength: 0 }, ...]
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Without it, a read lacking `MD` still reports its indels and clips, but no
|
|
70
|
+
substitutions — nothing in the record says where they are. See
|
|
71
|
+
[docs/api.md](docs/api.md#mismatches) for the field meanings and for reads
|
|
72
|
+
longer than the region you are looking at.
|
|
73
|
+
|
|
74
|
+
## Decompressing on a worker pool
|
|
75
|
+
|
|
76
|
+
BGZF decompression is 70-90% of a cold query, and BGZF blocks are independently
|
|
77
|
+
inflatable. Hand `BamFile` a
|
|
78
|
+
[`@gmod/bgzf-filehandle`](https://github.com/GMOD/bgzf-filehandle) worker pool
|
|
79
|
+
and it inflates chunks there instead of on the calling thread — measured
|
|
80
|
+
2.7-4.1x on the pool's own fixtures.
|
|
81
|
+
|
|
82
|
+
```typescript
|
|
83
|
+
import { getSharedWorkerPool } from '@gmod/bgzf-filehandle'
|
|
84
|
+
|
|
85
|
+
const bam = new BamFile({
|
|
86
|
+
bamUrl: 'https://example.com/yourfile.bam',
|
|
87
|
+
// the pending promise is fine — it is awaited at the point of use
|
|
88
|
+
bgzfWorkerPool: getSharedWorkerPool(),
|
|
89
|
+
})
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
No cross-origin isolation is needed. `getSharedWorkerPool()` gives back
|
|
93
|
+
`undefined` under node, or anywhere Workers cannot be created, which keeps the
|
|
94
|
+
in-process path — so this is safe to pass unconditionally. bam-js never creates
|
|
95
|
+
a pool on its own: the thread budget belongs to the consumer. For worker counts,
|
|
96
|
+
lifecycle and the pool's own benchmarks, see
|
|
97
|
+
[bgzf-filehandle's worker pool docs](https://github.com/GMOD/bgzf-filehandle/blob/main/docs/worker-pool.md).
|
|
98
|
+
|
|
36
99
|
## Usage with htsget
|
|
37
100
|
|
|
38
101
|
```typescript
|
|
@@ -46,10 +109,8 @@ const records = await bam.getRecordsForRange('1', 2000000, 2000001)
|
|
|
46
109
|
```
|
|
47
110
|
|
|
48
111
|
htsget fetches the server's range as-is, so `viewAsPairs`, `pairAcrossChr` and
|
|
49
|
-
`maxInsertSize` are ignored.
|
|
50
|
-
|
|
51
|
-
For a server that requires authentication, pass a `fetch` that adds the bearer
|
|
52
|
-
token the spec calls for:
|
|
112
|
+
`maxInsertSize` are ignored. For a server that requires authentication, pass a
|
|
113
|
+
`fetch` that adds the bearer token:
|
|
53
114
|
|
|
54
115
|
```typescript
|
|
55
116
|
const bam = new HtsgetFile({
|
|
@@ -63,227 +124,19 @@ const bam = new HtsgetFile({
|
|
|
63
124
|
})
|
|
64
125
|
```
|
|
65
126
|
|
|
66
|
-
Your `fetch` is called for the ticket request
|
|
67
|
-
ticket points at,
|
|
68
|
-
|
|
69
|
-
in each url's own `headers` field, which is applied either way.
|
|
70
|
-
|
|
71
|
-
## Documentation
|
|
72
|
-
|
|
73
|
-
### BamFile constructor
|
|
74
|
-
|
|
75
|
-
- `bamPath`/`bamUrl`/`bamFilehandle` - local path, remote URL, or a
|
|
76
|
-
generic-filehandle2 object
|
|
77
|
-
- `baiPath`/`baiUrl`/`baiFilehandle` - BAI index. Defaults to the `.bai` sibling
|
|
78
|
-
of `bamPath`/`bamUrl`
|
|
79
|
-
- `csiPath`/`csiUrl`/`csiFilehandle` - CSI index, required for chromosomes
|
|
80
|
-
longer than 2^29
|
|
81
|
-
- `renameRefSeqs` - `(refName: string) => string` applied to header ref names
|
|
82
|
-
- `recordClass` - custom class extending `BamRecord` (see below)
|
|
83
|
-
- `maxCacheBytes` - ceiling for the parsed-chunk cache, in decompressed bytes.
|
|
84
|
-
default: 1GB, `Infinity` for none. See [Caching](#caching)
|
|
85
|
-
- `cacheIdleTimeoutMs` - drop a cached chunk once nothing has read it for this
|
|
86
|
-
long. default: 3 minutes; `0` disables it. See [Caching](#caching)
|
|
87
|
-
|
|
88
|
-
The `path`/`url` forms are convenience wrappers for generic-filehandle2's
|
|
89
|
-
`LocalFile` and `RemoteFile`.
|
|
90
|
-
|
|
91
|
-
### Caching
|
|
92
|
-
|
|
93
|
-
Parsed chunks — the unit the BAM index hands out — are cached, so overlapping
|
|
94
|
-
and adjacent queries reuse decompressed records instead of re-fetching them. Two
|
|
95
|
-
options bound that cache, and they answer different questions.
|
|
96
|
-
|
|
97
|
-
**`maxCacheBytes` is a ceiling under load, not a limit on what you can ask
|
|
98
|
-
for.** Nothing is ever refused for being too large: a chunk bigger than the
|
|
99
|
-
whole budget is still cached, reads in flight are never evicted, and eviction
|
|
100
|
-
only drops a value that has already been returned once. The worst a budget can
|
|
101
|
-
cost you is a re-read. It can make a query slower; it can never make one fail or
|
|
102
|
-
come back short.
|
|
103
|
-
|
|
104
|
-
**It binds less often than its size suggests.** On the deepest data we measure —
|
|
105
|
-
1000x coverage long reads, 240 windows over six laps — the cache settles at
|
|
106
|
-
573MB across 60 entries and eviction never runs at the 1GB default. Treat it as
|
|
107
|
-
a backstop against a session that pans forever, not as an operating constraint.
|
|
108
|
-
|
|
109
|
-
**Don't pick a number between one query and several.** Below one query's working
|
|
110
|
-
set the cache inverts: each chunk is evicted before the next pan can reuse it,
|
|
111
|
-
so the hit rate is zero, the full re-decompress is paid every time, and the
|
|
112
|
-
unevictable entries retain the memory anyway. At a 200MB budget on that same
|
|
113
|
-
file the cache holds exactly one entry, because one chunk there decompresses to
|
|
114
|
-
181MB. Either size it above the working set or pass `Infinity` and bound memory
|
|
115
|
-
some other way.
|
|
116
|
-
|
|
117
|
-
**`cacheIdleTimeoutMs` is the only thing that gives memory back.**
|
|
118
|
-
`maxCacheBytes` is enforced when a read settles, so an idle cache sits at
|
|
119
|
-
whatever it reached and never lowers — and for a page that holds a `BamFile` for
|
|
120
|
-
the life of a track, that resting level is the number that actually matters. The
|
|
121
|
-
idle sweep is what makes a generous ceiling affordable, by turning it into a
|
|
122
|
-
peak under panning rather than a level a parked tab holds indefinitely. The
|
|
123
|
-
clock runs from the last _read_ of a chunk, or from its parse landing if nothing
|
|
124
|
-
has read it since, so panning back and forth over one region never expires it
|
|
125
|
-
and a slow chunk still gets the full timeout to be reused in. Measured on a pan
|
|
126
|
-
that held 331MB: 0MB once idle.
|
|
127
|
-
|
|
128
|
-
**Neither bounds peak memory**, and on deep data the gap is not small:
|
|
129
|
-
|
|
130
|
-
- Up to six chunk reads run at once and an in-flight read is never evicted. At
|
|
131
|
-
181MB a chunk, that is ~476MB in flight before retention holds anything.
|
|
132
|
-
- A query holds every chunk it parsed until it returns, whether or not the cache
|
|
133
|
-
still does.
|
|
134
|
-
- An entry is weighed once, when its read settles, and records grow after that.
|
|
135
|
-
`end`, `CIGAR` and `tags` each memoize onto the record the first time they are
|
|
136
|
-
read, which is what a renderer does to every visible read — measured at +38%
|
|
137
|
-
over the weighed size.
|
|
138
|
-
|
|
139
|
-
So these are bounds on retained decompressed bytes, not on the heap. Size
|
|
140
|
-
against what you want to keep, and bound total memory at a level that can see
|
|
141
|
-
the whole process.
|
|
142
|
-
|
|
143
|
-
### HtsgetFile constructor
|
|
144
|
-
|
|
145
|
-
- `baseUrl` - htsget reads endpoint, e.g. `https://htsget.example.com/reads`
|
|
146
|
-
- `trackId` - id of the resource under `baseUrl`
|
|
147
|
-
- `fetch` - `fetch` replacement for adding auth headers (see above)
|
|
148
|
-
- `recordClass` - custom class extending `BamRecord` (see below)
|
|
149
|
-
|
|
150
|
-
### async getRecordsForRange(refName, start, end, opts?)
|
|
151
|
-
|
|
152
|
-
- `refName` - chromosome to fetch from
|
|
153
|
-
- `start`/`end` - 0-based half-open coordinates
|
|
154
|
-
- `opts.signal` - `AbortSignal` to stop processing
|
|
155
|
-
- `opts.viewAsPairs` - re-dispatch requests to find mate pairs. default: false
|
|
156
|
-
- `opts.pairAcrossChr` - let `viewAsPairs` pair across chromosomes. default:
|
|
157
|
-
false
|
|
158
|
-
- `opts.maxInsertSize` - distance limit for `viewAsPairs` within a chromosome.
|
|
159
|
-
default: 200kb
|
|
160
|
-
- `opts.onProgress` - `(bytesDownloaded, totalBytes?) => void`, called per BGZF
|
|
161
|
-
chunk for a determinate progress bar
|
|
162
|
-
|
|
163
|
-
Returned records are cached and shared between overlapping queries, so treat
|
|
164
|
-
them as read-only — attaching your own fields to a record mutates it for every
|
|
165
|
-
other query holding it.
|
|
166
|
-
|
|
167
|
-
Records come back unfiltered. Filter them yourself with the flag helpers and
|
|
168
|
-
`getTag`, which decodes one tag instead of all of them:
|
|
127
|
+
Your `fetch` is called for the ticket request _and_ for the data-block urls the
|
|
128
|
+
ticket points at, which may live on a third-party host — so only attach
|
|
129
|
+
credentials to hosts you trust.
|
|
169
130
|
|
|
170
|
-
|
|
171
|
-
const records = (await bam.getRecordsForRange('chr1', 0, 100000)).filter(
|
|
172
|
-
r => r.isProperlyPaired() && !r.isSecondary() && r.getTag('RG') === 'rg1',
|
|
173
|
-
)
|
|
174
|
-
```
|
|
175
|
-
|
|
176
|
-
### async getHeader(opts?)
|
|
177
|
-
|
|
178
|
-
Returns the parsed SAM header. Called automatically by the query methods and
|
|
179
|
-
cached, so you only need it when you want the header itself.
|
|
180
|
-
`getHeaderText(opts?)` returns the raw header string.
|
|
181
|
-
|
|
182
|
-
### async indexCov(refName, start?, end?)
|
|
183
|
-
|
|
184
|
-
Returns `{start, end, score}` features estimating read density over 16kb
|
|
185
|
-
windows, derived from the BAI linear index. CSI has no linear index, so a
|
|
186
|
-
CSI-indexed file returns `[]`.
|
|
187
|
-
|
|
188
|
-
### async lineCount(refName)
|
|
189
|
-
|
|
190
|
-
Number of records on `refName` from the index's pseudo-bin (bin 37450 in BAI,
|
|
191
|
-
`n_mapped` in the SAM spec), or 0 if `refName` is absent.
|
|
131
|
+
## Docs
|
|
192
132
|
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
Compressed bytes the given `{refName, start, end}[]` would fetch — useful for
|
|
200
|
-
warning before a large query.
|
|
201
|
-
|
|
202
|
-
### clearFeatureCache()
|
|
203
|
-
|
|
204
|
-
Drops the parsed-chunk cache immediately, rather than waiting for
|
|
205
|
-
`maxCacheBytes` or `cacheIdleTimeoutMs` to reclaim it. See [Caching](#caching).
|
|
206
|
-
|
|
207
|
-
### BamRecord
|
|
208
|
-
|
|
209
|
-
```typescript
|
|
210
|
-
// Core alignment fields
|
|
211
|
-
record.fileOffset // "file offset" based id -- not a true file offset
|
|
212
|
-
record.ref_id // numerical sequence id from SAM header
|
|
213
|
-
record.start // 0-based start coordinate
|
|
214
|
-
record.end // 0-based end coordinate
|
|
215
|
-
record.name // QNAME
|
|
216
|
-
record.seq // sequence string
|
|
217
|
-
record.qual // Uint8Array of quality scores (null if SEQ is empty)
|
|
218
|
-
record.CIGAR // CIGAR string e.g. "50M2I48M"
|
|
219
|
-
record.flags // SAM flags integer
|
|
220
|
-
record.mq // mapping quality (undefined if 255)
|
|
221
|
-
record.strand // 1 or -1
|
|
222
|
-
record.template_length // TLEN
|
|
223
|
-
|
|
224
|
-
// Mate info
|
|
225
|
-
record.next_refid
|
|
226
|
-
record.next_pos
|
|
227
|
-
|
|
228
|
-
// Auxiliary data
|
|
229
|
-
record.tags // all aux tags e.g. {MD: "100", NM: 0}
|
|
230
|
-
record.getTag('MD') // one tag, without decoding the rest
|
|
231
|
-
record.getTagRaw('MD') // string tag as Uint8Array, skipping string conversion
|
|
232
|
-
|
|
233
|
-
// Typed-array views, for rendering without allocating strings
|
|
234
|
-
record.NUMERIC_MD // MD tag as Uint8Array
|
|
235
|
-
record.NUMERIC_CIGAR // Uint32Array of packed CIGAR operations
|
|
236
|
-
record.NUMERIC_SEQ // Uint8Array of 4-bit encoded sequence
|
|
237
|
-
|
|
238
|
-
// Flag methods
|
|
239
|
-
record.isPaired()
|
|
240
|
-
record.isProperlyPaired()
|
|
241
|
-
record.isSegmentUnmapped()
|
|
242
|
-
record.isMateUnmapped()
|
|
243
|
-
record.isReverseComplemented()
|
|
244
|
-
record.isMateReverseComplemented()
|
|
245
|
-
record.isRead1()
|
|
246
|
-
record.isRead2()
|
|
247
|
-
record.isSecondary()
|
|
248
|
-
record.isFailedQc()
|
|
249
|
-
record.isDuplicate()
|
|
250
|
-
record.isSupplementary()
|
|
251
|
-
|
|
252
|
-
// Utility
|
|
253
|
-
record.seqAt(idx) // single base at position
|
|
254
|
-
record.toJSON()
|
|
255
|
-
```
|
|
256
|
-
|
|
257
|
-
### Custom BamRecord class
|
|
258
|
-
|
|
259
|
-
```typescript
|
|
260
|
-
import { BamFile, BamRecord } from '@gmod/bam'
|
|
261
|
-
|
|
262
|
-
class CustomBamRecord extends BamRecord {
|
|
263
|
-
get customProperty() {
|
|
264
|
-
return `custom-${this.name}`
|
|
265
|
-
}
|
|
266
|
-
}
|
|
267
|
-
|
|
268
|
-
const bam = new BamFile<CustomBamRecord>({
|
|
269
|
-
bamPath: 'test.bam',
|
|
270
|
-
recordClass: CustomBamRecord,
|
|
271
|
-
})
|
|
272
|
-
|
|
273
|
-
// records are typed as CustomBamRecord[]
|
|
274
|
-
const records = await bam.getRecordsForRange('ctgA', 0, 50000)
|
|
275
|
-
console.log(records[0].customProperty)
|
|
276
|
-
```
|
|
133
|
+
- [docs/api.md](docs/api.md) — every constructor option, method and `BamRecord`
|
|
134
|
+
field, plus custom record classes
|
|
135
|
+
- [docs/caching.md](docs/caching.md) — sizing the parsed-chunk cache
|
|
136
|
+
- [agent-docs/adr/](agent-docs/adr/) — the measurements behind the performance
|
|
137
|
+
and caching decisions
|
|
138
|
+
- [CONTRIBUTING.md](CONTRIBUTING.md) — development and release steps
|
|
277
139
|
|
|
278
140
|
## License
|
|
279
141
|
|
|
280
142
|
MIT © [Colin Diesh](https://github.com/cmdcolin)
|
|
281
|
-
|
|
282
|
-
## Publishing
|
|
283
|
-
|
|
284
|
-
[Trusted publishing](https://docs.npmjs.com/about-trusted-publishing) via GitHub
|
|
285
|
-
Actions.
|
|
286
|
-
|
|
287
|
-
```bash
|
|
288
|
-
pnpm version patch # or minor/major
|
|
289
|
-
```
|
package/dist/bamFile.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import BAI from './bai.ts';
|
|
|
3
3
|
import CSI from './csi.ts';
|
|
4
4
|
import BAMFeature from './record.ts';
|
|
5
5
|
import type Chunk from './chunk.ts';
|
|
6
|
+
import type { PackedReference } from './reference.ts';
|
|
6
7
|
import type { BamOpts, BaseOpts } from './util.ts';
|
|
7
8
|
import type { BgzfWorkerPool } from '@gmod/bgzf-filehandle';
|
|
8
9
|
import type { SharedBudget } from '@gmod/shared-read-cache';
|
|
@@ -17,7 +18,35 @@ export interface BamRecordLike {
|
|
|
17
18
|
next_refid: number;
|
|
18
19
|
flags: number;
|
|
19
20
|
tags: Record<string, unknown>;
|
|
21
|
+
/**
|
|
22
|
+
* Optional, and read only to decide whether a read needs reference bases at
|
|
23
|
+
* all — a read with an MD tag carries its own. A `recordClass` without it is
|
|
24
|
+
* treated as having no MD, i.e. as always wanting the reference.
|
|
25
|
+
*/
|
|
26
|
+
NUMERIC_MD?: Uint8Array | undefined;
|
|
27
|
+
/**
|
|
28
|
+
* Optional; see {@link BamRecord.setReference}. A `recordClass` that does not
|
|
29
|
+
* implement it simply never gets a reference bound, whatever
|
|
30
|
+
* `fetchReferenceSequence` returns.
|
|
31
|
+
*/
|
|
32
|
+
setReference?: (ref: PackedReference) => void;
|
|
20
33
|
}
|
|
34
|
+
/**
|
|
35
|
+
* Supplies reference bases for a region, so reads with no MD tag can still
|
|
36
|
+
* report their substitutions. See {@link BamFile}'s option of the same name.
|
|
37
|
+
*
|
|
38
|
+
* `refName` is the name the query used, unchanged — `renameRefSeqs` maps the
|
|
39
|
+
* FILE's names into the caller's namespace, and this callback is on the
|
|
40
|
+
* caller's side of that. `start`/`end` are 0-based half-open.
|
|
41
|
+
*
|
|
42
|
+
* **The bases returned must begin at `start`.** Returning fewer than asked for
|
|
43
|
+
* is fine and is taken as `[start, start + seq.length)` — the end of a contig,
|
|
44
|
+
* or a source declining to hand over a huge span — and reads the shorter region
|
|
45
|
+
* does not cover are then left unresolved. Returning bases from somewhere else,
|
|
46
|
+
* e.g. clipping the LEFT of the requested range, cannot be detected and
|
|
47
|
+
* resolves every read against the wrong position.
|
|
48
|
+
*/
|
|
49
|
+
export type ReferenceSequenceFetcher = (refName: string, start: number, end: number, opts?: BaseOpts) => Promise<string>;
|
|
21
50
|
export type BamRecordClass<T extends BamRecordLike = BAMFeature> = new (byteArray: Uint8Array, start: number, end: number, fileOffset: number, dataView: DataView) => T;
|
|
22
51
|
export declare const BAM_MAGIC = 21840194;
|
|
23
52
|
interface ChunkEntry<T> {
|
|
@@ -46,7 +75,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
46
75
|
private RecordClass;
|
|
47
76
|
/** see the constructor option of the same name */
|
|
48
77
|
private bgzfWorkerPool?;
|
|
49
|
-
|
|
78
|
+
/** see the constructor option of the same name */
|
|
79
|
+
fetchReferenceSequence?: ReferenceSequenceFetcher;
|
|
80
|
+
constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs, recordClass, maxCacheBytes, cacheIdleTimeoutMs, cacheBudget, bgzfWorkerPool, fetchReferenceSequence, }: {
|
|
50
81
|
bamFilehandle?: GenericFilehandle;
|
|
51
82
|
bamPath?: string;
|
|
52
83
|
bamUrl?: string;
|
|
@@ -135,6 +166,33 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
135
166
|
* in-process path, so this is safe to pass unconditionally.
|
|
136
167
|
*/
|
|
137
168
|
bgzfWorkerPool?: BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined;
|
|
169
|
+
/**
|
|
170
|
+
* Reference bases for a region, as `(refName, start, end, opts) =>
|
|
171
|
+
* Promise<string>`. Only reads that carry no `MD` tag need it, and it is
|
|
172
|
+
* only what {@link BamRecord.forEachMismatch} uses — nothing else in a
|
|
173
|
+
* query touches the reference.
|
|
174
|
+
*
|
|
175
|
+
* Without it, a read lacking MD still reports its indels and clips but no
|
|
176
|
+
* substitutions at all: neither the CIGAR nor SEQ says where they are.
|
|
177
|
+
* That is most aligners' output — minimap2 and bwa both leave MD off unless
|
|
178
|
+
* asked — so this is the difference between mismatches rendering and not.
|
|
179
|
+
*
|
|
180
|
+
* `getRecordsForRange` calls it at most ONCE per query, for the span of the
|
|
181
|
+
* reads that need it, and binds the result to each of them (see
|
|
182
|
+
* {@link BamRecord.setReference}). A query whose reads all carry MD does
|
|
183
|
+
* not call it at all.
|
|
184
|
+
*
|
|
185
|
+
* **The span asked for is the reads' union, not the query's**: the query's
|
|
186
|
+
* range plus however far its edge reads overhang it. Usually that is a few
|
|
187
|
+
* hundred bases more for Illumina and a couple of Mb for ultra-long ONT,
|
|
188
|
+
* but a BAM holding whole chromosomes as reads can make it a chromosome.
|
|
189
|
+
* Nothing here clamps that for you, because only you know what your
|
|
190
|
+
* sequence source can afford — clamp inside the callback and return the
|
|
191
|
+
* shorter region; reads it does not cover are then simply left unresolved,
|
|
192
|
+
* and {@link BamRecord.forEachMismatch}'s `opts.ref` is the windowed way to
|
|
193
|
+
* handle one of those reads.
|
|
194
|
+
*/
|
|
195
|
+
fetchReferenceSequence?: ReferenceSequenceFetcher;
|
|
138
196
|
});
|
|
139
197
|
getHeaderPre(opts?: BaseOpts): Promise<{
|
|
140
198
|
tag: string;
|
|
@@ -201,6 +259,41 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
201
259
|
* it for every other query still holding that read. See ADR 0006.
|
|
202
260
|
*/
|
|
203
261
|
getRecordsForRange(chr: string, min: number, max: number, opts?: BamOpts): Promise<T[]>;
|
|
262
|
+
/**
|
|
263
|
+
* Reference bases for a region, packed for
|
|
264
|
+
* {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
|
|
265
|
+
* `fetchReferenceSequence` was configured.
|
|
266
|
+
*
|
|
267
|
+
* The way to resolve a read that is longer than the region you are looking at
|
|
268
|
+
* — a contig or a whole chromosome stored as one BAM read. Those are never
|
|
269
|
+
* bound to a record automatically, since binding a partial region to a
|
|
270
|
+
* record shared between queries is the hazard ADR 0006 is about; walking one
|
|
271
|
+
* with a window and a region of your own choosing has no such problem:
|
|
272
|
+
*
|
|
273
|
+
* ```js
|
|
274
|
+
* const ref = await bam.getReferenceRegion('chr1', start, end)
|
|
275
|
+
* record.forEachMismatch(cb, { ref, start, end })
|
|
276
|
+
* ```
|
|
277
|
+
*/
|
|
278
|
+
getReferenceRegion(refName: string, start: number, end: number, opts?: BaseOpts): Promise<PackedReference | undefined>;
|
|
279
|
+
/**
|
|
280
|
+
* Fetch reference bases for the reads in `records` that have no MD tag, and
|
|
281
|
+
* bind them to those reads so their substitutions resolve. A no-op unless
|
|
282
|
+
* `fetchReferenceSequence` was configured, and it issues at most one fetch.
|
|
283
|
+
*
|
|
284
|
+
* The span fetched is the union of the reads that need it, NOT the queried
|
|
285
|
+
* range: a read overhanging the range is only bound to a region covering all
|
|
286
|
+
* of it, because the records are shared between queries and a binding that
|
|
287
|
+
* varied by query would make one query's reads answer out of another's region
|
|
288
|
+
* (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
|
|
289
|
+
* `pairAcrossChr` — are left out for the same reason, and unbound.
|
|
290
|
+
*
|
|
291
|
+
* Nothing clamps that union, since only the callback knows what its sequence
|
|
292
|
+
* source can afford; a callback that returns a shorter region than it was
|
|
293
|
+
* asked for leaves the reads it does not cover unbound, which is the same
|
|
294
|
+
* outcome by a route the consumer controls.
|
|
295
|
+
*/
|
|
296
|
+
protected _applyReferenceSequence(records: T[], chrId: number, refName: string, opts?: BaseOpts): Promise<void>;
|
|
204
297
|
private _cachedChunkFeatures;
|
|
205
298
|
private _fetchChunkFeatures;
|
|
206
299
|
fetchPairs(chrId: number, records: T[], opts: BamOpts): Promise<T[]>;
|
package/dist/bamFile.js
CHANGED
|
@@ -12,6 +12,7 @@ const bai_ts_1 = __importDefault(require("./bai.js"));
|
|
|
12
12
|
const csi_ts_1 = __importDefault(require("./csi.js"));
|
|
13
13
|
const nullFilehandle_ts_1 = __importDefault(require("./nullFilehandle.js"));
|
|
14
14
|
const record_ts_1 = __importDefault(require("./record.js"));
|
|
15
|
+
const reference_ts_1 = require("./reference.js");
|
|
15
16
|
const sam_ts_1 = require("./sam.js");
|
|
16
17
|
const util_ts_1 = require("./util.js");
|
|
17
18
|
exports.BAM_MAGIC = 21840194;
|
|
@@ -112,9 +113,12 @@ class BamFile {
|
|
|
112
113
|
RecordClass;
|
|
113
114
|
/** see the constructor option of the same name */
|
|
114
115
|
bgzfWorkerPool;
|
|
115
|
-
|
|
116
|
+
/** see the constructor option of the same name */
|
|
117
|
+
fetchReferenceSequence;
|
|
118
|
+
constructor({ bamFilehandle, bamPath, bamUrl, baiPath, baiFilehandle, baiUrl, csiPath, csiFilehandle, csiUrl, htsget, renameRefSeqs = n => n, recordClass, maxCacheBytes = exports.DEFAULT_MAX_CACHE_BYTES, cacheIdleTimeoutMs = exports.DEFAULT_CACHE_IDLE_TIMEOUT_MS, cacheBudget, bgzfWorkerPool, fetchReferenceSequence, }) {
|
|
116
119
|
this.renameRefSeq = renameRefSeqs;
|
|
117
120
|
this.bgzfWorkerPool = bgzfWorkerPool;
|
|
121
|
+
this.fetchReferenceSequence = fetchReferenceSequence;
|
|
118
122
|
this.RecordClass = (recordClass ?? record_ts_1.default);
|
|
119
123
|
this.chunkFeatureCache = new shared_read_cache_1.SharedReadCache({
|
|
120
124
|
maxSize: maxCacheBytes,
|
|
@@ -269,7 +273,90 @@ class BamFile {
|
|
|
269
273
|
return [];
|
|
270
274
|
}
|
|
271
275
|
const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts);
|
|
272
|
-
return this._fetchChunkFeatures(chunks, chrId, min, max, opts);
|
|
276
|
+
return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts);
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Reference bases for a region, packed for
|
|
280
|
+
* {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
|
|
281
|
+
* `fetchReferenceSequence` was configured.
|
|
282
|
+
*
|
|
283
|
+
* The way to resolve a read that is longer than the region you are looking at
|
|
284
|
+
* — a contig or a whole chromosome stored as one BAM read. Those are never
|
|
285
|
+
* bound to a record automatically, since binding a partial region to a
|
|
286
|
+
* record shared between queries is the hazard ADR 0006 is about; walking one
|
|
287
|
+
* with a window and a region of your own choosing has no such problem:
|
|
288
|
+
*
|
|
289
|
+
* ```js
|
|
290
|
+
* const ref = await bam.getReferenceRegion('chr1', start, end)
|
|
291
|
+
* record.forEachMismatch(cb, { ref, start, end })
|
|
292
|
+
* ```
|
|
293
|
+
*/
|
|
294
|
+
async getReferenceRegion(refName, start, end, opts) {
|
|
295
|
+
const fetchReferenceSequence = this.fetchReferenceSequence;
|
|
296
|
+
if (!fetchReferenceSequence) {
|
|
297
|
+
return undefined;
|
|
298
|
+
}
|
|
299
|
+
// Packed once for the region rather than per read — the walk compares two
|
|
300
|
+
// bases per byte against the read's own packed SEQ, and this is the only
|
|
301
|
+
// per-base pass in it. Length comes from what came back rather than from
|
|
302
|
+
// what was asked for, so a callback that clips at the end of the contig
|
|
303
|
+
// still leaves the bases it did return usable.
|
|
304
|
+
return (0, reference_ts_1.packReference)(await fetchReferenceSequence(refName, start, end, opts), start);
|
|
305
|
+
}
|
|
306
|
+
/**
|
|
307
|
+
* Fetch reference bases for the reads in `records` that have no MD tag, and
|
|
308
|
+
* bind them to those reads so their substitutions resolve. A no-op unless
|
|
309
|
+
* `fetchReferenceSequence` was configured, and it issues at most one fetch.
|
|
310
|
+
*
|
|
311
|
+
* The span fetched is the union of the reads that need it, NOT the queried
|
|
312
|
+
* range: a read overhanging the range is only bound to a region covering all
|
|
313
|
+
* of it, because the records are shared between queries and a binding that
|
|
314
|
+
* varied by query would make one query's reads answer out of another's region
|
|
315
|
+
* (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
|
|
316
|
+
* `pairAcrossChr` — are left out for the same reason, and unbound.
|
|
317
|
+
*
|
|
318
|
+
* Nothing clamps that union, since only the callback knows what its sequence
|
|
319
|
+
* source can afford; a callback that returns a shorter region than it was
|
|
320
|
+
* asked for leaves the reads it does not cover unbound, which is the same
|
|
321
|
+
* outcome by a route the consumer controls.
|
|
322
|
+
*/
|
|
323
|
+
async _applyReferenceSequence(records, chrId, refName, opts = {}) {
|
|
324
|
+
const fetchReferenceSequence = this.fetchReferenceSequence;
|
|
325
|
+
if (!fetchReferenceSequence) {
|
|
326
|
+
return;
|
|
327
|
+
}
|
|
328
|
+
let start = Infinity;
|
|
329
|
+
let end = 0;
|
|
330
|
+
for (let i = 0, l = records.length; i < l; i++) {
|
|
331
|
+
const record = records[i];
|
|
332
|
+
if (record.ref_id === chrId &&
|
|
333
|
+
!record.NUMERIC_MD &&
|
|
334
|
+
record.setReference) {
|
|
335
|
+
if (record.start < start) {
|
|
336
|
+
start = record.start;
|
|
337
|
+
}
|
|
338
|
+
if (record.end > end) {
|
|
339
|
+
end = record.end;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
// every read carries MD, or there are no reads: nothing to fetch
|
|
344
|
+
if (start >= end) {
|
|
345
|
+
return;
|
|
346
|
+
}
|
|
347
|
+
const ref = await this.getReferenceRegion(refName, start, end, opts);
|
|
348
|
+
if (!ref) {
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
351
|
+
for (let i = 0, l = records.length; i < l; i++) {
|
|
352
|
+
const record = records[i];
|
|
353
|
+
if (record.ref_id === chrId &&
|
|
354
|
+
!record.NUMERIC_MD &&
|
|
355
|
+
record.setReference &&
|
|
356
|
+
(0, reference_ts_1.referenceCovers)(ref, record.start, record.end)) {
|
|
357
|
+
record.setReference(ref);
|
|
358
|
+
}
|
|
359
|
+
}
|
|
273
360
|
}
|
|
274
361
|
// Parsed records for a chunk, reading and decompressing it only on a miss.
|
|
275
362
|
// Every path that wants a chunk's features goes through here — mate lookups
|
|
@@ -288,7 +375,7 @@ class BamFile {
|
|
|
288
375
|
const entry = await this.chunkFeatureCache.get(chunk, opts.signal);
|
|
289
376
|
return entry.features;
|
|
290
377
|
}
|
|
291
|
-
async _fetchChunkFeatures(chunks, chrId, min, max, opts = {}) {
|
|
378
|
+
async _fetchChunkFeatures(chunks, chrId, chr, min, max, opts = {}) {
|
|
292
379
|
const { viewAsPairs, onProgress } = opts;
|
|
293
380
|
const result = [];
|
|
294
381
|
let totalBytes = 0;
|
|
@@ -375,6 +462,10 @@ class BamFile {
|
|
|
375
462
|
result.push(pairs[i]);
|
|
376
463
|
}
|
|
377
464
|
}
|
|
465
|
+
// After the pairs, so a mate fetched from another chunk is resolved on the
|
|
466
|
+
// same terms as the reads it was fetched for. A no-op without
|
|
467
|
+
// `fetchReferenceSequence`, which is the default.
|
|
468
|
+
await this._applyReferenceSequence(result, chrId, chr, opts);
|
|
378
469
|
return result;
|
|
379
470
|
}
|
|
380
471
|
async fetchPairs(chrId, records, opts) {
|