@gmod/bam 8.5.1 → 8.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/dist/bamFile.d.ts +94 -1
- package/dist/bamFile.js +94 -3
- package/dist/bamFile.js.map +1 -1
- package/dist/htsget.d.ts +8 -1
- package/dist/htsget.js +4 -1
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +5 -1
- package/dist/index.js +15 -1
- package/dist/index.js.map +1 -1
- package/dist/mismatches.d.ts +114 -0
- package/dist/mismatches.js +397 -0
- package/dist/mismatches.js.map +1 -0
- package/dist/record.d.ts +45 -0
- package/dist/record.js +76 -10
- package/dist/record.js.map +1 -1
- package/dist/reference.d.ts +45 -0
- package/dist/reference.js +61 -0
- package/dist/reference.js.map +1 -0
- package/dist/seqAlphabet.d.ts +3 -0
- package/dist/seqAlphabet.js +14 -0
- package/dist/seqAlphabet.js.map +1 -0
- package/esm/bamFile.d.ts +94 -1
- package/esm/bamFile.js +94 -3
- package/esm/bamFile.js.map +1 -1
- package/esm/htsget.d.ts +8 -1
- package/esm/htsget.js +4 -1
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +5 -1
- package/esm/index.js +6 -0
- package/esm/index.js.map +1 -1
- package/esm/mismatches.d.ts +114 -0
- package/esm/mismatches.js +393 -0
- package/esm/mismatches.js.map +1 -0
- package/esm/record.d.ts +45 -0
- package/esm/record.js +69 -3
- package/esm/record.js.map +1 -1
- package/esm/reference.d.ts +45 -0
- package/esm/reference.js +55 -0
- package/esm/reference.js.map +1 -0
- package/esm/seqAlphabet.d.ts +3 -0
- package/esm/seqAlphabet.js +11 -0
- package/esm/seqAlphabet.js.map +1 -0
- package/package.json +1 -1
- package/src/bamFile.ts +178 -1
- package/src/htsget.ts +16 -2
- package/src/index.ts +26 -1
- package/src/mismatches.ts +623 -0
- package/src/record.ts +95 -3
- package/src/reference.ts +87 -0
- package/src/seqAlphabet.ts +10 -0
package/src/bamFile.ts
CHANGED
|
@@ -7,6 +7,7 @@ import BAI from './bai.ts'
|
|
|
7
7
|
import CSI from './csi.ts'
|
|
8
8
|
import NullFilehandle from './nullFilehandle.ts'
|
|
9
9
|
import BAMFeature from './record.ts'
|
|
10
|
+
import { packReference, referenceCovers } from './reference.ts'
|
|
10
11
|
import { parseHeaderText } from './sam.ts'
|
|
11
12
|
import {
|
|
12
13
|
MAX_CONCURRENT_CHUNK_READS,
|
|
@@ -16,6 +17,7 @@ import {
|
|
|
16
17
|
} from './util.ts'
|
|
17
18
|
|
|
18
19
|
import type Chunk from './chunk.ts'
|
|
20
|
+
import type { PackedReference } from './reference.ts'
|
|
19
21
|
import type { BamOpts, BaseOpts } from './util.ts'
|
|
20
22
|
import type { BgzfWorkerPool } from '@gmod/bgzf-filehandle'
|
|
21
23
|
import type { SharedBudget } from '@gmod/shared-read-cache'
|
|
@@ -31,8 +33,42 @@ export interface BamRecordLike {
|
|
|
31
33
|
next_refid: number
|
|
32
34
|
flags: number
|
|
33
35
|
tags: Record<string, unknown>
|
|
36
|
+
/**
|
|
37
|
+
* Optional, and read only to decide whether a read needs reference bases at
|
|
38
|
+
* all — a read with an MD tag carries its own. A `recordClass` without it is
|
|
39
|
+
* treated as having no MD, i.e. as always wanting the reference.
|
|
40
|
+
*/
|
|
41
|
+
NUMERIC_MD?: Uint8Array | undefined
|
|
42
|
+
/**
|
|
43
|
+
* Optional; see {@link BamRecord.setReference}. A `recordClass` that does not
|
|
44
|
+
* implement it simply never gets a reference bound, whatever
|
|
45
|
+
* `fetchReferenceSequence` returns.
|
|
46
|
+
*/
|
|
47
|
+
setReference?: (ref: PackedReference) => void
|
|
34
48
|
}
|
|
35
49
|
|
|
50
|
+
/**
|
|
51
|
+
* Supplies reference bases for a region, so reads with no MD tag can still
|
|
52
|
+
* report their substitutions. See {@link BamFile}'s option of the same name.
|
|
53
|
+
*
|
|
54
|
+
* `refName` is the name the query used, unchanged — `renameRefSeqs` maps the
|
|
55
|
+
* FILE's names into the caller's namespace, and this callback is on the
|
|
56
|
+
* caller's side of that. `start`/`end` are 0-based half-open.
|
|
57
|
+
*
|
|
58
|
+
* **The bases returned must begin at `start`.** Returning fewer than asked for
|
|
59
|
+
* is fine and is taken as `[start, start + seq.length)` — the end of a contig,
|
|
60
|
+
* or a source declining to hand over a huge span — and reads the shorter region
|
|
61
|
+
* does not cover are then left unresolved. Returning bases from somewhere else,
|
|
62
|
+
* e.g. clipping the LEFT of the requested range, cannot be detected and
|
|
63
|
+
* resolves every read against the wrong position.
|
|
64
|
+
*/
|
|
65
|
+
export type ReferenceSequenceFetcher = (
|
|
66
|
+
refName: string,
|
|
67
|
+
start: number,
|
|
68
|
+
end: number,
|
|
69
|
+
opts?: BaseOpts,
|
|
70
|
+
) => Promise<string>
|
|
71
|
+
|
|
36
72
|
export type BamRecordClass<T extends BamRecordLike = BAMFeature> = new (
|
|
37
73
|
byteArray: Uint8Array,
|
|
38
74
|
start: number,
|
|
@@ -174,6 +210,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
174
210
|
private bgzfWorkerPool?:
|
|
175
211
|
BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined
|
|
176
212
|
|
|
213
|
+
/** see the constructor option of the same name */
|
|
214
|
+
public fetchReferenceSequence?: ReferenceSequenceFetcher
|
|
215
|
+
|
|
177
216
|
constructor({
|
|
178
217
|
bamFilehandle,
|
|
179
218
|
bamPath,
|
|
@@ -191,6 +230,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
191
230
|
cacheIdleTimeoutMs = DEFAULT_CACHE_IDLE_TIMEOUT_MS,
|
|
192
231
|
cacheBudget,
|
|
193
232
|
bgzfWorkerPool,
|
|
233
|
+
fetchReferenceSequence,
|
|
194
234
|
}: {
|
|
195
235
|
bamFilehandle?: GenericFilehandle
|
|
196
236
|
bamPath?: string
|
|
@@ -281,9 +321,37 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
281
321
|
*/
|
|
282
322
|
bgzfWorkerPool?:
|
|
283
323
|
BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined
|
|
324
|
+
/**
|
|
325
|
+
* Reference bases for a region, as `(refName, start, end, opts) =>
|
|
326
|
+
* Promise<string>`. Only reads that carry no `MD` tag need it, and it is
|
|
327
|
+
* only what {@link BamRecord.forEachMismatch} uses — nothing else in a
|
|
328
|
+
* query touches the reference.
|
|
329
|
+
*
|
|
330
|
+
* Without it, a read lacking MD still reports its indels and clips but no
|
|
331
|
+
* substitutions at all: neither the CIGAR nor SEQ says where they are.
|
|
332
|
+
* That is most aligners' output — minimap2 and bwa both leave MD off unless
|
|
333
|
+
* asked — so this is the difference between mismatches rendering and not.
|
|
334
|
+
*
|
|
335
|
+
* `getRecordsForRange` calls it at most ONCE per query, for the span of the
|
|
336
|
+
* reads that need it, and binds the result to each of them (see
|
|
337
|
+
* {@link BamRecord.setReference}). A query whose reads all carry MD does
|
|
338
|
+
* not call it at all.
|
|
339
|
+
*
|
|
340
|
+
* **The span asked for is the reads' union, not the query's**: the query's
|
|
341
|
+
* range plus however far its edge reads overhang it. Usually that is a few
|
|
342
|
+
* hundred bases more for Illumina and a couple of Mb for ultra-long ONT,
|
|
343
|
+
* but a BAM holding whole chromosomes as reads can make it a chromosome.
|
|
344
|
+
* Nothing here clamps that for you, because only you know what your
|
|
345
|
+
* sequence source can afford — clamp inside the callback and return the
|
|
346
|
+
* shorter region; reads it does not cover are then simply left unresolved,
|
|
347
|
+
* and {@link BamRecord.forEachMismatch}'s `opts.ref` is the windowed way to
|
|
348
|
+
* handle one of those reads.
|
|
349
|
+
*/
|
|
350
|
+
fetchReferenceSequence?: ReferenceSequenceFetcher
|
|
284
351
|
}) {
|
|
285
352
|
this.renameRefSeq = renameRefSeqs
|
|
286
353
|
this.bgzfWorkerPool = bgzfWorkerPool
|
|
354
|
+
this.fetchReferenceSequence = fetchReferenceSequence
|
|
287
355
|
this.RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
|
|
288
356
|
this.chunkFeatureCache = new SharedReadCache<Chunk, ChunkEntry<T>>({
|
|
289
357
|
maxSize: maxCacheBytes,
|
|
@@ -469,7 +537,110 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
469
537
|
return []
|
|
470
538
|
}
|
|
471
539
|
const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
|
|
472
|
-
return this._fetchChunkFeatures(chunks, chrId, min, max, opts)
|
|
540
|
+
return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts)
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
/**
|
|
544
|
+
* Reference bases for a region, packed for
|
|
545
|
+
* {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
|
|
546
|
+
* `fetchReferenceSequence` was configured.
|
|
547
|
+
*
|
|
548
|
+
* The way to resolve a read that is longer than the region you are looking at
|
|
549
|
+
* — a contig or a whole chromosome stored as one BAM read. Those are never
|
|
550
|
+
* bound to a record automatically, since binding a partial region to a
|
|
551
|
+
* record shared between queries is the hazard ADR 0006 is about; walking one
|
|
552
|
+
* with a window and a region of your own choosing has no such problem:
|
|
553
|
+
*
|
|
554
|
+
* ```js
|
|
555
|
+
* const ref = await bam.getReferenceRegion('chr1', start, end)
|
|
556
|
+
* record.forEachMismatch(cb, { ref, start, end })
|
|
557
|
+
* ```
|
|
558
|
+
*/
|
|
559
|
+
async getReferenceRegion(
|
|
560
|
+
refName: string,
|
|
561
|
+
start: number,
|
|
562
|
+
end: number,
|
|
563
|
+
opts?: BaseOpts,
|
|
564
|
+
) {
|
|
565
|
+
const fetchReferenceSequence = this.fetchReferenceSequence
|
|
566
|
+
if (!fetchReferenceSequence) {
|
|
567
|
+
return undefined
|
|
568
|
+
}
|
|
569
|
+
// Packed once for the region rather than per read — the walk compares two
|
|
570
|
+
// bases per byte against the read's own packed SEQ, and this is the only
|
|
571
|
+
// per-base pass in it. Length comes from what came back rather than from
|
|
572
|
+
// what was asked for, so a callback that clips at the end of the contig
|
|
573
|
+
// still leaves the bases it did return usable.
|
|
574
|
+
return packReference(
|
|
575
|
+
await fetchReferenceSequence(refName, start, end, opts),
|
|
576
|
+
start,
|
|
577
|
+
)
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
/**
|
|
581
|
+
* Fetch reference bases for the reads in `records` that have no MD tag, and
|
|
582
|
+
* bind them to those reads so their substitutions resolve. A no-op unless
|
|
583
|
+
* `fetchReferenceSequence` was configured, and it issues at most one fetch.
|
|
584
|
+
*
|
|
585
|
+
* The span fetched is the union of the reads that need it, NOT the queried
|
|
586
|
+
* range: a read overhanging the range is only bound to a region covering all
|
|
587
|
+
* of it, because the records are shared between queries and a binding that
|
|
588
|
+
* varied by query would make one query's reads answer out of another's region
|
|
589
|
+
* (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
|
|
590
|
+
* `pairAcrossChr` — are left out for the same reason, and unbound.
|
|
591
|
+
*
|
|
592
|
+
* Nothing clamps that union, since only the callback knows what its sequence
|
|
593
|
+
* source can afford; a callback that returns a shorter region than it was
|
|
594
|
+
* asked for leaves the reads it does not cover unbound, which is the same
|
|
595
|
+
* outcome by a route the consumer controls.
|
|
596
|
+
*/
|
|
597
|
+
protected async _applyReferenceSequence(
|
|
598
|
+
records: T[],
|
|
599
|
+
chrId: number,
|
|
600
|
+
refName: string,
|
|
601
|
+
opts: BaseOpts = {},
|
|
602
|
+
) {
|
|
603
|
+
const fetchReferenceSequence = this.fetchReferenceSequence
|
|
604
|
+
if (!fetchReferenceSequence) {
|
|
605
|
+
return
|
|
606
|
+
}
|
|
607
|
+
let start = Infinity
|
|
608
|
+
let end = 0
|
|
609
|
+
for (let i = 0, l = records.length; i < l; i++) {
|
|
610
|
+
const record = records[i]!
|
|
611
|
+
if (
|
|
612
|
+
record.ref_id === chrId &&
|
|
613
|
+
!record.NUMERIC_MD &&
|
|
614
|
+
record.setReference
|
|
615
|
+
) {
|
|
616
|
+
if (record.start < start) {
|
|
617
|
+
start = record.start
|
|
618
|
+
}
|
|
619
|
+
if (record.end > end) {
|
|
620
|
+
end = record.end
|
|
621
|
+
}
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
// every read carries MD, or there are no reads: nothing to fetch
|
|
625
|
+
if (start >= end) {
|
|
626
|
+
return
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
const ref = await this.getReferenceRegion(refName, start, end, opts)
|
|
630
|
+
if (!ref) {
|
|
631
|
+
return
|
|
632
|
+
}
|
|
633
|
+
for (let i = 0, l = records.length; i < l; i++) {
|
|
634
|
+
const record = records[i]!
|
|
635
|
+
if (
|
|
636
|
+
record.ref_id === chrId &&
|
|
637
|
+
!record.NUMERIC_MD &&
|
|
638
|
+
record.setReference &&
|
|
639
|
+
referenceCovers(ref, record.start, record.end)
|
|
640
|
+
) {
|
|
641
|
+
record.setReference(ref)
|
|
642
|
+
}
|
|
643
|
+
}
|
|
473
644
|
}
|
|
474
645
|
|
|
475
646
|
// Parsed records for a chunk, reading and decompressing it only on a miss.
|
|
@@ -496,6 +667,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
496
667
|
private async _fetchChunkFeatures(
|
|
497
668
|
chunks: Chunk[],
|
|
498
669
|
chrId: number,
|
|
670
|
+
chr: string,
|
|
499
671
|
min: number,
|
|
500
672
|
max: number,
|
|
501
673
|
opts: BamOpts = {},
|
|
@@ -595,6 +767,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
|
|
|
595
767
|
}
|
|
596
768
|
}
|
|
597
769
|
|
|
770
|
+
// After the pairs, so a mate fetched from another chunk is resolved on the
|
|
771
|
+
// same terms as the reads it was fetched for. A no-op without
|
|
772
|
+
// `fetchReferenceSequence`, which is the default.
|
|
773
|
+
await this._applyReferenceSequence(result, chrId, chr, opts)
|
|
774
|
+
|
|
598
775
|
return result
|
|
599
776
|
}
|
|
600
777
|
|
package/src/htsget.ts
CHANGED
|
@@ -5,7 +5,11 @@ import Chunk from './chunk.ts'
|
|
|
5
5
|
import { appendInRange, concatUint8Array, parseRefSeqs } from './util.ts'
|
|
6
6
|
import { VirtualOffset } from './virtualOffset.ts'
|
|
7
7
|
|
|
8
|
-
import type {
|
|
8
|
+
import type {
|
|
9
|
+
BamRecordClass,
|
|
10
|
+
BamRecordLike,
|
|
11
|
+
ReferenceSequenceFetcher,
|
|
12
|
+
} from './bamFile.ts'
|
|
9
13
|
import type BamRecord from './record.ts'
|
|
10
14
|
import type { BamOpts, BaseOpts } from './util.ts'
|
|
11
15
|
import type { Fetcher } from 'generic-filehandle2'
|
|
@@ -145,12 +149,20 @@ export default class HtsgetFile<
|
|
|
145
149
|
maxCacheBytes?: number
|
|
146
150
|
/** see {@link maxCacheBytes}: inherited, and inert in htsget mode */
|
|
147
151
|
cacheIdleTimeoutMs?: number
|
|
152
|
+
/**
|
|
153
|
+
* Reference bases for reads with no `MD` tag — see {@link BamFile}'s option
|
|
154
|
+
* of the same name. Unlike the cache options this one is live here: an
|
|
155
|
+
* htsget query resolves its reads' mismatches exactly as an indexed one
|
|
156
|
+
* does.
|
|
157
|
+
*/
|
|
158
|
+
fetchReferenceSequence?: ReferenceSequenceFetcher
|
|
148
159
|
}) {
|
|
149
160
|
super({
|
|
150
161
|
htsget: true,
|
|
151
162
|
recordClass: args.recordClass,
|
|
152
163
|
maxCacheBytes: args.maxCacheBytes,
|
|
153
164
|
cacheIdleTimeoutMs: args.cacheIdleTimeoutMs,
|
|
165
|
+
fetchReferenceSequence: args.fetchReferenceSequence,
|
|
154
166
|
})
|
|
155
167
|
this.baseUrl = args.baseUrl
|
|
156
168
|
this.trackId = args.trackId
|
|
@@ -192,7 +204,9 @@ export default class HtsgetFile<
|
|
|
192
204
|
[],
|
|
193
205
|
new Chunk(zero, zero, 0),
|
|
194
206
|
)
|
|
195
|
-
|
|
207
|
+
const result = appendInRange(records, chrId, min, max)
|
|
208
|
+
await this._applyReferenceSequence(result, chrId, chr, opts)
|
|
209
|
+
return result
|
|
196
210
|
}
|
|
197
211
|
|
|
198
212
|
async getHeaderPre(opts: BaseOpts = {}) {
|
package/src/index.ts
CHANGED
|
@@ -8,8 +8,33 @@ export { default as CSI } from './csi.ts'
|
|
|
8
8
|
export { default as BamRecord } from './record.ts'
|
|
9
9
|
export { default as HtsgetFile } from './htsget.ts'
|
|
10
10
|
|
|
11
|
+
// The mismatch walk, and the codes it reports differences as. The walk is
|
|
12
|
+
// exported alongside `record.forEachMismatch` for callers holding BAM's packed
|
|
13
|
+
// arrays without a record around them — a SAM parser, or a worker that was
|
|
14
|
+
// posted the typed arrays.
|
|
15
|
+
export {
|
|
16
|
+
MISMATCH_DELETION,
|
|
17
|
+
MISMATCH_HARD_CLIP,
|
|
18
|
+
MISMATCH_INSERTION,
|
|
19
|
+
MISMATCH_REF_SKIP,
|
|
20
|
+
MISMATCH_SOFT_CLIP,
|
|
21
|
+
MISMATCH_SUBST,
|
|
22
|
+
forEachMismatchNumeric,
|
|
23
|
+
} from './mismatches.ts'
|
|
24
|
+
export { packReference } from './reference.ts'
|
|
25
|
+
|
|
11
26
|
export type { NumericCigar } from './record.ts'
|
|
12
|
-
export type {
|
|
27
|
+
export type {
|
|
28
|
+
BamRecordClass,
|
|
29
|
+
BamRecordLike,
|
|
30
|
+
ReferenceSequenceFetcher,
|
|
31
|
+
} from './bamFile.ts'
|
|
32
|
+
export type {
|
|
33
|
+
Mismatch,
|
|
34
|
+
MismatchCallback,
|
|
35
|
+
MismatchOptions,
|
|
36
|
+
} from './mismatches.ts'
|
|
37
|
+
export type { PackedReference } from './reference.ts'
|
|
13
38
|
// the options every query method takes, and the shapes they hand back. The
|
|
14
39
|
// package has no subpath exports, so a consumer typing a wrapper around
|
|
15
40
|
// getRecordsForRange/indexCov can only name these if they come out of here.
|