@gmod/bam 8.5.1 → 8.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +30 -0
  2. package/dist/bamFile.d.ts +94 -1
  3. package/dist/bamFile.js +94 -3
  4. package/dist/bamFile.js.map +1 -1
  5. package/dist/htsget.d.ts +8 -1
  6. package/dist/htsget.js +4 -1
  7. package/dist/htsget.js.map +1 -1
  8. package/dist/index.d.ts +5 -1
  9. package/dist/index.js +15 -1
  10. package/dist/index.js.map +1 -1
  11. package/dist/mismatches.d.ts +114 -0
  12. package/dist/mismatches.js +397 -0
  13. package/dist/mismatches.js.map +1 -0
  14. package/dist/record.d.ts +45 -0
  15. package/dist/record.js +76 -10
  16. package/dist/record.js.map +1 -1
  17. package/dist/reference.d.ts +45 -0
  18. package/dist/reference.js +61 -0
  19. package/dist/reference.js.map +1 -0
  20. package/dist/seqAlphabet.d.ts +3 -0
  21. package/dist/seqAlphabet.js +14 -0
  22. package/dist/seqAlphabet.js.map +1 -0
  23. package/esm/bamFile.d.ts +94 -1
  24. package/esm/bamFile.js +94 -3
  25. package/esm/bamFile.js.map +1 -1
  26. package/esm/htsget.d.ts +8 -1
  27. package/esm/htsget.js +4 -1
  28. package/esm/htsget.js.map +1 -1
  29. package/esm/index.d.ts +5 -1
  30. package/esm/index.js +6 -0
  31. package/esm/index.js.map +1 -1
  32. package/esm/mismatches.d.ts +114 -0
  33. package/esm/mismatches.js +393 -0
  34. package/esm/mismatches.js.map +1 -0
  35. package/esm/record.d.ts +45 -0
  36. package/esm/record.js +69 -3
  37. package/esm/record.js.map +1 -1
  38. package/esm/reference.d.ts +45 -0
  39. package/esm/reference.js +55 -0
  40. package/esm/reference.js.map +1 -0
  41. package/esm/seqAlphabet.d.ts +3 -0
  42. package/esm/seqAlphabet.js +11 -0
  43. package/esm/seqAlphabet.js.map +1 -0
  44. package/package.json +1 -1
  45. package/src/bamFile.ts +178 -1
  46. package/src/htsget.ts +16 -2
  47. package/src/index.ts +26 -1
  48. package/src/mismatches.ts +623 -0
  49. package/src/record.ts +95 -3
  50. package/src/reference.ts +87 -0
  51. package/src/seqAlphabet.ts +10 -0
package/src/bamFile.ts CHANGED
@@ -7,6 +7,7 @@ import BAI from './bai.ts'
7
7
  import CSI from './csi.ts'
8
8
  import NullFilehandle from './nullFilehandle.ts'
9
9
  import BAMFeature from './record.ts'
10
+ import { packReference, referenceCovers } from './reference.ts'
10
11
  import { parseHeaderText } from './sam.ts'
11
12
  import {
12
13
  MAX_CONCURRENT_CHUNK_READS,
@@ -16,6 +17,7 @@ import {
16
17
  } from './util.ts'
17
18
 
18
19
  import type Chunk from './chunk.ts'
20
+ import type { PackedReference } from './reference.ts'
19
21
  import type { BamOpts, BaseOpts } from './util.ts'
20
22
  import type { BgzfWorkerPool } from '@gmod/bgzf-filehandle'
21
23
  import type { SharedBudget } from '@gmod/shared-read-cache'
@@ -31,8 +33,42 @@ export interface BamRecordLike {
31
33
  next_refid: number
32
34
  flags: number
33
35
  tags: Record<string, unknown>
36
+ /**
37
+ * Optional, and read only to decide whether a read needs reference bases at
38
+ * all — a read with an MD tag carries its own. A `recordClass` without it is
39
+ * treated as having no MD, i.e. as always wanting the reference.
40
+ */
41
+ NUMERIC_MD?: Uint8Array | undefined
42
+ /**
43
+ * Optional; see {@link BamRecord.setReference}. A `recordClass` that does not
44
+ * implement it simply never gets a reference bound, whatever
45
+ * `fetchReferenceSequence` returns.
46
+ */
47
+ setReference?: (ref: PackedReference) => void
34
48
  }
35
49
 
50
+ /**
51
+ * Supplies reference bases for a region, so reads with no MD tag can still
52
+ * report their substitutions. See {@link BamFile}'s option of the same name.
53
+ *
54
+ * `refName` is the name the query used, unchanged — `renameRefSeqs` maps the
55
+ * FILE's names into the caller's namespace, and this callback is on the
56
+ * caller's side of that. `start`/`end` are 0-based half-open.
57
+ *
58
+ * **The bases returned must begin at `start`.** Returning fewer than asked for
59
+ * is fine and is taken as `[start, start + seq.length)` — the end of a contig,
60
+ * or a source declining to hand over a huge span — and reads the shorter region
61
+ * does not cover are then left unresolved. Returning bases from somewhere else,
62
+ * e.g. clipping the LEFT of the requested range, cannot be detected and
63
+ * resolves every read against the wrong position.
64
+ */
65
+ export type ReferenceSequenceFetcher = (
66
+ refName: string,
67
+ start: number,
68
+ end: number,
69
+ opts?: BaseOpts,
70
+ ) => Promise<string>
71
+
36
72
  export type BamRecordClass<T extends BamRecordLike = BAMFeature> = new (
37
73
  byteArray: Uint8Array,
38
74
  start: number,
@@ -174,6 +210,9 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
174
210
  private bgzfWorkerPool?:
175
211
  BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined
176
212
 
213
+ /** see the constructor option of the same name */
214
+ public fetchReferenceSequence?: ReferenceSequenceFetcher
215
+
177
216
  constructor({
178
217
  bamFilehandle,
179
218
  bamPath,
@@ -191,6 +230,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
191
230
  cacheIdleTimeoutMs = DEFAULT_CACHE_IDLE_TIMEOUT_MS,
192
231
  cacheBudget,
193
232
  bgzfWorkerPool,
233
+ fetchReferenceSequence,
194
234
  }: {
195
235
  bamFilehandle?: GenericFilehandle
196
236
  bamPath?: string
@@ -281,9 +321,37 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
281
321
  */
282
322
  bgzfWorkerPool?:
283
323
  BgzfWorkerPool | Promise<BgzfWorkerPool | undefined> | undefined
324
+ /**
325
+ * Reference bases for a region, as `(refName, start, end, opts) =>
326
+ * Promise<string>`. Only reads that carry no `MD` tag need it, and it is
327
+ * only what {@link BamRecord.forEachMismatch} uses — nothing else in a
328
+ * query touches the reference.
329
+ *
330
+ * Without it, a read lacking MD still reports its indels and clips but no
331
+ * substitutions at all: neither the CIGAR nor SEQ says where they are.
332
+ * That is most aligners' output — minimap2 and bwa both leave MD off unless
333
+ * asked — so this is the difference between mismatches rendering and not.
334
+ *
335
+ * `getRecordsForRange` calls it at most ONCE per query, for the span of the
336
+ * reads that need it, and binds the result to each of them (see
337
+ * {@link BamRecord.setReference}). A query whose reads all carry MD does
338
+ * not call it at all.
339
+ *
340
+ * **The span asked for is the reads' union, not the query's**: the query's
341
+ * range plus however far its edge reads overhang it. Usually that is a few
342
+ * hundred bases more for Illumina and a couple of Mb for ultra-long ONT,
343
+ * but a BAM holding whole chromosomes as reads can make it a chromosome.
344
+ * Nothing here clamps that for you, because only you know what your
345
+ * sequence source can afford — clamp inside the callback and return the
346
+ * shorter region; reads it does not cover are then simply left unresolved,
347
+ * and {@link BamRecord.forEachMismatch}'s `opts.ref` is the windowed way to
348
+ * handle one of those reads.
349
+ */
350
+ fetchReferenceSequence?: ReferenceSequenceFetcher
284
351
  }) {
285
352
  this.renameRefSeq = renameRefSeqs
286
353
  this.bgzfWorkerPool = bgzfWorkerPool
354
+ this.fetchReferenceSequence = fetchReferenceSequence
287
355
  this.RecordClass = (recordClass ?? BAMFeature) as BamRecordClass<T>
288
356
  this.chunkFeatureCache = new SharedReadCache<Chunk, ChunkEntry<T>>({
289
357
  maxSize: maxCacheBytes,
@@ -469,7 +537,110 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
469
537
  return []
470
538
  }
471
539
  const chunks = await this.index.blocksForRange(chrId, min - 1, max, opts)
472
- return this._fetchChunkFeatures(chunks, chrId, min, max, opts)
540
+ return this._fetchChunkFeatures(chunks, chrId, chr, min, max, opts)
541
+ }
542
+
543
+ /**
544
+ * Reference bases for a region, packed for
545
+ * {@link BamRecord.forEachMismatch}'s `opts.ref`, or undefined when no
546
+ * `fetchReferenceSequence` was configured.
547
+ *
548
+ * The way to resolve a read that is longer than the region you are looking at
549
+ * — a contig or a whole chromosome stored as one BAM read. Those are never
550
+ * bound to a record automatically, since binding a partial region to a
551
+ * record shared between queries is the hazard ADR 0006 is about; walking one
552
+ * with a window and a region of your own choosing has no such problem:
553
+ *
554
+ * ```js
555
+ * const ref = await bam.getReferenceRegion('chr1', start, end)
556
+ * record.forEachMismatch(cb, { ref, start, end })
557
+ * ```
558
+ */
559
+ async getReferenceRegion(
560
+ refName: string,
561
+ start: number,
562
+ end: number,
563
+ opts?: BaseOpts,
564
+ ) {
565
+ const fetchReferenceSequence = this.fetchReferenceSequence
566
+ if (!fetchReferenceSequence) {
567
+ return undefined
568
+ }
569
+ // Packed once for the region rather than per read — the walk compares two
570
+ // bases per byte against the read's own packed SEQ, and this is the only
571
+ // per-base pass in it. Length comes from what came back rather than from
572
+ // what was asked for, so a callback that clips at the end of the contig
573
+ // still leaves the bases it did return usable.
574
+ return packReference(
575
+ await fetchReferenceSequence(refName, start, end, opts),
576
+ start,
577
+ )
578
+ }
579
+
580
+ /**
581
+ * Fetch reference bases for the reads in `records` that have no MD tag, and
582
+ * bind them to those reads so their substitutions resolve. A no-op unless
583
+ * `fetchReferenceSequence` was configured, and it issues at most one fetch.
584
+ *
585
+ * The span fetched is the union of the reads that need it, NOT the queried
586
+ * range: a read overhanging the range is only bound to a region covering all
587
+ * of it, because the records are shared between queries and a binding that
588
+ * varied by query would make one query's reads answer out of another's region
589
+ * (ADR 0006, ADR 0020). Reads on another reference — `viewAsPairs` mates with
590
+ * `pairAcrossChr` — are left out for the same reason, and unbound.
591
+ *
592
+ * Nothing clamps that union, since only the callback knows what its sequence
593
+ * source can afford; a callback that returns a shorter region than it was
594
+ * asked for leaves the reads it does not cover unbound, which is the same
595
+ * outcome by a route the consumer controls.
596
+ */
597
+ protected async _applyReferenceSequence(
598
+ records: T[],
599
+ chrId: number,
600
+ refName: string,
601
+ opts: BaseOpts = {},
602
+ ) {
603
+ const fetchReferenceSequence = this.fetchReferenceSequence
604
+ if (!fetchReferenceSequence) {
605
+ return
606
+ }
607
+ let start = Infinity
608
+ let end = 0
609
+ for (let i = 0, l = records.length; i < l; i++) {
610
+ const record = records[i]!
611
+ if (
612
+ record.ref_id === chrId &&
613
+ !record.NUMERIC_MD &&
614
+ record.setReference
615
+ ) {
616
+ if (record.start < start) {
617
+ start = record.start
618
+ }
619
+ if (record.end > end) {
620
+ end = record.end
621
+ }
622
+ }
623
+ }
624
+ // every read carries MD, or there are no reads: nothing to fetch
625
+ if (start >= end) {
626
+ return
627
+ }
628
+
629
+ const ref = await this.getReferenceRegion(refName, start, end, opts)
630
+ if (!ref) {
631
+ return
632
+ }
633
+ for (let i = 0, l = records.length; i < l; i++) {
634
+ const record = records[i]!
635
+ if (
636
+ record.ref_id === chrId &&
637
+ !record.NUMERIC_MD &&
638
+ record.setReference &&
639
+ referenceCovers(ref, record.start, record.end)
640
+ ) {
641
+ record.setReference(ref)
642
+ }
643
+ }
473
644
  }
474
645
 
475
646
  // Parsed records for a chunk, reading and decompressing it only on a miss.
@@ -496,6 +667,7 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
496
667
  private async _fetchChunkFeatures(
497
668
  chunks: Chunk[],
498
669
  chrId: number,
670
+ chr: string,
499
671
  min: number,
500
672
  max: number,
501
673
  opts: BamOpts = {},
@@ -595,6 +767,11 @@ export default class BamFile<T extends BamRecordLike = BAMFeature> {
595
767
  }
596
768
  }
597
769
 
770
+ // After the pairs, so a mate fetched from another chunk is resolved on the
771
+ // same terms as the reads it was fetched for. A no-op without
772
+ // `fetchReferenceSequence`, which is the default.
773
+ await this._applyReferenceSequence(result, chrId, chr, opts)
774
+
598
775
  return result
599
776
  }
600
777
 
package/src/htsget.ts CHANGED
@@ -5,7 +5,11 @@ import Chunk from './chunk.ts'
5
5
  import { appendInRange, concatUint8Array, parseRefSeqs } from './util.ts'
6
6
  import { VirtualOffset } from './virtualOffset.ts'
7
7
 
8
- import type { BamRecordClass, BamRecordLike } from './bamFile.ts'
8
+ import type {
9
+ BamRecordClass,
10
+ BamRecordLike,
11
+ ReferenceSequenceFetcher,
12
+ } from './bamFile.ts'
9
13
  import type BamRecord from './record.ts'
10
14
  import type { BamOpts, BaseOpts } from './util.ts'
11
15
  import type { Fetcher } from 'generic-filehandle2'
@@ -145,12 +149,20 @@ export default class HtsgetFile<
145
149
  maxCacheBytes?: number
146
150
  /** see {@link maxCacheBytes}: inherited, and inert in htsget mode */
147
151
  cacheIdleTimeoutMs?: number
152
+ /**
153
+ * Reference bases for reads with no `MD` tag — see {@link BamFile}'s option
154
+ * of the same name. Unlike the cache options this one is live here: an
155
+ * htsget query resolves its reads' mismatches exactly as an indexed one
156
+ * does.
157
+ */
158
+ fetchReferenceSequence?: ReferenceSequenceFetcher
148
159
  }) {
149
160
  super({
150
161
  htsget: true,
151
162
  recordClass: args.recordClass,
152
163
  maxCacheBytes: args.maxCacheBytes,
153
164
  cacheIdleTimeoutMs: args.cacheIdleTimeoutMs,
165
+ fetchReferenceSequence: args.fetchReferenceSequence,
154
166
  })
155
167
  this.baseUrl = args.baseUrl
156
168
  this.trackId = args.trackId
@@ -192,7 +204,9 @@ export default class HtsgetFile<
192
204
  [],
193
205
  new Chunk(zero, zero, 0),
194
206
  )
195
- return appendInRange(records, chrId, min, max)
207
+ const result = appendInRange(records, chrId, min, max)
208
+ await this._applyReferenceSequence(result, chrId, chr, opts)
209
+ return result
196
210
  }
197
211
 
198
212
  async getHeaderPre(opts: BaseOpts = {}) {
package/src/index.ts CHANGED
@@ -8,8 +8,33 @@ export { default as CSI } from './csi.ts'
8
8
  export { default as BamRecord } from './record.ts'
9
9
  export { default as HtsgetFile } from './htsget.ts'
10
10
 
11
+ // The mismatch walk, and the codes it reports differences as. The walk is
12
+ // exported alongside `record.forEachMismatch` for callers holding BAM's packed
13
+ // arrays without a record around them — a SAM parser, or a worker that was
14
+ // posted the typed arrays.
15
+ export {
16
+ MISMATCH_DELETION,
17
+ MISMATCH_HARD_CLIP,
18
+ MISMATCH_INSERTION,
19
+ MISMATCH_REF_SKIP,
20
+ MISMATCH_SOFT_CLIP,
21
+ MISMATCH_SUBST,
22
+ forEachMismatchNumeric,
23
+ } from './mismatches.ts'
24
+ export { packReference } from './reference.ts'
25
+
11
26
  export type { NumericCigar } from './record.ts'
12
- export type { BamRecordClass, BamRecordLike } from './bamFile.ts'
27
+ export type {
28
+ BamRecordClass,
29
+ BamRecordLike,
30
+ ReferenceSequenceFetcher,
31
+ } from './bamFile.ts'
32
+ export type {
33
+ Mismatch,
34
+ MismatchCallback,
35
+ MismatchOptions,
36
+ } from './mismatches.ts'
37
+ export type { PackedReference } from './reference.ts'
13
38
  // the options every query method takes, and the shapes they hand back. The
14
39
  // package has no subpath exports, so a consumer typing a wrapper around
15
40
  // getRecordsForRange/indexCov can only name these if they come out of here.