@gmod/bam 9.0.1 → 10.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +18 -17
  2. package/dist/bai.d.ts +2 -0
  3. package/dist/bai.js +19 -0
  4. package/dist/bai.js.map +1 -1
  5. package/dist/bamFile.js +12 -21
  6. package/dist/bamFile.js.map +1 -1
  7. package/dist/csi.d.ts +20 -29
  8. package/dist/csi.js +27 -10
  9. package/dist/csi.js.map +1 -1
  10. package/dist/indexFile.d.ts +21 -1
  11. package/dist/indexFile.js +60 -2
  12. package/dist/indexFile.js.map +1 -1
  13. package/dist/record.d.ts +1 -0
  14. package/dist/record.js +33 -21
  15. package/dist/record.js.map +1 -1
  16. package/dist/streamBam.js +9 -8
  17. package/dist/streamBam.js.map +1 -1
  18. package/dist/util.d.ts +24 -4
  19. package/dist/util.js +52 -4
  20. package/dist/util.js.map +1 -1
  21. package/dist/virtualOffset.d.ts +1 -0
  22. package/dist/virtualOffset.js +4 -0
  23. package/dist/virtualOffset.js.map +1 -1
  24. package/esm/bai.d.ts +2 -0
  25. package/esm/bai.js +19 -0
  26. package/esm/bai.js.map +1 -1
  27. package/esm/bamFile.js +13 -22
  28. package/esm/bamFile.js.map +1 -1
  29. package/esm/csi.d.ts +20 -29
  30. package/esm/csi.js +27 -10
  31. package/esm/csi.js.map +1 -1
  32. package/esm/indexFile.d.ts +21 -1
  33. package/esm/indexFile.js +59 -2
  34. package/esm/indexFile.js.map +1 -1
  35. package/esm/record.d.ts +1 -0
  36. package/esm/record.js +33 -21
  37. package/esm/record.js.map +1 -1
  38. package/esm/streamBam.js +10 -9
  39. package/esm/streamBam.js.map +1 -1
  40. package/esm/util.d.ts +24 -4
  41. package/esm/util.js +49 -4
  42. package/esm/util.js.map +1 -1
  43. package/esm/virtualOffset.d.ts +1 -0
  44. package/esm/virtualOffset.js +3 -0
  45. package/esm/virtualOffset.js.map +1 -1
  46. package/package.json +8 -8
  47. package/src/bai.ts +21 -0
  48. package/src/bamFile.ts +16 -23
  49. package/src/csi.ts +41 -14
  50. package/src/indexFile.ts +72 -3
  51. package/src/record.ts +39 -20
  52. package/src/streamBam.ts +11 -10
  53. package/src/util.ts +56 -4
  54. package/src/virtualOffset.ts +4 -0
package/src/streamBam.ts CHANGED
@@ -9,7 +9,9 @@ import { parseHeaderText } from './sam.ts'
9
9
  import {
10
10
  BAM_MAGIC,
11
11
  concatUint8Array,
12
+ decodeHeaderText,
12
13
  parseRefSeqs,
14
+ readBlockSize,
13
15
  resolveFilehandle,
14
16
  throwIfAborted,
15
17
  } from './util.ts'
@@ -273,9 +275,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
273
275
  recordCarry = bytes
274
276
  continue
275
277
  }
276
- const headerText = new TextDecoder('utf8').decode(
277
- bytes.subarray(8, 8 + lText),
278
- )
278
+ const headerText = decodeHeaderText(bytes, lText)
279
279
  onHeader?.({
280
280
  headerText,
281
281
  samHeader: parseHeaderText(headerText),
@@ -288,7 +288,7 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
288
288
 
289
289
  const sink: T[] = []
290
290
  while (blockStart + 4 <= bytes.length) {
291
- const blockSize = dataView.getInt32(blockStart, true)
291
+ const blockSize = readBlockSize(dataView, blockStart)
292
292
  const blockEnd = blockStart + 4 + blockSize - 1
293
293
  if (blockEnd >= bytes.length) {
294
294
  break
@@ -305,12 +305,13 @@ export async function* streamBamRecords<T extends BamRecordLike = BAMFeature>({
305
305
  // the window size, so it is as stable as a virtual offset without
306
306
  // pretending to be one.
307
307
  //
308
- // NOT the crc32 of the record bytes that readBamFeatures falls back
309
- // to when it has no positions. That is a content hash, so the two
310
- // byte-identical records in exact_duplicate.bam collide on it, and
311
- // hashing every record costs ~40% of the walk (135ms of 340ms over
312
- // out.bam) where a counter costs nothing. The fallback only fires on
313
- // an unusual path there; here it would fire on every record.
308
+ // NOT the hash of the record bytes that readBamFeatures falls back
309
+ // to when it has no positions: the two byte-identical records in
310
+ // exact_duplicate.bam collide on any content hash, and hashing
311
+ // every record cost ~40% of the walk (135ms of 340ms over out.bam,
312
+ // measured with crc32) where a counter costs nothing. The fallback
313
+ // only fires on an unusual path there; here it would fire on every
314
+ // record.
314
315
  recordIndex++,
315
316
  dataView,
316
317
  ),
package/src/util.ts CHANGED
@@ -9,6 +9,45 @@ import type { GenericFilehandle } from 'generic-filehandle2'
9
9
  /** 'BAM\1' read as a little-endian int32 */
10
10
  export const BAM_MAGIC = 21840194
11
11
 
12
+ /**
13
+ * The `block_size` of the record at `offset`, rejecting one too small to hold a
14
+ * record's 32 bytes of fixed fields, as htslib's `bam_read1` does. A negative
15
+ * one would otherwise leave the loops that advance by it standing still,
16
+ * pushing a record per turn until the process runs out of memory.
17
+ */
18
+ export function readBlockSize(dataView: DataView, offset: number) {
19
+ const blockSize = dataView.getInt32(offset, true)
20
+ if (blockSize < 32) {
21
+ throw new Error(
22
+ `corrupt BAM record: block_size ${blockSize} at byte ${offset}`,
23
+ )
24
+ }
25
+ return blockSize
26
+ }
27
+
28
+ /**
29
+ * A 53-bit hash of `bytes[start, end)`: cyrb53 (public domain), taken over
30
+ * bytes. The id of a record read with no file positions, which has to be the
31
+ * same for the same record in every query, so it hashes content. It is 53 bits
32
+ * rather than crc32's 32 because ids are deduplicated: at 32 bits a fetch of
33
+ * 150,000 records expects about 2.6 distinct pairs to collide, each silently
34
+ * dropping a read, where 53 bits expect about 1e-6.
35
+ */
36
+ export function contentHash53(bytes: Uint8Array, start: number, end: number) {
37
+ let h1 = 0xdeadbeef
38
+ let h2 = 0x41c6ce57
39
+ for (let i = start; i < end; i++) {
40
+ const b = bytes[i]!
41
+ h1 = Math.imul(h1 ^ b, 2654435761)
42
+ h2 = Math.imul(h2 ^ b, 1597334677)
43
+ }
44
+ h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507)
45
+ h1 ^= Math.imul(h2 ^ (h2 >>> 13), 3266489909)
46
+ h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507)
47
+ h2 ^= Math.imul(h1 ^ (h1 >>> 13), 3266489909)
48
+ return 4294967296 * (2097151 & h2) + (h1 >>> 0)
49
+ }
50
+
12
51
  export function resolveFilehandle(
13
52
  filehandle?: GenericFilehandle,
14
53
  path?: string,
@@ -66,10 +105,13 @@ export const MAX_CONCURRENT_CHUNK_READS = 6
66
105
  *
67
106
  * `blocksForRange` returns every chunk of every bin overlapping the query, at
68
107
  * every level of the binning scheme, minus the ones the linear index puts
69
- * entirely before it. On a long-read file that is wildly more than the query
70
- * reads: a coarse bin's chunks run to the end of the bin's span, so a 380bp
71
- * window on a deep ONT BAM resolves to 90 chunks / 43.5MB, of which
72
- * `getRecordsForRange` reads 6 / 7.8MB before the early stop fires.
108
+ * entirely before it. Until `max_off` (ADR 0023) also dropped the ones past it,
109
+ * that was wildly more than a long-read query reads: a coarse bin's chunks run
110
+ * to the end of the bin's span, so a 380bp window on a deep ONT BAM resolved to
111
+ * 90 chunks / 43.5MB, of which `getRecordsForRange` read 6 / 7.8MB before the
112
+ * early stop fired. Since then this changes the forecast in 4 of 320 fixture
113
+ * windows, undershooting in all 4; ADR 0023 says what to check before
114
+ * removing it.
73
115
  *
74
116
  * Two bounds, and the answer is the larger:
75
117
  *
@@ -325,6 +367,16 @@ export function clampChunkEnds(
325
367
  }
326
368
  }
327
369
 
370
+ // SAMv1 §4.2 counts NUL padding in l_text, and htslib reads the text as a C
371
+ // string, so the header ends at the first NUL.
372
+ export function decodeHeaderText(bytes: Uint8Array, lText: number) {
373
+ const text = bytes.subarray(8, 8 + lText)
374
+ const nul = text.indexOf(0)
375
+ return new TextDecoder('utf8').decode(
376
+ nul === -1 ? text : text.subarray(0, nul),
377
+ )
378
+ }
379
+
328
380
  // Parse the BAM reference-sequence table (SAMv1.pdf §4.2). Returns undefined
329
381
  // if `uncba` doesn't yet contain the full table — caller fetches more bytes
330
382
  // and retries.
@@ -35,3 +35,7 @@ export function fromBytes(bytes: Uint8Array, offset = 0) {
35
35
  (bytes[offset + 1]! << 8) | bytes[offset]!,
36
36
  )
37
37
  }
38
+
39
+ export function compareOffsets(a: OffsetCoords, b: OffsetCoords) {
40
+ return a.blockPosition - b.blockPosition || a.dataPosition - b.dataPosition
41
+ }