@gmod/bam 7.6.1 → 7.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/htsget.ts CHANGED
@@ -2,42 +2,111 @@ import { unzip } from '@gmod/bgzf-filehandle'
2
2
 
3
3
  import BamFile, { BAM_MAGIC } from './bamFile.ts'
4
4
  import Chunk from './chunk.ts'
5
- import { parseHeaderText } from './sam.ts'
6
- import { appendInRange, concatUint8Array } from './util.ts'
5
+ import { appendInRange, concatUint8Array, parseRefSeqs } from './util.ts'
7
6
  import { VirtualOffset } from './virtualOffset.ts'
8
7
 
9
8
  import type { BamRecordClass, BamRecordLike } from './bamFile.ts'
10
9
  import type BamRecord from './record.ts'
11
10
  import type { BamOpts, BaseOpts } from './util.ts'
11
+ import type { Fetcher } from 'generic-filehandle2'
12
12
 
13
13
  interface HtsgetChunk {
14
14
  url: string
15
15
  headers?: Record<string, string>
16
+ // present on either all of a ticket's urls or none; only a hint, since the
17
+ // blocks still have to be concatenated in ticket order either way
18
+ class?: 'header' | 'body'
16
19
  }
17
20
 
18
- async function fetchOk(url: string, opts?: RequestInit) {
19
- const res = await fetch(url, opts)
21
+ interface HtsgetTicket {
22
+ htsget: { urls: HtsgetChunk[] }
23
+ }
24
+
25
+ interface HtsgetErrorBody {
26
+ htsget?: { error?: string; message?: string }
27
+ }
28
+
29
+ // Errors carry a JSON body naming the error type, e.g. {"htsget":
30
+ // {"error":"InvalidAuthentication","message":"..."}}. Surface that rather than
31
+ // the raw payload, since 401/403 are the first thing a misconfigured token hits.
32
+ function htsgetErrorMessage(text: string) {
33
+ let parsed: HtsgetErrorBody | undefined
34
+ try {
35
+ parsed = JSON.parse(text)
36
+ } catch {
37
+ // not JSON, fall through to the raw body
38
+ }
39
+ const err = parsed?.htsget
40
+ return err?.error === undefined
41
+ ? text
42
+ : err.message === undefined
43
+ ? err.error
44
+ : `${err.error}: ${err.message}`
45
+ }
46
+
47
+ /**
48
+ * Offset of the first alignment record in the concatenated ticket blocks.
49
+ *
50
+ * The spec has the client concatenate the blocks in ticket order to get a
51
+ * complete data stream, so whether the header is its own block is up to the
52
+ * server: htsnexus splits it into a leading `data:` url, while htsget-rs
53
+ * returns header and records together in a single url. Either way the
54
+ * concatenation starts with the BAM header, so seek past it rather than
55
+ * dropping blocks — a url count or class attribute can't tell the two layouts
56
+ * apart. Returns 0 when the stream has no header, which is what a server
57
+ * sending body-only blocks produces.
58
+ */
59
+ function recordsOffset(uncba: Uint8Array) {
60
+ const dataView = new DataView(
61
+ uncba.buffer,
62
+ uncba.byteOffset,
63
+ uncba.byteLength,
64
+ )
65
+ if (uncba.byteLength < 8 || dataView.getInt32(0, true) !== BAM_MAGIC) {
66
+ return 0
67
+ }
68
+ const refs = parseRefSeqs(uncba, 8 + dataView.getInt32(4, true), n => n)
69
+ if (!refs) {
70
+ throw new Error('truncated BAM header in htsget response')
71
+ }
72
+ return refs.end
73
+ }
74
+
75
+ async function fetchOk(fetchFn: Fetcher, url: string, opts?: RequestInit) {
76
+ const res = await fetchFn(url, opts)
20
77
  if (!res.ok) {
21
- throw new Error(`HTTP ${res.status} fetching ${url}: ${await res.text()}`)
78
+ throw new Error(
79
+ `HTTP ${res.status} fetching ${url}: ${htsgetErrorMessage(await res.text())}`,
80
+ )
22
81
  }
23
82
  return res
24
83
  }
25
84
 
26
- async function fetchChunk({ url, headers }: HtsgetChunk, opts?: RequestInit) {
85
+ async function fetchChunk(
86
+ fetchFn: Fetcher,
87
+ { url, headers }: HtsgetChunk,
88
+ opts?: RequestInit,
89
+ ) {
27
90
  // pass base64 data URLs straight to fetch; otherwise apply headers (minus
28
91
  // referer, which isn't a permitted client-set header).
29
92
  // https://stackoverflow.com/a/54123275/2129219
30
93
  const { referer: _referer, ...rest } = headers ?? {}
31
94
  const res = url.startsWith('data:')
32
- ? await fetchOk(url)
33
- : await fetchOk(url, { ...opts, headers: rest })
95
+ ? await fetchOk(fetchFn, url)
96
+ : await fetchOk(fetchFn, url, { ...opts, headers: rest })
34
97
  return new Uint8Array(await res.arrayBuffer())
35
98
  }
36
99
 
37
- async function fetchAndConcat(arr: HtsgetChunk[], opts?: RequestInit) {
100
+ async function fetchAndConcat(
101
+ fetchFn: Fetcher,
102
+ arr: HtsgetChunk[],
103
+ opts?: RequestInit,
104
+ ) {
38
105
  // Pipeline unzip after each fetch so decompression overlaps later fetches.
39
106
  return concatUint8Array(
40
- await Promise.all(arr.map(async c => unzip(await fetchChunk(c, opts)))),
107
+ await Promise.all(
108
+ arr.map(async c => unzip(await fetchChunk(fetchFn, c, opts))),
109
+ ),
41
110
  )
42
111
  }
43
112
 
@@ -48,14 +117,39 @@ export default class HtsgetFile<
48
117
 
49
118
  private trackId: string
50
119
 
120
+ private fetchFn: Fetcher
121
+
51
122
  constructor(args: {
52
123
  trackId: string
53
124
  baseUrl: string
54
125
  recordClass?: BamRecordClass<T>
126
+ /**
127
+ * fetch implementation used for every request, so an `Authorization: Bearer
128
+ * <token>` header can be added for servers that require one. It is also
129
+ * called with the data-block urls from the ticket, which may point at
130
+ * third-party hosts, so only attach credentials to hosts you trust — the
131
+ * spec has servers put whatever a data block needs in that url's own
132
+ * `headers` field, which is applied either way.
133
+ */
134
+ fetch?: Fetcher
55
135
  }) {
56
136
  super({ htsget: true, recordClass: args.recordClass })
57
137
  this.baseUrl = args.baseUrl
58
138
  this.trackId = args.trackId
139
+ this.fetchFn = args.fetch ?? ((input, init) => fetch(input, init))
140
+ }
141
+
142
+ /**
143
+ * Requests a ticket and returns its data blocks decompressed and
144
+ * concatenated, which per the spec is a complete BAM stream.
145
+ */
146
+ private async fetchTicket(query: string, opts?: BaseOpts) {
147
+ const url = `${this.baseUrl}/${this.trackId}?${query}`
148
+ const res = await fetchOk(this.fetchFn, url, { signal: opts?.signal })
149
+ const ticket: HtsgetTicket = await res.json()
150
+ return fetchAndConcat(this.fetchFn, ticket.htsget.urls, {
151
+ signal: opts?.signal,
152
+ })
59
153
  }
60
154
 
61
155
  async getRecordsForRange(
@@ -65,71 +159,32 @@ export default class HtsgetFile<
65
159
  opts?: BamOpts,
66
160
  ) {
67
161
  await this.getHeader(opts)
68
- const base = `${this.baseUrl}/${this.trackId}`
69
- const url = `${base}?referenceName=${chr}&start=${min}&end=${max}&format=BAM`
70
162
  const chrId = this.chrToIndex?.[chr]
71
163
  if (chrId === undefined) {
72
164
  return []
73
165
  }
74
- const result = await fetchOk(url, opts)
75
- const data = await result.json()
76
- const uncba = await fetchAndConcat(data.htsget.urls.slice(1), {
77
- signal: opts?.signal,
78
- })
79
-
166
+ const uncba = await this.fetchTicket(
167
+ `referenceName=${chr}&start=${min}&end=${max}&format=BAM`,
168
+ opts,
169
+ )
80
170
  const zero = new VirtualOffset(0, 0)
81
- const allRecords = this.readBamFeatures(
82
- uncba,
171
+ const records = this.readBamFeatures(
172
+ uncba.subarray(recordsOffset(uncba)),
83
173
  [],
84
174
  [],
85
175
  new Chunk(zero, zero, 0),
86
176
  )
87
-
88
- return appendInRange(allRecords, chrId, min, max)
177
+ return appendInRange(records, chrId, min, max)
89
178
  }
90
179
 
91
180
  async getHeaderPre(opts: BaseOpts = {}) {
92
- const url = `${this.baseUrl}/${this.trackId}?referenceName=na&class=header`
93
- const result = await fetchOk(url, opts)
94
- const data = await result.json()
95
- const uncba = await fetchAndConcat(data.htsget.urls, {
96
- signal: opts.signal,
97
- })
98
- const dataView = new DataView(
99
- uncba.buffer,
100
- uncba.byteOffset,
101
- uncba.byteLength,
102
- )
103
-
104
- if (dataView.getInt32(0, true) !== BAM_MAGIC) {
105
- throw new Error('Not a BAM file')
106
- }
107
- const headLen = dataView.getInt32(4, true)
108
-
109
- const decoder = new TextDecoder()
110
- const headerText = decoder.decode(uncba.subarray(8, 8 + headLen))
111
- const samHeader = parseHeaderText(headerText)
112
-
113
- // use the @SQ lines in the header to figure out the
114
- // mapping between ref ref ID numbers and names
115
- const idToName: { refName: string; length: number }[] = []
116
- const nameToId: Record<string, number> = {}
117
- const sqLines = samHeader.filter(l => l.tag === 'SQ')
118
- for (const [refId, sqLine] of sqLines.entries()) {
119
- let refName = ''
120
- let length = 0
121
- for (const item of sqLine.data) {
122
- if (item.tag === 'SN') {
123
- refName = item.value
124
- } else if (item.tag === 'LN') {
125
- length = +item.value
126
- }
127
- }
128
- nameToId[refName] = refId
129
- idToName[refId] = { refName, length }
181
+ // format is the only parameter the spec permits alongside class=header;
182
+ // servers SHOULD reject anything else with InvalidInput
183
+ const uncba = await this.fetchTicket('class=header&format=BAM', opts)
184
+ const samHeader = this.applyHeader(uncba)
185
+ if (!samHeader) {
186
+ throw new Error('Insufficient data for reference sequences')
130
187
  }
131
- this.chrToIndex = nameToId
132
- this.indexToChr = idToName
133
188
  return samHeader
134
189
  }
135
190
  }
package/src/index.ts CHANGED
@@ -6,3 +6,5 @@ export { default as HtsgetFile } from './htsget.ts'
6
6
 
7
7
  export type { Bytes } from './record.ts'
8
8
  export type { BamRecordClass, BamRecordLike } from './bamFile.ts'
9
+ // for typing the HtsgetFile `fetch` option
10
+ export type { Fetcher } from 'generic-filehandle2'
package/src/util.ts CHANGED
@@ -184,14 +184,24 @@ export function parseRefSeqs(
184
184
  indexToChr.push({ refName, length: lRef })
185
185
  p += 8 + lName
186
186
  }
187
- return { chrToIndex, indexToChr }
187
+ // end is the offset just past the header, i.e. where alignment records start
188
+ return { chrToIndex, indexToChr, end: p }
188
189
  }
189
190
 
190
- // SYNC: ~/src/gmod/tabix-js/src/util.ts minVirtualOffset
191
+ // SYNC: ~/src/gmod/tabix-js/src/util.ts minVirtualOffset — but NOT the 0:0
192
+ // skip below, which is only sound for BAM. A tabix'd file with no header lines
193
+ // really does have its first record at 0:0.
191
194
  /**
192
195
  * The smallest of `current` and the `count` packed virtual offsets starting at
193
196
  * `offset`, allocating at most one VirtualOffset rather than one per entry.
194
197
  *
198
+ * 0:0 is skipped rather than treated as the minimum. No BAM record can live
199
+ * there — the magic and header occupy the start of the file — so it is the
200
+ * "unset" placeholder htslib leaves in linear-index windows ahead of a
201
+ * reference's first read (see test/data/HG00096_illumina_lowcov.bam.bai, whose
202
+ * first three windows are 0). Counting it collapses firstDataLine to 0:0 and
203
+ * makes callers size a header read from nothing.
204
+ *
195
205
  * The index first pass exists only to find this minimum, and it visits every
196
206
  * linear-index entry in the file to do it. Building a VirtualOffset per entry
197
207
  * to compare and discard it is the bulk of that pass.
@@ -215,7 +225,10 @@ export function minVirtualOffset(
215
225
  bytes[p + 3]! * 0x100 +
216
226
  bytes[p + 2]!
217
227
  const data = (bytes[p + 1]! << 8) | bytes[p]!
218
- if (block < minBlock || (block === minBlock && data < minData)) {
228
+ if (
229
+ (block !== 0 || data !== 0) &&
230
+ (block < minBlock || (block === minBlock && data < minData))
231
+ ) {
219
232
  minBlock = block
220
233
  minData = data
221
234
  found = true