@gmod/bam 7.6.1 → 7.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +109 -98
- package/dist/bamFile.d.ts +13 -0
- package/dist/bamFile.js +44 -24
- package/dist/bamFile.js.map +1 -1
- package/dist/htsget.d.ts +16 -0
- package/dist/htsget.js +72 -52
- package/dist/htsget.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/util.d.ts +8 -0
- package/dist/util.js +14 -3
- package/dist/util.js.map +1 -1
- package/esm/bamFile.d.ts +13 -0
- package/esm/bamFile.js +44 -24
- package/esm/bamFile.js.map +1 -1
- package/esm/htsget.d.ts +16 -0
- package/esm/htsget.js +73 -53
- package/esm/htsget.js.map +1 -1
- package/esm/index.d.ts +1 -0
- package/esm/util.d.ts +8 -0
- package/esm/util.js +14 -3
- package/esm/util.js.map +1 -1
- package/package.json +3 -3
- package/src/bamFile.ts +48 -27
- package/src/htsget.ts +117 -62
- package/src/index.ts +2 -0
- package/src/util.ts +16 -3
package/src/htsget.ts
CHANGED
|
@@ -2,42 +2,111 @@ import { unzip } from '@gmod/bgzf-filehandle'
|
|
|
2
2
|
|
|
3
3
|
import BamFile, { BAM_MAGIC } from './bamFile.ts'
|
|
4
4
|
import Chunk from './chunk.ts'
|
|
5
|
-
import {
|
|
6
|
-
import { appendInRange, concatUint8Array } from './util.ts'
|
|
5
|
+
import { appendInRange, concatUint8Array, parseRefSeqs } from './util.ts'
|
|
7
6
|
import { VirtualOffset } from './virtualOffset.ts'
|
|
8
7
|
|
|
9
8
|
import type { BamRecordClass, BamRecordLike } from './bamFile.ts'
|
|
10
9
|
import type BamRecord from './record.ts'
|
|
11
10
|
import type { BamOpts, BaseOpts } from './util.ts'
|
|
11
|
+
import type { Fetcher } from 'generic-filehandle2'
|
|
12
12
|
|
|
13
13
|
interface HtsgetChunk {
|
|
14
14
|
url: string
|
|
15
15
|
headers?: Record<string, string>
|
|
16
|
+
// present on either all of a ticket's urls or none; only a hint, since the
|
|
17
|
+
// blocks still have to be concatenated in ticket order either way
|
|
18
|
+
class?: 'header' | 'body'
|
|
16
19
|
}
|
|
17
20
|
|
|
18
|
-
|
|
19
|
-
|
|
21
|
+
interface HtsgetTicket {
|
|
22
|
+
htsget: { urls: HtsgetChunk[] }
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
interface HtsgetErrorBody {
|
|
26
|
+
htsget?: { error?: string; message?: string }
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// Errors carry a JSON body naming the error type, e.g. {"htsget":
|
|
30
|
+
// {"error":"InvalidAuthentication","message":"..."}}. Surface that rather than
|
|
31
|
+
// the raw payload, since 401/403 are the first thing a misconfigured token hits.
|
|
32
|
+
function htsgetErrorMessage(text: string) {
|
|
33
|
+
let parsed: HtsgetErrorBody | undefined
|
|
34
|
+
try {
|
|
35
|
+
parsed = JSON.parse(text)
|
|
36
|
+
} catch {
|
|
37
|
+
// not JSON, fall through to the raw body
|
|
38
|
+
}
|
|
39
|
+
const err = parsed?.htsget
|
|
40
|
+
return err?.error === undefined
|
|
41
|
+
? text
|
|
42
|
+
: err.message === undefined
|
|
43
|
+
? err.error
|
|
44
|
+
: `${err.error}: ${err.message}`
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Offset of the first alignment record in the concatenated ticket blocks.
|
|
49
|
+
*
|
|
50
|
+
* The spec has the client concatenate the blocks in ticket order to get a
|
|
51
|
+
* complete data stream, so whether the header is its own block is up to the
|
|
52
|
+
* server: htsnexus splits it into a leading `data:` url, while htsget-rs
|
|
53
|
+
* returns header and records together in a single url. Either way the
|
|
54
|
+
* concatenation starts with the BAM header, so seek past it rather than
|
|
55
|
+
* dropping blocks — a url count or class attribute can't tell the two layouts
|
|
56
|
+
* apart. Returns 0 when the stream has no header, which is what a server
|
|
57
|
+
* sending body-only blocks produces.
|
|
58
|
+
*/
|
|
59
|
+
function recordsOffset(uncba: Uint8Array) {
|
|
60
|
+
const dataView = new DataView(
|
|
61
|
+
uncba.buffer,
|
|
62
|
+
uncba.byteOffset,
|
|
63
|
+
uncba.byteLength,
|
|
64
|
+
)
|
|
65
|
+
if (uncba.byteLength < 8 || dataView.getInt32(0, true) !== BAM_MAGIC) {
|
|
66
|
+
return 0
|
|
67
|
+
}
|
|
68
|
+
const refs = parseRefSeqs(uncba, 8 + dataView.getInt32(4, true), n => n)
|
|
69
|
+
if (!refs) {
|
|
70
|
+
throw new Error('truncated BAM header in htsget response')
|
|
71
|
+
}
|
|
72
|
+
return refs.end
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
async function fetchOk(fetchFn: Fetcher, url: string, opts?: RequestInit) {
|
|
76
|
+
const res = await fetchFn(url, opts)
|
|
20
77
|
if (!res.ok) {
|
|
21
|
-
throw new Error(
|
|
78
|
+
throw new Error(
|
|
79
|
+
`HTTP ${res.status} fetching ${url}: ${htsgetErrorMessage(await res.text())}`,
|
|
80
|
+
)
|
|
22
81
|
}
|
|
23
82
|
return res
|
|
24
83
|
}
|
|
25
84
|
|
|
26
|
-
async function fetchChunk(
|
|
85
|
+
async function fetchChunk(
|
|
86
|
+
fetchFn: Fetcher,
|
|
87
|
+
{ url, headers }: HtsgetChunk,
|
|
88
|
+
opts?: RequestInit,
|
|
89
|
+
) {
|
|
27
90
|
// pass base64 data URLs straight to fetch; otherwise apply headers (minus
|
|
28
91
|
// referer, which isn't a permitted client-set header).
|
|
29
92
|
// https://stackoverflow.com/a/54123275/2129219
|
|
30
93
|
const { referer: _referer, ...rest } = headers ?? {}
|
|
31
94
|
const res = url.startsWith('data:')
|
|
32
|
-
? await fetchOk(url)
|
|
33
|
-
: await fetchOk(url, { ...opts, headers: rest })
|
|
95
|
+
? await fetchOk(fetchFn, url)
|
|
96
|
+
: await fetchOk(fetchFn, url, { ...opts, headers: rest })
|
|
34
97
|
return new Uint8Array(await res.arrayBuffer())
|
|
35
98
|
}
|
|
36
99
|
|
|
37
|
-
async function fetchAndConcat(
|
|
100
|
+
async function fetchAndConcat(
|
|
101
|
+
fetchFn: Fetcher,
|
|
102
|
+
arr: HtsgetChunk[],
|
|
103
|
+
opts?: RequestInit,
|
|
104
|
+
) {
|
|
38
105
|
// Pipeline unzip after each fetch so decompression overlaps later fetches.
|
|
39
106
|
return concatUint8Array(
|
|
40
|
-
await Promise.all(
|
|
107
|
+
await Promise.all(
|
|
108
|
+
arr.map(async c => unzip(await fetchChunk(fetchFn, c, opts))),
|
|
109
|
+
),
|
|
41
110
|
)
|
|
42
111
|
}
|
|
43
112
|
|
|
@@ -48,14 +117,39 @@ export default class HtsgetFile<
|
|
|
48
117
|
|
|
49
118
|
private trackId: string
|
|
50
119
|
|
|
120
|
+
private fetchFn: Fetcher
|
|
121
|
+
|
|
51
122
|
constructor(args: {
|
|
52
123
|
trackId: string
|
|
53
124
|
baseUrl: string
|
|
54
125
|
recordClass?: BamRecordClass<T>
|
|
126
|
+
/**
|
|
127
|
+
* fetch implementation used for every request, so an `Authorization: Bearer
|
|
128
|
+
* <token>` header can be added for servers that require one. It is also
|
|
129
|
+
* called with the data-block urls from the ticket, which may point at
|
|
130
|
+
* third-party hosts, so only attach credentials to hosts you trust — the
|
|
131
|
+
* spec has servers put whatever a data block needs in that url's own
|
|
132
|
+
* `headers` field, which is applied either way.
|
|
133
|
+
*/
|
|
134
|
+
fetch?: Fetcher
|
|
55
135
|
}) {
|
|
56
136
|
super({ htsget: true, recordClass: args.recordClass })
|
|
57
137
|
this.baseUrl = args.baseUrl
|
|
58
138
|
this.trackId = args.trackId
|
|
139
|
+
this.fetchFn = args.fetch ?? ((input, init) => fetch(input, init))
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* Requests a ticket and returns its data blocks decompressed and
|
|
144
|
+
* concatenated, which per the spec is a complete BAM stream.
|
|
145
|
+
*/
|
|
146
|
+
private async fetchTicket(query: string, opts?: BaseOpts) {
|
|
147
|
+
const url = `${this.baseUrl}/${this.trackId}?${query}`
|
|
148
|
+
const res = await fetchOk(this.fetchFn, url, { signal: opts?.signal })
|
|
149
|
+
const ticket: HtsgetTicket = await res.json()
|
|
150
|
+
return fetchAndConcat(this.fetchFn, ticket.htsget.urls, {
|
|
151
|
+
signal: opts?.signal,
|
|
152
|
+
})
|
|
59
153
|
}
|
|
60
154
|
|
|
61
155
|
async getRecordsForRange(
|
|
@@ -65,71 +159,32 @@ export default class HtsgetFile<
|
|
|
65
159
|
opts?: BamOpts,
|
|
66
160
|
) {
|
|
67
161
|
await this.getHeader(opts)
|
|
68
|
-
const base = `${this.baseUrl}/${this.trackId}`
|
|
69
|
-
const url = `${base}?referenceName=${chr}&start=${min}&end=${max}&format=BAM`
|
|
70
162
|
const chrId = this.chrToIndex?.[chr]
|
|
71
163
|
if (chrId === undefined) {
|
|
72
164
|
return []
|
|
73
165
|
}
|
|
74
|
-
const
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
})
|
|
79
|
-
|
|
166
|
+
const uncba = await this.fetchTicket(
|
|
167
|
+
`referenceName=${chr}&start=${min}&end=${max}&format=BAM`,
|
|
168
|
+
opts,
|
|
169
|
+
)
|
|
80
170
|
const zero = new VirtualOffset(0, 0)
|
|
81
|
-
const
|
|
82
|
-
uncba,
|
|
171
|
+
const records = this.readBamFeatures(
|
|
172
|
+
uncba.subarray(recordsOffset(uncba)),
|
|
83
173
|
[],
|
|
84
174
|
[],
|
|
85
175
|
new Chunk(zero, zero, 0),
|
|
86
176
|
)
|
|
87
|
-
|
|
88
|
-
return appendInRange(allRecords, chrId, min, max)
|
|
177
|
+
return appendInRange(records, chrId, min, max)
|
|
89
178
|
}
|
|
90
179
|
|
|
91
180
|
async getHeaderPre(opts: BaseOpts = {}) {
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
const
|
|
95
|
-
const
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
const dataView = new DataView(
|
|
99
|
-
uncba.buffer,
|
|
100
|
-
uncba.byteOffset,
|
|
101
|
-
uncba.byteLength,
|
|
102
|
-
)
|
|
103
|
-
|
|
104
|
-
if (dataView.getInt32(0, true) !== BAM_MAGIC) {
|
|
105
|
-
throw new Error('Not a BAM file')
|
|
106
|
-
}
|
|
107
|
-
const headLen = dataView.getInt32(4, true)
|
|
108
|
-
|
|
109
|
-
const decoder = new TextDecoder()
|
|
110
|
-
const headerText = decoder.decode(uncba.subarray(8, 8 + headLen))
|
|
111
|
-
const samHeader = parseHeaderText(headerText)
|
|
112
|
-
|
|
113
|
-
// use the @SQ lines in the header to figure out the
|
|
114
|
-
// mapping between ref ref ID numbers and names
|
|
115
|
-
const idToName: { refName: string; length: number }[] = []
|
|
116
|
-
const nameToId: Record<string, number> = {}
|
|
117
|
-
const sqLines = samHeader.filter(l => l.tag === 'SQ')
|
|
118
|
-
for (const [refId, sqLine] of sqLines.entries()) {
|
|
119
|
-
let refName = ''
|
|
120
|
-
let length = 0
|
|
121
|
-
for (const item of sqLine.data) {
|
|
122
|
-
if (item.tag === 'SN') {
|
|
123
|
-
refName = item.value
|
|
124
|
-
} else if (item.tag === 'LN') {
|
|
125
|
-
length = +item.value
|
|
126
|
-
}
|
|
127
|
-
}
|
|
128
|
-
nameToId[refName] = refId
|
|
129
|
-
idToName[refId] = { refName, length }
|
|
181
|
+
// format is the only parameter the spec permits alongside class=header;
|
|
182
|
+
// servers SHOULD reject anything else with InvalidInput
|
|
183
|
+
const uncba = await this.fetchTicket('class=header&format=BAM', opts)
|
|
184
|
+
const samHeader = this.applyHeader(uncba)
|
|
185
|
+
if (!samHeader) {
|
|
186
|
+
throw new Error('Insufficient data for reference sequences')
|
|
130
187
|
}
|
|
131
|
-
this.chrToIndex = nameToId
|
|
132
|
-
this.indexToChr = idToName
|
|
133
188
|
return samHeader
|
|
134
189
|
}
|
|
135
190
|
}
|
package/src/index.ts
CHANGED
package/src/util.ts
CHANGED
|
@@ -184,14 +184,24 @@ export function parseRefSeqs(
|
|
|
184
184
|
indexToChr.push({ refName, length: lRef })
|
|
185
185
|
p += 8 + lName
|
|
186
186
|
}
|
|
187
|
-
|
|
187
|
+
// end is the offset just past the header, i.e. where alignment records start
|
|
188
|
+
return { chrToIndex, indexToChr, end: p }
|
|
188
189
|
}
|
|
189
190
|
|
|
190
|
-
// SYNC: ~/src/gmod/tabix-js/src/util.ts minVirtualOffset
|
|
191
|
+
// SYNC: ~/src/gmod/tabix-js/src/util.ts minVirtualOffset — but NOT the 0:0
|
|
192
|
+
// skip below, which is only sound for BAM. A tabix'd file with no header lines
|
|
193
|
+
// really does have its first record at 0:0.
|
|
191
194
|
/**
|
|
192
195
|
* The smallest of `current` and the `count` packed virtual offsets starting at
|
|
193
196
|
* `offset`, allocating at most one VirtualOffset rather than one per entry.
|
|
194
197
|
*
|
|
198
|
+
* 0:0 is skipped rather than treated as the minimum. No BAM record can live
|
|
199
|
+
* there — the magic and header occupy the start of the file — so it is the
|
|
200
|
+
* "unset" placeholder htslib leaves in linear-index windows ahead of a
|
|
201
|
+
* reference's first read (see test/data/HG00096_illumina_lowcov.bam.bai, whose
|
|
202
|
+
* first three windows are 0). Counting it collapses firstDataLine to 0:0 and
|
|
203
|
+
* makes callers size a header read from nothing.
|
|
204
|
+
*
|
|
195
205
|
* The index first pass exists only to find this minimum, and it visits every
|
|
196
206
|
* linear-index entry in the file to do it. Building a VirtualOffset per entry
|
|
197
207
|
* to compare and discard it is the bulk of that pass.
|
|
@@ -215,7 +225,10 @@ export function minVirtualOffset(
|
|
|
215
225
|
bytes[p + 3]! * 0x100 +
|
|
216
226
|
bytes[p + 2]!
|
|
217
227
|
const data = (bytes[p + 1]! << 8) | bytes[p]!
|
|
218
|
-
if (
|
|
228
|
+
if (
|
|
229
|
+
(block !== 0 || data !== 0) &&
|
|
230
|
+
(block < minBlock || (block === minBlock && data < minData))
|
|
231
|
+
) {
|
|
219
232
|
minBlock = block
|
|
220
233
|
minData = data
|
|
221
234
|
found = true
|