jbrowse-plugin-msaview 3.2.0 → 3.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,21 +6,37 @@ import {
6
6
  fetchProteinForGene,
7
7
  resolveGeneId,
8
8
  } from '../utils/ncbiOrthologs'
9
+ import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
9
10
  import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
10
11
 
11
12
  import type { JBrowsePluginMsaViewModel } from './model'
12
13
  import type { OrthologRow } from '../utils/ncbiOrthologs'
13
14
 
15
+ interface Representative {
16
+ accession: string
17
+ sequence: string
18
+ }
19
+
20
+ /** What either source hands the shared tail of the launch. */
21
+ interface FoundOrthologs {
22
+ /** the query gene's id at the source: an NCBI GeneID, or PANTHER's gene xref */
23
+ geneId: string
24
+ representative: Representative | undefined
25
+ rows: OrthologRow[]
26
+ }
27
+
14
28
  /**
15
29
  * The no-search-job alternative to doLaunchBlast.
16
30
  *
17
31
  * BLAST spends 10+ minutes answering "what looks like this sequence" and
18
- * returns a redundant, accession-labelled hit list. This asks NCBI the question
19
- * the alignment actually wants — "what is this gene's ortholog in each species"
20
- * — which NCBI has already computed, so the whole NCBI half returns in about a
21
- * second and only the EBI alignment (~10s) costs real time.
32
+ * returns a redundant, accession-labelled hit list. This asks the question the
33
+ * alignment actually wants — "what is this gene's ortholog in each species" —
34
+ * which NCBI and PANTHER have already computed, so the lookup returns in
35
+ * seconds and only the EBI alignment (~10s) costs real time. `source` picks
36
+ * which of the two answers: NCBI for vertebrates and insects, PANTHER for
37
+ * everything else (yeast, worm, plants, and a fly gene's vertebrate relatives).
22
38
  *
23
- * The query row is the user's OWN selected transcript, not NCBI's
39
+ * The query row is the user's OWN selected transcript, not the source's
24
40
  * representative protein for the query species, because `connectedFeature`
25
41
  * maps genome coordinates through that row — swapping in a different isoform
26
42
  * would silently break the genome<->MSA linkage. The query species is therefore
@@ -38,49 +54,43 @@ export async function doLaunchOrthologs({
38
54
  geneCandidates,
39
55
  msaAlgorithm,
40
56
  proteinSequence,
57
+ source = 'ncbi',
41
58
  } = self.orthologParams!
42
59
 
43
60
  const onProgress = (arg: string) => {
44
61
  self.setProgress(arg)
45
62
  }
46
63
 
47
- onProgress('Resolving gene at NCBI...')
48
- const resolved = await resolveGeneId(geneCandidates, taxId)
49
- if (!resolved) {
50
- throw new Error(
51
- `Could not resolve any of ${geneCandidates.join(', ')} to an NCBI gene in taxon ${taxId}. Try the NCBI BLAST tab, which needs no gene identifier.`,
52
- )
64
+ const request = {
65
+ taxId,
66
+ geneCandidates,
67
+ taxa: taxa ? new Set(taxa) : undefined,
68
+ // the query species is represented by the query row below
69
+ exclude: taxId,
70
+ limit: maxSpecies,
71
+ onProgress,
53
72
  }
73
+ const { geneId, representative, rows } =
74
+ source === 'panther'
75
+ ? await findPantherOrthologs(request)
76
+ : await findNcbiOrthologs(request)
54
77
 
55
78
  // The query row. The dialog always supplies it — it is the user's OWN
56
79
  // selected transcript, which is what makes `connectedFeature` map genome
57
80
  // coordinates through this row. A launch that has no transcript to translate
58
- // (a session spec naming only a gene) falls back to NCBI's representative
59
- // protein for the resolved gene, which is the same choice made for every
60
- // other row, so the alignment is the one NCBI would build for that gene.
61
- const representative = await fetchRepresentativeQueryProtein(resolved.geneId)
81
+ // (a session spec naming only a gene) falls back to the source's
82
+ // representative protein for the resolved gene, which is the same choice
83
+ // made for every other row, so the alignment is the one the source would
84
+ // build for that gene.
62
85
  const cleanedSeq = proteinSequence
63
86
  ? cleanProteinSequence(proteinSequence)
64
87
  : representative?.sequence
65
88
  if (!cleanedSeq) {
66
89
  throw new Error(
67
- `No query protein: none was supplied and NCBI returned no representative protein for gene ${resolved.geneId}.`,
90
+ `No query protein: none was supplied and ${source === 'panther' ? 'PANTHER' : 'NCBI'} returned no representative protein for gene ${geneId}.`,
68
91
  )
69
92
  }
70
93
 
71
- // Every species NCBI has an ortholog for, when a launch names none, capped at
72
- // maxSpecies. A launch that wants specific species lists them; one that just
73
- // wants "this gene across species" gets NCBI's own order, which leads with the
74
- // reference organisms.
75
- const rows = await fetchOrthologRows({
76
- geneId: resolved.geneId,
77
- taxa: taxa ? new Set(taxa) : undefined,
78
- // the query species is represented by the query row above
79
- exclude: taxId,
80
- limit: maxSpecies,
81
- onProgress,
82
- })
83
-
84
94
  // The query row is named for its species like every other row, with a suffix
85
95
  // marking it as the one the genome view is linked to. A bare `QUERY` among
86
96
  // ninety-nine named species reads as a row whose species failed to resolve,
@@ -96,12 +106,7 @@ export async function doLaunchOrthologs({
96
106
  self.setQuerySeqName(queryLabel)
97
107
 
98
108
  const treeMetadata: Record<string, Record<string, string>> = {
99
- [queryLabel]: buildQueryMetadata(
100
- self,
101
- resolved.geneId,
102
- cleanedSeq,
103
- representative,
104
- ),
109
+ [queryLabel]: buildQueryMetadata(self, geneId, cleanedSeq, representative),
105
110
  }
106
111
  for (const row of rows) {
107
112
  treeMetadata[row.label] = buildRowMetadata(row)
@@ -122,6 +127,63 @@ export async function doLaunchOrthologs({
122
127
  }
123
128
  }
124
129
 
130
+ interface OrthologRequest {
131
+ taxId: number
132
+ geneCandidates: string[]
133
+ taxa: Set<number> | undefined
134
+ exclude: number
135
+ limit: number | undefined
136
+ onProgress: (arg: string) => void
137
+ }
138
+
139
+ /**
140
+ * Every species NCBI has an ortholog for, when a launch names none, capped at
141
+ * `limit`. A launch that wants specific species lists them; one that just
142
+ * wants "this gene across species" gets NCBI's own order, which leads with the
143
+ * reference organisms.
144
+ */
145
+ async function findNcbiOrthologs({
146
+ taxId,
147
+ geneCandidates,
148
+ onProgress,
149
+ ...rest
150
+ }: OrthologRequest): Promise<FoundOrthologs> {
151
+ onProgress('Resolving gene at NCBI...')
152
+ const resolved = await resolveGeneId(geneCandidates, taxId)
153
+ if (!resolved) {
154
+ throw new Error(
155
+ `Could not resolve any of ${geneCandidates.join(', ')} to an NCBI gene in taxon ${taxId}. Try the NCBI BLAST tab, which needs no gene identifier.`,
156
+ )
157
+ }
158
+ const representative = await fetchRepresentativeQueryProtein(resolved.geneId)
159
+ const rows = await fetchOrthologRows({
160
+ geneId: resolved.geneId,
161
+ onProgress,
162
+ ...rest,
163
+ })
164
+ return { geneId: resolved.geneId, representative, rows }
165
+ }
166
+
167
+ /**
168
+ * One `matchortho` call resolves the gene, names its own UniProt entry and
169
+ * lists an ortholog per genome, so the representative protein needs no second
170
+ * lookup here.
171
+ */
172
+ async function findPantherOrthologs({
173
+ geneCandidates,
174
+ ...rest
175
+ }: OrthologRequest): Promise<FoundOrthologs> {
176
+ const found = await fetchPantherOrthologs({
177
+ candidates: geneCandidates,
178
+ ...rest,
179
+ })
180
+ return {
181
+ geneId: found.query?.geneRef ?? found.matched,
182
+ representative: found.query,
183
+ rows: found.rows,
184
+ }
185
+ }
186
+
125
187
  /**
126
188
  * `<species>_query`, unique against the ortholog labels. Falls back to the bare
127
189
  * marker when NCBI cannot name the taxon, which is a naming failure and must not
@@ -158,7 +220,7 @@ async function fetchRepresentativeQueryProtein(geneId: string) {
158
220
  /**
159
221
  * The query row carries an Accession — which is what drives the automatic CDD
160
222
  * overlay (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains) —
161
- * ONLY when its sequence is byte-identical to the RefSeq protein that accession
223
+ * ONLY when its sequence is byte-identical to the protein that accession
162
224
  * names. Attaching it unconditionally would put every domain box at an offset
163
225
  * whenever the user picked a non-representative isoform, which is a silently
164
226
  * wrong figure rather than a missing one. A launch that took the representative
@@ -168,7 +230,7 @@ function buildQueryMetadata(
168
230
  self: JBrowsePluginMsaViewModel,
169
231
  geneId: string,
170
232
  proteinSequence: string,
171
- representative: { accession: string; sequence: string } | undefined,
233
+ representative: Representative | undefined,
172
234
  ): Record<string, string> {
173
235
  const transcript = self.orthologParams?.selectedTranscript
174
236
  const metadata: Record<string, string> = { 'Gene ID': geneId }
@@ -51,9 +51,19 @@ export interface BlastParams {
51
51
  proteinSequence: string
52
52
  }
53
53
 
54
+ /**
55
+ * Where the ortholog set comes from. NCBI's sets cover vertebrates and
56
+ * insects; PANTHER's span its 144 reference proteomes, human to yeast to
57
+ * Arabidopsis, so a gene from outside NCBI's scope aligns only through it.
58
+ */
59
+ export const orthologSources = ['ncbi', 'panther'] as const
60
+ export type OrthologSource = (typeof orthologSources)[number]
61
+
54
62
  export interface OrthologParams {
55
63
  /** NCBI taxon id of the assembly the query gene came from */
56
64
  taxId: number
65
+ /** `ncbi` when omitted, so every launch written before this key keeps its meaning */
66
+ source?: OrthologSource
57
67
  /**
58
68
  * taxon ids to include as rows. The query taxon has its own row already, so
59
69
  * it is excluded from this set whether or not it is named.
@@ -0,0 +1,399 @@
1
+ // The second ortholog source, for the genes NCBI's ortholog sets leave out.
2
+ //
3
+ // NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
4
+ // worm or plant gene comes back with orthologs only in its own clade, and a fly
5
+ // gene gets insects and nothing else. PANTHER's ortholog sets span its 144
6
+ // reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
7
+ // answers "this gene's ortholog in every genome" with a UniProt accession per
8
+ // target. Sequences come from one UniProt batch call. Both hosts send
9
+ // `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
10
+ // OMA, OrthoDB and Ensembl are in react-msaview's
11
+ // agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
12
+ //
13
+ // Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
14
+ // labels, the aligner and the CDD overlay do not know which source ran.
15
+ // `protein` is the UniProt accession; efetch serves UniProt accessions as
16
+ // GenPept records with CDD Region features, so the overlay attaches as it does
17
+ // to a RefSeq accession.
18
+
19
+ import { jsonfetch } from './fetch'
20
+ import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs'
21
+ import { fetchTaxonomyInfo } from './taxonomyNames'
22
+
23
+ import type { OrthologRow } from './ncbiOrthologs'
24
+
25
+ const PANTHER = 'https://pantherdb.org/services/oai/pantherdb'
26
+ const UNIPROT = 'https://rest.uniprot.org/uniprotkb'
27
+
28
+ export interface PantherGenome {
29
+ /** PANTHER's organism code, e.g. HUMAN, DROME */
30
+ code: string
31
+ taxId: number
32
+ /** short common name, e.g. fruit_fly */
33
+ name: string
34
+ /** scientific name */
35
+ longName: string
36
+ }
37
+
38
+ interface GenomesResponse {
39
+ search?: {
40
+ output?: {
41
+ genomes?: {
42
+ genome?: {
43
+ short_name?: string
44
+ taxon_id?: number
45
+ name?: string
46
+ long_name?: string
47
+ }[]
48
+ }
49
+ }
50
+ }
51
+ }
52
+
53
+ /**
54
+ * `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
55
+ * names organisms by code only in ortholog results.
56
+ */
57
+ export function parseGenomes(json: unknown): PantherGenome[] {
58
+ const list = (json as GenomesResponse).search?.output?.genomes?.genome ?? []
59
+ return list.flatMap(g =>
60
+ g.short_name && g.taxon_id
61
+ ? [
62
+ {
63
+ code: g.short_name,
64
+ taxId: g.taxon_id,
65
+ name: g.name ?? g.short_name,
66
+ longName: g.long_name ?? g.short_name,
67
+ },
68
+ ]
69
+ : [],
70
+ )
71
+ }
72
+
73
+ export interface PantherGene {
74
+ code: string
75
+ /** UniProt accession */
76
+ accession: string
77
+ /** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
78
+ geneRef: string
79
+ }
80
+
81
+ export interface PantherHit extends PantherGene {
82
+ symbol?: string
83
+ /**
84
+ * LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
85
+ * other ortholog in a one-to-many or many-to-many family
86
+ */
87
+ type: 'LDO' | 'O'
88
+ }
89
+
90
+ interface Mapping {
91
+ id?: string
92
+ gene?: string
93
+ target_gene?: string
94
+ target_gene_symbol?: string | number
95
+ ortholog?: string
96
+ }
97
+
98
+ interface MatchResponse {
99
+ search?: {
100
+ mapping?: {
101
+ /** one object for a single (or empty) match, an array otherwise */
102
+ mapped?: Mapping | Mapping[]
103
+ unmapped_ids?: unknown
104
+ }
105
+ }
106
+ }
107
+
108
+ // "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
109
+ function parseGeneRef(ref: string | undefined): PantherGene | undefined {
110
+ const [code, ...xrefs] = (ref ?? '').split('|')
111
+ const accession = xrefs
112
+ .find(x => x.startsWith('UniProtKB='))
113
+ ?.slice('UniProtKB='.length)
114
+ const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession
115
+ return code && accession && geneRef ? { code, accession, geneRef } : undefined
116
+ }
117
+
118
+ /**
119
+ * `matchortho` -> the query gene (PANTHER names it in every row) and one hit
120
+ * per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
121
+ * no ortholog in the target set comes back as a bare `{ id }`.
122
+ */
123
+ export function parseMatches(json: unknown): {
124
+ unmapped: boolean
125
+ query?: PantherGene
126
+ hits: PantherHit[]
127
+ } {
128
+ const mapping = (json as MatchResponse).search?.mapping
129
+ const mapped = mapping?.mapped
130
+ const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : []
131
+ const hits: PantherHit[] = []
132
+ let query: PantherGene | undefined
133
+ for (const row of rows) {
134
+ query ??= parseGeneRef(row.gene)
135
+ const target = parseGeneRef(row.target_gene)
136
+ if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
137
+ hits.push({
138
+ ...target,
139
+ symbol:
140
+ row.target_gene_symbol === undefined
141
+ ? undefined
142
+ : String(row.target_gene_symbol),
143
+ type: row.ortholog,
144
+ })
145
+ }
146
+ }
147
+ return { unmapped: !!mapping?.unmapped_ids, query, hits }
148
+ }
149
+
150
+ /**
151
+ * One hit per organism, in first-seen order: the LDO where PANTHER named one,
152
+ * else the first other ortholog it listed. A many-to-many family (the Hox
153
+ * genes) has no LDO at all, so dropping to "first O" is what keeps those
154
+ * species in the alignment.
155
+ */
156
+ export function pickOnePerGenome(hits: PantherHit[]): PantherHit[] {
157
+ const byCode = new Map<string, PantherHit>()
158
+ for (const hit of hits) {
159
+ const current = byCode.get(hit.code)
160
+ if (!current || (current.type === 'O' && hit.type === 'LDO')) {
161
+ byCode.set(hit.code, hit)
162
+ }
163
+ }
164
+ return [...byCode.values()]
165
+ }
166
+
167
+ interface UniProtResponse {
168
+ results?: {
169
+ primaryAccession?: string
170
+ sequence?: { value?: string }
171
+ }[]
172
+ }
173
+
174
+ /** `uniprotkb/accessions` -> accession -> sequence */
175
+ export function parseSequences(json: unknown): Map<string, string> {
176
+ const map = new Map<string, string>()
177
+ for (const r of (json as UniProtResponse).results ?? []) {
178
+ if (r.primaryAccession && r.sequence?.value) {
179
+ map.set(r.primaryAccession, r.sequence.value)
180
+ }
181
+ }
182
+ return map
183
+ }
184
+
185
+ let genomes: Promise<PantherGenome[]> | undefined
186
+
187
+ /** The proteome list, fetched once per page and forgotten on failure. */
188
+ export function fetchGenomes() {
189
+ genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
190
+ .then(parseGenomes)
191
+ .catch((e: unknown) => {
192
+ genomes = undefined
193
+ throw e
194
+ })
195
+ return genomes
196
+ }
197
+
198
+ // UniProt caps one `accessions` call at 100 ids
199
+ const UNIPROT_CHUNK = 100
200
+
201
+ async function fetchSequences(accessions: string[]) {
202
+ const map = new Map<string, string>()
203
+ for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
204
+ const chunk = accessions.slice(i, i + UNIPROT_CHUNK)
205
+ const json = await jsonfetch(
206
+ `${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`,
207
+ )
208
+ for (const [acc, seq] of parseSequences(json)) {
209
+ map.set(acc, seq)
210
+ }
211
+ }
212
+ return map
213
+ }
214
+
215
+ // strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
216
+ // resolveGeneId does for NCBI
217
+ function cleanCandidate(raw: string) {
218
+ return raw
219
+ .trim()
220
+ .replace(/^\w+:/, '')
221
+ .replace(/\.\d+$/, '')
222
+ }
223
+
224
+ /**
225
+ * One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
226
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
227
+ * some of those are names PANTHER knows. `targets` omitted asks for every
228
+ * genome PANTHER has, which is one call rather than one per genome.
229
+ */
230
+ async function matchOrthologs(
231
+ candidates: string[],
232
+ taxId: number,
233
+ targets: PantherGenome[] | undefined,
234
+ ) {
235
+ let matched: string | undefined
236
+ for (const raw of candidates) {
237
+ const query = cleanCandidate(raw)
238
+ if (!query) {
239
+ continue
240
+ }
241
+ const params = new URLSearchParams({
242
+ geneInputList: query,
243
+ organism: String(taxId),
244
+ orthologType: 'all',
245
+ })
246
+ if (targets) {
247
+ params.set('targetOrganism', targets.map(t => t.taxId).join(','))
248
+ }
249
+ const parsed = parseMatches(
250
+ await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`),
251
+ )
252
+ if (!parsed.unmapped) {
253
+ matched ??= query
254
+ if (parsed.query) {
255
+ return { ...parsed, matched: query }
256
+ }
257
+ }
258
+ }
259
+ return matched === undefined
260
+ ? undefined
261
+ : { matched, hits: [], query: undefined }
262
+ }
263
+
264
+ export interface PantherOrthologs {
265
+ /** the candidate PANTHER recognised */
266
+ matched: string
267
+ /** the query gene as PANTHER knows it, with its UniProt sequence */
268
+ query?: PantherGene & { sequence: string }
269
+ rows: OrthologRow[]
270
+ }
271
+
272
+ /**
273
+ * The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
274
+ * labels, accessions and sequences, plus the query gene's own protein for the
275
+ * query row. Two lookups (genomes, orthologs), one taxonomy batch for the
276
+ * labels, one UniProt batch for the sequences.
277
+ *
278
+ * `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
279
+ * `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
280
+ * `exclude` drops the query taxon, `limit` caps the rows before their
281
+ * sequences are fetched.
282
+ */
283
+ export async function fetchPantherOrthologs({
284
+ candidates,
285
+ taxId,
286
+ taxa,
287
+ exclude,
288
+ limit = defaultMaxSpecies,
289
+ onProgress,
290
+ }: {
291
+ candidates: string[]
292
+ taxId: number
293
+ taxa?: Set<number>
294
+ exclude?: number
295
+ limit?: number
296
+ onProgress: (arg: string) => void
297
+ }): Promise<PantherOrthologs> {
298
+ const all = await fetchGenomes()
299
+ const byTaxId = new Map(all.map(g => [g.taxId, g]))
300
+ const byCode = new Map(all.map(g => [g.code, g]))
301
+ const queryGenome = byTaxId.get(taxId)
302
+ if (!queryGenome) {
303
+ throw new Error(
304
+ `PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`,
305
+ )
306
+ }
307
+ const targets = taxa
308
+ ? [...taxa].flatMap(t => {
309
+ const g = byTaxId.get(t)
310
+ return g && t !== exclude ? [g] : []
311
+ })
312
+ : undefined
313
+
314
+ onProgress('Matching orthologs at PANTHER...')
315
+ const match = await matchOrthologs(candidates, taxId, targets)
316
+ if (!match) {
317
+ throw new Error(
318
+ `PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`,
319
+ )
320
+ }
321
+
322
+ const rank = new Map(targets?.map((t, i) => [t.taxId, i]))
323
+ const picks = pickOnePerGenome(match.hits)
324
+ .map(hit => ({ hit, genome: byCode.get(hit.code) }))
325
+ .filter(
326
+ (p): p is { hit: PantherHit; genome: PantherGenome } =>
327
+ !!p.genome &&
328
+ p.genome.taxId !== exclude &&
329
+ (targets ? rank.has(p.genome.taxId) : true),
330
+ )
331
+ .sort((a, b) =>
332
+ targets ? rank.get(a.genome.taxId)! - rank.get(b.genome.taxId)! : 0,
333
+ )
334
+ .slice(0, limit)
335
+ if (picks.length < 2) {
336
+ throw new Error(
337
+ `Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`,
338
+ )
339
+ }
340
+
341
+ onProgress(`Fetching ${picks.length} protein sequences from UniProt...`)
342
+ const [names, sequences] = await Promise.all([
343
+ taxonomyNames(picks.map(p => p.genome.taxId)),
344
+ fetchSequences([
345
+ ...(match.query ? [match.query.accession] : []),
346
+ ...picks.map(p => p.hit.accession),
347
+ ]),
348
+ ])
349
+
350
+ const described = picks.map(({ hit, genome }) => {
351
+ const info = names.get(genome.taxId)
352
+ return {
353
+ hit,
354
+ genome,
355
+ name: info ? (info.commonName ?? info.sciname) : genome.name,
356
+ scientificName: info?.sciname || genome.longName,
357
+ commonName: info?.commonName,
358
+ }
359
+ })
360
+ const labels = dedupeLabels(described.map(d => d.name))
361
+ const rows = described
362
+ .map(({ hit, genome, scientificName, commonName }, i) => ({
363
+ taxId: genome.taxId,
364
+ label: labels[i]!,
365
+ scientificName,
366
+ commonName,
367
+ geneId: hit.geneRef,
368
+ protein: hit.accession,
369
+ sequence: sequences.get(hit.accession) ?? '',
370
+ }))
371
+ .filter(r => r.sequence)
372
+ if (rows.length < 2) {
373
+ throw new Error('Could not fetch protein sequences for the orthologs')
374
+ }
375
+
376
+ const querySequence = match.query && sequences.get(match.query.accession)
377
+ return {
378
+ matched: match.matched,
379
+ query:
380
+ match.query && querySequence
381
+ ? { ...match.query, sequence: querySequence }
382
+ : undefined,
383
+ rows,
384
+ }
385
+ }
386
+
387
+ /**
388
+ * NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
389
+ * are. A failed lookup only costs the labels, which fall back to PANTHER's own
390
+ * short names, so it is logged rather than thrown.
391
+ */
392
+ async function taxonomyNames(taxIds: number[]) {
393
+ try {
394
+ return await fetchTaxonomyInfo(taxIds)
395
+ } catch (e) {
396
+ console.warn('[msaview-orthologs] taxonomy name lookup failed:', e)
397
+ return new Map<number, { sciname: string; commonName?: string }>()
398
+ }
399
+ }
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const version = '3.2.0'
1
+ export const version = '3.3.0'