jbrowse-plugin-msaview 3.2.0 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +7 -2
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.d.ts +9 -0
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.js +20 -0
- package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +7 -5
- package/dist/MsaViewPanel/doLaunchOrthologs.js +64 -30
- package/dist/MsaViewPanel/doLaunchOrthologs.test.js +105 -0
- package/dist/MsaViewPanel/model.d.ts +9 -0
- package/dist/MsaViewPanel/model.js +6 -0
- package/dist/jbrowse-plugin-msaview.umd.production.min.js +30 -30
- package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
- package/dist/utils/pantherOrthologs.d.ts +79 -0
- package/dist/utils/pantherOrthologs.js +262 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +1 -1
- package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +19 -3
- package/src/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.tsx +52 -0
- package/src/MsaViewPanel/doLaunchOrthologs.test.ts +117 -0
- package/src/MsaViewPanel/doLaunchOrthologs.ts +99 -37
- package/src/MsaViewPanel/model.ts +10 -0
- package/src/utils/pantherOrthologs.ts +399 -0
- package/src/version.ts +1 -1
|
@@ -6,21 +6,37 @@ import {
|
|
|
6
6
|
fetchProteinForGene,
|
|
7
7
|
resolveGeneId,
|
|
8
8
|
} from '../utils/ncbiOrthologs'
|
|
9
|
+
import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
|
|
9
10
|
import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
|
|
10
11
|
|
|
11
12
|
import type { JBrowsePluginMsaViewModel } from './model'
|
|
12
13
|
import type { OrthologRow } from '../utils/ncbiOrthologs'
|
|
13
14
|
|
|
15
|
+
interface Representative {
|
|
16
|
+
accession: string
|
|
17
|
+
sequence: string
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** What either source hands the shared tail of the launch. */
|
|
21
|
+
interface FoundOrthologs {
|
|
22
|
+
/** the query gene's id at the source: an NCBI GeneID, or PANTHER's gene xref */
|
|
23
|
+
geneId: string
|
|
24
|
+
representative: Representative | undefined
|
|
25
|
+
rows: OrthologRow[]
|
|
26
|
+
}
|
|
27
|
+
|
|
14
28
|
/**
|
|
15
29
|
* The no-search-job alternative to doLaunchBlast.
|
|
16
30
|
*
|
|
17
31
|
* BLAST spends 10+ minutes answering "what looks like this sequence" and
|
|
18
|
-
* returns a redundant, accession-labelled hit list. This asks
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
32
|
+
* returns a redundant, accession-labelled hit list. This asks the question the
|
|
33
|
+
* alignment actually wants — "what is this gene's ortholog in each species" —
|
|
34
|
+
* which NCBI and PANTHER have already computed, so the lookup returns in
|
|
35
|
+
* seconds and only the EBI alignment (~10s) costs real time. `source` picks
|
|
36
|
+
* which of the two answers: NCBI for vertebrates and insects, PANTHER for
|
|
37
|
+
* everything else (yeast, worm, plants, and a fly gene's vertebrate relatives).
|
|
22
38
|
*
|
|
23
|
-
* The query row is the user's OWN selected transcript, not
|
|
39
|
+
* The query row is the user's OWN selected transcript, not the source's
|
|
24
40
|
* representative protein for the query species, because `connectedFeature`
|
|
25
41
|
* maps genome coordinates through that row — swapping in a different isoform
|
|
26
42
|
* would silently break the genome<->MSA linkage. The query species is therefore
|
|
@@ -38,49 +54,43 @@ export async function doLaunchOrthologs({
|
|
|
38
54
|
geneCandidates,
|
|
39
55
|
msaAlgorithm,
|
|
40
56
|
proteinSequence,
|
|
57
|
+
source = 'ncbi',
|
|
41
58
|
} = self.orthologParams!
|
|
42
59
|
|
|
43
60
|
const onProgress = (arg: string) => {
|
|
44
61
|
self.setProgress(arg)
|
|
45
62
|
}
|
|
46
63
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
64
|
+
const request = {
|
|
65
|
+
taxId,
|
|
66
|
+
geneCandidates,
|
|
67
|
+
taxa: taxa ? new Set(taxa) : undefined,
|
|
68
|
+
// the query species is represented by the query row below
|
|
69
|
+
exclude: taxId,
|
|
70
|
+
limit: maxSpecies,
|
|
71
|
+
onProgress,
|
|
53
72
|
}
|
|
73
|
+
const { geneId, representative, rows } =
|
|
74
|
+
source === 'panther'
|
|
75
|
+
? await findPantherOrthologs(request)
|
|
76
|
+
: await findNcbiOrthologs(request)
|
|
54
77
|
|
|
55
78
|
// The query row. The dialog always supplies it — it is the user's OWN
|
|
56
79
|
// selected transcript, which is what makes `connectedFeature` map genome
|
|
57
80
|
// coordinates through this row. A launch that has no transcript to translate
|
|
58
|
-
// (a session spec naming only a gene) falls back to
|
|
59
|
-
// protein for the resolved gene, which is the same choice
|
|
60
|
-
// other row, so the alignment is the one
|
|
61
|
-
|
|
81
|
+
// (a session spec naming only a gene) falls back to the source's
|
|
82
|
+
// representative protein for the resolved gene, which is the same choice
|
|
83
|
+
// made for every other row, so the alignment is the one the source would
|
|
84
|
+
// build for that gene.
|
|
62
85
|
const cleanedSeq = proteinSequence
|
|
63
86
|
? cleanProteinSequence(proteinSequence)
|
|
64
87
|
: representative?.sequence
|
|
65
88
|
if (!cleanedSeq) {
|
|
66
89
|
throw new Error(
|
|
67
|
-
`No query protein: none was supplied and NCBI returned no representative protein for gene ${
|
|
90
|
+
`No query protein: none was supplied and ${source === 'panther' ? 'PANTHER' : 'NCBI'} returned no representative protein for gene ${geneId}.`,
|
|
68
91
|
)
|
|
69
92
|
}
|
|
70
93
|
|
|
71
|
-
// Every species NCBI has an ortholog for, when a launch names none, capped at
|
|
72
|
-
// maxSpecies. A launch that wants specific species lists them; one that just
|
|
73
|
-
// wants "this gene across species" gets NCBI's own order, which leads with the
|
|
74
|
-
// reference organisms.
|
|
75
|
-
const rows = await fetchOrthologRows({
|
|
76
|
-
geneId: resolved.geneId,
|
|
77
|
-
taxa: taxa ? new Set(taxa) : undefined,
|
|
78
|
-
// the query species is represented by the query row above
|
|
79
|
-
exclude: taxId,
|
|
80
|
-
limit: maxSpecies,
|
|
81
|
-
onProgress,
|
|
82
|
-
})
|
|
83
|
-
|
|
84
94
|
// The query row is named for its species like every other row, with a suffix
|
|
85
95
|
// marking it as the one the genome view is linked to. A bare `QUERY` among
|
|
86
96
|
// ninety-nine named species reads as a row whose species failed to resolve,
|
|
@@ -96,12 +106,7 @@ export async function doLaunchOrthologs({
|
|
|
96
106
|
self.setQuerySeqName(queryLabel)
|
|
97
107
|
|
|
98
108
|
const treeMetadata: Record<string, Record<string, string>> = {
|
|
99
|
-
[queryLabel]: buildQueryMetadata(
|
|
100
|
-
self,
|
|
101
|
-
resolved.geneId,
|
|
102
|
-
cleanedSeq,
|
|
103
|
-
representative,
|
|
104
|
-
),
|
|
109
|
+
[queryLabel]: buildQueryMetadata(self, geneId, cleanedSeq, representative),
|
|
105
110
|
}
|
|
106
111
|
for (const row of rows) {
|
|
107
112
|
treeMetadata[row.label] = buildRowMetadata(row)
|
|
@@ -122,6 +127,63 @@ export async function doLaunchOrthologs({
|
|
|
122
127
|
}
|
|
123
128
|
}
|
|
124
129
|
|
|
130
|
+
interface OrthologRequest {
|
|
131
|
+
taxId: number
|
|
132
|
+
geneCandidates: string[]
|
|
133
|
+
taxa: Set<number> | undefined
|
|
134
|
+
exclude: number
|
|
135
|
+
limit: number | undefined
|
|
136
|
+
onProgress: (arg: string) => void
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Every species NCBI has an ortholog for, when a launch names none, capped at
|
|
141
|
+
* `limit`. A launch that wants specific species lists them; one that just
|
|
142
|
+
* wants "this gene across species" gets NCBI's own order, which leads with the
|
|
143
|
+
* reference organisms.
|
|
144
|
+
*/
|
|
145
|
+
async function findNcbiOrthologs({
|
|
146
|
+
taxId,
|
|
147
|
+
geneCandidates,
|
|
148
|
+
onProgress,
|
|
149
|
+
...rest
|
|
150
|
+
}: OrthologRequest): Promise<FoundOrthologs> {
|
|
151
|
+
onProgress('Resolving gene at NCBI...')
|
|
152
|
+
const resolved = await resolveGeneId(geneCandidates, taxId)
|
|
153
|
+
if (!resolved) {
|
|
154
|
+
throw new Error(
|
|
155
|
+
`Could not resolve any of ${geneCandidates.join(', ')} to an NCBI gene in taxon ${taxId}. Try the NCBI BLAST tab, which needs no gene identifier.`,
|
|
156
|
+
)
|
|
157
|
+
}
|
|
158
|
+
const representative = await fetchRepresentativeQueryProtein(resolved.geneId)
|
|
159
|
+
const rows = await fetchOrthologRows({
|
|
160
|
+
geneId: resolved.geneId,
|
|
161
|
+
onProgress,
|
|
162
|
+
...rest,
|
|
163
|
+
})
|
|
164
|
+
return { geneId: resolved.geneId, representative, rows }
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* One `matchortho` call resolves the gene, names its own UniProt entry and
|
|
169
|
+
* lists an ortholog per genome, so the representative protein needs no second
|
|
170
|
+
* lookup here.
|
|
171
|
+
*/
|
|
172
|
+
async function findPantherOrthologs({
|
|
173
|
+
geneCandidates,
|
|
174
|
+
...rest
|
|
175
|
+
}: OrthologRequest): Promise<FoundOrthologs> {
|
|
176
|
+
const found = await fetchPantherOrthologs({
|
|
177
|
+
candidates: geneCandidates,
|
|
178
|
+
...rest,
|
|
179
|
+
})
|
|
180
|
+
return {
|
|
181
|
+
geneId: found.query?.geneRef ?? found.matched,
|
|
182
|
+
representative: found.query,
|
|
183
|
+
rows: found.rows,
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
125
187
|
/**
|
|
126
188
|
* `<species>_query`, unique against the ortholog labels. Falls back to the bare
|
|
127
189
|
* marker when NCBI cannot name the taxon, which is a naming failure and must not
|
|
@@ -158,7 +220,7 @@ async function fetchRepresentativeQueryProtein(geneId: string) {
|
|
|
158
220
|
/**
|
|
159
221
|
* The query row carries an Accession — which is what drives the automatic CDD
|
|
160
222
|
* overlay (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains) —
|
|
161
|
-
* ONLY when its sequence is byte-identical to the
|
|
223
|
+
* ONLY when its sequence is byte-identical to the protein that accession
|
|
162
224
|
* names. Attaching it unconditionally would put every domain box at an offset
|
|
163
225
|
* whenever the user picked a non-representative isoform, which is a silently
|
|
164
226
|
* wrong figure rather than a missing one. A launch that took the representative
|
|
@@ -168,7 +230,7 @@ function buildQueryMetadata(
|
|
|
168
230
|
self: JBrowsePluginMsaViewModel,
|
|
169
231
|
geneId: string,
|
|
170
232
|
proteinSequence: string,
|
|
171
|
-
representative:
|
|
233
|
+
representative: Representative | undefined,
|
|
172
234
|
): Record<string, string> {
|
|
173
235
|
const transcript = self.orthologParams?.selectedTranscript
|
|
174
236
|
const metadata: Record<string, string> = { 'Gene ID': geneId }
|
|
@@ -51,9 +51,19 @@ export interface BlastParams {
|
|
|
51
51
|
proteinSequence: string
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
/**
|
|
55
|
+
* Where the ortholog set comes from. NCBI's sets cover vertebrates and
|
|
56
|
+
* insects; PANTHER's span its 144 reference proteomes, human to yeast to
|
|
57
|
+
* Arabidopsis, so a gene from outside NCBI's scope aligns only through it.
|
|
58
|
+
*/
|
|
59
|
+
export const orthologSources = ['ncbi', 'panther'] as const
|
|
60
|
+
export type OrthologSource = (typeof orthologSources)[number]
|
|
61
|
+
|
|
54
62
|
export interface OrthologParams {
|
|
55
63
|
/** NCBI taxon id of the assembly the query gene came from */
|
|
56
64
|
taxId: number
|
|
65
|
+
/** `ncbi` when omitted, so every launch written before this key keeps its meaning */
|
|
66
|
+
source?: OrthologSource
|
|
57
67
|
/**
|
|
58
68
|
* taxon ids to include as rows. The query taxon has its own row already, so
|
|
59
69
|
* it is excluded from this set whether or not it is named.
|
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
// The second ortholog source, for the genes NCBI's ortholog sets leave out.
|
|
2
|
+
//
|
|
3
|
+
// NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
|
|
4
|
+
// worm or plant gene comes back with orthologs only in its own clade, and a fly
|
|
5
|
+
// gene gets insects and nothing else. PANTHER's ortholog sets span its 144
|
|
6
|
+
// reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
|
|
7
|
+
// answers "this gene's ortholog in every genome" with a UniProt accession per
|
|
8
|
+
// target. Sequences come from one UniProt batch call. Both hosts send
|
|
9
|
+
// `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
|
|
10
|
+
// OMA, OrthoDB and Ensembl are in react-msaview's
|
|
11
|
+
// agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
|
|
12
|
+
//
|
|
13
|
+
// Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
|
|
14
|
+
// labels, the aligner and the CDD overlay do not know which source ran.
|
|
15
|
+
// `protein` is the UniProt accession; efetch serves UniProt accessions as
|
|
16
|
+
// GenPept records with CDD Region features, so the overlay attaches as it does
|
|
17
|
+
// to a RefSeq accession.
|
|
18
|
+
|
|
19
|
+
import { jsonfetch } from './fetch'
|
|
20
|
+
import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs'
|
|
21
|
+
import { fetchTaxonomyInfo } from './taxonomyNames'
|
|
22
|
+
|
|
23
|
+
import type { OrthologRow } from './ncbiOrthologs'
|
|
24
|
+
|
|
25
|
+
const PANTHER = 'https://pantherdb.org/services/oai/pantherdb'
|
|
26
|
+
const UNIPROT = 'https://rest.uniprot.org/uniprotkb'
|
|
27
|
+
|
|
28
|
+
export interface PantherGenome {
|
|
29
|
+
/** PANTHER's organism code, e.g. HUMAN, DROME */
|
|
30
|
+
code: string
|
|
31
|
+
taxId: number
|
|
32
|
+
/** short common name, e.g. fruit_fly */
|
|
33
|
+
name: string
|
|
34
|
+
/** scientific name */
|
|
35
|
+
longName: string
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
interface GenomesResponse {
|
|
39
|
+
search?: {
|
|
40
|
+
output?: {
|
|
41
|
+
genomes?: {
|
|
42
|
+
genome?: {
|
|
43
|
+
short_name?: string
|
|
44
|
+
taxon_id?: number
|
|
45
|
+
name?: string
|
|
46
|
+
long_name?: string
|
|
47
|
+
}[]
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
|
|
55
|
+
* names organisms by code only in ortholog results.
|
|
56
|
+
*/
|
|
57
|
+
export function parseGenomes(json: unknown): PantherGenome[] {
|
|
58
|
+
const list = (json as GenomesResponse).search?.output?.genomes?.genome ?? []
|
|
59
|
+
return list.flatMap(g =>
|
|
60
|
+
g.short_name && g.taxon_id
|
|
61
|
+
? [
|
|
62
|
+
{
|
|
63
|
+
code: g.short_name,
|
|
64
|
+
taxId: g.taxon_id,
|
|
65
|
+
name: g.name ?? g.short_name,
|
|
66
|
+
longName: g.long_name ?? g.short_name,
|
|
67
|
+
},
|
|
68
|
+
]
|
|
69
|
+
: [],
|
|
70
|
+
)
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface PantherGene {
|
|
74
|
+
code: string
|
|
75
|
+
/** UniProt accession */
|
|
76
|
+
accession: string
|
|
77
|
+
/** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
|
|
78
|
+
geneRef: string
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export interface PantherHit extends PantherGene {
|
|
82
|
+
symbol?: string
|
|
83
|
+
/**
|
|
84
|
+
* LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
|
|
85
|
+
* other ortholog in a one-to-many or many-to-many family
|
|
86
|
+
*/
|
|
87
|
+
type: 'LDO' | 'O'
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
interface Mapping {
|
|
91
|
+
id?: string
|
|
92
|
+
gene?: string
|
|
93
|
+
target_gene?: string
|
|
94
|
+
target_gene_symbol?: string | number
|
|
95
|
+
ortholog?: string
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
interface MatchResponse {
|
|
99
|
+
search?: {
|
|
100
|
+
mapping?: {
|
|
101
|
+
/** one object for a single (or empty) match, an array otherwise */
|
|
102
|
+
mapped?: Mapping | Mapping[]
|
|
103
|
+
unmapped_ids?: unknown
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
|
|
109
|
+
function parseGeneRef(ref: string | undefined): PantherGene | undefined {
|
|
110
|
+
const [code, ...xrefs] = (ref ?? '').split('|')
|
|
111
|
+
const accession = xrefs
|
|
112
|
+
.find(x => x.startsWith('UniProtKB='))
|
|
113
|
+
?.slice('UniProtKB='.length)
|
|
114
|
+
const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession
|
|
115
|
+
return code && accession && geneRef ? { code, accession, geneRef } : undefined
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* `matchortho` -> the query gene (PANTHER names it in every row) and one hit
|
|
120
|
+
* per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
|
|
121
|
+
* no ortholog in the target set comes back as a bare `{ id }`.
|
|
122
|
+
*/
|
|
123
|
+
export function parseMatches(json: unknown): {
|
|
124
|
+
unmapped: boolean
|
|
125
|
+
query?: PantherGene
|
|
126
|
+
hits: PantherHit[]
|
|
127
|
+
} {
|
|
128
|
+
const mapping = (json as MatchResponse).search?.mapping
|
|
129
|
+
const mapped = mapping?.mapped
|
|
130
|
+
const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : []
|
|
131
|
+
const hits: PantherHit[] = []
|
|
132
|
+
let query: PantherGene | undefined
|
|
133
|
+
for (const row of rows) {
|
|
134
|
+
query ??= parseGeneRef(row.gene)
|
|
135
|
+
const target = parseGeneRef(row.target_gene)
|
|
136
|
+
if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
|
|
137
|
+
hits.push({
|
|
138
|
+
...target,
|
|
139
|
+
symbol:
|
|
140
|
+
row.target_gene_symbol === undefined
|
|
141
|
+
? undefined
|
|
142
|
+
: String(row.target_gene_symbol),
|
|
143
|
+
type: row.ortholog,
|
|
144
|
+
})
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
return { unmapped: !!mapping?.unmapped_ids, query, hits }
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* One hit per organism, in first-seen order: the LDO where PANTHER named one,
|
|
152
|
+
* else the first other ortholog it listed. A many-to-many family (the Hox
|
|
153
|
+
* genes) has no LDO at all, so dropping to "first O" is what keeps those
|
|
154
|
+
* species in the alignment.
|
|
155
|
+
*/
|
|
156
|
+
export function pickOnePerGenome(hits: PantherHit[]): PantherHit[] {
|
|
157
|
+
const byCode = new Map<string, PantherHit>()
|
|
158
|
+
for (const hit of hits) {
|
|
159
|
+
const current = byCode.get(hit.code)
|
|
160
|
+
if (!current || (current.type === 'O' && hit.type === 'LDO')) {
|
|
161
|
+
byCode.set(hit.code, hit)
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return [...byCode.values()]
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
interface UniProtResponse {
|
|
168
|
+
results?: {
|
|
169
|
+
primaryAccession?: string
|
|
170
|
+
sequence?: { value?: string }
|
|
171
|
+
}[]
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** `uniprotkb/accessions` -> accession -> sequence */
|
|
175
|
+
export function parseSequences(json: unknown): Map<string, string> {
|
|
176
|
+
const map = new Map<string, string>()
|
|
177
|
+
for (const r of (json as UniProtResponse).results ?? []) {
|
|
178
|
+
if (r.primaryAccession && r.sequence?.value) {
|
|
179
|
+
map.set(r.primaryAccession, r.sequence.value)
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
return map
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
let genomes: Promise<PantherGenome[]> | undefined
|
|
186
|
+
|
|
187
|
+
/** The proteome list, fetched once per page and forgotten on failure. */
|
|
188
|
+
export function fetchGenomes() {
|
|
189
|
+
genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
|
|
190
|
+
.then(parseGenomes)
|
|
191
|
+
.catch((e: unknown) => {
|
|
192
|
+
genomes = undefined
|
|
193
|
+
throw e
|
|
194
|
+
})
|
|
195
|
+
return genomes
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// UniProt caps one `accessions` call at 100 ids
|
|
199
|
+
const UNIPROT_CHUNK = 100
|
|
200
|
+
|
|
201
|
+
async function fetchSequences(accessions: string[]) {
|
|
202
|
+
const map = new Map<string, string>()
|
|
203
|
+
for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
|
|
204
|
+
const chunk = accessions.slice(i, i + UNIPROT_CHUNK)
|
|
205
|
+
const json = await jsonfetch(
|
|
206
|
+
`${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`,
|
|
207
|
+
)
|
|
208
|
+
for (const [acc, seq] of parseSequences(json)) {
|
|
209
|
+
map.set(acc, seq)
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
return map
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
|
|
216
|
+
// resolveGeneId does for NCBI
|
|
217
|
+
function cleanCandidate(raw: string) {
|
|
218
|
+
return raw
|
|
219
|
+
.trim()
|
|
220
|
+
.replace(/^\w+:/, '')
|
|
221
|
+
.replace(/\.\d+$/, '')
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
|
|
226
|
+
* carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
|
|
227
|
+
* some of those are names PANTHER knows. `targets` omitted asks for every
|
|
228
|
+
* genome PANTHER has, which is one call rather than one per genome.
|
|
229
|
+
*/
|
|
230
|
+
async function matchOrthologs(
|
|
231
|
+
candidates: string[],
|
|
232
|
+
taxId: number,
|
|
233
|
+
targets: PantherGenome[] | undefined,
|
|
234
|
+
) {
|
|
235
|
+
let matched: string | undefined
|
|
236
|
+
for (const raw of candidates) {
|
|
237
|
+
const query = cleanCandidate(raw)
|
|
238
|
+
if (!query) {
|
|
239
|
+
continue
|
|
240
|
+
}
|
|
241
|
+
const params = new URLSearchParams({
|
|
242
|
+
geneInputList: query,
|
|
243
|
+
organism: String(taxId),
|
|
244
|
+
orthologType: 'all',
|
|
245
|
+
})
|
|
246
|
+
if (targets) {
|
|
247
|
+
params.set('targetOrganism', targets.map(t => t.taxId).join(','))
|
|
248
|
+
}
|
|
249
|
+
const parsed = parseMatches(
|
|
250
|
+
await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`),
|
|
251
|
+
)
|
|
252
|
+
if (!parsed.unmapped) {
|
|
253
|
+
matched ??= query
|
|
254
|
+
if (parsed.query) {
|
|
255
|
+
return { ...parsed, matched: query }
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
return matched === undefined
|
|
260
|
+
? undefined
|
|
261
|
+
: { matched, hits: [], query: undefined }
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
export interface PantherOrthologs {
|
|
265
|
+
/** the candidate PANTHER recognised */
|
|
266
|
+
matched: string
|
|
267
|
+
/** the query gene as PANTHER knows it, with its UniProt sequence */
|
|
268
|
+
query?: PantherGene & { sequence: string }
|
|
269
|
+
rows: OrthologRow[]
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
|
|
274
|
+
* labels, accessions and sequences, plus the query gene's own protein for the
|
|
275
|
+
* query row. Two lookups (genomes, orthologs), one taxonomy batch for the
|
|
276
|
+
* labels, one UniProt batch for the sequences.
|
|
277
|
+
*
|
|
278
|
+
* `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
|
|
279
|
+
* `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
|
|
280
|
+
* `exclude` drops the query taxon, `limit` caps the rows before their
|
|
281
|
+
* sequences are fetched.
|
|
282
|
+
*/
|
|
283
|
+
export async function fetchPantherOrthologs({
|
|
284
|
+
candidates,
|
|
285
|
+
taxId,
|
|
286
|
+
taxa,
|
|
287
|
+
exclude,
|
|
288
|
+
limit = defaultMaxSpecies,
|
|
289
|
+
onProgress,
|
|
290
|
+
}: {
|
|
291
|
+
candidates: string[]
|
|
292
|
+
taxId: number
|
|
293
|
+
taxa?: Set<number>
|
|
294
|
+
exclude?: number
|
|
295
|
+
limit?: number
|
|
296
|
+
onProgress: (arg: string) => void
|
|
297
|
+
}): Promise<PantherOrthologs> {
|
|
298
|
+
const all = await fetchGenomes()
|
|
299
|
+
const byTaxId = new Map(all.map(g => [g.taxId, g]))
|
|
300
|
+
const byCode = new Map(all.map(g => [g.code, g]))
|
|
301
|
+
const queryGenome = byTaxId.get(taxId)
|
|
302
|
+
if (!queryGenome) {
|
|
303
|
+
throw new Error(
|
|
304
|
+
`PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`,
|
|
305
|
+
)
|
|
306
|
+
}
|
|
307
|
+
const targets = taxa
|
|
308
|
+
? [...taxa].flatMap(t => {
|
|
309
|
+
const g = byTaxId.get(t)
|
|
310
|
+
return g && t !== exclude ? [g] : []
|
|
311
|
+
})
|
|
312
|
+
: undefined
|
|
313
|
+
|
|
314
|
+
onProgress('Matching orthologs at PANTHER...')
|
|
315
|
+
const match = await matchOrthologs(candidates, taxId, targets)
|
|
316
|
+
if (!match) {
|
|
317
|
+
throw new Error(
|
|
318
|
+
`PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`,
|
|
319
|
+
)
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
const rank = new Map(targets?.map((t, i) => [t.taxId, i]))
|
|
323
|
+
const picks = pickOnePerGenome(match.hits)
|
|
324
|
+
.map(hit => ({ hit, genome: byCode.get(hit.code) }))
|
|
325
|
+
.filter(
|
|
326
|
+
(p): p is { hit: PantherHit; genome: PantherGenome } =>
|
|
327
|
+
!!p.genome &&
|
|
328
|
+
p.genome.taxId !== exclude &&
|
|
329
|
+
(targets ? rank.has(p.genome.taxId) : true),
|
|
330
|
+
)
|
|
331
|
+
.sort((a, b) =>
|
|
332
|
+
targets ? rank.get(a.genome.taxId)! - rank.get(b.genome.taxId)! : 0,
|
|
333
|
+
)
|
|
334
|
+
.slice(0, limit)
|
|
335
|
+
if (picks.length < 2) {
|
|
336
|
+
throw new Error(
|
|
337
|
+
`Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`,
|
|
338
|
+
)
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
onProgress(`Fetching ${picks.length} protein sequences from UniProt...`)
|
|
342
|
+
const [names, sequences] = await Promise.all([
|
|
343
|
+
taxonomyNames(picks.map(p => p.genome.taxId)),
|
|
344
|
+
fetchSequences([
|
|
345
|
+
...(match.query ? [match.query.accession] : []),
|
|
346
|
+
...picks.map(p => p.hit.accession),
|
|
347
|
+
]),
|
|
348
|
+
])
|
|
349
|
+
|
|
350
|
+
const described = picks.map(({ hit, genome }) => {
|
|
351
|
+
const info = names.get(genome.taxId)
|
|
352
|
+
return {
|
|
353
|
+
hit,
|
|
354
|
+
genome,
|
|
355
|
+
name: info ? (info.commonName ?? info.sciname) : genome.name,
|
|
356
|
+
scientificName: info?.sciname || genome.longName,
|
|
357
|
+
commonName: info?.commonName,
|
|
358
|
+
}
|
|
359
|
+
})
|
|
360
|
+
const labels = dedupeLabels(described.map(d => d.name))
|
|
361
|
+
const rows = described
|
|
362
|
+
.map(({ hit, genome, scientificName, commonName }, i) => ({
|
|
363
|
+
taxId: genome.taxId,
|
|
364
|
+
label: labels[i]!,
|
|
365
|
+
scientificName,
|
|
366
|
+
commonName,
|
|
367
|
+
geneId: hit.geneRef,
|
|
368
|
+
protein: hit.accession,
|
|
369
|
+
sequence: sequences.get(hit.accession) ?? '',
|
|
370
|
+
}))
|
|
371
|
+
.filter(r => r.sequence)
|
|
372
|
+
if (rows.length < 2) {
|
|
373
|
+
throw new Error('Could not fetch protein sequences for the orthologs')
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
const querySequence = match.query && sequences.get(match.query.accession)
|
|
377
|
+
return {
|
|
378
|
+
matched: match.matched,
|
|
379
|
+
query:
|
|
380
|
+
match.query && querySequence
|
|
381
|
+
? { ...match.query, sequence: querySequence }
|
|
382
|
+
: undefined,
|
|
383
|
+
rows,
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
/**
|
|
388
|
+
* NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
|
|
389
|
+
* are. A failed lookup only costs the labels, which fall back to PANTHER's own
|
|
390
|
+
* short names, so it is logged rather than thrown.
|
|
391
|
+
*/
|
|
392
|
+
async function taxonomyNames(taxIds: number[]) {
|
|
393
|
+
try {
|
|
394
|
+
return await fetchTaxonomyInfo(taxIds)
|
|
395
|
+
} catch (e) {
|
|
396
|
+
console.warn('[msaview-orthologs] taxonomy name lookup failed:', e)
|
|
397
|
+
return new Map<number, { sciname: string; commonName?: string }>()
|
|
398
|
+
}
|
|
399
|
+
}
|
package/src/version.ts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export const version = '3.
|
|
1
|
+
export const version = '3.3.0'
|