jbrowse-plugin-msaview 3.2.0 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +7 -2
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.d.ts +9 -0
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.js +20 -0
- package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +7 -5
- package/dist/MsaViewPanel/doLaunchOrthologs.js +64 -30
- package/dist/MsaViewPanel/doLaunchOrthologs.test.js +105 -0
- package/dist/MsaViewPanel/model.d.ts +9 -0
- package/dist/MsaViewPanel/model.js +6 -0
- package/dist/jbrowse-plugin-msaview.umd.production.min.js +30 -30
- package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
- package/dist/utils/pantherOrthologs.d.ts +79 -0
- package/dist/utils/pantherOrthologs.js +262 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +1 -1
- package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +19 -3
- package/src/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.tsx +52 -0
- package/src/MsaViewPanel/doLaunchOrthologs.test.ts +117 -0
- package/src/MsaViewPanel/doLaunchOrthologs.ts +99 -37
- package/src/MsaViewPanel/model.ts +10 -0
- package/src/utils/pantherOrthologs.ts +399 -0
- package/src/version.ts +1 -1
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import type { OrthologRow } from './ncbiOrthologs';
|
|
2
|
+
export interface PantherGenome {
|
|
3
|
+
/** PANTHER's organism code, e.g. HUMAN, DROME */
|
|
4
|
+
code: string;
|
|
5
|
+
taxId: number;
|
|
6
|
+
/** short common name, e.g. fruit_fly */
|
|
7
|
+
name: string;
|
|
8
|
+
/** scientific name */
|
|
9
|
+
longName: string;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
|
|
13
|
+
* names organisms by code only in ortholog results.
|
|
14
|
+
*/
|
|
15
|
+
export declare function parseGenomes(json: unknown): PantherGenome[];
|
|
16
|
+
export interface PantherGene {
|
|
17
|
+
code: string;
|
|
18
|
+
/** UniProt accession */
|
|
19
|
+
accession: string;
|
|
20
|
+
/** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
|
|
21
|
+
geneRef: string;
|
|
22
|
+
}
|
|
23
|
+
export interface PantherHit extends PantherGene {
|
|
24
|
+
symbol?: string;
|
|
25
|
+
/**
|
|
26
|
+
* LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
|
|
27
|
+
* other ortholog in a one-to-many or many-to-many family
|
|
28
|
+
*/
|
|
29
|
+
type: 'LDO' | 'O';
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* `matchortho` -> the query gene (PANTHER names it in every row) and one hit
|
|
33
|
+
* per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
|
|
34
|
+
* no ortholog in the target set comes back as a bare `{ id }`.
|
|
35
|
+
*/
|
|
36
|
+
export declare function parseMatches(json: unknown): {
|
|
37
|
+
unmapped: boolean;
|
|
38
|
+
query?: PantherGene;
|
|
39
|
+
hits: PantherHit[];
|
|
40
|
+
};
|
|
41
|
+
/**
|
|
42
|
+
* One hit per organism, in first-seen order: the LDO where PANTHER named one,
|
|
43
|
+
* else the first other ortholog it listed. A many-to-many family (the Hox
|
|
44
|
+
* genes) has no LDO at all, so dropping to "first O" is what keeps those
|
|
45
|
+
* species in the alignment.
|
|
46
|
+
*/
|
|
47
|
+
export declare function pickOnePerGenome(hits: PantherHit[]): PantherHit[];
|
|
48
|
+
/** `uniprotkb/accessions` -> accession -> sequence */
|
|
49
|
+
export declare function parseSequences(json: unknown): Map<string, string>;
|
|
50
|
+
/** The proteome list, fetched once per page and forgotten on failure. */
|
|
51
|
+
export declare function fetchGenomes(): Promise<PantherGenome[]>;
|
|
52
|
+
export interface PantherOrthologs {
|
|
53
|
+
/** the candidate PANTHER recognised */
|
|
54
|
+
matched: string;
|
|
55
|
+
/** the query gene as PANTHER knows it, with its UniProt sequence */
|
|
56
|
+
query?: PantherGene & {
|
|
57
|
+
sequence: string;
|
|
58
|
+
};
|
|
59
|
+
rows: OrthologRow[];
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
|
|
63
|
+
* labels, accessions and sequences, plus the query gene's own protein for the
|
|
64
|
+
* query row. Two lookups (genomes, orthologs), one taxonomy batch for the
|
|
65
|
+
* labels, one UniProt batch for the sequences.
|
|
66
|
+
*
|
|
67
|
+
* `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
|
|
68
|
+
* `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
|
|
69
|
+
* `exclude` drops the query taxon, `limit` caps the rows before their
|
|
70
|
+
* sequences are fetched.
|
|
71
|
+
*/
|
|
72
|
+
export declare function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit, onProgress, }: {
|
|
73
|
+
candidates: string[];
|
|
74
|
+
taxId: number;
|
|
75
|
+
taxa?: Set<number>;
|
|
76
|
+
exclude?: number;
|
|
77
|
+
limit?: number;
|
|
78
|
+
onProgress: (arg: string) => void;
|
|
79
|
+
}): Promise<PantherOrthologs>;
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
// The second ortholog source, for the genes NCBI's ortholog sets leave out.
|
|
2
|
+
//
|
|
3
|
+
// NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
|
|
4
|
+
// worm or plant gene comes back with orthologs only in its own clade, and a fly
|
|
5
|
+
// gene gets insects and nothing else. PANTHER's ortholog sets span its 144
|
|
6
|
+
// reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
|
|
7
|
+
// answers "this gene's ortholog in every genome" with a UniProt accession per
|
|
8
|
+
// target. Sequences come from one UniProt batch call. Both hosts send
|
|
9
|
+
// `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
|
|
10
|
+
// OMA, OrthoDB and Ensembl are in react-msaview's
|
|
11
|
+
// agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
|
|
12
|
+
//
|
|
13
|
+
// Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
|
|
14
|
+
// labels, the aligner and the CDD overlay do not know which source ran.
|
|
15
|
+
// `protein` is the UniProt accession; efetch serves UniProt accessions as
|
|
16
|
+
// GenPept records with CDD Region features, so the overlay attaches as it does
|
|
17
|
+
// to a RefSeq accession.
|
|
18
|
+
import { jsonfetch } from './fetch';
|
|
19
|
+
import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs';
|
|
20
|
+
import { fetchTaxonomyInfo } from './taxonomyNames';
|
|
21
|
+
const PANTHER = 'https://pantherdb.org/services/oai/pantherdb';
|
|
22
|
+
const UNIPROT = 'https://rest.uniprot.org/uniprotkb';
|
|
23
|
+
/**
|
|
24
|
+
* `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
|
|
25
|
+
* names organisms by code only in ortholog results.
|
|
26
|
+
*/
|
|
27
|
+
export function parseGenomes(json) {
|
|
28
|
+
const list = json.search?.output?.genomes?.genome ?? [];
|
|
29
|
+
return list.flatMap(g => g.short_name && g.taxon_id
|
|
30
|
+
? [
|
|
31
|
+
{
|
|
32
|
+
code: g.short_name,
|
|
33
|
+
taxId: g.taxon_id,
|
|
34
|
+
name: g.name ?? g.short_name,
|
|
35
|
+
longName: g.long_name ?? g.short_name,
|
|
36
|
+
},
|
|
37
|
+
]
|
|
38
|
+
: []);
|
|
39
|
+
}
|
|
40
|
+
// "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
|
|
41
|
+
function parseGeneRef(ref) {
|
|
42
|
+
const [code, ...xrefs] = (ref ?? '').split('|');
|
|
43
|
+
const accession = xrefs
|
|
44
|
+
.find(x => x.startsWith('UniProtKB='))
|
|
45
|
+
?.slice('UniProtKB='.length);
|
|
46
|
+
const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession;
|
|
47
|
+
return code && accession && geneRef ? { code, accession, geneRef } : undefined;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* `matchortho` -> the query gene (PANTHER names it in every row) and one hit
|
|
51
|
+
* per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
|
|
52
|
+
* no ortholog in the target set comes back as a bare `{ id }`.
|
|
53
|
+
*/
|
|
54
|
+
export function parseMatches(json) {
|
|
55
|
+
const mapping = json.search?.mapping;
|
|
56
|
+
const mapped = mapping?.mapped;
|
|
57
|
+
const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : [];
|
|
58
|
+
const hits = [];
|
|
59
|
+
let query;
|
|
60
|
+
for (const row of rows) {
|
|
61
|
+
query ??= parseGeneRef(row.gene);
|
|
62
|
+
const target = parseGeneRef(row.target_gene);
|
|
63
|
+
if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
|
|
64
|
+
hits.push({
|
|
65
|
+
...target,
|
|
66
|
+
symbol: row.target_gene_symbol === undefined
|
|
67
|
+
? undefined
|
|
68
|
+
: String(row.target_gene_symbol),
|
|
69
|
+
type: row.ortholog,
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
return { unmapped: !!mapping?.unmapped_ids, query, hits };
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* One hit per organism, in first-seen order: the LDO where PANTHER named one,
|
|
77
|
+
* else the first other ortholog it listed. A many-to-many family (the Hox
|
|
78
|
+
* genes) has no LDO at all, so dropping to "first O" is what keeps those
|
|
79
|
+
* species in the alignment.
|
|
80
|
+
*/
|
|
81
|
+
export function pickOnePerGenome(hits) {
|
|
82
|
+
const byCode = new Map();
|
|
83
|
+
for (const hit of hits) {
|
|
84
|
+
const current = byCode.get(hit.code);
|
|
85
|
+
if (!current || (current.type === 'O' && hit.type === 'LDO')) {
|
|
86
|
+
byCode.set(hit.code, hit);
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return [...byCode.values()];
|
|
90
|
+
}
|
|
91
|
+
/** `uniprotkb/accessions` -> accession -> sequence */
|
|
92
|
+
export function parseSequences(json) {
|
|
93
|
+
const map = new Map();
|
|
94
|
+
for (const r of json.results ?? []) {
|
|
95
|
+
if (r.primaryAccession && r.sequence?.value) {
|
|
96
|
+
map.set(r.primaryAccession, r.sequence.value);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return map;
|
|
100
|
+
}
|
|
101
|
+
let genomes;
|
|
102
|
+
/** The proteome list, fetched once per page and forgotten on failure. */
|
|
103
|
+
export function fetchGenomes() {
|
|
104
|
+
genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
|
|
105
|
+
.then(parseGenomes)
|
|
106
|
+
.catch((e) => {
|
|
107
|
+
genomes = undefined;
|
|
108
|
+
throw e;
|
|
109
|
+
});
|
|
110
|
+
return genomes;
|
|
111
|
+
}
|
|
112
|
+
// UniProt caps one `accessions` call at 100 ids
|
|
113
|
+
const UNIPROT_CHUNK = 100;
|
|
114
|
+
async function fetchSequences(accessions) {
|
|
115
|
+
const map = new Map();
|
|
116
|
+
for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
|
|
117
|
+
const chunk = accessions.slice(i, i + UNIPROT_CHUNK);
|
|
118
|
+
const json = await jsonfetch(`${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`);
|
|
119
|
+
for (const [acc, seq] of parseSequences(json)) {
|
|
120
|
+
map.set(acc, seq);
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
return map;
|
|
124
|
+
}
|
|
125
|
+
// strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
|
|
126
|
+
// resolveGeneId does for NCBI
|
|
127
|
+
function cleanCandidate(raw) {
|
|
128
|
+
return raw
|
|
129
|
+
.trim()
|
|
130
|
+
.replace(/^\w+:/, '')
|
|
131
|
+
.replace(/\.\d+$/, '');
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
|
|
135
|
+
* carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
|
|
136
|
+
* some of those are names PANTHER knows. `targets` omitted asks for every
|
|
137
|
+
* genome PANTHER has, which is one call rather than one per genome.
|
|
138
|
+
*/
|
|
139
|
+
async function matchOrthologs(candidates, taxId, targets) {
|
|
140
|
+
let matched;
|
|
141
|
+
for (const raw of candidates) {
|
|
142
|
+
const query = cleanCandidate(raw);
|
|
143
|
+
if (!query) {
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
const params = new URLSearchParams({
|
|
147
|
+
geneInputList: query,
|
|
148
|
+
organism: String(taxId),
|
|
149
|
+
orthologType: 'all',
|
|
150
|
+
});
|
|
151
|
+
if (targets) {
|
|
152
|
+
params.set('targetOrganism', targets.map(t => t.taxId).join(','));
|
|
153
|
+
}
|
|
154
|
+
const parsed = parseMatches(await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`));
|
|
155
|
+
if (!parsed.unmapped) {
|
|
156
|
+
matched ??= query;
|
|
157
|
+
if (parsed.query) {
|
|
158
|
+
return { ...parsed, matched: query };
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
return matched === undefined
|
|
163
|
+
? undefined
|
|
164
|
+
: { matched, hits: [], query: undefined };
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
|
|
168
|
+
* labels, accessions and sequences, plus the query gene's own protein for the
|
|
169
|
+
* query row. Two lookups (genomes, orthologs), one taxonomy batch for the
|
|
170
|
+
* labels, one UniProt batch for the sequences.
|
|
171
|
+
*
|
|
172
|
+
* `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
|
|
173
|
+
* `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
|
|
174
|
+
* `exclude` drops the query taxon, `limit` caps the rows before their
|
|
175
|
+
* sequences are fetched.
|
|
176
|
+
*/
|
|
177
|
+
export async function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit = defaultMaxSpecies, onProgress, }) {
|
|
178
|
+
const all = await fetchGenomes();
|
|
179
|
+
const byTaxId = new Map(all.map(g => [g.taxId, g]));
|
|
180
|
+
const byCode = new Map(all.map(g => [g.code, g]));
|
|
181
|
+
const queryGenome = byTaxId.get(taxId);
|
|
182
|
+
if (!queryGenome) {
|
|
183
|
+
throw new Error(`PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`);
|
|
184
|
+
}
|
|
185
|
+
const targets = taxa
|
|
186
|
+
? [...taxa].flatMap(t => {
|
|
187
|
+
const g = byTaxId.get(t);
|
|
188
|
+
return g && t !== exclude ? [g] : [];
|
|
189
|
+
})
|
|
190
|
+
: undefined;
|
|
191
|
+
onProgress('Matching orthologs at PANTHER...');
|
|
192
|
+
const match = await matchOrthologs(candidates, taxId, targets);
|
|
193
|
+
if (!match) {
|
|
194
|
+
throw new Error(`PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`);
|
|
195
|
+
}
|
|
196
|
+
const rank = new Map(targets?.map((t, i) => [t.taxId, i]));
|
|
197
|
+
const picks = pickOnePerGenome(match.hits)
|
|
198
|
+
.map(hit => ({ hit, genome: byCode.get(hit.code) }))
|
|
199
|
+
.filter((p) => !!p.genome &&
|
|
200
|
+
p.genome.taxId !== exclude &&
|
|
201
|
+
(targets ? rank.has(p.genome.taxId) : true))
|
|
202
|
+
.sort((a, b) => targets ? rank.get(a.genome.taxId) - rank.get(b.genome.taxId) : 0)
|
|
203
|
+
.slice(0, limit);
|
|
204
|
+
if (picks.length < 2) {
|
|
205
|
+
throw new Error(`Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`);
|
|
206
|
+
}
|
|
207
|
+
onProgress(`Fetching ${picks.length} protein sequences from UniProt...`);
|
|
208
|
+
const [names, sequences] = await Promise.all([
|
|
209
|
+
taxonomyNames(picks.map(p => p.genome.taxId)),
|
|
210
|
+
fetchSequences([
|
|
211
|
+
...(match.query ? [match.query.accession] : []),
|
|
212
|
+
...picks.map(p => p.hit.accession),
|
|
213
|
+
]),
|
|
214
|
+
]);
|
|
215
|
+
const described = picks.map(({ hit, genome }) => {
|
|
216
|
+
const info = names.get(genome.taxId);
|
|
217
|
+
return {
|
|
218
|
+
hit,
|
|
219
|
+
genome,
|
|
220
|
+
name: info ? (info.commonName ?? info.sciname) : genome.name,
|
|
221
|
+
scientificName: info?.sciname || genome.longName,
|
|
222
|
+
commonName: info?.commonName,
|
|
223
|
+
};
|
|
224
|
+
});
|
|
225
|
+
const labels = dedupeLabels(described.map(d => d.name));
|
|
226
|
+
const rows = described
|
|
227
|
+
.map(({ hit, genome, scientificName, commonName }, i) => ({
|
|
228
|
+
taxId: genome.taxId,
|
|
229
|
+
label: labels[i],
|
|
230
|
+
scientificName,
|
|
231
|
+
commonName,
|
|
232
|
+
geneId: hit.geneRef,
|
|
233
|
+
protein: hit.accession,
|
|
234
|
+
sequence: sequences.get(hit.accession) ?? '',
|
|
235
|
+
}))
|
|
236
|
+
.filter(r => r.sequence);
|
|
237
|
+
if (rows.length < 2) {
|
|
238
|
+
throw new Error('Could not fetch protein sequences for the orthologs');
|
|
239
|
+
}
|
|
240
|
+
const querySequence = match.query && sequences.get(match.query.accession);
|
|
241
|
+
return {
|
|
242
|
+
matched: match.matched,
|
|
243
|
+
query: match.query && querySequence
|
|
244
|
+
? { ...match.query, sequence: querySequence }
|
|
245
|
+
: undefined,
|
|
246
|
+
rows,
|
|
247
|
+
};
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
|
|
251
|
+
* are. A failed lookup only costs the labels, which fall back to PANTHER's own
|
|
252
|
+
* short names, so it is logged rather than thrown.
|
|
253
|
+
*/
|
|
254
|
+
async function taxonomyNames(taxIds) {
|
|
255
|
+
try {
|
|
256
|
+
return await fetchTaxonomyInfo(taxIds);
|
|
257
|
+
}
|
|
258
|
+
catch (e) {
|
|
259
|
+
console.warn('[msaview-orthologs] taxonomy name lookup failed:', e);
|
|
260
|
+
return new Map();
|
|
261
|
+
}
|
|
262
|
+
}
|
package/dist/version.d.ts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export declare const version = "3.
|
|
1
|
+
export declare const version = "3.3.0";
|
package/dist/version.js
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export const version = '3.
|
|
1
|
+
export const version = '3.3.0';
|
package/package.json
CHANGED
|
@@ -4,10 +4,14 @@ import { Typography } from '@mui/material'
|
|
|
4
4
|
import { observer } from 'mobx-react'
|
|
5
5
|
import { makeStyles } from 'tss-react/mui'
|
|
6
6
|
|
|
7
|
+
import OrthologSourceSelect, {
|
|
8
|
+
ORTHOLOG_SOURCE_STORAGE_KEY,
|
|
9
|
+
} from './OrthologSourceSelect'
|
|
7
10
|
import QuerySpeciesSelect from './QuerySpeciesSelect'
|
|
8
11
|
import { orthologLaunchView } from './orthologLaunchView'
|
|
9
12
|
import TextField2 from '../../../components/TextField2'
|
|
10
13
|
import { defaultMaxSpecies } from '../../../utils/ncbiOrthologs'
|
|
14
|
+
import { useLocalStorage } from '../../../utils/useLocalStorage'
|
|
11
15
|
import {
|
|
12
16
|
getGeneDisplayName,
|
|
13
17
|
getGeneIdentifiers,
|
|
@@ -20,6 +24,7 @@ import SubmitCancelActions from '../SubmitCancelActions'
|
|
|
20
24
|
import TranscriptSelector from '../TranscriptSelector'
|
|
21
25
|
import { useTranscriptSelection } from '../useTranscriptSelection'
|
|
22
26
|
|
|
27
|
+
import type { OrthologSource } from '../../../MsaViewPanel/model'
|
|
23
28
|
import type { MsaAlgorithm } from '../BlastQuery/consts'
|
|
24
29
|
import type { AbstractTrackModel, Feature } from '@jbrowse/core/util'
|
|
25
30
|
|
|
@@ -42,6 +47,10 @@ const OrthologPanel = observer(function ({
|
|
|
42
47
|
const view = getLinearGenomeView(model)
|
|
43
48
|
const [launchViewError, setLaunchViewError] = useState<unknown>()
|
|
44
49
|
const [taxId, setTaxId] = useState(9606)
|
|
50
|
+
const [source, setSource] = useLocalStorage<OrthologSource>(
|
|
51
|
+
ORTHOLOG_SOURCE_STORAGE_KEY,
|
|
52
|
+
'ncbi',
|
|
53
|
+
)
|
|
45
54
|
const [msaAlgorithm, setMsaAlgorithm] = useState<MsaAlgorithm>('clustalo')
|
|
46
55
|
const [maxSpecies, setMaxSpecies] = useState(String(defaultMaxSpecies))
|
|
47
56
|
|
|
@@ -65,11 +74,17 @@ const OrthologPanel = observer(function ({
|
|
|
65
74
|
instant, and it is now the aligner that costs the wait, about half a
|
|
66
75
|
second per row. */}
|
|
67
76
|
<Typography variant="body2">
|
|
68
|
-
|
|
69
|
-
|
|
77
|
+
Precomputed orthologs, one gene per species, looked up rather than
|
|
78
|
+
searched for. No BLAST job to queue.
|
|
70
79
|
</Typography>
|
|
71
80
|
|
|
72
81
|
<div>
|
|
82
|
+
<OrthologSourceSelect
|
|
83
|
+
className={classes.selectField}
|
|
84
|
+
value={source}
|
|
85
|
+
onChange={setSource}
|
|
86
|
+
/>
|
|
87
|
+
|
|
73
88
|
<QuerySpeciesSelect
|
|
74
89
|
className={classes.selectField}
|
|
75
90
|
value={taxId}
|
|
@@ -93,7 +108,7 @@ const OrthologPanel = observer(function ({
|
|
|
93
108
|
setMaxSpecies(event.target.value)
|
|
94
109
|
}}
|
|
95
110
|
error={!rowCountValid}
|
|
96
|
-
helperText=
|
|
111
|
+
helperText={`the closest N species ${source === 'panther' ? 'PANTHER' : 'NCBI'} has`}
|
|
97
112
|
/>
|
|
98
113
|
</div>
|
|
99
114
|
|
|
@@ -112,6 +127,7 @@ const OrthologPanel = observer(function ({
|
|
|
112
127
|
newViewTitle: `Orthologs - ${getGeneDisplayName(feature)} - ${getTranscriptDisplayName(selectedTranscript)}`,
|
|
113
128
|
orthologParams: {
|
|
114
129
|
taxId,
|
|
130
|
+
source,
|
|
115
131
|
maxSpecies: rowCount,
|
|
116
132
|
geneCandidates,
|
|
117
133
|
msaAlgorithm,
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import React from 'react'
|
|
2
|
+
|
|
3
|
+
import { MenuItem } from '@mui/material'
|
|
4
|
+
|
|
5
|
+
import TextField2 from '../../../components/TextField2'
|
|
6
|
+
|
|
7
|
+
import type { OrthologSource } from '../../../MsaViewPanel/model'
|
|
8
|
+
|
|
9
|
+
export const ORTHOLOG_SOURCE_STORAGE_KEY = 'msaview-ortholog-source'
|
|
10
|
+
|
|
11
|
+
export const orthologSourceLabels: Record<OrthologSource, string> = {
|
|
12
|
+
ncbi: 'NCBI orthologs',
|
|
13
|
+
panther: 'PANTHER',
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
// Which species a source can answer for, in the words a reader picking one
|
|
17
|
+
// needs: NCBI's ortholog sets stop at vertebrates and insects, PANTHER's run
|
|
18
|
+
// from human to yeast and Arabidopsis.
|
|
19
|
+
const hints: Record<OrthologSource, string> = {
|
|
20
|
+
ncbi: 'vertebrates and insects',
|
|
21
|
+
panther: 'also yeast, worm, fly and plants',
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export default function OrthologSourceSelect({
|
|
25
|
+
value,
|
|
26
|
+
onChange,
|
|
27
|
+
className,
|
|
28
|
+
}: {
|
|
29
|
+
value: OrthologSource
|
|
30
|
+
onChange: (val: OrthologSource) => void
|
|
31
|
+
className?: string
|
|
32
|
+
}) {
|
|
33
|
+
return (
|
|
34
|
+
<TextField2
|
|
35
|
+
variant="outlined"
|
|
36
|
+
label="Source"
|
|
37
|
+
className={className}
|
|
38
|
+
select
|
|
39
|
+
value={value}
|
|
40
|
+
helperText={hints[value]}
|
|
41
|
+
onChange={event => {
|
|
42
|
+
onChange(event.target.value as OrthologSource)
|
|
43
|
+
}}
|
|
44
|
+
>
|
|
45
|
+
{(Object.keys(orthologSourceLabels) as OrthologSource[]).map(val => (
|
|
46
|
+
<MenuItem value={val} key={val}>
|
|
47
|
+
{orthologSourceLabels[val]}
|
|
48
|
+
</MenuItem>
|
|
49
|
+
))}
|
|
50
|
+
</TextField2>
|
|
51
|
+
)
|
|
52
|
+
}
|
|
@@ -8,6 +8,7 @@ import {
|
|
|
8
8
|
fetchProteinForGene,
|
|
9
9
|
resolveGeneId,
|
|
10
10
|
} from '../utils/ncbiOrthologs'
|
|
11
|
+
import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
|
|
11
12
|
import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
|
|
12
13
|
|
|
13
14
|
import type { JBrowsePluginMsaViewModel } from './model'
|
|
@@ -24,12 +25,16 @@ vi.mock('../utils/ncbiOrthologs', async importOriginal => ({
|
|
|
24
25
|
fetchProteinForGene: vi.fn(),
|
|
25
26
|
fetchOrthologRows: vi.fn(),
|
|
26
27
|
}))
|
|
28
|
+
vi.mock('../utils/pantherOrthologs', () => ({
|
|
29
|
+
fetchPantherOrthologs: vi.fn(),
|
|
30
|
+
}))
|
|
27
31
|
vi.mock('../utils/msa', () => ({ launchMSA: vi.fn() }))
|
|
28
32
|
vi.mock('../utils/taxonomyNames', () => ({ fetchTaxonomyInfo: vi.fn() }))
|
|
29
33
|
|
|
30
34
|
const mockResolveGeneId = vi.mocked(resolveGeneId)
|
|
31
35
|
const mockFetchProtein = vi.mocked(fetchProteinForGene)
|
|
32
36
|
const mockFetchRows = vi.mocked(fetchOrthologRows)
|
|
37
|
+
const mockFetchPanther = vi.mocked(fetchPantherOrthologs)
|
|
33
38
|
const mockLaunchMSA = vi.mocked(launchMSA)
|
|
34
39
|
const mockFetchTaxonomy = vi.mocked(fetchTaxonomyInfo)
|
|
35
40
|
|
|
@@ -245,3 +250,115 @@ describe('the Accession that drives the domain overlay', () => {
|
|
|
245
250
|
expect(queryMetadata(result).Accession).toBeUndefined()
|
|
246
251
|
})
|
|
247
252
|
})
|
|
253
|
+
|
|
254
|
+
// The second source. What is under test is the dispatch and what the PANTHER
|
|
255
|
+
// result becomes on the query row -- the rows themselves are shaped upstream,
|
|
256
|
+
// and the tail of the launch (labels, aligner, metadata) is the same code the
|
|
257
|
+
// NCBI tests above already cover.
|
|
258
|
+
describe('the PANTHER source', () => {
|
|
259
|
+
const YEAST = 559292
|
|
260
|
+
const found = {
|
|
261
|
+
matched: 'CDC28',
|
|
262
|
+
query: {
|
|
263
|
+
code: 'YEAST',
|
|
264
|
+
accession: 'P00546',
|
|
265
|
+
geneRef: 'SGD=S000000364',
|
|
266
|
+
sequence: 'MSGELANYKRLEKVGEGTYGVVYKA',
|
|
267
|
+
},
|
|
268
|
+
rows: [
|
|
269
|
+
{
|
|
270
|
+
taxId: HUMAN,
|
|
271
|
+
label: 'human',
|
|
272
|
+
scientificName: 'Homo sapiens',
|
|
273
|
+
commonName: 'human',
|
|
274
|
+
geneId: 'HGNC=1771',
|
|
275
|
+
protein: 'P24941',
|
|
276
|
+
sequence: 'MENFQKVEKIGEGTYGVVYKARNK',
|
|
277
|
+
},
|
|
278
|
+
] as OrthologRow[],
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
beforeEach(() => {
|
|
282
|
+
mockFetchPanther.mockResolvedValue(found)
|
|
283
|
+
mockFetchTaxonomy.mockResolvedValue(
|
|
284
|
+
new Map([[YEAST, { sciname: 'Saccharomyces cerevisiae' }]]),
|
|
285
|
+
)
|
|
286
|
+
})
|
|
287
|
+
|
|
288
|
+
test('source omitted is NCBI, so an old launch never reaches PANTHER', async () => {
|
|
289
|
+
await doLaunchOrthologs({ self: makeModel(params()) })
|
|
290
|
+
expect(mockFetchPanther).not.toHaveBeenCalled()
|
|
291
|
+
expect(mockResolveGeneId).toHaveBeenCalled()
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
test('source panther asks PANTHER with the same species semantics, and skips NCBI', async () => {
|
|
295
|
+
await doLaunchOrthologs({
|
|
296
|
+
self: makeModel({
|
|
297
|
+
taxId: YEAST,
|
|
298
|
+
source: 'panther',
|
|
299
|
+
geneCandidates: ['CDC28'],
|
|
300
|
+
msaAlgorithm: 'clustalo',
|
|
301
|
+
taxa: [HUMAN, YEAST],
|
|
302
|
+
maxSpecies: 7,
|
|
303
|
+
}),
|
|
304
|
+
})
|
|
305
|
+
expect(mockResolveGeneId).not.toHaveBeenCalled()
|
|
306
|
+
expect(mockFetchRows).not.toHaveBeenCalled()
|
|
307
|
+
const { candidates, taxId, taxa, exclude, limit } =
|
|
308
|
+
mockFetchPanther.mock.calls[0]![0]
|
|
309
|
+
expect(candidates).toEqual(['CDC28'])
|
|
310
|
+
expect(taxId).toBe(YEAST)
|
|
311
|
+
expect([...taxa!]).toEqual([HUMAN, YEAST])
|
|
312
|
+
expect(exclude).toBe(YEAST)
|
|
313
|
+
expect(limit).toBe(7)
|
|
314
|
+
})
|
|
315
|
+
|
|
316
|
+
test("the query row is PANTHER's own entry for the gene when no sequence was supplied, and carries its UniProt accession for the domain overlay", async () => {
|
|
317
|
+
const result = await doLaunchOrthologs({
|
|
318
|
+
self: makeModel({
|
|
319
|
+
taxId: YEAST,
|
|
320
|
+
source: 'panther',
|
|
321
|
+
geneCandidates: ['CDC28'],
|
|
322
|
+
msaAlgorithm: 'clustalo',
|
|
323
|
+
}),
|
|
324
|
+
})
|
|
325
|
+
expect(queryRowName()).toBe('Saccharomyces_cerevisiae_query')
|
|
326
|
+
expect(queryRowSent()).toBe(found.query.sequence)
|
|
327
|
+
expect(queryMetadata(result)).toEqual({
|
|
328
|
+
'Gene ID': 'SGD=S000000364',
|
|
329
|
+
Accession: 'P00546',
|
|
330
|
+
})
|
|
331
|
+
expect(JSON.parse(result.treeMetadata).human).toMatchObject({
|
|
332
|
+
Accession: 'P24941',
|
|
333
|
+
'Gene ID': 'HGNC=1771',
|
|
334
|
+
})
|
|
335
|
+
})
|
|
336
|
+
|
|
337
|
+
test('a supplied sequence still wins, and a different isoform earns no Accession', async () => {
|
|
338
|
+
const result = await doLaunchOrthologs({
|
|
339
|
+
self: makeModel({
|
|
340
|
+
taxId: YEAST,
|
|
341
|
+
source: 'panther',
|
|
342
|
+
geneCandidates: ['CDC28'],
|
|
343
|
+
msaAlgorithm: 'clustalo',
|
|
344
|
+
proteinSequence: 'MDIFFERENTISOFORM',
|
|
345
|
+
}),
|
|
346
|
+
})
|
|
347
|
+
expect(queryRowSent()).toBe('MDIFFERENTISOFORM')
|
|
348
|
+
expect(queryMetadata(result).Accession).toBeUndefined()
|
|
349
|
+
})
|
|
350
|
+
|
|
351
|
+
test('names PANTHER when it has no protein for the query row', async () => {
|
|
352
|
+
mockFetchPanther.mockResolvedValue({ ...found, query: undefined })
|
|
353
|
+
await expect(
|
|
354
|
+
doLaunchOrthologs({
|
|
355
|
+
self: makeModel({
|
|
356
|
+
taxId: YEAST,
|
|
357
|
+
source: 'panther',
|
|
358
|
+
geneCandidates: ['CDC28'],
|
|
359
|
+
msaAlgorithm: 'clustalo',
|
|
360
|
+
}),
|
|
361
|
+
}),
|
|
362
|
+
).rejects.toThrow(/PANTHER returned no representative protein/)
|
|
363
|
+
})
|
|
364
|
+
})
|