jbrowse-plugin-msaview 2.7.3 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/dist/LaunchMsaView/components/LaunchMsaViewDialog.js +7 -1
  2. package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.d.ts +8 -0
  3. package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +89 -0
  4. package/dist/LaunchMsaView/components/OrthologQuery/orthologLaunchView.d.ts +9 -0
  5. package/dist/LaunchMsaView/components/OrthologQuery/orthologLaunchView.js +13 -0
  6. package/dist/LaunchMsaView/components/TranscriptSelector.js +7 -1
  7. package/dist/MsaViewPanel/afterCreateAutoruns.d.ts +8 -0
  8. package/dist/MsaViewPanel/afterCreateAutoruns.js +28 -0
  9. package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +23 -0
  10. package/dist/MsaViewPanel/doLaunchOrthologs.js +97 -0
  11. package/dist/MsaViewPanel/model.d.ts +21 -5
  12. package/dist/MsaViewPanel/model.js +12 -1
  13. package/dist/jbrowse-plugin-msaview.umd.production.min.js +31 -27
  14. package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
  15. package/dist/utils/ncbiOrthologs.d.ts +135 -0
  16. package/dist/utils/ncbiOrthologs.js +241 -0
  17. package/dist/utils/ncbiOrthologs.test.d.ts +1 -0
  18. package/dist/utils/ncbiOrthologs.test.js +41 -0
  19. package/dist/version.d.ts +1 -1
  20. package/dist/version.js +1 -1
  21. package/package.json +3 -3
  22. package/src/LaunchMsaView/components/LaunchMsaViewDialog.tsx +13 -2
  23. package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +172 -0
  24. package/src/LaunchMsaView/components/OrthologQuery/orthologLaunchView.ts +28 -0
  25. package/src/LaunchMsaView/components/TranscriptSelector.tsx +6 -0
  26. package/src/MsaViewPanel/afterCreateAutoruns.ts +27 -0
  27. package/src/MsaViewPanel/doLaunchOrthologs.ts +123 -0
  28. package/src/MsaViewPanel/model.ts +24 -0
  29. package/src/utils/ncbiOrthologs.test.ts +56 -0
  30. package/src/utils/ncbiOrthologs.ts +350 -0
  31. package/src/version.ts +1 -1
@@ -0,0 +1,135 @@
1
+ export declare const COMMON_SPECIES: readonly [{
2
+ readonly label: "Human";
3
+ readonly taxId: 9606;
4
+ }, {
5
+ readonly label: "Chimpanzee";
6
+ readonly taxId: 9598;
7
+ }, {
8
+ readonly label: "Gorilla";
9
+ readonly taxId: 9595;
10
+ }, {
11
+ readonly label: "Rhesus macaque";
12
+ readonly taxId: 9544;
13
+ }, {
14
+ readonly label: "Marmoset";
15
+ readonly taxId: 9483;
16
+ }, {
17
+ readonly label: "Mouse";
18
+ readonly taxId: 10090;
19
+ }, {
20
+ readonly label: "Rat";
21
+ readonly taxId: 10116;
22
+ }, {
23
+ readonly label: "Guinea pig";
24
+ readonly taxId: 10141;
25
+ }, {
26
+ readonly label: "Rabbit";
27
+ readonly taxId: 9986;
28
+ }, {
29
+ readonly label: "Cat";
30
+ readonly taxId: 9685;
31
+ }, {
32
+ readonly label: "Dog";
33
+ readonly taxId: 9615;
34
+ }, {
35
+ readonly label: "Horse";
36
+ readonly taxId: 9796;
37
+ }, {
38
+ readonly label: "Pig";
39
+ readonly taxId: 9823;
40
+ }, {
41
+ readonly label: "Cow";
42
+ readonly taxId: 9913;
43
+ }, {
44
+ readonly label: "Sheep";
45
+ readonly taxId: 9940;
46
+ }, {
47
+ readonly label: "Opossum";
48
+ readonly taxId: 13616;
49
+ }, {
50
+ readonly label: "Chicken";
51
+ readonly taxId: 9031;
52
+ }, {
53
+ readonly label: "Frog";
54
+ readonly taxId: 8364;
55
+ }, {
56
+ readonly label: "Zebrafish";
57
+ readonly taxId: 7955;
58
+ }, {
59
+ readonly label: "Fruitfly";
60
+ readonly taxId: 7227;
61
+ }, {
62
+ readonly label: "C. elegans";
63
+ readonly taxId: 6239;
64
+ }, {
65
+ readonly label: "Yeast";
66
+ readonly taxId: 4932;
67
+ }, {
68
+ readonly label: "Arabidopsis";
69
+ readonly taxId: 3702;
70
+ }];
71
+ export declare const COMMON_TAX_RANK: Map<number, number>;
72
+ export interface OrthologRow {
73
+ taxId: number;
74
+ /** single-token id used identically in the FASTA, the tree and the domain GFF */
75
+ label: string;
76
+ scientificName: string;
77
+ commonName?: string;
78
+ geneId: string;
79
+ /** accession.version */
80
+ protein: string;
81
+ sequence: string;
82
+ }
83
+ /**
84
+ * A free-text gene reference -> NCBI gene id. A bare number is taken as the id
85
+ * itself; anything else is searched as a gene name within the query taxon.
86
+ * Several candidate identifiers are tried in order, because a JBrowse feature
87
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
88
+ * some of those are real symbols.
89
+ */
90
+ export declare function resolveGeneId(candidates: string[], taxId: number): Promise<{
91
+ geneId: string;
92
+ matched: string;
93
+ } | undefined>;
94
+ /** One ortholog gene per species, restricted to the requested taxa. */
95
+ export declare function fetchOrthologGenes(geneId: string, taxa: Set<number>): Promise<{
96
+ taxId: number;
97
+ geneId: string;
98
+ scientificName: string;
99
+ commonName?: string;
100
+ }[]>;
101
+ /**
102
+ * geneId -> representative protein accession: MANE Select where flagged, else
103
+ * the longest isoform. A stable, comparable choice across species — picking
104
+ * "the first" would silently vary with NCBI's ordering.
105
+ */
106
+ export declare function fetchRepresentativeProteins(geneIds: string[]): Promise<Map<string, string>>;
107
+ /** accession (first header token) -> ungapped sequence, from a multi-FASTA. */
108
+ export declare function parseFasta(text: string): Map<string, string>;
109
+ /**
110
+ * Sanitized, unique single-token labels used identically in the FASTA headers,
111
+ * the tree leaf names and the domain GFF seq_ids — that identity is how the
112
+ * viewer pairs a tree leaf to its alignment row to its domain track. Collisions
113
+ * get a numeric suffix rather than silently overwriting a row.
114
+ */
115
+ export declare function dedupeLabels(names: string[]): string[];
116
+ /**
117
+ * The representative protein for a single gene, with its sequence. Used to
118
+ * decide whether the user's own translated transcript is byte-identical to the
119
+ * RefSeq protein — if it is, that accession's precomputed CDD domains apply to
120
+ * the query row exactly, and if it isn't, they would land at an offset.
121
+ */
122
+ export declare function fetchProteinForGene(geneId: string): Promise<{
123
+ accession: string;
124
+ sequence: string;
125
+ } | undefined>;
126
+ /**
127
+ * The whole NCBI half of the pipeline: gene -> ortholog rows carrying labels,
128
+ * accessions and sequences. Everything here is a precomputed lookup, so this
129
+ * returns in seconds rather than the 10+ minutes a BLAST submission costs.
130
+ */
131
+ export declare function fetchOrthologRows({ geneId, taxa, onProgress, }: {
132
+ geneId: string;
133
+ taxa: Set<number>;
134
+ onProgress: (arg: string) => void;
135
+ }): Promise<OrthologRow[]>;
@@ -0,0 +1,241 @@
1
+ // Homolog discovery WITHOUT a search job.
2
+ //
3
+ // The BLAST path answers "what looks like this sequence", which is not the
4
+ // question an MSA row set wants — it wants "what is homologous to this gene,
5
+ // one per species, labelled by species". BLAST then costs 10+ minutes to
6
+ // return a redundant, accession-labelled hit list that has to be deduplicated
7
+ // before it reads. NCBI has already computed the answer: the Datasets
8
+ // orthologs endpoint returns one ortholog gene per species, instantly.
9
+ //
10
+ // gene symbol -> gene id -> orthologs -> a representative protein each ->
11
+ // sequences, all from NCBI, in a handful of requests. The caller aligns them
12
+ // (EBI Clustal Omega, ~10s) and overlays CDD domains, which are already baked
13
+ // into the GenPept records (see ncbiDomains.ts).
14
+ //
15
+ // Mirrors jb2hubs' website/src/components/proteinMsa.ts assembler, trimmed to
16
+ // what the launch dialog needs and using this plugin's fetch/eutils helpers.
17
+ import { NCBI_EMAIL, NCBI_TOOL } from './eutils';
18
+ import { jsonfetch, textfetch } from './fetch';
19
+ // v2, not v2alpha: the alpha path still answers /orthologs but 404s
20
+ // /product_report, so an assembler pointed at it silently resolves zero
21
+ // representative proteins and reports "no orthologs" for every gene.
22
+ const DATASETS = 'https://api.ncbi.nlm.nih.gov/datasets/v2';
23
+ const EUTILS = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils';
24
+ // The species panel offered in the launch dialog, ordered from the reference
25
+ // outward so a run that finds only close relatives still reads as a ladder.
26
+ // Orthologs absent for a given gene are skipped rather than erroring, and the
27
+ // index order here is the ROW order of the alignment (COMMON_TAX_RANK below).
28
+ //
29
+ // THE MAMMALS EARN THEIR PLACE, and the reason is measured rather than aesthetic.
30
+ // The thirteen this list used to hold were one per major clade, which reads well
31
+ // on a gene conserved to yeast and produces almost nothing on a gene that is not:
32
+ // NCBI publishes 165 orthologs for human NLRP1 and every one of them is a mammal,
33
+ // so of the old thirteen only Human, Mouse, Cow, Pig and Dog returned a row --
34
+ // five, and Rat not among them, since NLRP1 is absent in Rattus norvegicus. The
35
+ // same query against this list returns twelve. An inflammasome gene is not an
36
+ // unusual case; anything immune, reproductive or lineage-specific behaves the
37
+ // same way, and those are the genes a person opens an ortholog alignment on.
38
+ //
39
+ // Cat, rabbit and opossum are here despite contributing nothing to that gene.
40
+ // They are the three that most often separate "absent in this clade" from
41
+ // "absent in this species", which is the question a gap in the alignment raises.
42
+ //
43
+ // The cost is the run, and it is roughly linear: one NCBI protein fetch per
44
+ // species and a Clustal Omega job over what comes back, so ~23 rows is about
45
+ // twice the ~13-row wait. Still seconds rather than the minutes BLAST takes,
46
+ // which is the comparison the panel's own text makes.
47
+ export const COMMON_SPECIES = [
48
+ { label: 'Human', taxId: 9606 },
49
+ { label: 'Chimpanzee', taxId: 9598 },
50
+ { label: 'Gorilla', taxId: 9595 },
51
+ { label: 'Rhesus macaque', taxId: 9544 },
52
+ { label: 'Marmoset', taxId: 9483 },
53
+ { label: 'Mouse', taxId: 10090 },
54
+ { label: 'Rat', taxId: 10116 },
55
+ { label: 'Guinea pig', taxId: 10141 },
56
+ { label: 'Rabbit', taxId: 9986 },
57
+ { label: 'Cat', taxId: 9685 },
58
+ { label: 'Dog', taxId: 9615 },
59
+ { label: 'Horse', taxId: 9796 },
60
+ { label: 'Pig', taxId: 9823 },
61
+ { label: 'Cow', taxId: 9913 },
62
+ { label: 'Sheep', taxId: 9940 },
63
+ { label: 'Opossum', taxId: 13616 },
64
+ { label: 'Chicken', taxId: 9031 },
65
+ { label: 'Frog', taxId: 8364 },
66
+ { label: 'Zebrafish', taxId: 7955 },
67
+ { label: 'Fruitfly', taxId: 7227 },
68
+ { label: 'C. elegans', taxId: 6239 },
69
+ { label: 'Yeast', taxId: 4932 },
70
+ { label: 'Arabidopsis', taxId: 3702 },
71
+ ];
72
+ export const COMMON_TAX_RANK = new Map(COMMON_SPECIES.map((s, i) => [s.taxId, i]));
73
+ function ncbiUrl(url) {
74
+ const sep = url.includes('?') ? '&' : '?';
75
+ return `${url}${sep}tool=${NCBI_TOOL}&email=${encodeURIComponent(NCBI_EMAIL)}`;
76
+ }
77
+ /**
78
+ * A free-text gene reference -> NCBI gene id. A bare number is taken as the id
79
+ * itself; anything else is searched as a gene name within the query taxon.
80
+ * Several candidate identifiers are tried in order, because a JBrowse feature
81
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
82
+ * some of those are real symbols.
83
+ */
84
+ export async function resolveGeneId(candidates, taxId) {
85
+ for (const raw of candidates) {
86
+ const query = raw.trim();
87
+ if (!query) {
88
+ continue;
89
+ }
90
+ if (/^\d+$/.test(query)) {
91
+ return { geneId: query, matched: query };
92
+ }
93
+ // strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53)
94
+ const cleaned = query.replace(/^\w+:/, '').replace(/\.\d+$/, '');
95
+ const term = `${cleaned}[Gene Name] AND ${taxId}[taxid]`;
96
+ const json = await jsonfetch(ncbiUrl(`${EUTILS}/esearch.fcgi?db=gene&term=${encodeURIComponent(term)}&retmode=json&retmax=1`));
97
+ const geneId = json.esearchresult?.idlist?.[0];
98
+ if (geneId) {
99
+ return { geneId, matched: cleaned };
100
+ }
101
+ }
102
+ return undefined;
103
+ }
104
+ /** One ortholog gene per species, restricted to the requested taxa. */
105
+ export async function fetchOrthologGenes(geneId, taxa) {
106
+ const json = await jsonfetch(ncbiUrl(`${DATASETS}/gene/id/${geneId}/orthologs?returned_content=COMPLETE`));
107
+ const byTaxon = new Map();
108
+ for (const { gene } of json.reports ?? []) {
109
+ const taxId = Number(gene?.tax_id);
110
+ if (gene?.gene_id && taxa.has(taxId) && !byTaxon.has(taxId)) {
111
+ byTaxon.set(taxId, {
112
+ taxId,
113
+ geneId: gene.gene_id,
114
+ scientificName: gene.taxname ?? String(taxId),
115
+ commonName: gene.common_name,
116
+ });
117
+ }
118
+ }
119
+ return [...byTaxon.values()].sort((a, b) => (COMMON_TAX_RANK.get(a.taxId) ?? Infinity) -
120
+ (COMMON_TAX_RANK.get(b.taxId) ?? Infinity));
121
+ }
122
+ /**
123
+ * geneId -> representative protein accession: MANE Select where flagged, else
124
+ * the longest isoform. A stable, comparable choice across species — picking
125
+ * "the first" would silently vary with NCBI's ordering.
126
+ */
127
+ export async function fetchRepresentativeProteins(geneIds) {
128
+ const byGene = new Map();
129
+ if (geneIds.length > 0) {
130
+ const json = await jsonfetch(ncbiUrl(`${DATASETS}/gene/id/${geneIds.join(',')}/product_report`));
131
+ for (const { product } of json.reports ?? []) {
132
+ const candidates = (product?.transcripts ?? [])
133
+ .map(t => ({
134
+ acc: t.protein?.accession_version,
135
+ len: t.protein?.length ?? 0,
136
+ mane: /select/i.test(t.select_category ?? ''),
137
+ }))
138
+ .filter((c) => !!c.acc);
139
+ const best = candidates.find(c => c.mane) ??
140
+ [...candidates].sort((a, b) => b.len - a.len).at(0);
141
+ if (product?.gene_id && best) {
142
+ byGene.set(product.gene_id, best.acc);
143
+ }
144
+ }
145
+ }
146
+ return byGene;
147
+ }
148
+ /** accession (first header token) -> ungapped sequence, from a multi-FASTA. */
149
+ export function parseFasta(text) {
150
+ const map = new Map();
151
+ let acc;
152
+ let buf = [];
153
+ for (const line of text.split('\n')) {
154
+ if (line.startsWith('>')) {
155
+ if (acc) {
156
+ map.set(acc, buf.join(''));
157
+ }
158
+ acc = line.slice(1).split(/\s+/)[0];
159
+ buf = [];
160
+ }
161
+ else {
162
+ buf.push(line.trim());
163
+ }
164
+ }
165
+ if (acc) {
166
+ map.set(acc, buf.join(''));
167
+ }
168
+ return map;
169
+ }
170
+ function sanitize(name) {
171
+ return name.replace(/[^A-Za-z0-9]+/g, '_').replace(/^_+|_+$/g, '');
172
+ }
173
+ /**
174
+ * Sanitized, unique single-token labels used identically in the FASTA headers,
175
+ * the tree leaf names and the domain GFF seq_ids — that identity is how the
176
+ * viewer pairs a tree leaf to its alignment row to its domain track. Collisions
177
+ * get a numeric suffix rather than silently overwriting a row.
178
+ */
179
+ export function dedupeLabels(names) {
180
+ const seen = new Map();
181
+ return names.map(name => {
182
+ const base = sanitize(name) || 'row';
183
+ const n = seen.get(base) ?? 0;
184
+ seen.set(base, n + 1);
185
+ return n === 0 ? base : `${base}_${n + 1}`;
186
+ });
187
+ }
188
+ /**
189
+ * The representative protein for a single gene, with its sequence. Used to
190
+ * decide whether the user's own translated transcript is byte-identical to the
191
+ * RefSeq protein — if it is, that accession's precomputed CDD domains apply to
192
+ * the query row exactly, and if it isn't, they would land at an offset.
193
+ */
194
+ export async function fetchProteinForGene(geneId) {
195
+ const acc = (await fetchRepresentativeProteins([geneId])).get(geneId);
196
+ if (!acc) {
197
+ return undefined;
198
+ }
199
+ const seq = parseFasta(await textfetch(ncbiUrl(`${EUTILS}/efetch.fcgi?db=protein&id=${acc}&rettype=fasta&retmode=text`))).get(acc);
200
+ return seq ? { accession: acc, sequence: seq } : undefined;
201
+ }
202
+ /**
203
+ * The whole NCBI half of the pipeline: gene -> ortholog rows carrying labels,
204
+ * accessions and sequences. Everything here is a precomputed lookup, so this
205
+ * returns in seconds rather than the 10+ minutes a BLAST submission costs.
206
+ */
207
+ export async function fetchOrthologRows({ geneId, taxa, onProgress, }) {
208
+ onProgress('Finding orthologs across species...');
209
+ const genes = await fetchOrthologGenes(geneId, taxa);
210
+ if (genes.length < 2) {
211
+ throw new Error(`Only ${genes.length} ortholog(s) found among the selected species — not enough to align`);
212
+ }
213
+ onProgress('Selecting a representative protein per species...');
214
+ const proteinByGene = await fetchRepresentativeProteins(genes.map(g => g.geneId));
215
+ const withProtein = genes.filter(g => proteinByGene.has(g.geneId));
216
+ if (withProtein.length < 2) {
217
+ throw new Error('Could not resolve representative proteins for the orthologs');
218
+ }
219
+ onProgress(`Fetching ${withProtein.length} protein sequences...`);
220
+ const accessions = withProtein.map(g => proteinByGene.get(g.geneId));
221
+ const seqByAcc = parseFasta(await textfetch(ncbiUrl(`${EUTILS}/efetch.fcgi?db=protein&id=${accessions.join(',')}&rettype=fasta&retmode=text`)));
222
+ const labels = dedupeLabels(withProtein.map(g => g.commonName ?? g.scientificName));
223
+ const rows = withProtein
224
+ .map((g, i) => {
225
+ const protein = proteinByGene.get(g.geneId);
226
+ return {
227
+ taxId: g.taxId,
228
+ label: labels[i],
229
+ scientificName: g.scientificName,
230
+ commonName: g.commonName,
231
+ geneId: g.geneId,
232
+ protein,
233
+ sequence: seqByAcc.get(protein) ?? '',
234
+ };
235
+ })
236
+ .filter(r => r.sequence);
237
+ if (rows.length < 2) {
238
+ throw new Error('Could not fetch protein sequences for the orthologs');
239
+ }
240
+ return rows;
241
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,41 @@
1
+ import { describe, expect, test } from 'vitest';
2
+ import { dedupeLabels, parseFasta } from './ncbiOrthologs';
3
+ describe('dedupeLabels', () => {
4
+ test('sanitizes to single tokens', () => {
5
+ // labels are used identically as FASTA headers, Newick leaf names and GFF
6
+ // seq_ids, so anything that would need quoting in one of those is stripped
7
+ expect(dedupeLabels(['house mouse', 'Norway rat'])).toEqual([
8
+ 'house_mouse',
9
+ 'Norway_rat',
10
+ ]);
11
+ expect(dedupeLabels(['Frog (X. tropicalis)'])).toEqual([
12
+ 'Frog_X_tropicalis',
13
+ ]);
14
+ });
15
+ test('suffixes collisions rather than overwriting a row', () => {
16
+ expect(dedupeLabels(['a b', 'a-b', 'a_b'])).toEqual([
17
+ 'a_b',
18
+ 'a_b_2',
19
+ 'a_b_3',
20
+ ]);
21
+ });
22
+ test('falls back for a name with no usable characters', () => {
23
+ expect(dedupeLabels(['...', '...'])).toEqual(['row', 'row_2']);
24
+ });
25
+ });
26
+ describe('parseFasta', () => {
27
+ test('keys by the first header token and joins wrapped lines', () => {
28
+ const map = parseFasta(['>NP_000537.3 cellular tumor antigen p53', 'MEEP', 'QSDP', ''].join('\n'));
29
+ expect(map.get('NP_000537.3')).toBe('MEEPQSDP');
30
+ });
31
+ test('reads every record of a multi-FASTA', () => {
32
+ const map = parseFasta(['>A one', 'MMM', '>B two', 'KKK', '>C three', 'LLL'].join('\n'));
33
+ expect([...map.keys()]).toEqual(['A', 'B', 'C']);
34
+ expect(map.get('C')).toBe('LLL');
35
+ });
36
+ test('returns nothing for a response that carried no records', () => {
37
+ // efetch answers an unknown accession with an error body, not a 4xx, so a
38
+ // caller that assumed "text back = sequences" would build empty rows
39
+ expect(parseFasta('Error: CEFetchPApplication::proxy_stream()').size).toBe(0);
40
+ });
41
+ });
package/dist/version.d.ts CHANGED
@@ -1 +1 @@
1
- export declare const version = "2.7.3";
1
+ export declare const version = "2.8.0";
package/dist/version.js CHANGED
@@ -1 +1 @@
1
- export const version = '2.7.3';
1
+ export const version = '2.8.0';
package/package.json CHANGED
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "2.7.3",
2
+ "version": "2.8.0",
3
3
  "license": "MIT",
4
4
  "name": "jbrowse-plugin-msaview",
5
5
  "repository": {
@@ -43,7 +43,7 @@
43
43
  "eslint-plugin-unicorn": "^72.0.0",
44
44
  "mobx": "^6.16.1",
45
45
  "mobx-react": "^9.2.2",
46
- "msa-parsers": "^5.6.1",
46
+ "msa-parsers": "^5.7.1",
47
47
  "pixelmatch": "^7.2.0",
48
48
  "pngjs": "^7.0.0",
49
49
  "prettier": "^3.9.6",
@@ -51,7 +51,7 @@
51
51
  "puppeteer": "^25.3.0",
52
52
  "react": "^19.2.8",
53
53
  "react-dom": "^19.2.8",
54
- "react-msaview": "^5.7.0",
54
+ "react-msaview": "^5.7.1",
55
55
  "rimraf": "^6.1.3",
56
56
  "rxjs": "^7.8.2",
57
57
  "serve": "^14.2.6",
@@ -6,6 +6,7 @@ import { Tab, Tabs } from '@mui/material'
6
6
 
7
7
  import ManualMSALoader from './ManualMSALoader/ManualMSALoader'
8
8
  import NCBIBlastPanel from './NCBIBlastQuery/NCBIBlastPanel'
9
+ import OrthologPanel from './OrthologQuery/OrthologPanel'
9
10
  import PreLoadedMSA from './PreLoadedMSA/PreLoadedMSADataPanel'
10
11
  import { readMsaDatasets } from './PreLoadedMSA/types'
11
12
  import TabPanel from './TabPanel'
@@ -25,9 +26,11 @@ export default function LaunchMsaViewDialog({
25
26
  const datasets = readMsaDatasets(session.jbrowse)
26
27
  const hasPreloadedDatasets = !!datasets?.length
27
28
 
29
+ // orthologs first, and the default: it answers the same question in ~10s
30
+ // that BLAST takes 10+ minutes to answer worse (see utils/ncbiOrthologs.ts)
28
31
  const [value, setValue] = useState<
29
- 'ncbi_blast' | 'preloaded_msa' | 'manual_msa'
30
- >('ncbi_blast')
32
+ 'orthologs' | 'ncbi_blast' | 'preloaded_msa' | 'manual_msa'
33
+ >('orthologs')
31
34
 
32
35
  return (
33
36
  <Dialog maxWidth="xl" title="Launch MSA view" open onClose={handleClose}>
@@ -37,12 +40,20 @@ export default function LaunchMsaViewDialog({
37
40
  setValue(newValue)
38
41
  }}
39
42
  >
43
+ <Tab label="Orthologs (fast)" value="orthologs" />
40
44
  <Tab label="NCBI BLAST query" value="ncbi_blast" />
41
45
  {hasPreloadedDatasets ? (
42
46
  <Tab label="Pre-loaded MSA datasets" value="preloaded_msa" />
43
47
  ) : null}
44
48
  <Tab label="Manual upload" value="manual_msa" />
45
49
  </Tabs>
50
+ <TabPanel value={value} index="orthologs">
51
+ <OrthologPanel
52
+ handleClose={handleClose}
53
+ feature={feature}
54
+ model={model}
55
+ />
56
+ </TabPanel>
46
57
  <TabPanel value={value} index="ncbi_blast">
47
58
  <NCBIBlastPanel
48
59
  handleClose={handleClose}
@@ -0,0 +1,172 @@
1
+ import React, { useMemo, useState } from 'react'
2
+
3
+ import { Checkbox, FormControlLabel, MenuItem, Typography } from '@mui/material'
4
+ import { observer } from 'mobx-react'
5
+ import { makeStyles } from 'tss-react/mui'
6
+
7
+ import { orthologLaunchView } from './orthologLaunchView'
8
+ import TextField2 from '../../../components/TextField2'
9
+ import { COMMON_SPECIES } from '../../../utils/ncbiOrthologs'
10
+ import {
11
+ getGeneDisplayName,
12
+ getGeneIdentifiers,
13
+ getLinearGenomeView,
14
+ getTranscriptDisplayName,
15
+ } from '../../util'
16
+ import LaunchPanelContent from '../LaunchPanelContent'
17
+ import MsaAlgorithmSelect from '../NCBIBlastQuery/MsaAlgorithmSelect'
18
+ import SubmitCancelActions from '../SubmitCancelActions'
19
+ import TranscriptSelector from '../TranscriptSelector'
20
+ import { useTranscriptSelection } from '../useTranscriptSelection'
21
+
22
+ import type { MsaAlgorithm } from '../NCBIBlastQuery/consts'
23
+ import type { AbstractTrackModel, Feature } from '@jbrowse/core/util'
24
+
25
+ const useStyles = makeStyles()({
26
+ selectField: {
27
+ width: 180,
28
+ },
29
+ // A GRID, not a wrapping flex row of fixed-width items. The old form was three
30
+ // 160px columns inside a 560px box, which is five rows for thirteen species and
31
+ // eight for twenty-three -- and the checkbox list is the tallest thing in the
32
+ // dialog, so those rows are the dialog's height. Five auto-fitted columns is
33
+ // five rows for twenty-three, i.e. more species in less space, and it reflows
34
+ // rather than being pinned to a width the dialog may not have.
35
+ speciesBox: {
36
+ display: 'grid',
37
+ gridTemplateColumns: 'repeat(auto-fit, minmax(130px, 1fr))',
38
+ maxWidth: 700,
39
+ marginTop: 4,
40
+ },
41
+ // The label carries the row height; the default control padding is what makes
42
+ // 23 rows of it tall.
43
+ species: {
44
+ marginRight: 0,
45
+ },
46
+ })
47
+
48
+ const OrthologPanel = observer(function ({
49
+ handleClose,
50
+ feature,
51
+ model,
52
+ }: {
53
+ model: AbstractTrackModel
54
+ feature: Feature
55
+ handleClose: () => void
56
+ }) {
57
+ const { classes } = useStyles()
58
+ const view = getLinearGenomeView(model)
59
+ const [launchViewError, setLaunchViewError] = useState<unknown>()
60
+ const [taxId, setTaxId] = useState(9606)
61
+ const [msaAlgorithm, setMsaAlgorithm] = useState<MsaAlgorithm>('clustalo')
62
+ const [excluded, setExcluded] = useState<number[]>([])
63
+
64
+ const geneCandidates = useMemo(() => getGeneIdentifiers(feature), [feature])
65
+ const transcriptSelection = useTranscriptSelection({ feature, view })
66
+ const { selectedTranscript, proteinSequence } = transcriptSelection
67
+ const e = transcriptSelection.error ?? launchViewError
68
+
69
+ const taxa = COMMON_SPECIES.map(s => s.taxId).filter(
70
+ t => !excluded.includes(t),
71
+ )
72
+
73
+ return (
74
+ <>
75
+ <LaunchPanelContent error={e}>
76
+ {/* One line rather than seven. What a reader needs here is which tab
77
+ to pick, and that is the seconds-against-minutes comparison; the
78
+ rest (species labels, CDD overlay, the query row being the selected
79
+ transcript) is visible in the result or documented, and as prose it
80
+ was most of the dialog's height. */}
81
+ <Typography variant="body2">
82
+ NCBI&apos;s precomputed orthologs, one gene per species, aligned at EBI
83
+ in seconds rather than the 10+ minutes BLAST takes.
84
+ </Typography>
85
+
86
+ <div>
87
+ <TextField2
88
+ variant="outlined"
89
+ label="Query species"
90
+ className={classes.selectField}
91
+ select
92
+ value={taxId}
93
+ onChange={event => {
94
+ setTaxId(Number(event.target.value))
95
+ }}
96
+ helperText="the species this gene is from"
97
+ >
98
+ {COMMON_SPECIES.map(s => (
99
+ <MenuItem value={s.taxId} key={s.taxId}>
100
+ {s.label}
101
+ </MenuItem>
102
+ ))}
103
+ </TextField2>
104
+
105
+ <MsaAlgorithmSelect
106
+ className={classes.selectField}
107
+ value={msaAlgorithm}
108
+ onChange={setMsaAlgorithm}
109
+ />
110
+ </div>
111
+
112
+ <Typography variant="subtitle2" style={{ marginTop: 8 }}>
113
+ Species to include (those without an ortholog are skipped)
114
+ </Typography>
115
+ <div className={classes.speciesBox}>
116
+ {COMMON_SPECIES.map(s => (
117
+ <FormControlLabel
118
+ className={classes.species}
119
+ key={s.taxId}
120
+ control={
121
+ <Checkbox
122
+ checked={!excluded.includes(s.taxId)}
123
+ onChange={event => {
124
+ setExcluded(
125
+ event.target.checked
126
+ ? excluded.filter(t => t !== s.taxId)
127
+ : [...excluded, s.taxId],
128
+ )
129
+ }}
130
+ />
131
+ }
132
+ label={s.label}
133
+ />
134
+ ))}
135
+ </div>
136
+
137
+ <TranscriptSelector feature={feature} {...transcriptSelection} />
138
+
139
+ </LaunchPanelContent>
140
+ <SubmitCancelActions
141
+ submitDisabled={!proteinSequence || taxa.length < 2}
142
+ onSubmit={() => {
143
+ try {
144
+ if (selectedTranscript) {
145
+ setLaunchViewError(undefined)
146
+ orthologLaunchView({
147
+ feature: selectedTranscript,
148
+ view,
149
+ newViewTitle: `Orthologs - ${getGeneDisplayName(feature)} - ${getTranscriptDisplayName(selectedTranscript)}`,
150
+ orthologParams: {
151
+ taxId,
152
+ taxa,
153
+ geneCandidates,
154
+ msaAlgorithm,
155
+ selectedTranscript,
156
+ proteinSequence,
157
+ },
158
+ })
159
+ handleClose()
160
+ }
161
+ } catch (e) {
162
+ console.error(e)
163
+ setLaunchViewError(e)
164
+ }
165
+ }}
166
+ onCancel={handleClose}
167
+ />
168
+ </>
169
+ )
170
+ })
171
+
172
+ export default OrthologPanel