jbrowse-plugin-msaview 3.2.0 → 3.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,79 @@
1
+ import type { OrthologRow } from './ncbiOrthologs';
2
+ export interface PantherGenome {
3
+ /** PANTHER's organism code, e.g. HUMAN, DROME */
4
+ code: string;
5
+ taxId: number;
6
+ /** short common name, e.g. fruit_fly */
7
+ name: string;
8
+ /** scientific name */
9
+ longName: string;
10
+ }
11
+ /**
12
+ * `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
13
+ * names organisms by code only in ortholog results.
14
+ */
15
+ export declare function parseGenomes(json: unknown): PantherGenome[];
16
+ export interface PantherGene {
17
+ code: string;
18
+ /** UniProt accession */
19
+ accession: string;
20
+ /** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
21
+ geneRef: string;
22
+ }
23
+ export interface PantherHit extends PantherGene {
24
+ symbol?: string;
25
+ /**
26
+ * LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
27
+ * other ortholog in a one-to-many or many-to-many family
28
+ */
29
+ type: 'LDO' | 'O';
30
+ }
31
+ /**
32
+ * `matchortho` -> the query gene (PANTHER names it in every row) and one hit
33
+ * per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
34
+ * no ortholog in the target set comes back as a bare `{ id }`.
35
+ */
36
+ export declare function parseMatches(json: unknown): {
37
+ unmapped: boolean;
38
+ query?: PantherGene;
39
+ hits: PantherHit[];
40
+ };
41
+ /**
42
+ * One hit per organism, in first-seen order: the LDO where PANTHER named one,
43
+ * else the first other ortholog it listed. A many-to-many family (the Hox
44
+ * genes) has no LDO at all, so dropping to "first O" is what keeps those
45
+ * species in the alignment.
46
+ */
47
+ export declare function pickOnePerGenome(hits: PantherHit[]): PantherHit[];
48
+ /** `uniprotkb/accessions` -> accession -> sequence */
49
+ export declare function parseSequences(json: unknown): Map<string, string>;
50
+ /** The proteome list, fetched once per page and forgotten on failure. */
51
+ export declare function fetchGenomes(): Promise<PantherGenome[]>;
52
+ export interface PantherOrthologs {
53
+ /** the candidate PANTHER recognised */
54
+ matched: string;
55
+ /** the query gene as PANTHER knows it, with its UniProt sequence */
56
+ query?: PantherGene & {
57
+ sequence: string;
58
+ };
59
+ rows: OrthologRow[];
60
+ }
61
+ /**
62
+ * The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
63
+ * labels, accessions and sequences, plus the query gene's own protein for the
64
+ * query row. Two lookups (genomes, orthologs), one taxonomy batch for the
65
+ * labels, one UniProt batch for the sequences.
66
+ *
67
+ * `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
68
+ * `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
69
+ * `exclude` drops the query taxon, `limit` caps the rows before their
70
+ * sequences are fetched.
71
+ */
72
+ export declare function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit, onProgress, }: {
73
+ candidates: string[];
74
+ taxId: number;
75
+ taxa?: Set<number>;
76
+ exclude?: number;
77
+ limit?: number;
78
+ onProgress: (arg: string) => void;
79
+ }): Promise<PantherOrthologs>;
@@ -0,0 +1,262 @@
1
+ // The second ortholog source, for the genes NCBI's ortholog sets leave out.
2
+ //
3
+ // NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
4
+ // worm or plant gene comes back with orthologs only in its own clade, and a fly
5
+ // gene gets insects and nothing else. PANTHER's ortholog sets span its 144
6
+ // reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
7
+ // answers "this gene's ortholog in every genome" with a UniProt accession per
8
+ // target. Sequences come from one UniProt batch call. Both hosts send
9
+ // `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
10
+ // OMA, OrthoDB and Ensembl are in react-msaview's
11
+ // agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
12
+ //
13
+ // Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
14
+ // labels, the aligner and the CDD overlay do not know which source ran.
15
+ // `protein` is the UniProt accession; efetch serves UniProt accessions as
16
+ // GenPept records with CDD Region features, so the overlay attaches as it does
17
+ // to a RefSeq accession.
18
+ import { jsonfetch } from './fetch';
19
+ import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs';
20
+ import { fetchTaxonomyInfo } from './taxonomyNames';
21
+ const PANTHER = 'https://pantherdb.org/services/oai/pantherdb';
22
+ const UNIPROT = 'https://rest.uniprot.org/uniprotkb';
23
+ /**
24
+ * `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
25
+ * names organisms by code only in ortholog results.
26
+ */
27
+ export function parseGenomes(json) {
28
+ const list = json.search?.output?.genomes?.genome ?? [];
29
+ return list.flatMap(g => g.short_name && g.taxon_id
30
+ ? [
31
+ {
32
+ code: g.short_name,
33
+ taxId: g.taxon_id,
34
+ name: g.name ?? g.short_name,
35
+ longName: g.long_name ?? g.short_name,
36
+ },
37
+ ]
38
+ : []);
39
+ }
40
+ // "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
41
+ function parseGeneRef(ref) {
42
+ const [code, ...xrefs] = (ref ?? '').split('|');
43
+ const accession = xrefs
44
+ .find(x => x.startsWith('UniProtKB='))
45
+ ?.slice('UniProtKB='.length);
46
+ const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession;
47
+ return code && accession && geneRef ? { code, accession, geneRef } : undefined;
48
+ }
49
+ /**
50
+ * `matchortho` -> the query gene (PANTHER names it in every row) and one hit
51
+ * per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
52
+ * no ortholog in the target set comes back as a bare `{ id }`.
53
+ */
54
+ export function parseMatches(json) {
55
+ const mapping = json.search?.mapping;
56
+ const mapped = mapping?.mapped;
57
+ const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : [];
58
+ const hits = [];
59
+ let query;
60
+ for (const row of rows) {
61
+ query ??= parseGeneRef(row.gene);
62
+ const target = parseGeneRef(row.target_gene);
63
+ if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
64
+ hits.push({
65
+ ...target,
66
+ symbol: row.target_gene_symbol === undefined
67
+ ? undefined
68
+ : String(row.target_gene_symbol),
69
+ type: row.ortholog,
70
+ });
71
+ }
72
+ }
73
+ return { unmapped: !!mapping?.unmapped_ids, query, hits };
74
+ }
75
+ /**
76
+ * One hit per organism, in first-seen order: the LDO where PANTHER named one,
77
+ * else the first other ortholog it listed. A many-to-many family (the Hox
78
+ * genes) has no LDO at all, so dropping to "first O" is what keeps those
79
+ * species in the alignment.
80
+ */
81
+ export function pickOnePerGenome(hits) {
82
+ const byCode = new Map();
83
+ for (const hit of hits) {
84
+ const current = byCode.get(hit.code);
85
+ if (!current || (current.type === 'O' && hit.type === 'LDO')) {
86
+ byCode.set(hit.code, hit);
87
+ }
88
+ }
89
+ return [...byCode.values()];
90
+ }
91
+ /** `uniprotkb/accessions` -> accession -> sequence */
92
+ export function parseSequences(json) {
93
+ const map = new Map();
94
+ for (const r of json.results ?? []) {
95
+ if (r.primaryAccession && r.sequence?.value) {
96
+ map.set(r.primaryAccession, r.sequence.value);
97
+ }
98
+ }
99
+ return map;
100
+ }
101
+ let genomes;
102
+ /** The proteome list, fetched once per page and forgotten on failure. */
103
+ export function fetchGenomes() {
104
+ genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
105
+ .then(parseGenomes)
106
+ .catch((e) => {
107
+ genomes = undefined;
108
+ throw e;
109
+ });
110
+ return genomes;
111
+ }
112
+ // UniProt caps one `accessions` call at 100 ids
113
+ const UNIPROT_CHUNK = 100;
114
+ async function fetchSequences(accessions) {
115
+ const map = new Map();
116
+ for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
117
+ const chunk = accessions.slice(i, i + UNIPROT_CHUNK);
118
+ const json = await jsonfetch(`${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`);
119
+ for (const [acc, seq] of parseSequences(json)) {
120
+ map.set(acc, seq);
121
+ }
122
+ }
123
+ return map;
124
+ }
125
+ // strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
126
+ // resolveGeneId does for NCBI
127
+ function cleanCandidate(raw) {
128
+ return raw
129
+ .trim()
130
+ .replace(/^\w+:/, '')
131
+ .replace(/\.\d+$/, '');
132
+ }
133
+ /**
134
+ * One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
135
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
136
+ * some of those are names PANTHER knows. `targets` omitted asks for every
137
+ * genome PANTHER has, which is one call rather than one per genome.
138
+ */
139
+ async function matchOrthologs(candidates, taxId, targets) {
140
+ let matched;
141
+ for (const raw of candidates) {
142
+ const query = cleanCandidate(raw);
143
+ if (!query) {
144
+ continue;
145
+ }
146
+ const params = new URLSearchParams({
147
+ geneInputList: query,
148
+ organism: String(taxId),
149
+ orthologType: 'all',
150
+ });
151
+ if (targets) {
152
+ params.set('targetOrganism', targets.map(t => t.taxId).join(','));
153
+ }
154
+ const parsed = parseMatches(await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`));
155
+ if (!parsed.unmapped) {
156
+ matched ??= query;
157
+ if (parsed.query) {
158
+ return { ...parsed, matched: query };
159
+ }
160
+ }
161
+ }
162
+ return matched === undefined
163
+ ? undefined
164
+ : { matched, hits: [], query: undefined };
165
+ }
166
+ /**
167
+ * The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
168
+ * labels, accessions and sequences, plus the query gene's own protein for the
169
+ * query row. Two lookups (genomes, orthologs), one taxonomy batch for the
170
+ * labels, one UniProt batch for the sequences.
171
+ *
172
+ * `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
173
+ * `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
174
+ * `exclude` drops the query taxon, `limit` caps the rows before their
175
+ * sequences are fetched.
176
+ */
177
+ export async function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit = defaultMaxSpecies, onProgress, }) {
178
+ const all = await fetchGenomes();
179
+ const byTaxId = new Map(all.map(g => [g.taxId, g]));
180
+ const byCode = new Map(all.map(g => [g.code, g]));
181
+ const queryGenome = byTaxId.get(taxId);
182
+ if (!queryGenome) {
183
+ throw new Error(`PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`);
184
+ }
185
+ const targets = taxa
186
+ ? [...taxa].flatMap(t => {
187
+ const g = byTaxId.get(t);
188
+ return g && t !== exclude ? [g] : [];
189
+ })
190
+ : undefined;
191
+ onProgress('Matching orthologs at PANTHER...');
192
+ const match = await matchOrthologs(candidates, taxId, targets);
193
+ if (!match) {
194
+ throw new Error(`PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`);
195
+ }
196
+ const rank = new Map(targets?.map((t, i) => [t.taxId, i]));
197
+ const picks = pickOnePerGenome(match.hits)
198
+ .map(hit => ({ hit, genome: byCode.get(hit.code) }))
199
+ .filter((p) => !!p.genome &&
200
+ p.genome.taxId !== exclude &&
201
+ (targets ? rank.has(p.genome.taxId) : true))
202
+ .sort((a, b) => targets ? rank.get(a.genome.taxId) - rank.get(b.genome.taxId) : 0)
203
+ .slice(0, limit);
204
+ if (picks.length < 2) {
205
+ throw new Error(`Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`);
206
+ }
207
+ onProgress(`Fetching ${picks.length} protein sequences from UniProt...`);
208
+ const [names, sequences] = await Promise.all([
209
+ taxonomyNames(picks.map(p => p.genome.taxId)),
210
+ fetchSequences([
211
+ ...(match.query ? [match.query.accession] : []),
212
+ ...picks.map(p => p.hit.accession),
213
+ ]),
214
+ ]);
215
+ const described = picks.map(({ hit, genome }) => {
216
+ const info = names.get(genome.taxId);
217
+ return {
218
+ hit,
219
+ genome,
220
+ name: info ? (info.commonName ?? info.sciname) : genome.name,
221
+ scientificName: info?.sciname || genome.longName,
222
+ commonName: info?.commonName,
223
+ };
224
+ });
225
+ const labels = dedupeLabels(described.map(d => d.name));
226
+ const rows = described
227
+ .map(({ hit, genome, scientificName, commonName }, i) => ({
228
+ taxId: genome.taxId,
229
+ label: labels[i],
230
+ scientificName,
231
+ commonName,
232
+ geneId: hit.geneRef,
233
+ protein: hit.accession,
234
+ sequence: sequences.get(hit.accession) ?? '',
235
+ }))
236
+ .filter(r => r.sequence);
237
+ if (rows.length < 2) {
238
+ throw new Error('Could not fetch protein sequences for the orthologs');
239
+ }
240
+ const querySequence = match.query && sequences.get(match.query.accession);
241
+ return {
242
+ matched: match.matched,
243
+ query: match.query && querySequence
244
+ ? { ...match.query, sequence: querySequence }
245
+ : undefined,
246
+ rows,
247
+ };
248
+ }
249
+ /**
250
+ * NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
251
+ * are. A failed lookup only costs the labels, which fall back to PANTHER's own
252
+ * short names, so it is logged rather than thrown.
253
+ */
254
+ async function taxonomyNames(taxIds) {
255
+ try {
256
+ return await fetchTaxonomyInfo(taxIds);
257
+ }
258
+ catch (e) {
259
+ console.warn('[msaview-orthologs] taxonomy name lookup failed:', e);
260
+ return new Map();
261
+ }
262
+ }
package/dist/version.d.ts CHANGED
@@ -1 +1 @@
1
- export declare const version = "3.2.0";
1
+ export declare const version = "3.3.0";
package/dist/version.js CHANGED
@@ -1 +1 @@
1
- export const version = '3.2.0';
1
+ export const version = '3.3.0';
package/package.json CHANGED
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "3.2.0",
2
+ "version": "3.3.0",
3
3
  "license": "MIT",
4
4
  "name": "jbrowse-plugin-msaview",
5
5
  "repository": {
@@ -4,10 +4,14 @@ import { Typography } from '@mui/material'
4
4
  import { observer } from 'mobx-react'
5
5
  import { makeStyles } from 'tss-react/mui'
6
6
 
7
+ import OrthologSourceSelect, {
8
+ ORTHOLOG_SOURCE_STORAGE_KEY,
9
+ } from './OrthologSourceSelect'
7
10
  import QuerySpeciesSelect from './QuerySpeciesSelect'
8
11
  import { orthologLaunchView } from './orthologLaunchView'
9
12
  import TextField2 from '../../../components/TextField2'
10
13
  import { defaultMaxSpecies } from '../../../utils/ncbiOrthologs'
14
+ import { useLocalStorage } from '../../../utils/useLocalStorage'
11
15
  import {
12
16
  getGeneDisplayName,
13
17
  getGeneIdentifiers,
@@ -20,6 +24,7 @@ import SubmitCancelActions from '../SubmitCancelActions'
20
24
  import TranscriptSelector from '../TranscriptSelector'
21
25
  import { useTranscriptSelection } from '../useTranscriptSelection'
22
26
 
27
+ import type { OrthologSource } from '../../../MsaViewPanel/model'
23
28
  import type { MsaAlgorithm } from '../BlastQuery/consts'
24
29
  import type { AbstractTrackModel, Feature } from '@jbrowse/core/util'
25
30
 
@@ -42,6 +47,10 @@ const OrthologPanel = observer(function ({
42
47
  const view = getLinearGenomeView(model)
43
48
  const [launchViewError, setLaunchViewError] = useState<unknown>()
44
49
  const [taxId, setTaxId] = useState(9606)
50
+ const [source, setSource] = useLocalStorage<OrthologSource>(
51
+ ORTHOLOG_SOURCE_STORAGE_KEY,
52
+ 'ncbi',
53
+ )
45
54
  const [msaAlgorithm, setMsaAlgorithm] = useState<MsaAlgorithm>('clustalo')
46
55
  const [maxSpecies, setMaxSpecies] = useState(String(defaultMaxSpecies))
47
56
 
@@ -65,11 +74,17 @@ const OrthologPanel = observer(function ({
65
74
  instant, and it is now the aligner that costs the wait, about half a
66
75
  second per row. */}
67
76
  <Typography variant="body2">
68
- NCBI&apos;s precomputed orthologs, one gene per species, looked up
69
- rather than searched for. No BLAST job to queue.
77
+ Precomputed orthologs, one gene per species, looked up rather than
78
+ searched for. No BLAST job to queue.
70
79
  </Typography>
71
80
 
72
81
  <div>
82
+ <OrthologSourceSelect
83
+ className={classes.selectField}
84
+ value={source}
85
+ onChange={setSource}
86
+ />
87
+
73
88
  <QuerySpeciesSelect
74
89
  className={classes.selectField}
75
90
  value={taxId}
@@ -93,7 +108,7 @@ const OrthologPanel = observer(function ({
93
108
  setMaxSpecies(event.target.value)
94
109
  }}
95
110
  error={!rowCountValid}
96
- helperText="the closest N species NCBI has"
111
+ helperText={`the closest N species ${source === 'panther' ? 'PANTHER' : 'NCBI'} has`}
97
112
  />
98
113
  </div>
99
114
 
@@ -112,6 +127,7 @@ const OrthologPanel = observer(function ({
112
127
  newViewTitle: `Orthologs - ${getGeneDisplayName(feature)} - ${getTranscriptDisplayName(selectedTranscript)}`,
113
128
  orthologParams: {
114
129
  taxId,
130
+ source,
115
131
  maxSpecies: rowCount,
116
132
  geneCandidates,
117
133
  msaAlgorithm,
@@ -0,0 +1,52 @@
1
+ import React from 'react'
2
+
3
+ import { MenuItem } from '@mui/material'
4
+
5
+ import TextField2 from '../../../components/TextField2'
6
+
7
+ import type { OrthologSource } from '../../../MsaViewPanel/model'
8
+
9
+ export const ORTHOLOG_SOURCE_STORAGE_KEY = 'msaview-ortholog-source'
10
+
11
+ export const orthologSourceLabels: Record<OrthologSource, string> = {
12
+ ncbi: 'NCBI orthologs',
13
+ panther: 'PANTHER',
14
+ }
15
+
16
+ // Which species a source can answer for, in the words a reader picking one
17
+ // needs: NCBI's ortholog sets stop at vertebrates and insects, PANTHER's run
18
+ // from human to yeast and Arabidopsis.
19
+ const hints: Record<OrthologSource, string> = {
20
+ ncbi: 'vertebrates and insects',
21
+ panther: 'also yeast, worm, fly and plants',
22
+ }
23
+
24
+ export default function OrthologSourceSelect({
25
+ value,
26
+ onChange,
27
+ className,
28
+ }: {
29
+ value: OrthologSource
30
+ onChange: (val: OrthologSource) => void
31
+ className?: string
32
+ }) {
33
+ return (
34
+ <TextField2
35
+ variant="outlined"
36
+ label="Source"
37
+ className={className}
38
+ select
39
+ value={value}
40
+ helperText={hints[value]}
41
+ onChange={event => {
42
+ onChange(event.target.value as OrthologSource)
43
+ }}
44
+ >
45
+ {(Object.keys(orthologSourceLabels) as OrthologSource[]).map(val => (
46
+ <MenuItem value={val} key={val}>
47
+ {orthologSourceLabels[val]}
48
+ </MenuItem>
49
+ ))}
50
+ </TextField2>
51
+ )
52
+ }
@@ -8,6 +8,7 @@ import {
8
8
  fetchProteinForGene,
9
9
  resolveGeneId,
10
10
  } from '../utils/ncbiOrthologs'
11
+ import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
11
12
  import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
12
13
 
13
14
  import type { JBrowsePluginMsaViewModel } from './model'
@@ -24,12 +25,16 @@ vi.mock('../utils/ncbiOrthologs', async importOriginal => ({
24
25
  fetchProteinForGene: vi.fn(),
25
26
  fetchOrthologRows: vi.fn(),
26
27
  }))
28
+ vi.mock('../utils/pantherOrthologs', () => ({
29
+ fetchPantherOrthologs: vi.fn(),
30
+ }))
27
31
  vi.mock('../utils/msa', () => ({ launchMSA: vi.fn() }))
28
32
  vi.mock('../utils/taxonomyNames', () => ({ fetchTaxonomyInfo: vi.fn() }))
29
33
 
30
34
  const mockResolveGeneId = vi.mocked(resolveGeneId)
31
35
  const mockFetchProtein = vi.mocked(fetchProteinForGene)
32
36
  const mockFetchRows = vi.mocked(fetchOrthologRows)
37
+ const mockFetchPanther = vi.mocked(fetchPantherOrthologs)
33
38
  const mockLaunchMSA = vi.mocked(launchMSA)
34
39
  const mockFetchTaxonomy = vi.mocked(fetchTaxonomyInfo)
35
40
 
@@ -245,3 +250,115 @@ describe('the Accession that drives the domain overlay', () => {
245
250
  expect(queryMetadata(result).Accession).toBeUndefined()
246
251
  })
247
252
  })
253
+
254
+ // The second source. What is under test is the dispatch and what the PANTHER
255
+ // result becomes on the query row -- the rows themselves are shaped upstream,
256
+ // and the tail of the launch (labels, aligner, metadata) is the same code the
257
+ // NCBI tests above already cover.
258
+ describe('the PANTHER source', () => {
259
+ const YEAST = 559292
260
+ const found = {
261
+ matched: 'CDC28',
262
+ query: {
263
+ code: 'YEAST',
264
+ accession: 'P00546',
265
+ geneRef: 'SGD=S000000364',
266
+ sequence: 'MSGELANYKRLEKVGEGTYGVVYKA',
267
+ },
268
+ rows: [
269
+ {
270
+ taxId: HUMAN,
271
+ label: 'human',
272
+ scientificName: 'Homo sapiens',
273
+ commonName: 'human',
274
+ geneId: 'HGNC=1771',
275
+ protein: 'P24941',
276
+ sequence: 'MENFQKVEKIGEGTYGVVYKARNK',
277
+ },
278
+ ] as OrthologRow[],
279
+ }
280
+
281
+ beforeEach(() => {
282
+ mockFetchPanther.mockResolvedValue(found)
283
+ mockFetchTaxonomy.mockResolvedValue(
284
+ new Map([[YEAST, { sciname: 'Saccharomyces cerevisiae' }]]),
285
+ )
286
+ })
287
+
288
+ test('source omitted is NCBI, so an old launch never reaches PANTHER', async () => {
289
+ await doLaunchOrthologs({ self: makeModel(params()) })
290
+ expect(mockFetchPanther).not.toHaveBeenCalled()
291
+ expect(mockResolveGeneId).toHaveBeenCalled()
292
+ })
293
+
294
+ test('source panther asks PANTHER with the same species semantics, and skips NCBI', async () => {
295
+ await doLaunchOrthologs({
296
+ self: makeModel({
297
+ taxId: YEAST,
298
+ source: 'panther',
299
+ geneCandidates: ['CDC28'],
300
+ msaAlgorithm: 'clustalo',
301
+ taxa: [HUMAN, YEAST],
302
+ maxSpecies: 7,
303
+ }),
304
+ })
305
+ expect(mockResolveGeneId).not.toHaveBeenCalled()
306
+ expect(mockFetchRows).not.toHaveBeenCalled()
307
+ const { candidates, taxId, taxa, exclude, limit } =
308
+ mockFetchPanther.mock.calls[0]![0]
309
+ expect(candidates).toEqual(['CDC28'])
310
+ expect(taxId).toBe(YEAST)
311
+ expect([...taxa!]).toEqual([HUMAN, YEAST])
312
+ expect(exclude).toBe(YEAST)
313
+ expect(limit).toBe(7)
314
+ })
315
+
316
+ test("the query row is PANTHER's own entry for the gene when no sequence was supplied, and carries its UniProt accession for the domain overlay", async () => {
317
+ const result = await doLaunchOrthologs({
318
+ self: makeModel({
319
+ taxId: YEAST,
320
+ source: 'panther',
321
+ geneCandidates: ['CDC28'],
322
+ msaAlgorithm: 'clustalo',
323
+ }),
324
+ })
325
+ expect(queryRowName()).toBe('Saccharomyces_cerevisiae_query')
326
+ expect(queryRowSent()).toBe(found.query.sequence)
327
+ expect(queryMetadata(result)).toEqual({
328
+ 'Gene ID': 'SGD=S000000364',
329
+ Accession: 'P00546',
330
+ })
331
+ expect(JSON.parse(result.treeMetadata).human).toMatchObject({
332
+ Accession: 'P24941',
333
+ 'Gene ID': 'HGNC=1771',
334
+ })
335
+ })
336
+
337
+ test('a supplied sequence still wins, and a different isoform earns no Accession', async () => {
338
+ const result = await doLaunchOrthologs({
339
+ self: makeModel({
340
+ taxId: YEAST,
341
+ source: 'panther',
342
+ geneCandidates: ['CDC28'],
343
+ msaAlgorithm: 'clustalo',
344
+ proteinSequence: 'MDIFFERENTISOFORM',
345
+ }),
346
+ })
347
+ expect(queryRowSent()).toBe('MDIFFERENTISOFORM')
348
+ expect(queryMetadata(result).Accession).toBeUndefined()
349
+ })
350
+
351
+ test('names PANTHER when it has no protein for the query row', async () => {
352
+ mockFetchPanther.mockResolvedValue({ ...found, query: undefined })
353
+ await expect(
354
+ doLaunchOrthologs({
355
+ self: makeModel({
356
+ taxId: YEAST,
357
+ source: 'panther',
358
+ geneCandidates: ['CDC28'],
359
+ msaAlgorithm: 'clustalo',
360
+ }),
361
+ }),
362
+ ).rejects.toThrow(/PANTHER returned no representative protein/)
363
+ })
364
+ })