jbrowse-plugin-msaview 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/AddHighlightModel/GenomeMouseoverHighlight.js +1 -1
- package/dist/AddHighlightModel/MsaToGenomeHighlight.js +1 -1
- package/dist/AddHighlightModel/index.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.js +61 -17
- package/dist/LaunchMsaView/components/BlastQuery/BlastManualPanel.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/BlastPanel.js +2 -2
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.d.ts +12 -0
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.js +21 -2
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.d.ts +1 -0
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.js +29 -0
- package/dist/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/consts.d.ts +28 -0
- package/dist/LaunchMsaView/components/BlastQuery/consts.js +21 -0
- package/dist/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.js +1 -1
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +9 -4
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.d.ts +9 -0
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.js +20 -0
- package/dist/LaunchMsaView/components/SubmitCancelActions.test.js +1 -1
- package/dist/LaunchMsaView/components/useFeatureSequence.js +1 -1
- package/dist/LaunchMsaView/detectQueryRow.d.ts +15 -2
- package/dist/LaunchMsaView/detectQueryRow.js +20 -21
- package/dist/LaunchMsaView/detectQueryRow.test.js +15 -15
- package/dist/LaunchMsaView/useQueryRowName.js +5 -8
- package/dist/MsaViewPanel/afterCreateAutoruns.js +2 -2
- package/dist/MsaViewPanel/components/ErrorBoundary.d.ts +2 -2
- package/dist/MsaViewPanel/components/JobLink.js +7 -1
- package/dist/MsaViewPanel/components/LaunchProgress.d.ts +17 -0
- package/dist/MsaViewPanel/components/LaunchProgress.js +41 -0
- package/dist/MsaViewPanel/components/MsaViewPanel.js +6 -3
- package/dist/MsaViewPanel/components/MsaViewPanel.test.d.ts +1 -0
- package/dist/MsaViewPanel/components/MsaViewPanel.test.js +68 -0
- package/dist/MsaViewPanel/doLaunchBlast.d.ts +1 -1
- package/dist/MsaViewPanel/doLaunchBlast.js +84 -52
- package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +7 -5
- package/dist/MsaViewPanel/doLaunchOrthologs.js +64 -30
- package/dist/MsaViewPanel/doLaunchOrthologs.test.js +106 -1
- package/dist/MsaViewPanel/genomeToMSA.js +4 -2
- package/dist/MsaViewPanel/genomeToMSA.test.js +34 -0
- package/dist/MsaViewPanel/model.d.ts +41 -11
- package/dist/MsaViewPanel/model.js +6 -0
- package/dist/MsaViewPanel/observeProteinHighlights.test.js +11 -0
- package/dist/MsaViewPanel/syncGenomeHoverToMsaColumn.test.js +1 -0
- package/dist/MsaViewPanel/util.d.ts +18 -0
- package/dist/MsaViewPanel/util.js +17 -0
- package/dist/jbrowse-plugin-msaview.umd.production.min.js +47 -35
- package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
- package/dist/utils/blastCache.d.ts +10 -6
- package/dist/utils/blastCache.js +15 -3
- package/dist/utils/ebiBlast.d.ts +1 -1
- package/dist/utils/msa.d.ts +12 -0
- package/dist/utils/msa.js +35 -12
- package/dist/utils/msaRows.d.ts +31 -0
- package/dist/utils/msaRows.js +67 -0
- package/dist/utils/pantherOrthologs.d.ts +79 -0
- package/dist/utils/pantherOrthologs.js +262 -0
- package/dist/utils/phmmer.d.ts +53 -0
- package/dist/utils/phmmer.js +118 -0
- package/dist/utils/taxonomyNames.d.ts +1 -1
- package/dist/utils/taxonomyNames.js +6 -1
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -21
- package/src/AddHighlightModel/GenomeMouseoverHighlight.tsx +1 -1
- package/src/AddHighlightModel/MsaToGenomeHighlight.tsx +1 -1
- package/src/AddHighlightModel/index.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.tsx +88 -30
- package/src/LaunchMsaView/components/BlastQuery/BlastManualPanel.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/BlastPanel.tsx +4 -4
- package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.ts +50 -0
- package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.tsx +23 -3
- package/src/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/consts.ts +40 -0
- package/src/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.tsx +1 -1
- package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +21 -5
- package/src/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.tsx +52 -0
- package/src/LaunchMsaView/components/SubmitCancelActions.test.tsx +1 -1
- package/src/LaunchMsaView/components/useFeatureSequence.ts +1 -1
- package/src/LaunchMsaView/detectQueryRow.test.ts +17 -15
- package/src/LaunchMsaView/detectQueryRow.ts +34 -23
- package/src/LaunchMsaView/useQueryRowName.ts +6 -9
- package/src/MsaViewPanel/afterCreateAutoruns.ts +2 -2
- package/src/MsaViewPanel/components/ErrorBoundary.tsx +2 -1
- package/src/MsaViewPanel/components/JobLink.tsx +7 -2
- package/src/MsaViewPanel/components/LaunchProgress.tsx +62 -0
- package/src/MsaViewPanel/components/MsaViewPanel.test.tsx +83 -0
- package/src/MsaViewPanel/components/MsaViewPanel.tsx +7 -4
- package/src/MsaViewPanel/doLaunchBlast.ts +127 -69
- package/src/MsaViewPanel/doLaunchOrthologs.test.ts +119 -2
- package/src/MsaViewPanel/doLaunchOrthologs.ts +100 -38
- package/src/MsaViewPanel/genomeToMSA.test.ts +37 -0
- package/src/MsaViewPanel/genomeToMSA.ts +6 -2
- package/src/MsaViewPanel/model.ts +38 -5
- package/src/MsaViewPanel/observeProteinHighlights.test.ts +13 -0
- package/src/MsaViewPanel/syncGenomeHoverToMsaColumn.test.ts +1 -0
- package/src/MsaViewPanel/util.ts +18 -0
- package/src/utils/blastCache.ts +33 -12
- package/src/utils/ebiBlast.ts +1 -1
- package/src/utils/msa.ts +43 -12
- package/src/utils/msaRows.ts +95 -0
- package/src/utils/pantherOrthologs.ts +399 -0
- package/src/utils/phmmer.ts +174 -0
- package/src/utils/taxonomyNames.ts +6 -1
- package/src/version.ts +1 -1
- package/dist/MsaViewPanel/components/LoadingBLAST.d.ts +0 -6
- package/dist/MsaViewPanel/components/LoadingBLAST.js +0 -26
- package/src/MsaViewPanel/components/LoadingBLAST.tsx +0 -48
|
@@ -2,45 +2,112 @@ import { makeId, strip } from '../LaunchMsaView/components/util'
|
|
|
2
2
|
import { cleanProteinSequence } from '../LaunchMsaView/util'
|
|
3
3
|
import { saveBlastResult } from '../utils/blastCache'
|
|
4
4
|
import { queryEbiBlast } from '../utils/ebiBlast'
|
|
5
|
-
import { launchMSA } from '../utils/msa'
|
|
5
|
+
import { launchMSA, launchTree } from '../utils/msa'
|
|
6
|
+
import { buildPhmmerMsa, buildRowMetadata } from '../utils/msaRows'
|
|
7
|
+
import { queryPhmmer } from '../utils/phmmer'
|
|
6
8
|
import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
|
|
7
9
|
|
|
10
|
+
import type {
|
|
11
|
+
BlastDatabase,
|
|
12
|
+
MsaAlgorithm,
|
|
13
|
+
PhmmerDatabase,
|
|
14
|
+
} from '../LaunchMsaView/components/BlastQuery/consts'
|
|
8
15
|
import type { JBrowsePluginMsaViewModel } from './model'
|
|
9
|
-
|
|
10
|
-
|
|
16
|
+
|
|
17
|
+
type TreeMetadata = Record<string, Record<string, string>>
|
|
11
18
|
|
|
12
19
|
export async function doLaunchBlast({
|
|
13
20
|
self,
|
|
14
21
|
}: {
|
|
15
22
|
self: JBrowsePluginMsaViewModel
|
|
16
23
|
}) {
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
const
|
|
24
|
+
// kept whole rather than destructured: the database's type depends on
|
|
25
|
+
// searchProgram, and pulling the two apart loses the link between them
|
|
26
|
+
const params = self.blastParams!
|
|
27
|
+
const { selectedTranscript } = params
|
|
28
|
+
const cleanedSeq = cleanProteinSequence(params.proteinSequence)
|
|
20
29
|
|
|
21
30
|
const onProgress = (arg: string) => {
|
|
22
31
|
self.setProgress(arg)
|
|
23
32
|
}
|
|
33
|
+
// publish the job id before the first poll so the view can link out while the
|
|
34
|
+
// job is still running
|
|
35
|
+
const onRid = (r: string) => {
|
|
36
|
+
self.setRid(r)
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const { msa, tree, treeMetadata, rid } =
|
|
40
|
+
params.searchProgram === 'phmmer'
|
|
41
|
+
? await runPhmmer({
|
|
42
|
+
query: cleanedSeq,
|
|
43
|
+
database: params.blastDatabase,
|
|
44
|
+
onProgress,
|
|
45
|
+
onRid,
|
|
46
|
+
})
|
|
47
|
+
: await runBlast({
|
|
48
|
+
query: cleanedSeq,
|
|
49
|
+
blastDatabase: params.blastDatabase,
|
|
50
|
+
msaAlgorithm: params.msaAlgorithm,
|
|
51
|
+
onProgress,
|
|
52
|
+
onRid,
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
const treeMetadataJson = JSON.stringify(treeMetadata)
|
|
56
|
+
|
|
57
|
+
await saveBlastResult({
|
|
58
|
+
proteinSequence: cleanedSeq,
|
|
59
|
+
blastDatabase: params.blastDatabase,
|
|
60
|
+
msaAlgorithm: params.msaAlgorithm,
|
|
61
|
+
searchProgram: params.searchProgram,
|
|
62
|
+
msa,
|
|
63
|
+
tree,
|
|
64
|
+
treeMetadata: treeMetadataJson,
|
|
65
|
+
rid,
|
|
66
|
+
geneId: selectedTranscript?.get('parentId'),
|
|
67
|
+
transcriptId: selectedTranscript?.id(),
|
|
68
|
+
transcriptName:
|
|
69
|
+
selectedTranscript?.get('name') ?? selectedTranscript?.get('id'),
|
|
70
|
+
geneName:
|
|
71
|
+
selectedTranscript?.get('gene_name') ??
|
|
72
|
+
selectedTranscript?.get('parentId'),
|
|
73
|
+
})
|
|
24
74
|
|
|
75
|
+
return { msa, tree, treeMetadata: treeMetadataJson }
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* BLAST returns each hit already aligned to the query, but pairwise and one hit
|
|
80
|
+
* at a time, so the alignments are stripped back off and every hit is realigned
|
|
81
|
+
* together by a dedicated aligner.
|
|
82
|
+
*/
|
|
83
|
+
async function runBlast({
|
|
84
|
+
query,
|
|
85
|
+
blastDatabase,
|
|
86
|
+
msaAlgorithm,
|
|
87
|
+
onProgress,
|
|
88
|
+
onRid,
|
|
89
|
+
}: {
|
|
90
|
+
query: string
|
|
91
|
+
blastDatabase: BlastDatabase
|
|
92
|
+
msaAlgorithm: MsaAlgorithm
|
|
93
|
+
onProgress: (arg: string) => void
|
|
94
|
+
onRid: (arg: string) => void
|
|
95
|
+
}) {
|
|
25
96
|
const { hits, rid } = await queryEbiBlast({
|
|
26
|
-
query
|
|
97
|
+
query,
|
|
27
98
|
blastDatabase,
|
|
28
99
|
onProgress,
|
|
29
|
-
|
|
30
|
-
// the job is still running
|
|
31
|
-
onRid: r => {
|
|
32
|
-
self.setRid(r)
|
|
33
|
-
},
|
|
100
|
+
onRid,
|
|
34
101
|
})
|
|
35
102
|
|
|
36
|
-
|
|
37
|
-
const
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
const treeMetadata: Record<string, Record<string, string>> = {}
|
|
103
|
+
onProgress('Fetching species taxonomy info...')
|
|
104
|
+
const taxonomyInfo = await fetchTaxonomyInfo(
|
|
105
|
+
hits
|
|
106
|
+
.map(h => h.description[0]?.taxid)
|
|
107
|
+
.filter((t): t is number => t !== undefined),
|
|
108
|
+
)
|
|
43
109
|
|
|
110
|
+
const treeMetadata: TreeMetadata = {}
|
|
44
111
|
const sequences = hits.map(h => {
|
|
45
112
|
const desc = h.description[0] ?? {
|
|
46
113
|
accession: 'unknown',
|
|
@@ -48,66 +115,57 @@ export async function doLaunchBlast({
|
|
|
48
115
|
sciname: 'unknown',
|
|
49
116
|
}
|
|
50
117
|
const rowName = makeId(desc, taxonomyInfo)
|
|
51
|
-
const seq = strip(h.hsps[0]?.hseq ?? '')
|
|
52
|
-
|
|
53
118
|
treeMetadata[rowName] = buildRowMetadata(desc, taxonomyInfo)
|
|
54
|
-
|
|
55
|
-
return `>${rowName}\n${seq}`
|
|
119
|
+
return `>${rowName}\n${strip(h.hsps[0]?.hseq ?? '')}`
|
|
56
120
|
})
|
|
57
121
|
|
|
58
122
|
const result = await launchMSA({
|
|
59
123
|
algorithm: msaAlgorithm,
|
|
60
|
-
sequence: [`>QUERY\n${
|
|
124
|
+
sequence: [`>QUERY\n${query}`, ...sequences].join('\n'),
|
|
61
125
|
onProgress,
|
|
62
126
|
})
|
|
127
|
+
return { ...result, treeMetadata, rid }
|
|
128
|
+
}
|
|
63
129
|
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
130
|
+
/**
|
|
131
|
+
* phmmer aligns every hit to a profile of the query as it searches, so its own
|
|
132
|
+
* output is the MSA and there is no realignment step — the hits keep the
|
|
133
|
+
* placement HMMER gave them, and the query row is derived from the alignment's
|
|
134
|
+
* match columns rather than being aligned back in afterwards. That leaves no
|
|
135
|
+
* aligner run to take a tree from, so the tree is built from this alignment.
|
|
136
|
+
*/
|
|
137
|
+
async function runPhmmer({
|
|
138
|
+
query,
|
|
139
|
+
database,
|
|
140
|
+
onProgress,
|
|
141
|
+
onRid,
|
|
142
|
+
}: {
|
|
143
|
+
query: string
|
|
144
|
+
database: PhmmerDatabase
|
|
145
|
+
onProgress: (arg: string) => void
|
|
146
|
+
onRid: (arg: string) => void
|
|
147
|
+
}) {
|
|
148
|
+
const { rows, queryRow, rid } = await queryPhmmer({
|
|
149
|
+
query,
|
|
150
|
+
database,
|
|
151
|
+
onProgress,
|
|
152
|
+
onRid,
|
|
81
153
|
})
|
|
82
154
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
function buildRowMetadata(
|
|
90
|
-
desc: BlastHitDescription,
|
|
91
|
-
taxonomyInfo: Map<number, TaxonomyInfo>,
|
|
92
|
-
) {
|
|
93
|
-
const metadata: Record<string, string> = {}
|
|
94
|
-
const taxInfo = desc.taxid ? taxonomyInfo.get(desc.taxid) : undefined
|
|
155
|
+
onProgress('Fetching species taxonomy info...')
|
|
156
|
+
const taxonomyInfo = await fetchTaxonomyInfo(
|
|
157
|
+
rows.map(r => r.taxid).filter((t): t is number => t !== undefined),
|
|
158
|
+
)
|
|
95
159
|
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
metadata.ID = desc.id
|
|
107
|
-
}
|
|
108
|
-
if (desc.title) {
|
|
109
|
-
metadata.Description = desc.title
|
|
160
|
+
const { msa, treeMetadata } = buildPhmmerMsa({
|
|
161
|
+
rows,
|
|
162
|
+
queryRow,
|
|
163
|
+
taxonomyInfo,
|
|
164
|
+
})
|
|
165
|
+
return {
|
|
166
|
+
msa,
|
|
167
|
+
tree: await launchTree({ alignment: msa, onProgress }),
|
|
168
|
+
treeMetadata,
|
|
169
|
+
rid,
|
|
110
170
|
}
|
|
111
|
-
|
|
112
|
-
return metadata
|
|
113
171
|
}
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
import { beforeEach, describe, expect, test, vi } from 'vitest'
|
|
2
2
|
|
|
3
|
-
import { doLaunchOrthologs } from './doLaunchOrthologs'
|
|
4
3
|
import { launchMSA } from '../utils/msa'
|
|
5
4
|
import {
|
|
6
5
|
defaultMaxSpecies,
|
|
@@ -8,10 +7,12 @@ import {
|
|
|
8
7
|
fetchProteinForGene,
|
|
9
8
|
resolveGeneId,
|
|
10
9
|
} from '../utils/ncbiOrthologs'
|
|
10
|
+
import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
|
|
11
11
|
import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
|
|
12
|
+
import { doLaunchOrthologs } from './doLaunchOrthologs'
|
|
12
13
|
|
|
13
|
-
import type { JBrowsePluginMsaViewModel } from './model'
|
|
14
14
|
import type { OrthologRow } from '../utils/ncbiOrthologs'
|
|
15
|
+
import type { JBrowsePluginMsaViewModel } from './model'
|
|
15
16
|
|
|
16
17
|
// Every network call is mocked and nothing else is. What is under test is the
|
|
17
18
|
// argument shaping either side of those calls -- which species get asked for,
|
|
@@ -24,12 +25,16 @@ vi.mock('../utils/ncbiOrthologs', async importOriginal => ({
|
|
|
24
25
|
fetchProteinForGene: vi.fn(),
|
|
25
26
|
fetchOrthologRows: vi.fn(),
|
|
26
27
|
}))
|
|
28
|
+
vi.mock('../utils/pantherOrthologs', () => ({
|
|
29
|
+
fetchPantherOrthologs: vi.fn(),
|
|
30
|
+
}))
|
|
27
31
|
vi.mock('../utils/msa', () => ({ launchMSA: vi.fn() }))
|
|
28
32
|
vi.mock('../utils/taxonomyNames', () => ({ fetchTaxonomyInfo: vi.fn() }))
|
|
29
33
|
|
|
30
34
|
const mockResolveGeneId = vi.mocked(resolveGeneId)
|
|
31
35
|
const mockFetchProtein = vi.mocked(fetchProteinForGene)
|
|
32
36
|
const mockFetchRows = vi.mocked(fetchOrthologRows)
|
|
37
|
+
const mockFetchPanther = vi.mocked(fetchPantherOrthologs)
|
|
33
38
|
const mockLaunchMSA = vi.mocked(launchMSA)
|
|
34
39
|
const mockFetchTaxonomy = vi.mocked(fetchTaxonomyInfo)
|
|
35
40
|
|
|
@@ -245,3 +250,115 @@ describe('the Accession that drives the domain overlay', () => {
|
|
|
245
250
|
expect(queryMetadata(result).Accession).toBeUndefined()
|
|
246
251
|
})
|
|
247
252
|
})
|
|
253
|
+
|
|
254
|
+
// The second source. What is under test is the dispatch and what the PANTHER
|
|
255
|
+
// result becomes on the query row -- the rows themselves are shaped upstream,
|
|
256
|
+
// and the tail of the launch (labels, aligner, metadata) is the same code the
|
|
257
|
+
// NCBI tests above already cover.
|
|
258
|
+
describe('the PANTHER source', () => {
|
|
259
|
+
const YEAST = 559292
|
|
260
|
+
const found = {
|
|
261
|
+
matched: 'CDC28',
|
|
262
|
+
query: {
|
|
263
|
+
code: 'YEAST',
|
|
264
|
+
accession: 'P00546',
|
|
265
|
+
geneRef: 'SGD=S000000364',
|
|
266
|
+
sequence: 'MSGELANYKRLEKVGEGTYGVVYKA',
|
|
267
|
+
},
|
|
268
|
+
rows: [
|
|
269
|
+
{
|
|
270
|
+
taxId: HUMAN,
|
|
271
|
+
label: 'human',
|
|
272
|
+
scientificName: 'Homo sapiens',
|
|
273
|
+
commonName: 'human',
|
|
274
|
+
geneId: 'HGNC=1771',
|
|
275
|
+
protein: 'P24941',
|
|
276
|
+
sequence: 'MENFQKVEKIGEGTYGVVYKARNK',
|
|
277
|
+
},
|
|
278
|
+
] as OrthologRow[],
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
beforeEach(() => {
|
|
282
|
+
mockFetchPanther.mockResolvedValue(found)
|
|
283
|
+
mockFetchTaxonomy.mockResolvedValue(
|
|
284
|
+
new Map([[YEAST, { sciname: 'Saccharomyces cerevisiae' }]]),
|
|
285
|
+
)
|
|
286
|
+
})
|
|
287
|
+
|
|
288
|
+
test('source omitted is NCBI, so an old launch never reaches PANTHER', async () => {
|
|
289
|
+
await doLaunchOrthologs({ self: makeModel(params()) })
|
|
290
|
+
expect(mockFetchPanther).not.toHaveBeenCalled()
|
|
291
|
+
expect(mockResolveGeneId).toHaveBeenCalled()
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
test('source panther asks PANTHER with the same species semantics, and skips NCBI', async () => {
|
|
295
|
+
await doLaunchOrthologs({
|
|
296
|
+
self: makeModel({
|
|
297
|
+
taxId: YEAST,
|
|
298
|
+
source: 'panther',
|
|
299
|
+
geneCandidates: ['CDC28'],
|
|
300
|
+
msaAlgorithm: 'clustalo',
|
|
301
|
+
taxa: [HUMAN, YEAST],
|
|
302
|
+
maxSpecies: 7,
|
|
303
|
+
}),
|
|
304
|
+
})
|
|
305
|
+
expect(mockResolveGeneId).not.toHaveBeenCalled()
|
|
306
|
+
expect(mockFetchRows).not.toHaveBeenCalled()
|
|
307
|
+
const { candidates, taxId, taxa, exclude, limit } =
|
|
308
|
+
mockFetchPanther.mock.calls[0]![0]
|
|
309
|
+
expect(candidates).toEqual(['CDC28'])
|
|
310
|
+
expect(taxId).toBe(YEAST)
|
|
311
|
+
expect([...taxa!]).toEqual([HUMAN, YEAST])
|
|
312
|
+
expect(exclude).toBe(YEAST)
|
|
313
|
+
expect(limit).toBe(7)
|
|
314
|
+
})
|
|
315
|
+
|
|
316
|
+
test("the query row is PANTHER's own entry for the gene when no sequence was supplied, and carries its UniProt accession for the domain overlay", async () => {
|
|
317
|
+
const result = await doLaunchOrthologs({
|
|
318
|
+
self: makeModel({
|
|
319
|
+
taxId: YEAST,
|
|
320
|
+
source: 'panther',
|
|
321
|
+
geneCandidates: ['CDC28'],
|
|
322
|
+
msaAlgorithm: 'clustalo',
|
|
323
|
+
}),
|
|
324
|
+
})
|
|
325
|
+
expect(queryRowName()).toBe('Saccharomyces_cerevisiae_query')
|
|
326
|
+
expect(queryRowSent()).toBe(found.query.sequence)
|
|
327
|
+
expect(queryMetadata(result)).toEqual({
|
|
328
|
+
'Gene ID': 'SGD=S000000364',
|
|
329
|
+
Accession: 'P00546',
|
|
330
|
+
})
|
|
331
|
+
expect(JSON.parse(result.treeMetadata).human).toMatchObject({
|
|
332
|
+
Accession: 'P24941',
|
|
333
|
+
'Gene ID': 'HGNC=1771',
|
|
334
|
+
})
|
|
335
|
+
})
|
|
336
|
+
|
|
337
|
+
test('a supplied sequence still wins, and a different isoform earns no Accession', async () => {
|
|
338
|
+
const result = await doLaunchOrthologs({
|
|
339
|
+
self: makeModel({
|
|
340
|
+
taxId: YEAST,
|
|
341
|
+
source: 'panther',
|
|
342
|
+
geneCandidates: ['CDC28'],
|
|
343
|
+
msaAlgorithm: 'clustalo',
|
|
344
|
+
proteinSequence: 'MDIFFERENTISOFORM',
|
|
345
|
+
}),
|
|
346
|
+
})
|
|
347
|
+
expect(queryRowSent()).toBe('MDIFFERENTISOFORM')
|
|
348
|
+
expect(queryMetadata(result).Accession).toBeUndefined()
|
|
349
|
+
})
|
|
350
|
+
|
|
351
|
+
test('names PANTHER when it has no protein for the query row', async () => {
|
|
352
|
+
mockFetchPanther.mockResolvedValue({ ...found, query: undefined })
|
|
353
|
+
await expect(
|
|
354
|
+
doLaunchOrthologs({
|
|
355
|
+
self: makeModel({
|
|
356
|
+
taxId: YEAST,
|
|
357
|
+
source: 'panther',
|
|
358
|
+
geneCandidates: ['CDC28'],
|
|
359
|
+
msaAlgorithm: 'clustalo',
|
|
360
|
+
}),
|
|
361
|
+
}),
|
|
362
|
+
).rejects.toThrow(/PANTHER returned no representative protein/)
|
|
363
|
+
})
|
|
364
|
+
})
|
|
@@ -6,21 +6,37 @@ import {
|
|
|
6
6
|
fetchProteinForGene,
|
|
7
7
|
resolveGeneId,
|
|
8
8
|
} from '../utils/ncbiOrthologs'
|
|
9
|
+
import { fetchPantherOrthologs } from '../utils/pantherOrthologs'
|
|
9
10
|
import { fetchTaxonomyInfo } from '../utils/taxonomyNames'
|
|
10
11
|
|
|
11
|
-
import type { JBrowsePluginMsaViewModel } from './model'
|
|
12
12
|
import type { OrthologRow } from '../utils/ncbiOrthologs'
|
|
13
|
+
import type { JBrowsePluginMsaViewModel } from './model'
|
|
14
|
+
|
|
15
|
+
interface Representative {
|
|
16
|
+
accession: string
|
|
17
|
+
sequence: string
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** What either source hands the shared tail of the launch. */
|
|
21
|
+
interface FoundOrthologs {
|
|
22
|
+
/** the query gene's id at the source: an NCBI GeneID, or PANTHER's gene xref */
|
|
23
|
+
geneId: string
|
|
24
|
+
representative: Representative | undefined
|
|
25
|
+
rows: OrthologRow[]
|
|
26
|
+
}
|
|
13
27
|
|
|
14
28
|
/**
|
|
15
29
|
* The no-search-job alternative to doLaunchBlast.
|
|
16
30
|
*
|
|
17
31
|
* BLAST spends 10+ minutes answering "what looks like this sequence" and
|
|
18
|
-
* returns a redundant, accession-labelled hit list. This asks
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
32
|
+
* returns a redundant, accession-labelled hit list. This asks the question the
|
|
33
|
+
* alignment actually wants — "what is this gene's ortholog in each species" —
|
|
34
|
+
* which NCBI and PANTHER have already computed, so the lookup returns in
|
|
35
|
+
* seconds and only the EBI alignment (~10s) costs real time. `source` picks
|
|
36
|
+
* which of the two answers: NCBI for vertebrates and insects, PANTHER for
|
|
37
|
+
* everything else (yeast, worm, plants, and a fly gene's vertebrate relatives).
|
|
22
38
|
*
|
|
23
|
-
* The query row is the user's OWN selected transcript, not
|
|
39
|
+
* The query row is the user's OWN selected transcript, not the source's
|
|
24
40
|
* representative protein for the query species, because `connectedFeature`
|
|
25
41
|
* maps genome coordinates through that row — swapping in a different isoform
|
|
26
42
|
* would silently break the genome<->MSA linkage. The query species is therefore
|
|
@@ -38,49 +54,43 @@ export async function doLaunchOrthologs({
|
|
|
38
54
|
geneCandidates,
|
|
39
55
|
msaAlgorithm,
|
|
40
56
|
proteinSequence,
|
|
57
|
+
source = 'ncbi',
|
|
41
58
|
} = self.orthologParams!
|
|
42
59
|
|
|
43
60
|
const onProgress = (arg: string) => {
|
|
44
61
|
self.setProgress(arg)
|
|
45
62
|
}
|
|
46
63
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
64
|
+
const request = {
|
|
65
|
+
taxId,
|
|
66
|
+
geneCandidates,
|
|
67
|
+
taxa: taxa ? new Set(taxa) : undefined,
|
|
68
|
+
// the query species is represented by the query row below
|
|
69
|
+
exclude: taxId,
|
|
70
|
+
limit: maxSpecies,
|
|
71
|
+
onProgress,
|
|
53
72
|
}
|
|
73
|
+
const { geneId, representative, rows } =
|
|
74
|
+
source === 'panther'
|
|
75
|
+
? await findPantherOrthologs(request)
|
|
76
|
+
: await findNcbiOrthologs(request)
|
|
54
77
|
|
|
55
78
|
// The query row. The dialog always supplies it — it is the user's OWN
|
|
56
79
|
// selected transcript, which is what makes `connectedFeature` map genome
|
|
57
80
|
// coordinates through this row. A launch that has no transcript to translate
|
|
58
|
-
// (a session spec naming only a gene) falls back to
|
|
59
|
-
// protein for the resolved gene, which is the same choice
|
|
60
|
-
// other row, so the alignment is the one
|
|
61
|
-
|
|
81
|
+
// (a session spec naming only a gene) falls back to the source's
|
|
82
|
+
// representative protein for the resolved gene, which is the same choice
|
|
83
|
+
// made for every other row, so the alignment is the one the source would
|
|
84
|
+
// build for that gene.
|
|
62
85
|
const cleanedSeq = proteinSequence
|
|
63
86
|
? cleanProteinSequence(proteinSequence)
|
|
64
87
|
: representative?.sequence
|
|
65
88
|
if (!cleanedSeq) {
|
|
66
89
|
throw new Error(
|
|
67
|
-
`No query protein: none was supplied and NCBI returned no representative protein for gene ${
|
|
90
|
+
`No query protein: none was supplied and ${source === 'panther' ? 'PANTHER' : 'NCBI'} returned no representative protein for gene ${geneId}.`,
|
|
68
91
|
)
|
|
69
92
|
}
|
|
70
93
|
|
|
71
|
-
// Every species NCBI has an ortholog for, when a launch names none, capped at
|
|
72
|
-
// maxSpecies. A launch that wants specific species lists them; one that just
|
|
73
|
-
// wants "this gene across species" gets NCBI's own order, which leads with the
|
|
74
|
-
// reference organisms.
|
|
75
|
-
const rows = await fetchOrthologRows({
|
|
76
|
-
geneId: resolved.geneId,
|
|
77
|
-
taxa: taxa ? new Set(taxa) : undefined,
|
|
78
|
-
// the query species is represented by the query row above
|
|
79
|
-
exclude: taxId,
|
|
80
|
-
limit: maxSpecies,
|
|
81
|
-
onProgress,
|
|
82
|
-
})
|
|
83
|
-
|
|
84
94
|
// The query row is named for its species like every other row, with a suffix
|
|
85
95
|
// marking it as the one the genome view is linked to. A bare `QUERY` among
|
|
86
96
|
// ninety-nine named species reads as a row whose species failed to resolve,
|
|
@@ -96,12 +106,7 @@ export async function doLaunchOrthologs({
|
|
|
96
106
|
self.setQuerySeqName(queryLabel)
|
|
97
107
|
|
|
98
108
|
const treeMetadata: Record<string, Record<string, string>> = {
|
|
99
|
-
[queryLabel]: buildQueryMetadata(
|
|
100
|
-
self,
|
|
101
|
-
resolved.geneId,
|
|
102
|
-
cleanedSeq,
|
|
103
|
-
representative,
|
|
104
|
-
),
|
|
109
|
+
[queryLabel]: buildQueryMetadata(self, geneId, cleanedSeq, representative),
|
|
105
110
|
}
|
|
106
111
|
for (const row of rows) {
|
|
107
112
|
treeMetadata[row.label] = buildRowMetadata(row)
|
|
@@ -122,6 +127,63 @@ export async function doLaunchOrthologs({
|
|
|
122
127
|
}
|
|
123
128
|
}
|
|
124
129
|
|
|
130
|
+
interface OrthologRequest {
|
|
131
|
+
taxId: number
|
|
132
|
+
geneCandidates: string[]
|
|
133
|
+
taxa: Set<number> | undefined
|
|
134
|
+
exclude: number
|
|
135
|
+
limit: number | undefined
|
|
136
|
+
onProgress: (arg: string) => void
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Every species NCBI has an ortholog for, when a launch names none, capped at
|
|
141
|
+
* `limit`. A launch that wants specific species lists them; one that just
|
|
142
|
+
* wants "this gene across species" gets NCBI's own order, which leads with the
|
|
143
|
+
* reference organisms.
|
|
144
|
+
*/
|
|
145
|
+
async function findNcbiOrthologs({
|
|
146
|
+
taxId,
|
|
147
|
+
geneCandidates,
|
|
148
|
+
onProgress,
|
|
149
|
+
...rest
|
|
150
|
+
}: OrthologRequest): Promise<FoundOrthologs> {
|
|
151
|
+
onProgress('Resolving gene at NCBI...')
|
|
152
|
+
const resolved = await resolveGeneId(geneCandidates, taxId)
|
|
153
|
+
if (!resolved) {
|
|
154
|
+
throw new Error(
|
|
155
|
+
`Could not resolve any of ${geneCandidates.join(', ')} to an NCBI gene in taxon ${taxId}. Try the NCBI BLAST tab, which needs no gene identifier.`,
|
|
156
|
+
)
|
|
157
|
+
}
|
|
158
|
+
const representative = await fetchRepresentativeQueryProtein(resolved.geneId)
|
|
159
|
+
const rows = await fetchOrthologRows({
|
|
160
|
+
geneId: resolved.geneId,
|
|
161
|
+
onProgress,
|
|
162
|
+
...rest,
|
|
163
|
+
})
|
|
164
|
+
return { geneId: resolved.geneId, representative, rows }
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* One `matchortho` call resolves the gene, names its own UniProt entry and
|
|
169
|
+
* lists an ortholog per genome, so the representative protein needs no second
|
|
170
|
+
* lookup here.
|
|
171
|
+
*/
|
|
172
|
+
async function findPantherOrthologs({
|
|
173
|
+
geneCandidates,
|
|
174
|
+
...rest
|
|
175
|
+
}: OrthologRequest): Promise<FoundOrthologs> {
|
|
176
|
+
const found = await fetchPantherOrthologs({
|
|
177
|
+
candidates: geneCandidates,
|
|
178
|
+
...rest,
|
|
179
|
+
})
|
|
180
|
+
return {
|
|
181
|
+
geneId: found.query?.geneRef ?? found.matched,
|
|
182
|
+
representative: found.query,
|
|
183
|
+
rows: found.rows,
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
125
187
|
/**
|
|
126
188
|
* `<species>_query`, unique against the ortholog labels. Falls back to the bare
|
|
127
189
|
* marker when NCBI cannot name the taxon, which is a naming failure and must not
|
|
@@ -158,7 +220,7 @@ async function fetchRepresentativeQueryProtein(geneId: string) {
|
|
|
158
220
|
/**
|
|
159
221
|
* The query row carries an Accession — which is what drives the automatic CDD
|
|
160
222
|
* overlay (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains) —
|
|
161
|
-
* ONLY when its sequence is byte-identical to the
|
|
223
|
+
* ONLY when its sequence is byte-identical to the protein that accession
|
|
162
224
|
* names. Attaching it unconditionally would put every domain box at an offset
|
|
163
225
|
* whenever the user picked a non-representative isoform, which is a silently
|
|
164
226
|
* wrong figure rather than a missing one. A launch that took the representative
|
|
@@ -168,7 +230,7 @@ function buildQueryMetadata(
|
|
|
168
230
|
self: JBrowsePluginMsaViewModel,
|
|
169
231
|
geneId: string,
|
|
170
232
|
proteinSequence: string,
|
|
171
|
-
representative:
|
|
233
|
+
representative: Representative | undefined,
|
|
172
234
|
): Record<string, string> {
|
|
173
235
|
const transcript = self.orthologParams?.selectedTranscript
|
|
174
236
|
const metadata: Record<string, string> = { 'Gene ID': geneId }
|