jbrowse-plugin-msaview 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/AddHighlightModel/GenomeMouseoverHighlight.js +1 -1
- package/dist/AddHighlightModel/MsaToGenomeHighlight.js +1 -1
- package/dist/AddHighlightModel/index.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.js +61 -17
- package/dist/LaunchMsaView/components/BlastQuery/BlastManualPanel.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/BlastPanel.js +2 -2
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.d.ts +12 -0
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.js +21 -2
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.d.ts +1 -0
- package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.js +29 -0
- package/dist/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.js +1 -1
- package/dist/LaunchMsaView/components/BlastQuery/consts.d.ts +28 -0
- package/dist/LaunchMsaView/components/BlastQuery/consts.js +21 -0
- package/dist/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.js +1 -1
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +9 -4
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.d.ts +9 -0
- package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.js +20 -0
- package/dist/LaunchMsaView/components/SubmitCancelActions.test.js +1 -1
- package/dist/LaunchMsaView/components/useFeatureSequence.js +1 -1
- package/dist/LaunchMsaView/detectQueryRow.d.ts +15 -2
- package/dist/LaunchMsaView/detectQueryRow.js +20 -21
- package/dist/LaunchMsaView/detectQueryRow.test.js +15 -15
- package/dist/LaunchMsaView/useQueryRowName.js +5 -8
- package/dist/MsaViewPanel/afterCreateAutoruns.js +2 -2
- package/dist/MsaViewPanel/components/ErrorBoundary.d.ts +2 -2
- package/dist/MsaViewPanel/components/JobLink.js +7 -1
- package/dist/MsaViewPanel/components/LaunchProgress.d.ts +17 -0
- package/dist/MsaViewPanel/components/LaunchProgress.js +41 -0
- package/dist/MsaViewPanel/components/MsaViewPanel.js +6 -3
- package/dist/MsaViewPanel/components/MsaViewPanel.test.d.ts +1 -0
- package/dist/MsaViewPanel/components/MsaViewPanel.test.js +68 -0
- package/dist/MsaViewPanel/doLaunchBlast.d.ts +1 -1
- package/dist/MsaViewPanel/doLaunchBlast.js +84 -52
- package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +7 -5
- package/dist/MsaViewPanel/doLaunchOrthologs.js +64 -30
- package/dist/MsaViewPanel/doLaunchOrthologs.test.js +106 -1
- package/dist/MsaViewPanel/genomeToMSA.js +4 -2
- package/dist/MsaViewPanel/genomeToMSA.test.js +34 -0
- package/dist/MsaViewPanel/model.d.ts +41 -11
- package/dist/MsaViewPanel/model.js +6 -0
- package/dist/MsaViewPanel/observeProteinHighlights.test.js +11 -0
- package/dist/MsaViewPanel/syncGenomeHoverToMsaColumn.test.js +1 -0
- package/dist/MsaViewPanel/util.d.ts +18 -0
- package/dist/MsaViewPanel/util.js +17 -0
- package/dist/jbrowse-plugin-msaview.umd.production.min.js +47 -35
- package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
- package/dist/utils/blastCache.d.ts +10 -6
- package/dist/utils/blastCache.js +15 -3
- package/dist/utils/ebiBlast.d.ts +1 -1
- package/dist/utils/msa.d.ts +12 -0
- package/dist/utils/msa.js +35 -12
- package/dist/utils/msaRows.d.ts +31 -0
- package/dist/utils/msaRows.js +67 -0
- package/dist/utils/pantherOrthologs.d.ts +79 -0
- package/dist/utils/pantherOrthologs.js +262 -0
- package/dist/utils/phmmer.d.ts +53 -0
- package/dist/utils/phmmer.js +118 -0
- package/dist/utils/taxonomyNames.d.ts +1 -1
- package/dist/utils/taxonomyNames.js +6 -1
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +27 -21
- package/src/AddHighlightModel/GenomeMouseoverHighlight.tsx +1 -1
- package/src/AddHighlightModel/MsaToGenomeHighlight.tsx +1 -1
- package/src/AddHighlightModel/index.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.tsx +88 -30
- package/src/LaunchMsaView/components/BlastQuery/BlastManualPanel.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/BlastPanel.tsx +4 -4
- package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.ts +50 -0
- package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.tsx +23 -3
- package/src/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.tsx +1 -1
- package/src/LaunchMsaView/components/BlastQuery/consts.ts +40 -0
- package/src/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.tsx +1 -1
- package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +21 -5
- package/src/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.tsx +52 -0
- package/src/LaunchMsaView/components/SubmitCancelActions.test.tsx +1 -1
- package/src/LaunchMsaView/components/useFeatureSequence.ts +1 -1
- package/src/LaunchMsaView/detectQueryRow.test.ts +17 -15
- package/src/LaunchMsaView/detectQueryRow.ts +34 -23
- package/src/LaunchMsaView/useQueryRowName.ts +6 -9
- package/src/MsaViewPanel/afterCreateAutoruns.ts +2 -2
- package/src/MsaViewPanel/components/ErrorBoundary.tsx +2 -1
- package/src/MsaViewPanel/components/JobLink.tsx +7 -2
- package/src/MsaViewPanel/components/LaunchProgress.tsx +62 -0
- package/src/MsaViewPanel/components/MsaViewPanel.test.tsx +83 -0
- package/src/MsaViewPanel/components/MsaViewPanel.tsx +7 -4
- package/src/MsaViewPanel/doLaunchBlast.ts +127 -69
- package/src/MsaViewPanel/doLaunchOrthologs.test.ts +119 -2
- package/src/MsaViewPanel/doLaunchOrthologs.ts +100 -38
- package/src/MsaViewPanel/genomeToMSA.test.ts +37 -0
- package/src/MsaViewPanel/genomeToMSA.ts +6 -2
- package/src/MsaViewPanel/model.ts +38 -5
- package/src/MsaViewPanel/observeProteinHighlights.test.ts +13 -0
- package/src/MsaViewPanel/syncGenomeHoverToMsaColumn.test.ts +1 -0
- package/src/MsaViewPanel/util.ts +18 -0
- package/src/utils/blastCache.ts +33 -12
- package/src/utils/ebiBlast.ts +1 -1
- package/src/utils/msa.ts +43 -12
- package/src/utils/msaRows.ts +95 -0
- package/src/utils/pantherOrthologs.ts +399 -0
- package/src/utils/phmmer.ts +174 -0
- package/src/utils/taxonomyNames.ts +6 -1
- package/src/version.ts +1 -1
- package/dist/MsaViewPanel/components/LoadingBLAST.d.ts +0 -6
- package/dist/MsaViewPanel/components/LoadingBLAST.js +0 -26
- package/src/MsaViewPanel/components/LoadingBLAST.tsx +0 -48
|
@@ -1,15 +1,18 @@
|
|
|
1
|
-
import type { BlastDatabase, MsaAlgorithm } from '../LaunchMsaView/components/BlastQuery/consts';
|
|
1
|
+
import type { BlastDatabase, MsaAlgorithm, PhmmerDatabase, SearchProgram } from '../LaunchMsaView/components/BlastQuery/consts';
|
|
2
2
|
export interface CachedBlastResult {
|
|
3
3
|
id: string;
|
|
4
4
|
proteinSequence: string;
|
|
5
|
-
blastDatabase: BlastDatabase;
|
|
5
|
+
blastDatabase: BlastDatabase | PhmmerDatabase;
|
|
6
6
|
/**
|
|
7
7
|
* Only ever set on rows cached by a version that still queried NCBI, where
|
|
8
8
|
* the choice between blastp and quick-blastp was real. Kept so those rows
|
|
9
9
|
* still display; never written now.
|
|
10
10
|
*/
|
|
11
11
|
blastProgram?: string;
|
|
12
|
-
|
|
12
|
+
/** absent on rows cached before phmmer existed, which were all blastp */
|
|
13
|
+
searchProgram?: SearchProgram;
|
|
14
|
+
/** absent on phmmer rows, which are aligned by the search itself */
|
|
15
|
+
msaAlgorithm?: MsaAlgorithm;
|
|
13
16
|
msa: string;
|
|
14
17
|
tree: string;
|
|
15
18
|
treeMetadata: string;
|
|
@@ -20,10 +23,11 @@ export interface CachedBlastResult {
|
|
|
20
23
|
transcriptName?: string;
|
|
21
24
|
geneName?: string;
|
|
22
25
|
}
|
|
23
|
-
export declare function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }: {
|
|
26
|
+
export declare function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }: {
|
|
24
27
|
proteinSequence: string;
|
|
25
|
-
blastDatabase: BlastDatabase;
|
|
26
|
-
msaAlgorithm
|
|
28
|
+
blastDatabase: BlastDatabase | PhmmerDatabase;
|
|
29
|
+
msaAlgorithm?: MsaAlgorithm;
|
|
30
|
+
searchProgram?: SearchProgram;
|
|
27
31
|
msa: string;
|
|
28
32
|
tree: string;
|
|
29
33
|
treeMetadata: string;
|
package/dist/utils/blastCache.js
CHANGED
|
@@ -10,21 +10,33 @@ const getDB = createDbOpener(DB_NAME, DB_VERSION, (db, oldVersion) => {
|
|
|
10
10
|
db.createObjectStore(STORE_NAME, { keyPath: 'id' });
|
|
11
11
|
}
|
|
12
12
|
});
|
|
13
|
-
function createCacheKey(proteinSequence, blastDatabase, msaAlgorithm, transcriptId) {
|
|
13
|
+
function createCacheKey({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, transcriptId, }) {
|
|
14
14
|
const idPart = transcriptId ? `:${transcriptId}` : '';
|
|
15
|
+
// phmmer keys are prefixed and blastp keys are left exactly as they were, so
|
|
16
|
+
// results cached before phmmer existed still resolve
|
|
17
|
+
if (searchProgram === 'phmmer') {
|
|
18
|
+
return `phmmer:${blastDatabase}${idPart}:${proteinSequence}`;
|
|
19
|
+
}
|
|
15
20
|
// msaAlgorithm is part of the key because the stored msa/tree are produced by
|
|
16
21
|
// it — without it, re-running the same query under a different algorithm
|
|
17
22
|
// overwrites the earlier result and drops it from the history list
|
|
18
23
|
return `${blastDatabase}:${msaAlgorithm}${idPart}:${proteinSequence}`;
|
|
19
24
|
}
|
|
20
|
-
export async function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }) {
|
|
25
|
+
export async function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }) {
|
|
21
26
|
const db = await getDB();
|
|
22
|
-
const id = createCacheKey(
|
|
27
|
+
const id = createCacheKey({
|
|
28
|
+
proteinSequence,
|
|
29
|
+
blastDatabase,
|
|
30
|
+
msaAlgorithm,
|
|
31
|
+
searchProgram,
|
|
32
|
+
transcriptId,
|
|
33
|
+
});
|
|
23
34
|
const entry = {
|
|
24
35
|
id,
|
|
25
36
|
proteinSequence,
|
|
26
37
|
blastDatabase,
|
|
27
38
|
msaAlgorithm,
|
|
39
|
+
searchProgram,
|
|
28
40
|
msa,
|
|
29
41
|
tree,
|
|
30
42
|
treeMetadata,
|
package/dist/utils/ebiBlast.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import type { BlastHit } from './types';
|
|
2
1
|
import type { BlastDatabase } from '../LaunchMsaView/components/BlastQuery/consts';
|
|
2
|
+
import type { BlastHit } from './types';
|
|
3
3
|
/**
|
|
4
4
|
* The subset of EBI's ncbiblast JSON result this plugin reads. The service
|
|
5
5
|
* returns a great deal more per hit (urls, bit scores, e-values, the match
|
package/dist/utils/msa.d.ts
CHANGED
|
@@ -1,4 +1,16 @@
|
|
|
1
1
|
import type { MsaAlgorithm } from '../LaunchMsaView/components/BlastQuery/consts';
|
|
2
|
+
/**
|
|
3
|
+
* Build a tree from an alignment that already exists, which is what the phmmer
|
|
4
|
+
* path needs: phmmer produces the alignment itself, so there is no aligner run
|
|
5
|
+
* to take a guide tree from — and a guide tree is a byproduct of deciding
|
|
6
|
+
* progressive alignment order, not a phylogeny, so it is not what we would want
|
|
7
|
+
* even if there were one. simple_phylogeny is clustalw2's neighbour-joining on
|
|
8
|
+
* a real distance matrix, Kimura-corrected for protein distances.
|
|
9
|
+
*/
|
|
10
|
+
export declare function launchTree({ alignment, onProgress, }: {
|
|
11
|
+
alignment: string;
|
|
12
|
+
onProgress: (arg: string) => void;
|
|
13
|
+
}): Promise<string>;
|
|
2
14
|
export declare function launchMSA({ algorithm, sequence, onProgress, }: {
|
|
3
15
|
algorithm: MsaAlgorithm;
|
|
4
16
|
sequence: string;
|
package/dist/utils/msa.js
CHANGED
|
@@ -21,6 +21,35 @@ const algorithms = {
|
|
|
21
21
|
treeResult: 'phylotree',
|
|
22
22
|
},
|
|
23
23
|
};
|
|
24
|
+
/**
|
|
25
|
+
* Build a tree from an alignment that already exists, which is what the phmmer
|
|
26
|
+
* path needs: phmmer produces the alignment itself, so there is no aligner run
|
|
27
|
+
* to take a guide tree from — and a guide tree is a byproduct of deciding
|
|
28
|
+
* progressive alignment order, not a phylogeny, so it is not what we would want
|
|
29
|
+
* even if there were one. simple_phylogeny is clustalw2's neighbour-joining on
|
|
30
|
+
* a real distance matrix, Kimura-corrected for protein distances.
|
|
31
|
+
*/
|
|
32
|
+
export async function launchTree({ alignment, onProgress, }) {
|
|
33
|
+
const tool = 'simple_phylogeny';
|
|
34
|
+
onProgress('Building tree...');
|
|
35
|
+
const jobId = await submitEbiJob({
|
|
36
|
+
tool,
|
|
37
|
+
params: {
|
|
38
|
+
sequence: alignment,
|
|
39
|
+
tree: 'phylip',
|
|
40
|
+
clustering: 'Neighbour-joining',
|
|
41
|
+
kimura: 'true',
|
|
42
|
+
},
|
|
43
|
+
});
|
|
44
|
+
await waitForEbiJob({
|
|
45
|
+
tool,
|
|
46
|
+
jobId,
|
|
47
|
+
onCountdown: s => {
|
|
48
|
+
onProgress(`Re-checking tree status in... ${s}`);
|
|
49
|
+
},
|
|
50
|
+
});
|
|
51
|
+
return fetchEbiResult({ tool, jobId, type: 'tree' });
|
|
52
|
+
}
|
|
24
53
|
export async function launchMSA({ algorithm, sequence, onProgress, }) {
|
|
25
54
|
const config = algorithms[algorithm];
|
|
26
55
|
onProgress(`Launching ${algorithm} MSA...`);
|
|
@@ -35,16 +64,10 @@ export async function launchMSA({ algorithm, sequence, onProgress, }) {
|
|
|
35
64
|
onProgress(`Re-checking MSA status in... ${s}`);
|
|
36
65
|
},
|
|
37
66
|
});
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
tree: await fetchEbiResult({
|
|
45
|
-
tool: algorithm,
|
|
46
|
-
jobId,
|
|
47
|
-
type: config.treeResult,
|
|
48
|
-
}),
|
|
49
|
-
};
|
|
67
|
+
// one finished job, two result files, neither derived from the other
|
|
68
|
+
const [msa, tree] = await Promise.all([
|
|
69
|
+
fetchEbiResult({ tool: algorithm, jobId, type: config.msaResult }),
|
|
70
|
+
fetchEbiResult({ tool: algorithm, jobId, type: config.treeResult }),
|
|
71
|
+
]);
|
|
72
|
+
return { msa, tree };
|
|
50
73
|
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import type { PhmmerRow } from './phmmer';
|
|
2
|
+
import type { TaxonomyInfo } from './taxonomyNames';
|
|
3
|
+
import type { BlastHitDescription } from './types';
|
|
4
|
+
/**
|
|
5
|
+
* Turning search results into the rows the view is given, kept free of any
|
|
6
|
+
* jbrowse or network import so the whole assembly can be run and checked
|
|
7
|
+
* outside a browser — see test/phmmerLive.test.ts.
|
|
8
|
+
*/
|
|
9
|
+
export declare function buildRowMetadata(desc: BlastHitDescription, taxonomyInfo: Map<number, TaxonomyInfo>): Record<string, string>;
|
|
10
|
+
/**
|
|
11
|
+
* One target can match the query in several places and phmmer emits a row per
|
|
12
|
+
* matched envelope — four for lamprey albumin against human albumin, which has
|
|
13
|
+
* three domains. Those rows share an accession and so would share a name, and
|
|
14
|
+
* duplicate names silently collapse rows in both the MSA and the tree, so the
|
|
15
|
+
* envelope disambiguates them.
|
|
16
|
+
*/
|
|
17
|
+
export declare function makeRowNames(rows: PhmmerRow[], taxonomyInfo: Map<number, TaxonomyInfo>): string[];
|
|
18
|
+
/**
|
|
19
|
+
* The phmmer alignment as the view receives it: aligned FASTA whose first row
|
|
20
|
+
* is the query, plus the per-row metadata keyed by the same names, which are
|
|
21
|
+
* also what the tree's leaves are labelled with.
|
|
22
|
+
*/
|
|
23
|
+
export declare function buildPhmmerMsa({ rows, queryRow, taxonomyInfo, querySeqName, }: {
|
|
24
|
+
rows: PhmmerRow[];
|
|
25
|
+
queryRow: string;
|
|
26
|
+
taxonomyInfo: Map<number, TaxonomyInfo>;
|
|
27
|
+
querySeqName?: string;
|
|
28
|
+
}): {
|
|
29
|
+
msa: string;
|
|
30
|
+
treeMetadata: Record<string, Record<string, string>>;
|
|
31
|
+
};
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { makeId } from '../LaunchMsaView/components/util';
|
|
2
|
+
/**
|
|
3
|
+
* Turning search results into the rows the view is given, kept free of any
|
|
4
|
+
* jbrowse or network import so the whole assembly can be run and checked
|
|
5
|
+
* outside a browser — see test/phmmerLive.test.ts.
|
|
6
|
+
*/
|
|
7
|
+
export function buildRowMetadata(desc, taxonomyInfo) {
|
|
8
|
+
const metadata = {};
|
|
9
|
+
const taxInfo = desc.taxid ? taxonomyInfo.get(desc.taxid) : undefined;
|
|
10
|
+
if (taxInfo?.sciname) {
|
|
11
|
+
metadata['Scientific name'] = taxInfo.sciname;
|
|
12
|
+
}
|
|
13
|
+
if (taxInfo?.commonName) {
|
|
14
|
+
metadata['Common name'] = taxInfo.commonName;
|
|
15
|
+
}
|
|
16
|
+
if (desc.accession) {
|
|
17
|
+
metadata.Accession = desc.accession;
|
|
18
|
+
}
|
|
19
|
+
if (desc.id) {
|
|
20
|
+
metadata.ID = desc.id;
|
|
21
|
+
}
|
|
22
|
+
if (desc.title) {
|
|
23
|
+
metadata.Description = desc.title;
|
|
24
|
+
}
|
|
25
|
+
return metadata;
|
|
26
|
+
}
|
|
27
|
+
/**
|
|
28
|
+
* One target can match the query in several places and phmmer emits a row per
|
|
29
|
+
* matched envelope — four for lamprey albumin against human albumin, which has
|
|
30
|
+
* three domains. Those rows share an accession and so would share a name, and
|
|
31
|
+
* duplicate names silently collapse rows in both the MSA and the tree, so the
|
|
32
|
+
* envelope disambiguates them.
|
|
33
|
+
*/
|
|
34
|
+
export function makeRowNames(rows, taxonomyInfo) {
|
|
35
|
+
const baseNames = rows.map(row => makeId(row, taxonomyInfo));
|
|
36
|
+
const counts = new Map();
|
|
37
|
+
for (const name of baseNames) {
|
|
38
|
+
counts.set(name, (counts.get(name) ?? 0) + 1);
|
|
39
|
+
}
|
|
40
|
+
const used = new Set();
|
|
41
|
+
return baseNames.map((base, i) => {
|
|
42
|
+
let name = counts.get(base) > 1 ? `${base}_${rows[i].range ?? i + 1}` : base;
|
|
43
|
+
while (used.has(name)) {
|
|
44
|
+
name = `${name}_${i + 1}`;
|
|
45
|
+
}
|
|
46
|
+
used.add(name);
|
|
47
|
+
return name;
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* The phmmer alignment as the view receives it: aligned FASTA whose first row
|
|
52
|
+
* is the query, plus the per-row metadata keyed by the same names, which are
|
|
53
|
+
* also what the tree's leaves are labelled with.
|
|
54
|
+
*/
|
|
55
|
+
export function buildPhmmerMsa({ rows, queryRow, taxonomyInfo, querySeqName = 'QUERY', }) {
|
|
56
|
+
const treeMetadata = {};
|
|
57
|
+
const rowNames = makeRowNames(rows, taxonomyInfo);
|
|
58
|
+
const sequences = rows.map((row, i) => {
|
|
59
|
+
const rowName = rowNames[i];
|
|
60
|
+
treeMetadata[rowName] = buildRowMetadata(row, taxonomyInfo);
|
|
61
|
+
return `>${rowName}\n${row.aligned}`;
|
|
62
|
+
});
|
|
63
|
+
return {
|
|
64
|
+
msa: [`>${querySeqName}\n${queryRow}`, ...sequences].join('\n'),
|
|
65
|
+
treeMetadata,
|
|
66
|
+
};
|
|
67
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import type { OrthologRow } from './ncbiOrthologs';
|
|
2
|
+
export interface PantherGenome {
|
|
3
|
+
/** PANTHER's organism code, e.g. HUMAN, DROME */
|
|
4
|
+
code: string;
|
|
5
|
+
taxId: number;
|
|
6
|
+
/** short common name, e.g. fruit_fly */
|
|
7
|
+
name: string;
|
|
8
|
+
/** scientific name */
|
|
9
|
+
longName: string;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
|
|
13
|
+
* names organisms by code only in ortholog results.
|
|
14
|
+
*/
|
|
15
|
+
export declare function parseGenomes(json: unknown): PantherGenome[];
|
|
16
|
+
export interface PantherGene {
|
|
17
|
+
code: string;
|
|
18
|
+
/** UniProt accession */
|
|
19
|
+
accession: string;
|
|
20
|
+
/** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
|
|
21
|
+
geneRef: string;
|
|
22
|
+
}
|
|
23
|
+
export interface PantherHit extends PantherGene {
|
|
24
|
+
symbol?: string;
|
|
25
|
+
/**
|
|
26
|
+
* LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
|
|
27
|
+
* other ortholog in a one-to-many or many-to-many family
|
|
28
|
+
*/
|
|
29
|
+
type: 'LDO' | 'O';
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* `matchortho` -> the query gene (PANTHER names it in every row) and one hit
|
|
33
|
+
* per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
|
|
34
|
+
* no ortholog in the target set comes back as a bare `{ id }`.
|
|
35
|
+
*/
|
|
36
|
+
export declare function parseMatches(json: unknown): {
|
|
37
|
+
unmapped: boolean;
|
|
38
|
+
query?: PantherGene;
|
|
39
|
+
hits: PantherHit[];
|
|
40
|
+
};
|
|
41
|
+
/**
|
|
42
|
+
* One hit per organism, in first-seen order: the LDO where PANTHER named one,
|
|
43
|
+
* else the first other ortholog it listed. A many-to-many family (the Hox
|
|
44
|
+
* genes) has no LDO at all, so dropping to "first O" is what keeps those
|
|
45
|
+
* species in the alignment.
|
|
46
|
+
*/
|
|
47
|
+
export declare function pickOnePerGenome(hits: PantherHit[]): PantherHit[];
|
|
48
|
+
/** `uniprotkb/accessions` -> accession -> sequence */
|
|
49
|
+
export declare function parseSequences(json: unknown): Map<string, string>;
|
|
50
|
+
/** The proteome list, fetched once per page and forgotten on failure. */
|
|
51
|
+
export declare function fetchGenomes(): Promise<PantherGenome[]>;
|
|
52
|
+
export interface PantherOrthologs {
|
|
53
|
+
/** the candidate PANTHER recognised */
|
|
54
|
+
matched: string;
|
|
55
|
+
/** the query gene as PANTHER knows it, with its UniProt sequence */
|
|
56
|
+
query?: PantherGene & {
|
|
57
|
+
sequence: string;
|
|
58
|
+
};
|
|
59
|
+
rows: OrthologRow[];
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
|
|
63
|
+
* labels, accessions and sequences, plus the query gene's own protein for the
|
|
64
|
+
* query row. Two lookups (genomes, orthologs), one taxonomy batch for the
|
|
65
|
+
* labels, one UniProt batch for the sequences.
|
|
66
|
+
*
|
|
67
|
+
* `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
|
|
68
|
+
* `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
|
|
69
|
+
* `exclude` drops the query taxon, `limit` caps the rows before their
|
|
70
|
+
* sequences are fetched.
|
|
71
|
+
*/
|
|
72
|
+
export declare function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit, onProgress, }: {
|
|
73
|
+
candidates: string[];
|
|
74
|
+
taxId: number;
|
|
75
|
+
taxa?: Set<number>;
|
|
76
|
+
exclude?: number;
|
|
77
|
+
limit?: number;
|
|
78
|
+
onProgress: (arg: string) => void;
|
|
79
|
+
}): Promise<PantherOrthologs>;
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
// The second ortholog source, for the genes NCBI's ortholog sets leave out.
|
|
2
|
+
//
|
|
3
|
+
// NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
|
|
4
|
+
// worm or plant gene comes back with orthologs only in its own clade, and a fly
|
|
5
|
+
// gene gets insects and nothing else. PANTHER's ortholog sets span its 144
|
|
6
|
+
// reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
|
|
7
|
+
// answers "this gene's ortholog in every genome" with a UniProt accession per
|
|
8
|
+
// target. Sequences come from one UniProt batch call. Both hosts send
|
|
9
|
+
// `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
|
|
10
|
+
// OMA, OrthoDB and Ensembl are in react-msaview's
|
|
11
|
+
// agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
|
|
12
|
+
//
|
|
13
|
+
// Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
|
|
14
|
+
// labels, the aligner and the CDD overlay do not know which source ran.
|
|
15
|
+
// `protein` is the UniProt accession; efetch serves UniProt accessions as
|
|
16
|
+
// GenPept records with CDD Region features, so the overlay attaches as it does
|
|
17
|
+
// to a RefSeq accession.
|
|
18
|
+
import { jsonfetch } from './fetch';
|
|
19
|
+
import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs';
|
|
20
|
+
import { fetchTaxonomyInfo } from './taxonomyNames';
|
|
21
|
+
const PANTHER = 'https://pantherdb.org/services/oai/pantherdb';
|
|
22
|
+
const UNIPROT = 'https://rest.uniprot.org/uniprotkb';
|
|
23
|
+
/**
|
|
24
|
+
* `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
|
|
25
|
+
* names organisms by code only in ortholog results.
|
|
26
|
+
*/
|
|
27
|
+
export function parseGenomes(json) {
|
|
28
|
+
const list = json.search?.output?.genomes?.genome ?? [];
|
|
29
|
+
return list.flatMap(g => g.short_name && g.taxon_id
|
|
30
|
+
? [
|
|
31
|
+
{
|
|
32
|
+
code: g.short_name,
|
|
33
|
+
taxId: g.taxon_id,
|
|
34
|
+
name: g.name ?? g.short_name,
|
|
35
|
+
longName: g.long_name ?? g.short_name,
|
|
36
|
+
},
|
|
37
|
+
]
|
|
38
|
+
: []);
|
|
39
|
+
}
|
|
40
|
+
// "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
|
|
41
|
+
function parseGeneRef(ref) {
|
|
42
|
+
const [code, ...xrefs] = (ref ?? '').split('|');
|
|
43
|
+
const accession = xrefs
|
|
44
|
+
.find(x => x.startsWith('UniProtKB='))
|
|
45
|
+
?.slice('UniProtKB='.length);
|
|
46
|
+
const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession;
|
|
47
|
+
return code && accession && geneRef ? { code, accession, geneRef } : undefined;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* `matchortho` -> the query gene (PANTHER names it in every row) and one hit
|
|
51
|
+
* per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
|
|
52
|
+
* no ortholog in the target set comes back as a bare `{ id }`.
|
|
53
|
+
*/
|
|
54
|
+
export function parseMatches(json) {
|
|
55
|
+
const mapping = json.search?.mapping;
|
|
56
|
+
const mapped = mapping?.mapped;
|
|
57
|
+
const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : [];
|
|
58
|
+
const hits = [];
|
|
59
|
+
let query;
|
|
60
|
+
for (const row of rows) {
|
|
61
|
+
query ??= parseGeneRef(row.gene);
|
|
62
|
+
const target = parseGeneRef(row.target_gene);
|
|
63
|
+
if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
|
|
64
|
+
hits.push({
|
|
65
|
+
...target,
|
|
66
|
+
symbol: row.target_gene_symbol === undefined
|
|
67
|
+
? undefined
|
|
68
|
+
: String(row.target_gene_symbol),
|
|
69
|
+
type: row.ortholog,
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
return { unmapped: !!mapping?.unmapped_ids, query, hits };
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* One hit per organism, in first-seen order: the LDO where PANTHER named one,
|
|
77
|
+
* else the first other ortholog it listed. A many-to-many family (the Hox
|
|
78
|
+
* genes) has no LDO at all, so dropping to "first O" is what keeps those
|
|
79
|
+
* species in the alignment.
|
|
80
|
+
*/
|
|
81
|
+
export function pickOnePerGenome(hits) {
|
|
82
|
+
const byCode = new Map();
|
|
83
|
+
for (const hit of hits) {
|
|
84
|
+
const current = byCode.get(hit.code);
|
|
85
|
+
if (!current || (current.type === 'O' && hit.type === 'LDO')) {
|
|
86
|
+
byCode.set(hit.code, hit);
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return [...byCode.values()];
|
|
90
|
+
}
|
|
91
|
+
/** `uniprotkb/accessions` -> accession -> sequence */
|
|
92
|
+
export function parseSequences(json) {
|
|
93
|
+
const map = new Map();
|
|
94
|
+
for (const r of json.results ?? []) {
|
|
95
|
+
if (r.primaryAccession && r.sequence?.value) {
|
|
96
|
+
map.set(r.primaryAccession, r.sequence.value);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return map;
|
|
100
|
+
}
|
|
101
|
+
let genomes;
|
|
102
|
+
/** The proteome list, fetched once per page and forgotten on failure. */
|
|
103
|
+
export function fetchGenomes() {
|
|
104
|
+
genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
|
|
105
|
+
.then(parseGenomes)
|
|
106
|
+
.catch((e) => {
|
|
107
|
+
genomes = undefined;
|
|
108
|
+
throw e;
|
|
109
|
+
});
|
|
110
|
+
return genomes;
|
|
111
|
+
}
|
|
112
|
+
// UniProt caps one `accessions` call at 100 ids
|
|
113
|
+
const UNIPROT_CHUNK = 100;
|
|
114
|
+
async function fetchSequences(accessions) {
|
|
115
|
+
const map = new Map();
|
|
116
|
+
for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
|
|
117
|
+
const chunk = accessions.slice(i, i + UNIPROT_CHUNK);
|
|
118
|
+
const json = await jsonfetch(`${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`);
|
|
119
|
+
for (const [acc, seq] of parseSequences(json)) {
|
|
120
|
+
map.set(acc, seq);
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
return map;
|
|
124
|
+
}
|
|
125
|
+
// strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
|
|
126
|
+
// resolveGeneId does for NCBI
|
|
127
|
+
function cleanCandidate(raw) {
|
|
128
|
+
return raw
|
|
129
|
+
.trim()
|
|
130
|
+
.replace(/^\w+:/, '')
|
|
131
|
+
.replace(/\.\d+$/, '');
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
|
|
135
|
+
* carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
|
|
136
|
+
* some of those are names PANTHER knows. `targets` omitted asks for every
|
|
137
|
+
* genome PANTHER has, which is one call rather than one per genome.
|
|
138
|
+
*/
|
|
139
|
+
async function matchOrthologs(candidates, taxId, targets) {
|
|
140
|
+
let matched;
|
|
141
|
+
for (const raw of candidates) {
|
|
142
|
+
const query = cleanCandidate(raw);
|
|
143
|
+
if (!query) {
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
const params = new URLSearchParams({
|
|
147
|
+
geneInputList: query,
|
|
148
|
+
organism: String(taxId),
|
|
149
|
+
orthologType: 'all',
|
|
150
|
+
});
|
|
151
|
+
if (targets) {
|
|
152
|
+
params.set('targetOrganism', targets.map(t => t.taxId).join(','));
|
|
153
|
+
}
|
|
154
|
+
const parsed = parseMatches(await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`));
|
|
155
|
+
if (!parsed.unmapped) {
|
|
156
|
+
matched ??= query;
|
|
157
|
+
if (parsed.query) {
|
|
158
|
+
return { ...parsed, matched: query };
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
return matched === undefined
|
|
163
|
+
? undefined
|
|
164
|
+
: { matched, hits: [], query: undefined };
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
|
|
168
|
+
* labels, accessions and sequences, plus the query gene's own protein for the
|
|
169
|
+
* query row. Two lookups (genomes, orthologs), one taxonomy batch for the
|
|
170
|
+
* labels, one UniProt batch for the sequences.
|
|
171
|
+
*
|
|
172
|
+
* `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
|
|
173
|
+
* `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
|
|
174
|
+
* `exclude` drops the query taxon, `limit` caps the rows before their
|
|
175
|
+
* sequences are fetched.
|
|
176
|
+
*/
|
|
177
|
+
export async function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit = defaultMaxSpecies, onProgress, }) {
|
|
178
|
+
const all = await fetchGenomes();
|
|
179
|
+
const byTaxId = new Map(all.map(g => [g.taxId, g]));
|
|
180
|
+
const byCode = new Map(all.map(g => [g.code, g]));
|
|
181
|
+
const queryGenome = byTaxId.get(taxId);
|
|
182
|
+
if (!queryGenome) {
|
|
183
|
+
throw new Error(`PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`);
|
|
184
|
+
}
|
|
185
|
+
const targets = taxa
|
|
186
|
+
? [...taxa].flatMap(t => {
|
|
187
|
+
const g = byTaxId.get(t);
|
|
188
|
+
return g && t !== exclude ? [g] : [];
|
|
189
|
+
})
|
|
190
|
+
: undefined;
|
|
191
|
+
onProgress('Matching orthologs at PANTHER...');
|
|
192
|
+
const match = await matchOrthologs(candidates, taxId, targets);
|
|
193
|
+
if (!match) {
|
|
194
|
+
throw new Error(`PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`);
|
|
195
|
+
}
|
|
196
|
+
const rank = new Map(targets?.map((t, i) => [t.taxId, i]));
|
|
197
|
+
const picks = pickOnePerGenome(match.hits)
|
|
198
|
+
.map(hit => ({ hit, genome: byCode.get(hit.code) }))
|
|
199
|
+
.filter((p) => !!p.genome &&
|
|
200
|
+
p.genome.taxId !== exclude &&
|
|
201
|
+
(targets ? rank.has(p.genome.taxId) : true))
|
|
202
|
+
.sort((a, b) => targets ? rank.get(a.genome.taxId) - rank.get(b.genome.taxId) : 0)
|
|
203
|
+
.slice(0, limit);
|
|
204
|
+
if (picks.length < 2) {
|
|
205
|
+
throw new Error(`Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`);
|
|
206
|
+
}
|
|
207
|
+
onProgress(`Fetching ${picks.length} protein sequences from UniProt...`);
|
|
208
|
+
const [names, sequences] = await Promise.all([
|
|
209
|
+
taxonomyNames(picks.map(p => p.genome.taxId)),
|
|
210
|
+
fetchSequences([
|
|
211
|
+
...(match.query ? [match.query.accession] : []),
|
|
212
|
+
...picks.map(p => p.hit.accession),
|
|
213
|
+
]),
|
|
214
|
+
]);
|
|
215
|
+
const described = picks.map(({ hit, genome }) => {
|
|
216
|
+
const info = names.get(genome.taxId);
|
|
217
|
+
return {
|
|
218
|
+
hit,
|
|
219
|
+
genome,
|
|
220
|
+
name: info ? (info.commonName ?? info.sciname) : genome.name,
|
|
221
|
+
scientificName: info?.sciname || genome.longName,
|
|
222
|
+
commonName: info?.commonName,
|
|
223
|
+
};
|
|
224
|
+
});
|
|
225
|
+
const labels = dedupeLabels(described.map(d => d.name));
|
|
226
|
+
const rows = described
|
|
227
|
+
.map(({ hit, genome, scientificName, commonName }, i) => ({
|
|
228
|
+
taxId: genome.taxId,
|
|
229
|
+
label: labels[i],
|
|
230
|
+
scientificName,
|
|
231
|
+
commonName,
|
|
232
|
+
geneId: hit.geneRef,
|
|
233
|
+
protein: hit.accession,
|
|
234
|
+
sequence: sequences.get(hit.accession) ?? '',
|
|
235
|
+
}))
|
|
236
|
+
.filter(r => r.sequence);
|
|
237
|
+
if (rows.length < 2) {
|
|
238
|
+
throw new Error('Could not fetch protein sequences for the orthologs');
|
|
239
|
+
}
|
|
240
|
+
const querySequence = match.query && sequences.get(match.query.accession);
|
|
241
|
+
return {
|
|
242
|
+
matched: match.matched,
|
|
243
|
+
query: match.query && querySequence
|
|
244
|
+
? { ...match.query, sequence: querySequence }
|
|
245
|
+
: undefined,
|
|
246
|
+
rows,
|
|
247
|
+
};
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
|
|
251
|
+
* are. A failed lookup only costs the labels, which fall back to PANTHER's own
|
|
252
|
+
* short names, so it is logged rather than thrown.
|
|
253
|
+
*/
|
|
254
|
+
async function taxonomyNames(taxIds) {
|
|
255
|
+
try {
|
|
256
|
+
return await fetchTaxonomyInfo(taxIds);
|
|
257
|
+
}
|
|
258
|
+
catch (e) {
|
|
259
|
+
console.warn('[msaview-orthologs] taxonomy name lookup failed:', e);
|
|
260
|
+
return new Map();
|
|
261
|
+
}
|
|
262
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import type { PhmmerDatabase } from '../LaunchMsaView/components/BlastQuery/consts';
|
|
2
|
+
export interface PhmmerRow {
|
|
3
|
+
accession: string;
|
|
4
|
+
id: string;
|
|
5
|
+
sciname: string;
|
|
6
|
+
taxid?: number;
|
|
7
|
+
title?: string;
|
|
8
|
+
/** the matched envelope on the target, e.g. '503-912', absent if unparseable */
|
|
9
|
+
range?: string;
|
|
10
|
+
/** the row as phmmer aligned it, uppercased with '.' inserts turned into '-' */
|
|
11
|
+
aligned: string;
|
|
12
|
+
}
|
|
13
|
+
export interface PhmmerAlignment {
|
|
14
|
+
rows: PhmmerRow[];
|
|
15
|
+
/**
|
|
16
|
+
* the query, placed into the same columns. phmmer does not put the query in
|
|
17
|
+
* its own output, so this is derived — see buildQueryRow.
|
|
18
|
+
*/
|
|
19
|
+
queryRow: string;
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* Human-facing link to a job, shown while it runs and on error.
|
|
23
|
+
*
|
|
24
|
+
* The category has to be sss: jdispatcher serves its shell with a 200 for any
|
|
25
|
+
* category, so /pfa/ and /psa/ look fine to a fetch and render "Page Not Found"
|
|
26
|
+
* in a browser.
|
|
27
|
+
*/
|
|
28
|
+
export declare function phmmerResultUrl(jobId: string): string;
|
|
29
|
+
/** EBI job ids are prefixed with the tool that made them */
|
|
30
|
+
export declare function isPhmmerJobId(jobId: string): boolean;
|
|
31
|
+
/**
|
|
32
|
+
* Exported for testing against a captured .sto — the annotation names (RF, the
|
|
33
|
+
* DE line's OS=/OX=) are the whole risk in this mapping, and nothing else in CI
|
|
34
|
+
* would notice if HMMER or EBI changed one.
|
|
35
|
+
*/
|
|
36
|
+
export declare function parsePhmmerAlignment({ stockholm, query, }: {
|
|
37
|
+
stockholm: string;
|
|
38
|
+
query: string;
|
|
39
|
+
}): PhmmerAlignment;
|
|
40
|
+
export declare function queryPhmmer({ query, database, onProgress, onRid, }: {
|
|
41
|
+
query: string;
|
|
42
|
+
database: PhmmerDatabase;
|
|
43
|
+
onProgress: (arg: string) => void;
|
|
44
|
+
onRid: (arg: string) => void;
|
|
45
|
+
}): Promise<{
|
|
46
|
+
rows: PhmmerRow[];
|
|
47
|
+
/**
|
|
48
|
+
* the query, placed into the same columns. phmmer does not put the query in
|
|
49
|
+
* its own output, so this is derived — see buildQueryRow.
|
|
50
|
+
*/
|
|
51
|
+
queryRow: string;
|
|
52
|
+
rid: string;
|
|
53
|
+
}>;
|