jbrowse-plugin-msaview 3.2.0 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/AddHighlightModel/GenomeMouseoverHighlight.js +1 -1
  2. package/dist/AddHighlightModel/MsaToGenomeHighlight.js +1 -1
  3. package/dist/AddHighlightModel/index.js +1 -1
  4. package/dist/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.js +61 -17
  5. package/dist/LaunchMsaView/components/BlastQuery/BlastManualPanel.js +1 -1
  6. package/dist/LaunchMsaView/components/BlastQuery/BlastPanel.js +2 -2
  7. package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.d.ts +12 -0
  8. package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.js +21 -2
  9. package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.d.ts +1 -0
  10. package/dist/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.js +29 -0
  11. package/dist/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.js +1 -1
  12. package/dist/LaunchMsaView/components/BlastQuery/consts.d.ts +28 -0
  13. package/dist/LaunchMsaView/components/BlastQuery/consts.js +21 -0
  14. package/dist/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.js +1 -1
  15. package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +9 -4
  16. package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.d.ts +9 -0
  17. package/dist/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.js +20 -0
  18. package/dist/LaunchMsaView/components/SubmitCancelActions.test.js +1 -1
  19. package/dist/LaunchMsaView/components/useFeatureSequence.js +1 -1
  20. package/dist/LaunchMsaView/detectQueryRow.d.ts +15 -2
  21. package/dist/LaunchMsaView/detectQueryRow.js +20 -21
  22. package/dist/LaunchMsaView/detectQueryRow.test.js +15 -15
  23. package/dist/LaunchMsaView/useQueryRowName.js +5 -8
  24. package/dist/MsaViewPanel/afterCreateAutoruns.js +2 -2
  25. package/dist/MsaViewPanel/components/ErrorBoundary.d.ts +2 -2
  26. package/dist/MsaViewPanel/components/JobLink.js +7 -1
  27. package/dist/MsaViewPanel/components/LaunchProgress.d.ts +17 -0
  28. package/dist/MsaViewPanel/components/LaunchProgress.js +41 -0
  29. package/dist/MsaViewPanel/components/MsaViewPanel.js +6 -3
  30. package/dist/MsaViewPanel/components/MsaViewPanel.test.d.ts +1 -0
  31. package/dist/MsaViewPanel/components/MsaViewPanel.test.js +68 -0
  32. package/dist/MsaViewPanel/doLaunchBlast.d.ts +1 -1
  33. package/dist/MsaViewPanel/doLaunchBlast.js +84 -52
  34. package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +7 -5
  35. package/dist/MsaViewPanel/doLaunchOrthologs.js +64 -30
  36. package/dist/MsaViewPanel/doLaunchOrthologs.test.js +106 -1
  37. package/dist/MsaViewPanel/genomeToMSA.js +4 -2
  38. package/dist/MsaViewPanel/genomeToMSA.test.js +34 -0
  39. package/dist/MsaViewPanel/model.d.ts +41 -11
  40. package/dist/MsaViewPanel/model.js +6 -0
  41. package/dist/MsaViewPanel/observeProteinHighlights.test.js +11 -0
  42. package/dist/MsaViewPanel/syncGenomeHoverToMsaColumn.test.js +1 -0
  43. package/dist/MsaViewPanel/util.d.ts +18 -0
  44. package/dist/MsaViewPanel/util.js +17 -0
  45. package/dist/jbrowse-plugin-msaview.umd.production.min.js +47 -35
  46. package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
  47. package/dist/utils/blastCache.d.ts +10 -6
  48. package/dist/utils/blastCache.js +15 -3
  49. package/dist/utils/ebiBlast.d.ts +1 -1
  50. package/dist/utils/msa.d.ts +12 -0
  51. package/dist/utils/msa.js +35 -12
  52. package/dist/utils/msaRows.d.ts +31 -0
  53. package/dist/utils/msaRows.js +67 -0
  54. package/dist/utils/pantherOrthologs.d.ts +79 -0
  55. package/dist/utils/pantherOrthologs.js +262 -0
  56. package/dist/utils/phmmer.d.ts +53 -0
  57. package/dist/utils/phmmer.js +118 -0
  58. package/dist/utils/taxonomyNames.d.ts +1 -1
  59. package/dist/utils/taxonomyNames.js +6 -1
  60. package/dist/version.d.ts +1 -1
  61. package/dist/version.js +1 -1
  62. package/package.json +27 -21
  63. package/src/AddHighlightModel/GenomeMouseoverHighlight.tsx +1 -1
  64. package/src/AddHighlightModel/MsaToGenomeHighlight.tsx +1 -1
  65. package/src/AddHighlightModel/index.tsx +1 -1
  66. package/src/LaunchMsaView/components/BlastQuery/BlastAutomaticPanel.tsx +88 -30
  67. package/src/LaunchMsaView/components/BlastQuery/BlastManualPanel.tsx +1 -1
  68. package/src/LaunchMsaView/components/BlastQuery/BlastPanel.tsx +4 -4
  69. package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.test.ts +50 -0
  70. package/src/LaunchMsaView/components/BlastQuery/CachedBlastResults.tsx +23 -3
  71. package/src/LaunchMsaView/components/BlastQuery/MsaAlgorithmSelect.tsx +1 -1
  72. package/src/LaunchMsaView/components/BlastQuery/consts.ts +40 -0
  73. package/src/LaunchMsaView/components/ManualMSALoader/ManualMSALoader.tsx +1 -1
  74. package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +21 -5
  75. package/src/LaunchMsaView/components/OrthologQuery/OrthologSourceSelect.tsx +52 -0
  76. package/src/LaunchMsaView/components/SubmitCancelActions.test.tsx +1 -1
  77. package/src/LaunchMsaView/components/useFeatureSequence.ts +1 -1
  78. package/src/LaunchMsaView/detectQueryRow.test.ts +17 -15
  79. package/src/LaunchMsaView/detectQueryRow.ts +34 -23
  80. package/src/LaunchMsaView/useQueryRowName.ts +6 -9
  81. package/src/MsaViewPanel/afterCreateAutoruns.ts +2 -2
  82. package/src/MsaViewPanel/components/ErrorBoundary.tsx +2 -1
  83. package/src/MsaViewPanel/components/JobLink.tsx +7 -2
  84. package/src/MsaViewPanel/components/LaunchProgress.tsx +62 -0
  85. package/src/MsaViewPanel/components/MsaViewPanel.test.tsx +83 -0
  86. package/src/MsaViewPanel/components/MsaViewPanel.tsx +7 -4
  87. package/src/MsaViewPanel/doLaunchBlast.ts +127 -69
  88. package/src/MsaViewPanel/doLaunchOrthologs.test.ts +119 -2
  89. package/src/MsaViewPanel/doLaunchOrthologs.ts +100 -38
  90. package/src/MsaViewPanel/genomeToMSA.test.ts +37 -0
  91. package/src/MsaViewPanel/genomeToMSA.ts +6 -2
  92. package/src/MsaViewPanel/model.ts +38 -5
  93. package/src/MsaViewPanel/observeProteinHighlights.test.ts +13 -0
  94. package/src/MsaViewPanel/syncGenomeHoverToMsaColumn.test.ts +1 -0
  95. package/src/MsaViewPanel/util.ts +18 -0
  96. package/src/utils/blastCache.ts +33 -12
  97. package/src/utils/ebiBlast.ts +1 -1
  98. package/src/utils/msa.ts +43 -12
  99. package/src/utils/msaRows.ts +95 -0
  100. package/src/utils/pantherOrthologs.ts +399 -0
  101. package/src/utils/phmmer.ts +174 -0
  102. package/src/utils/taxonomyNames.ts +6 -1
  103. package/src/version.ts +1 -1
  104. package/dist/MsaViewPanel/components/LoadingBLAST.d.ts +0 -6
  105. package/dist/MsaViewPanel/components/LoadingBLAST.js +0 -26
  106. package/src/MsaViewPanel/components/LoadingBLAST.tsx +0 -48
@@ -1,15 +1,18 @@
1
- import type { BlastDatabase, MsaAlgorithm } from '../LaunchMsaView/components/BlastQuery/consts';
1
+ import type { BlastDatabase, MsaAlgorithm, PhmmerDatabase, SearchProgram } from '../LaunchMsaView/components/BlastQuery/consts';
2
2
  export interface CachedBlastResult {
3
3
  id: string;
4
4
  proteinSequence: string;
5
- blastDatabase: BlastDatabase;
5
+ blastDatabase: BlastDatabase | PhmmerDatabase;
6
6
  /**
7
7
  * Only ever set on rows cached by a version that still queried NCBI, where
8
8
  * the choice between blastp and quick-blastp was real. Kept so those rows
9
9
  * still display; never written now.
10
10
  */
11
11
  blastProgram?: string;
12
- msaAlgorithm: MsaAlgorithm;
12
+ /** absent on rows cached before phmmer existed, which were all blastp */
13
+ searchProgram?: SearchProgram;
14
+ /** absent on phmmer rows, which are aligned by the search itself */
15
+ msaAlgorithm?: MsaAlgorithm;
13
16
  msa: string;
14
17
  tree: string;
15
18
  treeMetadata: string;
@@ -20,10 +23,11 @@ export interface CachedBlastResult {
20
23
  transcriptName?: string;
21
24
  geneName?: string;
22
25
  }
23
- export declare function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }: {
26
+ export declare function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }: {
24
27
  proteinSequence: string;
25
- blastDatabase: BlastDatabase;
26
- msaAlgorithm: MsaAlgorithm;
28
+ blastDatabase: BlastDatabase | PhmmerDatabase;
29
+ msaAlgorithm?: MsaAlgorithm;
30
+ searchProgram?: SearchProgram;
27
31
  msa: string;
28
32
  tree: string;
29
33
  treeMetadata: string;
@@ -10,21 +10,33 @@ const getDB = createDbOpener(DB_NAME, DB_VERSION, (db, oldVersion) => {
10
10
  db.createObjectStore(STORE_NAME, { keyPath: 'id' });
11
11
  }
12
12
  });
13
- function createCacheKey(proteinSequence, blastDatabase, msaAlgorithm, transcriptId) {
13
+ function createCacheKey({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, transcriptId, }) {
14
14
  const idPart = transcriptId ? `:${transcriptId}` : '';
15
+ // phmmer keys are prefixed and blastp keys are left exactly as they were, so
16
+ // results cached before phmmer existed still resolve
17
+ if (searchProgram === 'phmmer') {
18
+ return `phmmer:${blastDatabase}${idPart}:${proteinSequence}`;
19
+ }
15
20
  // msaAlgorithm is part of the key because the stored msa/tree are produced by
16
21
  // it — without it, re-running the same query under a different algorithm
17
22
  // overwrites the earlier result and drops it from the history list
18
23
  return `${blastDatabase}:${msaAlgorithm}${idPart}:${proteinSequence}`;
19
24
  }
20
- export async function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }) {
25
+ export async function saveBlastResult({ proteinSequence, blastDatabase, msaAlgorithm, searchProgram, msa, tree, treeMetadata, rid, geneId, transcriptId, transcriptName, geneName, }) {
21
26
  const db = await getDB();
22
- const id = createCacheKey(proteinSequence, blastDatabase, msaAlgorithm, transcriptId);
27
+ const id = createCacheKey({
28
+ proteinSequence,
29
+ blastDatabase,
30
+ msaAlgorithm,
31
+ searchProgram,
32
+ transcriptId,
33
+ });
23
34
  const entry = {
24
35
  id,
25
36
  proteinSequence,
26
37
  blastDatabase,
27
38
  msaAlgorithm,
39
+ searchProgram,
28
40
  msa,
29
41
  tree,
30
42
  treeMetadata,
@@ -1,5 +1,5 @@
1
- import type { BlastHit } from './types';
2
1
  import type { BlastDatabase } from '../LaunchMsaView/components/BlastQuery/consts';
2
+ import type { BlastHit } from './types';
3
3
  /**
4
4
  * The subset of EBI's ncbiblast JSON result this plugin reads. The service
5
5
  * returns a great deal more per hit (urls, bit scores, e-values, the match
@@ -1,4 +1,16 @@
1
1
  import type { MsaAlgorithm } from '../LaunchMsaView/components/BlastQuery/consts';
2
+ /**
3
+ * Build a tree from an alignment that already exists, which is what the phmmer
4
+ * path needs: phmmer produces the alignment itself, so there is no aligner run
5
+ * to take a guide tree from — and a guide tree is a byproduct of deciding
6
+ * progressive alignment order, not a phylogeny, so it is not what we would want
7
+ * even if there were one. simple_phylogeny is clustalw2's neighbour-joining on
8
+ * a real distance matrix, Kimura-corrected for protein distances.
9
+ */
10
+ export declare function launchTree({ alignment, onProgress, }: {
11
+ alignment: string;
12
+ onProgress: (arg: string) => void;
13
+ }): Promise<string>;
2
14
  export declare function launchMSA({ algorithm, sequence, onProgress, }: {
3
15
  algorithm: MsaAlgorithm;
4
16
  sequence: string;
package/dist/utils/msa.js CHANGED
@@ -21,6 +21,35 @@ const algorithms = {
21
21
  treeResult: 'phylotree',
22
22
  },
23
23
  };
24
+ /**
25
+ * Build a tree from an alignment that already exists, which is what the phmmer
26
+ * path needs: phmmer produces the alignment itself, so there is no aligner run
27
+ * to take a guide tree from — and a guide tree is a byproduct of deciding
28
+ * progressive alignment order, not a phylogeny, so it is not what we would want
29
+ * even if there were one. simple_phylogeny is clustalw2's neighbour-joining on
30
+ * a real distance matrix, Kimura-corrected for protein distances.
31
+ */
32
+ export async function launchTree({ alignment, onProgress, }) {
33
+ const tool = 'simple_phylogeny';
34
+ onProgress('Building tree...');
35
+ const jobId = await submitEbiJob({
36
+ tool,
37
+ params: {
38
+ sequence: alignment,
39
+ tree: 'phylip',
40
+ clustering: 'Neighbour-joining',
41
+ kimura: 'true',
42
+ },
43
+ });
44
+ await waitForEbiJob({
45
+ tool,
46
+ jobId,
47
+ onCountdown: s => {
48
+ onProgress(`Re-checking tree status in... ${s}`);
49
+ },
50
+ });
51
+ return fetchEbiResult({ tool, jobId, type: 'tree' });
52
+ }
24
53
  export async function launchMSA({ algorithm, sequence, onProgress, }) {
25
54
  const config = algorithms[algorithm];
26
55
  onProgress(`Launching ${algorithm} MSA...`);
@@ -35,16 +64,10 @@ export async function launchMSA({ algorithm, sequence, onProgress, }) {
35
64
  onProgress(`Re-checking MSA status in... ${s}`);
36
65
  },
37
66
  });
38
- return {
39
- msa: await fetchEbiResult({
40
- tool: algorithm,
41
- jobId,
42
- type: config.msaResult,
43
- }),
44
- tree: await fetchEbiResult({
45
- tool: algorithm,
46
- jobId,
47
- type: config.treeResult,
48
- }),
49
- };
67
+ // one finished job, two result files, neither derived from the other
68
+ const [msa, tree] = await Promise.all([
69
+ fetchEbiResult({ tool: algorithm, jobId, type: config.msaResult }),
70
+ fetchEbiResult({ tool: algorithm, jobId, type: config.treeResult }),
71
+ ]);
72
+ return { msa, tree };
50
73
  }
@@ -0,0 +1,31 @@
1
+ import type { PhmmerRow } from './phmmer';
2
+ import type { TaxonomyInfo } from './taxonomyNames';
3
+ import type { BlastHitDescription } from './types';
4
+ /**
5
+ * Turning search results into the rows the view is given, kept free of any
6
+ * jbrowse or network import so the whole assembly can be run and checked
7
+ * outside a browser — see test/phmmerLive.test.ts.
8
+ */
9
+ export declare function buildRowMetadata(desc: BlastHitDescription, taxonomyInfo: Map<number, TaxonomyInfo>): Record<string, string>;
10
+ /**
11
+ * One target can match the query in several places and phmmer emits a row per
12
+ * matched envelope — four for lamprey albumin against human albumin, which has
13
+ * three domains. Those rows share an accession and so would share a name, and
14
+ * duplicate names silently collapse rows in both the MSA and the tree, so the
15
+ * envelope disambiguates them.
16
+ */
17
+ export declare function makeRowNames(rows: PhmmerRow[], taxonomyInfo: Map<number, TaxonomyInfo>): string[];
18
+ /**
19
+ * The phmmer alignment as the view receives it: aligned FASTA whose first row
20
+ * is the query, plus the per-row metadata keyed by the same names, which are
21
+ * also what the tree's leaves are labelled with.
22
+ */
23
+ export declare function buildPhmmerMsa({ rows, queryRow, taxonomyInfo, querySeqName, }: {
24
+ rows: PhmmerRow[];
25
+ queryRow: string;
26
+ taxonomyInfo: Map<number, TaxonomyInfo>;
27
+ querySeqName?: string;
28
+ }): {
29
+ msa: string;
30
+ treeMetadata: Record<string, Record<string, string>>;
31
+ };
@@ -0,0 +1,67 @@
1
+ import { makeId } from '../LaunchMsaView/components/util';
2
+ /**
3
+ * Turning search results into the rows the view is given, kept free of any
4
+ * jbrowse or network import so the whole assembly can be run and checked
5
+ * outside a browser — see test/phmmerLive.test.ts.
6
+ */
7
+ export function buildRowMetadata(desc, taxonomyInfo) {
8
+ const metadata = {};
9
+ const taxInfo = desc.taxid ? taxonomyInfo.get(desc.taxid) : undefined;
10
+ if (taxInfo?.sciname) {
11
+ metadata['Scientific name'] = taxInfo.sciname;
12
+ }
13
+ if (taxInfo?.commonName) {
14
+ metadata['Common name'] = taxInfo.commonName;
15
+ }
16
+ if (desc.accession) {
17
+ metadata.Accession = desc.accession;
18
+ }
19
+ if (desc.id) {
20
+ metadata.ID = desc.id;
21
+ }
22
+ if (desc.title) {
23
+ metadata.Description = desc.title;
24
+ }
25
+ return metadata;
26
+ }
27
+ /**
28
+ * One target can match the query in several places and phmmer emits a row per
29
+ * matched envelope — four for lamprey albumin against human albumin, which has
30
+ * three domains. Those rows share an accession and so would share a name, and
31
+ * duplicate names silently collapse rows in both the MSA and the tree, so the
32
+ * envelope disambiguates them.
33
+ */
34
+ export function makeRowNames(rows, taxonomyInfo) {
35
+ const baseNames = rows.map(row => makeId(row, taxonomyInfo));
36
+ const counts = new Map();
37
+ for (const name of baseNames) {
38
+ counts.set(name, (counts.get(name) ?? 0) + 1);
39
+ }
40
+ const used = new Set();
41
+ return baseNames.map((base, i) => {
42
+ let name = counts.get(base) > 1 ? `${base}_${rows[i].range ?? i + 1}` : base;
43
+ while (used.has(name)) {
44
+ name = `${name}_${i + 1}`;
45
+ }
46
+ used.add(name);
47
+ return name;
48
+ });
49
+ }
50
+ /**
51
+ * The phmmer alignment as the view receives it: aligned FASTA whose first row
52
+ * is the query, plus the per-row metadata keyed by the same names, which are
53
+ * also what the tree's leaves are labelled with.
54
+ */
55
+ export function buildPhmmerMsa({ rows, queryRow, taxonomyInfo, querySeqName = 'QUERY', }) {
56
+ const treeMetadata = {};
57
+ const rowNames = makeRowNames(rows, taxonomyInfo);
58
+ const sequences = rows.map((row, i) => {
59
+ const rowName = rowNames[i];
60
+ treeMetadata[rowName] = buildRowMetadata(row, taxonomyInfo);
61
+ return `>${rowName}\n${row.aligned}`;
62
+ });
63
+ return {
64
+ msa: [`>${querySeqName}\n${queryRow}`, ...sequences].join('\n'),
65
+ treeMetadata,
66
+ };
67
+ }
@@ -0,0 +1,79 @@
1
+ import type { OrthologRow } from './ncbiOrthologs';
2
+ export interface PantherGenome {
3
+ /** PANTHER's organism code, e.g. HUMAN, DROME */
4
+ code: string;
5
+ taxId: number;
6
+ /** short common name, e.g. fruit_fly */
7
+ name: string;
8
+ /** scientific name */
9
+ longName: string;
10
+ }
11
+ /**
12
+ * `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
13
+ * names organisms by code only in ortholog results.
14
+ */
15
+ export declare function parseGenomes(json: unknown): PantherGenome[];
16
+ export interface PantherGene {
17
+ code: string;
18
+ /** UniProt accession */
19
+ accession: string;
20
+ /** the source database's own id, e.g. HGNC=1773, FlyBase=FBgn0016131 */
21
+ geneRef: string;
22
+ }
23
+ export interface PantherHit extends PantherGene {
24
+ symbol?: string;
25
+ /**
26
+ * LDO = least diverged ortholog, PANTHER's pick of the one-to-one; O = any
27
+ * other ortholog in a one-to-many or many-to-many family
28
+ */
29
+ type: 'LDO' | 'O';
30
+ }
31
+ /**
32
+ * `matchortho` -> the query gene (PANTHER names it in every row) and one hit
33
+ * per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
34
+ * no ortholog in the target set comes back as a bare `{ id }`.
35
+ */
36
+ export declare function parseMatches(json: unknown): {
37
+ unmapped: boolean;
38
+ query?: PantherGene;
39
+ hits: PantherHit[];
40
+ };
41
+ /**
42
+ * One hit per organism, in first-seen order: the LDO where PANTHER named one,
43
+ * else the first other ortholog it listed. A many-to-many family (the Hox
44
+ * genes) has no LDO at all, so dropping to "first O" is what keeps those
45
+ * species in the alignment.
46
+ */
47
+ export declare function pickOnePerGenome(hits: PantherHit[]): PantherHit[];
48
+ /** `uniprotkb/accessions` -> accession -> sequence */
49
+ export declare function parseSequences(json: unknown): Map<string, string>;
50
+ /** The proteome list, fetched once per page and forgotten on failure. */
51
+ export declare function fetchGenomes(): Promise<PantherGenome[]>;
52
+ export interface PantherOrthologs {
53
+ /** the candidate PANTHER recognised */
54
+ matched: string;
55
+ /** the query gene as PANTHER knows it, with its UniProt sequence */
56
+ query?: PantherGene & {
57
+ sequence: string;
58
+ };
59
+ rows: OrthologRow[];
60
+ }
61
+ /**
62
+ * The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
63
+ * labels, accessions and sequences, plus the query gene's own protein for the
64
+ * query row. Two lookups (genomes, orthologs), one taxonomy batch for the
65
+ * labels, one UniProt batch for the sequences.
66
+ *
67
+ * `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
68
+ * `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
69
+ * `exclude` drops the query taxon, `limit` caps the rows before their
70
+ * sequences are fetched.
71
+ */
72
+ export declare function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit, onProgress, }: {
73
+ candidates: string[];
74
+ taxId: number;
75
+ taxa?: Set<number>;
76
+ exclude?: number;
77
+ limit?: number;
78
+ onProgress: (arg: string) => void;
79
+ }): Promise<PantherOrthologs>;
@@ -0,0 +1,262 @@
1
+ // The second ortholog source, for the genes NCBI's ortholog sets leave out.
2
+ //
3
+ // NCBI Datasets computes orthologs for vertebrates and insects, so a yeast,
4
+ // worm or plant gene comes back with orthologs only in its own clade, and a fly
5
+ // gene gets insects and nothing else. PANTHER's ortholog sets span its 144
6
+ // reference proteomes, human to yeast to Arabidopsis, and one `matchortho` call
7
+ // answers "this gene's ortholog in every genome" with a UniProt accession per
8
+ // target. Sequences come from one UniProt batch call. Both hosts send
9
+ // `Access-Control-Allow-Origin: *`. The measurements that picked PANTHER over
10
+ // OMA, OrthoDB and Ensembl are in react-msaview's
11
+ // agent-docs/ideas/ortholog-sources-beyond-ncbi.md.
12
+ //
13
+ // Rows come out in the same shape as ncbiOrthologs.ts's, so the launch, the
14
+ // labels, the aligner and the CDD overlay do not know which source ran.
15
+ // `protein` is the UniProt accession; efetch serves UniProt accessions as
16
+ // GenPept records with CDD Region features, so the overlay attaches as it does
17
+ // to a RefSeq accession.
18
+ import { jsonfetch } from './fetch';
19
+ import { dedupeLabels, defaultMaxSpecies } from './ncbiOrthologs';
20
+ import { fetchTaxonomyInfo } from './taxonomyNames';
21
+ const PANTHER = 'https://pantherdb.org/services/oai/pantherdb';
22
+ const UNIPROT = 'https://rest.uniprot.org/uniprotkb';
23
+ /**
24
+ * `supportedgenomes` -> the code<->taxon map every other parse needs. PANTHER
25
+ * names organisms by code only in ortholog results.
26
+ */
27
+ export function parseGenomes(json) {
28
+ const list = json.search?.output?.genomes?.genome ?? [];
29
+ return list.flatMap(g => g.short_name && g.taxon_id
30
+ ? [
31
+ {
32
+ code: g.short_name,
33
+ taxId: g.taxon_id,
34
+ name: g.name ?? g.short_name,
35
+ longName: g.long_name ?? g.short_name,
36
+ },
37
+ ]
38
+ : []);
39
+ }
40
+ // "HUMAN|HGNC=1773|UniProtKB=P11802" -> { code, geneRef, accession }
41
+ function parseGeneRef(ref) {
42
+ const [code, ...xrefs] = (ref ?? '').split('|');
43
+ const accession = xrefs
44
+ .find(x => x.startsWith('UniProtKB='))
45
+ ?.slice('UniProtKB='.length);
46
+ const geneRef = xrefs.find(x => !x.startsWith('UniProtKB=')) ?? accession;
47
+ return code && accession && geneRef ? { code, accession, geneRef } : undefined;
48
+ }
49
+ /**
50
+ * `matchortho` -> the query gene (PANTHER names it in every row) and one hit
51
+ * per target gene. An unknown gene comes back under `unmapped_ids`; a gene with
52
+ * no ortholog in the target set comes back as a bare `{ id }`.
53
+ */
54
+ export function parseMatches(json) {
55
+ const mapping = json.search?.mapping;
56
+ const mapped = mapping?.mapped;
57
+ const rows = Array.isArray(mapped) ? mapped : mapped ? [mapped] : [];
58
+ const hits = [];
59
+ let query;
60
+ for (const row of rows) {
61
+ query ??= parseGeneRef(row.gene);
62
+ const target = parseGeneRef(row.target_gene);
63
+ if (target && (row.ortholog === 'LDO' || row.ortholog === 'O')) {
64
+ hits.push({
65
+ ...target,
66
+ symbol: row.target_gene_symbol === undefined
67
+ ? undefined
68
+ : String(row.target_gene_symbol),
69
+ type: row.ortholog,
70
+ });
71
+ }
72
+ }
73
+ return { unmapped: !!mapping?.unmapped_ids, query, hits };
74
+ }
75
+ /**
76
+ * One hit per organism, in first-seen order: the LDO where PANTHER named one,
77
+ * else the first other ortholog it listed. A many-to-many family (the Hox
78
+ * genes) has no LDO at all, so dropping to "first O" is what keeps those
79
+ * species in the alignment.
80
+ */
81
+ export function pickOnePerGenome(hits) {
82
+ const byCode = new Map();
83
+ for (const hit of hits) {
84
+ const current = byCode.get(hit.code);
85
+ if (!current || (current.type === 'O' && hit.type === 'LDO')) {
86
+ byCode.set(hit.code, hit);
87
+ }
88
+ }
89
+ return [...byCode.values()];
90
+ }
91
+ /** `uniprotkb/accessions` -> accession -> sequence */
92
+ export function parseSequences(json) {
93
+ const map = new Map();
94
+ for (const r of json.results ?? []) {
95
+ if (r.primaryAccession && r.sequence?.value) {
96
+ map.set(r.primaryAccession, r.sequence.value);
97
+ }
98
+ }
99
+ return map;
100
+ }
101
+ let genomes;
102
+ /** The proteome list, fetched once per page and forgotten on failure. */
103
+ export function fetchGenomes() {
104
+ genomes ??= jsonfetch(`${PANTHER}/supportedgenomes`)
105
+ .then(parseGenomes)
106
+ .catch((e) => {
107
+ genomes = undefined;
108
+ throw e;
109
+ });
110
+ return genomes;
111
+ }
112
+ // UniProt caps one `accessions` call at 100 ids
113
+ const UNIPROT_CHUNK = 100;
114
+ async function fetchSequences(accessions) {
115
+ const map = new Map();
116
+ for (let i = 0; i < accessions.length; i += UNIPROT_CHUNK) {
117
+ const chunk = accessions.slice(i, i + UNIPROT_CHUNK);
118
+ const json = await jsonfetch(`${UNIPROT}/accessions?accessions=${chunk.join(',')}&fields=accession,sequence&format=json`);
119
+ for (const [acc, seq] of parseSequences(json)) {
120
+ map.set(acc, seq);
121
+ }
122
+ }
123
+ return map;
124
+ }
125
+ // strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53), as
126
+ // resolveGeneId does for NCBI
127
+ function cleanCandidate(raw) {
128
+ return raw
129
+ .trim()
130
+ .replace(/^\w+:/, '')
131
+ .replace(/\.\d+$/, '');
132
+ }
133
+ /**
134
+ * One `matchortho` per candidate until PANTHER maps one. A JBrowse feature
135
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
136
+ * some of those are names PANTHER knows. `targets` omitted asks for every
137
+ * genome PANTHER has, which is one call rather than one per genome.
138
+ */
139
+ async function matchOrthologs(candidates, taxId, targets) {
140
+ let matched;
141
+ for (const raw of candidates) {
142
+ const query = cleanCandidate(raw);
143
+ if (!query) {
144
+ continue;
145
+ }
146
+ const params = new URLSearchParams({
147
+ geneInputList: query,
148
+ organism: String(taxId),
149
+ orthologType: 'all',
150
+ });
151
+ if (targets) {
152
+ params.set('targetOrganism', targets.map(t => t.taxId).join(','));
153
+ }
154
+ const parsed = parseMatches(await jsonfetch(`${PANTHER}/ortholog/matchortho?${params.toString()}`));
155
+ if (!parsed.unmapped) {
156
+ matched ??= query;
157
+ if (parsed.query) {
158
+ return { ...parsed, matched: query };
159
+ }
160
+ }
161
+ }
162
+ return matched === undefined
163
+ ? undefined
164
+ : { matched, hits: [], query: undefined };
165
+ }
166
+ /**
167
+ * The whole PANTHER half of the pipeline: gene -> ortholog rows carrying
168
+ * labels, accessions and sequences, plus the query gene's own protein for the
169
+ * query row. Two lookups (genomes, orthologs), one taxonomy batch for the
170
+ * labels, one UniProt batch for the sequences.
171
+ *
172
+ * `taxa`, `exclude` and `limit` mean what they mean for fetchOrthologRows:
173
+ * `taxa` narrows the targets (omitted, every genome PANTHER has, in its order),
174
+ * `exclude` drops the query taxon, `limit` caps the rows before their
175
+ * sequences are fetched.
176
+ */
177
+ export async function fetchPantherOrthologs({ candidates, taxId, taxa, exclude, limit = defaultMaxSpecies, onProgress, }) {
178
+ const all = await fetchGenomes();
179
+ const byTaxId = new Map(all.map(g => [g.taxId, g]));
180
+ const byCode = new Map(all.map(g => [g.code, g]));
181
+ const queryGenome = byTaxId.get(taxId);
182
+ if (!queryGenome) {
183
+ throw new Error(`PANTHER has no reference proteome for taxon ${taxId}. NCBI orthologs cover vertebrates and insects; try that source.`);
184
+ }
185
+ const targets = taxa
186
+ ? [...taxa].flatMap(t => {
187
+ const g = byTaxId.get(t);
188
+ return g && t !== exclude ? [g] : [];
189
+ })
190
+ : undefined;
191
+ onProgress('Matching orthologs at PANTHER...');
192
+ const match = await matchOrthologs(candidates, taxId, targets);
193
+ if (!match) {
194
+ throw new Error(`PANTHER has no entry for ${candidates.join(', ')} in ${queryGenome.longName}. Try the NCBI BLAST tab, which needs no gene identifier.`);
195
+ }
196
+ const rank = new Map(targets?.map((t, i) => [t.taxId, i]));
197
+ const picks = pickOnePerGenome(match.hits)
198
+ .map(hit => ({ hit, genome: byCode.get(hit.code) }))
199
+ .filter((p) => !!p.genome &&
200
+ p.genome.taxId !== exclude &&
201
+ (targets ? rank.has(p.genome.taxId) : true))
202
+ .sort((a, b) => targets ? rank.get(a.genome.taxId) - rank.get(b.genome.taxId) : 0)
203
+ .slice(0, limit);
204
+ if (picks.length < 2) {
205
+ throw new Error(`Only ${picks.length} PANTHER ortholog(s) found for ${match.matched} — not enough to align`);
206
+ }
207
+ onProgress(`Fetching ${picks.length} protein sequences from UniProt...`);
208
+ const [names, sequences] = await Promise.all([
209
+ taxonomyNames(picks.map(p => p.genome.taxId)),
210
+ fetchSequences([
211
+ ...(match.query ? [match.query.accession] : []),
212
+ ...picks.map(p => p.hit.accession),
213
+ ]),
214
+ ]);
215
+ const described = picks.map(({ hit, genome }) => {
216
+ const info = names.get(genome.taxId);
217
+ return {
218
+ hit,
219
+ genome,
220
+ name: info ? (info.commonName ?? info.sciname) : genome.name,
221
+ scientificName: info?.sciname || genome.longName,
222
+ commonName: info?.commonName,
223
+ };
224
+ });
225
+ const labels = dedupeLabels(described.map(d => d.name));
226
+ const rows = described
227
+ .map(({ hit, genome, scientificName, commonName }, i) => ({
228
+ taxId: genome.taxId,
229
+ label: labels[i],
230
+ scientificName,
231
+ commonName,
232
+ geneId: hit.geneRef,
233
+ protein: hit.accession,
234
+ sequence: sequences.get(hit.accession) ?? '',
235
+ }))
236
+ .filter(r => r.sequence);
237
+ if (rows.length < 2) {
238
+ throw new Error('Could not fetch protein sequences for the orthologs');
239
+ }
240
+ const querySequence = match.query && sequences.get(match.query.accession);
241
+ return {
242
+ matched: match.matched,
243
+ query: match.query && querySequence
244
+ ? { ...match.query, sequence: querySequence }
245
+ : undefined,
246
+ rows,
247
+ };
248
+ }
249
+ /**
250
+ * NCBI's names for the taxa, so PANTHER rows are labelled exactly as NCBI rows
251
+ * are. A failed lookup only costs the labels, which fall back to PANTHER's own
252
+ * short names, so it is logged rather than thrown.
253
+ */
254
+ async function taxonomyNames(taxIds) {
255
+ try {
256
+ return await fetchTaxonomyInfo(taxIds);
257
+ }
258
+ catch (e) {
259
+ console.warn('[msaview-orthologs] taxonomy name lookup failed:', e);
260
+ return new Map();
261
+ }
262
+ }
@@ -0,0 +1,53 @@
1
+ import type { PhmmerDatabase } from '../LaunchMsaView/components/BlastQuery/consts';
2
+ export interface PhmmerRow {
3
+ accession: string;
4
+ id: string;
5
+ sciname: string;
6
+ taxid?: number;
7
+ title?: string;
8
+ /** the matched envelope on the target, e.g. '503-912', absent if unparseable */
9
+ range?: string;
10
+ /** the row as phmmer aligned it, uppercased with '.' inserts turned into '-' */
11
+ aligned: string;
12
+ }
13
+ export interface PhmmerAlignment {
14
+ rows: PhmmerRow[];
15
+ /**
16
+ * the query, placed into the same columns. phmmer does not put the query in
17
+ * its own output, so this is derived — see buildQueryRow.
18
+ */
19
+ queryRow: string;
20
+ }
21
+ /**
22
+ * Human-facing link to a job, shown while it runs and on error.
23
+ *
24
+ * The category has to be sss: jdispatcher serves its shell with a 200 for any
25
+ * category, so /pfa/ and /psa/ look fine to a fetch and render "Page Not Found"
26
+ * in a browser.
27
+ */
28
+ export declare function phmmerResultUrl(jobId: string): string;
29
+ /** EBI job ids are prefixed with the tool that made them */
30
+ export declare function isPhmmerJobId(jobId: string): boolean;
31
+ /**
32
+ * Exported for testing against a captured .sto — the annotation names (RF, the
33
+ * DE line's OS=/OX=) are the whole risk in this mapping, and nothing else in CI
34
+ * would notice if HMMER or EBI changed one.
35
+ */
36
+ export declare function parsePhmmerAlignment({ stockholm, query, }: {
37
+ stockholm: string;
38
+ query: string;
39
+ }): PhmmerAlignment;
40
+ export declare function queryPhmmer({ query, database, onProgress, onRid, }: {
41
+ query: string;
42
+ database: PhmmerDatabase;
43
+ onProgress: (arg: string) => void;
44
+ onRid: (arg: string) => void;
45
+ }): Promise<{
46
+ rows: PhmmerRow[];
47
+ /**
48
+ * the query, placed into the same columns. phmmer does not put the query in
49
+ * its own output, so this is derived — see buildQueryRow.
50
+ */
51
+ queryRow: string;
52
+ rid: string;
53
+ }>;