jbrowse-plugin-msaview 2.8.0 → 2.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/LaunchMsaViewExtensionPoint/index.js +11 -4
- package/dist/MsaViewPanel/doLaunchOrthologs.js +44 -21
- package/dist/MsaViewPanel/model.d.ts +18 -7
- package/dist/jbrowse-plugin-msaview.umd.production.min.js +26 -26
- package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
- package/dist/utils/ncbiOrthologs.js +8 -2
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +2 -2
- package/src/LaunchMsaViewExtensionPoint/index.ts +31 -4
- package/src/MsaViewPanel/doLaunchOrthologs.ts +49 -20
- package/src/MsaViewPanel/model.ts +14 -3
- package/src/utils/ncbiOrthologs.ts +8 -2
- package/src/version.ts +1 -1
|
@@ -23,8 +23,14 @@ const DATASETS = 'https://api.ncbi.nlm.nih.gov/datasets/v2';
|
|
|
23
23
|
const EUTILS = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils';
|
|
24
24
|
// The species panel offered in the launch dialog, ordered from the reference
|
|
25
25
|
// outward so a run that finds only close relatives still reads as a ladder.
|
|
26
|
-
// Orthologs absent for a given gene are skipped rather than erroring
|
|
27
|
-
//
|
|
26
|
+
// Orthologs absent for a given gene are skipped rather than erroring.
|
|
27
|
+
//
|
|
28
|
+
// The index order is the order the sequences are SUBMITTED in
|
|
29
|
+
// (`COMMON_TAX_RANK` below sorts `fetchOrthologGenes`' return), not the order the
|
|
30
|
+
// rows are drawn in: the view lays rows out by the guide tree the aligner returns,
|
|
31
|
+
// so a run on this list comes out grouped by relatedness rather than by this
|
|
32
|
+
// list's own sequence. Reordering here changes what Clustal is handed, not the
|
|
33
|
+
// picture.
|
|
28
34
|
//
|
|
29
35
|
// THE MAMMALS EARN THEIR PLACE, and the reason is measured rather than aesthetic.
|
|
30
36
|
// The thirteen this list used to hold were one per major clade, which reads well
|
package/dist/version.d.ts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export declare const version = "2.8.
|
|
1
|
+
export declare const version = "2.8.2";
|
package/dist/version.js
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export const version = '2.8.
|
|
1
|
+
export const version = '2.8.2';
|
package/package.json
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"version": "2.8.
|
|
2
|
+
"version": "2.8.2",
|
|
3
3
|
"license": "MIT",
|
|
4
4
|
"name": "jbrowse-plugin-msaview",
|
|
5
5
|
"repository": {
|
|
@@ -51,7 +51,7 @@
|
|
|
51
51
|
"puppeteer": "^25.3.0",
|
|
52
52
|
"react": "^19.2.8",
|
|
53
53
|
"react-dom": "^19.2.8",
|
|
54
|
-
"react-msaview": "^5.7.
|
|
54
|
+
"react-msaview": "^5.7.2",
|
|
55
55
|
"rimraf": "^6.1.3",
|
|
56
56
|
"rxjs": "^7.8.2",
|
|
57
57
|
"serve": "^14.2.6",
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { OrthologParams } from '../MsaViewPanel/model'
|
|
1
2
|
import type PluginManager from '@jbrowse/core/PluginManager'
|
|
2
3
|
import type { AbstractSessionModel } from '@jbrowse/core/util'
|
|
3
4
|
|
|
@@ -21,6 +22,23 @@ interface LaunchMsaViewArgs {
|
|
|
21
22
|
showBranchLen?: boolean
|
|
22
23
|
querySeqName?: string
|
|
23
24
|
highlightColumns?: number[]
|
|
25
|
+
/**
|
|
26
|
+
* Build the alignment from NCBI orthologs at launch time instead of naming a
|
|
27
|
+
* file: `{ taxId, geneCandidates }` is enough, and `taxa` and
|
|
28
|
+
* `proteinSequence` both default (see OrthologParams). This is the launch
|
|
29
|
+
* dialog's Orthologs tab, reachable from a session spec — so a link can say
|
|
30
|
+
* "NLRP1 across species" and the view builds it.
|
|
31
|
+
*/
|
|
32
|
+
orthologParams?: OrthologParams
|
|
33
|
+
/**
|
|
34
|
+
* Hide any column gappier than this percentage, 100 being "hide nothing".
|
|
35
|
+
* A native react-msaview property, named here because it is the one setting
|
|
36
|
+
* that decides WHICH columns a freshly launched view opens on: proteins that
|
|
37
|
+
* differ in length put one row's private N-terminal extension at column 0,
|
|
38
|
+
* and everything else is gap there. Anything else react-msaview takes as a
|
|
39
|
+
* snapshot property passes through the same way.
|
|
40
|
+
*/
|
|
41
|
+
allowedGappyness?: number
|
|
24
42
|
}
|
|
25
43
|
|
|
26
44
|
export default function LaunchMsaViewExtensionPointF(
|
|
@@ -40,15 +58,24 @@ export default function LaunchMsaViewExtensionPointF(
|
|
|
40
58
|
...rest
|
|
41
59
|
} = args
|
|
42
60
|
|
|
43
|
-
|
|
61
|
+
// `orthologParams` is a fourth source, and unlike the other three it names
|
|
62
|
+
// no alignment at all — the view builds one from NCBI at launch, which is
|
|
63
|
+
// the dialog's Orthologs tab reached declaratively.
|
|
64
|
+
if (
|
|
65
|
+
!data &&
|
|
66
|
+
!msaFileLocation &&
|
|
67
|
+
!msaIndexedLocation &&
|
|
68
|
+
!rest.orthologParams
|
|
69
|
+
) {
|
|
44
70
|
throw new Error(
|
|
45
|
-
'No MSA data
|
|
71
|
+
'No MSA data, file location or orthologParams provided when launching MSA view',
|
|
46
72
|
)
|
|
47
73
|
}
|
|
48
74
|
|
|
49
75
|
// inline data and the tree URL are native react-msaview snapshot props, set
|
|
50
|
-
// directly
|
|
51
|
-
//
|
|
76
|
+
// directly, and so is orthologParams (the model's own autorun picks it up).
|
|
77
|
+
// Only sources needing launch-time resolution go through `init`: msaUrl
|
|
78
|
+
// (AlphaFold sniff) and the name-indexed bgzip block (no native loader).
|
|
52
79
|
session.addView('MsaView', {
|
|
53
80
|
type: 'MsaView',
|
|
54
81
|
...rest,
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { cleanProteinSequence } from '../LaunchMsaView/util'
|
|
2
2
|
import { launchMSA } from '../utils/msa'
|
|
3
3
|
import {
|
|
4
|
+
COMMON_SPECIES,
|
|
4
5
|
fetchOrthologRows,
|
|
5
6
|
fetchProteinForGene,
|
|
6
7
|
resolveGeneId,
|
|
@@ -31,7 +32,6 @@ export async function doLaunchOrthologs({
|
|
|
31
32
|
}) {
|
|
32
33
|
const { taxId, taxa, geneCandidates, msaAlgorithm, proteinSequence } =
|
|
33
34
|
self.orthologParams!
|
|
34
|
-
const cleanedSeq = cleanProteinSequence(proteinSequence)
|
|
35
35
|
|
|
36
36
|
const onProgress = (arg: string) => {
|
|
37
37
|
self.setProgress(arg)
|
|
@@ -45,8 +45,29 @@ export async function doLaunchOrthologs({
|
|
|
45
45
|
)
|
|
46
46
|
}
|
|
47
47
|
|
|
48
|
-
//
|
|
49
|
-
|
|
48
|
+
// The query row. The dialog always supplies it — it is the user's OWN
|
|
49
|
+
// selected transcript, which is what makes `connectedFeature` map genome
|
|
50
|
+
// coordinates through this row. A launch that has no transcript to translate
|
|
51
|
+
// (a session spec naming only a gene) falls back to NCBI's representative
|
|
52
|
+
// protein for the resolved gene, which is the same choice made for every
|
|
53
|
+
// other row, so the alignment is the one NCBI would build for that gene.
|
|
54
|
+
const representative = await fetchRepresentativeQueryProtein(resolved.geneId)
|
|
55
|
+
const cleanedSeq = proteinSequence
|
|
56
|
+
? cleanProteinSequence(proteinSequence)
|
|
57
|
+
: representative?.sequence
|
|
58
|
+
if (!cleanedSeq) {
|
|
59
|
+
throw new Error(
|
|
60
|
+
`No query protein: none was supplied and NCBI returned no representative protein for gene ${resolved.geneId}.`,
|
|
61
|
+
)
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
// Every species the panel offers, when a launch names none. A spec that wants
|
|
65
|
+
// a narrower comparison says so; one that just wants "this gene across
|
|
66
|
+
// species" should not have to enumerate the list the dialog would have
|
|
67
|
+
// checked for it.
|
|
68
|
+
const wantedTaxa = taxa ?? COMMON_SPECIES.map(s => s.taxId as number)
|
|
69
|
+
// the query species is represented by the query row above
|
|
70
|
+
const wanted = new Set(wantedTaxa.filter(t => t !== taxId))
|
|
50
71
|
const rows = await fetchOrthologRows({
|
|
51
72
|
geneId: resolved.geneId,
|
|
52
73
|
taxa: wanted,
|
|
@@ -54,7 +75,7 @@ export async function doLaunchOrthologs({
|
|
|
54
75
|
})
|
|
55
76
|
|
|
56
77
|
const treeMetadata: Record<string, Record<string, string>> = {
|
|
57
|
-
QUERY:
|
|
78
|
+
QUERY: buildQueryMetadata(self, resolved.geneId, cleanedSeq, representative),
|
|
58
79
|
}
|
|
59
80
|
for (const row of rows) {
|
|
60
81
|
treeMetadata[row.label] = buildRowMetadata(row)
|
|
@@ -76,34 +97,42 @@ export async function doLaunchOrthologs({
|
|
|
76
97
|
}
|
|
77
98
|
|
|
78
99
|
/**
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
*
|
|
82
|
-
|
|
100
|
+
* A failed lookup only costs the query row its domain overlay and, for a launch
|
|
101
|
+
* that supplied no sequence of its own, the alignment — so it is reported by
|
|
102
|
+
* returning nothing rather than by throwing here.
|
|
103
|
+
*/
|
|
104
|
+
async function fetchRepresentativeQueryProtein(geneId: string) {
|
|
105
|
+
try {
|
|
106
|
+
return await fetchProteinForGene(geneId)
|
|
107
|
+
} catch (e) {
|
|
108
|
+
console.warn('[msaview-orthologs] query protein lookup failed:', e)
|
|
109
|
+
return undefined
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* The query row carries an Accession — which is what drives the automatic CDD
|
|
115
|
+
* overlay (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains) —
|
|
116
|
+
* ONLY when its sequence is byte-identical to the RefSeq protein that accession
|
|
83
117
|
* names. Attaching it unconditionally would put every domain box at an offset
|
|
84
118
|
* whenever the user picked a non-representative isoform, which is a silently
|
|
85
|
-
* wrong figure rather than a missing one.
|
|
119
|
+
* wrong figure rather than a missing one. A launch that took the representative
|
|
120
|
+
* protein as its query row passes that test by construction.
|
|
86
121
|
*/
|
|
87
|
-
|
|
122
|
+
function buildQueryMetadata(
|
|
88
123
|
self: JBrowsePluginMsaViewModel,
|
|
89
124
|
geneId: string,
|
|
90
125
|
proteinSequence: string,
|
|
91
|
-
|
|
126
|
+
representative: { accession: string; sequence: string } | undefined,
|
|
127
|
+
): Record<string, string> {
|
|
92
128
|
const transcript = self.orthologParams?.selectedTranscript
|
|
93
129
|
const metadata: Record<string, string> = { 'Gene ID': geneId }
|
|
94
130
|
const name = transcript?.get('name') ?? transcript?.get('id')
|
|
95
131
|
if (name) {
|
|
96
132
|
metadata.Transcript = name
|
|
97
133
|
}
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
if (representative?.sequence === proteinSequence) {
|
|
101
|
-
metadata.Accession = representative.accession
|
|
102
|
-
}
|
|
103
|
-
} catch (e) {
|
|
104
|
-
// a failed lookup only costs the query row its domain overlay, so it must
|
|
105
|
-
// not take down an alignment that is otherwise complete
|
|
106
|
-
console.warn('[msaview-orthologs] query protein lookup failed:', e)
|
|
134
|
+
if (representative?.sequence === proteinSequence) {
|
|
135
|
+
metadata.Accession = representative.accession
|
|
107
136
|
}
|
|
108
137
|
return metadata
|
|
109
138
|
}
|
|
@@ -58,13 +58,24 @@ export interface BlastParams {
|
|
|
58
58
|
export interface OrthologParams {
|
|
59
59
|
/** NCBI taxon id of the assembly the query gene came from */
|
|
60
60
|
taxId: number
|
|
61
|
-
/**
|
|
62
|
-
|
|
61
|
+
/**
|
|
62
|
+
* taxon ids to include as rows (the query taxon is represented by QUERY).
|
|
63
|
+
* Omitted means every species the launch dialog offers, which is what a
|
|
64
|
+
* launch that just wants "this gene across species" wants.
|
|
65
|
+
*/
|
|
66
|
+
taxa?: number[]
|
|
63
67
|
/** candidate gene identifiers off the feature, tried in order */
|
|
64
68
|
geneCandidates: string[]
|
|
65
69
|
msaAlgorithm: MsaAlgorithm
|
|
66
70
|
selectedTranscript?: Feature
|
|
67
|
-
|
|
71
|
+
/**
|
|
72
|
+
* The QUERY row. The launch dialog always supplies it, translated from the
|
|
73
|
+
* transcript the user picked, which is what `connectedFeature` maps genome
|
|
74
|
+
* coordinates through. Omitted — a session spec naming a gene and nothing
|
|
75
|
+
* else — the query row becomes NCBI's representative protein for the resolved
|
|
76
|
+
* gene, the same choice every other row makes.
|
|
77
|
+
*/
|
|
78
|
+
proteinSequence?: string
|
|
68
79
|
}
|
|
69
80
|
|
|
70
81
|
/**
|
|
@@ -26,8 +26,14 @@ const EUTILS = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils'
|
|
|
26
26
|
|
|
27
27
|
// The species panel offered in the launch dialog, ordered from the reference
|
|
28
28
|
// outward so a run that finds only close relatives still reads as a ladder.
|
|
29
|
-
// Orthologs absent for a given gene are skipped rather than erroring
|
|
30
|
-
//
|
|
29
|
+
// Orthologs absent for a given gene are skipped rather than erroring.
|
|
30
|
+
//
|
|
31
|
+
// The index order is the order the sequences are SUBMITTED in
|
|
32
|
+
// (`COMMON_TAX_RANK` below sorts `fetchOrthologGenes`' return), not the order the
|
|
33
|
+
// rows are drawn in: the view lays rows out by the guide tree the aligner returns,
|
|
34
|
+
// so a run on this list comes out grouped by relatedness rather than by this
|
|
35
|
+
// list's own sequence. Reordering here changes what Clustal is handed, not the
|
|
36
|
+
// picture.
|
|
31
37
|
//
|
|
32
38
|
// THE MAMMALS EARN THEIR PLACE, and the reason is measured rather than aesthetic.
|
|
33
39
|
// The thirteen this list used to hold were one per major clade, which reads well
|
package/src/version.ts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export const version = '2.8.
|
|
1
|
+
export const version = '2.8.2'
|