jbrowse-plugin-msaview 2.7.3 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/dist/LaunchMsaView/components/LaunchMsaViewDialog.js +7 -1
  2. package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.d.ts +8 -0
  3. package/dist/LaunchMsaView/components/OrthologQuery/OrthologPanel.js +89 -0
  4. package/dist/LaunchMsaView/components/OrthologQuery/orthologLaunchView.d.ts +9 -0
  5. package/dist/LaunchMsaView/components/OrthologQuery/orthologLaunchView.js +13 -0
  6. package/dist/LaunchMsaView/components/TranscriptSelector.js +7 -1
  7. package/dist/MsaViewPanel/afterCreateAutoruns.d.ts +8 -0
  8. package/dist/MsaViewPanel/afterCreateAutoruns.js +28 -0
  9. package/dist/MsaViewPanel/doLaunchOrthologs.d.ts +23 -0
  10. package/dist/MsaViewPanel/doLaunchOrthologs.js +97 -0
  11. package/dist/MsaViewPanel/model.d.ts +21 -5
  12. package/dist/MsaViewPanel/model.js +12 -1
  13. package/dist/jbrowse-plugin-msaview.umd.production.min.js +31 -27
  14. package/dist/jbrowse-plugin-msaview.umd.production.min.js.map +4 -4
  15. package/dist/utils/ncbiOrthologs.d.ts +135 -0
  16. package/dist/utils/ncbiOrthologs.js +241 -0
  17. package/dist/utils/ncbiOrthologs.test.d.ts +1 -0
  18. package/dist/utils/ncbiOrthologs.test.js +41 -0
  19. package/dist/version.d.ts +1 -1
  20. package/dist/version.js +1 -1
  21. package/package.json +3 -3
  22. package/src/LaunchMsaView/components/LaunchMsaViewDialog.tsx +13 -2
  23. package/src/LaunchMsaView/components/OrthologQuery/OrthologPanel.tsx +172 -0
  24. package/src/LaunchMsaView/components/OrthologQuery/orthologLaunchView.ts +28 -0
  25. package/src/LaunchMsaView/components/TranscriptSelector.tsx +6 -0
  26. package/src/MsaViewPanel/afterCreateAutoruns.ts +27 -0
  27. package/src/MsaViewPanel/doLaunchOrthologs.ts +123 -0
  28. package/src/MsaViewPanel/model.ts +24 -0
  29. package/src/utils/ncbiOrthologs.test.ts +56 -0
  30. package/src/utils/ncbiOrthologs.ts +350 -0
  31. package/src/version.ts +1 -1
@@ -0,0 +1,28 @@
1
+ import { getSession } from '@jbrowse/core/util'
2
+
3
+ import type { OrthologParams } from '../../../MsaViewPanel/model'
4
+ import type { Feature } from '@jbrowse/core/util'
5
+ import type { LinearGenomeViewModel } from '@jbrowse/plugin-linear-genome-view'
6
+
7
+ export function orthologLaunchView({
8
+ newViewTitle,
9
+ view,
10
+ feature,
11
+ orthologParams,
12
+ }: {
13
+ newViewTitle: string
14
+ view: LinearGenomeViewModel
15
+ feature: Feature
16
+ orthologParams: OrthologParams
17
+ }) {
18
+ getSession(view).addView('MsaView', {
19
+ type: 'MsaView',
20
+ displayName: newViewTitle,
21
+ connectedViewId: view.id,
22
+ connectedFeature: feature.toJSON(),
23
+ drawNodeBubbles: true,
24
+ colWidth: 10,
25
+ rowHeight: 12,
26
+ orthologParams,
27
+ })
28
+ }
@@ -53,6 +53,12 @@ export default function TranscriptSelector({
53
53
  <TextField
54
54
  variant="outlined"
55
55
  label={`Choose isoform of ${getGeneDisplayName(feature)}`}
56
+ // The query row is this transcript rather than NCBI's representative
57
+ // protein, which is what keeps the alignment linked to the genome view
58
+ // at codon resolution. It used to be a paragraph under the panel; as
59
+ // helper text it says the same thing where the choice is made and
60
+ // costs no height of its own.
61
+ helperText="the query row, so the alignment stays linked to the genome view"
56
62
  select
57
63
  className={classes.minWidth}
58
64
  value={selectedId}
@@ -1,6 +1,7 @@
1
1
  import { getSession } from '@jbrowse/core/util'
2
2
 
3
3
  import { doLaunchBlast } from './doLaunchBlast'
4
+ import { doLaunchOrthologs } from './doLaunchOrthologs'
4
5
  import { fetchIndexedMsa } from './fetchIndexedMsa'
5
6
  import { genomeToMSA } from './genomeToMSA'
6
7
  import { loadProteinDomains } from './loadProteinDomains'
@@ -78,6 +79,32 @@ export function storeDataToIndexedDB(self: JBrowsePluginMsaViewModel) {
78
79
  }
79
80
  }
80
81
 
82
+ /**
83
+ * Same shape as launchBlastIfNeeded, for the ortholog path: the params ARE the
84
+ * request, and clearing them on success is what marks it done. They are left in
85
+ * place on failure so the error stays attributable to a specific request; the
86
+ * autorun's only tracked read is orthologParams itself, so nothing refires
87
+ * until a new request replaces them.
88
+ */
89
+ export function launchOrthologsIfNeeded(self: JBrowsePluginMsaViewModel) {
90
+ if (self.orthologParams) {
91
+ void (async () => {
92
+ try {
93
+ self.setProgress('Resolving orthologs')
94
+ self.setError(undefined)
95
+ const data = await doLaunchOrthologs({ self })
96
+ self.setData(data)
97
+ self.setOrthologParams(undefined)
98
+ } catch (e) {
99
+ self.setError(e)
100
+ console.error(e)
101
+ } finally {
102
+ self.setProgress('')
103
+ }
104
+ })()
105
+ }
106
+ }
107
+
81
108
  export function launchBlastIfNeeded(self: JBrowsePluginMsaViewModel) {
82
109
  if (self.blastParams) {
83
110
  void (async () => {
@@ -0,0 +1,123 @@
1
+ import { cleanProteinSequence } from '../LaunchMsaView/util'
2
+ import { launchMSA } from '../utils/msa'
3
+ import {
4
+ fetchOrthologRows,
5
+ fetchProteinForGene,
6
+ resolveGeneId,
7
+ } from '../utils/ncbiOrthologs'
8
+
9
+ import type { JBrowsePluginMsaViewModel } from './model'
10
+ import type { OrthologRow } from '../utils/ncbiOrthologs'
11
+
12
+ /**
13
+ * The no-search-job alternative to doLaunchBlast.
14
+ *
15
+ * BLAST spends 10+ minutes answering "what looks like this sequence" and
16
+ * returns a redundant, accession-labelled hit list. This asks NCBI the question
17
+ * the alignment actually wants — "what is this gene's ortholog in each species"
18
+ * — which NCBI has already computed, so the whole NCBI half returns in about a
19
+ * second and only the EBI alignment (~10s) costs real time.
20
+ *
21
+ * The query row is the user's OWN selected transcript, not NCBI's
22
+ * representative protein for the query species, because `connectedFeature`
23
+ * maps genome coordinates through that row — swapping in a different isoform
24
+ * would silently break the genome<->MSA linkage. The query species is therefore
25
+ * excluded from the ortholog set rather than appearing twice.
26
+ */
27
+ export async function doLaunchOrthologs({
28
+ self,
29
+ }: {
30
+ self: JBrowsePluginMsaViewModel
31
+ }) {
32
+ const { taxId, taxa, geneCandidates, msaAlgorithm, proteinSequence } =
33
+ self.orthologParams!
34
+ const cleanedSeq = cleanProteinSequence(proteinSequence)
35
+
36
+ const onProgress = (arg: string) => {
37
+ self.setProgress(arg)
38
+ }
39
+
40
+ onProgress('Resolving gene at NCBI...')
41
+ const resolved = await resolveGeneId(geneCandidates, taxId)
42
+ if (!resolved) {
43
+ throw new Error(
44
+ `Could not resolve any of ${geneCandidates.join(', ')} to an NCBI gene in taxon ${taxId}. Try the NCBI BLAST tab, which needs no gene identifier.`,
45
+ )
46
+ }
47
+
48
+ // the query species is represented by the user's own transcript below
49
+ const wanted = new Set(taxa.filter(t => t !== taxId))
50
+ const rows = await fetchOrthologRows({
51
+ geneId: resolved.geneId,
52
+ taxa: wanted,
53
+ onProgress,
54
+ })
55
+
56
+ const treeMetadata: Record<string, Record<string, string>> = {
57
+ QUERY: await buildQueryMetadata(self, resolved.geneId, cleanedSeq),
58
+ }
59
+ for (const row of rows) {
60
+ treeMetadata[row.label] = buildRowMetadata(row)
61
+ }
62
+
63
+ const result = await launchMSA({
64
+ algorithm: msaAlgorithm,
65
+ sequence: [
66
+ `>QUERY\n${cleanedSeq}`,
67
+ ...rows.map(r => `>${r.label}\n${r.sequence}`),
68
+ ].join('\n'),
69
+ onProgress,
70
+ })
71
+
72
+ return {
73
+ ...result,
74
+ treeMetadata: JSON.stringify(treeMetadata),
75
+ }
76
+ }
77
+
78
+ /**
79
+ * The query row is the user's own translated transcript, so it carries an
80
+ * Accession — which is what drives the automatic CDD overlay
81
+ * (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains) — ONLY
82
+ * when its sequence is byte-identical to the RefSeq protein that accession
83
+ * names. Attaching it unconditionally would put every domain box at an offset
84
+ * whenever the user picked a non-representative isoform, which is a silently
85
+ * wrong figure rather than a missing one.
86
+ */
87
+ async function buildQueryMetadata(
88
+ self: JBrowsePluginMsaViewModel,
89
+ geneId: string,
90
+ proteinSequence: string,
91
+ ): Promise<Record<string, string>> {
92
+ const transcript = self.orthologParams?.selectedTranscript
93
+ const metadata: Record<string, string> = { 'Gene ID': geneId }
94
+ const name = transcript?.get('name') ?? transcript?.get('id')
95
+ if (name) {
96
+ metadata.Transcript = name
97
+ }
98
+ try {
99
+ const representative = await fetchProteinForGene(geneId)
100
+ if (representative?.sequence === proteinSequence) {
101
+ metadata.Accession = representative.accession
102
+ }
103
+ } catch (e) {
104
+ // a failed lookup only costs the query row its domain overlay, so it must
105
+ // not take down an alignment that is otherwise complete
106
+ console.warn('[msaview-orthologs] query protein lookup failed:', e)
107
+ }
108
+ return metadata
109
+ }
110
+
111
+ function buildRowMetadata(row: OrthologRow): Record<string, string> {
112
+ const metadata: Record<string, string> = {
113
+ 'Scientific name': row.scientificName,
114
+ // Accession drives the automatic CDD domain overlay
115
+ // (afterCreateAutoruns.autoLoadProteinDomains -> loadProteinDomains)
116
+ Accession: row.protein,
117
+ 'Gene ID': row.geneId,
118
+ }
119
+ if (row.commonName) {
120
+ metadata['Common name'] = row.commonName
121
+ }
122
+ return metadata
123
+ }
@@ -12,6 +12,7 @@ export type { MSAFormat } from 'msa-parsers'
12
12
  import {
13
13
  autoLoadProteinDomains,
14
14
  launchBlastIfNeeded,
15
+ launchOrthologsIfNeeded,
15
16
  loadStoredData,
16
17
  observeProteinHighlights,
17
18
  processInit,
@@ -54,6 +55,18 @@ export interface BlastParams {
54
55
  rid?: string
55
56
  }
56
57
 
58
+ export interface OrthologParams {
59
+ /** NCBI taxon id of the assembly the query gene came from */
60
+ taxId: number
61
+ /** taxon ids to include as rows (the query taxon is represented by QUERY) */
62
+ taxa: number[]
63
+ /** candidate gene identifiers off the feature, tried in order */
64
+ geneCandidates: string[]
65
+ msaAlgorithm: MsaAlgorithm
66
+ selectedTranscript?: Feature
67
+ proteinSequence: string
68
+ }
69
+
57
70
  /**
58
71
  * #stateModel MsaViewPlugin
59
72
  * extends
@@ -77,6 +90,10 @@ export default function stateModelFactory() {
77
90
  * #property
78
91
  */
79
92
  blastParams: types.frozen<BlastParams | undefined>(),
93
+ /**
94
+ * #property
95
+ */
96
+ orthologParams: types.frozen<OrthologParams | undefined>(),
80
97
  /**
81
98
  * #property
82
99
  */
@@ -256,6 +273,12 @@ export default function stateModelFactory() {
256
273
  setBlastParams(args?: BlastParams) {
257
274
  self.blastParams = args
258
275
  },
276
+ /**
277
+ * #action
278
+ */
279
+ setOrthologParams(args?: OrthologParams) {
280
+ self.orthologParams = args
281
+ },
259
282
  /**
260
283
  * #action
261
284
  */
@@ -366,6 +389,7 @@ export default function stateModelFactory() {
366
389
  loadStoredData,
367
390
  storeDataToIndexedDB,
368
391
  launchBlastIfNeeded,
392
+ launchOrthologsIfNeeded,
369
393
  processInit,
370
394
  autoLoadProteinDomains,
371
395
  ]) {
@@ -0,0 +1,56 @@
1
+ import { describe, expect, test } from 'vitest'
2
+
3
+ import { dedupeLabels, parseFasta } from './ncbiOrthologs'
4
+
5
+ describe('dedupeLabels', () => {
6
+ test('sanitizes to single tokens', () => {
7
+ // labels are used identically as FASTA headers, Newick leaf names and GFF
8
+ // seq_ids, so anything that would need quoting in one of those is stripped
9
+ expect(dedupeLabels(['house mouse', 'Norway rat'])).toEqual([
10
+ 'house_mouse',
11
+ 'Norway_rat',
12
+ ])
13
+ expect(dedupeLabels(['Frog (X. tropicalis)'])).toEqual([
14
+ 'Frog_X_tropicalis',
15
+ ])
16
+ })
17
+
18
+ test('suffixes collisions rather than overwriting a row', () => {
19
+ expect(dedupeLabels(['a b', 'a-b', 'a_b'])).toEqual([
20
+ 'a_b',
21
+ 'a_b_2',
22
+ 'a_b_3',
23
+ ])
24
+ })
25
+
26
+ test('falls back for a name with no usable characters', () => {
27
+ expect(dedupeLabels(['...', '...'])).toEqual(['row', 'row_2'])
28
+ })
29
+ })
30
+
31
+ describe('parseFasta', () => {
32
+ test('keys by the first header token and joins wrapped lines', () => {
33
+ const map = parseFasta(
34
+ ['>NP_000537.3 cellular tumor antigen p53', 'MEEP', 'QSDP', ''].join(
35
+ '\n',
36
+ ),
37
+ )
38
+ expect(map.get('NP_000537.3')).toBe('MEEPQSDP')
39
+ })
40
+
41
+ test('reads every record of a multi-FASTA', () => {
42
+ const map = parseFasta(
43
+ ['>A one', 'MMM', '>B two', 'KKK', '>C three', 'LLL'].join('\n'),
44
+ )
45
+ expect([...map.keys()]).toEqual(['A', 'B', 'C'])
46
+ expect(map.get('C')).toBe('LLL')
47
+ })
48
+
49
+ test('returns nothing for a response that carried no records', () => {
50
+ // efetch answers an unknown accession with an error body, not a 4xx, so a
51
+ // caller that assumed "text back = sequences" would build empty rows
52
+ expect(parseFasta('Error: CEFetchPApplication::proxy_stream()').size).toBe(
53
+ 0,
54
+ )
55
+ })
56
+ })
@@ -0,0 +1,350 @@
1
+ // Homolog discovery WITHOUT a search job.
2
+ //
3
+ // The BLAST path answers "what looks like this sequence", which is not the
4
+ // question an MSA row set wants — it wants "what is homologous to this gene,
5
+ // one per species, labelled by species". BLAST then costs 10+ minutes to
6
+ // return a redundant, accession-labelled hit list that has to be deduplicated
7
+ // before it reads. NCBI has already computed the answer: the Datasets
8
+ // orthologs endpoint returns one ortholog gene per species, instantly.
9
+ //
10
+ // gene symbol -> gene id -> orthologs -> a representative protein each ->
11
+ // sequences, all from NCBI, in a handful of requests. The caller aligns them
12
+ // (EBI Clustal Omega, ~10s) and overlays CDD domains, which are already baked
13
+ // into the GenPept records (see ncbiDomains.ts).
14
+ //
15
+ // Mirrors jb2hubs' website/src/components/proteinMsa.ts assembler, trimmed to
16
+ // what the launch dialog needs and using this plugin's fetch/eutils helpers.
17
+
18
+ import { NCBI_EMAIL, NCBI_TOOL } from './eutils'
19
+ import { jsonfetch, textfetch } from './fetch'
20
+
21
+ // v2, not v2alpha: the alpha path still answers /orthologs but 404s
22
+ // /product_report, so an assembler pointed at it silently resolves zero
23
+ // representative proteins and reports "no orthologs" for every gene.
24
+ const DATASETS = 'https://api.ncbi.nlm.nih.gov/datasets/v2'
25
+ const EUTILS = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils'
26
+
27
+ // The species panel offered in the launch dialog, ordered from the reference
28
+ // outward so a run that finds only close relatives still reads as a ladder.
29
+ // Orthologs absent for a given gene are skipped rather than erroring, and the
30
+ // index order here is the ROW order of the alignment (COMMON_TAX_RANK below).
31
+ //
32
+ // THE MAMMALS EARN THEIR PLACE, and the reason is measured rather than aesthetic.
33
+ // The thirteen this list used to hold were one per major clade, which reads well
34
+ // on a gene conserved to yeast and produces almost nothing on a gene that is not:
35
+ // NCBI publishes 165 orthologs for human NLRP1 and every one of them is a mammal,
36
+ // so of the old thirteen only Human, Mouse, Cow, Pig and Dog returned a row --
37
+ // five, and Rat not among them, since NLRP1 is absent in Rattus norvegicus. The
38
+ // same query against this list returns twelve. An inflammasome gene is not an
39
+ // unusual case; anything immune, reproductive or lineage-specific behaves the
40
+ // same way, and those are the genes a person opens an ortholog alignment on.
41
+ //
42
+ // Cat, rabbit and opossum are here despite contributing nothing to that gene.
43
+ // They are the three that most often separate "absent in this clade" from
44
+ // "absent in this species", which is the question a gap in the alignment raises.
45
+ //
46
+ // The cost is the run, and it is roughly linear: one NCBI protein fetch per
47
+ // species and a Clustal Omega job over what comes back, so ~23 rows is about
48
+ // twice the ~13-row wait. Still seconds rather than the minutes BLAST takes,
49
+ // which is the comparison the panel's own text makes.
50
+ export const COMMON_SPECIES = [
51
+ { label: 'Human', taxId: 9606 },
52
+ { label: 'Chimpanzee', taxId: 9598 },
53
+ { label: 'Gorilla', taxId: 9595 },
54
+ { label: 'Rhesus macaque', taxId: 9544 },
55
+ { label: 'Marmoset', taxId: 9483 },
56
+ { label: 'Mouse', taxId: 10090 },
57
+ { label: 'Rat', taxId: 10116 },
58
+ { label: 'Guinea pig', taxId: 10141 },
59
+ { label: 'Rabbit', taxId: 9986 },
60
+ { label: 'Cat', taxId: 9685 },
61
+ { label: 'Dog', taxId: 9615 },
62
+ { label: 'Horse', taxId: 9796 },
63
+ { label: 'Pig', taxId: 9823 },
64
+ { label: 'Cow', taxId: 9913 },
65
+ { label: 'Sheep', taxId: 9940 },
66
+ { label: 'Opossum', taxId: 13616 },
67
+ { label: 'Chicken', taxId: 9031 },
68
+ { label: 'Frog', taxId: 8364 },
69
+ { label: 'Zebrafish', taxId: 7955 },
70
+ { label: 'Fruitfly', taxId: 7227 },
71
+ { label: 'C. elegans', taxId: 6239 },
72
+ { label: 'Yeast', taxId: 4932 },
73
+ { label: 'Arabidopsis', taxId: 3702 },
74
+ ] as const
75
+
76
+ export const COMMON_TAX_RANK = new Map(
77
+ COMMON_SPECIES.map((s, i) => [s.taxId as number, i]),
78
+ )
79
+
80
+ export interface OrthologRow {
81
+ taxId: number
82
+ /** single-token id used identically in the FASTA, the tree and the domain GFF */
83
+ label: string
84
+ scientificName: string
85
+ commonName?: string
86
+ geneId: string
87
+ /** accession.version */
88
+ protein: string
89
+ sequence: string
90
+ }
91
+
92
+ function ncbiUrl(url: string) {
93
+ const sep = url.includes('?') ? '&' : '?'
94
+ return `${url}${sep}tool=${NCBI_TOOL}&email=${encodeURIComponent(NCBI_EMAIL)}`
95
+ }
96
+
97
+ /**
98
+ * A free-text gene reference -> NCBI gene id. A bare number is taken as the id
99
+ * itself; anything else is searched as a gene name within the query taxon.
100
+ * Several candidate identifiers are tried in order, because a JBrowse feature
101
+ * carries whatever its GFF/BigBed had — `id()`, `name`, `gene_name` — and only
102
+ * some of those are real symbols.
103
+ */
104
+ export async function resolveGeneId(
105
+ candidates: string[],
106
+ taxId: number,
107
+ ): Promise<{ geneId: string; matched: string } | undefined> {
108
+ for (const raw of candidates) {
109
+ const query = raw.trim()
110
+ if (!query) {
111
+ continue
112
+ }
113
+ if (/^\d+$/.test(query)) {
114
+ return { geneId: query, matched: query }
115
+ }
116
+ // strip a version suffix (NM_000546.6) and any GFF ID prefix (gene:TP53)
117
+ const cleaned = query.replace(/^\w+:/, '').replace(/\.\d+$/, '')
118
+ const term = `${cleaned}[Gene Name] AND ${taxId}[taxid]`
119
+ const json = await jsonfetch<{
120
+ esearchresult?: { idlist?: string[] }
121
+ }>(
122
+ ncbiUrl(
123
+ `${EUTILS}/esearch.fcgi?db=gene&term=${encodeURIComponent(term)}&retmode=json&retmax=1`,
124
+ ),
125
+ )
126
+ const geneId = json.esearchresult?.idlist?.[0]
127
+ if (geneId) {
128
+ return { geneId, matched: cleaned }
129
+ }
130
+ }
131
+ return undefined
132
+ }
133
+
134
+ interface OrthologReport {
135
+ reports?: {
136
+ gene?: {
137
+ gene_id?: string
138
+ tax_id?: string | number
139
+ taxname?: string
140
+ common_name?: string
141
+ }
142
+ }[]
143
+ }
144
+
145
+ /** One ortholog gene per species, restricted to the requested taxa. */
146
+ export async function fetchOrthologGenes(geneId: string, taxa: Set<number>) {
147
+ const json = await jsonfetch<OrthologReport>(
148
+ ncbiUrl(
149
+ `${DATASETS}/gene/id/${geneId}/orthologs?returned_content=COMPLETE`,
150
+ ),
151
+ )
152
+ const byTaxon = new Map<
153
+ number,
154
+ {
155
+ taxId: number
156
+ geneId: string
157
+ scientificName: string
158
+ commonName?: string
159
+ }
160
+ >()
161
+ for (const { gene } of json.reports ?? []) {
162
+ const taxId = Number(gene?.tax_id)
163
+ if (gene?.gene_id && taxa.has(taxId) && !byTaxon.has(taxId)) {
164
+ byTaxon.set(taxId, {
165
+ taxId,
166
+ geneId: gene.gene_id,
167
+ scientificName: gene.taxname ?? String(taxId),
168
+ commonName: gene.common_name,
169
+ })
170
+ }
171
+ }
172
+ return [...byTaxon.values()].sort(
173
+ (a, b) =>
174
+ (COMMON_TAX_RANK.get(a.taxId) ?? Infinity) -
175
+ (COMMON_TAX_RANK.get(b.taxId) ?? Infinity),
176
+ )
177
+ }
178
+
179
+ interface ProductReport {
180
+ reports?: {
181
+ product?: {
182
+ gene_id?: string
183
+ transcripts?: {
184
+ select_category?: string
185
+ protein?: { accession_version?: string; length?: number }
186
+ }[]
187
+ }
188
+ }[]
189
+ }
190
+
191
+ /**
192
+ * geneId -> representative protein accession: MANE Select where flagged, else
193
+ * the longest isoform. A stable, comparable choice across species — picking
194
+ * "the first" would silently vary with NCBI's ordering.
195
+ */
196
+ export async function fetchRepresentativeProteins(geneIds: string[]) {
197
+ const byGene = new Map<string, string>()
198
+ if (geneIds.length > 0) {
199
+ const json = await jsonfetch<ProductReport>(
200
+ ncbiUrl(`${DATASETS}/gene/id/${geneIds.join(',')}/product_report`),
201
+ )
202
+ for (const { product } of json.reports ?? []) {
203
+ const candidates = (product?.transcripts ?? [])
204
+ .map(t => ({
205
+ acc: t.protein?.accession_version,
206
+ len: t.protein?.length ?? 0,
207
+ mane: /select/i.test(t.select_category ?? ''),
208
+ }))
209
+ .filter(
210
+ (c): c is { acc: string; len: number; mane: boolean } => !!c.acc,
211
+ )
212
+ const best =
213
+ candidates.find(c => c.mane) ??
214
+ [...candidates].sort((a, b) => b.len - a.len).at(0)
215
+ if (product?.gene_id && best) {
216
+ byGene.set(product.gene_id, best.acc)
217
+ }
218
+ }
219
+ }
220
+ return byGene
221
+ }
222
+
223
+ /** accession (first header token) -> ungapped sequence, from a multi-FASTA. */
224
+ export function parseFasta(text: string) {
225
+ const map = new Map<string, string>()
226
+ let acc: string | undefined
227
+ let buf: string[] = []
228
+ for (const line of text.split('\n')) {
229
+ if (line.startsWith('>')) {
230
+ if (acc) {
231
+ map.set(acc, buf.join(''))
232
+ }
233
+ acc = line.slice(1).split(/\s+/)[0]
234
+ buf = []
235
+ } else {
236
+ buf.push(line.trim())
237
+ }
238
+ }
239
+ if (acc) {
240
+ map.set(acc, buf.join(''))
241
+ }
242
+ return map
243
+ }
244
+
245
+ function sanitize(name: string) {
246
+ return name.replace(/[^A-Za-z0-9]+/g, '_').replace(/^_+|_+$/g, '')
247
+ }
248
+
249
+ /**
250
+ * Sanitized, unique single-token labels used identically in the FASTA headers,
251
+ * the tree leaf names and the domain GFF seq_ids — that identity is how the
252
+ * viewer pairs a tree leaf to its alignment row to its domain track. Collisions
253
+ * get a numeric suffix rather than silently overwriting a row.
254
+ */
255
+ export function dedupeLabels(names: string[]) {
256
+ const seen = new Map<string, number>()
257
+ return names.map(name => {
258
+ const base = sanitize(name) || 'row'
259
+ const n = seen.get(base) ?? 0
260
+ seen.set(base, n + 1)
261
+ return n === 0 ? base : `${base}_${n + 1}`
262
+ })
263
+ }
264
+
265
+ /**
266
+ * The representative protein for a single gene, with its sequence. Used to
267
+ * decide whether the user's own translated transcript is byte-identical to the
268
+ * RefSeq protein — if it is, that accession's precomputed CDD domains apply to
269
+ * the query row exactly, and if it isn't, they would land at an offset.
270
+ */
271
+ export async function fetchProteinForGene(geneId: string) {
272
+ const acc = (await fetchRepresentativeProteins([geneId])).get(geneId)
273
+ if (!acc) {
274
+ return undefined
275
+ }
276
+ const seq = parseFasta(
277
+ await textfetch(
278
+ ncbiUrl(
279
+ `${EUTILS}/efetch.fcgi?db=protein&id=${acc}&rettype=fasta&retmode=text`,
280
+ ),
281
+ ),
282
+ ).get(acc)
283
+ return seq ? { accession: acc, sequence: seq } : undefined
284
+ }
285
+
286
+ /**
287
+ * The whole NCBI half of the pipeline: gene -> ortholog rows carrying labels,
288
+ * accessions and sequences. Everything here is a precomputed lookup, so this
289
+ * returns in seconds rather than the 10+ minutes a BLAST submission costs.
290
+ */
291
+ export async function fetchOrthologRows({
292
+ geneId,
293
+ taxa,
294
+ onProgress,
295
+ }: {
296
+ geneId: string
297
+ taxa: Set<number>
298
+ onProgress: (arg: string) => void
299
+ }): Promise<OrthologRow[]> {
300
+ onProgress('Finding orthologs across species...')
301
+ const genes = await fetchOrthologGenes(geneId, taxa)
302
+ if (genes.length < 2) {
303
+ throw new Error(
304
+ `Only ${genes.length} ortholog(s) found among the selected species — not enough to align`,
305
+ )
306
+ }
307
+
308
+ onProgress('Selecting a representative protein per species...')
309
+ const proteinByGene = await fetchRepresentativeProteins(
310
+ genes.map(g => g.geneId),
311
+ )
312
+ const withProtein = genes.filter(g => proteinByGene.has(g.geneId))
313
+ if (withProtein.length < 2) {
314
+ throw new Error(
315
+ 'Could not resolve representative proteins for the orthologs',
316
+ )
317
+ }
318
+
319
+ onProgress(`Fetching ${withProtein.length} protein sequences...`)
320
+ const accessions = withProtein.map(g => proteinByGene.get(g.geneId)!)
321
+ const seqByAcc = parseFasta(
322
+ await textfetch(
323
+ ncbiUrl(
324
+ `${EUTILS}/efetch.fcgi?db=protein&id=${accessions.join(',')}&rettype=fasta&retmode=text`,
325
+ ),
326
+ ),
327
+ )
328
+
329
+ const labels = dedupeLabels(
330
+ withProtein.map(g => g.commonName ?? g.scientificName),
331
+ )
332
+ const rows = withProtein
333
+ .map((g, i) => {
334
+ const protein = proteinByGene.get(g.geneId)!
335
+ return {
336
+ taxId: g.taxId,
337
+ label: labels[i]!,
338
+ scientificName: g.scientificName,
339
+ commonName: g.commonName,
340
+ geneId: g.geneId,
341
+ protein,
342
+ sequence: seqByAcc.get(protein) ?? '',
343
+ }
344
+ })
345
+ .filter(r => r.sequence)
346
+ if (rows.length < 2) {
347
+ throw new Error('Could not fetch protein sequences for the orthologs')
348
+ }
349
+ return rows
350
+ }
package/src/version.ts CHANGED
@@ -1 +1 @@
1
- export const version = '2.7.3'
1
+ export const version = '2.8.0'