@sjcrh/proteinpaint-server 2.216.1 → 2.217.1-0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,7 @@ import fs from 'fs'
7
7
  import os from 'os'
8
8
  import path from 'path'
9
9
  import { fileURLToPath } from 'url'
10
+ import crypto from 'crypto'
10
11
 
11
12
  // import.meta.dirname is undefined when using docker dev environment
12
13
  // use __dirname and __filename global variable convention from commonjs
@@ -364,6 +365,128 @@ if (process.env.PP_MODE?.startsWith('container')) {
364
365
  })
365
366
  }
366
367
 
368
+ /*
369
+ The key for deriving the cache file names and the cachedir subdir name, kept module-local, not in serverconfig,
370
+ so that it is never part of a config dump or a response.
371
+
372
+ Set PP_CACHEID_CREDS to the same value on every instance that shares a cachedir, so that they derive the same names
373
+ and keep finding the existing cache files across restarts. When the server is started with container/envHelpers.mjs,
374
+ as in the container images, PP_CACHEID_CREDS_FILE may name a file with the value instead: envHelpers.mjs reads that
375
+ file and passes its content like the other <NAME>_CREDS values, so that it is not in the initial env of the server
376
+ process. This module does not read PP_CACHEID_CREDS_FILE, so without envHelpers.mjs it is ignored. When not set, a
377
+ random key is generated per process: the cache still works, but every restart or other instance misses on the
378
+ files that were written before.
379
+
380
+ Read here, after the handoff values are set in process.env above, and removed from process.env in every mode, so
381
+ that it is not inherited by a child process. Only native modules are imported by this file, since it is also
382
+ loaded unbundled from the published package, such as by genome/copyDataFilesFromRepo2Tp.js, besides its bundled
383
+ copy in app.js. Only the first loaded copy, which is the bundled one in a server process, reads the key and sets
384
+ up the cache subdir, see firstLoad below.
385
+ */
386
+ const cacheIdKeyValue = process.env.PP_CACHEID_CREDS
387
+ const hasPersistentCacheKey = !!cacheIdKeyValue
388
+ const cacheIdKey = cacheIdKeyValue ? Buffer.from(cacheIdKeyValue, 'utf8') : crypto.randomBytes(32)
389
+ delete process.env.PP_CACHEID_CREDS
390
+
391
+ /** Derive a 32-hex-char cacheId from the given object via
392
+ * HMAC-sha256(key, JSON.stringify([scope, args])). Truncation at 32 chars is safe
393
+ * for cache keys — collision probability is negligible at realistic cache sizes.
394
+ * Callers shape `args` to include only the fields whose identity
395
+ * determines the cache key, and must construct it with a stable key order
396
+ * (object literals do this naturally).
397
+ *
398
+ * `scope` is optional, and separates cacheIds for the same args, e.g. per user or session
399
+ * when the result depends on what the requester may access. Without a scope, identical
400
+ * args share one cacheId, which is the intended behavior for results that are the same for
401
+ * every requester. */
402
+ export function generateHash(args, scope = '') {
403
+ return crypto
404
+ .createHmac('sha256', cacheIdKey)
405
+ .update(JSON.stringify([scope, args]))
406
+ .digest('hex')
407
+ .slice(0, 32)
408
+ }
409
+
410
+ /*
411
+ true for the first loaded copy of this module in a process. A later copy, such as the unbundled file that a genome
412
+ file imports, finds no PP_CACHEID_CREDS since the first copy removed it, so it would derive a different cache
413
+ subdir. The flag only marks that the cache subdir is set up, and does not hold its name.
414
+ */
415
+ const firstLoadFlag = Symbol.for('proteinpaint.serverconfig.firstLoad')
416
+ const firstLoad = !globalThis[firstLoadFlag]
417
+ // configurable, so that a test can evaluate another first copy
418
+ if (firstLoad) Object.defineProperty(globalThis, firstLoadFlag, { value: true, configurable: true })
419
+
420
+ /*
421
+ Use an unlisted subdir of the configured cachedir, whose name is derived from the PP_CACHEID_CREDS key, so that the
422
+ cache files cannot be found by listing directories. On by default in a container, and may be set with
423
+ serverconfig.hideCachedir in any mode. The configured cachedir must then be owned by another user than the server
424
+ process, such as root, with mode 1733: the server user may create and use a subdir with a known name, but may not
425
+ list the entries or change the mode. A dir that is owned by the server user can always be listed by that user,
426
+ since the owner may change its mode.
427
+
428
+ All code that uses the path must copy serverconfig.cachedir to a module-local variable when it is loaded:
429
+ app.ts deletes serverconfig.cachedir before the server starts listening.
430
+ */
431
+ if (serverconfig.hideCachedir ?? process.env.PP_MODE?.startsWith('container')) {
432
+ // a later copy does not use the cache, and must not use the configured cachedir, which is the parent of the subdir
433
+ if (!firstLoad) delete serverconfig.cachedir
434
+ else {
435
+ const parent = serverconfig.cachedir
436
+ if (!parent) throw 'serverconfig.cachedir missing'
437
+ // a parent dir that is created here is owned by the server user, which the warning below reports
438
+ if (!fs.existsSync(parent)) fs.mkdirSync(parent, { recursive: true })
439
+ let parentEntries
440
+ try {
441
+ parentEntries = fs.readdirSync(parent)
442
+ console.warn(
443
+ `WARNING: serverconfig.cachedir='${parent}' can be listed by the server process, so the cache subdir name ` +
444
+ `is not hidden; the dir should be owned by another user, such as root, with mode 1733`
445
+ )
446
+ } catch (e) {
447
+ // EACCES is the expected result; another error is reported by the mkdirSync() below
448
+ }
449
+ if (!hasPersistentCacheKey) {
450
+ console.warn(
451
+ `WARNING: PP_CACHEID_CREDS is not set, so the cache subdir name is generated for this process only, ` +
452
+ `and the files cached by an earlier process are not found or removed`
453
+ )
454
+ }
455
+ // not a generateHash() value, since the HMAC input is not a JSON array
456
+ const dirName = crypto.createHmac('sha256', cacheIdKey).update('cachedir').digest('hex').slice(0, 32)
457
+ serverconfig.cachedir = path.join(parent, dirName)
458
+ // the name is not logged, so that it is not in any log output
459
+ fs.mkdirSync(serverconfig.cachedir, { recursive: true, mode: 0o700 })
460
+ if (parentEntries) moveEarlierCacheEntries(parent, parentEntries, dirName)
461
+ }
462
+ }
463
+ delete serverconfig.hideCachedir
464
+
465
+ /*
466
+ Moves the entries of the earlier cache layout, such as the saved sessions in massSession/, from the configured
467
+ cachedir into the derived subdir, where they are still found by the server and evicted by the cache monitor.
468
+ This is only possible while the configured cachedir can be listed, such as on the first start with the derived
469
+ subdir, before the dir owner and mode are changed. An entry with the name shape of a derived subdir, such as from
470
+ another instance with a different key, is not moved, and neither is an entry whose name already exists in the
471
+ derived subdir.
472
+ */
473
+ function moveEarlierCacheEntries(parent, entries, dirName) {
474
+ let moved = 0
475
+ for (const name of entries) {
476
+ if (name == dirName || /^[0-9a-f]{32}$/.test(name)) continue
477
+ const dest = path.join(parent, dirName, name)
478
+ if (fs.existsSync(dest)) continue
479
+ try {
480
+ fs.renameSync(path.join(parent, name), dest)
481
+ moved++
482
+ } catch (e) {
483
+ // such as EXDEV for a separately mounted entry, which then stays where it is
484
+ console.warn(`WARNING: unable to move the earlier cache entry '${name}' into the cache subdir: ${e.code || e}`)
485
+ }
486
+ }
487
+ if (moved) console.log(`moved ${moved} earlier cache entries into the cache subdir`)
488
+ }
489
+
367
490
  // when a mandatory setting is not defined in any ds, declare its default here
368
491
 
369
492
  if (process.argv.find(a => a == 'validate')) {
@@ -417,7 +540,8 @@ if (fs.existsSync('./package.json')) {
417
540
  serverconfig.version = JSON.parse(pkg).version
418
541
  }
419
542
 
420
- if (!serverconfig.cache_snpgt) {
543
+ // a later copy may have no cachedir, see firstLoad above
544
+ if (!serverconfig.cache_snpgt && serverconfig.cachedir) {
421
545
  serverconfig.cache_snpgt = {
422
546
  dir: path.join(serverconfig.cachedir, 'snpgt'),
423
547
  fileNameRegexp: /[^\w]/, // client-provided cache file name matching with this are denied
@@ -1,178 +0,0 @@
1
- import path from "path";
2
- import fs from "fs";
3
- import { run_R } from "@sjcrh/proteinpaint-r";
4
- import serverconfig from "#src/serverconfig.js";
5
- import { clusterMethodLst, distanceMethodLst } from "#shared/clustering.js";
6
- function init({ genomes }) {
7
- return async (req, res) => {
8
- try {
9
- const q = (req.method === "POST" ? req.body : req.query) || {};
10
- if (q.for === "cluster") {
11
- await handleCluster(q, res);
12
- } else {
13
- await handleData(q, res, genomes);
14
- }
15
- } catch (e) {
16
- if (e instanceof Error && e.stack) console.log(e);
17
- res.send({ error: e?.message || String(e) });
18
- }
19
- };
20
- }
21
- const fileCache = /* @__PURE__ */ new Map();
22
- function stripQuotes(s) {
23
- if (s.length >= 2 && s[0] === '"' && s[s.length - 1] === '"') return s.slice(1, -1);
24
- return s;
25
- }
26
- async function parseTsv(absPath) {
27
- const text = await fs.promises.readFile(absPath, "utf8");
28
- const lines = text.split(/\r?\n/).filter((l) => l.length > 0);
29
- if (lines.length === 0) return { columns: [], rows: [], counts: [], geneIndex: /* @__PURE__ */ new Map() };
30
- const columns = lines[0].split(" ").map(stripQuotes);
31
- const rows = [];
32
- for (let i = 1; i < lines.length; i++) {
33
- const fields = lines[i].split(" ").map(stripQuotes);
34
- const normalizedFields = fields.length < columns.length ? fields.concat(Array(columns.length - fields.length).fill("")) : fields.slice(0, columns.length);
35
- const row = normalizedFields.map((v) => {
36
- if (v === "" || v === "NA" || v === "NaN") return null;
37
- const n = Number(v);
38
- return Number.isFinite(n) && v.trim() !== "" ? n : v;
39
- });
40
- rows.push(row);
41
- }
42
- const counts = columns.map(
43
- (_, c) => c === 0 ? rows.length : rows.filter((r) => r[c] !== null && r[c] !== void 0).length
44
- );
45
- const geneIndex = /* @__PURE__ */ new Map();
46
- for (const [i, r] of rows.entries()) {
47
- if (typeof r[0] === "string") {
48
- const g = r[0].toLowerCase();
49
- if (!geneIndex.has(g)) geneIndex.set(g, i);
50
- }
51
- }
52
- return { columns, rows, counts, geneIndex };
53
- }
54
- function zscorePerColumnIgnoringNull(matrix, ncol) {
55
- const out = matrix.map((r) => [...r]);
56
- for (let c = 0; c < ncol; c++) {
57
- const vals = [];
58
- for (const r of matrix) {
59
- const v = r[c];
60
- if (v !== null && v !== void 0 && Number.isFinite(v)) vals.push(v);
61
- }
62
- if (vals.length === 0) continue;
63
- const mean = vals.reduce((s, v) => s + v, 0) / vals.length;
64
- const sd = Math.sqrt(vals.reduce((s, v) => s + (v - mean) ** 2, 0) / vals.length);
65
- for (let r = 0; r < matrix.length; r++) {
66
- const v = matrix[r][c];
67
- if (v === null || v === void 0 || !Number.isFinite(v)) out[r][c] = null;
68
- else out[r][c] = sd === 0 ? 0 : (v - mean) / sd;
69
- }
70
- }
71
- return out;
72
- }
73
- async function handleData(q, res, genomes) {
74
- const genome = genomes[q.genome];
75
- if (!genome) throw "invalid genome";
76
- const ds = genome.datasets[q.dslabel];
77
- if (!ds) throw "invalid dslabel";
78
- const cfg = ds.queries?.geneRanking;
79
- if (!cfg || !cfg.rankings) throw "geneRanking not configured for this dataset";
80
- if (q.gene) {
81
- const wanted = String(q.gene).trim().toLowerCase();
82
- const geneRanks = {};
83
- const keys = Object.keys(cfg.rankings);
84
- const loaded = await Promise.all(
85
- keys.map(
86
- (key) => loadRanking(q, cfg.rankings[key], key).catch((e) => {
87
- console.log(`geneRanking: cannot load ranking "${key}": ${e?.message || e}`);
88
- return null;
89
- })
90
- )
91
- );
92
- for (const [i, parsed2] of loaded.entries()) {
93
- if (!parsed2) continue;
94
- const ri = parsed2.geneIndex.get(wanted);
95
- geneRanks[keys[i]] = {
96
- columns: parsed2.columns,
97
- row: ri === void 0 ? null : parsed2.rows[ri],
98
- counts: parsed2.counts,
99
- total: parsed2.rows.length
100
- };
101
- }
102
- res.send({ geneRanks });
103
- return;
104
- }
105
- if (!q.key) {
106
- res.send({ keys: Object.keys(cfg.rankings) });
107
- return;
108
- }
109
- const relPath = cfg.rankings[q.key];
110
- if (!relPath) throw "invalid key";
111
- const parsed = await loadRanking(q, relPath, q.key);
112
- res.send({ columns: parsed.columns, rows: parsed.rows });
113
- }
114
- async function loadRanking(q, relPath, key) {
115
- if (path.isAbsolute(relPath) || relPath.split(/[\\/]/).includes("..")) throw "invalid file path";
116
- const absPath = path.resolve(serverconfig.tpmasterdir, relPath);
117
- const tpRoot = path.resolve(serverconfig.tpmasterdir) + path.sep;
118
- if (!absPath.startsWith(tpRoot)) throw "invalid file path";
119
- const stat = await fs.promises.stat(absPath);
120
- const cacheKey = `${q.genome}|${q.dslabel}|${key}`;
121
- let entry = fileCache.get(cacheKey);
122
- if (!entry || entry.mtimeMs !== stat.mtimeMs) {
123
- entry = { parsed: await parseTsv(absPath), mtimeMs: stat.mtimeMs };
124
- fileCache.set(cacheKey, entry);
125
- }
126
- return entry.parsed;
127
- }
128
- async function handleCluster(q, res) {
129
- const { row_names, col_names } = q;
130
- if (!Array.isArray(q.matrix) || !Array.isArray(row_names) || !Array.isArray(col_names)) {
131
- throw "matrix, row_names, and col_names are required";
132
- }
133
- if (q.matrix.length !== row_names.length) throw "matrix.length must equal row_names.length";
134
- if (col_names.length < 2) throw "need at least 2 modalities to cluster";
135
- const minAssays = Math.max(2, q.minAssays ?? 3);
136
- const keptRows = [];
137
- const keptNames = [];
138
- for (let i = 0; i < q.matrix.length; i++) {
139
- const row = q.matrix[i];
140
- if (!Array.isArray(row) || row.length !== col_names.length) continue;
141
- const nonNull = row.filter((v) => v !== null && v !== void 0 && Number.isFinite(v)).length;
142
- if (nonNull < minAssays) continue;
143
- keptRows.push(row.map((v) => v === null || v === void 0 || !Number.isFinite(v) ? null : v));
144
- keptNames.push(row_names[i]);
145
- }
146
- if (keptRows.length < 3) throw `need at least 3 rows with \u2265${minAssays} non-null values for clustering`;
147
- const zMatrix = zscorePerColumnIgnoringNull(keptRows, col_names.length);
148
- const matrixForR = zMatrix.map((row) => row.map((v) => v === null ? 0 : v));
149
- const clusterMethod = q.clusterMethod || "average";
150
- const distanceMethod = q.distanceMethod || "euclidean";
151
- if (!clusterMethodLst.find((i) => i.value == clusterMethod)) throw "Invalid cluster method";
152
- if (!distanceMethodLst.find((i) => i.value == distanceMethod)) throw "Invalid distance method";
153
- const inputData = {
154
- matrix: matrixForR,
155
- row_names: keptNames,
156
- col_names,
157
- cluster_method: clusterMethod,
158
- distance_method: distanceMethod,
159
- plot_image: false
160
- };
161
- const Routput = JSON.parse(await run_R("hclust.R", JSON.stringify(inputData)));
162
- const rowOrderIdx = Routput.RowOrder.map((row) => keptNames.indexOf(row.name));
163
- const orderedMatrix = rowOrderIdx.map((i) => zMatrix[i]);
164
- res.send({
165
- row: {
166
- merge: Routput.RowMerge,
167
- height: Routput.RowHeight,
168
- order: Routput.RowOrder,
169
- inputOrder: keptNames
170
- },
171
- usedRowNames: keptNames,
172
- usedColNames: col_names,
173
- matrix: orderedMatrix
174
- });
175
- }
176
- export {
177
- init
178
- };
@@ -1,359 +0,0 @@
1
- import path from "path";
2
- import { get_ds_tdb } from "#src/termdb.js";
3
- import * as utils from "#src/utils.js";
4
- import { mayLimitSamples } from "#src/mds3.filter.js";
5
- import serverconfig from "#src/serverconfig.js";
6
- import { sql } from "#src/sql.ts";
7
- import { readGeneRows, baseUniProtAcc } from "../src/routes/termdb.bubbleHeatmap.ts";
8
- const missingDapWarned = /* @__PURE__ */ new Set();
9
- function init({ genomes }) {
10
- return async (req, res) => {
11
- const q = req.query;
12
- try {
13
- const genome = genomes[q.genome];
14
- if (!genome) throw "invalid genome";
15
- const [ds] = get_ds_tdb(genome, q);
16
- if (!ds.queries?.proteome?.organisms) throw "queries.proteome not configured";
17
- const term = q.term?.term || q.term;
18
- if (!term?.name) throw "term.name missing";
19
- const cohorts = [];
20
- const brConfig = ds.queries.proteome.brainRegions;
21
- const regionRemap = brConfig?.regionValueRemap || {};
22
- const sampleRegions = {};
23
- const identifierAnno = /* @__PURE__ */ new Map();
24
- for (const organismName in ds.queries.proteome.organisms) {
25
- const organism = ds.queries.proteome.organisms[organismName];
26
- for (const assayName in organism.assays) {
27
- const assay = organism.assays[assayName];
28
- for (const cohortName in assay.cohorts || {}) {
29
- const cohortCfg = assay.cohorts[cohortName];
30
- const dapPath = path.join(serverconfig.tpmasterdir, cohortCfg.DAPfile);
31
- const rows = await readGeneRows(dapPath, String(term.name).toLowerCase());
32
- if (!rows) {
33
- if (!missingDapWarned.has(dapPath)) {
34
- missingDapWarned.add(dapPath);
35
- console.warn(
36
- `proteome: DAPfile missing or unreadable for ${organismName}/${assayName}/${cohortName}: ${dapPath}`
37
- );
38
- }
39
- continue;
40
- }
41
- if (!rows.length) continue;
42
- const organismFilter = [{ columnIdx: organism.columnIdx, columnValue: organism.columnValue }];
43
- const assayFilter = [{ columnIdx: assay.columnIdx, columnValue: assay.columnValue }];
44
- let caseSamples = [];
45
- let controlSamples = [];
46
- try {
47
- caseSamples = listCohortSamples(ds.queries.proteome.db, [
48
- ...organismFilter,
49
- ...assayFilter,
50
- ...cohortCfg.caseFilter
51
- ]);
52
- controlSamples = listCohortSamples(ds.queries.proteome.db, [
53
- ...organismFilter,
54
- ...assayFilter,
55
- ...cohortCfg.controlFilter
56
- ]);
57
- } catch {
58
- }
59
- const annoKey = `${organismName}|${assayName}`;
60
- if (!identifierAnno.has(annoKey)) {
61
- let m = /* @__PURE__ */ new Map();
62
- try {
63
- m = listIdentifierAnnotations(ds.queries.proteome.db, term.name, [...organismFilter, ...assayFilter]);
64
- } catch {
65
- }
66
- identifierAnno.set(annoKey, m);
67
- }
68
- const anno = identifierAnno.get(annoKey);
69
- const sampleIds = brConfig ? [...caseSamples, ...controlSamples] : [];
70
- if (brConfig) {
71
- const regionOf = (filters) => {
72
- const f = filters.find((f2) => f2.columnIdx === brConfig.regionColumnIdx);
73
- if (!f) return void 0;
74
- const code = regionRemap[String(f.columnValue)] ?? String(f.columnValue);
75
- return brConfig.regions[code] !== void 0 ? code : void 0;
76
- };
77
- const caseRegion = regionOf(cohortCfg.caseFilter);
78
- const controlRegion = regionOf(cohortCfg.controlFilter);
79
- if (caseRegion) for (const sid of caseSamples) sampleRegions[sid] = caseRegion;
80
- if (controlRegion) for (const sid of controlSamples) sampleRegions[sid] = controlRegion;
81
- }
82
- for (const row of rows) {
83
- const entry = {
84
- organism: organismName,
85
- // the study catalog (dataset config) is the authority on disease
86
- disease: cohortCfg.catalog?.disease,
87
- assayName,
88
- cohortName,
89
- uniqueIdentifier: row.identifier,
90
- proteinAccession: row.acc,
91
- geneName: term.name,
92
- // client computes log2(foldChange); DAP stores log2FC directly
93
- foldChange: Math.pow(2, row.fc),
94
- // significance is the DAP file's FDR (BH-adjusted p), consistent with
95
- // the other DAP-driven tools; a nominal p is only present when the
96
- // DAP file carries one (6th column)
97
- fdr: row.fdr,
98
- pValue: row.fdr,
99
- testedN: caseSamples.length,
100
- controlN: controlSamples.length
101
- };
102
- const a = anno.get(row.identifier);
103
- if (a?.isoform) entry.isoform = a.isoform;
104
- if (assay.PTMType) {
105
- entry.PTMType = assay.PTMType;
106
- if (a?.modsite) entry.modSites = a.modsite;
107
- }
108
- if (assay.mclassOverride) entry.mclassOverride = assay.mclassOverride;
109
- if (organism.genomeName) entry.genomeName = organism.genomeName;
110
- if (brConfig) entry.sampleIds = sampleIds.slice();
111
- cohorts.push(entry);
112
- }
113
- }
114
- }
115
- }
116
- const refAssay = ds.queries.proteome.proteinReferenceAssay;
117
- if (refAssay) {
118
- const refFcByKey = /* @__PURE__ */ new Map();
119
- for (const e of cohorts) {
120
- if (e.assayName !== refAssay || !Number.isFinite(e.foldChange)) continue;
121
- const baseAcc = baseUniProtAcc(e.proteinAccession);
122
- if (!baseAcc) continue;
123
- const key = `${e.organism}|${e.cohortName}|${baseAcc}`;
124
- const p = Number.isFinite(e.fdr) ? e.fdr : Infinity;
125
- const cur = refFcByKey.get(key);
126
- if (!cur || p < cur.p) refFcByKey.set(key, { fc: e.foldChange, p });
127
- }
128
- for (const e of cohorts) {
129
- if (!e.PTMType) continue;
130
- const baseAcc = baseUniProtAcc(e.proteinAccession);
131
- if (!baseAcc) continue;
132
- const ref = refFcByKey.get(`${e.organism}|${e.cohortName}|${baseAcc}`);
133
- if (ref) e.proteinFoldChange = ref.fc;
134
- }
135
- }
136
- res.send({ protein: term.name, cohorts, sampleRegions: brConfig ? sampleRegions : void 0 });
137
- } catch (e) {
138
- if (e?.stack) console.log(e.stack);
139
- res.send({ error: e.message || e });
140
- }
141
- };
142
- }
143
- async function validate_query_proteome(ds) {
144
- const q = ds.queries.proteome;
145
- if (!q) return;
146
- if (!q.organisms) {
147
- throw "queries.proteome.organisms is missing";
148
- }
149
- if (!q.dbfile) {
150
- throw "queries.proteome.dbfile is missing";
151
- }
152
- try {
153
- q.db = utils.connect_db(q.dbfile);
154
- } catch (e) {
155
- throw `Cannot connect to proteome db ${q.dbfile}: ${e.message || e}`;
156
- }
157
- for (const organismName in q.organisms) {
158
- const organism = q.organisms[organismName];
159
- if (organism.columnIdx == null) throw `queries.proteome.organisms.${organismName}.columnIdx missing`;
160
- if (organism.columnValue == null) throw `queries.proteome.organisms.${organismName}.columnValue missing`;
161
- if (!organism.assays || typeof organism.assays != "object")
162
- throw `queries.proteome.organisms.${organismName}.assays missing or invalid`;
163
- for (const assayName in organism.assays) {
164
- const assay = organism.assays[assayName];
165
- if (assay.columnIdx == null)
166
- throw `queries.proteome.organisms.${organismName}.assays.${assayName}.columnIdx missing`;
167
- if (assay.columnValue == null)
168
- throw `queries.proteome.organisms.${organismName}.assays.${assayName}.columnValue missing`;
169
- if (assay.cohorts) {
170
- for (const cohortName in assay.cohorts) {
171
- const cohort = assay.cohorts[cohortName];
172
- if (!cohort.controlFilter)
173
- throw `Missing controlFilter in queries.proteome.organisms.${organismName}.assays.${assayName}.cohorts.${cohortName}`;
174
- if (!cohort.caseFilter)
175
- throw `Missing caseFilter in queries.proteome.organisms.${organismName}.assays.${assayName}.cohorts.${cohortName}`;
176
- if (!cohort.DAPfile)
177
- throw `Missing DAPfile in queries.proteome.organisms.${organismName}.assays.${assayName}.cohorts.${cohortName}`;
178
- }
179
- } else {
180
- throw `Invalid assay structure for "${assayName}". Must have .cohorts`;
181
- }
182
- }
183
- }
184
- const geneIndexHint = q.db.prepare("SELECT 1 FROM sqlite_master WHERE type = ? AND name = ?").get("index", "proteome_abundance_gene") ? sql` INDEXED BY proteome_abundance_gene` : sql``;
185
- q.find = async (arg) => {
186
- const proteins = arg?.proteins;
187
- if (!Array.isArray(proteins) || proteins.length == 0) throw "queries.proteome.find arg.proteins[] missing";
188
- const matches = /* @__PURE__ */ new Set();
189
- const details = arg?.dataTypeDetails || {};
190
- const organism = details.organism;
191
- const assay = details.assay;
192
- const cohort = details.cohort;
193
- const MAX_FIND_RESULTS = 500;
194
- const filters = [];
195
- if (Object.keys(details).length) {
196
- if (!organism || !assay || !cohort)
197
- throw "queries.proteome.find arg.dataTypeDetails.{organism,assay,cohort} missing";
198
- const organismConfig = q.organisms?.[organism];
199
- if (!organismConfig) throw `queries.proteome.find invalid organism: ${organism}`;
200
- const assayConfig = organismConfig.assays?.[assay];
201
- if (!assayConfig) throw `queries.proteome.find invalid assay: ${assay}`;
202
- const cohortConfig = assayConfig?.cohorts?.[cohort];
203
- if (!cohortConfig) throw `queries.proteome.find invalid cohort: ${cohort}`;
204
- const organismFilter = [{ columnIdx: organismConfig.columnIdx, columnValue: organismConfig.columnValue }];
205
- const assayFilter = [{ columnIdx: assayConfig.columnIdx, columnValue: assayConfig.columnValue }];
206
- const cohortFilter = (Array.isArray(cohortConfig.caseFilter) ? cohortConfig.caseFilter : []).filter(
207
- (filter) => !!filter
208
- );
209
- if (!cohortFilter.length) throw `queries.proteome.find invalid cohort caseFilter: ${cohort}`;
210
- filters.push(...organismFilter, ...assayFilter, ...cohortFilter);
211
- }
212
- for (const p of proteins) {
213
- if (!p) continue;
214
- const token = String(p).trim();
215
- if (token.length < 2) continue;
216
- const upperToken = `${token}\uFFFF`;
217
- const rawRows = [];
218
- if (filters?.length) {
219
- const query = sql`SELECT DISTINCT gene, identifier FROM proteome_abundance${geneIndexHint} WHERE gene >= ${token} COLLATE NOCASE AND gene < ${upperToken} COLLATE NOCASE AND ${buildFilterClause(
220
- filters
221
- )} LIMIT ${MAX_FIND_RESULTS}`;
222
- rawRows.push(...q.db.prepare(query).all());
223
- } else {
224
- rawRows.push(
225
- ...q.db.prepare(
226
- sql`SELECT DISTINCT gene, identifier FROM proteome_abundance WHERE gene >= ${token} COLLATE NOCASE AND gene < ${upperToken} COLLATE NOCASE LIMIT ${MAX_FIND_RESULTS}`
227
- ).all()
228
- );
229
- }
230
- for (const row of rawRows) {
231
- if (!row?.gene || !row?.identifier) continue;
232
- matches.add(`${row.gene}: ${row.identifier}`);
233
- }
234
- }
235
- return [...matches];
236
- };
237
- q.get = async (param) => {
238
- if (!param?.terms?.length) throw "queries.proteome.get param.terms[] missing";
239
- if (!param.dataTypeDetails?.assay || !param.dataTypeDetails?.cohort || !param.dataTypeDetails?.organism)
240
- throw "queries.proteome.get param.dataTypeDetails.{assay,cohort,organism} missing";
241
- return await getProteomeValuesFromCohort(ds, param, q);
242
- };
243
- }
244
- const columnIdxToName = {
245
- 0: "organism",
246
- 1: "disease",
247
- 2: "tissue",
248
- 3: "brain_region",
249
- 4: "tech1",
250
- 5: "tech2",
251
- 6: "cohort"
252
- };
253
- function resolveColumnName(idx) {
254
- const name = columnIdxToName[idx];
255
- if (!name) throw `Invalid columnIdx: ${idx}, must be one of ${Object.keys(columnIdxToName).join(",")}`;
256
- return name;
257
- }
258
- function buildFilterClause(filters) {
259
- return sql.join(
260
- filters.map((f) => sql`${sql.id(resolveColumnName(f.columnIdx))} = ${f.columnValue}`),
261
- sql` AND `
262
- );
263
- }
264
- function listCohortSamples(db, filters) {
265
- if (!filters?.length) throw "listCohortSamples: filters must not be empty";
266
- let perDb = cohortSampleCache.get(db);
267
- if (!perDb) cohortSampleCache.set(db, perDb = /* @__PURE__ */ new Map());
268
- const key = JSON.stringify(filters);
269
- const hit = perDb.get(key);
270
- if (hit) return hit;
271
- const samples = db.prepare(sql`SELECT DISTINCT sample FROM proteome_abundance WHERE ${buildFilterClause(filters)}`).all().map((r) => String(r.sample));
272
- perDb.set(key, samples);
273
- return samples;
274
- }
275
- const cohortSampleCache = /* @__PURE__ */ new WeakMap();
276
- function listIdentifierAnnotations(db, gene, filters) {
277
- const rows = db.prepare(
278
- sql`SELECT identifier, modsite, isoform FROM proteome_abundance WHERE gene = ${gene} COLLATE NOCASE${filters.length ? sql` AND ${buildFilterClause(filters)}` : sql``} GROUP BY identifier`
279
- ).all();
280
- return new Map(rows.map((r) => [r.identifier, { modsite: r.modsite, isoform: r.isoform }]));
281
- }
282
- function queryDbRows(db, identifier, filters) {
283
- const query = sql`SELECT organism, disease, identifier, protein_accession, isoform, modsite, gene, sample, value, brain_region
284
- FROM proteome_abundance
285
- WHERE identifier = ${identifier} COLLATE NOCASE${filters.length ? sql` AND ${buildFilterClause(filters)}` : sql``}`;
286
- return db.prepare(query).all();
287
- }
288
- async function getProteomeValuesFromCohort(ds, param, q) {
289
- const db = ds.queries.proteome.db;
290
- const { assay, cohort, organism } = param.dataTypeDetails;
291
- const organismConfig = q.organisms?.[organism];
292
- if (!organismConfig) throw `queries.proteome invalid organism: ${organism}`;
293
- const organismColumnIdx = organismConfig.columnIdx;
294
- const organismColumnValue = organismConfig.columnValue;
295
- const assayConfig = organismConfig.assays?.[assay];
296
- if (!assayConfig) throw `queries.proteome.get invalid assay: ${assay}`;
297
- const assayColumnIdx = assayConfig.columnIdx;
298
- const assayColumnValue = assayConfig.columnValue;
299
- const cohortConfig = assayConfig?.cohorts?.[cohort];
300
- if (!cohortConfig) throw `queries.proteome.get invalid cohort: ${cohort}`;
301
- const cohortControlFilter = cohortConfig.controlFilter;
302
- const cohortCaseFilter = cohortConfig.caseFilter;
303
- const organismFilter = [{ columnIdx: organismColumnIdx, columnValue: organismColumnValue }];
304
- const assayFilter = [{ columnIdx: assayColumnIdx, columnValue: assayColumnValue }];
305
- const term2sample2value = /* @__PURE__ */ new Map();
306
- const controlSampleIds = /* @__PURE__ */ new Set();
307
- for (const tw of param.terms) {
308
- if (!tw) continue;
309
- const fullGeneName = tw.term.name;
310
- const identifier = fullGeneName.split(":")[1]?.trim();
311
- const geneName = fullGeneName.split(":")[0]?.trim();
312
- if (!identifier || !geneName)
313
- throw "invalid term name for proteome query, must be in format geneName: uniqueIdentifier";
314
- const caseRows = queryDbRows(db, identifier, [...organismFilter, ...assayFilter, ...cohortCaseFilter]);
315
- const controlRows = queryDbRows(db, identifier, [...organismFilter, ...assayFilter, ...cohortControlFilter]);
316
- for (const row of controlRows) {
317
- const sid = ds.cohort.termdb.q.sampleName2id(row.sample);
318
- if (sid !== void 0) controlSampleIds.add(String(sid));
319
- }
320
- const allRows = [...caseRows, ...controlRows];
321
- const allSampleIds = [];
322
- for (const row of allRows) {
323
- const sid = ds.cohort.termdb.q.sampleName2id(row.sample);
324
- if (sid !== void 0) allSampleIds.push(sid);
325
- }
326
- const uniqueSampleIds = [...new Set(allSampleIds)];
327
- const allowedSampleIds = await mayLimitSamples(param, uniqueSampleIds, ds);
328
- if (allowedSampleIds?.size == 0) {
329
- return { term2sample2value: /* @__PURE__ */ new Map(), byTermId: {}, bySampleId: {} };
330
- }
331
- const s2v = {};
332
- for (const row of allRows) {
333
- const sid = ds.cohort.termdb.q.sampleName2id(row.sample);
334
- if (sid === void 0) continue;
335
- if (allowedSampleIds && !allowedSampleIds.has(sid)) continue;
336
- s2v[sid] = row.value;
337
- }
338
- if (Object.keys(s2v).length) {
339
- term2sample2value.set(tw.$id, s2v);
340
- }
341
- }
342
- const bySampleId = {};
343
- if (term2sample2value.size == 0) {
344
- throw `No data available for: ${param.terms?.map((t) => t.term.name).join(", ")}`;
345
- }
346
- for (const s2v of term2sample2value.values()) {
347
- for (const sid of Object.keys(s2v)) {
348
- bySampleId[sid] = { label: ds.cohort.termdb.q.id2sampleName(Number(sid)) };
349
- }
350
- }
351
- return { term2sample2value, controlSampleIds, bySampleId };
352
- }
353
- export {
354
- init,
355
- listCohortSamples,
356
- listIdentifierAnnotations,
357
- queryDbRows,
358
- validate_query_proteome
359
- };