dasmixer-cli 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dasmixer/cli/__init__.py +3 -0
- dasmixer/cli/commands/__init__.py +3 -0
- dasmixer/cli/commands/calculate.py +513 -0
- dasmixer/cli/commands/import_data.py +499 -0
- dasmixer/cli/commands/import_project.py +75 -0
- dasmixer/cli/commands/portable.py +54 -0
- dasmixer/cli/commands/project.py +94 -0
- dasmixer/cli/commands/subset.py +130 -0
- dasmixer/cli/commands/tool.py +147 -0
- dasmixer/cli/main.py +52 -0
- dasmixer_cli-0.6.0.dist-info/METADATA +51 -0
- dasmixer_cli-0.6.0.dist-info/RECORD +14 -0
- dasmixer_cli-0.6.0.dist-info/WHEEL +4 -0
- dasmixer_cli-0.6.0.dist-info/entry_points.txt +3 -0
dasmixer/cli/__init__.py
ADDED
|
@@ -0,0 +1,513 @@
|
|
|
1
|
+
"""CLI commands for running pipeline calculations."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import typer
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Annotated
|
|
7
|
+
import asyncio
|
|
8
|
+
import pandas as pd
|
|
9
|
+
from dasmixer.api.project.project import Project
|
|
10
|
+
from dasmixer.api.config import config as app_config
|
|
11
|
+
|
|
12
|
+
app = typer.Typer(help="Run pipeline calculations")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _get_setting(project, key: str, default: str) -> str:
|
|
16
|
+
"""Get a setting from project_settings with fallback."""
|
|
17
|
+
import asyncio
|
|
18
|
+
try:
|
|
19
|
+
rows = asyncio.run_coroutine_threadsafe(
|
|
20
|
+
project.execute_query(
|
|
21
|
+
"SELECT value FROM project_settings WHERE key=?",
|
|
22
|
+
[key],
|
|
23
|
+
),
|
|
24
|
+
None,
|
|
25
|
+
)
|
|
26
|
+
# Fallback: use get_setting directly
|
|
27
|
+
return default
|
|
28
|
+
except Exception:
|
|
29
|
+
return default
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _parse_bool(val: str) -> bool:
|
|
33
|
+
return val.lower() in ("true", "1", "yes")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
# ion-coverage
|
|
38
|
+
# ---------------------------------------------------------------------------
|
|
39
|
+
|
|
40
|
+
@app.command()
|
|
41
|
+
def ion_coverage(
|
|
42
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
43
|
+
recalc_all: Annotated[bool, typer.Option("--recalc-all", help="Recalculate all, including already processed")] = False,
|
|
44
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
45
|
+
tolerance: Annotated[float, typer.Option("--tolerance", help="PPM tolerance (overrides project setting)")] = None,
|
|
46
|
+
ions: Annotated[str, typer.Option("--ions", help="Ion types, comma-separated (overrides project setting)")] = None,
|
|
47
|
+
):
|
|
48
|
+
"""
|
|
49
|
+
Calculate ion coverage for identifications.
|
|
50
|
+
|
|
51
|
+
Reads settings from project_settings, can override with CLI options.
|
|
52
|
+
"""
|
|
53
|
+
async def _run():
|
|
54
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
55
|
+
# Load settings from project_settings
|
|
56
|
+
ion_match_ions = await project.get_setting("ion_match_ions", "b,y")
|
|
57
|
+
ion_match_tolerance = await project.get_setting("ion_match_tolerance", "20.0")
|
|
58
|
+
ion_match_mode = await project.get_setting("ion_match_mode", "largest")
|
|
59
|
+
ion_match_water_loss = await project.get_setting("ion_match_water_loss", "True")
|
|
60
|
+
ion_match_ammonia_loss = await project.get_setting("ion_match_ammonia_loss", "True")
|
|
61
|
+
ion_fragment_charges = await project.get_setting("ion_fragment_charges", "1,2")
|
|
62
|
+
seqfixer_min_charge = await project.get_setting("seqfixer_min_charge", "1")
|
|
63
|
+
seqfixer_max_charge = await project.get_setting("seqfixer_max_charge", "4")
|
|
64
|
+
seqfixer_max_isotope_offset = await project.get_setting("seqfixer_max_isotope_offset", "3")
|
|
65
|
+
seqfixer_max_ptm = await project.get_setting("seqfixer_max_ptm", "2")
|
|
66
|
+
seqfixer_max_ptm_sites = await project.get_setting("seqfixer_max_ptm_sites", "3")
|
|
67
|
+
|
|
68
|
+
# Override with CLI options
|
|
69
|
+
if tolerance is not None:
|
|
70
|
+
ion_match_tolerance = str(tolerance)
|
|
71
|
+
if ions is not None:
|
|
72
|
+
ion_match_ions = ions
|
|
73
|
+
|
|
74
|
+
from dasmixer.api.calculations.ppm.seqfixer import SeqfixerParams
|
|
75
|
+
from dasmixer.api.calculations.spectra.ion_match import IonMatchParameters, process_identificatons_batch
|
|
76
|
+
|
|
77
|
+
seqfixer_params = SeqfixerParams(
|
|
78
|
+
min_charge=int(seqfixer_min_charge),
|
|
79
|
+
max_charge=int(seqfixer_max_charge),
|
|
80
|
+
max_isotope_offset=int(seqfixer_max_isotope_offset),
|
|
81
|
+
max_ptm=int(seqfixer_max_ptm),
|
|
82
|
+
max_ptm_sites=int(seqfixer_max_ptm_sites),
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
ion_params = IonMatchParameters(
|
|
86
|
+
ion_types=ion_match_ions.split(","),
|
|
87
|
+
ppm_tolerance=float(ion_match_tolerance),
|
|
88
|
+
match_mode=ion_match_mode,
|
|
89
|
+
water_loss=_parse_bool(ion_match_water_loss),
|
|
90
|
+
ammonia_loss=_parse_bool(ion_match_ammonia_loss),
|
|
91
|
+
fragment_charges=[int(c) for c in ion_fragment_charges.split(",")],
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
# Get identifications without coverage (or all if recalc_all)
|
|
95
|
+
if recalc_all:
|
|
96
|
+
idents_df = await project.get_identifications(sample_id=sample_id)
|
|
97
|
+
else:
|
|
98
|
+
idents_df = await project.get_identifications(sample_id=sample_id)
|
|
99
|
+
if not idents_df.empty and "ppm" in idents_df.columns:
|
|
100
|
+
idents_df = idents_df[idents_df["ppm"].isna()]
|
|
101
|
+
|
|
102
|
+
if idents_df.empty:
|
|
103
|
+
typer.echo("No identifications to process.")
|
|
104
|
+
return
|
|
105
|
+
|
|
106
|
+
total = len(idents_df)
|
|
107
|
+
batch_size = getattr(app_config, 'identification_processing_batch_size', 500)
|
|
108
|
+
processed = 0
|
|
109
|
+
|
|
110
|
+
import concurrent.futures
|
|
111
|
+
with typer.progressbar(length=total, label="Processing") as progress:
|
|
112
|
+
for start in range(0, total, batch_size):
|
|
113
|
+
batch = idents_df.iloc[start:start + batch_size]
|
|
114
|
+
data_rows = process_identificatons_batch(
|
|
115
|
+
batch, ion_params, seqfixer_params,
|
|
116
|
+
)
|
|
117
|
+
if data_rows:
|
|
118
|
+
await project.put_identification_data_batch(data_rows)
|
|
119
|
+
processed += len(batch)
|
|
120
|
+
progress.update(len(batch))
|
|
121
|
+
|
|
122
|
+
await project.save()
|
|
123
|
+
typer.echo(f"✓ Processed {processed} identifications")
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
asyncio.run(_run())
|
|
127
|
+
except typer.Exit:
|
|
128
|
+
raise
|
|
129
|
+
except Exception as e:
|
|
130
|
+
typer.echo(f"Error: {e}", err=True)
|
|
131
|
+
raise typer.Exit(1)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
# preferred
|
|
136
|
+
# ---------------------------------------------------------------------------
|
|
137
|
+
|
|
138
|
+
@app.command()
|
|
139
|
+
def preferred(
|
|
140
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
141
|
+
criterion: Annotated[str, typer.Option("--criterion", help="Selection criterion: ppm or intensity")] = None,
|
|
142
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
143
|
+
):
|
|
144
|
+
"""
|
|
145
|
+
Select preferred identifications per spectrum.
|
|
146
|
+
"""
|
|
147
|
+
async def _run():
|
|
148
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
149
|
+
if criterion is None:
|
|
150
|
+
criterion = await project.get_setting("preferred_criterion", "intensity")
|
|
151
|
+
|
|
152
|
+
from dasmixer.api.calculations.peptides.matching import select_preferred_identifications
|
|
153
|
+
|
|
154
|
+
# Build tool_settings from project_settings
|
|
155
|
+
tools = await project.get_tools()
|
|
156
|
+
tool_settings = {}
|
|
157
|
+
for t in tools:
|
|
158
|
+
tid = t.id
|
|
159
|
+
ts = dict(t.settings or {})
|
|
160
|
+
ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
|
|
161
|
+
ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
|
|
162
|
+
ts.setdefault("coverage_min", await project.get_setting(f"tool_{tid}_coverage_min", "0"))
|
|
163
|
+
ts.setdefault("length_min", await project.get_setting(f"tool_{tid}_length_min", "5"))
|
|
164
|
+
tool_settings[tid] = ts
|
|
165
|
+
|
|
166
|
+
count = await select_preferred_identifications(
|
|
167
|
+
project, criterion, tool_settings, sample_id=sample_id,
|
|
168
|
+
)
|
|
169
|
+
await project.save()
|
|
170
|
+
typer.echo(f"✓ Selected {count} preferred identifications")
|
|
171
|
+
|
|
172
|
+
try:
|
|
173
|
+
asyncio.run(_run())
|
|
174
|
+
except typer.Exit:
|
|
175
|
+
raise
|
|
176
|
+
except Exception as e:
|
|
177
|
+
typer.echo(f"Error: {e}", err=True)
|
|
178
|
+
raise typer.Exit(1)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
# ---------------------------------------------------------------------------
|
|
182
|
+
# peptide-match
|
|
183
|
+
# ---------------------------------------------------------------------------
|
|
184
|
+
|
|
185
|
+
@app.command()
|
|
186
|
+
def peptide_match(
|
|
187
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
188
|
+
fasta: Annotated[str, typer.Option("--fasta", help="Path to FASTA file (overrides project setting)")] = None,
|
|
189
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
190
|
+
threshold: Annotated[float, typer.Option("--threshold", help="Identity threshold")] = None,
|
|
191
|
+
batch_size: Annotated[int, typer.Option("--batch-size", help="Batch size")] = None,
|
|
192
|
+
):
|
|
193
|
+
"""
|
|
194
|
+
Match peptide identifications to protein sequences.
|
|
195
|
+
"""
|
|
196
|
+
async def _run():
|
|
197
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
198
|
+
fasta_path = fasta
|
|
199
|
+
if not fasta_path:
|
|
200
|
+
fasta_path = await project.get_setting("fasta_path", "")
|
|
201
|
+
if not fasta_path:
|
|
202
|
+
typer.echo("Error: No FASTA file specified. Use --fasta or set fasta_path in project settings.", err=True)
|
|
203
|
+
raise typer.Exit(1)
|
|
204
|
+
|
|
205
|
+
fasta_path_obj = Path(fasta_path)
|
|
206
|
+
if not fasta_path_obj.exists():
|
|
207
|
+
typer.echo(f"Error: FASTA file not found: {fasta_path}", err=True)
|
|
208
|
+
raise typer.Exit(1)
|
|
209
|
+
|
|
210
|
+
ident_threshold = threshold if threshold is not None else float(
|
|
211
|
+
await project.get_setting("blast_identity_threshold", "0.8")
|
|
212
|
+
)
|
|
213
|
+
mapping_batch_size = batch_size or getattr(app_config, 'protein_mapping_batch_size', 1000)
|
|
214
|
+
|
|
215
|
+
from dasmixer.api.calculations.peptides.protein_map import map_proteins
|
|
216
|
+
|
|
217
|
+
tools = await project.get_tools()
|
|
218
|
+
tool_settings = {}
|
|
219
|
+
for t in tools:
|
|
220
|
+
tid = t.id
|
|
221
|
+
ts = dict(t.settings or {})
|
|
222
|
+
ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
|
|
223
|
+
ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
|
|
224
|
+
tool_settings[tid] = ts
|
|
225
|
+
|
|
226
|
+
await project.set_setting("fasta_path", str(fasta_path_obj))
|
|
227
|
+
total = await map_proteins(
|
|
228
|
+
project, tool_settings,
|
|
229
|
+
fasta_path=str(fasta_path_obj),
|
|
230
|
+
identity_threshold=ident_threshold,
|
|
231
|
+
mapping_batch_size=mapping_batch_size,
|
|
232
|
+
sample_id=sample_id,
|
|
233
|
+
)
|
|
234
|
+
await project.save()
|
|
235
|
+
typer.echo(f"✓ Mapped {total} peptide matches")
|
|
236
|
+
|
|
237
|
+
try:
|
|
238
|
+
asyncio.run(_run())
|
|
239
|
+
except typer.Exit:
|
|
240
|
+
raise
|
|
241
|
+
except Exception as e:
|
|
242
|
+
typer.echo(f"Error: {e}", err=True)
|
|
243
|
+
raise typer.Exit(1)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
# ---------------------------------------------------------------------------
|
|
247
|
+
# protein-idents
|
|
248
|
+
# ---------------------------------------------------------------------------
|
|
249
|
+
|
|
250
|
+
@app.command()
|
|
251
|
+
def protein_idents(
|
|
252
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
253
|
+
min_peptides: Annotated[int, typer.Option("--min-peptides", help="Minimum peptides per protein")] = None,
|
|
254
|
+
min_unique: Annotated[int, typer.Option("--min-unique", help="Minimum unique evidence")] = None,
|
|
255
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
256
|
+
):
|
|
257
|
+
"""
|
|
258
|
+
Calculate protein identifications from peptide matches.
|
|
259
|
+
"""
|
|
260
|
+
async def _run():
|
|
261
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
262
|
+
if min_peptides is None:
|
|
263
|
+
min_peptides = int(await project.get_setting("proteins_min_peptides", "2"))
|
|
264
|
+
if min_unique is None:
|
|
265
|
+
min_unique = int(await project.get_setting("proteins_min_unique_evidence", "1"))
|
|
266
|
+
|
|
267
|
+
await project.set_setting("proteins_min_peptides", str(min_peptides))
|
|
268
|
+
await project.set_setting("proteins_min_unique_evidence", str(min_unique))
|
|
269
|
+
|
|
270
|
+
from dasmixer.api.calculations.proteins.map_identifications import find_protein_identifications
|
|
271
|
+
|
|
272
|
+
# Get joined peptide data
|
|
273
|
+
filters = {}
|
|
274
|
+
if sample_id is not None:
|
|
275
|
+
filters['sample_id'] = sample_id
|
|
276
|
+
joined_data = await project.get_joined_peptide_data(**filters)
|
|
277
|
+
|
|
278
|
+
if joined_data.empty:
|
|
279
|
+
typer.echo("No peptide data found. Run 'calculate peptide-match' first.")
|
|
280
|
+
return
|
|
281
|
+
|
|
282
|
+
# Get all proteins as sequences_db
|
|
283
|
+
proteins_df = await project.get_proteins()
|
|
284
|
+
if proteins_df.empty:
|
|
285
|
+
typer.echo("No proteins in project. Import a FASTA file first.")
|
|
286
|
+
return
|
|
287
|
+
|
|
288
|
+
sequences_db = dict(zip(proteins_df['id'], proteins_df['sequence']))
|
|
289
|
+
|
|
290
|
+
prot_idents_df = find_protein_identifications(
|
|
291
|
+
joined_data, sequences_db, min_peptides, min_unique,
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
if prot_idents_df.empty:
|
|
295
|
+
typer.echo("No protein identifications found with current thresholds.")
|
|
296
|
+
return
|
|
297
|
+
|
|
298
|
+
await project.add_protein_identifications_batch(prot_idents_df)
|
|
299
|
+
await project.save()
|
|
300
|
+
typer.echo(f"✓ Found {len(prot_idents_df)} protein identifications")
|
|
301
|
+
|
|
302
|
+
try:
|
|
303
|
+
asyncio.run(_run())
|
|
304
|
+
except typer.Exit:
|
|
305
|
+
raise
|
|
306
|
+
except Exception as e:
|
|
307
|
+
typer.echo(f"Error: {e}", err=True)
|
|
308
|
+
raise typer.Exit(1)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
# ---------------------------------------------------------------------------
|
|
312
|
+
# lfq
|
|
313
|
+
# ---------------------------------------------------------------------------
|
|
314
|
+
|
|
315
|
+
@app.command()
|
|
316
|
+
def lfq(
|
|
317
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
318
|
+
empai: Annotated[bool, typer.Option("--empai", help="Enable emPAI")] = None,
|
|
319
|
+
ibaq: Annotated[bool, typer.Option("--ibaq", help="Enable iBAQ")] = None,
|
|
320
|
+
nsaf: Annotated[bool, typer.Option("--nsaf", help="Enable NSAF")] = None,
|
|
321
|
+
top3: Annotated[bool, typer.Option("--top3", help="Enable Top3")] = None,
|
|
322
|
+
enzyme: Annotated[str, typer.Option("--enzyme", help="Digestion enzyme")] = None,
|
|
323
|
+
min_length: Annotated[int, typer.Option("--min-length", help="Min peptide length")] = None,
|
|
324
|
+
max_length: Annotated[int, typer.Option("--max-length", help="Max peptide length")] = None,
|
|
325
|
+
max_cleavage: Annotated[int, typer.Option("--max-cleavage", help="Max missed cleavages")] = None,
|
|
326
|
+
empai_base: Annotated[float, typer.Option("--empai-base", help="emPAI base value")] = None,
|
|
327
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
328
|
+
):
|
|
329
|
+
"""
|
|
330
|
+
Calculate protein quantification (LFQ).
|
|
331
|
+
|
|
332
|
+
If no method flags (--empai, --ibaq, etc.) are specified,
|
|
333
|
+
methods are read from project_settings.
|
|
334
|
+
"""
|
|
335
|
+
async def _run():
|
|
336
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
337
|
+
# Determine methods
|
|
338
|
+
cli_methods = []
|
|
339
|
+
if empai is not None:
|
|
340
|
+
cli_methods.append("empai")
|
|
341
|
+
if ibaq is not None:
|
|
342
|
+
cli_methods.append("ibaq")
|
|
343
|
+
if nsaf is not None:
|
|
344
|
+
cli_methods.append("nsaf")
|
|
345
|
+
if top3 is not None:
|
|
346
|
+
cli_methods.append("top3")
|
|
347
|
+
|
|
348
|
+
if cli_methods:
|
|
349
|
+
methods = cli_methods
|
|
350
|
+
else:
|
|
351
|
+
methods_str = await project.get_setting("lfq_methods", "empai,ibaq")
|
|
352
|
+
methods = [m.strip() for m in methods_str.split(",") if m.strip()]
|
|
353
|
+
|
|
354
|
+
if not methods:
|
|
355
|
+
typer.echo("Error: No LFQ methods selected", err=True)
|
|
356
|
+
raise typer.Exit(1)
|
|
357
|
+
|
|
358
|
+
# Load / override parameters
|
|
359
|
+
lfq_enzyme = enzyme or await project.get_setting("lfq_enzyme", "trypsin")
|
|
360
|
+
lfq_min_length = min_length or int(await project.get_setting("lfq_min_peptide_length", "7"))
|
|
361
|
+
lfq_max_length = max_length or int(await project.get_setting("lfq_max_peptide_length", "25"))
|
|
362
|
+
lfq_max_cleavage = max_cleavage or int(await project.get_setting("lfq_max_cleavage_sites", "2"))
|
|
363
|
+
lfq_empai_base = empai_base or float(await project.get_setting("lfq_empai_base", "10.0"))
|
|
364
|
+
|
|
365
|
+
# Save settings
|
|
366
|
+
await project.set_setting("lfq_methods", ",".join(methods))
|
|
367
|
+
await project.set_setting("lfq_enzyme", lfq_enzyme)
|
|
368
|
+
await project.set_setting("lfq_min_peptide_length", str(lfq_min_length))
|
|
369
|
+
await project.set_setting("lfq_max_peptide_length", str(lfq_max_length))
|
|
370
|
+
await project.set_setting("lfq_max_cleavage_sites", str(lfq_max_cleavage))
|
|
371
|
+
await project.set_setting("lfq_empai_base", str(lfq_empai_base))
|
|
372
|
+
|
|
373
|
+
from dasmixer.api.calculations.proteins.lfq import calculate_lfq
|
|
374
|
+
|
|
375
|
+
samples = await project.get_samples()
|
|
376
|
+
if sample_id is not None:
|
|
377
|
+
samples = [s for s in samples if s.id == sample_id]
|
|
378
|
+
if not samples:
|
|
379
|
+
typer.echo(f"Error: Sample ID {sample_id} not found", err=True)
|
|
380
|
+
raise typer.Exit(1)
|
|
381
|
+
|
|
382
|
+
calculated_count = 0
|
|
383
|
+
for s in samples:
|
|
384
|
+
await calculate_lfq(
|
|
385
|
+
project,
|
|
386
|
+
sample_id=s.id,
|
|
387
|
+
methods=methods,
|
|
388
|
+
enzyme=lfq_enzyme,
|
|
389
|
+
min_peptide_length=lfq_min_length,
|
|
390
|
+
max_peptide_length=lfq_max_length,
|
|
391
|
+
max_cleavage_sites=lfq_max_cleavage,
|
|
392
|
+
empai_base=lfq_empai_base,
|
|
393
|
+
)
|
|
394
|
+
calculated_count += 1
|
|
395
|
+
typer.echo(f" Sample '{s.name}': done")
|
|
396
|
+
|
|
397
|
+
await project.save()
|
|
398
|
+
typer.echo(f"✓ LFQ calculated for {calculated_count} samples, methods: {methods}")
|
|
399
|
+
|
|
400
|
+
try:
|
|
401
|
+
asyncio.run(_run())
|
|
402
|
+
except typer.Exit:
|
|
403
|
+
raise
|
|
404
|
+
except Exception as e:
|
|
405
|
+
typer.echo(f"Error: {e}", err=True)
|
|
406
|
+
raise typer.Exit(1)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
# ---------------------------------------------------------------------------
|
|
410
|
+
# peptides (full pipeline)
|
|
411
|
+
# ---------------------------------------------------------------------------
|
|
412
|
+
|
|
413
|
+
@app.command()
|
|
414
|
+
def peptides(
|
|
415
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
416
|
+
fasta: Annotated[str, typer.Option("--fasta", help="Path to FASTA file")] = None,
|
|
417
|
+
criterion: Annotated[str, typer.Option("--criterion", help="Selection criterion: ppm or intensity")] = None,
|
|
418
|
+
sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
|
|
419
|
+
):
|
|
420
|
+
"""
|
|
421
|
+
Run full peptide calculation pipeline:
|
|
422
|
+
1. Match proteins
|
|
423
|
+
2. Calculate ion coverage
|
|
424
|
+
3. Select preferred identifications
|
|
425
|
+
"""
|
|
426
|
+
async def _run():
|
|
427
|
+
async with Project(path=Path(project_path), create_if_not_exists=False) as project:
|
|
428
|
+
# Step 1: Match proteins
|
|
429
|
+
typer.echo("Step 1/3: Matching proteins...")
|
|
430
|
+
fasta_path = fasta or await project.get_setting("fasta_path", "")
|
|
431
|
+
if not fasta_path:
|
|
432
|
+
typer.echo("Error: No FASTA file specified", err=True)
|
|
433
|
+
raise typer.Exit(1)
|
|
434
|
+
if not Path(fasta_path).exists():
|
|
435
|
+
typer.echo(f"Error: FASTA file not found: {fasta_path}", err=True)
|
|
436
|
+
raise typer.Exit(1)
|
|
437
|
+
|
|
438
|
+
tools = await project.get_tools()
|
|
439
|
+
tool_settings = {}
|
|
440
|
+
for t in tools:
|
|
441
|
+
tid = t.id
|
|
442
|
+
ts = dict(t.settings or {})
|
|
443
|
+
ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
|
|
444
|
+
ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
|
|
445
|
+
tool_settings[tid] = ts
|
|
446
|
+
|
|
447
|
+
ident_threshold = float(await project.get_setting("blast_identity_threshold", "0.8"))
|
|
448
|
+
mapping_batch_size = getattr(app_config, 'protein_mapping_batch_size', 1000)
|
|
449
|
+
|
|
450
|
+
from dasmixer.api.calculations.peptides.protein_map import map_proteins
|
|
451
|
+
await map_proteins(
|
|
452
|
+
project, tool_settings,
|
|
453
|
+
fasta_path=fasta_path,
|
|
454
|
+
identity_threshold=ident_threshold,
|
|
455
|
+
mapping_batch_size=mapping_batch_size,
|
|
456
|
+
sample_id=sample_id,
|
|
457
|
+
)
|
|
458
|
+
await project.save()
|
|
459
|
+
|
|
460
|
+
# Step 2: Ion coverage
|
|
461
|
+
typer.echo("Step 2/3: Calculating ion coverage...")
|
|
462
|
+
from dasmixer.api.calculations.ppm.seqfixer import SeqfixerParams
|
|
463
|
+
from dasmixer.api.calculations.spectra.ion_match import IonMatchParameters, process_identificatons_batch
|
|
464
|
+
|
|
465
|
+
ion_match_ions = await project.get_setting("ion_match_ions", "b,y")
|
|
466
|
+
ion_match_tolerance = await project.get_setting("ion_match_tolerance", "20.0")
|
|
467
|
+
ion_match_mode = await project.get_setting("ion_match_mode", "largest")
|
|
468
|
+
ion_match_water_loss = await project.get_setting("ion_match_water_loss", "True")
|
|
469
|
+
ion_match_ammonia_loss = await project.get_setting("ion_match_ammonia_loss", "True")
|
|
470
|
+
ion_fragment_charges = await project.get_setting("ion_fragment_charges", "1,2")
|
|
471
|
+
|
|
472
|
+
seqfixer_params = SeqfixerParams(
|
|
473
|
+
min_charge=1, max_charge=4, max_isotope_offset=3, max_ptm=2, max_ptm_sites=3,
|
|
474
|
+
)
|
|
475
|
+
ion_params = IonMatchParameters(
|
|
476
|
+
ion_types=ion_match_ions.split(","),
|
|
477
|
+
ppm_tolerance=float(ion_match_tolerance),
|
|
478
|
+
match_mode=ion_match_mode,
|
|
479
|
+
water_loss=_parse_bool(ion_match_water_loss),
|
|
480
|
+
ammonia_loss=_parse_bool(ion_match_ammonia_loss),
|
|
481
|
+
fragment_charges=[int(c) for c in ion_fragment_charges.split(",")],
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
idents_df = await project.get_identifications(sample_id=sample_id)
|
|
485
|
+
if not idents_df.empty:
|
|
486
|
+
batch_size = getattr(app_config, 'identification_processing_batch_size', 500)
|
|
487
|
+
for start in range(0, len(idents_df), batch_size):
|
|
488
|
+
batch = idents_df.iloc[start:start + batch_size]
|
|
489
|
+
data_rows = process_identificatons_batch(batch, ion_params, seqfixer_params)
|
|
490
|
+
if data_rows:
|
|
491
|
+
await project.put_identification_data_batch(data_rows)
|
|
492
|
+
await project.save()
|
|
493
|
+
|
|
494
|
+
# Step 3: Select preferred
|
|
495
|
+
typer.echo("Step 3/3: Selecting preferred identifications...")
|
|
496
|
+
if criterion is None:
|
|
497
|
+
criterion = await project.get_setting("preferred_criterion", "intensity")
|
|
498
|
+
|
|
499
|
+
from dasmixer.api.calculations.peptides.matching import select_preferred_identifications
|
|
500
|
+
await select_preferred_identifications(
|
|
501
|
+
project, criterion, tool_settings, sample_id=sample_id,
|
|
502
|
+
)
|
|
503
|
+
await project.save()
|
|
504
|
+
|
|
505
|
+
typer.echo("✓ Peptide calculations complete!")
|
|
506
|
+
|
|
507
|
+
try:
|
|
508
|
+
asyncio.run(_run())
|
|
509
|
+
except typer.Exit:
|
|
510
|
+
raise
|
|
511
|
+
except Exception as e:
|
|
512
|
+
typer.echo(f"Error: {e}", err=True)
|
|
513
|
+
raise typer.Exit(1)
|