dasmixer-cli 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """CLI package for DASMixer."""
2
+
3
+ # CLI implementation will be added in stage 3
@@ -0,0 +1,3 @@
1
+ """CLI command modules."""
2
+
3
+ __all__ = ['project', 'subset', 'import_data']
@@ -0,0 +1,513 @@
1
+ """CLI commands for running pipeline calculations."""
2
+
3
+ import json
4
+ import typer
5
+ from pathlib import Path
6
+ from typing import Annotated
7
+ import asyncio
8
+ import pandas as pd
9
+ from dasmixer.api.project.project import Project
10
+ from dasmixer.api.config import config as app_config
11
+
12
+ app = typer.Typer(help="Run pipeline calculations")
13
+
14
+
15
+ def _get_setting(project, key: str, default: str) -> str:
16
+ """Get a setting from project_settings with fallback."""
17
+ import asyncio
18
+ try:
19
+ rows = asyncio.run_coroutine_threadsafe(
20
+ project.execute_query(
21
+ "SELECT value FROM project_settings WHERE key=?",
22
+ [key],
23
+ ),
24
+ None,
25
+ )
26
+ # Fallback: use get_setting directly
27
+ return default
28
+ except Exception:
29
+ return default
30
+
31
+
32
+ def _parse_bool(val: str) -> bool:
33
+ return val.lower() in ("true", "1", "yes")
34
+
35
+
36
+ # ---------------------------------------------------------------------------
37
+ # ion-coverage
38
+ # ---------------------------------------------------------------------------
39
+
40
+ @app.command()
41
+ def ion_coverage(
42
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
43
+ recalc_all: Annotated[bool, typer.Option("--recalc-all", help="Recalculate all, including already processed")] = False,
44
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
45
+ tolerance: Annotated[float, typer.Option("--tolerance", help="PPM tolerance (overrides project setting)")] = None,
46
+ ions: Annotated[str, typer.Option("--ions", help="Ion types, comma-separated (overrides project setting)")] = None,
47
+ ):
48
+ """
49
+ Calculate ion coverage for identifications.
50
+
51
+ Reads settings from project_settings, can override with CLI options.
52
+ """
53
+ async def _run():
54
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
55
+ # Load settings from project_settings
56
+ ion_match_ions = await project.get_setting("ion_match_ions", "b,y")
57
+ ion_match_tolerance = await project.get_setting("ion_match_tolerance", "20.0")
58
+ ion_match_mode = await project.get_setting("ion_match_mode", "largest")
59
+ ion_match_water_loss = await project.get_setting("ion_match_water_loss", "True")
60
+ ion_match_ammonia_loss = await project.get_setting("ion_match_ammonia_loss", "True")
61
+ ion_fragment_charges = await project.get_setting("ion_fragment_charges", "1,2")
62
+ seqfixer_min_charge = await project.get_setting("seqfixer_min_charge", "1")
63
+ seqfixer_max_charge = await project.get_setting("seqfixer_max_charge", "4")
64
+ seqfixer_max_isotope_offset = await project.get_setting("seqfixer_max_isotope_offset", "3")
65
+ seqfixer_max_ptm = await project.get_setting("seqfixer_max_ptm", "2")
66
+ seqfixer_max_ptm_sites = await project.get_setting("seqfixer_max_ptm_sites", "3")
67
+
68
+ # Override with CLI options
69
+ if tolerance is not None:
70
+ ion_match_tolerance = str(tolerance)
71
+ if ions is not None:
72
+ ion_match_ions = ions
73
+
74
+ from dasmixer.api.calculations.ppm.seqfixer import SeqfixerParams
75
+ from dasmixer.api.calculations.spectra.ion_match import IonMatchParameters, process_identificatons_batch
76
+
77
+ seqfixer_params = SeqfixerParams(
78
+ min_charge=int(seqfixer_min_charge),
79
+ max_charge=int(seqfixer_max_charge),
80
+ max_isotope_offset=int(seqfixer_max_isotope_offset),
81
+ max_ptm=int(seqfixer_max_ptm),
82
+ max_ptm_sites=int(seqfixer_max_ptm_sites),
83
+ )
84
+
85
+ ion_params = IonMatchParameters(
86
+ ion_types=ion_match_ions.split(","),
87
+ ppm_tolerance=float(ion_match_tolerance),
88
+ match_mode=ion_match_mode,
89
+ water_loss=_parse_bool(ion_match_water_loss),
90
+ ammonia_loss=_parse_bool(ion_match_ammonia_loss),
91
+ fragment_charges=[int(c) for c in ion_fragment_charges.split(",")],
92
+ )
93
+
94
+ # Get identifications without coverage (or all if recalc_all)
95
+ if recalc_all:
96
+ idents_df = await project.get_identifications(sample_id=sample_id)
97
+ else:
98
+ idents_df = await project.get_identifications(sample_id=sample_id)
99
+ if not idents_df.empty and "ppm" in idents_df.columns:
100
+ idents_df = idents_df[idents_df["ppm"].isna()]
101
+
102
+ if idents_df.empty:
103
+ typer.echo("No identifications to process.")
104
+ return
105
+
106
+ total = len(idents_df)
107
+ batch_size = getattr(app_config, 'identification_processing_batch_size', 500)
108
+ processed = 0
109
+
110
+ import concurrent.futures
111
+ with typer.progressbar(length=total, label="Processing") as progress:
112
+ for start in range(0, total, batch_size):
113
+ batch = idents_df.iloc[start:start + batch_size]
114
+ data_rows = process_identificatons_batch(
115
+ batch, ion_params, seqfixer_params,
116
+ )
117
+ if data_rows:
118
+ await project.put_identification_data_batch(data_rows)
119
+ processed += len(batch)
120
+ progress.update(len(batch))
121
+
122
+ await project.save()
123
+ typer.echo(f"✓ Processed {processed} identifications")
124
+
125
+ try:
126
+ asyncio.run(_run())
127
+ except typer.Exit:
128
+ raise
129
+ except Exception as e:
130
+ typer.echo(f"Error: {e}", err=True)
131
+ raise typer.Exit(1)
132
+
133
+
134
+ # ---------------------------------------------------------------------------
135
+ # preferred
136
+ # ---------------------------------------------------------------------------
137
+
138
+ @app.command()
139
+ def preferred(
140
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
141
+ criterion: Annotated[str, typer.Option("--criterion", help="Selection criterion: ppm or intensity")] = None,
142
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
143
+ ):
144
+ """
145
+ Select preferred identifications per spectrum.
146
+ """
147
+ async def _run():
148
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
149
+ if criterion is None:
150
+ criterion = await project.get_setting("preferred_criterion", "intensity")
151
+
152
+ from dasmixer.api.calculations.peptides.matching import select_preferred_identifications
153
+
154
+ # Build tool_settings from project_settings
155
+ tools = await project.get_tools()
156
+ tool_settings = {}
157
+ for t in tools:
158
+ tid = t.id
159
+ ts = dict(t.settings or {})
160
+ ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
161
+ ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
162
+ ts.setdefault("coverage_min", await project.get_setting(f"tool_{tid}_coverage_min", "0"))
163
+ ts.setdefault("length_min", await project.get_setting(f"tool_{tid}_length_min", "5"))
164
+ tool_settings[tid] = ts
165
+
166
+ count = await select_preferred_identifications(
167
+ project, criterion, tool_settings, sample_id=sample_id,
168
+ )
169
+ await project.save()
170
+ typer.echo(f"✓ Selected {count} preferred identifications")
171
+
172
+ try:
173
+ asyncio.run(_run())
174
+ except typer.Exit:
175
+ raise
176
+ except Exception as e:
177
+ typer.echo(f"Error: {e}", err=True)
178
+ raise typer.Exit(1)
179
+
180
+
181
+ # ---------------------------------------------------------------------------
182
+ # peptide-match
183
+ # ---------------------------------------------------------------------------
184
+
185
+ @app.command()
186
+ def peptide_match(
187
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
188
+ fasta: Annotated[str, typer.Option("--fasta", help="Path to FASTA file (overrides project setting)")] = None,
189
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
190
+ threshold: Annotated[float, typer.Option("--threshold", help="Identity threshold")] = None,
191
+ batch_size: Annotated[int, typer.Option("--batch-size", help="Batch size")] = None,
192
+ ):
193
+ """
194
+ Match peptide identifications to protein sequences.
195
+ """
196
+ async def _run():
197
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
198
+ fasta_path = fasta
199
+ if not fasta_path:
200
+ fasta_path = await project.get_setting("fasta_path", "")
201
+ if not fasta_path:
202
+ typer.echo("Error: No FASTA file specified. Use --fasta or set fasta_path in project settings.", err=True)
203
+ raise typer.Exit(1)
204
+
205
+ fasta_path_obj = Path(fasta_path)
206
+ if not fasta_path_obj.exists():
207
+ typer.echo(f"Error: FASTA file not found: {fasta_path}", err=True)
208
+ raise typer.Exit(1)
209
+
210
+ ident_threshold = threshold if threshold is not None else float(
211
+ await project.get_setting("blast_identity_threshold", "0.8")
212
+ )
213
+ mapping_batch_size = batch_size or getattr(app_config, 'protein_mapping_batch_size', 1000)
214
+
215
+ from dasmixer.api.calculations.peptides.protein_map import map_proteins
216
+
217
+ tools = await project.get_tools()
218
+ tool_settings = {}
219
+ for t in tools:
220
+ tid = t.id
221
+ ts = dict(t.settings or {})
222
+ ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
223
+ ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
224
+ tool_settings[tid] = ts
225
+
226
+ await project.set_setting("fasta_path", str(fasta_path_obj))
227
+ total = await map_proteins(
228
+ project, tool_settings,
229
+ fasta_path=str(fasta_path_obj),
230
+ identity_threshold=ident_threshold,
231
+ mapping_batch_size=mapping_batch_size,
232
+ sample_id=sample_id,
233
+ )
234
+ await project.save()
235
+ typer.echo(f"✓ Mapped {total} peptide matches")
236
+
237
+ try:
238
+ asyncio.run(_run())
239
+ except typer.Exit:
240
+ raise
241
+ except Exception as e:
242
+ typer.echo(f"Error: {e}", err=True)
243
+ raise typer.Exit(1)
244
+
245
+
246
+ # ---------------------------------------------------------------------------
247
+ # protein-idents
248
+ # ---------------------------------------------------------------------------
249
+
250
+ @app.command()
251
+ def protein_idents(
252
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
253
+ min_peptides: Annotated[int, typer.Option("--min-peptides", help="Minimum peptides per protein")] = None,
254
+ min_unique: Annotated[int, typer.Option("--min-unique", help="Minimum unique evidence")] = None,
255
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
256
+ ):
257
+ """
258
+ Calculate protein identifications from peptide matches.
259
+ """
260
+ async def _run():
261
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
262
+ if min_peptides is None:
263
+ min_peptides = int(await project.get_setting("proteins_min_peptides", "2"))
264
+ if min_unique is None:
265
+ min_unique = int(await project.get_setting("proteins_min_unique_evidence", "1"))
266
+
267
+ await project.set_setting("proteins_min_peptides", str(min_peptides))
268
+ await project.set_setting("proteins_min_unique_evidence", str(min_unique))
269
+
270
+ from dasmixer.api.calculations.proteins.map_identifications import find_protein_identifications
271
+
272
+ # Get joined peptide data
273
+ filters = {}
274
+ if sample_id is not None:
275
+ filters['sample_id'] = sample_id
276
+ joined_data = await project.get_joined_peptide_data(**filters)
277
+
278
+ if joined_data.empty:
279
+ typer.echo("No peptide data found. Run 'calculate peptide-match' first.")
280
+ return
281
+
282
+ # Get all proteins as sequences_db
283
+ proteins_df = await project.get_proteins()
284
+ if proteins_df.empty:
285
+ typer.echo("No proteins in project. Import a FASTA file first.")
286
+ return
287
+
288
+ sequences_db = dict(zip(proteins_df['id'], proteins_df['sequence']))
289
+
290
+ prot_idents_df = find_protein_identifications(
291
+ joined_data, sequences_db, min_peptides, min_unique,
292
+ )
293
+
294
+ if prot_idents_df.empty:
295
+ typer.echo("No protein identifications found with current thresholds.")
296
+ return
297
+
298
+ await project.add_protein_identifications_batch(prot_idents_df)
299
+ await project.save()
300
+ typer.echo(f"✓ Found {len(prot_idents_df)} protein identifications")
301
+
302
+ try:
303
+ asyncio.run(_run())
304
+ except typer.Exit:
305
+ raise
306
+ except Exception as e:
307
+ typer.echo(f"Error: {e}", err=True)
308
+ raise typer.Exit(1)
309
+
310
+
311
+ # ---------------------------------------------------------------------------
312
+ # lfq
313
+ # ---------------------------------------------------------------------------
314
+
315
+ @app.command()
316
+ def lfq(
317
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
318
+ empai: Annotated[bool, typer.Option("--empai", help="Enable emPAI")] = None,
319
+ ibaq: Annotated[bool, typer.Option("--ibaq", help="Enable iBAQ")] = None,
320
+ nsaf: Annotated[bool, typer.Option("--nsaf", help="Enable NSAF")] = None,
321
+ top3: Annotated[bool, typer.Option("--top3", help="Enable Top3")] = None,
322
+ enzyme: Annotated[str, typer.Option("--enzyme", help="Digestion enzyme")] = None,
323
+ min_length: Annotated[int, typer.Option("--min-length", help="Min peptide length")] = None,
324
+ max_length: Annotated[int, typer.Option("--max-length", help="Max peptide length")] = None,
325
+ max_cleavage: Annotated[int, typer.Option("--max-cleavage", help="Max missed cleavages")] = None,
326
+ empai_base: Annotated[float, typer.Option("--empai-base", help="emPAI base value")] = None,
327
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
328
+ ):
329
+ """
330
+ Calculate protein quantification (LFQ).
331
+
332
+ If no method flags (--empai, --ibaq, etc.) are specified,
333
+ methods are read from project_settings.
334
+ """
335
+ async def _run():
336
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
337
+ # Determine methods
338
+ cli_methods = []
339
+ if empai is not None:
340
+ cli_methods.append("empai")
341
+ if ibaq is not None:
342
+ cli_methods.append("ibaq")
343
+ if nsaf is not None:
344
+ cli_methods.append("nsaf")
345
+ if top3 is not None:
346
+ cli_methods.append("top3")
347
+
348
+ if cli_methods:
349
+ methods = cli_methods
350
+ else:
351
+ methods_str = await project.get_setting("lfq_methods", "empai,ibaq")
352
+ methods = [m.strip() for m in methods_str.split(",") if m.strip()]
353
+
354
+ if not methods:
355
+ typer.echo("Error: No LFQ methods selected", err=True)
356
+ raise typer.Exit(1)
357
+
358
+ # Load / override parameters
359
+ lfq_enzyme = enzyme or await project.get_setting("lfq_enzyme", "trypsin")
360
+ lfq_min_length = min_length or int(await project.get_setting("lfq_min_peptide_length", "7"))
361
+ lfq_max_length = max_length or int(await project.get_setting("lfq_max_peptide_length", "25"))
362
+ lfq_max_cleavage = max_cleavage or int(await project.get_setting("lfq_max_cleavage_sites", "2"))
363
+ lfq_empai_base = empai_base or float(await project.get_setting("lfq_empai_base", "10.0"))
364
+
365
+ # Save settings
366
+ await project.set_setting("lfq_methods", ",".join(methods))
367
+ await project.set_setting("lfq_enzyme", lfq_enzyme)
368
+ await project.set_setting("lfq_min_peptide_length", str(lfq_min_length))
369
+ await project.set_setting("lfq_max_peptide_length", str(lfq_max_length))
370
+ await project.set_setting("lfq_max_cleavage_sites", str(lfq_max_cleavage))
371
+ await project.set_setting("lfq_empai_base", str(lfq_empai_base))
372
+
373
+ from dasmixer.api.calculations.proteins.lfq import calculate_lfq
374
+
375
+ samples = await project.get_samples()
376
+ if sample_id is not None:
377
+ samples = [s for s in samples if s.id == sample_id]
378
+ if not samples:
379
+ typer.echo(f"Error: Sample ID {sample_id} not found", err=True)
380
+ raise typer.Exit(1)
381
+
382
+ calculated_count = 0
383
+ for s in samples:
384
+ await calculate_lfq(
385
+ project,
386
+ sample_id=s.id,
387
+ methods=methods,
388
+ enzyme=lfq_enzyme,
389
+ min_peptide_length=lfq_min_length,
390
+ max_peptide_length=lfq_max_length,
391
+ max_cleavage_sites=lfq_max_cleavage,
392
+ empai_base=lfq_empai_base,
393
+ )
394
+ calculated_count += 1
395
+ typer.echo(f" Sample '{s.name}': done")
396
+
397
+ await project.save()
398
+ typer.echo(f"✓ LFQ calculated for {calculated_count} samples, methods: {methods}")
399
+
400
+ try:
401
+ asyncio.run(_run())
402
+ except typer.Exit:
403
+ raise
404
+ except Exception as e:
405
+ typer.echo(f"Error: {e}", err=True)
406
+ raise typer.Exit(1)
407
+
408
+
409
+ # ---------------------------------------------------------------------------
410
+ # peptides (full pipeline)
411
+ # ---------------------------------------------------------------------------
412
+
413
+ @app.command()
414
+ def peptides(
415
+ project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
416
+ fasta: Annotated[str, typer.Option("--fasta", help="Path to FASTA file")] = None,
417
+ criterion: Annotated[str, typer.Option("--criterion", help="Selection criterion: ppm or intensity")] = None,
418
+ sample_id: Annotated[int, typer.Option("--sample-id", help="Limit to one sample ID")] = None,
419
+ ):
420
+ """
421
+ Run full peptide calculation pipeline:
422
+ 1. Match proteins
423
+ 2. Calculate ion coverage
424
+ 3. Select preferred identifications
425
+ """
426
+ async def _run():
427
+ async with Project(path=Path(project_path), create_if_not_exists=False) as project:
428
+ # Step 1: Match proteins
429
+ typer.echo("Step 1/3: Matching proteins...")
430
+ fasta_path = fasta or await project.get_setting("fasta_path", "")
431
+ if not fasta_path:
432
+ typer.echo("Error: No FASTA file specified", err=True)
433
+ raise typer.Exit(1)
434
+ if not Path(fasta_path).exists():
435
+ typer.echo(f"Error: FASTA file not found: {fasta_path}", err=True)
436
+ raise typer.Exit(1)
437
+
438
+ tools = await project.get_tools()
439
+ tool_settings = {}
440
+ for t in tools:
441
+ tid = t.id
442
+ ts = dict(t.settings or {})
443
+ ts.setdefault("score_min", await project.get_setting(f"tool_{tid}_score_min", "0"))
444
+ ts.setdefault("ppm_max", await project.get_setting(f"tool_{tid}_ppm_max", "20"))
445
+ tool_settings[tid] = ts
446
+
447
+ ident_threshold = float(await project.get_setting("blast_identity_threshold", "0.8"))
448
+ mapping_batch_size = getattr(app_config, 'protein_mapping_batch_size', 1000)
449
+
450
+ from dasmixer.api.calculations.peptides.protein_map import map_proteins
451
+ await map_proteins(
452
+ project, tool_settings,
453
+ fasta_path=fasta_path,
454
+ identity_threshold=ident_threshold,
455
+ mapping_batch_size=mapping_batch_size,
456
+ sample_id=sample_id,
457
+ )
458
+ await project.save()
459
+
460
+ # Step 2: Ion coverage
461
+ typer.echo("Step 2/3: Calculating ion coverage...")
462
+ from dasmixer.api.calculations.ppm.seqfixer import SeqfixerParams
463
+ from dasmixer.api.calculations.spectra.ion_match import IonMatchParameters, process_identificatons_batch
464
+
465
+ ion_match_ions = await project.get_setting("ion_match_ions", "b,y")
466
+ ion_match_tolerance = await project.get_setting("ion_match_tolerance", "20.0")
467
+ ion_match_mode = await project.get_setting("ion_match_mode", "largest")
468
+ ion_match_water_loss = await project.get_setting("ion_match_water_loss", "True")
469
+ ion_match_ammonia_loss = await project.get_setting("ion_match_ammonia_loss", "True")
470
+ ion_fragment_charges = await project.get_setting("ion_fragment_charges", "1,2")
471
+
472
+ seqfixer_params = SeqfixerParams(
473
+ min_charge=1, max_charge=4, max_isotope_offset=3, max_ptm=2, max_ptm_sites=3,
474
+ )
475
+ ion_params = IonMatchParameters(
476
+ ion_types=ion_match_ions.split(","),
477
+ ppm_tolerance=float(ion_match_tolerance),
478
+ match_mode=ion_match_mode,
479
+ water_loss=_parse_bool(ion_match_water_loss),
480
+ ammonia_loss=_parse_bool(ion_match_ammonia_loss),
481
+ fragment_charges=[int(c) for c in ion_fragment_charges.split(",")],
482
+ )
483
+
484
+ idents_df = await project.get_identifications(sample_id=sample_id)
485
+ if not idents_df.empty:
486
+ batch_size = getattr(app_config, 'identification_processing_batch_size', 500)
487
+ for start in range(0, len(idents_df), batch_size):
488
+ batch = idents_df.iloc[start:start + batch_size]
489
+ data_rows = process_identificatons_batch(batch, ion_params, seqfixer_params)
490
+ if data_rows:
491
+ await project.put_identification_data_batch(data_rows)
492
+ await project.save()
493
+
494
+ # Step 3: Select preferred
495
+ typer.echo("Step 3/3: Selecting preferred identifications...")
496
+ if criterion is None:
497
+ criterion = await project.get_setting("preferred_criterion", "intensity")
498
+
499
+ from dasmixer.api.calculations.peptides.matching import select_preferred_identifications
500
+ await select_preferred_identifications(
501
+ project, criterion, tool_settings, sample_id=sample_id,
502
+ )
503
+ await project.save()
504
+
505
+ typer.echo("✓ Peptide calculations complete!")
506
+
507
+ try:
508
+ asyncio.run(_run())
509
+ except typer.Exit:
510
+ raise
511
+ except Exception as e:
512
+ typer.echo(f"Error: {e}", err=True)
513
+ raise typer.Exit(1)