dasmixer-cli 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dasmixer/cli/__init__.py +3 -0
- dasmixer/cli/commands/__init__.py +3 -0
- dasmixer/cli/commands/calculate.py +513 -0
- dasmixer/cli/commands/import_data.py +499 -0
- dasmixer/cli/commands/import_project.py +75 -0
- dasmixer/cli/commands/portable.py +54 -0
- dasmixer/cli/commands/project.py +94 -0
- dasmixer/cli/commands/subset.py +130 -0
- dasmixer/cli/commands/tool.py +147 -0
- dasmixer/cli/main.py +52 -0
- dasmixer_cli-0.6.0.dist-info/METADATA +51 -0
- dasmixer_cli-0.6.0.dist-info/RECORD +14 -0
- dasmixer_cli-0.6.0.dist-info/WHEEL +4 -0
- dasmixer_cli-0.6.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,499 @@
|
|
|
1
|
+
"""CLI commands for importing data files."""
|
|
2
|
+
|
|
3
|
+
import typer
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import asyncio
|
|
6
|
+
from typing import Annotated
|
|
7
|
+
from dasmixer.api.project.project import Project
|
|
8
|
+
from dasmixer.api.inputs.registry import registry
|
|
9
|
+
from dasmixer.api.config import config
|
|
10
|
+
from dasmixer.utils.seek_files import seek_files
|
|
11
|
+
|
|
12
|
+
app = typer.Typer(help="Import data files")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@app.command()
|
|
16
|
+
def mgf_pattern(
|
|
17
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
18
|
+
folder: Annotated[str, typer.Option("--folder", "-f", help="Folder to search")] = ...,
|
|
19
|
+
file_pattern: Annotated[str, typer.Option("--pattern", "-p", help="File pattern (e.g., *.mgf)")] = "*.mgf",
|
|
20
|
+
id_pattern: Annotated[str, typer.Option("--id-pattern", "-i", help="Sample ID pattern (e.g., {id}_*.mgf)")] = "{id}*.mgf",
|
|
21
|
+
parser: Annotated[str, typer.Option("--parser", help="Parser name")] = "MGF",
|
|
22
|
+
group: Annotated[str, typer.Option("--group", "-g", help="Group to assign samples")] = "Control"
|
|
23
|
+
):
|
|
24
|
+
"""
|
|
25
|
+
Import MGF files using pattern matching.
|
|
26
|
+
|
|
27
|
+
Example:
|
|
28
|
+
dasmixer project.dasmix import mgf-pattern \\
|
|
29
|
+
--folder /data/spectra \\
|
|
30
|
+
--pattern "*.mgf" \\
|
|
31
|
+
--id-pattern "{id}_run*.mgf" \\
|
|
32
|
+
--group Control
|
|
33
|
+
"""
|
|
34
|
+
project_path = Path(project_path)
|
|
35
|
+
|
|
36
|
+
if not project_path.exists():
|
|
37
|
+
typer.echo(f"Error: Project file not found: {project_path}", err=True)
|
|
38
|
+
raise typer.Exit(1)
|
|
39
|
+
|
|
40
|
+
folder_path = Path(folder)
|
|
41
|
+
if not folder_path.exists():
|
|
42
|
+
typer.echo(f"Error: Folder not found: {folder}", err=True)
|
|
43
|
+
raise typer.Exit(1)
|
|
44
|
+
|
|
45
|
+
# Find files
|
|
46
|
+
try:
|
|
47
|
+
files = seek_files(folder_path, file_pattern, id_pattern)
|
|
48
|
+
except Exception as e:
|
|
49
|
+
typer.echo(f"Error searching files: {e}", err=True)
|
|
50
|
+
raise typer.Exit(1)
|
|
51
|
+
|
|
52
|
+
if not files:
|
|
53
|
+
typer.echo("No files found matching pattern", err=True)
|
|
54
|
+
raise typer.Exit(1)
|
|
55
|
+
|
|
56
|
+
# Show found files
|
|
57
|
+
typer.echo(f"\nFound {len(files)} file(s):")
|
|
58
|
+
typer.echo("-" * 60)
|
|
59
|
+
for file_path, sample_id in files:
|
|
60
|
+
display_id = sample_id or "UNKNOWN"
|
|
61
|
+
typer.echo(f" {file_path.name} → Sample ID: {display_id}")
|
|
62
|
+
|
|
63
|
+
if not typer.confirm("\nProceed with import?"):
|
|
64
|
+
typer.echo("Cancelled")
|
|
65
|
+
raise typer.Exit(0)
|
|
66
|
+
|
|
67
|
+
# Get parser
|
|
68
|
+
try:
|
|
69
|
+
parser_class = registry.get_parser(parser, "spectra")
|
|
70
|
+
except KeyError as e:
|
|
71
|
+
typer.echo(f"Error: {e}", err=True)
|
|
72
|
+
raise typer.Exit(1)
|
|
73
|
+
|
|
74
|
+
# Import files
|
|
75
|
+
async def _import():
|
|
76
|
+
async with Project(path=project_path, create_if_not_exists=False) as project:
|
|
77
|
+
# Get or create group
|
|
78
|
+
subsets = await project.get_subsets()
|
|
79
|
+
subset = next((s for s in subsets if s.name == group), None)
|
|
80
|
+
|
|
81
|
+
if not subset:
|
|
82
|
+
subset = await project.add_subset(group)
|
|
83
|
+
typer.echo(f"✓ Created group: {group}")
|
|
84
|
+
|
|
85
|
+
# Import with progress
|
|
86
|
+
with typer.progressbar(
|
|
87
|
+
files,
|
|
88
|
+
label="Importing",
|
|
89
|
+
show_pos=True
|
|
90
|
+
) as progress:
|
|
91
|
+
for file_path, sample_id in progress:
|
|
92
|
+
# Use filename as sample_id if not detected
|
|
93
|
+
if not sample_id:
|
|
94
|
+
sample_id = file_path.stem
|
|
95
|
+
|
|
96
|
+
try:
|
|
97
|
+
# Parse file
|
|
98
|
+
parser_instance = parser_class(str(file_path))
|
|
99
|
+
spectra_df = await parser_instance.parse_batch()
|
|
100
|
+
|
|
101
|
+
# Add sample if not exists
|
|
102
|
+
sample = await project.get_sample_by_name(sample_id)
|
|
103
|
+
if not sample:
|
|
104
|
+
sample = await project.add_sample(
|
|
105
|
+
sample_id,
|
|
106
|
+
subset_id=subset.id
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Add spectra file
|
|
110
|
+
spectra_file_id = await project.add_spectra_file(
|
|
111
|
+
sample.id,
|
|
112
|
+
parser,
|
|
113
|
+
str(file_path)
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
# Add spectra
|
|
117
|
+
await project.add_spectra_batch(spectra_file_id, spectra_df)
|
|
118
|
+
|
|
119
|
+
except Exception as e:
|
|
120
|
+
typer.echo(f"\n Error importing {file_path.name}: {e}", err=True)
|
|
121
|
+
|
|
122
|
+
typer.echo(f"\n✓ Imported {len(files)} file(s) successfully")
|
|
123
|
+
|
|
124
|
+
try:
|
|
125
|
+
asyncio.run(_import())
|
|
126
|
+
config.update_last_import_folder(folder)
|
|
127
|
+
except Exception as e:
|
|
128
|
+
typer.echo(f"\nError during import: {e}", err=True)
|
|
129
|
+
raise typer.Exit(1)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@app.command()
|
|
133
|
+
def mgf_file(
|
|
134
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
135
|
+
file: Annotated[str, typer.Option("--file", "-f", help="Path to MGF file")] = ...,
|
|
136
|
+
sample_id: Annotated[str, typer.Option("--sample-id", "-s", help="Sample ID")] = ...,
|
|
137
|
+
parser: Annotated[str, typer.Option("--parser", help="Parser name")] = "MGF",
|
|
138
|
+
group: Annotated[str, typer.Option("--group", "-g", help="Group to assign sample")] = "Control"
|
|
139
|
+
):
|
|
140
|
+
"""
|
|
141
|
+
Import single MGF file.
|
|
142
|
+
|
|
143
|
+
Example:
|
|
144
|
+
dasmixer project.dasmix import mgf-file \\
|
|
145
|
+
--file /data/sample1.mgf \\
|
|
146
|
+
--sample-id "Sample1" \\
|
|
147
|
+
--group Control
|
|
148
|
+
"""
|
|
149
|
+
project_path = Path(project_path)
|
|
150
|
+
file_path = Path(file)
|
|
151
|
+
|
|
152
|
+
if not project_path.exists():
|
|
153
|
+
typer.echo(f"Error: Project file not found: {project_path}", err=True)
|
|
154
|
+
raise typer.Exit(1)
|
|
155
|
+
|
|
156
|
+
if not file_path.exists():
|
|
157
|
+
typer.echo(f"Error: File not found: {file}", err=True)
|
|
158
|
+
raise typer.Exit(1)
|
|
159
|
+
|
|
160
|
+
# Get parser
|
|
161
|
+
try:
|
|
162
|
+
parser_class = registry.get_parser(parser, "spectra")
|
|
163
|
+
except KeyError as e:
|
|
164
|
+
typer.echo(f"Error: {e}", err=True)
|
|
165
|
+
raise typer.Exit(1)
|
|
166
|
+
|
|
167
|
+
# Import file
|
|
168
|
+
async def _import():
|
|
169
|
+
async with Project(path=project_path, create_if_not_exists=False) as project:
|
|
170
|
+
# Get or create group
|
|
171
|
+
subsets = await project.get_subsets()
|
|
172
|
+
subset = next((s for s in subsets if s.name == group), None)
|
|
173
|
+
|
|
174
|
+
if not subset:
|
|
175
|
+
subset = await project.add_subset(group)
|
|
176
|
+
typer.echo(f"✓ Created group: {group}")
|
|
177
|
+
|
|
178
|
+
typer.echo(f"Importing {file_path.name}...")
|
|
179
|
+
|
|
180
|
+
# Parse file
|
|
181
|
+
parser_instance = parser_class(str(file_path))
|
|
182
|
+
spectra_df = await parser_instance.parse_batch()
|
|
183
|
+
|
|
184
|
+
# Add sample if not exists
|
|
185
|
+
sample = await project.get_sample_by_name(sample_id)
|
|
186
|
+
if not sample:
|
|
187
|
+
sample = await project.add_sample(
|
|
188
|
+
sample_id,
|
|
189
|
+
subset_id=subset.id
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
# Add spectra file
|
|
193
|
+
spectra_file_id = await project.add_spectra_file(
|
|
194
|
+
sample.id,
|
|
195
|
+
parser,
|
|
196
|
+
str(file_path)
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
# Add spectra
|
|
200
|
+
await project.add_spectra_batch(spectra_file_id, spectra_df)
|
|
201
|
+
|
|
202
|
+
typer.echo(f"✓ Imported {len(spectra_df)} spectra from {file_path.name}")
|
|
203
|
+
typer.echo(f" Sample: {sample_id}")
|
|
204
|
+
typer.echo(f" Group: {group}")
|
|
205
|
+
|
|
206
|
+
try:
|
|
207
|
+
asyncio.run(_import())
|
|
208
|
+
config.update_last_import_folder(str(file_path.parent))
|
|
209
|
+
except Exception as e:
|
|
210
|
+
typer.echo(f"Error importing file: {e}", err=True)
|
|
211
|
+
raise typer.Exit(1)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
@app.command()
|
|
215
|
+
async def ident_file(
|
|
216
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
217
|
+
file: Annotated[str, typer.Option("--file", "-f", help="Path to identification file")] = ...,
|
|
218
|
+
sample_id: Annotated[str, typer.Option("--sample-id", "-s", help="Sample name (must exist)")] = ...,
|
|
219
|
+
parser: Annotated[str, typer.Option("--parser", help="Parser name (e.g., PowerNovo2)")] = ...,
|
|
220
|
+
tool: Annotated[str, typer.Option("--tool", help="Tool name (must exist in project)")] = ...,
|
|
221
|
+
spectra_file_id: Annotated[int, typer.Option("--spectra-file-id", help="Spectra file ID (auto-detected if omitted)")] = None,
|
|
222
|
+
):
|
|
223
|
+
"""
|
|
224
|
+
Import single identification file.
|
|
225
|
+
|
|
226
|
+
Requires that corresponding spectra file is already imported for the sample
|
|
227
|
+
and the tool has been added to the project.
|
|
228
|
+
|
|
229
|
+
Example:
|
|
230
|
+
dasmixer project.dasmix import ident-file \\
|
|
231
|
+
--file /data/sample1_powernovo.csv \\
|
|
232
|
+
--sample-id "Sample1" \\
|
|
233
|
+
--parser PowerNovo2 \\
|
|
234
|
+
--tool PowerNovo2
|
|
235
|
+
"""
|
|
236
|
+
await _import_ident_file_internal(
|
|
237
|
+
project_path=Path(project_path),
|
|
238
|
+
file_path=Path(file),
|
|
239
|
+
sample_name=sample_id,
|
|
240
|
+
parser_name=parser,
|
|
241
|
+
tool_name=tool,
|
|
242
|
+
spectra_file_id=spectra_file_id,
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
async def _import_ident_file_internal(
|
|
247
|
+
project_path: Path,
|
|
248
|
+
file_path: Path,
|
|
249
|
+
sample_name: str,
|
|
250
|
+
parser_name: str,
|
|
251
|
+
tool_name: str,
|
|
252
|
+
spectra_file_id: int | None = None,
|
|
253
|
+
quiet: bool = False,
|
|
254
|
+
) -> int:
|
|
255
|
+
"""Internal helper to import a single identification file."""
|
|
256
|
+
from dasmixer.api.config import config as app_config
|
|
257
|
+
|
|
258
|
+
if not project_path.exists():
|
|
259
|
+
typer.echo(f"Error: Project file not found: {project_path}", err=True)
|
|
260
|
+
raise typer.Exit(1)
|
|
261
|
+
|
|
262
|
+
if not file_path.exists():
|
|
263
|
+
typer.echo(f"Error: File not found: {file_path}", err=True)
|
|
264
|
+
raise typer.Exit(1)
|
|
265
|
+
|
|
266
|
+
async with Project(path=project_path, create_if_not_exists=False) as project:
|
|
267
|
+
# Find sample by name
|
|
268
|
+
samples = await project.get_samples()
|
|
269
|
+
sample = next((s for s in samples if s.name == sample_name), None)
|
|
270
|
+
if not sample:
|
|
271
|
+
typer.echo(f"Error: Sample '{sample_name}' not found", err=True)
|
|
272
|
+
raise typer.Exit(1)
|
|
273
|
+
|
|
274
|
+
# Find tool by name
|
|
275
|
+
tools = await project.get_tools()
|
|
276
|
+
tool_obj = next((t for t in tools if t.name == tool_name), None)
|
|
277
|
+
if not tool_obj:
|
|
278
|
+
typer.echo(f"Error: Tool '{tool_name}' not found. Use 'dasmixer-cli tool add' first", err=True)
|
|
279
|
+
raise typer.Exit(1)
|
|
280
|
+
|
|
281
|
+
# Determine spectra_file_id
|
|
282
|
+
if spectra_file_id is None:
|
|
283
|
+
# Get first spectra file for this sample
|
|
284
|
+
rows = await project.execute_query(
|
|
285
|
+
"SELECT id FROM spectre_file WHERE sample_id=? ORDER BY id LIMIT 1",
|
|
286
|
+
[sample.id],
|
|
287
|
+
)
|
|
288
|
+
if not rows:
|
|
289
|
+
typer.echo(f"Error: No spectra files found for sample '{sample_name}'", err=True)
|
|
290
|
+
raise typer.Exit(1)
|
|
291
|
+
spectra_file_id = rows[0]["id"]
|
|
292
|
+
|
|
293
|
+
# Get parser
|
|
294
|
+
try:
|
|
295
|
+
parser_class = registry.get_parser(parser_name, "identification")
|
|
296
|
+
except KeyError:
|
|
297
|
+
typer.echo(f"Error: Unknown identification parser '{parser_name}'", err=True)
|
|
298
|
+
raise typer.Exit(1)
|
|
299
|
+
|
|
300
|
+
# Create identification file entry
|
|
301
|
+
ident_file_id = await project.add_identification_file(
|
|
302
|
+
spectra_file_id=spectra_file_id,
|
|
303
|
+
tool_id=tool_obj.id,
|
|
304
|
+
file_path=str(file_path),
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
# Get spectra ID list
|
|
308
|
+
parser_instance = parser_class(str(file_path))
|
|
309
|
+
spectra_id_field = getattr(parser_instance, 'spectra_id_field', 'spectrum_id')
|
|
310
|
+
spectra_list = await project.get_spectra_idlist(spectra_file_id, by=spectra_id_field)
|
|
311
|
+
|
|
312
|
+
if not spectra_list:
|
|
313
|
+
typer.echo(f"Warning: No spectra found for spectra file {spectra_file_id}", err=True)
|
|
314
|
+
return 0
|
|
315
|
+
|
|
316
|
+
# Build lookup: ID → spectre_id
|
|
317
|
+
spectra_map = {str(s[spectra_id_field]): s['spectre_id'] for s in spectra_list}
|
|
318
|
+
|
|
319
|
+
if not quiet:
|
|
320
|
+
typer.echo(f"Importing {file_path.name}...")
|
|
321
|
+
|
|
322
|
+
total = 0
|
|
323
|
+
batch_size = getattr(app_config, 'identification_batch_size', 1000)
|
|
324
|
+
async for batch_df, _ in parser_instance.parse_batch(batch_size=batch_size):
|
|
325
|
+
if batch_df.empty:
|
|
326
|
+
continue
|
|
327
|
+
batch_df['ident_file_id'] = ident_file_id
|
|
328
|
+
# Map spectra IDs
|
|
329
|
+
id_col = spectra_id_field
|
|
330
|
+
if id_col in batch_df.columns:
|
|
331
|
+
batch_df['spectre_id'] = batch_df[id_col].astype(str).map(spectra_map)
|
|
332
|
+
matched = batch_df['spectre_id'].notna()
|
|
333
|
+
unmatched_count = (~matched).sum()
|
|
334
|
+
if unmatched_count > 0 and not quiet:
|
|
335
|
+
typer.echo(f" Warning: {unmatched_count} identifications unmatched to spectra")
|
|
336
|
+
batch_df = batch_df[matched].copy()
|
|
337
|
+
if not batch_df.empty:
|
|
338
|
+
await project.add_identifications_batch(batch_df)
|
|
339
|
+
total += len(batch_df)
|
|
340
|
+
|
|
341
|
+
await project.save()
|
|
342
|
+
|
|
343
|
+
if not quiet:
|
|
344
|
+
typer.echo(f"✓ Imported {total} identifications from {file_path.name}")
|
|
345
|
+
typer.echo(f" Tool: {tool_name}")
|
|
346
|
+
typer.echo(f" Sample: {sample_name}")
|
|
347
|
+
typer.echo(f" Spectra file ID: {spectra_file_id}")
|
|
348
|
+
return total
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
@app.command()
|
|
352
|
+
async def ident_pattern(
|
|
353
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
354
|
+
folder: Annotated[str, typer.Option("--folder", "-f", help="Folder to search")] = ...,
|
|
355
|
+
file_pattern: Annotated[str, typer.Option("--pattern", "-p", help="File pattern")] = "*.csv",
|
|
356
|
+
id_pattern: Annotated[str, typer.Option("--id-pattern", "-i", help="Sample ID pattern")] = "{id}*.csv",
|
|
357
|
+
parser: Annotated[str, typer.Option("--parser", help="Parser name (e.g., PowerNovo2)")] = ...,
|
|
358
|
+
tool: Annotated[str, typer.Option("--tool", help="Tool name")] = ...,
|
|
359
|
+
):
|
|
360
|
+
"""
|
|
361
|
+
Import identification files using pattern matching.
|
|
362
|
+
|
|
363
|
+
Requires that corresponding spectra files are already imported for the samples
|
|
364
|
+
and the tool has been added to the project.
|
|
365
|
+
|
|
366
|
+
Example:
|
|
367
|
+
dasmixer project.dasmix import ident-pattern \\
|
|
368
|
+
--folder /data/results \\
|
|
369
|
+
--pattern "*.csv" \\
|
|
370
|
+
--id-pattern "{id}_powernovo.csv" \\
|
|
371
|
+
--parser PowerNovo2 \\
|
|
372
|
+
--tool PowerNovo2
|
|
373
|
+
"""
|
|
374
|
+
project_path_obj = Path(project_path)
|
|
375
|
+
folder_path = Path(folder)
|
|
376
|
+
|
|
377
|
+
if not project_path_obj.exists():
|
|
378
|
+
typer.echo(f"Error: Project file not found: {project_path}", err=True)
|
|
379
|
+
raise typer.Exit(1)
|
|
380
|
+
|
|
381
|
+
if not folder_path.exists():
|
|
382
|
+
typer.echo(f"Error: Folder not found: {folder}", err=True)
|
|
383
|
+
raise typer.Exit(1)
|
|
384
|
+
|
|
385
|
+
try:
|
|
386
|
+
files = seek_files(folder_path, file_pattern, id_pattern)
|
|
387
|
+
except Exception as e:
|
|
388
|
+
typer.echo(f"Error searching files: {e}", err=True)
|
|
389
|
+
raise typer.Exit(1)
|
|
390
|
+
|
|
391
|
+
if not files:
|
|
392
|
+
typer.echo("No files found matching pattern", err=True)
|
|
393
|
+
raise typer.Exit(1)
|
|
394
|
+
|
|
395
|
+
typer.echo(f"\nFound {len(files)} file(s):")
|
|
396
|
+
typer.echo("-" * 60)
|
|
397
|
+
for file_path, sid in files:
|
|
398
|
+
display_id = sid or "UNKNOWN"
|
|
399
|
+
typer.echo(f" {file_path.name} → Sample ID: {display_id}")
|
|
400
|
+
|
|
401
|
+
if not typer.confirm("\nProceed with import?"):
|
|
402
|
+
typer.echo("Cancelled")
|
|
403
|
+
raise typer.Exit(0)
|
|
404
|
+
|
|
405
|
+
total_imported = 0
|
|
406
|
+
total_files = 0
|
|
407
|
+
errors = []
|
|
408
|
+
|
|
409
|
+
for file_path, sid in files:
|
|
410
|
+
if not sid:
|
|
411
|
+
sid = file_path.stem
|
|
412
|
+
try:
|
|
413
|
+
result = await _import_ident_file_internal(
|
|
414
|
+
project_path=project_path_obj,
|
|
415
|
+
file_path=file_path,
|
|
416
|
+
sample_name=sid,
|
|
417
|
+
parser_name=parser,
|
|
418
|
+
tool_name=tool,
|
|
419
|
+
spectra_file_id=None,
|
|
420
|
+
quiet=True,
|
|
421
|
+
)
|
|
422
|
+
total_imported += result
|
|
423
|
+
total_files += 1
|
|
424
|
+
typer.echo(f" ✓ {file_path.name}: {result} identifications")
|
|
425
|
+
except typer.Exit:
|
|
426
|
+
raise
|
|
427
|
+
except Exception as e:
|
|
428
|
+
errors.append(f"{file_path.name}: {e}")
|
|
429
|
+
typer.echo(f" ✗ {file_path.name}: {e}", err=True)
|
|
430
|
+
|
|
431
|
+
typer.echo(f"\n✓ Imported {total_files} files, {total_imported} identifications total")
|
|
432
|
+
if errors:
|
|
433
|
+
typer.echo(f" {len(errors)} file(s) had errors", err=True)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
@app.command()
|
|
437
|
+
def fasta(
|
|
438
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
439
|
+
file: Annotated[str, typer.Option("--file", "-f", help="Path to FASTA file")] = ...,
|
|
440
|
+
batch_size: Annotated[int, typer.Option("--batch-size", help="Batch size for import")] = 100,
|
|
441
|
+
):
|
|
442
|
+
"""
|
|
443
|
+
Import proteins from a FASTA file.
|
|
444
|
+
|
|
445
|
+
Example:
|
|
446
|
+
dasmixer project.dasmix import fasta \\
|
|
447
|
+
--file /data/uniprot.fasta \\
|
|
448
|
+
--batch-size 200
|
|
449
|
+
"""
|
|
450
|
+
project_path_obj = Path(project_path)
|
|
451
|
+
file_path = Path(file)
|
|
452
|
+
|
|
453
|
+
if not project_path_obj.exists():
|
|
454
|
+
typer.echo(f"Error: Project file not found: {project_path}", err=True)
|
|
455
|
+
raise typer.Exit(1)
|
|
456
|
+
|
|
457
|
+
if not file_path.exists():
|
|
458
|
+
typer.echo(f"Error: FASTA file not found: {file}", err=True)
|
|
459
|
+
raise typer.Exit(1)
|
|
460
|
+
|
|
461
|
+
from dasmixer.api.inputs.proteins.fasta import FastaParser
|
|
462
|
+
|
|
463
|
+
async def _import():
|
|
464
|
+
parser = FastaParser(str(file_path))
|
|
465
|
+
valid = await parser.validate()
|
|
466
|
+
if not valid:
|
|
467
|
+
typer.echo("Error: Invalid FASTA file", err=True)
|
|
468
|
+
raise typer.Exit(1)
|
|
469
|
+
|
|
470
|
+
async with Project(path=project_path_obj, create_if_not_exists=False) as project:
|
|
471
|
+
total_imported = 0
|
|
472
|
+
batch_count = 0
|
|
473
|
+
uniprot_count = 0
|
|
474
|
+
generic_count = 0
|
|
475
|
+
|
|
476
|
+
async for batch_df in parser.parse_batch(batch_size=batch_size):
|
|
477
|
+
if "is_uniprot" in batch_df.columns:
|
|
478
|
+
uniprot_count += batch_df["is_uniprot"].sum()
|
|
479
|
+
uniprot_in_batch = batch_df.get("is_uniprot", pd.Series([False] * len(batch_df))).sum()
|
|
480
|
+
generic_count += len(batch_df) - uniprot_in_batch
|
|
481
|
+
await project.add_proteins_batch(batch_df)
|
|
482
|
+
total_imported += len(batch_df)
|
|
483
|
+
batch_count += 1
|
|
484
|
+
typer.echo(f" Batch {batch_count}: {total_imported} proteins...")
|
|
485
|
+
|
|
486
|
+
await project.save()
|
|
487
|
+
typer.echo(f"✓ Imported {total_imported} proteins from {file_path.name}")
|
|
488
|
+
typer.echo(f" UniProt entries: {int(uniprot_count)}")
|
|
489
|
+
typer.echo(f" Generic entries: {int(generic_count)}")
|
|
490
|
+
|
|
491
|
+
try:
|
|
492
|
+
import pandas as pd
|
|
493
|
+
import asyncio
|
|
494
|
+
asyncio.run(_import())
|
|
495
|
+
except typer.Exit:
|
|
496
|
+
raise
|
|
497
|
+
except Exception as e:
|
|
498
|
+
typer.echo(f"Error importing FASTA: {e}", err=True)
|
|
499
|
+
raise typer.Exit(1)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""CLI command for importing/merging another project."""
|
|
2
|
+
|
|
3
|
+
import typer
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Annotated
|
|
6
|
+
import asyncio
|
|
7
|
+
from dasmixer.api.project.project import Project
|
|
8
|
+
|
|
9
|
+
app = typer.Typer(help="Merge another project into this one")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@app.command()
|
|
13
|
+
def import_project(
|
|
14
|
+
project_path: Annotated[str, typer.Argument(help="Target project (.dasmix)")],
|
|
15
|
+
source_path: Annotated[str, typer.Argument(help="Source project to import from (.dasmix)")],
|
|
16
|
+
tool_match: Annotated[str, typer.Option(help="Tool merge strategy: 'parser'|'name'|'none'")] = "parser",
|
|
17
|
+
no_subset_match: Annotated[bool, typer.Option("--no-subset-match", help="Do not merge subsets by name")] = False,
|
|
18
|
+
no_sample_match: Annotated[bool, typer.Option("--no-sample-match", help="Do not merge samples by name")] = False,
|
|
19
|
+
update_settings: Annotated[bool, typer.Option("--update-settings", help="Replace target settings with source")] = False,
|
|
20
|
+
conflict_suffix: Annotated[str, typer.Option(help="Suffix for conflicting names")] = "_1",
|
|
21
|
+
):
|
|
22
|
+
"""Merge another project into target project."""
|
|
23
|
+
tgt = Path(project_path)
|
|
24
|
+
src = Path(source_path)
|
|
25
|
+
|
|
26
|
+
if not tgt.exists():
|
|
27
|
+
typer.echo(f"Error: target project not found: {tgt}", err=True)
|
|
28
|
+
raise typer.Exit(1)
|
|
29
|
+
|
|
30
|
+
if not src.exists():
|
|
31
|
+
typer.echo(f"Error: source project not found: {src}", err=True)
|
|
32
|
+
raise typer.Exit(1)
|
|
33
|
+
|
|
34
|
+
# Convert tool_match
|
|
35
|
+
if tool_match == "none":
|
|
36
|
+
tool_match_value = None
|
|
37
|
+
elif tool_match == "name":
|
|
38
|
+
tool_match_value = "name"
|
|
39
|
+
else:
|
|
40
|
+
tool_match_value = "parser"
|
|
41
|
+
|
|
42
|
+
# Show summary
|
|
43
|
+
typer.echo(f"Target: {tgt}")
|
|
44
|
+
typer.echo(f"Source: {src}")
|
|
45
|
+
typer.echo(f"Tool match: {tool_match_value}")
|
|
46
|
+
typer.echo(f"Merge subsets: {not no_subset_match}")
|
|
47
|
+
typer.echo(f"Merge samples: {not no_sample_match}")
|
|
48
|
+
typer.echo(f"Update settings: {update_settings}")
|
|
49
|
+
typer.echo(f"Conflict suffix: {conflict_suffix}")
|
|
50
|
+
|
|
51
|
+
if not typer.confirm("Proceed with merge?"):
|
|
52
|
+
typer.echo("Cancelled")
|
|
53
|
+
raise typer.Exit(0)
|
|
54
|
+
|
|
55
|
+
async def _run():
|
|
56
|
+
async with Project(path=tgt, create_if_not_exists=False) as project:
|
|
57
|
+
def status_callback(table: str, fraction: float):
|
|
58
|
+
typer.echo(f" [{fraction*100:3.0f}%] Importing {table}...")
|
|
59
|
+
|
|
60
|
+
await project.import_project(
|
|
61
|
+
source_path=src,
|
|
62
|
+
tool_match=tool_match_value,
|
|
63
|
+
subset_match=not no_subset_match,
|
|
64
|
+
sample_match=not no_sample_match,
|
|
65
|
+
project_settings_match=update_settings,
|
|
66
|
+
conflict_suffix=conflict_suffix,
|
|
67
|
+
status_callback=status_callback,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
try:
|
|
71
|
+
asyncio.run(_run())
|
|
72
|
+
typer.echo("✓ Import complete")
|
|
73
|
+
except Exception as e:
|
|
74
|
+
typer.echo(f"Error during import: {e}", err=True)
|
|
75
|
+
raise typer.Exit(1)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""CLI commands for portable project utilities (checkpoint, vacuum)."""
|
|
2
|
+
|
|
3
|
+
import typer
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Annotated
|
|
6
|
+
import asyncio
|
|
7
|
+
from dasmixer.api.project.project import Project
|
|
8
|
+
|
|
9
|
+
app = typer.Typer(help="Portable project utilities")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@app.command()
|
|
13
|
+
def checkpoint(
|
|
14
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
15
|
+
):
|
|
16
|
+
"""Save uncommitted WAL changes into the main database file."""
|
|
17
|
+
path = Path(project_path)
|
|
18
|
+
if not path.exists():
|
|
19
|
+
typer.echo(f"Error: file not found: {path}", err=True)
|
|
20
|
+
raise typer.Exit(1)
|
|
21
|
+
|
|
22
|
+
async def _run():
|
|
23
|
+
async with Project(path=path, create_if_not_exists=False) as project:
|
|
24
|
+
await project.save(checkpoint=True)
|
|
25
|
+
|
|
26
|
+
try:
|
|
27
|
+
asyncio.run(_run())
|
|
28
|
+
typer.echo(f"✓ Checkpoint done: {path}")
|
|
29
|
+
except Exception as e:
|
|
30
|
+
typer.echo(f"Error during checkpoint: {e}", err=True)
|
|
31
|
+
raise typer.Exit(1)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@app.command()
|
|
35
|
+
def vacuum(
|
|
36
|
+
project_path: Annotated[str, typer.Argument(help="Path to .dasmix project file")],
|
|
37
|
+
):
|
|
38
|
+
"""Compact the database file by running SQLite VACUUM."""
|
|
39
|
+
path = Path(project_path)
|
|
40
|
+
if not path.exists():
|
|
41
|
+
typer.echo(f"Error: file not found: {path}", err=True)
|
|
42
|
+
raise typer.Exit(1)
|
|
43
|
+
|
|
44
|
+
async def _run():
|
|
45
|
+
async with Project(path=path, create_if_not_exists=False) as project:
|
|
46
|
+
await project.save(checkpoint=True)
|
|
47
|
+
await project.vacuum()
|
|
48
|
+
|
|
49
|
+
try:
|
|
50
|
+
asyncio.run(_run())
|
|
51
|
+
typer.echo(f"✓ Vacuum complete: {path}")
|
|
52
|
+
except Exception as e:
|
|
53
|
+
typer.echo(f"Error during vacuum: {e}", err=True)
|
|
54
|
+
raise typer.Exit(1)
|