tabtk 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tabtk/tabtk.py ADDED
@@ -0,0 +1,1556 @@
1
+ #!/usr/bin/env python
2
+
3
+ import os
4
+ import sys
5
+ from Bio import SeqIO
6
+ import multiprocessing
7
+ import argparse
8
+ import itertools
9
+ from progress.bar import ChargingBar
10
+ import re
11
+ import subprocess
12
+ from pathlib import Path
13
+ import numpy as np
14
+ import shutil
15
+ import pandas as pd
16
+ import networkx as nx
17
+ import math
18
+ import datetime
19
+ from .__init__ import __version__
20
+
21
+ def pairwiseSimilarity(firstRecord,secondRecord):
22
+ recordMatch = 0
23
+ recordMisMatch = 0
24
+
25
+ firstRecordName = list(firstRecord.keys())[0]
26
+ secondRecordName = list(secondRecord.keys())[0]
27
+
28
+ for r,s in zip(firstRecord[firstRecordName],secondRecord[secondRecordName]):
29
+ if not (r == 0 and s == 0):
30
+ if r == s:
31
+ if (r in [0,1] and s in [0,1]):
32
+ recordMatch = recordMatch + 1
33
+ else:
34
+ print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
35
+ sys.exit()
36
+ else:
37
+ if (r in [0,1] and s in [0,1]):
38
+ recordMisMatch = recordMisMatch + 1
39
+ else:
40
+ print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
41
+ sys.exit()
42
+ else:
43
+ if (r in [0,1] and s in [0,1]):
44
+ pass
45
+ else:
46
+ print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
47
+ sys.exit()
48
+
49
+ percentSimilarity = recordMatch/(recordMatch + recordMisMatch)*100
50
+
51
+ similarityData = [firstRecordName,secondRecordName,percentSimilarity]
52
+
53
+ #print('.', end='', flush=True)
54
+
55
+ return(similarityData)
56
+
57
+
58
+ def alignmentSimilarity(ij):
59
+ seqMatch=0
60
+ seqMisMatch=0
61
+
62
+ for r,s in zip(str(i.seq).replace("-","N").replace("X","N").upper(),str(j.seq).replace("-","N").replace("X","N").upper()):
63
+ if r==s and not (r=="N" or s=="N"):
64
+ seqMatch=seqMatch+1
65
+ else:
66
+ if r=="N" or s=="N":
67
+ continue
68
+ else:
69
+ seqMisMatch=seqMisMatch+1
70
+
71
+ seqPID=round(seqMatch/(seqMatch+seqMisMatch)*100,4) if (seqMatch+seqMisMatch)!=0 else 0
72
+
73
+ seqLen=round(len(i.seq),4)
74
+
75
+ seqCov=round((seqMatch+seqMisMatch)/seqLen*100,4) if seqLen!=0 else 0
76
+
77
+ retValue = f"{i.id}\t{j.id}\t{seqPID}"
78
+
79
+ return(retValue)
80
+
81
+
82
+ def mashSimilarity(cmdValues):
83
+ if not shutil.which("mash"):
84
+ print("Install MASH and try again... exiting...")
85
+ sys.exit()
86
+
87
+ else:
88
+ pass
89
+
90
+ tmpOut = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
91
+ tmpOutR = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
92
+
93
+ print("Calculating percent similarity... may take longer... be patient...")
94
+
95
+ mashSketchCmd = f"mash sketch -p {cmdValues["threads"]} -k {cmdValues["length"]} -o {tmpOut} -l {cmdValues["data"]} -s {cmdValues["size"]}"
96
+ mashDistCmd = f"mash dist -p {cmdValues["threads"]} -s {cmdValues["size"]} {f"{tmpOut}.msh"} {f"{tmpOut}.msh"} > {tmpOutR}"
97
+
98
+ subprocess.call(mashSketchCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
99
+ subprocess.call(mashDistCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
100
+
101
+ mashResults = []
102
+
103
+ with open(tmpOutR,"r") as mashFile:
104
+ for mashOutput in mashFile:
105
+ item = mashOutput.strip().replace("/","\t").split("\t")
106
+ mashResults.append(f"{Path(item[0]).stem}\t{Path(item[1]).stem}\t{float(item[4])/float(item[5])*100}")
107
+
108
+ if os.path.exists(tmpOut):
109
+ os.remove(tmpOut)
110
+ else:
111
+ pass
112
+
113
+ if os.path.exists(tmpOutR):
114
+ os.remove(tmpOutR)
115
+ else:
116
+ pass
117
+
118
+ if os.path.exists(f"{tmpOut}.msh"):
119
+ os.remove(f"{tmpOut}.msh")
120
+ else:
121
+ pass
122
+
123
+ return(mashResults)
124
+
125
+
126
+ def fastaniSimilarity(cmdValues):
127
+ if not shutil.which("fastANI"):
128
+ print("Install fastANI and try again... exiting...")
129
+ sys.exit()
130
+
131
+ else:
132
+ pass
133
+
134
+ tmpOut = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
135
+
136
+ fastaniCmd = f"fastANI --rl {cmdValues["data"]} --ql {cmdValues["data"]} -k {cmdValues["length"]} --matrix --fragLen {cmdValues["fragment"]} --minFraction {cmdValues["fraction"]} --output {tmpOut} -t {cmdValues["threads"]}"
137
+
138
+ print("Calculating percent similarity... may take longer... be patient...")
139
+
140
+ subprocess.call(fastaniCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
141
+
142
+ fastaniResults = []
143
+
144
+ with open(tmpOut,"r") as fastaniFile:
145
+ for itemPos,fastaniOutput in enumerate(fastaniFile):
146
+ if itemPos == 0:
147
+ pass
148
+ else:
149
+ item = fastaniOutput.strip().split("\t")
150
+ fastaniResults.append(f"{Path(item[0]).stem}\t{Path(item[1]).stem}\t{float(item[2])}")
151
+
152
+ if os.path.exists(tmpOut):
153
+ os.remove(tmpOut)
154
+ else:
155
+ pass
156
+
157
+ if os.path.exists(f"{tmpOut}.matrix"):
158
+ os.remove(f"{tmpOut}.matrix")
159
+ else:
160
+ pass
161
+
162
+ return(fastaniResults)
163
+
164
+
165
+ def similarity(cmdValues):
166
+
167
+ if sum([1 if cmdValues["aln"] else 0, 1 if cmdValues["mash"] else 0, 1 if cmdValues["fastani"] else 0]) > 1:
168
+ print("Specify none or only one of these options --aln/-a, --mash/-m, and --fastani/-n... exiting...")
169
+ sys.exit()
170
+ else:
171
+ pass
172
+
173
+ if (not cmdValues["aln"]) and (not cmdValues["mash"]) and (not cmdValues["fastani"]):
174
+ dataMatrixDF = loadDataTable(cmdValues)
175
+
176
+ itemPairs = itertools.permutations([item for pos,item in enumerate(dataMatrixDF.keys()) if pos>0], r=2)
177
+ tmpPairs = itertools.permutations([item for pos,item in enumerate(dataMatrixDF.keys()) if pos>0], r=2)
178
+
179
+ print("Calculating pairwise similarity... may take longer... be patient...")
180
+
181
+ with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
182
+ args = [({firstItem: dataMatrixDF[firstItem]}, {secondItem: dataMatrixDF[secondItem]}) for firstItem,secondItem in itemPairs]
183
+ similarityResults = pool.starmap(pairwiseSimilarity, args)
184
+
185
+ with open(cmdValues["output"],"w") as simData:
186
+ simData.write("seq1\tseq2\tpid\n")
187
+
188
+ bar = ChargingBar('Writing similarity values', max=len(similarityResults))
189
+ for itemPair in similarityResults:
190
+ simData.write(f"{itemPair[0]}\t{itemPair[1]}\t{itemPair[2]}\n")
191
+
192
+ bar.next()
193
+ bar.finish()
194
+
195
+ elif (not cmdValues["aln"]) and (cmdValues["mash"]) and (not cmdValues["fastani"]):
196
+ similarityResults = mashSimilarity(cmdValues)
197
+
198
+ with open(cmdValues["output"],"w") as simData:
199
+ simData.write("seq1\tseq2\tpid\n")
200
+
201
+ bar = ChargingBar('Writing similarity values', max=len(similarityResults))
202
+ for itemPair in similarityResults:
203
+ simData.write(f"{itemPair}\n")
204
+
205
+ bar.next()
206
+ bar.finish()
207
+
208
+ elif (not cmdValues["aln"]) and (not cmdValues["mash"]) and (cmdValues["fastani"]):
209
+ similarityResults = fastaniSimilarity(cmdValues)
210
+
211
+ with open(cmdValues["output"],"w") as simData:
212
+ simData.write("seq1\tseq2\tpid\n")
213
+
214
+ bar = ChargingBar('Writing similarity values', max=len(similarityResults))
215
+ for itemPair in similarityResults:
216
+ simData.write(f"{itemPair}\n")
217
+
218
+ bar.next()
219
+ bar.finish()
220
+
221
+ elif (cmdValues["aln"]) and (not cmdValues["mash"]) and (not cmdValues["fastani"]):
222
+ alignment={}
223
+
224
+ for i in SeqIO.parse(cmdValues['data'],"fasta"):
225
+ alignment[i.id]=i
226
+
227
+ fhandle=open(str(cmdValues['outputFile']),"w")
228
+
229
+ fhandle.write("seq1\tseq2\tpid\n")
230
+
231
+ tmpList = sorted(map(sorted, combinations(set([i for i in alignment.keys()]), 2)))
232
+
233
+ with multiprocessing.Pool(processes=cmdValues['threads']) as pool:
234
+ args=[(alignment[m],alignment[n]) for m,n in tmpList]
235
+ results=pool.starmap(alignmentSimilarity, args)
236
+
237
+ else:
238
+ pass
239
+
240
+
241
+ def transpose(cmdValues):
242
+ if not shutil.which("datamash"):
243
+ print("Install 'datamash' software and try again... exiting...")
244
+ sys.exit()
245
+
246
+ else:
247
+ pass
248
+
249
+ bar = ChargingBar("Transposing table", max=1)
250
+ datamashCmd = f"datamash transpose < {cmdValues["data"]} > {f"{Path(cmdValues["output"]).stem}.transposed.tsv"}"
251
+
252
+ subprocess.call(datamashCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
253
+
254
+ bar.next()
255
+ bar.finish()
256
+
257
+
258
+ def lineages1Levels(cmdValues):
259
+ np.random.rand(1)
260
+
261
+ snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
262
+
263
+ Graph = nx.Graph()
264
+ Graph.add_nodes_from(snpDistDF.seq2.to_list())
265
+ Graph.add_nodes_from(snpDistDF.seq1.to_list())
266
+
267
+ snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
268
+ seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
269
+
270
+ Graph.add_edges_from(seqPair)
271
+ networkClusters = tuple(nx.connected_components(Graph))
272
+
273
+ fhandle = open(f"{cmdValues['output']}","w")
274
+
275
+ for i,j in enumerate(networkClusters):
276
+ for s in j:
277
+ FirstLevelLin = str(i+1)
278
+ seqLabel = s
279
+ clusterPrefix = cmdValues["prefix"]
280
+ separator = cmdValues["separator"]
281
+
282
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
283
+
284
+ fhandle.close();
285
+
286
+
287
+ def lineages2Levels(cmdValues):
288
+ np.random.rand(1)
289
+
290
+ snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
291
+
292
+ Graph = nx.Graph()
293
+ Graph.add_nodes_from(snpDistDF.seq2.to_list())
294
+ Graph.add_nodes_from(snpDistDF.seq1.to_list())
295
+
296
+ snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
297
+ seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
298
+
299
+ Graph.add_edges_from(seqPair)
300
+ networkClusters = tuple(nx.connected_components(Graph))
301
+
302
+ fhandle = open(f"{cmdValues['output']}","w")
303
+
304
+ for i,j in enumerate(networkClusters):
305
+ snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
306
+ snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
307
+
308
+ Graph1 = nx.Graph()
309
+ Graph1.add_nodes_from(j)
310
+
311
+ seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
312
+
313
+ Graph1.add_edges_from(seqPair1)
314
+ networkClusters1 = tuple(nx.connected_components(Graph1))
315
+
316
+ for r,k in enumerate(networkClusters1):
317
+ for s in k:
318
+ FirstLevelLin = str(i+1)
319
+ seqLabel = s
320
+ SecondLevelLin = str(r+1)
321
+ clusterPrefix = cmdValues["prefix"]
322
+ separator = cmdValues["separator"]
323
+
324
+ if len(networkClusters1)==1:
325
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
326
+ else:
327
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
328
+
329
+ fhandle.close();
330
+
331
+
332
+ def lineages3Levels(cmdValues):
333
+ np.random.rand(1)
334
+
335
+ snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
336
+
337
+ Graph = nx.Graph()
338
+ Graph.add_nodes_from(snpDistDF.seq2.to_list())
339
+ Graph.add_nodes_from(snpDistDF.seq1.to_list())
340
+
341
+ snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
342
+ seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
343
+
344
+ Graph.add_edges_from(seqPair)
345
+ networkClusters = tuple(nx.connected_components(Graph))
346
+
347
+ fhandle = open(f"{cmdValues['output']}","w")
348
+
349
+ for i,j in enumerate(networkClusters):
350
+ snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
351
+ snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
352
+
353
+ Graph1 = nx.Graph()
354
+ Graph1.add_nodes_from(j)
355
+
356
+ seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
357
+
358
+ Graph1.add_edges_from(seqPair1)
359
+ networkClusters1 = tuple(nx.connected_components(Graph1))
360
+
361
+ for p,q in enumerate(networkClusters1):
362
+ snpDistDF2R = snpDistDF[snpDistDF.seq1.isin(q) & snpDistDF.seq2.isin(q)]
363
+ snpDistDF3R = snpDistDF2R[snpDistDF2R.pid >= cmdValues["thresholds"][2]]
364
+
365
+ Graph2 = nx.Graph()
366
+ Graph2.add_nodes_from(q)
367
+
368
+ seqPair1R = [(snpDistDF3R.seq1[z],snpDistDF3R.seq2[z]) for z in snpDistDF3R.index]
369
+
370
+ Graph2.add_edges_from(seqPair1R)
371
+ networkClusters2 = tuple(nx.connected_components(Graph2))
372
+
373
+ for r,k in enumerate(networkClusters2):
374
+ for s in k:
375
+ FirstLevelLin = str(i+1)
376
+ SecondLevelLin = str(p+1)
377
+ ThirdLevelLin = str(r+1)
378
+ clusterPrefix = cmdValues["prefix"]
379
+ seqLabel = s
380
+ separator = cmdValues["separator"]
381
+
382
+ if len(networkClusters1)==1:
383
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
384
+ else:
385
+ if len(networkClusters2)==1:
386
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
387
+ else:
388
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}\n")
389
+
390
+ fhandle.close();
391
+
392
+
393
+ def lineages4Levels(cmdValues):
394
+ np.random.rand(1)
395
+
396
+ snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
397
+
398
+ Graph = nx.Graph()
399
+ Graph.add_nodes_from(snpDistDF.seq2.to_list())
400
+ Graph.add_nodes_from(snpDistDF.seq1.to_list())
401
+
402
+ snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
403
+ seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
404
+
405
+ Graph.add_edges_from(seqPair)
406
+ networkClusters = tuple(nx.connected_components(Graph))
407
+
408
+ fhandle = open(f"{cmdValues['output']}","w")
409
+
410
+ for i,j in enumerate(networkClusters):
411
+ snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
412
+ snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
413
+
414
+ Graph1 = nx.Graph()
415
+ Graph1.add_nodes_from(j)
416
+
417
+ seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
418
+
419
+ Graph1.add_edges_from(seqPair1)
420
+ networkClusters1 = tuple(nx.connected_components(Graph1))
421
+
422
+ for p,q in enumerate(networkClusters1):
423
+ snpDistDF2R = snpDistDF[snpDistDF.seq1.isin(q) & snpDistDF.seq2.isin(q)]
424
+ snpDistDF3R = snpDistDF2R[snpDistDF2R.pid >= cmdValues["thresholds"][2]]
425
+
426
+ Graph2 = nx.Graph()
427
+ Graph2.add_nodes_from(q)
428
+
429
+ seqPair1R = [(snpDistDF3R.seq1[z],snpDistDF3R.seq2[z]) for z in snpDistDF3R.index]
430
+
431
+ Graph2.add_edges_from(seqPair1R)
432
+ networkClusters2 = tuple(nx.connected_components(Graph2))
433
+
434
+ for x,y in enumerate(networkClusters2):
435
+
436
+ snpDistDF2RR = snpDistDF[snpDistDF.seq1.isin(y) & snpDistDF.seq2.isin(y)]
437
+ snpDistDF3RR = snpDistDF2RR[snpDistDF2RR.pid >= cmdValues["thresholds"][3]]
438
+
439
+ Graph3 = nx.Graph()
440
+ Graph3.add_nodes_from(y)
441
+
442
+ seqPair1RR = [(snpDistDF3RR.seq1[z],snpDistDF3RR.seq2[z]) for z in snpDistDF3RR.index]
443
+
444
+ Graph3.add_edges_from(seqPair1RR)
445
+ networkClusters3 = tuple(nx.connected_components(Graph3))
446
+
447
+ for r,k in enumerate(networkClusters3):
448
+
449
+ for s in k:
450
+ FirstLevelLin = str(i+1)
451
+ SecondLevelLin = str(p+1)
452
+ ThirdLevelLin = str(x+1)
453
+ FourthLevelLin = str(r+1)
454
+ clusterPrefix = cmdValues["prefix"]
455
+ seqLabel = s
456
+ separator = cmdValues["separator"]
457
+
458
+ if len(networkClusters1)==1:
459
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
460
+ else:
461
+ if len(networkClusters2)==1:
462
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
463
+ else:
464
+ if len(networkClusters3)==1:
465
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}\n")
466
+ else:
467
+ fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}.{FourthLevelLin}\n")
468
+
469
+ fhandle.close();
470
+
471
+
472
+ def cluster(cmdValues):
473
+ for item in cmdValues['thresholds']:
474
+ if float(item)<0 or float(item)>100:
475
+ print("Clustering thresholds should be between 0 and 100")
476
+ sys.exit()
477
+
478
+ else:
479
+ pass
480
+
481
+ for itemPos,item in enumerate(cmdValues['thresholds']):
482
+ if itemPos == 0:
483
+ pass
484
+ else:
485
+ if cmdValues['thresholds'][itemPos-1]<cmdValues['thresholds'][itemPos]:
486
+ pass
487
+ else:
488
+ print("Clustering thresholds should be provided in increasing order... exiting...")
489
+ sys.exit()
490
+
491
+ if len(cmdValues['thresholds'])==1:
492
+ lineages1Levels(cmdValues)
493
+ elif len(cmdValues['thresholds'])==2:
494
+ lineages2Levels(cmdValues)
495
+ elif len(cmdValues['thresholds'])==3:
496
+ lineages3Levels(cmdValues)
497
+ elif len(cmdValues['thresholds'])==4:
498
+ lineages4Levels(cmdValues)
499
+ else:
500
+ print("Maximum number of nested should be 4... exiting...")
501
+ sys.exit()
502
+
503
+
504
+ def order(cmdValues):
505
+ with open(cmdValues["output"],"w") as outputFile:
506
+ dataMatrixDF = loadDataTable(cmdValues)
507
+
508
+ itemNames = [re.split("\t|,|;",str(item).strip())[0] for item in open(cmdValues["rownames"],"r")]
509
+
510
+ bar = ChargingBar('Saving ordered table', max=len(itemNames))
511
+ for itemNum,itemRowName in enumerate(itemNames):
512
+ if itemNum == 0:
513
+ headerName = list(dataMatrixDF.keys())[0]
514
+ outputFile.write(f"{headerName}\t{"\t".join([str(i) for i in dataMatrixDF[headerName]])}\n")
515
+ else:
516
+ pass
517
+
518
+ if itemRowName in list(dataMatrixDF.keys()):
519
+ outputFile.write(f"{itemRowName}\t{"\t".join([str(i) for i in dataMatrixDF[itemRowName]])}\n")
520
+
521
+ else:
522
+ outputFile.write(f"{itemRowName}\n")
523
+
524
+ bar.next()
525
+ bar.finish()
526
+
527
+ def sample(cmdValues):
528
+ np.random.seed(1)
529
+
530
+ dataMatrixDF = loadDataTable(cmdValues)
531
+
532
+ rows = list(dataMatrixDF.keys())
533
+ columns = np.arange(len(dataMatrixDF[rows[0]])).tolist()
534
+
535
+ selectedColumns = np.random.choice(columns,int(np.ceil(float(cmdValues["colfreq"])*len(columns))),replace=False).tolist()
536
+ selectedRows = [rows[0]]
537
+ selectedRows.extend(np.random.choice(rows[1:],int(np.ceil(float(cmdValues["rowfreq"])*len(rows[1:]))),replace=False).tolist())
538
+
539
+ with open(cmdValues["output"],"w") as outputFile:
540
+
541
+ bar = ChargingBar('Saving table', max=len(selectedRows)*len(selectedColumns))
542
+ for i,j in enumerate(selectedRows):
543
+ tmpStr = []
544
+
545
+ for r,s in enumerate(selectedColumns):
546
+ if r == 0:
547
+ tmpStr.append(str(j))
548
+ tmpStr.append(str(dataMatrixDF[j][r]))
549
+ else:
550
+ tmpStr.append(str(dataMatrixDF[j][r]))
551
+
552
+ bar.next()
553
+ outputFile.write(f"{"\t".join(tmpStr)}\n")
554
+
555
+ bar.finish()
556
+
557
+
558
+ def rename(cmdValues):
559
+ with open(cmdValues["output"],"w") as outputFile:
560
+ dataMatrixDF = loadDataTable(cmdValues)
561
+
562
+ columnIndexValues = []
563
+
564
+ if not os.path.exists(cmdValues["newdata"]):
565
+ print(f"File {cmdValues["newdata"]} does not exist... exiting...")
566
+ sys.exit()
567
+ else:
568
+ pass
569
+
570
+ if not isinstance(cmdValues["columns"],str):
571
+ columnIndexValues = [int(i) for i in [re.split("\t|,|;",str(item).strip()) for item in cmdValues["columns"]][0]]
572
+
573
+ else:
574
+ columnIndexValues = [int(i) for i in cmdValues["columns"]]
575
+
576
+ newRowNameValuesDF = {}
577
+
578
+ tmpCmd="wc -l "+str(cmdValues["newdata"])+" | awk '{print $1}'"
579
+ dataFileLines=subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
580
+ dataLines = int(dataFileLines.communicate()[0])
581
+ dataFileLines.kill()
582
+
583
+ bar = ChargingBar('Loading file with new names', max=int(dataLines))
584
+ with open(cmdValues["newdata"],"r") as newData:
585
+ for rowData in newData:
586
+ newRowNameValues = []
587
+
588
+ if len(rowData.strip().split("\t")) > 0:
589
+ if max(columnIndexValues) < 0 or max(columnIndexValues) > len(rowData.strip().split("\t")):
590
+ print(f"Values for --column/-c should be >0 and <number of columns in the new data table... exiting...")
591
+ sys.exit()
592
+ else:
593
+ pass
594
+ else:
595
+ pass
596
+
597
+ for columnIndex in columnIndexValues:
598
+ newRowNameValues.append(rowData.strip().split("\t")[columnIndex-1])
599
+
600
+ newRowNameValuesDF[rowData.strip().split("\t")[0]] = "_".join(newRowNameValues)
601
+
602
+ bar.next()
603
+ bar.finish()
604
+
605
+ bar = ChargingBar('Saving renamed table', max=len(dataMatrixDF.keys()))
606
+ for rowIndex,rowName in enumerate(dataMatrixDF.keys()):
607
+
608
+ if rowIndex == 0:
609
+ outputFile.write(f"{rowName}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
610
+ else:
611
+ if rowName in newRowNameValuesDF.keys():
612
+ outputFile.write(f"{newRowNameValuesDF[rowName]}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
613
+
614
+ else:
615
+ outputFile.write(f"{rowName}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
616
+
617
+ bar.next()
618
+ bar.finish()
619
+
620
+
621
+ def filter(cmdValues):
622
+ if cmdValues["minCutoff"]>=0 and cmdValues["maxCutoff"]<=100:
623
+ if cmdValues["minCutoff"] > cmdValues["maxCutoff"]:
624
+ print(f"--->>>Value of --min/-a should be less than or equal to value of --max/-b... exiting...")
625
+ sys.exit()
626
+ else:
627
+ pass
628
+
629
+ dataMatrixDF = loadDataTable(cmdValues)
630
+
631
+ tmpMatrixData = {}
632
+
633
+ with open(cmdValues["output"],"w") as outputFile:
634
+ columnValues = []
635
+
636
+ bar = ChargingBar('Filtering table', max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
637
+ for itemPos,itemValue in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
638
+ itemMatrixColValues = []
639
+
640
+ for tmpPos,itemMatrixName in enumerate(dataMatrixDF.keys()):
641
+ if tmpPos != 0:
642
+ itemMatrixColValues.append(dataMatrixDF[itemMatrixName][itemPos])
643
+ else:
644
+ pass
645
+
646
+ rowMatrixSum = sum(itemMatrixColValues)
647
+
648
+ if rowMatrixSum >= (cmdValues["minCutoff"]/100)*len(itemMatrixColValues) and \
649
+ rowMatrixSum <= (cmdValues["maxCutoff"]/100)*len(itemMatrixColValues):
650
+
651
+ columnValues.append(itemPos)
652
+ else:
653
+ pass
654
+
655
+ bar.next()
656
+ bar.finish()
657
+
658
+ bar = ChargingBar('Saving filtered table', max=len(list(dataMatrixDF.keys())))
659
+ for itemMatrixName in dataMatrixDF.keys():
660
+ tmpColData = []
661
+
662
+ for colPosition in columnValues:
663
+ tmpColData.append(str(dataMatrixDF[itemMatrixName][colPosition]))
664
+
665
+ bar.next()
666
+ outputFile.write(f"{str(itemMatrixName)}\t{"\t".join(tmpColData)}\n")
667
+ bar.finish()
668
+
669
+ else:
670
+ if cmdValues["minCutoff"] > cmdValues["maxCutoff"]:
671
+ print(f"--->>>Value of --min/-a should be less than or equal to value of --max/-b... exiting...")
672
+ sys.exit()
673
+ else:
674
+ print(f"--->>>Value of --min/-a snd --max/-b values should be between 0 and 100... exiting...")
675
+ sys.exit()
676
+
677
+
678
+ def makeMapPED(i,j,k,l):
679
+ if i == 0:
680
+ pass
681
+ else:
682
+ tmpStr = ""
683
+
684
+ for m,n in enumerate(k):
685
+ if m == 0:
686
+ tmpStr = f"{str(j)}\t{str(j)}\t0\t0\t1\t{str(l)}"
687
+
688
+ else:
689
+ tmpN = 0
690
+ if(int(n) == 0):
691
+ tmpN = 2
692
+ else:
693
+ tmpN = 1
694
+
695
+ tmpStr = f"{tmpStr}\t{str(tmpN)} {str(tmpN)}"
696
+
697
+ print('.', end='', flush=True)
698
+
699
+ return(tmpStr)
700
+
701
+
702
+ def gwas(cmdValues):
703
+ checkFlags = [1 for flag in [cmdValues["pyseer"],cmdValues["plink"],cmdValues["gemma"],cmdValues["fastlmm"]] if flag == True]
704
+
705
+ if sum(checkFlags) > 1:
706
+ print(f"Only one these flags should be specified: --pyseer/-s, --plink/-l, --gemma/g, and --fastlmm/-m... exiting...")
707
+ sys.exit()
708
+ else:
709
+ if sum(checkFlags) == 0:
710
+ cmdValues["plink"] = True
711
+
712
+ else:
713
+ pass
714
+
715
+ dataMatrixDF = loadDataTable(cmdValues)
716
+
717
+ phenoData={}
718
+
719
+ if os.path.exists(cmdValues["phenotype"]):
720
+ phenoDataTmp = [str(item).strip() for item in open(cmdValues["phenotype"],"r")]
721
+
722
+ phenoData = {}
723
+
724
+ bar = ChargingBar('Processing phenotype data', max=len(phenoDataTmp))
725
+ for m in phenoDataTmp:
726
+ tmpR = re.split("\t|,|;",str(m).strip())
727
+ phenoData[tmpR[0]]=tmpR[1]
728
+
729
+ bar.next()
730
+ bar.finish()
731
+
732
+ else:
733
+ print(f"Phenotype file {cmdValues["phenotype"]} not found... exiting...")
734
+ sys.exit()
735
+
736
+ if cmdValues["plink"]:
737
+ with open(f"{Path(cmdValues["output"]).stem}.plink.map","w") as mapFile:
738
+ with open(f"{Path(cmdValues["output"]).stem}.plink.ped","w") as pedFile:
739
+
740
+ tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
741
+
742
+ with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
743
+ args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
744
+ resultsMapPED = pool.starmap(makeMapPED, args)
745
+
746
+ bar = ChargingBar("Saving PLINK PED file", max=len(list(dataMatrixDF.keys())))
747
+ for eachResult in resultsMapPED:
748
+ pedFile.write(f"{eachResult}\n")
749
+
750
+ bar.next()
751
+ bar.finish()
752
+
753
+ bar = ChargingBar("Saving PLINK MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
754
+ for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
755
+ if k != 0:
756
+ mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
757
+
758
+ else:
759
+ pass
760
+
761
+ bar.next()
762
+ bar.finish()
763
+
764
+ elif cmdValues["fastlmm"]:
765
+ with open(f"{Path(cmdValues["output"]).stem}.fastlmm.map","w") as mapFile:
766
+ with open(f"{Path(cmdValues["output"]).stem}.fastlmm.ped","w") as pedFile:
767
+
768
+ tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
769
+
770
+ with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
771
+ args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
772
+ resultsMapPED = pool.starmap(makeMapPED, args)
773
+
774
+ bar = ChargingBar("Saving FASTLMM PED file", max=len(list(dataMatrixDF.keys())))
775
+ for eachResult in resultsMapPED:
776
+ pedFile.write(f"{eachResult}\n")
777
+
778
+ bar.next()
779
+ bar.finish()
780
+
781
+
782
+ bar = ChargingBar("Saving FASTLMM MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
783
+ for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
784
+ if k != 0:
785
+ mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
786
+
787
+ else:
788
+ pass
789
+
790
+ bar.next()
791
+ bar.finish()
792
+
793
+ elif cmdValues["gemma"]:
794
+ with open(f"{Path(cmdValues["output"]).stem}.plink.map","w") as mapFile:
795
+ with open(f"{Path(cmdValues["output"]).stem}.plink.ped","w") as pedFile:
796
+
797
+ tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
798
+
799
+ with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
800
+ args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
801
+ resultsMapPED = pool.starmap(makeMapPED, args)
802
+
803
+ bar = ChargingBar("Saving PLINK PED file", max=len(list(dataMatrixDF.keys())))
804
+ for eachResult in resultsMapPED:
805
+ pedFile.write(f"{eachResult}\n")
806
+
807
+ bar.next()
808
+ bar.finish()
809
+
810
+ bar = ChargingBar("Saving PLINK MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
811
+ for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
812
+ if k != 0:
813
+ mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
814
+
815
+ else:
816
+ pass
817
+
818
+ bar.next()
819
+ bar.finish()
820
+
821
+ if not shutil.which("plink"):
822
+ print("Install 'plink' software dependency and try again... exiting...")
823
+ sys.exit()
824
+ else:
825
+ pass
826
+
827
+ bar = ChargingBar("Generating GEMMA binary PED files", max=1)
828
+ plinkCmd = f"plink -file {f"{Path(cmdValues["output"]).stem}.plink"} -make-bed -out {f"{Path(cmdValues["output"]).stem}.gemma"} --threads {cmdValues["threads"]}"
829
+
830
+ subprocess.call(plinkCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
831
+ bar.next()
832
+ bar.finish()
833
+
834
+ os.remove(f"{Path(cmdValues["output"]).stem}.plink.map")
835
+ os.remove(f"{Path(cmdValues["output"]).stem}.plink.ped")
836
+ os.remove(f"{Path(cmdValues["output"]).stem}.gemma.log")
837
+
838
+ else:
839
+ with open(f"{Path(cmdValues["output"]).stem}.pyseer.tsv","w") as pyseerTab:
840
+ with open(f"{Path(cmdValues["output"]).stem}.pyseer.pheno.tsv","w") as pyseerPheno:
841
+
842
+ transpose(cmdValues)
843
+
844
+ transposedFileName = f"{Path(cmdValues["output"]).stem}.transposed.tsv"
845
+ pyseerFileName = f"{Path(cmdValues["output"]).stem}.pyseer.tsv"
846
+
847
+ shutil.move(transposedFileName,pyseerFileName)
848
+
849
+ bar = ChargingBar("Generating PYSEER files", max=len(phenoData.keys()))
850
+ for itemPos,item in enumerate(phenoData.keys()):
851
+ if itemPos == 0:
852
+ pyseerPheno.write("samples\tphenotype\n")
853
+ else:
854
+ pyseerPheno.write(f"{item}\t{phenoData[item]}\n")
855
+
856
+ bar.next()
857
+ bar.finish()
858
+
859
+
860
+ def queryKmers(seqNum,seqFile,kmerFile):
861
+ seqFileName = Path(os.path.basename(seqFile)).stem
862
+ tmpOutFileA = f"bifrost.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
863
+
864
+ tmpCmd1 = f"Bifrost build -t 3 -k 31 -i -d -r {seqFile} -o {tmpOutFileA}.graph"
865
+ subprocess.call(tmpCmd1,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
866
+
867
+ tmpCmd2 = f"Bifrost query -t 5 -e 1 -q {kmerFile} -g {tmpOutFileA}.graph.gfa.gz -o {tmpOutFileA}"
868
+ subprocess.call(tmpCmd2,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
869
+
870
+ tmpCmd3 = f"echo {seqFileName} > {tmpOutFileA}.out.tsv"
871
+ subprocess.call(tmpCmd3,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
872
+
873
+ tmpCmd4 = f"grep -v query_name {tmpOutFileA}.tsv | sort -k1 -n > {tmpOutFileA}.out.tmp.tsv"
874
+ subprocess.call(tmpCmd4,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
875
+
876
+ tmpCmd5 = "awk '{print $2}' "+f"{tmpOutFileA}.out.tmp.tsv >> {tmpOutFileA}.out.tsv"
877
+ subprocess.call(tmpCmd5,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
878
+
879
+ tmpCmd6 = f"echo KmerID > {tmpOutFileA}.head.tsv"
880
+ subprocess.call(tmpCmd6,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
881
+
882
+ tmpCmd7 = "awk '{print $1}' "+f"{tmpOutFileA}.out.tmp.tsv | sed \"s:query_name:KmerID:g\" >> {tmpOutFileA}.head.tsv";
883
+ subprocess.call(tmpCmd7,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
884
+
885
+ tmpCmd8 = f"datamash transpose < {tmpOutFileA}.head.tsv > {tmpOutFileA}.header.tsv"
886
+ subprocess.call(tmpCmd8,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
887
+
888
+ tmpCmd9 = f"datamash transpose < {tmpOutFileA}.out.tsv > {tmpOutFileA}.tmp.tsv"
889
+ subprocess.call(tmpCmd9,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
890
+
891
+ kmerData = {seqFileName: {"kmerFile": [str(i).strip() for i in open(f"{tmpOutFileA}.tmp.tsv","r")][0],"kmerHeaderFile": [str(j).strip() for j in open(f"{tmpOutFileA}.header.tsv","r")][0] }}
892
+
893
+ if os.path.exists(f"{tmpOutFileA}.graph.bfi"):
894
+ os.remove(f"{tmpOutFileA}.graph.bfi")
895
+ else:
896
+ pass
897
+
898
+ if os.path.exists(f"{tmpOutFileA}.graph.gfa.gz"):
899
+ os.remove(f"{tmpOutFileA}.graph.gfa.gz")
900
+ else:
901
+ pass
902
+
903
+ if os.path.exists(f"{tmpOutFileA}.out.tsv"):
904
+ os.remove(f"{tmpOutFileA}.out.tsv")
905
+ else:
906
+ pass
907
+
908
+ if os.path.exists(f"{tmpOutFileA}.tsv"):
909
+ os.remove(f"{tmpOutFileA}.tsv")
910
+ else:
911
+ pass
912
+
913
+ if os.path.exists(f"{tmpOutFileA}.out.tmp.tsv"):
914
+ os.remove(f"{tmpOutFileA}.out.tmp.tsv")
915
+ else:
916
+ pass
917
+
918
+ if os.path.exists(f"{tmpOutFileA}.out.tsv"):
919
+ os.remove(f"{tmpOutFileA}.out.tsv")
920
+ else:
921
+ pass
922
+
923
+ if os.path.exists(f"{tmpOutFileA}.head.tsv"):
924
+ os.remove(f"{tmpOutFileA}.head.tsv")
925
+ else:
926
+ pass
927
+
928
+ if os.path.exists(f"{tmpOutFileA}.header.tsv"):
929
+ os.remove(f"{tmpOutFileA}.header.tsv")
930
+ else:
931
+ pass
932
+
933
+ if os.path.exists(f"{tmpOutFileA}.tmp.tsv"):
934
+ os.remove(f"{tmpOutFileA}.tmp.tsv")
935
+ else:
936
+ pass
937
+
938
+ return(kmerData)
939
+
940
+
941
+ def table(cmdValues):
942
+ if not os.path.exists(cmdValues["data"]):
943
+ print(f"File {cmdValues["data"]} not found... exiting...")
944
+ sys.exit()
945
+ else:
946
+ pass
947
+
948
+ if not shutil.which("awk"):
949
+ print("Install awk and try again... exiting...")
950
+ sys.exit()
951
+ else:
952
+ pass
953
+
954
+ if not shutil.which("dsk") and not shutil.which("dsk2ascii"):
955
+ print("Install DSK and try again... exiting...")
956
+ sys.exit()
957
+ else:
958
+ pass
959
+
960
+ if not shutil.which("Bifrost"):
961
+ print("Install Bifrost and try again... exiting...")
962
+ sys.exit()
963
+ else:
964
+ pass
965
+
966
+ if not shutil.which("datamash"):
967
+ print("Install datamash and try again... exiting...")
968
+ sys.exit()
969
+ else:
970
+ pass
971
+
972
+ kmersFile = ""
973
+
974
+ if cmdValues["kmers"] != "":
975
+ kmersFile = cmdValues["kmers"]
976
+ else:
977
+ kmersFile = f"{Path(cmdValues["output"]).stem}.fasta"
978
+
979
+ with open(cmdValues["data"],"r") as seqFile:
980
+
981
+ tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
982
+ dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
983
+ dataLines = int(dataFileLines.communicate()[0])
984
+ dataFileLines.kill()
985
+
986
+ bar = ChargingBar("Generating k-mers (DSK)", max=int(dataLines))
987
+ for fileNameValue in seqFile:
988
+ fileName = fileNameValue.strip()
989
+
990
+ if os.path.exists(fileName):
991
+ if os.path.isfile(fileName) and os.path.getsize(fileName) > 0:
992
+ pass
993
+ else:
994
+ print(f"File {fileName} in {cmdValues["data"]} is empty... exiting...")
995
+ sys.exit()
996
+ else:
997
+ print(f"File {fileName} in {cmdValues["data"]} does not exist... exiting...")
998
+ sys.exit()
999
+
1000
+ bar.next()
1001
+ bar.finish()
1002
+
1003
+ if cmdValues["dsk"] and not cmdValues["bifrost"]:
1004
+
1005
+ if cmdValues["kmers"] == "":
1006
+ bar = ChargingBar("Generating k-mers (DSK)", max=2)
1007
+ dskCmdTmpFile = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
1008
+
1009
+ dsk1Cmd = f"dsk -kmer-size {cmdValues["length"]} -abundance-min {cmdValues["minabundance"]} -max-memory {cmdValues["maxmemory"]} -file {cmdValues["data"]} -out {dskCmdTmpFile} -nb-cores {cmdValues["threads"]}"
1010
+ subprocess.call(dsk1Cmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
1011
+ bar.next()
1012
+
1013
+ tmpKmersFileR = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
1014
+
1015
+ dsk2Cmd = f"dsk2ascii -file {dskCmdTmpFile}.h5 -out {tmpKmersFileR} -nb-cores {cmdValues["threads"]}"
1016
+ subprocess.call(dsk2Cmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
1017
+ bar.next()
1018
+
1019
+ awkCmd = "awk '{print \">\"NR\"\\n\"$1}' "+str(tmpKmersFileR)+" > "+str(kmersFile)
1020
+ subprocess.call(awkCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
1021
+ bar.next()
1022
+ bar.finish()
1023
+
1024
+ if os.path.exists(f"{dskCmdTmpFile}.h5"):
1025
+ os.remove(f"{dskCmdTmpFile}.h5")
1026
+ else:
1027
+ pass
1028
+
1029
+ if os.path.exists(f"{dskCmdTmpFile}"):
1030
+ os.remove(f"{dskCmdTmpFile}")
1031
+ else:
1032
+ pass
1033
+
1034
+ if os.path.exists(f"{dskCmdTmpFile}.txt"):
1035
+ os.remove(f"{dskCmdTmpFile}.txt")
1036
+ else:
1037
+ pass
1038
+
1039
+ if os.path.exists(f"{tmpKmersFileR}.txt"):
1040
+ os.remove(f"{tmpKmersFileR}.txt")
1041
+ else:
1042
+ pass
1043
+
1044
+ if os.path.exists(f"{tmpKmersFileR}"):
1045
+ os.remove(f"{tmpKmersFileR}")
1046
+ else:
1047
+ pass
1048
+
1049
+ else:
1050
+ pass
1051
+
1052
+ else:
1053
+ if cmdValues["kmers"] != "":
1054
+ tmpKmersFile = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
1055
+
1056
+ bar = ChargingBar("Generating unitigs (Bifrost)", max=1)
1057
+ bifrostCmd = f"Bifrost build -t {cmdValues["threads"]} -k {cmdValues["length"]} -i -d -r {cmdValues["data"]} -o {tmpKmersFile} -v -a -f -n"
1058
+
1059
+ subprocess.call(bifrostCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
1060
+ bar.next()
1061
+ bar.finish()
1062
+
1063
+ with open(kmersFile,"w") as kmerData:
1064
+ for seqNum,seq in enumerate(SeqIO.parse(f"{tmpKmersFile}.fasta","fasta")):
1065
+ kmerData.write(f">{(seqNum+1)}\n{seq.seq}\n")
1066
+
1067
+ if os.path.exists(f"{tmpKmersFile}.bfi"):
1068
+ os.remove(f"{tmpKmersFile}.bfi")
1069
+ else:
1070
+ pass
1071
+
1072
+ if os.path.exists(f"{tmpKmersFile}.fasta"):
1073
+ os.remove(f"{tmpKmersFile}.fasta")
1074
+ else:
1075
+ pass
1076
+
1077
+ else:
1078
+ pass
1079
+
1080
+ tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
1081
+ dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
1082
+ dataLines = int(dataFileLines.communicate()[0])
1083
+ dataFileLines.kill()
1084
+
1085
+ dataMatrixDF = []
1086
+
1087
+ bar = ChargingBar("Processing sequence files", max=int(dataLines))
1088
+ for seqFile in open(cmdValues["data"],"r"):
1089
+ dataMatrixDF.append(str(seqFile).strip().split("\t")[0])
1090
+
1091
+ bar.next()
1092
+ bar.finish()
1093
+
1094
+ seqFileDF = [(pos,item) for pos,item in enumerate(dataMatrixDF)]
1095
+
1096
+ print("Query k-mers against sequences... may take longer... be patient...")
1097
+
1098
+ with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
1099
+ args = [(seqFileNum,seqFileName,kmersFile) for seqFileNum,seqFileName in seqFileDF]
1100
+ kmerResults = pool.starmap(queryKmers, args)
1101
+
1102
+ with open(f"{Path(cmdValues["output"]).stem}.table.tsv","w") as kmerData:
1103
+ kmerFiles = []
1104
+ kmerHeaderFiles = []
1105
+
1106
+ bar = ChargingBar("Writing k-mers table", max=len(kmerResults))
1107
+ for eachResultPos,eachResult in enumerate(kmerResults):
1108
+ seqName = list(eachResult.keys())[0]
1109
+
1110
+ kmerFiles.append(eachResult[seqName])
1111
+ kmerHeaderFiles.append(eachResult[seqName])
1112
+
1113
+ if eachResultPos == 0:
1114
+ kmerData.write(f"{eachResult[seqName]["kmerHeaderFile"]}\n")
1115
+ kmerData.write(f"{eachResult[seqName]["kmerFile"]}\n")
1116
+ else:
1117
+ kmerData.write(f"{eachResult[seqName]["kmerFile"]}\n")
1118
+
1119
+ bar.next()
1120
+ bar.finish()
1121
+
1122
+
1123
+ def loadDataTable(cmdValues):
1124
+ if not (os.path.exists(cmdValues["data"]) and os.path.isfile(cmdValues["data"])):
1125
+ print(f"--->>>Input file {cmdValues["data"]} does not exist... exiting...")
1126
+ sys.exit()
1127
+
1128
+ else:
1129
+ dataMatrixDF = {}
1130
+
1131
+ sepFormat = "\t" if cmdValues["format"] == "tsv" else ","
1132
+
1133
+ if not (os.path.exists(cmdValues["data"]) and os.path.isfile(cmdValues["data"])):
1134
+ print(f"--->>>Input file {cmdValues["data"]} does not exist... exiting...")
1135
+ sys.exit()
1136
+ else:
1137
+ with open(cmdValues["data"],"r") as dataFile:
1138
+
1139
+ if not shutil.which("awk"):
1140
+ print("Install awk and try again... exiting...")
1141
+ sys.exit()
1142
+ else:
1143
+ pass
1144
+
1145
+ tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
1146
+ dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
1147
+ dataLines = int(dataFileLines.communicate()[0])
1148
+ dataFileLines.kill()
1149
+
1150
+ bar = ChargingBar('Loading data table', max=int(dataLines))
1151
+ for dataPos,dataRow in enumerate(dataFile):
1152
+ if len(dataRow.strip().split(sepFormat)) == 0:
1153
+ pass
1154
+ else:
1155
+ if dataPos == 0:
1156
+ dataRowValues = dataRow.strip().split(sepFormat)
1157
+ dataRowValuesName = dataRowValues[0]
1158
+ dataRowValuesCols = [item for item in dataRowValues[1:]]
1159
+ dataMatrixDF[dataRowValuesName] = dataRowValuesCols
1160
+ else:
1161
+ dataRowValues = dataRow.strip().split(sepFormat)
1162
+ dataRowValuesName = dataRowValues[0]
1163
+ dataRowValuesCols = [int(item) for item in dataRowValues[1:]]
1164
+ dataMatrixDF[dataRowValuesName] = dataRowValuesCols
1165
+
1166
+ if set(dataRowValuesCols) == {0} or set(dataRowValuesCols) == {1} or set(dataRowValuesCols) == {0,1}:
1167
+ pass
1168
+ else:
1169
+ print(f"--->>>Row {dataPos} in {cmdValues["data"]} contains unexpected other values besides 0 and 1... exiting...")
1170
+ sys.exit()
1171
+
1172
+ bar.next()
1173
+ bar.finish()
1174
+
1175
+ return(dataMatrixDF)
1176
+
1177
+
1178
+ def citation():
1179
+ print(f"Chrispin Chaguza. 2026. matrixToSimilarity: calculating percent similarity between pairs of items (rows) \
1180
+ given tabular text file. https://github.com/ChrispinChaguza/matrixToSimilarity")
1181
+
1182
+
1183
+ def printVersion():
1184
+ print(f"tabkit {__version__}")
1185
+
1186
+
1187
+ def tabtkMain():
1188
+
1189
+ optionsCmd = argparse.ArgumentParser(sys.argv[0],
1190
+ usage=argparse.SUPPRESS,
1191
+ description='tabtk: A toolkit for sequence similarity analysis based on tabular data',
1192
+ prefix_chars='-',
1193
+ add_help=True,
1194
+ epilog='Written by Chrispin Chaguza, St Jude Children\'s Research Hospital, 2026')
1195
+
1196
+ optionsCommand = optionsCmd.add_subparsers(dest="command")
1197
+
1198
+ similarityOptions = optionsCommand.add_parser("similarity")
1199
+
1200
+ similarityOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1201
+ metavar='data',dest='data',
1202
+ help='Input tabular text file (input should be file of fasta files when used with --mash/-m and --fastani/-n options or alignment file when used with --aln/-a option)')
1203
+ similarityOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1204
+ metavar='format',dest='format',default="tsv",
1205
+ choices=['tsv','csv'],
1206
+ help='Data format (default: tsv)')
1207
+ similarityOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1208
+ metavar='output',dest='output',default="out.similarity.tsv",
1209
+ help='Output percent similarity')
1210
+ similarityOptions.add_argument('--length','-l',action='store',required=False,nargs=1,
1211
+ metavar='length',dest='length',default=31,type=int,
1212
+ help='K-mer length for --mash/-m or --fastani/-n (default: 31)')
1213
+ similarityOptions.add_argument('--size','-s',action='store',required=False,nargs=1,
1214
+ metavar='size',dest='size',default=1000000,type=int,
1215
+ help='Sketch size for MASH; use with --mash (default: 1000000)')
1216
+ similarityOptions.add_argument('--mash','-m',action='store_true',default=False,
1217
+ dest='mash',help='Calculate similarity using MASH')
1218
+ similarityOptions.add_argument('--fastani','-n',action='store_true',default=False,
1219
+ dest='fastani',help='Calculate similarity using fastANI')
1220
+ similarityOptions.add_argument('--aln','-a',action='store_true',default=False,
1221
+ dest='aln',help='Calculate similarity based on a given sequence alignment')
1222
+ similarityOptions.add_argument('--fraction','-c',action='store',required=False,nargs=1,
1223
+ metavar='fraction',dest='fraction',default=0.70,type=float,
1224
+ help='Minimum overlap fraction with shorter sequence for ANI calculation; use with --fastani (default: 0.70)')
1225
+ similarityOptions.add_argument('--fragment','-r',action='store',required=False,nargs=1,
1226
+ metavar='fragment',dest='fragment',default=500,type=int,
1227
+ help='Fragment length for calculating ANI; use with --fastani (default: 500)')
1228
+ similarityOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
1229
+ metavar='threads',dest='threads',default=5,type=int,
1230
+ help='Number of threads (default: 5)')
1231
+
1232
+
1233
+ orderOptions = optionsCommand.add_parser("order")
1234
+
1235
+ orderOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1236
+ metavar='data',dest='data',
1237
+ help='Input tabular text file')
1238
+ orderOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1239
+ metavar='format',dest='format',default="tsv",
1240
+ choices=['tsv','csv'],
1241
+ help='Data format (default: tsv)')
1242
+ orderOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1243
+ metavar='output',dest='output',default="out.ordered.tsv",
1244
+ help='Output ordered data file')
1245
+ orderOptions.add_argument('--alphabetic','-a',action='store_true',required=False,
1246
+ dest='alphabetic',default=False,
1247
+ help='Show similarity values as percentage instead of fraction')
1248
+ orderOptions.add_argument('--rownames','-r',action='store',required=False,nargs=1,
1249
+ metavar='rownames',dest='rownames',
1250
+ help='Order by item names in a given text file (one name per row)')
1251
+ orderOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
1252
+ metavar='threads',dest='threads',default=5,type=int,
1253
+ help='Number of threads (default: 5)')
1254
+
1255
+ renameOptions = optionsCommand.add_parser("rename")
1256
+
1257
+ renameOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1258
+ metavar='data',dest='data',
1259
+ help='Input tabular text file')
1260
+ renameOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1261
+ metavar='format',dest='format',default="tsv",
1262
+ choices=['tsv','csv'],
1263
+ help='Data format (default: tsv)')
1264
+ renameOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1265
+ metavar='output',dest='output',default="out.renamed.tsv",
1266
+ help='Output ordered data file')
1267
+ renameOptions.add_argument('--columns','-c',action='store',required=True,nargs="*",
1268
+ metavar='columns',dest='columns',
1269
+ help='Selected column numbers (e.g., --columns 1 2 3 4)')
1270
+ renameOptions.add_argument('--newdata','-n',action='store',required=True,nargs=1,
1271
+ metavar='newdata',dest='newdata',
1272
+ help='Order by item names in a given text file (one name per row)')
1273
+
1274
+
1275
+ filterOptions = optionsCommand.add_parser("filter")
1276
+
1277
+ filterOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1278
+ metavar='data',dest='data',
1279
+ help='Input tabular text file')
1280
+ filterOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1281
+ metavar='format',dest='format',default="tsv",
1282
+ choices=['tsv','csv'],
1283
+ help='Data format (default: tsv)')
1284
+ filterOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1285
+ metavar='output',dest='output',default="out.filter.tsv",
1286
+ help='Output percent similarity')
1287
+ filterOptions.add_argument('--min','-a',action='store',required=False,nargs=1,
1288
+ metavar='minCutoff',dest='minCutoff',default=0,type=int,
1289
+ help='Select with frequency ≥minCutoff percent (overrides --columns/-c and --rows/-r)')
1290
+ filterOptions.add_argument('--max','-b',action='store',required=False,nargs=1,
1291
+ metavar='maxCutoff',dest='maxCutoff',default=100,type=int,
1292
+ help='Select with frequency ≤maxCutoff percent (overrides --columns/-c and --rows/-r)')
1293
+ filterOptions.add_argument('--columns','-c',action='store',required=False,nargs=1,
1294
+ metavar='columns',dest='columns',
1295
+ help='Select only columns present in a given text file (one column value per row)')
1296
+ filterOptions.add_argument('--rows','-r',action='store',required=False,nargs=1,
1297
+ metavar='rows',dest='rows',
1298
+ help='Select only rows present in a given text file (one column value per row)')
1299
+ filterOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
1300
+ metavar='threads',dest='threads',default=5,type=int,
1301
+ help='Number of threads (default: 5)')
1302
+
1303
+
1304
+ sampleOptions = optionsCommand.add_parser("sample")
1305
+
1306
+ sampleOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1307
+ metavar='data',dest='data',
1308
+ help='Input tabular text file')
1309
+ sampleOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1310
+ metavar='format',dest='format',default="tsv",
1311
+ choices=['tsv','csv'],
1312
+ help='Data format (default: tsv)')
1313
+ sampleOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1314
+ metavar='output',dest='output',default="out.sample.tsv",
1315
+ help='Output file')
1316
+ sampleOptions.add_argument('--cf','-c',action='store',required=False,nargs=1,
1317
+ metavar='colfreq',dest='colfreq',default=0,type=float,
1318
+ help='Randomly select ≥colfreq percent of the columns')
1319
+ sampleOptions.add_argument('--rf','-r',action='store',required=False,nargs=1,
1320
+ metavar='rowfreq',dest='rowfreq',default=0,type=float,
1321
+ help='Randomly select ≥rowfreq percent of the rows')
1322
+
1323
+
1324
+ transposeOptions = optionsCommand.add_parser("transpose")
1325
+
1326
+ transposeOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1327
+ metavar='data',dest='data',
1328
+ help='Input tabular text file')
1329
+ transposeOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1330
+ metavar='format',dest='format',default="tsv",
1331
+ choices=['tsv','csv'],
1332
+ help='Data format (default: tsv)')
1333
+ transposeOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1334
+ metavar='output',dest='output',default="out.filter.tsv",
1335
+ help='Output percent similarity')
1336
+
1337
+
1338
+ clusterOptions = optionsCommand.add_parser("cluster")
1339
+
1340
+ clusterOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1341
+ metavar='data',dest='data',
1342
+ help='Input tabular text file')
1343
+ clusterOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1344
+ metavar='format',dest='format',default="tsv",
1345
+ choices=['tsv','csv'],
1346
+ help='Data format (default: tsv)')
1347
+ clusterOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1348
+ metavar='output',dest='output',default="out.clusters.tsv",
1349
+ help='Output clusters')
1350
+ clusterOptions.add_argument('--prefix','-p',action='store',required=False,nargs=1,
1351
+ metavar='prefix',dest='prefix',default="CLS",
1352
+ help='Cluster label prefix')
1353
+ clusterOptions.add_argument('--sep','-s',action='store',required=False,nargs=1,
1354
+ metavar='separator',dest='separator',default="-",
1355
+ help='Lineage label separator')
1356
+ clusterOptions.add_argument('--thresholds','-t',action='store',required=True,nargs="*",
1357
+ metavar='thresholds',dest='thresholds',
1358
+ help='Cluster or lineage thresholds')
1359
+
1360
+
1361
+ gwasOptions = optionsCommand.add_parser("gwas")
1362
+
1363
+ gwasOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1364
+ metavar='data',dest='data',
1365
+ help='Input tabular text file')
1366
+ gwasOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
1367
+ metavar='format',dest='format',default="tsv",
1368
+ choices=['tsv','csv'],
1369
+ help='Data format (default: tsv)')
1370
+ gwasOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1371
+ metavar='output',dest='output',default="out.gwas.tsv",
1372
+ help='Output percent similarity')
1373
+ gwasOptions.add_argument('--phenotype','-p',action='store',required=False,nargs=1,
1374
+ metavar='phenotype',dest='phenotype',
1375
+ help='Input phenotype text file (row name,phenotype); no header')
1376
+ gwasOptions.add_argument('--pyseer','-s',action='store_true',required=False,
1377
+ dest='pyseer',default=False,
1378
+ help='Output presence and absence data for Pyseer')
1379
+ gwasOptions.add_argument('--plink','-l',action='store_true',required=False,
1380
+ dest='plink',default=False,
1381
+ help='Output presence and absence data for Pyseer')
1382
+ gwasOptions.add_argument('--fastlmm','-m',action='store_true',required=False,
1383
+ dest='fastlmm',default=False,
1384
+ help='Output presence and absence data for Pyseer')
1385
+ gwasOptions.add_argument('--gemma','-g',action='store_true',required=False,
1386
+ dest='gemma',default=False,
1387
+ help='Output presence and absence data for GEMMA')
1388
+ gwasOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
1389
+ metavar='threads',dest='threads',default=5,type=int,
1390
+ help='Number of threads (default: 5)')
1391
+
1392
+
1393
+ kmerMatrixOptions = optionsCommand.add_parser("table")
1394
+
1395
+ kmerMatrixOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
1396
+ metavar='data',dest='data',
1397
+ help='Input text files containing fasta sequence file names')
1398
+ kmerMatrixOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
1399
+ metavar='output',dest='output',default="db.kmers.fasta",
1400
+ help='Output percent similarity')
1401
+ kmerMatrixOptions.add_argument('--length','-l',action='store',required=False,nargs=1,
1402
+ metavar='length',dest='length',default=31,type=int,
1403
+ help='K-mer length (default: 31)')
1404
+ kmerMatrixOptions.add_argument('--min','-m',action='store',required=False,nargs=1,
1405
+ metavar='minabundance',dest='minabundance',default=1,type=int,
1406
+ help='Minimum k-mer abundance (minabundance: 1)')
1407
+ kmerMatrixOptions.add_argument('--maxmem','-x',action='store',required=False,nargs=1,
1408
+ metavar='maxmemory',dest='maxmemory',default=31,type=int,
1409
+ help='Maximum memory (default: 5000)')
1410
+ kmerMatrixOptions.add_argument('--dsk','-s',action='store_true',default=True,
1411
+ dest='dsk',help='DSK to identify k-mers')
1412
+ kmerMatrixOptions.add_argument('--bifrost','-b',action='store_true',default=False,
1413
+ dest='bifrost',help='Bifrost to identify variable length k-mers or unitigs')
1414
+ kmerMatrixOptions.add_argument('--kmers','-k',action='store',required=False,nargs=1,
1415
+ metavar='kmers',dest='kmers',default="",
1416
+ help='Only query input sequence against a given k-mer database')
1417
+ kmerMatrixOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
1418
+ metavar='threads',dest='threads',default=5,type=int,
1419
+ help='Number of threads (default: 5)')
1420
+
1421
+
1422
+ versionOptions = optionsCommand.add_parser("version")
1423
+
1424
+ versionOptions.add_argument('--version','-v',action='store_false',default=True,
1425
+ dest='version',help='Show software version')
1426
+
1427
+
1428
+ citationOptions = optionsCommand.add_parser("citation")
1429
+
1430
+ citationOptions.add_argument('--citation','-c',action='store_false',default=True,
1431
+ dest='citation',help='Show software citation information')
1432
+
1433
+
1434
+ options = optionsCmd.parse_args(args=None if sys.argv[1:] else ['--help'])
1435
+
1436
+ cmdValues = {}
1437
+
1438
+ if options.command == "similarity":
1439
+ cmdValues = {'data': options.data[0:][0],
1440
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1441
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1442
+ 'length': int(options.length[0:][0]) if isinstance(options.length,list) else int(options.length),
1443
+ 'fragment': int(options.fragment[0:][0]) if isinstance(options.fragment,list) else int(options.fragment),
1444
+ 'fraction': int(options.fraction[0:][0]) if isinstance(options.fraction,list) else int(options.fraction),
1445
+ 'size': int(options.size[0:][0]) if isinstance(options.size,list) else int(options.size),
1446
+ 'mash': options.mash,
1447
+ 'fastani': options.fastani,
1448
+ 'aln': options.aln,
1449
+ 'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
1450
+ }
1451
+
1452
+ similarity(cmdValues)
1453
+
1454
+ elif options.command == "filter":
1455
+ cmdValues = {'data': options.data[0:][0],
1456
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1457
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1458
+ 'minCutoff': int(options.minCutoff[0:][0]) if isinstance(options.minCutoff,list) else int(options.minCutoff),
1459
+ 'maxCutoff': int(options.maxCutoff[0:][0]) if isinstance(options.maxCutoff,list) else int(options.maxCutoff),
1460
+ 'rows': options.rows[0:][0] if isinstance(options.rows,list) else options.rows,
1461
+ 'columns': options.columns[0:][0] if isinstance(options.columns,list) else options.columns,
1462
+ 'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
1463
+ }
1464
+
1465
+ filter(cmdValues)
1466
+
1467
+ elif options.command == "sample":
1468
+ cmdValues = {'data': options.data[0:][0],
1469
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1470
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1471
+ 'colfreq': float(options.colfreq[0:][0]) if isinstance(options.colfreq,list) else float(options.colfreq),
1472
+ 'rowfreq': float(options.rowfreq[0:][0]) if isinstance(options.rowfreq,list) else float(options.rowfreq)
1473
+ }
1474
+
1475
+ sample(cmdValues)
1476
+
1477
+ elif options.command == "order":
1478
+ cmdValues = {'data': options.data[0:][0],
1479
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1480
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1481
+ 'rownames': options.rownames[0:][0] if isinstance(options.rownames,list) else options.rownames,
1482
+ 'alphabetic': options.alphabetic,
1483
+ 'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
1484
+ }
1485
+
1486
+ order(cmdValues)
1487
+
1488
+ elif options.command == "rename":
1489
+ cmdValues = {'data': options.data[0:][0],
1490
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1491
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1492
+ 'newdata': options.newdata[0:][0] if isinstance(options.newdata,list) else options.newdata,
1493
+ 'columns': options.columns[0:] if isinstance(options.columns,list) else options.columns
1494
+ }
1495
+
1496
+ rename(cmdValues)
1497
+
1498
+ elif options.command == "cluster":
1499
+ cmdValues = {'data': options.data[0:][0],
1500
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1501
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1502
+ 'prefix': options.prefix[0:][0] if isinstance(options.prefix,list) else options.prefix,
1503
+ 'separator': options.separator[0:][0] if isinstance(options.separator,list) else options.separator,
1504
+ 'thresholds': [float(i) for i in options.thresholds[0:]] if isinstance(options.thresholds,list) else float(options.thresholds)
1505
+ }
1506
+
1507
+ cluster(cmdValues)
1508
+
1509
+ elif options.command == "transpose":
1510
+ cmdValues = {'data': options.data[0:][0],
1511
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1512
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output
1513
+ }
1514
+
1515
+ transpose(cmdValues)
1516
+
1517
+ elif options.command == "gwas":
1518
+ cmdValues = {'data': options.data[0:][0],
1519
+ 'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
1520
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1521
+ 'phenotype': options.phenotype[0:][0] if isinstance(options.phenotype,list) else options.phenotype,
1522
+ 'pyseer': options.pyseer,
1523
+ 'fastlmm': options.fastlmm,
1524
+ 'plink': options.plink,
1525
+ 'gemma': options.gemma,
1526
+ 'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
1527
+ }
1528
+
1529
+ gwas(cmdValues)
1530
+
1531
+ elif options.command == "table":
1532
+ cmdValues = {'data': options.data[0:][0],
1533
+ 'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
1534
+ 'dsk': options.dsk,
1535
+ 'bifrost': options.bifrost,
1536
+ 'minabundance': int(options.minabundance[0:][0]) if isinstance(options.minabundance,list) else int(options.minabundance),
1537
+ 'maxmemory': int(options.maxmemory[0:][0]) if isinstance(options.maxmemory,list) else int(options.maxmemory),
1538
+ 'length': int(options.length[0:][0]) if isinstance(options.length,list) else int(options.length),
1539
+ 'kmers': options.kmers[0:][0] if isinstance(options.kmers,list) else options.kmers,
1540
+ 'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
1541
+ }
1542
+
1543
+ table(cmdValues)
1544
+
1545
+ elif options.command == "version":
1546
+ printVersion()
1547
+
1548
+ elif options.command == "citation":
1549
+ citation()
1550
+
1551
+ else:
1552
+ sys.exit()
1553
+
1554
+
1555
+ if __name__=="__main__":
1556
+ tabtkMain()