tabtk 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tabtk/__init__.py +4 -0
- tabtk/__main__.py +28 -0
- tabtk/tabtk.py +1556 -0
- tabtk-1.0.0.dist-info/METADATA +44 -0
- tabtk-1.0.0.dist-info/RECORD +9 -0
- tabtk-1.0.0.dist-info/WHEEL +5 -0
- tabtk-1.0.0.dist-info/entry_points.txt +2 -0
- tabtk-1.0.0.dist-info/licenses/LICENSE +21 -0
- tabtk-1.0.0.dist-info/top_level.txt +1 -0
tabtk/tabtk.py
ADDED
|
@@ -0,0 +1,1556 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
from Bio import SeqIO
|
|
6
|
+
import multiprocessing
|
|
7
|
+
import argparse
|
|
8
|
+
import itertools
|
|
9
|
+
from progress.bar import ChargingBar
|
|
10
|
+
import re
|
|
11
|
+
import subprocess
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
import numpy as np
|
|
14
|
+
import shutil
|
|
15
|
+
import pandas as pd
|
|
16
|
+
import networkx as nx
|
|
17
|
+
import math
|
|
18
|
+
import datetime
|
|
19
|
+
from .__init__ import __version__
|
|
20
|
+
|
|
21
|
+
def pairwiseSimilarity(firstRecord,secondRecord):
|
|
22
|
+
recordMatch = 0
|
|
23
|
+
recordMisMatch = 0
|
|
24
|
+
|
|
25
|
+
firstRecordName = list(firstRecord.keys())[0]
|
|
26
|
+
secondRecordName = list(secondRecord.keys())[0]
|
|
27
|
+
|
|
28
|
+
for r,s in zip(firstRecord[firstRecordName],secondRecord[secondRecordName]):
|
|
29
|
+
if not (r == 0 and s == 0):
|
|
30
|
+
if r == s:
|
|
31
|
+
if (r in [0,1] and s in [0,1]):
|
|
32
|
+
recordMatch = recordMatch + 1
|
|
33
|
+
else:
|
|
34
|
+
print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
|
|
35
|
+
sys.exit()
|
|
36
|
+
else:
|
|
37
|
+
if (r in [0,1] and s in [0,1]):
|
|
38
|
+
recordMisMatch = recordMisMatch + 1
|
|
39
|
+
else:
|
|
40
|
+
print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
|
|
41
|
+
sys.exit()
|
|
42
|
+
else:
|
|
43
|
+
if (r in [0,1] and s in [0,1]):
|
|
44
|
+
pass
|
|
45
|
+
else:
|
|
46
|
+
print(f"--->>>Found characters other than 0 and 1 when comparing items {firstRecordName} and {secondRecordName}... exiting...")
|
|
47
|
+
sys.exit()
|
|
48
|
+
|
|
49
|
+
percentSimilarity = recordMatch/(recordMatch + recordMisMatch)*100
|
|
50
|
+
|
|
51
|
+
similarityData = [firstRecordName,secondRecordName,percentSimilarity]
|
|
52
|
+
|
|
53
|
+
#print('.', end='', flush=True)
|
|
54
|
+
|
|
55
|
+
return(similarityData)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def alignmentSimilarity(ij):
|
|
59
|
+
seqMatch=0
|
|
60
|
+
seqMisMatch=0
|
|
61
|
+
|
|
62
|
+
for r,s in zip(str(i.seq).replace("-","N").replace("X","N").upper(),str(j.seq).replace("-","N").replace("X","N").upper()):
|
|
63
|
+
if r==s and not (r=="N" or s=="N"):
|
|
64
|
+
seqMatch=seqMatch+1
|
|
65
|
+
else:
|
|
66
|
+
if r=="N" or s=="N":
|
|
67
|
+
continue
|
|
68
|
+
else:
|
|
69
|
+
seqMisMatch=seqMisMatch+1
|
|
70
|
+
|
|
71
|
+
seqPID=round(seqMatch/(seqMatch+seqMisMatch)*100,4) if (seqMatch+seqMisMatch)!=0 else 0
|
|
72
|
+
|
|
73
|
+
seqLen=round(len(i.seq),4)
|
|
74
|
+
|
|
75
|
+
seqCov=round((seqMatch+seqMisMatch)/seqLen*100,4) if seqLen!=0 else 0
|
|
76
|
+
|
|
77
|
+
retValue = f"{i.id}\t{j.id}\t{seqPID}"
|
|
78
|
+
|
|
79
|
+
return(retValue)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def mashSimilarity(cmdValues):
|
|
83
|
+
if not shutil.which("mash"):
|
|
84
|
+
print("Install MASH and try again... exiting...")
|
|
85
|
+
sys.exit()
|
|
86
|
+
|
|
87
|
+
else:
|
|
88
|
+
pass
|
|
89
|
+
|
|
90
|
+
tmpOut = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
91
|
+
tmpOutR = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
92
|
+
|
|
93
|
+
print("Calculating percent similarity... may take longer... be patient...")
|
|
94
|
+
|
|
95
|
+
mashSketchCmd = f"mash sketch -p {cmdValues["threads"]} -k {cmdValues["length"]} -o {tmpOut} -l {cmdValues["data"]} -s {cmdValues["size"]}"
|
|
96
|
+
mashDistCmd = f"mash dist -p {cmdValues["threads"]} -s {cmdValues["size"]} {f"{tmpOut}.msh"} {f"{tmpOut}.msh"} > {tmpOutR}"
|
|
97
|
+
|
|
98
|
+
subprocess.call(mashSketchCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
99
|
+
subprocess.call(mashDistCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
100
|
+
|
|
101
|
+
mashResults = []
|
|
102
|
+
|
|
103
|
+
with open(tmpOutR,"r") as mashFile:
|
|
104
|
+
for mashOutput in mashFile:
|
|
105
|
+
item = mashOutput.strip().replace("/","\t").split("\t")
|
|
106
|
+
mashResults.append(f"{Path(item[0]).stem}\t{Path(item[1]).stem}\t{float(item[4])/float(item[5])*100}")
|
|
107
|
+
|
|
108
|
+
if os.path.exists(tmpOut):
|
|
109
|
+
os.remove(tmpOut)
|
|
110
|
+
else:
|
|
111
|
+
pass
|
|
112
|
+
|
|
113
|
+
if os.path.exists(tmpOutR):
|
|
114
|
+
os.remove(tmpOutR)
|
|
115
|
+
else:
|
|
116
|
+
pass
|
|
117
|
+
|
|
118
|
+
if os.path.exists(f"{tmpOut}.msh"):
|
|
119
|
+
os.remove(f"{tmpOut}.msh")
|
|
120
|
+
else:
|
|
121
|
+
pass
|
|
122
|
+
|
|
123
|
+
return(mashResults)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def fastaniSimilarity(cmdValues):
|
|
127
|
+
if not shutil.which("fastANI"):
|
|
128
|
+
print("Install fastANI and try again... exiting...")
|
|
129
|
+
sys.exit()
|
|
130
|
+
|
|
131
|
+
else:
|
|
132
|
+
pass
|
|
133
|
+
|
|
134
|
+
tmpOut = f"mash.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
135
|
+
|
|
136
|
+
fastaniCmd = f"fastANI --rl {cmdValues["data"]} --ql {cmdValues["data"]} -k {cmdValues["length"]} --matrix --fragLen {cmdValues["fragment"]} --minFraction {cmdValues["fraction"]} --output {tmpOut} -t {cmdValues["threads"]}"
|
|
137
|
+
|
|
138
|
+
print("Calculating percent similarity... may take longer... be patient...")
|
|
139
|
+
|
|
140
|
+
subprocess.call(fastaniCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
141
|
+
|
|
142
|
+
fastaniResults = []
|
|
143
|
+
|
|
144
|
+
with open(tmpOut,"r") as fastaniFile:
|
|
145
|
+
for itemPos,fastaniOutput in enumerate(fastaniFile):
|
|
146
|
+
if itemPos == 0:
|
|
147
|
+
pass
|
|
148
|
+
else:
|
|
149
|
+
item = fastaniOutput.strip().split("\t")
|
|
150
|
+
fastaniResults.append(f"{Path(item[0]).stem}\t{Path(item[1]).stem}\t{float(item[2])}")
|
|
151
|
+
|
|
152
|
+
if os.path.exists(tmpOut):
|
|
153
|
+
os.remove(tmpOut)
|
|
154
|
+
else:
|
|
155
|
+
pass
|
|
156
|
+
|
|
157
|
+
if os.path.exists(f"{tmpOut}.matrix"):
|
|
158
|
+
os.remove(f"{tmpOut}.matrix")
|
|
159
|
+
else:
|
|
160
|
+
pass
|
|
161
|
+
|
|
162
|
+
return(fastaniResults)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def similarity(cmdValues):
|
|
166
|
+
|
|
167
|
+
if sum([1 if cmdValues["aln"] else 0, 1 if cmdValues["mash"] else 0, 1 if cmdValues["fastani"] else 0]) > 1:
|
|
168
|
+
print("Specify none or only one of these options --aln/-a, --mash/-m, and --fastani/-n... exiting...")
|
|
169
|
+
sys.exit()
|
|
170
|
+
else:
|
|
171
|
+
pass
|
|
172
|
+
|
|
173
|
+
if (not cmdValues["aln"]) and (not cmdValues["mash"]) and (not cmdValues["fastani"]):
|
|
174
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
175
|
+
|
|
176
|
+
itemPairs = itertools.permutations([item for pos,item in enumerate(dataMatrixDF.keys()) if pos>0], r=2)
|
|
177
|
+
tmpPairs = itertools.permutations([item for pos,item in enumerate(dataMatrixDF.keys()) if pos>0], r=2)
|
|
178
|
+
|
|
179
|
+
print("Calculating pairwise similarity... may take longer... be patient...")
|
|
180
|
+
|
|
181
|
+
with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
|
|
182
|
+
args = [({firstItem: dataMatrixDF[firstItem]}, {secondItem: dataMatrixDF[secondItem]}) for firstItem,secondItem in itemPairs]
|
|
183
|
+
similarityResults = pool.starmap(pairwiseSimilarity, args)
|
|
184
|
+
|
|
185
|
+
with open(cmdValues["output"],"w") as simData:
|
|
186
|
+
simData.write("seq1\tseq2\tpid\n")
|
|
187
|
+
|
|
188
|
+
bar = ChargingBar('Writing similarity values', max=len(similarityResults))
|
|
189
|
+
for itemPair in similarityResults:
|
|
190
|
+
simData.write(f"{itemPair[0]}\t{itemPair[1]}\t{itemPair[2]}\n")
|
|
191
|
+
|
|
192
|
+
bar.next()
|
|
193
|
+
bar.finish()
|
|
194
|
+
|
|
195
|
+
elif (not cmdValues["aln"]) and (cmdValues["mash"]) and (not cmdValues["fastani"]):
|
|
196
|
+
similarityResults = mashSimilarity(cmdValues)
|
|
197
|
+
|
|
198
|
+
with open(cmdValues["output"],"w") as simData:
|
|
199
|
+
simData.write("seq1\tseq2\tpid\n")
|
|
200
|
+
|
|
201
|
+
bar = ChargingBar('Writing similarity values', max=len(similarityResults))
|
|
202
|
+
for itemPair in similarityResults:
|
|
203
|
+
simData.write(f"{itemPair}\n")
|
|
204
|
+
|
|
205
|
+
bar.next()
|
|
206
|
+
bar.finish()
|
|
207
|
+
|
|
208
|
+
elif (not cmdValues["aln"]) and (not cmdValues["mash"]) and (cmdValues["fastani"]):
|
|
209
|
+
similarityResults = fastaniSimilarity(cmdValues)
|
|
210
|
+
|
|
211
|
+
with open(cmdValues["output"],"w") as simData:
|
|
212
|
+
simData.write("seq1\tseq2\tpid\n")
|
|
213
|
+
|
|
214
|
+
bar = ChargingBar('Writing similarity values', max=len(similarityResults))
|
|
215
|
+
for itemPair in similarityResults:
|
|
216
|
+
simData.write(f"{itemPair}\n")
|
|
217
|
+
|
|
218
|
+
bar.next()
|
|
219
|
+
bar.finish()
|
|
220
|
+
|
|
221
|
+
elif (cmdValues["aln"]) and (not cmdValues["mash"]) and (not cmdValues["fastani"]):
|
|
222
|
+
alignment={}
|
|
223
|
+
|
|
224
|
+
for i in SeqIO.parse(cmdValues['data'],"fasta"):
|
|
225
|
+
alignment[i.id]=i
|
|
226
|
+
|
|
227
|
+
fhandle=open(str(cmdValues['outputFile']),"w")
|
|
228
|
+
|
|
229
|
+
fhandle.write("seq1\tseq2\tpid\n")
|
|
230
|
+
|
|
231
|
+
tmpList = sorted(map(sorted, combinations(set([i for i in alignment.keys()]), 2)))
|
|
232
|
+
|
|
233
|
+
with multiprocessing.Pool(processes=cmdValues['threads']) as pool:
|
|
234
|
+
args=[(alignment[m],alignment[n]) for m,n in tmpList]
|
|
235
|
+
results=pool.starmap(alignmentSimilarity, args)
|
|
236
|
+
|
|
237
|
+
else:
|
|
238
|
+
pass
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def transpose(cmdValues):
|
|
242
|
+
if not shutil.which("datamash"):
|
|
243
|
+
print("Install 'datamash' software and try again... exiting...")
|
|
244
|
+
sys.exit()
|
|
245
|
+
|
|
246
|
+
else:
|
|
247
|
+
pass
|
|
248
|
+
|
|
249
|
+
bar = ChargingBar("Transposing table", max=1)
|
|
250
|
+
datamashCmd = f"datamash transpose < {cmdValues["data"]} > {f"{Path(cmdValues["output"]).stem}.transposed.tsv"}"
|
|
251
|
+
|
|
252
|
+
subprocess.call(datamashCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
253
|
+
|
|
254
|
+
bar.next()
|
|
255
|
+
bar.finish()
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def lineages1Levels(cmdValues):
|
|
259
|
+
np.random.rand(1)
|
|
260
|
+
|
|
261
|
+
snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
|
|
262
|
+
|
|
263
|
+
Graph = nx.Graph()
|
|
264
|
+
Graph.add_nodes_from(snpDistDF.seq2.to_list())
|
|
265
|
+
Graph.add_nodes_from(snpDistDF.seq1.to_list())
|
|
266
|
+
|
|
267
|
+
snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
|
|
268
|
+
seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
|
|
269
|
+
|
|
270
|
+
Graph.add_edges_from(seqPair)
|
|
271
|
+
networkClusters = tuple(nx.connected_components(Graph))
|
|
272
|
+
|
|
273
|
+
fhandle = open(f"{cmdValues['output']}","w")
|
|
274
|
+
|
|
275
|
+
for i,j in enumerate(networkClusters):
|
|
276
|
+
for s in j:
|
|
277
|
+
FirstLevelLin = str(i+1)
|
|
278
|
+
seqLabel = s
|
|
279
|
+
clusterPrefix = cmdValues["prefix"]
|
|
280
|
+
separator = cmdValues["separator"]
|
|
281
|
+
|
|
282
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
|
|
283
|
+
|
|
284
|
+
fhandle.close();
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def lineages2Levels(cmdValues):
|
|
288
|
+
np.random.rand(1)
|
|
289
|
+
|
|
290
|
+
snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
|
|
291
|
+
|
|
292
|
+
Graph = nx.Graph()
|
|
293
|
+
Graph.add_nodes_from(snpDistDF.seq2.to_list())
|
|
294
|
+
Graph.add_nodes_from(snpDistDF.seq1.to_list())
|
|
295
|
+
|
|
296
|
+
snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
|
|
297
|
+
seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
|
|
298
|
+
|
|
299
|
+
Graph.add_edges_from(seqPair)
|
|
300
|
+
networkClusters = tuple(nx.connected_components(Graph))
|
|
301
|
+
|
|
302
|
+
fhandle = open(f"{cmdValues['output']}","w")
|
|
303
|
+
|
|
304
|
+
for i,j in enumerate(networkClusters):
|
|
305
|
+
snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
|
|
306
|
+
snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
|
|
307
|
+
|
|
308
|
+
Graph1 = nx.Graph()
|
|
309
|
+
Graph1.add_nodes_from(j)
|
|
310
|
+
|
|
311
|
+
seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
|
|
312
|
+
|
|
313
|
+
Graph1.add_edges_from(seqPair1)
|
|
314
|
+
networkClusters1 = tuple(nx.connected_components(Graph1))
|
|
315
|
+
|
|
316
|
+
for r,k in enumerate(networkClusters1):
|
|
317
|
+
for s in k:
|
|
318
|
+
FirstLevelLin = str(i+1)
|
|
319
|
+
seqLabel = s
|
|
320
|
+
SecondLevelLin = str(r+1)
|
|
321
|
+
clusterPrefix = cmdValues["prefix"]
|
|
322
|
+
separator = cmdValues["separator"]
|
|
323
|
+
|
|
324
|
+
if len(networkClusters1)==1:
|
|
325
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
|
|
326
|
+
else:
|
|
327
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
|
|
328
|
+
|
|
329
|
+
fhandle.close();
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def lineages3Levels(cmdValues):
|
|
333
|
+
np.random.rand(1)
|
|
334
|
+
|
|
335
|
+
snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
|
|
336
|
+
|
|
337
|
+
Graph = nx.Graph()
|
|
338
|
+
Graph.add_nodes_from(snpDistDF.seq2.to_list())
|
|
339
|
+
Graph.add_nodes_from(snpDistDF.seq1.to_list())
|
|
340
|
+
|
|
341
|
+
snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
|
|
342
|
+
seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
|
|
343
|
+
|
|
344
|
+
Graph.add_edges_from(seqPair)
|
|
345
|
+
networkClusters = tuple(nx.connected_components(Graph))
|
|
346
|
+
|
|
347
|
+
fhandle = open(f"{cmdValues['output']}","w")
|
|
348
|
+
|
|
349
|
+
for i,j in enumerate(networkClusters):
|
|
350
|
+
snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
|
|
351
|
+
snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
|
|
352
|
+
|
|
353
|
+
Graph1 = nx.Graph()
|
|
354
|
+
Graph1.add_nodes_from(j)
|
|
355
|
+
|
|
356
|
+
seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
|
|
357
|
+
|
|
358
|
+
Graph1.add_edges_from(seqPair1)
|
|
359
|
+
networkClusters1 = tuple(nx.connected_components(Graph1))
|
|
360
|
+
|
|
361
|
+
for p,q in enumerate(networkClusters1):
|
|
362
|
+
snpDistDF2R = snpDistDF[snpDistDF.seq1.isin(q) & snpDistDF.seq2.isin(q)]
|
|
363
|
+
snpDistDF3R = snpDistDF2R[snpDistDF2R.pid >= cmdValues["thresholds"][2]]
|
|
364
|
+
|
|
365
|
+
Graph2 = nx.Graph()
|
|
366
|
+
Graph2.add_nodes_from(q)
|
|
367
|
+
|
|
368
|
+
seqPair1R = [(snpDistDF3R.seq1[z],snpDistDF3R.seq2[z]) for z in snpDistDF3R.index]
|
|
369
|
+
|
|
370
|
+
Graph2.add_edges_from(seqPair1R)
|
|
371
|
+
networkClusters2 = tuple(nx.connected_components(Graph2))
|
|
372
|
+
|
|
373
|
+
for r,k in enumerate(networkClusters2):
|
|
374
|
+
for s in k:
|
|
375
|
+
FirstLevelLin = str(i+1)
|
|
376
|
+
SecondLevelLin = str(p+1)
|
|
377
|
+
ThirdLevelLin = str(r+1)
|
|
378
|
+
clusterPrefix = cmdValues["prefix"]
|
|
379
|
+
seqLabel = s
|
|
380
|
+
separator = cmdValues["separator"]
|
|
381
|
+
|
|
382
|
+
if len(networkClusters1)==1:
|
|
383
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
|
|
384
|
+
else:
|
|
385
|
+
if len(networkClusters2)==1:
|
|
386
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
|
|
387
|
+
else:
|
|
388
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}\n")
|
|
389
|
+
|
|
390
|
+
fhandle.close();
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def lineages4Levels(cmdValues):
|
|
394
|
+
np.random.rand(1)
|
|
395
|
+
|
|
396
|
+
snpDistDF = pd.read_csv(cmdValues["data"],delimiter="\t",header=0)
|
|
397
|
+
|
|
398
|
+
Graph = nx.Graph()
|
|
399
|
+
Graph.add_nodes_from(snpDistDF.seq2.to_list())
|
|
400
|
+
Graph.add_nodes_from(snpDistDF.seq1.to_list())
|
|
401
|
+
|
|
402
|
+
snpDistDF1 = snpDistDF[snpDistDF.pid >= cmdValues["thresholds"][0]]
|
|
403
|
+
seqPair = [(snpDistDF1.seq1[i],snpDistDF1.seq2[i]) for i in snpDistDF1.index]
|
|
404
|
+
|
|
405
|
+
Graph.add_edges_from(seqPair)
|
|
406
|
+
networkClusters = tuple(nx.connected_components(Graph))
|
|
407
|
+
|
|
408
|
+
fhandle = open(f"{cmdValues['output']}","w")
|
|
409
|
+
|
|
410
|
+
for i,j in enumerate(networkClusters):
|
|
411
|
+
snpDistDF2 = snpDistDF[snpDistDF.seq1.isin(j) & snpDistDF.seq2.isin(j)]
|
|
412
|
+
snpDistDF3 = snpDistDF2[snpDistDF2.pid >= cmdValues["thresholds"][1]]
|
|
413
|
+
|
|
414
|
+
Graph1 = nx.Graph()
|
|
415
|
+
Graph1.add_nodes_from(j)
|
|
416
|
+
|
|
417
|
+
seqPair1 = [(snpDistDF3.seq1[z],snpDistDF3.seq2[z]) for z in snpDistDF3.index]
|
|
418
|
+
|
|
419
|
+
Graph1.add_edges_from(seqPair1)
|
|
420
|
+
networkClusters1 = tuple(nx.connected_components(Graph1))
|
|
421
|
+
|
|
422
|
+
for p,q in enumerate(networkClusters1):
|
|
423
|
+
snpDistDF2R = snpDistDF[snpDistDF.seq1.isin(q) & snpDistDF.seq2.isin(q)]
|
|
424
|
+
snpDistDF3R = snpDistDF2R[snpDistDF2R.pid >= cmdValues["thresholds"][2]]
|
|
425
|
+
|
|
426
|
+
Graph2 = nx.Graph()
|
|
427
|
+
Graph2.add_nodes_from(q)
|
|
428
|
+
|
|
429
|
+
seqPair1R = [(snpDistDF3R.seq1[z],snpDistDF3R.seq2[z]) for z in snpDistDF3R.index]
|
|
430
|
+
|
|
431
|
+
Graph2.add_edges_from(seqPair1R)
|
|
432
|
+
networkClusters2 = tuple(nx.connected_components(Graph2))
|
|
433
|
+
|
|
434
|
+
for x,y in enumerate(networkClusters2):
|
|
435
|
+
|
|
436
|
+
snpDistDF2RR = snpDistDF[snpDistDF.seq1.isin(y) & snpDistDF.seq2.isin(y)]
|
|
437
|
+
snpDistDF3RR = snpDistDF2RR[snpDistDF2RR.pid >= cmdValues["thresholds"][3]]
|
|
438
|
+
|
|
439
|
+
Graph3 = nx.Graph()
|
|
440
|
+
Graph3.add_nodes_from(y)
|
|
441
|
+
|
|
442
|
+
seqPair1RR = [(snpDistDF3RR.seq1[z],snpDistDF3RR.seq2[z]) for z in snpDistDF3RR.index]
|
|
443
|
+
|
|
444
|
+
Graph3.add_edges_from(seqPair1RR)
|
|
445
|
+
networkClusters3 = tuple(nx.connected_components(Graph3))
|
|
446
|
+
|
|
447
|
+
for r,k in enumerate(networkClusters3):
|
|
448
|
+
|
|
449
|
+
for s in k:
|
|
450
|
+
FirstLevelLin = str(i+1)
|
|
451
|
+
SecondLevelLin = str(p+1)
|
|
452
|
+
ThirdLevelLin = str(x+1)
|
|
453
|
+
FourthLevelLin = str(r+1)
|
|
454
|
+
clusterPrefix = cmdValues["prefix"]
|
|
455
|
+
seqLabel = s
|
|
456
|
+
separator = cmdValues["separator"]
|
|
457
|
+
|
|
458
|
+
if len(networkClusters1)==1:
|
|
459
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}\n")
|
|
460
|
+
else:
|
|
461
|
+
if len(networkClusters2)==1:
|
|
462
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}\n")
|
|
463
|
+
else:
|
|
464
|
+
if len(networkClusters3)==1:
|
|
465
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}\n")
|
|
466
|
+
else:
|
|
467
|
+
fhandle.write(f"{seqLabel}\t{clusterPrefix}{separator}{FirstLevelLin}.{SecondLevelLin}.{ThirdLevelLin}.{FourthLevelLin}\n")
|
|
468
|
+
|
|
469
|
+
fhandle.close();
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
def cluster(cmdValues):
|
|
473
|
+
for item in cmdValues['thresholds']:
|
|
474
|
+
if float(item)<0 or float(item)>100:
|
|
475
|
+
print("Clustering thresholds should be between 0 and 100")
|
|
476
|
+
sys.exit()
|
|
477
|
+
|
|
478
|
+
else:
|
|
479
|
+
pass
|
|
480
|
+
|
|
481
|
+
for itemPos,item in enumerate(cmdValues['thresholds']):
|
|
482
|
+
if itemPos == 0:
|
|
483
|
+
pass
|
|
484
|
+
else:
|
|
485
|
+
if cmdValues['thresholds'][itemPos-1]<cmdValues['thresholds'][itemPos]:
|
|
486
|
+
pass
|
|
487
|
+
else:
|
|
488
|
+
print("Clustering thresholds should be provided in increasing order... exiting...")
|
|
489
|
+
sys.exit()
|
|
490
|
+
|
|
491
|
+
if len(cmdValues['thresholds'])==1:
|
|
492
|
+
lineages1Levels(cmdValues)
|
|
493
|
+
elif len(cmdValues['thresholds'])==2:
|
|
494
|
+
lineages2Levels(cmdValues)
|
|
495
|
+
elif len(cmdValues['thresholds'])==3:
|
|
496
|
+
lineages3Levels(cmdValues)
|
|
497
|
+
elif len(cmdValues['thresholds'])==4:
|
|
498
|
+
lineages4Levels(cmdValues)
|
|
499
|
+
else:
|
|
500
|
+
print("Maximum number of nested should be 4... exiting...")
|
|
501
|
+
sys.exit()
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def order(cmdValues):
|
|
505
|
+
with open(cmdValues["output"],"w") as outputFile:
|
|
506
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
507
|
+
|
|
508
|
+
itemNames = [re.split("\t|,|;",str(item).strip())[0] for item in open(cmdValues["rownames"],"r")]
|
|
509
|
+
|
|
510
|
+
bar = ChargingBar('Saving ordered table', max=len(itemNames))
|
|
511
|
+
for itemNum,itemRowName in enumerate(itemNames):
|
|
512
|
+
if itemNum == 0:
|
|
513
|
+
headerName = list(dataMatrixDF.keys())[0]
|
|
514
|
+
outputFile.write(f"{headerName}\t{"\t".join([str(i) for i in dataMatrixDF[headerName]])}\n")
|
|
515
|
+
else:
|
|
516
|
+
pass
|
|
517
|
+
|
|
518
|
+
if itemRowName in list(dataMatrixDF.keys()):
|
|
519
|
+
outputFile.write(f"{itemRowName}\t{"\t".join([str(i) for i in dataMatrixDF[itemRowName]])}\n")
|
|
520
|
+
|
|
521
|
+
else:
|
|
522
|
+
outputFile.write(f"{itemRowName}\n")
|
|
523
|
+
|
|
524
|
+
bar.next()
|
|
525
|
+
bar.finish()
|
|
526
|
+
|
|
527
|
+
def sample(cmdValues):
|
|
528
|
+
np.random.seed(1)
|
|
529
|
+
|
|
530
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
531
|
+
|
|
532
|
+
rows = list(dataMatrixDF.keys())
|
|
533
|
+
columns = np.arange(len(dataMatrixDF[rows[0]])).tolist()
|
|
534
|
+
|
|
535
|
+
selectedColumns = np.random.choice(columns,int(np.ceil(float(cmdValues["colfreq"])*len(columns))),replace=False).tolist()
|
|
536
|
+
selectedRows = [rows[0]]
|
|
537
|
+
selectedRows.extend(np.random.choice(rows[1:],int(np.ceil(float(cmdValues["rowfreq"])*len(rows[1:]))),replace=False).tolist())
|
|
538
|
+
|
|
539
|
+
with open(cmdValues["output"],"w") as outputFile:
|
|
540
|
+
|
|
541
|
+
bar = ChargingBar('Saving table', max=len(selectedRows)*len(selectedColumns))
|
|
542
|
+
for i,j in enumerate(selectedRows):
|
|
543
|
+
tmpStr = []
|
|
544
|
+
|
|
545
|
+
for r,s in enumerate(selectedColumns):
|
|
546
|
+
if r == 0:
|
|
547
|
+
tmpStr.append(str(j))
|
|
548
|
+
tmpStr.append(str(dataMatrixDF[j][r]))
|
|
549
|
+
else:
|
|
550
|
+
tmpStr.append(str(dataMatrixDF[j][r]))
|
|
551
|
+
|
|
552
|
+
bar.next()
|
|
553
|
+
outputFile.write(f"{"\t".join(tmpStr)}\n")
|
|
554
|
+
|
|
555
|
+
bar.finish()
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def rename(cmdValues):
|
|
559
|
+
with open(cmdValues["output"],"w") as outputFile:
|
|
560
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
561
|
+
|
|
562
|
+
columnIndexValues = []
|
|
563
|
+
|
|
564
|
+
if not os.path.exists(cmdValues["newdata"]):
|
|
565
|
+
print(f"File {cmdValues["newdata"]} does not exist... exiting...")
|
|
566
|
+
sys.exit()
|
|
567
|
+
else:
|
|
568
|
+
pass
|
|
569
|
+
|
|
570
|
+
if not isinstance(cmdValues["columns"],str):
|
|
571
|
+
columnIndexValues = [int(i) for i in [re.split("\t|,|;",str(item).strip()) for item in cmdValues["columns"]][0]]
|
|
572
|
+
|
|
573
|
+
else:
|
|
574
|
+
columnIndexValues = [int(i) for i in cmdValues["columns"]]
|
|
575
|
+
|
|
576
|
+
newRowNameValuesDF = {}
|
|
577
|
+
|
|
578
|
+
tmpCmd="wc -l "+str(cmdValues["newdata"])+" | awk '{print $1}'"
|
|
579
|
+
dataFileLines=subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
|
|
580
|
+
dataLines = int(dataFileLines.communicate()[0])
|
|
581
|
+
dataFileLines.kill()
|
|
582
|
+
|
|
583
|
+
bar = ChargingBar('Loading file with new names', max=int(dataLines))
|
|
584
|
+
with open(cmdValues["newdata"],"r") as newData:
|
|
585
|
+
for rowData in newData:
|
|
586
|
+
newRowNameValues = []
|
|
587
|
+
|
|
588
|
+
if len(rowData.strip().split("\t")) > 0:
|
|
589
|
+
if max(columnIndexValues) < 0 or max(columnIndexValues) > len(rowData.strip().split("\t")):
|
|
590
|
+
print(f"Values for --column/-c should be >0 and <number of columns in the new data table... exiting...")
|
|
591
|
+
sys.exit()
|
|
592
|
+
else:
|
|
593
|
+
pass
|
|
594
|
+
else:
|
|
595
|
+
pass
|
|
596
|
+
|
|
597
|
+
for columnIndex in columnIndexValues:
|
|
598
|
+
newRowNameValues.append(rowData.strip().split("\t")[columnIndex-1])
|
|
599
|
+
|
|
600
|
+
newRowNameValuesDF[rowData.strip().split("\t")[0]] = "_".join(newRowNameValues)
|
|
601
|
+
|
|
602
|
+
bar.next()
|
|
603
|
+
bar.finish()
|
|
604
|
+
|
|
605
|
+
bar = ChargingBar('Saving renamed table', max=len(dataMatrixDF.keys()))
|
|
606
|
+
for rowIndex,rowName in enumerate(dataMatrixDF.keys()):
|
|
607
|
+
|
|
608
|
+
if rowIndex == 0:
|
|
609
|
+
outputFile.write(f"{rowName}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
|
|
610
|
+
else:
|
|
611
|
+
if rowName in newRowNameValuesDF.keys():
|
|
612
|
+
outputFile.write(f"{newRowNameValuesDF[rowName]}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
|
|
613
|
+
|
|
614
|
+
else:
|
|
615
|
+
outputFile.write(f"{rowName}\t{"\t".join([str(i) for i in dataMatrixDF[rowName]])}\n")
|
|
616
|
+
|
|
617
|
+
bar.next()
|
|
618
|
+
bar.finish()
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
def filter(cmdValues):
|
|
622
|
+
if cmdValues["minCutoff"]>=0 and cmdValues["maxCutoff"]<=100:
|
|
623
|
+
if cmdValues["minCutoff"] > cmdValues["maxCutoff"]:
|
|
624
|
+
print(f"--->>>Value of --min/-a should be less than or equal to value of --max/-b... exiting...")
|
|
625
|
+
sys.exit()
|
|
626
|
+
else:
|
|
627
|
+
pass
|
|
628
|
+
|
|
629
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
630
|
+
|
|
631
|
+
tmpMatrixData = {}
|
|
632
|
+
|
|
633
|
+
with open(cmdValues["output"],"w") as outputFile:
|
|
634
|
+
columnValues = []
|
|
635
|
+
|
|
636
|
+
bar = ChargingBar('Filtering table', max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
|
|
637
|
+
for itemPos,itemValue in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
|
|
638
|
+
itemMatrixColValues = []
|
|
639
|
+
|
|
640
|
+
for tmpPos,itemMatrixName in enumerate(dataMatrixDF.keys()):
|
|
641
|
+
if tmpPos != 0:
|
|
642
|
+
itemMatrixColValues.append(dataMatrixDF[itemMatrixName][itemPos])
|
|
643
|
+
else:
|
|
644
|
+
pass
|
|
645
|
+
|
|
646
|
+
rowMatrixSum = sum(itemMatrixColValues)
|
|
647
|
+
|
|
648
|
+
if rowMatrixSum >= (cmdValues["minCutoff"]/100)*len(itemMatrixColValues) and \
|
|
649
|
+
rowMatrixSum <= (cmdValues["maxCutoff"]/100)*len(itemMatrixColValues):
|
|
650
|
+
|
|
651
|
+
columnValues.append(itemPos)
|
|
652
|
+
else:
|
|
653
|
+
pass
|
|
654
|
+
|
|
655
|
+
bar.next()
|
|
656
|
+
bar.finish()
|
|
657
|
+
|
|
658
|
+
bar = ChargingBar('Saving filtered table', max=len(list(dataMatrixDF.keys())))
|
|
659
|
+
for itemMatrixName in dataMatrixDF.keys():
|
|
660
|
+
tmpColData = []
|
|
661
|
+
|
|
662
|
+
for colPosition in columnValues:
|
|
663
|
+
tmpColData.append(str(dataMatrixDF[itemMatrixName][colPosition]))
|
|
664
|
+
|
|
665
|
+
bar.next()
|
|
666
|
+
outputFile.write(f"{str(itemMatrixName)}\t{"\t".join(tmpColData)}\n")
|
|
667
|
+
bar.finish()
|
|
668
|
+
|
|
669
|
+
else:
|
|
670
|
+
if cmdValues["minCutoff"] > cmdValues["maxCutoff"]:
|
|
671
|
+
print(f"--->>>Value of --min/-a should be less than or equal to value of --max/-b... exiting...")
|
|
672
|
+
sys.exit()
|
|
673
|
+
else:
|
|
674
|
+
print(f"--->>>Value of --min/-a snd --max/-b values should be between 0 and 100... exiting...")
|
|
675
|
+
sys.exit()
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def makeMapPED(i,j,k,l):
|
|
679
|
+
if i == 0:
|
|
680
|
+
pass
|
|
681
|
+
else:
|
|
682
|
+
tmpStr = ""
|
|
683
|
+
|
|
684
|
+
for m,n in enumerate(k):
|
|
685
|
+
if m == 0:
|
|
686
|
+
tmpStr = f"{str(j)}\t{str(j)}\t0\t0\t1\t{str(l)}"
|
|
687
|
+
|
|
688
|
+
else:
|
|
689
|
+
tmpN = 0
|
|
690
|
+
if(int(n) == 0):
|
|
691
|
+
tmpN = 2
|
|
692
|
+
else:
|
|
693
|
+
tmpN = 1
|
|
694
|
+
|
|
695
|
+
tmpStr = f"{tmpStr}\t{str(tmpN)} {str(tmpN)}"
|
|
696
|
+
|
|
697
|
+
print('.', end='', flush=True)
|
|
698
|
+
|
|
699
|
+
return(tmpStr)
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
def gwas(cmdValues):
|
|
703
|
+
checkFlags = [1 for flag in [cmdValues["pyseer"],cmdValues["plink"],cmdValues["gemma"],cmdValues["fastlmm"]] if flag == True]
|
|
704
|
+
|
|
705
|
+
if sum(checkFlags) > 1:
|
|
706
|
+
print(f"Only one these flags should be specified: --pyseer/-s, --plink/-l, --gemma/g, and --fastlmm/-m... exiting...")
|
|
707
|
+
sys.exit()
|
|
708
|
+
else:
|
|
709
|
+
if sum(checkFlags) == 0:
|
|
710
|
+
cmdValues["plink"] = True
|
|
711
|
+
|
|
712
|
+
else:
|
|
713
|
+
pass
|
|
714
|
+
|
|
715
|
+
dataMatrixDF = loadDataTable(cmdValues)
|
|
716
|
+
|
|
717
|
+
phenoData={}
|
|
718
|
+
|
|
719
|
+
if os.path.exists(cmdValues["phenotype"]):
|
|
720
|
+
phenoDataTmp = [str(item).strip() for item in open(cmdValues["phenotype"],"r")]
|
|
721
|
+
|
|
722
|
+
phenoData = {}
|
|
723
|
+
|
|
724
|
+
bar = ChargingBar('Processing phenotype data', max=len(phenoDataTmp))
|
|
725
|
+
for m in phenoDataTmp:
|
|
726
|
+
tmpR = re.split("\t|,|;",str(m).strip())
|
|
727
|
+
phenoData[tmpR[0]]=tmpR[1]
|
|
728
|
+
|
|
729
|
+
bar.next()
|
|
730
|
+
bar.finish()
|
|
731
|
+
|
|
732
|
+
else:
|
|
733
|
+
print(f"Phenotype file {cmdValues["phenotype"]} not found... exiting...")
|
|
734
|
+
sys.exit()
|
|
735
|
+
|
|
736
|
+
if cmdValues["plink"]:
|
|
737
|
+
with open(f"{Path(cmdValues["output"]).stem}.plink.map","w") as mapFile:
|
|
738
|
+
with open(f"{Path(cmdValues["output"]).stem}.plink.ped","w") as pedFile:
|
|
739
|
+
|
|
740
|
+
tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
|
|
741
|
+
|
|
742
|
+
with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
|
|
743
|
+
args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
|
|
744
|
+
resultsMapPED = pool.starmap(makeMapPED, args)
|
|
745
|
+
|
|
746
|
+
bar = ChargingBar("Saving PLINK PED file", max=len(list(dataMatrixDF.keys())))
|
|
747
|
+
for eachResult in resultsMapPED:
|
|
748
|
+
pedFile.write(f"{eachResult}\n")
|
|
749
|
+
|
|
750
|
+
bar.next()
|
|
751
|
+
bar.finish()
|
|
752
|
+
|
|
753
|
+
bar = ChargingBar("Saving PLINK MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
|
|
754
|
+
for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
|
|
755
|
+
if k != 0:
|
|
756
|
+
mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
|
|
757
|
+
|
|
758
|
+
else:
|
|
759
|
+
pass
|
|
760
|
+
|
|
761
|
+
bar.next()
|
|
762
|
+
bar.finish()
|
|
763
|
+
|
|
764
|
+
elif cmdValues["fastlmm"]:
|
|
765
|
+
with open(f"{Path(cmdValues["output"]).stem}.fastlmm.map","w") as mapFile:
|
|
766
|
+
with open(f"{Path(cmdValues["output"]).stem}.fastlmm.ped","w") as pedFile:
|
|
767
|
+
|
|
768
|
+
tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
|
|
769
|
+
|
|
770
|
+
with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
|
|
771
|
+
args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
|
|
772
|
+
resultsMapPED = pool.starmap(makeMapPED, args)
|
|
773
|
+
|
|
774
|
+
bar = ChargingBar("Saving FASTLMM PED file", max=len(list(dataMatrixDF.keys())))
|
|
775
|
+
for eachResult in resultsMapPED:
|
|
776
|
+
pedFile.write(f"{eachResult}\n")
|
|
777
|
+
|
|
778
|
+
bar.next()
|
|
779
|
+
bar.finish()
|
|
780
|
+
|
|
781
|
+
|
|
782
|
+
bar = ChargingBar("Saving FASTLMM MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
|
|
783
|
+
for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
|
|
784
|
+
if k != 0:
|
|
785
|
+
mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
|
|
786
|
+
|
|
787
|
+
else:
|
|
788
|
+
pass
|
|
789
|
+
|
|
790
|
+
bar.next()
|
|
791
|
+
bar.finish()
|
|
792
|
+
|
|
793
|
+
elif cmdValues["gemma"]:
|
|
794
|
+
with open(f"{Path(cmdValues["output"]).stem}.plink.map","w") as mapFile:
|
|
795
|
+
with open(f"{Path(cmdValues["output"]).stem}.plink.ped","w") as pedFile:
|
|
796
|
+
|
|
797
|
+
tmpItems = [(i,j) for i,j in enumerate(dataMatrixDF.keys()) if i>0]
|
|
798
|
+
|
|
799
|
+
with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
|
|
800
|
+
args = [(i,j,dataMatrixDF[j],phenoData[j]) for i,j in tmpItems]
|
|
801
|
+
resultsMapPED = pool.starmap(makeMapPED, args)
|
|
802
|
+
|
|
803
|
+
bar = ChargingBar("Saving PLINK PED file", max=len(list(dataMatrixDF.keys())))
|
|
804
|
+
for eachResult in resultsMapPED:
|
|
805
|
+
pedFile.write(f"{eachResult}\n")
|
|
806
|
+
|
|
807
|
+
bar.next()
|
|
808
|
+
bar.finish()
|
|
809
|
+
|
|
810
|
+
bar = ChargingBar("Saving PLINK MAP file", max=len(dataMatrixDF[list(dataMatrixDF.keys())[0]]))
|
|
811
|
+
for k,l in enumerate(dataMatrixDF[list(dataMatrixDF.keys())[0]]):
|
|
812
|
+
if k != 0:
|
|
813
|
+
mapFile.write(f"26\t{str(l)}\t0\t{str(k)}\n")
|
|
814
|
+
|
|
815
|
+
else:
|
|
816
|
+
pass
|
|
817
|
+
|
|
818
|
+
bar.next()
|
|
819
|
+
bar.finish()
|
|
820
|
+
|
|
821
|
+
if not shutil.which("plink"):
|
|
822
|
+
print("Install 'plink' software dependency and try again... exiting...")
|
|
823
|
+
sys.exit()
|
|
824
|
+
else:
|
|
825
|
+
pass
|
|
826
|
+
|
|
827
|
+
bar = ChargingBar("Generating GEMMA binary PED files", max=1)
|
|
828
|
+
plinkCmd = f"plink -file {f"{Path(cmdValues["output"]).stem}.plink"} -make-bed -out {f"{Path(cmdValues["output"]).stem}.gemma"} --threads {cmdValues["threads"]}"
|
|
829
|
+
|
|
830
|
+
subprocess.call(plinkCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
831
|
+
bar.next()
|
|
832
|
+
bar.finish()
|
|
833
|
+
|
|
834
|
+
os.remove(f"{Path(cmdValues["output"]).stem}.plink.map")
|
|
835
|
+
os.remove(f"{Path(cmdValues["output"]).stem}.plink.ped")
|
|
836
|
+
os.remove(f"{Path(cmdValues["output"]).stem}.gemma.log")
|
|
837
|
+
|
|
838
|
+
else:
|
|
839
|
+
with open(f"{Path(cmdValues["output"]).stem}.pyseer.tsv","w") as pyseerTab:
|
|
840
|
+
with open(f"{Path(cmdValues["output"]).stem}.pyseer.pheno.tsv","w") as pyseerPheno:
|
|
841
|
+
|
|
842
|
+
transpose(cmdValues)
|
|
843
|
+
|
|
844
|
+
transposedFileName = f"{Path(cmdValues["output"]).stem}.transposed.tsv"
|
|
845
|
+
pyseerFileName = f"{Path(cmdValues["output"]).stem}.pyseer.tsv"
|
|
846
|
+
|
|
847
|
+
shutil.move(transposedFileName,pyseerFileName)
|
|
848
|
+
|
|
849
|
+
bar = ChargingBar("Generating PYSEER files", max=len(phenoData.keys()))
|
|
850
|
+
for itemPos,item in enumerate(phenoData.keys()):
|
|
851
|
+
if itemPos == 0:
|
|
852
|
+
pyseerPheno.write("samples\tphenotype\n")
|
|
853
|
+
else:
|
|
854
|
+
pyseerPheno.write(f"{item}\t{phenoData[item]}\n")
|
|
855
|
+
|
|
856
|
+
bar.next()
|
|
857
|
+
bar.finish()
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def queryKmers(seqNum,seqFile,kmerFile):
|
|
861
|
+
seqFileName = Path(os.path.basename(seqFile)).stem
|
|
862
|
+
tmpOutFileA = f"bifrost.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
863
|
+
|
|
864
|
+
tmpCmd1 = f"Bifrost build -t 3 -k 31 -i -d -r {seqFile} -o {tmpOutFileA}.graph"
|
|
865
|
+
subprocess.call(tmpCmd1,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
866
|
+
|
|
867
|
+
tmpCmd2 = f"Bifrost query -t 5 -e 1 -q {kmerFile} -g {tmpOutFileA}.graph.gfa.gz -o {tmpOutFileA}"
|
|
868
|
+
subprocess.call(tmpCmd2,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
869
|
+
|
|
870
|
+
tmpCmd3 = f"echo {seqFileName} > {tmpOutFileA}.out.tsv"
|
|
871
|
+
subprocess.call(tmpCmd3,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
872
|
+
|
|
873
|
+
tmpCmd4 = f"grep -v query_name {tmpOutFileA}.tsv | sort -k1 -n > {tmpOutFileA}.out.tmp.tsv"
|
|
874
|
+
subprocess.call(tmpCmd4,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
875
|
+
|
|
876
|
+
tmpCmd5 = "awk '{print $2}' "+f"{tmpOutFileA}.out.tmp.tsv >> {tmpOutFileA}.out.tsv"
|
|
877
|
+
subprocess.call(tmpCmd5,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
878
|
+
|
|
879
|
+
tmpCmd6 = f"echo KmerID > {tmpOutFileA}.head.tsv"
|
|
880
|
+
subprocess.call(tmpCmd6,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
881
|
+
|
|
882
|
+
tmpCmd7 = "awk '{print $1}' "+f"{tmpOutFileA}.out.tmp.tsv | sed \"s:query_name:KmerID:g\" >> {tmpOutFileA}.head.tsv";
|
|
883
|
+
subprocess.call(tmpCmd7,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
884
|
+
|
|
885
|
+
tmpCmd8 = f"datamash transpose < {tmpOutFileA}.head.tsv > {tmpOutFileA}.header.tsv"
|
|
886
|
+
subprocess.call(tmpCmd8,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
887
|
+
|
|
888
|
+
tmpCmd9 = f"datamash transpose < {tmpOutFileA}.out.tsv > {tmpOutFileA}.tmp.tsv"
|
|
889
|
+
subprocess.call(tmpCmd9,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
890
|
+
|
|
891
|
+
kmerData = {seqFileName: {"kmerFile": [str(i).strip() for i in open(f"{tmpOutFileA}.tmp.tsv","r")][0],"kmerHeaderFile": [str(j).strip() for j in open(f"{tmpOutFileA}.header.tsv","r")][0] }}
|
|
892
|
+
|
|
893
|
+
if os.path.exists(f"{tmpOutFileA}.graph.bfi"):
|
|
894
|
+
os.remove(f"{tmpOutFileA}.graph.bfi")
|
|
895
|
+
else:
|
|
896
|
+
pass
|
|
897
|
+
|
|
898
|
+
if os.path.exists(f"{tmpOutFileA}.graph.gfa.gz"):
|
|
899
|
+
os.remove(f"{tmpOutFileA}.graph.gfa.gz")
|
|
900
|
+
else:
|
|
901
|
+
pass
|
|
902
|
+
|
|
903
|
+
if os.path.exists(f"{tmpOutFileA}.out.tsv"):
|
|
904
|
+
os.remove(f"{tmpOutFileA}.out.tsv")
|
|
905
|
+
else:
|
|
906
|
+
pass
|
|
907
|
+
|
|
908
|
+
if os.path.exists(f"{tmpOutFileA}.tsv"):
|
|
909
|
+
os.remove(f"{tmpOutFileA}.tsv")
|
|
910
|
+
else:
|
|
911
|
+
pass
|
|
912
|
+
|
|
913
|
+
if os.path.exists(f"{tmpOutFileA}.out.tmp.tsv"):
|
|
914
|
+
os.remove(f"{tmpOutFileA}.out.tmp.tsv")
|
|
915
|
+
else:
|
|
916
|
+
pass
|
|
917
|
+
|
|
918
|
+
if os.path.exists(f"{tmpOutFileA}.out.tsv"):
|
|
919
|
+
os.remove(f"{tmpOutFileA}.out.tsv")
|
|
920
|
+
else:
|
|
921
|
+
pass
|
|
922
|
+
|
|
923
|
+
if os.path.exists(f"{tmpOutFileA}.head.tsv"):
|
|
924
|
+
os.remove(f"{tmpOutFileA}.head.tsv")
|
|
925
|
+
else:
|
|
926
|
+
pass
|
|
927
|
+
|
|
928
|
+
if os.path.exists(f"{tmpOutFileA}.header.tsv"):
|
|
929
|
+
os.remove(f"{tmpOutFileA}.header.tsv")
|
|
930
|
+
else:
|
|
931
|
+
pass
|
|
932
|
+
|
|
933
|
+
if os.path.exists(f"{tmpOutFileA}.tmp.tsv"):
|
|
934
|
+
os.remove(f"{tmpOutFileA}.tmp.tsv")
|
|
935
|
+
else:
|
|
936
|
+
pass
|
|
937
|
+
|
|
938
|
+
return(kmerData)
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def table(cmdValues):
|
|
942
|
+
if not os.path.exists(cmdValues["data"]):
|
|
943
|
+
print(f"File {cmdValues["data"]} not found... exiting...")
|
|
944
|
+
sys.exit()
|
|
945
|
+
else:
|
|
946
|
+
pass
|
|
947
|
+
|
|
948
|
+
if not shutil.which("awk"):
|
|
949
|
+
print("Install awk and try again... exiting...")
|
|
950
|
+
sys.exit()
|
|
951
|
+
else:
|
|
952
|
+
pass
|
|
953
|
+
|
|
954
|
+
if not shutil.which("dsk") and not shutil.which("dsk2ascii"):
|
|
955
|
+
print("Install DSK and try again... exiting...")
|
|
956
|
+
sys.exit()
|
|
957
|
+
else:
|
|
958
|
+
pass
|
|
959
|
+
|
|
960
|
+
if not shutil.which("Bifrost"):
|
|
961
|
+
print("Install Bifrost and try again... exiting...")
|
|
962
|
+
sys.exit()
|
|
963
|
+
else:
|
|
964
|
+
pass
|
|
965
|
+
|
|
966
|
+
if not shutil.which("datamash"):
|
|
967
|
+
print("Install datamash and try again... exiting...")
|
|
968
|
+
sys.exit()
|
|
969
|
+
else:
|
|
970
|
+
pass
|
|
971
|
+
|
|
972
|
+
kmersFile = ""
|
|
973
|
+
|
|
974
|
+
if cmdValues["kmers"] != "":
|
|
975
|
+
kmersFile = cmdValues["kmers"]
|
|
976
|
+
else:
|
|
977
|
+
kmersFile = f"{Path(cmdValues["output"]).stem}.fasta"
|
|
978
|
+
|
|
979
|
+
with open(cmdValues["data"],"r") as seqFile:
|
|
980
|
+
|
|
981
|
+
tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
|
|
982
|
+
dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
|
|
983
|
+
dataLines = int(dataFileLines.communicate()[0])
|
|
984
|
+
dataFileLines.kill()
|
|
985
|
+
|
|
986
|
+
bar = ChargingBar("Generating k-mers (DSK)", max=int(dataLines))
|
|
987
|
+
for fileNameValue in seqFile:
|
|
988
|
+
fileName = fileNameValue.strip()
|
|
989
|
+
|
|
990
|
+
if os.path.exists(fileName):
|
|
991
|
+
if os.path.isfile(fileName) and os.path.getsize(fileName) > 0:
|
|
992
|
+
pass
|
|
993
|
+
else:
|
|
994
|
+
print(f"File {fileName} in {cmdValues["data"]} is empty... exiting...")
|
|
995
|
+
sys.exit()
|
|
996
|
+
else:
|
|
997
|
+
print(f"File {fileName} in {cmdValues["data"]} does not exist... exiting...")
|
|
998
|
+
sys.exit()
|
|
999
|
+
|
|
1000
|
+
bar.next()
|
|
1001
|
+
bar.finish()
|
|
1002
|
+
|
|
1003
|
+
if cmdValues["dsk"] and not cmdValues["bifrost"]:
|
|
1004
|
+
|
|
1005
|
+
if cmdValues["kmers"] == "":
|
|
1006
|
+
bar = ChargingBar("Generating k-mers (DSK)", max=2)
|
|
1007
|
+
dskCmdTmpFile = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
1008
|
+
|
|
1009
|
+
dsk1Cmd = f"dsk -kmer-size {cmdValues["length"]} -abundance-min {cmdValues["minabundance"]} -max-memory {cmdValues["maxmemory"]} -file {cmdValues["data"]} -out {dskCmdTmpFile} -nb-cores {cmdValues["threads"]}"
|
|
1010
|
+
subprocess.call(dsk1Cmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
1011
|
+
bar.next()
|
|
1012
|
+
|
|
1013
|
+
tmpKmersFileR = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
1014
|
+
|
|
1015
|
+
dsk2Cmd = f"dsk2ascii -file {dskCmdTmpFile}.h5 -out {tmpKmersFileR} -nb-cores {cmdValues["threads"]}"
|
|
1016
|
+
subprocess.call(dsk2Cmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
1017
|
+
bar.next()
|
|
1018
|
+
|
|
1019
|
+
awkCmd = "awk '{print \">\"NR\"\\n\"$1}' "+str(tmpKmersFileR)+" > "+str(kmersFile)
|
|
1020
|
+
subprocess.call(awkCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
1021
|
+
bar.next()
|
|
1022
|
+
bar.finish()
|
|
1023
|
+
|
|
1024
|
+
if os.path.exists(f"{dskCmdTmpFile}.h5"):
|
|
1025
|
+
os.remove(f"{dskCmdTmpFile}.h5")
|
|
1026
|
+
else:
|
|
1027
|
+
pass
|
|
1028
|
+
|
|
1029
|
+
if os.path.exists(f"{dskCmdTmpFile}"):
|
|
1030
|
+
os.remove(f"{dskCmdTmpFile}")
|
|
1031
|
+
else:
|
|
1032
|
+
pass
|
|
1033
|
+
|
|
1034
|
+
if os.path.exists(f"{dskCmdTmpFile}.txt"):
|
|
1035
|
+
os.remove(f"{dskCmdTmpFile}.txt")
|
|
1036
|
+
else:
|
|
1037
|
+
pass
|
|
1038
|
+
|
|
1039
|
+
if os.path.exists(f"{tmpKmersFileR}.txt"):
|
|
1040
|
+
os.remove(f"{tmpKmersFileR}.txt")
|
|
1041
|
+
else:
|
|
1042
|
+
pass
|
|
1043
|
+
|
|
1044
|
+
if os.path.exists(f"{tmpKmersFileR}"):
|
|
1045
|
+
os.remove(f"{tmpKmersFileR}")
|
|
1046
|
+
else:
|
|
1047
|
+
pass
|
|
1048
|
+
|
|
1049
|
+
else:
|
|
1050
|
+
pass
|
|
1051
|
+
|
|
1052
|
+
else:
|
|
1053
|
+
if cmdValues["kmers"] != "":
|
|
1054
|
+
tmpKmersFile = f"KMERS.{str(datetime.datetime.now()).replace(" ","").replace("-",".").replace(":",".")}"
|
|
1055
|
+
|
|
1056
|
+
bar = ChargingBar("Generating unitigs (Bifrost)", max=1)
|
|
1057
|
+
bifrostCmd = f"Bifrost build -t {cmdValues["threads"]} -k {cmdValues["length"]} -i -d -r {cmdValues["data"]} -o {tmpKmersFile} -v -a -f -n"
|
|
1058
|
+
|
|
1059
|
+
subprocess.call(bifrostCmd,shell=True,stdout=subprocess.DEVNULL,stderr=subprocess.STDOUT)
|
|
1060
|
+
bar.next()
|
|
1061
|
+
bar.finish()
|
|
1062
|
+
|
|
1063
|
+
with open(kmersFile,"w") as kmerData:
|
|
1064
|
+
for seqNum,seq in enumerate(SeqIO.parse(f"{tmpKmersFile}.fasta","fasta")):
|
|
1065
|
+
kmerData.write(f">{(seqNum+1)}\n{seq.seq}\n")
|
|
1066
|
+
|
|
1067
|
+
if os.path.exists(f"{tmpKmersFile}.bfi"):
|
|
1068
|
+
os.remove(f"{tmpKmersFile}.bfi")
|
|
1069
|
+
else:
|
|
1070
|
+
pass
|
|
1071
|
+
|
|
1072
|
+
if os.path.exists(f"{tmpKmersFile}.fasta"):
|
|
1073
|
+
os.remove(f"{tmpKmersFile}.fasta")
|
|
1074
|
+
else:
|
|
1075
|
+
pass
|
|
1076
|
+
|
|
1077
|
+
else:
|
|
1078
|
+
pass
|
|
1079
|
+
|
|
1080
|
+
tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
|
|
1081
|
+
dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
|
|
1082
|
+
dataLines = int(dataFileLines.communicate()[0])
|
|
1083
|
+
dataFileLines.kill()
|
|
1084
|
+
|
|
1085
|
+
dataMatrixDF = []
|
|
1086
|
+
|
|
1087
|
+
bar = ChargingBar("Processing sequence files", max=int(dataLines))
|
|
1088
|
+
for seqFile in open(cmdValues["data"],"r"):
|
|
1089
|
+
dataMatrixDF.append(str(seqFile).strip().split("\t")[0])
|
|
1090
|
+
|
|
1091
|
+
bar.next()
|
|
1092
|
+
bar.finish()
|
|
1093
|
+
|
|
1094
|
+
seqFileDF = [(pos,item) for pos,item in enumerate(dataMatrixDF)]
|
|
1095
|
+
|
|
1096
|
+
print("Query k-mers against sequences... may take longer... be patient...")
|
|
1097
|
+
|
|
1098
|
+
with multiprocessing.Pool(processes=cmdValues["threads"]) as pool:
|
|
1099
|
+
args = [(seqFileNum,seqFileName,kmersFile) for seqFileNum,seqFileName in seqFileDF]
|
|
1100
|
+
kmerResults = pool.starmap(queryKmers, args)
|
|
1101
|
+
|
|
1102
|
+
with open(f"{Path(cmdValues["output"]).stem}.table.tsv","w") as kmerData:
|
|
1103
|
+
kmerFiles = []
|
|
1104
|
+
kmerHeaderFiles = []
|
|
1105
|
+
|
|
1106
|
+
bar = ChargingBar("Writing k-mers table", max=len(kmerResults))
|
|
1107
|
+
for eachResultPos,eachResult in enumerate(kmerResults):
|
|
1108
|
+
seqName = list(eachResult.keys())[0]
|
|
1109
|
+
|
|
1110
|
+
kmerFiles.append(eachResult[seqName])
|
|
1111
|
+
kmerHeaderFiles.append(eachResult[seqName])
|
|
1112
|
+
|
|
1113
|
+
if eachResultPos == 0:
|
|
1114
|
+
kmerData.write(f"{eachResult[seqName]["kmerHeaderFile"]}\n")
|
|
1115
|
+
kmerData.write(f"{eachResult[seqName]["kmerFile"]}\n")
|
|
1116
|
+
else:
|
|
1117
|
+
kmerData.write(f"{eachResult[seqName]["kmerFile"]}\n")
|
|
1118
|
+
|
|
1119
|
+
bar.next()
|
|
1120
|
+
bar.finish()
|
|
1121
|
+
|
|
1122
|
+
|
|
1123
|
+
def loadDataTable(cmdValues):
|
|
1124
|
+
if not (os.path.exists(cmdValues["data"]) and os.path.isfile(cmdValues["data"])):
|
|
1125
|
+
print(f"--->>>Input file {cmdValues["data"]} does not exist... exiting...")
|
|
1126
|
+
sys.exit()
|
|
1127
|
+
|
|
1128
|
+
else:
|
|
1129
|
+
dataMatrixDF = {}
|
|
1130
|
+
|
|
1131
|
+
sepFormat = "\t" if cmdValues["format"] == "tsv" else ","
|
|
1132
|
+
|
|
1133
|
+
if not (os.path.exists(cmdValues["data"]) and os.path.isfile(cmdValues["data"])):
|
|
1134
|
+
print(f"--->>>Input file {cmdValues["data"]} does not exist... exiting...")
|
|
1135
|
+
sys.exit()
|
|
1136
|
+
else:
|
|
1137
|
+
with open(cmdValues["data"],"r") as dataFile:
|
|
1138
|
+
|
|
1139
|
+
if not shutil.which("awk"):
|
|
1140
|
+
print("Install awk and try again... exiting...")
|
|
1141
|
+
sys.exit()
|
|
1142
|
+
else:
|
|
1143
|
+
pass
|
|
1144
|
+
|
|
1145
|
+
tmpCmd="wc -l "+str(cmdValues["data"])+" | awk '{print $1}'"
|
|
1146
|
+
dataFileLines = subprocess.Popen(tmpCmd,shell=True, stderr=subprocess.STDOUT, stdout=subprocess.PIPE)
|
|
1147
|
+
dataLines = int(dataFileLines.communicate()[0])
|
|
1148
|
+
dataFileLines.kill()
|
|
1149
|
+
|
|
1150
|
+
bar = ChargingBar('Loading data table', max=int(dataLines))
|
|
1151
|
+
for dataPos,dataRow in enumerate(dataFile):
|
|
1152
|
+
if len(dataRow.strip().split(sepFormat)) == 0:
|
|
1153
|
+
pass
|
|
1154
|
+
else:
|
|
1155
|
+
if dataPos == 0:
|
|
1156
|
+
dataRowValues = dataRow.strip().split(sepFormat)
|
|
1157
|
+
dataRowValuesName = dataRowValues[0]
|
|
1158
|
+
dataRowValuesCols = [item for item in dataRowValues[1:]]
|
|
1159
|
+
dataMatrixDF[dataRowValuesName] = dataRowValuesCols
|
|
1160
|
+
else:
|
|
1161
|
+
dataRowValues = dataRow.strip().split(sepFormat)
|
|
1162
|
+
dataRowValuesName = dataRowValues[0]
|
|
1163
|
+
dataRowValuesCols = [int(item) for item in dataRowValues[1:]]
|
|
1164
|
+
dataMatrixDF[dataRowValuesName] = dataRowValuesCols
|
|
1165
|
+
|
|
1166
|
+
if set(dataRowValuesCols) == {0} or set(dataRowValuesCols) == {1} or set(dataRowValuesCols) == {0,1}:
|
|
1167
|
+
pass
|
|
1168
|
+
else:
|
|
1169
|
+
print(f"--->>>Row {dataPos} in {cmdValues["data"]} contains unexpected other values besides 0 and 1... exiting...")
|
|
1170
|
+
sys.exit()
|
|
1171
|
+
|
|
1172
|
+
bar.next()
|
|
1173
|
+
bar.finish()
|
|
1174
|
+
|
|
1175
|
+
return(dataMatrixDF)
|
|
1176
|
+
|
|
1177
|
+
|
|
1178
|
+
def citation():
|
|
1179
|
+
print(f"Chrispin Chaguza. 2026. matrixToSimilarity: calculating percent similarity between pairs of items (rows) \
|
|
1180
|
+
given tabular text file. https://github.com/ChrispinChaguza/matrixToSimilarity")
|
|
1181
|
+
|
|
1182
|
+
|
|
1183
|
+
def printVersion():
|
|
1184
|
+
print(f"tabkit {__version__}")
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def tabtkMain():
|
|
1188
|
+
|
|
1189
|
+
optionsCmd = argparse.ArgumentParser(sys.argv[0],
|
|
1190
|
+
usage=argparse.SUPPRESS,
|
|
1191
|
+
description='tabtk: A toolkit for sequence similarity analysis based on tabular data',
|
|
1192
|
+
prefix_chars='-',
|
|
1193
|
+
add_help=True,
|
|
1194
|
+
epilog='Written by Chrispin Chaguza, St Jude Children\'s Research Hospital, 2026')
|
|
1195
|
+
|
|
1196
|
+
optionsCommand = optionsCmd.add_subparsers(dest="command")
|
|
1197
|
+
|
|
1198
|
+
similarityOptions = optionsCommand.add_parser("similarity")
|
|
1199
|
+
|
|
1200
|
+
similarityOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1201
|
+
metavar='data',dest='data',
|
|
1202
|
+
help='Input tabular text file (input should be file of fasta files when used with --mash/-m and --fastani/-n options or alignment file when used with --aln/-a option)')
|
|
1203
|
+
similarityOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1204
|
+
metavar='format',dest='format',default="tsv",
|
|
1205
|
+
choices=['tsv','csv'],
|
|
1206
|
+
help='Data format (default: tsv)')
|
|
1207
|
+
similarityOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1208
|
+
metavar='output',dest='output',default="out.similarity.tsv",
|
|
1209
|
+
help='Output percent similarity')
|
|
1210
|
+
similarityOptions.add_argument('--length','-l',action='store',required=False,nargs=1,
|
|
1211
|
+
metavar='length',dest='length',default=31,type=int,
|
|
1212
|
+
help='K-mer length for --mash/-m or --fastani/-n (default: 31)')
|
|
1213
|
+
similarityOptions.add_argument('--size','-s',action='store',required=False,nargs=1,
|
|
1214
|
+
metavar='size',dest='size',default=1000000,type=int,
|
|
1215
|
+
help='Sketch size for MASH; use with --mash (default: 1000000)')
|
|
1216
|
+
similarityOptions.add_argument('--mash','-m',action='store_true',default=False,
|
|
1217
|
+
dest='mash',help='Calculate similarity using MASH')
|
|
1218
|
+
similarityOptions.add_argument('--fastani','-n',action='store_true',default=False,
|
|
1219
|
+
dest='fastani',help='Calculate similarity using fastANI')
|
|
1220
|
+
similarityOptions.add_argument('--aln','-a',action='store_true',default=False,
|
|
1221
|
+
dest='aln',help='Calculate similarity based on a given sequence alignment')
|
|
1222
|
+
similarityOptions.add_argument('--fraction','-c',action='store',required=False,nargs=1,
|
|
1223
|
+
metavar='fraction',dest='fraction',default=0.70,type=float,
|
|
1224
|
+
help='Minimum overlap fraction with shorter sequence for ANI calculation; use with --fastani (default: 0.70)')
|
|
1225
|
+
similarityOptions.add_argument('--fragment','-r',action='store',required=False,nargs=1,
|
|
1226
|
+
metavar='fragment',dest='fragment',default=500,type=int,
|
|
1227
|
+
help='Fragment length for calculating ANI; use with --fastani (default: 500)')
|
|
1228
|
+
similarityOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
|
|
1229
|
+
metavar='threads',dest='threads',default=5,type=int,
|
|
1230
|
+
help='Number of threads (default: 5)')
|
|
1231
|
+
|
|
1232
|
+
|
|
1233
|
+
orderOptions = optionsCommand.add_parser("order")
|
|
1234
|
+
|
|
1235
|
+
orderOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1236
|
+
metavar='data',dest='data',
|
|
1237
|
+
help='Input tabular text file')
|
|
1238
|
+
orderOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1239
|
+
metavar='format',dest='format',default="tsv",
|
|
1240
|
+
choices=['tsv','csv'],
|
|
1241
|
+
help='Data format (default: tsv)')
|
|
1242
|
+
orderOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1243
|
+
metavar='output',dest='output',default="out.ordered.tsv",
|
|
1244
|
+
help='Output ordered data file')
|
|
1245
|
+
orderOptions.add_argument('--alphabetic','-a',action='store_true',required=False,
|
|
1246
|
+
dest='alphabetic',default=False,
|
|
1247
|
+
help='Show similarity values as percentage instead of fraction')
|
|
1248
|
+
orderOptions.add_argument('--rownames','-r',action='store',required=False,nargs=1,
|
|
1249
|
+
metavar='rownames',dest='rownames',
|
|
1250
|
+
help='Order by item names in a given text file (one name per row)')
|
|
1251
|
+
orderOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
|
|
1252
|
+
metavar='threads',dest='threads',default=5,type=int,
|
|
1253
|
+
help='Number of threads (default: 5)')
|
|
1254
|
+
|
|
1255
|
+
renameOptions = optionsCommand.add_parser("rename")
|
|
1256
|
+
|
|
1257
|
+
renameOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1258
|
+
metavar='data',dest='data',
|
|
1259
|
+
help='Input tabular text file')
|
|
1260
|
+
renameOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1261
|
+
metavar='format',dest='format',default="tsv",
|
|
1262
|
+
choices=['tsv','csv'],
|
|
1263
|
+
help='Data format (default: tsv)')
|
|
1264
|
+
renameOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1265
|
+
metavar='output',dest='output',default="out.renamed.tsv",
|
|
1266
|
+
help='Output ordered data file')
|
|
1267
|
+
renameOptions.add_argument('--columns','-c',action='store',required=True,nargs="*",
|
|
1268
|
+
metavar='columns',dest='columns',
|
|
1269
|
+
help='Selected column numbers (e.g., --columns 1 2 3 4)')
|
|
1270
|
+
renameOptions.add_argument('--newdata','-n',action='store',required=True,nargs=1,
|
|
1271
|
+
metavar='newdata',dest='newdata',
|
|
1272
|
+
help='Order by item names in a given text file (one name per row)')
|
|
1273
|
+
|
|
1274
|
+
|
|
1275
|
+
filterOptions = optionsCommand.add_parser("filter")
|
|
1276
|
+
|
|
1277
|
+
filterOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1278
|
+
metavar='data',dest='data',
|
|
1279
|
+
help='Input tabular text file')
|
|
1280
|
+
filterOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1281
|
+
metavar='format',dest='format',default="tsv",
|
|
1282
|
+
choices=['tsv','csv'],
|
|
1283
|
+
help='Data format (default: tsv)')
|
|
1284
|
+
filterOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1285
|
+
metavar='output',dest='output',default="out.filter.tsv",
|
|
1286
|
+
help='Output percent similarity')
|
|
1287
|
+
filterOptions.add_argument('--min','-a',action='store',required=False,nargs=1,
|
|
1288
|
+
metavar='minCutoff',dest='minCutoff',default=0,type=int,
|
|
1289
|
+
help='Select with frequency ≥minCutoff percent (overrides --columns/-c and --rows/-r)')
|
|
1290
|
+
filterOptions.add_argument('--max','-b',action='store',required=False,nargs=1,
|
|
1291
|
+
metavar='maxCutoff',dest='maxCutoff',default=100,type=int,
|
|
1292
|
+
help='Select with frequency ≤maxCutoff percent (overrides --columns/-c and --rows/-r)')
|
|
1293
|
+
filterOptions.add_argument('--columns','-c',action='store',required=False,nargs=1,
|
|
1294
|
+
metavar='columns',dest='columns',
|
|
1295
|
+
help='Select only columns present in a given text file (one column value per row)')
|
|
1296
|
+
filterOptions.add_argument('--rows','-r',action='store',required=False,nargs=1,
|
|
1297
|
+
metavar='rows',dest='rows',
|
|
1298
|
+
help='Select only rows present in a given text file (one column value per row)')
|
|
1299
|
+
filterOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
|
|
1300
|
+
metavar='threads',dest='threads',default=5,type=int,
|
|
1301
|
+
help='Number of threads (default: 5)')
|
|
1302
|
+
|
|
1303
|
+
|
|
1304
|
+
sampleOptions = optionsCommand.add_parser("sample")
|
|
1305
|
+
|
|
1306
|
+
sampleOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1307
|
+
metavar='data',dest='data',
|
|
1308
|
+
help='Input tabular text file')
|
|
1309
|
+
sampleOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1310
|
+
metavar='format',dest='format',default="tsv",
|
|
1311
|
+
choices=['tsv','csv'],
|
|
1312
|
+
help='Data format (default: tsv)')
|
|
1313
|
+
sampleOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1314
|
+
metavar='output',dest='output',default="out.sample.tsv",
|
|
1315
|
+
help='Output file')
|
|
1316
|
+
sampleOptions.add_argument('--cf','-c',action='store',required=False,nargs=1,
|
|
1317
|
+
metavar='colfreq',dest='colfreq',default=0,type=float,
|
|
1318
|
+
help='Randomly select ≥colfreq percent of the columns')
|
|
1319
|
+
sampleOptions.add_argument('--rf','-r',action='store',required=False,nargs=1,
|
|
1320
|
+
metavar='rowfreq',dest='rowfreq',default=0,type=float,
|
|
1321
|
+
help='Randomly select ≥rowfreq percent of the rows')
|
|
1322
|
+
|
|
1323
|
+
|
|
1324
|
+
transposeOptions = optionsCommand.add_parser("transpose")
|
|
1325
|
+
|
|
1326
|
+
transposeOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1327
|
+
metavar='data',dest='data',
|
|
1328
|
+
help='Input tabular text file')
|
|
1329
|
+
transposeOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1330
|
+
metavar='format',dest='format',default="tsv",
|
|
1331
|
+
choices=['tsv','csv'],
|
|
1332
|
+
help='Data format (default: tsv)')
|
|
1333
|
+
transposeOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1334
|
+
metavar='output',dest='output',default="out.filter.tsv",
|
|
1335
|
+
help='Output percent similarity')
|
|
1336
|
+
|
|
1337
|
+
|
|
1338
|
+
clusterOptions = optionsCommand.add_parser("cluster")
|
|
1339
|
+
|
|
1340
|
+
clusterOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1341
|
+
metavar='data',dest='data',
|
|
1342
|
+
help='Input tabular text file')
|
|
1343
|
+
clusterOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1344
|
+
metavar='format',dest='format',default="tsv",
|
|
1345
|
+
choices=['tsv','csv'],
|
|
1346
|
+
help='Data format (default: tsv)')
|
|
1347
|
+
clusterOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1348
|
+
metavar='output',dest='output',default="out.clusters.tsv",
|
|
1349
|
+
help='Output clusters')
|
|
1350
|
+
clusterOptions.add_argument('--prefix','-p',action='store',required=False,nargs=1,
|
|
1351
|
+
metavar='prefix',dest='prefix',default="CLS",
|
|
1352
|
+
help='Cluster label prefix')
|
|
1353
|
+
clusterOptions.add_argument('--sep','-s',action='store',required=False,nargs=1,
|
|
1354
|
+
metavar='separator',dest='separator',default="-",
|
|
1355
|
+
help='Lineage label separator')
|
|
1356
|
+
clusterOptions.add_argument('--thresholds','-t',action='store',required=True,nargs="*",
|
|
1357
|
+
metavar='thresholds',dest='thresholds',
|
|
1358
|
+
help='Cluster or lineage thresholds')
|
|
1359
|
+
|
|
1360
|
+
|
|
1361
|
+
gwasOptions = optionsCommand.add_parser("gwas")
|
|
1362
|
+
|
|
1363
|
+
gwasOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1364
|
+
metavar='data',dest='data',
|
|
1365
|
+
help='Input tabular text file')
|
|
1366
|
+
gwasOptions.add_argument('--format','-f',action='store',required=False,nargs=1,
|
|
1367
|
+
metavar='format',dest='format',default="tsv",
|
|
1368
|
+
choices=['tsv','csv'],
|
|
1369
|
+
help='Data format (default: tsv)')
|
|
1370
|
+
gwasOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1371
|
+
metavar='output',dest='output',default="out.gwas.tsv",
|
|
1372
|
+
help='Output percent similarity')
|
|
1373
|
+
gwasOptions.add_argument('--phenotype','-p',action='store',required=False,nargs=1,
|
|
1374
|
+
metavar='phenotype',dest='phenotype',
|
|
1375
|
+
help='Input phenotype text file (row name,phenotype); no header')
|
|
1376
|
+
gwasOptions.add_argument('--pyseer','-s',action='store_true',required=False,
|
|
1377
|
+
dest='pyseer',default=False,
|
|
1378
|
+
help='Output presence and absence data for Pyseer')
|
|
1379
|
+
gwasOptions.add_argument('--plink','-l',action='store_true',required=False,
|
|
1380
|
+
dest='plink',default=False,
|
|
1381
|
+
help='Output presence and absence data for Pyseer')
|
|
1382
|
+
gwasOptions.add_argument('--fastlmm','-m',action='store_true',required=False,
|
|
1383
|
+
dest='fastlmm',default=False,
|
|
1384
|
+
help='Output presence and absence data for Pyseer')
|
|
1385
|
+
gwasOptions.add_argument('--gemma','-g',action='store_true',required=False,
|
|
1386
|
+
dest='gemma',default=False,
|
|
1387
|
+
help='Output presence and absence data for GEMMA')
|
|
1388
|
+
gwasOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
|
|
1389
|
+
metavar='threads',dest='threads',default=5,type=int,
|
|
1390
|
+
help='Number of threads (default: 5)')
|
|
1391
|
+
|
|
1392
|
+
|
|
1393
|
+
kmerMatrixOptions = optionsCommand.add_parser("table")
|
|
1394
|
+
|
|
1395
|
+
kmerMatrixOptions.add_argument('--data','-d',action='store',required=True,nargs=1,
|
|
1396
|
+
metavar='data',dest='data',
|
|
1397
|
+
help='Input text files containing fasta sequence file names')
|
|
1398
|
+
kmerMatrixOptions.add_argument('--output','-o',action='store',required=False,nargs=1,
|
|
1399
|
+
metavar='output',dest='output',default="db.kmers.fasta",
|
|
1400
|
+
help='Output percent similarity')
|
|
1401
|
+
kmerMatrixOptions.add_argument('--length','-l',action='store',required=False,nargs=1,
|
|
1402
|
+
metavar='length',dest='length',default=31,type=int,
|
|
1403
|
+
help='K-mer length (default: 31)')
|
|
1404
|
+
kmerMatrixOptions.add_argument('--min','-m',action='store',required=False,nargs=1,
|
|
1405
|
+
metavar='minabundance',dest='minabundance',default=1,type=int,
|
|
1406
|
+
help='Minimum k-mer abundance (minabundance: 1)')
|
|
1407
|
+
kmerMatrixOptions.add_argument('--maxmem','-x',action='store',required=False,nargs=1,
|
|
1408
|
+
metavar='maxmemory',dest='maxmemory',default=31,type=int,
|
|
1409
|
+
help='Maximum memory (default: 5000)')
|
|
1410
|
+
kmerMatrixOptions.add_argument('--dsk','-s',action='store_true',default=True,
|
|
1411
|
+
dest='dsk',help='DSK to identify k-mers')
|
|
1412
|
+
kmerMatrixOptions.add_argument('--bifrost','-b',action='store_true',default=False,
|
|
1413
|
+
dest='bifrost',help='Bifrost to identify variable length k-mers or unitigs')
|
|
1414
|
+
kmerMatrixOptions.add_argument('--kmers','-k',action='store',required=False,nargs=1,
|
|
1415
|
+
metavar='kmers',dest='kmers',default="",
|
|
1416
|
+
help='Only query input sequence against a given k-mer database')
|
|
1417
|
+
kmerMatrixOptions.add_argument('--threads','-t',action='store',required=False,nargs=1,
|
|
1418
|
+
metavar='threads',dest='threads',default=5,type=int,
|
|
1419
|
+
help='Number of threads (default: 5)')
|
|
1420
|
+
|
|
1421
|
+
|
|
1422
|
+
versionOptions = optionsCommand.add_parser("version")
|
|
1423
|
+
|
|
1424
|
+
versionOptions.add_argument('--version','-v',action='store_false',default=True,
|
|
1425
|
+
dest='version',help='Show software version')
|
|
1426
|
+
|
|
1427
|
+
|
|
1428
|
+
citationOptions = optionsCommand.add_parser("citation")
|
|
1429
|
+
|
|
1430
|
+
citationOptions.add_argument('--citation','-c',action='store_false',default=True,
|
|
1431
|
+
dest='citation',help='Show software citation information')
|
|
1432
|
+
|
|
1433
|
+
|
|
1434
|
+
options = optionsCmd.parse_args(args=None if sys.argv[1:] else ['--help'])
|
|
1435
|
+
|
|
1436
|
+
cmdValues = {}
|
|
1437
|
+
|
|
1438
|
+
if options.command == "similarity":
|
|
1439
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1440
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1441
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1442
|
+
'length': int(options.length[0:][0]) if isinstance(options.length,list) else int(options.length),
|
|
1443
|
+
'fragment': int(options.fragment[0:][0]) if isinstance(options.fragment,list) else int(options.fragment),
|
|
1444
|
+
'fraction': int(options.fraction[0:][0]) if isinstance(options.fraction,list) else int(options.fraction),
|
|
1445
|
+
'size': int(options.size[0:][0]) if isinstance(options.size,list) else int(options.size),
|
|
1446
|
+
'mash': options.mash,
|
|
1447
|
+
'fastani': options.fastani,
|
|
1448
|
+
'aln': options.aln,
|
|
1449
|
+
'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1452
|
+
similarity(cmdValues)
|
|
1453
|
+
|
|
1454
|
+
elif options.command == "filter":
|
|
1455
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1456
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1457
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1458
|
+
'minCutoff': int(options.minCutoff[0:][0]) if isinstance(options.minCutoff,list) else int(options.minCutoff),
|
|
1459
|
+
'maxCutoff': int(options.maxCutoff[0:][0]) if isinstance(options.maxCutoff,list) else int(options.maxCutoff),
|
|
1460
|
+
'rows': options.rows[0:][0] if isinstance(options.rows,list) else options.rows,
|
|
1461
|
+
'columns': options.columns[0:][0] if isinstance(options.columns,list) else options.columns,
|
|
1462
|
+
'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
|
|
1463
|
+
}
|
|
1464
|
+
|
|
1465
|
+
filter(cmdValues)
|
|
1466
|
+
|
|
1467
|
+
elif options.command == "sample":
|
|
1468
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1469
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1470
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1471
|
+
'colfreq': float(options.colfreq[0:][0]) if isinstance(options.colfreq,list) else float(options.colfreq),
|
|
1472
|
+
'rowfreq': float(options.rowfreq[0:][0]) if isinstance(options.rowfreq,list) else float(options.rowfreq)
|
|
1473
|
+
}
|
|
1474
|
+
|
|
1475
|
+
sample(cmdValues)
|
|
1476
|
+
|
|
1477
|
+
elif options.command == "order":
|
|
1478
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1479
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1480
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1481
|
+
'rownames': options.rownames[0:][0] if isinstance(options.rownames,list) else options.rownames,
|
|
1482
|
+
'alphabetic': options.alphabetic,
|
|
1483
|
+
'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
|
|
1484
|
+
}
|
|
1485
|
+
|
|
1486
|
+
order(cmdValues)
|
|
1487
|
+
|
|
1488
|
+
elif options.command == "rename":
|
|
1489
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1490
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1491
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1492
|
+
'newdata': options.newdata[0:][0] if isinstance(options.newdata,list) else options.newdata,
|
|
1493
|
+
'columns': options.columns[0:] if isinstance(options.columns,list) else options.columns
|
|
1494
|
+
}
|
|
1495
|
+
|
|
1496
|
+
rename(cmdValues)
|
|
1497
|
+
|
|
1498
|
+
elif options.command == "cluster":
|
|
1499
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1500
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1501
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1502
|
+
'prefix': options.prefix[0:][0] if isinstance(options.prefix,list) else options.prefix,
|
|
1503
|
+
'separator': options.separator[0:][0] if isinstance(options.separator,list) else options.separator,
|
|
1504
|
+
'thresholds': [float(i) for i in options.thresholds[0:]] if isinstance(options.thresholds,list) else float(options.thresholds)
|
|
1505
|
+
}
|
|
1506
|
+
|
|
1507
|
+
cluster(cmdValues)
|
|
1508
|
+
|
|
1509
|
+
elif options.command == "transpose":
|
|
1510
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1511
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1512
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output
|
|
1513
|
+
}
|
|
1514
|
+
|
|
1515
|
+
transpose(cmdValues)
|
|
1516
|
+
|
|
1517
|
+
elif options.command == "gwas":
|
|
1518
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1519
|
+
'format': options.format[0:][0] if isinstance(options.format,list) else options.format,
|
|
1520
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1521
|
+
'phenotype': options.phenotype[0:][0] if isinstance(options.phenotype,list) else options.phenotype,
|
|
1522
|
+
'pyseer': options.pyseer,
|
|
1523
|
+
'fastlmm': options.fastlmm,
|
|
1524
|
+
'plink': options.plink,
|
|
1525
|
+
'gemma': options.gemma,
|
|
1526
|
+
'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
|
|
1527
|
+
}
|
|
1528
|
+
|
|
1529
|
+
gwas(cmdValues)
|
|
1530
|
+
|
|
1531
|
+
elif options.command == "table":
|
|
1532
|
+
cmdValues = {'data': options.data[0:][0],
|
|
1533
|
+
'output': options.output[0:][0] if isinstance(options.output,list) else options.output,
|
|
1534
|
+
'dsk': options.dsk,
|
|
1535
|
+
'bifrost': options.bifrost,
|
|
1536
|
+
'minabundance': int(options.minabundance[0:][0]) if isinstance(options.minabundance,list) else int(options.minabundance),
|
|
1537
|
+
'maxmemory': int(options.maxmemory[0:][0]) if isinstance(options.maxmemory,list) else int(options.maxmemory),
|
|
1538
|
+
'length': int(options.length[0:][0]) if isinstance(options.length,list) else int(options.length),
|
|
1539
|
+
'kmers': options.kmers[0:][0] if isinstance(options.kmers,list) else options.kmers,
|
|
1540
|
+
'threads': int(options.threads[0:][0]) if isinstance(options.threads,list) else int(options.threads)
|
|
1541
|
+
}
|
|
1542
|
+
|
|
1543
|
+
table(cmdValues)
|
|
1544
|
+
|
|
1545
|
+
elif options.command == "version":
|
|
1546
|
+
printVersion()
|
|
1547
|
+
|
|
1548
|
+
elif options.command == "citation":
|
|
1549
|
+
citation()
|
|
1550
|
+
|
|
1551
|
+
else:
|
|
1552
|
+
sys.exit()
|
|
1553
|
+
|
|
1554
|
+
|
|
1555
|
+
if __name__=="__main__":
|
|
1556
|
+
tabtkMain()
|