MetaPont 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- MetaPont/Extraction_By_Function.py +95 -0
- MetaPont/Function_By_Taxa.py +304 -0
- MetaPont/__init__.py +0 -0
- MetaPont/constants.py +2 -0
- MetaPont/emapper_readmap_cds_combiner_v1.py +235 -0
- MetaPont/kraken_emapper_readmap_combiner_v1.py +222 -0
- MetaPont/lineage_matrix_collapse_v1.py +80 -0
- MetaPont/lineage_matrix_collapse_v2.py +121 -0
- MetaPont-0.0.1.dist-info/LICENSE +674 -0
- MetaPont-0.0.1.dist-info/METADATA +111 -0
- MetaPont-0.0.1.dist-info/RECORD +14 -0
- MetaPont-0.0.1.dist-info/WHEEL +5 -0
- MetaPont-0.0.1.dist-info/entry_points.txt +2 -0
- MetaPont-0.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import os
|
|
3
|
+
import csv
|
|
4
|
+
import sys
|
|
5
|
+
from collections import Counter
|
|
6
|
+
|
|
7
|
+
from HuwsLab.MetaPont.src.MetaPont.constants import MetaPont_Version
|
|
8
|
+
|
|
9
|
+
# Needed to account for the large CSV/TSV files we are working with
|
|
10
|
+
csv.field_size_limit(sys.maxsize)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def process_tsv(file_path, function_id):
|
|
14
|
+
"""
|
|
15
|
+
Processes a TSV file to calculate taxa counts and total matches for a given function ID.
|
|
16
|
+
"""
|
|
17
|
+
taxa_counts = Counter()
|
|
18
|
+
total_matches = 0
|
|
19
|
+
|
|
20
|
+
with open(file_path, "r") as tsv_file:
|
|
21
|
+
reader = csv.reader(tsv_file, delimiter="\t")
|
|
22
|
+
next(reader) # Skip the first row (sample name)
|
|
23
|
+
headers = next(reader) # Read headers from the second row
|
|
24
|
+
|
|
25
|
+
# Locate the Lineage column
|
|
26
|
+
taxa_idx = headers.index("Lineage")
|
|
27
|
+
|
|
28
|
+
# Process each row to find matches
|
|
29
|
+
for idx, row in enumerate(reader):
|
|
30
|
+
if len(row) < len(headers):
|
|
31
|
+
continue # Skip malformed rows
|
|
32
|
+
|
|
33
|
+
lineage = row[taxa_idx]
|
|
34
|
+
|
|
35
|
+
# Search functional columns starting from column 6
|
|
36
|
+
for cell in row[6:]:
|
|
37
|
+
if cell and any(function_id in part for part in cell.replace(',', '|').split('|')):
|
|
38
|
+
# Extract genus from the Lineage field
|
|
39
|
+
if "g__" in lineage:
|
|
40
|
+
genus = lineage.split("g__")[1].split("|")[0]
|
|
41
|
+
taxa_counts[genus] += 1
|
|
42
|
+
total_matches += 1
|
|
43
|
+
break # Stop checking further functional columns for this row
|
|
44
|
+
|
|
45
|
+
return taxa_counts, total_matches
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def main():
|
|
49
|
+
|
|
50
|
+
parser = argparse.ArgumentParser(description='MetaPont ' + MetaPont_Version + ': Extract-By-Function - Identify taxa contributing to a specific function.')
|
|
51
|
+
parser.add_argument(
|
|
52
|
+
"-d", "--directory", required=True,
|
|
53
|
+
help="Directory containing TSV files to analyse."
|
|
54
|
+
)
|
|
55
|
+
parser.add_argument(
|
|
56
|
+
"-f", "--function_id", required=True,
|
|
57
|
+
help="Specific function ID to search for (e.g., 'GO:0002')."
|
|
58
|
+
)
|
|
59
|
+
parser.add_argument(
|
|
60
|
+
"-o", "--output", default="output_taxa_proportions.tsv",
|
|
61
|
+
help="Output file to save results (default: output_taxa_proportions.tsv)."
|
|
62
|
+
)
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
"-m", "--min_proportion", type=float, default=0.05,
|
|
65
|
+
help="Minimum proportion threshold for taxa to be included in the output (default: 0.05)."
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
options = parser.parse_args()
|
|
69
|
+
print("Running MetaPont: Extract-By-Function " + MetaPont_Version)
|
|
70
|
+
|
|
71
|
+
all_results = {}
|
|
72
|
+
|
|
73
|
+
# Process each TSV file in the directory
|
|
74
|
+
for file_name in os.listdir(options.directory):
|
|
75
|
+
if file_name.endswith("_Final_Contig.tsv"):
|
|
76
|
+
file_path = os.path.join(options.directory, file_name)
|
|
77
|
+
print(f"Processing file: {file_name}")
|
|
78
|
+
taxa_counts, total_matches = process_tsv(file_path, options.function_id)
|
|
79
|
+
all_results[file_name] = (taxa_counts, total_matches)
|
|
80
|
+
|
|
81
|
+
# Write results to output
|
|
82
|
+
with open(options.output, "w") as out:
|
|
83
|
+
out.write("Function ID: " + options.function_id + "\n")
|
|
84
|
+
out.write("Sample\tTaxa\tProportion\n")
|
|
85
|
+
for sample, (taxa_counts, total_matches) in all_results.items():
|
|
86
|
+
for taxa, count in taxa_counts.items():
|
|
87
|
+
proportion = count / total_matches if total_matches > 0 else 0
|
|
88
|
+
if proportion >= options.min_proportion: # Apply minimum proportion filter
|
|
89
|
+
out.write(f"{sample}\t{taxa}\t{proportion:.6f}\n")
|
|
90
|
+
|
|
91
|
+
print(f"Results saved to {options.output}")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
if __name__ == "__main__":
|
|
95
|
+
main()
|
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
import glob
|
|
4
|
+
import argparse
|
|
5
|
+
import sys
|
|
6
|
+
import collections
|
|
7
|
+
################
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def read_files(files_list):
|
|
12
|
+
all_entries = []
|
|
13
|
+
read_counts = collections.defaultdict(lambda: collections.defaultdict(lambda: collections.defaultdict(int)))
|
|
14
|
+
|
|
15
|
+
total_reads = collections.defaultdict(int)
|
|
16
|
+
for file_path in files_list:
|
|
17
|
+
sample = file_path.split('/')[-1].split('_Final')[0] # _Final is specific to this dataset
|
|
18
|
+
with open(file_path, 'r') as file:
|
|
19
|
+
for line in file:
|
|
20
|
+
line = line.replace('\n','')
|
|
21
|
+
if line.startswith('NODE'): # Only works for this dataset/metaspades
|
|
22
|
+
line_data = line.split('\t')
|
|
23
|
+
mapped_reads = int(line_data[3])
|
|
24
|
+
total_reads[sample] = int(line_data[2])
|
|
25
|
+
lineage = line_data[5]
|
|
26
|
+
domain = lineage.split('|')[0]
|
|
27
|
+
## COGs
|
|
28
|
+
COGs = line_data[6]
|
|
29
|
+
if COGs != '-':
|
|
30
|
+
cog_list = [cog for cog in COGs]
|
|
31
|
+
for cog in cog_list:
|
|
32
|
+
if cog != '|':
|
|
33
|
+
read_counts[sample][domain][cog] += mapped_reads
|
|
34
|
+
|
|
35
|
+
else:
|
|
36
|
+
if line.startswith('Contig'):
|
|
37
|
+
columns = line.split('\t')
|
|
38
|
+
return all_entries, read_counts, total_reads
|
|
39
|
+
|
|
40
|
+
################
|
|
41
|
+
|
|
42
|
+
parent_directory_path = sys.argv[1]
|
|
43
|
+
|
|
44
|
+
all_COGs = []
|
|
45
|
+
all_GOs = []
|
|
46
|
+
all_ECs = []
|
|
47
|
+
all_KEGG_kos = []
|
|
48
|
+
all_KEGG_Pathways = []
|
|
49
|
+
all_KEGG_Modules = []
|
|
50
|
+
all_KEGG_Reactions = []
|
|
51
|
+
all_KEGG_rclasses = []
|
|
52
|
+
all_BRITEs = []
|
|
53
|
+
all_KEGG_TCs = []
|
|
54
|
+
all_CAZys =[]
|
|
55
|
+
all_BiGG_Reactions = []
|
|
56
|
+
all_PFAMs = []
|
|
57
|
+
|
|
58
|
+
functions = defaultdict(lambda: defaultdict(lambda: defaultdict(int)))
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
|
|
62
|
+
for dir in dirs:
|
|
63
|
+
try:
|
|
64
|
+
current_dir = os.path.basename(dir)
|
|
65
|
+
parent_of_current_dir = os.path.basename(subdir)
|
|
66
|
+
if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
|
|
67
|
+
print(current_dir)
|
|
68
|
+
specific_dir_path = os.path.join(subdir, dir)
|
|
69
|
+
first_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_cds_summary.txt")
|
|
70
|
+
second_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
|
|
71
|
+
|
|
72
|
+
if os.path.exists(first_file_path) and os.path.exists(second_file_path):
|
|
73
|
+
print(f"Master directory: {specific_dir_path}")
|
|
74
|
+
print(f"First file path: {first_file_path}")
|
|
75
|
+
print(f"Second file path: {second_file_path}")
|
|
76
|
+
|
|
77
|
+
genes = defaultdict(int)
|
|
78
|
+
with open(first_file_path, 'r') as readmap_cds_in:
|
|
79
|
+
for line in readmap_cds_in:
|
|
80
|
+
reads = line.split()[0]
|
|
81
|
+
gene = line.split()[1].replace('ID=','')
|
|
82
|
+
contig_length = int(gene.split('length_')[1].split('_cov')[0])
|
|
83
|
+
if contig_length >= 2500:
|
|
84
|
+
genes.update({gene:reads})
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
lineages = {}
|
|
90
|
+
taxa_ids = []
|
|
91
|
+
|
|
92
|
+
with open(second_file_path, 'r') as emapper_in:
|
|
93
|
+
for line in emapper_in:
|
|
94
|
+
if not line.startswith('#'):
|
|
95
|
+
line_data = line.strip().split('\t')
|
|
96
|
+
gene = line_data[0]
|
|
97
|
+
if gene in genes:
|
|
98
|
+
gene_read_count = int(genes[gene])
|
|
99
|
+
|
|
100
|
+
COGs = line_data[6]
|
|
101
|
+
if COGs != '-':
|
|
102
|
+
#cog_list = COGs.split(',')
|
|
103
|
+
cog_list = [cog for cog in COGs]
|
|
104
|
+
for cog in cog_list:
|
|
105
|
+
functions[dir]['COG'][cog] += gene_read_count
|
|
106
|
+
if cog not in all_COGs:
|
|
107
|
+
all_COGs.append(cog)
|
|
108
|
+
|
|
109
|
+
GOs = line_data[9]
|
|
110
|
+
if GOs != '-':
|
|
111
|
+
go_list = GOs.split(',')
|
|
112
|
+
for go in go_list:
|
|
113
|
+
functions[dir]['GO'][go] += gene_read_count
|
|
114
|
+
if go not in all_GOs:
|
|
115
|
+
all_GOs.append(go)
|
|
116
|
+
|
|
117
|
+
EC = line_data[10]
|
|
118
|
+
if EC != '-':
|
|
119
|
+
ec_list = EC.split(',')
|
|
120
|
+
for ec in ec_list:
|
|
121
|
+
functions[dir]['EC'][ec] += gene_read_count
|
|
122
|
+
if ec not in all_ECs:
|
|
123
|
+
all_ECs.append(ec)
|
|
124
|
+
|
|
125
|
+
KEGG_ko = line_data[11]
|
|
126
|
+
if KEGG_ko != '-':
|
|
127
|
+
kegg_ko_list = KEGG_ko.split(',')
|
|
128
|
+
for kegg_ko in kegg_ko_list:
|
|
129
|
+
functions[dir]['KEGG_ko'][kegg_ko] += gene_read_count
|
|
130
|
+
if kegg_ko not in all_KEGG_kos:
|
|
131
|
+
all_KEGG_kos.append(kegg_ko)
|
|
132
|
+
|
|
133
|
+
KEGG_Pathway = line_data[12]
|
|
134
|
+
if KEGG_Pathway != '-':
|
|
135
|
+
kegg_pathway_list = KEGG_Pathway.split(',')
|
|
136
|
+
for kegg_pathway in kegg_pathway_list:
|
|
137
|
+
functions[dir]['KEGG_Pathway'][kegg_pathway] += gene_read_count
|
|
138
|
+
if kegg_pathway not in all_KEGG_Pathways:
|
|
139
|
+
all_KEGG_Pathways.append(kegg_pathway)
|
|
140
|
+
|
|
141
|
+
KEGG_Module = line_data[13]
|
|
142
|
+
if KEGG_Module != '-':
|
|
143
|
+
kegg_module_list = KEGG_Module.split(',')
|
|
144
|
+
for kegg_module in kegg_module_list:
|
|
145
|
+
functions[dir]['KEGG_Module'][kegg_module] += gene_read_count
|
|
146
|
+
if kegg_module not in all_KEGG_Modules:
|
|
147
|
+
all_KEGG_Modules.append(kegg_module)
|
|
148
|
+
|
|
149
|
+
KEGG_Reaction = line_data[14]
|
|
150
|
+
if KEGG_Reaction != '-':
|
|
151
|
+
kegg_reaction_list = KEGG_Reaction.split(',')
|
|
152
|
+
for kegg_reaction in kegg_reaction_list:
|
|
153
|
+
functions[dir]['KEGG_Reaction'][kegg_reaction] += gene_read_count
|
|
154
|
+
if kegg_reaction not in all_KEGG_Reactions:
|
|
155
|
+
all_KEGG_Reactions.append(kegg_reaction)
|
|
156
|
+
|
|
157
|
+
KEGG_rclass = line_data[15]
|
|
158
|
+
if KEGG_rclass != '-':
|
|
159
|
+
kegg_rclass_list = KEGG_rclass.split(',')
|
|
160
|
+
for kegg_rclass in kegg_rclass_list:
|
|
161
|
+
functions[dir]['KEGG_rclass'][kegg_rclass] += gene_read_count
|
|
162
|
+
if kegg_rclass not in all_KEGG_rclasses:
|
|
163
|
+
all_KEGG_rclasses.append(kegg_rclass)
|
|
164
|
+
|
|
165
|
+
BRITE = line_data[16]
|
|
166
|
+
if BRITE != '-':
|
|
167
|
+
brite_list = BRITE.split(',')
|
|
168
|
+
for brite in brite_list:
|
|
169
|
+
functions[dir]['BRITE'][brite] += gene_read_count
|
|
170
|
+
if brite not in all_BRITEs:
|
|
171
|
+
all_BRITEs.append(brite)
|
|
172
|
+
|
|
173
|
+
KEGG_TC = line_data[17]
|
|
174
|
+
if KEGG_TC != '-':
|
|
175
|
+
kegg_tc_list = KEGG_TC.split(',')
|
|
176
|
+
for kegg_tc in kegg_tc_list:
|
|
177
|
+
functions[dir]['KEGG_TC'][kegg_tc] += gene_read_count
|
|
178
|
+
if kegg_tc not in all_KEGG_TCs:
|
|
179
|
+
all_KEGG_TCs.append(kegg_tc)
|
|
180
|
+
|
|
181
|
+
CAZy = line_data[18]
|
|
182
|
+
if CAZy != '-':
|
|
183
|
+
cazy_list = CAZy.split(',')
|
|
184
|
+
for cazy in cazy_list:
|
|
185
|
+
functions[dir]['CAZy'][cazy] += gene_read_count
|
|
186
|
+
if cazy not in all_CAZys:
|
|
187
|
+
all_CAZys.append(cazy)
|
|
188
|
+
|
|
189
|
+
BiGG_Reaction = line_data[19]
|
|
190
|
+
if BiGG_Reaction != '-':
|
|
191
|
+
bigg_reaction_list = BiGG_Reaction.split(',')
|
|
192
|
+
for bigg_reaction in bigg_reaction_list:
|
|
193
|
+
functions[dir]['BiGG_Reaction'][bigg_reaction] += gene_read_count
|
|
194
|
+
if bigg_reaction not in all_BiGG_Reactions:
|
|
195
|
+
all_BiGG_Reactions.append(bigg_reaction)
|
|
196
|
+
|
|
197
|
+
PFAMs = line_data[20]
|
|
198
|
+
if PFAMs != '-':
|
|
199
|
+
pfams_list = PFAMs.split(',')
|
|
200
|
+
for pfam in pfams_list:
|
|
201
|
+
functions[dir]['PFAM'][pfam] += gene_read_count
|
|
202
|
+
if pfam not in all_PFAMs:
|
|
203
|
+
all_PFAMs.append(pfam)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
print("Sample " + dir + " Done")
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
except FileNotFoundError as e:
|
|
211
|
+
print(e)
|
|
212
|
+
|
|
213
|
+
functions = dict(sorted(functions.items()))
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def write_transposed_output(output_path, category, all_items, functions):
|
|
217
|
+
output = open(output_path, 'w')
|
|
218
|
+
output.write('\t' + '\t'.join(functions.keys()) + '\n')
|
|
219
|
+
|
|
220
|
+
for item in all_items:
|
|
221
|
+
row_values = [item] + [str(functions[sample][category][item]) for sample in functions.keys()]
|
|
222
|
+
output.write('\t'.join(row_values) + '\n')
|
|
223
|
+
|
|
224
|
+
output.close()
|
|
225
|
+
|
|
226
|
+
output_path = os.path.join(parent_directory_path, 'CDS_Final_Outputs')
|
|
227
|
+
if not os.path.exists(output_path):
|
|
228
|
+
os.makedirs(output_path)
|
|
229
|
+
|
|
230
|
+
# COG output
|
|
231
|
+
write_transposed_output(output_path + '/Final_CDS_COG.tsv', 'COG', all_COGs, functions)
|
|
232
|
+
|
|
233
|
+
# GO output
|
|
234
|
+
write_transposed_output(output_path + '/Final_CDS_GO.tsv', 'GO', all_GOs, functions)
|
|
235
|
+
|
|
236
|
+
# EC output
|
|
237
|
+
write_transposed_output(output_path + '/Final_CDS_EC.tsv', 'EC', all_ECs, functions)
|
|
238
|
+
|
|
239
|
+
# KEGG_ko output
|
|
240
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_ko.tsv', 'KEGG_ko', all_KEGG_kos, functions)
|
|
241
|
+
|
|
242
|
+
# KEGG_Pathway output
|
|
243
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Pathway.tsv', 'KEGG_Pathway', all_KEGG_Pathways, functions)
|
|
244
|
+
|
|
245
|
+
# KEGG_Module output
|
|
246
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Module.tsv', 'KEGG_Module', all_KEGG_Modules, functions)
|
|
247
|
+
|
|
248
|
+
# KEGG_Reaction output
|
|
249
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Reaction.tsv', 'KEGG_Reaction', all_KEGG_Reactions, functions)
|
|
250
|
+
|
|
251
|
+
# KEGG_rclass output
|
|
252
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_rclass.tsv', 'KEGG_rclass', all_KEGG_rclasses, functions)
|
|
253
|
+
|
|
254
|
+
# BRITE output
|
|
255
|
+
write_transposed_output(output_path + '/Final_CDS_BRITE.tsv', 'BRITE', all_BRITEs, functions)
|
|
256
|
+
|
|
257
|
+
# KEGG_TC output
|
|
258
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_TC.tsv', 'KEGG_TC', all_KEGG_TCs, functions)
|
|
259
|
+
|
|
260
|
+
# CAZy output
|
|
261
|
+
write_transposed_output(output_path + '/Final_CDS_CAZy.tsv', 'CAZy', all_CAZys, functions)
|
|
262
|
+
|
|
263
|
+
# BiGG_Reaction output
|
|
264
|
+
write_transposed_output(output_path + '/Final_CDS_BiGG_Reaction.tsv', 'BiGG_Reaction', all_BiGG_Reactions, functions)
|
|
265
|
+
|
|
266
|
+
# PFAMs output
|
|
267
|
+
write_transposed_output(output_path + '/Final_CDS_PFAMs.tsv', 'PFAM', all_PFAMs, functions)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def main():
|
|
271
|
+
|
|
272
|
+
parser = argparse.ArgumentParser(description='....')
|
|
273
|
+
parser._action_groups.pop()
|
|
274
|
+
|
|
275
|
+
required = parser.add_argument_group('Required Arguments')
|
|
276
|
+
required.add_argument('-d', action='store', dest='dir_path', required=True,
|
|
277
|
+
help='Define the directory path containing the files')
|
|
278
|
+
required.add_argument('-o', action='store', dest='output', help='Outdir',
|
|
279
|
+
required=True)
|
|
280
|
+
|
|
281
|
+
options = parser.parse_args()
|
|
282
|
+
|
|
283
|
+
# Use glob to find files ending with '_Final_Output.tsv'
|
|
284
|
+
files_list = glob.glob(f"{options.dir_path}/*_Final_*.tsv")
|
|
285
|
+
all_entries, read_counts, total_reads = read_files(files_list)
|
|
286
|
+
|
|
287
|
+
separate_taxa = ['d__Bacteria', 'd__Archaea', 'd__Eukaryota', 'd__Viruses','k__Fungi',
|
|
288
|
+
'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown']
|
|
289
|
+
remove_taxa = ['c__Mammalia','k__Viridiplantae'
|
|
290
|
+
,'d__Eukaryota|k__unknown|p__Evosea|c__Eumycetozoa|o__Dictyosteliales|f__Dictyosteliaceae|g__Dictyostelium'
|
|
291
|
+
,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Leishmania'
|
|
292
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Toxoplasma'
|
|
293
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Haemosporida|f__Plasmodiidae|g__Plasmodium'
|
|
294
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Piroplasmida|f__Theileriidae|g__Theileria'
|
|
295
|
+
,'d__Eukaryota|k__unknown|p__Parabasalia|c__unknown|o__Trichomonadida|f__Trichomonadidae|g__Trichomonas'
|
|
296
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Besnoitia'
|
|
297
|
+
,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Trypanosoma'
|
|
298
|
+
,'d__Eukaryota|k__unknown|p__Ciliophora|c__Oligohymenophorea|o__Peniculida|f__Parameciidae|g__Paramecium'
|
|
299
|
+
,'d__Eukaryota|k__Fungi|p__Microsporidia|c__unknown|o__unknown|f__Unikaryonidae|g__Encephalitozoon'
|
|
300
|
+
,'d__Eukaryota|k__unknown|p__unknown|c__Cryptophyceae|o__Cryptomonadales|f__Cryptomonadaceae|g__Cryptomonas']
|
|
301
|
+
|
|
302
|
+
if __name__ == "__main__":
|
|
303
|
+
main()
|
|
304
|
+
print("Complete")
|
MetaPont/__init__.py
ADDED
|
File without changes
|
MetaPont/constants.py
ADDED
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
import re
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
################
|
|
8
|
+
|
|
9
|
+
parent_directory_path = sys.argv[1]
|
|
10
|
+
|
|
11
|
+
all_COGs = []
|
|
12
|
+
all_GOs = []
|
|
13
|
+
all_ECs = []
|
|
14
|
+
all_KEGG_kos = []
|
|
15
|
+
all_KEGG_Pathways = []
|
|
16
|
+
all_KEGG_Modules = []
|
|
17
|
+
all_KEGG_Reactions = []
|
|
18
|
+
all_KEGG_rclasses = []
|
|
19
|
+
all_BRITEs = []
|
|
20
|
+
all_KEGG_TCs = []
|
|
21
|
+
all_CAZys =[]
|
|
22
|
+
all_BiGG_Reactions = []
|
|
23
|
+
all_PFAMs = []
|
|
24
|
+
|
|
25
|
+
functions = defaultdict(lambda: defaultdict(lambda: defaultdict(int)))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
|
|
29
|
+
for dir in dirs:
|
|
30
|
+
try:
|
|
31
|
+
current_dir = os.path.basename(dir)
|
|
32
|
+
parent_of_current_dir = os.path.basename(subdir)
|
|
33
|
+
if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
|
|
34
|
+
print(current_dir)
|
|
35
|
+
specific_dir_path = os.path.join(subdir, dir)
|
|
36
|
+
first_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_cds_summary.txt")
|
|
37
|
+
second_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
|
|
38
|
+
|
|
39
|
+
if os.path.exists(first_file_path) and os.path.exists(second_file_path):
|
|
40
|
+
print(f"Master directory: {specific_dir_path}")
|
|
41
|
+
print(f"First file path: {first_file_path}")
|
|
42
|
+
print(f"Second file path: {second_file_path}")
|
|
43
|
+
|
|
44
|
+
genes = defaultdict(int)
|
|
45
|
+
with open(first_file_path, 'r') as readmap_cds_in:
|
|
46
|
+
for line in readmap_cds_in:
|
|
47
|
+
reads = line.split()[0]
|
|
48
|
+
gene = line.split()[1].replace('ID=','')
|
|
49
|
+
contig_length = int(gene.split('length_')[1].split('_cov')[0])
|
|
50
|
+
if contig_length >= 2500:
|
|
51
|
+
genes.update({gene:reads})
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
lineages = {}
|
|
57
|
+
taxa_ids = []
|
|
58
|
+
|
|
59
|
+
with open(second_file_path, 'r') as emapper_in:
|
|
60
|
+
for line in emapper_in:
|
|
61
|
+
if not line.startswith('#'):
|
|
62
|
+
line_data = line.strip().split('\t')
|
|
63
|
+
gene = line_data[0]
|
|
64
|
+
if gene in genes:
|
|
65
|
+
gene_read_count = int(genes[gene])
|
|
66
|
+
|
|
67
|
+
COGs = line_data[6]
|
|
68
|
+
if COGs != '-':
|
|
69
|
+
#cog_list = COGs.split(',')
|
|
70
|
+
cog_list = [cog for cog in COGs]
|
|
71
|
+
for cog in cog_list:
|
|
72
|
+
functions[dir]['COG'][cog] += gene_read_count
|
|
73
|
+
if cog not in all_COGs:
|
|
74
|
+
all_COGs.append(cog)
|
|
75
|
+
|
|
76
|
+
GOs = line_data[9]
|
|
77
|
+
if GOs != '-':
|
|
78
|
+
go_list = GOs.split(',')
|
|
79
|
+
for go in go_list:
|
|
80
|
+
functions[dir]['GO'][go] += gene_read_count
|
|
81
|
+
if go not in all_GOs:
|
|
82
|
+
all_GOs.append(go)
|
|
83
|
+
|
|
84
|
+
EC = line_data[10]
|
|
85
|
+
if EC != '-':
|
|
86
|
+
ec_list = EC.split(',')
|
|
87
|
+
for ec in ec_list:
|
|
88
|
+
functions[dir]['EC'][ec] += gene_read_count
|
|
89
|
+
if ec not in all_ECs:
|
|
90
|
+
all_ECs.append(ec)
|
|
91
|
+
|
|
92
|
+
KEGG_ko = line_data[11]
|
|
93
|
+
if KEGG_ko != '-':
|
|
94
|
+
kegg_ko_list = KEGG_ko.split(',')
|
|
95
|
+
for kegg_ko in kegg_ko_list:
|
|
96
|
+
functions[dir]['KEGG_ko'][kegg_ko] += gene_read_count
|
|
97
|
+
if kegg_ko not in all_KEGG_kos:
|
|
98
|
+
all_KEGG_kos.append(kegg_ko)
|
|
99
|
+
|
|
100
|
+
KEGG_Pathway = line_data[12]
|
|
101
|
+
if KEGG_Pathway != '-':
|
|
102
|
+
kegg_pathway_list = KEGG_Pathway.split(',')
|
|
103
|
+
for kegg_pathway in kegg_pathway_list:
|
|
104
|
+
functions[dir]['KEGG_Pathway'][kegg_pathway] += gene_read_count
|
|
105
|
+
if kegg_pathway not in all_KEGG_Pathways:
|
|
106
|
+
all_KEGG_Pathways.append(kegg_pathway)
|
|
107
|
+
|
|
108
|
+
KEGG_Module = line_data[13]
|
|
109
|
+
if KEGG_Module != '-':
|
|
110
|
+
kegg_module_list = KEGG_Module.split(',')
|
|
111
|
+
for kegg_module in kegg_module_list:
|
|
112
|
+
functions[dir]['KEGG_Module'][kegg_module] += gene_read_count
|
|
113
|
+
if kegg_module not in all_KEGG_Modules:
|
|
114
|
+
all_KEGG_Modules.append(kegg_module)
|
|
115
|
+
|
|
116
|
+
KEGG_Reaction = line_data[14]
|
|
117
|
+
if KEGG_Reaction != '-':
|
|
118
|
+
kegg_reaction_list = KEGG_Reaction.split(',')
|
|
119
|
+
for kegg_reaction in kegg_reaction_list:
|
|
120
|
+
functions[dir]['KEGG_Reaction'][kegg_reaction] += gene_read_count
|
|
121
|
+
if kegg_reaction not in all_KEGG_Reactions:
|
|
122
|
+
all_KEGG_Reactions.append(kegg_reaction)
|
|
123
|
+
|
|
124
|
+
KEGG_rclass = line_data[15]
|
|
125
|
+
if KEGG_rclass != '-':
|
|
126
|
+
kegg_rclass_list = KEGG_rclass.split(',')
|
|
127
|
+
for kegg_rclass in kegg_rclass_list:
|
|
128
|
+
functions[dir]['KEGG_rclass'][kegg_rclass] += gene_read_count
|
|
129
|
+
if kegg_rclass not in all_KEGG_rclasses:
|
|
130
|
+
all_KEGG_rclasses.append(kegg_rclass)
|
|
131
|
+
|
|
132
|
+
BRITE = line_data[16]
|
|
133
|
+
if BRITE != '-':
|
|
134
|
+
brite_list = BRITE.split(',')
|
|
135
|
+
for brite in brite_list:
|
|
136
|
+
functions[dir]['BRITE'][brite] += gene_read_count
|
|
137
|
+
if brite not in all_BRITEs:
|
|
138
|
+
all_BRITEs.append(brite)
|
|
139
|
+
|
|
140
|
+
KEGG_TC = line_data[17]
|
|
141
|
+
if KEGG_TC != '-':
|
|
142
|
+
kegg_tc_list = KEGG_TC.split(',')
|
|
143
|
+
for kegg_tc in kegg_tc_list:
|
|
144
|
+
functions[dir]['KEGG_TC'][kegg_tc] += gene_read_count
|
|
145
|
+
if kegg_tc not in all_KEGG_TCs:
|
|
146
|
+
all_KEGG_TCs.append(kegg_tc)
|
|
147
|
+
|
|
148
|
+
CAZy = line_data[18]
|
|
149
|
+
if CAZy != '-':
|
|
150
|
+
cazy_list = CAZy.split(',')
|
|
151
|
+
for cazy in cazy_list:
|
|
152
|
+
functions[dir]['CAZy'][cazy] += gene_read_count
|
|
153
|
+
if cazy not in all_CAZys:
|
|
154
|
+
all_CAZys.append(cazy)
|
|
155
|
+
|
|
156
|
+
BiGG_Reaction = line_data[19]
|
|
157
|
+
if BiGG_Reaction != '-':
|
|
158
|
+
bigg_reaction_list = BiGG_Reaction.split(',')
|
|
159
|
+
for bigg_reaction in bigg_reaction_list:
|
|
160
|
+
functions[dir]['BiGG_Reaction'][bigg_reaction] += gene_read_count
|
|
161
|
+
if bigg_reaction not in all_BiGG_Reactions:
|
|
162
|
+
all_BiGG_Reactions.append(bigg_reaction)
|
|
163
|
+
|
|
164
|
+
PFAMs = line_data[20]
|
|
165
|
+
if PFAMs != '-':
|
|
166
|
+
pfams_list = PFAMs.split(',')
|
|
167
|
+
for pfam in pfams_list:
|
|
168
|
+
functions[dir]['PFAM'][pfam] += gene_read_count
|
|
169
|
+
if pfam not in all_PFAMs:
|
|
170
|
+
all_PFAMs.append(pfam)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
print("Sample " + dir + " Done")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
except FileNotFoundError as e:
|
|
178
|
+
print(e)
|
|
179
|
+
|
|
180
|
+
functions = dict(sorted(functions.items()))
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def write_transposed_output(output_path, category, all_items, functions):
|
|
184
|
+
output = open(output_path, 'w')
|
|
185
|
+
output.write('\t' + '\t'.join(functions.keys()) + '\n')
|
|
186
|
+
|
|
187
|
+
for item in all_items:
|
|
188
|
+
row_values = [item] + [str(functions[sample][category][item]) for sample in functions.keys()]
|
|
189
|
+
output.write('\t'.join(row_values) + '\n')
|
|
190
|
+
|
|
191
|
+
output.close()
|
|
192
|
+
|
|
193
|
+
output_path = os.path.join(parent_directory_path, 'CDS_Final_Outputs')
|
|
194
|
+
if not os.path.exists(output_path):
|
|
195
|
+
os.makedirs(output_path)
|
|
196
|
+
|
|
197
|
+
# COG output
|
|
198
|
+
write_transposed_output(output_path + '/Final_CDS_COG.tsv', 'COG', all_COGs, functions)
|
|
199
|
+
|
|
200
|
+
# GO output
|
|
201
|
+
write_transposed_output(output_path + '/Final_CDS_GO.tsv', 'GO', all_GOs, functions)
|
|
202
|
+
|
|
203
|
+
# EC output
|
|
204
|
+
write_transposed_output(output_path + '/Final_CDS_EC.tsv', 'EC', all_ECs, functions)
|
|
205
|
+
|
|
206
|
+
# KEGG_ko output
|
|
207
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_ko.tsv', 'KEGG_ko', all_KEGG_kos, functions)
|
|
208
|
+
|
|
209
|
+
# KEGG_Pathway output
|
|
210
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Pathway.tsv', 'KEGG_Pathway', all_KEGG_Pathways, functions)
|
|
211
|
+
|
|
212
|
+
# KEGG_Module output
|
|
213
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Module.tsv', 'KEGG_Module', all_KEGG_Modules, functions)
|
|
214
|
+
|
|
215
|
+
# KEGG_Reaction output
|
|
216
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_Reaction.tsv', 'KEGG_Reaction', all_KEGG_Reactions, functions)
|
|
217
|
+
|
|
218
|
+
# KEGG_rclass output
|
|
219
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_rclass.tsv', 'KEGG_rclass', all_KEGG_rclasses, functions)
|
|
220
|
+
|
|
221
|
+
# BRITE output
|
|
222
|
+
write_transposed_output(output_path + '/Final_CDS_BRITE.tsv', 'BRITE', all_BRITEs, functions)
|
|
223
|
+
|
|
224
|
+
# KEGG_TC output
|
|
225
|
+
write_transposed_output(output_path + '/Final_CDS_KEGG_TC.tsv', 'KEGG_TC', all_KEGG_TCs, functions)
|
|
226
|
+
|
|
227
|
+
# CAZy output
|
|
228
|
+
write_transposed_output(output_path + '/Final_CDS_CAZy.tsv', 'CAZy', all_CAZys, functions)
|
|
229
|
+
|
|
230
|
+
# BiGG_Reaction output
|
|
231
|
+
write_transposed_output(output_path + '/Final_CDS_BiGG_Reaction.tsv', 'BiGG_Reaction', all_BiGG_Reactions, functions)
|
|
232
|
+
|
|
233
|
+
# PFAMs output
|
|
234
|
+
write_transposed_output(output_path + '/Final_CDS_PFAMs.tsv', 'PFAM', all_PFAMs, functions)
|
|
235
|
+
|