MetaPont 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,95 @@
1
+ import argparse
2
+ import os
3
+ import csv
4
+ import sys
5
+ from collections import Counter
6
+
7
+ from HuwsLab.MetaPont.src.MetaPont.constants import MetaPont_Version
8
+
9
+ # Needed to account for the large CSV/TSV files we are working with
10
+ csv.field_size_limit(sys.maxsize)
11
+
12
+
13
+ def process_tsv(file_path, function_id):
14
+ """
15
+ Processes a TSV file to calculate taxa counts and total matches for a given function ID.
16
+ """
17
+ taxa_counts = Counter()
18
+ total_matches = 0
19
+
20
+ with open(file_path, "r") as tsv_file:
21
+ reader = csv.reader(tsv_file, delimiter="\t")
22
+ next(reader) # Skip the first row (sample name)
23
+ headers = next(reader) # Read headers from the second row
24
+
25
+ # Locate the Lineage column
26
+ taxa_idx = headers.index("Lineage")
27
+
28
+ # Process each row to find matches
29
+ for idx, row in enumerate(reader):
30
+ if len(row) < len(headers):
31
+ continue # Skip malformed rows
32
+
33
+ lineage = row[taxa_idx]
34
+
35
+ # Search functional columns starting from column 6
36
+ for cell in row[6:]:
37
+ if cell and any(function_id in part for part in cell.replace(',', '|').split('|')):
38
+ # Extract genus from the Lineage field
39
+ if "g__" in lineage:
40
+ genus = lineage.split("g__")[1].split("|")[0]
41
+ taxa_counts[genus] += 1
42
+ total_matches += 1
43
+ break # Stop checking further functional columns for this row
44
+
45
+ return taxa_counts, total_matches
46
+
47
+
48
+ def main():
49
+
50
+ parser = argparse.ArgumentParser(description='MetaPont ' + MetaPont_Version + ': Extract-By-Function - Identify taxa contributing to a specific function.')
51
+ parser.add_argument(
52
+ "-d", "--directory", required=True,
53
+ help="Directory containing TSV files to analyse."
54
+ )
55
+ parser.add_argument(
56
+ "-f", "--function_id", required=True,
57
+ help="Specific function ID to search for (e.g., 'GO:0002')."
58
+ )
59
+ parser.add_argument(
60
+ "-o", "--output", default="output_taxa_proportions.tsv",
61
+ help="Output file to save results (default: output_taxa_proportions.tsv)."
62
+ )
63
+ parser.add_argument(
64
+ "-m", "--min_proportion", type=float, default=0.05,
65
+ help="Minimum proportion threshold for taxa to be included in the output (default: 0.05)."
66
+ )
67
+
68
+ options = parser.parse_args()
69
+ print("Running MetaPont: Extract-By-Function " + MetaPont_Version)
70
+
71
+ all_results = {}
72
+
73
+ # Process each TSV file in the directory
74
+ for file_name in os.listdir(options.directory):
75
+ if file_name.endswith("_Final_Contig.tsv"):
76
+ file_path = os.path.join(options.directory, file_name)
77
+ print(f"Processing file: {file_name}")
78
+ taxa_counts, total_matches = process_tsv(file_path, options.function_id)
79
+ all_results[file_name] = (taxa_counts, total_matches)
80
+
81
+ # Write results to output
82
+ with open(options.output, "w") as out:
83
+ out.write("Function ID: " + options.function_id + "\n")
84
+ out.write("Sample\tTaxa\tProportion\n")
85
+ for sample, (taxa_counts, total_matches) in all_results.items():
86
+ for taxa, count in taxa_counts.items():
87
+ proportion = count / total_matches if total_matches > 0 else 0
88
+ if proportion >= options.min_proportion: # Apply minimum proportion filter
89
+ out.write(f"{sample}\t{taxa}\t{proportion:.6f}\n")
90
+
91
+ print(f"Results saved to {options.output}")
92
+
93
+
94
+ if __name__ == "__main__":
95
+ main()
@@ -0,0 +1,304 @@
1
+ import os
2
+ from collections import defaultdict
3
+ import glob
4
+ import argparse
5
+ import sys
6
+ import collections
7
+ ################
8
+
9
+
10
+
11
+ def read_files(files_list):
12
+ all_entries = []
13
+ read_counts = collections.defaultdict(lambda: collections.defaultdict(lambda: collections.defaultdict(int)))
14
+
15
+ total_reads = collections.defaultdict(int)
16
+ for file_path in files_list:
17
+ sample = file_path.split('/')[-1].split('_Final')[0] # _Final is specific to this dataset
18
+ with open(file_path, 'r') as file:
19
+ for line in file:
20
+ line = line.replace('\n','')
21
+ if line.startswith('NODE'): # Only works for this dataset/metaspades
22
+ line_data = line.split('\t')
23
+ mapped_reads = int(line_data[3])
24
+ total_reads[sample] = int(line_data[2])
25
+ lineage = line_data[5]
26
+ domain = lineage.split('|')[0]
27
+ ## COGs
28
+ COGs = line_data[6]
29
+ if COGs != '-':
30
+ cog_list = [cog for cog in COGs]
31
+ for cog in cog_list:
32
+ if cog != '|':
33
+ read_counts[sample][domain][cog] += mapped_reads
34
+
35
+ else:
36
+ if line.startswith('Contig'):
37
+ columns = line.split('\t')
38
+ return all_entries, read_counts, total_reads
39
+
40
+ ################
41
+
42
+ parent_directory_path = sys.argv[1]
43
+
44
+ all_COGs = []
45
+ all_GOs = []
46
+ all_ECs = []
47
+ all_KEGG_kos = []
48
+ all_KEGG_Pathways = []
49
+ all_KEGG_Modules = []
50
+ all_KEGG_Reactions = []
51
+ all_KEGG_rclasses = []
52
+ all_BRITEs = []
53
+ all_KEGG_TCs = []
54
+ all_CAZys =[]
55
+ all_BiGG_Reactions = []
56
+ all_PFAMs = []
57
+
58
+ functions = defaultdict(lambda: defaultdict(lambda: defaultdict(int)))
59
+
60
+
61
+ for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
62
+ for dir in dirs:
63
+ try:
64
+ current_dir = os.path.basename(dir)
65
+ parent_of_current_dir = os.path.basename(subdir)
66
+ if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
67
+ print(current_dir)
68
+ specific_dir_path = os.path.join(subdir, dir)
69
+ first_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_cds_summary.txt")
70
+ second_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
71
+
72
+ if os.path.exists(first_file_path) and os.path.exists(second_file_path):
73
+ print(f"Master directory: {specific_dir_path}")
74
+ print(f"First file path: {first_file_path}")
75
+ print(f"Second file path: {second_file_path}")
76
+
77
+ genes = defaultdict(int)
78
+ with open(first_file_path, 'r') as readmap_cds_in:
79
+ for line in readmap_cds_in:
80
+ reads = line.split()[0]
81
+ gene = line.split()[1].replace('ID=','')
82
+ contig_length = int(gene.split('length_')[1].split('_cov')[0])
83
+ if contig_length >= 2500:
84
+ genes.update({gene:reads})
85
+
86
+
87
+
88
+
89
+ lineages = {}
90
+ taxa_ids = []
91
+
92
+ with open(second_file_path, 'r') as emapper_in:
93
+ for line in emapper_in:
94
+ if not line.startswith('#'):
95
+ line_data = line.strip().split('\t')
96
+ gene = line_data[0]
97
+ if gene in genes:
98
+ gene_read_count = int(genes[gene])
99
+
100
+ COGs = line_data[6]
101
+ if COGs != '-':
102
+ #cog_list = COGs.split(',')
103
+ cog_list = [cog for cog in COGs]
104
+ for cog in cog_list:
105
+ functions[dir]['COG'][cog] += gene_read_count
106
+ if cog not in all_COGs:
107
+ all_COGs.append(cog)
108
+
109
+ GOs = line_data[9]
110
+ if GOs != '-':
111
+ go_list = GOs.split(',')
112
+ for go in go_list:
113
+ functions[dir]['GO'][go] += gene_read_count
114
+ if go not in all_GOs:
115
+ all_GOs.append(go)
116
+
117
+ EC = line_data[10]
118
+ if EC != '-':
119
+ ec_list = EC.split(',')
120
+ for ec in ec_list:
121
+ functions[dir]['EC'][ec] += gene_read_count
122
+ if ec not in all_ECs:
123
+ all_ECs.append(ec)
124
+
125
+ KEGG_ko = line_data[11]
126
+ if KEGG_ko != '-':
127
+ kegg_ko_list = KEGG_ko.split(',')
128
+ for kegg_ko in kegg_ko_list:
129
+ functions[dir]['KEGG_ko'][kegg_ko] += gene_read_count
130
+ if kegg_ko not in all_KEGG_kos:
131
+ all_KEGG_kos.append(kegg_ko)
132
+
133
+ KEGG_Pathway = line_data[12]
134
+ if KEGG_Pathway != '-':
135
+ kegg_pathway_list = KEGG_Pathway.split(',')
136
+ for kegg_pathway in kegg_pathway_list:
137
+ functions[dir]['KEGG_Pathway'][kegg_pathway] += gene_read_count
138
+ if kegg_pathway not in all_KEGG_Pathways:
139
+ all_KEGG_Pathways.append(kegg_pathway)
140
+
141
+ KEGG_Module = line_data[13]
142
+ if KEGG_Module != '-':
143
+ kegg_module_list = KEGG_Module.split(',')
144
+ for kegg_module in kegg_module_list:
145
+ functions[dir]['KEGG_Module'][kegg_module] += gene_read_count
146
+ if kegg_module not in all_KEGG_Modules:
147
+ all_KEGG_Modules.append(kegg_module)
148
+
149
+ KEGG_Reaction = line_data[14]
150
+ if KEGG_Reaction != '-':
151
+ kegg_reaction_list = KEGG_Reaction.split(',')
152
+ for kegg_reaction in kegg_reaction_list:
153
+ functions[dir]['KEGG_Reaction'][kegg_reaction] += gene_read_count
154
+ if kegg_reaction not in all_KEGG_Reactions:
155
+ all_KEGG_Reactions.append(kegg_reaction)
156
+
157
+ KEGG_rclass = line_data[15]
158
+ if KEGG_rclass != '-':
159
+ kegg_rclass_list = KEGG_rclass.split(',')
160
+ for kegg_rclass in kegg_rclass_list:
161
+ functions[dir]['KEGG_rclass'][kegg_rclass] += gene_read_count
162
+ if kegg_rclass not in all_KEGG_rclasses:
163
+ all_KEGG_rclasses.append(kegg_rclass)
164
+
165
+ BRITE = line_data[16]
166
+ if BRITE != '-':
167
+ brite_list = BRITE.split(',')
168
+ for brite in brite_list:
169
+ functions[dir]['BRITE'][brite] += gene_read_count
170
+ if brite not in all_BRITEs:
171
+ all_BRITEs.append(brite)
172
+
173
+ KEGG_TC = line_data[17]
174
+ if KEGG_TC != '-':
175
+ kegg_tc_list = KEGG_TC.split(',')
176
+ for kegg_tc in kegg_tc_list:
177
+ functions[dir]['KEGG_TC'][kegg_tc] += gene_read_count
178
+ if kegg_tc not in all_KEGG_TCs:
179
+ all_KEGG_TCs.append(kegg_tc)
180
+
181
+ CAZy = line_data[18]
182
+ if CAZy != '-':
183
+ cazy_list = CAZy.split(',')
184
+ for cazy in cazy_list:
185
+ functions[dir]['CAZy'][cazy] += gene_read_count
186
+ if cazy not in all_CAZys:
187
+ all_CAZys.append(cazy)
188
+
189
+ BiGG_Reaction = line_data[19]
190
+ if BiGG_Reaction != '-':
191
+ bigg_reaction_list = BiGG_Reaction.split(',')
192
+ for bigg_reaction in bigg_reaction_list:
193
+ functions[dir]['BiGG_Reaction'][bigg_reaction] += gene_read_count
194
+ if bigg_reaction not in all_BiGG_Reactions:
195
+ all_BiGG_Reactions.append(bigg_reaction)
196
+
197
+ PFAMs = line_data[20]
198
+ if PFAMs != '-':
199
+ pfams_list = PFAMs.split(',')
200
+ for pfam in pfams_list:
201
+ functions[dir]['PFAM'][pfam] += gene_read_count
202
+ if pfam not in all_PFAMs:
203
+ all_PFAMs.append(pfam)
204
+
205
+
206
+ print("Sample " + dir + " Done")
207
+
208
+
209
+
210
+ except FileNotFoundError as e:
211
+ print(e)
212
+
213
+ functions = dict(sorted(functions.items()))
214
+
215
+
216
+ def write_transposed_output(output_path, category, all_items, functions):
217
+ output = open(output_path, 'w')
218
+ output.write('\t' + '\t'.join(functions.keys()) + '\n')
219
+
220
+ for item in all_items:
221
+ row_values = [item] + [str(functions[sample][category][item]) for sample in functions.keys()]
222
+ output.write('\t'.join(row_values) + '\n')
223
+
224
+ output.close()
225
+
226
+ output_path = os.path.join(parent_directory_path, 'CDS_Final_Outputs')
227
+ if not os.path.exists(output_path):
228
+ os.makedirs(output_path)
229
+
230
+ # COG output
231
+ write_transposed_output(output_path + '/Final_CDS_COG.tsv', 'COG', all_COGs, functions)
232
+
233
+ # GO output
234
+ write_transposed_output(output_path + '/Final_CDS_GO.tsv', 'GO', all_GOs, functions)
235
+
236
+ # EC output
237
+ write_transposed_output(output_path + '/Final_CDS_EC.tsv', 'EC', all_ECs, functions)
238
+
239
+ # KEGG_ko output
240
+ write_transposed_output(output_path + '/Final_CDS_KEGG_ko.tsv', 'KEGG_ko', all_KEGG_kos, functions)
241
+
242
+ # KEGG_Pathway output
243
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Pathway.tsv', 'KEGG_Pathway', all_KEGG_Pathways, functions)
244
+
245
+ # KEGG_Module output
246
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Module.tsv', 'KEGG_Module', all_KEGG_Modules, functions)
247
+
248
+ # KEGG_Reaction output
249
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Reaction.tsv', 'KEGG_Reaction', all_KEGG_Reactions, functions)
250
+
251
+ # KEGG_rclass output
252
+ write_transposed_output(output_path + '/Final_CDS_KEGG_rclass.tsv', 'KEGG_rclass', all_KEGG_rclasses, functions)
253
+
254
+ # BRITE output
255
+ write_transposed_output(output_path + '/Final_CDS_BRITE.tsv', 'BRITE', all_BRITEs, functions)
256
+
257
+ # KEGG_TC output
258
+ write_transposed_output(output_path + '/Final_CDS_KEGG_TC.tsv', 'KEGG_TC', all_KEGG_TCs, functions)
259
+
260
+ # CAZy output
261
+ write_transposed_output(output_path + '/Final_CDS_CAZy.tsv', 'CAZy', all_CAZys, functions)
262
+
263
+ # BiGG_Reaction output
264
+ write_transposed_output(output_path + '/Final_CDS_BiGG_Reaction.tsv', 'BiGG_Reaction', all_BiGG_Reactions, functions)
265
+
266
+ # PFAMs output
267
+ write_transposed_output(output_path + '/Final_CDS_PFAMs.tsv', 'PFAM', all_PFAMs, functions)
268
+
269
+
270
+ def main():
271
+
272
+ parser = argparse.ArgumentParser(description='....')
273
+ parser._action_groups.pop()
274
+
275
+ required = parser.add_argument_group('Required Arguments')
276
+ required.add_argument('-d', action='store', dest='dir_path', required=True,
277
+ help='Define the directory path containing the files')
278
+ required.add_argument('-o', action='store', dest='output', help='Outdir',
279
+ required=True)
280
+
281
+ options = parser.parse_args()
282
+
283
+ # Use glob to find files ending with '_Final_Output.tsv'
284
+ files_list = glob.glob(f"{options.dir_path}/*_Final_*.tsv")
285
+ all_entries, read_counts, total_reads = read_files(files_list)
286
+
287
+ separate_taxa = ['d__Bacteria', 'd__Archaea', 'd__Eukaryota', 'd__Viruses','k__Fungi',
288
+ 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown']
289
+ remove_taxa = ['c__Mammalia','k__Viridiplantae'
290
+ ,'d__Eukaryota|k__unknown|p__Evosea|c__Eumycetozoa|o__Dictyosteliales|f__Dictyosteliaceae|g__Dictyostelium'
291
+ ,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Leishmania'
292
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Toxoplasma'
293
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Haemosporida|f__Plasmodiidae|g__Plasmodium'
294
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Piroplasmida|f__Theileriidae|g__Theileria'
295
+ ,'d__Eukaryota|k__unknown|p__Parabasalia|c__unknown|o__Trichomonadida|f__Trichomonadidae|g__Trichomonas'
296
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Besnoitia'
297
+ ,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Trypanosoma'
298
+ ,'d__Eukaryota|k__unknown|p__Ciliophora|c__Oligohymenophorea|o__Peniculida|f__Parameciidae|g__Paramecium'
299
+ ,'d__Eukaryota|k__Fungi|p__Microsporidia|c__unknown|o__unknown|f__Unikaryonidae|g__Encephalitozoon'
300
+ ,'d__Eukaryota|k__unknown|p__unknown|c__Cryptophyceae|o__Cryptomonadales|f__Cryptomonadaceae|g__Cryptomonas']
301
+
302
+ if __name__ == "__main__":
303
+ main()
304
+ print("Complete")
MetaPont/__init__.py ADDED
File without changes
MetaPont/constants.py ADDED
@@ -0,0 +1,2 @@
1
+ MetaPont_Version = 'v0.0.1'
2
+
@@ -0,0 +1,235 @@
1
+ import os
2
+ from collections import defaultdict
3
+ import re
4
+ import sys
5
+
6
+
7
+ ################
8
+
9
+ parent_directory_path = sys.argv[1]
10
+
11
+ all_COGs = []
12
+ all_GOs = []
13
+ all_ECs = []
14
+ all_KEGG_kos = []
15
+ all_KEGG_Pathways = []
16
+ all_KEGG_Modules = []
17
+ all_KEGG_Reactions = []
18
+ all_KEGG_rclasses = []
19
+ all_BRITEs = []
20
+ all_KEGG_TCs = []
21
+ all_CAZys =[]
22
+ all_BiGG_Reactions = []
23
+ all_PFAMs = []
24
+
25
+ functions = defaultdict(lambda: defaultdict(lambda: defaultdict(int)))
26
+
27
+
28
+ for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
29
+ for dir in dirs:
30
+ try:
31
+ current_dir = os.path.basename(dir)
32
+ parent_of_current_dir = os.path.basename(subdir)
33
+ if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
34
+ print(current_dir)
35
+ specific_dir_path = os.path.join(subdir, dir)
36
+ first_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_cds_summary.txt")
37
+ second_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
38
+
39
+ if os.path.exists(first_file_path) and os.path.exists(second_file_path):
40
+ print(f"Master directory: {specific_dir_path}")
41
+ print(f"First file path: {first_file_path}")
42
+ print(f"Second file path: {second_file_path}")
43
+
44
+ genes = defaultdict(int)
45
+ with open(first_file_path, 'r') as readmap_cds_in:
46
+ for line in readmap_cds_in:
47
+ reads = line.split()[0]
48
+ gene = line.split()[1].replace('ID=','')
49
+ contig_length = int(gene.split('length_')[1].split('_cov')[0])
50
+ if contig_length >= 2500:
51
+ genes.update({gene:reads})
52
+
53
+
54
+
55
+
56
+ lineages = {}
57
+ taxa_ids = []
58
+
59
+ with open(second_file_path, 'r') as emapper_in:
60
+ for line in emapper_in:
61
+ if not line.startswith('#'):
62
+ line_data = line.strip().split('\t')
63
+ gene = line_data[0]
64
+ if gene in genes:
65
+ gene_read_count = int(genes[gene])
66
+
67
+ COGs = line_data[6]
68
+ if COGs != '-':
69
+ #cog_list = COGs.split(',')
70
+ cog_list = [cog for cog in COGs]
71
+ for cog in cog_list:
72
+ functions[dir]['COG'][cog] += gene_read_count
73
+ if cog not in all_COGs:
74
+ all_COGs.append(cog)
75
+
76
+ GOs = line_data[9]
77
+ if GOs != '-':
78
+ go_list = GOs.split(',')
79
+ for go in go_list:
80
+ functions[dir]['GO'][go] += gene_read_count
81
+ if go not in all_GOs:
82
+ all_GOs.append(go)
83
+
84
+ EC = line_data[10]
85
+ if EC != '-':
86
+ ec_list = EC.split(',')
87
+ for ec in ec_list:
88
+ functions[dir]['EC'][ec] += gene_read_count
89
+ if ec not in all_ECs:
90
+ all_ECs.append(ec)
91
+
92
+ KEGG_ko = line_data[11]
93
+ if KEGG_ko != '-':
94
+ kegg_ko_list = KEGG_ko.split(',')
95
+ for kegg_ko in kegg_ko_list:
96
+ functions[dir]['KEGG_ko'][kegg_ko] += gene_read_count
97
+ if kegg_ko not in all_KEGG_kos:
98
+ all_KEGG_kos.append(kegg_ko)
99
+
100
+ KEGG_Pathway = line_data[12]
101
+ if KEGG_Pathway != '-':
102
+ kegg_pathway_list = KEGG_Pathway.split(',')
103
+ for kegg_pathway in kegg_pathway_list:
104
+ functions[dir]['KEGG_Pathway'][kegg_pathway] += gene_read_count
105
+ if kegg_pathway not in all_KEGG_Pathways:
106
+ all_KEGG_Pathways.append(kegg_pathway)
107
+
108
+ KEGG_Module = line_data[13]
109
+ if KEGG_Module != '-':
110
+ kegg_module_list = KEGG_Module.split(',')
111
+ for kegg_module in kegg_module_list:
112
+ functions[dir]['KEGG_Module'][kegg_module] += gene_read_count
113
+ if kegg_module not in all_KEGG_Modules:
114
+ all_KEGG_Modules.append(kegg_module)
115
+
116
+ KEGG_Reaction = line_data[14]
117
+ if KEGG_Reaction != '-':
118
+ kegg_reaction_list = KEGG_Reaction.split(',')
119
+ for kegg_reaction in kegg_reaction_list:
120
+ functions[dir]['KEGG_Reaction'][kegg_reaction] += gene_read_count
121
+ if kegg_reaction not in all_KEGG_Reactions:
122
+ all_KEGG_Reactions.append(kegg_reaction)
123
+
124
+ KEGG_rclass = line_data[15]
125
+ if KEGG_rclass != '-':
126
+ kegg_rclass_list = KEGG_rclass.split(',')
127
+ for kegg_rclass in kegg_rclass_list:
128
+ functions[dir]['KEGG_rclass'][kegg_rclass] += gene_read_count
129
+ if kegg_rclass not in all_KEGG_rclasses:
130
+ all_KEGG_rclasses.append(kegg_rclass)
131
+
132
+ BRITE = line_data[16]
133
+ if BRITE != '-':
134
+ brite_list = BRITE.split(',')
135
+ for brite in brite_list:
136
+ functions[dir]['BRITE'][brite] += gene_read_count
137
+ if brite not in all_BRITEs:
138
+ all_BRITEs.append(brite)
139
+
140
+ KEGG_TC = line_data[17]
141
+ if KEGG_TC != '-':
142
+ kegg_tc_list = KEGG_TC.split(',')
143
+ for kegg_tc in kegg_tc_list:
144
+ functions[dir]['KEGG_TC'][kegg_tc] += gene_read_count
145
+ if kegg_tc not in all_KEGG_TCs:
146
+ all_KEGG_TCs.append(kegg_tc)
147
+
148
+ CAZy = line_data[18]
149
+ if CAZy != '-':
150
+ cazy_list = CAZy.split(',')
151
+ for cazy in cazy_list:
152
+ functions[dir]['CAZy'][cazy] += gene_read_count
153
+ if cazy not in all_CAZys:
154
+ all_CAZys.append(cazy)
155
+
156
+ BiGG_Reaction = line_data[19]
157
+ if BiGG_Reaction != '-':
158
+ bigg_reaction_list = BiGG_Reaction.split(',')
159
+ for bigg_reaction in bigg_reaction_list:
160
+ functions[dir]['BiGG_Reaction'][bigg_reaction] += gene_read_count
161
+ if bigg_reaction not in all_BiGG_Reactions:
162
+ all_BiGG_Reactions.append(bigg_reaction)
163
+
164
+ PFAMs = line_data[20]
165
+ if PFAMs != '-':
166
+ pfams_list = PFAMs.split(',')
167
+ for pfam in pfams_list:
168
+ functions[dir]['PFAM'][pfam] += gene_read_count
169
+ if pfam not in all_PFAMs:
170
+ all_PFAMs.append(pfam)
171
+
172
+
173
+ print("Sample " + dir + " Done")
174
+
175
+
176
+
177
+ except FileNotFoundError as e:
178
+ print(e)
179
+
180
+ functions = dict(sorted(functions.items()))
181
+
182
+
183
+ def write_transposed_output(output_path, category, all_items, functions):
184
+ output = open(output_path, 'w')
185
+ output.write('\t' + '\t'.join(functions.keys()) + '\n')
186
+
187
+ for item in all_items:
188
+ row_values = [item] + [str(functions[sample][category][item]) for sample in functions.keys()]
189
+ output.write('\t'.join(row_values) + '\n')
190
+
191
+ output.close()
192
+
193
+ output_path = os.path.join(parent_directory_path, 'CDS_Final_Outputs')
194
+ if not os.path.exists(output_path):
195
+ os.makedirs(output_path)
196
+
197
+ # COG output
198
+ write_transposed_output(output_path + '/Final_CDS_COG.tsv', 'COG', all_COGs, functions)
199
+
200
+ # GO output
201
+ write_transposed_output(output_path + '/Final_CDS_GO.tsv', 'GO', all_GOs, functions)
202
+
203
+ # EC output
204
+ write_transposed_output(output_path + '/Final_CDS_EC.tsv', 'EC', all_ECs, functions)
205
+
206
+ # KEGG_ko output
207
+ write_transposed_output(output_path + '/Final_CDS_KEGG_ko.tsv', 'KEGG_ko', all_KEGG_kos, functions)
208
+
209
+ # KEGG_Pathway output
210
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Pathway.tsv', 'KEGG_Pathway', all_KEGG_Pathways, functions)
211
+
212
+ # KEGG_Module output
213
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Module.tsv', 'KEGG_Module', all_KEGG_Modules, functions)
214
+
215
+ # KEGG_Reaction output
216
+ write_transposed_output(output_path + '/Final_CDS_KEGG_Reaction.tsv', 'KEGG_Reaction', all_KEGG_Reactions, functions)
217
+
218
+ # KEGG_rclass output
219
+ write_transposed_output(output_path + '/Final_CDS_KEGG_rclass.tsv', 'KEGG_rclass', all_KEGG_rclasses, functions)
220
+
221
+ # BRITE output
222
+ write_transposed_output(output_path + '/Final_CDS_BRITE.tsv', 'BRITE', all_BRITEs, functions)
223
+
224
+ # KEGG_TC output
225
+ write_transposed_output(output_path + '/Final_CDS_KEGG_TC.tsv', 'KEGG_TC', all_KEGG_TCs, functions)
226
+
227
+ # CAZy output
228
+ write_transposed_output(output_path + '/Final_CDS_CAZy.tsv', 'CAZy', all_CAZys, functions)
229
+
230
+ # BiGG_Reaction output
231
+ write_transposed_output(output_path + '/Final_CDS_BiGG_Reaction.tsv', 'BiGG_Reaction', all_BiGG_Reactions, functions)
232
+
233
+ # PFAMs output
234
+ write_transposed_output(output_path + '/Final_CDS_PFAMs.tsv', 'PFAM', all_PFAMs, functions)
235
+