MetaPont 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,222 @@
1
+ import os
2
+ from collections import defaultdict
3
+ import re
4
+ import sys
5
+
6
+ ################
7
+
8
+ parent_directory_path = sys.argv[1]
9
+
10
+
11
+ for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
12
+ for dir in dirs:
13
+ try:
14
+ current_dir = os.path.basename(dir)
15
+ parent_of_current_dir = os.path.basename(subdir)
16
+ if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
17
+ print(current_dir)
18
+ specific_dir_path = os.path.join(subdir, dir)
19
+ first_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
20
+ second_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2_report.txt")
21
+ third_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2_report_mpa.txt")
22
+ forth_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2.txt")
23
+ fifth_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_contig_summary.txt")
24
+
25
+ if os.path.exists(first_file_path) and os.path.exists(second_file_path) and os.path.exists(third_file_path) and os.path.exists(forth_file_path) and os.path.exists(fifth_file_path):
26
+ print(f"Master directory: {specific_dir_path}")
27
+ print(f"First file path: {first_file_path}")
28
+ print(f"Second file path: {second_file_path}")
29
+ print(f"Third file path: {third_file_path}")
30
+ print(f"Forth file path: {forth_file_path}")
31
+ print(f"Fifth file path: {fifth_file_path}")
32
+
33
+ contigs = defaultdict(list)
34
+ genes = defaultdict(list)
35
+ lineages = {}
36
+ taxa_ids = []
37
+
38
+ with open(first_file_path, 'r') as emapper_in:
39
+ for line in emapper_in:
40
+ if not line.startswith('#'):
41
+ line_data = line.strip().split('\t')
42
+ contig = line_data[0].rsplit('_', 1)[0]
43
+ COGs = line_data[6]
44
+ GOs = line_data[9]
45
+ EC = line_data[10]
46
+ KEGG_ko = line_data[11]
47
+ KEGG_Pathway = line_data[12]
48
+ KEGG_Module = line_data[13]
49
+ KEGG_Reaction = line_data[14]
50
+ KEGG_rclass = line_data[15]
51
+ BRITE = line_data[16]
52
+ KEGG_TC = line_data[17]
53
+ CAZy = line_data[18]
54
+ BiGG_Reaction = line_data[19]
55
+ PFAMs = line_data[20]
56
+ genes[contig].append(
57
+ [COGs, GOs, EC, KEGG_ko, KEGG_Pathway, KEGG_Module, KEGG_Reaction, KEGG_rclass, BRITE,
58
+ KEGG_TC, CAZy, BiGG_Reaction, PFAMs])
59
+
60
+ with open(second_file_path, 'r') as f:
61
+ kraken2_lines = f.readlines()
62
+
63
+
64
+ with open(third_file_path, 'r') as kraken_lineage_in:
65
+ for line in kraken_lineage_in:
66
+ lineages_to_check = {'d__':'d__unknown', 'k__':'k__unknown', 'p__':'p__unknown', 'c__':'c__unknown',
67
+ 'o__':'o__unknown', 'f__':'f__unknown', 'g__':'g__unknown', 's__':'s__unknown'}
68
+ if line.startswith('d__'):
69
+ line_data = line.split('|')
70
+ contig_length = line_data[-1].split('\t')[1].strip() # check
71
+ line_data[-1] = line_data[-1].split('\t')[0] #check
72
+ # Check for missing lineages
73
+ for data in line_data:
74
+ first_three_chars = data[:3]
75
+ if first_three_chars in lineages_to_check:
76
+ lineages_to_check[first_three_chars] = data
77
+
78
+ combined = ''
79
+ for key, value in lineages_to_check.items():
80
+ #if value is not None and value != '' and '__unknown' not in value:
81
+ combined+=value+'|'
82
+ combined = combined[:-1]
83
+ #print(combined)
84
+
85
+ lineages[line.split('\t')[0].split('_')[-1]] = combined
86
+
87
+ counter = 0
88
+ with open(forth_file_path, 'r') as kraken_in:
89
+ for line in kraken_in:
90
+ counter +=1
91
+ line_data = line.strip().split('\t')
92
+ contig = line_data[1]
93
+ taxa_id = line_data[2].split(' (taxid')[0]
94
+ contig_length = int(line_data[3])
95
+ if contig_length >= 2500:
96
+ full_lineage = None
97
+ if taxa_id == 'unclassified' or taxa_id == 'root':
98
+ full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
99
+ else:
100
+ try:
101
+ full_lineage = lineages[taxa_id]
102
+ except KeyError:
103
+ while True:
104
+ taxa_id = taxa_id.rsplit(' ', 1)[0]
105
+ try:
106
+ full_lineage = lineages[taxa_id]
107
+ break
108
+ except KeyError:
109
+ tmp_size = taxa_id.split(' ')
110
+ if len(tmp_size) == 1:
111
+ taxa_id = line_data[2].split(' (taxid')[0]
112
+ escaped_pattern = re.escape(taxa_id)
113
+ matching_line = next((line for line in kraken2_lines if re.search(escaped_pattern, line)), None)
114
+ # Loop through the lines and search for the pattern
115
+
116
+ index = kraken2_lines.index(matching_line)
117
+ line_above = kraken2_lines[index - 1]
118
+ while True:
119
+ if not any(char.isdigit() for char in line_above.split('\t')[3]):
120
+ break
121
+ else:
122
+ index = kraken2_lines.index(line_above)
123
+ line_above = kraken2_lines[index - 1]
124
+ taxa_id = line_above.strip().split('\t')[-1].lstrip()
125
+ try:
126
+ full_lineage = lineages[taxa_id]
127
+ break
128
+ except KeyError:
129
+ if taxa_id == 'unclassified' or taxa_id == 'root':
130
+ full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
131
+ break
132
+ continue
133
+ if full_lineage == None:
134
+ full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
135
+ print("Contig without lineage. ")
136
+ contigs[contig] = [line_data[2],line_data[3],full_lineage]
137
+ #print(counter)
138
+ else:
139
+ print("Contig length less than 2,500")
140
+ break
141
+
142
+
143
+
144
+
145
+ print("RM")
146
+ with open(fifth_file_path, 'r') as readmapped_in:
147
+ for line in readmapped_in:
148
+ if not line.startswith('Contig'): #sloppy
149
+ line_data = line.strip().split('\t')
150
+ contig = line_data[0]
151
+ if contig in contigs:
152
+ info = [line_data[1],line_data[2]]
153
+ info.extend(contigs[contig])
154
+ contigs[contig] = info
155
+
156
+ final_dict = defaultdict(list)
157
+
158
+ for contig, data in contigs.items(): # Easy to read but not efficient at all
159
+ data.extend([[], [], [], [], [], [], [], [], [], [], [], [], []])
160
+ final_dict[contig] = data
161
+ #for gene in data[0]:
162
+ gene_info = genes[contig]
163
+ for current in gene_info:
164
+
165
+ try:
166
+ if current[0] != '-':
167
+ final_dict[contig][5].append(current[0])
168
+ if current[1] != '-':
169
+ final_dict[contig][6].append(current[1])
170
+ if current[2] != '-':
171
+ final_dict[contig][7].append(current[2])
172
+ if current[3] != '-':
173
+ final_dict[contig][8].append(current[3])
174
+ if current[4] != '-':
175
+ final_dict[contig][9].append(current[4])
176
+ if current[5] != '-':
177
+ final_dict[contig][10].append(current[5])
178
+ if current[6] != '-':
179
+ final_dict[contig][11].append(current[6])
180
+ if current[7] != '-':
181
+ final_dict[contig][12].append(current[7])
182
+ if current[8] != '-':
183
+ final_dict[contig][13].append(current[8])
184
+ if current[9] != '-':
185
+ final_dict[contig][14].append(current[9])
186
+ if current[10] != '-':
187
+ final_dict[contig][15].append(current[10])
188
+ if current[11] != '-':
189
+ final_dict[contig][16].append(current[11])
190
+ if current[12] != '-':
191
+ final_dict[contig][17].append(current[12])
192
+ if current[13] != '-':
193
+ final_dict[contig][18].append(current[13])
194
+ except IndexError:
195
+ continue
196
+
197
+ output_path = os.path.join(parent_directory_path, 'Contig_Final_Outputs')
198
+ if not os.path.exists(output_path):
199
+ os.makedirs(output_path)
200
+ output = open(os.path.join(output_path, dir + '_Final_Contig.tsv'),'w')
201
+
202
+
203
+ output.write(dir+'\n')
204
+ output.write('Contig\tContig_Length\tTotal_Reads\tMapped_Reads\tScientific_Name\tLineage\tCOG\tGO\tEC\tKEGG_KO\tKEGG_Pathway\tKEGG_Module\t'
205
+ 'KEGG_Reaction\tKEGG_rclass\tBRITE\tKEGG_TC\tCAZy\tBiGG_Reaction\tPFAMs\n')
206
+
207
+ for contig, data in final_dict.items():
208
+ try:
209
+ #print(contig)
210
+ output.write(contig+'\t'+data[3]+'\t'+data[1]+'\t'+data[0]+'\t'+data[2]+'\t'+data[4]+'\t'+'|'.join(data[5])+'\t'+'|'.join(data[6])+'\t'+'|'.join(data[7])+'\t'+'|'.join(data[8])
211
+ +'\t'+'|'.join(data[9])+'\t'+'|'.join(data[10])+'\t'+'|'.join(data[11])+'\t'+'|'.join(data[12])+'\t'+'|'.join(data[13])+'\t'+'|'.join(data[14])
212
+ +'\t'+'|'.join(data[15])+'\t'+'|'.join(data[16])+'\t'+'|'.join(data[17])+'\n')
213
+ except TypeError as e:
214
+ print("Some Error: " + e)
215
+
216
+ print("Done")
217
+
218
+
219
+ except FileNotFoundError as e:
220
+ print(f"Error processing directory {dir}: {e}")
221
+
222
+
@@ -0,0 +1,80 @@
1
+ import collections
2
+ import re
3
+ import glob
4
+
5
+
6
+ outfile = open('/mnt/Internal/Nextcloud/Collab/Zhenbin/2024_project/Contig_Final_Outputs/Lineage_Genus.tsv', 'w')
7
+
8
+ KO_reads = collections.defaultdict(lambda: collections.defaultdict(int))
9
+
10
+ # Define the directory path containing the files
11
+ directory_path = '/mnt/Internal/Nextcloud/Collab/Zhenbin/2024_project/Contig_Final_Outputs/'
12
+
13
+
14
+
15
+ all_KOs = []
16
+
17
+ # Iterate through each file found
18
+ for file_path in files_list:
19
+ sample = file_path.split('/')[-1].split('_Final')[0]
20
+ # Perform actions on each file (for example, read the file content)
21
+ with open(file_path, 'r') as file:
22
+ for line in file:
23
+ line = line.replace('\n','')
24
+ seen_KOs = []
25
+ if line.startswith('NODE'): # Only works for this dataset
26
+ line_data = line.split('\t')
27
+ mapped_reads = int(line_data[3])
28
+ lineage = line_data[5]
29
+ #for KO in KOs:
30
+ # if KO not in seen_KOs:
31
+ # KO_reads[sample][KO] += mapped_reads
32
+ # seen_KOs.append(KO)
33
+ genus = lineage.rsplit('|', 1)[0]
34
+ KO_reads[sample][genus] += mapped_reads
35
+
36
+ if genus not in all_KOs:
37
+ all_KOs.append(genus)
38
+ print("")
39
+
40
+
41
+ def extract_numeric_suffix(key):
42
+ match = re.search(r'\d+$', key)
43
+ return int(match.group()) if match else 0
44
+
45
+ # Reorder the dictionary based on the last numeric characters of the keys
46
+ KO_reads = dict(sorted(KO_reads.items(), key=lambda x: extract_numeric_suffix(x[0])))
47
+
48
+
49
+
50
+
51
+ keys_as_string = '\t'.join(KO_reads.keys())
52
+
53
+
54
+ outfile.write('Lineage_Genus\t'+keys_as_string+'\n')
55
+
56
+ for KO in all_KOs:
57
+ outfile.write(KO)
58
+ for sample in KO_reads.keys():
59
+ outfile.write('\t'+str(KO_reads[sample][KO]))
60
+ outfile.write('\n')
61
+
62
+
63
+
64
+
65
+
66
+
67
+
68
+
69
+
70
+
71
+
72
+
73
+
74
+
75
+
76
+
77
+
78
+
79
+
80
+
@@ -0,0 +1,121 @@
1
+ import collections
2
+ import re
3
+ import glob
4
+ import argparse
5
+ import os
6
+
7
+
8
+ def read_files(files_list):
9
+ all_entries = []
10
+ read_counts = collections.defaultdict(lambda: collections.defaultdict(int))
11
+ total_reads = collections.defaultdict(int)
12
+ for file_path in files_list:
13
+ sample = file_path.split('/')[-1].split('_Final')[0] # _Final is specific to this dataset
14
+ with open(file_path, 'r') as file:
15
+ for line in file:
16
+ line = line.replace('\n','')
17
+ if line.startswith('NODE'): # Only works for this dataset/metaspades
18
+ line_data = line.split('\t')
19
+ mapped_reads = int(line_data[3])
20
+ total_reads[sample] = int(line_data[2])
21
+ lineage = line_data[5]
22
+ genus = lineage.rsplit('|', 1)[0] # We are reporting only down to the genus level
23
+ read_counts[sample][genus] += mapped_reads
24
+ if genus not in all_entries:
25
+ all_entries.append(genus)
26
+ return all_entries, read_counts, total_reads
27
+
28
+
29
+ def write_out(output_dir, read_counts, total_reads):
30
+ for substr, counts in read_counts.items():
31
+ if substr == 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown':
32
+ substr = 'd__unknown'
33
+ output_file = os.path.join(output_dir, f"{substr}_output.tsv")
34
+ sample_names = sorted(total_reads.keys())
35
+ keys_as_string = '\t'.join(sample_names)
36
+ values_as_string = '\t'.join([str(total_reads[key]) for key in sample_names])
37
+
38
+
39
+ all_current_taxa = list({key for sample_counts in counts.values() for key in sample_counts})
40
+
41
+
42
+
43
+ with open(output_file, 'w') as outfile:
44
+ outfile.write('Lineage_Genus\t'+keys_as_string+'\n')
45
+ outfile.write('Total_Num_Reads')
46
+ outfile.write('\t' + str(values_as_string))
47
+ outfile.write('\n')
48
+ for current_taxa in all_current_taxa:
49
+ outfile.write(current_taxa)
50
+ for sample_key in sample_names:
51
+ outfile.write('\t' + str(counts[sample_key][current_taxa]))
52
+ outfile.write('\n')
53
+
54
+
55
+
56
+ def extract_numeric_suffix(key):
57
+ match = re.search(r'\d+$', key)
58
+ return int(match.group()) if match else 0
59
+
60
+ def group_by_substrings(read_counts, substrings, remove_substrings):
61
+ grouped_counts = {substr: collections.defaultdict(lambda: collections.defaultdict(int)) for substr in substrings}
62
+ for sample, counts in read_counts.items():
63
+ for key, value in counts.items():
64
+ if any(remove_substr in key for remove_substr in remove_substrings):
65
+ continue
66
+ for substr in substrings:
67
+ if substr in key:
68
+ grouped_counts[substr][sample][key] += value
69
+ return grouped_counts
70
+
71
+
72
+ def main():
73
+
74
+ parser = argparse.ArgumentParser(description='....')
75
+ parser._action_groups.pop()
76
+
77
+ required = parser.add_argument_group('Required Arguments')
78
+ required.add_argument('-d', action='store', dest='dir_path', required=True,
79
+ help='Define the directory path containing the files')
80
+ required.add_argument('-o', action='store', dest='output', help='Outdir',
81
+ required=True)
82
+
83
+ options = parser.parse_args()
84
+
85
+ # Use glob to find files ending with '_Final_Output.tsv'
86
+ files_list = glob.glob(f"{options.dir_path}/*_Final_*.tsv")
87
+ all_entries, read_counts, total_reads = read_files(files_list)
88
+
89
+ separate_taxa = ['d__Bacteria', 'd__Archaea', 'd__Eukaryota', 'd__Viruses','k__Fungi',
90
+ 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown']
91
+ remove_taxa = ['c__Mammalia','k__Viridiplantae'
92
+ ,'d__Eukaryota|k__unknown|p__Evosea|c__Eumycetozoa|o__Dictyosteliales|f__Dictyosteliaceae|g__Dictyostelium'
93
+ ,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Leishmania'
94
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Toxoplasma'
95
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Haemosporida|f__Plasmodiidae|g__Plasmodium'
96
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Piroplasmida|f__Theileriidae|g__Theileria'
97
+ ,'d__Eukaryota|k__unknown|p__Parabasalia|c__unknown|o__Trichomonadida|f__Trichomonadidae|g__Trichomonas'
98
+ ,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Besnoitia'
99
+ ,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Trypanosoma'
100
+ ,'d__Eukaryota|k__unknown|p__Ciliophora|c__Oligohymenophorea|o__Peniculida|f__Parameciidae|g__Paramecium'
101
+ ,'d__Eukaryota|k__Fungi|p__Microsporidia|c__unknown|o__unknown|f__Unikaryonidae|g__Encephalitozoon'
102
+ ,'d__Eukaryota|k__unknown|p__unknown|c__Cryptophyceae|o__Cryptomonadales|f__Cryptomonadaceae|g__Cryptomonas']
103
+ #make a overview list of samples and report 0s even if group has none for that sample
104
+ #output total number of reads
105
+
106
+
107
+ read_counts = group_by_substrings(read_counts, separate_taxa, remove_taxa)
108
+
109
+ read_counts = dict(sorted(read_counts.items(), key=lambda x: extract_numeric_suffix(x[0])))
110
+
111
+ write_out(options.output, read_counts, total_reads)
112
+
113
+
114
+
115
+
116
+ if __name__ == "__main__":
117
+ main()
118
+ print("Complete")
119
+
120
+
121
+