MetaPont 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- MetaPont/Extraction_By_Function.py +95 -0
- MetaPont/Function_By_Taxa.py +304 -0
- MetaPont/__init__.py +0 -0
- MetaPont/constants.py +2 -0
- MetaPont/emapper_readmap_cds_combiner_v1.py +235 -0
- MetaPont/kraken_emapper_readmap_combiner_v1.py +222 -0
- MetaPont/lineage_matrix_collapse_v1.py +80 -0
- MetaPont/lineage_matrix_collapse_v2.py +121 -0
- MetaPont-0.0.1.dist-info/LICENSE +674 -0
- MetaPont-0.0.1.dist-info/METADATA +111 -0
- MetaPont-0.0.1.dist-info/RECORD +14 -0
- MetaPont-0.0.1.dist-info/WHEEL +5 -0
- MetaPont-0.0.1.dist-info/entry_points.txt +2 -0
- MetaPont-0.0.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
import re
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
################
|
|
7
|
+
|
|
8
|
+
parent_directory_path = sys.argv[1]
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
for subdir, dirs, files in os.walk(parent_directory_path, topdown=True):
|
|
12
|
+
for dir in dirs:
|
|
13
|
+
try:
|
|
14
|
+
current_dir = os.path.basename(dir)
|
|
15
|
+
parent_of_current_dir = os.path.basename(subdir)
|
|
16
|
+
if current_dir.startswith('PN') and parent_of_current_dir == os.path.basename(parent_directory_path):
|
|
17
|
+
print(current_dir)
|
|
18
|
+
specific_dir_path = os.path.join(subdir, dir)
|
|
19
|
+
first_file_path = os.path.join(specific_dir_path, f"{dir}_eggnog_mapper/{dir}_pyrodigal_eggnog_mapped.emapper.annotations")
|
|
20
|
+
second_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2_report.txt")
|
|
21
|
+
third_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2_report_mpa.txt")
|
|
22
|
+
forth_file_path = os.path.join(specific_dir_path, f"{dir}_kraken2/{dir}_kraken2.txt")
|
|
23
|
+
fifth_file_path = os.path.join(specific_dir_path, f"{dir}_readmapped/{dir}_readmapped_contig_summary.txt")
|
|
24
|
+
|
|
25
|
+
if os.path.exists(first_file_path) and os.path.exists(second_file_path) and os.path.exists(third_file_path) and os.path.exists(forth_file_path) and os.path.exists(fifth_file_path):
|
|
26
|
+
print(f"Master directory: {specific_dir_path}")
|
|
27
|
+
print(f"First file path: {first_file_path}")
|
|
28
|
+
print(f"Second file path: {second_file_path}")
|
|
29
|
+
print(f"Third file path: {third_file_path}")
|
|
30
|
+
print(f"Forth file path: {forth_file_path}")
|
|
31
|
+
print(f"Fifth file path: {fifth_file_path}")
|
|
32
|
+
|
|
33
|
+
contigs = defaultdict(list)
|
|
34
|
+
genes = defaultdict(list)
|
|
35
|
+
lineages = {}
|
|
36
|
+
taxa_ids = []
|
|
37
|
+
|
|
38
|
+
with open(first_file_path, 'r') as emapper_in:
|
|
39
|
+
for line in emapper_in:
|
|
40
|
+
if not line.startswith('#'):
|
|
41
|
+
line_data = line.strip().split('\t')
|
|
42
|
+
contig = line_data[0].rsplit('_', 1)[0]
|
|
43
|
+
COGs = line_data[6]
|
|
44
|
+
GOs = line_data[9]
|
|
45
|
+
EC = line_data[10]
|
|
46
|
+
KEGG_ko = line_data[11]
|
|
47
|
+
KEGG_Pathway = line_data[12]
|
|
48
|
+
KEGG_Module = line_data[13]
|
|
49
|
+
KEGG_Reaction = line_data[14]
|
|
50
|
+
KEGG_rclass = line_data[15]
|
|
51
|
+
BRITE = line_data[16]
|
|
52
|
+
KEGG_TC = line_data[17]
|
|
53
|
+
CAZy = line_data[18]
|
|
54
|
+
BiGG_Reaction = line_data[19]
|
|
55
|
+
PFAMs = line_data[20]
|
|
56
|
+
genes[contig].append(
|
|
57
|
+
[COGs, GOs, EC, KEGG_ko, KEGG_Pathway, KEGG_Module, KEGG_Reaction, KEGG_rclass, BRITE,
|
|
58
|
+
KEGG_TC, CAZy, BiGG_Reaction, PFAMs])
|
|
59
|
+
|
|
60
|
+
with open(second_file_path, 'r') as f:
|
|
61
|
+
kraken2_lines = f.readlines()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
with open(third_file_path, 'r') as kraken_lineage_in:
|
|
65
|
+
for line in kraken_lineage_in:
|
|
66
|
+
lineages_to_check = {'d__':'d__unknown', 'k__':'k__unknown', 'p__':'p__unknown', 'c__':'c__unknown',
|
|
67
|
+
'o__':'o__unknown', 'f__':'f__unknown', 'g__':'g__unknown', 's__':'s__unknown'}
|
|
68
|
+
if line.startswith('d__'):
|
|
69
|
+
line_data = line.split('|')
|
|
70
|
+
contig_length = line_data[-1].split('\t')[1].strip() # check
|
|
71
|
+
line_data[-1] = line_data[-1].split('\t')[0] #check
|
|
72
|
+
# Check for missing lineages
|
|
73
|
+
for data in line_data:
|
|
74
|
+
first_three_chars = data[:3]
|
|
75
|
+
if first_three_chars in lineages_to_check:
|
|
76
|
+
lineages_to_check[first_three_chars] = data
|
|
77
|
+
|
|
78
|
+
combined = ''
|
|
79
|
+
for key, value in lineages_to_check.items():
|
|
80
|
+
#if value is not None and value != '' and '__unknown' not in value:
|
|
81
|
+
combined+=value+'|'
|
|
82
|
+
combined = combined[:-1]
|
|
83
|
+
#print(combined)
|
|
84
|
+
|
|
85
|
+
lineages[line.split('\t')[0].split('_')[-1]] = combined
|
|
86
|
+
|
|
87
|
+
counter = 0
|
|
88
|
+
with open(forth_file_path, 'r') as kraken_in:
|
|
89
|
+
for line in kraken_in:
|
|
90
|
+
counter +=1
|
|
91
|
+
line_data = line.strip().split('\t')
|
|
92
|
+
contig = line_data[1]
|
|
93
|
+
taxa_id = line_data[2].split(' (taxid')[0]
|
|
94
|
+
contig_length = int(line_data[3])
|
|
95
|
+
if contig_length >= 2500:
|
|
96
|
+
full_lineage = None
|
|
97
|
+
if taxa_id == 'unclassified' or taxa_id == 'root':
|
|
98
|
+
full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
|
|
99
|
+
else:
|
|
100
|
+
try:
|
|
101
|
+
full_lineage = lineages[taxa_id]
|
|
102
|
+
except KeyError:
|
|
103
|
+
while True:
|
|
104
|
+
taxa_id = taxa_id.rsplit(' ', 1)[0]
|
|
105
|
+
try:
|
|
106
|
+
full_lineage = lineages[taxa_id]
|
|
107
|
+
break
|
|
108
|
+
except KeyError:
|
|
109
|
+
tmp_size = taxa_id.split(' ')
|
|
110
|
+
if len(tmp_size) == 1:
|
|
111
|
+
taxa_id = line_data[2].split(' (taxid')[0]
|
|
112
|
+
escaped_pattern = re.escape(taxa_id)
|
|
113
|
+
matching_line = next((line for line in kraken2_lines if re.search(escaped_pattern, line)), None)
|
|
114
|
+
# Loop through the lines and search for the pattern
|
|
115
|
+
|
|
116
|
+
index = kraken2_lines.index(matching_line)
|
|
117
|
+
line_above = kraken2_lines[index - 1]
|
|
118
|
+
while True:
|
|
119
|
+
if not any(char.isdigit() for char in line_above.split('\t')[3]):
|
|
120
|
+
break
|
|
121
|
+
else:
|
|
122
|
+
index = kraken2_lines.index(line_above)
|
|
123
|
+
line_above = kraken2_lines[index - 1]
|
|
124
|
+
taxa_id = line_above.strip().split('\t')[-1].lstrip()
|
|
125
|
+
try:
|
|
126
|
+
full_lineage = lineages[taxa_id]
|
|
127
|
+
break
|
|
128
|
+
except KeyError:
|
|
129
|
+
if taxa_id == 'unclassified' or taxa_id == 'root':
|
|
130
|
+
full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
|
|
131
|
+
break
|
|
132
|
+
continue
|
|
133
|
+
if full_lineage == None:
|
|
134
|
+
full_lineage = 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown|s__unknown'
|
|
135
|
+
print("Contig without lineage. ")
|
|
136
|
+
contigs[contig] = [line_data[2],line_data[3],full_lineage]
|
|
137
|
+
#print(counter)
|
|
138
|
+
else:
|
|
139
|
+
print("Contig length less than 2,500")
|
|
140
|
+
break
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
print("RM")
|
|
146
|
+
with open(fifth_file_path, 'r') as readmapped_in:
|
|
147
|
+
for line in readmapped_in:
|
|
148
|
+
if not line.startswith('Contig'): #sloppy
|
|
149
|
+
line_data = line.strip().split('\t')
|
|
150
|
+
contig = line_data[0]
|
|
151
|
+
if contig in contigs:
|
|
152
|
+
info = [line_data[1],line_data[2]]
|
|
153
|
+
info.extend(contigs[contig])
|
|
154
|
+
contigs[contig] = info
|
|
155
|
+
|
|
156
|
+
final_dict = defaultdict(list)
|
|
157
|
+
|
|
158
|
+
for contig, data in contigs.items(): # Easy to read but not efficient at all
|
|
159
|
+
data.extend([[], [], [], [], [], [], [], [], [], [], [], [], []])
|
|
160
|
+
final_dict[contig] = data
|
|
161
|
+
#for gene in data[0]:
|
|
162
|
+
gene_info = genes[contig]
|
|
163
|
+
for current in gene_info:
|
|
164
|
+
|
|
165
|
+
try:
|
|
166
|
+
if current[0] != '-':
|
|
167
|
+
final_dict[contig][5].append(current[0])
|
|
168
|
+
if current[1] != '-':
|
|
169
|
+
final_dict[contig][6].append(current[1])
|
|
170
|
+
if current[2] != '-':
|
|
171
|
+
final_dict[contig][7].append(current[2])
|
|
172
|
+
if current[3] != '-':
|
|
173
|
+
final_dict[contig][8].append(current[3])
|
|
174
|
+
if current[4] != '-':
|
|
175
|
+
final_dict[contig][9].append(current[4])
|
|
176
|
+
if current[5] != '-':
|
|
177
|
+
final_dict[contig][10].append(current[5])
|
|
178
|
+
if current[6] != '-':
|
|
179
|
+
final_dict[contig][11].append(current[6])
|
|
180
|
+
if current[7] != '-':
|
|
181
|
+
final_dict[contig][12].append(current[7])
|
|
182
|
+
if current[8] != '-':
|
|
183
|
+
final_dict[contig][13].append(current[8])
|
|
184
|
+
if current[9] != '-':
|
|
185
|
+
final_dict[contig][14].append(current[9])
|
|
186
|
+
if current[10] != '-':
|
|
187
|
+
final_dict[contig][15].append(current[10])
|
|
188
|
+
if current[11] != '-':
|
|
189
|
+
final_dict[contig][16].append(current[11])
|
|
190
|
+
if current[12] != '-':
|
|
191
|
+
final_dict[contig][17].append(current[12])
|
|
192
|
+
if current[13] != '-':
|
|
193
|
+
final_dict[contig][18].append(current[13])
|
|
194
|
+
except IndexError:
|
|
195
|
+
continue
|
|
196
|
+
|
|
197
|
+
output_path = os.path.join(parent_directory_path, 'Contig_Final_Outputs')
|
|
198
|
+
if not os.path.exists(output_path):
|
|
199
|
+
os.makedirs(output_path)
|
|
200
|
+
output = open(os.path.join(output_path, dir + '_Final_Contig.tsv'),'w')
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
output.write(dir+'\n')
|
|
204
|
+
output.write('Contig\tContig_Length\tTotal_Reads\tMapped_Reads\tScientific_Name\tLineage\tCOG\tGO\tEC\tKEGG_KO\tKEGG_Pathway\tKEGG_Module\t'
|
|
205
|
+
'KEGG_Reaction\tKEGG_rclass\tBRITE\tKEGG_TC\tCAZy\tBiGG_Reaction\tPFAMs\n')
|
|
206
|
+
|
|
207
|
+
for contig, data in final_dict.items():
|
|
208
|
+
try:
|
|
209
|
+
#print(contig)
|
|
210
|
+
output.write(contig+'\t'+data[3]+'\t'+data[1]+'\t'+data[0]+'\t'+data[2]+'\t'+data[4]+'\t'+'|'.join(data[5])+'\t'+'|'.join(data[6])+'\t'+'|'.join(data[7])+'\t'+'|'.join(data[8])
|
|
211
|
+
+'\t'+'|'.join(data[9])+'\t'+'|'.join(data[10])+'\t'+'|'.join(data[11])+'\t'+'|'.join(data[12])+'\t'+'|'.join(data[13])+'\t'+'|'.join(data[14])
|
|
212
|
+
+'\t'+'|'.join(data[15])+'\t'+'|'.join(data[16])+'\t'+'|'.join(data[17])+'\n')
|
|
213
|
+
except TypeError as e:
|
|
214
|
+
print("Some Error: " + e)
|
|
215
|
+
|
|
216
|
+
print("Done")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
except FileNotFoundError as e:
|
|
220
|
+
print(f"Error processing directory {dir}: {e}")
|
|
221
|
+
|
|
222
|
+
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import collections
|
|
2
|
+
import re
|
|
3
|
+
import glob
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
outfile = open('/mnt/Internal/Nextcloud/Collab/Zhenbin/2024_project/Contig_Final_Outputs/Lineage_Genus.tsv', 'w')
|
|
7
|
+
|
|
8
|
+
KO_reads = collections.defaultdict(lambda: collections.defaultdict(int))
|
|
9
|
+
|
|
10
|
+
# Define the directory path containing the files
|
|
11
|
+
directory_path = '/mnt/Internal/Nextcloud/Collab/Zhenbin/2024_project/Contig_Final_Outputs/'
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
all_KOs = []
|
|
16
|
+
|
|
17
|
+
# Iterate through each file found
|
|
18
|
+
for file_path in files_list:
|
|
19
|
+
sample = file_path.split('/')[-1].split('_Final')[0]
|
|
20
|
+
# Perform actions on each file (for example, read the file content)
|
|
21
|
+
with open(file_path, 'r') as file:
|
|
22
|
+
for line in file:
|
|
23
|
+
line = line.replace('\n','')
|
|
24
|
+
seen_KOs = []
|
|
25
|
+
if line.startswith('NODE'): # Only works for this dataset
|
|
26
|
+
line_data = line.split('\t')
|
|
27
|
+
mapped_reads = int(line_data[3])
|
|
28
|
+
lineage = line_data[5]
|
|
29
|
+
#for KO in KOs:
|
|
30
|
+
# if KO not in seen_KOs:
|
|
31
|
+
# KO_reads[sample][KO] += mapped_reads
|
|
32
|
+
# seen_KOs.append(KO)
|
|
33
|
+
genus = lineage.rsplit('|', 1)[0]
|
|
34
|
+
KO_reads[sample][genus] += mapped_reads
|
|
35
|
+
|
|
36
|
+
if genus not in all_KOs:
|
|
37
|
+
all_KOs.append(genus)
|
|
38
|
+
print("")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def extract_numeric_suffix(key):
|
|
42
|
+
match = re.search(r'\d+$', key)
|
|
43
|
+
return int(match.group()) if match else 0
|
|
44
|
+
|
|
45
|
+
# Reorder the dictionary based on the last numeric characters of the keys
|
|
46
|
+
KO_reads = dict(sorted(KO_reads.items(), key=lambda x: extract_numeric_suffix(x[0])))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
keys_as_string = '\t'.join(KO_reads.keys())
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
outfile.write('Lineage_Genus\t'+keys_as_string+'\n')
|
|
55
|
+
|
|
56
|
+
for KO in all_KOs:
|
|
57
|
+
outfile.write(KO)
|
|
58
|
+
for sample in KO_reads.keys():
|
|
59
|
+
outfile.write('\t'+str(KO_reads[sample][KO]))
|
|
60
|
+
outfile.write('\n')
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import collections
|
|
2
|
+
import re
|
|
3
|
+
import glob
|
|
4
|
+
import argparse
|
|
5
|
+
import os
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def read_files(files_list):
|
|
9
|
+
all_entries = []
|
|
10
|
+
read_counts = collections.defaultdict(lambda: collections.defaultdict(int))
|
|
11
|
+
total_reads = collections.defaultdict(int)
|
|
12
|
+
for file_path in files_list:
|
|
13
|
+
sample = file_path.split('/')[-1].split('_Final')[0] # _Final is specific to this dataset
|
|
14
|
+
with open(file_path, 'r') as file:
|
|
15
|
+
for line in file:
|
|
16
|
+
line = line.replace('\n','')
|
|
17
|
+
if line.startswith('NODE'): # Only works for this dataset/metaspades
|
|
18
|
+
line_data = line.split('\t')
|
|
19
|
+
mapped_reads = int(line_data[3])
|
|
20
|
+
total_reads[sample] = int(line_data[2])
|
|
21
|
+
lineage = line_data[5]
|
|
22
|
+
genus = lineage.rsplit('|', 1)[0] # We are reporting only down to the genus level
|
|
23
|
+
read_counts[sample][genus] += mapped_reads
|
|
24
|
+
if genus not in all_entries:
|
|
25
|
+
all_entries.append(genus)
|
|
26
|
+
return all_entries, read_counts, total_reads
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def write_out(output_dir, read_counts, total_reads):
|
|
30
|
+
for substr, counts in read_counts.items():
|
|
31
|
+
if substr == 'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown':
|
|
32
|
+
substr = 'd__unknown'
|
|
33
|
+
output_file = os.path.join(output_dir, f"{substr}_output.tsv")
|
|
34
|
+
sample_names = sorted(total_reads.keys())
|
|
35
|
+
keys_as_string = '\t'.join(sample_names)
|
|
36
|
+
values_as_string = '\t'.join([str(total_reads[key]) for key in sample_names])
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
all_current_taxa = list({key for sample_counts in counts.values() for key in sample_counts})
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
with open(output_file, 'w') as outfile:
|
|
44
|
+
outfile.write('Lineage_Genus\t'+keys_as_string+'\n')
|
|
45
|
+
outfile.write('Total_Num_Reads')
|
|
46
|
+
outfile.write('\t' + str(values_as_string))
|
|
47
|
+
outfile.write('\n')
|
|
48
|
+
for current_taxa in all_current_taxa:
|
|
49
|
+
outfile.write(current_taxa)
|
|
50
|
+
for sample_key in sample_names:
|
|
51
|
+
outfile.write('\t' + str(counts[sample_key][current_taxa]))
|
|
52
|
+
outfile.write('\n')
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def extract_numeric_suffix(key):
|
|
57
|
+
match = re.search(r'\d+$', key)
|
|
58
|
+
return int(match.group()) if match else 0
|
|
59
|
+
|
|
60
|
+
def group_by_substrings(read_counts, substrings, remove_substrings):
|
|
61
|
+
grouped_counts = {substr: collections.defaultdict(lambda: collections.defaultdict(int)) for substr in substrings}
|
|
62
|
+
for sample, counts in read_counts.items():
|
|
63
|
+
for key, value in counts.items():
|
|
64
|
+
if any(remove_substr in key for remove_substr in remove_substrings):
|
|
65
|
+
continue
|
|
66
|
+
for substr in substrings:
|
|
67
|
+
if substr in key:
|
|
68
|
+
grouped_counts[substr][sample][key] += value
|
|
69
|
+
return grouped_counts
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def main():
|
|
73
|
+
|
|
74
|
+
parser = argparse.ArgumentParser(description='....')
|
|
75
|
+
parser._action_groups.pop()
|
|
76
|
+
|
|
77
|
+
required = parser.add_argument_group('Required Arguments')
|
|
78
|
+
required.add_argument('-d', action='store', dest='dir_path', required=True,
|
|
79
|
+
help='Define the directory path containing the files')
|
|
80
|
+
required.add_argument('-o', action='store', dest='output', help='Outdir',
|
|
81
|
+
required=True)
|
|
82
|
+
|
|
83
|
+
options = parser.parse_args()
|
|
84
|
+
|
|
85
|
+
# Use glob to find files ending with '_Final_Output.tsv'
|
|
86
|
+
files_list = glob.glob(f"{options.dir_path}/*_Final_*.tsv")
|
|
87
|
+
all_entries, read_counts, total_reads = read_files(files_list)
|
|
88
|
+
|
|
89
|
+
separate_taxa = ['d__Bacteria', 'd__Archaea', 'd__Eukaryota', 'd__Viruses','k__Fungi',
|
|
90
|
+
'd__unknown|k__unknown|p__unknown|c__unknown|o__unknown|f__unknown|g__unknown']
|
|
91
|
+
remove_taxa = ['c__Mammalia','k__Viridiplantae'
|
|
92
|
+
,'d__Eukaryota|k__unknown|p__Evosea|c__Eumycetozoa|o__Dictyosteliales|f__Dictyosteliaceae|g__Dictyostelium'
|
|
93
|
+
,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Leishmania'
|
|
94
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Toxoplasma'
|
|
95
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Haemosporida|f__Plasmodiidae|g__Plasmodium'
|
|
96
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Aconoidasida|o__Piroplasmida|f__Theileriidae|g__Theileria'
|
|
97
|
+
,'d__Eukaryota|k__unknown|p__Parabasalia|c__unknown|o__Trichomonadida|f__Trichomonadidae|g__Trichomonas'
|
|
98
|
+
,'d__Eukaryota|k__unknown|p__Apicomplexa|c__Conoidasida|o__Eucoccidiorida|f__Sarcocystidae|g__Besnoitia'
|
|
99
|
+
,'d__Eukaryota|k__unknown|p__Euglenozoa|c__Kinetoplastea|o__Trypanosomatida|f__Trypanosomatidae|g__Trypanosoma'
|
|
100
|
+
,'d__Eukaryota|k__unknown|p__Ciliophora|c__Oligohymenophorea|o__Peniculida|f__Parameciidae|g__Paramecium'
|
|
101
|
+
,'d__Eukaryota|k__Fungi|p__Microsporidia|c__unknown|o__unknown|f__Unikaryonidae|g__Encephalitozoon'
|
|
102
|
+
,'d__Eukaryota|k__unknown|p__unknown|c__Cryptophyceae|o__Cryptomonadales|f__Cryptomonadaceae|g__Cryptomonas']
|
|
103
|
+
#make a overview list of samples and report 0s even if group has none for that sample
|
|
104
|
+
#output total number of reads
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
read_counts = group_by_substrings(read_counts, separate_taxa, remove_taxa)
|
|
108
|
+
|
|
109
|
+
read_counts = dict(sorted(read_counts.items(), key=lambda x: extract_numeric_suffix(x[0])))
|
|
110
|
+
|
|
111
|
+
write_out(options.output, read_counts, total_reads)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
if __name__ == "__main__":
|
|
117
|
+
main()
|
|
118
|
+
print("Complete")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
|