MetaPont 0.0.1__tar.gz → 0.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {metapont-0.0.1/src/MetaPont.egg-info → metapont-0.0.2}/PKG-INFO +25 -6
- {metapont-0.0.1 → metapont-0.0.2}/README.md +24 -5
- {metapont-0.0.1 → metapont-0.0.2}/setup.cfg +1 -1
- metapont-0.0.2/src/MetaPont/Extract_By_Function.py +108 -0
- metapont-0.0.2/src/MetaPont/constants.py +2 -0
- {metapont-0.0.1 → metapont-0.0.2/src/MetaPont.egg-info}/PKG-INFO +25 -6
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont.egg-info/SOURCES.txt +1 -1
- metapont-0.0.1/src/MetaPont/Extraction_By_Function.py +0 -95
- metapont-0.0.1/src/MetaPont/constants.py +0 -2
- {metapont-0.0.1 → metapont-0.0.2}/LICENSE +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/pyproject.toml +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/Function_By_Taxa.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/__init__.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/emapper_readmap_cds_combiner_v1.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/kraken_emapper_readmap_combiner_v1.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/lineage_matrix_collapse_v1.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont/lineage_matrix_collapse_v2.py +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont.egg-info/dependency_links.txt +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont.egg-info/entry_points.txt +0 -0
- {metapont-0.0.1 → metapont-0.0.2}/src/MetaPont.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: MetaPont
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.2
|
|
4
4
|
Summary: MetaPont - A tool to bridge the gap between the output of metagenomic tools and the analysis of the data
|
|
5
5
|
Home-page: https://github.com/TheHuwsLab/MetaPont
|
|
6
6
|
Author: Nicholas Dimonaco
|
|
@@ -47,6 +47,23 @@ pip install MetaPont
|
|
|
47
47
|
## Usage
|
|
48
48
|
|
|
49
49
|
### Command-line Arguments
|
|
50
|
+
```Extract-By-Function -h ```
|
|
51
|
+
```bash
|
|
52
|
+
usage: Extract-By-Function [-h] -d DIRECTORY -f FUNCTION_ID [-o OUTPUT] [-m MIN_PROPORTION]
|
|
53
|
+
|
|
54
|
+
MetaPont v0.0.2: Extract-By-Function - Identify taxa contributing to a specific function.
|
|
55
|
+
|
|
56
|
+
options:
|
|
57
|
+
-h, --help show this help message and exit
|
|
58
|
+
-d DIRECTORY, --directory DIRECTORY
|
|
59
|
+
Directory containing TSV files to analyse.
|
|
60
|
+
-f FUNCTION_ID, --function_id FUNCTION_ID
|
|
61
|
+
Specific function ID to search for (e.g., 'GO:0002').
|
|
62
|
+
-o OUTPUT, --output OUTPUT
|
|
63
|
+
Output file to save results (default: output_taxa_details.tsv).
|
|
64
|
+
-m MIN_PROPORTION, --min_proportion MIN_PROPORTION
|
|
65
|
+
Minimum proportion threshold for taxa to be included in the output (default: 0.05).
|
|
66
|
+
```
|
|
50
67
|
|
|
51
68
|
The `Extract-By-Function` tool provides several command-line options:
|
|
52
69
|
|
|
@@ -79,11 +96,13 @@ Example output:
|
|
|
79
96
|
|
|
80
97
|
```
|
|
81
98
|
Function ID: GO:0002
|
|
82
|
-
Sample Taxa Proportion
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
99
|
+
Sample Taxa Reads Assigned (Function) Proportion (Function) Proportion (Total Reads)
|
|
100
|
+
PN0536_0003_S83_Final_Contig.tsv Gordonibacter 60788 0.075 0.002
|
|
101
|
+
PN0536_0003_S83_Final_Contig.tsv Streptomyces 115671 0.142 0.004
|
|
102
|
+
PN0536_0003_S83_Final_Contig.tsv unknown 80890 0.099 0.003
|
|
103
|
+
PN0536_0003_S83_Final_Contig.tsv Clostridium 51018 0.063 0.002
|
|
104
|
+
PN0536_0003_S83_Final_Contig.tsv Lactobacillus 149909 0.184 0.005
|
|
105
|
+
PN0536_0003_S83_Final_Contig.tsv Limosilactobacillus 79694 0.098 0.003
|
|
87
106
|
```
|
|
88
107
|
|
|
89
108
|
---
|
|
@@ -32,6 +32,23 @@ pip install MetaPont
|
|
|
32
32
|
## Usage
|
|
33
33
|
|
|
34
34
|
### Command-line Arguments
|
|
35
|
+
```Extract-By-Function -h ```
|
|
36
|
+
```bash
|
|
37
|
+
usage: Extract-By-Function [-h] -d DIRECTORY -f FUNCTION_ID [-o OUTPUT] [-m MIN_PROPORTION]
|
|
38
|
+
|
|
39
|
+
MetaPont v0.0.2: Extract-By-Function - Identify taxa contributing to a specific function.
|
|
40
|
+
|
|
41
|
+
options:
|
|
42
|
+
-h, --help show this help message and exit
|
|
43
|
+
-d DIRECTORY, --directory DIRECTORY
|
|
44
|
+
Directory containing TSV files to analyse.
|
|
45
|
+
-f FUNCTION_ID, --function_id FUNCTION_ID
|
|
46
|
+
Specific function ID to search for (e.g., 'GO:0002').
|
|
47
|
+
-o OUTPUT, --output OUTPUT
|
|
48
|
+
Output file to save results (default: output_taxa_details.tsv).
|
|
49
|
+
-m MIN_PROPORTION, --min_proportion MIN_PROPORTION
|
|
50
|
+
Minimum proportion threshold for taxa to be included in the output (default: 0.05).
|
|
51
|
+
```
|
|
35
52
|
|
|
36
53
|
The `Extract-By-Function` tool provides several command-line options:
|
|
37
54
|
|
|
@@ -64,11 +81,13 @@ Example output:
|
|
|
64
81
|
|
|
65
82
|
```
|
|
66
83
|
Function ID: GO:0002
|
|
67
|
-
Sample Taxa Proportion
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
84
|
+
Sample Taxa Reads Assigned (Function) Proportion (Function) Proportion (Total Reads)
|
|
85
|
+
PN0536_0003_S83_Final_Contig.tsv Gordonibacter 60788 0.075 0.002
|
|
86
|
+
PN0536_0003_S83_Final_Contig.tsv Streptomyces 115671 0.142 0.004
|
|
87
|
+
PN0536_0003_S83_Final_Contig.tsv unknown 80890 0.099 0.003
|
|
88
|
+
PN0536_0003_S83_Final_Contig.tsv Clostridium 51018 0.063 0.002
|
|
89
|
+
PN0536_0003_S83_Final_Contig.tsv Lactobacillus 149909 0.184 0.005
|
|
90
|
+
PN0536_0003_S83_Final_Contig.tsv Limosilactobacillus 79694 0.098 0.003
|
|
72
91
|
```
|
|
73
92
|
|
|
74
93
|
---
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import os
|
|
3
|
+
import csv
|
|
4
|
+
import sys
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
from .constants import *
|
|
9
|
+
except (ModuleNotFoundError, ImportError, NameError, TypeError) as error:
|
|
10
|
+
from constants import *
|
|
11
|
+
|
|
12
|
+
# Needed to account for the large CSV/TSV files we are working with
|
|
13
|
+
csv.field_size_limit(sys.maxsize)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def process_tsv(file_path, function_id):
|
|
17
|
+
"""
|
|
18
|
+
Processes a TSV file to calculate:
|
|
19
|
+
1. Total reads for each taxon (within the function).
|
|
20
|
+
2. Total reads for the specified function.
|
|
21
|
+
3. Total reads across all contigs in the file.
|
|
22
|
+
"""
|
|
23
|
+
taxa_reads_function = defaultdict(int) # Reads for each taxon within the function
|
|
24
|
+
total_reads_function = 0 # Total reads for the specified function
|
|
25
|
+
total_reads_all = 0 # Total reads across all contigs
|
|
26
|
+
|
|
27
|
+
with open(file_path, "r") as tsv_file:
|
|
28
|
+
reader = csv.reader(tsv_file, delimiter="\t")
|
|
29
|
+
next(reader) # Skip the first row (sample name)
|
|
30
|
+
headers = next(reader) # Read headers from the second row
|
|
31
|
+
|
|
32
|
+
# Locate the necessary columns
|
|
33
|
+
taxa_idx = headers.index("Lineage")
|
|
34
|
+
reads_idx = headers.index("Mapped_Reads")
|
|
35
|
+
|
|
36
|
+
# Process each row
|
|
37
|
+
for idx, row in enumerate(reader):
|
|
38
|
+
if len(row) < len(headers):
|
|
39
|
+
continue # Skip malformed rows
|
|
40
|
+
|
|
41
|
+
lineage = row[taxa_idx]
|
|
42
|
+
reads = int(row[reads_idx]) # Get the number of reads for this row
|
|
43
|
+
total_reads_all += reads # Increment total reads for all contigs
|
|
44
|
+
|
|
45
|
+
# Check if this row matches the specified function ID
|
|
46
|
+
for cell in row[6:]:
|
|
47
|
+
if cell and any(function_id in part for part in cell.replace(',', '|').split('|')):
|
|
48
|
+
if "g__" in lineage:
|
|
49
|
+
genus = lineage.split("g__")[1].split("|")[0]
|
|
50
|
+
taxa_reads_function[genus] += reads
|
|
51
|
+
total_reads_function += reads
|
|
52
|
+
break # Stop checking further functional columns for this row
|
|
53
|
+
|
|
54
|
+
return taxa_reads_function, total_reads_function, total_reads_all
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def main():
|
|
58
|
+
parser = argparse.ArgumentParser(description='MetaPont ' + MetaPont_Version + ': Extract-By-Function - Identify taxa contributing to a specific function.')
|
|
59
|
+
parser.add_argument(
|
|
60
|
+
"-d", "--directory", required=True,
|
|
61
|
+
help="Directory containing TSV files to analyse."
|
|
62
|
+
)
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
"-f", "--function_id", required=True,
|
|
65
|
+
help="Specific function ID to search for (e.g., 'GO:0002')."
|
|
66
|
+
)
|
|
67
|
+
parser.add_argument(
|
|
68
|
+
"-o", "--output", default="output_taxa_details.tsv",
|
|
69
|
+
help="Output file to save results (default: output_taxa_details.tsv)."
|
|
70
|
+
)
|
|
71
|
+
parser.add_argument(
|
|
72
|
+
"-m", "--min_proportion", type=float, default=0.05,
|
|
73
|
+
help="Minimum proportion threshold for taxa to be included in the output (default: 0.05)."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
options = parser.parse_args()
|
|
77
|
+
print("Running MetaPont: Extract-By-Function " + MetaPont_Version)
|
|
78
|
+
|
|
79
|
+
input_path = os.path.abspath(options.directory)
|
|
80
|
+
output_path = os.path.abspath(options.output)
|
|
81
|
+
|
|
82
|
+
all_results = {}
|
|
83
|
+
|
|
84
|
+
# Process each TSV file in the directory
|
|
85
|
+
for file_name in os.listdir(input_path):
|
|
86
|
+
if file_name.endswith("_Final_Contig.tsv"):
|
|
87
|
+
file_path = os.path.join(options.directory, file_name)
|
|
88
|
+
print(f"Processing file: {file_name}")
|
|
89
|
+
taxa_reads_function, total_reads_function, total_reads_all = process_tsv(file_path, options.function_id)
|
|
90
|
+
all_results[file_name] = (taxa_reads_function, total_reads_function, total_reads_all)
|
|
91
|
+
|
|
92
|
+
# Write results to output
|
|
93
|
+
with open(output_path, "w") as out:
|
|
94
|
+
out.write("Function ID: " + options.function_id + "\n")
|
|
95
|
+
out.write("Sample\tTaxa\tReads Assigned (Function)\tProportion (Function)\tProportion (Total Reads)\n")
|
|
96
|
+
for sample, (taxa_reads_function, total_reads_function, total_reads_all) in all_results.items():
|
|
97
|
+
for taxa, reads_function in taxa_reads_function.items():
|
|
98
|
+
proportion_function = reads_function / total_reads_function if total_reads_function > 0 else 0
|
|
99
|
+
proportion_total = reads_function / total_reads_all if total_reads_all > 0 else 0
|
|
100
|
+
|
|
101
|
+
if proportion_function >= options.min_proportion: # Apply minimum proportion filter
|
|
102
|
+
out.write(f"{sample}\t{taxa}\t{reads_function}\t{proportion_function:.3f}\t{proportion_total:.3f}\n")
|
|
103
|
+
|
|
104
|
+
print(f"Results saved to {options.output}")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
if __name__ == "__main__":
|
|
108
|
+
main()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: MetaPont
|
|
3
|
-
Version: 0.0.
|
|
3
|
+
Version: 0.0.2
|
|
4
4
|
Summary: MetaPont - A tool to bridge the gap between the output of metagenomic tools and the analysis of the data
|
|
5
5
|
Home-page: https://github.com/TheHuwsLab/MetaPont
|
|
6
6
|
Author: Nicholas Dimonaco
|
|
@@ -47,6 +47,23 @@ pip install MetaPont
|
|
|
47
47
|
## Usage
|
|
48
48
|
|
|
49
49
|
### Command-line Arguments
|
|
50
|
+
```Extract-By-Function -h ```
|
|
51
|
+
```bash
|
|
52
|
+
usage: Extract-By-Function [-h] -d DIRECTORY -f FUNCTION_ID [-o OUTPUT] [-m MIN_PROPORTION]
|
|
53
|
+
|
|
54
|
+
MetaPont v0.0.2: Extract-By-Function - Identify taxa contributing to a specific function.
|
|
55
|
+
|
|
56
|
+
options:
|
|
57
|
+
-h, --help show this help message and exit
|
|
58
|
+
-d DIRECTORY, --directory DIRECTORY
|
|
59
|
+
Directory containing TSV files to analyse.
|
|
60
|
+
-f FUNCTION_ID, --function_id FUNCTION_ID
|
|
61
|
+
Specific function ID to search for (e.g., 'GO:0002').
|
|
62
|
+
-o OUTPUT, --output OUTPUT
|
|
63
|
+
Output file to save results (default: output_taxa_details.tsv).
|
|
64
|
+
-m MIN_PROPORTION, --min_proportion MIN_PROPORTION
|
|
65
|
+
Minimum proportion threshold for taxa to be included in the output (default: 0.05).
|
|
66
|
+
```
|
|
50
67
|
|
|
51
68
|
The `Extract-By-Function` tool provides several command-line options:
|
|
52
69
|
|
|
@@ -79,11 +96,13 @@ Example output:
|
|
|
79
96
|
|
|
80
97
|
```
|
|
81
98
|
Function ID: GO:0002
|
|
82
|
-
Sample Taxa Proportion
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
99
|
+
Sample Taxa Reads Assigned (Function) Proportion (Function) Proportion (Total Reads)
|
|
100
|
+
PN0536_0003_S83_Final_Contig.tsv Gordonibacter 60788 0.075 0.002
|
|
101
|
+
PN0536_0003_S83_Final_Contig.tsv Streptomyces 115671 0.142 0.004
|
|
102
|
+
PN0536_0003_S83_Final_Contig.tsv unknown 80890 0.099 0.003
|
|
103
|
+
PN0536_0003_S83_Final_Contig.tsv Clostridium 51018 0.063 0.002
|
|
104
|
+
PN0536_0003_S83_Final_Contig.tsv Lactobacillus 149909 0.184 0.005
|
|
105
|
+
PN0536_0003_S83_Final_Contig.tsv Limosilactobacillus 79694 0.098 0.003
|
|
87
106
|
```
|
|
88
107
|
|
|
89
108
|
---
|
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
import argparse
|
|
2
|
-
import os
|
|
3
|
-
import csv
|
|
4
|
-
import sys
|
|
5
|
-
from collections import Counter
|
|
6
|
-
|
|
7
|
-
from HuwsLab.MetaPont.src.MetaPont.constants import MetaPont_Version
|
|
8
|
-
|
|
9
|
-
# Needed to account for the large CSV/TSV files we are working with
|
|
10
|
-
csv.field_size_limit(sys.maxsize)
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def process_tsv(file_path, function_id):
|
|
14
|
-
"""
|
|
15
|
-
Processes a TSV file to calculate taxa counts and total matches for a given function ID.
|
|
16
|
-
"""
|
|
17
|
-
taxa_counts = Counter()
|
|
18
|
-
total_matches = 0
|
|
19
|
-
|
|
20
|
-
with open(file_path, "r") as tsv_file:
|
|
21
|
-
reader = csv.reader(tsv_file, delimiter="\t")
|
|
22
|
-
next(reader) # Skip the first row (sample name)
|
|
23
|
-
headers = next(reader) # Read headers from the second row
|
|
24
|
-
|
|
25
|
-
# Locate the Lineage column
|
|
26
|
-
taxa_idx = headers.index("Lineage")
|
|
27
|
-
|
|
28
|
-
# Process each row to find matches
|
|
29
|
-
for idx, row in enumerate(reader):
|
|
30
|
-
if len(row) < len(headers):
|
|
31
|
-
continue # Skip malformed rows
|
|
32
|
-
|
|
33
|
-
lineage = row[taxa_idx]
|
|
34
|
-
|
|
35
|
-
# Search functional columns starting from column 6
|
|
36
|
-
for cell in row[6:]:
|
|
37
|
-
if cell and any(function_id in part for part in cell.replace(',', '|').split('|')):
|
|
38
|
-
# Extract genus from the Lineage field
|
|
39
|
-
if "g__" in lineage:
|
|
40
|
-
genus = lineage.split("g__")[1].split("|")[0]
|
|
41
|
-
taxa_counts[genus] += 1
|
|
42
|
-
total_matches += 1
|
|
43
|
-
break # Stop checking further functional columns for this row
|
|
44
|
-
|
|
45
|
-
return taxa_counts, total_matches
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
def main():
|
|
49
|
-
|
|
50
|
-
parser = argparse.ArgumentParser(description='MetaPont ' + MetaPont_Version + ': Extract-By-Function - Identify taxa contributing to a specific function.')
|
|
51
|
-
parser.add_argument(
|
|
52
|
-
"-d", "--directory", required=True,
|
|
53
|
-
help="Directory containing TSV files to analyse."
|
|
54
|
-
)
|
|
55
|
-
parser.add_argument(
|
|
56
|
-
"-f", "--function_id", required=True,
|
|
57
|
-
help="Specific function ID to search for (e.g., 'GO:0002')."
|
|
58
|
-
)
|
|
59
|
-
parser.add_argument(
|
|
60
|
-
"-o", "--output", default="output_taxa_proportions.tsv",
|
|
61
|
-
help="Output file to save results (default: output_taxa_proportions.tsv)."
|
|
62
|
-
)
|
|
63
|
-
parser.add_argument(
|
|
64
|
-
"-m", "--min_proportion", type=float, default=0.05,
|
|
65
|
-
help="Minimum proportion threshold for taxa to be included in the output (default: 0.05)."
|
|
66
|
-
)
|
|
67
|
-
|
|
68
|
-
options = parser.parse_args()
|
|
69
|
-
print("Running MetaPont: Extract-By-Function " + MetaPont_Version)
|
|
70
|
-
|
|
71
|
-
all_results = {}
|
|
72
|
-
|
|
73
|
-
# Process each TSV file in the directory
|
|
74
|
-
for file_name in os.listdir(options.directory):
|
|
75
|
-
if file_name.endswith("_Final_Contig.tsv"):
|
|
76
|
-
file_path = os.path.join(options.directory, file_name)
|
|
77
|
-
print(f"Processing file: {file_name}")
|
|
78
|
-
taxa_counts, total_matches = process_tsv(file_path, options.function_id)
|
|
79
|
-
all_results[file_name] = (taxa_counts, total_matches)
|
|
80
|
-
|
|
81
|
-
# Write results to output
|
|
82
|
-
with open(options.output, "w") as out:
|
|
83
|
-
out.write("Function ID: " + options.function_id + "\n")
|
|
84
|
-
out.write("Sample\tTaxa\tProportion\n")
|
|
85
|
-
for sample, (taxa_counts, total_matches) in all_results.items():
|
|
86
|
-
for taxa, count in taxa_counts.items():
|
|
87
|
-
proportion = count / total_matches if total_matches > 0 else 0
|
|
88
|
-
if proportion >= options.min_proportion: # Apply minimum proportion filter
|
|
89
|
-
out.write(f"{sample}\t{taxa}\t{proportion:.6f}\n")
|
|
90
|
-
|
|
91
|
-
print(f"Results saved to {options.output}")
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
if __name__ == "__main__":
|
|
95
|
-
main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|