instanexus 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- instanexus/__init__.py +0 -0
- instanexus/__main__.py +54 -0
- instanexus/alignment.py +50 -0
- instanexus/clustering.py +105 -0
- instanexus/clustering_stats.py +73 -0
- instanexus/compute_statistics.py +94 -0
- instanexus/consensus.py +336 -0
- instanexus/dbg.py +384 -0
- instanexus/generate_cluster_fasta.py +50 -0
- instanexus/greedy_method.py +322 -0
- instanexus/mapping.py +639 -0
- instanexus/model_peptide_selector.py +467 -0
- instanexus/opt/__init__.py +0 -0
- instanexus/opt/gridsearch.py +99 -0
- instanexus/opt/opt_dbg.py +221 -0
- instanexus/opt/opt_greedy.py +203 -0
- instanexus/preprocessing.py +817 -0
- instanexus/scaffolding.py +259 -0
- instanexus/script_dbg.py +294 -0
- instanexus/script_greedy.py +303 -0
- instanexus-0.1.0.dist-info/METADATA +238 -0
- instanexus-0.1.0.dist-info/RECORD +25 -0
- instanexus-0.1.0.dist-info/WHEEL +4 -0
- instanexus-0.1.0.dist-info/entry_points.txt +2 -0
- instanexus-0.1.0.dist-info/licenses/LICENSE +21 -0
instanexus/__init__.py
ADDED
|
File without changes
|
instanexus/__main__.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import argparse
|
|
3
|
+
from . import script_dbg, script_greedy
|
|
4
|
+
|
|
5
|
+
def main():
|
|
6
|
+
banner = r"""
|
|
7
|
+
______ __ __ __
|
|
8
|
+
/\__ _\ /\ \__ /\ \/\ \
|
|
9
|
+
\/_/\ \/ ___ ____\ \ ,_\ __ \ \ `\\ \ __ __ _ __ __ ____
|
|
10
|
+
\ \ \ /' _ `\ /',__\\ \ \/ /'__`\ \ \ , ` \ /'__`\/\ \/'\/\ \/\ \ /',__\
|
|
11
|
+
\_\ \__/\ \/\ \/\__, `\\ \ \_/\ \L\.\_\ \ \`\ \/\ __/\/> </\ \ \_\ \/\__, `\
|
|
12
|
+
/\_____\ \_\ \_\/\____/ \ \__\ \__/.\_\\ \_\ \_\ \____\/\_/\_\\ \____/\/\____/
|
|
13
|
+
\/_____/\/_/\/_/\/___/ \/__/\/__/\/_/ \/_/\/_/\/____/\//\/_/ \/___/ \/___/
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
prog="instanexus",
|
|
18
|
+
description=(banner + "\n"
|
|
19
|
+
"InstaNexus CLI: de novo protein sequencing based on InstaNovo,\n\n" \
|
|
20
|
+
"an end-to-end workflow from de novo peptides to proteins\n\n"
|
|
21
|
+
"Usage:\n"
|
|
22
|
+
" instanexus <command> [options]\n\n"
|
|
23
|
+
"Available commands:\n"
|
|
24
|
+
" dbg Run De Bruijn Graph assembly pipeline\n"
|
|
25
|
+
" greedy Run greedy assembly pipeline\n\n"
|
|
26
|
+
"Examples:\n"
|
|
27
|
+
" instanexus dbg --input_csv inputs/sample.csv --chain light --folder_outputs outputs --reference\n"
|
|
28
|
+
" instanexus greedy --input_csv inputs/sample.csv --folder_outputs outputs\n\n"
|
|
29
|
+
"Use 'instanexus <command> --help' for detailed options."
|
|
30
|
+
),
|
|
31
|
+
formatter_class=argparse.RawTextHelpFormatter,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
parser.add_argument('--version', action='version', version='InstaNexus 0.1.0'),
|
|
35
|
+
|
|
36
|
+
subparsers = parser.add_subparsers(dest="command", help="subcommands")
|
|
37
|
+
|
|
38
|
+
# subcommands
|
|
39
|
+
subparsers.add_parser("dbg", help="Run de Bruijn graph assembly pipeline")
|
|
40
|
+
subparsers.add_parser("greedy", help="Run greedy assembly pipeline")
|
|
41
|
+
|
|
42
|
+
args, extra = parser.parse_known_args()
|
|
43
|
+
|
|
44
|
+
if args.command == "dbg":
|
|
45
|
+
sys.argv = [sys.argv[0]] + extra
|
|
46
|
+
script_dbg.cli()
|
|
47
|
+
elif args.command == "greedy":
|
|
48
|
+
sys.argv = [sys.argv[0]] + extra
|
|
49
|
+
script_greedy.cli()
|
|
50
|
+
else:
|
|
51
|
+
parser.print_help()
|
|
52
|
+
|
|
53
|
+
if __name__ == "__main__":
|
|
54
|
+
main()
|
instanexus/alignment.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import shutil
|
|
3
|
+
import subprocess
|
|
4
|
+
|
|
5
|
+
from Bio import SeqIO
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def align_or_copy_fasta(fasta_file, output_file):
|
|
9
|
+
|
|
10
|
+
sequences = list(SeqIO.parse(fasta_file, "fasta"))
|
|
11
|
+
|
|
12
|
+
if len(sequences) == 1:
|
|
13
|
+
shutil.copy(fasta_file, output_file)
|
|
14
|
+
else:
|
|
15
|
+
subprocess.run(
|
|
16
|
+
["clustalo", "-i", fasta_file, "-o", output_file, "--outfmt", "fa"]
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def process_alignment(input_folder):
|
|
21
|
+
"""
|
|
22
|
+
Process all fasta files in the cluster_fasta folder, align them if necessary,
|
|
23
|
+
and save the results in the align folder.
|
|
24
|
+
"""
|
|
25
|
+
cluster_fasta_folder = os.path.join(input_folder, "cluster_fasta")
|
|
26
|
+
align_folder = os.path.join(input_folder, "align")
|
|
27
|
+
|
|
28
|
+
# Create the align folder if it does not exist
|
|
29
|
+
os.makedirs(align_folder, exist_ok=True)
|
|
30
|
+
|
|
31
|
+
# Iterate over all folders in the cluster fasta folder
|
|
32
|
+
for cluster_folder in os.listdir(cluster_fasta_folder):
|
|
33
|
+
cluster_folder_path = os.path.join(cluster_fasta_folder, cluster_folder)
|
|
34
|
+
if os.path.isdir(cluster_folder_path):
|
|
35
|
+
# Create a corresponding folder in the align folder
|
|
36
|
+
output_cluster_folder = os.path.join(align_folder, cluster_folder)
|
|
37
|
+
os.makedirs(output_cluster_folder, exist_ok=True)
|
|
38
|
+
|
|
39
|
+
# Iterate over all fasta files in the cluster folder
|
|
40
|
+
for fasta_file in os.listdir(cluster_folder_path):
|
|
41
|
+
if fasta_file.endswith(".fasta"):
|
|
42
|
+
fasta_file_path = os.path.join(cluster_folder_path, fasta_file)
|
|
43
|
+
base_filename = os.path.splitext(fasta_file)[0]
|
|
44
|
+
output_file = os.path.join(
|
|
45
|
+
output_cluster_folder, f"{base_filename}_out.afa"
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
align_or_copy_fasta(fasta_file_path, output_file)
|
|
49
|
+
|
|
50
|
+
print("All alignment tasks completed.")
|
instanexus/clustering.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
|
|
3
|
+
r"""
|
|
4
|
+
_____ _______ _ _
|
|
5
|
+
| __ \|__ __|| | | |
|
|
6
|
+
| | | | | | | | | |
|
|
7
|
+
| | | | | | | | | |
|
|
8
|
+
| |__| | | | | |__| |
|
|
9
|
+
|_____/ |_| |______|
|
|
10
|
+
|
|
11
|
+
__authors__ = Marco Reverenna & Konstantinos Kalogeropoulus
|
|
12
|
+
__copyright__ = Copyright 2024-2025
|
|
13
|
+
__research-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
|
|
14
|
+
__date__ = 21 Mar 2025
|
|
15
|
+
__maintainer__ = Marco Reverenna
|
|
16
|
+
__email__ = marcor@dtu.dk
|
|
17
|
+
__status__ = Dev
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import shutil
|
|
22
|
+
import subprocess
|
|
23
|
+
from tempfile import mkdtemp
|
|
24
|
+
|
|
25
|
+
import Bio.SeqIO
|
|
26
|
+
import pandas as pd
|
|
27
|
+
from tqdm import tqdm
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def cluster_fasta_files(input_folder):
|
|
31
|
+
|
|
32
|
+
cluster_folder = os.path.join(input_folder, "cluster")
|
|
33
|
+
os.makedirs(cluster_folder, exist_ok=True)
|
|
34
|
+
|
|
35
|
+
temp_dir = mkdtemp(prefix="mmseqs-") # create a temporary directory for mmseqs
|
|
36
|
+
|
|
37
|
+
# Iterate over all fasta files in the folder
|
|
38
|
+
for fasta_file in os.listdir(input_folder):
|
|
39
|
+
if fasta_file.endswith(".fasta"):
|
|
40
|
+
fasta_path = os.path.join(input_folder, fasta_file)
|
|
41
|
+
print(f"the current fasta path is: {fasta_path}")
|
|
42
|
+
|
|
43
|
+
if os.path.isfile(fasta_path):
|
|
44
|
+
|
|
45
|
+
base_filename = os.path.splitext(fasta_file)[
|
|
46
|
+
0
|
|
47
|
+
] # get the base filename with no extension
|
|
48
|
+
|
|
49
|
+
prefix = os.path.join(
|
|
50
|
+
cluster_folder, base_filename
|
|
51
|
+
) # define the prefix for mmseqs easy-cluster
|
|
52
|
+
|
|
53
|
+
print(f"Clustering {fasta_file}...") # run mmseqs easy-cluster
|
|
54
|
+
subprocess.run(
|
|
55
|
+
[
|
|
56
|
+
"mmseqs",
|
|
57
|
+
"easy-cluster",
|
|
58
|
+
fasta_path,
|
|
59
|
+
prefix,
|
|
60
|
+
temp_dir,
|
|
61
|
+
"--min-seq-id",
|
|
62
|
+
"0.85",
|
|
63
|
+
"-c",
|
|
64
|
+
"0.8",
|
|
65
|
+
"--cov-mode",
|
|
66
|
+
"1",
|
|
67
|
+
"-v",
|
|
68
|
+
"1",
|
|
69
|
+
]
|
|
70
|
+
)
|
|
71
|
+
print(
|
|
72
|
+
f"Clustering completed for {fasta_file}, results stored with prefix {prefix}"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
shutil.rmtree(temp_dir)
|
|
76
|
+
print("All clustering tasks completed.")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def process_fasta_and_clusters(fasta_file, cluster_tsv_folder, output_base_folder):
|
|
80
|
+
|
|
81
|
+
base_filename = os.path.basename(fasta_file).rsplit(".", 1)[0]
|
|
82
|
+
|
|
83
|
+
cluster_tsv = os.path.join(cluster_tsv_folder, f"{base_filename}_cluster.tsv")
|
|
84
|
+
|
|
85
|
+
if not os.path.isfile(cluster_tsv):
|
|
86
|
+
print(f"Cluster TSV file not found for {fasta_file}, skipping.")
|
|
87
|
+
return
|
|
88
|
+
|
|
89
|
+
output_folder = os.path.join(output_base_folder, f"{base_filename}_cluster_fasta")
|
|
90
|
+
os.makedirs(output_folder, exist_ok=True)
|
|
91
|
+
|
|
92
|
+
cluster_df = pd.read_csv(
|
|
93
|
+
cluster_tsv, sep="\t", header=None, names=["cluster", "contig"]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
records = list(Bio.SeqIO.parse(fasta_file, "fasta"))
|
|
97
|
+
|
|
98
|
+
clusters = cluster_df["cluster"].unique()
|
|
99
|
+
|
|
100
|
+
for cluster in tqdm(clusters, desc=f"Processing clusters for {base_filename}"):
|
|
101
|
+
contigs = cluster_df[cluster_df["cluster"] == cluster]["contig"].values
|
|
102
|
+
contig_records = [record for record in records if record.id in contigs]
|
|
103
|
+
Bio.SeqIO.write(
|
|
104
|
+
contig_records, os.path.join(output_folder, f"{cluster}.fasta"), "fasta"
|
|
105
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
|
|
3
|
+
r"""
|
|
4
|
+
_____ _______ _ _
|
|
5
|
+
| __ \|__ __|| | | |
|
|
6
|
+
| | | | | | | | | |
|
|
7
|
+
| | | | | | | | | |
|
|
8
|
+
| |__| | | | | |__| |
|
|
9
|
+
|_____/ |_| |______|
|
|
10
|
+
|
|
11
|
+
__authors__ = Marco Reverenna
|
|
12
|
+
__copyright__ = Copyright 2025-2026
|
|
13
|
+
__research-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
|
|
14
|
+
__date__ = 26 Jun 2025
|
|
15
|
+
__maintainer__ = Marco Reverenna
|
|
16
|
+
__email__ = marcor@dtu.dk
|
|
17
|
+
__status__ = Dev
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
base_directory = (
|
|
25
|
+
"/home/marcor/works/assembly/outputs/ma1/light" # Change this to your actual path
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
merged_data = []
|
|
29
|
+
|
|
30
|
+
log_file_path = os.path.join(base_directory, "missing_statistics_log.txt")
|
|
31
|
+
missing_folders = []
|
|
32
|
+
|
|
33
|
+
for folder in os.listdir(base_directory):
|
|
34
|
+
folder_path = os.path.join(base_directory, folder)
|
|
35
|
+
statistics_path = os.path.join(folder_path, "statistics")
|
|
36
|
+
|
|
37
|
+
if os.path.isdir(statistics_path):
|
|
38
|
+
contigs_file = os.path.join(statistics_path, "contigs_stats_default.tsv")
|
|
39
|
+
scaffolds_file = os.path.join(statistics_path, "scaffolds_stats_default.tsv")
|
|
40
|
+
|
|
41
|
+
if os.path.exists(contigs_file) and os.path.exists(scaffolds_file):
|
|
42
|
+
# Read both files
|
|
43
|
+
df_contigs = pd.read_csv(contigs_file, sep="\t")
|
|
44
|
+
df_scaffolds = pd.read_csv(scaffolds_file, sep="\t")
|
|
45
|
+
|
|
46
|
+
# Add a column indicating the source folder
|
|
47
|
+
df_contigs["Source_Folder"] = folder
|
|
48
|
+
df_scaffolds["Source_Folder"] = folder
|
|
49
|
+
|
|
50
|
+
# Append to the merged list
|
|
51
|
+
merged_data.append(df_contigs)
|
|
52
|
+
merged_data.append(df_scaffolds)
|
|
53
|
+
else:
|
|
54
|
+
# Log missing files
|
|
55
|
+
missing_folders.append(folder)
|
|
56
|
+
|
|
57
|
+
if merged_data:
|
|
58
|
+
final_df = pd.concat(merged_data, ignore_index=True)
|
|
59
|
+
final_df.to_csv(
|
|
60
|
+
os.path.join(base_directory, "merged_statistics.tsv"), sep="\t", index=False
|
|
61
|
+
)
|
|
62
|
+
print("Merged DataFrame saved as 'merged_statistics.tsv'.")
|
|
63
|
+
else:
|
|
64
|
+
print("No valid statistics files found.")
|
|
65
|
+
|
|
66
|
+
# Write the log file
|
|
67
|
+
if missing_folders:
|
|
68
|
+
with open(log_file_path, "w") as log_file:
|
|
69
|
+
log_file.write("Missing statistics files in the following folders:\n")
|
|
70
|
+
log_file.write("\n".join(missing_folders))
|
|
71
|
+
print(f"Log file saved as '{log_file_path}'.")
|
|
72
|
+
else:
|
|
73
|
+
print("All statistics files were found.")
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
|
|
3
|
+
"""Calculates and saves statistics.
|
|
4
|
+
_____ _______ _ _
|
|
5
|
+
| __ \|__ __|| | | |
|
|
6
|
+
| | | | | | | | | |
|
|
7
|
+
| | | | | | | | | |
|
|
8
|
+
| |__| | | | | |__| |
|
|
9
|
+
|_____/ |_| |______|
|
|
10
|
+
|
|
11
|
+
__authors__ = Marco Reverenna
|
|
12
|
+
__copyright__ = Copyright 2024-2025
|
|
13
|
+
__reserach-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
|
|
14
|
+
__date__ = 26 Jun 2024
|
|
15
|
+
__maintainer__ = Marco Reverenna
|
|
16
|
+
__email__ = marcor@dtu.dk
|
|
17
|
+
__status__ = Dev
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def compute_assembly_statistics(df, sequence_type, output_folder, reference, **params):
|
|
25
|
+
"""Statistics for contigs and scaffolds
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
df: DataFrame with mapped values
|
|
29
|
+
sequence_type: either 'contigs' or 'scaffold'
|
|
30
|
+
output_folder: folder to save output
|
|
31
|
+
reference: reference protein normalized
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
statistics = {}
|
|
35
|
+
statistics.update(params) # add the hyperparameters to the statistics
|
|
36
|
+
|
|
37
|
+
df["sequence_length"] = df["end"] - df["start"] + 1
|
|
38
|
+
|
|
39
|
+
# Reference coordinates
|
|
40
|
+
statistics["reference_start"] = int(0)
|
|
41
|
+
statistics["reference_end"] = int(len(reference) + 1)
|
|
42
|
+
|
|
43
|
+
# Sequences statistics
|
|
44
|
+
statistics["total_sequences"] = int(len(df))
|
|
45
|
+
statistics["average_length"] = float(df["sequence_length"].mean())
|
|
46
|
+
statistics["min_length"] = int(df["sequence_length"].min())
|
|
47
|
+
statistics["max_length"] = int(df["sequence_length"].max())
|
|
48
|
+
|
|
49
|
+
# Create a set of covered positions (adjusting for 0-based indexing)
|
|
50
|
+
covered_positions = set()
|
|
51
|
+
for start, end in zip(df["start"], df["end"]):
|
|
52
|
+
covered_positions.update(
|
|
53
|
+
range(start - 1, end)
|
|
54
|
+
) # Convert 1-based to 0-based indexing
|
|
55
|
+
statistics["coverage"] = float(len(covered_positions) / statistics["reference_end"])
|
|
56
|
+
|
|
57
|
+
# Identity score statistics
|
|
58
|
+
statistics["mean_identity"] = float(df["identity_score"].mean())
|
|
59
|
+
statistics["median_identity"] = float(df["identity_score"].median())
|
|
60
|
+
# statistics['std_identity'] = float(df['identity_score'].std())
|
|
61
|
+
|
|
62
|
+
# Mismatch statistics
|
|
63
|
+
statistics["perfect_matches"] = int(
|
|
64
|
+
sum(df["mismatches_pos"].apply(len) == 0)
|
|
65
|
+
) # sequences with no mismatches
|
|
66
|
+
all_mismatches = [pos for mismatches in df["mismatches_pos"] for pos in mismatches]
|
|
67
|
+
statistics["total_mismatches"] = int(len(set(all_mismatches)))
|
|
68
|
+
|
|
69
|
+
# N50 and N90 calculations
|
|
70
|
+
lengths = sorted(df["sequence_length"], reverse=True)
|
|
71
|
+
total_length = sum(lengths)
|
|
72
|
+
|
|
73
|
+
cumulative_length = 0
|
|
74
|
+
n50 = None
|
|
75
|
+
n90 = None
|
|
76
|
+
for length in lengths:
|
|
77
|
+
cumulative_length += length
|
|
78
|
+
if n50 is None and cumulative_length >= total_length * 0.5:
|
|
79
|
+
n50 = length
|
|
80
|
+
if n90 is None and cumulative_length >= total_length * 0.9:
|
|
81
|
+
n90 = length
|
|
82
|
+
if n50 is not None and n90 is not None:
|
|
83
|
+
break
|
|
84
|
+
|
|
85
|
+
statistics["N50"] = int(n50)
|
|
86
|
+
statistics["N90"] = int(n90)
|
|
87
|
+
|
|
88
|
+
# Save JSON file
|
|
89
|
+
file_name = f"{sequence_type}_stats.json"
|
|
90
|
+
output_path = os.path.join(output_folder, file_name)
|
|
91
|
+
with open(output_path, "w") as file:
|
|
92
|
+
json.dump(statistics, file, indent=4)
|
|
93
|
+
|
|
94
|
+
return statistics
|