instanexus 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
instanexus/__init__.py ADDED
File without changes
instanexus/__main__.py ADDED
@@ -0,0 +1,54 @@
1
+ import sys
2
+ import argparse
3
+ from . import script_dbg, script_greedy
4
+
5
+ def main():
6
+ banner = r"""
7
+ ______ __ __ __
8
+ /\__ _\ /\ \__ /\ \/\ \
9
+ \/_/\ \/ ___ ____\ \ ,_\ __ \ \ `\\ \ __ __ _ __ __ ____
10
+ \ \ \ /' _ `\ /',__\\ \ \/ /'__`\ \ \ , ` \ /'__`\/\ \/'\/\ \/\ \ /',__\
11
+ \_\ \__/\ \/\ \/\__, `\\ \ \_/\ \L\.\_\ \ \`\ \/\ __/\/> </\ \ \_\ \/\__, `\
12
+ /\_____\ \_\ \_\/\____/ \ \__\ \__/.\_\\ \_\ \_\ \____\/\_/\_\\ \____/\/\____/
13
+ \/_____/\/_/\/_/\/___/ \/__/\/__/\/_/ \/_/\/_/\/____/\//\/_/ \/___/ \/___/
14
+ """
15
+
16
+ parser = argparse.ArgumentParser(
17
+ prog="instanexus",
18
+ description=(banner + "\n"
19
+ "InstaNexus CLI: de novo protein sequencing based on InstaNovo,\n\n" \
20
+ "an end-to-end workflow from de novo peptides to proteins\n\n"
21
+ "Usage:\n"
22
+ " instanexus <command> [options]\n\n"
23
+ "Available commands:\n"
24
+ " dbg Run De Bruijn Graph assembly pipeline\n"
25
+ " greedy Run greedy assembly pipeline\n\n"
26
+ "Examples:\n"
27
+ " instanexus dbg --input_csv inputs/sample.csv --chain light --folder_outputs outputs --reference\n"
28
+ " instanexus greedy --input_csv inputs/sample.csv --folder_outputs outputs\n\n"
29
+ "Use 'instanexus <command> --help' for detailed options."
30
+ ),
31
+ formatter_class=argparse.RawTextHelpFormatter,
32
+ )
33
+
34
+ parser.add_argument('--version', action='version', version='InstaNexus 0.1.0'),
35
+
36
+ subparsers = parser.add_subparsers(dest="command", help="subcommands")
37
+
38
+ # subcommands
39
+ subparsers.add_parser("dbg", help="Run de Bruijn graph assembly pipeline")
40
+ subparsers.add_parser("greedy", help="Run greedy assembly pipeline")
41
+
42
+ args, extra = parser.parse_known_args()
43
+
44
+ if args.command == "dbg":
45
+ sys.argv = [sys.argv[0]] + extra
46
+ script_dbg.cli()
47
+ elif args.command == "greedy":
48
+ sys.argv = [sys.argv[0]] + extra
49
+ script_greedy.cli()
50
+ else:
51
+ parser.print_help()
52
+
53
+ if __name__ == "__main__":
54
+ main()
@@ -0,0 +1,50 @@
1
+ import os
2
+ import shutil
3
+ import subprocess
4
+
5
+ from Bio import SeqIO
6
+
7
+
8
+ def align_or_copy_fasta(fasta_file, output_file):
9
+
10
+ sequences = list(SeqIO.parse(fasta_file, "fasta"))
11
+
12
+ if len(sequences) == 1:
13
+ shutil.copy(fasta_file, output_file)
14
+ else:
15
+ subprocess.run(
16
+ ["clustalo", "-i", fasta_file, "-o", output_file, "--outfmt", "fa"]
17
+ )
18
+
19
+
20
+ def process_alignment(input_folder):
21
+ """
22
+ Process all fasta files in the cluster_fasta folder, align them if necessary,
23
+ and save the results in the align folder.
24
+ """
25
+ cluster_fasta_folder = os.path.join(input_folder, "cluster_fasta")
26
+ align_folder = os.path.join(input_folder, "align")
27
+
28
+ # Create the align folder if it does not exist
29
+ os.makedirs(align_folder, exist_ok=True)
30
+
31
+ # Iterate over all folders in the cluster fasta folder
32
+ for cluster_folder in os.listdir(cluster_fasta_folder):
33
+ cluster_folder_path = os.path.join(cluster_fasta_folder, cluster_folder)
34
+ if os.path.isdir(cluster_folder_path):
35
+ # Create a corresponding folder in the align folder
36
+ output_cluster_folder = os.path.join(align_folder, cluster_folder)
37
+ os.makedirs(output_cluster_folder, exist_ok=True)
38
+
39
+ # Iterate over all fasta files in the cluster folder
40
+ for fasta_file in os.listdir(cluster_folder_path):
41
+ if fasta_file.endswith(".fasta"):
42
+ fasta_file_path = os.path.join(cluster_folder_path, fasta_file)
43
+ base_filename = os.path.splitext(fasta_file)[0]
44
+ output_file = os.path.join(
45
+ output_cluster_folder, f"{base_filename}_out.afa"
46
+ )
47
+
48
+ align_or_copy_fasta(fasta_file_path, output_file)
49
+
50
+ print("All alignment tasks completed.")
@@ -0,0 +1,105 @@
1
+ #!/usr/bin/env python
2
+
3
+ r"""
4
+ _____ _______ _ _
5
+ | __ \|__ __|| | | |
6
+ | | | | | | | | | |
7
+ | | | | | | | | | |
8
+ | |__| | | | | |__| |
9
+ |_____/ |_| |______|
10
+
11
+ __authors__ = Marco Reverenna & Konstantinos Kalogeropoulus
12
+ __copyright__ = Copyright 2024-2025
13
+ __research-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
14
+ __date__ = 21 Mar 2025
15
+ __maintainer__ = Marco Reverenna
16
+ __email__ = marcor@dtu.dk
17
+ __status__ = Dev
18
+ """
19
+
20
+ import os
21
+ import shutil
22
+ import subprocess
23
+ from tempfile import mkdtemp
24
+
25
+ import Bio.SeqIO
26
+ import pandas as pd
27
+ from tqdm import tqdm
28
+
29
+
30
+ def cluster_fasta_files(input_folder):
31
+
32
+ cluster_folder = os.path.join(input_folder, "cluster")
33
+ os.makedirs(cluster_folder, exist_ok=True)
34
+
35
+ temp_dir = mkdtemp(prefix="mmseqs-") # create a temporary directory for mmseqs
36
+
37
+ # Iterate over all fasta files in the folder
38
+ for fasta_file in os.listdir(input_folder):
39
+ if fasta_file.endswith(".fasta"):
40
+ fasta_path = os.path.join(input_folder, fasta_file)
41
+ print(f"the current fasta path is: {fasta_path}")
42
+
43
+ if os.path.isfile(fasta_path):
44
+
45
+ base_filename = os.path.splitext(fasta_file)[
46
+ 0
47
+ ] # get the base filename with no extension
48
+
49
+ prefix = os.path.join(
50
+ cluster_folder, base_filename
51
+ ) # define the prefix for mmseqs easy-cluster
52
+
53
+ print(f"Clustering {fasta_file}...") # run mmseqs easy-cluster
54
+ subprocess.run(
55
+ [
56
+ "mmseqs",
57
+ "easy-cluster",
58
+ fasta_path,
59
+ prefix,
60
+ temp_dir,
61
+ "--min-seq-id",
62
+ "0.85",
63
+ "-c",
64
+ "0.8",
65
+ "--cov-mode",
66
+ "1",
67
+ "-v",
68
+ "1",
69
+ ]
70
+ )
71
+ print(
72
+ f"Clustering completed for {fasta_file}, results stored with prefix {prefix}"
73
+ )
74
+
75
+ shutil.rmtree(temp_dir)
76
+ print("All clustering tasks completed.")
77
+
78
+
79
+ def process_fasta_and_clusters(fasta_file, cluster_tsv_folder, output_base_folder):
80
+
81
+ base_filename = os.path.basename(fasta_file).rsplit(".", 1)[0]
82
+
83
+ cluster_tsv = os.path.join(cluster_tsv_folder, f"{base_filename}_cluster.tsv")
84
+
85
+ if not os.path.isfile(cluster_tsv):
86
+ print(f"Cluster TSV file not found for {fasta_file}, skipping.")
87
+ return
88
+
89
+ output_folder = os.path.join(output_base_folder, f"{base_filename}_cluster_fasta")
90
+ os.makedirs(output_folder, exist_ok=True)
91
+
92
+ cluster_df = pd.read_csv(
93
+ cluster_tsv, sep="\t", header=None, names=["cluster", "contig"]
94
+ )
95
+
96
+ records = list(Bio.SeqIO.parse(fasta_file, "fasta"))
97
+
98
+ clusters = cluster_df["cluster"].unique()
99
+
100
+ for cluster in tqdm(clusters, desc=f"Processing clusters for {base_filename}"):
101
+ contigs = cluster_df[cluster_df["cluster"] == cluster]["contig"].values
102
+ contig_records = [record for record in records if record.id in contigs]
103
+ Bio.SeqIO.write(
104
+ contig_records, os.path.join(output_folder, f"{cluster}.fasta"), "fasta"
105
+ )
@@ -0,0 +1,73 @@
1
+ #!/usr/bin/env python
2
+
3
+ r"""
4
+ _____ _______ _ _
5
+ | __ \|__ __|| | | |
6
+ | | | | | | | | | |
7
+ | | | | | | | | | |
8
+ | |__| | | | | |__| |
9
+ |_____/ |_| |______|
10
+
11
+ __authors__ = Marco Reverenna
12
+ __copyright__ = Copyright 2025-2026
13
+ __research-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
14
+ __date__ = 26 Jun 2025
15
+ __maintainer__ = Marco Reverenna
16
+ __email__ = marcor@dtu.dk
17
+ __status__ = Dev
18
+ """
19
+
20
+ import os
21
+
22
+ import pandas as pd
23
+
24
+ base_directory = (
25
+ "/home/marcor/works/assembly/outputs/ma1/light" # Change this to your actual path
26
+ )
27
+
28
+ merged_data = []
29
+
30
+ log_file_path = os.path.join(base_directory, "missing_statistics_log.txt")
31
+ missing_folders = []
32
+
33
+ for folder in os.listdir(base_directory):
34
+ folder_path = os.path.join(base_directory, folder)
35
+ statistics_path = os.path.join(folder_path, "statistics")
36
+
37
+ if os.path.isdir(statistics_path):
38
+ contigs_file = os.path.join(statistics_path, "contigs_stats_default.tsv")
39
+ scaffolds_file = os.path.join(statistics_path, "scaffolds_stats_default.tsv")
40
+
41
+ if os.path.exists(contigs_file) and os.path.exists(scaffolds_file):
42
+ # Read both files
43
+ df_contigs = pd.read_csv(contigs_file, sep="\t")
44
+ df_scaffolds = pd.read_csv(scaffolds_file, sep="\t")
45
+
46
+ # Add a column indicating the source folder
47
+ df_contigs["Source_Folder"] = folder
48
+ df_scaffolds["Source_Folder"] = folder
49
+
50
+ # Append to the merged list
51
+ merged_data.append(df_contigs)
52
+ merged_data.append(df_scaffolds)
53
+ else:
54
+ # Log missing files
55
+ missing_folders.append(folder)
56
+
57
+ if merged_data:
58
+ final_df = pd.concat(merged_data, ignore_index=True)
59
+ final_df.to_csv(
60
+ os.path.join(base_directory, "merged_statistics.tsv"), sep="\t", index=False
61
+ )
62
+ print("Merged DataFrame saved as 'merged_statistics.tsv'.")
63
+ else:
64
+ print("No valid statistics files found.")
65
+
66
+ # Write the log file
67
+ if missing_folders:
68
+ with open(log_file_path, "w") as log_file:
69
+ log_file.write("Missing statistics files in the following folders:\n")
70
+ log_file.write("\n".join(missing_folders))
71
+ print(f"Log file saved as '{log_file_path}'.")
72
+ else:
73
+ print("All statistics files were found.")
@@ -0,0 +1,94 @@
1
+ #!/usr/bin/env python
2
+
3
+ """Calculates and saves statistics.
4
+ _____ _______ _ _
5
+ | __ \|__ __|| | | |
6
+ | | | | | | | | | |
7
+ | | | | | | | | | |
8
+ | |__| | | | | |__| |
9
+ |_____/ |_| |______|
10
+
11
+ __authors__ = Marco Reverenna
12
+ __copyright__ = Copyright 2024-2025
13
+ __reserach-group__ = DTU Biosustain (Multi-omics Network Analytics) and DTU Bioengineering
14
+ __date__ = 26 Jun 2024
15
+ __maintainer__ = Marco Reverenna
16
+ __email__ = marcor@dtu.dk
17
+ __status__ = Dev
18
+ """
19
+
20
+ import json
21
+ import os
22
+
23
+
24
+ def compute_assembly_statistics(df, sequence_type, output_folder, reference, **params):
25
+ """Statistics for contigs and scaffolds
26
+
27
+ Args:
28
+ df: DataFrame with mapped values
29
+ sequence_type: either 'contigs' or 'scaffold'
30
+ output_folder: folder to save output
31
+ reference: reference protein normalized
32
+ """
33
+
34
+ statistics = {}
35
+ statistics.update(params) # add the hyperparameters to the statistics
36
+
37
+ df["sequence_length"] = df["end"] - df["start"] + 1
38
+
39
+ # Reference coordinates
40
+ statistics["reference_start"] = int(0)
41
+ statistics["reference_end"] = int(len(reference) + 1)
42
+
43
+ # Sequences statistics
44
+ statistics["total_sequences"] = int(len(df))
45
+ statistics["average_length"] = float(df["sequence_length"].mean())
46
+ statistics["min_length"] = int(df["sequence_length"].min())
47
+ statistics["max_length"] = int(df["sequence_length"].max())
48
+
49
+ # Create a set of covered positions (adjusting for 0-based indexing)
50
+ covered_positions = set()
51
+ for start, end in zip(df["start"], df["end"]):
52
+ covered_positions.update(
53
+ range(start - 1, end)
54
+ ) # Convert 1-based to 0-based indexing
55
+ statistics["coverage"] = float(len(covered_positions) / statistics["reference_end"])
56
+
57
+ # Identity score statistics
58
+ statistics["mean_identity"] = float(df["identity_score"].mean())
59
+ statistics["median_identity"] = float(df["identity_score"].median())
60
+ # statistics['std_identity'] = float(df['identity_score'].std())
61
+
62
+ # Mismatch statistics
63
+ statistics["perfect_matches"] = int(
64
+ sum(df["mismatches_pos"].apply(len) == 0)
65
+ ) # sequences with no mismatches
66
+ all_mismatches = [pos for mismatches in df["mismatches_pos"] for pos in mismatches]
67
+ statistics["total_mismatches"] = int(len(set(all_mismatches)))
68
+
69
+ # N50 and N90 calculations
70
+ lengths = sorted(df["sequence_length"], reverse=True)
71
+ total_length = sum(lengths)
72
+
73
+ cumulative_length = 0
74
+ n50 = None
75
+ n90 = None
76
+ for length in lengths:
77
+ cumulative_length += length
78
+ if n50 is None and cumulative_length >= total_length * 0.5:
79
+ n50 = length
80
+ if n90 is None and cumulative_length >= total_length * 0.9:
81
+ n90 = length
82
+ if n50 is not None and n90 is not None:
83
+ break
84
+
85
+ statistics["N50"] = int(n50)
86
+ statistics["N90"] = int(n90)
87
+
88
+ # Save JSON file
89
+ file_name = f"{sequence_type}_stats.json"
90
+ output_path = os.path.join(output_folder, file_name)
91
+ with open(output_path, "w") as file:
92
+ json.dump(statistics, file, indent=4)
93
+
94
+ return statistics