eefinder 1.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,16 @@
1
+ scripts/__pycache__/
2
+ *.pyc
3
+ *.pin
4
+ *.phr
5
+ *.psq
6
+ output/
7
+
8
+ docs/_build/
9
+
10
+ # Build artifacts
11
+ dist/
12
+ build/
13
+ *.egg-info/
14
+
15
+ # macOS
16
+ .DS_Store
eefinder-1.1.2/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 WallauBioinfo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.5
2
+ Name: eefinder
3
+ Version: 1.1.2
4
+ Summary: Identification of endogenous viral and bacterial elements in eukaryotic genomes
5
+ Project-URL: Homepage, https://github.com/WallauBioinfo/EEfinder
6
+ Project-URL: Documentation, https://eefinder.readthedocs.io
7
+ Project-URL: Repository, https://github.com/WallauBioinfo/EEfinder
8
+ Project-URL: Issues, https://github.com/WallauBioinfo/EEfinder/issues
9
+ Project-URL: Publication, https://doi.org/10.1016/j.csbj.2024.10.012
10
+ Author-email: Filipe Zimmer Dezordi <zimmer.filipe@gmail.com>, Yago Dias <yag.dias@gmail.com>
11
+ Maintainer-email: Filipe Zimmer Dezordi <zimmer.filipe@gmail.com>
12
+ License-Expression: MIT
13
+ License-File: LICENSE
14
+ Keywords: EVE,bioinformatics,endogenous elements,genomics,horizontal gene transfer,virus
15
+ Classifier: Development Status :: 5 - Production/Stable
16
+ Classifier: Environment :: Console
17
+ Classifier: Intended Audience :: Science/Research
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3.9
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
24
+ Requires-Python: <3.12,>=3.9
25
+ Requires-Dist: biopython<1.86,>=1.79
26
+ Requires-Dist: click>=8.1
27
+ Requires-Dist: numpy<3,>=1.22
28
+ Requires-Dist: pandas<3,>=1.4
29
+ Description-Content-Type: text/markdown
30
+
31
+ # EEfinder
32
+
33
+ EEfinder is a tool/python package that automatizes several tasks related to identification of Endogenous Elements present on Eukaryotic Genomes.
34
+
35
+ #### Install
36
+
37
+ EEfinder drives BLAST, DIAMOND and bedtools, which are not pip-installable — install them first, then the package.
38
+
39
+ ##### From PyPI
40
+
41
+ ```bash
42
+ conda create -n EEfinder -c conda-forge -c bioconda \
43
+ "python>=3.9,<3.12" "blast>=2.5" "diamond>=2.0.15" "bedtools>=2.27"
44
+ conda activate EEfinder
45
+
46
+ pip install eefinder
47
+ ```
48
+
49
+ ##### From source
50
+
51
+ Cloning also gives you `env.yml`, which pins the exact versions EEfinder was developed and tested against (BLAST 2.5.0, DIAMOND 2.0.15, bedtools 2.27.1), plus the example data in `test_files/`.
52
+
53
+ ```bash
54
+ git clone https://github.com/WallauBioinfo/EEfinder.git
55
+ cd EEfinder
56
+ conda env create -f env.yml
57
+ conda activate EEfinder
58
+ pip install .
59
+ ```
60
+
61
+ #### Check tool
62
+
63
+ ```bash
64
+ eefinder --version
65
+
66
+ #eefinder, version 1.1.2
67
+ ```
68
+
69
+ #### Documentation
70
+
71
+ Full documentation is hosted on Read the Docs: **https://eefinder.readthedocs.io**
72
+
73
+ #### Cite us
74
+
75
+ If you use EEfinder in your research, please cite https://www.sciencedirect.com/science/article/pii/S2001037024003325:
76
+
77
+ ```
78
+ Dias, Y. J. M., Dezordi, F. Z., & Wallau, G. L. (2024). EEFinder: A general-purpose tool for identification of bacterial and viral endogenized elements in eukaryotic genomes. Computational and Structural Biotechnology Journal. https://doi.org/10.1016/j.csbj.2024.10.012
79
+ ```
@@ -0,0 +1,49 @@
1
+ # EEfinder
2
+
3
+ EEfinder is a tool/python package that automatizes several tasks related to identification of Endogenous Elements present on Eukaryotic Genomes.
4
+
5
+ #### Install
6
+
7
+ EEfinder drives BLAST, DIAMOND and bedtools, which are not pip-installable — install them first, then the package.
8
+
9
+ ##### From PyPI
10
+
11
+ ```bash
12
+ conda create -n EEfinder -c conda-forge -c bioconda \
13
+ "python>=3.9,<3.12" "blast>=2.5" "diamond>=2.0.15" "bedtools>=2.27"
14
+ conda activate EEfinder
15
+
16
+ pip install eefinder
17
+ ```
18
+
19
+ ##### From source
20
+
21
+ Cloning also gives you `env.yml`, which pins the exact versions EEfinder was developed and tested against (BLAST 2.5.0, DIAMOND 2.0.15, bedtools 2.27.1), plus the example data in `test_files/`.
22
+
23
+ ```bash
24
+ git clone https://github.com/WallauBioinfo/EEfinder.git
25
+ cd EEfinder
26
+ conda env create -f env.yml
27
+ conda activate EEfinder
28
+ pip install .
29
+ ```
30
+
31
+ #### Check tool
32
+
33
+ ```bash
34
+ eefinder --version
35
+
36
+ #eefinder, version 1.1.2
37
+ ```
38
+
39
+ #### Documentation
40
+
41
+ Full documentation is hosted on Read the Docs: **https://eefinder.readthedocs.io**
42
+
43
+ #### Cite us
44
+
45
+ If you use EEfinder in your research, please cite https://www.sciencedirect.com/science/article/pii/S2001037024003325:
46
+
47
+ ```
48
+ Dias, Y. J. M., Dezordi, F. Z., & Wallau, G. L. (2024). EEFinder: A general-purpose tool for identification of bacterial and viral endogenized elements in eukaryotic genomes. Computational and Structural Biotechnology Journal. https://doi.org/10.1016/j.csbj.2024.10.012
49
+ ```
@@ -0,0 +1,6 @@
1
+ from importlib.metadata import PackageNotFoundError, version
2
+
3
+ try:
4
+ __version__ = version("eefinder")
5
+ except PackageNotFoundError: # pragma: no cover - package not installed
6
+ __version__ = "unknown"
@@ -0,0 +1,206 @@
1
+ import pandas as pd
2
+ import numpy as np
3
+ import shlex
4
+ import subprocess
5
+ import re
6
+
7
+
8
+ class GetFasta:
9
+ """
10
+ This function execute the bedtools getfasta.
11
+
12
+ Keyword arguments:
13
+ input_file: input_file, parsed with -in argument.
14
+ bed_file: bed file, genereated along the pipeline
15
+ out_file: output file
16
+ """
17
+
18
+ def __init__(self, input_file: str, bed_file: str, out_file: str) -> object:
19
+ self.input_file = input_file
20
+ self.bed_file = bed_file
21
+ self.out_file = out_file
22
+
23
+ self.get_fasta()
24
+
25
+ def get_fasta(self) -> None:
26
+ get_fasta = f"bedtools getfasta -fi {self.input_file} -bed {self.bed_file} -fo {self.out_file}"
27
+ get_fasta = shlex.split(get_fasta)
28
+ cmd_get_fasta = subprocess.Popen(
29
+ get_fasta, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
30
+ )
31
+ cmd_get_fasta.wait()
32
+
33
+
34
+ class GetAnnotBed:
35
+ """
36
+ Create a bed file that will be used to merge truncated EVEs of the same
37
+ family in the same sense based on a limite length treshold.
38
+
39
+ Keyword arguments:
40
+ blast_tax_info: csv file generated in the get_taxonomy_info function on get_taxonomy.py
41
+ merge_level: genus or family, choose which level going to merge nearby elements
42
+ """
43
+
44
+ def __init__(self, blast_tax_info: str, merge_level: str) -> object:
45
+ self.blast_tax_info = blast_tax_info
46
+ self.merge_level = merge_level
47
+
48
+ self.get_annotated_bed()
49
+
50
+ def get_annotated_bed(self) -> None:
51
+ df_blast_tax_info = pd.read_csv(self.blast_tax_info, sep=",")
52
+ df_blast_tax_info["qseqid"] = df_blast_tax_info["qseqid"].str.replace(
53
+ r"\:.*", "", regex=True
54
+ )
55
+ df_blast_tax_info["sseqid"] = (
56
+ df_blast_tax_info["sseqid"]
57
+ + "|"
58
+ + df_blast_tax_info["sense"]
59
+ + "|"
60
+ + df_blast_tax_info["pident"].astype(str)
61
+ )
62
+ df_blast_tax_info["Family"] = df_blast_tax_info["Family"].fillna("Unknown")
63
+ df_blast_tax_info["Genus"] = df_blast_tax_info["Genus"].fillna("Unknown")
64
+
65
+ if self.merge_level == "genus":
66
+ df_blast_tax_info["formated_name"] = np.where(
67
+ df_blast_tax_info["Genus"] != "Unknown",
68
+ df_blast_tax_info["qseqid"]
69
+ + "|"
70
+ + df_blast_tax_info["Family"]
71
+ + "|"
72
+ + df_blast_tax_info["Genus"]
73
+ + "|"
74
+ + df_blast_tax_info["sense"],
75
+ df_blast_tax_info["qseqid"]
76
+ + "|"
77
+ + df_blast_tax_info["sseqid"]
78
+ + "|"
79
+ + df_blast_tax_info["Genus"],
80
+ )
81
+ else:
82
+ df_blast_tax_info["formated_name"] = np.where(
83
+ df_blast_tax_info["Family"] != "Unknown",
84
+ df_blast_tax_info["qseqid"]
85
+ + "|"
86
+ + df_blast_tax_info["Family"]
87
+ + "|"
88
+ + df_blast_tax_info["sense"],
89
+ df_blast_tax_info["qseqid"]
90
+ + "|"
91
+ + df_blast_tax_info["sseqid"]
92
+ + "|"
93
+ + df_blast_tax_info["Family"],
94
+ )
95
+ bed_blast_info = df_blast_tax_info[
96
+ ["formated_name", "qstart", "qend", "sseqid"]
97
+ ].copy()
98
+ bed_blast_info = bed_blast_info.sort_values(
99
+ ["formated_name", "qstart"], ascending=(True, True)
100
+ )
101
+ bed_blast_info.to_csv(
102
+ f"{self.blast_tax_info}.bed", index=False, header=False, sep="\t"
103
+ )
104
+
105
+
106
+ class RemoveAnnotation:
107
+ """
108
+ Remove the annotated information generate into the get_annotated_bed function.
109
+
110
+ Keyword arguments:
111
+ bed_annotated_merged_file: tsv file generated in the merge_bedfile function on bed_merge.py
112
+ """
113
+
114
+ def __init__(self, bed_annotated_merged_file: str) -> object:
115
+ self.bed_annotated_merged_file = bed_annotated_merged_file
116
+
117
+ self.reformat_bed()
118
+
119
+ def reformat_bed(self) -> None:
120
+ df_merge_file = pd.read_csv(
121
+ self.bed_annotated_merged_file, sep="\t", header=None
122
+ )
123
+ df_merge_file.iloc[:, 0] = df_merge_file.iloc[:, 0].str.replace(
124
+ "\|.*", "", regex=True
125
+ )
126
+ df_merge_file.to_csv(
127
+ f"{self.bed_annotated_merged_file}.fmt", index=False, header=False, sep="\t"
128
+ )
129
+
130
+
131
+ class MergeBed:
132
+ """
133
+ Execute the bedtools merge.
134
+
135
+ Keyword arguments:
136
+ bed_annotated_file: annotated bed file created at get_annotated_bed function
137
+ limit_merge: Limit of bases to merge regions, parsed with -lm argument
138
+ """
139
+
140
+ def __init__(self, bed_annotated_file: str, limit_merge: int) -> object:
141
+ self.bed_annotated_file = bed_annotated_file
142
+ self.limit_merge = limit_merge
143
+
144
+ self.merge_bed()
145
+
146
+ def merge_bed(self) -> None:
147
+ bed_merge_output = open(f"{self.bed_annotated_file}.merge", "w")
148
+ bed_merge_cmd = f'bedtools merge -d {int(self.limit_merge)} -i {self.bed_annotated_file} -c 4 -o collapse -delim " AND "'
149
+ bed_merge_cmd = shlex.split(bed_merge_cmd)
150
+ bed_merge_process = subprocess.Popen(bed_merge_cmd, stdout=bed_merge_output)
151
+ bed_merge_process.wait()
152
+
153
+
154
+ class BedFlank:
155
+ """
156
+ Extract flanking regions of EEs using bedtools slop.
157
+
158
+ Keyword arguments:
159
+ input_file: bed file generated by get_bed function
160
+ lenght_file: lenght file produced by get_length function
161
+ flank_region: desired lenght regions for extraction, parsed from
162
+ """
163
+
164
+ def __init__(self, input_file: str, length_file: str, flank_region: int) -> object:
165
+ self.input_file = input_file
166
+ self.length_file = length_file
167
+ self.flank_region = flank_region
168
+
169
+ self.bedtools_flank()
170
+
171
+ def bedtools_flank(self) -> None:
172
+ with open(f"{self.input_file}.flank", "w") as flank_out:
173
+ bed_flank_cmd = f"bedtools slop -i {self.input_file} -g {self.length_file} -b {str(self.flank_region)}"
174
+ bed_flank_cmd = shlex.split(bed_flank_cmd)
175
+ bed_flank_process = subprocess.Popen(bed_flank_cmd, stdout=flank_out)
176
+ bed_flank_process.wait()
177
+
178
+
179
+ class GetBed:
180
+ """
181
+ Create a bed file from fasta file using replace logic.
182
+
183
+ Keyword arguments:
184
+ input_file: fasta file for desired bed file
185
+ """
186
+
187
+ def __init__(self, input_file: str) -> object:
188
+ self.input_file = input_file
189
+
190
+ self.get_bed()
191
+
192
+ def get_bed(self) -> None:
193
+ with open(f"{self.input_file}", "r") as repeat_eves, open(
194
+ f"{self.input_file}.bed", "w"
195
+ ) as repeat_eves_bed_out:
196
+ repeat_eves_lines = repeat_eves.readlines()
197
+ for line in repeat_eves_lines:
198
+ if ">" in line:
199
+ line_name = line.replace(">", "")
200
+ line_name = re.sub(":.*", "", line_name).rstrip("\n")
201
+ line_start = re.sub(".*:", "", line)
202
+ line_start = re.sub("-.*", "", line_start).rstrip("\n")
203
+ line_end = re.sub(".*-", "", line).rstrip("\n")
204
+ repeat_eves_bed_out.write(
205
+ f"{line_name}\t{line_start}\t{line_end}\n"
206
+ )
@@ -0,0 +1,64 @@
1
+ from Bio import SeqIO
2
+
3
+
4
+ class RemoveShortSequences:
5
+ """
6
+ Remove sequences bellow the cutoff threshold.
7
+
8
+ Keyword arguments:
9
+ input_file: input fasta file
10
+ cutoff: cutoff length, parsed by -ln
11
+ """
12
+
13
+ def __init__(self, input_file: str, cutoff: int) -> object:
14
+ self.input_file = input_file
15
+ self.cutoff = cutoff
16
+
17
+ self.cut_seq()
18
+
19
+ def cut_seq(self) -> None:
20
+ new_sequences = []
21
+ input_handle = open(self.input_file, "r")
22
+ output_handle = open(self.input_file + ".fmt", "w")
23
+ for record in SeqIO.parse(input_handle, "fasta"):
24
+ if len(record.seq) >= int(self.cutoff):
25
+ new_sequences.append(record)
26
+ SeqIO.write(new_sequences, output_handle, "fasta")
27
+
28
+
29
+ class MaskClean:
30
+ """
31
+ Remove sequences of EE on regions with a certain % of soft masked bases.
32
+
33
+ Keyword arguments:
34
+ input_file: fasta file, with putative EEs
35
+ m_per: treshold masked percentage value, parsed with -mp argument
36
+ """
37
+
38
+ def __init__(self, input_file: str, m_per: int) -> object:
39
+ self.input_file = input_file
40
+ self.m_per = m_per
41
+
42
+ self.mask_clean()
43
+
44
+ def mask_clean(self) -> None:
45
+ sequences = {}
46
+ for seq_record in SeqIO.parse(self.input_file, "fasta"):
47
+ sequence = str(seq_record.seq)
48
+ sequence_id = str(seq_record.id)
49
+ if (
50
+ float(
51
+ sequence.count("a")
52
+ + sequence.count("t")
53
+ + sequence.count("c")
54
+ + sequence.count("g")
55
+ + sequence.count("n")
56
+ + sequence.count("N")
57
+ )
58
+ / float(len(sequence))
59
+ ) * 100 <= float(self.m_per):
60
+ if sequence_id not in sequences:
61
+ sequences[sequence_id] = sequence
62
+ with open(self.input_file + ".cl", "w+") as output_file:
63
+ for sequence_id, sequence in sequences.items():
64
+ output_file.write(f">{sequence_id}\n{sequence}\n")
@@ -0,0 +1,30 @@
1
+ import pandas as pd
2
+
3
+
4
+ class CompareResults:
5
+ """
6
+ This function compares 2 blast results, for queries with same ID, only the one
7
+ with the major bitscore is keept. In a final step only queries with tag EE are maintained
8
+
9
+ Keyword arguments:
10
+ vir_result: filtred blast against ee database
11
+ host_result: filtred blast against filter database
12
+ """
13
+
14
+ def __init__(self, vir_result: str, host_result: str) -> object:
15
+ self.vir_result = vir_result
16
+ self.host_result = host_result
17
+
18
+ self.compare_results()
19
+
20
+ def compare_results(self) -> None:
21
+ df_vir = pd.read_csv(self.vir_result, sep="\t")
22
+ df_vir["qseqid"] = df_vir["bed_name"]
23
+ df_host = pd.read_csv(self.host_result, sep="\t")
24
+ df_hybrid = pd.concat([df_vir, df_host], ignore_index=True)
25
+ df_hybrid = df_hybrid.sort_values(by=["qseqid", "bitscore"], ascending=False)
26
+ df_hybrid.to_csv(self.host_result + ".concat", sep="\t", index=False)
27
+ df_nr = df_hybrid.drop_duplicates(subset=["qseqid"])
28
+ df_nr.to_csv(self.host_result + ".concat.nr", sep="\t", index=False)
29
+ df_nr_vir = df_nr[df_nr.tag == "EE"]
30
+ df_nr_vir.to_csv(self.host_result + ".concat.nr", sep="\t", index=False)
@@ -0,0 +1,135 @@
1
+ import pandas as pd
2
+ import csv
3
+ import os
4
+ import re
5
+ import glob
6
+ import shutil
7
+
8
+
9
+ class FilterTable:
10
+ """
11
+ Receives a blastx result and filter based on query ID and ranges of qstart and qend.
12
+
13
+ Keyword arguments:
14
+ blast_result: input blastx result
15
+ rangejunction: range for filter redundant hits
16
+ tag: HOST or EE, tells which blastx is
17
+ out_dir: output directory, parsed by -od
18
+ """
19
+
20
+ def __init__(
21
+ self, blast_result: str, rangejunction: int, tag: str, out_dir: str
22
+ ) -> object:
23
+ self.blast_result = blast_result
24
+ self.rangejunction = rangejunction
25
+ self.tag = tag
26
+ self.out_dir = out_dir
27
+
28
+ self.filter_blast()
29
+
30
+ def filter_blast(self) -> None:
31
+ header_outfmt6 = [
32
+ "qseqid",
33
+ "sseqid",
34
+ "pident",
35
+ "length",
36
+ "mismatch",
37
+ "gapopen",
38
+ "qstart",
39
+ "qend",
40
+ "sstart",
41
+ "send",
42
+ "evalue",
43
+ "bitscore",
44
+ ] # creates a blast header output in format = 6
45
+ df = pd.read_csv(
46
+ self.blast_result, sep="\t", header=None, names=header_outfmt6
47
+ ).sort_values(by="bitscore", ascending=False)
48
+ df["sense"] = ""
49
+ df["bed_name"] = ""
50
+ df["tag"] = ""
51
+ df["new_qstart"] = df["qstart"]
52
+ df["new_qend"] = df["qend"]
53
+ df.to_csv(self.blast_result + ".csv", sep="\t")
54
+ chunks = df = pd.read_csv(
55
+ f"{self.blast_result}.csv", sep="\t", chunksize=200000
56
+ )
57
+ count = 0
58
+ tmp_path = f"{self.out_dir}/tmp/"
59
+ if os.path.exists(tmp_path) == False:
60
+ os.mkdir(tmp_path)
61
+ for df in chunks:
62
+ df["sense"] = df["sense"].astype(object)
63
+ df.loc[
64
+ df["qstart"].astype(int) > df["qend"].values.astype(int), "sense"
65
+ ] = "neg"
66
+ df.loc[
67
+ df["qend"].values.astype(int) > df["qstart"].astype(int), "sense"
68
+ ] = "pos"
69
+ df.loc[df["sense"] == "neg", "new_qstart"] = df["qend"]
70
+ df.loc[df["sense"] == "neg", "new_qend"] = df["qstart"]
71
+ df.loc[df["sense"] == "neg", "qstart"] = df["new_qstart"]
72
+ df.loc[df["sense"] == "neg", "qend"] = df["new_qend"]
73
+ df.drop(columns=["new_qstart", "new_qend"], inplace=True)
74
+ if self.tag == "EE":
75
+ df["tag"] = "EE"
76
+ df["bed_name"] = df.apply(
77
+ lambda x: "%s:%s-%s" % (x["qseqid"], x["qstart"], x["qend"]), axis=1
78
+ )
79
+ else:
80
+ df["tag"] = "HOST"
81
+ df["bed_name"] = df["qseqid"]
82
+ pd.options.display.float_format = "{:,.2f}".format
83
+ df["evalue"] = pd.to_numeric(df["evalue"], downcast="float")
84
+ df = df[df.length >= 33]
85
+ header = [
86
+ "qseqid",
87
+ "sseqid",
88
+ "pident",
89
+ "length",
90
+ "mismatch",
91
+ "gapopen",
92
+ "qstart",
93
+ "qend",
94
+ "sstart",
95
+ "send",
96
+ "evalue",
97
+ "bitscore",
98
+ "sense",
99
+ "bed_name",
100
+ "tag",
101
+ ]
102
+ df = df[header]
103
+ with open(f"{tmp_path}chunk.{count}.tsv", "w") as chunk_writer:
104
+ df.to_csv(chunk_writer, sep="\t", index=False)
105
+ count += 1
106
+ all_chunks = glob.glob(f"{tmp_path}/*.tsv")
107
+ final_filtred_file = pd.DataFrame()
108
+ chunks_list = []
109
+ for chunk in all_chunks:
110
+ df = pd.read_csv(chunk, sep="\t")
111
+ chunks_list.append(df)
112
+ final_filtred_file = pd.concat(chunks_list, ignore_index=True)
113
+ final_filtred_file["qstart_rng"] = final_filtred_file.qstart.floordiv(
114
+ self.rangejunction
115
+ )
116
+ final_filtred_file["qend_rng"] = final_filtred_file.qend.floordiv(
117
+ self.rangejunction
118
+ )
119
+ final_filtred_file = (
120
+ final_filtred_file.drop_duplicates(subset=["qseqid", "qstart_rng", "sense"])
121
+ .drop_duplicates(subset=["qseqid", "qstart_rng", "sense"])
122
+ .sort_values(by=["qseqid"])
123
+ )
124
+ final_filtred_file.to_csv(
125
+ f"{self.blast_result}.filtred", sep="\t", index=False, columns=header
126
+ )
127
+ if self.tag == "EE":
128
+ final_filtred_file.to_csv(
129
+ f"{self.blast_result}.filtred.bed",
130
+ header=False,
131
+ sep="\t",
132
+ index=False,
133
+ columns=["qseqid", "qstart", "qend"],
134
+ )
135
+ shutil.rmtree(tmp_path, ignore_errors=True)
@@ -0,0 +1,21 @@
1
+ from Bio import SeqIO
2
+
3
+
4
+ class GetLength:
5
+ """
6
+ Creates a length file with the module SeqIO.
7
+
8
+ Keywords arguments:
9
+ input_file: formated genome, generated by cut_seq function
10
+ """
11
+
12
+ def __init__(self, input_file: str) -> object:
13
+ self.input_file = input_file
14
+
15
+ self.get_length()
16
+
17
+ def get_length(self) -> None:
18
+ with open(f"{self.input_file}.rn.fmt.lenght", "w") as output_length:
19
+ length_list = []
20
+ for seq_record in SeqIO.parse(self.input_file, "fasta"):
21
+ output_length.write(f"{seq_record.id}\t{str(len(seq_record))}\n")