FunVIP 0.3.20__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ Metadata-Version: 2.1
2
+ Name: FunVIP
3
+ Version: 0.3.20
4
+ Summary: Fungal Validation & Identification Pipeline
5
+ Author-email: Changwan Seo <wan101010@snu.ac.kr>
6
+ License: GPL-3.0
7
+ Project-URL: Homepage, https://github.com/Changwanseo/FunVIP
8
+ Requires-Python: <3.13,>3.8
9
+ License-File: LICENSE
10
+ Requires-Dist: biopython ==1.78
11
+ Requires-Dist: ete3 ==3.1.3
12
+ Requires-Dist: Cython
13
+ Requires-Dist: dendropy
14
+ Requires-Dist: GenMine >=1.0.13
15
+ Requires-Dist: lxml
16
+ Requires-Dist: matplotlib
17
+ Requires-Dist: numpy
18
+ Requires-Dist: openpyxl ==3.0.9
19
+ Requires-Dist: pandas ==1.4.2
20
+ Requires-Dist: plotly ==5.9.0
21
+ Requires-Dist: psutil
22
+ Requires-Dist: pyyaml
23
+ Requires-Dist: sip >=4.19.4
24
+ Requires-Dist: scikit-learn
25
+ Requires-Dist: scipy
26
+ Requires-Dist: tabulate
27
+ Requires-Dist: unidecode ==1.2.0
28
+ Requires-Dist: xlrd ==2.0.1
29
+ Requires-Dist: xlsxwriter
30
+ Requires-Dist: xmltodict ==0.12.0
31
+ Requires-Dist: PyQt5 >=5.9.2 ; sys_platform != "darwin"
32
+
@@ -0,0 +1,36 @@
1
+ data/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ external/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ external/BLAST_Windows/bin/cleanup-blastdb-volumes.py,sha256=RhlY30loTnSWp_xGc43GPyv1My9QD2NaxWWqy_fwHEk,6243
4
+ src/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ src/align.py,sha256=HVHwur7q6JS5lIvGCRCSUe7fzaCwvqAqcpA9SvMYCmQ,4100
6
+ src/cluster.py,sha256=ZNbVJSzRAeDhW9K76fiLT0O0giTqdSGBrC0KhZeAqHM,19438
7
+ src/command.py,sha256=nAFIB4IAGVKqGkuOGYx98tIOM4T80lyDzfNI3Swxx1U,13840
8
+ src/concatenate.py,sha256=1GA_akqcEKXrsqC0nkNFmrRo49hWH5tcngGFcRLawis,13871
9
+ src/dataset.py,sha256=sbxrmZBQZ5Zf8sCwowtJRFKO4p_p_CDXCTpdIs-M_h4,31797
10
+ src/ext.py,sha256=CQ_8LKjGgZIq5p_tWzmSxgw1NmB0ULMVAlLJXpZSMcw,14722
11
+ src/hasher.py,sha256=6GV8b_sY8nP5_7diV1UoFMDEjMpINpdCynw9cS_-ni8,2810
12
+ src/initialize.py,sha256=13ci7eudMN0JjPZ4u-ChUYfKQ7ziyjrgX4VmbA0uUtc,11522
13
+ src/logger.py,sha256=ssUoj-MOK1E_M3fYalfVY7vdHxqteLw7sGxYbXI6uVg,2089
14
+ src/logics.py,sha256=o6_buuuy9qCeqj5sWR56NVAWRI14i5lOd6ui6yprm9U,2785
15
+ src/modeltest.py,sha256=e6tIuALrziVbqWCErJBNhKbvxv9ELZ3zTliTjI_oF34,16418
16
+ src/ncbi.py,sha256=Ed14A519oalurEVCckaoFNV6dIqHZU55gPllOtJgHVo,5072
17
+ src/opt_generator.py,sha256=Ts1IIMrd3xV9lKwtgGp-StzVU37NKu_xu_njRTKQTss,3088
18
+ src/patch.py,sha256=Mt5OzPrT-voxOVm5m9sUUuoVZQAY2zjPmTvzFBgYiUY,7674
19
+ src/reporter.py,sha256=MO9Q8wXpTksw4hI7xGDAntMX7vu3IXz92KLPrvz7lSg,40744
20
+ src/save.py,sha256=k40XRmawNFJH7w2GK44WubnT3TS7tE9UmpmKqyVwjbk,5678
21
+ src/search.py,sha256=KWy7JPlfsvJoD-JbPyTZpaeA-_J4AOiJKX-hTuMv2cI,15087
22
+ src/tool.py,sha256=4rAYeEmHIxgztDgCrqfRXFumactCcxrHe0HTmDd1L7w,9781
23
+ src/tree.py,sha256=6RqcfrVM3_EITzMACsKzo62epwQpGtFeF6XiLSa3bsc,9266
24
+ src/tree_interpretation.py,sha256=DA7jxbayCWm6AbDEKo-KNbnExjF1q8FxBcCp9kF4quI,54919
25
+ src/tree_interpretation_pipe.py,sha256=IypkGyYW6pnJx5grtTGvMpaSzn0K9R_QJd7XQXBxoJ4,26991
26
+ src/trim.py,sha256=XAKywgv_d4pHpqs7V6_xmMRG2M_RxNHHUbGRHigX2Ys,4238
27
+ src/validate_input.py,sha256=8DUkCn0_WfB4BJBkUF2BnnsZenzpoOgOLSnURzxzfTQ,31610
28
+ src/validate_option.py,sha256=mOfF61uCRGw9kRD56prvjQ9UERYcDK1_20bCbR1NYhY,60353
29
+ src/validation.py,sha256=sUuzvhiq3FpQbU6BIVwJsLgiCBlpYnMBZXZuj_zN5R4,1040
30
+ src/version.py,sha256=TRisvZ5lVreWU87DPgDgMB4tEIefjWZyJOkCbTX2d-Q,12275
31
+ FunVIP-0.3.20.dist-info/LICENSE,sha256=IwGE9guuL-ryRPEKi6wFPI_zOhg7zDZbTYuHbSt_SAk,35823
32
+ FunVIP-0.3.20.dist-info/METADATA,sha256=8qfvnXHDYfhk9a8ABOIM5HI89bPlQA7aAWVcg_b2uyo,901
33
+ FunVIP-0.3.20.dist-info/WHEEL,sha256=MxOBbn7YEn8Le9UJ_3pkKUGGC2zJe2Y86VEFkJgyT7E,115
34
+ FunVIP-0.3.20.dist-info/entry_points.txt,sha256=32U6s3fPYXUVP4xRuBkownB-gzt-QAUxAEVnD8V4F9E,69
35
+ FunVIP-0.3.20.dist-info/top_level.txt,sha256=Bc9RcY5o1hc_9FJtBHbyryHn86AjoeyB4Cuw-YHOQgg,18
36
+ FunVIP-0.3.20.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: bdist_wheel (0.43.0) + setuptools-ext (0.9)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ FunID = funvip.main:main
3
+ FunVIP = funvip.main:main
@@ -0,0 +1,3 @@
1
+ data
2
+ external
3
+ src
data/__init__.py ADDED
File without changes
@@ -0,0 +1,162 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ # $Id: cleanup-blastdb-volumes.py 590894 2019-08-07 14:59:53Z camacho $
4
+ # ===========================================================================
5
+ #
6
+ # PUBLIC DOMAIN NOTICE
7
+ # National Center for Biotechnology Information
8
+ #
9
+ # This software/database is a "United States Government Work" under the
10
+ # terms of the United States Copyright Act. It was written as part of
11
+ # the author's official duties as a United States Government employee and
12
+ # thus cannot be copyrighted. This software/database is freely available
13
+ # to the public for use. The National Library of Medicine and the U.S.
14
+ # Government have not placed any restriction on its use or reproduction.
15
+ #
16
+ # Although all reasonable efforts have been taken to ensure the accuracy
17
+ # and reliability of the software and data, the NLM and the U.S.
18
+ # Government do not and cannot warrant the performance or results that
19
+ # may be obtained by using this software or data. The NLM and the U.S.
20
+ # Government disclaim all warranties, express or implied, including
21
+ # warranties of performance, merchantability or fitness for any particular
22
+ # purpose.
23
+ #
24
+ # Please cite the author in any work or product based on this material.
25
+ #
26
+ # ===========================================================================
27
+ #
28
+ # Author: Christiam Camacho
29
+ #
30
+ # File Description:
31
+ # Script to remove needless BLAST database files.
32
+ #
33
+ # ===========================================================================
34
+ """
35
+ import argparse, os, configparser
36
+ import unittest, tempfile
37
+ from pathlib import Path
38
+ from glob import glob
39
+
40
+ VERSION = '1.0'
41
+ DESC = r""" Remove needless BLAST database volumes. """
42
+
43
+
44
+ class Tester(unittest.TestCase):
45
+ """ Testing class for this script. """
46
+
47
+ def test_blastdb_config_invalid(self):
48
+ rv = get_blastdb_from_ncbi_config("/dev/null")
49
+ self.assertIsNone(rv)
50
+
51
+ def test_blastdb_config(self):
52
+ config = configparser.ConfigParser()
53
+ expected = "/blast/db/blast"
54
+ config['BLAST'] = {'BLASTDB': expected}
55
+ tf = tempfile.NamedTemporaryFile(mode="wt")
56
+ config.write(tf)
57
+ tf.flush()
58
+ rv = get_blastdb_from_ncbi_config(tf.name)
59
+ self.assertEqual(expected, rv)
60
+
61
+ def test_blastdb_finder(self):
62
+ tal = tempfile.NamedTemporaryFile(suffix=".pin")
63
+ dbname = find_blastdb(tal.name[:-4], True)
64
+ self.assertEqual(dbname, tal.name[:-4])
65
+
66
+
67
+ def find_blastdb(name: str, is_prot: bool) -> str:
68
+ """ Returns full path to BLAST database or None. """
69
+ alias_file = "{}.{}al".format(name, "p" if is_prot else "n")
70
+ index_file = "{}.{}in".format(name, "p" if is_prot else "n")
71
+ if os.path.exists(alias_file) or os.path.exists(index_file):
72
+ return name
73
+
74
+ if "BLASTDB" in os.environ:
75
+ alf = os.path.join(os.environ["BLASTDB"], alias_file)
76
+ idxf = os.path.join(os.environ["BLASTDB"], index_file)
77
+ if os.path.exists(alf) or os.path.exists(idxf):
78
+ return os.path.join(os.environ["BLASTDB"], name)
79
+
80
+ paths = [ os.getcwd(), str(Path.home()) ]
81
+ if "NCBI" in os.environ:
82
+ paths.append(os.path.join(os.environ["NCBI"]))
83
+
84
+ for path in paths:
85
+ for fname in [ ".ncbirc", "ncbi.ini" ]:
86
+ ncbirc = os.path.join(path, fname)
87
+ if os.path.exists(ncbirc):
88
+ blastdb = get_blastdb_from_ncbi_config(ncbirc)
89
+ if blastdb is not None:
90
+ alf = os.path.join(blastdb, alias_file)
91
+ idxf = os.path.join(blastdb, index_file)
92
+ if os.path.exists(alf) or os.path.exists(idxf):
93
+ return os.path.join(blastdb, name)
94
+
95
+
96
+ def get_blastdb_from_ncbi_config(config_file: str) -> str:
97
+ """ Return the BLASTDB setting from the NCBI configuration file or None. """
98
+ config = configparser.ConfigParser()
99
+ config.read(config_file)
100
+ if 'BLAST' in config and 'BLASTDB' in config['BLAST']:
101
+ return config['BLAST']['BLASTDB']
102
+
103
+
104
+ def main():
105
+ """ Entry point into this program. """
106
+ parser = create_arg_parser()
107
+ args = parser.parse_args()
108
+
109
+ ext = args.dbtype[0]
110
+ db = find_blastdb(args.db, ext == 'p')
111
+ if db == None:
112
+ print("Cannot find {} {} BLAST database".
113
+ format("protein" if ext == 'p' else "nucleotide", args.db),
114
+ file=sys.stderr)
115
+ return 1
116
+
117
+ alias_file = "{}.{}al".format(db, ext)
118
+ if not os.path.exists(alias_file):
119
+ return 1
120
+
121
+ with open(alias_file, "rt") as al:
122
+ for line in al:
123
+ if not line.startswith("DBLIST"):
124
+ continue
125
+ vols = list(map(lambda x: x.replace('"', ''), line.split()[1:]))
126
+ for existing_vols in sorted(glob("{}.*.{}in".format(db, ext))):
127
+ vol_name = os.path.basename(existing_vols)[:-4]
128
+ if vol_name in vols:
129
+ continue
130
+ if args.dry_run:
131
+ print("Will remove extra volume {}".format(existing_vols[:-4]))
132
+ to_rm = glob("{}??".format(existing_vols[:-2]))
133
+ to_rm += glob("{}.tar.gz.md5".format(existing_vols[:-4]))
134
+ for f in to_rm:
135
+ if not args.dry_run:
136
+ os.unlink(f)
137
+ print("Removed {}".format(f))
138
+ elif args.verbose > 0:
139
+ print("Will remove {}".format(f))
140
+
141
+ return 0
142
+
143
+
144
+ def create_arg_parser():
145
+ """ Create the command line options parser object for this script. """
146
+ parser = argparse.ArgumentParser(description=DESC)
147
+ parser.add_argument("-db", required=True, help="BLAST database name")
148
+ parser.add_argument("-dbtype", help="Molecule type", required=True,
149
+ choices=["prot", "nucl"])
150
+ parser.add_argument("-dry-run", action='store_true',
151
+ help="Do not delete any files, just list them")
152
+ parser.add_argument('-version', action='version',
153
+ version='%(prog)s ' + VERSION)
154
+ parser.add_argument("-verbose", action="count", default=0,
155
+ help="Increase output verbosity")
156
+ return parser
157
+
158
+
159
+ if __name__ == "__main__":
160
+ import sys
161
+ sys.exit(main())
162
+
external/__init__.py ADDED
File without changes
src/__init__.py ADDED
File without changes
src/align.py ADDED
@@ -0,0 +1,124 @@
1
+ from funvip.src import ext
2
+ from funvip.src.opt_generator import opt_generator
3
+ import logging
4
+ from Bio import SeqIO
5
+ import multiprocessing as mp
6
+ import math
7
+
8
+
9
+ # Each module to be run in alignment multiprocessing
10
+ def module_alignment(
11
+ in_fasta,
12
+ out_fasta,
13
+ path,
14
+ thread,
15
+ mafft_algorithm,
16
+ adjustdirection,
17
+ op,
18
+ ep,
19
+ ):
20
+ # Running MAFFT
21
+ ext.MAFFT(
22
+ fasta=in_fasta,
23
+ out=out_fasta,
24
+ path=path,
25
+ thread=thread,
26
+ algorithm=mafft_algorithm,
27
+ adjust=adjustdirection,
28
+ maxiterate=1000,
29
+ op=op,
30
+ ep=ep,
31
+ )
32
+
33
+ # to prevent reversed sequence making error
34
+ fasta_list = list(SeqIO.parse(out_fasta, "fasta"))
35
+ for seq in fasta_list:
36
+ if seq.description.startswith("_R_"):
37
+ seq.id = ""
38
+ seq.description = seq.description[3:]
39
+ SeqIO.write(fasta_list, out_fasta, "fasta")
40
+
41
+ # Fix unexpected spaces on mafft
42
+ with open(out_fasta, "r") as fr:
43
+ alignment = fr.read()
44
+
45
+ with open(out_fasta, "w") as fw:
46
+ fw.write(alignment.replace(" HS", "HS"))
47
+
48
+
49
+ # Alignment pipeline
50
+ def pipe_alignment(V, path, opt):
51
+ alignment_opt = opt_generator(V, opt, path, step="alignment")
52
+
53
+ # Multiprocessing start
54
+ # Thread optimizations
55
+ # Using O(N^2L^2) of MAFFT-G-ins-i
56
+ # Use multithreading optimizations in non-verbose mode
57
+ if opt.verbose < 3:
58
+ time_consumptions = []
59
+ ## Simulate time comsumption
60
+ # Parsing probable time_consumption
61
+ for aln_opt in alignment_opt:
62
+ before_aln_file = aln_opt[0]
63
+ before_seq_list = list(SeqIO.parse(before_aln_file, "fasta"))
64
+ N = len(before_seq_list)
65
+ L = max([len(seq.seq) for seq in before_seq_list])
66
+ time_consumptions.append(N * N * L * L)
67
+
68
+ time_consumptions = sorted(time_consumptions, reverse=True)
69
+ best_distribute = 1
70
+ best_thread = opt.thread
71
+ best_time = 99999999999999999999
72
+ # Optimize thread numbers
73
+ for distribute_num in range(opt.thread):
74
+ time_buffers = {}
75
+
76
+ # Threads to be distributed
77
+ each_thread_num = int(opt.thread / (distribute_num + 1))
78
+ # Efficiency multipler estimated from
79
+ # Rubio-Largo, A., Castelli, M., Vanneschi, L., & Vega-Rodríguez, M. A. (2018). A parallel multiobjective metaheuristic for multiple sequence alignment. Journal of Computational Biology, 25(9), 1009-1022.
80
+
81
+ efficiency_multiplier = 1.0258 * math.log(each_thread_num) + 1
82
+
83
+ for i in range(distribute_num + 1):
84
+ time_buffers[i + 1] = 0
85
+
86
+ for t in time_consumptions:
87
+ time_buffers[min(time_buffers, key=time_buffers.get)] += int(
88
+ t / efficiency_multiplier
89
+ )
90
+
91
+ total_time_consumption = max(time_buffers.values())
92
+ logging.debug(
93
+ f"Time buffers for {distribute_num+1} workers: {time_buffers}"
94
+ )
95
+ logging.debug(total_time_consumption)
96
+
97
+ if total_time_consumption < best_time:
98
+ best_thread = each_thread_num
99
+ best_distribute = distribute_num + 1
100
+ best_time = total_time_consumption
101
+
102
+ # logging.info(f"time_consumption_simulation: {time_buffers}")
103
+
104
+ logging.info(
105
+ f"Using {best_thread} threads for {best_distribute} workers in alignment by optimization"
106
+ )
107
+
108
+ # Update thread numbers by simulation
109
+ alignment_opt = opt_generator(
110
+ V, opt, path, step="alignment", thread=best_thread
111
+ )
112
+ p = mp.Pool(best_distribute)
113
+
114
+ alignment_result = p.starmap(module_alignment, alignment_opt)
115
+ p.close()
116
+ p.join()
117
+
118
+ else:
119
+ # non-multithreading mode for debugging
120
+ alignment_result = []
121
+ for option in alignment_opt:
122
+ alignment_result.append(module_alignment(*option))
123
+
124
+ return V, path, opt