krakenparser 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. krakenparser-0.6.0/KrakenParser.egg-info/PKG-INFO +306 -0
  2. krakenparser-0.6.0/KrakenParser.egg-info/SOURCES.txt +34 -0
  3. krakenparser-0.6.0/KrakenParser.egg-info/dependency_links.txt +1 -0
  4. krakenparser-0.6.0/KrakenParser.egg-info/entry_points.txt +2 -0
  5. krakenparser-0.6.0/KrakenParser.egg-info/requires.txt +8 -0
  6. krakenparser-0.6.0/KrakenParser.egg-info/top_level.txt +1 -0
  7. krakenparser-0.6.0/LICENSE +21 -0
  8. krakenparser-0.6.0/MANIFEST.in +12 -0
  9. krakenparser-0.6.0/PKG-INFO +306 -0
  10. krakenparser-0.6.0/README_PyPI.md +279 -0
  11. krakenparser-0.6.0/krakenparser/__init__.py +9 -0
  12. krakenparser-0.6.0/krakenparser/convert2csv.py +54 -0
  13. krakenparser-0.6.0/krakenparser/decombine.sh +123 -0
  14. krakenparser-0.6.0/krakenparser/decombine_viruses.sh +111 -0
  15. krakenparser-0.6.0/krakenparser/diversity.py +114 -0
  16. krakenparser-0.6.0/krakenparser/kpplot/__init__.py +1 -0
  17. krakenparser-0.6.0/krakenparser/kpplot/base.py +33 -0
  18. krakenparser-0.6.0/krakenparser/kpplot/clustermap.py +196 -0
  19. krakenparser-0.6.0/krakenparser/kpplot/stackedbar.py +208 -0
  20. krakenparser-0.6.0/krakenparser/kpplot/streamgraph.py +213 -0
  21. krakenparser-0.6.0/krakenparser/kraken2csv.sh +90 -0
  22. krakenparser-0.6.0/krakenparser/krakenparser.py +106 -0
  23. krakenparser-0.6.0/krakenparser/processing_script.py +70 -0
  24. krakenparser-0.6.0/krakenparser/relabund.py +71 -0
  25. krakenparser-0.6.0/krakenparser/run_kreport2mpa.sh +64 -0
  26. krakenparser-0.6.0/krakenparser/version.py +1 -0
  27. krakenparser-0.6.0/requirements.txt +8 -0
  28. krakenparser-0.6.0/setup.cfg +4 -0
  29. krakenparser-0.6.0/setup.py +102 -0
  30. krakenparser-0.6.0/tests/test_full_pipeline.py +49 -0
@@ -0,0 +1,306 @@
1
+ Metadata-Version: 2.2
2
+ Name: krakenparser
3
+ Version: 0.6.0
4
+ Summary: A collection of scripts designed to process Kraken2 reports and convert them into CSV format.
5
+ Home-page: https://github.com/PopovIILab/KrakenParser
6
+ Author: Ilia Popov
7
+ Author-email: iljapopov17@gmail.com
8
+ Requires-Python: >=3.6
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: pandas==2.2.3
12
+ Requires-Dist: matplotlib==3.10.0
13
+ Requires-Dist: numpy==2.2.0
14
+ Requires-Dist: pandas==2.2.3
15
+ Requires-Dist: plotly==5.24.1
16
+ Requires-Dist: seaborn==0.13.2
17
+ Requires-Dist: scipy==1.14.1
18
+ Requires-Dist: scikit-bio==0.6.3
19
+ Dynamic: author
20
+ Dynamic: author-email
21
+ Dynamic: description
22
+ Dynamic: description-content-type
23
+ Dynamic: home-page
24
+ Dynamic: requires-dist
25
+ Dynamic: requires-python
26
+ Dynamic: summary
27
+
28
+ # KrakenParser: Convert Kraken2 Reports to CSV
29
+
30
+ ## Overview
31
+ KrakenParser is a collection of scripts designed to process Kraken2 reports and convert them into CSV format. This pipeline extracts taxonomic abundance data at six levels:
32
+ - **Phylum**
33
+ - **Class**
34
+ - **Order**
35
+ - **Family**
36
+ - **Genus**
37
+ - **Species**
38
+
39
+ You can run the entire pipeline with **a single command**, or use the scripts **individually** depending on your needs.
40
+
41
+ 🔗 Please visit [KrakenParser wiki](https://github.com/PopovIILab/KrakenParser/wiki) page
42
+
43
+ ## Output example
44
+
45
+ ### Total abundance output
46
+
47
+ `counts_phylum.csv` parsed from 7 kraken2 reports of metagenomic samples using `KrakenParser`:
48
+
49
+ ```
50
+ Sample_id,Calditrichota,Caldisericota,Thermosulfidibacterota,Elusimicrobiota,Candidatus Fervidibacterota,Lentisphaerota,Kiritimatiellota,Vulcanimicrobiota,Thermodesulfobiota,Atribacterota,Dictyoglomota,Nitrospinota,Chrysiogenota,Coprothermobacterota,Aquificota,Thermotogota,Bdellovibrionota,Nitrospirota,Deferribacterota,Synergistota,Myxococcota,Acidobacteriota,Candidatus Bipolaricaulota,Candidatus Saccharibacteria,Candidatus Absconditabacteria,Fusobacteriota,Spirochaetota,Candidatus Omnitrophota,Chlamydiota,Verrucomicrobiota,Planctomycetota,Thermodesulfobacteriota,Campylobacterota,Candidatus Cloacimonadota,Fibrobacterota,Gemmatimonadota,Balneolota,Rhodothermota,Ignavibacteriota,Chlorobiota,Bacteroidota,Deinococcota,Thermomicrobiota,Armatimonadota,Chloroflexota,Cyanobacteriota,Mycoplasmatota,Actinomycetota,Bacillota,Pseudomonadota,Heterolobosea,Parabasalia,Fornicata,Evosea,Bacillariophyta,Cercozoa,Euglenozoa,Apicomplexa,Microsporidia,Basidiomycota,Ascomycota,Nanoarchaeota,Candidatus Micrarchaeota,Candidatus Thermoplasmatota,Candidatus Lokiarchaeota,Nitrososphaerota,Euryarchaeota,Thermoproteota,Hofneiviricota,Artverviricota,Nucleocytoviricota,Cossaviricota,Kitrinoviricota,Negarnaviricota,Lenarviricota,Pisuviricota,Peploviricota,Uroviricota
51
+ X1,0,0,0,0,0,0,0,0,1,1,1,1,2,3,4,5,7,8,9,17,23,25,5,13,22,47,54,1,6,27,31,128,151,2,6,13,1,3,7,44,14991,7,9,11,61,414,449,3551,55304,438645,0,0,0,0,0,0,1,22,0,4,15,0,0,0,0,0,3,191,0,0,1,88,0,0,0,161,0,1241
52
+ X2,1,4,14,20,5,12,15,6,8,15,2,15,109,68,182,97,79,196,70,272,331,149,36,77,35,562,1237,21,33,129,427,1044,543,8,98,25,16,45,11,1043,41374,160,28,161,1348,1196,2709,15864,431170,2747842,22,7,301,373,134,136,107,3239,54,1151,2905,0,0,3,5,6,7,410,0,0,0,736,0,3,11,26,1,1552
53
+ ...
54
+ X8,1,19,0,47,0,1,6,20,28,0,1,1,47,7,336,110,30,32,10,93,85,48,9,7,7,154,386,0,14,19,106,358,242,14,5,134,15,11,7,18,54057,106,10,24,212,340,1128,16220,567908,650264,95,4,193,402,314,300,187,4376,37,9796,8653,0,1,0,1,5,23,1778,1,1,0,1,1,4,66,30,4,1263
55
+ X9,0,3,2,16,7,1,23,12,10,9,1,2,134,40,390,289,29,372,27,81,150,90,9,88,32,287,881,14,33,60,319,1045,328,15,22,22,10,72,8,63,35301,127,15,48,412,935,2343,11500,380765,2613854,0,0,0,0,0,0,5,74,0,38,40,3,0,0,0,1,3,275,0,0,0,0,0,2,118,25,0,1675
56
+
57
+ ```
58
+
59
+ ### Relative abundance output
60
+
61
+ `ra_phylum.csv` calculated from 7 kraken2 reports of metagenomic samples using `KrakenParser`:
62
+
63
+ ```
64
+ Sample_id,taxon,rel_abund_perc
65
+ X1,Pseudomonadota,85.03558294577552
66
+ X1,Bacillota,10.72121619814011
67
+ X1,Other (<4.0%),4.243200856084384
68
+ X2,Pseudomonadota,84.28702055549813
69
+ X2,Bacillota,13.225663867469137
70
+ X2,Other (<4.0%),2.487315577032736
71
+ ...
72
+ X8,Pseudomonadota,49.25373021277305
73
+ X8,Bacillota,43.01574040339849
74
+ X8,Bacteroidota,4.094504530639667
75
+ X8,Other (<4.0%),3.6360248531887933
76
+ X9,Pseudomonadota,85.62839981589192
77
+ X9,Bacillota,12.473649123439218
78
+ X9,Other (<4.0%),1.8979510606688494
79
+ ```
80
+
81
+ ### α-diversity output
82
+
83
+ `alpha_div.csv` calculated from 7 kraken2 reports of metagenomic samples using `KrakenParser`:
84
+
85
+ ```
86
+ Sample,Shannon,Pielou,Chao1
87
+ X1,3.911345447107001,0.5269245043289149,2274.533185840708
88
+ X2,3.9944130792536563,0.4906424221265042,4155.0
89
+ ...
90
+ X8,3.442077115880119,0.42753293021330063,4177.251358695652
91
+ X9,4.033664950188261,0.5050385978575492,3492.16
92
+ ```
93
+
94
+ ### β-diversity output
95
+
96
+ `beta_div_bray.csv` calculated from 7 kraken2 reports of metagenomic samples using `KrakenParser`:
97
+
98
+ ```
99
+ ,X1,X2,...,X8,X9
100
+ X1,0.0,0.398,...,0.61,0.353
101
+ X2,0.398,0.0,...,0.723,0.388
102
+ ...
103
+ X8,0.61,0.723,...,0.0,0.665
104
+ X9,0.353,0.388,...,0.665,0.0
105
+ ```
106
+
107
+ `beta_div_jaccard.csv` calculated from 7 kraken2 reports of metagenomic samples using `KrakenParser`:
108
+
109
+ ```
110
+ ,X1,X2,...,X8,X9
111
+ X1,0.0,0.7073170731707317,...,0.8223938223938224,0.7232472324723247
112
+ X2,0.7073170731707317,0.0,...,0.835016835016835,0.7352941176470589
113
+ ...
114
+ X8,0.8223938223938224,0.835016835016835,...,0.0,0.8066914498141264
115
+ X9,0.7232472324723247,0.7352941176470589,...,0.8066914498141264,0.0
116
+ ```
117
+
118
+ ### Visualization examples gallery
119
+
120
+ |[Stacked Barplot](https://github.com/PopovIILab/KrakenParser/wiki/Stacked-Barplot-API)|[Streamgraph](https://github.com/PopovIILab/KrakenParser/wiki/Streamgraph-API)|
121
+ |-------|-------|
122
+ |![kpstbar](https://github.com/user-attachments/assets/916b0164-28be-4f49-9634-707408487b85)|![kpstream](https://github.com/user-attachments/assets/5c6d930c-e85f-4e2e-9dbf-8caefca49a76)|
123
+
124
+ [Stacked Barplot + Streamgraph](https://github.com/PopovIILab/KrakenParser/wiki/Combined-Stacked-Barplot-&-Streamgraph)|[Clustermap](https://github.com/PopovIILab/KrakenParser/wiki/Clustermap)|
125
+ |-------|-------|
126
+ |![combined_white](https://github.com/user-attachments/assets/58acea93-f079-46fd-ac4b-d2ac83098c59)|![kpclust](https://github.com/user-attachments/assets/98a4d540-7c43-4802-8f77-277a5637a7a1)|
127
+
128
+ ## Quick Start (Full Pipeline)
129
+ To run the full pipeline, use the following command:
130
+ ```bash
131
+ KrakenParser --complete -i data/kreports
132
+ #Having troubles? Run KrakenParser --complete -h
133
+ ```
134
+ This will:
135
+ 1. Convert Kraken2 reports to MPA format
136
+ 2. Combine MPA files into a single file
137
+ 3. Extract taxonomic levels into separate text files
138
+ 4. Process extracted text files
139
+ 5. Convert them into CSV format
140
+ 6. Calculate relative abundance
141
+ 7. Calculate α & β-diversities
142
+
143
+ ### **Input Requirements**
144
+ - The Kraken2 reports must be inside a **subdirectory** (e.g., `data/kreports`).
145
+ - The script automatically creates output directories and processes the data.
146
+
147
+ ## Installation
148
+
149
+ ```
150
+ pip install krakenparser
151
+ ```
152
+
153
+ ## Using Individual Modules
154
+ You can also run each step manually if needed.
155
+
156
+ ### **Step 1: Convert Kraken2 Reports to MPA Format**
157
+ ```bash
158
+ KrakenParser --kreport2mpa -i data/kreports -o data/mpa
159
+ #Having troubles? Run KrakenParser --kreport2mpa -h
160
+ ```
161
+ This script converts Kraken2 `.kreport` files into **MPA format** using KrakenTools.
162
+
163
+ ### **Step 2: Combine MPA Files**
164
+ ```bash
165
+ KrakenParser --combine_mpa -i data/mpa/* -o data/COMBINED.txt
166
+ #Having troubles? Run KrakenParser --combine_mpa -h
167
+ ```
168
+ This merges multiple MPA files into a single combined file.
169
+
170
+ ### **Step 3: Extract Taxonomic Levels**
171
+ ```bash
172
+ KrakenParser --deconstruct -i data/COMBINED.txt -o data/counts
173
+ #Having troubles? Run KrakenParser --deconstruct -h
174
+ ```
175
+
176
+ If user wants to inspect **Viruses** domain separately:
177
+ ```bash
178
+ KrakenParser --deconstruct_viruses -i data/COMBINED.txt -o data/counts_viruses
179
+ #Having troubles? Run KrakenParser --deconstruct_viruses -h
180
+ ```
181
+
182
+ This step extracts only species-level data (excluding human reads).
183
+
184
+ ### **Step 4: Process Extracted Taxonomic Data**
185
+ ```bash
186
+ KrakenParser --process -i data/COMBINED.txt -o data/counts/txt/counts_phylum.txt
187
+ #Having troubles? Run KrakenParser --process -h
188
+ ```
189
+
190
+ Repeat on other 5 taxonomical levels (class, order, family, genus, species) or wrap up `KrakenParser --process` to a loop!
191
+
192
+ This script cleans up taxonomic names (removes prefixes, replaces underscores with spaces).
193
+
194
+ ### **Step 5: Convert TXT to CSV**
195
+ ```bash
196
+ KrakenParser --txt2csv -i data/counts/txt/counts_phylum.txt -o data/counts/csv/counts_phylum.csv
197
+ #Having troubles? Run KrakenParser --txt2csv -h
198
+ ```
199
+ Repeat on other 5 taxonomical levels (class, order, family, genus, species) or wrap up `KrakenParser --txt2csv` to a loop!
200
+
201
+ This converts the processed text files into structured CSV format.
202
+
203
+ ### **Step 6: Calculate relative abundance**
204
+ ```bash
205
+ KrakenParser --relabund -i data/counts/csv/counts_phylum.csv -o data/counts/csv_relabund/counts_phylum.csv
206
+ #Having troubles? Run KrakenParser --relabund -h
207
+ ```
208
+ Repeat on other 5 taxonomical levels (class, order, family, genus, species) or wrap up `KrakenParser --relabund` to a loop!
209
+
210
+ This calculates relative abundance and saves as CSV format.
211
+
212
+ If user wants to group low abundant taxa in "Other" group:
213
+ ```bash
214
+ KrakenParser --relabund -i data/counts/csv/counts_phylum.csv -o data/counts/csv_relabund/counts_phylum.csv --other 3.5
215
+ #Having troubles? Run KrakenParser --relabund -h
216
+ ```
217
+
218
+ This will group all the taxa that have abundance <3.5 into "Other <3.5%" group. Other parameters are welcome!
219
+
220
+ ### **Step 7: Calculate α & β-diversities**
221
+ ```bash
222
+ KrakenParser --diversity -i data/counts/csv/counts_species.csv -o data/diversity
223
+ #Having troubles? Run KrakenParser --diversity -h
224
+ ```
225
+
226
+ This calculates α & β-diversities and saves them as CSV format to directory provided in the output.
227
+
228
+ If user wants to use another depth for β-diversity calculations:
229
+ ```bash
230
+ KrakenParser --diversity -i data/counts/csv/counts_species.csv -o data/diversity --depth 750
231
+ #Having troubles? Run KrakenParser --diversity -h
232
+ ```
233
+
234
+ Other parameters are welcome!
235
+
236
+ ## Arguments Breakdown
237
+ ### **KrakenParser** (Main Pipeline)
238
+ - Automates the entire workflow.
239
+ - Takes **one argument**: the path to Kraken2 reports (`data/kreports`).
240
+ - Runs all the scripts in sequence.
241
+
242
+ ### **--kreport2mpa** (Step 1)
243
+ - Converts Kraken2 reports to MPA format.
244
+ - Uses `KrakenTools/kreport2mpa.py`.
245
+
246
+ ### **--combine_mpa** (Step 2)
247
+ - Combines multiple MPA files into one.
248
+ - Uses `KrakenTools/combine_mpa.py`.
249
+
250
+ ### **--deconstruct** & **--deconstruct_viruses** (Step 3)
251
+ - Extracts **phylum, class, order, family, genus, species** into separate text files.
252
+ - Removes human-related reads (**--deconstruct** only).
253
+
254
+ ### **--process** (Step 4)
255
+ - Cleans and formats extracted taxonomic data.
256
+ - Removes prefixes (`s__`, `g__`, etc.), replaces underscores with spaces.
257
+
258
+ ### **--txt2csv** (Step 5)
259
+ - Converts cleaned text files to CSV.
260
+ - Transposes data so that sample names become rows.
261
+
262
+ ### **--relabund** (Step 6)
263
+ - Calculates relative abundance based on total abundance CSV.
264
+ - Optionally can group low abundant taxa.
265
+
266
+ ### **--diversity** (Step 7)
267
+ - Calculates α & β-diversities based on total species abundance CSV.
268
+ - Shannon, Pielou & Chao1 indices for α-diversity
269
+ - Bray-Curtis & Jaccard indices for β-diversity
270
+ - Uses 1000 depth for β-diversity as default (can be adjusted with -d)
271
+
272
+ ## Example Output Structure
273
+ After running the full pipeline, the output directory will look like this:
274
+ ```
275
+ data/
276
+ ├─ kreports/ # Input Kraken2 reports
277
+ ├─ mpa/ # Converted MPA files
278
+ ├─ COMBINED.txt # Merged MPA file
279
+ ├─ counts/
280
+ │ ├─ txt/ # Extracted taxonomic levels in TXT
281
+ │ │ ├─ counts_species.txt
282
+ │ │ ├─ counts_genus.txt
283
+ │ │ ├─ counts_family.txt
284
+ │ │ ├─ ...
285
+ │ └─ csv/ # Total abundance CSV output
286
+ │ │ ├─ counts_species.csv
287
+ │ │ ├─ counts_genus.csv
288
+ │ │ ├─ counts_family.csv
289
+ │ │ ├─ ...
290
+ ├─ rel_abund/ # Relative abundance CSV output
291
+ │ ├─ ra_species.csv
292
+ │ ├─ ra_genus.csv
293
+ │ ├─ ra_family.csv
294
+ │ ├─ ...
295
+ └─ diversity/
296
+ ├─ alpha_div.csv
297
+ ├─ beta_div_bray.csv
298
+ └─ beta_div_jaccard.csv
299
+ ```
300
+
301
+ ## Conclusion
302
+ KrakenParser provides a **simple and automated** way to convert Kraken2 reports into usable CSV files for downstream analysis. You can run the **full pipeline** with a single command or use **individual scripts** as needed.
303
+
304
+ For any issues or feature requests, feel free to open an issue on GitHub!
305
+
306
+ 🚀 Happy analyzing!
@@ -0,0 +1,34 @@
1
+ LICENSE
2
+ MANIFEST.in
3
+ README_PyPI.md
4
+ requirements.txt
5
+ setup.py
6
+ KrakenParser.egg-info/PKG-INFO
7
+ KrakenParser.egg-info/SOURCES.txt
8
+ KrakenParser.egg-info/dependency_links.txt
9
+ KrakenParser.egg-info/entry_points.txt
10
+ KrakenParser.egg-info/requires.txt
11
+ KrakenParser.egg-info/top_level.txt
12
+ krakenparser/__init__.py
13
+ krakenparser/convert2csv.py
14
+ krakenparser/decombine.sh
15
+ krakenparser/decombine_viruses.sh
16
+ krakenparser/diversity.py
17
+ krakenparser/kraken2csv.sh
18
+ krakenparser/krakenparser.py
19
+ krakenparser/processing_script.py
20
+ krakenparser/relabund.py
21
+ krakenparser/run_kreport2mpa.sh
22
+ krakenparser/version.py
23
+ krakenparser.egg-info/PKG-INFO
24
+ krakenparser.egg-info/SOURCES.txt
25
+ krakenparser.egg-info/dependency_links.txt
26
+ krakenparser.egg-info/entry_points.txt
27
+ krakenparser.egg-info/requires.txt
28
+ krakenparser.egg-info/top_level.txt
29
+ krakenparser/kpplot/__init__.py
30
+ krakenparser/kpplot/base.py
31
+ krakenparser/kpplot/clustermap.py
32
+ krakenparser/kpplot/stackedbar.py
33
+ krakenparser/kpplot/streamgraph.py
34
+ tests/test_full_pipeline.py
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ KrakenParser = krakenparser:krakenparser.main
@@ -0,0 +1,8 @@
1
+ pandas==2.2.3
2
+ matplotlib==3.10.0
3
+ numpy==2.2.0
4
+ pandas==2.2.3
5
+ plotly==5.24.1
6
+ seaborn==0.13.2
7
+ scipy==1.14.1
8
+ scikit-bio==0.6.3
@@ -0,0 +1 @@
1
+ krakenparser
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Ilia Popov
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,12 @@
1
+ include LICENSE
2
+ include README_PyPI.md
3
+ include requirements.txt
4
+ exclude demo_data/*
5
+ exclude imgs/*.png
6
+ exclude imgs/Layout/*
7
+ exclude __pycache__
8
+ exclude README.md
9
+ exclude .pypirc
10
+ exclude CITATION.cff
11
+ exclude TESTS
12
+ exclude miscellaneous