PPanGGOLiN 2.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. ppanggolin-2.2.2/.github/workflows/black_lint.yml +12 -0
  2. ppanggolin-2.2.2/.github/workflows/check_doc.yml +33 -0
  3. ppanggolin-2.2.2/.github/workflows/check_recipes.yml +46 -0
  4. ppanggolin-2.2.2/.github/workflows/main.yml +264 -0
  5. ppanggolin-2.2.2/.github/workflows/pypi.yml +93 -0
  6. ppanggolin-2.2.2/.readthedocs.yaml +35 -0
  7. ppanggolin-2.2.2/LICENSE.txt +519 -0
  8. ppanggolin-2.2.2/MANIFEST.in +6 -0
  9. ppanggolin-2.2.2/PKG-INFO +692 -0
  10. ppanggolin-2.2.2/PPanGGOLiN.egg-info/PKG-INFO +692 -0
  11. ppanggolin-2.2.2/PPanGGOLiN.egg-info/SOURCES.txt +130 -0
  12. ppanggolin-2.2.2/PPanGGOLiN.egg-info/dependency_links.txt +1 -0
  13. ppanggolin-2.2.2/PPanGGOLiN.egg-info/entry_points.txt +2 -0
  14. ppanggolin-2.2.2/PPanGGOLiN.egg-info/requires.txt +23 -0
  15. ppanggolin-2.2.2/PPanGGOLiN.egg-info/top_level.txt +2 -0
  16. ppanggolin-2.2.2/README.md +128 -0
  17. ppanggolin-2.2.2/VERSION +1 -0
  18. ppanggolin-2.2.2/ppanggolin/RGP/__init__.py +3 -0
  19. ppanggolin-2.2.2/ppanggolin/RGP/genomicIsland.py +479 -0
  20. ppanggolin-2.2.2/ppanggolin/RGP/rgp_cluster.py +903 -0
  21. ppanggolin-2.2.2/ppanggolin/RGP/spot.py +422 -0
  22. ppanggolin-2.2.2/ppanggolin/__init__.py +67 -0
  23. ppanggolin-2.2.2/ppanggolin/align/__init__.py +1 -0
  24. ppanggolin-2.2.2/ppanggolin/align/alignOnPang.py +973 -0
  25. ppanggolin-2.2.2/ppanggolin/annotate/__init__.py +1 -0
  26. ppanggolin-2.2.2/ppanggolin/annotate/annotate.py +1966 -0
  27. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/RF00001.cm +877 -0
  28. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/RF00177.cm +10617 -0
  29. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/RF01959.cm +10221 -0
  30. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/RF02540.cm +20826 -0
  31. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/RF02541.cm +20329 -0
  32. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_arch.cm +31924 -0
  33. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_arch.cm.i1f +0 -0
  34. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_arch.cm.i1i +0 -0
  35. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_arch.cm.i1m +0 -0
  36. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_arch.cm.i1p +0 -0
  37. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_bact.cm +31823 -0
  38. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_bact.cm.i1f +0 -0
  39. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_bact.cm.i1i +0 -0
  40. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_bact.cm.i1m +0 -0
  41. ppanggolin-2.2.2/ppanggolin/annotate/rRNA_DB/rRNA_bact.cm.i1p +0 -0
  42. ppanggolin-2.2.2/ppanggolin/annotate/synta.py +571 -0
  43. ppanggolin-2.2.2/ppanggolin/cluster/__init__.py +1 -0
  44. ppanggolin-2.2.2/ppanggolin/cluster/cluster.py +1009 -0
  45. ppanggolin-2.2.2/ppanggolin/context/__init__.py +1 -0
  46. ppanggolin-2.2.2/ppanggolin/context/searchGeneContext.py +957 -0
  47. ppanggolin-2.2.2/ppanggolin/edge.py +120 -0
  48. ppanggolin-2.2.2/ppanggolin/figures/__init__.py +1 -0
  49. ppanggolin-2.2.2/ppanggolin/figures/draw_spot.py +984 -0
  50. ppanggolin-2.2.2/ppanggolin/figures/drawing.py +195 -0
  51. ppanggolin-2.2.2/ppanggolin/figures/tile_plot.py +611 -0
  52. ppanggolin-2.2.2/ppanggolin/figures/ucurve.py +146 -0
  53. ppanggolin-2.2.2/ppanggolin/formats/__init__.py +7 -0
  54. ppanggolin-2.2.2/ppanggolin/formats/readBinaries.py +1799 -0
  55. ppanggolin-2.2.2/ppanggolin/formats/writeAnnotations.py +645 -0
  56. ppanggolin-2.2.2/ppanggolin/formats/writeBinaries.py +997 -0
  57. ppanggolin-2.2.2/ppanggolin/formats/writeFlatGenomes.py +986 -0
  58. ppanggolin-2.2.2/ppanggolin/formats/writeFlatMetadata.py +257 -0
  59. ppanggolin-2.2.2/ppanggolin/formats/writeFlatPangenome.py +1729 -0
  60. ppanggolin-2.2.2/ppanggolin/formats/writeMSA.py +589 -0
  61. ppanggolin-2.2.2/ppanggolin/formats/writeMetadata.py +506 -0
  62. ppanggolin-2.2.2/ppanggolin/formats/writeSequences.py +748 -0
  63. ppanggolin-2.2.2/ppanggolin/formats/write_proksee.py +474 -0
  64. ppanggolin-2.2.2/ppanggolin/geneFamily.py +492 -0
  65. ppanggolin-2.2.2/ppanggolin/genetic_codes.py +8945 -0
  66. ppanggolin-2.2.2/ppanggolin/genome.py +1210 -0
  67. ppanggolin-2.2.2/ppanggolin/graph/__init__.py +1 -0
  68. ppanggolin-2.2.2/ppanggolin/graph/makeGraph.py +215 -0
  69. ppanggolin-2.2.2/ppanggolin/info/__init__.py +1 -0
  70. ppanggolin-2.2.2/ppanggolin/info/info.py +185 -0
  71. ppanggolin-2.2.2/ppanggolin/main.py +293 -0
  72. ppanggolin-2.2.2/ppanggolin/meta/__init__.py +1 -0
  73. ppanggolin-2.2.2/ppanggolin/meta/meta.py +354 -0
  74. ppanggolin-2.2.2/ppanggolin/metadata.py +296 -0
  75. ppanggolin-2.2.2/ppanggolin/metrics/__init__.py +1 -0
  76. ppanggolin-2.2.2/ppanggolin/metrics/fluidity.py +139 -0
  77. ppanggolin-2.2.2/ppanggolin/metrics/metrics.py +220 -0
  78. ppanggolin-2.2.2/ppanggolin/mod/__init__.py +1 -0
  79. ppanggolin-2.2.2/ppanggolin/mod/module.py +293 -0
  80. ppanggolin-2.2.2/ppanggolin/nem/NEM/genmemo.c +54 -0
  81. ppanggolin-2.2.2/ppanggolin/nem/NEM/genmemo.h +49 -0
  82. ppanggolin-2.2.2/ppanggolin/nem/NEM/lib_io.c +686 -0
  83. ppanggolin-2.2.2/ppanggolin/nem/NEM/lib_io.h +107 -0
  84. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_alg.c +2973 -0
  85. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_alg.h +35 -0
  86. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_exe.c +1903 -0
  87. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_exe.h +39 -0
  88. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_hlp.c +643 -0
  89. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_hlp.h +18 -0
  90. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_mod.c +2408 -0
  91. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_mod.h +33 -0
  92. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_nei.c +143 -0
  93. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_nei.h +14 -0
  94. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_rnd.c +153 -0
  95. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_rnd.h +10 -0
  96. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_stats.pyx +17 -0
  97. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_typ.h +591 -0
  98. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_ver.c +28 -0
  99. ppanggolin-2.2.2/ppanggolin/nem/NEM/nem_ver.h +6 -0
  100. ppanggolin-2.2.2/ppanggolin/nem/__init__.py +0 -0
  101. ppanggolin-2.2.2/ppanggolin/nem/partition.py +1057 -0
  102. ppanggolin-2.2.2/ppanggolin/nem/rarefaction.py +927 -0
  103. ppanggolin-2.2.2/ppanggolin/pangenome.py +933 -0
  104. ppanggolin-2.2.2/ppanggolin/projection/__init__.py +1 -0
  105. ppanggolin-2.2.2/ppanggolin/projection/projection.py +1937 -0
  106. ppanggolin-2.2.2/ppanggolin/region.py +1199 -0
  107. ppanggolin-2.2.2/ppanggolin/utility/__init__.py +1 -0
  108. ppanggolin-2.2.2/ppanggolin/utility/utils.py +357 -0
  109. ppanggolin-2.2.2/ppanggolin/utils.py +1605 -0
  110. ppanggolin-2.2.2/ppanggolin/workflow/__init__.py +4 -0
  111. ppanggolin-2.2.2/ppanggolin/workflow/all.py +643 -0
  112. ppanggolin-2.2.2/ppanggolin/workflow/panModule.py +34 -0
  113. ppanggolin-2.2.2/ppanggolin/workflow/panRGP.py +35 -0
  114. ppanggolin-2.2.2/ppanggolin/workflow/workflow.py +34 -0
  115. ppanggolin-2.2.2/ppanggolin_env.yaml +21 -0
  116. ppanggolin-2.2.2/pyproject.toml +81 -0
  117. ppanggolin-2.2.2/setup.cfg +4 -0
  118. ppanggolin-2.2.2/setup.py +27 -0
  119. ppanggolin-2.2.2/tests/align/test_align.py +88 -0
  120. ppanggolin-2.2.2/tests/annotate/test_annotate.py +571 -0
  121. ppanggolin-2.2.2/tests/context/test_context.py +280 -0
  122. ppanggolin-2.2.2/tests/formats/test_writeFlatGenomes.py +56 -0
  123. ppanggolin-2.2.2/tests/pytest.ini +4 -0
  124. ppanggolin-2.2.2/tests/region/test_genomicIsland.py +87 -0
  125. ppanggolin-2.2.2/tests/region/test_rgp_cluster.py +236 -0
  126. ppanggolin-2.2.2/tests/test_edge.py +121 -0
  127. ppanggolin-2.2.2/tests/test_genefamily.py +362 -0
  128. ppanggolin-2.2.2/tests/test_genome.py +631 -0
  129. ppanggolin-2.2.2/tests/test_metadata.py +132 -0
  130. ppanggolin-2.2.2/tests/test_pangenome.py +908 -0
  131. ppanggolin-2.2.2/tests/test_region.py +1052 -0
  132. ppanggolin-2.2.2/tests/utils/test_utilities.py +199 -0
@@ -0,0 +1,12 @@
1
+ name: Lint
2
+
3
+ on: [pull_request]
4
+
5
+ jobs:
6
+ lint:
7
+ runs-on: ubuntu-latest
8
+ steps:
9
+ - uses: actions/checkout@v4
10
+ - uses: psf/black@stable
11
+ with:
12
+ use_pyproject: true
@@ -0,0 +1,33 @@
1
+ name: Check documentation
2
+
3
+ on:
4
+ push:
5
+ paths:
6
+ - 'docs/**'
7
+ - '.readthedocs.yaml'
8
+ - '.github/workflows/check_doc.yml'
9
+
10
+ jobs:
11
+ build:
12
+
13
+ runs-on: ubuntu-latest
14
+
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: actions/setup-python@v5
18
+ with:
19
+ python-version: '3.9'
20
+ - name: install ppanggolin with python deps and doc deps
21
+ run: pip install .[doc]
22
+
23
+ - name: Complete workflow
24
+ shell: bash -l {0}
25
+ run: |
26
+ cd docs/
27
+ sphinx-build -b html . build/
28
+ # Great extra actions to compose with:
29
+ # Create an artifact of the html output.
30
+ - uses: actions/upload-artifact@v4
31
+ with:
32
+ name: DocumentationHTML
33
+ path: docs/build/
@@ -0,0 +1,46 @@
1
+ name: conda recipe
2
+
3
+ # Controls when the workflow will run
4
+ on:
5
+ # Triggers the workflow on schedule but only for the default branch (which is master)
6
+ schedule:
7
+ - cron: '0 7 * * 1-5'
8
+
9
+ # Allows you to run this workflow manually from the Actions tab
10
+ workflow_dispatch:
11
+
12
+ # A workflow run is made up of one or more jobs that can run sequentially or in parallel
13
+ jobs:
14
+ # This workflow contains a single job called "check_recipe"
15
+ check_recipe:
16
+ name: test bioconda recipes on ${{ matrix.os }} with python ${{ matrix.python-version }}
17
+ runs-on: ${{ matrix.os }}
18
+ strategy:
19
+ matrix:
20
+ os: ['ubuntu-latest','macos-13']
21
+ python-version: ['3.9','3.10','3.11']
22
+ # Steps represent a sequence of tasks that will be executed as part of the job
23
+ steps:
24
+ # Checks-out your repository under $GITHUB_WORKSPACE, so your job can access it
25
+ - uses: actions/checkout@v4
26
+ # Setting up miniconda
27
+ - uses: conda-incubator/setup-miniconda@v3
28
+ with:
29
+ python-version: ${{ matrix.python-version }}
30
+ channels: bioconda,conda-forge,anaconda,defaults
31
+ activate-environment: test
32
+ - name: Set up test environment
33
+ shell: bash -l {0}
34
+ run: |
35
+ conda install -y ppanggolin
36
+ - name: check installation
37
+ shell: bash -l {0}
38
+ run: |
39
+ python --version
40
+ ppanggolin --version
41
+ - name: all workflow run
42
+ shell: bash -l {0}
43
+ run: |
44
+ cd testingDataset/
45
+ ppanggolin all --cpu 1 --anno genomes.gbff.list -o pango
46
+ ppanggolin info -p pango/pangenome.h5 --content --parameters --status
@@ -0,0 +1,264 @@
1
+ name: CI
2
+
3
+ on:
4
+ pull_request:
5
+ branches:
6
+ - '*'
7
+ paths:
8
+ # if any of this files or directory changed, trigger the CI
9
+ # The only case where it is not triggerd is when docs/ is modified
10
+ - 'tests/**'
11
+ - 'testingDataset/**'
12
+ - '.github/**'
13
+ - 'ppanggolin/**'
14
+ - 'MANIFEST.in'
15
+ - 'VERSION'
16
+ - 'ppanggolin_env.yaml'
17
+ - 'pyproject.toml'
18
+ - 'setup.py'
19
+ # Allows you to run this workflow manually from the Actions tab
20
+ workflow_dispatch:
21
+
22
+ env:
23
+ NUM_CPUS: 1
24
+
25
+ # A workflow run is made up of one or more jobs that can run sequentially or in parallel
26
+ jobs:
27
+ test:
28
+ name: test PPanGGOLiN on ${{ matrix.os }} with python ${{ matrix.python-version }}
29
+ # The type of runner that the job will run on
30
+ runs-on: ${{ matrix.os }}
31
+ strategy:
32
+ matrix:
33
+ os: ['ubuntu-latest', 'macos-latest']
34
+ python-version: ['3.9', '3.12']
35
+ steps:
36
+
37
+ # Get number of cpu available on the current runner
38
+ - name: Get core number on linux
39
+ if: matrix.os == 'ubuntu-latest'
40
+ run: |
41
+ nb_cpu_linux=`nproc`
42
+ echo "Number of cores avalaible on the current linux runner $nb_cpu_linux"
43
+ echo "NUM_CPUS=$nb_cpu_linux" >> "$GITHUB_ENV"
44
+
45
+ - name: Get core number on macos
46
+ if: matrix.os == 'macos-latest'
47
+ run: |
48
+ nb_cpu_macos=`sysctl -n hw.ncpu`
49
+ echo "Number of cores avalaible on the current macos runner $nb_cpu_macos"
50
+ echo "NUM_CPUS=$nb_cpu_macos" >> "$GITHUB_ENV"
51
+
52
+ # Checks-out your repository under $GITHUB_WORKSPACE, so your job can access it
53
+ - uses: actions/checkout@v4
54
+ # Install requirements with miniconda
55
+ - uses: conda-incubator/setup-miniconda@v3
56
+ with:
57
+ python-version: ${{ matrix.python-version }}
58
+ channels: conda-forge,bioconda,defaults
59
+ environment-file: ppanggolin_env.yaml
60
+ activate-environment: ppanggolin
61
+
62
+ - name: Install ppanggolin
63
+ shell: bash -l {0}
64
+ run: |
65
+ pip install .[test]
66
+ mmseqs version
67
+
68
+ # Check that it is installed and displays help without error
69
+ - name: Check that PPanGGOLiN is installed
70
+ shell: bash -l {0}
71
+ run: |
72
+ ppanggolin --version
73
+ ppanggolin --help
74
+
75
+ # Check that unit tests are all passing
76
+ - name: Unit tests
77
+ shell: bash -l {0}
78
+ run: pytest
79
+
80
+ # Test the complete workflow
81
+ - name: Complete workflow
82
+ shell: bash -l {0}
83
+ run: |
84
+ cd testingDataset
85
+ mkdir info_to_test
86
+ ppanggolin all --cpu $NUM_CPUS --fasta genomes.fasta.list --output mybasicpangenome
87
+ ppanggolin info --pangenome mybasicpangenome/pangenome.h5 --content --parameters --status > info_to_test/mybasicpangenome_info.yaml
88
+ cat info_to_test/mybasicpangenome_info.yaml
89
+ echo "$(grep 'mybasicpangenome/gene_families.tsv' expected_info_files/checksum.txt | cut -d' ' -f1) mybasicpangenome/gene_families.tsv" | shasum -a 256 -c - || { echo 'Checksum verification failed.' >&2; exit 1; }
90
+ shasum -a 256 mybasicpangenome/gene_families.tsv > info_to_test/checksum.txt
91
+ cd -
92
+ # test most options calls. If there is a change in the API somewhere that was not taken into account (whether in the options for the users, or the classes for the devs), this should fail, otherwise everything is probably good.
93
+ #--draw_hotspots option is problematic on macOS.
94
+ - name: Step by Step workflow with most options calls
95
+ shell: bash -l {0}
96
+ run: |
97
+ cd testingDataset
98
+ ppanggolin annotate --fasta genomes.fasta.list --output stepbystep --kingdom bacteria --cpu $NUM_CPUS
99
+ ppanggolin cluster -p stepbystep/pangenome.h5 --coverage 0.8 --identity 0.8 --cpu $NUM_CPUS
100
+ ppanggolin graph -p stepbystep/pangenome.h5 -r 10
101
+ ppanggolin partition --output stepbystep -f -p stepbystep/pangenome.h5 --cpu $NUM_CPUS -b 2.6 -ms 10 -fd -ck 500 -Kmm 3 12 -im 0.04 --draw_ICL
102
+ ppanggolin rarefaction --output stepbystep -f -p stepbystep/pangenome.h5 --depth 5 --min 1 --max 50 -ms 10 -fd -ck 30 -K 3 --soft_core 0.9 -se $RANDOM
103
+ ppanggolin draw -p stepbystep/pangenome.h5 --tile_plot --nocloud --soft_core 0.92 --ucurve --output stepbystep -f
104
+ ppanggolin rgp -p stepbystep/pangenome.h5 --persistent_penalty 2 --variable_gain 1 --min_score 3 --dup_margin 0.05
105
+ ppanggolin spot -p stepbystep/pangenome.h5 --output stepbystep --spot_graph --overlapping_match 2 --set_size 3 --exact_match_size 1 -f
106
+ ppanggolin draw -p stepbystep/pangenome.h5 --draw_spots -o stepbystep -f
107
+ ppanggolin module -p stepbystep/pangenome.h5 --transitive 4 --size 3 --jaccard 0.86 --dup_margin 0.05
108
+ ppanggolin write_pangenome -p stepbystep/pangenome.h5 --output stepbystep -f --soft_core 0.9 --dup_margin 0.06 --gexf --light_gexf --csv --Rtab --stats --partitions --compress --json --spots --regions --borders --families_tsv --cpu 1
109
+ ppanggolin write_genomes -p stepbystep/pangenome.h5 --output stepbystep -f --fasta genomes.fasta.list --gff --proksee --table
110
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --prot_families all --gene_families shell --regions all --fasta genomes.fasta.list
111
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --prot_families rgp --gene_families rgp --compress
112
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --prot_families softcore --gene_families softcore
113
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --prot_families module_0
114
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --genes core --proteins cloud
115
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --gene_families module_0 --genes module_0 --compress
116
+ ppanggolin fasta -p stepbystep/pangenome.h5 --output stepbystep -f --proteins cloud --cpu $NUM_CPUS --keep_tmp --compress
117
+
118
+ ppanggolin draw -p stepbystep/pangenome.h5 --draw_spots --spots all -o stepbystep -f
119
+ ppanggolin metrics -p stepbystep/pangenome.h5 --genome_fluidity --no_print_info --recompute_metrics --log metrics.log
120
+ ppanggolin info --pangenome stepbystep/pangenome.h5 > info_to_test/stepbystep_info.yaml
121
+ cat info_to_test/stepbystep_info.yaml
122
+ gzip -d stepbystep/gene_families.tsv.gz
123
+ echo "$(grep 'stepbystep/gene_families.tsv' expected_info_files/checksum.txt | cut -d' ' -f1) stepbystep/gene_families.tsv" | shasum -a 256 -c - || { echo 'Checksum verification failed.' >&2; exit 1; }
124
+ shasum -a 256 stepbystep/gene_families.tsv >> info_to_test/checksum.txt
125
+ cd -
126
+ - name: gbff parsing and MSA computing
127
+ shell: bash -l {0}
128
+ run: |
129
+ cd testingDataset
130
+ ppanggolin workflow --cpu $NUM_CPUS --anno genomes.gbff.list --output myannopang
131
+ ppanggolin msa --pangenome myannopang/pangenome.h5 --source dna --partition core -o myannopang/ -f --use_gene_id --phylo --single_copy --cpu $NUM_CPUS
132
+ ppanggolin info --pangenome myannopang/pangenome.h5 > info_to_test/myannopang_info.yaml
133
+ cat info_to_test/myannopang_info.yaml
134
+ echo "$(grep 'myannopang/gene_families.tsv' expected_info_files/checksum.txt | cut -d' ' -f1) myannopang/gene_families.tsv" | shasum -a 256 -c - || { echo 'Checksum verification failed.' >&2; exit 1; }
135
+ shasum -a 256 myannopang/gene_families.tsv >> info_to_test/checksum.txt
136
+ cd -
137
+ - name: clusters reading from external file
138
+ shell: bash -l {0}
139
+ run: |
140
+ cd testingDataset
141
+ cat myannopang/gene_families.tsv | cut -f1,2,4 > clusters.tsv
142
+ ppanggolin panrgp --anno genomes.gbff.list --cluster clusters.tsv --output readclusterpang --cpu $NUM_CPUS
143
+ ppanggolin annotate --anno genomes.gbff.list --output readclusters --cpu $NUM_CPUS
144
+ awk 'BEGIN{FS=OFS="\t"} {$1 = $1 OFS $1} 1' clusters.tsv > clusters_with_reprez.tsv;
145
+ ppanggolin cluster --clusters clusters_with_reprez.tsv -p readclusters/pangenome.h5 --cpu $NUM_CPUS
146
+ ppanggolin msa --pangenome readclusterpang/pangenome.h5 --partition persistent --phylo -o readclusterpang/msa/ -f --cpu $NUM_CPUS
147
+ echo "$(grep 'readclusterpang/gene_families.tsv' expected_info_files/checksum.txt | cut -d' ' -f1) readclusterpang/gene_families.tsv" | shasum -a 256 -c - || { echo 'Checksum verification failed.' >&2; exit 1; }
148
+ shasum -a 256 readclusterpang/gene_families.tsv >> info_to_test/checksum.txt
149
+ cd -
150
+ - name: testing rgp_cluster command
151
+ shell: bash -l {0}
152
+ run: |
153
+ cd testingDataset
154
+ ppanggolin rgp_cluster --pangenome mybasicpangenome/pangenome.h5
155
+ ppanggolin rgp_cluster --pangenome mybasicpangenome/pangenome.h5 --ignore_incomplete_rgp --grr_metric max_grr -f --graph_formats graphml gexf
156
+ ppanggolin rgp_cluster --pangenome mybasicpangenome/pangenome.h5 --no_identical_rgp_merging -o rgp_clustering_no_identical_rgp_merging --graph_formats graphml
157
+ cd -
158
+ - name: testing align command
159
+ shell: bash -l {0}
160
+ run: |
161
+ cd testingDataset
162
+ ppanggolin align --pangenome mybasicpangenome/pangenome.h5 --sequences some_chlam_proteins.fasta \
163
+ --output test_align --draw_related --getinfo --fast --cpu $NUM_CPUS
164
+ cd -
165
+ - name: testing context command
166
+ shell: bash -l {0}
167
+ run: |
168
+ cd testingDataset
169
+ ppanggolin context --pangenome myannopang/pangenome.h5 --sequences some_chlam_proteins.fasta --output test_context --fast --cpu $NUM_CPUS
170
+
171
+ # test from gene family ids. Test here with one family of module 1. The context should find all families of module 1
172
+ echo AP288_RS05055 > one_family_of_module_1.txt
173
+ ppanggolin context --pangenome myannopang/pangenome.h5 --family one_family_of_module_1.txt --output test_context_from_id --cpu $NUM_CPUS
174
+ cd -
175
+ - name: testing metadata command
176
+ shell: bash -l {0}
177
+ run: |
178
+ cd testingDataset
179
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db1 -m metadata/metadata_genes.tsv -a genes
180
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db2 -m metadata/metadata_genomes.tsv -a genomes
181
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db3 -m metadata/metadata_families.tsv -a families --omit
182
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db4 -m metadata/metadata_rgps.tsv -a RGPs
183
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db5 -m metadata/metadata_contigs.tsv -a contigs
184
+ ppanggolin metadata -p mybasicpangenome/pangenome.h5 -s db6 -m metadata/metadata_modules.tsv -a modules
185
+ ppanggolin write_metadata -p mybasicpangenome/pangenome.h5 -o metadata_flat_output
186
+
187
+
188
+ ppanggolin write_pangenome -p mybasicpangenome/pangenome.h5 --output mybasicpangenome -f --gexf --light_gexf --cpu $NUM_CPUS
189
+ ppanggolin rgp_cluster --pangenome mybasicpangenome/pangenome.h5 -o rgp_cluster_with_metadata --graph_formats graphml
190
+ cd -
191
+ - name: testing config file
192
+ shell: bash -l {0}
193
+ run: |
194
+ cd testingDataset
195
+ ppanggolin utils --default_config panrgp -o panrgp_default_config.yaml
196
+ cut -f1,2 clusters.tsv > clusters_without_frag.tsv
197
+ ppanggolin panrgp --anno genomes.gbff.list --cluster clusters_without_frag.tsv -o test_config --config panrgp_default_config.yaml --cpu $NUM_CPUS
198
+ echo "$(grep 'test_config/gene_families.tsv' expected_info_files/checksum.txt | cut -d' ' -f1) test_config/gene_families.tsv" | shasum -a 256 -c - || { echo 'Checksum verification failed.' >&2; exit 1; }
199
+ shasum -a 256 test_config/gene_families.tsv >> info_to_test/checksum.txt
200
+ cd -
201
+ - name: testing projection cmd
202
+ shell: bash -l {0}
203
+ run: |
204
+ cd testingDataset
205
+ head genomes.gbff.list | sed 's/^/input_genome_/g' > genomes.gbff.head.list
206
+ ppanggolin projection --pangenome stepbystep/pangenome.h5 -o projection_from_list_of_gbff --anno genomes.gbff.head.list --gff --proksee --cpu $NUM_CPUS
207
+
208
+ head genomes.fasta.list | sed 's/^/input_genome_/g' > genomes.fasta.head.list
209
+ ppanggolin projection --pangenome myannopang/pangenome.h5 -o projection_from_list_of_fasta --fasta genomes.fasta.head.list --gff --proksee --cpu $NUM_CPUS
210
+
211
+ ppanggolin projection --pangenome mybasicpangenome/pangenome.h5 -o projection_from_single_fasta \
212
+ --genome_name chlam_A --fasta FASTA/GCF_002776845.1_ASM277684v1_genomic.fna.gz \
213
+ --spot_graph --graph_formats graphml --fast --keep_tmp -f --add_sequences --gff --proksee --table --add_metadata --cpu $NUM_CPUS
214
+
215
+ ppanggolin projection --pangenome mybasicpangenome/pangenome.h5 -o projection_from_gff_prodigal \
216
+ --genome_name chlam_annotated_with_prodigal --anno GBFF/GCF_003788785.1_ct114V1_genomic_prodigal_annotation.gff.gz \
217
+ --gff --table --cpu $NUM_CPUS
218
+
219
+ # projection of a plasmid with chevron that have been added manually to test chevron handeling in GFF
220
+ ppanggolin projection --pangenome myannopang/pangenome.h5 --anno GBFF/plasmid_NZ_CP007132_with_manually_added_chevrons.gff.gz --cpu $NUM_CPUS -o projection_plasmid_with_chevron
221
+
222
+ # projection with GFF with no sequence and fasta sequence
223
+ ppanggolin projection -p myannopang/pangenome.h5 --anno GBFF/plasmid_GCF_000093005.1_ASM9300v1.gff.gz --fasta GBFF/plasmid_GCF_000093005.1_ASM9300v1.fna.gz
224
+
225
+ # projection with GFF with no sequence and fasta sequence specified in a TSV file with other GFF (but with sequences)
226
+ head -n 3 genomes.gbff.head.list > genomes.gbff.h3_and_GFFplasmidNoSeq.list
227
+
228
+ echo GFF_plasmid_No_seq$'\t'GBFF/plasmid_GCF_000093005.1_ASM9300v1.gff.gz >> genomes.gbff.h3_and_GFFplasmidNoSeq.list
229
+ echo GFF_plasmid_No_seq$'\t'GBFF/plasmid_GCF_000093005.1_ASM9300v1.fna.gz >> genomes.fna.GFFplasmidNoSeq.list
230
+ ppanggolin projection -p myannopang/pangenome.h5 --anno genomes.gbff.h3_and_GFFplasmidNoSeq.list --fasta genomes.fna.GFFplasmidNoSeq.list
231
+
232
+ - name: testing write_genome_cmds
233
+ shell: bash -l {0}
234
+ run: |
235
+ cd testingDataset
236
+ head genomes.gbff.list | cut -f1 > genome_names.gbff.head.list
237
+
238
+ ppanggolin write_genomes -p myannopang/pangenome.h5 --output flat_genomes_from_genome_files -f \
239
+ --anno genomes.gbff.list --gff --table --genomes genome_names.gbff.head.list
240
+
241
+ ppanggolin write_genomes -p stepbystep/pangenome.h5 --output flat_genomes_from_cmdline_genomes --proksee \
242
+ --genomes GCF_006508185.1_ASM650818v1_genomic,GCF_002088315.1_ASM208831v1_genomic
243
+
244
+ head genomes.fasta.list | cut -f1 > genome_names.fasta.head.list
245
+ # Default separator is a pipe but a pipe is found in a value of metadata db1. That is why we use another separator here.
246
+ ppanggolin write_genomes -p mybasicpangenome/pangenome.h5 --output mybasicpangenome/genomes_outputs \
247
+ --genomes genome_names.fasta.head.list \
248
+ -f --gff --add_metadata --table --metadata_sep § --proksee
249
+
250
+ # Pipe separatore is found in metadata source db1. if we don't require this source then the writting with pipe is work fine.
251
+ ppanggolin write_genomes -p mybasicpangenome/pangenome.h5 --output mybasicpangenome/genomes_outputs_with_metadata -f --gff --proksee --table --add_metadata --metadata_sources db2 db3 db4
252
+
253
+ - name: Archive diff files
254
+ uses: actions/upload-artifact@v4
255
+ with:
256
+ name: comparison-results_${{ matrix.os }}_python${{ matrix.python-version }}
257
+ path: testingDataset/info_to_test/*
258
+
259
+
260
+ - name: testing info output
261
+ shell: bash -l {0}
262
+ run: |
263
+ cd testingDataset
264
+ python compare_results.py -e expected_info_files/ -t info_to_test/ -o diff_output
@@ -0,0 +1,93 @@
1
+ name: Build and upload to PyPI
2
+
3
+ # With the current configuration, the workflow (except PyPI upload)
4
+ # will run on every pull request and on every push to dev branch.
5
+ # This way we can track if some changes break the wheel builds.
6
+ #
7
+ # The PyPI publishing job will only run on release events.
8
+ on:
9
+ workflow_dispatch: # This allows manual triggering of the workflow for testing on test.pypi.org
10
+ inputs:
11
+ publish_test:
12
+ description: "Publish on test.pypi.org"
13
+ required: false
14
+ default: false
15
+
16
+ publish_to_pypi:
17
+ description: "Publish on pypi.org. Use with caution!"
18
+ required: false
19
+ default: false
20
+
21
+
22
+ pull_request:
23
+ push:
24
+ branches:
25
+ - dev
26
+ release:
27
+ types:
28
+ - released # Run only on release events, use 'published' to run on draft or pre-release
29
+
30
+ jobs:
31
+ build_wheels:
32
+ name: Build wheels on ${{ matrix.os }}
33
+ runs-on: ${{ matrix.os }}
34
+ strategy:
35
+ matrix:
36
+ # macos-13 is an Intel runner, macos-14 is an Apple Silicon runner
37
+ # Should we support Linux ARM? If so, add 'ubuntu-24.04-arm'
38
+ os: [ubuntu-latest, macos-13, macos-14, ubuntu-24.04-arm]
39
+
40
+ steps:
41
+ - uses: actions/checkout@v4
42
+
43
+ - name: Build wheels
44
+ uses: pypa/cibuildwheel@v2.23.0
45
+ # We skip:
46
+ # - 32-bit builds
47
+ # - PyPy builds (check if ppanggolin works on PyPy first)
48
+ env:
49
+ CIBW_SKIP: "*_i686 pp*"
50
+
51
+ - uses: actions/upload-artifact@v4
52
+ with:
53
+ name: cibw-wheels-${{ matrix.os }}-${{ strategy.job-index }}
54
+ path: ./wheelhouse/*.whl
55
+
56
+ build_sdist:
57
+ name: Build source distribution
58
+ runs-on: ubuntu-latest
59
+ steps:
60
+ - uses: actions/checkout@v4
61
+
62
+ - name: Build sdist
63
+ run: pipx run build --sdist
64
+
65
+ - uses: actions/upload-artifact@v4
66
+ with:
67
+ name: cibw-sdist
68
+ path: dist/*.tar.gz
69
+
70
+ upload_pypi:
71
+ needs: [build_wheels, build_sdist]
72
+ runs-on: ubuntu-latest
73
+ environment: pypi
74
+ permissions:
75
+ id-token: write
76
+ # Publish on PyPI only on release events
77
+ if: (github.event_name == 'release' && github.event.action == 'released') || (github.event_name == 'workflow_dispatch' && (inputs.publish_test == 'true' || inputs.publish_to_pypi == 'true'))
78
+ steps:
79
+ - uses: actions/download-artifact@v4
80
+ with:
81
+ pattern: cibw-*
82
+ path: dist
83
+ merge-multiple: true
84
+
85
+ - name: Publish to PyPI
86
+ if: (github.event_name == 'release' && github.event.action == 'released') || (github.event_name == 'workflow_dispatch' && inputs.publish_to_pypi == 'true')
87
+ uses: pypa/gh-action-pypi-publish@release/v1
88
+
89
+ - name: Publish to test.pypi.org
90
+ if: github.event_name == 'workflow_dispatch' && inputs.publish_test == 'true'
91
+ uses: pypa/gh-action-pypi-publish@release/v1
92
+ with:
93
+ repository-url: https://test.pypi.org/legacy/
@@ -0,0 +1,35 @@
1
+ # .readthedocs.yaml
2
+ # Read the Docs configuration file
3
+ # See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
4
+
5
+ # Required
6
+ version: 2
7
+ python:
8
+ install:
9
+ - method: pip
10
+ path: .
11
+ extra_requirements:
12
+ - doc
13
+
14
+ # Set the OS, Python version and other tools you might need
15
+ build:
16
+ os: ubuntu-22.04
17
+ tools:
18
+ python: "3.10"
19
+
20
+
21
+ # Build documentation in the "docs/" directory with Sphinx
22
+ sphinx:
23
+ configuration: docs/conf.py
24
+
25
+ # Optionally build your docs in additional formats such as PDF and ePub
26
+ # formats:
27
+ # - pdf
28
+ # - epub
29
+
30
+ # Optional but recommended, declare the Python requirements required
31
+ # to build your documentation
32
+ # See https://docs.readthedocs.io/en/stable/guides/reproducible-builds.html
33
+ # python:
34
+ # install:
35
+ # - requirements: docs/requirements.txt