py-semtools 1.3__tar.gz → 1.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. {py_semtools-1.3 → py_semtools-1.3.3}/PKG-INFO +6 -3
  2. py_semtools-1.3.3/README.md +6 -0
  3. {py_semtools-1.3 → py_semtools-1.3.3}/setup.cfg +11 -2
  4. py_semtools-1.3.3/src/py_semtools/__init__.py +16 -0
  5. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/cli_manager.py +82 -16
  6. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/indexers/text_indexer.py +91 -54
  7. py_semtools-1.3.3/src/py_semtools/lexical_engines/engine_baseclass.py +156 -0
  8. py_semtools-1.3.3/src/py_semtools/lexical_engines/fmEngine.py +99 -0
  9. {py_semtools-1.3/src/py_semtools → py_semtools-1.3.3/src/py_semtools/lexical_engines}/stEngine.py +83 -145
  10. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/main_modules.py +81 -29
  11. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/ontology.py +120 -89
  12. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parallelizer.py +0 -1
  13. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/json_parser.py +1 -1
  14. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/oboparser.py +1 -1
  15. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_abstract_parser.py +20 -15
  16. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_paper_parser.py +23 -19
  17. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/report_ont.py +1 -2
  18. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/sim_handler.py +0 -1
  19. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontoplot.txt +1 -1
  20. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/PKG-INFO +6 -3
  21. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/SOURCES.txt +4 -1
  22. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/entry_points.txt +1 -0
  23. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/requires.txt +6 -2
  24. {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/Report.html +221 -80
  25. {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/launch.sh +1 -1
  26. py_semtools-1.3.3/tests/cli_examples/styles.css +130 -0
  27. py_semtools-1.3.3/tests/cli_examples/template.txt +84 -0
  28. py_semtools-1.3.3/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981088 +1 -0
  29. py_semtools-1.3.3/tests/data/stEngine/expected/raw_indexes/papers/chunk1/38108203 +1 -0
  30. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/launch_plots.py +3 -3
  31. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_cli_manager.py +26 -27
  32. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_go.py +2 -3
  33. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_jsonparser.py +3 -5
  34. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_oboparser.py +3 -4
  35. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_ontology.py +3 -1
  36. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_abs_parser.py +15 -15
  37. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_pap_parser.py +38 -10
  38. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_text_parser.py +2 -4
  39. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_similitudes.py +1 -4
  40. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_stengine.py +2 -7
  41. {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_text_indexer.py +45 -36
  42. {py_semtools-1.3 → py_semtools-1.3.3}/tox.ini +2 -0
  43. py_semtools-1.3/README.md +0 -0
  44. py_semtools-1.3/src/py_semtools/__init__.py +0 -12
  45. py_semtools-1.3/tests/cli_examples/template.txt +0 -119
  46. py_semtools-1.3/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981088 +0 -0
  47. py_semtools-1.3/tests/data/stEngine/expected/raw_indexes/papers/chunk1/38108203 +0 -0
  48. {py_semtools-1.3 → py_semtools-1.3.3}/.coveragerc +0 -0
  49. {py_semtools-1.3 → py_semtools-1.3.3}/.gitignore +0 -0
  50. {py_semtools-1.3 → py_semtools-1.3.3}/.readthedocs.yml +0 -0
  51. {py_semtools-1.3 → py_semtools-1.3.3}/AUTHORS.rst +0 -0
  52. {py_semtools-1.3 → py_semtools-1.3.3}/CHANGELOG.rst +0 -0
  53. {py_semtools-1.3 → py_semtools-1.3.3}/CONTRIBUTING.rst +0 -0
  54. {py_semtools-1.3 → py_semtools-1.3.3}/LICENSE.txt +0 -0
  55. {py_semtools-1.3 → py_semtools-1.3.3}/MANIFEST.in +0 -0
  56. {py_semtools-1.3 → py_semtools-1.3.3}/README.rst +0 -0
  57. {py_semtools-1.3 → py_semtools-1.3.3}/docs/Makefile +0 -0
  58. {py_semtools-1.3 → py_semtools-1.3.3}/docs/_static/.gitignore +0 -0
  59. {py_semtools-1.3 → py_semtools-1.3.3}/docs/authors.rst +0 -0
  60. {py_semtools-1.3 → py_semtools-1.3.3}/docs/changelog.rst +0 -0
  61. {py_semtools-1.3 → py_semtools-1.3.3}/docs/conf.py +0 -0
  62. {py_semtools-1.3 → py_semtools-1.3.3}/docs/contributing.rst +0 -0
  63. {py_semtools-1.3 → py_semtools-1.3.3}/docs/index.rst +0 -0
  64. {py_semtools-1.3 → py_semtools-1.3.3}/docs/license.rst +0 -0
  65. {py_semtools-1.3 → py_semtools-1.3.3}/docs/readme.rst +0 -0
  66. {py_semtools-1.3 → py_semtools-1.3.3}/docs/requirements.txt +0 -0
  67. {py_semtools-1.3 → py_semtools-1.3.3}/pyproject.toml +0 -0
  68. {py_semtools-1.3 → py_semtools-1.3.3}/setup.py +0 -0
  69. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/external_data/ontologies.txt +0 -0
  70. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/file_parser.py +0 -0
  71. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_basic_parser.py +0 -0
  72. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_parser.py +0 -0
  73. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/comorb_sugg.txt +0 -0
  74. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/makeTermFreqTable.txt +0 -0
  75. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontoICdist.txt +0 -0
  76. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontodist.txt +0 -0
  77. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/plotClust.txt +0 -0
  78. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/plotProfRed.txt +0 -0
  79. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/report.txt +0 -0
  80. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/similarity_heatmap.txt +0 -0
  81. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/similarity_matrix.txt +0 -0
  82. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/stEngine.txt +0 -0
  83. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/dependency_links.txt +0 -0
  84. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/not-zip-safe +0 -0
  85. {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/top_level.txt +0 -0
  86. {py_semtools-1.3 → py_semtools-1.3.3}/tests/__init__.py +0 -0
  87. {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/example +0 -0
  88. {py_semtools-1.3 → py_semtools-1.3.3}/tests/conftest.py +0 -0
  89. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/branched.obo +0 -0
  90. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/circular_sample.obo +0 -0
  91. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology.obo +0 -0
  92. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology2.obo +0 -0
  93. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology3.obo +0 -0
  94. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_profs/report.html +0 -0
  95. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_profs/report.txt +0 -0
  96. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_query_parentals.txt +0 -0
  97. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_target_and_query_parentals.txt +0 -0
  98. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_target_parentals.txt +0 -0
  99. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/no_filter_limit_2.txt +0 -0
  100. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/no_filter_no_limit.txt +0 -0
  101. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/enrichment_ontology3.obo +0 -0
  102. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/query_hps.txt +0 -0
  103. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/relations.txt +0 -0
  104. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/go-basic_sample.obo +0 -0
  105. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/go-onlyOne.obo +0 -0
  106. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_alt_chain.obo +0 -0
  107. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_compressed.obo.gz +0 -0
  108. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_sample.obo +0 -0
  109. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles +0 -0
  110. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles_2cols +0 -0
  111. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles_with_removedTerms +0 -0
  112. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/string_values +0 -0
  113. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms +0 -0
  114. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms_for_xref +0 -0
  115. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms_list +0 -0
  116. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/only_header_sample.obo +0 -0
  117. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/partial_go.json +0 -0
  118. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/partial_go.obo +0 -0
  119. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/blacklisted.txt +0 -0
  120. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/cleaned_profiles +0 -0
  121. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/cleaned_profiles_2cols +0 -0
  122. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/expanded_profiles +0 -0
  123. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/expected_IC_ont +0 -0
  124. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/parental_from_terms +0 -0
  125. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/profile_stats +0 -0
  126. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/profiles_IC_onto_freq +0 -0
  127. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/strsimnet +0 -0
  128. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/strsimnet_cutoff2 +0 -0
  129. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/terms_attr +0 -0
  130. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/translated_profiles_names +0 -0
  131. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/translated_terms_codes +0 -0
  132. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/white_and_blacklisted.txt +0 -0
  133. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/whitelisted.txt +0 -0
  134. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/short_hierarchical_sample.obo +0 -0
  135. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/sparse2_sample.obo +0 -0
  136. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/sparse_sample.obo +0 -0
  137. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981089 +0 -0
  138. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981091 +0 -0
  139. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981092 +0 -0
  140. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/None +0 -0
  141. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/single_abs_text +0 -0
  142. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/18382669 +0 -0
  143. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/31356151 +0 -0
  144. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/35042469 +0 -0
  145. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/None +0 -0
  146. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/single_pap_text +0 -0
  147. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/abstracts/single_abs_index_nonsplit +0 -0
  148. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/abstracts/single_abs_index_split +0 -0
  149. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/papers/single_pap_index_nonsplit +0 -0
  150. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/papers/single_pap_index_split +0 -0
  151. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/abs_chunk1.xml.gz +0 -0
  152. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/abs_chunk2.xml.gz +0 -0
  153. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/single_abstract.xml +0 -0
  154. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/single_abstract.xml.gz +0 -0
  155. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words1.txt +0 -0
  156. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words2.txt +0 -0
  157. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words3.txt +0 -0
  158. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/example.xml +0 -0
  159. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/example_zipped.xml.gz +0 -0
  160. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/pap_chunk1.tar.gz +0 -0
  161. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/pap_chunk2.tar.gz +0 -0
  162. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/single_paper.tar.gz +0 -0
  163. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/single_paper.xml +0 -0
  164. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/abstracts/example1.txt.gz +0 -0
  165. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/queries/hpo_list +0 -0
  166. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/queries/mondo_list +0 -0
  167. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/st_engine_report/abs_profiles.txt +0 -0
  168. {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/st_engine_report/pmids_and_titles.txt +0 -0
  169. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/data/profiles.txt +0 -0
  170. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/data/ref_profile.txt +0 -0
  171. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/launch.sh +0 -0
  172. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/profiles_freqs.html +0 -0
  173. {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/template.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: py_semtools
3
- Version: 1.3
3
+ Version: 1.3.3
4
4
  Summary: Library to handle ontologies that allows queries and calculations such as information coefficients, semantic similarity, ontology representations, etc in a easy way. It can load any ontology that complies the obo format supported by OBO Foundry.
5
5
  Home-page: https://github.com/seoanezonjic/py_semtools
6
6
  Author: seoanezonjic
@@ -18,8 +18,6 @@ Requires-Dist: numpy
18
18
  Requires-Dist: networkx
19
19
  Requires-Dist: entrezpy
20
20
  Requires-Dist: scipy
21
- Requires-Dist: sentence_transformers
22
- Requires-Dist: langchain
23
21
  Requires-Dist: pynvml
24
22
  Requires-Dist: requests
25
23
  Requires-Dist: py_exp_calc
@@ -27,6 +25,11 @@ Requires-Dist: py_cmdtabs
27
25
  Requires-Dist: py_report_html
28
26
  Requires-Dist: loguru
29
27
  Requires-Dist: loky
28
+ Requires-Dist: bm25s
29
+ Provides-Extra: models
30
+ Requires-Dist: sentence_transformers; extra == "models"
31
+ Requires-Dist: langchain; extra == "models"
32
+ Requires-Dist: langchain-text-splitters; extra == "models"
30
33
  Provides-Extra: testing
31
34
  Requires-Dist: setuptools; extra == "testing"
32
35
  Requires-Dist: pytest; extra == "testing"
@@ -0,0 +1,6 @@
1
+ If you want to use lexical models capabilities (stEngine and fmEngine binaries), you should install extra required dependencies that are not installed
2
+ with the base installation. For that end, you can install semtools like the following:
3
+
4
+ ```bash
5
+ pip install py_semtools[MODELS]
6
+ ```
@@ -28,8 +28,6 @@ install_requires =
28
28
  networkx
29
29
  entrezpy
30
30
  scipy
31
- sentence_transformers
32
- langchain
33
31
  pynvml
34
32
  requests
35
33
  py_exp_calc
@@ -37,6 +35,7 @@ install_requires =
37
35
  py_report_html
38
36
  loguru
39
37
  loky
38
+ bm25s
40
39
 
41
40
  [options.packages.find]
42
41
  where = src
@@ -50,6 +49,7 @@ py_semtools.templates =
50
49
  *.txt
51
50
 
52
51
  [options.extras_require]
52
+ MODELS = sentence_transformers; langchain; langchain-text-splitters
53
53
  testing =
54
54
  setuptools
55
55
  pytest
@@ -62,11 +62,13 @@ console_scripts =
62
62
  get_sorted_suggestions = py_semtools.cli_manager:get_sorted_suggestions
63
63
  get_sorted_profs = py_semtools.cli_manager:get_sorted_profs
64
64
  remote_retriever = py_semtools.cli_manager:remote_retriever
65
+ fmEngine = py_semtools.cli_manager:fmEngine
65
66
  stEngine = py_semtools.cli_manager:stEngine
66
67
  stEngine_report = py_semtools.cli_manager:stEngine_report
67
68
  get_corpus_index = py_semtools.cli_manager:get_corpus_index
68
69
 
69
70
  [tool:pytest]
71
+ python_files = tests.py test_*.py *_tests.py
70
72
  addopts =
71
73
  --cov py_semtools --cov-report term-missing
72
74
  --verbose
@@ -74,6 +76,13 @@ norecursedirs =
74
76
  dist
75
77
  build
76
78
  .tox
79
+ .git
80
+ docs
81
+ src/py_semtools/templates
82
+ tests/cli_examples
83
+ tests/data
84
+ tests/demo_examples
85
+ tests/__pycache__
77
86
  testpaths = tests
78
87
 
79
88
  [devpi:upload]
@@ -0,0 +1,16 @@
1
+ import sys
2
+
3
+ if sys.version_info[:2] >= (3, 8):
4
+ # TODO: Import directly (no need for conditional) when `python_requires = >= 3.8`
5
+ from importlib.metadata import PackageNotFoundError, version # pragma: no cover
6
+ else:
7
+ from importlib_metadata import PackageNotFoundError, version # pragma: no cover
8
+
9
+ try:
10
+ # Change here if project is renamed and does not equal the package name
11
+ dist_name = __name__
12
+ __version__ = version(dist_name)
13
+ except PackageNotFoundError: # pragma: no cover
14
+ __version__ = "unknown"
15
+ finally:
16
+ del version, PackageNotFoundError
@@ -1,5 +1,7 @@
1
1
  import argparse, re, sys, inspect
2
2
  from py_semtools.main_modules import *
3
+ from collections import defaultdict
4
+
3
5
 
4
6
  ###########################################################################
5
7
  ## TYPES
@@ -9,6 +11,12 @@ def one_column_file(file): return [line.strip() for line in open(file).readlines
9
11
 
10
12
  def text_list(string): return string.split(',')
11
13
 
14
+ def text_to_default_dict(string):
15
+ d = defaultdict(lambda: False)
16
+ for item in text_list(string):
17
+ d[item] = True
18
+ return d
19
+
12
20
  def filter_regex(string):
13
21
  filters = []
14
22
  pattern = re.compile(r"([pn])\(([A-Za-z:0-9,]*)\)")
@@ -73,8 +81,8 @@ def remote_retriever(args = None):
73
81
  main_remote_retriever(opts)
74
82
 
75
83
  def semtools(args = None):
76
- if args is None:
77
- args = sys.argv[1:]
84
+
85
+ if args is None: args = sys.argv[1:]
78
86
 
79
87
  parser = argparse.ArgumentParser(description='Perform Ontology driven analysis ')
80
88
  parser.add_argument("--return_all_terms_with_user_defined_attributes", dest="return_all_terms_with_user_defined_attributes", default=None,
@@ -98,10 +106,10 @@ def semtools(args = None):
98
106
  parser.add_argument("-t", "--translate", dest="translate", default= None,
99
107
  help="Translate to 'names' or to 'codes'")
100
108
  parser.add_argument("-s", "--similarity_method", dest="similarity", default= None,
101
- help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods.")
109
+ help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods. Recently added 'eric', 'neric', 'nweric' and 'erlin' methods.")
102
110
  parser.add_argument("--reference_profiles", dest="reference_profiles", default= None,
103
111
  help="Path to file tabulated file with first column as id profile and second column with ontology terms separated by separator.")
104
- parser.add_argument('-c', "--clean_profiles", dest="clean_profiles", default= False, action='store_true',
112
+ parser.add_argument("--disable_cleaning_profiles", dest="clean_profiles", default= True, action='store_false',
105
113
  help="Removes ancestors, descendants and obsolete terms from profiles.")
106
114
  parser.add_argument('-r', "--removed_path", dest="removed_path", default= 'rejected_profs',
107
115
  help="Desired path to write removed profiles file.")
@@ -157,6 +165,8 @@ def semtools(args = None):
157
165
  help="For the output report, it sets the root term to show in center of ontoplot")
158
166
  parser.add_argument("--ref_term", dest="ref_term", default=None,
159
167
  help="For the output report, it sets the term whose childs will be in the color legend to indicate the branches of descendants")
168
+ parser.add_argument("--similarity_cluster", dest="similarity_cluster", default=None,
169
+ help="Use to activate profiles clustering with a similarity method and generate basefiles (resnik', 'lin' or 'jiang_conrath). Not active by default")
160
170
  parser.add_argument("--similarity_cluster_plot", dest="similarity_cluster_plot", default=None,
161
171
  help="For output report, use to activate profiles clustering and clustermap plot with a similarity method (resnik', 'lin' or 'jiang_conrath). Not active by default")
162
172
  parser.add_argument("--cl_size_factor", dest="cl_size_factor", default=1.0, type=float,
@@ -167,9 +177,22 @@ def semtools(args = None):
167
177
  help="For the ontoplot, use it to deactivate propagating frequency to parentals terms")
168
178
  parser.add_argument("--onto_min_freq", dest="onto_min_freq", default=0.005, type=float,
169
179
  help="For the ontoplot, it sets the minimum frequency of terms to be shown in the plot. Default is 0.5 percent. Set to 0 to show all terms.")
180
+ parser.add_argument("--similarity_index", dest="similarity_index", default=None,
181
+ help="Term to term similarity index to perform profile similarity calculations")
182
+ parser.add_argument("--get_LCA_from_profile", dest="get_LCA_from_profile", default=None, type=float,
183
+ help="For a list if profiles, give for each one the list of LCA that occurs for the selected frequency in the profile terms.")
184
+ parser.add_argument("--get_MICA_from_profile", dest="get_MICA_from_profile", default=None, type=float,
185
+ help="For a list of profiles, give for each one the list of LCA that occurs for the selected frequency in the profile terms.")
186
+ parser.add_argument("--LCA_list", dest="LCA_list", default=None, type=str,
187
+ help="With flags 'get_LCA_from_profile' and 'get_MICA_from_profile' constraint the ancestors to the given list of terms (one per line).")
188
+ parser.add_argument("--get_representative_profile", dest="get_representative_profile", default=None, type=float,
189
+ help="For a list of profiles, give one single profile with terms presented with al least FLOAT frecuency in the profile list. Parents are infered and computed too. The most spefic terms are retained.")
190
+
170
191
  opts = parser.parse_args(args)
192
+
171
193
  main_semtools(opts)
172
194
 
195
+
173
196
  def get_sorted_suggestions(args = None):
174
197
  if args is None:
175
198
  args = sys.argv[1:]
@@ -210,6 +233,37 @@ def get_sorted_suggestions(args = None):
210
233
  opts = parser.parse_args(args)
211
234
  main_get_sorted_suggestions(opts)
212
235
 
236
+
237
+ def fmEngine(args = None):
238
+ if args is None:
239
+ args = sys.argv[1:]
240
+
241
+ parser = argparse.ArgumentParser(description='Perform Ontology driven analysis with Sentence Transformer')
242
+
243
+ parser.add_argument('-m', "--model_name", dest="model_name", default= None,
244
+ help="Model to be used. Current options include 'bm25' and 'tfidf'")
245
+ parser.add_argument('-q', "--query", dest="query", default= None,
246
+ help="Path to the query file. Wildcards are accepted to process multiple files")
247
+ parser.add_argument('-c', "--corpus", dest="corpus", default= None,
248
+ help="Path to the corpus file. Wildcards are accepted to process multiple files")
249
+ parser.add_argument('-o', "--output_file", dest="output_file", default= None,
250
+ help="Path to save the output file with semantic scores")
251
+ parser.add_argument('-k', "--top_k", dest="top_k", default= 20, type = int,
252
+ help="Get top scores per keyword")
253
+ parser.add_argument('-t', "--threshold", dest="threshold", default= 0, type = float,
254
+ help="Similarity threshold to filter results to write")
255
+ parser.add_argument('-v', "--verbose", dest="verbose", default= False, action='store_true',
256
+ help="Toogle on to get verbose output")
257
+ parser.add_argument("--order", dest="order", default= "corpus-query",
258
+ help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
259
+ parser.add_argument('-s', "--split", dest="split", default= False, action='store_true',
260
+ help="Use it if your corpus comes splitted in smaller parts as list of lists (embedded in a json)")
261
+ parser.add_argument("--print_relevant_pairs", dest="print_relevant_pairs", default= False, action='store_true',
262
+ help="Use it to print the relevant pairs of query-corpus with their scores")
263
+ opts = parser.parse_args(args)
264
+ main_fmEngine(opts)
265
+
266
+
213
267
  def stEngine(args = None):
214
268
  if args is None:
215
269
  args = sys.argv[1:]
@@ -218,6 +272,8 @@ def stEngine(args = None):
218
272
 
219
273
  parser.add_argument('-m', "--model_name", dest="model_name", default= None,
220
274
  help="Name of the model to be used")
275
+ parser.add_argument('-r', "--reranker_model_name", dest="reranker_model_name", default= None,
276
+ help="(Optional) Name of the reranker model to be used")
221
277
  parser.add_argument('-p', "--model_path", dest="model_path", default= None,
222
278
  help="Path where the model is cached or where it will be stored")
223
279
  parser.add_argument('-q', "--query", dest="query", default= None,
@@ -235,21 +291,25 @@ def stEngine(args = None):
235
291
  parser.add_argument('-t', "--threshold", dest="threshold", default= 0, type = float,
236
292
  help="Similarity threshold to filter results to write")
237
293
  parser.add_argument('-v', "--verbose", dest="verbose", default= False, action='store_true',
238
- help="Toogle on to get verbose output")
239
- parser.add_argument('-g', "--gpu_device", dest="gpu_device", default= [], type=text_list,
240
- help="Use to specify the GPU device to be used for speed-up (if available). The format is like: 'cuda:0' or 'cuda:0,cuda:1' to use multiple GPUs or cpu,cpu to use multiple CPUs")
241
- parser.add_argument("-b", "--batch_size", dest="batch_size", default= 32, type=int,
242
- help="Use to specify batch size for multi-GPU embedding step")
243
- parser.add_argument("--use_gpu_for_sim_calculation", dest="use_gpu_for_sim_calculation", default= False, action='store_true',
244
- help="Toogle on if you want to use GPU not only for the embedding process, but also for calculating query-corpus similarities (then dot score if used instead of cosine similarity)")
294
+ help="Toogle on to get verbose output")
245
295
  parser.add_argument('-s', "--split", dest="split", default= False, action='store_true',
246
296
  help="Use it if your corpus comes splitted in smaller parts as list of lists (embedded in a json)")
247
297
  parser.add_argument("--order", dest="order", default= "corpus-query",
248
- help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
249
- parser.add_argument("--chunk_size", dest="chunk_size", default=10000, type=int,
250
- help="Size to be accumulating corpora until a threshold is reached before proceeding to embedd")
298
+ help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
251
299
  parser.add_argument("--print_relevant_pairs", dest="print_relevant_pairs", default= False, action='store_true',
252
- help="Use it to print the relevant pairs of query-corpus with their scores")
300
+ help="Use it to print the relevant pairs of query-corpus with their scores")
301
+ parser.add_argument('-g', "--gpu_device", dest="gpu_device", default= [], type=text_list,
302
+ help="Use to specify the GPU device to be used for speed-up (if available). The format is like: 'cuda:0' or 'cuda:0,cuda:1' to use multiple GPUs or cpu,cpu to use multiple CPUs")
303
+ parser.add_argument("--use_gpu_for_sim_calculation", dest="use_gpu_for_sim_calculation", default= False, action='store_true',
304
+ help="Toogle on if you want to use GPU not only for the embedding process, but also for calculating query-corpus similarities (then dot score if used instead of cosine similarity)")
305
+ parser.add_argument("--chunk_size", dest="chunk_size", default=10000, type=int,
306
+ help="Size to be accumulating corpora (documents) until a threshold is reached before proceeding to embedd")
307
+ parser.add_argument("--chunk_size_sentences", dest="chunk_size_sentences", default=0, type=int,
308
+ help="(If toogled on, takes precedence over --chunk_size) Size to be accumulating sentences until a threshold is reached before proceeding to embedd")
309
+ parser.add_argument("--single_worker_chunk_size", dest="single_worker_chunk_size", default=10000, type=int,
310
+ help="(Only for multi GPU processing) Chunk size for each worker when multi-GPU processing is used.")
311
+ parser.add_argument("-b", "--batch_size", dest="batch_size", default= 32, type=int,
312
+ help="Use to specify batch size for multi-GPU embedding step")
253
313
  opts = parser.parse_args(args)
254
314
  main_stEngine(opts)
255
315
 
@@ -328,6 +388,10 @@ def get_corpus_index(args = None):
328
388
  help="Use to set the type of cleaning to be performed on the text. Options: basic (do not clean at all), soft or hard (default).")
329
389
  parser.add_argument("--split_type", dest="split_type", default= 'classic',
330
390
  help="Use to set the type of splitting to be performed on the text. Options: classic (default, splitting by '\n\n', \n', '.', ';', ',') or space_overlap ('\n\n', '\n', ' ', '' with overlap to avoid losing information).")
391
+ parser.add_argument("--extra_fields", dest="extra_fields", default= defaultdict(lambda: False), type=text_to_default_dict,
392
+ help="Comma-separated list of extra fields to include in the processed corpus. Available: title and/or keywords. Default: none")
393
+ parser.add_argument("--remain_empty_documents", dest="remain_empty_documents", default= False, action='store_true',
394
+ help="Use this option to keep entries with (main text) empty documents in the processed corpus (if they have at least title or keywords)")
331
395
  opts = parser.parse_args(args)
332
396
  main_get_corpus_index(opts)
333
397
 
@@ -358,7 +422,9 @@ def get_sorted_profs(args=None):
358
422
  parser.add_argument("-X", "--excluded_terms", dest="excluded_terms", default= [], type = one_column_file,
359
423
  help="File with excluded terms. One term code per line.")
360
424
  parser.add_argument("--hard_check", dest="hard_check", default= True, action="store_false",
361
- help="Set to disable hard check cleaning. Default true")
425
+ help="Set to disable hard check cleaning. Default true")
426
+ parser.add_argument("-s", "--similarity_method", dest="similarity", default= 'lin',
427
+ help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods. Recently added 'eric', 'neric', 'nweric' and 'erlin' methods.")
362
428
  parser.add_argument("-o", "--output_file", dest="output_file", default= 'report.html',
363
429
  help="Output report file")
364
430
  parser.add_argument("-f", "--general_prof_freq", dest="term_freq", default= 0, type= float,
@@ -1,8 +1,9 @@
1
+ from collections import defaultdict
1
2
  import sys, os, glob, re, warnings, json, gzip
2
- from langchain.text_splitter import RecursiveCharacterTextSplitter
3
- from py_exp_calc.exp_calc import invert_nested_hash, flatten
3
+ from langchain_text_splitters import RecursiveCharacterTextSplitter
4
+ from py_exp_calc.exp_calc import flatten
4
5
 
5
- from py_cmdtabs import CmdTabs
6
+ from py_cmdtabs.cmdtabs import CmdTabs
6
7
  from py_semtools.parallelizer import Parallelizer
7
8
  from py_semtools.parsers.text.text_basic_parser import TextBasicParser
8
9
  from py_semtools.parsers.text.text_pubmed_abstract_parser import TextPubmedAbstractParser
@@ -31,7 +32,11 @@ class TextIndexer:
31
32
  else:
32
33
  chunks = manager.get_chunks(filenames, workload_balance='disperse_max', workload_function= lambda filename: os.stat(filename).st_size)
33
34
  items = [[[options, chunk, idx], {}] for idx,chunk in enumerate(chunks)]
34
- manager.execute(items, cls.process_files)
35
+ res_status= manager.execute(items, cls.process_files)
36
+ if set(list(res_status)) == {True}:
37
+ print("All processes finished succesfully")
38
+ else:
39
+ print("Some processes finished with errors")
35
40
 
36
41
  @classmethod
37
42
  def process_files(cls, options, filenames, sup_counter, logger = None):
@@ -54,6 +59,7 @@ class TextIndexer:
54
59
  counter += 1
55
60
  # For records saved in accumulated_texts that the loop has not writed
56
61
  cls.write_indexes(accumulated_texts, options["output"], options["tag"], a_suffix=sup_counter, b_suffix=counter, balancer = balancer, split_output_files = options["split_output_files"], items_per_file = options["text_balancing_size"])
62
+ return True
57
63
 
58
64
  @classmethod
59
65
  def write_indexes(cls, indexes, folder, name, a_suffix=None, b_suffix=None, balancer = None, split_output_files = False, items_per_file = 0):
@@ -71,13 +77,13 @@ class TextIndexer:
71
77
  f = gzip.open(out_filename, 'wt')
72
78
  item_count = 0
73
79
  while indexes:
74
- pmid, text, original_filename, year, text_length, number_of_sentences, length_of_sentences, title, article_type, article_category = indexes.pop()
80
+ pmid, text, original_filename, year, text_length, number_of_sentences, length_of_sentences, title, article_type, article_category, keywords = indexes.pop()
75
81
  if split_output_files and item_count >= items_per_file:
76
82
  f.close()
77
83
  file_count += 1
78
84
  item_count = 0
79
85
  f = gzip.open(os.path.join(folder, name+f"{a_suffix}{b_suffix}_{file_count}.gz" ), 'wt')
80
- f.write(f"{pmid}\t{text}\t{original_filename}\t{year}\t{text_length}\t{number_of_sentences}\t{length_of_sentences}\t{title}\t{article_type}\t{article_category}\n")
86
+ f.write(f"{pmid}\t{text}\t{original_filename}\t{year}\t{text_length}\t{number_of_sentences}\t{length_of_sentences}\t{title}\t{article_type}\t{article_category}\t{keywords}\n")
81
87
  item_count += 1
82
88
  f.close()
83
89
 
@@ -112,19 +118,17 @@ class TextIndexer:
112
118
  if logger != None: logger.info(f"The file {file} {file_exist}")
113
119
  parsed_texts, stats = TextPubmedAbstractParser.parse(file, logger= logger, options=options)
114
120
  for parsed_text in parsed_texts:
115
- pmid, text, year, title, article_type, article_category = parsed_text
116
- if pmid == None or text == "" or (len(text.split(" ")) < 10): continue
117
-
118
- #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
119
- if options['filter_by_blacklist'] != None:
120
- blacklisted_words = [word.strip() for word in open(options['filter_by_blacklist']).readlines()]
121
- filtered_out, w, c, t = TextIndexer._check_to_filter_out(blacklisted_words, title, article_category, options['blacklisted_mode'])
122
- if filtered_out:
123
- if logger != None: logger.warning(f"Blacklisted PMID {pmid} for having word {w} in {c} with content: {t}")
124
- continue
125
-
126
- pmid_content_and_stats = cls.prepare_indexes(text, pmid, file, year, title, article_type, article_category, options)
127
- texts.append(pmid_content_and_stats)
121
+ pmid, text, year, title, article_type, article_category, keywords = parsed_text
122
+ if pmid == "None": continue
123
+ if text == "None" and title == "None" and len(keywords) == 0: continue
124
+
125
+ if options['remain_empty_documents'] or text != "None":
126
+ if options['filter_by_blacklist'] != None: #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
127
+ is_blacklisted = TextIndexer._check_if_blacklisted_pmid(pmid, title, article_category, options, logger)
128
+ if is_blacklisted: continue
129
+
130
+ pmid_content_and_stats = cls.prepare_indexes(text, pmid, file, year, title, article_type, article_category, keywords, options)
131
+ texts.append(pmid_content_and_stats)
128
132
 
129
133
  if logger != None: logger.warning(f"stats:file={file},total={stats['total']},no_abstract={stats['no_abstract']},no_pmid={stats['no_pmid']}")
130
134
  return texts
@@ -139,54 +143,58 @@ class TextIndexer:
139
143
  if options["equivalences_file"] != None: PMC_PMID_dict = dict(CmdTabs.load_input_data(options["equivalences_file"]))
140
144
  texts = [] # aggregate all papers in XML file (Technically it is just one for each xml file, but it is emulating the original abstract part logic)
141
145
  for parsed_text in parsed_texts:
142
- pmid, pmc, filename, year, whole_content, title, article_type, article_category = parsed_text
143
- if pmid == None and PMC_PMID_dict != None: pmid = PMC_PMID_dict.get(pmc)
146
+ pmid, pmc, filename, year, whole_content, title, article_type, article_category, keywords = parsed_text
147
+ pmid = cls._set_alternative_pmid_if_needed_and_available(pmid, pmc, PMC_PMID_dict, file_path, logger)
144
148
 
145
- #If there is no pmid in the paper or in the PMC_PMID_dict, we will use the pmc as the document identifier, specifying it is PMC in the index.
146
- #If not PMC nor PMID, we will warn the user
147
- if pmid == None and pmc == None: logger.warning(f"ERROR: The file {file_path} has a paper without PMID and PMC")
148
- elif pmid == None and pmc != None: pmid = pmc
149
+ if pmid == "None": continue
150
+ if whole_content == "None" and title == "None" and len(keywords) == 0: continue
149
151
 
150
- #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
151
- if options['filter_by_blacklist'] != None:
152
- if os.path.exists(options['filter_by_blacklist']):
153
- blacklisted_words = [word.strip() for word in open(options['filter_by_blacklist']).readlines()]
154
- filtered_out, w, c, t = TextIndexer._check_to_filter_out(blacklisted_words, title, article_category, options['blacklisted_mode'])
155
- if filtered_out:
156
- if logger != None: logger.warning(f"Blacklisted PMID {pmid} for having word {w} in {c} with content: {t}")
157
- continue
158
- else:
159
- raise Exception("Blacklisted words filepath given does not exist")
160
-
161
- if pmid != None and whole_content != "" and len(whole_content) > 1000:
152
+ if options['remain_empty_documents'] or (whole_content != "None"):
153
+ if options['filter_by_blacklist'] != None: #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
154
+ is_blacklisted = TextIndexer._check_if_blacklisted_pmid(pmid, title, article_category, options, logger)
155
+ if is_blacklisted: continue
156
+
162
157
  #pmid_content_and_stats = cls.prepare_indexes(whole_content, pmc+"-"+pmid, filename, year, options)
163
- pmid_content_and_stats = cls.prepare_indexes(whole_content, pmid, filename, year, title, article_type, article_category, options)
158
+ pmid_content_and_stats = cls.prepare_indexes(whole_content, pmid, filename, year, title, article_type, article_category, keywords, options)
164
159
  texts.append(pmid_content_and_stats)
165
160
 
166
161
  if logger != None: logger.warning(f"stats:file={file_path},total={stats['total']},no_abstract={stats['no_abstract']},no_pmid={stats['no_pmid']}")
167
162
  if logger != None: logger.warning(f"logs_errors:file={file_path},errors_number={stats['errors']}")
168
163
  return texts
169
-
164
+
170
165
 
171
166
  @classmethod
172
- def prepare_indexes(cls, text, pmid, file, year, title, article_type, article_category, options):
167
+ def prepare_indexes(cls, text, pmid, file, year, title, article_type, article_category, keywords, options):
168
+ extra_fields = options.get("extra_fields", defaultdict(lambda: False))
173
169
  pmid = pmid.replace("\n", "")
174
170
  file = file.replace("\n", "")
175
171
  year = str(year).replace("\n", "")
176
-
177
172
  document_length = str(len(text))
178
- if options["split"]:
179
- document_parts = cls.split_document(text, pmid, split_type = options["split_type"])
173
+
174
+ cleaned_text = TextPubmedParser.perform_soft_cleaning(text, type=options["clean_type"])
175
+ if text == "None":
176
+ doc_to_jsonify = []
177
+ document_length = "0"
178
+ number_of_sentences = "0"
179
+ length_of_sentences = "0"
180
+ elif options["split"]:
181
+ document_parts = cls.split_document(cleaned_text, pmid, split_type = options["split_type"])
182
+ doc_to_jsonify = document_parts
180
183
  flattened_document = flatten(document_parts)
181
184
  number_of_sentences = str(len(flattened_document))
182
185
  length_of_sentences = ",".join([str(len(sentence)) for sentence in flattened_document])
183
- document_parts_json = json.dumps(document_parts)
184
- prepared_index = [pmid, document_parts_json, file, year, document_length, number_of_sentences, length_of_sentences,
185
- title, article_type, article_category]
186
186
  else:
187
- cleaned_document = TextPubmedParser.perform_soft_cleaning(text)
188
- prepared_index = [pmid, cleaned_document, file, year, document_length, "1", document_length,
189
- title, article_type, article_category]
187
+ number_of_sentences = "1"
188
+ length_of_sentences = document_length
189
+ doc_to_jsonify = [[cleaned_text]]
190
+
191
+ keywords_to_add = keywords if len(keywords) > 0 else ["None"]
192
+ if extra_fields["keywords"]: doc_to_jsonify.insert(0, ["KEYWORDS"] + keywords_to_add)
193
+ if extra_fields["title"]: doc_to_jsonify.insert(0, ["TITLE", title])
194
+
195
+ document_json = json.dumps(doc_to_jsonify)
196
+ prepared_index = [pmid, document_json, file, year, document_length, number_of_sentences, length_of_sentences,
197
+ title, article_type, article_category, ",".join(keywords_to_add)]
190
198
  return prepared_index
191
199
 
192
200
 
@@ -198,25 +206,54 @@ class TextIndexer:
198
206
  if split_type == "classic":
199
207
  sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 10, chunk_overlap = 0, length_function = len, separators=["\n",".", ";", ","], keep_separator=False, is_separator_regex=False)
200
208
  elif split_type == "space_overlap":
201
- sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 60, chunk_overlap = 20, length_function = len, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
209
+ sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 100, chunk_overlap = 20, length_function = len, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
210
+ #sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 20, chunk_overlap = 2, length_function = cls._len_by_words_func, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
202
211
  #sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 120, chunk_overlap = 20, length_function = len, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
203
212
  paragraphs = [paragraph.strip() for paragraph in paragraph_splitter.split_text(text)]
204
- sentences = [ sentences_splitter.split_text(paragraph.replace("\n", "")) for paragraph in paragraphs ]
205
- sentences = [ list(map(lambda sentence: sentence.strip(), paragraph)) for paragraph in sentences ]
213
+ paragraph_sentences = [ sentences_splitter.split_text(paragraph) for paragraph in paragraphs ]
206
214
 
207
215
  formated = []
208
- for paragraph in sentences:
216
+ for paragraph in paragraph_sentences:
209
217
  formated_paragraph = []
210
218
  for sentence in paragraph:
211
219
  if len(sentence) > 2:
212
220
  if len(sentence) > 3000: warnings.warn(f"ERROR: The pmid {pmid} has an unusual sentence lenght even after splitting. Total length of characters: {len(sentence)}. Content: {sentence}")
213
- else: formated_paragraph.append(sentence)
221
+ else: formated_paragraph.append(sentence.strip())
214
222
  if len(formated_paragraph) > 0: formated.append(formated_paragraph)
215
223
 
216
224
  return formated
217
225
 
218
226
 
219
227
  ######### UTILS METHODS
228
+ @classmethod
229
+ def _len_by_words_func(cls, text):
230
+ return text.count(" ")
231
+
232
+ @classmethod
233
+ def _set_alternative_pmid_if_needed_and_available(cls, pmid, pmc, PMC_PMID_dict, file_path, logger):
234
+ #If there is no pmid in the paper or in the PMC_PMID_dict, we will use the pmc as the document identifier, specifying it is PMC in the index.
235
+ #If not PMC nor PMID, we will warn the user
236
+ if pmid == "None" and PMC_PMID_dict != None:
237
+ pmid = PMC_PMID_dict.get(pmc, "None")
238
+ if pmid == "None" and pmc == "None":
239
+ logger.warning(f"ERROR: The file {file_path} has a paper without PMID and PMC")
240
+ elif pmid == "None" and pmc != "None":
241
+ pmid = pmc
242
+ return pmid
243
+
244
+ @classmethod
245
+ def _check_if_blacklisted_pmid(cls, pmid, title, article_category, options, logger):
246
+ if os.path.exists(options['filter_by_blacklist']):
247
+ is_blacklisted = False
248
+ blacklisted_words = [word.strip() for word in open(options['filter_by_blacklist']).readlines()]
249
+ filtered_out, w, c, t = TextIndexer._check_to_filter_out(blacklisted_words, title, article_category, options['blacklisted_mode'])
250
+ if filtered_out:
251
+ if logger != None: logger.warning(f"Blacklisted PMID {pmid} for having word {w} in {c} with content: {t}")
252
+ is_blacklisted = True
253
+ return is_blacklisted
254
+ else:
255
+ raise Exception("Blacklisted words filepath given does not exist")
256
+
220
257
 
221
258
  @classmethod
222
259
  def _check_to_filter_out(cls, blacklisted_words, title, article_category, mode):