py-semtools 1.3__tar.gz → 1.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {py_semtools-1.3 → py_semtools-1.3.3}/PKG-INFO +6 -3
- py_semtools-1.3.3/README.md +6 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/setup.cfg +11 -2
- py_semtools-1.3.3/src/py_semtools/__init__.py +16 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/cli_manager.py +82 -16
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/indexers/text_indexer.py +91 -54
- py_semtools-1.3.3/src/py_semtools/lexical_engines/engine_baseclass.py +156 -0
- py_semtools-1.3.3/src/py_semtools/lexical_engines/fmEngine.py +99 -0
- {py_semtools-1.3/src/py_semtools → py_semtools-1.3.3/src/py_semtools/lexical_engines}/stEngine.py +83 -145
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/main_modules.py +81 -29
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/ontology.py +120 -89
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parallelizer.py +0 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/json_parser.py +1 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/oboparser.py +1 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_abstract_parser.py +20 -15
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_paper_parser.py +23 -19
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/report_ont.py +1 -2
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/sim_handler.py +0 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontoplot.txt +1 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/PKG-INFO +6 -3
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/SOURCES.txt +4 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/entry_points.txt +1 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/requires.txt +6 -2
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/Report.html +221 -80
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/launch.sh +1 -1
- py_semtools-1.3.3/tests/cli_examples/styles.css +130 -0
- py_semtools-1.3.3/tests/cli_examples/template.txt +84 -0
- py_semtools-1.3.3/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981088 +1 -0
- py_semtools-1.3.3/tests/data/stEngine/expected/raw_indexes/papers/chunk1/38108203 +1 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/launch_plots.py +3 -3
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_cli_manager.py +26 -27
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_go.py +2 -3
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_jsonparser.py +3 -5
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_oboparser.py +3 -4
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_ontology.py +3 -1
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_abs_parser.py +15 -15
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_pap_parser.py +38 -10
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_pubmed_text_parser.py +2 -4
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_similitudes.py +1 -4
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_stengine.py +2 -7
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/test_text_indexer.py +45 -36
- {py_semtools-1.3 → py_semtools-1.3.3}/tox.ini +2 -0
- py_semtools-1.3/README.md +0 -0
- py_semtools-1.3/src/py_semtools/__init__.py +0 -12
- py_semtools-1.3/tests/cli_examples/template.txt +0 -119
- py_semtools-1.3/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981088 +0 -0
- py_semtools-1.3/tests/data/stEngine/expected/raw_indexes/papers/chunk1/38108203 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/.coveragerc +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/.gitignore +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/.readthedocs.yml +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/AUTHORS.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/CHANGELOG.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/CONTRIBUTING.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/LICENSE.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/MANIFEST.in +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/README.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/Makefile +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/_static/.gitignore +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/authors.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/changelog.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/conf.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/contributing.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/index.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/license.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/readme.rst +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/docs/requirements.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/pyproject.toml +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/setup.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/external_data/ontologies.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/ont/file_parser.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_basic_parser.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/parsers/text/text_pubmed_parser.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/comorb_sugg.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/makeTermFreqTable.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontoICdist.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/ontodist.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/plotClust.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/plotProfRed.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/report.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/similarity_heatmap.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/similarity_matrix.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools/templates/stEngine.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/dependency_links.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/not-zip-safe +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/src/py_semtools.egg-info/top_level.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/__init__.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/cli_examples/example +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/conftest.py +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/branched.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/circular_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology2.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/enrichment_ontology3.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_profs/report.html +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_profs/report.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_query_parentals.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_target_and_query_parentals.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/filter_target_parentals.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/no_filter_limit_2.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/expected/no_filter_no_limit.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/enrichment_ontology3.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/query_hps.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/get_sorted_suggestions/input_data/relations.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/go-basic_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/go-onlyOne.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_alt_chain.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_compressed.obo.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/hierarchical_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles_2cols +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/profiles_with_removedTerms +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/string_values +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms_for_xref +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/input_scripts/terms_list +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/only_header_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/partial_go.json +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/partial_go.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/blacklisted.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/cleaned_profiles +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/cleaned_profiles_2cols +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/expanded_profiles +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/expected_IC_ont +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/parental_from_terms +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/profile_stats +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/profiles_IC_onto_freq +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/strsimnet +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/strsimnet_cutoff2 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/terms_attr +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/translated_profiles_names +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/translated_terms_codes +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/white_and_blacklisted.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/ref_output_scripts/whitelisted.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/short_hierarchical_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/sparse2_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/sparse_sample.obo +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981089 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981091 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/22981092 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/chunk1/None +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/abstracts/single_abs_text +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/18382669 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/31356151 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/35042469 +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/chunk1/None +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/raw_indexes/papers/single_pap_text +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/abstracts/single_abs_index_nonsplit +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/abstracts/single_abs_index_split +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/papers/single_pap_index_nonsplit +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/expected/ready_indexes/papers/single_pap_index_split +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/abs_chunk1.xml.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/abs_chunk2.xml.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/single_abstract.xml +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/abstracts/single_abstract.xml.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words1.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words2.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/blacklisted_words3.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/example.xml +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/example_zipped.xml.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/pap_chunk1.tar.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/pap_chunk2.tar.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/single_paper.tar.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/papers/single_paper.xml +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/abstracts/example1.txt.gz +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/queries/hpo_list +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/stEngine/inputs/prepared_indexes/queries/mondo_list +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/st_engine_report/abs_profiles.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/data/st_engine_report/pmids_and_titles.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/data/profiles.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/data/ref_profile.txt +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/launch.sh +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/profiles_freqs.html +0 -0
- {py_semtools-1.3 → py_semtools-1.3.3}/tests/demo_examples/template.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: py_semtools
|
|
3
|
-
Version: 1.3
|
|
3
|
+
Version: 1.3.3
|
|
4
4
|
Summary: Library to handle ontologies that allows queries and calculations such as information coefficients, semantic similarity, ontology representations, etc in a easy way. It can load any ontology that complies the obo format supported by OBO Foundry.
|
|
5
5
|
Home-page: https://github.com/seoanezonjic/py_semtools
|
|
6
6
|
Author: seoanezonjic
|
|
@@ -18,8 +18,6 @@ Requires-Dist: numpy
|
|
|
18
18
|
Requires-Dist: networkx
|
|
19
19
|
Requires-Dist: entrezpy
|
|
20
20
|
Requires-Dist: scipy
|
|
21
|
-
Requires-Dist: sentence_transformers
|
|
22
|
-
Requires-Dist: langchain
|
|
23
21
|
Requires-Dist: pynvml
|
|
24
22
|
Requires-Dist: requests
|
|
25
23
|
Requires-Dist: py_exp_calc
|
|
@@ -27,6 +25,11 @@ Requires-Dist: py_cmdtabs
|
|
|
27
25
|
Requires-Dist: py_report_html
|
|
28
26
|
Requires-Dist: loguru
|
|
29
27
|
Requires-Dist: loky
|
|
28
|
+
Requires-Dist: bm25s
|
|
29
|
+
Provides-Extra: models
|
|
30
|
+
Requires-Dist: sentence_transformers; extra == "models"
|
|
31
|
+
Requires-Dist: langchain; extra == "models"
|
|
32
|
+
Requires-Dist: langchain-text-splitters; extra == "models"
|
|
30
33
|
Provides-Extra: testing
|
|
31
34
|
Requires-Dist: setuptools; extra == "testing"
|
|
32
35
|
Requires-Dist: pytest; extra == "testing"
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
If you want to use lexical models capabilities (stEngine and fmEngine binaries), you should install extra required dependencies that are not installed
|
|
2
|
+
with the base installation. For that end, you can install semtools like the following:
|
|
3
|
+
|
|
4
|
+
```bash
|
|
5
|
+
pip install py_semtools[MODELS]
|
|
6
|
+
```
|
|
@@ -28,8 +28,6 @@ install_requires =
|
|
|
28
28
|
networkx
|
|
29
29
|
entrezpy
|
|
30
30
|
scipy
|
|
31
|
-
sentence_transformers
|
|
32
|
-
langchain
|
|
33
31
|
pynvml
|
|
34
32
|
requests
|
|
35
33
|
py_exp_calc
|
|
@@ -37,6 +35,7 @@ install_requires =
|
|
|
37
35
|
py_report_html
|
|
38
36
|
loguru
|
|
39
37
|
loky
|
|
38
|
+
bm25s
|
|
40
39
|
|
|
41
40
|
[options.packages.find]
|
|
42
41
|
where = src
|
|
@@ -50,6 +49,7 @@ py_semtools.templates =
|
|
|
50
49
|
*.txt
|
|
51
50
|
|
|
52
51
|
[options.extras_require]
|
|
52
|
+
MODELS = sentence_transformers; langchain; langchain-text-splitters
|
|
53
53
|
testing =
|
|
54
54
|
setuptools
|
|
55
55
|
pytest
|
|
@@ -62,11 +62,13 @@ console_scripts =
|
|
|
62
62
|
get_sorted_suggestions = py_semtools.cli_manager:get_sorted_suggestions
|
|
63
63
|
get_sorted_profs = py_semtools.cli_manager:get_sorted_profs
|
|
64
64
|
remote_retriever = py_semtools.cli_manager:remote_retriever
|
|
65
|
+
fmEngine = py_semtools.cli_manager:fmEngine
|
|
65
66
|
stEngine = py_semtools.cli_manager:stEngine
|
|
66
67
|
stEngine_report = py_semtools.cli_manager:stEngine_report
|
|
67
68
|
get_corpus_index = py_semtools.cli_manager:get_corpus_index
|
|
68
69
|
|
|
69
70
|
[tool:pytest]
|
|
71
|
+
python_files = tests.py test_*.py *_tests.py
|
|
70
72
|
addopts =
|
|
71
73
|
--cov py_semtools --cov-report term-missing
|
|
72
74
|
--verbose
|
|
@@ -74,6 +76,13 @@ norecursedirs =
|
|
|
74
76
|
dist
|
|
75
77
|
build
|
|
76
78
|
.tox
|
|
79
|
+
.git
|
|
80
|
+
docs
|
|
81
|
+
src/py_semtools/templates
|
|
82
|
+
tests/cli_examples
|
|
83
|
+
tests/data
|
|
84
|
+
tests/demo_examples
|
|
85
|
+
tests/__pycache__
|
|
77
86
|
testpaths = tests
|
|
78
87
|
|
|
79
88
|
[devpi:upload]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
|
|
3
|
+
if sys.version_info[:2] >= (3, 8):
|
|
4
|
+
# TODO: Import directly (no need for conditional) when `python_requires = >= 3.8`
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version # pragma: no cover
|
|
6
|
+
else:
|
|
7
|
+
from importlib_metadata import PackageNotFoundError, version # pragma: no cover
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
# Change here if project is renamed and does not equal the package name
|
|
11
|
+
dist_name = __name__
|
|
12
|
+
__version__ = version(dist_name)
|
|
13
|
+
except PackageNotFoundError: # pragma: no cover
|
|
14
|
+
__version__ = "unknown"
|
|
15
|
+
finally:
|
|
16
|
+
del version, PackageNotFoundError
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import argparse, re, sys, inspect
|
|
2
2
|
from py_semtools.main_modules import *
|
|
3
|
+
from collections import defaultdict
|
|
4
|
+
|
|
3
5
|
|
|
4
6
|
###########################################################################
|
|
5
7
|
## TYPES
|
|
@@ -9,6 +11,12 @@ def one_column_file(file): return [line.strip() for line in open(file).readlines
|
|
|
9
11
|
|
|
10
12
|
def text_list(string): return string.split(',')
|
|
11
13
|
|
|
14
|
+
def text_to_default_dict(string):
|
|
15
|
+
d = defaultdict(lambda: False)
|
|
16
|
+
for item in text_list(string):
|
|
17
|
+
d[item] = True
|
|
18
|
+
return d
|
|
19
|
+
|
|
12
20
|
def filter_regex(string):
|
|
13
21
|
filters = []
|
|
14
22
|
pattern = re.compile(r"([pn])\(([A-Za-z:0-9,]*)\)")
|
|
@@ -73,8 +81,8 @@ def remote_retriever(args = None):
|
|
|
73
81
|
main_remote_retriever(opts)
|
|
74
82
|
|
|
75
83
|
def semtools(args = None):
|
|
76
|
-
|
|
77
|
-
|
|
84
|
+
|
|
85
|
+
if args is None: args = sys.argv[1:]
|
|
78
86
|
|
|
79
87
|
parser = argparse.ArgumentParser(description='Perform Ontology driven analysis ')
|
|
80
88
|
parser.add_argument("--return_all_terms_with_user_defined_attributes", dest="return_all_terms_with_user_defined_attributes", default=None,
|
|
@@ -98,10 +106,10 @@ def semtools(args = None):
|
|
|
98
106
|
parser.add_argument("-t", "--translate", dest="translate", default= None,
|
|
99
107
|
help="Translate to 'names' or to 'codes'")
|
|
100
108
|
parser.add_argument("-s", "--similarity_method", dest="similarity", default= None,
|
|
101
|
-
help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods.")
|
|
109
|
+
help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods. Recently added 'eric', 'neric', 'nweric' and 'erlin' methods.")
|
|
102
110
|
parser.add_argument("--reference_profiles", dest="reference_profiles", default= None,
|
|
103
111
|
help="Path to file tabulated file with first column as id profile and second column with ontology terms separated by separator.")
|
|
104
|
-
parser.add_argument(
|
|
112
|
+
parser.add_argument("--disable_cleaning_profiles", dest="clean_profiles", default= True, action='store_false',
|
|
105
113
|
help="Removes ancestors, descendants and obsolete terms from profiles.")
|
|
106
114
|
parser.add_argument('-r', "--removed_path", dest="removed_path", default= 'rejected_profs',
|
|
107
115
|
help="Desired path to write removed profiles file.")
|
|
@@ -157,6 +165,8 @@ def semtools(args = None):
|
|
|
157
165
|
help="For the output report, it sets the root term to show in center of ontoplot")
|
|
158
166
|
parser.add_argument("--ref_term", dest="ref_term", default=None,
|
|
159
167
|
help="For the output report, it sets the term whose childs will be in the color legend to indicate the branches of descendants")
|
|
168
|
+
parser.add_argument("--similarity_cluster", dest="similarity_cluster", default=None,
|
|
169
|
+
help="Use to activate profiles clustering with a similarity method and generate basefiles (resnik', 'lin' or 'jiang_conrath). Not active by default")
|
|
160
170
|
parser.add_argument("--similarity_cluster_plot", dest="similarity_cluster_plot", default=None,
|
|
161
171
|
help="For output report, use to activate profiles clustering and clustermap plot with a similarity method (resnik', 'lin' or 'jiang_conrath). Not active by default")
|
|
162
172
|
parser.add_argument("--cl_size_factor", dest="cl_size_factor", default=1.0, type=float,
|
|
@@ -167,9 +177,22 @@ def semtools(args = None):
|
|
|
167
177
|
help="For the ontoplot, use it to deactivate propagating frequency to parentals terms")
|
|
168
178
|
parser.add_argument("--onto_min_freq", dest="onto_min_freq", default=0.005, type=float,
|
|
169
179
|
help="For the ontoplot, it sets the minimum frequency of terms to be shown in the plot. Default is 0.5 percent. Set to 0 to show all terms.")
|
|
180
|
+
parser.add_argument("--similarity_index", dest="similarity_index", default=None,
|
|
181
|
+
help="Term to term similarity index to perform profile similarity calculations")
|
|
182
|
+
parser.add_argument("--get_LCA_from_profile", dest="get_LCA_from_profile", default=None, type=float,
|
|
183
|
+
help="For a list if profiles, give for each one the list of LCA that occurs for the selected frequency in the profile terms.")
|
|
184
|
+
parser.add_argument("--get_MICA_from_profile", dest="get_MICA_from_profile", default=None, type=float,
|
|
185
|
+
help="For a list of profiles, give for each one the list of LCA that occurs for the selected frequency in the profile terms.")
|
|
186
|
+
parser.add_argument("--LCA_list", dest="LCA_list", default=None, type=str,
|
|
187
|
+
help="With flags 'get_LCA_from_profile' and 'get_MICA_from_profile' constraint the ancestors to the given list of terms (one per line).")
|
|
188
|
+
parser.add_argument("--get_representative_profile", dest="get_representative_profile", default=None, type=float,
|
|
189
|
+
help="For a list of profiles, give one single profile with terms presented with al least FLOAT frecuency in the profile list. Parents are infered and computed too. The most spefic terms are retained.")
|
|
190
|
+
|
|
170
191
|
opts = parser.parse_args(args)
|
|
192
|
+
|
|
171
193
|
main_semtools(opts)
|
|
172
194
|
|
|
195
|
+
|
|
173
196
|
def get_sorted_suggestions(args = None):
|
|
174
197
|
if args is None:
|
|
175
198
|
args = sys.argv[1:]
|
|
@@ -210,6 +233,37 @@ def get_sorted_suggestions(args = None):
|
|
|
210
233
|
opts = parser.parse_args(args)
|
|
211
234
|
main_get_sorted_suggestions(opts)
|
|
212
235
|
|
|
236
|
+
|
|
237
|
+
def fmEngine(args = None):
|
|
238
|
+
if args is None:
|
|
239
|
+
args = sys.argv[1:]
|
|
240
|
+
|
|
241
|
+
parser = argparse.ArgumentParser(description='Perform Ontology driven analysis with Sentence Transformer')
|
|
242
|
+
|
|
243
|
+
parser.add_argument('-m', "--model_name", dest="model_name", default= None,
|
|
244
|
+
help="Model to be used. Current options include 'bm25' and 'tfidf'")
|
|
245
|
+
parser.add_argument('-q', "--query", dest="query", default= None,
|
|
246
|
+
help="Path to the query file. Wildcards are accepted to process multiple files")
|
|
247
|
+
parser.add_argument('-c', "--corpus", dest="corpus", default= None,
|
|
248
|
+
help="Path to the corpus file. Wildcards are accepted to process multiple files")
|
|
249
|
+
parser.add_argument('-o', "--output_file", dest="output_file", default= None,
|
|
250
|
+
help="Path to save the output file with semantic scores")
|
|
251
|
+
parser.add_argument('-k', "--top_k", dest="top_k", default= 20, type = int,
|
|
252
|
+
help="Get top scores per keyword")
|
|
253
|
+
parser.add_argument('-t', "--threshold", dest="threshold", default= 0, type = float,
|
|
254
|
+
help="Similarity threshold to filter results to write")
|
|
255
|
+
parser.add_argument('-v', "--verbose", dest="verbose", default= False, action='store_true',
|
|
256
|
+
help="Toogle on to get verbose output")
|
|
257
|
+
parser.add_argument("--order", dest="order", default= "corpus-query",
|
|
258
|
+
help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
|
|
259
|
+
parser.add_argument('-s', "--split", dest="split", default= False, action='store_true',
|
|
260
|
+
help="Use it if your corpus comes splitted in smaller parts as list of lists (embedded in a json)")
|
|
261
|
+
parser.add_argument("--print_relevant_pairs", dest="print_relevant_pairs", default= False, action='store_true',
|
|
262
|
+
help="Use it to print the relevant pairs of query-corpus with their scores")
|
|
263
|
+
opts = parser.parse_args(args)
|
|
264
|
+
main_fmEngine(opts)
|
|
265
|
+
|
|
266
|
+
|
|
213
267
|
def stEngine(args = None):
|
|
214
268
|
if args is None:
|
|
215
269
|
args = sys.argv[1:]
|
|
@@ -218,6 +272,8 @@ def stEngine(args = None):
|
|
|
218
272
|
|
|
219
273
|
parser.add_argument('-m', "--model_name", dest="model_name", default= None,
|
|
220
274
|
help="Name of the model to be used")
|
|
275
|
+
parser.add_argument('-r', "--reranker_model_name", dest="reranker_model_name", default= None,
|
|
276
|
+
help="(Optional) Name of the reranker model to be used")
|
|
221
277
|
parser.add_argument('-p', "--model_path", dest="model_path", default= None,
|
|
222
278
|
help="Path where the model is cached or where it will be stored")
|
|
223
279
|
parser.add_argument('-q', "--query", dest="query", default= None,
|
|
@@ -235,21 +291,25 @@ def stEngine(args = None):
|
|
|
235
291
|
parser.add_argument('-t', "--threshold", dest="threshold", default= 0, type = float,
|
|
236
292
|
help="Similarity threshold to filter results to write")
|
|
237
293
|
parser.add_argument('-v', "--verbose", dest="verbose", default= False, action='store_true',
|
|
238
|
-
help="Toogle on to get verbose output")
|
|
239
|
-
parser.add_argument('-g', "--gpu_device", dest="gpu_device", default= [], type=text_list,
|
|
240
|
-
help="Use to specify the GPU device to be used for speed-up (if available). The format is like: 'cuda:0' or 'cuda:0,cuda:1' to use multiple GPUs or cpu,cpu to use multiple CPUs")
|
|
241
|
-
parser.add_argument("-b", "--batch_size", dest="batch_size", default= 32, type=int,
|
|
242
|
-
help="Use to specify batch size for multi-GPU embedding step")
|
|
243
|
-
parser.add_argument("--use_gpu_for_sim_calculation", dest="use_gpu_for_sim_calculation", default= False, action='store_true',
|
|
244
|
-
help="Toogle on if you want to use GPU not only for the embedding process, but also for calculating query-corpus similarities (then dot score if used instead of cosine similarity)")
|
|
294
|
+
help="Toogle on to get verbose output")
|
|
245
295
|
parser.add_argument('-s', "--split", dest="split", default= False, action='store_true',
|
|
246
296
|
help="Use it if your corpus comes splitted in smaller parts as list of lists (embedded in a json)")
|
|
247
297
|
parser.add_argument("--order", dest="order", default= "corpus-query",
|
|
248
|
-
help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
|
|
249
|
-
parser.add_argument("--chunk_size", dest="chunk_size", default=10000, type=int,
|
|
250
|
-
help="Size to be accumulating corpora until a threshold is reached before proceeding to embedd")
|
|
298
|
+
help="Order of the semantic search. Options: 'corpus-query' or 'query-corpus'")
|
|
251
299
|
parser.add_argument("--print_relevant_pairs", dest="print_relevant_pairs", default= False, action='store_true',
|
|
252
|
-
help="Use it to print the relevant pairs of query-corpus with their scores")
|
|
300
|
+
help="Use it to print the relevant pairs of query-corpus with their scores")
|
|
301
|
+
parser.add_argument('-g', "--gpu_device", dest="gpu_device", default= [], type=text_list,
|
|
302
|
+
help="Use to specify the GPU device to be used for speed-up (if available). The format is like: 'cuda:0' or 'cuda:0,cuda:1' to use multiple GPUs or cpu,cpu to use multiple CPUs")
|
|
303
|
+
parser.add_argument("--use_gpu_for_sim_calculation", dest="use_gpu_for_sim_calculation", default= False, action='store_true',
|
|
304
|
+
help="Toogle on if you want to use GPU not only for the embedding process, but also for calculating query-corpus similarities (then dot score if used instead of cosine similarity)")
|
|
305
|
+
parser.add_argument("--chunk_size", dest="chunk_size", default=10000, type=int,
|
|
306
|
+
help="Size to be accumulating corpora (documents) until a threshold is reached before proceeding to embedd")
|
|
307
|
+
parser.add_argument("--chunk_size_sentences", dest="chunk_size_sentences", default=0, type=int,
|
|
308
|
+
help="(If toogled on, takes precedence over --chunk_size) Size to be accumulating sentences until a threshold is reached before proceeding to embedd")
|
|
309
|
+
parser.add_argument("--single_worker_chunk_size", dest="single_worker_chunk_size", default=10000, type=int,
|
|
310
|
+
help="(Only for multi GPU processing) Chunk size for each worker when multi-GPU processing is used.")
|
|
311
|
+
parser.add_argument("-b", "--batch_size", dest="batch_size", default= 32, type=int,
|
|
312
|
+
help="Use to specify batch size for multi-GPU embedding step")
|
|
253
313
|
opts = parser.parse_args(args)
|
|
254
314
|
main_stEngine(opts)
|
|
255
315
|
|
|
@@ -328,6 +388,10 @@ def get_corpus_index(args = None):
|
|
|
328
388
|
help="Use to set the type of cleaning to be performed on the text. Options: basic (do not clean at all), soft or hard (default).")
|
|
329
389
|
parser.add_argument("--split_type", dest="split_type", default= 'classic',
|
|
330
390
|
help="Use to set the type of splitting to be performed on the text. Options: classic (default, splitting by '\n\n', \n', '.', ';', ',') or space_overlap ('\n\n', '\n', ' ', '' with overlap to avoid losing information).")
|
|
391
|
+
parser.add_argument("--extra_fields", dest="extra_fields", default= defaultdict(lambda: False), type=text_to_default_dict,
|
|
392
|
+
help="Comma-separated list of extra fields to include in the processed corpus. Available: title and/or keywords. Default: none")
|
|
393
|
+
parser.add_argument("--remain_empty_documents", dest="remain_empty_documents", default= False, action='store_true',
|
|
394
|
+
help="Use this option to keep entries with (main text) empty documents in the processed corpus (if they have at least title or keywords)")
|
|
331
395
|
opts = parser.parse_args(args)
|
|
332
396
|
main_get_corpus_index(opts)
|
|
333
397
|
|
|
@@ -358,7 +422,9 @@ def get_sorted_profs(args=None):
|
|
|
358
422
|
parser.add_argument("-X", "--excluded_terms", dest="excluded_terms", default= [], type = one_column_file,
|
|
359
423
|
help="File with excluded terms. One term code per line.")
|
|
360
424
|
parser.add_argument("--hard_check", dest="hard_check", default= True, action="store_false",
|
|
361
|
-
help="Set to disable hard check cleaning. Default true")
|
|
425
|
+
help="Set to disable hard check cleaning. Default true")
|
|
426
|
+
parser.add_argument("-s", "--similarity_method", dest="similarity", default= 'lin',
|
|
427
|
+
help="Calculate similarity between profile IDs computed by 'resnik', 'lin' or 'jiang_conrath' methods. Recently added 'eric', 'neric', 'nweric' and 'erlin' methods.")
|
|
362
428
|
parser.add_argument("-o", "--output_file", dest="output_file", default= 'report.html',
|
|
363
429
|
help="Output report file")
|
|
364
430
|
parser.add_argument("-f", "--general_prof_freq", dest="term_freq", default= 0, type= float,
|
|
@@ -1,8 +1,9 @@
|
|
|
1
|
+
from collections import defaultdict
|
|
1
2
|
import sys, os, glob, re, warnings, json, gzip
|
|
2
|
-
from
|
|
3
|
-
from py_exp_calc.exp_calc import
|
|
3
|
+
from langchain_text_splitters import RecursiveCharacterTextSplitter
|
|
4
|
+
from py_exp_calc.exp_calc import flatten
|
|
4
5
|
|
|
5
|
-
from py_cmdtabs import CmdTabs
|
|
6
|
+
from py_cmdtabs.cmdtabs import CmdTabs
|
|
6
7
|
from py_semtools.parallelizer import Parallelizer
|
|
7
8
|
from py_semtools.parsers.text.text_basic_parser import TextBasicParser
|
|
8
9
|
from py_semtools.parsers.text.text_pubmed_abstract_parser import TextPubmedAbstractParser
|
|
@@ -31,7 +32,11 @@ class TextIndexer:
|
|
|
31
32
|
else:
|
|
32
33
|
chunks = manager.get_chunks(filenames, workload_balance='disperse_max', workload_function= lambda filename: os.stat(filename).st_size)
|
|
33
34
|
items = [[[options, chunk, idx], {}] for idx,chunk in enumerate(chunks)]
|
|
34
|
-
manager.execute(items, cls.process_files)
|
|
35
|
+
res_status= manager.execute(items, cls.process_files)
|
|
36
|
+
if set(list(res_status)) == {True}:
|
|
37
|
+
print("All processes finished succesfully")
|
|
38
|
+
else:
|
|
39
|
+
print("Some processes finished with errors")
|
|
35
40
|
|
|
36
41
|
@classmethod
|
|
37
42
|
def process_files(cls, options, filenames, sup_counter, logger = None):
|
|
@@ -54,6 +59,7 @@ class TextIndexer:
|
|
|
54
59
|
counter += 1
|
|
55
60
|
# For records saved in accumulated_texts that the loop has not writed
|
|
56
61
|
cls.write_indexes(accumulated_texts, options["output"], options["tag"], a_suffix=sup_counter, b_suffix=counter, balancer = balancer, split_output_files = options["split_output_files"], items_per_file = options["text_balancing_size"])
|
|
62
|
+
return True
|
|
57
63
|
|
|
58
64
|
@classmethod
|
|
59
65
|
def write_indexes(cls, indexes, folder, name, a_suffix=None, b_suffix=None, balancer = None, split_output_files = False, items_per_file = 0):
|
|
@@ -71,13 +77,13 @@ class TextIndexer:
|
|
|
71
77
|
f = gzip.open(out_filename, 'wt')
|
|
72
78
|
item_count = 0
|
|
73
79
|
while indexes:
|
|
74
|
-
pmid, text, original_filename, year, text_length, number_of_sentences, length_of_sentences, title, article_type, article_category = indexes.pop()
|
|
80
|
+
pmid, text, original_filename, year, text_length, number_of_sentences, length_of_sentences, title, article_type, article_category, keywords = indexes.pop()
|
|
75
81
|
if split_output_files and item_count >= items_per_file:
|
|
76
82
|
f.close()
|
|
77
83
|
file_count += 1
|
|
78
84
|
item_count = 0
|
|
79
85
|
f = gzip.open(os.path.join(folder, name+f"{a_suffix}{b_suffix}_{file_count}.gz" ), 'wt')
|
|
80
|
-
f.write(f"{pmid}\t{text}\t{original_filename}\t{year}\t{text_length}\t{number_of_sentences}\t{length_of_sentences}\t{title}\t{article_type}\t{article_category}\n")
|
|
86
|
+
f.write(f"{pmid}\t{text}\t{original_filename}\t{year}\t{text_length}\t{number_of_sentences}\t{length_of_sentences}\t{title}\t{article_type}\t{article_category}\t{keywords}\n")
|
|
81
87
|
item_count += 1
|
|
82
88
|
f.close()
|
|
83
89
|
|
|
@@ -112,19 +118,17 @@ class TextIndexer:
|
|
|
112
118
|
if logger != None: logger.info(f"The file {file} {file_exist}")
|
|
113
119
|
parsed_texts, stats = TextPubmedAbstractParser.parse(file, logger= logger, options=options)
|
|
114
120
|
for parsed_text in parsed_texts:
|
|
115
|
-
pmid, text, year, title, article_type, article_category = parsed_text
|
|
116
|
-
if pmid == None
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
if options['
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
pmid_content_and_stats = cls.prepare_indexes(text, pmid, file, year, title, article_type, article_category, options)
|
|
127
|
-
texts.append(pmid_content_and_stats)
|
|
121
|
+
pmid, text, year, title, article_type, article_category, keywords = parsed_text
|
|
122
|
+
if pmid == "None": continue
|
|
123
|
+
if text == "None" and title == "None" and len(keywords) == 0: continue
|
|
124
|
+
|
|
125
|
+
if options['remain_empty_documents'] or text != "None":
|
|
126
|
+
if options['filter_by_blacklist'] != None: #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
|
|
127
|
+
is_blacklisted = TextIndexer._check_if_blacklisted_pmid(pmid, title, article_category, options, logger)
|
|
128
|
+
if is_blacklisted: continue
|
|
129
|
+
|
|
130
|
+
pmid_content_and_stats = cls.prepare_indexes(text, pmid, file, year, title, article_type, article_category, keywords, options)
|
|
131
|
+
texts.append(pmid_content_and_stats)
|
|
128
132
|
|
|
129
133
|
if logger != None: logger.warning(f"stats:file={file},total={stats['total']},no_abstract={stats['no_abstract']},no_pmid={stats['no_pmid']}")
|
|
130
134
|
return texts
|
|
@@ -139,54 +143,58 @@ class TextIndexer:
|
|
|
139
143
|
if options["equivalences_file"] != None: PMC_PMID_dict = dict(CmdTabs.load_input_data(options["equivalences_file"]))
|
|
140
144
|
texts = [] # aggregate all papers in XML file (Technically it is just one for each xml file, but it is emulating the original abstract part logic)
|
|
141
145
|
for parsed_text in parsed_texts:
|
|
142
|
-
pmid, pmc, filename, year, whole_content, title, article_type, article_category = parsed_text
|
|
143
|
-
|
|
146
|
+
pmid, pmc, filename, year, whole_content, title, article_type, article_category, keywords = parsed_text
|
|
147
|
+
pmid = cls._set_alternative_pmid_if_needed_and_available(pmid, pmc, PMC_PMID_dict, file_path, logger)
|
|
144
148
|
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
if pmid == None and pmc == None: logger.warning(f"ERROR: The file {file_path} has a paper without PMID and PMC")
|
|
148
|
-
elif pmid == None and pmc != None: pmid = pmc
|
|
149
|
+
if pmid == "None": continue
|
|
150
|
+
if whole_content == "None" and title == "None" and len(keywords) == 0: continue
|
|
149
151
|
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
if filtered_out:
|
|
156
|
-
if logger != None: logger.warning(f"Blacklisted PMID {pmid} for having word {w} in {c} with content: {t}")
|
|
157
|
-
continue
|
|
158
|
-
else:
|
|
159
|
-
raise Exception("Blacklisted words filepath given does not exist")
|
|
160
|
-
|
|
161
|
-
if pmid != None and whole_content != "" and len(whole_content) > 1000:
|
|
152
|
+
if options['remain_empty_documents'] or (whole_content != "None"):
|
|
153
|
+
if options['filter_by_blacklist'] != None: #If a blacklist word file is given, check to filter out documents whose title or article_category contains any of the words
|
|
154
|
+
is_blacklisted = TextIndexer._check_if_blacklisted_pmid(pmid, title, article_category, options, logger)
|
|
155
|
+
if is_blacklisted: continue
|
|
156
|
+
|
|
162
157
|
#pmid_content_and_stats = cls.prepare_indexes(whole_content, pmc+"-"+pmid, filename, year, options)
|
|
163
|
-
pmid_content_and_stats = cls.prepare_indexes(whole_content, pmid, filename, year, title, article_type, article_category, options)
|
|
158
|
+
pmid_content_and_stats = cls.prepare_indexes(whole_content, pmid, filename, year, title, article_type, article_category, keywords, options)
|
|
164
159
|
texts.append(pmid_content_and_stats)
|
|
165
160
|
|
|
166
161
|
if logger != None: logger.warning(f"stats:file={file_path},total={stats['total']},no_abstract={stats['no_abstract']},no_pmid={stats['no_pmid']}")
|
|
167
162
|
if logger != None: logger.warning(f"logs_errors:file={file_path},errors_number={stats['errors']}")
|
|
168
163
|
return texts
|
|
169
|
-
|
|
164
|
+
|
|
170
165
|
|
|
171
166
|
@classmethod
|
|
172
|
-
def prepare_indexes(cls, text, pmid, file, year, title, article_type, article_category, options):
|
|
167
|
+
def prepare_indexes(cls, text, pmid, file, year, title, article_type, article_category, keywords, options):
|
|
168
|
+
extra_fields = options.get("extra_fields", defaultdict(lambda: False))
|
|
173
169
|
pmid = pmid.replace("\n", "")
|
|
174
170
|
file = file.replace("\n", "")
|
|
175
171
|
year = str(year).replace("\n", "")
|
|
176
|
-
|
|
177
172
|
document_length = str(len(text))
|
|
178
|
-
|
|
179
|
-
|
|
173
|
+
|
|
174
|
+
cleaned_text = TextPubmedParser.perform_soft_cleaning(text, type=options["clean_type"])
|
|
175
|
+
if text == "None":
|
|
176
|
+
doc_to_jsonify = []
|
|
177
|
+
document_length = "0"
|
|
178
|
+
number_of_sentences = "0"
|
|
179
|
+
length_of_sentences = "0"
|
|
180
|
+
elif options["split"]:
|
|
181
|
+
document_parts = cls.split_document(cleaned_text, pmid, split_type = options["split_type"])
|
|
182
|
+
doc_to_jsonify = document_parts
|
|
180
183
|
flattened_document = flatten(document_parts)
|
|
181
184
|
number_of_sentences = str(len(flattened_document))
|
|
182
185
|
length_of_sentences = ",".join([str(len(sentence)) for sentence in flattened_document])
|
|
183
|
-
document_parts_json = json.dumps(document_parts)
|
|
184
|
-
prepared_index = [pmid, document_parts_json, file, year, document_length, number_of_sentences, length_of_sentences,
|
|
185
|
-
title, article_type, article_category]
|
|
186
186
|
else:
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
187
|
+
number_of_sentences = "1"
|
|
188
|
+
length_of_sentences = document_length
|
|
189
|
+
doc_to_jsonify = [[cleaned_text]]
|
|
190
|
+
|
|
191
|
+
keywords_to_add = keywords if len(keywords) > 0 else ["None"]
|
|
192
|
+
if extra_fields["keywords"]: doc_to_jsonify.insert(0, ["KEYWORDS"] + keywords_to_add)
|
|
193
|
+
if extra_fields["title"]: doc_to_jsonify.insert(0, ["TITLE", title])
|
|
194
|
+
|
|
195
|
+
document_json = json.dumps(doc_to_jsonify)
|
|
196
|
+
prepared_index = [pmid, document_json, file, year, document_length, number_of_sentences, length_of_sentences,
|
|
197
|
+
title, article_type, article_category, ",".join(keywords_to_add)]
|
|
190
198
|
return prepared_index
|
|
191
199
|
|
|
192
200
|
|
|
@@ -198,25 +206,54 @@ class TextIndexer:
|
|
|
198
206
|
if split_type == "classic":
|
|
199
207
|
sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 10, chunk_overlap = 0, length_function = len, separators=["\n",".", ";", ","], keep_separator=False, is_separator_regex=False)
|
|
200
208
|
elif split_type == "space_overlap":
|
|
201
|
-
sentences_splitter = RecursiveCharacterTextSplitter(chunk_size =
|
|
209
|
+
sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 100, chunk_overlap = 20, length_function = len, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
|
|
210
|
+
#sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 20, chunk_overlap = 2, length_function = cls._len_by_words_func, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
|
|
202
211
|
#sentences_splitter = RecursiveCharacterTextSplitter(chunk_size = 120, chunk_overlap = 20, length_function = len, separators=["\n", " ", ""], keep_separator=False, is_separator_regex=False)
|
|
203
212
|
paragraphs = [paragraph.strip() for paragraph in paragraph_splitter.split_text(text)]
|
|
204
|
-
|
|
205
|
-
sentences = [ list(map(lambda sentence: sentence.strip(), paragraph)) for paragraph in sentences ]
|
|
213
|
+
paragraph_sentences = [ sentences_splitter.split_text(paragraph) for paragraph in paragraphs ]
|
|
206
214
|
|
|
207
215
|
formated = []
|
|
208
|
-
for paragraph in
|
|
216
|
+
for paragraph in paragraph_sentences:
|
|
209
217
|
formated_paragraph = []
|
|
210
218
|
for sentence in paragraph:
|
|
211
219
|
if len(sentence) > 2:
|
|
212
220
|
if len(sentence) > 3000: warnings.warn(f"ERROR: The pmid {pmid} has an unusual sentence lenght even after splitting. Total length of characters: {len(sentence)}. Content: {sentence}")
|
|
213
|
-
else: formated_paragraph.append(sentence)
|
|
221
|
+
else: formated_paragraph.append(sentence.strip())
|
|
214
222
|
if len(formated_paragraph) > 0: formated.append(formated_paragraph)
|
|
215
223
|
|
|
216
224
|
return formated
|
|
217
225
|
|
|
218
226
|
|
|
219
227
|
######### UTILS METHODS
|
|
228
|
+
@classmethod
|
|
229
|
+
def _len_by_words_func(cls, text):
|
|
230
|
+
return text.count(" ")
|
|
231
|
+
|
|
232
|
+
@classmethod
|
|
233
|
+
def _set_alternative_pmid_if_needed_and_available(cls, pmid, pmc, PMC_PMID_dict, file_path, logger):
|
|
234
|
+
#If there is no pmid in the paper or in the PMC_PMID_dict, we will use the pmc as the document identifier, specifying it is PMC in the index.
|
|
235
|
+
#If not PMC nor PMID, we will warn the user
|
|
236
|
+
if pmid == "None" and PMC_PMID_dict != None:
|
|
237
|
+
pmid = PMC_PMID_dict.get(pmc, "None")
|
|
238
|
+
if pmid == "None" and pmc == "None":
|
|
239
|
+
logger.warning(f"ERROR: The file {file_path} has a paper without PMID and PMC")
|
|
240
|
+
elif pmid == "None" and pmc != "None":
|
|
241
|
+
pmid = pmc
|
|
242
|
+
return pmid
|
|
243
|
+
|
|
244
|
+
@classmethod
|
|
245
|
+
def _check_if_blacklisted_pmid(cls, pmid, title, article_category, options, logger):
|
|
246
|
+
if os.path.exists(options['filter_by_blacklist']):
|
|
247
|
+
is_blacklisted = False
|
|
248
|
+
blacklisted_words = [word.strip() for word in open(options['filter_by_blacklist']).readlines()]
|
|
249
|
+
filtered_out, w, c, t = TextIndexer._check_to_filter_out(blacklisted_words, title, article_category, options['blacklisted_mode'])
|
|
250
|
+
if filtered_out:
|
|
251
|
+
if logger != None: logger.warning(f"Blacklisted PMID {pmid} for having word {w} in {c} with content: {t}")
|
|
252
|
+
is_blacklisted = True
|
|
253
|
+
return is_blacklisted
|
|
254
|
+
else:
|
|
255
|
+
raise Exception("Blacklisted words filepath given does not exist")
|
|
256
|
+
|
|
220
257
|
|
|
221
258
|
@classmethod
|
|
222
259
|
def _check_to_filter_out(cls, blacklisted_words, title, article_category, mode):
|