text2rel 0.2__tar.gz → 0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
text2rel-0.5/PKG-INFO ADDED
@@ -0,0 +1,86 @@
1
+ Metadata-Version: 2.4
2
+ Name: text2rel
3
+ Version: 0.5
4
+ Summary: An HTML cleaner customized for kinship ties extraction from large text corpra.
5
+ Requires-Python: >=3.10,<3.13
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Python :: 3.10
8
+ Classifier: Programming Language :: Python :: 3.11
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Requires-Dist: beautifulsoup4 (>=4.12,<5.0)
11
+ Requires-Dist: numpy (==1.26.4)
12
+ Requires-Dist: openai
13
+ Requires-Dist: pandas (==2.2.2)
14
+ Requires-Dist: pymupdf (>=1.24,<2)
15
+ Requires-Dist: regex
16
+ Requires-Dist: tiktoken
17
+ Description-Content-Type: text/markdown
18
+
19
+ ![Alt text](images/Kinship_Ties_Extraction.png)
20
+ ![PyPI version](https://img.shields.io/pypi/v/html-cleaner-kinship)
21
+
22
+
23
+
24
+ This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
25
+
26
+ ### Important information:
27
+ - For example package usage, see ```test_workflow.ipynb```
28
+ - For information about individual functions before the package implementation and release, see ```examples``` folder.
29
+
30
+ ### Expected input files
31
+
32
+ Each document should normally have these three files together in the same
33
+ folder under `documents_root`:
34
+
35
+ ```text
36
+ 0_GenelogiesAndBiographies/
37
+ ├── Example genealogy/
38
+ │ ├── Example genealogy_mod.htm # source HTML used for the inventory
39
+ │ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
40
+ │ └── Example genealogy_Original.pdf # original PDF without OCR
41
+ └── Biographies/
42
+ └── Example biography/
43
+ ├── Example biography_mod.htm
44
+ ├── Example biography_mod.pdf
45
+ └── Example biography_Original.pdf
46
+ ```
47
+
48
+ The filenames must share the same base name. The package uses the HTML for
49
+ inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
50
+ `*_Original.pdf`, which can be OCRed with Tesseract when necessary.
51
+
52
+ ### Current release:
53
+ Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
54
+
55
+ ### Files structure:
56
+
57
+ 1. #### cleaner.py
58
+ Contains the HTMLCleaner class and thus the main logic.
59
+
60
+ 2. #### utils.py
61
+ Contains helper functions (like JumPJumP insertion)
62
+
63
+ 3. #### io.py
64
+ Contains the file processing logic, such as loading and saving the JSON inventory.
65
+
66
+ 4. #### page_matching.py
67
+ Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
68
+ exact and fuzzy text matching. Missing PDFs are skipped with null page values
69
+ and explicit filename guidance.
70
+
71
+ ### Add PDF page numbers
72
+
73
+ If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
74
+ only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
75
+
76
+ ```python
77
+ summary = cleaner.add_pdf_pages(
78
+ documents_root="../0_GenelogiesAndBiographies",
79
+ output_path="exact_html_inventory_new_ids_cleaned_pages.json",
80
+ )
81
+ ```
82
+
83
+ The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
84
+ back to `_Original.pdf` when needed, and can use Tesseract for image-only
85
+ originals when Tesseract is installed.
86
+
@@ -0,0 +1,24 @@
1
+ [tool.poetry]
2
+ name = "text2rel"
3
+ version = "0.5"
4
+ description = "An HTML cleaner customized for kinship ties extraction from large text corpra."
5
+
6
+ readme = "readme.md"
7
+ include = [
8
+ { path = "text2rel/legacy_preprocessing_files", format = ["sdist", "wheel"] },
9
+ ]
10
+
11
+ [tool.poetry.dependencies]
12
+ python = ">=3.10,<3.13"
13
+ pandas = "2.2.2"
14
+ numpy = "1.26.4"
15
+ regex = "*"
16
+ beautifulsoup4 = "^4.12"
17
+ pymupdf = ">=1.24,<2"
18
+ openai = "*"
19
+ tiktoken = "*"
20
+
21
+
22
+ [build-system]
23
+ build-backend = "poetry.core.masonry.api"
24
+ requires = ["poetry-core>=1.0.0"]
text2rel-0.5/readme.md ADDED
@@ -0,0 +1,67 @@
1
+ ![Alt text](images/Kinship_Ties_Extraction.png)
2
+ ![PyPI version](https://img.shields.io/pypi/v/html-cleaner-kinship)
3
+
4
+
5
+
6
+ This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
7
+
8
+ ### Important information:
9
+ - For example package usage, see ```test_workflow.ipynb```
10
+ - For information about individual functions before the package implementation and release, see ```examples``` folder.
11
+
12
+ ### Expected input files
13
+
14
+ Each document should normally have these three files together in the same
15
+ folder under `documents_root`:
16
+
17
+ ```text
18
+ 0_GenelogiesAndBiographies/
19
+ ├── Example genealogy/
20
+ │ ├── Example genealogy_mod.htm # source HTML used for the inventory
21
+ │ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
22
+ │ └── Example genealogy_Original.pdf # original PDF without OCR
23
+ └── Biographies/
24
+ └── Example biography/
25
+ ├── Example biography_mod.htm
26
+ ├── Example biography_mod.pdf
27
+ └── Example biography_Original.pdf
28
+ ```
29
+
30
+ The filenames must share the same base name. The package uses the HTML for
31
+ inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
32
+ `*_Original.pdf`, which can be OCRed with Tesseract when necessary.
33
+
34
+ ### Current release:
35
+ Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
36
+
37
+ ### Files structure:
38
+
39
+ 1. #### cleaner.py
40
+ Contains the HTMLCleaner class and thus the main logic.
41
+
42
+ 2. #### utils.py
43
+ Contains helper functions (like JumPJumP insertion)
44
+
45
+ 3. #### io.py
46
+ Contains the file processing logic, such as loading and saving the JSON inventory.
47
+
48
+ 4. #### page_matching.py
49
+ Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
50
+ exact and fuzzy text matching. Missing PDFs are skipped with null page values
51
+ and explicit filename guidance.
52
+
53
+ ### Add PDF page numbers
54
+
55
+ If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
56
+ only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
57
+
58
+ ```python
59
+ summary = cleaner.add_pdf_pages(
60
+ documents_root="../0_GenelogiesAndBiographies",
61
+ output_path="exact_html_inventory_new_ids_cleaned_pages.json",
62
+ )
63
+ ```
64
+
65
+ The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
66
+ back to `_Original.pdf` when needed, and can use Tesseract for image-only
67
+ originals when Tesseract is installed.
@@ -0,0 +1,46 @@
1
+ from .cleaner import HTMLCleaner
2
+ from .llm_processor import LLMProcessor
3
+ from .reliability_assessor import ReliabilityAssessment
4
+ from .state import set_default_inventory_path, get_default_inventory_path
5
+ from .page_matching import add_pages_to_inventory, add_pages_to_inventory_file
6
+
7
+ from .labelbox import (
8
+ export_labelbox_project,
9
+ download_handlabels_from_projects,
10
+ build_reproducible_handlabel_allocation,
11
+ save_handlabel_allocation,
12
+ save_labelbox_tasks_per_labeler,
13
+ build_labelbox_chunk_items,
14
+ upload_selected_chunks_to_labelbox,
15
+ upload_labelers_data_to_labelbox,
16
+ parse_labelbox_export,
17
+ labelbox_project_to_tables,
18
+ labelbox_relations_to_dataframe,
19
+ labelbox_relations_from_project,
20
+ labelbox_events_to_dataframe,
21
+ labelbox_events_from_project,
22
+ )
23
+
24
+ __all__ = [
25
+ "HTMLCleaner",
26
+ "LLMProcessor",
27
+ "ReliabilityAssessment",
28
+ "set_default_inventory_path",
29
+ "get_default_inventory_path",
30
+ "add_pages_to_inventory",
31
+ "add_pages_to_inventory_file",
32
+ "export_labelbox_project",
33
+ "download_handlabels_from_projects",
34
+ "build_reproducible_handlabel_allocation",
35
+ "save_handlabel_allocation",
36
+ "save_labelbox_tasks_per_labeler",
37
+ "build_labelbox_chunk_items",
38
+ "upload_selected_chunks_to_labelbox",
39
+ "upload_labelers_data_to_labelbox",
40
+ "parse_labelbox_export",
41
+ "labelbox_project_to_tables",
42
+ "labelbox_relations_to_dataframe",
43
+ "labelbox_relations_from_project",
44
+ "labelbox_events_to_dataframe",
45
+ "labelbox_events_from_project",
46
+ ]