text2rel 0.2__tar.gz → 0.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- text2rel-0.5/PKG-INFO +86 -0
- text2rel-0.5/pyproject.toml +24 -0
- text2rel-0.5/readme.md +67 -0
- text2rel-0.5/text2rel/__init__.py +46 -0
- {text2rel-0.2 → text2rel-0.5}/text2rel/cleaner.py +314 -84
- text2rel-0.5/text2rel/historical_adapter.py +255 -0
- text2rel-0.5/text2rel/labelbox.py +2102 -0
- text2rel-0.5/text2rel/legacy_preprocessing_files/2026_07_27_historical_cleaner_and_batch_mode.md +150 -0
- text2rel-0.5/text2rel/legacy_preprocessing_files/historical_2025_prompts.txt +118 -0
- text2rel-0.5/text2rel/legacy_preprocessing_files/historical_document_manifest.txt +63 -0
- text2rel-0.5/text2rel/legacy_preprocessing_files/historical_reproducibility.py +301 -0
- text2rel-0.5/text2rel/llm_processor.py +734 -0
- text2rel-0.5/text2rel/page_matching.py +964 -0
- {text2rel-0.2 → text2rel-0.5}/text2rel/reliability_assessor.py +107 -43
- text2rel-0.5/text2rel/utils.py +279 -0
- text2rel-0.2/PKG-INFO +0 -42
- text2rel-0.2/pyproject.toml +0 -18
- text2rel-0.2/readme.md +0 -26
- text2rel-0.2/text2rel/__init__.py +0 -12
- text2rel-0.2/text2rel/llm_processor.py +0 -442
- text2rel-0.2/text2rel/utils.py +0 -175
- {text2rel-0.2 → text2rel-0.5}/text2rel/io.py +0 -0
- {text2rel-0.2 → text2rel-0.5}/text2rel/state.py +0 -0
text2rel-0.5/PKG-INFO
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: text2rel
|
|
3
|
+
Version: 0.5
|
|
4
|
+
Summary: An HTML cleaner customized for kinship ties extraction from large text corpra.
|
|
5
|
+
Requires-Python: >=3.10,<3.13
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Requires-Dist: beautifulsoup4 (>=4.12,<5.0)
|
|
11
|
+
Requires-Dist: numpy (==1.26.4)
|
|
12
|
+
Requires-Dist: openai
|
|
13
|
+
Requires-Dist: pandas (==2.2.2)
|
|
14
|
+
Requires-Dist: pymupdf (>=1.24,<2)
|
|
15
|
+
Requires-Dist: regex
|
|
16
|
+
Requires-Dist: tiktoken
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+

|
|
20
|
+

|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
|
|
25
|
+
|
|
26
|
+
### Important information:
|
|
27
|
+
- For example package usage, see ```test_workflow.ipynb```
|
|
28
|
+
- For information about individual functions before the package implementation and release, see ```examples``` folder.
|
|
29
|
+
|
|
30
|
+
### Expected input files
|
|
31
|
+
|
|
32
|
+
Each document should normally have these three files together in the same
|
|
33
|
+
folder under `documents_root`:
|
|
34
|
+
|
|
35
|
+
```text
|
|
36
|
+
0_GenelogiesAndBiographies/
|
|
37
|
+
├── Example genealogy/
|
|
38
|
+
│ ├── Example genealogy_mod.htm # source HTML used for the inventory
|
|
39
|
+
│ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
|
|
40
|
+
│ └── Example genealogy_Original.pdf # original PDF without OCR
|
|
41
|
+
└── Biographies/
|
|
42
|
+
└── Example biography/
|
|
43
|
+
├── Example biography_mod.htm
|
|
44
|
+
├── Example biography_mod.pdf
|
|
45
|
+
└── Example biography_Original.pdf
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
The filenames must share the same base name. The package uses the HTML for
|
|
49
|
+
inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
|
|
50
|
+
`*_Original.pdf`, which can be OCRed with Tesseract when necessary.
|
|
51
|
+
|
|
52
|
+
### Current release:
|
|
53
|
+
Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
|
|
54
|
+
|
|
55
|
+
### Files structure:
|
|
56
|
+
|
|
57
|
+
1. #### cleaner.py
|
|
58
|
+
Contains the HTMLCleaner class and thus the main logic.
|
|
59
|
+
|
|
60
|
+
2. #### utils.py
|
|
61
|
+
Contains helper functions (like JumPJumP insertion)
|
|
62
|
+
|
|
63
|
+
3. #### io.py
|
|
64
|
+
Contains the file processing logic, such as loading and saving the JSON inventory.
|
|
65
|
+
|
|
66
|
+
4. #### page_matching.py
|
|
67
|
+
Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
|
|
68
|
+
exact and fuzzy text matching. Missing PDFs are skipped with null page values
|
|
69
|
+
and explicit filename guidance.
|
|
70
|
+
|
|
71
|
+
### Add PDF page numbers
|
|
72
|
+
|
|
73
|
+
If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
|
|
74
|
+
only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
summary = cleaner.add_pdf_pages(
|
|
78
|
+
documents_root="../0_GenelogiesAndBiographies",
|
|
79
|
+
output_path="exact_html_inventory_new_ids_cleaned_pages.json",
|
|
80
|
+
)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
|
|
84
|
+
back to `_Original.pdf` when needed, and can use Tesseract for image-only
|
|
85
|
+
originals when Tesseract is installed.
|
|
86
|
+
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "text2rel"
|
|
3
|
+
version = "0.5"
|
|
4
|
+
description = "An HTML cleaner customized for kinship ties extraction from large text corpra."
|
|
5
|
+
|
|
6
|
+
readme = "readme.md"
|
|
7
|
+
include = [
|
|
8
|
+
{ path = "text2rel/legacy_preprocessing_files", format = ["sdist", "wheel"] },
|
|
9
|
+
]
|
|
10
|
+
|
|
11
|
+
[tool.poetry.dependencies]
|
|
12
|
+
python = ">=3.10,<3.13"
|
|
13
|
+
pandas = "2.2.2"
|
|
14
|
+
numpy = "1.26.4"
|
|
15
|
+
regex = "*"
|
|
16
|
+
beautifulsoup4 = "^4.12"
|
|
17
|
+
pymupdf = ">=1.24,<2"
|
|
18
|
+
openai = "*"
|
|
19
|
+
tiktoken = "*"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
[build-system]
|
|
23
|
+
build-backend = "poetry.core.masonry.api"
|
|
24
|
+
requires = ["poetry-core>=1.0.0"]
|
text2rel-0.5/readme.md
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+

|
|
2
|
+

|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
|
|
7
|
+
|
|
8
|
+
### Important information:
|
|
9
|
+
- For example package usage, see ```test_workflow.ipynb```
|
|
10
|
+
- For information about individual functions before the package implementation and release, see ```examples``` folder.
|
|
11
|
+
|
|
12
|
+
### Expected input files
|
|
13
|
+
|
|
14
|
+
Each document should normally have these three files together in the same
|
|
15
|
+
folder under `documents_root`:
|
|
16
|
+
|
|
17
|
+
```text
|
|
18
|
+
0_GenelogiesAndBiographies/
|
|
19
|
+
├── Example genealogy/
|
|
20
|
+
│ ├── Example genealogy_mod.htm # source HTML used for the inventory
|
|
21
|
+
│ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
|
|
22
|
+
│ └── Example genealogy_Original.pdf # original PDF without OCR
|
|
23
|
+
└── Biographies/
|
|
24
|
+
└── Example biography/
|
|
25
|
+
├── Example biography_mod.htm
|
|
26
|
+
├── Example biography_mod.pdf
|
|
27
|
+
└── Example biography_Original.pdf
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The filenames must share the same base name. The package uses the HTML for
|
|
31
|
+
inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
|
|
32
|
+
`*_Original.pdf`, which can be OCRed with Tesseract when necessary.
|
|
33
|
+
|
|
34
|
+
### Current release:
|
|
35
|
+
Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
|
|
36
|
+
|
|
37
|
+
### Files structure:
|
|
38
|
+
|
|
39
|
+
1. #### cleaner.py
|
|
40
|
+
Contains the HTMLCleaner class and thus the main logic.
|
|
41
|
+
|
|
42
|
+
2. #### utils.py
|
|
43
|
+
Contains helper functions (like JumPJumP insertion)
|
|
44
|
+
|
|
45
|
+
3. #### io.py
|
|
46
|
+
Contains the file processing logic, such as loading and saving the JSON inventory.
|
|
47
|
+
|
|
48
|
+
4. #### page_matching.py
|
|
49
|
+
Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
|
|
50
|
+
exact and fuzzy text matching. Missing PDFs are skipped with null page values
|
|
51
|
+
and explicit filename guidance.
|
|
52
|
+
|
|
53
|
+
### Add PDF page numbers
|
|
54
|
+
|
|
55
|
+
If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
|
|
56
|
+
only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
summary = cleaner.add_pdf_pages(
|
|
60
|
+
documents_root="../0_GenelogiesAndBiographies",
|
|
61
|
+
output_path="exact_html_inventory_new_ids_cleaned_pages.json",
|
|
62
|
+
)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
|
|
66
|
+
back to `_Original.pdf` when needed, and can use Tesseract for image-only
|
|
67
|
+
originals when Tesseract is installed.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from .cleaner import HTMLCleaner
|
|
2
|
+
from .llm_processor import LLMProcessor
|
|
3
|
+
from .reliability_assessor import ReliabilityAssessment
|
|
4
|
+
from .state import set_default_inventory_path, get_default_inventory_path
|
|
5
|
+
from .page_matching import add_pages_to_inventory, add_pages_to_inventory_file
|
|
6
|
+
|
|
7
|
+
from .labelbox import (
|
|
8
|
+
export_labelbox_project,
|
|
9
|
+
download_handlabels_from_projects,
|
|
10
|
+
build_reproducible_handlabel_allocation,
|
|
11
|
+
save_handlabel_allocation,
|
|
12
|
+
save_labelbox_tasks_per_labeler,
|
|
13
|
+
build_labelbox_chunk_items,
|
|
14
|
+
upload_selected_chunks_to_labelbox,
|
|
15
|
+
upload_labelers_data_to_labelbox,
|
|
16
|
+
parse_labelbox_export,
|
|
17
|
+
labelbox_project_to_tables,
|
|
18
|
+
labelbox_relations_to_dataframe,
|
|
19
|
+
labelbox_relations_from_project,
|
|
20
|
+
labelbox_events_to_dataframe,
|
|
21
|
+
labelbox_events_from_project,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"HTMLCleaner",
|
|
26
|
+
"LLMProcessor",
|
|
27
|
+
"ReliabilityAssessment",
|
|
28
|
+
"set_default_inventory_path",
|
|
29
|
+
"get_default_inventory_path",
|
|
30
|
+
"add_pages_to_inventory",
|
|
31
|
+
"add_pages_to_inventory_file",
|
|
32
|
+
"export_labelbox_project",
|
|
33
|
+
"download_handlabels_from_projects",
|
|
34
|
+
"build_reproducible_handlabel_allocation",
|
|
35
|
+
"save_handlabel_allocation",
|
|
36
|
+
"save_labelbox_tasks_per_labeler",
|
|
37
|
+
"build_labelbox_chunk_items",
|
|
38
|
+
"upload_selected_chunks_to_labelbox",
|
|
39
|
+
"upload_labelers_data_to_labelbox",
|
|
40
|
+
"parse_labelbox_export",
|
|
41
|
+
"labelbox_project_to_tables",
|
|
42
|
+
"labelbox_relations_to_dataframe",
|
|
43
|
+
"labelbox_relations_from_project",
|
|
44
|
+
"labelbox_events_to_dataframe",
|
|
45
|
+
"labelbox_events_from_project",
|
|
46
|
+
]
|