text2rel 0.2__tar.gz → 1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- text2rel-1.0/PKG-INFO +153 -0
- text2rel-1.0/pyproject.toml +32 -0
- text2rel-1.0/readme.md +129 -0
- text2rel-1.0/text2rel/__init__.py +119 -0
- {text2rel-0.2 → text2rel-1.0}/text2rel/cleaner.py +314 -84
- text2rel-1.0/text2rel/contracts.py +209 -0
- text2rel-1.0/text2rel/default_taxonomy.txt +2075 -0
- text2rel-1.0/text2rel/historical_adapter.py +255 -0
- text2rel-1.0/text2rel/labelbox.py +3106 -0
- text2rel-1.0/text2rel/legacy_preprocessing_files/2026_07_27_historical_cleaner_and_batch_mode.md +150 -0
- text2rel-1.0/text2rel/legacy_preprocessing_files/historical_2025_prompts.txt +118 -0
- text2rel-1.0/text2rel/legacy_preprocessing_files/historical_document_manifest.txt +63 -0
- text2rel-1.0/text2rel/legacy_preprocessing_files/historical_reproducibility.py +301 -0
- text2rel-1.0/text2rel/llm_processor.py +734 -0
- text2rel-1.0/text2rel/page_matching.py +964 -0
- {text2rel-0.2 → text2rel-1.0}/text2rel/reliability_assessor.py +1129 -75
- text2rel-1.0/text2rel/reliability_plots.py +888 -0
- text2rel-1.0/text2rel/similarity.py +76 -0
- text2rel-1.0/text2rel/taxonomy.py +271 -0
- text2rel-1.0/text2rel/utils.py +490 -0
- text2rel-0.2/PKG-INFO +0 -42
- text2rel-0.2/pyproject.toml +0 -18
- text2rel-0.2/readme.md +0 -26
- text2rel-0.2/text2rel/__init__.py +0 -12
- text2rel-0.2/text2rel/llm_processor.py +0 -442
- text2rel-0.2/text2rel/utils.py +0 -175
- {text2rel-0.2 → text2rel-1.0}/text2rel/io.py +0 -0
- {text2rel-0.2 → text2rel-1.0}/text2rel/state.py +0 -0
text2rel-1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: text2rel
|
|
3
|
+
Version: 1.0
|
|
4
|
+
Summary: An HTML cleaner customized for kinship ties extraction from large text corpra.
|
|
5
|
+
Requires-Python: >=3.10,<3.13
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Provides-Extra: plots
|
|
11
|
+
Requires-Dist: beautifulsoup4 (>=4.12,<5.0)
|
|
12
|
+
Requires-Dist: jellyfish
|
|
13
|
+
Requires-Dist: matplotlib ; extra == "plots"
|
|
14
|
+
Requires-Dist: numpy (==1.26.4)
|
|
15
|
+
Requires-Dist: openai
|
|
16
|
+
Requires-Dist: pandas (==2.2.2)
|
|
17
|
+
Requires-Dist: pymupdf (>=1.24,<2)
|
|
18
|
+
Requires-Dist: rapidfuzz
|
|
19
|
+
Requires-Dist: regex
|
|
20
|
+
Requires-Dist: scipy ; extra == "plots"
|
|
21
|
+
Requires-Dist: tiktoken
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+

|
|
25
|
+

|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
|
|
30
|
+
|
|
31
|
+
### Important information:
|
|
32
|
+
- For example package usage, see ```test_workflow.ipynb```
|
|
33
|
+
- For information about individual functions before the package implementation and release, see ```examples``` folder.
|
|
34
|
+
|
|
35
|
+
### Expected input files
|
|
36
|
+
|
|
37
|
+
Each document should normally have these three files together in the same
|
|
38
|
+
folder under `documents_root`:
|
|
39
|
+
|
|
40
|
+
```text
|
|
41
|
+
0_GenelogiesAndBiographies/
|
|
42
|
+
├── Example genealogy/
|
|
43
|
+
│ ├── Example genealogy_mod.htm # source HTML used for the inventory
|
|
44
|
+
│ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
|
|
45
|
+
│ └── Example genealogy_Original.pdf # original PDF without OCR
|
|
46
|
+
└── Biographies/
|
|
47
|
+
└── Example biography/
|
|
48
|
+
├── Example biography_mod.htm
|
|
49
|
+
├── Example biography_mod.pdf
|
|
50
|
+
└── Example biography_Original.pdf
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
The filenames must share the same base name. The package uses the HTML for
|
|
54
|
+
inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
|
|
55
|
+
`*_Original.pdf`, which can be OCRed with Tesseract when necessary.
|
|
56
|
+
|
|
57
|
+
### Current release:
|
|
58
|
+
Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
|
|
59
|
+
|
|
60
|
+
### Files structure:
|
|
61
|
+
|
|
62
|
+
1. #### cleaner.py
|
|
63
|
+
Contains the HTMLCleaner class and thus the main logic.
|
|
64
|
+
|
|
65
|
+
2. #### utils.py
|
|
66
|
+
Contains helper functions (like JumPJumP insertion)
|
|
67
|
+
|
|
68
|
+
3. #### io.py
|
|
69
|
+
Contains the file processing logic, such as loading and saving the JSON inventory.
|
|
70
|
+
|
|
71
|
+
4. #### page_matching.py
|
|
72
|
+
Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
|
|
73
|
+
exact and fuzzy text matching. Missing PDFs are skipped with null page values
|
|
74
|
+
and explicit filename guidance.
|
|
75
|
+
|
|
76
|
+
### Add PDF page numbers
|
|
77
|
+
|
|
78
|
+
If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
|
|
79
|
+
only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
summary = cleaner.add_pdf_pages(
|
|
83
|
+
documents_root="../0_GenelogiesAndBiographies",
|
|
84
|
+
output_path="exact_html_inventory_new_ids_cleaned_pages.json",
|
|
85
|
+
)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
|
|
89
|
+
back to `_Original.pdf` when needed, and can use Tesseract for image-only
|
|
90
|
+
originals when Tesseract is installed.
|
|
91
|
+
|
|
92
|
+
### Human Labeler Allocation
|
|
93
|
+
|
|
94
|
+
When text chunks are allocated to labelers, each generated TXT filename
|
|
95
|
+
indicates its task type:
|
|
96
|
+
|
|
97
|
+
- `Task_0` contains chunks shared across all labelers.
|
|
98
|
+
- `Task_1` through `Task_9` contain labeler-specific, unshared chunks.
|
|
99
|
+
|
|
100
|
+
### New section 4 assessment workflow
|
|
101
|
+
|
|
102
|
+
The reusable assessment follows sections 4.2.2.1 and 4.2.2.2 of
|
|
103
|
+
`240426 Information extraction system 260801.ipynb`. Install the optional plot
|
|
104
|
+
dependencies, load the processed human and AI tables, and then create the
|
|
105
|
+
assessment tables before drawing anything:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
pip install -e ".[plots]"
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
from text2rel import (
|
|
113
|
+
ReliabilityAssessment,
|
|
114
|
+
build_sweep_report,
|
|
115
|
+
plot_sweep_report,
|
|
116
|
+
plot_threshold_curves,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
assessment = ReliabilityAssessment("inventory.json")
|
|
120
|
+
assessment.load_relations_data(relations_csv="human_relations.csv")
|
|
121
|
+
assessment.load_chatgpt_relations("machine_relations.csv")
|
|
122
|
+
|
|
123
|
+
# one_to_one is the package's stricter default for headline metrics.
|
|
124
|
+
relation_results = assessment.assess_llm_against_humans(data="relations")
|
|
125
|
+
plot_threshold_curves(relation_results["threshold_metrics"], direction="one_to_one")
|
|
126
|
+
|
|
127
|
+
# directional_best reproduces the notebook's two reusable-candidate views.
|
|
128
|
+
diagnostics = assessment.assess_llm_against_humans(
|
|
129
|
+
data="relations", matching_mode="directional_best"
|
|
130
|
+
)
|
|
131
|
+
fn_report = build_sweep_report(
|
|
132
|
+
diagnostics["human_to_llm_best_pairs"], data="relations"
|
|
133
|
+
)
|
|
134
|
+
plot_sweep_report(fn_report)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
`assess_llm_against_humans()` returns the selected pairs, threshold metrics,
|
|
138
|
+
false-negative and false-positive review tables, structural exclusions, and
|
|
139
|
+
configuration metadata. Plot functions only render these returned tables; they
|
|
140
|
+
do not change matching or scores. See the separate **New section 4 assessment
|
|
141
|
+
plots and tables** section at the end of `test_workflow.ipynb` for relations
|
|
142
|
+
and events, notebook-style titles, confidence/length diagnostics, and review
|
|
143
|
+
summaries.
|
|
144
|
+
|
|
145
|
+
### Labelbox event construction modes
|
|
146
|
+
|
|
147
|
+
`labelbox_events_to_dataframe(..., construction_mode="graph")` is the default.
|
|
148
|
+
It follows section 2.4.3.2 of the 260801 notebook: connected annotations are
|
|
149
|
+
rebuilt around one Verb, safe reversed arrows are corrected, and incomplete or
|
|
150
|
+
ambiguous structures remain auditable. Use `construction_mode="legacy_simple"`
|
|
151
|
+
only when explicitly reproducing the package's earlier permissive conversion.
|
|
152
|
+
Relation conversion uses its audited endpoint joins separately.
|
|
153
|
+
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "text2rel"
|
|
3
|
+
version = "1.0"
|
|
4
|
+
description = "An HTML cleaner customized for kinship ties extraction from large text corpra."
|
|
5
|
+
|
|
6
|
+
readme = "readme.md"
|
|
7
|
+
include = [
|
|
8
|
+
{ path = "text2rel/legacy_preprocessing_files", format = ["sdist", "wheel"] },
|
|
9
|
+
{ path = "text2rel/default_taxonomy.txt", format = ["sdist", "wheel"] },
|
|
10
|
+
]
|
|
11
|
+
|
|
12
|
+
[tool.poetry.dependencies]
|
|
13
|
+
python = ">=3.10,<3.13"
|
|
14
|
+
pandas = "2.2.2"
|
|
15
|
+
numpy = "1.26.4"
|
|
16
|
+
regex = "*"
|
|
17
|
+
beautifulsoup4 = "^4.12"
|
|
18
|
+
pymupdf = ">=1.24,<2"
|
|
19
|
+
openai = "*"
|
|
20
|
+
tiktoken = "*"
|
|
21
|
+
jellyfish = "*"
|
|
22
|
+
rapidfuzz = "*"
|
|
23
|
+
matplotlib = { version = "*", optional = true }
|
|
24
|
+
scipy = { version = "*", optional = true }
|
|
25
|
+
|
|
26
|
+
[tool.poetry.extras]
|
|
27
|
+
plots = ["matplotlib", "scipy"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
[build-system]
|
|
31
|
+
build-backend = "poetry.core.masonry.api"
|
|
32
|
+
requires = ["poetry-core>=1.0.0"]
|
text2rel-1.0/readme.md
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+

|
|
2
|
+

|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
This is the package implementation of the code for the kinship ties and relational information from large text corpora, such as genealogies, biographies, and historical dictionaries.
|
|
7
|
+
|
|
8
|
+
### Important information:
|
|
9
|
+
- For example package usage, see ```test_workflow.ipynb```
|
|
10
|
+
- For information about individual functions before the package implementation and release, see ```examples``` folder.
|
|
11
|
+
|
|
12
|
+
### Expected input files
|
|
13
|
+
|
|
14
|
+
Each document should normally have these three files together in the same
|
|
15
|
+
folder under `documents_root`:
|
|
16
|
+
|
|
17
|
+
```text
|
|
18
|
+
0_GenelogiesAndBiographies/
|
|
19
|
+
├── Example genealogy/
|
|
20
|
+
│ ├── Example genealogy_mod.htm # source HTML used for the inventory
|
|
21
|
+
│ ├── Example genealogy_mod.pdf # PDF with a readable OCR text layer
|
|
22
|
+
│ └── Example genealogy_Original.pdf # original PDF without OCR
|
|
23
|
+
└── Biographies/
|
|
24
|
+
└── Example biography/
|
|
25
|
+
├── Example biography_mod.htm
|
|
26
|
+
├── Example biography_mod.pdf
|
|
27
|
+
└── Example biography_Original.pdf
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The filenames must share the same base name. The package uses the HTML for
|
|
31
|
+
inventory creation, prefers `*_mod.pdf` for page matching, and falls back to
|
|
32
|
+
`*_Original.pdf`, which can be OCRed with Tesseract when necessary.
|
|
33
|
+
|
|
34
|
+
### Current release:
|
|
35
|
+
Current release includes inventory creation, setting text bounds, assigning font usage and selecting appropriate text chunks from the files (including restriction to "MainText" fonts and JumPJumP insertion).
|
|
36
|
+
|
|
37
|
+
### Files structure:
|
|
38
|
+
|
|
39
|
+
1. #### cleaner.py
|
|
40
|
+
Contains the HTMLCleaner class and thus the main logic.
|
|
41
|
+
|
|
42
|
+
2. #### utils.py
|
|
43
|
+
Contains helper functions (like JumPJumP insertion)
|
|
44
|
+
|
|
45
|
+
3. #### io.py
|
|
46
|
+
Contains the file processing logic, such as loading and saving the JSON inventory.
|
|
47
|
+
|
|
48
|
+
4. #### page_matching.py
|
|
49
|
+
Adds one-based PDF `start_page` and `end_page` values to chunk metadata using
|
|
50
|
+
exact and fuzzy text matching. Missing PDFs are skipped with null page values
|
|
51
|
+
and explicit filename guidance.
|
|
52
|
+
|
|
53
|
+
### Add PDF page numbers
|
|
54
|
+
|
|
55
|
+
If you already created an `HTMLCleaner`, use `cleaner.add_pdf_pages()`. If you
|
|
56
|
+
only have an inventory JSON file, use `add_pages_to_inventory_file()` directly.
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
summary = cleaner.add_pdf_pages(
|
|
60
|
+
documents_root="../0_GenelogiesAndBiographies",
|
|
61
|
+
output_path="exact_html_inventory_new_ids_cleaned_pages.json",
|
|
62
|
+
)
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
The default fuzzy threshold is `0.80`. The matcher prefers `*_mod.pdf`, falls
|
|
66
|
+
back to `_Original.pdf` when needed, and can use Tesseract for image-only
|
|
67
|
+
originals when Tesseract is installed.
|
|
68
|
+
|
|
69
|
+
### Human Labeler Allocation
|
|
70
|
+
|
|
71
|
+
When text chunks are allocated to labelers, each generated TXT filename
|
|
72
|
+
indicates its task type:
|
|
73
|
+
|
|
74
|
+
- `Task_0` contains chunks shared across all labelers.
|
|
75
|
+
- `Task_1` through `Task_9` contain labeler-specific, unshared chunks.
|
|
76
|
+
|
|
77
|
+
### New section 4 assessment workflow
|
|
78
|
+
|
|
79
|
+
The reusable assessment follows sections 4.2.2.1 and 4.2.2.2 of
|
|
80
|
+
`240426 Information extraction system 260801.ipynb`. Install the optional plot
|
|
81
|
+
dependencies, load the processed human and AI tables, and then create the
|
|
82
|
+
assessment tables before drawing anything:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install -e ".[plots]"
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from text2rel import (
|
|
90
|
+
ReliabilityAssessment,
|
|
91
|
+
build_sweep_report,
|
|
92
|
+
plot_sweep_report,
|
|
93
|
+
plot_threshold_curves,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
assessment = ReliabilityAssessment("inventory.json")
|
|
97
|
+
assessment.load_relations_data(relations_csv="human_relations.csv")
|
|
98
|
+
assessment.load_chatgpt_relations("machine_relations.csv")
|
|
99
|
+
|
|
100
|
+
# one_to_one is the package's stricter default for headline metrics.
|
|
101
|
+
relation_results = assessment.assess_llm_against_humans(data="relations")
|
|
102
|
+
plot_threshold_curves(relation_results["threshold_metrics"], direction="one_to_one")
|
|
103
|
+
|
|
104
|
+
# directional_best reproduces the notebook's two reusable-candidate views.
|
|
105
|
+
diagnostics = assessment.assess_llm_against_humans(
|
|
106
|
+
data="relations", matching_mode="directional_best"
|
|
107
|
+
)
|
|
108
|
+
fn_report = build_sweep_report(
|
|
109
|
+
diagnostics["human_to_llm_best_pairs"], data="relations"
|
|
110
|
+
)
|
|
111
|
+
plot_sweep_report(fn_report)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
`assess_llm_against_humans()` returns the selected pairs, threshold metrics,
|
|
115
|
+
false-negative and false-positive review tables, structural exclusions, and
|
|
116
|
+
configuration metadata. Plot functions only render these returned tables; they
|
|
117
|
+
do not change matching or scores. See the separate **New section 4 assessment
|
|
118
|
+
plots and tables** section at the end of `test_workflow.ipynb` for relations
|
|
119
|
+
and events, notebook-style titles, confidence/length diagnostics, and review
|
|
120
|
+
summaries.
|
|
121
|
+
|
|
122
|
+
### Labelbox event construction modes
|
|
123
|
+
|
|
124
|
+
`labelbox_events_to_dataframe(..., construction_mode="graph")` is the default.
|
|
125
|
+
It follows section 2.4.3.2 of the 260801 notebook: connected annotations are
|
|
126
|
+
rebuilt around one Verb, safe reversed arrows are corrected, and incomplete or
|
|
127
|
+
ambiguous structures remain auditable. Use `construction_mode="legacy_simple"`
|
|
128
|
+
only when explicitly reproducing the package's earlier permissive conversion.
|
|
129
|
+
Relation conversion uses its audited endpoint joins separately.
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
from .cleaner import HTMLCleaner
|
|
2
|
+
from .llm_processor import LLMProcessor
|
|
3
|
+
from .reliability_assessor import ReliabilityAssessment
|
|
4
|
+
from .state import set_default_inventory_path, get_default_inventory_path
|
|
5
|
+
from .page_matching import add_pages_to_inventory, add_pages_to_inventory_file
|
|
6
|
+
from .contracts import (
|
|
7
|
+
MatchingMode,
|
|
8
|
+
HEADLINE_MATCHING_MODE,
|
|
9
|
+
DIAGNOSTIC_MATCHING_MODE,
|
|
10
|
+
MATCHING_MODE_OUTPUT_COLUMN,
|
|
11
|
+
MATCHING_MODE_DESCRIPTIONS,
|
|
12
|
+
RELATION_FRAME_SCHEMA,
|
|
13
|
+
EVENT_FRAME_SCHEMA,
|
|
14
|
+
resolve_matching_mode,
|
|
15
|
+
)
|
|
16
|
+
from .similarity import component_similarity, four_measure_similarity
|
|
17
|
+
from .reliability_plots import (
|
|
18
|
+
plot_threshold_curves,
|
|
19
|
+
build_sweep_report,
|
|
20
|
+
plot_sweep_report,
|
|
21
|
+
analyze_similarity_by_confidence,
|
|
22
|
+
plot_similarity_by_confidence,
|
|
23
|
+
attach_input_tokens,
|
|
24
|
+
analyze_similarity_by_length,
|
|
25
|
+
plot_similarity_by_length,
|
|
26
|
+
export_review_table,
|
|
27
|
+
summarize_adjudication_reviews,
|
|
28
|
+
)
|
|
29
|
+
from .taxonomy import (
|
|
30
|
+
apply_taxonomy_groups,
|
|
31
|
+
build_taxonomy_mapping,
|
|
32
|
+
invert_relation_directions,
|
|
33
|
+
load_taxonomy,
|
|
34
|
+
)
|
|
35
|
+
from .utils import (
|
|
36
|
+
canonicalize_event_date,
|
|
37
|
+
canonicalize_event_place,
|
|
38
|
+
normalize_person_name,
|
|
39
|
+
normalize_text,
|
|
40
|
+
unwrap_singleton_list,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
from .labelbox import (
|
|
44
|
+
export_labelbox_project,
|
|
45
|
+
download_handlabels_from_projects,
|
|
46
|
+
build_reproducible_handlabel_allocation,
|
|
47
|
+
save_handlabel_allocation,
|
|
48
|
+
save_labelbox_tasks_per_labeler,
|
|
49
|
+
build_labelbox_chunk_items,
|
|
50
|
+
upload_selected_chunks_to_labelbox,
|
|
51
|
+
upload_labelers_data_to_labelbox,
|
|
52
|
+
parse_labelbox_export,
|
|
53
|
+
RelationFrameResult,
|
|
54
|
+
EventFrameResult,
|
|
55
|
+
build_labelbox_relation_frames,
|
|
56
|
+
build_labelbox_event_frames,
|
|
57
|
+
labelbox_project_to_tables,
|
|
58
|
+
labelbox_relations_to_dataframe,
|
|
59
|
+
labelbox_relations_from_project,
|
|
60
|
+
labelbox_events_to_dataframe,
|
|
61
|
+
labelbox_events_from_project,
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
__all__ = [
|
|
65
|
+
"HTMLCleaner",
|
|
66
|
+
"LLMProcessor",
|
|
67
|
+
"ReliabilityAssessment",
|
|
68
|
+
"set_default_inventory_path",
|
|
69
|
+
"get_default_inventory_path",
|
|
70
|
+
"add_pages_to_inventory",
|
|
71
|
+
"add_pages_to_inventory_file",
|
|
72
|
+
"MatchingMode",
|
|
73
|
+
"HEADLINE_MATCHING_MODE",
|
|
74
|
+
"DIAGNOSTIC_MATCHING_MODE",
|
|
75
|
+
"MATCHING_MODE_OUTPUT_COLUMN",
|
|
76
|
+
"MATCHING_MODE_DESCRIPTIONS",
|
|
77
|
+
"RELATION_FRAME_SCHEMA",
|
|
78
|
+
"EVENT_FRAME_SCHEMA",
|
|
79
|
+
"resolve_matching_mode",
|
|
80
|
+
"component_similarity",
|
|
81
|
+
"four_measure_similarity",
|
|
82
|
+
"plot_threshold_curves",
|
|
83
|
+
"build_sweep_report",
|
|
84
|
+
"plot_sweep_report",
|
|
85
|
+
"analyze_similarity_by_confidence",
|
|
86
|
+
"plot_similarity_by_confidence",
|
|
87
|
+
"attach_input_tokens",
|
|
88
|
+
"analyze_similarity_by_length",
|
|
89
|
+
"plot_similarity_by_length",
|
|
90
|
+
"export_review_table",
|
|
91
|
+
"summarize_adjudication_reviews",
|
|
92
|
+
"load_taxonomy",
|
|
93
|
+
"build_taxonomy_mapping",
|
|
94
|
+
"apply_taxonomy_groups",
|
|
95
|
+
"invert_relation_directions",
|
|
96
|
+
"normalize_text",
|
|
97
|
+
"normalize_person_name",
|
|
98
|
+
"unwrap_singleton_list",
|
|
99
|
+
"canonicalize_event_date",
|
|
100
|
+
"canonicalize_event_place",
|
|
101
|
+
"export_labelbox_project",
|
|
102
|
+
"download_handlabels_from_projects",
|
|
103
|
+
"build_reproducible_handlabel_allocation",
|
|
104
|
+
"save_handlabel_allocation",
|
|
105
|
+
"save_labelbox_tasks_per_labeler",
|
|
106
|
+
"build_labelbox_chunk_items",
|
|
107
|
+
"upload_selected_chunks_to_labelbox",
|
|
108
|
+
"upload_labelers_data_to_labelbox",
|
|
109
|
+
"parse_labelbox_export",
|
|
110
|
+
"RelationFrameResult",
|
|
111
|
+
"EventFrameResult",
|
|
112
|
+
"build_labelbox_relation_frames",
|
|
113
|
+
"build_labelbox_event_frames",
|
|
114
|
+
"labelbox_project_to_tables",
|
|
115
|
+
"labelbox_relations_to_dataframe",
|
|
116
|
+
"labelbox_relations_from_project",
|
|
117
|
+
"labelbox_events_to_dataframe",
|
|
118
|
+
"labelbox_events_from_project",
|
|
119
|
+
]
|