retexo 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- retexo-0.1.0/LICENSE +21 -0
- retexo-0.1.0/PKG-INFO +154 -0
- retexo-0.1.0/README.md +91 -0
- retexo-0.1.0/pyproject.toml +118 -0
- retexo-0.1.0/src/retexo/__init__.py +62 -0
- retexo-0.1.0/src/retexo/aligners/__init__.py +27 -0
- retexo-0.1.0/src/retexo/aligners/agreement.py +181 -0
- retexo-0.1.0/src/retexo/aligners/aligner.py +87 -0
- retexo-0.1.0/src/retexo/aligners/assignment.py +269 -0
- retexo-0.1.0/src/retexo/aligners/decode.py +328 -0
- retexo-0.1.0/src/retexo/aligners/global_decode.py +91 -0
- retexo-0.1.0/src/retexo/aligners/sameness.py +123 -0
- retexo-0.1.0/src/retexo/aligners/sinkhorn.py +109 -0
- retexo-0.1.0/src/retexo/baselines/__init__.py +78 -0
- retexo-0.1.0/src/retexo/baselines/adapters.py +410 -0
- retexo-0.1.0/src/retexo/baselines/annotations.py +373 -0
- retexo-0.1.0/src/retexo/baselines/augment.py +99 -0
- retexo-0.1.0/src/retexo/baselines/base.py +219 -0
- retexo-0.1.0/src/retexo/baselines/curriculum.py +102 -0
- retexo-0.1.0/src/retexo/baselines/curves.py +139 -0
- retexo-0.1.0/src/retexo/baselines/decoder.py +526 -0
- retexo-0.1.0/src/retexo/baselines/downstream.py +355 -0
- retexo-0.1.0/src/retexo/baselines/early_stopping.py +485 -0
- retexo-0.1.0/src/retexo/baselines/em_aligner.py +628 -0
- retexo-0.1.0/src/retexo/baselines/floors.py +84 -0
- retexo-0.1.0/src/retexo/baselines/full_system.py +420 -0
- retexo-0.1.0/src/retexo/baselines/labels.py +333 -0
- retexo-0.1.0/src/retexo/baselines/llm.py +699 -0
- retexo-0.1.0/src/retexo/baselines/nmt_aligner.py +695 -0
- retexo-0.1.0/src/retexo/baselines/record.py +495 -0
- retexo-0.1.0/src/retexo/baselines/refinement.py +252 -0
- retexo-0.1.0/src/retexo/baselines/regimes.py +463 -0
- retexo-0.1.0/src/retexo/baselines/samples.py +302 -0
- retexo-0.1.0/src/retexo/baselines/schedule.py +63 -0
- retexo-0.1.0/src/retexo/baselines/scorer.py +645 -0
- retexo-0.1.0/src/retexo/baselines/sim_aligner.py +534 -0
- retexo-0.1.0/src/retexo/baselines/span_aligner.py +543 -0
- retexo-0.1.0/src/retexo/baselines/span_pair.py +1059 -0
- retexo-0.1.0/src/retexo/baselines/splits.py +25 -0
- retexo-0.1.0/src/retexo/baselines/stored_rows.py +80 -0
- retexo-0.1.0/src/retexo/baselines/sultan_aligner.py +668 -0
- retexo-0.1.0/src/retexo/baselines/synthetic.py +290 -0
- retexo-0.1.0/src/retexo/baselines/tables.py +141 -0
- retexo-0.1.0/src/retexo/baselines/tagger.py +482 -0
- retexo-0.1.0/src/retexo/baselines/typed_pointer.py +769 -0
- retexo-0.1.0/src/retexo/baselines/typer.py +372 -0
- retexo-0.1.0/src/retexo/cli.py +41 -0
- retexo-0.1.0/src/retexo/core/__init__.py +31 -0
- retexo-0.1.0/src/retexo/core/builder.py +309 -0
- retexo-0.1.0/src/retexo/core/normalize.py +63 -0
- retexo-0.1.0/src/retexo/core/oracle.py +293 -0
- retexo-0.1.0/src/retexo/core/scriba.py +122 -0
- retexo-0.1.0/src/retexo/core/script.py +137 -0
- retexo-0.1.0/src/retexo/datasets/__init__.py +46 -0
- retexo-0.1.0/src/retexo/datasets/composite.py +138 -0
- retexo-0.1.0/src/retexo/datasets/dataset.py +205 -0
- retexo-0.1.0/src/retexo/datasets/detection.py +58 -0
- retexo-0.1.0/src/retexo/datasets/generation.py +637 -0
- retexo-0.1.0/src/retexo/datasets/gold.py +147 -0
- retexo-0.1.0/src/retexo/datasets/localize.py +96 -0
- retexo-0.1.0/src/retexo/datasets/mlm_subst.py +179 -0
- retexo-0.1.0/src/retexo/datasets/negatives.py +187 -0
- retexo-0.1.0/src/retexo/datasets/parsing.py +75 -0
- retexo-0.1.0/src/retexo/datasets/passage.py +360 -0
- retexo-0.1.0/src/retexo/datasets/substitution.py +395 -0
- retexo-0.1.0/src/retexo/datasets/synthetic.py +1209 -0
- retexo-0.1.0/src/retexo/datasets/teacher.py +153 -0
- retexo-0.1.0/src/retexo/edit_typing/__init__.py +28 -0
- retexo-0.1.0/src/retexo/edit_typing/attest.py +338 -0
- retexo-0.1.0/src/retexo/edit_typing/dep_features.py +76 -0
- retexo-0.1.0/src/retexo/edit_typing/downstream.py +308 -0
- retexo-0.1.0/src/retexo/edit_typing/glosses.py +75 -0
- retexo-0.1.0/src/retexo/edit_typing/link_features.py +311 -0
- retexo-0.1.0/src/retexo/edit_typing/repair.py +167 -0
- retexo-0.1.0/src/retexo/export.py +188 -0
- retexo-0.1.0/src/retexo/formulations/__init__.py +41 -0
- retexo-0.1.0/src/retexo/formulations/base.py +146 -0
- retexo-0.1.0/src/retexo/formulations/change_detector.py +1358 -0
- retexo-0.1.0/src/retexo/formulations/checkpoint.py +79 -0
- retexo-0.1.0/src/retexo/formulations/encoding.py +111 -0
- retexo-0.1.0/src/retexo/formulations/operation_typer.py +160 -0
- retexo-0.1.0/src/retexo/formulations/pair_encoding.py +293 -0
- retexo-0.1.0/src/retexo/formulations/seq2seq_full.py +160 -0
- retexo-0.1.0/src/retexo/formulations/seq2seq_stepwise.py +312 -0
- retexo-0.1.0/src/retexo/formulations/token_classifier.py +431 -0
- retexo-0.1.0/src/retexo/formulations/typed_pointer.py +1346 -0
- retexo-0.1.0/src/retexo/formulations/word_channels.py +207 -0
- retexo-0.1.0/src/retexo/llm/__init__.py +19 -0
- retexo-0.1.0/src/retexo/llm/llm_benchmark.py +424 -0
- retexo-0.1.0/src/retexo/llm/llm_prompts.py +422 -0
- retexo-0.1.0/src/retexo/llm/script_prompts.py +222 -0
- retexo-0.1.0/src/retexo/metrics.py +298 -0
- retexo-0.1.0/src/retexo/operations/__init__.py +86 -0
- retexo-0.1.0/src/retexo/operations/base.py +238 -0
- retexo-0.1.0/src/retexo/operations/cardinality.py +53 -0
- retexo-0.1.0/src/retexo/operations/span.py +98 -0
- retexo-0.1.0/src/retexo/operations/structural.py +58 -0
- retexo-0.1.0/src/retexo/operations/token.py +360 -0
- retexo-0.1.0/src/retexo/paths.py +18 -0
- retexo-0.1.0/src/retexo/pretraining/__init__.py +30 -0
- retexo-0.1.0/src/retexo/pretraining/overlap.py +133 -0
- retexo-0.1.0/src/retexo/pretraining/pool.py +306 -0
- retexo-0.1.0/src/retexo/pretraining/resource_pairs.py +332 -0
- retexo-0.1.0/src/retexo/pretraining/stage0.py +668 -0
- retexo-0.1.0/src/retexo/refinement/__init__.py +14 -0
- retexo-0.1.0/src/retexo/refinement/grid_refiner.py +429 -0
- retexo-0.1.0/src/retexo/refinement/refine.py +159 -0
- retexo-0.1.0/src/retexo/resources/__init__.py +155 -0
- retexo-0.1.0/src/retexo/resources/entities.py +83 -0
- retexo-0.1.0/src/retexo/resources/morphology.py +229 -0
- retexo-0.1.0/src/retexo/resources/vectors.py +83 -0
- retexo-0.1.0/src/retexo/resources/wordnet.py +226 -0
- retexo-0.1.0/src/retexo/training.py +220 -0
- retexo-0.1.0/src/retexo/unified/__init__.py +33 -0
- retexo-0.1.0/src/retexo/unified/decoder.py +367 -0
- retexo-0.1.0/src/retexo/unified/gate.py +93 -0
- retexo-0.1.0/src/retexo/unified/model.py +199 -0
- retexo-0.1.0/src/retexo/unified/runs.py +130 -0
- retexo-0.1.0/src/retexo/unified/train.py +320 -0
- retexo-0.1.0/src/retexo_gui/__init__.py +0 -0
- retexo-0.1.0/src/retexo_gui/annotate.py +396 -0
- retexo-0.1.0/src/retexo_gui/app.py +696 -0
retexo-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Julian Schelb, Michael Wittweiler, Marie Revellio
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
retexo-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: retexo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Retexo explains Latin text reuse word by word: it aligns a reusing passage to its source and names the operation behind every link.
|
|
5
|
+
License: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Keywords: latin,intertextuality,text reuse,word alignment,edit scripts,digital humanities
|
|
8
|
+
Author: Julian Schelb
|
|
9
|
+
Author-email: julian.schelb@uni-konstanz.de
|
|
10
|
+
Requires-Python: >=3.10,<3.15
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Natural Language :: Latin
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Provides-Extra: gui
|
|
24
|
+
Provides-Extra: lexical
|
|
25
|
+
Provides-Extra: llm
|
|
26
|
+
Provides-Extra: plots
|
|
27
|
+
Requires-Dist: anthropic (>=0.30.0) ; extra == "llm"
|
|
28
|
+
Requires-Dist: build ; extra == "dev"
|
|
29
|
+
Requires-Dist: cltk (>=1.1.0,<2.0.0) ; (python_version < "3.13") and (extra == "lexical")
|
|
30
|
+
Requires-Dist: gensim (>=4.3.0,<5.0.0) ; extra == "lexical"
|
|
31
|
+
Requires-Dist: gradio (>=5.49.1) ; extra == "gui"
|
|
32
|
+
Requires-Dist: huggingface-hub (>=0.20.0)
|
|
33
|
+
Requires-Dist: locisimiles (>=2.1.3)
|
|
34
|
+
Requires-Dist: matplotlib (>=3.7.0) ; extra == "plots"
|
|
35
|
+
Requires-Dist: mkdocs (>=1.5.0) ; extra == "dev"
|
|
36
|
+
Requires-Dist: mkdocs-material (>=9.0.0) ; extra == "dev"
|
|
37
|
+
Requires-Dist: mkdocstrings[python] (>=0.24.0) ; extra == "dev"
|
|
38
|
+
Requires-Dist: nltk (>=3.8.0) ; extra == "lexical"
|
|
39
|
+
Requires-Dist: numpy (>=1.24.0,<3.0.0)
|
|
40
|
+
Requires-Dist: pandas (>=2.0.0,<3.0.0)
|
|
41
|
+
Requires-Dist: peft (>=0.10.0) ; extra == "llm"
|
|
42
|
+
Requires-Dist: poethepoet (>=0.24.0) ; extra == "dev"
|
|
43
|
+
Requires-Dist: pre-commit (>=3.5.0) ; extra == "dev"
|
|
44
|
+
Requires-Dist: pytest (>=8.0.0,<10.0.0) ; extra == "dev"
|
|
45
|
+
Requires-Dist: pytest-cov (>=4.0.0) ; extra == "dev"
|
|
46
|
+
Requires-Dist: ruff (>=0.8.0) ; extra == "dev"
|
|
47
|
+
Requires-Dist: scikit-learn (>=1.3.0,<2.0.0)
|
|
48
|
+
Requires-Dist: scipy (>=1.10.0,<2.0.0)
|
|
49
|
+
Requires-Dist: spacy (>=3.5.0,<4.0.0) ; extra == "lexical"
|
|
50
|
+
Requires-Dist: stanza (>=1.7.0) ; extra == "lexical"
|
|
51
|
+
Requires-Dist: torch (>=2.0.0,<3.0.0)
|
|
52
|
+
Requires-Dist: transformers (>=4.30.0,<5.0.0)
|
|
53
|
+
Requires-Dist: twine ; extra == "dev"
|
|
54
|
+
Project-URL: Changelog, https://github.com/julianschelb/retexo/blob/main/CHANGELOG.md
|
|
55
|
+
Project-URL: Documentation, https://julianschelb.github.io/retexo/
|
|
56
|
+
Project-URL: Demo, https://julianschelb.github.io/retexo/demo/
|
|
57
|
+
Project-URL: Dataset, https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts
|
|
58
|
+
Project-URL: Homepage, https://julianschelb.github.io/retexo/
|
|
59
|
+
Project-URL: Issues, https://github.com/julianschelb/retexo/issues
|
|
60
|
+
Project-URL: Repository, https://github.com/julianschelb/retexo
|
|
61
|
+
Description-Content-Type: text/markdown
|
|
62
|
+
|
|
63
|
+
# Retexo
|
|
64
|
+
|
|
65
|
+
[](https://github.com/julianschelb/retexo/actions/workflows/ci.yml)
|
|
66
|
+
[](https://julianschelb.github.io/retexo/)
|
|
67
|
+
[](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
|
|
68
|
+
[](https://github.com/julianschelb/retexo/blob/main/LICENSE)
|
|
69
|
+
|
|
70
|
+
**Word-level explanations of text reuse in Latin literature.**
|
|
71
|
+
|
|
72
|
+
When a later author quotes or alludes to an earlier one, the interesting question is *how* the source was reworked.
|
|
73
|
+
Retexo answers it word by word: it aligns every word of the reusing passage to the source word it takes up, or to none,
|
|
74
|
+
and names the operation behind each link, such as a copy, a change of inflection, or a substitution. The result is an
|
|
75
|
+
*edit script*: a set of labeled links between two passages that a philologist can read and contest.
|
|
76
|
+
|
|
77
|
+
Retexo learns these scripts from very little annotation. Synthetic pairs, built by applying known operations to Latin
|
|
78
|
+
text, teach a pointer network the task before a single pair is annotated, and active learning chooses the few pairs an
|
|
79
|
+
annotator corrects.
|
|
80
|
+
|
|
81
|
+
- **Predicted edit scripts** for all 1,490 references of the [Loci Similes](https://arxiv.org/abs/2601.07533) benchmark:
|
|
82
|
+
[Hugging Face dataset](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
|
|
83
|
+
- **Review app** to browse the script of every reference, drawn as arrows between the passages:
|
|
84
|
+
[julianschelb.github.io/retexo/demo](https://julianschelb.github.io/retexo/demo/)
|
|
85
|
+
- **Documentation:** [julianschelb.github.io/retexo](https://julianschelb.github.io/retexo/)
|
|
86
|
+
|
|
87
|
+
## Quick start
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install "retexo @ git+https://github.com/julianschelb/retexo"
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Read the predictions, with the Hugging Face `datasets` library:
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
from datasets import load_dataset
|
|
97
|
+
|
|
98
|
+
scripts = load_dataset("julian-schelb/latin-classical-intertextuality-edit-scripts", split="fold_4")
|
|
99
|
+
pair = scripts[0]
|
|
100
|
+
|
|
101
|
+
for link in pair["links"]:
|
|
102
|
+
print(pair["reuse"]["tokens"][link["reuse"]], "<-", pair["source"]["tokens"][link["source"]], link["label"])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Convert the raw predictions of a run into the same format:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
retexo export runs/f0/predictions.jsonl runs/f1/predictions.jsonl runs/f2/predictions.jsonl \
|
|
109
|
+
runs/f3/predictions.jsonl runs/f4/predictions.jsonl --out export/ --format parquet
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Labels
|
|
113
|
+
|
|
114
|
+
| Label | Meaning |
|
|
115
|
+
|---|---|
|
|
116
|
+
| `COPY` | the same form, spelling variants folded |
|
|
117
|
+
| `INFLECT` | the same lemma in another form |
|
|
118
|
+
| `SUBST` | another lemma in the same slot, optionally with a named relation such as a synonym |
|
|
119
|
+
| `SPLIT`, `MERGE` | two reuse words for one source word, or one for two |
|
|
120
|
+
| `INS`, `FRAME`, `DEL` | an unlinked reuse word, a word of a citing formula, an unclaimed source word |
|
|
121
|
+
|
|
122
|
+
See the [documentation](https://julianschelb.github.io/retexo/labels/) for the conventions.
|
|
123
|
+
|
|
124
|
+
## Repository layout
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
src/retexo/ the package: model, generator, active learning, baselines, scorer
|
|
128
|
+
src/retexo_gui/ Gradio demo and annotation app (gui extra)
|
|
129
|
+
tests/ pytest suite
|
|
130
|
+
examples/ notebooks on the operations, the generator, the passage class and the decoder
|
|
131
|
+
docs/ documentation (MkDocs)
|
|
132
|
+
webapp/ the review app (React, Vite), published under /demo/
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
The experiments behind the paper live in a separate repository that installs retexo as a dependency.
|
|
136
|
+
|
|
137
|
+
## Development
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
pip install -e ".[dev,lexical]"
|
|
141
|
+
pytest
|
|
142
|
+
mkdocs serve
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
See [docs/development.md](https://julianschelb.github.io/retexo/development/).
|
|
146
|
+
|
|
147
|
+
## Authors
|
|
148
|
+
|
|
149
|
+
Julian Schelb (University of Konstanz), Michael Wittweiler (University of Zurich), Marie Revellio (University of Konstanz).
|
|
150
|
+
|
|
151
|
+
## License
|
|
152
|
+
|
|
153
|
+
MIT
|
|
154
|
+
|
retexo-0.1.0/README.md
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Retexo
|
|
2
|
+
|
|
3
|
+
[](https://github.com/julianschelb/retexo/actions/workflows/ci.yml)
|
|
4
|
+
[](https://julianschelb.github.io/retexo/)
|
|
5
|
+
[](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
|
|
6
|
+
[](https://github.com/julianschelb/retexo/blob/main/LICENSE)
|
|
7
|
+
|
|
8
|
+
**Word-level explanations of text reuse in Latin literature.**
|
|
9
|
+
|
|
10
|
+
When a later author quotes or alludes to an earlier one, the interesting question is *how* the source was reworked.
|
|
11
|
+
Retexo answers it word by word: it aligns every word of the reusing passage to the source word it takes up, or to none,
|
|
12
|
+
and names the operation behind each link, such as a copy, a change of inflection, or a substitution. The result is an
|
|
13
|
+
*edit script*: a set of labeled links between two passages that a philologist can read and contest.
|
|
14
|
+
|
|
15
|
+
Retexo learns these scripts from very little annotation. Synthetic pairs, built by applying known operations to Latin
|
|
16
|
+
text, teach a pointer network the task before a single pair is annotated, and active learning chooses the few pairs an
|
|
17
|
+
annotator corrects.
|
|
18
|
+
|
|
19
|
+
- **Predicted edit scripts** for all 1,490 references of the [Loci Similes](https://arxiv.org/abs/2601.07533) benchmark:
|
|
20
|
+
[Hugging Face dataset](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
|
|
21
|
+
- **Review app** to browse the script of every reference, drawn as arrows between the passages:
|
|
22
|
+
[julianschelb.github.io/retexo/demo](https://julianschelb.github.io/retexo/demo/)
|
|
23
|
+
- **Documentation:** [julianschelb.github.io/retexo](https://julianschelb.github.io/retexo/)
|
|
24
|
+
|
|
25
|
+
## Quick start
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install "retexo @ git+https://github.com/julianschelb/retexo"
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Read the predictions, with the Hugging Face `datasets` library:
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from datasets import load_dataset
|
|
35
|
+
|
|
36
|
+
scripts = load_dataset("julian-schelb/latin-classical-intertextuality-edit-scripts", split="fold_4")
|
|
37
|
+
pair = scripts[0]
|
|
38
|
+
|
|
39
|
+
for link in pair["links"]:
|
|
40
|
+
print(pair["reuse"]["tokens"][link["reuse"]], "<-", pair["source"]["tokens"][link["source"]], link["label"])
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Convert the raw predictions of a run into the same format:
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
retexo export runs/f0/predictions.jsonl runs/f1/predictions.jsonl runs/f2/predictions.jsonl \
|
|
47
|
+
runs/f3/predictions.jsonl runs/f4/predictions.jsonl --out export/ --format parquet
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Labels
|
|
51
|
+
|
|
52
|
+
| Label | Meaning |
|
|
53
|
+
|---|---|
|
|
54
|
+
| `COPY` | the same form, spelling variants folded |
|
|
55
|
+
| `INFLECT` | the same lemma in another form |
|
|
56
|
+
| `SUBST` | another lemma in the same slot, optionally with a named relation such as a synonym |
|
|
57
|
+
| `SPLIT`, `MERGE` | two reuse words for one source word, or one for two |
|
|
58
|
+
| `INS`, `FRAME`, `DEL` | an unlinked reuse word, a word of a citing formula, an unclaimed source word |
|
|
59
|
+
|
|
60
|
+
See the [documentation](https://julianschelb.github.io/retexo/labels/) for the conventions.
|
|
61
|
+
|
|
62
|
+
## Repository layout
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
src/retexo/ the package: model, generator, active learning, baselines, scorer
|
|
66
|
+
src/retexo_gui/ Gradio demo and annotation app (gui extra)
|
|
67
|
+
tests/ pytest suite
|
|
68
|
+
examples/ notebooks on the operations, the generator, the passage class and the decoder
|
|
69
|
+
docs/ documentation (MkDocs)
|
|
70
|
+
webapp/ the review app (React, Vite), published under /demo/
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The experiments behind the paper live in a separate repository that installs retexo as a dependency.
|
|
74
|
+
|
|
75
|
+
## Development
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install -e ".[dev,lexical]"
|
|
79
|
+
pytest
|
|
80
|
+
mkdocs serve
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
See [docs/development.md](https://julianschelb.github.io/retexo/development/).
|
|
84
|
+
|
|
85
|
+
## Authors
|
|
86
|
+
|
|
87
|
+
Julian Schelb (University of Konstanz), Michael Wittweiler (University of Zurich), Marie Revellio (University of Konstanz).
|
|
88
|
+
|
|
89
|
+
## License
|
|
90
|
+
|
|
91
|
+
MIT
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "retexo"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Retexo explains Latin text reuse word by word: it aligns a reusing passage to its source and names the operation behind every link."
|
|
5
|
+
authors = [
|
|
6
|
+
{name = "Julian Schelb", email = "julian.schelb@uni-konstanz.de"},
|
|
7
|
+
{name = "Michael Wittweiler", email = "michael.wittweiler@sglp.uzh.ch"},
|
|
8
|
+
{name = "Marie Revellio"},
|
|
9
|
+
]
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
requires-python = ">=3.10,<3.15"
|
|
13
|
+
keywords = ["latin", "intertextuality", "text reuse", "word alignment", "edit scripts", "digital humanities"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Natural Language :: Latin",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
25
|
+
"Topic :: Text Processing :: Linguistic",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"numpy (>=1.24.0,<3.0.0)",
|
|
29
|
+
"scipy (>=1.10.0,<2.0.0)",
|
|
30
|
+
"pandas (>=2.0.0,<3.0.0)",
|
|
31
|
+
"scikit-learn (>=1.3.0,<2.0.0)",
|
|
32
|
+
"torch (>=2.0.0,<3.0.0)",
|
|
33
|
+
"transformers (>=4.30.0,<5.0.0)",
|
|
34
|
+
"huggingface-hub (>=0.20.0)",
|
|
35
|
+
"locisimiles (>=2.1.3)",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Homepage = "https://julianschelb.github.io/retexo/"
|
|
40
|
+
Documentation = "https://julianschelb.github.io/retexo/"
|
|
41
|
+
Repository = "https://github.com/julianschelb/retexo"
|
|
42
|
+
Demo = "https://julianschelb.github.io/retexo/demo/"
|
|
43
|
+
Dataset = "https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts"
|
|
44
|
+
Changelog = "https://github.com/julianschelb/retexo/blob/main/CHANGELOG.md"
|
|
45
|
+
Issues = "https://github.com/julianschelb/retexo/issues"
|
|
46
|
+
|
|
47
|
+
[project.optional-dependencies]
|
|
48
|
+
gui = [
|
|
49
|
+
"gradio (>=5.49.1)",
|
|
50
|
+
]
|
|
51
|
+
lexical = [
|
|
52
|
+
# The symbolic typer and the lexical evidence look words up in these resources.
|
|
53
|
+
"cltk (>=1.1.0,<2.0.0) ; python_version < '3.13'",
|
|
54
|
+
"spacy (>=3.5.0,<4.0.0)",
|
|
55
|
+
"stanza (>=1.7.0)",
|
|
56
|
+
"nltk (>=3.8.0)",
|
|
57
|
+
"gensim (>=4.3.0,<5.0.0)",
|
|
58
|
+
]
|
|
59
|
+
llm = [
|
|
60
|
+
"anthropic (>=0.30.0)",
|
|
61
|
+
"peft (>=0.10.0)",
|
|
62
|
+
]
|
|
63
|
+
plots = [
|
|
64
|
+
"matplotlib (>=3.7.0)",
|
|
65
|
+
]
|
|
66
|
+
dev = [
|
|
67
|
+
"pytest (>=8.0.0,<10.0.0)",
|
|
68
|
+
"pytest-cov (>=4.0.0)",
|
|
69
|
+
"poethepoet (>=0.24.0)",
|
|
70
|
+
"mkdocs (>=1.5.0)",
|
|
71
|
+
"mkdocs-material (>=9.0.0)",
|
|
72
|
+
"mkdocstrings[python] (>=0.24.0)",
|
|
73
|
+
"ruff (>=0.8.0)",
|
|
74
|
+
"pre-commit (>=3.5.0)",
|
|
75
|
+
"build",
|
|
76
|
+
"twine",
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
[project.scripts]
|
|
80
|
+
retexo = "retexo.cli:main"
|
|
81
|
+
retexo-gui = "retexo_gui.app:main"
|
|
82
|
+
|
|
83
|
+
[tool.poetry]
|
|
84
|
+
packages = [
|
|
85
|
+
{include = "retexo", from = "src"},
|
|
86
|
+
{include = "retexo_gui", from = "src"},
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
[build-system]
|
|
90
|
+
requires = ["poetry-core>=2.0.0,<3.0.0"]
|
|
91
|
+
build-backend = "poetry.core.masonry.api"
|
|
92
|
+
|
|
93
|
+
[tool.pytest.ini_options]
|
|
94
|
+
testpaths = ["tests"]
|
|
95
|
+
python_files = ["test_*.py"]
|
|
96
|
+
python_classes = ["Test*"]
|
|
97
|
+
python_functions = ["test_*"]
|
|
98
|
+
addopts = "-v"
|
|
99
|
+
pythonpath = ["src"]
|
|
100
|
+
|
|
101
|
+
[tool.poe.tasks]
|
|
102
|
+
test = "pytest"
|
|
103
|
+
test-cov = "pytest --cov=retexo --cov-report=term-missing"
|
|
104
|
+
lint = "ruff check src/ tests/"
|
|
105
|
+
docs = "mkdocs serve"
|
|
106
|
+
docs-build = "mkdocs build"
|
|
107
|
+
webapp = "npm --prefix webapp run dev"
|
|
108
|
+
|
|
109
|
+
# ---------- Ruff ----------
|
|
110
|
+
# The research code was written without a formatter; CI checks for errors only (syntax errors, undefined names).
|
|
111
|
+
|
|
112
|
+
[tool.ruff]
|
|
113
|
+
target-version = "py310"
|
|
114
|
+
line-length = 120
|
|
115
|
+
src = ["src", "tests"]
|
|
116
|
+
|
|
117
|
+
[tool.ruff.lint]
|
|
118
|
+
select = ["E9", "F63", "F7", "F82"]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# retexo/__init__.py
|
|
2
|
+
"""
|
|
3
|
+
Typed edit scripts for Latin intertextual reuse.
|
|
4
|
+
|
|
5
|
+
Building a variant and checking it round-trips:
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from retexo import VariantBuilder, Scriba
|
|
9
|
+
|
|
10
|
+
builder = VariantBuilder("uox faucibus haesit")
|
|
11
|
+
builder.keep(0).syn(1, "gutture").keep(2)
|
|
12
|
+
variant, script = builder.build()
|
|
13
|
+
|
|
14
|
+
Scriba().verify(script, "uox faucibus haesit".split(), variant) # True
|
|
15
|
+
Scriba().execute(script.invert(), variant) # back to the source
|
|
16
|
+
```
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
__version__ = "0.1.0"
|
|
21
|
+
|
|
22
|
+
from retexo.core.builder import VariantBuilder
|
|
23
|
+
from retexo.core.oracle import (
|
|
24
|
+
Aligner,
|
|
25
|
+
EditPlanOracle,
|
|
26
|
+
GreedyAligner,
|
|
27
|
+
OptimalAligner,
|
|
28
|
+
OracleConfig,
|
|
29
|
+
)
|
|
30
|
+
from retexo.resources import Resources
|
|
31
|
+
from retexo.core.normalize import normalize
|
|
32
|
+
from retexo.operations import (
|
|
33
|
+
ALL_OPERATIONS,
|
|
34
|
+
EditOperation,
|
|
35
|
+
Level,
|
|
36
|
+
Operation,
|
|
37
|
+
OperationRegistry,
|
|
38
|
+
Role,
|
|
39
|
+
)
|
|
40
|
+
from retexo.core.scriba import Scriba, Validity
|
|
41
|
+
from retexo.core.script import CostModel, EditScript
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"ALL_OPERATIONS",
|
|
45
|
+
"Aligner",
|
|
46
|
+
"EditPlanOracle",
|
|
47
|
+
"GreedyAligner",
|
|
48
|
+
"OptimalAligner",
|
|
49
|
+
"OracleConfig",
|
|
50
|
+
"Resources",
|
|
51
|
+
"CostModel",
|
|
52
|
+
"EditOperation",
|
|
53
|
+
"EditScript",
|
|
54
|
+
"Level",
|
|
55
|
+
"Operation",
|
|
56
|
+
"OperationRegistry",
|
|
57
|
+
"Role",
|
|
58
|
+
"Scriba",
|
|
59
|
+
"Validity",
|
|
60
|
+
"VariantBuilder",
|
|
61
|
+
"normalize",
|
|
62
|
+
]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# retexo/aligners/__init__.py
|
|
2
|
+
"""Alignment models and the corpus-level and pair-level policies that turn their scores into
|
|
3
|
+
one source per reuse word: the classical pointer-score assignment policies, the bidirectional
|
|
4
|
+
agreement rules, the orthographic sameness predicates, the decoder that derives a full script
|
|
5
|
+
from a typed alignment, the Sinkhorn and whole-script DP variants, and the contextual embedding
|
|
6
|
+
aligner."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from retexo.aligners.agreement import AgreementDecoder, PairSwap
|
|
10
|
+
from retexo.aligners.aligner import ContextualAligner
|
|
11
|
+
from retexo.aligners.assignment import AssignmentPolicy, Reranker
|
|
12
|
+
from retexo.aligners.decode import ScriptDecoder
|
|
13
|
+
from retexo.aligners.global_decode import GlobalDecoder
|
|
14
|
+
from retexo.aligners.sameness import SamenessPolicy
|
|
15
|
+
from retexo.aligners.sinkhorn import SinkhornBalancer
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"AgreementDecoder",
|
|
19
|
+
"AssignmentPolicy",
|
|
20
|
+
"ContextualAligner",
|
|
21
|
+
"GlobalDecoder",
|
|
22
|
+
"PairSwap",
|
|
23
|
+
"Reranker",
|
|
24
|
+
"SamenessPolicy",
|
|
25
|
+
"ScriptDecoder",
|
|
26
|
+
"SinkhornBalancer",
|
|
27
|
+
]
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# retexo/aligners/agreement.py
|
|
2
|
+
"""
|
|
3
|
+
Bidirectional-agreement decoding rules: score both directions, keep what both accept.
|
|
4
|
+
|
|
5
|
+
Renamed from the preliminary experiment module E29; kept under its own name,
|
|
6
|
+
unchanged, because ``retexo.baselines.decoder`` imports it directly and
|
|
7
|
+
several implementation notes say explicitly to reuse it verbatim.
|
|
8
|
+
|
|
9
|
+
Our pointer scores reuse -> source and its null is a scalar per reuse word.
|
|
10
|
+
awesome-align's null is a test on two directions: a link exists only if the
|
|
11
|
+
reuse word picks the source word *and* the source word picks it back.
|
|
12
|
+
SimAlign's mutual argmax is the same test without a threshold; its entropy
|
|
13
|
+
rule is a third signal. All three work on two sets of rows -- the forward
|
|
14
|
+
rows the pointer scores, and the reverse rows a swapped-sides pass produces --
|
|
15
|
+
and need no training.
|
|
16
|
+
|
|
17
|
+
Rows are the pointer's score-row shape: per reuse word, [(source, p) ...] best
|
|
18
|
+
first, with the null as source -1. Reverse rows are the same shape with the
|
|
19
|
+
roles swapped: per *source* word, [(reuse, p) ...].
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import math
|
|
25
|
+
from typing import ClassVar, Dict, List, Sequence, Tuple
|
|
26
|
+
|
|
27
|
+
from retexo.formulations.change_detector import ChangeExample
|
|
28
|
+
|
|
29
|
+
Row = List[Tuple[int, float]]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class PairSwap:
|
|
33
|
+
"""Building a pair with the roles reversed, for the reverse-direction pass
|
|
34
|
+
and for symmetric training.
|
|
35
|
+
|
|
36
|
+
Example:
|
|
37
|
+
```python
|
|
38
|
+
reverse_pair = PairSwap.example(example) # untrained: p(reuse word | source word)
|
|
39
|
+
reverse_training_pair = PairSwap.labelled(example) # a training pair with inverted labels
|
|
40
|
+
```
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
#: relations that change name when the sides change
|
|
44
|
+
INVERT: ClassVar[Dict[str, str]] = {"HYPER": "HYPO", "HYPO": "HYPER", "SPLIT": "MERGE", "MERGE": "SPLIT"}
|
|
45
|
+
|
|
46
|
+
@staticmethod
|
|
47
|
+
def example(example: ChangeExample) -> ChangeExample:
|
|
48
|
+
"""The pair with the roles exchanged: the source passage sits in the reuse
|
|
49
|
+
slot, so the model scores p(reuse word | source word)."""
|
|
50
|
+
s, t = list(example.target_tokens), list(example.source_tokens) # swapped on purpose
|
|
51
|
+
return ChangeExample(source_tokens=s, target_tokens=t, labels=[0] * len(t),
|
|
52
|
+
operations=["COPY"] * len(t), n_operations=0,
|
|
53
|
+
source_labels=[0] * len(s), source_operations=["COPY"] * len(s),
|
|
54
|
+
alignments=[-1] * len(t), fine_operations=["INS"] * len(t),
|
|
55
|
+
frame_labels=[0] * len(t), link_features=[None] * len(t))
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def labelled(cls, example: ChangeExample) -> ChangeExample:
|
|
59
|
+
"""A *training* pair with the sides exchanged and its labels inverted.
|
|
60
|
+
|
|
61
|
+
One-to-one links invert exactly; HYPER/HYPO and SPLIT/MERGE change name;
|
|
62
|
+
a gold lexical change of unknown kind ("?") stays unknown; the former reuse
|
|
63
|
+
side's inserted words become deleted source words; frames do not exist on
|
|
64
|
+
a source passage, so the new reuse side carries none. Evidence grids are
|
|
65
|
+
not transposed (the featurizer is not symmetric) -- recompute them."""
|
|
66
|
+
old_s, old_t = list(example.source_tokens), list(example.target_tokens)
|
|
67
|
+
links = example.alignments or [-1] * len(old_t)
|
|
68
|
+
fine = example.fine_operations or ["INS"] * len(old_t)
|
|
69
|
+
new_s, new_t = old_t, old_s
|
|
70
|
+
new_links = [-1] * len(new_t)
|
|
71
|
+
new_fine = ["INS"] * len(new_t)
|
|
72
|
+
for t, s in enumerate(links):
|
|
73
|
+
if s is not None and 0 <= s < len(new_t) and new_links[s] < 0:
|
|
74
|
+
new_links[s] = t
|
|
75
|
+
tag = fine[t] if t < len(fine) else "?"
|
|
76
|
+
new_fine[s] = cls.INVERT.get(tag, tag)
|
|
77
|
+
consumed = {t for t in new_links if t >= 0}
|
|
78
|
+
new_source_labels = [0 if t in consumed else 1 for t in range(len(new_s))]
|
|
79
|
+
coarse = ["COPY" if (k == "NOP") else ("INS" if k == "INS" else "SUBST") for k in new_fine]
|
|
80
|
+
ex = ChangeExample(source_tokens=new_s, target_tokens=new_t,
|
|
81
|
+
labels=[0 if op == "COPY" else 1 for op in coarse], operations=coarse,
|
|
82
|
+
n_operations=sum(1 for op in coarse if op != "COPY"),
|
|
83
|
+
source_labels=new_source_labels,
|
|
84
|
+
source_operations=["DEL" if d else "COPY" for d in new_source_labels],
|
|
85
|
+
alignments=new_links, fine_operations=new_fine,
|
|
86
|
+
frame_labels=[0] * len(new_t), link_features=[None] * len(new_t))
|
|
87
|
+
object.__setattr__(ex, "swapped_from", True)
|
|
88
|
+
return ex
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class AgreementDecoder:
|
|
92
|
+
"""Bidirectional-agreement rules, each returning restricted forward rows for
|
|
93
|
+
the Hungarian assignment.
|
|
94
|
+
|
|
95
|
+
Example:
|
|
96
|
+
```python
|
|
97
|
+
kept = AgreementDecoder.mutual_argmax(rows, reverse_rows)
|
|
98
|
+
kept = AgreementDecoder.intersect(rows, reverse_rows, c=0.4)
|
|
99
|
+
```
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
@staticmethod
|
|
103
|
+
def prob(row: Row, index: int) -> float:
|
|
104
|
+
return dict(row).get(index, 0.0)
|
|
105
|
+
|
|
106
|
+
@staticmethod
|
|
107
|
+
def top(row: Row) -> int:
|
|
108
|
+
"""The best entry, the null (-1) included."""
|
|
109
|
+
return row[0][0] if row else -1
|
|
110
|
+
|
|
111
|
+
@classmethod
|
|
112
|
+
def reverse_links(cls, rows_rev: Sequence[Row], n_reuse: int) -> List[int]:
|
|
113
|
+
"""The reverse direction read as an aligner of the reuse side: reuse word t
|
|
114
|
+
links to the source word s whose best pick is t (first such s), else -1."""
|
|
115
|
+
links = [-1] * n_reuse
|
|
116
|
+
for s, row in enumerate(rows_rev):
|
|
117
|
+
t = cls.top(row)
|
|
118
|
+
if 0 <= t < n_reuse and links[t] < 0:
|
|
119
|
+
links[t] = s
|
|
120
|
+
return links
|
|
121
|
+
|
|
122
|
+
@staticmethod
|
|
123
|
+
def entropy(row: Row) -> float:
|
|
124
|
+
ps = [p for _, p in row if p > 1e-12]
|
|
125
|
+
z = sum(ps) or 1.0
|
|
126
|
+
return -sum(p / z * math.log(p / z) for p in ps)
|
|
127
|
+
|
|
128
|
+
@classmethod
|
|
129
|
+
def intersect(cls, rows: Sequence[Row], rows_rev: Sequence[Row], c: float) -> List[Row]:
|
|
130
|
+
"""awesome-align: keep (t, s) iff p(s|t) > c and p(t|s) > c. A word whose
|
|
131
|
+
every candidate fails keeps only its null."""
|
|
132
|
+
out = []
|
|
133
|
+
for t, row in enumerate(rows):
|
|
134
|
+
kept = [(s, p) for s, p in row
|
|
135
|
+
if s >= 0 and p > c and s < len(rows_rev) and cls.prob(rows_rev[s], t) > c]
|
|
136
|
+
out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
@classmethod
|
|
140
|
+
def mutual_argmax(cls, rows: Sequence[Row], rows_rev: Sequence[Row]) -> List[Row]:
|
|
141
|
+
"""SimAlign Argmax: keep (t, s) iff s is t's best and t is s's best."""
|
|
142
|
+
out = []
|
|
143
|
+
for t, row in enumerate(rows):
|
|
144
|
+
s = cls.top(row)
|
|
145
|
+
kept = [(s, cls.prob(row, s))] if s >= 0 and s < len(rows_rev) and cls.top(rows_rev[s]) == t else []
|
|
146
|
+
out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
|
|
147
|
+
return out
|
|
148
|
+
|
|
149
|
+
@classmethod
|
|
150
|
+
def entropy_filter(cls, rows: Sequence[Row], rows_rev: Sequence[Row], tau: float) -> List[Row]:
|
|
151
|
+
"""SimAlign's null: drop every link of a reuse word whose row entropy, and of
|
|
152
|
+
a source word whose column entropy, are both above tau (normalised by log n)."""
|
|
153
|
+
out = []
|
|
154
|
+
h_col = [cls.entropy(r) / max(math.log(max(len(r), 2)), 1e-9) for r in rows_rev]
|
|
155
|
+
for t, row in enumerate(rows):
|
|
156
|
+
h_row = cls.entropy(row) / max(math.log(max(len(row), 2)), 1e-9)
|
|
157
|
+
kept = [(s, p) for s, p in row if s >= 0
|
|
158
|
+
and min(h_row, h_col[s] if s < len(h_col) else 1.0) <= tau]
|
|
159
|
+
out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
|
|
160
|
+
return out
|
|
161
|
+
|
|
162
|
+
@classmethod
|
|
163
|
+
def null_scale(cls, rows: Sequence[Row], k: float) -> List[Row]:
|
|
164
|
+
"""Our null re-weighted by k and renormalised (E18's move, at the raw pointer)."""
|
|
165
|
+
out = []
|
|
166
|
+
for row in rows:
|
|
167
|
+
scaled = [(s, p * (k if s < 0 else 1.0)) for s, p in row]
|
|
168
|
+
z = sum(p for _, p in scaled) or 1.0
|
|
169
|
+
out.append(sorted([(s, p / z) for s, p in scaled], key=lambda x: -x[1]))
|
|
170
|
+
return out
|
|
171
|
+
|
|
172
|
+
@classmethod
|
|
173
|
+
def compose(cls, *restricted: Sequence[Row]) -> List[Row]:
|
|
174
|
+
"""Keep a candidate only if every rule kept it; the null is the first rule's."""
|
|
175
|
+
out = []
|
|
176
|
+
for rows in zip(*restricted):
|
|
177
|
+
keep = set.intersection(*[{s for s, _ in r if s >= 0} for r in rows])
|
|
178
|
+
first = rows[0]
|
|
179
|
+
out.append(sorted([(s, p) for s, p in first if s in keep] + [(-1, cls.prob(first, -1))],
|
|
180
|
+
key=lambda x: -x[1]))
|
|
181
|
+
return out
|