retexo 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. retexo-0.1.0/LICENSE +21 -0
  2. retexo-0.1.0/PKG-INFO +154 -0
  3. retexo-0.1.0/README.md +91 -0
  4. retexo-0.1.0/pyproject.toml +118 -0
  5. retexo-0.1.0/src/retexo/__init__.py +62 -0
  6. retexo-0.1.0/src/retexo/aligners/__init__.py +27 -0
  7. retexo-0.1.0/src/retexo/aligners/agreement.py +181 -0
  8. retexo-0.1.0/src/retexo/aligners/aligner.py +87 -0
  9. retexo-0.1.0/src/retexo/aligners/assignment.py +269 -0
  10. retexo-0.1.0/src/retexo/aligners/decode.py +328 -0
  11. retexo-0.1.0/src/retexo/aligners/global_decode.py +91 -0
  12. retexo-0.1.0/src/retexo/aligners/sameness.py +123 -0
  13. retexo-0.1.0/src/retexo/aligners/sinkhorn.py +109 -0
  14. retexo-0.1.0/src/retexo/baselines/__init__.py +78 -0
  15. retexo-0.1.0/src/retexo/baselines/adapters.py +410 -0
  16. retexo-0.1.0/src/retexo/baselines/annotations.py +373 -0
  17. retexo-0.1.0/src/retexo/baselines/augment.py +99 -0
  18. retexo-0.1.0/src/retexo/baselines/base.py +219 -0
  19. retexo-0.1.0/src/retexo/baselines/curriculum.py +102 -0
  20. retexo-0.1.0/src/retexo/baselines/curves.py +139 -0
  21. retexo-0.1.0/src/retexo/baselines/decoder.py +526 -0
  22. retexo-0.1.0/src/retexo/baselines/downstream.py +355 -0
  23. retexo-0.1.0/src/retexo/baselines/early_stopping.py +485 -0
  24. retexo-0.1.0/src/retexo/baselines/em_aligner.py +628 -0
  25. retexo-0.1.0/src/retexo/baselines/floors.py +84 -0
  26. retexo-0.1.0/src/retexo/baselines/full_system.py +420 -0
  27. retexo-0.1.0/src/retexo/baselines/labels.py +333 -0
  28. retexo-0.1.0/src/retexo/baselines/llm.py +699 -0
  29. retexo-0.1.0/src/retexo/baselines/nmt_aligner.py +695 -0
  30. retexo-0.1.0/src/retexo/baselines/record.py +495 -0
  31. retexo-0.1.0/src/retexo/baselines/refinement.py +252 -0
  32. retexo-0.1.0/src/retexo/baselines/regimes.py +463 -0
  33. retexo-0.1.0/src/retexo/baselines/samples.py +302 -0
  34. retexo-0.1.0/src/retexo/baselines/schedule.py +63 -0
  35. retexo-0.1.0/src/retexo/baselines/scorer.py +645 -0
  36. retexo-0.1.0/src/retexo/baselines/sim_aligner.py +534 -0
  37. retexo-0.1.0/src/retexo/baselines/span_aligner.py +543 -0
  38. retexo-0.1.0/src/retexo/baselines/span_pair.py +1059 -0
  39. retexo-0.1.0/src/retexo/baselines/splits.py +25 -0
  40. retexo-0.1.0/src/retexo/baselines/stored_rows.py +80 -0
  41. retexo-0.1.0/src/retexo/baselines/sultan_aligner.py +668 -0
  42. retexo-0.1.0/src/retexo/baselines/synthetic.py +290 -0
  43. retexo-0.1.0/src/retexo/baselines/tables.py +141 -0
  44. retexo-0.1.0/src/retexo/baselines/tagger.py +482 -0
  45. retexo-0.1.0/src/retexo/baselines/typed_pointer.py +769 -0
  46. retexo-0.1.0/src/retexo/baselines/typer.py +372 -0
  47. retexo-0.1.0/src/retexo/cli.py +41 -0
  48. retexo-0.1.0/src/retexo/core/__init__.py +31 -0
  49. retexo-0.1.0/src/retexo/core/builder.py +309 -0
  50. retexo-0.1.0/src/retexo/core/normalize.py +63 -0
  51. retexo-0.1.0/src/retexo/core/oracle.py +293 -0
  52. retexo-0.1.0/src/retexo/core/scriba.py +122 -0
  53. retexo-0.1.0/src/retexo/core/script.py +137 -0
  54. retexo-0.1.0/src/retexo/datasets/__init__.py +46 -0
  55. retexo-0.1.0/src/retexo/datasets/composite.py +138 -0
  56. retexo-0.1.0/src/retexo/datasets/dataset.py +205 -0
  57. retexo-0.1.0/src/retexo/datasets/detection.py +58 -0
  58. retexo-0.1.0/src/retexo/datasets/generation.py +637 -0
  59. retexo-0.1.0/src/retexo/datasets/gold.py +147 -0
  60. retexo-0.1.0/src/retexo/datasets/localize.py +96 -0
  61. retexo-0.1.0/src/retexo/datasets/mlm_subst.py +179 -0
  62. retexo-0.1.0/src/retexo/datasets/negatives.py +187 -0
  63. retexo-0.1.0/src/retexo/datasets/parsing.py +75 -0
  64. retexo-0.1.0/src/retexo/datasets/passage.py +360 -0
  65. retexo-0.1.0/src/retexo/datasets/substitution.py +395 -0
  66. retexo-0.1.0/src/retexo/datasets/synthetic.py +1209 -0
  67. retexo-0.1.0/src/retexo/datasets/teacher.py +153 -0
  68. retexo-0.1.0/src/retexo/edit_typing/__init__.py +28 -0
  69. retexo-0.1.0/src/retexo/edit_typing/attest.py +338 -0
  70. retexo-0.1.0/src/retexo/edit_typing/dep_features.py +76 -0
  71. retexo-0.1.0/src/retexo/edit_typing/downstream.py +308 -0
  72. retexo-0.1.0/src/retexo/edit_typing/glosses.py +75 -0
  73. retexo-0.1.0/src/retexo/edit_typing/link_features.py +311 -0
  74. retexo-0.1.0/src/retexo/edit_typing/repair.py +167 -0
  75. retexo-0.1.0/src/retexo/export.py +188 -0
  76. retexo-0.1.0/src/retexo/formulations/__init__.py +41 -0
  77. retexo-0.1.0/src/retexo/formulations/base.py +146 -0
  78. retexo-0.1.0/src/retexo/formulations/change_detector.py +1358 -0
  79. retexo-0.1.0/src/retexo/formulations/checkpoint.py +79 -0
  80. retexo-0.1.0/src/retexo/formulations/encoding.py +111 -0
  81. retexo-0.1.0/src/retexo/formulations/operation_typer.py +160 -0
  82. retexo-0.1.0/src/retexo/formulations/pair_encoding.py +293 -0
  83. retexo-0.1.0/src/retexo/formulations/seq2seq_full.py +160 -0
  84. retexo-0.1.0/src/retexo/formulations/seq2seq_stepwise.py +312 -0
  85. retexo-0.1.0/src/retexo/formulations/token_classifier.py +431 -0
  86. retexo-0.1.0/src/retexo/formulations/typed_pointer.py +1346 -0
  87. retexo-0.1.0/src/retexo/formulations/word_channels.py +207 -0
  88. retexo-0.1.0/src/retexo/llm/__init__.py +19 -0
  89. retexo-0.1.0/src/retexo/llm/llm_benchmark.py +424 -0
  90. retexo-0.1.0/src/retexo/llm/llm_prompts.py +422 -0
  91. retexo-0.1.0/src/retexo/llm/script_prompts.py +222 -0
  92. retexo-0.1.0/src/retexo/metrics.py +298 -0
  93. retexo-0.1.0/src/retexo/operations/__init__.py +86 -0
  94. retexo-0.1.0/src/retexo/operations/base.py +238 -0
  95. retexo-0.1.0/src/retexo/operations/cardinality.py +53 -0
  96. retexo-0.1.0/src/retexo/operations/span.py +98 -0
  97. retexo-0.1.0/src/retexo/operations/structural.py +58 -0
  98. retexo-0.1.0/src/retexo/operations/token.py +360 -0
  99. retexo-0.1.0/src/retexo/paths.py +18 -0
  100. retexo-0.1.0/src/retexo/pretraining/__init__.py +30 -0
  101. retexo-0.1.0/src/retexo/pretraining/overlap.py +133 -0
  102. retexo-0.1.0/src/retexo/pretraining/pool.py +306 -0
  103. retexo-0.1.0/src/retexo/pretraining/resource_pairs.py +332 -0
  104. retexo-0.1.0/src/retexo/pretraining/stage0.py +668 -0
  105. retexo-0.1.0/src/retexo/refinement/__init__.py +14 -0
  106. retexo-0.1.0/src/retexo/refinement/grid_refiner.py +429 -0
  107. retexo-0.1.0/src/retexo/refinement/refine.py +159 -0
  108. retexo-0.1.0/src/retexo/resources/__init__.py +155 -0
  109. retexo-0.1.0/src/retexo/resources/entities.py +83 -0
  110. retexo-0.1.0/src/retexo/resources/morphology.py +229 -0
  111. retexo-0.1.0/src/retexo/resources/vectors.py +83 -0
  112. retexo-0.1.0/src/retexo/resources/wordnet.py +226 -0
  113. retexo-0.1.0/src/retexo/training.py +220 -0
  114. retexo-0.1.0/src/retexo/unified/__init__.py +33 -0
  115. retexo-0.1.0/src/retexo/unified/decoder.py +367 -0
  116. retexo-0.1.0/src/retexo/unified/gate.py +93 -0
  117. retexo-0.1.0/src/retexo/unified/model.py +199 -0
  118. retexo-0.1.0/src/retexo/unified/runs.py +130 -0
  119. retexo-0.1.0/src/retexo/unified/train.py +320 -0
  120. retexo-0.1.0/src/retexo_gui/__init__.py +0 -0
  121. retexo-0.1.0/src/retexo_gui/annotate.py +396 -0
  122. retexo-0.1.0/src/retexo_gui/app.py +696 -0
retexo-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Julian Schelb, Michael Wittweiler, Marie Revellio
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
retexo-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,154 @@
1
+ Metadata-Version: 2.4
2
+ Name: retexo
3
+ Version: 0.1.0
4
+ Summary: Retexo explains Latin text reuse word by word: it aligns a reusing passage to its source and names the operation behind every link.
5
+ License: MIT
6
+ License-File: LICENSE
7
+ Keywords: latin,intertextuality,text reuse,word alignment,edit scripts,digital humanities
8
+ Author: Julian Schelb
9
+ Author-email: julian.schelb@uni-konstanz.de
10
+ Requires-Python: >=3.10,<3.15
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Latin
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Text Processing :: Linguistic
22
+ Provides-Extra: dev
23
+ Provides-Extra: gui
24
+ Provides-Extra: lexical
25
+ Provides-Extra: llm
26
+ Provides-Extra: plots
27
+ Requires-Dist: anthropic (>=0.30.0) ; extra == "llm"
28
+ Requires-Dist: build ; extra == "dev"
29
+ Requires-Dist: cltk (>=1.1.0,<2.0.0) ; (python_version < "3.13") and (extra == "lexical")
30
+ Requires-Dist: gensim (>=4.3.0,<5.0.0) ; extra == "lexical"
31
+ Requires-Dist: gradio (>=5.49.1) ; extra == "gui"
32
+ Requires-Dist: huggingface-hub (>=0.20.0)
33
+ Requires-Dist: locisimiles (>=2.1.3)
34
+ Requires-Dist: matplotlib (>=3.7.0) ; extra == "plots"
35
+ Requires-Dist: mkdocs (>=1.5.0) ; extra == "dev"
36
+ Requires-Dist: mkdocs-material (>=9.0.0) ; extra == "dev"
37
+ Requires-Dist: mkdocstrings[python] (>=0.24.0) ; extra == "dev"
38
+ Requires-Dist: nltk (>=3.8.0) ; extra == "lexical"
39
+ Requires-Dist: numpy (>=1.24.0,<3.0.0)
40
+ Requires-Dist: pandas (>=2.0.0,<3.0.0)
41
+ Requires-Dist: peft (>=0.10.0) ; extra == "llm"
42
+ Requires-Dist: poethepoet (>=0.24.0) ; extra == "dev"
43
+ Requires-Dist: pre-commit (>=3.5.0) ; extra == "dev"
44
+ Requires-Dist: pytest (>=8.0.0,<10.0.0) ; extra == "dev"
45
+ Requires-Dist: pytest-cov (>=4.0.0) ; extra == "dev"
46
+ Requires-Dist: ruff (>=0.8.0) ; extra == "dev"
47
+ Requires-Dist: scikit-learn (>=1.3.0,<2.0.0)
48
+ Requires-Dist: scipy (>=1.10.0,<2.0.0)
49
+ Requires-Dist: spacy (>=3.5.0,<4.0.0) ; extra == "lexical"
50
+ Requires-Dist: stanza (>=1.7.0) ; extra == "lexical"
51
+ Requires-Dist: torch (>=2.0.0,<3.0.0)
52
+ Requires-Dist: transformers (>=4.30.0,<5.0.0)
53
+ Requires-Dist: twine ; extra == "dev"
54
+ Project-URL: Changelog, https://github.com/julianschelb/retexo/blob/main/CHANGELOG.md
55
+ Project-URL: Documentation, https://julianschelb.github.io/retexo/
56
+ Project-URL: Demo, https://julianschelb.github.io/retexo/demo/
57
+ Project-URL: Dataset, https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts
58
+ Project-URL: Homepage, https://julianschelb.github.io/retexo/
59
+ Project-URL: Issues, https://github.com/julianschelb/retexo/issues
60
+ Project-URL: Repository, https://github.com/julianschelb/retexo
61
+ Description-Content-Type: text/markdown
62
+
63
+ # Retexo
64
+
65
+ [![CI](https://github.com/julianschelb/retexo/actions/workflows/ci.yml/badge.svg)](https://github.com/julianschelb/retexo/actions/workflows/ci.yml)
66
+ [![Docs](https://img.shields.io/badge/docs-julianschelb.github.io%2Fretexo-blue)](https://julianschelb.github.io/retexo/)
67
+ [![Dataset](https://img.shields.io/badge/%F0%9F%A4%97%20dataset-edit--scripts-yellow)](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
68
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](https://github.com/julianschelb/retexo/blob/main/LICENSE)
69
+
70
+ **Word-level explanations of text reuse in Latin literature.**
71
+
72
+ When a later author quotes or alludes to an earlier one, the interesting question is *how* the source was reworked.
73
+ Retexo answers it word by word: it aligns every word of the reusing passage to the source word it takes up, or to none,
74
+ and names the operation behind each link, such as a copy, a change of inflection, or a substitution. The result is an
75
+ *edit script*: a set of labeled links between two passages that a philologist can read and contest.
76
+
77
+ Retexo learns these scripts from very little annotation. Synthetic pairs, built by applying known operations to Latin
78
+ text, teach a pointer network the task before a single pair is annotated, and active learning chooses the few pairs an
79
+ annotator corrects.
80
+
81
+ - **Predicted edit scripts** for all 1,490 references of the [Loci Similes](https://arxiv.org/abs/2601.07533) benchmark:
82
+ [Hugging Face dataset](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
83
+ - **Review app** to browse the script of every reference, drawn as arrows between the passages:
84
+ [julianschelb.github.io/retexo/demo](https://julianschelb.github.io/retexo/demo/)
85
+ - **Documentation:** [julianschelb.github.io/retexo](https://julianschelb.github.io/retexo/)
86
+
87
+ ## Quick start
88
+
89
+ ```bash
90
+ pip install "retexo @ git+https://github.com/julianschelb/retexo"
91
+ ```
92
+
93
+ Read the predictions, with the Hugging Face `datasets` library:
94
+
95
+ ```python
96
+ from datasets import load_dataset
97
+
98
+ scripts = load_dataset("julian-schelb/latin-classical-intertextuality-edit-scripts", split="fold_4")
99
+ pair = scripts[0]
100
+
101
+ for link in pair["links"]:
102
+ print(pair["reuse"]["tokens"][link["reuse"]], "<-", pair["source"]["tokens"][link["source"]], link["label"])
103
+ ```
104
+
105
+ Convert the raw predictions of a run into the same format:
106
+
107
+ ```bash
108
+ retexo export runs/f0/predictions.jsonl runs/f1/predictions.jsonl runs/f2/predictions.jsonl \
109
+ runs/f3/predictions.jsonl runs/f4/predictions.jsonl --out export/ --format parquet
110
+ ```
111
+
112
+ ## Labels
113
+
114
+ | Label | Meaning |
115
+ |---|---|
116
+ | `COPY` | the same form, spelling variants folded |
117
+ | `INFLECT` | the same lemma in another form |
118
+ | `SUBST` | another lemma in the same slot, optionally with a named relation such as a synonym |
119
+ | `SPLIT`, `MERGE` | two reuse words for one source word, or one for two |
120
+ | `INS`, `FRAME`, `DEL` | an unlinked reuse word, a word of a citing formula, an unclaimed source word |
121
+
122
+ See the [documentation](https://julianschelb.github.io/retexo/labels/) for the conventions.
123
+
124
+ ## Repository layout
125
+
126
+ ```
127
+ src/retexo/ the package: model, generator, active learning, baselines, scorer
128
+ src/retexo_gui/ Gradio demo and annotation app (gui extra)
129
+ tests/ pytest suite
130
+ examples/ notebooks on the operations, the generator, the passage class and the decoder
131
+ docs/ documentation (MkDocs)
132
+ webapp/ the review app (React, Vite), published under /demo/
133
+ ```
134
+
135
+ The experiments behind the paper live in a separate repository that installs retexo as a dependency.
136
+
137
+ ## Development
138
+
139
+ ```bash
140
+ pip install -e ".[dev,lexical]"
141
+ pytest
142
+ mkdocs serve
143
+ ```
144
+
145
+ See [docs/development.md](https://julianschelb.github.io/retexo/development/).
146
+
147
+ ## Authors
148
+
149
+ Julian Schelb (University of Konstanz), Michael Wittweiler (University of Zurich), Marie Revellio (University of Konstanz).
150
+
151
+ ## License
152
+
153
+ MIT
154
+
retexo-0.1.0/README.md ADDED
@@ -0,0 +1,91 @@
1
+ # Retexo
2
+
3
+ [![CI](https://github.com/julianschelb/retexo/actions/workflows/ci.yml/badge.svg)](https://github.com/julianschelb/retexo/actions/workflows/ci.yml)
4
+ [![Docs](https://img.shields.io/badge/docs-julianschelb.github.io%2Fretexo-blue)](https://julianschelb.github.io/retexo/)
5
+ [![Dataset](https://img.shields.io/badge/%F0%9F%A4%97%20dataset-edit--scripts-yellow)](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
6
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](https://github.com/julianschelb/retexo/blob/main/LICENSE)
7
+
8
+ **Word-level explanations of text reuse in Latin literature.**
9
+
10
+ When a later author quotes or alludes to an earlier one, the interesting question is *how* the source was reworked.
11
+ Retexo answers it word by word: it aligns every word of the reusing passage to the source word it takes up, or to none,
12
+ and names the operation behind each link, such as a copy, a change of inflection, or a substitution. The result is an
13
+ *edit script*: a set of labeled links between two passages that a philologist can read and contest.
14
+
15
+ Retexo learns these scripts from very little annotation. Synthetic pairs, built by applying known operations to Latin
16
+ text, teach a pointer network the task before a single pair is annotated, and active learning chooses the few pairs an
17
+ annotator corrects.
18
+
19
+ - **Predicted edit scripts** for all 1,490 references of the [Loci Similes](https://arxiv.org/abs/2601.07533) benchmark:
20
+ [Hugging Face dataset](https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts)
21
+ - **Review app** to browse the script of every reference, drawn as arrows between the passages:
22
+ [julianschelb.github.io/retexo/demo](https://julianschelb.github.io/retexo/demo/)
23
+ - **Documentation:** [julianschelb.github.io/retexo](https://julianschelb.github.io/retexo/)
24
+
25
+ ## Quick start
26
+
27
+ ```bash
28
+ pip install "retexo @ git+https://github.com/julianschelb/retexo"
29
+ ```
30
+
31
+ Read the predictions, with the Hugging Face `datasets` library:
32
+
33
+ ```python
34
+ from datasets import load_dataset
35
+
36
+ scripts = load_dataset("julian-schelb/latin-classical-intertextuality-edit-scripts", split="fold_4")
37
+ pair = scripts[0]
38
+
39
+ for link in pair["links"]:
40
+ print(pair["reuse"]["tokens"][link["reuse"]], "<-", pair["source"]["tokens"][link["source"]], link["label"])
41
+ ```
42
+
43
+ Convert the raw predictions of a run into the same format:
44
+
45
+ ```bash
46
+ retexo export runs/f0/predictions.jsonl runs/f1/predictions.jsonl runs/f2/predictions.jsonl \
47
+ runs/f3/predictions.jsonl runs/f4/predictions.jsonl --out export/ --format parquet
48
+ ```
49
+
50
+ ## Labels
51
+
52
+ | Label | Meaning |
53
+ |---|---|
54
+ | `COPY` | the same form, spelling variants folded |
55
+ | `INFLECT` | the same lemma in another form |
56
+ | `SUBST` | another lemma in the same slot, optionally with a named relation such as a synonym |
57
+ | `SPLIT`, `MERGE` | two reuse words for one source word, or one for two |
58
+ | `INS`, `FRAME`, `DEL` | an unlinked reuse word, a word of a citing formula, an unclaimed source word |
59
+
60
+ See the [documentation](https://julianschelb.github.io/retexo/labels/) for the conventions.
61
+
62
+ ## Repository layout
63
+
64
+ ```
65
+ src/retexo/ the package: model, generator, active learning, baselines, scorer
66
+ src/retexo_gui/ Gradio demo and annotation app (gui extra)
67
+ tests/ pytest suite
68
+ examples/ notebooks on the operations, the generator, the passage class and the decoder
69
+ docs/ documentation (MkDocs)
70
+ webapp/ the review app (React, Vite), published under /demo/
71
+ ```
72
+
73
+ The experiments behind the paper live in a separate repository that installs retexo as a dependency.
74
+
75
+ ## Development
76
+
77
+ ```bash
78
+ pip install -e ".[dev,lexical]"
79
+ pytest
80
+ mkdocs serve
81
+ ```
82
+
83
+ See [docs/development.md](https://julianschelb.github.io/retexo/development/).
84
+
85
+ ## Authors
86
+
87
+ Julian Schelb (University of Konstanz), Michael Wittweiler (University of Zurich), Marie Revellio (University of Konstanz).
88
+
89
+ ## License
90
+
91
+ MIT
@@ -0,0 +1,118 @@
1
+ [project]
2
+ name = "retexo"
3
+ version = "0.1.0"
4
+ description = "Retexo explains Latin text reuse word by word: it aligns a reusing passage to its source and names the operation behind every link."
5
+ authors = [
6
+ {name = "Julian Schelb", email = "julian.schelb@uni-konstanz.de"},
7
+ {name = "Michael Wittweiler", email = "michael.wittweiler@sglp.uzh.ch"},
8
+ {name = "Marie Revellio"},
9
+ ]
10
+ readme = "README.md"
11
+ license = {text = "MIT"}
12
+ requires-python = ">=3.10,<3.15"
13
+ keywords = ["latin", "intertextuality", "text reuse", "word alignment", "edit scripts", "digital humanities"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Science/Research",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Natural Language :: Latin",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3.10",
21
+ "Programming Language :: Python :: 3.11",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
25
+ "Topic :: Text Processing :: Linguistic",
26
+ ]
27
+ dependencies = [
28
+ "numpy (>=1.24.0,<3.0.0)",
29
+ "scipy (>=1.10.0,<2.0.0)",
30
+ "pandas (>=2.0.0,<3.0.0)",
31
+ "scikit-learn (>=1.3.0,<2.0.0)",
32
+ "torch (>=2.0.0,<3.0.0)",
33
+ "transformers (>=4.30.0,<5.0.0)",
34
+ "huggingface-hub (>=0.20.0)",
35
+ "locisimiles (>=2.1.3)",
36
+ ]
37
+
38
+ [project.urls]
39
+ Homepage = "https://julianschelb.github.io/retexo/"
40
+ Documentation = "https://julianschelb.github.io/retexo/"
41
+ Repository = "https://github.com/julianschelb/retexo"
42
+ Demo = "https://julianschelb.github.io/retexo/demo/"
43
+ Dataset = "https://huggingface.co/datasets/julian-schelb/latin-classical-intertextuality-edit-scripts"
44
+ Changelog = "https://github.com/julianschelb/retexo/blob/main/CHANGELOG.md"
45
+ Issues = "https://github.com/julianschelb/retexo/issues"
46
+
47
+ [project.optional-dependencies]
48
+ gui = [
49
+ "gradio (>=5.49.1)",
50
+ ]
51
+ lexical = [
52
+ # The symbolic typer and the lexical evidence look words up in these resources.
53
+ "cltk (>=1.1.0,<2.0.0) ; python_version < '3.13'",
54
+ "spacy (>=3.5.0,<4.0.0)",
55
+ "stanza (>=1.7.0)",
56
+ "nltk (>=3.8.0)",
57
+ "gensim (>=4.3.0,<5.0.0)",
58
+ ]
59
+ llm = [
60
+ "anthropic (>=0.30.0)",
61
+ "peft (>=0.10.0)",
62
+ ]
63
+ plots = [
64
+ "matplotlib (>=3.7.0)",
65
+ ]
66
+ dev = [
67
+ "pytest (>=8.0.0,<10.0.0)",
68
+ "pytest-cov (>=4.0.0)",
69
+ "poethepoet (>=0.24.0)",
70
+ "mkdocs (>=1.5.0)",
71
+ "mkdocs-material (>=9.0.0)",
72
+ "mkdocstrings[python] (>=0.24.0)",
73
+ "ruff (>=0.8.0)",
74
+ "pre-commit (>=3.5.0)",
75
+ "build",
76
+ "twine",
77
+ ]
78
+
79
+ [project.scripts]
80
+ retexo = "retexo.cli:main"
81
+ retexo-gui = "retexo_gui.app:main"
82
+
83
+ [tool.poetry]
84
+ packages = [
85
+ {include = "retexo", from = "src"},
86
+ {include = "retexo_gui", from = "src"},
87
+ ]
88
+
89
+ [build-system]
90
+ requires = ["poetry-core>=2.0.0,<3.0.0"]
91
+ build-backend = "poetry.core.masonry.api"
92
+
93
+ [tool.pytest.ini_options]
94
+ testpaths = ["tests"]
95
+ python_files = ["test_*.py"]
96
+ python_classes = ["Test*"]
97
+ python_functions = ["test_*"]
98
+ addopts = "-v"
99
+ pythonpath = ["src"]
100
+
101
+ [tool.poe.tasks]
102
+ test = "pytest"
103
+ test-cov = "pytest --cov=retexo --cov-report=term-missing"
104
+ lint = "ruff check src/ tests/"
105
+ docs = "mkdocs serve"
106
+ docs-build = "mkdocs build"
107
+ webapp = "npm --prefix webapp run dev"
108
+
109
+ # ---------- Ruff ----------
110
+ # The research code was written without a formatter; CI checks for errors only (syntax errors, undefined names).
111
+
112
+ [tool.ruff]
113
+ target-version = "py310"
114
+ line-length = 120
115
+ src = ["src", "tests"]
116
+
117
+ [tool.ruff.lint]
118
+ select = ["E9", "F63", "F7", "F82"]
@@ -0,0 +1,62 @@
1
+ # retexo/__init__.py
2
+ """
3
+ Typed edit scripts for Latin intertextual reuse.
4
+
5
+ Building a variant and checking it round-trips:
6
+
7
+ ```python
8
+ from retexo import VariantBuilder, Scriba
9
+
10
+ builder = VariantBuilder("uox faucibus haesit")
11
+ builder.keep(0).syn(1, "gutture").keep(2)
12
+ variant, script = builder.build()
13
+
14
+ Scriba().verify(script, "uox faucibus haesit".split(), variant) # True
15
+ Scriba().execute(script.invert(), variant) # back to the source
16
+ ```
17
+ """
18
+ from __future__ import annotations
19
+
20
+ __version__ = "0.1.0"
21
+
22
+ from retexo.core.builder import VariantBuilder
23
+ from retexo.core.oracle import (
24
+ Aligner,
25
+ EditPlanOracle,
26
+ GreedyAligner,
27
+ OptimalAligner,
28
+ OracleConfig,
29
+ )
30
+ from retexo.resources import Resources
31
+ from retexo.core.normalize import normalize
32
+ from retexo.operations import (
33
+ ALL_OPERATIONS,
34
+ EditOperation,
35
+ Level,
36
+ Operation,
37
+ OperationRegistry,
38
+ Role,
39
+ )
40
+ from retexo.core.scriba import Scriba, Validity
41
+ from retexo.core.script import CostModel, EditScript
42
+
43
+ __all__ = [
44
+ "ALL_OPERATIONS",
45
+ "Aligner",
46
+ "EditPlanOracle",
47
+ "GreedyAligner",
48
+ "OptimalAligner",
49
+ "OracleConfig",
50
+ "Resources",
51
+ "CostModel",
52
+ "EditOperation",
53
+ "EditScript",
54
+ "Level",
55
+ "Operation",
56
+ "OperationRegistry",
57
+ "Role",
58
+ "Scriba",
59
+ "Validity",
60
+ "VariantBuilder",
61
+ "normalize",
62
+ ]
@@ -0,0 +1,27 @@
1
+ # retexo/aligners/__init__.py
2
+ """Alignment models and the corpus-level and pair-level policies that turn their scores into
3
+ one source per reuse word: the classical pointer-score assignment policies, the bidirectional
4
+ agreement rules, the orthographic sameness predicates, the decoder that derives a full script
5
+ from a typed alignment, the Sinkhorn and whole-script DP variants, and the contextual embedding
6
+ aligner."""
7
+ from __future__ import annotations
8
+
9
+ from retexo.aligners.agreement import AgreementDecoder, PairSwap
10
+ from retexo.aligners.aligner import ContextualAligner
11
+ from retexo.aligners.assignment import AssignmentPolicy, Reranker
12
+ from retexo.aligners.decode import ScriptDecoder
13
+ from retexo.aligners.global_decode import GlobalDecoder
14
+ from retexo.aligners.sameness import SamenessPolicy
15
+ from retexo.aligners.sinkhorn import SinkhornBalancer
16
+
17
+ __all__ = [
18
+ "AgreementDecoder",
19
+ "AssignmentPolicy",
20
+ "ContextualAligner",
21
+ "GlobalDecoder",
22
+ "PairSwap",
23
+ "Reranker",
24
+ "SamenessPolicy",
25
+ "ScriptDecoder",
26
+ "SinkhornBalancer",
27
+ ]
@@ -0,0 +1,181 @@
1
+ # retexo/aligners/agreement.py
2
+ """
3
+ Bidirectional-agreement decoding rules: score both directions, keep what both accept.
4
+
5
+ Renamed from the preliminary experiment module E29; kept under its own name,
6
+ unchanged, because ``retexo.baselines.decoder`` imports it directly and
7
+ several implementation notes say explicitly to reuse it verbatim.
8
+
9
+ Our pointer scores reuse -> source and its null is a scalar per reuse word.
10
+ awesome-align's null is a test on two directions: a link exists only if the
11
+ reuse word picks the source word *and* the source word picks it back.
12
+ SimAlign's mutual argmax is the same test without a threshold; its entropy
13
+ rule is a third signal. All three work on two sets of rows -- the forward
14
+ rows the pointer scores, and the reverse rows a swapped-sides pass produces --
15
+ and need no training.
16
+
17
+ Rows are the pointer's score-row shape: per reuse word, [(source, p) ...] best
18
+ first, with the null as source -1. Reverse rows are the same shape with the
19
+ roles swapped: per *source* word, [(reuse, p) ...].
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import math
25
+ from typing import ClassVar, Dict, List, Sequence, Tuple
26
+
27
+ from retexo.formulations.change_detector import ChangeExample
28
+
29
+ Row = List[Tuple[int, float]]
30
+
31
+
32
+ class PairSwap:
33
+ """Building a pair with the roles reversed, for the reverse-direction pass
34
+ and for symmetric training.
35
+
36
+ Example:
37
+ ```python
38
+ reverse_pair = PairSwap.example(example) # untrained: p(reuse word | source word)
39
+ reverse_training_pair = PairSwap.labelled(example) # a training pair with inverted labels
40
+ ```
41
+ """
42
+
43
+ #: relations that change name when the sides change
44
+ INVERT: ClassVar[Dict[str, str]] = {"HYPER": "HYPO", "HYPO": "HYPER", "SPLIT": "MERGE", "MERGE": "SPLIT"}
45
+
46
+ @staticmethod
47
+ def example(example: ChangeExample) -> ChangeExample:
48
+ """The pair with the roles exchanged: the source passage sits in the reuse
49
+ slot, so the model scores p(reuse word | source word)."""
50
+ s, t = list(example.target_tokens), list(example.source_tokens) # swapped on purpose
51
+ return ChangeExample(source_tokens=s, target_tokens=t, labels=[0] * len(t),
52
+ operations=["COPY"] * len(t), n_operations=0,
53
+ source_labels=[0] * len(s), source_operations=["COPY"] * len(s),
54
+ alignments=[-1] * len(t), fine_operations=["INS"] * len(t),
55
+ frame_labels=[0] * len(t), link_features=[None] * len(t))
56
+
57
+ @classmethod
58
+ def labelled(cls, example: ChangeExample) -> ChangeExample:
59
+ """A *training* pair with the sides exchanged and its labels inverted.
60
+
61
+ One-to-one links invert exactly; HYPER/HYPO and SPLIT/MERGE change name;
62
+ a gold lexical change of unknown kind ("?") stays unknown; the former reuse
63
+ side's inserted words become deleted source words; frames do not exist on
64
+ a source passage, so the new reuse side carries none. Evidence grids are
65
+ not transposed (the featurizer is not symmetric) -- recompute them."""
66
+ old_s, old_t = list(example.source_tokens), list(example.target_tokens)
67
+ links = example.alignments or [-1] * len(old_t)
68
+ fine = example.fine_operations or ["INS"] * len(old_t)
69
+ new_s, new_t = old_t, old_s
70
+ new_links = [-1] * len(new_t)
71
+ new_fine = ["INS"] * len(new_t)
72
+ for t, s in enumerate(links):
73
+ if s is not None and 0 <= s < len(new_t) and new_links[s] < 0:
74
+ new_links[s] = t
75
+ tag = fine[t] if t < len(fine) else "?"
76
+ new_fine[s] = cls.INVERT.get(tag, tag)
77
+ consumed = {t for t in new_links if t >= 0}
78
+ new_source_labels = [0 if t in consumed else 1 for t in range(len(new_s))]
79
+ coarse = ["COPY" if (k == "NOP") else ("INS" if k == "INS" else "SUBST") for k in new_fine]
80
+ ex = ChangeExample(source_tokens=new_s, target_tokens=new_t,
81
+ labels=[0 if op == "COPY" else 1 for op in coarse], operations=coarse,
82
+ n_operations=sum(1 for op in coarse if op != "COPY"),
83
+ source_labels=new_source_labels,
84
+ source_operations=["DEL" if d else "COPY" for d in new_source_labels],
85
+ alignments=new_links, fine_operations=new_fine,
86
+ frame_labels=[0] * len(new_t), link_features=[None] * len(new_t))
87
+ object.__setattr__(ex, "swapped_from", True)
88
+ return ex
89
+
90
+
91
+ class AgreementDecoder:
92
+ """Bidirectional-agreement rules, each returning restricted forward rows for
93
+ the Hungarian assignment.
94
+
95
+ Example:
96
+ ```python
97
+ kept = AgreementDecoder.mutual_argmax(rows, reverse_rows)
98
+ kept = AgreementDecoder.intersect(rows, reverse_rows, c=0.4)
99
+ ```
100
+ """
101
+
102
+ @staticmethod
103
+ def prob(row: Row, index: int) -> float:
104
+ return dict(row).get(index, 0.0)
105
+
106
+ @staticmethod
107
+ def top(row: Row) -> int:
108
+ """The best entry, the null (-1) included."""
109
+ return row[0][0] if row else -1
110
+
111
+ @classmethod
112
+ def reverse_links(cls, rows_rev: Sequence[Row], n_reuse: int) -> List[int]:
113
+ """The reverse direction read as an aligner of the reuse side: reuse word t
114
+ links to the source word s whose best pick is t (first such s), else -1."""
115
+ links = [-1] * n_reuse
116
+ for s, row in enumerate(rows_rev):
117
+ t = cls.top(row)
118
+ if 0 <= t < n_reuse and links[t] < 0:
119
+ links[t] = s
120
+ return links
121
+
122
+ @staticmethod
123
+ def entropy(row: Row) -> float:
124
+ ps = [p for _, p in row if p > 1e-12]
125
+ z = sum(ps) or 1.0
126
+ return -sum(p / z * math.log(p / z) for p in ps)
127
+
128
+ @classmethod
129
+ def intersect(cls, rows: Sequence[Row], rows_rev: Sequence[Row], c: float) -> List[Row]:
130
+ """awesome-align: keep (t, s) iff p(s|t) > c and p(t|s) > c. A word whose
131
+ every candidate fails keeps only its null."""
132
+ out = []
133
+ for t, row in enumerate(rows):
134
+ kept = [(s, p) for s, p in row
135
+ if s >= 0 and p > c and s < len(rows_rev) and cls.prob(rows_rev[s], t) > c]
136
+ out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
137
+ return out
138
+
139
+ @classmethod
140
+ def mutual_argmax(cls, rows: Sequence[Row], rows_rev: Sequence[Row]) -> List[Row]:
141
+ """SimAlign Argmax: keep (t, s) iff s is t's best and t is s's best."""
142
+ out = []
143
+ for t, row in enumerate(rows):
144
+ s = cls.top(row)
145
+ kept = [(s, cls.prob(row, s))] if s >= 0 and s < len(rows_rev) and cls.top(rows_rev[s]) == t else []
146
+ out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
147
+ return out
148
+
149
+ @classmethod
150
+ def entropy_filter(cls, rows: Sequence[Row], rows_rev: Sequence[Row], tau: float) -> List[Row]:
151
+ """SimAlign's null: drop every link of a reuse word whose row entropy, and of
152
+ a source word whose column entropy, are both above tau (normalised by log n)."""
153
+ out = []
154
+ h_col = [cls.entropy(r) / max(math.log(max(len(r), 2)), 1e-9) for r in rows_rev]
155
+ for t, row in enumerate(rows):
156
+ h_row = cls.entropy(row) / max(math.log(max(len(row), 2)), 1e-9)
157
+ kept = [(s, p) for s, p in row if s >= 0
158
+ and min(h_row, h_col[s] if s < len(h_col) else 1.0) <= tau]
159
+ out.append(sorted(kept + [(-1, cls.prob(row, -1))], key=lambda x: -x[1]))
160
+ return out
161
+
162
+ @classmethod
163
+ def null_scale(cls, rows: Sequence[Row], k: float) -> List[Row]:
164
+ """Our null re-weighted by k and renormalised (E18's move, at the raw pointer)."""
165
+ out = []
166
+ for row in rows:
167
+ scaled = [(s, p * (k if s < 0 else 1.0)) for s, p in row]
168
+ z = sum(p for _, p in scaled) or 1.0
169
+ out.append(sorted([(s, p / z) for s, p in scaled], key=lambda x: -x[1]))
170
+ return out
171
+
172
+ @classmethod
173
+ def compose(cls, *restricted: Sequence[Row]) -> List[Row]:
174
+ """Keep a candidate only if every rule kept it; the null is the first rule's."""
175
+ out = []
176
+ for rows in zip(*restricted):
177
+ keep = set.intersection(*[{s for s, _ in r if s >= 0} for r in rows])
178
+ first = rows[0]
179
+ out.append(sorted([(s, p) for s, p in first if s in keep] + [(-1, cls.prob(first, -1))],
180
+ key=lambda x: -x[1]))
181
+ return out