emalign-phonology 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- emalign_phonology-0.1.0/LICENSE +21 -0
- emalign_phonology-0.1.0/PKG-INFO +281 -0
- emalign_phonology-0.1.0/README.md +248 -0
- emalign_phonology-0.1.0/pyproject.toml +64 -0
- emalign_phonology-0.1.0/setup.cfg +4 -0
- emalign_phonology-0.1.0/src/emalign/__init__.py +200 -0
- emalign_phonology-0.1.0/src/emalign/aligner.py +514 -0
- emalign_phonology-0.1.0/src/emalign/alignment.py +294 -0
- emalign_phonology-0.1.0/src/emalign/cldf_io.py +186 -0
- emalign_phonology-0.1.0/src/emalign/cli.py +157 -0
- emalign_phonology-0.1.0/src/emalign/features.py +142 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/PKG-INFO +281 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/SOURCES.txt +19 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/dependency_links.txt +1 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/entry_points.txt +2 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/requires.txt +9 -0
- emalign_phonology-0.1.0/src/emalign_phonology.egg-info/top_level.txt +1 -0
- emalign_phonology-0.1.0/tests/test_aligner.py +278 -0
- emalign_phonology-0.1.0/tests/test_alignment.py +210 -0
- emalign_phonology-0.1.0/tests/test_cldf_io.py +172 -0
- emalign_phonology-0.1.0/tests/test_features.py +111 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ChangeLing Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: emalign-phonology
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: EM-based alignment of cognate sets in comparative dictionaries using articulatory features
|
|
5
|
+
Author-email: Changeling Lab <changeling@example.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/changelinglab/emalign
|
|
8
|
+
Project-URL: Repository, https://github.com/changelinglab/emalign
|
|
9
|
+
Project-URL: Issues, https://github.com/changelinglab/emalign/issues
|
|
10
|
+
Keywords: linguistics,phonology,alignment,cognates,historical-linguistics,CLDF,panphon,expectation-maximization
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: numpy>=1.21.0
|
|
25
|
+
Requires-Dist: panphon>=0.20.0
|
|
26
|
+
Requires-Dist: pycldf>=1.30.0
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
29
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
30
|
+
Requires-Dist: build>=1.0.0; extra == "dev"
|
|
31
|
+
Requires-Dist: twine>=4.0.0; extra == "dev"
|
|
32
|
+
Dynamic: license-file
|
|
33
|
+
|
|
34
|
+
# emalign-phonology
|
|
35
|
+
|
|
36
|
+
A Python package for aligning cognate sets in comparative dictionaries using articulatory features, with weights learned via expectation maximization.
|
|
37
|
+
|
|
38
|
+
[](https://badge.fury.io/py/emalign-phonology)
|
|
39
|
+
[](https://www.python.org/downloads/)
|
|
40
|
+
[](https://opensource.org/licenses/MIT)
|
|
41
|
+
|
|
42
|
+
## Overview
|
|
43
|
+
|
|
44
|
+
`emalign-phonology` generates phoneme alignments for cognate forms in CLDF-formatted comparative dictionaries. It uses:
|
|
45
|
+
|
|
46
|
+
- **PanPhon's 24 articulatory features** to compute phoneme similarity
|
|
47
|
+
- **Expectation-Maximization (EM)** with SGD to learn optimal feature weights
|
|
48
|
+
- **Anchor-based alignment** to handle multi-language cognate sets efficiently
|
|
49
|
+
|
|
50
|
+
The package can be used both as a **command-line utility** and as a **Python library** with an API that mirrors the CLI interface.
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install emalign-phonology
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Or for development:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
git clone https://github.com/changelinglab/emalign.git
|
|
62
|
+
cd emalign
|
|
63
|
+
pip install -e ".[dev]"
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
## Quick Start
|
|
67
|
+
|
|
68
|
+
### Command Line
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
# Basic usage
|
|
72
|
+
emalign input_cldf/ output_alignments.csv
|
|
73
|
+
|
|
74
|
+
# With options
|
|
75
|
+
emalign input_cldf/ output.csv --verbose --seed 42 --max-iterations 20
|
|
76
|
+
|
|
77
|
+
# Select optimal anchor language
|
|
78
|
+
emalign input_cldf/ output.csv --select-anchor --verbose
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
### Python API (Recommended)
|
|
82
|
+
|
|
83
|
+
```python
|
|
84
|
+
from emalign import align
|
|
85
|
+
|
|
86
|
+
# Simple usage - equivalent to CLI
|
|
87
|
+
alignments = align("path/to/cldf/", "output.csv", verbose=True)
|
|
88
|
+
|
|
89
|
+
# With options - mirrors CLI arguments
|
|
90
|
+
alignments = align(
|
|
91
|
+
"path/to/cldf/",
|
|
92
|
+
"output.csv",
|
|
93
|
+
langs="kach1286,chal1279.1", # Filter by language
|
|
94
|
+
seed=42,
|
|
95
|
+
max_iterations=20,
|
|
96
|
+
select_anchor=True,
|
|
97
|
+
verbose=True,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
# Without writing to file - just get the results
|
|
101
|
+
alignments = align("path/to/cldf/")
|
|
102
|
+
for a in alignments:
|
|
103
|
+
print(f"{a.form_id}: {a.aligned_form}")
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## Usage
|
|
107
|
+
|
|
108
|
+
### Command Line Interface
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
emalign <input_cldf_path> <output_csv_path> [options]
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
#### Options
|
|
115
|
+
|
|
116
|
+
| Option | Description | Default |
|
|
117
|
+
|--------|-------------|---------|
|
|
118
|
+
| `-l, --langs LIST` | Comma-separated list of language IDs or Glottocodes to include | All languages |
|
|
119
|
+
| `--gap-penalty FLOAT` | Initial gap penalty for alignment | 0.9 |
|
|
120
|
+
| `--no-learn-gap` | Disable learning gap penalty (use fixed value) | Learn gap |
|
|
121
|
+
| `--gap-learning-rate FLOAT` | SGD learning rate for gap penalty | 0.05 |
|
|
122
|
+
| `--learning-rate FLOAT` | SGD learning rate for feature weights | 0.01 |
|
|
123
|
+
| `--max-iterations INT` | Maximum EM iterations | 40 |
|
|
124
|
+
| `--convergence-threshold FLOAT` | Stop when weight change is below this | 1e-4 |
|
|
125
|
+
| `--seed INT` | Random seed for reproducibility | None |
|
|
126
|
+
| `--select-anchor` | Try all languages as anchors and select the best | False |
|
|
127
|
+
| `-v, --verbose` | Print progress information | False |
|
|
128
|
+
|
|
129
|
+
### Python Library API
|
|
130
|
+
|
|
131
|
+
#### High-Level API: `align()`
|
|
132
|
+
|
|
133
|
+
The `align()` function provides a CLI-equivalent interface for programmatic use:
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
from emalign import align
|
|
137
|
+
|
|
138
|
+
alignments = align(
|
|
139
|
+
input_path, # Path to CLDF dataset
|
|
140
|
+
output_path=None, # Optional: write results to CSV
|
|
141
|
+
langs=None, # Language filter (string, list, or set)
|
|
142
|
+
gap_penalty=0.9,
|
|
143
|
+
no_learn_gap=False,
|
|
144
|
+
gap_learning_rate=0.05,
|
|
145
|
+
learning_rate=0.01,
|
|
146
|
+
max_iterations=40,
|
|
147
|
+
convergence_threshold=1e-4,
|
|
148
|
+
seed=None,
|
|
149
|
+
select_anchor=False,
|
|
150
|
+
verbose=False,
|
|
151
|
+
)
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
#### Low-Level API
|
|
155
|
+
|
|
156
|
+
For more control, use the component classes directly:
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
from emalign import (
|
|
160
|
+
CognateAligner,
|
|
161
|
+
load_cldf_dataset,
|
|
162
|
+
write_alignments,
|
|
163
|
+
select_best_anchor_language,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
# Load CLDF data
|
|
167
|
+
forms, cognate_sets, languages = load_cldf_dataset("path/to/cldf/")
|
|
168
|
+
|
|
169
|
+
# Create and configure aligner
|
|
170
|
+
aligner = CognateAligner(
|
|
171
|
+
gap_penalty=0.9,
|
|
172
|
+
learning_rate=0.01,
|
|
173
|
+
max_iterations=40,
|
|
174
|
+
random_seed=42,
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
# Optionally select the best anchor language
|
|
178
|
+
best_lang, best_prob, best_weights, best_gap = select_best_anchor_language(
|
|
179
|
+
cognate_sets, aligner
|
|
180
|
+
)
|
|
181
|
+
aligner.weights = best_weights
|
|
182
|
+
aligner.gap_penalty = best_gap
|
|
183
|
+
|
|
184
|
+
# Fit model and generate alignments
|
|
185
|
+
alignments = aligner.fit_and_align(cognate_sets, verbose=True)
|
|
186
|
+
|
|
187
|
+
# Write output
|
|
188
|
+
write_alignments(alignments, "alignments.csv")
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
### Language Filtering
|
|
192
|
+
|
|
193
|
+
You can restrict alignments to a subset of languages using either language IDs or Glottocodes:
|
|
194
|
+
|
|
195
|
+
```bash
|
|
196
|
+
# CLI: Filter by language IDs
|
|
197
|
+
emalign input_cldf/ output.csv --langs 1,2,3,4
|
|
198
|
+
|
|
199
|
+
# CLI: Filter by Glottocodes
|
|
200
|
+
emalign input_cldf/ output.csv --langs kach1286,chal1279.1,east2902
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
# Python: Various filter formats
|
|
205
|
+
align("data/", langs="kach1286,chal1279.1") # String
|
|
206
|
+
align("data/", langs=["kach1286", "chal1279.1"]) # List
|
|
207
|
+
align("data/", langs={"kach1286", "chal1279.1"}) # Set
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
## Data Formats
|
|
211
|
+
|
|
212
|
+
### Input Format
|
|
213
|
+
|
|
214
|
+
The input must be a CLDF dataset with:
|
|
215
|
+
|
|
216
|
+
- `forms.csv`: Lexical forms with `ID`, `Language_ID`, `Form` columns
|
|
217
|
+
- `cognates.csv`: Cognate judgments with `Form_ID`, `Cognateset_ID`, `Morph_Index` columns
|
|
218
|
+
- `languages.csv`: Language metadata with `ID`, `Name`, `Glottocode` columns
|
|
219
|
+
- `*-metadata.json`: CLDF metadata file
|
|
220
|
+
|
|
221
|
+
Forms should be in IPA with morphs separated by `+` (e.g., `pre+fix`).
|
|
222
|
+
|
|
223
|
+
### Output Format
|
|
224
|
+
|
|
225
|
+
The output is a CLDF-compliant CSV with columns:
|
|
226
|
+
|
|
227
|
+
| Column | Description |
|
|
228
|
+
|--------|-------------|
|
|
229
|
+
| `ID` | Unique alignment identifier |
|
|
230
|
+
| `Form_ID` | Reference to the original form |
|
|
231
|
+
| `Cognateset_ID` | Reference to the cognate set |
|
|
232
|
+
| `Aligned_Form` | Pipe-delimited aligned segments (e.g., `p\|a\|t\|-`) |
|
|
233
|
+
|
|
234
|
+
## Algorithm
|
|
235
|
+
|
|
236
|
+
1. **Initialization**: Feature weights initialized with small random perturbations, with critical features (syllabic, consonantal) receiving higher initial weights
|
|
237
|
+
2. **E-step**: Compute optimal alignments using weighted Levenshtein distance based on articulatory feature differences
|
|
238
|
+
3. **M-step**: Update weights via SGD based on alignment statistics; optionally learn gap penalty
|
|
239
|
+
4. **Iteration**: Repeat until convergence or max iterations reached
|
|
240
|
+
|
|
241
|
+
The alignment cost between segments is computed as:
|
|
242
|
+
|
|
243
|
+
```
|
|
244
|
+
cost = weights · |features(seg1) - features(seg2)|
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
where `features()` returns PanPhon's 24-dimensional articulatory feature vector.
|
|
248
|
+
|
|
249
|
+
## API Reference
|
|
250
|
+
|
|
251
|
+
### Classes
|
|
252
|
+
|
|
253
|
+
- **`CognateAligner`**: Main aligner class with EM-based weight learning
|
|
254
|
+
- **`AlignmentResult`**: Dataclass holding alignment results (form_id, cognateset_id, aligned_form)
|
|
255
|
+
- **`CognateSet`**: Dataclass representing a cognate set with its entries
|
|
256
|
+
- **`Form`**: Dataclass representing a lexical form
|
|
257
|
+
- **`Language`**: Dataclass representing a language with metadata
|
|
258
|
+
|
|
259
|
+
### Functions
|
|
260
|
+
|
|
261
|
+
- **`align()`**: High-level function providing CLI-equivalent API
|
|
262
|
+
- **`load_cldf_dataset()`**: Load a CLDF dataset from path
|
|
263
|
+
- **`write_alignments()`**: Write alignment results to CSV
|
|
264
|
+
- **`select_best_anchor_language()`**: Find optimal anchor language for alignment
|
|
265
|
+
|
|
266
|
+
## License
|
|
267
|
+
|
|
268
|
+
MIT License
|
|
269
|
+
|
|
270
|
+
## Citation
|
|
271
|
+
|
|
272
|
+
If you use this package in your research, please cite:
|
|
273
|
+
|
|
274
|
+
```bibtex
|
|
275
|
+
@software{emalign,
|
|
276
|
+
title = {emalign-phonology: EM-based cognate alignment},
|
|
277
|
+
author = {Changeling Lab},
|
|
278
|
+
url = {https://github.com/changelinglab/emalign},
|
|
279
|
+
year = {2024}
|
|
280
|
+
}
|
|
281
|
+
```
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
# emalign-phonology
|
|
2
|
+
|
|
3
|
+
A Python package for aligning cognate sets in comparative dictionaries using articulatory features, with weights learned via expectation maximization.
|
|
4
|
+
|
|
5
|
+
[](https://badge.fury.io/py/emalign-phonology)
|
|
6
|
+
[](https://www.python.org/downloads/)
|
|
7
|
+
[](https://opensource.org/licenses/MIT)
|
|
8
|
+
|
|
9
|
+
## Overview
|
|
10
|
+
|
|
11
|
+
`emalign-phonology` generates phoneme alignments for cognate forms in CLDF-formatted comparative dictionaries. It uses:
|
|
12
|
+
|
|
13
|
+
- **PanPhon's 24 articulatory features** to compute phoneme similarity
|
|
14
|
+
- **Expectation-Maximization (EM)** with SGD to learn optimal feature weights
|
|
15
|
+
- **Anchor-based alignment** to handle multi-language cognate sets efficiently
|
|
16
|
+
|
|
17
|
+
The package can be used both as a **command-line utility** and as a **Python library** with an API that mirrors the CLI interface.
|
|
18
|
+
|
|
19
|
+
## Installation
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
pip install emalign-phonology
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Or for development:
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
git clone https://github.com/changelinglab/emalign.git
|
|
29
|
+
cd emalign
|
|
30
|
+
pip install -e ".[dev]"
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Quick Start
|
|
34
|
+
|
|
35
|
+
### Command Line
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
# Basic usage
|
|
39
|
+
emalign input_cldf/ output_alignments.csv
|
|
40
|
+
|
|
41
|
+
# With options
|
|
42
|
+
emalign input_cldf/ output.csv --verbose --seed 42 --max-iterations 20
|
|
43
|
+
|
|
44
|
+
# Select optimal anchor language
|
|
45
|
+
emalign input_cldf/ output.csv --select-anchor --verbose
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Python API (Recommended)
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from emalign import align
|
|
52
|
+
|
|
53
|
+
# Simple usage - equivalent to CLI
|
|
54
|
+
alignments = align("path/to/cldf/", "output.csv", verbose=True)
|
|
55
|
+
|
|
56
|
+
# With options - mirrors CLI arguments
|
|
57
|
+
alignments = align(
|
|
58
|
+
"path/to/cldf/",
|
|
59
|
+
"output.csv",
|
|
60
|
+
langs="kach1286,chal1279.1", # Filter by language
|
|
61
|
+
seed=42,
|
|
62
|
+
max_iterations=20,
|
|
63
|
+
select_anchor=True,
|
|
64
|
+
verbose=True,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# Without writing to file - just get the results
|
|
68
|
+
alignments = align("path/to/cldf/")
|
|
69
|
+
for a in alignments:
|
|
70
|
+
print(f"{a.form_id}: {a.aligned_form}")
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Usage
|
|
74
|
+
|
|
75
|
+
### Command Line Interface
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
emalign <input_cldf_path> <output_csv_path> [options]
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
#### Options
|
|
82
|
+
|
|
83
|
+
| Option | Description | Default |
|
|
84
|
+
|--------|-------------|---------|
|
|
85
|
+
| `-l, --langs LIST` | Comma-separated list of language IDs or Glottocodes to include | All languages |
|
|
86
|
+
| `--gap-penalty FLOAT` | Initial gap penalty for alignment | 0.9 |
|
|
87
|
+
| `--no-learn-gap` | Disable learning gap penalty (use fixed value) | Learn gap |
|
|
88
|
+
| `--gap-learning-rate FLOAT` | SGD learning rate for gap penalty | 0.05 |
|
|
89
|
+
| `--learning-rate FLOAT` | SGD learning rate for feature weights | 0.01 |
|
|
90
|
+
| `--max-iterations INT` | Maximum EM iterations | 40 |
|
|
91
|
+
| `--convergence-threshold FLOAT` | Stop when weight change is below this | 1e-4 |
|
|
92
|
+
| `--seed INT` | Random seed for reproducibility | None |
|
|
93
|
+
| `--select-anchor` | Try all languages as anchors and select the best | False |
|
|
94
|
+
| `-v, --verbose` | Print progress information | False |
|
|
95
|
+
|
|
96
|
+
### Python Library API
|
|
97
|
+
|
|
98
|
+
#### High-Level API: `align()`
|
|
99
|
+
|
|
100
|
+
The `align()` function provides a CLI-equivalent interface for programmatic use:
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from emalign import align
|
|
104
|
+
|
|
105
|
+
alignments = align(
|
|
106
|
+
input_path, # Path to CLDF dataset
|
|
107
|
+
output_path=None, # Optional: write results to CSV
|
|
108
|
+
langs=None, # Language filter (string, list, or set)
|
|
109
|
+
gap_penalty=0.9,
|
|
110
|
+
no_learn_gap=False,
|
|
111
|
+
gap_learning_rate=0.05,
|
|
112
|
+
learning_rate=0.01,
|
|
113
|
+
max_iterations=40,
|
|
114
|
+
convergence_threshold=1e-4,
|
|
115
|
+
seed=None,
|
|
116
|
+
select_anchor=False,
|
|
117
|
+
verbose=False,
|
|
118
|
+
)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
#### Low-Level API
|
|
122
|
+
|
|
123
|
+
For more control, use the component classes directly:
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
from emalign import (
|
|
127
|
+
CognateAligner,
|
|
128
|
+
load_cldf_dataset,
|
|
129
|
+
write_alignments,
|
|
130
|
+
select_best_anchor_language,
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
# Load CLDF data
|
|
134
|
+
forms, cognate_sets, languages = load_cldf_dataset("path/to/cldf/")
|
|
135
|
+
|
|
136
|
+
# Create and configure aligner
|
|
137
|
+
aligner = CognateAligner(
|
|
138
|
+
gap_penalty=0.9,
|
|
139
|
+
learning_rate=0.01,
|
|
140
|
+
max_iterations=40,
|
|
141
|
+
random_seed=42,
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
# Optionally select the best anchor language
|
|
145
|
+
best_lang, best_prob, best_weights, best_gap = select_best_anchor_language(
|
|
146
|
+
cognate_sets, aligner
|
|
147
|
+
)
|
|
148
|
+
aligner.weights = best_weights
|
|
149
|
+
aligner.gap_penalty = best_gap
|
|
150
|
+
|
|
151
|
+
# Fit model and generate alignments
|
|
152
|
+
alignments = aligner.fit_and_align(cognate_sets, verbose=True)
|
|
153
|
+
|
|
154
|
+
# Write output
|
|
155
|
+
write_alignments(alignments, "alignments.csv")
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
### Language Filtering
|
|
159
|
+
|
|
160
|
+
You can restrict alignments to a subset of languages using either language IDs or Glottocodes:
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
# CLI: Filter by language IDs
|
|
164
|
+
emalign input_cldf/ output.csv --langs 1,2,3,4
|
|
165
|
+
|
|
166
|
+
# CLI: Filter by Glottocodes
|
|
167
|
+
emalign input_cldf/ output.csv --langs kach1286,chal1279.1,east2902
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
# Python: Various filter formats
|
|
172
|
+
align("data/", langs="kach1286,chal1279.1") # String
|
|
173
|
+
align("data/", langs=["kach1286", "chal1279.1"]) # List
|
|
174
|
+
align("data/", langs={"kach1286", "chal1279.1"}) # Set
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
## Data Formats
|
|
178
|
+
|
|
179
|
+
### Input Format
|
|
180
|
+
|
|
181
|
+
The input must be a CLDF dataset with:
|
|
182
|
+
|
|
183
|
+
- `forms.csv`: Lexical forms with `ID`, `Language_ID`, `Form` columns
|
|
184
|
+
- `cognates.csv`: Cognate judgments with `Form_ID`, `Cognateset_ID`, `Morph_Index` columns
|
|
185
|
+
- `languages.csv`: Language metadata with `ID`, `Name`, `Glottocode` columns
|
|
186
|
+
- `*-metadata.json`: CLDF metadata file
|
|
187
|
+
|
|
188
|
+
Forms should be in IPA with morphs separated by `+` (e.g., `pre+fix`).
|
|
189
|
+
|
|
190
|
+
### Output Format
|
|
191
|
+
|
|
192
|
+
The output is a CLDF-compliant CSV with columns:
|
|
193
|
+
|
|
194
|
+
| Column | Description |
|
|
195
|
+
|--------|-------------|
|
|
196
|
+
| `ID` | Unique alignment identifier |
|
|
197
|
+
| `Form_ID` | Reference to the original form |
|
|
198
|
+
| `Cognateset_ID` | Reference to the cognate set |
|
|
199
|
+
| `Aligned_Form` | Pipe-delimited aligned segments (e.g., `p\|a\|t\|-`) |
|
|
200
|
+
|
|
201
|
+
## Algorithm
|
|
202
|
+
|
|
203
|
+
1. **Initialization**: Feature weights initialized with small random perturbations, with critical features (syllabic, consonantal) receiving higher initial weights
|
|
204
|
+
2. **E-step**: Compute optimal alignments using weighted Levenshtein distance based on articulatory feature differences
|
|
205
|
+
3. **M-step**: Update weights via SGD based on alignment statistics; optionally learn gap penalty
|
|
206
|
+
4. **Iteration**: Repeat until convergence or max iterations reached
|
|
207
|
+
|
|
208
|
+
The alignment cost between segments is computed as:
|
|
209
|
+
|
|
210
|
+
```
|
|
211
|
+
cost = weights · |features(seg1) - features(seg2)|
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
where `features()` returns PanPhon's 24-dimensional articulatory feature vector.
|
|
215
|
+
|
|
216
|
+
## API Reference
|
|
217
|
+
|
|
218
|
+
### Classes
|
|
219
|
+
|
|
220
|
+
- **`CognateAligner`**: Main aligner class with EM-based weight learning
|
|
221
|
+
- **`AlignmentResult`**: Dataclass holding alignment results (form_id, cognateset_id, aligned_form)
|
|
222
|
+
- **`CognateSet`**: Dataclass representing a cognate set with its entries
|
|
223
|
+
- **`Form`**: Dataclass representing a lexical form
|
|
224
|
+
- **`Language`**: Dataclass representing a language with metadata
|
|
225
|
+
|
|
226
|
+
### Functions
|
|
227
|
+
|
|
228
|
+
- **`align()`**: High-level function providing CLI-equivalent API
|
|
229
|
+
- **`load_cldf_dataset()`**: Load a CLDF dataset from path
|
|
230
|
+
- **`write_alignments()`**: Write alignment results to CSV
|
|
231
|
+
- **`select_best_anchor_language()`**: Find optimal anchor language for alignment
|
|
232
|
+
|
|
233
|
+
## License
|
|
234
|
+
|
|
235
|
+
MIT License
|
|
236
|
+
|
|
237
|
+
## Citation
|
|
238
|
+
|
|
239
|
+
If you use this package in your research, please cite:
|
|
240
|
+
|
|
241
|
+
```bibtex
|
|
242
|
+
@software{emalign,
|
|
243
|
+
title = {emalign-phonology: EM-based cognate alignment},
|
|
244
|
+
author = {Changeling Lab},
|
|
245
|
+
url = {https://github.com/changelinglab/emalign},
|
|
246
|
+
year = {2024}
|
|
247
|
+
}
|
|
248
|
+
```
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "emalign-phonology"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "EM-based alignment of cognate sets in comparative dictionaries using articulatory features"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [
|
|
13
|
+
{name = "Changeling Lab", email = "changeling@example.com"},
|
|
14
|
+
]
|
|
15
|
+
keywords = [
|
|
16
|
+
"linguistics",
|
|
17
|
+
"phonology",
|
|
18
|
+
"alignment",
|
|
19
|
+
"cognates",
|
|
20
|
+
"historical-linguistics",
|
|
21
|
+
"CLDF",
|
|
22
|
+
"panphon",
|
|
23
|
+
"expectation-maximization",
|
|
24
|
+
]
|
|
25
|
+
classifiers = [
|
|
26
|
+
"Development Status :: 3 - Alpha",
|
|
27
|
+
"Intended Audience :: Science/Research",
|
|
28
|
+
"Operating System :: OS Independent",
|
|
29
|
+
"Programming Language :: Python :: 3",
|
|
30
|
+
"Programming Language :: Python :: 3.9",
|
|
31
|
+
"Programming Language :: Python :: 3.10",
|
|
32
|
+
"Programming Language :: Python :: 3.11",
|
|
33
|
+
"Programming Language :: Python :: 3.12",
|
|
34
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
35
|
+
"Topic :: Text Processing :: Linguistic",
|
|
36
|
+
]
|
|
37
|
+
dependencies = [
|
|
38
|
+
"numpy>=1.21.0",
|
|
39
|
+
"panphon>=0.20.0",
|
|
40
|
+
"pycldf>=1.30.0",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[project.optional-dependencies]
|
|
44
|
+
dev = [
|
|
45
|
+
"pytest>=7.0.0",
|
|
46
|
+
"pytest-cov>=4.0.0",
|
|
47
|
+
"build>=1.0.0",
|
|
48
|
+
"twine>=4.0.0",
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
[project.urls]
|
|
52
|
+
Homepage = "https://github.com/changelinglab/emalign"
|
|
53
|
+
Repository = "https://github.com/changelinglab/emalign"
|
|
54
|
+
Issues = "https://github.com/changelinglab/emalign/issues"
|
|
55
|
+
|
|
56
|
+
[project.scripts]
|
|
57
|
+
emalign = "emalign.cli:main"
|
|
58
|
+
|
|
59
|
+
[tool.setuptools.packages.find]
|
|
60
|
+
where = ["src"]
|
|
61
|
+
|
|
62
|
+
[tool.pytest.ini_options]
|
|
63
|
+
testpaths = ["tests"]
|
|
64
|
+
python_files = ["test_*.py"]
|