conlanggen 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- conlanggen-0.4.2/.gitattributes +6 -0
- conlanggen-0.4.2/.github/workflows/ci.yml +45 -0
- conlanggen-0.4.2/.github/workflows/release.yml +89 -0
- conlanggen-0.4.2/.gitignore +22 -0
- conlanggen-0.4.2/.pre-commit-config.yaml +12 -0
- conlanggen-0.4.2/CONTEXT.md +67 -0
- conlanggen-0.4.2/LICENSE +30 -0
- conlanggen-0.4.2/PKG-INFO +377 -0
- conlanggen-0.4.2/README.md +324 -0
- conlanggen-0.4.2/docs/adr/0001-architecture.md +119 -0
- conlanggen-0.4.2/docs/pipeline.md +166 -0
- conlanggen-0.4.2/pyproject.toml +64 -0
- conlanggen-0.4.2/spec/example-ipa.yaml +22 -0
- conlanggen-0.4.2/spec/example.yaml +18 -0
- conlanggen-0.4.2/src/conlang/__init__.py +29 -0
- conlanggen-0.4.2/src/conlang/bench/__init__.py +5 -0
- conlanggen-0.4.2/src/conlang/bench/kah.py +138 -0
- conlanggen-0.4.2/src/conlang/cli.py +430 -0
- conlanggen-0.4.2/src/conlang/data/the_quiet_morning_sentences.csv +112 -0
- conlanggen-0.4.2/src/conlang/lexicon/__init__.py +5 -0
- conlanggen-0.4.2/src/conlang/lexicon/generate.py +323 -0
- conlanggen-0.4.2/src/conlang/lint/__init__.py +181 -0
- conlanggen-0.4.2/src/conlang/model/__init__.py +168 -0
- conlanggen-0.4.2/src/conlang/morphology/__init__.py +4 -0
- conlanggen-0.4.2/src/conlang/morphology/generate.py +402 -0
- conlanggen-0.4.2/src/conlang/phonology/__init__.py +5 -0
- conlanggen-0.4.2/src/conlang/phonology/romanize.py +170 -0
- conlanggen-0.4.2/src/conlang/phonology/syllable.py +159 -0
- conlanggen-0.4.2/src/conlang/pipeline/__init__.py +214 -0
- conlanggen-0.4.2/src/conlang/pipeline/compose.py +264 -0
- conlanggen-0.4.2/src/conlang/pipeline/extras.py +155 -0
- conlanggen-0.4.2/src/conlang/pipeline/formats.py +58 -0
- conlanggen-0.4.2/src/conlang/pipeline/grow.py +88 -0
- conlanggen-0.4.2/src/conlang/pipeline/port.py +77 -0
- conlanggen-0.4.2/src/conlang/pipeline/scaffold.py +174 -0
- conlanggen-0.4.2/src/conlang/pipeline/validate.py +119 -0
- conlanggen-0.4.2/src/conlang/render/__init__.py +5 -0
- conlanggen-0.4.2/src/conlang/render/markdown.py +176 -0
- conlanggen-0.4.2/src/conlang/sampling/__init__.py +1 -0
- conlanggen-0.4.2/src/conlang/sampling/concepts.py +190 -0
- conlanggen-0.4.2/src/conlang/sampling/defaults.py +145 -0
- conlanggen-0.4.2/src/conlang/sampling/rng.py +76 -0
- conlanggen-0.4.2/src/conlang/sampling/segments.py +100 -0
- conlanggen-0.4.2/src/conlang/soundchange/__init__.py +208 -0
- conlanggen-0.4.2/src/conlang/spec.py +145 -0
- conlanggen-0.4.2/src/conlang/syntax/__init__.py +302 -0
- conlanggen-0.4.2/tests/golden.json +5 -0
- conlanggen-0.4.2/tests/test_adpositions.py +91 -0
- conlanggen-0.4.2/tests/test_aspect.py +50 -0
- conlanggen-0.4.2/tests/test_bench.py +93 -0
- conlanggen-0.4.2/tests/test_clause_sequencing.py +74 -0
- conlanggen-0.4.2/tests/test_cli.py +62 -0
- conlanggen-0.4.2/tests/test_golden.py +82 -0
- conlanggen-0.4.2/tests/test_intransitive_clauses.py +65 -0
- conlanggen-0.4.2/tests/test_invariants.py +174 -0
- conlanggen-0.4.2/tests/test_ipa.py +81 -0
- conlanggen-0.4.2/tests/test_lexicon.py +27 -0
- conlanggen-0.4.2/tests/test_lint.py +27 -0
- conlanggen-0.4.2/tests/test_lint_invariants.py +87 -0
- conlanggen-0.4.2/tests/test_morphology.py +281 -0
- conlanggen-0.4.2/tests/test_noun_phrases.py +92 -0
- conlanggen-0.4.2/tests/test_phonotactics.py +92 -0
- conlanggen-0.4.2/tests/test_pipeline_commands.py +179 -0
- conlanggen-0.4.2/tests/test_predication.py +83 -0
- conlanggen-0.4.2/tests/test_repo_hygiene.py +43 -0
- conlanggen-0.4.2/tests/test_reproducibility.py +21 -0
- conlanggen-0.4.2/tests/test_rng.py +16 -0
- conlanggen-0.4.2/tests/test_soundchange.py +161 -0
- conlanggen-0.4.2/tests/test_syntax.py +102 -0
- conlanggen-0.4.2/wayfinder/map.md +55 -0
- conlanggen-0.4.2/wayfinder/research/capture-kit-contracts.md +402 -0
- conlanggen-0.4.2/wayfinder/tickets/build-pipeline-commands.md +44 -0
- conlanggen-0.4.2/wayfinder/tickets/capture-kit-contracts.md +40 -0
- conlanggen-0.4.2/wayfinder/tickets/design-proof-language.md +30 -0
- conlanggen-0.4.2/wayfinder/tickets/grammar-match-checklist.md +42 -0
- conlanggen-0.4.2/wayfinder/tickets/pin-grammar-in-new.md +26 -0
- conlanggen-0.4.2/wayfinder/tickets/pipeline-command-surface.md +58 -0
- conlanggen-0.4.2/wayfinder/tickets/pipeline-runbook.md +26 -0
- conlanggen-0.4.2/wayfinder/tickets/prove-fidelity.md +32 -0
- conlanggen-0.4.2/wayfinder/tickets/standardize-workspaces.md +40 -0
- conlanggen-0.4.2/wayfinder/tickets/translate-proof-language.md +29 -0
- conlanggen-0.4.2/wayfinder/tickets/workspace-layout.md +53 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main, master]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
# Reproducibility is a hard invariant and the bytes on disk are the artifact
|
|
11
|
+
# users get, so the suite runs on more than the author's platform. A passing
|
|
12
|
+
# Linux-only run would not catch a platform-dependent newline or path.
|
|
13
|
+
runs-on: ${{ matrix.os }}
|
|
14
|
+
strategy:
|
|
15
|
+
fail-fast: false
|
|
16
|
+
matrix:
|
|
17
|
+
os: [ubuntu-latest, windows-latest]
|
|
18
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
- uses: actions/setup-python@v5
|
|
22
|
+
with:
|
|
23
|
+
python-version: ${{ matrix.python-version }}
|
|
24
|
+
- name: Install
|
|
25
|
+
run: |
|
|
26
|
+
python -m pip install --upgrade pip
|
|
27
|
+
pip install -e ".[dev]"
|
|
28
|
+
- name: Lint
|
|
29
|
+
run: ruff check .
|
|
30
|
+
- name: Type-check
|
|
31
|
+
run: mypy src
|
|
32
|
+
- name: Test
|
|
33
|
+
run: pytest
|
|
34
|
+
- name: Build (sdist is platform-independent)
|
|
35
|
+
# Built on one platform only: the sdist and wheel must not vary by runner,
|
|
36
|
+
# so building per-matrix would create artifacts that differ for no reason.
|
|
37
|
+
if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.12'
|
|
38
|
+
run: |
|
|
39
|
+
python -m pip install --upgrade build
|
|
40
|
+
python -m build
|
|
41
|
+
- uses: actions/upload-artifact@v4
|
|
42
|
+
if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.12'
|
|
43
|
+
with:
|
|
44
|
+
name: dist
|
|
45
|
+
path: dist/
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
|
|
7
|
+
permissions:
|
|
8
|
+
contents: read
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
test:
|
|
12
|
+
# A release is gated on the full matrix, not just the version being cut.
|
|
13
|
+
strategy:
|
|
14
|
+
fail-fast: false
|
|
15
|
+
matrix:
|
|
16
|
+
os: [ubuntu-latest, windows-latest]
|
|
17
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
18
|
+
runs-on: ${{ matrix.os }}
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
- uses: actions/setup-python@v5
|
|
22
|
+
with:
|
|
23
|
+
python-version: ${{ matrix.python-version }}
|
|
24
|
+
- name: Install
|
|
25
|
+
run: |
|
|
26
|
+
python -m pip install --upgrade pip
|
|
27
|
+
pip install -e ".[dev]"
|
|
28
|
+
- name: Lint
|
|
29
|
+
run: ruff check .
|
|
30
|
+
- name: Type-check
|
|
31
|
+
run: mypy src
|
|
32
|
+
- name: Test
|
|
33
|
+
run: pytest
|
|
34
|
+
|
|
35
|
+
publish:
|
|
36
|
+
needs: test
|
|
37
|
+
runs-on: ubuntu-latest
|
|
38
|
+
environment: pypi
|
|
39
|
+
permissions:
|
|
40
|
+
# Job-level `permissions` *replace* the workflow-level ones (unspecified
|
|
41
|
+
# scopes become `none`), so `contents: read` must be listed here too or
|
|
42
|
+
# actions/checkout in this job cannot read the repo.
|
|
43
|
+
contents: read
|
|
44
|
+
# Required for OIDC trusted publishing: no API token is stored anywhere.
|
|
45
|
+
id-token: write
|
|
46
|
+
steps:
|
|
47
|
+
- uses: actions/checkout@v4
|
|
48
|
+
- uses: actions/setup-python@v5
|
|
49
|
+
with:
|
|
50
|
+
python-version: "3.12"
|
|
51
|
+
|
|
52
|
+
- name: Check tag matches project version
|
|
53
|
+
run: |
|
|
54
|
+
tag="${GITHUB_REF_NAME#v}"
|
|
55
|
+
version="$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text(encoding='utf-8'))['project']['version'])")"
|
|
56
|
+
if [ "$tag" != "$version" ]; then
|
|
57
|
+
echo "tag $tag does not match project version $version" >&2
|
|
58
|
+
exit 1
|
|
59
|
+
fi
|
|
60
|
+
echo "publishing $version"
|
|
61
|
+
|
|
62
|
+
- name: Build
|
|
63
|
+
run: |
|
|
64
|
+
python -m pip install --upgrade build
|
|
65
|
+
python -m build
|
|
66
|
+
|
|
67
|
+
- uses: actions/upload-artifact@v4
|
|
68
|
+
with:
|
|
69
|
+
name: dist
|
|
70
|
+
path: dist/
|
|
71
|
+
|
|
72
|
+
- name: Show OIDC token claims (debug)
|
|
73
|
+
continue-on-error: true
|
|
74
|
+
run: |
|
|
75
|
+
python - <<'PY'
|
|
76
|
+
import base64, json, os, urllib.request
|
|
77
|
+
url = os.environ["ACTIONS_ID_TOKEN_REQUEST_URL"] + "&audience=pypi"
|
|
78
|
+
req = urllib.request.Request(
|
|
79
|
+
url,
|
|
80
|
+
headers={"Authorization": "bearer " + os.environ["ACTIONS_ID_TOKEN_REQUEST_TOKEN"]},
|
|
81
|
+
)
|
|
82
|
+
token = json.load(urllib.request.urlopen(req))["value"]
|
|
83
|
+
payload = token.split(".")[1]
|
|
84
|
+
payload += "=" * (-len(payload) % 4)
|
|
85
|
+
print(json.dumps(json.loads(base64.urlsafe_b64decode(payload)), indent=2))
|
|
86
|
+
PY
|
|
87
|
+
|
|
88
|
+
- name: Publish to PyPI
|
|
89
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Byte-compiled / optimized / DLL files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
.eggs/
|
|
6
|
+
build/
|
|
7
|
+
dist/
|
|
8
|
+
|
|
9
|
+
# Virtual environments
|
|
10
|
+
.venv/
|
|
11
|
+
venv/
|
|
12
|
+
|
|
13
|
+
# Tooling caches
|
|
14
|
+
.pytest_cache/
|
|
15
|
+
.ruff_cache/
|
|
16
|
+
.mypy_cache/
|
|
17
|
+
coverage.xml
|
|
18
|
+
.coverage
|
|
19
|
+
|
|
20
|
+
# Generated output
|
|
21
|
+
out/
|
|
22
|
+
*.local.yaml
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
repos:
|
|
2
|
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
3
|
+
rev: v0.6.9
|
|
4
|
+
hooks:
|
|
5
|
+
- id: ruff
|
|
6
|
+
args: [--fix]
|
|
7
|
+
- id: ruff-format
|
|
8
|
+
- repo: https://github.com/pre-commit/mirrors-mypy
|
|
9
|
+
rev: v1.11.2
|
|
10
|
+
hooks:
|
|
11
|
+
- id: mypy
|
|
12
|
+
additional_dependencies: ["pydantic>=2.7", "pyyaml>=6", "types-PyYAML>=6"]
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# CONTEXT — domain model and vocabulary
|
|
2
|
+
|
|
3
|
+
Shared language for this project. Update it when terminology shifts; the terms
|
|
4
|
+
here are load-bearing in code, specs, and issues.
|
|
5
|
+
|
|
6
|
+
## Core objects
|
|
7
|
+
|
|
8
|
+
| Term | Definition |
|
|
9
|
+
| --- | --- |
|
|
10
|
+
| **Spec** | What the user writes (`spec/*.yaml`). Partial by design; omitted fields are sampled. |
|
|
11
|
+
| **Model** | The canonical JSON (`Language`) produced by a run. Single source of truth; every view is rendered from it. |
|
|
12
|
+
| **Lexeme** | A dictionary entry: id, gloss, part of speech, and a **root**. |
|
|
13
|
+
| **Root** | The bare, citation form of a lexeme. |
|
|
14
|
+
| **Affix** | A generated prefix/suffix with a gloss (e.g. `PL`) and the parts of speech it applies to. |
|
|
15
|
+
| **Allomorph** | A conditioned surface form of an affix, used when the primary form would be illegal in context. |
|
|
16
|
+
| **Adposition** | A generated, invariable word (pos `adp`) for a place/direction relation (`in`, `above`, `toward`…). Its order is fixed by word order: pre- for VO, post- for OV. |
|
|
17
|
+
| **Oblique** | An optional clause adjunct: an adposition plus a bare noun, appended after the object. |
|
|
18
|
+
| **Noun phrase** | A head noun plus optional modifiers (adjectives, possessor, apposition, an adpositional phrase), built on demand. Modifiers sit on the head's side, fixed by head direction. |
|
|
19
|
+
| **Clause** | A predicate with its arguments, linearised on demand: **transitive** (subject + object) or **intransitive** (one argument, which takes the alignment's intransitive case: `NOM` if accusative, `ABS` if ergative). |
|
|
20
|
+
| **Copula** | A generated `be` word (pos `cop`) that predicates a complement. It draws verb morphology (past/future, agreement) by sharing the verb classes, but is not a content verb. |
|
|
21
|
+
| **Predication** | A copular clause: subject + copula + a bare predicative complement (adjective or noun). Existential *there is* is paraphrased as this shape plus a locative oblique. |
|
|
22
|
+
| **Aspect** | A verb category stacked with tense: **progressive** and **perfect** affixes. Pluperfect is `past + perfect`; "had been …-ing" is `past + perfect + progressive`. |
|
|
23
|
+
| **Conjunction** | A generated, invariable connective (`and`, `but`, `then`, `so`; pos `conj`) that joins clauses. |
|
|
24
|
+
| **Sentence** | Two or more clauses joined by **parataxis** (juxtaposition) or a conjunction; relatives and reported speech ride on parataxis. Built on demand, not stored. |
|
|
25
|
+
| **WordRecord** | A surface form plus its morphemes, syllables, and tags (`CIT`, `PL`). |
|
|
26
|
+
| **Origin** | Where a value came from: `specified`, `sampled`, or `derived`. |
|
|
27
|
+
| **Lock** | An entry the user pinned. Locked entries survive re-sampling. |
|
|
28
|
+
| **Seed** | Integer driving all sampling. Same spec + seed ⇒ identical bytes. |
|
|
29
|
+
|
|
30
|
+
## Phonology
|
|
31
|
+
|
|
32
|
+
| Term | Definition |
|
|
33
|
+
| --- | --- |
|
|
34
|
+
| **Inventory** | The consonants and vowels a language uses. Symbols are user-facing; v1 has no feature layer beneath them. A spec may declare the inventory phonemic (`ipa: true`), in which case each symbol is an IPA phoneme. |
|
|
35
|
+
| **Orthography** | The romanised spelling derived from an IPA inventory (symbol → spelling). A *view* of the phonemic form, stored on the inventory and on each lexeme/word; absence means the symbol spells itself. |
|
|
36
|
+
| **Syllable template** | A shape over `C` and `V` slots, e.g. `CVC`, with a sampling weight. |
|
|
37
|
+
| **Phonotactics** | The legality rules: a word is legal iff its `C`/`V` skeleton can be partitioned into the templates with allowed onsets/codas. Constraints bind *during* the search, so the verdict never depends on which partition is found first. |
|
|
38
|
+
| **Segment** | One inventory symbol, possibly multi-character (`sh`, `ch`). |
|
|
39
|
+
|
|
40
|
+
## Process
|
|
41
|
+
|
|
42
|
+
| Term | Definition |
|
|
43
|
+
| --- | --- |
|
|
44
|
+
| **Generate** | `spec → model`. Samples the gaps, assigns stable per-item seeds, assembles word forms. |
|
|
45
|
+
| **Stable per-item seeding** | Each item hashes its own label into its RNG, so generating one extra word never reshuffles the others. |
|
|
46
|
+
| **Lint** | Machine-checkable invariants run after generation. Errors mark a model `invalid`; warnings do not. |
|
|
47
|
+
| **Render** | Model → Markdown dictionary + grammar sketch. Never hand-maintained. |
|
|
48
|
+
| **Bench** | Adapter + plausibility scorecard against the real Kah dictionary. Reported, never a pass/fail gate. |
|
|
49
|
+
| **Lineage** | Proto-language/rule hooks carried in the schema for phase-2 diachrony. |
|
|
50
|
+
|
|
51
|
+
## Translation pipeline
|
|
52
|
+
|
|
53
|
+
| Term | Definition |
|
|
54
|
+
| --- | --- |
|
|
55
|
+
| **Pipeline command** | One of `conlang new` / `grow` / `translate` / `validate-translation` / `extras` — the repeatable way to build a generated language and translate the fixed story into it. |
|
|
56
|
+
| **Workspace** | The standard home of a generated language: `MISSION`/`NOTES`/`README`/`RESOURCES`, `spec/`, `out/`, `reference/`, `learning-records/`, and the bundled story. One per language. |
|
|
57
|
+
| **Composer** | The library object (`conlang.Composer`) that realises a token (`gloss.TAG…`) as a surface word the model licenses, and reads a surface back to its gloss. |
|
|
58
|
+
| **Gloss script** | A hand-authored TSV (`index<TAB>gloss`) that `translate` composes into a translation. |
|
|
59
|
+
| **Port** | Re-composing a translation from a sibling language's gloss column instead of writing a gloss script. |
|
|
60
|
+
| **Portability** | The grammar-match check that decides port vs hand-translate: alignment + word order (adposition/modifier order asserted) and the source's tag vocabulary present in the target. |
|
|
61
|
+
|
|
62
|
+
## Non-negotiable invariants
|
|
63
|
+
|
|
64
|
+
1. **Reproducibility is hard.** Same spec + seed produces byte-identical output.
|
|
65
|
+
2. **The model is the source of truth.** Views are derived; never edit a view.
|
|
66
|
+
3. **Lint never crashes a run.** Broken languages are generated and reported.
|
|
67
|
+
4. **Kah data stays out of the repo.** The benchmark reads a local cache at runtime.
|
conlanggen-0.4.2/LICENSE
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Brian McKeen
|
|
4
|
+
|
|
5
|
+
Contact: bmckeen1919@gmail.com
|
|
6
|
+
|
|
7
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
8
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
9
|
+
in the Software without restriction, including without limitation the rights
|
|
10
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
11
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
12
|
+
furnished to do so, subject to the following conditions:
|
|
13
|
+
|
|
14
|
+
The above copyright notice and this permission notice shall be included in all
|
|
15
|
+
copies or substantial portions of the Software.
|
|
16
|
+
|
|
17
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
18
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
19
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
20
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
21
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
22
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
23
|
+
SOFTWARE.
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
Note on third-party data: this project does not include the Kwesho Kah
|
|
28
|
+
dictionary or grammar, which are the copyrighted work of Kwesho. The Kah
|
|
29
|
+
benchmark reads them from a local cache at runtime and they are not distributed
|
|
30
|
+
here. See README.md.
|
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: conlanggen
|
|
3
|
+
Version: 0.4.2
|
|
4
|
+
Summary: A partial-spec conlang generator: specify what you care about, sample the rest.
|
|
5
|
+
Project-URL: Homepage, https://github.com/bmckeen1919-lab/conlanggen
|
|
6
|
+
Project-URL: Repository, https://github.com/bmckeen1919-lab/conlanggen
|
|
7
|
+
Project-URL: Issues, https://github.com/bmckeen1919-lab/conlanggen/issues
|
|
8
|
+
Author: Brian McKeen
|
|
9
|
+
License: MIT License
|
|
10
|
+
|
|
11
|
+
Copyright (c) 2026 Brian McKeen
|
|
12
|
+
|
|
13
|
+
Contact: bmckeen1919@gmail.com
|
|
14
|
+
|
|
15
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
16
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
17
|
+
in the Software without restriction, including without limitation the rights
|
|
18
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
19
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
20
|
+
furnished to do so, subject to the following conditions:
|
|
21
|
+
|
|
22
|
+
The above copyright notice and this permission notice shall be included in all
|
|
23
|
+
copies or substantial portions of the Software.
|
|
24
|
+
|
|
25
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
26
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
27
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
28
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
29
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
30
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
31
|
+
SOFTWARE.
|
|
32
|
+
|
|
33
|
+
---
|
|
34
|
+
|
|
35
|
+
Note on third-party data: this project does not include the Kwesho Kah
|
|
36
|
+
dictionary or grammar, which are the copyrighted work of Kwesho. The Kah
|
|
37
|
+
benchmark reads them from a local cache at runtime and they are not distributed
|
|
38
|
+
here. See README.md.
|
|
39
|
+
License-File: LICENSE
|
|
40
|
+
Keywords: conlang,linguistics,procedural-generation,worldbuilding
|
|
41
|
+
Requires-Python: >=3.11
|
|
42
|
+
Requires-Dist: pydantic>=2.7
|
|
43
|
+
Requires-Dist: pyyaml>=6
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
46
|
+
Requires-Dist: pre-commit>=3.7; extra == 'dev'
|
|
47
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
48
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
49
|
+
Requires-Dist: types-pyyaml>=6; extra == 'dev'
|
|
50
|
+
Provides-Extra: phoible
|
|
51
|
+
Provides-Extra: wals
|
|
52
|
+
Description-Content-Type: text/markdown
|
|
53
|
+
|
|
54
|
+
# conlang
|
|
55
|
+
|
|
56
|
+
A partial-spec constructed-language generator.
|
|
57
|
+
|
|
58
|
+
You describe only what you care about — phonology, syllable shapes, a handful of
|
|
59
|
+
words — and the generator **samples the rest** from typologically-plausible
|
|
60
|
+
defaults. Same spec plus same seed always produces byte-identical output, and
|
|
61
|
+
sampled entries can be locked so they survive re-sampling.
|
|
62
|
+
|
|
63
|
+
```text
|
|
64
|
+
spec (YAML) ──▶ generate ──▶ model (canonical JSON) ──▶ render ──▶ dictionary + grammar sketch
|
|
65
|
+
│
|
|
66
|
+
├── lint (machine-checkable invariants)
|
|
67
|
+
└── bench (plausibility scorecard vs. Kah)
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
The model JSON is the single source of truth; the dictionary, grammar sketch,
|
|
71
|
+
and any future views are rendered from it.
|
|
72
|
+
|
|
73
|
+
## Install
|
|
74
|
+
|
|
75
|
+
```powershell
|
|
76
|
+
pip install conlanggen
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
The distribution is `conlanggen`, but the import package and the command are both
|
|
80
|
+
`conlang` — `import conlang`, `conlang generate`.
|
|
81
|
+
|
|
82
|
+
## Quickstart
|
|
83
|
+
|
|
84
|
+
```powershell
|
|
85
|
+
python -m venv .venv
|
|
86
|
+
.\.venv\Scripts\python.exe -m pip install -e ".[dev]"
|
|
87
|
+
|
|
88
|
+
# Generate from the example spec, and render a Markdown dictionary.
|
|
89
|
+
.\.venv\Scripts\python.exe -m conlang.cli generate --spec spec\example.yaml --render
|
|
90
|
+
|
|
91
|
+
# Happy with it? Freeze it into an explicit spec so the values stick.
|
|
92
|
+
conlang accept --model out\veshtari.json
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Output lands in `out/`: a model JSON and a rendered `.md`.
|
|
96
|
+
|
|
97
|
+
```powershell
|
|
98
|
+
# Re-lint or re-render an existing model.
|
|
99
|
+
conlang lint --model out\veshtari.json
|
|
100
|
+
conlang render --model out\veshtari.json --out out\veshtari.md
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
## Translating a fixed story
|
|
104
|
+
|
|
105
|
+
The toolkit ships a repeatable pipeline for translating the fixed 111-sentence
|
|
106
|
+
story *The Quiet Morning* into a **generated** language: `conlang new` → `grow` →
|
|
107
|
+
`translate` → `validate-translation` → `extras`. The runbook is
|
|
108
|
+
[`docs/pipeline.md`](docs/pipeline.md).
|
|
109
|
+
|
|
110
|
+
## Writing a spec
|
|
111
|
+
|
|
112
|
+
Everything is optional. Omitted fields are sampled.
|
|
113
|
+
|
|
114
|
+
```yaml
|
|
115
|
+
name: Veshtari
|
|
116
|
+
seed: 42
|
|
117
|
+
|
|
118
|
+
phonology:
|
|
119
|
+
consonants: [p, t, k, m, n, s, l, r, v, sh, ch]
|
|
120
|
+
vowels: [a, e, i, o, u]
|
|
121
|
+
|
|
122
|
+
syllables: [CV, CVC, V] # C = consonant slot, V = vowel slot
|
|
123
|
+
word_count: 150
|
|
124
|
+
|
|
125
|
+
lexicon: # pin specific words; they are locked by default
|
|
126
|
+
- gloss: water
|
|
127
|
+
root: vela
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Accepted keys: `name`, `seed`, `phonology.{consonants,vowels,ipa,orthography}`,
|
|
131
|
+
`syllables`, `concepts`, `word_count`, `lexicon`, `affixes`, `onsets`, `codas`,
|
|
132
|
+
`forbidden`.
|
|
133
|
+
|
|
134
|
+
An explicit, unsized `lexicon` is treated as the complete word list (that is
|
|
135
|
+
what `accept` writes back); otherwise the built-in concept list is used and
|
|
136
|
+
`word_count` trims or extends it.
|
|
137
|
+
|
|
138
|
+
If your `syllables` include `CCV`/`CCVC`, the generator samples a legal onset
|
|
139
|
+
set unless you supply `onsets` yourself, and `accept` records whichever was
|
|
140
|
+
used so a frozen spec regenerates identically.
|
|
141
|
+
|
|
142
|
+
### IPA inventories
|
|
143
|
+
|
|
144
|
+
By default a symbol *is* its spelling (`sh` is written `sh`). Set
|
|
145
|
+
`phonology.ipa: true` to declare the inventory phonemic: each symbol is an IPA
|
|
146
|
+
phoneme, and the tool derives a romanised **orthography** for the dictionary.
|
|
147
|
+
|
|
148
|
+
```yaml
|
|
149
|
+
name: Shibo
|
|
150
|
+
seed: 11
|
|
151
|
+
|
|
152
|
+
phonology:
|
|
153
|
+
ipa: true
|
|
154
|
+
consonants: [p, t, k, ʃ, tʃ, m, n, s, l]
|
|
155
|
+
vowels: [a, e, i, o, u]
|
|
156
|
+
orthography: # optional; corrects or extends the built-in defaults
|
|
157
|
+
tʃ: ch
|
|
158
|
+
ʃ: sh
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
`/ʃ/` spells `sh` and `/tʃ/` spells `ch` from a built-in table; anything absent
|
|
162
|
+
spells itself, so a plain Latin inventory is unchanged. `orthography` overrides
|
|
163
|
+
the table (set an entry back to its own symbol to cancel a default), and the
|
|
164
|
+
resolved map is stored on the inventory. The dictionary and word-form tables
|
|
165
|
+
then show a `spelling` column beside the `ipa` one, and `accept` writes the map
|
|
166
|
+
back so a frozen spec regenerates identically. Lint warns on a map entry that is
|
|
167
|
+
not an inventory symbol, or two symbols that would spell the same thing.
|
|
168
|
+
|
|
169
|
+
## Freezing a language (`accept`)
|
|
170
|
+
|
|
171
|
+
A partial spec samples whatever you leave out. Once you like the result,
|
|
172
|
+
`conlang accept` writes it back as a **fully-explicit spec**: the phonology,
|
|
173
|
+
syllable shapes, constraints, affixes, and every root become pinned entries.
|
|
174
|
+
Regenerating from that spec reproduces the same language exactly — sampled
|
|
175
|
+
values now survive re-sampling, and changing the seed can no longer disturb
|
|
176
|
+
them.
|
|
177
|
+
|
|
178
|
+
```powershell
|
|
179
|
+
conlang generate --spec spec\example.yaml --out out\veshtari.json
|
|
180
|
+
conlang accept --model out\veshtari.json --out spec/veshtari.yaml
|
|
181
|
+
conlang generate --spec spec/veshtari.yaml --out out/veshtari.json
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Two fields legitimately change across the round trip: `origin` becomes
|
|
185
|
+
`specified` for every accepted entry (it is pinned now, not sampled), and
|
|
186
|
+
`spec_hash` changes because the spec text is different. Everything linguistic
|
|
187
|
+
is identical, and `tests/test_invariants.py` asserts exactly that.
|
|
188
|
+
|
|
189
|
+
## Guarantees and lint rules
|
|
190
|
+
|
|
191
|
+
Reproducibility is the one **hard** guarantee: `spec + seed ⇒ identical bytes`.
|
|
192
|
+
Everything else is reported by `lint` as errors or warnings, and a broken
|
|
193
|
+
language is still generated (conlangers want to explore broken states).
|
|
194
|
+
|
|
195
|
+
| rule | severity | meaning |
|
|
196
|
+
| --- | --- | --- |
|
|
197
|
+
| `schema` | error/warning | schema version present and known |
|
|
198
|
+
| `lexeme.root` | error | lexeme has a non-empty root |
|
|
199
|
+
| `phonotactics.lexeme` | error | every root is a legal word |
|
|
200
|
+
| `phonotactics.word` | error | every inflected form is a legal word |
|
|
201
|
+
| `lexicon.homophones` | warning | two roots collide (Kah has polysemy too) |
|
|
202
|
+
| `morphology.paradigm` | warning | a word is missing a form for a category its class realises |
|
|
203
|
+
| `morphology.class` | error | a word names an inflection class the language does not define |
|
|
204
|
+
| `lexicon.nonempty` | warning | the lexicon is empty |
|
|
205
|
+
| `orthography.unknown` | warning | the spelling map names a symbol that is not in the inventory |
|
|
206
|
+
| `orthography.collision` | warning | two symbols would spell the same thing |
|
|
207
|
+
|
|
208
|
+
Every rule above has a test in `tests/test_lint.py`, `tests/test_lint_invariants.py`,
|
|
209
|
+
or `tests/test_morphology.py`.
|
|
210
|
+
|
|
211
|
+
## Phonotactics
|
|
212
|
+
|
|
213
|
+
A word is legal when its `C`/`V` skeleton can be partitioned into the language's
|
|
214
|
+
syllable templates **and** every resulting onset and coda is allowed. Two details
|
|
215
|
+
matter more than they look:
|
|
216
|
+
|
|
217
|
+
- **Constraints are checked while syllabifying, not after.** A word often has
|
|
218
|
+
several valid partitions; `deplango` is `de|plan|go` with legal onsets, and
|
|
219
|
+
also `de|pla|ngo` with an illegal one. Verdict must depend on the phonology,
|
|
220
|
+
not on which partition the search happens to return first.
|
|
221
|
+
- **Onsets and codas are compared by segment, not by string.** An onset of
|
|
222
|
+
`ny` + `m` renders as `"nym"` — three characters, two segments. Matching
|
|
223
|
+
strings would silently reject legal words in any language with a digraph.
|
|
224
|
+
|
|
225
|
+
When a template set admits onset clusters (`CCV`, `CCVC`), the generator samples
|
|
226
|
+
a legal onset set and **builds roots from it**, so generated words are legal by
|
|
227
|
+
construction rather than by luck. `constraints.onsets` in the model records what
|
|
228
|
+
was sampled; an explicit `onsets` in the spec always wins.
|
|
229
|
+
|
|
230
|
+
Affixes carry **allomorphs**: conditioned surface forms. A single form cannot
|
|
231
|
+
serve both consonant-final and vowel-final roots, because the root's coda plus a
|
|
232
|
+
consonant-initial affix resyllabify into an onset cluster. When the primary form
|
|
233
|
+
would be illegal, `build_words` uses the first alternative that is legal — which
|
|
234
|
+
is what real languages do. It is a rare fallback (~2 in 8,500 affixed forms), but
|
|
235
|
+
without it those words would be illegal and lint would fail the language.
|
|
236
|
+
|
|
237
|
+
## The Kah benchmark
|
|
238
|
+
|
|
239
|
+
Kah (Kwesho's language) is hand-made, so the benchmark does **not** try to
|
|
240
|
+
regenerate it. It does two things:
|
|
241
|
+
|
|
242
|
+
1. **Adapter fixture** — parses the real Kwesho dictionary cache into the model
|
|
243
|
+
schema, proving the schema can represent real data.
|
|
244
|
+
2. **Plausibility scorecard** — compares a generated language against Kah.
|
|
245
|
+
|
|
246
|
+
The scorecard deliberately reports **three separate numbers**, because a word can
|
|
247
|
+
fail for two unrelated reasons and collapsing them makes the report unreadable:
|
|
248
|
+
|
|
249
|
+
| metric | question |
|
|
250
|
+
| --- | --- |
|
|
251
|
+
| symbol coverage | which Kah sounds does this language have? |
|
|
252
|
+
| usable sounds | how many Kah headwords use only sounds this language has? |
|
|
253
|
+
| legal shapes | of those, how many fit the phonotactics? |
|
|
254
|
+
|
|
255
|
+
Scoring the built-in `kah` preset against Kah is circular, so the report says so
|
|
256
|
+
explicitly when it detects it. Treat the scorecard as a lever, not a target: the
|
|
257
|
+
presets are deliberately small, and a language that only allows `CV` syllables is
|
|
258
|
+
a legitimate (if monotonous) language type, not a bug to tune away.
|
|
259
|
+
|
|
260
|
+
```powershell
|
|
261
|
+
conlang bench --spec spec\example.yaml
|
|
262
|
+
# or point at a cache explicitly:
|
|
263
|
+
conlang bench --spec spec\example.yaml --kah C:\path\to\kah-lexicon.txt
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
Kah's dictionary is Kwesho's copyrighted work and is **never** vendored into this
|
|
267
|
+
repository. Without `--kah`, the benchmark looks for the glossary at a per-user
|
|
268
|
+
cache path (`%LOCALAPPDATA%\conlang\kah-lexicon.txt` on Windows,
|
|
269
|
+
`~/.cache/conlang/kah-lexicon.txt` elsewhere, or `$XDG_CACHE_HOME/conlang/`);
|
|
270
|
+
override it with `--kah` or `CONLANG_KAH_LEXICON`.
|
|
271
|
+
|
|
272
|
+
On the reference machine, the adapter parses **8,153** entries with a **97.8%**
|
|
273
|
+
validity rate.
|
|
274
|
+
|
|
275
|
+
## Evolving a language
|
|
276
|
+
|
|
277
|
+
A generated model can be the parent of a daughter, evolved by ordered sound
|
|
278
|
+
changes. A rule is `old > new` (`new` empty deletes the segment), optionally
|
|
279
|
+
conditioned by an environment after `/`. The environment names the context on
|
|
280
|
+
each side of the target, with `_` marking the target slot, `#` the word boundary,
|
|
281
|
+
and `V`/`C` the vowel/consonant classes:
|
|
282
|
+
|
|
283
|
+
```powershell
|
|
284
|
+
conlang generate --spec spec\example.yaml --out out\proto.json
|
|
285
|
+
conlang evolve --model out\proto.json --rule "a > e" --rule "p > f / V_V" --out out\daughter.json
|
|
286
|
+
conlang evolve --model out\proto.json --rule "t > d / #_" --rule "s > z / V_" --out out\voiced.json
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
`V_V` is "between vowels", `#_` is word-initial, `_#` is word-final. Rules are
|
|
290
|
+
applied left to right so a later rule sees an earlier one's output (feeding).
|
|
291
|
+
|
|
292
|
+
The daughter is a full model: its inventory gains any new sounds, its lexicon
|
|
293
|
+
and affixes are rewritten, and its word forms are rebuilt. It records where it
|
|
294
|
+
came from in `lineage`:
|
|
295
|
+
|
|
296
|
+
```json
|
|
297
|
+
"lineage": { "parent": "proto", "rules": ["a > e", "p > f / V_V"] }
|
|
298
|
+
```
|
|
299
|
+
|
|
300
|
+
That is the whole point of the field: the derivation is data, so a daughter can
|
|
301
|
+
itself be evolved, and `accept` keeps the chain across a round trip. Mergers
|
|
302
|
+
introduced by a change show up as `lexicon.homophones` warnings, never errors.
|
|
303
|
+
|
|
304
|
+
## Syntax
|
|
305
|
+
|
|
306
|
+
Each language samples a constituent order (`SOV`, `SVO`, or `VSO`, weighted
|
|
307
|
+
toward the first two) and an alignment (`accusative` or `ergative`). Both are
|
|
308
|
+
recorded in the model and survive `accept`. A transitive clause is linearised
|
|
309
|
+
from the lexicon and inflected: the subject and object take case by alignment
|
|
310
|
+
(accusative marks `NOM`/`ACC`; ergative marks `ERG`/`ABS`) and the verb agrees
|
|
311
|
+
with the subject:
|
|
312
|
+
|
|
313
|
+
```python
|
|
314
|
+
from conlang.syntax import sample_clause
|
|
315
|
+
words, gloss = sample_clause(language)
|
|
316
|
+
# e.g. (['pyople', 'klavragprekbra', 'blonyto'], 'water.ERG say.3SG fire.ABS')
|
|
317
|
+
```
|
|
318
|
+
|
|
319
|
+
Case and agreement are applied when the clause is built, not woven into every
|
|
320
|
+
word's standing paradigm — a deliberate split, so a noun's form count does not
|
|
321
|
+
double again for case.
|
|
322
|
+
|
|
323
|
+
## Layout
|
|
324
|
+
|
|
325
|
+
```text
|
|
326
|
+
src/conlang/
|
|
327
|
+
model/ canonical pydantic schema (the source of truth)
|
|
328
|
+
sampling/ deterministic RNG, inventory presets, concepts, defaults
|
|
329
|
+
phonology/ tokenization, syllabification, legality checks
|
|
330
|
+
lexicon/ seeded root generation, stable per-item seeding
|
|
331
|
+
morphology/ affixes, inflection classes, and word-form assembly
|
|
332
|
+
soundchange/ ordered sound changes; derive a daughter language
|
|
333
|
+
syntax/ word order, alignment, simple clause linearisation
|
|
334
|
+
render/ Markdown dictionary + grammar sketch
|
|
335
|
+
lint/ machine-checkable invariants
|
|
336
|
+
bench/ Kah adapter + scorecard
|
|
337
|
+
cli.py thin command-line view over the library
|
|
338
|
+
```
|
|
339
|
+
|
|
340
|
+
## Development
|
|
341
|
+
|
|
342
|
+
```powershell
|
|
343
|
+
.\.venv\Scripts\python.exe -m pytest
|
|
344
|
+
.\.venv\Scripts\python.exe -m ruff check .
|
|
345
|
+
.\.venv\Scripts\python.exe -m mypy src
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
## Roadmap
|
|
349
|
+
|
|
350
|
+
Shipped so far:
|
|
351
|
+
|
|
352
|
+
- **Phonology** — inventory, phonotactics, seeded word generation.
|
|
353
|
+
- **IPA inventories** — declare the inventory phonemic (`ipa: true`) and the tool
|
|
354
|
+
derives a romanised orthography alongside the phonemic forms.
|
|
355
|
+
- **Morphology** — affix breadth, inflection classes, and stacking (a word can
|
|
356
|
+
carry several categories at once).
|
|
357
|
+
- **Diachrony** — `conlang evolve`: ordered sound changes derive a daughter
|
|
358
|
+
language, now with **environments** (`p > f / V_V`, `#`, `_`), and the chain
|
|
359
|
+
recorded in `lineage`.
|
|
360
|
+
- **Syntax** — sampled word order and alignment, case marking with subject
|
|
361
|
+
agreement, the **predication tier** (adpositions, noun phrases, intransitives,
|
|
362
|
+
copula/existential), **clause sequencing** (parataxis and conjunctions), and
|
|
363
|
+
**aspect** (progressive/perfect, stacking with tense).
|
|
364
|
+
- **Translation pipeline** — `conlang new` / `grow` / `translate` /
|
|
365
|
+
`validate-translation` / `extras`: build a generated language and translate the
|
|
366
|
+
fixed story into it, including a grammar-match **port** from a sibling language.
|
|
367
|
+
|
|
368
|
+
Next:
|
|
369
|
+
|
|
370
|
+
- Optional data packs: PHOIBLE (inventories) and WALS (typology) as CC-BY
|
|
371
|
+
adapters behind the built-in defaults.
|
|
372
|
+
- Semantics/pragmatics and writing systems (deferred from v1).
|
|
373
|
+
|
|
374
|
+
## License
|
|
375
|
+
|
|
376
|
+
MIT (code). Kah dictionary and grammar content remains Kwesho's and is not part
|
|
377
|
+
of this repository.
|