conlanggen 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. conlanggen-0.4.2/.gitattributes +6 -0
  2. conlanggen-0.4.2/.github/workflows/ci.yml +45 -0
  3. conlanggen-0.4.2/.github/workflows/release.yml +89 -0
  4. conlanggen-0.4.2/.gitignore +22 -0
  5. conlanggen-0.4.2/.pre-commit-config.yaml +12 -0
  6. conlanggen-0.4.2/CONTEXT.md +67 -0
  7. conlanggen-0.4.2/LICENSE +30 -0
  8. conlanggen-0.4.2/PKG-INFO +377 -0
  9. conlanggen-0.4.2/README.md +324 -0
  10. conlanggen-0.4.2/docs/adr/0001-architecture.md +119 -0
  11. conlanggen-0.4.2/docs/pipeline.md +166 -0
  12. conlanggen-0.4.2/pyproject.toml +64 -0
  13. conlanggen-0.4.2/spec/example-ipa.yaml +22 -0
  14. conlanggen-0.4.2/spec/example.yaml +18 -0
  15. conlanggen-0.4.2/src/conlang/__init__.py +29 -0
  16. conlanggen-0.4.2/src/conlang/bench/__init__.py +5 -0
  17. conlanggen-0.4.2/src/conlang/bench/kah.py +138 -0
  18. conlanggen-0.4.2/src/conlang/cli.py +430 -0
  19. conlanggen-0.4.2/src/conlang/data/the_quiet_morning_sentences.csv +112 -0
  20. conlanggen-0.4.2/src/conlang/lexicon/__init__.py +5 -0
  21. conlanggen-0.4.2/src/conlang/lexicon/generate.py +323 -0
  22. conlanggen-0.4.2/src/conlang/lint/__init__.py +181 -0
  23. conlanggen-0.4.2/src/conlang/model/__init__.py +168 -0
  24. conlanggen-0.4.2/src/conlang/morphology/__init__.py +4 -0
  25. conlanggen-0.4.2/src/conlang/morphology/generate.py +402 -0
  26. conlanggen-0.4.2/src/conlang/phonology/__init__.py +5 -0
  27. conlanggen-0.4.2/src/conlang/phonology/romanize.py +170 -0
  28. conlanggen-0.4.2/src/conlang/phonology/syllable.py +159 -0
  29. conlanggen-0.4.2/src/conlang/pipeline/__init__.py +214 -0
  30. conlanggen-0.4.2/src/conlang/pipeline/compose.py +264 -0
  31. conlanggen-0.4.2/src/conlang/pipeline/extras.py +155 -0
  32. conlanggen-0.4.2/src/conlang/pipeline/formats.py +58 -0
  33. conlanggen-0.4.2/src/conlang/pipeline/grow.py +88 -0
  34. conlanggen-0.4.2/src/conlang/pipeline/port.py +77 -0
  35. conlanggen-0.4.2/src/conlang/pipeline/scaffold.py +174 -0
  36. conlanggen-0.4.2/src/conlang/pipeline/validate.py +119 -0
  37. conlanggen-0.4.2/src/conlang/render/__init__.py +5 -0
  38. conlanggen-0.4.2/src/conlang/render/markdown.py +176 -0
  39. conlanggen-0.4.2/src/conlang/sampling/__init__.py +1 -0
  40. conlanggen-0.4.2/src/conlang/sampling/concepts.py +190 -0
  41. conlanggen-0.4.2/src/conlang/sampling/defaults.py +145 -0
  42. conlanggen-0.4.2/src/conlang/sampling/rng.py +76 -0
  43. conlanggen-0.4.2/src/conlang/sampling/segments.py +100 -0
  44. conlanggen-0.4.2/src/conlang/soundchange/__init__.py +208 -0
  45. conlanggen-0.4.2/src/conlang/spec.py +145 -0
  46. conlanggen-0.4.2/src/conlang/syntax/__init__.py +302 -0
  47. conlanggen-0.4.2/tests/golden.json +5 -0
  48. conlanggen-0.4.2/tests/test_adpositions.py +91 -0
  49. conlanggen-0.4.2/tests/test_aspect.py +50 -0
  50. conlanggen-0.4.2/tests/test_bench.py +93 -0
  51. conlanggen-0.4.2/tests/test_clause_sequencing.py +74 -0
  52. conlanggen-0.4.2/tests/test_cli.py +62 -0
  53. conlanggen-0.4.2/tests/test_golden.py +82 -0
  54. conlanggen-0.4.2/tests/test_intransitive_clauses.py +65 -0
  55. conlanggen-0.4.2/tests/test_invariants.py +174 -0
  56. conlanggen-0.4.2/tests/test_ipa.py +81 -0
  57. conlanggen-0.4.2/tests/test_lexicon.py +27 -0
  58. conlanggen-0.4.2/tests/test_lint.py +27 -0
  59. conlanggen-0.4.2/tests/test_lint_invariants.py +87 -0
  60. conlanggen-0.4.2/tests/test_morphology.py +281 -0
  61. conlanggen-0.4.2/tests/test_noun_phrases.py +92 -0
  62. conlanggen-0.4.2/tests/test_phonotactics.py +92 -0
  63. conlanggen-0.4.2/tests/test_pipeline_commands.py +179 -0
  64. conlanggen-0.4.2/tests/test_predication.py +83 -0
  65. conlanggen-0.4.2/tests/test_repo_hygiene.py +43 -0
  66. conlanggen-0.4.2/tests/test_reproducibility.py +21 -0
  67. conlanggen-0.4.2/tests/test_rng.py +16 -0
  68. conlanggen-0.4.2/tests/test_soundchange.py +161 -0
  69. conlanggen-0.4.2/tests/test_syntax.py +102 -0
  70. conlanggen-0.4.2/wayfinder/map.md +55 -0
  71. conlanggen-0.4.2/wayfinder/research/capture-kit-contracts.md +402 -0
  72. conlanggen-0.4.2/wayfinder/tickets/build-pipeline-commands.md +44 -0
  73. conlanggen-0.4.2/wayfinder/tickets/capture-kit-contracts.md +40 -0
  74. conlanggen-0.4.2/wayfinder/tickets/design-proof-language.md +30 -0
  75. conlanggen-0.4.2/wayfinder/tickets/grammar-match-checklist.md +42 -0
  76. conlanggen-0.4.2/wayfinder/tickets/pin-grammar-in-new.md +26 -0
  77. conlanggen-0.4.2/wayfinder/tickets/pipeline-command-surface.md +58 -0
  78. conlanggen-0.4.2/wayfinder/tickets/pipeline-runbook.md +26 -0
  79. conlanggen-0.4.2/wayfinder/tickets/prove-fidelity.md +32 -0
  80. conlanggen-0.4.2/wayfinder/tickets/standardize-workspaces.md +40 -0
  81. conlanggen-0.4.2/wayfinder/tickets/translate-proof-language.md +29 -0
  82. conlanggen-0.4.2/wayfinder/tickets/workspace-layout.md +53 -0
@@ -0,0 +1,6 @@
1
+ # Normalise line endings in the repository so checksums match across platforms.
2
+ * text=auto eol=lf
3
+
4
+ *.png binary
5
+ *.jpg binary
6
+ *.pdf binary
@@ -0,0 +1,45 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ branches: [main, master]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ # Reproducibility is a hard invariant and the bytes on disk are the artifact
11
+ # users get, so the suite runs on more than the author's platform. A passing
12
+ # Linux-only run would not catch a platform-dependent newline or path.
13
+ runs-on: ${{ matrix.os }}
14
+ strategy:
15
+ fail-fast: false
16
+ matrix:
17
+ os: [ubuntu-latest, windows-latest]
18
+ python-version: ["3.11", "3.12", "3.13"]
19
+ steps:
20
+ - uses: actions/checkout@v4
21
+ - uses: actions/setup-python@v5
22
+ with:
23
+ python-version: ${{ matrix.python-version }}
24
+ - name: Install
25
+ run: |
26
+ python -m pip install --upgrade pip
27
+ pip install -e ".[dev]"
28
+ - name: Lint
29
+ run: ruff check .
30
+ - name: Type-check
31
+ run: mypy src
32
+ - name: Test
33
+ run: pytest
34
+ - name: Build (sdist is platform-independent)
35
+ # Built on one platform only: the sdist and wheel must not vary by runner,
36
+ # so building per-matrix would create artifacts that differ for no reason.
37
+ if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.12'
38
+ run: |
39
+ python -m pip install --upgrade build
40
+ python -m build
41
+ - uses: actions/upload-artifact@v4
42
+ if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.12'
43
+ with:
44
+ name: dist
45
+ path: dist/
@@ -0,0 +1,89 @@
1
+ name: release
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ test:
12
+ # A release is gated on the full matrix, not just the version being cut.
13
+ strategy:
14
+ fail-fast: false
15
+ matrix:
16
+ os: [ubuntu-latest, windows-latest]
17
+ python-version: ["3.11", "3.12", "3.13"]
18
+ runs-on: ${{ matrix.os }}
19
+ steps:
20
+ - uses: actions/checkout@v4
21
+ - uses: actions/setup-python@v5
22
+ with:
23
+ python-version: ${{ matrix.python-version }}
24
+ - name: Install
25
+ run: |
26
+ python -m pip install --upgrade pip
27
+ pip install -e ".[dev]"
28
+ - name: Lint
29
+ run: ruff check .
30
+ - name: Type-check
31
+ run: mypy src
32
+ - name: Test
33
+ run: pytest
34
+
35
+ publish:
36
+ needs: test
37
+ runs-on: ubuntu-latest
38
+ environment: pypi
39
+ permissions:
40
+ # Job-level `permissions` *replace* the workflow-level ones (unspecified
41
+ # scopes become `none`), so `contents: read` must be listed here too or
42
+ # actions/checkout in this job cannot read the repo.
43
+ contents: read
44
+ # Required for OIDC trusted publishing: no API token is stored anywhere.
45
+ id-token: write
46
+ steps:
47
+ - uses: actions/checkout@v4
48
+ - uses: actions/setup-python@v5
49
+ with:
50
+ python-version: "3.12"
51
+
52
+ - name: Check tag matches project version
53
+ run: |
54
+ tag="${GITHUB_REF_NAME#v}"
55
+ version="$(python -c "import tomllib,pathlib; print(tomllib.loads(pathlib.Path('pyproject.toml').read_text(encoding='utf-8'))['project']['version'])")"
56
+ if [ "$tag" != "$version" ]; then
57
+ echo "tag $tag does not match project version $version" >&2
58
+ exit 1
59
+ fi
60
+ echo "publishing $version"
61
+
62
+ - name: Build
63
+ run: |
64
+ python -m pip install --upgrade build
65
+ python -m build
66
+
67
+ - uses: actions/upload-artifact@v4
68
+ with:
69
+ name: dist
70
+ path: dist/
71
+
72
+ - name: Show OIDC token claims (debug)
73
+ continue-on-error: true
74
+ run: |
75
+ python - <<'PY'
76
+ import base64, json, os, urllib.request
77
+ url = os.environ["ACTIONS_ID_TOKEN_REQUEST_URL"] + "&audience=pypi"
78
+ req = urllib.request.Request(
79
+ url,
80
+ headers={"Authorization": "bearer " + os.environ["ACTIONS_ID_TOKEN_REQUEST_TOKEN"]},
81
+ )
82
+ token = json.load(urllib.request.urlopen(req))["value"]
83
+ payload = token.split(".")[1]
84
+ payload += "=" * (-len(payload) % 4)
85
+ print(json.dumps(json.loads(base64.urlsafe_b64decode(payload)), indent=2))
86
+ PY
87
+
88
+ - name: Publish to PyPI
89
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,22 @@
1
+ # Byte-compiled / optimized / DLL files
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ .eggs/
6
+ build/
7
+ dist/
8
+
9
+ # Virtual environments
10
+ .venv/
11
+ venv/
12
+
13
+ # Tooling caches
14
+ .pytest_cache/
15
+ .ruff_cache/
16
+ .mypy_cache/
17
+ coverage.xml
18
+ .coverage
19
+
20
+ # Generated output
21
+ out/
22
+ *.local.yaml
@@ -0,0 +1,12 @@
1
+ repos:
2
+ - repo: https://github.com/astral-sh/ruff-pre-commit
3
+ rev: v0.6.9
4
+ hooks:
5
+ - id: ruff
6
+ args: [--fix]
7
+ - id: ruff-format
8
+ - repo: https://github.com/pre-commit/mirrors-mypy
9
+ rev: v1.11.2
10
+ hooks:
11
+ - id: mypy
12
+ additional_dependencies: ["pydantic>=2.7", "pyyaml>=6", "types-PyYAML>=6"]
@@ -0,0 +1,67 @@
1
+ # CONTEXT — domain model and vocabulary
2
+
3
+ Shared language for this project. Update it when terminology shifts; the terms
4
+ here are load-bearing in code, specs, and issues.
5
+
6
+ ## Core objects
7
+
8
+ | Term | Definition |
9
+ | --- | --- |
10
+ | **Spec** | What the user writes (`spec/*.yaml`). Partial by design; omitted fields are sampled. |
11
+ | **Model** | The canonical JSON (`Language`) produced by a run. Single source of truth; every view is rendered from it. |
12
+ | **Lexeme** | A dictionary entry: id, gloss, part of speech, and a **root**. |
13
+ | **Root** | The bare, citation form of a lexeme. |
14
+ | **Affix** | A generated prefix/suffix with a gloss (e.g. `PL`) and the parts of speech it applies to. |
15
+ | **Allomorph** | A conditioned surface form of an affix, used when the primary form would be illegal in context. |
16
+ | **Adposition** | A generated, invariable word (pos `adp`) for a place/direction relation (`in`, `above`, `toward`…). Its order is fixed by word order: pre- for VO, post- for OV. |
17
+ | **Oblique** | An optional clause adjunct: an adposition plus a bare noun, appended after the object. |
18
+ | **Noun phrase** | A head noun plus optional modifiers (adjectives, possessor, apposition, an adpositional phrase), built on demand. Modifiers sit on the head's side, fixed by head direction. |
19
+ | **Clause** | A predicate with its arguments, linearised on demand: **transitive** (subject + object) or **intransitive** (one argument, which takes the alignment's intransitive case: `NOM` if accusative, `ABS` if ergative). |
20
+ | **Copula** | A generated `be` word (pos `cop`) that predicates a complement. It draws verb morphology (past/future, agreement) by sharing the verb classes, but is not a content verb. |
21
+ | **Predication** | A copular clause: subject + copula + a bare predicative complement (adjective or noun). Existential *there is* is paraphrased as this shape plus a locative oblique. |
22
+ | **Aspect** | A verb category stacked with tense: **progressive** and **perfect** affixes. Pluperfect is `past + perfect`; "had been …-ing" is `past + perfect + progressive`. |
23
+ | **Conjunction** | A generated, invariable connective (`and`, `but`, `then`, `so`; pos `conj`) that joins clauses. |
24
+ | **Sentence** | Two or more clauses joined by **parataxis** (juxtaposition) or a conjunction; relatives and reported speech ride on parataxis. Built on demand, not stored. |
25
+ | **WordRecord** | A surface form plus its morphemes, syllables, and tags (`CIT`, `PL`). |
26
+ | **Origin** | Where a value came from: `specified`, `sampled`, or `derived`. |
27
+ | **Lock** | An entry the user pinned. Locked entries survive re-sampling. |
28
+ | **Seed** | Integer driving all sampling. Same spec + seed ⇒ identical bytes. |
29
+
30
+ ## Phonology
31
+
32
+ | Term | Definition |
33
+ | --- | --- |
34
+ | **Inventory** | The consonants and vowels a language uses. Symbols are user-facing; v1 has no feature layer beneath them. A spec may declare the inventory phonemic (`ipa: true`), in which case each symbol is an IPA phoneme. |
35
+ | **Orthography** | The romanised spelling derived from an IPA inventory (symbol → spelling). A *view* of the phonemic form, stored on the inventory and on each lexeme/word; absence means the symbol spells itself. |
36
+ | **Syllable template** | A shape over `C` and `V` slots, e.g. `CVC`, with a sampling weight. |
37
+ | **Phonotactics** | The legality rules: a word is legal iff its `C`/`V` skeleton can be partitioned into the templates with allowed onsets/codas. Constraints bind *during* the search, so the verdict never depends on which partition is found first. |
38
+ | **Segment** | One inventory symbol, possibly multi-character (`sh`, `ch`). |
39
+
40
+ ## Process
41
+
42
+ | Term | Definition |
43
+ | --- | --- |
44
+ | **Generate** | `spec → model`. Samples the gaps, assigns stable per-item seeds, assembles word forms. |
45
+ | **Stable per-item seeding** | Each item hashes its own label into its RNG, so generating one extra word never reshuffles the others. |
46
+ | **Lint** | Machine-checkable invariants run after generation. Errors mark a model `invalid`; warnings do not. |
47
+ | **Render** | Model → Markdown dictionary + grammar sketch. Never hand-maintained. |
48
+ | **Bench** | Adapter + plausibility scorecard against the real Kah dictionary. Reported, never a pass/fail gate. |
49
+ | **Lineage** | Proto-language/rule hooks carried in the schema for phase-2 diachrony. |
50
+
51
+ ## Translation pipeline
52
+
53
+ | Term | Definition |
54
+ | --- | --- |
55
+ | **Pipeline command** | One of `conlang new` / `grow` / `translate` / `validate-translation` / `extras` — the repeatable way to build a generated language and translate the fixed story into it. |
56
+ | **Workspace** | The standard home of a generated language: `MISSION`/`NOTES`/`README`/`RESOURCES`, `spec/`, `out/`, `reference/`, `learning-records/`, and the bundled story. One per language. |
57
+ | **Composer** | The library object (`conlang.Composer`) that realises a token (`gloss.TAG…`) as a surface word the model licenses, and reads a surface back to its gloss. |
58
+ | **Gloss script** | A hand-authored TSV (`index<TAB>gloss`) that `translate` composes into a translation. |
59
+ | **Port** | Re-composing a translation from a sibling language's gloss column instead of writing a gloss script. |
60
+ | **Portability** | The grammar-match check that decides port vs hand-translate: alignment + word order (adposition/modifier order asserted) and the source's tag vocabulary present in the target. |
61
+
62
+ ## Non-negotiable invariants
63
+
64
+ 1. **Reproducibility is hard.** Same spec + seed produces byte-identical output.
65
+ 2. **The model is the source of truth.** Views are derived; never edit a view.
66
+ 3. **Lint never crashes a run.** Broken languages are generated and reported.
67
+ 4. **Kah data stays out of the repo.** The benchmark reads a local cache at runtime.
@@ -0,0 +1,30 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Brian McKeen
4
+
5
+ Contact: bmckeen1919@gmail.com
6
+
7
+ Permission is hereby granted, free of charge, to any person obtaining a copy
8
+ of this software and associated documentation files (the "Software"), to deal
9
+ in the Software without restriction, including without limitation the rights
10
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11
+ copies of the Software, and to permit persons to whom the Software is
12
+ furnished to do so, subject to the following conditions:
13
+
14
+ The above copyright notice and this permission notice shall be included in all
15
+ copies or substantial portions of the Software.
16
+
17
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
20
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
23
+ SOFTWARE.
24
+
25
+ ---
26
+
27
+ Note on third-party data: this project does not include the Kwesho Kah
28
+ dictionary or grammar, which are the copyrighted work of Kwesho. The Kah
29
+ benchmark reads them from a local cache at runtime and they are not distributed
30
+ here. See README.md.
@@ -0,0 +1,377 @@
1
+ Metadata-Version: 2.5
2
+ Name: conlanggen
3
+ Version: 0.4.2
4
+ Summary: A partial-spec conlang generator: specify what you care about, sample the rest.
5
+ Project-URL: Homepage, https://github.com/bmckeen1919-lab/conlanggen
6
+ Project-URL: Repository, https://github.com/bmckeen1919-lab/conlanggen
7
+ Project-URL: Issues, https://github.com/bmckeen1919-lab/conlanggen/issues
8
+ Author: Brian McKeen
9
+ License: MIT License
10
+
11
+ Copyright (c) 2026 Brian McKeen
12
+
13
+ Contact: bmckeen1919@gmail.com
14
+
15
+ Permission is hereby granted, free of charge, to any person obtaining a copy
16
+ of this software and associated documentation files (the "Software"), to deal
17
+ in the Software without restriction, including without limitation the rights
18
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
19
+ copies of the Software, and to permit persons to whom the Software is
20
+ furnished to do so, subject to the following conditions:
21
+
22
+ The above copyright notice and this permission notice shall be included in all
23
+ copies or substantial portions of the Software.
24
+
25
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
26
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
27
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
28
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
29
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
30
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
31
+ SOFTWARE.
32
+
33
+ ---
34
+
35
+ Note on third-party data: this project does not include the Kwesho Kah
36
+ dictionary or grammar, which are the copyrighted work of Kwesho. The Kah
37
+ benchmark reads them from a local cache at runtime and they are not distributed
38
+ here. See README.md.
39
+ License-File: LICENSE
40
+ Keywords: conlang,linguistics,procedural-generation,worldbuilding
41
+ Requires-Python: >=3.11
42
+ Requires-Dist: pydantic>=2.7
43
+ Requires-Dist: pyyaml>=6
44
+ Provides-Extra: dev
45
+ Requires-Dist: mypy>=1.11; extra == 'dev'
46
+ Requires-Dist: pre-commit>=3.7; extra == 'dev'
47
+ Requires-Dist: pytest>=8; extra == 'dev'
48
+ Requires-Dist: ruff>=0.6; extra == 'dev'
49
+ Requires-Dist: types-pyyaml>=6; extra == 'dev'
50
+ Provides-Extra: phoible
51
+ Provides-Extra: wals
52
+ Description-Content-Type: text/markdown
53
+
54
+ # conlang
55
+
56
+ A partial-spec constructed-language generator.
57
+
58
+ You describe only what you care about — phonology, syllable shapes, a handful of
59
+ words — and the generator **samples the rest** from typologically-plausible
60
+ defaults. Same spec plus same seed always produces byte-identical output, and
61
+ sampled entries can be locked so they survive re-sampling.
62
+
63
+ ```text
64
+ spec (YAML) ──▶ generate ──▶ model (canonical JSON) ──▶ render ──▶ dictionary + grammar sketch
65
+ │
66
+ ├── lint (machine-checkable invariants)
67
+ └── bench (plausibility scorecard vs. Kah)
68
+ ```
69
+
70
+ The model JSON is the single source of truth; the dictionary, grammar sketch,
71
+ and any future views are rendered from it.
72
+
73
+ ## Install
74
+
75
+ ```powershell
76
+ pip install conlanggen
77
+ ```
78
+
79
+ The distribution is `conlanggen`, but the import package and the command are both
80
+ `conlang` — `import conlang`, `conlang generate`.
81
+
82
+ ## Quickstart
83
+
84
+ ```powershell
85
+ python -m venv .venv
86
+ .\.venv\Scripts\python.exe -m pip install -e ".[dev]"
87
+
88
+ # Generate from the example spec, and render a Markdown dictionary.
89
+ .\.venv\Scripts\python.exe -m conlang.cli generate --spec spec\example.yaml --render
90
+
91
+ # Happy with it? Freeze it into an explicit spec so the values stick.
92
+ conlang accept --model out\veshtari.json
93
+ ```
94
+
95
+ Output lands in `out/`: a model JSON and a rendered `.md`.
96
+
97
+ ```powershell
98
+ # Re-lint or re-render an existing model.
99
+ conlang lint --model out\veshtari.json
100
+ conlang render --model out\veshtari.json --out out\veshtari.md
101
+ ```
102
+
103
+ ## Translating a fixed story
104
+
105
+ The toolkit ships a repeatable pipeline for translating the fixed 111-sentence
106
+ story *The Quiet Morning* into a **generated** language: `conlang new` → `grow` →
107
+ `translate` → `validate-translation` → `extras`. The runbook is
108
+ [`docs/pipeline.md`](docs/pipeline.md).
109
+
110
+ ## Writing a spec
111
+
112
+ Everything is optional. Omitted fields are sampled.
113
+
114
+ ```yaml
115
+ name: Veshtari
116
+ seed: 42
117
+
118
+ phonology:
119
+ consonants: [p, t, k, m, n, s, l, r, v, sh, ch]
120
+ vowels: [a, e, i, o, u]
121
+
122
+ syllables: [CV, CVC, V] # C = consonant slot, V = vowel slot
123
+ word_count: 150
124
+
125
+ lexicon: # pin specific words; they are locked by default
126
+ - gloss: water
127
+ root: vela
128
+ ```
129
+
130
+ Accepted keys: `name`, `seed`, `phonology.{consonants,vowels,ipa,orthography}`,
131
+ `syllables`, `concepts`, `word_count`, `lexicon`, `affixes`, `onsets`, `codas`,
132
+ `forbidden`.
133
+
134
+ An explicit, unsized `lexicon` is treated as the complete word list (that is
135
+ what `accept` writes back); otherwise the built-in concept list is used and
136
+ `word_count` trims or extends it.
137
+
138
+ If your `syllables` include `CCV`/`CCVC`, the generator samples a legal onset
139
+ set unless you supply `onsets` yourself, and `accept` records whichever was
140
+ used so a frozen spec regenerates identically.
141
+
142
+ ### IPA inventories
143
+
144
+ By default a symbol *is* its spelling (`sh` is written `sh`). Set
145
+ `phonology.ipa: true` to declare the inventory phonemic: each symbol is an IPA
146
+ phoneme, and the tool derives a romanised **orthography** for the dictionary.
147
+
148
+ ```yaml
149
+ name: Shibo
150
+ seed: 11
151
+
152
+ phonology:
153
+ ipa: true
154
+ consonants: [p, t, k, ʃ, tʃ, m, n, s, l]
155
+ vowels: [a, e, i, o, u]
156
+ orthography: # optional; corrects or extends the built-in defaults
157
+ tʃ: ch
158
+ ʃ: sh
159
+ ```
160
+
161
+ `/ʃ/` spells `sh` and `/tʃ/` spells `ch` from a built-in table; anything absent
162
+ spells itself, so a plain Latin inventory is unchanged. `orthography` overrides
163
+ the table (set an entry back to its own symbol to cancel a default), and the
164
+ resolved map is stored on the inventory. The dictionary and word-form tables
165
+ then show a `spelling` column beside the `ipa` one, and `accept` writes the map
166
+ back so a frozen spec regenerates identically. Lint warns on a map entry that is
167
+ not an inventory symbol, or two symbols that would spell the same thing.
168
+
169
+ ## Freezing a language (`accept`)
170
+
171
+ A partial spec samples whatever you leave out. Once you like the result,
172
+ `conlang accept` writes it back as a **fully-explicit spec**: the phonology,
173
+ syllable shapes, constraints, affixes, and every root become pinned entries.
174
+ Regenerating from that spec reproduces the same language exactly — sampled
175
+ values now survive re-sampling, and changing the seed can no longer disturb
176
+ them.
177
+
178
+ ```powershell
179
+ conlang generate --spec spec\example.yaml --out out\veshtari.json
180
+ conlang accept --model out\veshtari.json --out spec/veshtari.yaml
181
+ conlang generate --spec spec/veshtari.yaml --out out/veshtari.json
182
+ ```
183
+
184
+ Two fields legitimately change across the round trip: `origin` becomes
185
+ `specified` for every accepted entry (it is pinned now, not sampled), and
186
+ `spec_hash` changes because the spec text is different. Everything linguistic
187
+ is identical, and `tests/test_invariants.py` asserts exactly that.
188
+
189
+ ## Guarantees and lint rules
190
+
191
+ Reproducibility is the one **hard** guarantee: `spec + seed ⇒ identical bytes`.
192
+ Everything else is reported by `lint` as errors or warnings, and a broken
193
+ language is still generated (conlangers want to explore broken states).
194
+
195
+ | rule | severity | meaning |
196
+ | --- | --- | --- |
197
+ | `schema` | error/warning | schema version present and known |
198
+ | `lexeme.root` | error | lexeme has a non-empty root |
199
+ | `phonotactics.lexeme` | error | every root is a legal word |
200
+ | `phonotactics.word` | error | every inflected form is a legal word |
201
+ | `lexicon.homophones` | warning | two roots collide (Kah has polysemy too) |
202
+ | `morphology.paradigm` | warning | a word is missing a form for a category its class realises |
203
+ | `morphology.class` | error | a word names an inflection class the language does not define |
204
+ | `lexicon.nonempty` | warning | the lexicon is empty |
205
+ | `orthography.unknown` | warning | the spelling map names a symbol that is not in the inventory |
206
+ | `orthography.collision` | warning | two symbols would spell the same thing |
207
+
208
+ Every rule above has a test in `tests/test_lint.py`, `tests/test_lint_invariants.py`,
209
+ or `tests/test_morphology.py`.
210
+
211
+ ## Phonotactics
212
+
213
+ A word is legal when its `C`/`V` skeleton can be partitioned into the language's
214
+ syllable templates **and** every resulting onset and coda is allowed. Two details
215
+ matter more than they look:
216
+
217
+ - **Constraints are checked while syllabifying, not after.** A word often has
218
+ several valid partitions; `deplango` is `de|plan|go` with legal onsets, and
219
+ also `de|pla|ngo` with an illegal one. Verdict must depend on the phonology,
220
+ not on which partition the search happens to return first.
221
+ - **Onsets and codas are compared by segment, not by string.** An onset of
222
+ `ny` + `m` renders as `"nym"` — three characters, two segments. Matching
223
+ strings would silently reject legal words in any language with a digraph.
224
+
225
+ When a template set admits onset clusters (`CCV`, `CCVC`), the generator samples
226
+ a legal onset set and **builds roots from it**, so generated words are legal by
227
+ construction rather than by luck. `constraints.onsets` in the model records what
228
+ was sampled; an explicit `onsets` in the spec always wins.
229
+
230
+ Affixes carry **allomorphs**: conditioned surface forms. A single form cannot
231
+ serve both consonant-final and vowel-final roots, because the root's coda plus a
232
+ consonant-initial affix resyllabify into an onset cluster. When the primary form
233
+ would be illegal, `build_words` uses the first alternative that is legal — which
234
+ is what real languages do. It is a rare fallback (~2 in 8,500 affixed forms), but
235
+ without it those words would be illegal and lint would fail the language.
236
+
237
+ ## The Kah benchmark
238
+
239
+ Kah (Kwesho's language) is hand-made, so the benchmark does **not** try to
240
+ regenerate it. It does two things:
241
+
242
+ 1. **Adapter fixture** — parses the real Kwesho dictionary cache into the model
243
+ schema, proving the schema can represent real data.
244
+ 2. **Plausibility scorecard** — compares a generated language against Kah.
245
+
246
+ The scorecard deliberately reports **three separate numbers**, because a word can
247
+ fail for two unrelated reasons and collapsing them makes the report unreadable:
248
+
249
+ | metric | question |
250
+ | --- | --- |
251
+ | symbol coverage | which Kah sounds does this language have? |
252
+ | usable sounds | how many Kah headwords use only sounds this language has? |
253
+ | legal shapes | of those, how many fit the phonotactics? |
254
+
255
+ Scoring the built-in `kah` preset against Kah is circular, so the report says so
256
+ explicitly when it detects it. Treat the scorecard as a lever, not a target: the
257
+ presets are deliberately small, and a language that only allows `CV` syllables is
258
+ a legitimate (if monotonous) language type, not a bug to tune away.
259
+
260
+ ```powershell
261
+ conlang bench --spec spec\example.yaml
262
+ # or point at a cache explicitly:
263
+ conlang bench --spec spec\example.yaml --kah C:\path\to\kah-lexicon.txt
264
+ ```
265
+
266
+ Kah's dictionary is Kwesho's copyrighted work and is **never** vendored into this
267
+ repository. Without `--kah`, the benchmark looks for the glossary at a per-user
268
+ cache path (`%LOCALAPPDATA%\conlang\kah-lexicon.txt` on Windows,
269
+ `~/.cache/conlang/kah-lexicon.txt` elsewhere, or `$XDG_CACHE_HOME/conlang/`);
270
+ override it with `--kah` or `CONLANG_KAH_LEXICON`.
271
+
272
+ On the reference machine, the adapter parses **8,153** entries with a **97.8%**
273
+ validity rate.
274
+
275
+ ## Evolving a language
276
+
277
+ A generated model can be the parent of a daughter, evolved by ordered sound
278
+ changes. A rule is `old > new` (`new` empty deletes the segment), optionally
279
+ conditioned by an environment after `/`. The environment names the context on
280
+ each side of the target, with `_` marking the target slot, `#` the word boundary,
281
+ and `V`/`C` the vowel/consonant classes:
282
+
283
+ ```powershell
284
+ conlang generate --spec spec\example.yaml --out out\proto.json
285
+ conlang evolve --model out\proto.json --rule "a > e" --rule "p > f / V_V" --out out\daughter.json
286
+ conlang evolve --model out\proto.json --rule "t > d / #_" --rule "s > z / V_" --out out\voiced.json
287
+ ```
288
+
289
+ `V_V` is "between vowels", `#_` is word-initial, `_#` is word-final. Rules are
290
+ applied left to right so a later rule sees an earlier one's output (feeding).
291
+
292
+ The daughter is a full model: its inventory gains any new sounds, its lexicon
293
+ and affixes are rewritten, and its word forms are rebuilt. It records where it
294
+ came from in `lineage`:
295
+
296
+ ```json
297
+ "lineage": { "parent": "proto", "rules": ["a > e", "p > f / V_V"] }
298
+ ```
299
+
300
+ That is the whole point of the field: the derivation is data, so a daughter can
301
+ itself be evolved, and `accept` keeps the chain across a round trip. Mergers
302
+ introduced by a change show up as `lexicon.homophones` warnings, never errors.
303
+
304
+ ## Syntax
305
+
306
+ Each language samples a constituent order (`SOV`, `SVO`, or `VSO`, weighted
307
+ toward the first two) and an alignment (`accusative` or `ergative`). Both are
308
+ recorded in the model and survive `accept`. A transitive clause is linearised
309
+ from the lexicon and inflected: the subject and object take case by alignment
310
+ (accusative marks `NOM`/`ACC`; ergative marks `ERG`/`ABS`) and the verb agrees
311
+ with the subject:
312
+
313
+ ```python
314
+ from conlang.syntax import sample_clause
315
+ words, gloss = sample_clause(language)
316
+ # e.g. (['pyople', 'klavragprekbra', 'blonyto'], 'water.ERG say.3SG fire.ABS')
317
+ ```
318
+
319
+ Case and agreement are applied when the clause is built, not woven into every
320
+ word's standing paradigm — a deliberate split, so a noun's form count does not
321
+ double again for case.
322
+
323
+ ## Layout
324
+
325
+ ```text
326
+ src/conlang/
327
+ model/ canonical pydantic schema (the source of truth)
328
+ sampling/ deterministic RNG, inventory presets, concepts, defaults
329
+ phonology/ tokenization, syllabification, legality checks
330
+ lexicon/ seeded root generation, stable per-item seeding
331
+ morphology/ affixes, inflection classes, and word-form assembly
332
+ soundchange/ ordered sound changes; derive a daughter language
333
+ syntax/ word order, alignment, simple clause linearisation
334
+ render/ Markdown dictionary + grammar sketch
335
+ lint/ machine-checkable invariants
336
+ bench/ Kah adapter + scorecard
337
+ cli.py thin command-line view over the library
338
+ ```
339
+
340
+ ## Development
341
+
342
+ ```powershell
343
+ .\.venv\Scripts\python.exe -m pytest
344
+ .\.venv\Scripts\python.exe -m ruff check .
345
+ .\.venv\Scripts\python.exe -m mypy src
346
+ ```
347
+
348
+ ## Roadmap
349
+
350
+ Shipped so far:
351
+
352
+ - **Phonology** — inventory, phonotactics, seeded word generation.
353
+ - **IPA inventories** — declare the inventory phonemic (`ipa: true`) and the tool
354
+ derives a romanised orthography alongside the phonemic forms.
355
+ - **Morphology** — affix breadth, inflection classes, and stacking (a word can
356
+ carry several categories at once).
357
+ - **Diachrony** — `conlang evolve`: ordered sound changes derive a daughter
358
+ language, now with **environments** (`p > f / V_V`, `#`, `_`), and the chain
359
+ recorded in `lineage`.
360
+ - **Syntax** — sampled word order and alignment, case marking with subject
361
+ agreement, the **predication tier** (adpositions, noun phrases, intransitives,
362
+ copula/existential), **clause sequencing** (parataxis and conjunctions), and
363
+ **aspect** (progressive/perfect, stacking with tense).
364
+ - **Translation pipeline** — `conlang new` / `grow` / `translate` /
365
+ `validate-translation` / `extras`: build a generated language and translate the
366
+ fixed story into it, including a grammar-match **port** from a sibling language.
367
+
368
+ Next:
369
+
370
+ - Optional data packs: PHOIBLE (inventories) and WALS (typology) as CC-BY
371
+ adapters behind the built-in defaults.
372
+ - Semantics/pragmatics and writing systems (deferred from v1).
373
+
374
+ ## License
375
+
376
+ MIT (code). Kah dictionary and grammar content remains Kwesho's and is not part
377
+ of this repository.