conlanggen 0.4.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- conlang/__init__.py +29 -0
- conlang/bench/__init__.py +5 -0
- conlang/bench/kah.py +138 -0
- conlang/cli.py +430 -0
- conlang/data/the_quiet_morning_sentences.csv +112 -0
- conlang/lexicon/__init__.py +5 -0
- conlang/lexicon/generate.py +323 -0
- conlang/lint/__init__.py +181 -0
- conlang/model/__init__.py +168 -0
- conlang/morphology/__init__.py +4 -0
- conlang/morphology/generate.py +402 -0
- conlang/phonology/__init__.py +5 -0
- conlang/phonology/romanize.py +170 -0
- conlang/phonology/syllable.py +159 -0
- conlang/pipeline/__init__.py +214 -0
- conlang/pipeline/compose.py +264 -0
- conlang/pipeline/extras.py +155 -0
- conlang/pipeline/formats.py +58 -0
- conlang/pipeline/grow.py +88 -0
- conlang/pipeline/port.py +77 -0
- conlang/pipeline/scaffold.py +174 -0
- conlang/pipeline/validate.py +119 -0
- conlang/render/__init__.py +5 -0
- conlang/render/markdown.py +176 -0
- conlang/sampling/__init__.py +1 -0
- conlang/sampling/concepts.py +190 -0
- conlang/sampling/defaults.py +145 -0
- conlang/sampling/rng.py +76 -0
- conlang/sampling/segments.py +100 -0
- conlang/soundchange/__init__.py +208 -0
- conlang/spec.py +145 -0
- conlang/syntax/__init__.py +302 -0
- conlanggen-0.4.2.dist-info/METADATA +377 -0
- conlanggen-0.4.2.dist-info/RECORD +37 -0
- conlanggen-0.4.2.dist-info/WHEEL +4 -0
- conlanggen-0.4.2.dist-info/entry_points.txt +2 -0
- conlanggen-0.4.2.dist-info/licenses/LICENSE +30 -0
conlang/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""conlang: a partial-spec constructed-language generator."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError
|
|
4
|
+
from importlib.metadata import version as _dist_version
|
|
5
|
+
|
|
6
|
+
from .lint import LintReport, lint
|
|
7
|
+
from .model import Language
|
|
8
|
+
from .pipeline import generate
|
|
9
|
+
from .pipeline.compose import Composer
|
|
10
|
+
from .render.markdown import render_markdown
|
|
11
|
+
from .spec import Spec, dump_spec, load_spec, to_spec
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
__version__ = _dist_version("conlanggen")
|
|
15
|
+
except PackageNotFoundError: # running from a source tree without an install
|
|
16
|
+
__version__ = "0.0.0+unknown"
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"Composer",
|
|
20
|
+
"Language",
|
|
21
|
+
"LintReport",
|
|
22
|
+
"Spec",
|
|
23
|
+
"dump_spec",
|
|
24
|
+
"generate",
|
|
25
|
+
"lint",
|
|
26
|
+
"load_spec",
|
|
27
|
+
"render_markdown",
|
|
28
|
+
"to_spec",
|
|
29
|
+
]
|
conlang/bench/kah.py
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Kah benchmark adapter.
|
|
2
|
+
|
|
3
|
+
Kah is hand-made, so this does *not* try to regenerate Kah. It does two things:
|
|
4
|
+
|
|
5
|
+
1. Parse the real Kwesho dictionary cache into ``KahEntry`` models -- an
|
|
6
|
+
integration check that the schema can represent real data. Kah data is
|
|
7
|
+
copyrighted and is never vendored into this repo; point ``--kah`` at a local
|
|
8
|
+
cache (see the ``kah-translation`` skill).
|
|
9
|
+
2. Produce a plausibility scorecard comparing a generated language against Kah's
|
|
10
|
+
statistics. The scorecard is reported, never used as a pass/fail gate.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, Field
|
|
19
|
+
|
|
20
|
+
from ..model import Language
|
|
21
|
+
from ..phonology.syllable import is_legal_word, skeleton, tokenize
|
|
22
|
+
from ..sampling.segments import PRESETS
|
|
23
|
+
|
|
24
|
+
POS = r"(?:a|s|n|i|c|v|adj|adv|prep|pron|num|conj|intj|interj|art|part|aux)\."
|
|
25
|
+
HEADWORD_RE = re.compile(rf"^(?P<hw>\S+)\s{{1,}}(?P<pos>{POS})\s*(?P<gloss>.*)$")
|
|
26
|
+
ARTIFACT_RE = re.compile(r"^(Kah Dictionary|\d+|[A-Z])$")
|
|
27
|
+
|
|
28
|
+
# The inventory the dictionary is tokenized with is the single curated Kah
|
|
29
|
+
# inventory, imported from the preset rather than restated, so the preset and
|
|
30
|
+
# the adapter cannot drift apart.
|
|
31
|
+
KAH_INVENTORY = PRESETS["kah"]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class KahEntry(BaseModel):
|
|
35
|
+
headword: str
|
|
36
|
+
pos: str
|
|
37
|
+
gloss: str
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class BenchReport(BaseModel):
|
|
41
|
+
source: str
|
|
42
|
+
entries_parsed: int
|
|
43
|
+
valid_models: int
|
|
44
|
+
validity_rate: float
|
|
45
|
+
symbol_coverage: float
|
|
46
|
+
tokenizable_rate: float
|
|
47
|
+
phonotactic_legality: float
|
|
48
|
+
kah_syllable_shapes: list[str] = Field(default_factory=list)
|
|
49
|
+
lang_syllable_shapes: list[str] = Field(default_factory=list)
|
|
50
|
+
notes: list[str] = Field(default_factory=list)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def section_bounds(lines: list[str]) -> tuple[int, int]:
|
|
54
|
+
marks = [i for i, line in enumerate(lines) if line.strip() == "A"]
|
|
55
|
+
if len(marks) >= 2:
|
|
56
|
+
return marks[0], marks[1]
|
|
57
|
+
return 0, len(lines)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def parse_kah_lexicon(path: str | Path) -> list[KahEntry]:
|
|
61
|
+
lines = Path(path).read_text(encoding="utf-8").splitlines()
|
|
62
|
+
lo, hi = section_bounds(lines)
|
|
63
|
+
entries: list[KahEntry] = []
|
|
64
|
+
for i in range(lo, hi):
|
|
65
|
+
stripped = lines[i].strip()
|
|
66
|
+
if ARTIFACT_RE.match(stripped):
|
|
67
|
+
continue
|
|
68
|
+
match = HEADWORD_RE.match(stripped)
|
|
69
|
+
if match:
|
|
70
|
+
entries.append(
|
|
71
|
+
KahEntry(
|
|
72
|
+
headword=match.group("hw"),
|
|
73
|
+
pos=match.group("pos").rstrip("."),
|
|
74
|
+
gloss=match.group("gloss").strip(),
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
return entries
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def kah_syllable_shapes(entries: list[KahEntry], limit: int = 64) -> list[str]:
|
|
81
|
+
shapes: set[str] = set()
|
|
82
|
+
for entry in entries:
|
|
83
|
+
tokens = tokenize(entry.headword, KAH_INVENTORY)
|
|
84
|
+
if tokens is None:
|
|
85
|
+
continue
|
|
86
|
+
shapes.add(skeleton(tokens, KAH_INVENTORY))
|
|
87
|
+
return sorted(shapes)[:limit]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def bench_kah(language: Language, path: str | Path) -> BenchReport:
|
|
91
|
+
entries = parse_kah_lexicon(path)
|
|
92
|
+
valid = [e for e in entries if e.headword and e.gloss and e.pos and e.headword.isalpha()]
|
|
93
|
+
|
|
94
|
+
kah_symbols = set(KAH_INVENTORY.symbols())
|
|
95
|
+
lang_symbols = set(language.inventory.symbols())
|
|
96
|
+
coverage = len(kah_symbols & lang_symbols) / len(kah_symbols) if kah_symbols else 0.0
|
|
97
|
+
|
|
98
|
+
# Separate the two ways a Kah headword can fail. A headword this language
|
|
99
|
+
# cannot even tokenize is missing a *sound*; one that tokenizes but will not
|
|
100
|
+
# syllabify has a forbidden *shape*. Collapsing the two makes the scorecard
|
|
101
|
+
# unreadable -- a phonotactics improvement looks like a vocabulary one.
|
|
102
|
+
tokenizable: list[KahEntry] = []
|
|
103
|
+
for entry in valid:
|
|
104
|
+
if tokenize(entry.headword, language.inventory) is not None:
|
|
105
|
+
tokenizable.append(entry)
|
|
106
|
+
|
|
107
|
+
legal = sum(
|
|
108
|
+
1
|
|
109
|
+
for entry in tokenizable
|
|
110
|
+
if is_legal_word(entry.headword, language.inventory, language.syllable_templates)
|
|
111
|
+
)
|
|
112
|
+
tokenizable_rate = len(tokenizable) / len(valid) if valid else 0.0
|
|
113
|
+
legality = legal / len(tokenizable) if tokenizable else 0.0
|
|
114
|
+
|
|
115
|
+
notes = [
|
|
116
|
+
f"{len(tokenizable)}/{len(valid)} headwords use only sounds this language has",
|
|
117
|
+
f"{legal}/{len(tokenizable)} of those are legal under this language's phonotactics",
|
|
118
|
+
f"overall: {legal}/{len(valid)} ({legal / len(valid) * 100 if valid else 0:.1f}%)",
|
|
119
|
+
"Kah is hand-made; this is a plausibility scorecard, not a pass/fail gate",
|
|
120
|
+
]
|
|
121
|
+
if PRESETS["kah"].symbols() == language.inventory.symbols():
|
|
122
|
+
notes.append(
|
|
123
|
+
"this language uses the Kah preset, so its scorecard is a self-check, "
|
|
124
|
+
"not an independent measurement"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
return BenchReport(
|
|
128
|
+
source=str(path),
|
|
129
|
+
entries_parsed=len(entries),
|
|
130
|
+
valid_models=len(valid),
|
|
131
|
+
validity_rate=len(valid) / len(entries) if entries else 0.0,
|
|
132
|
+
symbol_coverage=coverage,
|
|
133
|
+
tokenizable_rate=tokenizable_rate,
|
|
134
|
+
phonotactic_legality=legality,
|
|
135
|
+
kah_syllable_shapes=kah_syllable_shapes(valid),
|
|
136
|
+
lang_syllable_shapes=[t.pattern for t in language.syllable_templates],
|
|
137
|
+
notes=notes,
|
|
138
|
+
)
|
conlang/cli.py
ADDED
|
@@ -0,0 +1,430 @@
|
|
|
1
|
+
"""Command-line interface: a thin view over the library."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from .bench.kah import bench_kah
|
|
10
|
+
from .lint import lint
|
|
11
|
+
from .model import Language
|
|
12
|
+
from .pipeline import generate
|
|
13
|
+
from .pipeline.compose import compose_translation
|
|
14
|
+
from .pipeline.extras import build_extras
|
|
15
|
+
from .pipeline.grow import grow_lexicon
|
|
16
|
+
from .pipeline.port import PortabilityError, PortabilityReport
|
|
17
|
+
from .pipeline.scaffold import new_language
|
|
18
|
+
from .pipeline.validate import validate_translation
|
|
19
|
+
from .render.markdown import render_markdown
|
|
20
|
+
from .soundchange import evolve
|
|
21
|
+
from .spec import Spec, dump_spec, load_spec, to_spec
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _user_cache_dir() -> Path:
|
|
25
|
+
"""A conventional per-user cache directory, with no third-party dependency."""
|
|
26
|
+
xdg = os.environ.get("XDG_CACHE_HOME")
|
|
27
|
+
if xdg:
|
|
28
|
+
return Path(xdg) / "conlang"
|
|
29
|
+
if os.name == "nt":
|
|
30
|
+
local = os.environ.get("LOCALAPPDATA")
|
|
31
|
+
if local:
|
|
32
|
+
return Path(local) / "conlang"
|
|
33
|
+
return Path.home() / ".cache" / "conlang"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# Where `bench` looks for the Kah glossary when nothing is passed. The data is
|
|
37
|
+
# third-party, so it is never bundled: the user provides it via --kah, the
|
|
38
|
+
# CONLANG_KAH_LEXICON env var, or by dropping it at this path.
|
|
39
|
+
DEFAULT_KAH_CACHE = _user_cache_dir() / "kah-lexicon.txt"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _write(path: Path, text: str) -> None:
|
|
43
|
+
"""Write UTF-8 with LF endings so output bytes are identical on every platform."""
|
|
44
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
with path.open("w", encoding="utf-8", newline="\n") as handle:
|
|
46
|
+
handle.write(text)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _load_language(path: str) -> Language:
|
|
50
|
+
return Language.model_validate_json(Path(path).read_text(encoding="utf-8"))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _cmd_generate(args: argparse.Namespace) -> int:
|
|
54
|
+
spec = load_spec(args.spec)
|
|
55
|
+
if args.seed is not None:
|
|
56
|
+
spec = spec.model_copy(update={"seed": args.seed})
|
|
57
|
+
language = generate(spec)
|
|
58
|
+
out = Path(args.out) if args.out else Path("out") / f"{language.id}.json"
|
|
59
|
+
_write(out, language.model_dump_json(indent=2))
|
|
60
|
+
report = lint(language)
|
|
61
|
+
print(
|
|
62
|
+
f"{language.id}: {len(language.lexemes)} lexemes, "
|
|
63
|
+
f"{len(language.words)} word forms, seed {language.seed}"
|
|
64
|
+
)
|
|
65
|
+
print(f"saved {out}")
|
|
66
|
+
print(f"lint: {report.status} ({len(report.errors)} errors, {len(report.warnings)} warnings)")
|
|
67
|
+
for issue in report.issues[:20]:
|
|
68
|
+
print(f" [{issue.severity}] {issue.rule}: {issue.message}")
|
|
69
|
+
if args.render:
|
|
70
|
+
md = out.with_suffix(".md")
|
|
71
|
+
_write(md, render_markdown(language))
|
|
72
|
+
print(f"rendered {md}")
|
|
73
|
+
return 0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _cmd_render(args: argparse.Namespace) -> int:
|
|
77
|
+
language = _load_language(args.model)
|
|
78
|
+
text = render_markdown(language)
|
|
79
|
+
if args.out:
|
|
80
|
+
_write(Path(args.out), text)
|
|
81
|
+
print(f"rendered {args.out}")
|
|
82
|
+
else:
|
|
83
|
+
print(text)
|
|
84
|
+
return 0
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _cmd_lint(args: argparse.Namespace) -> int:
|
|
88
|
+
language = _load_language(args.model)
|
|
89
|
+
report = lint(language)
|
|
90
|
+
print(f"{language.id}: {report.status}")
|
|
91
|
+
for issue in report.issues:
|
|
92
|
+
print(f" [{issue.severity}] {issue.rule}: {issue.message}")
|
|
93
|
+
return 0 if report.status == "valid" else 1
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _cmd_bench(args: argparse.Namespace) -> int:
|
|
97
|
+
path = args.kah or os.environ.get("CONLANG_KAH_LEXICON") or str(DEFAULT_KAH_CACHE)
|
|
98
|
+
if not Path(path).exists():
|
|
99
|
+
print(f"Kah cache not found: {path}")
|
|
100
|
+
print("Pass --kah PATH or set CONLANG_KAH_LEXICON.")
|
|
101
|
+
return 2
|
|
102
|
+
if args.model:
|
|
103
|
+
language = _load_language(args.model)
|
|
104
|
+
elif args.spec:
|
|
105
|
+
language = generate(load_spec(args.spec))
|
|
106
|
+
else:
|
|
107
|
+
language = generate(Spec())
|
|
108
|
+
|
|
109
|
+
report = bench_kah(language, path)
|
|
110
|
+
print(f"Kah benchmark: {report.source}")
|
|
111
|
+
print(f" entries parsed: {report.entries_parsed}")
|
|
112
|
+
print(f" valid models: {report.valid_models} ({report.validity_rate:.1%})")
|
|
113
|
+
print(f" symbol coverage: {report.symbol_coverage:.1%}")
|
|
114
|
+
print(f" usable sounds: {report.tokenizable_rate:.1%}")
|
|
115
|
+
print(f" legal shapes: {report.phonotactic_legality:.1%}")
|
|
116
|
+
print(f" Kah syllable shapes: {' '.join(report.kah_syllable_shapes)}")
|
|
117
|
+
print(f" language templates: {' '.join(report.lang_syllable_shapes)}")
|
|
118
|
+
for note in report.notes:
|
|
119
|
+
print(f" note: {note}")
|
|
120
|
+
return 0
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _cmd_evolve(args: argparse.Namespace) -> int:
|
|
124
|
+
"""Derive a daughter language from a model by ordered sound changes."""
|
|
125
|
+
parent = _load_language(args.model)
|
|
126
|
+
try:
|
|
127
|
+
daughter = evolve(parent, args.rule, name=args.name)
|
|
128
|
+
except ValueError as error:
|
|
129
|
+
print(f"bad rule: {error}")
|
|
130
|
+
return 2
|
|
131
|
+
out = Path(args.out) if args.out else Path(args.model).with_name(f"{daughter.id}.json")
|
|
132
|
+
_write(out, daughter.model_dump_json(indent=2))
|
|
133
|
+
report = lint(daughter)
|
|
134
|
+
print(f"{daughter.id}: {len(daughter.lexemes)} lexemes, {len(daughter.words)} word forms")
|
|
135
|
+
print(f" derived from {daughter.lineage.parent} by {len(daughter.lineage.rules)} rule(s)")
|
|
136
|
+
print(f" saved {out}")
|
|
137
|
+
print(f"lint: {report.status} ({len(report.errors)} errors, {len(report.warnings)} warnings)")
|
|
138
|
+
return 0
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _cmd_accept(args: argparse.Namespace) -> int:
|
|
142
|
+
"""Freeze a generated model into a fully-explicit spec (write-back)."""
|
|
143
|
+
language = _load_language(args.model)
|
|
144
|
+
spec = to_spec(language)
|
|
145
|
+
out = (
|
|
146
|
+
Path(args.out)
|
|
147
|
+
if args.out
|
|
148
|
+
else Path(args.model).with_name(Path(args.model).stem + ".spec.yaml")
|
|
149
|
+
)
|
|
150
|
+
_write(out, dump_spec(spec))
|
|
151
|
+
affix_count = len(spec.affixes or [])
|
|
152
|
+
print(f"wrote {out}")
|
|
153
|
+
print(f" pinned {len(spec.lexicon)} lexemes and {affix_count} affixes")
|
|
154
|
+
print(" regenerating from this spec reproduces the same language")
|
|
155
|
+
return 0
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _resolve_under(
|
|
159
|
+
workspace: str, subdir: str, suffix: str, explicit: str | None, what: str
|
|
160
|
+
) -> Path:
|
|
161
|
+
"""Resolve a workspace path: an explicit flag, else the single match under `subdir`."""
|
|
162
|
+
if explicit:
|
|
163
|
+
return Path(explicit)
|
|
164
|
+
candidates = sorted((Path(workspace) / subdir).glob(f"*{suffix}"))
|
|
165
|
+
if len(candidates) == 1:
|
|
166
|
+
return candidates[0]
|
|
167
|
+
if not candidates:
|
|
168
|
+
raise ValueError(f"no {what} found under {Path(workspace) / subdir}")
|
|
169
|
+
raise ValueError(f"multiple {what}s under {Path(workspace) / subdir}; pass one explicitly")
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _print_port_report(report: PortabilityReport) -> None:
|
|
173
|
+
axes = ", ".join(f"{name}={'ok' if ok else 'MISMATCH'}" for name, ok in report.axes.items())
|
|
174
|
+
print(f"portable: {report.portable} ({axes})")
|
|
175
|
+
if report.missing_tags:
|
|
176
|
+
print(f" missing tags: {report.missing_tags}")
|
|
177
|
+
if report.missing_glosses:
|
|
178
|
+
print(f" missing glosses: {len(report.missing_glosses)} (grow these): "
|
|
179
|
+
f"{report.missing_glosses[:10]}")
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _cmd_new(args: argparse.Namespace) -> int:
|
|
183
|
+
def _split(value: str | None) -> list[str] | None:
|
|
184
|
+
return value.split(",") if value else None
|
|
185
|
+
|
|
186
|
+
orthography = (
|
|
187
|
+
dict(pair.split("=", 1) for pair in args.orthography) if args.orthography else None
|
|
188
|
+
)
|
|
189
|
+
result = new_language(
|
|
190
|
+
args.name,
|
|
191
|
+
seed=args.seed,
|
|
192
|
+
consonants=_split(args.consonants),
|
|
193
|
+
vowels=_split(args.vowels),
|
|
194
|
+
syllables=_split(args.syllables),
|
|
195
|
+
ipa=args.ipa,
|
|
196
|
+
orthography=orthography,
|
|
197
|
+
word_order=args.word_order,
|
|
198
|
+
alignment=args.alignment,
|
|
199
|
+
workspace=args.workspace,
|
|
200
|
+
git=not args.no_git,
|
|
201
|
+
)
|
|
202
|
+
print(f"{result.language_id}: {result.lexemes} lexemes, {result.word_forms} word forms")
|
|
203
|
+
print(f"wrote {result.spec_path}")
|
|
204
|
+
print(f"wrote {result.model_path}")
|
|
205
|
+
print(f"scaffolded {len(result.files)} files under {args.workspace}")
|
|
206
|
+
return 0
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _cmd_grow(args: argparse.Namespace) -> int:
|
|
210
|
+
try:
|
|
211
|
+
spec = _resolve_under(args.workspace, "spec", ".yaml", args.spec, "spec")
|
|
212
|
+
except ValueError as error:
|
|
213
|
+
print(error)
|
|
214
|
+
return 2
|
|
215
|
+
result = grow_lexicon(spec, args.words, out=args.out)
|
|
216
|
+
print(f"added {len(result.added)} -> {[(gloss, root) for gloss, root, _ in result.added]}")
|
|
217
|
+
if result.duplicates:
|
|
218
|
+
print(f"duplicates skipped: {result.duplicates}")
|
|
219
|
+
if result.reserved:
|
|
220
|
+
print(f"reserved pos skipped: {result.reserved}")
|
|
221
|
+
print(f"{result.language_id}: {result.lexemes} lexemes, {result.word_forms} word forms")
|
|
222
|
+
print(f"wrote {result.spec_path} and {result.model_path}")
|
|
223
|
+
return 0
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _cmd_translate(args: argparse.Namespace) -> int:
|
|
227
|
+
if bool(args.glosses) == bool(args.port_from):
|
|
228
|
+
print("pass exactly one of --glosses or --port-from")
|
|
229
|
+
return 2
|
|
230
|
+
try:
|
|
231
|
+
model = _resolve_under(args.workspace, "out", ".json", args.model, "model")
|
|
232
|
+
except ValueError as error:
|
|
233
|
+
print(error)
|
|
234
|
+
return 2
|
|
235
|
+
try:
|
|
236
|
+
result = compose_translation(
|
|
237
|
+
model,
|
|
238
|
+
args.out,
|
|
239
|
+
args.sentences,
|
|
240
|
+
glosses=args.glosses,
|
|
241
|
+
port_from=args.port_from,
|
|
242
|
+
source_model=args.source_model,
|
|
243
|
+
column=args.column,
|
|
244
|
+
check=args.check,
|
|
245
|
+
force=args.force,
|
|
246
|
+
)
|
|
247
|
+
except PortabilityError as error:
|
|
248
|
+
print(f"not portable: {error}")
|
|
249
|
+
_print_port_report(error.report)
|
|
250
|
+
return 1
|
|
251
|
+
if result.report is not None:
|
|
252
|
+
_print_port_report(result.report)
|
|
253
|
+
if args.check:
|
|
254
|
+
print("check only: nothing written")
|
|
255
|
+
return 0 if result.report and result.report.portable else 1
|
|
256
|
+
print(f"{result.language_id}: {result.translated}/{result.total} translated -> {result.out}")
|
|
257
|
+
for problem in result.problems:
|
|
258
|
+
print(f" problem: {problem}")
|
|
259
|
+
return 1 if result.problems else 0
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _cmd_validate_translation(args: argparse.Namespace) -> int:
|
|
263
|
+
try:
|
|
264
|
+
model = _resolve_under(args.workspace, "out", ".json", args.model, "model")
|
|
265
|
+
except ValueError as error:
|
|
266
|
+
print(error)
|
|
267
|
+
return 2
|
|
268
|
+
sentences = args.sentences or str(Path(args.workspace) / "the_quiet_morning_sentences.csv")
|
|
269
|
+
sentences_arg = sentences if Path(sentences).exists() else None
|
|
270
|
+
report = validate_translation(
|
|
271
|
+
args.csv,
|
|
272
|
+
model,
|
|
273
|
+
column=args.column,
|
|
274
|
+
rows=args.rows,
|
|
275
|
+
header=args.header,
|
|
276
|
+
sentences=sentences_arg,
|
|
277
|
+
extras=args.extras,
|
|
278
|
+
)
|
|
279
|
+
for issue in report.errors[:50]:
|
|
280
|
+
print(issue)
|
|
281
|
+
print(
|
|
282
|
+
f"translated {report.translated}/{report.total} · "
|
|
283
|
+
f"slots {report.slots} · errors {len(report.errors)}"
|
|
284
|
+
)
|
|
285
|
+
return 1 if report.errors else 0
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _cmd_extras(args: argparse.Namespace) -> int:
|
|
289
|
+
try:
|
|
290
|
+
model = _resolve_under(args.workspace, "out", ".json", args.model, "model")
|
|
291
|
+
except ValueError as error:
|
|
292
|
+
print(error)
|
|
293
|
+
return 2
|
|
294
|
+
result = build_extras(
|
|
295
|
+
args.csv, model, column=args.column, out_dir=args.out_dir, deck=args.deck
|
|
296
|
+
)
|
|
297
|
+
print(f"gloss CSV: {result['gloss_csv']}")
|
|
298
|
+
print(f"docx: {result['docx']}")
|
|
299
|
+
print(f"deck: {result['deck']} ({result['cards']} cards)")
|
|
300
|
+
return 0
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
304
|
+
parser = argparse.ArgumentParser(
|
|
305
|
+
prog="conlang",
|
|
306
|
+
description="Generate a conlang from a partial spec; sample the rest.",
|
|
307
|
+
)
|
|
308
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
309
|
+
|
|
310
|
+
gen = sub.add_parser("generate", help="Generate a language model from a spec.")
|
|
311
|
+
gen.add_argument("--spec", required=True, help="Path to a YAML spec.")
|
|
312
|
+
gen.add_argument("--seed", type=int, default=None, help="Override the spec seed.")
|
|
313
|
+
gen.add_argument("--out", default=None, help="Output model JSON path.")
|
|
314
|
+
gen.add_argument("--render", action="store_true", help="Also render Markdown.")
|
|
315
|
+
gen.set_defaults(func=_cmd_generate)
|
|
316
|
+
|
|
317
|
+
ren = sub.add_parser("render", help="Render a model JSON to Markdown.")
|
|
318
|
+
ren.add_argument("--model", required=True, help="Path to a model JSON.")
|
|
319
|
+
ren.add_argument("--out", default=None, help="Output Markdown path (default stdout).")
|
|
320
|
+
ren.set_defaults(func=_cmd_render)
|
|
321
|
+
|
|
322
|
+
lin = sub.add_parser("lint", help="Lint a model JSON.")
|
|
323
|
+
lin.add_argument("--model", required=True, help="Path to a model JSON.")
|
|
324
|
+
lin.set_defaults(func=_cmd_lint)
|
|
325
|
+
|
|
326
|
+
acc = sub.add_parser(
|
|
327
|
+
"accept",
|
|
328
|
+
help="Freeze a model into an explicit spec (write-back), so sampled values stick.",
|
|
329
|
+
)
|
|
330
|
+
acc.add_argument("--model", required=True, help="Path to a model JSON.")
|
|
331
|
+
acc.add_argument("--out", default=None, help="Output spec path (default <model>.spec.yaml).")
|
|
332
|
+
acc.set_defaults(func=_cmd_accept)
|
|
333
|
+
|
|
334
|
+
ben = sub.add_parser("bench", help="Score a language against the local Kah cache.")
|
|
335
|
+
ben.add_argument("--model", default=None, help="Path to a model JSON.")
|
|
336
|
+
ben.add_argument("--spec", default=None, help="Generate from this spec, then bench.")
|
|
337
|
+
ben.add_argument("--kah", default=None, help="Path to kah-lexicon.txt.")
|
|
338
|
+
ben.set_defaults(func=_cmd_bench)
|
|
339
|
+
|
|
340
|
+
evo = sub.add_parser(
|
|
341
|
+
"evolve", help="Derive a daughter language by ordered sound changes."
|
|
342
|
+
)
|
|
343
|
+
evo.add_argument("--model", required=True, help="Path to the parent model JSON.")
|
|
344
|
+
evo.add_argument(
|
|
345
|
+
"--rule",
|
|
346
|
+
action="append",
|
|
347
|
+
required=True,
|
|
348
|
+
metavar="OLD > NEW",
|
|
349
|
+
help="A sound change, e.g. 'a > e'. Repeatable; applied in order.",
|
|
350
|
+
)
|
|
351
|
+
evo.add_argument("--name", default=None, help="Daughter language id.")
|
|
352
|
+
evo.add_argument("--out", default=None, help="Output model JSON path.")
|
|
353
|
+
evo.set_defaults(func=_cmd_evolve)
|
|
354
|
+
|
|
355
|
+
new = sub.add_parser("new", help="Scaffold and freeze a new generated language.")
|
|
356
|
+
new.add_argument("--name", required=True)
|
|
357
|
+
new.add_argument("--seed", type=int, default=1)
|
|
358
|
+
new.add_argument("--consonants", default=None, help="comma-separated symbols")
|
|
359
|
+
new.add_argument("--vowels", default=None, help="comma-separated symbols")
|
|
360
|
+
new.add_argument("--syllables", default=None, help="comma-separated, e.g. CV,CVC")
|
|
361
|
+
new.add_argument("--ipa", action="store_true", help="treat the inventory as IPA")
|
|
362
|
+
new.add_argument(
|
|
363
|
+
"--orthography", action="append", default=None, metavar="SYM=SPELL",
|
|
364
|
+
help="orthography override, repeatable (e.g. ʃ=sh)",
|
|
365
|
+
)
|
|
366
|
+
new.add_argument(
|
|
367
|
+
"--word-order", dest="word_order", default=None, choices=["SOV", "SVO", "VSO"],
|
|
368
|
+
help="pin the word order (default: sampled)",
|
|
369
|
+
)
|
|
370
|
+
new.add_argument(
|
|
371
|
+
"--alignment", default=None, choices=["accusative", "ergative"],
|
|
372
|
+
help="pin the alignment (default: sampled)",
|
|
373
|
+
)
|
|
374
|
+
new.add_argument("--workspace", default=".", help="workspace root (default: .)")
|
|
375
|
+
new.add_argument("--no-git", action="store_true", help="do not git init the workspace")
|
|
376
|
+
new.set_defaults(func=_cmd_new)
|
|
377
|
+
|
|
378
|
+
grow = sub.add_parser("grow", help="Grow a frozen spec's lexicon, in place.")
|
|
379
|
+
grow.add_argument("--words", required=True, help="JSON list/map of gloss -> pos")
|
|
380
|
+
grow.add_argument("--workspace", default=".")
|
|
381
|
+
grow.add_argument("--spec", default=None, help="default: <workspace>/spec/*.yaml")
|
|
382
|
+
grow.add_argument("--out", default=None, help="model JSON path")
|
|
383
|
+
grow.set_defaults(func=_cmd_grow)
|
|
384
|
+
|
|
385
|
+
tr = sub.add_parser("translate", help="Compose a translation, or port a sibling's.")
|
|
386
|
+
tr.add_argument("--workspace", default=".")
|
|
387
|
+
tr.add_argument("--model", default=None, help="default: <workspace>/out/*.json")
|
|
388
|
+
tr.add_argument("--glosses", default=None, help="hand-authored gloss-script TSV")
|
|
389
|
+
tr.add_argument("--port-from", dest="port_from", default=None, help="source _gloss.csv")
|
|
390
|
+
tr.add_argument("--source-model", dest="source_model", default=None, help="the source model")
|
|
391
|
+
tr.add_argument("--sentences", required=True, help="source sentences CSV")
|
|
392
|
+
tr.add_argument("--out", required=True, help="output translation CSV")
|
|
393
|
+
tr.add_argument("--column", default=None, help="language column (default: the model id)")
|
|
394
|
+
tr.add_argument("--check", action="store_true", help="report portability, write nothing")
|
|
395
|
+
tr.add_argument("--force", action="store_true", help="port despite a grammar mismatch")
|
|
396
|
+
tr.set_defaults(func=_cmd_translate)
|
|
397
|
+
|
|
398
|
+
val = sub.add_parser("validate-translation", help="Validate a translation CSV.")
|
|
399
|
+
val.add_argument("--csv", required=True)
|
|
400
|
+
val.add_argument("--workspace", default=".")
|
|
401
|
+
val.add_argument("--model", default=None, help="default: <workspace>/out/*.json")
|
|
402
|
+
val.add_argument("--column", default=None)
|
|
403
|
+
val.add_argument("--rows", type=int, default=0, help="expected rows (default: derived)")
|
|
404
|
+
val.add_argument("--header", default=None, help="expected header, comma-separated")
|
|
405
|
+
val.add_argument("--sentences", default=None, help="source sentences CSV")
|
|
406
|
+
val.add_argument(
|
|
407
|
+
"--extras", action="store_true",
|
|
408
|
+
help="emit the gloss CSV / docx / deck, then check completeness",
|
|
409
|
+
)
|
|
410
|
+
val.set_defaults(func=_cmd_validate_translation)
|
|
411
|
+
|
|
412
|
+
ex = sub.add_parser("extras", help="Render a translation's gloss CSV, docx, and deck.")
|
|
413
|
+
ex.add_argument("--csv", required=True)
|
|
414
|
+
ex.add_argument("--workspace", default=".")
|
|
415
|
+
ex.add_argument("--model", default=None, help="default: <workspace>/out/*.json")
|
|
416
|
+
ex.add_argument("--column", default=None)
|
|
417
|
+
ex.add_argument("--out-dir", dest="out_dir", default=None)
|
|
418
|
+
ex.add_argument("--deck", default=None)
|
|
419
|
+
ex.set_defaults(func=_cmd_extras)
|
|
420
|
+
|
|
421
|
+
return parser
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def main(argv: list[str] | None = None) -> int:
|
|
425
|
+
args = build_parser().parse_args(argv)
|
|
426
|
+
return int(args.func(args))
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
if __name__ == "__main__":
|
|
430
|
+
raise SystemExit(main())
|