khamyang 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- khamyang/__init__.py +75 -0
- khamyang/cli.py +416 -0
- khamyang/core/__init__.py +19 -0
- khamyang/core/models.py +130 -0
- khamyang/corpora/ai/instruction_dataset.jsonl +5 -0
- khamyang/corpora/audio_metadata/audio_index.json +98 -0
- khamyang/corpora/conversations/conversations.json +124 -0
- khamyang/corpora/culture/material_culture.json +196 -0
- khamyang/corpora/geography/geography.json +62 -0
- khamyang/corpora/instruction_dataset.jsonl +5 -0
- khamyang/corpora/parallel_corpus/parallel_corpus.jsonl +16 -0
- khamyang/corpora/proverbs/proverbs.json +110 -0
- khamyang/corpora/sources/licenses.json +39 -0
- khamyang/corpora/sources/source_registry.json +220 -0
- khamyang/corpus/__init__.py +26 -0
- khamyang/corpus/exporters.py +97 -0
- khamyang/corpus/igt.py +77 -0
- khamyang/corpus/songs.py +112 -0
- khamyang/corpus/texts.py +276 -0
- khamyang/data/__init__.py +10 -0
- khamyang/data/bibliography.json +247 -0
- khamyang/data/build_data.py +1461 -0
- khamyang/data/kinship.json +848 -0
- khamyang/data/lexicon.json +20800 -0
- khamyang/data/phrases.json +308 -0
- khamyang/data/songs.json +388 -0
- khamyang/data/texts.json +475 -0
- khamyang/export/__init__.py +13 -0
- khamyang/export/exporter.py +130 -0
- khamyang/facade.py +243 -0
- khamyang/grammar/__init__.py +54 -0
- khamyang/grammar/classifiers.py +203 -0
- khamyang/grammar/kinship.py +218 -0
- khamyang/grammar/naming.py +132 -0
- khamyang/grammar/numerals.py +196 -0
- khamyang/grammar/pronouns.py +55 -0
- khamyang/grammar/syntax.py +130 -0
- khamyang/lexicon/__init__.py +17 -0
- khamyang/lexicon/dictionary.py +219 -0
- khamyang/lexicon/search.py +46 -0
- khamyang/nlp/__init__.py +86 -0
- khamyang/nlp/language_detector.py +95 -0
- khamyang/nlp/lemmatizer.py +48 -0
- khamyang/nlp/normalizer.py +53 -0
- khamyang/nlp/pos_tagger.py +93 -0
- khamyang/nlp/tokenizer.py +59 -0
- khamyang/phonology/__init__.py +37 -0
- khamyang/phonology/phonemes.py +158 -0
- khamyang/phonology/synth.py +43 -0
- khamyang/phonology/tones.py +208 -0
- khamyang/phrases/__init__.py +13 -0
- khamyang/phrases/phrasebook.py +397 -0
- khamyang/scraper/__init__.py +7 -0
- khamyang/scraper/downloader.py +235 -0
- khamyang/scraper/harvester.py +109 -0
- khamyang/transliteration/__init__.py +25 -0
- khamyang/transliteration/converter.py +236 -0
- khamyang/transliteration/scripts.py +177 -0
- khamyang/version.py +7 -0
- khamyang/web/__init__.py +5 -0
- khamyang/web/server.py +268 -0
- khamyang/web/static/app.js +680 -0
- khamyang/web/static/style.css +1130 -0
- khamyang/web/templates/index.html +552 -0
- khamyang-1.0.0.dist-info/METADATA +382 -0
- khamyang-1.0.0.dist-info/RECORD +72 -0
- khamyang-1.0.0.dist-info/WHEEL +5 -0
- khamyang-1.0.0.dist-info/entry_points.txt +2 -0
- khamyang-1.0.0.dist-info/licenses/LICENSE +21 -0
- khamyang-1.0.0.dist-info/top_level.txt +3 -0
- tai_khamyang/__init__.py +128 -0
- taikhamyang/__init__.py +113 -0
khamyang/__init__.py
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tai Khamyang (ksu) Language Library
|
|
3
|
+
===================================
|
|
4
|
+
A comprehensive Python library for the revitalization, documentation,
|
|
5
|
+
and computational processing of the critically endangered Tai Khamyang
|
|
6
|
+
language (ISO 639-3: ksu, Glottolog: kham1282) spoken in Assam, India.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from khamyang.version import __version__, __author__, __license__
|
|
10
|
+
from khamyang.lexicon.dictionary import KhamyangDictionary, get_dictionary
|
|
11
|
+
from khamyang.phonology.tones import ToneSystem, KhamyangTone, gedney_lookup
|
|
12
|
+
from khamyang.transliteration.converter import (
|
|
13
|
+
transliterate,
|
|
14
|
+
roman_to_liktai,
|
|
15
|
+
liktai_to_roman,
|
|
16
|
+
roman_to_assamese,
|
|
17
|
+
assamese_to_roman,
|
|
18
|
+
roman_to_ipa,
|
|
19
|
+
)
|
|
20
|
+
from khamyang.grammar.classifiers import ClassifierEngine
|
|
21
|
+
from khamyang.grammar.naming import get_birth_name, list_birth_names
|
|
22
|
+
from khamyang.grammar.numerals import num_to_khamyang, khamyang_to_num
|
|
23
|
+
from khamyang.phrases.phrasebook import KhamyangPhrasebook, get_phrasebook
|
|
24
|
+
from khamyang.corpus.texts import KhamyangCorpus, get_corpus
|
|
25
|
+
from khamyang.corpus.igt import InterlinearGloss
|
|
26
|
+
from khamyang.corpus.songs import SongEngine
|
|
27
|
+
from khamyang.grammar.kinship import KinshipEngine, CLANS
|
|
28
|
+
from khamyang.scraper.downloader import BibliographyManager
|
|
29
|
+
from khamyang.facade import Khamyang
|
|
30
|
+
from khamyang.nlp import (
|
|
31
|
+
normalize,
|
|
32
|
+
tokenize,
|
|
33
|
+
detokenize,
|
|
34
|
+
pos_tag,
|
|
35
|
+
detect_language,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
"__version__",
|
|
40
|
+
"__author__",
|
|
41
|
+
"__license__",
|
|
42
|
+
"Khamyang",
|
|
43
|
+
"KhamyangDictionary",
|
|
44
|
+
"get_dictionary",
|
|
45
|
+
"ToneSystem",
|
|
46
|
+
"KhamyangTone",
|
|
47
|
+
"gedney_lookup",
|
|
48
|
+
"transliterate",
|
|
49
|
+
"roman_to_liktai",
|
|
50
|
+
"liktai_to_roman",
|
|
51
|
+
"roman_to_assamese",
|
|
52
|
+
"assamese_to_roman",
|
|
53
|
+
"roman_to_ipa",
|
|
54
|
+
"ClassifierEngine",
|
|
55
|
+
"get_birth_name",
|
|
56
|
+
"list_birth_names",
|
|
57
|
+
"num_to_khamyang",
|
|
58
|
+
"khamyang_to_num",
|
|
59
|
+
"KhamyangPhrasebook",
|
|
60
|
+
"get_phrasebook",
|
|
61
|
+
"KhamyangCorpus",
|
|
62
|
+
"get_corpus",
|
|
63
|
+
"InterlinearGloss",
|
|
64
|
+
"SongEngine",
|
|
65
|
+
"KinshipEngine",
|
|
66
|
+
"CLANS",
|
|
67
|
+
"BibliographyManager",
|
|
68
|
+
"normalize",
|
|
69
|
+
"tokenize",
|
|
70
|
+
"detokenize",
|
|
71
|
+
"pos_tag",
|
|
72
|
+
"detect_language",
|
|
73
|
+
]
|
|
74
|
+
|
|
75
|
+
|
khamyang/cli.py
ADDED
|
@@ -0,0 +1,416 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tai Khamyang Command-Line Interface
|
|
3
|
+
===================================
|
|
4
|
+
Interactive CLI for developers, linguists, and researchers.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
import click
|
|
10
|
+
from rich.console import Console
|
|
11
|
+
from rich.table import Table
|
|
12
|
+
from rich.panel import Panel
|
|
13
|
+
from rich.text import Text
|
|
14
|
+
|
|
15
|
+
import khamyang
|
|
16
|
+
from khamyang.phonology.tones import ToneSystem, TONES
|
|
17
|
+
from khamyang.grammar.naming import list_birth_names, generate_full_name, KHAMYANG_CLANS
|
|
18
|
+
from khamyang.grammar.classifiers import ClassifierEngine
|
|
19
|
+
from khamyang.corpus.igt import InterlinearGloss
|
|
20
|
+
|
|
21
|
+
console = Console()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@click.group()
|
|
25
|
+
@click.version_option(version=khamyang.__version__, prog_name="khamyang")
|
|
26
|
+
def cli():
|
|
27
|
+
"""Tai Khamyang (ksu) Language Revitalization & Linguistic Toolkit."""
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@cli.command("lookup")
|
|
32
|
+
@click.argument("term")
|
|
33
|
+
def lookup_cmd(term: str):
|
|
34
|
+
"""Search for a word in Tai Khamyang (by Roman, Lik Tai, Assamese, or English)."""
|
|
35
|
+
d = khamyang.get_dictionary()
|
|
36
|
+
results = d.lookup(term)
|
|
37
|
+
|
|
38
|
+
if not results:
|
|
39
|
+
console.print(f"[bold red]No entries found for:[/bold red] '{term}'")
|
|
40
|
+
return
|
|
41
|
+
|
|
42
|
+
for w in results[:5]:
|
|
43
|
+
tone_obj = TONES.get(w.tone)
|
|
44
|
+
tone_str = f"Tone {w.tone} ({tone_obj.name if tone_obj else ''}) [chao: {tone_obj.pitch_chao if tone_obj else ''}]"
|
|
45
|
+
|
|
46
|
+
panel_content = Text()
|
|
47
|
+
panel_content.append(f"Lik Tai: ", style="bold cyan")
|
|
48
|
+
panel_content.append(f"{w.lik_tai}\n", style="bold yellow")
|
|
49
|
+
panel_content.append(f"Romanized: ", style="bold cyan")
|
|
50
|
+
panel_content.append(f"{w.roman}\n", style="bold green")
|
|
51
|
+
panel_content.append(f"Assamese: ", style="bold cyan")
|
|
52
|
+
panel_content.append(f"{w.assamese}\n", style="bold magenta")
|
|
53
|
+
panel_content.append(f"IPA: ", style="bold cyan")
|
|
54
|
+
panel_content.append(f"{w.ipa}\n", style="italic")
|
|
55
|
+
panel_content.append(f"Part of Speech: ", style="bold cyan")
|
|
56
|
+
panel_content.append(f"{w.pos}\n")
|
|
57
|
+
panel_content.append(f"Tone: ", style="bold cyan")
|
|
58
|
+
panel_content.append(f"{tone_str}\n", style="bold blue")
|
|
59
|
+
panel_content.append(f"English: ", style="bold cyan")
|
|
60
|
+
panel_content.append(f"{w.gloss_en}\n", style="bold white")
|
|
61
|
+
panel_content.append(f"Assamese: ", style="bold cyan")
|
|
62
|
+
panel_content.append(f"{w.gloss_as}\n", style="bold white")
|
|
63
|
+
panel_content.append(f"Category: ", style="bold cyan")
|
|
64
|
+
panel_content.append(f"{w.category}\n")
|
|
65
|
+
|
|
66
|
+
if w.cognates:
|
|
67
|
+
cogs = ", ".join(f"{k}: {v}" for k, v in w.cognates.items())
|
|
68
|
+
panel_content.append(f"Cognates: ", style="bold cyan")
|
|
69
|
+
panel_content.append(f"{cogs}\n", style="dim")
|
|
70
|
+
|
|
71
|
+
console.print(Panel(panel_content, title=f"[bold]Tai Khamyang: {w.roman}[/bold]", border_style="cyan"))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@cli.command("search")
|
|
75
|
+
@click.argument("query")
|
|
76
|
+
@click.option("--category", "-c", default=None, help="Filter by category (e.g. nature, body, food, kinship)")
|
|
77
|
+
@click.option("--tone", "-t", default=None, type=int, help="Filter by tone (1 to 6)")
|
|
78
|
+
@click.option("--limit", "-n", default=20, help="Max number of results")
|
|
79
|
+
def search_cmd(query: str, category: str, tone: int, limit: int):
|
|
80
|
+
"""Full-text search across dictionary meanings and transcriptions."""
|
|
81
|
+
d = khamyang.get_dictionary()
|
|
82
|
+
results = d.search(query, category=category, tone=tone, limit=limit)
|
|
83
|
+
|
|
84
|
+
if not results:
|
|
85
|
+
console.print(f"[yellow]No matches found for query:[/yellow] '{query}'")
|
|
86
|
+
return
|
|
87
|
+
|
|
88
|
+
table = Table(title=f"Khamyang Search Results: '{query}' ({len(results)} found)")
|
|
89
|
+
table.add_column("Lik Tai", style="yellow")
|
|
90
|
+
table.add_column("Roman", style="bold green")
|
|
91
|
+
table.add_column("Assamese", style="magenta")
|
|
92
|
+
table.add_column("Tone", style="cyan")
|
|
93
|
+
table.add_column("IPA", style="dim")
|
|
94
|
+
table.add_column("POS", style="italic")
|
|
95
|
+
table.add_column("English", style="bold white")
|
|
96
|
+
table.add_column("Assamese Gloss", style="white")
|
|
97
|
+
|
|
98
|
+
for w in results:
|
|
99
|
+
table.add_row(
|
|
100
|
+
w.lik_tai,
|
|
101
|
+
w.roman,
|
|
102
|
+
w.assamese,
|
|
103
|
+
str(w.tone),
|
|
104
|
+
w.ipa,
|
|
105
|
+
w.pos,
|
|
106
|
+
w.gloss_en,
|
|
107
|
+
w.gloss_as,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
console.print(table)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@cli.command("transliterate")
|
|
114
|
+
@click.argument("text")
|
|
115
|
+
@click.option("--to", "target", default="liktai", type=click.Choice(["liktai", "assamese", "roman", "ipa"], case_sensitive=False))
|
|
116
|
+
@click.option("--from", "source", default="roman", type=click.Choice(["roman", "liktai", "assamese"], case_sensitive=False))
|
|
117
|
+
def transliterate_cmd(text: str, target: str, source: str):
|
|
118
|
+
"""Convert text between Romanization, Lik Tai, Assamese, and IPA."""
|
|
119
|
+
out = khamyang.transliterate(text, source_script=source, target_script=target)
|
|
120
|
+
console.print(f"[bold cyan]Input ({source}):[/bold cyan] {text}")
|
|
121
|
+
console.print(f"[bold yellow]Output ({target}):[/bold yellow] {out}")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@cli.command("tone")
|
|
125
|
+
@click.argument("tone_or_syllable")
|
|
126
|
+
def tone_cmd(tone_or_syllable: str):
|
|
127
|
+
"""Explain one of the 6 tones or analyze a syllable's pitch."""
|
|
128
|
+
if tone_or_syllable.isdigit():
|
|
129
|
+
t_num = int(tone_or_syllable)
|
|
130
|
+
else:
|
|
131
|
+
t_num = ToneSystem.extract_tone(tone_or_syllable) or 1
|
|
132
|
+
|
|
133
|
+
t_info = ToneSystem.explain_tone(t_num)
|
|
134
|
+
p = Text()
|
|
135
|
+
p.append(f"Tone Number: ", style="bold")
|
|
136
|
+
p.append(f"{t_info['number']}\n", style="bold yellow")
|
|
137
|
+
p.append(f"Linguistic Name: ", style="bold")
|
|
138
|
+
p.append(f"{t_info['name']}\n", style="bold green")
|
|
139
|
+
p.append(f"Chao Contour: ", style="bold")
|
|
140
|
+
p.append(f"{t_info['pitch_contour']}\n", style="bold cyan")
|
|
141
|
+
p.append(f"IPA Symbol: ", style="bold")
|
|
142
|
+
p.append(f"{t_info['ipa']}\n", style="bold magenta")
|
|
143
|
+
p.append(f"Creaky Voice: ", style="bold")
|
|
144
|
+
p.append(f"{t_info['creaky_voice']}\n")
|
|
145
|
+
p.append(f"Glottal Stop: ", style="bold")
|
|
146
|
+
p.append(f"{t_info['glottal_stop']}\n")
|
|
147
|
+
p.append(f"Gedney Box: ", style="bold")
|
|
148
|
+
p.append(f"{t_info['gedney_origin']}\n", style="dim")
|
|
149
|
+
p.append(f"Description: ", style="bold")
|
|
150
|
+
p.append(f"{t_info['description']}\n")
|
|
151
|
+
|
|
152
|
+
console.print(Panel(p, title=f"Tone {t_num} Analysis", border_style="blue"))
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@cli.command("phrase")
|
|
156
|
+
@click.option("--category", "-c", default=None, help="Category (fieldwork, greeting, daily, food, identity, culture)")
|
|
157
|
+
@click.option("--search", "-s", default=None, help="Search text in phrases")
|
|
158
|
+
def phrase_cmd(category: str, search: str):
|
|
159
|
+
"""Browse the conversational phrasebook and fieldwork elicitation sentences."""
|
|
160
|
+
pb = khamyang.get_phrasebook()
|
|
161
|
+
if search:
|
|
162
|
+
phrases = pb.search(search)
|
|
163
|
+
elif category:
|
|
164
|
+
phrases = pb.get_by_category(category)
|
|
165
|
+
else:
|
|
166
|
+
phrases = pb.get_all()
|
|
167
|
+
|
|
168
|
+
table = Table(title=f"Khamyang Phrasebook ({len(phrases)} phrases)")
|
|
169
|
+
table.add_column("Lik Tai", style="yellow")
|
|
170
|
+
table.add_column("Roman", style="bold green")
|
|
171
|
+
table.add_column("Assamese", style="magenta")
|
|
172
|
+
table.add_column("English Meaning", style="bold white")
|
|
173
|
+
table.add_column("Assamese Meaning", style="cyan")
|
|
174
|
+
|
|
175
|
+
for p in phrases[:25]:
|
|
176
|
+
table.add_row(
|
|
177
|
+
p.lik_tai,
|
|
178
|
+
p.roman,
|
|
179
|
+
p.assamese,
|
|
180
|
+
p.english,
|
|
181
|
+
p.assamese_trans,
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
console.print(table)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@cli.command("num")
|
|
188
|
+
@click.argument("number", type=int)
|
|
189
|
+
@click.option("--script", default="roman", type=click.Choice(["roman", "liktai", "assamese"]))
|
|
190
|
+
def num_cmd(number: int, script: str):
|
|
191
|
+
"""Convert an integer (0 to 1,000,000) into Tai Khamyang words."""
|
|
192
|
+
result = khamyang.num_to_khamyang(number, script=script)
|
|
193
|
+
console.print(f"[bold cyan]Number {number}:[/bold cyan] [bold yellow]{result}[/bold yellow]")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
@cli.command("name")
|
|
197
|
+
@click.option("--gender", "-g", default="m", type=click.Choice(["m", "f", "male", "female"]))
|
|
198
|
+
@click.option("--order", "-o", default=1, type=int, help="Birth order (1 to 8)")
|
|
199
|
+
@click.option("--personal", "-p", default="", help="Optional personal name")
|
|
200
|
+
@click.option("--clan", default="Thomung", help="Clan surname (Thomung, Chowlik, Chaohai, Pangyok, Wailong)")
|
|
201
|
+
def name_cmd(gender: str, order: int, personal: str, clan: str):
|
|
202
|
+
"""Generate or look up traditional Tai Khamyang birth-order names."""
|
|
203
|
+
full = generate_full_name(order, gender, personal_name=personal, clan=clan)
|
|
204
|
+
p = Text()
|
|
205
|
+
p.append(f"Full Name: ", style="bold")
|
|
206
|
+
p.append(f"{full['full_roman']}\n", style="bold yellow")
|
|
207
|
+
p.append(f"Birth Order Name: ", style="bold")
|
|
208
|
+
p.append(f"{full['birth_order_name']}\n", style="bold green")
|
|
209
|
+
p.append(f"Clan Surname: ", style="bold")
|
|
210
|
+
p.append(f"{full['clan']}\n", style="bold cyan")
|
|
211
|
+
p.append(f"Significance: ", style="bold")
|
|
212
|
+
p.append(f"{full['meaning']}\n", style="white")
|
|
213
|
+
|
|
214
|
+
console.print(Panel(p, title=f"Khamyang Birth-Order Name ({gender.upper()} #{order})", border_style="green"))
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@cli.command("stats")
|
|
218
|
+
def stats_cmd():
|
|
219
|
+
"""Display comprehensive linguistic metrics of the Khamyang language library."""
|
|
220
|
+
d = khamyang.get_dictionary()
|
|
221
|
+
st = d.statistics()
|
|
222
|
+
|
|
223
|
+
table = Table(title="Tai Khamyang Linguistic Database Metrics")
|
|
224
|
+
table.add_column("Metric", style="bold cyan")
|
|
225
|
+
table.add_column("Value", style="bold yellow")
|
|
226
|
+
|
|
227
|
+
table.add_row("Total Lexical Entries", str(st["total_words"]))
|
|
228
|
+
table.add_row("Total Categories", str(st["total_categories"]))
|
|
229
|
+
for t_num, count in st["tones_distribution"].items():
|
|
230
|
+
table.add_row(f" • Tone {t_num} words", str(count))
|
|
231
|
+
|
|
232
|
+
console.print(table)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
@cli.command("export")
|
|
236
|
+
@click.option("--format", "-f", default="csv", type=click.Choice(["csv", "anki", "html"]))
|
|
237
|
+
@click.option("--output", "-o", required=True, help="Output destination file path")
|
|
238
|
+
def export_cmd(format: str, output: str):
|
|
239
|
+
"""Export dictionary to CSV, Anki flashcard TSV, or interactive HTML."""
|
|
240
|
+
from khamyang.export.exporter import export_to_csv, export_to_anki_tsv, export_to_html
|
|
241
|
+
if format == "csv":
|
|
242
|
+
export_to_csv(output)
|
|
243
|
+
elif format == "anki":
|
|
244
|
+
export_to_anki_tsv(output)
|
|
245
|
+
elif format == "html":
|
|
246
|
+
export_to_html(output)
|
|
247
|
+
console.print(f"[bold green]Successfully exported dictionary ({format}) to:[/bold green] {output}")
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
@cli.command("web")
|
|
251
|
+
@click.option("--port", "-p", default=8000, help="Web server port (default: 8000)")
|
|
252
|
+
@click.option("--host", "-h", default="127.0.0.1", help="Host address (default: 127.0.0.1)")
|
|
253
|
+
def web_cmd(port: int, host: str):
|
|
254
|
+
"""Start the interactive Tai Khamyang web revitalization interface."""
|
|
255
|
+
from khamyang.web.server import run_server
|
|
256
|
+
console.print(f"[bold green]Starting Tai Khamyang Web Platform at[/bold green] [bold cyan]http://{host}:{port}[/bold cyan]")
|
|
257
|
+
run_server(host=host, port=port)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
@cli.command("song")
|
|
261
|
+
@click.argument("action", default="list", type=click.Choice(["list", "show", "search"]))
|
|
262
|
+
@click.argument("query", required=False, default="")
|
|
263
|
+
def song_cmd(action: str, query: str):
|
|
264
|
+
"""Browse, view lyrics, and search authentic Tai Khamyang traditional songs."""
|
|
265
|
+
from khamyang.corpus.songs import SongEngine
|
|
266
|
+
|
|
267
|
+
if action == "list":
|
|
268
|
+
songs = SongEngine.all_songs()
|
|
269
|
+
table = Table(title="Tai Khamyang Traditional Songs & Ballads (Fɯn)")
|
|
270
|
+
table.add_column("ID", style="bold cyan")
|
|
271
|
+
table.add_column("Title (Khamyang)", style="bold yellow")
|
|
272
|
+
table.add_column("English Title", style="bold green")
|
|
273
|
+
table.add_column("Genre", style="magenta")
|
|
274
|
+
table.add_column("Informant / Origin", style="white")
|
|
275
|
+
|
|
276
|
+
for s in songs:
|
|
277
|
+
table.add_row(s["id"], s["title_khamyang"], s["title_english"], s["genre"], s["informant"])
|
|
278
|
+
console.print(table)
|
|
279
|
+
console.print("\n[italic dim]Use 'khamyang song show <id>' to read complete lyrics, glosses, and cultural context.[/italic dim]")
|
|
280
|
+
|
|
281
|
+
elif action == "show":
|
|
282
|
+
if not query:
|
|
283
|
+
console.print("[bold red]Please specify a song ID, e.g.:[/bold red] 'khamyang song show song_001'")
|
|
284
|
+
return
|
|
285
|
+
s = SongEngine.get_song(query)
|
|
286
|
+
if not s:
|
|
287
|
+
console.print(f"[bold red]Song not found:[/bold red] '{query}'")
|
|
288
|
+
return
|
|
289
|
+
text = SongEngine.format_song_text(s)
|
|
290
|
+
console.print(Panel(text, title=f"🎵 {s['title_khamyang']} ({s['title_english']})", border_style="yellow"))
|
|
291
|
+
|
|
292
|
+
elif action == "search":
|
|
293
|
+
if not query:
|
|
294
|
+
console.print("[bold red]Please provide a search term.[/bold red]")
|
|
295
|
+
return
|
|
296
|
+
res = SongEngine.search(query)
|
|
297
|
+
if not res:
|
|
298
|
+
console.print(f"[bold red]No songs found matching:[/bold red] '{query}'")
|
|
299
|
+
return
|
|
300
|
+
for s in res:
|
|
301
|
+
console.print(f"[bold yellow]{s['id']}[/bold yellow] - [bold cyan]{s['title_khamyang']}[/bold cyan] ({s['title_english']}): {s['description']}")
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
@cli.command("kinship")
|
|
305
|
+
@click.argument("action", default="resolve", type=click.Choice(["resolve", "search", "generation", "clans"]))
|
|
306
|
+
@click.argument("query", required=False, default="")
|
|
307
|
+
def kinship_cmd(action: str, query: str):
|
|
308
|
+
"""Solve and inspect Tai Khamyang kinship relations, terms, and clan lineages."""
|
|
309
|
+
from khamyang.grammar.kinship import KinshipEngine
|
|
310
|
+
|
|
311
|
+
if action == "resolve":
|
|
312
|
+
if not query:
|
|
313
|
+
console.print("[bold cyan]Common codes:[/bold cyan] Fa (father), Mo (mother), FaFa (paternal grandfather), MoMo (maternal grandmother), FaElBr (elder paternal uncle), YoBr (younger brother), So (son), DaHu (son-in-law)")
|
|
314
|
+
query = "FaElBr"
|
|
315
|
+
res = KinshipEngine.resolve(query)
|
|
316
|
+
if not res:
|
|
317
|
+
console.print(f"[bold red]Could not resolve code:[/bold red] '{query}'")
|
|
318
|
+
return
|
|
319
|
+
explanation = KinshipEngine.explain_relationship(query)
|
|
320
|
+
console.print(Panel(explanation, title=f"👨👩👧👦 Kinship Term: {res['khamyang_roman']}", border_style="cyan"))
|
|
321
|
+
|
|
322
|
+
elif action == "search":
|
|
323
|
+
if not query:
|
|
324
|
+
console.print("[bold red]Please provide a search term (e.g. 'uncle', 'grandmother', 'lung1').[/bold red]")
|
|
325
|
+
return
|
|
326
|
+
matches = KinshipEngine.lookup(query)
|
|
327
|
+
if not matches:
|
|
328
|
+
console.print(f"[bold red]No kinship terms matching:[/bold red] '{query}'")
|
|
329
|
+
return
|
|
330
|
+
table = Table(title=f"Kinship Search Results for '{query}'")
|
|
331
|
+
table.add_column("Roman", style="bold cyan")
|
|
332
|
+
table.add_column("Lik Tai", style="yellow")
|
|
333
|
+
table.add_column("Assamese", style="magenta")
|
|
334
|
+
table.add_column("English", style="green")
|
|
335
|
+
table.add_column("Generation", style="blue")
|
|
336
|
+
for m in matches:
|
|
337
|
+
table.add_row(m["khamyang_roman"], m["lik_tai"], m["assamese_script"], m["english"], str(m["generation"]))
|
|
338
|
+
console.print(table)
|
|
339
|
+
|
|
340
|
+
elif action == "generation":
|
|
341
|
+
gen = int(query) if query.lstrip("-").isdigit() else 1
|
|
342
|
+
terms = KinshipEngine.list_by_generation(gen)
|
|
343
|
+
table = Table(title=f"Generation {gen} Relatives ({'Ancestors' if gen > 0 else 'Descendants' if gen < 0 else 'Ego Generation'})")
|
|
344
|
+
table.add_column("Roman", style="bold cyan")
|
|
345
|
+
table.add_column("Lik Tai", style="yellow")
|
|
346
|
+
table.add_column("English", style="green")
|
|
347
|
+
table.add_column("Assamese", style="magenta")
|
|
348
|
+
table.add_column("Relation Codes", style="dim")
|
|
349
|
+
for t in terms:
|
|
350
|
+
table.add_row(t["khamyang_roman"], t["lik_tai"], t["english"], t["assamese"], ", ".join(t.get("relation_codes", [])))
|
|
351
|
+
console.print(table)
|
|
352
|
+
|
|
353
|
+
elif action == "clans":
|
|
354
|
+
clans = KinshipEngine.list_clans()
|
|
355
|
+
table = Table(title="Tai Khamyang Historical Clans (Khɯa1)")
|
|
356
|
+
table.add_column("Clan Name", style="bold yellow")
|
|
357
|
+
table.add_column("Lik Tai", style="bold cyan")
|
|
358
|
+
table.add_column("Meaning", style="bold green")
|
|
359
|
+
table.add_column("Traditional Hereditary Role", style="white")
|
|
360
|
+
for c in clans:
|
|
361
|
+
table.add_row(c["name_tai"], c["lik_tai"], c["meaning"], c["role"])
|
|
362
|
+
console.print(table)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
@cli.command("biblio")
|
|
366
|
+
@click.argument("action", default="list", type=click.Choice(["list", "search", "bibtex", "markdown"]))
|
|
367
|
+
@click.argument("query", required=False, default="")
|
|
368
|
+
@click.option("--output", "-o", default="", help="Optional output file destination")
|
|
369
|
+
def biblio_cmd(action: str, query: str, output: str):
|
|
370
|
+
"""Browse scholarly literature, books, articles, and export BibTeX citations."""
|
|
371
|
+
from khamyang.scraper.downloader import BibliographyManager
|
|
372
|
+
|
|
373
|
+
if action == "list":
|
|
374
|
+
entries = BibliographyManager.all_entries()
|
|
375
|
+
table = Table(title="Tai Khamyang Scholarly Bibliography & Resource Archive")
|
|
376
|
+
table.add_column("ID", style="bold cyan")
|
|
377
|
+
table.add_column("Year", style="bold yellow")
|
|
378
|
+
table.add_column("Type", style="magenta")
|
|
379
|
+
table.add_column("Author(s)", style="green")
|
|
380
|
+
table.add_column("Title", style="white")
|
|
381
|
+
|
|
382
|
+
for e in entries:
|
|
383
|
+
authors = ", ".join(e.get("authors", []))
|
|
384
|
+
table.add_row(e["id"], str(e.get("year", "")), e.get("entry_type", ""), authors[:28], e["title"][:48])
|
|
385
|
+
console.print(table)
|
|
386
|
+
|
|
387
|
+
elif action == "search":
|
|
388
|
+
if not query:
|
|
389
|
+
console.print("[bold red]Please enter search term (e.g. 'Morey', 'songs', 'tones', 'Nath').[/bold red]")
|
|
390
|
+
return
|
|
391
|
+
matches = BibliographyManager.search(query)
|
|
392
|
+
if not matches:
|
|
393
|
+
console.print(f"[bold red]No entries found matching:[/bold red] '{query}'")
|
|
394
|
+
return
|
|
395
|
+
for m in matches:
|
|
396
|
+
apa = BibliographyManager.format_apa(m)
|
|
397
|
+
console.print(f"• [bold cyan]{m['id']}[/bold cyan] ({m['entry_type']}):\n {apa}\n [dim]{m.get('description', '')}[/dim]\n")
|
|
398
|
+
|
|
399
|
+
elif action == "bibtex":
|
|
400
|
+
content = BibliographyManager.export_all_bibtex()
|
|
401
|
+
if output:
|
|
402
|
+
with open(output, "w", encoding="utf-8") as f:
|
|
403
|
+
f.write(content)
|
|
404
|
+
console.print(f"[bold green]BibTeX citations successfully saved to:[/bold green] {output}")
|
|
405
|
+
else:
|
|
406
|
+
console.print(content)
|
|
407
|
+
|
|
408
|
+
elif action == "markdown":
|
|
409
|
+
out = output or "BIBLIOGRAPHY.md"
|
|
410
|
+
BibliographyManager.save_markdown_file(out)
|
|
411
|
+
console.print(f"[bold green]Markdown bibliography saved to:[/bold green] {out}")
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
if __name__ == "__main__":
|
|
415
|
+
cli()
|
|
416
|
+
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Core module initialization."""
|
|
2
|
+
|
|
3
|
+
from khamyang.core.models import (
|
|
4
|
+
Word,
|
|
5
|
+
Phrase,
|
|
6
|
+
ExampleSentence,
|
|
7
|
+
InterlinearUnit,
|
|
8
|
+
FolkText,
|
|
9
|
+
GlossUnit,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"Word",
|
|
14
|
+
"Phrase",
|
|
15
|
+
"ExampleSentence",
|
|
16
|
+
"InterlinearUnit",
|
|
17
|
+
"FolkText",
|
|
18
|
+
"GlossUnit",
|
|
19
|
+
]
|
khamyang/core/models.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Core data models for Tai Khamyang language entities."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field, asdict
|
|
4
|
+
from typing import List, Dict, Optional, Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass
|
|
8
|
+
class ExampleSentence:
|
|
9
|
+
"""An example sentence demonstrating word usage."""
|
|
10
|
+
khamyang_roman: str
|
|
11
|
+
lik_tai: str = ""
|
|
12
|
+
assamese_script: str = ""
|
|
13
|
+
english: str = ""
|
|
14
|
+
assamese_meaning: str = ""
|
|
15
|
+
literal_gloss: str = ""
|
|
16
|
+
|
|
17
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
18
|
+
return asdict(self)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Word:
|
|
23
|
+
"""Lexical item in Tai Khamyang."""
|
|
24
|
+
id: str
|
|
25
|
+
roman: str
|
|
26
|
+
gloss_en: str
|
|
27
|
+
pos: str
|
|
28
|
+
tone: int
|
|
29
|
+
lik_tai: str = ""
|
|
30
|
+
assamese: str = ""
|
|
31
|
+
ipa: str = ""
|
|
32
|
+
definition_en: str = ""
|
|
33
|
+
gloss_as: str = ""
|
|
34
|
+
definition_as: str = ""
|
|
35
|
+
category: str = "general"
|
|
36
|
+
gedney_box: str = ""
|
|
37
|
+
cognates: Dict[str, str] = field(default_factory=dict)
|
|
38
|
+
examples: List[ExampleSentence] = field(default_factory=list)
|
|
39
|
+
notes: str = ""
|
|
40
|
+
|
|
41
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
42
|
+
res = asdict(self)
|
|
43
|
+
res["examples"] = [ex.to_dict() if isinstance(ex, ExampleSentence) else ex for ex in self.examples]
|
|
44
|
+
return res
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def from_dict(cls, data: Dict[str, Any]) -> "Word":
|
|
48
|
+
examples = []
|
|
49
|
+
for ex in data.get("examples", []):
|
|
50
|
+
if isinstance(ex, dict):
|
|
51
|
+
examples.append(ExampleSentence(**ex))
|
|
52
|
+
elif isinstance(ex, ExampleSentence):
|
|
53
|
+
examples.append(ex)
|
|
54
|
+
data_copy = dict(data)
|
|
55
|
+
data_copy["examples"] = examples
|
|
56
|
+
return cls(**data_copy)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Phrase:
|
|
61
|
+
"""Conversational phrase or fieldwork elicitation expression."""
|
|
62
|
+
id: str
|
|
63
|
+
roman: str
|
|
64
|
+
english: str
|
|
65
|
+
category: str
|
|
66
|
+
lik_tai: str = ""
|
|
67
|
+
assamese: str = ""
|
|
68
|
+
ipa: str = ""
|
|
69
|
+
assamese_trans: str = ""
|
|
70
|
+
literal: str = ""
|
|
71
|
+
context: str = ""
|
|
72
|
+
|
|
73
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
74
|
+
return asdict(self)
|
|
75
|
+
|
|
76
|
+
@classmethod
|
|
77
|
+
def from_dict(cls, data: Dict[str, Any]) -> "Phrase":
|
|
78
|
+
return cls(**data)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass
|
|
82
|
+
class GlossUnit:
|
|
83
|
+
"""Morpheme and its linguistic gloss."""
|
|
84
|
+
morpheme: str
|
|
85
|
+
gloss: str
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class InterlinearUnit:
|
|
90
|
+
"""A sentence with interlinear glossing for linguistic research."""
|
|
91
|
+
id: str
|
|
92
|
+
text_roman: str
|
|
93
|
+
lik_tai: str
|
|
94
|
+
assamese_script: str
|
|
95
|
+
morphemes: List[str]
|
|
96
|
+
glosses: List[str]
|
|
97
|
+
free_translation_en: str
|
|
98
|
+
free_translation_as: str
|
|
99
|
+
speaker: str = "Traditional / Elder"
|
|
100
|
+
location: str = "Powaimukh, Assam"
|
|
101
|
+
audio_ref: str = ""
|
|
102
|
+
notes: str = ""
|
|
103
|
+
|
|
104
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
105
|
+
return asdict(self)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass
|
|
109
|
+
class FolkText:
|
|
110
|
+
"""A cultural narrative, folktale, or chant in Khamyang."""
|
|
111
|
+
id: str
|
|
112
|
+
title_khamyang: str
|
|
113
|
+
title_english: str
|
|
114
|
+
genre: str # folktale, migration_history, buddhist_chant, oral_history
|
|
115
|
+
informant: str
|
|
116
|
+
village: str
|
|
117
|
+
sentences: List[InterlinearUnit] = field(default_factory=list)
|
|
118
|
+
description: str = ""
|
|
119
|
+
|
|
120
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
121
|
+
return {
|
|
122
|
+
"id": self.id,
|
|
123
|
+
"title_khamyang": self.title_khamyang,
|
|
124
|
+
"title_english": self.title_english,
|
|
125
|
+
"genre": self.genre,
|
|
126
|
+
"informant": self.informant,
|
|
127
|
+
"village": self.village,
|
|
128
|
+
"description": self.description,
|
|
129
|
+
"sentences": [s.to_dict() for s in self.sentences],
|
|
130
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
{"id": "ai_inst_001", "task_type": "translation", "source_type": "human_verified", "data_level": "level_1_verified", "instruction": "Translate the following Tai Khamyang sentence into English and provide the grammatical breakdown.", "input": "pu5 yaa3 haw1 khao5 maa1 thaang1 doi1 pat4 kai1", "output": "Translation: 'Our ancestors crossed over and came along the path of the Patkai mountains.'\nMorphemes: pu5-yaa3 (ancestor/grandparents) haw1 (1PL/our) khao5 (3PL) maa1 (come) thaang1 (path/way) doi1 (mountain) pat4-kai1 (Patkai). Lik Tai: ပူႈယႃႈႁဝ်းၶဝ်မႃးထၢင်ႇလွႆပႅတ်ႇၵၢႆး."}
|
|
2
|
+
{"id": "ai_inst_002", "task_type": "kinship_resolution", "source_type": "human_verified", "data_level": "level_1_verified", "instruction": "Identify the Tai Khamyang kinship term for a father's elder brother and explain the cultural address etiquette.", "input": "Relationship: Father's Elder Brother (FaElBr)", "output": "The Tai Khamyang term is 'lung1' (Lik Tai: လုင်, Assamese: লুং¹, IPA: /luŋ³³/ - Tone 1 Mid Level). He is addressed directly as 'Lung1'. In traditional Powaimukh village hierarchy, the lung1 holds the highest ceremonial authority in the patrilineage after the paternal grandfather."}
|
|
3
|
+
{"id": "ai_inst_003", "task_type": "tone_identification", "source_type": "human_verified", "data_level": "level_1_verified", "instruction": "Analyze the tonal value, pitch contour, and Gedney tone box assignment for the Tai Khamyang word 'nam6' (water).", "input": "Word: nam6 (water / ၼမ်ႉ)", "output": "Word: nam6 (Lik Tai: ၼမ်ႉ, Assamese: নাম⁶).\nTone Number: Tone 6.\nPitch Contour: Level Rising (Chao pitch 35, IPA [nam˧˥]).\nGedney Tone Box: A2 (Proto-Tai voiced initial Proto-*n- with Proto-Tai tone *A).\nPhonetic note: In Tai Phake this cognate is pronounced with Tone 4 (nam4), showing regular Tai tone correspondence."}
|
|
4
|
+
{"id": "ai_inst_004", "task_type": "cultural_qa", "source_type": "human_verified", "data_level": "level_1_verified", "instruction": "What is the historical meaning and geographical origin of the Tai Khamyang village name Powaimukh?", "input": "Topic: Etymology of Powaimukh", "output": "Powaimukh is the Assamese rendition of the authentic Tai Khamyang name 'Maan3 Paa1 Waai6' (Lik Tai: မၢၼ်ႈပႃႉဝၢႆး). It is composed of three morphemes: maan3 ('village') + paa1 ('estuary/river mouth') + waai6 ('rattan cane / Calamus rotang'). It literally means 'Village at the River-Mouth of Rattan', reflecting the abundance of wild rattan along the Buridihing river banks when the Nora ancestors settled there."}
|
|
5
|
+
{"id": "ai_inst_005", "task_type": "classifier_syntax", "source_type": "human_verified", "data_level": "level_1_verified", "instruction": "Form a quantified nominal phrase in Tai Khamyang for 'two water buffaloes' following Southwestern Tai syntax.", "input": "Noun: kwaai1 (buffalo), Quantity: 2 (sɔŋ1)", "output": "In Tai Khamyang, numeral quantification strictly follows the Noun + Numeral + Classifier order: 'kwaai1 sɔŋ1 to1' (Lik Tai: ၵႂၢႆးသွင်တူဝ်). The classifier for animals and four-legged creatures is 'to1'."}
|