champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
package/lib/scripts.js
ADDED
|
@@ -0,0 +1,994 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Script conversion registry — deterministic orthography converters.
|
|
3
|
+
*
|
|
4
|
+
* WHY: Some languages have multiple scripts for the same spoken language.
|
|
5
|
+
* Translation workflows often prefer a "working script" (easier to type,
|
|
6
|
+
* edit, and version-control) that gets converted to a "display script"
|
|
7
|
+
* as a post-translation step.
|
|
8
|
+
*
|
|
9
|
+
* Examples:
|
|
10
|
+
* - Plains Cree: SRO (Standard Roman Orthography) → Syllabics (ᓀᐦᐃᔭᐍᐏᐣ)
|
|
11
|
+
* - Serbian: Latin → Cyrillic
|
|
12
|
+
* - Japanese: Romaji → Hiragana/Katakana
|
|
13
|
+
* - Hindi: Romanized → Devanagari
|
|
14
|
+
*
|
|
15
|
+
* All converters here are DETERMINISTIC — no LLM needed, pure lookup tables.
|
|
16
|
+
* They run as a post-translation hook: translate in working script, then
|
|
17
|
+
* convert to display script.
|
|
18
|
+
*
|
|
19
|
+
* ADDING A NEW CONVERTER:
|
|
20
|
+
* 1. Add the conversion map below
|
|
21
|
+
* 2. Create the converter function (input string → output string)
|
|
22
|
+
* 3. Register it in SCRIPT_CONVERTERS with the locale code
|
|
23
|
+
* 4. Add the `scripts` field to the language's register entry in registers.js
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
// -----------------------------------------------------------------
|
|
27
|
+
// Plains Cree: SRO → Syllabics
|
|
28
|
+
// -----------------------------------------------------------------
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* SRO to Cree Syllabics conversion table.
|
|
32
|
+
*
|
|
33
|
+
* This is the standard mapping used by the University of Alberta's
|
|
34
|
+
* ALTLab and documented in Wolvengrey's Cree: Words dictionary.
|
|
35
|
+
*
|
|
36
|
+
* The mapping is context-sensitive: consonant+vowel combinations map
|
|
37
|
+
* to specific syllabic characters, while standalone consonants use
|
|
38
|
+
* finals (small superscript forms).
|
|
39
|
+
*
|
|
40
|
+
* ORDER MATTERS: Longer sequences must be matched before shorter ones
|
|
41
|
+
* (e.g., "twê" before "tw" before "t").
|
|
42
|
+
*/
|
|
43
|
+
const SRO_TO_SYLLABICS_MAP = [
|
|
44
|
+
// Long vowels with w-glide (must come before short vowel w-glide)
|
|
45
|
+
['pwê', 'ᐻ'], ['pwî', 'ᐽ'], ['pwô', 'ᐿ'], ['pwâ', 'ᑁ'],
|
|
46
|
+
['twê', 'ᑗ'], ['twî', 'ᑙ'], ['twô', 'ᑛ'], ['twâ', 'ᑝ'],
|
|
47
|
+
['kwê', 'ᑵ'], ['kwî', 'ᑷ'], ['kwô', 'ᑹ'], ['kwâ', 'ᑻ'],
|
|
48
|
+
['cwê', 'ᒑ'], ['cwî', 'ᒓ'], ['cwô', 'ᒕ'], ['cwâ', 'ᒗ'],
|
|
49
|
+
['mwê', 'ᒫ'], ['mwî', 'ᒭ'], ['mwô', 'ᒯ'], ['mwâ', 'ᒱ'],
|
|
50
|
+
['nwê', 'ᓇ'], ['nwî', 'ᓉ'], ['nwô', 'ᓋ'], ['nwâ', 'ᓍ'],
|
|
51
|
+
['swê', 'ᓭ'], ['swî', 'ᓯ'], ['swô', 'ᓱ'], ['swâ', 'ᓳ'],
|
|
52
|
+
['ywê', 'ᔋ'], ['ywî', 'ᔍ'], ['ywô', 'ᔏ'], ['ywâ', 'ᔑ'],
|
|
53
|
+
|
|
54
|
+
// Short vowels with w-glide
|
|
55
|
+
['pwe', 'ᐺ'], ['pwi', 'ᐼ'], ['pwo', 'ᐾ'], ['pwa', 'ᑀ'],
|
|
56
|
+
['twe', 'ᑖ'], ['twi', 'ᑘ'], ['two', 'ᑚ'], ['twa', 'ᑜ'],
|
|
57
|
+
['kwe', 'ᑴ'], ['kwi', 'ᑶ'], ['kwo', 'ᑸ'], ['kwa', 'ᑺ'],
|
|
58
|
+
['cwe', 'ᒐ'], ['cwi', 'ᒒ'], ['cwo', 'ᒔ'], ['cwa', 'ᒖ'],
|
|
59
|
+
['mwe', 'ᒪ'], ['mwi', 'ᒬ'], ['mwo', 'ᒮ'], ['mwa', 'ᒰ'],
|
|
60
|
+
['nwe', 'ᓈ'], ['nwi', 'ᓊ'], ['nwo', 'ᓌ'], ['nwa', 'ᓎ'],
|
|
61
|
+
['swe', 'ᓬ'], ['swi', 'ᓮ'], ['swo', 'ᓰ'], ['swa', 'ᓲ'],
|
|
62
|
+
['ywe', 'ᔊ'], ['ywi', 'ᔌ'], ['ywo', 'ᔎ'], ['ywa', 'ᔐ'],
|
|
63
|
+
|
|
64
|
+
// Long vowels (macron forms — these must come before short vowels)
|
|
65
|
+
['pê', 'ᐯ'], ['pî', 'ᐲ'], ['pô', 'ᐴ'], ['pâ', 'ᐹ'],
|
|
66
|
+
['tê', 'ᑌ'], ['tî', 'ᑏ'], ['tô', 'ᑑ'], ['tâ', 'ᑖ'],
|
|
67
|
+
['kê', 'ᑫ'], ['kî', 'ᑮ'], ['kô', 'ᑰ'], ['kâ', 'ᑳ'],
|
|
68
|
+
['cê', 'ᒉ'], ['cî', 'ᒌ'], ['cô', 'ᒎ'], ['câ', 'ᒑ'],
|
|
69
|
+
['mê', 'ᒣ'], ['mî', 'ᒦ'], ['mô', 'ᒨ'], ['mâ', 'ᒫ'],
|
|
70
|
+
['nê', 'ᓀ'], ['nî', 'ᓃ'], ['nô', 'ᓅ'], ['nâ', 'ᓈ'],
|
|
71
|
+
['sê', 'ᓭ'], ['sî', 'ᓰ'], ['sô', 'ᓲ'], ['sâ', 'ᓵ'],
|
|
72
|
+
['yê', 'ᔦ'], ['yî', 'ᔩ'], ['yô', 'ᔫ'], ['yâ', 'ᔮ'],
|
|
73
|
+
|
|
74
|
+
// Short vowels (consonant+vowel)
|
|
75
|
+
['pe', 'ᐯ'], ['pi', 'ᐱ'], ['po', 'ᐳ'], ['pa', 'ᐸ'],
|
|
76
|
+
['te', 'ᑌ'], ['ti', 'ᑎ'], ['to', 'ᑐ'], ['ta', 'ᑕ'],
|
|
77
|
+
['ke', 'ᑫ'], ['ki', 'ᑭ'], ['ko', 'ᑯ'], ['ka', 'ᑲ'],
|
|
78
|
+
['ce', 'ᒉ'], ['ci', 'ᒋ'], ['co', 'ᒍ'], ['ca', 'ᒐ'],
|
|
79
|
+
['me', 'ᒣ'], ['mi', 'ᒥ'], ['mo', 'ᒧ'], ['ma', 'ᒪ'],
|
|
80
|
+
['ne', 'ᓀ'], ['ni', 'ᓂ'], ['no', 'ᓄ'], ['na', 'ᓇ'],
|
|
81
|
+
['se', 'ᓭ'], ['si', 'ᓯ'], ['so', 'ᓱ'], ['sa', 'ᓴ'],
|
|
82
|
+
['ye', 'ᔦ'], ['yi', 'ᔨ'], ['yo', 'ᔪ'], ['ya', 'ᔭ'],
|
|
83
|
+
|
|
84
|
+
// Standalone vowels (long first)
|
|
85
|
+
['ê', 'ᐁ'], ['î', 'ᐄ'], ['ô', 'ᐆ'], ['â', 'ᐋ'],
|
|
86
|
+
['e', 'ᐁ'], ['i', 'ᐃ'], ['o', 'ᐅ'], ['a', 'ᐊ'],
|
|
87
|
+
|
|
88
|
+
// Digraphs (must come before single-char finals)
|
|
89
|
+
['th', 'ᖧ'],
|
|
90
|
+
|
|
91
|
+
// Finals (standalone consonants — no following vowel)
|
|
92
|
+
['p', 'ᑊ'], ['t', 'ᐟ'], ['k', 'ᐠ'], ['c', 'ᐨ'],
|
|
93
|
+
['m', 'ᒼ'], ['n', 'ᐣ'], ['s', 'ᐢ'], ['y', 'ᐩ'],
|
|
94
|
+
|
|
95
|
+
// Special characters
|
|
96
|
+
['h', 'ᐦ'], ['w', 'ᐤ'], ['l', 'ᓬ'], ['r', 'ᕒ'],
|
|
97
|
+
];
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Convert SRO text to Cree Syllabics.
|
|
101
|
+
*
|
|
102
|
+
* This is a greedy left-to-right scan: at each position, try the longest
|
|
103
|
+
* possible match first. Characters that don't match any pattern (spaces,
|
|
104
|
+
* punctuation, numbers) pass through unchanged.
|
|
105
|
+
*
|
|
106
|
+
* @param {string} sro - SRO text to convert
|
|
107
|
+
* @returns {string} Syllabics text
|
|
108
|
+
*/
|
|
109
|
+
function sroToSyllabics(sro) {
|
|
110
|
+
const input = sro.toLowerCase();
|
|
111
|
+
let result = '';
|
|
112
|
+
let i = 0;
|
|
113
|
+
|
|
114
|
+
while (i < input.length) {
|
|
115
|
+
let matched = false;
|
|
116
|
+
|
|
117
|
+
// Try longest matches first (up to 3 characters)
|
|
118
|
+
for (const [from, to] of SRO_TO_SYLLABICS_MAP) {
|
|
119
|
+
if (input.startsWith(from, i)) {
|
|
120
|
+
result += to;
|
|
121
|
+
i += from.length;
|
|
122
|
+
matched = true;
|
|
123
|
+
break;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
// No match — pass character through (space, punctuation, etc.)
|
|
128
|
+
if (!matched) {
|
|
129
|
+
result += input[i];
|
|
130
|
+
i++;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
return result;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// -----------------------------------------------------------------
|
|
138
|
+
// Serbian: Latin → Cyrillic
|
|
139
|
+
// -----------------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
const LATIN_TO_CYRILLIC_SR = {
|
|
142
|
+
'lj': 'љ', 'nj': 'њ', 'dž': 'џ',
|
|
143
|
+
'Lj': 'Љ', 'Nj': 'Њ', 'Dž': 'Џ',
|
|
144
|
+
'LJ': 'Љ', 'NJ': 'Њ', 'DŽ': 'Џ',
|
|
145
|
+
'a': 'а', 'b': 'б', 'v': 'в', 'g': 'г', 'd': 'д',
|
|
146
|
+
'đ': 'ђ', 'e': 'е', 'ž': 'ж', 'z': 'з', 'i': 'и',
|
|
147
|
+
'j': 'ј', 'k': 'к', 'l': 'л', 'm': 'м', 'n': 'н',
|
|
148
|
+
'o': 'о', 'p': 'п', 'r': 'р', 's': 'с', 't': 'т',
|
|
149
|
+
'ć': 'ћ', 'u': 'у', 'f': 'ф', 'h': 'х', 'c': 'ц',
|
|
150
|
+
'č': 'ч', 'š': 'ш',
|
|
151
|
+
'A': 'А', 'B': 'Б', 'V': 'В', 'G': 'Г', 'D': 'Д',
|
|
152
|
+
'Đ': 'Ђ', 'E': 'Е', 'Ž': 'Ж', 'Z': 'З', 'I': 'И',
|
|
153
|
+
'J': 'Ј', 'K': 'К', 'L': 'Л', 'M': 'М', 'N': 'Н',
|
|
154
|
+
'O': 'О', 'P': 'П', 'R': 'Р', 'S': 'С', 'T': 'Т',
|
|
155
|
+
'Ć': 'Ћ', 'U': 'У', 'F': 'Ф', 'H': 'Х', 'C': 'Ц',
|
|
156
|
+
'Č': 'Ч', 'Š': 'Ш',
|
|
157
|
+
};
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Convert Serbian Latin text to Cyrillic.
|
|
161
|
+
* Digraphs (lj, nj, dž) are matched first.
|
|
162
|
+
*
|
|
163
|
+
* @param {string} latin - Latin text
|
|
164
|
+
* @returns {string} Cyrillic text
|
|
165
|
+
*/
|
|
166
|
+
function latinToCyrillicSr(latin) {
|
|
167
|
+
let result = '';
|
|
168
|
+
let i = 0;
|
|
169
|
+
|
|
170
|
+
while (i < latin.length) {
|
|
171
|
+
// Try digraphs first (2 characters)
|
|
172
|
+
if (i + 1 < latin.length) {
|
|
173
|
+
const digraph = latin.slice(i, i + 2);
|
|
174
|
+
if (LATIN_TO_CYRILLIC_SR[digraph]) {
|
|
175
|
+
result += LATIN_TO_CYRILLIC_SR[digraph];
|
|
176
|
+
i += 2;
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Single character
|
|
182
|
+
const ch = latin[i];
|
|
183
|
+
result += LATIN_TO_CYRILLIC_SR[ch] || ch;
|
|
184
|
+
i++;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
return result;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
// -----------------------------------------------------------------
|
|
191
|
+
// Klingon: Romanization → pIqaD (CSUR PUA U+F8D0–F8FF)
|
|
192
|
+
// -----------------------------------------------------------------
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Klingon romanization to pIqaD conversion table.
|
|
196
|
+
*
|
|
197
|
+
* Based on the ConScript Unicode Registry (CSUR) mapping maintained
|
|
198
|
+
* at evertype.com. Characters are in the Unicode Private Use Area
|
|
199
|
+
* — they require a pIqaD-compatible web font to render visually.
|
|
200
|
+
*
|
|
201
|
+
* Klingon romanization is case-sensitive: 'D' ≠ 'd', 'S' ≠ 's',
|
|
202
|
+
* 'I' ≠ 'i', 'Q' ≠ 'q'. The table preserves this distinction.
|
|
203
|
+
*
|
|
204
|
+
* ORDER: Trigraphs (tlh) → digraphs (ch, gh, ng) → single chars.
|
|
205
|
+
*/
|
|
206
|
+
const KLINGON_TO_PIQAD_MAP = [
|
|
207
|
+
// Trigraph (must come first)
|
|
208
|
+
['tlh', '\uF8E4'],
|
|
209
|
+
|
|
210
|
+
// Digraphs
|
|
211
|
+
['ch', '\uF8D2'], ['gh', '\uF8D5'], ['ng', '\uF8DC'],
|
|
212
|
+
|
|
213
|
+
// Case-sensitive single characters
|
|
214
|
+
// Uppercase-only letters (distinct phonemes in Klingon)
|
|
215
|
+
['D', '\uF8D3'], ['H', '\uF8D6'], ['I', '\uF8D7'],
|
|
216
|
+
['Q', '\uF8E0'], ['S', '\uF8E2'],
|
|
217
|
+
|
|
218
|
+
// Lowercase letters
|
|
219
|
+
['a', '\uF8D0'], ['b', '\uF8D1'], ['e', '\uF8D4'],
|
|
220
|
+
['j', '\uF8D8'], ['l', '\uF8D9'], ['m', '\uF8DA'],
|
|
221
|
+
['n', '\uF8DB'], ['o', '\uF8DD'], ['p', '\uF8DE'],
|
|
222
|
+
['q', '\uF8DF'], ['r', '\uF8E1'], ['t', '\uF8E3'],
|
|
223
|
+
['u', '\uF8E5'], ['v', '\uF8E6'], ['w', '\uF8E7'],
|
|
224
|
+
['y', '\uF8E8'],
|
|
225
|
+
|
|
226
|
+
// Glottal stop (apostrophe)
|
|
227
|
+
["'", '\uF8E9'],
|
|
228
|
+
['\u2019', '\uF8E9'], // right single quote (common in copy-pasted text)
|
|
229
|
+
];
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Convert Klingon romanization to pIqaD script.
|
|
233
|
+
*
|
|
234
|
+
* Greedy left-to-right scan, longest match first.
|
|
235
|
+
* Case-sensitive: 'D' (retroflex) ≠ 'd' (not a Klingon phoneme).
|
|
236
|
+
* Non-Klingon characters (spaces, punctuation, numbers) pass through.
|
|
237
|
+
*
|
|
238
|
+
* NOTE: Output uses Unicode PUA characters (U+F8D0–F8FF).
|
|
239
|
+
* A pIqaD web font (e.g., "pIqaD qolqoS" or "Klingon pIqaD HaSta")
|
|
240
|
+
* must be loaded for visual rendering.
|
|
241
|
+
*
|
|
242
|
+
* @param {string} romanized - Klingon text in standard romanization
|
|
243
|
+
* @returns {string} pIqaD text
|
|
244
|
+
*/
|
|
245
|
+
function romanizationToPiqad(romanized) {
|
|
246
|
+
let result = '';
|
|
247
|
+
let i = 0;
|
|
248
|
+
|
|
249
|
+
while (i < romanized.length) {
|
|
250
|
+
let matched = false;
|
|
251
|
+
|
|
252
|
+
for (const [from, to] of KLINGON_TO_PIQAD_MAP) {
|
|
253
|
+
if (romanized.startsWith(from, i)) {
|
|
254
|
+
result += to;
|
|
255
|
+
i += from.length;
|
|
256
|
+
matched = true;
|
|
257
|
+
break;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
if (!matched) {
|
|
262
|
+
result += romanized[i];
|
|
263
|
+
i++;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
return result;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// -----------------------------------------------------------------
|
|
271
|
+
// Tengwar: Sindarin Latin → Tengwar (CSUR PUA U+E000–E07F)
|
|
272
|
+
// Mode of Beleriand — full vowel letters (not diacritics)
|
|
273
|
+
// -----------------------------------------------------------------
|
|
274
|
+
|
|
275
|
+
/**
|
|
276
|
+
* Sindarin Latin to Tengwar conversion table.
|
|
277
|
+
*
|
|
278
|
+
* Uses the "Mode of Beleriand" where vowels are full tengwar letters
|
|
279
|
+
* rather than tehtar (diacritics). This is the most deterministic
|
|
280
|
+
* mode — the Ómatehtar mode requires context-dependent diacritic
|
|
281
|
+
* placement which is significantly more complex.
|
|
282
|
+
*
|
|
283
|
+
* Based on the CSUR Tengwar block (U+E000–E07F) as documented by
|
|
284
|
+
* the Free Tengwar Font Project. Requires a CSUR-compatible Tengwar
|
|
285
|
+
* font (e.g., "Tengwar Formal CSUR", "Tengwar Annatar") to render.
|
|
286
|
+
*
|
|
287
|
+
* This is a simplified converter — it handles the most common
|
|
288
|
+
* Sindarin consonants and vowels but does not implement:
|
|
289
|
+
* - Double consonant bars (nasal signs)
|
|
290
|
+
* - Sa-rincë (s-hooks)
|
|
291
|
+
* - Ligatures for common combinations
|
|
292
|
+
*
|
|
293
|
+
* ORDER: Digraphs → single characters.
|
|
294
|
+
*/
|
|
295
|
+
const SINDARIN_TO_TENGWAR_MAP = [
|
|
296
|
+
// Digraphs (must come before single chars)
|
|
297
|
+
['th', '\uE003'], // thúlë (voiceless th)
|
|
298
|
+
['dh', '\uE004'], // anto (voiced th/dh)
|
|
299
|
+
['ch', '\uE002'], // hwesta (voiceless velar fricative)
|
|
300
|
+
['ph', '\uE00E'], // formen (labialized)
|
|
301
|
+
['ng', '\uE016'], // noldo
|
|
302
|
+
['nd', '\uE022'], // ando+númen combo — using ando
|
|
303
|
+
['mb', '\uE022'], // umbar area
|
|
304
|
+
['nn', '\uE015'], // doubled númen
|
|
305
|
+
['mm', '\uE012'], // doubled malta
|
|
306
|
+
['ll', '\uE00B'], // doubled lambe
|
|
307
|
+
['rh', '\uE00C'], // rómen (voiceless r)
|
|
308
|
+
['lh', '\uE00D'], // silmë (voiceless l)
|
|
309
|
+
['hw', '\uE017'], // hwesta sindarinwa
|
|
310
|
+
|
|
311
|
+
// Consonants (single)
|
|
312
|
+
['t', '\uE001'], // tinco
|
|
313
|
+
['p', '\uE00E'], // parma
|
|
314
|
+
['c', '\uE002'], // calma (hard c/k)
|
|
315
|
+
['k', '\uE002'], // calma
|
|
316
|
+
['d', '\uE005'], // ando
|
|
317
|
+
['b', '\uE00F'], // umbar
|
|
318
|
+
['g', '\uE006'], // anga (hard g)
|
|
319
|
+
['f', '\uE010'], // formen
|
|
320
|
+
['v', '\uE011'], // ampa
|
|
321
|
+
['n', '\uE015'], // númen
|
|
322
|
+
['m', '\uE012'], // malta
|
|
323
|
+
['r', '\uE00C'], // óre/rómen
|
|
324
|
+
['l', '\uE00B'], // lambe
|
|
325
|
+
['s', '\uE008'], // silmë
|
|
326
|
+
['h', '\uE017'], // hyarmen
|
|
327
|
+
['w', '\uE013'], // vilya/vala
|
|
328
|
+
['y', '\uE014'], // anna
|
|
329
|
+
|
|
330
|
+
// Vowels — Mode of Beleriand uses full letters, not diacritics
|
|
331
|
+
// Long vowels (circumflex or macron) mapped to long carriers
|
|
332
|
+
['á', '\uE040'], // long a carrier
|
|
333
|
+
['é', '\uE042'], // long e carrier
|
|
334
|
+
['í', '\uE044'], // long i carrier
|
|
335
|
+
['ó', '\uE046'], // long o carrier
|
|
336
|
+
['ú', '\uE048'], // long u carrier
|
|
337
|
+
['â', '\uE040'],
|
|
338
|
+
['ê', '\uE042'],
|
|
339
|
+
['î', '\uE044'],
|
|
340
|
+
['ô', '\uE046'],
|
|
341
|
+
['û', '\uE048'],
|
|
342
|
+
|
|
343
|
+
// Short vowels
|
|
344
|
+
['a', '\uE03F'], // short a
|
|
345
|
+
['e', '\uE041'], // short e
|
|
346
|
+
['i', '\uE043'], // short i
|
|
347
|
+
['o', '\uE045'], // short o
|
|
348
|
+
['u', '\uE047'], // short u
|
|
349
|
+
];
|
|
350
|
+
|
|
351
|
+
/**
|
|
352
|
+
* Convert Sindarin Latin text to Tengwar script (Mode of Beleriand).
|
|
353
|
+
*
|
|
354
|
+
* NOTE: Output uses Unicode PUA characters (U+E000–E07F).
|
|
355
|
+
* A CSUR-compatible Tengwar font must be loaded for visual rendering.
|
|
356
|
+
*
|
|
357
|
+
* @param {string} latin - Sindarin text in Latin script
|
|
358
|
+
* @returns {string} Tengwar text
|
|
359
|
+
*/
|
|
360
|
+
function latinToTengwar(latin) {
|
|
361
|
+
const input = latin.toLowerCase();
|
|
362
|
+
let result = '';
|
|
363
|
+
let i = 0;
|
|
364
|
+
|
|
365
|
+
while (i < input.length) {
|
|
366
|
+
let matched = false;
|
|
367
|
+
|
|
368
|
+
for (const [from, to] of SINDARIN_TO_TENGWAR_MAP) {
|
|
369
|
+
if (input.startsWith(from, i)) {
|
|
370
|
+
result += to;
|
|
371
|
+
i += from.length;
|
|
372
|
+
matched = true;
|
|
373
|
+
break;
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
if (!matched) {
|
|
378
|
+
result += input[i];
|
|
379
|
+
i++;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
return result;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
// -----------------------------------------------------------------
|
|
387
|
+
// Kryptonian: Latin → Kryptonian (font-based cipher)
|
|
388
|
+
// -----------------------------------------------------------------
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* Kryptonian "script conversion" — 1:1 Latin alphabet cipher.
|
|
392
|
+
*
|
|
393
|
+
* Unlike the other converters, Kryptonian has NO standard Unicode
|
|
394
|
+
* assignment (not even PUA/CSUR). The DC Comics script is a pure
|
|
395
|
+
* substitution cipher of the Latin alphabet, rendered via custom fonts.
|
|
396
|
+
*
|
|
397
|
+
* This converter maps A-Z to Unicode PUA characters (U+E100–E119)
|
|
398
|
+
* using a conventional fan-community assignment. The mapping is:
|
|
399
|
+
* A=U+E100, B=U+E101, ..., Z=U+E119
|
|
400
|
+
*
|
|
401
|
+
* FONT REQUIRED: A Kryptonian font mapped to these PUA codepoints
|
|
402
|
+
* (e.g., "Kryptonian" from kryptonian.info). Without the font,
|
|
403
|
+
* output will render as empty boxes.
|
|
404
|
+
*
|
|
405
|
+
* Alternative approach: skip this converter entirely and use
|
|
406
|
+
* CSS `font-family: 'Kryptonian'` on the element. The text stays
|
|
407
|
+
* as Latin characters but renders in Kryptonian glyphs. This is
|
|
408
|
+
* often simpler for web deployments.
|
|
409
|
+
*/
|
|
410
|
+
function latinToKryptonian(text) {
|
|
411
|
+
let result = '';
|
|
412
|
+
|
|
413
|
+
for (const ch of text) {
|
|
414
|
+
const upper = ch.toUpperCase();
|
|
415
|
+
const code = upper.charCodeAt(0);
|
|
416
|
+
|
|
417
|
+
// Map A-Z (65-90) to PUA U+E100-E119
|
|
418
|
+
if (code >= 65 && code <= 90) {
|
|
419
|
+
result += String.fromCharCode(0xE100 + (code - 65));
|
|
420
|
+
} else {
|
|
421
|
+
// Non-alpha characters (spaces, punctuation, numbers) pass through
|
|
422
|
+
result += ch;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
return result;
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
// -----------------------------------------------------------------
|
|
430
|
+
// Converter Registry
|
|
431
|
+
// -----------------------------------------------------------------
|
|
432
|
+
|
|
433
|
+
/**
|
|
434
|
+
* Registry of available script converters.
|
|
435
|
+
*
|
|
436
|
+
* Each entry maps a locale code to its converter configuration:
|
|
437
|
+
* - from: source script name
|
|
438
|
+
* - to: target script name
|
|
439
|
+
* - fromScript ISO 15924 code of the working script the LLM emits
|
|
440
|
+
* - toScript: ISO 15924 code of the converted output, or null when the
|
|
441
|
+
* target script has no registered code (Kryptonian). A null
|
|
442
|
+
* toScript CANNOT be selected via `script:` config — see
|
|
443
|
+
* resolveTargetScript.
|
|
444
|
+
* - type: 'deterministic' (pure lookup), or 'font-based' (needs web font)
|
|
445
|
+
* - converter: function(string) → string
|
|
446
|
+
* - map: the forward table, used to derive letter coverage and the
|
|
447
|
+
* reverse table. Null for converters that are pure arithmetic.
|
|
448
|
+
* - puaRange: [lo, hi] of the Private Use Area block the converter emits,
|
|
449
|
+
* or null when output is in assigned Unicode.
|
|
450
|
+
* - fontNote: (optional) font requirement for PUA-based converters
|
|
451
|
+
*/
|
|
452
|
+
const SCRIPT_CONVERTERS = {
|
|
453
|
+
crk: {
|
|
454
|
+
from: 'SRO (Standard Roman Orthography)',
|
|
455
|
+
to: 'Cree Syllabics',
|
|
456
|
+
fromScript: 'Latn',
|
|
457
|
+
toScript: 'Cans',
|
|
458
|
+
type: 'deterministic',
|
|
459
|
+
map: SRO_TO_SYLLABICS_MAP,
|
|
460
|
+
puaRange: null,
|
|
461
|
+
converter: sroToSyllabics,
|
|
462
|
+
},
|
|
463
|
+
sr: {
|
|
464
|
+
from: 'Latin',
|
|
465
|
+
to: 'Cyrillic',
|
|
466
|
+
fromScript: 'Latn',
|
|
467
|
+
toScript: 'Cyrl',
|
|
468
|
+
type: 'deterministic',
|
|
469
|
+
map: Object.entries(LATIN_TO_CYRILLIC_SR),
|
|
470
|
+
puaRange: null,
|
|
471
|
+
converter: latinToCyrillicSr,
|
|
472
|
+
},
|
|
473
|
+
tlh: {
|
|
474
|
+
from: 'Romanization',
|
|
475
|
+
to: 'pIqaD',
|
|
476
|
+
fromScript: 'Latn',
|
|
477
|
+
toScript: 'Piqd',
|
|
478
|
+
type: 'deterministic',
|
|
479
|
+
map: KLINGON_TO_PIQAD_MAP,
|
|
480
|
+
puaRange: [0xF8D0, 0xF8FF],
|
|
481
|
+
fontNote: 'Requires pIqaD web font (PUA U+F8D0–F8FF)',
|
|
482
|
+
converter: romanizationToPiqad,
|
|
483
|
+
},
|
|
484
|
+
'x-elvish-s': {
|
|
485
|
+
from: 'Latin',
|
|
486
|
+
to: 'Tengwar (Mode of Beleriand)',
|
|
487
|
+
fromScript: 'Latn',
|
|
488
|
+
toScript: 'Teng',
|
|
489
|
+
type: 'deterministic',
|
|
490
|
+
map: SINDARIN_TO_TENGWAR_MAP,
|
|
491
|
+
puaRange: [0xE000, 0xE07F],
|
|
492
|
+
fontNote: 'Requires CSUR Tengwar font (PUA U+E000–E07F)',
|
|
493
|
+
converter: latinToTengwar,
|
|
494
|
+
},
|
|
495
|
+
'x-kryptonian': {
|
|
496
|
+
from: 'Latin',
|
|
497
|
+
to: 'Kryptonian',
|
|
498
|
+
fromScript: 'Latn',
|
|
499
|
+
// Kryptonian has no ISO 15924 code — not even a provisional one. The
|
|
500
|
+
// encoded form is therefore unreachable through `script:` and must be
|
|
501
|
+
// requested with the explicit escape hatch (see resolveTargetScript).
|
|
502
|
+
toScript: null,
|
|
503
|
+
type: 'font-based',
|
|
504
|
+
map: null,
|
|
505
|
+
puaRange: [0xE100, 0xE119],
|
|
506
|
+
fontNote: 'Requires Kryptonian font mapped to PUA U+E100–E119',
|
|
507
|
+
converter: latinToKryptonian,
|
|
508
|
+
},
|
|
509
|
+
};
|
|
510
|
+
|
|
511
|
+
// -----------------------------------------------------------------
|
|
512
|
+
// Target-script resolution
|
|
513
|
+
// -----------------------------------------------------------------
|
|
514
|
+
|
|
515
|
+
/**
|
|
516
|
+
* ISO 15924 codes are exactly four letters, initial capital: Latn, Cans, Cyrl.
|
|
517
|
+
*/
|
|
518
|
+
const ISO_15924 = /^[A-Z][a-z]{3}$/;
|
|
519
|
+
|
|
520
|
+
/**
|
|
521
|
+
* Legacy/informal script names that used to appear in `script:` config, mapped
|
|
522
|
+
* to the ISO 15924 code that replaces them. The field was accepted but never
|
|
523
|
+
* read, so no project can have been depending on the old values working —
|
|
524
|
+
* but they were documented, so name the replacement rather than just rejecting.
|
|
525
|
+
*/
|
|
526
|
+
const LEGACY_SCRIPT_ALIASES = {
|
|
527
|
+
syllabics: 'Cans',
|
|
528
|
+
latin: 'Latn',
|
|
529
|
+
roman: 'Latn',
|
|
530
|
+
cyrillic: 'Cyrl',
|
|
531
|
+
piqad: 'Piqd',
|
|
532
|
+
tengwar: 'Teng',
|
|
533
|
+
};
|
|
534
|
+
|
|
535
|
+
/**
|
|
536
|
+
* The explicit escape hatch for converters whose output script has no ISO
|
|
537
|
+
* 15924 code. Set `script: "x-kryptonian"` (the converter key) to opt in.
|
|
538
|
+
*/
|
|
539
|
+
function isConverterKeyEscapeHatch(value) {
|
|
540
|
+
return Object.prototype.hasOwnProperty.call(SCRIPT_CONVERTERS, value);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
/**
|
|
544
|
+
* The converter registered for a locale, resolved through the card.
|
|
545
|
+
*
|
|
546
|
+
* The registry is keyed by converter key, which is USUALLY the locale code but
|
|
547
|
+
* not always: Serbian's converter is keyed `sr` while the card's canonical
|
|
548
|
+
* code is `srp`. The card's `scriptConverter` field records the key, so the
|
|
549
|
+
* card is the join — never guess from the code alone.
|
|
550
|
+
*
|
|
551
|
+
* @param {string} localeCode - Target locale code
|
|
552
|
+
* @param {object|null} card - The language card, or null
|
|
553
|
+
* @returns {string|null} The converter key, or null when none is registered
|
|
554
|
+
*/
|
|
555
|
+
function converterKeyForLocale(localeCode, card = null) {
|
|
556
|
+
if (card?.scriptConverter && SCRIPT_CONVERTERS[card.scriptConverter]) {
|
|
557
|
+
return card.scriptConverter;
|
|
558
|
+
}
|
|
559
|
+
return SCRIPT_CONVERTERS[localeCode] ? localeCode : null;
|
|
560
|
+
}
|
|
561
|
+
|
|
562
|
+
/**
|
|
563
|
+
* Resolve which script a locale's output should be written in.
|
|
564
|
+
*
|
|
565
|
+
* WHY this exists: until 0.2.0 the decision was a bare lookup —
|
|
566
|
+
* `hasScriptConverter(target)` — so every project targeting crk/sr/tlh/
|
|
567
|
+
* x-elvish-s/x-kryptonian had its output rewritten into the converter's
|
|
568
|
+
* display script unconditionally, with no way to decline. For the PUA
|
|
569
|
+
* converters that shipped unrenderable text to anyone whose font was keyed
|
|
570
|
+
* to Latin transliteration rather than Private Use Area codepoints; for
|
|
571
|
+
* crk it silently chose a community's display orthography on their behalf.
|
|
572
|
+
*
|
|
573
|
+
* Resolution:
|
|
574
|
+
* 1. Explicit `script:` config (per-language or per-pair) — user intent
|
|
575
|
+
* wins. ISO 15924, any casing accepted ("cans" → "Cans"); a value the
|
|
576
|
+
* locale's converter cannot produce fails loud listing what it can.
|
|
577
|
+
* 2. No config, and the locale's converter targets a REAL Unicode script
|
|
578
|
+
* (puaRange null — crk → Cans, sr → Cyrl): `{ source: 'choice-required',
|
|
579
|
+
* choices }`. Both orthographies are legitimate; picking one is not a
|
|
580
|
+
* default we get to make. Translation lanes refuse to run until the
|
|
581
|
+
* config says which; read-only lanes (status, integrity, repair) may
|
|
582
|
+
* proceed and display the state.
|
|
583
|
+
* 3. No config, and the converter targets a PUA block (tlh, x-elvish-s,
|
|
584
|
+
* x-kryptonian — scripts NOT in Unicode): default to the working script
|
|
585
|
+
* (Latn romanization), the only output that renders without a custom
|
|
586
|
+
* font. Opting into the PUA form is one config line away.
|
|
587
|
+
* 4. No converter at all: `script` passes through informationally,
|
|
588
|
+
* converterKey null, zero behavior change.
|
|
589
|
+
*
|
|
590
|
+
* @param {string} localeCode - Target locale code
|
|
591
|
+
* @param {object} langConfig - Per-language / per-pair config (reads .script)
|
|
592
|
+
* @param {object|null} card - The language card, or null when none exists
|
|
593
|
+
* @returns {{ script: string|null, source: 'config'|'default'|'choice-required'|'none',
|
|
594
|
+
* converterKey: string|null, choices?: Array<{script: string, label: string}> }}
|
|
595
|
+
* @throws {Error} on an unusable `script:` value — fail loud, never guess
|
|
596
|
+
*/
|
|
597
|
+
function resolveTargetScript(localeCode, langConfig = {}, card = null) {
|
|
598
|
+
const registeredKey = converterKeyForLocale(localeCode, card);
|
|
599
|
+
const conv = registeredKey ? SCRIPT_CONVERTERS[registeredKey] : null;
|
|
600
|
+
const requested = langConfig?.script;
|
|
601
|
+
|
|
602
|
+
if (requested != null && requested !== '') {
|
|
603
|
+
if (typeof requested !== 'string') {
|
|
604
|
+
throw new Error(
|
|
605
|
+
`Invalid "script" for ${localeCode}: expected an ISO 15924 code string, got ${typeof requested}.`
|
|
606
|
+
);
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
// The escape hatch for scripts with no ISO 15924 code (Kryptonian).
|
|
610
|
+
// Only valid on the locale that owns the converter — running the
|
|
611
|
+
// Kryptonian cipher on some other language is never what anyone meant.
|
|
612
|
+
if (isConverterKeyEscapeHatch(requested) && SCRIPT_CONVERTERS[requested].toScript === null) {
|
|
613
|
+
if (requested !== registeredKey) {
|
|
614
|
+
throw new Error(
|
|
615
|
+
`Invalid "script" for ${localeCode}: "${requested}" is the ${SCRIPT_CONVERTERS[requested].to} ` +
|
|
616
|
+
`converter, which belongs to the ${requested} locale, not ${localeCode}.`
|
|
617
|
+
);
|
|
618
|
+
}
|
|
619
|
+
return { script: null, source: 'config', converterKey: requested };
|
|
620
|
+
}
|
|
621
|
+
|
|
622
|
+
const alias = LEGACY_SCRIPT_ALIASES[requested.toLowerCase()];
|
|
623
|
+
if (alias && !ISO_15924.test(requested)) {
|
|
624
|
+
throw new Error(
|
|
625
|
+
`Invalid "script" for ${localeCode}: "${requested}" is not an ISO 15924 code. Use "${alias}".`
|
|
626
|
+
);
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
// Case-normalize a 4-letter value before validating: "cans", "PIQD" and
|
|
630
|
+
// "Cans" all mean the same code, and the project's own docs used the
|
|
631
|
+
// lowercase form for years — erroring on it would punish people for
|
|
632
|
+
// following us.
|
|
633
|
+
const normalized = /^[A-Za-z]{4}$/.test(requested)
|
|
634
|
+
? requested[0].toUpperCase() + requested.slice(1).toLowerCase()
|
|
635
|
+
: requested;
|
|
636
|
+
if (!ISO_15924.test(normalized)) {
|
|
637
|
+
throw new Error(
|
|
638
|
+
`Invalid "script" for ${localeCode}: "${requested}" is not an ISO 15924 code ` +
|
|
639
|
+
`(four letters — e.g. "Latn", "Cans", "Cyrl").`
|
|
640
|
+
);
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
// A locale WITH a converter can produce exactly two scripts: the working
|
|
644
|
+
// script and the converter's target. Anything else is a config mistake
|
|
645
|
+
// that would otherwise no-op silently — name what IS available.
|
|
646
|
+
if (conv && normalized !== conv.fromScript && normalized !== conv.toScript) {
|
|
647
|
+
const options = [`"${conv.fromScript}" (${conv.from})`, conv.toScript ? `"${conv.toScript}" (${conv.to})` : `"${registeredKey}" (${conv.to})`];
|
|
648
|
+
throw new Error(
|
|
649
|
+
`Invalid "script" for ${localeCode}: "${requested}" is not a script this locale can produce. ` +
|
|
650
|
+
`Available: ${options.join(' or ')}.`
|
|
651
|
+
);
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
return {
|
|
655
|
+
script: normalized,
|
|
656
|
+
source: 'config',
|
|
657
|
+
converterKey: conv && normalized === conv.toScript ? registeredKey : null,
|
|
658
|
+
};
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
// No explicit choice. What happens next depends on what KIND of converter
|
|
662
|
+
// this locale has — a real-Unicode orthography choice is the user's to
|
|
663
|
+
// make; an out-of-Unicode display encoding defaults safely off.
|
|
664
|
+
if (conv) {
|
|
665
|
+
if (conv.puaRange === null) {
|
|
666
|
+
return {
|
|
667
|
+
script: null,
|
|
668
|
+
source: 'choice-required',
|
|
669
|
+
converterKey: null,
|
|
670
|
+
choices: [
|
|
671
|
+
{ script: conv.fromScript, label: conv.from },
|
|
672
|
+
{ script: conv.toScript, label: conv.to },
|
|
673
|
+
],
|
|
674
|
+
};
|
|
675
|
+
}
|
|
676
|
+
return { script: conv.fromScript, source: 'default', converterKey: null };
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
// No converter — nothing to decide. An informational `script` from config
|
|
680
|
+
// never reaches here (handled above); absence means absence.
|
|
681
|
+
return { script: null, source: 'none', converterKey: null };
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
/**
|
|
685
|
+
* Format the choice-required error for a locale, shared by every lane that
|
|
686
|
+
* refuses to translate without the decision — one message, everywhere.
|
|
687
|
+
*
|
|
688
|
+
* @param {string} localeCode
|
|
689
|
+
* @param {{choices: Array<{script: string, label: string}>}} resolution
|
|
690
|
+
* @returns {string}
|
|
691
|
+
*/
|
|
692
|
+
function formatScriptChoiceError(localeCode, resolution) {
|
|
693
|
+
const opts = resolution.choices
|
|
694
|
+
.map(c => `"script": "${c.script}" (${c.label})`)
|
|
695
|
+
.join(' or ');
|
|
696
|
+
return (
|
|
697
|
+
`${localeCode} has more than one real orthography and Champollion will not pick one ` +
|
|
698
|
+
`for a community. Set ${opts} for ${localeCode} in champollion.config.json.`
|
|
699
|
+
);
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
// -----------------------------------------------------------------
|
|
703
|
+
// Transliteration fallbacks — user-declared rules for unmapped letters
|
|
704
|
+
// -----------------------------------------------------------------
|
|
705
|
+
|
|
706
|
+
/**
|
|
707
|
+
* Validate a `scriptFallback` map against a converter.
|
|
708
|
+
*
|
|
709
|
+
* Each entry maps a working-script sequence the converter does NOT cover to a
|
|
710
|
+
* replacement it DOES ("d" → "D", "c" → "ch" for Klingon). The replacement is
|
|
711
|
+
* itself converted through the normal table, so it must be fully mapped — a
|
|
712
|
+
* fallback that lands on another unmapped letter would just move the hole.
|
|
713
|
+
*
|
|
714
|
+
* Champollion ships NO fallbacks of its own: inventing orthographic
|
|
715
|
+
* adaptations — especially for a real language's orthography — is not ours to
|
|
716
|
+
* do. The docs list conventions with their sources; adopting one is a
|
|
717
|
+
* deliberate, per-project act.
|
|
718
|
+
*
|
|
719
|
+
* @param {object} fallbackMap - { sequence: replacement }
|
|
720
|
+
* @param {string} converterKey - Key into SCRIPT_CONVERTERS
|
|
721
|
+
* @throws {Error} naming the offending entry
|
|
722
|
+
*/
|
|
723
|
+
function validateScriptFallback(fallbackMap, converterKey) {
|
|
724
|
+
if (fallbackMap == null) return;
|
|
725
|
+
if (typeof fallbackMap !== 'object' || Array.isArray(fallbackMap)) {
|
|
726
|
+
throw new Error(
|
|
727
|
+
`"scriptFallback" must be an object mapping letters to replacements, got ${Array.isArray(fallbackMap) ? 'array' : typeof fallbackMap}.`
|
|
728
|
+
);
|
|
729
|
+
}
|
|
730
|
+
for (const [from, to] of Object.entries(fallbackMap)) {
|
|
731
|
+
if (typeof to !== 'string' || to === '') {
|
|
732
|
+
throw new Error(
|
|
733
|
+
`"scriptFallback" entry "${from}": replacement must be a non-empty string, got ${JSON.stringify(to)}.`
|
|
734
|
+
);
|
|
735
|
+
}
|
|
736
|
+
if (from === '') {
|
|
737
|
+
throw new Error('"scriptFallback" has an empty-string key — nothing to replace.');
|
|
738
|
+
}
|
|
739
|
+
const holes = unmappedLetters(to, converterKey);
|
|
740
|
+
if (holes.length > 0) {
|
|
741
|
+
throw new Error(
|
|
742
|
+
`"scriptFallback" entry "${from}" → "${to}": the replacement itself contains ` +
|
|
743
|
+
`letter(s) the ${converterKey} converter cannot map (${holes.join(', ')}). ` +
|
|
744
|
+
'A fallback must land on fully-mapped text.'
|
|
745
|
+
);
|
|
746
|
+
}
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
|
|
750
|
+
/**
|
|
751
|
+
* Apply a scriptFallback map to working-script text, longest keys first so
|
|
752
|
+
* "ck" wins over "c" + "k". Pure textual substitution — the result then runs
|
|
753
|
+
* through the normal conversion table.
|
|
754
|
+
*
|
|
755
|
+
* @param {string} text - Working-script text
|
|
756
|
+
* @param {object|null} fallbackMap - Validated { sequence: replacement } map
|
|
757
|
+
* @returns {string}
|
|
758
|
+
*/
|
|
759
|
+
function applyScriptFallback(text, fallbackMap) {
|
|
760
|
+
if (!fallbackMap || typeof text !== 'string' || text === '') return text;
|
|
761
|
+
const keys = Object.keys(fallbackMap).sort((a, b) => b.length - a.length);
|
|
762
|
+
if (keys.length === 0) return text;
|
|
763
|
+
|
|
764
|
+
let out = '';
|
|
765
|
+
let i = 0;
|
|
766
|
+
while (i < text.length) {
|
|
767
|
+
let matched = null;
|
|
768
|
+
for (const k of keys) {
|
|
769
|
+
if (text.startsWith(k, i)) { matched = k; break; }
|
|
770
|
+
}
|
|
771
|
+
if (matched) {
|
|
772
|
+
out += fallbackMap[matched];
|
|
773
|
+
i += matched.length;
|
|
774
|
+
} else {
|
|
775
|
+
out += text[i];
|
|
776
|
+
i++;
|
|
777
|
+
}
|
|
778
|
+
}
|
|
779
|
+
return out;
|
|
780
|
+
}
|
|
781
|
+
|
|
782
|
+
// -----------------------------------------------------------------
|
|
783
|
+
// Coverage and reversal — making the converters honest
|
|
784
|
+
// -----------------------------------------------------------------
|
|
785
|
+
|
|
786
|
+
/**
|
|
787
|
+
* The set of source sequences a converter's map covers, longest first.
|
|
788
|
+
* Derived from the map itself so coverage can never drift from behaviour.
|
|
789
|
+
*/
|
|
790
|
+
function mappedSequences(converterKey) {
|
|
791
|
+
const conv = SCRIPT_CONVERTERS[converterKey];
|
|
792
|
+
if (!conv?.map) return null;
|
|
793
|
+
return conv.map.map(([from]) => from).sort((a, b) => b.length - a.length);
|
|
794
|
+
}
|
|
795
|
+
|
|
796
|
+
/**
|
|
797
|
+
* Report the letters a converter would silently pass through untranslated.
|
|
798
|
+
*
|
|
799
|
+
* WHY: every converter here passes unmatched characters through, which is
|
|
800
|
+
* correct for spaces, digits and punctuation and wrong for letters. Klingon
|
|
801
|
+
* romanization has no `d`, `c`, `f`, `g`, `i`, `k`, `s`, `x` or `z`, so text
|
|
802
|
+
* containing them is not Klingon romanization — but the converter mapped what
|
|
803
|
+
* it recognised and emitted the rest as Latin, producing strings that are half
|
|
804
|
+
* pIqaD and half English with nothing raising a hand. 23 of the 32 affected
|
|
805
|
+
* strings found in the wild were this shape.
|
|
806
|
+
*
|
|
807
|
+
* Digits, whitespace and punctuation are legitimate passthrough and are never
|
|
808
|
+
* reported.
|
|
809
|
+
*
|
|
810
|
+
* @param {string} text - Text in the converter's working script
|
|
811
|
+
* @param {string} converterKey - Key into SCRIPT_CONVERTERS
|
|
812
|
+
* @returns {string[]} Distinct unmapped letters, in first-appearance order
|
|
813
|
+
*/
|
|
814
|
+
function unmappedLetters(text, converterKey) {
|
|
815
|
+
const conv = SCRIPT_CONVERTERS[converterKey];
|
|
816
|
+
if (!conv || typeof text !== 'string' || text === '') return [];
|
|
817
|
+
|
|
818
|
+
// Kryptonian is arithmetic over A–Z with no table; every letter outside
|
|
819
|
+
// the basic Latin alphabet is unmapped.
|
|
820
|
+
if (!conv.map) {
|
|
821
|
+
const out = [];
|
|
822
|
+
for (const ch of text) {
|
|
823
|
+
if (!/\p{L}/u.test(ch)) continue;
|
|
824
|
+
if (/[A-Za-z]/.test(ch)) continue;
|
|
825
|
+
if (!out.includes(ch)) out.push(ch);
|
|
826
|
+
}
|
|
827
|
+
return out;
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
const sequences = mappedSequences(converterKey);
|
|
831
|
+
// Converters that lowercase (or uppercase) their input before matching must
|
|
832
|
+
// be probed in that same normalised form, or every capital reads as unmapped.
|
|
833
|
+
const probe = conv.converter === latinToTengwar || conv.converter === sroToSyllabics
|
|
834
|
+
? text.toLowerCase()
|
|
835
|
+
: text;
|
|
836
|
+
|
|
837
|
+
const out = [];
|
|
838
|
+
let i = 0;
|
|
839
|
+
while (i < probe.length) {
|
|
840
|
+
let matched = 0;
|
|
841
|
+
for (const seq of sequences) {
|
|
842
|
+
if (probe.startsWith(seq, i)) { matched = seq.length; break; }
|
|
843
|
+
}
|
|
844
|
+
if (matched) { i += matched; continue; }
|
|
845
|
+
const ch = probe[i];
|
|
846
|
+
if (/\p{L}/u.test(ch) && !out.includes(ch)) out.push(ch);
|
|
847
|
+
i++;
|
|
848
|
+
}
|
|
849
|
+
return out;
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
/**
|
|
853
|
+
* Reverse a converter's output back to its working script.
|
|
854
|
+
*
|
|
855
|
+
* Used by `champollion repair-script` to undo conversions that should never
|
|
856
|
+
* have happened. Reversal is exact for pIqaD (the map is injective — only the
|
|
857
|
+
* straight and curly apostrophe share a codepoint, and both restore as `'`).
|
|
858
|
+
* Tengwar, Cree syllabics and Kryptonian normalise case on the way in, so the
|
|
859
|
+
* reverse cannot recover the original capitalisation; callers are told this
|
|
860
|
+
* via `caseLossy` rather than being left to discover it.
|
|
861
|
+
*
|
|
862
|
+
* @param {string} text - Converted text
|
|
863
|
+
* @param {string} converterKey - Key into SCRIPT_CONVERTERS
|
|
864
|
+
* @returns {{ reversed: string, caseLossy: boolean, unreversed: string[] }}
|
|
865
|
+
*/
|
|
866
|
+
function reverseScript(text, converterKey) {
|
|
867
|
+
const conv = SCRIPT_CONVERTERS[converterKey];
|
|
868
|
+
if (!conv || typeof text !== 'string' || text === '') {
|
|
869
|
+
return { reversed: text, caseLossy: false, unreversed: [] };
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
const caseLossy = conv.converter !== romanizationToPiqad;
|
|
873
|
+
|
|
874
|
+
if (!conv.map) {
|
|
875
|
+
// Kryptonian: U+E100–E119 → A–Z.
|
|
876
|
+
const [lo, hi] = conv.puaRange;
|
|
877
|
+
let out = '';
|
|
878
|
+
const unreversed = [];
|
|
879
|
+
for (const ch of text) {
|
|
880
|
+
const cp = ch.codePointAt(0);
|
|
881
|
+
if (cp >= lo && cp <= hi) out += String.fromCharCode(65 + (cp - lo));
|
|
882
|
+
else {
|
|
883
|
+
out += ch;
|
|
884
|
+
if (isPrivateUse(cp) && !unreversed.includes(ch)) unreversed.push(ch);
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
return { reversed: out, caseLossy, unreversed };
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
// Build the reverse table. Later duplicate targets do not overwrite earlier
|
|
891
|
+
// ones, so `'` wins over `’` for the shared glottal-stop codepoint.
|
|
892
|
+
const reverse = new Map();
|
|
893
|
+
for (const [from, to] of conv.map) {
|
|
894
|
+
if (!reverse.has(to)) reverse.set(to, from);
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
let out = '';
|
|
898
|
+
const unreversed = [];
|
|
899
|
+
for (const ch of text) {
|
|
900
|
+
const hit = reverse.get(ch);
|
|
901
|
+
if (hit !== undefined) { out += hit; continue; }
|
|
902
|
+
out += ch;
|
|
903
|
+
const cp = ch.codePointAt(0);
|
|
904
|
+
if (isPrivateUse(cp) && !unreversed.includes(ch)) unreversed.push(ch);
|
|
905
|
+
}
|
|
906
|
+
return { reversed: out, caseLossy, unreversed };
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
/**
|
|
910
|
+
* Unicode Private Use Area membership — the BMP block plus both supplementary
|
|
911
|
+
* planes. `\p{Co}` in one predicate, without a regex per call.
|
|
912
|
+
*/
|
|
913
|
+
function isPrivateUse(codePoint) {
|
|
914
|
+
return (codePoint >= 0xE000 && codePoint <= 0xF8FF)
|
|
915
|
+
|| (codePoint >= 0xF0000 && codePoint <= 0xFFFFD)
|
|
916
|
+
|| (codePoint >= 0x100000 && codePoint <= 0x10FFFD);
|
|
917
|
+
}
|
|
918
|
+
|
|
919
|
+
/**
|
|
920
|
+
* Convert text using the registered converter for a locale.
|
|
921
|
+
*
|
|
922
|
+
* `unmapped` lists letters the converter could not translate and passed
|
|
923
|
+
* through as-is. A non-empty `unmapped` means the input was not valid text in
|
|
924
|
+
* the converter's working script, and the output is a mix of both scripts —
|
|
925
|
+
* callers must treat it as a failure rather than writing it out.
|
|
926
|
+
*
|
|
927
|
+
* @param {string} text - Text in the source script
|
|
928
|
+
* @param {string} localeCode - Locale code (e.g., "crk", "sr")
|
|
929
|
+
* @returns {{ converted: string, converterUsed: string|null, unmapped: string[] }}
|
|
930
|
+
*/
|
|
931
|
+
function convertScript(text, localeCode) {
|
|
932
|
+
const converter = SCRIPT_CONVERTERS[localeCode];
|
|
933
|
+
if (!converter) {
|
|
934
|
+
return { converted: text, converterUsed: null, unmapped: [] };
|
|
935
|
+
}
|
|
936
|
+
|
|
937
|
+
return {
|
|
938
|
+
converted: converter.converter(text),
|
|
939
|
+
converterUsed: `${converter.from} → ${converter.to}`,
|
|
940
|
+
unmapped: unmappedLetters(text, localeCode),
|
|
941
|
+
};
|
|
942
|
+
}
|
|
943
|
+
|
|
944
|
+
/**
|
|
945
|
+
* Check if a locale has a registered script converter.
|
|
946
|
+
*
|
|
947
|
+
* @param {string} localeCode - Locale code
|
|
948
|
+
* @returns {boolean}
|
|
949
|
+
*/
|
|
950
|
+
function hasScriptConverter(localeCode) {
|
|
951
|
+
return localeCode in SCRIPT_CONVERTERS;
|
|
952
|
+
}
|
|
953
|
+
|
|
954
|
+
/**
|
|
955
|
+
* Get converter info for a locale (without the function reference).
|
|
956
|
+
* Safe for serialization into config/reports.
|
|
957
|
+
*
|
|
958
|
+
* @param {string} localeCode - Locale code
|
|
959
|
+
* @returns {object|null}
|
|
960
|
+
*/
|
|
961
|
+
function getConverterInfo(localeCode) {
|
|
962
|
+
const conv = SCRIPT_CONVERTERS[localeCode];
|
|
963
|
+
if (!conv) return null;
|
|
964
|
+
const info = {
|
|
965
|
+
from: conv.from,
|
|
966
|
+
to: conv.to,
|
|
967
|
+
type: conv.type,
|
|
968
|
+
fromScript: conv.fromScript,
|
|
969
|
+
toScript: conv.toScript,
|
|
970
|
+
puaRange: conv.puaRange,
|
|
971
|
+
};
|
|
972
|
+
if (conv.fontNote) info.fontNote = conv.fontNote;
|
|
973
|
+
return info;
|
|
974
|
+
}
|
|
975
|
+
|
|
976
|
+
export {
|
|
977
|
+
sroToSyllabics,
|
|
978
|
+
latinToCyrillicSr,
|
|
979
|
+
romanizationToPiqad,
|
|
980
|
+
latinToTengwar,
|
|
981
|
+
latinToKryptonian,
|
|
982
|
+
convertScript,
|
|
983
|
+
hasScriptConverter,
|
|
984
|
+
getConverterInfo,
|
|
985
|
+
resolveTargetScript,
|
|
986
|
+
converterKeyForLocale,
|
|
987
|
+
formatScriptChoiceError,
|
|
988
|
+
validateScriptFallback,
|
|
989
|
+
applyScriptFallback,
|
|
990
|
+
unmappedLetters,
|
|
991
|
+
reverseScript,
|
|
992
|
+
isPrivateUse,
|
|
993
|
+
SCRIPT_CONVERTERS,
|
|
994
|
+
};
|