champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
package/lib/scripts.js ADDED
@@ -0,0 +1,994 @@
1
+ /**
2
+ * Script conversion registry — deterministic orthography converters.
3
+ *
4
+ * WHY: Some languages have multiple scripts for the same spoken language.
5
+ * Translation workflows often prefer a "working script" (easier to type,
6
+ * edit, and version-control) that gets converted to a "display script"
7
+ * as a post-translation step.
8
+ *
9
+ * Examples:
10
+ * - Plains Cree: SRO (Standard Roman Orthography) → Syllabics (ᓀᐦᐃᔭᐍᐏᐣ)
11
+ * - Serbian: Latin → Cyrillic
12
+ * - Japanese: Romaji → Hiragana/Katakana
13
+ * - Hindi: Romanized → Devanagari
14
+ *
15
+ * All converters here are DETERMINISTIC — no LLM needed, pure lookup tables.
16
+ * They run as a post-translation hook: translate in working script, then
17
+ * convert to display script.
18
+ *
19
+ * ADDING A NEW CONVERTER:
20
+ * 1. Add the conversion map below
21
+ * 2. Create the converter function (input string → output string)
22
+ * 3. Register it in SCRIPT_CONVERTERS with the locale code
23
+ * 4. Add the `scripts` field to the language's register entry in registers.js
24
+ */
25
+
26
+ // -----------------------------------------------------------------
27
+ // Plains Cree: SRO → Syllabics
28
+ // -----------------------------------------------------------------
29
+
30
+ /**
31
+ * SRO to Cree Syllabics conversion table.
32
+ *
33
+ * This is the standard mapping used by the University of Alberta's
34
+ * ALTLab and documented in Wolvengrey's Cree: Words dictionary.
35
+ *
36
+ * The mapping is context-sensitive: consonant+vowel combinations map
37
+ * to specific syllabic characters, while standalone consonants use
38
+ * finals (small superscript forms).
39
+ *
40
+ * ORDER MATTERS: Longer sequences must be matched before shorter ones
41
+ * (e.g., "twê" before "tw" before "t").
42
+ */
43
+ const SRO_TO_SYLLABICS_MAP = [
44
+ // Long vowels with w-glide (must come before short vowel w-glide)
45
+ ['pwê', 'ᐻ'], ['pwî', 'ᐽ'], ['pwô', 'ᐿ'], ['pwâ', 'ᑁ'],
46
+ ['twê', 'ᑗ'], ['twî', 'ᑙ'], ['twô', 'ᑛ'], ['twâ', 'ᑝ'],
47
+ ['kwê', 'ᑵ'], ['kwî', 'ᑷ'], ['kwô', 'ᑹ'], ['kwâ', 'ᑻ'],
48
+ ['cwê', 'ᒑ'], ['cwî', 'ᒓ'], ['cwô', 'ᒕ'], ['cwâ', 'ᒗ'],
49
+ ['mwê', 'ᒫ'], ['mwî', 'ᒭ'], ['mwô', 'ᒯ'], ['mwâ', 'ᒱ'],
50
+ ['nwê', 'ᓇ'], ['nwî', 'ᓉ'], ['nwô', 'ᓋ'], ['nwâ', 'ᓍ'],
51
+ ['swê', 'ᓭ'], ['swî', 'ᓯ'], ['swô', 'ᓱ'], ['swâ', 'ᓳ'],
52
+ ['ywê', 'ᔋ'], ['ywî', 'ᔍ'], ['ywô', 'ᔏ'], ['ywâ', 'ᔑ'],
53
+
54
+ // Short vowels with w-glide
55
+ ['pwe', 'ᐺ'], ['pwi', 'ᐼ'], ['pwo', 'ᐾ'], ['pwa', 'ᑀ'],
56
+ ['twe', 'ᑖ'], ['twi', 'ᑘ'], ['two', 'ᑚ'], ['twa', 'ᑜ'],
57
+ ['kwe', 'ᑴ'], ['kwi', 'ᑶ'], ['kwo', 'ᑸ'], ['kwa', 'ᑺ'],
58
+ ['cwe', 'ᒐ'], ['cwi', 'ᒒ'], ['cwo', 'ᒔ'], ['cwa', 'ᒖ'],
59
+ ['mwe', 'ᒪ'], ['mwi', 'ᒬ'], ['mwo', 'ᒮ'], ['mwa', 'ᒰ'],
60
+ ['nwe', 'ᓈ'], ['nwi', 'ᓊ'], ['nwo', 'ᓌ'], ['nwa', 'ᓎ'],
61
+ ['swe', 'ᓬ'], ['swi', 'ᓮ'], ['swo', 'ᓰ'], ['swa', 'ᓲ'],
62
+ ['ywe', 'ᔊ'], ['ywi', 'ᔌ'], ['ywo', 'ᔎ'], ['ywa', 'ᔐ'],
63
+
64
+ // Long vowels (macron forms — these must come before short vowels)
65
+ ['pê', 'ᐯ'], ['pî', 'ᐲ'], ['pô', 'ᐴ'], ['pâ', 'ᐹ'],
66
+ ['tê', 'ᑌ'], ['tî', 'ᑏ'], ['tô', 'ᑑ'], ['tâ', 'ᑖ'],
67
+ ['kê', 'ᑫ'], ['kî', 'ᑮ'], ['kô', 'ᑰ'], ['kâ', 'ᑳ'],
68
+ ['cê', 'ᒉ'], ['cî', 'ᒌ'], ['cô', 'ᒎ'], ['câ', 'ᒑ'],
69
+ ['mê', 'ᒣ'], ['mî', 'ᒦ'], ['mô', 'ᒨ'], ['mâ', 'ᒫ'],
70
+ ['nê', 'ᓀ'], ['nî', 'ᓃ'], ['nô', 'ᓅ'], ['nâ', 'ᓈ'],
71
+ ['sê', 'ᓭ'], ['sî', 'ᓰ'], ['sô', 'ᓲ'], ['sâ', 'ᓵ'],
72
+ ['yê', 'ᔦ'], ['yî', 'ᔩ'], ['yô', 'ᔫ'], ['yâ', 'ᔮ'],
73
+
74
+ // Short vowels (consonant+vowel)
75
+ ['pe', 'ᐯ'], ['pi', 'ᐱ'], ['po', 'ᐳ'], ['pa', 'ᐸ'],
76
+ ['te', 'ᑌ'], ['ti', 'ᑎ'], ['to', 'ᑐ'], ['ta', 'ᑕ'],
77
+ ['ke', 'ᑫ'], ['ki', 'ᑭ'], ['ko', 'ᑯ'], ['ka', 'ᑲ'],
78
+ ['ce', 'ᒉ'], ['ci', 'ᒋ'], ['co', 'ᒍ'], ['ca', 'ᒐ'],
79
+ ['me', 'ᒣ'], ['mi', 'ᒥ'], ['mo', 'ᒧ'], ['ma', 'ᒪ'],
80
+ ['ne', 'ᓀ'], ['ni', 'ᓂ'], ['no', 'ᓄ'], ['na', 'ᓇ'],
81
+ ['se', 'ᓭ'], ['si', 'ᓯ'], ['so', 'ᓱ'], ['sa', 'ᓴ'],
82
+ ['ye', 'ᔦ'], ['yi', 'ᔨ'], ['yo', 'ᔪ'], ['ya', 'ᔭ'],
83
+
84
+ // Standalone vowels (long first)
85
+ ['ê', 'ᐁ'], ['î', 'ᐄ'], ['ô', 'ᐆ'], ['â', 'ᐋ'],
86
+ ['e', 'ᐁ'], ['i', 'ᐃ'], ['o', 'ᐅ'], ['a', 'ᐊ'],
87
+
88
+ // Digraphs (must come before single-char finals)
89
+ ['th', 'ᖧ'],
90
+
91
+ // Finals (standalone consonants — no following vowel)
92
+ ['p', 'ᑊ'], ['t', 'ᐟ'], ['k', 'ᐠ'], ['c', 'ᐨ'],
93
+ ['m', 'ᒼ'], ['n', 'ᐣ'], ['s', 'ᐢ'], ['y', 'ᐩ'],
94
+
95
+ // Special characters
96
+ ['h', 'ᐦ'], ['w', 'ᐤ'], ['l', 'ᓬ'], ['r', 'ᕒ'],
97
+ ];
98
+
99
+ /**
100
+ * Convert SRO text to Cree Syllabics.
101
+ *
102
+ * This is a greedy left-to-right scan: at each position, try the longest
103
+ * possible match first. Characters that don't match any pattern (spaces,
104
+ * punctuation, numbers) pass through unchanged.
105
+ *
106
+ * @param {string} sro - SRO text to convert
107
+ * @returns {string} Syllabics text
108
+ */
109
+ function sroToSyllabics(sro) {
110
+ const input = sro.toLowerCase();
111
+ let result = '';
112
+ let i = 0;
113
+
114
+ while (i < input.length) {
115
+ let matched = false;
116
+
117
+ // Try longest matches first (up to 3 characters)
118
+ for (const [from, to] of SRO_TO_SYLLABICS_MAP) {
119
+ if (input.startsWith(from, i)) {
120
+ result += to;
121
+ i += from.length;
122
+ matched = true;
123
+ break;
124
+ }
125
+ }
126
+
127
+ // No match — pass character through (space, punctuation, etc.)
128
+ if (!matched) {
129
+ result += input[i];
130
+ i++;
131
+ }
132
+ }
133
+
134
+ return result;
135
+ }
136
+
137
+ // -----------------------------------------------------------------
138
+ // Serbian: Latin → Cyrillic
139
+ // -----------------------------------------------------------------
140
+
141
+ const LATIN_TO_CYRILLIC_SR = {
142
+ 'lj': 'љ', 'nj': 'њ', 'dž': 'џ',
143
+ 'Lj': 'Љ', 'Nj': 'Њ', 'Dž': 'Џ',
144
+ 'LJ': 'Љ', 'NJ': 'Њ', 'DŽ': 'Џ',
145
+ 'a': 'а', 'b': 'б', 'v': 'в', 'g': 'г', 'd': 'д',
146
+ 'đ': 'ђ', 'e': 'е', 'ž': 'ж', 'z': 'з', 'i': 'и',
147
+ 'j': 'ј', 'k': 'к', 'l': 'л', 'm': 'м', 'n': 'н',
148
+ 'o': 'о', 'p': 'п', 'r': 'р', 's': 'с', 't': 'т',
149
+ 'ć': 'ћ', 'u': 'у', 'f': 'ф', 'h': 'х', 'c': 'ц',
150
+ 'č': 'ч', 'š': 'ш',
151
+ 'A': 'А', 'B': 'Б', 'V': 'В', 'G': 'Г', 'D': 'Д',
152
+ 'Đ': 'Ђ', 'E': 'Е', 'Ž': 'Ж', 'Z': 'З', 'I': 'И',
153
+ 'J': 'Ј', 'K': 'К', 'L': 'Л', 'M': 'М', 'N': 'Н',
154
+ 'O': 'О', 'P': 'П', 'R': 'Р', 'S': 'С', 'T': 'Т',
155
+ 'Ć': 'Ћ', 'U': 'У', 'F': 'Ф', 'H': 'Х', 'C': 'Ц',
156
+ 'Č': 'Ч', 'Š': 'Ш',
157
+ };
158
+
159
+ /**
160
+ * Convert Serbian Latin text to Cyrillic.
161
+ * Digraphs (lj, nj, dž) are matched first.
162
+ *
163
+ * @param {string} latin - Latin text
164
+ * @returns {string} Cyrillic text
165
+ */
166
+ function latinToCyrillicSr(latin) {
167
+ let result = '';
168
+ let i = 0;
169
+
170
+ while (i < latin.length) {
171
+ // Try digraphs first (2 characters)
172
+ if (i + 1 < latin.length) {
173
+ const digraph = latin.slice(i, i + 2);
174
+ if (LATIN_TO_CYRILLIC_SR[digraph]) {
175
+ result += LATIN_TO_CYRILLIC_SR[digraph];
176
+ i += 2;
177
+ continue;
178
+ }
179
+ }
180
+
181
+ // Single character
182
+ const ch = latin[i];
183
+ result += LATIN_TO_CYRILLIC_SR[ch] || ch;
184
+ i++;
185
+ }
186
+
187
+ return result;
188
+ }
189
+
190
+ // -----------------------------------------------------------------
191
+ // Klingon: Romanization → pIqaD (CSUR PUA U+F8D0–F8FF)
192
+ // -----------------------------------------------------------------
193
+
194
+ /**
195
+ * Klingon romanization to pIqaD conversion table.
196
+ *
197
+ * Based on the ConScript Unicode Registry (CSUR) mapping maintained
198
+ * at evertype.com. Characters are in the Unicode Private Use Area
199
+ * — they require a pIqaD-compatible web font to render visually.
200
+ *
201
+ * Klingon romanization is case-sensitive: 'D' ≠ 'd', 'S' ≠ 's',
202
+ * 'I' ≠ 'i', 'Q' ≠ 'q'. The table preserves this distinction.
203
+ *
204
+ * ORDER: Trigraphs (tlh) → digraphs (ch, gh, ng) → single chars.
205
+ */
206
+ const KLINGON_TO_PIQAD_MAP = [
207
+ // Trigraph (must come first)
208
+ ['tlh', '\uF8E4'],
209
+
210
+ // Digraphs
211
+ ['ch', '\uF8D2'], ['gh', '\uF8D5'], ['ng', '\uF8DC'],
212
+
213
+ // Case-sensitive single characters
214
+ // Uppercase-only letters (distinct phonemes in Klingon)
215
+ ['D', '\uF8D3'], ['H', '\uF8D6'], ['I', '\uF8D7'],
216
+ ['Q', '\uF8E0'], ['S', '\uF8E2'],
217
+
218
+ // Lowercase letters
219
+ ['a', '\uF8D0'], ['b', '\uF8D1'], ['e', '\uF8D4'],
220
+ ['j', '\uF8D8'], ['l', '\uF8D9'], ['m', '\uF8DA'],
221
+ ['n', '\uF8DB'], ['o', '\uF8DD'], ['p', '\uF8DE'],
222
+ ['q', '\uF8DF'], ['r', '\uF8E1'], ['t', '\uF8E3'],
223
+ ['u', '\uF8E5'], ['v', '\uF8E6'], ['w', '\uF8E7'],
224
+ ['y', '\uF8E8'],
225
+
226
+ // Glottal stop (apostrophe)
227
+ ["'", '\uF8E9'],
228
+ ['\u2019', '\uF8E9'], // right single quote (common in copy-pasted text)
229
+ ];
230
+
231
+ /**
232
+ * Convert Klingon romanization to pIqaD script.
233
+ *
234
+ * Greedy left-to-right scan, longest match first.
235
+ * Case-sensitive: 'D' (retroflex) ≠ 'd' (not a Klingon phoneme).
236
+ * Non-Klingon characters (spaces, punctuation, numbers) pass through.
237
+ *
238
+ * NOTE: Output uses Unicode PUA characters (U+F8D0–F8FF).
239
+ * A pIqaD web font (e.g., "pIqaD qolqoS" or "Klingon pIqaD HaSta")
240
+ * must be loaded for visual rendering.
241
+ *
242
+ * @param {string} romanized - Klingon text in standard romanization
243
+ * @returns {string} pIqaD text
244
+ */
245
+ function romanizationToPiqad(romanized) {
246
+ let result = '';
247
+ let i = 0;
248
+
249
+ while (i < romanized.length) {
250
+ let matched = false;
251
+
252
+ for (const [from, to] of KLINGON_TO_PIQAD_MAP) {
253
+ if (romanized.startsWith(from, i)) {
254
+ result += to;
255
+ i += from.length;
256
+ matched = true;
257
+ break;
258
+ }
259
+ }
260
+
261
+ if (!matched) {
262
+ result += romanized[i];
263
+ i++;
264
+ }
265
+ }
266
+
267
+ return result;
268
+ }
269
+
270
+ // -----------------------------------------------------------------
271
+ // Tengwar: Sindarin Latin → Tengwar (CSUR PUA U+E000–E07F)
272
+ // Mode of Beleriand — full vowel letters (not diacritics)
273
+ // -----------------------------------------------------------------
274
+
275
+ /**
276
+ * Sindarin Latin to Tengwar conversion table.
277
+ *
278
+ * Uses the "Mode of Beleriand" where vowels are full tengwar letters
279
+ * rather than tehtar (diacritics). This is the most deterministic
280
+ * mode — the Ómatehtar mode requires context-dependent diacritic
281
+ * placement which is significantly more complex.
282
+ *
283
+ * Based on the CSUR Tengwar block (U+E000–E07F) as documented by
284
+ * the Free Tengwar Font Project. Requires a CSUR-compatible Tengwar
285
+ * font (e.g., "Tengwar Formal CSUR", "Tengwar Annatar") to render.
286
+ *
287
+ * This is a simplified converter — it handles the most common
288
+ * Sindarin consonants and vowels but does not implement:
289
+ * - Double consonant bars (nasal signs)
290
+ * - Sa-rincë (s-hooks)
291
+ * - Ligatures for common combinations
292
+ *
293
+ * ORDER: Digraphs → single characters.
294
+ */
295
+ const SINDARIN_TO_TENGWAR_MAP = [
296
+ // Digraphs (must come before single chars)
297
+ ['th', '\uE003'], // thúlë (voiceless th)
298
+ ['dh', '\uE004'], // anto (voiced th/dh)
299
+ ['ch', '\uE002'], // hwesta (voiceless velar fricative)
300
+ ['ph', '\uE00E'], // formen (labialized)
301
+ ['ng', '\uE016'], // noldo
302
+ ['nd', '\uE022'], // ando+númen combo — using ando
303
+ ['mb', '\uE022'], // umbar area
304
+ ['nn', '\uE015'], // doubled númen
305
+ ['mm', '\uE012'], // doubled malta
306
+ ['ll', '\uE00B'], // doubled lambe
307
+ ['rh', '\uE00C'], // rómen (voiceless r)
308
+ ['lh', '\uE00D'], // silmë (voiceless l)
309
+ ['hw', '\uE017'], // hwesta sindarinwa
310
+
311
+ // Consonants (single)
312
+ ['t', '\uE001'], // tinco
313
+ ['p', '\uE00E'], // parma
314
+ ['c', '\uE002'], // calma (hard c/k)
315
+ ['k', '\uE002'], // calma
316
+ ['d', '\uE005'], // ando
317
+ ['b', '\uE00F'], // umbar
318
+ ['g', '\uE006'], // anga (hard g)
319
+ ['f', '\uE010'], // formen
320
+ ['v', '\uE011'], // ampa
321
+ ['n', '\uE015'], // númen
322
+ ['m', '\uE012'], // malta
323
+ ['r', '\uE00C'], // óre/rómen
324
+ ['l', '\uE00B'], // lambe
325
+ ['s', '\uE008'], // silmë
326
+ ['h', '\uE017'], // hyarmen
327
+ ['w', '\uE013'], // vilya/vala
328
+ ['y', '\uE014'], // anna
329
+
330
+ // Vowels — Mode of Beleriand uses full letters, not diacritics
331
+ // Long vowels (circumflex or macron) mapped to long carriers
332
+ ['á', '\uE040'], // long a carrier
333
+ ['é', '\uE042'], // long e carrier
334
+ ['í', '\uE044'], // long i carrier
335
+ ['ó', '\uE046'], // long o carrier
336
+ ['ú', '\uE048'], // long u carrier
337
+ ['â', '\uE040'],
338
+ ['ê', '\uE042'],
339
+ ['î', '\uE044'],
340
+ ['ô', '\uE046'],
341
+ ['û', '\uE048'],
342
+
343
+ // Short vowels
344
+ ['a', '\uE03F'], // short a
345
+ ['e', '\uE041'], // short e
346
+ ['i', '\uE043'], // short i
347
+ ['o', '\uE045'], // short o
348
+ ['u', '\uE047'], // short u
349
+ ];
350
+
351
+ /**
352
+ * Convert Sindarin Latin text to Tengwar script (Mode of Beleriand).
353
+ *
354
+ * NOTE: Output uses Unicode PUA characters (U+E000–E07F).
355
+ * A CSUR-compatible Tengwar font must be loaded for visual rendering.
356
+ *
357
+ * @param {string} latin - Sindarin text in Latin script
358
+ * @returns {string} Tengwar text
359
+ */
360
+ function latinToTengwar(latin) {
361
+ const input = latin.toLowerCase();
362
+ let result = '';
363
+ let i = 0;
364
+
365
+ while (i < input.length) {
366
+ let matched = false;
367
+
368
+ for (const [from, to] of SINDARIN_TO_TENGWAR_MAP) {
369
+ if (input.startsWith(from, i)) {
370
+ result += to;
371
+ i += from.length;
372
+ matched = true;
373
+ break;
374
+ }
375
+ }
376
+
377
+ if (!matched) {
378
+ result += input[i];
379
+ i++;
380
+ }
381
+ }
382
+
383
+ return result;
384
+ }
385
+
386
+ // -----------------------------------------------------------------
387
+ // Kryptonian: Latin → Kryptonian (font-based cipher)
388
+ // -----------------------------------------------------------------
389
+
390
+ /**
391
+ * Kryptonian "script conversion" — 1:1 Latin alphabet cipher.
392
+ *
393
+ * Unlike the other converters, Kryptonian has NO standard Unicode
394
+ * assignment (not even PUA/CSUR). The DC Comics script is a pure
395
+ * substitution cipher of the Latin alphabet, rendered via custom fonts.
396
+ *
397
+ * This converter maps A-Z to Unicode PUA characters (U+E100–E119)
398
+ * using a conventional fan-community assignment. The mapping is:
399
+ * A=U+E100, B=U+E101, ..., Z=U+E119
400
+ *
401
+ * FONT REQUIRED: A Kryptonian font mapped to these PUA codepoints
402
+ * (e.g., "Kryptonian" from kryptonian.info). Without the font,
403
+ * output will render as empty boxes.
404
+ *
405
+ * Alternative approach: skip this converter entirely and use
406
+ * CSS `font-family: 'Kryptonian'` on the element. The text stays
407
+ * as Latin characters but renders in Kryptonian glyphs. This is
408
+ * often simpler for web deployments.
409
+ */
410
+ function latinToKryptonian(text) {
411
+ let result = '';
412
+
413
+ for (const ch of text) {
414
+ const upper = ch.toUpperCase();
415
+ const code = upper.charCodeAt(0);
416
+
417
+ // Map A-Z (65-90) to PUA U+E100-E119
418
+ if (code >= 65 && code <= 90) {
419
+ result += String.fromCharCode(0xE100 + (code - 65));
420
+ } else {
421
+ // Non-alpha characters (spaces, punctuation, numbers) pass through
422
+ result += ch;
423
+ }
424
+ }
425
+
426
+ return result;
427
+ }
428
+
429
+ // -----------------------------------------------------------------
430
+ // Converter Registry
431
+ // -----------------------------------------------------------------
432
+
433
+ /**
434
+ * Registry of available script converters.
435
+ *
436
+ * Each entry maps a locale code to its converter configuration:
437
+ * - from: source script name
438
+ * - to: target script name
439
+ * - fromScript ISO 15924 code of the working script the LLM emits
440
+ * - toScript: ISO 15924 code of the converted output, or null when the
441
+ * target script has no registered code (Kryptonian). A null
442
+ * toScript CANNOT be selected via `script:` config — see
443
+ * resolveTargetScript.
444
+ * - type: 'deterministic' (pure lookup), or 'font-based' (needs web font)
445
+ * - converter: function(string) → string
446
+ * - map: the forward table, used to derive letter coverage and the
447
+ * reverse table. Null for converters that are pure arithmetic.
448
+ * - puaRange: [lo, hi] of the Private Use Area block the converter emits,
449
+ * or null when output is in assigned Unicode.
450
+ * - fontNote: (optional) font requirement for PUA-based converters
451
+ */
452
+ const SCRIPT_CONVERTERS = {
453
+ crk: {
454
+ from: 'SRO (Standard Roman Orthography)',
455
+ to: 'Cree Syllabics',
456
+ fromScript: 'Latn',
457
+ toScript: 'Cans',
458
+ type: 'deterministic',
459
+ map: SRO_TO_SYLLABICS_MAP,
460
+ puaRange: null,
461
+ converter: sroToSyllabics,
462
+ },
463
+ sr: {
464
+ from: 'Latin',
465
+ to: 'Cyrillic',
466
+ fromScript: 'Latn',
467
+ toScript: 'Cyrl',
468
+ type: 'deterministic',
469
+ map: Object.entries(LATIN_TO_CYRILLIC_SR),
470
+ puaRange: null,
471
+ converter: latinToCyrillicSr,
472
+ },
473
+ tlh: {
474
+ from: 'Romanization',
475
+ to: 'pIqaD',
476
+ fromScript: 'Latn',
477
+ toScript: 'Piqd',
478
+ type: 'deterministic',
479
+ map: KLINGON_TO_PIQAD_MAP,
480
+ puaRange: [0xF8D0, 0xF8FF],
481
+ fontNote: 'Requires pIqaD web font (PUA U+F8D0–F8FF)',
482
+ converter: romanizationToPiqad,
483
+ },
484
+ 'x-elvish-s': {
485
+ from: 'Latin',
486
+ to: 'Tengwar (Mode of Beleriand)',
487
+ fromScript: 'Latn',
488
+ toScript: 'Teng',
489
+ type: 'deterministic',
490
+ map: SINDARIN_TO_TENGWAR_MAP,
491
+ puaRange: [0xE000, 0xE07F],
492
+ fontNote: 'Requires CSUR Tengwar font (PUA U+E000–E07F)',
493
+ converter: latinToTengwar,
494
+ },
495
+ 'x-kryptonian': {
496
+ from: 'Latin',
497
+ to: 'Kryptonian',
498
+ fromScript: 'Latn',
499
+ // Kryptonian has no ISO 15924 code — not even a provisional one. The
500
+ // encoded form is therefore unreachable through `script:` and must be
501
+ // requested with the explicit escape hatch (see resolveTargetScript).
502
+ toScript: null,
503
+ type: 'font-based',
504
+ map: null,
505
+ puaRange: [0xE100, 0xE119],
506
+ fontNote: 'Requires Kryptonian font mapped to PUA U+E100–E119',
507
+ converter: latinToKryptonian,
508
+ },
509
+ };
510
+
511
+ // -----------------------------------------------------------------
512
+ // Target-script resolution
513
+ // -----------------------------------------------------------------
514
+
515
+ /**
516
+ * ISO 15924 codes are exactly four letters, initial capital: Latn, Cans, Cyrl.
517
+ */
518
+ const ISO_15924 = /^[A-Z][a-z]{3}$/;
519
+
520
+ /**
521
+ * Legacy/informal script names that used to appear in `script:` config, mapped
522
+ * to the ISO 15924 code that replaces them. The field was accepted but never
523
+ * read, so no project can have been depending on the old values working —
524
+ * but they were documented, so name the replacement rather than just rejecting.
525
+ */
526
+ const LEGACY_SCRIPT_ALIASES = {
527
+ syllabics: 'Cans',
528
+ latin: 'Latn',
529
+ roman: 'Latn',
530
+ cyrillic: 'Cyrl',
531
+ piqad: 'Piqd',
532
+ tengwar: 'Teng',
533
+ };
534
+
535
+ /**
536
+ * The explicit escape hatch for converters whose output script has no ISO
537
+ * 15924 code. Set `script: "x-kryptonian"` (the converter key) to opt in.
538
+ */
539
+ function isConverterKeyEscapeHatch(value) {
540
+ return Object.prototype.hasOwnProperty.call(SCRIPT_CONVERTERS, value);
541
+ }
542
+
543
+ /**
544
+ * The converter registered for a locale, resolved through the card.
545
+ *
546
+ * The registry is keyed by converter key, which is USUALLY the locale code but
547
+ * not always: Serbian's converter is keyed `sr` while the card's canonical
548
+ * code is `srp`. The card's `scriptConverter` field records the key, so the
549
+ * card is the join — never guess from the code alone.
550
+ *
551
+ * @param {string} localeCode - Target locale code
552
+ * @param {object|null} card - The language card, or null
553
+ * @returns {string|null} The converter key, or null when none is registered
554
+ */
555
+ function converterKeyForLocale(localeCode, card = null) {
556
+ if (card?.scriptConverter && SCRIPT_CONVERTERS[card.scriptConverter]) {
557
+ return card.scriptConverter;
558
+ }
559
+ return SCRIPT_CONVERTERS[localeCode] ? localeCode : null;
560
+ }
561
+
562
+ /**
563
+ * Resolve which script a locale's output should be written in.
564
+ *
565
+ * WHY this exists: until 0.2.0 the decision was a bare lookup —
566
+ * `hasScriptConverter(target)` — so every project targeting crk/sr/tlh/
567
+ * x-elvish-s/x-kryptonian had its output rewritten into the converter's
568
+ * display script unconditionally, with no way to decline. For the PUA
569
+ * converters that shipped unrenderable text to anyone whose font was keyed
570
+ * to Latin transliteration rather than Private Use Area codepoints; for
571
+ * crk it silently chose a community's display orthography on their behalf.
572
+ *
573
+ * Resolution:
574
+ * 1. Explicit `script:` config (per-language or per-pair) — user intent
575
+ * wins. ISO 15924, any casing accepted ("cans" → "Cans"); a value the
576
+ * locale's converter cannot produce fails loud listing what it can.
577
+ * 2. No config, and the locale's converter targets a REAL Unicode script
578
+ * (puaRange null — crk → Cans, sr → Cyrl): `{ source: 'choice-required',
579
+ * choices }`. Both orthographies are legitimate; picking one is not a
580
+ * default we get to make. Translation lanes refuse to run until the
581
+ * config says which; read-only lanes (status, integrity, repair) may
582
+ * proceed and display the state.
583
+ * 3. No config, and the converter targets a PUA block (tlh, x-elvish-s,
584
+ * x-kryptonian — scripts NOT in Unicode): default to the working script
585
+ * (Latn romanization), the only output that renders without a custom
586
+ * font. Opting into the PUA form is one config line away.
587
+ * 4. No converter at all: `script` passes through informationally,
588
+ * converterKey null, zero behavior change.
589
+ *
590
+ * @param {string} localeCode - Target locale code
591
+ * @param {object} langConfig - Per-language / per-pair config (reads .script)
592
+ * @param {object|null} card - The language card, or null when none exists
593
+ * @returns {{ script: string|null, source: 'config'|'default'|'choice-required'|'none',
594
+ * converterKey: string|null, choices?: Array<{script: string, label: string}> }}
595
+ * @throws {Error} on an unusable `script:` value — fail loud, never guess
596
+ */
597
+ function resolveTargetScript(localeCode, langConfig = {}, card = null) {
598
+ const registeredKey = converterKeyForLocale(localeCode, card);
599
+ const conv = registeredKey ? SCRIPT_CONVERTERS[registeredKey] : null;
600
+ const requested = langConfig?.script;
601
+
602
+ if (requested != null && requested !== '') {
603
+ if (typeof requested !== 'string') {
604
+ throw new Error(
605
+ `Invalid "script" for ${localeCode}: expected an ISO 15924 code string, got ${typeof requested}.`
606
+ );
607
+ }
608
+
609
+ // The escape hatch for scripts with no ISO 15924 code (Kryptonian).
610
+ // Only valid on the locale that owns the converter — running the
611
+ // Kryptonian cipher on some other language is never what anyone meant.
612
+ if (isConverterKeyEscapeHatch(requested) && SCRIPT_CONVERTERS[requested].toScript === null) {
613
+ if (requested !== registeredKey) {
614
+ throw new Error(
615
+ `Invalid "script" for ${localeCode}: "${requested}" is the ${SCRIPT_CONVERTERS[requested].to} ` +
616
+ `converter, which belongs to the ${requested} locale, not ${localeCode}.`
617
+ );
618
+ }
619
+ return { script: null, source: 'config', converterKey: requested };
620
+ }
621
+
622
+ const alias = LEGACY_SCRIPT_ALIASES[requested.toLowerCase()];
623
+ if (alias && !ISO_15924.test(requested)) {
624
+ throw new Error(
625
+ `Invalid "script" for ${localeCode}: "${requested}" is not an ISO 15924 code. Use "${alias}".`
626
+ );
627
+ }
628
+
629
+ // Case-normalize a 4-letter value before validating: "cans", "PIQD" and
630
+ // "Cans" all mean the same code, and the project's own docs used the
631
+ // lowercase form for years — erroring on it would punish people for
632
+ // following us.
633
+ const normalized = /^[A-Za-z]{4}$/.test(requested)
634
+ ? requested[0].toUpperCase() + requested.slice(1).toLowerCase()
635
+ : requested;
636
+ if (!ISO_15924.test(normalized)) {
637
+ throw new Error(
638
+ `Invalid "script" for ${localeCode}: "${requested}" is not an ISO 15924 code ` +
639
+ `(four letters — e.g. "Latn", "Cans", "Cyrl").`
640
+ );
641
+ }
642
+
643
+ // A locale WITH a converter can produce exactly two scripts: the working
644
+ // script and the converter's target. Anything else is a config mistake
645
+ // that would otherwise no-op silently — name what IS available.
646
+ if (conv && normalized !== conv.fromScript && normalized !== conv.toScript) {
647
+ const options = [`"${conv.fromScript}" (${conv.from})`, conv.toScript ? `"${conv.toScript}" (${conv.to})` : `"${registeredKey}" (${conv.to})`];
648
+ throw new Error(
649
+ `Invalid "script" for ${localeCode}: "${requested}" is not a script this locale can produce. ` +
650
+ `Available: ${options.join(' or ')}.`
651
+ );
652
+ }
653
+
654
+ return {
655
+ script: normalized,
656
+ source: 'config',
657
+ converterKey: conv && normalized === conv.toScript ? registeredKey : null,
658
+ };
659
+ }
660
+
661
+ // No explicit choice. What happens next depends on what KIND of converter
662
+ // this locale has — a real-Unicode orthography choice is the user's to
663
+ // make; an out-of-Unicode display encoding defaults safely off.
664
+ if (conv) {
665
+ if (conv.puaRange === null) {
666
+ return {
667
+ script: null,
668
+ source: 'choice-required',
669
+ converterKey: null,
670
+ choices: [
671
+ { script: conv.fromScript, label: conv.from },
672
+ { script: conv.toScript, label: conv.to },
673
+ ],
674
+ };
675
+ }
676
+ return { script: conv.fromScript, source: 'default', converterKey: null };
677
+ }
678
+
679
+ // No converter — nothing to decide. An informational `script` from config
680
+ // never reaches here (handled above); absence means absence.
681
+ return { script: null, source: 'none', converterKey: null };
682
+ }
683
+
684
+ /**
685
+ * Format the choice-required error for a locale, shared by every lane that
686
+ * refuses to translate without the decision — one message, everywhere.
687
+ *
688
+ * @param {string} localeCode
689
+ * @param {{choices: Array<{script: string, label: string}>}} resolution
690
+ * @returns {string}
691
+ */
692
+ function formatScriptChoiceError(localeCode, resolution) {
693
+ const opts = resolution.choices
694
+ .map(c => `"script": "${c.script}" (${c.label})`)
695
+ .join(' or ');
696
+ return (
697
+ `${localeCode} has more than one real orthography and Champollion will not pick one ` +
698
+ `for a community. Set ${opts} for ${localeCode} in champollion.config.json.`
699
+ );
700
+ }
701
+
702
+ // -----------------------------------------------------------------
703
+ // Transliteration fallbacks — user-declared rules for unmapped letters
704
+ // -----------------------------------------------------------------
705
+
706
+ /**
707
+ * Validate a `scriptFallback` map against a converter.
708
+ *
709
+ * Each entry maps a working-script sequence the converter does NOT cover to a
710
+ * replacement it DOES ("d" → "D", "c" → "ch" for Klingon). The replacement is
711
+ * itself converted through the normal table, so it must be fully mapped — a
712
+ * fallback that lands on another unmapped letter would just move the hole.
713
+ *
714
+ * Champollion ships NO fallbacks of its own: inventing orthographic
715
+ * adaptations — especially for a real language's orthography — is not ours to
716
+ * do. The docs list conventions with their sources; adopting one is a
717
+ * deliberate, per-project act.
718
+ *
719
+ * @param {object} fallbackMap - { sequence: replacement }
720
+ * @param {string} converterKey - Key into SCRIPT_CONVERTERS
721
+ * @throws {Error} naming the offending entry
722
+ */
723
+ function validateScriptFallback(fallbackMap, converterKey) {
724
+ if (fallbackMap == null) return;
725
+ if (typeof fallbackMap !== 'object' || Array.isArray(fallbackMap)) {
726
+ throw new Error(
727
+ `"scriptFallback" must be an object mapping letters to replacements, got ${Array.isArray(fallbackMap) ? 'array' : typeof fallbackMap}.`
728
+ );
729
+ }
730
+ for (const [from, to] of Object.entries(fallbackMap)) {
731
+ if (typeof to !== 'string' || to === '') {
732
+ throw new Error(
733
+ `"scriptFallback" entry "${from}": replacement must be a non-empty string, got ${JSON.stringify(to)}.`
734
+ );
735
+ }
736
+ if (from === '') {
737
+ throw new Error('"scriptFallback" has an empty-string key — nothing to replace.');
738
+ }
739
+ const holes = unmappedLetters(to, converterKey);
740
+ if (holes.length > 0) {
741
+ throw new Error(
742
+ `"scriptFallback" entry "${from}" → "${to}": the replacement itself contains ` +
743
+ `letter(s) the ${converterKey} converter cannot map (${holes.join(', ')}). ` +
744
+ 'A fallback must land on fully-mapped text.'
745
+ );
746
+ }
747
+ }
748
+ }
749
+
750
+ /**
751
+ * Apply a scriptFallback map to working-script text, longest keys first so
752
+ * "ck" wins over "c" + "k". Pure textual substitution — the result then runs
753
+ * through the normal conversion table.
754
+ *
755
+ * @param {string} text - Working-script text
756
+ * @param {object|null} fallbackMap - Validated { sequence: replacement } map
757
+ * @returns {string}
758
+ */
759
+ function applyScriptFallback(text, fallbackMap) {
760
+ if (!fallbackMap || typeof text !== 'string' || text === '') return text;
761
+ const keys = Object.keys(fallbackMap).sort((a, b) => b.length - a.length);
762
+ if (keys.length === 0) return text;
763
+
764
+ let out = '';
765
+ let i = 0;
766
+ while (i < text.length) {
767
+ let matched = null;
768
+ for (const k of keys) {
769
+ if (text.startsWith(k, i)) { matched = k; break; }
770
+ }
771
+ if (matched) {
772
+ out += fallbackMap[matched];
773
+ i += matched.length;
774
+ } else {
775
+ out += text[i];
776
+ i++;
777
+ }
778
+ }
779
+ return out;
780
+ }
781
+
782
+ // -----------------------------------------------------------------
783
+ // Coverage and reversal — making the converters honest
784
+ // -----------------------------------------------------------------
785
+
786
+ /**
787
+ * The set of source sequences a converter's map covers, longest first.
788
+ * Derived from the map itself so coverage can never drift from behaviour.
789
+ */
790
+ function mappedSequences(converterKey) {
791
+ const conv = SCRIPT_CONVERTERS[converterKey];
792
+ if (!conv?.map) return null;
793
+ return conv.map.map(([from]) => from).sort((a, b) => b.length - a.length);
794
+ }
795
+
796
+ /**
797
+ * Report the letters a converter would silently pass through untranslated.
798
+ *
799
+ * WHY: every converter here passes unmatched characters through, which is
800
+ * correct for spaces, digits and punctuation and wrong for letters. Klingon
801
+ * romanization has no `d`, `c`, `f`, `g`, `i`, `k`, `s`, `x` or `z`, so text
802
+ * containing them is not Klingon romanization — but the converter mapped what
803
+ * it recognised and emitted the rest as Latin, producing strings that are half
804
+ * pIqaD and half English with nothing raising a hand. 23 of the 32 affected
805
+ * strings found in the wild were this shape.
806
+ *
807
+ * Digits, whitespace and punctuation are legitimate passthrough and are never
808
+ * reported.
809
+ *
810
+ * @param {string} text - Text in the converter's working script
811
+ * @param {string} converterKey - Key into SCRIPT_CONVERTERS
812
+ * @returns {string[]} Distinct unmapped letters, in first-appearance order
813
+ */
814
+ function unmappedLetters(text, converterKey) {
815
+ const conv = SCRIPT_CONVERTERS[converterKey];
816
+ if (!conv || typeof text !== 'string' || text === '') return [];
817
+
818
+ // Kryptonian is arithmetic over A–Z with no table; every letter outside
819
+ // the basic Latin alphabet is unmapped.
820
+ if (!conv.map) {
821
+ const out = [];
822
+ for (const ch of text) {
823
+ if (!/\p{L}/u.test(ch)) continue;
824
+ if (/[A-Za-z]/.test(ch)) continue;
825
+ if (!out.includes(ch)) out.push(ch);
826
+ }
827
+ return out;
828
+ }
829
+
830
+ const sequences = mappedSequences(converterKey);
831
+ // Converters that lowercase (or uppercase) their input before matching must
832
+ // be probed in that same normalised form, or every capital reads as unmapped.
833
+ const probe = conv.converter === latinToTengwar || conv.converter === sroToSyllabics
834
+ ? text.toLowerCase()
835
+ : text;
836
+
837
+ const out = [];
838
+ let i = 0;
839
+ while (i < probe.length) {
840
+ let matched = 0;
841
+ for (const seq of sequences) {
842
+ if (probe.startsWith(seq, i)) { matched = seq.length; break; }
843
+ }
844
+ if (matched) { i += matched; continue; }
845
+ const ch = probe[i];
846
+ if (/\p{L}/u.test(ch) && !out.includes(ch)) out.push(ch);
847
+ i++;
848
+ }
849
+ return out;
850
+ }
851
+
852
+ /**
853
+ * Reverse a converter's output back to its working script.
854
+ *
855
+ * Used by `champollion repair-script` to undo conversions that should never
856
+ * have happened. Reversal is exact for pIqaD (the map is injective — only the
857
+ * straight and curly apostrophe share a codepoint, and both restore as `'`).
858
+ * Tengwar, Cree syllabics and Kryptonian normalise case on the way in, so the
859
+ * reverse cannot recover the original capitalisation; callers are told this
860
+ * via `caseLossy` rather than being left to discover it.
861
+ *
862
+ * @param {string} text - Converted text
863
+ * @param {string} converterKey - Key into SCRIPT_CONVERTERS
864
+ * @returns {{ reversed: string, caseLossy: boolean, unreversed: string[] }}
865
+ */
866
+ function reverseScript(text, converterKey) {
867
+ const conv = SCRIPT_CONVERTERS[converterKey];
868
+ if (!conv || typeof text !== 'string' || text === '') {
869
+ return { reversed: text, caseLossy: false, unreversed: [] };
870
+ }
871
+
872
+ const caseLossy = conv.converter !== romanizationToPiqad;
873
+
874
+ if (!conv.map) {
875
+ // Kryptonian: U+E100–E119 → A–Z.
876
+ const [lo, hi] = conv.puaRange;
877
+ let out = '';
878
+ const unreversed = [];
879
+ for (const ch of text) {
880
+ const cp = ch.codePointAt(0);
881
+ if (cp >= lo && cp <= hi) out += String.fromCharCode(65 + (cp - lo));
882
+ else {
883
+ out += ch;
884
+ if (isPrivateUse(cp) && !unreversed.includes(ch)) unreversed.push(ch);
885
+ }
886
+ }
887
+ return { reversed: out, caseLossy, unreversed };
888
+ }
889
+
890
+ // Build the reverse table. Later duplicate targets do not overwrite earlier
891
+ // ones, so `'` wins over `’` for the shared glottal-stop codepoint.
892
+ const reverse = new Map();
893
+ for (const [from, to] of conv.map) {
894
+ if (!reverse.has(to)) reverse.set(to, from);
895
+ }
896
+
897
+ let out = '';
898
+ const unreversed = [];
899
+ for (const ch of text) {
900
+ const hit = reverse.get(ch);
901
+ if (hit !== undefined) { out += hit; continue; }
902
+ out += ch;
903
+ const cp = ch.codePointAt(0);
904
+ if (isPrivateUse(cp) && !unreversed.includes(ch)) unreversed.push(ch);
905
+ }
906
+ return { reversed: out, caseLossy, unreversed };
907
+ }
908
+
909
+ /**
910
+ * Unicode Private Use Area membership — the BMP block plus both supplementary
911
+ * planes. `\p{Co}` in one predicate, without a regex per call.
912
+ */
913
+ function isPrivateUse(codePoint) {
914
+ return (codePoint >= 0xE000 && codePoint <= 0xF8FF)
915
+ || (codePoint >= 0xF0000 && codePoint <= 0xFFFFD)
916
+ || (codePoint >= 0x100000 && codePoint <= 0x10FFFD);
917
+ }
918
+
919
+ /**
920
+ * Convert text using the registered converter for a locale.
921
+ *
922
+ * `unmapped` lists letters the converter could not translate and passed
923
+ * through as-is. A non-empty `unmapped` means the input was not valid text in
924
+ * the converter's working script, and the output is a mix of both scripts —
925
+ * callers must treat it as a failure rather than writing it out.
926
+ *
927
+ * @param {string} text - Text in the source script
928
+ * @param {string} localeCode - Locale code (e.g., "crk", "sr")
929
+ * @returns {{ converted: string, converterUsed: string|null, unmapped: string[] }}
930
+ */
931
+ function convertScript(text, localeCode) {
932
+ const converter = SCRIPT_CONVERTERS[localeCode];
933
+ if (!converter) {
934
+ return { converted: text, converterUsed: null, unmapped: [] };
935
+ }
936
+
937
+ return {
938
+ converted: converter.converter(text),
939
+ converterUsed: `${converter.from} → ${converter.to}`,
940
+ unmapped: unmappedLetters(text, localeCode),
941
+ };
942
+ }
943
+
944
+ /**
945
+ * Check if a locale has a registered script converter.
946
+ *
947
+ * @param {string} localeCode - Locale code
948
+ * @returns {boolean}
949
+ */
950
+ function hasScriptConverter(localeCode) {
951
+ return localeCode in SCRIPT_CONVERTERS;
952
+ }
953
+
954
+ /**
955
+ * Get converter info for a locale (without the function reference).
956
+ * Safe for serialization into config/reports.
957
+ *
958
+ * @param {string} localeCode - Locale code
959
+ * @returns {object|null}
960
+ */
961
+ function getConverterInfo(localeCode) {
962
+ const conv = SCRIPT_CONVERTERS[localeCode];
963
+ if (!conv) return null;
964
+ const info = {
965
+ from: conv.from,
966
+ to: conv.to,
967
+ type: conv.type,
968
+ fromScript: conv.fromScript,
969
+ toScript: conv.toScript,
970
+ puaRange: conv.puaRange,
971
+ };
972
+ if (conv.fontNote) info.fontNote = conv.fontNote;
973
+ return info;
974
+ }
975
+
976
+ export {
977
+ sroToSyllabics,
978
+ latinToCyrillicSr,
979
+ romanizationToPiqad,
980
+ latinToTengwar,
981
+ latinToKryptonian,
982
+ convertScript,
983
+ hasScriptConverter,
984
+ getConverterInfo,
985
+ resolveTargetScript,
986
+ converterKeyForLocale,
987
+ formatScriptChoiceError,
988
+ validateScriptFallback,
989
+ applyScriptFallback,
990
+ unmappedLetters,
991
+ reverseScript,
992
+ isPrivateUse,
993
+ SCRIPT_CONVERTERS,
994
+ };