champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,510 @@
1
+ /**
2
+ * Translation Quality Gate — deterministic output validation.
3
+ *
4
+ * WHY: LLMs producing conlang/fictional translations often generate:
5
+ * - Repetitive nonsense ("Qo' Qo' Qo' Qo'") — hallucination loop
6
+ * - Drastically inflated output (300+ chars for a 5-char source) — padding
7
+ * - ASCII text for non-Latin scripts — wrong script entirely
8
+ * - Source text echoed back verbatim — lazy passthrough
9
+ * - Source text with the untranslatable letters DELETED, punctuation and
10
+ * spacing left standing ("low-resource nmt · tokenizers · nêhiyawêwin"
11
+ * → " · · êhiêi") — hollowing
12
+ *
13
+ * This module provides fast, deterministic checks that catch these failure
14
+ * modes BEFORE translations are written to locale files. Failed keys are
15
+ * logged loudly and excluded from the result.
16
+ *
17
+ * HOW IT WORKS:
18
+ * The sync loop calls `validateTranslations()` on the merged output of
19
+ * each language pair. Each key-value pair is checked against:
20
+ * 1. Repetition detector (trigram + long-8-gram frequency analysis,
21
+ * source-relative caps)
22
+ * 2. Length ratio check (source vs translated length)
23
+ * 3. Script compliance (non-Latin locales must produce non-ASCII)
24
+ * 4. Source echo check (translated value ≠ source value)
25
+ * 5. Content preservation (the output is not the source, hollowed out)
26
+ *
27
+ * Keys that fail any check are removed and logged as [GATE] failures.
28
+ * The caller receives only validated translations.
29
+ *
30
+ * CONFIGURATION:
31
+ * Per-language overrides can be set via the pair config:
32
+ * "languages": { "tlh": { "maxLengthRatio": 5, "requireNonLatin": true } }
33
+ */
34
+
35
+ import { getAllLanguageCodes, getLanguageCard } from './registers.js';
36
+
37
+ /**
38
+ * Locales whose scripts are predominantly non-Latin.
39
+ *
40
+ * DERIVED FROM LANGUAGE CARDS — not hardcoded. At module load, we scan
41
+ * every registered language card and collect those with a non-Latin script.
42
+ * Adding a new card with script: "Geor" (Georgian) or "Cans" (Cree Syllabics)
43
+ * automatically includes it here — no manual set maintenance needed.
44
+ *
45
+ * WHY DYNAMIC: The old hardcoded set drifted from reality whenever a new
46
+ * language card was added. Georgian was added in the v5 refactor but the
47
+ * set already had 'ka' — what about Quechua ('qu', Latn)? Yoruba ('yo', Latn)?
48
+ * By reading the card data, we always match the source of truth.
49
+ */
50
+ function _buildNonLatinSet() {
51
+ const set = new Set();
52
+ for (const code of getAllLanguageCodes()) {
53
+ const card = getLanguageCard(code);
54
+ if (!card) continue;
55
+
56
+ // Cards with non-Latin script are flagged — UNLESS they have a
57
+ // scriptConverter, meaning the LLM produces Latin output (e.g., SRO
58
+ // for Plains Cree) and script conversion is a post-processing step.
59
+ // The quality gate runs before conversion, so Latin output is correct.
60
+ if (card.script && card.script !== 'Latn' && !card.scriptConverter) {
61
+ set.add(code);
62
+
63
+ // Also add aliases so lookups like 'zh-CN' hit without base-locale fallback
64
+ if (Array.isArray(card.aliases)) {
65
+ for (const alias of card.aliases) {
66
+ set.add(alias);
67
+ }
68
+ }
69
+ }
70
+ }
71
+ return set;
72
+ }
73
+
74
+ const NON_LATIN_LOCALES = _buildNonLatinSet();
75
+
76
+ /**
77
+ * Default validation thresholds.
78
+ * These are intentionally generous — the goal is to catch gross failures,
79
+ * not nitpick edge cases. Tighter thresholds can be set per-language.
80
+ */
81
+ const DEFAULT_THRESHOLDS = {
82
+ // Max ratio of translated length to source length before flagging.
83
+ // e.g., 4.0 means translated text can be up to 4x longer than source.
84
+ // Some languages (German, Finnish) legitimately produce longer text.
85
+ maxLengthRatio: 4.0,
86
+
87
+ // Min ratio of translated length to source length before flagging.
88
+ // Catches truncation/empty output masquerading as translation.
89
+ minLengthRatio: 0.1,
90
+
91
+ // Max percentage of repeated trigrams before flagging as hallucination.
92
+ // A hallucinated output like "Qo' Qo' Qo'" has ~100% repetition.
93
+ // Particle-heavy languages legitimately run high here (formal Tagalog
94
+ // measures 60-70% from kung/ng/mga/paano alone), which is why exceeding
95
+ // this cap is necessary but NOT sufficient to flag — see
96
+ // maxLongRepetitionRate below.
97
+ maxRepetitionRate: 0.60,
98
+
99
+ // Max percentage of repeated LONG (8-char) n-grams before flagging.
100
+ // Degeneration loops repeat long substrings, so they score ~100% at this
101
+ // window too; particle-heavy text repeats only short function words and
102
+ // stays low (correct Tagalog ~24%, deliberate phrase repetition ~45%).
103
+ // A repetition flag requires BOTH this and maxRepetitionRate to be
104
+ // exceeded.
105
+ maxLongRepetitionRate: 0.50,
106
+
107
+ // Whether to require non-ASCII characters for non-Latin locales.
108
+ // When true, a translation containing only ASCII for a CJK/Cyrillic/etc
109
+ // locale is flagged as wrong-script.
110
+ requireNonLatin: true,
111
+
112
+ // Min fraction of the source's CONTENT characters (letters + digits) the
113
+ // translation must retain before the hollowing check looks harder. This is
114
+ // deliberately NOT a standalone rule — see checkContentPreservation for why
115
+ // a bare density ratio cannot work.
116
+ minContentRetention: 0.35,
117
+ };
118
+
119
+ // Window size for the long-n-gram repetition confirmation signal.
120
+ const REPETITION_LONG_N = 8;
121
+
122
+ // A source with fewer content characters than this is too short to measure a
123
+ // meaningful retention ratio ("OK", "Blog", "npm"), and the empty/echo/script
124
+ // checks already cover that range.
125
+ const MIN_MEASURABLE_CONTENT = 6;
126
+
127
+ /** Letters and digits — the characters that actually carry meaning. */
128
+ const CONTENT_CHAR = /[\p{L}\p{N}]/u;
129
+
130
+ /**
131
+ * Characters that are invisible AND survive String.prototype.trim().
132
+ *
133
+ * trim() strips White_Space only. The Cf (format) category — U+200B ZERO
134
+ * WIDTH SPACE, U+200E LEFT-TO-RIGHT MARK, U+2060 WORD JOINER, U+180E — is
135
+ * not White_Space, so a value built entirely from those has trim().length > 0
136
+ * and used to pass the empty check while rendering as a blank string on the
137
+ * page. That is the same corruption family as the URL incident.
138
+ */
139
+ const INVISIBLE_NON_WHITESPACE = /\p{Cf}/gu;
140
+
141
+ /**
142
+ * Extract the content characters (letters + digits) of a string.
143
+ *
144
+ * NFC-normalized first so a decomposed "ê" (e + U+0302) counts as one
145
+ * character on both sides of a comparison rather than one letter plus a
146
+ * combining mark.
147
+ *
148
+ * @param {string} text
149
+ * @returns {string[]} Content characters, in order
150
+ */
151
+ function contentCharacters(text) {
152
+ return [...String(text).normalize('NFC')].filter(ch => CONTENT_CHAR.test(ch));
153
+ }
154
+
155
+ /**
156
+ * Is `needle` a subsequence of `haystack` — i.e. can it be produced by
157
+ * DELETING characters from it, without reordering?
158
+ *
159
+ * Case-insensitive: the observed corruption preserves case, but a model that
160
+ * also lowercased while deleting is the same defect.
161
+ *
162
+ * @param {string[]} needle
163
+ * @param {string[]} haystack
164
+ * @returns {boolean}
165
+ */
166
+ function isSubsequence(needle, haystack) {
167
+ let i = 0;
168
+ for (const ch of haystack) {
169
+ if (i < needle.length && needle[i].toLowerCase() === ch.toLowerCase()) i++;
170
+ }
171
+ return i === needle.length;
172
+ }
173
+
174
+ /**
175
+ * Detect a HOLLOWED translation: the source with its letters deleted.
176
+ *
177
+ * THE BUG THIS CATCHES, observed in production:
178
+ * "low-resource nmt · tokenizers · nêhiyawêwin" → " · · êhiêi"
179
+ * "the simple-builder approach" → " "
180
+ * Every letter the model had no vocabulary for was deleted and the source's
181
+ * punctuation and spacing skeleton was left standing. The result passed every
182
+ * existing check: not empty (after trim), not an echo, not repetitive, and at
183
+ * 33% of the source LENGTH it cleared minLengthRatio (0.1) comfortably.
184
+ *
185
+ * WHY DENSITY ALONE CANNOT WORK — the obvious rule ("reject below X% of the
186
+ * source's alphanumeric density") is unshippable, because legitimate dense
187
+ * scripts sit in exactly the same place:
188
+ *
189
+ * "low-resource nmt · …" → " · · êhiêi" 0.14 retained ← BUG
190
+ * "Getting started" → "入门" 0.14 retained ← CORRECT
191
+ * "Frequently asked …" → "常见问题" 0.17 retained ← CORRECT
192
+ *
193
+ * Any threshold that catches the first rejects Chinese, Japanese and Korean
194
+ * outright. What actually separates them is not how MUCH survived but WHERE
195
+ * it came from: the hollowed output is a subsequence of its own source, while
196
+ * a real translation shares essentially nothing with it.
197
+ *
198
+ * isSubsequence("êhiêi", "lowresourcenmttokenizersnêhiyawêwin") → true
199
+ * isSubsequence("入门", "gettingstarted") → false
200
+ *
201
+ * So a flag requires BOTH signals — the same necessary-but-not-sufficient
202
+ * design the repetition detector uses. Verified against real Klingon output
203
+ * from the same run ("Doing things with logic" → "meqmo' vay' vita'", 0.60
204
+ * retained, not a subsequence): correct conlang translation is unaffected.
205
+ *
206
+ * @param {string} source - Source value
207
+ * @param {string} translated - Candidate translation
208
+ * @param {number} minRetention - Retention floor below which the subsequence
209
+ * signal is consulted (DEFAULT_THRESHOLDS.minContentRetention)
210
+ * @returns {{ reason: string, retention: number }|null} Failure, or null if OK
211
+ */
212
+ function checkContentPreservation(source, translated, minRetention = DEFAULT_THRESHOLDS.minContentRetention) {
213
+ const sourceContent = contentCharacters(source);
214
+ if (sourceContent.length < MIN_MEASURABLE_CONTENT) return null;
215
+
216
+ const targetContent = contentCharacters(translated);
217
+
218
+ // Total hollowing: the source carries real words and the output has no
219
+ // letter or digit at all. No language translates six letters into none, so
220
+ // this needs no second signal — and it is what catches a value built from
221
+ // punctuation, spaces, or invisible U+200B/U+200E characters.
222
+ if (targetContent.length === 0) {
223
+ return {
224
+ reason: 'no translatable content (every letter and digit removed from the source)',
225
+ retention: 0,
226
+ };
227
+ }
228
+
229
+ const retention = targetContent.length / sourceContent.length;
230
+ if (retention >= minRetention) return null;
231
+
232
+ // Below the floor — necessary, not sufficient. Confirm the output is the
233
+ // source with characters deleted rather than a legitimately terse
234
+ // translation in a denser script.
235
+ if (!isSubsequence(targetContent, sourceContent)) return null;
236
+
237
+ return {
238
+ reason:
239
+ `content deleted (only ${(retention * 100).toFixed(0)}% of the source's letters/digits remain, `
240
+ + 'and the result is the source with characters removed — the model had no vocabulary for this string)',
241
+ retention,
242
+ };
243
+ }
244
+
245
+ // A deliberately repetitive source ("Every language, into every language.")
246
+ // licenses an equally repetitive translation: the effective caps are raised
247
+ // to the source's own measured repetition plus this margin.
248
+ const REPETITION_SOURCE_MARGIN = 0.10;
249
+
250
+ /**
251
+ * Validate a batch of translations and return only passing keys.
252
+ *
253
+ * @param {object} translations - Key → translated value map
254
+ * @param {object} sourceFlat - Key → source value map (for comparison)
255
+ * @param {object} pairConfig - Pair config (target locale, thresholds)
256
+ * @param {object} [options] - Override thresholds for testing
257
+ * @returns {{ validated: object, failures: Array<{ key: string, reason: string, value: string }> }}
258
+ */
259
+ function validateTranslations(translations, sourceFlat, pairConfig, options = {}) {
260
+ const targetLocale = pairConfig.target || pairConfig.locale || '';
261
+ const isNonLatin = NON_LATIN_LOCALES.has(targetLocale) || NON_LATIN_LOCALES.has(targetLocale.split('-')[0]);
262
+
263
+ // Merge thresholds: options > pairConfig > defaults
264
+ const thresholds = {
265
+ maxLengthRatio: options.maxLengthRatio ?? pairConfig.maxLengthRatio ?? DEFAULT_THRESHOLDS.maxLengthRatio,
266
+ minLengthRatio: options.minLengthRatio ?? pairConfig.minLengthRatio ?? DEFAULT_THRESHOLDS.minLengthRatio,
267
+ maxRepetitionRate: options.maxRepetitionRate ?? pairConfig.maxRepetitionRate ?? DEFAULT_THRESHOLDS.maxRepetitionRate,
268
+ maxLongRepetitionRate: options.maxLongRepetitionRate ?? pairConfig.maxLongRepetitionRate ?? DEFAULT_THRESHOLDS.maxLongRepetitionRate,
269
+ requireNonLatin: options.requireNonLatin ?? pairConfig.requireNonLatin ?? DEFAULT_THRESHOLDS.requireNonLatin,
270
+ minContentRetention: options.minContentRetention ?? pairConfig.minContentRetention ?? DEFAULT_THRESHOLDS.minContentRetention,
271
+ };
272
+
273
+ const validated = {};
274
+ const failures = [];
275
+
276
+ for (const [key, translated] of Object.entries(translations)) {
277
+ const source = sourceFlat[key] || '';
278
+
279
+ // Skip non-string values (shouldn't happen, but defense-in-depth)
280
+ if (typeof translated !== 'string') {
281
+ failures.push({ key, reason: 'non-string value', value: String(translated) });
282
+ continue;
283
+ }
284
+
285
+ // Check 1: Empty translation.
286
+ // Format characters are stripped BEFORE the emptiness test: trim() only
287
+ // removes White_Space, so a value made of U+200B / U+200E / U+2060 has
288
+ // trim().length > 0 and used to pass here while rendering as blank. That
289
+ // is how a hollowed value reached disk looking like " ".
290
+ if (translated.replace(INVISIBLE_NON_WHITESPACE, '').trim().length === 0) {
291
+ failures.push({
292
+ key,
293
+ reason: translated.trim().length === 0
294
+ ? 'empty translation'
295
+ : 'empty translation (only invisible formatting characters)',
296
+ value: translated,
297
+ });
298
+ continue;
299
+ }
300
+
301
+ // Check 2: Source echo — translated value is identical to source.
302
+ // EXEMPTION: Short strings (≤30 chars) that are mostly ASCII are likely
303
+ // proper nouns, brand names, or technical terms (e.g. "Blog", "GitHub",
304
+ // "npm", "CLI Reference") that legitimately stay in English across all
305
+ // languages. Rejecting these creates an infinite retry loop where the
306
+ // correct translation is rejected every time, burning API calls forever.
307
+ if (translated === source) {
308
+ const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
309
+ const isShortAscii = source.length <= 30 && asciiRatio > 0.8;
310
+ if (!isShortAscii) {
311
+ failures.push({ key, reason: 'source echo (identical to English)', value: translated });
312
+ continue;
313
+ }
314
+ // Short ASCII string echoed back — accept it as a valid translation
315
+ }
316
+
317
+ // Check 3: Repetition detection — catches hallucination loops.
318
+ // For pipe-delimited plural strings (e.g. "one doc|{count} docs"),
319
+ // measure each variant independently — plural forms legitimately
320
+ // share most of their text, which inflates the trigram count.
321
+ //
322
+ // Two-signal design: a flag requires a segment to exceed BOTH the
323
+ // trigram cap AND the long-8-gram cap. Trigram repetition alone
324
+ // false-positives on particle-heavy languages (correct formal Tagalog
325
+ // measures 60-70% from kung/ng/mga alone), but only degeneration loops
326
+ // repeat 8-char substrings at high rates. Both caps are also raised to
327
+ // the source's own repetition + margin, so deliberately repetitive copy
328
+ // licenses a matching translation.
329
+ const sourceSegments = splitPluralSegments(source);
330
+ const trigramCap = Math.max(
331
+ thresholds.maxRepetitionRate,
332
+ Math.max(...sourceSegments.map(seg => measureRepetition(seg))) + REPETITION_SOURCE_MARGIN
333
+ );
334
+ const longGramCap = Math.max(
335
+ thresholds.maxLongRepetitionRate,
336
+ Math.max(...sourceSegments.map(seg => measureRepetition(seg, REPETITION_LONG_N))) + REPETITION_SOURCE_MARGIN
337
+ );
338
+ const degenerateSegment = splitPluralSegments(translated)
339
+ .map(seg => ({
340
+ trigramRate: measureRepetition(seg),
341
+ longGramRate: measureRepetition(seg, REPETITION_LONG_N),
342
+ }))
343
+ .find(m => m.trigramRate > trigramCap && m.longGramRate > longGramCap);
344
+ if (degenerateSegment) {
345
+ failures.push({
346
+ key,
347
+ reason: `repetition hallucination (${(degenerateSegment.trigramRate * 100).toFixed(0)}% repeated trigrams, ${(degenerateSegment.longGramRate * 100).toFixed(0)}% repeated ${REPETITION_LONG_N}-grams)`,
348
+ value: translated.slice(0, 80) + (translated.length > 80 ? '...' : ''),
349
+ });
350
+ continue;
351
+ }
352
+
353
+ // Check 4: Length ratio — catches padding and truncation
354
+ if (source.length > 0) {
355
+ const ratio = translated.length / source.length;
356
+ if (ratio > thresholds.maxLengthRatio) {
357
+ failures.push({
358
+ key,
359
+ reason: `length inflation (${ratio.toFixed(1)}x source, max ${thresholds.maxLengthRatio}x)`,
360
+ value: translated.slice(0, 80) + (translated.length > 80 ? '...' : ''),
361
+ });
362
+ continue;
363
+ }
364
+ if (ratio < thresholds.minLengthRatio) {
365
+ failures.push({
366
+ key,
367
+ reason: `suspiciously short (${(ratio * 100).toFixed(0)}% of source length)`,
368
+ value: translated,
369
+ });
370
+ continue;
371
+ }
372
+ }
373
+
374
+ // Check 5: Content preservation — catches a source hollowed of its
375
+ // letters. Runs AFTER the length ratio because a merely truncated output
376
+ // should report as truncation; what reaches here cleared that bar.
377
+ const hollowed = checkContentPreservation(source, translated, thresholds.minContentRetention);
378
+ if (hollowed) {
379
+ failures.push({ key, reason: hollowed.reason, value: translated });
380
+ continue;
381
+ }
382
+
383
+ // Check 6: Script compliance — non-Latin locales must have non-ASCII chars.
384
+ // EXEMPTIONS:
385
+ // - Strings with no translatable text after stripping ICU placeholders
386
+ // ({...}), digits, punctuation, and whitespace. e.g. "{authorName} - {nPosts}"
387
+ // or version strings like "3.2.0" have nothing to write in another script.
388
+ // - Short ASCII strings (≤30 chars, >80% ASCII) are likely proper nouns
389
+ // or brand names (e.g. "GitHub", "npm") that stay in English everywhere.
390
+ if (isNonLatin && thresholds.requireNonLatin) {
391
+ // Strip ICU placeholders, digits, punctuation, whitespace → what's left?
392
+ const translatableText = translated
393
+ .replace(/\{[^}]*\}/g, '') // ICU placeholders
394
+ .replace(/[\d\s\p{P}\p{S}]/gu, '') // digits, whitespace, punctuation, symbols
395
+ .trim();
396
+ const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
397
+ const isShortAsciiPropNoun = source.length <= 30 && asciiRatio > 0.8;
398
+ if (translatableText.length > 0 && isAsciiOnly(translated) && !isShortAsciiPropNoun) {
399
+ failures.push({
400
+ key,
401
+ reason: `wrong script (ASCII-only for ${targetLocale}, expected non-Latin characters)`,
402
+ value: translated.slice(0, 80),
403
+ });
404
+ continue;
405
+ }
406
+ }
407
+
408
+ // All checks passed
409
+ validated[key] = translated;
410
+ }
411
+
412
+ return { validated, failures };
413
+ }
414
+
415
+ /**
416
+ * Split a value into independently-measurable segments: pipe-delimited
417
+ * plural variants ("one doc|{count} docs") legitimately share most of
418
+ * their text, which would inflate a whole-string repetition measure.
419
+ *
420
+ * @param {string} text
421
+ * @returns {string[]} Trimmed segments (always at least one)
422
+ */
423
+ function splitPluralSegments(text) {
424
+ return (text.includes('|') ? text.split('|') : [text]).map(seg => seg.trim());
425
+ }
426
+
427
+ /**
428
+ * Measure repetition rate using character n-gram frequency analysis.
429
+ *
430
+ * Splits the text into overlapping n-character grams and counts how
431
+ * many are repeated. A hallucinated output like "Qo' Qo' Qo'" produces
432
+ * a very high rate because the same grams appear over and over.
433
+ *
434
+ * The window size matters: at n=3 particle-heavy languages (Tagalog
435
+ * kung/ng/mga) score high on normal text, while at n=8 only genuinely
436
+ * looping output repeats — callers combine both signals.
437
+ *
438
+ * @param {string} text - Text to analyze
439
+ * @param {number} [n=3] - Gram window size in characters
440
+ * @returns {number} Repetition rate (0.0 = no repetition, 1.0 = all repeated)
441
+ */
442
+ function measureRepetition(text, n = 3) {
443
+ // Short texts can't meaningfully repeat — skip
444
+ if (text.length < 12) return 0;
445
+
446
+ const grams = {};
447
+ let totalGrams = 0;
448
+
449
+ for (let i = 0; i <= text.length - n; i++) {
450
+ const gram = text.slice(i, i + n);
451
+ grams[gram] = (grams[gram] || 0) + 1;
452
+ totalGrams++;
453
+ }
454
+
455
+ if (totalGrams === 0) return 0;
456
+
457
+ // Count how many grams appear more than once
458
+ let repeatedCount = 0;
459
+ for (const count of Object.values(grams)) {
460
+ if (count > 1) {
461
+ repeatedCount += count;
462
+ }
463
+ }
464
+
465
+ return repeatedCount / totalGrams;
466
+ }
467
+
468
+ /**
469
+ * Check if a string contains only ASCII characters (codes 0-127).
470
+ * Used to detect wrong-script output for non-Latin locales.
471
+ *
472
+ * @param {string} text - Text to check
473
+ * @returns {boolean} True if text is ASCII-only
474
+ */
475
+ function isAsciiOnly(text) {
476
+ // eslint-disable-next-line no-control-regex
477
+ return /^[\x00-\x7F]*$/.test(text);
478
+ }
479
+
480
+ /**
481
+ * Log quality gate failures in a structured, actionable format.
482
+ *
483
+ * @param {Array<{ key: string, reason: string, value: string }>} failures
484
+ * @param {string} pairKey - e.g., "en:tlh"
485
+ */
486
+ function logGateFailures(failures, pairKey) {
487
+ if (failures.length === 0) return;
488
+
489
+ console.error(`\n [GATE] ${pairKey}: ${failures.length} key(s) failed quality validation:`);
490
+ for (const { key, reason, value } of failures) {
491
+ console.error(` ✗ "${key}": ${reason}`);
492
+ if (value) {
493
+ console.error(` → "${value}"`);
494
+ }
495
+ }
496
+ console.error('');
497
+ }
498
+
499
+ export {
500
+ validateTranslations,
501
+ measureRepetition,
502
+ isAsciiOnly,
503
+ logGateFailures,
504
+ checkContentPreservation,
505
+ contentCharacters,
506
+ isSubsequence,
507
+ NON_LATIN_LOCALES,
508
+ DEFAULT_THRESHOLDS,
509
+ MIN_MEASURABLE_CONTENT,
510
+ };