champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
package/lib/validate.js
ADDED
|
@@ -0,0 +1,510 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Translation Quality Gate — deterministic output validation.
|
|
3
|
+
*
|
|
4
|
+
* WHY: LLMs producing conlang/fictional translations often generate:
|
|
5
|
+
* - Repetitive nonsense ("Qo' Qo' Qo' Qo'") — hallucination loop
|
|
6
|
+
* - Drastically inflated output (300+ chars for a 5-char source) — padding
|
|
7
|
+
* - ASCII text for non-Latin scripts — wrong script entirely
|
|
8
|
+
* - Source text echoed back verbatim — lazy passthrough
|
|
9
|
+
* - Source text with the untranslatable letters DELETED, punctuation and
|
|
10
|
+
* spacing left standing ("low-resource nmt · tokenizers · nêhiyawêwin"
|
|
11
|
+
* → " · · êhiêi") — hollowing
|
|
12
|
+
*
|
|
13
|
+
* This module provides fast, deterministic checks that catch these failure
|
|
14
|
+
* modes BEFORE translations are written to locale files. Failed keys are
|
|
15
|
+
* logged loudly and excluded from the result.
|
|
16
|
+
*
|
|
17
|
+
* HOW IT WORKS:
|
|
18
|
+
* The sync loop calls `validateTranslations()` on the merged output of
|
|
19
|
+
* each language pair. Each key-value pair is checked against:
|
|
20
|
+
* 1. Repetition detector (trigram + long-8-gram frequency analysis,
|
|
21
|
+
* source-relative caps)
|
|
22
|
+
* 2. Length ratio check (source vs translated length)
|
|
23
|
+
* 3. Script compliance (non-Latin locales must produce non-ASCII)
|
|
24
|
+
* 4. Source echo check (translated value ≠ source value)
|
|
25
|
+
* 5. Content preservation (the output is not the source, hollowed out)
|
|
26
|
+
*
|
|
27
|
+
* Keys that fail any check are removed and logged as [GATE] failures.
|
|
28
|
+
* The caller receives only validated translations.
|
|
29
|
+
*
|
|
30
|
+
* CONFIGURATION:
|
|
31
|
+
* Per-language overrides can be set via the pair config:
|
|
32
|
+
* "languages": { "tlh": { "maxLengthRatio": 5, "requireNonLatin": true } }
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
import { getAllLanguageCodes, getLanguageCard } from './registers.js';
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Locales whose scripts are predominantly non-Latin.
|
|
39
|
+
*
|
|
40
|
+
* DERIVED FROM LANGUAGE CARDS — not hardcoded. At module load, we scan
|
|
41
|
+
* every registered language card and collect those with a non-Latin script.
|
|
42
|
+
* Adding a new card with script: "Geor" (Georgian) or "Cans" (Cree Syllabics)
|
|
43
|
+
* automatically includes it here — no manual set maintenance needed.
|
|
44
|
+
*
|
|
45
|
+
* WHY DYNAMIC: The old hardcoded set drifted from reality whenever a new
|
|
46
|
+
* language card was added. Georgian was added in the v5 refactor but the
|
|
47
|
+
* set already had 'ka' — what about Quechua ('qu', Latn)? Yoruba ('yo', Latn)?
|
|
48
|
+
* By reading the card data, we always match the source of truth.
|
|
49
|
+
*/
|
|
50
|
+
function _buildNonLatinSet() {
|
|
51
|
+
const set = new Set();
|
|
52
|
+
for (const code of getAllLanguageCodes()) {
|
|
53
|
+
const card = getLanguageCard(code);
|
|
54
|
+
if (!card) continue;
|
|
55
|
+
|
|
56
|
+
// Cards with non-Latin script are flagged — UNLESS they have a
|
|
57
|
+
// scriptConverter, meaning the LLM produces Latin output (e.g., SRO
|
|
58
|
+
// for Plains Cree) and script conversion is a post-processing step.
|
|
59
|
+
// The quality gate runs before conversion, so Latin output is correct.
|
|
60
|
+
if (card.script && card.script !== 'Latn' && !card.scriptConverter) {
|
|
61
|
+
set.add(code);
|
|
62
|
+
|
|
63
|
+
// Also add aliases so lookups like 'zh-CN' hit without base-locale fallback
|
|
64
|
+
if (Array.isArray(card.aliases)) {
|
|
65
|
+
for (const alias of card.aliases) {
|
|
66
|
+
set.add(alias);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
return set;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const NON_LATIN_LOCALES = _buildNonLatinSet();
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Default validation thresholds.
|
|
78
|
+
* These are intentionally generous — the goal is to catch gross failures,
|
|
79
|
+
* not nitpick edge cases. Tighter thresholds can be set per-language.
|
|
80
|
+
*/
|
|
81
|
+
const DEFAULT_THRESHOLDS = {
|
|
82
|
+
// Max ratio of translated length to source length before flagging.
|
|
83
|
+
// e.g., 4.0 means translated text can be up to 4x longer than source.
|
|
84
|
+
// Some languages (German, Finnish) legitimately produce longer text.
|
|
85
|
+
maxLengthRatio: 4.0,
|
|
86
|
+
|
|
87
|
+
// Min ratio of translated length to source length before flagging.
|
|
88
|
+
// Catches truncation/empty output masquerading as translation.
|
|
89
|
+
minLengthRatio: 0.1,
|
|
90
|
+
|
|
91
|
+
// Max percentage of repeated trigrams before flagging as hallucination.
|
|
92
|
+
// A hallucinated output like "Qo' Qo' Qo'" has ~100% repetition.
|
|
93
|
+
// Particle-heavy languages legitimately run high here (formal Tagalog
|
|
94
|
+
// measures 60-70% from kung/ng/mga/paano alone), which is why exceeding
|
|
95
|
+
// this cap is necessary but NOT sufficient to flag — see
|
|
96
|
+
// maxLongRepetitionRate below.
|
|
97
|
+
maxRepetitionRate: 0.60,
|
|
98
|
+
|
|
99
|
+
// Max percentage of repeated LONG (8-char) n-grams before flagging.
|
|
100
|
+
// Degeneration loops repeat long substrings, so they score ~100% at this
|
|
101
|
+
// window too; particle-heavy text repeats only short function words and
|
|
102
|
+
// stays low (correct Tagalog ~24%, deliberate phrase repetition ~45%).
|
|
103
|
+
// A repetition flag requires BOTH this and maxRepetitionRate to be
|
|
104
|
+
// exceeded.
|
|
105
|
+
maxLongRepetitionRate: 0.50,
|
|
106
|
+
|
|
107
|
+
// Whether to require non-ASCII characters for non-Latin locales.
|
|
108
|
+
// When true, a translation containing only ASCII for a CJK/Cyrillic/etc
|
|
109
|
+
// locale is flagged as wrong-script.
|
|
110
|
+
requireNonLatin: true,
|
|
111
|
+
|
|
112
|
+
// Min fraction of the source's CONTENT characters (letters + digits) the
|
|
113
|
+
// translation must retain before the hollowing check looks harder. This is
|
|
114
|
+
// deliberately NOT a standalone rule — see checkContentPreservation for why
|
|
115
|
+
// a bare density ratio cannot work.
|
|
116
|
+
minContentRetention: 0.35,
|
|
117
|
+
};
|
|
118
|
+
|
|
119
|
+
// Window size for the long-n-gram repetition confirmation signal.
|
|
120
|
+
const REPETITION_LONG_N = 8;
|
|
121
|
+
|
|
122
|
+
// A source with fewer content characters than this is too short to measure a
|
|
123
|
+
// meaningful retention ratio ("OK", "Blog", "npm"), and the empty/echo/script
|
|
124
|
+
// checks already cover that range.
|
|
125
|
+
const MIN_MEASURABLE_CONTENT = 6;
|
|
126
|
+
|
|
127
|
+
/** Letters and digits — the characters that actually carry meaning. */
|
|
128
|
+
const CONTENT_CHAR = /[\p{L}\p{N}]/u;
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Characters that are invisible AND survive String.prototype.trim().
|
|
132
|
+
*
|
|
133
|
+
* trim() strips White_Space only. The Cf (format) category — U+200B ZERO
|
|
134
|
+
* WIDTH SPACE, U+200E LEFT-TO-RIGHT MARK, U+2060 WORD JOINER, U+180E — is
|
|
135
|
+
* not White_Space, so a value built entirely from those has trim().length > 0
|
|
136
|
+
* and used to pass the empty check while rendering as a blank string on the
|
|
137
|
+
* page. That is the same corruption family as the URL incident.
|
|
138
|
+
*/
|
|
139
|
+
const INVISIBLE_NON_WHITESPACE = /\p{Cf}/gu;
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Extract the content characters (letters + digits) of a string.
|
|
143
|
+
*
|
|
144
|
+
* NFC-normalized first so a decomposed "ê" (e + U+0302) counts as one
|
|
145
|
+
* character on both sides of a comparison rather than one letter plus a
|
|
146
|
+
* combining mark.
|
|
147
|
+
*
|
|
148
|
+
* @param {string} text
|
|
149
|
+
* @returns {string[]} Content characters, in order
|
|
150
|
+
*/
|
|
151
|
+
function contentCharacters(text) {
|
|
152
|
+
return [...String(text).normalize('NFC')].filter(ch => CONTENT_CHAR.test(ch));
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Is `needle` a subsequence of `haystack` — i.e. can it be produced by
|
|
157
|
+
* DELETING characters from it, without reordering?
|
|
158
|
+
*
|
|
159
|
+
* Case-insensitive: the observed corruption preserves case, but a model that
|
|
160
|
+
* also lowercased while deleting is the same defect.
|
|
161
|
+
*
|
|
162
|
+
* @param {string[]} needle
|
|
163
|
+
* @param {string[]} haystack
|
|
164
|
+
* @returns {boolean}
|
|
165
|
+
*/
|
|
166
|
+
function isSubsequence(needle, haystack) {
|
|
167
|
+
let i = 0;
|
|
168
|
+
for (const ch of haystack) {
|
|
169
|
+
if (i < needle.length && needle[i].toLowerCase() === ch.toLowerCase()) i++;
|
|
170
|
+
}
|
|
171
|
+
return i === needle.length;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Detect a HOLLOWED translation: the source with its letters deleted.
|
|
176
|
+
*
|
|
177
|
+
* THE BUG THIS CATCHES, observed in production:
|
|
178
|
+
* "low-resource nmt · tokenizers · nêhiyawêwin" → " · · êhiêi"
|
|
179
|
+
* "the simple-builder approach" → " "
|
|
180
|
+
* Every letter the model had no vocabulary for was deleted and the source's
|
|
181
|
+
* punctuation and spacing skeleton was left standing. The result passed every
|
|
182
|
+
* existing check: not empty (after trim), not an echo, not repetitive, and at
|
|
183
|
+
* 33% of the source LENGTH it cleared minLengthRatio (0.1) comfortably.
|
|
184
|
+
*
|
|
185
|
+
* WHY DENSITY ALONE CANNOT WORK — the obvious rule ("reject below X% of the
|
|
186
|
+
* source's alphanumeric density") is unshippable, because legitimate dense
|
|
187
|
+
* scripts sit in exactly the same place:
|
|
188
|
+
*
|
|
189
|
+
* "low-resource nmt · …" → " · · êhiêi" 0.14 retained ← BUG
|
|
190
|
+
* "Getting started" → "入门" 0.14 retained ← CORRECT
|
|
191
|
+
* "Frequently asked …" → "常见问题" 0.17 retained ← CORRECT
|
|
192
|
+
*
|
|
193
|
+
* Any threshold that catches the first rejects Chinese, Japanese and Korean
|
|
194
|
+
* outright. What actually separates them is not how MUCH survived but WHERE
|
|
195
|
+
* it came from: the hollowed output is a subsequence of its own source, while
|
|
196
|
+
* a real translation shares essentially nothing with it.
|
|
197
|
+
*
|
|
198
|
+
* isSubsequence("êhiêi", "lowresourcenmttokenizersnêhiyawêwin") → true
|
|
199
|
+
* isSubsequence("入门", "gettingstarted") → false
|
|
200
|
+
*
|
|
201
|
+
* So a flag requires BOTH signals — the same necessary-but-not-sufficient
|
|
202
|
+
* design the repetition detector uses. Verified against real Klingon output
|
|
203
|
+
* from the same run ("Doing things with logic" → "meqmo' vay' vita'", 0.60
|
|
204
|
+
* retained, not a subsequence): correct conlang translation is unaffected.
|
|
205
|
+
*
|
|
206
|
+
* @param {string} source - Source value
|
|
207
|
+
* @param {string} translated - Candidate translation
|
|
208
|
+
* @param {number} minRetention - Retention floor below which the subsequence
|
|
209
|
+
* signal is consulted (DEFAULT_THRESHOLDS.minContentRetention)
|
|
210
|
+
* @returns {{ reason: string, retention: number }|null} Failure, or null if OK
|
|
211
|
+
*/
|
|
212
|
+
function checkContentPreservation(source, translated, minRetention = DEFAULT_THRESHOLDS.minContentRetention) {
|
|
213
|
+
const sourceContent = contentCharacters(source);
|
|
214
|
+
if (sourceContent.length < MIN_MEASURABLE_CONTENT) return null;
|
|
215
|
+
|
|
216
|
+
const targetContent = contentCharacters(translated);
|
|
217
|
+
|
|
218
|
+
// Total hollowing: the source carries real words and the output has no
|
|
219
|
+
// letter or digit at all. No language translates six letters into none, so
|
|
220
|
+
// this needs no second signal — and it is what catches a value built from
|
|
221
|
+
// punctuation, spaces, or invisible U+200B/U+200E characters.
|
|
222
|
+
if (targetContent.length === 0) {
|
|
223
|
+
return {
|
|
224
|
+
reason: 'no translatable content (every letter and digit removed from the source)',
|
|
225
|
+
retention: 0,
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
const retention = targetContent.length / sourceContent.length;
|
|
230
|
+
if (retention >= minRetention) return null;
|
|
231
|
+
|
|
232
|
+
// Below the floor — necessary, not sufficient. Confirm the output is the
|
|
233
|
+
// source with characters deleted rather than a legitimately terse
|
|
234
|
+
// translation in a denser script.
|
|
235
|
+
if (!isSubsequence(targetContent, sourceContent)) return null;
|
|
236
|
+
|
|
237
|
+
return {
|
|
238
|
+
reason:
|
|
239
|
+
`content deleted (only ${(retention * 100).toFixed(0)}% of the source's letters/digits remain, `
|
|
240
|
+
+ 'and the result is the source with characters removed — the model had no vocabulary for this string)',
|
|
241
|
+
retention,
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
// A deliberately repetitive source ("Every language, into every language.")
|
|
246
|
+
// licenses an equally repetitive translation: the effective caps are raised
|
|
247
|
+
// to the source's own measured repetition plus this margin.
|
|
248
|
+
const REPETITION_SOURCE_MARGIN = 0.10;
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* Validate a batch of translations and return only passing keys.
|
|
252
|
+
*
|
|
253
|
+
* @param {object} translations - Key → translated value map
|
|
254
|
+
* @param {object} sourceFlat - Key → source value map (for comparison)
|
|
255
|
+
* @param {object} pairConfig - Pair config (target locale, thresholds)
|
|
256
|
+
* @param {object} [options] - Override thresholds for testing
|
|
257
|
+
* @returns {{ validated: object, failures: Array<{ key: string, reason: string, value: string }> }}
|
|
258
|
+
*/
|
|
259
|
+
function validateTranslations(translations, sourceFlat, pairConfig, options = {}) {
|
|
260
|
+
const targetLocale = pairConfig.target || pairConfig.locale || '';
|
|
261
|
+
const isNonLatin = NON_LATIN_LOCALES.has(targetLocale) || NON_LATIN_LOCALES.has(targetLocale.split('-')[0]);
|
|
262
|
+
|
|
263
|
+
// Merge thresholds: options > pairConfig > defaults
|
|
264
|
+
const thresholds = {
|
|
265
|
+
maxLengthRatio: options.maxLengthRatio ?? pairConfig.maxLengthRatio ?? DEFAULT_THRESHOLDS.maxLengthRatio,
|
|
266
|
+
minLengthRatio: options.minLengthRatio ?? pairConfig.minLengthRatio ?? DEFAULT_THRESHOLDS.minLengthRatio,
|
|
267
|
+
maxRepetitionRate: options.maxRepetitionRate ?? pairConfig.maxRepetitionRate ?? DEFAULT_THRESHOLDS.maxRepetitionRate,
|
|
268
|
+
maxLongRepetitionRate: options.maxLongRepetitionRate ?? pairConfig.maxLongRepetitionRate ?? DEFAULT_THRESHOLDS.maxLongRepetitionRate,
|
|
269
|
+
requireNonLatin: options.requireNonLatin ?? pairConfig.requireNonLatin ?? DEFAULT_THRESHOLDS.requireNonLatin,
|
|
270
|
+
minContentRetention: options.minContentRetention ?? pairConfig.minContentRetention ?? DEFAULT_THRESHOLDS.minContentRetention,
|
|
271
|
+
};
|
|
272
|
+
|
|
273
|
+
const validated = {};
|
|
274
|
+
const failures = [];
|
|
275
|
+
|
|
276
|
+
for (const [key, translated] of Object.entries(translations)) {
|
|
277
|
+
const source = sourceFlat[key] || '';
|
|
278
|
+
|
|
279
|
+
// Skip non-string values (shouldn't happen, but defense-in-depth)
|
|
280
|
+
if (typeof translated !== 'string') {
|
|
281
|
+
failures.push({ key, reason: 'non-string value', value: String(translated) });
|
|
282
|
+
continue;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
// Check 1: Empty translation.
|
|
286
|
+
// Format characters are stripped BEFORE the emptiness test: trim() only
|
|
287
|
+
// removes White_Space, so a value made of U+200B / U+200E / U+2060 has
|
|
288
|
+
// trim().length > 0 and used to pass here while rendering as blank. That
|
|
289
|
+
// is how a hollowed value reached disk looking like " ".
|
|
290
|
+
if (translated.replace(INVISIBLE_NON_WHITESPACE, '').trim().length === 0) {
|
|
291
|
+
failures.push({
|
|
292
|
+
key,
|
|
293
|
+
reason: translated.trim().length === 0
|
|
294
|
+
? 'empty translation'
|
|
295
|
+
: 'empty translation (only invisible formatting characters)',
|
|
296
|
+
value: translated,
|
|
297
|
+
});
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// Check 2: Source echo — translated value is identical to source.
|
|
302
|
+
// EXEMPTION: Short strings (≤30 chars) that are mostly ASCII are likely
|
|
303
|
+
// proper nouns, brand names, or technical terms (e.g. "Blog", "GitHub",
|
|
304
|
+
// "npm", "CLI Reference") that legitimately stay in English across all
|
|
305
|
+
// languages. Rejecting these creates an infinite retry loop where the
|
|
306
|
+
// correct translation is rejected every time, burning API calls forever.
|
|
307
|
+
if (translated === source) {
|
|
308
|
+
const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
|
|
309
|
+
const isShortAscii = source.length <= 30 && asciiRatio > 0.8;
|
|
310
|
+
if (!isShortAscii) {
|
|
311
|
+
failures.push({ key, reason: 'source echo (identical to English)', value: translated });
|
|
312
|
+
continue;
|
|
313
|
+
}
|
|
314
|
+
// Short ASCII string echoed back — accept it as a valid translation
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
// Check 3: Repetition detection — catches hallucination loops.
|
|
318
|
+
// For pipe-delimited plural strings (e.g. "one doc|{count} docs"),
|
|
319
|
+
// measure each variant independently — plural forms legitimately
|
|
320
|
+
// share most of their text, which inflates the trigram count.
|
|
321
|
+
//
|
|
322
|
+
// Two-signal design: a flag requires a segment to exceed BOTH the
|
|
323
|
+
// trigram cap AND the long-8-gram cap. Trigram repetition alone
|
|
324
|
+
// false-positives on particle-heavy languages (correct formal Tagalog
|
|
325
|
+
// measures 60-70% from kung/ng/mga alone), but only degeneration loops
|
|
326
|
+
// repeat 8-char substrings at high rates. Both caps are also raised to
|
|
327
|
+
// the source's own repetition + margin, so deliberately repetitive copy
|
|
328
|
+
// licenses a matching translation.
|
|
329
|
+
const sourceSegments = splitPluralSegments(source);
|
|
330
|
+
const trigramCap = Math.max(
|
|
331
|
+
thresholds.maxRepetitionRate,
|
|
332
|
+
Math.max(...sourceSegments.map(seg => measureRepetition(seg))) + REPETITION_SOURCE_MARGIN
|
|
333
|
+
);
|
|
334
|
+
const longGramCap = Math.max(
|
|
335
|
+
thresholds.maxLongRepetitionRate,
|
|
336
|
+
Math.max(...sourceSegments.map(seg => measureRepetition(seg, REPETITION_LONG_N))) + REPETITION_SOURCE_MARGIN
|
|
337
|
+
);
|
|
338
|
+
const degenerateSegment = splitPluralSegments(translated)
|
|
339
|
+
.map(seg => ({
|
|
340
|
+
trigramRate: measureRepetition(seg),
|
|
341
|
+
longGramRate: measureRepetition(seg, REPETITION_LONG_N),
|
|
342
|
+
}))
|
|
343
|
+
.find(m => m.trigramRate > trigramCap && m.longGramRate > longGramCap);
|
|
344
|
+
if (degenerateSegment) {
|
|
345
|
+
failures.push({
|
|
346
|
+
key,
|
|
347
|
+
reason: `repetition hallucination (${(degenerateSegment.trigramRate * 100).toFixed(0)}% repeated trigrams, ${(degenerateSegment.longGramRate * 100).toFixed(0)}% repeated ${REPETITION_LONG_N}-grams)`,
|
|
348
|
+
value: translated.slice(0, 80) + (translated.length > 80 ? '...' : ''),
|
|
349
|
+
});
|
|
350
|
+
continue;
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
// Check 4: Length ratio — catches padding and truncation
|
|
354
|
+
if (source.length > 0) {
|
|
355
|
+
const ratio = translated.length / source.length;
|
|
356
|
+
if (ratio > thresholds.maxLengthRatio) {
|
|
357
|
+
failures.push({
|
|
358
|
+
key,
|
|
359
|
+
reason: `length inflation (${ratio.toFixed(1)}x source, max ${thresholds.maxLengthRatio}x)`,
|
|
360
|
+
value: translated.slice(0, 80) + (translated.length > 80 ? '...' : ''),
|
|
361
|
+
});
|
|
362
|
+
continue;
|
|
363
|
+
}
|
|
364
|
+
if (ratio < thresholds.minLengthRatio) {
|
|
365
|
+
failures.push({
|
|
366
|
+
key,
|
|
367
|
+
reason: `suspiciously short (${(ratio * 100).toFixed(0)}% of source length)`,
|
|
368
|
+
value: translated,
|
|
369
|
+
});
|
|
370
|
+
continue;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
// Check 5: Content preservation — catches a source hollowed of its
|
|
375
|
+
// letters. Runs AFTER the length ratio because a merely truncated output
|
|
376
|
+
// should report as truncation; what reaches here cleared that bar.
|
|
377
|
+
const hollowed = checkContentPreservation(source, translated, thresholds.minContentRetention);
|
|
378
|
+
if (hollowed) {
|
|
379
|
+
failures.push({ key, reason: hollowed.reason, value: translated });
|
|
380
|
+
continue;
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
// Check 6: Script compliance — non-Latin locales must have non-ASCII chars.
|
|
384
|
+
// EXEMPTIONS:
|
|
385
|
+
// - Strings with no translatable text after stripping ICU placeholders
|
|
386
|
+
// ({...}), digits, punctuation, and whitespace. e.g. "{authorName} - {nPosts}"
|
|
387
|
+
// or version strings like "3.2.0" have nothing to write in another script.
|
|
388
|
+
// - Short ASCII strings (≤30 chars, >80% ASCII) are likely proper nouns
|
|
389
|
+
// or brand names (e.g. "GitHub", "npm") that stay in English everywhere.
|
|
390
|
+
if (isNonLatin && thresholds.requireNonLatin) {
|
|
391
|
+
// Strip ICU placeholders, digits, punctuation, whitespace → what's left?
|
|
392
|
+
const translatableText = translated
|
|
393
|
+
.replace(/\{[^}]*\}/g, '') // ICU placeholders
|
|
394
|
+
.replace(/[\d\s\p{P}\p{S}]/gu, '') // digits, whitespace, punctuation, symbols
|
|
395
|
+
.trim();
|
|
396
|
+
const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
|
|
397
|
+
const isShortAsciiPropNoun = source.length <= 30 && asciiRatio > 0.8;
|
|
398
|
+
if (translatableText.length > 0 && isAsciiOnly(translated) && !isShortAsciiPropNoun) {
|
|
399
|
+
failures.push({
|
|
400
|
+
key,
|
|
401
|
+
reason: `wrong script (ASCII-only for ${targetLocale}, expected non-Latin characters)`,
|
|
402
|
+
value: translated.slice(0, 80),
|
|
403
|
+
});
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
// All checks passed
|
|
409
|
+
validated[key] = translated;
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
return { validated, failures };
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
/**
|
|
416
|
+
* Split a value into independently-measurable segments: pipe-delimited
|
|
417
|
+
* plural variants ("one doc|{count} docs") legitimately share most of
|
|
418
|
+
* their text, which would inflate a whole-string repetition measure.
|
|
419
|
+
*
|
|
420
|
+
* @param {string} text
|
|
421
|
+
* @returns {string[]} Trimmed segments (always at least one)
|
|
422
|
+
*/
|
|
423
|
+
function splitPluralSegments(text) {
|
|
424
|
+
return (text.includes('|') ? text.split('|') : [text]).map(seg => seg.trim());
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* Measure repetition rate using character n-gram frequency analysis.
|
|
429
|
+
*
|
|
430
|
+
* Splits the text into overlapping n-character grams and counts how
|
|
431
|
+
* many are repeated. A hallucinated output like "Qo' Qo' Qo'" produces
|
|
432
|
+
* a very high rate because the same grams appear over and over.
|
|
433
|
+
*
|
|
434
|
+
* The window size matters: at n=3 particle-heavy languages (Tagalog
|
|
435
|
+
* kung/ng/mga) score high on normal text, while at n=8 only genuinely
|
|
436
|
+
* looping output repeats — callers combine both signals.
|
|
437
|
+
*
|
|
438
|
+
* @param {string} text - Text to analyze
|
|
439
|
+
* @param {number} [n=3] - Gram window size in characters
|
|
440
|
+
* @returns {number} Repetition rate (0.0 = no repetition, 1.0 = all repeated)
|
|
441
|
+
*/
|
|
442
|
+
function measureRepetition(text, n = 3) {
|
|
443
|
+
// Short texts can't meaningfully repeat — skip
|
|
444
|
+
if (text.length < 12) return 0;
|
|
445
|
+
|
|
446
|
+
const grams = {};
|
|
447
|
+
let totalGrams = 0;
|
|
448
|
+
|
|
449
|
+
for (let i = 0; i <= text.length - n; i++) {
|
|
450
|
+
const gram = text.slice(i, i + n);
|
|
451
|
+
grams[gram] = (grams[gram] || 0) + 1;
|
|
452
|
+
totalGrams++;
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
if (totalGrams === 0) return 0;
|
|
456
|
+
|
|
457
|
+
// Count how many grams appear more than once
|
|
458
|
+
let repeatedCount = 0;
|
|
459
|
+
for (const count of Object.values(grams)) {
|
|
460
|
+
if (count > 1) {
|
|
461
|
+
repeatedCount += count;
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
return repeatedCount / totalGrams;
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
/**
|
|
469
|
+
* Check if a string contains only ASCII characters (codes 0-127).
|
|
470
|
+
* Used to detect wrong-script output for non-Latin locales.
|
|
471
|
+
*
|
|
472
|
+
* @param {string} text - Text to check
|
|
473
|
+
* @returns {boolean} True if text is ASCII-only
|
|
474
|
+
*/
|
|
475
|
+
function isAsciiOnly(text) {
|
|
476
|
+
// eslint-disable-next-line no-control-regex
|
|
477
|
+
return /^[\x00-\x7F]*$/.test(text);
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/**
|
|
481
|
+
* Log quality gate failures in a structured, actionable format.
|
|
482
|
+
*
|
|
483
|
+
* @param {Array<{ key: string, reason: string, value: string }>} failures
|
|
484
|
+
* @param {string} pairKey - e.g., "en:tlh"
|
|
485
|
+
*/
|
|
486
|
+
function logGateFailures(failures, pairKey) {
|
|
487
|
+
if (failures.length === 0) return;
|
|
488
|
+
|
|
489
|
+
console.error(`\n [GATE] ${pairKey}: ${failures.length} key(s) failed quality validation:`);
|
|
490
|
+
for (const { key, reason, value } of failures) {
|
|
491
|
+
console.error(` ✗ "${key}": ${reason}`);
|
|
492
|
+
if (value) {
|
|
493
|
+
console.error(` → "${value}"`);
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
console.error('');
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
export {
|
|
500
|
+
validateTranslations,
|
|
501
|
+
measureRepetition,
|
|
502
|
+
isAsciiOnly,
|
|
503
|
+
logGateFailures,
|
|
504
|
+
checkContentPreservation,
|
|
505
|
+
contentCharacters,
|
|
506
|
+
isSubsequence,
|
|
507
|
+
NON_LATIN_LOCALES,
|
|
508
|
+
DEFAULT_THRESHOLDS,
|
|
509
|
+
MIN_MEASURABLE_CONTENT,
|
|
510
|
+
};
|