champollion 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/README.md +41 -26
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +34 -0
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +286 -85
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +632 -125
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +15 -9
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +194 -35
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +6 -1
  100. package/lib/seal.mjs +4 -3
  101. package/lib/sealed-qualifier.mjs +1 -1
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +1 -1
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/cards-fallback.json +1 -1
  123. package/shared/curated-orthography-conventions.json +26 -8
  124. package/shared/gettext-plural-forms.json +45 -0
  125. package/shared/method-registry.json +2 -0
  126. package/shared/metric-registry.json +96 -18
  127. package/shared/schemas/champollion-plugin.schema.json +4 -0
  128. package/shared/schemas/corpora-card.schema.json +8 -2
  129. package/shared/schemas/method-index-record.schema.json +67 -0
  130. package/shared/schemas/method-registry.schema.json +4 -0
  131. package/shared/schemas/metric-registry.schema.json +55 -1
  132. package/shared/docent/corpus.json +0 -11739
package/lib/validate.js CHANGED
@@ -21,8 +21,26 @@
21
21
  * source-relative caps)
22
22
  * 2. Length ratio check (source vs translated length)
23
23
  * 3. Script compliance (non-Latin locales must produce non-ASCII)
24
- * 4. Source echo check (translated value ≠ source value)
24
+ * 4. Source echo check (translated value ≠ source value — and, for a
25
+ * source of 3+ words, not the source with only case, diacritics,
26
+ * spacing or invisible characters changed: see isDisguisedEcho)
25
27
  * 5. Content preservation (the output is not the source, hollowed out)
28
+ * 6. ICU / placeholder structure (lib/icu-structure.js): argument names,
29
+ * plural/select keywords, selectors, # and printf conversions are
30
+ * code — only the text inside the branches may change; markup tags
31
+ * are code too (checkMarkup: same tags opened, closed, nested); and
32
+ * i18next {{…}} / single-brace / tag tokens by the rule verify reports
33
+ * them by (lib/placeholders.js placeholderChanges); and a sentence
34
+ * break the translation inserts right beside a placeholder where the
35
+ * source has none (placeholderSentenceBreaks)
36
+ *
37
+ * 7. Plural forms (lib/icu-structure.js pluralGaps): a plural message
38
+ * that passed everything else but has no branch for a category the
39
+ * target language uses for ordinary counts (Russian few/many) is a
40
+ * SOFT failure (`pluralGap`): the caller asks once more, naming the
41
+ * missing forms; a second answer without them is accepted
42
+ * (acceptPluralGaps) and reported — never a retry loop, and never a
43
+ * form made up by the tool.
26
44
  *
27
45
  * Keys that fail any check are removed and logged as [GATE] failures.
28
46
  * The caller receives only validated translations.
@@ -33,6 +51,9 @@
33
51
  */
34
52
 
35
53
  import { getAllLanguageCodes, getLanguageCard } from './registers.js';
54
+ import { output } from './output.js';
55
+ import { checkICUStructure, checkMarkup, pluralGaps, describeCategories, pluralBranchPairs, pluralBranchTexts } from './icu-structure.js';
56
+ import { placeholderChanges, placeholderSentenceBreaks, PLACEHOLDER_SYNTAX_NAMES } from './placeholders.js';
36
57
 
37
58
  /**
38
59
  * Locales whose scripts are predominantly non-Latin.
@@ -242,11 +263,210 @@ function checkContentPreservation(source, translated, minRetention = DEFAULT_THR
242
263
  };
243
264
  }
244
265
 
266
+ // ---------------------------------------------------------------------------
267
+ // Disguised echoes — the source handed back with accents sprinkled on
268
+ // ---------------------------------------------------------------------------
269
+
270
+ /**
271
+ * Fewest words (whitespace-delimited, each carrying a letter) a source needs
272
+ * before a FOLDED match counts as an echo. Below it, only exact equality does.
273
+ *
274
+ * WHY A FLOOR: in a Latin-script target a short legitimate translation can
275
+ * differ from English only by accents — 'cafe' → 'café', 'Resume' → 'Résumé',
276
+ * 'Cafe Menu' → 'Café Menu' — and refusing those would cost users a retry and
277
+ * then the key. Three independent words that all survive translation
278
+ * unchanged apart from accents and case is what a copy looks like.
279
+ */
280
+ const MIN_FOLDED_ECHO_WORDS = 3;
281
+
282
+ /** Combining marks (Mn) and invisible format characters (Cf). */
283
+ const FOLD_STRIP = /[\p{Mn}\p{Cf}]/gu;
284
+
285
+ /**
286
+ * The form two strings are compared in to decide "is this output the source?".
287
+ *
288
+ * Twin of the fold in arena/mt_eval_harness/text_compare.py (its steps 1–4,
289
+ * `_fold` / token_compare_key, plus its whitespace collapse), so the CLI gate
290
+ * sees through the same disguises the harness does:
291
+ * 1. NFKD — precomposed letters split into base + combining mark, and
292
+ * compatibility forms (fullwidth letters, ligatures) fold to plain ones;
293
+ * 2. case folding (JavaScript has no casefold(); upper-then-lower makes the
294
+ * same equality decisions for this purpose: 'ß' and 'ss' meet, and the
295
+ * Greek sigma forms meet — JS lands on the contextual 'ς' where Python
296
+ * lands on 'σ', but both sides of a comparison land together), then
297
+ * NFKD again because a case mapping can produce a decomposable
298
+ * character;
299
+ * 3. drop combining marks (Mn: 'á' → 'a'; 'ŋ' is a letter and stays) and
300
+ * format characters (Cf: zero-width space/joiners, bidi marks, soft
301
+ * hyphen) — they change the bytes, not the text a reader sees;
302
+ * 4. collapse whitespace runs and trim.
303
+ * Punctuation is compared AS WRITTEN (the harness's per-token rule). Its
304
+ * whole-string form also turns punctuation into spaces; this gate does not —
305
+ * a gate that REFUSES output (and spends a retry) keeps the narrower fold, so
306
+ * a copy that also edits its punctuation is not caught here, as before.
307
+ *
308
+ * @param {string} text
309
+ * @returns {string}
310
+ */
311
+ function foldForEchoCompare(text) {
312
+ return String(text)
313
+ .normalize('NFKD')
314
+ .toUpperCase()
315
+ .toLowerCase()
316
+ .normalize('NFKD')
317
+ .replace(FOLD_STRIP, '')
318
+ .replace(/\s+/gu, ' ')
319
+ .trim();
320
+ }
321
+
322
+ /**
323
+ * How many words carrying a letter `text` has — after setting aside what is
324
+ * code rather than words: ICU/brace placeholders ("{count}"), printf
325
+ * conversions ("%s", "%1$d") and markup tags ("<b>", "</a>").
326
+ *
327
+ * @param {string} text
328
+ * @returns {number}
329
+ */
330
+ function letterWordCount(text) {
331
+ return String(text)
332
+ .replace(/\{[^}]*\}/g, ' ')
333
+ .replace(/%(\d+\$)?[-+ 0#]*\d*(\.\d+)?[a-zA-Z@]/g, ' ')
334
+ .replace(/<[^>]*>/g, ' ')
335
+ .split(/\s+/u)
336
+ .filter((w) => /\p{L}/u.test(w))
337
+ .length;
338
+ }
339
+
340
+ /**
341
+ * Is `translated` the source in disguise — identical once case, diacritics,
342
+ * spacing and invisible format characters are folded away, without being
343
+ * byte-identical? ('Thank you very much!' → 'Thánk yóú véry múch!')
344
+ *
345
+ * Only for a source of MIN_FOLDED_ECHO_WORDS+ words with letters; a shorter
346
+ * source keeps exact equality as its only echo test, exactly as before.
347
+ * An exact copy is NOT a disguised echo — it keeps its own rule (and the
348
+ * short-name lane), so this predicate changes nothing for it.
349
+ *
350
+ * Why the short-name lane does not apply here: that lane exists because a
351
+ * short value is often a NAME correctly kept as written ("GitHub", "Curtis
352
+ * Forbes"). A disguised copy is by construction not kept as written — it was
353
+ * altered without being translated — so it is refused like any long echo.
354
+ *
355
+ * Exported so an on-disk auditor that adopts it applies the same floor.
356
+ *
357
+ * @param {string} source
358
+ * @param {string} translated
359
+ * @returns {boolean}
360
+ */
361
+ function isDisguisedEcho(source, translated) {
362
+ if (typeof source !== 'string' || typeof translated !== 'string') return false;
363
+ if (translated === source) return false;
364
+ if (letterWordCount(source) < MIN_FOLDED_ECHO_WORDS) return false;
365
+ return foldForEchoCompare(translated) === foldForEchoCompare(source);
366
+ }
367
+
368
+ /**
369
+ * The first echo among a plural message's branches (validateTranslations
370
+ * check 2c), as a failure record's fields, or null.
371
+ *
372
+ * @returns {{ reason: string, disguisedEcho?: true, nameOrLabel?: true }|null}
373
+ */
374
+ function pluralBranchEcho(source, translated, { isNonLatin = false, requireNonLatin = true, acceptLatinNames = false, protectedTerms = [] } = {}) {
375
+ const pairs = pluralBranchPairs(source, translated);
376
+ if (pairs.length === 0) return null;
377
+ const exact = [];
378
+ const disguised = [];
379
+ const nameLike = [];
380
+ for (const p of pairs) {
381
+ const src = p.source.trim();
382
+ const tgt = p.translated.trim();
383
+ if (!/\p{L}/u.test(src) || isProtectedTermValue(tgt, protectedTerms)) continue;
384
+ if (tgt === src) {
385
+ const asciiRatio = src.replace(/[^\x20-\x7E]/g, '').length / Math.max(src.length, 1);
386
+ const isShortAscii = src.length <= 30 && asciiRatio > 0.8;
387
+ if (!isShortAscii) exact.push(p.selector);
388
+ else if (isNonLatin && requireNonLatin && !acceptLatinNames) nameLike.push(p.selector);
389
+ } else if (isDisguisedEcho(src, tgt)) {
390
+ disguised.push(p.selector);
391
+ }
392
+ }
393
+ const forms = (sels) => [...new Set(sels)].map(x => `"${x}"`).join(', ');
394
+ if (disguised.length > 0) {
395
+ return { reason: `${DISGUISED_ECHO} — in the plural form(s) ${forms(disguised)}`, disguisedEcho: true };
396
+ }
397
+ if (exact.length > 0) {
398
+ return { reason: `source echo (identical to English) in the plural form(s) ${forms(exact)}` };
399
+ }
400
+ if (nameLike.length > 0) {
401
+ return { reason: `${LATIN_NAME_OR_LABEL} (plural form(s) ${forms(nameLike)})`, nameOrLabel: true };
402
+ }
403
+ return null;
404
+ }
405
+
406
+ /** Gate reason for a disguised echo. */
407
+ const DISGUISED_ECHO =
408
+ 'source echo (the source handed back with only case, accents, spacing or invisible characters changed — not a translation)';
409
+
245
410
  // A deliberately repetitive source ("Every language, into every language.")
246
411
  // licenses an equally repetitive translation: the effective caps are raised
247
412
  // to the source's own measured repetition plus this margin.
248
413
  const REPETITION_SOURCE_MARGIN = 0.10;
249
414
 
415
+ /** Gate reason for a short Latin-script value in a non-Latin target. */
416
+ const LATIN_NAME_OR_LABEL = 'kept in Latin script — a name, or a label left untranslated?';
417
+
418
+ /**
419
+ * Is this value made ONLY of names the project declared (config.protectedTerms)?
420
+ *
421
+ * Strips every declared term, then ICU placeholders, digits, punctuation,
422
+ * symbols and whitespace; nothing left means the value is a name (or names)
423
+ * kept as written — correct in every script, never an "untranslated" error.
424
+ * "Curtis Forbes", "Game Day Suits", "Curtis Forbes · Game Day Suits" all
425
+ * qualify; "Curtis Forbes's portfolio" does not (the rest needs translating).
426
+ *
427
+ * @param {string} value
428
+ * @param {string[]} [protectedTerms]
429
+ * @returns {boolean}
430
+ */
431
+ function isProtectedTermValue(value, protectedTerms = []) {
432
+ if (typeof value !== 'string' || !protectedTerms || protectedTerms.length === 0) return false;
433
+ let rest = value;
434
+ let matched = false;
435
+ // Longest first, so "Game Day Suits Ltd" is removed before "Game Day Suits".
436
+ for (const term of [...protectedTerms].sort((a, b) => b.length - a.length)) {
437
+ if (typeof term === 'string' && term && rest.includes(term)) {
438
+ rest = rest.split(term).join(' ');
439
+ matched = true;
440
+ }
441
+ }
442
+ if (!matched) return false;
443
+ return rest.replace(/\{[^}]*\}/g, '').replace(/[\d\s\p{P}\p{S}]/gu, '').length === 0;
444
+ }
445
+
446
+ /**
447
+ * Why the quality gate refuses one Markdown block or front-matter field, or
448
+ * null when it passes — the key-value gate's own checks (empty, source echo,
449
+ * disguised echo, repetition, length inflation and truncation, hollowing,
450
+ * script), so a short heading turned into a sentence is refused in a
451
+ * newsletter exactly as it is for an app key (Round 7, school persona: "##
452
+ * Feast" became a full sentence and was accepted, while the same output for
453
+ * the key Nav.home was refused at 9.5x). Structure (code, links, markup) is
454
+ * checked by the lanes through their protected placeholders; a short
455
+ * Latin-script value kept as written is a name, as on the key-value lane's
456
+ * second ask (there is no per-block retry to ask once first).
457
+ *
458
+ * @param {string} source - The block or field's source text (placeholders restored)
459
+ * @param {string} value - The translation
460
+ * @param {object} [pairConfig] - The pair (target locale, thresholds, protectedTerms)
461
+ * @returns {string|null}
462
+ */
463
+ function contentGateFault(source, value, pairConfig = {}) {
464
+ if (typeof source !== 'string' || typeof value !== 'string') return null;
465
+ const { failures } = validateTranslations({ block: value }, { block: source }, pairConfig || {},
466
+ { prose: true, acceptLatinNames: true, acceptPluralGaps: true });
467
+ return failures.length > 0 ? failures[0].reason : null;
468
+ }
469
+
250
470
  /**
251
471
  * Validate a batch of translations and return only passing keys.
252
472
  *
@@ -272,6 +492,17 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
272
492
 
273
493
  const validated = {};
274
494
  const failures = [];
495
+ // config.protectedTerms rides the pair config (lib/pairs.js).
496
+ const protectedTerms = options.protectedTerms ?? pairConfig.protectedTerms ?? [];
497
+ // Second pass of the name-or-label retry: a short Latin-script value the
498
+ // model returned twice is accepted as a name.
499
+ const acceptLatinNames = options.acceptLatinNames === true;
500
+ // Second pass of the plural-forms retry: accepted, and reported by the caller.
501
+ const acceptPluralGaps = options.acceptPluralGaps === true;
502
+ // A Markdown block or front-matter field (contentGateFault): its structure
503
+ // (code, links, markup) rides protected placeholders, checked by the lane;
504
+ // the ICU/markup/plural checks are for key-value messages.
505
+ const prose = options.prose === true;
275
506
 
276
507
  for (const [key, translated] of Object.entries(translations)) {
277
508
  const source = sourceFlat[key] || '';
@@ -298,12 +529,23 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
298
529
  continue;
299
530
  }
300
531
 
532
+ // Names the project declared (config.protectedTerms) are correct as
533
+ // written in every language — no echo or script check applies.
534
+ if (isProtectedTermValue(translated, protectedTerms)) {
535
+ validated[key] = translated;
536
+ continue;
537
+ }
538
+
301
539
  // Check 2: Source echo — translated value is identical to source.
302
- // EXEMPTION: Short strings (≤30 chars) that are mostly ASCII are likely
303
- // proper nouns, brand names, or technical terms (e.g. "Blog", "GitHub",
304
- // "npm", "CLI Reference") that legitimately stay in English across all
305
- // languages. Rejecting these creates an infinite retry loop where the
306
- // correct translation is rejected every time, burning API calls forever.
540
+ // Short mostly-ASCII strings (≤30 chars) are often names ("GitHub",
541
+ // "Curtis Forbes") that legitimately stay as written — but just as often
542
+ // descriptive labels the model failed to translate ("Translation CLI").
543
+ // - Latin-script targets: accepted (an English label in French reads
544
+ // as a missed translation, not a broken page; verify warns).
545
+ // - Non-Latin targets: rejected ONCE with a name-or-label hint (the
546
+ // caller's feedback retry). If the model returns the same text again
547
+ // it is taken at its word as a name (acceptLatinNames) and cached, so
548
+ // it is never re-billed. One bounded retry — never a retry loop.
307
549
  if (translated === source) {
308
550
  const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
309
551
  const isShortAscii = source.length <= 30 && asciiRatio > 0.8;
@@ -311,7 +553,94 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
311
553
  failures.push({ key, reason: 'source echo (identical to English)', value: translated });
312
554
  continue;
313
555
  }
314
- // Short ASCII string echoed back — accept it as a valid translation
556
+ if (isNonLatin && thresholds.requireNonLatin && !acceptLatinNames) {
557
+ failures.push({ key, reason: LATIN_NAME_OR_LABEL, value: translated, nameOrLabel: true });
558
+ continue;
559
+ }
560
+ } else if (isDisguisedEcho(source, translated)) {
561
+ // Check 2a: the same echo, disguised — 'Thank you very much!' →
562
+ // 'Thánk yóú véry múch!'. Byte comparison calls it different, and the
563
+ // script check cannot fire (Latin in, Latin out; and in a non-Latin
564
+ // target the accents make it not-ASCII). Same suppressions as the
565
+ // exact check: declared names were accepted above, no-translate keys
566
+ // never reach the gate, letter-free and short (< 3-word) sources keep
567
+ // exact equality. Refused like a long echo — the caller's feedback
568
+ // retry, then the fallback method — never accepted as a name, because
569
+ // a name kept as written is byte-identical (isDisguisedEcho).
570
+ failures.push({ key, reason: DISGUISED_ECHO, value: translated, disguisedEcho: true });
571
+ continue;
572
+ }
573
+
574
+ // Check 2b: ICU MessageFormat / placeholder structure. Argument names,
575
+ // plural/select keywords, selectors, # and printf conversions are code:
576
+ // "{cóúnt, plúrál, óné {…} óthér {…}}" breaks the app (next-intl throws,
577
+ // Flutter gen-l10n refuses to build) while reading as a fine French
578
+ // string. Only the text inside the branches may change; a plural may
579
+ // ADD the CLDR categories the target language uses. Runs before the
580
+ // statistical checks so the retry hears the precise reason.
581
+ if (translated !== source && !prose) {
582
+ const icu = checkICUStructure(source, translated, targetLocale);
583
+ if (icu) {
584
+ failures.push({ key, reason: icu.reason, value: translated, icu: true });
585
+ continue;
586
+ }
587
+ // Markup is code as well: every tag opened, closed and nested as in the
588
+ // source ("<strong>Book</strong>" → "<strong>Réserver" breaks the page).
589
+ const markup = checkMarkup(source, translated);
590
+ if (markup) {
591
+ failures.push({ key, reason: markup.reason, value: translated, markup: true });
592
+ continue;
593
+ }
594
+ // The other placeholders verify reports — i18next {{…}} (renamed, lost,
595
+ // or written as a single-brace {name}, which i18next prints as is), a
596
+ // single-brace {name} outside an ICU message, a tag token — by the one
597
+ // rule verify reports them by (lib/placeholders.js). The gate let
598
+ // {{name}} → {{nom}} through, verify flagged it after the sync, and the
599
+ // run exited 2; refused here, the pair's fallback gets the text instead
600
+ // (Round 12, i18next persona).
601
+ const tokens = placeholderChanges(source, translated);
602
+ if (tokens.length > 0) {
603
+ failures.push({
604
+ key,
605
+ reason: `placeholder structure damaged: ${tokens.map(t => `${PLACEHOLDER_SYNTAX_NAMES[t.syntax]} ${t.issue}`).join('; ')}`,
606
+ value: translated,
607
+ placeholder: true,
608
+ });
609
+ continue;
610
+ }
611
+ // A sentence break the translation put right beside a placeholder
612
+ // where the source has none: "Take this medicine at {time}." →
613
+ // "… sina. {time}." passed every check (Round 14, hospital persona).
614
+ // The same rule verify flags by (lib/placeholders.js); refused here,
615
+ // the retry hears why and the pair's fallback runs.
616
+ const breaks = placeholderSentenceBreaks(source, translated);
617
+ if (breaks.length > 0) {
618
+ failures.push({
619
+ key,
620
+ reason: `sentence break beside a placeholder: ${breaks.map(b => b.issue).join('; ')}`,
621
+ value: translated,
622
+ placeholder: true,
623
+ });
624
+ continue;
625
+ }
626
+ }
627
+
628
+ // Check 2c: the echo checks, per plural/select branch. A plural message
629
+ // differs from its source as a WHOLE as soon as the target adds a form
630
+ // (or one branch is translated), so the two checks above never saw a
631
+ // branch handed back as the English — accented, or as it was (Round 4,
632
+ // Django persona: the plural entry passed while the identical singular
633
+ // change was refused). Each branch is held to the singular rules: an
634
+ // exact copy of a short Latin-script branch is accepted in a Latin-script
635
+ // target (a name, "emails") and asked about once in a non-Latin one; any
636
+ // longer copy is refused; a disguised copy (3+ words) is refused. A
637
+ // branch made only of declared names (protectedTerms) is never an echo.
638
+ if (translated !== source && !prose) {
639
+ const branchEcho = pluralBranchEcho(source, translated, { isNonLatin, requireNonLatin: thresholds.requireNonLatin, acceptLatinNames, protectedTerms });
640
+ if (branchEcho) {
641
+ failures.push({ key, value: translated, ...branchEcho });
642
+ continue;
643
+ }
315
644
  }
316
645
 
317
646
  // Check 3: Repetition detection — catches hallucination loops.
@@ -387,6 +716,17 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
387
716
  // or version strings like "3.2.0" have nothing to write in another script.
388
717
  // - Short ASCII strings (≤30 chars, >80% ASCII) are likely proper nouns
389
718
  // or brand names (e.g. "GitHub", "npm") that stay in English everywhere.
719
+ // Fullwidth Latin letters ("Book an appointment") outside CJK typography:
720
+ // English in disguise, whatever the target's script. Never a name — a
721
+ // name kept as written is in plain letters.
722
+ if (hasForeignFullwidthLatin(translated, targetLocale)) {
723
+ failures.push({
724
+ key,
725
+ reason: `wrong script (fullwidth Latin letters in a ${targetLocale} value — English in disguise, not a translation)`,
726
+ value: translated.slice(0, 80),
727
+ });
728
+ continue;
729
+ }
390
730
  if (isNonLatin && thresholds.requireNonLatin) {
391
731
  // Strip ICU placeholders, digits, punctuation, whitespace → what's left?
392
732
  const translatableText = translated
@@ -394,12 +734,43 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
394
734
  .replace(/[\d\s\p{P}\p{S}]/gu, '') // digits, whitespace, punctuation, symbols
395
735
  .trim();
396
736
  const asciiRatio = source.replace(/[^\x20-\x7E]/g, '').length / Math.max(source.length, 1);
397
- const isShortAsciiPropNoun = source.length <= 30 && asciiRatio > 0.8;
398
- if (translatableText.length > 0 && isAsciiOnly(translated) && !isShortAsciiPropNoun) {
737
+ const isShortAscii = source.length <= 30 && asciiRatio > 0.8;
738
+ // Letters classified by Unicode script: accented Latin is Latin too
739
+ // (an ASCII test passed "Thánk yóú" in a Russian catalog).
740
+ if (translatableText.length > 0 && isLatinOnly(translated, targetLocale)) {
741
+ if (isShortAscii && acceptLatinNames) {
742
+ // Second answer still Latin-script: the model says it is a name.
743
+ } else if (isShortAscii) {
744
+ failures.push({ key, reason: LATIN_NAME_OR_LABEL, value: translated.slice(0, 80), nameOrLabel: true });
745
+ continue;
746
+ } else {
747
+ failures.push({
748
+ key,
749
+ reason: `wrong script (ASCII-only for ${targetLocale}, expected non-Latin characters)`,
750
+ value: translated.slice(0, 80),
751
+ });
752
+ continue;
753
+ }
754
+ }
755
+ }
756
+
757
+ // Check 7 (last, so a soft failure means everything else passed): a
758
+ // plural message without a form the target language uses for ordinary
759
+ // counts. Soft — the caller asks once more; see the module header.
760
+ if (!acceptPluralGaps && !prose && translated !== source) {
761
+ // A gettext catalog's slots (pairConfig.pluralSlots, set by sync for a
762
+ // .po file): a form it has no msgstr[] for is never asked again for.
763
+ const gaps = pluralGaps(source, translated, targetLocale, options.pluralSlots ?? pairConfig.pluralSlots ?? null)
764
+ .filter(g => g.everyday.length > 0);
765
+ if (gaps.length > 0) {
766
+ const missing = [...new Set(gaps.flatMap(g => g.everyday))];
767
+ const type = gaps[0].type;
399
768
  failures.push({
400
769
  key,
401
- reason: `wrong script (ASCII-only for ${targetLocale}, expected non-Latin characters)`,
402
- value: translated.slice(0, 80),
770
+ reason: `plural form(s) ${missing.map(c => `"${c}"`).join(', ')} missing — ${targetLocale} uses `
771
+ + `${describeCategories(targetLocale, missing, type)}`,
772
+ value: translated,
773
+ pluralGap: { missing, type },
403
774
  });
404
775
  continue;
405
776
  }
@@ -412,6 +783,336 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
412
783
  return { validated, failures };
413
784
  }
414
785
 
786
+ // ---------------------------------------------------------------------------
787
+ // Different inputs, same output
788
+ // ---------------------------------------------------------------------------
789
+
790
+ /** Distinct source strings one output may answer before it is suspect. */
791
+ const SHARED_OUTPUT_MIN_SOURCES = 3;
792
+ /** Words (with letters) the shared output needs, on its own, to be suspect. */
793
+ const SHARED_OUTPUT_MIN_WORDS = 4;
794
+
795
+ /**
796
+ * Simple placeholders and markup tags in an output — not words of the
797
+ * translation (SharedOutputIndex.outputForm): {name} {0} {{count}} %(name)s
798
+ * %s %1$d <b> </a>. A plural/select argument ({n, plural, …}) has a comma and
799
+ * is not matched.
800
+ */
801
+ const PLACEHOLDER_IN_OUTPUT = /\{\{\s*[\w.$-]+\s*\}\}|\{\s*[\w.$-]+\s*\}|%\([\w.]+\)[sdifr]|%(?:\d+\$)?[-+0#]*\d*(?:\.\d+)?[sdifuxXeEgGcp@]|<\/?[A-Za-z][^<>]*>/gu;
802
+ /** The same without single-brace arguments: for a whole plural/select message. */
803
+ const PLACEHOLDER_NOT_BRACED = /\{\{\s*[\w.$-]+\s*\}\}|%\([\w.]+\)[sdifr]|%(?:\d+\$)?[-+0#]*\d*(?:\.\d+)?[sdifuxXeEgGcp@]|<\/?[A-Za-z][^<>]*>/gu;
804
+ /** A plural/select/selectordinal argument ("{n, plural, …"). */
805
+ const ICU_COMPLEX_ARGUMENT = /\{\s*[\w.$-]+\s*,\s*(?:plural|select|selectordinal)\s*,/u;
806
+
807
+ /**
808
+ * The sentences of a text, for the repeat check: split after . ! ? (and the
809
+ * CJK 。!?) followed by space, each kept only when it has a letter once
810
+ * its placeholders are gone ("S. {name}!" is one sentence).
811
+ *
812
+ * @param {string} text
813
+ * @returns {string[]}
814
+ */
815
+ function sentencesOf(text) {
816
+ return String(text)
817
+ .split(/(?<=[.!?\u3002\uFF01\uFF1F])\s+/u)
818
+ .map(t => t.trim())
819
+ .filter(t => /\p{L}/u.test(t.replace(PLACEHOLDER_IN_OUTPUT, ' ')));
820
+ }
821
+
822
+ /** A source in the form two sources are compared in ("Save", "save!" meet). */
823
+ function sourceIdentity(text) {
824
+ return foldForEchoCompare(String(text).replace(/[\p{P}\p{S}]+/gu, ' '));
825
+ }
826
+
827
+ /** The words of a folded string, as a set. */
828
+ function wordSet(text) {
829
+ return new Set(sourceIdentity(text).split(' ').filter(w => /\p{L}/u.test(w)));
830
+ }
831
+
832
+ /** True when every pair of sources shares under half its words (Jaccard). */
833
+ function clearlyDifferent(sources) {
834
+ const sets = sources.map(wordSet);
835
+ for (let i = 0; i < sets.length; i++) {
836
+ for (let j = i + 1; j < sets.length; j++) {
837
+ const a = sets[i];
838
+ const b = sets[j];
839
+ const inter = [...a].filter(w => b.has(w)).length;
840
+ const union = new Set([...a, ...b]).size;
841
+ if (union === 0 || inter / union >= 0.5) return false;
842
+ }
843
+ }
844
+ return true;
845
+ }
846
+
847
+ /**
848
+ * Different inputs, same output: ONE translation text returned for several
849
+ * DIFFERENT source strings is what a model that memorized a training sentence
850
+ * does (Round 4 school persona: a trained eng→crk model answered the app
851
+ * title, "Contact the school", the newsletter title and its heading with the
852
+ * same sentence — and every per-key check passed it).
853
+ *
854
+ * Suspect when one output answers SHARED_OUTPUT_MIN_SOURCES+ distinct sources
855
+ * (compared folded, punctuation aside — "Save" and "Save!" are one source)
856
+ * AND either the output has SHARED_OUTPUT_MIN_WORDS+ words, or every source
857
+ * has two or more words and no two of them share half their words. So
858
+ * synonyms collapsing to one short translation pass ('OK'/'Okay'/'Sure' →
859
+ * "D'accord"; 'Close'/'Dismiss'/'Cancel' → "Fermer"), and so does the same
860
+ * source text under several keys.
861
+ *
862
+ * TWO sources are already enough when the evidence is strong (Round 6,
863
+ * school persona: 'Thank you, {name}!' and 'Please bring the forms.' got one
864
+ * memorized sentence and passed, because only a third source counted): the
865
+ * output has SHARED_OUTPUT_MIN_WORDS+ words, both sources have two or more
866
+ * words, and they share under half their words (clearly different). Short
867
+ * outputs and synonym-like sources (sharing half their words or more) still
868
+ * need a third source. The cost of a false alarm is one more ask (an LLM is
869
+ * told to translate this string on its own; a method that takes no
870
+ * instructions is asked once more), then the pair's fallback.
871
+ *
872
+ * MEMORIZED outputs (markMemorized): an output an earlier sync found
873
+ * answering different source strings is suspect from its first source on —
874
+ * a repair that asks again and gets the same sentence back must not write it
875
+ * (Round 6, school persona: `--force-content` re-served it in silence).
876
+ *
877
+ * Same suppressions as the echo checks: a value made only of declared names
878
+ * (protectedTerms) is never suspect, nor a value equal to its own source
879
+ * (an echo — its own rule), nor one without letters. No-translate keys never
880
+ * reach a gate.
881
+ *
882
+ * The index spans a run's locale: the caller adds what it accepted, so a
883
+ * later batch (or a content file's block) that repeats an earlier answer is
884
+ * caught; only the NEW items are returned as suspect — what was written
885
+ * earlier is reported by `verify`. An output found suspect earlier in the
886
+ * run is suspect from its first new source on, like a MEMORIZED one: its
887
+ * refused members were never added, and the run remembers it only when it
888
+ * ends.
889
+ */
890
+ class SharedOutputIndex {
891
+ constructor({ protectedTerms = [] } = {}) {
892
+ this.protectedTerms = protectedTerms;
893
+ // output form → Map(source identity → { key, source })
894
+ this.byOutput = new Map();
895
+ // output form → the first text seen in that form (what reports show)
896
+ this.display = new Map();
897
+ // output text → { value, sources, keys } — every group found suspect,
898
+ // with ALL its members (earlier, accepted ones too), for the run report.
899
+ this.flagged = new Map();
900
+ // output form → the text as first seen: outputs an earlier sync found
901
+ // answering different source strings (markMemorized).
902
+ this.memorized = new Map();
903
+ }
904
+
905
+ /**
906
+ * Outputs an earlier run found answering several different source strings:
907
+ * suspect from their first source on (see the class comment).
908
+ *
909
+ * @param {Iterable<string>} values
910
+ */
911
+ markMemorized(values) {
912
+ for (const value of values || []) {
913
+ if (typeof value !== 'string') continue;
914
+ const form = SharedOutputIndex.outputForm(value);
915
+ if (form && /\p{L}/u.test(form)) this.memorized.set(form, String(value).replace(/\s+/gu, ' ').trim());
916
+ }
917
+ }
918
+
919
+ /**
920
+ * The form outputs are grouped by: case, punctuation and symbols aside, and
921
+ * without a Markdown block marker — so "S?", "S." and a heading "# S" are
922
+ * one output (Round 5: an NMT model copies the source's end punctuation,
923
+ * and a newsletter H1 kept its "# "). Letters and their accents are kept:
924
+ * two words that differ by a diacritic are two words.
925
+ *
926
+ * Placeholders and markup tags are not words of the output: "S. {name}!"
927
+ * and "S." are one output (Round 9, school persona: the memorized sentence
928
+ * came back as "S. {name}!" for 'Thank you, {name}!' and as "S." inside a
929
+ * newsletter paragraph, and the "name" left over from "{name}" kept the two
930
+ * apart). Simple arguments only ({name}, {0}, {{count}}, %(name)s, %s,
931
+ * %1$d, <b>…</b>); an unsplit plural/select message keeps its branches.
932
+ */
933
+ static outputForm(value) {
934
+ const text = String(value);
935
+ // Inside a whole plural/select message, `{un}` is a branch, not an argument.
936
+ const placeholders = ICU_COMPLEX_ARGUMENT.test(text) ? PLACEHOLDER_NOT_BRACED : PLACEHOLDER_IN_OUTPUT;
937
+ return text
938
+ .normalize('NFC')
939
+ .replace(/^\s{0,3}(?:#{1,6}\s+|[-*+]\s+|\d+[.)]\s+|>\s*)/u, '')
940
+ .replace(placeholders, ' ')
941
+ .toLowerCase()
942
+ .replace(/[\p{P}\p{S}]+/gu, ' ')
943
+ .replace(/\s+/gu, ' ')
944
+ .trim();
945
+ }
946
+
947
+ /**
948
+ * The items, plus — for a value of several sentences whose source has as
949
+ * many — each sentence on its own, paired with its source sentence (the
950
+ * parent kept in `whole`). One memorized sentence inside a paragraph is
951
+ * then the same output as that sentence answering a UI string (Round 9,
952
+ * school persona: 'Please bring the forms.' in the newsletter and 'Thank
953
+ * you, {name}!' got one sentence, and the whole-paragraph comparison never
954
+ * met it). Pairing by position keeps a sentence two paragraphs genuinely
955
+ * share ("Please bring the forms." in both) as ONE source — not suspect;
956
+ * a value whose sentence count differs from its source's is compared whole
957
+ * only.
958
+ *
959
+ * @param {Array<{ key: string, source: string, value: string }>} items
960
+ * @returns {Array<{ key: string, source: string, value: string, whole?: { source: string, value: string } }>}
961
+ */
962
+ static withSentences(items) {
963
+ const out = [];
964
+ for (const it of items) {
965
+ out.push(it);
966
+ if (typeof it.source !== 'string' || typeof it.value !== 'string' || it.whole) continue;
967
+ const vs = sentencesOf(it.value);
968
+ if (vs.length < 2) continue;
969
+ const ss = sentencesOf(it.source);
970
+ if (ss.length !== vs.length) continue;
971
+ vs.forEach((v, i) => out.push({ key: it.key, source: ss[i], value: v, whole: { source: it.source, value: it.value } }));
972
+ }
973
+ return out;
974
+ }
975
+
976
+ eligible(source, value) {
977
+ if (typeof source !== 'string' || typeof value !== 'string') return false;
978
+ if (value === source || !/\p{L}/u.test(value) || !SharedOutputIndex.outputForm(value)) return false;
979
+ if (isProtectedTermValue(value, this.protectedTerms)) return false;
980
+ return true;
981
+ }
982
+
983
+ /**
984
+ * Items that would make (or join) a suspect group.
985
+ *
986
+ * @param {Array<{ key: string, source: string, value: string }>} items
987
+ * @returns {Map<string, { value: string, sources: string[] }>} key → its group
988
+ */
989
+ suspects(items) {
990
+ const groups = new Map();
991
+ for (const it of SharedOutputIndex.withSentences(items)) {
992
+ if (!this.eligible(it.source, it.value)) continue;
993
+ const form = SharedOutputIndex.outputForm(it.value);
994
+ if (!this.display.has(form)) this.display.set(form, String(it.value).replace(/\s+/gu, ' ').trim());
995
+ if (!groups.has(form)) {
996
+ groups.set(form, { sources: new Map(this.byOutput.get(form) || []), fresh: [] });
997
+ }
998
+ const g = groups.get(form);
999
+ const id = sourceIdentity(it.source);
1000
+ if (!g.sources.has(id)) g.sources.set(id, { key: it.key, source: it.source });
1001
+ g.fresh.push(it);
1002
+ }
1003
+ const out = new Map();
1004
+ for (const [form, g] of groups) {
1005
+ let sources = [...g.sources.values()].map(s => s.source);
1006
+ const byRule = SharedOutputIndex.suspectGroup(form, sources);
1007
+ // `memorized` marks a group suspect ONLY because an earlier sync
1008
+ // remembered its output (reports word it that way); a group the rule
1009
+ // catches on its own is reported as before.
1010
+ const memorized = !byRule && this.memorized.has(form);
1011
+ // An output this index already found answering different source
1012
+ // strings EARLIER IN THIS RUN is suspect from its first new source on,
1013
+ // as a remembered one is from an earlier sync. The members refused
1014
+ // then were never added (only accepted items are), so without this the
1015
+ // same sentence came back for the newsletter's title after the app's
1016
+ // three keys were refused for it, and was written — the run remembers
1017
+ // it only when it ends, and `verify` flagged it (Round 10, school
1018
+ // persona). A gate retry that returns the refused sentence for one key
1019
+ // is refused the same way.
1020
+ const earlier = !byRule && !memorized ? this.flagged.get(form) : null;
1021
+ if (earlier) sources = [...new Set([...earlier.sources, ...sources])];
1022
+ if (!byRule && !memorized && !earlier) continue;
1023
+ const shown = this.memorized.get(form) || this.display.get(form) || form;
1024
+ for (const it of g.fresh) out.set(it.key, { value: shown, sources, ...(memorized && { memorized: true }) });
1025
+ const prior = this.flagged.get(form) || { value: shown, sources: [], keys: [] };
1026
+ this.flagged.set(form, {
1027
+ value: shown,
1028
+ sources: [...new Set([...prior.sources, ...sources])],
1029
+ keys: [...new Set([...prior.keys, ...[...g.sources.values()].map(v => v.key), ...g.fresh.map(it => it.key)])],
1030
+ ...(memorized && { memorized: true }),
1031
+ });
1032
+ }
1033
+ return out;
1034
+ }
1035
+
1036
+ /**
1037
+ * Is one output form answering these distinct sources the pattern? Three or
1038
+ * more: a long output, or clearly different multi-word sources. Two: only
1039
+ * the strong case (see the class comment).
1040
+ *
1041
+ * @param {string} form - outputForm() of the shared output
1042
+ * @param {string[]} sources - Distinct sources it answers
1043
+ * @returns {boolean}
1044
+ */
1045
+ static suspectGroup(form, sources) {
1046
+ const outWords = letterWordCount(form);
1047
+ if (sources.length >= SHARED_OUTPUT_MIN_SOURCES) {
1048
+ return outWords >= SHARED_OUTPUT_MIN_WORDS
1049
+ || (sources.every(src => letterWordCount(src) >= 2) && clearlyDifferent(sources));
1050
+ }
1051
+ if (sources.length !== 2) return false;
1052
+ return outWords >= SHARED_OUTPUT_MIN_WORDS
1053
+ && sources.every(src => letterWordCount(src) >= 2)
1054
+ && clearlyDifferent(sources);
1055
+ }
1056
+
1057
+ /**
1058
+ * Record accepted items (what was written) for later batches. A sentence
1059
+ * of a longer value is recorded with its parent (`whole`): what was written,
1060
+ * and cached, is the whole value.
1061
+ */
1062
+ add(items) {
1063
+ for (const it of SharedOutputIndex.withSentences(items)) {
1064
+ if (!this.eligible(it.source, it.value)) continue;
1065
+ const form = SharedOutputIndex.outputForm(it.value);
1066
+ if (!this.display.has(form)) this.display.set(form, String(it.value).replace(/\s+/gu, ' ').trim());
1067
+ if (!this.byOutput.has(form)) this.byOutput.set(form, new Map());
1068
+ const id = sourceIdentity(it.source);
1069
+ if (!this.byOutput.get(form).has(id)) {
1070
+ this.byOutput.get(form).set(id, { key: it.key, source: it.source, value: it.value, ...(it.whole && { whole: it.whole }) });
1071
+ }
1072
+ }
1073
+ }
1074
+ }
1075
+
1076
+ /**
1077
+ * What one translated value puts in the shared-output index: the value
1078
+ * itself, or — for an ICU plural/select message — each leaf branch on its
1079
+ * own (Round 5, school persona: one memorized sentence filled BOTH branches
1080
+ * of a plural message and two other keys, and nothing counted the branches).
1081
+ *
1082
+ * Every branch of one `plural`/`selectordinal` argument counts as ONE source
1083
+ * (the message's): a language without number inflection writes the same
1084
+ * text in every branch by design, so repeating it across branches is never
1085
+ * evidence on its own. `select` branches count as their own source branches —
1086
+ * different only when those differ.
1087
+ *
1088
+ * @param {string} key
1089
+ * @param {string} source
1090
+ * @param {string} value
1091
+ * @returns {Array<{ key: string, source: string, value: string }>}
1092
+ */
1093
+ function sharedOutputItems(key, source, value) {
1094
+ if (typeof source !== 'string' || typeof value !== 'string') return [];
1095
+ const pairs = value.includes('{') && source.includes('{') ? pluralBranchPairs(source, value) : [];
1096
+ if (pairs.length === 0) return [{ key, source, value }];
1097
+ return pairs.map(p => ({
1098
+ key,
1099
+ source: p.keyword === 'select' ? p.source : source,
1100
+ value: p.translated,
1101
+ }));
1102
+ }
1103
+
1104
+ /** Gate reason for an output shared by several different sources. */
1105
+ function sharedOutputReason(group) {
1106
+ if (group.memorized) {
1107
+ const text = group.value.length > 60 ? `${group.value.slice(0, 57)}...` : group.value;
1108
+ return `the sentence the model gave for other, different source strings on an earlier sync (${JSON.stringify(text)}) `
1109
+ + '— a memorized sentence, not a translation of this string; asking the same model again returns it again';
1110
+ }
1111
+ const shown = group.sources.slice(0, 4).map(s => JSON.stringify(s.length > 40 ? `${s.slice(0, 37)}...` : s)).join(', ');
1112
+ return `same output for ${group.sources.length} different source strings (${shown}${group.sources.length > 4 ? ', …' : ''}) `
1113
+ + '— a model repeating one memorized sentence, not a translation of this string';
1114
+ }
1115
+
415
1116
  /**
416
1117
  * Split a value into independently-measurable segments: pipe-delimited
417
1118
  * plural variants ("one doc|{count} docs") legitimately share most of
@@ -421,6 +1122,10 @@ function validateTranslations(translations, sourceFlat, pairConfig, options = {}
421
1122
  * @returns {string[]} Trimmed segments (always at least one)
422
1123
  */
423
1124
  function splitPluralSegments(text) {
1125
+ // ICU plural/select branches too: a Russian plural repeats one sentence in
1126
+ // four forms by design, and measured as one string it read as a loop.
1127
+ const branches = text.includes('{') ? pluralBranchTexts(text) : null;
1128
+ if (branches) return branches.map(seg => seg.trim());
424
1129
  return (text.includes('|') ? text.split('|') : [text]).map(seg => seg.trim());
425
1130
  }
426
1131
 
@@ -477,6 +1182,138 @@ function isAsciiOnly(text) {
477
1182
  return /^[\x00-\x7F]*$/.test(text);
478
1183
  }
479
1184
 
1185
+ /** Scripts whose text legitimately uses fullwidth Latin letters (CJK typography). */
1186
+ const FULLWIDTH_SCRIPTS = new Set(['Hani', 'Hans', 'Hant', 'Jpan', 'Kore', 'Hira', 'Kana', 'Hang', 'Bopo']);
1187
+
1188
+ /** Fullwidth Latin letters: A–Z, a–z. */
1189
+ const FULLWIDTH_LATIN = /[A-Za-z]/u;
1190
+
1191
+ /**
1192
+ * The text of a value with its code set aside: ICU/brace placeholders,
1193
+ * printf conversions, markup tags. (A URL stays: in a non-Latin locale a
1194
+ * URL value is a wrong-script answer unless the key is declared
1195
+ * no-translate — which never reaches this check.)
1196
+ */
1197
+ function proseOf(text) {
1198
+ // A plural/select message: its prose is the text of its branches (the
1199
+ // keywords and selectors — "plural", "one", "other" — are code).
1200
+ const str = String(text);
1201
+ const branches = str.includes('{') ? pluralBranchTexts(str) : null;
1202
+ return (branches ? branches.join(' ') : str)
1203
+ // Simple placeholders only: {name}, {count, number}. Branch text such as
1204
+ // "{Один файл}" is prose, never stripped.
1205
+ .replace(/\{\s*[\w.$-]+\s*(?:,[^{}]*)?\}/g, ' ')
1206
+ .replace(/\{\{\s*[\w.$-]+\s*\}\}/g, ' ')
1207
+ .replace(/%(\([^)]*\))?[-#0 +]*\d*(\.\d+)?[a-zA-Z@]/g, ' ')
1208
+ .replace(/<\/?[A-Za-z0-9][^<>]*>/g, ' ');
1209
+ }
1210
+
1211
+ /**
1212
+ * Are ALL the letters of a value's prose Latin script — by Unicode script,
1213
+ * not by byte? Fullwidth Latin ("Hello") and accented Latin ("Thánk")
1214
+ * are Latin: an ASCII test let both through as "not English" in a Russian
1215
+ * catalog (Round 4, Django persona). False when the prose has no letters.
1216
+ *
1217
+ * @param {string} text
1218
+ * @returns {boolean}
1219
+ */
1220
+ function isLatinOnly(text, locale = null) {
1221
+ const prose = proseOf(text);
1222
+ const letters = prose.match(/\p{L}/gu);
1223
+ if (!letters) return false;
1224
+ // CJK typography sets Latin in fullwidth forms ("OK" in Japanese): there,
1225
+ // fullwidth letters are the target's own usage, as they always were here.
1226
+ if (locale && FULLWIDTH_LATIN.test(prose) && usesFullwidthLatin(locale)) return false;
1227
+ return letters.every(ch => /\p{Script=Latin}/u.test(ch));
1228
+ }
1229
+
1230
+ /** Does this locale's script set Latin in fullwidth forms (CJK)? */
1231
+ function usesFullwidthLatin(locale) {
1232
+ const card = getLanguageCard(String(locale)) || getLanguageCard(String(locale).split(/[-_]/)[0]);
1233
+ return FULLWIDTH_SCRIPTS.has(card?.script);
1234
+ }
1235
+
1236
+ /**
1237
+ * Fullwidth Latin letters in a value whose language does not set Latin in
1238
+ * fullwidth forms (anything but CJK typography): a disguised copy of English,
1239
+ * never a translation.
1240
+ *
1241
+ * @param {string} text
1242
+ * @param {string} locale - Target locale
1243
+ * @returns {boolean}
1244
+ */
1245
+ function hasForeignFullwidthLatin(text, locale) {
1246
+ if (!FULLWIDTH_LATIN.test(proseOf(text))) return false;
1247
+ return !usesFullwidthLatin(locale);
1248
+ }
1249
+
1250
+ // ---------------------------------------------------------------------------
1251
+ // A question or exclamation that lost its mark
1252
+ // ---------------------------------------------------------------------------
1253
+
1254
+ // Marks that end a question / an exclamation in some writing system: Latin
1255
+ // and most scripts (?), CJK fullwidth (?), Arabic script (؟), Greek (U+037E
1256
+ // and the ASCII semicolon Greek text is typed with), Armenian (՞), Ethiopic
1257
+ // (፧), and the interrobang family. Spanish closes with "?" / "!" as well.
1258
+ const QUESTION_ENDS = new Set(['?', '?', '؟', '\u037e', ';', '՞', '፧', '⸮', '⁇', '⁈', '‽', '⁉']);
1259
+ const EXCLAMATION_ENDS = new Set(['!', '!', '‼', '⁉', '⁈', '‽', '՜', '¡']);
1260
+ // Closing quotes, brackets, ICU braces and direction marks after the mark.
1261
+ const TRAILING_CLOSERS = /[\s"'»”’)\]}›〉》」』】〕\u200e\u200f]+$/u;
1262
+
1263
+ /** The last visible character of a text, closers aside. */
1264
+ function terminalChar(text) {
1265
+ const chars = Array.from(String(text).replace(TRAILING_CLOSERS, ''));
1266
+ return chars.length > 0 ? chars[chars.length - 1] : '';
1267
+ }
1268
+
1269
+ /**
1270
+ * A source that ends with "?" or "!" whose translation ends with neither it
1271
+ * nor an equivalent the target's script uses (Round 6, hospital persona:
1272
+ * "Where does it hurt?" passed as a statement). A WARNING, never a refusal:
1273
+ * some languages mark a question with a word or particle instead of a mark,
1274
+ * and a translation that does so is right.
1275
+ *
1276
+ * @param {string} source
1277
+ * @param {string} translated
1278
+ * @returns {{ mark: '?'|'!' } | null}
1279
+ */
1280
+ function droppedTerminalMark(source, translated) {
1281
+ if (typeof source !== 'string' || typeof translated !== 'string' || translated === source) return null;
1282
+ if (!/\p{L}/u.test(translated)) return null;
1283
+ const end = terminalChar(source);
1284
+ if (end !== '?' && end !== '!') return null;
1285
+ const got = terminalChar(translated);
1286
+ if (end === '?' ? QUESTION_ENDS.has(got) : EXCLAMATION_ENDS.has(got)) return null;
1287
+ return { mark: end };
1288
+ }
1289
+
1290
+ /**
1291
+ * Keys whose translation dropped the source's closing "?" / "!" — one
1292
+ * finding for sync and verify alike. Declared names and no-translate keys
1293
+ * are skipped by the callers' own rules.
1294
+ *
1295
+ * @param {Array<[string, string, string]>} entries - [key, source, translated]
1296
+ * @returns {Array<{ key: string, mark: string, value: string }>}
1297
+ */
1298
+ function droppedTerminalMarks(entries) {
1299
+ const out = [];
1300
+ for (const [key, source, value] of entries) {
1301
+ const d = droppedTerminalMark(source, value);
1302
+ if (d) out.push({ key, mark: d.mark, value });
1303
+ }
1304
+ return out;
1305
+ }
1306
+
1307
+ /** The warning text for droppedTerminalMarks findings (sync and verify say the same). */
1308
+ function describeDroppedMarks(found, fix) {
1309
+ const marks = [...new Set(found.map(f => `"${f.mark}"`))].join(' or ');
1310
+ const shown = found.slice(0, 3).map(f => `${f.key} (${JSON.stringify(f.value.length > 40 ? `${f.value.slice(0, 37)}...` : f.value)})`).join(', ');
1311
+ return `${found.length} translation(s) dropped the source's closing ${marks}: ${shown}${found.length > 3 ? ', …' : ''} — `
1312
+ + 'a question or exclamation may now read as a statement. Some languages mark a question with a word or '
1313
+ + 'particle instead of a mark, so this is a warning, not a refusal: check them'
1314
+ + (fix ? `, or ask again (--fresh: the cache holds this answer): \`${fix}\`` : '') + '.';
1315
+ }
1316
+
480
1317
  /**
481
1318
  * Log quality gate failures in a structured, actionable format.
482
1319
  *
@@ -485,21 +1322,31 @@ function isAsciiOnly(text) {
485
1322
  */
486
1323
  function logGateFailures(failures, pairKey) {
487
1324
  if (failures.length === 0) return;
1325
+ if (output.getMode() === 'json') {
1326
+ // One structured record instead of indented prose on stderr.
1327
+ output.warn(`${pairKey}: ${failures.length} key(s) failed quality validation`, {
1328
+ pairKey, gateFailures: failures.map(({ key, reason, value }) => ({ key, reason, value })),
1329
+ });
1330
+ return;
1331
+ }
488
1332
 
489
- console.error(`\n [GATE] ${pairKey}: ${failures.length} key(s) failed quality validation:`);
1333
+ const lines = ['', ` [GATE] ${pairKey}: ${failures.length} key(s) failed quality validation:`];
490
1334
  for (const { key, reason, value } of failures) {
491
- console.error(` ✗ "${key}": ${reason}`);
492
- if (value) {
493
- console.error(` → "${value}"`);
494
- }
1335
+ lines.push(` ✗ "${key}": ${reason}`);
1336
+ if (value) lines.push(` → "${value}"`);
495
1337
  }
496
- console.error('');
1338
+ lines.push('');
1339
+ output.block(lines);
497
1340
  }
498
1341
 
499
1342
  export {
1343
+ checkICUStructure,
500
1344
  validateTranslations,
1345
+ contentGateFault,
501
1346
  measureRepetition,
502
1347
  isAsciiOnly,
1348
+ isLatinOnly,
1349
+ hasForeignFullwidthLatin,
503
1350
  logGateFailures,
504
1351
  checkContentPreservation,
505
1352
  contentCharacters,
@@ -507,4 +1354,20 @@ export {
507
1354
  NON_LATIN_LOCALES,
508
1355
  DEFAULT_THRESHOLDS,
509
1356
  MIN_MEASURABLE_CONTENT,
1357
+ isProtectedTermValue,
1358
+ LATIN_NAME_OR_LABEL,
1359
+ foldForEchoCompare,
1360
+ letterWordCount,
1361
+ isDisguisedEcho,
1362
+ pluralBranchEcho,
1363
+ MIN_FOLDED_ECHO_WORDS,
1364
+ DISGUISED_ECHO,
1365
+ SharedOutputIndex,
1366
+ sharedOutputItems,
1367
+ sharedOutputReason,
1368
+ SHARED_OUTPUT_MIN_SOURCES,
1369
+ droppedTerminalMark,
1370
+ droppedTerminalMarks,
1371
+ describeDroppedMarks,
1372
+ SHARED_OUTPUT_MIN_WORDS,
510
1373
  };