champollion 0.3.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -26
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +34 -0
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +286 -85
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +632 -125
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +15 -9
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +194 -35
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +6 -1
- package/lib/seal.mjs +4 -3
- package/lib/sealed-qualifier.mjs +1 -1
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +1 -1
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/cards-fallback.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +8 -2
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11739
package/lib/integrity.js
CHANGED
|
@@ -14,67 +14,32 @@
|
|
|
14
14
|
* or values that are clearly just the source value copy-pasted (same
|
|
15
15
|
* string in a different locale file = untranslated).
|
|
16
16
|
*
|
|
17
|
+
* 4. ICU STRUCTURE DAMAGE: a translated variable name, plural/select
|
|
18
|
+
* keyword or selector ("{cóúnt, plúrál, óné {…} óthér {…}}"), a lost #
|
|
19
|
+
* or printf conversion — the SAME check the sync quality gate applies
|
|
20
|
+
* (lib/icu-structure.js), so damage written before the gate existed is
|
|
21
|
+
* reported instead of waiting for the app to crash.
|
|
22
|
+
*
|
|
17
23
|
* Zero external dependencies. All string-based analysis.
|
|
18
24
|
*/
|
|
19
25
|
|
|
20
26
|
import fs from 'node:fs';
|
|
21
27
|
import path from 'node:path';
|
|
28
|
+
import { parseARB, flutterLocale } from './format.js';
|
|
22
29
|
import { isICUString, parseICU, getRequiredPluralCategories } from './icu.js';
|
|
23
|
-
import { checkContentPreservation } from './validate.js';
|
|
30
|
+
import { checkContentPreservation, checkICUStructure } from './validate.js';
|
|
31
|
+
import { checkMarkup, pluralGaps, describePluralGap } from './icu-structure.js';
|
|
32
|
+
import { getLanguageCard } from './registers.js';
|
|
33
|
+
import { extractPlaceholders, comparePlaceholders, placeholderSyntax, showPlaceholder } from './placeholders.js';
|
|
24
34
|
|
|
25
35
|
// -----------------------------------------------------------------
|
|
26
36
|
// Placeholder extraction
|
|
27
37
|
// -----------------------------------------------------------------
|
|
28
38
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
* - Simple: {name}, {count}
|
|
34
|
-
* - Nested ICU: {count, plural, one {# item} other {# items}}
|
|
35
|
-
* - React-intl: <bold>text</bold>
|
|
36
|
-
*
|
|
37
|
-
* @param {string} text - Translation string
|
|
38
|
-
* @returns {string[]} Sorted array of placeholder tokens
|
|
39
|
-
*/
|
|
40
|
-
function extractPlaceholders(text) {
|
|
41
|
-
if (typeof text !== 'string') return [];
|
|
42
|
-
|
|
43
|
-
const placeholders = new Set();
|
|
44
|
-
|
|
45
|
-
// Simple ICU placeholders: {name}, {count}
|
|
46
|
-
// Match top-level braces only (not nested plurals)
|
|
47
|
-
const simplePattern = /\{(\w+)(?:[,}])/g;
|
|
48
|
-
let match;
|
|
49
|
-
while ((match = simplePattern.exec(text)) !== null) {
|
|
50
|
-
placeholders.add(match[1]);
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
// React-intl XML tags: <bold>, </bold>, <link>, </link>
|
|
54
|
-
const xmlPattern = /<\/?(\w+)>/g;
|
|
55
|
-
while ((match = xmlPattern.exec(text)) !== null) {
|
|
56
|
-
placeholders.add(`<${match[1]}>`);
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
return [...placeholders].sort();
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
/**
|
|
63
|
-
* Compare placeholders between source and target strings.
|
|
64
|
-
*
|
|
65
|
-
* @param {string} sourceValue - Source locale value
|
|
66
|
-
* @param {string} targetValue - Target locale value
|
|
67
|
-
* @returns {{ missing: string[], extra: string[] }} Placeholder differences
|
|
68
|
-
*/
|
|
69
|
-
function comparePlaceholders(sourceValue, targetValue) {
|
|
70
|
-
const sourcePH = extractPlaceholders(sourceValue);
|
|
71
|
-
const targetPH = extractPlaceholders(targetValue);
|
|
72
|
-
|
|
73
|
-
const missing = sourcePH.filter(p => !targetPH.includes(p));
|
|
74
|
-
const extra = targetPH.filter(p => !sourcePH.includes(p));
|
|
75
|
-
|
|
76
|
-
return { missing, extra };
|
|
77
|
-
}
|
|
39
|
+
// extractPlaceholders, comparePlaceholders, placeholderSyntax and
|
|
40
|
+
// showPlaceholder live in lib/placeholders.js — one extraction shared with
|
|
41
|
+
// the sync quality gate (lib/validate.js), so sync refuses what verify flags
|
|
42
|
+
// (Round 12). Re-exported below for every existing caller.
|
|
78
43
|
|
|
79
44
|
// -----------------------------------------------------------------
|
|
80
45
|
// Encoding checks
|
|
@@ -316,6 +281,101 @@ function findHollowedValues(sourceFlat, targetFlat, noTranslate = null) {
|
|
|
316
281
|
return findings;
|
|
317
282
|
}
|
|
318
283
|
|
|
284
|
+
/**
|
|
285
|
+
* Find values ON DISK whose ICU MessageFormat / placeholder structure was
|
|
286
|
+
* damaged — the sync gate's own check (validate.js check 2b), applied to
|
|
287
|
+
* what is already written. Values identical to their source and no-translate
|
|
288
|
+
* keys are other checks' business.
|
|
289
|
+
*
|
|
290
|
+
* @param {object} sourceFlat - Flattened source locale
|
|
291
|
+
* @param {object} targetFlat - Flattened target locale
|
|
292
|
+
* @param {string} targetLang - Target locale (decides which plural categories it may add)
|
|
293
|
+
* @param {import('./no-translate.js').NoTranslateMatcher} [noTranslate]
|
|
294
|
+
* @returns {Array<{ key: string, actual: string, reason: string, issues: string[], syntaxes: Array<'icu'|'printf'> }>}
|
|
295
|
+
*/
|
|
296
|
+
function findICUStructureIssues(sourceFlat, targetFlat, targetLang, noTranslate = null) {
|
|
297
|
+
const findings = [];
|
|
298
|
+
for (const [key, sourceVal] of Object.entries(sourceFlat)) {
|
|
299
|
+
const targetVal = targetFlat[key];
|
|
300
|
+
if (typeof sourceVal !== 'string' || typeof targetVal !== 'string') continue;
|
|
301
|
+
if (sourceVal === targetVal) continue;
|
|
302
|
+
if (noTranslate && noTranslate.matches(key, sourceVal)) continue;
|
|
303
|
+
const damage = checkICUStructure(sourceVal, targetVal, targetLang);
|
|
304
|
+
// `syntaxes` (parallel to `issues`): 'icu' or 'printf' — which syntax
|
|
305
|
+
// each finding is about, so a report names a lost %(name)s as printf.
|
|
306
|
+
if (damage) findings.push({ key, actual: targetVal, reason: damage.reason, issues: damage.issues, syntaxes: damage.syntaxes });
|
|
307
|
+
}
|
|
308
|
+
return findings;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* Find values ON DISK whose markup no longer matches the source: a tag
|
|
313
|
+
* opened, closed or nested differently (lib/icu-structure.js checkMarkup —
|
|
314
|
+
* the gate's own check). Echoes and no-translate keys are other checks'.
|
|
315
|
+
*
|
|
316
|
+
* @returns {Array<{ key: string, actual: string, reason: string, issues: string[] }>}
|
|
317
|
+
*/
|
|
318
|
+
function findMarkupIssues(sourceFlat, targetFlat, noTranslate = null) {
|
|
319
|
+
const findings = [];
|
|
320
|
+
for (const [key, sourceVal] of Object.entries(sourceFlat)) {
|
|
321
|
+
const targetVal = targetFlat[key];
|
|
322
|
+
if (typeof sourceVal !== 'string' || typeof targetVal !== 'string') continue;
|
|
323
|
+
if (sourceVal === targetVal) continue;
|
|
324
|
+
if (noTranslate && noTranslate.matches(key, sourceVal)) continue;
|
|
325
|
+
const damage = checkMarkup(sourceVal, targetVal);
|
|
326
|
+
if (damage) findings.push({ key, actual: targetVal, reason: damage.reason, issues: damage.issues });
|
|
327
|
+
}
|
|
328
|
+
return findings;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* Damage in an ARB file OUTSIDE its messages — the parts no flat-map check
|
|
333
|
+
* sees, and the parts an older JSON sync translated:
|
|
334
|
+
* - `@@locale` must name the file's own locale (Flutter's underscore form);
|
|
335
|
+
* gen-l10n refuses a file whose @@locale disagrees with its file name;
|
|
336
|
+
* - a message's `placeholders` metadata (names and Dart types) is code and
|
|
337
|
+
* must equal the source's.
|
|
338
|
+
* Descriptions are not compared — a translated description is legitimate.
|
|
339
|
+
* Any sync that writes the file repairs both (it rebuilds the document from
|
|
340
|
+
* the source), so the finding names a TM-served rebuild.
|
|
341
|
+
*
|
|
342
|
+
* @param {string} sourcePath
|
|
343
|
+
* @param {string} targetPath
|
|
344
|
+
* @param {string} code - The target file's locale
|
|
345
|
+
* @returns {Array<{ key: string, reason: string }>}
|
|
346
|
+
*/
|
|
347
|
+
function auditARBDocument(sourcePath, targetPath, code) {
|
|
348
|
+
const read = (p) => {
|
|
349
|
+
if (!p || !fs.existsSync(p)) return null;
|
|
350
|
+
const raw = fs.readFileSync(p, 'utf-8');
|
|
351
|
+
return raw.trim() ? parseARB(raw, p) : null;
|
|
352
|
+
};
|
|
353
|
+
const source = read(sourcePath);
|
|
354
|
+
const target = read(targetPath);
|
|
355
|
+
if (!target) return [];
|
|
356
|
+
const findings = [];
|
|
357
|
+
const expected = flutterLocale(code);
|
|
358
|
+
if (Object.prototype.hasOwnProperty.call(target, '@@locale') && target['@@locale'] !== expected) {
|
|
359
|
+
findings.push({
|
|
360
|
+
key: '@@locale',
|
|
361
|
+
reason: `@@locale is ${JSON.stringify(target['@@locale'])} — expected "${expected}" (flutter gen-l10n refuses a file whose @@locale disagrees with its name)`,
|
|
362
|
+
});
|
|
363
|
+
}
|
|
364
|
+
for (const [key, meta] of Object.entries(target)) {
|
|
365
|
+
if (!key.startsWith('@') || key.startsWith('@@')) continue;
|
|
366
|
+
const ref = source ? source[key] : undefined;
|
|
367
|
+
if (!ref || typeof ref !== 'object' || !meta || typeof meta !== 'object') continue;
|
|
368
|
+
if (ref.placeholders === undefined && meta.placeholders === undefined) continue;
|
|
369
|
+
if (JSON.stringify(ref.placeholders ?? null) !== JSON.stringify(meta.placeholders ?? null)) {
|
|
370
|
+
findings.push({
|
|
371
|
+
key,
|
|
372
|
+
reason: `placeholder metadata of "${key.slice(1)}" differs from the source (placeholder names and types are code, not text)`,
|
|
373
|
+
});
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
return findings;
|
|
377
|
+
}
|
|
378
|
+
|
|
319
379
|
/**
|
|
320
380
|
* Find Private Use Area codepoints where script conversion is switched OFF.
|
|
321
381
|
*
|
|
@@ -369,17 +429,17 @@ function findOrphanedKeys(sourceFlat, targetFlat) {
|
|
|
369
429
|
// -----------------------------------------------------------------
|
|
370
430
|
|
|
371
431
|
/**
|
|
372
|
-
* Check whether ICU plural strings have the
|
|
432
|
+
* Check whether ICU plural strings have the plural forms the target locale uses.
|
|
373
433
|
*
|
|
374
|
-
* WHY: Arabic
|
|
434
|
+
* WHY: Arabic uses {zero, one, two, few, many, other} but if the LLM only
|
|
375
435
|
* produces {one, other}, the runtime silently falls back to 'other' for all
|
|
376
436
|
* other quantities — producing wrong text for 0, 2, 3-10, 11-99, etc.
|
|
377
437
|
*
|
|
378
|
-
*
|
|
379
|
-
*
|
|
380
|
-
*
|
|
381
|
-
*
|
|
382
|
-
*
|
|
438
|
+
* Missing forms follow the SAME everyday/rare rule sync uses
|
|
439
|
+
* (icu-structure.js pluralGaps): a form ordinary counts reach (Russian
|
|
440
|
+
* few/many) is an issue; one reached only above 1000 or by fractions (French
|
|
441
|
+
* `many`) is a note, not a warning — sync already says that is fine, and
|
|
442
|
+
* integrity warning about it contradicted it (Round 5, Next.js persona).
|
|
383
443
|
*
|
|
384
444
|
* Only triggers on keys where the SOURCE value is an ICU plural string.
|
|
385
445
|
* If the source doesn't use ICU plurals, we don't check (the target shouldn't
|
|
@@ -388,10 +448,14 @@ function findOrphanedKeys(sourceFlat, targetFlat) {
|
|
|
388
448
|
* @param {object} sourceFlat - Flattened source locale
|
|
389
449
|
* @param {object} targetFlat - Flattened target locale
|
|
390
450
|
* @param {string} targetLang - Target locale code
|
|
391
|
-
* @returns {Array<{ key: string, missing: string[], extra: string[] }
|
|
451
|
+
* @returns {{ issues: Array<{ key: string, missing: string[], extra: string[], type: string }>,
|
|
452
|
+
* notes: Array<{ key: string, rare: string[], type: string }> }}
|
|
453
|
+
* issues: everyday forms missing or categories the locale does not have;
|
|
454
|
+
* notes: only forms for large numbers / fractions missing
|
|
392
455
|
*/
|
|
393
|
-
function
|
|
456
|
+
function checkPluralForms(sourceFlat, targetFlat, targetLang, pluralSlots = null) {
|
|
394
457
|
const issues = [];
|
|
458
|
+
const notes = [];
|
|
395
459
|
const requiredCategories = getRequiredPluralCategories(targetLang);
|
|
396
460
|
|
|
397
461
|
for (const [key, sourceVal] of Object.entries(sourceFlat)) {
|
|
@@ -413,12 +477,11 @@ function checkPluralCategories(sourceFlat, targetFlat, targetLang) {
|
|
|
413
477
|
if (!targetPluralNode || !targetPluralNode.options) continue;
|
|
414
478
|
|
|
415
479
|
const targetCategories = Object.keys(targetPluralNode.options);
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
const missing =
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
);
|
|
480
|
+
// A gettext catalog: only the forms its header has slots for (lib/po.js).
|
|
481
|
+
const gaps = pluralGaps(sourceVal, targetVal, targetLang, pluralSlots);
|
|
482
|
+
const missing = [...new Set(gaps.flatMap(g => g.everyday))];
|
|
483
|
+
const rare = [...new Set(gaps.flatMap(g => g.rare))];
|
|
484
|
+
const type = gaps[0]?.type || 'cardinal';
|
|
422
485
|
|
|
423
486
|
// Find unexpected categories (not in CLDR for this locale)
|
|
424
487
|
// Exclude exact-match categories (=0, =1) which are always valid
|
|
@@ -428,11 +491,26 @@ function checkPluralCategories(sourceFlat, targetFlat, targetLang) {
|
|
|
428
491
|
);
|
|
429
492
|
|
|
430
493
|
if (missing.length > 0 || extra.length > 0) {
|
|
431
|
-
issues.push({ key, missing, extra });
|
|
494
|
+
issues.push({ key, missing, extra, type });
|
|
495
|
+
} else if (rare.length > 0) {
|
|
496
|
+
notes.push({ key, rare, type });
|
|
432
497
|
}
|
|
433
498
|
}
|
|
434
499
|
|
|
435
|
-
return issues;
|
|
500
|
+
return { issues, notes };
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
/**
|
|
504
|
+
* Plural issues only (everyday forms missing, unexpected categories) — the
|
|
505
|
+
* shape older callers read. See checkPluralForms for the notes.
|
|
506
|
+
*
|
|
507
|
+
* @param {object} sourceFlat
|
|
508
|
+
* @param {object} targetFlat
|
|
509
|
+
* @param {string} targetLang
|
|
510
|
+
* @returns {Array<{ key: string, missing: string[], extra: string[] }>}
|
|
511
|
+
*/
|
|
512
|
+
function checkPluralCategories(sourceFlat, targetFlat, targetLang) {
|
|
513
|
+
return checkPluralForms(sourceFlat, targetFlat, targetLang).issues;
|
|
436
514
|
}
|
|
437
515
|
|
|
438
516
|
// -----------------------------------------------------------------
|
|
@@ -485,8 +563,17 @@ function auditLocalePair(sourceFlat, targetFlat, targetLang, options = {}) {
|
|
|
485
563
|
const noTranslateDrift = findNoTranslateDrift(sourceFlat, targetFlat, noTranslate);
|
|
486
564
|
const unexpectedPua = findUnexpectedPua(targetFlat, options.scriptExpectation || null);
|
|
487
565
|
const hollowedValues = findHollowedValues(sourceFlat, targetFlat, noTranslate);
|
|
566
|
+
const icuIssues = findICUStructureIssues(sourceFlat, targetFlat, targetLang, noTranslate);
|
|
567
|
+
const markupIssues = findMarkupIssues(sourceFlat, targetFlat, noTranslate);
|
|
568
|
+
// A tag-only placeholder difference is the markup check's finding, said
|
|
569
|
+
// precisely there (which tag, opened or closed) — not twice.
|
|
570
|
+
const markupKeys = new Set(markupIssues.map(m => m.key));
|
|
571
|
+
for (let i = placeholderIssues.length - 1; i >= 0; i--) {
|
|
572
|
+
const p = placeholderIssues[i];
|
|
573
|
+
if (markupKeys.has(p.key) && [...p.missing, ...p.extra].every(t => t.startsWith('<'))) placeholderIssues.splice(i, 1);
|
|
574
|
+
}
|
|
488
575
|
const orphans = findOrphanedKeys(sourceFlat, targetFlat);
|
|
489
|
-
const pluralIssues =
|
|
576
|
+
const { issues: pluralIssues, notes: pluralNotes } = checkPluralForms(sourceFlat, targetFlat, targetLang, options.pluralSlots || null);
|
|
490
577
|
|
|
491
578
|
// File-level BOM check. When the caller passes file paths, REPORT a UTF-8
|
|
492
579
|
// BOM as an issue rather than letting it silently corrupt the first key /
|
|
@@ -498,7 +585,7 @@ function auditLocalePair(sourceFlat, targetFlat, targetLang, options = {}) {
|
|
|
498
585
|
} catch { /* unreadable file — not an integrity finding */ }
|
|
499
586
|
}
|
|
500
587
|
|
|
501
|
-
return { placeholderIssues, encodingIssues, copies, orphans, noTranslateDrift, unexpectedPua, hollowedValues, pluralIssues, bomFiles };
|
|
588
|
+
return { placeholderIssues, encodingIssues, copies, orphans, noTranslateDrift, unexpectedPua, hollowedValues, icuIssues, markupIssues, pluralIssues, pluralNotes, bomFiles };
|
|
502
589
|
}
|
|
503
590
|
|
|
504
591
|
/**
|
|
@@ -519,6 +606,26 @@ function visualize(value) {
|
|
|
519
606
|
return JSON.stringify(escaped);
|
|
520
607
|
}
|
|
521
608
|
|
|
609
|
+
/**
|
|
610
|
+
* Plural forms only large numbers or fractions reach (French `many`): said,
|
|
611
|
+
* never warned — the rule sync uses (icu-structure.js pluralGaps).
|
|
612
|
+
*
|
|
613
|
+
* @param {string} targetLang
|
|
614
|
+
* @param {object} audit
|
|
615
|
+
* @returns {string[]}
|
|
616
|
+
*/
|
|
617
|
+
function pluralNoteLines(targetLang, audit) {
|
|
618
|
+
const notes = audit.pluralNotes || [];
|
|
619
|
+
if (notes.length === 0) return [];
|
|
620
|
+
const langName = getLanguageCard(targetLang)?.name || targetLang;
|
|
621
|
+
const lines = [`\n [INFO] PLURAL FORMS FOR LARGE NUMBERS ONLY (${notes.length}) — not a problem`];
|
|
622
|
+
for (const note of notes.slice(0, 10)) {
|
|
623
|
+
lines.push(` ├── ${note.key}: ${describePluralGap(targetLang, note.rare, note.type || 'cardinal', langName, 'rare')}`);
|
|
624
|
+
}
|
|
625
|
+
if (notes.length > 10) lines.push(` └── ... and ${notes.length - 10} more`);
|
|
626
|
+
return lines;
|
|
627
|
+
}
|
|
628
|
+
|
|
522
629
|
/**
|
|
523
630
|
* Format an integrity audit result as a console report.
|
|
524
631
|
*
|
|
@@ -534,17 +641,22 @@ function formatIntegrityReport(targetLang, audit) {
|
|
|
534
641
|
const drift = audit.noTranslateDrift || [];
|
|
535
642
|
const unexpectedPua = audit.unexpectedPua || [];
|
|
536
643
|
const hollowedValues = audit.hollowedValues || [];
|
|
644
|
+
const icuIssues = audit.icuIssues || [];
|
|
645
|
+
const documentIssues = audit.documentIssues || [];
|
|
537
646
|
const totalIssues = placeholderIssues.length + encodingIssues.length +
|
|
538
647
|
copies.length + orphans.length + (audit.pluralIssues?.length || 0) +
|
|
539
|
-
bomFiles.length + drift.length + unexpectedPua.length + hollowedValues.length
|
|
648
|
+
bomFiles.length + drift.length + unexpectedPua.length + hollowedValues.length + icuIssues.length
|
|
649
|
+
+ documentIssues.length;
|
|
540
650
|
|
|
541
651
|
lines.push(`\n Integrity Audit: ${targetLang}`);
|
|
542
652
|
lines.push(` ${'─'.repeat(40)}`);
|
|
543
653
|
|
|
544
654
|
if (totalIssues === 0) {
|
|
545
|
-
|
|
655
|
+
// Structure, not meaning — say which (Round 3, hospital persona).
|
|
656
|
+
lines.push(' [OK] Structural checks passed — no issues found (the meaning is not checked)');
|
|
657
|
+
lines.push(...pluralNoteLines(targetLang, audit));
|
|
546
658
|
lines.push('');
|
|
547
|
-
return lines.join('\n');
|
|
659
|
+
return lines.join('\n').replace(/\u0004/g, '\u2404');
|
|
548
660
|
}
|
|
549
661
|
|
|
550
662
|
// Placeholder issues
|
|
@@ -564,6 +676,32 @@ function formatIntegrityReport(targetLang, audit) {
|
|
|
564
676
|
}
|
|
565
677
|
}
|
|
566
678
|
|
|
679
|
+
// ICU structure damage — translated keywords/variables, lost # or %d
|
|
680
|
+
if (icuIssues.length > 0) {
|
|
681
|
+
lines.push(`\n ICU STRUCTURE DAMAGE (${icuIssues.length})`);
|
|
682
|
+
lines.push(' └── Syntax that was translated or lost — the app cannot format these.');
|
|
683
|
+
lines.push(' Re-translate: `champollion sync --force-keys <key>` (the cached copy is evicted):');
|
|
684
|
+
for (const issue of icuIssues.slice(0, 10)) {
|
|
685
|
+
lines.push(` ${issue.key}: ${issue.issues.join('; ')}`);
|
|
686
|
+
lines.push(` actual: ${visualize(issue.actual)}`);
|
|
687
|
+
}
|
|
688
|
+
if (icuIssues.length > 10) {
|
|
689
|
+
lines.push(` ... and ${icuIssues.length - 10} more`);
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// Document damage (ARB @@locale / placeholder metadata)
|
|
694
|
+
if (documentIssues.length > 0) {
|
|
695
|
+
lines.push(`\n FILE STRUCTURE DAMAGE (${documentIssues.length})`);
|
|
696
|
+
lines.push(` └── Outside the messages. Any sync that rewrites the file repairs it — e.g. \`champollion sync --pair <src>:${targetLang} --force\` (served from the cache):`);
|
|
697
|
+
for (const issue of documentIssues.slice(0, 10)) {
|
|
698
|
+
lines.push(` ${issue.key}: ${issue.reason}`);
|
|
699
|
+
}
|
|
700
|
+
if (documentIssues.length > 10) {
|
|
701
|
+
lines.push(` ... and ${documentIssues.length - 10} more`);
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
|
|
567
705
|
// Encoding issues
|
|
568
706
|
if (encodingIssues.length > 0) {
|
|
569
707
|
lines.push(`\n [WARN] ENCODING ISSUES (${encodingIssues.length})`);
|
|
@@ -649,13 +787,16 @@ function formatIntegrityReport(targetLang, audit) {
|
|
|
649
787
|
}
|
|
650
788
|
}
|
|
651
789
|
|
|
652
|
-
// Plural category issues
|
|
790
|
+
// Plural category issues — the same everyday/rare rule and wording as sync
|
|
791
|
+
// (icu-structure.js pluralGaps / describePluralGap).
|
|
792
|
+
const langName = getLanguageCard(targetLang)?.name || targetLang;
|
|
653
793
|
if (audit.pluralIssues && audit.pluralIssues.length > 0) {
|
|
654
794
|
lines.push(`\n [WARN] PLURAL CATEGORY ISSUES (${audit.pluralIssues.length})`);
|
|
655
795
|
for (const issue of audit.pluralIssues.slice(0, 10)) {
|
|
656
796
|
lines.push(` ├── ${issue.key}`);
|
|
657
797
|
if (issue.missing.length > 0) {
|
|
658
|
-
lines.push(` │ Missing
|
|
798
|
+
lines.push(` │ Missing: ${describePluralGap(targetLang, issue.missing, issue.type || 'cardinal', langName, 'everyday')}`
|
|
799
|
+
+ ' — the app shows the "other" form for those counts');
|
|
659
800
|
}
|
|
660
801
|
if (issue.extra.length > 0) {
|
|
661
802
|
lines.push(` │ Unexpected categories: ${issue.extra.join(', ')}`);
|
|
@@ -665,13 +806,17 @@ function formatIntegrityReport(targetLang, audit) {
|
|
|
665
806
|
lines.push(` └── ... and ${audit.pluralIssues.length - 10} more`);
|
|
666
807
|
}
|
|
667
808
|
}
|
|
809
|
+
lines.push(...pluralNoteLines(targetLang, audit));
|
|
668
810
|
|
|
669
811
|
lines.push('');
|
|
670
|
-
|
|
812
|
+
// gettext context keys (msgctxt\u0004msgid): print U+0004 visibly as "␄".
|
|
813
|
+
return lines.join('\n').replace(/\u0004/g, '\u2404');
|
|
671
814
|
}
|
|
672
815
|
|
|
673
816
|
export {
|
|
674
817
|
extractPlaceholders,
|
|
818
|
+
placeholderSyntax,
|
|
819
|
+
showPlaceholder,
|
|
675
820
|
comparePlaceholders,
|
|
676
821
|
checkEncoding,
|
|
677
822
|
hasBOM,
|
|
@@ -679,9 +824,12 @@ export {
|
|
|
679
824
|
findNoTranslateDrift,
|
|
680
825
|
findUnexpectedPua,
|
|
681
826
|
findHollowedValues,
|
|
827
|
+
findICUStructureIssues,
|
|
828
|
+
auditARBDocument,
|
|
682
829
|
isLocaleInvariant,
|
|
683
830
|
findOrphanedKeys,
|
|
684
831
|
checkPluralCategories,
|
|
832
|
+
checkPluralForms,
|
|
685
833
|
auditLocalePair,
|
|
686
834
|
formatIntegrityReport,
|
|
687
835
|
visualize,
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How the network commands read and write a language pair — one parser for
|
|
3
|
+
* `register-corpus --pair`, `leaderboard --pair`, `recommend <pair>` and the
|
|
4
|
+
* pair fields of `submit`, so they accept the same spellings and print the
|
|
5
|
+
* same one.
|
|
6
|
+
*
|
|
7
|
+
* WRITTEN (what these commands print): source>target — `eng>crk`. It is the
|
|
8
|
+
* form the leaderboard stores (mt-eval's publish writes `eng>crk`), the form
|
|
9
|
+
* the harness's own --pair flags take, and the form the docs show. In a
|
|
10
|
+
* command it is quoted, `--pair "eng>crk"`: an unquoted > makes the shell
|
|
11
|
+
* write the output to a file.
|
|
12
|
+
*
|
|
13
|
+
* READ (what they accept):
|
|
14
|
+
* eng>crk eng-crk eng:crk eng→crk eng->crk eng,crk "eng crk"
|
|
15
|
+
*
|
|
16
|
+
* The rule that keeps this unambiguous: `-` is also the separator INSIDE a
|
|
17
|
+
* locale code (pt-BR, sr-Latn, crk-Cans). So
|
|
18
|
+
* - with an explicit separator (> : → -> , or a space), the value splits
|
|
19
|
+
* there and either side may carry its own hyphens: "en>pt-BR";
|
|
20
|
+
* - with hyphens only, the value is a pair only when it is exactly two bare
|
|
21
|
+
* language codes of two or three letters: eng-crk, en-fr. crk-Cans is one
|
|
22
|
+
* code with a script subtag, not crk → Cans; en-pt-BR could be en + pt-BR
|
|
23
|
+
* or en-PT + BR (and `br` is Breton). Those are refused, never guessed,
|
|
24
|
+
* and the refusal names the readings in the written form.
|
|
25
|
+
* This is the harness's own rule (arena/mt_eval_harness/pair_notation.py),
|
|
26
|
+
* so `mt-eval` and `champollion` read a dash pair the same way.
|
|
27
|
+
*
|
|
28
|
+
* Project pairs are a different key space: `sync`, `verify` and `serve`
|
|
29
|
+
* --pair name pairs of champollion.config.json, whose keys are written
|
|
30
|
+
* `en:fr`. They go through lib/pairs.js (filterPairGraph), which reads the
|
|
31
|
+
* same separators and settles a many-hyphen dash form against the pairs the
|
|
32
|
+
* project configures.
|
|
33
|
+
*
|
|
34
|
+
* @module language-pair
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
/** The separator a written pair uses. */
|
|
38
|
+
export const PAIR_SEPARATOR = '>';
|
|
39
|
+
|
|
40
|
+
/** One line for help text: how to write a pair, and what else is accepted. */
|
|
41
|
+
export const PAIR_NOTATION_HELP =
|
|
42
|
+
'source>target, e.g. "eng>crk" (quote it: an unquoted > sends the output to a file). '
|
|
43
|
+
+ 'eng-crk, eng:crk and "eng crk" are read the same way; a code with its own hyphen (pt-BR) needs >, e.g. "eng>pt-BR"';
|
|
44
|
+
|
|
45
|
+
// Longest first, so "->" is never read as "-" followed by ">".
|
|
46
|
+
const EXPLICIT_SEPARATORS = ['->', '→', '>', ':', ','];
|
|
47
|
+
|
|
48
|
+
// A language code or locale tag: letters first, then letters/digits, with
|
|
49
|
+
// optional -/_ subtags (eng, crk, pt-BR, sr-Latn, zh-Hant-TW, ace_Arab, qaa).
|
|
50
|
+
const CODE_SHAPE = /^[A-Za-z][A-Za-z0-9]*(?:[-_][A-Za-z0-9]+)*$/;
|
|
51
|
+
|
|
52
|
+
// The dash form: exactly two bare ISO 639 codes (the harness's _HYPHEN_PAIR).
|
|
53
|
+
const DASH_PAIR = /^([A-Za-z]{2,3})-([A-Za-z]{2,3})$/;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Write a pair in the one form these commands print: `eng>crk`.
|
|
57
|
+
*
|
|
58
|
+
* @param {{source: string, target: string}} pair
|
|
59
|
+
* @returns {string}
|
|
60
|
+
*/
|
|
61
|
+
export function formatLanguagePair(pair) {
|
|
62
|
+
return `${pair.source}${PAIR_SEPARATOR}${pair.target}`;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The same, ready to paste into a command: `"eng>crk"` (quoted for the > ).
|
|
67
|
+
*
|
|
68
|
+
* @param {{source: string, target: string}} pair
|
|
69
|
+
* @returns {string}
|
|
70
|
+
*/
|
|
71
|
+
export function quotedLanguagePair(pair) {
|
|
72
|
+
return `"${formatLanguagePair(pair)}"`;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function sides(source, target, shown) {
|
|
76
|
+
const s = source.trim();
|
|
77
|
+
const t = target.trim();
|
|
78
|
+
if (!s) return { ok: false, error: `${shown} names no source language — write source>target, e.g. "eng>crk".` };
|
|
79
|
+
if (!t) return { ok: false, error: `${shown} names no target language — write source>target, e.g. "eng>crk".` };
|
|
80
|
+
for (const code of [s, t]) {
|
|
81
|
+
if (!CODE_SHAPE.test(code)) {
|
|
82
|
+
return {
|
|
83
|
+
ok: false,
|
|
84
|
+
error: `${shown}: "${code}" is not a language code (letters and digits, with -subtags such as pt-BR). `
|
|
85
|
+
+ 'Write the pair source>target, e.g. "eng>crk".',
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return { ok: true, source: s, target: t };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Read a language pair in any accepted spelling (see the module comment).
|
|
94
|
+
* Codes keep the case they were typed in — callers that key on lower case
|
|
95
|
+
* (card ids, the leaderboard) lower it themselves.
|
|
96
|
+
*
|
|
97
|
+
* @param {*} value the raw value (a flag, a positional, a form line)
|
|
98
|
+
* @param {object} [o]
|
|
99
|
+
* @param {string} [o.label] how to name the value in an error (default: the value quoted)
|
|
100
|
+
* @returns {{ok: true, source: string, target: string} | {ok: false, error: string}}
|
|
101
|
+
*/
|
|
102
|
+
export function parseLanguagePair(value, { label } = {}) {
|
|
103
|
+
if (value === undefined || value === null || value === true || value === false) {
|
|
104
|
+
return { ok: false, error: `${label || 'The pair'} needs a value: source>target, e.g. "eng>crk" (or eng-crk).` };
|
|
105
|
+
}
|
|
106
|
+
const raw = String(value).trim();
|
|
107
|
+
const shown = label ? `${label} ${raw}` : `"${raw}"`;
|
|
108
|
+
if (!raw) return { ok: false, error: `${label || 'The pair'} is empty — write source>target, e.g. "eng>crk" (or eng-crk).` };
|
|
109
|
+
|
|
110
|
+
for (const sep of EXPLICIT_SEPARATORS) {
|
|
111
|
+
if (!raw.includes(sep)) continue;
|
|
112
|
+
const parts = raw.split(sep);
|
|
113
|
+
if (parts.length !== 2) {
|
|
114
|
+
return { ok: false, error: `${shown} has more than one "${sep}" — a pair is two codes: source>target, e.g. "eng>crk".` };
|
|
115
|
+
}
|
|
116
|
+
return sides(parts[0], parts[1], shown);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
if (/\s/.test(raw)) {
|
|
120
|
+
const words = raw.split(/\s+/);
|
|
121
|
+
if (words.length === 2) return sides(words[0], words[1], shown);
|
|
122
|
+
if (words.length === 3 && words[1] === '-') return sides(words[0], words[2], shown);
|
|
123
|
+
return { ok: false, error: `${shown} is more than two codes — a pair is source>target, e.g. "eng>crk".` };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const dash = DASH_PAIR.exec(raw);
|
|
127
|
+
if (dash) return { ok: true, source: dash[1], target: dash[2] };
|
|
128
|
+
const hyphens = [];
|
|
129
|
+
for (let i = raw.indexOf('-'); i !== -1; i = raw.indexOf('-', i + 1)) hyphens.push(i);
|
|
130
|
+
if (hyphens.length === 1) {
|
|
131
|
+
// One hyphen, but a side is not a bare 2–3 letter code: crk-Cans is a
|
|
132
|
+
// code with a script subtag, not a pair.
|
|
133
|
+
return {
|
|
134
|
+
ok: false,
|
|
135
|
+
error: `${shown} is not read as a pair: with a hyphen, a pair is two language codes of two or three `
|
|
136
|
+
+ 'letters (eng-crk), and a side of this one is not (crk-Cans, say, is one code with a script subtag). '
|
|
137
|
+
+ 'Write the pair with > between the two codes, e.g. "eng>crk-Cans".',
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
if (hyphens.length === 0) {
|
|
141
|
+
// One code, no separator. The commonest way to get here: `--pair eng>crk`
|
|
142
|
+
// typed without quotes — the shell takes `>crk` as "write the output to a
|
|
143
|
+
// file named crk" and passes just `eng`.
|
|
144
|
+
return {
|
|
145
|
+
ok: false,
|
|
146
|
+
error: `${shown} names one language; a pair names two, source first: "eng>crk" or eng-crk. `
|
|
147
|
+
+ `If you typed ${raw}>… without quotes, the shell read the > as "send the output to a file" `
|
|
148
|
+
+ '(and may have made a file named after the second code): quote the pair, or use the dash form.',
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
const readings = hyphens.map((i) => quotedLanguagePair({ source: raw.slice(0, i), target: raw.slice(i + 1) }));
|
|
152
|
+
return {
|
|
153
|
+
ok: false,
|
|
154
|
+
error: `${shown} has ${hyphens.length} hyphens, so it can be read more than one way (${readings.join(' or ')}). `
|
|
155
|
+
+ 'A code with its own hyphen (pt-BR, sr-Latn) needs > between the two codes — write the one you mean.',
|
|
156
|
+
};
|
|
157
|
+
}
|