champollion 0.3.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -37
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +51 -3
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +289 -88
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +649 -130
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +16 -10
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +197 -38
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +193 -106
- package/lib/seal.mjs +6 -5
- package/lib/sealed-qualifier.mjs +2 -2
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +3 -2
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/DATA-SOVEREIGNTY.md +19 -20
- package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
- package/shared/cards-fallback.json +1 -1
- package/shared/catalogue/card-config.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/docent/faq.en.json +14 -16
- package/shared/docent/system-prompt.md +17 -19
- package/shared/explainers/tc-features.json +15 -15
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/human-services.json +1 -1
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +20 -10
- package/shared/schemas/human-services.schema.json +2 -2
- package/shared/schemas/language-card.schema.json +1 -1
- package/shared/schemas/method-card.schema.json +1 -1
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11333
|
@@ -0,0 +1,929 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ICU MessageFormat STRUCTURE check — what a translation may and may not
|
|
3
|
+
* change in a message like
|
|
4
|
+
*
|
|
5
|
+
* {count, plural, =0 {No events} one {# event this week} other {# events this week}}
|
|
6
|
+
*
|
|
7
|
+
* THE FINDING THIS EXISTS FOR (synthetic users, 2026-10):
|
|
8
|
+
* - next-intl: an ICU plural came back as
|
|
9
|
+
* "{cóúnt, plúrál, óné {# event this week} óthér …}" — the variable name,
|
|
10
|
+
* the `plural` keyword and the selectors were "translated". The quality
|
|
11
|
+
* gate passed it, sync wrote it, locked it and cached it.
|
|
12
|
+
* - Flutter: ICU plurals in an .arb had their keywords translated, so
|
|
13
|
+
* `flutter gen-l10n` refused to build — while sync's gate, `integrity`
|
|
14
|
+
* and `verify` all said "All checks passed".
|
|
15
|
+
*
|
|
16
|
+
* THE RULE — only the message TEXT inside the branches may change:
|
|
17
|
+
* - argument names ({count}, {name}) are code: never renamed, never dropped;
|
|
18
|
+
* - argument types (plural / select / selectordinal / number / date …) are
|
|
19
|
+
* syntax: never translated, never swapped;
|
|
20
|
+
* - selectors (=0, one, other, male …) are syntax:
|
|
21
|
+
* select exactly the source's options;
|
|
22
|
+
* plural/ordinal the source's selectors, PLUS any CLDR category the
|
|
23
|
+
* TARGET language uses (Polish adds few/many, French
|
|
24
|
+
* many) and exact `=N` matches; a source category the
|
|
25
|
+
* target language does not have may be dropped
|
|
26
|
+
* (Japanese has only `other`); `other` is required;
|
|
27
|
+
* - `#` (the number) stays in every plural branch where the source has it,
|
|
28
|
+
* except zero / one / two / =N, where a language may spell the number;
|
|
29
|
+
* - `offset:N` is kept; nested structure is compared branch by branch;
|
|
30
|
+
* - printf conversions (%s, %d, %(name)s, %1$s) are kept — gettext
|
|
31
|
+
* catalogs carry them and `msgfmt --check-format` rejects a mismatch.
|
|
32
|
+
*
|
|
33
|
+
* APOSTROPHES. ICU MessageFormat uses ' as a quote character before a
|
|
34
|
+
* syntax character ('{' is a literal brace); Flutter's gen-l10n does NOT
|
|
35
|
+
* (its use-escaping option is off by default), and neither do most
|
|
36
|
+
* hand-written messages. A value is accepted when its structure matches the
|
|
37
|
+
* source's under EITHER reading, so French "d'{name}" is never rejected for
|
|
38
|
+
* an apostrophe — the check is about damage, not about which runtime reads it.
|
|
39
|
+
*
|
|
40
|
+
* Zero dependencies. The plural categories come from CLDR through
|
|
41
|
+
* Intl.PluralRules (lib/plurals.js) — no hardcoded language table.
|
|
42
|
+
*/
|
|
43
|
+
|
|
44
|
+
import { pluralCategoriesFor } from './plurals.js';
|
|
45
|
+
|
|
46
|
+
/** Argument types whose body is a list of `selector {message}` branches. */
|
|
47
|
+
const BRANCHING_TYPES = new Set(['plural', 'select', 'selectordinal']);
|
|
48
|
+
|
|
49
|
+
/** Argument types with an optional style and no branches. */
|
|
50
|
+
const SIMPLE_TYPES = new Set(['number', 'date', 'time', 'spellout', 'ordinal', 'duration', 'list']);
|
|
51
|
+
|
|
52
|
+
/** Every CLDR plural category name. */
|
|
53
|
+
const CLDR_CATEGORIES = ['zero', 'one', 'two', 'few', 'many', 'other'];
|
|
54
|
+
|
|
55
|
+
/** Plural selectors where a language may write the number as a word (no `#`). */
|
|
56
|
+
const NUMBER_WORD_SELECTOR = /^(?:zero|one|two|=\d+)$/;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* CLDR categories per (locale, type), memoized: the gate and verify ask
|
|
60
|
+
* once per ICU value, and a project can hold thousands of them.
|
|
61
|
+
*/
|
|
62
|
+
const categoryCache = new Map();
|
|
63
|
+
function categoriesFor(locale, type) {
|
|
64
|
+
const id = `${locale}\u0000${type}`;
|
|
65
|
+
if (!categoryCache.has(id)) categoryCache.set(id, pluralCategoriesFor(locale, type));
|
|
66
|
+
return categoryCache.get(id);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
// -----------------------------------------------------------------
|
|
70
|
+
// Parser
|
|
71
|
+
// -----------------------------------------------------------------
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Parse an ICU MessageFormat string strictly.
|
|
75
|
+
*
|
|
76
|
+
* Node shapes:
|
|
77
|
+
* { type: 'text', value, raw } literal text (value has quoting resolved)
|
|
78
|
+
* { type: 'pound', raw: '#' } the number, inside a plural branch
|
|
79
|
+
* { type: 'arg', name, argType, style, raw } {name} / {name, number} / {name, date, short}
|
|
80
|
+
* { type: 'branching', name, keyword, offset, options: [{ selector, nodes, raw }], raw }
|
|
81
|
+
* keyword is what was WRITTEN — 'plural', or a translated 'plúrál'
|
|
82
|
+
* (an unknown keyword followed by `sel {…}` branches still parses as
|
|
83
|
+
* branching, so the damage can be named precisely)
|
|
84
|
+
*
|
|
85
|
+
* @param {string} str
|
|
86
|
+
* @param {{ apostrophes?: 'icu'|'literal' }} [options] - 'icu': ' quotes a
|
|
87
|
+
* following syntax character (ICU / formatjs); 'literal': ' is ordinary
|
|
88
|
+
* text (Flutter gen-l10n default)
|
|
89
|
+
* @returns {{ ok: true, nodes: object[] } | { ok: false, error: string }}
|
|
90
|
+
*/
|
|
91
|
+
function parseMessage(str, { apostrophes = 'icu' } = {}) {
|
|
92
|
+
if (typeof str !== 'string') return { ok: false, error: 'not a string' };
|
|
93
|
+
let pos = 0;
|
|
94
|
+
const quoting = apostrophes === 'icu';
|
|
95
|
+
|
|
96
|
+
const fail = (msg) => {
|
|
97
|
+
const e = new Error(msg);
|
|
98
|
+
e.icuPos = pos;
|
|
99
|
+
throw e;
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
const isSpace = (ch) => /\s/.test(ch);
|
|
103
|
+
const skipSpace = () => { while (pos < str.length && isSpace(str[pos])) pos++; };
|
|
104
|
+
|
|
105
|
+
/** A word: everything up to whitespace or a syntax character. */
|
|
106
|
+
const readWord = () => {
|
|
107
|
+
const start = pos;
|
|
108
|
+
while (pos < str.length && !isSpace(str[pos]) && !',{}'.includes(str[pos])) pos++;
|
|
109
|
+
return str.slice(start, pos);
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
function parseNodes(depth, inPlural) {
|
|
113
|
+
const nodes = [];
|
|
114
|
+
let text = '';
|
|
115
|
+
let rawStart = pos;
|
|
116
|
+
const flush = () => {
|
|
117
|
+
if (pos > rawStart || text) nodes.push({ type: 'text', value: text, raw: str.slice(rawStart, pos) });
|
|
118
|
+
text = '';
|
|
119
|
+
};
|
|
120
|
+
|
|
121
|
+
while (pos < str.length) {
|
|
122
|
+
const ch = str[pos];
|
|
123
|
+
if (ch === '}') {
|
|
124
|
+
if (depth === 0) fail(`unmatched '}' at position ${pos}`);
|
|
125
|
+
flush();
|
|
126
|
+
return nodes;
|
|
127
|
+
}
|
|
128
|
+
if (ch === '{') {
|
|
129
|
+
flush();
|
|
130
|
+
nodes.push(parseArgument(depth, inPlural));
|
|
131
|
+
rawStart = pos;
|
|
132
|
+
continue;
|
|
133
|
+
}
|
|
134
|
+
if (ch === '#' && inPlural) {
|
|
135
|
+
flush();
|
|
136
|
+
nodes.push({ type: 'pound', raw: '#' });
|
|
137
|
+
pos++;
|
|
138
|
+
rawStart = pos;
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
if (ch === "'" && quoting) {
|
|
142
|
+
const next = str[pos + 1];
|
|
143
|
+
if (next === "'") { text += "'"; pos += 2; continue; }
|
|
144
|
+
if (next === '{' || next === '}' || next === '|' || (next === '#' && inPlural)) {
|
|
145
|
+
// Quoted literal: runs to the next lone apostrophe ('' inside = ').
|
|
146
|
+
pos++;
|
|
147
|
+
while (pos < str.length) {
|
|
148
|
+
if (str[pos] === "'") {
|
|
149
|
+
if (str[pos + 1] === "'") { text += "'"; pos += 2; continue; }
|
|
150
|
+
pos++;
|
|
151
|
+
break;
|
|
152
|
+
}
|
|
153
|
+
text += str[pos];
|
|
154
|
+
pos++;
|
|
155
|
+
}
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
text += ch;
|
|
160
|
+
pos++;
|
|
161
|
+
}
|
|
162
|
+
if (depth > 0) fail('unclosed \'{\' — a branch or argument is never closed');
|
|
163
|
+
flush();
|
|
164
|
+
return nodes;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function parseArgument(depth, inPlural) {
|
|
168
|
+
const start = pos;
|
|
169
|
+
pos++; // {
|
|
170
|
+
skipSpace();
|
|
171
|
+
if (str[pos] === '{') fail(`unexpected '{' at position ${pos} (an argument name was expected)`);
|
|
172
|
+
const name = readWord();
|
|
173
|
+
skipSpace();
|
|
174
|
+
if (pos >= str.length) fail(`argument "{${name}" is never closed`);
|
|
175
|
+
|
|
176
|
+
if (str[pos] === '}') {
|
|
177
|
+
pos++;
|
|
178
|
+
return { type: 'arg', name, argType: null, style: null, raw: str.slice(start, pos) };
|
|
179
|
+
}
|
|
180
|
+
if (str[pos] !== ',') fail(`unexpected '${str[pos]}' in argument "{${name}" at position ${pos}`);
|
|
181
|
+
pos++;
|
|
182
|
+
skipSpace();
|
|
183
|
+
const keyword = readWord();
|
|
184
|
+
skipSpace();
|
|
185
|
+
if (!keyword) fail(`argument "{${name}," has no type`);
|
|
186
|
+
|
|
187
|
+
if (str[pos] === '}') {
|
|
188
|
+
pos++;
|
|
189
|
+
return { type: 'arg', name, argType: keyword, style: null, raw: str.slice(start, pos) };
|
|
190
|
+
}
|
|
191
|
+
if (str[pos] !== ',') fail(`unexpected '${str[pos]}' after "{${name}, ${keyword}" at position ${pos}`);
|
|
192
|
+
pos++;
|
|
193
|
+
|
|
194
|
+
if (BRANCHING_TYPES.has(keyword)) {
|
|
195
|
+
const body = parseBranches(depth, inPlural || keyword !== 'select');
|
|
196
|
+
return { type: 'branching', name, keyword, ...body, raw: str.slice(start, pos) };
|
|
197
|
+
}
|
|
198
|
+
if (!SIMPLE_TYPES.has(keyword)) {
|
|
199
|
+
// An unknown keyword followed by branches is a translated `plural` /
|
|
200
|
+
// `select` — parse it as branching so the report can say exactly that.
|
|
201
|
+
const save = pos;
|
|
202
|
+
try {
|
|
203
|
+
const body = parseBranches(depth, true);
|
|
204
|
+
if (body.options.length > 0) {
|
|
205
|
+
return { type: 'branching', name, keyword, ...body, raw: str.slice(start, pos) };
|
|
206
|
+
}
|
|
207
|
+
} catch { /* not branches — fall through to a style */ }
|
|
208
|
+
pos = save;
|
|
209
|
+
}
|
|
210
|
+
// Style: everything up to the matching close brace.
|
|
211
|
+
const styleStart = pos;
|
|
212
|
+
let level = 1;
|
|
213
|
+
while (pos < str.length) {
|
|
214
|
+
if (str[pos] === '{') level++;
|
|
215
|
+
else if (str[pos] === '}') { level--; if (level === 0) break; }
|
|
216
|
+
pos++;
|
|
217
|
+
}
|
|
218
|
+
if (pos >= str.length) fail(`argument "{${name}, ${keyword}, …" is never closed`);
|
|
219
|
+
const style = str.slice(styleStart, pos).trim();
|
|
220
|
+
pos++;
|
|
221
|
+
return { type: 'arg', name, argType: keyword, style, raw: str.slice(start, pos) };
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function parseBranches(depth, inPlural) {
|
|
225
|
+
const options = [];
|
|
226
|
+
let offset = null;
|
|
227
|
+
const seen = new Set();
|
|
228
|
+
for (;;) {
|
|
229
|
+
skipSpace();
|
|
230
|
+
if (pos >= str.length) fail('a plural/select argument is never closed');
|
|
231
|
+
if (str[pos] === '}') { pos++; break; }
|
|
232
|
+
if (str.startsWith('offset:', pos)) {
|
|
233
|
+
pos += 'offset:'.length;
|
|
234
|
+
skipSpace();
|
|
235
|
+
const m = /^\d+/.exec(str.slice(pos));
|
|
236
|
+
if (!m) fail(`offset: needs a number at position ${pos}`);
|
|
237
|
+
offset = Number(m[0]);
|
|
238
|
+
pos += m[0].length;
|
|
239
|
+
continue;
|
|
240
|
+
}
|
|
241
|
+
const selector = readWord();
|
|
242
|
+
if (!selector) fail(`expected a selector at position ${pos}`);
|
|
243
|
+
skipSpace();
|
|
244
|
+
if (str[pos] !== '{') fail(`selector '${selector}' must be followed by '{' (position ${pos})`);
|
|
245
|
+
pos++;
|
|
246
|
+
const contentStart = pos;
|
|
247
|
+
const nodes = parseNodes(depth + 1, inPlural);
|
|
248
|
+
const raw = str.slice(contentStart, pos);
|
|
249
|
+
pos++; // }
|
|
250
|
+
if (seen.has(selector)) fail(`selector '${selector}' appears twice`);
|
|
251
|
+
seen.add(selector);
|
|
252
|
+
options.push({ selector, nodes, raw });
|
|
253
|
+
}
|
|
254
|
+
return { offset, options };
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
try {
|
|
258
|
+
const nodes = parseNodes(0, false);
|
|
259
|
+
return { ok: true, nodes };
|
|
260
|
+
} catch (err) {
|
|
261
|
+
return { ok: false, error: err.message };
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
// -----------------------------------------------------------------
|
|
266
|
+
// Inspection helpers
|
|
267
|
+
// -----------------------------------------------------------------
|
|
268
|
+
|
|
269
|
+
/** Every node of a tree, depth first. */
|
|
270
|
+
function* walk(nodes) {
|
|
271
|
+
for (const node of nodes) {
|
|
272
|
+
yield node;
|
|
273
|
+
if (node.type === 'branching') {
|
|
274
|
+
for (const opt of node.options) yield* walk(opt.nodes);
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** True when the parsed message has at least one argument. */
|
|
280
|
+
function hasArguments(nodes) {
|
|
281
|
+
for (const node of walk(nodes)) {
|
|
282
|
+
if (node.type === 'arg' || node.type === 'branching') return true;
|
|
283
|
+
}
|
|
284
|
+
return false;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/** True when the message has a plural / select / selectordinal argument. */
|
|
288
|
+
function hasBranching(nodes) {
|
|
289
|
+
for (const node of walk(nodes)) {
|
|
290
|
+
if (node.type === 'branching') return true;
|
|
291
|
+
}
|
|
292
|
+
return false;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/** `#` directly in this branch (not inside a nested branching argument). */
|
|
296
|
+
function hasPound(nodes) {
|
|
297
|
+
return nodes.some(n => n.type === 'pound');
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/** "name" or "name, number" — how a simple argument is shown in a report. */
|
|
301
|
+
function argLabel(a) {
|
|
302
|
+
return a.argType ? `{${a.name}, ${a.argType}}` : `{${a.name}}`;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/**
|
|
306
|
+
* The simple arguments (name + type) used anywhere in the tree, in order of
|
|
307
|
+
* first appearance. A SET, not a multiset: a language that drops a plural
|
|
308
|
+
* branch (Japanese has no `one`) legitimately uses {count} fewer times.
|
|
309
|
+
*/
|
|
310
|
+
function simpleArgIds(nodes) {
|
|
311
|
+
const ids = [];
|
|
312
|
+
for (const node of walk(nodes)) {
|
|
313
|
+
if (node.type !== 'arg') continue;
|
|
314
|
+
const id = `${node.name}\u0000${node.argType || ''}`;
|
|
315
|
+
if (!ids.includes(id)) ids.push(id);
|
|
316
|
+
}
|
|
317
|
+
return ids;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
// -----------------------------------------------------------------
|
|
321
|
+
// printf conversions (gettext c-format / python-format, Android, Rails)
|
|
322
|
+
// -----------------------------------------------------------------
|
|
323
|
+
|
|
324
|
+
/**
|
|
325
|
+
* printf conversions in a string, normalized for comparison: positional
|
|
326
|
+
* indexes are dropped ("%1$s" → "%s", so a translation may reorder), named
|
|
327
|
+
* ones keep their name ("%(count)d"). "%%" is a literal percent sign.
|
|
328
|
+
* The space flag is deliberately NOT recognised: "50% off" is text.
|
|
329
|
+
*
|
|
330
|
+
* @param {string} text
|
|
331
|
+
* @returns {string[]} Sorted conversions
|
|
332
|
+
*/
|
|
333
|
+
function printfConversions(text) {
|
|
334
|
+
if (typeof text !== 'string' || !text.includes('%')) return [];
|
|
335
|
+
const out = [];
|
|
336
|
+
const re = /%(?:(\d+)\$)?(\([^)\s]+\))?[-+#0]*(?:\d+|\*)?(?:\.(?:\d+|\*))?(?:hh|h|ll|l|L|q|j|z|t)?([sdifuxXoeEgGcpr@])/g;
|
|
337
|
+
const stripped = text.replace(/%%/g, '');
|
|
338
|
+
let m;
|
|
339
|
+
while ((m = re.exec(stripped)) !== null) {
|
|
340
|
+
// %i and %d are the same conversion (msgfmt --check-format agrees).
|
|
341
|
+
out.push(`%${m[2] || ''}${m[3] === 'i' ? 'd' : m[3]}`);
|
|
342
|
+
}
|
|
343
|
+
return out.sort();
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
// -----------------------------------------------------------------
|
|
347
|
+
// Structural comparison
|
|
348
|
+
// -----------------------------------------------------------------
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* Compare a translation's ICU structure with its source's (one apostrophe
|
|
352
|
+
* reading). Returns the list of damage found; empty means only text changed.
|
|
353
|
+
*/
|
|
354
|
+
function compareParsed(srcNodes, tgtNodes, locale) {
|
|
355
|
+
const issues = [];
|
|
356
|
+
compareLevel(srcNodes, tgtNodes, locale, issues);
|
|
357
|
+
|
|
358
|
+
// Simple arguments, anywhere in the message: never dropped, renamed or
|
|
359
|
+
// retyped. (Compared over the whole message — a translation may move
|
|
360
|
+
// {name} from one branch to the sentence around it.)
|
|
361
|
+
const src = simpleArgIds(srcNodes);
|
|
362
|
+
const tgt = simpleArgIds(tgtNodes);
|
|
363
|
+
const label = (id) => {
|
|
364
|
+
const [name, argType] = id.split('\u0000');
|
|
365
|
+
return argLabel({ name, argType: argType || null });
|
|
366
|
+
};
|
|
367
|
+
const missing = src.filter(id => !tgt.includes(id));
|
|
368
|
+
const extra = tgt.filter(id => !src.includes(id));
|
|
369
|
+
// A renamed placeholder reads as one missing + one extra: say so.
|
|
370
|
+
while (missing.length > 0 && extra.length > 0) {
|
|
371
|
+
issues.push(`placeholder ${label(missing.shift())} was changed to ${label(extra.shift())}`);
|
|
372
|
+
}
|
|
373
|
+
for (const id of missing) issues.push(`placeholder ${label(id)} is missing`);
|
|
374
|
+
for (const id of extra) issues.push(`placeholder ${label(id)} is not in the source`);
|
|
375
|
+
return issues;
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* Compare the branching arguments of one message level (the top level, or
|
|
380
|
+
* the inside of one branch) and recurse into their branches.
|
|
381
|
+
*/
|
|
382
|
+
function compareLevel(srcNodes, tgtNodes, locale, issues) {
|
|
383
|
+
const srcB = srcNodes.filter(n => n.type === 'branching');
|
|
384
|
+
const tgtB = tgtNodes.filter(n => n.type === 'branching');
|
|
385
|
+
const used = new Set();
|
|
386
|
+
const pairs = [];
|
|
387
|
+
const unmatchedSrc = [];
|
|
388
|
+
for (const s of srcB) {
|
|
389
|
+
const t = tgtB.find(c => !used.has(c) && c.name === s.name);
|
|
390
|
+
if (t) { used.add(t); pairs.push([s, t]); } else unmatchedSrc.push(s);
|
|
391
|
+
}
|
|
392
|
+
const unmatchedTgt = tgtB.filter(t => !used.has(t));
|
|
393
|
+
// A variable that is "missing" while an unknown one appears in its place
|
|
394
|
+
// was translated: pair them in order and report the rename.
|
|
395
|
+
while (unmatchedSrc.length > 0 && unmatchedTgt.length > 0) {
|
|
396
|
+
const s = unmatchedSrc.shift();
|
|
397
|
+
const t = unmatchedTgt.shift();
|
|
398
|
+
issues.push(`ICU variable '${s.name}' was translated to '${t.name}'`);
|
|
399
|
+
pairs.push([s, t]);
|
|
400
|
+
}
|
|
401
|
+
for (const s of unmatchedSrc) issues.push(`ICU ${s.keyword} argument '${s.name}' is missing`);
|
|
402
|
+
for (const t of unmatchedTgt) {
|
|
403
|
+
issues.push(`ICU argument '{${t.name}, ${t.keyword}, …}' is not in the source`);
|
|
404
|
+
}
|
|
405
|
+
for (const [s, t] of pairs) compareBranching(s, t, locale, issues);
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
/** Compare one plural/select/selectordinal argument with its translation. */
|
|
409
|
+
function compareBranching(s, t, locale, issues) {
|
|
410
|
+
const kind = s.keyword; // the source is the reference
|
|
411
|
+
if (t.keyword !== s.keyword) {
|
|
412
|
+
issues.push(BRANCHING_TYPES.has(t.keyword)
|
|
413
|
+
? `ICU argument '${s.name}' changed from ${s.keyword} to ${t.keyword}`
|
|
414
|
+
: `ICU keyword '${s.keyword}' was translated to '${t.keyword}'`);
|
|
415
|
+
}
|
|
416
|
+
if ((s.offset ?? null) !== (t.offset ?? null)) {
|
|
417
|
+
issues.push(`ICU offset for '${s.name}' changed (${s.offset ?? 'none'} → ${t.offset ?? 'none'})`);
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
const isPlural = kind === 'plural' || kind === 'selectordinal';
|
|
421
|
+
const targetCats = isPlural
|
|
422
|
+
? categoriesFor(locale, kind === 'selectordinal' ? 'ordinal' : 'cardinal')
|
|
423
|
+
: null;
|
|
424
|
+
const srcSel = s.options.map(o => o.selector);
|
|
425
|
+
const tgtSel = t.options.map(o => o.selector);
|
|
426
|
+
|
|
427
|
+
const mayAdd = (sel) => isPlural && (/^=\d+$/.test(sel)
|
|
428
|
+
|| (targetCats ? targetCats.includes(sel) : CLDR_CATEGORIES.includes(sel)));
|
|
429
|
+
const mayDrop = (sel) => isPlural && sel !== 'other' && targetCats
|
|
430
|
+
&& CLDR_CATEGORIES.includes(sel) && !targetCats.includes(sel);
|
|
431
|
+
|
|
432
|
+
const absent = srcSel.filter(sel => !tgtSel.includes(sel));
|
|
433
|
+
let extra = tgtSel.filter(sel => !srcSel.includes(sel) && !mayAdd(sel));
|
|
434
|
+
const renamedFrom = new Map(); // target selector → source selector it replaced
|
|
435
|
+
const rename = (from, to) => {
|
|
436
|
+
issues.push(`ICU keyword '${from}' was translated to '${to}'`);
|
|
437
|
+
renamedFrom.set(to, from);
|
|
438
|
+
};
|
|
439
|
+
// An unknown selector standing where a source selector was is that
|
|
440
|
+
// selector, translated ("óthér" in the slot of "other") — even one the
|
|
441
|
+
// target language could have dropped. A real CLDR category name in the
|
|
442
|
+
// wrong language ("few" in French) is not a translation; it is reported
|
|
443
|
+
// as a category the language does not have.
|
|
444
|
+
const translatable = (sel) => !isPlural || !(CLDR_CATEGORIES.includes(sel) || /^=\d+$/.test(sel));
|
|
445
|
+
for (const to of [...extra]) {
|
|
446
|
+
if (!translatable(to)) continue;
|
|
447
|
+
const from = srcSel[tgtSel.indexOf(to)];
|
|
448
|
+
if (from && absent.includes(from) && ![...renamedFrom.values()].includes(from)) {
|
|
449
|
+
rename(from, to);
|
|
450
|
+
extra = extra.filter(e => e !== to);
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
const missing = absent.filter(sel => !mayDrop(sel) && ![...renamedFrom.values()].includes(sel));
|
|
454
|
+
const words = extra.filter(translatable);
|
|
455
|
+
while (missing.length > 0 && words.length > 0) {
|
|
456
|
+
const to = words.shift();
|
|
457
|
+
rename(missing.shift(), to);
|
|
458
|
+
extra = extra.filter(e => e !== to);
|
|
459
|
+
}
|
|
460
|
+
for (const sel of missing) issues.push(`ICU selector '${sel}' of '${s.name}' was removed`);
|
|
461
|
+
for (const sel of extra) {
|
|
462
|
+
issues.push(isPlural
|
|
463
|
+
? `'${sel}' is not one of the ${kind === 'selectordinal' ? 'ordinal' : 'plural'} categories of ${locale}`
|
|
464
|
+
+ (targetCats ? ` (${targetCats.join(', ')})` : '')
|
|
465
|
+
: `select option '${sel}' of '${s.name}' is not in the source`);
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
// Branch by branch: nested structure, and the number placeholder.
|
|
469
|
+
const sourcePrintf = [...new Set(s.options.flatMap(o => printfConversions(o.raw)))];
|
|
470
|
+
const sourceUsesPrintf = sourcePrintf.length > 0;
|
|
471
|
+
const srcBySel = new Map(s.options.map(o => [o.selector, o]));
|
|
472
|
+
const srcOther = srcBySel.get('other') || s.options[s.options.length - 1];
|
|
473
|
+
for (const opt of t.options) {
|
|
474
|
+
const ref = srcBySel.get(opt.selector)
|
|
475
|
+
|| srcBySel.get(renamedFrom.get(opt.selector))
|
|
476
|
+
|| srcOther;
|
|
477
|
+
if (!ref) continue;
|
|
478
|
+
compareLevel(ref.nodes, opt.nodes, locale, issues);
|
|
479
|
+
if (isPlural && hasPound(ref.nodes) && !hasPound(opt.nodes)
|
|
480
|
+
&& !NUMBER_WORD_SELECTOR.test(opt.selector)) {
|
|
481
|
+
issues.push(`the number placeholder # is missing from the '${opt.selector}' branch of '${s.name}'`);
|
|
482
|
+
}
|
|
483
|
+
// A plural whose branches count with printf conversions (a gettext
|
|
484
|
+
// plural) has no `#`: an ICU `#` there would be written out literally.
|
|
485
|
+
if (isPlural && sourceUsesPrintf && hasPound(opt.nodes) && !hasPound(ref.nodes)) {
|
|
486
|
+
issues.push(`'#' in the '${opt.selector}' branch of '${s.name}' — this message counts with printf placeholders (${sourcePrintf.join(', ')}); keep those instead of #`);
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
/**
|
|
492
|
+
* Check that a translation kept its source's ICU MessageFormat structure.
|
|
493
|
+
*
|
|
494
|
+
* @param {string} source - Source value
|
|
495
|
+
* @param {string} translated - Candidate translation
|
|
496
|
+
* @param {string} locale - Target locale (decides which plural categories it may add)
|
|
497
|
+
* @returns {{ reason: string, issues: string[], syntaxes: Array<'icu'|'printf'> }|null}
|
|
498
|
+
* null when the source carries no ICU arguments / printf conversions, or
|
|
499
|
+
* the structure survived. `syntaxes[i]` names the placeholder syntax
|
|
500
|
+
* `issues[i]` is about — 'icu' (MessageFormat arguments, keywords,
|
|
501
|
+
* selectors, #) or 'printf' (%s, %d, %(name)s) — so a report can name a
|
|
502
|
+
* lost `%(name)s` in a gettext catalog as printf, not as ICU (Round 12,
|
|
503
|
+
* Django persona: verify called it an "ICU structure error")
|
|
504
|
+
*/
|
|
505
|
+
function checkICUStructure(source, translated, locale) {
|
|
506
|
+
if (typeof source !== 'string' || typeof translated !== 'string') return null;
|
|
507
|
+
|
|
508
|
+
let issues = null;
|
|
509
|
+
let syntaxes = null;
|
|
510
|
+
|
|
511
|
+
if (source.includes('{')) {
|
|
512
|
+
let firstIssues = null;
|
|
513
|
+
let applicable = false;
|
|
514
|
+
for (const apostrophes of ['icu', 'literal']) {
|
|
515
|
+
const src = parseMessage(source, { apostrophes });
|
|
516
|
+
// A source that is not a well-formed message under this reading has
|
|
517
|
+
// no structure to protect under it ({{count}}, Hugo "{{ .Count }}").
|
|
518
|
+
if (!src.ok || !hasArguments(src.nodes)) continue;
|
|
519
|
+
applicable = true;
|
|
520
|
+
const tgt = parseMessage(translated, { apostrophes });
|
|
521
|
+
const found = tgt.ok
|
|
522
|
+
? compareParsed(src.nodes, tgt.nodes, locale)
|
|
523
|
+
: [`ICU syntax broken (${tgt.error})`];
|
|
524
|
+
if (found.length === 0) { firstIssues = []; break; }
|
|
525
|
+
if (firstIssues === null) firstIssues = found;
|
|
526
|
+
}
|
|
527
|
+
if (applicable && firstIssues && firstIssues.length > 0) {
|
|
528
|
+
issues = firstIssues;
|
|
529
|
+
syntaxes = firstIssues.map(() => 'icu');
|
|
530
|
+
}
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
// printf conversions — independent of ICU (a gettext msgid has no braces).
|
|
534
|
+
// Compared as SETS: a plural translation that adds a category (Russian
|
|
535
|
+
// few/many) repeats %d legitimately.
|
|
536
|
+
const srcPrintf = [...new Set(printfConversions(source))];
|
|
537
|
+
if (srcPrintf.length > 0) {
|
|
538
|
+
const tgtPrintf = [...new Set(printfConversions(translated))];
|
|
539
|
+
const missing = srcPrintf.filter(c => !tgtPrintf.includes(c));
|
|
540
|
+
const extra = tgtPrintf.filter(c => !srcPrintf.includes(c));
|
|
541
|
+
if (missing.length > 0 || extra.length > 0) {
|
|
542
|
+
issues = issues || [];
|
|
543
|
+
syntaxes = syntaxes || [];
|
|
544
|
+
for (const c of missing) { issues.push(`printf placeholder ${c} is missing`); syntaxes.push('printf'); }
|
|
545
|
+
for (const c of extra) { issues.push(`printf placeholder ${c} is not in the source`); syntaxes.push('printf'); }
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
if (!issues || issues.length === 0) return null;
|
|
550
|
+
return { reason: `ICU/placeholder structure damaged: ${issues.join('; ')}`, issues, syntaxes };
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
/**
|
|
554
|
+
* Prompt guidance for a source value carrying plural / select / selectordinal
|
|
555
|
+
* syntax, naming the plural categories the TARGET language uses (CLDR).
|
|
556
|
+
*
|
|
557
|
+
* @param {string} source
|
|
558
|
+
* @param {string} locale - Target locale
|
|
559
|
+
* @param {string} [languageName] - Display name for the prompt
|
|
560
|
+
* @returns {string|null}
|
|
561
|
+
*/
|
|
562
|
+
function icuGuidance(source, locale, languageName = locale) {
|
|
563
|
+
if (typeof source !== 'string' || !source.includes('{')) return null;
|
|
564
|
+
const parsed = parseMessage(source, { apostrophes: 'icu' });
|
|
565
|
+
const nodes = parsed.ok ? parsed.nodes : (parseMessage(source, { apostrophes: 'literal' }).nodes || null);
|
|
566
|
+
if (!nodes || !hasBranching(nodes)) return null;
|
|
567
|
+
|
|
568
|
+
const kinds = new Set();
|
|
569
|
+
const names = new Set();
|
|
570
|
+
for (const node of walk(nodes)) {
|
|
571
|
+
if (node.type === 'branching') { kinds.add(node.keyword); names.add(node.name); }
|
|
572
|
+
}
|
|
573
|
+
const parts = [
|
|
574
|
+
`ICU MessageFormat: keep the syntax exactly — the variable name(s) ${[...names].map(n => `"${n}"`).join(', ')}, `
|
|
575
|
+
+ `the word(s) ${[...kinds].join('/')}, every selector (=0, one, other, …), # and {placeholders} — `
|
|
576
|
+
+ 'and translate only the text inside the branches.',
|
|
577
|
+
];
|
|
578
|
+
if (kinds.has('plural')) {
|
|
579
|
+
const cats = categoriesFor(locale, 'cardinal');
|
|
580
|
+
if (cats) {
|
|
581
|
+
parts.push(`${languageName} plural categories (CLDR): ${describeCategories(locale, cats, 'cardinal')}`
|
|
582
|
+
+ ' — write a branch for each one it needs, keep "other".');
|
|
583
|
+
}
|
|
584
|
+
}
|
|
585
|
+
if (kinds.has('selectordinal')) {
|
|
586
|
+
const cats = categoriesFor(locale, 'ordinal');
|
|
587
|
+
if (cats) parts.push(`${languageName} ordinal categories (CLDR): ${describeCategories(locale, cats, 'ordinal')}.`);
|
|
588
|
+
}
|
|
589
|
+
return parts.join(' ');
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
// -----------------------------------------------------------------
|
|
593
|
+
// Plural forms a translation did not supply
|
|
594
|
+
// -----------------------------------------------------------------
|
|
595
|
+
|
|
596
|
+
/** The largest count treated as "everyday" (see pluralCategoryUse). */
|
|
597
|
+
const EVERYDAY_MAX = 1000;
|
|
598
|
+
const useCache = new Map();
|
|
599
|
+
|
|
600
|
+
/**
|
|
601
|
+
* Which of a locale's CLDR plural categories ordinary counts reach.
|
|
602
|
+
*
|
|
603
|
+
* everyday categories CLDR assigns to some integer 0…1000 (besides
|
|
604
|
+
* `other`, which ICU requires anyway): Russian one/few/many,
|
|
605
|
+
* Polish one/few/many, Arabic zero/one/two/few/many. A message
|
|
606
|
+
* without one of them shows the wrong form for counts a UI
|
|
607
|
+
* displays all the time (Russian "2 файлов").
|
|
608
|
+
* rare categories reached only by larger or fractional numbers
|
|
609
|
+
* (French/Spanish/Italian `many`: 1 000 000; Lithuanian `many`:
|
|
610
|
+
* fractions). Missing one is worth saying, not worth a warning.
|
|
611
|
+
*
|
|
612
|
+
* @param {string} locale
|
|
613
|
+
* @param {'cardinal'|'ordinal'} [type='cardinal']
|
|
614
|
+
* @returns {{ categories: string[], everyday: string[], rare: string[],
|
|
615
|
+
* members: Map<string, number[]> } | null} null when CLDR has no rules
|
|
616
|
+
* for the locale; `members` = the integers 0…EVERYDAY_MAX per category
|
|
617
|
+
*/
|
|
618
|
+
function pluralCategoryUse(locale, type = 'cardinal') {
|
|
619
|
+
const id = `${locale}\u0000${type}`;
|
|
620
|
+
if (useCache.has(id)) return useCache.get(id);
|
|
621
|
+
const categories = categoriesFor(locale, type);
|
|
622
|
+
let result = null;
|
|
623
|
+
if (categories) {
|
|
624
|
+
let rules = null;
|
|
625
|
+
try { rules = new Intl.PluralRules(String(locale).replace(/_/g, '-'), { type }); } catch { rules = null; }
|
|
626
|
+
const members = new Map(categories.map(c => [c, []]));
|
|
627
|
+
if (rules) {
|
|
628
|
+
for (let n = 0; n <= EVERYDAY_MAX; n++) members.get(rules.select(n))?.push(n);
|
|
629
|
+
}
|
|
630
|
+
const everyday = categories.filter(c => c !== 'other' && members.get(c).length > 0);
|
|
631
|
+
const rare = categories.filter(c => c !== 'other' && members.get(c).length === 0);
|
|
632
|
+
result = { categories, everyday, rare, members };
|
|
633
|
+
}
|
|
634
|
+
useCache.set(id, result);
|
|
635
|
+
return result;
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
/** Parse under the first apostrophe reading that yields a plural/select. */
|
|
639
|
+
function parseBranching(text) {
|
|
640
|
+
if (typeof text !== 'string' || !text.includes('{')) return null;
|
|
641
|
+
for (const apostrophes of ['literal', 'icu']) {
|
|
642
|
+
const parsed = parseMessage(text, { apostrophes });
|
|
643
|
+
if (parsed.ok && hasBranching(parsed.nodes)) return parsed.nodes;
|
|
644
|
+
}
|
|
645
|
+
return null;
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
/**
|
|
649
|
+
* The CLDR plural categories a translated ICU message does not supply for
|
|
650
|
+
* its target locale — the forms the runtime will take from `other`
|
|
651
|
+
* (ICU, next-intl, Flutter ARB) or that a gettext catalog repeats from it.
|
|
652
|
+
*
|
|
653
|
+
* Only arguments whose SOURCE inflects are judged: a source plural with
|
|
654
|
+
* nothing but `other` ("{n, plural, other {Items: #}}") is count-agnostic
|
|
655
|
+
* by design. An everyday category is also supplied when exact `=N`
|
|
656
|
+
* branches cover every count it takes (French `one` is 0 and 1: `=0` and
|
|
657
|
+
* `=1` together cover it).
|
|
658
|
+
*
|
|
659
|
+
* A gettext catalog holds only the forms its Plural-Forms header has slots
|
|
660
|
+
* for (`slots`, lib/po.js poPluralSlots): a cardinal form without a slot
|
|
661
|
+
* cannot be written, so it is not a gap (French "many" in a two-form
|
|
662
|
+
* catalog; Hebrew "two" in msginit's two-form header).
|
|
663
|
+
*
|
|
664
|
+
* @param {string} source - Source message
|
|
665
|
+
* @param {string} translated - Translated message
|
|
666
|
+
* @param {string} locale - Target locale
|
|
667
|
+
* @param {string[]|null} [slots] - Cardinal categories the target can hold (null = all CLDR's)
|
|
668
|
+
* @returns {Array<{ name: string, type: 'cardinal'|'ordinal', everyday: string[], rare: string[] }>}
|
|
669
|
+
* one entry per plural/selectordinal argument with something missing
|
|
670
|
+
*/
|
|
671
|
+
function pluralGaps(source, translated, locale, slots = null) {
|
|
672
|
+
const tgt = parseBranching(translated);
|
|
673
|
+
const src = tgt ? parseBranching(source) : null;
|
|
674
|
+
if (!tgt || !src) return [];
|
|
675
|
+
const srcArgs = [...walk(src)].filter(n => n.type === 'branching');
|
|
676
|
+
const gaps = [];
|
|
677
|
+
for (const node of walk(tgt)) {
|
|
678
|
+
if (node.type !== 'branching' || (node.keyword !== 'plural' && node.keyword !== 'selectordinal')) continue;
|
|
679
|
+
const ref = srcArgs.find(s => s.name === node.name && s.keyword === node.keyword);
|
|
680
|
+
if (!ref || ref.options.every(o => o.selector === 'other')) continue;
|
|
681
|
+
const type = node.keyword === 'selectordinal' ? 'ordinal' : 'cardinal';
|
|
682
|
+
const use = pluralCategoryUse(locale, type);
|
|
683
|
+
if (!use) continue;
|
|
684
|
+
const present = new Set(node.options.map(o => o.selector));
|
|
685
|
+
const exact = new Set([...present].filter(sel => /^=\d+$/.test(sel)).map(sel => Number(sel.slice(1))));
|
|
686
|
+
const covered = (cat) => present.has(cat)
|
|
687
|
+
|| (use.members.get(cat).length > 0 && use.members.get(cat).length <= exact.size
|
|
688
|
+
&& use.members.get(cat).every(n => exact.has(n)));
|
|
689
|
+
const holds = (c) => !slots || type !== 'cardinal' || slots.includes(c);
|
|
690
|
+
const everyday = use.everyday.filter(c => holds(c) && !covered(c));
|
|
691
|
+
const rare = use.rare.filter(c => holds(c) && !present.has(c));
|
|
692
|
+
if (everyday.length > 0 || rare.length > 0) gaps.push({ name: node.name, type, everyday, rare });
|
|
693
|
+
}
|
|
694
|
+
return gaps;
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
/**
|
|
698
|
+
* The text of each translated plural/select branch beside the source branch
|
|
699
|
+
* it translates: the same selector, or — for a category the target adds
|
|
700
|
+
* (Russian few/many from English one/other) — the source's `other`. Nested
|
|
701
|
+
* branching pairs recursively by argument name. So per-branch checks (an
|
|
702
|
+
* echo of the English in ONE form) can run on a message whose whole string
|
|
703
|
+
* differs from the source — a gettext msgstr[n] plural reads as one ICU
|
|
704
|
+
* message here (lib/po.js).
|
|
705
|
+
*
|
|
706
|
+
* @param {string} source
|
|
707
|
+
* @param {string} translated
|
|
708
|
+
* @returns {Array<{ selector: string, source: string, translated: string }>} [] when
|
|
709
|
+
* either side has no plural/select
|
|
710
|
+
*/
|
|
711
|
+
function pluralBranchPairs(source, translated) {
|
|
712
|
+
const tgt = parseBranching(translated);
|
|
713
|
+
const src = tgt ? parseBranching(source) : null;
|
|
714
|
+
if (!tgt || !src) return [];
|
|
715
|
+
const out = [];
|
|
716
|
+
const visit = (srcNodes, tgtNodes) => {
|
|
717
|
+
const srcArgs = srcNodes.filter(n => n.type === 'branching');
|
|
718
|
+
for (const node of tgtNodes) {
|
|
719
|
+
if (node.type !== 'branching') continue;
|
|
720
|
+
const ref = srcArgs.find(a => a.name === node.name) || null;
|
|
721
|
+
if (!ref) continue;
|
|
722
|
+
for (const opt of node.options) {
|
|
723
|
+
const match = ref.options.find(o => o.selector === opt.selector)
|
|
724
|
+
|| ref.options.find(o => o.selector === 'other');
|
|
725
|
+
if (!match) continue;
|
|
726
|
+
if (opt.nodes.some(n => n.type === 'branching') && match.nodes.some(n => n.type === 'branching')) {
|
|
727
|
+
visit(match.nodes, opt.nodes);
|
|
728
|
+
} else {
|
|
729
|
+
out.push({ selector: opt.selector, source: match.raw, translated: opt.raw, keyword: node.keyword, name: node.name });
|
|
730
|
+
}
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
};
|
|
734
|
+
visit(src, tgt);
|
|
735
|
+
return out;
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
/**
|
|
739
|
+
* The leaf branch texts of a plural/select message (nested branching
|
|
740
|
+
* flattened), or null when it has none. The repetition detector measures
|
|
741
|
+
* each branch on its own: a Russian plural repeats one sentence in four
|
|
742
|
+
* forms by design.
|
|
743
|
+
*
|
|
744
|
+
* @param {string} text
|
|
745
|
+
* @returns {string[]|null}
|
|
746
|
+
*/
|
|
747
|
+
function pluralBranchTexts(text) {
|
|
748
|
+
const nodes = parseBranching(text);
|
|
749
|
+
if (!nodes) return null;
|
|
750
|
+
const out = [];
|
|
751
|
+
const visit = (list) => {
|
|
752
|
+
for (const node of list) {
|
|
753
|
+
if (node.type !== 'branching') continue;
|
|
754
|
+
for (const opt of node.options) {
|
|
755
|
+
if (opt.nodes.some(n => n.type === 'branching')) visit(opt.nodes);
|
|
756
|
+
else out.push(opt.raw);
|
|
757
|
+
}
|
|
758
|
+
}
|
|
759
|
+
};
|
|
760
|
+
visit(nodes);
|
|
761
|
+
return out.length > 0 ? out : null;
|
|
762
|
+
}
|
|
763
|
+
|
|
764
|
+
// -----------------------------------------------------------------
|
|
765
|
+
// Markup: tags are code too
|
|
766
|
+
// -----------------------------------------------------------------
|
|
767
|
+
|
|
768
|
+
/** A tag: <b>, </b>, <br/>, <a href="…">, react-intl <link>, react-i18next <0>. */
|
|
769
|
+
const TAG = /<(\/?)([A-Za-z][\w:.-]*|\d+)((?:\s+[^<>]*?)?)\s*(\/?)>/g;
|
|
770
|
+
/** HTML elements that never close. */
|
|
771
|
+
const VOID_ELEMENTS = new Set(['area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', 'meta', 'param', 'source', 'track', 'wbr']);
|
|
772
|
+
|
|
773
|
+
/**
|
|
774
|
+
* Tags in a text, with their nesting: per tag name the open / close /
|
|
775
|
+
* self-closing counts, the (tag, parent) pairs, and whether it nests.
|
|
776
|
+
*/
|
|
777
|
+
function readTags(text) {
|
|
778
|
+
const counts = new Map();
|
|
779
|
+
const parents = [];
|
|
780
|
+
const stack = [];
|
|
781
|
+
let wellNested = true;
|
|
782
|
+
const bump = (name, field) => {
|
|
783
|
+
if (!counts.has(name)) counts.set(name, { open: 0, close: 0, self: 0 });
|
|
784
|
+
counts.get(name)[field]++;
|
|
785
|
+
};
|
|
786
|
+
for (const m of String(text).matchAll(TAG)) {
|
|
787
|
+
const [, slash, name, , selfSlash] = m;
|
|
788
|
+
const isVoid = VOID_ELEMENTS.has(name.toLowerCase());
|
|
789
|
+
if (slash) {
|
|
790
|
+
bump(name, 'close');
|
|
791
|
+
if (stack.length > 0 && stack[stack.length - 1] === name) stack.pop();
|
|
792
|
+
else wellNested = false;
|
|
793
|
+
} else if (selfSlash || isVoid) {
|
|
794
|
+
bump(name, 'self');
|
|
795
|
+
parents.push(`${name}<${stack[stack.length - 1] || ''}`);
|
|
796
|
+
} else {
|
|
797
|
+
bump(name, 'open');
|
|
798
|
+
parents.push(`${name}<${stack[stack.length - 1] || ''}`);
|
|
799
|
+
stack.push(name);
|
|
800
|
+
}
|
|
801
|
+
}
|
|
802
|
+
if (stack.length > 0) wellNested = false;
|
|
803
|
+
return { counts, parents: parents.sort(), wellNested, any: counts.size > 0 };
|
|
804
|
+
}
|
|
805
|
+
|
|
806
|
+
/** Problems with one target text's tags against its source's. */
|
|
807
|
+
function compareTags(source, translated) {
|
|
808
|
+
const src = readTags(source);
|
|
809
|
+
const tgt = readTags(translated);
|
|
810
|
+
if (!src.any && !tgt.any) return [];
|
|
811
|
+
const issues = [];
|
|
812
|
+
const names = [...new Set([...src.counts.keys(), ...tgt.counts.keys()])].sort();
|
|
813
|
+
for (const name of names) {
|
|
814
|
+
const a = src.counts.get(name) || { open: 0, close: 0, self: 0 };
|
|
815
|
+
const b = tgt.counts.get(name) || { open: 0, close: 0, self: 0 };
|
|
816
|
+
if (a.open === b.open && a.close === b.close && a.self === b.self) continue;
|
|
817
|
+
if (a.open + a.close + a.self === 0) { issues.push(`<${name}> is not in the source`); continue; }
|
|
818
|
+
if (b.open + b.close + b.self === 0) { issues.push(`<${name}> is missing`); continue; }
|
|
819
|
+
if (a.open !== b.open) issues.push(`<${name}> is opened ${b.open} time(s), ${a.open} in the source`);
|
|
820
|
+
if (a.close !== b.close) issues.push(`</${name}> closes ${b.close} time(s), ${a.close} in the source`);
|
|
821
|
+
if (a.self !== b.self) issues.push(`<${name}/> appears ${b.self} time(s), ${a.self} in the source`);
|
|
822
|
+
}
|
|
823
|
+
if (issues.length === 0) {
|
|
824
|
+
if (src.wellNested && !tgt.wellNested) issues.push('tags no longer nest (a tag closes inside another it opened before)');
|
|
825
|
+
else if (src.parents.join('|') !== tgt.parents.join('|')) issues.push('tags nest differently (a tag moved into or out of another)');
|
|
826
|
+
}
|
|
827
|
+
return issues;
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
/**
|
|
831
|
+
* Check that a translation kept its source's markup: per tag name the same
|
|
832
|
+
* number of opening, closing and self-closing tags, nesting the same way.
|
|
833
|
+
* Sibling order may change (word order does); what a tag contains may not.
|
|
834
|
+
* `verify` used to compare tag NAMES as a set, so a lost `</strong>` passed
|
|
835
|
+
* — only dropping both tags was caught (Round 4, Django persona). Shared by
|
|
836
|
+
* the quality gate (lib/validate.js) and verify (lib/integrity.js).
|
|
837
|
+
*
|
|
838
|
+
* In a plural/select message each branch is compared with the source branch
|
|
839
|
+
* it translates (a Russian few/many form repeats `other`'s tags).
|
|
840
|
+
*
|
|
841
|
+
* @param {string} source
|
|
842
|
+
* @param {string} translated
|
|
843
|
+
* @returns {{ reason: string, issues: string[] }|null}
|
|
844
|
+
*/
|
|
845
|
+
function checkMarkup(source, translated) {
|
|
846
|
+
if (typeof source !== 'string' || typeof translated !== 'string') return null;
|
|
847
|
+
if (!source.includes('<') && !translated.includes('<')) return null;
|
|
848
|
+
const pairs = source.includes('{') ? pluralBranchPairs(source, translated) : [];
|
|
849
|
+
let issues;
|
|
850
|
+
if (pairs.length > 0) {
|
|
851
|
+
issues = [];
|
|
852
|
+
for (const p of pairs) {
|
|
853
|
+
for (const issue of compareTags(p.source, p.translated)) issues.push(`${issue} (plural form "${p.selector}")`);
|
|
854
|
+
}
|
|
855
|
+
} else {
|
|
856
|
+
issues = compareTags(source, translated);
|
|
857
|
+
}
|
|
858
|
+
if (issues.length === 0) return null;
|
|
859
|
+
return { reason: `markup damaged: ${[...new Set(issues)].join('; ')}`, issues: [...new Set(issues)] };
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
/**
|
|
863
|
+
* The words for plural forms a message lacks — ONE wording, and one
|
|
864
|
+
* everyday/rare rule (pluralGaps), for sync and integrity alike (Round 5,
|
|
865
|
+
* Next.js persona: sync called a missing French `many` fine while integrity
|
|
866
|
+
* warned about it).
|
|
867
|
+
*
|
|
868
|
+
* everyday → `no "few" and "many" form, which Russian uses for few (2, 3, 4), many (0, 5, 6)`
|
|
869
|
+
* rare → `no "many" form — French uses it only above 1000 or for fractions (many (1000000)); the "other" form is used there`
|
|
870
|
+
*
|
|
871
|
+
* @param {string} locale
|
|
872
|
+
* @param {string[]} cats - Missing categories
|
|
873
|
+
* @param {'cardinal'|'ordinal'} type
|
|
874
|
+
* @param {string} name - The language's display name
|
|
875
|
+
* @param {'everyday'|'rare'} kind
|
|
876
|
+
* @returns {string}
|
|
877
|
+
*/
|
|
878
|
+
function describePluralGap(locale, cats, type, name, kind) {
|
|
879
|
+
const quoted = cats.map(c => `"${c}"`);
|
|
880
|
+
const words = quoted.length <= 1 ? quoted.join('') : `${quoted.slice(0, -1).join(', ')} and ${quoted[quoted.length - 1]}`;
|
|
881
|
+
const counts = describeCategories(locale, cats, type);
|
|
882
|
+
if (kind === 'rare') {
|
|
883
|
+
return `no ${words} form — ${name} uses ${cats.length === 1 ? 'it' : 'them'} only above ${EVERYDAY_MAX} or for fractions `
|
|
884
|
+
+ `(${counts}); the "other" form is used there`;
|
|
885
|
+
}
|
|
886
|
+
return `no ${words} form, which ${name} uses for ${counts}`;
|
|
887
|
+
}
|
|
888
|
+
|
|
889
|
+
/** "one (1, 21, 31), few (2, 3, 4), many (0, 5, 6), other" — counts from Intl.PluralRules. */
|
|
890
|
+
const descriptionCache = new Map();
|
|
891
|
+
function describeCategories(locale, cats, type) {
|
|
892
|
+
// Keyed by the categories asked for too: a report about Russian few/many
|
|
893
|
+
// must not get the description cached for all four.
|
|
894
|
+
const id = `${locale}\u0000${type}\u0000${cats.join(',')}`;
|
|
895
|
+
if (!descriptionCache.has(id)) descriptionCache.set(id, computeDescription(locale, cats, type));
|
|
896
|
+
return descriptionCache.get(id);
|
|
897
|
+
}
|
|
898
|
+
|
|
899
|
+
function computeDescription(locale, cats, type) {
|
|
900
|
+
let rules;
|
|
901
|
+
try { rules = new Intl.PluralRules(String(locale).replace(/_/g, '-'), { type }); } catch { return cats.join(', '); }
|
|
902
|
+
const samples = new Map(cats.map(c => [c, []]));
|
|
903
|
+
const probe = [];
|
|
904
|
+
for (let n = 0; n <= 120; n++) probe.push(n);
|
|
905
|
+
probe.push(1000, 1000000);
|
|
906
|
+
for (const n of probe) {
|
|
907
|
+
const c = rules.select(n);
|
|
908
|
+
const list = samples.get(c);
|
|
909
|
+
if (list && list.length < 3) list.push(n);
|
|
910
|
+
}
|
|
911
|
+
return cats.map(c => (samples.get(c)?.length ? `${c} (${samples.get(c).join(', ')})` : c)).join(', ');
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
export {
|
|
915
|
+
checkMarkup,
|
|
916
|
+
pluralBranchPairs,
|
|
917
|
+
pluralBranchTexts,
|
|
918
|
+
parseMessage,
|
|
919
|
+
checkICUStructure,
|
|
920
|
+
icuGuidance,
|
|
921
|
+
pluralGaps,
|
|
922
|
+
pluralCategoryUse,
|
|
923
|
+
describeCategories,
|
|
924
|
+
describePluralGap,
|
|
925
|
+
printfConversions,
|
|
926
|
+
hasArguments,
|
|
927
|
+
hasBranching,
|
|
928
|
+
CLDR_CATEGORIES,
|
|
929
|
+
};
|