champollion 0.3.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -37
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +51 -3
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +289 -88
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +649 -130
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +16 -10
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +197 -38
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +193 -106
- package/lib/seal.mjs +6 -5
- package/lib/sealed-qualifier.mjs +2 -2
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +3 -2
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/DATA-SOVEREIGNTY.md +19 -20
- package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
- package/shared/cards-fallback.json +1 -1
- package/shared/catalogue/card-config.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/docent/faq.en.json +14 -16
- package/shared/docent/system-prompt.md +17 -19
- package/shared/explainers/tc-features.json +15 -15
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/human-services.json +1 -1
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +20 -10
- package/shared/schemas/human-services.schema.json +2 -2
- package/shared/schemas/language-card.schema.json +1 -1
- package/shared/schemas/method-card.schema.json +1 -1
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11333
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* placeholders.js — the placeholders a value carries, and what a translation
|
|
3
|
+
* did to them. ONE rule for the sync quality gate (lib/validate.js) and for
|
|
4
|
+
* verify (lib/verify.js, through lib/integrity.js): sync refuses what verify
|
|
5
|
+
* would flag, so the fallback runs instead of the damage being written and
|
|
6
|
+
* then reported (Round 12, i18next: the gate did not check {{…}}, so sync
|
|
7
|
+
* wrote {{nom}} for {{name}}, verify flagged it, and the run exited 2).
|
|
8
|
+
*
|
|
9
|
+
* A leaf module (it imports only lib/icu-structure.js), so the gate and the
|
|
10
|
+
* integrity audit can both use it without importing each other.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { parseMessage, hasArguments } from './icu-structure.js';
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* i18next interpolation — `{{name}}`, `{{ user.name }}`, `{{- html}}`
|
|
17
|
+
* (unescaped), `{{val, number}}` (formatted). A leading dot is Go template
|
|
18
|
+
* syntax (Hugo "{{ .Count }}") and is not matched.
|
|
19
|
+
*/
|
|
20
|
+
const I18NEXT_INTERPOLATION = /\{\{\s*(-\s*)?([A-Za-z_$][\w.$]*)\s*(?:,\s*([^{}]*?))?\s*\}\}/g;
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Extract ICU-style placeholders from a string.
|
|
24
|
+
*
|
|
25
|
+
* Handles:
|
|
26
|
+
* - Simple: {name}, {count} → "name", "count"
|
|
27
|
+
* - Nested ICU: {count, plural, one {# item} other {# items}}
|
|
28
|
+
* - React-intl: <bold>text</bold> → "<bold>"
|
|
29
|
+
* - i18next: {{name}}, {{val, number}} → "{{name}}", "{{val, number}}"
|
|
30
|
+
*
|
|
31
|
+
* The token's form names its syntax (placeholderSyntax). An i18next token
|
|
32
|
+
* keeps its double braces: read as "name" it was indistinguishable from
|
|
33
|
+
* "{name}", which i18next prints literally — a `{{name}}` → `{name}` loss
|
|
34
|
+
* passed, and a `{{name}}` loss could not be reported as i18next (Round 12).
|
|
35
|
+
*
|
|
36
|
+
* @param {string} text - Translation string
|
|
37
|
+
* @returns {string[]} Sorted array of placeholder tokens
|
|
38
|
+
*/
|
|
39
|
+
function extractPlaceholders(text) {
|
|
40
|
+
if (typeof text !== 'string') return [];
|
|
41
|
+
|
|
42
|
+
const placeholders = new Set();
|
|
43
|
+
|
|
44
|
+
// i18next {{…}} first, then blanked out, so the single-brace scan below
|
|
45
|
+
// does not read the inner "{name}" of "{{name}}" as an ICU argument.
|
|
46
|
+
const rest = text.replace(I18NEXT_INTERPOLATION, (whole, unescaped, name, format) => {
|
|
47
|
+
placeholders.add(`{{${unescaped ? '- ' : ''}${name}${format ? `, ${format.trim().replace(/\s+/g, ' ')}` : ''}}}`);
|
|
48
|
+
return ' ';
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
// Simple ICU placeholders: {name}, {count}
|
|
52
|
+
// Match top-level braces only (not nested plurals)
|
|
53
|
+
const simplePattern = /\{(\w+)(?:[,}])/g;
|
|
54
|
+
let match;
|
|
55
|
+
while ((match = simplePattern.exec(rest)) !== null) {
|
|
56
|
+
placeholders.add(match[1]);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// React-intl XML tags: <bold>, </bold>, <link>, </link>
|
|
60
|
+
const xmlPattern = /<\/?(\w+)>/g;
|
|
61
|
+
while ((match = xmlPattern.exec(text)) !== null) {
|
|
62
|
+
placeholders.add(`<${match[1]}>`);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
return [...placeholders].sort();
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* The syntax a token from extractPlaceholders is written in — read off the
|
|
70
|
+
* token's own form, which is how extractPlaceholders tells them apart:
|
|
71
|
+
* 'i18next' "{{name}}" (i18next interpolation)
|
|
72
|
+
* 'markup' "<bold>" (a tag: react-intl, react-i18next, HTML)
|
|
73
|
+
* 'brace' "name" (a single-brace "{name}" — an ICU/MessageFormat
|
|
74
|
+
* argument when the message parses as ICU, which
|
|
75
|
+
* lib/icu-structure.js checkICUStructure judges)
|
|
76
|
+
* printf conversions (%s, %(name)s) are not tokens here: checkICUStructure
|
|
77
|
+
* compares them and tags its findings 'printf'.
|
|
78
|
+
*
|
|
79
|
+
* @param {string} token
|
|
80
|
+
* @returns {'i18next'|'markup'|'brace'}
|
|
81
|
+
*/
|
|
82
|
+
function placeholderSyntax(token) {
|
|
83
|
+
if (token.startsWith('{{')) return 'i18next';
|
|
84
|
+
if (token.startsWith('<')) return 'markup';
|
|
85
|
+
return 'brace';
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* A token as the user wrote it: "name" was "{name}".
|
|
90
|
+
*
|
|
91
|
+
* @param {string} token
|
|
92
|
+
* @returns {string}
|
|
93
|
+
*/
|
|
94
|
+
function showPlaceholder(token) {
|
|
95
|
+
return placeholderSyntax(token) === 'brace' ? `{${token}}` : token;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Compare placeholders between source and target strings.
|
|
100
|
+
*
|
|
101
|
+
* @param {string} sourceValue - Source locale value
|
|
102
|
+
* @param {string} targetValue - Target locale value
|
|
103
|
+
* @returns {{ missing: string[], extra: string[] }} Placeholder differences
|
|
104
|
+
*/
|
|
105
|
+
function comparePlaceholders(sourceValue, targetValue) {
|
|
106
|
+
const sourcePH = extractPlaceholders(sourceValue);
|
|
107
|
+
const targetPH = extractPlaceholders(targetValue);
|
|
108
|
+
|
|
109
|
+
const missing = sourcePH.filter(p => !targetPH.includes(p));
|
|
110
|
+
const extra = targetPH.filter(p => !sourcePH.includes(p));
|
|
111
|
+
|
|
112
|
+
return { missing, extra };
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Does `text` parse as an ICU message with arguments (either apostrophe
|
|
117
|
+
* reading — checkICUStructure's own test)? Its single-brace tokens are then
|
|
118
|
+
* the ICU check's to judge, argument by argument.
|
|
119
|
+
*
|
|
120
|
+
* @param {unknown} text
|
|
121
|
+
* @returns {boolean}
|
|
122
|
+
*/
|
|
123
|
+
function parsesAsICU(text) {
|
|
124
|
+
if (typeof text !== 'string' || !text.includes('{')) return false;
|
|
125
|
+
return ['icu', 'literal'].some((apostrophes) => {
|
|
126
|
+
const parsed = parseMessage(text, { apostrophes });
|
|
127
|
+
return parsed.ok && hasArguments(parsed.nodes);
|
|
128
|
+
});
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** What each syntax is called in a finding: "i18next {{…}} placeholder …". */
|
|
132
|
+
const PLACEHOLDER_SYNTAX_NAMES = Object.freeze({
|
|
133
|
+
i18next: 'i18next {{…}}',
|
|
134
|
+
brace: '{…}',
|
|
135
|
+
markup: 'tag',
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* What a translation did to the placeholders its source carries, each change
|
|
140
|
+
* named by its syntax (placeholderSyntax) — the findings verify reports, and
|
|
141
|
+
* the ones the sync gate refuses:
|
|
142
|
+
* { syntax: 'i18next', token: '{{name}}', issue: 'placeholder {{name}} was changed to {{nom}}' }
|
|
143
|
+
*
|
|
144
|
+
* A missing token and an extra one of the same name read as one change
|
|
145
|
+
* ("{{name}} was changed to {name}") and are named by the SOURCE's syntax;
|
|
146
|
+
* the remaining missing/extra tokens pair up in order. Left out:
|
|
147
|
+
* - a brace token (single or double) when the source parses as an ICU
|
|
148
|
+
* message — the ICU check compares its arguments, and the token scan
|
|
149
|
+
* reads a one-word plural branch ("other {articles}") as a placeholder
|
|
150
|
+
* and a branch holding only an argument ("other{{count}}") as i18next;
|
|
151
|
+
* - a tag when the markup check already reported the value
|
|
152
|
+
* (`markupReported`), said precisely there.
|
|
153
|
+
* printf conversions are not tokens here: checkICUStructure compares them.
|
|
154
|
+
*
|
|
155
|
+
* @param {string} sourceValue
|
|
156
|
+
* @param {string} targetValue
|
|
157
|
+
* @param {{ markupReported?: boolean }} [options]
|
|
158
|
+
* @returns {Array<{ syntax: 'i18next'|'brace'|'markup', token: string, issue: string }>}
|
|
159
|
+
*/
|
|
160
|
+
function placeholderChanges(sourceValue, targetValue, { markupReported = false } = {}) {
|
|
161
|
+
if (typeof sourceValue !== 'string' || typeof targetValue !== 'string') return [];
|
|
162
|
+
const { missing, extra } = comparePlaceholders(sourceValue, targetValue);
|
|
163
|
+
if (missing.length === 0 && extra.length === 0) return [];
|
|
164
|
+
const nameOf = (token) => (/[A-Za-z_$][\w.$]*/.exec(token) || [token])[0];
|
|
165
|
+
const changes = [];
|
|
166
|
+
for (const m of [...missing]) {
|
|
167
|
+
const e = extra.find(x => nameOf(x) === nameOf(m));
|
|
168
|
+
if (!e) continue;
|
|
169
|
+
changes.push([m, e]);
|
|
170
|
+
missing.splice(missing.indexOf(m), 1);
|
|
171
|
+
extra.splice(extra.indexOf(e), 1);
|
|
172
|
+
}
|
|
173
|
+
while (missing.length > 0 && extra.length > 0) changes.push([missing.shift(), extra.shift()]);
|
|
174
|
+
const items = [
|
|
175
|
+
...changes.map(([m, e]) => [m, `placeholder ${showPlaceholder(m)} was changed to ${showPlaceholder(e)}`]),
|
|
176
|
+
...missing.map(m => [m, `placeholder ${showPlaceholder(m)} is missing`]),
|
|
177
|
+
...extra.map(e => [e, `placeholder ${showPlaceholder(e)} is not in the source`]),
|
|
178
|
+
];
|
|
179
|
+
const icuSource = parsesAsICU(sourceValue);
|
|
180
|
+
const out = [];
|
|
181
|
+
for (const [token, issue] of items) {
|
|
182
|
+
const syntax = placeholderSyntax(token);
|
|
183
|
+
// In an ICU message, braces are ICU's: "other{{count}}" is a branch
|
|
184
|
+
// holding the argument, not i18next (an i18next "{{name}}" message never
|
|
185
|
+
// parses as ICU).
|
|
186
|
+
if ((syntax === 'brace' || syntax === 'i18next') && icuSource) continue;
|
|
187
|
+
if (syntax === 'markup' && markupReported) continue;
|
|
188
|
+
out.push({ syntax, token, issue });
|
|
189
|
+
}
|
|
190
|
+
return out;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// ── A sentence break inserted beside a placeholder ──────────────────────────
|
|
194
|
+
//
|
|
195
|
+
// "Take this medicine at {time}." came back as "… sina. {time}." — a sentence
|
|
196
|
+
// end before the placeholder, so the app shows the time as a sentence of its
|
|
197
|
+
// own — and the gate and verify both passed it: every placeholder was there
|
|
198
|
+
// (Round 14, hospital persona; a medical string). The rule below is
|
|
199
|
+
// structural and narrow: it does not judge fluency, only a sentence boundary
|
|
200
|
+
// the TRANSLATION puts right beside a placeholder where the SOURCE has none.
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Sentence-final marks, across scripts: Latin/Cyrillic/Greek . ! ? (Greek's
|
|
204
|
+
* question mark ";" is left out: it is also the semicolon), CJK 。!? and the
|
|
205
|
+
* halfwidth/fullwidth forms, Devanagari/Bengali danda । ॥, Arabic ؟ and the
|
|
206
|
+
* Urdu full stop ۔, Ethiopic ። ፧, Armenian ։, Myanmar ။, Khmer ។ ៕,
|
|
207
|
+
* Canadian Syllabics ᙮ (Cree, Inuktitut), and the doubled marks ‼ ⁇ ⁈ ⁉ ‽.
|
|
208
|
+
* Script-agnostic: a mark from any script is a boundary in any text.
|
|
209
|
+
*/
|
|
210
|
+
const SENTENCE_MARKS = new Set([...'.!?。!?。.।॥؟۔።፧։။។៕᙮‼⁇⁈⁉‽']);
|
|
211
|
+
|
|
212
|
+
/** Closing quotes and brackets that may stand between a mark and what follows it. */
|
|
213
|
+
const CLOSERS = new Set([...'"\'”’»›」』))]]']);
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Clause punctuation: where the SOURCE already sets a placeholder apart
|
|
217
|
+
* ("… corpora; {spec}.", "Error: {message}", "— {noncomp}, and …"), a
|
|
218
|
+
* translation may make that clause a sentence of its own.
|
|
219
|
+
*/
|
|
220
|
+
const CLAUSE_MARKS = new Set([...';:,—–·|(([[、,;:']);
|
|
221
|
+
|
|
222
|
+
/** Whitespace (JavaScript's \s includes no-break and narrow no-break space). */
|
|
223
|
+
const SPACE = /\s/;
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Placeholder spans with their positions: i18next / Hugo / Handlebars
|
|
227
|
+
* `{{…}}`, a single-brace argument `{name}` / `{0}` / `{n, number}` (also
|
|
228
|
+
* Ruby's `%{name}`), and printf `%(name)s`, `%1$s`, `%s`, `%d`, `%@`.
|
|
229
|
+
* Nested ICU (plural/select) is not matched as one span — messages holding
|
|
230
|
+
* it are left out of this check (see placeholderSentenceBreaks).
|
|
231
|
+
*/
|
|
232
|
+
const PLACEHOLDER_SPAN = /\{\{[^{}]*\}\}|%?\{\s*(?:[A-Za-z_$][\w.$]*|\d+)\s*(?:,[^{}]*)?\}|%\(\w+\)[a-zA-Z]|%\d+\$[a-zA-Z@]|%[sd@]/g;
|
|
233
|
+
|
|
234
|
+
/** An ICU plural/select/selectordinal argument — its branches are not placeholder spans. */
|
|
235
|
+
const ICU_BRANCHING = /\{\s*[\w.$]+\s*,\s*(?:plural|select|selectordinal)\s*,/;
|
|
236
|
+
|
|
237
|
+
function placeholderSpans(text) {
|
|
238
|
+
const spans = [];
|
|
239
|
+
for (const m of text.matchAll(PLACEHOLDER_SPAN)) {
|
|
240
|
+
spans.push({ token: m[0].replace(/\s+/g, ''), shown: m[0], start: m.index, end: m.index + m[0].length });
|
|
241
|
+
}
|
|
242
|
+
return spans;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* Is text[i] a sentence boundary? Any mark in SENTENCE_MARKS but a full stop
|
|
247
|
+
* is one. A full stop "." is one unless it is:
|
|
248
|
+
* - part of an ellipsis ("..." — a pause, not an end),
|
|
249
|
+
* - followed directly by a letter, digit or "_" ("3.5", "{host}.com", "e.g"),
|
|
250
|
+
* - the stop of a one-letter word — an initial or an abbreviation
|
|
251
|
+
* ("M. {name}", "p. {page}", "z. B. {x}"),
|
|
252
|
+
* - the stop of a short capitalised word (two or three letters) — an
|
|
253
|
+
* abbreviation before a number or name ("Nr. {id}", "Dr. {name}",
|
|
254
|
+
* "Tel. {phone}", "Ca. {count}").
|
|
255
|
+
* A stop right after a placeholder ("{time}.", "%(dose)s.") is never read as
|
|
256
|
+
* an abbreviation: `afterPlaceholder` says the stop follows one.
|
|
257
|
+
*
|
|
258
|
+
* @param {string} text
|
|
259
|
+
* @param {number} i
|
|
260
|
+
* @param {boolean} [afterPlaceholder]
|
|
261
|
+
* @returns {boolean}
|
|
262
|
+
*/
|
|
263
|
+
function isSentenceBoundaryAt(text, i, afterPlaceholder = false) {
|
|
264
|
+
const ch = text[i];
|
|
265
|
+
if (!SENTENCE_MARKS.has(ch)) return false;
|
|
266
|
+
if (ch !== '.') return true;
|
|
267
|
+
if (text[i - 1] === '.' || text[i + 1] === '.') return false;
|
|
268
|
+
const next = text[i + 1];
|
|
269
|
+
if (next !== undefined && /[\p{L}\p{N}_]/u.test(next)) return false;
|
|
270
|
+
if (afterPlaceholder) return true;
|
|
271
|
+
const word = /(\p{L}+)$/u.exec(text.slice(0, i))?.[1] || '';
|
|
272
|
+
if (word.length === 1) return false;
|
|
273
|
+
if (word.length <= 3 && /^\p{Lu}/u.test(word)) return false;
|
|
274
|
+
return true;
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* What stands on one side of a placeholder span, skipping spaces (and, on
|
|
279
|
+
* the left, closing quotes/brackets): the message's edge, a sentence
|
|
280
|
+
* boundary, clause punctuation, or a word (anything else, another
|
|
281
|
+
* placeholder included).
|
|
282
|
+
*
|
|
283
|
+
* @returns {{ kind: 'edge'|'boundary'|'clause'|'word', at: number }}
|
|
284
|
+
*/
|
|
285
|
+
function sideOf(text, span, side, placeholderEnds = new Set()) {
|
|
286
|
+
if (side === 'left') {
|
|
287
|
+
let j = span.start - 1;
|
|
288
|
+
while (j >= 0 && (SPACE.test(text[j]) || CLOSERS.has(text[j]))) j--;
|
|
289
|
+
if (j < 0) return { kind: 'edge', at: -1 };
|
|
290
|
+
if (isSentenceBoundaryAt(text, j, placeholderEnds.has(j))) return { kind: 'boundary', at: j };
|
|
291
|
+
if (SENTENCE_MARKS.has(text[j]) || CLAUSE_MARKS.has(text[j])) return { kind: 'clause', at: j };
|
|
292
|
+
return { kind: 'word', at: j };
|
|
293
|
+
}
|
|
294
|
+
let k = span.end;
|
|
295
|
+
while (k < text.length && SPACE.test(text[k])) k++;
|
|
296
|
+
if (k >= text.length) return { kind: 'edge', at: k };
|
|
297
|
+
if (isSentenceBoundaryAt(text, k, k === span.end || /^\s*$/.test(text.slice(span.end, k)))) return { kind: 'boundary', at: k };
|
|
298
|
+
if (SENTENCE_MARKS.has(text[k]) || CLAUSE_MARKS.has(text[k]) || CLOSERS.has(text[k])) return { kind: 'clause', at: k };
|
|
299
|
+
return { kind: 'word', at: k };
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* The sentence breaks a translation INSERTED beside a placeholder, leaving
|
|
304
|
+
* it standing as a sentence of its own — the one rule the sync gate refuses
|
|
305
|
+
* by (lib/validate.js) and verify flags by (lib/verify.js):
|
|
306
|
+
* { token: '{time}', side: 'before', issue: 'a sentence break was inserted before {time} ("medisina. {time}.") — {time} now stands as a sentence of its own; in the source it is part of one' }
|
|
307
|
+
*
|
|
308
|
+
* A finding needs:
|
|
309
|
+
* 1. a placeholder the translation shares with the source, with a
|
|
310
|
+
* sentence boundary right BEFORE or right AFTER it in the translation
|
|
311
|
+
* ("medisina. {time}", "{time}. Kisik") — text before it, so a mark
|
|
312
|
+
* that opens nothing is not one;
|
|
313
|
+
* 2. nothing of its sentence around it: on each side only a boundary or
|
|
314
|
+
* the message's edge (". {time}." — the time shown as a sentence);
|
|
315
|
+
* 3. in the source, that placeholder inside a sentence: a word on one
|
|
316
|
+
* side, not a boundary, clause punctuation or the edge
|
|
317
|
+
* ("Take this medicine at {time}.").
|
|
318
|
+
* Narrow on purpose (Round 14 measurement, lib/placeholders.js tests): a
|
|
319
|
+
* placeholder that only starts or ends a sentence in the translation is
|
|
320
|
+
* word order, not damage — "Open an issue on {github} or …" is rightly
|
|
321
|
+
* "{github}에 이슈를 열거나 …" in Korean, and "Shipped by {carrier} on
|
|
322
|
+
* {date}." rightly "Expédié le {date} par {carrier}." Left out: ICU
|
|
323
|
+
* plural/select messages (the ICU check judges their branches).
|
|
324
|
+
*
|
|
325
|
+
* @param {string} sourceValue
|
|
326
|
+
* @param {string} targetValue
|
|
327
|
+
* @returns {Array<{ token: string, side: 'before'|'after', issue: string }>}
|
|
328
|
+
*/
|
|
329
|
+
function placeholderSentenceBreaks(sourceValue, targetValue) {
|
|
330
|
+
if (typeof sourceValue !== 'string' || typeof targetValue !== 'string') return [];
|
|
331
|
+
if (sourceValue === targetValue) return [];
|
|
332
|
+
if (ICU_BRANCHING.test(sourceValue) || ICU_BRANCHING.test(targetValue)) return [];
|
|
333
|
+
const targetSpans = placeholderSpans(targetValue);
|
|
334
|
+
if (targetSpans.length === 0) return [];
|
|
335
|
+
const sourceSpans = placeholderSpans(sourceValue);
|
|
336
|
+
|
|
337
|
+
// In the source the placeholder is part of a sentence: a word beside it.
|
|
338
|
+
const inSentence = (token) => sourceSpans.some(s => s.token === token
|
|
339
|
+
&& (sideOf(sourceValue, s, 'left').kind === 'word' || sideOf(sourceValue, s, 'right').kind === 'word'));
|
|
340
|
+
|
|
341
|
+
const ends = new Set(targetSpans.map(t => t.end));
|
|
342
|
+
const found = [];
|
|
343
|
+
const seen = new Set();
|
|
344
|
+
for (const s of targetSpans) {
|
|
345
|
+
if (seen.has(s.token) || !inSentence(s.token)) continue; // a missing/extra placeholder is placeholderChanges' finding
|
|
346
|
+
const left = sideOf(targetValue, s, 'left', ends);
|
|
347
|
+
const right = sideOf(targetValue, s, 'right');
|
|
348
|
+
const alone = (left.kind === 'boundary' || left.kind === 'edge') && (right.kind === 'boundary' || right.kind === 'edge');
|
|
349
|
+
if (!alone || (left.kind !== 'boundary' && right.kind !== 'boundary')) continue;
|
|
350
|
+
// A boundary before it must close some text; one after it at the very
|
|
351
|
+
// end of the message, with nothing before, is the message's own end.
|
|
352
|
+
if (left.kind === 'boundary' && !/[\p{L}\p{N}]/u.test(targetValue.slice(0, left.at))) continue;
|
|
353
|
+
seen.add(s.token);
|
|
354
|
+
const side = left.kind === 'boundary' ? 'before' : 'after';
|
|
355
|
+
// The word before the mark, the placeholder, the mark after it: "medisina. {time}."
|
|
356
|
+
const from = left.kind === 'boundary' ? Math.max(0, targetValue.slice(0, left.at).search(/\S+\s*$/)) : s.start;
|
|
357
|
+
const to = right.kind === 'boundary' ? right.at + 1 : s.end;
|
|
358
|
+
const shown = targetValue.slice(from, to).trim();
|
|
359
|
+
found.push({
|
|
360
|
+
token: s.token,
|
|
361
|
+
side,
|
|
362
|
+
issue: `a sentence break was inserted ${side} ${s.shown} ("${shown}") — ${s.shown} now stands as a sentence of its own; in the source it is part of one`,
|
|
363
|
+
});
|
|
364
|
+
}
|
|
365
|
+
return found;
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
/** Sentence boundaries in a text: runs of marks ("?!", "。") count once. */
|
|
369
|
+
function countSentenceBoundaries(text) {
|
|
370
|
+
const ends = new Set(placeholderSpans(text).map(s => s.end));
|
|
371
|
+
let n = 0;
|
|
372
|
+
for (let i = 0; i < text.length; i++) {
|
|
373
|
+
if (!SENTENCE_MARKS.has(text[i])) continue;
|
|
374
|
+
let j = i;
|
|
375
|
+
while (j + 1 < text.length && SENTENCE_MARKS.has(text[j + 1])) j++;
|
|
376
|
+
for (let k = i; k <= j; k++) {
|
|
377
|
+
if (isSentenceBoundaryAt(text, k, ends.has(k))) { n++; break; }
|
|
378
|
+
}
|
|
379
|
+
i = j;
|
|
380
|
+
}
|
|
381
|
+
return n;
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
export {
|
|
385
|
+
extractPlaceholders,
|
|
386
|
+
comparePlaceholders,
|
|
387
|
+
placeholderSyntax,
|
|
388
|
+
showPlaceholder,
|
|
389
|
+
parsesAsICU,
|
|
390
|
+
placeholderChanges,
|
|
391
|
+
placeholderSentenceBreaks,
|
|
392
|
+
countSentenceBoundaries,
|
|
393
|
+
PLACEHOLDER_SYNTAX_NAMES,
|
|
394
|
+
};
|
package/lib/plugins.js
CHANGED
|
@@ -148,6 +148,9 @@ function validateManifest(manifest) {
|
|
|
148
148
|
if (manifest.type === 'api' && !manifest.endpoint) {
|
|
149
149
|
errors.push('API plugins must include an "endpoint" URL');
|
|
150
150
|
}
|
|
151
|
+
if (manifest.acceptsInstructions !== undefined && typeof manifest.acceptsInstructions !== 'boolean') {
|
|
152
|
+
errors.push('"acceptsInstructions" must be true or false');
|
|
153
|
+
}
|
|
151
154
|
|
|
152
155
|
// Benchmarks: validate structure if present
|
|
153
156
|
if (manifest.benchmarks) {
|
|
@@ -375,6 +378,11 @@ function resolvePluginForPair(plugins, pairConfig) {
|
|
|
375
378
|
|
|
376
379
|
// Plugin-specific fields
|
|
377
380
|
endpoint: plugin.endpoint || pairConfig.endpoint || null,
|
|
381
|
+
// The endpoint's declared instruction capability: the pair's own
|
|
382
|
+
// statement wins, then the manifest's; null = unknown.
|
|
383
|
+
acceptsInstructions: typeof pairConfig.acceptsInstructions === 'boolean'
|
|
384
|
+
? pairConfig.acceptsInstructions
|
|
385
|
+
: (typeof plugin.acceptsInstructions === 'boolean' ? plugin.acceptsInstructions : null),
|
|
378
386
|
pluginName: plugin.name,
|
|
379
387
|
pluginVersion: plugin.version,
|
|
380
388
|
pluginDir: plugin._pluginDir,
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* plural-gap-redo.js — a plural message left incomplete is asked again by a
|
|
3
|
+
* setup that has not tried it yet.
|
|
4
|
+
*
|
|
5
|
+
* THE FINDING (Round 13, Django persona): a local model left a Russian
|
|
6
|
+
* plural entry without its few/many forms. Sync wrote the entry with the
|
|
7
|
+
* "other" form standing in and marked it (`# champollion:` in a gettext
|
|
8
|
+
* catalog; a missing ICU branch in JSON), and every later sync treated it as
|
|
9
|
+
* done — including CI's, which runs a stronger hosted model. Every sync then
|
|
10
|
+
* exited 2 until a person re-ran with a key or wrote the forms by hand, and
|
|
11
|
+
* the CI workflow had no way to do either.
|
|
12
|
+
*
|
|
13
|
+
* THE RULE. A marked gap is not a translation. A sync asks for it again —
|
|
14
|
+
* from the model, not the cache, which holds the incomplete answer — when it
|
|
15
|
+
* runs a setup (method, model, register, coaching: the cache key) that has
|
|
16
|
+
* not answered this message's current text yet:
|
|
17
|
+
* - the setup that wrote it is the one the lock records for the key
|
|
18
|
+
* (`by`), while the record still matches the value on disk (a person who
|
|
19
|
+
* edited the entry is never overruled);
|
|
20
|
+
* - the setups that answered it without the forms before are kept in the
|
|
21
|
+
* lock (`gaps`: source hash + method keys), so a local run and a CI run
|
|
22
|
+
* do not take turns paying for the same incomplete answer;
|
|
23
|
+
* - the pair's own fallback is part of the setup: a gap it left is not
|
|
24
|
+
* re-asked by the same configuration.
|
|
25
|
+
* `--redo gaps` asks for every gap on disk, whoever left it (an explicit
|
|
26
|
+
* redo). A setup that also leaves the forms out leaves the gap marked, as
|
|
27
|
+
* before; the run exits 2.
|
|
28
|
+
*
|
|
29
|
+
* Shared by the sync and its cost estimate, so the estimate prices exactly
|
|
30
|
+
* what is asked again.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
import { pluralGapsInFile } from './verify.js';
|
|
34
|
+
import { tmMethodKey } from './tm.js';
|
|
35
|
+
import { decodeWritten, shortSourceHash, valueHash } from './locale-state.js';
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* The setups a gap record says answered this text without the forms.
|
|
39
|
+
*
|
|
40
|
+
* @param {object|undefined} record - LockState.of(code).gaps[lockKey]
|
|
41
|
+
* @param {string} sourceValue
|
|
42
|
+
* @returns {string[]}
|
|
43
|
+
*/
|
|
44
|
+
export function gapTriedBy(record, sourceValue) {
|
|
45
|
+
if (!record || typeof record !== 'object' || typeof sourceValue !== 'string') return [];
|
|
46
|
+
if (record.source !== shortSourceHash(sourceValue) || !Array.isArray(record.methods)) return [];
|
|
47
|
+
return record.methods;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Remember that `methodKey` answered this key's current text without a form
|
|
52
|
+
* the language uses for ordinary counts.
|
|
53
|
+
*
|
|
54
|
+
* @param {object} localeState - LockState.of(code)
|
|
55
|
+
* @param {string} lockKey
|
|
56
|
+
* @param {string} sourceValue
|
|
57
|
+
* @param {string} methodKey
|
|
58
|
+
*/
|
|
59
|
+
export function recordGap(localeState, lockKey, sourceValue, methodKey) {
|
|
60
|
+
if (typeof sourceValue !== 'string' || !methodKey) return;
|
|
61
|
+
localeState.gaps = localeState.gaps || {};
|
|
62
|
+
const known = gapTriedBy(localeState.gaps[lockKey], sourceValue);
|
|
63
|
+
localeState.gaps[lockKey] = { source: shortSourceHash(sourceValue), methods: [...new Set([...known, methodKey])] };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Plural messages of one target file that this run asks for again.
|
|
68
|
+
*
|
|
69
|
+
* @param {object} p
|
|
70
|
+
* @param {object} p.file - layout file ({ path, format, rel })
|
|
71
|
+
* @param {object} p.expected - The target's expected map (source text per target key)
|
|
72
|
+
* @param {object} p.targetFlat - Values on disk
|
|
73
|
+
* @param {string} p.locale
|
|
74
|
+
* @param {object} p.localeState - LockState.peek(code)
|
|
75
|
+
* @param {(key: string) => string} p.lockKeyOf
|
|
76
|
+
* @param {object} p.pairConfig
|
|
77
|
+
* @param {boolean} [p.all] - `--redo gaps`: every gap on disk
|
|
78
|
+
* @returns {{ keys: string[], gaps: Array<{ key: string, missing: string[], marked: boolean }>,
|
|
79
|
+
* by: Object<string, string|null>, missing: Object<string, string[]> }}
|
|
80
|
+
* keys — target keys to ask again; gaps — every gap in the file; by — the
|
|
81
|
+
* setup that left each asked key (null: no record); missing — its forms
|
|
82
|
+
*/
|
|
83
|
+
export function planGapRedo({ file, expected, targetFlat, locale, localeState, lockKeyOf, pairConfig, all = false }) {
|
|
84
|
+
const out = { keys: [], gaps: [], by: {}, missing: {} };
|
|
85
|
+
if (!file || !targetFlat) return out;
|
|
86
|
+
try { out.gaps = pluralGapsInFile({ file, expected, targetFlat, locale }); } catch { out.gaps = []; }
|
|
87
|
+
if (out.gaps.length === 0) return out;
|
|
88
|
+
const current = tmMethodKey(pairConfig);
|
|
89
|
+
const fallback = pairConfig.fallback ? tmMethodKey(pairConfig.fallback) : null;
|
|
90
|
+
for (const g of out.gaps) {
|
|
91
|
+
const src = expected[g.key];
|
|
92
|
+
const value = targetFlat[g.key];
|
|
93
|
+
if (typeof src !== 'string' || typeof value !== 'string') continue;
|
|
94
|
+
const lk = lockKeyOf(g.key);
|
|
95
|
+
// Who wrote what is on disk — only while the record still describes it.
|
|
96
|
+
const record = decodeWritten(localeState?.written?.[lk]);
|
|
97
|
+
const writer = record && record.source === shortSourceHash(src) && record.value === valueHash(value)
|
|
98
|
+
? (localeState?.by?.[lk] || null) : null;
|
|
99
|
+
if (!all) {
|
|
100
|
+
if (!writer) continue; // unknown, or a person's edit: never automatic
|
|
101
|
+
if (writer === current || writer === fallback) continue; // this setup left it
|
|
102
|
+
if (gapTriedBy(localeState?.gaps?.[lk], src).includes(current)) continue; // tried already
|
|
103
|
+
}
|
|
104
|
+
out.keys.push(g.key);
|
|
105
|
+
out.by[g.key] = writer;
|
|
106
|
+
out.missing[g.key] = g.missing;
|
|
107
|
+
}
|
|
108
|
+
return out;
|
|
109
|
+
}
|