champollion 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/README.md +41 -26
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +34 -0
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +286 -85
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +632 -125
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +15 -9
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +194 -35
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +6 -1
  100. package/lib/seal.mjs +4 -3
  101. package/lib/sealed-qualifier.mjs +1 -1
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +1 -1
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/cards-fallback.json +1 -1
  123. package/shared/curated-orthography-conventions.json +26 -8
  124. package/shared/gettext-plural-forms.json +45 -0
  125. package/shared/method-registry.json +2 -0
  126. package/shared/metric-registry.json +96 -18
  127. package/shared/schemas/champollion-plugin.schema.json +4 -0
  128. package/shared/schemas/corpora-card.schema.json +8 -2
  129. package/shared/schemas/method-index-record.schema.json +67 -0
  130. package/shared/schemas/method-registry.schema.json +4 -0
  131. package/shared/schemas/metric-registry.schema.json +55 -1
  132. package/shared/docent/corpus.json +0 -11739
@@ -0,0 +1,394 @@
1
+ /**
2
+ * placeholders.js — the placeholders a value carries, and what a translation
3
+ * did to them. ONE rule for the sync quality gate (lib/validate.js) and for
4
+ * verify (lib/verify.js, through lib/integrity.js): sync refuses what verify
5
+ * would flag, so the fallback runs instead of the damage being written and
6
+ * then reported (Round 12, i18next: the gate did not check {{…}}, so sync
7
+ * wrote {{nom}} for {{name}}, verify flagged it, and the run exited 2).
8
+ *
9
+ * A leaf module (it imports only lib/icu-structure.js), so the gate and the
10
+ * integrity audit can both use it without importing each other.
11
+ */
12
+
13
+ import { parseMessage, hasArguments } from './icu-structure.js';
14
+
15
+ /**
16
+ * i18next interpolation — `{{name}}`, `{{ user.name }}`, `{{- html}}`
17
+ * (unescaped), `{{val, number}}` (formatted). A leading dot is Go template
18
+ * syntax (Hugo "{{ .Count }}") and is not matched.
19
+ */
20
+ const I18NEXT_INTERPOLATION = /\{\{\s*(-\s*)?([A-Za-z_$][\w.$]*)\s*(?:,\s*([^{}]*?))?\s*\}\}/g;
21
+
22
+ /**
23
+ * Extract ICU-style placeholders from a string.
24
+ *
25
+ * Handles:
26
+ * - Simple: {name}, {count} → "name", "count"
27
+ * - Nested ICU: {count, plural, one {# item} other {# items}}
28
+ * - React-intl: <bold>text</bold> → "<bold>"
29
+ * - i18next: {{name}}, {{val, number}} → "{{name}}", "{{val, number}}"
30
+ *
31
+ * The token's form names its syntax (placeholderSyntax). An i18next token
32
+ * keeps its double braces: read as "name" it was indistinguishable from
33
+ * "{name}", which i18next prints literally — a `{{name}}` → `{name}` loss
34
+ * passed, and a `{{name}}` loss could not be reported as i18next (Round 12).
35
+ *
36
+ * @param {string} text - Translation string
37
+ * @returns {string[]} Sorted array of placeholder tokens
38
+ */
39
+ function extractPlaceholders(text) {
40
+ if (typeof text !== 'string') return [];
41
+
42
+ const placeholders = new Set();
43
+
44
+ // i18next {{…}} first, then blanked out, so the single-brace scan below
45
+ // does not read the inner "{name}" of "{{name}}" as an ICU argument.
46
+ const rest = text.replace(I18NEXT_INTERPOLATION, (whole, unescaped, name, format) => {
47
+ placeholders.add(`{{${unescaped ? '- ' : ''}${name}${format ? `, ${format.trim().replace(/\s+/g, ' ')}` : ''}}}`);
48
+ return ' ';
49
+ });
50
+
51
+ // Simple ICU placeholders: {name}, {count}
52
+ // Match top-level braces only (not nested plurals)
53
+ const simplePattern = /\{(\w+)(?:[,}])/g;
54
+ let match;
55
+ while ((match = simplePattern.exec(rest)) !== null) {
56
+ placeholders.add(match[1]);
57
+ }
58
+
59
+ // React-intl XML tags: <bold>, </bold>, <link>, </link>
60
+ const xmlPattern = /<\/?(\w+)>/g;
61
+ while ((match = xmlPattern.exec(text)) !== null) {
62
+ placeholders.add(`<${match[1]}>`);
63
+ }
64
+
65
+ return [...placeholders].sort();
66
+ }
67
+
68
+ /**
69
+ * The syntax a token from extractPlaceholders is written in — read off the
70
+ * token's own form, which is how extractPlaceholders tells them apart:
71
+ * 'i18next' "{{name}}" (i18next interpolation)
72
+ * 'markup' "<bold>" (a tag: react-intl, react-i18next, HTML)
73
+ * 'brace' "name" (a single-brace "{name}" — an ICU/MessageFormat
74
+ * argument when the message parses as ICU, which
75
+ * lib/icu-structure.js checkICUStructure judges)
76
+ * printf conversions (%s, %(name)s) are not tokens here: checkICUStructure
77
+ * compares them and tags its findings 'printf'.
78
+ *
79
+ * @param {string} token
80
+ * @returns {'i18next'|'markup'|'brace'}
81
+ */
82
+ function placeholderSyntax(token) {
83
+ if (token.startsWith('{{')) return 'i18next';
84
+ if (token.startsWith('<')) return 'markup';
85
+ return 'brace';
86
+ }
87
+
88
+ /**
89
+ * A token as the user wrote it: "name" was "{name}".
90
+ *
91
+ * @param {string} token
92
+ * @returns {string}
93
+ */
94
+ function showPlaceholder(token) {
95
+ return placeholderSyntax(token) === 'brace' ? `{${token}}` : token;
96
+ }
97
+
98
+ /**
99
+ * Compare placeholders between source and target strings.
100
+ *
101
+ * @param {string} sourceValue - Source locale value
102
+ * @param {string} targetValue - Target locale value
103
+ * @returns {{ missing: string[], extra: string[] }} Placeholder differences
104
+ */
105
+ function comparePlaceholders(sourceValue, targetValue) {
106
+ const sourcePH = extractPlaceholders(sourceValue);
107
+ const targetPH = extractPlaceholders(targetValue);
108
+
109
+ const missing = sourcePH.filter(p => !targetPH.includes(p));
110
+ const extra = targetPH.filter(p => !sourcePH.includes(p));
111
+
112
+ return { missing, extra };
113
+ }
114
+
115
+ /**
116
+ * Does `text` parse as an ICU message with arguments (either apostrophe
117
+ * reading — checkICUStructure's own test)? Its single-brace tokens are then
118
+ * the ICU check's to judge, argument by argument.
119
+ *
120
+ * @param {unknown} text
121
+ * @returns {boolean}
122
+ */
123
+ function parsesAsICU(text) {
124
+ if (typeof text !== 'string' || !text.includes('{')) return false;
125
+ return ['icu', 'literal'].some((apostrophes) => {
126
+ const parsed = parseMessage(text, { apostrophes });
127
+ return parsed.ok && hasArguments(parsed.nodes);
128
+ });
129
+ }
130
+
131
+ /** What each syntax is called in a finding: "i18next {{…}} placeholder …". */
132
+ const PLACEHOLDER_SYNTAX_NAMES = Object.freeze({
133
+ i18next: 'i18next {{…}}',
134
+ brace: '{…}',
135
+ markup: 'tag',
136
+ });
137
+
138
+ /**
139
+ * What a translation did to the placeholders its source carries, each change
140
+ * named by its syntax (placeholderSyntax) — the findings verify reports, and
141
+ * the ones the sync gate refuses:
142
+ * { syntax: 'i18next', token: '{{name}}', issue: 'placeholder {{name}} was changed to {{nom}}' }
143
+ *
144
+ * A missing token and an extra one of the same name read as one change
145
+ * ("{{name}} was changed to {name}") and are named by the SOURCE's syntax;
146
+ * the remaining missing/extra tokens pair up in order. Left out:
147
+ * - a brace token (single or double) when the source parses as an ICU
148
+ * message — the ICU check compares its arguments, and the token scan
149
+ * reads a one-word plural branch ("other {articles}") as a placeholder
150
+ * and a branch holding only an argument ("other{{count}}") as i18next;
151
+ * - a tag when the markup check already reported the value
152
+ * (`markupReported`), said precisely there.
153
+ * printf conversions are not tokens here: checkICUStructure compares them.
154
+ *
155
+ * @param {string} sourceValue
156
+ * @param {string} targetValue
157
+ * @param {{ markupReported?: boolean }} [options]
158
+ * @returns {Array<{ syntax: 'i18next'|'brace'|'markup', token: string, issue: string }>}
159
+ */
160
+ function placeholderChanges(sourceValue, targetValue, { markupReported = false } = {}) {
161
+ if (typeof sourceValue !== 'string' || typeof targetValue !== 'string') return [];
162
+ const { missing, extra } = comparePlaceholders(sourceValue, targetValue);
163
+ if (missing.length === 0 && extra.length === 0) return [];
164
+ const nameOf = (token) => (/[A-Za-z_$][\w.$]*/.exec(token) || [token])[0];
165
+ const changes = [];
166
+ for (const m of [...missing]) {
167
+ const e = extra.find(x => nameOf(x) === nameOf(m));
168
+ if (!e) continue;
169
+ changes.push([m, e]);
170
+ missing.splice(missing.indexOf(m), 1);
171
+ extra.splice(extra.indexOf(e), 1);
172
+ }
173
+ while (missing.length > 0 && extra.length > 0) changes.push([missing.shift(), extra.shift()]);
174
+ const items = [
175
+ ...changes.map(([m, e]) => [m, `placeholder ${showPlaceholder(m)} was changed to ${showPlaceholder(e)}`]),
176
+ ...missing.map(m => [m, `placeholder ${showPlaceholder(m)} is missing`]),
177
+ ...extra.map(e => [e, `placeholder ${showPlaceholder(e)} is not in the source`]),
178
+ ];
179
+ const icuSource = parsesAsICU(sourceValue);
180
+ const out = [];
181
+ for (const [token, issue] of items) {
182
+ const syntax = placeholderSyntax(token);
183
+ // In an ICU message, braces are ICU's: "other{{count}}" is a branch
184
+ // holding the argument, not i18next (an i18next "{{name}}" message never
185
+ // parses as ICU).
186
+ if ((syntax === 'brace' || syntax === 'i18next') && icuSource) continue;
187
+ if (syntax === 'markup' && markupReported) continue;
188
+ out.push({ syntax, token, issue });
189
+ }
190
+ return out;
191
+ }
192
+
193
+ // ── A sentence break inserted beside a placeholder ──────────────────────────
194
+ //
195
+ // "Take this medicine at {time}." came back as "… sina. {time}." — a sentence
196
+ // end before the placeholder, so the app shows the time as a sentence of its
197
+ // own — and the gate and verify both passed it: every placeholder was there
198
+ // (Round 14, hospital persona; a medical string). The rule below is
199
+ // structural and narrow: it does not judge fluency, only a sentence boundary
200
+ // the TRANSLATION puts right beside a placeholder where the SOURCE has none.
201
+
202
+ /**
203
+ * Sentence-final marks, across scripts: Latin/Cyrillic/Greek . ! ? (Greek's
204
+ * question mark ";" is left out: it is also the semicolon), CJK 。!? and the
205
+ * halfwidth/fullwidth forms, Devanagari/Bengali danda । ॥, Arabic ؟ and the
206
+ * Urdu full stop ۔, Ethiopic ። ፧, Armenian ։, Myanmar ။, Khmer ។ ៕,
207
+ * Canadian Syllabics ᙮ (Cree, Inuktitut), and the doubled marks ‼ ⁇ ⁈ ⁉ ‽.
208
+ * Script-agnostic: a mark from any script is a boundary in any text.
209
+ */
210
+ const SENTENCE_MARKS = new Set([...'.!?。!?。.।॥؟۔።፧։။។៕᙮‼⁇⁈⁉‽']);
211
+
212
+ /** Closing quotes and brackets that may stand between a mark and what follows it. */
213
+ const CLOSERS = new Set([...'"\'”’»›」』))]]']);
214
+
215
+ /**
216
+ * Clause punctuation: where the SOURCE already sets a placeholder apart
217
+ * ("… corpora; {spec}.", "Error: {message}", "— {noncomp}, and …"), a
218
+ * translation may make that clause a sentence of its own.
219
+ */
220
+ const CLAUSE_MARKS = new Set([...';:,—–·|(([[、,;:']);
221
+
222
+ /** Whitespace (JavaScript's \s includes no-break and narrow no-break space). */
223
+ const SPACE = /\s/;
224
+
225
+ /**
226
+ * Placeholder spans with their positions: i18next / Hugo / Handlebars
227
+ * `{{…}}`, a single-brace argument `{name}` / `{0}` / `{n, number}` (also
228
+ * Ruby's `%{name}`), and printf `%(name)s`, `%1$s`, `%s`, `%d`, `%@`.
229
+ * Nested ICU (plural/select) is not matched as one span — messages holding
230
+ * it are left out of this check (see placeholderSentenceBreaks).
231
+ */
232
+ const PLACEHOLDER_SPAN = /\{\{[^{}]*\}\}|%?\{\s*(?:[A-Za-z_$][\w.$]*|\d+)\s*(?:,[^{}]*)?\}|%\(\w+\)[a-zA-Z]|%\d+\$[a-zA-Z@]|%[sd@]/g;
233
+
234
+ /** An ICU plural/select/selectordinal argument — its branches are not placeholder spans. */
235
+ const ICU_BRANCHING = /\{\s*[\w.$]+\s*,\s*(?:plural|select|selectordinal)\s*,/;
236
+
237
+ function placeholderSpans(text) {
238
+ const spans = [];
239
+ for (const m of text.matchAll(PLACEHOLDER_SPAN)) {
240
+ spans.push({ token: m[0].replace(/\s+/g, ''), shown: m[0], start: m.index, end: m.index + m[0].length });
241
+ }
242
+ return spans;
243
+ }
244
+
245
+ /**
246
+ * Is text[i] a sentence boundary? Any mark in SENTENCE_MARKS but a full stop
247
+ * is one. A full stop "." is one unless it is:
248
+ * - part of an ellipsis ("..." — a pause, not an end),
249
+ * - followed directly by a letter, digit or "_" ("3.5", "{host}.com", "e.g"),
250
+ * - the stop of a one-letter word — an initial or an abbreviation
251
+ * ("M. {name}", "p. {page}", "z. B. {x}"),
252
+ * - the stop of a short capitalised word (two or three letters) — an
253
+ * abbreviation before a number or name ("Nr. {id}", "Dr. {name}",
254
+ * "Tel. {phone}", "Ca. {count}").
255
+ * A stop right after a placeholder ("{time}.", "%(dose)s.") is never read as
256
+ * an abbreviation: `afterPlaceholder` says the stop follows one.
257
+ *
258
+ * @param {string} text
259
+ * @param {number} i
260
+ * @param {boolean} [afterPlaceholder]
261
+ * @returns {boolean}
262
+ */
263
+ function isSentenceBoundaryAt(text, i, afterPlaceholder = false) {
264
+ const ch = text[i];
265
+ if (!SENTENCE_MARKS.has(ch)) return false;
266
+ if (ch !== '.') return true;
267
+ if (text[i - 1] === '.' || text[i + 1] === '.') return false;
268
+ const next = text[i + 1];
269
+ if (next !== undefined && /[\p{L}\p{N}_]/u.test(next)) return false;
270
+ if (afterPlaceholder) return true;
271
+ const word = /(\p{L}+)$/u.exec(text.slice(0, i))?.[1] || '';
272
+ if (word.length === 1) return false;
273
+ if (word.length <= 3 && /^\p{Lu}/u.test(word)) return false;
274
+ return true;
275
+ }
276
+
277
+ /**
278
+ * What stands on one side of a placeholder span, skipping spaces (and, on
279
+ * the left, closing quotes/brackets): the message's edge, a sentence
280
+ * boundary, clause punctuation, or a word (anything else, another
281
+ * placeholder included).
282
+ *
283
+ * @returns {{ kind: 'edge'|'boundary'|'clause'|'word', at: number }}
284
+ */
285
+ function sideOf(text, span, side, placeholderEnds = new Set()) {
286
+ if (side === 'left') {
287
+ let j = span.start - 1;
288
+ while (j >= 0 && (SPACE.test(text[j]) || CLOSERS.has(text[j]))) j--;
289
+ if (j < 0) return { kind: 'edge', at: -1 };
290
+ if (isSentenceBoundaryAt(text, j, placeholderEnds.has(j))) return { kind: 'boundary', at: j };
291
+ if (SENTENCE_MARKS.has(text[j]) || CLAUSE_MARKS.has(text[j])) return { kind: 'clause', at: j };
292
+ return { kind: 'word', at: j };
293
+ }
294
+ let k = span.end;
295
+ while (k < text.length && SPACE.test(text[k])) k++;
296
+ if (k >= text.length) return { kind: 'edge', at: k };
297
+ if (isSentenceBoundaryAt(text, k, k === span.end || /^\s*$/.test(text.slice(span.end, k)))) return { kind: 'boundary', at: k };
298
+ if (SENTENCE_MARKS.has(text[k]) || CLAUSE_MARKS.has(text[k]) || CLOSERS.has(text[k])) return { kind: 'clause', at: k };
299
+ return { kind: 'word', at: k };
300
+ }
301
+
302
+ /**
303
+ * The sentence breaks a translation INSERTED beside a placeholder, leaving
304
+ * it standing as a sentence of its own — the one rule the sync gate refuses
305
+ * by (lib/validate.js) and verify flags by (lib/verify.js):
306
+ * { token: '{time}', side: 'before', issue: 'a sentence break was inserted before {time} ("medisina. {time}.") — {time} now stands as a sentence of its own; in the source it is part of one' }
307
+ *
308
+ * A finding needs:
309
+ * 1. a placeholder the translation shares with the source, with a
310
+ * sentence boundary right BEFORE or right AFTER it in the translation
311
+ * ("medisina. {time}", "{time}. Kisik") — text before it, so a mark
312
+ * that opens nothing is not one;
313
+ * 2. nothing of its sentence around it: on each side only a boundary or
314
+ * the message's edge (". {time}." — the time shown as a sentence);
315
+ * 3. in the source, that placeholder inside a sentence: a word on one
316
+ * side, not a boundary, clause punctuation or the edge
317
+ * ("Take this medicine at {time}.").
318
+ * Narrow on purpose (Round 14 measurement, lib/placeholders.js tests): a
319
+ * placeholder that only starts or ends a sentence in the translation is
320
+ * word order, not damage — "Open an issue on {github} or …" is rightly
321
+ * "{github}에 이슈를 열거나 …" in Korean, and "Shipped by {carrier} on
322
+ * {date}." rightly "Expédié le {date} par {carrier}." Left out: ICU
323
+ * plural/select messages (the ICU check judges their branches).
324
+ *
325
+ * @param {string} sourceValue
326
+ * @param {string} targetValue
327
+ * @returns {Array<{ token: string, side: 'before'|'after', issue: string }>}
328
+ */
329
+ function placeholderSentenceBreaks(sourceValue, targetValue) {
330
+ if (typeof sourceValue !== 'string' || typeof targetValue !== 'string') return [];
331
+ if (sourceValue === targetValue) return [];
332
+ if (ICU_BRANCHING.test(sourceValue) || ICU_BRANCHING.test(targetValue)) return [];
333
+ const targetSpans = placeholderSpans(targetValue);
334
+ if (targetSpans.length === 0) return [];
335
+ const sourceSpans = placeholderSpans(sourceValue);
336
+
337
+ // In the source the placeholder is part of a sentence: a word beside it.
338
+ const inSentence = (token) => sourceSpans.some(s => s.token === token
339
+ && (sideOf(sourceValue, s, 'left').kind === 'word' || sideOf(sourceValue, s, 'right').kind === 'word'));
340
+
341
+ const ends = new Set(targetSpans.map(t => t.end));
342
+ const found = [];
343
+ const seen = new Set();
344
+ for (const s of targetSpans) {
345
+ if (seen.has(s.token) || !inSentence(s.token)) continue; // a missing/extra placeholder is placeholderChanges' finding
346
+ const left = sideOf(targetValue, s, 'left', ends);
347
+ const right = sideOf(targetValue, s, 'right');
348
+ const alone = (left.kind === 'boundary' || left.kind === 'edge') && (right.kind === 'boundary' || right.kind === 'edge');
349
+ if (!alone || (left.kind !== 'boundary' && right.kind !== 'boundary')) continue;
350
+ // A boundary before it must close some text; one after it at the very
351
+ // end of the message, with nothing before, is the message's own end.
352
+ if (left.kind === 'boundary' && !/[\p{L}\p{N}]/u.test(targetValue.slice(0, left.at))) continue;
353
+ seen.add(s.token);
354
+ const side = left.kind === 'boundary' ? 'before' : 'after';
355
+ // The word before the mark, the placeholder, the mark after it: "medisina. {time}."
356
+ const from = left.kind === 'boundary' ? Math.max(0, targetValue.slice(0, left.at).search(/\S+\s*$/)) : s.start;
357
+ const to = right.kind === 'boundary' ? right.at + 1 : s.end;
358
+ const shown = targetValue.slice(from, to).trim();
359
+ found.push({
360
+ token: s.token,
361
+ side,
362
+ issue: `a sentence break was inserted ${side} ${s.shown} ("${shown}") — ${s.shown} now stands as a sentence of its own; in the source it is part of one`,
363
+ });
364
+ }
365
+ return found;
366
+ }
367
+
368
+ /** Sentence boundaries in a text: runs of marks ("?!", "。") count once. */
369
+ function countSentenceBoundaries(text) {
370
+ const ends = new Set(placeholderSpans(text).map(s => s.end));
371
+ let n = 0;
372
+ for (let i = 0; i < text.length; i++) {
373
+ if (!SENTENCE_MARKS.has(text[i])) continue;
374
+ let j = i;
375
+ while (j + 1 < text.length && SENTENCE_MARKS.has(text[j + 1])) j++;
376
+ for (let k = i; k <= j; k++) {
377
+ if (isSentenceBoundaryAt(text, k, ends.has(k))) { n++; break; }
378
+ }
379
+ i = j;
380
+ }
381
+ return n;
382
+ }
383
+
384
+ export {
385
+ extractPlaceholders,
386
+ comparePlaceholders,
387
+ placeholderSyntax,
388
+ showPlaceholder,
389
+ parsesAsICU,
390
+ placeholderChanges,
391
+ placeholderSentenceBreaks,
392
+ countSentenceBoundaries,
393
+ PLACEHOLDER_SYNTAX_NAMES,
394
+ };
package/lib/plugins.js CHANGED
@@ -148,6 +148,9 @@ function validateManifest(manifest) {
148
148
  if (manifest.type === 'api' && !manifest.endpoint) {
149
149
  errors.push('API plugins must include an "endpoint" URL');
150
150
  }
151
+ if (manifest.acceptsInstructions !== undefined && typeof manifest.acceptsInstructions !== 'boolean') {
152
+ errors.push('"acceptsInstructions" must be true or false');
153
+ }
151
154
 
152
155
  // Benchmarks: validate structure if present
153
156
  if (manifest.benchmarks) {
@@ -375,6 +378,11 @@ function resolvePluginForPair(plugins, pairConfig) {
375
378
 
376
379
  // Plugin-specific fields
377
380
  endpoint: plugin.endpoint || pairConfig.endpoint || null,
381
+ // The endpoint's declared instruction capability: the pair's own
382
+ // statement wins, then the manifest's; null = unknown.
383
+ acceptsInstructions: typeof pairConfig.acceptsInstructions === 'boolean'
384
+ ? pairConfig.acceptsInstructions
385
+ : (typeof plugin.acceptsInstructions === 'boolean' ? plugin.acceptsInstructions : null),
378
386
  pluginName: plugin.name,
379
387
  pluginVersion: plugin.version,
380
388
  pluginDir: plugin._pluginDir,
@@ -0,0 +1,109 @@
1
+ /**
2
+ * plural-gap-redo.js — a plural message left incomplete is asked again by a
3
+ * setup that has not tried it yet.
4
+ *
5
+ * THE FINDING (Round 13, Django persona): a local model left a Russian
6
+ * plural entry without its few/many forms. Sync wrote the entry with the
7
+ * "other" form standing in and marked it (`# champollion:` in a gettext
8
+ * catalog; a missing ICU branch in JSON), and every later sync treated it as
9
+ * done — including CI's, which runs a stronger hosted model. Every sync then
10
+ * exited 2 until a person re-ran with a key or wrote the forms by hand, and
11
+ * the CI workflow had no way to do either.
12
+ *
13
+ * THE RULE. A marked gap is not a translation. A sync asks for it again —
14
+ * from the model, not the cache, which holds the incomplete answer — when it
15
+ * runs a setup (method, model, register, coaching: the cache key) that has
16
+ * not answered this message's current text yet:
17
+ * - the setup that wrote it is the one the lock records for the key
18
+ * (`by`), while the record still matches the value on disk (a person who
19
+ * edited the entry is never overruled);
20
+ * - the setups that answered it without the forms before are kept in the
21
+ * lock (`gaps`: source hash + method keys), so a local run and a CI run
22
+ * do not take turns paying for the same incomplete answer;
23
+ * - the pair's own fallback is part of the setup: a gap it left is not
24
+ * re-asked by the same configuration.
25
+ * `--redo gaps` asks for every gap on disk, whoever left it (an explicit
26
+ * redo). A setup that also leaves the forms out leaves the gap marked, as
27
+ * before; the run exits 2.
28
+ *
29
+ * Shared by the sync and its cost estimate, so the estimate prices exactly
30
+ * what is asked again.
31
+ */
32
+
33
+ import { pluralGapsInFile } from './verify.js';
34
+ import { tmMethodKey } from './tm.js';
35
+ import { decodeWritten, shortSourceHash, valueHash } from './locale-state.js';
36
+
37
+ /**
38
+ * The setups a gap record says answered this text without the forms.
39
+ *
40
+ * @param {object|undefined} record - LockState.of(code).gaps[lockKey]
41
+ * @param {string} sourceValue
42
+ * @returns {string[]}
43
+ */
44
+ export function gapTriedBy(record, sourceValue) {
45
+ if (!record || typeof record !== 'object' || typeof sourceValue !== 'string') return [];
46
+ if (record.source !== shortSourceHash(sourceValue) || !Array.isArray(record.methods)) return [];
47
+ return record.methods;
48
+ }
49
+
50
+ /**
51
+ * Remember that `methodKey` answered this key's current text without a form
52
+ * the language uses for ordinary counts.
53
+ *
54
+ * @param {object} localeState - LockState.of(code)
55
+ * @param {string} lockKey
56
+ * @param {string} sourceValue
57
+ * @param {string} methodKey
58
+ */
59
+ export function recordGap(localeState, lockKey, sourceValue, methodKey) {
60
+ if (typeof sourceValue !== 'string' || !methodKey) return;
61
+ localeState.gaps = localeState.gaps || {};
62
+ const known = gapTriedBy(localeState.gaps[lockKey], sourceValue);
63
+ localeState.gaps[lockKey] = { source: shortSourceHash(sourceValue), methods: [...new Set([...known, methodKey])] };
64
+ }
65
+
66
+ /**
67
+ * Plural messages of one target file that this run asks for again.
68
+ *
69
+ * @param {object} p
70
+ * @param {object} p.file - layout file ({ path, format, rel })
71
+ * @param {object} p.expected - The target's expected map (source text per target key)
72
+ * @param {object} p.targetFlat - Values on disk
73
+ * @param {string} p.locale
74
+ * @param {object} p.localeState - LockState.peek(code)
75
+ * @param {(key: string) => string} p.lockKeyOf
76
+ * @param {object} p.pairConfig
77
+ * @param {boolean} [p.all] - `--redo gaps`: every gap on disk
78
+ * @returns {{ keys: string[], gaps: Array<{ key: string, missing: string[], marked: boolean }>,
79
+ * by: Object<string, string|null>, missing: Object<string, string[]> }}
80
+ * keys — target keys to ask again; gaps — every gap in the file; by — the
81
+ * setup that left each asked key (null: no record); missing — its forms
82
+ */
83
+ export function planGapRedo({ file, expected, targetFlat, locale, localeState, lockKeyOf, pairConfig, all = false }) {
84
+ const out = { keys: [], gaps: [], by: {}, missing: {} };
85
+ if (!file || !targetFlat) return out;
86
+ try { out.gaps = pluralGapsInFile({ file, expected, targetFlat, locale }); } catch { out.gaps = []; }
87
+ if (out.gaps.length === 0) return out;
88
+ const current = tmMethodKey(pairConfig);
89
+ const fallback = pairConfig.fallback ? tmMethodKey(pairConfig.fallback) : null;
90
+ for (const g of out.gaps) {
91
+ const src = expected[g.key];
92
+ const value = targetFlat[g.key];
93
+ if (typeof src !== 'string' || typeof value !== 'string') continue;
94
+ const lk = lockKeyOf(g.key);
95
+ // Who wrote what is on disk — only while the record still describes it.
96
+ const record = decodeWritten(localeState?.written?.[lk]);
97
+ const writer = record && record.source === shortSourceHash(src) && record.value === valueHash(value)
98
+ ? (localeState?.by?.[lk] || null) : null;
99
+ if (!all) {
100
+ if (!writer) continue; // unknown, or a person's edit: never automatic
101
+ if (writer === current || writer === fallback) continue; // this setup left it
102
+ if (gapTriedBy(localeState?.gaps?.[lk], src).includes(current)) continue; // tried already
103
+ }
104
+ out.keys.push(g.key);
105
+ out.by[g.key] = writer;
106
+ out.missing[g.key] = g.missing;
107
+ }
108
+ return out;
109
+ }