claude-translator 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,449 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * i18n step 6 — translation quality assessment.
4
+ *
5
+ * node scripts/i18n/tqa.mjs --lang es
6
+ * node scripts/i18n/tqa.mjs --lang es,fr --sample 200
7
+ * node scripts/i18n/tqa.mjs --lang es --judge-provider openai --judge-model gpt-5.6-luna
8
+ * node scripts/i18n/tqa.mjs --lang es --dry # cost estimate only
9
+ * node scripts/i18n/tqa.mjs --lang es --repeat # measure the judge's own variance
10
+ *
11
+ * Reads i18n/source.json, i18n/tm/{lang}.json
12
+ * Writes i18n/tqa/{lang}.json and i18n/tqa/{lang}.md
13
+ *
14
+ * ── What this is ─────────────────────────────────────────────────────────────
15
+ * An MQM-style error-typology assessment of a stratified sample, scored per 100 words.
16
+ * MQM is the standard the industry actually uses, so the output is comparable with a
17
+ * human review rather than being a number invented here.
18
+ *
19
+ * score = 100 - (weighted error points / words) * 100
20
+ *
21
+ * ── What this is NOT ─────────────────────────────────────────────────────────
22
+ * It is not a human certification, and nothing here should be reported as one. It is one
23
+ * model's opinion of another model's output, and models are known to prefer their own
24
+ * work — which is why the judge defaults to a DIFFERENT provider than the translator
25
+ * whenever one is configured, and why --repeat exists to show how much the judge moves
26
+ * between identical runs. A quality number without its variance is marketing.
27
+ *
28
+ * The honest reading: this reliably finds the bottom of the distribution — the units
29
+ * that are actually wrong — and should not be trusted to the second decimal place.
30
+ */
31
+
32
+ import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'fs';
33
+ import { join } from 'path';
34
+
35
+ import {
36
+ SOURCE_FILE, TM_DIR, I18N_DIR, LOCALES, GLOSSARY, ROOT_DIR as ROOT,
37
+ SITE_NAME, SITE_DESCRIPTION, SOURCE_LANGUAGE, MODEL as CFG_MODEL,
38
+ PROVIDER as CFG_PROVIDER, API_BASE_URL, API_KEY_ENV, JSON_MODE, PRICING,
39
+ } from './config.mjs';
40
+ import { creditBlock } from './credit.mjs';
41
+ import { loadProvider } from './providers/index.mjs';
42
+ import { termsForBatch } from './glossary.mjs';
43
+ import { WEIGHT, CATEGORIES, rng, words, stratifiedSample, parseErrors, score } from './tqa-score.mjs';
44
+
45
+ // ── CLI ──────────────────────────────────────────────────────────────────────
46
+
47
+ const args = Object.fromEntries(
48
+ process.argv
49
+ .slice(2)
50
+ .join(' ')
51
+ .split('--')
52
+ .filter(Boolean)
53
+ .map((s) => s.trim().split(/\s+/))
54
+ .map(([k, ...v]) => [k, v.join(' ') || true])
55
+ );
56
+
57
+ const LANGS = String(args.lang ?? '').split(',').map((s) => s.trim()).filter(Boolean);
58
+ const SAMPLE = Number(args.sample ?? 100);
59
+ const SEED = Number(args.seed ?? 20260829);
60
+ const BATCH = Number(args.batch ?? 10);
61
+ const CONCURRENCY = Number(args.concurrency ?? 6);
62
+ const DRY = Boolean(args.dry);
63
+ const REPEAT = Boolean(args.repeat);
64
+
65
+ if (LANGS.length === 0) {
66
+ console.error(
67
+ 'Usage: node scripts/i18n/tqa.mjs --lang es[,fr] [--sample N] [--seed N] [--dry] [--repeat]\n' +
68
+ ' [--judge-provider P] [--judge-model M]'
69
+ );
70
+ process.exit(1);
71
+ }
72
+
73
+ // ── MQM ──────────────────────────────────────────────────────────────────────
74
+
75
+ // ── Judge ────────────────────────────────────────────────────────────────────
76
+
77
+ const JUDGE_PROVIDER_NAME = args['judge-provider'] ? String(args['judge-provider']) : null;
78
+ const JUDGE_MODEL_ARG = args['judge-model'] ? String(args['judge-model']) : null;
79
+
80
+ /**
81
+ * Pick a judge that is not the translator, when we can.
82
+ *
83
+ * Self-preference is a documented failure of LLM-as-judge setups: a model scores its own
84
+ * output higher than a peer's. Defaulting the judge to a different provider costs nothing
85
+ * and removes the most obvious objection to the number. If only one provider is
86
+ * configured, we use it and SAY SO in the report rather than hiding it.
87
+ */
88
+ function pickJudge() {
89
+ if (JUDGE_PROVIDER_NAME) return { provider: JUDGE_PROVIDER_NAME, reason: 'chosen with --judge-provider' };
90
+ const translator = CFG_PROVIDER ?? 'anthropic';
91
+ const alternatives = ['anthropic', 'gemini', 'openai'].filter((p) => p !== translator);
92
+ for (const alt of alternatives) {
93
+ const keys = { anthropic: 'ANTHROPIC_API_KEY', gemini: 'GEMINI_API_KEY', openai: 'OPENAI_API_KEY' }[alt];
94
+ if (process.env[keys]) return { provider: alt, reason: `differs from the translator (${translator})` };
95
+ }
96
+ return { provider: translator, reason: `SAME as the translator — no other provider key found` };
97
+ }
98
+
99
+ function judgePrompt(langName, langCode, terms) {
100
+ const glossaryLines = terms.length
101
+ ? [
102
+ `The site glossary constrains these terms:`,
103
+ ...terms.map((t) =>
104
+ t.rule === 'keep'
105
+ ? ` - "${t.source}" must appear unchanged`
106
+ : ` - "${t.source}" must be rendered as "${t.targets[langCode]}"`
107
+ ),
108
+ ]
109
+ : [];
110
+
111
+ return [
112
+ `You are a senior localization reviewer assessing ${langName} translations of ${SITE_NAME}${SITE_DESCRIPTION ? `, ${SITE_DESCRIPTION}` : ''}.`,
113
+ `The source language is ${SOURCE_LANGUAGE}. This is marketing and product copy for a website.`,
114
+ ``,
115
+ `For each item, list the translation errors using MQM categories and severities.`,
116
+ ``,
117
+ `CATEGORIES: ${CATEGORIES.join(', ')}`,
118
+ `SEVERITIES: minor (noticeable, does not mislead), major (misleads or reads as wrong),`,
119
+ ` critical (reverses meaning, breaks a promise, or is unusable)`,
120
+ ``,
121
+ ...glossaryLines,
122
+ ``,
123
+ `RULES`,
124
+ `1. <0>, </0>, <1/> are markup placeholders. They must appear in the translation with`,
125
+ ` the same count and numbers. Report any difference as markup/placeholder, critical.`,
126
+ `2. Numbers, prices and measurements must keep their VALUE. Different digit grouping or`,
127
+ ` symbol placement is correct localization, not an error. A changed value is critical.`,
128
+ `3. A term left in the source language is only an error if it should have been`,
129
+ ` translated. Brand names, formats and protocol names are correct unchanged.`,
130
+ `4. Judge the translation on its own terms as ${langName} copy. Do not reward literalness.`,
131
+ `5. Report NO errors when there are none. An empty list is the expected result for`,
132
+ ` most items, and inventing a minor error to look thorough makes the whole score useless.`,
133
+ ``,
134
+ `Return a JSON array. For each input item return { "id": <same id>, "text": "<errors>" }`,
135
+ `where <errors> is a JSON array serialised as a string, e.g.`,
136
+ `"[{\\"c\\":\\"fluency/grammar\\",\\"s\\":\\"minor\\",\\"n\\":\\"wrong gender agreement\\"}]"`,
137
+ `or "[]" when the translation is correct.`,
138
+ ]
139
+ .filter((l) => l !== null)
140
+ .join('\n');
141
+ }
142
+
143
+ // ── API ──────────────────────────────────────────────────────────────────────
144
+
145
+ const JUDGE = pickJudge();
146
+ const PROVIDER = await loadProvider({ provider: JUDGE.provider, model: JUDGE_MODEL_ARG, root: ROOT });
147
+ const MODEL = JUDGE_MODEL_ARG ?? PROVIDER.defaultModel;
148
+
149
+ function loadKey() {
150
+ const names = API_KEY_ENV ? [API_KEY_ENV] : (PROVIDER.envKeys ?? []);
151
+ for (const name of names) if (process.env[name]) return process.env[name];
152
+ const envFile = join(ROOT, '.env');
153
+ if (existsSync(envFile)) {
154
+ const text = readFileSync(envFile, 'utf8');
155
+ for (const name of names) {
156
+ const m = new RegExp(`^${name}=(.*)$`, 'm').exec(text);
157
+ if (m) return m[1].trim();
158
+ }
159
+ }
160
+ if (PROVIDER.keyOptional) return null;
161
+ console.error(`No API key for ${PROVIDER.label ?? PROVIDER.id}. Set ${names.join(' or ')}.`);
162
+ process.exit(1);
163
+ }
164
+ const KEY = DRY ? null : loadKey();
165
+
166
+ /**
167
+ * One judging request. Reuses the translator's retry and batch-splitting behaviour by
168
+ * shape rather than by import: the same safety-block and truncation failures apply to a
169
+ * review pass, and a batch that trips one is halved rather than lost.
170
+ */
171
+ async function callJudge(system, items, attempt = 1, jsonMode = JSON_MODE) {
172
+ const { url, headers, body } = PROVIDER.request({
173
+ model: MODEL, system, items, temperature: 0, key: KEY, baseUrl: API_BASE_URL, jsonMode,
174
+ });
175
+
176
+ const split = async (why) => {
177
+ if (items.length === 1) return { rows: [], usage: { inTok: 0, outTok: 0 } };
178
+ const mid = Math.ceil(items.length / 2);
179
+ process.stderr.write(` ${why} on ${items.length} items — splitting\n`);
180
+ const [a, b] = await Promise.all([
181
+ callJudge(system, items.slice(0, mid), 1, jsonMode),
182
+ callJudge(system, items.slice(mid), 1, jsonMode),
183
+ ]);
184
+ return {
185
+ rows: [...a.rows, ...b.rows],
186
+ usage: { inTok: a.usage.inTok + b.usage.inTok, outTok: a.usage.outTok + b.usage.outTok },
187
+ };
188
+ };
189
+
190
+ let res;
191
+ try {
192
+ res = await fetch(url, { method: 'POST', headers, body: JSON.stringify(body) });
193
+ } catch (err) {
194
+ if (attempt <= 5) {
195
+ await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
196
+ return callJudge(system, items, attempt + 1, jsonMode);
197
+ }
198
+ throw err;
199
+ }
200
+
201
+ if (!res.ok) {
202
+ const text = await res.text();
203
+ if (PROVIDER.unsupportedJsonMode?.(res.status, text) && (jsonMode ?? 'schema') !== 'none') {
204
+ return callJudge(system, items, attempt, (jsonMode ?? 'schema') === 'schema' ? 'object' : 'none');
205
+ }
206
+ if ((res.status === 429 || res.status >= 500) && attempt <= 5) {
207
+ await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
208
+ return callJudge(system, items, attempt + 1, jsonMode);
209
+ }
210
+ throw new Error(`${PROVIDER.label ?? PROVIDER.id} HTTP ${res.status}: ${text.slice(0, 300)}`);
211
+ }
212
+
213
+ const data = await res.json();
214
+ const { text: part, usage, retryable, detail } = PROVIDER.parse(data);
215
+ if (retryable === 'safety') return split(detail ?? 'safety block');
216
+ if (retryable === 'truncated') return split('truncated');
217
+
218
+ try {
219
+ const rows = PROVIDER.unwrap ? PROVIDER.unwrap(JSON.parse(part)) : JSON.parse(part);
220
+ return { rows: Array.isArray(rows) ? rows : [], usage };
221
+ } catch {
222
+ return split('unparseable response');
223
+ }
224
+ }
225
+
226
+ // ── Run ──────────────────────────────────────────────────────────────────────
227
+
228
+ const LANG_NAMES = Object.fromEntries(LOCALES.map((l) => [l.pathCode, l.nativeLabel ?? l.pathCode]));
229
+
230
+ if (!existsSync(SOURCE_FILE)) {
231
+ console.error(`No ${SOURCE_FILE}. Run extract.mjs first.`);
232
+ process.exit(1);
233
+ }
234
+ const SOURCE = JSON.parse(readFileSync(SOURCE_FILE, 'utf8'));
235
+
236
+ console.log(`judge: ${PROVIDER.label ?? PROVIDER.id} · model: ${MODEL}`);
237
+ console.log(` ${JUDGE.reason}`);
238
+ if (JUDGE.reason.startsWith('SAME')) {
239
+ console.log(' ⚠ a model judging its own output scores it generously — treat the number as a floor, not a grade');
240
+ }
241
+
242
+ mkdirSync(join(I18N_DIR, 'tqa'), { recursive: true });
243
+
244
+ async function assess(langCode, seed) {
245
+ const tmFile = join(TM_DIR, `${langCode}.json`);
246
+ if (!existsSync(tmFile)) {
247
+ console.log(`${langCode}: no memory — skipped`);
248
+ return null;
249
+ }
250
+ const tm = JSON.parse(readFileSync(tmFile, 'utf8'));
251
+
252
+ const pool = Object.entries(SOURCE)
253
+ .filter(([hash]) => typeof tm[hash] === 'string')
254
+ .map(([hash, unit]) => ({ hash, source: unit.text, target: tm[hash], count: unit.count ?? 1 }));
255
+
256
+ if (!pool.length) {
257
+ console.log(`${langCode}: nothing translated yet — skipped`);
258
+ return null;
259
+ }
260
+
261
+ const sample = stratifiedSample(pool, SAMPLE, seed).map((u) => ({ ...u, words: words(u.source) }));
262
+ const sampledWords = sample.reduce((n, u) => n + u.words, 0);
263
+
264
+ console.log(
265
+ `\n${langCode}: assessing ${sample.length} of ${pool.length.toLocaleString()} units ` +
266
+ `(${sampledWords.toLocaleString()} source words, seed ${seed})`
267
+ );
268
+
269
+ if (DRY) {
270
+ const estIn = Math.ceil(sampledWords * 3.2) + sample.length * 40;
271
+ const estOut = sample.length * 25;
272
+ console.log(` estimated ~${estIn.toLocaleString()} in / ~${estOut.toLocaleString()} out tokens`);
273
+ const rate = PRICING ?? PROVIDER.pricing?.(MODEL) ?? null;
274
+ if (rate) {
275
+ const cost = (estIn / 1e6) * rate[0] + (estOut / 1e6) * rate[1];
276
+ console.log(` estimated cost: $${cost.toFixed(4)}`);
277
+ } else {
278
+ console.log(' no price known for this endpoint — set "pricing" in i18n.config.json for a figure');
279
+ }
280
+ return null;
281
+ }
282
+
283
+ const langName = LANG_NAMES[langCode] ?? langCode;
284
+ const batches = [];
285
+ for (let i = 0; i < sample.length; i += BATCH) batches.push(sample.slice(i, i + BATCH));
286
+
287
+ const assessed = [];
288
+ let unreadable = 0;
289
+ const usage = { inTok: 0, outTok: 0 };
290
+ let cursor = 0;
291
+ let done = 0;
292
+
293
+ const runOne = async () => {
294
+ for (;;) {
295
+ const idx = cursor++;
296
+ if (idx >= batches.length) return;
297
+ const batch = batches[idx];
298
+ const terms = termsForBatch(GLOSSARY, batch.map((u) => u.source), langCode);
299
+ const system = judgePrompt(langName, langCode, terms);
300
+ const items = batch.map((u, i) => ({ id: i, text: `SOURCE: ${u.source}\nTARGET: ${u.target}` }));
301
+
302
+ let rows = [];
303
+ try {
304
+ const out = await callJudge(system, items);
305
+ rows = out.rows;
306
+ usage.inTok += out.usage.inTok;
307
+ usage.outTok += out.usage.outTok;
308
+ } catch (err) {
309
+ process.stderr.write(` batch ${idx} failed: ${String(err.message ?? err).slice(0, 90)}\n`);
310
+ }
311
+
312
+ const byId = new Map(rows.map((r) => [Number(r.id), r.text]));
313
+ for (let i = 0; i < batch.length; i++) {
314
+ const errors = parseErrors(byId.get(i));
315
+ // A unit the judge did not return, or returned unreadably, is EXCLUDED from the
316
+ // score rather than counted as clean. Counting it clean would inflate the result
317
+ // every time the judge failed, which is precisely backwards.
318
+ if (errors === null) {
319
+ unreadable++;
320
+ continue;
321
+ }
322
+ assessed.push({ ...batch[i], errors });
323
+ }
324
+ done += batch.length;
325
+ process.stdout.write(`\r ${langCode}: ${done}/${sample.length} units judged`);
326
+ }
327
+ };
328
+
329
+ await Promise.all(Array.from({ length: Math.min(CONCURRENCY, batches.length) }, runOne));
330
+ process.stdout.write('\n');
331
+
332
+ // A score computed from nothing is not a good score, it is a broken measurement — and
333
+ // it would read as a perfect one. Every unit here failed to come back readable at least
334
+ // once during development (a 401 against the wrong endpoint), and the run cheerfully
335
+ // printed "100.00 / 100" from zero assessed units. Refuse instead.
336
+ const judgedShare = sample.length ? (assessed.length * 100) / sample.length : 0;
337
+ if (assessed.length === 0) {
338
+ console.log(
339
+ ` \u2717 no unit could be assessed \u2014 the judge returned nothing readable for all ` +
340
+ `${sample.length} of them. No score is reported, because a score from an empty sample ` +
341
+ `would read as a perfect one.`
342
+ );
343
+ return null;
344
+ }
345
+ if (judgedShare < 80) {
346
+ console.log(
347
+ ` \u26a0 only ${judgedShare.toFixed(0)}% of the sample was assessed ` +
348
+ `(${unreadable} unit(s) unreadable). The score below is computed from what came back ` +
349
+ `and should not be compared with a clean run.`
350
+ );
351
+ }
352
+
353
+ const result = score(assessed);
354
+ result.lang = langCode;
355
+ result.seed = seed;
356
+ result.sampleRequested = SAMPLE;
357
+ result.poolSize = pool.length;
358
+ result.unreadable = unreadable;
359
+ result.judge = { provider: PROVIDER.id, model: MODEL, note: JUDGE.reason };
360
+ result.usage = usage;
361
+ result.worst = assessed
362
+ .filter((a) => a.errors.length)
363
+ .sort((a, b) => {
364
+ const w = (x) => x.errors.reduce((n, e) => n + WEIGHT[e.severity], 0);
365
+ return w(b) - w(a);
366
+ })
367
+ .slice(0, 15)
368
+ .map((a) => ({ source: a.source, target: a.target, errors: a.errors }));
369
+
370
+ return result;
371
+ }
372
+
373
+ const reports = [];
374
+ for (const lang of LANGS) {
375
+ const first = await assess(lang, SEED);
376
+ if (!first) continue;
377
+
378
+ if (REPEAT) {
379
+ // The same sample, judged again. Any gap between the two is the judge's own noise,
380
+ // and reporting a score without it invites the reader to over-read a decimal place.
381
+ console.log(` re-judging the same sample to measure judge variance…`);
382
+ const second = await assess(lang, SEED);
383
+ if (second) {
384
+ first.variance = {
385
+ secondRunMqm: second.mqm,
386
+ delta: Math.round((second.mqm - first.mqm) * 100) / 100,
387
+ };
388
+ }
389
+ }
390
+
391
+ writeFileSync(join(I18N_DIR, 'tqa', `${lang}.json`), JSON.stringify(first, null, 2));
392
+ reports.push(first);
393
+
394
+ const pctClean = first.unitsAssessed ? (first.unitsClean * 100) / first.unitsAssessed : 0;
395
+ console.log(` MQM score: ${first.mqm.toFixed(2)} / 100`);
396
+ console.log(
397
+ ` ${first.unitsClean}/${first.unitsAssessed} units error-free (${pctClean.toFixed(0)}%) · ` +
398
+ `${first.bySeverity.critical} critical, ${first.bySeverity.major} major, ${first.bySeverity.minor} minor`
399
+ );
400
+ if (first.unreadable) console.log(` ${first.unreadable} unit(s) excluded — judge returned nothing readable`);
401
+ if (first.variance) {
402
+ console.log(` re-run of the same sample scored ${first.variance.secondRunMqm.toFixed(2)} (Δ ${first.variance.delta >= 0 ? '+' : ''}${first.variance.delta})`);
403
+ }
404
+ }
405
+
406
+ // ── Markdown scorecard ───────────────────────────────────────────────────────
407
+
408
+ if (reports.length) {
409
+ const lines = [
410
+ `# Translation quality assessment — ${SITE_NAME}`,
411
+ ``,
412
+ `Generated ${new Date().toISOString().slice(0, 10)} by \`tqa.mjs\`.`,
413
+ ``,
414
+ `**Method.** MQM error typology on a stratified sample, weighted toward the strings that`,
415
+ `appear most often on the site. Score = 100 − (weighted error points ÷ words) × 100, with`,
416
+ `minor = 1, major = 5, critical = 10.`,
417
+ ``,
418
+ `**This is a machine assessment, not a human certification.** The judge is`,
419
+ `${PROVIDER.label ?? PROVIDER.id} (\`${MODEL}\`) — ${JUDGE.reason}. Read the score as a`,
420
+ `comparison between locales and across runs, not as an absolute grade.`,
421
+ ``,
422
+ `| Locale | MQM | Units | Error-free | Critical | Major | Minor |`,
423
+ `| --- | ---: | ---: | ---: | ---: | ---: | ---: |`,
424
+ ...reports.map((r) => {
425
+ const pct = r.unitsAssessed ? Math.round((r.unitsClean * 100) / r.unitsAssessed) : 0;
426
+ return `| ${r.lang} | **${r.mqm.toFixed(2)}** | ${r.unitsAssessed} | ${pct}% | ${r.bySeverity.critical} | ${r.bySeverity.major} | ${r.bySeverity.minor} |`;
427
+ }),
428
+ ``,
429
+ ];
430
+
431
+ for (const r of reports) {
432
+ if (!r.worst.length) continue;
433
+ lines.push(`## ${r.lang} — worst-scoring units`, ``);
434
+ for (const w of r.worst.slice(0, 8)) {
435
+ lines.push(
436
+ `- **${w.errors.map((e) => `${e.severity} ${e.category}`).join(', ')}**`,
437
+ ` - source: \`${w.source.slice(0, 140).replace(/`/g, "'")}\``,
438
+ ` - target: \`${w.target.slice(0, 140).replace(/`/g, "'")}\``,
439
+ ...(w.errors[0]?.note ? [` - note: ${w.errors[0].note}`] : [])
440
+ );
441
+ }
442
+ lines.push(``);
443
+ }
444
+
445
+ const md = join(I18N_DIR, 'tqa', 'scorecard.md');
446
+ writeFileSync(md, lines.join('\n'));
447
+ console.log(`\nwrote ${md}`);
448
+ creditBlock([`${reports.length} locale(s) assessed · MQM sample of ${SAMPLE} units each`]);
449
+ }
@@ -27,10 +27,12 @@ import { join } from 'path';
27
27
  import { fileURLToPath } from 'url';
28
28
 
29
29
  import {
30
- SOURCE_FILE as SRC_FILE, TM_DIR, LOCALES, RTL, DNT, MODEL as CFG_MODEL,
30
+ SOURCE_FILE as SRC_FILE, TM_DIR, LOCALES, RTL, DNT, GLOSSARY, MODEL as CFG_MODEL,
31
31
  ROOT_DIR as ROOT, SITE_NAME, SITE_DESCRIPTION, SOURCE_LANGUAGE,
32
32
  PROVIDER as CFG_PROVIDER, API_BASE_URL, API_KEY_ENV, JSON_MODE, PRICING,
33
33
  } from './config.mjs';
34
+ import { termsForBatch, glossaryPrompt, glossaryFingerprint, containsTerm } from './glossary.mjs';
35
+ import { roleOf, rolePrompt } from './roles.mjs';
34
36
  import { hint, link } from './credit.mjs';
35
37
  import { loadProvider, extractJson } from './providers/index.mjs';
36
38
 
@@ -115,10 +117,18 @@ const TECH_TOKENS = [
115
117
  'OCR', 'GDPR', 'SSL', 'TLS', 'API', 'SDK', 'URL', 'HTTP', 'HTTPS', 'SEO', 'CSS', 'RSS',
116
118
  ];
117
119
 
118
- const GLOSSARY = [...new Set([...DNT.brands, ...DNT.formats, ...TECH_TOKENS])];
120
+ /**
121
+ * The flat never-translate list that still goes into rule 2. The structured glossary
122
+ * (config.GLOSSARY) carries the same brands plus per-locale targets and is injected
123
+ * per batch; this line stays because format and protocol tokens are correct unchanged
124
+ * in every language and are cheap to state once.
125
+ */
126
+ const DNT_NAMES = [...new Set([...DNT.brands, ...DNT.formats, ...TECH_TOKENS])];
119
127
 
120
- function systemPrompt(langCode) {
128
+ function systemPrompt(langCode, batchTerms = [], batchRoles = []) {
121
129
  const name = LANG_NAMES[langCode] ?? langCode;
130
+ const terminology = glossaryPrompt(batchTerms, langCode);
131
+ const context = rolePrompt(batchRoles);
122
132
  return [
123
133
  `You are a professional translator localising the website of ${SITE_NAME}${SITE_DESCRIPTION ? `, ${SITE_DESCRIPTION}` : ''}.`,
124
134
  `Translate from ${SOURCE_LANGUAGE} into ${name} (${langCode}).`,
@@ -127,7 +137,9 @@ function systemPrompt(langCode) {
127
137
  `1. Preserve every placeholder EXACTLY: <0>, </0>, <1/> and so on. Same count, same numbers.`,
128
138
  ` Placeholders wrap inline markup — move them so they wrap the equivalent words in your`,
129
139
  ` translation, but never drop, add, renumber or reorder their nesting.`,
130
- `2. Never translate these names: ${GLOSSARY.join(', ')}.`,
140
+ `2. Never translate these names: ${DNT_NAMES.join(', ')}.`,
141
+ ` Match whole words, and respect capitalisation: a lowercase common noun that`,
142
+ ` happens to spell a brand name is the common noun, and must be translated.`,
131
143
  `3. This is marketing and product copy. Translate meaning and tone, not word for word.`,
132
144
  ` Keep it natural and idiomatic for a native reader.`,
133
145
  `4. Keep numbers, prices, file sizes and counts unchanged (e.g. "120+", "1 GB", "10 MB").`,
@@ -135,6 +147,8 @@ function systemPrompt(langCode) {
135
147
  ` target language allows it. Headings stay headings; button labels stay short.`,
136
148
  `6. Do not add explanations, notes or quotes around the result.`,
137
149
  RTL.has(langCode) ? `7. ${name} is right-to-left. Write natural RTL text; do not insert directional marks.` : ``,
150
+ terminology,
151
+ context,
138
152
  ``,
139
153
  `Return a JSON array. For each input item return { "id": <same id>, "text": "<translation>" }.`,
140
154
  ]
@@ -181,9 +195,14 @@ function validate(source, translated) {
181
195
  * own field names.
182
196
  */
183
197
  async function callModel(langCode, items, attempt = 1, jsonMode = JSON_MODE, drop = new Set()) {
198
+ // Recomputed per call, not per run, so a batch that gets halved on a safety block or a
199
+ // truncation carries exactly the terms its own half contains.
200
+ const batchTerms = termsForBatch(GLOSSARY, items.map((i) => i.text), langCode);
201
+ const batchRoles = items.map((i) => i.el).filter(Boolean);
202
+
184
203
  const { url, headers, body } = PROVIDER.request({
185
204
  model: MODEL,
186
- system: systemPrompt(langCode),
205
+ system: systemPrompt(langCode, batchTerms, batchRoles),
187
206
  items,
188
207
  temperature: attempt === 1 ? 0.2 : 0.4,
189
208
  key: KEY,
@@ -308,6 +327,53 @@ async function translateLang(langCode, units) {
308
327
  const tmFile = join(TM_DIR, `${langCode}${TAG}.json`);
309
328
  const tm = existsSync(tmFile) ? JSON.parse(readFileSync(tmFile, 'utf8')) : {};
310
329
 
330
+ // ── Glossary invalidation ──────────────────────────────────────────────────
331
+ // The memory is keyed by the source hash alone, so editing a glossary target used to
332
+ // change nothing: every affected unit was already in the memory and got skipped, and
333
+ // the new terminology silently never shipped. The sidecar records the fingerprint the
334
+ // memory was built against; when it moves, the units containing an affected term are
335
+ // dropped so they re-translate. Only those — a term change must not cost a full locale.
336
+ //
337
+ // It is a SIDECAR, not a key inside the memory, because README documents
338
+ // i18n/tm/{lang}.json as a hand-editable hash -> string map and three other scripts
339
+ // iterate it. A `__meta` key would have made every coverage count off by one.
340
+ const metaFile = join(TM_DIR, `${langCode}${TAG}.meta.json`);
341
+ const meta = existsSync(metaFile) ? JSON.parse(readFileSync(metaFile, 'utf8')) : {};
342
+ const fingerprint = glossaryFingerprint(GLOSSARY);
343
+
344
+ if (typeof meta.glossary === 'string' && meta.glossary !== fingerprint) {
345
+ const before = new Set(meta.glossary.split('\n').filter(Boolean));
346
+ const after = new Set(fingerprint.split('\n').filter(Boolean));
347
+ // A term whose line is missing from either side has been added, removed or edited.
348
+ const changed = GLOSSARY.filter((t) => {
349
+ const line = glossaryFingerprint([t]);
350
+ return !before.has(line) || !after.has(line);
351
+ });
352
+ // Removed terms are gone from GLOSSARY, so recover them from the old fingerprint to
353
+ // re-translate units that were constrained by a rule the user has just deleted.
354
+ const removedSources = [...before]
355
+ .filter((line) => !after.has(line))
356
+ .map((line) => line.split('|')[2])
357
+ .filter(Boolean);
358
+
359
+ let dropped = 0;
360
+ for (const [hash, unit] of units) {
361
+ if (!(hash in tm)) continue;
362
+ const hit =
363
+ changed.some((t) => containsTerm(unit.text, t)) ||
364
+ removedSources.some((src) => containsTerm(unit.text, { source: src, matchCase: false }));
365
+ if (hit) {
366
+ delete tm[hash];
367
+ dropped++;
368
+ }
369
+ }
370
+ if (dropped) {
371
+ process.stderr.write(
372
+ ` glossary changed — re-translating ${dropped.toLocaleString()} affected unit(s)\n`
373
+ );
374
+ }
375
+ }
376
+
311
377
  const pending = units.filter(([hash]) => !(hash in tm));
312
378
 
313
379
  // Re-run churn. The memory is keyed by source hash, so on an existing locale the
@@ -330,6 +396,11 @@ async function translateLang(langCode, units) {
330
396
 
331
397
  if (pending.length === 0) {
332
398
  console.log(`${langCode}: nothing to do (${Object.keys(tm).length} in memory)`);
399
+ // Nothing pending means nothing was dropped, so the memory on disk already matches
400
+ // this glossary — safe to record the fingerprint without a translation pass. The
401
+ // fingerprint is never written ahead of a TM flush: if a run dies mid-way, the next
402
+ // one must still see a stale fingerprint and re-drop the affected units.
403
+ writeFileSync(metaFile, JSON.stringify({ ...meta, glossary: fingerprint }, null, 2));
333
404
  return;
334
405
  }
335
406
 
@@ -352,6 +423,7 @@ async function translateLang(langCode, units) {
352
423
  let sinceFlush = 0;
353
424
  const flush = () => {
354
425
  writeFileSync(tmFile, JSON.stringify(tm, null, 2));
426
+ writeFileSync(metaFile, JSON.stringify({ ...meta, glossary: fingerprint }, null, 2));
355
427
  sinceFlush = 0;
356
428
  };
357
429
 
@@ -362,7 +434,12 @@ async function translateLang(langCode, units) {
362
434
  if (myIndex >= batches.length) return;
363
435
  const batch = batches[myIndex];
364
436
 
365
- const items = batch.map(([, unit], i) => ({ id: i, text: unit.text }));
437
+ // `el` is omitted for ordinary prose rather than sent as null: a site of nothing but
438
+ // paragraphs then produces a byte-identical payload to 1.x and costs not one extra token.
439
+ const items = batch.map(([, unit], i) => {
440
+ const el = roleOf(unit);
441
+ return el ? { id: i, text: unit.text, el } : { id: i, text: unit.text };
442
+ });
366
443
 
367
444
  let rows;
368
445
  let usage;
@@ -391,7 +468,10 @@ async function translateLang(langCode, units) {
391
468
 
392
469
  for (const [hash, unit, problem] of retry) {
393
470
  try {
394
- const solo = await callModel(langCode, [{ id: 0, text: unit.text }]);
471
+ const soloEl = roleOf(unit);
472
+ const solo = await callModel(langCode, [
473
+ soloEl ? { id: 0, text: unit.text, el: soloEl } : { id: 0, text: unit.text },
474
+ ]);
395
475
  const out = solo.rows.find((r) => r.id === 0)?.text;
396
476
  inTok += solo.usage.inTok;
397
477
  outTok += solo.usage.outTok;