claude-translator 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +14 -0
- package/CHANGELOG.md +180 -0
- package/PRIVACY.md +71 -0
- package/README.md +241 -39
- package/bin/claude-translator.mjs +11 -2
- package/bin/cli.test.mjs +20 -1
- package/glossary.example.json +23 -0
- package/i18n.config.example.json +7 -0
- package/package.json +9 -5
- package/scripts/audit-seo.mjs +6 -3
- package/scripts/build-locales.mjs +55 -6
- package/scripts/config.mjs +66 -0
- package/scripts/credit.mjs +12 -5
- package/scripts/extract.mjs +21 -6
- package/scripts/format-locale.mjs +290 -0
- package/scripts/format-locale.test.mjs +171 -0
- package/scripts/glossary.mjs +229 -0
- package/scripts/glossary.test.mjs +188 -0
- package/scripts/roles.mjs +142 -0
- package/scripts/roles.test.mjs +140 -0
- package/scripts/tqa-score.mjs +127 -0
- package/scripts/tqa-score.test.mjs +144 -0
- package/scripts/tqa.mjs +449 -0
- package/scripts/translate.mjs +87 -7
- package/scripts/verify.mjs +113 -5
- package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
- package/{references → skills/translate-site/references}/quality-review.md +28 -0
- /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
- /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
- /package/{references → skills/translate-site/references}/providers.md +0 -0
- /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
package/scripts/tqa.mjs
ADDED
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* i18n step 6 — translation quality assessment.
|
|
4
|
+
*
|
|
5
|
+
* node scripts/i18n/tqa.mjs --lang es
|
|
6
|
+
* node scripts/i18n/tqa.mjs --lang es,fr --sample 200
|
|
7
|
+
* node scripts/i18n/tqa.mjs --lang es --judge-provider openai --judge-model gpt-5.6-luna
|
|
8
|
+
* node scripts/i18n/tqa.mjs --lang es --dry # cost estimate only
|
|
9
|
+
* node scripts/i18n/tqa.mjs --lang es --repeat # measure the judge's own variance
|
|
10
|
+
*
|
|
11
|
+
* Reads i18n/source.json, i18n/tm/{lang}.json
|
|
12
|
+
* Writes i18n/tqa/{lang}.json and i18n/tqa/{lang}.md
|
|
13
|
+
*
|
|
14
|
+
* ── What this is ─────────────────────────────────────────────────────────────
|
|
15
|
+
* An MQM-style error-typology assessment of a stratified sample, scored per 100 words.
|
|
16
|
+
* MQM is the standard the industry actually uses, so the output is comparable with a
|
|
17
|
+
* human review rather than being a number invented here.
|
|
18
|
+
*
|
|
19
|
+
* score = 100 - (weighted error points / words) * 100
|
|
20
|
+
*
|
|
21
|
+
* ── What this is NOT ─────────────────────────────────────────────────────────
|
|
22
|
+
* It is not a human certification, and nothing here should be reported as one. It is one
|
|
23
|
+
* model's opinion of another model's output, and models are known to prefer their own
|
|
24
|
+
* work — which is why the judge defaults to a DIFFERENT provider than the translator
|
|
25
|
+
* whenever one is configured, and why --repeat exists to show how much the judge moves
|
|
26
|
+
* between identical runs. A quality number without its variance is marketing.
|
|
27
|
+
*
|
|
28
|
+
* The honest reading: this reliably finds the bottom of the distribution — the units
|
|
29
|
+
* that are actually wrong — and should not be trusted to the second decimal place.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'fs';
|
|
33
|
+
import { join } from 'path';
|
|
34
|
+
|
|
35
|
+
import {
|
|
36
|
+
SOURCE_FILE, TM_DIR, I18N_DIR, LOCALES, GLOSSARY, ROOT_DIR as ROOT,
|
|
37
|
+
SITE_NAME, SITE_DESCRIPTION, SOURCE_LANGUAGE, MODEL as CFG_MODEL,
|
|
38
|
+
PROVIDER as CFG_PROVIDER, API_BASE_URL, API_KEY_ENV, JSON_MODE, PRICING,
|
|
39
|
+
} from './config.mjs';
|
|
40
|
+
import { creditBlock } from './credit.mjs';
|
|
41
|
+
import { loadProvider } from './providers/index.mjs';
|
|
42
|
+
import { termsForBatch } from './glossary.mjs';
|
|
43
|
+
import { WEIGHT, CATEGORIES, rng, words, stratifiedSample, parseErrors, score } from './tqa-score.mjs';
|
|
44
|
+
|
|
45
|
+
// ── CLI ──────────────────────────────────────────────────────────────────────
|
|
46
|
+
|
|
47
|
+
const args = Object.fromEntries(
|
|
48
|
+
process.argv
|
|
49
|
+
.slice(2)
|
|
50
|
+
.join(' ')
|
|
51
|
+
.split('--')
|
|
52
|
+
.filter(Boolean)
|
|
53
|
+
.map((s) => s.trim().split(/\s+/))
|
|
54
|
+
.map(([k, ...v]) => [k, v.join(' ') || true])
|
|
55
|
+
);
|
|
56
|
+
|
|
57
|
+
const LANGS = String(args.lang ?? '').split(',').map((s) => s.trim()).filter(Boolean);
|
|
58
|
+
const SAMPLE = Number(args.sample ?? 100);
|
|
59
|
+
const SEED = Number(args.seed ?? 20260829);
|
|
60
|
+
const BATCH = Number(args.batch ?? 10);
|
|
61
|
+
const CONCURRENCY = Number(args.concurrency ?? 6);
|
|
62
|
+
const DRY = Boolean(args.dry);
|
|
63
|
+
const REPEAT = Boolean(args.repeat);
|
|
64
|
+
|
|
65
|
+
if (LANGS.length === 0) {
|
|
66
|
+
console.error(
|
|
67
|
+
'Usage: node scripts/i18n/tqa.mjs --lang es[,fr] [--sample N] [--seed N] [--dry] [--repeat]\n' +
|
|
68
|
+
' [--judge-provider P] [--judge-model M]'
|
|
69
|
+
);
|
|
70
|
+
process.exit(1);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// ── MQM ──────────────────────────────────────────────────────────────────────
|
|
74
|
+
|
|
75
|
+
// ── Judge ────────────────────────────────────────────────────────────────────
|
|
76
|
+
|
|
77
|
+
const JUDGE_PROVIDER_NAME = args['judge-provider'] ? String(args['judge-provider']) : null;
|
|
78
|
+
const JUDGE_MODEL_ARG = args['judge-model'] ? String(args['judge-model']) : null;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Pick a judge that is not the translator, when we can.
|
|
82
|
+
*
|
|
83
|
+
* Self-preference is a documented failure of LLM-as-judge setups: a model scores its own
|
|
84
|
+
* output higher than a peer's. Defaulting the judge to a different provider costs nothing
|
|
85
|
+
* and removes the most obvious objection to the number. If only one provider is
|
|
86
|
+
* configured, we use it and SAY SO in the report rather than hiding it.
|
|
87
|
+
*/
|
|
88
|
+
function pickJudge() {
|
|
89
|
+
if (JUDGE_PROVIDER_NAME) return { provider: JUDGE_PROVIDER_NAME, reason: 'chosen with --judge-provider' };
|
|
90
|
+
const translator = CFG_PROVIDER ?? 'anthropic';
|
|
91
|
+
const alternatives = ['anthropic', 'gemini', 'openai'].filter((p) => p !== translator);
|
|
92
|
+
for (const alt of alternatives) {
|
|
93
|
+
const keys = { anthropic: 'ANTHROPIC_API_KEY', gemini: 'GEMINI_API_KEY', openai: 'OPENAI_API_KEY' }[alt];
|
|
94
|
+
if (process.env[keys]) return { provider: alt, reason: `differs from the translator (${translator})` };
|
|
95
|
+
}
|
|
96
|
+
return { provider: translator, reason: `SAME as the translator — no other provider key found` };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function judgePrompt(langName, langCode, terms) {
|
|
100
|
+
const glossaryLines = terms.length
|
|
101
|
+
? [
|
|
102
|
+
`The site glossary constrains these terms:`,
|
|
103
|
+
...terms.map((t) =>
|
|
104
|
+
t.rule === 'keep'
|
|
105
|
+
? ` - "${t.source}" must appear unchanged`
|
|
106
|
+
: ` - "${t.source}" must be rendered as "${t.targets[langCode]}"`
|
|
107
|
+
),
|
|
108
|
+
]
|
|
109
|
+
: [];
|
|
110
|
+
|
|
111
|
+
return [
|
|
112
|
+
`You are a senior localization reviewer assessing ${langName} translations of ${SITE_NAME}${SITE_DESCRIPTION ? `, ${SITE_DESCRIPTION}` : ''}.`,
|
|
113
|
+
`The source language is ${SOURCE_LANGUAGE}. This is marketing and product copy for a website.`,
|
|
114
|
+
``,
|
|
115
|
+
`For each item, list the translation errors using MQM categories and severities.`,
|
|
116
|
+
``,
|
|
117
|
+
`CATEGORIES: ${CATEGORIES.join(', ')}`,
|
|
118
|
+
`SEVERITIES: minor (noticeable, does not mislead), major (misleads or reads as wrong),`,
|
|
119
|
+
` critical (reverses meaning, breaks a promise, or is unusable)`,
|
|
120
|
+
``,
|
|
121
|
+
...glossaryLines,
|
|
122
|
+
``,
|
|
123
|
+
`RULES`,
|
|
124
|
+
`1. <0>, </0>, <1/> are markup placeholders. They must appear in the translation with`,
|
|
125
|
+
` the same count and numbers. Report any difference as markup/placeholder, critical.`,
|
|
126
|
+
`2. Numbers, prices and measurements must keep their VALUE. Different digit grouping or`,
|
|
127
|
+
` symbol placement is correct localization, not an error. A changed value is critical.`,
|
|
128
|
+
`3. A term left in the source language is only an error if it should have been`,
|
|
129
|
+
` translated. Brand names, formats and protocol names are correct unchanged.`,
|
|
130
|
+
`4. Judge the translation on its own terms as ${langName} copy. Do not reward literalness.`,
|
|
131
|
+
`5. Report NO errors when there are none. An empty list is the expected result for`,
|
|
132
|
+
` most items, and inventing a minor error to look thorough makes the whole score useless.`,
|
|
133
|
+
``,
|
|
134
|
+
`Return a JSON array. For each input item return { "id": <same id>, "text": "<errors>" }`,
|
|
135
|
+
`where <errors> is a JSON array serialised as a string, e.g.`,
|
|
136
|
+
`"[{\\"c\\":\\"fluency/grammar\\",\\"s\\":\\"minor\\",\\"n\\":\\"wrong gender agreement\\"}]"`,
|
|
137
|
+
`or "[]" when the translation is correct.`,
|
|
138
|
+
]
|
|
139
|
+
.filter((l) => l !== null)
|
|
140
|
+
.join('\n');
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// ── API ──────────────────────────────────────────────────────────────────────
|
|
144
|
+
|
|
145
|
+
const JUDGE = pickJudge();
|
|
146
|
+
const PROVIDER = await loadProvider({ provider: JUDGE.provider, model: JUDGE_MODEL_ARG, root: ROOT });
|
|
147
|
+
const MODEL = JUDGE_MODEL_ARG ?? PROVIDER.defaultModel;
|
|
148
|
+
|
|
149
|
+
function loadKey() {
|
|
150
|
+
const names = API_KEY_ENV ? [API_KEY_ENV] : (PROVIDER.envKeys ?? []);
|
|
151
|
+
for (const name of names) if (process.env[name]) return process.env[name];
|
|
152
|
+
const envFile = join(ROOT, '.env');
|
|
153
|
+
if (existsSync(envFile)) {
|
|
154
|
+
const text = readFileSync(envFile, 'utf8');
|
|
155
|
+
for (const name of names) {
|
|
156
|
+
const m = new RegExp(`^${name}=(.*)$`, 'm').exec(text);
|
|
157
|
+
if (m) return m[1].trim();
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
if (PROVIDER.keyOptional) return null;
|
|
161
|
+
console.error(`No API key for ${PROVIDER.label ?? PROVIDER.id}. Set ${names.join(' or ')}.`);
|
|
162
|
+
process.exit(1);
|
|
163
|
+
}
|
|
164
|
+
const KEY = DRY ? null : loadKey();
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* One judging request. Reuses the translator's retry and batch-splitting behaviour by
|
|
168
|
+
* shape rather than by import: the same safety-block and truncation failures apply to a
|
|
169
|
+
* review pass, and a batch that trips one is halved rather than lost.
|
|
170
|
+
*/
|
|
171
|
+
async function callJudge(system, items, attempt = 1, jsonMode = JSON_MODE) {
|
|
172
|
+
const { url, headers, body } = PROVIDER.request({
|
|
173
|
+
model: MODEL, system, items, temperature: 0, key: KEY, baseUrl: API_BASE_URL, jsonMode,
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
const split = async (why) => {
|
|
177
|
+
if (items.length === 1) return { rows: [], usage: { inTok: 0, outTok: 0 } };
|
|
178
|
+
const mid = Math.ceil(items.length / 2);
|
|
179
|
+
process.stderr.write(` ${why} on ${items.length} items — splitting\n`);
|
|
180
|
+
const [a, b] = await Promise.all([
|
|
181
|
+
callJudge(system, items.slice(0, mid), 1, jsonMode),
|
|
182
|
+
callJudge(system, items.slice(mid), 1, jsonMode),
|
|
183
|
+
]);
|
|
184
|
+
return {
|
|
185
|
+
rows: [...a.rows, ...b.rows],
|
|
186
|
+
usage: { inTok: a.usage.inTok + b.usage.inTok, outTok: a.usage.outTok + b.usage.outTok },
|
|
187
|
+
};
|
|
188
|
+
};
|
|
189
|
+
|
|
190
|
+
let res;
|
|
191
|
+
try {
|
|
192
|
+
res = await fetch(url, { method: 'POST', headers, body: JSON.stringify(body) });
|
|
193
|
+
} catch (err) {
|
|
194
|
+
if (attempt <= 5) {
|
|
195
|
+
await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
|
|
196
|
+
return callJudge(system, items, attempt + 1, jsonMode);
|
|
197
|
+
}
|
|
198
|
+
throw err;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
if (!res.ok) {
|
|
202
|
+
const text = await res.text();
|
|
203
|
+
if (PROVIDER.unsupportedJsonMode?.(res.status, text) && (jsonMode ?? 'schema') !== 'none') {
|
|
204
|
+
return callJudge(system, items, attempt, (jsonMode ?? 'schema') === 'schema' ? 'object' : 'none');
|
|
205
|
+
}
|
|
206
|
+
if ((res.status === 429 || res.status >= 500) && attempt <= 5) {
|
|
207
|
+
await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
|
|
208
|
+
return callJudge(system, items, attempt + 1, jsonMode);
|
|
209
|
+
}
|
|
210
|
+
throw new Error(`${PROVIDER.label ?? PROVIDER.id} HTTP ${res.status}: ${text.slice(0, 300)}`);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const data = await res.json();
|
|
214
|
+
const { text: part, usage, retryable, detail } = PROVIDER.parse(data);
|
|
215
|
+
if (retryable === 'safety') return split(detail ?? 'safety block');
|
|
216
|
+
if (retryable === 'truncated') return split('truncated');
|
|
217
|
+
|
|
218
|
+
try {
|
|
219
|
+
const rows = PROVIDER.unwrap ? PROVIDER.unwrap(JSON.parse(part)) : JSON.parse(part);
|
|
220
|
+
return { rows: Array.isArray(rows) ? rows : [], usage };
|
|
221
|
+
} catch {
|
|
222
|
+
return split('unparseable response');
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
// ── Run ──────────────────────────────────────────────────────────────────────
|
|
227
|
+
|
|
228
|
+
const LANG_NAMES = Object.fromEntries(LOCALES.map((l) => [l.pathCode, l.nativeLabel ?? l.pathCode]));
|
|
229
|
+
|
|
230
|
+
if (!existsSync(SOURCE_FILE)) {
|
|
231
|
+
console.error(`No ${SOURCE_FILE}. Run extract.mjs first.`);
|
|
232
|
+
process.exit(1);
|
|
233
|
+
}
|
|
234
|
+
const SOURCE = JSON.parse(readFileSync(SOURCE_FILE, 'utf8'));
|
|
235
|
+
|
|
236
|
+
console.log(`judge: ${PROVIDER.label ?? PROVIDER.id} · model: ${MODEL}`);
|
|
237
|
+
console.log(` ${JUDGE.reason}`);
|
|
238
|
+
if (JUDGE.reason.startsWith('SAME')) {
|
|
239
|
+
console.log(' ⚠ a model judging its own output scores it generously — treat the number as a floor, not a grade');
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
mkdirSync(join(I18N_DIR, 'tqa'), { recursive: true });
|
|
243
|
+
|
|
244
|
+
async function assess(langCode, seed) {
|
|
245
|
+
const tmFile = join(TM_DIR, `${langCode}.json`);
|
|
246
|
+
if (!existsSync(tmFile)) {
|
|
247
|
+
console.log(`${langCode}: no memory — skipped`);
|
|
248
|
+
return null;
|
|
249
|
+
}
|
|
250
|
+
const tm = JSON.parse(readFileSync(tmFile, 'utf8'));
|
|
251
|
+
|
|
252
|
+
const pool = Object.entries(SOURCE)
|
|
253
|
+
.filter(([hash]) => typeof tm[hash] === 'string')
|
|
254
|
+
.map(([hash, unit]) => ({ hash, source: unit.text, target: tm[hash], count: unit.count ?? 1 }));
|
|
255
|
+
|
|
256
|
+
if (!pool.length) {
|
|
257
|
+
console.log(`${langCode}: nothing translated yet — skipped`);
|
|
258
|
+
return null;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
const sample = stratifiedSample(pool, SAMPLE, seed).map((u) => ({ ...u, words: words(u.source) }));
|
|
262
|
+
const sampledWords = sample.reduce((n, u) => n + u.words, 0);
|
|
263
|
+
|
|
264
|
+
console.log(
|
|
265
|
+
`\n${langCode}: assessing ${sample.length} of ${pool.length.toLocaleString()} units ` +
|
|
266
|
+
`(${sampledWords.toLocaleString()} source words, seed ${seed})`
|
|
267
|
+
);
|
|
268
|
+
|
|
269
|
+
if (DRY) {
|
|
270
|
+
const estIn = Math.ceil(sampledWords * 3.2) + sample.length * 40;
|
|
271
|
+
const estOut = sample.length * 25;
|
|
272
|
+
console.log(` estimated ~${estIn.toLocaleString()} in / ~${estOut.toLocaleString()} out tokens`);
|
|
273
|
+
const rate = PRICING ?? PROVIDER.pricing?.(MODEL) ?? null;
|
|
274
|
+
if (rate) {
|
|
275
|
+
const cost = (estIn / 1e6) * rate[0] + (estOut / 1e6) * rate[1];
|
|
276
|
+
console.log(` estimated cost: $${cost.toFixed(4)}`);
|
|
277
|
+
} else {
|
|
278
|
+
console.log(' no price known for this endpoint — set "pricing" in i18n.config.json for a figure');
|
|
279
|
+
}
|
|
280
|
+
return null;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
const langName = LANG_NAMES[langCode] ?? langCode;
|
|
284
|
+
const batches = [];
|
|
285
|
+
for (let i = 0; i < sample.length; i += BATCH) batches.push(sample.slice(i, i + BATCH));
|
|
286
|
+
|
|
287
|
+
const assessed = [];
|
|
288
|
+
let unreadable = 0;
|
|
289
|
+
const usage = { inTok: 0, outTok: 0 };
|
|
290
|
+
let cursor = 0;
|
|
291
|
+
let done = 0;
|
|
292
|
+
|
|
293
|
+
const runOne = async () => {
|
|
294
|
+
for (;;) {
|
|
295
|
+
const idx = cursor++;
|
|
296
|
+
if (idx >= batches.length) return;
|
|
297
|
+
const batch = batches[idx];
|
|
298
|
+
const terms = termsForBatch(GLOSSARY, batch.map((u) => u.source), langCode);
|
|
299
|
+
const system = judgePrompt(langName, langCode, terms);
|
|
300
|
+
const items = batch.map((u, i) => ({ id: i, text: `SOURCE: ${u.source}\nTARGET: ${u.target}` }));
|
|
301
|
+
|
|
302
|
+
let rows = [];
|
|
303
|
+
try {
|
|
304
|
+
const out = await callJudge(system, items);
|
|
305
|
+
rows = out.rows;
|
|
306
|
+
usage.inTok += out.usage.inTok;
|
|
307
|
+
usage.outTok += out.usage.outTok;
|
|
308
|
+
} catch (err) {
|
|
309
|
+
process.stderr.write(` batch ${idx} failed: ${String(err.message ?? err).slice(0, 90)}\n`);
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
const byId = new Map(rows.map((r) => [Number(r.id), r.text]));
|
|
313
|
+
for (let i = 0; i < batch.length; i++) {
|
|
314
|
+
const errors = parseErrors(byId.get(i));
|
|
315
|
+
// A unit the judge did not return, or returned unreadably, is EXCLUDED from the
|
|
316
|
+
// score rather than counted as clean. Counting it clean would inflate the result
|
|
317
|
+
// every time the judge failed, which is precisely backwards.
|
|
318
|
+
if (errors === null) {
|
|
319
|
+
unreadable++;
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
assessed.push({ ...batch[i], errors });
|
|
323
|
+
}
|
|
324
|
+
done += batch.length;
|
|
325
|
+
process.stdout.write(`\r ${langCode}: ${done}/${sample.length} units judged`);
|
|
326
|
+
}
|
|
327
|
+
};
|
|
328
|
+
|
|
329
|
+
await Promise.all(Array.from({ length: Math.min(CONCURRENCY, batches.length) }, runOne));
|
|
330
|
+
process.stdout.write('\n');
|
|
331
|
+
|
|
332
|
+
// A score computed from nothing is not a good score, it is a broken measurement — and
|
|
333
|
+
// it would read as a perfect one. Every unit here failed to come back readable at least
|
|
334
|
+
// once during development (a 401 against the wrong endpoint), and the run cheerfully
|
|
335
|
+
// printed "100.00 / 100" from zero assessed units. Refuse instead.
|
|
336
|
+
const judgedShare = sample.length ? (assessed.length * 100) / sample.length : 0;
|
|
337
|
+
if (assessed.length === 0) {
|
|
338
|
+
console.log(
|
|
339
|
+
` \u2717 no unit could be assessed \u2014 the judge returned nothing readable for all ` +
|
|
340
|
+
`${sample.length} of them. No score is reported, because a score from an empty sample ` +
|
|
341
|
+
`would read as a perfect one.`
|
|
342
|
+
);
|
|
343
|
+
return null;
|
|
344
|
+
}
|
|
345
|
+
if (judgedShare < 80) {
|
|
346
|
+
console.log(
|
|
347
|
+
` \u26a0 only ${judgedShare.toFixed(0)}% of the sample was assessed ` +
|
|
348
|
+
`(${unreadable} unit(s) unreadable). The score below is computed from what came back ` +
|
|
349
|
+
`and should not be compared with a clean run.`
|
|
350
|
+
);
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
const result = score(assessed);
|
|
354
|
+
result.lang = langCode;
|
|
355
|
+
result.seed = seed;
|
|
356
|
+
result.sampleRequested = SAMPLE;
|
|
357
|
+
result.poolSize = pool.length;
|
|
358
|
+
result.unreadable = unreadable;
|
|
359
|
+
result.judge = { provider: PROVIDER.id, model: MODEL, note: JUDGE.reason };
|
|
360
|
+
result.usage = usage;
|
|
361
|
+
result.worst = assessed
|
|
362
|
+
.filter((a) => a.errors.length)
|
|
363
|
+
.sort((a, b) => {
|
|
364
|
+
const w = (x) => x.errors.reduce((n, e) => n + WEIGHT[e.severity], 0);
|
|
365
|
+
return w(b) - w(a);
|
|
366
|
+
})
|
|
367
|
+
.slice(0, 15)
|
|
368
|
+
.map((a) => ({ source: a.source, target: a.target, errors: a.errors }));
|
|
369
|
+
|
|
370
|
+
return result;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
const reports = [];
|
|
374
|
+
for (const lang of LANGS) {
|
|
375
|
+
const first = await assess(lang, SEED);
|
|
376
|
+
if (!first) continue;
|
|
377
|
+
|
|
378
|
+
if (REPEAT) {
|
|
379
|
+
// The same sample, judged again. Any gap between the two is the judge's own noise,
|
|
380
|
+
// and reporting a score without it invites the reader to over-read a decimal place.
|
|
381
|
+
console.log(` re-judging the same sample to measure judge variance…`);
|
|
382
|
+
const second = await assess(lang, SEED);
|
|
383
|
+
if (second) {
|
|
384
|
+
first.variance = {
|
|
385
|
+
secondRunMqm: second.mqm,
|
|
386
|
+
delta: Math.round((second.mqm - first.mqm) * 100) / 100,
|
|
387
|
+
};
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
writeFileSync(join(I18N_DIR, 'tqa', `${lang}.json`), JSON.stringify(first, null, 2));
|
|
392
|
+
reports.push(first);
|
|
393
|
+
|
|
394
|
+
const pctClean = first.unitsAssessed ? (first.unitsClean * 100) / first.unitsAssessed : 0;
|
|
395
|
+
console.log(` MQM score: ${first.mqm.toFixed(2)} / 100`);
|
|
396
|
+
console.log(
|
|
397
|
+
` ${first.unitsClean}/${first.unitsAssessed} units error-free (${pctClean.toFixed(0)}%) · ` +
|
|
398
|
+
`${first.bySeverity.critical} critical, ${first.bySeverity.major} major, ${first.bySeverity.minor} minor`
|
|
399
|
+
);
|
|
400
|
+
if (first.unreadable) console.log(` ${first.unreadable} unit(s) excluded — judge returned nothing readable`);
|
|
401
|
+
if (first.variance) {
|
|
402
|
+
console.log(` re-run of the same sample scored ${first.variance.secondRunMqm.toFixed(2)} (Δ ${first.variance.delta >= 0 ? '+' : ''}${first.variance.delta})`);
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
// ── Markdown scorecard ───────────────────────────────────────────────────────
|
|
407
|
+
|
|
408
|
+
if (reports.length) {
|
|
409
|
+
const lines = [
|
|
410
|
+
`# Translation quality assessment — ${SITE_NAME}`,
|
|
411
|
+
``,
|
|
412
|
+
`Generated ${new Date().toISOString().slice(0, 10)} by \`tqa.mjs\`.`,
|
|
413
|
+
``,
|
|
414
|
+
`**Method.** MQM error typology on a stratified sample, weighted toward the strings that`,
|
|
415
|
+
`appear most often on the site. Score = 100 − (weighted error points ÷ words) × 100, with`,
|
|
416
|
+
`minor = 1, major = 5, critical = 10.`,
|
|
417
|
+
``,
|
|
418
|
+
`**This is a machine assessment, not a human certification.** The judge is`,
|
|
419
|
+
`${PROVIDER.label ?? PROVIDER.id} (\`${MODEL}\`) — ${JUDGE.reason}. Read the score as a`,
|
|
420
|
+
`comparison between locales and across runs, not as an absolute grade.`,
|
|
421
|
+
``,
|
|
422
|
+
`| Locale | MQM | Units | Error-free | Critical | Major | Minor |`,
|
|
423
|
+
`| --- | ---: | ---: | ---: | ---: | ---: | ---: |`,
|
|
424
|
+
...reports.map((r) => {
|
|
425
|
+
const pct = r.unitsAssessed ? Math.round((r.unitsClean * 100) / r.unitsAssessed) : 0;
|
|
426
|
+
return `| ${r.lang} | **${r.mqm.toFixed(2)}** | ${r.unitsAssessed} | ${pct}% | ${r.bySeverity.critical} | ${r.bySeverity.major} | ${r.bySeverity.minor} |`;
|
|
427
|
+
}),
|
|
428
|
+
``,
|
|
429
|
+
];
|
|
430
|
+
|
|
431
|
+
for (const r of reports) {
|
|
432
|
+
if (!r.worst.length) continue;
|
|
433
|
+
lines.push(`## ${r.lang} — worst-scoring units`, ``);
|
|
434
|
+
for (const w of r.worst.slice(0, 8)) {
|
|
435
|
+
lines.push(
|
|
436
|
+
`- **${w.errors.map((e) => `${e.severity} ${e.category}`).join(', ')}**`,
|
|
437
|
+
` - source: \`${w.source.slice(0, 140).replace(/`/g, "'")}\``,
|
|
438
|
+
` - target: \`${w.target.slice(0, 140).replace(/`/g, "'")}\``,
|
|
439
|
+
...(w.errors[0]?.note ? [` - note: ${w.errors[0].note}`] : [])
|
|
440
|
+
);
|
|
441
|
+
}
|
|
442
|
+
lines.push(``);
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
const md = join(I18N_DIR, 'tqa', 'scorecard.md');
|
|
446
|
+
writeFileSync(md, lines.join('\n'));
|
|
447
|
+
console.log(`\nwrote ${md}`);
|
|
448
|
+
creditBlock([`${reports.length} locale(s) assessed · MQM sample of ${SAMPLE} units each`]);
|
|
449
|
+
}
|
package/scripts/translate.mjs
CHANGED
|
@@ -27,10 +27,12 @@ import { join } from 'path';
|
|
|
27
27
|
import { fileURLToPath } from 'url';
|
|
28
28
|
|
|
29
29
|
import {
|
|
30
|
-
SOURCE_FILE as SRC_FILE, TM_DIR, LOCALES, RTL, DNT, MODEL as CFG_MODEL,
|
|
30
|
+
SOURCE_FILE as SRC_FILE, TM_DIR, LOCALES, RTL, DNT, GLOSSARY, MODEL as CFG_MODEL,
|
|
31
31
|
ROOT_DIR as ROOT, SITE_NAME, SITE_DESCRIPTION, SOURCE_LANGUAGE,
|
|
32
32
|
PROVIDER as CFG_PROVIDER, API_BASE_URL, API_KEY_ENV, JSON_MODE, PRICING,
|
|
33
33
|
} from './config.mjs';
|
|
34
|
+
import { termsForBatch, glossaryPrompt, glossaryFingerprint, containsTerm } from './glossary.mjs';
|
|
35
|
+
import { roleOf, rolePrompt } from './roles.mjs';
|
|
34
36
|
import { hint, link } from './credit.mjs';
|
|
35
37
|
import { loadProvider, extractJson } from './providers/index.mjs';
|
|
36
38
|
|
|
@@ -115,10 +117,18 @@ const TECH_TOKENS = [
|
|
|
115
117
|
'OCR', 'GDPR', 'SSL', 'TLS', 'API', 'SDK', 'URL', 'HTTP', 'HTTPS', 'SEO', 'CSS', 'RSS',
|
|
116
118
|
];
|
|
117
119
|
|
|
118
|
-
|
|
120
|
+
/**
|
|
121
|
+
* The flat never-translate list that still goes into rule 2. The structured glossary
|
|
122
|
+
* (config.GLOSSARY) carries the same brands plus per-locale targets and is injected
|
|
123
|
+
* per batch; this line stays because format and protocol tokens are correct unchanged
|
|
124
|
+
* in every language and are cheap to state once.
|
|
125
|
+
*/
|
|
126
|
+
const DNT_NAMES = [...new Set([...DNT.brands, ...DNT.formats, ...TECH_TOKENS])];
|
|
119
127
|
|
|
120
|
-
function systemPrompt(langCode) {
|
|
128
|
+
function systemPrompt(langCode, batchTerms = [], batchRoles = []) {
|
|
121
129
|
const name = LANG_NAMES[langCode] ?? langCode;
|
|
130
|
+
const terminology = glossaryPrompt(batchTerms, langCode);
|
|
131
|
+
const context = rolePrompt(batchRoles);
|
|
122
132
|
return [
|
|
123
133
|
`You are a professional translator localising the website of ${SITE_NAME}${SITE_DESCRIPTION ? `, ${SITE_DESCRIPTION}` : ''}.`,
|
|
124
134
|
`Translate from ${SOURCE_LANGUAGE} into ${name} (${langCode}).`,
|
|
@@ -127,7 +137,9 @@ function systemPrompt(langCode) {
|
|
|
127
137
|
`1. Preserve every placeholder EXACTLY: <0>, </0>, <1/> and so on. Same count, same numbers.`,
|
|
128
138
|
` Placeholders wrap inline markup — move them so they wrap the equivalent words in your`,
|
|
129
139
|
` translation, but never drop, add, renumber or reorder their nesting.`,
|
|
130
|
-
`2. Never translate these names: ${
|
|
140
|
+
`2. Never translate these names: ${DNT_NAMES.join(', ')}.`,
|
|
141
|
+
` Match whole words, and respect capitalisation: a lowercase common noun that`,
|
|
142
|
+
` happens to spell a brand name is the common noun, and must be translated.`,
|
|
131
143
|
`3. This is marketing and product copy. Translate meaning and tone, not word for word.`,
|
|
132
144
|
` Keep it natural and idiomatic for a native reader.`,
|
|
133
145
|
`4. Keep numbers, prices, file sizes and counts unchanged (e.g. "120+", "1 GB", "10 MB").`,
|
|
@@ -135,6 +147,8 @@ function systemPrompt(langCode) {
|
|
|
135
147
|
` target language allows it. Headings stay headings; button labels stay short.`,
|
|
136
148
|
`6. Do not add explanations, notes or quotes around the result.`,
|
|
137
149
|
RTL.has(langCode) ? `7. ${name} is right-to-left. Write natural RTL text; do not insert directional marks.` : ``,
|
|
150
|
+
terminology,
|
|
151
|
+
context,
|
|
138
152
|
``,
|
|
139
153
|
`Return a JSON array. For each input item return { "id": <same id>, "text": "<translation>" }.`,
|
|
140
154
|
]
|
|
@@ -181,9 +195,14 @@ function validate(source, translated) {
|
|
|
181
195
|
* own field names.
|
|
182
196
|
*/
|
|
183
197
|
async function callModel(langCode, items, attempt = 1, jsonMode = JSON_MODE, drop = new Set()) {
|
|
198
|
+
// Recomputed per call, not per run, so a batch that gets halved on a safety block or a
|
|
199
|
+
// truncation carries exactly the terms its own half contains.
|
|
200
|
+
const batchTerms = termsForBatch(GLOSSARY, items.map((i) => i.text), langCode);
|
|
201
|
+
const batchRoles = items.map((i) => i.el).filter(Boolean);
|
|
202
|
+
|
|
184
203
|
const { url, headers, body } = PROVIDER.request({
|
|
185
204
|
model: MODEL,
|
|
186
|
-
system: systemPrompt(langCode),
|
|
205
|
+
system: systemPrompt(langCode, batchTerms, batchRoles),
|
|
187
206
|
items,
|
|
188
207
|
temperature: attempt === 1 ? 0.2 : 0.4,
|
|
189
208
|
key: KEY,
|
|
@@ -308,6 +327,53 @@ async function translateLang(langCode, units) {
|
|
|
308
327
|
const tmFile = join(TM_DIR, `${langCode}${TAG}.json`);
|
|
309
328
|
const tm = existsSync(tmFile) ? JSON.parse(readFileSync(tmFile, 'utf8')) : {};
|
|
310
329
|
|
|
330
|
+
// ── Glossary invalidation ──────────────────────────────────────────────────
|
|
331
|
+
// The memory is keyed by the source hash alone, so editing a glossary target used to
|
|
332
|
+
// change nothing: every affected unit was already in the memory and got skipped, and
|
|
333
|
+
// the new terminology silently never shipped. The sidecar records the fingerprint the
|
|
334
|
+
// memory was built against; when it moves, the units containing an affected term are
|
|
335
|
+
// dropped so they re-translate. Only those — a term change must not cost a full locale.
|
|
336
|
+
//
|
|
337
|
+
// It is a SIDECAR, not a key inside the memory, because README documents
|
|
338
|
+
// i18n/tm/{lang}.json as a hand-editable hash -> string map and three other scripts
|
|
339
|
+
// iterate it. A `__meta` key would have made every coverage count off by one.
|
|
340
|
+
const metaFile = join(TM_DIR, `${langCode}${TAG}.meta.json`);
|
|
341
|
+
const meta = existsSync(metaFile) ? JSON.parse(readFileSync(metaFile, 'utf8')) : {};
|
|
342
|
+
const fingerprint = glossaryFingerprint(GLOSSARY);
|
|
343
|
+
|
|
344
|
+
if (typeof meta.glossary === 'string' && meta.glossary !== fingerprint) {
|
|
345
|
+
const before = new Set(meta.glossary.split('\n').filter(Boolean));
|
|
346
|
+
const after = new Set(fingerprint.split('\n').filter(Boolean));
|
|
347
|
+
// A term whose line is missing from either side has been added, removed or edited.
|
|
348
|
+
const changed = GLOSSARY.filter((t) => {
|
|
349
|
+
const line = glossaryFingerprint([t]);
|
|
350
|
+
return !before.has(line) || !after.has(line);
|
|
351
|
+
});
|
|
352
|
+
// Removed terms are gone from GLOSSARY, so recover them from the old fingerprint to
|
|
353
|
+
// re-translate units that were constrained by a rule the user has just deleted.
|
|
354
|
+
const removedSources = [...before]
|
|
355
|
+
.filter((line) => !after.has(line))
|
|
356
|
+
.map((line) => line.split('|')[2])
|
|
357
|
+
.filter(Boolean);
|
|
358
|
+
|
|
359
|
+
let dropped = 0;
|
|
360
|
+
for (const [hash, unit] of units) {
|
|
361
|
+
if (!(hash in tm)) continue;
|
|
362
|
+
const hit =
|
|
363
|
+
changed.some((t) => containsTerm(unit.text, t)) ||
|
|
364
|
+
removedSources.some((src) => containsTerm(unit.text, { source: src, matchCase: false }));
|
|
365
|
+
if (hit) {
|
|
366
|
+
delete tm[hash];
|
|
367
|
+
dropped++;
|
|
368
|
+
}
|
|
369
|
+
}
|
|
370
|
+
if (dropped) {
|
|
371
|
+
process.stderr.write(
|
|
372
|
+
` glossary changed — re-translating ${dropped.toLocaleString()} affected unit(s)\n`
|
|
373
|
+
);
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
|
|
311
377
|
const pending = units.filter(([hash]) => !(hash in tm));
|
|
312
378
|
|
|
313
379
|
// Re-run churn. The memory is keyed by source hash, so on an existing locale the
|
|
@@ -330,6 +396,11 @@ async function translateLang(langCode, units) {
|
|
|
330
396
|
|
|
331
397
|
if (pending.length === 0) {
|
|
332
398
|
console.log(`${langCode}: nothing to do (${Object.keys(tm).length} in memory)`);
|
|
399
|
+
// Nothing pending means nothing was dropped, so the memory on disk already matches
|
|
400
|
+
// this glossary — safe to record the fingerprint without a translation pass. The
|
|
401
|
+
// fingerprint is never written ahead of a TM flush: if a run dies mid-way, the next
|
|
402
|
+
// one must still see a stale fingerprint and re-drop the affected units.
|
|
403
|
+
writeFileSync(metaFile, JSON.stringify({ ...meta, glossary: fingerprint }, null, 2));
|
|
333
404
|
return;
|
|
334
405
|
}
|
|
335
406
|
|
|
@@ -352,6 +423,7 @@ async function translateLang(langCode, units) {
|
|
|
352
423
|
let sinceFlush = 0;
|
|
353
424
|
const flush = () => {
|
|
354
425
|
writeFileSync(tmFile, JSON.stringify(tm, null, 2));
|
|
426
|
+
writeFileSync(metaFile, JSON.stringify({ ...meta, glossary: fingerprint }, null, 2));
|
|
355
427
|
sinceFlush = 0;
|
|
356
428
|
};
|
|
357
429
|
|
|
@@ -362,7 +434,12 @@ async function translateLang(langCode, units) {
|
|
|
362
434
|
if (myIndex >= batches.length) return;
|
|
363
435
|
const batch = batches[myIndex];
|
|
364
436
|
|
|
365
|
-
|
|
437
|
+
// `el` is omitted for ordinary prose rather than sent as null: a site of nothing but
|
|
438
|
+
// paragraphs then produces a byte-identical payload to 1.x and costs not one extra token.
|
|
439
|
+
const items = batch.map(([, unit], i) => {
|
|
440
|
+
const el = roleOf(unit);
|
|
441
|
+
return el ? { id: i, text: unit.text, el } : { id: i, text: unit.text };
|
|
442
|
+
});
|
|
366
443
|
|
|
367
444
|
let rows;
|
|
368
445
|
let usage;
|
|
@@ -391,7 +468,10 @@ async function translateLang(langCode, units) {
|
|
|
391
468
|
|
|
392
469
|
for (const [hash, unit, problem] of retry) {
|
|
393
470
|
try {
|
|
394
|
-
const
|
|
471
|
+
const soloEl = roleOf(unit);
|
|
472
|
+
const solo = await callModel(langCode, [
|
|
473
|
+
soloEl ? { id: 0, text: unit.text, el: soloEl } : { id: 0, text: unit.text },
|
|
474
|
+
]);
|
|
395
475
|
const out = solo.rows.find((r) => r.id === 0)?.text;
|
|
396
476
|
inTok += solo.usage.inTok;
|
|
397
477
|
outTok += solo.usage.outTok;
|