claude-translator 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +14 -0
- package/CHANGELOG.md +249 -0
- package/PRIVACY.md +71 -0
- package/README.md +279 -44
- package/bin/claude-translator.mjs +11 -2
- package/bin/cli.test.mjs +20 -1
- package/glossary.example.json +23 -0
- package/i18n.config.example.json +7 -0
- package/package.json +9 -5
- package/scripts/audit-seo.mjs +6 -3
- package/scripts/build-locales.mjs +55 -6
- package/scripts/config.mjs +66 -0
- package/scripts/credit.mjs +12 -5
- package/scripts/extract.mjs +21 -6
- package/scripts/format-locale.mjs +290 -0
- package/scripts/format-locale.test.mjs +171 -0
- package/scripts/glossary.mjs +229 -0
- package/scripts/glossary.test.mjs +188 -0
- package/scripts/providers/openai.mjs +63 -3
- package/scripts/providers/providers.test.mjs +80 -0
- package/scripts/roles.mjs +142 -0
- package/scripts/roles.test.mjs +140 -0
- package/scripts/tqa-score.mjs +127 -0
- package/scripts/tqa-score.test.mjs +144 -0
- package/scripts/tqa.mjs +449 -0
- package/scripts/translate.mjs +104 -13
- package/scripts/verify.mjs +113 -5
- package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
- package/{references → skills/translate-site/references}/providers.md +16 -2
- package/{references → skills/translate-site/references}/quality-review.md +28 -0
- /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
- /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
- /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TQA scoring contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* A published quality score is only worth what its arithmetic is worth, so the MQM
|
|
5
|
+
* formula, the sampling determinism and the judge-failure handling are all pinned here.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { test } from 'node:test';
|
|
9
|
+
import assert from 'node:assert/strict';
|
|
10
|
+
|
|
11
|
+
import { WEIGHT, rng, words, stratifiedSample, parseErrors, score } from './tqa-score.mjs';
|
|
12
|
+
|
|
13
|
+
const unit = (n, errors = []) => ({ words: n, errors });
|
|
14
|
+
|
|
15
|
+
// ── MQM arithmetic ───────────────────────────────────────────────────────────
|
|
16
|
+
|
|
17
|
+
test('a clean sample scores 100', () => {
|
|
18
|
+
const r = score([unit(50), unit(50)]);
|
|
19
|
+
assert.equal(r.mqm, 100);
|
|
20
|
+
assert.equal(r.penalty, 0);
|
|
21
|
+
assert.equal(r.unitsClean, 2);
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
test('severity weights are the conventional MQM values', () => {
|
|
25
|
+
assert.deepEqual(WEIGHT, { minor: 1, major: 5, critical: 10 });
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
test('the score is penalty per 100 words', () => {
|
|
29
|
+
// one major (5 points) over 100 words = 5 points per 100 words = 95.
|
|
30
|
+
const r = score([unit(100, [{ category: 'accuracy/mistranslation', severity: 'major' }])]);
|
|
31
|
+
assert.equal(r.mqm, 95);
|
|
32
|
+
|
|
33
|
+
// the same error over 500 words is a fifth of the penalty density.
|
|
34
|
+
const r2 = score([unit(500, [{ category: 'accuracy/mistranslation', severity: 'major' }])]);
|
|
35
|
+
assert.equal(r2.mqm, 99);
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
test('a critical error costs ten times a minor one', () => {
|
|
39
|
+
const minor = score([unit(100, [{ category: 'fluency/grammar', severity: 'minor' }])]);
|
|
40
|
+
const critical = score([unit(100, [{ category: 'accuracy/mistranslation', severity: 'critical' }])]);
|
|
41
|
+
assert.equal(100 - minor.mqm, 1);
|
|
42
|
+
assert.equal(100 - critical.mqm, 10);
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
test('errors are counted by severity and category', () => {
|
|
46
|
+
const r = score([
|
|
47
|
+
unit(100, [
|
|
48
|
+
{ category: 'fluency/grammar', severity: 'minor' },
|
|
49
|
+
{ category: 'fluency/grammar', severity: 'major' },
|
|
50
|
+
]),
|
|
51
|
+
unit(100, [{ category: 'terminology/glossary', severity: 'critical' }]),
|
|
52
|
+
]);
|
|
53
|
+
assert.deepEqual(r.bySeverity, { minor: 1, major: 1, critical: 1 });
|
|
54
|
+
assert.deepEqual(r.byCategory, { 'fluency/grammar': 2, 'terminology/glossary': 1 });
|
|
55
|
+
assert.equal(r.unitsClean, 0);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test('an empty assessment scores 100 rather than dividing by zero', () => {
|
|
59
|
+
const r = score([]);
|
|
60
|
+
assert.equal(r.mqm, 100);
|
|
61
|
+
assert.equal(r.totalWords, 0);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
// ── Sampling ─────────────────────────────────────────────────────────────────
|
|
65
|
+
|
|
66
|
+
const pool = (n) => Array.from({ length: n }, (_, i) => ({ hash: `h${i}`, count: n - i }));
|
|
67
|
+
|
|
68
|
+
test('sampling is deterministic for a given seed', () => {
|
|
69
|
+
const a = stratifiedSample(pool(300), 50, 42).map((u) => u.hash);
|
|
70
|
+
const b = stratifiedSample(pool(300), 50, 42).map((u) => u.hash);
|
|
71
|
+
assert.deepEqual(a, b, 'the same seed must reproduce the same sample');
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test('a different seed gives a different sample', () => {
|
|
75
|
+
const a = stratifiedSample(pool(300), 50, 1).map((u) => u.hash);
|
|
76
|
+
const b = stratifiedSample(pool(300), 50, 2).map((u) => u.hash);
|
|
77
|
+
assert.notDeepEqual(a, b);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
test('a pool smaller than the sample is returned whole', () => {
|
|
81
|
+
assert.equal(stratifiedSample(pool(10), 100, 1).length, 10);
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test('the sample never exceeds the requested size', () => {
|
|
85
|
+
assert.equal(stratifiedSample(pool(1000), 100, 1).length, 100);
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
test('frequent strings are over-represented, because visitors read them more', () => {
|
|
89
|
+
// pool(300) has count descending, so the top third is the most-repeated third.
|
|
90
|
+
const picked = stratifiedSample(pool(300), 90, 7);
|
|
91
|
+
const fromTopThird = picked.filter((u) => Number(u.hash.slice(1)) < 100).length;
|
|
92
|
+
assert.ok(fromTopThird > 30, `expected the top third to be over-sampled, got ${fromTopThird}/90`);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
test('the PRNG is stable across runs', () => {
|
|
96
|
+
const a = [rng(123)(), rng(123)(), rng(123)()];
|
|
97
|
+
assert.equal(a[0], a[1]);
|
|
98
|
+
assert.equal(a[1], a[2]);
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
// ── Word counting ────────────────────────────────────────────────────────────
|
|
102
|
+
|
|
103
|
+
test('word counting ignores surrounding whitespace', () => {
|
|
104
|
+
assert.equal(words(' one two three '), 3);
|
|
105
|
+
assert.equal(words(''), 0);
|
|
106
|
+
assert.equal(words('single'), 1);
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
// ── Judge output parsing ─────────────────────────────────────────────────────
|
|
110
|
+
|
|
111
|
+
test('an empty error list parses as clean', () => {
|
|
112
|
+
assert.deepEqual(parseErrors('[]'), []);
|
|
113
|
+
assert.deepEqual(parseErrors(' '), []);
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
test('short and long field names both parse', () => {
|
|
117
|
+
const short = parseErrors('[{"c":"fluency/grammar","s":"minor","n":"agreement"}]');
|
|
118
|
+
assert.equal(short.length, 1);
|
|
119
|
+
assert.equal(short[0].category, 'fluency/grammar');
|
|
120
|
+
assert.equal(short[0].severity, 'minor');
|
|
121
|
+
|
|
122
|
+
const long = parseErrors('[{"category":"fluency/grammar","severity":"major","note":"x"}]');
|
|
123
|
+
assert.equal(long[0].severity, 'major');
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
test('an unknown severity is dropped rather than scored', () => {
|
|
127
|
+
const r = parseErrors('[{"c":"fluency/grammar","s":"catastrophic"}]');
|
|
128
|
+
assert.deepEqual(r, [], 'a severity with no weight cannot be scored, so it is not counted');
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test('unreadable judge output is null, NOT an empty list', () => {
|
|
132
|
+
// This distinction is load-bearing. Treating a failed judgement as "no errors found"
|
|
133
|
+
// would inflate the score every time the judge failed — exactly backwards.
|
|
134
|
+
assert.equal(parseErrors('not json'), null);
|
|
135
|
+
assert.equal(parseErrors('{"not":"an array"}'), null);
|
|
136
|
+
assert.equal(parseErrors(undefined), null);
|
|
137
|
+
assert.equal(parseErrors(null), null);
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
test('severity casing from the model is normalised', () => {
|
|
141
|
+
const r = parseErrors('[{"c":"fluency/grammar","s":"MAJOR"}]');
|
|
142
|
+
assert.equal(r.length, 1);
|
|
143
|
+
assert.equal(r[0].severity, 'major');
|
|
144
|
+
});
|
package/scripts/tqa.mjs
ADDED
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* i18n step 6 — translation quality assessment.
|
|
4
|
+
*
|
|
5
|
+
* node scripts/i18n/tqa.mjs --lang es
|
|
6
|
+
* node scripts/i18n/tqa.mjs --lang es,fr --sample 200
|
|
7
|
+
* node scripts/i18n/tqa.mjs --lang es --judge-provider openai --judge-model gpt-5.6-luna
|
|
8
|
+
* node scripts/i18n/tqa.mjs --lang es --dry # cost estimate only
|
|
9
|
+
* node scripts/i18n/tqa.mjs --lang es --repeat # measure the judge's own variance
|
|
10
|
+
*
|
|
11
|
+
* Reads i18n/source.json, i18n/tm/{lang}.json
|
|
12
|
+
* Writes i18n/tqa/{lang}.json and i18n/tqa/{lang}.md
|
|
13
|
+
*
|
|
14
|
+
* ── What this is ─────────────────────────────────────────────────────────────
|
|
15
|
+
* An MQM-style error-typology assessment of a stratified sample, scored per 100 words.
|
|
16
|
+
* MQM is the standard the industry actually uses, so the output is comparable with a
|
|
17
|
+
* human review rather than being a number invented here.
|
|
18
|
+
*
|
|
19
|
+
* score = 100 - (weighted error points / words) * 100
|
|
20
|
+
*
|
|
21
|
+
* ── What this is NOT ─────────────────────────────────────────────────────────
|
|
22
|
+
* It is not a human certification, and nothing here should be reported as one. It is one
|
|
23
|
+
* model's opinion of another model's output, and models are known to prefer their own
|
|
24
|
+
* work — which is why the judge defaults to a DIFFERENT provider than the translator
|
|
25
|
+
* whenever one is configured, and why --repeat exists to show how much the judge moves
|
|
26
|
+
* between identical runs. A quality number without its variance is marketing.
|
|
27
|
+
*
|
|
28
|
+
* The honest reading: this reliably finds the bottom of the distribution — the units
|
|
29
|
+
* that are actually wrong — and should not be trusted to the second decimal place.
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { readFileSync, writeFileSync, mkdirSync, existsSync } from 'fs';
|
|
33
|
+
import { join } from 'path';
|
|
34
|
+
|
|
35
|
+
import {
|
|
36
|
+
SOURCE_FILE, TM_DIR, I18N_DIR, LOCALES, GLOSSARY, ROOT_DIR as ROOT,
|
|
37
|
+
SITE_NAME, SITE_DESCRIPTION, SOURCE_LANGUAGE, MODEL as CFG_MODEL,
|
|
38
|
+
PROVIDER as CFG_PROVIDER, API_BASE_URL, API_KEY_ENV, JSON_MODE, PRICING,
|
|
39
|
+
} from './config.mjs';
|
|
40
|
+
import { creditBlock } from './credit.mjs';
|
|
41
|
+
import { loadProvider } from './providers/index.mjs';
|
|
42
|
+
import { termsForBatch } from './glossary.mjs';
|
|
43
|
+
import { WEIGHT, CATEGORIES, rng, words, stratifiedSample, parseErrors, score } from './tqa-score.mjs';
|
|
44
|
+
|
|
45
|
+
// ── CLI ──────────────────────────────────────────────────────────────────────
|
|
46
|
+
|
|
47
|
+
const args = Object.fromEntries(
|
|
48
|
+
process.argv
|
|
49
|
+
.slice(2)
|
|
50
|
+
.join(' ')
|
|
51
|
+
.split('--')
|
|
52
|
+
.filter(Boolean)
|
|
53
|
+
.map((s) => s.trim().split(/\s+/))
|
|
54
|
+
.map(([k, ...v]) => [k, v.join(' ') || true])
|
|
55
|
+
);
|
|
56
|
+
|
|
57
|
+
const LANGS = String(args.lang ?? '').split(',').map((s) => s.trim()).filter(Boolean);
|
|
58
|
+
const SAMPLE = Number(args.sample ?? 100);
|
|
59
|
+
const SEED = Number(args.seed ?? 20260829);
|
|
60
|
+
const BATCH = Number(args.batch ?? 10);
|
|
61
|
+
const CONCURRENCY = Number(args.concurrency ?? 6);
|
|
62
|
+
const DRY = Boolean(args.dry);
|
|
63
|
+
const REPEAT = Boolean(args.repeat);
|
|
64
|
+
|
|
65
|
+
if (LANGS.length === 0) {
|
|
66
|
+
console.error(
|
|
67
|
+
'Usage: node scripts/i18n/tqa.mjs --lang es[,fr] [--sample N] [--seed N] [--dry] [--repeat]\n' +
|
|
68
|
+
' [--judge-provider P] [--judge-model M]'
|
|
69
|
+
);
|
|
70
|
+
process.exit(1);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// ── MQM ──────────────────────────────────────────────────────────────────────
|
|
74
|
+
|
|
75
|
+
// ── Judge ────────────────────────────────────────────────────────────────────
|
|
76
|
+
|
|
77
|
+
const JUDGE_PROVIDER_NAME = args['judge-provider'] ? String(args['judge-provider']) : null;
|
|
78
|
+
const JUDGE_MODEL_ARG = args['judge-model'] ? String(args['judge-model']) : null;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Pick a judge that is not the translator, when we can.
|
|
82
|
+
*
|
|
83
|
+
* Self-preference is a documented failure of LLM-as-judge setups: a model scores its own
|
|
84
|
+
* output higher than a peer's. Defaulting the judge to a different provider costs nothing
|
|
85
|
+
* and removes the most obvious objection to the number. If only one provider is
|
|
86
|
+
* configured, we use it and SAY SO in the report rather than hiding it.
|
|
87
|
+
*/
|
|
88
|
+
function pickJudge() {
|
|
89
|
+
if (JUDGE_PROVIDER_NAME) return { provider: JUDGE_PROVIDER_NAME, reason: 'chosen with --judge-provider' };
|
|
90
|
+
const translator = CFG_PROVIDER ?? 'anthropic';
|
|
91
|
+
const alternatives = ['anthropic', 'gemini', 'openai'].filter((p) => p !== translator);
|
|
92
|
+
for (const alt of alternatives) {
|
|
93
|
+
const keys = { anthropic: 'ANTHROPIC_API_KEY', gemini: 'GEMINI_API_KEY', openai: 'OPENAI_API_KEY' }[alt];
|
|
94
|
+
if (process.env[keys]) return { provider: alt, reason: `differs from the translator (${translator})` };
|
|
95
|
+
}
|
|
96
|
+
return { provider: translator, reason: `SAME as the translator — no other provider key found` };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function judgePrompt(langName, langCode, terms) {
|
|
100
|
+
const glossaryLines = terms.length
|
|
101
|
+
? [
|
|
102
|
+
`The site glossary constrains these terms:`,
|
|
103
|
+
...terms.map((t) =>
|
|
104
|
+
t.rule === 'keep'
|
|
105
|
+
? ` - "${t.source}" must appear unchanged`
|
|
106
|
+
: ` - "${t.source}" must be rendered as "${t.targets[langCode]}"`
|
|
107
|
+
),
|
|
108
|
+
]
|
|
109
|
+
: [];
|
|
110
|
+
|
|
111
|
+
return [
|
|
112
|
+
`You are a senior localization reviewer assessing ${langName} translations of ${SITE_NAME}${SITE_DESCRIPTION ? `, ${SITE_DESCRIPTION}` : ''}.`,
|
|
113
|
+
`The source language is ${SOURCE_LANGUAGE}. This is marketing and product copy for a website.`,
|
|
114
|
+
``,
|
|
115
|
+
`For each item, list the translation errors using MQM categories and severities.`,
|
|
116
|
+
``,
|
|
117
|
+
`CATEGORIES: ${CATEGORIES.join(', ')}`,
|
|
118
|
+
`SEVERITIES: minor (noticeable, does not mislead), major (misleads or reads as wrong),`,
|
|
119
|
+
` critical (reverses meaning, breaks a promise, or is unusable)`,
|
|
120
|
+
``,
|
|
121
|
+
...glossaryLines,
|
|
122
|
+
``,
|
|
123
|
+
`RULES`,
|
|
124
|
+
`1. <0>, </0>, <1/> are markup placeholders. They must appear in the translation with`,
|
|
125
|
+
` the same count and numbers. Report any difference as markup/placeholder, critical.`,
|
|
126
|
+
`2. Numbers, prices and measurements must keep their VALUE. Different digit grouping or`,
|
|
127
|
+
` symbol placement is correct localization, not an error. A changed value is critical.`,
|
|
128
|
+
`3. A term left in the source language is only an error if it should have been`,
|
|
129
|
+
` translated. Brand names, formats and protocol names are correct unchanged.`,
|
|
130
|
+
`4. Judge the translation on its own terms as ${langName} copy. Do not reward literalness.`,
|
|
131
|
+
`5. Report NO errors when there are none. An empty list is the expected result for`,
|
|
132
|
+
` most items, and inventing a minor error to look thorough makes the whole score useless.`,
|
|
133
|
+
``,
|
|
134
|
+
`Return a JSON array. For each input item return { "id": <same id>, "text": "<errors>" }`,
|
|
135
|
+
`where <errors> is a JSON array serialised as a string, e.g.`,
|
|
136
|
+
`"[{\\"c\\":\\"fluency/grammar\\",\\"s\\":\\"minor\\",\\"n\\":\\"wrong gender agreement\\"}]"`,
|
|
137
|
+
`or "[]" when the translation is correct.`,
|
|
138
|
+
]
|
|
139
|
+
.filter((l) => l !== null)
|
|
140
|
+
.join('\n');
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
// ── API ──────────────────────────────────────────────────────────────────────
|
|
144
|
+
|
|
145
|
+
const JUDGE = pickJudge();
|
|
146
|
+
const PROVIDER = await loadProvider({ provider: JUDGE.provider, model: JUDGE_MODEL_ARG, root: ROOT });
|
|
147
|
+
const MODEL = JUDGE_MODEL_ARG ?? PROVIDER.defaultModel;
|
|
148
|
+
|
|
149
|
+
function loadKey() {
|
|
150
|
+
const names = API_KEY_ENV ? [API_KEY_ENV] : (PROVIDER.envKeys ?? []);
|
|
151
|
+
for (const name of names) if (process.env[name]) return process.env[name];
|
|
152
|
+
const envFile = join(ROOT, '.env');
|
|
153
|
+
if (existsSync(envFile)) {
|
|
154
|
+
const text = readFileSync(envFile, 'utf8');
|
|
155
|
+
for (const name of names) {
|
|
156
|
+
const m = new RegExp(`^${name}=(.*)$`, 'm').exec(text);
|
|
157
|
+
if (m) return m[1].trim();
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
if (PROVIDER.keyOptional) return null;
|
|
161
|
+
console.error(`No API key for ${PROVIDER.label ?? PROVIDER.id}. Set ${names.join(' or ')}.`);
|
|
162
|
+
process.exit(1);
|
|
163
|
+
}
|
|
164
|
+
const KEY = DRY ? null : loadKey();
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* One judging request. Reuses the translator's retry and batch-splitting behaviour by
|
|
168
|
+
* shape rather than by import: the same safety-block and truncation failures apply to a
|
|
169
|
+
* review pass, and a batch that trips one is halved rather than lost.
|
|
170
|
+
*/
|
|
171
|
+
async function callJudge(system, items, attempt = 1, jsonMode = JSON_MODE) {
|
|
172
|
+
const { url, headers, body } = PROVIDER.request({
|
|
173
|
+
model: MODEL, system, items, temperature: 0, key: KEY, baseUrl: API_BASE_URL, jsonMode,
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
const split = async (why) => {
|
|
177
|
+
if (items.length === 1) return { rows: [], usage: { inTok: 0, outTok: 0 } };
|
|
178
|
+
const mid = Math.ceil(items.length / 2);
|
|
179
|
+
process.stderr.write(` ${why} on ${items.length} items — splitting\n`);
|
|
180
|
+
const [a, b] = await Promise.all([
|
|
181
|
+
callJudge(system, items.slice(0, mid), 1, jsonMode),
|
|
182
|
+
callJudge(system, items.slice(mid), 1, jsonMode),
|
|
183
|
+
]);
|
|
184
|
+
return {
|
|
185
|
+
rows: [...a.rows, ...b.rows],
|
|
186
|
+
usage: { inTok: a.usage.inTok + b.usage.inTok, outTok: a.usage.outTok + b.usage.outTok },
|
|
187
|
+
};
|
|
188
|
+
};
|
|
189
|
+
|
|
190
|
+
let res;
|
|
191
|
+
try {
|
|
192
|
+
res = await fetch(url, { method: 'POST', headers, body: JSON.stringify(body) });
|
|
193
|
+
} catch (err) {
|
|
194
|
+
if (attempt <= 5) {
|
|
195
|
+
await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
|
|
196
|
+
return callJudge(system, items, attempt + 1, jsonMode);
|
|
197
|
+
}
|
|
198
|
+
throw err;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
if (!res.ok) {
|
|
202
|
+
const text = await res.text();
|
|
203
|
+
if (PROVIDER.unsupportedJsonMode?.(res.status, text) && (jsonMode ?? 'schema') !== 'none') {
|
|
204
|
+
return callJudge(system, items, attempt, (jsonMode ?? 'schema') === 'schema' ? 'object' : 'none');
|
|
205
|
+
}
|
|
206
|
+
if ((res.status === 429 || res.status >= 500) && attempt <= 5) {
|
|
207
|
+
await new Promise((r) => setTimeout(r, Math.min(2 ** attempt * 1000, 30000)));
|
|
208
|
+
return callJudge(system, items, attempt + 1, jsonMode);
|
|
209
|
+
}
|
|
210
|
+
throw new Error(`${PROVIDER.label ?? PROVIDER.id} HTTP ${res.status}: ${text.slice(0, 300)}`);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const data = await res.json();
|
|
214
|
+
const { text: part, usage, retryable, detail } = PROVIDER.parse(data);
|
|
215
|
+
if (retryable === 'safety') return split(detail ?? 'safety block');
|
|
216
|
+
if (retryable === 'truncated') return split('truncated');
|
|
217
|
+
|
|
218
|
+
try {
|
|
219
|
+
const rows = PROVIDER.unwrap ? PROVIDER.unwrap(JSON.parse(part)) : JSON.parse(part);
|
|
220
|
+
return { rows: Array.isArray(rows) ? rows : [], usage };
|
|
221
|
+
} catch {
|
|
222
|
+
return split('unparseable response');
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
// ── Run ──────────────────────────────────────────────────────────────────────
|
|
227
|
+
|
|
228
|
+
const LANG_NAMES = Object.fromEntries(LOCALES.map((l) => [l.pathCode, l.nativeLabel ?? l.pathCode]));
|
|
229
|
+
|
|
230
|
+
if (!existsSync(SOURCE_FILE)) {
|
|
231
|
+
console.error(`No ${SOURCE_FILE}. Run extract.mjs first.`);
|
|
232
|
+
process.exit(1);
|
|
233
|
+
}
|
|
234
|
+
const SOURCE = JSON.parse(readFileSync(SOURCE_FILE, 'utf8'));
|
|
235
|
+
|
|
236
|
+
console.log(`judge: ${PROVIDER.label ?? PROVIDER.id} · model: ${MODEL}`);
|
|
237
|
+
console.log(` ${JUDGE.reason}`);
|
|
238
|
+
if (JUDGE.reason.startsWith('SAME')) {
|
|
239
|
+
console.log(' ⚠ a model judging its own output scores it generously — treat the number as a floor, not a grade');
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
mkdirSync(join(I18N_DIR, 'tqa'), { recursive: true });
|
|
243
|
+
|
|
244
|
+
async function assess(langCode, seed) {
|
|
245
|
+
const tmFile = join(TM_DIR, `${langCode}.json`);
|
|
246
|
+
if (!existsSync(tmFile)) {
|
|
247
|
+
console.log(`${langCode}: no memory — skipped`);
|
|
248
|
+
return null;
|
|
249
|
+
}
|
|
250
|
+
const tm = JSON.parse(readFileSync(tmFile, 'utf8'));
|
|
251
|
+
|
|
252
|
+
const pool = Object.entries(SOURCE)
|
|
253
|
+
.filter(([hash]) => typeof tm[hash] === 'string')
|
|
254
|
+
.map(([hash, unit]) => ({ hash, source: unit.text, target: tm[hash], count: unit.count ?? 1 }));
|
|
255
|
+
|
|
256
|
+
if (!pool.length) {
|
|
257
|
+
console.log(`${langCode}: nothing translated yet — skipped`);
|
|
258
|
+
return null;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
const sample = stratifiedSample(pool, SAMPLE, seed).map((u) => ({ ...u, words: words(u.source) }));
|
|
262
|
+
const sampledWords = sample.reduce((n, u) => n + u.words, 0);
|
|
263
|
+
|
|
264
|
+
console.log(
|
|
265
|
+
`\n${langCode}: assessing ${sample.length} of ${pool.length.toLocaleString()} units ` +
|
|
266
|
+
`(${sampledWords.toLocaleString()} source words, seed ${seed})`
|
|
267
|
+
);
|
|
268
|
+
|
|
269
|
+
if (DRY) {
|
|
270
|
+
const estIn = Math.ceil(sampledWords * 3.2) + sample.length * 40;
|
|
271
|
+
const estOut = sample.length * 25;
|
|
272
|
+
console.log(` estimated ~${estIn.toLocaleString()} in / ~${estOut.toLocaleString()} out tokens`);
|
|
273
|
+
const rate = PRICING ?? PROVIDER.pricing?.(MODEL) ?? null;
|
|
274
|
+
if (rate) {
|
|
275
|
+
const cost = (estIn / 1e6) * rate[0] + (estOut / 1e6) * rate[1];
|
|
276
|
+
console.log(` estimated cost: $${cost.toFixed(4)}`);
|
|
277
|
+
} else {
|
|
278
|
+
console.log(' no price known for this endpoint — set "pricing" in i18n.config.json for a figure');
|
|
279
|
+
}
|
|
280
|
+
return null;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
const langName = LANG_NAMES[langCode] ?? langCode;
|
|
284
|
+
const batches = [];
|
|
285
|
+
for (let i = 0; i < sample.length; i += BATCH) batches.push(sample.slice(i, i + BATCH));
|
|
286
|
+
|
|
287
|
+
const assessed = [];
|
|
288
|
+
let unreadable = 0;
|
|
289
|
+
const usage = { inTok: 0, outTok: 0 };
|
|
290
|
+
let cursor = 0;
|
|
291
|
+
let done = 0;
|
|
292
|
+
|
|
293
|
+
const runOne = async () => {
|
|
294
|
+
for (;;) {
|
|
295
|
+
const idx = cursor++;
|
|
296
|
+
if (idx >= batches.length) return;
|
|
297
|
+
const batch = batches[idx];
|
|
298
|
+
const terms = termsForBatch(GLOSSARY, batch.map((u) => u.source), langCode);
|
|
299
|
+
const system = judgePrompt(langName, langCode, terms);
|
|
300
|
+
const items = batch.map((u, i) => ({ id: i, text: `SOURCE: ${u.source}\nTARGET: ${u.target}` }));
|
|
301
|
+
|
|
302
|
+
let rows = [];
|
|
303
|
+
try {
|
|
304
|
+
const out = await callJudge(system, items);
|
|
305
|
+
rows = out.rows;
|
|
306
|
+
usage.inTok += out.usage.inTok;
|
|
307
|
+
usage.outTok += out.usage.outTok;
|
|
308
|
+
} catch (err) {
|
|
309
|
+
process.stderr.write(` batch ${idx} failed: ${String(err.message ?? err).slice(0, 90)}\n`);
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
const byId = new Map(rows.map((r) => [Number(r.id), r.text]));
|
|
313
|
+
for (let i = 0; i < batch.length; i++) {
|
|
314
|
+
const errors = parseErrors(byId.get(i));
|
|
315
|
+
// A unit the judge did not return, or returned unreadably, is EXCLUDED from the
|
|
316
|
+
// score rather than counted as clean. Counting it clean would inflate the result
|
|
317
|
+
// every time the judge failed, which is precisely backwards.
|
|
318
|
+
if (errors === null) {
|
|
319
|
+
unreadable++;
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
assessed.push({ ...batch[i], errors });
|
|
323
|
+
}
|
|
324
|
+
done += batch.length;
|
|
325
|
+
process.stdout.write(`\r ${langCode}: ${done}/${sample.length} units judged`);
|
|
326
|
+
}
|
|
327
|
+
};
|
|
328
|
+
|
|
329
|
+
await Promise.all(Array.from({ length: Math.min(CONCURRENCY, batches.length) }, runOne));
|
|
330
|
+
process.stdout.write('\n');
|
|
331
|
+
|
|
332
|
+
// A score computed from nothing is not a good score, it is a broken measurement — and
|
|
333
|
+
// it would read as a perfect one. Every unit here failed to come back readable at least
|
|
334
|
+
// once during development (a 401 against the wrong endpoint), and the run cheerfully
|
|
335
|
+
// printed "100.00 / 100" from zero assessed units. Refuse instead.
|
|
336
|
+
const judgedShare = sample.length ? (assessed.length * 100) / sample.length : 0;
|
|
337
|
+
if (assessed.length === 0) {
|
|
338
|
+
console.log(
|
|
339
|
+
` \u2717 no unit could be assessed \u2014 the judge returned nothing readable for all ` +
|
|
340
|
+
`${sample.length} of them. No score is reported, because a score from an empty sample ` +
|
|
341
|
+
`would read as a perfect one.`
|
|
342
|
+
);
|
|
343
|
+
return null;
|
|
344
|
+
}
|
|
345
|
+
if (judgedShare < 80) {
|
|
346
|
+
console.log(
|
|
347
|
+
` \u26a0 only ${judgedShare.toFixed(0)}% of the sample was assessed ` +
|
|
348
|
+
`(${unreadable} unit(s) unreadable). The score below is computed from what came back ` +
|
|
349
|
+
`and should not be compared with a clean run.`
|
|
350
|
+
);
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
const result = score(assessed);
|
|
354
|
+
result.lang = langCode;
|
|
355
|
+
result.seed = seed;
|
|
356
|
+
result.sampleRequested = SAMPLE;
|
|
357
|
+
result.poolSize = pool.length;
|
|
358
|
+
result.unreadable = unreadable;
|
|
359
|
+
result.judge = { provider: PROVIDER.id, model: MODEL, note: JUDGE.reason };
|
|
360
|
+
result.usage = usage;
|
|
361
|
+
result.worst = assessed
|
|
362
|
+
.filter((a) => a.errors.length)
|
|
363
|
+
.sort((a, b) => {
|
|
364
|
+
const w = (x) => x.errors.reduce((n, e) => n + WEIGHT[e.severity], 0);
|
|
365
|
+
return w(b) - w(a);
|
|
366
|
+
})
|
|
367
|
+
.slice(0, 15)
|
|
368
|
+
.map((a) => ({ source: a.source, target: a.target, errors: a.errors }));
|
|
369
|
+
|
|
370
|
+
return result;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
const reports = [];
|
|
374
|
+
for (const lang of LANGS) {
|
|
375
|
+
const first = await assess(lang, SEED);
|
|
376
|
+
if (!first) continue;
|
|
377
|
+
|
|
378
|
+
if (REPEAT) {
|
|
379
|
+
// The same sample, judged again. Any gap between the two is the judge's own noise,
|
|
380
|
+
// and reporting a score without it invites the reader to over-read a decimal place.
|
|
381
|
+
console.log(` re-judging the same sample to measure judge variance…`);
|
|
382
|
+
const second = await assess(lang, SEED);
|
|
383
|
+
if (second) {
|
|
384
|
+
first.variance = {
|
|
385
|
+
secondRunMqm: second.mqm,
|
|
386
|
+
delta: Math.round((second.mqm - first.mqm) * 100) / 100,
|
|
387
|
+
};
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
writeFileSync(join(I18N_DIR, 'tqa', `${lang}.json`), JSON.stringify(first, null, 2));
|
|
392
|
+
reports.push(first);
|
|
393
|
+
|
|
394
|
+
const pctClean = first.unitsAssessed ? (first.unitsClean * 100) / first.unitsAssessed : 0;
|
|
395
|
+
console.log(` MQM score: ${first.mqm.toFixed(2)} / 100`);
|
|
396
|
+
console.log(
|
|
397
|
+
` ${first.unitsClean}/${first.unitsAssessed} units error-free (${pctClean.toFixed(0)}%) · ` +
|
|
398
|
+
`${first.bySeverity.critical} critical, ${first.bySeverity.major} major, ${first.bySeverity.minor} minor`
|
|
399
|
+
);
|
|
400
|
+
if (first.unreadable) console.log(` ${first.unreadable} unit(s) excluded — judge returned nothing readable`);
|
|
401
|
+
if (first.variance) {
|
|
402
|
+
console.log(` re-run of the same sample scored ${first.variance.secondRunMqm.toFixed(2)} (Δ ${first.variance.delta >= 0 ? '+' : ''}${first.variance.delta})`);
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
// ── Markdown scorecard ───────────────────────────────────────────────────────
|
|
407
|
+
|
|
408
|
+
if (reports.length) {
|
|
409
|
+
const lines = [
|
|
410
|
+
`# Translation quality assessment — ${SITE_NAME}`,
|
|
411
|
+
``,
|
|
412
|
+
`Generated ${new Date().toISOString().slice(0, 10)} by \`tqa.mjs\`.`,
|
|
413
|
+
``,
|
|
414
|
+
`**Method.** MQM error typology on a stratified sample, weighted toward the strings that`,
|
|
415
|
+
`appear most often on the site. Score = 100 − (weighted error points ÷ words) × 100, with`,
|
|
416
|
+
`minor = 1, major = 5, critical = 10.`,
|
|
417
|
+
``,
|
|
418
|
+
`**This is a machine assessment, not a human certification.** The judge is`,
|
|
419
|
+
`${PROVIDER.label ?? PROVIDER.id} (\`${MODEL}\`) — ${JUDGE.reason}. Read the score as a`,
|
|
420
|
+
`comparison between locales and across runs, not as an absolute grade.`,
|
|
421
|
+
``,
|
|
422
|
+
`| Locale | MQM | Units | Error-free | Critical | Major | Minor |`,
|
|
423
|
+
`| --- | ---: | ---: | ---: | ---: | ---: | ---: |`,
|
|
424
|
+
...reports.map((r) => {
|
|
425
|
+
const pct = r.unitsAssessed ? Math.round((r.unitsClean * 100) / r.unitsAssessed) : 0;
|
|
426
|
+
return `| ${r.lang} | **${r.mqm.toFixed(2)}** | ${r.unitsAssessed} | ${pct}% | ${r.bySeverity.critical} | ${r.bySeverity.major} | ${r.bySeverity.minor} |`;
|
|
427
|
+
}),
|
|
428
|
+
``,
|
|
429
|
+
];
|
|
430
|
+
|
|
431
|
+
for (const r of reports) {
|
|
432
|
+
if (!r.worst.length) continue;
|
|
433
|
+
lines.push(`## ${r.lang} — worst-scoring units`, ``);
|
|
434
|
+
for (const w of r.worst.slice(0, 8)) {
|
|
435
|
+
lines.push(
|
|
436
|
+
`- **${w.errors.map((e) => `${e.severity} ${e.category}`).join(', ')}**`,
|
|
437
|
+
` - source: \`${w.source.slice(0, 140).replace(/`/g, "'")}\``,
|
|
438
|
+
` - target: \`${w.target.slice(0, 140).replace(/`/g, "'")}\``,
|
|
439
|
+
...(w.errors[0]?.note ? [` - note: ${w.errors[0].note}`] : [])
|
|
440
|
+
);
|
|
441
|
+
}
|
|
442
|
+
lines.push(``);
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
const md = join(I18N_DIR, 'tqa', 'scorecard.md');
|
|
446
|
+
writeFileSync(md, lines.join('\n'));
|
|
447
|
+
console.log(`\nwrote ${md}`);
|
|
448
|
+
creditBlock([`${reports.length} locale(s) assessed · MQM sample of ${SAMPLE} units each`]);
|
|
449
|
+
}
|