claude-translator 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +14 -0
- package/CHANGELOG.md +249 -0
- package/PRIVACY.md +71 -0
- package/README.md +279 -44
- package/bin/claude-translator.mjs +11 -2
- package/bin/cli.test.mjs +20 -1
- package/glossary.example.json +23 -0
- package/i18n.config.example.json +7 -0
- package/package.json +9 -5
- package/scripts/audit-seo.mjs +6 -3
- package/scripts/build-locales.mjs +55 -6
- package/scripts/config.mjs +66 -0
- package/scripts/credit.mjs +12 -5
- package/scripts/extract.mjs +21 -6
- package/scripts/format-locale.mjs +290 -0
- package/scripts/format-locale.test.mjs +171 -0
- package/scripts/glossary.mjs +229 -0
- package/scripts/glossary.test.mjs +188 -0
- package/scripts/providers/openai.mjs +63 -3
- package/scripts/providers/providers.test.mjs +80 -0
- package/scripts/roles.mjs +142 -0
- package/scripts/roles.test.mjs +140 -0
- package/scripts/tqa-score.mjs +127 -0
- package/scripts/tqa-score.test.mjs +144 -0
- package/scripts/tqa.mjs +449 -0
- package/scripts/translate.mjs +104 -13
- package/scripts/verify.mjs +113 -5
- package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
- package/{references → skills/translate-site/references}/providers.md +16 -2
- package/{references → skills/translate-site/references}/quality-review.md +28 -0
- /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
- /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
- /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
|
@@ -22,18 +22,44 @@
|
|
|
22
22
|
* so a weaker guarantee here costs retries, not correctness. The rung is chosen by
|
|
23
23
|
* config (`jsonMode`) and falls back automatically when the server rejects a request
|
|
24
24
|
* for mentioning `response_format`.
|
|
25
|
+
*
|
|
26
|
+
* ── Sampling is not universal either ─────────────────────────────────────────
|
|
27
|
+
* The same "one adapter, many servers" problem applies to `temperature`. Reasoning
|
|
28
|
+
* models reject it outright — GPT-5.x answers a `temperature: 0.2` with
|
|
29
|
+
* 400 "Unsupported value: 'temperature' … Only the default (1) is supported" — and
|
|
30
|
+
* they take a reasoning budget instead. That is handled twice over, deliberately:
|
|
31
|
+
* proactively by the model table below, and reactively by `unsupportedParam()`, so a
|
|
32
|
+
* model family nobody has heard of yet degrades instead of failing the whole run.
|
|
25
33
|
*/
|
|
26
34
|
|
|
27
35
|
const API = 'https://api.openai.com/v1';
|
|
28
36
|
|
|
29
37
|
export const id = 'openai';
|
|
30
38
|
export const label = 'OpenAI-compatible';
|
|
31
|
-
export const defaultModel = 'gpt-
|
|
39
|
+
export const defaultModel = 'gpt-5.6-luna';
|
|
32
40
|
export const envKeys = ['OPENAI_API_KEY', 'OPENAI_COMPATIBLE_API_KEY'];
|
|
33
41
|
|
|
34
42
|
/** Local servers accept any key, but some reject a missing Authorization header. */
|
|
35
43
|
export const keyOptional = true;
|
|
36
44
|
|
|
45
|
+
/**
|
|
46
|
+
* Model families that reject sampling parameters and take `reasoning_effort` instead.
|
|
47
|
+
*
|
|
48
|
+
* Kept to a prefix match on purpose. This adapter points at dozens of servers, and a
|
|
49
|
+
* local model whose id merely CONTAINS "gpt-5" must not be caught by it — `qwen2.5:14b`
|
|
50
|
+
* and friends still get `temperature` exactly as before.
|
|
51
|
+
*/
|
|
52
|
+
const REASONING = [/^gpt-5/i, /^o[1-9](-|$)/i];
|
|
53
|
+
const isReasoning = (model) => REASONING.some((re) => re.test(String(model)));
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Parameters this adapter is willing to drop and retry without. Nothing else is
|
|
57
|
+
* droppable, so a 400 that happens to contain the word "unsupported" can never make the
|
|
58
|
+
* pipeline silently discard something load-bearing. `response_format` is deliberately
|
|
59
|
+
* absent — it has its own ladder in `unsupportedJsonMode()`.
|
|
60
|
+
*/
|
|
61
|
+
const DROPPABLE = ['temperature', 'reasoning_effort', 'top_p'];
|
|
62
|
+
|
|
37
63
|
const RESPONSE_SCHEMA = {
|
|
38
64
|
type: 'object',
|
|
39
65
|
properties: {
|
|
@@ -51,16 +77,28 @@ const RESPONSE_SCHEMA = {
|
|
|
51
77
|
additionalProperties: false,
|
|
52
78
|
};
|
|
53
79
|
|
|
54
|
-
export function request({ model, system, items, temperature, key, baseUrl, jsonMode }) {
|
|
80
|
+
export function request({ model, system, items, temperature, key, baseUrl, jsonMode, drop }) {
|
|
81
|
+
const dropped = drop ?? new Set();
|
|
55
82
|
const body = {
|
|
56
83
|
model,
|
|
57
|
-
temperature,
|
|
58
84
|
messages: [
|
|
59
85
|
{ role: 'system', content: system },
|
|
60
86
|
{ role: 'user', content: JSON.stringify(items) },
|
|
61
87
|
],
|
|
62
88
|
};
|
|
63
89
|
|
|
90
|
+
// Reasoning models: no sampling parameter, and no reasoning budget either. Bulk
|
|
91
|
+
// segment translation is a low-reasoning task, so paying for thinking tokens on every
|
|
92
|
+
// batch of forty is pure waste — the same call `anthropic.mjs` makes when it pins the
|
|
93
|
+
// thinking tiers to `effort: 'low'`. `temperature` is omitted rather than sent as 1 so
|
|
94
|
+
// the server applies its own default; sending a value it merely tolerates would be a
|
|
95
|
+
// claim we have no reason to make.
|
|
96
|
+
if (isReasoning(model)) {
|
|
97
|
+
if (!dropped.has('reasoning_effort')) body.reasoning_effort = 'none';
|
|
98
|
+
} else if (!dropped.has('temperature')) {
|
|
99
|
+
body.temperature = temperature;
|
|
100
|
+
}
|
|
101
|
+
|
|
64
102
|
const mode = jsonMode ?? 'schema';
|
|
65
103
|
if (mode === 'schema') {
|
|
66
104
|
body.response_format = {
|
|
@@ -111,9 +149,31 @@ export const unwrap = (parsed) => parsed?.translations ?? parsed;
|
|
|
111
149
|
*/
|
|
112
150
|
export function unsupportedJsonMode(status, errText) {
|
|
113
151
|
if (status !== 400 && status !== 404 && status !== 422) return false;
|
|
152
|
+
if (unsupportedParam(status, errText)) return false; // a sampling complaint, not a schema one
|
|
114
153
|
return /response_format|json_schema|json_object|not supported|unrecognized|unknown.*field/i.test(errText);
|
|
115
154
|
}
|
|
116
155
|
|
|
156
|
+
/**
|
|
157
|
+
* A rejection caused by ONE request parameter rather than by the request being wrong.
|
|
158
|
+
* Returns the parameter's name so `translate.mjs` can retry without it, or null.
|
|
159
|
+
*
|
|
160
|
+
* The bar is deliberately high: the error must read like a capability complaint AND name
|
|
161
|
+
* a parameter this adapter actually sends and is willing to lose. Anything looser would
|
|
162
|
+
* let a genuine 400 ("model not found") be mistaken for a recoverable one, and the run
|
|
163
|
+
* would keep quietly degrading its own request instead of telling the user.
|
|
164
|
+
*
|
|
165
|
+
* Real messages this must catch:
|
|
166
|
+
* 400 "Unsupported value: 'temperature' does not support 0.2 with this model.
|
|
167
|
+
* Only the default (1) is supported."
|
|
168
|
+
* 400 "Unrecognized request argument supplied: reasoning_effort"
|
|
169
|
+
*/
|
|
170
|
+
export function unsupportedParam(status, errText) {
|
|
171
|
+
if (status !== 400 && status !== 422) return null;
|
|
172
|
+
const text = String(errText ?? '');
|
|
173
|
+
if (!/unsupported|unrecognized|unknown|not supported|does not support/i.test(text)) return null;
|
|
174
|
+
return DROPPABLE.find((p) => new RegExp(`\\b${p}\\b`).test(text)) ?? null;
|
|
175
|
+
}
|
|
176
|
+
|
|
117
177
|
/** Unknown by design: this adapter points at dozens of providers, and local ones are free. */
|
|
118
178
|
export function pricing() {
|
|
119
179
|
return null;
|
|
@@ -131,6 +131,86 @@ test('openai: default host and bearer auth', () => {
|
|
|
131
131
|
assert.equal(r.body.messages[1].content, JSON.stringify(ITEMS));
|
|
132
132
|
});
|
|
133
133
|
|
|
134
|
+
// ── Reasoning models reject sampling ─────────────────────────────────────────
|
|
135
|
+
// GPT-5.x answers `temperature: 0.2` with a 400 and takes reasoning_effort instead.
|
|
136
|
+
// Both halves are asserted: what the reasoning models get, and — just as important —
|
|
137
|
+
// that nothing else changed for the models and local servers that were working.
|
|
138
|
+
|
|
139
|
+
test('openai: a reasoning model gets no temperature and no reasoning budget', () => {
|
|
140
|
+
const r = openai.request({ ...ARGS, model: 'gpt-5.6-luna' });
|
|
141
|
+
assert.equal(r.body.temperature, undefined, 'GPT-5.x returns 400 on any temperature but 1');
|
|
142
|
+
assert.equal(r.body.reasoning_effort, 'none', 'bulk translation must not pay for reasoning');
|
|
143
|
+
});
|
|
144
|
+
|
|
145
|
+
test('openai: the o-series is treated the same way', () => {
|
|
146
|
+
for (const model of ['o1', 'o3-mini', 'o4-mini']) {
|
|
147
|
+
const r = openai.request({ ...ARGS, model });
|
|
148
|
+
assert.equal(r.body.temperature, undefined, `${model} rejects sampling params`);
|
|
149
|
+
assert.equal(r.body.reasoning_effort, 'none');
|
|
150
|
+
}
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
test('openai: a pinned gpt-4o-mini still gets temperature, unchanged', () => {
|
|
154
|
+
const r = openai.request({ ...ARGS, model: 'gpt-4o-mini' });
|
|
155
|
+
assert.equal(r.body.temperature, 0.2);
|
|
156
|
+
assert.equal(r.body.reasoning_effort, undefined);
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
test('openai: a local model is never caught by the reasoning rule', () => {
|
|
160
|
+
// The match is anchored, so an id that merely CONTAINS "gpt-5" must not match —
|
|
161
|
+
// a local server would reject reasoning_effort and lose its sampling parameter.
|
|
162
|
+
for (const model of ['qwen2.5:14b', 'llama3.1:70b', 'my-finetune-of-gpt-5']) {
|
|
163
|
+
const r = openai.request({ ...ARGS, model });
|
|
164
|
+
assert.equal(r.body.temperature, 0.2, `${model} must keep sampling`);
|
|
165
|
+
assert.equal(r.body.reasoning_effort, undefined, `${model} must not be sent a reasoning budget`);
|
|
166
|
+
}
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
test('openai: request honours an explicit drop set', () => {
|
|
170
|
+
const noTemp = openai.request({ ...ARGS, model: 'gpt-4o-mini', drop: new Set(['temperature']) });
|
|
171
|
+
assert.equal(noTemp.body.temperature, undefined);
|
|
172
|
+
|
|
173
|
+
const noEffort = openai.request({ ...ARGS, model: 'gpt-5.6-luna', drop: new Set(['reasoning_effort']) });
|
|
174
|
+
assert.equal(noEffort.body.reasoning_effort, undefined);
|
|
175
|
+
assert.equal(noEffort.body.temperature, undefined, 'dropping the budget must not re-add sampling');
|
|
176
|
+
});
|
|
177
|
+
|
|
178
|
+
test('openai: unsupportedParam names the one parameter to drop', () => {
|
|
179
|
+
// The messages OpenAI actually returns.
|
|
180
|
+
assert.equal(
|
|
181
|
+
// Copied verbatim from a real 400 returned by gpt-5.6-luna on 2026-08-25.
|
|
182
|
+
openai.unsupportedParam(400, "Unsupported value: 'temperature' does not support 0.2 with this model. Only the default (1) value is supported."),
|
|
183
|
+
'temperature'
|
|
184
|
+
);
|
|
185
|
+
assert.equal(
|
|
186
|
+
openai.unsupportedParam(400, 'Unrecognized request argument supplied: reasoning_effort'),
|
|
187
|
+
'reasoning_effort'
|
|
188
|
+
);
|
|
189
|
+
|
|
190
|
+
// A genuine bad request must NOT be mistaken for a recoverable one, or the pipeline
|
|
191
|
+
// would quietly strip its own parameters instead of surfacing the real problem.
|
|
192
|
+
assert.equal(openai.unsupportedParam(400, 'model not found'), null);
|
|
193
|
+
assert.equal(openai.unsupportedParam(401, 'invalid api key'), null);
|
|
194
|
+
assert.equal(openai.unsupportedParam(429, 'rate limited'), null);
|
|
195
|
+
// Names a parameter we never send, so there is nothing to usefully drop.
|
|
196
|
+
assert.equal(openai.unsupportedParam(400, "Unsupported parameter: 'frequency_penalty'"), null);
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
test('openai: the two capability hooks never both claim the same error', () => {
|
|
200
|
+
const cases = [
|
|
201
|
+
"Unsupported value: 'temperature' does not support 0.2 with this model.",
|
|
202
|
+
'Unrecognized request argument supplied: reasoning_effort',
|
|
203
|
+
"Unknown field 'response_format'",
|
|
204
|
+
'json_schema is not supported',
|
|
205
|
+
];
|
|
206
|
+
for (const text of cases) {
|
|
207
|
+
const json = openai.unsupportedJsonMode(400, text);
|
|
208
|
+
const param = openai.unsupportedParam(400, text);
|
|
209
|
+
assert.ok(!(json && param), `both hooks claimed: ${text}`);
|
|
210
|
+
assert.ok(json || param, `neither hook claimed a capability error: ${text}`);
|
|
211
|
+
}
|
|
212
|
+
});
|
|
213
|
+
|
|
134
214
|
test('openai: a custom host is used verbatim — this is how local models work', () => {
|
|
135
215
|
const r = openai.request({ ...ARGS, model: 'qwen2.5:14b', baseUrl: 'http://localhost:11434/v1' });
|
|
136
216
|
assert.equal(r.url, 'http://localhost:11434/v1/chat/completions');
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Element context — telling the translator WHAT a string is, not just what it says.
|
|
3
|
+
*
|
|
4
|
+
* The prompt has always ended rule 5 with "Headings stay headings; button labels stay
|
|
5
|
+
* short." That sentence was unenforceable: the model received `{ id, text }` and had no
|
|
6
|
+
* way to tell a <button> from a paragraph. Every string looked like prose.
|
|
7
|
+
*
|
|
8
|
+
* extract.mjs already knows the answer — it walks the DOM and has `node.tagName` in hand
|
|
9
|
+
* at the moment it records a unit. It simply threw it away. This module turns that tag
|
|
10
|
+
* into a short label the model can act on.
|
|
11
|
+
*
|
|
12
|
+
* ── Why a separate field, not a richer `kind` ────────────────────────────────
|
|
13
|
+
* build-locales.mjs switches on `seg.kind === 'block' | 'jsonld' | startsWith('attr:')`
|
|
14
|
+
* to decide how to escape a replacement. A value like 'block:button' would fall through
|
|
15
|
+
* to escHtml and print literal <0> placeholder tokens as visible text on every
|
|
16
|
+
* localized page — silent corruption that no test catches. So `kind` is untouched and
|
|
17
|
+
* `el` is new.
|
|
18
|
+
*
|
|
19
|
+
* ── Why this costs nothing ───────────────────────────────────────────────────
|
|
20
|
+
* Ordinary prose returns null and the field is omitted from the payload entirely. Only
|
|
21
|
+
* the strings where the answer changes the translation carry it.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Block-level elements whose register genuinely differs from body copy.
|
|
26
|
+
*
|
|
27
|
+
* <a> is the interesting entry. It is an INLINE element in extract.mjs, so a link inside
|
|
28
|
+
* a sentence is absorbed into the surrounding block as a <0>…</0> placeholder and never
|
|
29
|
+
* recorded on its own. An <a> that DOES surface as its own unit is therefore a standalone
|
|
30
|
+
* link — a call to action or a nav item — which makes it a high-precision signal rather
|
|
31
|
+
* than noise.
|
|
32
|
+
*/
|
|
33
|
+
const BLOCK_ROLES = {
|
|
34
|
+
button: 'button',
|
|
35
|
+
a: 'button',
|
|
36
|
+
h1: 'heading',
|
|
37
|
+
h2: 'heading',
|
|
38
|
+
h3: 'heading',
|
|
39
|
+
h4: 'heading',
|
|
40
|
+
h5: 'heading',
|
|
41
|
+
h6: 'heading',
|
|
42
|
+
title: 'page title',
|
|
43
|
+
label: 'form label',
|
|
44
|
+
legend: 'form label',
|
|
45
|
+
summary: 'expander label',
|
|
46
|
+
th: 'table header',
|
|
47
|
+
figcaption: 'caption',
|
|
48
|
+
option: 'menu option',
|
|
49
|
+
optgroup: 'menu option',
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/** Attribute-sourced strings. The attribute name IS the role. */
|
|
53
|
+
const ATTR_ROLES = {
|
|
54
|
+
'attr:alt': 'image alt text',
|
|
55
|
+
'attr:placeholder': 'input placeholder',
|
|
56
|
+
'attr:aria-label': 'accessible label',
|
|
57
|
+
'attr:aria-description': 'accessible description',
|
|
58
|
+
'attr:title': 'tooltip',
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Meta tags carry their key in `el` as `meta:<key>`, because `kind` flattens every one of
|
|
63
|
+
* them to `attr:content` and a description behaves nothing like an og:title.
|
|
64
|
+
*/
|
|
65
|
+
const META_ROLES = {
|
|
66
|
+
description: 'meta description',
|
|
67
|
+
'og:description': 'social share description',
|
|
68
|
+
'twitter:description': 'social share description',
|
|
69
|
+
'og:title': 'social share title',
|
|
70
|
+
'twitter:title': 'social share title',
|
|
71
|
+
'og:site_name': 'site name',
|
|
72
|
+
'og:image:alt': 'image alt text',
|
|
73
|
+
'twitter:image:alt': 'image alt text',
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* The label for one unit, or null when it is ordinary prose.
|
|
78
|
+
*
|
|
79
|
+
* @param {{kind?: string, el?: string|null}} unit a record from i18n/source.json
|
|
80
|
+
*/
|
|
81
|
+
export function roleOf(unit) {
|
|
82
|
+
if (!unit || typeof unit !== 'object') return null;
|
|
83
|
+
const { kind, el } = unit;
|
|
84
|
+
|
|
85
|
+
// A conflicting element cleared the hint at extraction time. Saying nothing is correct
|
|
86
|
+
// here — a confident wrong answer is worse than no answer.
|
|
87
|
+
if (el === null || el === undefined) {
|
|
88
|
+
return kind === 'jsonld' ? 'structured data' : null;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
if (typeof el === 'string' && el.startsWith('meta:')) {
|
|
92
|
+
return META_ROLES[el.slice(5)] ?? 'page metadata';
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
if (typeof kind === 'string' && kind.startsWith('attr:')) {
|
|
96
|
+
return ATTR_ROLES[kind] ?? null;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if (kind === 'jsonld') return 'structured data';
|
|
100
|
+
|
|
101
|
+
return BLOCK_ROLES[el] ?? null;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* The guidance block appended to the system prompt, listing only the roles actually
|
|
106
|
+
* present in this batch. Empty string when the batch is all prose, so a run that
|
|
107
|
+
* translates nothing but paragraphs pays not a single extra token.
|
|
108
|
+
*/
|
|
109
|
+
export function rolePrompt(roles) {
|
|
110
|
+
const present = [...new Set(roles.filter(Boolean))].sort();
|
|
111
|
+
if (!present.length) return '';
|
|
112
|
+
|
|
113
|
+
const ADVICE = {
|
|
114
|
+
button: 'A control the user clicks. Keep it at or near the source length — it sits in a fixed-width box. Use whatever construction your language uses on buttons, which is often not the imperative: an infinitive, a verbal noun, or a bare noun may all read better than a literal command. No final period.',
|
|
115
|
+
'form label': 'Names a field. Short, nominal, no final period.',
|
|
116
|
+
'expander label': 'A short clickable summary. Keep it terse.',
|
|
117
|
+
heading: 'A headline. Keep it headline-shaped and roughly the source length; do not expand it into a sentence.',
|
|
118
|
+
'page title': 'The browser tab and search-result title. Under about 60 characters if the language allows.',
|
|
119
|
+
'meta description': 'Search-result copy. One or two sentences, under about 155 characters, written to be read in a results page.',
|
|
120
|
+
'social share title': 'A share-card headline. Short and concrete.',
|
|
121
|
+
'social share description': 'A share-card summary. One sentence.',
|
|
122
|
+
'image alt text': 'Describes an image for someone who cannot see it. Plain and factual, no "image of".',
|
|
123
|
+
'input placeholder': 'Example text inside an empty field. Very short.',
|
|
124
|
+
'accessible label': 'Read aloud by a screen reader. Say what the control does.',
|
|
125
|
+
'accessible description': 'Extra screen-reader detail. One short phrase.',
|
|
126
|
+
tooltip: 'A hover hint. One short phrase.',
|
|
127
|
+
'table header': 'A column heading. Very short, nominal.',
|
|
128
|
+
caption: 'A figure caption. One short sentence.',
|
|
129
|
+
'menu option': 'One choice in a dropdown. Very short.',
|
|
130
|
+
'site name': 'The name of the site. Usually left unchanged.',
|
|
131
|
+
'page metadata': 'Metadata, not body copy. Keep it compact.',
|
|
132
|
+
'structured data': 'A search-engine structured-data value. Plain text, no markup, no added punctuation.',
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
return [
|
|
136
|
+
`CONTEXT`,
|
|
137
|
+
`Some items carry an "el" field naming what the string is on the page. It is not part`,
|
|
138
|
+
`of the text and must not be translated or echoed. Where it appears, honour it:`,
|
|
139
|
+
...present.map((r) => ` - ${r}: ${ADVICE[r] ?? 'Keep the register appropriate to this element.'}`),
|
|
140
|
+
`Items with no "el" field are body copy — translate them normally.`,
|
|
141
|
+
].join('\n');
|
|
142
|
+
}
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Element-context contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* The prompt has always said "button labels stay short" while the model had no way to
|
|
5
|
+
* know what a button was. These tests pin the mapping that finally makes that actionable,
|
|
6
|
+
* and — more importantly — pin the cases where the tool must say NOTHING rather than
|
|
7
|
+
* guess.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { test } from 'node:test';
|
|
11
|
+
import assert from 'node:assert/strict';
|
|
12
|
+
|
|
13
|
+
import { roleOf, rolePrompt } from './roles.mjs';
|
|
14
|
+
|
|
15
|
+
const unit = (el, kind = 'block') => ({ kind, el });
|
|
16
|
+
|
|
17
|
+
// ── Controls ─────────────────────────────────────────────────────────────────
|
|
18
|
+
|
|
19
|
+
test('a button is a button', () => {
|
|
20
|
+
assert.equal(roleOf(unit('button')), 'button');
|
|
21
|
+
});
|
|
22
|
+
|
|
23
|
+
test('a standalone <a> is treated as a call to action', () => {
|
|
24
|
+
// <a> is INLINE in extract.mjs, so a link inside a sentence is absorbed into the
|
|
25
|
+
// surrounding block as a placeholder and never recorded on its own. An <a> that DOES
|
|
26
|
+
// surface as its own unit is a standalone link — a CTA or a nav item.
|
|
27
|
+
assert.equal(roleOf(unit('a')), 'button');
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
test('headings at every level map to one role', () => {
|
|
31
|
+
for (const h of ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']) {
|
|
32
|
+
assert.equal(roleOf(unit(h)), 'heading', h);
|
|
33
|
+
}
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
test('form and table furniture get their own roles', () => {
|
|
37
|
+
assert.equal(roleOf(unit('label')), 'form label');
|
|
38
|
+
assert.equal(roleOf(unit('legend')), 'form label');
|
|
39
|
+
assert.equal(roleOf(unit('summary')), 'expander label');
|
|
40
|
+
assert.equal(roleOf(unit('th')), 'table header');
|
|
41
|
+
assert.equal(roleOf(unit('figcaption')), 'caption');
|
|
42
|
+
assert.equal(roleOf(unit('option')), 'menu option');
|
|
43
|
+
assert.equal(roleOf(unit('title')), 'page title');
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
// ── Prose says nothing ───────────────────────────────────────────────────────
|
|
47
|
+
|
|
48
|
+
test('ordinary prose returns null so the field is omitted entirely', () => {
|
|
49
|
+
for (const el of ['p', 'div', 'span', 'li', 'td', 'blockquote', 'section']) {
|
|
50
|
+
assert.equal(roleOf(unit(el)), null, el);
|
|
51
|
+
}
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test('an unknown element is prose, not a guess', () => {
|
|
55
|
+
assert.equal(roleOf(unit('marquee')), null);
|
|
56
|
+
assert.equal(roleOf(unit('my-web-component')), null);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
// ── The collision case ───────────────────────────────────────────────────────
|
|
60
|
+
|
|
61
|
+
test('a cleared element yields no hint at all', () => {
|
|
62
|
+
// extract.mjs sets el to null when the same string appears as both a button and a
|
|
63
|
+
// paragraph. Saying nothing is correct: a confident wrong answer is worse than none.
|
|
64
|
+
assert.equal(roleOf(unit(null)), null);
|
|
65
|
+
assert.equal(roleOf({ kind: 'block' }), null, 'a missing el behaves like a cleared one');
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
// ── Attributes ───────────────────────────────────────────────────────────────
|
|
69
|
+
|
|
70
|
+
test('attribute roles come from the attribute, not the element', () => {
|
|
71
|
+
assert.equal(roleOf({ kind: 'attr:alt', el: 'img' }), 'image alt text');
|
|
72
|
+
assert.equal(roleOf({ kind: 'attr:placeholder', el: 'input' }), 'input placeholder');
|
|
73
|
+
assert.equal(roleOf({ kind: 'attr:aria-label', el: 'button' }), 'accessible label');
|
|
74
|
+
assert.equal(roleOf({ kind: 'attr:title', el: 'abbr' }), 'tooltip');
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
test('an unmapped attribute is silent rather than mislabelled', () => {
|
|
78
|
+
assert.equal(roleOf({ kind: 'attr:data-thing', el: 'div' }), null);
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
// ── Meta tags ────────────────────────────────────────────────────────────────
|
|
82
|
+
|
|
83
|
+
test('meta tags are distinguished by key, which kind alone cannot express', () => {
|
|
84
|
+
// Every meta tag arrives as kind "attr:content"; the key travels in el.
|
|
85
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:description' }), 'meta description');
|
|
86
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:title' }), 'social share title');
|
|
87
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:description' }), 'social share description');
|
|
88
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:image:alt' }), 'image alt text');
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
test('an unrecognised meta key still says "not body copy"', () => {
|
|
92
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:custom:thing' }), 'page metadata');
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
// ── JSON-LD ──────────────────────────────────────────────────────────────────
|
|
96
|
+
|
|
97
|
+
test('structured data is labelled even though it has no element', () => {
|
|
98
|
+
assert.equal(roleOf({ kind: 'jsonld', el: null }), 'structured data');
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
// ── Defensive ────────────────────────────────────────────────────────────────
|
|
102
|
+
|
|
103
|
+
test('junk input never throws', () => {
|
|
104
|
+
for (const junk of [null, undefined, 'string', 42, []]) {
|
|
105
|
+
assert.doesNotThrow(() => roleOf(junk));
|
|
106
|
+
}
|
|
107
|
+
assert.equal(roleOf(null), null);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
// ── The prompt block ─────────────────────────────────────────────────────────
|
|
111
|
+
|
|
112
|
+
test('a batch of pure prose produces no prompt text and costs no tokens', () => {
|
|
113
|
+
assert.equal(rolePrompt([]), '');
|
|
114
|
+
assert.equal(rolePrompt([null, null, undefined]), '');
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
test('only the roles present in the batch are described', () => {
|
|
118
|
+
const p = rolePrompt(['button', 'heading']);
|
|
119
|
+
assert.match(p, /- button:/);
|
|
120
|
+
assert.match(p, /- heading:/);
|
|
121
|
+
assert.ok(!/image alt text/.test(p), 'roles absent from the batch must not be described');
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
test('duplicates collapse, so forty buttons describe the role once', () => {
|
|
125
|
+
const p = rolePrompt(Array(40).fill('button'));
|
|
126
|
+
assert.equal((p.match(/- button:/g) ?? []).length, 1);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test('the button advice does not demand the imperative', () => {
|
|
130
|
+
// Languages differ: German UI prefers a verbal noun, French the infinitive. Ordering a
|
|
131
|
+
// literal imperative everywhere is precisely the defect this feature exists to avoid.
|
|
132
|
+
const p = rolePrompt(['button']);
|
|
133
|
+
assert.match(p, /not the imperative/i);
|
|
134
|
+
assert.match(p, /fixed-width/);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test('the prompt tells the model el is metadata, not text to translate', () => {
|
|
138
|
+
const p = rolePrompt(['button']);
|
|
139
|
+
assert.match(p, /must not be translated or echoed/);
|
|
140
|
+
});
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TQA scoring primitives — the parts with no I/O, no network and no config, so they can
|
|
3
|
+
* be tested directly. tqa.mjs owns the run; this file owns the arithmetic and the
|
|
4
|
+
* sampling, which are the parts a reader is entitled to check.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Severity weights. These are the conventional MQM values — minor 1, major 5,
|
|
9
|
+
* critical 10 — kept rather than tuned, so a score here means the same thing it means
|
|
10
|
+
* in any other MQM report.
|
|
11
|
+
*/
|
|
12
|
+
export const WEIGHT = { minor: 1, major: 5, critical: 10 };
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The typology offered to the judge. Deliberately short: a long taxonomy produces
|
|
16
|
+
* category-shopping and inconsistent labelling between runs, and the categories that
|
|
17
|
+
* matter for web localization are these.
|
|
18
|
+
*/
|
|
19
|
+
export const CATEGORIES = [
|
|
20
|
+
'accuracy/mistranslation',
|
|
21
|
+
'accuracy/omission',
|
|
22
|
+
'accuracy/addition',
|
|
23
|
+
'accuracy/untranslated',
|
|
24
|
+
'fluency/grammar',
|
|
25
|
+
'fluency/spelling',
|
|
26
|
+
'fluency/register',
|
|
27
|
+
'fluency/awkward',
|
|
28
|
+
'terminology/inconsistent',
|
|
29
|
+
'terminology/glossary',
|
|
30
|
+
'locale/number-date-currency',
|
|
31
|
+
'style/tone',
|
|
32
|
+
'markup/placeholder',
|
|
33
|
+
];
|
|
34
|
+
|
|
35
|
+
// ── Deterministic sampling ───────────────────────────────────────────────────
|
|
36
|
+
|
|
37
|
+
/** mulberry32 — a small seeded PRNG, so a reported score can be reproduced exactly. */
|
|
38
|
+
export function rng(seed) {
|
|
39
|
+
let a = seed >>> 0;
|
|
40
|
+
return () => {
|
|
41
|
+
a = (a + 0x6d2b79f5) >>> 0;
|
|
42
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
43
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
44
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export const words = (s) => (String(s).trim().match(/\S+/g) ?? []).length;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Stratified sample, weighted by how often a unit appears on the site.
|
|
52
|
+
*
|
|
53
|
+
* A uniform sample over-represents the long tail of one-off strings and under-represents
|
|
54
|
+
* the header, footer and nav that every visitor reads on every page. `count` is already
|
|
55
|
+
* recorded by extract.mjs, so weighting by it costs nothing and makes the score reflect
|
|
56
|
+
* what people actually see. Strata are frequency terciles, sampled proportionally.
|
|
57
|
+
*/
|
|
58
|
+
export function stratifiedSample(units, n, seed) {
|
|
59
|
+
if (units.length <= n) return units.slice();
|
|
60
|
+
const rand = rng(seed);
|
|
61
|
+
const sorted = units.slice().sort((a, b) => (b.count ?? 1) - (a.count ?? 1));
|
|
62
|
+
const third = Math.ceil(sorted.length / 3);
|
|
63
|
+
const strata = [sorted.slice(0, third), sorted.slice(third, third * 2), sorted.slice(third * 2)];
|
|
64
|
+
|
|
65
|
+
const picked = [];
|
|
66
|
+
strata.forEach((stratum, i) => {
|
|
67
|
+
if (!stratum.length) return;
|
|
68
|
+
// The most-repeated third gets half the sample; the rest split the remainder.
|
|
69
|
+
const share = i === 0 ? 0.5 : 0.25;
|
|
70
|
+
const want = Math.min(stratum.length, Math.max(1, Math.round(n * share)));
|
|
71
|
+
const pool = stratum.slice();
|
|
72
|
+
for (let k = 0; k < want && pool.length; k++) {
|
|
73
|
+
picked.push(pool.splice(Math.floor(rand() * pool.length), 1)[0]);
|
|
74
|
+
}
|
|
75
|
+
});
|
|
76
|
+
return picked.slice(0, n);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* MQM score for a set of assessed units.
|
|
82
|
+
*/
|
|
83
|
+
export function score(assessed) {
|
|
84
|
+
const totalWords = assessed.reduce((n, a) => n + a.words, 0);
|
|
85
|
+
let penalty = 0;
|
|
86
|
+
const byCategory = {};
|
|
87
|
+
const bySeverity = { minor: 0, major: 0, critical: 0 };
|
|
88
|
+
|
|
89
|
+
for (const a of assessed) {
|
|
90
|
+
for (const e of a.errors) {
|
|
91
|
+
penalty += WEIGHT[e.severity];
|
|
92
|
+
bySeverity[e.severity]++;
|
|
93
|
+
byCategory[e.category] = (byCategory[e.category] ?? 0) + 1;
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
const mqm = totalWords ? 100 - (penalty / totalWords) * 100 : 100;
|
|
97
|
+
return {
|
|
98
|
+
mqm: Math.round(mqm * 100) / 100,
|
|
99
|
+
penalty,
|
|
100
|
+
totalWords,
|
|
101
|
+
unitsAssessed: assessed.length,
|
|
102
|
+
unitsClean: assessed.filter((a) => a.errors.length === 0).length,
|
|
103
|
+
bySeverity,
|
|
104
|
+
byCategory,
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
/** Parse the judge's per-unit payload. Anything unreadable is dropped, never guessed. */
|
|
110
|
+
export function parseErrors(raw) {
|
|
111
|
+
if (typeof raw !== 'string') return null;
|
|
112
|
+
const text = raw.trim();
|
|
113
|
+
if (!text) return [];
|
|
114
|
+
try {
|
|
115
|
+
const parsed = JSON.parse(text);
|
|
116
|
+
if (!Array.isArray(parsed)) return null;
|
|
117
|
+
return parsed
|
|
118
|
+
.map((e) => ({
|
|
119
|
+
category: String(e.c ?? e.category ?? '').trim(),
|
|
120
|
+
severity: String(e.s ?? e.severity ?? '').trim().toLowerCase(),
|
|
121
|
+
note: String(e.n ?? e.note ?? '').trim(),
|
|
122
|
+
}))
|
|
123
|
+
.filter((e) => WEIGHT[e.severity] !== undefined);
|
|
124
|
+
} catch {
|
|
125
|
+
return null;
|
|
126
|
+
}
|
|
127
|
+
}
|