claude-translator 1.3.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/.claude-plugin/plugin.json +14 -0
  2. package/CHANGELOG.md +249 -0
  3. package/PRIVACY.md +71 -0
  4. package/README.md +279 -44
  5. package/bin/claude-translator.mjs +11 -2
  6. package/bin/cli.test.mjs +20 -1
  7. package/glossary.example.json +23 -0
  8. package/i18n.config.example.json +7 -0
  9. package/package.json +9 -5
  10. package/scripts/audit-seo.mjs +6 -3
  11. package/scripts/build-locales.mjs +55 -6
  12. package/scripts/config.mjs +66 -0
  13. package/scripts/credit.mjs +12 -5
  14. package/scripts/extract.mjs +21 -6
  15. package/scripts/format-locale.mjs +290 -0
  16. package/scripts/format-locale.test.mjs +171 -0
  17. package/scripts/glossary.mjs +229 -0
  18. package/scripts/glossary.test.mjs +188 -0
  19. package/scripts/providers/openai.mjs +63 -3
  20. package/scripts/providers/providers.test.mjs +80 -0
  21. package/scripts/roles.mjs +142 -0
  22. package/scripts/roles.test.mjs +140 -0
  23. package/scripts/tqa-score.mjs +127 -0
  24. package/scripts/tqa-score.test.mjs +144 -0
  25. package/scripts/tqa.mjs +449 -0
  26. package/scripts/translate.mjs +104 -13
  27. package/scripts/verify.mjs +113 -5
  28. package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
  29. package/{references → skills/translate-site/references}/providers.md +16 -2
  30. package/{references → skills/translate-site/references}/quality-review.md +28 -0
  31. /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
  32. /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
  33. /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
@@ -22,18 +22,44 @@
22
22
  * so a weaker guarantee here costs retries, not correctness. The rung is chosen by
23
23
  * config (`jsonMode`) and falls back automatically when the server rejects a request
24
24
  * for mentioning `response_format`.
25
+ *
26
+ * ── Sampling is not universal either ─────────────────────────────────────────
27
+ * The same "one adapter, many servers" problem applies to `temperature`. Reasoning
28
+ * models reject it outright — GPT-5.x answers a `temperature: 0.2` with
29
+ * 400 "Unsupported value: 'temperature' … Only the default (1) is supported" — and
30
+ * they take a reasoning budget instead. That is handled twice over, deliberately:
31
+ * proactively by the model table below, and reactively by `unsupportedParam()`, so a
32
+ * model family nobody has heard of yet degrades instead of failing the whole run.
25
33
  */
26
34
 
27
35
  const API = 'https://api.openai.com/v1';
28
36
 
29
37
  export const id = 'openai';
30
38
  export const label = 'OpenAI-compatible';
31
- export const defaultModel = 'gpt-4o-mini';
39
+ export const defaultModel = 'gpt-5.6-luna';
32
40
  export const envKeys = ['OPENAI_API_KEY', 'OPENAI_COMPATIBLE_API_KEY'];
33
41
 
34
42
  /** Local servers accept any key, but some reject a missing Authorization header. */
35
43
  export const keyOptional = true;
36
44
 
45
+ /**
46
+ * Model families that reject sampling parameters and take `reasoning_effort` instead.
47
+ *
48
+ * Kept to a prefix match on purpose. This adapter points at dozens of servers, and a
49
+ * local model whose id merely CONTAINS "gpt-5" must not be caught by it — `qwen2.5:14b`
50
+ * and friends still get `temperature` exactly as before.
51
+ */
52
+ const REASONING = [/^gpt-5/i, /^o[1-9](-|$)/i];
53
+ const isReasoning = (model) => REASONING.some((re) => re.test(String(model)));
54
+
55
+ /**
56
+ * Parameters this adapter is willing to drop and retry without. Nothing else is
57
+ * droppable, so a 400 that happens to contain the word "unsupported" can never make the
58
+ * pipeline silently discard something load-bearing. `response_format` is deliberately
59
+ * absent — it has its own ladder in `unsupportedJsonMode()`.
60
+ */
61
+ const DROPPABLE = ['temperature', 'reasoning_effort', 'top_p'];
62
+
37
63
  const RESPONSE_SCHEMA = {
38
64
  type: 'object',
39
65
  properties: {
@@ -51,16 +77,28 @@ const RESPONSE_SCHEMA = {
51
77
  additionalProperties: false,
52
78
  };
53
79
 
54
- export function request({ model, system, items, temperature, key, baseUrl, jsonMode }) {
80
+ export function request({ model, system, items, temperature, key, baseUrl, jsonMode, drop }) {
81
+ const dropped = drop ?? new Set();
55
82
  const body = {
56
83
  model,
57
- temperature,
58
84
  messages: [
59
85
  { role: 'system', content: system },
60
86
  { role: 'user', content: JSON.stringify(items) },
61
87
  ],
62
88
  };
63
89
 
90
+ // Reasoning models: no sampling parameter, and no reasoning budget either. Bulk
91
+ // segment translation is a low-reasoning task, so paying for thinking tokens on every
92
+ // batch of forty is pure waste — the same call `anthropic.mjs` makes when it pins the
93
+ // thinking tiers to `effort: 'low'`. `temperature` is omitted rather than sent as 1 so
94
+ // the server applies its own default; sending a value it merely tolerates would be a
95
+ // claim we have no reason to make.
96
+ if (isReasoning(model)) {
97
+ if (!dropped.has('reasoning_effort')) body.reasoning_effort = 'none';
98
+ } else if (!dropped.has('temperature')) {
99
+ body.temperature = temperature;
100
+ }
101
+
64
102
  const mode = jsonMode ?? 'schema';
65
103
  if (mode === 'schema') {
66
104
  body.response_format = {
@@ -111,9 +149,31 @@ export const unwrap = (parsed) => parsed?.translations ?? parsed;
111
149
  */
112
150
  export function unsupportedJsonMode(status, errText) {
113
151
  if (status !== 400 && status !== 404 && status !== 422) return false;
152
+ if (unsupportedParam(status, errText)) return false; // a sampling complaint, not a schema one
114
153
  return /response_format|json_schema|json_object|not supported|unrecognized|unknown.*field/i.test(errText);
115
154
  }
116
155
 
156
+ /**
157
+ * A rejection caused by ONE request parameter rather than by the request being wrong.
158
+ * Returns the parameter's name so `translate.mjs` can retry without it, or null.
159
+ *
160
+ * The bar is deliberately high: the error must read like a capability complaint AND name
161
+ * a parameter this adapter actually sends and is willing to lose. Anything looser would
162
+ * let a genuine 400 ("model not found") be mistaken for a recoverable one, and the run
163
+ * would keep quietly degrading its own request instead of telling the user.
164
+ *
165
+ * Real messages this must catch:
166
+ * 400 "Unsupported value: 'temperature' does not support 0.2 with this model.
167
+ * Only the default (1) is supported."
168
+ * 400 "Unrecognized request argument supplied: reasoning_effort"
169
+ */
170
+ export function unsupportedParam(status, errText) {
171
+ if (status !== 400 && status !== 422) return null;
172
+ const text = String(errText ?? '');
173
+ if (!/unsupported|unrecognized|unknown|not supported|does not support/i.test(text)) return null;
174
+ return DROPPABLE.find((p) => new RegExp(`\\b${p}\\b`).test(text)) ?? null;
175
+ }
176
+
117
177
  /** Unknown by design: this adapter points at dozens of providers, and local ones are free. */
118
178
  export function pricing() {
119
179
  return null;
@@ -131,6 +131,86 @@ test('openai: default host and bearer auth', () => {
131
131
  assert.equal(r.body.messages[1].content, JSON.stringify(ITEMS));
132
132
  });
133
133
 
134
+ // ── Reasoning models reject sampling ─────────────────────────────────────────
135
+ // GPT-5.x answers `temperature: 0.2` with a 400 and takes reasoning_effort instead.
136
+ // Both halves are asserted: what the reasoning models get, and — just as important —
137
+ // that nothing else changed for the models and local servers that were working.
138
+
139
+ test('openai: a reasoning model gets no temperature and no reasoning budget', () => {
140
+ const r = openai.request({ ...ARGS, model: 'gpt-5.6-luna' });
141
+ assert.equal(r.body.temperature, undefined, 'GPT-5.x returns 400 on any temperature but 1');
142
+ assert.equal(r.body.reasoning_effort, 'none', 'bulk translation must not pay for reasoning');
143
+ });
144
+
145
+ test('openai: the o-series is treated the same way', () => {
146
+ for (const model of ['o1', 'o3-mini', 'o4-mini']) {
147
+ const r = openai.request({ ...ARGS, model });
148
+ assert.equal(r.body.temperature, undefined, `${model} rejects sampling params`);
149
+ assert.equal(r.body.reasoning_effort, 'none');
150
+ }
151
+ });
152
+
153
+ test('openai: a pinned gpt-4o-mini still gets temperature, unchanged', () => {
154
+ const r = openai.request({ ...ARGS, model: 'gpt-4o-mini' });
155
+ assert.equal(r.body.temperature, 0.2);
156
+ assert.equal(r.body.reasoning_effort, undefined);
157
+ });
158
+
159
+ test('openai: a local model is never caught by the reasoning rule', () => {
160
+ // The match is anchored, so an id that merely CONTAINS "gpt-5" must not match —
161
+ // a local server would reject reasoning_effort and lose its sampling parameter.
162
+ for (const model of ['qwen2.5:14b', 'llama3.1:70b', 'my-finetune-of-gpt-5']) {
163
+ const r = openai.request({ ...ARGS, model });
164
+ assert.equal(r.body.temperature, 0.2, `${model} must keep sampling`);
165
+ assert.equal(r.body.reasoning_effort, undefined, `${model} must not be sent a reasoning budget`);
166
+ }
167
+ });
168
+
169
+ test('openai: request honours an explicit drop set', () => {
170
+ const noTemp = openai.request({ ...ARGS, model: 'gpt-4o-mini', drop: new Set(['temperature']) });
171
+ assert.equal(noTemp.body.temperature, undefined);
172
+
173
+ const noEffort = openai.request({ ...ARGS, model: 'gpt-5.6-luna', drop: new Set(['reasoning_effort']) });
174
+ assert.equal(noEffort.body.reasoning_effort, undefined);
175
+ assert.equal(noEffort.body.temperature, undefined, 'dropping the budget must not re-add sampling');
176
+ });
177
+
178
+ test('openai: unsupportedParam names the one parameter to drop', () => {
179
+ // The messages OpenAI actually returns.
180
+ assert.equal(
181
+ // Copied verbatim from a real 400 returned by gpt-5.6-luna on 2026-08-25.
182
+ openai.unsupportedParam(400, "Unsupported value: 'temperature' does not support 0.2 with this model. Only the default (1) value is supported."),
183
+ 'temperature'
184
+ );
185
+ assert.equal(
186
+ openai.unsupportedParam(400, 'Unrecognized request argument supplied: reasoning_effort'),
187
+ 'reasoning_effort'
188
+ );
189
+
190
+ // A genuine bad request must NOT be mistaken for a recoverable one, or the pipeline
191
+ // would quietly strip its own parameters instead of surfacing the real problem.
192
+ assert.equal(openai.unsupportedParam(400, 'model not found'), null);
193
+ assert.equal(openai.unsupportedParam(401, 'invalid api key'), null);
194
+ assert.equal(openai.unsupportedParam(429, 'rate limited'), null);
195
+ // Names a parameter we never send, so there is nothing to usefully drop.
196
+ assert.equal(openai.unsupportedParam(400, "Unsupported parameter: 'frequency_penalty'"), null);
197
+ });
198
+
199
+ test('openai: the two capability hooks never both claim the same error', () => {
200
+ const cases = [
201
+ "Unsupported value: 'temperature' does not support 0.2 with this model.",
202
+ 'Unrecognized request argument supplied: reasoning_effort',
203
+ "Unknown field 'response_format'",
204
+ 'json_schema is not supported',
205
+ ];
206
+ for (const text of cases) {
207
+ const json = openai.unsupportedJsonMode(400, text);
208
+ const param = openai.unsupportedParam(400, text);
209
+ assert.ok(!(json && param), `both hooks claimed: ${text}`);
210
+ assert.ok(json || param, `neither hook claimed a capability error: ${text}`);
211
+ }
212
+ });
213
+
134
214
  test('openai: a custom host is used verbatim — this is how local models work', () => {
135
215
  const r = openai.request({ ...ARGS, model: 'qwen2.5:14b', baseUrl: 'http://localhost:11434/v1' });
136
216
  assert.equal(r.url, 'http://localhost:11434/v1/chat/completions');
@@ -0,0 +1,142 @@
1
+ /**
2
+ * Element context — telling the translator WHAT a string is, not just what it says.
3
+ *
4
+ * The prompt has always ended rule 5 with "Headings stay headings; button labels stay
5
+ * short." That sentence was unenforceable: the model received `{ id, text }` and had no
6
+ * way to tell a <button> from a paragraph. Every string looked like prose.
7
+ *
8
+ * extract.mjs already knows the answer — it walks the DOM and has `node.tagName` in hand
9
+ * at the moment it records a unit. It simply threw it away. This module turns that tag
10
+ * into a short label the model can act on.
11
+ *
12
+ * ── Why a separate field, not a richer `kind` ────────────────────────────────
13
+ * build-locales.mjs switches on `seg.kind === 'block' | 'jsonld' | startsWith('attr:')`
14
+ * to decide how to escape a replacement. A value like 'block:button' would fall through
15
+ * to escHtml and print literal &lt;0&gt; placeholder tokens as visible text on every
16
+ * localized page — silent corruption that no test catches. So `kind` is untouched and
17
+ * `el` is new.
18
+ *
19
+ * ── Why this costs nothing ───────────────────────────────────────────────────
20
+ * Ordinary prose returns null and the field is omitted from the payload entirely. Only
21
+ * the strings where the answer changes the translation carry it.
22
+ */
23
+
24
+ /**
25
+ * Block-level elements whose register genuinely differs from body copy.
26
+ *
27
+ * <a> is the interesting entry. It is an INLINE element in extract.mjs, so a link inside
28
+ * a sentence is absorbed into the surrounding block as a <0>…</0> placeholder and never
29
+ * recorded on its own. An <a> that DOES surface as its own unit is therefore a standalone
30
+ * link — a call to action or a nav item — which makes it a high-precision signal rather
31
+ * than noise.
32
+ */
33
+ const BLOCK_ROLES = {
34
+ button: 'button',
35
+ a: 'button',
36
+ h1: 'heading',
37
+ h2: 'heading',
38
+ h3: 'heading',
39
+ h4: 'heading',
40
+ h5: 'heading',
41
+ h6: 'heading',
42
+ title: 'page title',
43
+ label: 'form label',
44
+ legend: 'form label',
45
+ summary: 'expander label',
46
+ th: 'table header',
47
+ figcaption: 'caption',
48
+ option: 'menu option',
49
+ optgroup: 'menu option',
50
+ };
51
+
52
+ /** Attribute-sourced strings. The attribute name IS the role. */
53
+ const ATTR_ROLES = {
54
+ 'attr:alt': 'image alt text',
55
+ 'attr:placeholder': 'input placeholder',
56
+ 'attr:aria-label': 'accessible label',
57
+ 'attr:aria-description': 'accessible description',
58
+ 'attr:title': 'tooltip',
59
+ };
60
+
61
+ /**
62
+ * Meta tags carry their key in `el` as `meta:<key>`, because `kind` flattens every one of
63
+ * them to `attr:content` and a description behaves nothing like an og:title.
64
+ */
65
+ const META_ROLES = {
66
+ description: 'meta description',
67
+ 'og:description': 'social share description',
68
+ 'twitter:description': 'social share description',
69
+ 'og:title': 'social share title',
70
+ 'twitter:title': 'social share title',
71
+ 'og:site_name': 'site name',
72
+ 'og:image:alt': 'image alt text',
73
+ 'twitter:image:alt': 'image alt text',
74
+ };
75
+
76
+ /**
77
+ * The label for one unit, or null when it is ordinary prose.
78
+ *
79
+ * @param {{kind?: string, el?: string|null}} unit a record from i18n/source.json
80
+ */
81
+ export function roleOf(unit) {
82
+ if (!unit || typeof unit !== 'object') return null;
83
+ const { kind, el } = unit;
84
+
85
+ // A conflicting element cleared the hint at extraction time. Saying nothing is correct
86
+ // here — a confident wrong answer is worse than no answer.
87
+ if (el === null || el === undefined) {
88
+ return kind === 'jsonld' ? 'structured data' : null;
89
+ }
90
+
91
+ if (typeof el === 'string' && el.startsWith('meta:')) {
92
+ return META_ROLES[el.slice(5)] ?? 'page metadata';
93
+ }
94
+
95
+ if (typeof kind === 'string' && kind.startsWith('attr:')) {
96
+ return ATTR_ROLES[kind] ?? null;
97
+ }
98
+
99
+ if (kind === 'jsonld') return 'structured data';
100
+
101
+ return BLOCK_ROLES[el] ?? null;
102
+ }
103
+
104
+ /**
105
+ * The guidance block appended to the system prompt, listing only the roles actually
106
+ * present in this batch. Empty string when the batch is all prose, so a run that
107
+ * translates nothing but paragraphs pays not a single extra token.
108
+ */
109
+ export function rolePrompt(roles) {
110
+ const present = [...new Set(roles.filter(Boolean))].sort();
111
+ if (!present.length) return '';
112
+
113
+ const ADVICE = {
114
+ button: 'A control the user clicks. Keep it at or near the source length — it sits in a fixed-width box. Use whatever construction your language uses on buttons, which is often not the imperative: an infinitive, a verbal noun, or a bare noun may all read better than a literal command. No final period.',
115
+ 'form label': 'Names a field. Short, nominal, no final period.',
116
+ 'expander label': 'A short clickable summary. Keep it terse.',
117
+ heading: 'A headline. Keep it headline-shaped and roughly the source length; do not expand it into a sentence.',
118
+ 'page title': 'The browser tab and search-result title. Under about 60 characters if the language allows.',
119
+ 'meta description': 'Search-result copy. One or two sentences, under about 155 characters, written to be read in a results page.',
120
+ 'social share title': 'A share-card headline. Short and concrete.',
121
+ 'social share description': 'A share-card summary. One sentence.',
122
+ 'image alt text': 'Describes an image for someone who cannot see it. Plain and factual, no "image of".',
123
+ 'input placeholder': 'Example text inside an empty field. Very short.',
124
+ 'accessible label': 'Read aloud by a screen reader. Say what the control does.',
125
+ 'accessible description': 'Extra screen-reader detail. One short phrase.',
126
+ tooltip: 'A hover hint. One short phrase.',
127
+ 'table header': 'A column heading. Very short, nominal.',
128
+ caption: 'A figure caption. One short sentence.',
129
+ 'menu option': 'One choice in a dropdown. Very short.',
130
+ 'site name': 'The name of the site. Usually left unchanged.',
131
+ 'page metadata': 'Metadata, not body copy. Keep it compact.',
132
+ 'structured data': 'A search-engine structured-data value. Plain text, no markup, no added punctuation.',
133
+ };
134
+
135
+ return [
136
+ `CONTEXT`,
137
+ `Some items carry an "el" field naming what the string is on the page. It is not part`,
138
+ `of the text and must not be translated or echoed. Where it appears, honour it:`,
139
+ ...present.map((r) => ` - ${r}: ${ADVICE[r] ?? 'Keep the register appropriate to this element.'}`),
140
+ `Items with no "el" field are body copy — translate them normally.`,
141
+ ].join('\n');
142
+ }
@@ -0,0 +1,140 @@
1
+ /**
2
+ * Element-context contract tests. No network, no key, no model.
3
+ *
4
+ * The prompt has always said "button labels stay short" while the model had no way to
5
+ * know what a button was. These tests pin the mapping that finally makes that actionable,
6
+ * and — more importantly — pin the cases where the tool must say NOTHING rather than
7
+ * guess.
8
+ */
9
+
10
+ import { test } from 'node:test';
11
+ import assert from 'node:assert/strict';
12
+
13
+ import { roleOf, rolePrompt } from './roles.mjs';
14
+
15
+ const unit = (el, kind = 'block') => ({ kind, el });
16
+
17
+ // ── Controls ─────────────────────────────────────────────────────────────────
18
+
19
+ test('a button is a button', () => {
20
+ assert.equal(roleOf(unit('button')), 'button');
21
+ });
22
+
23
+ test('a standalone <a> is treated as a call to action', () => {
24
+ // <a> is INLINE in extract.mjs, so a link inside a sentence is absorbed into the
25
+ // surrounding block as a placeholder and never recorded on its own. An <a> that DOES
26
+ // surface as its own unit is a standalone link — a CTA or a nav item.
27
+ assert.equal(roleOf(unit('a')), 'button');
28
+ });
29
+
30
+ test('headings at every level map to one role', () => {
31
+ for (const h of ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']) {
32
+ assert.equal(roleOf(unit(h)), 'heading', h);
33
+ }
34
+ });
35
+
36
+ test('form and table furniture get their own roles', () => {
37
+ assert.equal(roleOf(unit('label')), 'form label');
38
+ assert.equal(roleOf(unit('legend')), 'form label');
39
+ assert.equal(roleOf(unit('summary')), 'expander label');
40
+ assert.equal(roleOf(unit('th')), 'table header');
41
+ assert.equal(roleOf(unit('figcaption')), 'caption');
42
+ assert.equal(roleOf(unit('option')), 'menu option');
43
+ assert.equal(roleOf(unit('title')), 'page title');
44
+ });
45
+
46
+ // ── Prose says nothing ───────────────────────────────────────────────────────
47
+
48
+ test('ordinary prose returns null so the field is omitted entirely', () => {
49
+ for (const el of ['p', 'div', 'span', 'li', 'td', 'blockquote', 'section']) {
50
+ assert.equal(roleOf(unit(el)), null, el);
51
+ }
52
+ });
53
+
54
+ test('an unknown element is prose, not a guess', () => {
55
+ assert.equal(roleOf(unit('marquee')), null);
56
+ assert.equal(roleOf(unit('my-web-component')), null);
57
+ });
58
+
59
+ // ── The collision case ───────────────────────────────────────────────────────
60
+
61
+ test('a cleared element yields no hint at all', () => {
62
+ // extract.mjs sets el to null when the same string appears as both a button and a
63
+ // paragraph. Saying nothing is correct: a confident wrong answer is worse than none.
64
+ assert.equal(roleOf(unit(null)), null);
65
+ assert.equal(roleOf({ kind: 'block' }), null, 'a missing el behaves like a cleared one');
66
+ });
67
+
68
+ // ── Attributes ───────────────────────────────────────────────────────────────
69
+
70
+ test('attribute roles come from the attribute, not the element', () => {
71
+ assert.equal(roleOf({ kind: 'attr:alt', el: 'img' }), 'image alt text');
72
+ assert.equal(roleOf({ kind: 'attr:placeholder', el: 'input' }), 'input placeholder');
73
+ assert.equal(roleOf({ kind: 'attr:aria-label', el: 'button' }), 'accessible label');
74
+ assert.equal(roleOf({ kind: 'attr:title', el: 'abbr' }), 'tooltip');
75
+ });
76
+
77
+ test('an unmapped attribute is silent rather than mislabelled', () => {
78
+ assert.equal(roleOf({ kind: 'attr:data-thing', el: 'div' }), null);
79
+ });
80
+
81
+ // ── Meta tags ────────────────────────────────────────────────────────────────
82
+
83
+ test('meta tags are distinguished by key, which kind alone cannot express', () => {
84
+ // Every meta tag arrives as kind "attr:content"; the key travels in el.
85
+ assert.equal(roleOf({ kind: 'attr:content', el: 'meta:description' }), 'meta description');
86
+ assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:title' }), 'social share title');
87
+ assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:description' }), 'social share description');
88
+ assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:image:alt' }), 'image alt text');
89
+ });
90
+
91
+ test('an unrecognised meta key still says "not body copy"', () => {
92
+ assert.equal(roleOf({ kind: 'attr:content', el: 'meta:custom:thing' }), 'page metadata');
93
+ });
94
+
95
+ // ── JSON-LD ──────────────────────────────────────────────────────────────────
96
+
97
+ test('structured data is labelled even though it has no element', () => {
98
+ assert.equal(roleOf({ kind: 'jsonld', el: null }), 'structured data');
99
+ });
100
+
101
+ // ── Defensive ────────────────────────────────────────────────────────────────
102
+
103
+ test('junk input never throws', () => {
104
+ for (const junk of [null, undefined, 'string', 42, []]) {
105
+ assert.doesNotThrow(() => roleOf(junk));
106
+ }
107
+ assert.equal(roleOf(null), null);
108
+ });
109
+
110
+ // ── The prompt block ─────────────────────────────────────────────────────────
111
+
112
+ test('a batch of pure prose produces no prompt text and costs no tokens', () => {
113
+ assert.equal(rolePrompt([]), '');
114
+ assert.equal(rolePrompt([null, null, undefined]), '');
115
+ });
116
+
117
+ test('only the roles present in the batch are described', () => {
118
+ const p = rolePrompt(['button', 'heading']);
119
+ assert.match(p, /- button:/);
120
+ assert.match(p, /- heading:/);
121
+ assert.ok(!/image alt text/.test(p), 'roles absent from the batch must not be described');
122
+ });
123
+
124
+ test('duplicates collapse, so forty buttons describe the role once', () => {
125
+ const p = rolePrompt(Array(40).fill('button'));
126
+ assert.equal((p.match(/- button:/g) ?? []).length, 1);
127
+ });
128
+
129
+ test('the button advice does not demand the imperative', () => {
130
+ // Languages differ: German UI prefers a verbal noun, French the infinitive. Ordering a
131
+ // literal imperative everywhere is precisely the defect this feature exists to avoid.
132
+ const p = rolePrompt(['button']);
133
+ assert.match(p, /not the imperative/i);
134
+ assert.match(p, /fixed-width/);
135
+ });
136
+
137
+ test('the prompt tells the model el is metadata, not text to translate', () => {
138
+ const p = rolePrompt(['button']);
139
+ assert.match(p, /must not be translated or echoed/);
140
+ });
@@ -0,0 +1,127 @@
1
+ /**
2
+ * TQA scoring primitives — the parts with no I/O, no network and no config, so they can
3
+ * be tested directly. tqa.mjs owns the run; this file owns the arithmetic and the
4
+ * sampling, which are the parts a reader is entitled to check.
5
+ */
6
+
7
+ /**
8
+ * Severity weights. These are the conventional MQM values — minor 1, major 5,
9
+ * critical 10 — kept rather than tuned, so a score here means the same thing it means
10
+ * in any other MQM report.
11
+ */
12
+ export const WEIGHT = { minor: 1, major: 5, critical: 10 };
13
+
14
+ /**
15
+ * The typology offered to the judge. Deliberately short: a long taxonomy produces
16
+ * category-shopping and inconsistent labelling between runs, and the categories that
17
+ * matter for web localization are these.
18
+ */
19
+ export const CATEGORIES = [
20
+ 'accuracy/mistranslation',
21
+ 'accuracy/omission',
22
+ 'accuracy/addition',
23
+ 'accuracy/untranslated',
24
+ 'fluency/grammar',
25
+ 'fluency/spelling',
26
+ 'fluency/register',
27
+ 'fluency/awkward',
28
+ 'terminology/inconsistent',
29
+ 'terminology/glossary',
30
+ 'locale/number-date-currency',
31
+ 'style/tone',
32
+ 'markup/placeholder',
33
+ ];
34
+
35
+ // ── Deterministic sampling ───────────────────────────────────────────────────
36
+
37
+ /** mulberry32 — a small seeded PRNG, so a reported score can be reproduced exactly. */
38
+ export function rng(seed) {
39
+ let a = seed >>> 0;
40
+ return () => {
41
+ a = (a + 0x6d2b79f5) >>> 0;
42
+ let t = Math.imul(a ^ (a >>> 15), 1 | a);
43
+ t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
44
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
45
+ };
46
+ }
47
+
48
+ export const words = (s) => (String(s).trim().match(/\S+/g) ?? []).length;
49
+
50
+ /**
51
+ * Stratified sample, weighted by how often a unit appears on the site.
52
+ *
53
+ * A uniform sample over-represents the long tail of one-off strings and under-represents
54
+ * the header, footer and nav that every visitor reads on every page. `count` is already
55
+ * recorded by extract.mjs, so weighting by it costs nothing and makes the score reflect
56
+ * what people actually see. Strata are frequency terciles, sampled proportionally.
57
+ */
58
+ export function stratifiedSample(units, n, seed) {
59
+ if (units.length <= n) return units.slice();
60
+ const rand = rng(seed);
61
+ const sorted = units.slice().sort((a, b) => (b.count ?? 1) - (a.count ?? 1));
62
+ const third = Math.ceil(sorted.length / 3);
63
+ const strata = [sorted.slice(0, third), sorted.slice(third, third * 2), sorted.slice(third * 2)];
64
+
65
+ const picked = [];
66
+ strata.forEach((stratum, i) => {
67
+ if (!stratum.length) return;
68
+ // The most-repeated third gets half the sample; the rest split the remainder.
69
+ const share = i === 0 ? 0.5 : 0.25;
70
+ const want = Math.min(stratum.length, Math.max(1, Math.round(n * share)));
71
+ const pool = stratum.slice();
72
+ for (let k = 0; k < want && pool.length; k++) {
73
+ picked.push(pool.splice(Math.floor(rand() * pool.length), 1)[0]);
74
+ }
75
+ });
76
+ return picked.slice(0, n);
77
+ }
78
+
79
+
80
+ /**
81
+ * MQM score for a set of assessed units.
82
+ */
83
+ export function score(assessed) {
84
+ const totalWords = assessed.reduce((n, a) => n + a.words, 0);
85
+ let penalty = 0;
86
+ const byCategory = {};
87
+ const bySeverity = { minor: 0, major: 0, critical: 0 };
88
+
89
+ for (const a of assessed) {
90
+ for (const e of a.errors) {
91
+ penalty += WEIGHT[e.severity];
92
+ bySeverity[e.severity]++;
93
+ byCategory[e.category] = (byCategory[e.category] ?? 0) + 1;
94
+ }
95
+ }
96
+ const mqm = totalWords ? 100 - (penalty / totalWords) * 100 : 100;
97
+ return {
98
+ mqm: Math.round(mqm * 100) / 100,
99
+ penalty,
100
+ totalWords,
101
+ unitsAssessed: assessed.length,
102
+ unitsClean: assessed.filter((a) => a.errors.length === 0).length,
103
+ bySeverity,
104
+ byCategory,
105
+ };
106
+ }
107
+
108
+
109
+ /** Parse the judge's per-unit payload. Anything unreadable is dropped, never guessed. */
110
+ export function parseErrors(raw) {
111
+ if (typeof raw !== 'string') return null;
112
+ const text = raw.trim();
113
+ if (!text) return [];
114
+ try {
115
+ const parsed = JSON.parse(text);
116
+ if (!Array.isArray(parsed)) return null;
117
+ return parsed
118
+ .map((e) => ({
119
+ category: String(e.c ?? e.category ?? '').trim(),
120
+ severity: String(e.s ?? e.severity ?? '').trim().toLowerCase(),
121
+ note: String(e.n ?? e.note ?? '').trim(),
122
+ }))
123
+ .filter((e) => WEIGHT[e.severity] !== undefined);
124
+ } catch {
125
+ return null;
126
+ }
127
+ }