claude-translator 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +14 -0
- package/CHANGELOG.md +180 -0
- package/PRIVACY.md +71 -0
- package/README.md +241 -39
- package/bin/claude-translator.mjs +11 -2
- package/bin/cli.test.mjs +20 -1
- package/glossary.example.json +23 -0
- package/i18n.config.example.json +7 -0
- package/package.json +9 -5
- package/scripts/audit-seo.mjs +6 -3
- package/scripts/build-locales.mjs +55 -6
- package/scripts/config.mjs +66 -0
- package/scripts/credit.mjs +12 -5
- package/scripts/extract.mjs +21 -6
- package/scripts/format-locale.mjs +290 -0
- package/scripts/format-locale.test.mjs +171 -0
- package/scripts/glossary.mjs +229 -0
- package/scripts/glossary.test.mjs +188 -0
- package/scripts/roles.mjs +142 -0
- package/scripts/roles.test.mjs +140 -0
- package/scripts/tqa-score.mjs +127 -0
- package/scripts/tqa-score.test.mjs +144 -0
- package/scripts/tqa.mjs +449 -0
- package/scripts/translate.mjs +87 -7
- package/scripts/verify.mjs +113 -5
- package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
- package/{references → skills/translate-site/references}/quality-review.md +28 -0
- /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
- /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
- /package/{references → skills/translate-site/references}/providers.md +0 -0
- /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Element context — telling the translator WHAT a string is, not just what it says.
|
|
3
|
+
*
|
|
4
|
+
* The prompt has always ended rule 5 with "Headings stay headings; button labels stay
|
|
5
|
+
* short." That sentence was unenforceable: the model received `{ id, text }` and had no
|
|
6
|
+
* way to tell a <button> from a paragraph. Every string looked like prose.
|
|
7
|
+
*
|
|
8
|
+
* extract.mjs already knows the answer — it walks the DOM and has `node.tagName` in hand
|
|
9
|
+
* at the moment it records a unit. It simply threw it away. This module turns that tag
|
|
10
|
+
* into a short label the model can act on.
|
|
11
|
+
*
|
|
12
|
+
* ── Why a separate field, not a richer `kind` ────────────────────────────────
|
|
13
|
+
* build-locales.mjs switches on `seg.kind === 'block' | 'jsonld' | startsWith('attr:')`
|
|
14
|
+
* to decide how to escape a replacement. A value like 'block:button' would fall through
|
|
15
|
+
* to escHtml and print literal <0> placeholder tokens as visible text on every
|
|
16
|
+
* localized page — silent corruption that no test catches. So `kind` is untouched and
|
|
17
|
+
* `el` is new.
|
|
18
|
+
*
|
|
19
|
+
* ── Why this costs nothing ───────────────────────────────────────────────────
|
|
20
|
+
* Ordinary prose returns null and the field is omitted from the payload entirely. Only
|
|
21
|
+
* the strings where the answer changes the translation carry it.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Block-level elements whose register genuinely differs from body copy.
|
|
26
|
+
*
|
|
27
|
+
* <a> is the interesting entry. It is an INLINE element in extract.mjs, so a link inside
|
|
28
|
+
* a sentence is absorbed into the surrounding block as a <0>…</0> placeholder and never
|
|
29
|
+
* recorded on its own. An <a> that DOES surface as its own unit is therefore a standalone
|
|
30
|
+
* link — a call to action or a nav item — which makes it a high-precision signal rather
|
|
31
|
+
* than noise.
|
|
32
|
+
*/
|
|
33
|
+
const BLOCK_ROLES = {
|
|
34
|
+
button: 'button',
|
|
35
|
+
a: 'button',
|
|
36
|
+
h1: 'heading',
|
|
37
|
+
h2: 'heading',
|
|
38
|
+
h3: 'heading',
|
|
39
|
+
h4: 'heading',
|
|
40
|
+
h5: 'heading',
|
|
41
|
+
h6: 'heading',
|
|
42
|
+
title: 'page title',
|
|
43
|
+
label: 'form label',
|
|
44
|
+
legend: 'form label',
|
|
45
|
+
summary: 'expander label',
|
|
46
|
+
th: 'table header',
|
|
47
|
+
figcaption: 'caption',
|
|
48
|
+
option: 'menu option',
|
|
49
|
+
optgroup: 'menu option',
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/** Attribute-sourced strings. The attribute name IS the role. */
|
|
53
|
+
const ATTR_ROLES = {
|
|
54
|
+
'attr:alt': 'image alt text',
|
|
55
|
+
'attr:placeholder': 'input placeholder',
|
|
56
|
+
'attr:aria-label': 'accessible label',
|
|
57
|
+
'attr:aria-description': 'accessible description',
|
|
58
|
+
'attr:title': 'tooltip',
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Meta tags carry their key in `el` as `meta:<key>`, because `kind` flattens every one of
|
|
63
|
+
* them to `attr:content` and a description behaves nothing like an og:title.
|
|
64
|
+
*/
|
|
65
|
+
const META_ROLES = {
|
|
66
|
+
description: 'meta description',
|
|
67
|
+
'og:description': 'social share description',
|
|
68
|
+
'twitter:description': 'social share description',
|
|
69
|
+
'og:title': 'social share title',
|
|
70
|
+
'twitter:title': 'social share title',
|
|
71
|
+
'og:site_name': 'site name',
|
|
72
|
+
'og:image:alt': 'image alt text',
|
|
73
|
+
'twitter:image:alt': 'image alt text',
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* The label for one unit, or null when it is ordinary prose.
|
|
78
|
+
*
|
|
79
|
+
* @param {{kind?: string, el?: string|null}} unit a record from i18n/source.json
|
|
80
|
+
*/
|
|
81
|
+
export function roleOf(unit) {
|
|
82
|
+
if (!unit || typeof unit !== 'object') return null;
|
|
83
|
+
const { kind, el } = unit;
|
|
84
|
+
|
|
85
|
+
// A conflicting element cleared the hint at extraction time. Saying nothing is correct
|
|
86
|
+
// here — a confident wrong answer is worse than no answer.
|
|
87
|
+
if (el === null || el === undefined) {
|
|
88
|
+
return kind === 'jsonld' ? 'structured data' : null;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
if (typeof el === 'string' && el.startsWith('meta:')) {
|
|
92
|
+
return META_ROLES[el.slice(5)] ?? 'page metadata';
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
if (typeof kind === 'string' && kind.startsWith('attr:')) {
|
|
96
|
+
return ATTR_ROLES[kind] ?? null;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if (kind === 'jsonld') return 'structured data';
|
|
100
|
+
|
|
101
|
+
return BLOCK_ROLES[el] ?? null;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* The guidance block appended to the system prompt, listing only the roles actually
|
|
106
|
+
* present in this batch. Empty string when the batch is all prose, so a run that
|
|
107
|
+
* translates nothing but paragraphs pays not a single extra token.
|
|
108
|
+
*/
|
|
109
|
+
export function rolePrompt(roles) {
|
|
110
|
+
const present = [...new Set(roles.filter(Boolean))].sort();
|
|
111
|
+
if (!present.length) return '';
|
|
112
|
+
|
|
113
|
+
const ADVICE = {
|
|
114
|
+
button: 'A control the user clicks. Keep it at or near the source length — it sits in a fixed-width box. Use whatever construction your language uses on buttons, which is often not the imperative: an infinitive, a verbal noun, or a bare noun may all read better than a literal command. No final period.',
|
|
115
|
+
'form label': 'Names a field. Short, nominal, no final period.',
|
|
116
|
+
'expander label': 'A short clickable summary. Keep it terse.',
|
|
117
|
+
heading: 'A headline. Keep it headline-shaped and roughly the source length; do not expand it into a sentence.',
|
|
118
|
+
'page title': 'The browser tab and search-result title. Under about 60 characters if the language allows.',
|
|
119
|
+
'meta description': 'Search-result copy. One or two sentences, under about 155 characters, written to be read in a results page.',
|
|
120
|
+
'social share title': 'A share-card headline. Short and concrete.',
|
|
121
|
+
'social share description': 'A share-card summary. One sentence.',
|
|
122
|
+
'image alt text': 'Describes an image for someone who cannot see it. Plain and factual, no "image of".',
|
|
123
|
+
'input placeholder': 'Example text inside an empty field. Very short.',
|
|
124
|
+
'accessible label': 'Read aloud by a screen reader. Say what the control does.',
|
|
125
|
+
'accessible description': 'Extra screen-reader detail. One short phrase.',
|
|
126
|
+
tooltip: 'A hover hint. One short phrase.',
|
|
127
|
+
'table header': 'A column heading. Very short, nominal.',
|
|
128
|
+
caption: 'A figure caption. One short sentence.',
|
|
129
|
+
'menu option': 'One choice in a dropdown. Very short.',
|
|
130
|
+
'site name': 'The name of the site. Usually left unchanged.',
|
|
131
|
+
'page metadata': 'Metadata, not body copy. Keep it compact.',
|
|
132
|
+
'structured data': 'A search-engine structured-data value. Plain text, no markup, no added punctuation.',
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
return [
|
|
136
|
+
`CONTEXT`,
|
|
137
|
+
`Some items carry an "el" field naming what the string is on the page. It is not part`,
|
|
138
|
+
`of the text and must not be translated or echoed. Where it appears, honour it:`,
|
|
139
|
+
...present.map((r) => ` - ${r}: ${ADVICE[r] ?? 'Keep the register appropriate to this element.'}`),
|
|
140
|
+
`Items with no "el" field are body copy — translate them normally.`,
|
|
141
|
+
].join('\n');
|
|
142
|
+
}
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Element-context contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* The prompt has always said "button labels stay short" while the model had no way to
|
|
5
|
+
* know what a button was. These tests pin the mapping that finally makes that actionable,
|
|
6
|
+
* and — more importantly — pin the cases where the tool must say NOTHING rather than
|
|
7
|
+
* guess.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { test } from 'node:test';
|
|
11
|
+
import assert from 'node:assert/strict';
|
|
12
|
+
|
|
13
|
+
import { roleOf, rolePrompt } from './roles.mjs';
|
|
14
|
+
|
|
15
|
+
const unit = (el, kind = 'block') => ({ kind, el });
|
|
16
|
+
|
|
17
|
+
// ── Controls ─────────────────────────────────────────────────────────────────
|
|
18
|
+
|
|
19
|
+
test('a button is a button', () => {
|
|
20
|
+
assert.equal(roleOf(unit('button')), 'button');
|
|
21
|
+
});
|
|
22
|
+
|
|
23
|
+
test('a standalone <a> is treated as a call to action', () => {
|
|
24
|
+
// <a> is INLINE in extract.mjs, so a link inside a sentence is absorbed into the
|
|
25
|
+
// surrounding block as a placeholder and never recorded on its own. An <a> that DOES
|
|
26
|
+
// surface as its own unit is a standalone link — a CTA or a nav item.
|
|
27
|
+
assert.equal(roleOf(unit('a')), 'button');
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
test('headings at every level map to one role', () => {
|
|
31
|
+
for (const h of ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']) {
|
|
32
|
+
assert.equal(roleOf(unit(h)), 'heading', h);
|
|
33
|
+
}
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
test('form and table furniture get their own roles', () => {
|
|
37
|
+
assert.equal(roleOf(unit('label')), 'form label');
|
|
38
|
+
assert.equal(roleOf(unit('legend')), 'form label');
|
|
39
|
+
assert.equal(roleOf(unit('summary')), 'expander label');
|
|
40
|
+
assert.equal(roleOf(unit('th')), 'table header');
|
|
41
|
+
assert.equal(roleOf(unit('figcaption')), 'caption');
|
|
42
|
+
assert.equal(roleOf(unit('option')), 'menu option');
|
|
43
|
+
assert.equal(roleOf(unit('title')), 'page title');
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
// ── Prose says nothing ───────────────────────────────────────────────────────
|
|
47
|
+
|
|
48
|
+
test('ordinary prose returns null so the field is omitted entirely', () => {
|
|
49
|
+
for (const el of ['p', 'div', 'span', 'li', 'td', 'blockquote', 'section']) {
|
|
50
|
+
assert.equal(roleOf(unit(el)), null, el);
|
|
51
|
+
}
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test('an unknown element is prose, not a guess', () => {
|
|
55
|
+
assert.equal(roleOf(unit('marquee')), null);
|
|
56
|
+
assert.equal(roleOf(unit('my-web-component')), null);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
// ── The collision case ───────────────────────────────────────────────────────
|
|
60
|
+
|
|
61
|
+
test('a cleared element yields no hint at all', () => {
|
|
62
|
+
// extract.mjs sets el to null when the same string appears as both a button and a
|
|
63
|
+
// paragraph. Saying nothing is correct: a confident wrong answer is worse than none.
|
|
64
|
+
assert.equal(roleOf(unit(null)), null);
|
|
65
|
+
assert.equal(roleOf({ kind: 'block' }), null, 'a missing el behaves like a cleared one');
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
// ── Attributes ───────────────────────────────────────────────────────────────
|
|
69
|
+
|
|
70
|
+
test('attribute roles come from the attribute, not the element', () => {
|
|
71
|
+
assert.equal(roleOf({ kind: 'attr:alt', el: 'img' }), 'image alt text');
|
|
72
|
+
assert.equal(roleOf({ kind: 'attr:placeholder', el: 'input' }), 'input placeholder');
|
|
73
|
+
assert.equal(roleOf({ kind: 'attr:aria-label', el: 'button' }), 'accessible label');
|
|
74
|
+
assert.equal(roleOf({ kind: 'attr:title', el: 'abbr' }), 'tooltip');
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
test('an unmapped attribute is silent rather than mislabelled', () => {
|
|
78
|
+
assert.equal(roleOf({ kind: 'attr:data-thing', el: 'div' }), null);
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
// ── Meta tags ────────────────────────────────────────────────────────────────
|
|
82
|
+
|
|
83
|
+
test('meta tags are distinguished by key, which kind alone cannot express', () => {
|
|
84
|
+
// Every meta tag arrives as kind "attr:content"; the key travels in el.
|
|
85
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:description' }), 'meta description');
|
|
86
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:title' }), 'social share title');
|
|
87
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:description' }), 'social share description');
|
|
88
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:og:image:alt' }), 'image alt text');
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
test('an unrecognised meta key still says "not body copy"', () => {
|
|
92
|
+
assert.equal(roleOf({ kind: 'attr:content', el: 'meta:custom:thing' }), 'page metadata');
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
// ── JSON-LD ──────────────────────────────────────────────────────────────────
|
|
96
|
+
|
|
97
|
+
test('structured data is labelled even though it has no element', () => {
|
|
98
|
+
assert.equal(roleOf({ kind: 'jsonld', el: null }), 'structured data');
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
// ── Defensive ────────────────────────────────────────────────────────────────
|
|
102
|
+
|
|
103
|
+
test('junk input never throws', () => {
|
|
104
|
+
for (const junk of [null, undefined, 'string', 42, []]) {
|
|
105
|
+
assert.doesNotThrow(() => roleOf(junk));
|
|
106
|
+
}
|
|
107
|
+
assert.equal(roleOf(null), null);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
// ── The prompt block ─────────────────────────────────────────────────────────
|
|
111
|
+
|
|
112
|
+
test('a batch of pure prose produces no prompt text and costs no tokens', () => {
|
|
113
|
+
assert.equal(rolePrompt([]), '');
|
|
114
|
+
assert.equal(rolePrompt([null, null, undefined]), '');
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
test('only the roles present in the batch are described', () => {
|
|
118
|
+
const p = rolePrompt(['button', 'heading']);
|
|
119
|
+
assert.match(p, /- button:/);
|
|
120
|
+
assert.match(p, /- heading:/);
|
|
121
|
+
assert.ok(!/image alt text/.test(p), 'roles absent from the batch must not be described');
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
test('duplicates collapse, so forty buttons describe the role once', () => {
|
|
125
|
+
const p = rolePrompt(Array(40).fill('button'));
|
|
126
|
+
assert.equal((p.match(/- button:/g) ?? []).length, 1);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test('the button advice does not demand the imperative', () => {
|
|
130
|
+
// Languages differ: German UI prefers a verbal noun, French the infinitive. Ordering a
|
|
131
|
+
// literal imperative everywhere is precisely the defect this feature exists to avoid.
|
|
132
|
+
const p = rolePrompt(['button']);
|
|
133
|
+
assert.match(p, /not the imperative/i);
|
|
134
|
+
assert.match(p, /fixed-width/);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test('the prompt tells the model el is metadata, not text to translate', () => {
|
|
138
|
+
const p = rolePrompt(['button']);
|
|
139
|
+
assert.match(p, /must not be translated or echoed/);
|
|
140
|
+
});
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TQA scoring primitives — the parts with no I/O, no network and no config, so they can
|
|
3
|
+
* be tested directly. tqa.mjs owns the run; this file owns the arithmetic and the
|
|
4
|
+
* sampling, which are the parts a reader is entitled to check.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Severity weights. These are the conventional MQM values — minor 1, major 5,
|
|
9
|
+
* critical 10 — kept rather than tuned, so a score here means the same thing it means
|
|
10
|
+
* in any other MQM report.
|
|
11
|
+
*/
|
|
12
|
+
export const WEIGHT = { minor: 1, major: 5, critical: 10 };
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The typology offered to the judge. Deliberately short: a long taxonomy produces
|
|
16
|
+
* category-shopping and inconsistent labelling between runs, and the categories that
|
|
17
|
+
* matter for web localization are these.
|
|
18
|
+
*/
|
|
19
|
+
export const CATEGORIES = [
|
|
20
|
+
'accuracy/mistranslation',
|
|
21
|
+
'accuracy/omission',
|
|
22
|
+
'accuracy/addition',
|
|
23
|
+
'accuracy/untranslated',
|
|
24
|
+
'fluency/grammar',
|
|
25
|
+
'fluency/spelling',
|
|
26
|
+
'fluency/register',
|
|
27
|
+
'fluency/awkward',
|
|
28
|
+
'terminology/inconsistent',
|
|
29
|
+
'terminology/glossary',
|
|
30
|
+
'locale/number-date-currency',
|
|
31
|
+
'style/tone',
|
|
32
|
+
'markup/placeholder',
|
|
33
|
+
];
|
|
34
|
+
|
|
35
|
+
// ── Deterministic sampling ───────────────────────────────────────────────────
|
|
36
|
+
|
|
37
|
+
/** mulberry32 — a small seeded PRNG, so a reported score can be reproduced exactly. */
|
|
38
|
+
export function rng(seed) {
|
|
39
|
+
let a = seed >>> 0;
|
|
40
|
+
return () => {
|
|
41
|
+
a = (a + 0x6d2b79f5) >>> 0;
|
|
42
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
43
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
44
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export const words = (s) => (String(s).trim().match(/\S+/g) ?? []).length;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Stratified sample, weighted by how often a unit appears on the site.
|
|
52
|
+
*
|
|
53
|
+
* A uniform sample over-represents the long tail of one-off strings and under-represents
|
|
54
|
+
* the header, footer and nav that every visitor reads on every page. `count` is already
|
|
55
|
+
* recorded by extract.mjs, so weighting by it costs nothing and makes the score reflect
|
|
56
|
+
* what people actually see. Strata are frequency terciles, sampled proportionally.
|
|
57
|
+
*/
|
|
58
|
+
export function stratifiedSample(units, n, seed) {
|
|
59
|
+
if (units.length <= n) return units.slice();
|
|
60
|
+
const rand = rng(seed);
|
|
61
|
+
const sorted = units.slice().sort((a, b) => (b.count ?? 1) - (a.count ?? 1));
|
|
62
|
+
const third = Math.ceil(sorted.length / 3);
|
|
63
|
+
const strata = [sorted.slice(0, third), sorted.slice(third, third * 2), sorted.slice(third * 2)];
|
|
64
|
+
|
|
65
|
+
const picked = [];
|
|
66
|
+
strata.forEach((stratum, i) => {
|
|
67
|
+
if (!stratum.length) return;
|
|
68
|
+
// The most-repeated third gets half the sample; the rest split the remainder.
|
|
69
|
+
const share = i === 0 ? 0.5 : 0.25;
|
|
70
|
+
const want = Math.min(stratum.length, Math.max(1, Math.round(n * share)));
|
|
71
|
+
const pool = stratum.slice();
|
|
72
|
+
for (let k = 0; k < want && pool.length; k++) {
|
|
73
|
+
picked.push(pool.splice(Math.floor(rand() * pool.length), 1)[0]);
|
|
74
|
+
}
|
|
75
|
+
});
|
|
76
|
+
return picked.slice(0, n);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* MQM score for a set of assessed units.
|
|
82
|
+
*/
|
|
83
|
+
export function score(assessed) {
|
|
84
|
+
const totalWords = assessed.reduce((n, a) => n + a.words, 0);
|
|
85
|
+
let penalty = 0;
|
|
86
|
+
const byCategory = {};
|
|
87
|
+
const bySeverity = { minor: 0, major: 0, critical: 0 };
|
|
88
|
+
|
|
89
|
+
for (const a of assessed) {
|
|
90
|
+
for (const e of a.errors) {
|
|
91
|
+
penalty += WEIGHT[e.severity];
|
|
92
|
+
bySeverity[e.severity]++;
|
|
93
|
+
byCategory[e.category] = (byCategory[e.category] ?? 0) + 1;
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
const mqm = totalWords ? 100 - (penalty / totalWords) * 100 : 100;
|
|
97
|
+
return {
|
|
98
|
+
mqm: Math.round(mqm * 100) / 100,
|
|
99
|
+
penalty,
|
|
100
|
+
totalWords,
|
|
101
|
+
unitsAssessed: assessed.length,
|
|
102
|
+
unitsClean: assessed.filter((a) => a.errors.length === 0).length,
|
|
103
|
+
bySeverity,
|
|
104
|
+
byCategory,
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
/** Parse the judge's per-unit payload. Anything unreadable is dropped, never guessed. */
|
|
110
|
+
export function parseErrors(raw) {
|
|
111
|
+
if (typeof raw !== 'string') return null;
|
|
112
|
+
const text = raw.trim();
|
|
113
|
+
if (!text) return [];
|
|
114
|
+
try {
|
|
115
|
+
const parsed = JSON.parse(text);
|
|
116
|
+
if (!Array.isArray(parsed)) return null;
|
|
117
|
+
return parsed
|
|
118
|
+
.map((e) => ({
|
|
119
|
+
category: String(e.c ?? e.category ?? '').trim(),
|
|
120
|
+
severity: String(e.s ?? e.severity ?? '').trim().toLowerCase(),
|
|
121
|
+
note: String(e.n ?? e.note ?? '').trim(),
|
|
122
|
+
}))
|
|
123
|
+
.filter((e) => WEIGHT[e.severity] !== undefined);
|
|
124
|
+
} catch {
|
|
125
|
+
return null;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TQA scoring contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* A published quality score is only worth what its arithmetic is worth, so the MQM
|
|
5
|
+
* formula, the sampling determinism and the judge-failure handling are all pinned here.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { test } from 'node:test';
|
|
9
|
+
import assert from 'node:assert/strict';
|
|
10
|
+
|
|
11
|
+
import { WEIGHT, rng, words, stratifiedSample, parseErrors, score } from './tqa-score.mjs';
|
|
12
|
+
|
|
13
|
+
const unit = (n, errors = []) => ({ words: n, errors });
|
|
14
|
+
|
|
15
|
+
// ── MQM arithmetic ───────────────────────────────────────────────────────────
|
|
16
|
+
|
|
17
|
+
test('a clean sample scores 100', () => {
|
|
18
|
+
const r = score([unit(50), unit(50)]);
|
|
19
|
+
assert.equal(r.mqm, 100);
|
|
20
|
+
assert.equal(r.penalty, 0);
|
|
21
|
+
assert.equal(r.unitsClean, 2);
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
test('severity weights are the conventional MQM values', () => {
|
|
25
|
+
assert.deepEqual(WEIGHT, { minor: 1, major: 5, critical: 10 });
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
test('the score is penalty per 100 words', () => {
|
|
29
|
+
// one major (5 points) over 100 words = 5 points per 100 words = 95.
|
|
30
|
+
const r = score([unit(100, [{ category: 'accuracy/mistranslation', severity: 'major' }])]);
|
|
31
|
+
assert.equal(r.mqm, 95);
|
|
32
|
+
|
|
33
|
+
// the same error over 500 words is a fifth of the penalty density.
|
|
34
|
+
const r2 = score([unit(500, [{ category: 'accuracy/mistranslation', severity: 'major' }])]);
|
|
35
|
+
assert.equal(r2.mqm, 99);
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
test('a critical error costs ten times a minor one', () => {
|
|
39
|
+
const minor = score([unit(100, [{ category: 'fluency/grammar', severity: 'minor' }])]);
|
|
40
|
+
const critical = score([unit(100, [{ category: 'accuracy/mistranslation', severity: 'critical' }])]);
|
|
41
|
+
assert.equal(100 - minor.mqm, 1);
|
|
42
|
+
assert.equal(100 - critical.mqm, 10);
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
test('errors are counted by severity and category', () => {
|
|
46
|
+
const r = score([
|
|
47
|
+
unit(100, [
|
|
48
|
+
{ category: 'fluency/grammar', severity: 'minor' },
|
|
49
|
+
{ category: 'fluency/grammar', severity: 'major' },
|
|
50
|
+
]),
|
|
51
|
+
unit(100, [{ category: 'terminology/glossary', severity: 'critical' }]),
|
|
52
|
+
]);
|
|
53
|
+
assert.deepEqual(r.bySeverity, { minor: 1, major: 1, critical: 1 });
|
|
54
|
+
assert.deepEqual(r.byCategory, { 'fluency/grammar': 2, 'terminology/glossary': 1 });
|
|
55
|
+
assert.equal(r.unitsClean, 0);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test('an empty assessment scores 100 rather than dividing by zero', () => {
|
|
59
|
+
const r = score([]);
|
|
60
|
+
assert.equal(r.mqm, 100);
|
|
61
|
+
assert.equal(r.totalWords, 0);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
// ── Sampling ─────────────────────────────────────────────────────────────────
|
|
65
|
+
|
|
66
|
+
const pool = (n) => Array.from({ length: n }, (_, i) => ({ hash: `h${i}`, count: n - i }));
|
|
67
|
+
|
|
68
|
+
test('sampling is deterministic for a given seed', () => {
|
|
69
|
+
const a = stratifiedSample(pool(300), 50, 42).map((u) => u.hash);
|
|
70
|
+
const b = stratifiedSample(pool(300), 50, 42).map((u) => u.hash);
|
|
71
|
+
assert.deepEqual(a, b, 'the same seed must reproduce the same sample');
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
test('a different seed gives a different sample', () => {
|
|
75
|
+
const a = stratifiedSample(pool(300), 50, 1).map((u) => u.hash);
|
|
76
|
+
const b = stratifiedSample(pool(300), 50, 2).map((u) => u.hash);
|
|
77
|
+
assert.notDeepEqual(a, b);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
test('a pool smaller than the sample is returned whole', () => {
|
|
81
|
+
assert.equal(stratifiedSample(pool(10), 100, 1).length, 10);
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test('the sample never exceeds the requested size', () => {
|
|
85
|
+
assert.equal(stratifiedSample(pool(1000), 100, 1).length, 100);
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
test('frequent strings are over-represented, because visitors read them more', () => {
|
|
89
|
+
// pool(300) has count descending, so the top third is the most-repeated third.
|
|
90
|
+
const picked = stratifiedSample(pool(300), 90, 7);
|
|
91
|
+
const fromTopThird = picked.filter((u) => Number(u.hash.slice(1)) < 100).length;
|
|
92
|
+
assert.ok(fromTopThird > 30, `expected the top third to be over-sampled, got ${fromTopThird}/90`);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
test('the PRNG is stable across runs', () => {
|
|
96
|
+
const a = [rng(123)(), rng(123)(), rng(123)()];
|
|
97
|
+
assert.equal(a[0], a[1]);
|
|
98
|
+
assert.equal(a[1], a[2]);
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
// ── Word counting ────────────────────────────────────────────────────────────
|
|
102
|
+
|
|
103
|
+
test('word counting ignores surrounding whitespace', () => {
|
|
104
|
+
assert.equal(words(' one two three '), 3);
|
|
105
|
+
assert.equal(words(''), 0);
|
|
106
|
+
assert.equal(words('single'), 1);
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
// ── Judge output parsing ─────────────────────────────────────────────────────
|
|
110
|
+
|
|
111
|
+
test('an empty error list parses as clean', () => {
|
|
112
|
+
assert.deepEqual(parseErrors('[]'), []);
|
|
113
|
+
assert.deepEqual(parseErrors(' '), []);
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
test('short and long field names both parse', () => {
|
|
117
|
+
const short = parseErrors('[{"c":"fluency/grammar","s":"minor","n":"agreement"}]');
|
|
118
|
+
assert.equal(short.length, 1);
|
|
119
|
+
assert.equal(short[0].category, 'fluency/grammar');
|
|
120
|
+
assert.equal(short[0].severity, 'minor');
|
|
121
|
+
|
|
122
|
+
const long = parseErrors('[{"category":"fluency/grammar","severity":"major","note":"x"}]');
|
|
123
|
+
assert.equal(long[0].severity, 'major');
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
test('an unknown severity is dropped rather than scored', () => {
|
|
127
|
+
const r = parseErrors('[{"c":"fluency/grammar","s":"catastrophic"}]');
|
|
128
|
+
assert.deepEqual(r, [], 'a severity with no weight cannot be scored, so it is not counted');
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test('unreadable judge output is null, NOT an empty list', () => {
|
|
132
|
+
// This distinction is load-bearing. Treating a failed judgement as "no errors found"
|
|
133
|
+
// would inflate the score every time the judge failed — exactly backwards.
|
|
134
|
+
assert.equal(parseErrors('not json'), null);
|
|
135
|
+
assert.equal(parseErrors('{"not":"an array"}'), null);
|
|
136
|
+
assert.equal(parseErrors(undefined), null);
|
|
137
|
+
assert.equal(parseErrors(null), null);
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
test('severity casing from the model is normalised', () => {
|
|
141
|
+
const r = parseErrors('[{"c":"fluency/grammar","s":"MAJOR"}]');
|
|
142
|
+
assert.equal(r.length, 1);
|
|
143
|
+
assert.equal(r[0].severity, 'major');
|
|
144
|
+
});
|