claude-translator 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +14 -0
- package/CHANGELOG.md +249 -0
- package/PRIVACY.md +71 -0
- package/README.md +279 -44
- package/bin/claude-translator.mjs +11 -2
- package/bin/cli.test.mjs +20 -1
- package/glossary.example.json +23 -0
- package/i18n.config.example.json +7 -0
- package/package.json +9 -5
- package/scripts/audit-seo.mjs +6 -3
- package/scripts/build-locales.mjs +55 -6
- package/scripts/config.mjs +66 -0
- package/scripts/credit.mjs +12 -5
- package/scripts/extract.mjs +21 -6
- package/scripts/format-locale.mjs +290 -0
- package/scripts/format-locale.test.mjs +171 -0
- package/scripts/glossary.mjs +229 -0
- package/scripts/glossary.test.mjs +188 -0
- package/scripts/providers/openai.mjs +63 -3
- package/scripts/providers/providers.test.mjs +80 -0
- package/scripts/roles.mjs +142 -0
- package/scripts/roles.test.mjs +140 -0
- package/scripts/tqa-score.mjs +127 -0
- package/scripts/tqa-score.test.mjs +144 -0
- package/scripts/tqa.mjs +449 -0
- package/scripts/translate.mjs +104 -13
- package/scripts/verify.mjs +113 -5
- package/{SKILL.md → skills/translate-site/SKILL.md} +45 -13
- package/{references → skills/translate-site/references}/providers.md +16 -2
- package/{references → skills/translate-site/references}/quality-review.md +28 -0
- /package/{references → skills/translate-site/references}/adapting-generators.md +0 -0
- /package/{references → skills/translate-site/references}/failure-modes.md +0 -0
- /package/{references → skills/translate-site/references}/throughput-and-cost.md +0 -0
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Locale-formatting contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* The "must not touch" block is the important half. Rewriting a version number, a time
|
|
5
|
+
* or an IP address into a localised decimal is worse than doing nothing at all, and it
|
|
6
|
+
* would ship silently — the markup would still be byte-identical and every other gate
|
|
7
|
+
* would pass.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { test } from 'node:test';
|
|
11
|
+
import assert from 'node:assert/strict';
|
|
12
|
+
|
|
13
|
+
import { formatText, intlLocale } from './format-locale.mjs';
|
|
14
|
+
|
|
15
|
+
const fmt = (text, locale, opts) => formatText(text, locale, opts).text;
|
|
16
|
+
/** Intl uses NBSP/narrow-NBSP; normalise so assertions stay readable. */
|
|
17
|
+
const norm = (s) => s.replace(/[ ]/g, ' ');
|
|
18
|
+
|
|
19
|
+
// ── Grouped numbers ──────────────────────────────────────────────────────────
|
|
20
|
+
|
|
21
|
+
test('grouped numbers take the target locale separators', () => {
|
|
22
|
+
assert.equal(norm(fmt('We processed 1,234,567 files', 'de')), 'We processed 1.234.567 files');
|
|
23
|
+
assert.equal(norm(fmt('We processed 1,234,567 files', 'fr')), 'We processed 1 234 567 files');
|
|
24
|
+
assert.equal(norm(fmt('We processed 1,234,567 files', 'en')), 'We processed 1,234,567 files');
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
test('a grouped decimal keeps its value and its precision', () => {
|
|
28
|
+
assert.equal(norm(fmt('1,234.50 units', 'de')), '1.234,50 units');
|
|
29
|
+
assert.equal(norm(fmt('1,234.5 units', 'de')), '1.234,5 units');
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
test('an ungrouped number is left alone — there is no convention to apply', () => {
|
|
33
|
+
assert.equal(fmt('We support 50 languages', 'de'), 'We support 50 languages');
|
|
34
|
+
assert.equal(fmt('Founded in 2026', 'de'), 'Founded in 2026');
|
|
35
|
+
assert.equal(fmt('Page 7', 'fr'), 'Page 7');
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
// ── Currency: formatted, never converted ─────────────────────────────────────
|
|
39
|
+
|
|
40
|
+
test('currency symbol placement follows the locale, value unchanged', () => {
|
|
41
|
+
assert.equal(norm(fmt('$5', 'fr')), '5,00 $US');
|
|
42
|
+
assert.equal(norm(fmt('$1,234.50', 'de')), '1.234,50 $');
|
|
43
|
+
assert.match(norm(fmt('$5', 'en')), /^\$5\.00$/);
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
test('the numeric value of a price is never altered', () => {
|
|
47
|
+
for (const locale of ['de', 'fr', 'ja', 'es', 'ar']) {
|
|
48
|
+
const out = fmt('$49', locale);
|
|
49
|
+
assert.ok(/49/.test(out), `${locale}: 49 must survive, got ${out}`);
|
|
50
|
+
assert.ok(!/4[0-8](\D|$)/.test(out.replace(/49/g, '')), `${locale}: no other amount may appear`);
|
|
51
|
+
}
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test('the currency is held constant — no conversion, ever', () => {
|
|
55
|
+
const out = norm(fmt('$100', 'de'));
|
|
56
|
+
assert.ok(!out.includes('€'), `must not become euros, got ${out}`);
|
|
57
|
+
assert.ok(out.includes('$'), `must stay dollars, got ${out}`);
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
test('a trailing ISO currency code is recognised', () => {
|
|
61
|
+
assert.equal(norm(fmt('1,500 USD', 'de')), '1.500,00 $');
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test('currency formatting can be switched off', () => {
|
|
65
|
+
assert.equal(fmt('$1,234.50', 'de', { currency: 'off', numbers: false }), '$1,234.50');
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
// ── Percent ──────────────────────────────────────────────────────────────────
|
|
69
|
+
|
|
70
|
+
test('percent spacing follows the locale', () => {
|
|
71
|
+
assert.equal(norm(fmt('Save 50%', 'fr')), 'Save 50 %');
|
|
72
|
+
assert.equal(norm(fmt('Save 50%', 'en')), 'Save 50%');
|
|
73
|
+
assert.equal(norm(fmt('Save 12.5%', 'de')), 'Save 12,5 %');
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
// ── Must not touch ───────────────────────────────────────────────────────────
|
|
77
|
+
|
|
78
|
+
test('semantic version numbers are left alone', () => {
|
|
79
|
+
for (const s of ['Requires Node 20.5.1', 'Version 1.2.3 is out', 'v2.0 ships today']) {
|
|
80
|
+
assert.equal(fmt(s, 'de'), s, s);
|
|
81
|
+
}
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test('times are left alone', () => {
|
|
85
|
+
for (const s of ['Opens at 10:30', 'Between 9:00 and 17:45']) {
|
|
86
|
+
assert.equal(fmt(s, 'de'), s, s);
|
|
87
|
+
}
|
|
88
|
+
});
|
|
89
|
+
|
|
90
|
+
test('IP addresses are left alone', () => {
|
|
91
|
+
assert.equal(fmt('Connect to 192.168.1.1 now', 'de'), 'Connect to 192.168.1.1 now');
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
test('ISO dates are left alone', () => {
|
|
95
|
+
assert.equal(fmt('Released 2026-08-29', 'de'), 'Released 2026-08-29');
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
test('phone numbers are left alone', () => {
|
|
99
|
+
for (const s of ['Call +1-800-555-0199', 'Call 1-800-555-0199']) {
|
|
100
|
+
assert.equal(fmt(s, 'de'), s, s);
|
|
101
|
+
}
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
test('fractions and ratios are left alone', () => {
|
|
105
|
+
assert.equal(fmt('A 1/2 scale model', 'de'), 'A 1/2 scale model');
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
test('file sizes without grouping are left alone', () => {
|
|
109
|
+
assert.equal(fmt('Upload up to 10 MB', 'de'), 'Upload up to 10 MB');
|
|
110
|
+
assert.equal(fmt('1 GB of storage', 'fr'), '1 GB of storage');
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
test('placeholders are never entered or altered', () => {
|
|
114
|
+
const src = 'Save <0>1,234</0> files and <1/> more';
|
|
115
|
+
const out = fmt(src, 'de');
|
|
116
|
+
assert.ok(out.includes('<0>'), 'opening placeholder survives');
|
|
117
|
+
assert.ok(out.includes('</0>'), 'closing placeholder survives');
|
|
118
|
+
assert.ok(out.includes('<1/>'), 'self-closing placeholder survives');
|
|
119
|
+
assert.equal((out.match(/<\/?\d+\/?>/g) ?? []).length, 3);
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
// ── Reporting ────────────────────────────────────────────────────────────────
|
|
123
|
+
|
|
124
|
+
test('monetary amounts are reported for human review', () => {
|
|
125
|
+
const { money } = formatText('Plans from $9 to $99 per month', 'de');
|
|
126
|
+
assert.equal(money.length, 2);
|
|
127
|
+
assert.deepEqual(money.map((m) => m.value), [9, 99]);
|
|
128
|
+
assert.ok(money.every((m) => m.currency === 'USD'));
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test('changes are itemised so a build can show its work', () => {
|
|
132
|
+
const { changes } = formatText('1,000 items at 50%', 'de');
|
|
133
|
+
assert.ok(changes.some((c) => c.kind === 'number'));
|
|
134
|
+
assert.ok(changes.some((c) => c.kind === 'percent'));
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test('a string with nothing to format is returned byte-identical', () => {
|
|
138
|
+
const src = 'Translate your website into any language';
|
|
139
|
+
const { text, changes, money } = formatText(src, 'de');
|
|
140
|
+
assert.equal(text, src);
|
|
141
|
+
assert.equal(changes.length, 0);
|
|
142
|
+
assert.equal(money.length, 0);
|
|
143
|
+
});
|
|
144
|
+
|
|
145
|
+
test('empty and non-string input is survivable', () => {
|
|
146
|
+
assert.equal(formatText('', 'de').text, '');
|
|
147
|
+
assert.equal(formatText(null, 'de').text, null);
|
|
148
|
+
assert.equal(formatText(undefined, 'de').text, undefined);
|
|
149
|
+
});
|
|
150
|
+
|
|
151
|
+
// ── Locale tag ───────────────────────────────────────────────────────────────
|
|
152
|
+
|
|
153
|
+
test('Intl gets the hreflang tag, not the URL path code', () => {
|
|
154
|
+
assert.equal(intlLocale({ hreflang: 'pt-BR', pathCode: 'pt-br' }), 'pt-BR');
|
|
155
|
+
assert.equal(intlLocale({ hreflang: 'zh-Hant', pathCode: 'zh-tw' }), 'zh-Hant');
|
|
156
|
+
assert.equal(intlLocale({ pathCode: 'es' }), 'es');
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
test('an unknown locale degrades to leaving text alone, never to a crash', () => {
|
|
160
|
+
const out = formatText('1,234 items', 'not-a-locale');
|
|
161
|
+
assert.equal(typeof out.text, 'string');
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
test('Spanish does not group four-digit numbers, but German does — this is not a bug', () => {
|
|
165
|
+
// CLDR minimumGroupingDigits=2 for es. "1234,50" is correct Spanish and "1.234,50" is
|
|
166
|
+
// correct German for the same amount. This looked like a lost separator during
|
|
167
|
+
// development and is the exact thing a future reader will try to "fix".
|
|
168
|
+
assert.equal(norm(fmt('$1,234.50', 'es')), '1234,50 US$');
|
|
169
|
+
assert.equal(norm(fmt('$1,234.50', 'de')), '1.234,50 $');
|
|
170
|
+
assert.equal(norm(fmt('12,345 items', 'es')), '12.345 items', 'five digits DO group in Spanish');
|
|
171
|
+
});
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Glossary — terminology consistency and brand protection.
|
|
3
|
+
*
|
|
4
|
+
* Two questions this answers that `doNotTranslate` could not:
|
|
5
|
+
*
|
|
6
|
+
* 1. "Dashboard" must be "Panel de control" every time it appears, including inside
|
|
7
|
+
* sentences that are otherwise all different. The translation memory already keeps
|
|
8
|
+
* IDENTICAL strings consistent — it is keyed by the hash of the whole unit — but a
|
|
9
|
+
* term inside varying sentences lands in different units, different batches, and
|
|
10
|
+
* different stateless requests. Nothing compared them. A term base does.
|
|
11
|
+
*
|
|
12
|
+
* 2. "Apple" the company must survive; "apple" the fruit must be translated. The old
|
|
13
|
+
* list could not express the difference: it was a flat array of strings, matched
|
|
14
|
+
* against the WHOLE trimmed unit and pasted into one prompt line. Case sensitivity
|
|
15
|
+
* plus word boundaries is the whole answer, and it needs a per-term flag.
|
|
16
|
+
*
|
|
17
|
+
* ── Shape ────────────────────────────────────────────────────────────────────
|
|
18
|
+
* Deliberately the same four fields the managed ConveyThis product stores per glossary
|
|
19
|
+
* row (rule / source_text / translate_text / target_language), so a glossary can move
|
|
20
|
+
* between the two without a translation layer.
|
|
21
|
+
*
|
|
22
|
+
* { "source": "Dashboard", "rule": "translate",
|
|
23
|
+
* "targets": { "es": "Panel de control", "de": "Dashboard" } }
|
|
24
|
+
* { "source": "Apple", "rule": "keep", "matchCase": true }
|
|
25
|
+
*
|
|
26
|
+
* `rule: "keep"` means leave it in the source language. `rule: "translate"` with
|
|
27
|
+
* `targets` pins the wording per locale; a locale with no entry in `targets` is simply
|
|
28
|
+
* not constrained, which is the honest default for a term nobody has decided yet.
|
|
29
|
+
*
|
|
30
|
+
* ── Why matching is conservative ─────────────────────────────────────────────
|
|
31
|
+
* Every match here becomes either a prompt instruction or a compliance finding, and
|
|
32
|
+
* `references/quality-review.md` is unambiguous that a heuristic which over-flags is
|
|
33
|
+
* worse than no heuristic. So: word boundaries always, case-sensitivity opt-in per term,
|
|
34
|
+
* and longest-term-first so "Acme Cloud Inc" wins over "Acme".
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import { readFileSync, existsSync } from 'fs';
|
|
38
|
+
|
|
39
|
+
/** Unicode-aware-ish word boundary: the char before/after a term must not be a letter,
|
|
40
|
+
* digit or underscore. Built by hand because JS \b is ASCII-only, so "Straße" would
|
|
41
|
+
* boundary wrongly against a following accented letter. */
|
|
42
|
+
const WORDISH = /[\p{L}\p{N}_]/u;
|
|
43
|
+
|
|
44
|
+
const isWordish = (ch) => ch !== undefined && WORDISH.test(ch);
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Normalise one raw entry. Returns null for anything unusable rather than throwing,
|
|
48
|
+
* because a single malformed row must not take a 40-locale run down with it — the
|
|
49
|
+
* caller reports the count of skipped rows instead.
|
|
50
|
+
*/
|
|
51
|
+
function normalise(entry, index, problems) {
|
|
52
|
+
if (!entry || typeof entry !== 'object') {
|
|
53
|
+
problems.push(`entry ${index}: not an object`);
|
|
54
|
+
return null;
|
|
55
|
+
}
|
|
56
|
+
const source = typeof entry.source === 'string' ? entry.source.trim() : '';
|
|
57
|
+
if (!source) {
|
|
58
|
+
problems.push(`entry ${index}: missing "source"`);
|
|
59
|
+
return null;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const rule = entry.rule ?? 'translate';
|
|
63
|
+
if (rule !== 'keep' && rule !== 'translate') {
|
|
64
|
+
problems.push(`entry ${index} ("${source}"): rule must be "keep" or "translate", got "${rule}"`);
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const targets = {};
|
|
69
|
+
if (entry.targets && typeof entry.targets === 'object') {
|
|
70
|
+
for (const [lang, value] of Object.entries(entry.targets)) {
|
|
71
|
+
if (typeof value === 'string' && value.trim()) targets[lang] = value.trim();
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
if (rule === 'translate' && Object.keys(targets).length === 0) {
|
|
75
|
+
// Not an error: a term listed with no targets yet is still worth tracking for
|
|
76
|
+
// consistency reporting. It just cannot constrain anything.
|
|
77
|
+
problems.push(`entry ${index} ("${source}"): rule "translate" with no targets — ignored`);
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
return {
|
|
82
|
+
source,
|
|
83
|
+
rule,
|
|
84
|
+
targets,
|
|
85
|
+
// Brands are the case-sensitive case, so "keep" defaults to case-sensitive and
|
|
86
|
+
// "translate" does not. Either can be overridden explicitly.
|
|
87
|
+
matchCase: entry.matchCase ?? rule === 'keep',
|
|
88
|
+
note: typeof entry.note === 'string' ? entry.note : null,
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Load and validate. Accepts an array (inline in i18n.config.json) or a path to a JSON
|
|
94
|
+
* file — the same dual form `locales` already takes.
|
|
95
|
+
*/
|
|
96
|
+
export function loadGlossary(src, root, resolvePath) {
|
|
97
|
+
if (!src) return { terms: [], problems: [] };
|
|
98
|
+
|
|
99
|
+
let rawEntries = src;
|
|
100
|
+
if (typeof src === 'string') {
|
|
101
|
+
const p = resolvePath(root, src);
|
|
102
|
+
if (!existsSync(p)) {
|
|
103
|
+
console.error(`config.glossary points at ${p}, which does not exist`);
|
|
104
|
+
process.exit(1);
|
|
105
|
+
}
|
|
106
|
+
try {
|
|
107
|
+
rawEntries = JSON.parse(readFileSync(p, 'utf8'));
|
|
108
|
+
} catch (err) {
|
|
109
|
+
console.error(`config.glossary: ${p} is not valid JSON — ${err.message}`);
|
|
110
|
+
process.exit(1);
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
if (!Array.isArray(rawEntries)) {
|
|
115
|
+
console.error('config.glossary must be an array, or a path to a JSON file containing one');
|
|
116
|
+
process.exit(1);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
const problems = [];
|
|
120
|
+
const terms = rawEntries.map((e, i) => normalise(e, i, problems)).filter(Boolean);
|
|
121
|
+
|
|
122
|
+
// Longest first, so "Acme Cloud Inc" is matched and consumed before "Acme".
|
|
123
|
+
terms.sort((a, b) => b.source.length - a.source.length);
|
|
124
|
+
return { terms, problems };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Does `text` contain `term` as a whole word?
|
|
129
|
+
*
|
|
130
|
+
* Case-insensitive matching still requires the boundary check, so "apple" does not
|
|
131
|
+
* match inside "pineapple" — the false positive that would otherwise make every
|
|
132
|
+
* fruit-adjacent sentence look like a brand violation.
|
|
133
|
+
*/
|
|
134
|
+
export function containsTerm(text, term) {
|
|
135
|
+
if (!text || !term?.source) return false;
|
|
136
|
+
const haystack = term.matchCase ? text : text.toLowerCase();
|
|
137
|
+
const needle = term.matchCase ? term.source : term.source.toLowerCase();
|
|
138
|
+
|
|
139
|
+
let from = 0;
|
|
140
|
+
for (;;) {
|
|
141
|
+
const at = haystack.indexOf(needle, from);
|
|
142
|
+
if (at === -1) return false;
|
|
143
|
+
const before = at === 0 ? undefined : haystack[at - 1];
|
|
144
|
+
const after = haystack[at + needle.length];
|
|
145
|
+
if (!isWordish(before) && !isWordish(after)) return true;
|
|
146
|
+
from = at + 1;
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* The subset of the glossary that actually appears in this batch.
|
|
152
|
+
*
|
|
153
|
+
* This is the whole reason the feature is affordable. A 500-term glossary pasted into
|
|
154
|
+
* the system prompt of every 40-unit batch would dominate token cost on a large site
|
|
155
|
+
* and push the real instructions out of the model's attention. Sending only the terms
|
|
156
|
+
* present in the batch keeps prompt growth proportional to what is actually at stake.
|
|
157
|
+
*/
|
|
158
|
+
export function termsForBatch(terms, texts, lang) {
|
|
159
|
+
if (!terms.length) return [];
|
|
160
|
+
return terms.filter((term) => {
|
|
161
|
+
// A "translate" term with no target for THIS locale constrains nothing, so it would
|
|
162
|
+
// only be noise in the prompt.
|
|
163
|
+
if (term.rule === 'translate' && !term.targets[lang]) return false;
|
|
164
|
+
return texts.some((t) => containsTerm(t, term));
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** The prompt lines for a batch's terms. Empty string when there is nothing to say. */
|
|
169
|
+
export function glossaryPrompt(terms, lang) {
|
|
170
|
+
if (!terms.length) return '';
|
|
171
|
+
const lines = terms.map((t) => {
|
|
172
|
+
const how = t.rule === 'keep'
|
|
173
|
+
? `leave exactly as "${t.source}"`
|
|
174
|
+
: `always render as "${t.targets[lang]}"`;
|
|
175
|
+
const cased = t.matchCase ? ' (this capitalisation only)' : '';
|
|
176
|
+
return ` - "${t.source}"${cased}: ${how}${t.note ? ` — ${t.note}` : ''}`;
|
|
177
|
+
});
|
|
178
|
+
return [
|
|
179
|
+
`TERMINOLOGY — these terms appear in this batch and are not free choices:`,
|
|
180
|
+
...lines,
|
|
181
|
+
` Match whole words only. A term that is part of a longer word is a different word.`,
|
|
182
|
+
` Inflect for grammar where the target language requires it, but keep the stem.`,
|
|
183
|
+
].join('\n');
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Compliance check for one translated unit.
|
|
188
|
+
*
|
|
189
|
+
* Returns a list of violations, each { source, expected, rule }. Deliberately NOT a
|
|
190
|
+
* hard failure by default: target languages inflect ("Panel de control" →
|
|
191
|
+
* "del Panel de control"), compound ("Dashboard-Ansicht"), and decline, so a strict
|
|
192
|
+
* substring test would flag correct translations constantly. The caller decides.
|
|
193
|
+
*/
|
|
194
|
+
export function checkCompliance(sourceText, translatedText, terms, lang) {
|
|
195
|
+
const violations = [];
|
|
196
|
+
for (const term of terms) {
|
|
197
|
+
if (!containsTerm(sourceText, term)) continue;
|
|
198
|
+
|
|
199
|
+
if (term.rule === 'keep') {
|
|
200
|
+
// "keep" is the strict one and can afford to be: the term is supposed to come
|
|
201
|
+
// through byte-identical, so a plain case-sensitive substring test is right.
|
|
202
|
+
if (!translatedText.includes(term.source)) {
|
|
203
|
+
violations.push({ source: term.source, expected: term.source, rule: 'keep' });
|
|
204
|
+
}
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
const expected = term.targets[lang];
|
|
209
|
+
if (!expected) continue;
|
|
210
|
+
// Case-insensitive and boundary-free on purpose — this is the inflection-tolerant
|
|
211
|
+
// side of the check. "Panel de control" inside "del Panel de control" passes.
|
|
212
|
+
if (!translatedText.toLowerCase().includes(expected.toLowerCase())) {
|
|
213
|
+
violations.push({ source: term.source, expected, rule: 'translate' });
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
return violations;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* A stable fingerprint of the glossary, stored in the translation memory so a changed
|
|
221
|
+
* glossary can invalidate exactly the units it affects rather than the whole locale.
|
|
222
|
+
*/
|
|
223
|
+
export function glossaryFingerprint(terms) {
|
|
224
|
+
const canonical = terms
|
|
225
|
+
.map((t) => `${t.rule}|${t.matchCase ? 'C' : 'i'}|${t.source}|${JSON.stringify(t.targets)}`)
|
|
226
|
+
.sort()
|
|
227
|
+
.join('\n');
|
|
228
|
+
return canonical;
|
|
229
|
+
}
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Glossary contract tests. No network, no key, no model.
|
|
3
|
+
*
|
|
4
|
+
* The case that motivated the feature is the first one: a user asked whether the tool
|
|
5
|
+
* can tell Apple (the company) from apple (the fruit). Before this module it could not
|
|
6
|
+
* — the do-not-translate list was matched against the whole trimmed unit and pasted
|
|
7
|
+
* into one prompt line with no case guidance.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { test } from 'node:test';
|
|
11
|
+
import assert from 'node:assert/strict';
|
|
12
|
+
import { tmpdir } from 'os';
|
|
13
|
+
import { mkdtempSync, writeFileSync } from 'fs';
|
|
14
|
+
import { join, resolve } from 'path';
|
|
15
|
+
|
|
16
|
+
import {
|
|
17
|
+
loadGlossary, containsTerm, termsForBatch, glossaryPrompt,
|
|
18
|
+
checkCompliance, glossaryFingerprint,
|
|
19
|
+
} from './glossary.mjs';
|
|
20
|
+
|
|
21
|
+
const load = (entries) => loadGlossary(entries, '/nowhere', resolve);
|
|
22
|
+
const term = (entry) => load([entry]).terms[0];
|
|
23
|
+
|
|
24
|
+
// ── The Apple problem ────────────────────────────────────────────────────────
|
|
25
|
+
|
|
26
|
+
test('a case-sensitive brand matches the company but not the fruit', () => {
|
|
27
|
+
const apple = term({ source: 'Apple', rule: 'keep' });
|
|
28
|
+
assert.equal(apple.matchCase, true, 'rule "keep" must default to case-sensitive');
|
|
29
|
+
|
|
30
|
+
assert.ok(containsTerm('Apple announced a new device', apple));
|
|
31
|
+
assert.ok(!containsTerm('I ate an apple for lunch', apple));
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
test('a brand does not match inside a longer word', () => {
|
|
35
|
+
const apple = term({ source: 'Apple', rule: 'keep' });
|
|
36
|
+
assert.ok(!containsTerm('Applesauce is on sale', apple));
|
|
37
|
+
assert.ok(!containsTerm('The Appleton office', apple));
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
test('a case-insensitive term still respects word boundaries', () => {
|
|
41
|
+
const t = term({ source: 'apple', rule: 'keep', matchCase: false });
|
|
42
|
+
assert.ok(containsTerm('An apple a day', t));
|
|
43
|
+
assert.ok(containsTerm('Apple pie recipe', t), 'case-insensitive should match either case');
|
|
44
|
+
assert.ok(!containsTerm('pineapple juice', t), 'must not match inside pineapple');
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test('sentence-initial and punctuation-adjacent terms match', () => {
|
|
48
|
+
const acme = term({ source: 'Acme', rule: 'keep' });
|
|
49
|
+
assert.ok(containsTerm('Acme is great', acme), 'start of string');
|
|
50
|
+
assert.ok(containsTerm('We love Acme.', acme), 'followed by a period');
|
|
51
|
+
assert.ok(containsTerm('(Acme)', acme), 'wrapped in parentheses');
|
|
52
|
+
assert.ok(containsTerm('Acme', acme), 'the whole string');
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
// ── Term base ────────────────────────────────────────────────────────────────
|
|
56
|
+
|
|
57
|
+
test('longer terms sort before shorter ones so the specific name wins', () => {
|
|
58
|
+
const { terms } = load([
|
|
59
|
+
{ source: 'Acme', rule: 'keep' },
|
|
60
|
+
{ source: 'Acme Cloud Inc', rule: 'keep' },
|
|
61
|
+
]);
|
|
62
|
+
assert.equal(terms[0].source, 'Acme Cloud Inc');
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
test('per-locale targets pin the wording', () => {
|
|
66
|
+
const t = term({ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control', de: 'Dashboard' } });
|
|
67
|
+
assert.equal(t.targets.es, 'Panel de control');
|
|
68
|
+
assert.equal(t.matchCase, false, 'rule "translate" defaults to case-insensitive');
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
test('a "translate" term with no targets is dropped, with a reported reason', () => {
|
|
72
|
+
const { terms, problems } = load([{ source: 'Widget', rule: 'translate' }]);
|
|
73
|
+
assert.equal(terms.length, 0);
|
|
74
|
+
assert.match(problems[0], /no targets/);
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
test('malformed rows are skipped, not fatal', () => {
|
|
78
|
+
const { terms, problems } = load([
|
|
79
|
+
{ source: 'Acme', rule: 'keep' },
|
|
80
|
+
{ source: '', rule: 'keep' },
|
|
81
|
+
{ rule: 'keep' },
|
|
82
|
+
{ source: 'Bad', rule: 'nonsense' },
|
|
83
|
+
'not an object',
|
|
84
|
+
]);
|
|
85
|
+
assert.equal(terms.length, 1, 'the one good row survives');
|
|
86
|
+
assert.equal(problems.length, 4);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
// ── Batch subsetting (the cost control) ──────────────────────────────────────
|
|
90
|
+
|
|
91
|
+
test('only terms present in the batch are selected', () => {
|
|
92
|
+
const { terms } = load([
|
|
93
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control' } },
|
|
94
|
+
{ source: 'Workspace', rule: 'translate', targets: { es: 'Espacio de trabajo' } },
|
|
95
|
+
{ source: 'Apple', rule: 'keep' },
|
|
96
|
+
]);
|
|
97
|
+
const picked = termsForBatch(terms, ['Open the Dashboard to begin'], 'es');
|
|
98
|
+
assert.deepEqual(picked.map((t) => t.source), ['Dashboard']);
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
test('a term with no target for THIS locale is not sent to the model', () => {
|
|
102
|
+
const { terms } = load([
|
|
103
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control' } },
|
|
104
|
+
]);
|
|
105
|
+
assert.equal(termsForBatch(terms, ['Open the Dashboard'], 'es').length, 1);
|
|
106
|
+
assert.equal(termsForBatch(terms, ['Open the Dashboard'], 'fr').length, 0,
|
|
107
|
+
'fr has no pinned target, so the term constrains nothing and must not bloat the prompt');
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
test('an empty glossary produces no prompt text at all', () => {
|
|
111
|
+
assert.equal(glossaryPrompt([], 'es'), '');
|
|
112
|
+
assert.deepEqual(termsForBatch([], ['anything'], 'es'), []);
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
test('the prompt names the rule and the capitalisation constraint', () => {
|
|
116
|
+
const { terms } = load([
|
|
117
|
+
{ source: 'Apple', rule: 'keep' },
|
|
118
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control' } },
|
|
119
|
+
]);
|
|
120
|
+
const picked = termsForBatch(terms, ['Apple built the Dashboard'], 'es');
|
|
121
|
+
const prompt = glossaryPrompt(picked, 'es');
|
|
122
|
+
assert.match(prompt, /leave exactly as "Apple"/);
|
|
123
|
+
assert.match(prompt, /this capitalisation only/);
|
|
124
|
+
assert.match(prompt, /always render as "Panel de control"/);
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
// ── Compliance ───────────────────────────────────────────────────────────────
|
|
128
|
+
|
|
129
|
+
test('a dropped brand is a violation; a surviving one is not', () => {
|
|
130
|
+
const { terms } = load([{ source: 'Acme', rule: 'keep' }]);
|
|
131
|
+
assert.equal(checkCompliance('Acme is fast', 'Acme es rápido', terms, 'es').length, 0);
|
|
132
|
+
|
|
133
|
+
const bad = checkCompliance('Acme is fast', 'Cumbre es rápido', terms, 'es');
|
|
134
|
+
assert.equal(bad.length, 1);
|
|
135
|
+
assert.equal(bad[0].rule, 'keep');
|
|
136
|
+
assert.equal(bad[0].expected, 'Acme');
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
test('compliance tolerates inflection around a pinned term', () => {
|
|
140
|
+
const { terms } = load([
|
|
141
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control' } },
|
|
142
|
+
]);
|
|
143
|
+
const ok = checkCompliance('Open the Dashboard', 'Abre el Panel de control', terms, 'es');
|
|
144
|
+
assert.equal(ok.length, 0);
|
|
145
|
+
|
|
146
|
+
const alsoOk = checkCompliance('Dashboard settings', 'Ajustes del panel de control', terms, 'es');
|
|
147
|
+
assert.equal(alsoOk.length, 0, 'case and surrounding words must not trigger a false positive');
|
|
148
|
+
|
|
149
|
+
const bad = checkCompliance('Open the Dashboard', 'Abre el Tablero', terms, 'es');
|
|
150
|
+
assert.equal(bad.length, 1);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
test('a term absent from the source is never checked against the target', () => {
|
|
154
|
+
const { terms } = load([{ source: 'Acme', rule: 'keep' }]);
|
|
155
|
+
assert.equal(checkCompliance('Nothing to see', 'Nada que ver', terms, 'es').length, 0);
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
// ── Fingerprint ──────────────────────────────────────────────────────────────
|
|
159
|
+
|
|
160
|
+
test('the fingerprint is order-independent but content-sensitive', () => {
|
|
161
|
+
const a = load([
|
|
162
|
+
{ source: 'Acme', rule: 'keep' },
|
|
163
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel' } },
|
|
164
|
+
]).terms;
|
|
165
|
+
const b = load([
|
|
166
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel' } },
|
|
167
|
+
{ source: 'Acme', rule: 'keep' },
|
|
168
|
+
]).terms;
|
|
169
|
+
assert.equal(glossaryFingerprint(a), glossaryFingerprint(b), 'reordering must not invalidate a memory');
|
|
170
|
+
|
|
171
|
+
const c = load([
|
|
172
|
+
{ source: 'Acme', rule: 'keep' },
|
|
173
|
+
{ source: 'Dashboard', rule: 'translate', targets: { es: 'Panel de control' } },
|
|
174
|
+
]).terms;
|
|
175
|
+
assert.notEqual(glossaryFingerprint(a), glossaryFingerprint(c), 'a changed target must invalidate');
|
|
176
|
+
});
|
|
177
|
+
|
|
178
|
+
// ── Loading from a file ──────────────────────────────────────────────────────
|
|
179
|
+
|
|
180
|
+
test('a glossary loads from a JSON file path', () => {
|
|
181
|
+
const dir = mkdtempSync(join(tmpdir(), 'ct-glossary-'));
|
|
182
|
+
const file = join(dir, 'glossary.json');
|
|
183
|
+
writeFileSync(file, JSON.stringify([{ source: 'Acme', rule: 'keep' }]));
|
|
184
|
+
|
|
185
|
+
const { terms } = loadGlossary('glossary.json', dir, resolve);
|
|
186
|
+
assert.equal(terms.length, 1);
|
|
187
|
+
assert.equal(terms[0].source, 'Acme');
|
|
188
|
+
});
|