claude-translator 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -18,6 +18,14 @@
18
18
  * inLanguage / @id / url — the three sitewide defects the
19
19
  * 2026-08-13 audit found on the proxy-served pages.
20
20
  * 5. Coverage share of segments still in English, per locale.
21
+ * 6. Never offered visible text the extractor never picked up. Coverage measures
22
+ * translated-of-EXTRACTED and is structurally blind to this.
23
+ * 7. Glossary terms that were supposed to survive, or to be rendered a
24
+ * particular way, and were not. REPORTS by default; --strict
25
+ * makes it gate, because inflection makes strictness noisy.
26
+ * 8. Numeric integrity numbers that changed value between source and translation.
27
+ * Gates by default: a silently rewritten price is a commercial
28
+ * problem, and no other gate can see it.
21
29
  */
22
30
 
23
31
  import { readFileSync, readdirSync, existsSync } from 'fs';
@@ -27,11 +35,15 @@ import { parse } from 'parse5';
27
35
 
28
36
  import {
29
37
  BUILD_DIR as DIST, SEG_DIR, BASE_URL as BASE, LOCALES as LANG_ROWS,
30
- BY_PATH, RTL, getPages, I18N_DIR, DNT,
38
+ BY_PATH, RTL, getPages, I18N_DIR, TM_DIR, SOURCE_FILE, ROOT_DIR, DNT, GLOSSARY,
31
39
  } from './config.mjs';
32
40
  import { creditBlock, markerBytes, GENERATOR_NAME, PRIOR_GENERATOR_NAMES } from './credit.mjs';
41
+ import { checkCompliance } from './glossary.mjs';
33
42
 
34
- const ROOT = process.cwd();
43
+ // ROOT_DIR honours $I18N_ROOT; process.cwd() did not, so a run from another
44
+ // directory read the wrong tree. Paths under i18n/ come from config so that
45
+ // `i18nDir` actually takes effect — it was imported here and then ignored.
46
+ const ROOT = ROOT_DIR;
35
47
 
36
48
  /**
37
49
  * Format, protocol and standards tokens that are correct unchanged in every language.
@@ -54,6 +66,9 @@ const args = Object.fromEntries(
54
66
  .map(([k, ...v]) => [k, v.join(' ') || true])
55
67
  );
56
68
 
69
+ /** Gate 7 reports by default; --strict makes a terminology violation fail the build. */
70
+ const STRICT = Boolean(args.strict);
71
+
57
72
 
58
73
  const requested = String(args.lang ?? '').trim();
59
74
  if (!requested) {
@@ -238,7 +253,7 @@ if (identityBad.length) fail('identity', `${identityBad.length} pages`);
238
253
  console.log('\n[5] translation coverage');
239
254
  const totalSegments = pages.reduce((n, p) => n + p.segmentCount, 0);
240
255
  for (const lang of LANGS) {
241
- const tmFile = join(ROOT, 'i18n/tm', `${lang}.json`);
256
+ const tmFile = join(TM_DIR, `${lang}.json`);
242
257
  if (!existsSync(tmFile)) {
243
258
  console.log(` ${lang}: no memory`);
244
259
  continue;
@@ -331,8 +346,8 @@ function extractedBlob(slug) {
331
346
  return entry.segments.map((s) => (src[s.hash]?.text ?? '').replace(/<\/?\d+\/?>/g, ' ')).join('  ');
332
347
  }
333
348
 
334
- const SOURCE = existsSync(join(ROOT, 'i18n/source.json'))
335
- ? JSON.parse(readFileSync(join(ROOT, 'i18n/source.json'), 'utf8'))
349
+ const SOURCE = existsSync(SOURCE_FILE)
350
+ ? JSON.parse(readFileSync(SOURCE_FILE, 'utf8'))
336
351
  : null;
337
352
 
338
353
  console.log('\n[6] text never offered for translation');
@@ -365,6 +380,99 @@ for (const lang of LANGS) {
365
380
  if (holes > 0) fail('extraction-hole', `${lang}: ${holes} strings never offered for translation`);
366
381
  }
367
382
 
383
+ // ── Gate 7: glossary compliance ──────────────────────────────────────────────
384
+ // Terminology was previously unverifiable. A brand could be translated away and every
385
+ // gate still passed: gate 2 compares markup, gate 5 counts coverage, and gate 6 uses the
386
+ // brand list only to SUPPRESS false alarms. Nothing asserted a term survived.
387
+ //
388
+ // Reports by default and gates only under --strict, on purpose. Target languages inflect
389
+ // ("Panel de control" -> "del Panel de control"), compound ("Dashboard-Ansicht") and
390
+ // decline, so a strict test flags correct work. references/quality-review.md is explicit
391
+ // that a heuristic which over-flags is worse than none, and the remedy here —
392
+ // purge-and-retranslate — costs real money.
393
+
394
+ console.log('\n[7] glossary compliance');
395
+ if (GLOSSARY.length === 0) {
396
+ console.log(' no glossary configured \u2014 skipped');
397
+ } else {
398
+ for (const lang of LANGS) {
399
+ const tmFile = join(TM_DIR, `${lang}.json`);
400
+ if (!existsSync(tmFile)) {
401
+ console.log(` ${lang}: no memory`);
402
+ continue;
403
+ }
404
+ const tm = JSON.parse(readFileSync(tmFile, 'utf8'));
405
+ const violations = [];
406
+ for (const [hash, unit] of Object.entries(SOURCE)) {
407
+ const translated = tm[hash];
408
+ if (typeof translated !== 'string') continue;
409
+ for (const v of checkCompliance(unit.text, translated, GLOSSARY, lang)) {
410
+ violations.push({ hash, ...v, source: unit.text.slice(0, 60) , term: v.source });
411
+ }
412
+ }
413
+ const checked = Object.keys(tm).length;
414
+ const pct = checked ? (violations.length * 100) / checked : 0;
415
+ console.log(
416
+ ` ${lang}: ${violations.length} violation(s) in ${checked.toLocaleString()} units (${pct.toFixed(2)}%)`
417
+ );
418
+ for (const v of violations.slice(0, 5)) {
419
+ console.log(` "${v.term}" \u2192 expected "${v.expected}" in: ${v.source}\u2026`);
420
+ }
421
+ if (violations.length > 5) console.log(` \u2026 and ${violations.length - 5} more`);
422
+ if (STRICT && violations.length) fail('glossary', `${lang}: ${violations.length} violation(s)`);
423
+ }
424
+ if (!STRICT) console.log(' (reporting only \u2014 pass --strict to gate on this)');
425
+ }
426
+
427
+ // ── Gate 8: numeric integrity ────────────────────────────────────────────────
428
+ // A model that silently rewrites "$49/month" as "$39/month" passes every other gate:
429
+ // the markup is identical, the placeholder count matches, the length is plausible and
430
+ // the text is fluent target-language. Prices are commercial commitments, so this one
431
+ // DOES gate by default.
432
+ //
433
+ // Compares multisets, not sequences: languages legitimately reorder ("2 of 3" ->
434
+ // "3 dintre 2" never happens, but date and measurement order does move). Digit-group
435
+ // separators are stripped first, because locale formatting is applied at build time and
436
+ // is a correct difference, not a defect.
437
+
438
+ const NUM_RE = /\d[\d.,\u00a0\u202f ]*\d|\d/g;
439
+
440
+ /** Numbers reduced to a comparable form: separators stripped, trailing zeros normalised. */
441
+ function numeralMultiset(text) {
442
+ const found = String(text).match(NUM_RE) ?? [];
443
+ return found
444
+ .map((n) => n.replace(/[.,\u00a0\u202f ]/g, ''))
445
+ .filter((n) => n.length > 0)
446
+ .map((n) => n.replace(/^0+(?=\d)/, ''))
447
+ .sort();
448
+ }
449
+
450
+ console.log('\n[8] numeric integrity');
451
+ for (const lang of LANGS) {
452
+ const tmFile = join(TM_DIR, `${lang}.json`);
453
+ if (!existsSync(tmFile)) {
454
+ console.log(` ${lang}: no memory`);
455
+ continue;
456
+ }
457
+ const tm = JSON.parse(readFileSync(tmFile, 'utf8'));
458
+ const drifted = [];
459
+ for (const [hash, unit] of Object.entries(SOURCE)) {
460
+ const translated = tm[hash];
461
+ if (typeof translated !== 'string') continue;
462
+ const a = numeralMultiset(unit.text);
463
+ const b = numeralMultiset(translated);
464
+ if (a.join('|') !== b.join('|')) {
465
+ drifted.push({ src: unit.text.slice(0, 70), out: translated.slice(0, 70), a, b });
466
+ }
467
+ }
468
+ console.log(` ${lang}: ${drifted.length} unit(s) whose numbers changed`);
469
+ for (const d of drifted.slice(0, 5)) {
470
+ console.log(` [${d.a.join(', ')}] \u2192 [${d.b.join(', ')}] ${d.src}\u2026`);
471
+ }
472
+ if (drifted.length > 5) console.log(` \u2026 and ${drifted.length - 5} more`);
473
+ if (drifted.length) fail('numeric-drift', `${lang}: ${drifted.length} unit(s)`);
474
+ }
475
+
368
476
  // ── Result ───────────────────────────────────────────────────────────────────
369
477
 
370
478
  console.log('');
@@ -1,9 +1,9 @@
1
1
  ---
2
- name: claude-translator
2
+ name: translate-site
3
3
  description: >
4
4
  Localize a static website into many languages by substituting translations into
5
- already-built HTML, without re-rendering. Ships six proven scripts (extract,
6
- translate, review, build, verify, SEO audit) plus the failure modes that cost real
5
+ already-built HTML, without re-rendering. Ships proven scripts (extract,
6
+ translate, review, build, verify, SEO audit, glossary, locale formatting, MQM quality scoring) plus the failure modes that cost real
7
7
  money to discover. Use when the user says "translate the site", "localize",
8
8
  "multi-language site", "i18n", "add languages", "translate all pages", or is
9
9
  replacing a translation proxy (Weglot, Bablic, Localize, TranslatePress) with self-hosted pages.
@@ -14,7 +14,7 @@ argument-hint: "[project-dir]"
14
14
  license: AGPL-3.0
15
15
  metadata:
16
16
  author: ConveyThis
17
- version: "1.4.0"
17
+ version: "2.0.0"
18
18
  category: i18n
19
19
  ---
20
20
 
@@ -82,7 +82,7 @@ write; where that is missing, say so rather than spending the user's time and AP
82
82
  - **The user wants to edit translations in a UI, or needs human review** — there is neither
83
83
  here. Editing means hand-editing a hash in `i18n/tm/{lang}.json`.
84
84
  - **The user is wrapping a modified copy in a hosted service** — AGPL-3.0 §13 obliges them to
85
- publish their modifications. See `LICENSING.md`; a commercial licence exists.
85
+ publish their modifications. See `${CLAUDE_PLUGIN_ROOT}/LICENSING.md`; a commercial licence exists.
86
86
 
87
87
  Running it unmodified, on their own sites, and shipping the output is unrestricted. Do not
88
88
  warn them about the licence in that case — it does not apply.
@@ -114,16 +114,22 @@ do not take the same parameters and guessing costs money.
114
114
 
115
115
  ```bash
116
116
  cd <project>
117
- node ~/.claude/skills/claude-translator/bin/claude-translator.mjs init
117
+ npx claude-translator init # works anywhere
118
118
  npm install # parse5, the only dependency
119
119
  ```
120
120
 
121
+ Offline, or when the plugin is already installed, run the bundled copy instead of `npx`:
122
+
123
+ ```bash
124
+ node "${CLAUDE_PLUGIN_ROOT}"/bin/claude-translator.mjs init
125
+ ```
126
+
121
127
  That copies the pipeline into `scripts/i18n/`, writes `i18n.config.json`, declares
122
128
  `parse5` and adds the derived paths to `.gitignore`. It never overwrites without
123
129
  `--force`, so it is safe to re-run; `--dir <path>` puts the scripts elsewhere.
124
130
 
125
131
  Scripts run **from inside the project** so `parse5` and relative paths resolve. Edit
126
- `i18n.config.json` — five keys cover everything; see `i18n.config.example.json`.
132
+ `i18n.config.json` — five keys cover everything; see `${CLAUDE_PLUGIN_ROOT}/i18n.config.example.json`.
127
133
 
128
134
  **Commit `i18n/tm/{lang}.json`.** The scaffolder deliberately does not ignore it: the
129
135
  memory is the asset, and losing it means paying for a full re-translation. Everything
@@ -133,12 +139,13 @@ else under `i18n/` is derived and is ignored for you. Add `i18n` to `.prettierig
133
139
 
134
140
  ```bash
135
141
  npm run build # source language only
136
- node scripts/extract.mjs # → i18n/source.json + segments/
137
- node scripts/translate.mjs --lang es,fr # → i18n/tm/{lang}.json (needs a provider key)
138
- node scripts/review.mjs --lang es # quality flags
139
- node scripts/build-locales.mjs --lang all # → dist/{lang}/…
140
- node scripts/verify.mjs --lang all # six gates
141
- node scripts/audit-seo.mjs # canonical/hreflang/JSON-LD/sitemaps
142
+ node scripts/i18n/extract.mjs # → i18n/source.json + segments/
143
+ node scripts/i18n/translate.mjs --lang es,fr # → i18n/tm/{lang}.json (needs a provider key)
144
+ node scripts/i18n/review.mjs --lang es # quality flags
145
+ node scripts/i18n/build-locales.mjs --lang all # → dist/{lang}/…
146
+ node scripts/i18n/verify.mjs --lang all # six gates
147
+ node scripts/i18n/audit-seo.mjs # canonical/hreflang/JSON-LD/sitemaps
148
+ node scripts/i18n/tqa.mjs --lang es # MQM quality score (optional, costs money)
142
149
  ```
143
150
 
144
151
  `finalize.sh <locales…>` collapses the per-locale cycle (gap-fill → review → purge →
@@ -163,6 +170,31 @@ matches and warn on zero. Attribute order is not guaranteed — `<link href="…
163
170
  rel="canonical">` is as valid as `rel` first — and an order-dependent regex silently
164
171
  matches nothing while reporting success.
165
172
 
173
+ **Element context is a hint, never a constraint.** Units carry an `el` label (button,
174
+ heading, form label, meta description...) so the model can pick the right register. There is
175
+ deliberately no length enforcement: a unit that fails validation ships in the SOURCE
176
+ language, so gating on length would replace a slightly-long German button with an English
177
+ one. If a user asks for a character budget, explain that tradeoff first.
178
+
179
+ **Never convert a currency, and never offer to.** `localeFormat` reformats amounts and
180
+ the pipeline reports every one it saw to `i18n/locale-format.json`. Converting a price at
181
+ a build-time rate is how a translation tool starts publishing wrong offers, and there is
182
+ no config option for it. If the user asks for conversion, explain the report instead.
183
+
184
+ **Gate 8 gates; gate 7 reports.** A changed numeric value fails the build, because a
185
+ silently rewritten price passes every other check. Terminology only warns unless
186
+ `--strict` is passed, because target languages inflect pinned terms and a strict check
187
+ flags correct work — see `references/quality-review.md` on why over-flagging is worse
188
+ than nothing.
189
+
190
+ **This tool rewrites tags; it does not create them.** Every locale-identity rule replaces
191
+ an attribute value on a tag the template already emits — the only thing ever inserted is the
192
+ attribution marker. So the **full `hreflang` mesh and `sitemap.xml` must come from the user's
193
+ own build**: `build-locales.mjs` rewrites only the `x-default` and source-language hrefs, and
194
+ writes no sitemap at all. `audit-seo.mjs` checks both exhaustively, which is how a missing
195
+ mesh surfaces. If the user's template lacks the alternates, say so plainly and point at
196
+ `references/adapting-generators.md` — do not imply the pipeline will emit them.
197
+
166
198
  **Verify server-side state, not exit codes.** Especially with rsync on macOS
167
199
  (`openrsync` prints usage and exits **0** on an unsupported flag).
168
200
 
@@ -71,6 +71,34 @@ are legitimate — postal addresses, image filenames used as alt text, proper no
71
71
 
72
72
  ---
73
73
 
74
+ ## Where TQA fits
75
+
76
+ `review.mjs` finds *defects by shape* — a dropped placeholder, a wholesale source-language
77
+ return, a truncated string. It is cheap, offline, and blind to whether the text is any good.
78
+
79
+ `tqa.mjs` answers the other question, and costs money. It scores a seeded,
80
+ frequency-stratified sample against the MQM typology using a second model as judge.
81
+
82
+ | | `review.mjs` | `tqa.mjs` |
83
+ | --- | --- | --- |
84
+ | Cost | free | one API call per ~10 units |
85
+ | Finds | mechanical defects | mistranslation, register, terminology, awkwardness |
86
+ | Output | `{lang}.review.json` | `i18n/tqa/{lang}.json` + `scorecard.md` |
87
+ | Gates a build | no | no |
88
+
89
+ Run `review.mjs` on every locale, every time. Run `tqa.mjs` when you need a number to
90
+ compare — between locales, between models, or before and after a prompt change.
91
+
92
+ **The same over-flagging discipline applies to the score itself.** Three guards exist
93
+ because each of them failed once during development:
94
+
95
+ - the judge defaults to a *different provider* than the translator, since a model scores
96
+ its own work generously
97
+ - `--repeat` reports the gap between two runs on the same sample, so nobody reads a
98
+ decimal place that is really noise
99
+ - a unit the judge could not assess is **excluded**, never counted as clean. An early
100
+ version reported `100.00 / 100` from a sample where every unit had failed to parse
101
+
74
102
  ## Workflow
75
103
 
76
104
  ```bash