champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,202 @@
1
+ /**
2
+ * bcp47.js — parse a language tag into its parts.
3
+ *
4
+ * WHY WE PARSE RATHER THAN SPLIT ON HYPHENS
5
+ * Because the parts are not positional. `zh-Hans-CN`, `zh-CN`, `zh-Hans` and
6
+ * `zh-x-pirate` all have a different second element, and the only way to know
7
+ * which is which is by SHAPE: four letters is a script, two letters or three
8
+ * digits is a region, `x-` opens private use. Code that assumes position gets
9
+ * `sr-Latn` right and `sr-RS` wrong, which is the kind of bug that surfaces
10
+ * as one language quietly rendering in the wrong script.
11
+ *
12
+ * WELL-FORMEDNESS, NOT VALIDITY
13
+ * RFC 5646 draws the distinction and so does this file. Well-formed means the
14
+ * tag has legal shape. Valid means every subtag is registered with IANA. This
15
+ * parser answers the first question only; the resolver answers the second by
16
+ * looking the subtags up in the atlas, which is where the registry lives.
17
+ *
18
+ * Keeping them apart matters: a well-formed tag we cannot resolve is a
19
+ * language we may simply not index yet, and reporting it as malformed would
20
+ * blame the caller for our own coverage gap.
21
+ *
22
+ * @see https://datatracker.ietf.org/doc/html/rfc5646#section-2.1
23
+ */
24
+
25
+ /** RFC 5646 2.2.9 — the irregular grandfathered tags, which match no grammar. */
26
+ const IRREGULAR = new Set([
27
+ 'en-gb-oed', 'i-ami', 'i-bnn', 'i-default', 'i-enochian', 'i-hak',
28
+ 'i-klingon', 'i-lux', 'i-mingo', 'i-navajo', 'i-pwn', 'i-tao', 'i-tay',
29
+ 'i-tsu', 'sgn-be-fr', 'sgn-be-nl', 'sgn-ch-de',
30
+ ]);
31
+
32
+ const ALPHA = /^[a-z]+$/;
33
+ const ALPHANUM = /^[a-z0-9]+$/;
34
+ const DIGIT = /^[0-9]+$/;
35
+
36
+ /**
37
+ * @typedef {object} ParsedTag
38
+ * @property {string} input the tag as given
39
+ * @property {boolean} wellFormed
40
+ * @property {string|null} language primary language subtag, lower-cased
41
+ * @property {string[]} extlangs
42
+ * @property {string|null} script title-cased, as BCP 47 recommends
43
+ * @property {string|null} region upper-cased
44
+ * @property {string[]} variants
45
+ * @property {Record<string,string[]>} extensions keyed by singleton
46
+ * @property {string[]} privateUse subtags after `x-`
47
+ * @property {boolean} grandfathered
48
+ * @property {string|null} problem why it is not well-formed, in words
49
+ */
50
+
51
+ /**
52
+ * Parse a BCP 47 language tag.
53
+ *
54
+ * Never throws: a malformed tag is a fact about the input, and callers need to
55
+ * report it rather than catch it.
56
+ *
57
+ * @param {string} input
58
+ * @returns {ParsedTag}
59
+ */
60
+ export function parseTag(input) {
61
+ const base = {
62
+ input,
63
+ wellFormed: false,
64
+ language: null,
65
+ extlangs: [],
66
+ script: null,
67
+ region: null,
68
+ variants: [],
69
+ extensions: {},
70
+ privateUse: [],
71
+ grandfathered: false,
72
+ problem: null,
73
+ };
74
+
75
+ if (typeof input !== 'string' || input === '') {
76
+ return { ...base, problem: 'empty tag' };
77
+ }
78
+ // BCP 47 tags are case-insensitive; casing is presentational only.
79
+ const lower = input.toLowerCase();
80
+ if (lower.length > 255) {
81
+ return { ...base, problem: 'tag longer than 255 characters' };
82
+ }
83
+ if (/[^a-z0-9-]/.test(lower)) {
84
+ return { ...base, problem: 'contains characters other than letters, digits and hyphen' };
85
+ }
86
+ if (lower.startsWith('-') || lower.endsWith('-') || lower.includes('--')) {
87
+ return { ...base, problem: 'empty subtag (leading, trailing or doubled hyphen)' };
88
+ }
89
+
90
+ if (IRREGULAR.has(lower)) {
91
+ return { ...base, wellFormed: true, grandfathered: true, language: null };
92
+ }
93
+
94
+ const parts = lower.split('-');
95
+ let i = 0;
96
+
97
+ // ── Whole-tag private use: `x-pirate`. Legal, and denotes no registered
98
+ // language at all — which is exactly what a conlang code should say.
99
+ if (parts[0] === 'x') {
100
+ const sub = parts.slice(1);
101
+ if (!sub.length || sub.some((p) => !ALPHANUM.test(p) || p.length > 8)) {
102
+ return { ...base, problem: 'private-use tag needs 1-8 alphanumeric subtags after "x"' };
103
+ }
104
+ return { ...base, wellFormed: true, privateUse: sub };
105
+ }
106
+
107
+ // ── Primary language subtag ─────────────────────────────────────────────
108
+ const lang = parts[i];
109
+ if (!ALPHA.test(lang) || lang.length < 2 || lang.length > 8) {
110
+ return { ...base, problem: `"${lang}" is not a language subtag (2-8 letters)` };
111
+ }
112
+ const out = { ...base, language: lang };
113
+ i++;
114
+
115
+ // ── Extended language subtags: up to three, three letters each ──────────
116
+ // Only follow a 2-3 letter primary subtag, per the grammar.
117
+ while (
118
+ lang.length <= 3 && out.extlangs.length < 3
119
+ && parts[i] && parts[i].length === 3 && ALPHA.test(parts[i])
120
+ ) {
121
+ out.extlangs.push(parts[i]);
122
+ i++;
123
+ }
124
+
125
+ // ── Script: exactly four letters ────────────────────────────────────────
126
+ if (parts[i] && parts[i].length === 4 && ALPHA.test(parts[i])) {
127
+ out.script = parts[i][0].toUpperCase() + parts[i].slice(1);
128
+ i++;
129
+ }
130
+
131
+ // ── Region: two letters or three digits ─────────────────────────────────
132
+ if (parts[i]
133
+ && ((parts[i].length === 2 && ALPHA.test(parts[i]))
134
+ || (parts[i].length === 3 && DIGIT.test(parts[i])))) {
135
+ out.region = parts[i].toUpperCase();
136
+ i++;
137
+ }
138
+
139
+ // ── Variants: 5-8 alphanumerics, or a digit then three alphanumerics ────
140
+ while (parts[i]
141
+ && ((parts[i].length >= 5 && parts[i].length <= 8 && ALPHANUM.test(parts[i]))
142
+ || (parts[i].length === 4 && DIGIT.test(parts[i][0]) && ALPHANUM.test(parts[i])))) {
143
+ if (out.variants.includes(parts[i])) {
144
+ return { ...out, problem: `variant "${parts[i]}" repeated` };
145
+ }
146
+ out.variants.push(parts[i]);
147
+ i++;
148
+ }
149
+
150
+ // ── Extensions: a singleton (not x) then one or more 2-8 alphanumerics ──
151
+ while (parts[i] && parts[i].length === 1 && parts[i] !== 'x' && ALPHANUM.test(parts[i])) {
152
+ const singleton = parts[i];
153
+ if (out.extensions[singleton]) {
154
+ return { ...out, problem: `extension singleton "${singleton}" repeated` };
155
+ }
156
+ i++;
157
+ const sub = [];
158
+ while (parts[i] && parts[i].length >= 2 && parts[i].length <= 8 && ALPHANUM.test(parts[i])) {
159
+ sub.push(parts[i]);
160
+ i++;
161
+ }
162
+ if (!sub.length) {
163
+ return { ...out, problem: `extension "${singleton}" has no subtags` };
164
+ }
165
+ out.extensions[singleton] = sub;
166
+ }
167
+
168
+ // ── Private use, as a suffix ────────────────────────────────────────────
169
+ if (parts[i] === 'x') {
170
+ i++;
171
+ const sub = [];
172
+ while (parts[i] && parts[i].length >= 1 && parts[i].length <= 8 && ALPHANUM.test(parts[i])) {
173
+ sub.push(parts[i]);
174
+ i++;
175
+ }
176
+ if (!sub.length) return { ...out, problem: 'private-use "x" has no subtags' };
177
+ out.privateUse = sub;
178
+ }
179
+
180
+ if (i < parts.length) {
181
+ return { ...out, problem: `"${parts[i]}" is not a valid subtag in this position` };
182
+ }
183
+ return { ...out, wellFormed: true };
184
+ }
185
+
186
+ /**
187
+ * Re-assemble a parsed tag in BCP 47's recommended casing.
188
+ *
189
+ * @param {ParsedTag} t
190
+ * @returns {string}
191
+ */
192
+ export function formatTag(t) {
193
+ if (t.grandfathered) return t.input.toLowerCase();
194
+ if (!t.language && t.privateUse.length) return `x-${t.privateUse.join('-')}`;
195
+ const parts = [t.language, ...t.extlangs];
196
+ if (t.script) parts.push(t.script);
197
+ if (t.region) parts.push(t.region);
198
+ parts.push(...t.variants);
199
+ for (const [singleton, sub] of Object.entries(t.extensions)) parts.push(singleton, ...sub);
200
+ if (t.privateUse.length) parts.push('x', ...t.privateUse);
201
+ return parts.join('-');
202
+ }
@@ -0,0 +1,314 @@
1
+ /**
2
+ * resolve.js — the ONE way anything in this repo turns a code into a language.
3
+ *
4
+ * THE PROBLEM IT EXISTS FOR
5
+ * Every MT service publishes coverage in BCP 47 and every language card is
6
+ * keyed by ISO 639-3, and the two do not agree about what a language is.
7
+ * Google lists `zho`; nobody lists `cmn`. CLDR's rule is that Unicode
8
+ * identifiers always use the macrolanguage for the predominant form, so the
9
+ * services are conformant and we were the ones off-standard — bridging the
10
+ * gap with eight hand-written routings that cited nothing.
11
+ *
12
+ * The bridge is now data: three pinned registries, projected into a tag
13
+ * index at build time. This module reads it and answers two questions.
14
+ *
15
+ * QUESTION ONE — WHAT LANGUAGE IS THIS CODE?
16
+ * `resolveTag()`. It never guesses. Where a tag denotes a macrolanguage it
17
+ * says so and reports what the registries say about a predominant member,
18
+ * attributed; where they do not agree, or where several members fold in, it
19
+ * reports that instead of picking. Norwegian has no cited predominant member
20
+ * and `no` → `nob` is therefore visibly an editorial decision rather than a
21
+ * fact hiding in a config file.
22
+ *
23
+ * QUESTION TWO — DOES THIS METHOD COVER THIS LANGUAGE?
24
+ * `resolveCoverage()`. Coverage is stored at the granularity the METHOD
25
+ * published and is never propagated down a macrolanguage. Instead the
26
+ * relationship is named:
27
+ *
28
+ * exact the method lists this language
29
+ * via-predominant the method lists the macrolanguage, and both
30
+ * registries name this language as the member it denotes
31
+ * via-macrolanguage the method lists the macrolanguage, and this language
32
+ * is merely one of its members
33
+ * none nothing
34
+ *
35
+ * The distinction is the whole point. OPUS-MT publishes coverage for `cre`,
36
+ * and both CLDR and SIL name Woods Cree as the member `cre` denotes — so
37
+ * Plains Cree gets `via-macrolanguage`, never `via-predominant`. That is a
38
+ * weaker answer than a boolean and it is the true one, and it is what a Cree
39
+ * speaker deciding whether to trust a translation actually needs to know.
40
+ *
41
+ * NOTHING HERE DECIDES WHAT TO DO ABOUT IT
42
+ * A verdict is evidence, not a policy. Whether `via-macrolanguage` is good
43
+ * enough to translate with is the caller's call, made under the caller's
44
+ * flags, and recorded on the run — so a leaderboard row reads "crk via cre"
45
+ * and can never silently become a `crk` score.
46
+ */
47
+
48
+ import fs from 'node:fs';
49
+ import path from 'node:path';
50
+
51
+ import { parseTag } from './bcp47.js';
52
+
53
+ /**
54
+ * The index is BUILD OUTPUT of `champollion atlas build`. Overridable so tests
55
+ * and the cutover can point at a freshly built one.
56
+ */
57
+ export const TAG_INDEX_FILE =
58
+ process.env.CHAMPOLLION_TAG_INDEX
59
+ || path.join(import.meta.dirname, '..', '..', 'shared', 'tag-index.json');
60
+
61
+ /** Verdicts, exported so callers branch on constants rather than strings. */
62
+ export const ROUTE = Object.freeze({
63
+ EXACT: 'exact',
64
+ MACROLANGUAGE: 'macrolanguage',
65
+ AMBIGUOUS: 'ambiguous',
66
+ PRIVATE_USE: 'private-use',
67
+ UNKNOWN: 'unknown',
68
+ MALFORMED: 'malformed',
69
+ });
70
+
71
+ export const COVERAGE = Object.freeze({
72
+ EXACT: 'exact',
73
+ VIA_PREDOMINANT: 'via-predominant',
74
+ VIA_MACROLANGUAGE: 'via-macrolanguage',
75
+ NONE: 'none',
76
+ });
77
+
78
+ let cached = null;
79
+
80
+ /**
81
+ * Load the tag index.
82
+ *
83
+ * Fails loudly and specifically when it is missing. A resolver that quietly
84
+ * degraded to "code in, code out" would look like it worked for the 90% of
85
+ * lookups where a code IS its own language, and be wrong on exactly the
86
+ * macrolanguages this module exists for.
87
+ *
88
+ * @param {string} [file]
89
+ */
90
+ export function loadTagIndex(file = TAG_INDEX_FILE) {
91
+ if (cached && cached.file === file) return cached.index;
92
+ if (!fs.existsSync(file)) {
93
+ throw new Error(
94
+ `no tag index at ${file}. It is build output — run:\n`
95
+ + ' node cli/scripts/cldf/build-atlas.mjs\n'
96
+ + 'Resolving codes without it would silently succeed for individual languages '
97
+ + 'and silently mislead for every macrolanguage.',
98
+ );
99
+ }
100
+ const index = JSON.parse(fs.readFileSync(file, 'utf-8'));
101
+ if (!index.byTag || !index.languages) {
102
+ throw new Error(`${file} is not a tag index (no byTag/languages)`);
103
+ }
104
+ cached = { file, index };
105
+ return index;
106
+ }
107
+
108
+ /** Drop the memoised index. For tests that rebuild it between cases. */
109
+ export function clearTagIndexCache() {
110
+ cached = null;
111
+ }
112
+
113
+ /**
114
+ * @typedef {object} Resolution
115
+ * @property {string} input
116
+ * @property {import('./bcp47.js').ParsedTag} tag
117
+ * @property {string|null} language the atlas language ID, if one was found
118
+ * @property {string} route one of ROUTE
119
+ * @property {string|null} scope 'individual' | 'macrolanguage'
120
+ * @property {object|null} predominant {language, authorities[], agreement}
121
+ * @property {string|null} note why there is no clean answer, in words
122
+ * @property {string[]} candidates for an ambiguous tag, every reading
123
+ */
124
+
125
+ /**
126
+ * Resolve one code or language tag.
127
+ *
128
+ * @param {string} input
129
+ * @param {object} [opts]
130
+ * @param {object} [opts.index]
131
+ * @returns {Resolution}
132
+ */
133
+ export function resolveTag(input, { index = loadTagIndex() } = {}) {
134
+ // FLORES, NLLB and Meta's Omnilingual MT all identify a language as
135
+ // `lang_Script` — `arb_Arab`, `zho_Hans`, `crk_Cans`. That underscore is not
136
+ // BCP 47, which uses a hyphen, and `parseTag` is right to reject it: the
137
+ // parser answers "is this a well-formed language tag", and this is not one.
138
+ //
139
+ // But it is the dominant convention in machine translation, and rejecting it
140
+ // outright made the resolver useless for exactly the datasets this project
141
+ // reads. So the separator is normalised HERE, explicitly and reported, rather
142
+ // than by loosening the grammar — a caller gets the answer and is told the
143
+ // input was not a BCP 47 tag.
144
+ const flores = typeof input === 'string' && !input.includes('-') && input.includes('_')
145
+ ? input.replace(/_/g, '-')
146
+ : null;
147
+ const tag = parseTag(flores ?? input);
148
+ const base = {
149
+ input,
150
+ tag,
151
+ language: null,
152
+ route: ROUTE.UNKNOWN,
153
+ scope: null,
154
+ predominant: null,
155
+ note: flores
156
+ ? `read as "${flores}" — the input uses the FLORES/NLLB lang_Script `
157
+ + 'convention, which is not BCP 47'
158
+ : null,
159
+ candidates: [],
160
+ };
161
+
162
+ if (!tag.wellFormed) {
163
+ return { ...base, route: ROUTE.MALFORMED, note: tag.problem };
164
+ }
165
+ // A private-use tag denotes no registered language, which is exactly what a
166
+ // conlang code should say. Reporting it as unknown would invite a caller to
167
+ // treat it as a lookup failure and retry.
168
+ if (!tag.language && tag.privateUse.length) {
169
+ return {
170
+ ...base,
171
+ route: ROUTE.PRIVATE_USE,
172
+ note: `"${input}" is a private-use tag; it names no registered language`,
173
+ };
174
+ }
175
+ if (!tag.language) {
176
+ return {
177
+ ...base,
178
+ route: ROUTE.UNKNOWN,
179
+ note: tag.grandfathered
180
+ ? `"${input}" is a grandfathered tag with no primary language subtag`
181
+ : 'no primary language subtag',
182
+ };
183
+ }
184
+
185
+ const subtag = tag.language;
186
+ const ambiguous = index.ambiguousTags?.[subtag];
187
+ if (ambiguous && !ambiguous.resolvesTo) {
188
+ return {
189
+ ...base,
190
+ route: ROUTE.AMBIGUOUS,
191
+ note: ambiguous.reason,
192
+ candidates: ambiguous.alsoClaimedBy.map((c) => ({
193
+ language: c.language, sources: c.sources,
194
+ })),
195
+ };
196
+ }
197
+
198
+ const languageId = index.byTag[subtag] ?? null;
199
+ if (!languageId) {
200
+ return {
201
+ ...base,
202
+ route: ROUTE.UNKNOWN,
203
+ note: `"${subtag}" is well-formed but is not a code the atlas resolves`,
204
+ };
205
+ }
206
+
207
+ const entry = index.languages[languageId] ?? {};
208
+ if (entry.scope !== 'macrolanguage') {
209
+ return { ...base, language: languageId, route: ROUTE.EXACT, scope: 'individual' };
210
+ }
211
+
212
+ return {
213
+ ...base,
214
+ language: languageId,
215
+ route: ROUTE.MACROLANGUAGE,
216
+ scope: 'macrolanguage',
217
+ predominant: entry.predominant ?? null,
218
+ // Both facts can be true at once — `zho_Hans` is a FLORES-style tag AND a
219
+ // macrolanguage. Overwriting the first with the second lost the reading
220
+ // note on exactly the inputs that needed it most.
221
+ note: [
222
+ base.note,
223
+ entry.predominant
224
+ ? null
225
+ : entry.noPredominant
226
+ ?? `"${subtag}" is a macrolanguage and no registry names a member it denotes`,
227
+ ].filter(Boolean).join('; ') || null,
228
+ };
229
+ }
230
+
231
+ /**
232
+ * How does a method's published coverage relate to this language?
233
+ *
234
+ * @param {object} args
235
+ * @param {string} args.language atlas language ID being asked about
236
+ * @param {Iterable<string>} args.covered the codes the METHOD published, verbatim
237
+ * @param {object} [args.index]
238
+ * @returns {{verdict: string, via: string|null, authorities: string[], because: string}}
239
+ */
240
+ export function resolveCoverage({ language, covered, index = loadTagIndex() }) {
241
+ const set = covered instanceof Set ? covered : new Set(covered);
242
+
243
+ if (set.has(language)) {
244
+ return {
245
+ verdict: COVERAGE.EXACT,
246
+ via: null,
247
+ authorities: [],
248
+ because: `the method lists "${language}" itself`,
249
+ };
250
+ }
251
+
252
+ const entry = index.languages[language];
253
+ const macro = entry?.macrolanguage ?? null;
254
+ if (!macro || !set.has(macro)) {
255
+ return {
256
+ verdict: COVERAGE.NONE, via: null, authorities: [],
257
+ because: macro
258
+ ? `the method lists neither "${language}" nor its macrolanguage "${macro}"`
259
+ : `the method does not list "${language}", which belongs to no macrolanguage`,
260
+ };
261
+ }
262
+
263
+ const macroEntry = index.languages[macro] ?? {};
264
+ const predominant = macroEntry.predominant ?? null;
265
+ if (predominant && predominant.language === language) {
266
+ return {
267
+ verdict: COVERAGE.VIA_PREDOMINANT,
268
+ via: macro,
269
+ authorities: predominant.authorities ?? [],
270
+ because: `the method lists the macrolanguage "${macro}", and `
271
+ + `${(predominant.authorities ?? []).join(' and ')} name "${language}" as the `
272
+ + 'member that tag denotes. The method still never said so itself.',
273
+ };
274
+ }
275
+
276
+ return {
277
+ verdict: COVERAGE.VIA_MACROLANGUAGE,
278
+ via: macro,
279
+ authorities: predominant ? predominant.authorities ?? [] : [],
280
+ because: predominant
281
+ ? `the method lists the macrolanguage "${macro}", but the registries name `
282
+ + `"${predominant.language}" as the member that tag denotes, not "${language}"`
283
+ : `the method lists the macrolanguage "${macro}", and no registry names which `
284
+ + 'of its members that tag denotes',
285
+ };
286
+ }
287
+
288
+ /**
289
+ * A one-line, human-readable account of a resolution — for CLI output, prompts
290
+ * and run records, so the same words appear everywhere a route is reported.
291
+ *
292
+ * @param {Resolution} r
293
+ */
294
+ export function explainResolution(r) {
295
+ switch (r.route) {
296
+ case ROUTE.EXACT:
297
+ return `${r.input} → ${r.language}`;
298
+ case ROUTE.MACROLANGUAGE:
299
+ return r.predominant
300
+ ? `${r.input} → ${r.language} (a macrolanguage; ${r.predominant.authorities.join(' and ')} `
301
+ + `name ${r.predominant.language} as the language it denotes)`
302
+ : `${r.input} → ${r.language} (a macrolanguage; ${r.note})`;
303
+ case ROUTE.AMBIGUOUS:
304
+ return `${r.input} is ambiguous: ${r.candidates.map(
305
+ (c) => `${c.language} per ${c.sources.join(', ')}`,
306
+ ).join('; ')}`;
307
+ case ROUTE.PRIVATE_USE:
308
+ return `${r.input} is a private-use tag and names no registered language`;
309
+ case ROUTE.MALFORMED:
310
+ return `${r.input} is not a well-formed language tag: ${r.note}`;
311
+ default:
312
+ return `${r.input} does not resolve to a language the atlas knows`;
313
+ }
314
+ }
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Post-translation terminology enforcement — warns when dictionary terms
3
+ * were prompted but not used in the LLM output.
4
+ *
5
+ * WHY THIS EXISTS:
6
+ * The coached method (llm-coached.js) injects a dictionary into the LLM
7
+ * prompt: "REQUIRED TERMINOLOGY: dashboard → tableau de bord". But the LLM
8
+ * is free to ignore it. Without verification, a user who carefully built
9
+ * a dictionary has no way to know if their terms were actually applied.
10
+ *
11
+ * HOW IT WORKS:
12
+ * After translation, this module scans each translated value to check
13
+ * whether expected dictionary terms appear. If a source value contains
14
+ * a dictionary source term (e.g., "dashboard") AND the translated value
15
+ * does NOT contain the required translation (e.g., "tableau de bord"),
16
+ * a violation is recorded.
17
+ *
18
+ * Matching is case-insensitive substring search — the LLM might inflect
19
+ * the term ("tableaux de bord" for plural), so exact match would be too
20
+ * strict. Violations are warnings, not blocking errors.
21
+ *
22
+ * USAGE:
23
+ * import { verifyTerminology } from './terminology.js';
24
+ *
25
+ * const { violations } = verifyTerminology(translations, sourceFlat, dictionary);
26
+ * if (violations.length > 0) {
27
+ * console.warn(`[TERM] ${violations.length} term(s) may not have been applied`);
28
+ * }
29
+ */
30
+
31
+ /**
32
+ * Check whether required dictionary terms were used in translations.
33
+ *
34
+ * For each translated value:
35
+ * 1. Find which dictionary source terms appear in the corresponding source value
36
+ * 2. For each matching term, check if the required translation appears in the output
37
+ * 3. Record violations where the term was expected but not found
38
+ *
39
+ * @param {object} translations - key → translated value (LLM output)
40
+ * @param {object} sourceFlat - key → source value (English)
41
+ * @param {object} dictionary - source term → required translation
42
+ * e.g., { "dashboard": "tableau de bord", "sign in": "se connecter" }
43
+ * @returns {{ violations: Array<{ key: string, term: string, expected: string, got: string }> }}
44
+ */
45
+ function verifyTerminology(translations, sourceFlat, dictionary) {
46
+ const violations = [];
47
+
48
+ // Fast path: nothing to check
49
+ if (!dictionary || typeof dictionary !== 'object' || Object.keys(dictionary).length === 0) {
50
+ return { violations };
51
+ }
52
+
53
+ // Pre-lowercase dictionary entries for case-insensitive matching
54
+ const terms = Object.entries(dictionary).map(([src, tgt]) => ({
55
+ source: src,
56
+ sourceLower: src.toLowerCase(),
57
+ expected: tgt,
58
+ expectedLower: tgt.toLowerCase(),
59
+ }));
60
+
61
+ for (const [key, translated] of Object.entries(translations)) {
62
+ // Skip non-string values (defense-in-depth)
63
+ if (typeof translated !== 'string') continue;
64
+
65
+ const source = sourceFlat[key];
66
+ if (typeof source !== 'string') continue;
67
+
68
+ const sourceLower = source.toLowerCase();
69
+ const translatedLower = translated.toLowerCase();
70
+
71
+ for (const term of terms) {
72
+ // Step 1: Does the source value contain this dictionary source term?
73
+ if (!sourceLower.includes(term.sourceLower)) continue;
74
+
75
+ // Step 2: Does the translated value contain the required translation?
76
+ if (translatedLower.includes(term.expectedLower)) continue;
77
+
78
+ // Violation: term was expected but not found
79
+ violations.push({
80
+ key,
81
+ term: term.source,
82
+ expected: term.expected,
83
+ got: translated.length > 100 ? translated.slice(0, 100) + '…' : translated,
84
+ });
85
+ }
86
+ }
87
+
88
+ return { violations };
89
+ }
90
+
91
+ /**
92
+ * Log terminology violations in a structured, actionable format.
93
+ *
94
+ * Designed to sit alongside the existing [GATE] log output from validate.js.
95
+ * Violations are warnings — they don't block the translation from being written.
96
+ *
97
+ * @param {Array<{ key: string, term: string, expected: string, got: string }>} violations
98
+ * @param {string} pairKey - e.g., "en:fr"
99
+ */
100
+ function logTermViolations(violations, pairKey) {
101
+ if (violations.length === 0) return;
102
+
103
+ console.error(`\n [TERM] ${pairKey}: ${violations.length} dictionary term(s) may not have been applied:`);
104
+ for (const { key, term, expected, got } of violations) {
105
+ console.error(` ⚠ "${key}": expected "${expected}" for term "${term}"`);
106
+ console.error(` → got "${got}"`);
107
+ }
108
+ console.error('');
109
+ }
110
+
111
+ export { verifyTerminology, logTermViolations };