champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bcp47.js — parse a language tag into its parts.
|
|
3
|
+
*
|
|
4
|
+
* WHY WE PARSE RATHER THAN SPLIT ON HYPHENS
|
|
5
|
+
* Because the parts are not positional. `zh-Hans-CN`, `zh-CN`, `zh-Hans` and
|
|
6
|
+
* `zh-x-pirate` all have a different second element, and the only way to know
|
|
7
|
+
* which is which is by SHAPE: four letters is a script, two letters or three
|
|
8
|
+
* digits is a region, `x-` opens private use. Code that assumes position gets
|
|
9
|
+
* `sr-Latn` right and `sr-RS` wrong, which is the kind of bug that surfaces
|
|
10
|
+
* as one language quietly rendering in the wrong script.
|
|
11
|
+
*
|
|
12
|
+
* WELL-FORMEDNESS, NOT VALIDITY
|
|
13
|
+
* RFC 5646 draws the distinction and so does this file. Well-formed means the
|
|
14
|
+
* tag has legal shape. Valid means every subtag is registered with IANA. This
|
|
15
|
+
* parser answers the first question only; the resolver answers the second by
|
|
16
|
+
* looking the subtags up in the atlas, which is where the registry lives.
|
|
17
|
+
*
|
|
18
|
+
* Keeping them apart matters: a well-formed tag we cannot resolve is a
|
|
19
|
+
* language we may simply not index yet, and reporting it as malformed would
|
|
20
|
+
* blame the caller for our own coverage gap.
|
|
21
|
+
*
|
|
22
|
+
* @see https://datatracker.ietf.org/doc/html/rfc5646#section-2.1
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/** RFC 5646 2.2.9 — the irregular grandfathered tags, which match no grammar. */
|
|
26
|
+
const IRREGULAR = new Set([
|
|
27
|
+
'en-gb-oed', 'i-ami', 'i-bnn', 'i-default', 'i-enochian', 'i-hak',
|
|
28
|
+
'i-klingon', 'i-lux', 'i-mingo', 'i-navajo', 'i-pwn', 'i-tao', 'i-tay',
|
|
29
|
+
'i-tsu', 'sgn-be-fr', 'sgn-be-nl', 'sgn-ch-de',
|
|
30
|
+
]);
|
|
31
|
+
|
|
32
|
+
const ALPHA = /^[a-z]+$/;
|
|
33
|
+
const ALPHANUM = /^[a-z0-9]+$/;
|
|
34
|
+
const DIGIT = /^[0-9]+$/;
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* @typedef {object} ParsedTag
|
|
38
|
+
* @property {string} input the tag as given
|
|
39
|
+
* @property {boolean} wellFormed
|
|
40
|
+
* @property {string|null} language primary language subtag, lower-cased
|
|
41
|
+
* @property {string[]} extlangs
|
|
42
|
+
* @property {string|null} script title-cased, as BCP 47 recommends
|
|
43
|
+
* @property {string|null} region upper-cased
|
|
44
|
+
* @property {string[]} variants
|
|
45
|
+
* @property {Record<string,string[]>} extensions keyed by singleton
|
|
46
|
+
* @property {string[]} privateUse subtags after `x-`
|
|
47
|
+
* @property {boolean} grandfathered
|
|
48
|
+
* @property {string|null} problem why it is not well-formed, in words
|
|
49
|
+
*/
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Parse a BCP 47 language tag.
|
|
53
|
+
*
|
|
54
|
+
* Never throws: a malformed tag is a fact about the input, and callers need to
|
|
55
|
+
* report it rather than catch it.
|
|
56
|
+
*
|
|
57
|
+
* @param {string} input
|
|
58
|
+
* @returns {ParsedTag}
|
|
59
|
+
*/
|
|
60
|
+
export function parseTag(input) {
|
|
61
|
+
const base = {
|
|
62
|
+
input,
|
|
63
|
+
wellFormed: false,
|
|
64
|
+
language: null,
|
|
65
|
+
extlangs: [],
|
|
66
|
+
script: null,
|
|
67
|
+
region: null,
|
|
68
|
+
variants: [],
|
|
69
|
+
extensions: {},
|
|
70
|
+
privateUse: [],
|
|
71
|
+
grandfathered: false,
|
|
72
|
+
problem: null,
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
if (typeof input !== 'string' || input === '') {
|
|
76
|
+
return { ...base, problem: 'empty tag' };
|
|
77
|
+
}
|
|
78
|
+
// BCP 47 tags are case-insensitive; casing is presentational only.
|
|
79
|
+
const lower = input.toLowerCase();
|
|
80
|
+
if (lower.length > 255) {
|
|
81
|
+
return { ...base, problem: 'tag longer than 255 characters' };
|
|
82
|
+
}
|
|
83
|
+
if (/[^a-z0-9-]/.test(lower)) {
|
|
84
|
+
return { ...base, problem: 'contains characters other than letters, digits and hyphen' };
|
|
85
|
+
}
|
|
86
|
+
if (lower.startsWith('-') || lower.endsWith('-') || lower.includes('--')) {
|
|
87
|
+
return { ...base, problem: 'empty subtag (leading, trailing or doubled hyphen)' };
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
if (IRREGULAR.has(lower)) {
|
|
91
|
+
return { ...base, wellFormed: true, grandfathered: true, language: null };
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
const parts = lower.split('-');
|
|
95
|
+
let i = 0;
|
|
96
|
+
|
|
97
|
+
// ── Whole-tag private use: `x-pirate`. Legal, and denotes no registered
|
|
98
|
+
// language at all — which is exactly what a conlang code should say.
|
|
99
|
+
if (parts[0] === 'x') {
|
|
100
|
+
const sub = parts.slice(1);
|
|
101
|
+
if (!sub.length || sub.some((p) => !ALPHANUM.test(p) || p.length > 8)) {
|
|
102
|
+
return { ...base, problem: 'private-use tag needs 1-8 alphanumeric subtags after "x"' };
|
|
103
|
+
}
|
|
104
|
+
return { ...base, wellFormed: true, privateUse: sub };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// ── Primary language subtag ─────────────────────────────────────────────
|
|
108
|
+
const lang = parts[i];
|
|
109
|
+
if (!ALPHA.test(lang) || lang.length < 2 || lang.length > 8) {
|
|
110
|
+
return { ...base, problem: `"${lang}" is not a language subtag (2-8 letters)` };
|
|
111
|
+
}
|
|
112
|
+
const out = { ...base, language: lang };
|
|
113
|
+
i++;
|
|
114
|
+
|
|
115
|
+
// ── Extended language subtags: up to three, three letters each ──────────
|
|
116
|
+
// Only follow a 2-3 letter primary subtag, per the grammar.
|
|
117
|
+
while (
|
|
118
|
+
lang.length <= 3 && out.extlangs.length < 3
|
|
119
|
+
&& parts[i] && parts[i].length === 3 && ALPHA.test(parts[i])
|
|
120
|
+
) {
|
|
121
|
+
out.extlangs.push(parts[i]);
|
|
122
|
+
i++;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// ── Script: exactly four letters ────────────────────────────────────────
|
|
126
|
+
if (parts[i] && parts[i].length === 4 && ALPHA.test(parts[i])) {
|
|
127
|
+
out.script = parts[i][0].toUpperCase() + parts[i].slice(1);
|
|
128
|
+
i++;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// ── Region: two letters or three digits ─────────────────────────────────
|
|
132
|
+
if (parts[i]
|
|
133
|
+
&& ((parts[i].length === 2 && ALPHA.test(parts[i]))
|
|
134
|
+
|| (parts[i].length === 3 && DIGIT.test(parts[i])))) {
|
|
135
|
+
out.region = parts[i].toUpperCase();
|
|
136
|
+
i++;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// ── Variants: 5-8 alphanumerics, or a digit then three alphanumerics ────
|
|
140
|
+
while (parts[i]
|
|
141
|
+
&& ((parts[i].length >= 5 && parts[i].length <= 8 && ALPHANUM.test(parts[i]))
|
|
142
|
+
|| (parts[i].length === 4 && DIGIT.test(parts[i][0]) && ALPHANUM.test(parts[i])))) {
|
|
143
|
+
if (out.variants.includes(parts[i])) {
|
|
144
|
+
return { ...out, problem: `variant "${parts[i]}" repeated` };
|
|
145
|
+
}
|
|
146
|
+
out.variants.push(parts[i]);
|
|
147
|
+
i++;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// ── Extensions: a singleton (not x) then one or more 2-8 alphanumerics ──
|
|
151
|
+
while (parts[i] && parts[i].length === 1 && parts[i] !== 'x' && ALPHANUM.test(parts[i])) {
|
|
152
|
+
const singleton = parts[i];
|
|
153
|
+
if (out.extensions[singleton]) {
|
|
154
|
+
return { ...out, problem: `extension singleton "${singleton}" repeated` };
|
|
155
|
+
}
|
|
156
|
+
i++;
|
|
157
|
+
const sub = [];
|
|
158
|
+
while (parts[i] && parts[i].length >= 2 && parts[i].length <= 8 && ALPHANUM.test(parts[i])) {
|
|
159
|
+
sub.push(parts[i]);
|
|
160
|
+
i++;
|
|
161
|
+
}
|
|
162
|
+
if (!sub.length) {
|
|
163
|
+
return { ...out, problem: `extension "${singleton}" has no subtags` };
|
|
164
|
+
}
|
|
165
|
+
out.extensions[singleton] = sub;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// ── Private use, as a suffix ────────────────────────────────────────────
|
|
169
|
+
if (parts[i] === 'x') {
|
|
170
|
+
i++;
|
|
171
|
+
const sub = [];
|
|
172
|
+
while (parts[i] && parts[i].length >= 1 && parts[i].length <= 8 && ALPHANUM.test(parts[i])) {
|
|
173
|
+
sub.push(parts[i]);
|
|
174
|
+
i++;
|
|
175
|
+
}
|
|
176
|
+
if (!sub.length) return { ...out, problem: 'private-use "x" has no subtags' };
|
|
177
|
+
out.privateUse = sub;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
if (i < parts.length) {
|
|
181
|
+
return { ...out, problem: `"${parts[i]}" is not a valid subtag in this position` };
|
|
182
|
+
}
|
|
183
|
+
return { ...out, wellFormed: true };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Re-assemble a parsed tag in BCP 47's recommended casing.
|
|
188
|
+
*
|
|
189
|
+
* @param {ParsedTag} t
|
|
190
|
+
* @returns {string}
|
|
191
|
+
*/
|
|
192
|
+
export function formatTag(t) {
|
|
193
|
+
if (t.grandfathered) return t.input.toLowerCase();
|
|
194
|
+
if (!t.language && t.privateUse.length) return `x-${t.privateUse.join('-')}`;
|
|
195
|
+
const parts = [t.language, ...t.extlangs];
|
|
196
|
+
if (t.script) parts.push(t.script);
|
|
197
|
+
if (t.region) parts.push(t.region);
|
|
198
|
+
parts.push(...t.variants);
|
|
199
|
+
for (const [singleton, sub] of Object.entries(t.extensions)) parts.push(singleton, ...sub);
|
|
200
|
+
if (t.privateUse.length) parts.push('x', ...t.privateUse);
|
|
201
|
+
return parts.join('-');
|
|
202
|
+
}
|
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* resolve.js — the ONE way anything in this repo turns a code into a language.
|
|
3
|
+
*
|
|
4
|
+
* THE PROBLEM IT EXISTS FOR
|
|
5
|
+
* Every MT service publishes coverage in BCP 47 and every language card is
|
|
6
|
+
* keyed by ISO 639-3, and the two do not agree about what a language is.
|
|
7
|
+
* Google lists `zho`; nobody lists `cmn`. CLDR's rule is that Unicode
|
|
8
|
+
* identifiers always use the macrolanguage for the predominant form, so the
|
|
9
|
+
* services are conformant and we were the ones off-standard — bridging the
|
|
10
|
+
* gap with eight hand-written routings that cited nothing.
|
|
11
|
+
*
|
|
12
|
+
* The bridge is now data: three pinned registries, projected into a tag
|
|
13
|
+
* index at build time. This module reads it and answers two questions.
|
|
14
|
+
*
|
|
15
|
+
* QUESTION ONE — WHAT LANGUAGE IS THIS CODE?
|
|
16
|
+
* `resolveTag()`. It never guesses. Where a tag denotes a macrolanguage it
|
|
17
|
+
* says so and reports what the registries say about a predominant member,
|
|
18
|
+
* attributed; where they do not agree, or where several members fold in, it
|
|
19
|
+
* reports that instead of picking. Norwegian has no cited predominant member
|
|
20
|
+
* and `no` → `nob` is therefore visibly an editorial decision rather than a
|
|
21
|
+
* fact hiding in a config file.
|
|
22
|
+
*
|
|
23
|
+
* QUESTION TWO — DOES THIS METHOD COVER THIS LANGUAGE?
|
|
24
|
+
* `resolveCoverage()`. Coverage is stored at the granularity the METHOD
|
|
25
|
+
* published and is never propagated down a macrolanguage. Instead the
|
|
26
|
+
* relationship is named:
|
|
27
|
+
*
|
|
28
|
+
* exact the method lists this language
|
|
29
|
+
* via-predominant the method lists the macrolanguage, and both
|
|
30
|
+
* registries name this language as the member it denotes
|
|
31
|
+
* via-macrolanguage the method lists the macrolanguage, and this language
|
|
32
|
+
* is merely one of its members
|
|
33
|
+
* none nothing
|
|
34
|
+
*
|
|
35
|
+
* The distinction is the whole point. OPUS-MT publishes coverage for `cre`,
|
|
36
|
+
* and both CLDR and SIL name Woods Cree as the member `cre` denotes — so
|
|
37
|
+
* Plains Cree gets `via-macrolanguage`, never `via-predominant`. That is a
|
|
38
|
+
* weaker answer than a boolean and it is the true one, and it is what a Cree
|
|
39
|
+
* speaker deciding whether to trust a translation actually needs to know.
|
|
40
|
+
*
|
|
41
|
+
* NOTHING HERE DECIDES WHAT TO DO ABOUT IT
|
|
42
|
+
* A verdict is evidence, not a policy. Whether `via-macrolanguage` is good
|
|
43
|
+
* enough to translate with is the caller's call, made under the caller's
|
|
44
|
+
* flags, and recorded on the run — so a leaderboard row reads "crk via cre"
|
|
45
|
+
* and can never silently become a `crk` score.
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
import fs from 'node:fs';
|
|
49
|
+
import path from 'node:path';
|
|
50
|
+
|
|
51
|
+
import { parseTag } from './bcp47.js';
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The index is BUILD OUTPUT of `champollion atlas build`. Overridable so tests
|
|
55
|
+
* and the cutover can point at a freshly built one.
|
|
56
|
+
*/
|
|
57
|
+
export const TAG_INDEX_FILE =
|
|
58
|
+
process.env.CHAMPOLLION_TAG_INDEX
|
|
59
|
+
|| path.join(import.meta.dirname, '..', '..', 'shared', 'tag-index.json');
|
|
60
|
+
|
|
61
|
+
/** Verdicts, exported so callers branch on constants rather than strings. */
|
|
62
|
+
export const ROUTE = Object.freeze({
|
|
63
|
+
EXACT: 'exact',
|
|
64
|
+
MACROLANGUAGE: 'macrolanguage',
|
|
65
|
+
AMBIGUOUS: 'ambiguous',
|
|
66
|
+
PRIVATE_USE: 'private-use',
|
|
67
|
+
UNKNOWN: 'unknown',
|
|
68
|
+
MALFORMED: 'malformed',
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
export const COVERAGE = Object.freeze({
|
|
72
|
+
EXACT: 'exact',
|
|
73
|
+
VIA_PREDOMINANT: 'via-predominant',
|
|
74
|
+
VIA_MACROLANGUAGE: 'via-macrolanguage',
|
|
75
|
+
NONE: 'none',
|
|
76
|
+
});
|
|
77
|
+
|
|
78
|
+
let cached = null;
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Load the tag index.
|
|
82
|
+
*
|
|
83
|
+
* Fails loudly and specifically when it is missing. A resolver that quietly
|
|
84
|
+
* degraded to "code in, code out" would look like it worked for the 90% of
|
|
85
|
+
* lookups where a code IS its own language, and be wrong on exactly the
|
|
86
|
+
* macrolanguages this module exists for.
|
|
87
|
+
*
|
|
88
|
+
* @param {string} [file]
|
|
89
|
+
*/
|
|
90
|
+
export function loadTagIndex(file = TAG_INDEX_FILE) {
|
|
91
|
+
if (cached && cached.file === file) return cached.index;
|
|
92
|
+
if (!fs.existsSync(file)) {
|
|
93
|
+
throw new Error(
|
|
94
|
+
`no tag index at ${file}. It is build output — run:\n`
|
|
95
|
+
+ ' node cli/scripts/cldf/build-atlas.mjs\n'
|
|
96
|
+
+ 'Resolving codes without it would silently succeed for individual languages '
|
|
97
|
+
+ 'and silently mislead for every macrolanguage.',
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
const index = JSON.parse(fs.readFileSync(file, 'utf-8'));
|
|
101
|
+
if (!index.byTag || !index.languages) {
|
|
102
|
+
throw new Error(`${file} is not a tag index (no byTag/languages)`);
|
|
103
|
+
}
|
|
104
|
+
cached = { file, index };
|
|
105
|
+
return index;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** Drop the memoised index. For tests that rebuild it between cases. */
|
|
109
|
+
export function clearTagIndexCache() {
|
|
110
|
+
cached = null;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* @typedef {object} Resolution
|
|
115
|
+
* @property {string} input
|
|
116
|
+
* @property {import('./bcp47.js').ParsedTag} tag
|
|
117
|
+
* @property {string|null} language the atlas language ID, if one was found
|
|
118
|
+
* @property {string} route one of ROUTE
|
|
119
|
+
* @property {string|null} scope 'individual' | 'macrolanguage'
|
|
120
|
+
* @property {object|null} predominant {language, authorities[], agreement}
|
|
121
|
+
* @property {string|null} note why there is no clean answer, in words
|
|
122
|
+
* @property {string[]} candidates for an ambiguous tag, every reading
|
|
123
|
+
*/
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Resolve one code or language tag.
|
|
127
|
+
*
|
|
128
|
+
* @param {string} input
|
|
129
|
+
* @param {object} [opts]
|
|
130
|
+
* @param {object} [opts.index]
|
|
131
|
+
* @returns {Resolution}
|
|
132
|
+
*/
|
|
133
|
+
export function resolveTag(input, { index = loadTagIndex() } = {}) {
|
|
134
|
+
// FLORES, NLLB and Meta's Omnilingual MT all identify a language as
|
|
135
|
+
// `lang_Script` — `arb_Arab`, `zho_Hans`, `crk_Cans`. That underscore is not
|
|
136
|
+
// BCP 47, which uses a hyphen, and `parseTag` is right to reject it: the
|
|
137
|
+
// parser answers "is this a well-formed language tag", and this is not one.
|
|
138
|
+
//
|
|
139
|
+
// But it is the dominant convention in machine translation, and rejecting it
|
|
140
|
+
// outright made the resolver useless for exactly the datasets this project
|
|
141
|
+
// reads. So the separator is normalised HERE, explicitly and reported, rather
|
|
142
|
+
// than by loosening the grammar — a caller gets the answer and is told the
|
|
143
|
+
// input was not a BCP 47 tag.
|
|
144
|
+
const flores = typeof input === 'string' && !input.includes('-') && input.includes('_')
|
|
145
|
+
? input.replace(/_/g, '-')
|
|
146
|
+
: null;
|
|
147
|
+
const tag = parseTag(flores ?? input);
|
|
148
|
+
const base = {
|
|
149
|
+
input,
|
|
150
|
+
tag,
|
|
151
|
+
language: null,
|
|
152
|
+
route: ROUTE.UNKNOWN,
|
|
153
|
+
scope: null,
|
|
154
|
+
predominant: null,
|
|
155
|
+
note: flores
|
|
156
|
+
? `read as "${flores}" — the input uses the FLORES/NLLB lang_Script `
|
|
157
|
+
+ 'convention, which is not BCP 47'
|
|
158
|
+
: null,
|
|
159
|
+
candidates: [],
|
|
160
|
+
};
|
|
161
|
+
|
|
162
|
+
if (!tag.wellFormed) {
|
|
163
|
+
return { ...base, route: ROUTE.MALFORMED, note: tag.problem };
|
|
164
|
+
}
|
|
165
|
+
// A private-use tag denotes no registered language, which is exactly what a
|
|
166
|
+
// conlang code should say. Reporting it as unknown would invite a caller to
|
|
167
|
+
// treat it as a lookup failure and retry.
|
|
168
|
+
if (!tag.language && tag.privateUse.length) {
|
|
169
|
+
return {
|
|
170
|
+
...base,
|
|
171
|
+
route: ROUTE.PRIVATE_USE,
|
|
172
|
+
note: `"${input}" is a private-use tag; it names no registered language`,
|
|
173
|
+
};
|
|
174
|
+
}
|
|
175
|
+
if (!tag.language) {
|
|
176
|
+
return {
|
|
177
|
+
...base,
|
|
178
|
+
route: ROUTE.UNKNOWN,
|
|
179
|
+
note: tag.grandfathered
|
|
180
|
+
? `"${input}" is a grandfathered tag with no primary language subtag`
|
|
181
|
+
: 'no primary language subtag',
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const subtag = tag.language;
|
|
186
|
+
const ambiguous = index.ambiguousTags?.[subtag];
|
|
187
|
+
if (ambiguous && !ambiguous.resolvesTo) {
|
|
188
|
+
return {
|
|
189
|
+
...base,
|
|
190
|
+
route: ROUTE.AMBIGUOUS,
|
|
191
|
+
note: ambiguous.reason,
|
|
192
|
+
candidates: ambiguous.alsoClaimedBy.map((c) => ({
|
|
193
|
+
language: c.language, sources: c.sources,
|
|
194
|
+
})),
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
const languageId = index.byTag[subtag] ?? null;
|
|
199
|
+
if (!languageId) {
|
|
200
|
+
return {
|
|
201
|
+
...base,
|
|
202
|
+
route: ROUTE.UNKNOWN,
|
|
203
|
+
note: `"${subtag}" is well-formed but is not a code the atlas resolves`,
|
|
204
|
+
};
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
const entry = index.languages[languageId] ?? {};
|
|
208
|
+
if (entry.scope !== 'macrolanguage') {
|
|
209
|
+
return { ...base, language: languageId, route: ROUTE.EXACT, scope: 'individual' };
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
return {
|
|
213
|
+
...base,
|
|
214
|
+
language: languageId,
|
|
215
|
+
route: ROUTE.MACROLANGUAGE,
|
|
216
|
+
scope: 'macrolanguage',
|
|
217
|
+
predominant: entry.predominant ?? null,
|
|
218
|
+
// Both facts can be true at once — `zho_Hans` is a FLORES-style tag AND a
|
|
219
|
+
// macrolanguage. Overwriting the first with the second lost the reading
|
|
220
|
+
// note on exactly the inputs that needed it most.
|
|
221
|
+
note: [
|
|
222
|
+
base.note,
|
|
223
|
+
entry.predominant
|
|
224
|
+
? null
|
|
225
|
+
: entry.noPredominant
|
|
226
|
+
?? `"${subtag}" is a macrolanguage and no registry names a member it denotes`,
|
|
227
|
+
].filter(Boolean).join('; ') || null,
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* How does a method's published coverage relate to this language?
|
|
233
|
+
*
|
|
234
|
+
* @param {object} args
|
|
235
|
+
* @param {string} args.language atlas language ID being asked about
|
|
236
|
+
* @param {Iterable<string>} args.covered the codes the METHOD published, verbatim
|
|
237
|
+
* @param {object} [args.index]
|
|
238
|
+
* @returns {{verdict: string, via: string|null, authorities: string[], because: string}}
|
|
239
|
+
*/
|
|
240
|
+
export function resolveCoverage({ language, covered, index = loadTagIndex() }) {
|
|
241
|
+
const set = covered instanceof Set ? covered : new Set(covered);
|
|
242
|
+
|
|
243
|
+
if (set.has(language)) {
|
|
244
|
+
return {
|
|
245
|
+
verdict: COVERAGE.EXACT,
|
|
246
|
+
via: null,
|
|
247
|
+
authorities: [],
|
|
248
|
+
because: `the method lists "${language}" itself`,
|
|
249
|
+
};
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
const entry = index.languages[language];
|
|
253
|
+
const macro = entry?.macrolanguage ?? null;
|
|
254
|
+
if (!macro || !set.has(macro)) {
|
|
255
|
+
return {
|
|
256
|
+
verdict: COVERAGE.NONE, via: null, authorities: [],
|
|
257
|
+
because: macro
|
|
258
|
+
? `the method lists neither "${language}" nor its macrolanguage "${macro}"`
|
|
259
|
+
: `the method does not list "${language}", which belongs to no macrolanguage`,
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
const macroEntry = index.languages[macro] ?? {};
|
|
264
|
+
const predominant = macroEntry.predominant ?? null;
|
|
265
|
+
if (predominant && predominant.language === language) {
|
|
266
|
+
return {
|
|
267
|
+
verdict: COVERAGE.VIA_PREDOMINANT,
|
|
268
|
+
via: macro,
|
|
269
|
+
authorities: predominant.authorities ?? [],
|
|
270
|
+
because: `the method lists the macrolanguage "${macro}", and `
|
|
271
|
+
+ `${(predominant.authorities ?? []).join(' and ')} name "${language}" as the `
|
|
272
|
+
+ 'member that tag denotes. The method still never said so itself.',
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
return {
|
|
277
|
+
verdict: COVERAGE.VIA_MACROLANGUAGE,
|
|
278
|
+
via: macro,
|
|
279
|
+
authorities: predominant ? predominant.authorities ?? [] : [],
|
|
280
|
+
because: predominant
|
|
281
|
+
? `the method lists the macrolanguage "${macro}", but the registries name `
|
|
282
|
+
+ `"${predominant.language}" as the member that tag denotes, not "${language}"`
|
|
283
|
+
: `the method lists the macrolanguage "${macro}", and no registry names which `
|
|
284
|
+
+ 'of its members that tag denotes',
|
|
285
|
+
};
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* A one-line, human-readable account of a resolution — for CLI output, prompts
|
|
290
|
+
* and run records, so the same words appear everywhere a route is reported.
|
|
291
|
+
*
|
|
292
|
+
* @param {Resolution} r
|
|
293
|
+
*/
|
|
294
|
+
export function explainResolution(r) {
|
|
295
|
+
switch (r.route) {
|
|
296
|
+
case ROUTE.EXACT:
|
|
297
|
+
return `${r.input} → ${r.language}`;
|
|
298
|
+
case ROUTE.MACROLANGUAGE:
|
|
299
|
+
return r.predominant
|
|
300
|
+
? `${r.input} → ${r.language} (a macrolanguage; ${r.predominant.authorities.join(' and ')} `
|
|
301
|
+
+ `name ${r.predominant.language} as the language it denotes)`
|
|
302
|
+
: `${r.input} → ${r.language} (a macrolanguage; ${r.note})`;
|
|
303
|
+
case ROUTE.AMBIGUOUS:
|
|
304
|
+
return `${r.input} is ambiguous: ${r.candidates.map(
|
|
305
|
+
(c) => `${c.language} per ${c.sources.join(', ')}`,
|
|
306
|
+
).join('; ')}`;
|
|
307
|
+
case ROUTE.PRIVATE_USE:
|
|
308
|
+
return `${r.input} is a private-use tag and names no registered language`;
|
|
309
|
+
case ROUTE.MALFORMED:
|
|
310
|
+
return `${r.input} is not a well-formed language tag: ${r.note}`;
|
|
311
|
+
default:
|
|
312
|
+
return `${r.input} does not resolve to a language the atlas knows`;
|
|
313
|
+
}
|
|
314
|
+
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Post-translation terminology enforcement — warns when dictionary terms
|
|
3
|
+
* were prompted but not used in the LLM output.
|
|
4
|
+
*
|
|
5
|
+
* WHY THIS EXISTS:
|
|
6
|
+
* The coached method (llm-coached.js) injects a dictionary into the LLM
|
|
7
|
+
* prompt: "REQUIRED TERMINOLOGY: dashboard → tableau de bord". But the LLM
|
|
8
|
+
* is free to ignore it. Without verification, a user who carefully built
|
|
9
|
+
* a dictionary has no way to know if their terms were actually applied.
|
|
10
|
+
*
|
|
11
|
+
* HOW IT WORKS:
|
|
12
|
+
* After translation, this module scans each translated value to check
|
|
13
|
+
* whether expected dictionary terms appear. If a source value contains
|
|
14
|
+
* a dictionary source term (e.g., "dashboard") AND the translated value
|
|
15
|
+
* does NOT contain the required translation (e.g., "tableau de bord"),
|
|
16
|
+
* a violation is recorded.
|
|
17
|
+
*
|
|
18
|
+
* Matching is case-insensitive substring search — the LLM might inflect
|
|
19
|
+
* the term ("tableaux de bord" for plural), so exact match would be too
|
|
20
|
+
* strict. Violations are warnings, not blocking errors.
|
|
21
|
+
*
|
|
22
|
+
* USAGE:
|
|
23
|
+
* import { verifyTerminology } from './terminology.js';
|
|
24
|
+
*
|
|
25
|
+
* const { violations } = verifyTerminology(translations, sourceFlat, dictionary);
|
|
26
|
+
* if (violations.length > 0) {
|
|
27
|
+
* console.warn(`[TERM] ${violations.length} term(s) may not have been applied`);
|
|
28
|
+
* }
|
|
29
|
+
*/
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Check whether required dictionary terms were used in translations.
|
|
33
|
+
*
|
|
34
|
+
* For each translated value:
|
|
35
|
+
* 1. Find which dictionary source terms appear in the corresponding source value
|
|
36
|
+
* 2. For each matching term, check if the required translation appears in the output
|
|
37
|
+
* 3. Record violations where the term was expected but not found
|
|
38
|
+
*
|
|
39
|
+
* @param {object} translations - key → translated value (LLM output)
|
|
40
|
+
* @param {object} sourceFlat - key → source value (English)
|
|
41
|
+
* @param {object} dictionary - source term → required translation
|
|
42
|
+
* e.g., { "dashboard": "tableau de bord", "sign in": "se connecter" }
|
|
43
|
+
* @returns {{ violations: Array<{ key: string, term: string, expected: string, got: string }> }}
|
|
44
|
+
*/
|
|
45
|
+
function verifyTerminology(translations, sourceFlat, dictionary) {
|
|
46
|
+
const violations = [];
|
|
47
|
+
|
|
48
|
+
// Fast path: nothing to check
|
|
49
|
+
if (!dictionary || typeof dictionary !== 'object' || Object.keys(dictionary).length === 0) {
|
|
50
|
+
return { violations };
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// Pre-lowercase dictionary entries for case-insensitive matching
|
|
54
|
+
const terms = Object.entries(dictionary).map(([src, tgt]) => ({
|
|
55
|
+
source: src,
|
|
56
|
+
sourceLower: src.toLowerCase(),
|
|
57
|
+
expected: tgt,
|
|
58
|
+
expectedLower: tgt.toLowerCase(),
|
|
59
|
+
}));
|
|
60
|
+
|
|
61
|
+
for (const [key, translated] of Object.entries(translations)) {
|
|
62
|
+
// Skip non-string values (defense-in-depth)
|
|
63
|
+
if (typeof translated !== 'string') continue;
|
|
64
|
+
|
|
65
|
+
const source = sourceFlat[key];
|
|
66
|
+
if (typeof source !== 'string') continue;
|
|
67
|
+
|
|
68
|
+
const sourceLower = source.toLowerCase();
|
|
69
|
+
const translatedLower = translated.toLowerCase();
|
|
70
|
+
|
|
71
|
+
for (const term of terms) {
|
|
72
|
+
// Step 1: Does the source value contain this dictionary source term?
|
|
73
|
+
if (!sourceLower.includes(term.sourceLower)) continue;
|
|
74
|
+
|
|
75
|
+
// Step 2: Does the translated value contain the required translation?
|
|
76
|
+
if (translatedLower.includes(term.expectedLower)) continue;
|
|
77
|
+
|
|
78
|
+
// Violation: term was expected but not found
|
|
79
|
+
violations.push({
|
|
80
|
+
key,
|
|
81
|
+
term: term.source,
|
|
82
|
+
expected: term.expected,
|
|
83
|
+
got: translated.length > 100 ? translated.slice(0, 100) + '…' : translated,
|
|
84
|
+
});
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return { violations };
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Log terminology violations in a structured, actionable format.
|
|
93
|
+
*
|
|
94
|
+
* Designed to sit alongside the existing [GATE] log output from validate.js.
|
|
95
|
+
* Violations are warnings — they don't block the translation from being written.
|
|
96
|
+
*
|
|
97
|
+
* @param {Array<{ key: string, term: string, expected: string, got: string }>} violations
|
|
98
|
+
* @param {string} pairKey - e.g., "en:fr"
|
|
99
|
+
*/
|
|
100
|
+
function logTermViolations(violations, pairKey) {
|
|
101
|
+
if (violations.length === 0) return;
|
|
102
|
+
|
|
103
|
+
console.error(`\n [TERM] ${pairKey}: ${violations.length} dictionary term(s) may not have been applied:`);
|
|
104
|
+
for (const { key, term, expected, got } of violations) {
|
|
105
|
+
console.error(` ⚠ "${key}": expected "${expected}" for term "${term}"`);
|
|
106
|
+
console.error(` → got "${got}"`);
|
|
107
|
+
}
|
|
108
|
+
console.error('');
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
export { verifyTerminology, logTermViolations };
|