champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Google Translate Method — Google Cloud Translation API v2.
|
|
3
|
+
*
|
|
4
|
+
* The universal baseline. Works out of the box with just a Google API key.
|
|
5
|
+
* Zero prompt engineering, zero coaching data — pure neural MT. This gives
|
|
6
|
+
* champollion a free/cheap option that supports 130+ languages.
|
|
7
|
+
*
|
|
8
|
+
* HOW IT WORKS:
|
|
9
|
+
* 1. Reads GOOGLE_TRANSLATE_API_KEY from environment
|
|
10
|
+
* 2. Chunks keys into batches (max 128 segments per Google API call)
|
|
11
|
+
* 3. POSTs to Google Cloud Translation API v2 REST endpoint
|
|
12
|
+
* 4. Maps Google's response array back to champollion's key-value format
|
|
13
|
+
* 5. Returns translations
|
|
14
|
+
*
|
|
15
|
+
* WHY BUILT-IN (not a plugin):
|
|
16
|
+
* Google Translate is the universal i18n baseline. Every developer
|
|
17
|
+
* expects it. It should work with zero config — just an env var.
|
|
18
|
+
* No plugin install, no method manifest, no coaching data.
|
|
19
|
+
*
|
|
20
|
+
* COST PROFILE: ~$20 per 1M characters (Google's pricing)
|
|
21
|
+
* QUALITY TIER: standard — no post-processing or verification
|
|
22
|
+
*
|
|
23
|
+
* ZERO DEPENDENCIES: Uses Node.js built-in fetch() against the REST API.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { TranslationMethod } from './base.js';
|
|
27
|
+
import { resolveProviderEnv, envNamesLabel } from './provider-env.js';
|
|
28
|
+
import { EST_CHARS_PER_KEY, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
|
|
29
|
+
import { estimateProviderCost } from './provider-pricing.js';
|
|
30
|
+
import { extractContentBody } from './content-separator.js';
|
|
31
|
+
import { output } from '../output.js';
|
|
32
|
+
import { fetchWithRetry } from './fetch-with-retry.js';
|
|
33
|
+
import { pMap } from '../concurrent.js';
|
|
34
|
+
|
|
35
|
+
const GOOGLE_API_URL = 'https://translation.googleapis.com/language/translate/v2';
|
|
36
|
+
|
|
37
|
+
// Google's batch limit per request
|
|
38
|
+
const MAX_SEGMENTS_PER_REQUEST = 128;
|
|
39
|
+
|
|
40
|
+
// Google Translate responses are fast — use a shorter timeout than the default
|
|
41
|
+
const GOOGLE_REQUEST_TIMEOUT_MS = 15000;
|
|
42
|
+
|
|
43
|
+
class GoogleTranslateMethod extends TranslationMethod {
|
|
44
|
+
constructor(options = {}) {
|
|
45
|
+
super('google-translate', options);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// ── API resolution helpers ──────────────────────────────────────
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Resolve the Google Cloud Translation API key.
|
|
52
|
+
* Checks: options.googleApiKey, then the provider-env SSOT
|
|
53
|
+
* (GOOGLE_TRANSLATE_API_KEY canonical, GOOGLE_API_KEY alias) across
|
|
54
|
+
* process.env and .env / .env.local files.
|
|
55
|
+
* @param {object} options - Caller-provided options
|
|
56
|
+
* @returns {string|null}
|
|
57
|
+
*/
|
|
58
|
+
_resolveApiKey(options) {
|
|
59
|
+
return options.googleApiKey
|
|
60
|
+
|| resolveProviderEnv('google-translate', options.cwd).value;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Translate a batch of key-value pairs via Google Cloud Translation API.
|
|
65
|
+
*
|
|
66
|
+
* @param {string[]} keys - Flat dot-notation keys to translate
|
|
67
|
+
* @param {object} sourceFlat - Full flattened source locale
|
|
68
|
+
* @param {import('../types.js').PairConfig} pairConfig - Pair config (target, source, etc.)
|
|
69
|
+
* @param {object} options - { googleApiKey } or reads from env
|
|
70
|
+
* @returns {object|null} Map of key → translated value, or null
|
|
71
|
+
*/
|
|
72
|
+
async translate(keys, sourceFlat, pairConfig, options) {
|
|
73
|
+
const apiKey = this._resolveApiKey(options);
|
|
74
|
+
|
|
75
|
+
if (!apiKey) {
|
|
76
|
+
output.error('Google Translate: No API key found.');
|
|
77
|
+
output.error('Set GOOGLE_TRANSLATE_API_KEY in your environment.');
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const targetLocale = pairConfig.target;
|
|
82
|
+
const sourceLocale = pairConfig.source || 'en';
|
|
83
|
+
const allTranslated = {};
|
|
84
|
+
|
|
85
|
+
// Chunk keys into batches and translate in parallel
|
|
86
|
+
const batchChunks = [];
|
|
87
|
+
for (let i = 0; i < keys.length; i += MAX_SEGMENTS_PER_REQUEST) {
|
|
88
|
+
batchChunks.push(keys.slice(i, i + MAX_SEGMENTS_PER_REQUEST));
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
await pMap(batchChunks, async (chunk, idx) => {
|
|
92
|
+
// Build parallel arrays: ordered keys and their source values
|
|
93
|
+
const orderedKeys = [];
|
|
94
|
+
const sourceTexts = [];
|
|
95
|
+
for (const key of chunk) {
|
|
96
|
+
const value = sourceFlat[key];
|
|
97
|
+
if (value && typeof value === 'string') {
|
|
98
|
+
orderedKeys.push(key);
|
|
99
|
+
sourceTexts.push(value);
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (sourceTexts.length === 0) return;
|
|
104
|
+
|
|
105
|
+
const result = await this._translateBatchWithRetry(
|
|
106
|
+
orderedKeys,
|
|
107
|
+
sourceTexts,
|
|
108
|
+
sourceLocale,
|
|
109
|
+
targetLocale,
|
|
110
|
+
apiKey,
|
|
111
|
+
idx + 1,
|
|
112
|
+
);
|
|
113
|
+
|
|
114
|
+
if (result) {
|
|
115
|
+
Object.assign(allTranslated, result);
|
|
116
|
+
}
|
|
117
|
+
}, { concurrency: DEFAULT_METHOD_CONCURRENCY });
|
|
118
|
+
|
|
119
|
+
return Object.keys(allTranslated).length > 0 ? allTranslated : null;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Translate freeform Markdown content via Google Cloud Translation API.
|
|
124
|
+
*
|
|
125
|
+
* HOW: The caller (content.js) has already run protectBlocks() on the
|
|
126
|
+
* Markdown body, replacing code blocks, shortcodes, inline code, and
|
|
127
|
+
* HTML with ⟦PROTECTED_N⟧ placeholders. We send the protected text
|
|
128
|
+
* through Google Translate as a single string. GT's neural engine
|
|
129
|
+
* treats the Unicode sentinels as opaque tokens and passes them through.
|
|
130
|
+
*
|
|
131
|
+
* SAFETY NET: If GT mangles any placeholders (rare, but possible),
|
|
132
|
+
* the caller's hasOrphanedPlaceholders() check catches the corruption
|
|
133
|
+
* and falls back to the English body with a warning.
|
|
134
|
+
*
|
|
135
|
+
* @param {string} prompt - Complete translation prompt from buildContentPrompt()
|
|
136
|
+
* @param {import('../types.js').PairConfig} pairConfig - Pair config
|
|
137
|
+
* @param {object} options - { googleApiKey }
|
|
138
|
+
* @returns {string|null} Translated text, or null on failure
|
|
139
|
+
*/
|
|
140
|
+
async translateContent(prompt, pairConfig, options) {
|
|
141
|
+
const apiKey = this._resolveApiKey(options);
|
|
142
|
+
|
|
143
|
+
if (!apiKey) {
|
|
144
|
+
output.error('Google Translate: No API key found for content translation.');
|
|
145
|
+
output.error('Set GOOGLE_TRANSLATE_API_KEY in your environment.');
|
|
146
|
+
return null;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// Extract the Markdown body from the translation prompt.
|
|
150
|
+
// The prompt format (from content.js buildContentPrompt) ends with:
|
|
151
|
+
// ---\n<markdown body>
|
|
152
|
+
const bodyText = extractContentBody(prompt);
|
|
153
|
+
|
|
154
|
+
if (!bodyText.trim()) return null;
|
|
155
|
+
|
|
156
|
+
const targetLocale = pairConfig.target;
|
|
157
|
+
const sourceLocale = pairConfig.source || 'en';
|
|
158
|
+
|
|
159
|
+
return this._translateSingleText(bodyText, sourceLocale, targetLocale, apiKey);
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Translate a single text string via Google Cloud Translation API v2.
|
|
164
|
+
*
|
|
165
|
+
* Simpler variant of _translateBatchWithRetry — sends one string,
|
|
166
|
+
* returns one translated string. Used for content translation.
|
|
167
|
+
*
|
|
168
|
+
* @param {string} text - Text to translate
|
|
169
|
+
* @param {string} sourceLocale - Source language code
|
|
170
|
+
* @param {string} targetLocale - Target language code
|
|
171
|
+
* @param {string} apiKey - Google Cloud API key
|
|
172
|
+
* @returns {string|null} Translated text, or null on failure
|
|
173
|
+
*/
|
|
174
|
+
async _translateSingleText(text, sourceLocale, targetLocale, apiKey) {
|
|
175
|
+
const response = await fetchWithRetry(GOOGLE_API_URL, {
|
|
176
|
+
method: 'POST',
|
|
177
|
+
headers: {
|
|
178
|
+
'Content-Type': 'application/json',
|
|
179
|
+
'x-goog-api-key': apiKey,
|
|
180
|
+
},
|
|
181
|
+
body: JSON.stringify({
|
|
182
|
+
q: [text],
|
|
183
|
+
source: normalizeLocaleForGoogle(sourceLocale),
|
|
184
|
+
target: normalizeLocaleForGoogle(targetLocale),
|
|
185
|
+
format: 'text',
|
|
186
|
+
}),
|
|
187
|
+
}, {
|
|
188
|
+
label: 'Google Translate content',
|
|
189
|
+
timeoutMs: GOOGLE_REQUEST_TIMEOUT_MS * 2,
|
|
190
|
+
});
|
|
191
|
+
|
|
192
|
+
if (!response) return null;
|
|
193
|
+
|
|
194
|
+
if (!response.ok) {
|
|
195
|
+
const errorBody = await response.text();
|
|
196
|
+
output.error(`Google Translate content: ${describeGoogleError(response.status, errorBody)}`);
|
|
197
|
+
return null;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const json = await response.json();
|
|
201
|
+
const translations = json?.data?.translations;
|
|
202
|
+
|
|
203
|
+
if (!translations || translations.length === 0) {
|
|
204
|
+
output.error('Google Translate content: empty response');
|
|
205
|
+
return null;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
return translations[0].translatedText;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* Call Google Cloud Translation API v2 with retry.
|
|
213
|
+
*
|
|
214
|
+
* @param {string[]} orderedKeys - Keys in the same order as sourceTexts
|
|
215
|
+
* @param {string[]} sourceTexts - Source values to translate
|
|
216
|
+
* @param {string} sourceLocale - Source language code
|
|
217
|
+
* @param {string} targetLocale - Target language code
|
|
218
|
+
* @param {string} apiKey - Google Cloud API key
|
|
219
|
+
* @param {number} batchNum - Batch number for logging
|
|
220
|
+
* @returns {object|null} Map of key → translated value
|
|
221
|
+
*/
|
|
222
|
+
async _translateBatchWithRetry(orderedKeys, sourceTexts, sourceLocale, targetLocale, apiKey, batchNum) {
|
|
223
|
+
const response = await fetchWithRetry(GOOGLE_API_URL, {
|
|
224
|
+
method: 'POST',
|
|
225
|
+
headers: {
|
|
226
|
+
'Content-Type': 'application/json',
|
|
227
|
+
// API key sent via header (not query string) to avoid leaking in logs/proxies
|
|
228
|
+
'x-goog-api-key': apiKey,
|
|
229
|
+
},
|
|
230
|
+
body: JSON.stringify({
|
|
231
|
+
q: sourceTexts,
|
|
232
|
+
source: normalizeLocaleForGoogle(sourceLocale),
|
|
233
|
+
target: normalizeLocaleForGoogle(targetLocale),
|
|
234
|
+
format: 'text',
|
|
235
|
+
}),
|
|
236
|
+
}, {
|
|
237
|
+
label: `Google batch ${batchNum}`,
|
|
238
|
+
timeoutMs: GOOGLE_REQUEST_TIMEOUT_MS,
|
|
239
|
+
});
|
|
240
|
+
|
|
241
|
+
if (!response) return null;
|
|
242
|
+
|
|
243
|
+
if (!response.ok) {
|
|
244
|
+
const errorBody = await response.text();
|
|
245
|
+
output.error(`Google batch ${batchNum}: ${describeGoogleError(response.status, errorBody)}`);
|
|
246
|
+
return null;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
const json = await response.json();
|
|
250
|
+
const translations = json?.data?.translations;
|
|
251
|
+
|
|
252
|
+
if (!translations || translations.length !== orderedKeys.length) {
|
|
253
|
+
output.error(`Google batch ${batchNum}: Response length mismatch (expected ${orderedKeys.length}, got ${translations?.length || 0})`);
|
|
254
|
+
return null;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
// Map translations back to key-value pairs
|
|
258
|
+
const result = {};
|
|
259
|
+
for (let i = 0; i < orderedKeys.length; i++) {
|
|
260
|
+
result[orderedKeys[i]] = translations[i].translatedText;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
const charCount = sourceTexts.reduce((sum, t) => sum + t.length, 0);
|
|
264
|
+
output.progress(` ✓ Google batch ${batchNum} (${orderedKeys.length} keys, ${charCount} chars)`);
|
|
265
|
+
|
|
266
|
+
return result;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Estimate translation cost at Google's documented rate ($20/million chars).
|
|
271
|
+
* Source: https://cloud.google.com/translate/pricing
|
|
272
|
+
*/
|
|
273
|
+
estimateCost(keyCount) {
|
|
274
|
+
return estimateProviderCost('google-translate', keyCount);
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
checkReadiness(context = {}) {
|
|
278
|
+
// Resolve through the same SSOT the loader uses so readiness can never
|
|
279
|
+
// pass under a name translate() won't read (or vice versa).
|
|
280
|
+
const { value } = resolveProviderEnv('google-translate', context.cwd);
|
|
281
|
+
if (!value) {
|
|
282
|
+
return { ready: false, reason: `No Google Translate API key (${envNamesLabel('google-translate')}).` };
|
|
283
|
+
}
|
|
284
|
+
return { ready: true };
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
getQualityTier() {
|
|
288
|
+
return 'standard';
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
getProvenance() {
|
|
292
|
+
return {
|
|
293
|
+
resources: [
|
|
294
|
+
{
|
|
295
|
+
name: 'Google Cloud Translation API',
|
|
296
|
+
license: 'Proprietary (Google ToS)',
|
|
297
|
+
type: 'api',
|
|
298
|
+
},
|
|
299
|
+
],
|
|
300
|
+
commercialReady: true,
|
|
301
|
+
flags: [],
|
|
302
|
+
};
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
getSetupHelp() {
|
|
306
|
+
const { value: apiKey } = resolveProviderEnv('google-translate');
|
|
307
|
+
if (!apiKey) {
|
|
308
|
+
return [
|
|
309
|
+
'',
|
|
310
|
+
' ┌─ Missing API Key ─────────────────────────────────────────────┐',
|
|
311
|
+
' │ Google Translate requires a Google Cloud API key. │',
|
|
312
|
+
' │ │',
|
|
313
|
+
' │ 1. Enable the Cloud Translation API in Google Cloud Console │',
|
|
314
|
+
' │ 2. Create an API key under APIs & Services > Credentials │',
|
|
315
|
+
' │ 3. Run: export GOOGLE_TRANSLATE_API_KEY=... │',
|
|
316
|
+
' │ │',
|
|
317
|
+
' │ Note: Google Translate works for key-value pairs but cannot │',
|
|
318
|
+
' │ safely translate Markdown content (no code block awareness). │',
|
|
319
|
+
' └────────────────────────────────────────────────────────────────┘',
|
|
320
|
+
];
|
|
321
|
+
}
|
|
322
|
+
return this._apiFailureHelp('Google Cloud Console');
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
// -----------------------------------------------------------------
|
|
327
|
+
// Helpers
|
|
328
|
+
// -----------------------------------------------------------------
|
|
329
|
+
|
|
330
|
+
/**
|
|
331
|
+
* Turn a Google Cloud Translation API error response into an actionable
|
|
332
|
+
* one-line message.
|
|
333
|
+
*
|
|
334
|
+
* WHY: Google returns HTTP 400 for BOTH "API key not valid" and "Bad language
|
|
335
|
+
* pair" (an unsupported source→target combination). A bare `400 — <body>` log
|
|
336
|
+
* leaves the user — and getSetupHelp's HTTP-status classifier — unable to tell
|
|
337
|
+
* a key problem from an unsupported-pair problem; 400 isn't 401/403, so the
|
|
338
|
+
* classifier falls through to "unknown" and the CLI wrongly tells the user to
|
|
339
|
+
* "check billing/quota". Parsing the body's `reason`/`message` lets us name the
|
|
340
|
+
* real cause.
|
|
341
|
+
*
|
|
342
|
+
* The body is JSON of the shape:
|
|
343
|
+
* { error: { code, message, status, errors: [{ message, reason }] } }
|
|
344
|
+
* but we degrade gracefully if it isn't parseable.
|
|
345
|
+
*
|
|
346
|
+
* @param {number} status - HTTP status code
|
|
347
|
+
* @param {string} errorBody - Raw response body text
|
|
348
|
+
* @returns {string} A human-readable, status-prefixed error description
|
|
349
|
+
*/
|
|
350
|
+
function describeGoogleError(status, errorBody) {
|
|
351
|
+
let parsed = null;
|
|
352
|
+
try {
|
|
353
|
+
parsed = JSON.parse(errorBody);
|
|
354
|
+
} catch {
|
|
355
|
+
// Non-JSON body — fall through to the raw text below.
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
const err = parsed?.error;
|
|
359
|
+
const reason = err?.errors?.[0]?.reason || '';
|
|
360
|
+
const message = err?.message || '';
|
|
361
|
+
const haystack = `${reason} ${message}`.toLowerCase();
|
|
362
|
+
|
|
363
|
+
// Invalid / unauthorized API key. Google reports this as a 400 with
|
|
364
|
+
// reason API_KEY_INVALID (not a 401), so we map it to an auth-style message.
|
|
365
|
+
if (reason === 'API_KEY_INVALID' || haystack.includes('api key not valid')) {
|
|
366
|
+
return `${status} — API key not valid. Re-check the GOOGLE_TRANSLATE_API_KEY value itself, not your account settings. ${message}`.trim();
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
// Unsupported source→target language pair. Google reports "Bad language pair"
|
|
370
|
+
// (often with reason "invalid"/"badRequest"). This is a configuration error,
|
|
371
|
+
// not a billing or key problem.
|
|
372
|
+
if (haystack.includes('bad language pair')
|
|
373
|
+
|| haystack.includes('language pair')
|
|
374
|
+
|| haystack.includes('invalid value at \'target\'')
|
|
375
|
+
|| haystack.includes('invalid value at \'source\'')) {
|
|
376
|
+
return `${status} — Unsupported language pair for Google Translate. ${message || 'Check that both the source and target locales are supported by Google Translate.'}`.trim();
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// Any other error — surface the parsed message if we have one, else the raw body.
|
|
380
|
+
return `${status} — ${message || errorBody}`;
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Normalize champollion locale codes to Google Translate codes.
|
|
385
|
+
*
|
|
386
|
+
* Google uses BCP-47 but with some quirks:
|
|
387
|
+
* - 'zh-TW' → 'zh-TW' (fine)
|
|
388
|
+
* - 'crk' → not supported by Google (will return error)
|
|
389
|
+
* - Some codes need mapping: 'he' ↔ 'iw', 'jw' ↔ 'jv'
|
|
390
|
+
*
|
|
391
|
+
* @param {string} locale - Champollion locale code
|
|
392
|
+
* @returns {string} Google-compatible locale code
|
|
393
|
+
*/
|
|
394
|
+
function normalizeLocaleForGoogle(locale) {
|
|
395
|
+
const GOOGLE_LOCALE_MAP = {
|
|
396
|
+
'he': 'iw', // Hebrew: BCP-47 is 'he', Google uses 'iw'
|
|
397
|
+
'jv': 'jw', // Javanese: BCP-47 is 'jv', Google uses 'jw'
|
|
398
|
+
};
|
|
399
|
+
return GOOGLE_LOCALE_MAP[locale] || locale;
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
export { GoogleTranslateMethod, normalizeLocaleForGoogle, describeGoogleError };
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared HTTP utilities for translation methods.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS EXISTS:
|
|
5
|
+
* Every translation method (LLM, coached, Google, API) needs the same
|
|
6
|
+
* retry logic: exponential backoff with jitter, retryable status detection,
|
|
7
|
+
* and an async sleep helper. Before this module, these were copy-pasted
|
|
8
|
+
* identically across 4 files — a maintenance hazard where a bug fix in
|
|
9
|
+
* one copy wouldn't propagate to the others.
|
|
10
|
+
*
|
|
11
|
+
* WHAT'S SHARED:
|
|
12
|
+
* - Constants: MAX_RETRIES, BASE_DELAY_MS, REQUEST_TIMEOUT_MS
|
|
13
|
+
* - isRetryable(status) — should this HTTP status trigger a retry?
|
|
14
|
+
* - getBackoffDelay(attempt) — exponential backoff with random jitter
|
|
15
|
+
* - sleep(ms) — promise-based delay
|
|
16
|
+
* - stripCodeFences(text) — remove markdown code fences from LLM responses
|
|
17
|
+
*
|
|
18
|
+
* WHAT'S NOT SHARED:
|
|
19
|
+
* - Prompt building (method-specific)
|
|
20
|
+
* - Response validation (method-specific key filtering)
|
|
21
|
+
* - Endpoint URLs (method-specific)
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
// -----------------------------------------------------------------
|
|
25
|
+
// Retry constants
|
|
26
|
+
// -----------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
/** Maximum number of retry attempts after the initial request */
|
|
29
|
+
const MAX_RETRIES = 3;
|
|
30
|
+
|
|
31
|
+
/** Base delay for exponential backoff (doubles each attempt) */
|
|
32
|
+
const BASE_DELAY_MS = 1000;
|
|
33
|
+
|
|
34
|
+
/** Default request timeout — aborts if no response within this window */
|
|
35
|
+
const REQUEST_TIMEOUT_MS = 30000;
|
|
36
|
+
|
|
37
|
+
// -----------------------------------------------------------------
|
|
38
|
+
// Retry utilities
|
|
39
|
+
// -----------------------------------------------------------------
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Check if an HTTP status code should trigger a retry.
|
|
43
|
+
*
|
|
44
|
+
* Retries on:
|
|
45
|
+
* - 429 (Too Many Requests) — rate limiting, always transient
|
|
46
|
+
* - 5xx (Server Error) — server-side issues, often transient
|
|
47
|
+
*
|
|
48
|
+
* Does NOT retry on:
|
|
49
|
+
* - 4xx (Client Error) — our request is wrong, retrying won't help
|
|
50
|
+
* - 2xx (Success) — obviously
|
|
51
|
+
*
|
|
52
|
+
* @param {number} status - HTTP status code
|
|
53
|
+
* @returns {boolean} True if the request should be retried
|
|
54
|
+
*/
|
|
55
|
+
function isRetryable(status) {
|
|
56
|
+
return status === 429 || status >= 500;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Calculate backoff delay with jitter for a given retry attempt.
|
|
61
|
+
*
|
|
62
|
+
* Formula: (BASE_DELAY_MS × 2^attempt) + random(0..500ms)
|
|
63
|
+
*
|
|
64
|
+
* Examples:
|
|
65
|
+
* attempt 0 → 1000-1500ms
|
|
66
|
+
* attempt 1 → 2000-2500ms
|
|
67
|
+
* attempt 2 → 4000-4500ms
|
|
68
|
+
*
|
|
69
|
+
* The jitter prevents thundering herd when multiple requests
|
|
70
|
+
* hit a rate limit simultaneously.
|
|
71
|
+
*
|
|
72
|
+
* @param {number} attempt - Zero-based retry attempt number
|
|
73
|
+
* @returns {number} Delay in milliseconds
|
|
74
|
+
*/
|
|
75
|
+
function getBackoffDelay(attempt) {
|
|
76
|
+
const base = BASE_DELAY_MS * Math.pow(2, attempt);
|
|
77
|
+
const jitter = Math.random() * 500;
|
|
78
|
+
return base + jitter;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Promise-based sleep utility.
|
|
83
|
+
*
|
|
84
|
+
* @param {number} ms - Duration to sleep in milliseconds
|
|
85
|
+
* @returns {Promise<void>}
|
|
86
|
+
*/
|
|
87
|
+
function sleep(ms) {
|
|
88
|
+
return new Promise(resolve => setTimeout(resolve, ms));
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// -----------------------------------------------------------------
|
|
92
|
+
// Response parsing utilities
|
|
93
|
+
// -----------------------------------------------------------------
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Strip markdown code fences from LLM response text.
|
|
97
|
+
*
|
|
98
|
+
* LLMs frequently wrap JSON responses in ```json ... ``` blocks
|
|
99
|
+
* even when instructed not to. This strips those fences so the
|
|
100
|
+
* response can be parsed as raw JSON.
|
|
101
|
+
*
|
|
102
|
+
* Handles both ```json and bare ``` fences.
|
|
103
|
+
*
|
|
104
|
+
* @param {string} text - Raw LLM response text (already trimmed)
|
|
105
|
+
* @returns {string} Text with code fences removed
|
|
106
|
+
*/
|
|
107
|
+
function stripCodeFences(text) {
|
|
108
|
+
return text
|
|
109
|
+
.replace(/^```(?:json|markdown|md)?\n?/i, '')
|
|
110
|
+
.replace(/\n?```$/i, '')
|
|
111
|
+
.trim();
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export {
|
|
115
|
+
MAX_RETRIES,
|
|
116
|
+
BASE_DELAY_MS,
|
|
117
|
+
REQUEST_TIMEOUT_MS,
|
|
118
|
+
isRetryable,
|
|
119
|
+
getBackoffDelay,
|
|
120
|
+
sleep,
|
|
121
|
+
stripCodeFences,
|
|
122
|
+
};
|