champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,586 @@
1
+ /**
2
+ * DirectLLMMethod — shared base class for direct LLM provider integrations.
3
+ *
4
+ * WHY THIS EXISTS:
5
+ * OpenAI, Anthropic, and Gemini methods share ~90% of their logic:
6
+ * - Constructor with coaching cache
7
+ * - translate() with coaching loading, system message building, batch loop
8
+ * - translateContent() with coaching block prepending
9
+ * - Dictionary hint injection in per-batch user messages
10
+ * - JSON response parsing + key validation
11
+ * - Retry loop with exponential backoff
12
+ *
13
+ * The ONLY things that differ are:
14
+ * - API endpoint URL + auth header format
15
+ * - Request body shape (messages, system param, generationConfig)
16
+ * - Response parsing path (choices[0] vs content[0] vs candidates[0])
17
+ * - Pricing, provenance, model patterns
18
+ *
19
+ * This base class implements all shared logic. Subclasses override a small
20
+ * set of abstract methods to provide provider-specific HTTP details.
21
+ *
22
+ * RUNTIME MODEL VALIDATION:
23
+ * On first translate() call, the base class fetches the provider's available
24
+ * model list and validates the configured model against it. This catches:
25
+ * - Deprecated/retired model names (the exact bug we fixed twice already)
26
+ * - OpenRouter-format model strings used with direct providers
27
+ * - Models from the wrong provider (e.g., claude-* on OpenAI)
28
+ * The model list is cached per-process to avoid repeated API calls.
29
+ *
30
+ * INHERITANCE CHAIN:
31
+ * TranslationMethod → LLMMethod → DirectLLMMethod → OpenAIMethod
32
+ * → AnthropicMethod
33
+ * → GeminiMethod
34
+ */
35
+
36
+ import path from 'node:path';
37
+ import { LLMMethod, buildSystemMessage, buildUserMessage, isUnsafeKey, inferKeyTypes } from './llm.js';
38
+ import { loadCoachingData, findDictionaryMatches, buildCoachedSystemMessage, buildContentCoachingBlock, DEFAULT_COACHING_DIR } from './llm-coached.js';
39
+ import { getEnvOrFileVar } from '../api-key.js';
40
+ import {
41
+ MAX_RETRIES,
42
+ REQUEST_TIMEOUT_MS,
43
+ isRetryable,
44
+ getBackoffDelay,
45
+ sleep,
46
+ stripCodeFences,
47
+ } from './http-utils.js';
48
+ import { DEFAULT_BATCH_SIZE, DEFAULT_TEMPERATURE, DEFAULT_MAX_RETRIES, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
49
+ import { pMap } from '../concurrent.js';
50
+ import { output } from '../output.js';
51
+ import { recordTranslationError } from './translation-error.js';
52
+
53
+ // ── Model validation patterns ──────────────────────────────────────
54
+ // Used to detect when a model string belongs to the wrong provider.
55
+ // These are intentionally broad — we're catching obvious mismatches,
56
+ // not building a comprehensive model catalog.
57
+ const MODEL_PATTERNS = {
58
+ openai: {
59
+ prefixes: ['gpt-', 'o1', 'o3', 'o4', 'chatgpt-'],
60
+ label: 'OpenAI',
61
+ },
62
+ anthropic: {
63
+ prefixes: ['claude-'],
64
+ label: 'Anthropic',
65
+ },
66
+ gemini: {
67
+ prefixes: ['gemini-'],
68
+ label: 'Google Gemini',
69
+ },
70
+ };
71
+
72
+ /**
73
+ * Per-process cache of available models per provider.
74
+ * Keyed by provider name (e.g., 'openai'), value is a Set of model IDs
75
+ * or null if the fetch failed (so we don't retry on every batch).
76
+ */
77
+ const _modelListCache = new Map();
78
+
79
+ class DirectLLMMethod extends LLMMethod {
80
+ constructor(options = {}) {
81
+ super(options);
82
+ // Coaching data cache — avoids re-reading .champollion/coaching/<locale>.json per batch
83
+ this._coachingCache = new Map();
84
+ // Track whether we've already validated the model for this instance
85
+ this._modelValidated = false;
86
+ }
87
+
88
+ // ── Abstract methods — subclasses MUST implement these ──────────
89
+
90
+ /**
91
+ * Environment variable name for this provider's API key.
92
+ * @returns {string} e.g., 'OPENAI_API_KEY'
93
+ */
94
+ _getApiKeyEnvVar() {
95
+ throw new Error(`${this.name}._getApiKeyEnvVar() not implemented`);
96
+ }
97
+
98
+ /**
99
+ * Options key name for this provider's API key.
100
+ * @returns {string} e.g., 'openaiApiKey'
101
+ */
102
+ _getApiKeyOptionsKey() {
103
+ throw new Error(`${this.name}._getApiKeyOptionsKey() not implemented`);
104
+ }
105
+
106
+ /**
107
+ * Default model ID for this provider.
108
+ * @returns {string} e.g., 'gpt-4o'
109
+ */
110
+ _getDefaultModel() {
111
+ throw new Error(`${this.name}._getDefaultModel() not implemented`);
112
+ }
113
+
114
+ /**
115
+ * Human-readable provider name for log messages.
116
+ * @returns {string} e.g., 'OpenAI'
117
+ */
118
+ _getProviderLabel() {
119
+ throw new Error(`${this.name}._getProviderLabel() not implemented`);
120
+ }
121
+
122
+ /**
123
+ * Build the HTTP request for the provider's chat/generate endpoint.
124
+ *
125
+ * @param {object} params
126
+ * @param {string} params.prompt - User message content
127
+ * @param {string} params.systemMessage - System message (may be null)
128
+ * @param {string} params.apiKey - Provider API key
129
+ * @param {string} params.model - Model ID
130
+ * @param {number} params.temperature - Sampling temperature
131
+ * @param {boolean} params.isJsonMode - Whether to request JSON output
132
+ * @returns {{ url: string, headers: object, body: object }}
133
+ */
134
+ _buildApiRequest(params) {
135
+ throw new Error(`${this.name}._buildApiRequest() not implemented`);
136
+ }
137
+
138
+ /**
139
+ * Extract the text content from the provider's API response JSON.
140
+ *
141
+ * @param {object} json - Parsed response JSON
142
+ * @returns {string|null} Extracted text, or null if missing
143
+ */
144
+ _extractResponseText(json) {
145
+ throw new Error(`${this.name}._extractResponseText() not implemented`);
146
+ }
147
+
148
+ /**
149
+ * Fetch the list of available model IDs from the provider's API.
150
+ *
151
+ * @param {string} apiKey - Provider API key
152
+ * @returns {Promise<string[]|null>} Array of model IDs, or null on failure
153
+ */
154
+ async _fetchModels(apiKey) {
155
+ // Default: no model listing available. Subclasses override.
156
+ return null;
157
+ }
158
+
159
+ // ── Shared implementation ──────────────────────────────────────
160
+
161
+ /**
162
+ * Resolve the API key from options, env vars, or .env files.
163
+ * @param {object} options - Caller-provided options
164
+ * @returns {string|null}
165
+ */
166
+ _resolveApiKey(options) {
167
+ const envVar = this._getApiKeyEnvVar();
168
+ const optKey = this._getApiKeyOptionsKey();
169
+ return options[optKey]
170
+ || getEnvOrFileVar(envVar)
171
+ || getEnvOrFileVar(envVar, options.cwd);
172
+ }
173
+
174
+ // ── OpenAI-compatible endpoint base (base_url) ──────────────────
175
+ // Mirrors the harness OpenAIProvider/LocalProvider: lets a method point at
176
+ // any OpenAI-compatible server (Ollama, vLLM, LM Studio, Groq, Together).
177
+
178
+ /**
179
+ * Default API base (no /chat/completions) for this provider. Override in
180
+ * subclasses (e.g. OpenAI → api.openai.com/v1, Local → Ollama).
181
+ * @returns {string|null}
182
+ */
183
+ _getDefaultApiBase() { return null; }
184
+
185
+ /**
186
+ * Env var that overrides the API base. Default OPENAI_API_BASE (shared with
187
+ * the harness). Subclasses may override (e.g. LocalMethod → LOCAL_API_BASE).
188
+ * @returns {string}
189
+ */
190
+ _getApiBaseEnvVar() { return 'OPENAI_API_BASE'; }
191
+
192
+ /**
193
+ * Resolve the endpoint base (no /chat/completions). Precedence:
194
+ * options.baseUrl / this.options.baseUrl > env > subclass default.
195
+ * Normalizes a trailing slash and a full .../chat/completions path.
196
+ * @returns {string|null}
197
+ */
198
+ _resolveApiBase(options = {}) {
199
+ const envVar = this._getApiBaseEnvVar();
200
+ const fromEnv = envVar
201
+ ? (getEnvOrFileVar(envVar) || getEnvOrFileVar(envVar, options.cwd))
202
+ : null;
203
+ const base = options.baseUrl
204
+ || (this.options && this.options.baseUrl)
205
+ || fromEnv
206
+ || this._getDefaultApiBase();
207
+ if (!base) return null;
208
+ let b = String(base).replace(/\/+$/, '');
209
+ if (b.endsWith('/chat/completions')) {
210
+ b = b.slice(0, -'/chat/completions'.length);
211
+ }
212
+ return b;
213
+ }
214
+
215
+ /**
216
+ * Validate the model string before making API calls.
217
+ *
218
+ * Checks:
219
+ * 1. OpenRouter-format model strings (contain '/') — wrong method
220
+ * 2. Model belongs to a different provider (e.g., claude-* on OpenAI)
221
+ * 3. Model exists in the provider's API (runtime fetch, cached)
222
+ *
223
+ * Logs warnings but does NOT block — the provider API will give
224
+ * the definitive answer. This is a DX aid, not a gate.
225
+ *
226
+ * @param {string} model - Model ID to validate
227
+ * @param {string} apiKey - API key for model list fetch
228
+ */
229
+ async _validateModel(model, apiKey) {
230
+ if (this._modelValidated) return;
231
+ this._modelValidated = true;
232
+
233
+ const label = this._getProviderLabel();
234
+
235
+ // Check 1: OpenRouter-format model string (contains '/')
236
+ if (model.includes('/')) {
237
+ output.warn(`${label}: model "${model}" looks like an OpenRouter path.`);
238
+ output.warn(`Direct providers use bare model names (e.g., "${this._getDefaultModel()}").`);
239
+ output.warn(`To use OpenRouter models, set method to 'llm' instead.`);
240
+ return;
241
+ }
242
+
243
+ // Check 2: Model belongs to a different provider
244
+ for (const [provider, { prefixes, label: providerLabel }] of Object.entries(MODEL_PATTERNS)) {
245
+ if (provider === this.name) continue; // skip own provider
246
+ const matchesOther = prefixes.some(p => model.startsWith(p));
247
+ if (matchesOther) {
248
+ const article = /^[aeiou]/i.test(providerLabel) ? 'an' : 'a';
249
+ output.warn(`${label}: model "${model}" is ${article} ${providerLabel} model.`);
250
+ output.warn(`This provider (${this.name}) cannot serve ${providerLabel} models.`);
251
+ output.warn(`Use --method ${provider} or set "method": "${provider}" in config.`);
252
+ return;
253
+ }
254
+ }
255
+
256
+ // Check 3: Runtime model list validation (cached per-process)
257
+ try {
258
+ let modelSet = _modelListCache.get(this.name);
259
+
260
+ // null = already tried and failed; undefined = never tried
261
+ if (modelSet === undefined) {
262
+ const models = await this._fetchModels(apiKey);
263
+ if (models && models.length > 0) {
264
+ modelSet = new Set(models);
265
+ _modelListCache.set(this.name, modelSet);
266
+ } else {
267
+ // Mark as failed so we don't retry on every batch
268
+ _modelListCache.set(this.name, null);
269
+ }
270
+ }
271
+
272
+ if (modelSet && !modelSet.has(model)) {
273
+ // Find close matches to suggest
274
+ const suggestions = [...modelSet]
275
+ .filter(m => {
276
+ // Only suggest models that support generateContent-style operations
277
+ // (not embedding models, not vision-only, etc.)
278
+ const base = model.split('-')[0];
279
+ return m.startsWith(base);
280
+ })
281
+ .sort()
282
+ .slice(0, 5);
283
+
284
+ output.warn(`${label}: model "${model}" not found in available models.`);
285
+ if (suggestions.length > 0) {
286
+ output.warn(`Similar models: ${suggestions.join(', ')}`);
287
+ }
288
+ output.warn(`The API call will proceed — the provider will give the final verdict.`);
289
+ }
290
+ } catch {
291
+ // Model listing failed silently — don't block translation.
292
+ // The actual translate call will surface any real model errors.
293
+ }
294
+ }
295
+
296
+ /**
297
+ * Determine quality tier based on the model name.
298
+ *
299
+ * Provider subclasses can override _getModelTier() for provider-specific
300
+ * mappings. Default: 'standard'.
301
+ */
302
+ getQualityTier(pairConfig = {}) {
303
+ const model = pairConfig.model || this._getDefaultModel();
304
+ return this._getModelTier(model);
305
+ }
306
+
307
+ /**
308
+ * Map a model name to a quality tier. Override in subclasses.
309
+ * @param {string} model
310
+ * @returns {'budget'|'standard'|'premium'}
311
+ */
312
+ _getModelTier(model) {
313
+ return 'standard';
314
+ }
315
+
316
+ // ── Core translate() — shared across all direct LLM providers ──
317
+
318
+ async translate(keys, sourceFlat, pairConfig, options) {
319
+ const apiKey = this._resolveApiKey(options);
320
+
321
+ if (!apiKey) {
322
+ output.warn(`${this._getProviderLabel()}: no API key — skipping.`);
323
+ return null;
324
+ }
325
+
326
+ const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
327
+ const model = pairConfig.model || options.model || this._getDefaultModel();
328
+ const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
329
+ const langConfig = {
330
+ name: pairConfig.name,
331
+ register: pairConfig.register,
332
+ };
333
+
334
+ // Validate model on first call (logs warnings, does not block)
335
+ await this._validateModel(model, apiKey);
336
+
337
+ // Load coaching data if available (.champollion/coaching/<locale>.json)
338
+ const cwd = options.cwd || process.cwd();
339
+ const coachingDir = path.join(cwd, DEFAULT_COACHING_DIR);
340
+ const coaching = loadCoachingData(coachingDir, pairConfig.target, this._coachingCache);
341
+
342
+ // If coaching data exists, use the coached system message (grammar/style in system
343
+ // prompt for provider-level caching). Otherwise use the standard system message.
344
+ const systemMessage = coaching
345
+ ? buildCoachedSystemMessage(langConfig, coaching)
346
+ : buildSystemMessage(langConfig);
347
+ const allTranslated = {};
348
+
349
+ // Wrap the batch function to inject dictionary hints when coaching is active.
350
+ // Thread the resolved temperature so _callProviderBatch doesn't need pairConfig.
351
+ const resolvedTemperature = pairConfig.temperature ?? DEFAULT_TEMPERATURE;
352
+ const descriptions = options.descriptions || null;
353
+ const batchFn = (batch, opts) => this._callProviderBatch(batch, { ...opts, apiKey, model, coaching, temperature: resolvedTemperature, descriptions });
354
+
355
+ const batchChunks = [];
356
+ for (let i = 0; i < keys.length; i += batchSize) {
357
+ batchChunks.push(keys.slice(i, i + batchSize));
358
+ }
359
+
360
+ await pMap(batchChunks, async (chunk, idx) => {
361
+ const toTranslate = {};
362
+ for (const key of chunk) {
363
+ toTranslate[key] = sourceFlat[key];
364
+ }
365
+
366
+ const result = await this._translateWithCascade(toTranslate, langConfig, {
367
+ apiKey,
368
+ model,
369
+ batchNum: idx + 1,
370
+ maxRetries,
371
+ systemMessage,
372
+ }, batchFn);
373
+
374
+ if (result) {
375
+ Object.assign(allTranslated, result);
376
+ }
377
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
378
+
379
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
380
+ }
381
+
382
+ // ── Core translateContent() — shared across all direct LLM providers ──
383
+
384
+ async translateContent(prompt, pairConfig, options) {
385
+ const apiKey = this._resolveApiKey(options);
386
+
387
+ if (!apiKey) {
388
+ output.warn(`${this._getProviderLabel()}: no API key — skipping.`);
389
+ return null;
390
+ }
391
+
392
+ const model = pairConfig.model || options.model || this._getDefaultModel();
393
+
394
+ // Prepend coaching context (grammar/style rules) to content prompts when available.
395
+ // Dictionary matching is skipped for freeform content — it's too unpredictable.
396
+ const cwd = options.cwd || process.cwd();
397
+ const coachingDir = path.join(cwd, DEFAULT_COACHING_DIR);
398
+ const coaching = loadCoachingData(coachingDir, pairConfig.target, this._coachingCache);
399
+ let augmentedPrompt = prompt;
400
+ if (coaching) {
401
+ const block = buildContentCoachingBlock(coaching);
402
+ if (block) {
403
+ augmentedPrompt = block + '\n\n' + prompt;
404
+ }
405
+ }
406
+
407
+ // Round 1: standard timeout (2× base = 60s)
408
+ const result = await this._callProviderDirect({
409
+ prompt: augmentedPrompt,
410
+ apiKey,
411
+ model,
412
+ temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
413
+ timeoutMs: REQUEST_TIMEOUT_MS * 2,
414
+ label: `${this._getProviderLabel()} Content`,
415
+ });
416
+
417
+ if (result) return result;
418
+
419
+ // Round 2: escalated — longer cool-down and 4× timeout (120s)
420
+ const label = `${this._getProviderLabel()} Content (escalated)`;
421
+ output.warn(`⟳ ${label}: standard retries exhausted — escalating with extended timeout...`);
422
+ await sleep(10_000);
423
+
424
+ return this._callProviderDirect({
425
+ prompt: augmentedPrompt,
426
+ apiKey,
427
+ model,
428
+ temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
429
+ timeoutMs: REQUEST_TIMEOUT_MS * 4,
430
+ label,
431
+ });
432
+ }
433
+
434
+ // ── Shared batch call — builds user message with coaching hints ──
435
+
436
+ async _callProviderBatch(toTranslate, options) {
437
+ const { apiKey, model, batchNum, systemMessage, coaching, temperature } = options;
438
+
439
+ // Build user message — inject dictionary term matches when coaching is active.
440
+ // Dictionary hints go in the user message (per-batch) rather than the system
441
+ // message (cached) because they're specific to the current batch's source values.
442
+ let prompt;
443
+ if (coaching && coaching.dictionary) {
444
+ const dictHints = findDictionaryMatches(toTranslate, coaching.dictionary);
445
+ const typeHints = inferKeyTypes(toTranslate);
446
+ let userMessage = '';
447
+ if (dictHints.length > 0) {
448
+ userMessage += 'REQUIRED TERMINOLOGY (use these exact translations):\n';
449
+ userMessage += dictHints.map(h => ` • "${h.term}" → "${h.translation}"`).join('\n');
450
+ userMessage += '\n\n';
451
+ }
452
+ if (typeHints.length > 0) {
453
+ userMessage += `UI context for these keys:\n${typeHints.join('\n')}\n\n`;
454
+ }
455
+ userMessage += JSON.stringify(toTranslate, null, 2);
456
+ prompt = userMessage;
457
+ } else {
458
+ prompt = buildUserMessage(toTranslate, options.descriptions || undefined);
459
+ }
460
+
461
+ const label = this._getProviderLabel();
462
+ const content = await this._callProviderDirect({
463
+ prompt,
464
+ systemMessage,
465
+ apiKey,
466
+ model,
467
+ temperature: temperature ?? DEFAULT_TEMPERATURE,
468
+ label: `${label} Batch ${batchNum}`,
469
+ isJsonMode: true,
470
+ });
471
+
472
+ if (!content) return null;
473
+
474
+ try {
475
+ const parsed = JSON.parse(content);
476
+ const expectedKeys = new Set(Object.keys(toTranslate));
477
+ const validated = {};
478
+ for (const [key, value] of Object.entries(parsed)) {
479
+ if (expectedKeys.has(key) && typeof value === 'string' && !isUnsafeKey(key)) {
480
+ validated[key] = value;
481
+ }
482
+ }
483
+ return Object.keys(validated).length > 0 ? validated : null;
484
+ } catch (err) {
485
+ output.error(`${label} Batch ${batchNum}: JSON parse error — ${err.message}`);
486
+ return { _parseError: true, rawContent: content, error: err.message };
487
+ }
488
+ }
489
+
490
+ // ── Shared direct call — retry loop with exponential backoff ──
491
+
492
+ async _callProviderDirect({
493
+ prompt,
494
+ systemMessage,
495
+ apiKey,
496
+ model,
497
+ temperature = DEFAULT_TEMPERATURE,
498
+ timeoutMs = REQUEST_TIMEOUT_MS,
499
+ label,
500
+ isJsonMode = false,
501
+ }) {
502
+ label = label || this._getProviderLabel();
503
+
504
+ for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
505
+ try {
506
+ const controller = new AbortController();
507
+ const timeoutId = setTimeout(() => controller.abort(), timeoutMs);
508
+
509
+ const { url, headers, body } = this._buildApiRequest({
510
+ prompt,
511
+ systemMessage,
512
+ apiKey,
513
+ model,
514
+ temperature,
515
+ isJsonMode,
516
+ });
517
+
518
+ const response = await fetch(url, {
519
+ method: 'POST',
520
+ headers,
521
+ body: JSON.stringify(body),
522
+ signal: controller.signal,
523
+ });
524
+
525
+ clearTimeout(timeoutId);
526
+
527
+ if (isRetryable(response.status)) {
528
+ if (attempt < MAX_RETRIES) {
529
+ const delay = getBackoffDelay(attempt);
530
+ output.warn(`⏳ ${label}: ${response.status} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
531
+ await sleep(delay);
532
+ continue;
533
+ }
534
+ output.error(`${label}: ${response.status} after ${MAX_RETRIES + 1} attempts`);
535
+ recordTranslationError(response.status);
536
+ return null;
537
+ }
538
+
539
+ if (!response.ok) {
540
+ const errorBody = await response.text();
541
+ output.error(`${label}: API error ${response.status} — ${errorBody}`);
542
+ recordTranslationError(response.status);
543
+ return null;
544
+ }
545
+
546
+ const data = await response.json();
547
+ const content = this._extractResponseText(data);
548
+ if (!content) {
549
+ // Empty responses — retry with backoff, same as HTTP errors.
550
+ // Common with long content where the model hits output limits.
551
+ if (attempt < MAX_RETRIES) {
552
+ const delay = getBackoffDelay(attempt);
553
+ output.warn(`⏳ ${label}: empty response — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
554
+ await sleep(delay);
555
+ continue;
556
+ }
557
+ output.error(`${label}: empty response after ${MAX_RETRIES + 1} attempts`);
558
+ return null;
559
+ }
560
+
561
+ return stripCodeFences(content.trim());
562
+
563
+ } catch (err) {
564
+ const isTimeout = err.name === 'AbortError';
565
+ const errLabel = isTimeout ? 'timeout' : err.message;
566
+
567
+ if (attempt < MAX_RETRIES) {
568
+ const delay = getBackoffDelay(attempt);
569
+ output.warn(`⏳ ${label}: ${errLabel} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
570
+ await sleep(delay);
571
+ continue;
572
+ }
573
+
574
+ output.error(`${label} failed: ${errLabel}`);
575
+ return null;
576
+ }
577
+ }
578
+ return null;
579
+ }
580
+ }
581
+
582
+ export {
583
+ DirectLLMMethod,
584
+ MODEL_PATTERNS,
585
+ _modelListCache,
586
+ };