champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,592 @@
1
+ /**
2
+ * LLM Translation Method — direct LLM prompting via OpenRouter.
3
+ *
4
+ * This is the default translation method and the foundation all other
5
+ * methods build on. It extracts the existing translateBatch/translateRawContent
6
+ * logic from the v2 translate.js into a proper method class.
7
+ *
8
+ * HOW IT WORKS:
9
+ * 1. Receives keys + source values from the orchestrator
10
+ * 2. Chunks them into batches (default 80 keys per batch)
11
+ * 3. Builds a register-steered prompt per batch
12
+ * 4. Sends to OpenRouter with exponential backoff retry
13
+ * 5. Validates response (only accept keys we sent, block prototype pollution)
14
+ * 6. Returns merged results
15
+ *
16
+ * PROMPT CACHING:
17
+ * The prompt is split into two parts:
18
+ * - system message: register, rules, UI context (identical across batches)
19
+ * - user message: JSON payload (varies per batch)
20
+ * This enables provider-level prompt caching (Anthropic, Gemini) on the
21
+ * system preamble, reducing token costs by 30-50% for large syncs.
22
+ *
23
+ * RETRY CASCADE:
24
+ * On JSON parse failure (malformed model output), the cascade retries:
25
+ * 1. Full batch (original size)
26
+ * 2. Half-batches (split in two, retry each half)
27
+ * 3. Individual keys (batchSize=1) — capped per cascade invocation by
28
+ * CHAMPOLLION_MAX_KEY_FANOUT (default 16, 0 disables the per-key
29
+ * fallback). Keys beyond the cap are skipped with one summary warn
30
+ * and retry on the next sync via the failed-key machinery.
31
+ * Each level is tried before escalating to the next. A maxRetries budget
32
+ * cap prevents infinite token spend on the batch/half rounds.
33
+ *
34
+ * COST PROFILE: ~$0.01 per 1k keys at GPT-4o-mini pricing.
35
+ * QUALITY TIER: standard — no post-processing or verification.
36
+ */
37
+
38
+ import { TranslationMethod } from './base.js';
39
+ import { REQUEST_TIMEOUT_MS, sleep } from './http-utils.js';
40
+ import { callOpenRouter, callOpenRouterJSON } from './openrouter-client.js';
41
+ import { estimateOpenRouterCost } from './openrouter-pricing.js';
42
+ import { DEFAULT_OPENROUTER_MODEL, DEFAULT_BATCH_SIZE, DEFAULT_TEMPERATURE, DEFAULT_MAX_RETRIES, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
43
+ import { isUnsafeKey } from '../security.js';
44
+ import { pMap } from '../concurrent.js';
45
+ import { output } from '../output.js';
46
+ import { getEnvOrFileVar } from '../api-key.js';
47
+
48
+
49
+ /** Default per-cascade cap on individual-key fallback calls (see below). */
50
+ const DEFAULT_MAX_KEY_FANOUT = 16;
51
+
52
+ /**
53
+ * Resolve the per-key fallback fan-out cap.
54
+ *
55
+ * CHAMPOLLION_MAX_KEY_FANOUT bounds how many INDIVIDUAL-KEY API calls one
56
+ * cascade invocation may fan out into after half-batch parse failures.
57
+ * Unset/empty → 16; 0 → per-key fallback disabled entirely; anything
58
+ * non-numeric or negative falls back to the default (fail-safe, not free).
59
+ *
60
+ * @returns {number}
61
+ */
62
+ function maxKeyFanout() {
63
+ const raw = process.env.CHAMPOLLION_MAX_KEY_FANOUT;
64
+ if (raw === undefined || raw === null || String(raw).trim() === '') {
65
+ return DEFAULT_MAX_KEY_FANOUT;
66
+ }
67
+ const n = Number(raw);
68
+ if (!Number.isFinite(n) || n < 0) return DEFAULT_MAX_KEY_FANOUT;
69
+ return Math.floor(n);
70
+ }
71
+
72
+ class LLMMethod extends TranslationMethod {
73
+ constructor(options = {}) {
74
+ super('llm', options);
75
+ }
76
+
77
+ /**
78
+ * Translate a batch of key-value pairs via OpenRouter.
79
+ *
80
+ * @param {string[]} keys - Flat dot-notation keys to translate
81
+ * @param {object} sourceFlat - Full flattened source locale
82
+ * @param {object} pairConfig - Pair config (method, model, register, name, etc.)
83
+ * @param {object} options - { apiKey, batchSize }
84
+ * @returns {object|null} Map of key → translated value, or null if all failed
85
+ */
86
+ async translate(keys, sourceFlat, pairConfig, options) {
87
+ const { apiKey } = options;
88
+ const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
89
+ const model = pairConfig.model || options.model || DEFAULT_OPENROUTER_MODEL;
90
+ const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
91
+ if (!apiKey) {
92
+ output.warn('LLM translate: no API key provided — skipping batch.');
93
+ return null;
94
+ }
95
+
96
+ const langConfig = {
97
+ name: pairConfig.name,
98
+ register: pairConfig.register,
99
+ // Language-specific gender guidance from the card (e.g., écriture inclusive
100
+ // for French, Doppelpunkt notation for German). Falls back to a generic
101
+ // rule in buildSystemMessage when null.
102
+ genderGuidance: pairConfig.genderGuidance || null,
103
+ // User-provided global context (e.g., "This is a developer tool README").
104
+ // Injected into the system message to give the LLM domain awareness.
105
+ promptContext: pairConfig.promptContext || null,
106
+ // Coaching prompt text — free-text instructions read from a coaching file.
107
+ // Injected between register and rules in the system message.
108
+ coachingPrompt: pairConfig.coachingPrompt || null,
109
+ };
110
+
111
+ // Build the system message once — identical across all batches for this locale.
112
+ // This enables provider-level prompt caching (Anthropic, Gemini).
113
+ const systemMessage = buildSystemMessage(langConfig);
114
+
115
+ const allTranslated = {};
116
+
117
+ // ── Parallel batch execution ───────────────────────────────────────
118
+ // Fire all batch chunks concurrently (up to 4 in parallel).
119
+ // Each batch cascades independently on parse failure (full → half →
120
+ // individual), so parallelism doesn't break the retry logic.
121
+ // Object.assign on allTranslated is safe: keys don't overlap across
122
+ // batches, and JS property assignment is synchronous between awaits.
123
+ const batchChunks = [];
124
+ for (let i = 0; i < keys.length; i += batchSize) {
125
+ batchChunks.push({ chunk: keys.slice(i, i + batchSize), offset: i });
126
+ }
127
+
128
+ let completedKeys = 0;
129
+ await pMap(batchChunks, async ({ chunk, offset }, idx) => {
130
+ const toTranslate = {};
131
+ for (const key of chunk) {
132
+ toTranslate[key] = sourceFlat[key];
133
+ }
134
+
135
+ const result = await this._translateWithCascade(toTranslate, langConfig, {
136
+ apiKey,
137
+ model,
138
+ batchNum: idx + 1,
139
+ maxRetries,
140
+ systemMessage,
141
+ temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
142
+ descriptions: options.descriptions || null,
143
+ }, this._callOpenRouterBatch.bind(this));
144
+
145
+ if (result) {
146
+ Object.assign(allTranslated, result);
147
+ }
148
+
149
+ // Progress callback — report cumulative completion
150
+ completedKeys += chunk.length;
151
+ if (options.onProgress) {
152
+ options.onProgress(
153
+ Math.min(completedKeys, keys.length),
154
+ keys.length,
155
+ );
156
+ }
157
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
158
+
159
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
160
+ }
161
+
162
+ /**
163
+ * Translate freeform content (Markdown body, etc.) via OpenRouter.
164
+ *
165
+ * Uses a two-round escalation strategy for reliability:
166
+ * Round 1 (standard): 4 attempts, 2× base timeout (60s), normal backoff
167
+ * Round 2 (escalated): If round 1 fails, 10s cool-down, then 4 attempts
168
+ * with 4× base timeout (120s). Handles long docs where the model needs
169
+ * more time to generate the full output.
170
+ *
171
+ * Total worst-case: 8 API attempts before returning null.
172
+ *
173
+ * @param {string} prompt - Complete translation prompt
174
+ * @param {object} pairConfig - Pair config
175
+ * @param {object} options - { apiKey }
176
+ * @returns {string|null} Translated text, or null on failure
177
+ */
178
+ async translateContent(prompt, pairConfig, options) {
179
+ const { apiKey } = options;
180
+ const model = pairConfig.model || options.model || DEFAULT_OPENROUTER_MODEL;
181
+ if (!apiKey) {
182
+ output.warn('LLM translateContent: no API key provided — skipping.');
183
+ return null;
184
+ }
185
+
186
+ // Round 1: standard timeout (2× base = 60s)
187
+ const result = await callOpenRouter({
188
+ prompt,
189
+ apiKey,
190
+ model,
191
+ temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
192
+ timeoutMs: REQUEST_TIMEOUT_MS * 2,
193
+ label: 'Content',
194
+ });
195
+
196
+ if (result) return result;
197
+
198
+ // Round 2: escalated — longer cool-down and 4× timeout (120s).
199
+ // This handles long documents where the model needs more generation
200
+ // time, or transient API issues that resolve after a longer pause.
201
+ output.warn('⟳ Content: standard retries exhausted — escalating with extended timeout...');
202
+ await sleep(10_000);
203
+
204
+ return callOpenRouter({
205
+ prompt,
206
+ apiKey,
207
+ model,
208
+ temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
209
+ timeoutMs: REQUEST_TIMEOUT_MS * 4,
210
+ label: 'Content (escalated)',
211
+ });
212
+ }
213
+
214
+ /**
215
+ * Cost estimation — fetches live per-token pricing from OpenRouter.
216
+ *
217
+ * The pricing module fetches once per process and caches.
218
+ * Falls back to 'unknown' if offline or the model isn't found.
219
+ *
220
+ * @param {number} keyCount - Number of keys to translate
221
+ * @param {object} [pairConfig] - Pair config containing the model ID
222
+ */
223
+ async estimateCost(keyCount, pairConfig = {}) {
224
+ const model = pairConfig.model || DEFAULT_OPENROUTER_MODEL;
225
+ return estimateOpenRouterCost(keyCount, model, { coached: false });
226
+ }
227
+
228
+ checkReadiness(context) {
229
+ if (!context.apiKey) {
230
+ return { ready: false, reason: 'No OpenRouter API key (OPENROUTER_API_KEY).' };
231
+ }
232
+ return { ready: true };
233
+ }
234
+
235
+ getQualityTier() {
236
+ return 'standard';
237
+ }
238
+
239
+ getProvenance() {
240
+ return {
241
+ resources: [],
242
+ commercialReady: true,
243
+ flags: [],
244
+ };
245
+ }
246
+
247
+ getSetupHelp() {
248
+ // Resolve through the same env-or-file lookup the runtime uses, so a key
249
+ // that only lives in .env.local doesn't get misreported as "Missing API
250
+ // Key" after a real 401 — that would point the user at the wrong fix.
251
+ const apiKey = getEnvOrFileVar('OPENROUTER_API_KEY');
252
+ if (!apiKey) {
253
+ return [
254
+ '',
255
+ ' ┌─ Missing API Key ─────────────────────────────────────────────┐',
256
+ ' │ The LLM method requires an OpenRouter API key. │',
257
+ ' │ │',
258
+ ' │ 1. Sign up at https://openrouter.ai (free tier available) │',
259
+ ' │ 2. Run: export OPENROUTER_API_KEY=sk-or-v1-... │',
260
+ ' │ 3. Or add to .env.local: OPENROUTER_API_KEY=sk-or-v1-... │',
261
+ ' │ │',
262
+ ' │ Alternative: use Google Translate instead (key-value only): │',
263
+ ' │ export GOOGLE_TRANSLATE_API_KEY=... │',
264
+ ' │ champollion sync --method google-translate │',
265
+ ' └────────────────────────────────────────────────────────────────┘',
266
+ ];
267
+ }
268
+ return this._apiFailureHelp('OpenRouter dashboard');
269
+ }
270
+
271
+ // -----------------------------------------------------------------
272
+ // Private helpers
273
+ // -----------------------------------------------------------------
274
+
275
+ /**
276
+ * Retry cascade for a single batch.
277
+ *
278
+ * On JSON parse failure (model returned garbage), the cascade:
279
+ * 1. Retries the full batch once
280
+ * 2. Splits into two half-batches and retries each
281
+ * 3. Falls back to individual key translation (batchSize=1)
282
+ *
283
+ * The maxRetries budget limits total cascade depth to prevent
284
+ * infinite token spend on a language the model genuinely can't handle.
285
+ *
286
+ * This method is also used by LLMCoachedMethod via composition — the
287
+ * coached method passes its own batch function that injects dictionary
288
+ * hints into the prompt.
289
+ *
290
+ * @param {object} toTranslate - Key-value map to translate
291
+ * @param {object} langConfig - { name, register }
292
+ * @param {object} options - { apiKey, model, batchNum, maxRetries, systemMessage }
293
+ * @param {Function} batchFn - (toTranslate, options) => Promise<result>
294
+ * The function that makes a single batch API call.
295
+ * @param {string} label - Prefix for log messages (e.g., '' or 'Coached ')
296
+ * @returns {object|null} Validated key-value map, or null
297
+ */
298
+ async _translateWithCascade(toTranslate, langConfig, options, batchFn, label = '') {
299
+ const { maxRetries } = options;
300
+ let retriesUsed = 0;
301
+
302
+ // Attempt 1: Full batch
303
+ const result = await batchFn(toTranslate, options);
304
+
305
+ // Success — return the result
306
+ if (result && !result._parseError) return result;
307
+
308
+ // API failure (null) — nothing to retry with smaller batches
309
+ if (!result) return null;
310
+
311
+ // Parse error — start the cascade
312
+ retriesUsed++;
313
+ if (retriesUsed > maxRetries) {
314
+ this._logCascadeExhausted(toTranslate, options, label);
315
+ return null;
316
+ }
317
+
318
+ const keys = Object.keys(toTranslate);
319
+
320
+ // Attempt 2: Half-batches (only if batch has >1 key)
321
+ if (keys.length > 1) {
322
+ const mid = Math.ceil(keys.length / 2);
323
+ const firstHalf = {};
324
+ const secondHalf = {};
325
+ for (let i = 0; i < keys.length; i++) {
326
+ const key = keys[i];
327
+ if (i < mid) firstHalf[key] = toTranslate[key];
328
+ else secondHalf[key] = toTranslate[key];
329
+ }
330
+
331
+ output.warn(`⟳ ${label}Batch ${options.batchNum}: parse error — splitting ${keys.length} keys into halves (${Object.keys(firstHalf).length} + ${Object.keys(secondHalf).length})...`);
332
+
333
+ // Per-key fallback fan-out cap, counted across THIS cascade invocation
334
+ // (both halves share it). The retry budget gates batch/half rounds;
335
+ // this cap bounds the terminal per-key leaves so a big batch that
336
+ // parses badly can't fan out into one API call per key (80 calls at
337
+ // default batch size). Skipped keys return null and flow into the
338
+ // existing failed-key machinery — they retry on the next sync.
339
+ const fanoutCap = maxKeyFanout();
340
+ let fanoutTranslated = 0;
341
+ let fanoutFailed = 0;
342
+ let fanoutSkipped = 0;
343
+
344
+ const merged = {};
345
+ for (const half of [firstHalf, secondHalf]) {
346
+ retriesUsed++;
347
+ if (retriesUsed > maxRetries) {
348
+ this._logCascadeExhausted(half, options, label);
349
+ break;
350
+ }
351
+
352
+ const halfResult = await batchFn(half, options);
353
+
354
+ if (halfResult && !halfResult._parseError) {
355
+ Object.assign(merged, halfResult);
356
+ } else if (halfResult?._parseError) {
357
+ // Half-batch also failed — try individual keys.
358
+ //
359
+ // Terminal per-key calls do NOT count against the retry budget.
360
+ // The budget gates how many batch/half RETRY ROUNDS we attempt
361
+ // (the recoverable splitting cascade); once we've reached the
362
+ // individual-key leaves there is nothing left to split, so each
363
+ // key gets its one final attempt regardless of budget. Counting
364
+ // them made the advertised batch→half→individual cascade
365
+ // structurally unreachable at defaults (maxRetries=3, batchSize=80):
366
+ // a single parse failure would exhaust the budget after recovering
367
+ // just one key and abandon the other 79. The only bound on the
368
+ // per-key leaves is the fan-out cap above (CHAMPOLLION_MAX_KEY_FANOUT).
369
+ if (fanoutCap > 0) {
370
+ output.warn(`⟳ ${label}Half-batch parse error — falling back to individual keys...`);
371
+ }
372
+ for (const [key, value] of Object.entries(half)) {
373
+ if (fanoutTranslated + fanoutFailed >= fanoutCap) {
374
+ fanoutSkipped++;
375
+ continue;
376
+ }
377
+ const singleResult = await batchFn({ [key]: value }, options);
378
+ if (singleResult && !singleResult._parseError) {
379
+ fanoutTranslated++;
380
+ Object.assign(merged, singleResult);
381
+ } else {
382
+ fanoutFailed++;
383
+ output.error(`Key "${key}" failed even as individual ${label.toLowerCase().trim() || 'LLM'} translation.`);
384
+ }
385
+ }
386
+ }
387
+ // halfResult === null (API failure) — skip this half, nothing to retry
388
+ }
389
+
390
+ if (fanoutSkipped > 0) {
391
+ output.warn(
392
+ `${label}Batch ${options.batchNum}: per-key fallback capped at ${fanoutCap} ` +
393
+ `(CHAMPOLLION_MAX_KEY_FANOUT) — ${fanoutTranslated} key(s) translated individually, ` +
394
+ `${fanoutSkipped} skipped; skipped keys will retry next sync.`
395
+ );
396
+ }
397
+
398
+ return Object.keys(merged).length > 0 ? merged : null;
399
+ }
400
+
401
+ // Single key that failed to parse — one more try
402
+ retriesUsed++;
403
+ if (retriesUsed > maxRetries) {
404
+ this._logCascadeExhausted(toTranslate, options, label);
405
+ return null;
406
+ }
407
+
408
+ const retryResult = await batchFn(toTranslate, options);
409
+ if (retryResult && !retryResult._parseError) return retryResult;
410
+
411
+ this._logCascadeExhausted(toTranslate, options, label);
412
+ return null;
413
+ }
414
+
415
+ /**
416
+ * Make a single OpenRouter batch call.
417
+ * Builds the user message (JSON payload) and delegates to openrouter-client.
418
+ */
419
+ async _callOpenRouterBatch(toTranslate, options) {
420
+ const { apiKey, model, batchNum, systemMessage, temperature } = options;
421
+ const userMessage = buildUserMessage(toTranslate, options.descriptions || undefined);
422
+
423
+ return callOpenRouterJSON({
424
+ prompt: userMessage,
425
+ systemMessage,
426
+ apiKey,
427
+ model,
428
+ temperature: temperature ?? DEFAULT_TEMPERATURE,
429
+ label: `Batch ${batchNum}`,
430
+ expectedKeys: new Set(Object.keys(toTranslate)),
431
+ isUnsafeKey,
432
+ });
433
+ }
434
+
435
+ /**
436
+ * Log cascade exhaustion with actionable guidance.
437
+ *
438
+ * @param {object} toTranslate - The keys that failed
439
+ * @param {object} options - { batchNum, maxRetries }
440
+ * @param {string} label - Prefix for log messages (e.g., 'Coached ')
441
+ */
442
+ _logCascadeExhausted(toTranslate, options, label = '') {
443
+ const keys = Object.keys(toTranslate);
444
+ output.error(`${label}Batch ${options.batchNum}: retry cascade exhausted (${options.maxRetries} max).`);
445
+ output.error(`${keys.length} key(s) could not be translated.`);
446
+ if (!label) {
447
+ output.error(`Consider increasing maxRetries or using a more capable model:`);
448
+ output.error(`"languages": { "<code>": { "model": "google/gemini-2.5-pro", "batchSize": 5, "maxRetries": 5 } }`);
449
+ }
450
+ }
451
+ }
452
+
453
+ // -----------------------------------------------------------------
454
+ // Prompt building — split into system (cached) + user (per-batch)
455
+ // -----------------------------------------------------------------
456
+
457
+ /**
458
+ * Build the system message preamble (cached across batches).
459
+ *
460
+ * Contains: role definition, register, gender guidance, prompt context, translation rules.
461
+ * This is identical for every batch targeting the same locale,
462
+ * so providers cache it automatically after the first batch.
463
+ *
464
+ * Gender guidance: When the language card provides specific guidance
465
+ * (e.g., écriture inclusive for French), it replaces the generic rule.
466
+ * This ensures Hindi, Arabic, etc. get language-appropriate advice
467
+ * instead of a one-size-fits-all line.
468
+ *
469
+ * Prompt context: User-provided global context about the project/content
470
+ * being translated (e.g., "This is a developer tool for i18n"). Injected
471
+ * after the role line to give the LLM domain awareness.
472
+ *
473
+ * @param {object} langConfig - { name, register, genderGuidance?, promptContext? }
474
+ * @returns {string} System message
475
+ */
476
+ function buildSystemMessage(langConfig) {
477
+ // Use language-specific gender guidance when available from the card,
478
+ // otherwise fall back to a generic rule for unknown languages.
479
+ const genderRule = langConfig.genderGuidance
480
+ ? `- Gender: ${langConfig.genderGuidance}`
481
+ : '- When gender is ambiguous, prefer gender-neutral forms or the most inclusive option available in ' + langConfig.name + '.';
482
+
483
+ // Inject user-provided promptContext to give the LLM domain awareness.
484
+ // This is set in the top-level config and flows through the pair graph.
485
+ const contextBlock = langConfig.promptContext
486
+ ? `\nContext: ${langConfig.promptContext}\n`
487
+ : '';
488
+
489
+ // Coaching guidance: free-text instructions read from a coaching file.
490
+ // This sits between the register line and the rules block so the LLM
491
+ // treats it as authoritative style/domain guidance rather than a mere hint.
492
+ const coachingBlock = langConfig.coachingPrompt
493
+ ? `\nCoaching guidance:\n${langConfig.coachingPrompt}\n`
494
+ : '';
495
+
496
+ return `You are translating UI strings for a web/mobile application from English to ${langConfig.name}.
497
+ ${contextBlock}
498
+ Register/tone: ${langConfig.register}
499
+ ${coachingBlock}
500
+ Rules:
501
+ - Translate ONLY the values, keep the keys exactly as-is.
502
+ - Proper nouns (product names, company names, place names) should NOT be translated.
503
+ - Technical terms and role descriptions that are industry-standard should stay in English.
504
+ ${genderRule}
505
+ - Respect the UI element type: button labels should be concise, descriptions can be natural-length, error messages should be clear and direct.
506
+ - Quotation marks inside a translated value: use the target language's typographic quotation marks (e.g. “…”, «…», 「…」); if you must use ASCII double quotes, escape them as \\" so the JSON stays valid.
507
+ - Return ONLY valid JSON, no markdown fences, no explanation.`;
508
+ }
509
+
510
+ /**
511
+ * Build the user message (varies per batch).
512
+ *
513
+ * Contains: UI context hints for key names + the JSON payload.
514
+ * This changes with every batch, so it's never cached.
515
+ *
516
+ * Descriptions: When Docusaurus {message, description} files provide
517
+ * developer-written descriptions (e.g., "The title of the blog page"),
518
+ * those are included as additional context hints. This helps the LLM
519
+ * disambiguate polysemous terms — e.g., "Post" as "submit" vs "blog post".
520
+ *
521
+ * @param {object} toTranslate - Key-value map to translate
522
+ * @param {object} [descriptions] - Optional key→description map from Docusaurus
523
+ * @returns {string} User message
524
+ */
525
+ function buildUserMessage(toTranslate, descriptions) {
526
+ const typeHints = inferKeyTypes(toTranslate);
527
+
528
+ // Merge auto-inferred type hints with Docusaurus descriptions.
529
+ // Description context is appended after the type hint for richer context.
530
+ if (descriptions && typeof descriptions === 'object') {
531
+ for (const [key, desc] of Object.entries(descriptions)) {
532
+ if (key in toTranslate && typeof desc === 'string' && desc.length > 0) {
533
+ // Check if we already have a type hint for this key
534
+ const existingIdx = typeHints.findIndex(h => h.startsWith(`- "${key}":`));
535
+ if (existingIdx >= 0) {
536
+ // Append description to existing type hint
537
+ typeHints[existingIdx] += ` — "${desc}"`;
538
+ } else {
539
+ // No type hint inferred — add description-only hint
540
+ typeHints.push(`- "${key}": ${desc}`);
541
+ }
542
+ }
543
+ }
544
+ }
545
+
546
+ const hintsBlock = typeHints.length > 0
547
+ ? `UI context for these keys:\n${typeHints.join('\n')}\n\n`
548
+ : '';
549
+
550
+ return `${hintsBlock}${JSON.stringify(toTranslate, null, 2)}`;
551
+ }
552
+
553
+ /**
554
+ * Build a combined prompt (legacy interface for backward compat).
555
+ *
556
+ * Used by tests and any code that calls buildPrompt() directly.
557
+ * New code should use buildSystemMessage() + buildUserMessage() separately.
558
+ */
559
+ function buildPrompt(toTranslate, langConfig) {
560
+ return `${buildSystemMessage(langConfig)}\n\n${buildUserMessage(toTranslate)}`;
561
+ }
562
+
563
+ /**
564
+ * Infer UI element types from key naming patterns.
565
+ */
566
+ const KEY_TYPE_PATTERNS = [
567
+ { pattern: /(?:^|\.)(?:.*(?:btn|button|cta|action|submit|cancel|confirm|dismiss))/i, type: 'button label — keep concise' },
568
+ { pattern: /(?:^|\.)(?:.*(?:title|heading|h[1-6]))/i, type: 'heading/title' },
569
+ { pattern: /(?:^|\.)(?:.*(?:description|desc|subtitle|summary|body|paragraph))/i, type: 'description text — natural length OK' },
570
+ { pattern: /(?:^|\.)(?:.*(?:error|warning|validation|alert))/i, type: 'error/status message — be clear and direct' },
571
+ { pattern: /(?:^|\.)(?:.*(?:placeholder|hint))/i, type: 'input placeholder — keep very short' },
572
+ { pattern: /(?:^|\.)(?:.*(?:label|field))/i, type: 'form label' },
573
+ { pattern: /(?:^|\.)(?:.*(?:tooltip|popover|help))/i, type: 'tooltip/help text' },
574
+ { pattern: /(?:^|\.)(?:.*(?:toast|notification|snackbar))/i, type: 'notification message' },
575
+ { pattern: /(?:^|\.)(?:.*(?:nav|menu|tab|breadcrumb|link))/i, type: 'navigation element — keep concise' },
576
+ { pattern: /(?:^|\.)(?:.*(?:modal|dialog))/i, type: 'dialog/modal text' },
577
+ ];
578
+
579
+ function inferKeyTypes(toTranslate) {
580
+ const hints = [];
581
+ for (const key of Object.keys(toTranslate)) {
582
+ for (const { pattern, type } of KEY_TYPE_PATTERNS) {
583
+ if (pattern.test(key)) {
584
+ hints.push(`- "${key}": ${type}`);
585
+ break;
586
+ }
587
+ }
588
+ }
589
+ return hints;
590
+ }
591
+
592
+ export { LLMMethod, buildPrompt, buildSystemMessage, buildUserMessage, isUnsafeKey, inferKeyTypes };
@@ -0,0 +1,76 @@
1
+ /**
2
+ * LocalMethod — OpenAI-compatible local / self-hosted endpoint.
3
+ *
4
+ * Mirrors the harness LocalProvider: defaults to Ollama
5
+ * (http://localhost:11434/v1), and works with vLLM, LM Studio, llama.cpp
6
+ * servers, and OpenAI-compatible gateways (Groq, Together) via
7
+ * LOCAL_API_BASE / OPENAI_API_BASE. Inherits all OpenAI-format request/response
8
+ * logic from OpenAIMethod — only the endpoint default, key handling, and cost
9
+ * differ.
10
+ *
11
+ * FAIL-HONEST COST: estimateCost() returns null — local/self-hosted token cost
12
+ * is genuinely unknown to the tool and must never be fabricated as $0.
13
+ */
14
+
15
+ import { OpenAIMethod } from './openai.js';
16
+ import { getEnvOrFileVar } from '../api-key.js';
17
+
18
+ class LocalMethod extends OpenAIMethod {
19
+ constructor(options = {}) {
20
+ super(options);
21
+ this.name = 'local';
22
+ }
23
+
24
+ _getProviderLabel() { return 'Local (OpenAI-compatible)'; }
25
+ _getApiKeyEnvVar() { return 'OPENAI_API_KEY'; }
26
+ _getDefaultModel() { return 'llama3.1'; }
27
+
28
+ // Endpoint precedence: options.baseUrl > LOCAL_API_BASE > OPENAI_API_BASE >
29
+ // Ollama default. (Matches the harness LocalProvider resolution order.)
30
+ _getApiBaseEnvVar() { return 'LOCAL_API_BASE'; }
31
+ _getDefaultApiBase() {
32
+ return getEnvOrFileVar('OPENAI_API_BASE') || 'http://localhost:11434/v1';
33
+ }
34
+
35
+ // Local servers ignore auth — supply a placeholder when no key is set so
36
+ // translate() doesn't skip the run for a "missing" key.
37
+ _resolveApiKey(options) {
38
+ return super._resolveApiKey(options) || 'not-needed';
39
+ }
40
+
41
+ checkReadiness(_context) {
42
+ return { ready: true }; // local endpoint; no API key required
43
+ }
44
+
45
+ // Cost is genuinely unknown for local/self-hosted models — never $0.
46
+ estimateCost(_keyCount, _pairConfig = {}) {
47
+ return {
48
+ estimatedCost: null,
49
+ currency: 'USD',
50
+ source: 'local-unknown',
51
+ note: 'Local/self-hosted model — token cost unknown (never priced as $0).',
52
+ };
53
+ }
54
+
55
+ getProvenance() {
56
+ return {
57
+ resources: [
58
+ { name: 'Local OpenAI-compatible endpoint', license: 'varies', type: 'local' },
59
+ ],
60
+ commercialReady: false,
61
+ flags: ['local-endpoint'],
62
+ };
63
+ }
64
+
65
+ getSetupHelp() {
66
+ return [
67
+ ' Local (OpenAI-compatible) endpoint:',
68
+ ' • Ollama: install from https://ollama.com, run `ollama serve`,',
69
+ ' then `ollama pull llama3.1` (default http://localhost:11434/v1).',
70
+ ' • Or point at vLLM / LM Studio / Groq via LOCAL_API_BASE or OPENAI_API_BASE,',
71
+ ' and set the model with --model (e.g. qwen2.5, mistral).',
72
+ ];
73
+ }
74
+ }
75
+
76
+ export { LocalMethod };