champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,670 @@
1
+ /**
2
+ * LLM-Coached Translation Method — grammar/dictionary-injected LLM prompting.
3
+ *
4
+ * This method sits between raw LLM translation and a full FST-gated pipeline.
5
+ * It injects developer-provided linguistic hints into the prompt before each
6
+ * translation batch, giving the LLM explicit guidance for languages where
7
+ * naive prompting produces frequent errors.
8
+ *
9
+ * TWO COACHING CHANNELS (both honored):
10
+ * a) Free-text coaching PROMPT — `coachingFile`/`coachingPrompt` on the pair
11
+ * config. This is what the harness export carries (config_exporter.py emits
12
+ * `coachingFile`), and it rides in the system message exactly as the plain
13
+ * `llm` method injects it. If a coaching file/prompt is CONFIGURED but
14
+ * cannot be loaded (missing/unreadable/empty), this method FAILS LOUD — it
15
+ * must never silently degrade a coached run to an uncoached one (that would
16
+ * publish a different system than the harness validated).
17
+ * b) Structured coaching DATA — .champollion/coaching/<locale>.json with
18
+ * grammar_rules, a dictionary, and style_notes. Per-batch dictionary
19
+ * matches are injected as REQUIRED TERMINOLOGY hints.
20
+ *
21
+ * PROVIDER-AGNOSTIC:
22
+ * The actual API call is dispatched on `pairConfig.provider` so a coached run
23
+ * validated against openai / anthropic / gemini / local in the harness is
24
+ * reproducible on the CLI — not just OpenRouter. The provider only swaps the
25
+ * HTTP transport; the coached system message is built once and shared.
26
+ *
27
+ * FALLBACK (uncoached) is reached ONLY when NO coaching is configured at all
28
+ * (no coachingFile/coachingPrompt AND no structured file) — the documented
29
+ * "coached method, no coaching yet" path, routed to the chosen provider.
30
+ *
31
+ * COACHING DATA FORMAT (.champollion/coaching/<locale>.json):
32
+ * {
33
+ * "grammar_rules": [
34
+ * "French adjectives agree in gender and number with the noun",
35
+ * "Use 'vous' for formal contexts, 'tu' for informal"
36
+ * ],
37
+ * "dictionary": {
38
+ * "dashboard": "tableau de bord",
39
+ * "deployment": "déploiement",
40
+ * "settings": "paramètres"
41
+ * },
42
+ * "style_notes": "Prefer active voice. Avoid anglicisms where a native French term exists."
43
+ * }
44
+ *
45
+ * WHY .champollion/ AND NOT localesDir/:
46
+ * Coaching data is a development tool artifact, not a deployable asset.
47
+ * Locale files in localesDir/ get bundled into the app. Coaching hints
48
+ * are tool configuration — they live in the project's .champollion/ directory,
49
+ * following the same convention as .husky/, .eslintrc/, etc.
50
+ *
51
+ * COST PROFILE: ~$0.02–0.04 per 1k keys (longer prompts from coaching context)
52
+ * QUALITY TIER: high
53
+ */
54
+
55
+ import path from 'node:path';
56
+ import fs from 'node:fs';
57
+ import { TranslationMethod } from './base.js';
58
+ import { callOpenRouterJSON } from './openrouter-client.js';
59
+ import { estimateOpenRouterCost } from './openrouter-pricing.js';
60
+
61
+ // Re-use the LLM method's infrastructure (prompt building, key validation, cascade)
62
+ import { LLMMethod, inferKeyTypes, isUnsafeKey, buildSystemMessage } from './llm.js';
63
+ import { DEFAULT_OPENROUTER_MODEL, DEFAULT_BATCH_SIZE, DEFAULT_COACHED_TEMPERATURE, DEFAULT_MAX_RETRIES, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
64
+ import { pMap } from '../concurrent.js';
65
+ import { output } from '../output.js';
66
+
67
+ /**
68
+ * Default coaching data directory, relative to project root.
69
+ * Users can override via config: coaching.dir
70
+ */
71
+ const DEFAULT_COACHING_DIR = '.champollion/coaching';
72
+
73
+ /**
74
+ * Direct-provider methods a coached run can dispatch to (in addition to the
75
+ * default OpenRouter path). Imported lazily inside _getProviderMethod() because
76
+ * direct-llm.js already imports from THIS module — a static import would form a
77
+ * cycle. Mirrors the provider set the harness export emits (config_exporter.py).
78
+ */
79
+ const DIRECT_PROVIDER_MODULES = {
80
+ openai: { module: './openai.js', export: 'OpenAIMethod' },
81
+ anthropic: { module: './anthropic.js', export: 'AnthropicMethod' },
82
+ gemini: { module: './gemini.js', export: 'GeminiMethod' },
83
+ local: { module: './local.js', export: 'LocalMethod' },
84
+ };
85
+
86
+ /**
87
+ * Normalize a pair config's `provider` to a lowercase key. Absent/blank → the
88
+ * default OpenRouter path, matching config_exporter.py (which only emits
89
+ * `provider` when it is non-default).
90
+ *
91
+ * @param {string|null|undefined} provider
92
+ * @returns {string}
93
+ */
94
+ function normalizeProvider(provider) {
95
+ const p = (provider == null ? '' : String(provider)).trim().toLowerCase();
96
+ return p || 'openrouter';
97
+ }
98
+
99
+ /**
100
+ * Resolve the free-text coaching prompt for a coached run.
101
+ *
102
+ * Precedence:
103
+ * 1. pairConfig.coachingPrompt — already-resolved text (config.js and the
104
+ * harness export read the coaching file into this field).
105
+ * 2. pairConfig.coachingFile — read the file on demand (relative to cwd).
106
+ *
107
+ * FAIL-LOUD CONTRACT: when a coaching file/prompt is *configured* but cannot be
108
+ * turned into usable text (file missing, unreadable, or empty), this THROWS.
109
+ * The coached method must never silently degrade to an uncoached run when the
110
+ * operator asked for coaching — doing so would run a different system than the
111
+ * one validated in the harness. The plain "llm" method is the explicit choice
112
+ * for an intentionally-uncoached run.
113
+ *
114
+ * @param {object} pairConfig
115
+ * @param {object} options
116
+ * @returns {{ configured: boolean, text: string|null }}
117
+ */
118
+ function resolveCoachingPrompt(pairConfig, options) {
119
+ const hasFile = typeof pairConfig.coachingFile === 'string' && pairConfig.coachingFile.trim().length > 0;
120
+ const promptText = typeof pairConfig.coachingPrompt === 'string' ? pairConfig.coachingPrompt.trim() : '';
121
+
122
+ // Already-resolved coaching text wins.
123
+ if (promptText) {
124
+ return { configured: true, text: promptText };
125
+ }
126
+
127
+ // A coaching file is named but no text was resolved — read it now, and FAIL
128
+ // LOUD on any problem rather than running uncoached.
129
+ if (hasFile) {
130
+ const cwd = options.cwd || process.cwd();
131
+ const coachingPath = path.isAbsolute(pairConfig.coachingFile)
132
+ ? pairConfig.coachingFile
133
+ : path.resolve(cwd, pairConfig.coachingFile);
134
+
135
+ let raw;
136
+ try {
137
+ raw = fs.readFileSync(coachingPath, 'utf-8');
138
+ } catch (err) {
139
+ throw new Error(
140
+ `llm-coached: coachingFile "${pairConfig.coachingFile}" is configured but could not be read ` +
141
+ `(${err.code || err.message}; resolved to ${coachingPath}). ` +
142
+ `Refusing to run UNCOACHED — fix the path, or switch this pair to the plain "llm" method ` +
143
+ `if you intend an uncoached run.`
144
+ );
145
+ }
146
+
147
+ const text = raw.trim();
148
+ if (!text) {
149
+ throw new Error(
150
+ `llm-coached: coachingFile "${pairConfig.coachingFile}" (${coachingPath}) is empty. ` +
151
+ `Refusing to run UNCOACHED — add coaching guidance, or switch this pair to the plain "llm" method.`
152
+ );
153
+ }
154
+ return { configured: true, text };
155
+ }
156
+
157
+ // No free-text coaching configured.
158
+ return { configured: false, text: null };
159
+ }
160
+
161
+ /**
162
+ * Load coaching data for a locale from a JSON file, with caching.
163
+ *
164
+ * This is a standalone function so that any translation method (LLMCoached,
165
+ * OpenAI, Anthropic, Gemini) can load coaching data without instantiating
166
+ * the LLMCoachedMethod class.
167
+ *
168
+ * @param {string} coachingDir - Path to coaching data directory
169
+ * @param {string} locale - Target locale code (e.g., 'fr', 'crk')
170
+ * @param {Map} cache - Cache map to store loaded data (avoids re-reading files)
171
+ * @returns {object|null} Coaching data { grammar_rules, dictionary, style_notes }, or null
172
+ */
173
+ function loadCoachingData(coachingDir, locale, cache) {
174
+ if (!locale) return null;
175
+
176
+ // Check cache first
177
+ const cacheKey = `${coachingDir}:${locale}`;
178
+ if (cache.has(cacheKey)) {
179
+ return cache.get(cacheKey);
180
+ }
181
+
182
+ const filePath = path.join(coachingDir, `${locale}.json`);
183
+
184
+ if (!fs.existsSync(filePath)) {
185
+ cache.set(cacheKey, null);
186
+ return null;
187
+ }
188
+
189
+ try {
190
+ const raw = fs.readFileSync(filePath, 'utf-8');
191
+ const data = JSON.parse(raw);
192
+
193
+ // Validate required structure — normalize missing fields to safe defaults
194
+ const coaching = {
195
+ grammar_rules: Array.isArray(data.grammar_rules) ? data.grammar_rules : [],
196
+ dictionary: (data.dictionary && typeof data.dictionary === 'object') ? data.dictionary : {},
197
+ style_notes: typeof data.style_notes === 'string' ? data.style_notes : '',
198
+ };
199
+
200
+ cache.set(cacheKey, coaching);
201
+ return coaching;
202
+ } catch (err) {
203
+ output.warn(`Failed to load coaching data: ${filePath}`);
204
+ output.warn(err.message);
205
+ cache.set(cacheKey, null);
206
+ return null;
207
+ }
208
+ }
209
+
210
+ class LLMCoachedMethod extends TranslationMethod {
211
+ constructor(options = {}) {
212
+ super('llm-coached', options);
213
+ this._coachingCache = new Map();
214
+ }
215
+
216
+ /**
217
+ * Translate a batch of key-value pairs with coaching augmentation.
218
+ *
219
+ * Strategy:
220
+ * 1. Resolve the free-text coaching prompt (coachingFile/coachingPrompt).
221
+ * FAILS LOUD if configured but unloadable — never silently uncoached.
222
+ * 2. Load structured coaching data (.champollion/coaching/<locale>.json).
223
+ * 3. If NEITHER is configured, run the plain (uncoached) method for the
224
+ * chosen provider (documented fallback).
225
+ * 4. Otherwise build the coached system message once and dispatch the API
226
+ * call to the provider named by pairConfig.provider (openrouter default,
227
+ * or openai/anthropic/gemini/local).
228
+ *
229
+ * @param {string[]} keys - Flat dot-notation keys to translate
230
+ * @param {object} sourceFlat - Full flattened source locale
231
+ * @param {object} pairConfig - Pair config (method, provider, model, register, name, coachingFile, etc.)
232
+ * @param {object} options - { apiKey, batchSize, cwd }
233
+ * @returns {object|null} Map of key → translated value, or null
234
+ */
235
+ async translate(keys, sourceFlat, pairConfig, options) {
236
+ const provider = normalizeProvider(pairConfig.provider);
237
+
238
+ // ── Resolve coaching ───────────────────────────────────────────────
239
+ // 1. Free-text coaching prompt (coachingFile/coachingPrompt). FAILS LOUD
240
+ // if a coaching file/prompt is configured but cannot be loaded — the
241
+ // coached method must never silently fall through to an uncoached run.
242
+ const coachingPromptInfo = resolveCoachingPrompt(pairConfig, options); // throws on bad file
243
+
244
+ // 2. Structured coaching data (.champollion/coaching/<locale>.json).
245
+ const targetLocale = pairConfig.target || pairConfig.locale;
246
+ const cwd = options.cwd || process.cwd();
247
+ const coachingDir = options.coachingDir || path.join(cwd, DEFAULT_COACHING_DIR);
248
+ const structured = this._loadCoachingData(coachingDir, targetLocale);
249
+
250
+ const coachingConfigured = coachingPromptInfo.configured || Boolean(structured);
251
+
252
+ // No coaching configured at all → run the plain (uncoached) method for the
253
+ // chosen provider. This is the documented "coached method, no coaching yet"
254
+ // path — NOT the silent-drop bug (dropping coaching that WAS configured),
255
+ // which resolveCoachingPrompt() above now makes impossible.
256
+ if (!coachingConfigured) {
257
+ output.info(`No coaching configured for "${targetLocale}" (no coachingFile/coachingPrompt, no ${coachingDir}/${targetLocale}.json).`);
258
+ output.info('Running the standard (uncoached) LLM method. Add coaching for better results.');
259
+ // For OpenRouter, construct LLMMethod synchronously (no `await`) so the
260
+ // pipeline reaches fetch() within the same microtask — preserving the
261
+ // original fallback's timing. Direct providers need the lazy import.
262
+ const plain = provider === 'openrouter'
263
+ ? new LLMMethod()
264
+ : await this._getProviderMethod(provider);
265
+ return plain.translate(keys, sourceFlat, pairConfig, options);
266
+ }
267
+
268
+ // ── Build the coached system message ONCE ──────────────────────────
269
+ // The free-text coaching prompt rides in the system message via
270
+ // langConfig.coachingPrompt (exactly as the plain llm.js method injects
271
+ // it), so it is part of EVERY batch and can never be silently dropped —
272
+ // including for the openai/anthropic/gemini/local providers below.
273
+ const langConfig = {
274
+ name: pairConfig.name,
275
+ register: pairConfig.register,
276
+ genderGuidance: pairConfig.genderGuidance || null,
277
+ promptContext: pairConfig.promptContext || null,
278
+ coachingPrompt: coachingPromptInfo.text,
279
+ };
280
+ const systemMessage = structured
281
+ ? buildCoachedSystemMessage(langConfig, structured)
282
+ : buildSystemMessage(langConfig);
283
+
284
+ const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
285
+ const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
286
+
287
+ // ── Provider dispatch ──────────────────────────────────────────────
288
+ if (provider === 'openrouter') {
289
+ const { apiKey } = options;
290
+ if (!apiKey) {
291
+ output.warn('LLM-Coached translate: no API key provided — skipping batch.');
292
+ return null;
293
+ }
294
+ const model = pairConfig.model || options.model || DEFAULT_OPENROUTER_MODEL;
295
+ const llm = new LLMMethod();
296
+ const batchFn = (batch, opts) => this._callCoachedBatch(batch, structured, opts, pairConfig);
297
+ return this._runCoachedBatches(llm, keys, sourceFlat, langConfig, {
298
+ apiKey, model, maxRetries, systemMessage, batchSize,
299
+ }, batchFn);
300
+ }
301
+
302
+ // Direct provider (openai / anthropic / gemini / local). Reuses the direct
303
+ // method's transport + JSON parsing (_callProviderBatch) and shared cascade
304
+ // (_translateWithCascade), but with the coached system message WE built —
305
+ // so the coaching prompt is honored, not the default uncoached system.
306
+ const directMethod = await this._getProviderMethod(provider);
307
+ const apiKey = directMethod._resolveApiKey(options);
308
+ if (!apiKey) {
309
+ output.warn(`LLM-Coached translate (${provider}): no API key — set ${directMethod._getApiKeyEnvVar()}. Skipping batch.`);
310
+ return null;
311
+ }
312
+ const model = pairConfig.model || options.model || directMethod._getDefaultModel();
313
+ await directMethod._validateModel(model, apiKey); // DX warnings only; never blocks
314
+ const temperature = pairConfig.temperature ?? DEFAULT_COACHED_TEMPERATURE;
315
+ const batchFn = (batch, opts) => directMethod._callProviderBatch(batch, opts);
316
+ return this._runCoachedBatches(directMethod, keys, sourceFlat, langConfig, {
317
+ apiKey, model, maxRetries, systemMessage, batchSize,
318
+ coaching: structured, temperature, descriptions: options.descriptions || null,
319
+ }, batchFn);
320
+ }
321
+
322
+ /**
323
+ * Run the coached batch loop against a chosen method (OpenRouter LLMMethod or
324
+ * a direct provider). Shared by every provider branch so the parallel-batch +
325
+ * retry-cascade behavior is identical regardless of transport.
326
+ *
327
+ * @param {import('./llm.js').LLMMethod} method - Method whose _translateWithCascade + batchFn drive the calls
328
+ * @param {string[]} keys - Keys to translate
329
+ * @param {object} sourceFlat - Source values
330
+ * @param {object} langConfig - { name, register, coachingPrompt, ... }
331
+ * @param {object} baseOptions - { apiKey, model, maxRetries, systemMessage, batchSize, ... }
332
+ * @param {Function} batchFn - (toTranslate, options) => Promise<result>
333
+ * @returns {Promise<object|null>}
334
+ */
335
+ async _runCoachedBatches(method, keys, sourceFlat, langConfig, baseOptions, batchFn) {
336
+ const { batchSize } = baseOptions;
337
+ const allTranslated = {};
338
+
339
+ const batchChunks = [];
340
+ for (let i = 0; i < keys.length; i += batchSize) {
341
+ batchChunks.push(keys.slice(i, i + batchSize));
342
+ }
343
+
344
+ await pMap(batchChunks, async (chunk, idx) => {
345
+ const toTranslate = {};
346
+ for (const key of chunk) {
347
+ toTranslate[key] = sourceFlat[key];
348
+ }
349
+
350
+ const result = await method._translateWithCascade(
351
+ toTranslate,
352
+ langConfig,
353
+ { ...baseOptions, batchNum: idx + 1 },
354
+ batchFn,
355
+ 'Coached ',
356
+ );
357
+
358
+ if (result) {
359
+ Object.assign(allTranslated, result);
360
+ }
361
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
362
+
363
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
364
+ }
365
+
366
+ /**
367
+ * Resolve the underlying TranslationMethod for a provider name.
368
+ *
369
+ * 'openrouter' (default) → LLMMethod. Direct providers are imported lazily to
370
+ * avoid a static import cycle (direct-llm.js already imports from this file).
371
+ *
372
+ * @param {string} provider - Normalized provider name
373
+ * @returns {Promise<import('./base.js').TranslationMethod>}
374
+ */
375
+ async _getProviderMethod(provider) {
376
+ if (provider === 'openrouter') return new LLMMethod();
377
+ const spec = DIRECT_PROVIDER_MODULES[provider];
378
+ if (!spec) {
379
+ throw new Error(
380
+ `llm-coached: unknown provider "${provider}". ` +
381
+ `Supported providers: openrouter, ${Object.keys(DIRECT_PROVIDER_MODULES).join(', ')}.`
382
+ );
383
+ }
384
+ const mod = await import(spec.module);
385
+ return new mod[spec.export]();
386
+ }
387
+
388
+ /**
389
+ * Translate freeform content with coaching context.
390
+ *
391
+ * For content translation, we prepend coaching style notes and grammar
392
+ * rules to the existing prompt (which already contains the content).
393
+ */
394
+ async translateContent(prompt, pairConfig, options) {
395
+ const provider = normalizeProvider(pairConfig.provider);
396
+
397
+ // Same fail-loud coaching resolution as translate(): a configured coaching
398
+ // file/prompt that cannot be loaded throws rather than running uncoached.
399
+ const coachingPromptInfo = resolveCoachingPrompt(pairConfig, options); // throws on bad file
400
+
401
+ const targetLocale = pairConfig.target || pairConfig.locale;
402
+ const cwd = options.cwd || process.cwd();
403
+ const coachingDir = options.coachingDir || path.join(cwd, DEFAULT_COACHING_DIR);
404
+ const structured = this._loadCoachingData(coachingDir, targetLocale);
405
+
406
+ const coachingConfigured = coachingPromptInfo.configured || Boolean(structured);
407
+ const targetMethod = await this._getProviderMethod(provider);
408
+
409
+ if (!coachingConfigured) {
410
+ // Documented uncoached fallback — still routed to the chosen provider.
411
+ return targetMethod.translateContent(prompt, pairConfig, options);
412
+ }
413
+
414
+ // Build the coaching block (free-text prompt + structured grammar/style)
415
+ // and prepend it ourselves so coaching is applied for ANY provider.
416
+ const blocks = [];
417
+ if (coachingPromptInfo.text) {
418
+ blocks.push(`Coaching guidance:\n${coachingPromptInfo.text}`);
419
+ }
420
+ if (structured) {
421
+ const structuredBlock = buildContentCoachingBlock(structured);
422
+ if (structuredBlock) blocks.push(structuredBlock);
423
+ }
424
+ const coachingBlock = blocks.join('\n\n');
425
+ const augmentedPrompt = coachingBlock ? `${coachingBlock}\n\n${prompt}` : prompt;
426
+
427
+ // Strip the target locale so a direct provider's translateContent does NOT
428
+ // re-load and re-prepend the SAME structured coaching block (it keys
429
+ // coaching loading off pairConfig.target; LLMMethod ignores target).
430
+ const safePairConfig = { ...pairConfig, target: undefined, locale: undefined };
431
+ return targetMethod.translateContent(augmentedPrompt, safePairConfig, options);
432
+ }
433
+
434
+ /**
435
+ * Cost estimation — same as LLM but with coached:true flag
436
+ * for the 2.5x input token multiplier (grammar/dictionary injection).
437
+ *
438
+ * @param {number} keyCount - Number of keys to translate
439
+ * @param {object} [pairConfig] - Pair config containing the model ID
440
+ */
441
+ async estimateCost(keyCount, pairConfig = {}) {
442
+ const model = pairConfig.model || DEFAULT_OPENROUTER_MODEL;
443
+ return estimateOpenRouterCost(keyCount, model, { coached: true });
444
+ }
445
+
446
+ checkReadiness(context) {
447
+ if (!context.apiKey) {
448
+ return { ready: false, reason: 'No OpenRouter API key (OPENROUTER_API_KEY).' };
449
+ }
450
+ return { ready: true };
451
+ }
452
+
453
+ getQualityTier() {
454
+ return 'high';
455
+ }
456
+
457
+ getProvenance() {
458
+ return {
459
+ resources: [
460
+ { name: 'User-provided coaching data', license: 'project-local', type: 'dictionary/grammar' },
461
+ ],
462
+ commercialReady: true,
463
+ flags: [],
464
+ };
465
+ }
466
+
467
+ // -----------------------------------------------------------------
468
+ // Private helpers
469
+ // -----------------------------------------------------------------
470
+
471
+ /**
472
+ * Load coaching data for a locale, with caching.
473
+ * Thin wrapper around the standalone loadCoachingData() function.
474
+ *
475
+ * @param {string} coachingDir - Path to coaching data directory
476
+ * @param {string} locale - Target locale code
477
+ * @returns {object|null} Coaching data, or null if not found
478
+ */
479
+ _loadCoachingData(coachingDir, locale) {
480
+ return loadCoachingData(coachingDir, locale, this._coachingCache);
481
+ }
482
+
483
+ // NOTE: The coached cascade (_translateCoachedWithCascade) was removed.
484
+ // Coached translation now delegates to LLMMethod._translateWithCascade
485
+ // via composition, passing _callCoachedBatch as the batchFn parameter.
486
+ // This eliminates ~80 lines of duplicated cascade logic.
487
+
488
+ /**
489
+ * Make a single coached API call via the shared OpenRouter client.
490
+ * Builds per-batch user message with dictionary hints, uses shared system message.
491
+ */
492
+ async _callCoachedBatch(toTranslate, coaching, options, pairConfig = {}) {
493
+ const { apiKey, model, batchNum, systemMessage } = options;
494
+
495
+ // Build per-batch user message with dictionary hints specific to this batch's
496
+ // values. `coaching` may be null when only a free-text coaching prompt is
497
+ // configured (no structured .champollion/coaching/<locale>.json) — in that
498
+ // case there is no dictionary to match against.
499
+ const dictHints = (coaching && coaching.dictionary)
500
+ ? findDictionaryMatches(toTranslate, coaching.dictionary)
501
+ : [];
502
+ const typeHints = inferKeyTypes(toTranslate);
503
+
504
+ let userMessage = '';
505
+ if (dictHints.length > 0) {
506
+ userMessage += 'REQUIRED TERMINOLOGY (use these exact translations):\n';
507
+ userMessage += dictHints.map(h => ` • "${h.term}" → "${h.translation}"`).join('\n');
508
+ userMessage += '\n\n';
509
+ }
510
+ if (typeHints.length > 0) {
511
+ userMessage += `UI context for these keys:\n${typeHints.join('\n')}\n\n`;
512
+ }
513
+ userMessage += JSON.stringify(toTranslate, null, 2);
514
+
515
+ return callOpenRouterJSON({
516
+ prompt: userMessage,
517
+ systemMessage,
518
+ apiKey,
519
+ model,
520
+ temperature: pairConfig.temperature ?? DEFAULT_COACHED_TEMPERATURE, // Lower than standard for coached (more deterministic)
521
+ label: `Coached batch ${batchNum}`,
522
+ xTitle: 'champollion (coached)',
523
+ expectedKeys: new Set(Object.keys(toTranslate)),
524
+ isUnsafeKey,
525
+ });
526
+ }
527
+
528
+ }
529
+
530
+ // -----------------------------------------------------------------
531
+ // Coached prompt building
532
+ // -----------------------------------------------------------------
533
+
534
+ /**
535
+ * Build the system message for coached translation (cached across batches).
536
+ *
537
+ * Contains: base translation rules + coaching context (grammar, style).
538
+ * Dictionary hints are NOT included here because they vary per batch
539
+ * (only terms present in that batch's values are injected).
540
+ *
541
+ * @param {object} langConfig - { name, register }
542
+ * @param {object} coaching - { grammar_rules, dictionary, style_notes }
543
+ * @returns {string} System message for prompt caching
544
+ */
545
+ function buildCoachedSystemMessage(langConfig, coaching) {
546
+ // Start with the base system message (register + rules)
547
+ let system = buildSystemMessage(langConfig);
548
+
549
+ // Append coaching context
550
+ const coachingParts = [];
551
+
552
+ if (coaching.grammar_rules.length > 0) {
553
+ coachingParts.push(
554
+ 'GRAMMAR RULES (follow strictly):',
555
+ ...coaching.grammar_rules.map(r => ` • ${r}`)
556
+ );
557
+ }
558
+
559
+ if (coaching.style_notes) {
560
+ coachingParts.push(
561
+ '',
562
+ `STYLE GUIDE: ${coaching.style_notes}`
563
+ );
564
+ }
565
+
566
+ if (coachingParts.length > 0) {
567
+ system += `\n\n--- COACHING CONTEXT ---\n${coachingParts.join('\n')}\n--- END COACHING ---`;
568
+ }
569
+
570
+ return system;
571
+ }
572
+
573
+ /**
574
+ * Build a combined coached prompt (legacy interface for backward compat).
575
+ *
576
+ * Used by tests that call buildCoachedPrompt() directly.
577
+ * New code should use buildCoachedSystemMessage() + per-batch user message.
578
+ *
579
+ * @param {Object<string, string>} toTranslate - Key-value map to translate
580
+ * @param {{ name: string, register: string }} langConfig - Target language info
581
+ * @param {import('../types.js').CoachingData} coaching - Coaching data
582
+ * @returns {string} Combined system + user prompt
583
+ */
584
+ function buildCoachedPrompt(toTranslate, langConfig, coaching) {
585
+ const system = buildCoachedSystemMessage(langConfig, coaching);
586
+
587
+ // Build user message with dictionary hints + UI context + JSON payload
588
+ const dictHints = findDictionaryMatches(toTranslate, coaching.dictionary);
589
+ const typeHints = inferKeyTypes(toTranslate);
590
+
591
+ let userMessage = '';
592
+ if (dictHints.length > 0) {
593
+ userMessage += 'REQUIRED TERMINOLOGY (use these exact translations):\n';
594
+ userMessage += dictHints.map(h => ` • "${h.term}" → "${h.translation}"`).join('\n');
595
+ userMessage += '\n\n';
596
+ }
597
+ if (typeHints.length > 0) {
598
+ userMessage += `UI context for these keys:\n${typeHints.join('\n')}\n\n`;
599
+ }
600
+ userMessage += JSON.stringify(toTranslate, null, 2);
601
+
602
+ return `${system}\n\n${userMessage}`;
603
+ }
604
+
605
+ /**
606
+ * Build a coaching context block for freeform content translation.
607
+ *
608
+ * Lighter than the key-value version — only grammar and style, no dictionary
609
+ * matching (content is too freeform for term-level matching).
610
+ *
611
+ * @param {import('../types.js').CoachingData} coaching - Coaching data
612
+ * @returns {string} Coaching context block, or empty string if no data
613
+ */
614
+ function buildContentCoachingBlock(coaching) {
615
+ const parts = [];
616
+
617
+ if (coaching.grammar_rules.length > 0) {
618
+ parts.push(
619
+ 'IMPORTANT — Follow these grammar rules:',
620
+ ...coaching.grammar_rules.map(r => ` • ${r}`)
621
+ );
622
+ }
623
+
624
+ if (coaching.style_notes) {
625
+ parts.push('', `STYLE GUIDE: ${coaching.style_notes}`);
626
+ }
627
+
628
+ return parts.length > 0 ? parts.join('\n') : '';
629
+ }
630
+
631
+ /**
632
+ * Scan source values for dictionary term matches.
633
+ *
634
+ * Uses case-insensitive word-boundary matching to find terms from the
635
+ * coaching dictionary that appear in the current batch's source values.
636
+ *
637
+ * @param {object} toTranslate - Key-value map to scan
638
+ * @param {object} dictionary - Term → translation map
639
+ * @returns {Array<{ term: string, translation: string }>} Matched hints
640
+ */
641
+ function findDictionaryMatches(toTranslate, dictionary) {
642
+ if (!dictionary || Object.keys(dictionary).length === 0) return [];
643
+
644
+ const matches = [];
645
+ const seen = new Set();
646
+ const values = Object.values(toTranslate).join(' ').toLowerCase();
647
+
648
+ for (const [term, translation] of Object.entries(dictionary)) {
649
+ if (seen.has(term)) continue;
650
+
651
+ // Case-insensitive word-boundary check
652
+ // Use a simple indexOf for performance — the dictionary is usually small
653
+ if (values.includes(term.toLowerCase())) {
654
+ matches.push({ term, translation });
655
+ seen.add(term);
656
+ }
657
+ }
658
+
659
+ return matches;
660
+ }
661
+
662
+ export {
663
+ LLMCoachedMethod,
664
+ loadCoachingData,
665
+ buildCoachedPrompt,
666
+ buildCoachedSystemMessage,
667
+ buildContentCoachingBlock,
668
+ findDictionaryMatches,
669
+ DEFAULT_COACHING_DIR,
670
+ };