champollion 0.3.3 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. package/README.md +52 -37
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +51 -3
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +289 -88
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +649 -130
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +16 -10
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +197 -38
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +193 -106
  100. package/lib/seal.mjs +6 -5
  101. package/lib/sealed-qualifier.mjs +2 -2
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +3 -2
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/DATA-SOVEREIGNTY.md +19 -20
  123. package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
  124. package/shared/cards-fallback.json +1 -1
  125. package/shared/catalogue/card-config.json +1 -1
  126. package/shared/curated-orthography-conventions.json +26 -8
  127. package/shared/docent/faq.en.json +14 -16
  128. package/shared/docent/system-prompt.md +17 -19
  129. package/shared/explainers/tc-features.json +15 -15
  130. package/shared/gettext-plural-forms.json +45 -0
  131. package/shared/human-services.json +1 -1
  132. package/shared/method-registry.json +2 -0
  133. package/shared/metric-registry.json +96 -18
  134. package/shared/schemas/champollion-plugin.schema.json +4 -0
  135. package/shared/schemas/corpora-card.schema.json +20 -10
  136. package/shared/schemas/human-services.schema.json +2 -2
  137. package/shared/schemas/language-card.schema.json +1 -1
  138. package/shared/schemas/method-card.schema.json +1 -1
  139. package/shared/schemas/method-index-record.schema.json +67 -0
  140. package/shared/schemas/method-registry.schema.json +4 -0
  141. package/shared/schemas/metric-registry.schema.json +55 -1
  142. package/shared/docent/corpus.json +0 -11333
@@ -56,20 +56,18 @@ import path from 'node:path';
56
56
  import fs from 'node:fs';
57
57
  import { TranslationMethod } from './base.js';
58
58
  import { callOpenRouterJSON } from './openrouter-client.js';
59
+ import { isCapturing, PREVIEW_KEY } from './request-capture.js';
59
60
  import { estimateOpenRouterCost } from './openrouter-pricing.js';
60
61
 
61
62
  // Re-use the LLM method's infrastructure (prompt building, key validation, cascade)
62
- import { LLMMethod, inferKeyTypes, isUnsafeKey, buildSystemMessage } from './llm.js';
63
+ import { LLMMethod, inferKeyTypes, isUnsafeKey, buildSystemMessage, buildUserMessage, promptSettingsFor } from './llm.js';
64
+ // The coaching file + glossary helpers live in coaching-data.js (llm.js uses
65
+ // them too); re-exported below under their old names.
66
+ import { DEFAULT_COACHING_DIR, loadCoachingData, findDictionaryMatches, projectGlossary } from './coaching-data.js';
63
67
  import { DEFAULT_OPENROUTER_MODEL, DEFAULT_BATCH_SIZE, DEFAULT_COACHED_TEMPERATURE, DEFAULT_MAX_RETRIES, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
64
68
  import { pMap } from '../concurrent.js';
65
69
  import { output } from '../output.js';
66
70
 
67
- /**
68
- * Default coaching data directory, relative to project root.
69
- * Users can override via config: coaching.dir
70
- */
71
- const DEFAULT_COACHING_DIR = '.champollion/coaching';
72
-
73
71
  /**
74
72
  * Direct-provider methods a coached run can dispatch to (in addition to the
75
73
  * default OpenRouter path). Imported lazily inside _getProviderMethod() because
@@ -83,6 +81,13 @@ const DIRECT_PROVIDER_MODULES = {
83
81
  local: { module: './local.js', export: 'LocalMethod' },
84
82
  };
85
83
 
84
+ /**
85
+ * Every transport a coached (or plain llm) pair can name in `provider`.
86
+ * pairs.js validates config against this list, so an unknown provider fails at
87
+ * pair-graph build time instead of silently riding OpenRouter.
88
+ */
89
+ const COACHED_PROVIDERS = ['openrouter', ...Object.keys(DIRECT_PROVIDER_MODULES)];
90
+
86
91
  /**
87
92
  * Normalize a pair config's `provider` to a lowercase key. Absent/blank → the
88
93
  * default OpenRouter path, matching config_exporter.py (which only emits
@@ -158,59 +163,16 @@ function resolveCoachingPrompt(pairConfig, options) {
158
163
  return { configured: false, text: null };
159
164
  }
160
165
 
161
- /**
162
- * Load coaching data for a locale from a JSON file, with caching.
163
- *
164
- * This is a standalone function so that any translation method (LLMCoached,
165
- * OpenAI, Anthropic, Gemini) can load coaching data without instantiating
166
- * the LLMCoachedMethod class.
167
- *
168
- * @param {string} coachingDir - Path to coaching data directory
169
- * @param {string} locale - Target locale code (e.g., 'fr', 'crk')
170
- * @param {Map} cache - Cache map to store loaded data (avoids re-reading files)
171
- * @returns {object|null} Coaching data { grammar_rules, dictionary, style_notes }, or null
172
- */
173
- function loadCoachingData(coachingDir, locale, cache) {
174
- if (!locale) return null;
175
-
176
- // Check cache first
177
- const cacheKey = `${coachingDir}:${locale}`;
178
- if (cache.has(cacheKey)) {
179
- return cache.get(cacheKey);
180
- }
181
-
182
- const filePath = path.join(coachingDir, `${locale}.json`);
183
-
184
- if (!fs.existsSync(filePath)) {
185
- cache.set(cacheKey, null);
186
- return null;
187
- }
188
-
189
- try {
190
- const raw = fs.readFileSync(filePath, 'utf-8');
191
- const data = JSON.parse(raw);
192
-
193
- // Validate required structure — normalize missing fields to safe defaults
194
- const coaching = {
195
- grammar_rules: Array.isArray(data.grammar_rules) ? data.grammar_rules : [],
196
- dictionary: (data.dictionary && typeof data.dictionary === 'object') ? data.dictionary : {},
197
- style_notes: typeof data.style_notes === 'string' ? data.style_notes : '',
198
- };
199
-
200
- cache.set(cacheKey, coaching);
201
- return coaching;
202
- } catch (err) {
203
- output.warn(`Failed to load coaching data: ${filePath}`);
204
- output.warn(err.message);
205
- cache.set(cacheKey, null);
206
- return null;
207
- }
208
- }
209
-
210
166
  class LLMCoachedMethod extends TranslationMethod {
211
167
  constructor(options = {}) {
212
168
  super('llm-coached', options);
169
+ this.acceptsKeyInstructions = true; // prompt "UI context" lines (lib/methods/base.js)
170
+ this.supportsRequestPreview = true; // its transport reports to request-capture.js
213
171
  this._coachingCache = new Map();
172
+ // The pair's transport, known at construction (getMethod passes the pair
173
+ // config) so preflight and cost estimation — which receive no pairConfig
174
+ // of their own — check the provider the run will actually call.
175
+ this._provider = normalizeProvider(options.provider);
214
176
  }
215
177
 
216
178
  /**
@@ -270,32 +232,35 @@ class LLMCoachedMethod extends TranslationMethod {
270
232
  // langConfig.coachingPrompt (exactly as the plain llm.js method injects
271
233
  // it), so it is part of EVERY batch and can never be silently dropped —
272
234
  // including for the openai/anthropic/gemini/local providers below.
273
- const langConfig = {
274
- name: pairConfig.name,
275
- register: pairConfig.register,
276
- genderGuidance: pairConfig.genderGuidance || null,
277
- promptContext: pairConfig.promptContext || null,
278
- coachingPrompt: coachingPromptInfo.text,
279
- };
235
+ // The same reading of the pair as every LLM method (llm.js
236
+ // promptSettingsFor) — it used to leave out the protected terms — with
237
+ // the coaching prompt this method resolved itself.
238
+ const langConfig = promptSettingsFor(pairConfig, { coachingPrompt: coachingPromptInfo.text });
280
239
  const systemMessage = structured
281
240
  ? buildCoachedSystemMessage(langConfig, structured)
282
241
  : buildSystemMessage(langConfig);
242
+ // The glossary each batch is told about: the coaching file's dictionary,
243
+ // else the project's (the same file, unless a plugin brings its own).
244
+ const glossary = (structured && Object.keys(structured.dictionary).length > 0)
245
+ ? structured.dictionary
246
+ : projectGlossary(pairConfig, options, this._coachingCache);
283
247
 
284
248
  const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
285
249
  const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
286
250
 
287
251
  // ── Provider dispatch ──────────────────────────────────────────────
288
252
  if (provider === 'openrouter') {
289
- const { apiKey } = options;
253
+ // A request preview needs no key: it is shown, never sent.
254
+ const apiKey = options.apiKey || (isCapturing() ? PREVIEW_KEY : null);
290
255
  if (!apiKey) {
291
256
  output.warn('LLM-Coached translate: no API key provided — skipping batch.');
292
257
  return null;
293
258
  }
294
259
  const model = pairConfig.model || options.model || DEFAULT_OPENROUTER_MODEL;
295
260
  const llm = new LLMMethod();
296
- const batchFn = (batch, opts) => this._callCoachedBatch(batch, structured, opts, pairConfig);
261
+ const batchFn = (batch, opts) => this._callCoachedBatch(batch, glossary, opts, pairConfig);
297
262
  return this._runCoachedBatches(llm, keys, sourceFlat, langConfig, {
298
- apiKey, model, maxRetries, systemMessage, batchSize,
263
+ apiKey, model, maxRetries, systemMessage, batchSize, descriptions: options.descriptions || null,
299
264
  }, batchFn);
300
265
  }
301
266
 
@@ -304,18 +269,20 @@ class LLMCoachedMethod extends TranslationMethod {
304
269
  // (_translateWithCascade), but with the coached system message WE built —
305
270
  // so the coaching prompt is honored, not the default uncoached system.
306
271
  const directMethod = await this._getProviderMethod(provider);
307
- const apiKey = directMethod._resolveApiKey(options);
272
+ const apiKey = directMethod._resolveApiKey(options) || (isCapturing() ? PREVIEW_KEY : null);
308
273
  if (!apiKey) {
309
274
  output.warn(`LLM-Coached translate (${provider}): no API key — set ${directMethod._getApiKeyEnvVar()}. Skipping batch.`);
310
275
  return null;
311
276
  }
312
- const model = pairConfig.model || options.model || directMethod._getDefaultModel();
313
- await directMethod._validateModel(model, apiKey); // DX warnings only; never blocks
277
+ // The provider's own name for the model; an OpenRouter slug is mapped or
278
+ // refused (direct-llm.js resolveModelId), never sent to it as is.
279
+ const model = directMethod.resolveModelId(pairConfig.model || options.model || directMethod._getDefaultModel(), { cwd });
280
+ if (!isCapturing()) await directMethod._validateModel(model, apiKey, cwd); // DX warnings only; never blocks
314
281
  const temperature = pairConfig.temperature ?? DEFAULT_COACHED_TEMPERATURE;
315
282
  const batchFn = (batch, opts) => directMethod._callProviderBatch(batch, opts);
316
283
  return this._runCoachedBatches(directMethod, keys, sourceFlat, langConfig, {
317
284
  apiKey, model, maxRetries, systemMessage, batchSize,
318
- coaching: structured, temperature, descriptions: options.descriptions || null,
285
+ glossary, temperature, descriptions: options.descriptions || null, cwd,
319
286
  }, batchFn);
320
287
  }
321
288
 
@@ -424,26 +391,41 @@ class LLMCoachedMethod extends TranslationMethod {
424
391
  const coachingBlock = blocks.join('\n\n');
425
392
  const augmentedPrompt = coachingBlock ? `${coachingBlock}\n\n${prompt}` : prompt;
426
393
 
427
- // Strip the target locale so a direct provider's translateContent does NOT
428
- // re-load and re-prepend the SAME structured coaching block (it keys
429
- // coaching loading off pairConfig.target; LLMMethod ignores target).
430
- const safePairConfig = { ...pairConfig, target: undefined, locale: undefined };
431
- return targetMethod.translateContent(augmentedPrompt, safePairConfig, options);
394
+ // The plain methods send a content prompt as they get it, so the
395
+ // coaching is prepended once, here, whichever provider carries it.
396
+ return targetMethod.translateContent(augmentedPrompt, pairConfig, options);
432
397
  }
433
398
 
434
399
  /**
435
400
  * Cost estimation — same as LLM but with coached:true flag
436
401
  * for the 2.5x input token multiplier (grammar/dictionary injection).
402
+ * A direct provider prices through its own method (and `local` honestly
403
+ * returns null) — never through OpenRouter's rate for a model it isn't
404
+ * sending to OpenRouter.
437
405
  *
438
406
  * @param {number} keyCount - Number of keys to translate
439
407
  * @param {object} [pairConfig] - Pair config containing the model ID
440
408
  */
441
- async estimateCost(keyCount, pairConfig = {}) {
409
+ async estimateCost(keyCount, pairConfig = {}, context = {}) {
410
+ const provider = normalizeProvider(pairConfig.provider ?? this._provider);
411
+ if (provider !== 'openrouter') {
412
+ const directMethod = await this._getProviderMethod(provider);
413
+ return directMethod.estimateCost(keyCount, pairConfig, context);
414
+ }
442
415
  const model = pairConfig.model || DEFAULT_OPENROUTER_MODEL;
443
416
  return estimateOpenRouterCost(keyCount, model, { coached: true });
444
417
  }
445
418
 
446
- checkReadiness(context) {
419
+ /**
420
+ * Preflight: the credential the run needs is the chosen provider's, not
421
+ * OpenRouter's. A coached openai pair must pass with only OPENAI_API_KEY set,
422
+ * and must fail (naming OPENAI_API_KEY) without it.
423
+ */
424
+ async checkReadiness(context) {
425
+ if (this._provider !== 'openrouter') {
426
+ const directMethod = await this._getProviderMethod(this._provider);
427
+ return directMethod.checkReadiness(context);
428
+ }
447
429
  if (!context.apiKey) {
448
430
  return { ready: false, reason: 'No OpenRouter API key (OPENROUTER_API_KEY).' };
449
431
  }
@@ -489,28 +471,15 @@ class LLMCoachedMethod extends TranslationMethod {
489
471
  * Make a single coached API call via the shared OpenRouter client.
490
472
  * Builds per-batch user message with dictionary hints, uses shared system message.
491
473
  */
492
- async _callCoachedBatch(toTranslate, coaching, options, pairConfig = {}) {
474
+ async _callCoachedBatch(toTranslate, glossary, options, pairConfig = {}) {
493
475
  const { apiKey, model, batchNum, systemMessage } = options;
494
476
 
495
- // Build per-batch user message with dictionary hints specific to this batch's
496
- // values. `coaching` may be null when only a free-text coaching prompt is
497
- // configured (no structured .champollion/coaching/<locale>.json) — in that
498
- // case there is no dictionary to match against.
499
- const dictHints = (coaching && coaching.dictionary)
500
- ? findDictionaryMatches(toTranslate, coaching.dictionary)
501
- : [];
502
- const typeHints = inferKeyTypes(toTranslate);
503
-
504
- let userMessage = '';
505
- if (dictHints.length > 0) {
506
- userMessage += 'REQUIRED TERMINOLOGY (use these exact translations):\n';
507
- userMessage += dictHints.map(h => ` • "${h.term}" → "${h.translation}"`).join('\n');
508
- userMessage += '\n\n';
509
- }
510
- if (typeHints.length > 0) {
511
- userMessage += `UI context for these keys:\n${typeHints.join('\n')}\n\n`;
512
- }
513
- userMessage += JSON.stringify(toTranslate, null, 2);
477
+ // The user message every LLM method sends (llm.js buildUserMessage): the
478
+ // glossary terms this batch contains (none when only a free-text
479
+ // coaching prompt is configured and the project has no glossary), the
480
+ // per-key instructions (plural forms, ICU categories, gettext context,
481
+ // gate feedback), then the batch.
482
+ const userMessage = buildUserMessage(toTranslate, options.descriptions || undefined, glossary || null);
514
483
 
515
484
  return callOpenRouterJSON({
516
485
  prompt: userMessage,
@@ -628,39 +597,10 @@ function buildContentCoachingBlock(coaching) {
628
597
  return parts.length > 0 ? parts.join('\n') : '';
629
598
  }
630
599
 
631
- /**
632
- * Scan source values for dictionary term matches.
633
- *
634
- * Uses case-insensitive word-boundary matching to find terms from the
635
- * coaching dictionary that appear in the current batch's source values.
636
- *
637
- * @param {object} toTranslate - Key-value map to scan
638
- * @param {object} dictionary - Term → translation map
639
- * @returns {Array<{ term: string, translation: string }>} Matched hints
640
- */
641
- function findDictionaryMatches(toTranslate, dictionary) {
642
- if (!dictionary || Object.keys(dictionary).length === 0) return [];
643
-
644
- const matches = [];
645
- const seen = new Set();
646
- const values = Object.values(toTranslate).join(' ').toLowerCase();
647
-
648
- for (const [term, translation] of Object.entries(dictionary)) {
649
- if (seen.has(term)) continue;
650
-
651
- // Case-insensitive word-boundary check
652
- // Use a simple indexOf for performance — the dictionary is usually small
653
- if (values.includes(term.toLowerCase())) {
654
- matches.push({ term, translation });
655
- seen.add(term);
656
- }
657
- }
658
-
659
- return matches;
660
- }
661
-
662
600
  export {
663
601
  LLMCoachedMethod,
602
+ COACHED_PROVIDERS,
603
+ normalizeProvider,
664
604
  loadCoachingData,
665
605
  buildCoachedPrompt,
666
606
  buildCoachedSystemMessage,
@@ -44,7 +44,13 @@ import { isUnsafeKey } from '../security.js';
44
44
  import { pMap } from '../concurrent.js';
45
45
  import { output } from '../output.js';
46
46
  import { getEnvOrFileVar } from '../api-key.js';
47
-
47
+ import { nameRules } from '../name-rules.js';
48
+ import { isCapturing, PREVIEW_KEY } from './request-capture.js';
49
+ import { projectGlossary, terminologyBlock } from './coaching-data.js';
50
+ // The plain LLM methods: one prompt (promptSettingsFor + buildSystemMessage
51
+ // + buildUserMessage), different transports. llm-coached adds its coaching.
52
+ // Defined beside the set the cache key reads (lib/tm.js tmMethodKey).
53
+ import { PLAIN_LLM_METHODS } from './prompt-methods.js';
48
54
 
49
55
  /** Default per-cascade cap on individual-key fallback calls (see below). */
50
56
  const DEFAULT_MAX_KEY_FANOUT = 16;
@@ -72,6 +78,10 @@ function maxKeyFanout() {
72
78
  class LLMMethod extends TranslationMethod {
73
79
  constructor(options = {}) {
74
80
  super('llm', options);
81
+ this.acceptsKeyInstructions = true; // prompt "UI context" lines (lib/methods/base.js)
82
+ this.supportsRequestPreview = true; // its transport reports to request-capture.js
83
+ // Coaching-file cache (the project glossary, lib/methods/coaching-data.js)
84
+ this._coachingCache = new Map();
75
85
  }
76
86
 
77
87
  /**
@@ -84,7 +94,8 @@ class LLMMethod extends TranslationMethod {
84
94
  * @returns {object|null} Map of key → translated value, or null if all failed
85
95
  */
86
96
  async translate(keys, sourceFlat, pairConfig, options) {
87
- const { apiKey } = options;
97
+ // A request preview needs no key: it is shown, never sent.
98
+ const apiKey = options.apiKey || (isCapturing() ? PREVIEW_KEY : null);
88
99
  const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
89
100
  const model = pairConfig.model || options.model || DEFAULT_OPENROUTER_MODEL;
90
101
  const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
@@ -93,24 +104,16 @@ class LLMMethod extends TranslationMethod {
93
104
  return null;
94
105
  }
95
106
 
96
- const langConfig = {
97
- name: pairConfig.name,
98
- register: pairConfig.register,
99
- // Language-specific gender guidance from the card (e.g., écriture inclusive
100
- // for French, Doppelpunkt notation for German). Falls back to a generic
101
- // rule in buildSystemMessage when null.
102
- genderGuidance: pairConfig.genderGuidance || null,
103
- // User-provided global context (e.g., "This is a developer tool README").
104
- // Injected into the system message to give the LLM domain awareness.
105
- promptContext: pairConfig.promptContext || null,
106
- // Coaching prompt text — free-text instructions read from a coaching file.
107
- // Injected between register and rules in the system message.
108
- coachingPrompt: pairConfig.coachingPrompt || null,
109
- };
107
+ // What the pair tells its model — the same reading every LLM method
108
+ // uses (promptSettingsFor), so openai/anthropic/gemini/local send this
109
+ // exact system message too.
110
+ const langConfig = promptSettingsFor(pairConfig);
110
111
 
111
112
  // Build the system message once — identical across all batches for this locale.
112
113
  // This enables provider-level prompt caching (Anthropic, Gemini).
113
114
  const systemMessage = buildSystemMessage(langConfig);
115
+ // The project glossary: each batch is told the terms it contains.
116
+ const glossary = projectGlossary(pairConfig, options, this._coachingCache);
114
117
 
115
118
  const allTranslated = {};
116
119
 
@@ -140,6 +143,7 @@ class LLMMethod extends TranslationMethod {
140
143
  systemMessage,
141
144
  temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
142
145
  descriptions: options.descriptions || null,
146
+ glossary,
143
147
  }, this._callOpenRouterBatch.bind(this));
144
148
 
145
149
  if (result) {
@@ -418,7 +422,7 @@ class LLMMethod extends TranslationMethod {
418
422
  */
419
423
  async _callOpenRouterBatch(toTranslate, options) {
420
424
  const { apiKey, model, batchNum, systemMessage, temperature } = options;
421
- const userMessage = buildUserMessage(toTranslate, options.descriptions || undefined);
425
+ const userMessage = buildUserMessage(toTranslate, options.descriptions || undefined, options.glossary || null);
422
426
 
423
427
  return callOpenRouterJSON({
424
428
  prompt: userMessage,
@@ -445,7 +449,7 @@ class LLMMethod extends TranslationMethod {
445
449
  output.error(`${keys.length} key(s) could not be translated.`);
446
450
  if (!label) {
447
451
  output.error(`Consider increasing maxRetries or using a more capable model:`);
448
- output.error(`"languages": { "<code>": { "model": "google/gemini-2.5-pro", "batchSize": 5, "maxRetries": 5 } }`);
452
+ output.error(`"languages": { "<code>": { "model": "google/gemini-3.1-pro-preview", "batchSize": 5, "maxRetries": 5 } }`);
449
453
  }
450
454
  }
451
455
  }
@@ -454,6 +458,46 @@ class LLMMethod extends TranslationMethod {
454
458
  // Prompt building — split into system (cached) + user (per-batch)
455
459
  // -----------------------------------------------------------------
456
460
 
461
+ /**
462
+ * What a pair tells its model, read from the pair config — the ONE reading
463
+ * every LLM-backed method uses (llm, openai, anthropic, gemini, local and
464
+ * llm-coached on any provider). Only the transport differs between them.
465
+ *
466
+ * WHY: the direct-provider methods used to build their own, shorter
467
+ * settings ({ name, register }), so a project's protected terms, its prompt
468
+ * context and the language's gender guidance never reached a model run
469
+ * through `local`, `openai`, `anthropic` or `gemini` — a brand name could
470
+ * be translated there and not through `llm` (found 2026-10-03).
471
+ *
472
+ * @param {object} pairConfig - Resolved pair config (lib/pairs.js)
473
+ * @param {object} [overrides] - Fields a method resolves itself (llm-coached:
474
+ * the coaching prompt it read from coachingFile)
475
+ * @returns {{ name: string, register: string, genderGuidance: string|null,
476
+ * promptContext: string|null, protectedTerms: string[], coachingPrompt: string|null }}
477
+ */
478
+ function promptSettingsFor(pairConfig = {}, overrides = {}) {
479
+ return {
480
+ name: pairConfig.name,
481
+ register: pairConfig.register,
482
+ // Language-specific gender guidance from the card (e.g., écriture inclusive
483
+ // for French, Doppelpunkt notation for German). Falls back to a generic
484
+ // rule in buildSystemMessage when null.
485
+ genderGuidance: pairConfig.genderGuidance || null,
486
+ // "genderGuidance": false in the config — no gender instruction at all
487
+ // (not even the generic one).
488
+ genderOff: pairConfig.genderGuidanceSource === 'off',
489
+ // User-provided global context (e.g., "This is a developer tool README").
490
+ // Injected into the system message to give the LLM domain awareness.
491
+ promptContext: pairConfig.promptContext || null,
492
+ // Names the project keeps exactly as written (config.protectedTerms).
493
+ protectedTerms: pairConfig.protectedTerms || [],
494
+ // Coaching prompt text — free-text instructions read from a coaching file.
495
+ // Injected between register and rules in the system message.
496
+ coachingPrompt: pairConfig.coachingPrompt || null,
497
+ ...overrides,
498
+ };
499
+ }
500
+
457
501
  /**
458
502
  * Build the system message preamble (cached across batches).
459
503
  *
@@ -470,15 +514,18 @@ class LLMMethod extends TranslationMethod {
470
514
  * being translated (e.g., "This is a developer tool for i18n"). Injected
471
515
  * after the role line to give the LLM domain awareness.
472
516
  *
473
- * @param {object} langConfig - { name, register, genderGuidance?, promptContext? }
517
+ * @param {object} langConfig - { name, register, genderGuidance?, promptContext?, protectedTerms? }
474
518
  * @returns {string} System message
475
519
  */
476
520
  function buildSystemMessage(langConfig) {
477
521
  // Use language-specific gender guidance when available from the card,
478
522
  // otherwise fall back to a generic rule for unknown languages.
479
- const genderRule = langConfig.genderGuidance
480
- ? `- Gender: ${langConfig.genderGuidance}`
481
- : '- When gender is ambiguous, prefer gender-neutral forms or the most inclusive option available in ' + langConfig.name + '.';
523
+ // "genderGuidance": false — the config asked for none.
524
+ const genderRule = langConfig.genderOff
525
+ ? null
526
+ : langConfig.genderGuidance
527
+ ? `- Gender: ${langConfig.genderGuidance}`
528
+ : '- When gender is ambiguous, prefer gender-neutral forms or the most inclusive option available in ' + langConfig.name + '.';
482
529
 
483
530
  // Inject user-provided promptContext to give the LLM domain awareness.
484
531
  // This is set in the top-level config and flows through the pair graph.
@@ -499,10 +546,8 @@ Register/tone: ${langConfig.register}
499
546
  ${coachingBlock}
500
547
  Rules:
501
548
  - Translate ONLY the values, keep the keys exactly as-is.
502
- - Proper nouns (product names, company names, place names) should NOT be translated.
503
- - Technical terms and role descriptions that are industry-standard should stay in English.
504
- ${genderRule}
505
- - Respect the UI element type: button labels should be concise, descriptions can be natural-length, error messages should be clear and direct.
549
+ ${nameRules(langConfig.protectedTerms)}
550
+ ${genderRule ? `${genderRule}\n` : ''}- Respect the UI element type: button labels should be concise, descriptions can be natural-length, error messages should be clear and direct.
506
551
  - Quotation marks inside a translated value: use the target language's typographic quotation marks (e.g. “…”, «…», 「…」); if you must use ASCII double quotes, escape them as \\" so the JSON stays valid.
507
552
  - Return ONLY valid JSON, no markdown fences, no explanation.`;
508
553
  }
@@ -510,8 +555,9 @@ ${genderRule}
510
555
  /**
511
556
  * Build the user message (varies per batch).
512
557
  *
513
- * Contains: UI context hints for key names + the JSON payload.
514
- * This changes with every batch, so it's never cached.
558
+ * Contains: the glossary terms this batch contains (REQUIRED TERMINOLOGY),
559
+ * UI context hints for key names + the JSON payload. This changes with
560
+ * every batch, so it's never cached.
515
561
  *
516
562
  * Descriptions: When Docusaurus {message, description} files provide
517
563
  * developer-written descriptions (e.g., "The title of the blog page"),
@@ -520,9 +566,12 @@ ${genderRule}
520
566
  *
521
567
  * @param {object} toTranslate - Key-value map to translate
522
568
  * @param {object} [descriptions] - Optional key→description map from Docusaurus
569
+ * @param {Object<string, string>|null} [glossary] - The project glossary
570
+ * (lib/methods/coaching-data.js projectGlossary); only the terms this
571
+ * batch contains are listed
523
572
  * @returns {string} User message
524
573
  */
525
- function buildUserMessage(toTranslate, descriptions) {
574
+ function buildUserMessage(toTranslate, descriptions, glossary = null) {
526
575
  const typeHints = inferKeyTypes(toTranslate);
527
576
 
528
577
  // Merge auto-inferred type hints with Docusaurus descriptions.
@@ -547,7 +596,7 @@ function buildUserMessage(toTranslate, descriptions) {
547
596
  ? `UI context for these keys:\n${typeHints.join('\n')}\n\n`
548
597
  : '';
549
598
 
550
- return `${hintsBlock}${JSON.stringify(toTranslate, null, 2)}`;
599
+ return `${terminologyBlock(toTranslate, glossary)}${hintsBlock}${JSON.stringify(toTranslate, null, 2)}`;
551
600
  }
552
601
 
553
602
  /**
@@ -589,4 +638,4 @@ function inferKeyTypes(toTranslate) {
589
638
  return hints;
590
639
  }
591
640
 
592
- export { LLMMethod, buildPrompt, buildSystemMessage, buildUserMessage, isUnsafeKey, inferKeyTypes };
641
+ export { LLMMethod, PLAIN_LLM_METHODS, buildPrompt, buildSystemMessage, buildUserMessage, promptSettingsFor, isUnsafeKey, inferKeyTypes };
@@ -13,7 +13,34 @@
13
13
  */
14
14
 
15
15
  import { OpenAIMethod } from './openai.js';
16
- import { getEnvOrFileVar } from '../api-key.js';
16
+ import { isLoopbackEndpoint, localMachineCost } from './http-utils.js';
17
+ import { onCIRunner } from '../missing-key.js';
18
+
19
+ /** How long the readiness probe waits for a local endpoint to answer. */
20
+ const PROBE_TIMEOUT_MS = 3000;
21
+ // One probe per endpoint per process: a project with ten local pairs asks once.
22
+ const probes = new Map();
23
+
24
+ /** Does anything answer HTTP at `<base>/models`? (Any status counts.) */
25
+ function endpointAnswers(base) {
26
+ if (!probes.has(base)) {
27
+ probes.set(base, (async () => {
28
+ const ctl = new AbortController();
29
+ const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
30
+ try {
31
+ const res = await fetch(`${base}/models`, { signal: ctl.signal });
32
+ // Drain it: the answer is all that matters, not the body.
33
+ try { await res.arrayBuffer(); } catch { /* the status was enough */ }
34
+ return true;
35
+ } catch {
36
+ return false;
37
+ } finally {
38
+ clearTimeout(timer);
39
+ }
40
+ })());
41
+ }
42
+ return probes.get(base);
43
+ }
17
44
 
18
45
  class LocalMethod extends OpenAIMethod {
19
46
  constructor(options = {}) {
@@ -24,13 +51,16 @@ class LocalMethod extends OpenAIMethod {
24
51
  _getProviderLabel() { return 'Local (OpenAI-compatible)'; }
25
52
  _getApiKeyEnvVar() { return 'OPENAI_API_KEY'; }
26
53
  _getDefaultModel() { return 'llama3.1'; }
54
+ // A local server names its models itself ("llama3.1:8b",
55
+ // "Qwen/Qwen2.5-7B-Instruct"): ids are sent as written, never mapped.
56
+ _getModelVendor() { return null; }
27
57
 
28
58
  // Endpoint precedence: options.baseUrl > LOCAL_API_BASE > OPENAI_API_BASE >
29
- // Ollama default. (Matches the harness LocalProvider resolution order.)
59
+ // OPENAI_BASE_URL (the OpenAI SDK's name) > Ollama default. (Matches the harness LocalProvider resolution order.)
30
60
  _getApiBaseEnvVar() { return 'LOCAL_API_BASE'; }
31
- _getDefaultApiBase() {
32
- return getEnvOrFileVar('OPENAI_API_BASE') || 'http://localhost:11434/v1';
33
- }
61
+ _getApiBaseEnvVars() { return ['LOCAL_API_BASE', 'OPENAI_API_BASE', 'OPENAI_BASE_URL']; }
62
+ _getDefaultApiBase() { return 'http://localhost:11434/v1'; }
63
+ _getDefaultApiBaseLabel() { return 'the default, Ollama'; }
34
64
 
35
65
  // Local servers ignore auth — supply a placeholder when no key is set so
36
66
  // translate() doesn't skip the run for a "missing" key.
@@ -38,12 +68,60 @@ class LocalMethod extends OpenAIMethod {
38
68
  return super._resolveApiKey(options) || 'not-needed';
39
69
  }
40
70
 
41
- checkReadiness(_context) {
42
- return { ready: true }; // local endpoint; no API key required
71
+ /**
72
+ * A local endpoint needs no key — but it must answer. Checked even when
73
+ * nothing is queued (as hosted methods' keys are): a project whose config
74
+ * says "local", synced in CI with the guide's default line, went green on
75
+ * every push that changed no string and failed only on the first that did
76
+ * (Round 8, i18next persona). Any HTTP answer from `<base>/models` counts
77
+ * (a server that does not list models still answers); refused, unknown
78
+ * host or no answer within a few seconds does not. A dry run reports it
79
+ * as a warning (lib/sync.js preflight).
80
+ *
81
+ * Off a CI runner the answer is `unreachable: true` (with the `endpoint`):
82
+ * sync decides after its plan — a run that sends nothing to the model (a
83
+ * redo served from the cache) goes on with a warning, one that sends
84
+ * something stops as before (Round 11, Django persona). On a runner it
85
+ * still stops at once, for the reason above.
86
+ */
87
+ async checkReadiness({ cwd } = {}) {
88
+ let resolved;
89
+ try { resolved = this._resolveApiBaseSource(cwd ? { cwd } : {}); } catch { resolved = { base: null, from: null }; }
90
+ if (!resolved.base) return { ready: true };
91
+ if (await endpointAnswers(resolved.base)) return { ready: true };
92
+ const where = this._describeEndpoint(cwd ? { cwd } : {}) || resolved.base;
93
+ const inCI = onCIRunner();
94
+ return {
95
+ ready: false,
96
+ unreachable: true,
97
+ endpoint: where,
98
+ reason: `the config uses the "local" method, and no model server answers at ${where}. `
99
+ + (inCI
100
+ ? 'A CI runner has no model server: pass --method llm (with OPENROUTER_API_KEY added as a repository secret and passed to the sync step), '
101
+ + 'or start one in the job and set LOCAL_API_BASE to it'
102
+ : 'Start it (e.g. `ollama serve`), set LOCAL_API_BASE to a server that runs, or pass --method llm to use a hosted model'),
103
+ };
104
+ }
105
+
106
+ // No model listing. The inherited OpenAI listing called
107
+ // api.openai.com/v1/models — a request off the machine on every "local"
108
+ // run, the one method people choose to stay on it (found 2026-10-03). A
109
+ // local server's own /models is not a reliable allow-list either (many
110
+ // serve whatever model name is asked), so the model is not pre-validated.
111
+ async _fetchModels(_apiKey) {
112
+ return null;
43
113
  }
44
114
 
45
- // Cost is genuinely unknown for local/self-hosted models — never $0.
46
- estimateCost(_keyCount, _pairConfig = {}) {
115
+ // A model served on THIS machine (Ollama's default, LM Studio, a forge
116
+ // model on 127.0.0.1) costs $0 in API fees — said as such, with what is not
117
+ // counted. Any other endpoint (Groq, Together, a LAN box) is genuinely
118
+ // unknown to the tool — never $0.
119
+ estimateCost(_keyCount, pairConfig = {}, { cwd = null } = {}) {
120
+ const base = this._resolveApiBase({
121
+ ...(pairConfig?.baseUrl ? { baseUrl: pairConfig.baseUrl } : {}),
122
+ ...(cwd && { cwd }),
123
+ });
124
+ if (isLoopbackEndpoint(base)) return localMachineCost(base);
47
125
  return {
48
126
  estimatedCost: null,
49
127
  currency: 'USD',
@@ -63,11 +141,16 @@ class LocalMethod extends OpenAIMethod {
63
141
  }
64
142
 
65
143
  getSetupHelp() {
144
+ // Name the endpoint this run used and the setting that chose it: "fetch
145
+ // failed" alone left people guessing which of four settings was in play.
146
+ const endpoint = this._describeEndpoint();
66
147
  return [
67
148
  ' Local (OpenAI-compatible) endpoint:',
149
+ ...(endpoint ? [` • This run sent requests to ${endpoint}.`] : []),
68
150
  ' • Ollama: install from https://ollama.com, run `ollama serve`,',
69
151
  ' then `ollama pull llama3.1` (default http://localhost:11434/v1).',
70
- ' • Or point at vLLM / LM Studio / Groq via LOCAL_API_BASE or OPENAI_API_BASE,',
152
+ ' • Or point at vLLM / LM Studio / Groq via LOCAL_API_BASE (checked first),',
153
+ ' OPENAI_API_BASE or OPENAI_BASE_URL — in the environment or .env.local / .env —',
71
154
  ' and set the model with --model (e.g. qwen2.5, mistral).',
72
155
  ];
73
156
  }