champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,184 @@
1
+ /**
2
+ * TranslationMethod — base interface for all translation methods.
3
+ *
4
+ * WHY: champollion v2 hardcoded a single OpenRouter fetch in translate.js.
5
+ * v3 abstracts this into a pluggable method system so that different
6
+ * language pairs can use different translation strategies:
7
+ *
8
+ * - llm: Direct LLM prompt (current behavior, cheapest)
9
+ * - llm-coached: LLM + grammar/dictionary injection (better for complex morphology)
10
+ * - fst-gated: (planned) LLM + deterministic morphological gate
11
+ * - human-review: (planned) LLM draft flagged for human review
12
+ *
13
+ * Each method must implement:
14
+ * - translate(keys, sourceFlat, pairConfig, options) → { key: value } or null
15
+ * - estimateCost(keyCount) → { estimatedCost: number|null, currency, source }
16
+ * - getQualityTier() → 'standard' | 'high' | 'research' | 'verified'
17
+ * - getProvenance() → { resources: [], commercialReady: boolean, flags: [] }
18
+ *
19
+ * HOW IT WORKS:
20
+ * The translate orchestrator in translate.js looks up the method name
21
+ * from the pair config, instantiates the corresponding method class,
22
+ * and delegates the translation call. The method handles prompting,
23
+ * API communication, and any post-processing specific to its strategy.
24
+ */
25
+
26
+ import { getLastTranslationError } from './translation-error.js';
27
+
28
+ /**
29
+ * Base class for translation methods.
30
+ * Subclasses must override translate() at minimum.
31
+ */
32
+ class TranslationMethod {
33
+ constructor(name, options = {}) {
34
+ this.name = name;
35
+ this.options = options;
36
+ }
37
+
38
+ /**
39
+ * Translate a set of key-value pairs.
40
+ *
41
+ * @param {string[]} keys - Flat dot-notation keys to translate
42
+ * @param {object} sourceFlat - Full flattened source locale (for value lookup)
43
+ * @param {import('../types.js').PairConfig} pairConfig - Pair config from pairs.js (method, model, register, etc.)
44
+ * @param {object} options - { apiKey, batchSize, ... }
45
+ * @param {function} [options.onProgress] - Optional callback fired after each batch chunk: onProgress(completedKeys, totalKeys). Used by sync.js to drive the CLI progress bar.
46
+ * @returns {object|null} Map of key → translated value, or null if all batches failed
47
+ */
48
+ async translate(keys, sourceFlat, pairConfig, options) {
49
+ throw new Error(`TranslationMethod.translate() not implemented by ${this.name}`);
50
+ }
51
+
52
+ /**
53
+ * Translate freeform text content (e.g., Markdown body).
54
+ *
55
+ * Not all methods support this — freeform content is harder to gate
56
+ * deterministically. Methods that don't support it should return null,
57
+ * and the orchestrator will fall back to the default LLM method.
58
+ *
59
+ * @param {string} prompt - Complete translation prompt
60
+ * @param {import('../types.js').PairConfig} pairConfig - Pair config
61
+ * @param {object} options - { apiKey, model }
62
+ * @returns {string|null} Translated text, or null if unsupported
63
+ */
64
+ async translateContent(prompt, pairConfig, options) {
65
+ // Default: unsupported. Subclasses override if they can handle freeform.
66
+ return null;
67
+ }
68
+
69
+ /**
70
+ * Estimate the cost of translating N keys with this method.
71
+ *
72
+ * Subclasses should override with real pricing data.
73
+ * Returning `estimatedCost: null` means "unknown" — consumers must
74
+ * distinguish this from zero (which would mean "free").
75
+ *
76
+ * @param {number} keyCount - Number of keys to translate
77
+ * @param {import('../types.js').PairConfig} [pairConfig] - Pair config (method, model, etc.) for model-specific pricing
78
+ * @returns {Promise<{ estimatedCost: number|null, currency: string, source: string }>|{ estimatedCost: number|null, currency: string, source: string }}
79
+ */
80
+ estimateCost(keyCount, pairConfig) {
81
+ return { estimatedCost: null, currency: 'USD', source: 'none' };
82
+ }
83
+
84
+ /**
85
+ * Get the quality tier for this method.
86
+ *
87
+ * @deprecated Use plugin benchmarks instead of tier labels.
88
+ * Tier labels are subjective; benchmarks are measurable.
89
+ * Kept for backward compat — nothing makes dispatch decisions based on this.
90
+ *
91
+ * @returns {string} One of: 'standard', 'high', 'research', 'verified'
92
+ */
93
+ getQualityTier() {
94
+ return 'standard';
95
+ }
96
+
97
+ /**
98
+ * Get provenance information for this method.
99
+ *
100
+ * Lists all external resources (datasets, tools, APIs) that this method
101
+ * depends on, their licenses, and whether commercial use is cleared.
102
+ *
103
+ * @returns {{ resources: Array, commercialReady: boolean, flags: string[] }}
104
+ */
105
+ getProvenance() {
106
+ return {
107
+ resources: [],
108
+ commercialReady: true,
109
+ flags: [],
110
+ };
111
+ }
112
+
113
+ /**
114
+ * Preflight readiness check — can this method execute right now?
115
+ *
116
+ * Called by resolveRuntime() BEFORE entering the translation loop.
117
+ * If any pair's method returns { ready: false }, the CLI aborts with
118
+ * actionable guidance instead of silently producing garbage output.
119
+ *
120
+ * WHY: Without this, a missing API key was only discovered deep inside
121
+ * the translation loop. For JSON sync, the method returned null and
122
+ * sync.js caught it. But for content sync, the code checked `if (apiKey)`
123
+ * directly and silently wrote hundreds of [EN] fallback files — making
124
+ * it look like translation succeeded when nothing was translated.
125
+ *
126
+ * Now we check at startup: no gas, no ignition.
127
+ *
128
+ * @param {object} context - Runtime context for the readiness check
129
+ * @param {string|null} context.apiKey - The OpenRouter/primary API key
130
+ * @param {string} context.cwd - Working directory (for .env file lookups)
131
+ * @returns {{ ready: boolean, reason?: string }}
132
+ */
133
+ checkReadiness(context) {
134
+ // Base: always ready. Methods override to check their prerequisites.
135
+ return { ready: true };
136
+ }
137
+
138
+ /**
139
+ * Return setup guidance lines when translation fails.
140
+ *
141
+ * ⚠️ MAINTENANCE NOTE FOR FUTURE DEVELOPERS:
142
+ * Each method's setup help lives in its OWN class, not in sync.js.
143
+ * This prevents the sync orchestrator from accumulating provider-specific
144
+ * help text (which was a 125-line if/else chain before this refactor).
145
+ *
146
+ * When you add a new translation method, override this method in your
147
+ * subclass with provider-specific setup instructions.
148
+ *
149
+ * @returns {string[]} Array of lines to print to stderr, or empty array
150
+ */
151
+ getSetupHelp() {
152
+ return [` Check your API key and configuration for method "${this.name}".`];
153
+ }
154
+
155
+ /**
156
+ * Build the "key is set but translation failed" help line, tailored to the
157
+ * last-seen HTTP failure so we don't blame billing for a bad key.
158
+ *
159
+ * Subclasses call this from getSetupHelp()'s key-present branch instead of
160
+ * hardcoding a generic "check billing" line. When no failure was recorded
161
+ * (e.g. a unit test, or a non-HTTP failure), it returns the same generic
162
+ * guidance as before so behavior is unchanged for those cases.
163
+ *
164
+ * @param {string} dashboardLabel - Where the user manages quota/billing for
165
+ * this provider (e.g. 'OpenRouter dashboard', 'Azure Portal').
166
+ * @returns {string[]} One help line, indented to match getSetupHelp output.
167
+ */
168
+ _apiFailureHelp(dashboardLabel) {
169
+ const err = getLastTranslationError();
170
+ const kind = err?.kind;
171
+ if (kind === 'auth') {
172
+ return [` API key is set but the provider rejected it (HTTP ${err.status}) — the key is invalid, expired, or lacks access. Re-check the key value itself, not your billing.`];
173
+ }
174
+ if (kind === 'quota') {
175
+ return [` API key is set but the request was rate-limited / over quota (HTTP 429). Check your ${dashboardLabel} for quota/billing, then retry.`];
176
+ }
177
+ if (kind === 'server') {
178
+ return [` API key is set but the provider returned a server error (HTTP ${err.status}). This is an upstream issue — wait and retry; check the provider status page if it persists.`];
179
+ }
180
+ return [` API key is set but translation failed. Check your ${dashboardLabel} for quota/billing.`];
181
+ }
182
+ }
183
+
184
+ export { TranslationMethod };
@@ -0,0 +1,45 @@
1
+ /**
2
+ * Content prompt separator — shared between all methods that handle
3
+ * freeform Markdown content translation.
4
+ *
5
+ * buildContentPrompt() (content.js) constructs prompts as:
6
+ * [translation instructions]\n---\n[markdown body]
7
+ *
8
+ * API-based methods (DeepL, Google, Microsoft, LibreTranslate) need to
9
+ * extract just the Markdown body because they don't understand instruction
10
+ * prompts — they translate raw text. LLM methods send the full prompt as-is.
11
+ *
12
+ * WHY THIS MODULE: Four method files each defined `const separator = '\n---\n'`
13
+ * and had their own split logic. If buildContentPrompt() ever changed its
14
+ * separator format, all four would silently break. Single source of truth.
15
+ */
16
+
17
+ /**
18
+ * The separator string used by buildContentPrompt() to divide
19
+ * translation instructions from the Markdown body.
20
+ */
21
+ export const CONTENT_SEPARATOR = '\n---\n';
22
+
23
+ /**
24
+ * Extract the Markdown body from a content prompt.
25
+ *
26
+ * @param {string} prompt - Full prompt from buildContentPrompt()
27
+ * @returns {string} The Markdown body after the separator, or the full
28
+ * prompt if no separator is found (with a warning logged).
29
+ */
30
+ export function extractContentBody(prompt) {
31
+ const sepIdx = prompt.indexOf(CONTENT_SEPARATOR);
32
+ if (sepIdx === -1) {
33
+ // Warn but don't throw — the caller may be passing raw content
34
+ // that wasn't built by buildContentPrompt(). This is unexpected
35
+ // in production but shouldn't crash edge cases or tests.
36
+ if (typeof process !== 'undefined' && process.stderr) {
37
+ process.stderr.write(
38
+ '[WARN] Content prompt missing separator "\\n---\\n". ' +
39
+ 'Expected output from buildContentPrompt().\n'
40
+ );
41
+ }
42
+ return prompt;
43
+ }
44
+ return prompt.slice(sepIdx + CONTENT_SEPARATOR.length);
45
+ }
@@ -0,0 +1,426 @@
1
+ import path from 'node:path';
2
+ import fs from 'node:fs';
3
+ import { TranslationMethod } from './base.js';
4
+ import { getEnvOrFileVar } from '../api-key.js';
5
+ import { loadCoachingData, DEFAULT_COACHING_DIR } from './llm-coached.js';
6
+ import { getLanguageCard } from '../registers.js';
7
+ import { EST_CHARS_PER_KEY, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
8
+ import { estimateProviderCost } from './provider-pricing.js';
9
+ import { extractContentBody } from './content-separator.js';
10
+ import { output } from '../output.js';
11
+ import {
12
+ MAX_RETRIES,
13
+ } from './http-utils.js';
14
+ import { fetchWithRetry } from './fetch-with-retry.js';
15
+ import { pMap } from '../concurrent.js';
16
+
17
+ const DEEPL_REQUEST_TIMEOUT_MS = 15000;
18
+
19
+ class DeepLMethod extends TranslationMethod {
20
+ constructor(options = {}) {
21
+ super('deepl', options);
22
+ this._coachingCache = new Map();
23
+ }
24
+
25
+ // ── API resolution helpers ──────────────────────────────────────
26
+
27
+ /**
28
+ * Resolve the DeepL API key from options, env vars, or .env files.
29
+ * @param {object} options - Caller-provided options
30
+ * @returns {string|null}
31
+ */
32
+ _resolveApiKey(options) {
33
+ return options.deeplApiKey
34
+ || getEnvOrFileVar('DEEPL_API_KEY')
35
+ || getEnvOrFileVar('DEEPL_API_KEY', options.cwd);
36
+ }
37
+
38
+ /**
39
+ * Resolve the DeepL API base URL based on key type.
40
+ * Free keys (ending in ':fx') use the free-tier endpoint.
41
+ * @param {string} apiKey - Resolved API key
42
+ * @returns {string} API base URL
43
+ */
44
+ _resolveApiBase(apiKey) {
45
+ const isFree = apiKey.endsWith(':fx');
46
+ return isFree ? 'https://api-free.deepl.com' : 'https://api.deepl.com';
47
+ }
48
+
49
+ /**
50
+ * Translate a batch of key-value pairs via DeepL API.
51
+ *
52
+ * @param {string[]} keys - Flat dot-notation keys to translate
53
+ * @param {object} sourceFlat - Full flattened source locale
54
+ * @param {import('../types.js').PairConfig} pairConfig - Pair config
55
+ * @param {object} options - { apiKey, batchSize }
56
+ * @returns {object|null} Map of key → translated value, or null
57
+ */
58
+ async translate(keys, sourceFlat, pairConfig, options) {
59
+ const apiKey = this._resolveApiKey(options);
60
+
61
+ if (!apiKey) {
62
+ output.warn('DeepL: no API key — skipping.');
63
+ return null;
64
+ }
65
+
66
+ const targetLocale = pairConfig.target;
67
+ const sourceLocale = pairConfig.source || 'en';
68
+ const allTranslated = {};
69
+
70
+ const apiBase = this._resolveApiBase(apiKey);
71
+
72
+ // Determine DeepL formality parameter from language card metadata.
73
+ //
74
+ // HOW: Each register preset in a language card can declare a `deeplFormality`
75
+ // field ('prefer_more', 'prefer_less', or 'default') that maps directly to
76
+ // DeepL's API parameter. This is fully data-driven — no regex heuristics on
77
+ // preset key names.
78
+ //
79
+ // WHY data-driven: DeepL supports formality across multiple formality
80
+ // *systems* — T-V (French), keigo (Japanese), pronoun hierarchies (Vietnamese).
81
+ // A regex on key names (e.g., /formal|vous/) was fragile: it broke for
82
+ // neutral presets, non-Latin key names, and languages where the formal/casual
83
+ // boundary doesn't map to T-V. Putting the mapping in the data means each
84
+ // language card author explicitly declares what DeepL should do.
85
+ //
86
+ // WHY prefer_*: 'prefer_more'/'prefer_less' gracefully degrade to 'default'
87
+ // for unsupported languages instead of returning an API error.
88
+ let formality = 'default';
89
+ const card = getLanguageCard(targetLocale);
90
+ if (card?.methodSupport?.deepl?.formality) {
91
+ // Look up the active preset's declared DeepL formality mapping.
92
+ // pairConfig.registerPreset holds the preset key name (e.g., "casual-tu"),
93
+ // while pairConfig.register holds the resolved prompt text. We need
94
+ // the key to look up preset-specific metadata like deeplFormality.
95
+ const presetKey = pairConfig.registerPreset;
96
+ const activePreset = presetKey && card.registers?.[presetKey];
97
+ if (activePreset?.deeplFormality) {
98
+ // Preset has an explicit mapping — use it directly
99
+ formality = activePreset.deeplFormality;
100
+ } else {
101
+ // No active preset key (custom text) or no mapping — use card default
102
+ const defaultKey = card.formality?.default;
103
+ const defaultPreset = defaultKey && card.registers?.[defaultKey];
104
+ if (defaultPreset?.deeplFormality) {
105
+ formality = defaultPreset.deeplFormality;
106
+ }
107
+ }
108
+ }
109
+
110
+ // Load coaching data for glossary creation
111
+ const cwd = options.cwd || process.cwd();
112
+ const coachingDir = options.coachingDir || path.join(cwd, DEFAULT_COACHING_DIR);
113
+ const coaching = loadCoachingData(coachingDir, targetLocale, this._coachingCache);
114
+
115
+ let glossaryId = null;
116
+ if (coaching && coaching.dictionary && Object.keys(coaching.dictionary).length > 0) {
117
+ glossaryId = await this._syncGlossary(apiBase, apiKey, sourceLocale, targetLocale, coaching.dictionary);
118
+ }
119
+
120
+ const maxSegments = pairConfig.batchSize || options.batchSize || 128;
121
+
122
+ const batchChunks = [];
123
+ for (let i = 0; i < keys.length; i += maxSegments) {
124
+ batchChunks.push(keys.slice(i, i + maxSegments));
125
+ }
126
+
127
+ await pMap(batchChunks, async (chunk, idx) => {
128
+ const orderedKeys = [];
129
+ const sourceTexts = [];
130
+
131
+ for (const key of chunk) {
132
+ const val = sourceFlat[key];
133
+ if (val && typeof val === 'string') {
134
+ orderedKeys.push(key);
135
+ sourceTexts.push(val);
136
+ }
137
+ }
138
+
139
+ if (sourceTexts.length === 0) return;
140
+
141
+ const result = await this._translateBatchWithRetry({
142
+ apiBase,
143
+ apiKey,
144
+ orderedKeys,
145
+ sourceTexts,
146
+ sourceLocale,
147
+ targetLocale,
148
+ formality,
149
+ glossaryId,
150
+ batchNum: idx + 1,
151
+ });
152
+
153
+ if (result) {
154
+ Object.assign(allTranslated, result);
155
+ }
156
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
157
+
158
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
159
+ }
160
+
161
+ /**
162
+ * Translate freeform Markdown content via DeepL API.
163
+ *
164
+ * Uses the same protect/restore approach as Google Translate — the caller
165
+ * has already shielded code blocks and shortcodes with ⟦PROTECTED_N⟧
166
+ * placeholders. We send the protected text through DeepL as a single
167
+ * translation request.
168
+ *
169
+ * @param {string} prompt - Complete translation prompt from buildContentPrompt()
170
+ * @param {import('../types.js').PairConfig} pairConfig - Pair config
171
+ * @param {object} options - { apiKey }
172
+ * @returns {string|null} Translated text, or null on failure
173
+ */
174
+ async translateContent(prompt, pairConfig, options) {
175
+ const apiKey = this._resolveApiKey(options);
176
+
177
+ if (!apiKey) {
178
+ output.warn('DeepL: no API key — skipping.');
179
+ return null;
180
+ }
181
+
182
+ // Extract the Markdown body from the translation prompt
183
+ const bodyText = extractContentBody(prompt);
184
+ if (!bodyText.trim()) return null;
185
+
186
+ const apiBase = this._resolveApiBase(apiKey);
187
+ const targetLocale = pairConfig.target;
188
+ const sourceLocale = pairConfig.source || 'en';
189
+
190
+ const response = await fetchWithRetry(`${apiBase}/v2/translate`, {
191
+ method: 'POST',
192
+ headers: {
193
+ 'Content-Type': 'application/json',
194
+ 'Authorization': `DeepL-Auth-Key ${apiKey}`,
195
+ },
196
+ body: JSON.stringify({
197
+ text: [bodyText],
198
+ target_lang: targetLocale.toUpperCase(),
199
+ source_lang: sourceLocale.toUpperCase(),
200
+ }),
201
+ }, {
202
+ label: 'DeepL content',
203
+ timeoutMs: DEEPL_REQUEST_TIMEOUT_MS * 2,
204
+ });
205
+
206
+ if (!response) return null;
207
+
208
+ if (!response.ok) {
209
+ const errorBody = await response.text();
210
+ output.error(`DeepL content: ${response.status} — ${errorBody}`);
211
+ return null;
212
+ }
213
+
214
+ const json = await response.json();
215
+ const translations = json?.translations;
216
+ if (!translations || translations.length === 0) {
217
+ output.error('DeepL content: empty response');
218
+ return null;
219
+ }
220
+
221
+ return translations[0].text;
222
+ }
223
+
224
+ estimateCost(keyCount) {
225
+ return estimateProviderCost('deepl', keyCount);
226
+ }
227
+
228
+ checkReadiness(context) {
229
+ // Resolve through the same chain translate() uses (options → env →
230
+ // .env.local/.env in cwd) so readiness can never fail for a key the
231
+ // loader *would* read — e.g. DEEPL_API_KEY set only in .env.local.
232
+ if (!this._resolveApiKey(context || {})) {
233
+ return { ready: false, reason: 'No DeepL API key (DEEPL_API_KEY).' };
234
+ }
235
+ return { ready: true };
236
+ }
237
+
238
+ getQualityTier() {
239
+ return 'standard';
240
+ }
241
+
242
+ getProvenance() {
243
+ return {
244
+ resources: [
245
+ {
246
+ name: 'DeepL Translation API',
247
+ license: 'Proprietary (DeepL ToS)',
248
+ type: 'api',
249
+ },
250
+ ],
251
+ commercialReady: true,
252
+ flags: [],
253
+ };
254
+ }
255
+
256
+ getSetupHelp() {
257
+ // Resolve through the same env-or-file lookup the runtime uses, so a key
258
+ // that only lives in .env.local doesn't get misreported as "Missing API
259
+ // Key" after a real 401 — that would point the user at the wrong fix.
260
+ const apiKey = getEnvOrFileVar('DEEPL_API_KEY');
261
+ if (!apiKey) {
262
+ return [
263
+ '',
264
+ ' ┌─ Missing API Key ─────────────────────────────────────────────┐',
265
+ ' │ DeepL requires an API authentication key. │',
266
+ ' │ │',
267
+ ' │ 1. Sign up at https://www.deepl.com/pro-api (free tier avail.) │',
268
+ ' │ 2. Run: export DEEPL_API_KEY=... │',
269
+ ' │ 3. Or add to .env.local: DEEPL_API_KEY=... │',
270
+ ' │ │',
271
+ ' │ Free keys end with ":fx" — detected automatically. │',
272
+ ' └────────────────────────────────────────────────────────────────┘',
273
+ ];
274
+ }
275
+ return this._apiFailureHelp('DeepL dashboard');
276
+ }
277
+
278
+ // Helper to load coaching data
279
+ // _loadCoachingData — removed: now uses shared loadCoachingData from llm-coached.js
280
+
281
+
282
+ // Create or get glossary on DeepL
283
+ async _syncGlossary(apiBase, apiKey, sourceLang, targetLang, dictionary) {
284
+ const sortedEntries = Object.entries(dictionary).sort(([a], [b]) => a.localeCompare(b));
285
+ if (sortedEntries.length === 0) return null;
286
+
287
+ const contentStr = sortedEntries.map(([s, t]) => `${s}\t${t}`).join('\n');
288
+ let hash = 0;
289
+ for (let i = 0; i < contentStr.length; i++) {
290
+ hash = (hash << 5) - hash + contentStr.charCodeAt(i);
291
+ hash |= 0;
292
+ }
293
+ const hashStr = Math.abs(hash).toString(16);
294
+
295
+ const sLang = sourceLang.toUpperCase();
296
+ const tLang = targetLang.toUpperCase();
297
+ const glossaryName = `champollion_${sLang.toLowerCase()}_${tLang.toLowerCase()}_${hashStr}`;
298
+
299
+ try {
300
+ // 1. List existing glossaries to see if we already created it
301
+ const listResponse = await fetchWithRetry(`${apiBase}/v2/glossaries`, {
302
+ method: 'GET',
303
+ headers: {
304
+ 'Authorization': `DeepL-Auth-Key ${apiKey}`,
305
+ },
306
+ });
307
+
308
+ if (listResponse.ok) {
309
+ const listJson = await listResponse.json();
310
+ const existing = listJson.glossaries?.find(
311
+ g => g.name === glossaryName && g.source_lang === sLang && g.target_lang === tLang
312
+ );
313
+ if (existing) {
314
+ return existing.glossary_id;
315
+ }
316
+ }
317
+
318
+ // 2. Create glossary if not found
319
+ const createResponse = await fetchWithRetry(`${apiBase}/v2/glossaries`, {
320
+ method: 'POST',
321
+ headers: {
322
+ 'Authorization': `DeepL-Auth-Key ${apiKey}`,
323
+ 'Content-Type': 'application/json',
324
+ },
325
+ body: JSON.stringify({
326
+ name: glossaryName,
327
+ source_lang: sLang,
328
+ target_lang: tLang,
329
+ entries: contentStr,
330
+ entries_format: 'tsv',
331
+ }),
332
+ });
333
+
334
+ if (createResponse.ok) {
335
+ const createJson = await createResponse.json();
336
+ return createJson.glossary_id;
337
+ } else {
338
+ const errText = await createResponse.text();
339
+ output.warn(`DeepL: Failed to create glossary "${glossaryName}": ${createResponse.status} - ${errText}`);
340
+ return null;
341
+ }
342
+ } catch (err) {
343
+ output.warn(`DeepL: Glossary sync error: ${err.message}`);
344
+ return null;
345
+ }
346
+ }
347
+
348
+ async _translateBatchWithRetry(params, startAttempt = 0) {
349
+ const {
350
+ apiBase,
351
+ apiKey,
352
+ orderedKeys,
353
+ sourceTexts,
354
+ sourceLocale,
355
+ targetLocale,
356
+ formality,
357
+ glossaryId,
358
+ batchNum,
359
+ } = params;
360
+
361
+ const body = {
362
+ text: sourceTexts,
363
+ target_lang: targetLocale.toUpperCase(),
364
+ source_lang: sourceLocale.toUpperCase(),
365
+ };
366
+
367
+ if (formality && formality !== 'default') {
368
+ body.formality = formality;
369
+ }
370
+ if (glossaryId) {
371
+ body.glossary_id = glossaryId;
372
+ }
373
+
374
+ const response = await fetchWithRetry(`${apiBase}/v2/translate`, {
375
+ method: 'POST',
376
+ headers: {
377
+ 'Content-Type': 'application/json',
378
+ 'Authorization': `DeepL-Auth-Key ${apiKey}`,
379
+ },
380
+ body: JSON.stringify(body),
381
+ }, {
382
+ label: `DeepL batch ${batchNum}`,
383
+ timeoutMs: DEEPL_REQUEST_TIMEOUT_MS,
384
+ startAttempt,
385
+ });
386
+
387
+ if (!response) return null;
388
+
389
+ if (!response.ok) {
390
+ const errorBody = await response.text();
391
+ // If we passed a glossary_id and got a 400 bad request, try again without the glossary.
392
+ // Carry forward the current attempt count so the retry budget isn't reset —
393
+ // without this, the recursive call would restart from attempt 0 and allow
394
+ // up to 2× MAX_RETRIES total network requests.
395
+ if (response.status === 400 && glossaryId) {
396
+ output.warn(`DeepL batch ${batchNum} failed with glossary. Retrying without glossary...`);
397
+ return this._translateBatchWithRetry({
398
+ ...params,
399
+ glossaryId: null,
400
+ }, startAttempt + 1);
401
+ }
402
+ output.error(`DeepL batch ${batchNum}: ${response.status} — ${errorBody}`);
403
+ return null;
404
+ }
405
+
406
+ const json = await response.json();
407
+ const translations = json?.translations;
408
+
409
+ if (!translations || translations.length !== orderedKeys.length) {
410
+ output.error(`DeepL batch ${batchNum}: Response length mismatch (expected ${orderedKeys.length}, got ${translations?.length || 0})`);
411
+ return null;
412
+ }
413
+
414
+ const result = {};
415
+ for (let i = 0; i < orderedKeys.length; i++) {
416
+ result[orderedKeys[i]] = translations[i].text;
417
+ }
418
+
419
+ const charCount = sourceTexts.reduce((sum, t) => sum + t.length, 0);
420
+ output.progress(` ✓ DeepL batch ${batchNum} (${orderedKeys.length} keys, ${charCount} chars)`);
421
+
422
+ return result;
423
+ }
424
+ }
425
+
426
+ export { DeepLMethod };