champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,586 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DirectLLMMethod — shared base class for direct LLM provider integrations.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS EXISTS:
|
|
5
|
+
* OpenAI, Anthropic, and Gemini methods share ~90% of their logic:
|
|
6
|
+
* - Constructor with coaching cache
|
|
7
|
+
* - translate() with coaching loading, system message building, batch loop
|
|
8
|
+
* - translateContent() with coaching block prepending
|
|
9
|
+
* - Dictionary hint injection in per-batch user messages
|
|
10
|
+
* - JSON response parsing + key validation
|
|
11
|
+
* - Retry loop with exponential backoff
|
|
12
|
+
*
|
|
13
|
+
* The ONLY things that differ are:
|
|
14
|
+
* - API endpoint URL + auth header format
|
|
15
|
+
* - Request body shape (messages, system param, generationConfig)
|
|
16
|
+
* - Response parsing path (choices[0] vs content[0] vs candidates[0])
|
|
17
|
+
* - Pricing, provenance, model patterns
|
|
18
|
+
*
|
|
19
|
+
* This base class implements all shared logic. Subclasses override a small
|
|
20
|
+
* set of abstract methods to provide provider-specific HTTP details.
|
|
21
|
+
*
|
|
22
|
+
* RUNTIME MODEL VALIDATION:
|
|
23
|
+
* On first translate() call, the base class fetches the provider's available
|
|
24
|
+
* model list and validates the configured model against it. This catches:
|
|
25
|
+
* - Deprecated/retired model names (the exact bug we fixed twice already)
|
|
26
|
+
* - OpenRouter-format model strings used with direct providers
|
|
27
|
+
* - Models from the wrong provider (e.g., claude-* on OpenAI)
|
|
28
|
+
* The model list is cached per-process to avoid repeated API calls.
|
|
29
|
+
*
|
|
30
|
+
* INHERITANCE CHAIN:
|
|
31
|
+
* TranslationMethod → LLMMethod → DirectLLMMethod → OpenAIMethod
|
|
32
|
+
* → AnthropicMethod
|
|
33
|
+
* → GeminiMethod
|
|
34
|
+
*/
|
|
35
|
+
|
|
36
|
+
import path from 'node:path';
|
|
37
|
+
import { LLMMethod, buildSystemMessage, buildUserMessage, isUnsafeKey, inferKeyTypes } from './llm.js';
|
|
38
|
+
import { loadCoachingData, findDictionaryMatches, buildCoachedSystemMessage, buildContentCoachingBlock, DEFAULT_COACHING_DIR } from './llm-coached.js';
|
|
39
|
+
import { getEnvOrFileVar } from '../api-key.js';
|
|
40
|
+
import {
|
|
41
|
+
MAX_RETRIES,
|
|
42
|
+
REQUEST_TIMEOUT_MS,
|
|
43
|
+
isRetryable,
|
|
44
|
+
getBackoffDelay,
|
|
45
|
+
sleep,
|
|
46
|
+
stripCodeFences,
|
|
47
|
+
} from './http-utils.js';
|
|
48
|
+
import { DEFAULT_BATCH_SIZE, DEFAULT_TEMPERATURE, DEFAULT_MAX_RETRIES, DEFAULT_METHOD_CONCURRENCY } from '../config.js';
|
|
49
|
+
import { pMap } from '../concurrent.js';
|
|
50
|
+
import { output } from '../output.js';
|
|
51
|
+
import { recordTranslationError } from './translation-error.js';
|
|
52
|
+
|
|
53
|
+
// ── Model validation patterns ──────────────────────────────────────
|
|
54
|
+
// Used to detect when a model string belongs to the wrong provider.
|
|
55
|
+
// These are intentionally broad — we're catching obvious mismatches,
|
|
56
|
+
// not building a comprehensive model catalog.
|
|
57
|
+
const MODEL_PATTERNS = {
|
|
58
|
+
openai: {
|
|
59
|
+
prefixes: ['gpt-', 'o1', 'o3', 'o4', 'chatgpt-'],
|
|
60
|
+
label: 'OpenAI',
|
|
61
|
+
},
|
|
62
|
+
anthropic: {
|
|
63
|
+
prefixes: ['claude-'],
|
|
64
|
+
label: 'Anthropic',
|
|
65
|
+
},
|
|
66
|
+
gemini: {
|
|
67
|
+
prefixes: ['gemini-'],
|
|
68
|
+
label: 'Google Gemini',
|
|
69
|
+
},
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Per-process cache of available models per provider.
|
|
74
|
+
* Keyed by provider name (e.g., 'openai'), value is a Set of model IDs
|
|
75
|
+
* or null if the fetch failed (so we don't retry on every batch).
|
|
76
|
+
*/
|
|
77
|
+
const _modelListCache = new Map();
|
|
78
|
+
|
|
79
|
+
class DirectLLMMethod extends LLMMethod {
|
|
80
|
+
constructor(options = {}) {
|
|
81
|
+
super(options);
|
|
82
|
+
// Coaching data cache — avoids re-reading .champollion/coaching/<locale>.json per batch
|
|
83
|
+
this._coachingCache = new Map();
|
|
84
|
+
// Track whether we've already validated the model for this instance
|
|
85
|
+
this._modelValidated = false;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// ── Abstract methods — subclasses MUST implement these ──────────
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Environment variable name for this provider's API key.
|
|
92
|
+
* @returns {string} e.g., 'OPENAI_API_KEY'
|
|
93
|
+
*/
|
|
94
|
+
_getApiKeyEnvVar() {
|
|
95
|
+
throw new Error(`${this.name}._getApiKeyEnvVar() not implemented`);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Options key name for this provider's API key.
|
|
100
|
+
* @returns {string} e.g., 'openaiApiKey'
|
|
101
|
+
*/
|
|
102
|
+
_getApiKeyOptionsKey() {
|
|
103
|
+
throw new Error(`${this.name}._getApiKeyOptionsKey() not implemented`);
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Default model ID for this provider.
|
|
108
|
+
* @returns {string} e.g., 'gpt-4o'
|
|
109
|
+
*/
|
|
110
|
+
_getDefaultModel() {
|
|
111
|
+
throw new Error(`${this.name}._getDefaultModel() not implemented`);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Human-readable provider name for log messages.
|
|
116
|
+
* @returns {string} e.g., 'OpenAI'
|
|
117
|
+
*/
|
|
118
|
+
_getProviderLabel() {
|
|
119
|
+
throw new Error(`${this.name}._getProviderLabel() not implemented`);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Build the HTTP request for the provider's chat/generate endpoint.
|
|
124
|
+
*
|
|
125
|
+
* @param {object} params
|
|
126
|
+
* @param {string} params.prompt - User message content
|
|
127
|
+
* @param {string} params.systemMessage - System message (may be null)
|
|
128
|
+
* @param {string} params.apiKey - Provider API key
|
|
129
|
+
* @param {string} params.model - Model ID
|
|
130
|
+
* @param {number} params.temperature - Sampling temperature
|
|
131
|
+
* @param {boolean} params.isJsonMode - Whether to request JSON output
|
|
132
|
+
* @returns {{ url: string, headers: object, body: object }}
|
|
133
|
+
*/
|
|
134
|
+
_buildApiRequest(params) {
|
|
135
|
+
throw new Error(`${this.name}._buildApiRequest() not implemented`);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Extract the text content from the provider's API response JSON.
|
|
140
|
+
*
|
|
141
|
+
* @param {object} json - Parsed response JSON
|
|
142
|
+
* @returns {string|null} Extracted text, or null if missing
|
|
143
|
+
*/
|
|
144
|
+
_extractResponseText(json) {
|
|
145
|
+
throw new Error(`${this.name}._extractResponseText() not implemented`);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Fetch the list of available model IDs from the provider's API.
|
|
150
|
+
*
|
|
151
|
+
* @param {string} apiKey - Provider API key
|
|
152
|
+
* @returns {Promise<string[]|null>} Array of model IDs, or null on failure
|
|
153
|
+
*/
|
|
154
|
+
async _fetchModels(apiKey) {
|
|
155
|
+
// Default: no model listing available. Subclasses override.
|
|
156
|
+
return null;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// ── Shared implementation ──────────────────────────────────────
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Resolve the API key from options, env vars, or .env files.
|
|
163
|
+
* @param {object} options - Caller-provided options
|
|
164
|
+
* @returns {string|null}
|
|
165
|
+
*/
|
|
166
|
+
_resolveApiKey(options) {
|
|
167
|
+
const envVar = this._getApiKeyEnvVar();
|
|
168
|
+
const optKey = this._getApiKeyOptionsKey();
|
|
169
|
+
return options[optKey]
|
|
170
|
+
|| getEnvOrFileVar(envVar)
|
|
171
|
+
|| getEnvOrFileVar(envVar, options.cwd);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// ── OpenAI-compatible endpoint base (base_url) ──────────────────
|
|
175
|
+
// Mirrors the harness OpenAIProvider/LocalProvider: lets a method point at
|
|
176
|
+
// any OpenAI-compatible server (Ollama, vLLM, LM Studio, Groq, Together).
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Default API base (no /chat/completions) for this provider. Override in
|
|
180
|
+
* subclasses (e.g. OpenAI → api.openai.com/v1, Local → Ollama).
|
|
181
|
+
* @returns {string|null}
|
|
182
|
+
*/
|
|
183
|
+
_getDefaultApiBase() { return null; }
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Env var that overrides the API base. Default OPENAI_API_BASE (shared with
|
|
187
|
+
* the harness). Subclasses may override (e.g. LocalMethod → LOCAL_API_BASE).
|
|
188
|
+
* @returns {string}
|
|
189
|
+
*/
|
|
190
|
+
_getApiBaseEnvVar() { return 'OPENAI_API_BASE'; }
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Resolve the endpoint base (no /chat/completions). Precedence:
|
|
194
|
+
* options.baseUrl / this.options.baseUrl > env > subclass default.
|
|
195
|
+
* Normalizes a trailing slash and a full .../chat/completions path.
|
|
196
|
+
* @returns {string|null}
|
|
197
|
+
*/
|
|
198
|
+
_resolveApiBase(options = {}) {
|
|
199
|
+
const envVar = this._getApiBaseEnvVar();
|
|
200
|
+
const fromEnv = envVar
|
|
201
|
+
? (getEnvOrFileVar(envVar) || getEnvOrFileVar(envVar, options.cwd))
|
|
202
|
+
: null;
|
|
203
|
+
const base = options.baseUrl
|
|
204
|
+
|| (this.options && this.options.baseUrl)
|
|
205
|
+
|| fromEnv
|
|
206
|
+
|| this._getDefaultApiBase();
|
|
207
|
+
if (!base) return null;
|
|
208
|
+
let b = String(base).replace(/\/+$/, '');
|
|
209
|
+
if (b.endsWith('/chat/completions')) {
|
|
210
|
+
b = b.slice(0, -'/chat/completions'.length);
|
|
211
|
+
}
|
|
212
|
+
return b;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Validate the model string before making API calls.
|
|
217
|
+
*
|
|
218
|
+
* Checks:
|
|
219
|
+
* 1. OpenRouter-format model strings (contain '/') — wrong method
|
|
220
|
+
* 2. Model belongs to a different provider (e.g., claude-* on OpenAI)
|
|
221
|
+
* 3. Model exists in the provider's API (runtime fetch, cached)
|
|
222
|
+
*
|
|
223
|
+
* Logs warnings but does NOT block — the provider API will give
|
|
224
|
+
* the definitive answer. This is a DX aid, not a gate.
|
|
225
|
+
*
|
|
226
|
+
* @param {string} model - Model ID to validate
|
|
227
|
+
* @param {string} apiKey - API key for model list fetch
|
|
228
|
+
*/
|
|
229
|
+
async _validateModel(model, apiKey) {
|
|
230
|
+
if (this._modelValidated) return;
|
|
231
|
+
this._modelValidated = true;
|
|
232
|
+
|
|
233
|
+
const label = this._getProviderLabel();
|
|
234
|
+
|
|
235
|
+
// Check 1: OpenRouter-format model string (contains '/')
|
|
236
|
+
if (model.includes('/')) {
|
|
237
|
+
output.warn(`${label}: model "${model}" looks like an OpenRouter path.`);
|
|
238
|
+
output.warn(`Direct providers use bare model names (e.g., "${this._getDefaultModel()}").`);
|
|
239
|
+
output.warn(`To use OpenRouter models, set method to 'llm' instead.`);
|
|
240
|
+
return;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// Check 2: Model belongs to a different provider
|
|
244
|
+
for (const [provider, { prefixes, label: providerLabel }] of Object.entries(MODEL_PATTERNS)) {
|
|
245
|
+
if (provider === this.name) continue; // skip own provider
|
|
246
|
+
const matchesOther = prefixes.some(p => model.startsWith(p));
|
|
247
|
+
if (matchesOther) {
|
|
248
|
+
const article = /^[aeiou]/i.test(providerLabel) ? 'an' : 'a';
|
|
249
|
+
output.warn(`${label}: model "${model}" is ${article} ${providerLabel} model.`);
|
|
250
|
+
output.warn(`This provider (${this.name}) cannot serve ${providerLabel} models.`);
|
|
251
|
+
output.warn(`Use --method ${provider} or set "method": "${provider}" in config.`);
|
|
252
|
+
return;
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
// Check 3: Runtime model list validation (cached per-process)
|
|
257
|
+
try {
|
|
258
|
+
let modelSet = _modelListCache.get(this.name);
|
|
259
|
+
|
|
260
|
+
// null = already tried and failed; undefined = never tried
|
|
261
|
+
if (modelSet === undefined) {
|
|
262
|
+
const models = await this._fetchModels(apiKey);
|
|
263
|
+
if (models && models.length > 0) {
|
|
264
|
+
modelSet = new Set(models);
|
|
265
|
+
_modelListCache.set(this.name, modelSet);
|
|
266
|
+
} else {
|
|
267
|
+
// Mark as failed so we don't retry on every batch
|
|
268
|
+
_modelListCache.set(this.name, null);
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
if (modelSet && !modelSet.has(model)) {
|
|
273
|
+
// Find close matches to suggest
|
|
274
|
+
const suggestions = [...modelSet]
|
|
275
|
+
.filter(m => {
|
|
276
|
+
// Only suggest models that support generateContent-style operations
|
|
277
|
+
// (not embedding models, not vision-only, etc.)
|
|
278
|
+
const base = model.split('-')[0];
|
|
279
|
+
return m.startsWith(base);
|
|
280
|
+
})
|
|
281
|
+
.sort()
|
|
282
|
+
.slice(0, 5);
|
|
283
|
+
|
|
284
|
+
output.warn(`${label}: model "${model}" not found in available models.`);
|
|
285
|
+
if (suggestions.length > 0) {
|
|
286
|
+
output.warn(`Similar models: ${suggestions.join(', ')}`);
|
|
287
|
+
}
|
|
288
|
+
output.warn(`The API call will proceed — the provider will give the final verdict.`);
|
|
289
|
+
}
|
|
290
|
+
} catch {
|
|
291
|
+
// Model listing failed silently — don't block translation.
|
|
292
|
+
// The actual translate call will surface any real model errors.
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/**
|
|
297
|
+
* Determine quality tier based on the model name.
|
|
298
|
+
*
|
|
299
|
+
* Provider subclasses can override _getModelTier() for provider-specific
|
|
300
|
+
* mappings. Default: 'standard'.
|
|
301
|
+
*/
|
|
302
|
+
getQualityTier(pairConfig = {}) {
|
|
303
|
+
const model = pairConfig.model || this._getDefaultModel();
|
|
304
|
+
return this._getModelTier(model);
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Map a model name to a quality tier. Override in subclasses.
|
|
309
|
+
* @param {string} model
|
|
310
|
+
* @returns {'budget'|'standard'|'premium'}
|
|
311
|
+
*/
|
|
312
|
+
_getModelTier(model) {
|
|
313
|
+
return 'standard';
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
// ── Core translate() — shared across all direct LLM providers ──
|
|
317
|
+
|
|
318
|
+
async translate(keys, sourceFlat, pairConfig, options) {
|
|
319
|
+
const apiKey = this._resolveApiKey(options);
|
|
320
|
+
|
|
321
|
+
if (!apiKey) {
|
|
322
|
+
output.warn(`${this._getProviderLabel()}: no API key — skipping.`);
|
|
323
|
+
return null;
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
const batchSize = pairConfig.batchSize || options.batchSize || DEFAULT_BATCH_SIZE;
|
|
327
|
+
const model = pairConfig.model || options.model || this._getDefaultModel();
|
|
328
|
+
const maxRetries = pairConfig.maxRetries ?? DEFAULT_MAX_RETRIES;
|
|
329
|
+
const langConfig = {
|
|
330
|
+
name: pairConfig.name,
|
|
331
|
+
register: pairConfig.register,
|
|
332
|
+
};
|
|
333
|
+
|
|
334
|
+
// Validate model on first call (logs warnings, does not block)
|
|
335
|
+
await this._validateModel(model, apiKey);
|
|
336
|
+
|
|
337
|
+
// Load coaching data if available (.champollion/coaching/<locale>.json)
|
|
338
|
+
const cwd = options.cwd || process.cwd();
|
|
339
|
+
const coachingDir = path.join(cwd, DEFAULT_COACHING_DIR);
|
|
340
|
+
const coaching = loadCoachingData(coachingDir, pairConfig.target, this._coachingCache);
|
|
341
|
+
|
|
342
|
+
// If coaching data exists, use the coached system message (grammar/style in system
|
|
343
|
+
// prompt for provider-level caching). Otherwise use the standard system message.
|
|
344
|
+
const systemMessage = coaching
|
|
345
|
+
? buildCoachedSystemMessage(langConfig, coaching)
|
|
346
|
+
: buildSystemMessage(langConfig);
|
|
347
|
+
const allTranslated = {};
|
|
348
|
+
|
|
349
|
+
// Wrap the batch function to inject dictionary hints when coaching is active.
|
|
350
|
+
// Thread the resolved temperature so _callProviderBatch doesn't need pairConfig.
|
|
351
|
+
const resolvedTemperature = pairConfig.temperature ?? DEFAULT_TEMPERATURE;
|
|
352
|
+
const descriptions = options.descriptions || null;
|
|
353
|
+
const batchFn = (batch, opts) => this._callProviderBatch(batch, { ...opts, apiKey, model, coaching, temperature: resolvedTemperature, descriptions });
|
|
354
|
+
|
|
355
|
+
const batchChunks = [];
|
|
356
|
+
for (let i = 0; i < keys.length; i += batchSize) {
|
|
357
|
+
batchChunks.push(keys.slice(i, i + batchSize));
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
await pMap(batchChunks, async (chunk, idx) => {
|
|
361
|
+
const toTranslate = {};
|
|
362
|
+
for (const key of chunk) {
|
|
363
|
+
toTranslate[key] = sourceFlat[key];
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
const result = await this._translateWithCascade(toTranslate, langConfig, {
|
|
367
|
+
apiKey,
|
|
368
|
+
model,
|
|
369
|
+
batchNum: idx + 1,
|
|
370
|
+
maxRetries,
|
|
371
|
+
systemMessage,
|
|
372
|
+
}, batchFn);
|
|
373
|
+
|
|
374
|
+
if (result) {
|
|
375
|
+
Object.assign(allTranslated, result);
|
|
376
|
+
}
|
|
377
|
+
}, { concurrency: DEFAULT_METHOD_CONCURRENCY });
|
|
378
|
+
|
|
379
|
+
return Object.keys(allTranslated).length > 0 ? allTranslated : null;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
// ── Core translateContent() — shared across all direct LLM providers ──
|
|
383
|
+
|
|
384
|
+
async translateContent(prompt, pairConfig, options) {
|
|
385
|
+
const apiKey = this._resolveApiKey(options);
|
|
386
|
+
|
|
387
|
+
if (!apiKey) {
|
|
388
|
+
output.warn(`${this._getProviderLabel()}: no API key — skipping.`);
|
|
389
|
+
return null;
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
const model = pairConfig.model || options.model || this._getDefaultModel();
|
|
393
|
+
|
|
394
|
+
// Prepend coaching context (grammar/style rules) to content prompts when available.
|
|
395
|
+
// Dictionary matching is skipped for freeform content — it's too unpredictable.
|
|
396
|
+
const cwd = options.cwd || process.cwd();
|
|
397
|
+
const coachingDir = path.join(cwd, DEFAULT_COACHING_DIR);
|
|
398
|
+
const coaching = loadCoachingData(coachingDir, pairConfig.target, this._coachingCache);
|
|
399
|
+
let augmentedPrompt = prompt;
|
|
400
|
+
if (coaching) {
|
|
401
|
+
const block = buildContentCoachingBlock(coaching);
|
|
402
|
+
if (block) {
|
|
403
|
+
augmentedPrompt = block + '\n\n' + prompt;
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
// Round 1: standard timeout (2× base = 60s)
|
|
408
|
+
const result = await this._callProviderDirect({
|
|
409
|
+
prompt: augmentedPrompt,
|
|
410
|
+
apiKey,
|
|
411
|
+
model,
|
|
412
|
+
temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
|
|
413
|
+
timeoutMs: REQUEST_TIMEOUT_MS * 2,
|
|
414
|
+
label: `${this._getProviderLabel()} Content`,
|
|
415
|
+
});
|
|
416
|
+
|
|
417
|
+
if (result) return result;
|
|
418
|
+
|
|
419
|
+
// Round 2: escalated — longer cool-down and 4× timeout (120s)
|
|
420
|
+
const label = `${this._getProviderLabel()} Content (escalated)`;
|
|
421
|
+
output.warn(`⟳ ${label}: standard retries exhausted — escalating with extended timeout...`);
|
|
422
|
+
await sleep(10_000);
|
|
423
|
+
|
|
424
|
+
return this._callProviderDirect({
|
|
425
|
+
prompt: augmentedPrompt,
|
|
426
|
+
apiKey,
|
|
427
|
+
model,
|
|
428
|
+
temperature: pairConfig.temperature ?? DEFAULT_TEMPERATURE,
|
|
429
|
+
timeoutMs: REQUEST_TIMEOUT_MS * 4,
|
|
430
|
+
label,
|
|
431
|
+
});
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
// ── Shared batch call — builds user message with coaching hints ──
|
|
435
|
+
|
|
436
|
+
async _callProviderBatch(toTranslate, options) {
|
|
437
|
+
const { apiKey, model, batchNum, systemMessage, coaching, temperature } = options;
|
|
438
|
+
|
|
439
|
+
// Build user message — inject dictionary term matches when coaching is active.
|
|
440
|
+
// Dictionary hints go in the user message (per-batch) rather than the system
|
|
441
|
+
// message (cached) because they're specific to the current batch's source values.
|
|
442
|
+
let prompt;
|
|
443
|
+
if (coaching && coaching.dictionary) {
|
|
444
|
+
const dictHints = findDictionaryMatches(toTranslate, coaching.dictionary);
|
|
445
|
+
const typeHints = inferKeyTypes(toTranslate);
|
|
446
|
+
let userMessage = '';
|
|
447
|
+
if (dictHints.length > 0) {
|
|
448
|
+
userMessage += 'REQUIRED TERMINOLOGY (use these exact translations):\n';
|
|
449
|
+
userMessage += dictHints.map(h => ` • "${h.term}" → "${h.translation}"`).join('\n');
|
|
450
|
+
userMessage += '\n\n';
|
|
451
|
+
}
|
|
452
|
+
if (typeHints.length > 0) {
|
|
453
|
+
userMessage += `UI context for these keys:\n${typeHints.join('\n')}\n\n`;
|
|
454
|
+
}
|
|
455
|
+
userMessage += JSON.stringify(toTranslate, null, 2);
|
|
456
|
+
prompt = userMessage;
|
|
457
|
+
} else {
|
|
458
|
+
prompt = buildUserMessage(toTranslate, options.descriptions || undefined);
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
const label = this._getProviderLabel();
|
|
462
|
+
const content = await this._callProviderDirect({
|
|
463
|
+
prompt,
|
|
464
|
+
systemMessage,
|
|
465
|
+
apiKey,
|
|
466
|
+
model,
|
|
467
|
+
temperature: temperature ?? DEFAULT_TEMPERATURE,
|
|
468
|
+
label: `${label} Batch ${batchNum}`,
|
|
469
|
+
isJsonMode: true,
|
|
470
|
+
});
|
|
471
|
+
|
|
472
|
+
if (!content) return null;
|
|
473
|
+
|
|
474
|
+
try {
|
|
475
|
+
const parsed = JSON.parse(content);
|
|
476
|
+
const expectedKeys = new Set(Object.keys(toTranslate));
|
|
477
|
+
const validated = {};
|
|
478
|
+
for (const [key, value] of Object.entries(parsed)) {
|
|
479
|
+
if (expectedKeys.has(key) && typeof value === 'string' && !isUnsafeKey(key)) {
|
|
480
|
+
validated[key] = value;
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
return Object.keys(validated).length > 0 ? validated : null;
|
|
484
|
+
} catch (err) {
|
|
485
|
+
output.error(`${label} Batch ${batchNum}: JSON parse error — ${err.message}`);
|
|
486
|
+
return { _parseError: true, rawContent: content, error: err.message };
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
// ── Shared direct call — retry loop with exponential backoff ──
|
|
491
|
+
|
|
492
|
+
async _callProviderDirect({
|
|
493
|
+
prompt,
|
|
494
|
+
systemMessage,
|
|
495
|
+
apiKey,
|
|
496
|
+
model,
|
|
497
|
+
temperature = DEFAULT_TEMPERATURE,
|
|
498
|
+
timeoutMs = REQUEST_TIMEOUT_MS,
|
|
499
|
+
label,
|
|
500
|
+
isJsonMode = false,
|
|
501
|
+
}) {
|
|
502
|
+
label = label || this._getProviderLabel();
|
|
503
|
+
|
|
504
|
+
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
|
|
505
|
+
try {
|
|
506
|
+
const controller = new AbortController();
|
|
507
|
+
const timeoutId = setTimeout(() => controller.abort(), timeoutMs);
|
|
508
|
+
|
|
509
|
+
const { url, headers, body } = this._buildApiRequest({
|
|
510
|
+
prompt,
|
|
511
|
+
systemMessage,
|
|
512
|
+
apiKey,
|
|
513
|
+
model,
|
|
514
|
+
temperature,
|
|
515
|
+
isJsonMode,
|
|
516
|
+
});
|
|
517
|
+
|
|
518
|
+
const response = await fetch(url, {
|
|
519
|
+
method: 'POST',
|
|
520
|
+
headers,
|
|
521
|
+
body: JSON.stringify(body),
|
|
522
|
+
signal: controller.signal,
|
|
523
|
+
});
|
|
524
|
+
|
|
525
|
+
clearTimeout(timeoutId);
|
|
526
|
+
|
|
527
|
+
if (isRetryable(response.status)) {
|
|
528
|
+
if (attempt < MAX_RETRIES) {
|
|
529
|
+
const delay = getBackoffDelay(attempt);
|
|
530
|
+
output.warn(`⏳ ${label}: ${response.status} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
531
|
+
await sleep(delay);
|
|
532
|
+
continue;
|
|
533
|
+
}
|
|
534
|
+
output.error(`${label}: ${response.status} after ${MAX_RETRIES + 1} attempts`);
|
|
535
|
+
recordTranslationError(response.status);
|
|
536
|
+
return null;
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
if (!response.ok) {
|
|
540
|
+
const errorBody = await response.text();
|
|
541
|
+
output.error(`${label}: API error ${response.status} — ${errorBody}`);
|
|
542
|
+
recordTranslationError(response.status);
|
|
543
|
+
return null;
|
|
544
|
+
}
|
|
545
|
+
|
|
546
|
+
const data = await response.json();
|
|
547
|
+
const content = this._extractResponseText(data);
|
|
548
|
+
if (!content) {
|
|
549
|
+
// Empty responses — retry with backoff, same as HTTP errors.
|
|
550
|
+
// Common with long content where the model hits output limits.
|
|
551
|
+
if (attempt < MAX_RETRIES) {
|
|
552
|
+
const delay = getBackoffDelay(attempt);
|
|
553
|
+
output.warn(`⏳ ${label}: empty response — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
554
|
+
await sleep(delay);
|
|
555
|
+
continue;
|
|
556
|
+
}
|
|
557
|
+
output.error(`${label}: empty response after ${MAX_RETRIES + 1} attempts`);
|
|
558
|
+
return null;
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
return stripCodeFences(content.trim());
|
|
562
|
+
|
|
563
|
+
} catch (err) {
|
|
564
|
+
const isTimeout = err.name === 'AbortError';
|
|
565
|
+
const errLabel = isTimeout ? 'timeout' : err.message;
|
|
566
|
+
|
|
567
|
+
if (attempt < MAX_RETRIES) {
|
|
568
|
+
const delay = getBackoffDelay(attempt);
|
|
569
|
+
output.warn(`⏳ ${label}: ${errLabel} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
570
|
+
await sleep(delay);
|
|
571
|
+
continue;
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
output.error(`${label} failed: ${errLabel}`);
|
|
575
|
+
return null;
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
return null;
|
|
579
|
+
}
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
export {
|
|
583
|
+
DirectLLMMethod,
|
|
584
|
+
MODEL_PATTERNS,
|
|
585
|
+
_modelListCache,
|
|
586
|
+
};
|