champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenRouter API Client — shared HTTP client for all LLM-based methods.
|
|
3
|
+
*
|
|
4
|
+
* Centralizes the OpenRouter chat-completion call pattern that was
|
|
5
|
+
* previously duplicated across llm.js, llm-coached.js, and their
|
|
6
|
+
* translateContent variants. Each call site only needed to differ in:
|
|
7
|
+
* - prompt text
|
|
8
|
+
* - temperature
|
|
9
|
+
* - timeout multiplier
|
|
10
|
+
* - log label
|
|
11
|
+
*
|
|
12
|
+
* This module provides two functions:
|
|
13
|
+
* callOpenRouter() — raw text response (for content translation)
|
|
14
|
+
* callOpenRouterJSON() — parsed + validated JSON (for key-value translation)
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import {
|
|
18
|
+
MAX_RETRIES, REQUEST_TIMEOUT_MS,
|
|
19
|
+
isRetryable, getBackoffDelay, sleep, stripCodeFences,
|
|
20
|
+
} from './http-utils.js';
|
|
21
|
+
import { DEFAULT_TEMPERATURE } from '../config.js';
|
|
22
|
+
import { output } from '../output.js';
|
|
23
|
+
import { recordTranslationError } from './translation-error.js';
|
|
24
|
+
|
|
25
|
+
const OPENROUTER_URL = 'https://openrouter.ai/api/v1/chat/completions';
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Call OpenRouter's chat completion API with retry and backoff.
|
|
29
|
+
*
|
|
30
|
+
* Returns the raw text content from the model, with code fences stripped.
|
|
31
|
+
* Returns null on permanent failure (non-retryable error or exhausted retries).
|
|
32
|
+
*
|
|
33
|
+
* @param {object} options
|
|
34
|
+
* @param {string} options.prompt - The user message content
|
|
35
|
+
* @param {string} options.apiKey - OpenRouter API key
|
|
36
|
+
* @param {string} options.model - Model identifier (e.g. 'openai/gpt-4o-mini')
|
|
37
|
+
* @param {number} [options.temperature=0.3] - Sampling temperature
|
|
38
|
+
* @param {number} [options.timeoutMs] - Override timeout (default: REQUEST_TIMEOUT_MS)
|
|
39
|
+
* @param {string} [options.label='LLM'] - Label for log messages (e.g. 'Batch 3', 'Coached batch 1')
|
|
40
|
+
* @param {string} [options.xTitle='champollion'] - X-Title header value
|
|
41
|
+
* @param {string} [options.systemMessage] - Optional system message for prompt caching.
|
|
42
|
+
* When provided, the system preamble (register, rules) is sent as a separate
|
|
43
|
+
* role:system message, enabling provider-level prompt caching across batches
|
|
44
|
+
* that share the same preamble for a given locale. The prompt parameter then
|
|
45
|
+
* contains only the batch-specific payload (JSON key-value pairs).
|
|
46
|
+
* @returns {Promise<string|null>} Model response text, or null on failure
|
|
47
|
+
*/
|
|
48
|
+
async function callOpenRouter({
|
|
49
|
+
prompt,
|
|
50
|
+
apiKey,
|
|
51
|
+
model,
|
|
52
|
+
temperature = DEFAULT_TEMPERATURE,
|
|
53
|
+
timeoutMs = REQUEST_TIMEOUT_MS,
|
|
54
|
+
label = 'LLM',
|
|
55
|
+
xTitle = 'champollion',
|
|
56
|
+
systemMessage = null,
|
|
57
|
+
}) {
|
|
58
|
+
for (let attempt = 0; attempt <= MAX_RETRIES; attempt++) {
|
|
59
|
+
try {
|
|
60
|
+
const controller = new AbortController();
|
|
61
|
+
const timeoutId = setTimeout(() => controller.abort(), timeoutMs);
|
|
62
|
+
|
|
63
|
+
// Build messages array — when systemMessage is provided, split preamble
|
|
64
|
+
// from payload to enable provider-level prompt caching. The system message
|
|
65
|
+
// (register + rules) is identical across batches for a given locale, so
|
|
66
|
+
// providers like Anthropic and Google cache it automatically.
|
|
67
|
+
const messages = systemMessage
|
|
68
|
+
? [{ role: 'system', content: systemMessage }, { role: 'user', content: prompt }]
|
|
69
|
+
: [{ role: 'user', content: prompt }];
|
|
70
|
+
|
|
71
|
+
const response = await fetch(OPENROUTER_URL, {
|
|
72
|
+
method: 'POST',
|
|
73
|
+
headers: {
|
|
74
|
+
'Authorization': `Bearer ${apiKey}`,
|
|
75
|
+
'Content-Type': 'application/json',
|
|
76
|
+
'HTTP-Referer': 'https://github.com/gamedaysuits/Champollion',
|
|
77
|
+
'X-Title': xTitle,
|
|
78
|
+
},
|
|
79
|
+
body: JSON.stringify({
|
|
80
|
+
model,
|
|
81
|
+
messages,
|
|
82
|
+
temperature,
|
|
83
|
+
}),
|
|
84
|
+
signal: controller.signal,
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
clearTimeout(timeoutId);
|
|
88
|
+
|
|
89
|
+
// Retryable status (429, 5xx) — back off and try again
|
|
90
|
+
if (isRetryable(response.status)) {
|
|
91
|
+
if (attempt < MAX_RETRIES) {
|
|
92
|
+
const delay = getBackoffDelay(attempt);
|
|
93
|
+
output.warn(`⏳ ${label}: ${response.status} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
94
|
+
await sleep(delay);
|
|
95
|
+
continue;
|
|
96
|
+
}
|
|
97
|
+
output.error(`${label}: ${response.status} after ${MAX_RETRIES + 1} attempts`);
|
|
98
|
+
recordTranslationError(response.status);
|
|
99
|
+
return null;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
if (!response.ok) {
|
|
103
|
+
output.error(`${label}: API error ${response.status}`);
|
|
104
|
+
recordTranslationError(response.status);
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const data = await response.json();
|
|
109
|
+
const content = data.choices?.[0]?.message?.content?.trim();
|
|
110
|
+
if (!content) {
|
|
111
|
+
// Empty responses are common with some models (especially for long
|
|
112
|
+
// markdown). Retry with backoff — same as HTTP errors — instead of
|
|
113
|
+
// giving up immediately and returning null (which skips the key).
|
|
114
|
+
if (attempt < MAX_RETRIES) {
|
|
115
|
+
const delay = getBackoffDelay(attempt);
|
|
116
|
+
output.warn(`⏳ ${label}: empty response — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
117
|
+
await sleep(delay);
|
|
118
|
+
continue;
|
|
119
|
+
}
|
|
120
|
+
output.error(`${label}: empty response after ${MAX_RETRIES + 1} attempts`);
|
|
121
|
+
return null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// Strip markdown code fences if the model wraps its response
|
|
125
|
+
return stripCodeFences(content);
|
|
126
|
+
|
|
127
|
+
} catch (err) {
|
|
128
|
+
const isTimeout = err.name === 'AbortError';
|
|
129
|
+
const errLabel = isTimeout ? 'timeout' : err.message;
|
|
130
|
+
|
|
131
|
+
if (attempt < MAX_RETRIES) {
|
|
132
|
+
const delay = getBackoffDelay(attempt);
|
|
133
|
+
output.warn(`⏳ ${label}: ${errLabel} — retry ${attempt + 1}/${MAX_RETRIES} in ${Math.round(delay / 1000)}s...`);
|
|
134
|
+
await sleep(delay);
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
output.error(`${label} failed: ${errLabel}`);
|
|
139
|
+
return null;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
return null;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Call OpenRouter and parse the response as validated JSON key-value pairs.
|
|
148
|
+
*
|
|
149
|
+
* This is the standard pattern for key-value batch translation:
|
|
150
|
+
* 1. Send prompt to model
|
|
151
|
+
* 2. Parse JSON response
|
|
152
|
+
* 3. Validate: only accept keys from expectedKeys set
|
|
153
|
+
* 4. Block prototype-pollution keys
|
|
154
|
+
*
|
|
155
|
+
* @param {object} options - Same as callOpenRouter, plus:
|
|
156
|
+
* @param {Set<string>} options.expectedKeys - Set of keys to accept in the response
|
|
157
|
+
* @param {function} options.isUnsafeKey - Function to check for unsafe key segments
|
|
158
|
+
* @returns {Promise<object|null>} Validated key-value map, or null on failure
|
|
159
|
+
*/
|
|
160
|
+
async function callOpenRouterJSON({ expectedKeys, isUnsafeKey, ...callOptions }) {
|
|
161
|
+
const content = await callOpenRouter(callOptions);
|
|
162
|
+
if (!content) return null;
|
|
163
|
+
|
|
164
|
+
let parsed;
|
|
165
|
+
try {
|
|
166
|
+
parsed = JSON.parse(content);
|
|
167
|
+
} catch (err) {
|
|
168
|
+
// Models translating quoted text (e.g. `never “not a language.”`) often
|
|
169
|
+
// downgrade typographic quotes to unescaped ASCII quotes inside the JSON
|
|
170
|
+
// string value, which kills the parse. At temperature 0 a blind retry
|
|
171
|
+
// returns byte-identical output, so the cascade can never recover —
|
|
172
|
+
// repair the response instead of rejecting it.
|
|
173
|
+
try {
|
|
174
|
+
parsed = JSON.parse(repairUnescapedQuotes(content));
|
|
175
|
+
output.warn(`${callOptions.label || 'LLM'}: repaired unescaped quotes in model JSON.`);
|
|
176
|
+
} catch {
|
|
177
|
+
// Second recovery tier: the response keys are known exactly (models
|
|
178
|
+
// keep keys as-is per the prompt rules), so values can be extracted
|
|
179
|
+
// by anchoring on the key tokens even when quote damage confuses the
|
|
180
|
+
// in-string heuristic above (e.g. an inner quote directly before a
|
|
181
|
+
// comma: `... "geen taal", zei ...`).
|
|
182
|
+
parsed = extractByExpectedKeys(content, expectedKeys);
|
|
183
|
+
if (parsed) {
|
|
184
|
+
output.warn(`${callOptions.label || 'LLM'}: recovered ${Object.keys(parsed).length} value(s) from malformed model JSON via key anchors.`);
|
|
185
|
+
} else {
|
|
186
|
+
// Return structured error instead of null — the retry cascade needs to
|
|
187
|
+
// distinguish "API returned nothing" (null) from "API returned garbage
|
|
188
|
+
// JSON that we can retry with a smaller batch" (_parseError).
|
|
189
|
+
output.error(`${callOptions.label || 'LLM'}: JSON parse error — ${err.message}`);
|
|
190
|
+
return { _parseError: true, rawContent: content, error: err.message };
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// Validate: only accept keys we sent, block prototype pollution
|
|
196
|
+
const validated = {};
|
|
197
|
+
for (const [key, value] of Object.entries(parsed)) {
|
|
198
|
+
if (expectedKeys.has(key) && typeof value === 'string' && !isUnsafeKey(key)) {
|
|
199
|
+
validated[key] = value;
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
return Object.keys(validated).length > 0 ? validated : null;
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Escape unescaped ASCII double quotes inside JSON string values.
|
|
208
|
+
*
|
|
209
|
+
* Walks the text tracking in-string state. Inside a string, a `"` only
|
|
210
|
+
* terminates it when the next non-whitespace character is a JSON structural
|
|
211
|
+
* character (`:`, `,`, `}`, `]`) or end-of-input; any other `"` is an inner
|
|
212
|
+
* quote the model forgot to escape and becomes `\"`.
|
|
213
|
+
*
|
|
214
|
+
* The heuristic can misread an inner quote that directly precedes a comma
|
|
215
|
+
* (e.g. a value containing `"yes",`), in which case the repaired text still
|
|
216
|
+
* fails to parse and the caller falls back to the retry cascade — the repair
|
|
217
|
+
* never makes a previously-parseable response worse because it only runs
|
|
218
|
+
* after JSON.parse has already failed.
|
|
219
|
+
*
|
|
220
|
+
* @param {string} text - Raw model output that failed JSON.parse
|
|
221
|
+
* @returns {string} Text with inner quotes escaped
|
|
222
|
+
*/
|
|
223
|
+
function repairUnescapedQuotes(text) {
|
|
224
|
+
let out = '';
|
|
225
|
+
let inString = false;
|
|
226
|
+
for (let i = 0; i < text.length; i++) {
|
|
227
|
+
const ch = text[i];
|
|
228
|
+
if (!inString) {
|
|
229
|
+
if (ch === '"') inString = true;
|
|
230
|
+
out += ch;
|
|
231
|
+
continue;
|
|
232
|
+
}
|
|
233
|
+
if (ch === '\\') {
|
|
234
|
+
// Preserve existing escape pairs untouched
|
|
235
|
+
out += ch + (text[i + 1] ?? '');
|
|
236
|
+
i++;
|
|
237
|
+
continue;
|
|
238
|
+
}
|
|
239
|
+
if (ch === '"') {
|
|
240
|
+
let j = i + 1;
|
|
241
|
+
while (j < text.length && /\s/.test(text[j])) j++;
|
|
242
|
+
const next = text[j];
|
|
243
|
+
if (next === undefined || next === ':' || next === ',' || next === '}' || next === ']') {
|
|
244
|
+
inString = false;
|
|
245
|
+
out += ch;
|
|
246
|
+
} else {
|
|
247
|
+
out += '\\"';
|
|
248
|
+
}
|
|
249
|
+
continue;
|
|
250
|
+
}
|
|
251
|
+
out += ch;
|
|
252
|
+
}
|
|
253
|
+
return out;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Extract string values from a malformed JSON response by anchoring on the
|
|
258
|
+
* known key set.
|
|
259
|
+
*
|
|
260
|
+
* The prompt instructs models to keep keys exactly as sent, so every key in
|
|
261
|
+
* the response appears as a literal `"<key>"` token. The value for a key is
|
|
262
|
+
* the text between the quote that opens its value and the last quote before
|
|
263
|
+
* the next key's token (or the end of the response). This survives arbitrary
|
|
264
|
+
* unescaped-quote damage inside values — the failure mode that defeats both
|
|
265
|
+
* JSON.parse and the in-string repair heuristic.
|
|
266
|
+
*
|
|
267
|
+
* Only used as the last recovery tier after JSON.parse and
|
|
268
|
+
* repairUnescapedQuotes have both failed; extracted values still pass
|
|
269
|
+
* through the caller's key validation and the downstream quality gate.
|
|
270
|
+
*
|
|
271
|
+
* @param {string} text - Raw model output that failed to parse
|
|
272
|
+
* @param {Set<string>} expectedKeys - Keys that were sent in the batch
|
|
273
|
+
* @returns {object|null} Recovered key→value map, or null if nothing anchored
|
|
274
|
+
*/
|
|
275
|
+
function extractByExpectedKeys(text, expectedKeys) {
|
|
276
|
+
if (!expectedKeys || expectedKeys.size === 0) return null;
|
|
277
|
+
|
|
278
|
+
const anchors = [];
|
|
279
|
+
for (const key of expectedKeys) {
|
|
280
|
+
const token = JSON.stringify(key);
|
|
281
|
+
const idx = text.indexOf(token);
|
|
282
|
+
if (idx >= 0) anchors.push({ key, idx, token });
|
|
283
|
+
}
|
|
284
|
+
if (anchors.length === 0) return null;
|
|
285
|
+
anchors.sort((a, b) => a.idx - b.idx);
|
|
286
|
+
|
|
287
|
+
const result = {};
|
|
288
|
+
for (let i = 0; i < anchors.length; i++) {
|
|
289
|
+
const { key, idx, token } = anchors[i];
|
|
290
|
+
|
|
291
|
+
// Walk past `"<key>"` + whitespace + `:` + whitespace + opening `"`
|
|
292
|
+
let p = idx + token.length;
|
|
293
|
+
while (p < text.length && /\s/.test(text[p])) p++;
|
|
294
|
+
if (text[p] !== ':') continue;
|
|
295
|
+
p++;
|
|
296
|
+
while (p < text.length && /\s/.test(text[p])) p++;
|
|
297
|
+
if (text[p] !== '"') continue;
|
|
298
|
+
|
|
299
|
+
const valStart = p + 1;
|
|
300
|
+
const endLimit = i + 1 < anchors.length ? anchors[i + 1].idx : text.length;
|
|
301
|
+
const segment = text.slice(valStart, endLimit);
|
|
302
|
+
const lastQuote = segment.lastIndexOf('"');
|
|
303
|
+
if (lastQuote < 0) continue;
|
|
304
|
+
const raw = segment.slice(0, lastQuote);
|
|
305
|
+
|
|
306
|
+
// Normalize quotes to escaped form (both already-escaped and bare), keep
|
|
307
|
+
// every other escape sequence for JSON to decode, then parse as a single
|
|
308
|
+
// JSON string. A value that still fails to decode is skipped — the
|
|
309
|
+
// retry cascade handles it. U+0001 is a safe sentinel: raw control
|
|
310
|
+
// characters are invalid inside JSON strings, so it cannot occur in
|
|
311
|
+
// model output that reached us through the API's own JSON layer.
|
|
312
|
+
const SENTINEL = '\u0001';
|
|
313
|
+
const normalized = raw
|
|
314
|
+
.replace(/\\"/g, SENTINEL)
|
|
315
|
+
.replace(/"/g, '\\"')
|
|
316
|
+
.split(SENTINEL).join('\\"');
|
|
317
|
+
try {
|
|
318
|
+
result[key] = JSON.parse(`"${normalized}"`);
|
|
319
|
+
} catch {
|
|
320
|
+
continue;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
return Object.keys(result).length > 0 ? result : null;
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
export { callOpenRouter, callOpenRouterJSON, repairUnescapedQuotes, extractByExpectedKeys, OPENROUTER_URL };
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenRouter Pricing — fetch live model pricing for cost estimation.
|
|
3
|
+
*
|
|
4
|
+
* WHY: The LLM and LLM-Coached methods use OpenRouter, which aggregates
|
|
5
|
+
* 100+ models, each with different pricing. We can't hardcode rates.
|
|
6
|
+
* OpenRouter provides a public `/api/v1/models` endpoint with per-token
|
|
7
|
+
* pricing for every model — no auth required.
|
|
8
|
+
*
|
|
9
|
+
* HOW IT WORKS:
|
|
10
|
+
* 1. Fetches the model list from OpenRouter (cached for the process lifetime)
|
|
11
|
+
* 2. Looks up the specific model's input/output pricing
|
|
12
|
+
* 3. Estimates cost based on average tokens per key
|
|
13
|
+
*
|
|
14
|
+
* CACHE: Pricing is fetched once per process and cached in memory.
|
|
15
|
+
* This avoids hammering the API during large multi-pair syncs.
|
|
16
|
+
*
|
|
17
|
+
* FALLBACK: If the fetch fails (offline, rate-limited, etc.), returns
|
|
18
|
+
* null pricing — the cost table will show "unknown" for that method.
|
|
19
|
+
* This never blocks a sync.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { EST_INPUT_TOKENS_PER_KEY, EST_OUTPUT_TOKENS_PER_KEY } from '../config.js';
|
|
23
|
+
|
|
24
|
+
const OPENROUTER_MODELS_URL = 'https://openrouter.ai/api/v1/models';
|
|
25
|
+
|
|
26
|
+
// Coached methods inject grammar rules and dictionary matches into the
|
|
27
|
+
// prompt — roughly 2.5x the input tokens of a standard prompt.
|
|
28
|
+
const COACHED_INPUT_MULTIPLIER = 2.5;
|
|
29
|
+
|
|
30
|
+
// In-memory cache: fetched once per process
|
|
31
|
+
let _pricingCache = null;
|
|
32
|
+
let _pricingFetchPromise = null;
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Fetch pricing for all OpenRouter models.
|
|
36
|
+
*
|
|
37
|
+
* Returns a Map of model ID → { input, output } (cost per token in USD).
|
|
38
|
+
* Caches the result for the lifetime of the process.
|
|
39
|
+
*
|
|
40
|
+
* @returns {Promise<Map<string, {input: number, output: number}>>}
|
|
41
|
+
*/
|
|
42
|
+
async function fetchModelPricing() {
|
|
43
|
+
// Return cached result if available
|
|
44
|
+
if (_pricingCache) return _pricingCache;
|
|
45
|
+
|
|
46
|
+
// Deduplicate concurrent fetches — if one is in flight, share the promise
|
|
47
|
+
if (_pricingFetchPromise) return _pricingFetchPromise;
|
|
48
|
+
|
|
49
|
+
_pricingFetchPromise = _doFetch();
|
|
50
|
+
try {
|
|
51
|
+
_pricingCache = await _pricingFetchPromise;
|
|
52
|
+
return _pricingCache;
|
|
53
|
+
} finally {
|
|
54
|
+
_pricingFetchPromise = null;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
async function _doFetch() {
|
|
59
|
+
try {
|
|
60
|
+
const controller = new AbortController();
|
|
61
|
+
const timeoutId = setTimeout(() => controller.abort(), 5000);
|
|
62
|
+
|
|
63
|
+
const response = await fetch(OPENROUTER_MODELS_URL, {
|
|
64
|
+
headers: { 'User-Agent': 'champollion' },
|
|
65
|
+
signal: controller.signal,
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
clearTimeout(timeoutId);
|
|
69
|
+
|
|
70
|
+
if (!response.ok) return new Map();
|
|
71
|
+
|
|
72
|
+
const json = await response.json();
|
|
73
|
+
const models = json.data || [];
|
|
74
|
+
const pricing = new Map();
|
|
75
|
+
|
|
76
|
+
for (const model of models) {
|
|
77
|
+
if (model.id && model.pricing) {
|
|
78
|
+
pricing.set(model.id, {
|
|
79
|
+
input: parseFloat(model.pricing.prompt) || 0,
|
|
80
|
+
output: parseFloat(model.pricing.completion) || 0,
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
return pricing;
|
|
86
|
+
} catch {
|
|
87
|
+
// Offline, timeout, or API issue — return empty map
|
|
88
|
+
// Cost estimation degrades gracefully to "unknown"
|
|
89
|
+
return new Map();
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Estimate cost for translating N keys with a specific model via OpenRouter.
|
|
95
|
+
*
|
|
96
|
+
* Uses real pricing from the OpenRouter models API when available.
|
|
97
|
+
* Falls back to a rough estimate when the model isn't found.
|
|
98
|
+
*
|
|
99
|
+
* Token estimation uses the shared EST_*_TOKENS_PER_KEY constants from
|
|
100
|
+
* config.js (~200 in / ~30 out — see the rationale there). They are the
|
|
101
|
+
* single source of truth for per-key token heuristics: this module used to
|
|
102
|
+
* hardcode 200/30 while config.js exported 60/10, so estimates disagreed
|
|
103
|
+
* between OpenRouter-routed and direct-provider engines.
|
|
104
|
+
* Coached methods use a 2.5x multiplier on input due to grammar/dictionary injection.
|
|
105
|
+
*
|
|
106
|
+
* @param {number} keyCount - Number of keys to translate
|
|
107
|
+
* @param {string} model - OpenRouter model ID (e.g., 'openai/gpt-4o-mini')
|
|
108
|
+
* @param {object} options
|
|
109
|
+
* @param {boolean} [options.coached=false] - Whether this is a coached method (larger prompts)
|
|
110
|
+
* @returns {Promise<{estimatedCost: number|null, currency: string, source: string, note: string}>}
|
|
111
|
+
*/
|
|
112
|
+
async function estimateOpenRouterCost(keyCount, model, options = {}) {
|
|
113
|
+
const coached = options.coached || false;
|
|
114
|
+
const pricing = await fetchModelPricing();
|
|
115
|
+
|
|
116
|
+
const modelPricing = pricing.get(model);
|
|
117
|
+
if (!modelPricing) {
|
|
118
|
+
return {
|
|
119
|
+
estimatedCost: null,
|
|
120
|
+
currency: 'USD',
|
|
121
|
+
source: 'unknown',
|
|
122
|
+
note: `Model "${model}" not found in OpenRouter pricing. Cost cannot be estimated.`,
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
// Token estimation: amortized system message + per-key payload,
|
|
127
|
+
// from the shared config.js constants (SSOT for all engines).
|
|
128
|
+
const inputTokensPerKey = coached
|
|
129
|
+
? Math.round(EST_INPUT_TOKENS_PER_KEY * COACHED_INPUT_MULTIPLIER)
|
|
130
|
+
: EST_INPUT_TOKENS_PER_KEY;
|
|
131
|
+
const outputTokensPerKey = EST_OUTPUT_TOKENS_PER_KEY;
|
|
132
|
+
|
|
133
|
+
const totalInputTokens = keyCount * inputTokensPerKey;
|
|
134
|
+
const totalOutputTokens = keyCount * outputTokensPerKey;
|
|
135
|
+
|
|
136
|
+
const inputCost = totalInputTokens * modelPricing.input;
|
|
137
|
+
const outputCost = totalOutputTokens * modelPricing.output;
|
|
138
|
+
const totalCost = inputCost + outputCost;
|
|
139
|
+
|
|
140
|
+
return {
|
|
141
|
+
estimatedCost: Math.round(totalCost * 10000) / 10000,
|
|
142
|
+
currency: 'USD',
|
|
143
|
+
source: `openrouter (${model})`,
|
|
144
|
+
note: `Based on ${model} pricing: $${modelPricing.input}/tok in, $${modelPricing.output}/tok out.`,
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Clear the pricing cache. Useful for testing.
|
|
150
|
+
*/
|
|
151
|
+
function clearPricingCache() {
|
|
152
|
+
_pricingCache = null;
|
|
153
|
+
_pricingFetchPromise = null;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export { fetchModelPricing, estimateOpenRouterCost, clearPricingCache };
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider environment-variable SSOT.
|
|
3
|
+
*
|
|
4
|
+
* WHY: The translation method *loader* (`_resolveApiKey`), the preflight
|
|
5
|
+
* *readiness* check (`checkReadiness`), the `doctor` command, and the setup
|
|
6
|
+
* *help* text each used to hardcode their own env-var names. They drifted —
|
|
7
|
+
* e.g. the Microsoft loader read only MICROSOFT_TRANSLATOR_API_KEY while
|
|
8
|
+
* `doctor` checked only AZURE_TRANSLATOR_KEY. A user who set
|
|
9
|
+
* AZURE_TRANSLATOR_KEY got a green `doctor` ("✓ ready") but the loader
|
|
10
|
+
* returned null at run time and the sync silently produced nothing.
|
|
11
|
+
*
|
|
12
|
+
* This module is the single source of truth: one canonical name per provider
|
|
13
|
+
* plus the aliases the loader accepts. Every surface resolves keys through
|
|
14
|
+
* here, so `doctor` can never report a key "ready" under a name the loader
|
|
15
|
+
* won't read, and never report "not set" for a name the loader *would* read.
|
|
16
|
+
*
|
|
17
|
+
* The loader reads aliases too (so existing users with either name keep
|
|
18
|
+
* working); the canonical name is what we print in help/setup guidance.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { getEnvOrFileVar } from '../api-key.js';
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* provider method name → { canonical, aliases, kind }
|
|
25
|
+
*
|
|
26
|
+
* `kind: 'url'` marks an endpoint URL (LibreTranslate) rather than a secret
|
|
27
|
+
* key — callers that only care about secrets can skip those.
|
|
28
|
+
*
|
|
29
|
+
* Keep these in sync with the harness loaders in
|
|
30
|
+
* arena/mt_eval_harness/methods/*.py (which already accept the same aliases
|
|
31
|
+
* via `_env_first(canonical, alias)`).
|
|
32
|
+
*/
|
|
33
|
+
export const PROVIDER_ENV = {
|
|
34
|
+
'deepl': {
|
|
35
|
+
canonical: 'DEEPL_API_KEY',
|
|
36
|
+
aliases: [],
|
|
37
|
+
},
|
|
38
|
+
'google-translate': {
|
|
39
|
+
canonical: 'GOOGLE_TRANSLATE_API_KEY',
|
|
40
|
+
aliases: ['GOOGLE_API_KEY'],
|
|
41
|
+
},
|
|
42
|
+
'microsoft-translator': {
|
|
43
|
+
canonical: 'MICROSOFT_TRANSLATOR_API_KEY',
|
|
44
|
+
aliases: ['AZURE_TRANSLATOR_KEY'],
|
|
45
|
+
},
|
|
46
|
+
'libretranslate': {
|
|
47
|
+
canonical: 'LIBRETRANSLATE_API_URL',
|
|
48
|
+
aliases: ['LIBRETRANSLATE_URL', 'LIBRE_URL'],
|
|
49
|
+
kind: 'url',
|
|
50
|
+
},
|
|
51
|
+
'tilde': {
|
|
52
|
+
canonical: 'TILDE_API_KEY',
|
|
53
|
+
aliases: [],
|
|
54
|
+
},
|
|
55
|
+
// Lara authenticates with a key PAIR; the adapter's readiness probe checks
|
|
56
|
+
// the secret (LARA_ACCESS_KEY_SECRET) alongside this canonical id.
|
|
57
|
+
'translated': {
|
|
58
|
+
canonical: 'LARA_ACCESS_KEY_ID',
|
|
59
|
+
aliases: [],
|
|
60
|
+
},
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Ordered list of env var names for a provider: [canonical, ...aliases].
|
|
65
|
+
* Empty array for an unknown provider.
|
|
66
|
+
*
|
|
67
|
+
* @param {string} method - Provider method name (e.g. 'google-translate')
|
|
68
|
+
* @returns {string[]}
|
|
69
|
+
*/
|
|
70
|
+
export function providerEnvNames(method) {
|
|
71
|
+
const entry = PROVIDER_ENV[method];
|
|
72
|
+
return entry ? [entry.canonical, ...entry.aliases] : [];
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* The canonical env var name for a provider (what help/setup text should show).
|
|
77
|
+
*
|
|
78
|
+
* @param {string} method - Provider method name
|
|
79
|
+
* @returns {string|null}
|
|
80
|
+
*/
|
|
81
|
+
export function canonicalEnvName(method) {
|
|
82
|
+
return PROVIDER_ENV[method]?.canonical ?? null;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Resolve a provider's key by checking the canonical name then each alias,
|
|
87
|
+
* across process.env AND .env.local/.env (via getEnvOrFileVar) — exactly the
|
|
88
|
+
* surface the method loaders read. This is what readiness/doctor should use
|
|
89
|
+
* so they agree with the loader.
|
|
90
|
+
*
|
|
91
|
+
* @param {string} method - Provider method name
|
|
92
|
+
* @param {string} [cwd] - Project root for .env file lookup
|
|
93
|
+
* @returns {{ value: string|null, name: string|null }}
|
|
94
|
+
* value: the resolved key/url, or null. name: the env var it came from.
|
|
95
|
+
*/
|
|
96
|
+
export function resolveProviderEnv(method, cwd) {
|
|
97
|
+
for (const name of providerEnvNames(method)) {
|
|
98
|
+
const value = getEnvOrFileVar(name, cwd);
|
|
99
|
+
if (value) return { value, name };
|
|
100
|
+
}
|
|
101
|
+
return { value: null, name: null };
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Human-readable "NAME (or ALIAS)" string for messages.
|
|
106
|
+
*
|
|
107
|
+
* @param {string} method - Provider method name
|
|
108
|
+
* @returns {string}
|
|
109
|
+
*/
|
|
110
|
+
export function envNamesLabel(method) {
|
|
111
|
+
const names = providerEnvNames(method);
|
|
112
|
+
if (names.length === 0) return '';
|
|
113
|
+
if (names.length === 1) return names[0];
|
|
114
|
+
return `${names[0]} (or ${names.slice(1).join(', ')})`;
|
|
115
|
+
}
|