champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
@@ -0,0 +1,310 @@
1
+ /**
2
+ * Provider API Pricing — Static rates for non-LLM translation APIs.
3
+ *
4
+ * WHY STATIC: Unlike OpenRouter (which exposes /api/v1/models with live
5
+ * per-token pricing), Google, DeepL, and Microsoft do not have pricing
6
+ * APIs. These rates must be hardcoded and manually verified.
7
+ *
8
+ * WHY THIS MODULE: Four method files each had their own hardcoded rate.
9
+ * DeepL was wrong ($20 instead of $25). Single source of truth prevents
10
+ * that class of bug.
11
+ *
12
+ * LAST VERIFIED: 2026-06-08
13
+ * Sources:
14
+ * Google Cloud Translation v2: https://cloud.google.com/translate/pricing
15
+ * DeepL API Pro: https://www.deepl.com/pro-api
16
+ * Microsoft Translator S1: https://azure.microsoft.com/pricing/details/cognitive-services/translator/
17
+ */
18
+
19
+ import {
20
+ EST_CHARS_PER_KEY,
21
+ EST_INPUT_TOKENS_PER_KEY,
22
+ EST_OUTPUT_TOKENS_PER_KEY,
23
+ } from '../config.js';
24
+ import { fetchModelPricing } from './openrouter-pricing.js';
25
+
26
+ /**
27
+ * Per-provider pricing data.
28
+ *
29
+ * Each entry has:
30
+ * costPerMillionChars — USD cost per 1M characters translated
31
+ * source — pricing page identifier (for LAST VERIFIED audit trail)
32
+ * note — human-readable pricing context
33
+ */
34
+ export const PROVIDER_RATES = {
35
+ 'google-translate': {
36
+ costPerMillionChars: 20,
37
+ source: 'google-cloud-pricing',
38
+ note: 'Google Cloud Translation API v2 ($20/1M chars).',
39
+ },
40
+ 'deepl': {
41
+ costPerMillionChars: 25,
42
+ source: 'deepl-api-pro-pricing',
43
+ note: 'DeepL API Pro (~$25/1M chars). Free tier: 500K chars/month cap.',
44
+ },
45
+ 'microsoft-translator': {
46
+ costPerMillionChars: 10,
47
+ source: 'microsoft-translator-pricing',
48
+ note: 'Azure Translator S1 ($10/1M chars). 2M chars/month free tier.',
49
+ },
50
+ 'libretranslate': {
51
+ costPerMillionChars: 0,
52
+ source: 'libretranslate-self-hosted',
53
+ note: 'Self-hosted, free. Infrastructure costs not included.',
54
+ },
55
+ };
56
+
57
+ /**
58
+ * Per-token pricing for the DIRECT LLM providers (anthropic / openai / gemini),
59
+ * in USD per 1M tokens, keyed by EXACT model ID.
60
+ *
61
+ * WHY EXACT IDS, NOT SUBSTRINGS: each provider file used to guess by substring
62
+ * (`model.includes('opus')` → $15/$75). That silently mispriced two ways —
63
+ * an unrecognized model got the default tier's rate as though it were known,
64
+ * and a recognized-but-repriced model kept a rate that no longer existed.
65
+ * Anthropic's `opus` branch was still charging Claude 3 Opus rates ($15/$75)
66
+ * for an Opus generation that costs $5/$25 — a 3x overestimate presented to
67
+ * the user as fact. An exact-ID table cannot drift that way: a model is either
68
+ * priced from a verified row, or reported as UNKNOWN.
69
+ *
70
+ * LIVE DRAW (2026-08-01): none of these three providers expose a pricing API,
71
+ * but OpenRouter publishes /api/v1/models with live per-token pricing for all
72
+ * of them, and every model in the table below maps onto an OpenRouter id by a
73
+ * mechanical rule (see toOpenRouterId). So the rates are now drawn live and
74
+ * the table below is the OFFLINE FALLBACK, not the primary source.
75
+ *
76
+ * That change was not cosmetic. On the day it was made, three pinned rows
77
+ * disagreed with the live figures — and one disagreed in the direction that
78
+ * costs money:
79
+ *
80
+ * gemini-2.5-flash pinned $0.15/$0.60 live $0.30/$2.50
81
+ *
82
+ * A 4x UNDER-estimate on output. Under `--max-cost` an under-estimate is the
83
+ * dangerous kind: it lets a run pass a cap it should have failed. (The other
84
+ * two, claude-sonnet-5 and gpt-4o, were over-estimates.) A dated static table
85
+ * is only as good as the last time somebody remembered to date it, and the two
86
+ * blocks that were wrong are exactly the two marked "unverified".
87
+ *
88
+ * HONESTY ABOUT THE SOURCE: OpenRouter's number is what OPENROUTER charges to
89
+ * serve that model. A direct-provider call bills you at the provider's own list
90
+ * price, so the live figure is a PROXY for list, not an authority on it. It is
91
+ * a good proxy — all eight Anthropic rows below, verified against Anthropic's
92
+ * own pricing page on 2026-07-31, match OpenRouter exactly — but the returned
93
+ * `source` always says which number was used, and a live/pinned divergence is
94
+ * reported rather than silently resolved.
95
+ *
96
+ * Unknown models return null from estimateLlmCost() — "unknown", never a guess
97
+ * and never $0. Consumers already treat a null estimate as over-cap under
98
+ * --max-cost (see cost-report.js), so an unpriced model aborts a capped run
99
+ * rather than silently running up a bill.
100
+ */
101
+ export const LLM_RATES = {
102
+ // ── Anthropic ──────────────────────────────────────────────────────
103
+ // LAST VERIFIED: 2026-07-31 — https://platform.claude.com/docs/en/pricing
104
+ // Opus 4.6/4.7/4.8 and Opus 5 are all $5/$25; the $15/$75 this table
105
+ // replaces was Claude 3 Opus, retired 2026-01-05.
106
+ 'claude-opus-5': { provider: 'anthropic', input: 5.00, output: 25.00, verified: '2026-07-31' },
107
+ 'claude-opus-4-8': { provider: 'anthropic', input: 5.00, output: 25.00, verified: '2026-07-31' },
108
+ 'claude-opus-4-7': { provider: 'anthropic', input: 5.00, output: 25.00, verified: '2026-07-31' },
109
+ 'claude-opus-4-6': { provider: 'anthropic', input: 5.00, output: 25.00, verified: '2026-07-31' },
110
+ 'claude-sonnet-5': { provider: 'anthropic', input: 2.00, output: 10.00, verified: '2026-08-01' },
111
+ 'claude-sonnet-4-6': { provider: 'anthropic', input: 3.00, output: 15.00, verified: '2026-07-31' },
112
+ 'claude-haiku-4-5': { provider: 'anthropic', input: 1.00, output: 5.00, verified: '2026-07-31' },
113
+ 'claude-fable-5': { provider: 'anthropic', input: 10.00, output: 50.00, verified: '2026-07-31' },
114
+
115
+ // ── OpenAI ─────────────────────────────────────────────────────────
116
+ // LAST VERIFIED: 2026-08-01 against the live OpenRouter draw. The values
117
+ // carried over from openai.js were WRONG: gpt-4o was pinned at $5/$15,
118
+ // twice the real input rate. Corrected here.
119
+ 'gpt-4o': { provider: 'openai', input: 2.50, output: 10.00, verified: '2026-08-01' },
120
+ 'gpt-4o-mini': { provider: 'openai', input: 0.15, output: 0.60, verified: '2026-08-01' },
121
+
122
+ // ── Google Gemini ──────────────────────────────────────────────────
123
+ // LAST VERIFIED: 2026-08-01 against the live OpenRouter draw. The values
124
+ // carried over from gemini.js were WRONG, and wrong in the dangerous
125
+ // direction: gemini-2.5-flash output was pinned at $0.60/1M against a real
126
+ // $2.50/1M — a 4x UNDER-estimate, which lets a --max-cost run pass a cap it
127
+ // should have failed. Corrected here.
128
+ 'gemini-2.5-flash': { provider: 'gemini', input: 0.30, output: 2.50, verified: '2026-08-01' },
129
+ 'gemini-2.5-pro': { provider: 'gemini', input: 1.25, output: 10.00, verified: '2026-08-01' },
130
+ };
131
+
132
+ /**
133
+ * Map a native provider model id onto its OpenRouter id.
134
+ *
135
+ * The rule is mechanical and was verified against all 337 live OpenRouter
136
+ * models on 2026-08-01: every id in LLM_RATES resolved.
137
+ *
138
+ * - Anthropic prefixes `anthropic/` and writes the point release with a DOT
139
+ * where the native API uses a hyphen: claude-sonnet-4-6 → claude-sonnet-4.6.
140
+ * Single-segment names pass through: claude-opus-5 → claude-opus-5.
141
+ * - OpenAI prefixes `openai/`, 1:1.
142
+ * - Google prefixes `google/`, 1:1.
143
+ *
144
+ * @param {string} model native model id
145
+ * @returns {string|null} OpenRouter id, or null if the family is unrecognised
146
+ */
147
+ export function toOpenRouterId(model) {
148
+ if (typeof model !== 'string' || !model) return null;
149
+ if (model.startsWith('claude-')) {
150
+ const parts = model.split('-');
151
+ const [major, minor] = parts.slice(-2);
152
+ // …-4-6 is a point release; …-5 is not.
153
+ if (parts.length >= 4 && /^\d+$/.test(major) && /^\d+$/.test(minor)) {
154
+ return `anthropic/${parts.slice(0, -2).join('-')}-${major}.${minor}`;
155
+ }
156
+ return `anthropic/${model}`;
157
+ }
158
+ if (model.startsWith('gpt-') || model.startsWith('o1') || model.startsWith('o3')) {
159
+ return `openai/${model}`;
160
+ }
161
+ if (model.startsWith('gemini-')) return `google/${model}`;
162
+ return null;
163
+ }
164
+
165
+ // Divergence above this fraction between the live draw and the pinned row is
166
+ // reported. It means the pinned table has drifted and needs re-verifying —
167
+ // silence there is how a 4x under-estimate survived.
168
+ const DIVERGENCE_TOLERANCE = 0.10;
169
+
170
+ // One warning per model per process; a 3,000-key sync should not print 3,000
171
+ // identical lines.
172
+ const _warnedModels = new Set();
173
+
174
+ /**
175
+ * Is the live pricing draw disabled?
176
+ *
177
+ * `CHAMPOLLION_PRICING_OFFLINE=1` forces the pinned table. This is not a test
178
+ * hook — the sovereign-node and airgap-sandbox lanes run with no outbound
179
+ * network at all, and a cost estimate must still resolve there rather than
180
+ * hanging on a fetch that cannot complete. It also makes the test suite
181
+ * deterministic and network-free, which is why the suite sets it.
182
+ *
183
+ * @returns {boolean}
184
+ */
185
+ export function pricingOffline() {
186
+ return process.env.CHAMPOLLION_PRICING_OFFLINE === '1';
187
+ }
188
+
189
+ /**
190
+ * Estimate the token cost of translating a batch of keys with a direct LLM.
191
+ *
192
+ * Draws live pricing from OpenRouter and falls back to the pinned table when
193
+ * the network is unavailable. Returns null — "unknown", never $0 and never a
194
+ * guess — when neither source prices the model.
195
+ *
196
+ * @param {string} provider - 'anthropic' | 'openai' | 'gemini' (for the note)
197
+ * @param {string} model - EXACT model ID
198
+ * @param {number} keyCount - Number of translation keys in the batch
199
+ * @returns {Promise<{ estimatedCost: number|null, currency: string, source: string, note: string }>}
200
+ */
201
+ export async function estimateLlmCost(provider, model, keyCount) {
202
+ const pinned = LLM_RATES[model];
203
+ const pinnedUsable = pinned && pinned.provider === provider;
204
+
205
+ // ── live draw ────────────────────────────────────────────────────
206
+ let rate = null;
207
+ let source = null;
208
+ let divergence = '';
209
+
210
+ // The OpenRouter namespace must match the provider being billed. Without
211
+ // this, estimateLlmCost('anthropic', 'gpt-4o', …) would happily price the
212
+ // model through openai/gpt-4o — losing the cross-provider guard that the
213
+ // exact-ID table exists to enforce.
214
+ const OR_NAMESPACE = { anthropic: 'anthropic/', openai: 'openai/', gemini: 'google/' };
215
+ const orId = toOpenRouterId(model);
216
+ const namespaceOk = orId && OR_NAMESPACE[provider] && orId.startsWith(OR_NAMESPACE[provider]);
217
+ if (orId && namespaceOk && !pricingOffline()) {
218
+ let live = null;
219
+ try {
220
+ const pricing = await fetchModelPricing();
221
+ live = pricing.get(orId) || null;
222
+ } catch {
223
+ // fetchModelPricing already degrades to an empty Map offline; this
224
+ // catch only guards an unexpected throw. Fall through to pinned.
225
+ live = null;
226
+ }
227
+ if (live && (live.input > 0 || live.output > 0)) {
228
+ // openrouter-pricing.js stores cost PER TOKEN; this module works in
229
+ // dollars per million tokens.
230
+ rate = { input: live.input * 1_000_000, output: live.output * 1_000_000 };
231
+ source = `openrouter-live (proxy for ${provider} list price)`;
232
+
233
+ if (pinnedUsable) {
234
+ const drift = (a, b) => (b === 0 ? (a === 0 ? 0 : 1) : Math.abs(a - b) / b);
235
+ const dIn = drift(rate.input, pinned.input);
236
+ const dOut = drift(rate.output, pinned.output);
237
+ if (dIn > DIVERGENCE_TOLERANCE || dOut > DIVERGENCE_TOLERANCE) {
238
+ divergence =
239
+ ` PINNED TABLE HAS DRIFTED: provider-pricing.js says `
240
+ + `$${pinned.input}/$${pinned.output} per 1M, live says `
241
+ + `$${rate.input.toFixed(2)}/$${rate.output.toFixed(2)}. `
242
+ + `Using live. Re-verify the pinned row against the provider's `
243
+ + `pricing page and re-date it.`;
244
+ if (!_warnedModels.has(model)) {
245
+ _warnedModels.add(model);
246
+ console.warn(`⚠ pricing drift for ${model}:${divergence}`);
247
+ }
248
+ }
249
+ }
250
+ }
251
+ }
252
+
253
+ // ── offline fallback ─────────────────────────────────────────────
254
+ if (!rate && pinnedUsable) {
255
+ rate = { input: pinned.input, output: pinned.output };
256
+ source = `pinned-table (${pinned.verified ? `verified ${pinned.verified}` : 'UNVERIFIED'}; live draw unavailable)`;
257
+ }
258
+
259
+ if (!rate) {
260
+ return {
261
+ estimatedCost: null,
262
+ currency: 'USD',
263
+ source: `${provider}-pricing-unknown`,
264
+ note: `No rate for ${provider} model "${model}" — not on OpenRouter and `
265
+ + `no pinned row. Cost is UNKNOWN (not $0). Add a row to LLM_RATES in `
266
+ + `provider-pricing.js to price it.`,
267
+ };
268
+ }
269
+
270
+ const inputTokens = keyCount * EST_INPUT_TOKENS_PER_KEY;
271
+ const outputTokens = keyCount * EST_OUTPUT_TOKENS_PER_KEY;
272
+ const estimatedCost =
273
+ (inputTokens * rate.input + outputTokens * rate.output) / 1_000_000;
274
+
275
+ return {
276
+ estimatedCost: Math.round(estimatedCost * 10000) / 10000,
277
+ currency: 'USD',
278
+ source,
279
+ note: `Based on ${provider} ${model} pricing `
280
+ + `($${rate.input.toFixed(2)}/1M input, $${rate.output.toFixed(2)}/1M output, `
281
+ + `${source}).${divergence}`,
282
+ };
283
+ }
284
+
285
+ /**
286
+ * Estimate the API cost for translating a batch of keys.
287
+ *
288
+ * @param {string} provider - Provider name matching a PROVIDER_RATES key
289
+ * @param {number} keyCount - Number of translation keys in the batch
290
+ * @returns {{ estimatedCost: number, currency: string, source: string, note: string }}
291
+ * @throws {Error} If provider is not in PROVIDER_RATES — forces new
292
+ * providers to add pricing data before estimateCost() can be called.
293
+ */
294
+ export function estimateProviderCost(provider, keyCount) {
295
+ const rate = PROVIDER_RATES[provider];
296
+ if (!rate) {
297
+ throw new Error(
298
+ `No pricing data for provider "${provider}". ` +
299
+ `Add it to provider-pricing.js before using estimateCost().`
300
+ );
301
+ }
302
+ const estimatedChars = keyCount * EST_CHARS_PER_KEY;
303
+ const costPerChar = rate.costPerMillionChars / 1_000_000;
304
+ return {
305
+ estimatedCost: Math.round(estimatedChars * costPerChar * 10000) / 10000,
306
+ currency: 'USD',
307
+ source: rate.source,
308
+ note: rate.note,
309
+ };
310
+ }
@@ -0,0 +1,150 @@
1
+ /**
2
+ * Tilde MT Translation Method — European-developed neural MT via the Tilde MT
3
+ * REST API. Mirrors the harness TildeMethod
4
+ * (arena/mt_eval_harness/methods/tilde.py): same endpoint, env var, body, and
5
+ * response shape, so config is interchangeable between CLI and harness.
6
+ *
7
+ * Endpoint: POST https://translate.tilde.ai/api/translate/text
8
+ * Auth: header X-Api-Key: <key>
9
+ * Env: TILDE_API_KEY (free trial key at translate.tilde.ai/access-keys)
10
+ * Body: { srcLang, trgLang, domain:"general", text:[...], termCollections:[] }
11
+ * Response: json.translations[i].translation
12
+ * Locale: ISO 639-1.
13
+ */
14
+
15
+ import { TranslationMethod } from './base.js';
16
+ import { getEnvOrFileVar } from '../api-key.js';
17
+ import { DEFAULT_METHOD_CONCURRENCY } from '../config.js';
18
+ import { extractContentBody } from './content-separator.js';
19
+ import { output } from '../output.js';
20
+ import { fetchWithRetry } from './fetch-with-retry.js';
21
+ import { pMap } from '../concurrent.js';
22
+
23
+ const TILDE_API_URL = 'https://translate.tilde.ai/api/translate/text';
24
+ const TILDE_TIMEOUT_MS = 15000;
25
+ const TILDE_MAX_BATCH = 25;
26
+
27
+ class TildeMethod extends TranslationMethod {
28
+ constructor(options = {}) {
29
+ super('tilde', options);
30
+ }
31
+
32
+ _resolveApiKey(options = {}) {
33
+ return options.tildeApiKey
34
+ || getEnvOrFileVar('TILDE_API_KEY')
35
+ || getEnvOrFileVar('TILDE_API_KEY', options.cwd);
36
+ }
37
+
38
+ async _translateBatch(apiKey, orderedKeys, sourceTexts, srcLang, trgLang, batchNum) {
39
+ const res = await fetchWithRetry(TILDE_API_URL, {
40
+ method: 'POST',
41
+ headers: { 'Content-Type': 'application/json', 'X-Api-Key': apiKey },
42
+ body: JSON.stringify({
43
+ srcLang, trgLang, domain: 'general', text: sourceTexts, termCollections: [],
44
+ }),
45
+ }, { label: `Tilde batch ${batchNum}`, timeoutMs: TILDE_TIMEOUT_MS });
46
+
47
+ if (!res) return null;
48
+ if (!res.ok) {
49
+ const body = await res.text();
50
+ output.error(`Tilde batch ${batchNum}: ${res.status} — ${body.slice(0, 200)}`);
51
+ return null;
52
+ }
53
+ const json = await res.json();
54
+ const translations = json?.translations;
55
+ if (!Array.isArray(translations) || translations.length !== orderedKeys.length) {
56
+ output.error(`Tilde batch ${batchNum}: response length mismatch`);
57
+ return null;
58
+ }
59
+ const result = {};
60
+ for (let i = 0; i < orderedKeys.length; i++) {
61
+ result[orderedKeys[i]] = translations[i]?.translation;
62
+ }
63
+ return result;
64
+ }
65
+
66
+ async translate(keys, sourceFlat, pairConfig, options) {
67
+ const apiKey = this._resolveApiKey(options);
68
+ if (!apiKey) {
69
+ output.warn('Tilde: no API key (TILDE_API_KEY) — skipping.');
70
+ return null;
71
+ }
72
+ const srcLang = pairConfig.source || 'en';
73
+ const trgLang = pairConfig.target;
74
+ const maxBatch = pairConfig.batchSize || options.batchSize || TILDE_MAX_BATCH;
75
+ const allTranslated = {};
76
+
77
+ const chunks = [];
78
+ for (let i = 0; i < keys.length; i += maxBatch) chunks.push(keys.slice(i, i + maxBatch));
79
+
80
+ await pMap(chunks, async (chunk, idx) => {
81
+ const orderedKeys = [];
82
+ const sourceTexts = [];
83
+ for (const key of chunk) {
84
+ const val = sourceFlat[key];
85
+ if (typeof val === 'string' && val) { orderedKeys.push(key); sourceTexts.push(val); }
86
+ }
87
+ if (!sourceTexts.length) return;
88
+ const r = await this._translateBatch(apiKey, orderedKeys, sourceTexts, srcLang, trgLang, idx + 1);
89
+ if (r) Object.assign(allTranslated, r);
90
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
91
+
92
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
93
+ }
94
+
95
+ async translateContent(prompt, pairConfig, options) {
96
+ const apiKey = this._resolveApiKey(options);
97
+ if (!apiKey) return null;
98
+ const body = extractContentBody(prompt);
99
+ if (!body.trim()) return null;
100
+ const r = await this._translateBatch(
101
+ apiKey, ['__content__'], [body], pairConfig.source || 'en', pairConfig.target, 'content',
102
+ );
103
+ return r ? r['__content__'] : null;
104
+ }
105
+
106
+ checkReadiness(context) {
107
+ const apiKey = this._resolveApiKey(context || {});
108
+ if (!apiKey) {
109
+ return {
110
+ ready: false,
111
+ reason: 'No Tilde API key (TILDE_API_KEY). Free trial key at https://translate.tilde.ai/access-keys.',
112
+ };
113
+ }
114
+ return { ready: true };
115
+ }
116
+
117
+ estimateCost(_keyCount) {
118
+ // Tilde pricing is plan-based; per-character cost is not published, so we
119
+ // do not fabricate one (fail-honest — null, not a guessed $).
120
+ return {
121
+ estimatedCost: null,
122
+ currency: 'USD',
123
+ source: 'tilde-proprietary',
124
+ note: 'Tilde pricing is plan-based (free trial 250K chars); per-char cost not published.',
125
+ };
126
+ }
127
+
128
+ getQualityTier() {
129
+ return 'standard';
130
+ }
131
+
132
+ getProvenance() {
133
+ return {
134
+ resources: [{ name: 'Tilde MT API', license: 'Proprietary (Tilde ToS)', type: 'api' }],
135
+ commercialReady: true,
136
+ flags: [],
137
+ };
138
+ }
139
+
140
+ getSetupHelp() {
141
+ return [
142
+ ' Tilde MT (European neural MT):',
143
+ ' 1. Sign up: https://tilde.ai/translate/machine-translation-api/ (free trial 250K chars).',
144
+ ' 2. Create an access key: https://translate.tilde.ai/access-keys',
145
+ ' 3. export TILDE_API_KEY=... (or add to .env.local)',
146
+ ];
147
+ }
148
+ }
149
+
150
+ export { TildeMethod };
@@ -0,0 +1,229 @@
1
+ /**
2
+ * Translated (Lara) — professional MT via Translated's Lara Translate API.
3
+ *
4
+ * Endpoint: official @translated/lara SDK (REST /v2/translate under the
5
+ * hood; requests are HMAC-signed by the SDK — we never hand-roll
6
+ * the signature)
7
+ * Auth: LARA_ACCESS_KEY_ID + LARA_ACCESS_KEY_SECRET (dashboard →
8
+ * Settings → API Keys at laratranslate.com)
9
+ * Codes: BCP-47; base ISO 639-1 codes are accepted and auto-resolve to
10
+ * the default regional locale (en → en-US), so we pass our locale
11
+ * codes through unchanged.
12
+ * Batch: the SDK accepts string arrays; we chunk to keep request sizes
13
+ * polite and failures isolated.
14
+ *
15
+ * ModernMT note: Translated is sunsetting ModernMT into Lara (API-compatible
16
+ * migration path until end of 2026); this adapter targets Lara only.
17
+ */
18
+
19
+ import { TranslationMethod } from './base.js';
20
+ import { getEnvOrFileVar } from '../api-key.js';
21
+ import { DEFAULT_METHOD_CONCURRENCY } from '../config.js';
22
+ import { extractContentBody } from './content-separator.js';
23
+ import { output } from '../output.js';
24
+ import { pMap } from '../concurrent.js';
25
+
26
+ const TRANSLATED_MAX_BATCH = 50;
27
+
28
+ class TranslatedMethod extends TranslationMethod {
29
+ constructor(options = {}) {
30
+ super('translated', options);
31
+ this._client = options.laraClient || null; // injectable for tests
32
+ }
33
+
34
+ // ── Credential / client resolution ─────────────────────────────
35
+
36
+ /**
37
+ * Resolve the Lara access-key pair from options, env vars, or .env files.
38
+ * @param {object} options - Caller-provided options
39
+ * @returns {{id: string, secret: string}|null}
40
+ */
41
+ _resolveCredentials(options = {}) {
42
+ const id = options.laraAccessKeyId
43
+ || getEnvOrFileVar('LARA_ACCESS_KEY_ID')
44
+ || getEnvOrFileVar('LARA_ACCESS_KEY_ID', options.cwd);
45
+ const secret = options.laraAccessKeySecret
46
+ || getEnvOrFileVar('LARA_ACCESS_KEY_SECRET')
47
+ || getEnvOrFileVar('LARA_ACCESS_KEY_SECRET', options.cwd);
48
+ return id && secret ? { id, secret } : null;
49
+ }
50
+
51
+ /**
52
+ * Lazily construct the official Lara SDK client. The SDK owns request
53
+ * signing (HMAC over the request envelope), retries, and endpoint choice.
54
+ * @param {object} options
55
+ * @returns {Promise<object|null>} Translator instance or null
56
+ */
57
+ async _getClient(options = {}) {
58
+ if (this._client) return this._client;
59
+ const creds = this._resolveCredentials(options);
60
+ if (!creds) return null;
61
+ let lara;
62
+ try {
63
+ lara = await import('@translated/lara');
64
+ } catch {
65
+ output.error(
66
+ 'Translated (Lara): the @translated/lara SDK is not installed — '
67
+ + 'run `npm install` in the champollion package.',
68
+ );
69
+ return null;
70
+ }
71
+ const { Credentials, Translator } = lara;
72
+ this._client = new Translator(new Credentials(creds.id, creds.secret));
73
+ return this._client;
74
+ }
75
+
76
+ // ── Translation ────────────────────────────────────────────────
77
+
78
+ /**
79
+ * Translate one chunk of texts. Returns the translations array (in input
80
+ * order) or null on failure — callers treat null as a skipped batch.
81
+ */
82
+ async _translateBatch(client, sourceTexts, sourceLocale, targetLocale, batchNum) {
83
+ try {
84
+ const result = await client.translate(sourceTexts, sourceLocale, targetLocale);
85
+ const translations = Array.isArray(result?.translation)
86
+ ? result.translation
87
+ : [result?.translation];
88
+ if (translations.length !== sourceTexts.length) {
89
+ output.error(
90
+ `Translated batch ${batchNum}: response length mismatch `
91
+ + `(sent ${sourceTexts.length}, got ${translations.length})`,
92
+ );
93
+ return null;
94
+ }
95
+ return translations;
96
+ } catch (err) {
97
+ output.error(`Translated batch ${batchNum}: ${err?.message || err}`);
98
+ return null;
99
+ }
100
+ }
101
+
102
+ /**
103
+ * Translate a batch of key-value pairs via the Lara API.
104
+ *
105
+ * @param {string[]} keys - Flat dot-notation keys to translate
106
+ * @param {object} sourceFlat - Full flattened source locale
107
+ * @param {import('../types.js').PairConfig} pairConfig - Pair config
108
+ * @param {object} options - { laraAccessKeyId, laraAccessKeySecret, batchSize }
109
+ * @returns {object|null} Map of key → translated value, or null
110
+ */
111
+ async translate(keys, sourceFlat, pairConfig, options) {
112
+ const client = await this._getClient(options);
113
+ if (!client) {
114
+ output.warn(
115
+ 'Translated (Lara): no credentials (LARA_ACCESS_KEY_ID / '
116
+ + 'LARA_ACCESS_KEY_SECRET) — skipping.',
117
+ );
118
+ return null;
119
+ }
120
+
121
+ const sourceLocale = pairConfig.source || 'en';
122
+ const targetLocale = pairConfig.target;
123
+ const maxSegments = pairConfig.batchSize || options.batchSize || TRANSLATED_MAX_BATCH;
124
+ const allTranslated = {};
125
+
126
+ const batchChunks = [];
127
+ for (let i = 0; i < keys.length; i += maxSegments) {
128
+ batchChunks.push(keys.slice(i, i + maxSegments));
129
+ }
130
+
131
+ await pMap(batchChunks, async (chunk, idx) => {
132
+ const orderedKeys = [];
133
+ const sourceTexts = [];
134
+ for (const key of chunk) {
135
+ const val = sourceFlat[key];
136
+ if (val && typeof val === 'string') {
137
+ orderedKeys.push(key);
138
+ sourceTexts.push(val);
139
+ }
140
+ }
141
+ if (sourceTexts.length === 0) return;
142
+
143
+ const translations = await this._translateBatch(
144
+ client, sourceTexts, sourceLocale, targetLocale, idx + 1,
145
+ );
146
+ if (!translations) return;
147
+ for (let i = 0; i < orderedKeys.length; i++) {
148
+ if (typeof translations[i] === 'string') {
149
+ allTranslated[orderedKeys[i]] = translations[i];
150
+ }
151
+ }
152
+ }, { concurrency: DEFAULT_METHOD_CONCURRENCY });
153
+
154
+ return Object.keys(allTranslated).length > 0 ? allTranslated : null;
155
+ }
156
+
157
+ /**
158
+ * Translate freeform Markdown content via Lara.
159
+ * @param {string} prompt - Content prompt (body extracted before sending)
160
+ * @param {import('../types.js').PairConfig} pairConfig
161
+ * @param {object} options
162
+ * @returns {Promise<string|null>}
163
+ */
164
+ async translateContent(prompt, pairConfig, options) {
165
+ const client = await this._getClient(options);
166
+ if (!client) return null;
167
+ const body = extractContentBody(prompt);
168
+ if (!body.trim()) return null;
169
+ const translations = await this._translateBatch(
170
+ client, [body], pairConfig.source || 'en', pairConfig.target, 'content',
171
+ );
172
+ return translations ? translations[0] : null;
173
+ }
174
+
175
+ // ── Metadata ───────────────────────────────────────────────────
176
+
177
+ checkReadiness(context) {
178
+ const creds = this._resolveCredentials(context || {});
179
+ if (!creds) {
180
+ return {
181
+ ready: false,
182
+ reason: 'No Lara credentials (LARA_ACCESS_KEY_ID + LARA_ACCESS_KEY_SECRET). '
183
+ + 'Create an API key at laratranslate.com → Settings → API Keys.',
184
+ };
185
+ }
186
+ return { ready: true };
187
+ }
188
+
189
+ estimateCost(_keyCount) {
190
+ // Proprietary per-character pricing behind Translated's plans; never
191
+ // invent a number (Project rule: never invent pricing).
192
+ return {
193
+ estimatedCost: null,
194
+ currency: 'USD',
195
+ source: 'translated-lara-proprietary',
196
+ note: 'Lara pricing is plan-based; per-character cost not published.',
197
+ };
198
+ }
199
+
200
+ getQualityTier() {
201
+ return 'standard';
202
+ }
203
+
204
+ getProvenance() {
205
+ return {
206
+ resources: [
207
+ {
208
+ name: 'Translated Lara Translate API',
209
+ license: 'Proprietary (Translated ToS)',
210
+ type: 'api',
211
+ },
212
+ ],
213
+ commercialReady: true,
214
+ flags: [],
215
+ };
216
+ }
217
+
218
+ getSetupHelp() {
219
+ return [
220
+ ' Translated (Lara) — professional MT API, 200+ languages:',
221
+ ' 1. Sign up: https://laratranslate.com',
222
+ ' 2. Settings → API Keys → create an access key',
223
+ ' 3. export LARA_ACCESS_KEY_ID=... LARA_ACCESS_KEY_SECRET=...',
224
+ ' (or add both to .env.local)',
225
+ ];
226
+ }
227
+ }
228
+
229
+ export { TranslatedMethod, TRANSLATED_MAX_BATCH };