champollion 0.3.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -37
- package/bin/cli.js +53 -5
- package/index.js +63 -2
- package/lib/api-key.js +17 -4
- package/lib/autofix.js +83 -36
- package/lib/bridge/method_bridge.py +15 -3
- package/lib/cards/reader.js +51 -3
- package/lib/cards/remote.js +15 -0
- package/lib/cards/search-names.js +178 -0
- package/lib/command-help.js +289 -88
- package/lib/commands/audit.js +10 -3
- package/lib/commands/card.js +583 -226
- package/lib/commands/doctor.js +54 -18
- package/lib/commands/help.js +37 -32
- package/lib/commands/init.js +1689 -87
- package/lib/commands/integrity.js +127 -40
- package/lib/commands/leaderboard.js +187 -67
- package/lib/commands/models.js +9 -2
- package/lib/commands/provenance.js +7 -2
- package/lib/commands/recommend.js +43 -14
- package/lib/commands/register-corpus.js +649 -130
- package/lib/commands/seal-corpus.js +1 -1
- package/lib/commands/status.js +564 -27
- package/lib/commands/submit.js +17 -12
- package/lib/commands/sync.js +31 -7
- package/lib/commands/tm.js +16 -10
- package/lib/commands/verify.js +27 -3
- package/lib/commands/wrap.js +63 -5
- package/lib/commands/xliff.js +135 -64
- package/lib/commercial-eligibility.js +1 -1
- package/lib/config.js +196 -14
- package/lib/content-estimate.js +96 -0
- package/lib/content-refusals.js +270 -0
- package/lib/content-review.js +372 -0
- package/lib/content-sync.js +1127 -344
- package/lib/content.js +94 -7
- package/lib/corpus-registration.mjs +197 -38
- package/lib/cost-label.js +29 -0
- package/lib/cost-report.js +726 -78
- package/lib/diff.js +38 -4
- package/lib/docusaurus-sync.js +965 -253
- package/lib/edit-distance.js +31 -0
- package/lib/fallback.js +964 -0
- package/lib/file-scope.js +106 -0
- package/lib/flatten.js +80 -3
- package/lib/flutter-locales.js +124 -0
- package/lib/format.js +266 -12
- package/lib/hash.js +146 -21
- package/lib/icu-structure.js +929 -0
- package/lib/integrity.js +223 -75
- package/lib/language-pair.js +157 -0
- package/lib/lint.js +78 -16
- package/lib/local-only-marks.js +106 -0
- package/lib/locale-layout.js +1103 -0
- package/lib/locale-state.js +571 -0
- package/lib/methods/anthropic.js +5 -0
- package/lib/methods/apertium.js +6 -3
- package/lib/methods/api.js +138 -25
- package/lib/methods/base.js +17 -0
- package/lib/methods/coaching-data.js +153 -0
- package/lib/methods/content-separator.js +43 -0
- package/lib/methods/deepl.js +1 -1
- package/lib/methods/direct-llm.js +252 -103
- package/lib/methods/external.js +146 -63
- package/lib/methods/gemini.js +1 -0
- package/lib/methods/google-translate.js +1 -0
- package/lib/methods/http-utils.js +41 -0
- package/lib/methods/libretranslate.js +7 -2
- package/lib/methods/llm-coached.js +68 -128
- package/lib/methods/llm.js +80 -31
- package/lib/methods/local.js +93 -10
- package/lib/methods/microsoft-translator.js +1 -2
- package/lib/methods/openai.js +4 -2
- package/lib/methods/openrouter-client.js +20 -19
- package/lib/methods/openrouter-pricing.js +150 -13
- package/lib/methods/prompt-methods.js +20 -0
- package/lib/methods/provider-pricing.js +42 -1
- package/lib/methods/request-capture.js +104 -0
- package/lib/methods/tilde.js +1 -1
- package/lib/methods/translated.js +1 -2
- package/lib/missing-key.js +93 -0
- package/lib/models.js +11 -0
- package/lib/name-rules.js +32 -0
- package/lib/named-keys.js +172 -0
- package/lib/no-translate.js +4 -3
- package/lib/output.js +160 -19
- package/lib/pairs.js +586 -30
- package/lib/placeholders.js +394 -0
- package/lib/plugins.js +8 -0
- package/lib/plural-gap-redo.js +109 -0
- package/lib/plurals.js +323 -0
- package/lib/po.js +1187 -0
- package/lib/public-catalogue.js +74 -0
- package/lib/recommend.js +527 -32
- package/lib/redo.js +95 -0
- package/lib/refusal-category.js +44 -0
- package/lib/registers.js +255 -11
- package/lib/repair-script.js +20 -13
- package/lib/scripts.js +193 -106
- package/lib/seal.mjs +6 -5
- package/lib/sealed-qualifier.mjs +2 -2
- package/lib/segment.js +2 -1
- package/lib/seo.js +19 -9
- package/lib/serve.js +43 -6
- package/lib/shared-output-seed.js +164 -0
- package/lib/source-contexts.js +39 -0
- package/lib/submit.mjs +57 -5
- package/lib/sync.js +2923 -474
- package/lib/terminology.js +13 -4
- package/lib/tm-evict.js +179 -0
- package/lib/tm-seed.js +5 -2
- package/lib/tm.js +818 -36
- package/lib/translate-pair.js +639 -34
- package/lib/translate.js +78 -5
- package/lib/types.js +22 -3
- package/lib/validate.js +880 -17
- package/lib/verify.js +1296 -104
- package/lib/watch.js +32 -13
- package/lib/xliff.js +44 -3
- package/package.json +3 -2
- package/shared/CORPORA-CARDS.md +2 -0
- package/shared/DATA-SOVEREIGNTY.md +19 -20
- package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
- package/shared/cards-fallback.json +1 -1
- package/shared/catalogue/card-config.json +1 -1
- package/shared/curated-orthography-conventions.json +26 -8
- package/shared/docent/faq.en.json +14 -16
- package/shared/docent/system-prompt.md +17 -19
- package/shared/explainers/tc-features.json +15 -15
- package/shared/gettext-plural-forms.json +45 -0
- package/shared/human-services.json +1 -1
- package/shared/method-registry.json +2 -0
- package/shared/metric-registry.json +96 -18
- package/shared/schemas/champollion-plugin.schema.json +4 -0
- package/shared/schemas/corpora-card.schema.json +20 -10
- package/shared/schemas/human-services.schema.json +2 -2
- package/shared/schemas/language-card.schema.json +1 -1
- package/shared/schemas/method-card.schema.json +1 -1
- package/shared/schemas/method-index-record.schema.json +67 -0
- package/shared/schemas/method-registry.schema.json +4 -0
- package/shared/schemas/metric-registry.schema.json +55 -1
- package/shared/docent/corpus.json +0 -11333
package/lib/cost-report.js
CHANGED
|
@@ -38,11 +38,154 @@ import path from 'node:path';
|
|
|
38
38
|
import { flattenKeys } from './flatten.js';
|
|
39
39
|
import { diffLocale } from './diff.js';
|
|
40
40
|
import { readLocaleFile } from './format.js';
|
|
41
|
+
import { expectedForTarget, keysForNamespace, readLocaleFlat } from './locale-layout.js';
|
|
42
|
+
import { mapSourceKeysToTarget } from './plurals.js';
|
|
41
43
|
import { EST_CHARS_PER_KEY } from './config.js';
|
|
44
|
+
import { formatPerMillion } from './methods/openrouter-pricing.js';
|
|
42
45
|
import { countPendingContentTranslations } from './content-sync.js';
|
|
43
|
-
import { loadTM, lookupTM, partitionByTM, tmMethodKey } from './tm.js';
|
|
46
|
+
import { loadTM, lookupTM, partitionByTM, tmMethodKey, findModelSwitchStrandedEntries, reusableFromEarlierModels, carriedFromModel } from './tm.js';
|
|
47
|
+
import { tmTextFor } from './tm-evict.js';
|
|
44
48
|
import { compileNoTranslate } from './no-translate.js';
|
|
49
|
+
import { tmKeysForPair, tmHoldsValue } from './fallback.js';
|
|
45
50
|
import { output } from './output.js';
|
|
51
|
+
import { planQueue, createEditClassifier } from './locale-state.js';
|
|
52
|
+
import { planGapRedo } from './plural-gap-redo.js';
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Say what a model switch means for this run.
|
|
56
|
+
*
|
|
57
|
+
* By default (model carry-over, see lib/tm.js) translations made under the
|
|
58
|
+
* previous model are reused at no cost — the operator should KNOW that, both
|
|
59
|
+
* because it is a saving and because it is a choice they can reverse. With
|
|
60
|
+
* --fresh-on-model-change the same entries are deliberately bypassed, and
|
|
61
|
+
* the run will re-translate (and bill) that text — say so before spending.
|
|
62
|
+
* Returns the report so callers can carry it into the --json summary.
|
|
63
|
+
*
|
|
64
|
+
* `served` (target → { earlierModel: count }, from the cost estimate's
|
|
65
|
+
* options.collect) limits the carry-over notice to what THIS run reuses: a
|
|
66
|
+
* plain sync after a model switch that queues nothing reads nothing from the
|
|
67
|
+
* cache, and repeating "N cached translations will be reused" on every such
|
|
68
|
+
* run was noise (Round 5, Next.js persona). `champollion status` carries the
|
|
69
|
+
* standing fact (the files hold an earlier model's text). Without `served`
|
|
70
|
+
* the notice counts every cached translation of the current source.
|
|
71
|
+
*
|
|
72
|
+
* @param {object} tm - Loaded TM
|
|
73
|
+
* @param {Iterable<object>} pairConfigs - Resolved pair configs
|
|
74
|
+
* @param {{ fresh?: boolean, redoAll?: boolean, sourceTexts?: Iterable<string>|null,
|
|
75
|
+
* served?: Object<string, Object<string, number>>|null }} [options]
|
|
76
|
+
* @returns {ReturnType<typeof findModelSwitchStrandedEntries>}
|
|
77
|
+
*/
|
|
78
|
+
/** "1 translation" / "10 translations". */
|
|
79
|
+
const translationsWord = (n) => `${n} translation${n === 1 ? '' : 's'}`;
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* What a count of translations across target languages is made of, in plain
|
|
83
|
+
* units: "5 strings × 2 languages", "5 strings in fr", "3 in fr, 5 in de".
|
|
84
|
+
* The notices summed every locale's count and called the total "source
|
|
85
|
+
* string(s)" — 10 for a 5-key en.json with two targets (Round 12, Next.js
|
|
86
|
+
* persona).
|
|
87
|
+
*
|
|
88
|
+
* @param {Array<{ target: string, n: number }>} perTarget
|
|
89
|
+
* @returns {string}
|
|
90
|
+
*/
|
|
91
|
+
export function translationsBreakdown(perTarget) {
|
|
92
|
+
const rows = perTarget.filter(p => p.n > 0);
|
|
93
|
+
const strings = (n) => `${n} string${n === 1 ? '' : 's'}`;
|
|
94
|
+
if (rows.length === 1) return `${strings(rows[0].n)} in ${rows[0].target}`;
|
|
95
|
+
if (rows.length > 1 && rows.every(p => p.n === rows[0].n)) return `${strings(rows[0].n)} × ${rows.length} languages`;
|
|
96
|
+
return rows.map(p => `${p.n} in ${p.target}`).join(', ');
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export function warnModelSwitchStrandedTM(tm, pairConfigs, { fresh = false, redoAll = false, sourceTexts = null, served = null } = {}) {
|
|
100
|
+
// Iterated twice (the report, then each row's own pair) — an iterator
|
|
101
|
+
// (Map#values()) would be empty the second time.
|
|
102
|
+
const pcs = [...pairConfigs];
|
|
103
|
+
// sourceTexts: count only translations of strings the project still has
|
|
104
|
+
// (an entry for an edited or deleted string is never reused).
|
|
105
|
+
const report = findModelSwitchStrandedEntries(tm, pcs, { sourceTexts });
|
|
106
|
+
// With `served`, what the run reuses is said even when the report is
|
|
107
|
+
// empty (see below).
|
|
108
|
+
if (report.length === 0 && !served) return report;
|
|
109
|
+
// One count for the total and its breakdown (Round 11, Next.js persona:
|
|
110
|
+
// "12 cached translation(s)" above per-model counts adding up to 30 — the
|
|
111
|
+
// total summed each locale's FIRST earlier model, the lines listed every
|
|
112
|
+
// model's holdings, and a string two models translated counted twice).
|
|
113
|
+
// Each string of the current source that only an earlier model translated
|
|
114
|
+
// counts once, under the model whose translation carry-over would reuse
|
|
115
|
+
// (lib/tm.js reusableFromEarlierModels); a project whose Markdown shares
|
|
116
|
+
// the cache counts cache entries instead, and says so.
|
|
117
|
+
const reusable = (r) => reusableFromEarlierModels(tm, pcs.find(pc => pc.target === r.target) || { target: r.target }, r.stranded, { sourceTexts });
|
|
118
|
+
const sayCounts = (rows, verb) => {
|
|
119
|
+
const counted = rows.map(r => ({ r, u: reusable(r) })).filter(x => x.u.total > 0);
|
|
120
|
+
const total = counted.reduce((n, x) => n + x.u.total, 0);
|
|
121
|
+
const unit = counted.some(x => x.u.unit === 'entries') ? 'entries' : 'strings';
|
|
122
|
+
const perTarget = counted.map(x => ({ target: x.r.target, n: x.u.total }));
|
|
123
|
+
// Counted in translations — one per string per language — and said so.
|
|
124
|
+
const what = unit === 'strings'
|
|
125
|
+
? `${translationsWord(total)} only an earlier model wrote (${translationsBreakdown(perTarget)})`
|
|
126
|
+
: `${translationsWord(total)} cached from earlier models (counted per cache entry, so a string two earlier models `
|
|
127
|
+
+ `translated counts twice: ${perTarget.map(p => `${p.n} in ${p.target}`).join(', ')})`;
|
|
128
|
+
const lines = counted.map(({ r, u }) => ` ${r.target}: now ${r.currentModel || '(none)'}; ${u.total} ${verb} ${u.byModel.map(x => `${x.model || '(none)'} (${x.count})`).join(', ')}`);
|
|
129
|
+
return { total, what, lines };
|
|
130
|
+
};
|
|
131
|
+
if (fresh) {
|
|
132
|
+
const { total, what, lines } = sayCounts(report, 'from');
|
|
133
|
+
if (total === 0) return report;
|
|
134
|
+
// "Billed" read wrong above a "$0 (local)" estimate (Round 6, Next.js
|
|
135
|
+
// persona): what happens is that the text is sent to the model again, at
|
|
136
|
+
// the method's price — which the estimate below states.
|
|
137
|
+
const price = 'at the method\'s price (the estimate below; $0 API cost for a model on this machine)';
|
|
138
|
+
output.warn(redoAll
|
|
139
|
+
? `Model changed: re-translating everything with the new model (--redo all --fresh-on-model-change); ${what} will not be reused — that text is sent to the model again, ${price}.`
|
|
140
|
+
: `Model changed: --fresh-on-model-change set, so ${what} will NOT be reused — but only text this run translates anyway (new or changed keys) is sent to the model again, ${price}. To re-translate ALL of it with the new model, add --redo all.`);
|
|
141
|
+
for (const l of lines) output.warn(l);
|
|
142
|
+
return report;
|
|
143
|
+
}
|
|
144
|
+
if (served) {
|
|
145
|
+
// Only what this run actually serves from an earlier model's cache —
|
|
146
|
+
// counted from what the estimate found, for EVERY locale it serves one
|
|
147
|
+
// in, not only the locales the stranded-entries report names. That report
|
|
148
|
+
// stays silent once a switch is complete (or when the new model holds more
|
|
149
|
+
// entries than the old), so a string reverted to text only the earlier
|
|
150
|
+
// model had translated was served as a free cache hit with no word that
|
|
151
|
+
// the text was that model's — the real sync and `status` said it
|
|
152
|
+
// afterwards (Round 13, Next.js persona; the Translation Memory page
|
|
153
|
+
// promises it before the estimate).
|
|
154
|
+
const every = findModelSwitchStrandedEntries(tm, pcs, { sourceTexts, every: true });
|
|
155
|
+
for (const pc of pcs) {
|
|
156
|
+
const byModel = served[pc.target] || {};
|
|
157
|
+
if (Object.values(byModel).reduce((a, b) => a + b, 0) === 0 || report.some(r => r.target === pc.target)) continue;
|
|
158
|
+
const row = every.find(r => r.target === pc.target) || {
|
|
159
|
+
target: pc.target, currentModel: tmMethodKey(pc).split('|')[1] || '', current: 0,
|
|
160
|
+
stranded: Object.entries(byModel).map(([model, count]) => ({ model, count })),
|
|
161
|
+
};
|
|
162
|
+
report.push(row);
|
|
163
|
+
}
|
|
164
|
+
const lines = [];
|
|
165
|
+
const perTarget = [];
|
|
166
|
+
let total = 0;
|
|
167
|
+
for (const r of report) {
|
|
168
|
+
const byModel = served[r.target] || {};
|
|
169
|
+
const n = Object.values(byModel).reduce((a, b) => a + b, 0);
|
|
170
|
+
r.servedThisRun = n;
|
|
171
|
+
if (n === 0) continue;
|
|
172
|
+
total += n;
|
|
173
|
+
perTarget.push({ target: r.target, n });
|
|
174
|
+
const from = Object.entries(byModel).sort((a, b) => b[1] - a[1]).map(([m, c]) => `${m} (${c})`).join(', ');
|
|
175
|
+
lines.push(` ${r.target}: now ${r.currentModel || '(none)'}; reusing translations from ${from}`);
|
|
176
|
+
}
|
|
177
|
+
if (total === 0) return report;
|
|
178
|
+
output.info(`Model changed: ${translationsWord(total)} this run needs (${translationsBreakdown(perTarget)}) `
|
|
179
|
+
+ `${total === 1 ? 'is' : 'are'} served from the previous model's cached translations, at no cost. To have the new model translate them instead: --redo all --fresh-on-model-change (it sends what an earlier model translated; what the new model already translated comes from the cache).`);
|
|
180
|
+
for (const l of lines) output.info(l);
|
|
181
|
+
return report;
|
|
182
|
+
}
|
|
183
|
+
const { total, what, lines } = sayCounts(report, 'from');
|
|
184
|
+
if (total === 0) return report;
|
|
185
|
+
output.info(`Model changed: ${what} will be reused at no cost. To have the new model translate them instead: --redo all --fresh-on-model-change (it sends what an earlier model translated; what the new model already translated comes from the cache).`);
|
|
186
|
+
for (const l of lines) output.info(l);
|
|
187
|
+
return report;
|
|
188
|
+
}
|
|
46
189
|
|
|
47
190
|
/**
|
|
48
191
|
* Parse and validate a --max-cost flag value.
|
|
@@ -72,14 +215,134 @@ export function parseMaxCost(raw) {
|
|
|
72
215
|
* `tmHits` counts keys the Translation Memory already covers ($0).
|
|
73
216
|
* null estimatedCost = unknown, never $0.
|
|
74
217
|
* @property {number} keyCost - Sum of KNOWN per-pair key-value estimates
|
|
75
|
-
* @property {{ files: number, pendingTranslations: number, estimatedCost: number|null,
|
|
76
|
-
*
|
|
77
|
-
*
|
|
218
|
+
* @property {{ files: number, pendingTranslations: number, estimatedCost: number|null,
|
|
219
|
+
* costWithoutTM: number|null, rough: true }|null} content
|
|
220
|
+
* Rough content-file estimate when content is pending, else null.
|
|
221
|
+
* estimatedCost prices only TM misses (what the run bills, and what
|
|
222
|
+
* --max-cost compares); costWithoutTM prices every pending character, to
|
|
223
|
+
* show what the cache saves.
|
|
224
|
+
* @property {number|null} totalEstimatedCost - keyCost + content cost; null
|
|
225
|
+
* when any part is unknown (never a partial sum passed off as the total)
|
|
226
|
+
* @property {number} knownEstimatedCost - The priced part only
|
|
78
227
|
* @property {boolean} hasUnknownCosts - True when ANY pair or the content
|
|
79
228
|
* estimate is unknown. Consumers enforcing --max-cost must treat this as
|
|
80
229
|
* over-cap (unknown ≠ free).
|
|
230
|
+
* @property {{ reason: string, pairs: string[] }} [unknownCost] - Why the
|
|
231
|
+
* total is null (present only when hasUnknownCosts)
|
|
81
232
|
*/
|
|
82
233
|
|
|
234
|
+
/** "Estimated cost: ~$0.0068, --max-cost cap: $0.0010." — one wording for the gate and the dry run. */
|
|
235
|
+
function capLine(maxCost, estimatedCost) {
|
|
236
|
+
const estimateStr = estimatedCost !== null ? `~$${estimatedCost.toFixed(4)}` : 'unknown';
|
|
237
|
+
return `Estimated cost: ${estimateStr}, --max-cost cap: $${maxCost.toFixed(4)}.`;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* The --max-cost verdict on an estimate: would a real run stop at the cap,
|
|
242
|
+
* and why. The ONE rule both the gate (real runs) and the dry-run report use
|
|
243
|
+
* — a dry run never stops, but it must say what the real run would do
|
|
244
|
+
* (Round 5, Next.js persona: `sync --dry --max-cost 0.001` over a ~$0.0068
|
|
245
|
+
* estimate exited 0 and said nothing about the cap).
|
|
246
|
+
*
|
|
247
|
+
* Unknown is not free: a failed estimate or a pair with no published price
|
|
248
|
+
* stops a capped run.
|
|
249
|
+
*
|
|
250
|
+
* @param {number} maxCost - The cap in USD
|
|
251
|
+
* @param {CostEstimateSummary|null} costEstimate
|
|
252
|
+
* @returns {{ cap: number, estimatedCost: number|null, wouldStop: boolean, reason: string|null }}
|
|
253
|
+
*/
|
|
254
|
+
export function maxCostVerdict(maxCost, costEstimate) {
|
|
255
|
+
if (!costEstimate) {
|
|
256
|
+
return { cap: maxCost, estimatedCost: null, wouldStop: true,
|
|
257
|
+
reason: 'Cost estimation failed, so --max-cost cannot be enforced (unknown is not free).' };
|
|
258
|
+
}
|
|
259
|
+
if (costEstimate.hasUnknownCosts) {
|
|
260
|
+
const named = unpricedGroups(costEstimate.pairs || [], costEstimate.content || null)
|
|
261
|
+
.map(g => `${g.subject} (${g.pairs.join(', ')})`).join('; ');
|
|
262
|
+
return { cap: maxCost, estimatedCost: null, wouldStop: true,
|
|
263
|
+
reason: `Some pairs have unknown pricing${named ? ` — no price for ${named}` : ''}, so the total cost cannot be bounded (unknown is not free).` };
|
|
264
|
+
}
|
|
265
|
+
if (costEstimate.totalEstimatedCost > maxCost) {
|
|
266
|
+
return { cap: maxCost, estimatedCost: costEstimate.totalEstimatedCost, wouldStop: true,
|
|
267
|
+
reason: 'Estimated translation cost exceeds the --max-cost cap.' };
|
|
268
|
+
}
|
|
269
|
+
return { cap: maxCost, estimatedCost: costEstimate.totalEstimatedCost, wouldStop: false, reason: null };
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* A dry run's --max-cost report: what the real run would do at the cap.
|
|
274
|
+
* Returned, not printed: the caller says `message` ONCE, at the end of the
|
|
275
|
+
* run beside the preflight verdict (at `level`), and carries the rest in the
|
|
276
|
+
* --json summary (`maxCost: { cap, estimatedCost, wouldStop, exitCode }`).
|
|
277
|
+
* It used to be printed here AND again at the end — the same warning twice,
|
|
278
|
+
* --quiet included (Round 11, Next.js persona).
|
|
279
|
+
* The dry run itself still exits 0 — a preview never fails.
|
|
280
|
+
*
|
|
281
|
+
* @param {number} maxCost
|
|
282
|
+
* @param {CostEstimateSummary|null} costEstimate
|
|
283
|
+
* @param {{ stopsEarlier?: string|null }} [options] - why the real run would
|
|
284
|
+
* stop BEFORE the cap is checked (the preflight), in words; the report then
|
|
285
|
+
* says so instead of "a real run would go ahead"
|
|
286
|
+
* @returns {{ cap: number, estimatedCost: number|null, wouldStop: boolean, reason: string|null, exitCode: number|null,
|
|
287
|
+
* stopsEarlier?: string, message: string, level: 'warn'|'info' }}
|
|
288
|
+
*/
|
|
289
|
+
export function reportDryRunMaxCost(maxCost, costEstimate, { stopsEarlier = null } = {}) {
|
|
290
|
+
const v = maxCostVerdict(maxCost, costEstimate);
|
|
291
|
+
// The preflight runs before the cap is checked: a run that would stop there
|
|
292
|
+
// (a missing key, a model server that does not answer) never reaches the
|
|
293
|
+
// cap, so "a real run would go ahead" was false beside the preflight's
|
|
294
|
+
// "the real sync would STOP" (Round 9, Next.js persona).
|
|
295
|
+
if (stopsEarlier) {
|
|
296
|
+
const message = (v.wouldStop
|
|
297
|
+
? `--max-cost: the estimate is over the cap (${v.reason.replace(/\.$/, '')}), but the run would stop earlier, before the cap is checked, and exit 1: ${stopsEarlier}. ${capLine(maxCost, v.estimatedCost)}`
|
|
298
|
+
: `--max-cost: the estimate is under the cap, but the run would stop earlier and exit 1: ${stopsEarlier}. ${capLine(maxCost, v.estimatedCost)}`)
|
|
299
|
+
+ ` ${dryRunCiHint('preflight.ready')}`;
|
|
300
|
+
return { ...v, exitCode: 1, stopsEarlier, message, level: 'warn' };
|
|
301
|
+
}
|
|
302
|
+
const message = v.wouldStop
|
|
303
|
+
? `A real run would stop at the --max-cost cap before any API call and exit 2: ${v.reason} ${capLine(maxCost, v.estimatedCost)} ${dryRunCiHint('maxCost.wouldStop')}`
|
|
304
|
+
: `--max-cost: the estimate is within the cap — a real run would go ahead. ${capLine(maxCost, v.estimatedCost)}`;
|
|
305
|
+
return { ...v, exitCode: v.wouldStop ? 2 : null, message, level: v.wouldStop ? 'warn' : 'info' };
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/** The CI guide's dry-run check: a step that fails the job when the real run would not go through. */
|
|
309
|
+
export const CI_CHECK_DOC = 'https://champollion.dev/docs/guides/ci-cd#check-before-sync';
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* How to make a CI step fail on what a dry run predicts. A dry run exits 0
|
|
313
|
+
* on purpose (Round 11), so `sync --dry --max-cost` "warned and passed" and
|
|
314
|
+
* could not gate a step on its own (Round 14, Next.js persona): the warning
|
|
315
|
+
* now names the --json field to read, and the guide's step that reads it.
|
|
316
|
+
*
|
|
317
|
+
* @param {string} field - The summary field that says it ('maxCost.wouldStop', 'preflight.ready', …)
|
|
318
|
+
* @returns {string}
|
|
319
|
+
*/
|
|
320
|
+
export function dryRunCiHint(field) {
|
|
321
|
+
const fields = field === 'realRun.exitCode' ? '`realRun.exitCode`' : `\`${field}\` or \`realRun.exitCode\``;
|
|
322
|
+
return `This dry run exits 0 (a preview): to fail a CI step on it, read ${fields} `
|
|
323
|
+
+ `from \`champollion sync --dry --json\` — the CI guide's check step does, and prints the reason: ${CI_CHECK_DOC}`;
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
/**
|
|
327
|
+
* Why a real run would stop at the preflight, in one clause for the dry run's
|
|
328
|
+
* --max-cost line: each reason once, with the pairs it applies to
|
|
329
|
+
* ("No OpenRouter API key (OPENROUTER_API_KEY) for en:de, en:fr"). null when
|
|
330
|
+
* the preflight would pass.
|
|
331
|
+
*
|
|
332
|
+
* @param {Array<{ pair: string, reason: string }>} failures - resolveRuntime's preflightFailures
|
|
333
|
+
* @returns {string|null}
|
|
334
|
+
*/
|
|
335
|
+
export function preflightStopReason(failures) {
|
|
336
|
+
if (!Array.isArray(failures) || failures.length === 0) return null;
|
|
337
|
+
const byReason = new Map();
|
|
338
|
+
for (const f of failures) {
|
|
339
|
+
const reason = String(f.reason).replace(/[.;\s]+$/, '');
|
|
340
|
+
if (!byReason.has(reason)) byReason.set(reason, []);
|
|
341
|
+
byReason.get(reason).push(f.pair);
|
|
342
|
+
}
|
|
343
|
+
return [...byReason].map(([reason, pairs]) => `${reason} for ${pairs.join(', ')}`).join('; ');
|
|
344
|
+
}
|
|
345
|
+
|
|
83
346
|
/**
|
|
84
347
|
* Build the --max-cost abort: print the estimate-vs-cap verdict, emit a
|
|
85
348
|
* structured summary (--json consumers get {error} instead of regexing
|
|
@@ -96,9 +359,7 @@ export function parseMaxCost(raw) {
|
|
|
96
359
|
* @returns {object} runSync result with maxCostAborted set
|
|
97
360
|
*/
|
|
98
361
|
export function abortForMaxCost(maxCost, estimatedCost, reason) {
|
|
99
|
-
const
|
|
100
|
-
const message = `${reason} Estimated cost: ${estimateStr}, --max-cost cap: $${maxCost.toFixed(4)}. ` +
|
|
101
|
-
'Aborting before any API call.';
|
|
362
|
+
const message = `${reason} ${capLine(maxCost, estimatedCost)} Aborting before any API call.`;
|
|
102
363
|
output.error(message);
|
|
103
364
|
output.summary({
|
|
104
365
|
command: 'sync',
|
|
@@ -119,6 +380,199 @@ export function abortForMaxCost(maxCost, estimatedCost, reason) {
|
|
|
119
380
|
};
|
|
120
381
|
}
|
|
121
382
|
|
|
383
|
+
/**
|
|
384
|
+
* Price per-pair content characters through each pair's own estimateCost().
|
|
385
|
+
* Characters become EST_CHARS_PER_KEY-char key-equivalents (exact for
|
|
386
|
+
* char-priced providers; conservative for token-priced LLMs, which pay the
|
|
387
|
+
* per-key prompt overhead on every 25-char slice). A pair with 0 chars is a
|
|
388
|
+
* KNOWN $0 and never consults estimateCost — an unknown-pricing null there
|
|
389
|
+
* would wrongly abort a free run under --max-cost.
|
|
390
|
+
*
|
|
391
|
+
* @param {Map<object, number>} charsByPair - pairConfig → source chars
|
|
392
|
+
* @param {Array<[string, object]>} pairEntries
|
|
393
|
+
* @param {(keys: number, pairConfig: object) => Promise<{estimatedCost: number|null}>} estimateCost
|
|
394
|
+
* @returns {Promise<{ cost: number, unknown: boolean }>}
|
|
395
|
+
*/
|
|
396
|
+
export async function priceContentChars(charsByPair, pairEntries, estimateCost) {
|
|
397
|
+
let cost = 0;
|
|
398
|
+
let unknown = false;
|
|
399
|
+
// The pairs with no price, and why (the method's note), so the table names
|
|
400
|
+
// them instead of "some methods have unknown pricing".
|
|
401
|
+
const unpriced = [];
|
|
402
|
+
// The rates what was priced was priced at (each once).
|
|
403
|
+
const rates = [];
|
|
404
|
+
for (const [pairKey, pairConfig] of pairEntries) {
|
|
405
|
+
const chars = charsByPair.get(pairConfig);
|
|
406
|
+
if (!chars) continue;
|
|
407
|
+
// eslint-disable-next-line no-await-in-loop — sequential is fine for cost queries (cached)
|
|
408
|
+
const estimate = await estimateCost(Math.ceil(chars / EST_CHARS_PER_KEY), pairConfig);
|
|
409
|
+
if (estimate.estimatedCost !== null) {
|
|
410
|
+
cost += estimate.estimatedCost;
|
|
411
|
+
if (estimate.rate) rates.push(estimate.rate);
|
|
412
|
+
} else {
|
|
413
|
+
unknown = true;
|
|
414
|
+
unpriced.push({
|
|
415
|
+
pair: pairKey, method: pairConfig.method || 'llm',
|
|
416
|
+
...(estimate.model && { model: estimate.model }), ...(estimate.note && { note: estimate.note }),
|
|
417
|
+
});
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
return { cost, unknown, unpriced, rates: uniqueRates(rates) };
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
/** A rate's identity: the same model priced from the same list is one rate. */
|
|
424
|
+
const rateId = (r) => JSON.stringify([r.model, r.unit, r.inputPerMillion, r.outputPerMillion, r.perMillionChars, r.from, r.tokensPerKey?.input]);
|
|
425
|
+
|
|
426
|
+
/**
|
|
427
|
+
* Each rate once, in the order first seen.
|
|
428
|
+
*
|
|
429
|
+
* @param {object[]} rates
|
|
430
|
+
* @returns {object[]}
|
|
431
|
+
*/
|
|
432
|
+
function uniqueRates(rates) {
|
|
433
|
+
const seen = new Map();
|
|
434
|
+
for (const r of rates) if (r && !seen.has(rateId(r))) seen.set(rateId(r), r);
|
|
435
|
+
return [...seen.values()];
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
/**
|
|
439
|
+
* Every rate an estimate priced something at: the pairs' and the content's.
|
|
440
|
+
*
|
|
441
|
+
* @param {CostEstimateSummary['pairs']} costEstimates
|
|
442
|
+
* @param {CostEstimateSummary['content']} content
|
|
443
|
+
* @returns {object[]}
|
|
444
|
+
*/
|
|
445
|
+
export function ratesUsed(costEstimates, content) {
|
|
446
|
+
return uniqueRates([
|
|
447
|
+
...costEstimates.filter(e => e.keys > 0 && e.rate).map(e => e.rate),
|
|
448
|
+
...(content?.rates || []),
|
|
449
|
+
]);
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
/** "2026-10-04 14:02 UTC" from an ISO time (null → null). */
|
|
453
|
+
function utcMinute(iso) {
|
|
454
|
+
if (!iso) return null;
|
|
455
|
+
const d = new Date(iso);
|
|
456
|
+
return Number.isNaN(d.getTime()) ? null : `${d.toISOString().slice(0, 16).replace('T', ' ')} UTC`;
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
/**
|
|
460
|
+
* One rate in words: the price per million units, and where it came from.
|
|
461
|
+
*
|
|
462
|
+
* @param {object} r - A method estimate's `rate`
|
|
463
|
+
* @returns {string}
|
|
464
|
+
*/
|
|
465
|
+
function describeRate(r) {
|
|
466
|
+
const price = r.unit === 'char'
|
|
467
|
+
? `$${formatPerMillion(r.perMillionChars)} per 1M characters`
|
|
468
|
+
: `$${formatPerMillion(r.inputPerMillion)} input / $${formatPerMillion(r.outputPerMillion)} output per 1M tokens`;
|
|
469
|
+
let where;
|
|
470
|
+
if (r.from === 'openrouter-price-list') {
|
|
471
|
+
const when = utcMinute(r.fetchedAt);
|
|
472
|
+
where = `OpenRouter's price list${r.proxyFor ? ` (a stand-in for ${r.proxyFor}'s own price)` : ''}, ${when ? `read ${when}` : 'read this run'}`;
|
|
473
|
+
} else if (r.from === 'pinned-table') {
|
|
474
|
+
const why = {
|
|
475
|
+
off: 'the live price draw is off', 'no-price': 'OpenRouter\'s list has no price for it', 'not-listed': 'not on OpenRouter\'s list',
|
|
476
|
+
}[r.liveDraw] || 'OpenRouter\'s list could not be read';
|
|
477
|
+
where = `a copy kept in champollion${r.verified ? `, checked ${r.verified}` : ', never checked'} (${why})`;
|
|
478
|
+
} else if (r.from === 'published-rate') {
|
|
479
|
+
where = `the provider's published price${r.verified ? `, as checked ${r.verified}` : ''}`;
|
|
480
|
+
} else {
|
|
481
|
+
where = r.from || 'unknown source';
|
|
482
|
+
}
|
|
483
|
+
return `${r.model} ${price} — ${where}`;
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
/**
|
|
487
|
+
* The table's rate line: each rate used, where it came from and when, and
|
|
488
|
+
* what the estimate assumes per key — one line; --json carries the detail
|
|
489
|
+
* (`costEstimate.rates`, each pair's `rate`). Null when nothing was priced
|
|
490
|
+
* at a rate (all from the cache, a model on this machine, no price).
|
|
491
|
+
*
|
|
492
|
+
* @param {object[]} rates
|
|
493
|
+
* @returns {string|null}
|
|
494
|
+
*/
|
|
495
|
+
export function rateLine(rates) {
|
|
496
|
+
if (!rates || rates.length === 0) return null;
|
|
497
|
+
const tokens = rates.filter(r => r.unit === 'token');
|
|
498
|
+
const chars = rates.filter(r => r.unit === 'char');
|
|
499
|
+
const assumed = [
|
|
500
|
+
...new Set(tokens.map(r => `~${r.tokensPerKey.input} input + ~${r.tokensPerKey.output} output tokens`)),
|
|
501
|
+
...new Set(chars.map(r => `~${r.charsPerKey} characters`)),
|
|
502
|
+
].join(' or ');
|
|
503
|
+
return ` ${rates.length > 1 ? 'Rates' : 'Rate'}: ${rates.map(describeRate).join('; ')}. `
|
|
504
|
+
+ `An estimate: ${assumed} per key assumed — the bill depends on the real lengths (--json has the detail).`;
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
/**
|
|
508
|
+
* What has no price in an estimate, grouped by what it is: a model (the
|
|
509
|
+
* method looked its price up and found none — its note says why: not in the
|
|
510
|
+
* provider's list, listed without a price, or no list) or a method with no
|
|
511
|
+
* published price at all (a self-hosted endpoint elsewhere). The table's
|
|
512
|
+
* Total line names these; it used to say "no method in this run has
|
|
513
|
+
* published pricing" for a mistyped model slug too (Round 11, Next.js
|
|
514
|
+
* persona).
|
|
515
|
+
*
|
|
516
|
+
* @param {CostEstimateSummary['pairs']} costEstimates
|
|
517
|
+
* @param {CostEstimateSummary['content']} content
|
|
518
|
+
* @returns {Array<{ subject: string, pairs: string[], note: string|null }>}
|
|
519
|
+
*/
|
|
520
|
+
export function unpricedGroups(costEstimates, content) {
|
|
521
|
+
const items = [
|
|
522
|
+
...costEstimates.filter(e => e.estimatedCost === null),
|
|
523
|
+
...(content?.estimatedCost === null && Array.isArray(content.unpriced) ? content.unpriced : []),
|
|
524
|
+
];
|
|
525
|
+
const groups = new Map();
|
|
526
|
+
for (const e of items) {
|
|
527
|
+
const subject = e.model ? `model ${e.model}` : `the ${e.method || 'llm'} method`;
|
|
528
|
+
const note = e.note || null;
|
|
529
|
+
const id = `${subject}\x00${note}`;
|
|
530
|
+
if (!groups.has(id)) groups.set(id, { subject, pairs: [], note });
|
|
531
|
+
const g = groups.get(id);
|
|
532
|
+
if (!g.pairs.includes(e.pair)) g.pairs.push(e.pair);
|
|
533
|
+
}
|
|
534
|
+
return [...groups.values()];
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/**
|
|
538
|
+
* The structured estimate both sync paths return (and --json carries).
|
|
539
|
+
*
|
|
540
|
+
* totalEstimatedCost is null whenever any part is unknown: a number there
|
|
541
|
+
* read as the whole bill — 0 for a run whose only method had no price
|
|
542
|
+
* (Round 3, Next.js persona), while the docs promise unknown is never shown
|
|
543
|
+
* as $0. knownEstimatedCost keeps the priced part; unknownCost says why the
|
|
544
|
+
* total is missing (which pairs, and their methods' own notes).
|
|
545
|
+
*
|
|
546
|
+
* @returns {CostEstimateSummary}
|
|
547
|
+
*/
|
|
548
|
+
export function summarizeEstimate(costEstimates, content, knownTotal, hasUnknownCosts) {
|
|
549
|
+
const unknownPairs = costEstimates.filter(e => e.estimatedCost === null);
|
|
550
|
+
return {
|
|
551
|
+
currency: 'USD',
|
|
552
|
+
pairs: costEstimates,
|
|
553
|
+
keyCost: content && content.estimatedCost !== null
|
|
554
|
+
? knownTotal - content.estimatedCost
|
|
555
|
+
: knownTotal,
|
|
556
|
+
content,
|
|
557
|
+
totalEstimatedCost: hasUnknownCosts ? null : knownTotal,
|
|
558
|
+
knownEstimatedCost: knownTotal,
|
|
559
|
+
// Every rate something was priced at: per 1M tokens (or characters),
|
|
560
|
+
// the list it came from and when it was read (or the date of the copy).
|
|
561
|
+
rates: ratesUsed(costEstimates, content),
|
|
562
|
+
hasUnknownCosts,
|
|
563
|
+
...(hasUnknownCosts && {
|
|
564
|
+
unknownCost: {
|
|
565
|
+
reason: unknownPairs.length > 0
|
|
566
|
+
? 'no published price for: ' + unknownPairs.map(e => `${e.pair} (${e.method}${e.source ? `, ${e.source}` : ''})`).join(', ')
|
|
567
|
+
: 'the content-file estimate has no published price',
|
|
568
|
+
pairs: unknownPairs.map(e => e.pair),
|
|
569
|
+
// What has no price and why, in words (the table's lines).
|
|
570
|
+
notes: unpricedGroups(costEstimates, content),
|
|
571
|
+
},
|
|
572
|
+
}),
|
|
573
|
+
};
|
|
574
|
+
}
|
|
575
|
+
|
|
122
576
|
/**
|
|
123
577
|
* Print the formatted estimate table shared by the standard and Docusaurus
|
|
124
578
|
* sync paths. Pure display — takes an already-computed CostEstimateSummary
|
|
@@ -151,7 +605,13 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
|
|
|
151
605
|
output.raw(` ${'─'.repeat(maxPair)} ${'─'.repeat(maxMethod)} ${'─'.repeat(6)} ${'─'.repeat(7)} ${'─'.repeat(10)}`);
|
|
152
606
|
|
|
153
607
|
for (const e of costEstimates) {
|
|
154
|
-
|
|
608
|
+
// A model on this machine: no API bill, said plainly (not "~$0.0000",
|
|
609
|
+
// which reads like a rounding of a real price). Nothing to send — all
|
|
610
|
+
// of it from the cache, or held back — is free, said so (Round 4,
|
|
611
|
+
// Django persona: a redo served from the cache showed "~$0.0000").
|
|
612
|
+
const costStr = e.local && e.estimatedCost === 0 ? '$0 (local)'
|
|
613
|
+
: e.keys === 0 && e.estimatedCost === 0 ? (e.tmHits > 0 ? 'free (cache)' : '$0 (nothing sent)')
|
|
614
|
+
: e.estimatedCost !== null ? `~$${e.estimatedCost.toFixed(4)}` : 'unknown';
|
|
155
615
|
output.raw(` ${e.pair.padEnd(maxPair)} ${e.method.padEnd(maxMethod)} ${String(e.keys).padStart(6)} ${String(e.tmHits ?? 0).padStart(7)} ${costStr.padStart(10)}`);
|
|
156
616
|
}
|
|
157
617
|
}
|
|
@@ -164,13 +624,53 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
|
|
|
164
624
|
` Content files: ${content.pendingTranslations} pending translation(s) ` +
|
|
165
625
|
`across ${content.files} source file(s) ${contentCostStr} (rough estimate)`
|
|
166
626
|
);
|
|
627
|
+
// What the Translation Memory is worth on this run: the same work priced
|
|
628
|
+
// as if nothing were cached. Shown so the gate's figure is legible — the
|
|
629
|
+
// cap is compared against the TM-discounted number above, never this one.
|
|
630
|
+
if (typeof content.costWithoutTM === 'number' && content.estimatedCost !== null
|
|
631
|
+
&& content.costWithoutTM > content.estimatedCost) {
|
|
632
|
+
output.raw(
|
|
633
|
+
` Without the Translation Memory this content would cost ~$${content.costWithoutTM.toFixed(4)}; ` +
|
|
634
|
+
`cached text saves ~$${(content.costWithoutTM - content.estimatedCost).toFixed(4)}.`
|
|
635
|
+
);
|
|
636
|
+
}
|
|
167
637
|
}
|
|
168
638
|
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
639
|
+
// Nothing priced at all (every row "unknown" — e.g. a local model): say
|
|
640
|
+
// unknown, never "~$0.0000+", which reads as free.
|
|
641
|
+
const anyKnown = costEstimates.some((e) => e.estimatedCost !== null)
|
|
642
|
+
|| Boolean(content && content.estimatedCost !== null && content.estimatedCost !== undefined);
|
|
643
|
+
const allLocal = costEstimates.length > 0 && costEstimates.every(e => e.local || e.estimatedCost === 0)
|
|
644
|
+
&& costEstimates.some(e => e.local) && !hasContentLine;
|
|
645
|
+
// Nothing goes to any method: every queued key comes from the cache (or is held back).
|
|
646
|
+
const nothingSent = costEstimates.length > 0 && costEstimates.every(e => e.keys === 0)
|
|
647
|
+
&& (!hasContentLine || content.estimatedCost === 0) && !hasUnknownCosts;
|
|
648
|
+
// What has no price, named: a model (with why on the line below) or a
|
|
649
|
+
// method with no published price — never blamed on "the method" when it
|
|
650
|
+
// is the model name that is unknown (Round 11, Next.js persona).
|
|
651
|
+
const unpriced = hasUnknownCosts ? unpricedGroups(costEstimates, content) : [];
|
|
652
|
+
const named = unpriced.map(g => `${g.subject} (${g.pairs.join(', ')})`).join('; ');
|
|
653
|
+
const totalStr = nothingSent
|
|
654
|
+
? (costEstimates.some(e => e.tmHits > 0) ? 'free — everything queued is served from the cache' : '$0 — nothing is sent')
|
|
655
|
+
: hasUnknownCosts && !anyKnown
|
|
656
|
+
? (named ? `unknown — no price for ${named}` : 'unknown — no price for what this run sends')
|
|
657
|
+
: hasUnknownCosts
|
|
658
|
+
? `~$${totalEstimatedCost.toFixed(4)}+ — plus ${named || 'what has no price'}, which ${unpriced.length > 1 ? 'have' : 'has'} no price`
|
|
659
|
+
: allLocal
|
|
660
|
+
? '$0 API cost (runs on this machine; your own hardware and power are not counted)'
|
|
661
|
+
: `~$${totalEstimatedCost.toFixed(4)}`;
|
|
172
662
|
output.raw(`\n Total: ${totalStr}`);
|
|
173
|
-
|
|
663
|
+
for (const g of unpriced) {
|
|
664
|
+
if (g.note) output.raw(` ${g.pairs.join(', ')}: ${g.note}`);
|
|
665
|
+
}
|
|
666
|
+
const held = costEstimates.reduce((n, e) => n + (e.held || 0), 0);
|
|
667
|
+
if (held > 0) {
|
|
668
|
+
output.raw(` ${held} queued key(s) are held back — refused before by the same method, so not sent (not billed); the sync names them.`);
|
|
669
|
+
}
|
|
670
|
+
// The rate behind the figure, its source and date — one line (Round 14,
|
|
671
|
+
// Next.js persona: the hosted-model estimate gave no rate, source or date).
|
|
672
|
+
const rates = rateLine(ratesUsed(costEstimates, content));
|
|
673
|
+
output.raw(rates || ' Note: Estimates are approximate. Actual cost depends on string length and model pricing.');
|
|
174
674
|
output.raw('');
|
|
175
675
|
}
|
|
176
676
|
|
|
@@ -194,6 +694,29 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
|
|
|
194
694
|
* lib/no-translate.js. Pass the SAME instance the run will use so exempt
|
|
195
695
|
* keys are excluded from the bill by the same decision that excludes them
|
|
196
696
|
* from translation. Derived from config when omitted.
|
|
697
|
+
* @param {import('./locale-layout.js').LocaleLayout} [options.layout] - The
|
|
698
|
+
* run's locale layout. With `options.units` the estimate walks every
|
|
699
|
+
* source file × target file exactly like sync does (namespaces, real
|
|
700
|
+
* extensions, i18next plural expansion); without them it prices the
|
|
701
|
+
* single legacy file `<localesDir>/<code><ext>` from the positional args.
|
|
702
|
+
* @param {import('./locale-layout.js').SourceUnit[]} [options.units] - The
|
|
703
|
+
* run's source units, each with `changedKeys` in its own key space.
|
|
704
|
+
* @param {import('./locale-state.js').LockState} [options.lockState] - The
|
|
705
|
+
* per-locale record: with it the estimate applies the run's own plan
|
|
706
|
+
* (lib/locale-state.js planQueue) — pending redo keys are priced as sent,
|
|
707
|
+
* keys held back (refused before) and hand edits a bulk redo keeps are not.
|
|
708
|
+
* @param {{ named?: string[], bulk?: boolean, fresh?: boolean }} [options.redo]
|
|
709
|
+
* @param {object} [options.collect] - Filled per pair key (pairs with queued
|
|
710
|
+
* keys) with `{ target, pairConfig, sendTexts, carriedFrom }`: the cached
|
|
711
|
+
* texts of the keys the run would send, and `{ model: count }` of queued
|
|
712
|
+
* keys it would serve from an earlier model's translations (carry-over).
|
|
713
|
+
* Never part of the returned (JSON) estimate.
|
|
714
|
+
* @param {object} [options.collectContent] - Filled per target locale with
|
|
715
|
+
* `{ pendingTranslations, billedChars }` of its pending Markdown (billedChars:
|
|
716
|
+
* what the cache does not hold — what is sent). Never part of the estimate.
|
|
717
|
+
* @param {(ctx: { content: object|null }) => void} [options.beforeTable] - Called
|
|
718
|
+
* once the estimate is computed (options.collect filled), before the table
|
|
719
|
+
* is printed: what must be said before the figure
|
|
197
720
|
* @returns {Promise<CostEstimateSummary|null>} Structured estimate, or null
|
|
198
721
|
* when estimation itself failed (callers with --max-cost must fail safe)
|
|
199
722
|
*/
|
|
@@ -217,36 +740,144 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
|
|
|
217
740
|
let totalEstimatedCost = 0;
|
|
218
741
|
let hasUnknownCosts = false;
|
|
219
742
|
|
|
743
|
+
// The units to price: the run's own (one per source file, any layout),
|
|
744
|
+
// or — for callers that predate layouts — the one flat file the
|
|
745
|
+
// positional arguments describe.
|
|
746
|
+
const units = options.units || [{
|
|
747
|
+
ns: '', flat: sourceFlat, changedKeys, pluralGroups: new Map(), legacy: true,
|
|
748
|
+
}];
|
|
749
|
+
const layout = options.layout || { namespaced: false };
|
|
750
|
+
const legacyFile = (code) => {
|
|
751
|
+
const p = path.join(config.localesDir, `${code}${ext}`);
|
|
752
|
+
return { path: p, format, rel: `${code}${ext}` };
|
|
753
|
+
};
|
|
754
|
+
|
|
220
755
|
for (const [pairKey, pairConfig] of pairEntries) {
|
|
221
756
|
const code = pairConfig.target;
|
|
222
|
-
const
|
|
223
|
-
|
|
757
|
+
const tmKey = tmMethodKey(pairConfig);
|
|
758
|
+
// With a fallback: values it produced are cached under its own key.
|
|
759
|
+
// They confirm echoes, and the sync serves them before asking the
|
|
760
|
+
// primary — so they are free here too (lib/fallback.js).
|
|
761
|
+
const tmKeys = tmKeysForPair(pairConfig);
|
|
762
|
+
let stringKeyCount = 0;
|
|
763
|
+
let missCount = 0;
|
|
764
|
+
// Keys held back (refused before by this method): queued, not sent, $0.
|
|
765
|
+
let heldCount = 0;
|
|
766
|
+
// Source text already priced for this locale in an EARLIER file: the
|
|
767
|
+
// run translates files of one locale in sequence, so the second file
|
|
768
|
+
// gets that text from the TM for free. Mirror that, don't bill twice.
|
|
769
|
+
const pricedTexts = new Set();
|
|
770
|
+
// What the caller may want beyond the totals (options.collect): the
|
|
771
|
+
// texts this run would send, and the keys it would serve from an
|
|
772
|
+
// earlier model's translations (model carry-over), per earlier model.
|
|
773
|
+
const sendTexts = [];
|
|
774
|
+
const carriedFrom = {};
|
|
224
775
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
776
|
+
for (const unit of units) {
|
|
777
|
+
const { flat: expectedFlat, expansion } = unit.legacy
|
|
778
|
+
? { flat: unit.flat, expansion: null }
|
|
779
|
+
: expectedForTarget(unit, config.inputLocale || 'en', code);
|
|
780
|
+
let targetFlat = {};
|
|
781
|
+
let targetFile = null;
|
|
782
|
+
if (unit.legacy) {
|
|
783
|
+
const f = legacyFile(code);
|
|
784
|
+
if (fs.existsSync(f.path)) {
|
|
785
|
+
const data = readLocaleFile(f.path, f.format);
|
|
786
|
+
targetFlat = f.format === 'json' ? flattenKeys(data) : { ...data };
|
|
787
|
+
}
|
|
788
|
+
} else {
|
|
789
|
+
const f = layout.fileFor(code, unit.ns);
|
|
790
|
+
if (fs.existsSync(f.path)) { targetFlat = readLocaleFlat(f); targetFile = f; }
|
|
791
|
+
}
|
|
230
792
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
(
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
793
|
+
// Same confirmed-echo suppression the sync's diff applies, so the
|
|
794
|
+
// estimate prices exactly the keys the run will actually queue —
|
|
795
|
+
// including the run's plan over the per-locale record (pending
|
|
796
|
+
// keys, held-back keys, kept hand edits; lib/locale-state.js).
|
|
797
|
+
const localeState = options.lockState && !unit.legacy ? options.lockState.peek(code) : null;
|
|
798
|
+
const lockKeyOf = (k) => (layout.namespaced ? `${unit.ns}::${k}` : k);
|
|
799
|
+
const pendingKeys = new Set(localeState
|
|
800
|
+
? keysForNamespace(layout, Object.keys(localeState.pending), unit.ns).filter(k => typeof expectedFlat[k] === 'string')
|
|
801
|
+
: []);
|
|
802
|
+
// Plural messages a model left incomplete, asked again this run —
|
|
803
|
+
// the sync's own plan (lib/plural-gap-redo.js), priced as sends.
|
|
804
|
+
const gapKeys = localeState && targetFile
|
|
805
|
+
? planGapRedo({
|
|
806
|
+
file: targetFile, expected: expectedFlat, targetFlat, locale: code, localeState, lockKeyOf, pairConfig,
|
|
807
|
+
all: !!options.redo?.gaps,
|
|
808
|
+
}).keys
|
|
809
|
+
: [];
|
|
810
|
+
const forceKeys = [...new Set([
|
|
811
|
+
...mapSourceKeysToTarget(keysForNamespace(layout, config.forceKeys, unit.ns), expansion),
|
|
812
|
+
...pendingKeys,
|
|
813
|
+
...gapKeys,
|
|
814
|
+
])];
|
|
815
|
+
const unitChanged = mapSourceKeysToTarget(unit.changedKeys || [], expansion);
|
|
816
|
+
const diff = diffLocale(
|
|
817
|
+
expectedFlat, targetFlat, config.fallbackPrefix, forceKeys, unitChanged,
|
|
818
|
+
(key, sourceValue) => tmHoldsValue(tm, tmTextFor(key, sourceValue, expansion), code, tmKeys, sourceValue),
|
|
819
|
+
noTranslate.active ? noTranslate.matches : null
|
|
820
|
+
);
|
|
821
|
+
let queued = diff.toProcess;
|
|
822
|
+
let bypass = new Set();
|
|
823
|
+
let notSent = new Set();
|
|
824
|
+
if (localeState) {
|
|
825
|
+
const redo = options.redo || {};
|
|
826
|
+
const plan = planQueue({
|
|
827
|
+
diff, sourceFlat: expectedFlat, targetFlat, lockKeyOf, localeState,
|
|
828
|
+
named: new Set(mapSourceKeysToTarget(keysForNamespace(layout, redo.named || [], unit.ns), expansion)),
|
|
829
|
+
bulk: !!redo.bulk, pending: pendingKeys, fresh: !!redo.fresh, pairConfig,
|
|
830
|
+
classify: createEditClassifier({ tm, locale: code, written: localeState.written, expansion }),
|
|
831
|
+
fallbackPrefix: config.fallbackPrefix,
|
|
832
|
+
redoGaps: redo.gaps ? new Set(gapKeys) : null,
|
|
833
|
+
});
|
|
834
|
+
queued = plan.toProcess;
|
|
835
|
+
// Pending keys and plural gaps asked again bypass the cache.
|
|
836
|
+
bypass = new Set([...plan.pendingAll, ...gapKeys.filter(k => plan.toProcess.includes(k))]);
|
|
837
|
+
notSent = new Set([...plan.held, ...plan.heldFromPrimary]);
|
|
838
|
+
}
|
|
839
|
+
const stringKeys = queued.filter(k => typeof expectedFlat[k] === 'string');
|
|
840
|
+
if (stringKeys.length === 0) continue;
|
|
241
841
|
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
842
|
+
// Same partition translate-pair.js performs: full method key
|
|
843
|
+
// (method|model|register|coaching), keyed per target locale, on the
|
|
844
|
+
// text each key is CACHED under — a gettext msgctxt is folded in
|
|
845
|
+
// (lib/tm-evict.js), so "Open" (verb) and "Open" (adjective) are two
|
|
846
|
+
// cache entries and two billed strings, exactly as the run sees them.
|
|
847
|
+
const cachedText = {};
|
|
848
|
+
// A borrowed i18next plural form has its own entry (tmTextFor).
|
|
849
|
+
for (const k of stringKeys) cachedText[k] = tmTextFor(k, expectedFlat[k], expansion);
|
|
850
|
+
let { misses } = partitionByTM(tm, cachedText, stringKeys, code, tmKey);
|
|
851
|
+
if (tmKeys.length > 1) {
|
|
852
|
+
misses = misses.filter(k => lookupTM(tm, cachedText[k], code, tmKeys[1]) === null);
|
|
853
|
+
}
|
|
854
|
+
// Pending keys bypass the cache (the model is asked again); keys held
|
|
855
|
+
// back are never sent to the method, so they cost nothing.
|
|
856
|
+
const missSet = new Set(misses);
|
|
857
|
+
heldCount += stringKeys.filter(k => notSent.has(k) && (missSet.has(k) || bypass.has(k))).length;
|
|
858
|
+
misses = stringKeys.filter(k => (missSet.has(k) || bypass.has(k)) && !notSent.has(k));
|
|
859
|
+
for (const k of misses) sendTexts.push(cachedText[k]);
|
|
860
|
+
for (const k of stringKeys) {
|
|
861
|
+
if (missSet.has(k) || bypass.has(k)) continue;
|
|
862
|
+
const model = carriedFromModel(tm, cachedText[k], code, tmKey);
|
|
863
|
+
if (model !== null) carriedFrom[model] = (carriedFrom[model] || 0) + 1;
|
|
864
|
+
}
|
|
865
|
+
stringKeyCount += stringKeys.length;
|
|
866
|
+
for (const k of misses) {
|
|
867
|
+
if (pricedTexts.has(cachedText[k])) continue;
|
|
868
|
+
missCount++;
|
|
869
|
+
}
|
|
870
|
+
// Texts first priced in THIS file still cost here (one batch sends
|
|
871
|
+
// duplicates as separate keys); only later files reuse them.
|
|
872
|
+
for (const k of misses) pricedTexts.add(cachedText[k]);
|
|
873
|
+
}
|
|
874
|
+
if (stringKeyCount === 0) continue;
|
|
875
|
+
const tmHits = stringKeyCount - missCount - heldCount;
|
|
876
|
+
if (options.collect) options.collect[pairKey] = { target: code, pairConfig, sendTexts, carriedFrom };
|
|
246
877
|
|
|
247
|
-
if (
|
|
878
|
+
if (missCount > 0) {
|
|
248
879
|
// eslint-disable-next-line no-await-in-loop — sequential is fine for cost queries (cached)
|
|
249
|
-
const estimate = await estimateCost(
|
|
880
|
+
const estimate = await estimateCost(missCount, pairConfig, { cwd });
|
|
250
881
|
|
|
251
882
|
if (estimate.estimatedCost !== null) {
|
|
252
883
|
totalEstimatedCost += estimate.estimatedCost;
|
|
@@ -257,10 +888,19 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
|
|
|
257
888
|
costEstimates.push({
|
|
258
889
|
pair: pairKey,
|
|
259
890
|
method: pairConfig.method || 'llm',
|
|
260
|
-
keys:
|
|
891
|
+
keys: missCount,
|
|
261
892
|
tmHits,
|
|
893
|
+
...(heldCount > 0 && { held: heldCount }),
|
|
262
894
|
estimatedCost: estimate.estimatedCost,
|
|
263
895
|
source: estimate.source,
|
|
896
|
+
// A model on this machine: $0 API cost, and why (lib/methods/http-utils.js).
|
|
897
|
+
...(estimate.local && { local: true }),
|
|
898
|
+
...(estimate.note && { note: estimate.note }),
|
|
899
|
+
// The rate it was priced at: per 1M tokens (or characters), where
|
|
900
|
+
// the figure came from and when (printCostTable says it in a line).
|
|
901
|
+
...(estimate.rate && { rate: estimate.rate }),
|
|
902
|
+
// No price: the model it was looked up under, so the table can name it.
|
|
903
|
+
...(estimate.estimatedCost === null && estimate.model && { model: estimate.model }),
|
|
264
904
|
});
|
|
265
905
|
} else {
|
|
266
906
|
// Fully TM-covered: zero API calls, so the cost is a KNOWN $0 even
|
|
@@ -272,71 +912,79 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
|
|
|
272
912
|
method: pairConfig.method || 'llm',
|
|
273
913
|
keys: 0,
|
|
274
914
|
tmHits,
|
|
915
|
+
...(heldCount > 0 && { held: heldCount }),
|
|
275
916
|
estimatedCost: 0,
|
|
276
|
-
source: 'translation-memory',
|
|
917
|
+
source: tmHits > 0 ? 'translation-memory' : 'nothing-to-send',
|
|
277
918
|
});
|
|
278
919
|
}
|
|
279
920
|
}
|
|
280
921
|
|
|
281
|
-
// ── Content files (Hugo Markdown)
|
|
922
|
+
// ── Content files (Hugo Markdown) ─────────────────────────────
|
|
282
923
|
// Counts only the (file × pair) translations the sync would actually
|
|
283
|
-
// run (hash-manifest skip logic), then prices each
|
|
284
|
-
//
|
|
285
|
-
// the
|
|
286
|
-
//
|
|
287
|
-
//
|
|
288
|
-
//
|
|
289
|
-
// NOT TM-partitioned (unlike the key-value table above): content-sync
|
|
290
|
-
// caches whole bodies/front-matter fields in the TM, but mirroring that
|
|
291
|
-
// here would mean parsing every pending file — kept rough instead.
|
|
292
|
-
// Rough by design — fail-safe overestimates rather than under.
|
|
924
|
+
// run (hash-manifest skip logic), then prices what each one BILLS: the
|
|
925
|
+
// front-matter fields and body blocks the TM does not already hold —
|
|
926
|
+
// the same ladder runContentSync runs (lib/content-estimate.js). The
|
|
927
|
+
// character → cost step stays rough (see priceContentChars), but the
|
|
928
|
+
// TM discount is exact, so --max-cost compares against what the run
|
|
929
|
+
// will pay rather than the whole file (dogfood 2026-08-28, finding 1).
|
|
293
930
|
let content = null;
|
|
294
931
|
if (config.contentDir) {
|
|
295
932
|
const pending = countPendingContentTranslations(
|
|
296
|
-
config.contentDir, config.inputLocale || 'en', pairEntries, cwd
|
|
933
|
+
config.contentDir, config.inputLocale || 'en', pairEntries, cwd,
|
|
934
|
+
{
|
|
935
|
+
tm, translatableFields: config.translatableFields,
|
|
936
|
+
fileScope: options.fileScope || null, forceContent: !!options.forceContent,
|
|
937
|
+
fresh: !!options.redo?.fresh,
|
|
938
|
+
}
|
|
297
939
|
);
|
|
298
|
-
let contentCost = 0;
|
|
299
|
-
let contentUnknown = false;
|
|
300
940
|
if (pending.pendingTranslations > 0) {
|
|
941
|
+
const billedByPair = new Map();
|
|
942
|
+
const roughByPair = new Map();
|
|
301
943
|
for (const [, pairConfig] of pairEntries) {
|
|
302
944
|
const perTarget = pending.byTarget[pairConfig.target];
|
|
303
945
|
if (!perTarget || perTarget.pendingTranslations === 0) continue;
|
|
304
|
-
|
|
305
|
-
//
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
contentUnknown = true;
|
|
946
|
+
// What the caller may want per target (options.collectContent):
|
|
947
|
+
// whether this run sends Markdown to the pair's method at all.
|
|
948
|
+
if (options.collectContent) {
|
|
949
|
+
options.collectContent[pairConfig.target] = {
|
|
950
|
+
pendingTranslations: perTarget.pendingTranslations, billedChars: perTarget.billedChars,
|
|
951
|
+
};
|
|
311
952
|
}
|
|
953
|
+
billedByPair.set(pairConfig, perTarget.billedChars);
|
|
954
|
+
roughByPair.set(pairConfig, perTarget.pendingSourceChars);
|
|
312
955
|
}
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
956
|
+
const billed = await priceContentChars(billedByPair, pairEntries, estimateCost);
|
|
957
|
+
const rough = await priceContentChars(roughByPair, pairEntries, estimateCost);
|
|
958
|
+
content = {
|
|
959
|
+
files: pending.sourceFileCount,
|
|
960
|
+
pendingTranslations: pending.pendingTranslations,
|
|
961
|
+
estimatedCost: billed.unknown ? null : billed.cost,
|
|
962
|
+
costWithoutTM: rough.unknown ? null : rough.cost,
|
|
963
|
+
rough: true,
|
|
964
|
+
// Which pairs have no price, and why (printCostTable names them).
|
|
965
|
+
...(billed.unknown && { unpriced: billed.unpriced }),
|
|
966
|
+
// The rates the billed part was priced at.
|
|
967
|
+
...(billed.rates.length > 0 && { rates: billed.rates }),
|
|
968
|
+
};
|
|
969
|
+
if (billed.unknown) hasUnknownCosts = true;
|
|
970
|
+
else totalEstimatedCost += billed.cost;
|
|
322
971
|
} else {
|
|
323
|
-
|
|
972
|
+
content = {
|
|
973
|
+
files: pending.sourceFileCount, pendingTranslations: 0,
|
|
974
|
+
estimatedCost: 0, costWithoutTM: 0, rough: true,
|
|
975
|
+
};
|
|
324
976
|
}
|
|
325
977
|
}
|
|
326
978
|
|
|
327
979
|
// ── Display ────────────────────────────────────────────────────
|
|
980
|
+
// What the caller must say BEFORE the figure (options.beforeTable): the
|
|
981
|
+
// model carry-over notice — translations reused from the previous model
|
|
982
|
+
// are part of why the cost is what it is (Round 7, Next.js persona: the
|
|
983
|
+
// table said "TM hits 1, free (cache)" and the reuse was named after it).
|
|
984
|
+
if (typeof options.beforeTable === 'function') options.beforeTable({ content });
|
|
328
985
|
printCostTable(costEstimates, content, totalEstimatedCost, hasUnknownCosts);
|
|
329
986
|
|
|
330
|
-
return
|
|
331
|
-
currency: 'USD',
|
|
332
|
-
pairs: costEstimates,
|
|
333
|
-
keyCost: content && content.estimatedCost !== null
|
|
334
|
-
? totalEstimatedCost - content.estimatedCost
|
|
335
|
-
: totalEstimatedCost,
|
|
336
|
-
content,
|
|
337
|
-
totalEstimatedCost,
|
|
338
|
-
hasUnknownCosts,
|
|
339
|
-
};
|
|
987
|
+
return summarizeEstimate(costEstimates, content, totalEstimatedCost, hasUnknownCosts);
|
|
340
988
|
} catch (costError) {
|
|
341
989
|
// Cost estimation is non-blocking when no cap is set — log and continue.
|
|
342
990
|
// Callers enforcing --max-cost must treat the null return as unknown.
|