champollion 0.3.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/README.md +41 -26
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +34 -0
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +286 -85
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +632 -125
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +15 -9
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +194 -35
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +6 -1
  100. package/lib/seal.mjs +4 -3
  101. package/lib/sealed-qualifier.mjs +1 -1
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +1 -1
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/cards-fallback.json +1 -1
  123. package/shared/curated-orthography-conventions.json +26 -8
  124. package/shared/gettext-plural-forms.json +45 -0
  125. package/shared/method-registry.json +2 -0
  126. package/shared/metric-registry.json +96 -18
  127. package/shared/schemas/champollion-plugin.schema.json +4 -0
  128. package/shared/schemas/corpora-card.schema.json +8 -2
  129. package/shared/schemas/method-index-record.schema.json +67 -0
  130. package/shared/schemas/method-registry.schema.json +4 -0
  131. package/shared/schemas/metric-registry.schema.json +55 -1
  132. package/shared/docent/corpus.json +0 -11739
@@ -38,11 +38,154 @@ import path from 'node:path';
38
38
  import { flattenKeys } from './flatten.js';
39
39
  import { diffLocale } from './diff.js';
40
40
  import { readLocaleFile } from './format.js';
41
+ import { expectedForTarget, keysForNamespace, readLocaleFlat } from './locale-layout.js';
42
+ import { mapSourceKeysToTarget } from './plurals.js';
41
43
  import { EST_CHARS_PER_KEY } from './config.js';
44
+ import { formatPerMillion } from './methods/openrouter-pricing.js';
42
45
  import { countPendingContentTranslations } from './content-sync.js';
43
- import { loadTM, lookupTM, partitionByTM, tmMethodKey } from './tm.js';
46
+ import { loadTM, lookupTM, partitionByTM, tmMethodKey, findModelSwitchStrandedEntries, reusableFromEarlierModels, carriedFromModel } from './tm.js';
47
+ import { tmTextFor } from './tm-evict.js';
44
48
  import { compileNoTranslate } from './no-translate.js';
49
+ import { tmKeysForPair, tmHoldsValue } from './fallback.js';
45
50
  import { output } from './output.js';
51
+ import { planQueue, createEditClassifier } from './locale-state.js';
52
+ import { planGapRedo } from './plural-gap-redo.js';
53
+
54
+ /**
55
+ * Say what a model switch means for this run.
56
+ *
57
+ * By default (model carry-over, see lib/tm.js) translations made under the
58
+ * previous model are reused at no cost — the operator should KNOW that, both
59
+ * because it is a saving and because it is a choice they can reverse. With
60
+ * --fresh-on-model-change the same entries are deliberately bypassed, and
61
+ * the run will re-translate (and bill) that text — say so before spending.
62
+ * Returns the report so callers can carry it into the --json summary.
63
+ *
64
+ * `served` (target → { earlierModel: count }, from the cost estimate's
65
+ * options.collect) limits the carry-over notice to what THIS run reuses: a
66
+ * plain sync after a model switch that queues nothing reads nothing from the
67
+ * cache, and repeating "N cached translations will be reused" on every such
68
+ * run was noise (Round 5, Next.js persona). `champollion status` carries the
69
+ * standing fact (the files hold an earlier model's text). Without `served`
70
+ * the notice counts every cached translation of the current source.
71
+ *
72
+ * @param {object} tm - Loaded TM
73
+ * @param {Iterable<object>} pairConfigs - Resolved pair configs
74
+ * @param {{ fresh?: boolean, redoAll?: boolean, sourceTexts?: Iterable<string>|null,
75
+ * served?: Object<string, Object<string, number>>|null }} [options]
76
+ * @returns {ReturnType<typeof findModelSwitchStrandedEntries>}
77
+ */
78
+ /** "1 translation" / "10 translations". */
79
+ const translationsWord = (n) => `${n} translation${n === 1 ? '' : 's'}`;
80
+
81
+ /**
82
+ * What a count of translations across target languages is made of, in plain
83
+ * units: "5 strings × 2 languages", "5 strings in fr", "3 in fr, 5 in de".
84
+ * The notices summed every locale's count and called the total "source
85
+ * string(s)" — 10 for a 5-key en.json with two targets (Round 12, Next.js
86
+ * persona).
87
+ *
88
+ * @param {Array<{ target: string, n: number }>} perTarget
89
+ * @returns {string}
90
+ */
91
+ export function translationsBreakdown(perTarget) {
92
+ const rows = perTarget.filter(p => p.n > 0);
93
+ const strings = (n) => `${n} string${n === 1 ? '' : 's'}`;
94
+ if (rows.length === 1) return `${strings(rows[0].n)} in ${rows[0].target}`;
95
+ if (rows.length > 1 && rows.every(p => p.n === rows[0].n)) return `${strings(rows[0].n)} × ${rows.length} languages`;
96
+ return rows.map(p => `${p.n} in ${p.target}`).join(', ');
97
+ }
98
+
99
+ export function warnModelSwitchStrandedTM(tm, pairConfigs, { fresh = false, redoAll = false, sourceTexts = null, served = null } = {}) {
100
+ // Iterated twice (the report, then each row's own pair) — an iterator
101
+ // (Map#values()) would be empty the second time.
102
+ const pcs = [...pairConfigs];
103
+ // sourceTexts: count only translations of strings the project still has
104
+ // (an entry for an edited or deleted string is never reused).
105
+ const report = findModelSwitchStrandedEntries(tm, pcs, { sourceTexts });
106
+ // With `served`, what the run reuses is said even when the report is
107
+ // empty (see below).
108
+ if (report.length === 0 && !served) return report;
109
+ // One count for the total and its breakdown (Round 11, Next.js persona:
110
+ // "12 cached translation(s)" above per-model counts adding up to 30 — the
111
+ // total summed each locale's FIRST earlier model, the lines listed every
112
+ // model's holdings, and a string two models translated counted twice).
113
+ // Each string of the current source that only an earlier model translated
114
+ // counts once, under the model whose translation carry-over would reuse
115
+ // (lib/tm.js reusableFromEarlierModels); a project whose Markdown shares
116
+ // the cache counts cache entries instead, and says so.
117
+ const reusable = (r) => reusableFromEarlierModels(tm, pcs.find(pc => pc.target === r.target) || { target: r.target }, r.stranded, { sourceTexts });
118
+ const sayCounts = (rows, verb) => {
119
+ const counted = rows.map(r => ({ r, u: reusable(r) })).filter(x => x.u.total > 0);
120
+ const total = counted.reduce((n, x) => n + x.u.total, 0);
121
+ const unit = counted.some(x => x.u.unit === 'entries') ? 'entries' : 'strings';
122
+ const perTarget = counted.map(x => ({ target: x.r.target, n: x.u.total }));
123
+ // Counted in translations — one per string per language — and said so.
124
+ const what = unit === 'strings'
125
+ ? `${translationsWord(total)} only an earlier model wrote (${translationsBreakdown(perTarget)})`
126
+ : `${translationsWord(total)} cached from earlier models (counted per cache entry, so a string two earlier models `
127
+ + `translated counts twice: ${perTarget.map(p => `${p.n} in ${p.target}`).join(', ')})`;
128
+ const lines = counted.map(({ r, u }) => ` ${r.target}: now ${r.currentModel || '(none)'}; ${u.total} ${verb} ${u.byModel.map(x => `${x.model || '(none)'} (${x.count})`).join(', ')}`);
129
+ return { total, what, lines };
130
+ };
131
+ if (fresh) {
132
+ const { total, what, lines } = sayCounts(report, 'from');
133
+ if (total === 0) return report;
134
+ // "Billed" read wrong above a "$0 (local)" estimate (Round 6, Next.js
135
+ // persona): what happens is that the text is sent to the model again, at
136
+ // the method's price — which the estimate below states.
137
+ const price = 'at the method\'s price (the estimate below; $0 API cost for a model on this machine)';
138
+ output.warn(redoAll
139
+ ? `Model changed: re-translating everything with the new model (--redo all --fresh-on-model-change); ${what} will not be reused — that text is sent to the model again, ${price}.`
140
+ : `Model changed: --fresh-on-model-change set, so ${what} will NOT be reused — but only text this run translates anyway (new or changed keys) is sent to the model again, ${price}. To re-translate ALL of it with the new model, add --redo all.`);
141
+ for (const l of lines) output.warn(l);
142
+ return report;
143
+ }
144
+ if (served) {
145
+ // Only what this run actually serves from an earlier model's cache —
146
+ // counted from what the estimate found, for EVERY locale it serves one
147
+ // in, not only the locales the stranded-entries report names. That report
148
+ // stays silent once a switch is complete (or when the new model holds more
149
+ // entries than the old), so a string reverted to text only the earlier
150
+ // model had translated was served as a free cache hit with no word that
151
+ // the text was that model's — the real sync and `status` said it
152
+ // afterwards (Round 13, Next.js persona; the Translation Memory page
153
+ // promises it before the estimate).
154
+ const every = findModelSwitchStrandedEntries(tm, pcs, { sourceTexts, every: true });
155
+ for (const pc of pcs) {
156
+ const byModel = served[pc.target] || {};
157
+ if (Object.values(byModel).reduce((a, b) => a + b, 0) === 0 || report.some(r => r.target === pc.target)) continue;
158
+ const row = every.find(r => r.target === pc.target) || {
159
+ target: pc.target, currentModel: tmMethodKey(pc).split('|')[1] || '', current: 0,
160
+ stranded: Object.entries(byModel).map(([model, count]) => ({ model, count })),
161
+ };
162
+ report.push(row);
163
+ }
164
+ const lines = [];
165
+ const perTarget = [];
166
+ let total = 0;
167
+ for (const r of report) {
168
+ const byModel = served[r.target] || {};
169
+ const n = Object.values(byModel).reduce((a, b) => a + b, 0);
170
+ r.servedThisRun = n;
171
+ if (n === 0) continue;
172
+ total += n;
173
+ perTarget.push({ target: r.target, n });
174
+ const from = Object.entries(byModel).sort((a, b) => b[1] - a[1]).map(([m, c]) => `${m} (${c})`).join(', ');
175
+ lines.push(` ${r.target}: now ${r.currentModel || '(none)'}; reusing translations from ${from}`);
176
+ }
177
+ if (total === 0) return report;
178
+ output.info(`Model changed: ${translationsWord(total)} this run needs (${translationsBreakdown(perTarget)}) `
179
+ + `${total === 1 ? 'is' : 'are'} served from the previous model's cached translations, at no cost. To have the new model translate them instead: --redo all --fresh-on-model-change (it sends what an earlier model translated; what the new model already translated comes from the cache).`);
180
+ for (const l of lines) output.info(l);
181
+ return report;
182
+ }
183
+ const { total, what, lines } = sayCounts(report, 'from');
184
+ if (total === 0) return report;
185
+ output.info(`Model changed: ${what} will be reused at no cost. To have the new model translate them instead: --redo all --fresh-on-model-change (it sends what an earlier model translated; what the new model already translated comes from the cache).`);
186
+ for (const l of lines) output.info(l);
187
+ return report;
188
+ }
46
189
 
47
190
  /**
48
191
  * Parse and validate a --max-cost flag value.
@@ -72,14 +215,134 @@ export function parseMaxCost(raw) {
72
215
  * `tmHits` counts keys the Translation Memory already covers ($0).
73
216
  * null estimatedCost = unknown, never $0.
74
217
  * @property {number} keyCost - Sum of KNOWN per-pair key-value estimates
75
- * @property {{ files: number, pendingTranslations: number, estimatedCost: number|null, rough: true }|null} content
76
- * Rough content-file estimate when contentDir is configured, else null
77
- * @property {number} totalEstimatedCost - keyCost + known content cost
218
+ * @property {{ files: number, pendingTranslations: number, estimatedCost: number|null,
219
+ * costWithoutTM: number|null, rough: true }|null} content
220
+ * Rough content-file estimate when content is pending, else null.
221
+ * estimatedCost prices only TM misses (what the run bills, and what
222
+ * --max-cost compares); costWithoutTM prices every pending character, to
223
+ * show what the cache saves.
224
+ * @property {number|null} totalEstimatedCost - keyCost + content cost; null
225
+ * when any part is unknown (never a partial sum passed off as the total)
226
+ * @property {number} knownEstimatedCost - The priced part only
78
227
  * @property {boolean} hasUnknownCosts - True when ANY pair or the content
79
228
  * estimate is unknown. Consumers enforcing --max-cost must treat this as
80
229
  * over-cap (unknown ≠ free).
230
+ * @property {{ reason: string, pairs: string[] }} [unknownCost] - Why the
231
+ * total is null (present only when hasUnknownCosts)
81
232
  */
82
233
 
234
+ /** "Estimated cost: ~$0.0068, --max-cost cap: $0.0010." — one wording for the gate and the dry run. */
235
+ function capLine(maxCost, estimatedCost) {
236
+ const estimateStr = estimatedCost !== null ? `~$${estimatedCost.toFixed(4)}` : 'unknown';
237
+ return `Estimated cost: ${estimateStr}, --max-cost cap: $${maxCost.toFixed(4)}.`;
238
+ }
239
+
240
+ /**
241
+ * The --max-cost verdict on an estimate: would a real run stop at the cap,
242
+ * and why. The ONE rule both the gate (real runs) and the dry-run report use
243
+ * — a dry run never stops, but it must say what the real run would do
244
+ * (Round 5, Next.js persona: `sync --dry --max-cost 0.001` over a ~$0.0068
245
+ * estimate exited 0 and said nothing about the cap).
246
+ *
247
+ * Unknown is not free: a failed estimate or a pair with no published price
248
+ * stops a capped run.
249
+ *
250
+ * @param {number} maxCost - The cap in USD
251
+ * @param {CostEstimateSummary|null} costEstimate
252
+ * @returns {{ cap: number, estimatedCost: number|null, wouldStop: boolean, reason: string|null }}
253
+ */
254
+ export function maxCostVerdict(maxCost, costEstimate) {
255
+ if (!costEstimate) {
256
+ return { cap: maxCost, estimatedCost: null, wouldStop: true,
257
+ reason: 'Cost estimation failed, so --max-cost cannot be enforced (unknown is not free).' };
258
+ }
259
+ if (costEstimate.hasUnknownCosts) {
260
+ const named = unpricedGroups(costEstimate.pairs || [], costEstimate.content || null)
261
+ .map(g => `${g.subject} (${g.pairs.join(', ')})`).join('; ');
262
+ return { cap: maxCost, estimatedCost: null, wouldStop: true,
263
+ reason: `Some pairs have unknown pricing${named ? ` — no price for ${named}` : ''}, so the total cost cannot be bounded (unknown is not free).` };
264
+ }
265
+ if (costEstimate.totalEstimatedCost > maxCost) {
266
+ return { cap: maxCost, estimatedCost: costEstimate.totalEstimatedCost, wouldStop: true,
267
+ reason: 'Estimated translation cost exceeds the --max-cost cap.' };
268
+ }
269
+ return { cap: maxCost, estimatedCost: costEstimate.totalEstimatedCost, wouldStop: false, reason: null };
270
+ }
271
+
272
+ /**
273
+ * A dry run's --max-cost report: what the real run would do at the cap.
274
+ * Returned, not printed: the caller says `message` ONCE, at the end of the
275
+ * run beside the preflight verdict (at `level`), and carries the rest in the
276
+ * --json summary (`maxCost: { cap, estimatedCost, wouldStop, exitCode }`).
277
+ * It used to be printed here AND again at the end — the same warning twice,
278
+ * --quiet included (Round 11, Next.js persona).
279
+ * The dry run itself still exits 0 — a preview never fails.
280
+ *
281
+ * @param {number} maxCost
282
+ * @param {CostEstimateSummary|null} costEstimate
283
+ * @param {{ stopsEarlier?: string|null }} [options] - why the real run would
284
+ * stop BEFORE the cap is checked (the preflight), in words; the report then
285
+ * says so instead of "a real run would go ahead"
286
+ * @returns {{ cap: number, estimatedCost: number|null, wouldStop: boolean, reason: string|null, exitCode: number|null,
287
+ * stopsEarlier?: string, message: string, level: 'warn'|'info' }}
288
+ */
289
+ export function reportDryRunMaxCost(maxCost, costEstimate, { stopsEarlier = null } = {}) {
290
+ const v = maxCostVerdict(maxCost, costEstimate);
291
+ // The preflight runs before the cap is checked: a run that would stop there
292
+ // (a missing key, a model server that does not answer) never reaches the
293
+ // cap, so "a real run would go ahead" was false beside the preflight's
294
+ // "the real sync would STOP" (Round 9, Next.js persona).
295
+ if (stopsEarlier) {
296
+ const message = (v.wouldStop
297
+ ? `--max-cost: the estimate is over the cap (${v.reason.replace(/\.$/, '')}), but the run would stop earlier, before the cap is checked, and exit 1: ${stopsEarlier}. ${capLine(maxCost, v.estimatedCost)}`
298
+ : `--max-cost: the estimate is under the cap, but the run would stop earlier and exit 1: ${stopsEarlier}. ${capLine(maxCost, v.estimatedCost)}`)
299
+ + ` ${dryRunCiHint('preflight.ready')}`;
300
+ return { ...v, exitCode: 1, stopsEarlier, message, level: 'warn' };
301
+ }
302
+ const message = v.wouldStop
303
+ ? `A real run would stop at the --max-cost cap before any API call and exit 2: ${v.reason} ${capLine(maxCost, v.estimatedCost)} ${dryRunCiHint('maxCost.wouldStop')}`
304
+ : `--max-cost: the estimate is within the cap — a real run would go ahead. ${capLine(maxCost, v.estimatedCost)}`;
305
+ return { ...v, exitCode: v.wouldStop ? 2 : null, message, level: v.wouldStop ? 'warn' : 'info' };
306
+ }
307
+
308
+ /** The CI guide's dry-run check: a step that fails the job when the real run would not go through. */
309
+ export const CI_CHECK_DOC = 'https://champollion.dev/docs/guides/ci-cd#check-before-sync';
310
+
311
+ /**
312
+ * How to make a CI step fail on what a dry run predicts. A dry run exits 0
313
+ * on purpose (Round 11), so `sync --dry --max-cost` "warned and passed" and
314
+ * could not gate a step on its own (Round 14, Next.js persona): the warning
315
+ * now names the --json field to read, and the guide's step that reads it.
316
+ *
317
+ * @param {string} field - The summary field that says it ('maxCost.wouldStop', 'preflight.ready', …)
318
+ * @returns {string}
319
+ */
320
+ export function dryRunCiHint(field) {
321
+ const fields = field === 'realRun.exitCode' ? '`realRun.exitCode`' : `\`${field}\` or \`realRun.exitCode\``;
322
+ return `This dry run exits 0 (a preview): to fail a CI step on it, read ${fields} `
323
+ + `from \`champollion sync --dry --json\` — the CI guide's check step does, and prints the reason: ${CI_CHECK_DOC}`;
324
+ }
325
+
326
+ /**
327
+ * Why a real run would stop at the preflight, in one clause for the dry run's
328
+ * --max-cost line: each reason once, with the pairs it applies to
329
+ * ("No OpenRouter API key (OPENROUTER_API_KEY) for en:de, en:fr"). null when
330
+ * the preflight would pass.
331
+ *
332
+ * @param {Array<{ pair: string, reason: string }>} failures - resolveRuntime's preflightFailures
333
+ * @returns {string|null}
334
+ */
335
+ export function preflightStopReason(failures) {
336
+ if (!Array.isArray(failures) || failures.length === 0) return null;
337
+ const byReason = new Map();
338
+ for (const f of failures) {
339
+ const reason = String(f.reason).replace(/[.;\s]+$/, '');
340
+ if (!byReason.has(reason)) byReason.set(reason, []);
341
+ byReason.get(reason).push(f.pair);
342
+ }
343
+ return [...byReason].map(([reason, pairs]) => `${reason} for ${pairs.join(', ')}`).join('; ');
344
+ }
345
+
83
346
  /**
84
347
  * Build the --max-cost abort: print the estimate-vs-cap verdict, emit a
85
348
  * structured summary (--json consumers get {error} instead of regexing
@@ -96,9 +359,7 @@ export function parseMaxCost(raw) {
96
359
  * @returns {object} runSync result with maxCostAborted set
97
360
  */
98
361
  export function abortForMaxCost(maxCost, estimatedCost, reason) {
99
- const estimateStr = estimatedCost !== null ? `~$${estimatedCost.toFixed(4)}` : 'unknown';
100
- const message = `${reason} Estimated cost: ${estimateStr}, --max-cost cap: $${maxCost.toFixed(4)}. ` +
101
- 'Aborting before any API call.';
362
+ const message = `${reason} ${capLine(maxCost, estimatedCost)} Aborting before any API call.`;
102
363
  output.error(message);
103
364
  output.summary({
104
365
  command: 'sync',
@@ -119,6 +380,199 @@ export function abortForMaxCost(maxCost, estimatedCost, reason) {
119
380
  };
120
381
  }
121
382
 
383
+ /**
384
+ * Price per-pair content characters through each pair's own estimateCost().
385
+ * Characters become EST_CHARS_PER_KEY-char key-equivalents (exact for
386
+ * char-priced providers; conservative for token-priced LLMs, which pay the
387
+ * per-key prompt overhead on every 25-char slice). A pair with 0 chars is a
388
+ * KNOWN $0 and never consults estimateCost — an unknown-pricing null there
389
+ * would wrongly abort a free run under --max-cost.
390
+ *
391
+ * @param {Map<object, number>} charsByPair - pairConfig → source chars
392
+ * @param {Array<[string, object]>} pairEntries
393
+ * @param {(keys: number, pairConfig: object) => Promise<{estimatedCost: number|null}>} estimateCost
394
+ * @returns {Promise<{ cost: number, unknown: boolean }>}
395
+ */
396
+ export async function priceContentChars(charsByPair, pairEntries, estimateCost) {
397
+ let cost = 0;
398
+ let unknown = false;
399
+ // The pairs with no price, and why (the method's note), so the table names
400
+ // them instead of "some methods have unknown pricing".
401
+ const unpriced = [];
402
+ // The rates what was priced was priced at (each once).
403
+ const rates = [];
404
+ for (const [pairKey, pairConfig] of pairEntries) {
405
+ const chars = charsByPair.get(pairConfig);
406
+ if (!chars) continue;
407
+ // eslint-disable-next-line no-await-in-loop — sequential is fine for cost queries (cached)
408
+ const estimate = await estimateCost(Math.ceil(chars / EST_CHARS_PER_KEY), pairConfig);
409
+ if (estimate.estimatedCost !== null) {
410
+ cost += estimate.estimatedCost;
411
+ if (estimate.rate) rates.push(estimate.rate);
412
+ } else {
413
+ unknown = true;
414
+ unpriced.push({
415
+ pair: pairKey, method: pairConfig.method || 'llm',
416
+ ...(estimate.model && { model: estimate.model }), ...(estimate.note && { note: estimate.note }),
417
+ });
418
+ }
419
+ }
420
+ return { cost, unknown, unpriced, rates: uniqueRates(rates) };
421
+ }
422
+
423
+ /** A rate's identity: the same model priced from the same list is one rate. */
424
+ const rateId = (r) => JSON.stringify([r.model, r.unit, r.inputPerMillion, r.outputPerMillion, r.perMillionChars, r.from, r.tokensPerKey?.input]);
425
+
426
+ /**
427
+ * Each rate once, in the order first seen.
428
+ *
429
+ * @param {object[]} rates
430
+ * @returns {object[]}
431
+ */
432
+ function uniqueRates(rates) {
433
+ const seen = new Map();
434
+ for (const r of rates) if (r && !seen.has(rateId(r))) seen.set(rateId(r), r);
435
+ return [...seen.values()];
436
+ }
437
+
438
+ /**
439
+ * Every rate an estimate priced something at: the pairs' and the content's.
440
+ *
441
+ * @param {CostEstimateSummary['pairs']} costEstimates
442
+ * @param {CostEstimateSummary['content']} content
443
+ * @returns {object[]}
444
+ */
445
+ export function ratesUsed(costEstimates, content) {
446
+ return uniqueRates([
447
+ ...costEstimates.filter(e => e.keys > 0 && e.rate).map(e => e.rate),
448
+ ...(content?.rates || []),
449
+ ]);
450
+ }
451
+
452
+ /** "2026-10-04 14:02 UTC" from an ISO time (null → null). */
453
+ function utcMinute(iso) {
454
+ if (!iso) return null;
455
+ const d = new Date(iso);
456
+ return Number.isNaN(d.getTime()) ? null : `${d.toISOString().slice(0, 16).replace('T', ' ')} UTC`;
457
+ }
458
+
459
+ /**
460
+ * One rate in words: the price per million units, and where it came from.
461
+ *
462
+ * @param {object} r - A method estimate's `rate`
463
+ * @returns {string}
464
+ */
465
+ function describeRate(r) {
466
+ const price = r.unit === 'char'
467
+ ? `$${formatPerMillion(r.perMillionChars)} per 1M characters`
468
+ : `$${formatPerMillion(r.inputPerMillion)} input / $${formatPerMillion(r.outputPerMillion)} output per 1M tokens`;
469
+ let where;
470
+ if (r.from === 'openrouter-price-list') {
471
+ const when = utcMinute(r.fetchedAt);
472
+ where = `OpenRouter's price list${r.proxyFor ? ` (a stand-in for ${r.proxyFor}'s own price)` : ''}, ${when ? `read ${when}` : 'read this run'}`;
473
+ } else if (r.from === 'pinned-table') {
474
+ const why = {
475
+ off: 'the live price draw is off', 'no-price': 'OpenRouter\'s list has no price for it', 'not-listed': 'not on OpenRouter\'s list',
476
+ }[r.liveDraw] || 'OpenRouter\'s list could not be read';
477
+ where = `a copy kept in champollion${r.verified ? `, checked ${r.verified}` : ', never checked'} (${why})`;
478
+ } else if (r.from === 'published-rate') {
479
+ where = `the provider's published price${r.verified ? `, as checked ${r.verified}` : ''}`;
480
+ } else {
481
+ where = r.from || 'unknown source';
482
+ }
483
+ return `${r.model} ${price} — ${where}`;
484
+ }
485
+
486
+ /**
487
+ * The table's rate line: each rate used, where it came from and when, and
488
+ * what the estimate assumes per key — one line; --json carries the detail
489
+ * (`costEstimate.rates`, each pair's `rate`). Null when nothing was priced
490
+ * at a rate (all from the cache, a model on this machine, no price).
491
+ *
492
+ * @param {object[]} rates
493
+ * @returns {string|null}
494
+ */
495
+ export function rateLine(rates) {
496
+ if (!rates || rates.length === 0) return null;
497
+ const tokens = rates.filter(r => r.unit === 'token');
498
+ const chars = rates.filter(r => r.unit === 'char');
499
+ const assumed = [
500
+ ...new Set(tokens.map(r => `~${r.tokensPerKey.input} input + ~${r.tokensPerKey.output} output tokens`)),
501
+ ...new Set(chars.map(r => `~${r.charsPerKey} characters`)),
502
+ ].join(' or ');
503
+ return ` ${rates.length > 1 ? 'Rates' : 'Rate'}: ${rates.map(describeRate).join('; ')}. `
504
+ + `An estimate: ${assumed} per key assumed — the bill depends on the real lengths (--json has the detail).`;
505
+ }
506
+
507
+ /**
508
+ * What has no price in an estimate, grouped by what it is: a model (the
509
+ * method looked its price up and found none — its note says why: not in the
510
+ * provider's list, listed without a price, or no list) or a method with no
511
+ * published price at all (a self-hosted endpoint elsewhere). The table's
512
+ * Total line names these; it used to say "no method in this run has
513
+ * published pricing" for a mistyped model slug too (Round 11, Next.js
514
+ * persona).
515
+ *
516
+ * @param {CostEstimateSummary['pairs']} costEstimates
517
+ * @param {CostEstimateSummary['content']} content
518
+ * @returns {Array<{ subject: string, pairs: string[], note: string|null }>}
519
+ */
520
+ export function unpricedGroups(costEstimates, content) {
521
+ const items = [
522
+ ...costEstimates.filter(e => e.estimatedCost === null),
523
+ ...(content?.estimatedCost === null && Array.isArray(content.unpriced) ? content.unpriced : []),
524
+ ];
525
+ const groups = new Map();
526
+ for (const e of items) {
527
+ const subject = e.model ? `model ${e.model}` : `the ${e.method || 'llm'} method`;
528
+ const note = e.note || null;
529
+ const id = `${subject}\x00${note}`;
530
+ if (!groups.has(id)) groups.set(id, { subject, pairs: [], note });
531
+ const g = groups.get(id);
532
+ if (!g.pairs.includes(e.pair)) g.pairs.push(e.pair);
533
+ }
534
+ return [...groups.values()];
535
+ }
536
+
537
+ /**
538
+ * The structured estimate both sync paths return (and --json carries).
539
+ *
540
+ * totalEstimatedCost is null whenever any part is unknown: a number there
541
+ * read as the whole bill — 0 for a run whose only method had no price
542
+ * (Round 3, Next.js persona), while the docs promise unknown is never shown
543
+ * as $0. knownEstimatedCost keeps the priced part; unknownCost says why the
544
+ * total is missing (which pairs, and their methods' own notes).
545
+ *
546
+ * @returns {CostEstimateSummary}
547
+ */
548
+ export function summarizeEstimate(costEstimates, content, knownTotal, hasUnknownCosts) {
549
+ const unknownPairs = costEstimates.filter(e => e.estimatedCost === null);
550
+ return {
551
+ currency: 'USD',
552
+ pairs: costEstimates,
553
+ keyCost: content && content.estimatedCost !== null
554
+ ? knownTotal - content.estimatedCost
555
+ : knownTotal,
556
+ content,
557
+ totalEstimatedCost: hasUnknownCosts ? null : knownTotal,
558
+ knownEstimatedCost: knownTotal,
559
+ // Every rate something was priced at: per 1M tokens (or characters),
560
+ // the list it came from and when it was read (or the date of the copy).
561
+ rates: ratesUsed(costEstimates, content),
562
+ hasUnknownCosts,
563
+ ...(hasUnknownCosts && {
564
+ unknownCost: {
565
+ reason: unknownPairs.length > 0
566
+ ? 'no published price for: ' + unknownPairs.map(e => `${e.pair} (${e.method}${e.source ? `, ${e.source}` : ''})`).join(', ')
567
+ : 'the content-file estimate has no published price',
568
+ pairs: unknownPairs.map(e => e.pair),
569
+ // What has no price and why, in words (the table's lines).
570
+ notes: unpricedGroups(costEstimates, content),
571
+ },
572
+ }),
573
+ };
574
+ }
575
+
122
576
  /**
123
577
  * Print the formatted estimate table shared by the standard and Docusaurus
124
578
  * sync paths. Pure display — takes an already-computed CostEstimateSummary
@@ -151,7 +605,13 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
151
605
  output.raw(` ${'─'.repeat(maxPair)} ${'─'.repeat(maxMethod)} ${'─'.repeat(6)} ${'─'.repeat(7)} ${'─'.repeat(10)}`);
152
606
 
153
607
  for (const e of costEstimates) {
154
- const costStr = e.estimatedCost !== null ? `~$${e.estimatedCost.toFixed(4)}` : 'unknown';
608
+ // A model on this machine: no API bill, said plainly (not "~$0.0000",
609
+ // which reads like a rounding of a real price). Nothing to send — all
610
+ // of it from the cache, or held back — is free, said so (Round 4,
611
+ // Django persona: a redo served from the cache showed "~$0.0000").
612
+ const costStr = e.local && e.estimatedCost === 0 ? '$0 (local)'
613
+ : e.keys === 0 && e.estimatedCost === 0 ? (e.tmHits > 0 ? 'free (cache)' : '$0 (nothing sent)')
614
+ : e.estimatedCost !== null ? `~$${e.estimatedCost.toFixed(4)}` : 'unknown';
155
615
  output.raw(` ${e.pair.padEnd(maxPair)} ${e.method.padEnd(maxMethod)} ${String(e.keys).padStart(6)} ${String(e.tmHits ?? 0).padStart(7)} ${costStr.padStart(10)}`);
156
616
  }
157
617
  }
@@ -164,13 +624,53 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
164
624
  ` Content files: ${content.pendingTranslations} pending translation(s) ` +
165
625
  `across ${content.files} source file(s) ${contentCostStr} (rough estimate)`
166
626
  );
627
+ // What the Translation Memory is worth on this run: the same work priced
628
+ // as if nothing were cached. Shown so the gate's figure is legible — the
629
+ // cap is compared against the TM-discounted number above, never this one.
630
+ if (typeof content.costWithoutTM === 'number' && content.estimatedCost !== null
631
+ && content.costWithoutTM > content.estimatedCost) {
632
+ output.raw(
633
+ ` Without the Translation Memory this content would cost ~$${content.costWithoutTM.toFixed(4)}; ` +
634
+ `cached text saves ~$${(content.costWithoutTM - content.estimatedCost).toFixed(4)}.`
635
+ );
636
+ }
167
637
  }
168
638
 
169
- const totalStr = hasUnknownCosts
170
- ? `~$${totalEstimatedCost.toFixed(4)}+ (some methods have unknown pricing)`
171
- : `~$${totalEstimatedCost.toFixed(4)}`;
639
+ // Nothing priced at all (every row "unknown" — e.g. a local model): say
640
+ // unknown, never "~$0.0000+", which reads as free.
641
+ const anyKnown = costEstimates.some((e) => e.estimatedCost !== null)
642
+ || Boolean(content && content.estimatedCost !== null && content.estimatedCost !== undefined);
643
+ const allLocal = costEstimates.length > 0 && costEstimates.every(e => e.local || e.estimatedCost === 0)
644
+ && costEstimates.some(e => e.local) && !hasContentLine;
645
+ // Nothing goes to any method: every queued key comes from the cache (or is held back).
646
+ const nothingSent = costEstimates.length > 0 && costEstimates.every(e => e.keys === 0)
647
+ && (!hasContentLine || content.estimatedCost === 0) && !hasUnknownCosts;
648
+ // What has no price, named: a model (with why on the line below) or a
649
+ // method with no published price — never blamed on "the method" when it
650
+ // is the model name that is unknown (Round 11, Next.js persona).
651
+ const unpriced = hasUnknownCosts ? unpricedGroups(costEstimates, content) : [];
652
+ const named = unpriced.map(g => `${g.subject} (${g.pairs.join(', ')})`).join('; ');
653
+ const totalStr = nothingSent
654
+ ? (costEstimates.some(e => e.tmHits > 0) ? 'free — everything queued is served from the cache' : '$0 — nothing is sent')
655
+ : hasUnknownCosts && !anyKnown
656
+ ? (named ? `unknown — no price for ${named}` : 'unknown — no price for what this run sends')
657
+ : hasUnknownCosts
658
+ ? `~$${totalEstimatedCost.toFixed(4)}+ — plus ${named || 'what has no price'}, which ${unpriced.length > 1 ? 'have' : 'has'} no price`
659
+ : allLocal
660
+ ? '$0 API cost (runs on this machine; your own hardware and power are not counted)'
661
+ : `~$${totalEstimatedCost.toFixed(4)}`;
172
662
  output.raw(`\n Total: ${totalStr}`);
173
- output.raw(' Note: Estimates are approximate. Actual cost depends on string length and model pricing.');
663
+ for (const g of unpriced) {
664
+ if (g.note) output.raw(` ${g.pairs.join(', ')}: ${g.note}`);
665
+ }
666
+ const held = costEstimates.reduce((n, e) => n + (e.held || 0), 0);
667
+ if (held > 0) {
668
+ output.raw(` ${held} queued key(s) are held back — refused before by the same method, so not sent (not billed); the sync names them.`);
669
+ }
670
+ // The rate behind the figure, its source and date — one line (Round 14,
671
+ // Next.js persona: the hosted-model estimate gave no rate, source or date).
672
+ const rates = rateLine(ratesUsed(costEstimates, content));
673
+ output.raw(rates || ' Note: Estimates are approximate. Actual cost depends on string length and model pricing.');
174
674
  output.raw('');
175
675
  }
176
676
 
@@ -194,6 +694,29 @@ export function printCostTable(costEstimates, content, totalEstimatedCost, hasUn
194
694
  * lib/no-translate.js. Pass the SAME instance the run will use so exempt
195
695
  * keys are excluded from the bill by the same decision that excludes them
196
696
  * from translation. Derived from config when omitted.
697
+ * @param {import('./locale-layout.js').LocaleLayout} [options.layout] - The
698
+ * run's locale layout. With `options.units` the estimate walks every
699
+ * source file × target file exactly like sync does (namespaces, real
700
+ * extensions, i18next plural expansion); without them it prices the
701
+ * single legacy file `<localesDir>/<code><ext>` from the positional args.
702
+ * @param {import('./locale-layout.js').SourceUnit[]} [options.units] - The
703
+ * run's source units, each with `changedKeys` in its own key space.
704
+ * @param {import('./locale-state.js').LockState} [options.lockState] - The
705
+ * per-locale record: with it the estimate applies the run's own plan
706
+ * (lib/locale-state.js planQueue) — pending redo keys are priced as sent,
707
+ * keys held back (refused before) and hand edits a bulk redo keeps are not.
708
+ * @param {{ named?: string[], bulk?: boolean, fresh?: boolean }} [options.redo]
709
+ * @param {object} [options.collect] - Filled per pair key (pairs with queued
710
+ * keys) with `{ target, pairConfig, sendTexts, carriedFrom }`: the cached
711
+ * texts of the keys the run would send, and `{ model: count }` of queued
712
+ * keys it would serve from an earlier model's translations (carry-over).
713
+ * Never part of the returned (JSON) estimate.
714
+ * @param {object} [options.collectContent] - Filled per target locale with
715
+ * `{ pendingTranslations, billedChars }` of its pending Markdown (billedChars:
716
+ * what the cache does not hold — what is sent). Never part of the estimate.
717
+ * @param {(ctx: { content: object|null }) => void} [options.beforeTable] - Called
718
+ * once the estimate is computed (options.collect filled), before the table
719
+ * is printed: what must be said before the figure
197
720
  * @returns {Promise<CostEstimateSummary|null>} Structured estimate, or null
198
721
  * when estimation itself failed (callers with --max-cost must fail safe)
199
722
  */
@@ -217,36 +740,144 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
217
740
  let totalEstimatedCost = 0;
218
741
  let hasUnknownCosts = false;
219
742
 
743
+ // The units to price: the run's own (one per source file, any layout),
744
+ // or — for callers that predate layouts — the one flat file the
745
+ // positional arguments describe.
746
+ const units = options.units || [{
747
+ ns: '', flat: sourceFlat, changedKeys, pluralGroups: new Map(), legacy: true,
748
+ }];
749
+ const layout = options.layout || { namespaced: false };
750
+ const legacyFile = (code) => {
751
+ const p = path.join(config.localesDir, `${code}${ext}`);
752
+ return { path: p, format, rel: `${code}${ext}` };
753
+ };
754
+
220
755
  for (const [pairKey, pairConfig] of pairEntries) {
221
756
  const code = pairConfig.target;
222
- const filename = `${code}${ext}`;
223
- const filePath = path.join(config.localesDir, filename);
757
+ const tmKey = tmMethodKey(pairConfig);
758
+ // With a fallback: values it produced are cached under its own key.
759
+ // They confirm echoes, and the sync serves them before asking the
760
+ // primary — so they are free here too (lib/fallback.js).
761
+ const tmKeys = tmKeysForPair(pairConfig);
762
+ let stringKeyCount = 0;
763
+ let missCount = 0;
764
+ // Keys held back (refused before by this method): queued, not sent, $0.
765
+ let heldCount = 0;
766
+ // Source text already priced for this locale in an EARLIER file: the
767
+ // run translates files of one locale in sequence, so the second file
768
+ // gets that text from the TM for free. Mirror that, don't bill twice.
769
+ const pricedTexts = new Set();
770
+ // What the caller may want beyond the totals (options.collect): the
771
+ // texts this run would send, and the keys it would serve from an
772
+ // earlier model's translations (model carry-over), per earlier model.
773
+ const sendTexts = [];
774
+ const carriedFrom = {};
224
775
 
225
- let targetFlat = {};
226
- if (fs.existsSync(filePath)) {
227
- const data = readLocaleFile(filePath, format);
228
- targetFlat = format === 'json' ? flattenKeys(data) : { ...data };
229
- }
776
+ for (const unit of units) {
777
+ const { flat: expectedFlat, expansion } = unit.legacy
778
+ ? { flat: unit.flat, expansion: null }
779
+ : expectedForTarget(unit, config.inputLocale || 'en', code);
780
+ let targetFlat = {};
781
+ let targetFile = null;
782
+ if (unit.legacy) {
783
+ const f = legacyFile(code);
784
+ if (fs.existsSync(f.path)) {
785
+ const data = readLocaleFile(f.path, f.format);
786
+ targetFlat = f.format === 'json' ? flattenKeys(data) : { ...data };
787
+ }
788
+ } else {
789
+ const f = layout.fileFor(code, unit.ns);
790
+ if (fs.existsSync(f.path)) { targetFlat = readLocaleFlat(f); targetFile = f; }
791
+ }
230
792
 
231
- // Same confirmed-echo suppression the sync's diff applies, so the
232
- // estimate prices exactly the keys the run will actually queue.
233
- const tmKey = tmMethodKey(pairConfig);
234
- const diff = diffLocale(
235
- sourceFlat, targetFlat, config.fallbackPrefix, config.forceKeys, changedKeys,
236
- (key, sourceValue) => lookupTM(tm, sourceValue, code, tmKey) === sourceValue,
237
- noTranslate.active ? noTranslate.matches : null
238
- );
239
- const stringKeys = diff.toProcess.filter(k => typeof sourceFlat[k] === 'string');
240
- if (stringKeys.length === 0) continue;
793
+ // Same confirmed-echo suppression the sync's diff applies, so the
794
+ // estimate prices exactly the keys the run will actually queue —
795
+ // including the run's plan over the per-locale record (pending
796
+ // keys, held-back keys, kept hand edits; lib/locale-state.js).
797
+ const localeState = options.lockState && !unit.legacy ? options.lockState.peek(code) : null;
798
+ const lockKeyOf = (k) => (layout.namespaced ? `${unit.ns}::${k}` : k);
799
+ const pendingKeys = new Set(localeState
800
+ ? keysForNamespace(layout, Object.keys(localeState.pending), unit.ns).filter(k => typeof expectedFlat[k] === 'string')
801
+ : []);
802
+ // Plural messages a model left incomplete, asked again this run —
803
+ // the sync's own plan (lib/plural-gap-redo.js), priced as sends.
804
+ const gapKeys = localeState && targetFile
805
+ ? planGapRedo({
806
+ file: targetFile, expected: expectedFlat, targetFlat, locale: code, localeState, lockKeyOf, pairConfig,
807
+ all: !!options.redo?.gaps,
808
+ }).keys
809
+ : [];
810
+ const forceKeys = [...new Set([
811
+ ...mapSourceKeysToTarget(keysForNamespace(layout, config.forceKeys, unit.ns), expansion),
812
+ ...pendingKeys,
813
+ ...gapKeys,
814
+ ])];
815
+ const unitChanged = mapSourceKeysToTarget(unit.changedKeys || [], expansion);
816
+ const diff = diffLocale(
817
+ expectedFlat, targetFlat, config.fallbackPrefix, forceKeys, unitChanged,
818
+ (key, sourceValue) => tmHoldsValue(tm, tmTextFor(key, sourceValue, expansion), code, tmKeys, sourceValue),
819
+ noTranslate.active ? noTranslate.matches : null
820
+ );
821
+ let queued = diff.toProcess;
822
+ let bypass = new Set();
823
+ let notSent = new Set();
824
+ if (localeState) {
825
+ const redo = options.redo || {};
826
+ const plan = planQueue({
827
+ diff, sourceFlat: expectedFlat, targetFlat, lockKeyOf, localeState,
828
+ named: new Set(mapSourceKeysToTarget(keysForNamespace(layout, redo.named || [], unit.ns), expansion)),
829
+ bulk: !!redo.bulk, pending: pendingKeys, fresh: !!redo.fresh, pairConfig,
830
+ classify: createEditClassifier({ tm, locale: code, written: localeState.written, expansion }),
831
+ fallbackPrefix: config.fallbackPrefix,
832
+ redoGaps: redo.gaps ? new Set(gapKeys) : null,
833
+ });
834
+ queued = plan.toProcess;
835
+ // Pending keys and plural gaps asked again bypass the cache.
836
+ bypass = new Set([...plan.pendingAll, ...gapKeys.filter(k => plan.toProcess.includes(k))]);
837
+ notSent = new Set([...plan.held, ...plan.heldFromPrimary]);
838
+ }
839
+ const stringKeys = queued.filter(k => typeof expectedFlat[k] === 'string');
840
+ if (stringKeys.length === 0) continue;
241
841
 
242
- // Same partition translate-pair.js performs: full method key
243
- // (method|model|register|coaching), keyed per target locale.
244
- const { misses } = partitionByTM(tm, sourceFlat, stringKeys, code, tmKey);
245
- const tmHits = stringKeys.length - misses.length;
842
+ // Same partition translate-pair.js performs: full method key
843
+ // (method|model|register|coaching), keyed per target locale, on the
844
+ // text each key is CACHED under — a gettext msgctxt is folded in
845
+ // (lib/tm-evict.js), so "Open" (verb) and "Open" (adjective) are two
846
+ // cache entries and two billed strings, exactly as the run sees them.
847
+ const cachedText = {};
848
+ // A borrowed i18next plural form has its own entry (tmTextFor).
849
+ for (const k of stringKeys) cachedText[k] = tmTextFor(k, expectedFlat[k], expansion);
850
+ let { misses } = partitionByTM(tm, cachedText, stringKeys, code, tmKey);
851
+ if (tmKeys.length > 1) {
852
+ misses = misses.filter(k => lookupTM(tm, cachedText[k], code, tmKeys[1]) === null);
853
+ }
854
+ // Pending keys bypass the cache (the model is asked again); keys held
855
+ // back are never sent to the method, so they cost nothing.
856
+ const missSet = new Set(misses);
857
+ heldCount += stringKeys.filter(k => notSent.has(k) && (missSet.has(k) || bypass.has(k))).length;
858
+ misses = stringKeys.filter(k => (missSet.has(k) || bypass.has(k)) && !notSent.has(k));
859
+ for (const k of misses) sendTexts.push(cachedText[k]);
860
+ for (const k of stringKeys) {
861
+ if (missSet.has(k) || bypass.has(k)) continue;
862
+ const model = carriedFromModel(tm, cachedText[k], code, tmKey);
863
+ if (model !== null) carriedFrom[model] = (carriedFrom[model] || 0) + 1;
864
+ }
865
+ stringKeyCount += stringKeys.length;
866
+ for (const k of misses) {
867
+ if (pricedTexts.has(cachedText[k])) continue;
868
+ missCount++;
869
+ }
870
+ // Texts first priced in THIS file still cost here (one batch sends
871
+ // duplicates as separate keys); only later files reuse them.
872
+ for (const k of misses) pricedTexts.add(cachedText[k]);
873
+ }
874
+ if (stringKeyCount === 0) continue;
875
+ const tmHits = stringKeyCount - missCount - heldCount;
876
+ if (options.collect) options.collect[pairKey] = { target: code, pairConfig, sendTexts, carriedFrom };
246
877
 
247
- if (misses.length > 0) {
878
+ if (missCount > 0) {
248
879
  // eslint-disable-next-line no-await-in-loop — sequential is fine for cost queries (cached)
249
- const estimate = await estimateCost(misses.length, pairConfig);
880
+ const estimate = await estimateCost(missCount, pairConfig, { cwd });
250
881
 
251
882
  if (estimate.estimatedCost !== null) {
252
883
  totalEstimatedCost += estimate.estimatedCost;
@@ -257,10 +888,19 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
257
888
  costEstimates.push({
258
889
  pair: pairKey,
259
890
  method: pairConfig.method || 'llm',
260
- keys: misses.length,
891
+ keys: missCount,
261
892
  tmHits,
893
+ ...(heldCount > 0 && { held: heldCount }),
262
894
  estimatedCost: estimate.estimatedCost,
263
895
  source: estimate.source,
896
+ // A model on this machine: $0 API cost, and why (lib/methods/http-utils.js).
897
+ ...(estimate.local && { local: true }),
898
+ ...(estimate.note && { note: estimate.note }),
899
+ // The rate it was priced at: per 1M tokens (or characters), where
900
+ // the figure came from and when (printCostTable says it in a line).
901
+ ...(estimate.rate && { rate: estimate.rate }),
902
+ // No price: the model it was looked up under, so the table can name it.
903
+ ...(estimate.estimatedCost === null && estimate.model && { model: estimate.model }),
264
904
  });
265
905
  } else {
266
906
  // Fully TM-covered: zero API calls, so the cost is a KNOWN $0 even
@@ -272,71 +912,79 @@ export async function printCostEstimate(pairEntries, sourceFlat, config, format,
272
912
  method: pairConfig.method || 'llm',
273
913
  keys: 0,
274
914
  tmHits,
915
+ ...(heldCount > 0 && { held: heldCount }),
275
916
  estimatedCost: 0,
276
- source: 'translation-memory',
917
+ source: tmHits > 0 ? 'translation-memory' : 'nothing-to-send',
277
918
  });
278
919
  }
279
920
  }
280
921
 
281
- // ── Content files (Hugo Markdown) — ROUGH estimate ─────────────
922
+ // ── Content files (Hugo Markdown) ─────────────────────────────
282
923
  // Counts only the (file × pair) translations the sync would actually
283
- // run (hash-manifest skip logic), then prices each pair's pending
284
- // source characters as EST_CHARS_PER_KEY-char key-equivalents through
285
- // the pair's own estimateCost(). Exact for char-priced providers
286
- // (google/deepl/…: key-equivalents × 25 chars = the real characters);
287
- // deliberately CONSERVATIVE for token-priced LLM models, where each
288
- // 25-char slice is priced as if it paid full per-key prompt overhead.
289
- // NOT TM-partitioned (unlike the key-value table above): content-sync
290
- // caches whole bodies/front-matter fields in the TM, but mirroring that
291
- // here would mean parsing every pending file — kept rough instead.
292
- // Rough by design — fail-safe overestimates rather than under.
924
+ // run (hash-manifest skip logic), then prices what each one BILLS: the
925
+ // front-matter fields and body blocks the TM does not already hold —
926
+ // the same ladder runContentSync runs (lib/content-estimate.js). The
927
+ // character → cost step stays rough (see priceContentChars), but the
928
+ // TM discount is exact, so --max-cost compares against what the run
929
+ // will pay rather than the whole file (dogfood 2026-08-28, finding 1).
293
930
  let content = null;
294
931
  if (config.contentDir) {
295
932
  const pending = countPendingContentTranslations(
296
- config.contentDir, config.inputLocale || 'en', pairEntries, cwd
933
+ config.contentDir, config.inputLocale || 'en', pairEntries, cwd,
934
+ {
935
+ tm, translatableFields: config.translatableFields,
936
+ fileScope: options.fileScope || null, forceContent: !!options.forceContent,
937
+ fresh: !!options.redo?.fresh,
938
+ }
297
939
  );
298
- let contentCost = 0;
299
- let contentUnknown = false;
300
940
  if (pending.pendingTranslations > 0) {
941
+ const billedByPair = new Map();
942
+ const roughByPair = new Map();
301
943
  for (const [, pairConfig] of pairEntries) {
302
944
  const perTarget = pending.byTarget[pairConfig.target];
303
945
  if (!perTarget || perTarget.pendingTranslations === 0) continue;
304
- const keyEquivalents = Math.ceil(perTarget.pendingSourceChars / EST_CHARS_PER_KEY);
305
- // eslint-disable-next-line no-await-in-loop — sequential is fine for cost queries (cached)
306
- const estimate = await estimateCost(keyEquivalents, pairConfig);
307
- if (estimate.estimatedCost !== null) {
308
- contentCost += estimate.estimatedCost;
309
- } else {
310
- contentUnknown = true;
946
+ // What the caller may want per target (options.collectContent):
947
+ // whether this run sends Markdown to the pair's method at all.
948
+ if (options.collectContent) {
949
+ options.collectContent[pairConfig.target] = {
950
+ pendingTranslations: perTarget.pendingTranslations, billedChars: perTarget.billedChars,
951
+ };
311
952
  }
953
+ billedByPair.set(pairConfig, perTarget.billedChars);
954
+ roughByPair.set(pairConfig, perTarget.pendingSourceChars);
312
955
  }
313
- }
314
- content = {
315
- files: pending.sourceFileCount,
316
- pendingTranslations: pending.pendingTranslations,
317
- estimatedCost: contentUnknown ? null : contentCost,
318
- rough: true,
319
- };
320
- if (contentUnknown) {
321
- hasUnknownCosts = true;
956
+ const billed = await priceContentChars(billedByPair, pairEntries, estimateCost);
957
+ const rough = await priceContentChars(roughByPair, pairEntries, estimateCost);
958
+ content = {
959
+ files: pending.sourceFileCount,
960
+ pendingTranslations: pending.pendingTranslations,
961
+ estimatedCost: billed.unknown ? null : billed.cost,
962
+ costWithoutTM: rough.unknown ? null : rough.cost,
963
+ rough: true,
964
+ // Which pairs have no price, and why (printCostTable names them).
965
+ ...(billed.unknown && { unpriced: billed.unpriced }),
966
+ // The rates the billed part was priced at.
967
+ ...(billed.rates.length > 0 && { rates: billed.rates }),
968
+ };
969
+ if (billed.unknown) hasUnknownCosts = true;
970
+ else totalEstimatedCost += billed.cost;
322
971
  } else {
323
- totalEstimatedCost += contentCost;
972
+ content = {
973
+ files: pending.sourceFileCount, pendingTranslations: 0,
974
+ estimatedCost: 0, costWithoutTM: 0, rough: true,
975
+ };
324
976
  }
325
977
  }
326
978
 
327
979
  // ── Display ────────────────────────────────────────────────────
980
+ // What the caller must say BEFORE the figure (options.beforeTable): the
981
+ // model carry-over notice — translations reused from the previous model
982
+ // are part of why the cost is what it is (Round 7, Next.js persona: the
983
+ // table said "TM hits 1, free (cache)" and the reuse was named after it).
984
+ if (typeof options.beforeTable === 'function') options.beforeTable({ content });
328
985
  printCostTable(costEstimates, content, totalEstimatedCost, hasUnknownCosts);
329
986
 
330
- return {
331
- currency: 'USD',
332
- pairs: costEstimates,
333
- keyCost: content && content.estimatedCost !== null
334
- ? totalEstimatedCost - content.estimatedCost
335
- : totalEstimatedCost,
336
- content,
337
- totalEstimatedCost,
338
- hasUnknownCosts,
339
- };
987
+ return summarizeEstimate(costEstimates, content, totalEstimatedCost, hasUnknownCosts);
340
988
  } catch (costError) {
341
989
  // Cost estimation is non-blocking when no cap is set — log and continue.
342
990
  // Callers enforcing --max-cost must treat the null return as unknown.