champollion 0.3.3 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. package/README.md +52 -37
  2. package/bin/cli.js +53 -5
  3. package/index.js +63 -2
  4. package/lib/api-key.js +17 -4
  5. package/lib/autofix.js +83 -36
  6. package/lib/bridge/method_bridge.py +15 -3
  7. package/lib/cards/reader.js +51 -3
  8. package/lib/cards/remote.js +15 -0
  9. package/lib/cards/search-names.js +178 -0
  10. package/lib/command-help.js +289 -88
  11. package/lib/commands/audit.js +10 -3
  12. package/lib/commands/card.js +583 -226
  13. package/lib/commands/doctor.js +54 -18
  14. package/lib/commands/help.js +37 -32
  15. package/lib/commands/init.js +1689 -87
  16. package/lib/commands/integrity.js +127 -40
  17. package/lib/commands/leaderboard.js +187 -67
  18. package/lib/commands/models.js +9 -2
  19. package/lib/commands/provenance.js +7 -2
  20. package/lib/commands/recommend.js +43 -14
  21. package/lib/commands/register-corpus.js +649 -130
  22. package/lib/commands/seal-corpus.js +1 -1
  23. package/lib/commands/status.js +564 -27
  24. package/lib/commands/submit.js +17 -12
  25. package/lib/commands/sync.js +31 -7
  26. package/lib/commands/tm.js +16 -10
  27. package/lib/commands/verify.js +27 -3
  28. package/lib/commands/wrap.js +63 -5
  29. package/lib/commands/xliff.js +135 -64
  30. package/lib/commercial-eligibility.js +1 -1
  31. package/lib/config.js +196 -14
  32. package/lib/content-estimate.js +96 -0
  33. package/lib/content-refusals.js +270 -0
  34. package/lib/content-review.js +372 -0
  35. package/lib/content-sync.js +1127 -344
  36. package/lib/content.js +94 -7
  37. package/lib/corpus-registration.mjs +197 -38
  38. package/lib/cost-label.js +29 -0
  39. package/lib/cost-report.js +726 -78
  40. package/lib/diff.js +38 -4
  41. package/lib/docusaurus-sync.js +965 -253
  42. package/lib/edit-distance.js +31 -0
  43. package/lib/fallback.js +964 -0
  44. package/lib/file-scope.js +106 -0
  45. package/lib/flatten.js +80 -3
  46. package/lib/flutter-locales.js +124 -0
  47. package/lib/format.js +266 -12
  48. package/lib/hash.js +146 -21
  49. package/lib/icu-structure.js +929 -0
  50. package/lib/integrity.js +223 -75
  51. package/lib/language-pair.js +157 -0
  52. package/lib/lint.js +78 -16
  53. package/lib/local-only-marks.js +106 -0
  54. package/lib/locale-layout.js +1103 -0
  55. package/lib/locale-state.js +571 -0
  56. package/lib/methods/anthropic.js +5 -0
  57. package/lib/methods/apertium.js +6 -3
  58. package/lib/methods/api.js +138 -25
  59. package/lib/methods/base.js +17 -0
  60. package/lib/methods/coaching-data.js +153 -0
  61. package/lib/methods/content-separator.js +43 -0
  62. package/lib/methods/deepl.js +1 -1
  63. package/lib/methods/direct-llm.js +252 -103
  64. package/lib/methods/external.js +146 -63
  65. package/lib/methods/gemini.js +1 -0
  66. package/lib/methods/google-translate.js +1 -0
  67. package/lib/methods/http-utils.js +41 -0
  68. package/lib/methods/libretranslate.js +7 -2
  69. package/lib/methods/llm-coached.js +68 -128
  70. package/lib/methods/llm.js +80 -31
  71. package/lib/methods/local.js +93 -10
  72. package/lib/methods/microsoft-translator.js +1 -2
  73. package/lib/methods/openai.js +4 -2
  74. package/lib/methods/openrouter-client.js +20 -19
  75. package/lib/methods/openrouter-pricing.js +150 -13
  76. package/lib/methods/prompt-methods.js +20 -0
  77. package/lib/methods/provider-pricing.js +42 -1
  78. package/lib/methods/request-capture.js +104 -0
  79. package/lib/methods/tilde.js +1 -1
  80. package/lib/methods/translated.js +1 -2
  81. package/lib/missing-key.js +93 -0
  82. package/lib/models.js +11 -0
  83. package/lib/name-rules.js +32 -0
  84. package/lib/named-keys.js +172 -0
  85. package/lib/no-translate.js +4 -3
  86. package/lib/output.js +160 -19
  87. package/lib/pairs.js +586 -30
  88. package/lib/placeholders.js +394 -0
  89. package/lib/plugins.js +8 -0
  90. package/lib/plural-gap-redo.js +109 -0
  91. package/lib/plurals.js +323 -0
  92. package/lib/po.js +1187 -0
  93. package/lib/public-catalogue.js +74 -0
  94. package/lib/recommend.js +527 -32
  95. package/lib/redo.js +95 -0
  96. package/lib/refusal-category.js +44 -0
  97. package/lib/registers.js +255 -11
  98. package/lib/repair-script.js +20 -13
  99. package/lib/scripts.js +193 -106
  100. package/lib/seal.mjs +6 -5
  101. package/lib/sealed-qualifier.mjs +2 -2
  102. package/lib/segment.js +2 -1
  103. package/lib/seo.js +19 -9
  104. package/lib/serve.js +43 -6
  105. package/lib/shared-output-seed.js +164 -0
  106. package/lib/source-contexts.js +39 -0
  107. package/lib/submit.mjs +57 -5
  108. package/lib/sync.js +2923 -474
  109. package/lib/terminology.js +13 -4
  110. package/lib/tm-evict.js +179 -0
  111. package/lib/tm-seed.js +5 -2
  112. package/lib/tm.js +818 -36
  113. package/lib/translate-pair.js +639 -34
  114. package/lib/translate.js +78 -5
  115. package/lib/types.js +22 -3
  116. package/lib/validate.js +880 -17
  117. package/lib/verify.js +1296 -104
  118. package/lib/watch.js +32 -13
  119. package/lib/xliff.js +44 -3
  120. package/package.json +3 -2
  121. package/shared/CORPORA-CARDS.md +2 -0
  122. package/shared/DATA-SOVEREIGNTY.md +19 -20
  123. package/shared/LANGUAGE-CARD-FIELDS.md +1 -1
  124. package/shared/cards-fallback.json +1 -1
  125. package/shared/catalogue/card-config.json +1 -1
  126. package/shared/curated-orthography-conventions.json +26 -8
  127. package/shared/docent/faq.en.json +14 -16
  128. package/shared/docent/system-prompt.md +17 -19
  129. package/shared/explainers/tc-features.json +15 -15
  130. package/shared/gettext-plural-forms.json +45 -0
  131. package/shared/human-services.json +1 -1
  132. package/shared/method-registry.json +2 -0
  133. package/shared/metric-registry.json +96 -18
  134. package/shared/schemas/champollion-plugin.schema.json +4 -0
  135. package/shared/schemas/corpora-card.schema.json +20 -10
  136. package/shared/schemas/human-services.schema.json +2 -2
  137. package/shared/schemas/language-card.schema.json +1 -1
  138. package/shared/schemas/method-card.schema.json +1 -1
  139. package/shared/schemas/method-index-record.schema.json +67 -0
  140. package/shared/schemas/method-registry.schema.json +4 -0
  141. package/shared/schemas/metric-registry.schema.json +55 -1
  142. package/shared/docent/corpus.json +0 -11333
package/lib/tm.js CHANGED
@@ -49,6 +49,7 @@
49
49
  import fs from 'node:fs';
50
50
  import path from 'node:path';
51
51
  import crypto from 'node:crypto';
52
+ import { COACHING_PROMPT_METHODS } from './methods/prompt-methods.js';
52
53
 
53
54
  /**
54
55
  * Current TM format version. If the format changes in a backward-incompatible
@@ -114,13 +115,22 @@ function _shortHash(text) {
114
115
  * - model: the specific model, when the method uses one ('' otherwise)
115
116
  * - register: the preset key when known (human-readable), else a short
116
117
  * hash of the custom register text ('' when unset)
117
- * - coaching: for llm-coached pairs, a short hash of the resolved coaching
118
- * prompt text (config.js reads coachingFile into
119
- * coachingPrompt) plus any structured plugin coachingData.
120
- * When only a coachingFile path is available (not yet
121
- * resolved), the path is fingerprinted — a moved/renamed file
122
- * still invalidates, though an in-place edit that bypassed
123
- * config resolution would not. '' for non-coached methods.
118
+ * - coaching: for every method whose prompt carries the pair's
119
+ * free-text coaching — the plain LLM methods (llm, local,
120
+ * openai, anthropic, gemini) and llm-coached — a short hash of
121
+ * the coaching TEXT (lib/pairs.js reads a pair's, a language's
122
+ * or a fallback's own coachingFile into coachingPrompt; config.js
123
+ * the top-level one), plus, for llm-coached, any structured
124
+ * plugin coachingData. So an edit of the file re-translates.
125
+ * Only an llm-coached pair built without resolution (a library
126
+ * caller) falls back to the path; it reads the file itself.
127
+ * '' for the methods that send no coaching (api, the MT
128
+ * engines) — a coaching file there changes nothing.
129
+ * Until 2026-10 only llm-coached was keyed on it: a `local`
130
+ * fallback's coaching file never reached its prompt nor its
131
+ * key, and `--redo all` served the uncoached text (Round 10,
132
+ * school persona). Older entries made with the SAME coaching
133
+ * text stay reachable (adoptLegacyCoachingKeys).
124
134
  *
125
135
  * Changing any component makes old entries unreachable (a cache miss, so
126
136
  * the API is consulted) WITHOUT nuking valid entries for other pairs —
@@ -129,32 +139,97 @@ function _shortHash(text) {
129
139
  * @param {object} pairConfig - Pair config (method, model, register, registerPreset, coaching*)
130
140
  * @returns {string} Stable method-key string, e.g. "llm|google/gemini-3.5-flash|formal|"
131
141
  */
142
+ /** origin + path of an endpoint URL ('' when absent or unparseable). */
143
+ function _endpointIdentity(endpoint) {
144
+ if (!endpoint || typeof endpoint !== 'string') return '';
145
+ try {
146
+ const u = new URL(endpoint);
147
+ return `${u.origin}${u.pathname.replace(/\/+$/, '')}`;
148
+ } catch {
149
+ return '';
150
+ }
151
+ }
152
+
132
153
  function tmMethodKey(pairConfig) {
133
154
  const method = pairConfig.method || 'llm';
134
- const model = pairConfig.model || '';
155
+ // An `api` pair's system is its ENDPOINT (or its method plugin), not the
156
+ // global default model it inherits but never calls — a trained model served
157
+ // by nmt-forge was cached under "google/gemini-…" (synthetic review). Only
158
+ // origin + path: a query string or credentials must never land in the cache.
159
+ // Older entries keyed the old way are still served by model carry-over.
160
+ const model = method === 'api'
161
+ ? (pairConfig.methodPlugin || _endpointIdentity(pairConfig.endpoint) || pairConfig.model || '')
162
+ : (pairConfig.model || '');
135
163
 
136
- const register = pairConfig.registerPreset
164
+ let register = pairConfig.registerPreset
137
165
  || (typeof pairConfig.register === 'string' && pairConfig.register.length > 0
138
166
  ? _shortHash(pairConfig.register)
139
167
  : '');
168
+ // Gender guidance the config chose (its own instruction, or none): another
169
+ // prompt, so another cache entry. The catalogue's default adds nothing, so
170
+ // every existing entry keeps its key.
171
+ if (pairConfig.genderGuidanceSource === 'off') register += '+gender-off';
172
+ else if (pairConfig.genderGuidanceSource === 'config' && typeof pairConfig.genderGuidance === 'string') {
173
+ register += `+gender-${_shortHash(pairConfig.genderGuidance)}`;
174
+ }
140
175
 
141
- let coaching = '';
142
- if (method === 'llm-coached') {
143
- const parts = [];
144
- if (typeof pairConfig.coachingPrompt === 'string' && pairConfig.coachingPrompt.trim().length > 0) {
145
- parts.push(pairConfig.coachingPrompt);
146
- } else if (typeof pairConfig.coachingFile === 'string' && pairConfig.coachingFile.trim().length > 0) {
147
- parts.push(`file:${pairConfig.coachingFile}`);
148
- }
149
- if (pairConfig.coachingData) {
150
- parts.push(JSON.stringify(pairConfig.coachingData));
151
- }
152
- if (parts.length > 0) {
153
- coaching = _shortHash(parts.join('\x00'));
154
- }
176
+ return `${method}|${model}|${register}|${_coachingSegment(method, pairConfig, pairConfig.coachingPrompt)}`;
177
+ }
178
+
179
+ /**
180
+ * The coaching part of a method key: a short hash of the coaching text the
181
+ * method's prompt carries ('' when it carries none).
182
+ *
183
+ * @param {string} method
184
+ * @param {object} pairConfig
185
+ * @param {string|null|undefined} promptText - The coaching text to fingerprint
186
+ * @param {{ plainKeyed?: boolean }} [opts] - false: the plain LLM methods
187
+ * are not keyed (how keys were made before 2026-10 — legacyMethodKey)
188
+ * @returns {string}
189
+ */
190
+ function _coachingSegment(method, pairConfig, promptText, { plainKeyed = true } = {}) {
191
+ if (!COACHING_PROMPT_METHODS.has(method)) return '';
192
+ if (method !== 'llm-coached' && !plainKeyed) return '';
193
+ const parts = [];
194
+ if (typeof promptText === 'string' && promptText.trim().length > 0) {
195
+ parts.push(promptText);
196
+ } else if (method === 'llm-coached' && typeof pairConfig.coachingFile === 'string' && pairConfig.coachingFile.trim().length > 0) {
197
+ // Unresolved (a pair built by a library caller): llm-coached reads the
198
+ // file itself, so its path stands for it. The plain methods read only
199
+ // the resolved text — without it they send no coaching.
200
+ parts.push(`file:${pairConfig.coachingFile}`);
201
+ }
202
+ if (method === 'llm-coached' && pairConfig.coachingData) {
203
+ parts.push(JSON.stringify(pairConfig.coachingData));
155
204
  }
205
+ return parts.length > 0 ? _shortHash(parts.join('\x00')) : '';
206
+ }
156
207
 
157
- return `${method}|${model}|${register}|${coaching}`;
208
+ /**
209
+ * The method key this pair's entries were made under before its coaching
210
+ * was keyed the way tmMethodKey keys it now — or null when that is the same
211
+ * key, or when the coaching those entries were made with is not the
212
+ * coaching the pair has now (they were made with other text: not reusable).
213
+ *
214
+ * Before 2026-10 the plain LLM methods kept '' as coaching whatever their
215
+ * prompt carried, and llm-coached fingerprinted a pair's own coachingFile by
216
+ * its PATH. lib/pairs.js records on each pair the coaching text that
217
+ * version would have SENT (`_legacyCoachingSent`): the top-level or inline
218
+ * coaching it did send; never a language's, pair's or fallback's own
219
+ * coachingFile for a plain method, which it never read.
220
+ *
221
+ * @param {object} pairConfig - Resolved pair (lib/pairs.js)
222
+ * @returns {string|null}
223
+ */
224
+ function legacyMethodKey(pairConfig) {
225
+ if (!pairConfig || !Object.prototype.hasOwnProperty.call(pairConfig, '_legacyCoachingSent')) return null;
226
+ const now = typeof pairConfig.coachingPrompt === 'string' && pairConfig.coachingPrompt.trim() ? pairConfig.coachingPrompt.trim() : null;
227
+ const then = typeof pairConfig._legacyCoachingSent === 'string' && pairConfig._legacyCoachingSent.trim() ? pairConfig._legacyCoachingSent.trim() : null;
228
+ if (now !== then) return null;
229
+ const current = tmMethodKey(pairConfig);
230
+ const [method, model, register] = current.split('|');
231
+ const legacy = `${method}|${model}|${register}|${_coachingSegment(method, pairConfig, pairConfig._legacyCoachingPrompt, { plainKeyed: false })}`;
232
+ return legacy === current ? null : legacy;
158
233
  }
159
234
 
160
235
  // -----------------------------------------------------------------
@@ -247,7 +322,374 @@ function saveTM(cwd, tm) {
247
322
  }
248
323
 
249
324
  /**
250
- * Look up a cached translation.
325
+ * Model carry-over: a model switch reuses existing translations.
326
+ *
327
+ * tmMethodKey folds the model into every key, which on its own would make a
328
+ * model switch strand the whole cache — every unchanged block in a re-queued
329
+ * file would be re-translated and re-billed (the 2026-08 dogfood run priced
330
+ * 70k such entries). Founder direction 2026-10-01: switching model is NOT a
331
+ * reason to pay for a full re-translation. So on an exact miss, lookups fall
332
+ * back to the same source text translated under a DIFFERENT MODEL with the
333
+ * same method, register and coaching. Anything else that shapes output
334
+ * (method, register, coaching) still misses — those changes exist to get
335
+ * different text.
336
+ *
337
+ * Carried entries are served through the same gates as any TM hit
338
+ * (lookupTMValidated evicts the entry that actually served). Opt out per run
339
+ * with setModelCarryover(tm, false) — the CLI's --fresh-on-model-change.
340
+ */
341
+ const _NO_CARRYOVER = Symbol('tm-no-model-carryover');
342
+ // --fresh / --no-tm: nothing is SERVED from the cache this run, but what the
343
+ // run pays for is still STORED. These used to swap in a throwaway empty TM,
344
+ // so a fresh re-translation was never cached and a later --redo served the
345
+ // older text back (synthetic review, 2026-10-03). Never serialized.
346
+ const _NO_READS = Symbol('tm-no-reads');
347
+
348
+ /**
349
+ * Serve nothing from this TM for the rest of the run (writes still land).
350
+ * @param {object} tm
351
+ * @param {boolean} enabled - false = reads off
352
+ */
353
+ function setTMReads(tm, enabled) {
354
+ tm[_NO_READS] = !enabled;
355
+ }
356
+ const _SIBLINGS = Symbol('tm-model-siblings');
357
+ const _CARRIED = Symbol('tm-carried-hits');
358
+ const _BYPASS = Symbol('tm-bypass');
359
+
360
+ /**
361
+ * Make lookups for these (locale, source text) pairs MISS for the rest of
362
+ * this run — `sync --retranslate <glob>`: the named files are translated
363
+ * fresh even though the TM holds their text. New translations are still
364
+ * stored (replacing the old entries), so the next run is cached again.
365
+ *
366
+ * @param {object} tm - TM object
367
+ * @param {string} locale - Target locale
368
+ * @param {Iterable<string>} sourceTexts - Fields, body and block sources
369
+ */
370
+ function bypassTMFor(tm, locale, sourceTexts) {
371
+ if (!tm[_BYPASS]) tm[_BYPASS] = new Set();
372
+ for (const text of sourceTexts) {
373
+ if (typeof text === 'string' && text) tm[_BYPASS].add(`${locale}\x00${text}`);
374
+ }
375
+ }
376
+
377
+ /**
378
+ * Enable (default) or disable model carry-over for this TM object.
379
+ *
380
+ * @param {object} tm - TM object
381
+ * @param {boolean} enabled - false = only exact-model hits are served
382
+ */
383
+ function setModelCarryover(tm, enabled) {
384
+ tm[_NO_CARRYOVER] = !enabled;
385
+ }
386
+
387
+ /**
388
+ * How many lookups on this TM object were served by model carry-over.
389
+ *
390
+ * @param {object} tm - TM object
391
+ * @returns {number}
392
+ */
393
+ function carriedHitCount(tm) {
394
+ return tm[_CARRIED] || 0;
395
+ }
396
+
397
+ /**
398
+ * The model whose entry a lookup of (source, locale, method) would be served
399
+ * from by model carry-over — null when the lookup misses or hits the exact
400
+ * model. Read-only: counts no hit. Lets a run say "reused from the previous
401
+ * model" only when it actually reuses something (Round 5, Next.js persona).
402
+ *
403
+ * @param {object} tm
404
+ * @param {string} sourceValue - Text as cached
405
+ * @param {string} locale
406
+ * @param {string} method - Full method key (tmMethodKey)
407
+ * @returns {string|null} The earlier model's name ('(none)' when it had none)
408
+ */
409
+ function carriedFromModel(tm, sourceValue, locale, method) {
410
+ const found = _findEntry(tm, sourceValue, locale, method);
411
+ if (!found || !found.carried) return null;
412
+ return found.methodKey.split('|')[1] || '(none)';
413
+ }
414
+
415
+ /** "method|model|register|coaching" → "locale\0method|register|coaching" (null if not 4-part). */
416
+ function _siblingGroup(locale, methodKey) {
417
+ const parts = methodKey.split('|');
418
+ if (parts.length !== 4) return null;
419
+ return `${locale}\x00${parts[0]}|${parts[2]}|${parts[3]}`;
420
+ }
421
+
422
+ /** Lazily index the method keys present per (locale, model-less key). */
423
+ function _siblingIndex(tm) {
424
+ if (!tm[_SIBLINGS]) {
425
+ const index = new Map();
426
+ for (const [k, entry] of Object.entries(tm)) {
427
+ if (k === '_meta' || !entry || typeof entry.m !== 'string' || typeof entry.l !== 'string') continue;
428
+ const group = _siblingGroup(entry.l, entry.m);
429
+ if (group === null) continue;
430
+ if (!index.has(group)) index.set(group, new Set());
431
+ index.get(group).add(entry.m);
432
+ }
433
+ tm[_SIBLINGS] = index;
434
+ }
435
+ return tm[_SIBLINGS];
436
+ }
437
+
438
+ /**
439
+ * Find the entry that serves (source, locale, method): the exact key first,
440
+ * then — unless carry-over is off — the newest entry for the same text
441
+ * under another model. Returns the method key that actually holds it so a
442
+ * failed validation evicts the right entry.
443
+ *
444
+ * @returns {{ text: string, methodKey: string, carried: boolean } | null}
445
+ */
446
+ function _findEntry(tm, sourceValue, locale, method) {
447
+ if (tm[_NO_READS]) return null;
448
+ if (tm[_BYPASS] && tm[_BYPASS].has(`${locale}\x00${sourceValue}`)) return null;
449
+ const exact = tm[cacheKey(sourceValue, locale, method)];
450
+ if (exact && typeof exact.t === 'string') return { text: exact.t, methodKey: method, carried: false };
451
+ // The same setup's entries from before its coaching was keyed (same
452
+ // method, model, register and coaching TEXT): this pair's own work.
453
+ const legacy = _legacyKeyOf(tm, locale, method);
454
+ if (legacy) {
455
+ const old = tm[cacheKey(sourceValue, locale, legacy)];
456
+ // methodKey: the entry that holds it (what a failed check evicts);
457
+ // writer: the setup that made it, today's key (what the lock records).
458
+ if (old && typeof old.t === 'string') return { text: old.t, methodKey: legacy, writer: method, carried: false };
459
+ }
460
+ if (tm[_NO_CARRYOVER]) return null;
461
+
462
+ const groups = [_siblingGroup(locale, method), legacy ? _siblingGroup(locale, legacy) : null].filter(Boolean);
463
+ let best = null;
464
+ for (const group of groups) {
465
+ for (const other of _siblingIndex(tm).get(group) || []) {
466
+ if (other === method || other === legacy) continue;
467
+ const entry = tm[cacheKey(sourceValue, locale, other)];
468
+ if (entry && typeof entry.t === 'string' && (!best || String(entry.ts) > String(best.entry.ts))) {
469
+ best = { entry, methodKey: other };
470
+ }
471
+ }
472
+ }
473
+ return best ? { text: best.entry.t, methodKey: best.methodKey, carried: true } : null;
474
+ }
475
+
476
+ // ── Coaching keys from before 2026-10 ────────────────────────────────
477
+
478
+ /**
479
+ * Once per cache written before coaching was keyed: remember, in the cache
480
+ * itself, which of its keys (legacyMethodKey) hold THIS setup's work —
481
+ * method, model, register and the coaching text the pair has now. Lookups
482
+ * then serve those entries for the current key, so an upgrade re-translates
483
+ * nothing and no file is reported as "written by another coaching"
484
+ * (canonicalWriterKey).
485
+ *
486
+ * ONCE: the first run that loads such a cache records it and marks the cache
487
+ * `coachingKeyed`; a new cache is marked from the start. After that a key
488
+ * with an empty coaching part is just an uncoached entry, so coaching added
489
+ * or edited later is a change — it re-translates (what the key is for). A
490
+ * pair whose coaching the old version never sent (a fallback's own
491
+ * coachingFile with a plain method) gets no record: its old entries were
492
+ * made without that coaching. What cannot be known: old entries made before
493
+ * a coaching that was added in the same upgrade are taken as made with it —
494
+ * as the old version served them.
495
+ *
496
+ * @param {object} tm
497
+ * @param {Iterable<object>} pairConfigs - Resolved pairs (fallbacks are visited too)
498
+ * @returns {number} records added
499
+ */
500
+ function adoptLegacyCoachingKeys(tm, pairConfigs) {
501
+ if (!tm || tm._meta?.coachingKeyed) return 0;
502
+ let added = 0;
503
+ const visit = (pc) => {
504
+ if (!pc || typeof pc.target !== 'string') return;
505
+ const legacy = legacyMethodKey(pc);
506
+ if (!legacy) return;
507
+ if (!_localeMethodKeys(tm, pc.target).has(legacy)) return;
508
+ tm._meta.coachingKeys = tm._meta.coachingKeys || {};
509
+ tm._meta.coachingKeys[pc.target] = tm._meta.coachingKeys[pc.target] || {};
510
+ tm._meta.coachingKeys[pc.target][tmMethodKey(pc)] = legacy;
511
+ added++;
512
+ };
513
+ tm._meta = tm._meta || {};
514
+ for (const pc of pairConfigs || []) {
515
+ visit(pc);
516
+ visit(pc?.fallback);
517
+ }
518
+ tm._meta.coachingKeyed = true;
519
+ _markDirty(tm);
520
+ return added;
521
+ }
522
+
523
+ /** The pre-2026-10 key recorded for (locale, current key), or null. */
524
+ function _legacyKeyOf(tm, locale, method) {
525
+ const legacy = tm?._meta?.coachingKeys?.[locale]?.[method];
526
+ return typeof legacy === 'string' && legacy !== method ? legacy : null;
527
+ }
528
+
529
+ /**
530
+ * The key a writer recorded in the lock (`by`) stands for today: a key from
531
+ * before coaching was keyed, for a setup adoptLegacyCoachingKeys recorded,
532
+ * reads as that setup's current key — the same model kept, so an earlier
533
+ * model stays an earlier model. Any other key is returned as it is.
534
+ *
535
+ * @param {object} tm
536
+ * @param {string} locale
537
+ * @param {string} methodKey
538
+ * @returns {string}
539
+ */
540
+ function canonicalWriterKey(tm, locale, methodKey) {
541
+ const records = tm?._meta?.coachingKeys?.[locale];
542
+ if (!records || typeof methodKey !== 'string') return methodKey;
543
+ const parts = methodKey.split('|');
544
+ if (parts.length !== 4) return methodKey;
545
+ for (const [current, legacy] of Object.entries(records)) {
546
+ if (legacy === methodKey) return current;
547
+ const c = current.split('|');
548
+ const l = legacy.split('|');
549
+ if (c.length !== 4 || l.length !== 4) continue;
550
+ // Same method, register and old coaching, another model: that model's
551
+ // entry under today's coaching key.
552
+ if (parts[0] === l[0] && parts[2] === l[2] && parts[3] === l[3]) return `${c[0]}|${parts[1]}|${c[2]}|${c[3]}`;
553
+ }
554
+ return methodKey;
555
+ }
556
+
557
+ /**
558
+ * What a lookup of (source, locale, method) would serve — exact model, else
559
+ * model carry-over — without counting a carried hit. Honors --fresh and
560
+ * --retranslate (null when reads are off). For reports that must not change
561
+ * the run's tallies.
562
+ *
563
+ * @returns {string|null}
564
+ */
565
+ function peekTM(tm, sourceValue, locale, method) {
566
+ const found = _findEntry(tm, sourceValue, locale, method);
567
+ return found ? found.text : null;
568
+ }
569
+
570
+ /**
571
+ * The method key of the entry a lookup of (source, locale, method) serves
572
+ * from — `method` itself on an exact hit, another model's key on a carried
573
+ * one — or null on a miss. Read-only. Lets sync record which model wrote a
574
+ * value it served from the cache (lib/locale-state.js `by`).
575
+ *
576
+ * @returns {string|null}
577
+ */
578
+ function servingMethodKey(tm, sourceValue, locale, method) {
579
+ const found = _findEntry(tm, sourceValue, locale, method);
580
+ if (!found) return null;
581
+ return found.writer || canonicalWriterKey(tm, locale, found.methodKey);
582
+ }
583
+
584
+ /**
585
+ * Every translation the cache holds for (source text, locale), under ANY
586
+ * method key — the proof that a value on disk is something the pipeline
587
+ * produced (lib/locale-state.js: a value with no written-record counts as
588
+ * Champollion's only when the cache holds it). Read-only: ignores the
589
+ * --fresh read switch and --retranslate bypass, counts no carry-over hit.
590
+ *
591
+ * @param {object} tm
592
+ * @param {string} sourceValue - Text as cached (lib/tm-evict.js tmSourceText)
593
+ * @param {string} locale
594
+ * @returns {Set<string>}
595
+ */
596
+ function tmTranslationsOf(tm, sourceValue, locale) {
597
+ const out = new Set();
598
+ if (!tm || typeof sourceValue !== 'string') return out;
599
+ for (const mk of _localeMethodKeys(tm, locale)) {
600
+ const entry = tm[cacheKey(sourceValue, locale, mk)];
601
+ if (entry && typeof entry.t === 'string') out.add(entry.t);
602
+ }
603
+ return out;
604
+ }
605
+
606
+ /**
607
+ * The method keys whose cache entry for (source text, locale) holds exactly
608
+ * `value` — who could have produced a value on disk. Read-only.
609
+ *
610
+ * @param {object} tm
611
+ * @param {string} sourceValue - Text as cached
612
+ * @param {string} locale
613
+ * @param {string} value
614
+ * @returns {string[]}
615
+ */
616
+ function tmMethodKeysHolding(tm, sourceValue, locale, value) {
617
+ const out = [];
618
+ if (!tm || typeof sourceValue !== 'string' || typeof value !== 'string') return out;
619
+ for (const mk of _localeMethodKeys(tm, locale)) {
620
+ const entry = tm[cacheKey(sourceValue, locale, mk)];
621
+ // A key from before coaching was keyed reads as its setup's today.
622
+ if (entry && entry.t === value) out.push(canonicalWriterKey(tm, locale, mk));
623
+ }
624
+ return [...new Set(out)];
625
+ }
626
+
627
+ /**
628
+ * Every translation text the cache holds for a locale (any source, any
629
+ * method). For a value whose source has since changed — its old text is
630
+ * known only by hash — "the cache holds this exact text" is the remaining
631
+ * proof that the pipeline wrote it. Read-only.
632
+ *
633
+ * @param {object} tm
634
+ * @param {string} locale
635
+ * @returns {Set<string>}
636
+ */
637
+ function tmTranslationSet(tm, locale) {
638
+ const out = new Set();
639
+ for (const [k, entry] of Object.entries(tm || {})) {
640
+ if (k === '_meta' || !entry || entry.l !== locale || typeof entry.t !== 'string') continue;
641
+ out.add(entry.t);
642
+ }
643
+ return out;
644
+ }
645
+
646
+ /**
647
+ * Cached translations of these texts that this pair can NOT reuse because
648
+ * they were made with another method, register or coaching (a different
649
+ * MODEL alone is reused — model carry-over). Switching method (local → llm)
650
+ * showed "0 served from the cache" with no reason (Round 4, Next.js
651
+ * persona); this is the reason, counted. Read-only.
652
+ *
653
+ * @param {object} tm
654
+ * @param {object} pairConfig
655
+ * @param {string[]} texts - Texts as cached (lib/tm-evict.js tmSourceText)
656
+ * @returns {Array<{ methodKey: string, count: number }>} most first
657
+ */
658
+ function findOtherMethodEntries(tm, pairConfig, texts) {
659
+ const locale = pairConfig.target;
660
+ const current = tmMethodKey(pairConfig);
661
+ const [method, , register, coaching] = current.split('|');
662
+ const counts = new Map();
663
+ for (const mk of _localeMethodKeys(tm, locale)) {
664
+ if (mk === current) continue;
665
+ const parts = mk.split('|');
666
+ // Same method, register and coaching = a sibling model: carry-over reuses it.
667
+ if (parts.length === 4 && parts[0] === method && parts[2] === register && parts[3] === coaching) continue;
668
+ let n = 0;
669
+ for (const text of texts) if (tm[cacheKey(text, locale, mk)]) n++;
670
+ if (n > 0) counts.set(mk, n);
671
+ }
672
+ return [...counts.entries()].map(([methodKey, count]) => ({ methodKey, count })).sort((a, b) => b.count - a.count);
673
+ }
674
+
675
+ const _LOCALE_KEYS = Symbol('tm-locale-method-keys');
676
+ /** Method keys present per locale, indexed once per TM object. */
677
+ function _localeMethodKeys(tm, locale) {
678
+ if (!tm[_LOCALE_KEYS]) {
679
+ const index = new Map();
680
+ for (const [k, entry] of Object.entries(tm)) {
681
+ if (k === '_meta' || !entry || typeof entry.m !== 'string' || typeof entry.l !== 'string') continue;
682
+ if (!index.has(entry.l)) index.set(entry.l, new Set());
683
+ index.get(entry.l).add(entry.m);
684
+ }
685
+ tm[_LOCALE_KEYS] = index;
686
+ }
687
+ return tm[_LOCALE_KEYS].get(locale) || new Set();
688
+ }
689
+
690
+ /**
691
+ * Look up a cached translation — exact model first, then (by default) the
692
+ * same text under another model; see "Model carry-over" above.
251
693
  *
252
694
  * @param {object} tm - TM object (from loadTM)
253
695
  * @param {string} sourceValue - Source language value
@@ -256,12 +698,10 @@ function saveTM(cwd, tm) {
256
698
  * @returns {string|null} Cached translation, or null for cache miss
257
699
  */
258
700
  function lookupTM(tm, sourceValue, locale, method) {
259
- const key = cacheKey(sourceValue, locale, method);
260
- const entry = tm[key];
261
- if (entry && typeof entry.t === 'string') {
262
- return entry.t;
263
- }
264
- return null;
701
+ const found = _findEntry(tm, sourceValue, locale, method);
702
+ if (!found) return null;
703
+ if (found.carried) tm[_CARRIED] = (tm[_CARRIED] || 0) + 1;
704
+ return found.text;
265
705
  }
266
706
 
267
707
  /**
@@ -286,13 +726,21 @@ function lookupTM(tm, sourceValue, locale, method) {
286
726
  * @param {string} locale - Target locale code
287
727
  * @param {string} method - Translation method name
288
728
  * @param {(source: string, cached: string) => boolean} isValid - True to serve
729
+ * @param {{ countCarry?: boolean }} [opts] - countCarry false: a look before
730
+ * serving (the content lanes' repeat check) — a carried hit is counted
731
+ * when it is served, not twice
289
732
  * @returns {string|null} Validated cached translation, or null
290
733
  */
291
- function lookupTMValidated(tm, sourceValue, locale, method, isValid) {
292
- const cached = lookupTM(tm, sourceValue, locale, method);
293
- if (cached === null) return null;
294
- if (isValid(sourceValue, cached)) return cached;
295
- evictTM(tm, sourceValue, locale, method);
734
+ function lookupTMValidated(tm, sourceValue, locale, method, isValid, { countCarry = true } = {}) {
735
+ const found = _findEntry(tm, sourceValue, locale, method);
736
+ if (!found) return null;
737
+ if (isValid(sourceValue, found.text)) {
738
+ if (found.carried && countCarry) tm[_CARRIED] = (tm[_CARRIED] || 0) + 1;
739
+ return found.text;
740
+ }
741
+ // Evict the entry that actually served — for a carried hit that is the
742
+ // OTHER model's entry; evicting the exact key would leave it to re-fail.
743
+ evictTM(tm, sourceValue, locale, found.methodKey);
296
744
  return null;
297
745
  }
298
746
 
@@ -307,12 +755,28 @@ function lookupTMValidated(tm, sourceValue, locale, method, isValid) {
307
755
  */
308
756
  function storeTM(tm, sourceValue, locale, method, translation) {
309
757
  const key = cacheKey(sourceValue, locale, method);
758
+ const prior = tm[key];
759
+ const stats = _changeStats(tm);
760
+ if (!prior) stats.added++;
761
+ else if (prior.t !== translation) stats.replaced++;
762
+ else stats.refreshed++;
310
763
  tm[key] = {
311
764
  t: translation,
312
765
  ts: new Date().toISOString(),
313
766
  l: locale, // locale code — enables per-locale stats and filtering
314
767
  m: method, // method name — enables per-method stats
315
768
  };
769
+ if (tm[_LOCALE_KEYS]) {
770
+ if (!tm[_LOCALE_KEYS].has(locale)) tm[_LOCALE_KEYS].set(locale, new Set());
771
+ tm[_LOCALE_KEYS].get(locale).add(method);
772
+ }
773
+ if (tm[_SIBLINGS]) {
774
+ const group = _siblingGroup(locale, method);
775
+ if (group !== null) {
776
+ if (!tm[_SIBLINGS].has(group)) tm[_SIBLINGS].set(group, new Set());
777
+ tm[_SIBLINGS].get(group).add(method);
778
+ }
779
+ }
316
780
  _markDirty(tm);
317
781
  }
318
782
 
@@ -333,12 +797,49 @@ function evictTM(tm, sourceValue, locale, method) {
333
797
  const key = cacheKey(sourceValue, locale, method);
334
798
  if (key in tm) {
335
799
  delete tm[key];
800
+ _changeStats(tm).removed++;
336
801
  _markDirty(tm);
337
802
  return true;
338
803
  }
339
804
  return false;
340
805
  }
341
806
 
807
+ // What this run did to the cache — never serialized. "+0 this sync" after a
808
+ // --fresh redo read as if nothing was saved, when every entry had been
809
+ // replaced by the new translation (Round 3, Django persona).
810
+ const _STATS = Symbol('tm-change-stats');
811
+ function _changeStats(tm) {
812
+ if (!tm[_STATS]) tm[_STATS] = { added: 0, replaced: 0, refreshed: 0, removed: 0 };
813
+ return tm[_STATS];
814
+ }
815
+
816
+ /**
817
+ * Entries this TM object gained, replaced (new text for the same source,
818
+ * locale and method), re-stored with identical text, or removed since load.
819
+ *
820
+ * @param {object} tm
821
+ * @returns {{ added: number, replaced: number, refreshed: number, removed: number }}
822
+ */
823
+ function tmChangeStats(tm) {
824
+ return { ..._changeStats(tm) };
825
+ }
826
+
827
+ /**
828
+ * One line for the end of a sync: how big the cache is and what changed.
829
+ * e.g. "12 entries — 2 added, 6 replaced with a new translation, 1 removed".
830
+ *
831
+ * @param {object} tm
832
+ * @returns {string}
833
+ */
834
+ function describeTMChanges(tm) {
835
+ const { added, replaced, refreshed, removed } = _changeStats(tm);
836
+ const parts = [`${added} added`];
837
+ if (replaced > 0) parts.push(`${replaced} replaced with a new translation`);
838
+ if (refreshed > 0) parts.push(`${refreshed} re-translated to the same text`);
839
+ if (removed > 0) parts.push(`${removed} removed`);
840
+ return `${tmSize(tm)} entries — ${parts.join(', ')}`;
841
+ }
842
+
342
843
  /**
343
844
  * Has this TM been mutated (store or evict) since load/save?
344
845
  *
@@ -488,6 +989,10 @@ function _createEmptyTM() {
488
989
  _meta: {
489
990
  version: TM_VERSION,
490
991
  created: new Date().toISOString(),
992
+ // Every entry of this cache is keyed with its coaching (tmMethodKey):
993
+ // nothing in it is from before, so nothing is adopted
994
+ // (adoptLegacyCoachingKeys).
995
+ coachingKeyed: true,
491
996
  },
492
997
  };
493
998
  }
@@ -496,7 +1001,271 @@ function _createEmptyTM() {
496
1001
  // Exports
497
1002
  // -----------------------------------------------------------------
498
1003
 
1004
+ /**
1005
+ * Find TM entries a MODEL SWITCH stranded.
1006
+ *
1007
+ * tmMethodKey folds the model into every key on purpose (a new model should
1008
+ * not be served the old one's output by accident). The cost of that rule is
1009
+ * silent: after a model change, every entry for the pair becomes unreachable
1010
+ * and the next sync prices — and bills — a full re-translation of text that
1011
+ * never changed. The 2026-08 dogfood run hit this with 70k entries.
1012
+ *
1013
+ * The recovery already exists (`champollion tm seed` re-keys the on-disk
1014
+ * translations under the current method key, for free); what was missing is
1015
+ * anyone noticing. This reports, per target locale, entries whose method key
1016
+ * is identical to the current one EXCEPT for the model segment. Pure — the
1017
+ * caller decides how to say it.
1018
+ *
1019
+ * COUNTS ONLY WHAT THE PROJECT STILL USES when `sourceTexts` is given: an
1020
+ * entry for a string since edited or deleted cannot be reused, and counting
1021
+ * it announced "12 cached translations will be reused" for a project with
1022
+ * 10 keys (Round 3, Next.js persona).
1023
+ *
1024
+ * @param {object} tm - TM object (from loadTM)
1025
+ * @param {Iterable<object>} pairConfigs - Resolved pair configs for this run
1026
+ * @param {{ sourceTexts?: Iterable<string>|null, every?: boolean }} [options] - the texts the
1027
+ * project's current source strings are cached under (lib/tm-evict.js
1028
+ * tmSourceText); omitted = count every entry. `every`: report each locale
1029
+ * where an earlier model holds entries at all (no "more than the current
1030
+ * model" threshold, switches marked done included)
1031
+ * @returns {Array<{ target: string, currentModel: string, current: number,
1032
+ * stranded: Array<{ model: string, count: number }> }>} Locales where some
1033
+ * other model holds more entries than the current one. Empty when none.
1034
+ */
1035
+ function findModelSwitchStrandedEntries(tm, pairConfigList, { sourceTexts = null, every = false } = {}) {
1036
+ // Iterated twice below — an iterator (Map#values()) would be empty the 2nd time.
1037
+ const pairConfigs = [...pairConfigList];
1038
+ // Count entries per (locale, method key) once.
1039
+ const counts = new Map();
1040
+ const methodKeysByLocale = new Map();
1041
+ for (const [k, entry] of Object.entries(tm)) {
1042
+ if (k === '_meta' || !entry || typeof entry.m !== 'string' || typeof entry.l !== 'string') continue;
1043
+ const id = `${entry.l}\x00${entry.m}`;
1044
+ counts.set(id, (counts.get(id) || 0) + 1);
1045
+ if (!methodKeysByLocale.has(entry.l)) methodKeysByLocale.set(entry.l, new Set());
1046
+ methodKeysByLocale.get(entry.l).add(entry.m);
1047
+ }
1048
+ // Restricted to the current source: count, per (locale, method key), the
1049
+ // current texts that method key holds a translation for.
1050
+ if (sourceTexts) {
1051
+ const texts = [...new Set(sourceTexts)].filter(t => typeof t === 'string');
1052
+ counts.clear();
1053
+ const targets = new Set(pairConfigs.map(pc => pc.target));
1054
+ for (const locale of targets) {
1055
+ for (const mk of methodKeysByLocale.get(locale) || []) {
1056
+ let n = 0;
1057
+ for (const text of texts) if (tm[cacheKey(text, locale, mk)]) n++;
1058
+ if (n > 0) counts.set(`${locale}\x00${mk}`, n);
1059
+ }
1060
+ }
1061
+ }
1062
+
1063
+ const report = [];
1064
+ for (const pairConfig of pairConfigs) {
1065
+ const target = pairConfig.target;
1066
+ const currentKey = tmMethodKey(pairConfig);
1067
+ // Fully re-translated under this model already (sync records it after
1068
+ // --redo all --fresh-on-model-change): the switch is done, nothing to say.
1069
+ // `every`: a locale where any earlier model holds entries, whatever the
1070
+ // counts — for a run that serves one of them (Round 13: a string reverted
1071
+ // to text only the earlier model had translated).
1072
+ if (!every && tm._meta?.switchedTo?.[target] === currentKey) continue;
1073
+ const [method, currentModel, register, coaching] = currentKey.split('|');
1074
+ const current = counts.get(`${target}\x00${currentKey}`) || 0;
1075
+ const stranded = [];
1076
+ for (const [id, count] of counts) {
1077
+ const [locale, methodKey] = id.split('\x00');
1078
+ if (locale !== target || methodKey === currentKey) continue;
1079
+ const parts = methodKey.split('|');
1080
+ if (parts.length !== 4) continue;
1081
+ if (parts[0] === method && parts[2] === register && parts[3] === coaching && parts[1] !== currentModel) {
1082
+ stranded.push({ model: parts[1], count });
1083
+ }
1084
+ }
1085
+ stranded.sort((a, b) => b.count - a.count);
1086
+ // Only worth saying when another model holds MORE than the current one:
1087
+ // a few leftovers after a deliberate switch are not news.
1088
+ if (stranded.length > 0 && (every || stranded[0].count > current)) {
1089
+ report.push({ target, currentModel, current, stranded });
1090
+ }
1091
+ }
1092
+ return report;
1093
+ }
1094
+
1095
+ /**
1096
+ * What an earlier model's cache holds for one pair, counted ONCE per thing
1097
+ * so a total and its per-model breakdown always add up.
1098
+ *
1099
+ * `findModelSwitchStrandedEntries` counts, per earlier model, every string
1100
+ * that model holds — two earlier models that both translated a string count
1101
+ * it twice, and a model's count includes strings the current model has too.
1102
+ * A notice that summed the first model's count and then listed every model's
1103
+ * said "12 cached translation(s)" above per-model counts adding up to 30
1104
+ * (Round 11, Next.js persona).
1105
+ *
1106
+ * With `sourceTexts` it counts STRINGS of the current source that only an
1107
+ * earlier model translated (what carry-over would serve), each under the
1108
+ * model carry-over would serve it from (the newest entry, as lookups pick).
1109
+ * Without them (a project whose Markdown blocks share the cache) strings
1110
+ * cannot be told apart, so it counts cache ENTRIES per earlier model, and
1111
+ * the total is their sum.
1112
+ *
1113
+ * @param {object} tm
1114
+ * @param {object} pairConfig - The resolved pair
1115
+ * @param {Array<{ model: string }>} stranded - The pair's row from findModelSwitchStrandedEntries
1116
+ * @param {{ sourceTexts?: Iterable<string>|null }} [options]
1117
+ * @returns {{ unit: 'strings'|'entries', total: number, byModel: Array<{ model: string, count: number }> }}
1118
+ */
1119
+ function reusableFromEarlierModels(tm, pairConfig, stranded, { sourceTexts = null } = {}) {
1120
+ if (!sourceTexts) {
1121
+ const byModel = stranded.map(s => ({ model: s.model, count: s.count }));
1122
+ return { unit: 'entries', total: byModel.reduce((n, s) => n + s.count, 0), byModel };
1123
+ }
1124
+ const target = pairConfig.target;
1125
+ const currentKey = tmMethodKey(pairConfig);
1126
+ const [method, , register, coaching] = currentKey.split('|');
1127
+ const earlierKeys = stranded.map(s => ({ model: s.model, methodKey: [method, s.model, register, coaching].join('|') }));
1128
+ const counts = new Map();
1129
+ let total = 0;
1130
+ for (const text of new Set(sourceTexts)) {
1131
+ if (typeof text !== 'string' || tm[cacheKey(text, target, currentKey)]) continue;
1132
+ let best = null;
1133
+ for (const k of earlierKeys) {
1134
+ const entry = tm[cacheKey(text, target, k.methodKey)];
1135
+ if (entry && typeof entry.t === 'string' && (!best || String(entry.ts) > String(best.ts))) best = { ts: entry.ts, model: k.model };
1136
+ }
1137
+ if (!best) continue;
1138
+ counts.set(best.model, (counts.get(best.model) || 0) + 1);
1139
+ total++;
1140
+ }
1141
+ const byModel = [...counts].map(([model, count]) => ({ model, count })).sort((a, b) => b.count - a.count);
1142
+ return { unit: 'strings', total, byModel };
1143
+ }
1144
+
1145
+ /**
1146
+ * A method key in words, for reports: "llm|google/gemini-3.5-flash|formal-vous|"
1147
+ * → "llm · model google/gemini-3.5-flash · register formal-vous". `tm stats`
1148
+ * printed the raw key (Round 3, Next.js persona). Legacy bare keys ("llm")
1149
+ * are returned as they are.
1150
+ *
1151
+ * @param {string} methodKey
1152
+ * @returns {string}
1153
+ */
1154
+ function describeMethodKey(methodKey) {
1155
+ const parts = String(methodKey).split('|');
1156
+ if (parts.length !== 4) return String(methodKey);
1157
+ const [method, model, register, coaching] = parts;
1158
+ const bits = [method || 'llm'];
1159
+ if (model) bits.push(`model ${model}`);
1160
+ if (register) bits.push(`register ${register}`);
1161
+ if (coaching) bits.push(`coaching ${coaching}`);
1162
+ return bits.join(' · ');
1163
+ }
1164
+
1165
+ /**
1166
+ * Whether a writer key (the lock's `by`) is the pair's FALLBACK: 'current'
1167
+ * for the fallback as it is set up now, 'earlier' for the same fallback as it
1168
+ * was set up before — another model, register or coaching of the fallback's
1169
+ * method (and, when the pair and its fallback share a method, the fallback's
1170
+ * model) — else null. Values the fallback wrote are its by design; one its
1171
+ * earlier setup wrote is re-translated by a redo (Round 10, school persona:
1172
+ * a coaching file added to a `local` fallback).
1173
+ *
1174
+ * @param {object} pairConfig - Resolved pair with its `fallback`
1175
+ * @param {string} methodKey
1176
+ * @returns {'current'|'earlier'|null}
1177
+ */
1178
+ function fallbackWriterOf(pairConfig, methodKey) {
1179
+ const fb = pairConfig?.fallback;
1180
+ if (!fb || typeof methodKey !== 'string') return null;
1181
+ const fbKey = tmMethodKey(fb);
1182
+ if (methodKey === fbKey) return 'current';
1183
+ const parts = methodKey.split('|');
1184
+ const fbParts = fbKey.split('|');
1185
+ if (parts.length !== 4 || parts[0] !== fbParts[0]) return null;
1186
+ if (fb.method === pairConfig.method && parts[1] !== fbParts[1]) return null;
1187
+ // The pair's own setup (same method, register and coaching) is the pair's.
1188
+ const own = tmMethodKey(pairConfig).split('|');
1189
+ if (fb.method === pairConfig.method && parts[2] === own[2] && parts[3] === own[3]) return null;
1190
+ return 'earlier';
1191
+ }
1192
+
1193
+ /**
1194
+ * Which models produced the values a target file holds now. After a model
1195
+ * switch without a full re-translation the files mix two models' text, and
1196
+ * nothing said so (Round 3, Next.js persona: `status`).
1197
+ *
1198
+ * The lock's record comes first: `writtenBy` (lib/locale-state.js `by`) is
1199
+ * the method key that produced the value sync last wrote, and is used when
1200
+ * that value is still the one on disk. A value with no record (written
1201
+ * before 0.4.0) is attributed from the cache: the entry under this pair's
1202
+ * method, register and coaching — any model — whose text IS that value.
1203
+ * When several models cached the identical text, the value is "model
1204
+ * unknown": the older answer used to win, and `status` named a model that
1205
+ * had not written the files (Round 6, Next.js persona).
1206
+ *
1207
+ * Values produced by the pair's own fallback are not counted (they are the
1208
+ * fallback's by design); a value no record or entry explains (hand-written,
1209
+ * older than the cache) is not attributed.
1210
+ *
1211
+ * @param {object} tm
1212
+ * @param {object} pairConfig - Resolved pair config (target, method, model, fallback, …)
1213
+ * @param {Iterable<{ text?: string, texts?: string[], value: string, writtenBy?: string|null }>} items -
1214
+ * cached-under text(s) (lib/tm-evict.js), the value on disk, and the
1215
+ * lock's record of who wrote it when it is current
1216
+ * @returns {Array<{ model: string, keys: number, current: boolean, unknown?: true, candidates?: string[] }>} most keys first
1217
+ */
1218
+ function modelsBehindValues(tm, pairConfig, items) {
1219
+ const target = pairConfig.target;
1220
+ const currentKey = tmMethodKey(pairConfig);
1221
+ const fallbackKey = pairConfig.fallback ? tmMethodKey(pairConfig.fallback) : null;
1222
+ const group = _siblingGroup(target, currentKey);
1223
+ const sameGroup = (mk) => mk === currentKey || (group !== null && _siblingGroup(target, mk) === group);
1224
+ const modelOf = (mk) => mk.split('|')[1] || '(none)';
1225
+ const label = (mk) => (sameGroup(mk) ? modelOf(mk) : describeMethodKey(mk));
1226
+ const buckets = new Map();
1227
+ const add = (id, make) => {
1228
+ if (!buckets.has(id)) buckets.set(id, make());
1229
+ buckets.get(id).keys++;
1230
+ };
1231
+ for (const item of items) {
1232
+ const { value } = item;
1233
+ if (typeof value !== 'string') continue;
1234
+ if (typeof item.writtenBy === 'string' && item.writtenBy) {
1235
+ if (item.writtenBy === fallbackKey || fallbackWriterOf(pairConfig, item.writtenBy)) continue;
1236
+ add(item.writtenBy, () => ({ model: label(item.writtenBy), keys: 0, current: item.writtenBy === currentKey }));
1237
+ continue;
1238
+ }
1239
+ const texts = Array.isArray(item.texts) ? item.texts : [item.text];
1240
+ const holders = new Set();
1241
+ for (const text of texts) {
1242
+ if (typeof text !== 'string') continue;
1243
+ for (const mk of _localeMethodKeys(tm, target)) {
1244
+ if (!sameGroup(mk)) continue;
1245
+ const entry = tm[cacheKey(text, target, mk)];
1246
+ if (entry && entry.t === value) holders.add(mk);
1247
+ }
1248
+ }
1249
+ if (holders.size === 1) {
1250
+ const mk = [...holders][0];
1251
+ add(mk, () => ({ model: label(mk), keys: 0, current: mk === currentKey }));
1252
+ } else if (holders.size > 1) {
1253
+ const candidates = [...new Set([...holders].map(modelOf))].sort();
1254
+ add(`?${candidates.join('\u0000')}`, () => ({ model: 'model unknown', keys: 0, current: false, unknown: true, candidates }));
1255
+ }
1256
+ }
1257
+ return [...buckets.values()].sort((a, b) => b.keys - a.keys);
1258
+ }
1259
+
499
1260
  export {
1261
+ peekTM,
1262
+ servingMethodKey,
1263
+ tmMethodKeysHolding,
1264
+ findOtherMethodEntries,
1265
+ tmTranslationsOf,
1266
+ tmTranslationSet,
1267
+ describeMethodKey,
1268
+ modelsBehindValues,
500
1269
  loadTM,
501
1270
  saveTM,
502
1271
  lookupTM,
@@ -507,8 +1276,21 @@ export {
507
1276
  pruneTM,
508
1277
  partitionByTM,
509
1278
  tmSize,
1279
+ tmChangeStats,
1280
+ describeTMChanges,
510
1281
  cacheKey,
511
1282
  tmMethodKey,
1283
+ legacyMethodKey,
1284
+ adoptLegacyCoachingKeys,
1285
+ canonicalWriterKey,
1286
+ fallbackWriterOf,
1287
+ findModelSwitchStrandedEntries,
1288
+ reusableFromEarlierModels,
1289
+ setModelCarryover,
1290
+ setTMReads,
1291
+ bypassTMFor,
1292
+ carriedHitCount,
1293
+ carriedFromModel,
512
1294
  TM_VERSION,
513
1295
  TM_DIR,
514
1296
  TM_FILENAME,