champollion 0.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (170) hide show
  1. package/LICENSE +133 -0
  2. package/README.md +387 -0
  3. package/bin/cli.js +278 -0
  4. package/index.js +135 -0
  5. package/lib/api-key.js +127 -0
  6. package/lib/autofix.js +432 -0
  7. package/lib/bridge/method_bridge.py +430 -0
  8. package/lib/card-source-resolution.mjs +284 -0
  9. package/lib/cards/cache.js +169 -0
  10. package/lib/cards/env.js +82 -0
  11. package/lib/cards/fetch-card-child.js +38 -0
  12. package/lib/cards/reader.js +435 -0
  13. package/lib/cards/refresh.js +111 -0
  14. package/lib/cards/remote.js +387 -0
  15. package/lib/cldf-export.mjs +540 -0
  16. package/lib/cldf-terms.mjs +62 -0
  17. package/lib/command-help.js +790 -0
  18. package/lib/commands/audit.js +49 -0
  19. package/lib/commands/card.js +454 -0
  20. package/lib/commands/doctor.js +559 -0
  21. package/lib/commands/fonts.js +489 -0
  22. package/lib/commands/help.js +91 -0
  23. package/lib/commands/init.js +1259 -0
  24. package/lib/commands/integrity.js +148 -0
  25. package/lib/commands/leaderboard.js +478 -0
  26. package/lib/commands/lint.js +30 -0
  27. package/lib/commands/models.js +177 -0
  28. package/lib/commands/plugin.js +103 -0
  29. package/lib/commands/provenance.js +45 -0
  30. package/lib/commands/recommend.js +75 -0
  31. package/lib/commands/register-corpus.js +678 -0
  32. package/lib/commands/repair-script.js +42 -0
  33. package/lib/commands/seal-corpus.js +355 -0
  34. package/lib/commands/seo.js +72 -0
  35. package/lib/commands/serve.js +147 -0
  36. package/lib/commands/status.js +265 -0
  37. package/lib/commands/submit.js +332 -0
  38. package/lib/commands/sync.js +89 -0
  39. package/lib/commands/tm.js +573 -0
  40. package/lib/commands/verify.js +39 -0
  41. package/lib/commands/watch.js +20 -0
  42. package/lib/commands/wrap.js +138 -0
  43. package/lib/commands/xliff.js +327 -0
  44. package/lib/commercial-eligibility.js +235 -0
  45. package/lib/concurrent.js +87 -0
  46. package/lib/config.js +523 -0
  47. package/lib/contamination-lane.js +76 -0
  48. package/lib/content-sync.js +731 -0
  49. package/lib/content.js +733 -0
  50. package/lib/corpus-registration.mjs +608 -0
  51. package/lib/cost-report.js +346 -0
  52. package/lib/diff.js +155 -0
  53. package/lib/docusaurus-sync.js +1256 -0
  54. package/lib/flatten.js +55 -0
  55. package/lib/format.js +954 -0
  56. package/lib/hash.js +159 -0
  57. package/lib/icu.js +473 -0
  58. package/lib/integrity.js +689 -0
  59. package/lib/license-gate.mjs +478 -0
  60. package/lib/license-identify.mjs +229 -0
  61. package/lib/lint.js +629 -0
  62. package/lib/method-manifest.js +60 -0
  63. package/lib/methods/anthropic.js +140 -0
  64. package/lib/methods/apertium.js +163 -0
  65. package/lib/methods/api.js +316 -0
  66. package/lib/methods/base.js +184 -0
  67. package/lib/methods/content-separator.js +45 -0
  68. package/lib/methods/deepl.js +426 -0
  69. package/lib/methods/direct-llm.js +586 -0
  70. package/lib/methods/external.js +332 -0
  71. package/lib/methods/fetch-with-retry.js +124 -0
  72. package/lib/methods/gemini.js +147 -0
  73. package/lib/methods/google-translate.js +402 -0
  74. package/lib/methods/http-utils.js +122 -0
  75. package/lib/methods/libretranslate.js +314 -0
  76. package/lib/methods/llm-coached.js +670 -0
  77. package/lib/methods/llm.js +592 -0
  78. package/lib/methods/local.js +76 -0
  79. package/lib/methods/microsoft-translator.js +331 -0
  80. package/lib/methods/openai.js +131 -0
  81. package/lib/methods/openrouter-client.js +327 -0
  82. package/lib/methods/openrouter-pricing.js +156 -0
  83. package/lib/methods/provider-env.js +115 -0
  84. package/lib/methods/provider-pricing.js +310 -0
  85. package/lib/methods/tilde.js +150 -0
  86. package/lib/methods/translated.js +229 -0
  87. package/lib/methods/translation-error.js +80 -0
  88. package/lib/models.js +258 -0
  89. package/lib/no-translate.js +233 -0
  90. package/lib/output.js +238 -0
  91. package/lib/pairs.js +547 -0
  92. package/lib/plugins.js +447 -0
  93. package/lib/provenance.js +323 -0
  94. package/lib/recommend.js +648 -0
  95. package/lib/registers.js +1185 -0
  96. package/lib/repair-script.js +266 -0
  97. package/lib/scripts.js +994 -0
  98. package/lib/seal.mjs +464 -0
  99. package/lib/sealed-qualifier.mjs +211 -0
  100. package/lib/security.js +59 -0
  101. package/lib/segment.js +369 -0
  102. package/lib/seo.js +275 -0
  103. package/lib/serve.js +854 -0
  104. package/lib/string-classify.js +85 -0
  105. package/lib/submit.mjs +344 -0
  106. package/lib/sync.js +969 -0
  107. package/lib/tags/bcp47.js +202 -0
  108. package/lib/tags/resolve.js +314 -0
  109. package/lib/terminology.js +111 -0
  110. package/lib/tm-seed.js +294 -0
  111. package/lib/tm.js +515 -0
  112. package/lib/translate-pair.js +197 -0
  113. package/lib/translate.js +203 -0
  114. package/lib/types.js +230 -0
  115. package/lib/validate.js +510 -0
  116. package/lib/verify.js +451 -0
  117. package/lib/watch.js +145 -0
  118. package/lib/xliff.js +184 -0
  119. package/package.json +93 -0
  120. package/shared/ATTRIBUTION.md +145 -0
  121. package/shared/CORPORA-CARDS.md +288 -0
  122. package/shared/DATA-SOVEREIGNTY.md +500 -0
  123. package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
  124. package/shared/card-lint-baseline.json +3189 -0
  125. package/shared/cards-fallback.json +1 -0
  126. package/shared/catalogue/card-config.json +6091 -0
  127. package/shared/catalogue/external-results.json +3888 -0
  128. package/shared/catalogue/gender-guidance.json +1038 -0
  129. package/shared/catalogue/method-coverage.json +1751 -0
  130. package/shared/catalogue/metric-coverage.json +170 -0
  131. package/shared/catalogue/metric-reliability.json +1 -0
  132. package/shared/catalogue/register-presets.json +3180 -0
  133. package/shared/catalogue/vitality-scales.json +55 -0
  134. package/shared/cldr-index.json +1115 -0
  135. package/shared/code-bridge.json +253 -0
  136. package/shared/corpora-cards-v1-reference.md +281 -0
  137. package/shared/curated-dictionary-flags.json +35 -0
  138. package/shared/curated-endonyms.json +35 -0
  139. package/shared/curated-fsts.json +51 -0
  140. package/shared/curated-orthography-conventions.json +26 -0
  141. package/shared/curated-sil-resources.json +374 -0
  142. package/shared/curated-tools.json +41 -0
  143. package/shared/docent/corpus.json +11333 -0
  144. package/shared/docent/faq.en.json +564 -0
  145. package/shared/docent/register-blocks.json +60 -0
  146. package/shared/docent/system-prompt.md +144 -0
  147. package/shared/domain-taxonomy.json +35 -0
  148. package/shared/explainers/glossary.json +2975 -0
  149. package/shared/explainers/tc-features.json +20112 -0
  150. package/shared/explainers/term-watchlist.json +147 -0
  151. package/shared/human-services.json +59 -0
  152. package/shared/license-corrections.json +261 -0
  153. package/shared/license-evidence.json +13452 -0
  154. package/shared/licenses.json +6781 -0
  155. package/shared/method-registry.json +236 -0
  156. package/shared/metric-registry.json +620 -0
  157. package/shared/model-aliases.json +7 -0
  158. package/shared/schemas/champollion-plugin.schema.json +206 -0
  159. package/shared/schemas/corpora-card.schema.json +957 -0
  160. package/shared/schemas/domain-taxonomy.schema.json +64 -0
  161. package/shared/schemas/external-results.schema.json +314 -0
  162. package/shared/schemas/human-services.schema.json +90 -0
  163. package/shared/schemas/language-card.schema.json +1308 -0
  164. package/shared/schemas/licenses.schema.json +155 -0
  165. package/shared/schemas/method-card.schema.json +412 -0
  166. package/shared/schemas/method-registry.schema.json +85 -0
  167. package/shared/schemas/metric-registry.schema.json +96 -0
  168. package/shared/schemas/metric-reliability.schema.json +178 -0
  169. package/shared/schemas/model-aliases.schema.json +27 -0
  170. package/shared/schemas/source-snapshot.schema.json +96 -0
package/lib/content.js ADDED
@@ -0,0 +1,733 @@
1
+ /**
2
+ * Markdown content translation — translates Hugo content files.
3
+ *
4
+ * WHY: Hugo stores page content as Markdown files with YAML front matter.
5
+ * Unlike i18n string files (key→value pairs), content files need:
6
+ * 1. Front matter parsing — extract translatable fields (title, description)
7
+ * 2. Block protection — shield code blocks, shortcodes, and inline code
8
+ * from the translation engine so they pass through untouched
9
+ * 3. Body translation — send the protected Markdown to the LLM
10
+ * 4. Reassembly — restore protected blocks and rebuild the file
11
+ *
12
+ * Hugo's translation-by-filename convention:
13
+ * content/posts/my-post.md → default language
14
+ * content/posts/my-post.fr.md → French
15
+ * content/posts/my-post.ja.md → Japanese
16
+ *
17
+ * This module handles the parse→protect→translate→restore→write pipeline.
18
+ */
19
+
20
+ import fs from 'node:fs';
21
+ import path from 'node:path';
22
+
23
+ // Sentinel used for protected block placeholders. Uses Unicode brackets
24
+ // that are extremely unlikely to appear in real content, making them
25
+ // safe to use as delimiters even if the LLM generates creative output.
26
+ const PLACEHOLDER_PREFIX = '⟦PROTECTED_';
27
+ const PLACEHOLDER_SUFFIX = '⟧';
28
+
29
+ // Front matter fields that should be translated by default.
30
+ // Other fields (date, draft, tags, slug, weight, etc.) are preserved as-is.
31
+ // `sidebar_label` is the Docusaurus sidebar/menu entry — user-facing prose that
32
+ // was previously left in the source language on every translated page.
33
+ const DEFAULT_TRANSLATABLE_FIELDS = [
34
+ 'title',
35
+ 'description',
36
+ 'summary',
37
+ 'subtitle',
38
+ 'caption',
39
+ 'linkTitle',
40
+ 'sidebar_label',
41
+ ];
42
+
43
+ // Top-level front matter keys that are structural/taxonomy metadata, NOT human
44
+ // prose — safe to leave untranslated even when they appear as arrays or nested
45
+ // blocks the flat parser can't reach. Anything else that we skip as nested/array
46
+ // gets surfaced by findUntranslatableNestedFields() so it's never silently lost.
47
+ const NON_TRANSLATABLE_NESTED_FIELDS = new Set([
48
+ 'tags', 'keywords', 'slug', 'aliases', 'url', 'permalink',
49
+ 'date', 'lastmod', 'publishdate', 'expirydate', 'draft', 'weight',
50
+ 'type', 'layout', 'author', 'authors', 'categories', 'series',
51
+ 'image', 'images', 'id', 'sidebar_position', 'pagination_next',
52
+ 'pagination_prev', 'hide_table_of_contents', 'toc_min_heading_level',
53
+ 'toc_max_heading_level', 'sidebar_custom_props', 'tags_url',
54
+ ]);
55
+
56
+ // Regex for YAML front matter delimiters (--- ... ---)
57
+ const YAML_FM_REGEX = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/;
58
+
59
+ // Regex for TOML front matter delimiters (+++ ... +++)
60
+ const TOML_FM_REGEX = /^\+\+\+\r?\n([\s\S]*?)\r?\n\+\+\+\r?\n?([\s\S]*)$/;
61
+
62
+ // -----------------------------------------------------------------
63
+ // Content file parsing
64
+ // -----------------------------------------------------------------
65
+
66
+ /**
67
+ * Parse a Hugo Markdown content file into structured parts.
68
+ *
69
+ * Hugo supports both YAML (---) and TOML (+++) front matter.
70
+ * We try YAML first (more common), then fall back to TOML.
71
+ *
72
+ * @param {string} raw - Raw file content
73
+ * @returns {object} { frontMatter, rawFrontMatter, body, hasFrontMatter, frontMatterFormat }
74
+ * - frontMatter: parsed key→value map (simple flat parsing)
75
+ * - rawFrontMatter: the raw string between delimiters
76
+ * - body: everything after the front matter
77
+ * - hasFrontMatter: whether front matter was detected
78
+ * - frontMatterFormat: 'yaml' | 'toml' | null
79
+ */
80
+ function parseContentFile(raw) {
81
+ // Try YAML first (--- ... ---) — most common in Hugo
82
+ const yamlMatch = raw.match(YAML_FM_REGEX);
83
+ if (yamlMatch) {
84
+ return {
85
+ frontMatter: parseSimpleFrontMatter(yamlMatch[1]),
86
+ rawFrontMatter: yamlMatch[1],
87
+ body: yamlMatch[2],
88
+ hasFrontMatter: true,
89
+ frontMatterFormat: 'yaml',
90
+ };
91
+ }
92
+
93
+ // Try TOML (+++ ... +++)
94
+ const tomlMatch = raw.match(TOML_FM_REGEX);
95
+ if (tomlMatch) {
96
+ return {
97
+ frontMatter: parseSimpleTomlFrontMatter(tomlMatch[1]),
98
+ rawFrontMatter: tomlMatch[1],
99
+ body: tomlMatch[2],
100
+ hasFrontMatter: true,
101
+ frontMatterFormat: 'toml',
102
+ };
103
+ }
104
+
105
+ return { frontMatter: {}, rawFrontMatter: '', body: raw, hasFrontMatter: false, frontMatterFormat: null };
106
+ }
107
+
108
+ /**
109
+ * Parse simple YAML front matter into a key→value map.
110
+ *
111
+ * WHY hand-rolled: Hugo front matter is almost always flat key-value
112
+ * pairs (title, description, date, draft, etc.). We only need to
113
+ * extract the translatable string fields. Complex nested YAML,
114
+ * arrays, and multi-line values are preserved as raw strings so
115
+ * they pass through unchanged.
116
+ *
117
+ * @param {string} yaml - Raw YAML content (between --- delimiters)
118
+ * @returns {object} key→value map
119
+ */
120
+ function parseSimpleFrontMatter(yaml) {
121
+ const result = {};
122
+
123
+ for (const line of yaml.split('\n')) {
124
+ const trimmed = line.trim();
125
+ if (!trimmed || trimmed.startsWith('#')) continue;
126
+
127
+ // Simple key: value pairs only (skip arrays, nested objects)
128
+ if (line.startsWith(' ') || line.startsWith('\t')) continue;
129
+
130
+ const colonIdx = trimmed.indexOf(':');
131
+ if (colonIdx < 0) continue;
132
+
133
+ const key = trimmed.slice(0, colonIdx).trim();
134
+ let value = trimmed.slice(colonIdx + 1).trim();
135
+
136
+ // Skip keys with empty values — they're YAML map/array parent keys
137
+ // (e.g., "tags:" followed by indented array items)
138
+ if (!value) continue;
139
+
140
+ // Skip YAML block scalars (`>`, `|`, `>-`, `|-`, `>+`, indent variants).
141
+ // The value on this line is only the block indicator; the real text lives
142
+ // on the indented lines below (already skipped by the indent check above).
143
+ // Extracting the indicator as a translatable field and re-quoting it on
144
+ // rebuild would orphan those lines and emit invalid YAML — so, per this
145
+ // parser's contract, multi-line values pass through raw and untranslated.
146
+ if (/^[>|][0-9+-]*$/.test(value)) continue;
147
+
148
+ // Unquote if quoted
149
+ if (
150
+ (value.startsWith('"') && value.endsWith('"')) ||
151
+ (value.startsWith("'") && value.endsWith("'"))
152
+ ) {
153
+ value = value.slice(1, -1);
154
+ }
155
+
156
+ result[key] = value;
157
+ }
158
+
159
+ return result;
160
+ }
161
+
162
+ /**
163
+ * Find top-level front matter keys that hold array/nested values the flat
164
+ * parser cannot reach — and that are NOT known structural/taxonomy metadata.
165
+ *
166
+ * WHY: parseSimpleFrontMatter() silently skips every indented/array line by
167
+ * design. That's correct for `tags`, `slug`, `date`, etc., but it also means a
168
+ * human-prose container like `related:` (a list of `{title, url}` entries) and
169
+ * other nested fields pass through entirely untranslated with no signal. The
170
+ * never-silent doctrine says: at minimum, tell the user. Callers surface these
171
+ * as a warning so the omission is visible instead of shipping half-translated
172
+ * pages that look complete.
173
+ *
174
+ * Block scalars (`>`, `|`) are intentionally excluded — they're a separate,
175
+ * already-documented skip (extracting the indicator would emit invalid YAML).
176
+ *
177
+ * @param {string} rawFrontMatter - Raw YAML front matter (between --- delimiters)
178
+ * @returns {string[]} Names of skipped, non-allowlisted nested/array fields
179
+ */
180
+ function findUntranslatableNestedFields(rawFrontMatter) {
181
+ const found = [];
182
+ if (!rawFrontMatter) return found;
183
+
184
+ for (const line of rawFrontMatter.split('\n')) {
185
+ if (!line.trim() || line.trim().startsWith('#')) continue;
186
+ // Only consider top-level keys — indented lines are the nested content.
187
+ if (line.startsWith(' ') || line.startsWith('\t')) continue;
188
+
189
+ const trimmed = line.trim();
190
+ const colonIdx = trimmed.indexOf(':');
191
+ if (colonIdx < 0) continue;
192
+
193
+ const key = trimmed.slice(0, colonIdx).trim();
194
+ const value = trimmed.slice(colonIdx + 1).trim();
195
+
196
+ if (NON_TRANSLATABLE_NESTED_FIELDS.has(key)) continue;
197
+ if (/^[>|][0-9+-]*$/.test(value)) continue; // block scalar — separate case
198
+
199
+ // Empty value → an indented block/array follows; or an inline array/object.
200
+ if (value === '' || value.startsWith('[') || value.startsWith('{')) {
201
+ found.push(key);
202
+ }
203
+ }
204
+
205
+ return found;
206
+ }
207
+
208
+ /**
209
+ * Parse simple TOML front matter into a key→value map.
210
+ *
211
+ * TOML front matter uses `key = "value"` syntax (with = instead of :).
212
+ * Like the YAML parser, we only extract flat key-value pairs and
213
+ * skip complex nested structures.
214
+ *
215
+ * @param {string} toml - Raw TOML content (between +++ delimiters)
216
+ * @returns {object} key→value map
217
+ */
218
+ function parseSimpleTomlFrontMatter(toml) {
219
+ const result = {};
220
+ // Track whether we're inside a nested table section.
221
+ // Once we hit a [section] header, all subsequent keys belong to that
222
+ // table and should NOT be treated as top-level front matter keys.
223
+ // There's no way to "exit" a TOML table — a new top-level key after
224
+ // a section is technically invalid TOML, but we handle it gracefully
225
+ // by staying in nested mode until the end of the front matter block.
226
+ let insideNestedTable = false;
227
+
228
+ for (const line of toml.split('\n')) {
229
+ const trimmed = line.trim();
230
+ if (!trimmed || trimmed.startsWith('#')) continue;
231
+
232
+ // Detect TOML section headers ([section]) and array-of-tables ([[section]])
233
+ // WHY we warn: Keys inside nested tables (e.g. [params] description = "...")
234
+ // won't be parsed or translated. The user needs to know this so they don't
235
+ // ship partially-translated content thinking everything was handled.
236
+ if (trimmed.startsWith('[')) {
237
+ insideNestedTable = true;
238
+ const tableName = trimmed.replace(/^\[+|\]+$/g, '').trim();
239
+ console.warn(
240
+ ` [WARN] TOML front matter: nested table [${tableName}] detected — ` +
241
+ `keys inside it will not be translated. Flatten translatable fields to the top level.`
242
+ );
243
+ continue;
244
+ }
245
+
246
+ // Skip all keys inside nested tables — they belong to the section,
247
+ // not to the top-level front matter we're interested in.
248
+ if (insideNestedTable) continue;
249
+
250
+ // Match key = value pairs
251
+ const eqIdx = trimmed.indexOf('=');
252
+ if (eqIdx < 0) continue;
253
+
254
+ const key = trimmed.slice(0, eqIdx).trim();
255
+ let value = trimmed.slice(eqIdx + 1).trim();
256
+
257
+ // Skip keys with empty values
258
+ if (!value) continue;
259
+
260
+ // Unquote if quoted
261
+ if (
262
+ (value.startsWith('"') && value.endsWith('"')) ||
263
+ (value.startsWith("'") && value.endsWith("'"))
264
+ ) {
265
+ value = value.slice(1, -1);
266
+ }
267
+
268
+ result[key] = value;
269
+ }
270
+
271
+ return result;
272
+ }
273
+
274
+ /**
275
+ * Rebuild front matter YAML with translated fields.
276
+ *
277
+ * Preserves the original formatting for non-translated fields by
278
+ * doing line-by-line replacement rather than full re-serialization.
279
+ * This keeps array fields, comments, and complex YAML intact.
280
+ *
281
+ * @param {string} rawYaml - Original raw YAML front matter
282
+ * @param {object} translations - Map of field name → translated value
283
+ * @returns {string} Updated YAML front matter
284
+ */
285
+ function rebuildFrontMatter(rawYaml, translations) {
286
+ const lines = rawYaml.split('\n');
287
+ const result = [];
288
+
289
+ for (const line of lines) {
290
+ // Check if this line is a top-level key: value that we have a translation for
291
+ const trimmed = line.trim();
292
+ if (!trimmed.startsWith(' ') && !trimmed.startsWith('\t') && !trimmed.startsWith('#')) {
293
+ const colonIdx = trimmed.indexOf(':');
294
+ if (colonIdx > 0) {
295
+ const key = trimmed.slice(0, colonIdx).trim();
296
+ const rawValue = trimmed.slice(colonIdx + 1).trim();
297
+ // Never rewrite a block scalar (`>`, `|`, …): its text spans the
298
+ // indented lines below, so replacing just this line would orphan them
299
+ // and emit invalid YAML. Leave it raw (parseSimpleFrontMatter already
300
+ // skips these, so this is defense-in-depth).
301
+ if (key in translations && !/^[>|][0-9+-]*$/.test(rawValue)) {
302
+ // Replace the value, preserving the key and indentation
303
+ const value = translations[key];
304
+ const needsQuotes = value.includes(':') || value.includes('#') ||
305
+ value.includes('"') || value.includes("'") ||
306
+ value.startsWith(' ') || value.endsWith(' ');
307
+ const formatted = needsQuotes
308
+ ? `"${value.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"`
309
+ : `"${value}"`;
310
+ result.push(`${key}: ${formatted}`);
311
+ continue;
312
+ }
313
+ }
314
+ }
315
+
316
+ result.push(line);
317
+ }
318
+
319
+ return result.join('\n');
320
+ }
321
+
322
+ /**
323
+ * Rebuild TOML front matter with translated fields.
324
+ *
325
+ * Same approach as YAML — line-by-line replacement to preserve
326
+ * formatting for non-translated fields.
327
+ *
328
+ * @param {string} rawToml - Original raw TOML front matter
329
+ * @param {object} translations - Map of field name → translated value
330
+ * @returns {string} Updated TOML front matter
331
+ */
332
+ function rebuildTomlFrontMatter(rawToml, translations) {
333
+ const lines = rawToml.split('\n');
334
+ const result = [];
335
+
336
+ for (const line of lines) {
337
+ const trimmed = line.trim();
338
+ // Skip section headers and comments
339
+ if (trimmed.startsWith('[') || trimmed.startsWith('#') || !trimmed) {
340
+ result.push(line);
341
+ continue;
342
+ }
343
+
344
+ const eqIdx = trimmed.indexOf('=');
345
+ if (eqIdx > 0) {
346
+ const key = trimmed.slice(0, eqIdx).trim();
347
+ if (key in translations) {
348
+ const value = translations[key];
349
+ // TOML always uses double quotes for strings
350
+ const escaped = value.replace(/\\/g, '\\\\').replace(/"/g, '\\"');
351
+ result.push(`${key} = "${escaped}"`);
352
+ continue;
353
+ }
354
+ }
355
+
356
+ result.push(line);
357
+ }
358
+
359
+ return result.join('\n');
360
+ }
361
+
362
+ // -----------------------------------------------------------------
363
+ // Block protection — shield non-translatable content
364
+ // -----------------------------------------------------------------
365
+
366
+ /**
367
+ * Protect non-translatable blocks in Markdown body text.
368
+ *
369
+ * Replaces code blocks, Hugo shortcodes, inline code, and raw HTML
370
+ * with unique placeholders. The translation engine sees the
371
+ * placeholders and leaves them intact. After translation, we
372
+ * restore the original blocks.
373
+ *
374
+ * Protection order matters — we process larger/greedier patterns
375
+ * first to prevent inner patterns from matching within outer blocks.
376
+ *
377
+ * @param {string} body - Raw Markdown body
378
+ * @returns {object} { protectedBody, blocks }
379
+ * - protectedBody: body with placeholders
380
+ * - blocks: Map of placeholder → original content
381
+ */
382
+ function protectBlocks(body) {
383
+ const blocks = new Map();
384
+ let counter = 0;
385
+ let result = body;
386
+
387
+ /**
388
+ * Replace matches with numbered placeholders.
389
+ * Each placeholder is unique and maps back to the original.
390
+ */
391
+ function protect(regex) {
392
+ result = result.replace(regex, (match) => {
393
+ const id = `${PLACEHOLDER_PREFIX}${counter++}${PLACEHOLDER_SUFFIX}`;
394
+ blocks.set(id, match);
395
+ return id;
396
+ });
397
+ }
398
+
399
+ // 1. Fenced code blocks (```lang\n...\n```)
400
+ // Must be first — they can contain shortcodes, HTML, etc.
401
+ protect(/```[\s\S]*?```/g);
402
+
403
+ // 1b. Tilde-fenced code blocks (~~~lang\n...\n~~~)
404
+ // The CommonMark/MDX alternative fence. Without this, a ~~~ block's
405
+ // contents (often code or config that must NOT change) were fed to the
406
+ // LLM and silently translated/mangled.
407
+ protect(/~~~[\s\S]*?~~~/g);
408
+
409
+ // 1c. MDX import/export statements (whole line).
410
+ // MDX files begin lines with `import …` / `export …` that are JS, not
411
+ // prose — translating an identifier or a quoted path breaks the build.
412
+ // Anchored to line start (multiline) so it never matches the word
413
+ // "import"/"export" mid-sentence.
414
+ protect(/^[ \t]*(?:import|export)\b.*$/gm);
415
+
416
+ // 2. Hugo paired shortcodes: {{< name >}}...{{< /name >}} and {{% name %}}...{{% /name %}}
417
+ // Must come before standalone shortcodes so the entire block
418
+ // (including inner content like code) is protected as one unit.
419
+ // Example: {{% highlight go %}}...code...{{% /highlight %}}
420
+ protect(/\{\{[<%]\s*(\w+)[^%>]*[%>]\}\}[\s\S]*?\{\{[<%]\s*\/\1\s*[%>]\}\}/g);
421
+
422
+ // 3. Hugo standalone shortcodes: {{< name params >}} and {{% name params %}}
423
+ // These are unpaired (self-contained on one line).
424
+ protect(/\{\{[<%][^%>]*[%>]\}\}/g);
425
+
426
+ // 4. Inline code (`...`)
427
+ protect(/`[^`\n]+`/g);
428
+
429
+ // 5. HTML blocks and inline HTML tags
430
+ protect(/<[a-zA-Z\/][^>]*>/g);
431
+
432
+ return { protectedBody: result, blocks };
433
+ }
434
+
435
+ /**
436
+ * Restore protected blocks after translation.
437
+ *
438
+ * Restores in reverse order (last captured first) so that
439
+ * nested blocks resolve correctly. When a code block is inside
440
+ * a paired shortcode, the shortcode's stored content contains
441
+ * the code block's placeholder — restoring the shortcode first
442
+ * (reverse order) then the code block ensures full resolution.
443
+ *
444
+ * @param {string} translatedBody - Translated body with placeholders
445
+ * @param {Map} blocks - Map of placeholder → original content
446
+ * @returns {string} Body with original blocks restored
447
+ */
448
+ function restoreBlocks(translatedBody, blocks) {
449
+ let restored = translatedBody;
450
+ // Convert to array and reverse so innermost (last captured) blocks
451
+ // are restored first, resolving nested placeholders correctly
452
+ const entries = [...blocks.entries()].reverse();
453
+ for (const [placeholder, original] of entries) {
454
+ // Use split/join instead of replace to avoid regex special char issues
455
+ restored = restored.split(placeholder).join(original);
456
+ }
457
+ return restored;
458
+ }
459
+
460
+ /**
461
+ * Check if a restored body still contains orphaned placeholder tokens.
462
+ *
463
+ * WHY: The block protection system relies on the LLM preserving
464
+ * ⟦PROTECTED_N⟧ placeholders verbatim during translation. If the
465
+ * model drops, duplicates, or subtly mangles a placeholder (e.g.
466
+ * adds a space, changes 0 to O), restoreBlocks() will leave the
467
+ * broken token in the output. Rather than silently writing corrupted
468
+ * content with orphaned Unicode sentinels or missing code blocks,
469
+ * we detect this and let the caller fall back to the English body
470
+ * with a loud warning.
471
+ *
472
+ * @param {string} text - Body text after restoreBlocks()
473
+ * @returns {boolean} True if orphaned placeholders remain
474
+ */
475
+ function hasOrphanedPlaceholders(text) {
476
+ return text.includes(PLACEHOLDER_PREFIX);
477
+ }
478
+
479
+ // -----------------------------------------------------------------
480
+ // Content file discovery
481
+ // -----------------------------------------------------------------
482
+
483
+ /**
484
+ * Scan a Hugo content directory for source language Markdown files.
485
+ *
486
+ * Uses Hugo's filename convention: files without a language suffix
487
+ * (e.g., my-post.md) or with the source language suffix
488
+ * (e.g., my-post.en.md) are source files.
489
+ *
490
+ * @param {string} contentDir - Path to the content directory
491
+ * @param {string} sourceLocale - Source language code (e.g., 'en')
492
+ * @returns {string[]} Array of absolute paths to source content files
493
+ */
494
+ function discoverContentFiles(contentDir, sourceLocale) {
495
+ const files = [];
496
+
497
+ function walk(dir) {
498
+ if (!fs.existsSync(dir)) return;
499
+ for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
500
+ const fullPath = path.join(dir, entry.name);
501
+ if (entry.isDirectory()) {
502
+ walk(fullPath);
503
+ } else if (entry.isFile() && (entry.name.endsWith('.md') || entry.name.endsWith('.mdx'))) {
504
+ // Check if this is a source file (no language suffix or source language suffix).
505
+ // Strip the .md/.mdx extension before inspecting the language suffix so
506
+ // foo.fr.mdx is recognized the same way foo.fr.md is.
507
+ const base = entry.name.replace(/\.mdx?$/, '');
508
+ const parts = base.split('.');
509
+ const langSuffix = parts.length > 1 ? parts[parts.length - 1] : null;
510
+
511
+ // It's a source file if:
512
+ // 1. No language suffix (e.g., my-post.md)
513
+ // 2. Language suffix matches source locale (e.g., my-post.en.md)
514
+ if (!langSuffix || langSuffix === sourceLocale || !isLikelyLangCode(langSuffix)) {
515
+ files.push(fullPath);
516
+ }
517
+ }
518
+ }
519
+ }
520
+
521
+ walk(contentDir);
522
+ return files.sort();
523
+ }
524
+
525
+ /**
526
+ * Check if a string looks like a language code (2-3 lowercase letters,
527
+ * optionally with a region suffix like zh-TW).
528
+ *
529
+ * WHY: We need to distinguish "my-post.md" (no lang suffix) from
530
+ * "my-post.fr.md" (French) and "version.2.md" (not a lang code).
531
+ */
532
+ function isLikelyLangCode(str) {
533
+ return /^[a-z]{2,3}(-[A-Z]{2})?$/.test(str);
534
+ }
535
+
536
+ /**
537
+ * Generate the target file path for a translated content file.
538
+ *
539
+ * Follows Hugo's filename convention:
540
+ * my-post.md → my-post.fr.md
541
+ * my-post.en.md → my-post.fr.md
542
+ * index.md → index.fr.md
543
+ *
544
+ * @param {string} sourcePath - Path to the source content file
545
+ * @param {string} targetLang - Target language code
546
+ * @param {string} sourceLocale - Source language code
547
+ * @returns {string} Path to the target content file
548
+ */
549
+ function getTargetContentPath(sourcePath, targetLang, sourceLocale) {
550
+ const dir = path.dirname(sourcePath);
551
+ const ext = path.extname(sourcePath); // .md
552
+ const base = path.basename(sourcePath, ext);
553
+
554
+ // Remove source locale suffix if present (my-post.en → my-post)
555
+ const parts = base.split('.');
556
+ const langSuffix = parts.length > 1 ? parts[parts.length - 1] : null;
557
+ const cleanBase = langSuffix === sourceLocale
558
+ ? parts.slice(0, -1).join('.')
559
+ : base;
560
+
561
+ return path.join(dir, `${cleanBase}.${targetLang}${ext}`);
562
+ }
563
+
564
+ /**
565
+ * Build the prompt for translating Markdown content.
566
+ *
567
+ * @param {string} protectedBody - Markdown body with protected placeholders
568
+ * @param {object} langConfig - { name, register }
569
+ * @param {object} options - { sourceLanguageName, promptContext } (defaults to 'English', null)
570
+ * @returns {string} Translation prompt
571
+ */
572
+ function buildContentPrompt(protectedBody, langConfig, options = {}) {
573
+ const sourceLanguageName = options.sourceLanguageName || 'English';
574
+
575
+ // Inject user-provided promptContext between the role line and register.
576
+ // This gives the LLM global context about what it's translating (e.g.,
577
+ // "This is a developer tool README" or "This is medical documentation").
578
+ const contextBlock = options.promptContext
579
+ ? `\nContext: ${options.promptContext}\n`
580
+ : '';
581
+
582
+ // Only mention placeholders when the body actually contains them. When the
583
+ // instruction appears without any ⟦PROTECTED_N⟧ in the text, some models
584
+ // (observed: zh) echo the literal token from the rules into their output,
585
+ // which the orphaned-placeholder check then (rightly) rejects — a
586
+ // deterministic false "corruption" failure on placeholder-free documents.
587
+ const placeholderRule = protectedBody.includes(PLACEHOLDER_PREFIX)
588
+ ? '\n- DO NOT translate or modify anything inside ⟦PROTECTED_N⟧ placeholders. Leave them exactly as they appear.'
589
+ : '';
590
+
591
+ return `You are translating Markdown content from ${sourceLanguageName} to ${langConfig.name}.
592
+ ${contextBlock}
593
+ Register/tone: ${langConfig.register}
594
+
595
+ Rules:
596
+ - Translate ALL human-readable text in the Markdown.
597
+ - Preserve ALL Markdown formatting: headers (#), bold (**), italic (*), links, images, lists, blockquotes, etc.${placeholderRule}
598
+ - Preserve all line breaks, paragraph spacing, and document structure.
599
+ - Proper nouns, product names, and technical terms should remain in the source language.
600
+ - Translate link text but preserve link URLs. For example: [Read more](url) → [Lire la suite](url)
601
+ - Return ONLY the translated Markdown. No code fences, no explanation, no preamble.
602
+
603
+ ---
604
+ ${protectedBody}`;
605
+ }
606
+
607
+ /**
608
+ * Reassemble a complete Hugo content file from translated parts.
609
+ *
610
+ * @param {object} options
611
+ * @param {string} options.rawFrontMatter - Original raw front matter
612
+ * @param {object} options.translatedFields - Map of translated front matter fields
613
+ * @param {string} options.translatedBody - Translated Markdown body
614
+ * @param {boolean} options.hasFrontMatter - Whether the original had front matter
615
+ * @param {string} options.frontMatterFormat - 'yaml' | 'toml' | null
616
+ * @returns {string} Complete content file
617
+ */
618
+ function reassembleContentFile({ rawFrontMatter, translatedFields, translatedBody, hasFrontMatter, frontMatterFormat }) {
619
+ // Normalize the body's surrounding whitespace so we can re-apply the canonical
620
+ // Markdown shape. The parse regex consumes the blank line after the closing
621
+ // fence, and the LLM commonly strips the file's trailing newline — the old
622
+ // reassembly preserved neither, so every synced file lost its post-front-matter
623
+ // blank line and its final newline. We restore both here deterministically.
624
+ const body = translatedBody.replace(/^\n+/, '').replace(/\s+$/, '');
625
+
626
+ if (!hasFrontMatter) {
627
+ return body.length ? `${body}\n` : '';
628
+ }
629
+
630
+ const fence = frontMatterFormat === 'toml' ? '+++' : '---';
631
+ const updatedFrontMatter = frontMatterFormat === 'toml'
632
+ ? rebuildTomlFrontMatter(rawFrontMatter, translatedFields)
633
+ : rebuildFrontMatter(rawFrontMatter, translatedFields);
634
+
635
+ // <fence> … <fence>, then a blank line (the Markdown convention Docusaurus and
636
+ // Hugo emit), then the body, then exactly one trailing newline.
637
+ return body.length
638
+ ? `${fence}\n${updatedFrontMatter}\n${fence}\n\n${body}\n`
639
+ : `${fence}\n${updatedFrontMatter}\n${fence}\n`;
640
+ }
641
+
642
+ // -----------------------------------------------------------------
643
+ // Docusaurus content discovery
644
+ //
645
+ // Docusaurus uses a directory-per-locale layout for content:
646
+ // docs/intro.md → i18n/{locale}/docusaurus-plugin-content-docs/current/intro.md
647
+ // blog/my-post.md → i18n/{locale}/docusaurus-plugin-content-blog/my-post.md
648
+ //
649
+ // Unlike Hugo (which uses filename suffixes like my-post.fr.md),
650
+ // Docusaurus mirrors the entire directory tree under each locale.
651
+ // The content parsing (front matter, block protection) is the same.
652
+ // -----------------------------------------------------------------
653
+
654
+ /**
655
+ * Discover Docusaurus content files (Markdown and MDX) in a source directory.
656
+ *
657
+ * Walks the directory recursively. Unlike Hugo's discoverContentFiles,
658
+ * there's no filename-based language filtering — all .md/.mdx files
659
+ * in the source directory are source files.
660
+ *
661
+ * @param {string} contentDir - Path to the source content directory (e.g., docs/ or blog/)
662
+ * @returns {string[]} Array of absolute paths to Markdown/MDX files
663
+ */
664
+ function discoverDocusaurusContentFiles(contentDir) {
665
+ const files = [];
666
+
667
+ function walk(dir) {
668
+ if (!fs.existsSync(dir)) return;
669
+ for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
670
+ const fullPath = path.join(dir, entry.name);
671
+ if (entry.isDirectory()) {
672
+ // Skip hidden directories and common non-content dirs
673
+ if (entry.name.startsWith('.') || entry.name === 'node_modules') continue;
674
+ walk(fullPath);
675
+ } else if (entry.isFile() && (entry.name.endsWith('.md') || entry.name.endsWith('.mdx'))) {
676
+ files.push(fullPath);
677
+ }
678
+ }
679
+ }
680
+
681
+ walk(contentDir);
682
+ return files.sort();
683
+ }
684
+
685
+ /**
686
+ * Compute the Docusaurus i18n target path for a source content file.
687
+ *
688
+ * Docusaurus layout:
689
+ * Source: docs/guides/foo.md
690
+ * Target: i18n/fr/docusaurus-plugin-content-docs/current/guides/foo.md
691
+ *
692
+ * Source: blog/2026-01-01-hello.md
693
+ * Target: i18n/fr/docusaurus-plugin-content-blog/2026-01-01-hello.md
694
+ *
695
+ * @param {string} sourcePath - Absolute path to the source content file
696
+ * @param {string} sourceDir - Absolute path to the source directory (e.g., /project/docs)
697
+ * @param {string} targetLocale - Target language code (e.g., 'fr')
698
+ * @param {string} i18nDir - Absolute path to the i18n directory (e.g., /project/i18n)
699
+ * @param {string} pluginName - Docusaurus plugin name (e.g., 'docusaurus-plugin-content-docs')
700
+ * @param {string} [versionDir='current'] - Version directory name (Docusaurus versioned docs)
701
+ * @returns {string} Absolute path to the target content file
702
+ */
703
+ function getDocusaurusTargetPath(sourcePath, sourceDir, targetLocale, i18nDir, pluginName, versionDir = 'current') {
704
+ const relPath = path.relative(sourceDir, sourcePath);
705
+ // Docs go under a version directory ('current'), blog does not
706
+ if (pluginName.includes('content-docs')) {
707
+ return path.join(i18nDir, targetLocale, pluginName, versionDir, relPath);
708
+ }
709
+ return path.join(i18nDir, targetLocale, pluginName, relPath);
710
+ }
711
+
712
+ export {
713
+ parseContentFile,
714
+ parseSimpleFrontMatter,
715
+ findUntranslatableNestedFields,
716
+ parseSimpleTomlFrontMatter,
717
+ rebuildFrontMatter,
718
+ rebuildTomlFrontMatter,
719
+ protectBlocks,
720
+ restoreBlocks,
721
+ hasOrphanedPlaceholders,
722
+ discoverContentFiles,
723
+ getTargetContentPath,
724
+ buildContentPrompt,
725
+ reassembleContentFile,
726
+ isLikelyLangCode,
727
+ DEFAULT_TRANSLATABLE_FIELDS,
728
+ NON_TRANSLATABLE_NESTED_FIELDS,
729
+ PLACEHOLDER_PREFIX,
730
+ PLACEHOLDER_SUFFIX,
731
+ discoverDocusaurusContentFiles,
732
+ getDocusaurusTargetPath,
733
+ };