champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
package/lib/content.js
ADDED
|
@@ -0,0 +1,733 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Markdown content translation — translates Hugo content files.
|
|
3
|
+
*
|
|
4
|
+
* WHY: Hugo stores page content as Markdown files with YAML front matter.
|
|
5
|
+
* Unlike i18n string files (key→value pairs), content files need:
|
|
6
|
+
* 1. Front matter parsing — extract translatable fields (title, description)
|
|
7
|
+
* 2. Block protection — shield code blocks, shortcodes, and inline code
|
|
8
|
+
* from the translation engine so they pass through untouched
|
|
9
|
+
* 3. Body translation — send the protected Markdown to the LLM
|
|
10
|
+
* 4. Reassembly — restore protected blocks and rebuild the file
|
|
11
|
+
*
|
|
12
|
+
* Hugo's translation-by-filename convention:
|
|
13
|
+
* content/posts/my-post.md → default language
|
|
14
|
+
* content/posts/my-post.fr.md → French
|
|
15
|
+
* content/posts/my-post.ja.md → Japanese
|
|
16
|
+
*
|
|
17
|
+
* This module handles the parse→protect→translate→restore→write pipeline.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import path from 'node:path';
|
|
22
|
+
|
|
23
|
+
// Sentinel used for protected block placeholders. Uses Unicode brackets
|
|
24
|
+
// that are extremely unlikely to appear in real content, making them
|
|
25
|
+
// safe to use as delimiters even if the LLM generates creative output.
|
|
26
|
+
const PLACEHOLDER_PREFIX = '⟦PROTECTED_';
|
|
27
|
+
const PLACEHOLDER_SUFFIX = '⟧';
|
|
28
|
+
|
|
29
|
+
// Front matter fields that should be translated by default.
|
|
30
|
+
// Other fields (date, draft, tags, slug, weight, etc.) are preserved as-is.
|
|
31
|
+
// `sidebar_label` is the Docusaurus sidebar/menu entry — user-facing prose that
|
|
32
|
+
// was previously left in the source language on every translated page.
|
|
33
|
+
const DEFAULT_TRANSLATABLE_FIELDS = [
|
|
34
|
+
'title',
|
|
35
|
+
'description',
|
|
36
|
+
'summary',
|
|
37
|
+
'subtitle',
|
|
38
|
+
'caption',
|
|
39
|
+
'linkTitle',
|
|
40
|
+
'sidebar_label',
|
|
41
|
+
];
|
|
42
|
+
|
|
43
|
+
// Top-level front matter keys that are structural/taxonomy metadata, NOT human
|
|
44
|
+
// prose — safe to leave untranslated even when they appear as arrays or nested
|
|
45
|
+
// blocks the flat parser can't reach. Anything else that we skip as nested/array
|
|
46
|
+
// gets surfaced by findUntranslatableNestedFields() so it's never silently lost.
|
|
47
|
+
const NON_TRANSLATABLE_NESTED_FIELDS = new Set([
|
|
48
|
+
'tags', 'keywords', 'slug', 'aliases', 'url', 'permalink',
|
|
49
|
+
'date', 'lastmod', 'publishdate', 'expirydate', 'draft', 'weight',
|
|
50
|
+
'type', 'layout', 'author', 'authors', 'categories', 'series',
|
|
51
|
+
'image', 'images', 'id', 'sidebar_position', 'pagination_next',
|
|
52
|
+
'pagination_prev', 'hide_table_of_contents', 'toc_min_heading_level',
|
|
53
|
+
'toc_max_heading_level', 'sidebar_custom_props', 'tags_url',
|
|
54
|
+
]);
|
|
55
|
+
|
|
56
|
+
// Regex for YAML front matter delimiters (--- ... ---)
|
|
57
|
+
const YAML_FM_REGEX = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/;
|
|
58
|
+
|
|
59
|
+
// Regex for TOML front matter delimiters (+++ ... +++)
|
|
60
|
+
const TOML_FM_REGEX = /^\+\+\+\r?\n([\s\S]*?)\r?\n\+\+\+\r?\n?([\s\S]*)$/;
|
|
61
|
+
|
|
62
|
+
// -----------------------------------------------------------------
|
|
63
|
+
// Content file parsing
|
|
64
|
+
// -----------------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Parse a Hugo Markdown content file into structured parts.
|
|
68
|
+
*
|
|
69
|
+
* Hugo supports both YAML (---) and TOML (+++) front matter.
|
|
70
|
+
* We try YAML first (more common), then fall back to TOML.
|
|
71
|
+
*
|
|
72
|
+
* @param {string} raw - Raw file content
|
|
73
|
+
* @returns {object} { frontMatter, rawFrontMatter, body, hasFrontMatter, frontMatterFormat }
|
|
74
|
+
* - frontMatter: parsed key→value map (simple flat parsing)
|
|
75
|
+
* - rawFrontMatter: the raw string between delimiters
|
|
76
|
+
* - body: everything after the front matter
|
|
77
|
+
* - hasFrontMatter: whether front matter was detected
|
|
78
|
+
* - frontMatterFormat: 'yaml' | 'toml' | null
|
|
79
|
+
*/
|
|
80
|
+
function parseContentFile(raw) {
|
|
81
|
+
// Try YAML first (--- ... ---) — most common in Hugo
|
|
82
|
+
const yamlMatch = raw.match(YAML_FM_REGEX);
|
|
83
|
+
if (yamlMatch) {
|
|
84
|
+
return {
|
|
85
|
+
frontMatter: parseSimpleFrontMatter(yamlMatch[1]),
|
|
86
|
+
rawFrontMatter: yamlMatch[1],
|
|
87
|
+
body: yamlMatch[2],
|
|
88
|
+
hasFrontMatter: true,
|
|
89
|
+
frontMatterFormat: 'yaml',
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Try TOML (+++ ... +++)
|
|
94
|
+
const tomlMatch = raw.match(TOML_FM_REGEX);
|
|
95
|
+
if (tomlMatch) {
|
|
96
|
+
return {
|
|
97
|
+
frontMatter: parseSimpleTomlFrontMatter(tomlMatch[1]),
|
|
98
|
+
rawFrontMatter: tomlMatch[1],
|
|
99
|
+
body: tomlMatch[2],
|
|
100
|
+
hasFrontMatter: true,
|
|
101
|
+
frontMatterFormat: 'toml',
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
return { frontMatter: {}, rawFrontMatter: '', body: raw, hasFrontMatter: false, frontMatterFormat: null };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Parse simple YAML front matter into a key→value map.
|
|
110
|
+
*
|
|
111
|
+
* WHY hand-rolled: Hugo front matter is almost always flat key-value
|
|
112
|
+
* pairs (title, description, date, draft, etc.). We only need to
|
|
113
|
+
* extract the translatable string fields. Complex nested YAML,
|
|
114
|
+
* arrays, and multi-line values are preserved as raw strings so
|
|
115
|
+
* they pass through unchanged.
|
|
116
|
+
*
|
|
117
|
+
* @param {string} yaml - Raw YAML content (between --- delimiters)
|
|
118
|
+
* @returns {object} key→value map
|
|
119
|
+
*/
|
|
120
|
+
function parseSimpleFrontMatter(yaml) {
|
|
121
|
+
const result = {};
|
|
122
|
+
|
|
123
|
+
for (const line of yaml.split('\n')) {
|
|
124
|
+
const trimmed = line.trim();
|
|
125
|
+
if (!trimmed || trimmed.startsWith('#')) continue;
|
|
126
|
+
|
|
127
|
+
// Simple key: value pairs only (skip arrays, nested objects)
|
|
128
|
+
if (line.startsWith(' ') || line.startsWith('\t')) continue;
|
|
129
|
+
|
|
130
|
+
const colonIdx = trimmed.indexOf(':');
|
|
131
|
+
if (colonIdx < 0) continue;
|
|
132
|
+
|
|
133
|
+
const key = trimmed.slice(0, colonIdx).trim();
|
|
134
|
+
let value = trimmed.slice(colonIdx + 1).trim();
|
|
135
|
+
|
|
136
|
+
// Skip keys with empty values — they're YAML map/array parent keys
|
|
137
|
+
// (e.g., "tags:" followed by indented array items)
|
|
138
|
+
if (!value) continue;
|
|
139
|
+
|
|
140
|
+
// Skip YAML block scalars (`>`, `|`, `>-`, `|-`, `>+`, indent variants).
|
|
141
|
+
// The value on this line is only the block indicator; the real text lives
|
|
142
|
+
// on the indented lines below (already skipped by the indent check above).
|
|
143
|
+
// Extracting the indicator as a translatable field and re-quoting it on
|
|
144
|
+
// rebuild would orphan those lines and emit invalid YAML — so, per this
|
|
145
|
+
// parser's contract, multi-line values pass through raw and untranslated.
|
|
146
|
+
if (/^[>|][0-9+-]*$/.test(value)) continue;
|
|
147
|
+
|
|
148
|
+
// Unquote if quoted
|
|
149
|
+
if (
|
|
150
|
+
(value.startsWith('"') && value.endsWith('"')) ||
|
|
151
|
+
(value.startsWith("'") && value.endsWith("'"))
|
|
152
|
+
) {
|
|
153
|
+
value = value.slice(1, -1);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
result[key] = value;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
return result;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Find top-level front matter keys that hold array/nested values the flat
|
|
164
|
+
* parser cannot reach — and that are NOT known structural/taxonomy metadata.
|
|
165
|
+
*
|
|
166
|
+
* WHY: parseSimpleFrontMatter() silently skips every indented/array line by
|
|
167
|
+
* design. That's correct for `tags`, `slug`, `date`, etc., but it also means a
|
|
168
|
+
* human-prose container like `related:` (a list of `{title, url}` entries) and
|
|
169
|
+
* other nested fields pass through entirely untranslated with no signal. The
|
|
170
|
+
* never-silent doctrine says: at minimum, tell the user. Callers surface these
|
|
171
|
+
* as a warning so the omission is visible instead of shipping half-translated
|
|
172
|
+
* pages that look complete.
|
|
173
|
+
*
|
|
174
|
+
* Block scalars (`>`, `|`) are intentionally excluded — they're a separate,
|
|
175
|
+
* already-documented skip (extracting the indicator would emit invalid YAML).
|
|
176
|
+
*
|
|
177
|
+
* @param {string} rawFrontMatter - Raw YAML front matter (between --- delimiters)
|
|
178
|
+
* @returns {string[]} Names of skipped, non-allowlisted nested/array fields
|
|
179
|
+
*/
|
|
180
|
+
function findUntranslatableNestedFields(rawFrontMatter) {
|
|
181
|
+
const found = [];
|
|
182
|
+
if (!rawFrontMatter) return found;
|
|
183
|
+
|
|
184
|
+
for (const line of rawFrontMatter.split('\n')) {
|
|
185
|
+
if (!line.trim() || line.trim().startsWith('#')) continue;
|
|
186
|
+
// Only consider top-level keys — indented lines are the nested content.
|
|
187
|
+
if (line.startsWith(' ') || line.startsWith('\t')) continue;
|
|
188
|
+
|
|
189
|
+
const trimmed = line.trim();
|
|
190
|
+
const colonIdx = trimmed.indexOf(':');
|
|
191
|
+
if (colonIdx < 0) continue;
|
|
192
|
+
|
|
193
|
+
const key = trimmed.slice(0, colonIdx).trim();
|
|
194
|
+
const value = trimmed.slice(colonIdx + 1).trim();
|
|
195
|
+
|
|
196
|
+
if (NON_TRANSLATABLE_NESTED_FIELDS.has(key)) continue;
|
|
197
|
+
if (/^[>|][0-9+-]*$/.test(value)) continue; // block scalar — separate case
|
|
198
|
+
|
|
199
|
+
// Empty value → an indented block/array follows; or an inline array/object.
|
|
200
|
+
if (value === '' || value.startsWith('[') || value.startsWith('{')) {
|
|
201
|
+
found.push(key);
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
return found;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* Parse simple TOML front matter into a key→value map.
|
|
210
|
+
*
|
|
211
|
+
* TOML front matter uses `key = "value"` syntax (with = instead of :).
|
|
212
|
+
* Like the YAML parser, we only extract flat key-value pairs and
|
|
213
|
+
* skip complex nested structures.
|
|
214
|
+
*
|
|
215
|
+
* @param {string} toml - Raw TOML content (between +++ delimiters)
|
|
216
|
+
* @returns {object} key→value map
|
|
217
|
+
*/
|
|
218
|
+
function parseSimpleTomlFrontMatter(toml) {
|
|
219
|
+
const result = {};
|
|
220
|
+
// Track whether we're inside a nested table section.
|
|
221
|
+
// Once we hit a [section] header, all subsequent keys belong to that
|
|
222
|
+
// table and should NOT be treated as top-level front matter keys.
|
|
223
|
+
// There's no way to "exit" a TOML table — a new top-level key after
|
|
224
|
+
// a section is technically invalid TOML, but we handle it gracefully
|
|
225
|
+
// by staying in nested mode until the end of the front matter block.
|
|
226
|
+
let insideNestedTable = false;
|
|
227
|
+
|
|
228
|
+
for (const line of toml.split('\n')) {
|
|
229
|
+
const trimmed = line.trim();
|
|
230
|
+
if (!trimmed || trimmed.startsWith('#')) continue;
|
|
231
|
+
|
|
232
|
+
// Detect TOML section headers ([section]) and array-of-tables ([[section]])
|
|
233
|
+
// WHY we warn: Keys inside nested tables (e.g. [params] description = "...")
|
|
234
|
+
// won't be parsed or translated. The user needs to know this so they don't
|
|
235
|
+
// ship partially-translated content thinking everything was handled.
|
|
236
|
+
if (trimmed.startsWith('[')) {
|
|
237
|
+
insideNestedTable = true;
|
|
238
|
+
const tableName = trimmed.replace(/^\[+|\]+$/g, '').trim();
|
|
239
|
+
console.warn(
|
|
240
|
+
` [WARN] TOML front matter: nested table [${tableName}] detected — ` +
|
|
241
|
+
`keys inside it will not be translated. Flatten translatable fields to the top level.`
|
|
242
|
+
);
|
|
243
|
+
continue;
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
// Skip all keys inside nested tables — they belong to the section,
|
|
247
|
+
// not to the top-level front matter we're interested in.
|
|
248
|
+
if (insideNestedTable) continue;
|
|
249
|
+
|
|
250
|
+
// Match key = value pairs
|
|
251
|
+
const eqIdx = trimmed.indexOf('=');
|
|
252
|
+
if (eqIdx < 0) continue;
|
|
253
|
+
|
|
254
|
+
const key = trimmed.slice(0, eqIdx).trim();
|
|
255
|
+
let value = trimmed.slice(eqIdx + 1).trim();
|
|
256
|
+
|
|
257
|
+
// Skip keys with empty values
|
|
258
|
+
if (!value) continue;
|
|
259
|
+
|
|
260
|
+
// Unquote if quoted
|
|
261
|
+
if (
|
|
262
|
+
(value.startsWith('"') && value.endsWith('"')) ||
|
|
263
|
+
(value.startsWith("'") && value.endsWith("'"))
|
|
264
|
+
) {
|
|
265
|
+
value = value.slice(1, -1);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
result[key] = value;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
return result;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* Rebuild front matter YAML with translated fields.
|
|
276
|
+
*
|
|
277
|
+
* Preserves the original formatting for non-translated fields by
|
|
278
|
+
* doing line-by-line replacement rather than full re-serialization.
|
|
279
|
+
* This keeps array fields, comments, and complex YAML intact.
|
|
280
|
+
*
|
|
281
|
+
* @param {string} rawYaml - Original raw YAML front matter
|
|
282
|
+
* @param {object} translations - Map of field name → translated value
|
|
283
|
+
* @returns {string} Updated YAML front matter
|
|
284
|
+
*/
|
|
285
|
+
function rebuildFrontMatter(rawYaml, translations) {
|
|
286
|
+
const lines = rawYaml.split('\n');
|
|
287
|
+
const result = [];
|
|
288
|
+
|
|
289
|
+
for (const line of lines) {
|
|
290
|
+
// Check if this line is a top-level key: value that we have a translation for
|
|
291
|
+
const trimmed = line.trim();
|
|
292
|
+
if (!trimmed.startsWith(' ') && !trimmed.startsWith('\t') && !trimmed.startsWith('#')) {
|
|
293
|
+
const colonIdx = trimmed.indexOf(':');
|
|
294
|
+
if (colonIdx > 0) {
|
|
295
|
+
const key = trimmed.slice(0, colonIdx).trim();
|
|
296
|
+
const rawValue = trimmed.slice(colonIdx + 1).trim();
|
|
297
|
+
// Never rewrite a block scalar (`>`, `|`, …): its text spans the
|
|
298
|
+
// indented lines below, so replacing just this line would orphan them
|
|
299
|
+
// and emit invalid YAML. Leave it raw (parseSimpleFrontMatter already
|
|
300
|
+
// skips these, so this is defense-in-depth).
|
|
301
|
+
if (key in translations && !/^[>|][0-9+-]*$/.test(rawValue)) {
|
|
302
|
+
// Replace the value, preserving the key and indentation
|
|
303
|
+
const value = translations[key];
|
|
304
|
+
const needsQuotes = value.includes(':') || value.includes('#') ||
|
|
305
|
+
value.includes('"') || value.includes("'") ||
|
|
306
|
+
value.startsWith(' ') || value.endsWith(' ');
|
|
307
|
+
const formatted = needsQuotes
|
|
308
|
+
? `"${value.replace(/\\/g, '\\\\').replace(/"/g, '\\"')}"`
|
|
309
|
+
: `"${value}"`;
|
|
310
|
+
result.push(`${key}: ${formatted}`);
|
|
311
|
+
continue;
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
result.push(line);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
return result.join('\n');
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Rebuild TOML front matter with translated fields.
|
|
324
|
+
*
|
|
325
|
+
* Same approach as YAML — line-by-line replacement to preserve
|
|
326
|
+
* formatting for non-translated fields.
|
|
327
|
+
*
|
|
328
|
+
* @param {string} rawToml - Original raw TOML front matter
|
|
329
|
+
* @param {object} translations - Map of field name → translated value
|
|
330
|
+
* @returns {string} Updated TOML front matter
|
|
331
|
+
*/
|
|
332
|
+
function rebuildTomlFrontMatter(rawToml, translations) {
|
|
333
|
+
const lines = rawToml.split('\n');
|
|
334
|
+
const result = [];
|
|
335
|
+
|
|
336
|
+
for (const line of lines) {
|
|
337
|
+
const trimmed = line.trim();
|
|
338
|
+
// Skip section headers and comments
|
|
339
|
+
if (trimmed.startsWith('[') || trimmed.startsWith('#') || !trimmed) {
|
|
340
|
+
result.push(line);
|
|
341
|
+
continue;
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
const eqIdx = trimmed.indexOf('=');
|
|
345
|
+
if (eqIdx > 0) {
|
|
346
|
+
const key = trimmed.slice(0, eqIdx).trim();
|
|
347
|
+
if (key in translations) {
|
|
348
|
+
const value = translations[key];
|
|
349
|
+
// TOML always uses double quotes for strings
|
|
350
|
+
const escaped = value.replace(/\\/g, '\\\\').replace(/"/g, '\\"');
|
|
351
|
+
result.push(`${key} = "${escaped}"`);
|
|
352
|
+
continue;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
result.push(line);
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
return result.join('\n');
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
// -----------------------------------------------------------------
|
|
363
|
+
// Block protection — shield non-translatable content
|
|
364
|
+
// -----------------------------------------------------------------
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Protect non-translatable blocks in Markdown body text.
|
|
368
|
+
*
|
|
369
|
+
* Replaces code blocks, Hugo shortcodes, inline code, and raw HTML
|
|
370
|
+
* with unique placeholders. The translation engine sees the
|
|
371
|
+
* placeholders and leaves them intact. After translation, we
|
|
372
|
+
* restore the original blocks.
|
|
373
|
+
*
|
|
374
|
+
* Protection order matters — we process larger/greedier patterns
|
|
375
|
+
* first to prevent inner patterns from matching within outer blocks.
|
|
376
|
+
*
|
|
377
|
+
* @param {string} body - Raw Markdown body
|
|
378
|
+
* @returns {object} { protectedBody, blocks }
|
|
379
|
+
* - protectedBody: body with placeholders
|
|
380
|
+
* - blocks: Map of placeholder → original content
|
|
381
|
+
*/
|
|
382
|
+
function protectBlocks(body) {
|
|
383
|
+
const blocks = new Map();
|
|
384
|
+
let counter = 0;
|
|
385
|
+
let result = body;
|
|
386
|
+
|
|
387
|
+
/**
|
|
388
|
+
* Replace matches with numbered placeholders.
|
|
389
|
+
* Each placeholder is unique and maps back to the original.
|
|
390
|
+
*/
|
|
391
|
+
function protect(regex) {
|
|
392
|
+
result = result.replace(regex, (match) => {
|
|
393
|
+
const id = `${PLACEHOLDER_PREFIX}${counter++}${PLACEHOLDER_SUFFIX}`;
|
|
394
|
+
blocks.set(id, match);
|
|
395
|
+
return id;
|
|
396
|
+
});
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
// 1. Fenced code blocks (```lang\n...\n```)
|
|
400
|
+
// Must be first — they can contain shortcodes, HTML, etc.
|
|
401
|
+
protect(/```[\s\S]*?```/g);
|
|
402
|
+
|
|
403
|
+
// 1b. Tilde-fenced code blocks (~~~lang\n...\n~~~)
|
|
404
|
+
// The CommonMark/MDX alternative fence. Without this, a ~~~ block's
|
|
405
|
+
// contents (often code or config that must NOT change) were fed to the
|
|
406
|
+
// LLM and silently translated/mangled.
|
|
407
|
+
protect(/~~~[\s\S]*?~~~/g);
|
|
408
|
+
|
|
409
|
+
// 1c. MDX import/export statements (whole line).
|
|
410
|
+
// MDX files begin lines with `import …` / `export …` that are JS, not
|
|
411
|
+
// prose — translating an identifier or a quoted path breaks the build.
|
|
412
|
+
// Anchored to line start (multiline) so it never matches the word
|
|
413
|
+
// "import"/"export" mid-sentence.
|
|
414
|
+
protect(/^[ \t]*(?:import|export)\b.*$/gm);
|
|
415
|
+
|
|
416
|
+
// 2. Hugo paired shortcodes: {{< name >}}...{{< /name >}} and {{% name %}}...{{% /name %}}
|
|
417
|
+
// Must come before standalone shortcodes so the entire block
|
|
418
|
+
// (including inner content like code) is protected as one unit.
|
|
419
|
+
// Example: {{% highlight go %}}...code...{{% /highlight %}}
|
|
420
|
+
protect(/\{\{[<%]\s*(\w+)[^%>]*[%>]\}\}[\s\S]*?\{\{[<%]\s*\/\1\s*[%>]\}\}/g);
|
|
421
|
+
|
|
422
|
+
// 3. Hugo standalone shortcodes: {{< name params >}} and {{% name params %}}
|
|
423
|
+
// These are unpaired (self-contained on one line).
|
|
424
|
+
protect(/\{\{[<%][^%>]*[%>]\}\}/g);
|
|
425
|
+
|
|
426
|
+
// 4. Inline code (`...`)
|
|
427
|
+
protect(/`[^`\n]+`/g);
|
|
428
|
+
|
|
429
|
+
// 5. HTML blocks and inline HTML tags
|
|
430
|
+
protect(/<[a-zA-Z\/][^>]*>/g);
|
|
431
|
+
|
|
432
|
+
return { protectedBody: result, blocks };
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
/**
|
|
436
|
+
* Restore protected blocks after translation.
|
|
437
|
+
*
|
|
438
|
+
* Restores in reverse order (last captured first) so that
|
|
439
|
+
* nested blocks resolve correctly. When a code block is inside
|
|
440
|
+
* a paired shortcode, the shortcode's stored content contains
|
|
441
|
+
* the code block's placeholder — restoring the shortcode first
|
|
442
|
+
* (reverse order) then the code block ensures full resolution.
|
|
443
|
+
*
|
|
444
|
+
* @param {string} translatedBody - Translated body with placeholders
|
|
445
|
+
* @param {Map} blocks - Map of placeholder → original content
|
|
446
|
+
* @returns {string} Body with original blocks restored
|
|
447
|
+
*/
|
|
448
|
+
function restoreBlocks(translatedBody, blocks) {
|
|
449
|
+
let restored = translatedBody;
|
|
450
|
+
// Convert to array and reverse so innermost (last captured) blocks
|
|
451
|
+
// are restored first, resolving nested placeholders correctly
|
|
452
|
+
const entries = [...blocks.entries()].reverse();
|
|
453
|
+
for (const [placeholder, original] of entries) {
|
|
454
|
+
// Use split/join instead of replace to avoid regex special char issues
|
|
455
|
+
restored = restored.split(placeholder).join(original);
|
|
456
|
+
}
|
|
457
|
+
return restored;
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* Check if a restored body still contains orphaned placeholder tokens.
|
|
462
|
+
*
|
|
463
|
+
* WHY: The block protection system relies on the LLM preserving
|
|
464
|
+
* ⟦PROTECTED_N⟧ placeholders verbatim during translation. If the
|
|
465
|
+
* model drops, duplicates, or subtly mangles a placeholder (e.g.
|
|
466
|
+
* adds a space, changes 0 to O), restoreBlocks() will leave the
|
|
467
|
+
* broken token in the output. Rather than silently writing corrupted
|
|
468
|
+
* content with orphaned Unicode sentinels or missing code blocks,
|
|
469
|
+
* we detect this and let the caller fall back to the English body
|
|
470
|
+
* with a loud warning.
|
|
471
|
+
*
|
|
472
|
+
* @param {string} text - Body text after restoreBlocks()
|
|
473
|
+
* @returns {boolean} True if orphaned placeholders remain
|
|
474
|
+
*/
|
|
475
|
+
function hasOrphanedPlaceholders(text) {
|
|
476
|
+
return text.includes(PLACEHOLDER_PREFIX);
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
// -----------------------------------------------------------------
|
|
480
|
+
// Content file discovery
|
|
481
|
+
// -----------------------------------------------------------------
|
|
482
|
+
|
|
483
|
+
/**
|
|
484
|
+
* Scan a Hugo content directory for source language Markdown files.
|
|
485
|
+
*
|
|
486
|
+
* Uses Hugo's filename convention: files without a language suffix
|
|
487
|
+
* (e.g., my-post.md) or with the source language suffix
|
|
488
|
+
* (e.g., my-post.en.md) are source files.
|
|
489
|
+
*
|
|
490
|
+
* @param {string} contentDir - Path to the content directory
|
|
491
|
+
* @param {string} sourceLocale - Source language code (e.g., 'en')
|
|
492
|
+
* @returns {string[]} Array of absolute paths to source content files
|
|
493
|
+
*/
|
|
494
|
+
function discoverContentFiles(contentDir, sourceLocale) {
|
|
495
|
+
const files = [];
|
|
496
|
+
|
|
497
|
+
function walk(dir) {
|
|
498
|
+
if (!fs.existsSync(dir)) return;
|
|
499
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
500
|
+
const fullPath = path.join(dir, entry.name);
|
|
501
|
+
if (entry.isDirectory()) {
|
|
502
|
+
walk(fullPath);
|
|
503
|
+
} else if (entry.isFile() && (entry.name.endsWith('.md') || entry.name.endsWith('.mdx'))) {
|
|
504
|
+
// Check if this is a source file (no language suffix or source language suffix).
|
|
505
|
+
// Strip the .md/.mdx extension before inspecting the language suffix so
|
|
506
|
+
// foo.fr.mdx is recognized the same way foo.fr.md is.
|
|
507
|
+
const base = entry.name.replace(/\.mdx?$/, '');
|
|
508
|
+
const parts = base.split('.');
|
|
509
|
+
const langSuffix = parts.length > 1 ? parts[parts.length - 1] : null;
|
|
510
|
+
|
|
511
|
+
// It's a source file if:
|
|
512
|
+
// 1. No language suffix (e.g., my-post.md)
|
|
513
|
+
// 2. Language suffix matches source locale (e.g., my-post.en.md)
|
|
514
|
+
if (!langSuffix || langSuffix === sourceLocale || !isLikelyLangCode(langSuffix)) {
|
|
515
|
+
files.push(fullPath);
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
}
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
walk(contentDir);
|
|
522
|
+
return files.sort();
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
/**
|
|
526
|
+
* Check if a string looks like a language code (2-3 lowercase letters,
|
|
527
|
+
* optionally with a region suffix like zh-TW).
|
|
528
|
+
*
|
|
529
|
+
* WHY: We need to distinguish "my-post.md" (no lang suffix) from
|
|
530
|
+
* "my-post.fr.md" (French) and "version.2.md" (not a lang code).
|
|
531
|
+
*/
|
|
532
|
+
function isLikelyLangCode(str) {
|
|
533
|
+
return /^[a-z]{2,3}(-[A-Z]{2})?$/.test(str);
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
/**
|
|
537
|
+
* Generate the target file path for a translated content file.
|
|
538
|
+
*
|
|
539
|
+
* Follows Hugo's filename convention:
|
|
540
|
+
* my-post.md → my-post.fr.md
|
|
541
|
+
* my-post.en.md → my-post.fr.md
|
|
542
|
+
* index.md → index.fr.md
|
|
543
|
+
*
|
|
544
|
+
* @param {string} sourcePath - Path to the source content file
|
|
545
|
+
* @param {string} targetLang - Target language code
|
|
546
|
+
* @param {string} sourceLocale - Source language code
|
|
547
|
+
* @returns {string} Path to the target content file
|
|
548
|
+
*/
|
|
549
|
+
function getTargetContentPath(sourcePath, targetLang, sourceLocale) {
|
|
550
|
+
const dir = path.dirname(sourcePath);
|
|
551
|
+
const ext = path.extname(sourcePath); // .md
|
|
552
|
+
const base = path.basename(sourcePath, ext);
|
|
553
|
+
|
|
554
|
+
// Remove source locale suffix if present (my-post.en → my-post)
|
|
555
|
+
const parts = base.split('.');
|
|
556
|
+
const langSuffix = parts.length > 1 ? parts[parts.length - 1] : null;
|
|
557
|
+
const cleanBase = langSuffix === sourceLocale
|
|
558
|
+
? parts.slice(0, -1).join('.')
|
|
559
|
+
: base;
|
|
560
|
+
|
|
561
|
+
return path.join(dir, `${cleanBase}.${targetLang}${ext}`);
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
/**
|
|
565
|
+
* Build the prompt for translating Markdown content.
|
|
566
|
+
*
|
|
567
|
+
* @param {string} protectedBody - Markdown body with protected placeholders
|
|
568
|
+
* @param {object} langConfig - { name, register }
|
|
569
|
+
* @param {object} options - { sourceLanguageName, promptContext } (defaults to 'English', null)
|
|
570
|
+
* @returns {string} Translation prompt
|
|
571
|
+
*/
|
|
572
|
+
function buildContentPrompt(protectedBody, langConfig, options = {}) {
|
|
573
|
+
const sourceLanguageName = options.sourceLanguageName || 'English';
|
|
574
|
+
|
|
575
|
+
// Inject user-provided promptContext between the role line and register.
|
|
576
|
+
// This gives the LLM global context about what it's translating (e.g.,
|
|
577
|
+
// "This is a developer tool README" or "This is medical documentation").
|
|
578
|
+
const contextBlock = options.promptContext
|
|
579
|
+
? `\nContext: ${options.promptContext}\n`
|
|
580
|
+
: '';
|
|
581
|
+
|
|
582
|
+
// Only mention placeholders when the body actually contains them. When the
|
|
583
|
+
// instruction appears without any ⟦PROTECTED_N⟧ in the text, some models
|
|
584
|
+
// (observed: zh) echo the literal token from the rules into their output,
|
|
585
|
+
// which the orphaned-placeholder check then (rightly) rejects — a
|
|
586
|
+
// deterministic false "corruption" failure on placeholder-free documents.
|
|
587
|
+
const placeholderRule = protectedBody.includes(PLACEHOLDER_PREFIX)
|
|
588
|
+
? '\n- DO NOT translate or modify anything inside ⟦PROTECTED_N⟧ placeholders. Leave them exactly as they appear.'
|
|
589
|
+
: '';
|
|
590
|
+
|
|
591
|
+
return `You are translating Markdown content from ${sourceLanguageName} to ${langConfig.name}.
|
|
592
|
+
${contextBlock}
|
|
593
|
+
Register/tone: ${langConfig.register}
|
|
594
|
+
|
|
595
|
+
Rules:
|
|
596
|
+
- Translate ALL human-readable text in the Markdown.
|
|
597
|
+
- Preserve ALL Markdown formatting: headers (#), bold (**), italic (*), links, images, lists, blockquotes, etc.${placeholderRule}
|
|
598
|
+
- Preserve all line breaks, paragraph spacing, and document structure.
|
|
599
|
+
- Proper nouns, product names, and technical terms should remain in the source language.
|
|
600
|
+
- Translate link text but preserve link URLs. For example: [Read more](url) → [Lire la suite](url)
|
|
601
|
+
- Return ONLY the translated Markdown. No code fences, no explanation, no preamble.
|
|
602
|
+
|
|
603
|
+
---
|
|
604
|
+
${protectedBody}`;
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
/**
|
|
608
|
+
* Reassemble a complete Hugo content file from translated parts.
|
|
609
|
+
*
|
|
610
|
+
* @param {object} options
|
|
611
|
+
* @param {string} options.rawFrontMatter - Original raw front matter
|
|
612
|
+
* @param {object} options.translatedFields - Map of translated front matter fields
|
|
613
|
+
* @param {string} options.translatedBody - Translated Markdown body
|
|
614
|
+
* @param {boolean} options.hasFrontMatter - Whether the original had front matter
|
|
615
|
+
* @param {string} options.frontMatterFormat - 'yaml' | 'toml' | null
|
|
616
|
+
* @returns {string} Complete content file
|
|
617
|
+
*/
|
|
618
|
+
function reassembleContentFile({ rawFrontMatter, translatedFields, translatedBody, hasFrontMatter, frontMatterFormat }) {
|
|
619
|
+
// Normalize the body's surrounding whitespace so we can re-apply the canonical
|
|
620
|
+
// Markdown shape. The parse regex consumes the blank line after the closing
|
|
621
|
+
// fence, and the LLM commonly strips the file's trailing newline — the old
|
|
622
|
+
// reassembly preserved neither, so every synced file lost its post-front-matter
|
|
623
|
+
// blank line and its final newline. We restore both here deterministically.
|
|
624
|
+
const body = translatedBody.replace(/^\n+/, '').replace(/\s+$/, '');
|
|
625
|
+
|
|
626
|
+
if (!hasFrontMatter) {
|
|
627
|
+
return body.length ? `${body}\n` : '';
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
const fence = frontMatterFormat === 'toml' ? '+++' : '---';
|
|
631
|
+
const updatedFrontMatter = frontMatterFormat === 'toml'
|
|
632
|
+
? rebuildTomlFrontMatter(rawFrontMatter, translatedFields)
|
|
633
|
+
: rebuildFrontMatter(rawFrontMatter, translatedFields);
|
|
634
|
+
|
|
635
|
+
// <fence> … <fence>, then a blank line (the Markdown convention Docusaurus and
|
|
636
|
+
// Hugo emit), then the body, then exactly one trailing newline.
|
|
637
|
+
return body.length
|
|
638
|
+
? `${fence}\n${updatedFrontMatter}\n${fence}\n\n${body}\n`
|
|
639
|
+
: `${fence}\n${updatedFrontMatter}\n${fence}\n`;
|
|
640
|
+
}
|
|
641
|
+
|
|
642
|
+
// -----------------------------------------------------------------
|
|
643
|
+
// Docusaurus content discovery
|
|
644
|
+
//
|
|
645
|
+
// Docusaurus uses a directory-per-locale layout for content:
|
|
646
|
+
// docs/intro.md → i18n/{locale}/docusaurus-plugin-content-docs/current/intro.md
|
|
647
|
+
// blog/my-post.md → i18n/{locale}/docusaurus-plugin-content-blog/my-post.md
|
|
648
|
+
//
|
|
649
|
+
// Unlike Hugo (which uses filename suffixes like my-post.fr.md),
|
|
650
|
+
// Docusaurus mirrors the entire directory tree under each locale.
|
|
651
|
+
// The content parsing (front matter, block protection) is the same.
|
|
652
|
+
// -----------------------------------------------------------------
|
|
653
|
+
|
|
654
|
+
/**
|
|
655
|
+
* Discover Docusaurus content files (Markdown and MDX) in a source directory.
|
|
656
|
+
*
|
|
657
|
+
* Walks the directory recursively. Unlike Hugo's discoverContentFiles,
|
|
658
|
+
* there's no filename-based language filtering — all .md/.mdx files
|
|
659
|
+
* in the source directory are source files.
|
|
660
|
+
*
|
|
661
|
+
* @param {string} contentDir - Path to the source content directory (e.g., docs/ or blog/)
|
|
662
|
+
* @returns {string[]} Array of absolute paths to Markdown/MDX files
|
|
663
|
+
*/
|
|
664
|
+
function discoverDocusaurusContentFiles(contentDir) {
|
|
665
|
+
const files = [];
|
|
666
|
+
|
|
667
|
+
function walk(dir) {
|
|
668
|
+
if (!fs.existsSync(dir)) return;
|
|
669
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
670
|
+
const fullPath = path.join(dir, entry.name);
|
|
671
|
+
if (entry.isDirectory()) {
|
|
672
|
+
// Skip hidden directories and common non-content dirs
|
|
673
|
+
if (entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
|
674
|
+
walk(fullPath);
|
|
675
|
+
} else if (entry.isFile() && (entry.name.endsWith('.md') || entry.name.endsWith('.mdx'))) {
|
|
676
|
+
files.push(fullPath);
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
walk(contentDir);
|
|
682
|
+
return files.sort();
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
/**
|
|
686
|
+
* Compute the Docusaurus i18n target path for a source content file.
|
|
687
|
+
*
|
|
688
|
+
* Docusaurus layout:
|
|
689
|
+
* Source: docs/guides/foo.md
|
|
690
|
+
* Target: i18n/fr/docusaurus-plugin-content-docs/current/guides/foo.md
|
|
691
|
+
*
|
|
692
|
+
* Source: blog/2026-01-01-hello.md
|
|
693
|
+
* Target: i18n/fr/docusaurus-plugin-content-blog/2026-01-01-hello.md
|
|
694
|
+
*
|
|
695
|
+
* @param {string} sourcePath - Absolute path to the source content file
|
|
696
|
+
* @param {string} sourceDir - Absolute path to the source directory (e.g., /project/docs)
|
|
697
|
+
* @param {string} targetLocale - Target language code (e.g., 'fr')
|
|
698
|
+
* @param {string} i18nDir - Absolute path to the i18n directory (e.g., /project/i18n)
|
|
699
|
+
* @param {string} pluginName - Docusaurus plugin name (e.g., 'docusaurus-plugin-content-docs')
|
|
700
|
+
* @param {string} [versionDir='current'] - Version directory name (Docusaurus versioned docs)
|
|
701
|
+
* @returns {string} Absolute path to the target content file
|
|
702
|
+
*/
|
|
703
|
+
function getDocusaurusTargetPath(sourcePath, sourceDir, targetLocale, i18nDir, pluginName, versionDir = 'current') {
|
|
704
|
+
const relPath = path.relative(sourceDir, sourcePath);
|
|
705
|
+
// Docs go under a version directory ('current'), blog does not
|
|
706
|
+
if (pluginName.includes('content-docs')) {
|
|
707
|
+
return path.join(i18nDir, targetLocale, pluginName, versionDir, relPath);
|
|
708
|
+
}
|
|
709
|
+
return path.join(i18nDir, targetLocale, pluginName, relPath);
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
export {
|
|
713
|
+
parseContentFile,
|
|
714
|
+
parseSimpleFrontMatter,
|
|
715
|
+
findUntranslatableNestedFields,
|
|
716
|
+
parseSimpleTomlFrontMatter,
|
|
717
|
+
rebuildFrontMatter,
|
|
718
|
+
rebuildTomlFrontMatter,
|
|
719
|
+
protectBlocks,
|
|
720
|
+
restoreBlocks,
|
|
721
|
+
hasOrphanedPlaceholders,
|
|
722
|
+
discoverContentFiles,
|
|
723
|
+
getTargetContentPath,
|
|
724
|
+
buildContentPrompt,
|
|
725
|
+
reassembleContentFile,
|
|
726
|
+
isLikelyLangCode,
|
|
727
|
+
DEFAULT_TRANSLATABLE_FIELDS,
|
|
728
|
+
NON_TRANSLATABLE_NESTED_FIELDS,
|
|
729
|
+
PLACEHOLDER_PREFIX,
|
|
730
|
+
PLACEHOLDER_SUFFIX,
|
|
731
|
+
discoverDocusaurusContentFiles,
|
|
732
|
+
getDocusaurusTargetPath,
|
|
733
|
+
};
|