jekyll-theme-zer0 1.26.0 → 1.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +558 -5
  3. data/README.md +10 -27
  4. data/_data/authors.yml +4 -3
  5. data/_data/backlog.yml +28 -0
  6. data/_data/consumers.yml +147 -0
  7. data/_data/features.yml +142 -23
  8. data/_data/i18n/fr.yml +98 -0
  9. data/_data/i18n/languages.yml +36 -0
  10. data/_data/i18n/manifest.yml +1502 -0
  11. data/_data/navigation/docs.yml +2 -0
  12. data/_data/navigation/main.yml +6 -0
  13. data/_data/series.yml +19 -0
  14. data/_data/theme-manifest.yml +348 -268
  15. data/_data/ui-text.yml +36 -246
  16. data/_includes/README.md +25 -2
  17. data/_includes/analytics/google-tag-manager-body.html +11 -2
  18. data/_includes/analytics/google-tag-manager-head.html +16 -6
  19. data/_includes/analytics/posthog.html +33 -4
  20. data/_includes/components/abc-letter.html +43 -0
  21. data/_includes/components/author-avatar-url.html +4 -2
  22. data/_includes/components/book-card.html +42 -0
  23. data/_includes/components/book-nav.html +80 -0
  24. data/_includes/components/book-plate.html +31 -0
  25. data/_includes/components/book-toc.html +47 -0
  26. data/_includes/components/bookshelf.html +68 -0
  27. data/_includes/components/card-grid.html +60 -0
  28. data/_includes/components/data-card.html +95 -0
  29. data/_includes/components/env-switcher.html +3 -1
  30. data/_includes/components/halfmoon.html +5 -1
  31. data/_includes/components/language-toggle.html +81 -0
  32. data/_includes/components/page-feedback.html +65 -2
  33. data/_includes/components/search-modal.html +2 -2
  34. data/_includes/components/shortcuts-modal.html +1 -1
  35. data/_includes/components/theme-controls-bar.html +10 -2
  36. data/_includes/components/theme-customizer.html +8 -2
  37. data/_includes/components/translation-notice.html +27 -0
  38. data/_includes/content/intro.html +16 -15
  39. data/_includes/content/seo.html +9 -3
  40. data/_includes/core/color-mode-init.html +13 -4
  41. data/_includes/core/favicon.html +46 -0
  42. data/_includes/core/footer.html +9 -4
  43. data/_includes/core/head.html +18 -0
  44. data/_includes/core/header.html +9 -4
  45. data/_includes/core/hreflang.html +33 -0
  46. data/_includes/core/i18n.html +36 -0
  47. data/_includes/custom/body-end.html +18 -0
  48. data/_includes/custom/body-start.html +17 -0
  49. data/_includes/custom/footer.html +18 -0
  50. data/_includes/custom/head.html +18 -0
  51. data/_includes/navigation/breadcrumbs.html +1 -1
  52. data/_includes/navigation/local-graph.html +28 -2
  53. data/_includes/navigation/nav-tree.html +3 -3
  54. data/_includes/navigation/navbar.html +5 -4
  55. data/_includes/navigation/sidebar-config.html +21 -0
  56. data/_includes/navigation/sidebar-nav.html +4 -0
  57. data/_includes/navigation/sidebar-pagetree.html +150 -0
  58. data/_includes/navigation/sidebar-right.html +3 -2
  59. data/_includes/navigation/unified-drawer.html +9 -3
  60. data/_includes/obsidian/full-graph.html +165 -136
  61. data/_layouts/404.html +260 -0
  62. data/_layouts/README.md +2 -0
  63. data/_layouts/article.html +7 -1
  64. data/_layouts/book-abc.html +106 -0
  65. data/_layouts/book-story.html +91 -0
  66. data/_layouts/book.html +113 -0
  67. data/_layouts/collection.html +17 -5
  68. data/_layouts/default.html +5 -3
  69. data/_layouts/home.html +5 -2
  70. data/_layouts/landing.html +9 -0
  71. data/_layouts/news.html +159 -33
  72. data/_layouts/root.html +29 -9
  73. data/_layouts/section.html +61 -17
  74. data/_sass/components/_book.scss +423 -0
  75. data/_sass/core/_navbar.scss +13 -0
  76. data/_sass/core/_obsidian.scss +295 -7
  77. data/_sass/layouts/_navbar-extras.scss +6 -1
  78. data/_sass/theme/_backgrounds.scss +21 -8
  79. data/assets/css/main.scss +1 -0
  80. data/assets/js/auto-hide-nav.js +5 -1
  81. data/assets/js/halfmoon.js +26 -0
  82. data/assets/js/obsidian-graph.js +707 -265
  83. data/assets/js/obsidian-local-graph.js +161 -54
  84. data/assets/js/search-modal.js +4 -1
  85. data/scripts/README.md +20 -26
  86. data/scripts/bin/audit-consumer +1 -1
  87. data/scripts/bin/manifest +31 -5
  88. data/scripts/bin/sync-plugins +0 -1
  89. data/scripts/bin/validate +5 -1
  90. data/scripts/dev/rasterize-svg.js +65 -0
  91. data/scripts/features/generate-preview-images +49 -1390
  92. data/scripts/features/install-preview-generator +55 -33
  93. data/scripts/install/README.md +55 -25
  94. data/scripts/install/ai/client.sh +302 -93
  95. data/scripts/install/ai/prompts/spec.schema.json +1 -1
  96. data/scripts/install/ai/prompts/wizard.system.md +8 -17
  97. data/scripts/install/ai/wizard.sh +10 -5
  98. data/scripts/install/apply.sh +7 -3
  99. data/scripts/install/cli.sh +54 -4
  100. data/scripts/install/config.sh +167 -0
  101. data/scripts/install/doctor.sh +38 -0
  102. data/scripts/install/plan.sh +10 -0
  103. data/scripts/install/spec.sh +15 -7
  104. data/scripts/install/template.sh +4 -0
  105. data/scripts/lib/README.md +1 -5
  106. data/scripts/lib/install/deploy/README.md +3 -9
  107. data/scripts/lib/preview_generator.py +2261 -1341
  108. data/scripts/propagate.rb +277 -0
  109. data/scripts/translate.rb +1204 -0
  110. metadata +34 -3
  111. data/_plugins/preview_image_generator.rb +0 -351
@@ -0,0 +1,1204 @@
1
+ #!/usr/bin/env ruby
2
+ # Feature: ZER0-078
3
+ # frozen_string_literal: true
4
+
5
+ # ===================================================================
6
+ # translate.rb — AI translation pipeline for multilingual content
7
+ # ===================================================================
8
+ #
9
+ # Purpose: Generate alternate-language versions of the site's English
10
+ # content WITHOUT storing hand-written translations in source.
11
+ # English (pages/**, _data/ui-text.yml `en`) is the only
12
+ # human-maintained language; every other language is a build
13
+ # artifact produced by this utility and committed by the
14
+ # `translate.yml` workflow (never edited by hand).
15
+ #
16
+ # What it produces:
17
+ # fr/<area>/<file>.md Translated page files (plain Jekyll pages
18
+ # with explicit permalink /fr<en-url>, so the
19
+ # GitHub Pages safe-mode build needs no plugin)
20
+ # _data/i18n/<lang>.yml Translated UI strings (from ui-text.yml en)
21
+ # _data/i18n/manifest.yml Source-of-truth map: en URL -> per-language
22
+ # output URL + content SHA (drives the
23
+ # language toggle, hreflang tags, and
24
+ # incremental change detection)
25
+ #
26
+ # Modes:
27
+ # (default) Incremental — translate only new/changed sources
28
+ # --full Retranslate everything
29
+ # --check Report stale/missing translations; exit 1 if any (no API)
30
+ # --dry-run Plan only; no API calls, no writes
31
+ #
32
+ # Providers:
33
+ # claude (default) Anthropic Messages API. Credential precedence mirrors
34
+ # the chat proxy (templates/deploy/chat-proxy/worker.js):
35
+ # CLAUDE_CODE_OAUTH_TOKEN Bearer + oauth beta header
36
+ # ANTHROPIC_AUTH_TOKEN Bearer + oauth beta header
37
+ # ANTHROPIC_API_KEY x-api-key
38
+ # OAuth tokens require the first system block to carry the
39
+ # Claude Code identity (same rule as the chat proxy).
40
+ # For local runs, credentials are auto-loaded from the
41
+ # repo-root .env (gitignored; `claude setup-token` output
42
+ # goes there — same file the chat dev proxy reads).
43
+ # Real environment variables always win over .env.
44
+ # stub Deterministic offline pseudo-translation (tests/demo):
45
+ # appends " [<lang>]" to every segment. No network.
46
+ #
47
+ # Safety model (how markdown survives translation):
48
+ # - Fenced code blocks, {% highlight %}/{% raw %} regions and the YAML
49
+ # front matter are never sent to the model.
50
+ # - Inline code, Liquid tags/outputs, wiki-links, HTML tags and link
51
+ # destinations are masked as ⟦N⟧ placeholders before the request and
52
+ # restored after; a response that loses or invents placeholders, or
53
+ # changes the segment set/line shape, is rejected and retried once.
54
+ # - Translation is per-line ("one paragraph per line" house rule), sent
55
+ # as a JSON segment map — the file's structure is reassembled from the
56
+ # source, so code, blank lines and ordering are preserved by
57
+ # construction and the markdown-oneline CI check stays green.
58
+ #
59
+ # Usage:
60
+ # ruby scripts/translate.rb # incremental, configured langs
61
+ # ruby scripts/translate.rb --full --langs fr
62
+ # ruby scripts/translate.rb --dry-run --verbose
63
+ # ruby scripts/translate.rb --provider stub --root /tmp/sandbox # tests
64
+ #
65
+ # Configuration: `translation:` block in _config.yml (see there for keys).
66
+ # ===================================================================
67
+
68
+ require "date"
69
+ require "digest"
70
+ require "fileutils"
71
+ require "json"
72
+ require "net/http"
73
+ require "optparse"
74
+ require "time"
75
+ require "uri"
76
+ require "yaml"
77
+
78
+ module Zer0Translate
79
+ VERSION = "1.0.0"
80
+ PROMPT_VERSION = 1
81
+ MANIFEST_REL = File.join("_data", "i18n", "manifest.yml")
82
+ UI_TEXT_REL = File.join("_data", "ui-text.yml")
83
+ PLACEHOLDER_RE = /⟦\d+⟧/ # ⟦N⟧
84
+
85
+ FRONT_MATTER_FIELDS = %w[title sub-title subtitle description excerpt tagline].freeze
86
+ # Front-matter keys that must NOT be copied onto a generated translation
87
+ # (they would collide with the English page: duplicate redirects, wrong
88
+ # permalink, wiki aliases, stale translation metadata).
89
+ FRONT_MATTER_DROP = %w[
90
+ permalink redirect_from redirect_to aliases lang
91
+ translation_of translation_source_url machine_translated translated_from_sha
92
+ ].freeze
93
+ # Layouts that only work for a document INSIDE a Jekyll collection. The
94
+ # translated tree is deliberately flat — fr/** are plain pages, not collection
95
+ # documents — so Jekyll never sets `page.collection` there and these layouts
96
+ # have nothing to enumerate. `collection` is worse than useless: it sorts
97
+ # site[page.collection], which is nil for a plain page, and a nil sort aborts
98
+ # the whole Jekyll build. Drop the key so the `path: <lang>` front-matter
99
+ # default in _config.yml (layout: default) applies instead.
100
+ COLLECTION_ONLY_LAYOUTS = %w[collection].freeze
101
+
102
+ DEFAULT_CONFIG = {
103
+ "enabled" => false,
104
+ "source_lang" => "en",
105
+ "languages" => [],
106
+ "provider" => "claude",
107
+ "model" => "claude-opus-4-8",
108
+ "max_tokens" => 8192,
109
+ "max_chunk_chars" => 4000,
110
+ "max_chunk_segments" => 60,
111
+ "ui_text" => true,
112
+ "sources" => [
113
+ { "path" => "pages/_posts", "output" => "posts" },
114
+ { "path" => "pages/_docs", "output" => "docs" },
115
+ { "path" => "pages/_about", "output" => "about" },
116
+ { "path" => "pages/_quickstart", "output" => "quickstart" },
117
+ { "path" => "pages/_notes", "output" => "notes" },
118
+ ],
119
+ "exclude" => ["**/README.md", "**/_templates/**"],
120
+ }.freeze
121
+
122
+ # ----------------------------------------------------------------
123
+ # Small logging helpers (kept dependency-free; mirror scripts/lib tone)
124
+ # ----------------------------------------------------------------
125
+ module Log
126
+ class << self
127
+ attr_accessor :verbose
128
+
129
+ def info(msg) = puts(msg)
130
+ def debug(msg) = (puts(" [debug] #{msg}") if verbose)
131
+ def warn(msg) = Kernel.warn(" [warn] #{msg}")
132
+ def error(msg) = Kernel.warn("[error] #{msg}")
133
+ end
134
+ end
135
+
136
+ Segment = Struct.new(:key, :text, :placeholders, keyword_init: true)
137
+
138
+ # ----------------------------------------------------------------
139
+ # Masking: protect non-translatable spans inside a prose line
140
+ # ----------------------------------------------------------------
141
+ class Masker
142
+ # Order matters: coarser spans first so finer patterns never split them.
143
+ INLINE_PATTERNS = [
144
+ /\{%.*?%\}/m, # Liquid tags {% ... %}
145
+ /\{\{.*?\}\}/m, # Liquid output {{ ... }}
146
+ /!?\[\[[^\]]+\]\]/, # Obsidian wiki-links / embeds
147
+ /`[^`]*`/, # inline code spans
148
+ /\]\([^()\s]+\)/, # markdown link destinations "](url)"
149
+ /<https?:[^>\s]+>/, # autolinks
150
+ /<\/?[A-Za-z][^>]*>/, # inline HTML tags
151
+ ].freeze
152
+
153
+ def initialize
154
+ @map = {}
155
+ @counter = 0
156
+ end
157
+
158
+ attr_reader :map
159
+
160
+ def mask_line(line)
161
+ masked = line.dup
162
+ INLINE_PATTERNS.each do |pattern|
163
+ masked = masked.gsub(pattern) do |match|
164
+ @counter += 1
165
+ token = "⟦#{@counter}⟧"
166
+ @map[token] = match
167
+ token
168
+ end
169
+ end
170
+ masked
171
+ end
172
+
173
+ def unmask(text)
174
+ text.gsub(PLACEHOLDER_RE) { |token| @map.fetch(token, token) }
175
+ end
176
+ end
177
+
178
+ # ----------------------------------------------------------------
179
+ # Splits a markdown body into translatable segments + verbatim lines
180
+ # ----------------------------------------------------------------
181
+ class Segmenter
182
+ FENCE_RE = /\A(\s*)(`{3,}|~{3,})/
183
+ # A line that is a SINGLE Liquid construct only (e.g. `{% include x %}`,
184
+ # `{{ page.title }}`). The interior excludes braces so a line with prose
185
+ # BETWEEN two constructs (`{{ a }} text {{ b }}`) is NOT treated as pure
186
+ # Liquid — that text must still be translated.
187
+ PURE_LIQUID_RE = /\A\s*\{[%{][^{}]*[%}]\}\s*\z/
188
+ PURE_HTML_RE = %r{\A\s*</?[A-Za-z][^>]*/?>\s*\z}
189
+ HR_RE = /\A\s*(?:[-*_]\s*){3,}\z/
190
+ TABLE_RULE_RE = /\A\s*\|?[\s:|-]+\|?\s*\z/
191
+
192
+ attr_reader :lines, :segments, :masker
193
+
194
+ def initialize(body)
195
+ @lines = body.split("\n", -1)
196
+ @masker = Masker.new
197
+ @segments = []
198
+ scan
199
+ end
200
+
201
+ # Rebuild the body with translated segments swapped in.
202
+ def reassemble(translations)
203
+ out = @lines.dup
204
+ @segments.each do |seg|
205
+ translated = translations.fetch(seg.key)
206
+ out[seg.key.delete_prefix("s").to_i] = @masker.unmask(translated)
207
+ end
208
+ out.join("\n")
209
+ end
210
+
211
+ private
212
+
213
+ def scan
214
+ in_fence = false
215
+ fence_marker = nil
216
+ in_liquid_block = false
217
+
218
+ @lines.each_with_index do |line, idx|
219
+ if in_fence
220
+ in_fence = false if line.lstrip.start_with?(fence_marker)
221
+ next
222
+ end
223
+ if (m = line.match(FENCE_RE))
224
+ in_fence = true
225
+ fence_marker = m[2][0] * m[2].length
226
+ next
227
+ end
228
+ if in_liquid_block
229
+ in_liquid_block = false if line =~ /\{%-?\s*(endhighlight|endraw)\s*-?%\}/
230
+ next
231
+ end
232
+ if line =~ /\{%-?\s*(highlight|raw)\b/ && line !~ /\{%-?\s*end(highlight|raw)/
233
+ in_liquid_block = true
234
+ next
235
+ end
236
+
237
+ next if line.strip.empty?
238
+ next if line =~ PURE_LIQUID_RE || line =~ PURE_HTML_RE
239
+ next if line =~ HR_RE
240
+ next if line =~ TABLE_RULE_RE && line.include?("|")
241
+
242
+ masked = @masker.mask_line(line)
243
+ # Nothing human-readable left after masking → keep the line verbatim.
244
+ next unless masked =~ /\p{L}/
245
+
246
+ @segments << Segment.new(
247
+ key: "s#{idx}",
248
+ text: masked,
249
+ placeholders: masked.scan(PLACEHOLDER_RE).sort,
250
+ )
251
+ end
252
+ end
253
+ end
254
+
255
+ # ----------------------------------------------------------------
256
+ # Jekyll URL resolution for this repo's permalink patterns
257
+ # ----------------------------------------------------------------
258
+ class UrlBuilder
259
+ def initialize(site_config)
260
+ @config = site_config
261
+ end
262
+
263
+ # Returns the pretty URL ("/posts/2026/01/01/foo/") for a source file,
264
+ # or nil when the permalink template contains a placeholder we cannot
265
+ # resolve (the caller skips the file with a warning).
266
+ #
267
+ # Resolution matches the OBSERVED Jekyll 3.10 behavior for this repo's
268
+ # collection documents (verified against a real build): an explicit
269
+ # front-matter `permalink` wins, otherwise the collection's permalink
270
+ # template applies. Front-matter *defaults* permalinks do not affect
271
+ # collection documents on this Jekyll version.
272
+ def url_for(rel_path, front_matter, collection)
273
+ # An explicit front-matter permalink is served by Jekyll VERBATIM (it
274
+ # overrides the global `permalink: pretty`). Return it unchanged so the
275
+ # manifest key matches the page's real `page.url` — prettifying a
276
+ # non-trailing-slash permalink (`/faq`, `/x.html`) would key the
277
+ # manifest at `/faq/` and the toggle/hreflang lookup would miss.
278
+ if (explicit = front_matter["permalink"]) && !explicit.include?(":")
279
+ return explicit
280
+ end
281
+
282
+ template = front_matter["permalink"] || collection_permalink(collection)
283
+ return nil unless template
284
+ return prettify(template) unless template.include?(":")
285
+
286
+ fill_template(template, rel_path, front_matter, collection)
287
+ end
288
+
289
+ private
290
+
291
+ def collections_dir = @config["collections_dir"] || ""
292
+
293
+ def collection_permalink(collection)
294
+ cols = @config["collections"]
295
+ return nil unless cols.is_a?(Hash)
296
+
297
+ entry = cols[collection]
298
+ entry.is_a?(Hash) ? entry["permalink"] : nil
299
+ end
300
+
301
+ def fill_template(template, rel_path, front_matter, collection)
302
+ basename = File.basename(rel_path).sub(/\.[^.]+\z/, "")
303
+ date = resolve_date(front_matter, basename)
304
+ slug_base = basename
305
+ if (m = basename.match(/\A(\d{4})-(\d{2})-(\d{2})-(.+)\z/))
306
+ slug_base = m[4]
307
+ end
308
+ slug = front_matter["slug"] || slugify(slug_base)
309
+
310
+ categories = Array(front_matter["categories"] || front_matter["category"])
311
+ .flatten.compact.map { |c| slugify(c.to_s) }
312
+
313
+ in_collection = collection_relative(rel_path, collection)
314
+ subdir = File.dirname(in_collection)
315
+ subdir = "" if subdir == "."
316
+
317
+ url = template.dup
318
+ url = url.gsub(":collection", collection.to_s)
319
+ url = url.gsub(":categories", categories.join("/"))
320
+ url = url.gsub(":year", date ? format("%04d", date.year) : ":year")
321
+ url = url.gsub(":month", date ? format("%02d", date.month) : ":month")
322
+ url = url.gsub(":day", date ? format("%02d", date.day) : ":day")
323
+ url = url.gsub(":slug", slug)
324
+ url = url.gsub(":name", slugify(slug_base))
325
+ url = url.gsub(":title", slug)
326
+ url = url.gsub(":path", subdir)
327
+ url = url.gsub(":output_ext", "")
328
+
329
+ return nil if url.include?(":") # unresolved placeholder
330
+
331
+ prettify(url)
332
+ end
333
+
334
+ def collection_relative(rel_path, collection)
335
+ prefix = File.join(*[collections_dir, "_#{collection}"].reject(&:empty?))
336
+ rel_path.delete_prefix("#{prefix}/")
337
+ end
338
+
339
+ def resolve_date(front_matter, basename)
340
+ raw = front_matter["date"]
341
+ case raw
342
+ when Date, Time then return raw
343
+ when String
344
+ begin
345
+ return Time.parse(raw)
346
+ rescue ArgumentError
347
+ nil
348
+ end
349
+ end
350
+ m = basename.match(/\A(\d{4})-(\d{2})-(\d{2})-/)
351
+ m ? Date.new(m[1].to_i, m[2].to_i, m[3].to_i) : nil
352
+ end
353
+
354
+ def slugify(str)
355
+ str.to_s.downcase.gsub(/[^a-z0-9]+/, "-").gsub(/\A-+|-+\z/, "")
356
+ end
357
+
358
+ # `permalink: pretty` — collapse duplicate slashes, ensure trailing slash.
359
+ def prettify(url)
360
+ url = "/#{url}".gsub(%r{/+}, "/")
361
+ url.end_with?("/") ? url : "#{url}/"
362
+ end
363
+ end
364
+
365
+ # ----------------------------------------------------------------
366
+ # Providers
367
+ # ----------------------------------------------------------------
368
+ class StubProvider
369
+ def name = "stub"
370
+
371
+ # Deterministic, structure-preserving pseudo-translation: appends a
372
+ # visible marker to every segment. Placeholders survive by construction.
373
+ def translate(segments, target_lang, _context)
374
+ segments.to_h { |key, text| [key, "#{text} [#{target_lang}]"] }
375
+ end
376
+ end
377
+
378
+ # Test-only variant of StubProvider that soft-wraps its output, reproducing
379
+ # the one thing a real provider does that no prompt can reliably prevent.
380
+ # Exists so the prose-normalisation pass has something to actually fix —
381
+ # asserting "output is unwrapped" against the plain stub would pass whether
382
+ # or not normalisation ran.
383
+ class WrappingStubProvider
384
+ WRAP_AT = 40
385
+
386
+ def name = "stub-wrap"
387
+
388
+ def translate(segments, target_lang, _context)
389
+ segments.to_h { |key, text| [key, wrap("#{text} [#{target_lang}]")] }
390
+ end
391
+
392
+ private
393
+
394
+ # Greedy wrap on spaces. Never splits a line that has no space past the
395
+ # column (a long URL or placeholder token stays intact).
396
+ def wrap(text)
397
+ out = []
398
+ line = +""
399
+ text.split(" ").each do |word|
400
+ if line.empty?
401
+ line << word
402
+ elsif line.length + 1 + word.length > WRAP_AT
403
+ out << line
404
+ line = +word
405
+ else
406
+ line << " " << word
407
+ end
408
+ end
409
+ out << line unless line.empty?
410
+ out.join("\n")
411
+ end
412
+ end
413
+
414
+ class ClaudeProvider
415
+ ENDPOINT = URI("https://api.anthropic.com/v1/messages")
416
+ API_VERSION = "2023-06-01"
417
+ OAUTH_BETA = "oauth-2025-04-20"
418
+ # Claude Code OAuth tokens are gated to Claude Code: the FIRST system
419
+ # block must carry this identity or the API rejects the request (same
420
+ # rule the chat proxy implements — see chat-proxy/worker.js).
421
+ CLAUDE_CODE_IDENTITY = "You are Claude Code, Anthropic's official CLI for Claude."
422
+ MAX_ATTEMPTS = 4
423
+
424
+ def initialize(model:, max_tokens:)
425
+ @model = model
426
+ @max_tokens = max_tokens
427
+ @auth = resolve_auth
428
+ raise "No Anthropic credential found. Set CLAUDE_CODE_OAUTH_TOKEN " \
429
+ "(from `claude setup-token`), ANTHROPIC_AUTH_TOKEN, or ANTHROPIC_API_KEY." unless @auth
430
+ end
431
+
432
+ def name = "claude (#{@model})"
433
+
434
+ def translate(segments, target_lang, context)
435
+ payload = build_payload(segments, target_lang, context)
436
+ body = request_with_retries(payload)
437
+ text = (body["content"] || []).select { |b| b["type"] == "text" }
438
+ .map { |b| b["text"] }.join("\n")
439
+ parsed = extract_json(text)
440
+ raise ProviderError, "response was not a JSON object" unless parsed.is_a?(Hash)
441
+
442
+ parsed
443
+ end
444
+
445
+ class ProviderError < StandardError; end
446
+
447
+ private
448
+
449
+ def resolve_auth
450
+ if (token = ENV["CLAUDE_CODE_OAUTH_TOKEN"] || ENV["ANTHROPIC_AUTH_TOKEN"])
451
+ return { mode: :oauth, token: token } unless token.empty?
452
+ end
453
+ if (key = ENV["ANTHROPIC_API_KEY"])
454
+ return { mode: :api_key, token: key } unless key.empty?
455
+ end
456
+ nil
457
+ end
458
+
459
+ def system_blocks(target_lang)
460
+ instructions = <<~PROMPT
461
+ You are a professional technical translator for a software documentation website (Jekyll theme, Docker tooling, developer blog).
462
+ Translate the values of the JSON object the user provides from English into the language with IETF code "#{target_lang}".
463
+
464
+ Rules:
465
+ - Respond with ONLY a JSON object — no prose, no markdown fences — containing exactly the same keys; each value is the translation of the source value.
466
+ - Values are Markdown fragments. Preserve all Markdown/HTML syntax characters (#, *, -, >, |, [ ], ( ), :, emphasis markers) in their structural roles.
467
+ - Tokens like ⟦12⟧ are protected placeholders. Reproduce every placeholder EXACTLY as it appears, positioned where its content belongs in the translated sentence. Never translate, alter, merge, or drop a placeholder.
468
+ - Do not translate: code, shell commands, file paths, URLs, configuration keys, or product/proper names (Jekyll, Docker, Bootstrap, GitHub, Ruby, Obsidian, zer0-mistakes, ...).
469
+ - Every value must stay on a single line (no newline characters) because the site enforces one paragraph per line.
470
+ - Keys prefixed "fm:" are page metadata (title, description); keys prefixed "ui:" are short UI labels — translate them concisely. Keys prefixed "s" are body lines.
471
+ - Use natural, idiomatic phrasing in the target language with a professional technical register.
472
+ PROMPT
473
+
474
+ blocks = []
475
+ blocks << { type: "text", text: CLAUDE_CODE_IDENTITY } if @auth[:mode] == :oauth
476
+ blocks << { type: "text", text: instructions }
477
+ blocks
478
+ end
479
+
480
+ def build_payload(segments, target_lang, context)
481
+ user_text = +"Context: #{context}\n\nTranslate these segments:\n"
482
+ user_text << JSON.pretty_generate(segments)
483
+ {
484
+ model: @model,
485
+ max_tokens: @max_tokens,
486
+ system: system_blocks(target_lang),
487
+ messages: [{ role: "user", content: user_text }],
488
+ }
489
+ end
490
+
491
+ def headers
492
+ base = { "content-type" => "application/json", "anthropic-version" => API_VERSION }
493
+ if @auth[:mode] == :oauth
494
+ base["authorization"] = "Bearer #{@auth[:token]}"
495
+ base["anthropic-beta"] = OAUTH_BETA
496
+ else
497
+ base["x-api-key"] = @auth[:token]
498
+ end
499
+ base
500
+ end
501
+
502
+ def request_with_retries(payload)
503
+ attempt = 0
504
+ begin
505
+ attempt += 1
506
+ response = post(payload)
507
+ code = response.code.to_i
508
+ if [429, 500, 502, 503, 529].include?(code) && attempt < MAX_ATTEMPTS
509
+ delay = (response["retry-after"]&.to_i&.positive? ? response["retry-after"].to_i : 2**attempt)
510
+ Log.warn "API #{code}; retrying in #{delay}s (attempt #{attempt}/#{MAX_ATTEMPTS})"
511
+ sleep delay
512
+ raise RetryableError
513
+ end
514
+ body = JSON.parse(response.body)
515
+ unless code == 200
516
+ message = body.dig("error", "message") || response.body[0, 300]
517
+ raise ProviderError, "Anthropic API #{code}: #{message}"
518
+ end
519
+ body
520
+ rescue RetryableError
521
+ retry
522
+ rescue Errno::ECONNRESET, Net::OpenTimeout, Net::ReadTimeout, SocketError => e
523
+ raise ProviderError, "network error: #{e.message}" unless attempt < MAX_ATTEMPTS
524
+
525
+ sleep 2**attempt
526
+ retry
527
+ end
528
+ end
529
+
530
+ class RetryableError < StandardError; end
531
+
532
+ def post(payload)
533
+ http = Net::HTTP.new(ENDPOINT.host, ENDPOINT.port)
534
+ http.use_ssl = true
535
+ http.open_timeout = 30
536
+ http.read_timeout = 300
537
+ request = Net::HTTP::Post.new(ENDPOINT.request_uri, headers)
538
+ request.body = JSON.generate(payload)
539
+ http.request(request)
540
+ end
541
+
542
+ # Tolerate fences / stray prose around the JSON object: slicing from the
543
+ # first "{" to the last "}" covers fenced responses too, without any
544
+ # backtracking-prone regex over model-controlled text.
545
+ def extract_json(text)
546
+ clean = text.to_s
547
+ first = clean.index("{")
548
+ last = clean.rindex("}")
549
+ return nil if first.nil? || last.nil? || last < first
550
+
551
+ JSON.parse(clean[first..last])
552
+ rescue JSON::ParserError
553
+ nil
554
+ end
555
+ end
556
+
557
+ # ----------------------------------------------------------------
558
+ # Chunked, validated translation of a segment map
559
+ # ----------------------------------------------------------------
560
+ class Translator
561
+ def initialize(provider, chunk_chars:, chunk_segments:)
562
+ @provider = provider
563
+ @chunk_chars = chunk_chars
564
+ @chunk_segments = chunk_segments
565
+ end
566
+
567
+ # segments: { key => masked text }; returns { key => translated text }.
568
+ # Raises TranslationError when a chunk cannot be validated after a retry.
569
+ def translate_map(segments, target_lang, context)
570
+ result = {}
571
+ each_chunk(segments) do |chunk|
572
+ result.merge!(translate_chunk(chunk, target_lang, context))
573
+ end
574
+ result
575
+ end
576
+
577
+ class TranslationError < StandardError; end
578
+
579
+ private
580
+
581
+ def each_chunk(segments)
582
+ chunk = {}
583
+ size = 0
584
+ segments.each do |key, text|
585
+ if !chunk.empty? && (size + text.length > @chunk_chars || chunk.size >= @chunk_segments)
586
+ yield chunk
587
+ chunk = {}
588
+ size = 0
589
+ end
590
+ chunk[key] = text
591
+ size += text.length
592
+ end
593
+ yield chunk unless chunk.empty?
594
+ end
595
+
596
+ def translate_chunk(chunk, target_lang, context, retried: false)
597
+ output = @provider.translate(chunk, target_lang, context)
598
+ validate!(chunk, output)
599
+ output
600
+ rescue TranslationError, ClaudeProvider::ProviderError => e
601
+ raise TranslationError, e.message if retried
602
+
603
+ Log.warn "chunk rejected (#{e.message}); retrying once"
604
+ translate_chunk(chunk, target_lang, "#{context} | RETRY — previous attempt failed validation: #{e.message}. Follow the placeholder and single-line rules exactly.", retried: true)
605
+ end
606
+
607
+ def validate!(input, output)
608
+ raise TranslationError, "provider returned #{output.class}" unless output.is_a?(Hash)
609
+
610
+ missing = input.keys - output.keys
611
+ raise TranslationError, "missing keys: #{missing.first(5).join(', ')}" unless missing.empty?
612
+
613
+ input.each do |key, source|
614
+ value = output[key]
615
+ raise TranslationError, "#{key}: non-string value" unless value.is_a?(String)
616
+ raise TranslationError, "#{key}: empty translation" if value.strip.empty? && !source.strip.empty?
617
+ raise TranslationError, "#{key}: line break introduced" if value.include?("\n")
618
+
619
+ expected = source.scan(PLACEHOLDER_RE).sort
620
+ actual = value.scan(PLACEHOLDER_RE).sort
621
+ raise TranslationError, "#{key}: placeholder mismatch" unless expected == actual
622
+
623
+ assert_link_brackets_balanced!(key, source, value)
624
+ end
625
+ end
626
+
627
+ # A masked markdown link destination — `](url)` — carries the `]` that
628
+ # closes a preceding `[text]`. The multiset check above still passes if the
629
+ # model relocates that placeholder away from its bracket, but unmask would
630
+ # then sever the link (literal brackets on the page). Guard it structurally
631
+ # without needing the masker's map: any placeholder that closes a link in
632
+ # the SOURCE must, in the OUTPUT, sit where an unclosed `[` still awaits it
633
+ # (more literal `[` than `]` in the text before it — the placeholder's own
634
+ # `]` is still masked, so it isn't counted). A false alarm is impossible:
635
+ # a correctly-placed link always has its opening `[` earlier in the line.
636
+ def assert_link_brackets_balanced!(key, source, value)
637
+ link_tokens = link_closing_placeholders(source)
638
+ return if link_tokens.empty?
639
+
640
+ scan_placeholder_positions(value).each do |token, before|
641
+ next unless link_tokens.include?(token)
642
+ next if before.count("[") > before.count("]")
643
+
644
+ raise TranslationError, "#{key}: link placeholder #{token} detached from its [text]"
645
+ end
646
+ end
647
+
648
+ # Placeholders in `source` that sit immediately after link text with no
649
+ # intervening bracket — i.e. the token that provides a link's closing `]`.
650
+ # Identified structurally: in the source the token is preceded by an
651
+ # unclosed `[`.
652
+ def link_closing_placeholders(source)
653
+ scan_placeholder_positions(source).filter_map do |token, before|
654
+ token if before.count("[") > before.count("]")
655
+ end
656
+ end
657
+
658
+ def scan_placeholder_positions(text)
659
+ positions = []
660
+ text.scan(PLACEHOLDER_RE) do
661
+ positions << [Regexp.last_match(0), text[0...Regexp.last_match.begin(0)]]
662
+ end
663
+ positions
664
+ end
665
+ end
666
+
667
+ # ----------------------------------------------------------------
668
+ # Manifest — generated map read by Liquid (toggle/hreflang) and by
669
+ # this script for incremental change detection.
670
+ # ----------------------------------------------------------------
671
+ class Manifest
672
+ attr_reader :data
673
+
674
+ def initialize(root)
675
+ @path = File.join(root, MANIFEST_REL)
676
+ @data = load_data
677
+ end
678
+
679
+ def load_data
680
+ if File.file?(@path)
681
+ raw = File.read(@path, encoding: "bom|utf-8")
682
+ loaded = YAML.safe_load(raw, permitted_classes: [Date, Time], aliases: true)
683
+ return loaded if loaded.is_a?(Hash) && loaded["pages"].is_a?(Hash)
684
+ end
685
+ { "version" => 1, "pages" => {}, "ui_text" => {} }
686
+ end
687
+
688
+ def pages = @data["pages"]
689
+ def ui_text = @data["ui_text"] ||= {}
690
+
691
+ def entry_for(url)
692
+ pages[url] ||= {}
693
+ end
694
+
695
+ def save(source_lang, languages)
696
+ @data["version"] = 1
697
+ @data["source_lang"] = source_lang
698
+ @data["languages"] = languages
699
+ @data["updated_at"] = Time.now.utc.iso8601
700
+ @data["pages"] = pages.sort.to_h
701
+ FileUtils.mkdir_p(File.dirname(@path))
702
+ header = <<~HEADER
703
+ # ================================================================
704
+ # GENERATED FILE — do not edit by hand.
705
+ # Maintained by scripts/translate.rb (see .github/workflows/translate.yml).
706
+ # Maps each English page URL to its generated translations; read by
707
+ # _includes/components/language-toggle.html and _includes/core/hreflang.html.
708
+ # ================================================================
709
+ HEADER
710
+ File.write(@path, header + @data.to_yaml.sub(/\A---\n/, ""))
711
+ end
712
+ end
713
+
714
+ # ----------------------------------------------------------------
715
+ # A single translatable source page
716
+ # ----------------------------------------------------------------
717
+ class SourceFile
718
+ FRONT_MATTER_RE = /\A---\s*\n(.*?)\n---\s*\n?/m
719
+
720
+ attr_reader :rel_path, :collection, :output_area, :error
721
+
722
+ def initialize(root, rel_path, collection:, output_area:)
723
+ @root = root
724
+ @rel_path = rel_path
725
+ @collection = collection
726
+ @output_area = output_area
727
+ @raw = File.read(File.join(root, rel_path), encoding: "bom|utf-8")
728
+ if (m = @raw.match(FRONT_MATTER_RE))
729
+ @front_matter = YAML.safe_load(m[1], permitted_classes: [Date, Time], aliases: true) || {}
730
+ @body = m.post_match
731
+ else
732
+ @front_matter = nil
733
+ @body = @raw
734
+ end
735
+ rescue Psych::Exception => e
736
+ @front_matter = nil
737
+ @error = "front matter parse error: #{e.message}"
738
+ end
739
+
740
+ def front_matter = @front_matter || {}
741
+ def body = @body.to_s
742
+
743
+ def translatable?
744
+ return false unless @front_matter.is_a?(Hash)
745
+ return false if front_matter["published"] == false
746
+ return false if front_matter["translate"] == false
747
+ return false if front_matter["lang"] && front_matter["lang"].to_s[0, 2] != "en"
748
+
749
+ true
750
+ end
751
+
752
+ def sha
753
+ @sha ||= Digest::SHA256.hexdigest(@raw)
754
+ end
755
+
756
+ # Path of the generated translation, mirroring subfolders within the
757
+ # collection: pages/_posts/a.md -> <lang>/posts/a.md
758
+ def output_rel_path(lang, collections_dir)
759
+ prefix = File.join(*[collections_dir, "_#{collection}"].reject(&:empty?))
760
+ inner = rel_path.delete_prefix("#{prefix}/")
761
+ File.join(lang, output_area, inner)
762
+ end
763
+ end
764
+
765
+ # ----------------------------------------------------------------
766
+ # Local credential wiring: load KEY=VALUE pairs from the repo-root
767
+ # .env (gitignored) so `claude setup-token` credentials work for
768
+ # local runs — the same file the chat dev proxy reads. Real
769
+ # environment variables always win; values are never logged.
770
+ # ----------------------------------------------------------------
771
+ module DotEnv
772
+ def self.load(root)
773
+ path = File.join(root, ".env")
774
+ return unless File.file?(path)
775
+
776
+ File.foreach(path, encoding: "bom|utf-8") do |line|
777
+ line = line.strip
778
+ next if line.empty? || line.start_with?("#")
779
+
780
+ key, _, value = line.partition("=")
781
+ key = key.sub(/\Aexport\s+/, "").strip
782
+ next unless key.match?(/\A[A-Za-z_][A-Za-z0-9_]*\z/)
783
+ next if ENV.key?(key) # real environment always wins
784
+
785
+ ENV[key] = value.strip.gsub(/\A["']|["']\z/, "")
786
+ end
787
+ end
788
+ end
789
+
790
+ # ----------------------------------------------------------------
791
+ # CLI / orchestration
792
+ # ----------------------------------------------------------------
793
+ class CLI
794
+ def initialize(argv)
795
+ @options = {
796
+ root: Dir.pwd, mode: :incremental, dry_run: false, check: false,
797
+ provider: nil, model: nil, langs: nil, limit: nil, verbose: false
798
+ }
799
+ parse(argv)
800
+ Log.verbose = @options[:verbose]
801
+ @root = File.expand_path(@options[:root])
802
+ DotEnv.load(@root)
803
+ @site_config = load_yaml(File.join(@root, "_config.yml")) || {}
804
+ @config = DEFAULT_CONFIG.merge(@site_config["translation"] || {})
805
+ @config["languages"] = @options[:langs] if @options[:langs]
806
+ @url_builder = UrlBuilder.new(@site_config)
807
+ @manifest = Manifest.new(@root)
808
+ @stats = Hash.new(0)
809
+ # Absolute paths of every markdown file this run wrote, for the
810
+ # prose-normalisation pass in `run`.
811
+ @written_pages = []
812
+ end
813
+
814
+ def run
815
+ languages = Array(@config["languages"]).map(&:to_s)
816
+ source_lang = @config["source_lang"] || "en"
817
+ if languages.empty?
818
+ Log.info "No target languages configured (translation.languages) — nothing to do."
819
+ return 0
820
+ end
821
+ bad = languages.reject { |l| l.match?(/\A[a-z]{2}(-[A-Za-z]{2})?\z/) }
822
+ raise "invalid language code(s): #{bad.join(', ')}" unless bad.empty?
823
+
824
+ sources = discover_sources
825
+ plan = build_plan(sources, languages)
826
+ report_plan(plan, sources.size, languages)
827
+
828
+ return check_result(plan) if @options[:check]
829
+ if plan.empty? && !prune_needed?(sources, languages)
830
+ Log.info "Everything is up to date."
831
+ return 0
832
+ end
833
+ if @options[:dry_run]
834
+ Log.info "Dry run — no API calls made, no files written."
835
+ return 0
836
+ end
837
+
838
+ translator = build_translator
839
+ execute(plan, translator)
840
+ normalize_prose(@written_pages)
841
+ prune(sources, languages)
842
+ @manifest.save(source_lang, languages)
843
+ summary
844
+ @stats[:failed].positive? ? 1 : 0
845
+ end
846
+
847
+ private
848
+
849
+ # -- planning ---------------------------------------------------
850
+
851
+ Job = Struct.new(:type, :source, :lang, :url, keyword_init: true)
852
+
853
+ def discover_sources
854
+ collections_dir = @site_config["collections_dir"].to_s
855
+ list = []
856
+ Array(@config["sources"]).each do |src|
857
+ dir = src["path"].to_s
858
+ area = src["output"] || File.basename(dir).delete_prefix("_")
859
+ collection = File.basename(dir).delete_prefix("_")
860
+ Dir.glob(File.join(@root, dir, "**", "*.{md,markdown}")).sort.each do |abs|
861
+ rel = abs.delete_prefix("#{@root}/")
862
+ next if excluded?(rel)
863
+
864
+ next unless matches_only_filter?(rel)
865
+
866
+ file = SourceFile.new(@root, rel, collection: collection, output_area: area)
867
+ if file.error
868
+ Log.warn "#{rel}: #{file.error} — skipped"
869
+ next
870
+ end
871
+ next unless file.translatable?
872
+
873
+ list << file
874
+ end
875
+ end
876
+ @collections_dir = collections_dir
877
+ list
878
+ end
879
+
880
+ def excluded?(rel)
881
+ Array(@config["exclude"]).any? do |pattern|
882
+ File.fnmatch?(pattern, rel, File::FNM_PATHNAME | File::FNM_DOTMATCH) ||
883
+ File.fnmatch?(pattern, rel)
884
+ end
885
+ end
886
+
887
+ # --only accepts a glob or plain substring of the source path.
888
+ def matches_only_filter?(rel)
889
+ pattern = @options[:only]
890
+ return true unless pattern
891
+
892
+ File.fnmatch?(pattern, rel) || rel.include?(pattern)
893
+ end
894
+
895
+ def build_plan(sources, languages)
896
+ plan = []
897
+ sources.each do |file|
898
+ url = @url_builder.url_for(file.rel_path, file.front_matter, file.collection)
899
+ unless url
900
+ Log.warn "#{file.rel_path}: could not resolve permalink template — skipped"
901
+ next
902
+ end
903
+ entry = @manifest.pages[url]
904
+ languages.each do |lang|
905
+ state = entry && entry[lang]
906
+ out_path = file.output_rel_path(lang, @collections_dir)
907
+ stale = @options[:mode] == :full ||
908
+ state.nil? ||
909
+ state["sha"] != file.sha ||
910
+ !File.file?(File.join(@root, out_path))
911
+ plan << Job.new(type: :page, source: file, lang: lang, url: url) if stale
912
+ end
913
+ end
914
+ plan.concat(ui_text_jobs(languages))
915
+ plan = plan.first(@options[:limit]) if @options[:limit]
916
+ plan
917
+ end
918
+
919
+ def ui_text_jobs(languages)
920
+ return [] unless @config["ui_text"]
921
+
922
+ strings = ui_text_strings
923
+ return [] if strings.empty?
924
+
925
+ sha = Digest::SHA256.hexdigest(JSON.generate(strings.sort.to_h))
926
+ languages.filter_map do |lang|
927
+ state = @manifest.ui_text[lang]
928
+ out = File.join(@root, "_data", "i18n", "#{lang}.yml")
929
+ next if @options[:mode] != :full && state && state["sha"] == sha && File.file?(out)
930
+
931
+ Job.new(type: :ui_text, source: sha, lang: lang, url: nil)
932
+ end
933
+ end
934
+
935
+ def ui_text_strings
936
+ @ui_text_strings ||= begin
937
+ path = File.join(@root, UI_TEXT_REL)
938
+ data = File.file?(path) ? load_yaml(path) : nil
939
+ en = data && (data[@config["source_lang"]] || data["en"])
940
+ en.is_a?(Hash) ? en.select { |_k, v| v.is_a?(String) && !v.strip.empty? } : {}
941
+ end
942
+ end
943
+
944
+ def report_plan(plan, source_count, languages)
945
+ pages = plan.count { |j| j.type == :page }
946
+ ui = plan.count { |j| j.type == :ui_text }
947
+ Log.info "Translation plan: #{source_count} source page(s) × #{languages.join(', ')} → " \
948
+ "#{pages} page translation(s) + #{ui} UI-string set(s) needed " \
949
+ "(mode: #{@options[:mode]}#{@options[:limit] ? ", limit: #{@options[:limit]}" : ''})"
950
+ plan.first(20).each do |job|
951
+ label = job.type == :page ? job.source.rel_path : "_data/ui-text.yml"
952
+ Log.debug "→ [#{job.lang}] #{label}"
953
+ end
954
+ end
955
+
956
+ def check_result(plan)
957
+ if plan.empty?
958
+ Log.info "--check: translations are up to date."
959
+ 0
960
+ else
961
+ Log.info "--check: #{plan.size} translation job(s) pending. Run the translate workflow."
962
+ 1
963
+ end
964
+ end
965
+
966
+ # -- execution --------------------------------------------------
967
+
968
+ def build_translator
969
+ provider_name = @options[:provider] || @config["provider"] || "claude"
970
+ provider =
971
+ case provider_name
972
+ when "stub" then StubProvider.new
973
+ when "stub-wrap" then WrappingStubProvider.new
974
+ when "claude"
975
+ ClaudeProvider.new(
976
+ model: @options[:model] || ENV["TRANSLATE_MODEL"] || @config["model"],
977
+ max_tokens: @config["max_tokens"].to_i,
978
+ )
979
+ else
980
+ raise "unknown provider: #{provider_name}"
981
+ end
982
+ Log.info "Provider: #{provider.name}"
983
+ Translator.new(provider,
984
+ chunk_chars: @config["max_chunk_chars"].to_i,
985
+ chunk_segments: @config["max_chunk_segments"].to_i)
986
+ end
987
+
988
+ def execute(plan, translator)
989
+ plan.each do |job|
990
+ case job.type
991
+ when :page then translate_page(job, translator)
992
+ when :ui_text then translate_ui_text(job, translator)
993
+ end
994
+ rescue Translator::TranslationError, ClaudeProvider::ProviderError => e
995
+ label = job.type == :page ? job.source.rel_path : "ui-text"
996
+ Log.error "[#{job.lang}] #{label}: #{e.message}"
997
+ @stats[:failed] += 1
998
+ end
999
+ end
1000
+
1001
+ # Generated pages must satisfy the repo's one-paragraph-per-line rule, which
1002
+ # CI enforces via .github/workflows/markdown-oneline.yml. The provider is
1003
+ # free to soft-wrap a translated paragraph across several lines — nothing in
1004
+ # the prompt can guarantee otherwise — so normalise deterministically after
1005
+ # the fact rather than hoping the model complies.
1006
+ #
1007
+ # tools/unwrap-prose.py is the single source of truth for the rule (it is
1008
+ # what CI runs), so shell out to it instead of reimplementing the classifier
1009
+ # here and letting the two drift.
1010
+ #
1011
+ # Best-effort by design: a missing python3 should not fail an otherwise good
1012
+ # translation run. The Translate workflow runs the same tool as a guaranteed
1013
+ # backstop before committing, so the only cost of skipping here is a local
1014
+ # run leaving wrapped prose for that step to fix.
1015
+ def normalize_prose(paths)
1016
+ paths = paths.select { |p| File.file?(p) }
1017
+ return if paths.empty?
1018
+
1019
+ # Resolve the tool relative to THIS script, not to --root. translate.rb
1020
+ # and tools/ ship together; --root points at the content tree being
1021
+ # translated, which is a different thing (and is a throwaway sandbox
1022
+ # under test).
1023
+ tool = File.expand_path("../tools/unwrap-prose.py", __dir__)
1024
+ tool = File.join(@root, "tools", "unwrap-prose.py") unless File.file?(tool)
1025
+ unless File.file?(tool)
1026
+ Log.warn "tools/unwrap-prose.py not found — skipping prose normalisation"
1027
+ return
1028
+ end
1029
+
1030
+ ok = system("python3", tool, "--write", *paths,
1031
+ out: File::NULL, err: File::NULL)
1032
+ if ok
1033
+ Log.info "Normalised prose in #{paths.size} generated page(s)"
1034
+ else
1035
+ Log.warn "prose normalisation skipped (python3 unavailable or tool failed); " \
1036
+ "run: python3 tools/unwrap-prose.py --write"
1037
+ end
1038
+ end
1039
+
1040
+ def translate_page(job, translator)
1041
+ file = job.source
1042
+ segmenter = Segmenter.new(file.body)
1043
+ segments = segmenter.segments.to_h { |s| [s.key, s.text] }
1044
+
1045
+ fm_masker = Masker.new
1046
+ FRONT_MATTER_FIELDS.each do |field|
1047
+ value = file.front_matter[field]
1048
+ next unless value.is_a?(String) && !value.strip.empty?
1049
+
1050
+ # Front-matter fields render as single-line HTML attributes (title,
1051
+ # description, ...). Collapse any embedded newlines from a YAML block
1052
+ # scalar (`description: |` / `>`) BEFORE masking, so the single-line
1053
+ # validator can't reject an otherwise-faithful translation and leave
1054
+ # the whole page permanently untranslated.
1055
+ segments["fm:#{field}"] = fm_masker.mask_line(value.gsub(/\s*\n\s*/, " ").strip)
1056
+ end
1057
+
1058
+ context = "Page \"#{file.front_matter['title']}\" (#{file.rel_path}) from the zer0-mistakes Jekyll theme site."
1059
+ translated = segments.empty? ? {} : translator.translate_map(segments, job.lang, context)
1060
+
1061
+ out_rel = file.output_rel_path(job.lang, @collections_dir)
1062
+ write_page(file, job, translated, segmenter, fm_masker, out_rel)
1063
+
1064
+ entry = @manifest.entry_for(job.url)
1065
+ entry["source"] = file.rel_path
1066
+ entry["sha"] = file.sha
1067
+ entry[job.lang] = {
1068
+ "url" => "/#{job.lang}#{job.url}",
1069
+ "path" => out_rel,
1070
+ "sha" => file.sha,
1071
+ "translated_at" => Time.now.utc.iso8601,
1072
+ "prompt_version" => PROMPT_VERSION,
1073
+ }
1074
+ @stats[:pages] += 1
1075
+ Log.info " ✓ [#{job.lang}] #{file.rel_path} → #{out_rel}"
1076
+ end
1077
+
1078
+ def write_page(file, job, translated, segmenter, fm_masker, out_rel)
1079
+ fm = file.front_matter.reject { |k, _| FRONT_MATTER_DROP.include?(k) }
1080
+ FRONT_MATTER_FIELDS.each do |field|
1081
+ key = "fm:#{field}"
1082
+ fm[field] = fm_masker.unmask(translated[key]) if translated.key?(key)
1083
+ end
1084
+ fm.delete("layout") if COLLECTION_ONLY_LAYOUTS.include?(fm["layout"].to_s)
1085
+ fm["lang"] = job.lang
1086
+ fm["permalink"] = "/#{job.lang}#{job.url}"
1087
+ fm["translation_of"] = file.rel_path
1088
+ fm["translation_source_url"] = job.url
1089
+ fm["machine_translated"] = true
1090
+ fm["translated_from_sha"] = file.sha[0, 12]
1091
+
1092
+ body_out = segmenter.reassemble(
1093
+ segmenter.segments.to_h { |s| [s.key, translated.fetch(s.key, s.text)] },
1094
+ )
1095
+
1096
+ abs = File.join(@root, out_rel)
1097
+ FileUtils.mkdir_p(File.dirname(abs))
1098
+ File.write(abs, "#{fm.to_yaml}---\n\n#{body_out.sub(/\A\n+/, '')}")
1099
+ @written_pages << abs
1100
+ end
1101
+
1102
+ def translate_ui_text(job, translator)
1103
+ strings = ui_text_strings
1104
+ masker = Masker.new
1105
+ segments = strings.transform_values { |v| masker.mask_line(v) }
1106
+ .transform_keys { |k| "ui:#{k}" }
1107
+ context = "Short UI labels for the zer0-mistakes Jekyll theme (navigation, search, footer)."
1108
+ translated = translator.translate_map(segments, job.lang, context)
1109
+
1110
+ out = strings.keys.to_h { |k| [k, masker.unmask(translated.fetch("ui:#{k}"))] }
1111
+ path = File.join(@root, "_data", "i18n", "#{job.lang}.yml")
1112
+ FileUtils.mkdir_p(File.dirname(path))
1113
+ header = <<~HEADER
1114
+ # ================================================================
1115
+ # GENERATED FILE — do not edit by hand.
1116
+ # Machine-translated UI strings (#{job.lang}) produced by
1117
+ # scripts/translate.rb from _data/ui-text.yml (en). Regenerate via
1118
+ # the Translate workflow or: ruby scripts/translate.rb --langs #{job.lang}
1119
+ # ================================================================
1120
+ HEADER
1121
+ File.write(path, header + out.to_yaml.sub(/\A---\n/, ""))
1122
+
1123
+ @manifest.ui_text[job.lang] = {
1124
+ "sha" => job.source,
1125
+ "translated_at" => Time.now.utc.iso8601,
1126
+ "prompt_version" => PROMPT_VERSION,
1127
+ }
1128
+ @stats[:ui] += 1
1129
+ Log.info " ✓ [#{job.lang}] UI strings → _data/i18n/#{job.lang}.yml"
1130
+ end
1131
+
1132
+ # -- pruning ----------------------------------------------------
1133
+
1134
+ def orphaned_urls(sources, _languages)
1135
+ live = sources.filter_map { |f| @url_builder.url_for(f.rel_path, f.front_matter, f.collection) }
1136
+ @manifest.pages.keys - live
1137
+ end
1138
+
1139
+ def prune_needed?(sources, languages)
1140
+ orphaned_urls(sources, languages).any?
1141
+ end
1142
+
1143
+ def prune(sources, languages)
1144
+ orphaned_urls(sources, languages).each do |url|
1145
+ entry = @manifest.pages.delete(url)
1146
+ languages.each do |lang|
1147
+ path = entry.dig(lang, "path")
1148
+ next unless path
1149
+
1150
+ abs = File.join(@root, path)
1151
+ # Only ever delete inside a language output root.
1152
+ if abs.start_with?(File.join(@root, lang, "")) && File.file?(abs)
1153
+ File.delete(abs)
1154
+ Log.info " ✗ pruned #{path} (source removed)"
1155
+ @stats[:pruned] += 1
1156
+ end
1157
+ end
1158
+ end
1159
+ end
1160
+
1161
+ def summary
1162
+ Log.info "Done: #{@stats[:pages]} page(s), #{@stats[:ui]} UI set(s) translated; " \
1163
+ "#{@stats[:pruned]} pruned; #{@stats[:failed]} failed."
1164
+ end
1165
+
1166
+ # -- plumbing ---------------------------------------------------
1167
+
1168
+ def load_yaml(path)
1169
+ YAML.safe_load(File.read(path, encoding: "bom|utf-8"),
1170
+ permitted_classes: [Date, Time], aliases: true)
1171
+ rescue Psych::Exception => e
1172
+ raise "failed to parse #{path}: #{e.message}"
1173
+ end
1174
+
1175
+ def parse(argv)
1176
+ OptionParser.new do |o|
1177
+ o.banner = "Usage: ruby scripts/translate.rb [options]"
1178
+ o.on("--full", "Retranslate everything (ignore manifest state)") { @options[:mode] = :full }
1179
+ o.on("--incremental", "Translate only new/changed sources (default)") { @options[:mode] = :incremental }
1180
+ o.on("--check", "Report pending translations; exit 1 if any (no API calls)") { @options[:check] = true }
1181
+ o.on("-n", "--dry-run", "Plan only; no API calls, no writes") { @options[:dry_run] = true }
1182
+ o.on("--langs LANGS", "Comma-separated target languages (overrides config)") { |v| @options[:langs] = v.split(",").map(&:strip) }
1183
+ o.on("--only PATTERN", "Only sources matching this glob/substring") { |v| @options[:only] = v }
1184
+ o.on("--limit N", Integer, "Cap the number of translation jobs this run") { |v| @options[:limit] = v }
1185
+ o.on("--provider NAME", "claude | stub (overrides config)") { |v| @options[:provider] = v }
1186
+ o.on("--model MODEL", "Override the Claude model") { |v| @options[:model] = v }
1187
+ o.on("--root PATH", "Repo root (default: cwd; used by tests)") { |v| @options[:root] = v }
1188
+ o.on("-V", "--verbose", "Extra logging") { @options[:verbose] = true }
1189
+ o.on("-v", "--version", "Print version") { puts VERSION; exit 0 }
1190
+ o.on("-h", "--help", "Show this help") { puts o; exit 0 }
1191
+ end.parse!(argv)
1192
+ end
1193
+ end
1194
+ end
1195
+
1196
+ if $PROGRAM_NAME == __FILE__
1197
+ begin
1198
+ exit Zer0Translate::CLI.new(ARGV).run
1199
+ rescue StandardError => e
1200
+ Zer0Translate::Log.error e.message
1201
+ Zer0Translate::Log.debug e.backtrace.join("\n") if Zer0Translate::Log.verbose
1202
+ exit 1
1203
+ end
1204
+ end