jekyll-obsidian-site 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (191) hide show
  1. checksums.yaml +7 -0
  2. data/LICENSE +21 -0
  3. data/README.md +133 -0
  4. data/website/_config.yml +21 -0
  5. data/website/_includes/website/archive-timeline.html +8 -0
  6. data/website/_includes/website/blog-ledger.html +19 -0
  7. data/website/_includes/website/blog-navigation.html +7 -0
  8. data/website/_includes/website/comments.html +22 -0
  9. data/website/_includes/website/contact-icon.html +14 -0
  10. data/website/_includes/website/context-tags.html +7 -0
  11. data/website/_includes/website/context.html +3 -0
  12. data/website/_includes/website/dialogs.html +71 -0
  13. data/website/_includes/website/docs-navigation.html +7 -0
  14. data/website/_includes/website/docs-tree.html +8 -0
  15. data/website/_includes/website/footer.html +14 -0
  16. data/website/_includes/website/github-icon.html +3 -0
  17. data/website/_includes/website/head.html +41 -0
  18. data/website/_includes/website/header.html +35 -0
  19. data/website/_includes/website/language-switcher.html +16 -0
  20. data/website/_includes/website/local-graph.html +25 -0
  21. data/website/_includes/website/note-header.html +35 -0
  22. data/website/_includes/website/outline.html +9 -0
  23. data/website/_includes/website/page-actions.html +34 -0
  24. data/website/_includes/website/portfolio-grid.html +36 -0
  25. data/website/_includes/website/portfolio-topics.html +4 -0
  26. data/website/_includes/website/post-cards.html +16 -0
  27. data/website/_includes/website/related-articles.html +7 -0
  28. data/website/_includes/website/relations.html +16 -0
  29. data/website/_includes/website/source-actions.html +28 -0
  30. data/website/_includes/website/system-page.html +31 -0
  31. data/website/_includes/website/tab-collection.html +28 -0
  32. data/website/_includes/website/tag-filter.html +12 -0
  33. data/website/_layouts/website-docs.html +20 -0
  34. data/website/_layouts/website-minimal.html +93 -0
  35. data/website/_layouts/website-redirect.html +17 -0
  36. data/website/assets/chunks/abnfDiagram-N423BO3Z-TSORAEI2.js +1 -0
  37. data/website/assets/chunks/architecture-TIHT7OUA-OUC5PEC7.js +1 -0
  38. data/website/assets/chunks/architectureDiagram-T3A2C74G-6P5MB2DJ.js +36 -0
  39. data/website/assets/chunks/blockDiagram-VBNYF7ZC-ZWINYROZ.js +132 -0
  40. data/website/assets/chunks/c4Diagram-5PPSVZJV-R5RMEAOT.js +10 -0
  41. data/website/assets/chunks/chunk-22DRK7VS.js +1 -0
  42. data/website/assets/chunks/chunk-25REBZ7M.js +2 -0
  43. data/website/assets/chunks/chunk-2RQCUDTG.js +1 -0
  44. data/website/assets/chunks/chunk-34FHPXU4.js +1 -0
  45. data/website/assets/chunks/chunk-57XXJOQN.js +10 -0
  46. data/website/assets/chunks/chunk-6DCNDVBX.js +2 -0
  47. data/website/assets/chunks/chunk-6DHFEWGX.js +1 -0
  48. data/website/assets/chunks/chunk-6VQ47JJS.js +2 -0
  49. data/website/assets/chunks/chunk-7CSIRG7W.js +127 -0
  50. data/website/assets/chunks/chunk-A43YELH5.js +1 -0
  51. data/website/assets/chunks/chunk-AIX6IH4O.js +1 -0
  52. data/website/assets/chunks/chunk-B3K7RH5O.js +88 -0
  53. data/website/assets/chunks/chunk-CBGUD2NP.js +1 -0
  54. data/website/assets/chunks/chunk-D3WOCELE.js +1 -0
  55. data/website/assets/chunks/chunk-D6FURU6X.js +1 -0
  56. data/website/assets/chunks/chunk-DGEGT77F.js +1 -0
  57. data/website/assets/chunks/chunk-DN52R6U7.js +1 -0
  58. data/website/assets/chunks/chunk-E3T3IFAC.js +1 -0
  59. data/website/assets/chunks/chunk-EIBFAW37.js +1 -0
  60. data/website/assets/chunks/chunk-ERWOH47R.js +1 -0
  61. data/website/assets/chunks/chunk-ETOVTYWF.js +1 -0
  62. data/website/assets/chunks/chunk-GHD3PFWA.js +231 -0
  63. data/website/assets/chunks/chunk-GMBKQ7NZ.js +1 -0
  64. data/website/assets/chunks/chunk-GQM25EPP.js +156 -0
  65. data/website/assets/chunks/chunk-GSOU4UTT.js +1 -0
  66. data/website/assets/chunks/chunk-HEMELP7J.js +1 -0
  67. data/website/assets/chunks/chunk-HZ5I5NJO.js +1 -0
  68. data/website/assets/chunks/chunk-IP56FOSI.js +70 -0
  69. data/website/assets/chunks/chunk-L7BMR7WL.js +1 -0
  70. data/website/assets/chunks/chunk-LGUG3SQB.js +1 -0
  71. data/website/assets/chunks/chunk-LOMHZYS2.js +1 -0
  72. data/website/assets/chunks/chunk-MII4VZEB.js +3 -0
  73. data/website/assets/chunks/chunk-MYMJLHQL.js +62 -0
  74. data/website/assets/chunks/chunk-NBH7YOBD.js +15 -0
  75. data/website/assets/chunks/chunk-NCOVM3YF.js +1 -0
  76. data/website/assets/chunks/chunk-OAGTSILX.js +1 -0
  77. data/website/assets/chunks/chunk-OCX5OCIU.js +321 -0
  78. data/website/assets/chunks/chunk-OE3NOVIH.js +1 -0
  79. data/website/assets/chunks/chunk-PLP5XUTJ.js +1 -0
  80. data/website/assets/chunks/chunk-PNOERP6G.js +1 -0
  81. data/website/assets/chunks/chunk-QBPQ7DPE.js +1 -0
  82. data/website/assets/chunks/chunk-RWFFDL53.js +1 -0
  83. data/website/assets/chunks/chunk-SEGIW3MP.js +1 -0
  84. data/website/assets/chunks/chunk-SZDMZNUS.js +1 -0
  85. data/website/assets/chunks/chunk-T2QBAPVT.js +1 -0
  86. data/website/assets/chunks/chunk-UE533OZ7.js +1 -0
  87. data/website/assets/chunks/chunk-UTWOJE2A.js +1 -0
  88. data/website/assets/chunks/chunk-VOCDYA3H.js +1 -0
  89. data/website/assets/chunks/chunk-VYZZPKGK.js +161 -0
  90. data/website/assets/chunks/chunk-WWLZIRS6.js +1 -0
  91. data/website/assets/chunks/chunk-XNXYDBKX.js +32 -0
  92. data/website/assets/chunks/chunk-XQUIO562.js +1 -0
  93. data/website/assets/chunks/chunk-Y2XVFW7T.js +2 -0
  94. data/website/assets/chunks/chunk-YORIST7Z.js +1 -0
  95. data/website/assets/chunks/chunk-Z2WBQPVY.js +206 -0
  96. data/website/assets/chunks/classDiagram-JCYQIIEL-OFDVDDIM.js +1 -0
  97. data/website/assets/chunks/classDiagram-v2-OCEON4UE-E36B6BGQ.js +1 -0
  98. data/website/assets/chunks/cose-bilkent-JH36ORCC-CBAW64AB.js +1 -0
  99. data/website/assets/chunks/cynefin-VYW2F7L2-CBO6DLNV.js +1 -0
  100. data/website/assets/chunks/cynefinDiagram-MW4NZA55-DFINLOUY.js +62 -0
  101. data/website/assets/chunks/dagre-VZM6K2ZE-BZFBSZAH.js +4 -0
  102. data/website/assets/chunks/diagram-7IWD3JNH-XQHTVOU2.js +30 -0
  103. data/website/assets/chunks/diagram-B4RE2ZJO-FTSXBZJR.js +3 -0
  104. data/website/assets/chunks/diagram-LBJQPF4R-ZSZW4OLH.js +24 -0
  105. data/website/assets/chunks/diagram-Q27KOJAE-2IV5TVCJ.js +24 -0
  106. data/website/assets/chunks/diagram-UB23O5K3-D6PDUTFL.js +41 -0
  107. data/website/assets/chunks/ebnfDiagram-BXEA7PRR-C6MAER3M.js +1 -0
  108. data/website/assets/chunks/erDiagram-JOGREHBK-EDAD2H5M.js +85 -0
  109. data/website/assets/chunks/eventmodeling-45OFAUF4-4K577DMF.js +1 -0
  110. data/website/assets/chunks/flowDiagram-UKHOOZJN-545F4YX3.js +1 -0
  111. data/website/assets/chunks/ganttDiagram-PKOTCBZU-7A7EKMY3.js +292 -0
  112. data/website/assets/chunks/gitGraph-TEB2WS4Q-3WUB7WYQ.js +1 -0
  113. data/website/assets/chunks/gitGraphDiagram-DS77QQ5N-A5UWV5FF.js +106 -0
  114. data/website/assets/chunks/info-DKCQHKI2-EOOHI7XS.js +1 -0
  115. data/website/assets/chunks/infoDiagram-6WML65LV-CWIOOWJP.js +2 -0
  116. data/website/assets/chunks/ishikawaDiagram-WSZJBQD7-CEASQVDL.js +70 -0
  117. data/website/assets/chunks/journeyDiagram-NVQOT4AX-LYG3S2AW.js +139 -0
  118. data/website/assets/chunks/kanban-definition-27J2QSJJ-HCC6KAYI.js +89 -0
  119. data/website/assets/chunks/katex-F4MQ6FB3.js +257 -0
  120. data/website/assets/chunks/mindmap-definition-FAOFIHXS-JMGSM3RE.js +96 -0
  121. data/website/assets/chunks/packet-7NZHBO7P-W3SJA3MY.js +1 -0
  122. data/website/assets/chunks/pegDiagram-VL7TDLO6-4JO3Y3Z3.js +1 -0
  123. data/website/assets/chunks/pie-RZYD4A2V-YOIQBTY7.js +1 -0
  124. data/website/assets/chunks/pieDiagram-7S7Q4E2Y-VFQ7SE46.js +39 -0
  125. data/website/assets/chunks/quadrantDiagram-CIZ2JOQS-JISDI2S6.js +7 -0
  126. data/website/assets/chunks/radar-I7S5WNFK-NMKP5AZO.js +1 -0
  127. data/website/assets/chunks/railroad-3IZDKUUU-7MOIV7CI.js +1 -0
  128. data/website/assets/chunks/railroad-abnf-AHOZXSZD-HXNH5SLO.js +1 -0
  129. data/website/assets/chunks/railroad-ebnf-EBAXGLYW-OWDJEWMK.js +1 -0
  130. data/website/assets/chunks/railroad-peg-LSFZ7HO6-ASSINGP5.js +1 -0
  131. data/website/assets/chunks/railroadDiagram-AXF67PYL-AFV5VP5F.js +1 -0
  132. data/website/assets/chunks/requirementDiagram-LRYGKXZP-RNRPEHHN.js +84 -0
  133. data/website/assets/chunks/sankeyDiagram-W5VNT64P-O54R2L3E.js +40 -0
  134. data/website/assets/chunks/sequenceDiagram-SI44F4Z6-42WISKJ7.js +162 -0
  135. data/website/assets/chunks/sizeCapture-X5ZJPWSS-LRN2Q37T.js +1 -0
  136. data/website/assets/chunks/stateDiagram-OKZ733FA-BWIBJXIR.js +1 -0
  137. data/website/assets/chunks/stateDiagram-v2-UEYNNEHI-L73LHL3L.js +1 -0
  138. data/website/assets/chunks/swimlanes-SLNWSIFB-ECOAE7LQ.js +1 -0
  139. data/website/assets/chunks/swimlanesDiagram-ULZ7WXOC-3CKKKFZZ.js +8 -0
  140. data/website/assets/chunks/timeline-definition-Z64GVDOM-HR2OL5G7.js +120 -0
  141. data/website/assets/chunks/treeView-QDETBFTQ-C6D5WFJA.js +1 -0
  142. data/website/assets/chunks/treemap-6X3UGDF4-JFCDMSCA.js +1 -0
  143. data/website/assets/chunks/vennDiagram-T6HMQDX7-62WAAPTF.js +34 -0
  144. data/website/assets/chunks/wardley-OPB4EBWU-RQIOII7I.js +1 -0
  145. data/website/assets/chunks/wardleyDiagram-T6FBY63Y-KUXO3NHV.js +78 -0
  146. data/website/assets/chunks/xychartDiagram-ELKLHX3M-I54W4HJ7.js +7 -0
  147. data/website/assets/color-scheme-bootstrap-4H4ZR4BW.js +1 -0
  148. data/website/assets/docs-3WBBQT6K.css +1 -0
  149. data/website/assets/docs-W3D5DWQF.js +1 -0
  150. data/website/assets/docs-navigation-75ALFYBA.js +1 -0
  151. data/website/assets/graph-QWL54LWD.js +1 -0
  152. data/website/assets/manifest.json +314 -0
  153. data/website/assets/math-7JTH4OWZ.js +21 -0
  154. data/website/assets/mermaid-RXC6URG4.js +10 -0
  155. data/website/assets/minimal-3NPTDWQX.js +1 -0
  156. data/website/assets/minimal-E2VATXXS.css +1 -0
  157. data/website/assets/previews-FMHOX7UX.js +1 -0
  158. data/website/assets/search-BH73JZDU.js +1 -0
  159. data/website/assets/search-worker-SM4IM4J7.js +1 -0
  160. data/website/exe/jekyll-obsidian +5 -0
  161. data/website/lib/jekyll_obsidian/adapter.rb +733 -0
  162. data/website/lib/jekyll_obsidian/built_in_themes.rb +855 -0
  163. data/website/lib/jekyll_obsidian/cli.rb +84 -0
  164. data/website/lib/jekyll_obsidian/content_policy.rb +194 -0
  165. data/website/lib/jekyll_obsidian/destination_registry.rb +39 -0
  166. data/website/lib/jekyll_obsidian/directory_indexes.rb +30 -0
  167. data/website/lib/jekyll_obsidian/external_media.rb +371 -0
  168. data/website/lib/jekyll_obsidian/front_matter.rb +453 -0
  169. data/website/lib/jekyll_obsidian/github_markdown.rb +664 -0
  170. data/website/lib/jekyll_obsidian/html_publication.rb +206 -0
  171. data/website/lib/jekyll_obsidian/initializer.rb +68 -0
  172. data/website/lib/jekyll_obsidian/localized_compiler.rb +1044 -0
  173. data/website/lib/jekyll_obsidian/media_policy.rb +41 -0
  174. data/website/lib/jekyll_obsidian/ofm_scanner.rb +493 -0
  175. data/website/lib/jekyll_obsidian/output_text.rb +13 -0
  176. data/website/lib/jekyll_obsidian/preview.rb +93 -0
  177. data/website/lib/jekyll_obsidian/published_markdown.rb +32 -0
  178. data/website/lib/jekyll_obsidian/raw_html_manifest.rb +66 -0
  179. data/website/lib/jekyll_obsidian/runtime.rb +197 -0
  180. data/website/lib/jekyll_obsidian/site_compilation.rb +147 -0
  181. data/website/lib/jekyll_obsidian/site_navigation.rb +517 -0
  182. data/website/lib/jekyll_obsidian/url_builder.rb +163 -0
  183. data/website/lib/jekyll_obsidian/value_objects.rb +198 -0
  184. data/website/lib/jekyll_obsidian/vault_compiler.rb +2771 -0
  185. data/website/lib/jekyll_obsidian/version.rb +6 -0
  186. data/website/lib/jekyll_obsidian/workspace_layout.rb +104 -0
  187. data/website/lib/jekyll_obsidian.rb +22 -0
  188. data/website/scripts/audit-site.rb +120 -0
  189. data/website/scripts/templates/pages.yml +55 -0
  190. data/website/scripts/verify-site-urls.rb +585 -0
  191. metadata +319 -0
@@ -0,0 +1,2771 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi/escape"
4
+ require "commonmarker"
5
+ require "json"
6
+ require "nokogiri"
7
+ require "pathname"
8
+ require "set"
9
+ require "uri"
10
+
11
+ module JekyllObsidian
12
+ class VaultCompiler
13
+ COMMONMARK_OPTIONS = {
14
+ parse: {
15
+ smart: false,
16
+ relaxed_tasklist_matching: true,
17
+ relaxed_autolinks: true,
18
+ sourcepos_chars: true
19
+ },
20
+ render: {
21
+ unsafe: true,
22
+ hardbreaks: true,
23
+ tasklist_classes: true,
24
+ escaped_char_spans: true,
25
+ sourcepos: true
26
+ },
27
+ extension: {
28
+ strikethrough: true,
29
+ table: true,
30
+ autolink: true,
31
+ tasklist: true,
32
+ footnotes: true,
33
+ math_dollars: true,
34
+ math_code: true,
35
+ math_latex: true,
36
+ wikilinks_title_after_pipe: true,
37
+ highlight: true,
38
+ cjk_friendly_emphasis: true
39
+ }
40
+ }.freeze
41
+ COMMONMARK_PLUGINS = { syntax_highlighter: { theme: "" } }.freeze
42
+ REMOTE_COMMONMARK_OPTIONS = {
43
+ parse: COMMONMARK_OPTIONS.fetch(:parse),
44
+ render: COMMONMARK_OPTIONS.fetch(:render).merge(unsafe: false),
45
+ extension: {
46
+ strikethrough: true,
47
+ table: true,
48
+ autolink: true,
49
+ tasklist: true,
50
+ footnotes: true
51
+ }
52
+ }.freeze
53
+ DANGEROUS_SCHEMES = %w[data file javascript vbscript].freeze
54
+ EXTERNAL_SCHEMES = %w[http https mailto tel].freeze
55
+ NOTE_EXTENSION = ".md"
56
+ TRANSLATIONS_DIRECTORY = "_translations/"
57
+ THEMES = BuiltInThemes::IDS
58
+ FEATURE_KEYS = %w[search tags feed graph relations previews outline].freeze
59
+ COMMENT_KEYS = %w[enabled repository repository_id category category_id].freeze
60
+ ANALYTICS_PROVIDERS = %w[cloudflare google].freeze
61
+ CONTACT_KEYS = %w[label url].freeze
62
+ CONTACT_HOST_ICONS = {
63
+ "github.com" => "github",
64
+ "linkedin.com" => "linkedin",
65
+ "x.com" => "x",
66
+ "twitter.com" => "x",
67
+ "mastodon.social" => "mastodon",
68
+ "bsky.app" => "bluesky",
69
+ "bsky.social" => "bluesky",
70
+ "instagram.com" => "instagram",
71
+ "youtube.com" => "youtube",
72
+ "youtu.be" => "youtube",
73
+ "t.me" => "telegram",
74
+ "telegram.me" => "telegram",
75
+ "telegram.org" => "telegram"
76
+ }.freeze
77
+ NAVIGATION_BUILTINS = {
78
+ "home" => { "order" => 0, "visible" => true }.freeze,
79
+ "blog" => { "order" => 10, "visible" => true }.freeze,
80
+ "docs" => { "order" => 20, "visible" => true }.freeze
81
+ }.freeze
82
+ PORTFOLIO_NAVIGATION_DEFAULTS = {
83
+ "path" => "portfolio", "order" => 30, "visible" => true
84
+ }.freeze
85
+ NAVIGATION_OVERRIDE_KEYS = %w[label order visible].freeze
86
+ NAVIGATION_KEYS = (NAVIGATION_BUILTINS.keys + %w[portfolio]).freeze
87
+ PORTFOLIO_NAVIGATION_KEYS = (NAVIGATION_OVERRIDE_KEYS + %w[path]).freeze
88
+ MAX_CONTACTS = 12
89
+ GISCUS_LANGUAGES = Set.new(%w[
90
+ ar be bg ca cs da de en eo es eu fa fr gr hbs he hu id it ja kh ko nl pl
91
+ pt ro ru th tr uk uz vi zh-CN zh-HK zh-TW
92
+ ]).freeze
93
+ THEME_FEATURE_DEFAULTS = {
94
+ "minimal" => {
95
+ "search" => true, "tags" => true, "feed" => true,
96
+ "graph" => true, "relations" => true, "previews" => true, "outline" => true
97
+ },
98
+ "docs" => {
99
+ "search" => true, "tags" => false, "feed" => false,
100
+ "graph" => true, "relations" => true, "previews" => true, "outline" => true
101
+ },
102
+ }.transform_values(&:freeze).freeze
103
+ MutableNote = Struct.new(
104
+ :id,
105
+ :entry,
106
+ :properties,
107
+ :frontmatter_links,
108
+ :body,
109
+ :document,
110
+ :scanner,
111
+ :title,
112
+ :has_h1,
113
+ :route,
114
+ :occurrences,
115
+ :base_fragment,
116
+ :authored_text,
117
+ :preview,
118
+ :outline,
119
+ :anchors,
120
+ :updated,
121
+ :created,
122
+ :content_type,
123
+ :published_at,
124
+ :nav_order,
125
+ :nav_exclude,
126
+ :feature_flags,
127
+ :topics,
128
+ :external_document,
129
+ keyword_init: true
130
+ )
131
+
132
+ Occurrence = Struct.new(
133
+ :index,
134
+ :source_id,
135
+ :raw_target,
136
+ :display,
137
+ :kind,
138
+ :syntax,
139
+ :source_span,
140
+ :scanner_token,
141
+ :resolved_type,
142
+ :target_id,
143
+ :target_path,
144
+ :fragment,
145
+ :options,
146
+ :anchor_id,
147
+ :unresolved,
148
+ :property,
149
+ :external_media,
150
+ keyword_init: true
151
+ )
152
+
153
+ Topic = Struct.new(:kind, :name, :occurrence, keyword_init: true)
154
+ Anchor = Struct.new(:kind, :id, :label, :level, :chain, keyword_init: true)
155
+ TransclusionContext = Struct.new(:host_id, :sequence, :instances, :bytes, keyword_init: true)
156
+ MAX_TRANSCLUSION_DEPTH = 16
157
+ MAX_TRANSCLUSION_INSTANCES = 256
158
+ MAX_TRANSCLUSION_BYTES = 2 * 1024 * 1024
159
+
160
+ def self.compile(request)
161
+ return LocalizedCompiler.compile(request) unless request.config.i18n.nil?
162
+
163
+ compile_single(request)
164
+ end
165
+
166
+ def self.compile_single(request, html_publication: nil, include_html_projection: true)
167
+ new(request, html_publication:, include_html_projection:).compile
168
+ end
169
+
170
+ def self.giscus_language(locale)
171
+ GISCUS_LANGUAGES.include?(locale.to_s) ? locale.to_s : "en"
172
+ end
173
+
174
+ def initialize(request, html_publication: nil, include_html_projection: true)
175
+ @request = request
176
+ @config = request.config
177
+ @diagnostics = []
178
+ @notes = {}
179
+ @all_note_paths = Set.new
180
+ @directory_index_paths = DeepFreeze.call({})
181
+ @attachments = {}
182
+ @attachment_basename_index = Hash.new { |hash, key| hash[key] = [] }
183
+ @image_paths = {}
184
+ @relations = []
185
+ @projected_attachment_paths = Set.new
186
+ @html_publication = html_publication || HtmlPublication.empty_resolution
187
+ @provided_html_publication = html_publication
188
+ @include_html_projection = include_html_projection
189
+ @html_routes_by_source = {}
190
+ @html_document_sources = Set.new
191
+ @transclusion_selection_cache = {}
192
+ @url_builder = nil
193
+ @theme = "minimal"
194
+ @features = THEME_FEATURE_DEFAULTS.fetch(@theme)
195
+ @content_policy = ContentPolicy.resolve(nil).policy
196
+ @content = @content_policy.settings
197
+ @comments = disabled_comments_config
198
+ @analytics = disabled_analytics_config
199
+ @contacts = []
200
+ @navigation_config = default_navigation_config
201
+ end
202
+
203
+ def compile
204
+ validate_request
205
+ resolve_html_publication
206
+ index_snapshot
207
+ parse_public_notes
208
+ establish_identities
209
+ parse_markdown_once
210
+ resolve_all_occurrences
211
+ detect_embed_cycles
212
+ render_authored_documents
213
+ merge_content_features
214
+
215
+ published_site = build_published_site_model
216
+ navigation = SiteNavigation.build(
217
+ model: published_site,
218
+ settings: @navigation_config,
219
+ content: @content,
220
+ url_builder: @url_builder,
221
+ theme: @theme,
222
+ member_ids_by_tab_id: @request.tab_memberships
223
+ )
224
+ @diagnostics.concat(navigation.diagnostics)
225
+ theme_config = EffectiveThemeConfig.new(
226
+ theme: @theme,
227
+ features: @features,
228
+ content: @content,
229
+ comments: @comments,
230
+ analytics: @analytics,
231
+ contacts: @contacts,
232
+ navigation: navigation,
233
+ site: @config,
234
+ url_builder: @url_builder
235
+ )
236
+ theme_output = BuiltInThemes.resolve(@theme).render(model: published_site, config: theme_config)
237
+ pages = theme_output.pages
238
+ generated_files = theme_output.shared_files + build_generated_files(pages, published_site, theme_output)
239
+ raw_html_manifest = HtmlPublication.manifest(@html_publication) if @include_html_projection
240
+ generated_files << raw_html_manifest if raw_html_manifest
241
+ projected_files = build_projected_files
242
+ projected_files += @html_publication.projected_files if @include_html_projection
243
+ preflight_routes(pages, generated_files, projected_files, theme_output.reserved_namespaces)
244
+
245
+ note_outputs = published_site.notes.map do |note|
246
+ NoteOutput.new(id: note.id, title: note.title, route: note.route, properties: note.properties)
247
+ end
248
+ diagnostics = sorted_diagnostics
249
+ return BuildFailure.new(diagnostics: diagnostics) if diagnostics.any? { |item| item.severity == :error }
250
+
251
+ BuildSuccess.new(
252
+ pages: pages.sort_by(&:route),
253
+ generated_files: generated_files.sort_by(&:route),
254
+ projected_files: projected_files.sort_by(&:route),
255
+ diagnostics: diagnostics,
256
+ relations: published_site.relations,
257
+ notes: note_outputs,
258
+ theme: @theme,
259
+ features: @features,
260
+ site_data: theme_output.site_data
261
+ )
262
+ end
263
+
264
+ private
265
+
266
+ def validate_request
267
+ unless @request.is_a?(BuildRequest) && @request.snapshot.is_a?(Snapshot) && @config.is_a?(BuildConfig)
268
+ error("invalid_request", "compile expects a BuildRequest containing Snapshot and BuildConfig")
269
+ return
270
+ end
271
+
272
+ @config.to_h.each do |name, value|
273
+ next if FrontMatter.valid_output_text?(value.to_s)
274
+
275
+ error("invalid_config_character", "#{name} contains a character forbidden by XML 1.0")
276
+ end
277
+ if @config.syntax_profile != "ofm@1"
278
+ error("unsupported_syntax_profile", "syntax_profile must be ofm@1")
279
+ end
280
+ resolve_theme_config
281
+ resolve_content_config
282
+ resolve_comments_config
283
+ resolve_analytics_config
284
+ resolve_contacts_config
285
+ resolve_navigation_config
286
+ @url_builder = UrlBuilder.new(origin: @config.url, baseurl: @config.baseurl)
287
+ error("missing_origin", "production builds require a non-empty url origin") if production? && @url_builder.origin.empty?
288
+ rescue ArgumentError => exception
289
+ error("invalid_url_config", exception.message)
290
+ @url_builder = UrlBuilder.new(origin: "", baseurl: "")
291
+ end
292
+
293
+ def resolve_html_publication
294
+ if @provided_html_publication
295
+ @html_routes_by_source = @html_publication.routes_by_source
296
+ @html_document_sources = Set.new(@html_publication.document_sources)
297
+ return
298
+ end
299
+
300
+ @html_publication = HtmlPublication.resolve(
301
+ snapshot: @request.snapshot,
302
+ mappings: @config.html,
303
+ url_builder: @url_builder
304
+ )
305
+ @html_routes_by_source = @html_publication.routes_by_source
306
+ @html_document_sources = Set.new(@html_publication.document_sources)
307
+ @diagnostics.concat(@html_publication.diagnostics)
308
+ end
309
+
310
+ def resolve_theme_config
311
+ requested = @config.theme.to_s
312
+ requested = "minimal" if requested.empty?
313
+ if THEMES.include?(requested)
314
+ @theme = requested
315
+ else
316
+ error("invalid_theme", "theme must be one of: #{THEMES.join(', ')}")
317
+ @theme = "minimal"
318
+ end
319
+
320
+ overrides = @config.features
321
+ unless overrides.nil? || overrides.is_a?(Hash)
322
+ error("invalid_features", "features must be a mapping of supported feature names to YAML booleans")
323
+ overrides = {}
324
+ end
325
+ overrides ||= {}
326
+ normalized = {}
327
+ overrides.each do |key, value|
328
+ name = key.to_s
329
+ unless FEATURE_KEYS.include?(name)
330
+ error("invalid_feature", "unknown feature #{name.inspect}")
331
+ next
332
+ end
333
+ unless value == true || value == false
334
+ error("invalid_feature", "feature #{name.inspect} must be a YAML boolean")
335
+ next
336
+ end
337
+ normalized[name] = value
338
+ end
339
+ @features = THEME_FEATURE_DEFAULTS.fetch(@theme).merge(normalized).sort.to_h.freeze
340
+ end
341
+
342
+ def resolve_comments_config
343
+ raw = @config.comments
344
+ if raw.nil?
345
+ @comments = disabled_comments_config
346
+ return
347
+ end
348
+ unless raw.is_a?(Hash) && raw.keys.all? { |key| key.is_a?(String) }
349
+ error("invalid_comments", "website.comments must be a mapping with string keys")
350
+ @comments = disabled_comments_config
351
+ return
352
+ end
353
+
354
+ unknown = raw.keys - COMMENT_KEYS
355
+ unknown.sort.each { |key| error("invalid_comments", "unknown comments setting #{key.inspect}") }
356
+ enabled = raw.fetch("enabled", @theme == "minimal")
357
+ unless enabled == true || enabled == false
358
+ error("invalid_comments", "comments.enabled must be a YAML boolean")
359
+ enabled = false
360
+ end
361
+
362
+ repository = raw.key?("repository") ? raw["repository"] : @config.repository
363
+ repository = validate_comment_text("repository", repository)
364
+ repository_id = validate_comment_text("repository_id", raw["repository_id"])
365
+ category = validate_comment_text("category", raw["category"])
366
+ category_id = validate_comment_text("category_id", raw["category_id"])
367
+
368
+ repository_valid = repository&.match?(/\A[\w.-]+\/[\w.-]+\z/)
369
+ error("invalid_comments", "comments.repository must be an owner/repository pair") if enabled && !repository_valid
370
+ missing_provider_fields = {
371
+ "repository_id" => repository_id,
372
+ "category" => category,
373
+ "category_id" => category_id
374
+ }.filter_map { |key, value| "comments.#{key}" if value.to_s.empty? }
375
+ configured = enabled && repository_valid && missing_provider_fields.empty?
376
+ if enabled && repository_valid && missing_provider_fields.any?
377
+ warning(
378
+ "comments_unconfigured",
379
+ "comments are enabled but Giscus setup is incomplete; missing #{missing_provider_fields.join(', ')}"
380
+ )
381
+ end
382
+
383
+ @comments = CommentsConfig.new(
384
+ enabled: enabled,
385
+ configured: configured,
386
+ repository: repository.to_s,
387
+ repository_id: repository_id.to_s,
388
+ category: category.to_s,
389
+ category_id: category_id.to_s,
390
+ language: self.class.giscus_language(@config.lang),
391
+ load: configured && production?
392
+ )
393
+ end
394
+
395
+ def validate_comment_text(key, value)
396
+ return nil if value.nil?
397
+ unless value.is_a?(String) && FrontMatter.valid_output_text?(value) && value.length <= 256
398
+ error("invalid_comments", "comments.#{key} must be a string of at most 256 output-safe characters")
399
+ return nil
400
+ end
401
+
402
+ value.strip
403
+ end
404
+
405
+ def disabled_comments_config
406
+ CommentsConfig.new(
407
+ enabled: false,
408
+ configured: false,
409
+ repository: "",
410
+ repository_id: "",
411
+ category: "",
412
+ category_id: "",
413
+ language: "en",
414
+ load: false
415
+ )
416
+ end
417
+
418
+ def resolve_analytics_config
419
+ raw = @config.analytics
420
+ return @analytics = disabled_analytics_config if raw.nil?
421
+ unless raw.is_a?(Hash) && raw.keys.all? { |key| key.is_a?(String) }
422
+ error("invalid_analytics", "website.analytics must be a mapping with string keys")
423
+ return @analytics = disabled_analytics_config
424
+ end
425
+
426
+ provider = raw["provider"]
427
+ unless ANALYTICS_PROVIDERS.include?(provider)
428
+ error("invalid_analytics", "analytics.provider must be one of: #{ANALYTICS_PROVIDERS.join(', ')}")
429
+ return @analytics = disabled_analytics_config
430
+ end
431
+
432
+ identifier_key = provider == "cloudflare" ? "token" : "measurement_id"
433
+ (raw.keys - ["provider", identifier_key]).sort.each do |key|
434
+ error("invalid_analytics", "unknown analytics setting #{key.inspect} for provider #{provider.inspect}")
435
+ end
436
+ value = raw[identifier_key]
437
+ unless value.is_a?(String) && !value.empty? && value.length <= 256 &&
438
+ value == value.strip && !value.match?(/[\s<>"']/) && FrontMatter.valid_output_text?(value)
439
+ error("invalid_analytics", "analytics.#{identifier_key} must be a non-empty output-safe token of at most 256 characters")
440
+ return @analytics = disabled_analytics_config
441
+ end
442
+ identifier = value
443
+ if provider == "google" && !identifier.match?(/\AG-[A-Z0-9]+\z/)
444
+ error("invalid_analytics", "analytics.measurement_id must begin with G- and contain only uppercase ASCII letters and digits")
445
+ return @analytics = disabled_analytics_config
446
+ end
447
+
448
+ @analytics = AnalyticsConfig.new(provider: provider, identifier: identifier, load: production?)
449
+ end
450
+
451
+ def disabled_analytics_config
452
+ AnalyticsConfig.new(provider: "", identifier: "", load: false)
453
+ end
454
+
455
+ def resolve_contacts_config
456
+ raw = @config.contacts
457
+ return @contacts = [] if raw.nil?
458
+ unless raw.is_a?(Array)
459
+ error("invalid_contacts", "website.contacts must be a list of label and url mappings")
460
+ return @contacts = []
461
+ end
462
+ error("invalid_contacts", "website.contacts supports at most #{MAX_CONTACTS} entries") if raw.length > MAX_CONTACTS
463
+
464
+ @contacts = raw.first(MAX_CONTACTS).each_with_index.filter_map do |entry, index|
465
+ unless entry.is_a?(Hash) && entry.keys.all? { |key| key.is_a?(String) }
466
+ error("invalid_contacts", "contacts[#{index}] must be a mapping with string keys")
467
+ next
468
+ end
469
+ unknown = entry.keys - CONTACT_KEYS
470
+ unless unknown.empty?
471
+ error("invalid_contacts", "unknown contacts[#{index}] setting #{unknown.sort.first.inspect}")
472
+ next
473
+ end
474
+
475
+ label = contact_text(entry["label"], "contacts[#{index}].label", 64)
476
+ url = contact_text(entry["url"], "contacts[#{index}].url", 2048)
477
+ next unless label && url
478
+ unless contact_url?(url)
479
+ error("invalid_contacts", "contacts[#{index}].url must use https, mailto, or tel")
480
+ next
481
+ end
482
+
483
+ contact = { "label" => label, "url" => url }
484
+ icon = contact_icon(label, url)
485
+ contact["icon"] = icon if icon
486
+ contact
487
+ end
488
+ end
489
+
490
+ def resolve_navigation_config
491
+ raw = @config.navigation
492
+ return @navigation_config = default_navigation_config if raw.nil?
493
+ unless raw.is_a?(Hash) && raw.keys.all? { |key| key.is_a?(String) }
494
+ error("invalid_navigation_config", "website.navigation must be a mapping with string keys")
495
+ return @navigation_config = default_navigation_config
496
+ end
497
+
498
+ (raw.keys - NAVIGATION_KEYS).sort.each do |key|
499
+ error("invalid_navigation_config", "unknown website.navigation setting #{key.inspect}")
500
+ end
501
+
502
+ normalized = NAVIGATION_BUILTINS.to_h do |name, defaults|
503
+ value = if raw.key?(name)
504
+ normalize_navigation_override(raw[name], "website.navigation.#{name}", defaults)
505
+ else
506
+ defaults.dup
507
+ end
508
+ [name, value]
509
+ end
510
+ normalized["portfolio"] = if raw.key?("portfolio")
511
+ normalize_portfolio_navigation(raw["portfolio"])
512
+ else
513
+ PORTFOLIO_NAVIGATION_DEFAULTS.dup
514
+ end
515
+ @navigation_config = DeepFreeze.call(normalized)
516
+ end
517
+
518
+ def default_navigation_config
519
+ DeepFreeze.call(
520
+ NAVIGATION_BUILTINS.transform_values(&:dup).merge(
521
+ "portfolio" => PORTFOLIO_NAVIGATION_DEFAULTS.dup
522
+ )
523
+ )
524
+ end
525
+
526
+ def normalize_portfolio_navigation(value)
527
+ path = "website.navigation.portfolio"
528
+ unless value.is_a?(Hash) && value.keys.all? { |key| key.is_a?(String) }
529
+ error("invalid_navigation_config", "#{path} must be a mapping with string keys")
530
+ return PORTFOLIO_NAVIGATION_DEFAULTS.dup
531
+ end
532
+
533
+ (value.keys - PORTFOLIO_NAVIGATION_KEYS).sort.each do |key|
534
+ error("invalid_navigation_config", "unknown #{path} setting #{key.inspect}")
535
+ end
536
+ normalized = normalize_navigation_fields(value, path, PORTFOLIO_NAVIGATION_DEFAULTS)
537
+ return normalized unless value.key?("path")
538
+
539
+ folder_path = normalize_navigation_folder_path(value["path"], "#{path}.path")
540
+ folder_path ? normalized.merge("path" => folder_path) : normalized
541
+ end
542
+
543
+ def normalize_navigation_override(value, path, defaults)
544
+ unless value.is_a?(Hash) && value.keys.all? { |key| key.is_a?(String) }
545
+ error("invalid_navigation_config", "#{path} must be a mapping with string keys")
546
+ return defaults.dup
547
+ end
548
+
549
+ (value.keys - NAVIGATION_OVERRIDE_KEYS).sort.each do |key|
550
+ error("invalid_navigation_config", "unknown #{path} setting #{key.inspect}")
551
+ end
552
+ normalize_navigation_fields(value, path, defaults)
553
+ end
554
+
555
+ def normalize_navigation_fields(value, path, defaults)
556
+ normalized = defaults.dup
557
+ if value.key?("label")
558
+ label = value["label"]
559
+ if FrontMatter.valid_output_text?(label) && !label.strip.empty?
560
+ normalized["label"] = label
561
+ else
562
+ error("invalid_navigation_config", "#{path}.label must be a non-empty string containing only output-safe Unicode characters")
563
+ end
564
+ end
565
+ if value.key?("order")
566
+ order = value["order"]
567
+ if order.is_a?(Integer)
568
+ normalized["order"] = order
569
+ else
570
+ error("invalid_navigation_config", "#{path}.order must be an integer")
571
+ end
572
+ end
573
+ if value.key?("visible")
574
+ visible = value["visible"]
575
+ if visible == true || visible == false
576
+ normalized["visible"] = visible
577
+ else
578
+ error("invalid_navigation_config", "#{path}.visible must be a YAML boolean")
579
+ end
580
+ end
581
+ normalized
582
+ end
583
+
584
+ def normalize_navigation_folder_path(value, path)
585
+ unless FrontMatter.valid_output_text?(value)
586
+ error("invalid_navigation_config", "#{path} must be a string")
587
+ return nil
588
+ end
589
+ if value.empty? || value.start_with?("/", "\\") || value.include?("\\") || value != value.unicode_normalize(:nfc)
590
+ error("invalid_navigation_config", "#{path} must be a normalized vault-relative POSIX directory")
591
+ return nil
592
+ end
593
+ segments = value.split("/", -1)
594
+ if segments.any? { |segment| segment.empty? || segment == "." || segment == ".." }
595
+ error("invalid_navigation_config", "#{path} must not contain empty or traversal segments")
596
+ return nil
597
+ end
598
+ value
599
+ rescue EncodingError
600
+ error("invalid_navigation_config", "#{path} must contain valid Unicode")
601
+ nil
602
+ end
603
+
604
+ def contact_text(value, key, limit)
605
+ unless value.is_a?(String) && !value.strip.empty? && value.length <= limit && FrontMatter.valid_output_text?(value)
606
+ error("invalid_contacts", "#{key} must be a non-empty output-safe string of at most #{limit} characters")
607
+ return nil
608
+ end
609
+
610
+ value.strip
611
+ end
612
+
613
+ def contact_url?(value)
614
+ uri = URI.parse(value)
615
+ case uri.scheme&.downcase
616
+ when "https"
617
+ !uri.host.to_s.empty?
618
+ when "mailto", "tel"
619
+ !uri.opaque.to_s.empty? && !uri.opaque.match?(/\s/)
620
+ else
621
+ false
622
+ end
623
+ rescue URI::InvalidURIError
624
+ false
625
+ end
626
+
627
+ def contact_icon(label, value)
628
+ uri = URI.parse(value)
629
+ scheme = uri.scheme&.downcase
630
+ return "email" if scheme == "mailto"
631
+ return "phone" if scheme == "tel"
632
+
633
+ host = uri.host.to_s.downcase.delete_suffix(".")
634
+ CONTACT_HOST_ICONS.each do |domain, icon|
635
+ return icon if host == domain || host.end_with?(".#{domain}")
636
+ end
637
+
638
+ normalized_label = label.downcase.strip
639
+ return "mastodon" if normalized_label.match?(/\bmastodon\b/)
640
+ return "rss" if normalized_label.match?(/\b(?:rss|feed|atom)\b/) || feed_url_path?(uri.path)
641
+ return "website" if normalized_label.match?(/\A(?:website|web[ -]?site|site|homepage|home[ -]?page)\z/)
642
+
643
+ nil
644
+ rescue URI::InvalidURIError
645
+ nil
646
+ end
647
+
648
+ def feed_url_path?(path)
649
+ path.to_s.downcase.match?(%r{(?:\A|/)(?:feed|rss|atom)(?:\.(?:xml|rss|atom))?(?:/|\z)})
650
+ end
651
+
652
+ def resolve_content_config
653
+ resolution = ContentPolicy.resolve(@config.content)
654
+ @diagnostics.concat(resolution.diagnostics)
655
+ @content_policy = resolution.policy
656
+ @content = @content_policy.settings
657
+ end
658
+
659
+ def index_snapshot
660
+ seen = {}
661
+ Array(@request.snapshot&.entries).sort_by { |entry| entry.path.to_s.b }.each do |entry|
662
+ path = validated_path(entry)
663
+ next unless path
664
+ next if path.start_with?(TRANSLATIONS_DIRECTORY)
665
+
666
+ collision = path.unicode_normalize(:nfc).downcase(:fold)
667
+ if seen.key?(collision)
668
+ error("path_collision", "snapshot paths are equivalent under NFC/case-folding", path)
669
+ error("route_collision", "note paths would map to equivalent public routes", path) if entry.kind.to_sym == :note
670
+ next
671
+ end
672
+ seen[collision] = path
673
+
674
+ case entry.kind.to_sym
675
+ when :note
676
+ unless path.end_with?(NOTE_EXTENSION)
677
+ error("invalid_note_path", "note paths must end in .md", path)
678
+ next
679
+ end
680
+ @all_note_paths << path
681
+ parsed = FrontMatter.parse(path, entry.bytes.to_s)
682
+ @diagnostics.concat(parsed.diagnostics)
683
+ published = @content_policy.publish?(path, parsed.properties)
684
+ if (parsed.properties.key?("tab") || parsed.properties.key?("tabs")) && !published
685
+ error(
686
+ "unpublished_tab_declaration",
687
+ "tab and tabs frontmatter are only supported on published pages",
688
+ path
689
+ )
690
+ end
691
+ next unless published
692
+ unless FrontMatter.valid_output_text?(parsed.body)
693
+ error("invalid_character", "public note body contains a character forbidden by XML 1.0", path)
694
+ next
695
+ end
696
+
697
+ external_document = entry.external_document
698
+ if external_document && !parsed.body.strip.empty?
699
+ error(
700
+ "github_markdown_body_conflict",
701
+ "a github_markdown project wrapper must have an empty local body",
702
+ path
703
+ )
704
+ end
705
+ @notes[path] = MutableNote.new(
706
+ id: path,
707
+ entry: entry,
708
+ properties: parsed.properties,
709
+ frontmatter_links: parsed.property_links,
710
+ body: external_document ? external_document.markdown : parsed.body,
711
+ occurrences: [],
712
+ outline: [],
713
+ feature_flags: {},
714
+ external_document: external_document
715
+ )
716
+ when :attachment
717
+ @attachments[path] = entry
718
+ basename = File.basename(path).unicode_normalize(:nfc).downcase(:fold)
719
+ @attachment_basename_index[basename] << path
720
+ when :locale_manifest
721
+ # Locale manifests are compiler metadata, never public vault assets.
722
+ when :symlink
723
+ error("symlink_rejected", "symlinks are not accepted in a vault snapshot", path)
724
+ else
725
+ error("invalid_entry_kind", "snapshot entry kind must be note, attachment, or locale_manifest", path)
726
+ end
727
+ end
728
+ @directory_index_paths = DirectoryIndexes.resolve(@all_note_paths)
729
+ end
730
+
731
+ def parse_public_notes
732
+ if @notes.empty?
733
+ error("missing_public_notes", "the content directory must contain at least one public note; add publish: true to a note or configure website.content.publish_by_default", nil)
734
+ end
735
+ end
736
+
737
+ def establish_identities
738
+ @basename_index = Hash.new { |hash, key| hash[key] = [] }
739
+ @notes.each_value do |note|
740
+ route = note.properties["permalink"]
741
+ if route
742
+ route = @url_builder.validate_permalink(route)
743
+ error("invalid_permalink", "permalink must be a concrete site path beginning and ending with /", note.id) unless route
744
+ else
745
+ route = @url_builder.route_for_note(note.id, directory_index: directory_index_note?(note.id))
746
+ end
747
+ note.route = route || @url_builder.route_for_note(note.id, directory_index: directory_index_note?(note.id))
748
+ if root_index_note?(note.id) && note.route != "/"
749
+ error("invalid_home_permalink", "the public root index must publish at /", note.id)
750
+ end
751
+ note.updated = note.properties["updated"]
752
+ note.created = deterministic_created(note)
753
+ if root_index_note?(note.id) && note.properties["content_type"] && note.properties["content_type"] != "page"
754
+ error("invalid_root_content_type", "the public root index must have content_type: page", note.id)
755
+ end
756
+ note.content_type = effective_content_type(note)
757
+ note.nav_order = note.properties["nav_order"]
758
+ note.nav_exclude = note.properties["nav_exclude"] == true
759
+ note.published_at = published_at(note)
760
+ if note.content_type == "post" && note.published_at.nil?
761
+ error_or_warning(
762
+ "missing_post_date",
763
+ "published posts require date, created, or Git first commit time",
764
+ note.id,
765
+ nil,
766
+ fatal: production?
767
+ )
768
+ end
769
+ basename = File.basename(note.id, NOTE_EXTENSION).unicode_normalize(:nfc).downcase(:fold)
770
+ @basename_index[basename] << note.id
771
+ end
772
+
773
+ note_routes = {}
774
+ @notes.each_value do |note|
775
+ key = @url_builder.collision_key(note.route)
776
+ if note_routes.key?(key)
777
+ error("route_collision", "public notes map to equivalent routes", note.id)
778
+ else
779
+ note_routes[key] = note.id
780
+ end
781
+ end
782
+ end
783
+
784
+ def published_at(note)
785
+ return nil unless note.content_type == "post"
786
+
787
+ note.properties["date"] || note.properties["created"] || note.entry.first_committed_at
788
+ end
789
+
790
+ def effective_content_type(note)
791
+ return "page" if root_index_note?(note.id)
792
+
793
+ classified = @content_policy.classify(note.id, note.properties)
794
+ return classified unless portfolio_note?(note.id)
795
+
796
+ explicit = note.properties["content_type"]
797
+ if explicit && explicit != "page"
798
+ error(
799
+ "portfolio_content_type_conflict",
800
+ "published notes inside the portfolio path must use content_type: page",
801
+ note.id
802
+ )
803
+ end
804
+ "page"
805
+ end
806
+
807
+ def portfolio_note?(note_id)
808
+ return false unless @theme == "minimal"
809
+
810
+ path = @navigation_config.fetch("portfolio").fetch("path")
811
+ note_id.start_with?("#{path}/")
812
+ end
813
+
814
+ def directory_index_note?(note_id)
815
+ @directory_index_paths[File.dirname(note_id)] == note_id
816
+ end
817
+
818
+ def root_index_note?(note_id)
819
+ @directory_index_paths["."] == note_id
820
+ end
821
+
822
+ def parse_markdown_once
823
+ @notes.values.sort_by(&:id).each do |note|
824
+ prepared = OfmScanner.prepare(note.external_document ? "" : note.body)
825
+ note.scanner = prepared
826
+ merged_tags = (Array(note.properties["tags"]) + prepared.tags).uniq.sort
827
+ note.properties = note.properties.merge("tags" => merged_tags)
828
+ markdown = (note.external_document ? note.body : prepared.markdown).dup.force_encoding(Encoding::UTF_8)
829
+ options = note.external_document ? REMOTE_COMMONMARK_OPTIONS : COMMONMARK_OPTIONS
830
+ note.document = Commonmarker.parse(markdown, options: options)
831
+ if note.external_document
832
+ rewrite_github_markdown_urls(note)
833
+ note.body = note.document.to_commonmark(options: { render: { width: 0 } })
834
+ end
835
+ note.has_h1 = note.document.any? { |node| node.type == :heading && node.header_level == 1 }
836
+ note.title = note.properties["title"] || first_h1(note.document) || filename_title(note.id)
837
+ build_anchor_registry(note)
838
+ annotate_occurrences(note) unless note.external_document
839
+ annotate_frontmatter_links(note)
840
+ annotate_frontmatter_topics(note)
841
+ annotate_code_markers(note)
842
+ end
843
+ end
844
+
845
+ def rewrite_github_markdown_urls(note)
846
+ note.document.walk.each do |node|
847
+ next unless %i[link image].include?(node.type)
848
+
849
+ rewritten = github_markdown_url(note, node.url.to_s, image: node.type == :image)
850
+ node.url = rewritten if rewritten
851
+ end
852
+ end
853
+
854
+ def github_markdown_url(note, raw_url, image:)
855
+ return raw_url if raw_url.empty? || raw_url.start_with?("#")
856
+
857
+ uri = parse_github_markdown_uri(raw_url)
858
+ if uri.scheme
859
+ if image && uri.scheme.downcase != "https"
860
+ error("invalid_github_markdown_image", "GitHub Markdown images must use HTTPS", note.id)
861
+ return note.external_document.source_url
862
+ end
863
+ return raw_url
864
+ end
865
+ if uri.host || raw_url.start_with?("//")
866
+ error("invalid_github_markdown_link", "protocol-relative GitHub Markdown links are not supported", note.id)
867
+ return note.external_document.source_url
868
+ end
869
+
870
+ decoded = URI.decode_uri_component(uri.path.to_s)
871
+ if decoded.include?("\\") || decoded.include?("\0")
872
+ error("invalid_github_markdown_link", "GitHub Markdown links must use safe POSIX paths", note.id)
873
+ return note.external_document.source_url
874
+ end
875
+ base = decoded.start_with?("/") ? "" : File.dirname(note.external_document.path)
876
+ candidate = Pathname.new(File.join(base, decoded.delete_prefix("/"))).cleanpath.to_s.tr("\\", "/")
877
+ if candidate == ".." || candidate.start_with?("../")
878
+ error("invalid_github_markdown_link", "GitHub Markdown links must not escape the repository root", note.id)
879
+ return note.external_document.source_url
880
+ end
881
+
882
+ encoded = candidate == "." ? "" : candidate.split("/").map { |part| URI.encode_uri_component(part) }.join("/")
883
+ repository = note.external_document.repository
884
+ commit = note.external_document.resolved_commit
885
+ target = if image
886
+ "https://raw.githubusercontent.com/#{repository}/#{commit}"
887
+ else
888
+ directory = uri.path.to_s.end_with?("/")
889
+ "https://github.com/#{repository}/#{directory ? 'tree' : 'blob'}/#{commit}"
890
+ end
891
+ target = "#{target}/#{encoded}" unless encoded.empty?
892
+ target = "#{target}?#{uri.query}" if uri.query
893
+ target = "#{target}##{uri.fragment}" if uri.fragment
894
+ target
895
+ rescue URI::InvalidURIError, ArgumentError
896
+ error("invalid_github_markdown_link", "GitHub Markdown contains an invalid relative URL", note.id)
897
+ note.external_document.source_url
898
+ end
899
+
900
+ def parse_github_markdown_uri(raw_url)
901
+ ascii_url = if raw_url.ascii_only?
902
+ raw_url
903
+ else
904
+ raw_url.gsub(/[^\x00-\x7F]/) do |character|
905
+ character.bytes.map { |byte| format("%%%02X", byte) }.join
906
+ end
907
+ end
908
+ URI.parse(ascii_url)
909
+ end
910
+
911
+ def build_anchor_registry(note)
912
+ used_heading_ids = Hash.new(0)
913
+ identifiers = {}
914
+ heading_stack = []
915
+ anchors = []
916
+ outline = []
917
+
918
+ note.document.select { |node| node.type == :heading }.each do |heading|
919
+ label = plain_node_text(heading).strip
920
+ level = heading.header_level
921
+ heading_stack.pop while heading_stack.last && heading_stack.last.fetch(:level) >= level
922
+ chain = heading_stack.map { |item| item.fetch(:label) } + [label]
923
+ base = @url_builder.slug(label)
924
+ used_heading_ids[base] += 1
925
+ duplicate_number = note.external_document ? used_heading_ids[base] - 1 : used_heading_ids[base]
926
+ identifier = used_heading_ids[base] == 1 ? base : "#{base}-#{duplicate_number}"
927
+ anchor = Anchor.new(kind: :heading, id: identifier, label: label, level: level, chain: chain)
928
+ anchors << anchor
929
+ identifiers[identifier] = anchor
930
+ outline << { "id" => identifier, "label" => label, "level" => level }
931
+ heading_stack << { level: level, label: label }
932
+ end
933
+
934
+ note.scanner.block_ids.each do |identifier, line_number|
935
+ if identifiers.key?(identifier)
936
+ error(
937
+ "anchor_collision",
938
+ "block ID collides with an existing heading or block anchor",
939
+ note.id,
940
+ SourceSpan.new(start_line: line_number, start_column: 1, end_line: line_number, end_column: 1)
941
+ )
942
+ next
943
+ end
944
+
945
+ anchor = Anchor.new(kind: :block, id: identifier, label: identifier, level: nil, chain: nil)
946
+ anchors << anchor
947
+ identifiers[identifier] = anchor
948
+ end
949
+
950
+ note.anchors = anchors
951
+ note.outline = outline
952
+ end
953
+
954
+ def annotate_occurrences(note)
955
+ note.scanner.embeds.each do |embed|
956
+ note.occurrences << Occurrence.new(
957
+ index: note.occurrences.length,
958
+ source_id: note.id,
959
+ raw_target: embed.target,
960
+ display: nil,
961
+ kind: :embed,
962
+ syntax: :ofm_embed,
963
+ source_span: embed.source_span,
964
+ scanner_token: embed.token
965
+ )
966
+ end
967
+
968
+ note.scanner.iframes.each do |iframe|
969
+ descriptor = ExternalMedia.resolve_iframe(iframe.html, closed: iframe.closed)
970
+ note.occurrences << Occurrence.new(
971
+ index: note.occurrences.length,
972
+ source_id: note.id,
973
+ raw_target: descriptor.source_url,
974
+ display: descriptor.title,
975
+ kind: :embed,
976
+ syntax: :html_iframe,
977
+ source_span: iframe.source_span,
978
+ scanner_token: iframe.token,
979
+ resolved_type: :external_media,
980
+ external_media: descriptor
981
+ )
982
+ rescue ExternalMedia::Invalid => exception
983
+ error("invalid_external_media", exception.message, note.id, iframe.source_span)
984
+ end
985
+
986
+ note.scanner.wikilinks.each do |link|
987
+ note.occurrences << Occurrence.new(
988
+ index: note.occurrences.length,
989
+ source_id: note.id,
990
+ raw_target: link.target,
991
+ display: link.display,
992
+ kind: :link,
993
+ syntax: :wikilink,
994
+ source_span: link.source_span,
995
+ scanner_token: link.token
996
+ )
997
+ end
998
+
999
+ note.document.walk do |node|
1000
+ case node.type
1001
+ when :link, :image
1002
+ raw_url = node.url.to_s
1003
+ next if raw_url.empty?
1004
+ if external_url?(raw_url, note.id, source_span(node.source_position), media: node.type == :image)
1005
+ annotate_external_media(note, node, raw_url) if node.type == :image
1006
+ next
1007
+ end
1008
+
1009
+ display = plain_node_text(node)
1010
+ occurrence_target = raw_url
1011
+ if node.type == :image && (dimension = display.match(/\A(.*)\|(\d+(?:x\d+)?)\z/m))
1012
+ display = dimension[1]
1013
+ occurrence_target = "#{raw_url}|#{dimension[2]}"
1014
+ end
1015
+
1016
+ occurrence = Occurrence.new(
1017
+ index: note.occurrences.length,
1018
+ source_id: note.id,
1019
+ raw_target: occurrence_target,
1020
+ display: display,
1021
+ kind: node.type == :image ? :embed : :link,
1022
+ syntax: node.type == :image ? :markdown_image : :markdown_link,
1023
+ source_span: source_span(node.source_position)
1024
+ )
1025
+ note.occurrences << occurrence
1026
+ node.url = token_url(occurrence.index)
1027
+ end
1028
+ end
1029
+ end
1030
+
1031
+ def annotate_external_media(note, node, raw_url)
1032
+ descriptor = ExternalMedia.resolve(raw_url)
1033
+ return unless descriptor
1034
+
1035
+ if descriptor.kind != :image && node.parent&.type != :paragraph
1036
+ error(
1037
+ "invalid_external_media",
1038
+ "block external media must use a standalone Markdown image",
1039
+ note.id,
1040
+ source_span(node.source_position)
1041
+ )
1042
+ return
1043
+ end
1044
+
1045
+ display = plain_node_text(node)
1046
+ options = {}
1047
+ if descriptor.kind == :image
1048
+ display, options = external_image_display(display)
1049
+ end
1050
+ occurrence = Occurrence.new(
1051
+ index: note.occurrences.length,
1052
+ source_id: note.id,
1053
+ raw_target: raw_url,
1054
+ display: display,
1055
+ kind: :embed,
1056
+ syntax: :markdown_image,
1057
+ source_span: source_span(node.source_position),
1058
+ resolved_type: :external_media,
1059
+ options: options,
1060
+ external_media: descriptor
1061
+ )
1062
+ note.occurrences << occurrence
1063
+ node.url = token_url(occurrence.index)
1064
+ rescue ExternalMedia::Invalid => exception
1065
+ error("invalid_external_media", exception.message, note.id, source_span(node.source_position))
1066
+ end
1067
+
1068
+ def external_image_display(display)
1069
+ if (dimension = display.match(/\A(\d+)(?:x(\d+))?\z/))
1070
+ return ["", { "width" => dimension[1].to_i, "height" => dimension[2]&.to_i }.compact]
1071
+ end
1072
+ if (dimension = display.match(/\A(.*)\|(\d+)(?:x(\d+))?\z/m))
1073
+ return [dimension[1], { "width" => dimension[2].to_i, "height" => dimension[3]&.to_i }.compact]
1074
+ end
1075
+
1076
+ [display, {}]
1077
+ end
1078
+
1079
+ def annotate_frontmatter_topics(note)
1080
+ note.topics = Array(note.properties["tags"]).map do |tag|
1081
+ Topic.new(kind: "tag", name: tag)
1082
+ end
1083
+
1084
+ { "author" => "author", "categories" => "category" }.each do |property, kind|
1085
+ Array(note.properties[property]).each do |value|
1086
+ link = FrontMatter.parse_wiki_link(value)
1087
+ unless link
1088
+ note.topics << Topic.new(kind: kind, name: value)
1089
+ next
1090
+ end
1091
+ target, display = link
1092
+
1093
+ occurrence = Occurrence.new(
1094
+ index: note.occurrences.length,
1095
+ source_id: note.id,
1096
+ raw_target: target,
1097
+ display: display,
1098
+ kind: :link,
1099
+ syntax: :frontmatter_topic,
1100
+ property: property
1101
+ )
1102
+ note.occurrences << occurrence
1103
+ note.topics << Topic.new(kind: kind, occurrence: occurrence)
1104
+ end
1105
+ end
1106
+ end
1107
+
1108
+ def annotate_frontmatter_links(note)
1109
+ Array(note.frontmatter_links).each do |link|
1110
+ note.occurrences << Occurrence.new(
1111
+ index: note.occurrences.length,
1112
+ source_id: note.id,
1113
+ raw_target: link.target,
1114
+ display: link.display,
1115
+ kind: :link,
1116
+ syntax: :frontmatter_property,
1117
+ source_span: link.source_span,
1118
+ property: link.property
1119
+ )
1120
+ end
1121
+ end
1122
+
1123
+ def resolve_all_occurrences
1124
+ @notes.values.sort_by(&:id).each do |note|
1125
+ note.occurrences.each do |occurrence|
1126
+ resolve_occurrence(note, occurrence) unless occurrence.resolved_type == :external_media
1127
+ if occurrence.resolved_type == :note && !(occurrence.property == "related" && occurrence.unresolved)
1128
+ @relations << Relation.new(
1129
+ source_id: note.id,
1130
+ target_id: occurrence.target_id,
1131
+ kind: occurrence.kind,
1132
+ fragment: occurrence.fragment,
1133
+ source_span: occurrence.source_span,
1134
+ property: occurrence.property
1135
+ )
1136
+ elsif occurrence.resolved_type == :attachment
1137
+ @projected_attachment_paths << occurrence.target_path
1138
+ end
1139
+ end
1140
+
1141
+ image = note.properties["image"]
1142
+ next unless image
1143
+
1144
+ resolved, ambiguous = resolve_attachment_path(note.id, image)
1145
+ if ambiguous
1146
+ error_or_warning("ambiguous_attachment", "image property matches more than one attachment", note.id, nil, fatal: production?)
1147
+ elsif resolved && MediaPolicy.kind(resolved) == :image
1148
+ @projected_attachment_paths << resolved unless @html_routes_by_source.key?(resolved)
1149
+ @image_paths[note.id] = resolved
1150
+ else
1151
+ error_or_warning("missing_image_property", "image property does not resolve to an attachment", note.id, nil, fatal: production?)
1152
+ end
1153
+ end
1154
+ end
1155
+
1156
+ def resolve_occurrence(note, occurrence)
1157
+ target_text, fragment, options = split_target(occurrence.raw_target, occurrence.kind)
1158
+ occurrence.fragment = fragment
1159
+ occurrence.options = options
1160
+
1161
+ if target_text.empty?
1162
+ if occurrence.property == "related"
1163
+ occurrence.unresolved = true
1164
+ error_or_warning(
1165
+ "related_self_reference",
1166
+ "related must not link a note to itself",
1167
+ note.id,
1168
+ occurrence.source_span,
1169
+ fatal: production?,
1170
+ property: occurrence.property
1171
+ )
1172
+ return
1173
+ end
1174
+ occurrence.resolved_type = :note
1175
+ occurrence.target_id = note.id
1176
+ resolve_occurrence_fragment(note, occurrence)
1177
+ return
1178
+ end
1179
+
1180
+ if local_target_escapes_vault?(note.id, target_text)
1181
+ occurrence.unresolved = true
1182
+ error_or_warning(
1183
+ "path_escape",
1184
+ "local target escapes the vault root",
1185
+ note.id,
1186
+ occurrence.source_span,
1187
+ fatal: production?,
1188
+ property: occurrence.property
1189
+ )
1190
+ return
1191
+ end
1192
+
1193
+ note_target, ambiguous = resolve_note_path(note.id, target_text)
1194
+ if ambiguous
1195
+ occurrence.unresolved = true
1196
+ code = "ambiguous_target"
1197
+ if production?
1198
+ error(code, "target is ambiguous", note.id, occurrence.source_span, property: occurrence.property)
1199
+ else
1200
+ warning(
1201
+ code,
1202
+ "target is ambiguous; rendered as a placeholder",
1203
+ note.id,
1204
+ occurrence.source_span,
1205
+ property: occurrence.property
1206
+ )
1207
+ end
1208
+ return
1209
+ end
1210
+
1211
+ if note_target
1212
+ occurrence.resolved_type = :note
1213
+ occurrence.target_id = note_target
1214
+ if occurrence.property == "related" && note_target == note.id
1215
+ occurrence.unresolved = true
1216
+ error_or_warning(
1217
+ "related_self_reference",
1218
+ "related must not link a note to itself",
1219
+ note.id,
1220
+ occurrence.source_span,
1221
+ fatal: production?,
1222
+ property: occurrence.property
1223
+ )
1224
+ return
1225
+ end
1226
+ resolve_occurrence_fragment(@notes.fetch(note_target), occurrence)
1227
+ return
1228
+ end
1229
+
1230
+ if %i[frontmatter_property frontmatter_topic].include?(occurrence.syntax)
1231
+ occurrence.unresolved = true
1232
+ if occurrence.property == "related"
1233
+ error_or_warning(
1234
+ "unresolved_related",
1235
+ "related wiki link target is missing or not a public note",
1236
+ note.id,
1237
+ occurrence.source_span,
1238
+ fatal: production?,
1239
+ property: occurrence.property
1240
+ )
1241
+ else
1242
+ warning(
1243
+ "unresolved_property_link",
1244
+ "#{occurrence.property} wiki link target is missing or not a public note",
1245
+ note.id,
1246
+ occurrence.source_span,
1247
+ property: occurrence.property
1248
+ )
1249
+ end
1250
+ return
1251
+ end
1252
+
1253
+ attachment_target, ambiguous_attachment = resolve_attachment_path(note.id, target_text)
1254
+ if ambiguous_attachment
1255
+ occurrence.unresolved = true
1256
+ error_or_warning(
1257
+ "ambiguous_attachment",
1258
+ "attachment target is ambiguous",
1259
+ note.id,
1260
+ occurrence.source_span,
1261
+ fatal: production?
1262
+ )
1263
+ return
1264
+ end
1265
+
1266
+ if attachment_target
1267
+ if @html_routes_by_source.key?(attachment_target)
1268
+ if occurrence.kind == :embed && @html_document_sources.include?(attachment_target)
1269
+ occurrence.unresolved = true
1270
+ error(
1271
+ "html_embed_unsupported",
1272
+ "raw HTML can only be linked as an independent page",
1273
+ note.id,
1274
+ occurrence.source_span
1275
+ )
1276
+ elsif occurrence.kind == :embed && !MediaPolicy.kind(attachment_target)
1277
+ occurrence.unresolved = true
1278
+ error(
1279
+ "unsupported_attachment",
1280
+ "mapped bundle file type is not supported for embedding",
1281
+ note.id,
1282
+ occurrence.source_span
1283
+ )
1284
+ else
1285
+ occurrence.resolved_type = :projected_file
1286
+ occurrence.target_path = attachment_target
1287
+ end
1288
+ return
1289
+ end
1290
+ unless MediaPolicy.kind(attachment_target)
1291
+ occurrence.unresolved = true
1292
+ error(
1293
+ "unsupported_attachment",
1294
+ "attachment type is not supported for publication",
1295
+ note.id,
1296
+ occurrence.source_span
1297
+ )
1298
+ return
1299
+ end
1300
+ occurrence.resolved_type = :attachment
1301
+ occurrence.target_path = attachment_target
1302
+ return
1303
+ end
1304
+
1305
+ occurrence.unresolved = true
1306
+ if occurrence.kind == :embed
1307
+ error_or_warning("missing_embed", "embed target is missing or not public", note.id, occurrence.source_span, fatal: production?)
1308
+ else
1309
+ warning("unresolved_link", "link target is missing or not public", note.id, occurrence.source_span)
1310
+ end
1311
+ end
1312
+
1313
+ def resolve_occurrence_fragment(target_note, occurrence)
1314
+ fragment = occurrence.fragment
1315
+ return unless fragment && !fragment.empty?
1316
+
1317
+ anchor = find_fragment_anchor(target_note, fragment)
1318
+ if anchor
1319
+ occurrence.anchor_id = anchor.id
1320
+ return
1321
+ end
1322
+
1323
+ occurrence.unresolved = true
1324
+ if occurrence.kind == :embed
1325
+ error_or_warning(
1326
+ "missing_embed_fragment",
1327
+ "embed fragment does not exist in the public target",
1328
+ occurrence.source_id,
1329
+ occurrence.source_span,
1330
+ fatal: production?,
1331
+ property: occurrence.property
1332
+ )
1333
+ elsif occurrence.property == "related"
1334
+ error_or_warning(
1335
+ "unresolved_related_fragment",
1336
+ "related wiki link fragment does not exist in the public target",
1337
+ occurrence.source_id,
1338
+ occurrence.source_span,
1339
+ fatal: production?,
1340
+ property: occurrence.property
1341
+ )
1342
+ else
1343
+ warning(
1344
+ "unresolved_fragment",
1345
+ "link fragment does not exist in the public target; rendered as unresolved",
1346
+ occurrence.source_id,
1347
+ occurrence.source_span,
1348
+ property: occurrence.property
1349
+ )
1350
+ end
1351
+ end
1352
+
1353
+ def find_fragment_anchor(note, raw_fragment)
1354
+ anchors = note.anchors || []
1355
+ fragment = safe_decode(raw_fragment).unicode_normalize(:nfc)
1356
+ if fragment.start_with?("^")
1357
+ identifier = fragment.delete_prefix("^")
1358
+ return anchors.find { |anchor| anchor.kind == :block && anchor.id == identifier }
1359
+ end
1360
+
1361
+ direct = fragment
1362
+ by_id = anchors.find { |anchor| anchor.kind == :heading && anchor.id == direct }
1363
+ return by_id if by_id
1364
+
1365
+ chain = fragment.split("#").map(&:strip).reject(&:empty?).map { |label| @url_builder.slug(label) }
1366
+ return nil if chain.empty?
1367
+
1368
+ anchors.find do |anchor|
1369
+ next false unless anchor.kind == :heading
1370
+
1371
+ anchor_chain = anchor.chain.map { |label| @url_builder.slug(label) }
1372
+ anchor_chain.last(chain.length) == chain
1373
+ end
1374
+ end
1375
+
1376
+ def detect_embed_cycles
1377
+ graph = Hash.new { |hash, key| hash[key] = [] }
1378
+ @relations.each do |relation|
1379
+ graph[relation.source_id] << relation.target_id if relation.kind == :embed
1380
+ end
1381
+
1382
+ state = {}
1383
+ stack = []
1384
+ visit = lambda do |id|
1385
+ return if state[id] == :done
1386
+ if state[id] == :visiting
1387
+ cycle = stack.drop_while { |candidate| candidate != id } + [id]
1388
+ error_or_warning("embed_cycle", "embed cycle detected: #{cycle.join(" -> ")}", id, nil, fatal: production?)
1389
+ return
1390
+ end
1391
+
1392
+ state[id] = :visiting
1393
+ stack << id
1394
+ graph[id].sort.each { |target| visit.call(target) }
1395
+ stack.pop
1396
+ state[id] = :done
1397
+ end
1398
+ @notes.keys.sort.each { |id| visit.call(id) }
1399
+ end
1400
+
1401
+ def render_authored_documents
1402
+ @notes.values.sort_by(&:id).each do |note|
1403
+ options = note.external_document ? REMOTE_COMMONMARK_OPTIONS : COMMONMARK_OPTIONS
1404
+ html = note.document.to_html(options: options, plugins: COMMONMARK_PLUGINS)
1405
+ fragment = Nokogiri::HTML5.fragment(html)
1406
+ bind_occurrence_nodes(note, fragment)
1407
+ normalize_document(note, fragment)
1408
+ note.base_fragment = fragment
1409
+
1410
+ authored = fragment.dup
1411
+ authored.css("website-embed").remove
1412
+ note.authored_text = visible_text(authored)
1413
+ note.preview = truncate(note.properties["description"] || note.authored_text, 240)
1414
+ note.feature_flags = {
1415
+ "math" => !fragment.css("[data-math-style], .math, math").empty? || note.body.match?(/\$[^$]+\$/),
1416
+ "mermaid" => !fragment.css("pre code.language-mermaid").empty?
1417
+ }
1418
+ end
1419
+ end
1420
+
1421
+ def merge_content_features
1422
+ content_features = {
1423
+ "math" => @notes.values.any? { |note| note.feature_flags["math"] },
1424
+ "mermaid" => @notes.values.any? { |note| note.feature_flags["mermaid"] }
1425
+ }
1426
+ @features = @features.merge(content_features).sort.to_h.freeze
1427
+ end
1428
+
1429
+ def bind_occurrence_nodes(note, fragment)
1430
+ note.occurrences.select { |occurrence| occurrence.syntax == :wikilink }.each do |occurrence|
1431
+ node = fragment.at_css("a[data-website-wikilink-token='#{occurrence.scanner_token}']")
1432
+ node["data-website-occurrence"] = occurrence.index.to_s if node
1433
+ end
1434
+
1435
+ note.occurrences.select { |occurrence| occurrence.syntax == :ofm_embed }.each do |occurrence|
1436
+ node = fragment.at_css("website-ofm-embed[data-token='#{occurrence.scanner_token}']")
1437
+ node["data-website-occurrence"] = occurrence.index.to_s if node
1438
+ end
1439
+
1440
+ note.occurrences.select { |occurrence| occurrence.syntax == :html_iframe }.each do |occurrence|
1441
+ node = fragment.at_css("website-ofm-iframe[data-token='#{occurrence.scanner_token}']")
1442
+ node["data-website-occurrence"] = occurrence.index.to_s if node
1443
+ end
1444
+
1445
+ note.occurrences.select { |occurrence| %i[markdown_link markdown_image].include?(occurrence.syntax) }.each do |occurrence|
1446
+ selector = occurrence.syntax == :markdown_image ? "img[src='#{token_url(occurrence.index)}']" : "a[href='#{token_url(occurrence.index)}']"
1447
+ node = fragment.at_css(selector)
1448
+ node["data-website-occurrence"] = occurrence.index.to_s if node
1449
+ end
1450
+
1451
+ note.occurrences.each do |occurrence|
1452
+ node = fragment.at_css("[data-website-occurrence='#{occurrence.index}']")
1453
+ next unless node
1454
+
1455
+ transform_reference_node(note, occurrence, node)
1456
+ end
1457
+ promote_embed_placeholders(fragment)
1458
+ end
1459
+
1460
+ def promote_embed_placeholders(fragment)
1461
+ fragment.css("p").to_a.each do |paragraph|
1462
+ next unless paragraph.children.any? { |child| block_embed_node?(child) }
1463
+
1464
+ replacement = Nokogiri::HTML5.fragment("")
1465
+ inline = new_paragraph_like(paragraph)
1466
+ discard_break = false
1467
+ paragraph.children.to_a.each do |child|
1468
+ if block_embed_node?(child)
1469
+ replacement.add_child(inline) if meaningful_paragraph?(inline)
1470
+ inline = new_paragraph_like(paragraph)
1471
+ replacement.add_child(child.unlink)
1472
+ discard_break = true
1473
+ elsif discard_break && (child.name == "br" || (child.text? && child.text.strip.empty?))
1474
+ child.unlink
1475
+ else
1476
+ discard_break = false
1477
+ inline.add_child(child.unlink)
1478
+ end
1479
+ end
1480
+ replacement.add_child(inline) if meaningful_paragraph?(inline)
1481
+ paragraph.replace(replacement)
1482
+ end
1483
+ fragment.css("p").each { |paragraph| paragraph.remove unless meaningful_paragraph?(paragraph) }
1484
+ end
1485
+
1486
+ def block_embed_node?(node)
1487
+ return false unless node.element?
1488
+ return true if node.name == "website-embed"
1489
+
1490
+ classes = node["class"].to_s.split
1491
+ classes.any? do |name|
1492
+ %w[website-external-player website-external-video website-external-frame website-tweet].include?(name)
1493
+ end
1494
+ end
1495
+
1496
+ def new_paragraph_like(source)
1497
+ paragraph = Nokogiri::XML::Node.new("p", source.document)
1498
+ source.attribute_nodes.each { |attribute| paragraph[attribute.name] = attribute.value }
1499
+ paragraph
1500
+ end
1501
+
1502
+ def meaningful_paragraph?(paragraph)
1503
+ paragraph.children.any? { |child| child.element? || child.text.strip != "" }
1504
+ end
1505
+
1506
+ def transform_reference_node(note, occurrence, node)
1507
+ if occurrence.unresolved
1508
+ replacement = Nokogiri::XML::Node.new("span", node.document)
1509
+ replacement["class"] = occurrence.kind == :embed ? "website-embed website-embed--unresolved website-unresolved" : "website-link website-link--unresolved website-unresolved"
1510
+ replacement["role"] = "status"
1511
+ replacement.content = occurrence.display || occurrence.raw_target
1512
+ node.replace(replacement)
1513
+ return
1514
+ end
1515
+
1516
+ if occurrence.resolved_type == :note
1517
+ target = @notes.fetch(occurrence.target_id)
1518
+ anchor = occurrence.anchor_id ? "^#{occurrence.anchor_id}" : occurrence.fragment
1519
+ href = @url_builder.href(target.route) + @url_builder.fragment(anchor)
1520
+ if occurrence.kind == :embed
1521
+ placeholder = Nokogiri::XML::Node.new("website-embed", node.document)
1522
+ placeholder["data-source-id"] = target.id
1523
+ placeholder["data-fragment"] = occurrence.fragment.to_s
1524
+ placeholder["data-anchor-id"] = occurrence.anchor_id.to_s if occurrence.anchor_id
1525
+ placeholder["data-href"] = href
1526
+ node.replace(placeholder)
1527
+ else
1528
+ node.name = "a"
1529
+ node["href"] = href
1530
+ node["class"] = [node["class"], "website-link"].compact.join(" ")
1531
+ node["data-note-id"] = target.id
1532
+ node.remove_attribute("data-website-occurrence")
1533
+ end
1534
+ elsif occurrence.resolved_type == :attachment
1535
+ transform_attachment_node(occurrence, node)
1536
+ elsif occurrence.resolved_type == :projected_file
1537
+ transform_projected_file_node(occurrence, node)
1538
+ elsif occurrence.resolved_type == :external_media
1539
+ transform_external_media_node(occurrence, node)
1540
+ end
1541
+ end
1542
+
1543
+ def transform_external_media_node(occurrence, node)
1544
+ descriptor = occurrence.external_media
1545
+ if descriptor.kind != :image && node.parent&.name != "p"
1546
+ error(
1547
+ "invalid_external_media",
1548
+ "block external media must use a standalone Markdown image",
1549
+ occurrence.source_id,
1550
+ occurrence.source_span
1551
+ )
1552
+ return
1553
+ end
1554
+
1555
+ replacement = case descriptor.kind
1556
+ when :image
1557
+ image_node(node.document, occurrence, descriptor.source_url)
1558
+ when :direct_video
1559
+ external_video_node(node.document, descriptor, occurrence.display)
1560
+ when :player
1561
+ external_player_node(node.document, descriptor, occurrence.display)
1562
+ when :web_frame
1563
+ external_web_frame_node(node.document, descriptor)
1564
+ when :tweet
1565
+ tweet_node(node.document, descriptor)
1566
+ else
1567
+ raise "unsupported external media descriptor: #{descriptor.kind.inspect}"
1568
+ end
1569
+ node.replace(replacement)
1570
+ end
1571
+
1572
+ def transform_projected_file_node(occurrence, node)
1573
+ route = @html_routes_by_source.fetch(occurrence.target_path)
1574
+ if occurrence.kind == :embed
1575
+ transform_attachment_node(occurrence, node, route: route)
1576
+ return
1577
+ end
1578
+
1579
+ node.name = "a"
1580
+ node["href"] = @url_builder.href(route)
1581
+ node["class"] = [node["class"], "website-link"].compact.join(" ")
1582
+ node.remove_attribute("data-website-occurrence")
1583
+ end
1584
+
1585
+ def transform_attachment_node(occurrence, node, route: nil)
1586
+ entry = @attachments.fetch(occurrence.target_path)
1587
+ route ||= @url_builder.attachment_route(occurrence.target_path)
1588
+ href = @url_builder.href(route)
1589
+
1590
+ kind = MediaPolicy.kind(occurrence.target_path)
1591
+ media_type = MediaPolicy.media_type(occurrence.target_path, fallback: entry.media_type)
1592
+ replacement = if occurrence.kind == :link || kind == :download
1593
+ download_card(node.document, occurrence.target_path, href, media_type)
1594
+ elsif kind == :image
1595
+ image_node(node.document, occurrence, href)
1596
+ elsif kind == :audio
1597
+ media_node(node.document, "audio", href, media_type)
1598
+ elsif kind == :video
1599
+ media_node(node.document, "video", href, media_type)
1600
+ elsif kind == :pdf
1601
+ pdf_node(node.document, occurrence, href)
1602
+ else
1603
+ unresolved_attachment_node(node.document, occurrence.target_path)
1604
+ end
1605
+ node.replace(replacement)
1606
+ end
1607
+
1608
+ def normalize_document(note, fragment)
1609
+ fragment.xpath(".//comment()").remove
1610
+ assign_heading_ids(note, fragment)
1611
+ assign_block_ids(note, fragment)
1612
+ annotate_task_states(note, fragment)
1613
+ annotate_code_blocks(note, fragment)
1614
+ transform_callouts(fragment)
1615
+ fragment.css("img").each { |image| image["data-website-image"] = "true" }
1616
+ fragment.css("pre code.language-mermaid").each { |node| node.parent["data-website-mermaid"] = "true" }
1617
+ fragment.css("a[href]").each do |link|
1618
+ next unless link["href"].match?(%r{\Ahttps?://}i)
1619
+
1620
+ link["rel"] = "noopener noreferrer"
1621
+ end
1622
+ fragment.css("[data-sourcepos]").remove_attr("data-sourcepos")
1623
+ end
1624
+
1625
+ def assign_heading_ids(note, fragment)
1626
+ headings = note.anchors.select { |anchor| anchor.kind == :heading }
1627
+ source_headings = note.document.select { |node| node.type == :heading }
1628
+ headings.zip(source_headings).each do |anchor, source|
1629
+ heading = fragment.at_css("h#{anchor.level}[data-sourcepos='#{sourcepos_value(source.source_position)}']")
1630
+ next unless heading
1631
+
1632
+ heading["id"] = anchor.id
1633
+ heading.css("a.anchor").each do |permalink|
1634
+ permalink["href"] = "##{anchor.id}"
1635
+ label = permalink["data-heading-content"] || anchor.label
1636
+ permalink["aria-label"] = "Link to heading '#{label}'"
1637
+ end
1638
+ end
1639
+ end
1640
+
1641
+ def assign_block_ids(note, fragment)
1642
+ block_ids = note.anchors.select { |anchor| anchor.kind == :block }.map(&:id)
1643
+ fragment.css("[data-website-block-id]").each do |marker|
1644
+ parent = marker.parent
1645
+ identifier = marker["data-website-block-id"]
1646
+ if parent&.element? && block_ids.include?(identifier)
1647
+ standalone = parent.children.all? do |child|
1648
+ child == marker || (child.text? && child.text.strip.empty?)
1649
+ end
1650
+ previous = parent.previous_element if standalone
1651
+ if previous
1652
+ if previous["id"].to_s.empty?
1653
+ previous["id"] = identifier
1654
+ else
1655
+ # A heading already owns its public heading ID. Preserve it and
1656
+ # place a second scroll target immediately before the block;
1657
+ # overwriting the heading ID would break its outline and links.
1658
+ anchor = Nokogiri::XML::Node.new("span", previous.document)
1659
+ anchor["id"] = identifier
1660
+ anchor["class"] = "website-block-anchor"
1661
+ anchor["aria-hidden"] = "true"
1662
+ previous.add_previous_sibling(anchor)
1663
+ end
1664
+ parent.remove
1665
+ else
1666
+ parent["id"] = identifier
1667
+ end
1668
+ end
1669
+ marker.remove
1670
+ end
1671
+ end
1672
+
1673
+ def annotate_task_states(note, fragment)
1674
+ lines = note.scanner.markdown.lines
1675
+ fragment.css("li.task-list-item[data-sourcepos]").each do |item|
1676
+ position = parse_sourcepos_value(item["data-sourcepos"])
1677
+ line = lines.fetch(position.fetch(:start_line) - 1, "")
1678
+ source = line[(position.fetch(:start_column) - 1)..].to_s
1679
+ match = source.match(/\A(?:[-+*]|\d+[.)])\s+\[([^\]\r\n])\](?=\s|$)/)
1680
+ next unless match
1681
+
1682
+ state = match[1]
1683
+ input = item.at_css("input.task-list-item-checkbox")
1684
+ next unless input
1685
+ item["data-task"] = state
1686
+ input["data-task"] = state
1687
+ input["aria-label"] = "Task state: #{task_state_label(state)}"
1688
+ input.remove_attribute("checked") unless state.match?(/\A[xX]\z/)
1689
+ end
1690
+ end
1691
+
1692
+ def task_state_label(state)
1693
+ {
1694
+ " " => "open",
1695
+ "x" => "completed",
1696
+ "X" => "completed",
1697
+ "?" => "question",
1698
+ "/" => "in progress",
1699
+ "-" => "cancelled"
1700
+ }.fetch(state, state)
1701
+ end
1702
+
1703
+ def transform_callouts(fragment)
1704
+ fragment.css("blockquote[data-sourcepos]").to_a.reverse_each do |blockquote|
1705
+ first = blockquote.at_css("p")
1706
+ next unless first
1707
+
1708
+ text_node = first.xpath(".//text()").first
1709
+ next unless text_node
1710
+ match = text_node.text.match(/\A\s*\[!([a-z0-9_-]+)\]([+-])?\s*([^\n]*)/i)
1711
+ next unless match
1712
+
1713
+ type = match[1].downcase.gsub(/[^a-z0-9_-]/, "")
1714
+ fold = match[2]
1715
+ title = match[3].to_s.strip
1716
+ title = type.tr("-_", " ").split.map(&:capitalize).join(" ") if title.empty?
1717
+ text_node.content = text_node.text.sub(match[0], "").sub(/\A\s+/, "")
1718
+
1719
+ wrapper = Nokogiri::XML::Node.new(fold ? "details" : "aside", blockquote.document)
1720
+ wrapper["class"] = "website-callout website-callout--#{type} callout"
1721
+ wrapper["data-callout"] = type
1722
+ wrapper["role"] = "note" unless fold
1723
+ wrapper["open"] = "open" if fold == "+"
1724
+ transfer_replacement_identity(blockquote, wrapper)
1725
+ header = Nokogiri::XML::Node.new(fold ? "summary" : "header", blockquote.document)
1726
+ header["class"] = "website-callout__title callout__title"
1727
+ header.content = title
1728
+ wrapper.add_child(header)
1729
+ content = Nokogiri::XML::Node.new("div", blockquote.document)
1730
+ content["class"] = "website-callout__content callout__content"
1731
+ blockquote.children.to_a.each { |child| content.add_child(child.unlink) }
1732
+ content.css("p").first.remove if content.css("p").first&.text.to_s.strip.empty?
1733
+ first_paragraph = content.css("p").first
1734
+ first_paragraph.children.first.remove if first_paragraph&.children&.first&.name == "br"
1735
+ wrapper.add_child(content) unless content.children.empty?
1736
+ blockquote.replace(wrapper)
1737
+ end
1738
+ end
1739
+
1740
+ def annotate_code_blocks(note, fragment)
1741
+ note.document.select { |node| node.type == :code_block }.each_with_index do |source, index|
1742
+ marker = code_marker(index)
1743
+ pre = fragment.css("pre").find { |candidate| candidate.at_css("code")&.text&.start_with?(marker) }
1744
+ next unless pre
1745
+ remove_text_prefix(pre.at_css("code"), "#{marker}\n")
1746
+ language = source.fence_info.to_s.split.first.to_s.downcase.gsub(/[^a-z0-9_+-]/, "")
1747
+ next if language.empty?
1748
+
1749
+ code = pre.at_css("code")
1750
+ code["class"] = [code["class"], "language-#{language}"].compact.join(" ") if code
1751
+ pre["lang"] = language
1752
+ pre["data-website-mermaid"] = "true" if language == "mermaid"
1753
+ end
1754
+ end
1755
+
1756
+ def remove_text_prefix(node, prefix)
1757
+ remaining = prefix
1758
+ node.xpath(".//text()").each do |text|
1759
+ break if remaining.empty?
1760
+
1761
+ length = [text.text.length, remaining.length].min
1762
+ return false unless text.text[0, length] == remaining[0, length]
1763
+
1764
+ text.content = text.text[length..].to_s
1765
+ remaining = remaining[length..].to_s
1766
+ end
1767
+ remaining.empty?
1768
+ end
1769
+
1770
+ def annotate_code_markers(note)
1771
+ note.document.select { |node| node.type == :code_block }.each_with_index do |source, index|
1772
+ source.string_content = "#{code_marker(index)}\n#{source.string_content}"
1773
+ end
1774
+ end
1775
+
1776
+ def code_marker(index)
1777
+ "JEKYLL_OBSIDIAN_CODE_#{index}_START"
1778
+ end
1779
+
1780
+ def build_published_site_model
1781
+ relations = @relations.sort_by do |relation|
1782
+ [relation.source_id, relation.target_id, relation.kind.to_s, relation.fragment.to_s, span_key(relation.source_span)]
1783
+ end
1784
+ backlinks = relation_index(:link)
1785
+ embedded_by = relation_index(:embed)
1786
+ direct = @relations.group_by(&:source_id)
1787
+
1788
+ notes = @notes.values.sort_by(&:id).map do |note|
1789
+ context = TransclusionContext.new(host_id: note.id, sequence: 0, instances: 0, bytes: 0)
1790
+ content = render_with_transclusions(note.id, [], context, depth: 0).to_html
1791
+ assert_block_anchors_rendered(note, content)
1792
+ PublishedNote.new(
1793
+ id: note.id,
1794
+ title: note.title,
1795
+ route: note.route,
1796
+ content: content,
1797
+ properties: note.properties,
1798
+ markdown_source: PublishedMarkdown.content(
1799
+ title: note.title,
1800
+ body: note.body,
1801
+ has_h1: note.has_h1
1802
+ ),
1803
+ authored_text: note.authored_text,
1804
+ preview: note.preview,
1805
+ outline: note.outline,
1806
+ updated: note.updated,
1807
+ created: note.created,
1808
+ content_type: note.content_type,
1809
+ published_at: note.published_at,
1810
+ nav_order: note.nav_order,
1811
+ nav_exclude: note.nav_exclude,
1812
+ directory_index: directory_index_note?(note.id),
1813
+ has_h1: note.has_h1,
1814
+ feature_flags: note.feature_flags,
1815
+ content_security: content_security_needs(content),
1816
+ image_url: published_image_url(note),
1817
+ source_links: published_source_links(note),
1818
+ topics: published_topics(note),
1819
+ related: published_related_cards(note),
1820
+ links: relation_cards(direct.fetch(note.id, []).select { |item| item.kind == :link && item.property != "related" }),
1821
+ backlinks: relation_cards(backlinks.fetch(note.id, []), source: true),
1822
+ embedded_by: relation_cards(embedded_by.fetch(note.id, []), source: true)
1823
+ )
1824
+ end
1825
+ graph_edges = graph_edges_for(relations)
1826
+ PublishedSiteModel.new(
1827
+ notes: notes,
1828
+ notes_by_id: notes.to_h { |note| [note.id, note] },
1829
+ directory_index_paths: @directory_index_paths,
1830
+ relations: relations,
1831
+ graph_edges: graph_edges,
1832
+ graph_degrees: graph_degrees_for(notes, graph_edges)
1833
+ )
1834
+ end
1835
+
1836
+ def content_security_needs(content)
1837
+ fragment = Nokogiri::HTML5.fragment(content)
1838
+ media_sources = fragment.css("video[src], audio[src], video source[src], audio source[src]")
1839
+ .filter_map { |node| ExternalMedia.https_origin(node["src"]) }
1840
+ .uniq
1841
+ .sort
1842
+ frame_sources = fragment.css("iframe[src]")
1843
+ .filter_map { |node| ExternalMedia.https_origin(node["src"]) }
1844
+ .uniq
1845
+ .sort
1846
+ tweet = !fragment.at_css("[data-website-tweet]").nil?
1847
+ ContentSecurityNeeds.new(
1848
+ media_sources: media_sources,
1849
+ frame_sources: (frame_sources + (tweet ? ["https://platform.twitter.com"] : [])).uniq.sort,
1850
+ script_sources: tweet ? ["https://platform.twitter.com"] : [],
1851
+ connect_sources: []
1852
+ )
1853
+ end
1854
+
1855
+ def published_topics(note)
1856
+ note.topics.filter_map do |topic|
1857
+ occurrence = topic.occurrence
1858
+ unless occurrence
1859
+ next({ "kind" => topic.kind, "name" => topic.name })
1860
+ end
1861
+
1862
+ target = @notes[occurrence.target_id]
1863
+ name = occurrence.display || target&.title || occurrence.raw_target
1864
+ value = { "kind" => topic.kind, "name" => name }
1865
+ if target && !occurrence.unresolved
1866
+ fragment = occurrence.anchor_id || occurrence.fragment
1867
+ value["url"] = "#{@url_builder.href(target.route)}#{fragment && !fragment.empty? ? "##{fragment}" : ""}"
1868
+ end
1869
+ value
1870
+ end.uniq
1871
+ end
1872
+
1873
+ def render_with_transclusions(note_id, stack, context, depth:, raw_fragment: nil, resolved_anchor_id: nil)
1874
+ fragment = cached_transclusion_selection(note_id, raw_fragment, resolved_anchor_id)
1875
+ unless consume_transclusion_bytes(context, fragment.to_html.bytesize, note_id)
1876
+ return transclusion_limit_fragment(fragment.document, "expanded HTML exceeds #{MAX_TRANSCLUSION_BYTES} bytes")
1877
+ end
1878
+
1879
+ fragment.css("website-embed").to_a.each do |placeholder|
1880
+ target_id = placeholder["data-source-id"]
1881
+ if stack.include?(target_id) || target_id == note_id
1882
+ replacement = Nokogiri::XML::Node.new("span", fragment.document)
1883
+ replacement["class"] = "website-embed website-embed--cycle"
1884
+ replacement.content = "Embed cycle"
1885
+ transfer_replacement_identity(placeholder, replacement)
1886
+ placeholder.replace(replacement)
1887
+ next
1888
+ end
1889
+
1890
+ unless consume_transclusion_instance(context, depth + 1, note_id)
1891
+ placeholder.replace(transclusion_limit_node(fragment.document, "embed expansion limit reached"))
1892
+ next
1893
+ end
1894
+
1895
+ selected = render_with_transclusions(
1896
+ target_id,
1897
+ stack + [note_id],
1898
+ context,
1899
+ depth: depth + 1,
1900
+ raw_fragment: placeholder["data-fragment"],
1901
+ resolved_anchor_id: placeholder["data-anchor-id"]
1902
+ )
1903
+ context.sequence += 1
1904
+ prefix = "embed-#{@url_builder.slug(context.host_id)}-#{context.sequence}-"
1905
+ rewrite_fragment_ids(selected, prefix)
1906
+
1907
+ wrapper = Nokogiri::XML::Node.new("section", fragment.document)
1908
+ wrapper["class"] = "website-transclusion website-embed"
1909
+ wrapper["data-source-id"] = target_id
1910
+ transfer_replacement_identity(placeholder, wrapper)
1911
+ source = Nokogiri::XML::Node.new("a", fragment.document)
1912
+ source["class"] = "website-transclusion__source website-embed__source"
1913
+ source["href"] = placeholder["data-href"]
1914
+ source.content = "From #{@notes.fetch(target_id).title}"
1915
+ wrapper.add_child(source)
1916
+ embedded_content = Nokogiri::XML::Node.new("div", fragment.document)
1917
+ embedded_content["class"] = "website-transclusion__content website-embed__content"
1918
+ selected.children.to_a.each { |child| embedded_content.add_child(child.unlink) }
1919
+ wrapper.add_child(embedded_content)
1920
+ placeholder.replace(wrapper)
1921
+ end
1922
+ fragment
1923
+ end
1924
+
1925
+ def cached_transclusion_selection(note_id, raw_fragment, resolved_anchor_id)
1926
+ key = [note_id, raw_fragment.to_s, resolved_anchor_id.to_s]
1927
+ cached = @transclusion_selection_cache[key]
1928
+ return Nokogiri::HTML5.fragment(cached) if cached
1929
+
1930
+ fragment = @notes.fetch(note_id).base_fragment.dup
1931
+ selected = select_transclusion_fragment(fragment, raw_fragment, resolved_anchor_id)
1932
+ serialized = selected.to_html.freeze
1933
+ @transclusion_selection_cache[key] = serialized
1934
+ Nokogiri::HTML5.fragment(serialized)
1935
+ end
1936
+
1937
+ def consume_transclusion_instance(context, depth, path)
1938
+ return transclusion_limit(context, "embed depth exceeds #{MAX_TRANSCLUSION_DEPTH}", path) if depth > MAX_TRANSCLUSION_DEPTH
1939
+ return transclusion_limit(context, "embed instances exceed #{MAX_TRANSCLUSION_INSTANCES}", path) if context.instances >= MAX_TRANSCLUSION_INSTANCES
1940
+
1941
+ context.instances += 1
1942
+ true
1943
+ end
1944
+
1945
+ def consume_transclusion_bytes(context, amount, path)
1946
+ return transclusion_limit(context, "expanded HTML exceeds #{MAX_TRANSCLUSION_BYTES} bytes", path) if context.bytes + amount > MAX_TRANSCLUSION_BYTES
1947
+
1948
+ context.bytes += amount
1949
+ true
1950
+ end
1951
+
1952
+ def transclusion_limit(_context, message, path)
1953
+ error_or_warning("embed_budget_exceeded", message, path, nil, fatal: production?)
1954
+ false
1955
+ end
1956
+
1957
+ def transclusion_limit_fragment(document, message)
1958
+ fragment = Nokogiri::HTML5.fragment("")
1959
+ fragment.add_child(transclusion_limit_node(document, message))
1960
+ fragment
1961
+ end
1962
+
1963
+ def transclusion_limit_node(document, message)
1964
+ node = Nokogiri::XML::Node.new("span", document)
1965
+ node["class"] = "website-embed website-embed--limited"
1966
+ node["role"] = "status"
1967
+ node.content = message
1968
+ node
1969
+ end
1970
+
1971
+ def transfer_replacement_identity(source, replacement)
1972
+ replacement["id"] = source["id"] if source["id"]
1973
+ end
1974
+
1975
+ def assert_block_anchors_rendered(note, content)
1976
+ document = Nokogiri::HTML5.fragment(content)
1977
+ counts = document.css("[id]").each_with_object(Hash.new(0)) { |node, memo| memo[node["id"]] += 1 }
1978
+ note.anchors.select { |anchor| anchor.kind == :block }.each do |anchor|
1979
+ next if counts[anchor.id] == 1
1980
+
1981
+ error(
1982
+ "block_anchor_realization",
1983
+ "block ID #{anchor.id.inspect} rendered #{counts[anchor.id]} matching DOM targets instead of one",
1984
+ note.id
1985
+ )
1986
+ end
1987
+ end
1988
+
1989
+ def select_transclusion_fragment(fragment, raw_fragment, resolved_anchor_id = nil)
1990
+ return fragment unless raw_fragment && !raw_fragment.empty?
1991
+
1992
+ identifier = resolved_anchor_id.to_s
1993
+ identifier = raw_fragment.start_with?("^") ? raw_fragment.delete_prefix("^") : @url_builder.slug(raw_fragment.split("#").last) if identifier.empty?
1994
+ # Match the attribute value in Ruby instead of interpolating it into a
1995
+ # CSS ID selector. CSS selectors need a special escape for leading
1996
+ # digits, while HTML fragment IDs do not.
1997
+ target = fragment.css("[id]").find { |candidate| candidate["id"] == identifier }
1998
+ return empty_embed_fragment(fragment, raw_fragment) unless target
1999
+ if target["class"].to_s.split.include?("website-block-anchor") && target.next_element
2000
+ return fragment_for_node(target.next_element)
2001
+ end
2002
+ return fragment_for_node(target) unless target.name.match?(/\Ah[1-6]\z/)
2003
+
2004
+ level = target.name.delete_prefix("h").to_i
2005
+ selected = Nokogiri::HTML5.fragment("")
2006
+ cursor = target
2007
+ while cursor
2008
+ break if cursor != target && cursor.element? && cursor.name.match?(/\Ah[1-6]\z/) && cursor.name.delete_prefix("h").to_i <= level
2009
+
2010
+ following = cursor.next_sibling
2011
+ selected.add_child(cursor.unlink)
2012
+ cursor = following
2013
+ end
2014
+ selected
2015
+ end
2016
+
2017
+ def fragment_for_node(node)
2018
+ selected = Nokogiri::HTML5.fragment("")
2019
+ if node.name == "li" && %w[ul ol].include?(node.parent&.name)
2020
+ list = Nokogiri::XML::Node.new(node.parent.name, node.document)
2021
+ node.parent.attribute_nodes.each { |attribute| list[attribute.name] = attribute.value }
2022
+ list.add_child(node.unlink)
2023
+ selected.add_child(list)
2024
+ else
2025
+ selected.add_child(node.unlink)
2026
+ end
2027
+ selected
2028
+ end
2029
+
2030
+ def empty_embed_fragment(fragment, label)
2031
+ selected = Nokogiri::HTML5.fragment("")
2032
+ span = Nokogiri::XML::Node.new("span", fragment.document)
2033
+ span["class"] = "website-embed website-embed--unresolved"
2034
+ span.content = "Missing fragment: #{label}"
2035
+ selected.add_child(span)
2036
+ selected
2037
+ end
2038
+
2039
+ def rewrite_fragment_ids(fragment, prefix)
2040
+ mapping = {}
2041
+ fragment.css("[id]").each do |node|
2042
+ old = node["id"]
2043
+ mapping[old] = "#{prefix}#{old}"
2044
+ node["id"] = mapping[old]
2045
+ end
2046
+ fragment.css("*").select { |node| node["href"] || node["xlink:href"] }.each do |node|
2047
+ %w[href xlink:href].each do |attribute|
2048
+ next unless node[attribute]&.start_with?("#")
2049
+
2050
+ old = node[attribute].delete_prefix("#")
2051
+ node[attribute] = "##{mapping.fetch(old, "#{prefix}#{old}")}"
2052
+ end
2053
+ end
2054
+ %w[for list form aria-activedescendant aria-details aria-errormessage].each do |attribute|
2055
+ fragment.css("[#{attribute}]").each do |node|
2056
+ old = node[attribute]
2057
+ node[attribute] = mapping.fetch(old, "#{prefix}#{old}")
2058
+ end
2059
+ end
2060
+ %w[aria-labelledby aria-describedby aria-controls aria-owns headers itemref].each do |attribute|
2061
+ fragment.css("[#{attribute}]").each do |node|
2062
+ node[attribute] = node[attribute].split.map { |old| mapping.fetch(old, "#{prefix}#{old}") }.join(" ")
2063
+ end
2064
+ end
2065
+ %w[style clip-path fill filter mask marker-start marker-mid marker-end stroke].each do |attribute|
2066
+ fragment.css("[#{attribute}]").each do |node|
2067
+ node[attribute] = node[attribute].gsub(/url\(\s*(['"]?)#([^)'"\s]+)\1\s*\)/) do
2068
+ quote = Regexp.last_match(1)
2069
+ old = Regexp.last_match(2)
2070
+ "url(#{quote}##{mapping.fetch(old, "#{prefix}#{old}")}#{quote})"
2071
+ end
2072
+ end
2073
+ end
2074
+ end
2075
+
2076
+ def published_image_url(note)
2077
+ path = @image_paths[note.id]
2078
+ return nil unless path
2079
+
2080
+ route = @html_routes_by_source.fetch(path) { @url_builder.attachment_route(path) }
2081
+ @url_builder.absolute_url(route) || @url_builder.href(route)
2082
+ end
2083
+
2084
+ def repository_links(path)
2085
+ repository = @config.repository.to_s
2086
+ return {} unless repository.match?(/\A[\w.-]+\/[\w.-]+\z/)
2087
+
2088
+ branch = URI.encode_uri_component(@config.edit_branch.to_s.empty? ? "main" : @config.edit_branch.to_s)
2089
+ source = [@config.source.to_s, path].reject(&:empty?).map { |part| part.split("/").map { |segment| URI.encode_uri_component(segment) }.join("/") }.join("/")
2090
+ base = "https://github.com/#{repository}"
2091
+ {
2092
+ "edit" => "#{base}/edit/#{branch}/#{source}",
2093
+ "history" => "#{base}/commits/#{branch}/#{source}",
2094
+ "source" => "#{base}/blob/#{branch}/#{source}",
2095
+ "issue" => "#{base}/issues/new?title=#{URI.encode_uri_component("Issue with #{path}")}"
2096
+ }
2097
+ end
2098
+
2099
+ def published_source_links(note)
2100
+ links = repository_links(note.id)
2101
+ if note.external_document
2102
+ links = links.merge(
2103
+ "imported" => note.external_document.source_url,
2104
+ "repository" => GitHubMarkdown.repository_url(note.external_document.repository)
2105
+ )
2106
+ end
2107
+ links
2108
+ end
2109
+
2110
+ def build_generated_files(pages, model, theme_output)
2111
+ artifacts = theme_output.artifacts.filter_map do |artifact|
2112
+ case artifact
2113
+ when "catalog"
2114
+ json_file("/assets/website/catalog.v1.json", catalog_payload(model))
2115
+ when "graph"
2116
+ json_file("/assets/website/graph.v1.json", graph_payload(model))
2117
+ when "search"
2118
+ json_file("/assets/website/search.v1.json", search_payload(model))
2119
+ when "sitemap"
2120
+ GeneratedFile.new(route: "/sitemap.xml", content: sitemap_xml(pages), media_type: "application/xml")
2121
+ when "feed"
2122
+ candidates = theme_output.feed_note_ids.map { |id| model.notes_by_id.fetch(id) }
2123
+ feed = feed_xml(candidates)
2124
+ GeneratedFile.new(route: "/feed.xml", content: feed, media_type: "application/atom+xml") if feed
2125
+ else
2126
+ raise ArgumentError, "unknown generated artifact #{artifact.inspect}"
2127
+ end
2128
+ end
2129
+ markdown = model.notes.map do |note|
2130
+ GeneratedFile.new(
2131
+ route: PublishedMarkdown.route(note.route),
2132
+ content: note.markdown_source,
2133
+ media_type: "text/markdown"
2134
+ )
2135
+ end
2136
+ artifacts + markdown
2137
+ end
2138
+
2139
+ def catalog_payload(model)
2140
+ {
2141
+ "schema_version" => 1,
2142
+ "notes" => model.notes.map do |note|
2143
+ {
2144
+ "id" => note.id,
2145
+ "title" => note.title,
2146
+ "url" => @url_builder.href(note.route),
2147
+ "aliases" => Array(note.properties["aliases"]),
2148
+ "tags" => Array(note.properties["tags"]),
2149
+ "description" => note.properties["description"],
2150
+ "preview" => note.preview,
2151
+ "updated" => note.updated,
2152
+ "content_type" => note.content_type,
2153
+ "published_at" => note.published_at
2154
+ }
2155
+ end
2156
+ }
2157
+ end
2158
+
2159
+ def graph_payload(model)
2160
+ {
2161
+ "schema_version" => 1,
2162
+ "nodes" => model.notes.map do |note|
2163
+ {
2164
+ "id" => note.id,
2165
+ "title" => note.title,
2166
+ "url" => @url_builder.href(note.route),
2167
+ "tags" => Array(note.properties["tags"]),
2168
+ "degree" => model.graph_degrees.fetch(note.id)
2169
+ }
2170
+ end,
2171
+ "edges" => model.graph_edges
2172
+ }
2173
+ end
2174
+
2175
+ def search_payload(model)
2176
+ {
2177
+ "schema_version" => 1,
2178
+ "documents" => model.notes.map do |note|
2179
+ {
2180
+ "id" => note.id,
2181
+ "title" => note.title,
2182
+ "url" => @url_builder.href(note.route),
2183
+ "aliases" => Array(note.properties["aliases"]),
2184
+ "tags" => Array(note.properties["tags"]),
2185
+ "text" => note.authored_text
2186
+ }
2187
+ end
2188
+ }
2189
+ end
2190
+
2191
+ def graph_edges_for(relations)
2192
+ counts = Hash.new(0)
2193
+ relations.each { |relation| counts[[relation.source_id, relation.target_id, relation.kind.to_s]] += 1 }
2194
+ counts.keys.sort.map do |source, target, kind|
2195
+ { "source" => source, "target" => target, "kind" => kind, "count" => counts[[source, target, kind]] }
2196
+ end
2197
+ end
2198
+
2199
+ def graph_degrees_for(notes, edges)
2200
+ neighbours = notes.to_h { |note| [note.id, {}] }
2201
+ edges.each do |edge|
2202
+ source = edge.fetch("source")
2203
+ target = edge.fetch("target")
2204
+ neighbours.fetch(source)[target] = true
2205
+ neighbours.fetch(target)[source] = true
2206
+ end
2207
+ neighbours.transform_values(&:length)
2208
+ end
2209
+
2210
+ def json_file(route, payload)
2211
+ GeneratedFile.new(route: route, content: "#{JSON.generate(payload)}\n", media_type: "application/json")
2212
+ end
2213
+
2214
+ def sitemap_xml(pages)
2215
+ urls = pages.reject { |page| %w[404 redirect].include?(page.data.dig("website", "kind")) }
2216
+ .map(&:route).sort
2217
+ body = urls.map { |route| " <url><loc>#{h(@url_builder.absolute_url(route))}</loc></url>" }.join("\n")
2218
+ %(<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n#{body}\n</urlset>\n)
2219
+ end
2220
+
2221
+ def feed_xml(candidates)
2222
+ if candidates.empty?
2223
+ warning("feed_skipped_empty", "feed skipped because there are no public notes")
2224
+ return nil
2225
+ end
2226
+
2227
+ missing = candidates.select { |note| feed_timestamp(note).nil? }
2228
+ unless missing.empty?
2229
+ warning("feed_omitted_missing_time", "feed omitted #{missing.length} public note(s) without an explicit update time or post publication time")
2230
+ end
2231
+
2232
+ candidates -= missing
2233
+ return nil if candidates.empty?
2234
+
2235
+ notes = candidates.sort_by { |note| [chronology_key(feed_timestamp(note)), note.id] }.reverse
2236
+ updated = feed_timestamp(notes.first)
2237
+ entries = notes.map do |note|
2238
+ published = note.published_at ? "\n <published>#{h(note.published_at)}</published>" : ""
2239
+ <<~XML.chomp
2240
+ <entry>
2241
+ <id>#{h(@url_builder.absolute_url(note.route))}</id>
2242
+ <title>#{h(note.title)}</title>
2243
+ <link href="#{h(@url_builder.absolute_url(note.route))}" />
2244
+ <updated>#{h(feed_timestamp(note))}</updated>#{published}
2245
+ <summary>#{h(note.preview)}</summary>
2246
+ </entry>
2247
+ XML
2248
+ end.join("\n")
2249
+ <<~XML
2250
+ <?xml version="1.0" encoding="UTF-8"?>
2251
+ <feed xmlns="http://www.w3.org/2005/Atom">
2252
+ <id>#{h(@url_builder.absolute_url("/"))}</id>
2253
+ <title>#{h(@config.title.to_s)}</title>
2254
+ <updated>#{h(updated)}</updated>
2255
+ <link href="#{h(@url_builder.absolute_url("/feed.xml"))}" rel="self" />
2256
+ #{entries.lines.map { |line| " #{line}" }.join}</feed>
2257
+ XML
2258
+ end
2259
+
2260
+ def feed_timestamp(note)
2261
+ note.updated || (note.content_type == "post" ? note.published_at : nil)
2262
+ end
2263
+
2264
+ def chronology_key(value)
2265
+ [0, DateTime.iso8601(value.to_s).new_offset(0).ajd]
2266
+ rescue Date::Error
2267
+ [1, value.to_s]
2268
+ end
2269
+
2270
+ def build_projected_files
2271
+ @projected_attachment_paths.to_a.sort.map do |path|
2272
+ entry = @attachments.fetch(path)
2273
+ ProjectedFile.new(
2274
+ source_path: path,
2275
+ route: @url_builder.attachment_route(path),
2276
+ media_type: entry.media_type,
2277
+ size: entry.size,
2278
+ device: entry.device,
2279
+ inode: entry.inode,
2280
+ mtime_ns: entry.mtime_ns
2281
+ )
2282
+ end
2283
+ end
2284
+
2285
+ def preflight_routes(pages, generated_files, projected_files, reserved_namespaces)
2286
+ registry = DestinationRegistry.new
2287
+ (pages + generated_files + projected_files).each do |output|
2288
+ destination = destination_key(output)
2289
+ conflict = registry.add(destination, output)
2290
+ if conflict
2291
+ readme_index = [output, conflict].find { |candidate| readme_directory_index_output?(candidate) }
2292
+ if readme_index
2293
+ note_id = readme_index.data.dig("website", "id")
2294
+ conflicting_output = readme_index.equal?(output) ? conflict : output
2295
+ error(
2296
+ "route_collision",
2297
+ "README.md becomes its folder index route and collides with #{conflicting_output.route}; set publish: false or rename it",
2298
+ note_id
2299
+ )
2300
+ else
2301
+ error(
2302
+ "route_collision",
2303
+ "#{output_owner(output)} at #{output.route} collides with #{output_owner(conflict)} at #{conflict.route}",
2304
+ output.is_a?(ProjectedFile) ? output.source_path : output.route
2305
+ )
2306
+ end
2307
+ end
2308
+ end
2309
+
2310
+ namespace_keys = reserved_namespaces.map do |namespace|
2311
+ @url_builder.collision_key(namespace).delete_suffix("/")
2312
+ end
2313
+ @notes.each_value do |note|
2314
+ page_key = @url_builder.collision_key(note.route).delete_suffix("/")
2315
+ next unless namespace_keys.any? { |namespace| page_key == namespace || page_key.start_with?("#{namespace}/") }
2316
+
2317
+ error("route_collision", "note route collides with a generated system namespace", note.route)
2318
+ end
2319
+
2320
+ return unless production? && @notes.any?
2321
+ index_count = pages.count { |page| page.route == "/" }
2322
+ error("invalid_index_count", "production must generate exactly one /index.html", nil) unless index_count == 1
2323
+ end
2324
+
2325
+ def readme_directory_index_output?(output)
2326
+ return false unless output.is_a?(PageOutput)
2327
+
2328
+ website = output.data["website"]
2329
+ note_id = website.is_a?(Hash) && website["id"]
2330
+ website.is_a?(Hash) && website["directory_index"] == true &&
2331
+ File.basename(note_id.to_s) == DirectoryIndexes::FALLBACK_BASENAME
2332
+ end
2333
+
2334
+ def output_owner(output)
2335
+ if output.is_a?(ProjectedFile)
2336
+ "projected source #{output.source_path}"
2337
+ elsif output.is_a?(PageOutput)
2338
+ note_id = output.data.dig("website", "id")
2339
+ note_id ? "note #{note_id}" : "theme page"
2340
+ else
2341
+ "generated artifact"
2342
+ end
2343
+ end
2344
+
2345
+ def destination_key(output)
2346
+ key = @url_builder.collision_key(output.route)
2347
+ if output.is_a?(PageOutput)
2348
+ key.end_with?("/") ? "#{key}index.html" : key
2349
+ else
2350
+ output.route.end_with?("/") ? key : key.delete_suffix("/")
2351
+ end
2352
+ end
2353
+
2354
+ def resolve_note_path(source_id, raw_target)
2355
+ decoded = safe_decode(raw_target).unicode_normalize(:nfc).tr("\\", "/")
2356
+ return [nil, false] if decoded.empty?
2357
+ return [nil, false] if attachment_extension?(decoded)
2358
+
2359
+ rooted = decoded.delete_prefix("/")
2360
+ root_candidate = ensure_md(rooted)
2361
+ return [root_candidate, false] if @notes.key?(root_candidate)
2362
+
2363
+ if decoded.start_with?("./", "../")
2364
+ relative = clean_relative(File.dirname(source_id), decoded)
2365
+ return [nil, false] unless relative
2366
+ candidate = ensure_md(relative)
2367
+ return [candidate, false] if @notes.key?(candidate)
2368
+ return [nil, false]
2369
+ end
2370
+
2371
+ relative = clean_relative(File.dirname(source_id), decoded)
2372
+ if relative
2373
+ candidate = ensure_md(relative)
2374
+ return [candidate, false] if @notes.key?(candidate)
2375
+ end
2376
+
2377
+ basename = File.basename(rooted, NOTE_EXTENSION).unicode_normalize(:nfc).downcase(:fold)
2378
+ candidates = @basename_index.fetch(basename, [])
2379
+ return [candidates.first, false] if candidates.one?
2380
+ return [nil, true] if candidates.length > 1
2381
+
2382
+ [nil, false]
2383
+ end
2384
+
2385
+ def resolve_attachment_path(source_id, raw_target)
2386
+ decoded = safe_decode(raw_target).unicode_normalize(:nfc).tr("\\", "/").delete_prefix("/")
2387
+ return [nil, false] if decoded.empty?
2388
+
2389
+ candidates = []
2390
+ candidates << decoded
2391
+ relative = clean_relative(File.dirname(source_id), decoded)
2392
+ candidates << relative if relative && relative != decoded
2393
+ candidates.each { |candidate| return [candidate, false] if @attachments.key?(candidate) }
2394
+
2395
+ return [nil, false] if decoded.include?("/")
2396
+
2397
+ folded = File.basename(decoded).unicode_normalize(:nfc).downcase(:fold)
2398
+ matches = @attachment_basename_index.fetch(folded, [])
2399
+ return [matches.first, false] if matches.one?
2400
+
2401
+ [nil, matches.length > 1]
2402
+ end
2403
+
2404
+ def split_target(raw, kind)
2405
+ text = raw.to_s.strip
2406
+ target_and_fragment, size_option = text.split("|", 2)
2407
+ target, fragment = target_and_fragment.split("#", 2)
2408
+ options = {}
2409
+ if size_option&.match?(/\A\d+(?:x\d+)?\z/)
2410
+ width, height = size_option.split("x", 2)
2411
+ options["width"] = width.to_i
2412
+ options["height"] = height.to_i if height
2413
+ end
2414
+
2415
+ if File.extname(target.to_s).downcase == ".pdf" && fragment
2416
+ fragment.split("&").each do |part|
2417
+ key, value = part.split("=", 2)
2418
+ options[key] = value.to_i if %w[page height].include?(key) && value&.match?(/\A\d+\z/)
2419
+ end
2420
+ fragment = nil if options.any?
2421
+ end
2422
+ [target.to_s.strip, fragment&.strip, options]
2423
+ end
2424
+
2425
+ def external_url?(raw_url, path, span, media: false)
2426
+ text = raw_url.to_s.strip
2427
+ scheme = text[/\A([a-z][a-z0-9+.-]*):/i, 1]&.downcase
2428
+ return false unless scheme
2429
+
2430
+ if DANGEROUS_SCHEMES.include?(scheme) || !EXTERNAL_SCHEMES.include?(scheme)
2431
+ error("unsafe_url", "URL scheme is not allowed", path, span)
2432
+ elsif media && scheme != "https"
2433
+ error("unsafe_url", "author media URLs must use HTTPS", path, span)
2434
+ end
2435
+ true
2436
+ end
2437
+
2438
+ def attachment_extension?(target)
2439
+ extension = File.extname(target).downcase
2440
+ !extension.empty? && extension != NOTE_EXTENSION
2441
+ end
2442
+
2443
+ def deterministic_created(note)
2444
+ note.properties["created"] || note.entry.first_committed_at
2445
+ end
2446
+
2447
+ def relation_index(kind)
2448
+ index = Hash.new { |hash, key| hash[key] = [] }
2449
+ @relations.each { |relation| index[relation.target_id] << relation if relation.kind == kind }
2450
+ index
2451
+ end
2452
+
2453
+ def relation_cards(relations, source: false)
2454
+ ids = relations.map { |relation| source ? relation.source_id : relation.target_id }.uniq.sort
2455
+ ids.map do |id|
2456
+ note = @notes.fetch(id)
2457
+ { "id" => id, "title" => note.title, "url" => @url_builder.href(note.route) }
2458
+ end
2459
+ end
2460
+
2461
+ def published_related_cards(note)
2462
+ seen = {}
2463
+ note.occurrences.filter_map do |occurrence|
2464
+ next unless occurrence.syntax == :frontmatter_property && occurrence.property == "related"
2465
+ next unless occurrence.resolved_type == :note && !occurrence.unresolved
2466
+ next if seen[occurrence.target_id]
2467
+
2468
+ seen[occurrence.target_id] = true
2469
+ target = @notes.fetch(occurrence.target_id)
2470
+ fragment = occurrence.anchor_id ? "^#{occurrence.anchor_id}" : occurrence.fragment
2471
+ {
2472
+ "id" => target.id,
2473
+ "title" => occurrence.display || target.title,
2474
+ "url" => @url_builder.href(target.route) + @url_builder.fragment(fragment)
2475
+ }
2476
+ end
2477
+ end
2478
+
2479
+ def token_url(index)
2480
+ "https://obsidian.invalid/ref/#{index}"
2481
+ end
2482
+
2483
+ def source_span(position)
2484
+ return nil unless position
2485
+ SourceSpan.new(
2486
+ start_line: position[:start_line],
2487
+ start_column: position[:start_column],
2488
+ end_line: position[:end_line],
2489
+ end_column: position[:end_column]
2490
+ )
2491
+ end
2492
+
2493
+ def sourcepos_value(position)
2494
+ "#{position.fetch(:start_line)}:#{position.fetch(:start_column)}-#{position.fetch(:end_line)}:#{position.fetch(:end_column)}"
2495
+ end
2496
+
2497
+ def parse_sourcepos_value(value)
2498
+ match = value.to_s.match(/\A(\d+):(\d+)-(\d+):(\d+)\z/)
2499
+ return { start_line: 0, start_column: 0, end_line: 0, end_column: 0 } unless match
2500
+
2501
+ {
2502
+ start_line: match[1].to_i,
2503
+ start_column: match[2].to_i,
2504
+ end_line: match[3].to_i,
2505
+ end_column: match[4].to_i
2506
+ }
2507
+ end
2508
+
2509
+ def span_key(span)
2510
+ return [0, 0, 0, 0] unless span
2511
+ [span.start_line, span.start_column, span.end_line, span.end_column]
2512
+ end
2513
+
2514
+ def first_h1(document)
2515
+ heading = document.find { |node| node.type == :heading && node.header_level == 1 }
2516
+ heading && plain_node_text(heading).strip
2517
+ end
2518
+
2519
+ def plain_node_text(node)
2520
+ node.walk.filter_map do |child|
2521
+ child.string_content if %i[text code].include?(child.type)
2522
+ rescue TypeError
2523
+ nil
2524
+ end.join
2525
+ end
2526
+
2527
+ def filename_title(path)
2528
+ File.basename(path, NOTE_EXTENSION).tr("-_", " ")
2529
+ end
2530
+
2531
+ def visible_text(fragment)
2532
+ copy = fragment.dup
2533
+ copy.css("script, style, template, website-embed, .website-transclusion__source").remove
2534
+ copy.text.gsub(/\s+/, " ").strip
2535
+ end
2536
+
2537
+ def truncate(text, limit)
2538
+ value = text.to_s.gsub(/\s+/, " ").strip
2539
+ return value if value.length <= limit
2540
+
2541
+ "#{value[0, limit - 1].rstrip}…"
2542
+ end
2543
+
2544
+ def image_node(document, occurrence, href)
2545
+ node = Nokogiri::XML::Node.new("img", document)
2546
+ node["src"] = href
2547
+ node["alt"] = occurrence.display.to_s
2548
+ node["loading"] = "lazy"
2549
+ node["decoding"] = "async"
2550
+ node["width"] = occurrence.options["width"].to_s if occurrence.options&.key?("width")
2551
+ node["height"] = occurrence.options["height"].to_s if occurrence.options&.key?("height")
2552
+ node
2553
+ end
2554
+
2555
+ def media_node(document, name, href, media_type)
2556
+ node = Nokogiri::XML::Node.new(name, document)
2557
+ node["controls"] = "controls"
2558
+ node["preload"] = "metadata"
2559
+ node["playsinline"] = "playsinline" if name == "video"
2560
+ source = Nokogiri::XML::Node.new("source", document)
2561
+ source["src"] = href
2562
+ source["type"] = media_type.to_s unless media_type.to_s.empty?
2563
+ node.add_child(source)
2564
+ node
2565
+ end
2566
+
2567
+ def external_video_node(document, descriptor, label)
2568
+ node = media_node(document, "video", descriptor.source_url, descriptor.media_type)
2569
+ node["class"] = "website-external-video"
2570
+ node["aria-label"] = label.to_s.strip unless label.to_s.strip.empty?
2571
+ fallback = Nokogiri::XML::Node.new("a", document)
2572
+ fallback["href"] = descriptor.fallback_url
2573
+ fallback["rel"] = "noopener noreferrer"
2574
+ fallback.content = label.to_s.strip.empty? ? "Open video" : label.to_s.strip
2575
+ node.add_child(fallback)
2576
+ node
2577
+ end
2578
+
2579
+ def external_player_node(document, descriptor, label)
2580
+ node = Nokogiri::XML::Node.new("iframe", document)
2581
+ node["class"] = "website-external-player website-external-player--#{descriptor.provider}"
2582
+ node["data-website-external-player"] = descriptor.provider.to_s
2583
+ node["src"] = descriptor.source_url
2584
+ node["title"] = label.to_s.strip.empty? ? (descriptor.title || external_player_title(descriptor.provider)) : label.to_s.strip
2585
+ node["loading"] = "lazy"
2586
+ node["referrerpolicy"] = "strict-origin-when-cross-origin"
2587
+ node["allow"] = descriptor.iframe_allow if descriptor.iframe_allow
2588
+ node["allowfullscreen"] = "allowfullscreen"
2589
+ node["width"] = descriptor.width.to_s if descriptor.width
2590
+ node["height"] = descriptor.height.to_s if descriptor.height
2591
+ node
2592
+ end
2593
+
2594
+ def external_web_frame_node(document, descriptor)
2595
+ wrapper = Nokogiri::XML::Node.new("figure", document)
2596
+ wrapper["class"] = "website-external-frame"
2597
+ frame = Nokogiri::XML::Node.new("iframe", document)
2598
+ frame["class"] = "website-external-frame__viewport"
2599
+ frame["data-website-external-frame"] = "web"
2600
+ frame["src"] = descriptor.source_url
2601
+ frame["title"] = descriptor.title
2602
+ frame["loading"] = "lazy"
2603
+ frame["referrerpolicy"] = "strict-origin-when-cross-origin"
2604
+ frame["sandbox"] = descriptor.iframe_sandbox
2605
+ frame["allowfullscreen"] = "allowfullscreen"
2606
+ frame["width"] = descriptor.width.to_s if descriptor.width
2607
+ frame["height"] = descriptor.height.to_s
2608
+ wrapper.add_child(frame)
2609
+ fallback = Nokogiri::XML::Node.new("a", document)
2610
+ fallback["class"] = "website-external-frame__fallback"
2611
+ fallback["href"] = descriptor.fallback_url
2612
+ fallback["target"] = "_blank"
2613
+ fallback["rel"] = "noopener noreferrer"
2614
+ fallback.content = "Open embedded page"
2615
+ wrapper.add_child(fallback)
2616
+ wrapper
2617
+ end
2618
+
2619
+ def tweet_node(document, descriptor)
2620
+ wrapper = Nokogiri::XML::Node.new("figure", document)
2621
+ wrapper["class"] = "website-tweet"
2622
+ wrapper["data-website-tweet"] = descriptor.identifier
2623
+ mount = Nokogiri::XML::Node.new("div", document)
2624
+ mount["class"] = "website-tweet__mount"
2625
+ mount["data-website-tweet-mount"] = ""
2626
+ wrapper.add_child(mount)
2627
+ fallback = Nokogiri::XML::Node.new("a", document)
2628
+ fallback["class"] = "website-tweet__fallback"
2629
+ fallback["data-website-tweet-fallback"] = ""
2630
+ fallback["href"] = descriptor.fallback_url
2631
+ fallback["target"] = "_blank"
2632
+ fallback["rel"] = "noopener noreferrer"
2633
+ fallback.content = "View post on X"
2634
+ wrapper.add_child(fallback)
2635
+ wrapper
2636
+ end
2637
+
2638
+ def external_player_title(provider)
2639
+ {
2640
+ youtube: "YouTube video player",
2641
+ bilibili: "Bilibili video player",
2642
+ vimeo: "Vimeo video player"
2643
+ }.fetch(provider, "Video player")
2644
+ end
2645
+
2646
+ def pdf_node(document, occurrence, href)
2647
+ data = href
2648
+ data = "#{data}#page=#{occurrence.options["page"]}" if occurrence.options&.key?("page")
2649
+ node = Nokogiri::XML::Node.new("object", document)
2650
+ node["data"] = data
2651
+ node["type"] = "application/pdf"
2652
+ node["height"] = occurrence.options.fetch("height", 640).to_s
2653
+ fallback = Nokogiri::XML::Node.new("a", document)
2654
+ fallback["href"] = href
2655
+ fallback.content = "Download PDF"
2656
+ node.add_child(fallback)
2657
+ node
2658
+ end
2659
+
2660
+ def download_card(document, path, href, media_type)
2661
+ node = Nokogiri::XML::Node.new("a", document)
2662
+ node["class"] = "website-download-card attachment-card"
2663
+ node["href"] = href
2664
+ node["download"] = ""
2665
+ title = Nokogiri::XML::Node.new("span", document)
2666
+ title["class"] = "website-download-card__title attachment-card__title"
2667
+ title.content = File.basename(path)
2668
+ meta = Nokogiri::XML::Node.new("span", document)
2669
+ meta["class"] = "website-download-card__meta attachment-card__type"
2670
+ meta.content = media_type.to_s.empty? ? "Download" : media_type.to_s
2671
+ node.add_child(title)
2672
+ node.add_child(meta)
2673
+ node
2674
+ end
2675
+
2676
+ def clean_relative(base, target)
2677
+ joined = base.empty? || base == "." ? target : File.join(base, target)
2678
+ clean = Pathname.new(joined).cleanpath.to_s.tr("\\", "/")
2679
+ return nil if clean == ".." || clean.start_with?("../") || clean.start_with?("/")
2680
+
2681
+ clean.delete_prefix("./")
2682
+ end
2683
+
2684
+ def local_target_escapes_vault?(source_id, raw_target)
2685
+ decoded = safe_decode(raw_target).unicode_normalize(:nfc).tr("\\", "/")
2686
+ if decoded.start_with?("/")
2687
+ clean_relative("", decoded.delete_prefix("/")).nil?
2688
+ elsif decoded.start_with?("./", "../")
2689
+ clean_relative(File.dirname(source_id), decoded).nil?
2690
+ else
2691
+ clean_relative("", decoded).nil?
2692
+ end
2693
+ rescue ArgumentError, EncodingError
2694
+ true
2695
+ end
2696
+
2697
+ def ensure_md(path)
2698
+ path.end_with?(NOTE_EXTENSION) ? path : "#{path}#{NOTE_EXTENSION}"
2699
+ end
2700
+
2701
+ def safe_decode(value)
2702
+ URI.decode_uri_component(value.to_s)
2703
+ rescue ArgumentError
2704
+ value.to_s
2705
+ end
2706
+
2707
+ def validated_path(entry)
2708
+ path = entry.path.to_s
2709
+ if path.empty? || path.start_with?("/", "\\") || path.include?("\0") || path.include?("\\")
2710
+ error("invalid_path", "snapshot paths must be relative POSIX paths", path)
2711
+ return nil
2712
+ end
2713
+ segments = path.split("/")
2714
+ if segments.any? { |segment| segment.empty? || segment == "." || segment == ".." } || path != path.unicode_normalize(:nfc)
2715
+ error("invalid_path", "snapshot paths must be normalized NFC paths without traversal", path)
2716
+ return nil
2717
+ end
2718
+ path
2719
+ rescue Encoding::CompatibilityError
2720
+ error("invalid_path", "snapshot path is not valid Unicode", path)
2721
+ nil
2722
+ end
2723
+
2724
+ def production?
2725
+ @config&.environment.to_s != "development"
2726
+ end
2727
+
2728
+ def error_or_warning(code, message, path = nil, span = nil, fatal:, property: nil)
2729
+ if fatal
2730
+ error(code, message, path, span, property: property)
2731
+ else
2732
+ warning(code, message, path, span, property: property)
2733
+ end
2734
+ end
2735
+
2736
+ def error(code, message, path = nil, span = nil, property: nil)
2737
+ @diagnostics << Diagnostic.new(
2738
+ severity: :error,
2739
+ code: code,
2740
+ message: message,
2741
+ path: path,
2742
+ span: span,
2743
+ property: property
2744
+ )
2745
+ end
2746
+
2747
+ def warning(code, message, path = nil, span = nil, property: nil)
2748
+ @diagnostics << Diagnostic.new(
2749
+ severity: :warning,
2750
+ code: code,
2751
+ message: message,
2752
+ path: path,
2753
+ span: span,
2754
+ property: property
2755
+ )
2756
+ end
2757
+
2758
+ def sorted_diagnostics
2759
+ @diagnostics.uniq do |item|
2760
+ [item.severity, item.code, item.message, item.path, span_key(item.span), item.property]
2761
+ end.sort_by do |item|
2762
+ [item.path.to_s, span_key(item.span), item.severity.to_s, item.code, item.property.to_s]
2763
+ end
2764
+ end
2765
+
2766
+ def h(value)
2767
+ text = value.to_s.encode(Encoding::UTF_8, invalid: :replace, undef: :replace, replace: "\uFFFD")
2768
+ CGI.escapeHTML(text.gsub(FrontMatter::XML_INVALID_CHARACTER, "\uFFFD"))
2769
+ end
2770
+ end
2771
+ end