datalog-theme 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +125 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +23 -17
  5. data/_data/cdn-integrity.yml +0 -30
  6. data/_includes/analytics/dashboard.html +3 -1
  7. data/_includes/components/api-function.html +20 -1
  8. data/_includes/components/enhanced-code-block.html +1 -1
  9. data/_includes/csp-meta.html +115 -11
  10. data/_includes/footer.html +19 -20
  11. data/_includes/head.html +60 -23
  12. data/_includes/header/navigation.html +12 -15
  13. data/_includes/header.html +28 -19
  14. data/_includes/layouts/default/article.html +5 -1
  15. data/_includes/meta/math-config.html +9 -5
  16. data/_includes/meta/schema.html +5 -2
  17. data/_includes/meta/scripts-loader.html +16 -32
  18. data/_includes/post/related-posts.html +4 -7
  19. data/_includes/search/index-data.json +9 -34
  20. data/_layouts/dataset.html +1 -0
  21. data/_layouts/default.html +16 -8
  22. data/_layouts/notebook.html +1 -0
  23. data/_layouts/package.html +1 -0
  24. data/_layouts/portfolio.html +1 -0
  25. data/_layouts/post.html +20 -5
  26. data/_layouts/project.html +2 -1
  27. data/_plugins/analytics_dashboard.rb +9 -3
  28. data/_plugins/config_validator.rb +12 -7
  29. data/_plugins/csp_generator.rb +18 -28
  30. data/_plugins/datalog_bibliography.rb +9 -7
  31. data/_plugins/datalog_comments.rb +8 -5
  32. data/_plugins/datalog_slides.rb +9 -8
  33. data/_plugins/i18n.rb +7 -8
  34. data/_plugins/image_optimizer.rb +17 -6
  35. data/_plugins/math_preprocessor.rb +33 -3
  36. data/_plugins/notebook_converter.rb +23 -4
  37. data/_plugins/plugin_loader.rb +3 -1
  38. data/_plugins/publications_generator.rb +8 -2
  39. data/_plugins/rouge_highlight_filter.rb +42 -0
  40. data/_plugins/search_code_blocks.rb +30 -0
  41. data/_plugins/search_normalizer.rb +15 -53
  42. data/_plugins/search_pages.rb +3 -4
  43. data/_sass/_academic-dashboard.scss +262 -0
  44. data/_sass/_base.scss +12 -0
  45. data/_sass/_components.scss +64 -1153
  46. data/_sass/_features.scss +17 -0
  47. data/_sass/_layout.scss +1 -1
  48. data/_sass/_notebooks.scss +322 -0
  49. data/_sass/_open-science-badges.scss +56 -0
  50. data/_sass/{_phase1-enhancements.scss → _post-components.scss} +1 -1
  51. data/_sass/_search-page.scss +530 -0
  52. data/_sass/_search.scss +46 -0
  53. data/_sass/_syntax-highlighting.scss +212 -97
  54. data/_sass/_theme.scss +29 -19
  55. data/_sass/_typography.scss +14 -0
  56. data/assets/css/main.scss +14 -0
  57. data/assets/js/dist/academic.js +1 -1
  58. data/assets/js/dist/analytics-dashboard.js +1 -1
  59. data/assets/js/dist/chunks/chunk-225H5YXE.js +1 -0
  60. data/assets/js/dist/core.js +1 -1
  61. data/assets/js/dist/loader.js +1 -1
  62. data/assets/js/dist/notebook.js +1 -1
  63. data/assets/js/dist/search.js +1 -1
  64. data/assets/js/dist/visualizations.js +11 -2
  65. data/assets/js/loader.js +3 -1
  66. data/datalog-theme.gemspec +36 -23
  67. data/lib/datalog/cli.rb +34 -14
  68. data/lib/datalog/plugin_system/dependency_resolver.rb +0 -2
  69. data/lib/datalog/theme/package.rb +57 -0
  70. data/lib/datalog/theme/repository_checkout.rb +90 -0
  71. data/lib/datalog/theme/version.rb +1 -1
  72. data/lib/datalog/warning_filter.rb +5 -11
  73. data/lib/datalog-theme.rb +5 -0
  74. metadata +48 -133
  75. data/_data/academic.yml +0 -217
  76. data/_data/config/author.yml +0 -121
  77. data/_data/datasets.yml +0 -28
  78. data/_data/js_meta.json +0 -371
  79. data/_data/navigation.yml +0 -145
  80. data/_data/projects.yml +0 -41
  81. data/_data/publications.yml +0 -28
  82. data/_data/social.yml +0 -73
  83. data/_data/visualizations.yml +0 -51
  84. data/_includes/components/advanced-search.html +0 -682
  85. data/_includes/components/bookmark-system.html +0 -96
  86. data/_includes/components/comments.html +0 -244
  87. data/_includes/components/content-recommendations.html +0 -228
  88. data/_includes/components/email-preferences.html +0 -200
  89. data/_includes/components/enhanced-metadata.html +0 -228
  90. data/_includes/components/language-switcher.html +0 -396
  91. data/_includes/components/navigation-enhancements.html +0 -454
  92. data/_includes/components/newsletter-signup.html +0 -178
  93. data/_includes/components/popular-posts.html +0 -233
  94. data/_includes/components/reading-progress.html +0 -133
  95. data/_includes/components/reading-time.html +0 -121
  96. data/_includes/components/series-navigation.html +0 -124
  97. data/_includes/components/social-proof.html +0 -34
  98. data/_includes/components/user-preferences.html +0 -566
  99. data/_includes/meta/syntax-config.html +0 -19
  100. data/_layouts/archive.html +0 -282
  101. data/_layouts/post-sidebar.html +0 -183
  102. data/_sass/_phase3-enhancements.scss +0 -874
  103. data/_sass/_phase4-enhancements.scss +0 -1214
  104. data/_sass/_phase5-enhancements.scss +0 -414
  105. data/assets/js/academic.js +0 -262
  106. data/assets/js/analytics-dashboard.js +0 -382
  107. data/assets/js/core/dark-mode.js +0 -79
  108. data/assets/js/core/github-cards.js +0 -123
  109. data/assets/js/core/language-filter.js +0 -69
  110. data/assets/js/core/navigation.js +0 -184
  111. data/assets/js/core/scroll-progress.js +0 -45
  112. data/assets/js/core/search-hotkeys.js +0 -62
  113. data/assets/js/core/skip-links.js +0 -62
  114. data/assets/js/dist/manifest.json +0 -22
  115. data/assets/js/dist/meta.json +0 -371
  116. data/assets/js/main.js +0 -23
  117. data/assets/js/math.js +0 -818
  118. data/assets/js/notebook.js +0 -158
  119. data/assets/js/search/analytics.js +0 -91
  120. data/assets/js/search/app.js +0 -271
  121. data/assets/js/search/autocomplete.js +0 -120
  122. data/assets/js/search/engine.js +0 -260
  123. data/assets/js/search/filters.js +0 -45
  124. data/assets/js/search/render.js +0 -217
  125. data/assets/js/search/utils.js +0 -99
  126. data/assets/js/search.js +0 -354
  127. data/assets/js/visualizations.js +0 -816
  128. data/assets/publications/datalog-publications.bib +0 -8
  129. data/assets/publications/datalog-publications.ris +0 -9
  130. data/assets/publications/publications.bib +0 -30
  131. data/assets/templates/diogo-ribeiro-cv.md +0 -31
  132. data/assets/templates/diogo-ribeiro-cv.tex +0 -32
  133. data/lib/datalog/theme/theme.rb +0 -18
data/_layouts/post.html CHANGED
@@ -272,7 +272,12 @@ schema_type: TechnicalArticle
272
272
  const metaBar = document.createElement('div');
273
273
  metaBar.className = 'code-block-meta';
274
274
 
275
- const lang = codeBlock.getAttribute('class');
275
+ // Rouge puts the language on the block's wrapper rather than on the code.
276
+ // With line numbers on, the code is a table whose first column holds the
277
+ // numbers, and the copy leaves them out.
278
+ const langHolder = codeBlock.matches('[class*="language-"]') ? codeBlock : codeBlock.closest('.highlighter-rouge');
279
+ const lang = langHolder && langHolder.getAttribute('class');
280
+ const source = codeBlock.querySelector('.rouge-code') || codeBlock;
276
281
  if (lang) {
277
282
  const m = lang.match(/language-([a-z0-9+#]+)/i);
278
283
  if (m) {
@@ -291,7 +296,7 @@ schema_type: TechnicalArticle
291
296
 
292
297
  if (navigator.clipboard && navigator.clipboard.writeText) {
293
298
  button.addEventListener('click', function () {
294
- navigator.clipboard.writeText(codeBlock.textContent).then(function () {
299
+ navigator.clipboard.writeText(source.textContent).then(function () {
295
300
  button.textContent = 'Copied!';
296
301
  button.classList.add('is-copied');
297
302
  setTimeout(function () {
@@ -311,9 +316,19 @@ schema_type: TechnicalArticle
311
316
  wrapper.insertBefore(metaBar, pre);
312
317
  });
313
318
 
314
- // Make equation references focus their targets for accessibility.
315
- if (window.MathJax && window.MathJax.startup) {
316
- window.MathJax.startup.promise.then(function () {
319
+ // Make equation references focus their targets for accessibility. This runs
320
+ // on DOMContentLoaded, before MathJax has loaded, when window.MathJax is
321
+ // still the configuration object and has no startup.promise; reading it
322
+ // threw. The head's MathJax configuration announces typeset math instead.
323
+ const onMathReady = function (callback) {
324
+ if (document.body.classList.contains('math-ready') && document.querySelector('mjx-container')) {
325
+ callback();
326
+ } else {
327
+ document.addEventListener('datalog:math-ready', callback, { once: true });
328
+ }
329
+ };
330
+ if (window.MathJax) {
331
+ onMathReady(function () {
317
332
  const equations = document.querySelectorAll('.post-content mjx-container[display="true"]');
318
333
  let eqIndex = 1;
319
334
  equations.forEach(function (container) {
@@ -1,5 +1,6 @@
1
1
  ---
2
2
  layout: default
3
+ show_title: false # this layout renders the page's <h1>
3
4
  ---
4
5
  {% assign localization_config = site.theme_options.localization %}
5
6
  {% assign date_format = localization_config.date_format | default: 'long' %}
@@ -30,7 +31,7 @@ layout: default
30
31
  {% if page.github %}
31
32
  <section class="project-case__github" aria-labelledby="github-heading">
32
33
  <h2 id="github-heading">{% t 'project.header.github_repository' %}</h2>
33
- <div class="project-case__github-card" data-github-repo="{{ page.github.repo }}" data-github-owner="{{ page.github.owner }}">
34
+ <div class="project-case__github-card" data-github-repo="{{ page.github.repo }}" data-github-owner="{{ page.github.owner }}" data-github-unavailable="{% t 'project.github.unavailable' %}">
34
35
  <div class="project-case__github-meta">
35
36
  <p class="project-case__github-name">
36
37
  <a href="https://github.com/{{ page.github.owner }}/{{ page.github.repo }}" itemprop="url">{{ page.github.owner }}/{{ page.github.repo }}</a>
@@ -31,7 +31,10 @@ module Datalog
31
31
 
32
32
  data = query_analytics(site)
33
33
  data["fetched_at"] = Time.now.utc.iso8601
34
- write_cache(cache_path, data)
34
+ # Only a successful report is cached. A missing-configuration or error
35
+ # payload used to be kept for CACHE_TTL as well, so a site that had just
36
+ # set GA4_PROPERTY_ID went on reporting the old problem for a day.
37
+ write_cache(cache_path, data) if data["status"] == "ok"
35
38
  data
36
39
  rescue StandardError => e
37
40
  Jekyll.logger.warn("Analytics", "Falling back to cached analytics data: #{e.message}")
@@ -43,7 +46,7 @@ module Datalog
43
46
  end
44
47
 
45
48
  def fresh?(payload)
46
- return false unless payload.is_a?(Hash)
49
+ return false unless payload.is_a?(Hash) && payload["status"] == "ok"
47
50
 
48
51
  fetched_at = payload["fetched_at"]
49
52
  return false if fetched_at.to_s.empty?
@@ -95,7 +98,10 @@ module Datalog
95
98
  results["status"] = "ok"
96
99
  results
97
100
  rescue LoadError => e
98
- fallback_payload("missing_dependency", "Install googleauth to enable GA4 integration: #{e.message}")
101
+ # The theme does not depend on googleauth: only a site with a GA4 property
102
+ # configured needs it, and adds it to its own Gemfile.
103
+ fallback_payload("missing_dependency",
104
+ "Add gem \"googleauth\" to the site's Gemfile to enable the GA4 integration: #{e.message}")
99
105
  rescue StandardError => e
100
106
  fallback_payload("error", e.message)
101
107
  end
@@ -12,8 +12,9 @@ module Datalog
12
12
  SCHEMA = {
13
13
  title: { type: :string, required: true },
14
14
  url: { type: :string, required: true, format: :url },
15
+ # `author: Jane Doe` is a common Jekyll setting; the hash adds an email and profile links.
15
16
  author: {
16
- type: :hash,
17
+ type: %i[string hash],
17
18
  required: true,
18
19
  schema: {
19
20
  name: { type: :string, required: true },
@@ -68,12 +69,6 @@ module Datalog
68
69
  engine: { type: :string, enum: %w[mathjax katex] },
69
70
  enabled: { type: :boolean }
70
71
  }
71
- },
72
- syntax_highlighting: {
73
- type: :hash,
74
- schema: {
75
- cdn: { type: :string }
76
- }
77
72
  }
78
73
  }
79
74
  }
@@ -84,6 +79,10 @@ module Datalog
84
79
  replacement: "theme_options.math.engine",
85
80
  message: "'math_engine' has moved under theme_options.math.engine.",
86
81
  auto_migrate: true
82
+ },
83
+ "theme_options.syntax_highlighting" => {
84
+ message: "Code is highlighted by Rouge when the site builds and the theme no longer loads Prism, " \
85
+ "so these settings have no effect. Remove them."
87
86
  }
88
87
  }.freeze
89
88
 
@@ -204,6 +203,8 @@ module Datalog
204
203
  end
205
204
 
206
205
  def type_valid?(value, expected_type)
206
+ return expected_type.any? { |type| type_valid?(value, type) } if expected_type.is_a?(Array)
207
+
207
208
  case expected_type
208
209
  when :string
209
210
  value.is_a?(String)
@@ -222,6 +223,8 @@ module Datalog
222
223
 
223
224
  def url_valid?(value)
224
225
  return false unless value.is_a?(String)
226
+ # `url: ""` is what `jekyll new` writes, and what a site built locally keeps.
227
+ return true if value.strip.empty?
225
228
 
226
229
  uri = URI.parse(value)
227
230
  uri.is_a?(URI::HTTP) && !uri.host.nil?
@@ -311,6 +314,8 @@ module Datalog
311
314
  end
312
315
 
313
316
  def human_type(type)
317
+ return type.map { |entry| human_type(entry) }.join(" or ") if type.is_a?(Array)
318
+
314
319
  case type
315
320
  when :string then "a String"
316
321
  when :integer then "an Integer"
@@ -1,7 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "securerandom"
4
- require "digest"
5
4
 
6
5
  module Datalog
7
6
  module Security
@@ -18,7 +17,6 @@ module Datalog
18
17
  def generate(site)
19
18
  site.data["csp"] ||= {}
20
19
  site.data["csp"]["nonces"] ||= {}
21
- site.data["csp"]["hashes"] ||= {}
22
20
 
23
21
  assign_nonces(site, site.pages)
24
22
  site.collections.each_value do |collection|
@@ -40,7 +38,6 @@ module Datalog
40
38
 
41
39
  nonce = SecureRandom.base64(nonce_bytes)
42
40
  document.data["csp_nonce"] = nonce
43
- document.data["csp_hashes"] ||= []
44
41
 
45
42
  registry_site = site || (document.respond_to?(:site) ? document.site : nil)
46
43
  if registry_site.respond_to?(:data)
@@ -61,7 +58,12 @@ module Datalog
61
58
  end
62
59
  end
63
60
 
64
- def self.compute_hashes(document)
61
+ # Gives every inline script in the rendered page the page's nonce. The
62
+ # generator also took a SHA-256 of each inline script, but the policy is
63
+ # written into the head while the page renders, so the hashes of the
64
+ # finished page never reached it; with every script nonced they are not
65
+ # needed.
66
+ def self.add_nonces(document)
65
67
  return unless document.respond_to?(:output)
66
68
 
67
69
  output = document.output
@@ -77,39 +79,27 @@ module Datalog
77
79
  attributes = Regexp.last_match(1)
78
80
  "<script nonce=\"#{nonce}\"#{attributes}>"
79
81
  end
80
-
81
- document.output = output
82
-
83
- hashes = []
84
- output.scan(%r{<script(?![^>]*\bsrc=)[^>]*>(.*?)</script>}mi) do |match|
85
- content = match.first
86
- next if content.nil? || content.empty?
87
-
88
- hashes << Digest::SHA256.base64digest(content)
82
+ # A template that printed page.csp_nonce before the page had one left
83
+ # nonce="", which authorises nothing.
84
+ document.output = output.gsub(/(<(?:script|style)\b[^>]*\bnonce=)""/i) do
85
+ "#{Regexp.last_match(1)}\"#{nonce}\""
89
86
  end
90
-
91
- hashes.uniq!
92
- document.data["csp_hashes"] = hashes
93
-
94
- return unless site.respond_to?(:data)
95
-
96
- site.data["csp"] ||= {}
97
- site.data["csp"]["hashes"] ||= {}
98
-
99
- key = document_key(document)
100
-
101
- site.data["csp"]["hashes"][key] = hashes if key
102
87
  end
103
88
  end
104
89
  end
105
90
  end
106
91
 
107
92
  %i[pages documents].each do |target|
108
- Jekyll::Hooks.register target, :pre_render do |document|
109
- Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
93
+ Jekyll::Hooks.register target, :pre_render do |document, payload|
94
+ nonce = Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
95
+ # A page another generator creates after this one runs, such as a notebook
96
+ # page, only gets its nonce here. Jekyll has already copied a page's data
97
+ # into the hash its templates read, so page.csp_nonce rendered empty there.
98
+ page_data = payload && payload["page"]
99
+ page_data["csp_nonce"] ||= nonce if nonce && page_data.is_a?(Hash)
110
100
  end
111
101
 
112
102
  Jekyll::Hooks.register target, :post_render do |document|
113
- Datalog::Security::CspGenerator.compute_hashes(document)
103
+ Datalog::Security::CspGenerator.add_nonces(document)
114
104
  end
115
105
  end
@@ -1,17 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_bibliography` tag when the datalog-citations
5
+ # plugin is not enabled, so a layout that uses the tag still builds. The plugin
6
+ # registers the real tag and sets `datalog_bibliography` on the pages it
7
+ # handles; a page that sets the key by hand without the plugin gets nothing and
8
+ # a build warning, where it used to get a "coming soon" notice.
4
9
  class DatalogBibliographyTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
- bib_data = page["datalog_bibliography"]
8
- return "" unless bib_data
12
+ return "" unless page["datalog_bibliography"]
9
13
 
10
- <<~HTML
11
- <div class="datalog-bibliography-placeholder">
12
- <p>Bibliography feature coming soon.</p>
13
- </div>
14
- HTML
14
+ Jekyll.logger.warn("datalog-citations", "#{page['path']} sets datalog_bibliography, " \
15
+ "but datalog_plugins.enabled does not list datalog-citations")
16
+ ""
15
17
  end
16
18
  end
17
19
  end
@@ -1,16 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_comments` tag when the datalog-comments plugin is
5
+ # not enabled, so a layout that uses the tag still builds. The plugin registers
6
+ # the real tag and sets `datalog_comments` on the pages it handles; a page that
7
+ # sets the key by hand without the plugin gets nothing and a build warning,
8
+ # where it used to get a "coming soon" notice.
4
9
  class DatalogCommentsTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
12
  return "" unless page["datalog_comments"]
8
13
 
9
- <<~HTML
10
- <div class="datalog-comments-placeholder">
11
- <p>Comments feature coming soon.</p>
12
- </div>
13
- HTML
14
+ Jekyll.logger.warn("datalog-comments", "#{page['path']} sets datalog_comments, " \
15
+ "but datalog_plugins.enabled does not list datalog-comments")
16
+ ""
14
17
  end
15
18
  end
16
19
  end
@@ -1,18 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_slides` tag when the datalog-slides plugin is not
5
+ # enabled, so a layout that uses the tag still builds. The plugin registers the
6
+ # real tag and sets `datalog_slides` on the pages it handles; a page that sets
7
+ # the key by hand without the plugin gets nothing and a build warning, where it
8
+ # used to get a "coming soon" notice showing its raw configuration.
4
9
  class DatalogSlidesTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
- slides_data = page["datalog_slides"]
8
- return "" unless slides_data
12
+ return "" unless page["datalog_slides"]
9
13
 
10
- # Render a simple placeholder or embed
11
- <<~HTML
12
- <div class="datalog-slides-placeholder">
13
- <p>Slides feature coming soon. Configured: #{slides_data}</p>
14
- </div>
15
- HTML
14
+ Jekyll.logger.warn("datalog-slides", "#{page['path']} sets datalog_slides, " \
15
+ "but datalog_plugins.enabled does not list datalog-slides")
16
+ ""
16
17
  end
17
18
  end
18
19
  end
data/_plugins/i18n.rb CHANGED
@@ -116,6 +116,11 @@ class TranslateTag < Liquid::Tag
116
116
  # quote into the first option name, or interpolation silently fails.
117
117
  SYNTAX = /\A\s*(['"]?)(\w[\w.-]*)\1(.*)?\z/m
118
118
 
119
+ # `name: value` pairs, separated by commas or spaces. A quoted value may
120
+ # contain commas: splitting the markup on every comma cut `name: "Doe, Jane"`
121
+ # in two.
122
+ OPTION = /(\w+)\s*:\s*("(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|[^\s,]+)/
123
+
119
124
  def initialize(tag_name, markup, tokens)
120
125
  super
121
126
  raise Liquid::SyntaxError, "Syntax Error in 't' - Valid syntax: t key [arg: value]" unless markup.strip =~ SYNTAX
@@ -134,14 +139,8 @@ class TranslateTag < Liquid::Tag
134
139
  def parse_options(markup, context)
135
140
  return {} unless markup && !markup.strip.empty?
136
141
 
137
- tokens = markup.strip.split(",").map(&:strip)
138
- tokens.each_with_object({}) do |token, memo|
139
- next if token.empty?
140
-
141
- if token.include?(":")
142
- key, value = token.split(":", 2)
143
- memo[key.strip.to_sym] = context.evaluate(Liquid::Expression.parse(value.strip))
144
- end
142
+ markup.scan(OPTION).to_h do |key, value|
143
+ [key.to_sym, context.evaluate(Liquid::Expression.parse(value))]
145
144
  end
146
145
  end
147
146
  end
@@ -34,6 +34,18 @@ module Jekyll
34
34
  }.freeze
35
35
  RASTER_EXTENSIONS = %w[.jpg .jpeg .png].freeze
36
36
 
37
+ # The one image a page fetches early: the first in its post or page content,
38
+ # unless the author made it lazy or the page already preloads an image, as
39
+ # the hero does. The first <img> anywhere used to be marked both lazy and
40
+ # high priority; on the home page that was a post card below the hero, and
41
+ # on tutorials the thumbnail of a related post at the bottom.
42
+ def priority_image(fragment)
43
+ return if fragment.at_css('link[rel="preload"][as="image"]')
44
+
45
+ image = fragment.css(".post-content img, .page-content img").find { |img| img["data-no-optimize"] != "true" }
46
+ image unless image.nil? || image["loading"] == "lazy"
47
+ end
48
+
37
49
  def process(document)
38
50
  return unless document.output_ext == ".html"
39
51
  return if document.output.nil? || document.output.empty?
@@ -46,18 +58,17 @@ module Jekyll
46
58
  manifest = site&.data&.fetch("datalog_responsive_images", {}) || {}
47
59
  image_config = site&.config&.fetch("datalog_image_config", {}) || {}
48
60
  optimized = false
49
- first_priority_assigned = false
61
+ priority = priority_image(fragment)
50
62
 
51
63
  fragment.css("img").each do |img|
52
64
  next if img["data-no-optimize"] == "true"
53
65
 
54
- img["loading"] ||= "lazy"
55
- img["decoding"] ||= "async"
56
-
57
- unless first_priority_assigned
66
+ if img == priority
58
67
  img["fetchpriority"] ||= "high"
59
- first_priority_assigned = true
68
+ else
69
+ img["loading"] ||= "lazy"
60
70
  end
71
+ img["decoding"] ||= "async"
61
72
 
62
73
  normalized_src = normalize_src(img["src"], site)
63
74
  picture_entry = manifest[normalized_src]
@@ -19,8 +19,11 @@ module MathPreprocessor
19
19
  ].freeze
20
20
 
21
21
  INLINE_PATTERNS = [
22
+ # Pandoc's rule for inline math, so prices and shell variables stay text:
23
+ # the opening $ is followed by a non-space, the closing $ follows a
24
+ # non-space and is not followed by a digit, and a blank line ends it.
22
25
  {
23
- regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>[^$]+?)(?<close>\$)(?!\$)/m,
26
+ regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
24
27
  tag: "span"
25
28
  },
26
29
  {
@@ -29,6 +32,20 @@ module MathPreprocessor
29
32
  }
30
33
  ].freeze
31
34
 
35
+ # Code shows dollar signs literally (shell and R variables, amounts in SQL),
36
+ # so fenced blocks, highlight tags, <pre>/<code> elements and inline code
37
+ # spans are set aside before looking for math and put back afterwards.
38
+ CODE_PATTERNS = [
39
+ /^([ \t]*)(`{3,}|~{3,})[^\n]*\n.*?(?:^\1\2[ \t]*$|\z)/m,
40
+ /\{%-?\s*highlight\b.*?\{%-?\s*endhighlight\s*-?%\}/m,
41
+ %r{<(pre|code)\b[^>]*>.*?</\1>}mi,
42
+ /(?<!`)(`+)(?!`)(?:(?!\n[ \t]*\n).)+?(?<!`)\1(?!`)/m
43
+ ].freeze
44
+
45
+ # NUL marks masked code, since page content never contains it. It is written as
46
+ # an escape: a raw NUL byte in the source stopped RuboCop from parsing the file.
47
+ PLACEHOLDER = /\x00(\d+)\x00/
48
+
32
49
  class Processor
33
50
  attr_reader :expressions
34
51
 
@@ -40,9 +57,18 @@ module MathPreprocessor
40
57
  def process
41
58
  return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
42
59
 
43
- processed = @content.dup
60
+ code = []
61
+ processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
62
+ text.gsub(pattern) do |match|
63
+ code << match
64
+ "\x00#{code.size - 1}\x00"
65
+ end
66
+ end
44
67
  processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
45
- apply_patterns(processed, INLINE_PATTERNS, display: false)
68
+ processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
69
+ # A segment set aside can contain the placeholder of an earlier one.
70
+ processed = processed.gsub(PLACEHOLDER) { code[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
71
+ processed
46
72
  end
47
73
 
48
74
  private
@@ -66,8 +92,12 @@ module MathPreprocessor
66
92
  cleaned_source = cleanup_source(latex)
67
93
  record_expression(cleaned_source, alt_text)
68
94
 
95
+ # ARIA forbids aria-label on an element with no role, such as a plain span.
96
+ # axe let it pass while the span held the raw LaTeX as text, and failed it
97
+ # once MathJax rendered the expression; the math role allows the label.
69
98
  attributes = {
70
99
  "class" => display ? "math-expression math-expression--source" : "math-expression-inline math-expression--source",
100
+ "role" => "math",
71
101
  "data-math-alt" => alt_text,
72
102
  "data-math-source" => cleaned_source,
73
103
  "aria-label" => alt_text,
@@ -7,6 +7,7 @@ require "fileutils"
7
7
  require "cgi"
8
8
  require "loofah"
9
9
  require "base64"
10
+ require_relative "rouge_highlight_filter"
10
11
 
11
12
  module Datalog
12
13
  module NotebookRenderer
@@ -94,14 +95,30 @@ module Datalog
94
95
  metadata: build_sanitization_metadata(meta, cell_index: cell_index))
95
96
  return if sanitized.to_s.strip.empty?
96
97
 
97
- %(<section class="notebook-cell notebook-cell--markdown">\n#{sanitized}\n</section>)
98
+ %(<section class="notebook-cell notebook-cell--markdown">\n#{demote_headings(sanitized.to_s)}\n</section>)
99
+ end
100
+
101
+ # The notebook layout gives the page its <h1>, and a notebook's first
102
+ # markdown cell usually repeats the title as `# Title`. Each heading moves
103
+ # down a level (h1 to h2, and so on to h6), which keeps one <h1> on the
104
+ # page and the cells' own outline under it.
105
+ def demote_headings(html)
106
+ html.gsub(%r{<(/?)h([1-5])(?=[\s>])}i) { "<#{Regexp.last_match(1)}h#{Regexp.last_match(2).to_i + 1}" }
98
107
  end
99
108
 
100
109
  # Class names mirror the theme stylesheet (`.notebook-cell--input`,
101
110
  # `.notebook-cell__code` and `.notebook-cell__outputs` in _sass/_components.scss).
102
111
  def render_code(cell, source, metadata, site, cell_index)
103
112
  language = cell.dig("metadata", "language") || metadata[:language] || "text"
104
- code_html = %(<pre class="notebook-cell__code"><code class="language-#{language}">#{CGI.escapeHTML(source)}</code></pre>)
113
+ # The language comes from the notebook file, so it is cut down to the
114
+ # characters a class name can hold before it goes into the attribute.
115
+ language_class = language.to_s.gsub(/[^\w+#.-]/, "")
116
+ # Rouge highlights the cell as the site builds, as kramdown does for code
117
+ # blocks, and escapes it.
118
+ highlighted = Jekyll::RougeHighlightFilter.highlight(source, language_class)
119
+ pre_attributes = %(class="highlight notebook-cell__code" tabindex="0")
120
+ code_attributes = %(class="language-#{language_class}")
121
+ code_html = %(<pre #{pre_attributes}><code #{code_attributes}>#{highlighted}</code></pre>)
105
122
  base_metadata = metadata.respond_to?(:merge) ? metadata.merge(language: language) : { language: language }
106
123
  outputs_html = render_outputs(Array(cell["outputs"]), site: site, cell_index: cell_index, metadata: base_metadata)
107
124
  outputs_html = %(\n<div class="notebook-cell__outputs">\n#{outputs_html}\n</div>) unless outputs_html.empty?
@@ -157,11 +174,13 @@ module Datalog
157
174
  def image_output_html(output)
158
175
  data = output["data"] || {}
159
176
 
177
+ # Jupyter writes base64 image data split over lines or ending in a newline,
178
+ # and the data URI check rejects whitespace, which dropped those images.
160
179
  if (png = data["image/png"])
161
- html = %(<img src="data:image/png;base64,#{Array(png).join}" alt="Notebook output" />)
180
+ html = %(<img src="data:image/png;base64,#{Array(png).join.gsub(/\s+/, '')}" alt="Notebook output" />)
162
181
  return [html, "image/png"]
163
182
  elsif (jpeg = data["image/jpeg"])
164
- html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join}" alt="Notebook output" />)
183
+ html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join.gsub(/\s+/, '')}" alt="Notebook output" />)
165
184
  return [html, "image/jpeg"]
166
185
  elsif (svg = data["image/svg+xml"])
167
186
  encoded = Base64.strict_encode64(Array(svg).join)
@@ -149,7 +149,9 @@ end
149
149
 
150
150
  module Datalog
151
151
  module PluginLoaderHooks
152
- HOOK_SCOPES = %i[pages documents posts].freeze
152
+ # Posts are documents: Jekyll fires a post's `posts` hooks and then its
153
+ # `documents` hooks, so registering both ran every plugin hook twice a post.
154
+ HOOK_SCOPES = %i[pages documents].freeze
153
155
 
154
156
  module_function
155
157
 
@@ -6,7 +6,13 @@ module Jekyll
6
6
  priority :low
7
7
 
8
8
  def generate(site)
9
- publications_data = site.data["publications"] ||= {}
9
+ publications_data = site.data["publications"]
10
+ # _data/publications.yml is normally a map with `settings` and
11
+ # `manual_entries`; a plain list of entries is read as the manual
12
+ # entries rather than stopping the build with a TypeError.
13
+ publications_data = { "manual_entries" => publications_data } if publications_data.is_a?(Array)
14
+ publications_data = {} unless publications_data.is_a?(Hash)
15
+ site.data["publications"] = publications_data
10
16
  settings = publications_data["settings"] || {}
11
17
  config_source = site.config.dig("theme_options", "publications", "bibtex_source")
12
18
  bibtex_source = settings["bibtex_source"] || config_source
@@ -22,7 +28,7 @@ module Jekyll
22
28
  end
23
29
  end
24
30
 
25
- manual_entries = publications_data["manual_entries"] || []
31
+ manual_entries = Array(publications_data["manual_entries"]).select { |entry| entry.is_a?(Hash) }
26
32
  combined = (imported_entries + manual_entries).map { |entry| normalize_entry(entry) }
27
33
 
28
34
  academic_citations = site.data.dig("academic", "citations", "per_publication") || {}
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rouge"
4
+
5
+ # `rouge_highlight` highlights code with Rouge as the site builds, the way
6
+ # kramdown highlights fenced code blocks. Includes that print code passed to
7
+ # them, such as `components/api-function.html`, call it as
8
+ # `code | rouge_highlight: language` inside `<pre class="highlight"><code>`;
9
+ # the notebook converter calls `RougeHighlightFilter.highlight` for code cells.
10
+ # The result is escaped; a language Rouge does not know comes back as plain
11
+ # text.
12
+ module Jekyll
13
+ module RougeHighlightFilter
14
+ # kramdown's opening tag for a block Rouge highlighted.
15
+ KRAMDOWN_CODE_BLOCK = '<pre class="highlight">'
16
+
17
+ def self.highlight(code, language = nil)
18
+ return "" if code.nil?
19
+
20
+ source = code.to_s
21
+ lexer = Rouge::Lexer.find_fancy(language.to_s.strip.downcase, source) || Rouge::Lexers::PlainText
22
+ Rouge::Formatters::HTML.new.format(lexer.lex(source))
23
+ end
24
+
25
+ def rouge_highlight(code, language = nil)
26
+ RougeHighlightFilter.highlight(code, language)
27
+ end
28
+ end
29
+ end
30
+
31
+ Liquid::Template.register_filter(Jekyll::RougeHighlightFilter)
32
+
33
+ # A code block wider than the page scrolls, and a keyboard user can only scroll
34
+ # it once it takes focus. Prism made every block focusable in the browser; the
35
+ # blocks kramdown highlights get the attribute here, and the includes and the
36
+ # notebook converter write it themselves.
37
+ Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
38
+ block = Jekyll::RougeHighlightFilter::KRAMDOWN_CODE_BLOCK
39
+ next unless document.content&.include?(block)
40
+
41
+ document.content = document.content.gsub(block, '<pre class="highlight" tabindex="0">')
42
+ end
@@ -0,0 +1,30 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Datalog
4
+ # Collects the fenced code blocks of every page and document for the search
5
+ # index. The index template used to split `doc.content` on backticks, but
6
+ # Jekyll renders documents before pages, so by the time search.json rendered
7
+ # that content was HTML and every document's code list came out empty.
8
+ module SearchCodeBlocks
9
+ # An opening fence of three or more backticks or tildes with an optional
10
+ # language, the code, and a closing fence of the same characters.
11
+ FENCE = /^ {0,3}(`{3,}|~{3,})[ \t]*([^\s`~{]*)[^\n]*\n(.*?)^ {0,3}\1[ \t]*$/m
12
+
13
+ module_function
14
+
15
+ def extract(source)
16
+ text = source.to_s.gsub("\r\n", "\n")
17
+ text.scan(FENCE).map do |_fence, language, code|
18
+ { "language" => language.empty? ? "text" : language.downcase, "code" => code.chomp }
19
+ end
20
+ end
21
+ end
22
+ end
23
+
24
+ # Content is still the author's source before rendering starts. Collection
25
+ # docs only: site.documents also lists a collection's static files.
26
+ Jekyll::Hooks.register :site, :pre_render do |site|
27
+ (site.pages + site.collections.values.flat_map(&:docs)).each do |item|
28
+ item.data["search_code"] = Datalog::SearchCodeBlocks.extract(item.content)
29
+ end
30
+ end