datalog-theme 0.6.1 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +175 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +26 -19
  5. data/_data/cdn-integrity.yml +0 -30
  6. data/_includes/analytics/dashboard.html +3 -1
  7. data/_includes/components/api-function.html +20 -1
  8. data/_includes/components/author-bio.html +2 -2
  9. data/_includes/components/enhanced-code-block.html +1 -1
  10. data/_includes/components/hero.html +19 -1
  11. data/_includes/csp-meta.html +115 -11
  12. data/_includes/footer.html +19 -20
  13. data/_includes/head.html +61 -29
  14. data/_includes/header/navigation.html +12 -15
  15. data/_includes/header.html +29 -20
  16. data/_includes/layouts/default/article.html +5 -1
  17. data/_includes/meta/math-config.html +27 -1
  18. data/_includes/meta/schema.html +5 -2
  19. data/_includes/meta/scripts-loader.html +16 -31
  20. data/_includes/post/related-posts.html +4 -7
  21. data/_includes/scripts.html +1 -9
  22. data/_includes/search/index-data.json +131 -0
  23. data/_includes/search/page.html +136 -0
  24. data/_layouts/dataset.html +1 -0
  25. data/_layouts/default.html +16 -8
  26. data/_layouts/home.html +6 -0
  27. data/_layouts/notebook.html +1 -0
  28. data/_layouts/package.html +1 -0
  29. data/_layouts/portfolio.html +1 -0
  30. data/_layouts/post.html +20 -5
  31. data/_layouts/project.html +2 -1
  32. data/_plugins/analytics_dashboard.rb +9 -3
  33. data/_plugins/config_validator.rb +12 -7
  34. data/_plugins/csp_generator.rb +18 -28
  35. data/_plugins/datalog_bibliography.rb +9 -7
  36. data/_plugins/datalog_comments.rb +8 -5
  37. data/_plugins/datalog_slides.rb +9 -8
  38. data/_plugins/i18n.rb +7 -8
  39. data/_plugins/image_optimizer.rb +17 -6
  40. data/_plugins/math_preprocessor.rb +36 -3
  41. data/_plugins/notebook_converter.rb +23 -43
  42. data/_plugins/plugin_loader.rb +3 -1
  43. data/_plugins/publications_generator.rb +8 -2
  44. data/_plugins/rouge_highlight_filter.rb +42 -0
  45. data/_plugins/search_code_blocks.rb +30 -0
  46. data/_plugins/search_normalizer.rb +15 -53
  47. data/_plugins/search_pages.rb +72 -0
  48. data/_sass/_academic-dashboard.scss +262 -0
  49. data/_sass/_base.scss +19 -1
  50. data/_sass/_components.scss +77 -1144
  51. data/_sass/_features.scss +17 -0
  52. data/_sass/_font-fallbacks.scss +43 -0
  53. data/_sass/_header.scss +10 -0
  54. data/_sass/_layout.scss +18 -12
  55. data/_sass/_mathematical.scss +1 -1
  56. data/_sass/_notebooks.scss +322 -0
  57. data/_sass/_open-science-badges.scss +56 -0
  58. data/_sass/{_phase1-enhancements.scss → _post-components.scss} +46 -1
  59. data/_sass/_search-page.scss +530 -0
  60. data/_sass/_search.scss +47 -1
  61. data/_sass/_syntax-highlighting.scss +212 -97
  62. data/_sass/_theme.scss +42 -18
  63. data/_sass/_typography.scss +14 -0
  64. data/_sass/_variables.scss +7 -4
  65. data/assets/css/main.scss +14 -0
  66. data/assets/img/hero-detail-640.webp +0 -0
  67. data/assets/img/hero-detail.webp +0 -0
  68. data/assets/img/social-card.png +0 -0
  69. data/assets/js/dist/academic.js +1 -0
  70. data/assets/js/dist/analytics-dashboard.js +1 -0
  71. data/assets/js/dist/chunks/chunk-225H5YXE.js +1 -0
  72. data/assets/js/dist/core.js +1 -0
  73. data/assets/js/dist/loader.js +1 -0
  74. data/assets/js/dist/math.js +1 -0
  75. data/assets/js/dist/notebook.js +1 -0
  76. data/assets/js/dist/search.js +1 -0
  77. data/assets/js/dist/visualizations.js +11 -0
  78. data/assets/js/loader.js +3 -1
  79. data/datalog-theme.gemspec +43 -22
  80. data/lib/datalog/cli.rb +53 -15
  81. data/lib/datalog/plugin_system/dependency_resolver.rb +0 -2
  82. data/lib/datalog/theme/package.rb +57 -0
  83. data/lib/datalog/theme/repository_checkout.rb +90 -0
  84. data/lib/datalog/theme/version.rb +1 -1
  85. data/lib/datalog/warning_filter.rb +5 -11
  86. data/lib/datalog-theme.rb +21 -0
  87. metadata +75 -146
  88. data/_data/academic.yml +0 -217
  89. data/_data/config/author.yml +0 -121
  90. data/_data/config/features.yml +0 -262
  91. data/_data/config/site.yml +0 -42
  92. data/_data/config/theme.yml +0 -181
  93. data/_data/datasets.yml +0 -28
  94. data/_data/js_meta.json +0 -371
  95. data/_data/navigation.yml +0 -145
  96. data/_data/projects.yml +0 -41
  97. data/_data/publications.yml +0 -28
  98. data/_data/social.yml +0 -73
  99. data/_data/visualizations.yml +0 -51
  100. data/_includes/components/advanced-search.html +0 -682
  101. data/_includes/components/bookmark-system.html +0 -96
  102. data/_includes/components/comments.html +0 -244
  103. data/_includes/components/content-recommendations.html +0 -228
  104. data/_includes/components/email-preferences.html +0 -200
  105. data/_includes/components/enhanced-metadata.html +0 -228
  106. data/_includes/components/language-switcher.html +0 -396
  107. data/_includes/components/navigation-enhancements.html +0 -454
  108. data/_includes/components/newsletter-signup.html +0 -178
  109. data/_includes/components/popular-posts.html +0 -233
  110. data/_includes/components/reading-progress.html +0 -133
  111. data/_includes/components/reading-time.html +0 -121
  112. data/_includes/components/series-navigation.html +0 -124
  113. data/_includes/components/social-proof.html +0 -34
  114. data/_includes/components/user-preferences.html +0 -566
  115. data/_layouts/archive.html +0 -282
  116. data/_layouts/post-sidebar.html +0 -183
  117. data/_sass/_phase3-enhancements.scss +0 -874
  118. data/_sass/_phase4-enhancements.scss +0 -1214
  119. data/_sass/_phase5-enhancements.scss +0 -414
  120. data/assets/img/20220607123041_detail.001.png +0 -0
  121. data/assets/js/academic.js +0 -262
  122. data/assets/js/analytics-dashboard.js +0 -382
  123. data/assets/js/core/dark-mode.js +0 -79
  124. data/assets/js/core/github-cards.js +0 -123
  125. data/assets/js/core/language-filter.js +0 -69
  126. data/assets/js/core/navigation.js +0 -184
  127. data/assets/js/core/scroll-progress.js +0 -45
  128. data/assets/js/core/search-hotkeys.js +0 -62
  129. data/assets/js/core/skip-links.js +0 -62
  130. data/assets/js/main.js +0 -23
  131. data/assets/js/math.js +0 -818
  132. data/assets/js/notebook.js +0 -158
  133. data/assets/js/search/analytics.js +0 -91
  134. data/assets/js/search/app.js +0 -271
  135. data/assets/js/search/autocomplete.js +0 -120
  136. data/assets/js/search/engine.js +0 -260
  137. data/assets/js/search/filters.js +0 -38
  138. data/assets/js/search/render.js +0 -217
  139. data/assets/js/search/utils.js +0 -99
  140. data/assets/js/search.js +0 -354
  141. data/assets/js/visualizations.js +0 -816
  142. data/assets/publications/datalog-publications.bib +0 -8
  143. data/assets/publications/datalog-publications.ris +0 -9
  144. data/assets/publications/publications.bib +0 -30
  145. data/assets/templates/diogo-ribeiro-cv.md +0 -31
  146. data/assets/templates/diogo-ribeiro-cv.tex +0 -32
  147. data/lib/datalog/theme/theme.rb +0 -18
@@ -1,7 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "securerandom"
4
- require "digest"
5
4
 
6
5
  module Datalog
7
6
  module Security
@@ -18,7 +17,6 @@ module Datalog
18
17
  def generate(site)
19
18
  site.data["csp"] ||= {}
20
19
  site.data["csp"]["nonces"] ||= {}
21
- site.data["csp"]["hashes"] ||= {}
22
20
 
23
21
  assign_nonces(site, site.pages)
24
22
  site.collections.each_value do |collection|
@@ -40,7 +38,6 @@ module Datalog
40
38
 
41
39
  nonce = SecureRandom.base64(nonce_bytes)
42
40
  document.data["csp_nonce"] = nonce
43
- document.data["csp_hashes"] ||= []
44
41
 
45
42
  registry_site = site || (document.respond_to?(:site) ? document.site : nil)
46
43
  if registry_site.respond_to?(:data)
@@ -61,7 +58,12 @@ module Datalog
61
58
  end
62
59
  end
63
60
 
64
- def self.compute_hashes(document)
61
+ # Gives every inline script in the rendered page the page's nonce. The
62
+ # generator also took a SHA-256 of each inline script, but the policy is
63
+ # written into the head while the page renders, so the hashes of the
64
+ # finished page never reached it; with every script nonced they are not
65
+ # needed.
66
+ def self.add_nonces(document)
65
67
  return unless document.respond_to?(:output)
66
68
 
67
69
  output = document.output
@@ -77,39 +79,27 @@ module Datalog
77
79
  attributes = Regexp.last_match(1)
78
80
  "<script nonce=\"#{nonce}\"#{attributes}>"
79
81
  end
80
-
81
- document.output = output
82
-
83
- hashes = []
84
- output.scan(%r{<script(?![^>]*\bsrc=)[^>]*>(.*?)</script>}mi) do |match|
85
- content = match.first
86
- next if content.nil? || content.empty?
87
-
88
- hashes << Digest::SHA256.base64digest(content)
82
+ # A template that printed page.csp_nonce before the page had one left
83
+ # nonce="", which authorises nothing.
84
+ document.output = output.gsub(/(<(?:script|style)\b[^>]*\bnonce=)""/i) do
85
+ "#{Regexp.last_match(1)}\"#{nonce}\""
89
86
  end
90
-
91
- hashes.uniq!
92
- document.data["csp_hashes"] = hashes
93
-
94
- return unless site.respond_to?(:data)
95
-
96
- site.data["csp"] ||= {}
97
- site.data["csp"]["hashes"] ||= {}
98
-
99
- key = document_key(document)
100
-
101
- site.data["csp"]["hashes"][key] = hashes if key
102
87
  end
103
88
  end
104
89
  end
105
90
  end
106
91
 
107
92
  %i[pages documents].each do |target|
108
- Jekyll::Hooks.register target, :pre_render do |document|
109
- Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
93
+ Jekyll::Hooks.register target, :pre_render do |document, payload|
94
+ nonce = Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
95
+ # A page another generator creates after this one runs, such as a notebook
96
+ # page, only gets its nonce here. Jekyll has already copied a page's data
97
+ # into the hash its templates read, so page.csp_nonce rendered empty there.
98
+ page_data = payload && payload["page"]
99
+ page_data["csp_nonce"] ||= nonce if nonce && page_data.is_a?(Hash)
110
100
  end
111
101
 
112
102
  Jekyll::Hooks.register target, :post_render do |document|
113
- Datalog::Security::CspGenerator.compute_hashes(document)
103
+ Datalog::Security::CspGenerator.add_nonces(document)
114
104
  end
115
105
  end
@@ -1,17 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_bibliography` tag when the datalog-citations
5
+ # plugin is not enabled, so a layout that uses the tag still builds. The plugin
6
+ # registers the real tag and sets `datalog_bibliography` on the pages it
7
+ # handles; a page that sets the key by hand without the plugin gets nothing and
8
+ # a build warning, where it used to get a "coming soon" notice.
4
9
  class DatalogBibliographyTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
- bib_data = page["datalog_bibliography"]
8
- return "" unless bib_data
12
+ return "" unless page["datalog_bibliography"]
9
13
 
10
- <<~HTML
11
- <div class="datalog-bibliography-placeholder">
12
- <p>Bibliography feature coming soon.</p>
13
- </div>
14
- HTML
14
+ Jekyll.logger.warn("datalog-citations", "#{page['path']} sets datalog_bibliography, " \
15
+ "but datalog_plugins.enabled does not list datalog-citations")
16
+ ""
15
17
  end
16
18
  end
17
19
  end
@@ -1,16 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_comments` tag when the datalog-comments plugin is
5
+ # not enabled, so a layout that uses the tag still builds. The plugin registers
6
+ # the real tag and sets `datalog_comments` on the pages it handles; a page that
7
+ # sets the key by hand without the plugin gets nothing and a build warning,
8
+ # where it used to get a "coming soon" notice.
4
9
  class DatalogCommentsTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
12
  return "" unless page["datalog_comments"]
8
13
 
9
- <<~HTML
10
- <div class="datalog-comments-placeholder">
11
- <p>Comments feature coming soon.</p>
12
- </div>
13
- HTML
14
+ Jekyll.logger.warn("datalog-comments", "#{page['path']} sets datalog_comments, " \
15
+ "but datalog_plugins.enabled does not list datalog-comments")
16
+ ""
14
17
  end
15
18
  end
16
19
  end
@@ -1,18 +1,19 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Jekyll
4
+ # Stands in for the `datalog_slides` tag when the datalog-slides plugin is not
5
+ # enabled, so a layout that uses the tag still builds. The plugin registers the
6
+ # real tag and sets `datalog_slides` on the pages it handles; a page that sets
7
+ # the key by hand without the plugin gets nothing and a build warning, where it
8
+ # used to get a "coming soon" notice showing its raw configuration.
4
9
  class DatalogSlidesTag < Liquid::Tag
5
10
  def render(context)
6
11
  page = context.registers[:page]
7
- slides_data = page["datalog_slides"]
8
- return "" unless slides_data
12
+ return "" unless page["datalog_slides"]
9
13
 
10
- # Render a simple placeholder or embed
11
- <<~HTML
12
- <div class="datalog-slides-placeholder">
13
- <p>Slides feature coming soon. Configured: #{slides_data}</p>
14
- </div>
15
- HTML
14
+ Jekyll.logger.warn("datalog-slides", "#{page['path']} sets datalog_slides, " \
15
+ "but datalog_plugins.enabled does not list datalog-slides")
16
+ ""
16
17
  end
17
18
  end
18
19
  end
data/_plugins/i18n.rb CHANGED
@@ -116,6 +116,11 @@ class TranslateTag < Liquid::Tag
116
116
  # quote into the first option name, or interpolation silently fails.
117
117
  SYNTAX = /\A\s*(['"]?)(\w[\w.-]*)\1(.*)?\z/m
118
118
 
119
+ # `name: value` pairs, separated by commas or spaces. A quoted value may
120
+ # contain commas: splitting the markup on every comma cut `name: "Doe, Jane"`
121
+ # in two.
122
+ OPTION = /(\w+)\s*:\s*("(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|[^\s,]+)/
123
+
119
124
  def initialize(tag_name, markup, tokens)
120
125
  super
121
126
  raise Liquid::SyntaxError, "Syntax Error in 't' - Valid syntax: t key [arg: value]" unless markup.strip =~ SYNTAX
@@ -134,14 +139,8 @@ class TranslateTag < Liquid::Tag
134
139
  def parse_options(markup, context)
135
140
  return {} unless markup && !markup.strip.empty?
136
141
 
137
- tokens = markup.strip.split(",").map(&:strip)
138
- tokens.each_with_object({}) do |token, memo|
139
- next if token.empty?
140
-
141
- if token.include?(":")
142
- key, value = token.split(":", 2)
143
- memo[key.strip.to_sym] = context.evaluate(Liquid::Expression.parse(value.strip))
144
- end
142
+ markup.scan(OPTION).to_h do |key, value|
143
+ [key.to_sym, context.evaluate(Liquid::Expression.parse(value))]
145
144
  end
146
145
  end
147
146
  end
@@ -34,6 +34,18 @@ module Jekyll
34
34
  }.freeze
35
35
  RASTER_EXTENSIONS = %w[.jpg .jpeg .png].freeze
36
36
 
37
+ # The one image a page fetches early: the first in its post or page content,
38
+ # unless the author made it lazy or the page already preloads an image, as
39
+ # the hero does. The first <img> anywhere used to be marked both lazy and
40
+ # high priority; on the home page that was a post card below the hero, and
41
+ # on tutorials the thumbnail of a related post at the bottom.
42
+ def priority_image(fragment)
43
+ return if fragment.at_css('link[rel="preload"][as="image"]')
44
+
45
+ image = fragment.css(".post-content img, .page-content img").find { |img| img["data-no-optimize"] != "true" }
46
+ image unless image.nil? || image["loading"] == "lazy"
47
+ end
48
+
37
49
  def process(document)
38
50
  return unless document.output_ext == ".html"
39
51
  return if document.output.nil? || document.output.empty?
@@ -46,18 +58,17 @@ module Jekyll
46
58
  manifest = site&.data&.fetch("datalog_responsive_images", {}) || {}
47
59
  image_config = site&.config&.fetch("datalog_image_config", {}) || {}
48
60
  optimized = false
49
- first_priority_assigned = false
61
+ priority = priority_image(fragment)
50
62
 
51
63
  fragment.css("img").each do |img|
52
64
  next if img["data-no-optimize"] == "true"
53
65
 
54
- img["loading"] ||= "lazy"
55
- img["decoding"] ||= "async"
56
-
57
- unless first_priority_assigned
66
+ if img == priority
58
67
  img["fetchpriority"] ||= "high"
59
- first_priority_assigned = true
68
+ else
69
+ img["loading"] ||= "lazy"
60
70
  end
71
+ img["decoding"] ||= "async"
61
72
 
62
73
  normalized_src = normalize_src(img["src"], site)
63
74
  picture_entry = manifest[normalized_src]
@@ -19,8 +19,11 @@ module MathPreprocessor
19
19
  ].freeze
20
20
 
21
21
  INLINE_PATTERNS = [
22
+ # Pandoc's rule for inline math, so prices and shell variables stay text:
23
+ # the opening $ is followed by a non-space, the closing $ follows a
24
+ # non-space and is not followed by a digit, and a blank line ends it.
22
25
  {
23
- regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>[^$]+?)(?<close>\$)(?!\$)/m,
26
+ regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
24
27
  tag: "span"
25
28
  },
26
29
  {
@@ -29,6 +32,20 @@ module MathPreprocessor
29
32
  }
30
33
  ].freeze
31
34
 
35
+ # Code shows dollar signs literally (shell and R variables, amounts in SQL),
36
+ # so fenced blocks, highlight tags, <pre>/<code> elements and inline code
37
+ # spans are set aside before looking for math and put back afterwards.
38
+ CODE_PATTERNS = [
39
+ /^([ \t]*)(`{3,}|~{3,})[^\n]*\n.*?(?:^\1\2[ \t]*$|\z)/m,
40
+ /\{%-?\s*highlight\b.*?\{%-?\s*endhighlight\s*-?%\}/m,
41
+ %r{<(pre|code)\b[^>]*>.*?</\1>}mi,
42
+ /(?<!`)(`+)(?!`)(?:(?!\n[ \t]*\n).)+?(?<!`)\1(?!`)/m
43
+ ].freeze
44
+
45
+ # NUL marks masked code, since page content never contains it. It is written as
46
+ # an escape: a raw NUL byte in the source stopped RuboCop from parsing the file.
47
+ PLACEHOLDER = /\x00(\d+)\x00/
48
+
32
49
  class Processor
33
50
  attr_reader :expressions
34
51
 
@@ -40,9 +57,18 @@ module MathPreprocessor
40
57
  def process
41
58
  return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
42
59
 
43
- processed = @content.dup
60
+ code = []
61
+ processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
62
+ text.gsub(pattern) do |match|
63
+ code << match
64
+ "\x00#{code.size - 1}\x00"
65
+ end
66
+ end
44
67
  processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
45
- apply_patterns(processed, INLINE_PATTERNS, display: false)
68
+ processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
69
+ # A segment set aside can contain the placeholder of an earlier one.
70
+ processed = processed.gsub(PLACEHOLDER) { code[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
71
+ processed
46
72
  end
47
73
 
48
74
  private
@@ -66,8 +92,12 @@ module MathPreprocessor
66
92
  cleaned_source = cleanup_source(latex)
67
93
  record_expression(cleaned_source, alt_text)
68
94
 
95
+ # ARIA forbids aria-label on an element with no role, such as a plain span.
96
+ # axe let it pass while the span held the raw LaTeX as text, and failed it
97
+ # once MathJax rendered the expression; the math role allows the label.
69
98
  attributes = {
70
99
  "class" => display ? "math-expression math-expression--source" : "math-expression-inline math-expression--source",
100
+ "role" => "math",
71
101
  "data-math-alt" => alt_text,
72
102
  "data-math-source" => cleaned_source,
73
103
  "aria-label" => alt_text,
@@ -165,6 +195,9 @@ module MathPreprocessor
165
195
  def apply(document)
166
196
  return unless document.respond_to?(:content)
167
197
  return unless document.respond_to?(:output_ext) && document.output_ext == ".html"
198
+ # A page that opts out of math rendering (`math: false` or `mathjax: false`)
199
+ # keeps its dollar signs and TeX-looking text verbatim.
200
+ return if document.respond_to?(:data) && (document.data["math"] == false || document.data["mathjax"] == false)
168
201
 
169
202
  content = document.content
170
203
  return unless content&.match?(/\$|\\\(|\\\[|\\begin\{/)
@@ -7,6 +7,7 @@ require "fileutils"
7
7
  require "cgi"
8
8
  require "loofah"
9
9
  require "base64"
10
+ require_relative "rouge_highlight_filter"
10
11
 
11
12
  module Datalog
12
13
  module NotebookRenderer
@@ -94,14 +95,30 @@ module Datalog
94
95
  metadata: build_sanitization_metadata(meta, cell_index: cell_index))
95
96
  return if sanitized.to_s.strip.empty?
96
97
 
97
- %(<section class="notebook-cell notebook-cell--markdown">\n#{sanitized}\n</section>)
98
+ %(<section class="notebook-cell notebook-cell--markdown">\n#{demote_headings(sanitized.to_s)}\n</section>)
99
+ end
100
+
101
+ # The notebook layout gives the page its <h1>, and a notebook's first
102
+ # markdown cell usually repeats the title as `# Title`. Each heading moves
103
+ # down a level (h1 to h2, and so on to h6), which keeps one <h1> on the
104
+ # page and the cells' own outline under it.
105
+ def demote_headings(html)
106
+ html.gsub(%r{<(/?)h([1-5])(?=[\s>])}i) { "<#{Regexp.last_match(1)}h#{Regexp.last_match(2).to_i + 1}" }
98
107
  end
99
108
 
100
109
  # Class names mirror the theme stylesheet (`.notebook-cell--input`,
101
110
  # `.notebook-cell__code` and `.notebook-cell__outputs` in _sass/_components.scss).
102
111
  def render_code(cell, source, metadata, site, cell_index)
103
112
  language = cell.dig("metadata", "language") || metadata[:language] || "text"
104
- code_html = %(<pre class="notebook-cell__code"><code class="language-#{language}">#{CGI.escapeHTML(source)}</code></pre>)
113
+ # The language comes from the notebook file, so it is cut down to the
114
+ # characters a class name can hold before it goes into the attribute.
115
+ language_class = language.to_s.gsub(/[^\w+#.-]/, "")
116
+ # Rouge highlights the cell as the site builds, as kramdown does for code
117
+ # blocks, and escapes it.
118
+ highlighted = Jekyll::RougeHighlightFilter.highlight(source, language_class)
119
+ pre_attributes = %(class="highlight notebook-cell__code" tabindex="0")
120
+ code_attributes = %(class="language-#{language_class}")
121
+ code_html = %(<pre #{pre_attributes}><code #{code_attributes}>#{highlighted}</code></pre>)
105
122
  base_metadata = metadata.respond_to?(:merge) ? metadata.merge(language: language) : { language: language }
106
123
  outputs_html = render_outputs(Array(cell["outputs"]), site: site, cell_index: cell_index, metadata: base_metadata)
107
124
  outputs_html = %(\n<div class="notebook-cell__outputs">\n#{outputs_html}\n</div>) unless outputs_html.empty?
@@ -157,11 +174,13 @@ module Datalog
157
174
  def image_output_html(output)
158
175
  data = output["data"] || {}
159
176
 
177
+ # Jupyter writes base64 image data split over lines or ending in a newline,
178
+ # and the data URI check rejects whitespace, which dropped those images.
160
179
  if (png = data["image/png"])
161
- html = %(<img src="data:image/png;base64,#{Array(png).join}" alt="Notebook output" />)
180
+ html = %(<img src="data:image/png;base64,#{Array(png).join.gsub(/\s+/, '')}" alt="Notebook output" />)
162
181
  return [html, "image/png"]
163
182
  elsif (jpeg = data["image/jpeg"])
164
- html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join}" alt="Notebook output" />)
183
+ html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join.gsub(/\s+/, '')}" alt="Notebook output" />)
165
184
  return [html, "image/jpeg"]
166
185
  elsif (svg = data["image/svg+xml"])
167
186
  encoded = Base64.strict_encode64(Array(svg).join)
@@ -629,31 +648,9 @@ module Datalog
629
648
  nil
630
649
  end
631
650
 
632
- def patch_jupyter_converter!
633
- return unless defined?(JekyllJupyterNotebook::Converter)
634
- return if @converter_fallback_applied
635
-
636
- fallback = Module.new do
637
- def convert(content)
638
- super
639
- rescue Errno::ENOENT, StandardError => e
640
- Jekyll.logger.warn("notebook converter", "primary conversion failed: #{e.message}; using fallback renderer")
641
- Datalog::NotebookRenderer.render_from_raw(content) || ""
642
- end
643
- end
644
-
645
- JekyllJupyterNotebook::Converter.prepend(fallback)
646
- @converter_fallback_applied = true
647
- rescue StandardError => e
648
- Jekyll.logger.warn("notebook converter", "failed to apply converter fallback: #{e.message}")
649
- end
650
651
  end
651
652
  end
652
653
 
653
- Jekyll::Hooks.register :site, :after_init do |_site|
654
- Datalog::NotebookRenderer.patch_jupyter_converter!
655
- end
656
-
657
654
  module Jekyll
658
655
  # Converts Jupyter notebooks into HTML pages and downloadable assets.
659
656
  class NotebookConverter < Generator
@@ -671,8 +668,6 @@ module Jekyll
671
668
  return
672
669
  end
673
670
 
674
- ensure_dependency
675
-
676
671
  files = notebook_files
677
672
  logger.debug("notebook converter", "located #{files.size} notebooks")
678
673
  return if files.empty?
@@ -722,21 +717,6 @@ module Jekyll
722
717
  value.sub(%r{/+$}, "")
723
718
  end
724
719
 
725
- # Notebook pages do not need the gem (see #convert_notebook). When it is
726
- # present, its converter is given the same renderer as a fallback so the
727
- # gem's own `.ipynb` handling and `{% jupyter_notebook %}` tag keep working
728
- # on machines without a `jupyter` executable.
729
- def ensure_dependency
730
- return true if defined?(JekyllJupyterNotebook::Converter)
731
-
732
- require "jekyll-jupyter-notebook"
733
- Datalog::NotebookRenderer.patch_jupyter_converter!
734
- true
735
- rescue LoadError => e
736
- logger.debug("notebook converter", "jekyll-jupyter-notebook not loaded: #{e.message}")
737
- false
738
- end
739
-
740
720
  def notebook_files
741
721
  glob = File.join(site.source, config["source"], "**", "*.ipynb")
742
722
  Dir.glob(glob)
@@ -149,7 +149,9 @@ end
149
149
 
150
150
  module Datalog
151
151
  module PluginLoaderHooks
152
- HOOK_SCOPES = %i[pages documents posts].freeze
152
+ # Posts are documents: Jekyll fires a post's `posts` hooks and then its
153
+ # `documents` hooks, so registering both ran every plugin hook twice a post.
154
+ HOOK_SCOPES = %i[pages documents].freeze
153
155
 
154
156
  module_function
155
157
 
@@ -6,7 +6,13 @@ module Jekyll
6
6
  priority :low
7
7
 
8
8
  def generate(site)
9
- publications_data = site.data["publications"] ||= {}
9
+ publications_data = site.data["publications"]
10
+ # _data/publications.yml is normally a map with `settings` and
11
+ # `manual_entries`; a plain list of entries is read as the manual
12
+ # entries rather than stopping the build with a TypeError.
13
+ publications_data = { "manual_entries" => publications_data } if publications_data.is_a?(Array)
14
+ publications_data = {} unless publications_data.is_a?(Hash)
15
+ site.data["publications"] = publications_data
10
16
  settings = publications_data["settings"] || {}
11
17
  config_source = site.config.dig("theme_options", "publications", "bibtex_source")
12
18
  bibtex_source = settings["bibtex_source"] || config_source
@@ -22,7 +28,7 @@ module Jekyll
22
28
  end
23
29
  end
24
30
 
25
- manual_entries = publications_data["manual_entries"] || []
31
+ manual_entries = Array(publications_data["manual_entries"]).select { |entry| entry.is_a?(Hash) }
26
32
  combined = (imported_entries + manual_entries).map { |entry| normalize_entry(entry) }
27
33
 
28
34
  academic_citations = site.data.dig("academic", "citations", "per_publication") || {}
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "rouge"
4
+
5
+ # `rouge_highlight` highlights code with Rouge as the site builds, the way
6
+ # kramdown highlights fenced code blocks. Includes that print code passed to
7
+ # them, such as `components/api-function.html`, call it as
8
+ # `code | rouge_highlight: language` inside `<pre class="highlight"><code>`;
9
+ # the notebook converter calls `RougeHighlightFilter.highlight` for code cells.
10
+ # The result is escaped; a language Rouge does not know comes back as plain
11
+ # text.
12
+ module Jekyll
13
+ module RougeHighlightFilter
14
+ # kramdown's opening tag for a block Rouge highlighted.
15
+ KRAMDOWN_CODE_BLOCK = '<pre class="highlight">'
16
+
17
+ def self.highlight(code, language = nil)
18
+ return "" if code.nil?
19
+
20
+ source = code.to_s
21
+ lexer = Rouge::Lexer.find_fancy(language.to_s.strip.downcase, source) || Rouge::Lexers::PlainText
22
+ Rouge::Formatters::HTML.new.format(lexer.lex(source))
23
+ end
24
+
25
+ def rouge_highlight(code, language = nil)
26
+ RougeHighlightFilter.highlight(code, language)
27
+ end
28
+ end
29
+ end
30
+
31
+ Liquid::Template.register_filter(Jekyll::RougeHighlightFilter)
32
+
33
+ # A code block wider than the page scrolls, and a keyboard user can only scroll
34
+ # it once it takes focus. Prism made every block focusable in the browser; the
35
+ # blocks kramdown highlights get the attribute here, and the includes and the
36
+ # notebook converter write it themselves.
37
+ Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
38
+ block = Jekyll::RougeHighlightFilter::KRAMDOWN_CODE_BLOCK
39
+ next unless document.content&.include?(block)
40
+
41
+ document.content = document.content.gsub(block, '<pre class="highlight" tabindex="0">')
42
+ end
@@ -0,0 +1,30 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Datalog
4
+ # Collects the fenced code blocks of every page and document for the search
5
+ # index. The index template used to split `doc.content` on backticks, but
6
+ # Jekyll renders documents before pages, so by the time search.json rendered
7
+ # that content was HTML and every document's code list came out empty.
8
+ module SearchCodeBlocks
9
+ # An opening fence of three or more backticks or tildes with an optional
10
+ # language, the code, and a closing fence of the same characters.
11
+ FENCE = /^ {0,3}(`{3,}|~{3,})[ \t]*([^\s`~{]*)[^\n]*\n(.*?)^ {0,3}\1[ \t]*$/m
12
+
13
+ module_function
14
+
15
+ def extract(source)
16
+ text = source.to_s.gsub("\r\n", "\n")
17
+ text.scan(FENCE).map do |_fence, language, code|
18
+ { "language" => language.empty? ? "text" : language.downcase, "code" => code.chomp }
19
+ end
20
+ end
21
+ end
22
+ end
23
+
24
+ # Content is still the author's source before rendering starts. Collection
25
+ # docs only: site.documents also lists a collection's static files.
26
+ Jekyll::Hooks.register :site, :pre_render do |site|
27
+ (site.pages + site.collections.values.flat_map(&:docs)).each do |item|
28
+ item.data["search_code"] = Datalog::SearchCodeBlocks.extract(item.content)
29
+ end
30
+ end
@@ -1,66 +1,28 @@
1
1
  # frozen_string_literal: true
2
2
 
3
- begin
4
- require "unicode_normalize"
5
- UNICODE_NORMALIZE_SUPPORTED = true
6
- rescue LoadError
7
- UNICODE_NORMALIZE_SUPPORTED = false
8
- warn "[search_normalizer] unicode_normalize gem not available; falling back to basic normalization"
9
- end
10
-
11
3
  module Datalog
12
4
  module SearchFilters
13
5
  module_function
14
6
 
7
+ # Lower-cases text and strips combining marks, so "Café" and "cafe" match
8
+ # while letters outside ASCII ("ß", "ł", Cyrillic, CJK) are kept. The
9
+ # browser normalizes queries the same way (assets/js/search/utils.js), so
10
+ # the index and the query agree. String#unicode_normalize is part of Ruby:
11
+ # this file used to require it as if it were a gem, fail, and fall back to
12
+ # dropping every character outside ASCII.
15
13
  def normalize_search(input)
16
- value = input.to_s
14
+ # Invalid byte sequences are dropped first: unicode_normalize raises on them.
15
+ value = input.to_s.scrub("")
17
16
  return "" if value.empty?
18
17
 
19
- normalized = if UNICODE_NORMALIZE_SUPPORTED && value.respond_to?(:unicode_normalize)
20
- value.unicode_normalize(:nfkd).gsub(/\p{Mn}/, "")
21
- else
22
- transliterate(value)
23
- end
24
-
25
- normalized.downcase.strip
26
- rescue StandardError
27
- input.to_s.downcase
28
- end
29
-
30
- def transliterate(value)
31
- value.encode("ASCII", fallback: lambda { |char|
32
- approximate_character(char)
33
- }, invalid: :replace, undef: :replace, replace: "")
34
- rescue Encoding::UndefinedConversionError, Encoding::InvalidByteSequenceError
35
- value
36
- end
37
-
38
- def approximate_character(char)
39
- @transliteration_map ||= build_transliteration_map
40
- @transliteration_map.fetch(char, "")
41
- end
42
-
43
- def build_transliteration_map
44
- basic_map = {}
45
-
46
- accents = {
47
- "ÀÁÂÃÄÅàáâãäå" => "a",
48
- "ÈÉÊËèéêë" => "e",
49
- "ÌÍÎÏìíîï" => "i",
50
- "ÒÓÔÕÖØòóôõöø" => "o",
51
- "ÙÚÛÜùúûü" => "u",
52
- "Çç" => "c",
53
- "Ññ" => "n",
54
- "Ýýÿ" => "y",
55
- "Ææ" => "ae",
56
- "Œœ" => "oe"
57
- }
58
-
59
- accents.each do |chars, replacement|
60
- chars.each_char { |char| basic_map[char] = replacement }
18
+ stripped = begin
19
+ value.unicode_normalize(:nfkd).gsub(/\p{Mn}/, "")
20
+ rescue Encoding::CompatibilityError
21
+ # Text in an encoding other than Unicode cannot be normalized, so it is
22
+ # indexed as it is.
23
+ value
61
24
  end
62
-
63
- basic_map
25
+ stripped.downcase.strip
64
26
  end
65
27
 
66
28
  def normalize_search_array(values)