datalog-theme 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (203) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +188 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +25 -17
  5. data/_data/cdn-integrity.yml +0 -30
  6. data/_data/i18n/en.yml +297 -0
  7. data/_data/i18n/es.yml +297 -0
  8. data/_data/i18n/pt.yml +297 -0
  9. data/_data/js_manifest.json +16 -0
  10. data/_includes/analytics/dashboard.html +3 -1
  11. data/_includes/components/api-function.html +20 -1
  12. data/_includes/components/author-bio.html +21 -11
  13. data/_includes/components/author-list.html +32 -0
  14. data/_includes/components/citation-tools.html +33 -25
  15. data/_includes/components/comments-thread.html +90 -0
  16. data/_includes/components/contact-form.html +120 -0
  17. data/_includes/components/correction-report.html +82 -0
  18. data/_includes/components/enhanced-code-block.html +1 -1
  19. data/_includes/components/enhanced-toc.html +6 -7
  20. data/_includes/components/license-link.html +13 -0
  21. data/_includes/components/license-notice.html +28 -0
  22. data/_includes/components/moderation-inbox.html +139 -0
  23. data/_includes/components/reactions.html +49 -0
  24. data/_includes/components/reading-list.html +31 -0
  25. data/_includes/components/reading-mode-toggle.html +65 -0
  26. data/_includes/components/reading-state-bookmark.html +27 -0
  27. data/_includes/components/reading-state-panel.html +71 -0
  28. data/_includes/components/reproducibility.html +61 -0
  29. data/_includes/components/responsive-image.html +3 -3
  30. data/_includes/components/revision-history.html +39 -0
  31. data/_includes/components/revision-notice.html +35 -0
  32. data/_includes/components/series-nav.html +64 -0
  33. data/_includes/components/subscribe-form.html +88 -0
  34. data/_includes/components/subscription-manage.html +68 -0
  35. data/_includes/components/webmentions.html +48 -0
  36. data/_includes/csp-meta.html +135 -11
  37. data/_includes/footer.html +22 -20
  38. data/_includes/head.html +111 -63
  39. data/_includes/header/navigation.html +12 -15
  40. data/_includes/header.html +28 -19
  41. data/_includes/layouts/default/article.html +9 -7
  42. data/_includes/meta/dynamic-services-config.html +14 -0
  43. data/_includes/meta/math-config.html +18 -10
  44. data/_includes/meta/person-json.html +27 -0
  45. data/_includes/meta/publisher.html +45 -0
  46. data/_includes/meta/schema.html +67 -30
  47. data/_includes/meta/scholarly.html +121 -0
  48. data/_includes/meta/scripts-loader.html +18 -32
  49. data/_includes/meta/webmention-discovery.html +14 -0
  50. data/_includes/post/related-posts.html +4 -7
  51. data/_includes/scripts.html +57 -0
  52. data/_includes/search/index-data.json +9 -34
  53. data/_layouts/dataset.html +5 -3
  54. data/_layouts/default.html +16 -8
  55. data/_layouts/notebook.html +1 -0
  56. data/_layouts/package.html +4 -2
  57. data/_layouts/page.html +15 -0
  58. data/_layouts/portfolio.html +1 -0
  59. data/_layouts/post.html +102 -23
  60. data/_layouts/project.html +4 -3
  61. data/_layouts/research.html +20 -8
  62. data/_plugins/analytics_dashboard.rb +9 -3
  63. data/_plugins/authors.rb +133 -0
  64. data/_plugins/config_validator.rb +231 -23
  65. data/_plugins/critical_css_check.rb +42 -0
  66. data/_plugins/csp_generator.rb +18 -28
  67. data/_plugins/datalog_bibliography.rb +9 -7
  68. data/_plugins/datalog_comments.rb +8 -5
  69. data/_plugins/datalog_slides.rb +9 -8
  70. data/_plugins/i18n.rb +13 -12
  71. data/_plugins/image_optimizer.rb +241 -178
  72. data/_plugins/licenses.rb +135 -0
  73. data/_plugins/math_preprocessor.rb +59 -7
  74. data/_plugins/notebook_converter.rb +23 -4
  75. data/_plugins/plugin_loader.rb +3 -1
  76. data/_plugins/publications_generator.rb +8 -2
  77. data/_plugins/references.rb +238 -0
  78. data/_plugins/reproducibility.rb +150 -0
  79. data/_plugins/revisions.rb +101 -0
  80. data/_plugins/rouge_highlight_filter.rb +42 -0
  81. data/_plugins/scholarly.rb +50 -0
  82. data/_plugins/search_code_blocks.rb +30 -0
  83. data/_plugins/search_normalizer.rb +15 -53
  84. data/_plugins/search_pages.rb +3 -4
  85. data/_plugins/series.rb +104 -0
  86. data/_plugins/statements.rb +87 -0
  87. data/_sass/_academic-dashboard.scss +262 -0
  88. data/_sass/_base.scss +20 -1
  89. data/_sass/_comments-thread.scss +159 -0
  90. data/_sass/_components.scss +64 -1153
  91. data/_sass/_features.scss +17 -0
  92. data/_sass/_layout.scss +385 -1
  93. data/_sass/_mathematical.scss +27 -0
  94. data/_sass/_moderation.scss +222 -0
  95. data/_sass/_notebooks.scss +322 -0
  96. data/_sass/_open-science-badges.scss +56 -0
  97. data/_sass/{_phase1-enhancements.scss → _post-components.scss} +5 -3
  98. data/_sass/_print.scss +291 -0
  99. data/_sass/_reactions.scss +89 -0
  100. data/_sass/_reading-state.scss +290 -0
  101. data/_sass/_search-page.scss +530 -0
  102. data/_sass/_search.scss +46 -0
  103. data/_sass/_service-forms.scss +204 -0
  104. data/_sass/_subscriptions.scss +140 -0
  105. data/_sass/_syntax-highlighting.scss +212 -97
  106. data/_sass/_theme.scss +40 -19
  107. data/_sass/_typography.scss +130 -0
  108. data/_sass/_utilities.scss +5 -0
  109. data/_sass/_variables.scss +6 -0
  110. data/_sass/_webmentions.scss +125 -0
  111. data/assets/css/main.scss +14 -0
  112. data/assets/js/dist/academic.js +1 -1
  113. data/assets/js/dist/analytics-dashboard.js +1 -1
  114. data/assets/js/dist/chunks/chunk-2DYDWUFX.js +1 -0
  115. data/assets/js/dist/chunks/chunk-PATLC23F.js +1 -0
  116. data/assets/js/dist/chunks/chunk-V7734B2G.js +1 -0
  117. data/assets/js/dist/comments.js +2 -0
  118. data/assets/js/dist/contact.js +1 -0
  119. data/assets/js/dist/core.js +1 -1
  120. data/assets/js/dist/corrections.js +1 -0
  121. data/assets/js/dist/loader.js +1 -1
  122. data/assets/js/dist/math.js +1 -1
  123. data/assets/js/dist/moderation.js +1 -0
  124. data/assets/js/dist/notebook.js +1 -1
  125. data/assets/js/dist/reactions.js +1 -0
  126. data/assets/js/dist/reading-state.js +1 -0
  127. data/assets/js/dist/search.js +1 -1
  128. data/assets/js/dist/sources.json +52 -0
  129. data/assets/js/dist/subscriptions.js +1 -0
  130. data/assets/js/dist/visualizations.js +11 -2
  131. data/assets/js/dist/webmentions.js +1 -0
  132. data/assets/js/loader.js +37 -1
  133. data/datalog-theme.gemspec +35 -23
  134. data/lib/datalog/cli.rb +43 -15
  135. data/lib/datalog/critical_css.rb +168 -0
  136. data/lib/datalog/plugin_system/dependency_resolver.rb +0 -2
  137. data/lib/datalog/plugins/comments.rb +33 -3
  138. data/lib/datalog/theme/installed_files.rb +113 -0
  139. data/lib/datalog/theme/package.rb +57 -0
  140. data/lib/datalog/theme/repository_checkout.rb +94 -0
  141. data/lib/datalog/theme/version.rb +5 -1
  142. data/lib/datalog/warning_filter.rb +5 -11
  143. data/lib/datalog-theme.rb +6 -0
  144. metadata +96 -147
  145. data/_data/academic.yml +0 -217
  146. data/_data/config/author.yml +0 -121
  147. data/_data/datasets.yml +0 -28
  148. data/_data/js_meta.json +0 -371
  149. data/_data/navigation.yml +0 -145
  150. data/_data/projects.yml +0 -41
  151. data/_data/publications.yml +0 -28
  152. data/_data/social.yml +0 -73
  153. data/_data/visualizations.yml +0 -51
  154. data/_includes/components/advanced-search.html +0 -682
  155. data/_includes/components/bookmark-system.html +0 -96
  156. data/_includes/components/comments.html +0 -244
  157. data/_includes/components/content-recommendations.html +0 -228
  158. data/_includes/components/email-preferences.html +0 -200
  159. data/_includes/components/enhanced-metadata.html +0 -228
  160. data/_includes/components/language-switcher.html +0 -396
  161. data/_includes/components/navigation-enhancements.html +0 -454
  162. data/_includes/components/newsletter-signup.html +0 -178
  163. data/_includes/components/popular-posts.html +0 -233
  164. data/_includes/components/reading-progress.html +0 -133
  165. data/_includes/components/reading-time.html +0 -121
  166. data/_includes/components/series-navigation.html +0 -124
  167. data/_includes/components/social-proof.html +0 -34
  168. data/_includes/components/user-preferences.html +0 -566
  169. data/_includes/meta/syntax-config.html +0 -19
  170. data/_layouts/archive.html +0 -282
  171. data/_layouts/post-sidebar.html +0 -183
  172. data/_sass/_phase3-enhancements.scss +0 -874
  173. data/_sass/_phase4-enhancements.scss +0 -1214
  174. data/_sass/_phase5-enhancements.scss +0 -414
  175. data/assets/js/academic.js +0 -262
  176. data/assets/js/analytics-dashboard.js +0 -382
  177. data/assets/js/core/dark-mode.js +0 -79
  178. data/assets/js/core/github-cards.js +0 -123
  179. data/assets/js/core/language-filter.js +0 -69
  180. data/assets/js/core/navigation.js +0 -184
  181. data/assets/js/core/scroll-progress.js +0 -45
  182. data/assets/js/core/search-hotkeys.js +0 -62
  183. data/assets/js/core/skip-links.js +0 -62
  184. data/assets/js/dist/manifest.json +0 -22
  185. data/assets/js/dist/meta.json +0 -371
  186. data/assets/js/main.js +0 -23
  187. data/assets/js/math.js +0 -818
  188. data/assets/js/notebook.js +0 -158
  189. data/assets/js/search/analytics.js +0 -91
  190. data/assets/js/search/app.js +0 -271
  191. data/assets/js/search/autocomplete.js +0 -120
  192. data/assets/js/search/engine.js +0 -260
  193. data/assets/js/search/filters.js +0 -45
  194. data/assets/js/search/render.js +0 -217
  195. data/assets/js/search/utils.js +0 -99
  196. data/assets/js/search.js +0 -354
  197. data/assets/js/visualizations.js +0 -816
  198. data/assets/publications/datalog-publications.bib +0 -8
  199. data/assets/publications/datalog-publications.ris +0 -9
  200. data/assets/publications/publications.bib +0 -30
  201. data/assets/templates/diogo-ribeiro-cv.md +0 -31
  202. data/assets/templates/diogo-ribeiro-cv.tex +0 -32
  203. data/lib/datalog/theme/theme.rb +0 -18
@@ -0,0 +1,135 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "time"
4
+
5
+ module Datalog
6
+ # The licence of a page's text and figures, and of its code samples, for
7
+ # the reuse notice, the JSON-LD and the head's rel="license" link:
8
+ #
9
+ # content_license: CC-BY-4.0 # _config.yml: the default for every article
10
+ #
11
+ # license: CC-BY-SA-4.0 # front matter: this page's own
12
+ # license: false # front matter: none for this page
13
+ # license: # front matter: a licence the theme does not know
14
+ # name: Open Government Licence v3.0
15
+ # url: https://www.nationalarchives.gov.uk/doc/open-government-licence/version/3/
16
+ # holder: The Lab
17
+ # year: 2026
18
+ # code_license: MIT # the code samples, when their licence differs
19
+ #
20
+ # The repository's LICENSE covers the theme's software; these settings are
21
+ # for what a site publishes. Datasets and packages carry their own
22
+ # `license` and never take the site's default.
23
+ module Licenses
24
+ module_function
25
+
26
+ # SPDX identifiers, with the name readers see and the licence text.
27
+ KNOWN = {
28
+ "CC-BY-4.0" => ["CC BY 4.0", "https://creativecommons.org/licenses/by/4.0/"],
29
+ "CC-BY-SA-4.0" => ["CC BY-SA 4.0", "https://creativecommons.org/licenses/by-sa/4.0/"],
30
+ "CC-BY-ND-4.0" => ["CC BY-ND 4.0", "https://creativecommons.org/licenses/by-nd/4.0/"],
31
+ "CC-BY-NC-4.0" => ["CC BY-NC 4.0", "https://creativecommons.org/licenses/by-nc/4.0/"],
32
+ "CC-BY-NC-SA-4.0" => ["CC BY-NC-SA 4.0", "https://creativecommons.org/licenses/by-nc-sa/4.0/"],
33
+ "CC-BY-NC-ND-4.0" => ["CC BY-NC-ND 4.0", "https://creativecommons.org/licenses/by-nc-nd/4.0/"],
34
+ "CC0-1.0" => ["CC0 1.0", "https://creativecommons.org/publicdomain/zero/1.0/"],
35
+ "MIT" => ["MIT", "https://spdx.org/licenses/MIT.html"],
36
+ "Apache-2.0" => ["Apache 2.0", "https://spdx.org/licenses/Apache-2.0.html"],
37
+ "BSD-2-Clause" => ["BSD 2-Clause", "https://spdx.org/licenses/BSD-2-Clause.html"],
38
+ "BSD-3-Clause" => ["BSD 3-Clause", "https://spdx.org/licenses/BSD-3-Clause.html"],
39
+ "GPL-3.0-only" => ["GPL 3.0", "https://spdx.org/licenses/GPL-3.0-only.html"],
40
+ "GPL-3.0-or-later" => ["GPL 3.0 or later", "https://spdx.org/licenses/GPL-3.0-or-later.html"],
41
+ "LGPL-3.0-only" => ["LGPL 3.0", "https://spdx.org/licenses/LGPL-3.0-only.html"],
42
+ "AGPL-3.0-only" => ["AGPL 3.0", "https://spdx.org/licenses/AGPL-3.0-only.html"],
43
+ "MPL-2.0" => ["MPL 2.0", "https://spdx.org/licenses/MPL-2.0.html"],
44
+ "ISC" => ["ISC", "https://spdx.org/licenses/ISC.html"],
45
+ "Unlicense" => ["The Unlicense", "https://spdx.org/licenses/Unlicense.html"],
46
+ "all-rights-reserved" => ["All rights reserved", nil]
47
+ }.freeze
48
+
49
+ # Other spellings: "CC BY" and "cc0" take the current version; the GPL
50
+ # family without a suffix means "only", as SPDX reads it.
51
+ ALIASES = {
52
+ "CC-BY" => "CC-BY-4.0", "CC-BY-SA" => "CC-BY-SA-4.0", "CC-BY-ND" => "CC-BY-ND-4.0",
53
+ "CC-BY-NC" => "CC-BY-NC-4.0", "CC-BY-NC-SA" => "CC-BY-NC-SA-4.0", "CC-BY-NC-ND" => "CC-BY-NC-ND-4.0",
54
+ "CC0" => "CC0-1.0", "GPL-3.0" => "GPL-3.0-only", "LGPL-3.0" => "LGPL-3.0-only", "AGPL-3.0" => "AGPL-3.0-only"
55
+ }.freeze
56
+
57
+ # Collections whose pages carry their own licence and never the site's.
58
+ OWN_LICENSE = %w[datasets packages].freeze
59
+
60
+ def key(value)
61
+ value.to_s.strip.upcase.gsub(/[\s_]+/, "-")
62
+ end
63
+
64
+ LOOKUP = KNOWN.keys.to_h { |id| [key(id), id] }.merge(ALIASES.to_h { |from, to| [key(from), to] }).freeze
65
+
66
+ # The licence of the page's text and figures, or nil.
67
+ def content(page, site)
68
+ resolve(setting(page, "license", site, "content_license"), page, site)
69
+ end
70
+
71
+ # The licence of the page's code samples, or nil.
72
+ def code(page, site)
73
+ resolve(setting(page, "code_license", site, "code_license"), page, site)
74
+ end
75
+
76
+ # The page's own value, else the site's; `false` declines the site's.
77
+ def setting(page, page_key, site, site_key)
78
+ value = Authors.value(page, page_key)
79
+ return value unless value.nil?
80
+ return if OWN_LICENSE.include?(Authors.value(page, "collection").to_s)
81
+
82
+ Authors.value(site, site_key)
83
+ end
84
+
85
+ def resolve(value, page, site)
86
+ return if value.nil? || value == false || (value.is_a?(String) && value.strip.empty?)
87
+
88
+ given = value.is_a?(Hash) ? Authors.present(value) : { "name" => value.to_s.strip }
89
+ id = LOOKUP[key(given["id"] || given["name"])]
90
+ name, url = KNOWN[id] if id
91
+ # A name that is not itself an identifier is the label the page chose.
92
+ name = given["name"] if given["name"] && LOOKUP[key(given["name"])].nil?
93
+ licence = { "id" => id, "name" => name, "url" => given["url"] || url, "reserved" => id == "all-rights-reserved" }
94
+ licence.merge(holder(given, page, site)).compact
95
+ end
96
+
97
+ # The copyright holder and year: the page's, else the site default's,
98
+ # else the page's authors and its date.
99
+ def holder(given, page, site)
100
+ site_given = Authors.value(site, "content_license")
101
+ site_given = site_given.is_a?(Hash) ? Authors.present(site_given) : {}
102
+ names = Array(given["holder"] || site_given["holder"]).map(&:to_s).reject(&:empty?)
103
+ authors = Authors.authors(page, site).map { |author| author["name"] }
104
+ people = names.empty? || (names - authors).empty?
105
+ names = authors if names.empty?
106
+ year = year(given["year"] || site_given["year"] || Authors.value(page, "date"))
107
+ { "holders" => names, "holder" => (names.join(", ") unless names.empty?), "people" => people, "year" => year }
108
+ end
109
+
110
+ # A year, a date or a string naming either.
111
+ def year(value)
112
+ return value if value.is_a?(Integer)
113
+ return value.year if value.respond_to?(:year)
114
+
115
+ text = value.to_s.strip
116
+ return text.to_i if text.match?(/\A\d{4}\z/)
117
+
118
+ Time.parse(text).year unless text.empty?
119
+ rescue ArgumentError
120
+ nil
121
+ end
122
+ end
123
+
124
+ module LicenseFilters
125
+ def page_license(page)
126
+ Licenses.content(page, @context["site"])
127
+ end
128
+
129
+ def page_code_license(page)
130
+ Licenses.code(page, @context["site"])
131
+ end
132
+ end
133
+ end
134
+
135
+ Liquid::Template.register_filter(Datalog::LicenseFilters)
@@ -19,16 +19,43 @@ module MathPreprocessor
19
19
  ].freeze
20
20
 
21
21
  INLINE_PATTERNS = [
22
+ # Pandoc's rule for inline math, so prices and shell variables stay text:
23
+ # the opening $ is followed by a non-space, the closing $ follows a
24
+ # non-space and is not followed by a digit, and a blank line ends it.
22
25
  {
23
- regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>[^$]+?)(?<close>\$)(?!\$)/m,
26
+ regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
24
27
  tag: "span"
25
28
  },
29
+ # MathJax and KaTeX also render math with spaces inside the dollars, such
30
+ # as `$ \frac{a}{b} $`, which the rule above leaves out, so a page whose
31
+ # only math was written that way loaded no engine. Such a pair counts when
32
+ # its body holds a TeX command, a superscript or a subscript, which prices
33
+ # like `$ 5 or $ 10` do not.
34
+ {
35
+ regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<!\\)(?<close>\$)(?![$\d])/m,
36
+ tag: "span",
37
+ requires: /\\[a-zA-Z]+|[\^_]/
38
+ },
26
39
  {
27
40
  regex: /(?<open>\\\()(?<body>.+?)(?<close>\\\))/m,
28
41
  tag: "span"
29
42
  }
30
43
  ].freeze
31
44
 
45
+ # Code shows dollar signs literally (shell and R variables, amounts in SQL),
46
+ # so fenced blocks, highlight tags, <pre>/<code> elements and inline code
47
+ # spans are set aside before looking for math and put back afterwards.
48
+ CODE_PATTERNS = [
49
+ /^([ \t]*)(`{3,}|~{3,})[^\n]*\n.*?(?:^\1\2[ \t]*$|\z)/m,
50
+ /\{%-?\s*highlight\b.*?\{%-?\s*endhighlight\s*-?%\}/m,
51
+ %r{<(pre|code)\b[^>]*>.*?</\1>}mi,
52
+ /(?<!`)(`+)(?!`)(?:(?!\n[ \t]*\n).)+?(?<!`)\1(?!`)/m
53
+ ].freeze
54
+
55
+ # NUL marks masked code, since page content never contains it. It is written as
56
+ # an escape: a raw NUL byte in the source stopped RuboCop from parsing the file.
57
+ PLACEHOLDER = /\x00(\d+)\x00/
58
+
32
59
  class Processor
33
60
  attr_reader :expressions
34
61
 
@@ -40,13 +67,24 @@ module MathPreprocessor
40
67
  def process
41
68
  return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
42
69
 
43
- processed = @content.dup
70
+ @segments = []
71
+ processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
72
+ text.gsub(pattern) { |match| mask(match) }
73
+ end
44
74
  processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
45
- apply_patterns(processed, INLINE_PATTERNS, display: false)
75
+ processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
76
+ # A segment set aside can contain the placeholder of an earlier one.
77
+ processed = processed.gsub(PLACEHOLDER) { @segments[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
78
+ processed
46
79
  end
47
80
 
48
81
  private
49
82
 
83
+ def mask(segment)
84
+ @segments << segment
85
+ "\x00#{@segments.size - 1}\x00"
86
+ end
87
+
50
88
  def apply_patterns(text, patterns, display: false)
51
89
  patterns.reduce(text) do |result, pattern|
52
90
  result.gsub(pattern[:regex]) do |match|
@@ -55,8 +93,11 @@ module MathPreprocessor
55
93
  close = Regexp.last_match[:close]
56
94
 
57
95
  next match if body.nil? || body.strip.empty?
96
+ next match if pattern[:requires] && !pattern[:requires].match?(body)
58
97
 
59
- wrapper_for(match, body, open, close, pattern[:tag], display: display)
98
+ # Each wrapper is set aside like code. Otherwise a later pattern could
99
+ # pair a dollar sign inside it with one in the text that follows.
100
+ mask(wrapper_for(match, body, open, close, pattern[:tag], display: display))
60
101
  end
61
102
  end
62
103
  end
@@ -66,8 +107,12 @@ module MathPreprocessor
66
107
  cleaned_source = cleanup_source(latex)
67
108
  record_expression(cleaned_source, alt_text)
68
109
 
110
+ # ARIA forbids aria-label on an element with no role, such as a plain span.
111
+ # axe let it pass while the span held the raw LaTeX as text, and failed it
112
+ # once MathJax rendered the expression; the math role allows the label.
69
113
  attributes = {
70
114
  "class" => display ? "math-expression math-expression--source" : "math-expression-inline math-expression--source",
115
+ "role" => "math",
71
116
  "data-math-alt" => alt_text,
72
117
  "data-math-source" => cleaned_source,
73
118
  "aria-label" => alt_text,
@@ -165,9 +210,9 @@ module MathPreprocessor
165
210
  def apply(document)
166
211
  return unless document.respond_to?(:content)
167
212
  return unless document.respond_to?(:output_ext) && document.output_ext == ".html"
168
- # A page that opts out of math rendering (`math: false` or `mathjax: false`)
169
- # keeps its dollar signs and TeX-looking text verbatim.
170
- return if document.respond_to?(:data) && (document.data["math"] == false || document.data["mathjax"] == false)
213
+ # A page that opts out of math rendering keeps its dollar signs and
214
+ # TeX-looking text verbatim.
215
+ return if document.respond_to?(:data) && math_setting(document.data) == false
171
216
 
172
217
  content = document.content
173
218
  return unless content&.match?(/\$|\\\(|\\\[|\\begin\{/)
@@ -177,6 +222,13 @@ module MathPreprocessor
177
222
  document.content = updated_content
178
223
  document.data["math_expressions"] = processor.expressions if processor.expressions.any?
179
224
  end
225
+
226
+ # `math`, or its alias `mathjax` when `math` is unset, as
227
+ # _includes/meta/math-config.html reads them. A `mathjax: true` in front
228
+ # matter defaults used to win over a page's `math: false`.
229
+ def math_setting(data)
230
+ data["math"].nil? ? data["mathjax"] : data["math"]
231
+ end
180
232
  end
181
233
 
182
234
  # Posts are documents, so registering them separately ran the preprocessor
@@ -7,6 +7,7 @@ require "fileutils"
7
7
  require "cgi"
8
8
  require "loofah"
9
9
  require "base64"
10
+ require_relative "rouge_highlight_filter"
10
11
 
11
12
  module Datalog
12
13
  module NotebookRenderer
@@ -94,14 +95,30 @@ module Datalog
94
95
  metadata: build_sanitization_metadata(meta, cell_index: cell_index))
95
96
  return if sanitized.to_s.strip.empty?
96
97
 
97
- %(<section class="notebook-cell notebook-cell--markdown">\n#{sanitized}\n</section>)
98
+ %(<section class="notebook-cell notebook-cell--markdown">\n#{demote_headings(sanitized.to_s)}\n</section>)
99
+ end
100
+
101
+ # The notebook layout gives the page its <h1>, and a notebook's first
102
+ # markdown cell usually repeats the title as `# Title`. Each heading moves
103
+ # down a level (h1 to h2, and so on to h6), which keeps one <h1> on the
104
+ # page and the cells' own outline under it.
105
+ def demote_headings(html)
106
+ html.gsub(%r{<(/?)h([1-5])(?=[\s>])}i) { "<#{Regexp.last_match(1)}h#{Regexp.last_match(2).to_i + 1}" }
98
107
  end
99
108
 
100
109
  # Class names mirror the theme stylesheet (`.notebook-cell--input`,
101
110
  # `.notebook-cell__code` and `.notebook-cell__outputs` in _sass/_components.scss).
102
111
  def render_code(cell, source, metadata, site, cell_index)
103
112
  language = cell.dig("metadata", "language") || metadata[:language] || "text"
104
- code_html = %(<pre class="notebook-cell__code"><code class="language-#{language}">#{CGI.escapeHTML(source)}</code></pre>)
113
+ # The language comes from the notebook file, so it is cut down to the
114
+ # characters a class name can hold before it goes into the attribute.
115
+ language_class = language.to_s.gsub(/[^\w+#.-]/, "")
116
+ # Rouge highlights the cell as the site builds, as kramdown does for code
117
+ # blocks, and escapes it.
118
+ highlighted = Jekyll::RougeHighlightFilter.highlight(source, language_class)
119
+ pre_attributes = %(class="highlight notebook-cell__code" tabindex="0")
120
+ code_attributes = %(class="language-#{language_class}")
121
+ code_html = %(<pre #{pre_attributes}><code #{code_attributes}>#{highlighted}</code></pre>)
105
122
  base_metadata = metadata.respond_to?(:merge) ? metadata.merge(language: language) : { language: language }
106
123
  outputs_html = render_outputs(Array(cell["outputs"]), site: site, cell_index: cell_index, metadata: base_metadata)
107
124
  outputs_html = %(\n<div class="notebook-cell__outputs">\n#{outputs_html}\n</div>) unless outputs_html.empty?
@@ -157,11 +174,13 @@ module Datalog
157
174
  def image_output_html(output)
158
175
  data = output["data"] || {}
159
176
 
177
+ # Jupyter writes base64 image data split over lines or ending in a newline,
178
+ # and the data URI check rejects whitespace, which dropped those images.
160
179
  if (png = data["image/png"])
161
- html = %(<img src="data:image/png;base64,#{Array(png).join}" alt="Notebook output" />)
180
+ html = %(<img src="data:image/png;base64,#{Array(png).join.gsub(/\s+/, '')}" alt="Notebook output" />)
162
181
  return [html, "image/png"]
163
182
  elsif (jpeg = data["image/jpeg"])
164
- html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join}" alt="Notebook output" />)
183
+ html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join.gsub(/\s+/, '')}" alt="Notebook output" />)
165
184
  return [html, "image/jpeg"]
166
185
  elsif (svg = data["image/svg+xml"])
167
186
  encoded = Base64.strict_encode64(Array(svg).join)
@@ -149,7 +149,9 @@ end
149
149
 
150
150
  module Datalog
151
151
  module PluginLoaderHooks
152
- HOOK_SCOPES = %i[pages documents posts].freeze
152
+ # Posts are documents: Jekyll fires a post's `posts` hooks and then its
153
+ # `documents` hooks, so registering both ran every plugin hook twice a post.
154
+ HOOK_SCOPES = %i[pages documents].freeze
153
155
 
154
156
  module_function
155
157
 
@@ -6,7 +6,13 @@ module Jekyll
6
6
  priority :low
7
7
 
8
8
  def generate(site)
9
- publications_data = site.data["publications"] ||= {}
9
+ publications_data = site.data["publications"]
10
+ # _data/publications.yml is normally a map with `settings` and
11
+ # `manual_entries`; a plain list of entries is read as the manual
12
+ # entries rather than stopping the build with a TypeError.
13
+ publications_data = { "manual_entries" => publications_data } if publications_data.is_a?(Array)
14
+ publications_data = {} unless publications_data.is_a?(Hash)
15
+ site.data["publications"] = publications_data
10
16
  settings = publications_data["settings"] || {}
11
17
  config_source = site.config.dig("theme_options", "publications", "bibtex_source")
12
18
  bibtex_source = settings["bibtex_source"] || config_source
@@ -22,7 +28,7 @@ module Jekyll
22
28
  end
23
29
  end
24
30
 
25
- manual_entries = publications_data["manual_entries"] || []
31
+ manual_entries = Array(publications_data["manual_entries"]).select { |entry| entry.is_a?(Hash) }
26
32
  combined = (imported_entries + manual_entries).map { |entry| normalize_entry(entry) }
27
33
 
28
34
  academic_citations = site.data.dig("academic", "citations", "per_publication") || {}
@@ -0,0 +1,238 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi"
4
+
5
+ module Datalog
6
+ # Numbered figures and tables, and references to them, within a page:
7
+ #
8
+ # {% figure id="fig-power" src="/assets/img/power.png" alt="Power curve" %}
9
+ # Power as a function of effect size $\delta$.
10
+ # {% endfigure %}
11
+ #
12
+ # {% table id="tab-runs" %}
13
+ # Simulation runs by sample size.
14
+ #
15
+ # | n | runs |
16
+ # |---|------|
17
+ # {% endtable %}
18
+ #
19
+ # As {% ref fig-power %} shows, ...
20
+ #
21
+ # The tags write placeholders. Once the page's Markdown is converted, every
22
+ # target gets its number in the order it appears, so a reference may come
23
+ # before its figure, and each reference becomes a link reading "Figure 2".
24
+ # A duplicate id or a reference to nothing stops the build. Equations are
25
+ # numbered and referenced by MathJax (\label and \eqref) instead.
26
+ #
27
+ # A caption is the body of the tag, not an attribute, so math in it goes
28
+ # through the math preprocessor like the rest of the page.
29
+ #
30
+ # Theorems, definitions and the other statements in _plugins/statements.rb
31
+ # are numbered and referred to the same way.
32
+ module References
33
+ module_function
34
+
35
+ # Each kind counts separately; the label comes from _data/i18n.
36
+ KINDS = {
37
+ "figure" => "Figure", "table" => "Table",
38
+ "theorem" => "Theorem", "lemma" => "Lemma", "proposition" => "Proposition", "corollary" => "Corollary",
39
+ "definition" => "Definition", "assumption" => "Assumption", "example" => "Example", "remark" => "Remark"
40
+ }.freeze
41
+ ID = /\A[A-Za-z][\w.:-]*\z/
42
+ # A label given in place of the number, as in "Theorem A".
43
+ CUSTOM_LABEL = /\A[[:alnum:]][[:alnum:].'*-]*\z/
44
+ TARGET = /data-ref-target="([^"]+)" data-ref-kind="([a-z]+)"(?: data-ref-number="([^"]+)")?/
45
+ # A label ends with a full stop unless the tag chose another ending.
46
+ LABEL = %r{<span class="datalog-ref-label" data-ref-for="([^"]+)"(?: data-ref-end="([^"]*)")?></span>}
47
+ LINK = %r{<a class="datalog-ref" href="#([^"]+)" data-ref="\1">[^<]*</a>}
48
+ SOURCE_TAG = /\{%-?\s*(#{KINDS.keys.join('|')})\b([^%]*)-?%\}/
49
+
50
+ def number(document)
51
+ content = document.content
52
+ return unless content&.include?("data-ref")
53
+
54
+ targets = targets(content.scan(TARGET), document)
55
+ content = content.gsub(LABEL) do
56
+ id, ending = Regexp.last_match.captures
57
+ text = "#{CGI.escapeHTML(targets.fetch(id))}#{ending || '.'}"
58
+ %(<span class="datalog-ref-label" data-ref-for="#{id}">#{text}</span>)
59
+ end
60
+ document.content = link_references(content, targets, document)
61
+ end
62
+
63
+ # Jekyll runs no hooks for an excerpt, and renders it when a template first
64
+ # asks for it, which on a listing page can be before its post is converted.
65
+ # Its references take their numbers from the post: from the converted
66
+ # content when there is one, and otherwise from the tags in the source,
67
+ # which are in the same order. A missing target is the post's error to report.
68
+ def number_excerpt(excerpt, html)
69
+ return html unless html&.include?("data-ref")
70
+
71
+ post = excerpt.doc
72
+ found = post.content.to_s.scan(TARGET)
73
+ found = source_targets(post.content.to_s) if found.empty?
74
+ targets = targets(found.uniq(&:first), post)
75
+ html.gsub(LINK) do
76
+ id = Regexp.last_match(1)
77
+ text = targets[id] ? CGI.escapeHTML(targets[id]) : id
78
+ %(<a class="datalog-ref" href="#{post.site.baseurl}#{post.url}##{id}" data-ref="#{id}">#{text}</a>)
79
+ end
80
+ end
81
+
82
+ # [id, kind, label] for each numbered tag in a page's source.
83
+ def source_targets(source)
84
+ source.scan(SOURCE_TAG).filter_map do |kind, markup|
85
+ attributes = markup.scan(/(\w+)=["']([^"']*)["']/).to_h
86
+ [attributes["id"], kind, attributes["label"]] if attributes["id"]
87
+ end
88
+ end
89
+
90
+ # The id => "Figure 2" of every [id, kind, label] target, numbered per kind in
91
+ # page order. A target with its own label does not take a number.
92
+ def targets(found, document)
93
+ counts = Hash.new(0)
94
+ found.each_with_object({}) do |(id, kind, custom), targets|
95
+ if targets.key?(id)
96
+ raise Jekyll::Errors::FatalException,
97
+ "#{document.relative_path} has two numbered figures, tables or statements with the id " \
98
+ "\"#{id}\"; each id has to be unique"
99
+ end
100
+
101
+ number = custom ? CGI.unescapeHTML(custom) : (counts[kind] += 1)
102
+ targets[id] = "#{label(document, kind)} #{number}"
103
+ end
104
+ end
105
+
106
+ def link_references(content, targets, document)
107
+ missing = content.scan(LINK).flatten.uniq - targets.keys
108
+ unless missing.empty?
109
+ raise Jekyll::Errors::FatalException,
110
+ "#{document.relative_path} refers to #{missing.map { |id| "\"#{id}\"" }.join(', ')}, which no " \
111
+ "numbered figure, table or statement on the page has as its id"
112
+ end
113
+
114
+ content.gsub(LINK) do
115
+ id = Regexp.last_match(1)
116
+ %(<a class="datalog-ref" href="##{id}" data-ref="#{id}">#{CGI.escapeHTML(targets[id])}</a>)
117
+ end
118
+ end
119
+
120
+ def label(document, kind)
121
+ site = document.site
122
+ locale = I18n.locale_code(site, document.data["lang"])
123
+ I18n.lookup(site, locale, "references.#{kind}") || KINDS.fetch(kind)
124
+ end
125
+
126
+ def validate_label(label, tag)
127
+ return if label.nil? || label.match?(CUSTOM_LABEL)
128
+
129
+ raise Liquid::ArgumentError,
130
+ "{% #{tag} %} takes a label of letters, digits, \".\", \"'\", \"*\" or \"-\", such as label=\"A\"; " \
131
+ "got #{label.inspect}"
132
+ end
133
+
134
+ def validate_id(id, tag)
135
+ return id if id.to_s.match?(ID)
136
+
137
+ raise Liquid::ArgumentError,
138
+ "{% #{tag} %} needs an id that starts with a letter and holds only letters, digits, " \
139
+ "\"-\", \"_\", \".\" or \":\", such as id=\"fig-power\"; got #{id.inspect}"
140
+ end
141
+
142
+ # key="value", key='value' or key=variable.
143
+ def attributes(markup, context)
144
+ markup.scan(/(\w+)=(?:"([^"]*)"|'([^']*)'|([\w.\[\]-]+))/).to_h do |key, double, single, variable|
145
+ [key, double || single || context[variable].to_s]
146
+ end
147
+ end
148
+
149
+ def markdown(context, text)
150
+ site = context.registers[:site]
151
+ site.find_converter_instance(Jekyll::Converters::Markdown).convert(text.to_s.strip)
152
+ end
153
+
154
+ # A caption of one paragraph loses its <p>, which a <figcaption> or <caption> does not need.
155
+ def inline(html)
156
+ html = html.strip
157
+ paragraph = html.match(%r{\A<p>(.*)</p>\z}m)
158
+ paragraph && !paragraph[1].include?("<p>") ? paragraph[1] : html
159
+ end
160
+ end
161
+
162
+ class FigureTag < Liquid::Block
163
+ def initialize(tag_name, markup, options)
164
+ super
165
+ @markup = markup
166
+ end
167
+
168
+ def render(context)
169
+ attributes = References.attributes(@markup, context)
170
+ id = References.validate_id(attributes["id"], "figure")
171
+ src = attributes["src"].to_s
172
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs a src, the image it shows" if src.empty?
173
+ if attributes["alt"].to_s.strip.empty?
174
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs alt text for its image"
175
+ end
176
+
177
+ src = "#{context.registers[:site].config['baseurl'].to_s.chomp('/')}#{src}" if src.start_with?("/")
178
+ caption = References.inline(References.markdown(context, super))
179
+ classes = ["datalog-figure", attributes["class"]].compact.join(" ")
180
+
181
+ alt = CGI.escapeHTML(attributes["alt"])
182
+ %(<figure class="#{CGI.escapeHTML(classes)}" id="#{id}" data-ref-target="#{id}" data-ref-kind="figure">) +
183
+ %(<img src="#{CGI.escapeHTML(src)}" alt="#{alt}" loading="lazy" decoding="async">) +
184
+ %(<figcaption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</figcaption></figure>\n)
185
+ end
186
+ end
187
+
188
+ class TableTag < Liquid::Block
189
+ def initialize(tag_name, markup, options)
190
+ super
191
+ @markup = markup
192
+ end
193
+
194
+ # The body holds the caption and then a Markdown table.
195
+ def render(context)
196
+ id = References.validate_id(References.attributes(@markup, context)["id"], "table")
197
+ html = References.markdown(context, super)
198
+ tables = html.scan(/<table\b/).size
199
+ unless tables == 1
200
+ raise Liquid::ArgumentError, "{% table id=\"#{id}\" %} holds #{tables} tables; it needs one, after its caption"
201
+ end
202
+
203
+ caption, table = html.split(/(?=<table\b)/, 2)
204
+ caption = References.inline(caption)
205
+ numbered = table.sub(/<table\b([^>]*)>/) do
206
+ %(<table#{Regexp.last_match(1)} id="#{id}" data-ref-target="#{id}" data-ref-kind="table">) +
207
+ %(<caption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</caption>)
208
+ end
209
+ "#{numbered.strip}\n"
210
+ end
211
+ end
212
+
213
+ class ReferenceTag < Liquid::Tag
214
+ def initialize(tag_name, markup, options)
215
+ super
216
+ @id = References.validate_id(markup.strip, "ref")
217
+ end
218
+
219
+ def render(_context)
220
+ %(<a class="datalog-ref" href="##{@id}" data-ref="#{@id}">#{@id}</a>)
221
+ end
222
+ end
223
+ end
224
+
225
+ Liquid::Template.register_tag("figure", Datalog::FigureTag)
226
+ Liquid::Template.register_tag("table", Datalog::TableTag)
227
+ Liquid::Template.register_tag("ref", Datalog::ReferenceTag)
228
+
229
+ Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
230
+ Datalog::References.number(document)
231
+ end
232
+
233
+ Jekyll::Excerpt.prepend(Module.new do
234
+ # Excerpt#output renders once and keeps the result; numbering it is a quick gsub.
235
+ def output
236
+ Datalog::References.number_excerpt(self, super)
237
+ end
238
+ end)