datalog-theme 0.6.1 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +175 -0
- data/CITATION.cff +2 -2
- data/README.md +26 -19
- data/_data/cdn-integrity.yml +0 -30
- data/_includes/analytics/dashboard.html +3 -1
- data/_includes/components/api-function.html +20 -1
- data/_includes/components/author-bio.html +2 -2
- data/_includes/components/enhanced-code-block.html +1 -1
- data/_includes/components/hero.html +19 -1
- data/_includes/csp-meta.html +115 -11
- data/_includes/footer.html +19 -20
- data/_includes/head.html +61 -29
- data/_includes/header/navigation.html +12 -15
- data/_includes/header.html +29 -20
- data/_includes/layouts/default/article.html +5 -1
- data/_includes/meta/math-config.html +27 -1
- data/_includes/meta/schema.html +5 -2
- data/_includes/meta/scripts-loader.html +16 -31
- data/_includes/post/related-posts.html +4 -7
- data/_includes/scripts.html +1 -9
- data/_includes/search/index-data.json +131 -0
- data/_includes/search/page.html +136 -0
- data/_layouts/dataset.html +1 -0
- data/_layouts/default.html +16 -8
- data/_layouts/home.html +6 -0
- data/_layouts/notebook.html +1 -0
- data/_layouts/package.html +1 -0
- data/_layouts/portfolio.html +1 -0
- data/_layouts/post.html +20 -5
- data/_layouts/project.html +2 -1
- data/_plugins/analytics_dashboard.rb +9 -3
- data/_plugins/config_validator.rb +12 -7
- data/_plugins/csp_generator.rb +18 -28
- data/_plugins/datalog_bibliography.rb +9 -7
- data/_plugins/datalog_comments.rb +8 -5
- data/_plugins/datalog_slides.rb +9 -8
- data/_plugins/i18n.rb +7 -8
- data/_plugins/image_optimizer.rb +17 -6
- data/_plugins/math_preprocessor.rb +36 -3
- data/_plugins/notebook_converter.rb +23 -43
- data/_plugins/plugin_loader.rb +3 -1
- data/_plugins/publications_generator.rb +8 -2
- data/_plugins/rouge_highlight_filter.rb +42 -0
- data/_plugins/search_code_blocks.rb +30 -0
- data/_plugins/search_normalizer.rb +15 -53
- data/_plugins/search_pages.rb +72 -0
- data/_sass/_academic-dashboard.scss +262 -0
- data/_sass/_base.scss +19 -1
- data/_sass/_components.scss +77 -1144
- data/_sass/_features.scss +17 -0
- data/_sass/_font-fallbacks.scss +43 -0
- data/_sass/_header.scss +10 -0
- data/_sass/_layout.scss +18 -12
- data/_sass/_mathematical.scss +1 -1
- data/_sass/_notebooks.scss +322 -0
- data/_sass/_open-science-badges.scss +56 -0
- data/_sass/{_phase1-enhancements.scss → _post-components.scss} +46 -1
- data/_sass/_search-page.scss +530 -0
- data/_sass/_search.scss +47 -1
- data/_sass/_syntax-highlighting.scss +212 -97
- data/_sass/_theme.scss +42 -18
- data/_sass/_typography.scss +14 -0
- data/_sass/_variables.scss +7 -4
- data/assets/css/main.scss +14 -0
- data/assets/img/hero-detail-640.webp +0 -0
- data/assets/img/hero-detail.webp +0 -0
- data/assets/img/social-card.png +0 -0
- data/assets/js/dist/academic.js +1 -0
- data/assets/js/dist/analytics-dashboard.js +1 -0
- data/assets/js/dist/chunks/chunk-225H5YXE.js +1 -0
- data/assets/js/dist/core.js +1 -0
- data/assets/js/dist/loader.js +1 -0
- data/assets/js/dist/math.js +1 -0
- data/assets/js/dist/notebook.js +1 -0
- data/assets/js/dist/search.js +1 -0
- data/assets/js/dist/visualizations.js +11 -0
- data/assets/js/loader.js +3 -1
- data/datalog-theme.gemspec +43 -22
- data/lib/datalog/cli.rb +53 -15
- data/lib/datalog/plugin_system/dependency_resolver.rb +0 -2
- data/lib/datalog/theme/package.rb +57 -0
- data/lib/datalog/theme/repository_checkout.rb +90 -0
- data/lib/datalog/theme/version.rb +1 -1
- data/lib/datalog/warning_filter.rb +5 -11
- data/lib/datalog-theme.rb +21 -0
- metadata +75 -146
- data/_data/academic.yml +0 -217
- data/_data/config/author.yml +0 -121
- data/_data/config/features.yml +0 -262
- data/_data/config/site.yml +0 -42
- data/_data/config/theme.yml +0 -181
- data/_data/datasets.yml +0 -28
- data/_data/js_meta.json +0 -371
- data/_data/navigation.yml +0 -145
- data/_data/projects.yml +0 -41
- data/_data/publications.yml +0 -28
- data/_data/social.yml +0 -73
- data/_data/visualizations.yml +0 -51
- data/_includes/components/advanced-search.html +0 -682
- data/_includes/components/bookmark-system.html +0 -96
- data/_includes/components/comments.html +0 -244
- data/_includes/components/content-recommendations.html +0 -228
- data/_includes/components/email-preferences.html +0 -200
- data/_includes/components/enhanced-metadata.html +0 -228
- data/_includes/components/language-switcher.html +0 -396
- data/_includes/components/navigation-enhancements.html +0 -454
- data/_includes/components/newsletter-signup.html +0 -178
- data/_includes/components/popular-posts.html +0 -233
- data/_includes/components/reading-progress.html +0 -133
- data/_includes/components/reading-time.html +0 -121
- data/_includes/components/series-navigation.html +0 -124
- data/_includes/components/social-proof.html +0 -34
- data/_includes/components/user-preferences.html +0 -566
- data/_layouts/archive.html +0 -282
- data/_layouts/post-sidebar.html +0 -183
- data/_sass/_phase3-enhancements.scss +0 -874
- data/_sass/_phase4-enhancements.scss +0 -1214
- data/_sass/_phase5-enhancements.scss +0 -414
- data/assets/img/20220607123041_detail.001.png +0 -0
- data/assets/js/academic.js +0 -262
- data/assets/js/analytics-dashboard.js +0 -382
- data/assets/js/core/dark-mode.js +0 -79
- data/assets/js/core/github-cards.js +0 -123
- data/assets/js/core/language-filter.js +0 -69
- data/assets/js/core/navigation.js +0 -184
- data/assets/js/core/scroll-progress.js +0 -45
- data/assets/js/core/search-hotkeys.js +0 -62
- data/assets/js/core/skip-links.js +0 -62
- data/assets/js/main.js +0 -23
- data/assets/js/math.js +0 -818
- data/assets/js/notebook.js +0 -158
- data/assets/js/search/analytics.js +0 -91
- data/assets/js/search/app.js +0 -271
- data/assets/js/search/autocomplete.js +0 -120
- data/assets/js/search/engine.js +0 -260
- data/assets/js/search/filters.js +0 -38
- data/assets/js/search/render.js +0 -217
- data/assets/js/search/utils.js +0 -99
- data/assets/js/search.js +0 -354
- data/assets/js/visualizations.js +0 -816
- data/assets/publications/datalog-publications.bib +0 -8
- data/assets/publications/datalog-publications.ris +0 -9
- data/assets/publications/publications.bib +0 -30
- data/assets/templates/diogo-ribeiro-cv.md +0 -31
- data/assets/templates/diogo-ribeiro-cv.tex +0 -32
- data/lib/datalog/theme/theme.rb +0 -18
data/_plugins/csp_generator.rb
CHANGED
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "securerandom"
|
|
4
|
-
require "digest"
|
|
5
4
|
|
|
6
5
|
module Datalog
|
|
7
6
|
module Security
|
|
@@ -18,7 +17,6 @@ module Datalog
|
|
|
18
17
|
def generate(site)
|
|
19
18
|
site.data["csp"] ||= {}
|
|
20
19
|
site.data["csp"]["nonces"] ||= {}
|
|
21
|
-
site.data["csp"]["hashes"] ||= {}
|
|
22
20
|
|
|
23
21
|
assign_nonces(site, site.pages)
|
|
24
22
|
site.collections.each_value do |collection|
|
|
@@ -40,7 +38,6 @@ module Datalog
|
|
|
40
38
|
|
|
41
39
|
nonce = SecureRandom.base64(nonce_bytes)
|
|
42
40
|
document.data["csp_nonce"] = nonce
|
|
43
|
-
document.data["csp_hashes"] ||= []
|
|
44
41
|
|
|
45
42
|
registry_site = site || (document.respond_to?(:site) ? document.site : nil)
|
|
46
43
|
if registry_site.respond_to?(:data)
|
|
@@ -61,7 +58,12 @@ module Datalog
|
|
|
61
58
|
end
|
|
62
59
|
end
|
|
63
60
|
|
|
64
|
-
|
|
61
|
+
# Gives every inline script in the rendered page the page's nonce. The
|
|
62
|
+
# generator also took a SHA-256 of each inline script, but the policy is
|
|
63
|
+
# written into the head while the page renders, so the hashes of the
|
|
64
|
+
# finished page never reached it; with every script nonced they are not
|
|
65
|
+
# needed.
|
|
66
|
+
def self.add_nonces(document)
|
|
65
67
|
return unless document.respond_to?(:output)
|
|
66
68
|
|
|
67
69
|
output = document.output
|
|
@@ -77,39 +79,27 @@ module Datalog
|
|
|
77
79
|
attributes = Regexp.last_match(1)
|
|
78
80
|
"<script nonce=\"#{nonce}\"#{attributes}>"
|
|
79
81
|
end
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
output.scan(%r{<script(?![^>]*\bsrc=)[^>]*>(.*?)</script>}mi) do |match|
|
|
85
|
-
content = match.first
|
|
86
|
-
next if content.nil? || content.empty?
|
|
87
|
-
|
|
88
|
-
hashes << Digest::SHA256.base64digest(content)
|
|
82
|
+
# A template that printed page.csp_nonce before the page had one left
|
|
83
|
+
# nonce="", which authorises nothing.
|
|
84
|
+
document.output = output.gsub(/(<(?:script|style)\b[^>]*\bnonce=)""/i) do
|
|
85
|
+
"#{Regexp.last_match(1)}\"#{nonce}\""
|
|
89
86
|
end
|
|
90
|
-
|
|
91
|
-
hashes.uniq!
|
|
92
|
-
document.data["csp_hashes"] = hashes
|
|
93
|
-
|
|
94
|
-
return unless site.respond_to?(:data)
|
|
95
|
-
|
|
96
|
-
site.data["csp"] ||= {}
|
|
97
|
-
site.data["csp"]["hashes"] ||= {}
|
|
98
|
-
|
|
99
|
-
key = document_key(document)
|
|
100
|
-
|
|
101
|
-
site.data["csp"]["hashes"][key] = hashes if key
|
|
102
87
|
end
|
|
103
88
|
end
|
|
104
89
|
end
|
|
105
90
|
end
|
|
106
91
|
|
|
107
92
|
%i[pages documents].each do |target|
|
|
108
|
-
Jekyll::Hooks.register target, :pre_render do |document|
|
|
109
|
-
Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
|
|
93
|
+
Jekyll::Hooks.register target, :pre_render do |document, payload|
|
|
94
|
+
nonce = Datalog::Security::CspGenerator.assign_nonce(document.respond_to?(:site) ? document.site : nil, document)
|
|
95
|
+
# A page another generator creates after this one runs, such as a notebook
|
|
96
|
+
# page, only gets its nonce here. Jekyll has already copied a page's data
|
|
97
|
+
# into the hash its templates read, so page.csp_nonce rendered empty there.
|
|
98
|
+
page_data = payload && payload["page"]
|
|
99
|
+
page_data["csp_nonce"] ||= nonce if nonce && page_data.is_a?(Hash)
|
|
110
100
|
end
|
|
111
101
|
|
|
112
102
|
Jekyll::Hooks.register target, :post_render do |document|
|
|
113
|
-
Datalog::Security::CspGenerator.
|
|
103
|
+
Datalog::Security::CspGenerator.add_nonces(document)
|
|
114
104
|
end
|
|
115
105
|
end
|
|
@@ -1,17 +1,19 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Jekyll
|
|
4
|
+
# Stands in for the `datalog_bibliography` tag when the datalog-citations
|
|
5
|
+
# plugin is not enabled, so a layout that uses the tag still builds. The plugin
|
|
6
|
+
# registers the real tag and sets `datalog_bibliography` on the pages it
|
|
7
|
+
# handles; a page that sets the key by hand without the plugin gets nothing and
|
|
8
|
+
# a build warning, where it used to get a "coming soon" notice.
|
|
4
9
|
class DatalogBibliographyTag < Liquid::Tag
|
|
5
10
|
def render(context)
|
|
6
11
|
page = context.registers[:page]
|
|
7
|
-
|
|
8
|
-
return "" unless bib_data
|
|
12
|
+
return "" unless page["datalog_bibliography"]
|
|
9
13
|
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
</div>
|
|
14
|
-
HTML
|
|
14
|
+
Jekyll.logger.warn("datalog-citations", "#{page['path']} sets datalog_bibliography, " \
|
|
15
|
+
"but datalog_plugins.enabled does not list datalog-citations")
|
|
16
|
+
""
|
|
15
17
|
end
|
|
16
18
|
end
|
|
17
19
|
end
|
|
@@ -1,16 +1,19 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Jekyll
|
|
4
|
+
# Stands in for the `datalog_comments` tag when the datalog-comments plugin is
|
|
5
|
+
# not enabled, so a layout that uses the tag still builds. The plugin registers
|
|
6
|
+
# the real tag and sets `datalog_comments` on the pages it handles; a page that
|
|
7
|
+
# sets the key by hand without the plugin gets nothing and a build warning,
|
|
8
|
+
# where it used to get a "coming soon" notice.
|
|
4
9
|
class DatalogCommentsTag < Liquid::Tag
|
|
5
10
|
def render(context)
|
|
6
11
|
page = context.registers[:page]
|
|
7
12
|
return "" unless page["datalog_comments"]
|
|
8
13
|
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
</div>
|
|
13
|
-
HTML
|
|
14
|
+
Jekyll.logger.warn("datalog-comments", "#{page['path']} sets datalog_comments, " \
|
|
15
|
+
"but datalog_plugins.enabled does not list datalog-comments")
|
|
16
|
+
""
|
|
14
17
|
end
|
|
15
18
|
end
|
|
16
19
|
end
|
data/_plugins/datalog_slides.rb
CHANGED
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Jekyll
|
|
4
|
+
# Stands in for the `datalog_slides` tag when the datalog-slides plugin is not
|
|
5
|
+
# enabled, so a layout that uses the tag still builds. The plugin registers the
|
|
6
|
+
# real tag and sets `datalog_slides` on the pages it handles; a page that sets
|
|
7
|
+
# the key by hand without the plugin gets nothing and a build warning, where it
|
|
8
|
+
# used to get a "coming soon" notice showing its raw configuration.
|
|
4
9
|
class DatalogSlidesTag < Liquid::Tag
|
|
5
10
|
def render(context)
|
|
6
11
|
page = context.registers[:page]
|
|
7
|
-
|
|
8
|
-
return "" unless slides_data
|
|
12
|
+
return "" unless page["datalog_slides"]
|
|
9
13
|
|
|
10
|
-
#
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
<p>Slides feature coming soon. Configured: #{slides_data}</p>
|
|
14
|
-
</div>
|
|
15
|
-
HTML
|
|
14
|
+
Jekyll.logger.warn("datalog-slides", "#{page['path']} sets datalog_slides, " \
|
|
15
|
+
"but datalog_plugins.enabled does not list datalog-slides")
|
|
16
|
+
""
|
|
16
17
|
end
|
|
17
18
|
end
|
|
18
19
|
end
|
data/_plugins/i18n.rb
CHANGED
|
@@ -116,6 +116,11 @@ class TranslateTag < Liquid::Tag
|
|
|
116
116
|
# quote into the first option name, or interpolation silently fails.
|
|
117
117
|
SYNTAX = /\A\s*(['"]?)(\w[\w.-]*)\1(.*)?\z/m
|
|
118
118
|
|
|
119
|
+
# `name: value` pairs, separated by commas or spaces. A quoted value may
|
|
120
|
+
# contain commas: splitting the markup on every comma cut `name: "Doe, Jane"`
|
|
121
|
+
# in two.
|
|
122
|
+
OPTION = /(\w+)\s*:\s*("(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|[^\s,]+)/
|
|
123
|
+
|
|
119
124
|
def initialize(tag_name, markup, tokens)
|
|
120
125
|
super
|
|
121
126
|
raise Liquid::SyntaxError, "Syntax Error in 't' - Valid syntax: t key [arg: value]" unless markup.strip =~ SYNTAX
|
|
@@ -134,14 +139,8 @@ class TranslateTag < Liquid::Tag
|
|
|
134
139
|
def parse_options(markup, context)
|
|
135
140
|
return {} unless markup && !markup.strip.empty?
|
|
136
141
|
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
next if token.empty?
|
|
140
|
-
|
|
141
|
-
if token.include?(":")
|
|
142
|
-
key, value = token.split(":", 2)
|
|
143
|
-
memo[key.strip.to_sym] = context.evaluate(Liquid::Expression.parse(value.strip))
|
|
144
|
-
end
|
|
142
|
+
markup.scan(OPTION).to_h do |key, value|
|
|
143
|
+
[key.to_sym, context.evaluate(Liquid::Expression.parse(value))]
|
|
145
144
|
end
|
|
146
145
|
end
|
|
147
146
|
end
|
data/_plugins/image_optimizer.rb
CHANGED
|
@@ -34,6 +34,18 @@ module Jekyll
|
|
|
34
34
|
}.freeze
|
|
35
35
|
RASTER_EXTENSIONS = %w[.jpg .jpeg .png].freeze
|
|
36
36
|
|
|
37
|
+
# The one image a page fetches early: the first in its post or page content,
|
|
38
|
+
# unless the author made it lazy or the page already preloads an image, as
|
|
39
|
+
# the hero does. The first <img> anywhere used to be marked both lazy and
|
|
40
|
+
# high priority; on the home page that was a post card below the hero, and
|
|
41
|
+
# on tutorials the thumbnail of a related post at the bottom.
|
|
42
|
+
def priority_image(fragment)
|
|
43
|
+
return if fragment.at_css('link[rel="preload"][as="image"]')
|
|
44
|
+
|
|
45
|
+
image = fragment.css(".post-content img, .page-content img").find { |img| img["data-no-optimize"] != "true" }
|
|
46
|
+
image unless image.nil? || image["loading"] == "lazy"
|
|
47
|
+
end
|
|
48
|
+
|
|
37
49
|
def process(document)
|
|
38
50
|
return unless document.output_ext == ".html"
|
|
39
51
|
return if document.output.nil? || document.output.empty?
|
|
@@ -46,18 +58,17 @@ module Jekyll
|
|
|
46
58
|
manifest = site&.data&.fetch("datalog_responsive_images", {}) || {}
|
|
47
59
|
image_config = site&.config&.fetch("datalog_image_config", {}) || {}
|
|
48
60
|
optimized = false
|
|
49
|
-
|
|
61
|
+
priority = priority_image(fragment)
|
|
50
62
|
|
|
51
63
|
fragment.css("img").each do |img|
|
|
52
64
|
next if img["data-no-optimize"] == "true"
|
|
53
65
|
|
|
54
|
-
img
|
|
55
|
-
img["decoding"] ||= "async"
|
|
56
|
-
|
|
57
|
-
unless first_priority_assigned
|
|
66
|
+
if img == priority
|
|
58
67
|
img["fetchpriority"] ||= "high"
|
|
59
|
-
|
|
68
|
+
else
|
|
69
|
+
img["loading"] ||= "lazy"
|
|
60
70
|
end
|
|
71
|
+
img["decoding"] ||= "async"
|
|
61
72
|
|
|
62
73
|
normalized_src = normalize_src(img["src"], site)
|
|
63
74
|
picture_entry = manifest[normalized_src]
|
|
@@ -19,8 +19,11 @@ module MathPreprocessor
|
|
|
19
19
|
].freeze
|
|
20
20
|
|
|
21
21
|
INLINE_PATTERNS = [
|
|
22
|
+
# Pandoc's rule for inline math, so prices and shell variables stay text:
|
|
23
|
+
# the opening $ is followed by a non-space, the closing $ follows a
|
|
24
|
+
# non-space and is not followed by a digit, and a blank line ends it.
|
|
22
25
|
{
|
|
23
|
-
regex: /(?<![\\$])(?<open>\$)(
|
|
26
|
+
regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
|
|
24
27
|
tag: "span"
|
|
25
28
|
},
|
|
26
29
|
{
|
|
@@ -29,6 +32,20 @@ module MathPreprocessor
|
|
|
29
32
|
}
|
|
30
33
|
].freeze
|
|
31
34
|
|
|
35
|
+
# Code shows dollar signs literally (shell and R variables, amounts in SQL),
|
|
36
|
+
# so fenced blocks, highlight tags, <pre>/<code> elements and inline code
|
|
37
|
+
# spans are set aside before looking for math and put back afterwards.
|
|
38
|
+
CODE_PATTERNS = [
|
|
39
|
+
/^([ \t]*)(`{3,}|~{3,})[^\n]*\n.*?(?:^\1\2[ \t]*$|\z)/m,
|
|
40
|
+
/\{%-?\s*highlight\b.*?\{%-?\s*endhighlight\s*-?%\}/m,
|
|
41
|
+
%r{<(pre|code)\b[^>]*>.*?</\1>}mi,
|
|
42
|
+
/(?<!`)(`+)(?!`)(?:(?!\n[ \t]*\n).)+?(?<!`)\1(?!`)/m
|
|
43
|
+
].freeze
|
|
44
|
+
|
|
45
|
+
# NUL marks masked code, since page content never contains it. It is written as
|
|
46
|
+
# an escape: a raw NUL byte in the source stopped RuboCop from parsing the file.
|
|
47
|
+
PLACEHOLDER = /\x00(\d+)\x00/
|
|
48
|
+
|
|
32
49
|
class Processor
|
|
33
50
|
attr_reader :expressions
|
|
34
51
|
|
|
@@ -40,9 +57,18 @@ module MathPreprocessor
|
|
|
40
57
|
def process
|
|
41
58
|
return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
|
|
42
59
|
|
|
43
|
-
|
|
60
|
+
code = []
|
|
61
|
+
processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
|
|
62
|
+
text.gsub(pattern) do |match|
|
|
63
|
+
code << match
|
|
64
|
+
"\x00#{code.size - 1}\x00"
|
|
65
|
+
end
|
|
66
|
+
end
|
|
44
67
|
processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
|
|
45
|
-
apply_patterns(processed, INLINE_PATTERNS, display: false)
|
|
68
|
+
processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
|
|
69
|
+
# A segment set aside can contain the placeholder of an earlier one.
|
|
70
|
+
processed = processed.gsub(PLACEHOLDER) { code[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
|
|
71
|
+
processed
|
|
46
72
|
end
|
|
47
73
|
|
|
48
74
|
private
|
|
@@ -66,8 +92,12 @@ module MathPreprocessor
|
|
|
66
92
|
cleaned_source = cleanup_source(latex)
|
|
67
93
|
record_expression(cleaned_source, alt_text)
|
|
68
94
|
|
|
95
|
+
# ARIA forbids aria-label on an element with no role, such as a plain span.
|
|
96
|
+
# axe let it pass while the span held the raw LaTeX as text, and failed it
|
|
97
|
+
# once MathJax rendered the expression; the math role allows the label.
|
|
69
98
|
attributes = {
|
|
70
99
|
"class" => display ? "math-expression math-expression--source" : "math-expression-inline math-expression--source",
|
|
100
|
+
"role" => "math",
|
|
71
101
|
"data-math-alt" => alt_text,
|
|
72
102
|
"data-math-source" => cleaned_source,
|
|
73
103
|
"aria-label" => alt_text,
|
|
@@ -165,6 +195,9 @@ module MathPreprocessor
|
|
|
165
195
|
def apply(document)
|
|
166
196
|
return unless document.respond_to?(:content)
|
|
167
197
|
return unless document.respond_to?(:output_ext) && document.output_ext == ".html"
|
|
198
|
+
# A page that opts out of math rendering (`math: false` or `mathjax: false`)
|
|
199
|
+
# keeps its dollar signs and TeX-looking text verbatim.
|
|
200
|
+
return if document.respond_to?(:data) && (document.data["math"] == false || document.data["mathjax"] == false)
|
|
168
201
|
|
|
169
202
|
content = document.content
|
|
170
203
|
return unless content&.match?(/\$|\\\(|\\\[|\\begin\{/)
|
|
@@ -7,6 +7,7 @@ require "fileutils"
|
|
|
7
7
|
require "cgi"
|
|
8
8
|
require "loofah"
|
|
9
9
|
require "base64"
|
|
10
|
+
require_relative "rouge_highlight_filter"
|
|
10
11
|
|
|
11
12
|
module Datalog
|
|
12
13
|
module NotebookRenderer
|
|
@@ -94,14 +95,30 @@ module Datalog
|
|
|
94
95
|
metadata: build_sanitization_metadata(meta, cell_index: cell_index))
|
|
95
96
|
return if sanitized.to_s.strip.empty?
|
|
96
97
|
|
|
97
|
-
%(<section class="notebook-cell notebook-cell--markdown">\n#{sanitized}\n</section>)
|
|
98
|
+
%(<section class="notebook-cell notebook-cell--markdown">\n#{demote_headings(sanitized.to_s)}\n</section>)
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# The notebook layout gives the page its <h1>, and a notebook's first
|
|
102
|
+
# markdown cell usually repeats the title as `# Title`. Each heading moves
|
|
103
|
+
# down a level (h1 to h2, and so on to h6), which keeps one <h1> on the
|
|
104
|
+
# page and the cells' own outline under it.
|
|
105
|
+
def demote_headings(html)
|
|
106
|
+
html.gsub(%r{<(/?)h([1-5])(?=[\s>])}i) { "<#{Regexp.last_match(1)}h#{Regexp.last_match(2).to_i + 1}" }
|
|
98
107
|
end
|
|
99
108
|
|
|
100
109
|
# Class names mirror the theme stylesheet (`.notebook-cell--input`,
|
|
101
110
|
# `.notebook-cell__code` and `.notebook-cell__outputs` in _sass/_components.scss).
|
|
102
111
|
def render_code(cell, source, metadata, site, cell_index)
|
|
103
112
|
language = cell.dig("metadata", "language") || metadata[:language] || "text"
|
|
104
|
-
|
|
113
|
+
# The language comes from the notebook file, so it is cut down to the
|
|
114
|
+
# characters a class name can hold before it goes into the attribute.
|
|
115
|
+
language_class = language.to_s.gsub(/[^\w+#.-]/, "")
|
|
116
|
+
# Rouge highlights the cell as the site builds, as kramdown does for code
|
|
117
|
+
# blocks, and escapes it.
|
|
118
|
+
highlighted = Jekyll::RougeHighlightFilter.highlight(source, language_class)
|
|
119
|
+
pre_attributes = %(class="highlight notebook-cell__code" tabindex="0")
|
|
120
|
+
code_attributes = %(class="language-#{language_class}")
|
|
121
|
+
code_html = %(<pre #{pre_attributes}><code #{code_attributes}>#{highlighted}</code></pre>)
|
|
105
122
|
base_metadata = metadata.respond_to?(:merge) ? metadata.merge(language: language) : { language: language }
|
|
106
123
|
outputs_html = render_outputs(Array(cell["outputs"]), site: site, cell_index: cell_index, metadata: base_metadata)
|
|
107
124
|
outputs_html = %(\n<div class="notebook-cell__outputs">\n#{outputs_html}\n</div>) unless outputs_html.empty?
|
|
@@ -157,11 +174,13 @@ module Datalog
|
|
|
157
174
|
def image_output_html(output)
|
|
158
175
|
data = output["data"] || {}
|
|
159
176
|
|
|
177
|
+
# Jupyter writes base64 image data split over lines or ending in a newline,
|
|
178
|
+
# and the data URI check rejects whitespace, which dropped those images.
|
|
160
179
|
if (png = data["image/png"])
|
|
161
|
-
html = %(<img src="data:image/png;base64,#{Array(png).join}" alt="Notebook output" />)
|
|
180
|
+
html = %(<img src="data:image/png;base64,#{Array(png).join.gsub(/\s+/, '')}" alt="Notebook output" />)
|
|
162
181
|
return [html, "image/png"]
|
|
163
182
|
elsif (jpeg = data["image/jpeg"])
|
|
164
|
-
html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join}" alt="Notebook output" />)
|
|
183
|
+
html = %(<img src="data:image/jpeg;base64,#{Array(jpeg).join.gsub(/\s+/, '')}" alt="Notebook output" />)
|
|
165
184
|
return [html, "image/jpeg"]
|
|
166
185
|
elsif (svg = data["image/svg+xml"])
|
|
167
186
|
encoded = Base64.strict_encode64(Array(svg).join)
|
|
@@ -629,31 +648,9 @@ module Datalog
|
|
|
629
648
|
nil
|
|
630
649
|
end
|
|
631
650
|
|
|
632
|
-
def patch_jupyter_converter!
|
|
633
|
-
return unless defined?(JekyllJupyterNotebook::Converter)
|
|
634
|
-
return if @converter_fallback_applied
|
|
635
|
-
|
|
636
|
-
fallback = Module.new do
|
|
637
|
-
def convert(content)
|
|
638
|
-
super
|
|
639
|
-
rescue Errno::ENOENT, StandardError => e
|
|
640
|
-
Jekyll.logger.warn("notebook converter", "primary conversion failed: #{e.message}; using fallback renderer")
|
|
641
|
-
Datalog::NotebookRenderer.render_from_raw(content) || ""
|
|
642
|
-
end
|
|
643
|
-
end
|
|
644
|
-
|
|
645
|
-
JekyllJupyterNotebook::Converter.prepend(fallback)
|
|
646
|
-
@converter_fallback_applied = true
|
|
647
|
-
rescue StandardError => e
|
|
648
|
-
Jekyll.logger.warn("notebook converter", "failed to apply converter fallback: #{e.message}")
|
|
649
|
-
end
|
|
650
651
|
end
|
|
651
652
|
end
|
|
652
653
|
|
|
653
|
-
Jekyll::Hooks.register :site, :after_init do |_site|
|
|
654
|
-
Datalog::NotebookRenderer.patch_jupyter_converter!
|
|
655
|
-
end
|
|
656
|
-
|
|
657
654
|
module Jekyll
|
|
658
655
|
# Converts Jupyter notebooks into HTML pages and downloadable assets.
|
|
659
656
|
class NotebookConverter < Generator
|
|
@@ -671,8 +668,6 @@ module Jekyll
|
|
|
671
668
|
return
|
|
672
669
|
end
|
|
673
670
|
|
|
674
|
-
ensure_dependency
|
|
675
|
-
|
|
676
671
|
files = notebook_files
|
|
677
672
|
logger.debug("notebook converter", "located #{files.size} notebooks")
|
|
678
673
|
return if files.empty?
|
|
@@ -722,21 +717,6 @@ module Jekyll
|
|
|
722
717
|
value.sub(%r{/+$}, "")
|
|
723
718
|
end
|
|
724
719
|
|
|
725
|
-
# Notebook pages do not need the gem (see #convert_notebook). When it is
|
|
726
|
-
# present, its converter is given the same renderer as a fallback so the
|
|
727
|
-
# gem's own `.ipynb` handling and `{% jupyter_notebook %}` tag keep working
|
|
728
|
-
# on machines without a `jupyter` executable.
|
|
729
|
-
def ensure_dependency
|
|
730
|
-
return true if defined?(JekyllJupyterNotebook::Converter)
|
|
731
|
-
|
|
732
|
-
require "jekyll-jupyter-notebook"
|
|
733
|
-
Datalog::NotebookRenderer.patch_jupyter_converter!
|
|
734
|
-
true
|
|
735
|
-
rescue LoadError => e
|
|
736
|
-
logger.debug("notebook converter", "jekyll-jupyter-notebook not loaded: #{e.message}")
|
|
737
|
-
false
|
|
738
|
-
end
|
|
739
|
-
|
|
740
720
|
def notebook_files
|
|
741
721
|
glob = File.join(site.source, config["source"], "**", "*.ipynb")
|
|
742
722
|
Dir.glob(glob)
|
data/_plugins/plugin_loader.rb
CHANGED
|
@@ -149,7 +149,9 @@ end
|
|
|
149
149
|
|
|
150
150
|
module Datalog
|
|
151
151
|
module PluginLoaderHooks
|
|
152
|
-
|
|
152
|
+
# Posts are documents: Jekyll fires a post's `posts` hooks and then its
|
|
153
|
+
# `documents` hooks, so registering both ran every plugin hook twice a post.
|
|
154
|
+
HOOK_SCOPES = %i[pages documents].freeze
|
|
153
155
|
|
|
154
156
|
module_function
|
|
155
157
|
|
|
@@ -6,7 +6,13 @@ module Jekyll
|
|
|
6
6
|
priority :low
|
|
7
7
|
|
|
8
8
|
def generate(site)
|
|
9
|
-
publications_data = site.data["publications"]
|
|
9
|
+
publications_data = site.data["publications"]
|
|
10
|
+
# _data/publications.yml is normally a map with `settings` and
|
|
11
|
+
# `manual_entries`; a plain list of entries is read as the manual
|
|
12
|
+
# entries rather than stopping the build with a TypeError.
|
|
13
|
+
publications_data = { "manual_entries" => publications_data } if publications_data.is_a?(Array)
|
|
14
|
+
publications_data = {} unless publications_data.is_a?(Hash)
|
|
15
|
+
site.data["publications"] = publications_data
|
|
10
16
|
settings = publications_data["settings"] || {}
|
|
11
17
|
config_source = site.config.dig("theme_options", "publications", "bibtex_source")
|
|
12
18
|
bibtex_source = settings["bibtex_source"] || config_source
|
|
@@ -22,7 +28,7 @@ module Jekyll
|
|
|
22
28
|
end
|
|
23
29
|
end
|
|
24
30
|
|
|
25
|
-
manual_entries = publications_data["manual_entries"]
|
|
31
|
+
manual_entries = Array(publications_data["manual_entries"]).select { |entry| entry.is_a?(Hash) }
|
|
26
32
|
combined = (imported_entries + manual_entries).map { |entry| normalize_entry(entry) }
|
|
27
33
|
|
|
28
34
|
academic_citations = site.data.dig("academic", "citations", "per_publication") || {}
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "rouge"
|
|
4
|
+
|
|
5
|
+
# `rouge_highlight` highlights code with Rouge as the site builds, the way
|
|
6
|
+
# kramdown highlights fenced code blocks. Includes that print code passed to
|
|
7
|
+
# them, such as `components/api-function.html`, call it as
|
|
8
|
+
# `code | rouge_highlight: language` inside `<pre class="highlight"><code>`;
|
|
9
|
+
# the notebook converter calls `RougeHighlightFilter.highlight` for code cells.
|
|
10
|
+
# The result is escaped; a language Rouge does not know comes back as plain
|
|
11
|
+
# text.
|
|
12
|
+
module Jekyll
|
|
13
|
+
module RougeHighlightFilter
|
|
14
|
+
# kramdown's opening tag for a block Rouge highlighted.
|
|
15
|
+
KRAMDOWN_CODE_BLOCK = '<pre class="highlight">'
|
|
16
|
+
|
|
17
|
+
def self.highlight(code, language = nil)
|
|
18
|
+
return "" if code.nil?
|
|
19
|
+
|
|
20
|
+
source = code.to_s
|
|
21
|
+
lexer = Rouge::Lexer.find_fancy(language.to_s.strip.downcase, source) || Rouge::Lexers::PlainText
|
|
22
|
+
Rouge::Formatters::HTML.new.format(lexer.lex(source))
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def rouge_highlight(code, language = nil)
|
|
26
|
+
RougeHighlightFilter.highlight(code, language)
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
Liquid::Template.register_filter(Jekyll::RougeHighlightFilter)
|
|
32
|
+
|
|
33
|
+
# A code block wider than the page scrolls, and a keyboard user can only scroll
|
|
34
|
+
# it once it takes focus. Prism made every block focusable in the browser; the
|
|
35
|
+
# blocks kramdown highlights get the attribute here, and the includes and the
|
|
36
|
+
# notebook converter write it themselves.
|
|
37
|
+
Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
|
|
38
|
+
block = Jekyll::RougeHighlightFilter::KRAMDOWN_CODE_BLOCK
|
|
39
|
+
next unless document.content&.include?(block)
|
|
40
|
+
|
|
41
|
+
document.content = document.content.gsub(block, '<pre class="highlight" tabindex="0">')
|
|
42
|
+
end
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Datalog
|
|
4
|
+
# Collects the fenced code blocks of every page and document for the search
|
|
5
|
+
# index. The index template used to split `doc.content` on backticks, but
|
|
6
|
+
# Jekyll renders documents before pages, so by the time search.json rendered
|
|
7
|
+
# that content was HTML and every document's code list came out empty.
|
|
8
|
+
module SearchCodeBlocks
|
|
9
|
+
# An opening fence of three or more backticks or tildes with an optional
|
|
10
|
+
# language, the code, and a closing fence of the same characters.
|
|
11
|
+
FENCE = /^ {0,3}(`{3,}|~{3,})[ \t]*([^\s`~{]*)[^\n]*\n(.*?)^ {0,3}\1[ \t]*$/m
|
|
12
|
+
|
|
13
|
+
module_function
|
|
14
|
+
|
|
15
|
+
def extract(source)
|
|
16
|
+
text = source.to_s.gsub("\r\n", "\n")
|
|
17
|
+
text.scan(FENCE).map do |_fence, language, code|
|
|
18
|
+
{ "language" => language.empty? ? "text" : language.downcase, "code" => code.chomp }
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Content is still the author's source before rendering starts. Collection
|
|
25
|
+
# docs only: site.documents also lists a collection's static files.
|
|
26
|
+
Jekyll::Hooks.register :site, :pre_render do |site|
|
|
27
|
+
(site.pages + site.collections.values.flat_map(&:docs)).each do |item|
|
|
28
|
+
item.data["search_code"] = Datalog::SearchCodeBlocks.extract(item.content)
|
|
29
|
+
end
|
|
30
|
+
end
|
|
@@ -1,66 +1,28 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
-
begin
|
|
4
|
-
require "unicode_normalize"
|
|
5
|
-
UNICODE_NORMALIZE_SUPPORTED = true
|
|
6
|
-
rescue LoadError
|
|
7
|
-
UNICODE_NORMALIZE_SUPPORTED = false
|
|
8
|
-
warn "[search_normalizer] unicode_normalize gem not available; falling back to basic normalization"
|
|
9
|
-
end
|
|
10
|
-
|
|
11
3
|
module Datalog
|
|
12
4
|
module SearchFilters
|
|
13
5
|
module_function
|
|
14
6
|
|
|
7
|
+
# Lower-cases text and strips combining marks, so "Café" and "cafe" match
|
|
8
|
+
# while letters outside ASCII ("ß", "ł", Cyrillic, CJK) are kept. The
|
|
9
|
+
# browser normalizes queries the same way (assets/js/search/utils.js), so
|
|
10
|
+
# the index and the query agree. String#unicode_normalize is part of Ruby:
|
|
11
|
+
# this file used to require it as if it were a gem, fail, and fall back to
|
|
12
|
+
# dropping every character outside ASCII.
|
|
15
13
|
def normalize_search(input)
|
|
16
|
-
|
|
14
|
+
# Invalid byte sequences are dropped first: unicode_normalize raises on them.
|
|
15
|
+
value = input.to_s.scrub("")
|
|
17
16
|
return "" if value.empty?
|
|
18
17
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
normalized.downcase.strip
|
|
26
|
-
rescue StandardError
|
|
27
|
-
input.to_s.downcase
|
|
28
|
-
end
|
|
29
|
-
|
|
30
|
-
def transliterate(value)
|
|
31
|
-
value.encode("ASCII", fallback: lambda { |char|
|
|
32
|
-
approximate_character(char)
|
|
33
|
-
}, invalid: :replace, undef: :replace, replace: "")
|
|
34
|
-
rescue Encoding::UndefinedConversionError, Encoding::InvalidByteSequenceError
|
|
35
|
-
value
|
|
36
|
-
end
|
|
37
|
-
|
|
38
|
-
def approximate_character(char)
|
|
39
|
-
@transliteration_map ||= build_transliteration_map
|
|
40
|
-
@transliteration_map.fetch(char, "")
|
|
41
|
-
end
|
|
42
|
-
|
|
43
|
-
def build_transliteration_map
|
|
44
|
-
basic_map = {}
|
|
45
|
-
|
|
46
|
-
accents = {
|
|
47
|
-
"ÀÁÂÃÄÅàáâãäå" => "a",
|
|
48
|
-
"ÈÉÊËèéêë" => "e",
|
|
49
|
-
"ÌÍÎÏìíîï" => "i",
|
|
50
|
-
"ÒÓÔÕÖØòóôõöø" => "o",
|
|
51
|
-
"ÙÚÛÜùúûü" => "u",
|
|
52
|
-
"Çç" => "c",
|
|
53
|
-
"Ññ" => "n",
|
|
54
|
-
"Ýýÿ" => "y",
|
|
55
|
-
"Ææ" => "ae",
|
|
56
|
-
"Œœ" => "oe"
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
accents.each do |chars, replacement|
|
|
60
|
-
chars.each_char { |char| basic_map[char] = replacement }
|
|
18
|
+
stripped = begin
|
|
19
|
+
value.unicode_normalize(:nfkd).gsub(/\p{Mn}/, "")
|
|
20
|
+
rescue Encoding::CompatibilityError
|
|
21
|
+
# Text in an encoding other than Unicode cannot be normalized, so it is
|
|
22
|
+
# indexed as it is.
|
|
23
|
+
value
|
|
61
24
|
end
|
|
62
|
-
|
|
63
|
-
basic_map
|
|
25
|
+
stripped.downcase.strip
|
|
64
26
|
end
|
|
65
27
|
|
|
66
28
|
def normalize_search_array(values)
|