datalog-theme 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +63 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +4 -2
  5. data/_data/i18n/en.yml +297 -0
  6. data/_data/i18n/es.yml +297 -0
  7. data/_data/i18n/pt.yml +297 -0
  8. data/_data/js_manifest.json +16 -0
  9. data/_includes/components/author-bio.html +21 -11
  10. data/_includes/components/author-list.html +32 -0
  11. data/_includes/components/citation-tools.html +33 -25
  12. data/_includes/components/comments-thread.html +90 -0
  13. data/_includes/components/contact-form.html +120 -0
  14. data/_includes/components/correction-report.html +82 -0
  15. data/_includes/components/enhanced-toc.html +6 -7
  16. data/_includes/components/license-link.html +13 -0
  17. data/_includes/components/license-notice.html +28 -0
  18. data/_includes/components/moderation-inbox.html +139 -0
  19. data/_includes/components/reactions.html +49 -0
  20. data/_includes/components/reading-list.html +31 -0
  21. data/_includes/components/reading-mode-toggle.html +65 -0
  22. data/_includes/components/reading-state-bookmark.html +27 -0
  23. data/_includes/components/reading-state-panel.html +71 -0
  24. data/_includes/components/reproducibility.html +61 -0
  25. data/_includes/components/responsive-image.html +3 -3
  26. data/_includes/components/revision-history.html +39 -0
  27. data/_includes/components/revision-notice.html +35 -0
  28. data/_includes/components/series-nav.html +64 -0
  29. data/_includes/components/subscribe-form.html +88 -0
  30. data/_includes/components/subscription-manage.html +68 -0
  31. data/_includes/components/webmentions.html +48 -0
  32. data/_includes/csp-meta.html +21 -1
  33. data/_includes/footer.html +3 -0
  34. data/_includes/head.html +51 -40
  35. data/_includes/layouts/default/article.html +4 -6
  36. data/_includes/meta/dynamic-services-config.html +14 -0
  37. data/_includes/meta/math-config.html +9 -5
  38. data/_includes/meta/person-json.html +27 -0
  39. data/_includes/meta/publisher.html +45 -0
  40. data/_includes/meta/schema.html +62 -28
  41. data/_includes/meta/scholarly.html +121 -0
  42. data/_includes/meta/scripts-loader.html +2 -0
  43. data/_includes/meta/webmention-discovery.html +14 -0
  44. data/_includes/scripts.html +57 -0
  45. data/_layouts/dataset.html +4 -3
  46. data/_layouts/default.html +1 -1
  47. data/_layouts/package.html +3 -2
  48. data/_layouts/page.html +15 -0
  49. data/_layouts/post.html +82 -18
  50. data/_layouts/project.html +2 -2
  51. data/_layouts/research.html +20 -8
  52. data/_plugins/authors.rb +133 -0
  53. data/_plugins/config_validator.rb +223 -20
  54. data/_plugins/critical_css_check.rb +42 -0
  55. data/_plugins/i18n.rb +6 -4
  56. data/_plugins/image_optimizer.rb +224 -172
  57. data/_plugins/licenses.rb +135 -0
  58. data/_plugins/math_preprocessor.rb +32 -10
  59. data/_plugins/references.rb +238 -0
  60. data/_plugins/reproducibility.rb +150 -0
  61. data/_plugins/revisions.rb +101 -0
  62. data/_plugins/scholarly.rb +50 -0
  63. data/_plugins/series.rb +104 -0
  64. data/_plugins/statements.rb +87 -0
  65. data/_sass/_base.scss +8 -1
  66. data/_sass/_comments-thread.scss +159 -0
  67. data/_sass/_layout.scss +384 -0
  68. data/_sass/_mathematical.scss +27 -0
  69. data/_sass/_moderation.scss +222 -0
  70. data/_sass/_post-components.scss +4 -2
  71. data/_sass/_print.scss +291 -0
  72. data/_sass/_reactions.scss +89 -0
  73. data/_sass/_reading-state.scss +290 -0
  74. data/_sass/_service-forms.scss +204 -0
  75. data/_sass/_subscriptions.scss +140 -0
  76. data/_sass/_syntax-highlighting.scss +2 -2
  77. data/_sass/_theme.scss +11 -0
  78. data/_sass/_typography.scss +116 -0
  79. data/_sass/_utilities.scss +5 -0
  80. data/_sass/_variables.scss +6 -0
  81. data/_sass/_webmentions.scss +125 -0
  82. data/assets/js/dist/academic.js +1 -1
  83. data/assets/js/dist/analytics-dashboard.js +1 -1
  84. data/assets/js/dist/chunks/chunk-2DYDWUFX.js +1 -0
  85. data/assets/js/dist/chunks/chunk-PATLC23F.js +1 -0
  86. data/assets/js/dist/chunks/chunk-V7734B2G.js +1 -0
  87. data/assets/js/dist/comments.js +2 -0
  88. data/assets/js/dist/contact.js +1 -0
  89. data/assets/js/dist/core.js +1 -1
  90. data/assets/js/dist/corrections.js +1 -0
  91. data/assets/js/dist/loader.js +1 -1
  92. data/assets/js/dist/math.js +1 -1
  93. data/assets/js/dist/moderation.js +1 -0
  94. data/assets/js/dist/notebook.js +1 -1
  95. data/assets/js/dist/reactions.js +1 -0
  96. data/assets/js/dist/reading-state.js +1 -0
  97. data/assets/js/dist/search.js +1 -1
  98. data/assets/js/dist/sources.json +52 -0
  99. data/assets/js/dist/subscriptions.js +1 -0
  100. data/assets/js/dist/webmentions.js +1 -0
  101. data/assets/js/loader.js +34 -0
  102. data/datalog-theme.gemspec +1 -2
  103. data/lib/datalog/cli.rb +9 -1
  104. data/lib/datalog/critical_css.rb +168 -0
  105. data/lib/datalog/plugins/comments.rb +33 -3
  106. data/lib/datalog/theme/installed_files.rb +113 -0
  107. data/lib/datalog/theme/repository_checkout.rb +6 -2
  108. data/lib/datalog/theme/version.rb +5 -1
  109. data/lib/datalog-theme.rb +1 -0
  110. metadata +57 -23
  111. data/assets/js/dist/chunks/chunk-225H5YXE.js +0 -1
@@ -26,6 +26,16 @@ module MathPreprocessor
26
26
  regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
27
27
  tag: "span"
28
28
  },
29
+ # MathJax and KaTeX also render math with spaces inside the dollars, such
30
+ # as `$ \frac{a}{b} $`, which the rule above leaves out, so a page whose
31
+ # only math was written that way loaded no engine. Such a pair counts when
32
+ # its body holds a TeX command, a superscript or a subscript, which prices
33
+ # like `$ 5 or $ 10` do not.
34
+ {
35
+ regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<!\\)(?<close>\$)(?![$\d])/m,
36
+ tag: "span",
37
+ requires: /\\[a-zA-Z]+|[\^_]/
38
+ },
29
39
  {
30
40
  regex: /(?<open>\\\()(?<body>.+?)(?<close>\\\))/m,
31
41
  tag: "span"
@@ -57,22 +67,24 @@ module MathPreprocessor
57
67
  def process
58
68
  return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
59
69
 
60
- code = []
70
+ @segments = []
61
71
  processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
62
- text.gsub(pattern) do |match|
63
- code << match
64
- "\x00#{code.size - 1}\x00"
65
- end
72
+ text.gsub(pattern) { |match| mask(match) }
66
73
  end
67
74
  processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
68
75
  processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
69
76
  # A segment set aside can contain the placeholder of an earlier one.
70
- processed = processed.gsub(PLACEHOLDER) { code[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
77
+ processed = processed.gsub(PLACEHOLDER) { @segments[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
71
78
  processed
72
79
  end
73
80
 
74
81
  private
75
82
 
83
+ def mask(segment)
84
+ @segments << segment
85
+ "\x00#{@segments.size - 1}\x00"
86
+ end
87
+
76
88
  def apply_patterns(text, patterns, display: false)
77
89
  patterns.reduce(text) do |result, pattern|
78
90
  result.gsub(pattern[:regex]) do |match|
@@ -81,8 +93,11 @@ module MathPreprocessor
81
93
  close = Regexp.last_match[:close]
82
94
 
83
95
  next match if body.nil? || body.strip.empty?
96
+ next match if pattern[:requires] && !pattern[:requires].match?(body)
84
97
 
85
- wrapper_for(match, body, open, close, pattern[:tag], display: display)
98
+ # Each wrapper is set aside like code. Otherwise a later pattern could
99
+ # pair a dollar sign inside it with one in the text that follows.
100
+ mask(wrapper_for(match, body, open, close, pattern[:tag], display: display))
86
101
  end
87
102
  end
88
103
  end
@@ -195,9 +210,9 @@ module MathPreprocessor
195
210
  def apply(document)
196
211
  return unless document.respond_to?(:content)
197
212
  return unless document.respond_to?(:output_ext) && document.output_ext == ".html"
198
- # A page that opts out of math rendering (`math: false` or `mathjax: false`)
199
- # keeps its dollar signs and TeX-looking text verbatim.
200
- return if document.respond_to?(:data) && (document.data["math"] == false || document.data["mathjax"] == false)
213
+ # A page that opts out of math rendering keeps its dollar signs and
214
+ # TeX-looking text verbatim.
215
+ return if document.respond_to?(:data) && math_setting(document.data) == false
201
216
 
202
217
  content = document.content
203
218
  return unless content&.match?(/\$|\\\(|\\\[|\\begin\{/)
@@ -207,6 +222,13 @@ module MathPreprocessor
207
222
  document.content = updated_content
208
223
  document.data["math_expressions"] = processor.expressions if processor.expressions.any?
209
224
  end
225
+
226
+ # `math`, or its alias `mathjax` when `math` is unset, as
227
+ # _includes/meta/math-config.html reads them. A `mathjax: true` in front
228
+ # matter defaults used to win over a page's `math: false`.
229
+ def math_setting(data)
230
+ data["math"].nil? ? data["mathjax"] : data["math"]
231
+ end
210
232
  end
211
233
 
212
234
  # Posts are documents, so registering them separately ran the preprocessor
@@ -0,0 +1,238 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi"
4
+
5
+ module Datalog
6
+ # Numbered figures and tables, and references to them, within a page:
7
+ #
8
+ # {% figure id="fig-power" src="/assets/img/power.png" alt="Power curve" %}
9
+ # Power as a function of effect size $\delta$.
10
+ # {% endfigure %}
11
+ #
12
+ # {% table id="tab-runs" %}
13
+ # Simulation runs by sample size.
14
+ #
15
+ # | n | runs |
16
+ # |---|------|
17
+ # {% endtable %}
18
+ #
19
+ # As {% ref fig-power %} shows, ...
20
+ #
21
+ # The tags write placeholders. Once the page's Markdown is converted, every
22
+ # target gets its number in the order it appears, so a reference may come
23
+ # before its figure, and each reference becomes a link reading "Figure 2".
24
+ # A duplicate id or a reference to nothing stops the build. Equations are
25
+ # numbered and referenced by MathJax (\label and \eqref) instead.
26
+ #
27
+ # A caption is the body of the tag, not an attribute, so math in it goes
28
+ # through the math preprocessor like the rest of the page.
29
+ #
30
+ # Theorems, definitions and the other statements in _plugins/statements.rb
31
+ # are numbered and referred to the same way.
32
+ module References
33
+ module_function
34
+
35
+ # Each kind counts separately; the label comes from _data/i18n.
36
+ KINDS = {
37
+ "figure" => "Figure", "table" => "Table",
38
+ "theorem" => "Theorem", "lemma" => "Lemma", "proposition" => "Proposition", "corollary" => "Corollary",
39
+ "definition" => "Definition", "assumption" => "Assumption", "example" => "Example", "remark" => "Remark"
40
+ }.freeze
41
+ ID = /\A[A-Za-z][\w.:-]*\z/
42
+ # A label given in place of the number, as in "Theorem A".
43
+ CUSTOM_LABEL = /\A[[:alnum:]][[:alnum:].'*-]*\z/
44
+ TARGET = /data-ref-target="([^"]+)" data-ref-kind="([a-z]+)"(?: data-ref-number="([^"]+)")?/
45
+ # A label ends with a full stop unless the tag chose another ending.
46
+ LABEL = %r{<span class="datalog-ref-label" data-ref-for="([^"]+)"(?: data-ref-end="([^"]*)")?></span>}
47
+ LINK = %r{<a class="datalog-ref" href="#([^"]+)" data-ref="\1">[^<]*</a>}
48
+ SOURCE_TAG = /\{%-?\s*(#{KINDS.keys.join('|')})\b([^%]*)-?%\}/
49
+
50
+ def number(document)
51
+ content = document.content
52
+ return unless content&.include?("data-ref")
53
+
54
+ targets = targets(content.scan(TARGET), document)
55
+ content = content.gsub(LABEL) do
56
+ id, ending = Regexp.last_match.captures
57
+ text = "#{CGI.escapeHTML(targets.fetch(id))}#{ending || '.'}"
58
+ %(<span class="datalog-ref-label" data-ref-for="#{id}">#{text}</span>)
59
+ end
60
+ document.content = link_references(content, targets, document)
61
+ end
62
+
63
+ # Jekyll runs no hooks for an excerpt, and renders it when a template first
64
+ # asks for it, which on a listing page can be before its post is converted.
65
+ # Its references take their numbers from the post: from the converted
66
+ # content when there is one, and otherwise from the tags in the source,
67
+ # which are in the same order. A missing target is the post's error to report.
68
+ def number_excerpt(excerpt, html)
69
+ return html unless html&.include?("data-ref")
70
+
71
+ post = excerpt.doc
72
+ found = post.content.to_s.scan(TARGET)
73
+ found = source_targets(post.content.to_s) if found.empty?
74
+ targets = targets(found.uniq(&:first), post)
75
+ html.gsub(LINK) do
76
+ id = Regexp.last_match(1)
77
+ text = targets[id] ? CGI.escapeHTML(targets[id]) : id
78
+ %(<a class="datalog-ref" href="#{post.site.baseurl}#{post.url}##{id}" data-ref="#{id}">#{text}</a>)
79
+ end
80
+ end
81
+
82
+ # [id, kind, label] for each numbered tag in a page's source.
83
+ def source_targets(source)
84
+ source.scan(SOURCE_TAG).filter_map do |kind, markup|
85
+ attributes = markup.scan(/(\w+)=["']([^"']*)["']/).to_h
86
+ [attributes["id"], kind, attributes["label"]] if attributes["id"]
87
+ end
88
+ end
89
+
90
+ # The id => "Figure 2" of every [id, kind, label] target, numbered per kind in
91
+ # page order. A target with its own label does not take a number.
92
+ def targets(found, document)
93
+ counts = Hash.new(0)
94
+ found.each_with_object({}) do |(id, kind, custom), targets|
95
+ if targets.key?(id)
96
+ raise Jekyll::Errors::FatalException,
97
+ "#{document.relative_path} has two numbered figures, tables or statements with the id " \
98
+ "\"#{id}\"; each id has to be unique"
99
+ end
100
+
101
+ number = custom ? CGI.unescapeHTML(custom) : (counts[kind] += 1)
102
+ targets[id] = "#{label(document, kind)} #{number}"
103
+ end
104
+ end
105
+
106
+ def link_references(content, targets, document)
107
+ missing = content.scan(LINK).flatten.uniq - targets.keys
108
+ unless missing.empty?
109
+ raise Jekyll::Errors::FatalException,
110
+ "#{document.relative_path} refers to #{missing.map { |id| "\"#{id}\"" }.join(', ')}, which no " \
111
+ "numbered figure, table or statement on the page has as its id"
112
+ end
113
+
114
+ content.gsub(LINK) do
115
+ id = Regexp.last_match(1)
116
+ %(<a class="datalog-ref" href="##{id}" data-ref="#{id}">#{CGI.escapeHTML(targets[id])}</a>)
117
+ end
118
+ end
119
+
120
+ def label(document, kind)
121
+ site = document.site
122
+ locale = I18n.locale_code(site, document.data["lang"])
123
+ I18n.lookup(site, locale, "references.#{kind}") || KINDS.fetch(kind)
124
+ end
125
+
126
+ def validate_label(label, tag)
127
+ return if label.nil? || label.match?(CUSTOM_LABEL)
128
+
129
+ raise Liquid::ArgumentError,
130
+ "{% #{tag} %} takes a label of letters, digits, \".\", \"'\", \"*\" or \"-\", such as label=\"A\"; " \
131
+ "got #{label.inspect}"
132
+ end
133
+
134
+ def validate_id(id, tag)
135
+ return id if id.to_s.match?(ID)
136
+
137
+ raise Liquid::ArgumentError,
138
+ "{% #{tag} %} needs an id that starts with a letter and holds only letters, digits, " \
139
+ "\"-\", \"_\", \".\" or \":\", such as id=\"fig-power\"; got #{id.inspect}"
140
+ end
141
+
142
+ # key="value", key='value' or key=variable.
143
+ def attributes(markup, context)
144
+ markup.scan(/(\w+)=(?:"([^"]*)"|'([^']*)'|([\w.\[\]-]+))/).to_h do |key, double, single, variable|
145
+ [key, double || single || context[variable].to_s]
146
+ end
147
+ end
148
+
149
+ def markdown(context, text)
150
+ site = context.registers[:site]
151
+ site.find_converter_instance(Jekyll::Converters::Markdown).convert(text.to_s.strip)
152
+ end
153
+
154
+ # A caption of one paragraph loses its <p>, which a <figcaption> or <caption> does not need.
155
+ def inline(html)
156
+ html = html.strip
157
+ paragraph = html.match(%r{\A<p>(.*)</p>\z}m)
158
+ paragraph && !paragraph[1].include?("<p>") ? paragraph[1] : html
159
+ end
160
+ end
161
+
162
+ class FigureTag < Liquid::Block
163
+ def initialize(tag_name, markup, options)
164
+ super
165
+ @markup = markup
166
+ end
167
+
168
+ def render(context)
169
+ attributes = References.attributes(@markup, context)
170
+ id = References.validate_id(attributes["id"], "figure")
171
+ src = attributes["src"].to_s
172
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs a src, the image it shows" if src.empty?
173
+ if attributes["alt"].to_s.strip.empty?
174
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs alt text for its image"
175
+ end
176
+
177
+ src = "#{context.registers[:site].config['baseurl'].to_s.chomp('/')}#{src}" if src.start_with?("/")
178
+ caption = References.inline(References.markdown(context, super))
179
+ classes = ["datalog-figure", attributes["class"]].compact.join(" ")
180
+
181
+ alt = CGI.escapeHTML(attributes["alt"])
182
+ %(<figure class="#{CGI.escapeHTML(classes)}" id="#{id}" data-ref-target="#{id}" data-ref-kind="figure">) +
183
+ %(<img src="#{CGI.escapeHTML(src)}" alt="#{alt}" loading="lazy" decoding="async">) +
184
+ %(<figcaption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</figcaption></figure>\n)
185
+ end
186
+ end
187
+
188
+ class TableTag < Liquid::Block
189
+ def initialize(tag_name, markup, options)
190
+ super
191
+ @markup = markup
192
+ end
193
+
194
+ # The body holds the caption and then a Markdown table.
195
+ def render(context)
196
+ id = References.validate_id(References.attributes(@markup, context)["id"], "table")
197
+ html = References.markdown(context, super)
198
+ tables = html.scan(/<table\b/).size
199
+ unless tables == 1
200
+ raise Liquid::ArgumentError, "{% table id=\"#{id}\" %} holds #{tables} tables; it needs one, after its caption"
201
+ end
202
+
203
+ caption, table = html.split(/(?=<table\b)/, 2)
204
+ caption = References.inline(caption)
205
+ numbered = table.sub(/<table\b([^>]*)>/) do
206
+ %(<table#{Regexp.last_match(1)} id="#{id}" data-ref-target="#{id}" data-ref-kind="table">) +
207
+ %(<caption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</caption>)
208
+ end
209
+ "#{numbered.strip}\n"
210
+ end
211
+ end
212
+
213
+ class ReferenceTag < Liquid::Tag
214
+ def initialize(tag_name, markup, options)
215
+ super
216
+ @id = References.validate_id(markup.strip, "ref")
217
+ end
218
+
219
+ def render(_context)
220
+ %(<a class="datalog-ref" href="##{@id}" data-ref="#{@id}">#{@id}</a>)
221
+ end
222
+ end
223
+ end
224
+
225
+ Liquid::Template.register_tag("figure", Datalog::FigureTag)
226
+ Liquid::Template.register_tag("table", Datalog::TableTag)
227
+ Liquid::Template.register_tag("ref", Datalog::ReferenceTag)
228
+
229
+ Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
230
+ Datalog::References.number(document)
231
+ end
232
+
233
+ Jekyll::Excerpt.prepend(Module.new do
234
+ # Excerpt#output renders once and keeps the result; numbering it is a quick gsub.
235
+ def output
236
+ Datalog::References.number_excerpt(self, super)
237
+ end
238
+ end)
@@ -0,0 +1,150 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "uri"
4
+
5
+ module Datalog
6
+ # The computational artifacts behind an article, for the "Reproduce this
7
+ # analysis" panel (_includes/components/reproducibility.html):
8
+ #
9
+ # reproducibility:
10
+ # code:
11
+ # url: https://github.com/example/project
12
+ # ref: 4f2c1ab # the commit, tag or branch the article used
13
+ # data:
14
+ # doi: 10.5281/zenodo.1234567 # or url:
15
+ # version: v2
16
+ # environment:
17
+ # file: requirements.txt # in the code repository at the ref, a site path or a URL
18
+ # container: ghcr.io/example/project:1.4.0
19
+ # archive: https://doi.org/10.5281/zenodo.7654321
20
+ # notebook:
21
+ # url: /notebooks/example/
22
+ # results:
23
+ # url: https://github.com/example/project/releases/tag/results-v1
24
+ # version: results-v1
25
+ #
26
+ # Every artifact is optional, and each may be a bare URL. The panel shows
27
+ # what is given and claims nothing more: a link is a link, a ref is a ref.
28
+ # A URL that is not http(s), a site path or a DOI stops the build.
29
+ module Reproducibility
30
+ module_function
31
+
32
+ KINDS = %w[code data notebook environment results].freeze
33
+ # Where a ref and a file in the repository can be linked.
34
+ HOSTS = {
35
+ "github.com" => { tree: "/tree/%s", blob: "/blob/%s/%s" },
36
+ "gitlab.com" => { tree: "/-/tree/%s", blob: "/-/blob/%s/%s" }
37
+ }.freeze
38
+ URL = %r{\Ahttps?://[^\s"'<>]+\z}i
39
+ PATH = %r{\A/[^\s"'<>]*\z}
40
+
41
+ # The artifacts, in KINDS order, or nil when the page gives none.
42
+ def resolve(page, site)
43
+ value = Authors.value(page, "reproducibility")
44
+ return unless value.is_a?(Hash)
45
+
46
+ given = value.transform_keys(&:to_s)
47
+ baseurl = Authors.value(site, "baseurl").to_s
48
+ artifacts = KINDS.filter_map { |kind| artifact(kind, given[kind], given, page, baseurl) }
49
+ { "artifacts" => artifacts } unless artifacts.empty?
50
+ end
51
+
52
+ def artifact(kind, value, all, page, baseurl)
53
+ return if value.nil? || value == false || (value.is_a?(String) && value.strip.empty?)
54
+
55
+ given = value.is_a?(Hash) ? Authors.present(value) : { "url" => value.to_s }
56
+ url = link(given["url"] || doi_url(given["doi"]), page, "#{kind}.url", baseurl)
57
+ artifact = { "kind" => kind, "url" => url, "external" => external?(url), "version" => given["version"]&.to_s,
58
+ "label" => given["label"] || display(url), "doi" => bare_doi(given["doi"]) }
59
+ artifact.merge!(code(given, url)) if kind == "code"
60
+ artifact.merge!(environment(given, all, page, baseurl)) if kind == "environment"
61
+ artifact = artifact.compact
62
+ artifact if artifact.values_at("url", "file", "container", "archive").any?
63
+ end
64
+
65
+ def code(given, url)
66
+ ref = given["ref"].to_s.strip
67
+ return {} if ref.empty?
68
+
69
+ { "ref" => ref, "ref_url" => host_url(url, :tree, ref) }
70
+ end
71
+
72
+ # The environment file lives in the code repository at the article's ref
73
+ # unless it is a site path or a URL of its own.
74
+ def environment(given, all, page, baseurl)
75
+ file = given["file"].to_s.strip
76
+ archive = link(given["archive"], page, "environment.archive", baseurl)
77
+ {
78
+ "file" => (file unless file.empty?),
79
+ "file_url" => (file_url(file, all["code"], page, baseurl) unless file.empty?),
80
+ "container" => given["container"]&.to_s,
81
+ "archive" => archive, "archive_label" => display(archive)
82
+ }
83
+ end
84
+
85
+ def file_url(file, code, page, baseurl)
86
+ own = file.match?(URL) || file.match?(PATH) || file.match?(/\Adoi:/i)
87
+ return link(file, page, "environment.file", baseurl) if own
88
+
89
+ code = code.is_a?(Hash) ? Authors.present(code) : { "url" => code.to_s }
90
+ code_url = link(code["url"], page, "code.url", baseurl)
91
+ ref = code["ref"].to_s.strip
92
+ host_url(code_url, :blob, ref.empty? ? "HEAD" : ref, file)
93
+ end
94
+
95
+ # A page of a known host under the repository URL, such as a tree or a blob.
96
+ def host_url(url, kind, *parts)
97
+ return unless url
98
+
99
+ host = URI.parse(url).host.to_s.sub(/\Awww\./, "")
100
+ pattern = HOSTS.dig(host, kind)
101
+ "#{url.chomp('/')}#{format(pattern, *parts)}" if pattern
102
+ rescue URI::InvalidURIError
103
+ nil
104
+ end
105
+
106
+ # An http(s) URL as given, a site path with the baseurl, or a DOI as its
107
+ # URL; anything else, such as a javascript: URL or an address without its
108
+ # scheme, stops the build.
109
+ def link(value, page, field, baseurl)
110
+ text = value.to_s.strip
111
+ return if text.empty?
112
+ return doi_url(text) if text.match?(/\Adoi:/i)
113
+ return text if text.match?(URL)
114
+ return "#{baseurl}#{text}" if text.match?(PATH)
115
+
116
+ raise Jekyll::Errors::FatalException,
117
+ "#{Authors.value(page, 'path')} reproducibility.#{field} is #{text.inspect}, which is not an http(s) " \
118
+ "URL, a site path starting with / or a doi:; give the whole address, such as https://github.com/example/project"
119
+ end
120
+
121
+ def bare_doi(doi)
122
+ text = doi.to_s.strip.sub(%r{\Ahttps?://(dx\.)?doi\.org/}i, "").sub(/\Adoi:\s*/i, "")
123
+ text unless text.empty?
124
+ end
125
+
126
+ def doi_url(doi)
127
+ bare = bare_doi(doi)
128
+ "https://doi.org/#{bare}" if bare
129
+ end
130
+
131
+ # What a link reads: the address without its scheme, or the site path.
132
+ def display(url)
133
+ return url unless external?(url)
134
+
135
+ url.sub(%r{\Ahttps?://(www\.)?}i, "").chomp("/")
136
+ end
137
+
138
+ def external?(url)
139
+ url.to_s.match?(URL)
140
+ end
141
+ end
142
+
143
+ module ReproducibilityFilters
144
+ def page_reproducibility(page)
145
+ Reproducibility.resolve(page, @context["site"])
146
+ end
147
+ end
148
+ end
149
+
150
+ Liquid::Template.register_filter(Datalog::ReproducibilityFilters)
@@ -0,0 +1,101 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "date"
4
+ require "time"
5
+
6
+ module Datalog
7
+ # The revision history of an article, for a post that is corrected or
8
+ # rewritten while keeping its URL:
9
+ #
10
+ # revisions:
11
+ # - date: 2026-09-16
12
+ # type: correction
13
+ # summary: Replaced the sample-size rules with model diagnostics.
14
+ # details_url: https://github.com/example/repo/pull/425
15
+ # - date: 2024-03-12
16
+ # type: update
17
+ # summary: Updated the code examples for the current SciPy.
18
+ #
19
+ # Before the site renders, each page's list is checked, its dates are
20
+ # parsed and its entries sorted newest first, so the includes read one
21
+ # shape. A correction or an update is substantive: the newest one becomes
22
+ # `revision_notice`, which the post layout announces under the metadata.
23
+ # A review or an editorial change appears in the history only.
24
+ #
25
+ # Every revision but a review changed the article, so the newest one sets
26
+ # `last_modified_at` when it is later than the date the page gives, and
27
+ # the JSON-LD, the microdata, the feed and the sitemap all carry it.
28
+ module Revisions
29
+ module_function
30
+
31
+ TYPES = %w[correction update review editorial].freeze
32
+ SUBSTANTIVE = %w[correction update].freeze
33
+ DEFAULT_TYPE = "update"
34
+
35
+ def normalize!(document)
36
+ data = document.data
37
+ return unless data.key?("revisions")
38
+
39
+ revisions = entries(data["revisions"], document)
40
+ data["revisions"] = revisions
41
+ notice = revisions.find { |revision| revision["substantive"] }
42
+ data["revision_notice"] = notice unless data["revision_notice"] == false
43
+
44
+ changed = revisions.find { |revision| revision["type"] != "review" }
45
+ modified = to_time(data["last_modified_at"] || data["updated"])
46
+ data["last_modified_at"] = changed["date"] if changed && (modified.nil? || changed["date"] > modified)
47
+ end
48
+
49
+ # Newest first; revisions on one day keep their order.
50
+ def entries(list, document)
51
+ unless list.is_a?(Array)
52
+ stop(document, "has revisions that is not a list; each revision is a map with a date, a summary and, " \
53
+ "if wanted, a type and a details_url")
54
+ end
55
+
56
+ revisions = list.each_with_index.map { |entry, index| revision(entry, index + 1, document) }
57
+ revisions.each_with_index.sort_by { |revision, index| [-revision["date"].to_i, index] }.map(&:first)
58
+ end
59
+
60
+ def revision(entry, number, document)
61
+ stop(document, "revision #{number} is not a map with a date and a summary") unless entry.is_a?(Hash)
62
+
63
+ entry = entry.transform_keys(&:to_s)
64
+ type = (entry["type"] || DEFAULT_TYPE).to_s.strip.downcase
65
+ unless TYPES.include?(type)
66
+ stop(document, "revision #{number} has the type #{entry['type'].inspect}; it takes #{TYPES.join(', ')}")
67
+ end
68
+ summary = entry["summary"].to_s.strip
69
+ stop(document, "revision #{number} needs a summary saying what changed") if summary.empty?
70
+ date = to_time(entry["date"])
71
+ unless date
72
+ stop(document, "revision #{number} has the date #{entry['date'].inspect}, which is not a date such as " \
73
+ "2026-09-16")
74
+ end
75
+
76
+ details = entry["details_url"].to_s.strip
77
+ { "date" => date, "type" => type, "summary" => summary, "details_url" => (details unless details.empty?),
78
+ "substantive" => SUBSTANTIVE.include?(type) }.compact
79
+ end
80
+
81
+ def stop(document, problem)
82
+ raise Jekyll::Errors::FatalException, "#{document.relative_path} #{problem}"
83
+ end
84
+
85
+ # A Date, a Time or a string naming a day, such as "2026-09-16".
86
+ def to_time(value)
87
+ case value
88
+ when Time then value
89
+ when Date then value.to_time
90
+ when String
91
+ parts = Date._parse(value)
92
+ Time.parse(value) if parts[:year] && parts[:mon] && parts[:mday]
93
+ end
94
+ end
95
+ end
96
+ end
97
+
98
+ Jekyll::Hooks.register :site, :post_read do |site|
99
+ site.documents.each { |document| Datalog::Revisions.normalize!(document) }
100
+ site.pages.each { |page| Datalog::Revisions.normalize!(page) }
101
+ end
@@ -0,0 +1,50 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Datalog
4
+ # Which pages get scholarly discovery metadata (_includes/meta/scholarly.html):
5
+ # the Highwire meta tags Google Scholar and reference managers read, and
6
+ # their Dublin Core equivalents. Not every post is a paper, so the tags are
7
+ # opt-in:
8
+ #
9
+ # scholarly: true # front matter: this page is a research article
10
+ # scholarly: false # front matter: this one is not, whatever the site says
11
+ #
12
+ # scholarly: true # _config.yml: every post and research article
13
+ # scholarly: [notebooks] # _config.yml: these collections or layouts as well
14
+ #
15
+ # A page with the research layout, or in a research collection, is
16
+ # scholarly unless it says otherwise.
17
+ module Scholarly
18
+ module_function
19
+
20
+ ALWAYS = %w[research].freeze
21
+ POSTS = %w[posts post].freeze
22
+
23
+ def scholarly?(page, site)
24
+ own = Authors.value(page, "scholarly")
25
+ return own == true unless own.nil?
26
+
27
+ kinds = kinds(Authors.value(site, "scholarly"))
28
+ [Authors.value(page, "layout"), Authors.value(page, "collection")].any? { |kind| kinds.include?(kind.to_s) }
29
+ end
30
+
31
+ # The layouts and collections the site's setting covers, besides research.
32
+ def kinds(setting)
33
+ case setting
34
+ when true then ALWAYS + POSTS
35
+ when Array then ALWAYS + setting.map(&:to_s)
36
+ when String then ALWAYS + setting.split(/[\s,]+/)
37
+ else ALWAYS
38
+ end
39
+ end
40
+ end
41
+
42
+ module ScholarlyFilters
43
+ # A Liquid filter's name cannot end with "?".
44
+ def scholarly(page) # rubocop:disable Naming/PredicateMethod
45
+ Scholarly.scholarly?(page, @context["site"])
46
+ end
47
+ end
48
+ end
49
+
50
+ Liquid::Template.register_filter(Datalog::ScholarlyFilters)