datalog-theme 0.8.0 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +84 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +4 -2
  5. data/_data/i18n/en.yml +321 -0
  6. data/_data/i18n/es.yml +321 -0
  7. data/_data/i18n/pt.yml +321 -0
  8. data/_data/js_manifest.json +16 -0
  9. data/_includes/components/academic-dashboard.html +4 -4
  10. data/_includes/components/author-bio.html +21 -11
  11. data/_includes/components/author-list.html +32 -0
  12. data/_includes/components/breadcrumbs.html +6 -4
  13. data/_includes/components/citation-tools.html +18 -58
  14. data/_includes/components/comments-thread.html +90 -0
  15. data/_includes/components/contact-form.html +120 -0
  16. data/_includes/components/correction-report.html +82 -0
  17. data/_includes/components/enhanced-toc.html +6 -7
  18. data/_includes/components/hero.html +1 -1
  19. data/_includes/components/license-link.html +13 -0
  20. data/_includes/components/license-notice.html +28 -0
  21. data/_includes/components/moderation-inbox.html +139 -0
  22. data/_includes/components/post-hero.html +2 -2
  23. data/_includes/components/reactions.html +49 -0
  24. data/_includes/components/reading-list.html +34 -0
  25. data/_includes/components/reading-mode-toggle.html +67 -0
  26. data/_includes/components/reading-state-bookmark.html +27 -0
  27. data/_includes/components/reading-state-panel.html +62 -0
  28. data/_includes/components/reading-state-resume.html +17 -0
  29. data/_includes/components/reproducibility.html +61 -0
  30. data/_includes/components/responsive-image.html +3 -3
  31. data/_includes/components/revision-history.html +39 -0
  32. data/_includes/components/revision-notice.html +35 -0
  33. data/_includes/components/series-nav.html +64 -0
  34. data/_includes/components/subscribe-form.html +88 -0
  35. data/_includes/components/subscription-manage.html +69 -0
  36. data/_includes/components/visualization-card.html +1 -1
  37. data/_includes/components/webmentions.html +48 -0
  38. data/_includes/csp-meta.html +21 -1
  39. data/_includes/footer/nav-column.html +1 -1
  40. data/_includes/footer.html +5 -2
  41. data/_includes/head.html +64 -52
  42. data/_includes/header/navigation.html +2 -2
  43. data/_includes/header.html +1 -1
  44. data/_includes/layouts/default/article.html +6 -8
  45. data/_includes/meta/dynamic-services-config.html +14 -0
  46. data/_includes/meta/math-config.html +12 -5
  47. data/_includes/meta/person-json.html +27 -0
  48. data/_includes/meta/publisher.html +45 -0
  49. data/_includes/meta/schema.html +84 -48
  50. data/_includes/meta/scholarly.html +122 -0
  51. data/_includes/meta/scripts-loader.html +2 -0
  52. data/_includes/meta/webmention-discovery.html +14 -0
  53. data/_includes/post/related-posts.html +1 -1
  54. data/_includes/scripts.html +57 -0
  55. data/_layouts/dataset.html +5 -4
  56. data/_layouts/default.html +1 -1
  57. data/_layouts/home.html +2 -2
  58. data/_layouts/notebook.html +1 -1
  59. data/_layouts/package.html +8 -7
  60. data/_layouts/page.html +15 -0
  61. data/_layouts/portfolio.html +1 -1
  62. data/_layouts/post.html +94 -25
  63. data/_layouts/project.html +5 -5
  64. data/_layouts/research.html +27 -75
  65. data/_plugins/authors.rb +151 -0
  66. data/_plugins/citation_exports.rb +93 -0
  67. data/_plugins/config_validator.rb +223 -20
  68. data/_plugins/critical_css_check.rb +42 -0
  69. data/_plugins/i18n.rb +6 -4
  70. data/_plugins/image_optimizer.rb +240 -186
  71. data/_plugins/licenses.rb +147 -0
  72. data/_plugins/math_preprocessor.rb +55 -11
  73. data/_plugins/references.rb +260 -0
  74. data/_plugins/reproducibility.rb +167 -0
  75. data/_plugins/responsive_content.rb +40 -0
  76. data/_plugins/revisions.rb +102 -0
  77. data/_plugins/scholarly.rb +55 -0
  78. data/_plugins/series.rb +104 -0
  79. data/_plugins/statements.rb +87 -0
  80. data/_sass/_base.scss +13 -1
  81. data/_sass/_comments-thread.scss +159 -0
  82. data/_sass/_layout.scss +402 -1
  83. data/_sass/_mathematical.scss +38 -1
  84. data/_sass/_moderation.scss +222 -0
  85. data/_sass/_post-components.scss +4 -2
  86. data/_sass/_print.scss +299 -0
  87. data/_sass/_reactions.scss +89 -0
  88. data/_sass/_reading-state.scss +290 -0
  89. data/_sass/_service-forms.scss +204 -0
  90. data/_sass/_subscriptions.scss +140 -0
  91. data/_sass/_syntax-highlighting.scss +3 -3
  92. data/_sass/_theme.scss +11 -0
  93. data/_sass/_typography.scss +116 -0
  94. data/_sass/_utilities.scss +5 -0
  95. data/_sass/_variables.scss +6 -0
  96. data/_sass/_webmentions.scss +125 -0
  97. data/assets/js/dist/academic.js +1 -1
  98. data/assets/js/dist/analytics-dashboard.js +1 -1
  99. data/assets/js/dist/chunks/chunk-PATLC23F.js +1 -0
  100. data/assets/js/dist/chunks/chunk-VZ5WKQVA.js +1 -0
  101. data/assets/js/dist/chunks/chunk-WIRUK3OZ.js +1 -0
  102. data/assets/js/dist/comments.js +2 -0
  103. data/assets/js/dist/contact.js +1 -0
  104. data/assets/js/dist/core.js +1 -1
  105. data/assets/js/dist/corrections.js +1 -0
  106. data/assets/js/dist/loader.js +1 -1
  107. data/assets/js/dist/math.js +1 -1
  108. data/assets/js/dist/moderation.js +1 -0
  109. data/assets/js/dist/notebook.js +1 -1
  110. data/assets/js/dist/reactions.js +1 -0
  111. data/assets/js/dist/reading-state.js +1 -0
  112. data/assets/js/dist/search.js +1 -1
  113. data/assets/js/dist/sources.json +52 -0
  114. data/assets/js/dist/subscriptions.js +1 -0
  115. data/assets/js/dist/visualizations.js +2 -2
  116. data/assets/js/dist/webmentions.js +1 -0
  117. data/assets/js/loader.js +34 -0
  118. data/datalog-theme.gemspec +1 -2
  119. data/lib/datalog/cli.rb +9 -1
  120. data/lib/datalog/critical_css.rb +168 -0
  121. data/lib/datalog/plugins/comments.rb +33 -3
  122. data/lib/datalog/theme/installed_files.rb +113 -0
  123. data/lib/datalog/theme/repository_checkout.rb +6 -2
  124. data/lib/datalog/theme/version.rb +5 -1
  125. data/lib/datalog-theme.rb +1 -0
  126. metadata +60 -23
  127. data/assets/js/dist/chunks/chunk-225H5YXE.js +0 -1
@@ -0,0 +1,147 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "time"
4
+
5
+ module Datalog
6
+ # The licence of a page's text and figures, and of its code samples, for
7
+ # the reuse notice, the JSON-LD and the head's rel="license" link:
8
+ #
9
+ # content_license: CC-BY-4.0 # _config.yml: the default for every article
10
+ #
11
+ # license: CC-BY-SA-4.0 # front matter: this page's own
12
+ # license: false # front matter: none for this page
13
+ # license: # front matter: a licence the theme does not know
14
+ # name: Open Government Licence v3.0
15
+ # url: https://www.nationalarchives.gov.uk/doc/open-government-licence/version/3/
16
+ # holder: The Lab
17
+ # year: 2026
18
+ # code_license: MIT # the code samples, when their licence differs
19
+ #
20
+ # The repository's LICENSE covers the theme's software; these settings are
21
+ # for what a site publishes. Datasets and packages carry their own
22
+ # `license` and never take the site's default.
23
+ module Licenses
24
+ module_function
25
+
26
+ # SPDX identifiers, with the name readers see and the licence text.
27
+ KNOWN = {
28
+ "CC-BY-4.0" => ["CC BY 4.0", "https://creativecommons.org/licenses/by/4.0/"],
29
+ "CC-BY-SA-4.0" => ["CC BY-SA 4.0", "https://creativecommons.org/licenses/by-sa/4.0/"],
30
+ "CC-BY-ND-4.0" => ["CC BY-ND 4.0", "https://creativecommons.org/licenses/by-nd/4.0/"],
31
+ "CC-BY-NC-4.0" => ["CC BY-NC 4.0", "https://creativecommons.org/licenses/by-nc/4.0/"],
32
+ "CC-BY-NC-SA-4.0" => ["CC BY-NC-SA 4.0", "https://creativecommons.org/licenses/by-nc-sa/4.0/"],
33
+ "CC-BY-NC-ND-4.0" => ["CC BY-NC-ND 4.0", "https://creativecommons.org/licenses/by-nc-nd/4.0/"],
34
+ "CC0-1.0" => ["CC0 1.0", "https://creativecommons.org/publicdomain/zero/1.0/"],
35
+ "MIT" => ["MIT", "https://spdx.org/licenses/MIT.html"],
36
+ "Apache-2.0" => ["Apache 2.0", "https://spdx.org/licenses/Apache-2.0.html"],
37
+ "BSD-2-Clause" => ["BSD 2-Clause", "https://spdx.org/licenses/BSD-2-Clause.html"],
38
+ "BSD-3-Clause" => ["BSD 3-Clause", "https://spdx.org/licenses/BSD-3-Clause.html"],
39
+ "GPL-3.0-only" => ["GPL 3.0", "https://spdx.org/licenses/GPL-3.0-only.html"],
40
+ "GPL-3.0-or-later" => ["GPL 3.0 or later", "https://spdx.org/licenses/GPL-3.0-or-later.html"],
41
+ "LGPL-3.0-only" => ["LGPL 3.0", "https://spdx.org/licenses/LGPL-3.0-only.html"],
42
+ "AGPL-3.0-only" => ["AGPL 3.0", "https://spdx.org/licenses/AGPL-3.0-only.html"],
43
+ "MPL-2.0" => ["MPL 2.0", "https://spdx.org/licenses/MPL-2.0.html"],
44
+ "ISC" => ["ISC", "https://spdx.org/licenses/ISC.html"],
45
+ "Unlicense" => ["The Unlicense", "https://spdx.org/licenses/Unlicense.html"],
46
+ "all-rights-reserved" => ["All rights reserved", nil]
47
+ }.freeze
48
+
49
+ # Other spellings: "CC BY" and "cc0" take the current version; the GPL
50
+ # family without a suffix means "only", as SPDX reads it.
51
+ ALIASES = {
52
+ "CC-BY" => "CC-BY-4.0", "CC-BY-SA" => "CC-BY-SA-4.0", "CC-BY-ND" => "CC-BY-ND-4.0",
53
+ "CC-BY-NC" => "CC-BY-NC-4.0", "CC-BY-NC-SA" => "CC-BY-NC-SA-4.0", "CC-BY-NC-ND" => "CC-BY-NC-ND-4.0",
54
+ "CC0" => "CC0-1.0", "GPL-3.0" => "GPL-3.0-only", "LGPL-3.0" => "LGPL-3.0-only", "AGPL-3.0" => "AGPL-3.0-only"
55
+ }.freeze
56
+
57
+ # Collections whose pages carry their own licence and never the site's.
58
+ OWN_LICENSE = %w[datasets packages].freeze
59
+
60
+ def key(value)
61
+ value.to_s.strip.upcase.gsub(/[\s_]+/, "-")
62
+ end
63
+
64
+ LOOKUP = KNOWN.keys.to_h { |id| [key(id), id] }.merge(ALIASES.to_h { |from, to| [key(from), to] }).freeze
65
+
66
+ # The licence of the page's text and figures, or nil.
67
+ def content(page, site)
68
+ resolve(setting(page, "license", site, "content_license"), page, site)
69
+ end
70
+
71
+ # The licence of the page's code samples, or nil.
72
+ def code(page, site)
73
+ resolve(setting(page, "code_license", site, "code_license"), page, site)
74
+ end
75
+
76
+ # The page's own value, else the site's; `false` declines the site's.
77
+ def setting(page, page_key, site, site_key)
78
+ value = Authors.value(page, page_key)
79
+ own_only = OWN_LICENSE.include?(Authors.value(page, "collection").to_s)
80
+ fallback = Authors.value(site, site_key) unless own_only
81
+ return fallback if value.nil? || value == true
82
+
83
+ if value.is_a?(Hash) && !Authors.present(value).keys.intersect?(%w[id name url])
84
+ defaults = fallback.is_a?(Hash) ? Authors.present(fallback) : { "name" => fallback }
85
+ return defaults.merge(Authors.present(value))
86
+ end
87
+
88
+ value
89
+ end
90
+
91
+ def resolve(value, page, site)
92
+ return if value.nil? || value == false || (value.is_a?(String) && value.strip.empty?)
93
+ unless value.is_a?(String) || value.is_a?(Hash)
94
+ raise Jekyll::Errors::FatalException, "#{Authors.value(page, 'path')} license must be a name, a map or false"
95
+ end
96
+
97
+ given = value.is_a?(Hash) ? Authors.present(value) : { "name" => value.to_s.strip }
98
+ return unless given.keys.intersect?(%w[id name url])
99
+
100
+ id = LOOKUP[key(given["id"] || given["name"])]
101
+ name, url = KNOWN[id] if id
102
+ # A name that is not itself an identifier is the label the page chose.
103
+ name = given["name"] if given["name"] && LOOKUP[key(given["name"])].nil?
104
+ licence = { "id" => id, "name" => name, "url" => given["url"] || url, "reserved" => id == "all-rights-reserved" }
105
+ licence.merge(holder(given, page, site)).compact
106
+ end
107
+
108
+ # The copyright holder and year: the page's, else the site default's,
109
+ # else the page's authors and its date.
110
+ def holder(given, page, site)
111
+ site_given = Authors.value(site, "content_license")
112
+ site_given = site_given.is_a?(Hash) ? Authors.present(site_given) : {}
113
+ names = Array(given["holder"] || site_given["holder"]).map(&:to_s).reject(&:empty?)
114
+ authors = Authors.authors(page, site).map { |author| author["name"] }
115
+ people = names.empty? || (names - authors).empty?
116
+ names = authors if names.empty?
117
+ year = year(given["year"] || site_given["year"] || Authors.value(page, "date"))
118
+ { "holders" => names, "holder" => (names.join(", ") unless names.empty?), "people" => people, "year" => year }
119
+ end
120
+
121
+ # A year, a date or a string naming either.
122
+ def year(value)
123
+ return value if value.is_a?(Integer)
124
+ return value.year if value.respond_to?(:year)
125
+
126
+ text = value.to_s.strip
127
+ return text.to_i if text.match?(/\A\d{4}\z/)
128
+ return text if text.match?(/\A\d{4}[-–]\d{4}\z/)
129
+
130
+ Time.parse(text).year unless text.empty?
131
+ rescue ArgumentError
132
+ nil
133
+ end
134
+ end
135
+
136
+ module LicenseFilters
137
+ def page_license(page)
138
+ Licenses.content(page, @context["site"])
139
+ end
140
+
141
+ def page_code_license(page)
142
+ Licenses.code(page, @context["site"])
143
+ end
144
+ end
145
+ end
146
+
147
+ Liquid::Template.register_filter(Datalog::LicenseFilters)
@@ -26,6 +26,16 @@ module MathPreprocessor
26
26
  regex: /(?<![\\$])(?<open>\$)(?![\s$])(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<![\s\\])(?<close>\$)(?![$\d])/m,
27
27
  tag: "span"
28
28
  },
29
+ # MathJax and KaTeX also render math with spaces inside the dollars, such
30
+ # as `$ \frac{a}{b} $`, which the rule above leaves out, so a page whose
31
+ # only math was written that way loaded no engine. Such a pair counts when
32
+ # its body holds a TeX command, a superscript or a subscript, which prices
33
+ # like `$ 5 or $ 10` do not.
34
+ {
35
+ regex: /(?<![\\$])(?<open>\$)(?!\$)(?<body>(?:[^$\\\n]|\\.|\n(?![ \t]*\n))+?)(?<!\\)(?<close>\$)(?![$\d])/m,
36
+ tag: "span",
37
+ requires: /\\[a-zA-Z]+|[\^_]/
38
+ },
29
39
  {
30
40
  regex: /(?<open>\\\()(?<body>.+?)(?<close>\\\))/m,
31
41
  tag: "span"
@@ -57,22 +67,43 @@ module MathPreprocessor
57
67
  def process
58
68
  return @content unless @content&.match?(/\$|\\\(|\\\[|\\begin\{/)
59
69
 
60
- code = []
70
+ @segments = []
61
71
  processed = CODE_PATTERNS.reduce(@content.dup) do |text, pattern|
62
- text.gsub(pattern) do |match|
63
- code << match
64
- "\x00#{code.size - 1}\x00"
65
- end
72
+ text.gsub(pattern) { |match| mask(match) }
66
73
  end
74
+ processed = normalize_inline_dollars(processed)
67
75
  processed = apply_patterns(processed, DISPLAY_PATTERNS, display: true)
68
76
  processed = apply_patterns(processed, INLINE_PATTERNS, display: false)
69
77
  # A segment set aside can contain the placeholder of an earlier one.
70
- processed = processed.gsub(PLACEHOLDER) { code[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
78
+ processed = processed.gsub(PLACEHOLDER) { @segments[Regexp.last_match(1).to_i] } while processed.match?(PLACEHOLDER)
71
79
  processed
72
80
  end
73
81
 
74
82
  private
75
83
 
84
+ def mask(segment)
85
+ @segments << segment
86
+ "\x00#{@segments.size - 1}\x00"
87
+ end
88
+
89
+ # Kramdown accepts $$...$$ inside prose as inline math. A div there is
90
+ # escaped by Markdown and leaks its attributes into the visible article.
91
+ # Normalize before wrapping; code is already masked and standalone or
92
+ # multiline display equations retain their original delimiters.
93
+ def normalize_inline_dollars(text)
94
+ text.gsub(DISPLAY_PATTERNS.first[:regex]) do |expression|
95
+ match = Regexp.last_match
96
+ body = match[:body]
97
+ next expression if body.include?("\n") || body.strip.empty?
98
+
99
+ before = match.pre_match.split("\n", -1).last.to_s
100
+ after = match.post_match.split("\n", 2).first.to_s
101
+ next expression unless before.match?(/\S/) || after.match?(/\S/)
102
+
103
+ "$#{body.strip}$"
104
+ end
105
+ end
106
+
76
107
  def apply_patterns(text, patterns, display: false)
77
108
  patterns.reduce(text) do |result, pattern|
78
109
  result.gsub(pattern[:regex]) do |match|
@@ -81,8 +112,11 @@ module MathPreprocessor
81
112
  close = Regexp.last_match[:close]
82
113
 
83
114
  next match if body.nil? || body.strip.empty?
115
+ next match if pattern[:requires] && !pattern[:requires].match?(body)
84
116
 
85
- wrapper_for(match, body, open, close, pattern[:tag], display: display)
117
+ # Each wrapper is set aside like code. Otherwise a later pattern could
118
+ # pair a dollar sign inside it with one in the text that follows.
119
+ mask(wrapper_for(match, body, open, close, pattern[:tag], display: display))
86
120
  end
87
121
  end
88
122
  end
@@ -103,6 +137,9 @@ module MathPreprocessor
103
137
  "aria-label" => alt_text,
104
138
  "tabindex" => "0"
105
139
  }
140
+ # Kramdown must not interpret TeX's escaped delimiters or underscores as
141
+ # Markdown inside an inline HTML span.
142
+ attributes["markdown"] = "0" unless display
106
143
 
107
144
  attribute_string = attributes.map do |key, value|
108
145
  next if value.nil? || value.strip.empty?
@@ -110,7 +147,7 @@ module MathPreprocessor
110
147
  %(#{key}="#{CGI.escapeHTML(value)}")
111
148
  end.compact.join(" ")
112
149
 
113
- inner = "#{open}#{latex}#{close}"
150
+ inner = CGI.escapeHTML("#{open}#{latex}#{close}")
114
151
  "<#{tag} #{attribute_string}>#{inner}</#{tag}>"
115
152
  end
116
153
 
@@ -195,9 +232,9 @@ module MathPreprocessor
195
232
  def apply(document)
196
233
  return unless document.respond_to?(:content)
197
234
  return unless document.respond_to?(:output_ext) && document.output_ext == ".html"
198
- # A page that opts out of math rendering (`math: false` or `mathjax: false`)
199
- # keeps its dollar signs and TeX-looking text verbatim.
200
- return if document.respond_to?(:data) && (document.data["math"] == false || document.data["mathjax"] == false)
235
+ # A page that opts out of math rendering keeps its dollar signs and
236
+ # TeX-looking text verbatim.
237
+ return if document.respond_to?(:data) && math_setting(document.data) == false
201
238
 
202
239
  content = document.content
203
240
  return unless content&.match?(/\$|\\\(|\\\[|\\begin\{/)
@@ -207,6 +244,13 @@ module MathPreprocessor
207
244
  document.content = updated_content
208
245
  document.data["math_expressions"] = processor.expressions if processor.expressions.any?
209
246
  end
247
+
248
+ # `math`, or its alias `mathjax` when `math` is unset, as
249
+ # _includes/meta/math-config.html reads them. A `mathjax: true` in front
250
+ # matter defaults used to win over a page's `math: false`.
251
+ def math_setting(data)
252
+ data["math"].nil? ? data["mathjax"] : data["math"]
253
+ end
210
254
  end
211
255
 
212
256
  # Posts are documents, so registering them separately ran the preprocessor
@@ -0,0 +1,260 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi"
4
+ require "strscan"
5
+
6
+ module Datalog
7
+ # Numbered figures and tables, and references to them, within a page:
8
+ #
9
+ # {% figure id="fig-power" src="/assets/img/power.png" alt="Power curve" %}
10
+ # Power as a function of effect size $\delta$.
11
+ # {% endfigure %}
12
+ #
13
+ # {% table id="tab-runs" %}
14
+ # Simulation runs by sample size.
15
+ #
16
+ # | n | runs |
17
+ # |---|------|
18
+ # {% endtable %}
19
+ #
20
+ # As {% ref fig-power %} shows, ...
21
+ #
22
+ # The tags write placeholders. Once the page's Markdown is converted, every
23
+ # target gets its number in the order it appears, so a reference may come
24
+ # before its figure, and each reference becomes a link reading "Figure 2".
25
+ # A duplicate id or a reference to nothing stops the build. Equations are
26
+ # numbered and referenced by MathJax (\label and \eqref) instead.
27
+ #
28
+ # A caption is the body of the tag, not an attribute, so math in it goes
29
+ # through the math preprocessor like the rest of the page.
30
+ #
31
+ # Theorems, definitions and the other statements in _plugins/statements.rb
32
+ # are numbered and referred to the same way.
33
+ module References
34
+ module_function
35
+
36
+ # Each kind counts separately; the label comes from _data/i18n.
37
+ KINDS = {
38
+ "figure" => "Figure", "table" => "Table",
39
+ "theorem" => "Theorem", "lemma" => "Lemma", "proposition" => "Proposition", "corollary" => "Corollary",
40
+ "definition" => "Definition", "assumption" => "Assumption", "example" => "Example", "remark" => "Remark"
41
+ }.freeze
42
+ ID = /\A[A-Za-z][\w.:-]*\z/
43
+ # A label given in place of the number, as in "Theorem A".
44
+ CUSTOM_LABEL = /\A[[:alnum:]][[:alnum:].'*-]*\z/
45
+ TARGET = /data-ref-target="([^"]+)"[ ]data-ref-kind="([a-z]+)"(?:[ ]data-ref-number="([^"]+)")?
46
+ ([ ]data-ref-numbered="true")?/x
47
+ # A label ends with a full stop unless the tag chose another ending.
48
+ LABEL = %r{<span class="datalog-ref-label" data-ref-for="([^"]+)"(?: data-ref-end="([^"]*)")?></span>}
49
+ LINK = %r{<a class="datalog-ref" href="#([^"]+)" data-ref="\1">[^<]*</a>}
50
+ SOURCE_TAG = /\{%-?\s*(#{KINDS.keys.join('|')})\b([^%]*)-?%\}/
51
+
52
+ def number(document)
53
+ content = document.content
54
+ return unless content&.include?("data-ref")
55
+
56
+ # Feeds and listings embed content that has already been numbered on its
57
+ # own page. Only process this document's remaining placeholders.
58
+ targets = targets(content.scan(TARGET).reject { |entry| entry[3] }, document)
59
+ content = content.gsub(TARGET) do |target|
60
+ Regexp.last_match(4) ? target : %(#{target} data-ref-numbered="true")
61
+ end
62
+ content = content.gsub(LABEL) do
63
+ id, ending = Regexp.last_match.captures
64
+ text = "#{CGI.escapeHTML(targets.fetch(id))}#{ending || '.'}"
65
+ %(<span class="datalog-ref-label" data-ref-for="#{id}">#{text}</span>)
66
+ end
67
+ document.content = link_references(content, targets, document)
68
+ end
69
+
70
+ # Jekyll runs no hooks for an excerpt, and renders it when a template first
71
+ # asks for it, which on a listing page can be before its post is converted.
72
+ # Its references take their numbers from the post: from the converted
73
+ # content when there is one, and otherwise from the tags in the source,
74
+ # which are in the same order. A missing target is the post's error to report.
75
+ def number_excerpt(excerpt, html)
76
+ return html unless html&.include?("data-ref")
77
+
78
+ post = excerpt.doc
79
+ found = post.content.to_s.scan(TARGET)
80
+ found = source_targets(post.content.to_s) if found.empty?
81
+ targets = targets(found.uniq(&:first), post)
82
+ html.gsub(LINK) do
83
+ id = Regexp.last_match(1)
84
+ text = targets[id] ? CGI.escapeHTML(targets[id]) : id
85
+ %(<a class="datalog-ref" href="#{post.site.baseurl}#{post.url}##{id}" data-ref="#{id}">#{text}</a>)
86
+ end
87
+ end
88
+
89
+ # [id, kind, label] for each numbered tag in a page's source.
90
+ def source_targets(source)
91
+ source.scan(SOURCE_TAG).filter_map do |kind, markup|
92
+ attributes = markup.scan(/(\w+)=["']([^"']*)["']/).to_h
93
+ [attributes["id"], kind, attributes["label"]] if attributes["id"]
94
+ end
95
+ end
96
+
97
+ # The id => "Figure 2" of every [id, kind, label] target, numbered per kind in
98
+ # page order. A target with its own label does not take a number.
99
+ def targets(found, document)
100
+ counts = Hash.new(0)
101
+ found.each_with_object({}) do |(id, kind, custom), targets|
102
+ if targets.key?(id)
103
+ raise Jekyll::Errors::FatalException,
104
+ "#{document.relative_path} has two numbered figures, tables or statements with the id " \
105
+ "\"#{id}\"; each id has to be unique"
106
+ end
107
+
108
+ number = custom ? CGI.unescapeHTML(custom) : (counts[kind] += 1)
109
+ targets[id] = "#{label(document, kind)} #{number}"
110
+ end
111
+ end
112
+
113
+ def link_references(content, targets, document)
114
+ missing = content.scan(LINK).flatten.uniq - targets.keys
115
+ unless missing.empty?
116
+ raise Jekyll::Errors::FatalException,
117
+ "#{document.relative_path} refers to #{missing.map { |id| "\"#{id}\"" }.join(', ')}, which no " \
118
+ "numbered figure, table or statement on the page has as its id"
119
+ end
120
+
121
+ content.gsub(LINK) do
122
+ id = Regexp.last_match(1)
123
+ label = CGI.escapeHTML(targets[id])
124
+ %(<a class="datalog-ref" href="##{id}" data-ref="#{id}" data-ref-numbered="true">#{label}</a>)
125
+ end
126
+ end
127
+
128
+ def label(document, kind)
129
+ site = document.site
130
+ locale = I18n.locale_code(site, document.data["lang"])
131
+ I18n.lookup(site, locale, "references.#{kind}") || KINDS.fetch(kind)
132
+ end
133
+
134
+ def validate_label(label, tag)
135
+ return if label.nil? || label.match?(CUSTOM_LABEL)
136
+
137
+ raise Liquid::ArgumentError,
138
+ "{% #{tag} %} takes a label of letters, digits, \".\", \"'\", \"*\" or \"-\", such as label=\"A\"; " \
139
+ "got #{label.inspect}"
140
+ end
141
+
142
+ def validate_id(id, tag)
143
+ return id if id.to_s.match?(ID)
144
+
145
+ raise Liquid::ArgumentError,
146
+ "{% #{tag} %} needs an id that starts with a letter and holds only letters, digits, " \
147
+ "\"-\", \"_\", \".\" or \":\", such as id=\"fig-power\"; got #{id.inspect}"
148
+ end
149
+
150
+ # key="value", key='value' or key=variable.
151
+ def attributes(markup, context, tag = "reference")
152
+ scanner = StringScanner.new(markup)
153
+ result = {}
154
+ until scanner.eos?
155
+ scanner.skip(/\s+/)
156
+ break if scanner.eos?
157
+
158
+ unless scanner.scan(/(\w+)\s*=\s*(?:"((?:[^"\\]|\\.)*)"|'((?:[^'\\]|\\.)*)'|([\w.\[\]-]+))(?=\s|\z)/m)
159
+ page = context.registers[:page] || {}
160
+ raise Liquid::ArgumentError,
161
+ "#{page['path'] || page['url']} {% #{tag} %} has invalid attributes near #{scanner.rest.inspect}"
162
+ end
163
+ # Ruby 3.2's StringScanner#captures returns "" for unmatched groups.
164
+ # values_at preserves nil so quoted literals stay distinct from variables.
165
+ key, double, single, variable = scanner.values_at(1, 2, 3, 4)
166
+ result[key] = variable ? context[variable].to_s : (double || single).gsub(/\\([\\"'])/, '\\1')
167
+ end
168
+ result
169
+ end
170
+
171
+ def markdown(context, text)
172
+ site = context.registers[:site]
173
+ site.find_converter_instance(Jekyll::Converters::Markdown).convert(text.to_s.strip)
174
+ end
175
+
176
+ # A caption of one paragraph loses its <p>, which a <figcaption> or <caption> does not need.
177
+ def inline(html)
178
+ html = html.strip
179
+ paragraph = html.match(%r{\A<p>(.*)</p>\z}m)
180
+ paragraph && !paragraph[1].include?("<p>") ? paragraph[1] : html
181
+ end
182
+ end
183
+
184
+ class FigureTag < Liquid::Block
185
+ def initialize(tag_name, markup, options)
186
+ super
187
+ @markup = markup
188
+ end
189
+
190
+ def render(context)
191
+ attributes = References.attributes(@markup, context, "figure")
192
+ id = References.validate_id(attributes["id"], "figure")
193
+ src = attributes["src"].to_s
194
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs a src, the image it shows" if src.empty?
195
+ if attributes["alt"].to_s.strip.empty?
196
+ raise Liquid::ArgumentError, "{% figure id=\"#{id}\" %} needs alt text for its image"
197
+ end
198
+
199
+ src = "#{context.registers[:site].config['baseurl'].to_s.chomp('/')}#{src}" if src.start_with?("/")
200
+ caption = References.inline(References.markdown(context, super))
201
+ classes = ["datalog-figure", attributes["class"]].compact.join(" ")
202
+
203
+ alt = CGI.escapeHTML(attributes["alt"])
204
+ %(<figure class="#{CGI.escapeHTML(classes)}" id="#{id}" data-ref-target="#{id}" data-ref-kind="figure">) +
205
+ %(<img src="#{CGI.escapeHTML(src)}" alt="#{alt}" loading="lazy" decoding="async">) +
206
+ %(<figcaption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</figcaption></figure>\n)
207
+ end
208
+ end
209
+
210
+ class TableTag < Liquid::Block
211
+ def initialize(tag_name, markup, options)
212
+ super
213
+ @markup = markup
214
+ end
215
+
216
+ # The body holds the caption and then a Markdown table.
217
+ def render(context)
218
+ id = References.validate_id(References.attributes(@markup, context, "table")["id"], "table")
219
+ html = References.markdown(context, super)
220
+ tables = html.scan(/<table\b/).size
221
+ unless tables == 1
222
+ raise Liquid::ArgumentError, "{% table id=\"#{id}\" %} holds #{tables} tables; it needs one, after its caption"
223
+ end
224
+
225
+ caption, table = html.split(/(?=<table\b)/, 2)
226
+ caption = References.inline(caption)
227
+ numbered = table.sub(/<table\b([^>]*)>/) do
228
+ %(<table#{Regexp.last_match(1)} id="#{id}" data-ref-target="#{id}" data-ref-kind="table">) +
229
+ %(<caption><span class="datalog-ref-label" data-ref-for="#{id}"></span> #{caption}</caption>)
230
+ end
231
+ "#{numbered.strip}\n"
232
+ end
233
+ end
234
+
235
+ class ReferenceTag < Liquid::Tag
236
+ def initialize(tag_name, markup, options)
237
+ super
238
+ @id = References.validate_id(markup.strip, "ref")
239
+ end
240
+
241
+ def render(_context)
242
+ %(<a class="datalog-ref" href="##{@id}" data-ref="#{@id}">#{@id}</a>)
243
+ end
244
+ end
245
+ end
246
+
247
+ Liquid::Template.register_tag("figure", Datalog::FigureTag)
248
+ Liquid::Template.register_tag("table", Datalog::TableTag)
249
+ Liquid::Template.register_tag("ref", Datalog::ReferenceTag)
250
+
251
+ Jekyll::Hooks.register %i[pages documents], :post_convert do |document|
252
+ Datalog::References.number(document)
253
+ end
254
+
255
+ Jekyll::Excerpt.prepend(Module.new do
256
+ # Excerpt#output renders once and keeps the result; numbering it is a quick gsub.
257
+ def output
258
+ Datalog::References.number_excerpt(self, super)
259
+ end
260
+ end)