datalog-theme 0.10.1 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +127 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +4 -2
  5. data/_data/i18n/en.yml +55 -3
  6. data/_data/i18n/es.yml +55 -3
  7. data/_data/i18n/pt.yml +55 -3
  8. data/_includes/analytics/dashboard.html +24 -9
  9. data/_includes/components/academic-dashboard.html +168 -133
  10. data/_includes/components/author-bio.html +7 -5
  11. data/_includes/components/author-list.html +6 -5
  12. data/_includes/components/citation-tools.html +2 -1
  13. data/_includes/components/content-provenance.html +19 -1
  14. data/_includes/components/correction-report.html +15 -8
  15. data/_includes/components/hero.html +22 -15
  16. data/_includes/components/package-index.html +121 -0
  17. data/_includes/components/package-install.html +27 -50
  18. data/_includes/components/post-list.html +14 -0
  19. data/_includes/components/responsive-image.html +15 -21
  20. data/_includes/footer.html +1 -1
  21. data/_includes/head.html +70 -11
  22. data/_includes/layouts/default/article.html +76 -46
  23. data/_includes/meta/dataset-json.html +171 -0
  24. data/_includes/meta/math-config.html +15 -1
  25. data/_includes/meta/package-json.html +69 -0
  26. data/_includes/meta/person-json.html +36 -13
  27. data/_includes/meta/publisher.html +15 -0
  28. data/_includes/meta/schema.html +108 -23
  29. data/_includes/meta/scholarly.html +10 -5
  30. data/_includes/post/related-posts.html +17 -43
  31. data/_includes/scripts.html +2 -2
  32. data/_includes/search/index-data.json +10 -1
  33. data/_includes/search/page.html +11 -6
  34. data/_layouts/archive.html +67 -0
  35. data/_layouts/dataset.html +2 -1
  36. data/_layouts/default.html +13 -11
  37. data/_layouts/docs.html +53 -0
  38. data/_layouts/home.html +30 -1
  39. data/_layouts/notebook.html +5 -1
  40. data/_layouts/package.html +38 -21
  41. data/_layouts/portfolio.html +1 -0
  42. data/_layouts/post.html +21 -62
  43. data/_layouts/project.html +2 -1
  44. data/_layouts/research.html +14 -2
  45. data/_plugins/analytics_dashboard.rb +4 -1
  46. data/_plugins/archive.rb +114 -0
  47. data/_plugins/authors.rb +50 -1
  48. data/_plugins/citation_exports.rb +12 -1
  49. data/_plugins/config_validator.rb +77 -2
  50. data/_plugins/correction_fallback.rb +110 -0
  51. data/_plugins/image_optimizer.rb +129 -22
  52. data/_plugins/math_preprocessor.rb +7 -48
  53. data/_plugins/notebook_converter.rb +44 -5
  54. data/_plugins/packages.rb +84 -0
  55. data/_plugins/page_dates.rb +37 -0
  56. data/_plugins/references.rb +16 -2
  57. data/_plugins/related_posts.rb +149 -0
  58. data/_plugins/search_sections.rb +110 -0
  59. data/_plugins/site_identity.rb +42 -0
  60. data/_plugins/social_cards.rb +17 -0
  61. data/_sass/_academic-dashboard.scss +18 -30
  62. data/_sass/_base.scss +2 -0
  63. data/_sass/_components.scss +21 -23
  64. data/_sass/_docs.scss +143 -0
  65. data/_sass/_features.scss +6 -0
  66. data/_sass/_layout.scss +65 -6
  67. data/_sass/_mathematical.scss +61 -19
  68. data/_sass/_notebooks.scss +1 -32
  69. data/_sass/_package-docs.scss +6 -9
  70. data/_sass/_post-components.scss +23 -23
  71. data/_sass/_print.scss +1 -1
  72. data/_sass/_search-page.scss +75 -0
  73. data/_sass/_search.scss +17 -26
  74. data/_sass/_syntax-highlighting.scss +23 -1
  75. data/_sass/_theme.scss +1 -0
  76. data/_sass/_utilities.scss +87 -40
  77. data/_sass/_variables.scss +31 -29
  78. data/assets/css/main.scss +2 -1
  79. data/assets/js/dist/academic.js +1 -1
  80. data/assets/js/dist/analytics-dashboard.js +1 -1
  81. data/assets/js/dist/chunks/chunk-TNVD6UAM.js +1 -0
  82. data/assets/js/dist/comments.js +1 -1
  83. data/assets/js/dist/contact.js +1 -1
  84. data/assets/js/dist/core.js +1 -1
  85. data/assets/js/dist/corrections.js +5 -1
  86. data/assets/js/dist/loader.js +1 -1
  87. data/assets/js/dist/math.js +2 -1
  88. data/assets/js/dist/moderation.js +1 -1
  89. data/assets/js/dist/reactions.js +1 -1
  90. data/assets/js/dist/search.js +1 -1
  91. data/assets/js/dist/sources.json +20 -16
  92. data/assets/js/dist/subscriptions.js +1 -1
  93. data/assets/js/loader.js +6 -2
  94. data/lib/datalog/audit/checks.rb +269 -0
  95. data/lib/datalog/audit/known_keys.rb +61 -0
  96. data/lib/datalog/audit/site_reader.rb +118 -0
  97. data/lib/datalog/audit/source_file.rb +85 -0
  98. data/lib/datalog/audit.rb +145 -0
  99. data/lib/datalog/citations/bibtex.rb +200 -0
  100. data/lib/datalog/citations/entry.rb +333 -0
  101. data/lib/datalog/citations/markup.rb +79 -0
  102. data/lib/datalog/cli.rb +164 -89
  103. data/lib/datalog/critical_css.rb +126 -21
  104. data/lib/datalog/latex_speech/words.json +88 -0
  105. data/lib/datalog/latex_speech.rb +355 -0
  106. data/lib/datalog/packages/command.rb +51 -0
  107. data/lib/datalog/packages/refresh.rb +187 -0
  108. data/lib/datalog/packages.rb +177 -0
  109. data/lib/datalog/plugin_system.rb +5 -0
  110. data/lib/datalog/plugins/citations.rb +325 -109
  111. data/lib/datalog/site_config.rb +26 -0
  112. data/lib/datalog/social_cards/font.rb +286 -0
  113. data/lib/datalog/social_cards/fonts/IBMPlexSans-Regular.ttf +0 -0
  114. data/lib/datalog/social_cards/fonts/IBMPlexSerif-SemiBold.ttf +0 -0
  115. data/lib/datalog/social_cards/fonts/OFL.txt +92 -0
  116. data/lib/datalog/social_cards/geometry.rb +51 -0
  117. data/lib/datalog/social_cards/outline.rb +54 -0
  118. data/lib/datalog/social_cards/plain_text.rb +258 -0
  119. data/lib/datalog/social_cards/template.rb +322 -0
  120. data/lib/datalog/social_cards/template.svg +17 -0
  121. data/lib/datalog/social_cards/typesetter.rb +178 -0
  122. data/lib/datalog/social_cards.rb +369 -0
  123. data/lib/datalog/theme/updater.rb +307 -0
  124. data/lib/datalog/theme/version.rb +1 -1
  125. metadata +44 -7
  126. data/_includes/components/enhanced-code-block.html +0 -212
  127. data/_includes/components/performance-monitor.html +0 -170
  128. data/_includes/components/viz-table-fallback.html +0 -19
  129. data/_plugins/datalog_bibliography.rb +0 -21
  130. data/assets/js/dist/chunks/chunk-WIRUK3OZ.js +0 -1
@@ -0,0 +1,145 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require_relative "audit/source_file"
5
+ require_relative "audit/site_reader"
6
+ require_relative "audit/known_keys"
7
+ require_relative "audit/checks"
8
+
9
+ module Datalog
10
+ # `datalog audit`: which of a site's pages would gain from the theme's newer
11
+ # authoring features, and which have problems no build reports, each with
12
+ # its file and line. It reads the site and writes nothing.
13
+ class Audit
14
+ class Error < StandardError; end
15
+
16
+ Finding = Struct.new(:kind, :check, :file, :line, :message) do
17
+ def to_h
18
+ { "kind" => kind, "check" => check, "file" => file, "line" => line, "message" => message }
19
+ end
20
+ end
21
+
22
+ # Check id => [kind, heading]. Opportunities are advice; problems fail --strict.
23
+ CHECKS = {
24
+ "statements" => [:opportunity, "Statements typed by hand"],
25
+ "figures" => [:opportunity, "Figure and table numbers typed by hand"],
26
+ "series" => [:opportunity, "Series"],
27
+ "reproducibility" => [:opportunity, "Reproducibility"],
28
+ "revisions" => [:opportunity, "Revisions"],
29
+ "references" => [:opportunity, "References written by hand"],
30
+ "front-matter" => [:problem, "Front matter no one reads"],
31
+ "images" => [:problem, "Images without alt text"],
32
+ "links" => [:problem, "Links to pages the site does not build"],
33
+ "math" => [:problem, "Math switched against the content"]
34
+ }.freeze
35
+
36
+ THEME_ROOT = File.expand_path("../..", __dir__)
37
+
38
+ attr_reader :root, :findings, :files
39
+
40
+ def initialize(root:, only: nil, path: nil)
41
+ @root = File.expand_path(root)
42
+ @only = only ? Array(only).flat_map { |id| id.to_s.split(",") }.map(&:strip).reject(&:empty?) : CHECKS.keys
43
+ unknown = @only - CHECKS.keys
44
+ raise Error, "unknown check #{unknown.join(', ')}; the checks are #{CHECKS.keys.join(', ')}" unless unknown.empty?
45
+
46
+ @path = path&.delete_prefix("./")&.chomp("/")
47
+ raise Error, "#{@root} has no _config.yml: run the audit from a Jekyll site, or pass --root" unless
48
+ File.file?(File.join(@root, "_config.yml"))
49
+ end
50
+
51
+ def run
52
+ reader = SiteReader.new(root).read
53
+ settings = reader.config["audit"].is_a?(Hash) ? reader.config["audit"] : {}
54
+ keys = KnownKeys.build(theme_root: THEME_ROOT, site_root: root, extra: settings["known_keys"])
55
+ # Every file's Liquid, audited or not, may read another page's front matter.
56
+ sources = reader.content_files.map { |path, relative| SourceFile.new(path, relative) }
57
+ sources.each { |source| keys.merge(KnownKeys.from_content(source.body)) }
58
+ checks = Checks.new(reader: reader, known_keys: keys, settings: settings)
59
+ audited = sources.select { |source| within_path?(source.relative) }
60
+ @files = audited.map { |source| [source.path, source.relative] }
61
+ @findings = audited.flat_map do |file|
62
+ relative = file.relative
63
+ @only.flat_map do |check|
64
+ checks.public_send(check.tr("-", "_"), file).map do |line, message|
65
+ Finding.new(CHECKS[check].first, check, relative, line, message)
66
+ end
67
+ end
68
+ end
69
+ self
70
+ end
71
+
72
+ def problems
73
+ findings.select { |finding| finding.kind == :problem }
74
+ end
75
+
76
+ def opportunities
77
+ findings.select { |finding| finding.kind == :opportunity }
78
+ end
79
+
80
+ # --------------------------------------------------------------- reports
81
+
82
+ def to_json(*_args)
83
+ JSON.pretty_generate(
84
+ "root" => root, "files" => files.size,
85
+ "summary" => { "problems" => problems.size, "opportunities" => opportunities.size },
86
+ "problems" => problems.map(&:to_h), "opportunities" => opportunities.map(&:to_h)
87
+ )
88
+ end
89
+
90
+ def to_text
91
+ out = ["DataLog audit of #{root}: #{files.size} #{files.size == 1 ? 'file' : 'files'}", ""]
92
+ [[:problem, "Problems"], [:opportunity, "Opportunities"]].each do |kind, title|
93
+ group = findings.select { |finding| finding.kind == kind }
94
+ out << "#{title} (#{group.size})"
95
+ by_check(group).each do |check, items|
96
+ out << " #{CHECKS[check].last} (#{items.size})"
97
+ items.each { |item| out << " #{item.file}:#{item.line} #{item.message}" }
98
+ end
99
+ out << ""
100
+ end
101
+ out << summary
102
+ out.join("\n")
103
+ end
104
+
105
+ def to_markdown
106
+ out = ["## DataLog audit", "", "#{summary.chomp('.')} in #{files.size} files.", ""]
107
+ [[:problem, "Problems"], [:opportunity, "Opportunities"]].each do |kind, title|
108
+ group = findings.select { |finding| finding.kind == kind }
109
+ next if group.empty?
110
+
111
+ out << "### #{title}" << ""
112
+ by_check(group).each do |check, items|
113
+ out << "#### #{CHECKS[check].last}" << ""
114
+ items.each { |item| out << "- `#{item.file}:#{item.line}` #{markdown_text(item.message)}" }
115
+ out << ""
116
+ end
117
+ end
118
+ out.join("\n")
119
+ end
120
+
121
+ def summary
122
+ "#{problems.size} #{problems.size == 1 ? 'problem' : 'problems'}, #{opportunities.size} " \
123
+ "#{opportunities.size == 1 ? 'opportunity' : 'opportunities'}."
124
+ end
125
+
126
+ private
127
+
128
+ # A message as Markdown: Liquid tags as code, and the characters Markdown
129
+ # would read as emphasis, a link or a table cell escaped, so a quoted
130
+ # **Theorem 1.** reads as typed.
131
+ def markdown_text(message)
132
+ message.split(/(\{%.*?%\})/).each_with_index.map do |part, index|
133
+ index.odd? ? "`#{part}`" : part.gsub(/([*_\[\]<>|\\])/) { "\\#{Regexp.last_match(1)}" }
134
+ end.join
135
+ end
136
+
137
+ def by_check(group)
138
+ group.group_by(&:check).sort_by { |check, _| CHECKS.keys.index(check) }
139
+ end
140
+
141
+ def within_path?(relative)
142
+ @path.nil? || @path.empty? || relative == @path || relative.start_with?("#{@path}/")
143
+ end
144
+ end
145
+ end
@@ -0,0 +1,200 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "strscan"
4
+
5
+ module Datalog
6
+ module Citations
7
+ # A practical subset of BibTeX, read without a dependency.
8
+ #
9
+ # It reads every entry type (`@article`, `@book`, `@inproceedings`,
10
+ # `@incollection`, `@phdthesis`, `@techreport`, `@misc`, `@online` and the
11
+ # rest) and the fields `Entry.from_fields` knows; others are kept but not
12
+ # shown. Values may be braced, quoted, bare numbers or `@string` macros,
13
+ # joined with `#`; the month abbreviations (`jan` ... `dec`) are predefined.
14
+ # `@comment`, `@preamble` and text between entries are skipped. The common
15
+ # LaTeX accents and escapes become their characters, and the braces that
16
+ # protect capitals are dropped.
17
+ #
18
+ # A malformed entry raises ParseError with its line, so the build says where.
19
+ module BibTeX
20
+ class ParseError < StandardError; end
21
+
22
+ MONTHS = %w[jan feb mar apr may jun jul aug sep oct nov dec].each_with_index.to_h do |name, index|
23
+ [name, (index + 1).to_s]
24
+ end.freeze
25
+
26
+ ACCENTS = {
27
+ '"' => "\u0308", "'" => "\u0301", "`" => "\u0300", "^" => "\u0302", "~" => "\u0303",
28
+ "=" => "\u0304", "." => "\u0307", "u" => "\u0306", "v" => "\u030C", "H" => "\u030B",
29
+ "c" => "\u0327", "k" => "\u0328", "r" => "\u030A"
30
+ }.freeze
31
+ SYMBOLS = {
32
+ "ss" => "ß", "o" => "ø", "O" => "Ø", "aa" => "å", "AA" => "Å", "ae" => "æ", "AE" => "Æ",
33
+ "oe" => "œ", "OE" => "Œ", "l" => "ł", "L" => "Ł", "i" => "ı", "j" => "ȷ"
34
+ }.freeze
35
+
36
+ module_function
37
+
38
+ # [{ "type" => "article", "key" => "smith2020", "fields" => { "title" => "..." } }, ...]
39
+ # The values are as written, braces and LaTeX included: names need their
40
+ # braces to be split (Entry), and latex_to_text makes the rest plain text.
41
+ def parse(text)
42
+ scanner = StringScanner.new(text.to_s)
43
+ macros = MONTHS.dup
44
+ entries = []
45
+ until scanner.eos?
46
+ break unless scanner.skip_until(/@/)
47
+
48
+ type = scanner.scan(/[A-Za-z]+/).to_s.downcase
49
+ scanner.skip(/\s*/)
50
+ open = scanner.scan(/[{(]/)
51
+ raise ParseError, "line #{line(scanner)}: @#{type} is not followed by { or (" unless open
52
+
53
+ close = open == "{" ? "}" : ")"
54
+ case type
55
+ when "comment", "preamble", ""
56
+ skip_balanced(scanner, open, close)
57
+ when "string"
58
+ name, value = read_field(scanner, macros)
59
+ macros[name] = value if name
60
+ scanner.skip(/\s*[})]/)
61
+ else
62
+ entries << read_entry(scanner, type, close, macros)
63
+ end
64
+ end
65
+ entries
66
+ end
67
+
68
+ def read_entry(scanner, type, close, macros)
69
+ start = line(scanner)
70
+ scanner.skip(/\s*/)
71
+ key = scanner.scan(/[^,\s})]+/)
72
+ raise ParseError, "line #{start}: @#{type} has no citation key" unless key
73
+
74
+ fields = {}
75
+ loop do
76
+ scanner.skip(/\s*,?\s*/)
77
+ break if scanner.skip(Regexp.new("\\#{close}"))
78
+ raise ParseError, "line #{start}: @#{type}{#{key}, ...} is not closed" if scanner.eos?
79
+
80
+ name, value = read_field(scanner, macros)
81
+ raise ParseError, "line #{line(scanner)}: a field of #{key} has no name" unless name
82
+
83
+ fields[name] = value
84
+ end
85
+ { "type" => type, "key" => key, "fields" => fields }
86
+ end
87
+
88
+ def read_field(scanner, macros)
89
+ scanner.skip(/\s*/)
90
+ name = scanner.scan(/[A-Za-z][\w:.-]*/)
91
+ return [nil, nil] unless name
92
+
93
+ scanner.skip(/\s*=\s*/) || raise(ParseError, "line #{line(scanner)}: #{name} has no = and value")
94
+ parts = []
95
+ loop do
96
+ scanner.skip(/\s*/)
97
+ parts << read_value(scanner, macros)
98
+ break unless scanner.skip(/\s*#\s*/)
99
+ end
100
+ [name.downcase, parts.join.gsub(/\s+/, " ").strip]
101
+ end
102
+
103
+ def read_value(scanner, macros)
104
+ if scanner.skip(/\{/)
105
+ read_braced(scanner)
106
+ elsif scanner.skip(/"/)
107
+ read_quoted(scanner)
108
+ elsif (number = scanner.scan(/\d+/))
109
+ number
110
+ elsif (macro = scanner.scan(/[A-Za-z][\w:.-]*/))
111
+ macros.fetch(macro.downcase, macro)
112
+ else
113
+ raise ParseError, "line #{line(scanner)}: expected a value near #{scanner.peek(20).inspect}"
114
+ end
115
+ end
116
+
117
+ # The text up to the brace that closes the one already read, inner braces kept.
118
+ def read_braced(scanner)
119
+ depth = 1
120
+ value = +""
121
+ until depth.zero?
122
+ raise ParseError, "line #{line(scanner)}: a braced value is not closed" if scanner.eos?
123
+
124
+ char = scanner.getch
125
+ if char == "\\"
126
+ value << char << scanner.getch.to_s
127
+ next
128
+ end
129
+ depth += 1 if char == "{"
130
+ depth -= 1 if char == "}"
131
+ value << char unless depth.zero?
132
+ end
133
+ value
134
+ end
135
+
136
+ def read_quoted(scanner)
137
+ depth = 0
138
+ value = +""
139
+ loop do
140
+ raise ParseError, "line #{line(scanner)}: a quoted value is not closed" if scanner.eos?
141
+
142
+ char = scanner.getch
143
+ if char == "\\"
144
+ value << char << scanner.getch.to_s
145
+ next
146
+ end
147
+ break if char == '"' && depth.zero?
148
+
149
+ depth += 1 if char == "{"
150
+ depth -= 1 if char == "}"
151
+ value << char
152
+ end
153
+ value
154
+ end
155
+
156
+ def skip_balanced(scanner, open, close)
157
+ depth = 1
158
+ until depth.zero? || scanner.eos?
159
+ char = scanner.getch
160
+ depth += 1 if char == open
161
+ depth -= 1 if char == close
162
+ end
163
+ end
164
+
165
+ # LaTeX as BibTeX files write it, turned into the characters it stands for.
166
+ def latex_to_text(value)
167
+ text = accents(value.gsub(/\s+/, " ").strip)
168
+ text = text.gsub(/\{?\\(#{SYMBOLS.keys.sort_by(&:size).reverse.join('|')})(?![A-Za-z])\s*\}?/) do
169
+ SYMBOLS.fetch(Regexp.last_match(1))
170
+ end
171
+ # Escaped braces survive the removal of the braces that only protect case.
172
+ text = text.gsub("\\{", "@@LBRACE@@").gsub("\\}", "@@RBRACE@@")
173
+ .gsub(/\\([&%$#_])/, '\1')
174
+ .gsub("---", "\u2014").gsub("--", "\u2013")
175
+ .gsub(/(?<!\\)~/, "\u00A0")
176
+ .gsub(/\\(TeX|LaTeX|BibTeX)(?![A-Za-z])/, '\1')
177
+ .gsub(/\\[A-Za-z]+\s*/, "")
178
+ text.delete("{}").gsub("@@LBRACE@@", "{").gsub("@@RBRACE@@", "}").gsub(/\s+/, " ").strip
179
+ end
180
+
181
+ # An accent over one letter: \"{u}, {\"u}, \"u, \'{\i}. A letter command
182
+ # (\c, \v, \u, ...) counts only with a brace or a space after it, so \url
183
+ # is not \u followed by "rl".
184
+ SYMBOL_MARKS = %("'`^~=.)
185
+ LETTER_MARKS = "uvHckr"
186
+ ACCENT = /\{?\\(?:([#{Regexp.escape(SYMBOL_MARKS)}])\s*|([#{LETTER_MARKS}])(?:\s+|(?=\{)))\{?\\?([A-Za-z])\}?\}?/
187
+
188
+ def accents(text)
189
+ text.gsub(ACCENT) do
190
+ mark = Regexp.last_match(1) || Regexp.last_match(2)
191
+ (Regexp.last_match(3) + ACCENTS.fetch(mark)).unicode_normalize(:nfc)
192
+ end
193
+ end
194
+
195
+ def line(scanner)
196
+ scanner.string[0...scanner.pos].count("\n") + 1
197
+ end
198
+ end
199
+ end
200
+ end
@@ -0,0 +1,333 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "cgi"
4
+ require "strscan"
5
+ require_relative "bibtex"
6
+
7
+ module Datalog
8
+ module Citations
9
+ # A work a page can cite, in one shape whether it came from BibTeX, CSL-JSON
10
+ # or front matter. Every value is plain text; the HTML methods escape it.
11
+ class Entry
12
+ # A person, or an organisation written as one literal name.
13
+ Name = Struct.new(:family, :given, :literal) do
14
+ def family_name
15
+ literal || family.to_s
16
+ end
17
+
18
+ # "Müller, J." in the reference list.
19
+ def listed
20
+ return literal if literal
21
+ return family.to_s if given.to_s.empty?
22
+
23
+ "#{family}, #{initials}"
24
+ end
25
+
26
+ def initials
27
+ given.to_s.split(/[\s.]+/).reject(&:empty?).map do |part|
28
+ part.split("-").map { |piece| "#{piece[0]}." }.join("-")
29
+ end.join(" ")
30
+ end
31
+
32
+ # "Jörg Müller", as a name reads in metadata.
33
+ def full
34
+ literal || [given, family].compact.reject(&:empty?).join(" ")
35
+ end
36
+
37
+ # "Müller, Jörg", as Highwire's citation_author has it.
38
+ def inverted
39
+ literal || [family, given].compact.reject(&:empty?).join(", ")
40
+ end
41
+ end
42
+
43
+ FIELDS = %i[key type authors title container publisher year volume issue pages doi url note].freeze
44
+ attr_reader(*FIELDS)
45
+ attr_accessor :others
46
+
47
+ # Types whose own title is the published thing: set in italics, as in APA.
48
+ STANDALONE = %w[book thesis report webpage].freeze
49
+ DOI = %r{\A(?:https?://(?:dx\.)?doi\.org/|doi:)?(10\.\d{4,9}/\S+)\z}i
50
+ URL = %r{\Ahttps?://[^\s<>"'`]+\z}i
51
+
52
+ def initialize(**fields)
53
+ FIELDS.each { |field| instance_variable_set(:"@#{field}", fields[field]) }
54
+ @authors = Array(@authors)
55
+ @others = fields[:others] || false
56
+ @key = @key.to_s
57
+ end
58
+
59
+ # ---------------------------------------------------------------- sources
60
+
61
+ BIBTEX_TYPES = {
62
+ "article" => "article", "book" => "book", "booklet" => "book", "inproceedings" => "paper-conference",
63
+ "conference" => "paper-conference", "incollection" => "chapter", "inbook" => "chapter",
64
+ "phdthesis" => "thesis", "mastersthesis" => "thesis", "thesis" => "thesis", "techreport" => "report",
65
+ "report" => "report", "online" => "webpage", "electronic" => "webpage", "www" => "webpage"
66
+ }.freeze
67
+
68
+ # Where each field of an entry comes from, first match wins.
69
+ BIBTEX_FIELDS = {
70
+ title: %w[title], container: %w[journal journaltitle booktitle series],
71
+ publisher: %w[publisher school institution organization], year: %w[year date],
72
+ volume: %w[volume], issue: %w[number issue], pages: %w[pages], doi: %w[doi]
73
+ }.freeze
74
+
75
+ def self.from_bibtex(record)
76
+ fields = record["fields"]
77
+ values = BIBTEX_FIELDS.transform_values do |names|
78
+ value = fields.values_at(*names).compact.first
79
+ value && BibTeX.latex_to_text(value)
80
+ end
81
+ names, others = bibtex_names(fields["author"] || fields["editor"])
82
+ link = [fields["url"], fields["howpublished"]].compact.map { |value| BibTeX.latex_to_text(value) }.first
83
+ address = link if link&.match?(URL)
84
+ note = fields["note"] ? BibTeX.latex_to_text(fields["note"]) : (link unless address)
85
+ new(**values, key: record["key"], type: BIBTEX_TYPES.fetch(record["type"], "misc"), authors: names,
86
+ others: others, year: values[:year].to_s[/\d{4}/], url: address, note: note)
87
+ end
88
+
89
+ # "Last, First and First von Last and {Organisation} and others".
90
+ def self.bibtex_names(value)
91
+ return [[], false] if value.to_s.strip.empty?
92
+
93
+ parts = split_top_level(value, /\band\b/i).map(&:strip)
94
+ others = parts.last&.casecmp?("others") || false
95
+ parts.pop if others
96
+ [parts.map { |part| bibtex_name(part.strip) }, others]
97
+ end
98
+
99
+ def self.bibtex_name(raw)
100
+ return Name.new(nil, nil, BibTeX.latex_to_text(raw)) if raw.match?(/\A\{.*\}\z/m) && balanced?(raw[1..-2])
101
+
102
+ pieces = split_top_level(raw, /,/).map(&:strip)
103
+ if pieces.size >= 2
104
+ # "von Last, First" or "von Last, Jr, First"
105
+ family = pieces.first
106
+ family = "#{family}, #{pieces[1]}" if pieces.size > 2
107
+ return Name.new(BibTeX.latex_to_text(family), BibTeX.latex_to_text(pieces.last), nil)
108
+ end
109
+
110
+ words = split_top_level(raw, /\s+/)
111
+ return Name.new(BibTeX.latex_to_text(raw), nil, nil) if words.size == 1
112
+
113
+ # "First von Last": the family name is the last word with the lowercase
114
+ # particles before it.
115
+ family_start = words.size - 1
116
+ family_start -= 1 while family_start > 1 && words[family_start - 1].match?(/\A[[:lower:]]/)
117
+ given = words[0...family_start].join(" ")
118
+ Name.new(BibTeX.latex_to_text(words[family_start..].join(" ")), BibTeX.latex_to_text(given), nil)
119
+ end
120
+
121
+ # Splits at `separator` outside braces.
122
+ def self.split_top_level(value, separator)
123
+ parts = []
124
+ depth = 0
125
+ current = +""
126
+ scanner = StringScanner.new(value)
127
+ until scanner.eos?
128
+ if depth.zero? && scanner.scan(separator)
129
+ parts << current
130
+ current = +""
131
+ next
132
+ end
133
+ char = scanner.getch
134
+ depth += 1 if char == "{"
135
+ depth -= 1 if char == "}"
136
+ current << char
137
+ end
138
+ parts << current
139
+ parts.reject { |part| part.strip.empty? }
140
+ end
141
+
142
+ def self.balanced?(text)
143
+ depth = 0
144
+ text.each_char do |char|
145
+ depth += 1 if char == "{"
146
+ depth -= 1 if char == "}"
147
+ return false if depth.negative?
148
+ end
149
+ depth.zero?
150
+ end
151
+
152
+ CSL_TYPES = {
153
+ "article-journal" => "article", "article-magazine" => "article", "article-newspaper" => "article",
154
+ "article" => "article", "book" => "book", "chapter" => "chapter", "paper-conference" => "paper-conference",
155
+ "thesis" => "thesis", "report" => "report", "webpage" => "webpage", "post-weblog" => "webpage"
156
+ }.freeze
157
+
158
+ def self.from_csl(item)
159
+ names = Array(item["author"] || item["editor"]).map do |name|
160
+ next Name.new(nil, nil, name.to_s) unless name.is_a?(Hash)
161
+
162
+ name["literal"] ? Name.new(nil, nil, name["literal"].to_s) : Name.new(name["family"].to_s, name["given"], nil)
163
+ end
164
+ issued = item["issued"] || {}
165
+ year = Array(Array(issued["date-parts"]).first).first || issued["raw"] || issued["literal"]
166
+ url = item["URL"].to_s
167
+ new(key: item["id"], type: CSL_TYPES.fetch(item["type"].to_s, "misc"), authors: names,
168
+ title: item["title"], container: item["container-title"] || item["collection-title"],
169
+ publisher: item["publisher"], year: year.to_s[/\d{4}/], volume: item["volume"]&.to_s,
170
+ issue: item["issue"]&.to_s, pages: item["page"]&.to_s, doi: item["DOI"],
171
+ url: url.match?(URL) ? url : nil, note: item["note"])
172
+ end
173
+
174
+ FRONT_MATTER_FIELDS = {
175
+ title: %w[title], container: %w[journal booktitle conference container], publisher: %w[publisher],
176
+ volume: %w[volume], issue: %w[issue number], pages: %w[pages], doi: %w[doi], note: %w[note]
177
+ }.freeze
178
+
179
+ # The `citations:` entries a page lists: a map in the fields above, or a
180
+ # string, which is cited by its slug and listed as written.
181
+ def self.from_front_matter(value)
182
+ return new(key: Datalog::Citations.slug(value), type: "misc", title: value.to_s) unless value.is_a?(Hash)
183
+
184
+ data = value.transform_keys(&:to_s)
185
+ values = FRONT_MATTER_FIELDS.transform_values { |names| data.values_at(*names).compact.first&.to_s }
186
+ url = data["url"].to_s
187
+ new(**values, key: data["id"] || data["key"], type: data["type"] || "misc",
188
+ authors: front_matter_names(data["authors"] || data["author"]),
189
+ year: data["year"].to_s[/\d{4}/], url: url.match?(URL) ? url : nil)
190
+ end
191
+
192
+ # ["Doe, Jo", "Ann Roe"] or "Doe, Jo and Ann Roe".
193
+ def self.front_matter_names(authors)
194
+ authors = authors.split(/\band\b/).map(&:strip) if authors.is_a?(String)
195
+ Array(authors).map do |name|
196
+ next Name.new(nil, nil, name.to_s) unless name.is_a?(String)
197
+
198
+ family, given = name.split(",", 2).map(&:strip)
199
+ given ? Name.new(family, given, nil) : bibtex_name(name)
200
+ end
201
+ end
202
+
203
+ # ------------------------------------------------------------------ links
204
+
205
+ # https://doi.org/10.1234/x for a DOI, else the entry's http(s) address;
206
+ # nothing else (javascript:, data:, a relative path) is ever a link.
207
+ def link
208
+ doi_link || (url if url.to_s.match?(URL))
209
+ end
210
+
211
+ def doi_link
212
+ match = doi.to_s.strip.match(DOI)
213
+ "https://doi.org/#{match[1]}" if match
214
+ end
215
+
216
+ # --------------------------------------------------------------- in text
217
+
218
+ # "Smith", "Smith and Jones", "Smith et al.", or the title when there is
219
+ # no author.
220
+ def names_in_text(words)
221
+ family = authors.map(&:family_name)
222
+ return short_title if family.empty?
223
+ return "#{family.first} #{words.fetch(:et_al)}" if family.size > 2 || others
224
+ return "#{family.first} #{words.fetch(:and)} #{family.last}" if family.size == 2
225
+
226
+ family.first
227
+ end
228
+
229
+ def short_title
230
+ words = title.to_s.split
231
+ words.size > 4 ? "#{words.first(4).join(' ')}\u2026" : title.to_s
232
+ end
233
+
234
+ def year_or(no_date)
235
+ year.to_s.empty? ? no_date : year.to_s
236
+ end
237
+
238
+ # First author, year, title: the order of an author-year reference list.
239
+ def sort_key
240
+ [authors.first&.family_name.to_s.downcase, year.to_s, title.to_s.downcase]
241
+ end
242
+
243
+ # ----------------------------------------------------------------- HTML
244
+
245
+ # The entry as the reference list shows it, in the manner of APA:
246
+ # Müller, J., & Smith, A. (2020). Title. Journal, 12(2), 10-20. https://doi.org/...
247
+ def reference_html(words, suffix: "")
248
+ date = "(#{h(year_or(words.fetch(:no_date)))}#{h(suffix)})."
249
+ parts = []
250
+ if authors.empty?
251
+ parts << "#{title_html}." if title
252
+ parts << date
253
+ else
254
+ parts << "#{h(listed_authors(words))} #{date}"
255
+ parts << "#{title_html}." if title
256
+ end
257
+ parts << "#{container_html}." if container
258
+ parts << "#{h(publisher)}." if publisher && !publisher.empty?
259
+ parts << "#{h(note)}." if note && !note.empty?
260
+ address = link
261
+ parts << %(<a href="#{h(address)}" rel="noopener noreferrer">#{h(address)}</a>) if address
262
+ parts.join(" ").gsub(/([.?!])\./, '\1')
263
+ end
264
+
265
+ def listed_authors(words)
266
+ names = authors.first(20).map(&:listed)
267
+ names << words.fetch(:et_al) if others || authors.size > 20
268
+ return names.first.to_s if names.size == 1
269
+
270
+ "#{names[0..-2].join(', ')} #{words.fetch(:and)} #{names.last}"
271
+ end
272
+
273
+ def title_html
274
+ STANDALONE.include?(type) ? "<em>#{h(title)}</em>" : h(title)
275
+ end
276
+
277
+ def container_html
278
+ text = "<em>#{h(container)}</em>"
279
+ text += ", #{h(volume)}" if volume && !volume.empty?
280
+ text += "(#{h(issue)})" if issue && !issue.empty?
281
+ text += ", #{h(pages)}" if pages && !pages.empty?
282
+ text
283
+ end
284
+
285
+ # ------------------------------------------------------------- metadata
286
+
287
+ # One Highwire citation_reference: "citation_title=...; citation_author=...".
288
+ def highwire
289
+ pairs = [["citation_title", title]]
290
+ authors.each { |name| pairs << ["citation_author", name.inverted] }
291
+ pairs << ["citation_publication_date", year]
292
+ pairs << [type == "paper-conference" ? "citation_conference_title" : "citation_journal_title", container]
293
+ pairs << ["citation_publisher", publisher] if STANDALONE.include?(type)
294
+ pairs.push(["citation_volume", volume], ["citation_issue", issue])
295
+ first, last = pages.to_s.split(/[-\u2013\u2014]+/, 2).map(&:strip)
296
+ pairs.push(["citation_firstpage", first], ["citation_lastpage", last])
297
+ pairs << ["citation_doi", doi_link&.delete_prefix("https://doi.org/")]
298
+ pairs.reject { |_, value| value.to_s.strip.empty? }
299
+ .map { |name, value| "#{name}=#{value.to_s.tr(';', ',').strip}" }.join("; ")
300
+ end
301
+
302
+ # A schema.org CreativeWork for the JSON-LD `citation` list.
303
+ def json_ld
304
+ work = { "@type" => "CreativeWork", "name" => title.to_s }
305
+ people = authors.map do |name|
306
+ { "@type" => name.literal ? "Organization" : "Person", "name" => name.full }
307
+ end
308
+ work["author"] = people unless people.empty?
309
+ work["datePublished"] = year.to_s if year
310
+ work["isPartOf"] = container.to_s if container
311
+ if doi_link
312
+ work["sameAs"] = doi_link
313
+ elsif link
314
+ work["url"] = link
315
+ end
316
+ work
317
+ end
318
+
319
+ private
320
+
321
+ def h(value)
322
+ CGI.escapeHTML(value.to_s)
323
+ end
324
+ end
325
+
326
+ module_function
327
+
328
+ # "Smith and Jones (2020)" in a string entry's text becomes smith-and-jones-2020.
329
+ def slug(value)
330
+ value.to_s.downcase.gsub(/[^a-z0-9]+/, "-").gsub(/\A-|-\z/, "")
331
+ end
332
+ end
333
+ end