datalog-theme 0.10.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +127 -0
- data/CITATION.cff +2 -2
- data/README.md +4 -2
- data/_data/i18n/en.yml +55 -3
- data/_data/i18n/es.yml +55 -3
- data/_data/i18n/pt.yml +55 -3
- data/_includes/analytics/dashboard.html +24 -9
- data/_includes/components/academic-dashboard.html +168 -133
- data/_includes/components/author-bio.html +7 -5
- data/_includes/components/author-list.html +6 -5
- data/_includes/components/citation-tools.html +2 -1
- data/_includes/components/content-provenance.html +19 -1
- data/_includes/components/correction-report.html +15 -8
- data/_includes/components/hero.html +22 -15
- data/_includes/components/package-index.html +121 -0
- data/_includes/components/package-install.html +27 -50
- data/_includes/components/post-list.html +14 -0
- data/_includes/components/responsive-image.html +15 -21
- data/_includes/footer.html +1 -1
- data/_includes/head.html +70 -11
- data/_includes/layouts/default/article.html +76 -46
- data/_includes/meta/dataset-json.html +171 -0
- data/_includes/meta/math-config.html +15 -1
- data/_includes/meta/package-json.html +69 -0
- data/_includes/meta/person-json.html +36 -13
- data/_includes/meta/publisher.html +15 -0
- data/_includes/meta/schema.html +108 -23
- data/_includes/meta/scholarly.html +10 -5
- data/_includes/post/related-posts.html +17 -43
- data/_includes/scripts.html +2 -2
- data/_includes/search/index-data.json +10 -1
- data/_includes/search/page.html +11 -6
- data/_layouts/archive.html +67 -0
- data/_layouts/dataset.html +2 -1
- data/_layouts/default.html +13 -11
- data/_layouts/docs.html +53 -0
- data/_layouts/home.html +30 -1
- data/_layouts/notebook.html +5 -1
- data/_layouts/package.html +38 -21
- data/_layouts/portfolio.html +1 -0
- data/_layouts/post.html +21 -62
- data/_layouts/project.html +2 -1
- data/_layouts/research.html +14 -2
- data/_plugins/analytics_dashboard.rb +4 -1
- data/_plugins/archive.rb +114 -0
- data/_plugins/authors.rb +50 -1
- data/_plugins/citation_exports.rb +12 -1
- data/_plugins/config_validator.rb +77 -2
- data/_plugins/correction_fallback.rb +110 -0
- data/_plugins/image_optimizer.rb +129 -22
- data/_plugins/math_preprocessor.rb +7 -48
- data/_plugins/notebook_converter.rb +44 -5
- data/_plugins/packages.rb +84 -0
- data/_plugins/page_dates.rb +37 -0
- data/_plugins/references.rb +16 -2
- data/_plugins/related_posts.rb +149 -0
- data/_plugins/search_sections.rb +110 -0
- data/_plugins/site_identity.rb +42 -0
- data/_plugins/social_cards.rb +17 -0
- data/_sass/_academic-dashboard.scss +18 -30
- data/_sass/_base.scss +2 -0
- data/_sass/_components.scss +21 -23
- data/_sass/_docs.scss +143 -0
- data/_sass/_features.scss +6 -0
- data/_sass/_layout.scss +65 -6
- data/_sass/_mathematical.scss +61 -19
- data/_sass/_notebooks.scss +1 -32
- data/_sass/_package-docs.scss +6 -9
- data/_sass/_post-components.scss +23 -23
- data/_sass/_print.scss +1 -1
- data/_sass/_search-page.scss +75 -0
- data/_sass/_search.scss +17 -26
- data/_sass/_syntax-highlighting.scss +23 -1
- data/_sass/_theme.scss +1 -0
- data/_sass/_utilities.scss +87 -40
- data/_sass/_variables.scss +31 -29
- data/assets/css/main.scss +2 -1
- data/assets/js/dist/academic.js +1 -1
- data/assets/js/dist/analytics-dashboard.js +1 -1
- data/assets/js/dist/chunks/chunk-TNVD6UAM.js +1 -0
- data/assets/js/dist/comments.js +1 -1
- data/assets/js/dist/contact.js +1 -1
- data/assets/js/dist/core.js +1 -1
- data/assets/js/dist/corrections.js +5 -1
- data/assets/js/dist/loader.js +1 -1
- data/assets/js/dist/math.js +2 -1
- data/assets/js/dist/moderation.js +1 -1
- data/assets/js/dist/reactions.js +1 -1
- data/assets/js/dist/search.js +1 -1
- data/assets/js/dist/sources.json +20 -16
- data/assets/js/dist/subscriptions.js +1 -1
- data/assets/js/loader.js +6 -2
- data/lib/datalog/audit/checks.rb +269 -0
- data/lib/datalog/audit/known_keys.rb +61 -0
- data/lib/datalog/audit/site_reader.rb +118 -0
- data/lib/datalog/audit/source_file.rb +85 -0
- data/lib/datalog/audit.rb +145 -0
- data/lib/datalog/citations/bibtex.rb +200 -0
- data/lib/datalog/citations/entry.rb +333 -0
- data/lib/datalog/citations/markup.rb +79 -0
- data/lib/datalog/cli.rb +164 -89
- data/lib/datalog/critical_css.rb +126 -21
- data/lib/datalog/latex_speech/words.json +88 -0
- data/lib/datalog/latex_speech.rb +355 -0
- data/lib/datalog/packages/command.rb +51 -0
- data/lib/datalog/packages/refresh.rb +187 -0
- data/lib/datalog/packages.rb +177 -0
- data/lib/datalog/plugin_system.rb +5 -0
- data/lib/datalog/plugins/citations.rb +325 -109
- data/lib/datalog/site_config.rb +26 -0
- data/lib/datalog/social_cards/font.rb +286 -0
- data/lib/datalog/social_cards/fonts/IBMPlexSans-Regular.ttf +0 -0
- data/lib/datalog/social_cards/fonts/IBMPlexSerif-SemiBold.ttf +0 -0
- data/lib/datalog/social_cards/fonts/OFL.txt +92 -0
- data/lib/datalog/social_cards/geometry.rb +51 -0
- data/lib/datalog/social_cards/outline.rb +54 -0
- data/lib/datalog/social_cards/plain_text.rb +258 -0
- data/lib/datalog/social_cards/template.rb +322 -0
- data/lib/datalog/social_cards/template.svg +17 -0
- data/lib/datalog/social_cards/typesetter.rb +178 -0
- data/lib/datalog/social_cards.rb +369 -0
- data/lib/datalog/theme/updater.rb +307 -0
- data/lib/datalog/theme/version.rb +1 -1
- metadata +44 -7
- data/_includes/components/enhanced-code-block.html +0 -212
- data/_includes/components/performance-monitor.html +0 -170
- data/_includes/components/viz-table-fallback.html +0 -19
- data/_plugins/datalog_bibliography.rb +0 -21
- data/assets/js/dist/chunks/chunk-WIRUK3OZ.js +0 -1
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require_relative "audit/source_file"
|
|
5
|
+
require_relative "audit/site_reader"
|
|
6
|
+
require_relative "audit/known_keys"
|
|
7
|
+
require_relative "audit/checks"
|
|
8
|
+
|
|
9
|
+
module Datalog
|
|
10
|
+
# `datalog audit`: which of a site's pages would gain from the theme's newer
|
|
11
|
+
# authoring features, and which have problems no build reports, each with
|
|
12
|
+
# its file and line. It reads the site and writes nothing.
|
|
13
|
+
class Audit
|
|
14
|
+
class Error < StandardError; end
|
|
15
|
+
|
|
16
|
+
Finding = Struct.new(:kind, :check, :file, :line, :message) do
|
|
17
|
+
def to_h
|
|
18
|
+
{ "kind" => kind, "check" => check, "file" => file, "line" => line, "message" => message }
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Check id => [kind, heading]. Opportunities are advice; problems fail --strict.
|
|
23
|
+
CHECKS = {
|
|
24
|
+
"statements" => [:opportunity, "Statements typed by hand"],
|
|
25
|
+
"figures" => [:opportunity, "Figure and table numbers typed by hand"],
|
|
26
|
+
"series" => [:opportunity, "Series"],
|
|
27
|
+
"reproducibility" => [:opportunity, "Reproducibility"],
|
|
28
|
+
"revisions" => [:opportunity, "Revisions"],
|
|
29
|
+
"references" => [:opportunity, "References written by hand"],
|
|
30
|
+
"front-matter" => [:problem, "Front matter no one reads"],
|
|
31
|
+
"images" => [:problem, "Images without alt text"],
|
|
32
|
+
"links" => [:problem, "Links to pages the site does not build"],
|
|
33
|
+
"math" => [:problem, "Math switched against the content"]
|
|
34
|
+
}.freeze
|
|
35
|
+
|
|
36
|
+
THEME_ROOT = File.expand_path("../..", __dir__)
|
|
37
|
+
|
|
38
|
+
attr_reader :root, :findings, :files
|
|
39
|
+
|
|
40
|
+
def initialize(root:, only: nil, path: nil)
|
|
41
|
+
@root = File.expand_path(root)
|
|
42
|
+
@only = only ? Array(only).flat_map { |id| id.to_s.split(",") }.map(&:strip).reject(&:empty?) : CHECKS.keys
|
|
43
|
+
unknown = @only - CHECKS.keys
|
|
44
|
+
raise Error, "unknown check #{unknown.join(', ')}; the checks are #{CHECKS.keys.join(', ')}" unless unknown.empty?
|
|
45
|
+
|
|
46
|
+
@path = path&.delete_prefix("./")&.chomp("/")
|
|
47
|
+
raise Error, "#{@root} has no _config.yml: run the audit from a Jekyll site, or pass --root" unless
|
|
48
|
+
File.file?(File.join(@root, "_config.yml"))
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def run
|
|
52
|
+
reader = SiteReader.new(root).read
|
|
53
|
+
settings = reader.config["audit"].is_a?(Hash) ? reader.config["audit"] : {}
|
|
54
|
+
keys = KnownKeys.build(theme_root: THEME_ROOT, site_root: root, extra: settings["known_keys"])
|
|
55
|
+
# Every file's Liquid, audited or not, may read another page's front matter.
|
|
56
|
+
sources = reader.content_files.map { |path, relative| SourceFile.new(path, relative) }
|
|
57
|
+
sources.each { |source| keys.merge(KnownKeys.from_content(source.body)) }
|
|
58
|
+
checks = Checks.new(reader: reader, known_keys: keys, settings: settings)
|
|
59
|
+
audited = sources.select { |source| within_path?(source.relative) }
|
|
60
|
+
@files = audited.map { |source| [source.path, source.relative] }
|
|
61
|
+
@findings = audited.flat_map do |file|
|
|
62
|
+
relative = file.relative
|
|
63
|
+
@only.flat_map do |check|
|
|
64
|
+
checks.public_send(check.tr("-", "_"), file).map do |line, message|
|
|
65
|
+
Finding.new(CHECKS[check].first, check, relative, line, message)
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
self
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def problems
|
|
73
|
+
findings.select { |finding| finding.kind == :problem }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def opportunities
|
|
77
|
+
findings.select { |finding| finding.kind == :opportunity }
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# --------------------------------------------------------------- reports
|
|
81
|
+
|
|
82
|
+
def to_json(*_args)
|
|
83
|
+
JSON.pretty_generate(
|
|
84
|
+
"root" => root, "files" => files.size,
|
|
85
|
+
"summary" => { "problems" => problems.size, "opportunities" => opportunities.size },
|
|
86
|
+
"problems" => problems.map(&:to_h), "opportunities" => opportunities.map(&:to_h)
|
|
87
|
+
)
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def to_text
|
|
91
|
+
out = ["DataLog audit of #{root}: #{files.size} #{files.size == 1 ? 'file' : 'files'}", ""]
|
|
92
|
+
[[:problem, "Problems"], [:opportunity, "Opportunities"]].each do |kind, title|
|
|
93
|
+
group = findings.select { |finding| finding.kind == kind }
|
|
94
|
+
out << "#{title} (#{group.size})"
|
|
95
|
+
by_check(group).each do |check, items|
|
|
96
|
+
out << " #{CHECKS[check].last} (#{items.size})"
|
|
97
|
+
items.each { |item| out << " #{item.file}:#{item.line} #{item.message}" }
|
|
98
|
+
end
|
|
99
|
+
out << ""
|
|
100
|
+
end
|
|
101
|
+
out << summary
|
|
102
|
+
out.join("\n")
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def to_markdown
|
|
106
|
+
out = ["## DataLog audit", "", "#{summary.chomp('.')} in #{files.size} files.", ""]
|
|
107
|
+
[[:problem, "Problems"], [:opportunity, "Opportunities"]].each do |kind, title|
|
|
108
|
+
group = findings.select { |finding| finding.kind == kind }
|
|
109
|
+
next if group.empty?
|
|
110
|
+
|
|
111
|
+
out << "### #{title}" << ""
|
|
112
|
+
by_check(group).each do |check, items|
|
|
113
|
+
out << "#### #{CHECKS[check].last}" << ""
|
|
114
|
+
items.each { |item| out << "- `#{item.file}:#{item.line}` #{markdown_text(item.message)}" }
|
|
115
|
+
out << ""
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
out.join("\n")
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def summary
|
|
122
|
+
"#{problems.size} #{problems.size == 1 ? 'problem' : 'problems'}, #{opportunities.size} " \
|
|
123
|
+
"#{opportunities.size == 1 ? 'opportunity' : 'opportunities'}."
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
private
|
|
127
|
+
|
|
128
|
+
# A message as Markdown: Liquid tags as code, and the characters Markdown
|
|
129
|
+
# would read as emphasis, a link or a table cell escaped, so a quoted
|
|
130
|
+
# **Theorem 1.** reads as typed.
|
|
131
|
+
def markdown_text(message)
|
|
132
|
+
message.split(/(\{%.*?%\})/).each_with_index.map do |part, index|
|
|
133
|
+
index.odd? ? "`#{part}`" : part.gsub(/([*_\[\]<>|\\])/) { "\\#{Regexp.last_match(1)}" }
|
|
134
|
+
end.join
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def by_check(group)
|
|
138
|
+
group.group_by(&:check).sort_by { |check, _| CHECKS.keys.index(check) }
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def within_path?(relative)
|
|
142
|
+
@path.nil? || @path.empty? || relative == @path || relative.start_with?("#{@path}/")
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
end
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "strscan"
|
|
4
|
+
|
|
5
|
+
module Datalog
|
|
6
|
+
module Citations
|
|
7
|
+
# A practical subset of BibTeX, read without a dependency.
|
|
8
|
+
#
|
|
9
|
+
# It reads every entry type (`@article`, `@book`, `@inproceedings`,
|
|
10
|
+
# `@incollection`, `@phdthesis`, `@techreport`, `@misc`, `@online` and the
|
|
11
|
+
# rest) and the fields `Entry.from_fields` knows; others are kept but not
|
|
12
|
+
# shown. Values may be braced, quoted, bare numbers or `@string` macros,
|
|
13
|
+
# joined with `#`; the month abbreviations (`jan` ... `dec`) are predefined.
|
|
14
|
+
# `@comment`, `@preamble` and text between entries are skipped. The common
|
|
15
|
+
# LaTeX accents and escapes become their characters, and the braces that
|
|
16
|
+
# protect capitals are dropped.
|
|
17
|
+
#
|
|
18
|
+
# A malformed entry raises ParseError with its line, so the build says where.
|
|
19
|
+
module BibTeX
|
|
20
|
+
class ParseError < StandardError; end
|
|
21
|
+
|
|
22
|
+
MONTHS = %w[jan feb mar apr may jun jul aug sep oct nov dec].each_with_index.to_h do |name, index|
|
|
23
|
+
[name, (index + 1).to_s]
|
|
24
|
+
end.freeze
|
|
25
|
+
|
|
26
|
+
ACCENTS = {
|
|
27
|
+
'"' => "\u0308", "'" => "\u0301", "`" => "\u0300", "^" => "\u0302", "~" => "\u0303",
|
|
28
|
+
"=" => "\u0304", "." => "\u0307", "u" => "\u0306", "v" => "\u030C", "H" => "\u030B",
|
|
29
|
+
"c" => "\u0327", "k" => "\u0328", "r" => "\u030A"
|
|
30
|
+
}.freeze
|
|
31
|
+
SYMBOLS = {
|
|
32
|
+
"ss" => "ß", "o" => "ø", "O" => "Ø", "aa" => "å", "AA" => "Å", "ae" => "æ", "AE" => "Æ",
|
|
33
|
+
"oe" => "œ", "OE" => "Œ", "l" => "ł", "L" => "Ł", "i" => "ı", "j" => "ȷ"
|
|
34
|
+
}.freeze
|
|
35
|
+
|
|
36
|
+
module_function
|
|
37
|
+
|
|
38
|
+
# [{ "type" => "article", "key" => "smith2020", "fields" => { "title" => "..." } }, ...]
|
|
39
|
+
# The values are as written, braces and LaTeX included: names need their
|
|
40
|
+
# braces to be split (Entry), and latex_to_text makes the rest plain text.
|
|
41
|
+
def parse(text)
|
|
42
|
+
scanner = StringScanner.new(text.to_s)
|
|
43
|
+
macros = MONTHS.dup
|
|
44
|
+
entries = []
|
|
45
|
+
until scanner.eos?
|
|
46
|
+
break unless scanner.skip_until(/@/)
|
|
47
|
+
|
|
48
|
+
type = scanner.scan(/[A-Za-z]+/).to_s.downcase
|
|
49
|
+
scanner.skip(/\s*/)
|
|
50
|
+
open = scanner.scan(/[{(]/)
|
|
51
|
+
raise ParseError, "line #{line(scanner)}: @#{type} is not followed by { or (" unless open
|
|
52
|
+
|
|
53
|
+
close = open == "{" ? "}" : ")"
|
|
54
|
+
case type
|
|
55
|
+
when "comment", "preamble", ""
|
|
56
|
+
skip_balanced(scanner, open, close)
|
|
57
|
+
when "string"
|
|
58
|
+
name, value = read_field(scanner, macros)
|
|
59
|
+
macros[name] = value if name
|
|
60
|
+
scanner.skip(/\s*[})]/)
|
|
61
|
+
else
|
|
62
|
+
entries << read_entry(scanner, type, close, macros)
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
entries
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
def read_entry(scanner, type, close, macros)
|
|
69
|
+
start = line(scanner)
|
|
70
|
+
scanner.skip(/\s*/)
|
|
71
|
+
key = scanner.scan(/[^,\s})]+/)
|
|
72
|
+
raise ParseError, "line #{start}: @#{type} has no citation key" unless key
|
|
73
|
+
|
|
74
|
+
fields = {}
|
|
75
|
+
loop do
|
|
76
|
+
scanner.skip(/\s*,?\s*/)
|
|
77
|
+
break if scanner.skip(Regexp.new("\\#{close}"))
|
|
78
|
+
raise ParseError, "line #{start}: @#{type}{#{key}, ...} is not closed" if scanner.eos?
|
|
79
|
+
|
|
80
|
+
name, value = read_field(scanner, macros)
|
|
81
|
+
raise ParseError, "line #{line(scanner)}: a field of #{key} has no name" unless name
|
|
82
|
+
|
|
83
|
+
fields[name] = value
|
|
84
|
+
end
|
|
85
|
+
{ "type" => type, "key" => key, "fields" => fields }
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def read_field(scanner, macros)
|
|
89
|
+
scanner.skip(/\s*/)
|
|
90
|
+
name = scanner.scan(/[A-Za-z][\w:.-]*/)
|
|
91
|
+
return [nil, nil] unless name
|
|
92
|
+
|
|
93
|
+
scanner.skip(/\s*=\s*/) || raise(ParseError, "line #{line(scanner)}: #{name} has no = and value")
|
|
94
|
+
parts = []
|
|
95
|
+
loop do
|
|
96
|
+
scanner.skip(/\s*/)
|
|
97
|
+
parts << read_value(scanner, macros)
|
|
98
|
+
break unless scanner.skip(/\s*#\s*/)
|
|
99
|
+
end
|
|
100
|
+
[name.downcase, parts.join.gsub(/\s+/, " ").strip]
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def read_value(scanner, macros)
|
|
104
|
+
if scanner.skip(/\{/)
|
|
105
|
+
read_braced(scanner)
|
|
106
|
+
elsif scanner.skip(/"/)
|
|
107
|
+
read_quoted(scanner)
|
|
108
|
+
elsif (number = scanner.scan(/\d+/))
|
|
109
|
+
number
|
|
110
|
+
elsif (macro = scanner.scan(/[A-Za-z][\w:.-]*/))
|
|
111
|
+
macros.fetch(macro.downcase, macro)
|
|
112
|
+
else
|
|
113
|
+
raise ParseError, "line #{line(scanner)}: expected a value near #{scanner.peek(20).inspect}"
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# The text up to the brace that closes the one already read, inner braces kept.
|
|
118
|
+
def read_braced(scanner)
|
|
119
|
+
depth = 1
|
|
120
|
+
value = +""
|
|
121
|
+
until depth.zero?
|
|
122
|
+
raise ParseError, "line #{line(scanner)}: a braced value is not closed" if scanner.eos?
|
|
123
|
+
|
|
124
|
+
char = scanner.getch
|
|
125
|
+
if char == "\\"
|
|
126
|
+
value << char << scanner.getch.to_s
|
|
127
|
+
next
|
|
128
|
+
end
|
|
129
|
+
depth += 1 if char == "{"
|
|
130
|
+
depth -= 1 if char == "}"
|
|
131
|
+
value << char unless depth.zero?
|
|
132
|
+
end
|
|
133
|
+
value
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def read_quoted(scanner)
|
|
137
|
+
depth = 0
|
|
138
|
+
value = +""
|
|
139
|
+
loop do
|
|
140
|
+
raise ParseError, "line #{line(scanner)}: a quoted value is not closed" if scanner.eos?
|
|
141
|
+
|
|
142
|
+
char = scanner.getch
|
|
143
|
+
if char == "\\"
|
|
144
|
+
value << char << scanner.getch.to_s
|
|
145
|
+
next
|
|
146
|
+
end
|
|
147
|
+
break if char == '"' && depth.zero?
|
|
148
|
+
|
|
149
|
+
depth += 1 if char == "{"
|
|
150
|
+
depth -= 1 if char == "}"
|
|
151
|
+
value << char
|
|
152
|
+
end
|
|
153
|
+
value
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def skip_balanced(scanner, open, close)
|
|
157
|
+
depth = 1
|
|
158
|
+
until depth.zero? || scanner.eos?
|
|
159
|
+
char = scanner.getch
|
|
160
|
+
depth += 1 if char == open
|
|
161
|
+
depth -= 1 if char == close
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# LaTeX as BibTeX files write it, turned into the characters it stands for.
|
|
166
|
+
def latex_to_text(value)
|
|
167
|
+
text = accents(value.gsub(/\s+/, " ").strip)
|
|
168
|
+
text = text.gsub(/\{?\\(#{SYMBOLS.keys.sort_by(&:size).reverse.join('|')})(?![A-Za-z])\s*\}?/) do
|
|
169
|
+
SYMBOLS.fetch(Regexp.last_match(1))
|
|
170
|
+
end
|
|
171
|
+
# Escaped braces survive the removal of the braces that only protect case.
|
|
172
|
+
text = text.gsub("\\{", "@@LBRACE@@").gsub("\\}", "@@RBRACE@@")
|
|
173
|
+
.gsub(/\\([&%$#_])/, '\1')
|
|
174
|
+
.gsub("---", "\u2014").gsub("--", "\u2013")
|
|
175
|
+
.gsub(/(?<!\\)~/, "\u00A0")
|
|
176
|
+
.gsub(/\\(TeX|LaTeX|BibTeX)(?![A-Za-z])/, '\1')
|
|
177
|
+
.gsub(/\\[A-Za-z]+\s*/, "")
|
|
178
|
+
text.delete("{}").gsub("@@LBRACE@@", "{").gsub("@@RBRACE@@", "}").gsub(/\s+/, " ").strip
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# An accent over one letter: \"{u}, {\"u}, \"u, \'{\i}. A letter command
|
|
182
|
+
# (\c, \v, \u, ...) counts only with a brace or a space after it, so \url
|
|
183
|
+
# is not \u followed by "rl".
|
|
184
|
+
SYMBOL_MARKS = %("'`^~=.)
|
|
185
|
+
LETTER_MARKS = "uvHckr"
|
|
186
|
+
ACCENT = /\{?\\(?:([#{Regexp.escape(SYMBOL_MARKS)}])\s*|([#{LETTER_MARKS}])(?:\s+|(?=\{)))\{?\\?([A-Za-z])\}?\}?/
|
|
187
|
+
|
|
188
|
+
def accents(text)
|
|
189
|
+
text.gsub(ACCENT) do
|
|
190
|
+
mark = Regexp.last_match(1) || Regexp.last_match(2)
|
|
191
|
+
(Regexp.last_match(3) + ACCENTS.fetch(mark)).unicode_normalize(:nfc)
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
def line(scanner)
|
|
196
|
+
scanner.string[0...scanner.pos].count("\n") + 1
|
|
197
|
+
end
|
|
198
|
+
end
|
|
199
|
+
end
|
|
200
|
+
end
|
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "cgi"
|
|
4
|
+
require "strscan"
|
|
5
|
+
require_relative "bibtex"
|
|
6
|
+
|
|
7
|
+
module Datalog
|
|
8
|
+
module Citations
|
|
9
|
+
# A work a page can cite, in one shape whether it came from BibTeX, CSL-JSON
|
|
10
|
+
# or front matter. Every value is plain text; the HTML methods escape it.
|
|
11
|
+
class Entry
|
|
12
|
+
# A person, or an organisation written as one literal name.
|
|
13
|
+
Name = Struct.new(:family, :given, :literal) do
|
|
14
|
+
def family_name
|
|
15
|
+
literal || family.to_s
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# "Müller, J." in the reference list.
|
|
19
|
+
def listed
|
|
20
|
+
return literal if literal
|
|
21
|
+
return family.to_s if given.to_s.empty?
|
|
22
|
+
|
|
23
|
+
"#{family}, #{initials}"
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def initials
|
|
27
|
+
given.to_s.split(/[\s.]+/).reject(&:empty?).map do |part|
|
|
28
|
+
part.split("-").map { |piece| "#{piece[0]}." }.join("-")
|
|
29
|
+
end.join(" ")
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# "Jörg Müller", as a name reads in metadata.
|
|
33
|
+
def full
|
|
34
|
+
literal || [given, family].compact.reject(&:empty?).join(" ")
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# "Müller, Jörg", as Highwire's citation_author has it.
|
|
38
|
+
def inverted
|
|
39
|
+
literal || [family, given].compact.reject(&:empty?).join(", ")
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
FIELDS = %i[key type authors title container publisher year volume issue pages doi url note].freeze
|
|
44
|
+
attr_reader(*FIELDS)
|
|
45
|
+
attr_accessor :others
|
|
46
|
+
|
|
47
|
+
# Types whose own title is the published thing: set in italics, as in APA.
|
|
48
|
+
STANDALONE = %w[book thesis report webpage].freeze
|
|
49
|
+
DOI = %r{\A(?:https?://(?:dx\.)?doi\.org/|doi:)?(10\.\d{4,9}/\S+)\z}i
|
|
50
|
+
URL = %r{\Ahttps?://[^\s<>"'`]+\z}i
|
|
51
|
+
|
|
52
|
+
def initialize(**fields)
|
|
53
|
+
FIELDS.each { |field| instance_variable_set(:"@#{field}", fields[field]) }
|
|
54
|
+
@authors = Array(@authors)
|
|
55
|
+
@others = fields[:others] || false
|
|
56
|
+
@key = @key.to_s
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# ---------------------------------------------------------------- sources
|
|
60
|
+
|
|
61
|
+
BIBTEX_TYPES = {
|
|
62
|
+
"article" => "article", "book" => "book", "booklet" => "book", "inproceedings" => "paper-conference",
|
|
63
|
+
"conference" => "paper-conference", "incollection" => "chapter", "inbook" => "chapter",
|
|
64
|
+
"phdthesis" => "thesis", "mastersthesis" => "thesis", "thesis" => "thesis", "techreport" => "report",
|
|
65
|
+
"report" => "report", "online" => "webpage", "electronic" => "webpage", "www" => "webpage"
|
|
66
|
+
}.freeze
|
|
67
|
+
|
|
68
|
+
# Where each field of an entry comes from, first match wins.
|
|
69
|
+
BIBTEX_FIELDS = {
|
|
70
|
+
title: %w[title], container: %w[journal journaltitle booktitle series],
|
|
71
|
+
publisher: %w[publisher school institution organization], year: %w[year date],
|
|
72
|
+
volume: %w[volume], issue: %w[number issue], pages: %w[pages], doi: %w[doi]
|
|
73
|
+
}.freeze
|
|
74
|
+
|
|
75
|
+
def self.from_bibtex(record)
|
|
76
|
+
fields = record["fields"]
|
|
77
|
+
values = BIBTEX_FIELDS.transform_values do |names|
|
|
78
|
+
value = fields.values_at(*names).compact.first
|
|
79
|
+
value && BibTeX.latex_to_text(value)
|
|
80
|
+
end
|
|
81
|
+
names, others = bibtex_names(fields["author"] || fields["editor"])
|
|
82
|
+
link = [fields["url"], fields["howpublished"]].compact.map { |value| BibTeX.latex_to_text(value) }.first
|
|
83
|
+
address = link if link&.match?(URL)
|
|
84
|
+
note = fields["note"] ? BibTeX.latex_to_text(fields["note"]) : (link unless address)
|
|
85
|
+
new(**values, key: record["key"], type: BIBTEX_TYPES.fetch(record["type"], "misc"), authors: names,
|
|
86
|
+
others: others, year: values[:year].to_s[/\d{4}/], url: address, note: note)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# "Last, First and First von Last and {Organisation} and others".
|
|
90
|
+
def self.bibtex_names(value)
|
|
91
|
+
return [[], false] if value.to_s.strip.empty?
|
|
92
|
+
|
|
93
|
+
parts = split_top_level(value, /\band\b/i).map(&:strip)
|
|
94
|
+
others = parts.last&.casecmp?("others") || false
|
|
95
|
+
parts.pop if others
|
|
96
|
+
[parts.map { |part| bibtex_name(part.strip) }, others]
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def self.bibtex_name(raw)
|
|
100
|
+
return Name.new(nil, nil, BibTeX.latex_to_text(raw)) if raw.match?(/\A\{.*\}\z/m) && balanced?(raw[1..-2])
|
|
101
|
+
|
|
102
|
+
pieces = split_top_level(raw, /,/).map(&:strip)
|
|
103
|
+
if pieces.size >= 2
|
|
104
|
+
# "von Last, First" or "von Last, Jr, First"
|
|
105
|
+
family = pieces.first
|
|
106
|
+
family = "#{family}, #{pieces[1]}" if pieces.size > 2
|
|
107
|
+
return Name.new(BibTeX.latex_to_text(family), BibTeX.latex_to_text(pieces.last), nil)
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
words = split_top_level(raw, /\s+/)
|
|
111
|
+
return Name.new(BibTeX.latex_to_text(raw), nil, nil) if words.size == 1
|
|
112
|
+
|
|
113
|
+
# "First von Last": the family name is the last word with the lowercase
|
|
114
|
+
# particles before it.
|
|
115
|
+
family_start = words.size - 1
|
|
116
|
+
family_start -= 1 while family_start > 1 && words[family_start - 1].match?(/\A[[:lower:]]/)
|
|
117
|
+
given = words[0...family_start].join(" ")
|
|
118
|
+
Name.new(BibTeX.latex_to_text(words[family_start..].join(" ")), BibTeX.latex_to_text(given), nil)
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Splits at `separator` outside braces.
|
|
122
|
+
def self.split_top_level(value, separator)
|
|
123
|
+
parts = []
|
|
124
|
+
depth = 0
|
|
125
|
+
current = +""
|
|
126
|
+
scanner = StringScanner.new(value)
|
|
127
|
+
until scanner.eos?
|
|
128
|
+
if depth.zero? && scanner.scan(separator)
|
|
129
|
+
parts << current
|
|
130
|
+
current = +""
|
|
131
|
+
next
|
|
132
|
+
end
|
|
133
|
+
char = scanner.getch
|
|
134
|
+
depth += 1 if char == "{"
|
|
135
|
+
depth -= 1 if char == "}"
|
|
136
|
+
current << char
|
|
137
|
+
end
|
|
138
|
+
parts << current
|
|
139
|
+
parts.reject { |part| part.strip.empty? }
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def self.balanced?(text)
|
|
143
|
+
depth = 0
|
|
144
|
+
text.each_char do |char|
|
|
145
|
+
depth += 1 if char == "{"
|
|
146
|
+
depth -= 1 if char == "}"
|
|
147
|
+
return false if depth.negative?
|
|
148
|
+
end
|
|
149
|
+
depth.zero?
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
CSL_TYPES = {
|
|
153
|
+
"article-journal" => "article", "article-magazine" => "article", "article-newspaper" => "article",
|
|
154
|
+
"article" => "article", "book" => "book", "chapter" => "chapter", "paper-conference" => "paper-conference",
|
|
155
|
+
"thesis" => "thesis", "report" => "report", "webpage" => "webpage", "post-weblog" => "webpage"
|
|
156
|
+
}.freeze
|
|
157
|
+
|
|
158
|
+
def self.from_csl(item)
|
|
159
|
+
names = Array(item["author"] || item["editor"]).map do |name|
|
|
160
|
+
next Name.new(nil, nil, name.to_s) unless name.is_a?(Hash)
|
|
161
|
+
|
|
162
|
+
name["literal"] ? Name.new(nil, nil, name["literal"].to_s) : Name.new(name["family"].to_s, name["given"], nil)
|
|
163
|
+
end
|
|
164
|
+
issued = item["issued"] || {}
|
|
165
|
+
year = Array(Array(issued["date-parts"]).first).first || issued["raw"] || issued["literal"]
|
|
166
|
+
url = item["URL"].to_s
|
|
167
|
+
new(key: item["id"], type: CSL_TYPES.fetch(item["type"].to_s, "misc"), authors: names,
|
|
168
|
+
title: item["title"], container: item["container-title"] || item["collection-title"],
|
|
169
|
+
publisher: item["publisher"], year: year.to_s[/\d{4}/], volume: item["volume"]&.to_s,
|
|
170
|
+
issue: item["issue"]&.to_s, pages: item["page"]&.to_s, doi: item["DOI"],
|
|
171
|
+
url: url.match?(URL) ? url : nil, note: item["note"])
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
FRONT_MATTER_FIELDS = {
|
|
175
|
+
title: %w[title], container: %w[journal booktitle conference container], publisher: %w[publisher],
|
|
176
|
+
volume: %w[volume], issue: %w[issue number], pages: %w[pages], doi: %w[doi], note: %w[note]
|
|
177
|
+
}.freeze
|
|
178
|
+
|
|
179
|
+
# The `citations:` entries a page lists: a map in the fields above, or a
|
|
180
|
+
# string, which is cited by its slug and listed as written.
|
|
181
|
+
def self.from_front_matter(value)
|
|
182
|
+
return new(key: Datalog::Citations.slug(value), type: "misc", title: value.to_s) unless value.is_a?(Hash)
|
|
183
|
+
|
|
184
|
+
data = value.transform_keys(&:to_s)
|
|
185
|
+
values = FRONT_MATTER_FIELDS.transform_values { |names| data.values_at(*names).compact.first&.to_s }
|
|
186
|
+
url = data["url"].to_s
|
|
187
|
+
new(**values, key: data["id"] || data["key"], type: data["type"] || "misc",
|
|
188
|
+
authors: front_matter_names(data["authors"] || data["author"]),
|
|
189
|
+
year: data["year"].to_s[/\d{4}/], url: url.match?(URL) ? url : nil)
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
# ["Doe, Jo", "Ann Roe"] or "Doe, Jo and Ann Roe".
|
|
193
|
+
def self.front_matter_names(authors)
|
|
194
|
+
authors = authors.split(/\band\b/).map(&:strip) if authors.is_a?(String)
|
|
195
|
+
Array(authors).map do |name|
|
|
196
|
+
next Name.new(nil, nil, name.to_s) unless name.is_a?(String)
|
|
197
|
+
|
|
198
|
+
family, given = name.split(",", 2).map(&:strip)
|
|
199
|
+
given ? Name.new(family, given, nil) : bibtex_name(name)
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
# ------------------------------------------------------------------ links
|
|
204
|
+
|
|
205
|
+
# https://doi.org/10.1234/x for a DOI, else the entry's http(s) address;
|
|
206
|
+
# nothing else (javascript:, data:, a relative path) is ever a link.
|
|
207
|
+
def link
|
|
208
|
+
doi_link || (url if url.to_s.match?(URL))
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
def doi_link
|
|
212
|
+
match = doi.to_s.strip.match(DOI)
|
|
213
|
+
"https://doi.org/#{match[1]}" if match
|
|
214
|
+
end
|
|
215
|
+
|
|
216
|
+
# --------------------------------------------------------------- in text
|
|
217
|
+
|
|
218
|
+
# "Smith", "Smith and Jones", "Smith et al.", or the title when there is
|
|
219
|
+
# no author.
|
|
220
|
+
def names_in_text(words)
|
|
221
|
+
family = authors.map(&:family_name)
|
|
222
|
+
return short_title if family.empty?
|
|
223
|
+
return "#{family.first} #{words.fetch(:et_al)}" if family.size > 2 || others
|
|
224
|
+
return "#{family.first} #{words.fetch(:and)} #{family.last}" if family.size == 2
|
|
225
|
+
|
|
226
|
+
family.first
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def short_title
|
|
230
|
+
words = title.to_s.split
|
|
231
|
+
words.size > 4 ? "#{words.first(4).join(' ')}\u2026" : title.to_s
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def year_or(no_date)
|
|
235
|
+
year.to_s.empty? ? no_date : year.to_s
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# First author, year, title: the order of an author-year reference list.
|
|
239
|
+
def sort_key
|
|
240
|
+
[authors.first&.family_name.to_s.downcase, year.to_s, title.to_s.downcase]
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
# ----------------------------------------------------------------- HTML
|
|
244
|
+
|
|
245
|
+
# The entry as the reference list shows it, in the manner of APA:
|
|
246
|
+
# Müller, J., & Smith, A. (2020). Title. Journal, 12(2), 10-20. https://doi.org/...
|
|
247
|
+
def reference_html(words, suffix: "")
|
|
248
|
+
date = "(#{h(year_or(words.fetch(:no_date)))}#{h(suffix)})."
|
|
249
|
+
parts = []
|
|
250
|
+
if authors.empty?
|
|
251
|
+
parts << "#{title_html}." if title
|
|
252
|
+
parts << date
|
|
253
|
+
else
|
|
254
|
+
parts << "#{h(listed_authors(words))} #{date}"
|
|
255
|
+
parts << "#{title_html}." if title
|
|
256
|
+
end
|
|
257
|
+
parts << "#{container_html}." if container
|
|
258
|
+
parts << "#{h(publisher)}." if publisher && !publisher.empty?
|
|
259
|
+
parts << "#{h(note)}." if note && !note.empty?
|
|
260
|
+
address = link
|
|
261
|
+
parts << %(<a href="#{h(address)}" rel="noopener noreferrer">#{h(address)}</a>) if address
|
|
262
|
+
parts.join(" ").gsub(/([.?!])\./, '\1')
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
def listed_authors(words)
|
|
266
|
+
names = authors.first(20).map(&:listed)
|
|
267
|
+
names << words.fetch(:et_al) if others || authors.size > 20
|
|
268
|
+
return names.first.to_s if names.size == 1
|
|
269
|
+
|
|
270
|
+
"#{names[0..-2].join(', ')} #{words.fetch(:and)} #{names.last}"
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
def title_html
|
|
274
|
+
STANDALONE.include?(type) ? "<em>#{h(title)}</em>" : h(title)
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
def container_html
|
|
278
|
+
text = "<em>#{h(container)}</em>"
|
|
279
|
+
text += ", #{h(volume)}" if volume && !volume.empty?
|
|
280
|
+
text += "(#{h(issue)})" if issue && !issue.empty?
|
|
281
|
+
text += ", #{h(pages)}" if pages && !pages.empty?
|
|
282
|
+
text
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
# ------------------------------------------------------------- metadata
|
|
286
|
+
|
|
287
|
+
# One Highwire citation_reference: "citation_title=...; citation_author=...".
|
|
288
|
+
def highwire
|
|
289
|
+
pairs = [["citation_title", title]]
|
|
290
|
+
authors.each { |name| pairs << ["citation_author", name.inverted] }
|
|
291
|
+
pairs << ["citation_publication_date", year]
|
|
292
|
+
pairs << [type == "paper-conference" ? "citation_conference_title" : "citation_journal_title", container]
|
|
293
|
+
pairs << ["citation_publisher", publisher] if STANDALONE.include?(type)
|
|
294
|
+
pairs.push(["citation_volume", volume], ["citation_issue", issue])
|
|
295
|
+
first, last = pages.to_s.split(/[-\u2013\u2014]+/, 2).map(&:strip)
|
|
296
|
+
pairs.push(["citation_firstpage", first], ["citation_lastpage", last])
|
|
297
|
+
pairs << ["citation_doi", doi_link&.delete_prefix("https://doi.org/")]
|
|
298
|
+
pairs.reject { |_, value| value.to_s.strip.empty? }
|
|
299
|
+
.map { |name, value| "#{name}=#{value.to_s.tr(';', ',').strip}" }.join("; ")
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
# A schema.org CreativeWork for the JSON-LD `citation` list.
|
|
303
|
+
def json_ld
|
|
304
|
+
work = { "@type" => "CreativeWork", "name" => title.to_s }
|
|
305
|
+
people = authors.map do |name|
|
|
306
|
+
{ "@type" => name.literal ? "Organization" : "Person", "name" => name.full }
|
|
307
|
+
end
|
|
308
|
+
work["author"] = people unless people.empty?
|
|
309
|
+
work["datePublished"] = year.to_s if year
|
|
310
|
+
work["isPartOf"] = container.to_s if container
|
|
311
|
+
if doi_link
|
|
312
|
+
work["sameAs"] = doi_link
|
|
313
|
+
elsif link
|
|
314
|
+
work["url"] = link
|
|
315
|
+
end
|
|
316
|
+
work
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
private
|
|
320
|
+
|
|
321
|
+
def h(value)
|
|
322
|
+
CGI.escapeHTML(value.to_s)
|
|
323
|
+
end
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
module_function
|
|
327
|
+
|
|
328
|
+
# "Smith and Jones (2020)" in a string entry's text becomes smith-and-jones-2020.
|
|
329
|
+
def slug(value)
|
|
330
|
+
value.to_s.downcase.gsub(/[^a-z0-9]+/, "-").gsub(/\A-|-\z/, "")
|
|
331
|
+
end
|
|
332
|
+
end
|
|
333
|
+
end
|