datalog-theme 0.10.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +127 -0
- data/CITATION.cff +2 -2
- data/README.md +4 -2
- data/_data/i18n/en.yml +55 -3
- data/_data/i18n/es.yml +55 -3
- data/_data/i18n/pt.yml +55 -3
- data/_includes/analytics/dashboard.html +24 -9
- data/_includes/components/academic-dashboard.html +168 -133
- data/_includes/components/author-bio.html +7 -5
- data/_includes/components/author-list.html +6 -5
- data/_includes/components/citation-tools.html +2 -1
- data/_includes/components/content-provenance.html +19 -1
- data/_includes/components/correction-report.html +15 -8
- data/_includes/components/hero.html +22 -15
- data/_includes/components/package-index.html +121 -0
- data/_includes/components/package-install.html +27 -50
- data/_includes/components/post-list.html +14 -0
- data/_includes/components/responsive-image.html +15 -21
- data/_includes/footer.html +1 -1
- data/_includes/head.html +70 -11
- data/_includes/layouts/default/article.html +76 -46
- data/_includes/meta/dataset-json.html +171 -0
- data/_includes/meta/math-config.html +15 -1
- data/_includes/meta/package-json.html +69 -0
- data/_includes/meta/person-json.html +36 -13
- data/_includes/meta/publisher.html +15 -0
- data/_includes/meta/schema.html +108 -23
- data/_includes/meta/scholarly.html +10 -5
- data/_includes/post/related-posts.html +17 -43
- data/_includes/scripts.html +2 -2
- data/_includes/search/index-data.json +10 -1
- data/_includes/search/page.html +11 -6
- data/_layouts/archive.html +67 -0
- data/_layouts/dataset.html +2 -1
- data/_layouts/default.html +13 -11
- data/_layouts/docs.html +53 -0
- data/_layouts/home.html +30 -1
- data/_layouts/notebook.html +5 -1
- data/_layouts/package.html +38 -21
- data/_layouts/portfolio.html +1 -0
- data/_layouts/post.html +21 -62
- data/_layouts/project.html +2 -1
- data/_layouts/research.html +14 -2
- data/_plugins/analytics_dashboard.rb +4 -1
- data/_plugins/archive.rb +114 -0
- data/_plugins/authors.rb +50 -1
- data/_plugins/citation_exports.rb +12 -1
- data/_plugins/config_validator.rb +77 -2
- data/_plugins/correction_fallback.rb +110 -0
- data/_plugins/image_optimizer.rb +129 -22
- data/_plugins/math_preprocessor.rb +7 -48
- data/_plugins/notebook_converter.rb +44 -5
- data/_plugins/packages.rb +84 -0
- data/_plugins/page_dates.rb +37 -0
- data/_plugins/references.rb +16 -2
- data/_plugins/related_posts.rb +149 -0
- data/_plugins/search_sections.rb +110 -0
- data/_plugins/site_identity.rb +42 -0
- data/_plugins/social_cards.rb +17 -0
- data/_sass/_academic-dashboard.scss +18 -30
- data/_sass/_base.scss +2 -0
- data/_sass/_components.scss +21 -23
- data/_sass/_docs.scss +143 -0
- data/_sass/_features.scss +6 -0
- data/_sass/_layout.scss +65 -6
- data/_sass/_mathematical.scss +61 -19
- data/_sass/_notebooks.scss +1 -32
- data/_sass/_package-docs.scss +6 -9
- data/_sass/_post-components.scss +23 -23
- data/_sass/_print.scss +1 -1
- data/_sass/_search-page.scss +75 -0
- data/_sass/_search.scss +17 -26
- data/_sass/_syntax-highlighting.scss +23 -1
- data/_sass/_theme.scss +1 -0
- data/_sass/_utilities.scss +87 -40
- data/_sass/_variables.scss +31 -29
- data/assets/css/main.scss +2 -1
- data/assets/js/dist/academic.js +1 -1
- data/assets/js/dist/analytics-dashboard.js +1 -1
- data/assets/js/dist/chunks/chunk-TNVD6UAM.js +1 -0
- data/assets/js/dist/comments.js +1 -1
- data/assets/js/dist/contact.js +1 -1
- data/assets/js/dist/core.js +1 -1
- data/assets/js/dist/corrections.js +5 -1
- data/assets/js/dist/loader.js +1 -1
- data/assets/js/dist/math.js +2 -1
- data/assets/js/dist/moderation.js +1 -1
- data/assets/js/dist/reactions.js +1 -1
- data/assets/js/dist/search.js +1 -1
- data/assets/js/dist/sources.json +20 -16
- data/assets/js/dist/subscriptions.js +1 -1
- data/assets/js/loader.js +6 -2
- data/lib/datalog/audit/checks.rb +269 -0
- data/lib/datalog/audit/known_keys.rb +61 -0
- data/lib/datalog/audit/site_reader.rb +118 -0
- data/lib/datalog/audit/source_file.rb +85 -0
- data/lib/datalog/audit.rb +145 -0
- data/lib/datalog/citations/bibtex.rb +200 -0
- data/lib/datalog/citations/entry.rb +333 -0
- data/lib/datalog/citations/markup.rb +79 -0
- data/lib/datalog/cli.rb +164 -89
- data/lib/datalog/critical_css.rb +126 -21
- data/lib/datalog/latex_speech/words.json +88 -0
- data/lib/datalog/latex_speech.rb +355 -0
- data/lib/datalog/packages/command.rb +51 -0
- data/lib/datalog/packages/refresh.rb +187 -0
- data/lib/datalog/packages.rb +177 -0
- data/lib/datalog/plugin_system.rb +5 -0
- data/lib/datalog/plugins/citations.rb +325 -109
- data/lib/datalog/site_config.rb +26 -0
- data/lib/datalog/social_cards/font.rb +286 -0
- data/lib/datalog/social_cards/fonts/IBMPlexSans-Regular.ttf +0 -0
- data/lib/datalog/social_cards/fonts/IBMPlexSerif-SemiBold.ttf +0 -0
- data/lib/datalog/social_cards/fonts/OFL.txt +92 -0
- data/lib/datalog/social_cards/geometry.rb +51 -0
- data/lib/datalog/social_cards/outline.rb +54 -0
- data/lib/datalog/social_cards/plain_text.rb +258 -0
- data/lib/datalog/social_cards/template.rb +322 -0
- data/lib/datalog/social_cards/template.svg +17 -0
- data/lib/datalog/social_cards/typesetter.rb +178 -0
- data/lib/datalog/social_cards.rb +369 -0
- data/lib/datalog/theme/updater.rb +307 -0
- data/lib/datalog/theme/version.rb +1 -1
- metadata +44 -7
- data/_includes/components/enhanced-code-block.html +0 -212
- data/_includes/components/performance-monitor.html +0 -170
- data/_includes/components/viz-table-fallback.html +0 -19
- data/_plugins/datalog_bibliography.rb +0 -21
- data/assets/js/dist/chunks/chunk-WIRUK3OZ.js +0 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
import{a as
|
|
1
|
+
import{a as R,b as O,c as T,d as q,e as I,f as U}from"./chunks/chunk-TNVD6UAM.js";import{a as x,c as F}from"./chunks/chunk-VZ5WKQVA.js";import"./chunks/chunk-PATLC23F.js";var h="subscriptions",N=/^[^\s@]+@[^\s@]+\.[^\s@]+$/;function $(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function j(e,a={}){let s={email:e.email||""};return a.hasTopics&&(s.topics=[].concat(e.topics||[]).filter(Boolean)),a.sourceUrl&&(s.source_url=a.sourceUrl),a.locale&&(s.locale=a.locale),s}function P(e,a,s={}){let n={};return N.test(e.email)||(n.email=s.email_invalid||"That does not look like an email address."),Array.isArray(e.topics)&&e.topics.length===0&&(n.topics=s.topics_required||"Pick at least one."),a.website&&(n.website="spam"),Object.keys(n).length>0?n:null}function J(e,a={}){let s=a.client||F(),n=e.querySelector("form"),c=$(e,"[data-subscribe-labels]"),g=c.errors||{},S=e.ownerDocument,o=S.querySelector('link[rel="canonical"]'),k={sourceUrl:e.dataset.subscribeSource||o&&o.href||"",locale:e.dataset.subscribeLocale||S.documentElement.lang||"",hasTopics:n.querySelector('[name="topics"]')!==null},w=e.dataset.subscribeDoubleOptIn!=="false",E=R(n),v=d=>{n.dataset.result=d,q(n,"success",c[d]||"")};if(!s.enabled)return q(n,"disabled",g.disabled||""),{form:n,available:!1,submit:async()=>null};let m=O(s,h,n,g);T(n,()=>m().catch(()=>{}));let p={form:n,available:!0,async submit(){let d=I(n),f=j(d,k),l=P(f,d,c);if(delete n.dataset.result,l)return l.website?(v("pending"),null):(q(n,"invalid",g.invalid||"",l),null);q(n,"pending",c.sending||"");try{await m()}catch(i){return null}try{let i=await s.post(s.pathFor(h),f,{idempotencyKey:E.key(f)}),u=i&&i.data&&typeof i.data.status=="string"?i.data.status:"";return E.clear(),n.reset(),v(u==="confirmed"||u==="subscribed"||u===""&&!w?"confirmed":"pending"),i}catch(i){return i&&i.kind==="conflict"?v("duplicate"):U(n,i,g),null}}};return n.addEventListener("submit",d=>{d.preventDefault(),p.submit()}),p}function D(e=document){return Array.from(e.querySelectorAll("[data-subscribe]")).map(a=>J(a))}var K=["confirm","unsubscribe","manage"],z=/^[A-Za-z0-9._~-]{8,512}$/;function B(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function V(e){let a=new URLSearchParams(e||""),s=K.find(c=>a.has(c));if(!s)return null;let n=(a.get(s)||"").trim();return z.test(n)?{action:s,token:n}:null}function Z(e,a={}){let s=e.ownerDocument,n=s.defaultView,c=a.client||F(),g=a.location||n&&n.location||{search:"",pathname:"/"},S=a.history===void 0?n&&n.history:a.history,o=B(e,"[data-manage-labels]"),k=o.errors||{},w=e.querySelector("[data-manage-status]"),E=e.querySelector("[data-manage-idle]"),v=e.querySelector("[data-manage-unsubscribe-panel]"),m=e.querySelector("[data-manage-preferences]"),p=V(g.search),d=p?`${c.pathFor(h)}/${encodeURIComponent(p.token)}`:"",f=null,l=s.createElement("button");l.type="button",l.className="post-tool",l.dataset.manageRetry="",l.textContent=o.retry||"Try again",l.hidden=!0,l.addEventListener("click",()=>f==null?void 0:f()),e.appendChild(l);let i=(r,t="",A=!1)=>{r!=="error"&&(f=null),e.dataset.state=r,l.hidden=r!=="error"||!f,w&&(w.textContent=t,w.hidden=t==="",w.setAttribute("role",A?"alert":"status"))},u=r=>{[E,v,m].forEach(t=>{t&&(t.hidden=t!==r)})},L=(r,t,A)=>{let C=t&&r&&(r.kind==="not_found"||r.status===410);f=r!=null&&r.retryable?A:null,u(null),i("error",C?o.invalid_link||"":x(r,k),!0)},y=r=>{e.setAttribute("aria-busy",r?"true":"false"),e.querySelectorAll("button").forEach(t=>{t.disabled=r})},b={root:e,request:p,async confirm(){let r=!1;y(!0),u(null),i("pending",o.confirming||"");try{await c.feature(h),r=!0;let t=await c.post(`${c.pathFor(h)}/confirm`,{token:p.token});return i("confirmed",o.confirmed||""),t}catch(t){return L(t,r,()=>b.confirm()),null}finally{y(!1)}},async unsubscribe(){let r=!1;y(!0),i("pending",o.unsubscribing||"");try{await c.feature(h),r=!0;let t=await c.delete(d);return u(null),i("unsubscribed",o.unsubscribed||""),t}catch(t){return L(t,r,()=>b.unsubscribe()),null}finally{y(!1)}},async load(){let r=!1;y(!0),u(null),i("loading",o.loading||"");try{await c.feature(h),r=!0;let t=await c.get(d),A=t&&t.data&&Array.isArray(t.data.topics)?t.data.topics.map(String):[];return m&&m.querySelectorAll('[name="topics"]').forEach(C=>{C.checked=A.includes(C.value)}),u(m),i("loaded"),A}catch(t){return L(t,r,()=>b.load()),null}finally{y(!1)}},async save(){let r=Array.from(m.querySelectorAll('[name="topics"]:checked')).map(t=>t.value);y(!0),i("pending",o.saving||"");try{let t=await c.patch(d,{topics:r});return i("saved",o.saved||""),t}catch(t){return f=t!=null&&t.retryable?()=>b.save():null,i("error",x(t,k),!0),null}finally{y(!1)}}};return e.querySelectorAll("[data-manage-unsubscribe]").forEach(r=>{r.addEventListener("click",()=>b.unsubscribe())}),m&&m.addEventListener("submit",r=>{r.preventDefault(),b.save()}),c.enabled?p?(S&&typeof S.replaceState=="function"&&S.replaceState(null,"",g.pathname),p.action==="confirm"?b.confirm():p.action==="manage"?b.load():(u(v),i("idle")),b):(u(E),i("idle"),b):(u(null),i("disabled",k.disabled||""),b)}function _(e=document){return Array.from(e.querySelectorAll("[data-subscription-manage]")).map(a=>Z(a))}function M(){D(document),_(document)}document.readyState==="loading"?document.addEventListener("DOMContentLoaded",M,{once:!0}):M();
|
data/assets/js/loader.js
CHANGED
|
@@ -74,8 +74,12 @@ const FEATURE_CONFIG = [
|
|
|
74
74
|
test: () => document.body?.dataset.featureMath === "true"
|
|
75
75
|
},
|
|
76
76
|
{
|
|
77
|
+
// The citation hooks, and the academic dashboard's submission and calendar
|
|
78
|
+
// filters, which a dashboard without citation metrics still has.
|
|
77
79
|
name: "academic",
|
|
78
|
-
test: () => document.querySelector(
|
|
80
|
+
test: () => document.querySelector(
|
|
81
|
+
"[data-citation-metric], [data-citation-table], [data-citation-chart], [data-submission-filter], [data-calendar-filter]"
|
|
82
|
+
)
|
|
79
83
|
},
|
|
80
84
|
{
|
|
81
85
|
name: "notebook",
|
|
@@ -88,7 +92,7 @@ const FEATURE_CONFIG = [
|
|
|
88
92
|
},
|
|
89
93
|
{
|
|
90
94
|
name: "corrections",
|
|
91
|
-
test: () => document.querySelector("[data-correction-report]")
|
|
95
|
+
test: () => document.querySelector("[data-correction-report], [data-correction-fallback]")
|
|
92
96
|
},
|
|
93
97
|
{
|
|
94
98
|
name: "contact",
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "date"
|
|
4
|
+
require "time"
|
|
5
|
+
require "uri"
|
|
6
|
+
|
|
7
|
+
module Datalog
|
|
8
|
+
class Audit
|
|
9
|
+
# What each check looks for in one file. A check returns [line, message]
|
|
10
|
+
# pairs; Audit names the check, its kind and the file.
|
|
11
|
+
class Checks
|
|
12
|
+
KINDS = %w[theorem lemma proposition corollary definition assumption example remark].freeze
|
|
13
|
+
# **Definition 2.**, __Theorem (Bayes)__, > **Lemma**, *Proof.*
|
|
14
|
+
STATEMENT = /\A\s*(?:>\s*)*(?:[-*+]\s+)?(\*\*|__)(#{(KINDS + ['proof']).join('|')})\b[^*_\n]*\1/i
|
|
15
|
+
PROOF = /\A\s*(?:>\s*)*(\*|_)proof\.?\1/i
|
|
16
|
+
TYPED_NUMBER = /\b(Figure|Fig\.|Table)\s+(\d+)\b/
|
|
17
|
+
REFERENCES = /\A\s{0,3}(?:\#{1,6}\s+|\*\*)
|
|
18
|
+
(References|Bibliography|Works\ cited|Literature\ cited|Sources)(?:\*\*)?[\s\#]*\z/ix
|
|
19
|
+
# Addresses are found with one plain pattern and judged by their host,
|
|
20
|
+
# compared whole: a pattern for the host would also match it inside
|
|
21
|
+
# another address.
|
|
22
|
+
URL = %r{https?://[^\s)<>\]"'`]+}i
|
|
23
|
+
REPOSITORY_HOSTS = %w[github.com gitlab.com bitbucket.org codeberg.org].freeze
|
|
24
|
+
NOTEBOOK_HOSTS = %w[colab.research.google.com mybinder.org nbviewer.org nbviewer.jupyter.org].freeze
|
|
25
|
+
DOI_HOSTS = %w[doi.org dx.doi.org].freeze
|
|
26
|
+
NOTEBOOK_FILE = /\.ipynb\b/i
|
|
27
|
+
BARE_DOI = %r{\bdoi:\s*10\.\d{4,9}/}i
|
|
28
|
+
# Link and image text stops at a bracket, so every [ starts a scan that
|
|
29
|
+
# ends at the next one, and a line of brackets takes linear time.
|
|
30
|
+
MARKDOWN_IMAGE = /!\[([^\[\]\n]*)\]\(\s*<?([^)\s>]+)/
|
|
31
|
+
HTML_IMAGE = /<img\b[^>]*>/i
|
|
32
|
+
IMAGE_FILE = /\.(?:png|jpe?g|gif|svg|webp|avif|bmp|tiff?)\z/i
|
|
33
|
+
MARKDOWN_LINK = %r{(?<!!)\[[^\[\]\n]*\]\(\s*<?(/(?!/)[^)\s>]*)}
|
|
34
|
+
HTML_LINK = %r{\bhref\s*=\s*["'](/(?!/)[^"']*)["']}i
|
|
35
|
+
PART = /\bpart[\s_-]*(\d+|[ivx]+)\b/i
|
|
36
|
+
# A link to a page of the site, and link text that names another part.
|
|
37
|
+
SITE_LINK = %r{\[([^\[\]\n]*)\]\(\s*<?/}
|
|
38
|
+
SEQUENCE_TEXT = /\b(?:part\s+(?:\d+|[ivx]+)|(?:next|previous)\s+(?:part|post|article|chapter))\b/i
|
|
39
|
+
|
|
40
|
+
def initialize(reader:, known_keys:, settings:)
|
|
41
|
+
@reader = reader
|
|
42
|
+
@known_keys = known_keys
|
|
43
|
+
@settings = settings
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# --------------------------------------------------------- opportunities
|
|
47
|
+
|
|
48
|
+
def statements(file)
|
|
49
|
+
file.each_line.filter_map do |line, number|
|
|
50
|
+
match = line.match(STATEMENT) || line.match(PROOF)
|
|
51
|
+
next unless match
|
|
52
|
+
|
|
53
|
+
kind = (match[2] || "proof").downcase
|
|
54
|
+
[number, "#{kind.capitalize} typed by hand (#{match[0].strip}): {% #{kind} %} numbers it and " \
|
|
55
|
+
"{% ref %} links to it"]
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def figures(file)
|
|
60
|
+
file.each_line.filter_map do |line, number|
|
|
61
|
+
match = line.match(TYPED_NUMBER)
|
|
62
|
+
next unless match
|
|
63
|
+
|
|
64
|
+
tag = match[1] == "Table" ? "table" : "figure"
|
|
65
|
+
[number, "\"#{match[0]}\" typed by hand: {% #{tag} %} numbers it and {% ref %} keeps the " \
|
|
66
|
+
"number right when the order changes"]
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def references(file)
|
|
71
|
+
return [] if file.body.include?("{% cite") || file.body.include?("{%- cite")
|
|
72
|
+
|
|
73
|
+
file.each_line.filter_map do |line, number|
|
|
74
|
+
next unless line.match?(REFERENCES)
|
|
75
|
+
|
|
76
|
+
[number, "a references section written by hand: {% cite %} from a BibTeX file numbers the " \
|
|
77
|
+
"citations and lists the works they cite"]
|
|
78
|
+
end.first(1)
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def reproducibility(file)
|
|
82
|
+
return [] if !file.post? || file.front_matter.key?("reproducibility")
|
|
83
|
+
|
|
84
|
+
file.each_line do |line, number|
|
|
85
|
+
found = research_link(line)
|
|
86
|
+
if found
|
|
87
|
+
return [[number, "links #{found} but has no reproducibility: block, which sets out the code, " \
|
|
88
|
+
"data and environment behind the article"]]
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
[]
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def revisions(file)
|
|
95
|
+
data = file.front_matter
|
|
96
|
+
return [] unless file.post? && !data.key?("revisions")
|
|
97
|
+
|
|
98
|
+
published = to_time(data["date"]) || date_from_name(file.relative)
|
|
99
|
+
return [] unless published
|
|
100
|
+
|
|
101
|
+
key = %w[last_modified_at updated].find { |name| data[name] }
|
|
102
|
+
edited = key ? to_time(data[key]) : @reader.git_dates[file.relative]
|
|
103
|
+
return [] unless edited && edited - published > revision_gap
|
|
104
|
+
|
|
105
|
+
source = key ? "`#{key}`" : "its last commit"
|
|
106
|
+
[[key ? file.key_line(key) : 1,
|
|
107
|
+
"edited #{(edited.to_date - published.to_date).to_i} days after publication (#{source}) with no " \
|
|
108
|
+
"revisions: entry, which tells readers what changed"]]
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
# Posts whose titles or file names say "Part 2", or that link to the next
|
|
112
|
+
# part by hand, and are in no series.
|
|
113
|
+
def series(file)
|
|
114
|
+
return [] if file.front_matter.key?("series") || !file.post?
|
|
115
|
+
|
|
116
|
+
found = []
|
|
117
|
+
title = file.front_matter["title"].to_s
|
|
118
|
+
part = title[PART] || File.basename(file.relative)[PART]
|
|
119
|
+
if part
|
|
120
|
+
where = title[PART] ? "title" : "file name"
|
|
121
|
+
found << [file.key_line("title"), "\"#{part}\" in the #{where} and no series: series: numbers the " \
|
|
122
|
+
"parts and links each to the next"]
|
|
123
|
+
end
|
|
124
|
+
file.each_line do |line, number|
|
|
125
|
+
link = line.scan(SITE_LINK).flatten.find { |text| text.match?(SEQUENCE_TEXT) }
|
|
126
|
+
found << [number, "links to \"#{link}\" by hand: series: links the parts in order"] if link
|
|
127
|
+
end
|
|
128
|
+
found.first(2)
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# -------------------------------------------------------------- problems
|
|
132
|
+
|
|
133
|
+
# Fields other themes read and DataLog does not, named so the finding
|
|
134
|
+
# says where a key came from.
|
|
135
|
+
FOREIGN = {
|
|
136
|
+
"Minimal Mistakes" => %w[author_profile seo_type read_time share related sidebar toc_sticky toc_label
|
|
137
|
+
toc_icon header_image],
|
|
138
|
+
"Just the Docs" => %w[nav_exclude nav_order parent grand_parent has_children has_toc search_exclude],
|
|
139
|
+
"Chirpy" => %w[pin math_rendering mermaid media_subpath]
|
|
140
|
+
}.freeze
|
|
141
|
+
|
|
142
|
+
def front_matter(file)
|
|
143
|
+
file.front_matter.keys.reject { |key| @known_keys.include?(key.to_s) }.map do |key|
|
|
144
|
+
theme = FOREIGN.find { |_, keys| keys.include?(key.to_s) }&.first
|
|
145
|
+
reason = theme ? "a #{theme} field DataLog ignores" : "a typo, or a field of another theme"
|
|
146
|
+
[file.key_line(key), "front matter key \"#{key}\" is read by no layout, include or plugin: #{reason} " \
|
|
147
|
+
"(audit.known_keys lists a site's own)"]
|
|
148
|
+
end
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
def images(file)
|
|
152
|
+
found = []
|
|
153
|
+
file.each_line do |line, number|
|
|
154
|
+
line.scan(MARKDOWN_IMAGE) do |alt, src|
|
|
155
|
+
problem = alt_problem(alt, src)
|
|
156
|
+
found << [number, "image #{src} #{problem}"] if problem
|
|
157
|
+
end
|
|
158
|
+
line.scan(HTML_IMAGE) do
|
|
159
|
+
tag = Regexp.last_match(0)
|
|
160
|
+
src = tag[/\bsrc\s*=\s*["']([^"']*)["']/i, 1].to_s
|
|
161
|
+
alt = tag[/\balt\s*=\s*["']([^"']*)["']/i, 1]
|
|
162
|
+
problem = alt.nil? ? "has no alt attribute" : (alt_problem(alt, src) unless alt.empty?)
|
|
163
|
+
found << [number, "image #{src} #{problem}"] if problem
|
|
164
|
+
end
|
|
165
|
+
end
|
|
166
|
+
found
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
def links(file)
|
|
170
|
+
found = []
|
|
171
|
+
file.each_line do |line, number|
|
|
172
|
+
(line.scan(MARKDOWN_LINK) + line.scan(HTML_LINK)).flatten.each do |target|
|
|
173
|
+
path = strip_baseurl(target)
|
|
174
|
+
next if path.include?("{{") || ignored_link?(path) || @reader.built?(path)
|
|
175
|
+
|
|
176
|
+
found << [number, "links #{target}, which the site does not build"]
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
found
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Posts only: a page such as the search page loads the engine for math
|
|
183
|
+
# its script fetches, which the page's own source cannot show.
|
|
184
|
+
def math(file)
|
|
185
|
+
data = file.front_matter
|
|
186
|
+
return [] unless file.post?
|
|
187
|
+
|
|
188
|
+
key = data.key?("math") ? "math" : ("mathjax" if data.key?("mathjax"))
|
|
189
|
+
return [] unless key && [true, false].include?(data[key])
|
|
190
|
+
|
|
191
|
+
found = detect_math(file.body)
|
|
192
|
+
if data[key] == true && !found
|
|
193
|
+
[[file.key_line(key), "#{key}: true loads the math engine, but the page has no math it can find"]]
|
|
194
|
+
elsif data[key] == false && found
|
|
195
|
+
[[file.key_line(key), "#{key}: false leaves the page's math (#{found}) as LaTeX source"]]
|
|
196
|
+
else
|
|
197
|
+
[]
|
|
198
|
+
end
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
private
|
|
202
|
+
|
|
203
|
+
# The first address on the line that is a repository, a notebook or a DOI,
|
|
204
|
+
# or a notebook file or DOI written without an address.
|
|
205
|
+
def research_link(line)
|
|
206
|
+
line.scan(URL).find { |url| research_address?(url) } ||
|
|
207
|
+
line.split.find { |word| word.match?(NOTEBOOK_FILE) } || line[BARE_DOI]
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
def research_address?(url)
|
|
211
|
+
uri = URI.parse(url)
|
|
212
|
+
host = uri.host.to_s.downcase.delete_prefix("www.")
|
|
213
|
+
segments = uri.path.to_s.split("/").reject(&:empty?)
|
|
214
|
+
return segments.size >= 2 if REPOSITORY_HOSTS.include?(host)
|
|
215
|
+
return segments.first.to_s.start_with?("10.") if DOI_HOSTS.include?(host)
|
|
216
|
+
|
|
217
|
+
NOTEBOOK_HOSTS.include?(host) || uri.path.to_s.match?(NOTEBOOK_FILE)
|
|
218
|
+
rescue URI::InvalidURIError
|
|
219
|
+
false
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
def alt_problem(alt, src)
|
|
223
|
+
text = alt.to_s.strip
|
|
224
|
+
return "has no alt text" if text.empty?
|
|
225
|
+
|
|
226
|
+
name = File.basename(src.to_s.split(/[?#]/).first.to_s)
|
|
227
|
+
"has its file name as alt text" if text.match?(IMAGE_FILE) || text == name || text == File.basename(name, ".*")
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def strip_baseurl(target)
|
|
231
|
+
base = @reader.config["baseurl"].to_s.chomp("/")
|
|
232
|
+
base.empty? || !target.start_with?("#{base}/") ? target : target.delete_prefix(base)
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
def ignored_link?(path)
|
|
236
|
+
Array(@settings["ignore_links"]).any? { |prefix| path.start_with?(prefix.to_s) }
|
|
237
|
+
end
|
|
238
|
+
|
|
239
|
+
# The first expression the math preprocessor would render, or nil.
|
|
240
|
+
def detect_math(body)
|
|
241
|
+
return nil unless defined?(::MathPreprocessor::Processor)
|
|
242
|
+
|
|
243
|
+
processor = ::MathPreprocessor::Processor.new(body)
|
|
244
|
+
processor.process
|
|
245
|
+
expression = processor.expressions.first
|
|
246
|
+
expression && expression["latex"].to_s.strip[0, 40]
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
def revision_gap
|
|
250
|
+
(@settings["revision_after_days"] || 30).to_i * 86_400
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def to_time(value)
|
|
254
|
+
case value
|
|
255
|
+
when Time then value
|
|
256
|
+
when Date then value.to_time
|
|
257
|
+
when String then Time.parse(value)
|
|
258
|
+
end
|
|
259
|
+
rescue ArgumentError
|
|
260
|
+
nil
|
|
261
|
+
end
|
|
262
|
+
|
|
263
|
+
def date_from_name(relative)
|
|
264
|
+
stamp = File.basename(relative)[/\A\d{4}-\d{2}-\d{2}/]
|
|
265
|
+
stamp && Time.parse(stamp)
|
|
266
|
+
end
|
|
267
|
+
end
|
|
268
|
+
end
|
|
269
|
+
end
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Datalog
|
|
4
|
+
class Audit
|
|
5
|
+
# The front matter keys something reads, found in the sources that read
|
|
6
|
+
# them rather than kept as a list: the theme's layouts, includes, plugins
|
|
7
|
+
# and library, and the site's own layouts, includes and plugins. A key
|
|
8
|
+
# none of them names is a typo (`descripton`) or a leftover of another
|
|
9
|
+
# theme, which nothing will ever show.
|
|
10
|
+
module KnownKeys
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Keys that Jekyll and the plugins the theme depends on read, which the
|
|
14
|
+
# theme's own sources do not name.
|
|
15
|
+
JEKYLL = %w[
|
|
16
|
+
layout permalink published date categories category tags tag title excerpt excerpt_separator slug
|
|
17
|
+
lang render_with_liquid collection output path url id ext draft hidden
|
|
18
|
+
sitemap redirect_from redirect_to last_modified_at image seo canonical_url locale author description
|
|
19
|
+
paginate pagination feed
|
|
20
|
+
].freeze
|
|
21
|
+
|
|
22
|
+
TEMPLATE_DIRS = %w[_layouts _includes].freeze
|
|
23
|
+
CODE_DIRS = %w[_plugins lib].freeze
|
|
24
|
+
# page.title, include.page.summary, post.series and every other property
|
|
25
|
+
# a template reads, and quoted names in filters (where: "layout").
|
|
26
|
+
TEMPLATE_KEY = /[\w\]]\.([a-z_][a-z0-9_]*)|["']([a-z_][a-z0-9_]*)["']/
|
|
27
|
+
# data["reproducibility"], fetch("title"), Authors.value(page, "series").
|
|
28
|
+
CODE_KEY = /["']([a-z_][a-z0-9_]*)["']/
|
|
29
|
+
|
|
30
|
+
def build(theme_root:, site_root:, extra: [])
|
|
31
|
+
keys = Set.new(JEKYLL)
|
|
32
|
+
[theme_root, site_root].uniq.each do |root|
|
|
33
|
+
scan(root, TEMPLATE_DIRS, "**/*.{html,liquid,xml,json,md}", TEMPLATE_KEY, keys)
|
|
34
|
+
scan(root, CODE_DIRS, "**/*.rb", CODE_KEY, keys)
|
|
35
|
+
end
|
|
36
|
+
keys.merge(Array(extra).map(&:to_s))
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def scan(root, dirs, glob, pattern, keys)
|
|
40
|
+
dirs.each do |dir|
|
|
41
|
+
Dir.glob(File.join(root, dir, glob)).each do |file|
|
|
42
|
+
File.read(file, encoding: "utf-8").scan(pattern) { |match| keys.merge(Array(match).compact) }
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# Liquid output and tags; neither holds a brace, so a scan from one {{
|
|
48
|
+
# stops at the next brace and the file is read in linear time.
|
|
49
|
+
LIQUID = /\{\{[^{}]*\}\}|\{%(?:[^{}%]|%(?!\}))*%\}/
|
|
50
|
+
|
|
51
|
+
# The properties a page's own Liquid reads, such as a showcase page that
|
|
52
|
+
# passes `page.academic_demo` to an include, or a listing that shows
|
|
53
|
+
# `post.subtitle`: read by the site's content rather than its templates.
|
|
54
|
+
def from_content(text)
|
|
55
|
+
text.scan(LIQUID).each_with_object(Set.new) do |liquid, keys|
|
|
56
|
+
liquid.scan(TEMPLATE_KEY) { |match| keys.merge(Array(match).compact) }
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "open3"
|
|
4
|
+
require "time"
|
|
5
|
+
require "tmpdir"
|
|
6
|
+
|
|
7
|
+
module Datalog
|
|
8
|
+
class Audit
|
|
9
|
+
# What Jekyll would build, found without building it: the site's own
|
|
10
|
+
# content files and every URL the build has, read with Jekyll's reader and
|
|
11
|
+
# the generators that only add pages in memory. Nothing is rendered or
|
|
12
|
+
# written; the destination is a directory that never exists.
|
|
13
|
+
class SiteReader
|
|
14
|
+
# The generators that add pages (feeds, the sitemap, pagination pages,
|
|
15
|
+
# redirects, the search page, notebook pages) and write nothing until the
|
|
16
|
+
# site is written. The others are left out: they fetch data, encode
|
|
17
|
+
# images or fill caches.
|
|
18
|
+
PAGE_GENERATORS = %w[
|
|
19
|
+
JekyllFeed::Generator Jekyll::JekyllSitemap Jekyll::Paginate::Pagination
|
|
20
|
+
JekyllRedirectFrom::Generator Datalog::SearchPages Jekyll::NotebookConverter
|
|
21
|
+
].freeze
|
|
22
|
+
CONTENT = /\.(md|markdown|html?)\z/i
|
|
23
|
+
|
|
24
|
+
attr_reader :root, :site
|
|
25
|
+
|
|
26
|
+
def initialize(root)
|
|
27
|
+
@root = File.expand_path(root)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def read
|
|
31
|
+
require "jekyll"
|
|
32
|
+
Jekyll.logger.log_level = :error
|
|
33
|
+
config = Jekyll.configuration(
|
|
34
|
+
"source" => root, "destination" => File.join(Dir.tmpdir, "datalog-audit-#{Process.pid}", "_site"),
|
|
35
|
+
"quiet" => true, "disable_disk_cache" => true
|
|
36
|
+
)
|
|
37
|
+
@site = Jekyll::Site.new(config)
|
|
38
|
+
@site.reset
|
|
39
|
+
@site.read
|
|
40
|
+
@site.generators.each do |generator|
|
|
41
|
+
generator.generate(@site) if PAGE_GENERATORS.include?(generator.class.name)
|
|
42
|
+
end
|
|
43
|
+
self
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def config
|
|
47
|
+
site.config
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# [absolute path, path from the site root] of every page and document the
|
|
51
|
+
# site's own sources hold, in a stable order.
|
|
52
|
+
def content_files
|
|
53
|
+
documents = site.collections.values.flat_map(&:docs)
|
|
54
|
+
pages = site.pages.select { |page| page.instance_of?(Jekyll::Page) }
|
|
55
|
+
files = (documents + pages).filter_map do |item|
|
|
56
|
+
path = File.expand_path(item.relative_path, root)
|
|
57
|
+
next unless path.start_with?("#{root}/") && File.file?(path) && path.match?(CONTENT)
|
|
58
|
+
|
|
59
|
+
[path, path.delete_prefix("#{root}/")]
|
|
60
|
+
end
|
|
61
|
+
files.uniq.sort_by(&:last)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Every URL the build writes, normalised: pages, documents, generated
|
|
65
|
+
# pages and static files.
|
|
66
|
+
def urls
|
|
67
|
+
@urls ||= begin
|
|
68
|
+
items = site.pages + site.static_files + site.collections.values.flat_map(&:docs).select(&:write?)
|
|
69
|
+
items.to_set { |item| normalize(item.url) }
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def normalize(url)
|
|
74
|
+
path = url.to_s.split(/[?#]/, 2).first.to_s
|
|
75
|
+
path = "/#{path}" unless path.start_with?("/")
|
|
76
|
+
path = path.sub(%r{/index\.html?\z}, "/")
|
|
77
|
+
path.end_with?("/") || path.length == 1 ? path : path.sub(/\.html?\z/, "")
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def built?(url)
|
|
81
|
+
path = normalize(url)
|
|
82
|
+
[path, "#{path}/", path.delete_suffix("/")].any? { |candidate| urls.include?(candidate) }
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
# A commit that touches more files than this is a sweeping change (a
|
|
86
|
+
# reformat, a migration, a rename), not an edit of any one article.
|
|
87
|
+
SWEEPING_COMMIT = 10
|
|
88
|
+
|
|
89
|
+
# The date of the last commit to edit each file, from one `git log` over
|
|
90
|
+
# the site, or {} outside a repository.
|
|
91
|
+
def git_dates
|
|
92
|
+
@git_dates ||= begin
|
|
93
|
+
output, status = Open3.capture2("git", "-C", root, "log", "--format=@@date@@%cI", "--name-only", "--relative",
|
|
94
|
+
"--", ".", err: File::NULL)
|
|
95
|
+
status.success? ? parse_log(output) : {}
|
|
96
|
+
rescue SystemCallError
|
|
97
|
+
{}
|
|
98
|
+
end
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
private
|
|
102
|
+
|
|
103
|
+
def parse_log(output)
|
|
104
|
+
dates = {}
|
|
105
|
+
output.split("@@date@@").each do |commit|
|
|
106
|
+
stamp, *paths = commit.split("\n").map(&:strip).reject(&:empty?)
|
|
107
|
+
next if stamp.nil? || paths.size > SWEEPING_COMMIT
|
|
108
|
+
|
|
109
|
+
date = Time.iso8601(stamp)
|
|
110
|
+
paths.each { |path| dates[path] ||= date }
|
|
111
|
+
rescue ArgumentError
|
|
112
|
+
next
|
|
113
|
+
end
|
|
114
|
+
dates
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
end
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "date"
|
|
4
|
+
require "yaml"
|
|
5
|
+
|
|
6
|
+
module Datalog
|
|
7
|
+
class Audit
|
|
8
|
+
# One content file as its author wrote it: the front matter keys with the
|
|
9
|
+
# line each is on, and the body with its code set aside, line by line, so
|
|
10
|
+
# a finding names the line the author would edit.
|
|
11
|
+
class SourceFile
|
|
12
|
+
# Code shows what it shows: nothing inside it is a statement, a figure
|
|
13
|
+
# reference, a link or an image. Each match is blanked character for
|
|
14
|
+
# character, newlines kept, so line numbers do not move.
|
|
15
|
+
CODE = [
|
|
16
|
+
/^[ \t]*(`{3,}|~{3,})[^\n]*\n.*?(?:^[ \t]*\1[ \t]*$|\z)/m,
|
|
17
|
+
/\{%-?\s*(highlight|raw)\b.*?\{%-?\s*end\1\s*-?%\}/m,
|
|
18
|
+
%r{<(pre|code)\b[^>]*>.*?</\1>}mi,
|
|
19
|
+
/`[^`\n]+`/,
|
|
20
|
+
/<!--.*?-->/m,
|
|
21
|
+
/\{%-?\s*comment\s*-?%\}.*?\{%-?\s*endcomment\s*-?%\}/m
|
|
22
|
+
].freeze
|
|
23
|
+
|
|
24
|
+
attr_reader :path, :relative, :front_matter, :key_lines, :body_start, :body, :lines
|
|
25
|
+
|
|
26
|
+
def initialize(path, relative)
|
|
27
|
+
@path = path
|
|
28
|
+
@relative = relative
|
|
29
|
+
text = File.read(path, encoding: "bom|utf-8")
|
|
30
|
+
@front_matter, @key_lines, @body_start, @body = split(text)
|
|
31
|
+
@lines = mask(@body).split("\n", -1)
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# The body's lines with code blanked, each with its line in the file.
|
|
35
|
+
def each_line
|
|
36
|
+
return enum_for(:each_line) unless block_given?
|
|
37
|
+
|
|
38
|
+
@lines.each_with_index { |line, index| yield line, @body_start + index }
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# The line in the file of the body's `offset`th character.
|
|
42
|
+
def line_at(offset)
|
|
43
|
+
@body_start + masked_body[0, offset].count("\n")
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def masked_body
|
|
47
|
+
@masked_body ||= @lines.join("\n")
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def key_line(key)
|
|
51
|
+
@key_lines.fetch(key, 1)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def post?
|
|
55
|
+
relative.match?(%r{(\A|/)_posts/})
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
private
|
|
59
|
+
|
|
60
|
+
def split(text)
|
|
61
|
+
match = text.match(/\A---[ \t]*\r?\n(.*?)^---[ \t]*\r?$\n?/m)
|
|
62
|
+
return [{}, {}, 1, text] unless match
|
|
63
|
+
|
|
64
|
+
yaml = match[1]
|
|
65
|
+
data = begin
|
|
66
|
+
YAML.safe_load(yaml, permitted_classes: [Date, Time], aliases: true)
|
|
67
|
+
rescue Psych::Exception
|
|
68
|
+
nil
|
|
69
|
+
end
|
|
70
|
+
key_lines = {}
|
|
71
|
+
yaml.each_line.with_index(2) do |line, number|
|
|
72
|
+
key = line[/\A([A-Za-z_][\w-]*)\s*:/, 1]
|
|
73
|
+
key_lines[key] ||= number if key
|
|
74
|
+
end
|
|
75
|
+
[data.is_a?(Hash) ? data : {}, key_lines, match[0].count("\n") + 1, match.post_match]
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def mask(text)
|
|
79
|
+
CODE.reduce(text) do |masked, pattern|
|
|
80
|
+
masked.gsub(pattern) { |code| code.gsub(/[^\n]/, " ") }
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|