datalog-theme 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +133 -0
  3. data/CITATION.cff +2 -2
  4. data/README.md +4 -2
  5. data/_data/i18n/en.yml +55 -3
  6. data/_data/i18n/es.yml +55 -3
  7. data/_data/i18n/pt.yml +55 -3
  8. data/_includes/analytics/dashboard.html +24 -9
  9. data/_includes/components/academic-dashboard.html +168 -133
  10. data/_includes/components/author-bio.html +7 -5
  11. data/_includes/components/author-list.html +6 -5
  12. data/_includes/components/citation-tools.html +2 -1
  13. data/_includes/components/content-provenance.html +19 -1
  14. data/_includes/components/correction-report.html +15 -8
  15. data/_includes/components/hero.html +22 -15
  16. data/_includes/components/package-index.html +121 -0
  17. data/_includes/components/package-install.html +27 -50
  18. data/_includes/components/post-list.html +14 -0
  19. data/_includes/components/responsive-image.html +15 -21
  20. data/_includes/footer.html +1 -1
  21. data/_includes/head.html +70 -11
  22. data/_includes/layouts/default/article.html +76 -46
  23. data/_includes/meta/dataset-json.html +171 -0
  24. data/_includes/meta/math-config.html +15 -1
  25. data/_includes/meta/package-json.html +69 -0
  26. data/_includes/meta/person-json.html +36 -13
  27. data/_includes/meta/publisher.html +15 -0
  28. data/_includes/meta/schema.html +108 -23
  29. data/_includes/meta/scholarly.html +10 -5
  30. data/_includes/post/related-posts.html +17 -43
  31. data/_includes/scripts.html +2 -2
  32. data/_includes/search/index-data.json +10 -1
  33. data/_includes/search/page.html +11 -6
  34. data/_layouts/archive.html +67 -0
  35. data/_layouts/dataset.html +2 -1
  36. data/_layouts/default.html +13 -11
  37. data/_layouts/docs.html +53 -0
  38. data/_layouts/home.html +30 -1
  39. data/_layouts/notebook.html +5 -1
  40. data/_layouts/package.html +38 -21
  41. data/_layouts/portfolio.html +1 -0
  42. data/_layouts/post.html +21 -62
  43. data/_layouts/project.html +2 -1
  44. data/_layouts/research.html +14 -2
  45. data/_plugins/analytics_dashboard.rb +4 -1
  46. data/_plugins/archive.rb +114 -0
  47. data/_plugins/authors.rb +50 -1
  48. data/_plugins/citation_exports.rb +12 -1
  49. data/_plugins/config_validator.rb +77 -2
  50. data/_plugins/correction_fallback.rb +110 -0
  51. data/_plugins/image_optimizer.rb +129 -22
  52. data/_plugins/math_preprocessor.rb +7 -48
  53. data/_plugins/notebook_converter.rb +44 -5
  54. data/_plugins/packages.rb +84 -0
  55. data/_plugins/page_dates.rb +37 -0
  56. data/_plugins/references.rb +16 -2
  57. data/_plugins/related_posts.rb +149 -0
  58. data/_plugins/search_sections.rb +110 -0
  59. data/_plugins/site_identity.rb +42 -0
  60. data/_plugins/social_cards.rb +17 -0
  61. data/_sass/_academic-dashboard.scss +18 -30
  62. data/_sass/_base.scss +2 -0
  63. data/_sass/_components.scss +21 -23
  64. data/_sass/_docs.scss +143 -0
  65. data/_sass/_features.scss +6 -0
  66. data/_sass/_layout.scss +65 -6
  67. data/_sass/_mathematical.scss +61 -19
  68. data/_sass/_notebooks.scss +1 -32
  69. data/_sass/_package-docs.scss +6 -9
  70. data/_sass/_post-components.scss +23 -23
  71. data/_sass/_print.scss +1 -1
  72. data/_sass/_search-page.scss +75 -0
  73. data/_sass/_search.scss +17 -26
  74. data/_sass/_syntax-highlighting.scss +23 -1
  75. data/_sass/_theme.scss +1 -0
  76. data/_sass/_utilities.scss +87 -40
  77. data/_sass/_variables.scss +31 -29
  78. data/assets/css/main.scss +2 -1
  79. data/assets/js/dist/academic.js +1 -1
  80. data/assets/js/dist/analytics-dashboard.js +1 -1
  81. data/assets/js/dist/chunks/chunk-TNVD6UAM.js +1 -0
  82. data/assets/js/dist/comments.js +1 -1
  83. data/assets/js/dist/contact.js +1 -1
  84. data/assets/js/dist/core.js +1 -1
  85. data/assets/js/dist/corrections.js +5 -1
  86. data/assets/js/dist/loader.js +1 -1
  87. data/assets/js/dist/math.js +2 -1
  88. data/assets/js/dist/moderation.js +1 -1
  89. data/assets/js/dist/reactions.js +1 -1
  90. data/assets/js/dist/search.js +1 -1
  91. data/assets/js/dist/sources.json +20 -16
  92. data/assets/js/dist/subscriptions.js +1 -1
  93. data/assets/js/loader.js +6 -2
  94. data/datalog-theme.gemspec +1 -1
  95. data/lib/datalog/audit/checks.rb +269 -0
  96. data/lib/datalog/audit/known_keys.rb +61 -0
  97. data/lib/datalog/audit/site_reader.rb +118 -0
  98. data/lib/datalog/audit/source_file.rb +85 -0
  99. data/lib/datalog/audit.rb +145 -0
  100. data/lib/datalog/citations/bibtex.rb +200 -0
  101. data/lib/datalog/citations/entry.rb +333 -0
  102. data/lib/datalog/citations/markup.rb +79 -0
  103. data/lib/datalog/cli.rb +164 -89
  104. data/lib/datalog/critical_css.rb +126 -21
  105. data/lib/datalog/latex_speech/words.json +88 -0
  106. data/lib/datalog/latex_speech.rb +355 -0
  107. data/lib/datalog/packages/command.rb +51 -0
  108. data/lib/datalog/packages/refresh.rb +187 -0
  109. data/lib/datalog/packages.rb +177 -0
  110. data/lib/datalog/plugin_system.rb +5 -0
  111. data/lib/datalog/plugins/citations.rb +325 -109
  112. data/lib/datalog/site_config.rb +26 -0
  113. data/lib/datalog/social_cards/font.rb +286 -0
  114. data/lib/datalog/social_cards/fonts/IBMPlexSans-Regular.ttf +0 -0
  115. data/lib/datalog/social_cards/fonts/IBMPlexSerif-SemiBold.ttf +0 -0
  116. data/lib/datalog/social_cards/fonts/OFL.txt +92 -0
  117. data/lib/datalog/social_cards/geometry.rb +51 -0
  118. data/lib/datalog/social_cards/outline.rb +54 -0
  119. data/lib/datalog/social_cards/plain_text.rb +258 -0
  120. data/lib/datalog/social_cards/template.rb +322 -0
  121. data/lib/datalog/social_cards/template.svg +17 -0
  122. data/lib/datalog/social_cards/typesetter.rb +178 -0
  123. data/lib/datalog/social_cards.rb +369 -0
  124. data/lib/datalog/theme/updater.rb +307 -0
  125. data/lib/datalog/theme/version.rb +1 -1
  126. metadata +46 -9
  127. data/_includes/components/enhanced-code-block.html +0 -212
  128. data/_includes/components/performance-monitor.html +0 -170
  129. data/_includes/components/viz-table-fallback.html +0 -19
  130. data/_plugins/datalog_bibliography.rb +0 -21
  131. data/assets/js/dist/chunks/chunk-WIRUK3OZ.js +0 -1
@@ -1 +1 @@
1
- import{a as O,b as R,c as q,d as T,e as I}from"./chunks/chunk-WIRUK3OZ.js";import{a as x,c as C}from"./chunks/chunk-VZ5WKQVA.js";import"./chunks/chunk-PATLC23F.js";var h="subscriptions",M=/^[^\s@]+@[^\s@]+\.[^\s@]+$/;function N(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function $(e,a={}){let s={email:e.email||""};return a.hasTopics&&(s.topics=[].concat(e.topics||[]).filter(Boolean)),a.sourceUrl&&(s.source_url=a.sourceUrl),a.locale&&(s.locale=a.locale),s}function j(e,a,s={}){let n={};return M.test(e.email)||(n.email=s.email_invalid||"That does not look like an email address."),Array.isArray(e.topics)&&e.topics.length===0&&(n.topics=s.topics_required||"Pick at least one."),a.website&&(n.website="spam"),Object.keys(n).length>0?n:null}function P(e,a={}){let s=a.client||C(),n=e.querySelector("form"),c=N(e,"[data-subscribe-labels]"),g=c.errors||{},S=e.ownerDocument,o=S.querySelector('link[rel="canonical"]'),k={sourceUrl:e.dataset.subscribeSource||o&&o.href||"",locale:e.dataset.subscribeLocale||S.documentElement.lang||"",hasTopics:n.querySelector('[name="topics"]')!==null},v=e.dataset.subscribeDoubleOptIn!=="false",E=O(n),w=d=>{n.dataset.result=d,q(n,"success",c[d]||"")};if(!s.enabled)return q(n,"disabled",g.disabled||""),{form:n,available:!1,submit:async()=>null};let m=R(s,h,n,g);n.addEventListener("focusin",()=>m().catch(()=>{}),{once:!0});let p={form:n,available:!0,async submit(){let d=T(n),f=$(d,k),l=j(f,d,c);if(delete n.dataset.result,l)return l.website?(w("pending"),null):(q(n,"invalid",g.invalid||"",l),null);q(n,"pending",c.sending||"");try{await m()}catch(i){return null}try{let i=await s.post(s.pathFor(h),f,{idempotencyKey:E.key(f)}),u=i&&i.data&&typeof i.data.status=="string"?i.data.status:"";return E.clear(),n.reset(),w(u==="confirmed"||u==="subscribed"||u===""&&!v?"confirmed":"pending"),i}catch(i){return i&&i.kind==="conflict"?w("duplicate"):I(n,i,g),null}}};return n.addEventListener("submit",d=>{d.preventDefault(),p.submit()}),p}function U(e=document){return Array.from(e.querySelectorAll("[data-subscribe]")).map(a=>P(a))}var J=["confirm","unsubscribe","manage"],K=/^[A-Za-z0-9._~-]{8,512}$/;function z(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function B(e){let a=new URLSearchParams(e||""),s=J.find(c=>a.has(c));if(!s)return null;let n=(a.get(s)||"").trim();return K.test(n)?{action:s,token:n}:null}function V(e,a={}){let s=e.ownerDocument,n=s.defaultView,c=a.client||C(),g=a.location||n&&n.location||{search:"",pathname:"/"},S=a.history===void 0?n&&n.history:a.history,o=z(e,"[data-manage-labels]"),k=o.errors||{},v=e.querySelector("[data-manage-status]"),E=e.querySelector("[data-manage-idle]"),w=e.querySelector("[data-manage-unsubscribe-panel]"),m=e.querySelector("[data-manage-preferences]"),p=B(g.search),d=p?`${c.pathFor(h)}/${encodeURIComponent(p.token)}`:"",f=null,l=s.createElement("button");l.type="button",l.className="post-tool",l.dataset.manageRetry="",l.textContent=o.retry||"Try again",l.hidden=!0,l.addEventListener("click",()=>f==null?void 0:f()),e.appendChild(l);let i=(r,t="",A=!1)=>{r!=="error"&&(f=null),e.dataset.state=r,l.hidden=r!=="error"||!f,v&&(v.textContent=t,v.hidden=t==="",v.setAttribute("role",A?"alert":"status"))},u=r=>{[E,w,m].forEach(t=>{t&&(t.hidden=t!==r)})},F=(r,t,A)=>{let L=t&&r&&(r.kind==="not_found"||r.status===410);f=r!=null&&r.retryable?A:null,u(null),i("error",L?o.invalid_link||"":x(r,k),!0)},y=r=>{e.setAttribute("aria-busy",r?"true":"false"),e.querySelectorAll("button").forEach(t=>{t.disabled=r})},b={root:e,request:p,async confirm(){let r=!1;y(!0),u(null),i("pending",o.confirming||"");try{await c.feature(h),r=!0;let t=await c.post(`${c.pathFor(h)}/confirm`,{token:p.token});return i("confirmed",o.confirmed||""),t}catch(t){return F(t,r,()=>b.confirm()),null}finally{y(!1)}},async unsubscribe(){let r=!1;y(!0),i("pending",o.unsubscribing||"");try{await c.feature(h),r=!0;let t=await c.delete(d);return u(null),i("unsubscribed",o.unsubscribed||""),t}catch(t){return F(t,r,()=>b.unsubscribe()),null}finally{y(!1)}},async load(){let r=!1;y(!0),u(null),i("loading",o.loading||"");try{await c.feature(h),r=!0;let t=await c.get(d),A=t&&t.data&&Array.isArray(t.data.topics)?t.data.topics.map(String):[];return m&&m.querySelectorAll('[name="topics"]').forEach(L=>{L.checked=A.includes(L.value)}),u(m),i("loaded"),A}catch(t){return F(t,r,()=>b.load()),null}finally{y(!1)}},async save(){let r=Array.from(m.querySelectorAll('[name="topics"]:checked')).map(t=>t.value);y(!0),i("pending",o.saving||"");try{let t=await c.patch(d,{topics:r});return i("saved",o.saved||""),t}catch(t){return f=t!=null&&t.retryable?()=>b.save():null,i("error",x(t,k),!0),null}finally{y(!1)}}};return e.querySelectorAll("[data-manage-unsubscribe]").forEach(r=>{r.addEventListener("click",()=>b.unsubscribe())}),m&&m.addEventListener("submit",r=>{r.preventDefault(),b.save()}),c.enabled?p?(S&&typeof S.replaceState=="function"&&S.replaceState(null,"",g.pathname),p.action==="confirm"?b.confirm():p.action==="manage"?b.load():(u(w),i("idle")),b):(u(E),i("idle"),b):(u(null),i("disabled",k.disabled||""),b)}function D(e=document){return Array.from(e.querySelectorAll("[data-subscription-manage]")).map(a=>V(a))}function _(){U(document),D(document)}document.readyState==="loading"?document.addEventListener("DOMContentLoaded",_,{once:!0}):_();
1
+ import{a as R,b as O,c as T,d as q,e as I,f as U}from"./chunks/chunk-TNVD6UAM.js";import{a as x,c as F}from"./chunks/chunk-VZ5WKQVA.js";import"./chunks/chunk-PATLC23F.js";var h="subscriptions",N=/^[^\s@]+@[^\s@]+\.[^\s@]+$/;function $(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function j(e,a={}){let s={email:e.email||""};return a.hasTopics&&(s.topics=[].concat(e.topics||[]).filter(Boolean)),a.sourceUrl&&(s.source_url=a.sourceUrl),a.locale&&(s.locale=a.locale),s}function P(e,a,s={}){let n={};return N.test(e.email)||(n.email=s.email_invalid||"That does not look like an email address."),Array.isArray(e.topics)&&e.topics.length===0&&(n.topics=s.topics_required||"Pick at least one."),a.website&&(n.website="spam"),Object.keys(n).length>0?n:null}function J(e,a={}){let s=a.client||F(),n=e.querySelector("form"),c=$(e,"[data-subscribe-labels]"),g=c.errors||{},S=e.ownerDocument,o=S.querySelector('link[rel="canonical"]'),k={sourceUrl:e.dataset.subscribeSource||o&&o.href||"",locale:e.dataset.subscribeLocale||S.documentElement.lang||"",hasTopics:n.querySelector('[name="topics"]')!==null},w=e.dataset.subscribeDoubleOptIn!=="false",E=R(n),v=d=>{n.dataset.result=d,q(n,"success",c[d]||"")};if(!s.enabled)return q(n,"disabled",g.disabled||""),{form:n,available:!1,submit:async()=>null};let m=O(s,h,n,g);T(n,()=>m().catch(()=>{}));let p={form:n,available:!0,async submit(){let d=I(n),f=j(d,k),l=P(f,d,c);if(delete n.dataset.result,l)return l.website?(v("pending"),null):(q(n,"invalid",g.invalid||"",l),null);q(n,"pending",c.sending||"");try{await m()}catch(i){return null}try{let i=await s.post(s.pathFor(h),f,{idempotencyKey:E.key(f)}),u=i&&i.data&&typeof i.data.status=="string"?i.data.status:"";return E.clear(),n.reset(),v(u==="confirmed"||u==="subscribed"||u===""&&!w?"confirmed":"pending"),i}catch(i){return i&&i.kind==="conflict"?v("duplicate"):U(n,i,g),null}}};return n.addEventListener("submit",d=>{d.preventDefault(),p.submit()}),p}function D(e=document){return Array.from(e.querySelectorAll("[data-subscribe]")).map(a=>J(a))}var K=["confirm","unsubscribe","manage"],z=/^[A-Za-z0-9._~-]{8,512}$/;function B(e,a){let s=e.querySelector(a);if(!s)return{};try{return JSON.parse(s.textContent)||{}}catch(n){return{}}}function V(e){let a=new URLSearchParams(e||""),s=K.find(c=>a.has(c));if(!s)return null;let n=(a.get(s)||"").trim();return z.test(n)?{action:s,token:n}:null}function Z(e,a={}){let s=e.ownerDocument,n=s.defaultView,c=a.client||F(),g=a.location||n&&n.location||{search:"",pathname:"/"},S=a.history===void 0?n&&n.history:a.history,o=B(e,"[data-manage-labels]"),k=o.errors||{},w=e.querySelector("[data-manage-status]"),E=e.querySelector("[data-manage-idle]"),v=e.querySelector("[data-manage-unsubscribe-panel]"),m=e.querySelector("[data-manage-preferences]"),p=V(g.search),d=p?`${c.pathFor(h)}/${encodeURIComponent(p.token)}`:"",f=null,l=s.createElement("button");l.type="button",l.className="post-tool",l.dataset.manageRetry="",l.textContent=o.retry||"Try again",l.hidden=!0,l.addEventListener("click",()=>f==null?void 0:f()),e.appendChild(l);let i=(r,t="",A=!1)=>{r!=="error"&&(f=null),e.dataset.state=r,l.hidden=r!=="error"||!f,w&&(w.textContent=t,w.hidden=t==="",w.setAttribute("role",A?"alert":"status"))},u=r=>{[E,v,m].forEach(t=>{t&&(t.hidden=t!==r)})},L=(r,t,A)=>{let C=t&&r&&(r.kind==="not_found"||r.status===410);f=r!=null&&r.retryable?A:null,u(null),i("error",C?o.invalid_link||"":x(r,k),!0)},y=r=>{e.setAttribute("aria-busy",r?"true":"false"),e.querySelectorAll("button").forEach(t=>{t.disabled=r})},b={root:e,request:p,async confirm(){let r=!1;y(!0),u(null),i("pending",o.confirming||"");try{await c.feature(h),r=!0;let t=await c.post(`${c.pathFor(h)}/confirm`,{token:p.token});return i("confirmed",o.confirmed||""),t}catch(t){return L(t,r,()=>b.confirm()),null}finally{y(!1)}},async unsubscribe(){let r=!1;y(!0),i("pending",o.unsubscribing||"");try{await c.feature(h),r=!0;let t=await c.delete(d);return u(null),i("unsubscribed",o.unsubscribed||""),t}catch(t){return L(t,r,()=>b.unsubscribe()),null}finally{y(!1)}},async load(){let r=!1;y(!0),u(null),i("loading",o.loading||"");try{await c.feature(h),r=!0;let t=await c.get(d),A=t&&t.data&&Array.isArray(t.data.topics)?t.data.topics.map(String):[];return m&&m.querySelectorAll('[name="topics"]').forEach(C=>{C.checked=A.includes(C.value)}),u(m),i("loaded"),A}catch(t){return L(t,r,()=>b.load()),null}finally{y(!1)}},async save(){let r=Array.from(m.querySelectorAll('[name="topics"]:checked')).map(t=>t.value);y(!0),i("pending",o.saving||"");try{let t=await c.patch(d,{topics:r});return i("saved",o.saved||""),t}catch(t){return f=t!=null&&t.retryable?()=>b.save():null,i("error",x(t,k),!0),null}finally{y(!1)}}};return e.querySelectorAll("[data-manage-unsubscribe]").forEach(r=>{r.addEventListener("click",()=>b.unsubscribe())}),m&&m.addEventListener("submit",r=>{r.preventDefault(),b.save()}),c.enabled?p?(S&&typeof S.replaceState=="function"&&S.replaceState(null,"",g.pathname),p.action==="confirm"?b.confirm():p.action==="manage"?b.load():(u(v),i("idle")),b):(u(E),i("idle"),b):(u(null),i("disabled",k.disabled||""),b)}function _(e=document){return Array.from(e.querySelectorAll("[data-subscription-manage]")).map(a=>Z(a))}function M(){D(document),_(document)}document.readyState==="loading"?document.addEventListener("DOMContentLoaded",M,{once:!0}):M();
data/assets/js/loader.js CHANGED
@@ -74,8 +74,12 @@ const FEATURE_CONFIG = [
74
74
  test: () => document.body?.dataset.featureMath === "true"
75
75
  },
76
76
  {
77
+ // The citation hooks, and the academic dashboard's submission and calendar
78
+ // filters, which a dashboard without citation metrics still has.
77
79
  name: "academic",
78
- test: () => document.querySelector("[data-citation-metric], [data-citation-table], [data-citation-chart]")
80
+ test: () => document.querySelector(
81
+ "[data-citation-metric], [data-citation-table], [data-citation-chart], [data-submission-filter], [data-calendar-filter]"
82
+ )
79
83
  },
80
84
  {
81
85
  name: "notebook",
@@ -88,7 +92,7 @@ const FEATURE_CONFIG = [
88
92
  },
89
93
  {
90
94
  name: "corrections",
91
- test: () => document.querySelector("[data-correction-report]")
95
+ test: () => document.querySelector("[data-correction-report], [data-correction-fallback]")
92
96
  },
93
97
  {
94
98
  name: "contact",
@@ -65,7 +65,7 @@ Gem::Specification.new do |spec|
65
65
  # Each dependency stops below its next major version, so a breaking release
66
66
  # reaches sites through a pull request that updates this file rather than an
67
67
  # untested `bundle update`.
68
- spec.add_runtime_dependency "jekyll", "~> 4.3"
68
+ spec.add_runtime_dependency "jekyll", "~> 4.4"
69
69
  spec.add_runtime_dependency "jekyll-sass-converter", "~> 3.0"
70
70
  spec.add_runtime_dependency "jekyll-feed", "~> 0.16"
71
71
  spec.add_runtime_dependency "jekyll-seo-tag", "~> 2.8"
@@ -0,0 +1,269 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "date"
4
+ require "time"
5
+ require "uri"
6
+
7
+ module Datalog
8
+ class Audit
9
+ # What each check looks for in one file. A check returns [line, message]
10
+ # pairs; Audit names the check, its kind and the file.
11
+ class Checks
12
+ KINDS = %w[theorem lemma proposition corollary definition assumption example remark].freeze
13
+ # **Definition 2.**, __Theorem (Bayes)__, > **Lemma**, *Proof.*
14
+ STATEMENT = /\A\s*(?:>\s*)*(?:[-*+]\s+)?(\*\*|__)(#{(KINDS + ['proof']).join('|')})\b[^*_\n]*\1/i
15
+ PROOF = /\A\s*(?:>\s*)*(\*|_)proof\.?\1/i
16
+ TYPED_NUMBER = /\b(Figure|Fig\.|Table)\s+(\d+)\b/
17
+ REFERENCES = /\A\s{0,3}(?:\#{1,6}\s+|\*\*)
18
+ (References|Bibliography|Works\ cited|Literature\ cited|Sources)(?:\*\*)?[\s\#]*\z/ix
19
+ # Addresses are found with one plain pattern and judged by their host,
20
+ # compared whole: a pattern for the host would also match it inside
21
+ # another address.
22
+ URL = %r{https?://[^\s)<>\]"'`]+}i
23
+ REPOSITORY_HOSTS = %w[github.com gitlab.com bitbucket.org codeberg.org].freeze
24
+ NOTEBOOK_HOSTS = %w[colab.research.google.com mybinder.org nbviewer.org nbviewer.jupyter.org].freeze
25
+ DOI_HOSTS = %w[doi.org dx.doi.org].freeze
26
+ NOTEBOOK_FILE = /\.ipynb\b/i
27
+ BARE_DOI = %r{\bdoi:\s*10\.\d{4,9}/}i
28
+ # Link and image text stops at a bracket, so every [ starts a scan that
29
+ # ends at the next one, and a line of brackets takes linear time.
30
+ MARKDOWN_IMAGE = /!\[([^\[\]\n]*)\]\(\s*<?([^)\s>]+)/
31
+ HTML_IMAGE = /<img\b[^>]*>/i
32
+ IMAGE_FILE = /\.(?:png|jpe?g|gif|svg|webp|avif|bmp|tiff?)\z/i
33
+ MARKDOWN_LINK = %r{(?<!!)\[[^\[\]\n]*\]\(\s*<?(/(?!/)[^)\s>]*)}
34
+ HTML_LINK = %r{\bhref\s*=\s*["'](/(?!/)[^"']*)["']}i
35
+ PART = /\bpart[\s_-]*(\d+|[ivx]+)\b/i
36
+ # A link to a page of the site, and link text that names another part.
37
+ SITE_LINK = %r{\[([^\[\]\n]*)\]\(\s*<?/}
38
+ SEQUENCE_TEXT = /\b(?:part\s+(?:\d+|[ivx]+)|(?:next|previous)\s+(?:part|post|article|chapter))\b/i
39
+
40
+ def initialize(reader:, known_keys:, settings:)
41
+ @reader = reader
42
+ @known_keys = known_keys
43
+ @settings = settings
44
+ end
45
+
46
+ # --------------------------------------------------------- opportunities
47
+
48
+ def statements(file)
49
+ file.each_line.filter_map do |line, number|
50
+ match = line.match(STATEMENT) || line.match(PROOF)
51
+ next unless match
52
+
53
+ kind = (match[2] || "proof").downcase
54
+ [number, "#{kind.capitalize} typed by hand (#{match[0].strip}): {% #{kind} %} numbers it and " \
55
+ "{% ref %} links to it"]
56
+ end
57
+ end
58
+
59
+ def figures(file)
60
+ file.each_line.filter_map do |line, number|
61
+ match = line.match(TYPED_NUMBER)
62
+ next unless match
63
+
64
+ tag = match[1] == "Table" ? "table" : "figure"
65
+ [number, "\"#{match[0]}\" typed by hand: {% #{tag} %} numbers it and {% ref %} keeps the " \
66
+ "number right when the order changes"]
67
+ end
68
+ end
69
+
70
+ def references(file)
71
+ return [] if file.body.include?("{% cite") || file.body.include?("{%- cite")
72
+
73
+ file.each_line.filter_map do |line, number|
74
+ next unless line.match?(REFERENCES)
75
+
76
+ [number, "a references section written by hand: {% cite %} from a BibTeX file numbers the " \
77
+ "citations and lists the works they cite"]
78
+ end.first(1)
79
+ end
80
+
81
+ def reproducibility(file)
82
+ return [] if !file.post? || file.front_matter.key?("reproducibility")
83
+
84
+ file.each_line do |line, number|
85
+ found = research_link(line)
86
+ if found
87
+ return [[number, "links #{found} but has no reproducibility: block, which sets out the code, " \
88
+ "data and environment behind the article"]]
89
+ end
90
+ end
91
+ []
92
+ end
93
+
94
+ def revisions(file)
95
+ data = file.front_matter
96
+ return [] unless file.post? && !data.key?("revisions")
97
+
98
+ published = to_time(data["date"]) || date_from_name(file.relative)
99
+ return [] unless published
100
+
101
+ key = %w[last_modified_at updated].find { |name| data[name] }
102
+ edited = key ? to_time(data[key]) : @reader.git_dates[file.relative]
103
+ return [] unless edited && edited - published > revision_gap
104
+
105
+ source = key ? "`#{key}`" : "its last commit"
106
+ [[key ? file.key_line(key) : 1,
107
+ "edited #{(edited.to_date - published.to_date).to_i} days after publication (#{source}) with no " \
108
+ "revisions: entry, which tells readers what changed"]]
109
+ end
110
+
111
+ # Posts whose titles or file names say "Part 2", or that link to the next
112
+ # part by hand, and are in no series.
113
+ def series(file)
114
+ return [] if file.front_matter.key?("series") || !file.post?
115
+
116
+ found = []
117
+ title = file.front_matter["title"].to_s
118
+ part = title[PART] || File.basename(file.relative)[PART]
119
+ if part
120
+ where = title[PART] ? "title" : "file name"
121
+ found << [file.key_line("title"), "\"#{part}\" in the #{where} and no series: series: numbers the " \
122
+ "parts and links each to the next"]
123
+ end
124
+ file.each_line do |line, number|
125
+ link = line.scan(SITE_LINK).flatten.find { |text| text.match?(SEQUENCE_TEXT) }
126
+ found << [number, "links to \"#{link}\" by hand: series: links the parts in order"] if link
127
+ end
128
+ found.first(2)
129
+ end
130
+
131
+ # -------------------------------------------------------------- problems
132
+
133
+ # Fields other themes read and DataLog does not, named so the finding
134
+ # says where a key came from.
135
+ FOREIGN = {
136
+ "Minimal Mistakes" => %w[author_profile seo_type read_time share related sidebar toc_sticky toc_label
137
+ toc_icon header_image],
138
+ "Just the Docs" => %w[nav_exclude nav_order parent grand_parent has_children has_toc search_exclude],
139
+ "Chirpy" => %w[pin math_rendering mermaid media_subpath]
140
+ }.freeze
141
+
142
+ def front_matter(file)
143
+ file.front_matter.keys.reject { |key| @known_keys.include?(key.to_s) }.map do |key|
144
+ theme = FOREIGN.find { |_, keys| keys.include?(key.to_s) }&.first
145
+ reason = theme ? "a #{theme} field DataLog ignores" : "a typo, or a field of another theme"
146
+ [file.key_line(key), "front matter key \"#{key}\" is read by no layout, include or plugin: #{reason} " \
147
+ "(audit.known_keys lists a site's own)"]
148
+ end
149
+ end
150
+
151
+ def images(file)
152
+ found = []
153
+ file.each_line do |line, number|
154
+ line.scan(MARKDOWN_IMAGE) do |alt, src|
155
+ problem = alt_problem(alt, src)
156
+ found << [number, "image #{src} #{problem}"] if problem
157
+ end
158
+ line.scan(HTML_IMAGE) do
159
+ tag = Regexp.last_match(0)
160
+ src = tag[/\bsrc\s*=\s*["']([^"']*)["']/i, 1].to_s
161
+ alt = tag[/\balt\s*=\s*["']([^"']*)["']/i, 1]
162
+ problem = alt.nil? ? "has no alt attribute" : (alt_problem(alt, src) unless alt.empty?)
163
+ found << [number, "image #{src} #{problem}"] if problem
164
+ end
165
+ end
166
+ found
167
+ end
168
+
169
+ def links(file)
170
+ found = []
171
+ file.each_line do |line, number|
172
+ (line.scan(MARKDOWN_LINK) + line.scan(HTML_LINK)).flatten.each do |target|
173
+ path = strip_baseurl(target)
174
+ next if path.include?("{{") || ignored_link?(path) || @reader.built?(path)
175
+
176
+ found << [number, "links #{target}, which the site does not build"]
177
+ end
178
+ end
179
+ found
180
+ end
181
+
182
+ # Posts only: a page such as the search page loads the engine for math
183
+ # its script fetches, which the page's own source cannot show.
184
+ def math(file)
185
+ data = file.front_matter
186
+ return [] unless file.post?
187
+
188
+ key = data.key?("math") ? "math" : ("mathjax" if data.key?("mathjax"))
189
+ return [] unless key && [true, false].include?(data[key])
190
+
191
+ found = detect_math(file.body)
192
+ if data[key] == true && !found
193
+ [[file.key_line(key), "#{key}: true loads the math engine, but the page has no math it can find"]]
194
+ elsif data[key] == false && found
195
+ [[file.key_line(key), "#{key}: false leaves the page's math (#{found}) as LaTeX source"]]
196
+ else
197
+ []
198
+ end
199
+ end
200
+
201
+ private
202
+
203
+ # The first address on the line that is a repository, a notebook or a DOI,
204
+ # or a notebook file or DOI written without an address.
205
+ def research_link(line)
206
+ line.scan(URL).find { |url| research_address?(url) } ||
207
+ line.split.find { |word| word.match?(NOTEBOOK_FILE) } || line[BARE_DOI]
208
+ end
209
+
210
+ def research_address?(url)
211
+ uri = URI.parse(url)
212
+ host = uri.host.to_s.downcase.delete_prefix("www.")
213
+ segments = uri.path.to_s.split("/").reject(&:empty?)
214
+ return segments.size >= 2 if REPOSITORY_HOSTS.include?(host)
215
+ return segments.first.to_s.start_with?("10.") if DOI_HOSTS.include?(host)
216
+
217
+ NOTEBOOK_HOSTS.include?(host) || uri.path.to_s.match?(NOTEBOOK_FILE)
218
+ rescue URI::InvalidURIError
219
+ false
220
+ end
221
+
222
+ def alt_problem(alt, src)
223
+ text = alt.to_s.strip
224
+ return "has no alt text" if text.empty?
225
+
226
+ name = File.basename(src.to_s.split(/[?#]/).first.to_s)
227
+ "has its file name as alt text" if text.match?(IMAGE_FILE) || text == name || text == File.basename(name, ".*")
228
+ end
229
+
230
+ def strip_baseurl(target)
231
+ base = @reader.config["baseurl"].to_s.chomp("/")
232
+ base.empty? || !target.start_with?("#{base}/") ? target : target.delete_prefix(base)
233
+ end
234
+
235
+ def ignored_link?(path)
236
+ Array(@settings["ignore_links"]).any? { |prefix| path.start_with?(prefix.to_s) }
237
+ end
238
+
239
+ # The first expression the math preprocessor would render, or nil.
240
+ def detect_math(body)
241
+ return nil unless defined?(::MathPreprocessor::Processor)
242
+
243
+ processor = ::MathPreprocessor::Processor.new(body)
244
+ processor.process
245
+ expression = processor.expressions.first
246
+ expression && expression["latex"].to_s.strip[0, 40]
247
+ end
248
+
249
+ def revision_gap
250
+ (@settings["revision_after_days"] || 30).to_i * 86_400
251
+ end
252
+
253
+ def to_time(value)
254
+ case value
255
+ when Time then value
256
+ when Date then value.to_time
257
+ when String then Time.parse(value)
258
+ end
259
+ rescue ArgumentError
260
+ nil
261
+ end
262
+
263
+ def date_from_name(relative)
264
+ stamp = File.basename(relative)[/\A\d{4}-\d{2}-\d{2}/]
265
+ stamp && Time.parse(stamp)
266
+ end
267
+ end
268
+ end
269
+ end
@@ -0,0 +1,61 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Datalog
4
+ class Audit
5
+ # The front matter keys something reads, found in the sources that read
6
+ # them rather than kept as a list: the theme's layouts, includes, plugins
7
+ # and library, and the site's own layouts, includes and plugins. A key
8
+ # none of them names is a typo (`descripton`) or a leftover of another
9
+ # theme, which nothing will ever show.
10
+ module KnownKeys
11
+ module_function
12
+
13
+ # Keys that Jekyll and the plugins the theme depends on read, which the
14
+ # theme's own sources do not name.
15
+ JEKYLL = %w[
16
+ layout permalink published date categories category tags tag title excerpt excerpt_separator slug
17
+ lang render_with_liquid collection output path url id ext draft hidden
18
+ sitemap redirect_from redirect_to last_modified_at image seo canonical_url locale author description
19
+ paginate pagination feed
20
+ ].freeze
21
+
22
+ TEMPLATE_DIRS = %w[_layouts _includes].freeze
23
+ CODE_DIRS = %w[_plugins lib].freeze
24
+ # page.title, include.page.summary, post.series and every other property
25
+ # a template reads, and quoted names in filters (where: "layout").
26
+ TEMPLATE_KEY = /[\w\]]\.([a-z_][a-z0-9_]*)|["']([a-z_][a-z0-9_]*)["']/
27
+ # data["reproducibility"], fetch("title"), Authors.value(page, "series").
28
+ CODE_KEY = /["']([a-z_][a-z0-9_]*)["']/
29
+
30
+ def build(theme_root:, site_root:, extra: [])
31
+ keys = Set.new(JEKYLL)
32
+ [theme_root, site_root].uniq.each do |root|
33
+ scan(root, TEMPLATE_DIRS, "**/*.{html,liquid,xml,json,md}", TEMPLATE_KEY, keys)
34
+ scan(root, CODE_DIRS, "**/*.rb", CODE_KEY, keys)
35
+ end
36
+ keys.merge(Array(extra).map(&:to_s))
37
+ end
38
+
39
+ def scan(root, dirs, glob, pattern, keys)
40
+ dirs.each do |dir|
41
+ Dir.glob(File.join(root, dir, glob)).each do |file|
42
+ File.read(file, encoding: "utf-8").scan(pattern) { |match| keys.merge(Array(match).compact) }
43
+ end
44
+ end
45
+ end
46
+
47
+ # Liquid output and tags; neither holds a brace, so a scan from one {{
48
+ # stops at the next brace and the file is read in linear time.
49
+ LIQUID = /\{\{[^{}]*\}\}|\{%(?:[^{}%]|%(?!\}))*%\}/
50
+
51
+ # The properties a page's own Liquid reads, such as a showcase page that
52
+ # passes `page.academic_demo` to an include, or a listing that shows
53
+ # `post.subtitle`: read by the site's content rather than its templates.
54
+ def from_content(text)
55
+ text.scan(LIQUID).each_with_object(Set.new) do |liquid, keys|
56
+ liquid.scan(TEMPLATE_KEY) { |match| keys.merge(Array(match).compact) }
57
+ end
58
+ end
59
+ end
60
+ end
61
+ end
@@ -0,0 +1,118 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "open3"
4
+ require "time"
5
+ require "tmpdir"
6
+
7
+ module Datalog
8
+ class Audit
9
+ # What Jekyll would build, found without building it: the site's own
10
+ # content files and every URL the build has, read with Jekyll's reader and
11
+ # the generators that only add pages in memory. Nothing is rendered or
12
+ # written; the destination is a directory that never exists.
13
+ class SiteReader
14
+ # The generators that add pages (feeds, the sitemap, pagination pages,
15
+ # redirects, the search page, notebook pages) and write nothing until the
16
+ # site is written. The others are left out: they fetch data, encode
17
+ # images or fill caches.
18
+ PAGE_GENERATORS = %w[
19
+ JekyllFeed::Generator Jekyll::JekyllSitemap Jekyll::Paginate::Pagination
20
+ JekyllRedirectFrom::Generator Datalog::SearchPages Jekyll::NotebookConverter
21
+ ].freeze
22
+ CONTENT = /\.(md|markdown|html?)\z/i
23
+
24
+ attr_reader :root, :site
25
+
26
+ def initialize(root)
27
+ @root = File.expand_path(root)
28
+ end
29
+
30
+ def read
31
+ require "jekyll"
32
+ Jekyll.logger.log_level = :error
33
+ config = Jekyll.configuration(
34
+ "source" => root, "destination" => File.join(Dir.tmpdir, "datalog-audit-#{Process.pid}", "_site"),
35
+ "quiet" => true, "disable_disk_cache" => true
36
+ )
37
+ @site = Jekyll::Site.new(config)
38
+ @site.reset
39
+ @site.read
40
+ @site.generators.each do |generator|
41
+ generator.generate(@site) if PAGE_GENERATORS.include?(generator.class.name)
42
+ end
43
+ self
44
+ end
45
+
46
+ def config
47
+ site.config
48
+ end
49
+
50
+ # [absolute path, path from the site root] of every page and document the
51
+ # site's own sources hold, in a stable order.
52
+ def content_files
53
+ documents = site.collections.values.flat_map(&:docs)
54
+ pages = site.pages.select { |page| page.instance_of?(Jekyll::Page) }
55
+ files = (documents + pages).filter_map do |item|
56
+ path = File.expand_path(item.relative_path, root)
57
+ next unless path.start_with?("#{root}/") && File.file?(path) && path.match?(CONTENT)
58
+
59
+ [path, path.delete_prefix("#{root}/")]
60
+ end
61
+ files.uniq.sort_by(&:last)
62
+ end
63
+
64
+ # Every URL the build writes, normalised: pages, documents, generated
65
+ # pages and static files.
66
+ def urls
67
+ @urls ||= begin
68
+ items = site.pages + site.static_files + site.collections.values.flat_map(&:docs).select(&:write?)
69
+ items.to_set { |item| normalize(item.url) }
70
+ end
71
+ end
72
+
73
+ def normalize(url)
74
+ path = url.to_s.split(/[?#]/, 2).first.to_s
75
+ path = "/#{path}" unless path.start_with?("/")
76
+ path = path.sub(%r{/index\.html?\z}, "/")
77
+ path.end_with?("/") || path.length == 1 ? path : path.sub(/\.html?\z/, "")
78
+ end
79
+
80
+ def built?(url)
81
+ path = normalize(url)
82
+ [path, "#{path}/", path.delete_suffix("/")].any? { |candidate| urls.include?(candidate) }
83
+ end
84
+
85
+ # A commit that touches more files than this is a sweeping change (a
86
+ # reformat, a migration, a rename), not an edit of any one article.
87
+ SWEEPING_COMMIT = 10
88
+
89
+ # The date of the last commit to edit each file, from one `git log` over
90
+ # the site, or {} outside a repository.
91
+ def git_dates
92
+ @git_dates ||= begin
93
+ output, status = Open3.capture2("git", "-C", root, "log", "--format=@@date@@%cI", "--name-only", "--relative",
94
+ "--", ".", err: File::NULL)
95
+ status.success? ? parse_log(output) : {}
96
+ rescue SystemCallError
97
+ {}
98
+ end
99
+ end
100
+
101
+ private
102
+
103
+ def parse_log(output)
104
+ dates = {}
105
+ output.split("@@date@@").each do |commit|
106
+ stamp, *paths = commit.split("\n").map(&:strip).reject(&:empty?)
107
+ next if stamp.nil? || paths.size > SWEEPING_COMMIT
108
+
109
+ date = Time.iso8601(stamp)
110
+ paths.each { |path| dates[path] ||= date }
111
+ rescue ArgumentError
112
+ next
113
+ end
114
+ dates
115
+ end
116
+ end
117
+ end
118
+ end
@@ -0,0 +1,85 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "date"
4
+ require "yaml"
5
+
6
+ module Datalog
7
+ class Audit
8
+ # One content file as its author wrote it: the front matter keys with the
9
+ # line each is on, and the body with its code set aside, line by line, so
10
+ # a finding names the line the author would edit.
11
+ class SourceFile
12
+ # Code shows what it shows: nothing inside it is a statement, a figure
13
+ # reference, a link or an image. Each match is blanked character for
14
+ # character, newlines kept, so line numbers do not move.
15
+ CODE = [
16
+ /^[ \t]*(`{3,}|~{3,})[^\n]*\n.*?(?:^[ \t]*\1[ \t]*$|\z)/m,
17
+ /\{%-?\s*(highlight|raw)\b.*?\{%-?\s*end\1\s*-?%\}/m,
18
+ %r{<(pre|code)\b[^>]*>.*?</\1>}mi,
19
+ /`[^`\n]+`/,
20
+ /<!--.*?-->/m,
21
+ /\{%-?\s*comment\s*-?%\}.*?\{%-?\s*endcomment\s*-?%\}/m
22
+ ].freeze
23
+
24
+ attr_reader :path, :relative, :front_matter, :key_lines, :body_start, :body, :lines
25
+
26
+ def initialize(path, relative)
27
+ @path = path
28
+ @relative = relative
29
+ text = File.read(path, encoding: "bom|utf-8")
30
+ @front_matter, @key_lines, @body_start, @body = split(text)
31
+ @lines = mask(@body).split("\n", -1)
32
+ end
33
+
34
+ # The body's lines with code blanked, each with its line in the file.
35
+ def each_line
36
+ return enum_for(:each_line) unless block_given?
37
+
38
+ @lines.each_with_index { |line, index| yield line, @body_start + index }
39
+ end
40
+
41
+ # The line in the file of the body's `offset`th character.
42
+ def line_at(offset)
43
+ @body_start + masked_body[0, offset].count("\n")
44
+ end
45
+
46
+ def masked_body
47
+ @masked_body ||= @lines.join("\n")
48
+ end
49
+
50
+ def key_line(key)
51
+ @key_lines.fetch(key, 1)
52
+ end
53
+
54
+ def post?
55
+ relative.match?(%r{(\A|/)_posts/})
56
+ end
57
+
58
+ private
59
+
60
+ def split(text)
61
+ match = text.match(/\A---[ \t]*\r?\n(.*?)^---[ \t]*\r?$\n?/m)
62
+ return [{}, {}, 1, text] unless match
63
+
64
+ yaml = match[1]
65
+ data = begin
66
+ YAML.safe_load(yaml, permitted_classes: [Date, Time], aliases: true)
67
+ rescue Psych::Exception
68
+ nil
69
+ end
70
+ key_lines = {}
71
+ yaml.each_line.with_index(2) do |line, number|
72
+ key = line[/\A([A-Za-z_][\w-]*)\s*:/, 1]
73
+ key_lines[key] ||= number if key
74
+ end
75
+ [data.is_a?(Hash) ? data : {}, key_lines, match[0].count("\n") + 1, match.post_match]
76
+ end
77
+
78
+ def mask(text)
79
+ CODE.reduce(text) do |masked, pattern|
80
+ masked.gsub(pattern) { |code| code.gsub(/[^\n]/, " ") }
81
+ end
82
+ end
83
+ end
84
+ end
85
+ end