domma-cms 0.88.1 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -292,10 +292,16 @@ export async function renderPage(page, opts = {}) {
292
292
  }
293
293
  const pageBodyStyle = pageBodyStyleParts.join(';');
294
294
 
295
- const seoTitle = escapeHtml(page.seo?.title || `${page.title}${site.seo?.titleSeparator || ' | '}${site.seo?.defaultTitle || site.title}`);
296
- const seoDescription = escapeHtml(page.seo?.description || site.seo?.defaultDescription || '');
297
- const ogImage = escapeHtml(page.seo?.image || site.seo?.defaultImage || '');
298
- const seoTags = buildSeoTags({page, site, baseUrl, seoTitle, seoDescription, ogImage});
295
+ const seoMeta = transformSeoMeta(buildSeoMeta({
296
+ page, site, baseUrl,
297
+ title: page.seo?.title || `${page.title}${site.seo?.titleSeparator || ' | '}${site.seo?.defaultTitle || site.title}`,
298
+ description: page.seo?.description || site.seo?.defaultDescription || '',
299
+ image: page.seo?.image || site.seo?.defaultImage || ''
300
+ }), {page, site, baseUrl, kind: 'page'});
301
+ const seoTitle = escapeHtml(seoMeta.title);
302
+ const seoDescription = escapeHtml(seoMeta.description);
303
+ const ogImage = escapeHtml(seoMeta.og?.image || seoMeta.image || '');
304
+ const seoTags = renderSeoTags(seoMeta);
299
305
 
300
306
  const dconfig = page.dconfig || null;
301
307
  // Escape </script> to prevent injection via dconfig values in the inline script block
@@ -510,63 +516,43 @@ function toTitleCase(str) {
510
516
  }
511
517
 
512
518
  /**
513
- * Build the consolidated SEO `<meta>` tag block: canonical link, Open Graph,
514
- * Twitter cards, JSON-LD, and a noindex hint when requested.
519
+ * The SEO facts of one render, raw (not escaped): what the `<title>`, the
520
+ * description and the tag block are built from. Plugins change it through the
521
+ * 'seo:meta' transform (0.89) - the SEO plugin's overrides, JSON-LD presets and
522
+ * social variants - before anything is escaped.
515
523
  *
516
- * All values are HTML-escaped at the call site (seoTitle/Description/ogImage)
517
- * or here (canonicalUrl). JSON-LD is JSON.stringify'd and `</script>` is
518
- * escaped to prevent breaking out of the script context.
524
+ * Shape: {title, description, image, ogType, noindex, canonical, siteName,
525
+ * twitterSite, og: {title?, description?, image?}, twitter: {title?,
526
+ * description?, image?, card?}, robots?, jsonLd: object[], extra: [{tag:
527
+ * 'meta'|'link', attrs}]}
519
528
  *
520
529
  * @param {object} args
521
530
  * @param {object} args.page - Parsed page (urlPath, seo, title)
522
531
  * @param {object} args.site - Site config (seo defaults)
523
- * @param {string} args.baseUrl - Absolute origin for this request, derived by
524
- * the route handler. Pass empty string to skip canonical/og:url emission.
525
- * @param {string} args.seoTitle - Already HTML-escaped
526
- * @param {string} args.seoDescription - Already HTML-escaped
527
- * @param {string} args.ogImage - Already HTML-escaped (may be empty)
528
- * @returns {string}
532
+ * @param {string} args.baseUrl - Absolute origin for this request. Empty skips
533
+ * canonical/og:url and makes no URL absolute.
534
+ * @param {string} args.title - The document title (raw)
535
+ * @param {string} args.description - Raw
536
+ * @param {string} args.image - Raw, may be empty
537
+ * @returns {object}
529
538
  */
530
- function buildSeoTags({page, site, baseUrl, seoTitle, seoDescription, ogImage}) {
539
+ function buildSeoMeta({page, site, baseUrl, title, description, image}) {
531
540
  const origin = (baseUrl || '').toString().trim().replace(/\/+$/, '');
532
541
  const urlPath = page.urlPath || '/';
533
- const canonicalUrl = origin ? escapeHtml(`${origin}${urlPath}`) : '';
534
- const ogType = escapeHtml(page.seo?.ogType || 'article');
535
- const siteName = escapeHtml(site.title || '');
536
- const twitterHandle = escapeHtml(site.social?.twitter || '');
537
- const noindex = page.seo?.noindex === true;
538
-
539
- const tags = [];
540
- if (canonicalUrl) tags.push(`<link rel="canonical" href="${canonicalUrl}">`);
541
- if (noindex) tags.push('<meta name="robots" content="noindex, nofollow">');
542
-
543
- tags.push(`<meta property="og:title" content="${seoTitle}">`);
544
- tags.push(`<meta property="og:description" content="${seoDescription}">`);
545
- tags.push(`<meta property="og:type" content="${ogType}">`);
546
- if (siteName) tags.push(`<meta property="og:site_name" content="${siteName}">`);
547
- if (canonicalUrl) tags.push(`<meta property="og:url" content="${canonicalUrl}">`);
548
- if (ogImage) tags.push(`<meta property="og:image" content="${ogImage}">`);
549
-
550
- tags.push(`<meta name="twitter:card" content="${ogImage ? 'summary_large_image' : 'summary'}">`);
551
- tags.push(`<meta name="twitter:title" content="${seoTitle}">`);
552
- tags.push(`<meta name="twitter:description" content="${seoDescription}">`);
553
- if (ogImage) tags.push(`<meta name="twitter:image" content="${ogImage}">`);
554
- if (twitterHandle) tags.push(`<meta name="twitter:site" content="${twitterHandle}">`);
542
+ const canonical = origin ? `${origin}${urlPath}` : '';
543
+ const absImage = absoluteUrl(image, origin);
555
544
 
556
545
  // JSON-LD WebPage schema - gives search engines a structured summary.
557
546
  // Omit `url` when no baseUrl, since a relative URL in JSON-LD is invalid.
558
- const jsonLd = {
547
+ const webPage = {
559
548
  '@context': 'https://schema.org',
560
549
  '@type': 'WebPage',
561
550
  name: page.title || '',
562
551
  description: page.seo?.description || site.seo?.defaultDescription || ''
563
552
  };
564
- if (canonicalUrl) jsonLd.url = origin + urlPath;
565
- if (page.seo?.image || site.seo?.defaultImage) {
566
- jsonLd.image = page.seo?.image || site.seo?.defaultImage;
567
- }
568
- const jsonLdString = JSON.stringify(jsonLd).replace(/<\/script>/gi, '<\\/script>');
569
- tags.push(`<script type="application/ld+json">${jsonLdString}</script>`);
553
+ if (canonical) webPage.url = canonical;
554
+ if (absImage) webPage.image = absImage;
555
+ const jsonLd = [webPage];
570
556
 
571
557
  // BreadcrumbList JSON-LD - emit whenever the page has a non-trivial path
572
558
  // and an origin is known (Schema.org requires absolute URLs in `item`).
@@ -574,7 +560,7 @@ function buildSeoTags({page, site, baseUrl, seoTitle, seoDescription, ogImage})
574
560
  // even when the on-page nav hides it.
575
561
  const breadcrumbItems = buildBreadcrumbItems(page, site);
576
562
  if (origin && breadcrumbItems.length > 1) {
577
- const breadcrumbJsonLd = {
563
+ jsonLd.push({
578
564
  '@context': 'https://schema.org',
579
565
  '@type': 'BreadcrumbList',
580
566
  itemListElement: breadcrumbItems.map((item, i) => ({
@@ -583,9 +569,96 @@ function buildSeoTags({page, site, baseUrl, seoTitle, seoDescription, ogImage})
583
569
  name: item.name,
584
570
  item: origin + item.urlPath
585
571
  }))
586
- };
587
- const breadcrumbString = JSON.stringify(breadcrumbJsonLd).replace(/<\/script>/gi, '<\\/script>');
588
- tags.push(`<script type="application/ld+json">${breadcrumbString}</script>`);
572
+ });
573
+ }
574
+
575
+ return {
576
+ title,
577
+ description,
578
+ image: absImage,
579
+ ogType: page.seo?.ogType || 'article',
580
+ noindex: page.seo?.noindex === true,
581
+ canonical,
582
+ siteName: site.title || '',
583
+ twitterSite: site.social?.twitter || '',
584
+ og: {},
585
+ twitter: {},
586
+ jsonLd,
587
+ extra: []
588
+ };
589
+ }
590
+
591
+ /** An absolute URL for og:image and JSON-LD: crawlers ignore relative ones. */
592
+ function absoluteUrl(url, origin) {
593
+ const u = String(url || '').trim();
594
+ if (!u || !origin || /^[a-z][a-z0-9+.-]*:/i.test(u) || u.startsWith('//')) return u;
595
+ return `${origin}${u.startsWith('/') ? '' : '/'}${u}`;
596
+ }
597
+
598
+ /**
599
+ * Run the 'seo:meta' transforms. A transform that throws is logged and
600
+ * skipped: SEO extras never take a page down.
601
+ */
602
+ function transformSeoMeta(meta, context) {
603
+ try {
604
+ const out = applyTransforms('seo:meta', meta, context);
605
+ return out && typeof out === 'object' ? out : meta;
606
+ } catch (err) {
607
+ console.warn(`[seo] a seo:meta transform failed, using the page's own: ${err.message}`);
608
+ return meta;
609
+ }
610
+ }
611
+
612
+ const SAFE_ATTR = /^[a-z][a-z0-9:_-]*$/i;
613
+
614
+ /**
615
+ * Build the consolidated SEO `<meta>` tag block from buildSeoMeta()'s model:
616
+ * canonical link, Open Graph, Twitter cards, JSON-LD, and a robots hint.
617
+ *
618
+ * Every value is escaped here. JSON-LD is JSON.stringify'd with `<` escaped,
619
+ * so no `</script>` or `<!--` can break out of the script context.
620
+ *
621
+ * @param {object} meta
622
+ * @returns {string}
623
+ */
624
+ function renderSeoTags(meta) {
625
+ const e = escapeHtml;
626
+ const ogTitle = meta.og?.title || meta.title;
627
+ const ogDescription = meta.og?.description ?? meta.description;
628
+ const ogImage = meta.og?.image || meta.image;
629
+ const twTitle = meta.twitter?.title || ogTitle;
630
+ const twDescription = meta.twitter?.description ?? ogDescription;
631
+ const twImage = meta.twitter?.image || ogImage;
632
+ const robots = meta.robots || (meta.noindex ? 'noindex, nofollow' : '');
633
+
634
+ const tags = [];
635
+ if (meta.canonical) tags.push(`<link rel="canonical" href="${e(meta.canonical)}">`);
636
+ if (robots) tags.push(`<meta name="robots" content="${e(robots)}">`);
637
+
638
+ tags.push(`<meta property="og:title" content="${e(ogTitle)}">`);
639
+ tags.push(`<meta property="og:description" content="${e(ogDescription)}">`);
640
+ tags.push(`<meta property="og:type" content="${e(meta.ogType)}">`);
641
+ if (meta.siteName) tags.push(`<meta property="og:site_name" content="${e(meta.siteName)}">`);
642
+ if (meta.canonical) tags.push(`<meta property="og:url" content="${e(meta.canonical)}">`);
643
+ if (ogImage) tags.push(`<meta property="og:image" content="${e(ogImage)}">`);
644
+
645
+ tags.push(`<meta name="twitter:card" content="${e(meta.twitter?.card || (twImage ? 'summary_large_image' : 'summary'))}">`);
646
+ tags.push(`<meta name="twitter:title" content="${e(twTitle)}">`);
647
+ tags.push(`<meta name="twitter:description" content="${e(twDescription)}">`);
648
+ if (twImage) tags.push(`<meta name="twitter:image" content="${e(twImage)}">`);
649
+ if (meta.twitterSite) tags.push(`<meta name="twitter:site" content="${e(meta.twitterSite)}">`);
650
+
651
+ for (const x of Array.isArray(meta.extra) ? meta.extra : []) {
652
+ if (!x || (x.tag !== 'meta' && x.tag !== 'link') || !x.attrs || typeof x.attrs !== 'object') continue;
653
+ const attrs = Object.entries(x.attrs).filter(([k, v]) => SAFE_ATTR.test(k) && v !== undefined && v !== null)
654
+ .map(([k, v]) => `${k}="${e(v)}"`).join(' ');
655
+ if (attrs) tags.push(`<${x.tag} ${attrs}>`);
656
+ }
657
+
658
+ for (const block of Array.isArray(meta.jsonLd) ? meta.jsonLd : []) {
659
+ if (!block || typeof block !== 'object') continue;
660
+ const json = JSON.stringify(block).replace(/</g, '\\u003c');
661
+ tags.push(`<script type="application/ld+json">${json}</script>`);
589
662
  }
590
663
 
591
664
  return tags.join('\n ');
@@ -926,11 +999,9 @@ export async function renderBlogPage(templatePath, data = {}, seoMeta = {}) {
926
999
  return String(val);
927
1000
  });
928
1001
 
929
- const seoTitle = escapeHtml(seoMeta.title ?? site.seo?.defaultTitle ?? site.title ?? 'Blog');
930
- const seoDescription = escapeHtml(seoMeta.description ?? site.seo?.defaultDescription ?? '');
931
- const ogImage = escapeHtml(seoMeta.ogImage ?? '');
932
-
933
1002
  // Plugin fragments use the unified SEO builder with a synthetic page object.
1003
+ // `item` ({type, id}) lets the 'seo:meta' transform know WHAT is being
1004
+ // shown, not just where - optional; the SEO plugin keys its overrides by URL.
934
1005
  const syntheticPage = {
935
1006
  urlPath: seoMeta.urlPath || '/',
936
1007
  title: seoMeta.title || site.title || '',
@@ -939,9 +1010,19 @@ export async function renderBlogPage(templatePath, data = {}, seoMeta = {}) {
939
1010
  image: seoMeta.ogImage,
940
1011
  ogType: seoMeta.ogType || 'website',
941
1012
  noindex: seoMeta.noindex === true
942
- }
1013
+ },
1014
+ ...(seoMeta.item && {item: seoMeta.item})
943
1015
  };
944
- const seoTags = buildSeoTags({page: syntheticPage, site, baseUrl, seoTitle, seoDescription, ogImage});
1016
+ const meta = transformSeoMeta(buildSeoMeta({
1017
+ page: syntheticPage, site, baseUrl,
1018
+ title: seoMeta.title ?? site.seo?.defaultTitle ?? site.title ?? 'Blog',
1019
+ description: seoMeta.description ?? site.seo?.defaultDescription ?? '',
1020
+ image: seoMeta.ogImage ?? ''
1021
+ }), {page: syntheticPage, site, baseUrl, kind: 'plugin'});
1022
+ const seoTitle = escapeHtml(meta.title);
1023
+ const seoDescription = escapeHtml(meta.description);
1024
+ const ogImage = escapeHtml(meta.og?.image || meta.image || '');
1025
+ const seoTags = renderSeoTags(meta);
945
1026
 
946
1027
  // Both render paths resolve the theme the same way. Wiring one and not the
947
1028
  // other is the standing trap in this file.
@@ -1,13 +1,18 @@
1
1
  /**
2
2
  * Sitemap Service
3
- * Builds sitemap.xml from the page list. Filter logic lives in
4
- * shouldIncludeInSitemap() - see the TODO below.
3
+ * Builds sitemap.xml from the page list plus every plugin sitemap source
4
+ * (hooks.registerSitemapSource, 0.89). Page filter logic lives in
5
+ * shouldIncludeInSitemap().
5
6
  *
6
7
  * Cache: callers should wrap generate() with the 'sitemap' tag so it
7
8
  * invalidates whenever a page is created, updated, renamed, or deleted.
8
9
  */
9
10
  import {listPages} from './content.js';
10
11
  import {getConfig} from '../config.js';
12
+ import {applyTransforms, getSitemapSources} from './hooks.js';
13
+
14
+ const CHANGEFREQ = new Set(['always', 'hourly', 'daily', 'weekly', 'monthly', 'yearly', 'never']);
15
+ const validPriority = (p) => p !== undefined && p !== null && p !== '' && Number(p) >= 0 && Number(p) <= 1;
11
16
 
12
17
  /**
13
18
  * Decide whether a page should appear in sitemap.xml.
@@ -23,30 +28,84 @@ import {getConfig} from '../config.js';
23
28
  */
24
29
  function shouldIncludeInSitemap(page) {
25
30
  return page.status === 'published'
26
- && page.visibility === 'public'
31
+ && (page.visibility || 'public') === 'public'
27
32
  && page.urlPath !== '/404'
28
33
  && !page.seo?.noindex;
29
34
  }
30
35
 
36
+ /** The sitemap entry of a core page. */
37
+ function pageEntry(page) {
38
+ return {urlPath: page.urlPath, title: page.title || '', lastmod: page.updatedAt || null, source: 'pages'};
39
+ }
40
+
41
+ /**
42
+ * Every URL the sitemap lists: core pages, then each plugin's sitemap source
43
+ * (hooks.registerSitemapSource), de-duplicated by path (the first wins), then
44
+ * the 'sitemap:entries' transform - where the SEO plugin drops or re-weights
45
+ * entries. Plain objects, so the SEO audit can use the same list.
46
+ *
47
+ * @param {string} baseUrl - Absolute origin (trailing slash stripped here)
48
+ * @param {{log?: {warn: Function}}} [opts]
49
+ * @returns {Promise<Array<{urlPath: string, title?: string, lastmod?: string|null,
50
+ * changefreq?: string, priority?: number, image?: string, source: string}>>}
51
+ */
52
+ export async function listSitemapEntries(baseUrl, {log = console} = {}) {
53
+ const normalised = normaliseBaseUrl(baseUrl);
54
+ const pages = await listPages();
55
+ const entries = pages.filter(shouldIncludeInSitemap).map(pageEntry);
56
+ const seen = new Set(entries.map(e => e.urlPath));
57
+
58
+ for (const source of getSitemapSources()) {
59
+ let items = [];
60
+ try {
61
+ items = await source.list({baseUrl: normalised});
62
+ } catch (err) {
63
+ log.warn?.(`[sitemap] source "${source.id}" skipped: ${err.message}`);
64
+ continue;
65
+ }
66
+ for (const item of Array.isArray(items) ? items : []) {
67
+ const urlPath = typeof item?.urlPath === 'string' ? item.urlPath.trim() : '';
68
+ if (!urlPath.startsWith('/') || urlPath.startsWith('//') || seen.has(urlPath)) continue;
69
+ seen.add(urlPath);
70
+ entries.push({
71
+ urlPath,
72
+ title: typeof item.title === 'string' ? item.title : '',
73
+ lastmod: item.lastmod || null,
74
+ ...(CHANGEFREQ.has(item.changefreq) && {changefreq: item.changefreq}),
75
+ ...(validPriority(item.priority) && {priority: Number(item.priority)}),
76
+ ...(typeof item.image === 'string' && item.image && {image: item.image}),
77
+ source: source.id
78
+ });
79
+ }
80
+ }
81
+
82
+ try {
83
+ const out = applyTransforms('sitemap:entries', entries, {baseUrl: normalised});
84
+ return Array.isArray(out) ? out.filter(e => e && typeof e.urlPath === 'string') : entries;
85
+ } catch (err) {
86
+ log.warn?.(`[sitemap] a sitemap:entries transform failed, using the plain list: ${err.message}`);
87
+ return entries;
88
+ }
89
+ }
90
+
31
91
  /**
32
92
  * Build the full sitemap.xml document as a string.
33
93
  *
34
94
  * @param {string} baseUrl - Absolute origin (e.g. 'https://example.com'),
35
95
  * typically derived from the incoming request. Trailing slash is stripped.
96
+ * @param {{log?: {warn: Function}}} [opts]
36
97
  * @returns {Promise<string>}
37
98
  */
38
- export async function generate(baseUrl) {
99
+ export async function generate(baseUrl, opts = {}) {
39
100
  const normalised = normaliseBaseUrl(baseUrl);
40
- const pages = await listPages();
41
- const entries = pages
42
- .filter(shouldIncludeInSitemap)
43
- .map(page => buildUrlEntry(page, normalised));
44
-
45
- return wrapInUrlset(entries);
101
+ const entries = await listSitemapEntries(normalised, opts);
102
+ const withImages = entries.some(e => e.image);
103
+ return wrapInUrlset(entries.map(e => buildUrlEntry(e, normalised)), {withImages});
46
104
  }
47
105
 
48
106
  /**
49
107
  * Build the robots.txt document. Points at /sitemap.xml on the given base URL.
108
+ * The 'robots:lines' transform may change the lines (the SEO plugin's editor).
50
109
  *
51
110
  * @param {string} baseUrl - Absolute origin, typically derived from the
52
111
  * incoming request.
@@ -58,7 +117,7 @@ export function buildRobotsTxt(baseUrl) {
58
117
  const sitemapUrl = normalised ? `${normalised}/sitemap.xml` : '/sitemap.xml';
59
118
  const extra = (site.seo?.robotsExtra || '').toString().trim();
60
119
 
61
- const lines = [
120
+ let lines = [
62
121
  'User-agent: *',
63
122
  'Allow: /',
64
123
  'Disallow: /admin/',
@@ -67,6 +126,12 @@ export function buildRobotsTxt(baseUrl) {
67
126
  `Sitemap: ${sitemapUrl}`
68
127
  ];
69
128
  if (extra) lines.push('', extra);
129
+ try {
130
+ const out = applyTransforms('robots:lines', lines, {baseUrl: normalised, sitemapUrl});
131
+ if (Array.isArray(out)) lines = out.map(l => String(l ?? ''));
132
+ } catch (err) {
133
+ console.warn(`[sitemap] a robots:lines transform failed, using the default: ${err.message}`);
134
+ }
70
135
  return lines.join('\n') + '\n';
71
136
  }
72
137
 
@@ -74,20 +139,26 @@ export function buildRobotsTxt(baseUrl) {
74
139
  // Helpers
75
140
  // ---------------------------------------------------------------------------
76
141
 
77
- function buildUrlEntry(page, baseUrl) {
78
- const loc = escapeXml(baseUrl ? `${baseUrl}${page.urlPath}` : page.urlPath);
79
- const lastmod = page.updatedAt
80
- ? new Date(page.updatedAt).toISOString()
81
- : null;
142
+ function buildUrlEntry(entry, baseUrl) {
143
+ const loc = escapeXml(baseUrl ? `${baseUrl}${entry.urlPath}` : entry.urlPath);
144
+ const when = entry.lastmod ? new Date(entry.lastmod) : null;
145
+ const lastmod = when && !Number.isNaN(when.getTime()) ? when.toISOString() : null;
82
146
 
83
147
  const parts = [`<loc>${loc}</loc>`];
84
148
  if (lastmod) parts.push(`<lastmod>${lastmod}</lastmod>`);
149
+ if (entry.changefreq) parts.push(`<changefreq>${escapeXml(entry.changefreq)}</changefreq>`);
150
+ if (validPriority(entry.priority)) parts.push(`<priority>${Number(entry.priority).toFixed(1)}</priority>`);
151
+ if (entry.image) {
152
+ const img = /^https?:\/\//i.test(entry.image) ? entry.image : `${baseUrl}${entry.image.startsWith('/') ? '' : '/'}${entry.image}`;
153
+ parts.push(`<image:image><image:loc>${escapeXml(img)}</image:loc></image:image>`);
154
+ }
85
155
  return ` <url>\n ${parts.join('\n ')}\n </url>`;
86
156
  }
87
157
 
88
- function wrapInUrlset(entries) {
158
+ function wrapInUrlset(entries, {withImages = false} = {}) {
159
+ const ns = withImages ? ' xmlns:image="http://www.google.com/schemas/sitemap-image/1.1"' : '';
89
160
  return `<?xml version="1.0" encoding="UTF-8"?>
90
- <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
161
+ <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"${ns}>
91
162
  ${entries.join('\n')}
92
163
  </urlset>
93
164
  `;