react-cheminfo 0.35.0 → 0.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/check-seo.mjs +333 -0
- package/lib/ecosystem/core/sites.d.ts.map +1 -1
- package/lib/ecosystem/core/sites.js +17 -0
- package/lib/ecosystem/core/sites.js.map +1 -1
- package/lib/ecosystem/core/types.d.ts +1 -1
- package/lib/ecosystem/core/types.d.ts.map +1 -1
- package/lib/ecosystem/ui/glyphs.d.ts.map +1 -1
- package/lib/ecosystem/ui/glyphs.js +4 -0
- package/lib/ecosystem/ui/glyphs.js.map +1 -1
- package/lib/language/core/index.d.ts +2 -0
- package/lib/language/core/index.d.ts.map +1 -1
- package/lib/language/core/index.js +1 -0
- package/lib/language/core/index.js.map +1 -1
- package/lib/language/core/languagePath.d.ts +52 -0
- package/lib/language/core/languagePath.d.ts.map +1 -0
- package/lib/language/core/languagePath.js +65 -0
- package/lib/language/core/languagePath.js.map +1 -0
- package/lib/seo/core/alternates.d.ts +40 -0
- package/lib/seo/core/alternates.d.ts.map +1 -0
- package/lib/seo/core/alternates.js +42 -0
- package/lib/seo/core/alternates.js.map +1 -0
- package/lib/seo/core/index.d.ts +3 -0
- package/lib/seo/core/index.d.ts.map +1 -1
- package/lib/seo/core/index.js +2 -0
- package/lib/seo/core/index.js.map +1 -1
- package/lib/seo/core/noscript.d.ts +9 -0
- package/lib/seo/core/noscript.d.ts.map +1 -1
- package/lib/seo/core/noscript.js +7 -2
- package/lib/seo/core/noscript.js.map +1 -1
- package/lib/seo/core/pageMeta.d.ts +23 -0
- package/lib/seo/core/pageMeta.d.ts.map +1 -1
- package/lib/seo/core/pageMeta.js +52 -3
- package/lib/seo/core/pageMeta.js.map +1 -1
- package/lib/seo/core/pageProse.d.ts +76 -0
- package/lib/seo/core/pageProse.d.ts.map +1 -0
- package/lib/seo/core/pageProse.js +92 -0
- package/lib/seo/core/pageProse.js.map +1 -0
- package/lib/seo/core/routeProblems.d.ts +33 -0
- package/lib/seo/core/routeProblems.d.ts.map +1 -0
- package/lib/seo/core/routeProblems.js +89 -0
- package/lib/seo/core/routeProblems.js.map +1 -0
- package/lib/seo/core/routes.d.ts +14 -0
- package/lib/seo/core/routes.d.ts.map +1 -1
- package/lib/seo/core/routes.js.map +1 -1
- package/lib/seo/core/siteFiles.d.ts.map +1 -1
- package/lib/seo/core/siteFiles.js +2 -1
- package/lib/seo/core/siteFiles.js.map +1 -1
- package/lib/seo/vite/prerender.d.ts +13 -0
- package/lib/seo/vite/prerender.d.ts.map +1 -1
- package/lib/seo/vite/prerender.js +17 -10
- package/lib/seo/vite/prerender.js.map +1 -1
- package/package.json +2 -1
- package/src/ecosystem/core/sites.ts +17 -0
- package/src/ecosystem/core/types.ts +1 -0
- package/src/ecosystem/ui/glyphs.tsx +16 -0
- package/src/language/core/index.ts +2 -0
- package/src/language/core/languagePath.ts +78 -0
- package/src/seo/core/alternates.ts +65 -0
- package/src/seo/core/index.ts +3 -0
- package/src/seo/core/noscript.ts +18 -2
- package/src/seo/core/pageMeta.ts +74 -3
- package/src/seo/core/pageProse.ts +146 -0
- package/src/seo/core/routeProblems.ts +111 -0
- package/src/seo/core/routes.ts +14 -0
- package/src/seo/core/siteFiles.ts +2 -1
- package/src/seo/vite/prerender.ts +30 -11
package/src/seo/core/pageMeta.ts
CHANGED
|
@@ -14,14 +14,26 @@
|
|
|
14
14
|
|
|
15
15
|
import { siteDisplayName } from '../../ecosystem/core/lookup.ts';
|
|
16
16
|
import type { SiteId, SiteRecord } from '../../ecosystem/core/sites.ts';
|
|
17
|
+
import type { Language } from '../../i18n/core/languages.ts';
|
|
18
|
+
import { DEFAULT_LANGUAGE } from '../../i18n/core/languages.ts';
|
|
19
|
+
import {
|
|
20
|
+
readLanguagePath,
|
|
21
|
+
withLanguagePath,
|
|
22
|
+
} from '../../language/core/languagePath.ts';
|
|
23
|
+
import { withoutQueryOrFragment } from '../../router/core/address.ts';
|
|
24
|
+
import { stripBasePath } from '../../router/core/basePath.ts';
|
|
17
25
|
import { escapeAttribute, escapeText } from '../../share/core/escape.ts';
|
|
18
26
|
|
|
27
|
+
import { alternateLinkTags } from './alternates.ts';
|
|
19
28
|
import type { DocumentMeta } from './documentMeta.ts';
|
|
20
29
|
import type { RouteMeta } from './routes.ts';
|
|
21
30
|
import { pageMetaFor } from './routes.ts';
|
|
22
31
|
import { mountPathOf, originOf, resolveSite } from './siteFiles.ts';
|
|
23
32
|
import { PAGE_HEAD_MARKER, fill } from './template.ts';
|
|
24
33
|
|
|
34
|
+
// A scheme and an authority: what `location.href` hands out.
|
|
35
|
+
const ABSOLUTE_URL = /^[a-z][\d+.a-z-]*:\/\//i;
|
|
36
|
+
|
|
25
37
|
/** Which site is being served, and what it answers. */
|
|
26
38
|
export interface PageMetaOptions {
|
|
27
39
|
/** The site, named or passed. */
|
|
@@ -48,6 +60,17 @@ export interface PageMetaOptions {
|
|
|
48
60
|
* @default '/og.png'
|
|
49
61
|
*/
|
|
50
62
|
image?: string;
|
|
63
|
+
/**
|
|
64
|
+
* Every language the site is written in, the default one included.
|
|
65
|
+
*
|
|
66
|
+
* The language is read off the address — `/fr/tutorial` is the French
|
|
67
|
+
* tutorial — so the canonical, the card and the `hreflang` set are written
|
|
68
|
+
* for the page actually being served, and the routes passed are the table in
|
|
69
|
+
* that language. A site writing one language leaves this out and nothing
|
|
70
|
+
* about its head changes.
|
|
71
|
+
* @default [the default language]
|
|
72
|
+
*/
|
|
73
|
+
languages?: readonly Language[];
|
|
51
74
|
}
|
|
52
75
|
|
|
53
76
|
/**
|
|
@@ -74,12 +97,27 @@ export function injectPageMeta(html: string, options: PageMetaOptions): string {
|
|
|
74
97
|
*/
|
|
75
98
|
export function pageHeadTags(options: PageMetaOptions): string {
|
|
76
99
|
const name = siteDisplayName(resolveSite(options.site));
|
|
77
|
-
const
|
|
100
|
+
const route = routeMetaOf(options);
|
|
101
|
+
const description = route.description;
|
|
78
102
|
const origin = originOf(options);
|
|
79
103
|
const { title, canonical } = pageDocumentMeta(options);
|
|
80
104
|
const image = absolute(options.image ?? '/og.png', origin);
|
|
105
|
+
const alternates = alternateLinkTags({
|
|
106
|
+
origin,
|
|
107
|
+
path: route.path,
|
|
108
|
+
languages: languagesOf(options),
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
// A maintenance screen says so on the page itself. `robots.txt` cannot: it
|
|
112
|
+
// stops the crawl, and an address nobody crawled is still listed from
|
|
113
|
+
// whatever links to it, with no description to show for it.
|
|
114
|
+
const robots =
|
|
115
|
+
route.indexed === false
|
|
116
|
+
? ['<meta name="robots" content="noindex, nofollow" />']
|
|
117
|
+
: [];
|
|
81
118
|
|
|
82
119
|
return [
|
|
120
|
+
...robots,
|
|
83
121
|
`<title>${escapeText(title)}</title>`,
|
|
84
122
|
`<meta name="description" content="${escapeAttribute(description)}" />`,
|
|
85
123
|
`<link rel="canonical" href="${escapeAttribute(canonical)}" />`,
|
|
@@ -90,6 +128,7 @@ export function pageHeadTags(options: PageMetaOptions): string {
|
|
|
90
128
|
`<meta property="og:url" content="${escapeAttribute(canonical)}" />`,
|
|
91
129
|
`<meta property="og:image" content="${escapeAttribute(image)}" />`,
|
|
92
130
|
'<meta name="twitter:card" content="summary_large_image" />',
|
|
131
|
+
...(alternates === '' ? [] : [alternates]),
|
|
93
132
|
].join('\n');
|
|
94
133
|
}
|
|
95
134
|
|
|
@@ -112,15 +151,47 @@ export function pageDocumentMeta(
|
|
|
112
151
|
return {
|
|
113
152
|
title: `${meta.title} — ${siteDisplayName(site)}`,
|
|
114
153
|
description: meta.description,
|
|
115
|
-
canonical: `${originOf(options)}${meta.path}`,
|
|
154
|
+
canonical: `${originOf(options)}${withLanguagePath(pageLanguage(options), meta.path)}`,
|
|
116
155
|
};
|
|
117
156
|
}
|
|
118
157
|
|
|
158
|
+
/**
|
|
159
|
+
* The language an address is being served in.
|
|
160
|
+
*
|
|
161
|
+
* Read off the address itself rather than passed beside it, so the catalog a
|
|
162
|
+
* server picks and the canonical it writes cannot disagree.
|
|
163
|
+
* @param options - Which site, which address, and the languages it speaks.
|
|
164
|
+
* @returns The language named by the address, or the default one.
|
|
165
|
+
* @throws {Error} When the deployment names an origin that is not an absolute
|
|
166
|
+
* address.
|
|
167
|
+
*/
|
|
168
|
+
export function pageLanguage(options: PageMetaOptions): Language {
|
|
169
|
+
return readLanguagePath(ownPath(options), languagesOf(options)).language;
|
|
170
|
+
}
|
|
171
|
+
|
|
119
172
|
// A server behind a mount is handed the address the browser asked for, and the
|
|
120
173
|
// route table is written from the site's own root, so the mount the origin
|
|
121
174
|
// carries is taken off it before the table is read.
|
|
122
175
|
function routeMetaOf(options: PageMetaOptions): RouteMeta {
|
|
123
|
-
|
|
176
|
+
const { path } = readLanguagePath(ownPath(options), languagesOf(options));
|
|
177
|
+
return pageMetaFor(options.routes, path);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// The address from the site's own root, the mount taken off, with the query
|
|
181
|
+
// string and the fragment gone: what the language prefix is read off. An app
|
|
182
|
+
// handing over `location.href` is answered too, so the one reading serves a
|
|
183
|
+
// server, a build and a click alike.
|
|
184
|
+
function ownPath(options: PageMetaOptions): string {
|
|
185
|
+
const url = options.url;
|
|
186
|
+
const path =
|
|
187
|
+
ABSOLUTE_URL.test(url) && URL.canParse(url)
|
|
188
|
+
? new URL(url).pathname
|
|
189
|
+
: withoutQueryOrFragment(url);
|
|
190
|
+
return stripBasePath(mountPathOf(options), path);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
function languagesOf(options: PageMetaOptions): readonly Language[] {
|
|
194
|
+
return options.languages ?? [DEFAULT_LANGUAGE];
|
|
124
195
|
}
|
|
125
196
|
|
|
126
197
|
function absolute(target: string, origin: string): string {
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The text a page is indexed on, written into the HTML before anything runs.
|
|
3
|
+
*
|
|
4
|
+
* A prerendered site writes one file per address, and each of them carried the
|
|
5
|
+
* same body: the site's crawl path, and nothing else. A search engine clusters
|
|
6
|
+
* pages by the text it is handed, so a hundred and eighteen element pages
|
|
7
|
+
* differing only in their title are a hundred and seventeen duplicates — and
|
|
8
|
+
* which one it keeps is not ours to choose. A page heavy enough that the
|
|
9
|
+
* renderer never gets to it is indexed on that body alone.
|
|
10
|
+
*
|
|
11
|
+
* So what a site writes here is the text its running app already shows, drawn
|
|
12
|
+
* from the same data: the prose is not written twice, it is written where the
|
|
13
|
+
* build can reach it. Text only, and escaped — the addresses a reader follows
|
|
14
|
+
* are the crawl path's business.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { escapeText } from '../../share/core/escape.ts';
|
|
18
|
+
|
|
19
|
+
/** A table of facts under the prose: the rows a page would show anyway. */
|
|
20
|
+
export interface PageTable {
|
|
21
|
+
/** Header cells, left to right. */
|
|
22
|
+
columns: readonly string[];
|
|
23
|
+
/** Body rows, each as many cells as there are columns. */
|
|
24
|
+
rows: ReadonlyArray<readonly string[]>;
|
|
25
|
+
/**
|
|
26
|
+
* The sentence under the table, saying where the numbers come from.
|
|
27
|
+
* @default undefined — the table stands on its own
|
|
28
|
+
*/
|
|
29
|
+
caption?: string;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** One run of the page's text: a heading, prose, and the facts under it. */
|
|
33
|
+
export interface PageSection {
|
|
34
|
+
/**
|
|
35
|
+
* The heading it opens with, as an `h2`.
|
|
36
|
+
* @default undefined — the run carries no heading
|
|
37
|
+
*/
|
|
38
|
+
heading?: string;
|
|
39
|
+
/**
|
|
40
|
+
* The paragraphs under it, each taken as written.
|
|
41
|
+
* @default undefined — no prose
|
|
42
|
+
*/
|
|
43
|
+
paragraphs?: readonly string[];
|
|
44
|
+
/**
|
|
45
|
+
* A list under the prose, one item per line.
|
|
46
|
+
* @default undefined — no list
|
|
47
|
+
*/
|
|
48
|
+
list?: readonly string[];
|
|
49
|
+
/**
|
|
50
|
+
* A table under the prose.
|
|
51
|
+
* @default undefined — no table
|
|
52
|
+
*/
|
|
53
|
+
table?: PageTable;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** The whole of one page's text, as the page itself would say it. */
|
|
57
|
+
export interface PageContent extends PageSection {
|
|
58
|
+
/**
|
|
59
|
+
* The page's own name, written as its `h1`. It answers the title the page is
|
|
60
|
+
* indexed under, so a reader who searched for it reads the same words again.
|
|
61
|
+
*/
|
|
62
|
+
heading: string;
|
|
63
|
+
/**
|
|
64
|
+
* The runs under the opening prose, in reading order.
|
|
65
|
+
* @default undefined — the page is its opening prose
|
|
66
|
+
*/
|
|
67
|
+
sections?: readonly PageSection[];
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* The page's text as HTML, everything below its heading.
|
|
72
|
+
*
|
|
73
|
+
* The heading itself is left to the caller, which already writes the one `h1`
|
|
74
|
+
* the page carries — `noscriptIndex` does.
|
|
75
|
+
* @param content - What the page says.
|
|
76
|
+
* @param indent - The indentation every line is written at.
|
|
77
|
+
* @default ' '
|
|
78
|
+
* @returns The HTML, opening with a newline, or `''` when the page says nothing
|
|
79
|
+
* below its heading.
|
|
80
|
+
*/
|
|
81
|
+
export function pageProseHtml(content: PageContent, indent = ' '): string {
|
|
82
|
+
// The page's own heading is the caller's `h1`; what is left of its opening run
|
|
83
|
+
// is prose like any section's.
|
|
84
|
+
const opening: PageSection = {
|
|
85
|
+
paragraphs: content.paragraphs,
|
|
86
|
+
list: content.list,
|
|
87
|
+
table: content.table,
|
|
88
|
+
};
|
|
89
|
+
const parts = [
|
|
90
|
+
sectionHtml(opening, indent),
|
|
91
|
+
...(content.sections ?? []).map((section) => sectionHtml(section, indent)),
|
|
92
|
+
];
|
|
93
|
+
return parts.join('');
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function sectionHtml(section: PageSection, indent: string): string {
|
|
97
|
+
const lines: string[] = [];
|
|
98
|
+
if (section.heading !== undefined && section.heading.trim() !== '') {
|
|
99
|
+
lines.push(`${indent}<h2>${escapeText(section.heading)}</h2>`);
|
|
100
|
+
}
|
|
101
|
+
for (const paragraph of section.paragraphs ?? []) {
|
|
102
|
+
if (paragraph.trim() === '') continue;
|
|
103
|
+
lines.push(`${indent}<p>${escapeText(paragraph)}</p>`);
|
|
104
|
+
}
|
|
105
|
+
const list = listHtml(section.list, indent);
|
|
106
|
+
if (list !== '') lines.push(list);
|
|
107
|
+
const table = tableHtml(section.table, indent);
|
|
108
|
+
if (table !== '') lines.push(table);
|
|
109
|
+
return lines.length === 0 ? '' : `\n${lines.join('\n')}`;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function listHtml(list: readonly string[] | undefined, indent: string): string {
|
|
113
|
+
// A list with no item is not a list: `<ul>` holds at least one `<li>`.
|
|
114
|
+
if (list === undefined || list.length === 0) return '';
|
|
115
|
+
const items = list
|
|
116
|
+
.map((item) => `${indent} <li>${escapeText(item)}</li>`)
|
|
117
|
+
.join('\n');
|
|
118
|
+
return `${indent}<ul>\n${items}\n${indent}</ul>`;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function tableHtml(table: PageTable | undefined, indent: string): string {
|
|
122
|
+
// A table with no row says nothing, and an empty `<tbody>` is not markup a
|
|
123
|
+
// crawler is owed.
|
|
124
|
+
if (table === undefined || table.rows.length === 0) return '';
|
|
125
|
+
const caption =
|
|
126
|
+
table.caption === undefined || table.caption.trim() === ''
|
|
127
|
+
? ''
|
|
128
|
+
: `\n${indent} <caption>${escapeText(table.caption)}</caption>`;
|
|
129
|
+
const head = table.columns
|
|
130
|
+
.map((column) => `<th>${escapeText(column)}</th>`)
|
|
131
|
+
.join('');
|
|
132
|
+
const body = table.rows
|
|
133
|
+
.map(
|
|
134
|
+
(row) =>
|
|
135
|
+
`${indent} <tr>${row.map((cell) => `<td>${escapeText(cell)}</td>`).join('')}</tr>`,
|
|
136
|
+
)
|
|
137
|
+
.join('\n');
|
|
138
|
+
return `${indent}<table>${caption}
|
|
139
|
+
${indent} <thead>
|
|
140
|
+
${indent} <tr>${head}</tr>
|
|
141
|
+
${indent} </thead>
|
|
142
|
+
${indent} <tbody>
|
|
143
|
+
${body}
|
|
144
|
+
${indent} </tbody>
|
|
145
|
+
${indent}</table>`;
|
|
146
|
+
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What is wrong with the prose a site is indexed under.
|
|
3
|
+
*
|
|
4
|
+
* A route table is the only place a search result is written, and it is written
|
|
5
|
+
* once and read for years, so the limits that decide whether a result is
|
|
6
|
+
* readable are checked rather than trusted: a title Google cuts in half, a
|
|
7
|
+
* description too short to say anything or long enough to be clipped, two pages
|
|
8
|
+
* sharing a sentence — which is how a site asks to be deduplicated down to one
|
|
9
|
+
* result — and a snippet promising something the page does not carry.
|
|
10
|
+
*
|
|
11
|
+
* The paths are `assertRoutes`' business; this is about the words.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import type { RouteMeta } from './routes.ts';
|
|
15
|
+
|
|
16
|
+
/** The longest title that survives a search result whole. */
|
|
17
|
+
const TITLE_LIMIT = 60;
|
|
18
|
+
/** The shortest description that says anything. */
|
|
19
|
+
const DESCRIPTION_MIN = 110;
|
|
20
|
+
/** The longest description a result shows without clipping it. */
|
|
21
|
+
const DESCRIPTION_MAX = 160;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* What a site never says about itself, in the words it would say it in. A
|
|
25
|
+
* snippet naming a repository, a tracker or a licence is both a promise the
|
|
26
|
+
* page does not keep and a thing we do not publish.
|
|
27
|
+
*/
|
|
28
|
+
const WITHHELD =
|
|
29
|
+
/\b(?:licence|license|licensing|open[ -]source|repositor(?:y|ies)|issue tracker|source code|report a (?:problem|bug|issue))\b/i;
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* The phrase in a snippet that names something a site does not publish.
|
|
33
|
+
*
|
|
34
|
+
* Shared with the build-time checker, which reads the same sentences back out
|
|
35
|
+
* of the pages they were written into: one list of words, checked where they
|
|
36
|
+
* are authored and again where they landed.
|
|
37
|
+
* @param text - A title or a description.
|
|
38
|
+
* @returns The phrase, or `undefined` when there is none.
|
|
39
|
+
*/
|
|
40
|
+
export function withheldPhrase(text: string): string | undefined {
|
|
41
|
+
return WITHHELD.exec(text)?.[0];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Check the table a site is indexed under, before a build reads it.
|
|
46
|
+
*
|
|
47
|
+
* Returns the problems rather than throwing, so a site asserts an empty list in
|
|
48
|
+
* its own test and reads every one of them at once.
|
|
49
|
+
* @param routes - Every address the site answers.
|
|
50
|
+
* @returns One line per problem, empty when the table is fit to ship.
|
|
51
|
+
*/
|
|
52
|
+
export function routeProblems(routes: readonly RouteMeta[]): string[] {
|
|
53
|
+
const problems: string[] = [];
|
|
54
|
+
|
|
55
|
+
if (routes.length === 0) {
|
|
56
|
+
problems.push('the table is empty: a site answers at least one route.');
|
|
57
|
+
return problems;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const titles = new Map<string, string>();
|
|
61
|
+
const descriptions = new Map<string, string>();
|
|
62
|
+
|
|
63
|
+
for (const route of routes) {
|
|
64
|
+
const at = route.path;
|
|
65
|
+
|
|
66
|
+
if (route.title.trim() === '') {
|
|
67
|
+
problems.push(`${at}: the title is empty.`);
|
|
68
|
+
} else if (route.title.length > TITLE_LIMIT) {
|
|
69
|
+
problems.push(
|
|
70
|
+
`${at}: the title is ${route.title.length} characters, and the site name is appended to it — at most ${TITLE_LIMIT} survives a result whole.`,
|
|
71
|
+
);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const length = route.description.length;
|
|
75
|
+
if (length < DESCRIPTION_MIN) {
|
|
76
|
+
problems.push(
|
|
77
|
+
`${at}: the description is ${length} characters — at least ${DESCRIPTION_MIN}, or the result says half of what the page is.`,
|
|
78
|
+
);
|
|
79
|
+
} else if (length > DESCRIPTION_MAX) {
|
|
80
|
+
problems.push(
|
|
81
|
+
`${at}: the description is ${length} characters — at most ${DESCRIPTION_MAX}, or the sentence is cut off in the result itself.`,
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const titleFirst = titles.get(route.title);
|
|
86
|
+
if (titleFirst === undefined) {
|
|
87
|
+
titles.set(route.title, at);
|
|
88
|
+
} else {
|
|
89
|
+
problems.push(`${at}: the title repeats the one at ${titleFirst}.`);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const descriptionFirst = descriptions.get(route.description);
|
|
93
|
+
if (descriptionFirst === undefined) {
|
|
94
|
+
descriptions.set(route.description, at);
|
|
95
|
+
} else {
|
|
96
|
+
problems.push(
|
|
97
|
+
`${at}: the description repeats the one at ${descriptionFirst}.`,
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const withheld =
|
|
102
|
+
withheldPhrase(route.title) ?? withheldPhrase(route.description);
|
|
103
|
+
if (withheld !== undefined) {
|
|
104
|
+
problems.push(
|
|
105
|
+
`${at}: the snippet says "${withheld}" — a site names no repository, tracker or licence of ours.`,
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
return problems;
|
|
111
|
+
}
|
package/src/seo/core/routes.ts
CHANGED
|
@@ -42,6 +42,20 @@ export interface RouteMeta {
|
|
|
42
42
|
* @default undefined — the link stands on its own
|
|
43
43
|
*/
|
|
44
44
|
note?: string;
|
|
45
|
+
/**
|
|
46
|
+
* Whether a search engine is meant to list the page.
|
|
47
|
+
*
|
|
48
|
+
* A maintenance screen — a curation queue, an import run, an admin table — is
|
|
49
|
+
* a real address a signed-in person opens, so it belongs in the table the
|
|
50
|
+
* router and the tab title read. It is not a result anybody wants: it says
|
|
51
|
+
* nothing a visitor searched for, and a chemist who lands on it has been sent
|
|
52
|
+
* to the wrong place. Such a route is left out of the sitemap and the page
|
|
53
|
+
* answers `noindex`, which is how a page is kept out of the index — never a
|
|
54
|
+
* `Disallow`, which only stops the crawl and still lets the address be listed
|
|
55
|
+
* from a link somewhere else.
|
|
56
|
+
* @default true
|
|
57
|
+
*/
|
|
58
|
+
indexed?: boolean;
|
|
45
59
|
/**
|
|
46
60
|
* Whether the route also answers every address beneath it, so a section
|
|
47
61
|
* carrying more pages than a table can hold — an entry per structure, per
|
|
@@ -50,10 +50,11 @@ export interface SiteFilesOptions {
|
|
|
50
50
|
*/
|
|
51
51
|
export function sitemapXml(options: SiteFilesOptions): string {
|
|
52
52
|
const origin = originOf(options);
|
|
53
|
-
if (options.routes.
|
|
53
|
+
if (options.routes.every((route) => route.indexed === false)) {
|
|
54
54
|
throw new Error('a sitemap lists at least one address');
|
|
55
55
|
}
|
|
56
56
|
const entries = options.routes
|
|
57
|
+
.filter((route) => route.indexed !== false)
|
|
57
58
|
.map(
|
|
58
59
|
(route) =>
|
|
59
60
|
` <url><loc>${escapeText(`${origin}${route.path}`)}</loc></url>`,
|
|
@@ -27,6 +27,7 @@ import { trimTrailingSlash } from '../../router/core/address.ts';
|
|
|
27
27
|
import type { NoscriptText } from '../core/noscript.ts';
|
|
28
28
|
import { noscriptIndex } from '../core/noscript.ts';
|
|
29
29
|
import { pageHeadTags } from '../core/pageMeta.ts';
|
|
30
|
+
import type { PageContent } from '../core/pageProse.ts';
|
|
30
31
|
import type { RobotsDisallow } from '../core/robots.ts';
|
|
31
32
|
import { robotsTxt } from '../core/robots.ts';
|
|
32
33
|
import type { RouteMeta } from '../core/routes.ts';
|
|
@@ -91,6 +92,18 @@ export interface PrerenderOptions {
|
|
|
91
92
|
* @default true
|
|
92
93
|
*/
|
|
93
94
|
noscript?: boolean | NoscriptText;
|
|
95
|
+
/**
|
|
96
|
+
* What each page says for itself, above the crawl path: its own heading, its
|
|
97
|
+
* prose and the facts it would show anyway, read from the same data the app
|
|
98
|
+
* renders from.
|
|
99
|
+
*
|
|
100
|
+
* Without it every address ships the same body — the site's menu — and a
|
|
101
|
+
* search engine handed a hundred identical bodies keeps one of them. It is a
|
|
102
|
+
* function of the route rather than a field of it, so the prose stays out of
|
|
103
|
+
* the bundle the browser downloads: only the build ever calls it.
|
|
104
|
+
* @default undefined — every page carries the menu alone
|
|
105
|
+
*/
|
|
106
|
+
content?: (route: RouteMeta) => PageContent | undefined;
|
|
94
107
|
}
|
|
95
108
|
|
|
96
109
|
/**
|
|
@@ -104,18 +117,18 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
104
117
|
const { site, routes, origin, robots = [] } = options;
|
|
105
118
|
assertRoutes(routes);
|
|
106
119
|
const structuredData = structuredDataOf(options);
|
|
107
|
-
const crawlPath = crawlPathOf(options);
|
|
108
120
|
|
|
109
121
|
let out = 'dist';
|
|
110
122
|
let serve = false;
|
|
111
123
|
let logger: Logger | null = null;
|
|
112
124
|
|
|
113
|
-
const page = (template: string,
|
|
125
|
+
const page = (template: string, route: RouteMeta) => {
|
|
114
126
|
const head = fill(
|
|
115
127
|
template,
|
|
116
128
|
PAGE_HEAD_MARKER,
|
|
117
|
-
`${pageHeadTags({ site, routes, origin, url })}${structuredData}`,
|
|
129
|
+
`${pageHeadTags({ site, routes, origin, url: route.path })}${structuredData}`,
|
|
118
130
|
);
|
|
131
|
+
const crawlPath = crawlPathOf(options, route);
|
|
119
132
|
return crawlPath === '' ? head : fill(head, PAGE_BODY_MARKER, crawlPath);
|
|
120
133
|
};
|
|
121
134
|
|
|
@@ -132,17 +145,16 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
132
145
|
// from the home route rather than shipped with its markers showing.
|
|
133
146
|
transformIndexHtml: {
|
|
134
147
|
order: 'post',
|
|
135
|
-
handler: (html: string) =>
|
|
136
|
-
serve ? page(html, homeRoute(routes).path) : html,
|
|
148
|
+
handler: (html: string) => (serve ? page(html, homeRoute(routes)) : html),
|
|
137
149
|
},
|
|
138
150
|
|
|
139
151
|
closeBundle() {
|
|
140
152
|
if (serve) return;
|
|
141
153
|
const template = readFileSync(join(out, 'index.html'), 'utf8');
|
|
142
154
|
|
|
143
|
-
const write = (
|
|
155
|
+
const write = (route: RouteMeta, file: string) => {
|
|
144
156
|
mkdirSync(dirname(file), { recursive: true });
|
|
145
|
-
writeFileSync(file, page(template,
|
|
157
|
+
writeFileSync(file, page(template, route));
|
|
146
158
|
};
|
|
147
159
|
|
|
148
160
|
let root = false;
|
|
@@ -150,7 +162,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
150
162
|
const address = trimTrailingSlash(route.path);
|
|
151
163
|
if (address === '/') root = true;
|
|
152
164
|
write(
|
|
153
|
-
route
|
|
165
|
+
route,
|
|
154
166
|
address === '/'
|
|
155
167
|
? join(out, 'index.html')
|
|
156
168
|
: join(out, address.slice(1), 'index.html'),
|
|
@@ -159,7 +171,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
159
171
|
// The file a static server hands out for the mount itself. A table naming
|
|
160
172
|
// no root would otherwise leave the template vite built, and ship a site
|
|
161
173
|
// whose front page carries its markers instead of a head.
|
|
162
|
-
if (!root) write(homeRoute(routes)
|
|
174
|
+
if (!root) write(homeRoute(routes), join(out, 'index.html'));
|
|
163
175
|
|
|
164
176
|
writeFileSync(
|
|
165
177
|
join(out, 'sitemap.xml'),
|
|
@@ -196,9 +208,16 @@ function structuredDataOf(options: PrerenderOptions): string {
|
|
|
196
208
|
})}`;
|
|
197
209
|
}
|
|
198
210
|
|
|
199
|
-
function crawlPathOf(options: PrerenderOptions): string {
|
|
211
|
+
function crawlPathOf(options: PrerenderOptions, route: RouteMeta): string {
|
|
200
212
|
const { site, routes, origin, noscript = true } = options;
|
|
201
213
|
if (noscript === false) return '';
|
|
214
|
+
const content = options.content?.(route);
|
|
202
215
|
const { routes: listed, ...prose } = noscript === true ? {} : noscript;
|
|
203
|
-
return noscriptIndex({
|
|
216
|
+
return noscriptIndex({
|
|
217
|
+
site,
|
|
218
|
+
origin,
|
|
219
|
+
...prose,
|
|
220
|
+
routes: listed ?? routes,
|
|
221
|
+
content,
|
|
222
|
+
});
|
|
204
223
|
}
|