react-cheminfo 0.35.0 → 0.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/check-seo.mjs +333 -0
- package/lib/language/core/index.d.ts +2 -0
- package/lib/language/core/index.d.ts.map +1 -1
- package/lib/language/core/index.js +1 -0
- package/lib/language/core/index.js.map +1 -1
- package/lib/language/core/languagePath.d.ts +52 -0
- package/lib/language/core/languagePath.d.ts.map +1 -0
- package/lib/language/core/languagePath.js +65 -0
- package/lib/language/core/languagePath.js.map +1 -0
- package/lib/seo/core/alternates.d.ts +40 -0
- package/lib/seo/core/alternates.d.ts.map +1 -0
- package/lib/seo/core/alternates.js +42 -0
- package/lib/seo/core/alternates.js.map +1 -0
- package/lib/seo/core/index.d.ts +3 -0
- package/lib/seo/core/index.d.ts.map +1 -1
- package/lib/seo/core/index.js +2 -0
- package/lib/seo/core/index.js.map +1 -1
- package/lib/seo/core/noscript.d.ts +9 -0
- package/lib/seo/core/noscript.d.ts.map +1 -1
- package/lib/seo/core/noscript.js +7 -2
- package/lib/seo/core/noscript.js.map +1 -1
- package/lib/seo/core/pageMeta.d.ts +23 -0
- package/lib/seo/core/pageMeta.d.ts.map +1 -1
- package/lib/seo/core/pageMeta.js +52 -3
- package/lib/seo/core/pageMeta.js.map +1 -1
- package/lib/seo/core/pageProse.d.ts +76 -0
- package/lib/seo/core/pageProse.d.ts.map +1 -0
- package/lib/seo/core/pageProse.js +92 -0
- package/lib/seo/core/pageProse.js.map +1 -0
- package/lib/seo/core/routeProblems.d.ts +33 -0
- package/lib/seo/core/routeProblems.d.ts.map +1 -0
- package/lib/seo/core/routeProblems.js +89 -0
- package/lib/seo/core/routeProblems.js.map +1 -0
- package/lib/seo/core/routes.d.ts +14 -0
- package/lib/seo/core/routes.d.ts.map +1 -1
- package/lib/seo/core/routes.js.map +1 -1
- package/lib/seo/core/siteFiles.d.ts.map +1 -1
- package/lib/seo/core/siteFiles.js +2 -1
- package/lib/seo/core/siteFiles.js.map +1 -1
- package/lib/seo/vite/prerender.d.ts +13 -0
- package/lib/seo/vite/prerender.d.ts.map +1 -1
- package/lib/seo/vite/prerender.js +17 -10
- package/lib/seo/vite/prerender.js.map +1 -1
- package/package.json +2 -1
- package/src/language/core/index.ts +2 -0
- package/src/language/core/languagePath.ts +78 -0
- package/src/seo/core/alternates.ts +65 -0
- package/src/seo/core/index.ts +3 -0
- package/src/seo/core/noscript.ts +18 -2
- package/src/seo/core/pageMeta.ts +74 -3
- package/src/seo/core/pageProse.ts +146 -0
- package/src/seo/core/routeProblems.ts +111 -0
- package/src/seo/core/routes.ts +14 -0
- package/src/seo/core/siteFiles.ts +2 -1
- package/src/seo/vite/prerender.ts +30 -11
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What is wrong with the prose a site is indexed under.
|
|
3
|
+
*
|
|
4
|
+
* A route table is the only place a search result is written, and it is written
|
|
5
|
+
* once and read for years, so the limits that decide whether a result is
|
|
6
|
+
* readable are checked rather than trusted: a title Google cuts in half, a
|
|
7
|
+
* description too short to say anything or long enough to be clipped, two pages
|
|
8
|
+
* sharing a sentence — which is how a site asks to be deduplicated down to one
|
|
9
|
+
* result — and a snippet promising something the page does not carry.
|
|
10
|
+
*
|
|
11
|
+
* The paths are `assertRoutes`' business; this is about the words.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import type { RouteMeta } from './routes.ts';
|
|
15
|
+
|
|
16
|
+
/** The longest title that survives a search result whole. */
|
|
17
|
+
const TITLE_LIMIT = 60;
|
|
18
|
+
/** The shortest description that says anything. */
|
|
19
|
+
const DESCRIPTION_MIN = 110;
|
|
20
|
+
/** The longest description a result shows without clipping it. */
|
|
21
|
+
const DESCRIPTION_MAX = 160;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* What a site never says about itself, in the words it would say it in. A
|
|
25
|
+
* snippet naming a repository, a tracker or a licence is both a promise the
|
|
26
|
+
* page does not keep and a thing we do not publish.
|
|
27
|
+
*/
|
|
28
|
+
const WITHHELD =
|
|
29
|
+
/\b(?:licence|license|licensing|open[ -]source|repositor(?:y|ies)|issue tracker|source code|report a (?:problem|bug|issue))\b/i;
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* The phrase in a snippet that names something a site does not publish.
|
|
33
|
+
*
|
|
34
|
+
* Shared with the build-time checker, which reads the same sentences back out
|
|
35
|
+
* of the pages they were written into: one list of words, checked where they
|
|
36
|
+
* are authored and again where they landed.
|
|
37
|
+
* @param text - A title or a description.
|
|
38
|
+
* @returns The phrase, or `undefined` when there is none.
|
|
39
|
+
*/
|
|
40
|
+
export function withheldPhrase(text: string): string | undefined {
|
|
41
|
+
return WITHHELD.exec(text)?.[0];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Check the table a site is indexed under, before a build reads it.
|
|
46
|
+
*
|
|
47
|
+
* Returns the problems rather than throwing, so a site asserts an empty list in
|
|
48
|
+
* its own test and reads every one of them at once.
|
|
49
|
+
* @param routes - Every address the site answers.
|
|
50
|
+
* @returns One line per problem, empty when the table is fit to ship.
|
|
51
|
+
*/
|
|
52
|
+
export function routeProblems(routes: readonly RouteMeta[]): string[] {
|
|
53
|
+
const problems: string[] = [];
|
|
54
|
+
|
|
55
|
+
if (routes.length === 0) {
|
|
56
|
+
problems.push('the table is empty: a site answers at least one route.');
|
|
57
|
+
return problems;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const titles = new Map<string, string>();
|
|
61
|
+
const descriptions = new Map<string, string>();
|
|
62
|
+
|
|
63
|
+
for (const route of routes) {
|
|
64
|
+
const at = route.path;
|
|
65
|
+
|
|
66
|
+
if (route.title.trim() === '') {
|
|
67
|
+
problems.push(`${at}: the title is empty.`);
|
|
68
|
+
} else if (route.title.length > TITLE_LIMIT) {
|
|
69
|
+
problems.push(
|
|
70
|
+
`${at}: the title is ${route.title.length} characters, and the site name is appended to it — at most ${TITLE_LIMIT} survives a result whole.`,
|
|
71
|
+
);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
const length = route.description.length;
|
|
75
|
+
if (length < DESCRIPTION_MIN) {
|
|
76
|
+
problems.push(
|
|
77
|
+
`${at}: the description is ${length} characters — at least ${DESCRIPTION_MIN}, or the result says half of what the page is.`,
|
|
78
|
+
);
|
|
79
|
+
} else if (length > DESCRIPTION_MAX) {
|
|
80
|
+
problems.push(
|
|
81
|
+
`${at}: the description is ${length} characters — at most ${DESCRIPTION_MAX}, or the sentence is cut off in the result itself.`,
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const titleFirst = titles.get(route.title);
|
|
86
|
+
if (titleFirst === undefined) {
|
|
87
|
+
titles.set(route.title, at);
|
|
88
|
+
} else {
|
|
89
|
+
problems.push(`${at}: the title repeats the one at ${titleFirst}.`);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const descriptionFirst = descriptions.get(route.description);
|
|
93
|
+
if (descriptionFirst === undefined) {
|
|
94
|
+
descriptions.set(route.description, at);
|
|
95
|
+
} else {
|
|
96
|
+
problems.push(
|
|
97
|
+
`${at}: the description repeats the one at ${descriptionFirst}.`,
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const withheld =
|
|
102
|
+
withheldPhrase(route.title) ?? withheldPhrase(route.description);
|
|
103
|
+
if (withheld !== undefined) {
|
|
104
|
+
problems.push(
|
|
105
|
+
`${at}: the snippet says "${withheld}" — a site names no repository, tracker or licence of ours.`,
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
return problems;
|
|
111
|
+
}
|
package/src/seo/core/routes.ts
CHANGED
|
@@ -42,6 +42,20 @@ export interface RouteMeta {
|
|
|
42
42
|
* @default undefined — the link stands on its own
|
|
43
43
|
*/
|
|
44
44
|
note?: string;
|
|
45
|
+
/**
|
|
46
|
+
* Whether a search engine is meant to list the page.
|
|
47
|
+
*
|
|
48
|
+
* A maintenance screen — a curation queue, an import run, an admin table — is
|
|
49
|
+
* a real address a signed-in person opens, so it belongs in the table the
|
|
50
|
+
* router and the tab title read. It is not a result anybody wants: it says
|
|
51
|
+
* nothing a visitor searched for, and a chemist who lands on it has been sent
|
|
52
|
+
* to the wrong place. Such a route is left out of the sitemap and the page
|
|
53
|
+
* answers `noindex`, which is how a page is kept out of the index — never a
|
|
54
|
+
* `Disallow`, which only stops the crawl and still lets the address be listed
|
|
55
|
+
* from a link somewhere else.
|
|
56
|
+
* @default true
|
|
57
|
+
*/
|
|
58
|
+
indexed?: boolean;
|
|
45
59
|
/**
|
|
46
60
|
* Whether the route also answers every address beneath it, so a section
|
|
47
61
|
* carrying more pages than a table can hold — an entry per structure, per
|
|
@@ -50,10 +50,11 @@ export interface SiteFilesOptions {
|
|
|
50
50
|
*/
|
|
51
51
|
export function sitemapXml(options: SiteFilesOptions): string {
|
|
52
52
|
const origin = originOf(options);
|
|
53
|
-
if (options.routes.
|
|
53
|
+
if (options.routes.every((route) => route.indexed === false)) {
|
|
54
54
|
throw new Error('a sitemap lists at least one address');
|
|
55
55
|
}
|
|
56
56
|
const entries = options.routes
|
|
57
|
+
.filter((route) => route.indexed !== false)
|
|
57
58
|
.map(
|
|
58
59
|
(route) =>
|
|
59
60
|
` <url><loc>${escapeText(`${origin}${route.path}`)}</loc></url>`,
|
|
@@ -27,6 +27,7 @@ import { trimTrailingSlash } from '../../router/core/address.ts';
|
|
|
27
27
|
import type { NoscriptText } from '../core/noscript.ts';
|
|
28
28
|
import { noscriptIndex } from '../core/noscript.ts';
|
|
29
29
|
import { pageHeadTags } from '../core/pageMeta.ts';
|
|
30
|
+
import type { PageContent } from '../core/pageProse.ts';
|
|
30
31
|
import type { RobotsDisallow } from '../core/robots.ts';
|
|
31
32
|
import { robotsTxt } from '../core/robots.ts';
|
|
32
33
|
import type { RouteMeta } from '../core/routes.ts';
|
|
@@ -91,6 +92,18 @@ export interface PrerenderOptions {
|
|
|
91
92
|
* @default true
|
|
92
93
|
*/
|
|
93
94
|
noscript?: boolean | NoscriptText;
|
|
95
|
+
/**
|
|
96
|
+
* What each page says for itself, above the crawl path: its own heading, its
|
|
97
|
+
* prose and the facts it would show anyway, read from the same data the app
|
|
98
|
+
* renders from.
|
|
99
|
+
*
|
|
100
|
+
* Without it every address ships the same body — the site's menu — and a
|
|
101
|
+
* search engine handed a hundred identical bodies keeps one of them. It is a
|
|
102
|
+
* function of the route rather than a field of it, so the prose stays out of
|
|
103
|
+
* the bundle the browser downloads: only the build ever calls it.
|
|
104
|
+
* @default undefined — every page carries the menu alone
|
|
105
|
+
*/
|
|
106
|
+
content?: (route: RouteMeta) => PageContent | undefined;
|
|
94
107
|
}
|
|
95
108
|
|
|
96
109
|
/**
|
|
@@ -104,18 +117,18 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
104
117
|
const { site, routes, origin, robots = [] } = options;
|
|
105
118
|
assertRoutes(routes);
|
|
106
119
|
const structuredData = structuredDataOf(options);
|
|
107
|
-
const crawlPath = crawlPathOf(options);
|
|
108
120
|
|
|
109
121
|
let out = 'dist';
|
|
110
122
|
let serve = false;
|
|
111
123
|
let logger: Logger | null = null;
|
|
112
124
|
|
|
113
|
-
const page = (template: string,
|
|
125
|
+
const page = (template: string, route: RouteMeta) => {
|
|
114
126
|
const head = fill(
|
|
115
127
|
template,
|
|
116
128
|
PAGE_HEAD_MARKER,
|
|
117
|
-
`${pageHeadTags({ site, routes, origin, url })}${structuredData}`,
|
|
129
|
+
`${pageHeadTags({ site, routes, origin, url: route.path })}${structuredData}`,
|
|
118
130
|
);
|
|
131
|
+
const crawlPath = crawlPathOf(options, route);
|
|
119
132
|
return crawlPath === '' ? head : fill(head, PAGE_BODY_MARKER, crawlPath);
|
|
120
133
|
};
|
|
121
134
|
|
|
@@ -132,17 +145,16 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
132
145
|
// from the home route rather than shipped with its markers showing.
|
|
133
146
|
transformIndexHtml: {
|
|
134
147
|
order: 'post',
|
|
135
|
-
handler: (html: string) =>
|
|
136
|
-
serve ? page(html, homeRoute(routes).path) : html,
|
|
148
|
+
handler: (html: string) => (serve ? page(html, homeRoute(routes)) : html),
|
|
137
149
|
},
|
|
138
150
|
|
|
139
151
|
closeBundle() {
|
|
140
152
|
if (serve) return;
|
|
141
153
|
const template = readFileSync(join(out, 'index.html'), 'utf8');
|
|
142
154
|
|
|
143
|
-
const write = (
|
|
155
|
+
const write = (route: RouteMeta, file: string) => {
|
|
144
156
|
mkdirSync(dirname(file), { recursive: true });
|
|
145
|
-
writeFileSync(file, page(template,
|
|
157
|
+
writeFileSync(file, page(template, route));
|
|
146
158
|
};
|
|
147
159
|
|
|
148
160
|
let root = false;
|
|
@@ -150,7 +162,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
150
162
|
const address = trimTrailingSlash(route.path);
|
|
151
163
|
if (address === '/') root = true;
|
|
152
164
|
write(
|
|
153
|
-
route
|
|
165
|
+
route,
|
|
154
166
|
address === '/'
|
|
155
167
|
? join(out, 'index.html')
|
|
156
168
|
: join(out, address.slice(1), 'index.html'),
|
|
@@ -159,7 +171,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
|
|
|
159
171
|
// The file a static server hands out for the mount itself. A table naming
|
|
160
172
|
// no root would otherwise leave the template vite built, and ship a site
|
|
161
173
|
// whose front page carries its markers instead of a head.
|
|
162
|
-
if (!root) write(homeRoute(routes)
|
|
174
|
+
if (!root) write(homeRoute(routes), join(out, 'index.html'));
|
|
163
175
|
|
|
164
176
|
writeFileSync(
|
|
165
177
|
join(out, 'sitemap.xml'),
|
|
@@ -196,9 +208,16 @@ function structuredDataOf(options: PrerenderOptions): string {
|
|
|
196
208
|
})}`;
|
|
197
209
|
}
|
|
198
210
|
|
|
199
|
-
function crawlPathOf(options: PrerenderOptions): string {
|
|
211
|
+
function crawlPathOf(options: PrerenderOptions, route: RouteMeta): string {
|
|
200
212
|
const { site, routes, origin, noscript = true } = options;
|
|
201
213
|
if (noscript === false) return '';
|
|
214
|
+
const content = options.content?.(route);
|
|
202
215
|
const { routes: listed, ...prose } = noscript === true ? {} : noscript;
|
|
203
|
-
return noscriptIndex({
|
|
216
|
+
return noscriptIndex({
|
|
217
|
+
site,
|
|
218
|
+
origin,
|
|
219
|
+
...prose,
|
|
220
|
+
routes: listed ?? routes,
|
|
221
|
+
content,
|
|
222
|
+
});
|
|
204
223
|
}
|