react-cheminfo 0.35.0 → 0.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/bin/check-seo.mjs +333 -0
  2. package/lib/language/core/index.d.ts +2 -0
  3. package/lib/language/core/index.d.ts.map +1 -1
  4. package/lib/language/core/index.js +1 -0
  5. package/lib/language/core/index.js.map +1 -1
  6. package/lib/language/core/languagePath.d.ts +52 -0
  7. package/lib/language/core/languagePath.d.ts.map +1 -0
  8. package/lib/language/core/languagePath.js +65 -0
  9. package/lib/language/core/languagePath.js.map +1 -0
  10. package/lib/seo/core/alternates.d.ts +40 -0
  11. package/lib/seo/core/alternates.d.ts.map +1 -0
  12. package/lib/seo/core/alternates.js +42 -0
  13. package/lib/seo/core/alternates.js.map +1 -0
  14. package/lib/seo/core/index.d.ts +3 -0
  15. package/lib/seo/core/index.d.ts.map +1 -1
  16. package/lib/seo/core/index.js +2 -0
  17. package/lib/seo/core/index.js.map +1 -1
  18. package/lib/seo/core/noscript.d.ts +9 -0
  19. package/lib/seo/core/noscript.d.ts.map +1 -1
  20. package/lib/seo/core/noscript.js +7 -2
  21. package/lib/seo/core/noscript.js.map +1 -1
  22. package/lib/seo/core/pageMeta.d.ts +23 -0
  23. package/lib/seo/core/pageMeta.d.ts.map +1 -1
  24. package/lib/seo/core/pageMeta.js +52 -3
  25. package/lib/seo/core/pageMeta.js.map +1 -1
  26. package/lib/seo/core/pageProse.d.ts +76 -0
  27. package/lib/seo/core/pageProse.d.ts.map +1 -0
  28. package/lib/seo/core/pageProse.js +92 -0
  29. package/lib/seo/core/pageProse.js.map +1 -0
  30. package/lib/seo/core/routeProblems.d.ts +33 -0
  31. package/lib/seo/core/routeProblems.d.ts.map +1 -0
  32. package/lib/seo/core/routeProblems.js +89 -0
  33. package/lib/seo/core/routeProblems.js.map +1 -0
  34. package/lib/seo/core/routes.d.ts +14 -0
  35. package/lib/seo/core/routes.d.ts.map +1 -1
  36. package/lib/seo/core/routes.js.map +1 -1
  37. package/lib/seo/core/siteFiles.d.ts.map +1 -1
  38. package/lib/seo/core/siteFiles.js +2 -1
  39. package/lib/seo/core/siteFiles.js.map +1 -1
  40. package/lib/seo/vite/prerender.d.ts +13 -0
  41. package/lib/seo/vite/prerender.d.ts.map +1 -1
  42. package/lib/seo/vite/prerender.js +17 -10
  43. package/lib/seo/vite/prerender.js.map +1 -1
  44. package/package.json +2 -1
  45. package/src/language/core/index.ts +2 -0
  46. package/src/language/core/languagePath.ts +78 -0
  47. package/src/seo/core/alternates.ts +65 -0
  48. package/src/seo/core/index.ts +3 -0
  49. package/src/seo/core/noscript.ts +18 -2
  50. package/src/seo/core/pageMeta.ts +74 -3
  51. package/src/seo/core/pageProse.ts +146 -0
  52. package/src/seo/core/routeProblems.ts +111 -0
  53. package/src/seo/core/routes.ts +14 -0
  54. package/src/seo/core/siteFiles.ts +2 -1
  55. package/src/seo/vite/prerender.ts +30 -11
@@ -0,0 +1,111 @@
1
+ /**
2
+ * What is wrong with the prose a site is indexed under.
3
+ *
4
+ * A route table is the only place a search result is written, and it is written
5
+ * once and read for years, so the limits that decide whether a result is
6
+ * readable are checked rather than trusted: a title Google cuts in half, a
7
+ * description too short to say anything or long enough to be clipped, two pages
8
+ * sharing a sentence — which is how a site asks to be deduplicated down to one
9
+ * result — and a snippet promising something the page does not carry.
10
+ *
11
+ * The paths are `assertRoutes`' business; this is about the words.
12
+ */
13
+
14
+ import type { RouteMeta } from './routes.ts';
15
+
16
+ /** The longest title that survives a search result whole. */
17
+ const TITLE_LIMIT = 60;
18
+ /** The shortest description that says anything. */
19
+ const DESCRIPTION_MIN = 110;
20
+ /** The longest description a result shows without clipping it. */
21
+ const DESCRIPTION_MAX = 160;
22
+
23
+ /**
24
+ * What a site never says about itself, in the words it would say it in. A
25
+ * snippet naming a repository, a tracker or a licence is both a promise the
26
+ * page does not keep and a thing we do not publish.
27
+ */
28
+ const WITHHELD =
29
+ /\b(?:licence|license|licensing|open[ -]source|repositor(?:y|ies)|issue tracker|source code|report a (?:problem|bug|issue))\b/i;
30
+
31
+ /**
32
+ * The phrase in a snippet that names something a site does not publish.
33
+ *
34
+ * Shared with the build-time checker, which reads the same sentences back out
35
+ * of the pages they were written into: one list of words, checked where they
36
+ * are authored and again where they landed.
37
+ * @param text - A title or a description.
38
+ * @returns The phrase, or `undefined` when there is none.
39
+ */
40
+ export function withheldPhrase(text: string): string | undefined {
41
+ return WITHHELD.exec(text)?.[0];
42
+ }
43
+
44
+ /**
45
+ * Check the table a site is indexed under, before a build reads it.
46
+ *
47
+ * Returns the problems rather than throwing, so a site asserts an empty list in
48
+ * its own test and reads every one of them at once.
49
+ * @param routes - Every address the site answers.
50
+ * @returns One line per problem, empty when the table is fit to ship.
51
+ */
52
+ export function routeProblems(routes: readonly RouteMeta[]): string[] {
53
+ const problems: string[] = [];
54
+
55
+ if (routes.length === 0) {
56
+ problems.push('the table is empty: a site answers at least one route.');
57
+ return problems;
58
+ }
59
+
60
+ const titles = new Map<string, string>();
61
+ const descriptions = new Map<string, string>();
62
+
63
+ for (const route of routes) {
64
+ const at = route.path;
65
+
66
+ if (route.title.trim() === '') {
67
+ problems.push(`${at}: the title is empty.`);
68
+ } else if (route.title.length > TITLE_LIMIT) {
69
+ problems.push(
70
+ `${at}: the title is ${route.title.length} characters, and the site name is appended to it — at most ${TITLE_LIMIT} survives a result whole.`,
71
+ );
72
+ }
73
+
74
+ const length = route.description.length;
75
+ if (length < DESCRIPTION_MIN) {
76
+ problems.push(
77
+ `${at}: the description is ${length} characters — at least ${DESCRIPTION_MIN}, or the result says half of what the page is.`,
78
+ );
79
+ } else if (length > DESCRIPTION_MAX) {
80
+ problems.push(
81
+ `${at}: the description is ${length} characters — at most ${DESCRIPTION_MAX}, or the sentence is cut off in the result itself.`,
82
+ );
83
+ }
84
+
85
+ const titleFirst = titles.get(route.title);
86
+ if (titleFirst === undefined) {
87
+ titles.set(route.title, at);
88
+ } else {
89
+ problems.push(`${at}: the title repeats the one at ${titleFirst}.`);
90
+ }
91
+
92
+ const descriptionFirst = descriptions.get(route.description);
93
+ if (descriptionFirst === undefined) {
94
+ descriptions.set(route.description, at);
95
+ } else {
96
+ problems.push(
97
+ `${at}: the description repeats the one at ${descriptionFirst}.`,
98
+ );
99
+ }
100
+
101
+ const withheld =
102
+ withheldPhrase(route.title) ?? withheldPhrase(route.description);
103
+ if (withheld !== undefined) {
104
+ problems.push(
105
+ `${at}: the snippet says "${withheld}" — a site names no repository, tracker or licence of ours.`,
106
+ );
107
+ }
108
+ }
109
+
110
+ return problems;
111
+ }
@@ -42,6 +42,20 @@ export interface RouteMeta {
42
42
  * @default undefined — the link stands on its own
43
43
  */
44
44
  note?: string;
45
+ /**
46
+ * Whether a search engine is meant to list the page.
47
+ *
48
+ * A maintenance screen — a curation queue, an import run, an admin table — is
49
+ * a real address a signed-in person opens, so it belongs in the table the
50
+ * router and the tab title read. It is not a result anybody wants: it says
51
+ * nothing a visitor searched for, and a chemist who lands on it has been sent
52
+ * to the wrong place. Such a route is left out of the sitemap and the page
53
+ * answers `noindex`, which is how a page is kept out of the index — never a
54
+ * `Disallow`, which only stops the crawl and still lets the address be listed
55
+ * from a link somewhere else.
56
+ * @default true
57
+ */
58
+ indexed?: boolean;
45
59
  /**
46
60
  * Whether the route also answers every address beneath it, so a section
47
61
  * carrying more pages than a table can hold — an entry per structure, per
@@ -50,10 +50,11 @@ export interface SiteFilesOptions {
50
50
  */
51
51
  export function sitemapXml(options: SiteFilesOptions): string {
52
52
  const origin = originOf(options);
53
- if (options.routes.length === 0) {
53
+ if (options.routes.every((route) => route.indexed === false)) {
54
54
  throw new Error('a sitemap lists at least one address');
55
55
  }
56
56
  const entries = options.routes
57
+ .filter((route) => route.indexed !== false)
57
58
  .map(
58
59
  (route) =>
59
60
  ` <url><loc>${escapeText(`${origin}${route.path}`)}</loc></url>`,
@@ -27,6 +27,7 @@ import { trimTrailingSlash } from '../../router/core/address.ts';
27
27
  import type { NoscriptText } from '../core/noscript.ts';
28
28
  import { noscriptIndex } from '../core/noscript.ts';
29
29
  import { pageHeadTags } from '../core/pageMeta.ts';
30
+ import type { PageContent } from '../core/pageProse.ts';
30
31
  import type { RobotsDisallow } from '../core/robots.ts';
31
32
  import { robotsTxt } from '../core/robots.ts';
32
33
  import type { RouteMeta } from '../core/routes.ts';
@@ -91,6 +92,18 @@ export interface PrerenderOptions {
91
92
  * @default true
92
93
  */
93
94
  noscript?: boolean | NoscriptText;
95
+ /**
96
+ * What each page says for itself, above the crawl path: its own heading, its
97
+ * prose and the facts it would show anyway, read from the same data the app
98
+ * renders from.
99
+ *
100
+ * Without it every address ships the same body — the site's menu — and a
101
+ * search engine handed a hundred identical bodies keeps one of them. It is a
102
+ * function of the route rather than a field of it, so the prose stays out of
103
+ * the bundle the browser downloads: only the build ever calls it.
104
+ * @default undefined — every page carries the menu alone
105
+ */
106
+ content?: (route: RouteMeta) => PageContent | undefined;
94
107
  }
95
108
 
96
109
  /**
@@ -104,18 +117,18 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
104
117
  const { site, routes, origin, robots = [] } = options;
105
118
  assertRoutes(routes);
106
119
  const structuredData = structuredDataOf(options);
107
- const crawlPath = crawlPathOf(options);
108
120
 
109
121
  let out = 'dist';
110
122
  let serve = false;
111
123
  let logger: Logger | null = null;
112
124
 
113
- const page = (template: string, url: string) => {
125
+ const page = (template: string, route: RouteMeta) => {
114
126
  const head = fill(
115
127
  template,
116
128
  PAGE_HEAD_MARKER,
117
- `${pageHeadTags({ site, routes, origin, url })}${structuredData}`,
129
+ `${pageHeadTags({ site, routes, origin, url: route.path })}${structuredData}`,
118
130
  );
131
+ const crawlPath = crawlPathOf(options, route);
119
132
  return crawlPath === '' ? head : fill(head, PAGE_BODY_MARKER, crawlPath);
120
133
  };
121
134
 
@@ -132,17 +145,16 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
132
145
  // from the home route rather than shipped with its markers showing.
133
146
  transformIndexHtml: {
134
147
  order: 'post',
135
- handler: (html: string) =>
136
- serve ? page(html, homeRoute(routes).path) : html,
148
+ handler: (html: string) => (serve ? page(html, homeRoute(routes)) : html),
137
149
  },
138
150
 
139
151
  closeBundle() {
140
152
  if (serve) return;
141
153
  const template = readFileSync(join(out, 'index.html'), 'utf8');
142
154
 
143
- const write = (url: string, file: string) => {
155
+ const write = (route: RouteMeta, file: string) => {
144
156
  mkdirSync(dirname(file), { recursive: true });
145
- writeFileSync(file, page(template, url));
157
+ writeFileSync(file, page(template, route));
146
158
  };
147
159
 
148
160
  let root = false;
@@ -150,7 +162,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
150
162
  const address = trimTrailingSlash(route.path);
151
163
  if (address === '/') root = true;
152
164
  write(
153
- route.path,
165
+ route,
154
166
  address === '/'
155
167
  ? join(out, 'index.html')
156
168
  : join(out, address.slice(1), 'index.html'),
@@ -159,7 +171,7 @@ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
159
171
  // The file a static server hands out for the mount itself. A table naming
160
172
  // no root would otherwise leave the template vite built, and ship a site
161
173
  // whose front page carries its markers instead of a head.
162
- if (!root) write(homeRoute(routes).path, join(out, 'index.html'));
174
+ if (!root) write(homeRoute(routes), join(out, 'index.html'));
163
175
 
164
176
  writeFileSync(
165
177
  join(out, 'sitemap.xml'),
@@ -196,9 +208,16 @@ function structuredDataOf(options: PrerenderOptions): string {
196
208
  })}`;
197
209
  }
198
210
 
199
- function crawlPathOf(options: PrerenderOptions): string {
211
+ function crawlPathOf(options: PrerenderOptions, route: RouteMeta): string {
200
212
  const { site, routes, origin, noscript = true } = options;
201
213
  if (noscript === false) return '';
214
+ const content = options.content?.(route);
202
215
  const { routes: listed, ...prose } = noscript === true ? {} : noscript;
203
- return noscriptIndex({ site, origin, ...prose, routes: listed ?? routes });
216
+ return noscriptIndex({
217
+ site,
218
+ origin,
219
+ ...prose,
220
+ routes: listed ?? routes,
221
+ content,
222
+ });
204
223
  }