react-cheminfo 0.4.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/lib/ecosystem/core/index.d.ts +1 -1
  2. package/lib/ecosystem/core/index.d.ts.map +1 -1
  3. package/lib/ecosystem/core/index.js +1 -1
  4. package/lib/ecosystem/core/index.js.map +1 -1
  5. package/lib/ecosystem/core/lookup.d.ts +11 -0
  6. package/lib/ecosystem/core/lookup.d.ts.map +1 -1
  7. package/lib/ecosystem/core/lookup.js +15 -0
  8. package/lib/ecosystem/core/lookup.js.map +1 -1
  9. package/lib/ecosystem/core/sites.d.ts +1 -1
  10. package/lib/ecosystem/core/sites.d.ts.map +1 -1
  11. package/lib/ecosystem/core/sites.js +4 -4
  12. package/lib/ecosystem/core/sites.js.map +1 -1
  13. package/lib/ecosystem/ui/glyphs.js +1 -1
  14. package/lib/ecosystem/ui/glyphs.js.map +1 -1
  15. package/lib/seo/core/index.d.ts +6 -0
  16. package/lib/seo/core/index.d.ts.map +1 -1
  17. package/lib/seo/core/index.js +3 -0
  18. package/lib/seo/core/index.js.map +1 -1
  19. package/lib/seo/core/pageMeta.d.ts +60 -0
  20. package/lib/seo/core/pageMeta.d.ts.map +1 -0
  21. package/lib/seo/core/pageMeta.js +91 -0
  22. package/lib/seo/core/pageMeta.js.map +1 -0
  23. package/lib/seo/core/routes.d.ts +52 -0
  24. package/lib/seo/core/routes.d.ts.map +1 -0
  25. package/lib/seo/core/routes.js +62 -0
  26. package/lib/seo/core/routes.js.map +1 -0
  27. package/lib/seo/core/siteFiles.d.ts +73 -0
  28. package/lib/seo/core/siteFiles.d.ts.map +1 -0
  29. package/lib/seo/core/siteFiles.js +100 -0
  30. package/lib/seo/core/siteFiles.js.map +1 -0
  31. package/lib/seo/vite/index.d.ts +5 -0
  32. package/lib/seo/vite/index.d.ts.map +1 -0
  33. package/lib/seo/vite/index.js +3 -0
  34. package/lib/seo/vite/index.js.map +1 -0
  35. package/lib/seo/vite/ogCard.d.ts +33 -0
  36. package/lib/seo/vite/ogCard.d.ts.map +1 -0
  37. package/lib/seo/vite/ogCard.js +72 -0
  38. package/lib/seo/vite/ogCard.js.map +1 -0
  39. package/lib/seo/vite/prerender.d.ts +58 -0
  40. package/lib/seo/vite/prerender.d.ts.map +1 -0
  41. package/lib/seo/vite/prerender.js +76 -0
  42. package/lib/seo/vite/prerender.js.map +1 -0
  43. package/lib/vite.d.ts +2 -0
  44. package/lib/vite.d.ts.map +1 -0
  45. package/lib/vite.js +2 -0
  46. package/lib/vite.js.map +1 -0
  47. package/package.json +7 -2
  48. package/src/ecosystem/core/index.ts +1 -1
  49. package/src/ecosystem/core/lookup.ts +16 -0
  50. package/src/ecosystem/core/sites.ts +5 -5
  51. package/src/ecosystem/ui/glyphs.tsx +1 -1
  52. package/src/seo/core/index.ts +20 -0
  53. package/src/seo/core/pageMeta.ts +127 -0
  54. package/src/seo/core/routes.ts +79 -0
  55. package/src/seo/core/siteFiles.ts +150 -0
  56. package/src/seo/vite/index.ts +4 -0
  57. package/src/seo/vite/ogCard.ts +92 -0
  58. package/src/seo/vite/prerender.ts +156 -0
  59. package/src/vite.ts +1 -0
@@ -0,0 +1,79 @@
1
+ /**
2
+ * The addresses a site answers, each with the name and the sentence it is
3
+ * indexed under.
4
+ *
5
+ * One table per site, read by three things: the build, which writes an HTML
6
+ * file per entry and the sitemap listing them; the head injector; and the
7
+ * running app, which retitles the tab after an in-app move. A page missing from
8
+ * the table is a page a search engine only ever sees as the home page.
9
+ */
10
+
11
+ /** A page, as a crawler and a shared card see it. */
12
+ export interface RouteMeta {
13
+ /** Absolute path, without a trailing slash and without a query string. */
14
+ path: string;
15
+ /** Under ~60 characters: the site name is appended to it. */
16
+ title: string;
17
+ /** One sentence, in the words someone would search for. */
18
+ description: string;
19
+ }
20
+
21
+ /**
22
+ * The route an address names.
23
+ * @param routes - Every address the site answers.
24
+ * @param path - Absolute path, without a query string.
25
+ * @returns Its entry, or `undefined` when the site does not know the address.
26
+ */
27
+ export function routeFor(
28
+ routes: readonly RouteMeta[],
29
+ path: string,
30
+ ): RouteMeta | undefined {
31
+ const wanted = trimTrailingSlash(path) || '/';
32
+ for (const route of routes) {
33
+ if (trimTrailingSlash(route.path) === wanted) return route;
34
+ }
35
+ return undefined;
36
+ }
37
+
38
+ /**
39
+ * The page an address opens.
40
+ *
41
+ * An address the site does not know is described as the home page rather than
42
+ * invented on the fly, which is what the router does with it too. The query
43
+ * string never reaches the answer: the structure being drawn and the
44
+ * configuration a shared link carries are not pages of their own.
45
+ * @param routes - Every address the site answers.
46
+ * @param url - The address, query string and fragment included.
47
+ * @returns The route it is indexed as.
48
+ * @throws {Error} When the table is empty, so there is no page to fall back to.
49
+ */
50
+ export function pageMetaFor(
51
+ routes: readonly RouteMeta[],
52
+ url: string,
53
+ ): RouteMeta {
54
+ const home = homeRoute(routes);
55
+ const cut = url.search(/[?#]/);
56
+ const path = cut === -1 ? url : url.slice(0, cut);
57
+ return routeFor(routes, path) ?? home;
58
+ }
59
+
60
+ /**
61
+ * The page an unknown address falls back to.
62
+ * @param routes - Every address the site answers.
63
+ * @returns The `/` entry, or the first one when the table names no root.
64
+ * @throws {Error} When the table is empty.
65
+ */
66
+ export function homeRoute(routes: readonly RouteMeta[]): RouteMeta {
67
+ const first = routes[0];
68
+ if (first === undefined) throw new Error('a site answers at least one route');
69
+ return routeFor(routes, '/') ?? first;
70
+ }
71
+
72
+ /**
73
+ * Drop a trailing slash, so `/about/` and `/about` are one page.
74
+ * @param value - A path or an origin.
75
+ * @returns It, without the trailing slash `/` itself keeps.
76
+ */
77
+ export function trimTrailingSlash(value: string): string {
78
+ return value.length > 1 && value.endsWith('/') ? value.slice(0, -1) : value;
79
+ }
@@ -0,0 +1,150 @@
1
+ /**
2
+ * The files and blocks a crawler reads besides the head: the sitemap, the
3
+ * robots policy, the structured-data block and the list of addresses a visitor
4
+ * without JavaScript can still follow.
5
+ *
6
+ * All four are derived from the site's own record and its route table, so a
7
+ * page added to the table is added to every one of them at once.
8
+ */
9
+
10
+ import { siteById, siteDisplayName } from '../../ecosystem/core/lookup.ts';
11
+ import type { EcosystemSite, SiteId } from '../../ecosystem/core/sites.ts';
12
+ import { escapeAttribute, escapeText } from '../../share/core/escape.ts';
13
+
14
+ import type { RouteMeta } from './routes.ts';
15
+ import { trimTrailingSlash } from './routes.ts';
16
+
17
+ /** The sequence that must not appear raw inside a script element. */
18
+ const SCRIPT_SAFE_LESS_THAN = String.raw`\u003c`;
19
+
20
+ /** What a crawler is told about the site as a whole. */
21
+ export interface SiteFilesOptions {
22
+ /** The site, named or passed. */
23
+ site: EcosystemSite | SiteId;
24
+ /** Every address it answers. */
25
+ routes: readonly RouteMeta[];
26
+ /**
27
+ * Origin every absolute address is built on.
28
+ * @default `https://<the site's host>`
29
+ */
30
+ origin?: string;
31
+ }
32
+
33
+ /**
34
+ * Every routed address, as the sitemap lists them.
35
+ * @param options - The site and its routes.
36
+ * @returns The `sitemap.xml` document.
37
+ */
38
+ export function sitemapXml(options: SiteFilesOptions): string {
39
+ const origin = originOf(options);
40
+ const entries = options.routes
41
+ .map(
42
+ (route) =>
43
+ ` <url><loc>${escapeText(`${origin}${route.path}`)}</loc></url>`,
44
+ )
45
+ .join('\n');
46
+ return `<?xml version="1.0" encoding="UTF-8"?>
47
+ <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
48
+ ${entries}
49
+ </urlset>
50
+ `;
51
+ }
52
+
53
+ /**
54
+ * The crawl policy.
55
+ *
56
+ * Our tools are meant to be found, so only the endpoints are disallowed — an
57
+ * API prefix and its documentation are not pages. The sitemap is named only
58
+ * because this module also writes it: a `Sitemap:` line pointing at a 404 is
59
+ * reported as an error on every fetch.
60
+ * @param options - The site and its routes.
61
+ * @param disallow - Address prefixes to keep out of the index.
62
+ * @returns The `robots.txt` document.
63
+ */
64
+ export function robotsTxt(
65
+ options: SiteFilesOptions,
66
+ disallow: readonly string[] = [],
67
+ ): string {
68
+ const lines = ['User-agent: *', 'Allow: /'];
69
+ for (const path of disallow) lines.push(`Disallow: ${path}`);
70
+ lines.push('', `Sitemap: ${originOf(options)}/sitemap.xml`, '');
71
+ return lines.join('\n');
72
+ }
73
+
74
+ /** What the structured-data block says the tool is. */
75
+ export interface StructuredDataOptions extends SiteFilesOptions {
76
+ /**
77
+ * The schema.org application category.
78
+ * @default 'EducationalApplication'
79
+ */
80
+ category?: string;
81
+ /**
82
+ * What the tool needs to run.
83
+ * @default 'Any modern browser'
84
+ */
85
+ operatingSystem?: string;
86
+ }
87
+
88
+ /**
89
+ * One `application/ld+json` block describing the tool.
90
+ *
91
+ * It is the same on every page of a site — what varies per page is the head —
92
+ * so it is written into the built page once rather than per route.
93
+ * @param options - The site, and what kind of application it is.
94
+ * @returns The script tag, ready to put in the head.
95
+ */
96
+ export function structuredDataScript(options: StructuredDataOptions): string {
97
+ const site = resolveSite(options.site);
98
+ const data = {
99
+ '@context': 'https://schema.org',
100
+ '@type': 'WebApplication',
101
+ name: siteDisplayName(site),
102
+ url: `${originOf(options)}/`,
103
+ description: site.tagline,
104
+ applicationCategory: options.category ?? 'EducationalApplication',
105
+ operatingSystem: options.operatingSystem ?? 'Any modern browser',
106
+ offers: { '@type': 'Offer', price: '0', priceCurrency: 'EUR' },
107
+ publisher: { '@type': 'Organization', name: 'cheminfo' },
108
+ };
109
+ const json = JSON.stringify(data, null, 2).replaceAll(
110
+ '<',
111
+ SCRIPT_SAFE_LESS_THAN,
112
+ );
113
+ return `<script type="application/ld+json">\n${json}\n</script>`;
114
+ }
115
+
116
+ /**
117
+ * A readable page for a visitor, or a crawler, with no JavaScript.
118
+ *
119
+ * The body of our sites is an empty root element, so this is the only crawl
120
+ * path through them that costs nothing to render — and it is honest: it says
121
+ * the tool needs JavaScript, and links every address it answers.
122
+ * @param options - The site and its routes.
123
+ * @returns The `noscript` block, ready to put in the body.
124
+ */
125
+ export function noscriptIndex(options: SiteFilesOptions): string {
126
+ const site = resolveSite(options.site);
127
+ const items = options.routes
128
+ .map(
129
+ (route) =>
130
+ ` <li><a href="${escapeAttribute(route.path)}">${escapeText(route.title)}</a></li>`,
131
+ )
132
+ .join('\n');
133
+ return `<noscript>
134
+ <h1>${escapeText(siteDisplayName(site))}</h1>
135
+ <p>${escapeText(site.tagline)} This tool needs JavaScript; these are the pages it offers:</p>
136
+ <ul>
137
+ ${items}
138
+ </ul>
139
+ </noscript>`;
140
+ }
141
+
142
+ function resolveSite(site: EcosystemSite | SiteId): EcosystemSite {
143
+ return typeof site === 'string' ? siteById(site) : site;
144
+ }
145
+
146
+ function originOf(options: SiteFilesOptions): string {
147
+ return trimTrailingSlash(
148
+ options.origin ?? `https://${resolveSite(options.site).host}`,
149
+ );
150
+ }
@@ -0,0 +1,4 @@
1
+ export type { OgCardOptions } from './ogCard.ts';
2
+ export { OG_HEIGHT, OG_WIDTH, ogCardHtml } from './ogCard.ts';
3
+ export type { PrerenderOptions } from './prerender.ts';
4
+ export { cheminfoPrerender } from './prerender.ts';
@@ -0,0 +1,92 @@
1
+ /**
2
+ * The 1200×630 card a link to a site unfurls into, as a page to screenshot.
3
+ *
4
+ * The card is the site's own mark, its two colours and its name, all read from
5
+ * its record — so it is generated rather than hand-drawn. A mark redrawn in the
6
+ * card is a mark that drifts from the one the site shows.
7
+ */
8
+
9
+ import { createElement } from 'react';
10
+ import { renderToStaticMarkup } from 'react-dom/server';
11
+
12
+ import { siteById } from '../../ecosystem/core/lookup.ts';
13
+ import type { EcosystemSite, SiteId } from '../../ecosystem/core/sites.ts';
14
+ import { SiteMark } from '../../ecosystem/ui/marks.tsx';
15
+ import { escapeText } from '../../share/core/escape.ts';
16
+
17
+ /** The width every card is drawn at. */
18
+ export const OG_WIDTH = 1200;
19
+
20
+ /** The height every card is drawn at. */
21
+ export const OG_HEIGHT = 630;
22
+
23
+ /** What the card says, beyond the site's own name and mark. */
24
+ export interface OgCardOptions {
25
+ /** The site, named or passed. */
26
+ site: EcosystemSite | SiteId;
27
+ /**
28
+ * The sentence under the name.
29
+ * @default the site's tagline
30
+ */
31
+ description?: string;
32
+ }
33
+
34
+ /**
35
+ * The card, as a standalone page.
36
+ *
37
+ * Screenshot it at {@link OG_WIDTH} × {@link OG_HEIGHT} — a headless browser is
38
+ * the only thing here that can rasterise it, and every site already has one for
39
+ * its end-to-end tests.
40
+ * @param options - Which site, and what it says.
41
+ * @returns A complete HTML document.
42
+ */
43
+ export function ogCardHtml(options: OgCardOptions): string {
44
+ const site =
45
+ typeof options.site === 'string' ? siteById(options.site) : options.site;
46
+ const description = options.description ?? site.tagline;
47
+ const mark = renderToStaticMarkup(
48
+ createElement(SiteMark, { site, size: 132, colors: 'literal' }),
49
+ );
50
+ const dot = site.name.dot === true ? '<span class="dot">.</span>' : '';
51
+
52
+ return `<!doctype html>
53
+ <html lang="en">
54
+ <head>
55
+ <meta charset="utf-8" />
56
+ <style>
57
+ * { box-sizing: border-box; margin: 0; }
58
+ body {
59
+ display: flex;
60
+ width: ${OG_WIDTH}px;
61
+ height: ${OG_HEIGHT}px;
62
+ flex-direction: column;
63
+ justify-content: center;
64
+ padding: 88px;
65
+ background: #ffffff;
66
+ color: #16202c;
67
+ font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto,
68
+ Helvetica, Arial, sans-serif;
69
+ gap: 28px;
70
+ }
71
+ h1 { font-size: 76px; font-weight: 700; letter-spacing: -0.02em; }
72
+ .lead { color: ${site.brand}; }
73
+ .alt { color: ${site.brandAlt}; }
74
+ .dot { color: #8a96a3; }
75
+ p { max-width: 900px; color: #5b6875; font-size: 34px; line-height: 1.35; }
76
+ .rule {
77
+ width: 180px;
78
+ height: 10px;
79
+ border-radius: 5px;
80
+ background: ${site.brandAlt};
81
+ }
82
+ </style>
83
+ </head>
84
+ <body>
85
+ ${mark}
86
+ <h1><span class="lead">${escapeText(site.name.lead)}</span>${dot}<span class="alt">${escapeText(site.name.alt)}</span></h1>
87
+ <div class="rule"></div>
88
+ <p>${escapeText(description)}</p>
89
+ </body>
90
+ </html>
91
+ `;
92
+ }
@@ -0,0 +1,156 @@
1
+ /**
2
+ * Write one real HTML file per routed address, and everything else a crawler
3
+ * fetches on its own.
4
+ *
5
+ * A site served by a static image has nothing to rewrite a head per request: a
6
+ * crawler gets whatever came off the wire. Without this every address carries
7
+ * the same title and a search engine folds the whole site into one result.
8
+ *
9
+ * These files are also what makes the server's catch-all fallback unnecessary.
10
+ * Every address the tool answers is on disk, so an address that is *not* on
11
+ * disk is genuinely not a page and must 404 rather than serving the tool under
12
+ * a name it does not have.
13
+ */
14
+
15
+ import { mkdirSync, readFileSync, writeFileSync } from 'node:fs';
16
+ import { dirname, join, resolve } from 'node:path';
17
+
18
+ import type { Logger, Plugin } from 'vite';
19
+
20
+ import type { EcosystemSite, SiteId } from '../../ecosystem/core/sites.ts';
21
+ import { injectPageMeta, insertBeforeHeadEnd } from '../core/pageMeta.ts';
22
+ import type { RouteMeta } from '../core/routes.ts';
23
+ import {
24
+ noscriptIndex,
25
+ robotsTxt,
26
+ sitemapXml,
27
+ structuredDataScript,
28
+ } from '../core/siteFiles.ts';
29
+
30
+ /** What the build needs to know to write the site's addresses. */
31
+ export interface PrerenderOptions {
32
+ /** The site, named or passed. */
33
+ site: EcosystemSite | SiteId;
34
+ /** Every address it answers, each with its title and description. */
35
+ routes: readonly RouteMeta[];
36
+ /**
37
+ * Origin every absolute address is built on.
38
+ * @default `https://<the site's host>`
39
+ */
40
+ origin?: string;
41
+ /**
42
+ * Address prefixes `robots.txt` keeps out of the index, e.g. `/v1/`. Set to
43
+ * `false` to write no `robots.txt` at all, for a site that ships its own.
44
+ * @default []
45
+ */
46
+ robots?: false | readonly string[];
47
+ /**
48
+ * The schema.org category of the structured-data block, or `false` to write
49
+ * none.
50
+ * @default 'EducationalApplication'
51
+ */
52
+ category?: false | string;
53
+ /**
54
+ * What the tool needs to run, named in the structured-data block.
55
+ * @default 'Any modern browser'
56
+ */
57
+ operatingSystem?: string;
58
+ /**
59
+ * Whether the built page carries a `noscript` index of the addresses. It is
60
+ * the only crawl path through a site whose body is an empty root element.
61
+ * @default true
62
+ */
63
+ noscript?: boolean;
64
+ }
65
+
66
+ /**
67
+ * Prerender every routed address of a cheminfo site.
68
+ * @param options - The site, its routes, and what a crawler is told.
69
+ * @returns The Vite plugin.
70
+ */
71
+ export function cheminfoPrerender(options: PrerenderOptions): Plugin {
72
+ const {
73
+ site,
74
+ routes,
75
+ origin,
76
+ robots = [],
77
+ category,
78
+ operatingSystem,
79
+ noscript = true,
80
+ } = options;
81
+ let out = 'dist';
82
+ let logger: Logger | null = null;
83
+
84
+ return {
85
+ name: 'cheminfo:prerender',
86
+ apply: 'build',
87
+
88
+ configResolved(config) {
89
+ out = resolve(config.root, config.build.outDir);
90
+ logger = config.logger;
91
+ },
92
+
93
+ transformIndexHtml: {
94
+ order: 'post',
95
+ handler(html) {
96
+ let page = html;
97
+ if (category !== false) {
98
+ page = insertBeforeHeadEnd(
99
+ page,
100
+ structuredDataScript({
101
+ site,
102
+ routes,
103
+ origin,
104
+ category,
105
+ operatingSystem,
106
+ }),
107
+ );
108
+ }
109
+ if (noscript) {
110
+ page = insertBeforeBodyEnd(
111
+ page,
112
+ noscriptIndex({ site, routes, origin }),
113
+ );
114
+ }
115
+ return page;
116
+ },
117
+ },
118
+
119
+ closeBundle() {
120
+ const index = readFileSync(join(out, 'index.html'), 'utf8');
121
+
122
+ for (const route of routes) {
123
+ const file =
124
+ route.path === '/'
125
+ ? join(out, 'index.html')
126
+ : join(out, route.path.slice(1), 'index.html');
127
+ mkdirSync(dirname(file), { recursive: true });
128
+ writeFileSync(
129
+ file,
130
+ injectPageMeta(index, { site, routes, origin, url: route.path }),
131
+ );
132
+ }
133
+
134
+ writeFileSync(
135
+ join(out, 'sitemap.xml'),
136
+ sitemapXml({ site, routes, origin }),
137
+ );
138
+ if (robots !== false) {
139
+ writeFileSync(
140
+ join(out, 'robots.txt'),
141
+ robotsTxt({ site, routes, origin }, robots),
142
+ );
143
+ }
144
+
145
+ logger?.info(
146
+ `${routes.length} pages prerendered, and listed in sitemap.xml`,
147
+ );
148
+ },
149
+ };
150
+ }
151
+
152
+ function insertBeforeBodyEnd(html: string, addition: string): string {
153
+ const body = html.lastIndexOf('</body>');
154
+ if (body === -1) return `${html}\n${addition}\n`;
155
+ return `${html.slice(0, body)}${addition}\n${html.slice(body)}`;
156
+ }
package/src/vite.ts ADDED
@@ -0,0 +1 @@
1
+ export * from './seo/vite/index.ts';