@haruhimemoe/next-kit 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,135 @@
1
+ /**
2
+ * @file src/seo/ld-site.ts
3
+ * @desc schema.org JSON-LD for the site itself: the graph wrapper (the only place `@context`
4
+ * goes), Organization, WebSite with its SearchAction, WebApplication, BreadcrumbList and
5
+ * ItemList. Stable @ids tie every tool to one organization (audit A3, A6).
6
+ * @author David @dvhsh (https://dvh.sh)
7
+ * @created Mon Sep 28, 2026
8
+ * @modified Mon Sep 28, 2026
9
+ */
10
+ import { absoluteUrl, compact, nodeId } from "./site.js";
11
+ /** The placeholder a SearchAction's urlTemplate must hold. */
12
+ export const SEARCH_TERM = "{search_term_string}";
13
+ /**
14
+ * @function graph
15
+ * @param nodes {LdNode[]} the page's nodes
16
+ * @returns {LdGraph} one JSON-LD document with `@context` on the root only
17
+ */
18
+ export const graph = (...nodes) => ({
19
+ "@context": "https://schema.org",
20
+ "@graph": nodes,
21
+ });
22
+ /**
23
+ * @function organization
24
+ * @param org {Organization} the organization, like HARUHIME_ORG
25
+ * @returns {LdNode} Organization with @id `${origin}/#organization`
26
+ */
27
+ export const organization = (org) => compact({
28
+ "@type": "Organization",
29
+ "@id": nodeId(org.url, "organization"),
30
+ name: org.name,
31
+ url: org.url,
32
+ logo: org.logo,
33
+ email: org.email,
34
+ sameAs: org.sameAs?.length ? [...org.sameAs] : undefined,
35
+ });
36
+ /** A reference to the site's organization by @id. */
37
+ export const orgRef = (site) => ({
38
+ "@id": nodeId(site.organization.url, "organization"),
39
+ });
40
+ /**
41
+ * @function webSite
42
+ * @param site {Site} the site
43
+ * @param options {{ searchUrlTemplate?: string }} a search URL holding {search_term_string},
44
+ * like "/search?q={search_term_string}"
45
+ * @returns {LdNode} WebSite with @id `${origin}/#website`, publisher → the organization,
46
+ * isPartOf → the parent site, and a SearchAction when a template is given
47
+ * @throws {Error} when the template lacks {search_term_string} or isn't on the site
48
+ */
49
+ export const webSite = (site, options = {}) => {
50
+ const template = options.searchUrlTemplate;
51
+ if (template !== undefined && !template.includes(SEARCH_TERM)) {
52
+ throw new Error(`seo: searchUrlTemplate must hold ${SEARCH_TERM}`);
53
+ }
54
+ return compact({
55
+ "@type": "WebSite",
56
+ "@id": nodeId(site.url, "website"),
57
+ name: site.name,
58
+ alternateName: new URL(site.url).host,
59
+ url: absoluteUrl(site, "/"),
60
+ description: site.description,
61
+ inLanguage: (site.locale ?? "en_US").replace("_", "-"),
62
+ publisher: orgRef(site),
63
+ isPartOf: site.parent ? { "@id": nodeId(site.parent.url, "website") } : undefined,
64
+ potentialAction: template
65
+ ? {
66
+ "@type": "SearchAction",
67
+ target: { "@type": "EntryPoint", urlTemplate: absoluteUrl(site, template) },
68
+ "query-input": "required name=search_term_string",
69
+ }
70
+ : undefined,
71
+ });
72
+ };
73
+ /**
74
+ * @function webApplication
75
+ * @param site {Site} the site
76
+ * @param options {WebApplicationOptions} category, and optional name, features and path
77
+ * @returns {LdNode} WebApplication, free (Offer price 0), operatingSystem "Any", publisher →
78
+ * the organization
79
+ */
80
+ export const webApplication = (site, options) => {
81
+ const url = absoluteUrl(site, options.path ?? "/");
82
+ return compact({
83
+ "@type": "WebApplication",
84
+ "@id": options.path && options.path !== "/" ? `${url}#app` : nodeId(site.url, "app"),
85
+ name: options.name ?? site.name,
86
+ url,
87
+ description: options.description ?? site.description,
88
+ applicationCategory: options.category,
89
+ operatingSystem: "Any",
90
+ browserRequirements: options.browserRequirements ?? "Requires JavaScript. Works in any modern browser.",
91
+ featureList: options.features?.length ? [...options.features] : undefined,
92
+ isAccessibleForFree: true,
93
+ offers: { "@type": "Offer", price: "0", priceCurrency: "USD" },
94
+ publisher: orgRef(site),
95
+ isPartOf: { "@id": nodeId(site.url, "website") },
96
+ });
97
+ };
98
+ /**
99
+ * @function breadcrumbs
100
+ * @param site {Site} the site
101
+ * @param trail {readonly LdLink[]} from the home page down to this page
102
+ * @returns {LdNode} BreadcrumbList, positions from 1, absolute item URLs
103
+ * @throws {Error} when the trail is empty
104
+ */
105
+ export const breadcrumbs = (site, trail) => {
106
+ if (!trail.length)
107
+ throw new Error("seo: breadcrumbs need at least one page");
108
+ return {
109
+ "@type": "BreadcrumbList",
110
+ itemListElement: trail.map((link, i) => ({
111
+ "@type": "ListItem",
112
+ position: i + 1,
113
+ name: link.name,
114
+ item: absoluteUrl(site, link.path),
115
+ })),
116
+ };
117
+ };
118
+ /**
119
+ * @function itemList
120
+ * @param site {Site} the site
121
+ * @param items {readonly LdLink[]} the listed pages, in order
122
+ * @param options {{ name?: string }} the list's name
123
+ * @returns {LdNode} ItemList with numberOfItems and ListItems (position, name, url)
124
+ */
125
+ export const itemList = (site, items, options = {}) => compact({
126
+ "@type": "ItemList",
127
+ name: options.name,
128
+ numberOfItems: items.length,
129
+ itemListElement: items.map((item, i) => ({
130
+ "@type": "ListItem",
131
+ position: i + 1,
132
+ name: item.name,
133
+ url: absoluteUrl(site, item.path),
134
+ })),
135
+ });
@@ -0,0 +1,32 @@
1
+ /**
2
+ * @file src/seo/ld.ts
3
+ * @desc The `ld` namespace of JSON-LD builders, and serializeLd: JSON that is safe inside a
4
+ * `<script>` tag. @haruhimemoe/ui's JsonLd escapes "<" only; serializeLd also escapes ">",
5
+ * "&" and U+2028/U+2029, so the output is safe in any HTML or JavaScript context.
6
+ * @author David @dvhsh (https://dvh.sh)
7
+ * @created Mon Sep 28, 2026
8
+ * @modified Mon Sep 28, 2026
9
+ */
10
+ import { creativeWork, dataset, faq, howTo, techArticle } from "./ld-content.js";
11
+ import { breadcrumbs, graph, itemList, organization, webApplication, webSite } from "./ld-site.js";
12
+ /** The JSON-LD builders: plain schema.org objects, rendered with @haruhimemoe/ui's JsonLd. */
13
+ export declare const ld: {
14
+ readonly graph: typeof graph;
15
+ readonly organization: typeof organization;
16
+ readonly webSite: typeof webSite;
17
+ readonly webApplication: typeof webApplication;
18
+ readonly breadcrumbs: typeof breadcrumbs;
19
+ readonly itemList: typeof itemList;
20
+ readonly faq: typeof faq;
21
+ readonly howTo: typeof howTo;
22
+ readonly techArticle: typeof techArticle;
23
+ readonly creativeWork: typeof creativeWork;
24
+ readonly dataset: typeof dataset;
25
+ };
26
+ /**
27
+ * @function serializeLd
28
+ * @param data {object} a JSON-LD document or node
29
+ * @returns {string} JSON with <, >, & and U+2028/U+2029 written as \u escapes: the same data to
30
+ * a JSON parser, and nothing a `<script>` tag or an HTML comment can end on
31
+ */
32
+ export declare const serializeLd: (data: object) => string;
package/dist/seo/ld.js ADDED
@@ -0,0 +1,39 @@
1
+ /**
2
+ * @file src/seo/ld.ts
3
+ * @desc The `ld` namespace of JSON-LD builders, and serializeLd: JSON that is safe inside a
4
+ * `<script>` tag. @haruhimemoe/ui's JsonLd escapes "<" only; serializeLd also escapes ">",
5
+ * "&" and U+2028/U+2029, so the output is safe in any HTML or JavaScript context.
6
+ * @author David @dvhsh (https://dvh.sh)
7
+ * @created Mon Sep 28, 2026
8
+ * @modified Mon Sep 28, 2026
9
+ */
10
+ import { creativeWork, dataset, faq, howTo, techArticle } from "./ld-content.js";
11
+ import { breadcrumbs, graph, itemList, organization, webApplication, webSite } from "./ld-site.js";
12
+ /** The JSON-LD builders: plain schema.org objects, rendered with @haruhimemoe/ui's JsonLd. */
13
+ export const ld = {
14
+ graph,
15
+ organization,
16
+ webSite,
17
+ webApplication,
18
+ breadcrumbs,
19
+ itemList,
20
+ faq,
21
+ howTo,
22
+ techArticle,
23
+ creativeWork,
24
+ dataset,
25
+ };
26
+ const ESCAPES = {
27
+ "<": "\\u003c",
28
+ ">": "\\u003e",
29
+ "&": "\\u0026",
30
+ "\u2028": "\\u2028",
31
+ "\u2029": "\\u2029",
32
+ };
33
+ /**
34
+ * @function serializeLd
35
+ * @param data {object} a JSON-LD document or node
36
+ * @returns {string} JSON with <, >, & and U+2028/U+2029 written as \u escapes: the same data to
37
+ * a JSON parser, and nothing a `<script>` tag or an HTML comment can end on
38
+ */
39
+ export const serializeLd = (data) => JSON.stringify(data).replace(/[<>&\u2028\u2029]/g, (char) => ESCAPES[char]);
@@ -0,0 +1,72 @@
1
+ /**
2
+ * @file src/seo/llms.ts
3
+ * @desc /llms.txt (https://llmstxt.org): an H1, a blockquote summary, note paragraphs, then H2
4
+ * sections of "- [title](url): note" links. /llms-full.txt: whole documents in one file.
5
+ * And the text/plain response both are served with. One builder, so escaping and section
6
+ * format stay the same on every site (audit A2).
7
+ * @author David @dvhsh (https://dvh.sh)
8
+ * @created Mon Sep 28, 2026
9
+ * @modified Mon Sep 28, 2026
10
+ */
11
+ /** One link in a section. */
12
+ export type LlmsLink = {
13
+ title: string;
14
+ url: string;
15
+ note?: string;
16
+ };
17
+ /** An H2 section of links. A section with no links is left out. */
18
+ export type LlmsSection = {
19
+ heading: string;
20
+ links: readonly LlmsLink[];
21
+ };
22
+ /** llmsTxt's input. */
23
+ export type LlmsTxtOptions = {
24
+ /** The H1, like "pools.haruhime.moe". */
25
+ title: string;
26
+ /** The blockquote: what the site is, in one paragraph. */
27
+ summary: string;
28
+ /** Paragraphs a reader needs before following any link. */
29
+ notes?: readonly string[];
30
+ sections: readonly LlmsSection[];
31
+ };
32
+ /** One document in llms-full.txt. */
33
+ export type LlmsFullPart = {
34
+ title: string;
35
+ url?: string;
36
+ markdown: string;
37
+ };
38
+ /**
39
+ * @function llmsTxt
40
+ * @param options {LlmsTxtOptions} title, summary, notes and link sections
41
+ * @returns {string} the llms.txt body, ending in one newline
42
+ * @throws {Error} when the title, summary, a heading, a link title or a URL is blank
43
+ */
44
+ export declare const llmsTxt: ({ title, summary, notes, sections }: LlmsTxtOptions) => string;
45
+ /**
46
+ * @function llmsFull
47
+ * @param parts {readonly LlmsFullPart[]} the documents, in order, as Markdown
48
+ * @param head {{ title: string; summary?: string }} an H1 and summary for the whole file
49
+ * @returns {string} one Markdown file: the head, then each document with its source URL,
50
+ * separated by horizontal rules, ending in one newline. Each document's Markdown is
51
+ * kept as is (code blocks included).
52
+ */
53
+ export declare const llmsFull: (parts: readonly LlmsFullPart[], head?: {
54
+ title: string;
55
+ summary?: string;
56
+ }) => string;
57
+ /** textResponse's options. */
58
+ export type TextResponseOptions = {
59
+ /** Browser cache seconds. Default 3600. */
60
+ maxAge?: number;
61
+ /** CDN cache seconds. Defaults to maxAge. */
62
+ sMaxAge?: number;
63
+ /** Default "text/plain". /docs/*.md mirrors use "text/markdown". */
64
+ type?: "text/plain" | "text/markdown";
65
+ };
66
+ /**
67
+ * @function textResponse
68
+ * @param body {string} the text
69
+ * @param options {TextResponseOptions} cache lifetimes and type
70
+ * @returns {Response} 200 with `<type>; charset=utf-8`, public Cache-Control and nosniff
71
+ */
72
+ export declare const textResponse: (body: string, options?: TextResponseOptions) => Response;
@@ -0,0 +1,85 @@
1
+ /**
2
+ * @file src/seo/llms.ts
3
+ * @desc /llms.txt (https://llmstxt.org): an H1, a blockquote summary, note paragraphs, then H2
4
+ * sections of "- [title](url): note" links. /llms-full.txt: whole documents in one file.
5
+ * And the text/plain response both are served with. One builder, so escaping and section
6
+ * format stay the same on every site (audit A2).
7
+ * @author David @dvhsh (https://dvh.sh)
8
+ * @created Mon Sep 28, 2026
9
+ * @modified Mon Sep 28, 2026
10
+ */
11
+ const oneLine = (text) => text.replace(/\s+/g, " ").trim();
12
+ /** Link text can't end the link early: brackets and backslashes are escaped. */
13
+ const linkText = (text) => oneLine(text).replace(/[\\[\]]/g, "\\$&");
14
+ /** A URL can't hold a space or end the link early: those are percent-encoded. */
15
+ const linkUrl = (url) => url.trim().replace(/[\s()<>]/g, (char) => `%${char.charCodeAt(0).toString(16).toUpperCase()}`);
16
+ const required = (value, what) => {
17
+ const line = oneLine(value);
18
+ if (!line)
19
+ throw new Error(`seo: llms.txt ${what} can't be blank`);
20
+ return line;
21
+ };
22
+ /**
23
+ * @function llmsTxt
24
+ * @param options {LlmsTxtOptions} title, summary, notes and link sections
25
+ * @returns {string} the llms.txt body, ending in one newline
26
+ * @throws {Error} when the title, summary, a heading, a link title or a URL is blank
27
+ */
28
+ export const llmsTxt = ({ title, summary, notes = [], sections }) => {
29
+ const lines = [`# ${required(title, "title")}`, "", `> ${required(summary, "summary")}`];
30
+ for (const note of notes) {
31
+ const line = oneLine(note);
32
+ if (line)
33
+ lines.push("", line);
34
+ }
35
+ for (const { heading, links } of sections) {
36
+ if (!links.length)
37
+ continue;
38
+ lines.push("", `## ${required(heading, "heading")}`, "");
39
+ for (const link of links) {
40
+ const note = link.note ? oneLine(link.note) : "";
41
+ const text = linkText(required(link.title, "link title"));
42
+ const href = linkUrl(required(link.url, "link URL"));
43
+ lines.push(`- [${text}](${href})${note ? `: ${note}` : ""}`);
44
+ }
45
+ }
46
+ return `${lines.join("\n")}\n`;
47
+ };
48
+ /**
49
+ * @function llmsFull
50
+ * @param parts {readonly LlmsFullPart[]} the documents, in order, as Markdown
51
+ * @param head {{ title: string; summary?: string }} an H1 and summary for the whole file
52
+ * @returns {string} one Markdown file: the head, then each document with its source URL,
53
+ * separated by horizontal rules, ending in one newline. Each document's Markdown is
54
+ * kept as is (code blocks included).
55
+ */
56
+ export const llmsFull = (parts, head) => {
57
+ const blocks = [];
58
+ if (head) {
59
+ blocks.push([
60
+ `# ${required(head.title, "title")}`,
61
+ ...(head.summary ? ["", `> ${oneLine(head.summary)}`] : []),
62
+ ].join("\n"));
63
+ }
64
+ for (const part of parts) {
65
+ const source = part.url ? `\n\nSource: ${linkUrl(part.url)}` : "";
66
+ blocks.push(`# ${required(part.title, "part title")}${source}\n\n${part.markdown.trim()}`);
67
+ }
68
+ return `${blocks.join("\n\n---\n\n")}\n`;
69
+ };
70
+ /**
71
+ * @function textResponse
72
+ * @param body {string} the text
73
+ * @param options {TextResponseOptions} cache lifetimes and type
74
+ * @returns {Response} 200 with `<type>; charset=utf-8`, public Cache-Control and nosniff
75
+ */
76
+ export const textResponse = (body, options = {}) => {
77
+ const maxAge = options.maxAge ?? 3600;
78
+ return new Response(body, {
79
+ headers: {
80
+ "Content-Type": `${options.type ?? "text/plain"}; charset=utf-8`,
81
+ "Cache-Control": `public, max-age=${maxAge}, s-maxage=${options.sMaxAge ?? maxAge}`,
82
+ "X-Content-Type-Options": "nosniff",
83
+ },
84
+ });
85
+ };
@@ -0,0 +1,66 @@
1
+ /**
2
+ * @file src/seo/metadata.ts
3
+ * @desc Next.js Metadata for the root layout, the home page, every other page and not-found.
4
+ * A page's openGraph replaces the layout's whole object in Next (it isn't merged), so
5
+ * pageMetadata always rebuilds it in full: site name, locale, type, url and images (audit T2),
6
+ * with the canonical and og:url set together from one path (T1, T3).
7
+ * @author David @dvhsh (https://dvh.sh)
8
+ * @created Mon Sep 28, 2026
9
+ * @modified Mon Sep 28, 2026
10
+ */
11
+ import type { Metadata } from "next";
12
+ import { type OgImage, type Site } from "./site.js";
13
+ /** pageMetadata's input. */
14
+ export type PageMetadataOptions = {
15
+ /** The page's path, like "/search". Canonical and og:url both come from it. */
16
+ path: string;
17
+ /** The primary keyword; " · host" is added (pageTitle). */
18
+ title: string;
19
+ /** Defaults to the site's description. Clamped to 160 characters. */
20
+ description?: string;
21
+ /** false: noindex, follow. Default true. */
22
+ index?: boolean;
23
+ /** Default "website". "article" also writes the published and modified times. */
24
+ ogType?: "website" | "article";
25
+ /** Defaults to the site's ogImages, so a page never loses its preview. */
26
+ images?: readonly OgImage[];
27
+ modifiedTime?: string | Date;
28
+ publishedTime?: string | Date;
29
+ };
30
+ /**
31
+ * @function siteMetadata
32
+ * @param site {Site} the site
33
+ * @returns {Metadata} the root layout's metadata: metadataBase, the title default and template,
34
+ * description, applicationName, openGraph (type, site name, locale, images) and the
35
+ * twitter card. No canonical: a layout canonical would leak to every child page.
36
+ * Icons stay with the app (its icon files).
37
+ */
38
+ export declare const siteMetadata: (site: Site) => Metadata;
39
+ /**
40
+ * @function pageMetadata
41
+ * @param site {Site} the site
42
+ * @param options {PageMetadataOptions} the page
43
+ * @returns {Metadata} an absolute "keyword · host" title, the clamped description, the
44
+ * canonical and og:url (always together, as absolute URLs), a full openGraph and
45
+ * twitter card, and robots noindex when index is false
46
+ * @throws {Error} when path doesn't start with "/" or the title is blank
47
+ */
48
+ export declare const pageMetadata: (site: Site, options: PageMetadataOptions) => Metadata;
49
+ /**
50
+ * @function homeMetadata
51
+ * @param site {Site} the site
52
+ * @param options {{ title?: string; description?: string }} overrides for the site's own
53
+ * @returns {Metadata} pageMetadata for "/": "keyword · host", canonical and og:url on the origin
54
+ */
55
+ export declare const homeMetadata: (site: Site, options?: {
56
+ title?: string;
57
+ description?: string;
58
+ }) => Metadata;
59
+ /**
60
+ * @function notFoundMetadata
61
+ * @param site {Site} the site
62
+ * @param what {string} what's missing, like "Pack" (default "Page")
63
+ * @returns {Metadata} "Pack not found · host", noindex, and no canonical (audit T8: missing
64
+ * records returned {} and got the site's default title)
65
+ */
66
+ export declare const notFoundMetadata: (site: Site, what?: string) => Metadata;
@@ -0,0 +1,106 @@
1
+ /**
2
+ * @file src/seo/metadata.ts
3
+ * @desc Next.js Metadata for the root layout, the home page, every other page and not-found.
4
+ * A page's openGraph replaces the layout's whole object in Next (it isn't merged), so
5
+ * pageMetadata always rebuilds it in full: site name, locale, type, url and images (audit T2),
6
+ * with the canonical and og:url set together from one path (T1, T3).
7
+ * @author David @dvhsh (https://dvh.sh)
8
+ * @created Mon Sep 28, 2026
9
+ * @modified Mon Sep 28, 2026
10
+ */
11
+ import { clampDescription } from "./describe.js";
12
+ import { absoluteUrl, compact, isoDate, pageTitle, TITLE_SEPARATOR, titleSuffix, } from "./site.js";
13
+ const DEFAULT_LOCALE = "en_US";
14
+ const twitter = (site, extra = {}) => compact({
15
+ card: "summary_large_image",
16
+ site: site.twitter?.site,
17
+ creator: site.twitter?.creator,
18
+ ...extra,
19
+ });
20
+ /**
21
+ * @function siteMetadata
22
+ * @param site {Site} the site
23
+ * @returns {Metadata} the root layout's metadata: metadataBase, the title default and template,
24
+ * description, applicationName, openGraph (type, site name, locale, images) and the
25
+ * twitter card. No canonical: a layout canonical would leak to every child page.
26
+ * Icons stay with the app (its icon files).
27
+ */
28
+ export const siteMetadata = (site) => {
29
+ return {
30
+ metadataBase: new URL(site.url),
31
+ title: {
32
+ default: pageTitle(site, site.title),
33
+ template: `%s${TITLE_SEPARATOR}${titleSuffix(site)}`,
34
+ },
35
+ description: clampDescription(site.description),
36
+ applicationName: site.name,
37
+ openGraph: {
38
+ type: "website",
39
+ siteName: site.name,
40
+ locale: site.locale ?? DEFAULT_LOCALE,
41
+ images: [...site.ogImages],
42
+ },
43
+ twitter: twitter(site, { images: site.ogImages.map((image) => image.url) }),
44
+ };
45
+ };
46
+ /**
47
+ * @function pageMetadata
48
+ * @param site {Site} the site
49
+ * @param options {PageMetadataOptions} the page
50
+ * @returns {Metadata} an absolute "keyword · host" title, the clamped description, the
51
+ * canonical and og:url (always together, as absolute URLs), a full openGraph and
52
+ * twitter card, and robots noindex when index is false
53
+ * @throws {Error} when path doesn't start with "/" or the title is blank
54
+ */
55
+ export const pageMetadata = (site, options) => {
56
+ const url = absoluteUrl(site, options.path);
57
+ const title = pageTitle(site, options.title);
58
+ const description = clampDescription(options.description ?? site.description);
59
+ const images = [...(options.images ?? site.ogImages)];
60
+ const base = {
61
+ title,
62
+ description,
63
+ url,
64
+ siteName: site.name,
65
+ locale: site.locale ?? DEFAULT_LOCALE,
66
+ images,
67
+ };
68
+ const openGraph = options.ogType === "article"
69
+ ? compact({
70
+ ...base,
71
+ type: "article",
72
+ publishedTime: isoDate(options.publishedTime),
73
+ modifiedTime: isoDate(options.modifiedTime),
74
+ })
75
+ : { ...base, type: "website" };
76
+ return {
77
+ title: { absolute: title },
78
+ description,
79
+ alternates: { canonical: url },
80
+ openGraph,
81
+ twitter: twitter(site, { title, description, images: images.map((image) => image.url) }),
82
+ ...(options.index === false ? { robots: { index: false, follow: true } } : {}),
83
+ };
84
+ };
85
+ /**
86
+ * @function homeMetadata
87
+ * @param site {Site} the site
88
+ * @param options {{ title?: string; description?: string }} overrides for the site's own
89
+ * @returns {Metadata} pageMetadata for "/": "keyword · host", canonical and og:url on the origin
90
+ */
91
+ export const homeMetadata = (site, options = {}) => pageMetadata(site, {
92
+ path: "/",
93
+ title: options.title ?? site.title,
94
+ ...(options.description === undefined ? {} : { description: options.description }),
95
+ });
96
+ /**
97
+ * @function notFoundMetadata
98
+ * @param site {Site} the site
99
+ * @param what {string} what's missing, like "Pack" (default "Page")
100
+ * @returns {Metadata} "Pack not found · host", noindex, and no canonical (audit T8: missing
101
+ * records returned {} and got the site's default title)
102
+ */
103
+ export const notFoundMetadata = (site, what = "Page") => ({
104
+ title: { absolute: pageTitle(site, `${what} not found`) },
105
+ robots: { index: false, follow: false },
106
+ });
@@ -0,0 +1,43 @@
1
+ /**
2
+ * @file src/seo/robots.ts
3
+ * @desc robots.txt with the AI stance written down (audit A1). A crawler that matches a named
4
+ * group ignores the `*` group (RFC 9309), so every allowed AI bot gets a group with the same
5
+ * rules as `*`, and a blocked one gets `Disallow: /`. Changing the stance is one word.
6
+ * @author David @dvhsh (https://dvh.sh)
7
+ * @created Mon Sep 28, 2026
8
+ * @modified Mon Sep 28, 2026
9
+ */
10
+ import type { MetadataRoute } from "next";
11
+ import { type Site } from "./site.js";
12
+ /** What a bot fetches for: model training, or search and answers (user fetches included). */
13
+ export type AiBotKind = "training" | "search";
14
+ /** One AI or search crawler, by its robots.txt user-agent token. */
15
+ export type AiBot = {
16
+ userAgent: string;
17
+ operator: string;
18
+ kind: AiBotKind;
19
+ /** A web search engine's own crawler: "block-all" keeps it, or the site leaves that engine. */
20
+ searchEngine?: true;
21
+ };
22
+ /** The crawlers robots() names. Order is the order their groups list them. */
23
+ export declare const AI_BOTS: readonly AiBot[];
24
+ /** The AI stance: allow every bot, block training only, or block every AI bot. */
25
+ export type AiBotsPolicy = "allow" | "block-training" | "block-all";
26
+ /** robots()'s options. */
27
+ export type RobotsOptions = {
28
+ /** Paths every crawler may fetch. Default ["/"]. */
29
+ allow?: readonly string[];
30
+ /** Private paths, like "/api/" or "/admin". Longest match wins over allow. */
31
+ disallow?: readonly string[];
32
+ /** Default "allow". */
33
+ aiBots?: AiBotsPolicy;
34
+ };
35
+ /**
36
+ * @function robots
37
+ * @param site {Site} the site (sitemap and host come from its origin)
38
+ * @param options {RobotsOptions} allow, disallow and the AI stance
39
+ * @returns {MetadataRoute.Robots} the `*` group, one group for the allowed AI bots with the same
40
+ * rules (Bingbot is always there), one `Disallow: /` group for the blocked ones (when
41
+ * any are), the sitemap URL and the host
42
+ */
43
+ export declare const robots: (site: Site, options?: RobotsOptions) => MetadataRoute.Robots;
@@ -0,0 +1,55 @@
1
+ /**
2
+ * @file src/seo/robots.ts
3
+ * @desc robots.txt with the AI stance written down (audit A1). A crawler that matches a named
4
+ * group ignores the `*` group (RFC 9309), so every allowed AI bot gets a group with the same
5
+ * rules as `*`, and a blocked one gets `Disallow: /`. Changing the stance is one word.
6
+ * @author David @dvhsh (https://dvh.sh)
7
+ * @created Mon Sep 28, 2026
8
+ * @modified Mon Sep 28, 2026
9
+ */
10
+ import { absoluteUrl, origin } from "./site.js";
11
+ /** The crawlers robots() names. Order is the order their groups list them. */
12
+ export const AI_BOTS = [
13
+ { userAgent: "GPTBot", operator: "OpenAI", kind: "training" },
14
+ { userAgent: "OAI-SearchBot", operator: "OpenAI", kind: "search" },
15
+ { userAgent: "ChatGPT-User", operator: "OpenAI", kind: "search" },
16
+ { userAgent: "PerplexityBot", operator: "Perplexity", kind: "search" },
17
+ { userAgent: "Perplexity-User", operator: "Perplexity", kind: "search" },
18
+ { userAgent: "ClaudeBot", operator: "Anthropic", kind: "training" },
19
+ { userAgent: "Claude-SearchBot", operator: "Anthropic", kind: "search" },
20
+ { userAgent: "Claude-User", operator: "Anthropic", kind: "search" },
21
+ { userAgent: "anthropic-ai", operator: "Anthropic", kind: "training" },
22
+ { userAgent: "Google-Extended", operator: "Google", kind: "training" },
23
+ { userAgent: "Applebot-Extended", operator: "Apple", kind: "training" },
24
+ { userAgent: "Bingbot", operator: "Microsoft", kind: "search", searchEngine: true },
25
+ { userAgent: "CCBot", operator: "Common Crawl", kind: "training" },
26
+ { userAgent: "Bytespider", operator: "ByteDance", kind: "training" },
27
+ { userAgent: "meta-externalagent", operator: "Meta", kind: "training" },
28
+ ];
29
+ const blocked = (bot, policy) => policy === "block-all"
30
+ ? !bot.searchEngine
31
+ : policy === "block-training" && bot.kind === "training";
32
+ /**
33
+ * @function robots
34
+ * @param site {Site} the site (sitemap and host come from its origin)
35
+ * @param options {RobotsOptions} allow, disallow and the AI stance
36
+ * @returns {MetadataRoute.Robots} the `*` group, one group for the allowed AI bots with the same
37
+ * rules (Bingbot is always there), one `Disallow: /` group for the blocked ones (when
38
+ * any are), the sitemap URL and the host
39
+ */
40
+ export const robots = (site, options = {}) => {
41
+ const policy = options.aiBots ?? "allow";
42
+ const allow = [...(options.allow ?? ["/"])];
43
+ const rules = { allow, ...(options.disallow?.length ? { disallow: [...options.disallow] } : {}) };
44
+ const allowed = AI_BOTS.filter((bot) => !blocked(bot, policy)).map((bot) => bot.userAgent);
45
+ const denied = AI_BOTS.filter((bot) => blocked(bot, policy)).map((bot) => bot.userAgent);
46
+ return {
47
+ rules: [
48
+ { userAgent: "*", ...rules },
49
+ { userAgent: allowed, ...rules },
50
+ ...(denied.length ? [{ userAgent: denied, disallow: "/" }] : []),
51
+ ],
52
+ sitemap: absoluteUrl(site, "/sitemap.xml"),
53
+ host: origin(site.url),
54
+ };
55
+ };