@haruhimemoe/next-kit 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -1
- package/README.md +59 -1
- package/dist/seo/describe.d.ts +20 -0
- package/dist/seo/describe.js +34 -0
- package/dist/seo/index.d.ts +19 -0
- package/dist/seo/index.js +18 -0
- package/dist/seo/ld-content.d.ts +105 -0
- package/dist/seo/ld-content.js +120 -0
- package/dist/seo/ld-site.d.ts +96 -0
- package/dist/seo/ld-site.js +135 -0
- package/dist/seo/ld.d.ts +32 -0
- package/dist/seo/ld.js +39 -0
- package/dist/seo/llms.d.ts +72 -0
- package/dist/seo/llms.js +85 -0
- package/dist/seo/metadata.d.ts +66 -0
- package/dist/seo/metadata.js +106 -0
- package/dist/seo/robots.d.ts +43 -0
- package/dist/seo/robots.js +55 -0
- package/dist/seo/site.d.ts +108 -0
- package/dist/seo/site.js +95 -0
- package/dist/seo/sitemap.d.ts +39 -0
- package/dist/seo/sitemap.js +50 -0
- package/package.json +12 -4
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/ld-site.ts
|
|
3
|
+
* @desc schema.org JSON-LD for the site itself: the graph wrapper (the only place `@context`
|
|
4
|
+
* goes), Organization, WebSite with its SearchAction, WebApplication, BreadcrumbList and
|
|
5
|
+
* ItemList. Stable @ids tie every tool to one organization (audit A3, A6).
|
|
6
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
7
|
+
* @created Mon Sep 28, 2026
|
|
8
|
+
* @modified Mon Sep 28, 2026
|
|
9
|
+
*/
|
|
10
|
+
import { absoluteUrl, compact, nodeId } from "./site.js";
|
|
11
|
+
/** The placeholder a SearchAction's urlTemplate must hold. */
|
|
12
|
+
export const SEARCH_TERM = "{search_term_string}";
|
|
13
|
+
/**
|
|
14
|
+
* @function graph
|
|
15
|
+
* @param nodes {LdNode[]} the page's nodes
|
|
16
|
+
* @returns {LdGraph} one JSON-LD document with `@context` on the root only
|
|
17
|
+
*/
|
|
18
|
+
export const graph = (...nodes) => ({
|
|
19
|
+
"@context": "https://schema.org",
|
|
20
|
+
"@graph": nodes,
|
|
21
|
+
});
|
|
22
|
+
/**
|
|
23
|
+
* @function organization
|
|
24
|
+
* @param org {Organization} the organization, like HARUHIME_ORG
|
|
25
|
+
* @returns {LdNode} Organization with @id `${origin}/#organization`
|
|
26
|
+
*/
|
|
27
|
+
export const organization = (org) => compact({
|
|
28
|
+
"@type": "Organization",
|
|
29
|
+
"@id": nodeId(org.url, "organization"),
|
|
30
|
+
name: org.name,
|
|
31
|
+
url: org.url,
|
|
32
|
+
logo: org.logo,
|
|
33
|
+
email: org.email,
|
|
34
|
+
sameAs: org.sameAs?.length ? [...org.sameAs] : undefined,
|
|
35
|
+
});
|
|
36
|
+
/** A reference to the site's organization by @id. */
|
|
37
|
+
export const orgRef = (site) => ({
|
|
38
|
+
"@id": nodeId(site.organization.url, "organization"),
|
|
39
|
+
});
|
|
40
|
+
/**
|
|
41
|
+
* @function webSite
|
|
42
|
+
* @param site {Site} the site
|
|
43
|
+
* @param options {{ searchUrlTemplate?: string }} a search URL holding {search_term_string},
|
|
44
|
+
* like "/search?q={search_term_string}"
|
|
45
|
+
* @returns {LdNode} WebSite with @id `${origin}/#website`, publisher → the organization,
|
|
46
|
+
* isPartOf → the parent site, and a SearchAction when a template is given
|
|
47
|
+
* @throws {Error} when the template lacks {search_term_string} or isn't on the site
|
|
48
|
+
*/
|
|
49
|
+
export const webSite = (site, options = {}) => {
|
|
50
|
+
const template = options.searchUrlTemplate;
|
|
51
|
+
if (template !== undefined && !template.includes(SEARCH_TERM)) {
|
|
52
|
+
throw new Error(`seo: searchUrlTemplate must hold ${SEARCH_TERM}`);
|
|
53
|
+
}
|
|
54
|
+
return compact({
|
|
55
|
+
"@type": "WebSite",
|
|
56
|
+
"@id": nodeId(site.url, "website"),
|
|
57
|
+
name: site.name,
|
|
58
|
+
alternateName: new URL(site.url).host,
|
|
59
|
+
url: absoluteUrl(site, "/"),
|
|
60
|
+
description: site.description,
|
|
61
|
+
inLanguage: (site.locale ?? "en_US").replace("_", "-"),
|
|
62
|
+
publisher: orgRef(site),
|
|
63
|
+
isPartOf: site.parent ? { "@id": nodeId(site.parent.url, "website") } : undefined,
|
|
64
|
+
potentialAction: template
|
|
65
|
+
? {
|
|
66
|
+
"@type": "SearchAction",
|
|
67
|
+
target: { "@type": "EntryPoint", urlTemplate: absoluteUrl(site, template) },
|
|
68
|
+
"query-input": "required name=search_term_string",
|
|
69
|
+
}
|
|
70
|
+
: undefined,
|
|
71
|
+
});
|
|
72
|
+
};
|
|
73
|
+
/**
|
|
74
|
+
* @function webApplication
|
|
75
|
+
* @param site {Site} the site
|
|
76
|
+
* @param options {WebApplicationOptions} category, and optional name, features and path
|
|
77
|
+
* @returns {LdNode} WebApplication, free (Offer price 0), operatingSystem "Any", publisher →
|
|
78
|
+
* the organization
|
|
79
|
+
*/
|
|
80
|
+
export const webApplication = (site, options) => {
|
|
81
|
+
const url = absoluteUrl(site, options.path ?? "/");
|
|
82
|
+
return compact({
|
|
83
|
+
"@type": "WebApplication",
|
|
84
|
+
"@id": options.path && options.path !== "/" ? `${url}#app` : nodeId(site.url, "app"),
|
|
85
|
+
name: options.name ?? site.name,
|
|
86
|
+
url,
|
|
87
|
+
description: options.description ?? site.description,
|
|
88
|
+
applicationCategory: options.category,
|
|
89
|
+
operatingSystem: "Any",
|
|
90
|
+
browserRequirements: options.browserRequirements ?? "Requires JavaScript. Works in any modern browser.",
|
|
91
|
+
featureList: options.features?.length ? [...options.features] : undefined,
|
|
92
|
+
isAccessibleForFree: true,
|
|
93
|
+
offers: { "@type": "Offer", price: "0", priceCurrency: "USD" },
|
|
94
|
+
publisher: orgRef(site),
|
|
95
|
+
isPartOf: { "@id": nodeId(site.url, "website") },
|
|
96
|
+
});
|
|
97
|
+
};
|
|
98
|
+
/**
|
|
99
|
+
* @function breadcrumbs
|
|
100
|
+
* @param site {Site} the site
|
|
101
|
+
* @param trail {readonly LdLink[]} from the home page down to this page
|
|
102
|
+
* @returns {LdNode} BreadcrumbList, positions from 1, absolute item URLs
|
|
103
|
+
* @throws {Error} when the trail is empty
|
|
104
|
+
*/
|
|
105
|
+
export const breadcrumbs = (site, trail) => {
|
|
106
|
+
if (!trail.length)
|
|
107
|
+
throw new Error("seo: breadcrumbs need at least one page");
|
|
108
|
+
return {
|
|
109
|
+
"@type": "BreadcrumbList",
|
|
110
|
+
itemListElement: trail.map((link, i) => ({
|
|
111
|
+
"@type": "ListItem",
|
|
112
|
+
position: i + 1,
|
|
113
|
+
name: link.name,
|
|
114
|
+
item: absoluteUrl(site, link.path),
|
|
115
|
+
})),
|
|
116
|
+
};
|
|
117
|
+
};
|
|
118
|
+
/**
|
|
119
|
+
* @function itemList
|
|
120
|
+
* @param site {Site} the site
|
|
121
|
+
* @param items {readonly LdLink[]} the listed pages, in order
|
|
122
|
+
* @param options {{ name?: string }} the list's name
|
|
123
|
+
* @returns {LdNode} ItemList with numberOfItems and ListItems (position, name, url)
|
|
124
|
+
*/
|
|
125
|
+
export const itemList = (site, items, options = {}) => compact({
|
|
126
|
+
"@type": "ItemList",
|
|
127
|
+
name: options.name,
|
|
128
|
+
numberOfItems: items.length,
|
|
129
|
+
itemListElement: items.map((item, i) => ({
|
|
130
|
+
"@type": "ListItem",
|
|
131
|
+
position: i + 1,
|
|
132
|
+
name: item.name,
|
|
133
|
+
url: absoluteUrl(site, item.path),
|
|
134
|
+
})),
|
|
135
|
+
});
|
package/dist/seo/ld.d.ts
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/ld.ts
|
|
3
|
+
* @desc The `ld` namespace of JSON-LD builders, and serializeLd: JSON that is safe inside a
|
|
4
|
+
* `<script>` tag. @haruhimemoe/ui's JsonLd escapes "<" only; serializeLd also escapes ">",
|
|
5
|
+
* "&" and U+2028/U+2029, so the output is safe in any HTML or JavaScript context.
|
|
6
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
7
|
+
* @created Mon Sep 28, 2026
|
|
8
|
+
* @modified Mon Sep 28, 2026
|
|
9
|
+
*/
|
|
10
|
+
import { creativeWork, dataset, faq, howTo, techArticle } from "./ld-content.js";
|
|
11
|
+
import { breadcrumbs, graph, itemList, organization, webApplication, webSite } from "./ld-site.js";
|
|
12
|
+
/** The JSON-LD builders: plain schema.org objects, rendered with @haruhimemoe/ui's JsonLd. */
|
|
13
|
+
export declare const ld: {
|
|
14
|
+
readonly graph: typeof graph;
|
|
15
|
+
readonly organization: typeof organization;
|
|
16
|
+
readonly webSite: typeof webSite;
|
|
17
|
+
readonly webApplication: typeof webApplication;
|
|
18
|
+
readonly breadcrumbs: typeof breadcrumbs;
|
|
19
|
+
readonly itemList: typeof itemList;
|
|
20
|
+
readonly faq: typeof faq;
|
|
21
|
+
readonly howTo: typeof howTo;
|
|
22
|
+
readonly techArticle: typeof techArticle;
|
|
23
|
+
readonly creativeWork: typeof creativeWork;
|
|
24
|
+
readonly dataset: typeof dataset;
|
|
25
|
+
};
|
|
26
|
+
/**
|
|
27
|
+
* @function serializeLd
|
|
28
|
+
* @param data {object} a JSON-LD document or node
|
|
29
|
+
* @returns {string} JSON with <, >, & and U+2028/U+2029 written as \u escapes: the same data to
|
|
30
|
+
* a JSON parser, and nothing a `<script>` tag or an HTML comment can end on
|
|
31
|
+
*/
|
|
32
|
+
export declare const serializeLd: (data: object) => string;
|
package/dist/seo/ld.js
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/ld.ts
|
|
3
|
+
* @desc The `ld` namespace of JSON-LD builders, and serializeLd: JSON that is safe inside a
|
|
4
|
+
* `<script>` tag. @haruhimemoe/ui's JsonLd escapes "<" only; serializeLd also escapes ">",
|
|
5
|
+
* "&" and U+2028/U+2029, so the output is safe in any HTML or JavaScript context.
|
|
6
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
7
|
+
* @created Mon Sep 28, 2026
|
|
8
|
+
* @modified Mon Sep 28, 2026
|
|
9
|
+
*/
|
|
10
|
+
import { creativeWork, dataset, faq, howTo, techArticle } from "./ld-content.js";
|
|
11
|
+
import { breadcrumbs, graph, itemList, organization, webApplication, webSite } from "./ld-site.js";
|
|
12
|
+
/** The JSON-LD builders: plain schema.org objects, rendered with @haruhimemoe/ui's JsonLd. */
|
|
13
|
+
export const ld = {
|
|
14
|
+
graph,
|
|
15
|
+
organization,
|
|
16
|
+
webSite,
|
|
17
|
+
webApplication,
|
|
18
|
+
breadcrumbs,
|
|
19
|
+
itemList,
|
|
20
|
+
faq,
|
|
21
|
+
howTo,
|
|
22
|
+
techArticle,
|
|
23
|
+
creativeWork,
|
|
24
|
+
dataset,
|
|
25
|
+
};
|
|
26
|
+
const ESCAPES = {
|
|
27
|
+
"<": "\\u003c",
|
|
28
|
+
">": "\\u003e",
|
|
29
|
+
"&": "\\u0026",
|
|
30
|
+
"\u2028": "\\u2028",
|
|
31
|
+
"\u2029": "\\u2029",
|
|
32
|
+
};
|
|
33
|
+
/**
|
|
34
|
+
* @function serializeLd
|
|
35
|
+
* @param data {object} a JSON-LD document or node
|
|
36
|
+
* @returns {string} JSON with <, >, & and U+2028/U+2029 written as \u escapes: the same data to
|
|
37
|
+
* a JSON parser, and nothing a `<script>` tag or an HTML comment can end on
|
|
38
|
+
*/
|
|
39
|
+
export const serializeLd = (data) => JSON.stringify(data).replace(/[<>&\u2028\u2029]/g, (char) => ESCAPES[char]);
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/llms.ts
|
|
3
|
+
* @desc /llms.txt (https://llmstxt.org): an H1, a blockquote summary, note paragraphs, then H2
|
|
4
|
+
* sections of "- [title](url): note" links. /llms-full.txt: whole documents in one file.
|
|
5
|
+
* And the text/plain response both are served with. One builder, so escaping and section
|
|
6
|
+
* format stay the same on every site (audit A2).
|
|
7
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
8
|
+
* @created Mon Sep 28, 2026
|
|
9
|
+
* @modified Mon Sep 28, 2026
|
|
10
|
+
*/
|
|
11
|
+
/** One link in a section. */
|
|
12
|
+
export type LlmsLink = {
|
|
13
|
+
title: string;
|
|
14
|
+
url: string;
|
|
15
|
+
note?: string;
|
|
16
|
+
};
|
|
17
|
+
/** An H2 section of links. A section with no links is left out. */
|
|
18
|
+
export type LlmsSection = {
|
|
19
|
+
heading: string;
|
|
20
|
+
links: readonly LlmsLink[];
|
|
21
|
+
};
|
|
22
|
+
/** llmsTxt's input. */
|
|
23
|
+
export type LlmsTxtOptions = {
|
|
24
|
+
/** The H1, like "pools.haruhime.moe". */
|
|
25
|
+
title: string;
|
|
26
|
+
/** The blockquote: what the site is, in one paragraph. */
|
|
27
|
+
summary: string;
|
|
28
|
+
/** Paragraphs a reader needs before following any link. */
|
|
29
|
+
notes?: readonly string[];
|
|
30
|
+
sections: readonly LlmsSection[];
|
|
31
|
+
};
|
|
32
|
+
/** One document in llms-full.txt. */
|
|
33
|
+
export type LlmsFullPart = {
|
|
34
|
+
title: string;
|
|
35
|
+
url?: string;
|
|
36
|
+
markdown: string;
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* @function llmsTxt
|
|
40
|
+
* @param options {LlmsTxtOptions} title, summary, notes and link sections
|
|
41
|
+
* @returns {string} the llms.txt body, ending in one newline
|
|
42
|
+
* @throws {Error} when the title, summary, a heading, a link title or a URL is blank
|
|
43
|
+
*/
|
|
44
|
+
export declare const llmsTxt: ({ title, summary, notes, sections }: LlmsTxtOptions) => string;
|
|
45
|
+
/**
|
|
46
|
+
* @function llmsFull
|
|
47
|
+
* @param parts {readonly LlmsFullPart[]} the documents, in order, as Markdown
|
|
48
|
+
* @param head {{ title: string; summary?: string }} an H1 and summary for the whole file
|
|
49
|
+
* @returns {string} one Markdown file: the head, then each document with its source URL,
|
|
50
|
+
* separated by horizontal rules, ending in one newline. Each document's Markdown is
|
|
51
|
+
* kept as is (code blocks included).
|
|
52
|
+
*/
|
|
53
|
+
export declare const llmsFull: (parts: readonly LlmsFullPart[], head?: {
|
|
54
|
+
title: string;
|
|
55
|
+
summary?: string;
|
|
56
|
+
}) => string;
|
|
57
|
+
/** textResponse's options. */
|
|
58
|
+
export type TextResponseOptions = {
|
|
59
|
+
/** Browser cache seconds. Default 3600. */
|
|
60
|
+
maxAge?: number;
|
|
61
|
+
/** CDN cache seconds. Defaults to maxAge. */
|
|
62
|
+
sMaxAge?: number;
|
|
63
|
+
/** Default "text/plain". /docs/*.md mirrors use "text/markdown". */
|
|
64
|
+
type?: "text/plain" | "text/markdown";
|
|
65
|
+
};
|
|
66
|
+
/**
|
|
67
|
+
* @function textResponse
|
|
68
|
+
* @param body {string} the text
|
|
69
|
+
* @param options {TextResponseOptions} cache lifetimes and type
|
|
70
|
+
* @returns {Response} 200 with `<type>; charset=utf-8`, public Cache-Control and nosniff
|
|
71
|
+
*/
|
|
72
|
+
export declare const textResponse: (body: string, options?: TextResponseOptions) => Response;
|
package/dist/seo/llms.js
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/llms.ts
|
|
3
|
+
* @desc /llms.txt (https://llmstxt.org): an H1, a blockquote summary, note paragraphs, then H2
|
|
4
|
+
* sections of "- [title](url): note" links. /llms-full.txt: whole documents in one file.
|
|
5
|
+
* And the text/plain response both are served with. One builder, so escaping and section
|
|
6
|
+
* format stay the same on every site (audit A2).
|
|
7
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
8
|
+
* @created Mon Sep 28, 2026
|
|
9
|
+
* @modified Mon Sep 28, 2026
|
|
10
|
+
*/
|
|
11
|
+
const oneLine = (text) => text.replace(/\s+/g, " ").trim();
|
|
12
|
+
/** Link text can't end the link early: brackets and backslashes are escaped. */
|
|
13
|
+
const linkText = (text) => oneLine(text).replace(/[\\[\]]/g, "\\$&");
|
|
14
|
+
/** A URL can't hold a space or end the link early: those are percent-encoded. */
|
|
15
|
+
const linkUrl = (url) => url.trim().replace(/[\s()<>]/g, (char) => `%${char.charCodeAt(0).toString(16).toUpperCase()}`);
|
|
16
|
+
const required = (value, what) => {
|
|
17
|
+
const line = oneLine(value);
|
|
18
|
+
if (!line)
|
|
19
|
+
throw new Error(`seo: llms.txt ${what} can't be blank`);
|
|
20
|
+
return line;
|
|
21
|
+
};
|
|
22
|
+
/**
|
|
23
|
+
* @function llmsTxt
|
|
24
|
+
* @param options {LlmsTxtOptions} title, summary, notes and link sections
|
|
25
|
+
* @returns {string} the llms.txt body, ending in one newline
|
|
26
|
+
* @throws {Error} when the title, summary, a heading, a link title or a URL is blank
|
|
27
|
+
*/
|
|
28
|
+
export const llmsTxt = ({ title, summary, notes = [], sections }) => {
|
|
29
|
+
const lines = [`# ${required(title, "title")}`, "", `> ${required(summary, "summary")}`];
|
|
30
|
+
for (const note of notes) {
|
|
31
|
+
const line = oneLine(note);
|
|
32
|
+
if (line)
|
|
33
|
+
lines.push("", line);
|
|
34
|
+
}
|
|
35
|
+
for (const { heading, links } of sections) {
|
|
36
|
+
if (!links.length)
|
|
37
|
+
continue;
|
|
38
|
+
lines.push("", `## ${required(heading, "heading")}`, "");
|
|
39
|
+
for (const link of links) {
|
|
40
|
+
const note = link.note ? oneLine(link.note) : "";
|
|
41
|
+
const text = linkText(required(link.title, "link title"));
|
|
42
|
+
const href = linkUrl(required(link.url, "link URL"));
|
|
43
|
+
lines.push(`- [${text}](${href})${note ? `: ${note}` : ""}`);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
return `${lines.join("\n")}\n`;
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* @function llmsFull
|
|
50
|
+
* @param parts {readonly LlmsFullPart[]} the documents, in order, as Markdown
|
|
51
|
+
* @param head {{ title: string; summary?: string }} an H1 and summary for the whole file
|
|
52
|
+
* @returns {string} one Markdown file: the head, then each document with its source URL,
|
|
53
|
+
* separated by horizontal rules, ending in one newline. Each document's Markdown is
|
|
54
|
+
* kept as is (code blocks included).
|
|
55
|
+
*/
|
|
56
|
+
export const llmsFull = (parts, head) => {
|
|
57
|
+
const blocks = [];
|
|
58
|
+
if (head) {
|
|
59
|
+
blocks.push([
|
|
60
|
+
`# ${required(head.title, "title")}`,
|
|
61
|
+
...(head.summary ? ["", `> ${oneLine(head.summary)}`] : []),
|
|
62
|
+
].join("\n"));
|
|
63
|
+
}
|
|
64
|
+
for (const part of parts) {
|
|
65
|
+
const source = part.url ? `\n\nSource: ${linkUrl(part.url)}` : "";
|
|
66
|
+
blocks.push(`# ${required(part.title, "part title")}${source}\n\n${part.markdown.trim()}`);
|
|
67
|
+
}
|
|
68
|
+
return `${blocks.join("\n\n---\n\n")}\n`;
|
|
69
|
+
};
|
|
70
|
+
/**
|
|
71
|
+
* @function textResponse
|
|
72
|
+
* @param body {string} the text
|
|
73
|
+
* @param options {TextResponseOptions} cache lifetimes and type
|
|
74
|
+
* @returns {Response} 200 with `<type>; charset=utf-8`, public Cache-Control and nosniff
|
|
75
|
+
*/
|
|
76
|
+
export const textResponse = (body, options = {}) => {
|
|
77
|
+
const maxAge = options.maxAge ?? 3600;
|
|
78
|
+
return new Response(body, {
|
|
79
|
+
headers: {
|
|
80
|
+
"Content-Type": `${options.type ?? "text/plain"}; charset=utf-8`,
|
|
81
|
+
"Cache-Control": `public, max-age=${maxAge}, s-maxage=${options.sMaxAge ?? maxAge}`,
|
|
82
|
+
"X-Content-Type-Options": "nosniff",
|
|
83
|
+
},
|
|
84
|
+
});
|
|
85
|
+
};
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/metadata.ts
|
|
3
|
+
* @desc Next.js Metadata for the root layout, the home page, every other page and not-found.
|
|
4
|
+
* A page's openGraph replaces the layout's whole object in Next (it isn't merged), so
|
|
5
|
+
* pageMetadata always rebuilds it in full: site name, locale, type, url and images (audit T2),
|
|
6
|
+
* with the canonical and og:url set together from one path (T1, T3).
|
|
7
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
8
|
+
* @created Mon Sep 28, 2026
|
|
9
|
+
* @modified Mon Sep 28, 2026
|
|
10
|
+
*/
|
|
11
|
+
import type { Metadata } from "next";
|
|
12
|
+
import { type OgImage, type Site } from "./site.js";
|
|
13
|
+
/** pageMetadata's input. */
|
|
14
|
+
export type PageMetadataOptions = {
|
|
15
|
+
/** The page's path, like "/search". Canonical and og:url both come from it. */
|
|
16
|
+
path: string;
|
|
17
|
+
/** The primary keyword; " · host" is added (pageTitle). */
|
|
18
|
+
title: string;
|
|
19
|
+
/** Defaults to the site's description. Clamped to 160 characters. */
|
|
20
|
+
description?: string;
|
|
21
|
+
/** false: noindex, follow. Default true. */
|
|
22
|
+
index?: boolean;
|
|
23
|
+
/** Default "website". "article" also writes the published and modified times. */
|
|
24
|
+
ogType?: "website" | "article";
|
|
25
|
+
/** Defaults to the site's ogImages, so a page never loses its preview. */
|
|
26
|
+
images?: readonly OgImage[];
|
|
27
|
+
modifiedTime?: string | Date;
|
|
28
|
+
publishedTime?: string | Date;
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* @function siteMetadata
|
|
32
|
+
* @param site {Site} the site
|
|
33
|
+
* @returns {Metadata} the root layout's metadata: metadataBase, the title default and template,
|
|
34
|
+
* description, applicationName, openGraph (type, site name, locale, images) and the
|
|
35
|
+
* twitter card. No canonical: a layout canonical would leak to every child page.
|
|
36
|
+
* Icons stay with the app (its icon files).
|
|
37
|
+
*/
|
|
38
|
+
export declare const siteMetadata: (site: Site) => Metadata;
|
|
39
|
+
/**
|
|
40
|
+
* @function pageMetadata
|
|
41
|
+
* @param site {Site} the site
|
|
42
|
+
* @param options {PageMetadataOptions} the page
|
|
43
|
+
* @returns {Metadata} an absolute "keyword · host" title, the clamped description, the
|
|
44
|
+
* canonical and og:url (always together, as absolute URLs), a full openGraph and
|
|
45
|
+
* twitter card, and robots noindex when index is false
|
|
46
|
+
* @throws {Error} when path doesn't start with "/" or the title is blank
|
|
47
|
+
*/
|
|
48
|
+
export declare const pageMetadata: (site: Site, options: PageMetadataOptions) => Metadata;
|
|
49
|
+
/**
|
|
50
|
+
* @function homeMetadata
|
|
51
|
+
* @param site {Site} the site
|
|
52
|
+
* @param options {{ title?: string; description?: string }} overrides for the site's own
|
|
53
|
+
* @returns {Metadata} pageMetadata for "/": "keyword · host", canonical and og:url on the origin
|
|
54
|
+
*/
|
|
55
|
+
export declare const homeMetadata: (site: Site, options?: {
|
|
56
|
+
title?: string;
|
|
57
|
+
description?: string;
|
|
58
|
+
}) => Metadata;
|
|
59
|
+
/**
|
|
60
|
+
* @function notFoundMetadata
|
|
61
|
+
* @param site {Site} the site
|
|
62
|
+
* @param what {string} what's missing, like "Pack" (default "Page")
|
|
63
|
+
* @returns {Metadata} "Pack not found · host", noindex, and no canonical (audit T8: missing
|
|
64
|
+
* records returned {} and got the site's default title)
|
|
65
|
+
*/
|
|
66
|
+
export declare const notFoundMetadata: (site: Site, what?: string) => Metadata;
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/metadata.ts
|
|
3
|
+
* @desc Next.js Metadata for the root layout, the home page, every other page and not-found.
|
|
4
|
+
* A page's openGraph replaces the layout's whole object in Next (it isn't merged), so
|
|
5
|
+
* pageMetadata always rebuilds it in full: site name, locale, type, url and images (audit T2),
|
|
6
|
+
* with the canonical and og:url set together from one path (T1, T3).
|
|
7
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
8
|
+
* @created Mon Sep 28, 2026
|
|
9
|
+
* @modified Mon Sep 28, 2026
|
|
10
|
+
*/
|
|
11
|
+
import { clampDescription } from "./describe.js";
|
|
12
|
+
import { absoluteUrl, compact, isoDate, pageTitle, TITLE_SEPARATOR, titleSuffix, } from "./site.js";
|
|
13
|
+
const DEFAULT_LOCALE = "en_US";
|
|
14
|
+
const twitter = (site, extra = {}) => compact({
|
|
15
|
+
card: "summary_large_image",
|
|
16
|
+
site: site.twitter?.site,
|
|
17
|
+
creator: site.twitter?.creator,
|
|
18
|
+
...extra,
|
|
19
|
+
});
|
|
20
|
+
/**
|
|
21
|
+
* @function siteMetadata
|
|
22
|
+
* @param site {Site} the site
|
|
23
|
+
* @returns {Metadata} the root layout's metadata: metadataBase, the title default and template,
|
|
24
|
+
* description, applicationName, openGraph (type, site name, locale, images) and the
|
|
25
|
+
* twitter card. No canonical: a layout canonical would leak to every child page.
|
|
26
|
+
* Icons stay with the app (its icon files).
|
|
27
|
+
*/
|
|
28
|
+
export const siteMetadata = (site) => {
|
|
29
|
+
return {
|
|
30
|
+
metadataBase: new URL(site.url),
|
|
31
|
+
title: {
|
|
32
|
+
default: pageTitle(site, site.title),
|
|
33
|
+
template: `%s${TITLE_SEPARATOR}${titleSuffix(site)}`,
|
|
34
|
+
},
|
|
35
|
+
description: clampDescription(site.description),
|
|
36
|
+
applicationName: site.name,
|
|
37
|
+
openGraph: {
|
|
38
|
+
type: "website",
|
|
39
|
+
siteName: site.name,
|
|
40
|
+
locale: site.locale ?? DEFAULT_LOCALE,
|
|
41
|
+
images: [...site.ogImages],
|
|
42
|
+
},
|
|
43
|
+
twitter: twitter(site, { images: site.ogImages.map((image) => image.url) }),
|
|
44
|
+
};
|
|
45
|
+
};
|
|
46
|
+
/**
|
|
47
|
+
* @function pageMetadata
|
|
48
|
+
* @param site {Site} the site
|
|
49
|
+
* @param options {PageMetadataOptions} the page
|
|
50
|
+
* @returns {Metadata} an absolute "keyword · host" title, the clamped description, the
|
|
51
|
+
* canonical and og:url (always together, as absolute URLs), a full openGraph and
|
|
52
|
+
* twitter card, and robots noindex when index is false
|
|
53
|
+
* @throws {Error} when path doesn't start with "/" or the title is blank
|
|
54
|
+
*/
|
|
55
|
+
export const pageMetadata = (site, options) => {
|
|
56
|
+
const url = absoluteUrl(site, options.path);
|
|
57
|
+
const title = pageTitle(site, options.title);
|
|
58
|
+
const description = clampDescription(options.description ?? site.description);
|
|
59
|
+
const images = [...(options.images ?? site.ogImages)];
|
|
60
|
+
const base = {
|
|
61
|
+
title,
|
|
62
|
+
description,
|
|
63
|
+
url,
|
|
64
|
+
siteName: site.name,
|
|
65
|
+
locale: site.locale ?? DEFAULT_LOCALE,
|
|
66
|
+
images,
|
|
67
|
+
};
|
|
68
|
+
const openGraph = options.ogType === "article"
|
|
69
|
+
? compact({
|
|
70
|
+
...base,
|
|
71
|
+
type: "article",
|
|
72
|
+
publishedTime: isoDate(options.publishedTime),
|
|
73
|
+
modifiedTime: isoDate(options.modifiedTime),
|
|
74
|
+
})
|
|
75
|
+
: { ...base, type: "website" };
|
|
76
|
+
return {
|
|
77
|
+
title: { absolute: title },
|
|
78
|
+
description,
|
|
79
|
+
alternates: { canonical: url },
|
|
80
|
+
openGraph,
|
|
81
|
+
twitter: twitter(site, { title, description, images: images.map((image) => image.url) }),
|
|
82
|
+
...(options.index === false ? { robots: { index: false, follow: true } } : {}),
|
|
83
|
+
};
|
|
84
|
+
};
|
|
85
|
+
/**
|
|
86
|
+
* @function homeMetadata
|
|
87
|
+
* @param site {Site} the site
|
|
88
|
+
* @param options {{ title?: string; description?: string }} overrides for the site's own
|
|
89
|
+
* @returns {Metadata} pageMetadata for "/": "keyword · host", canonical and og:url on the origin
|
|
90
|
+
*/
|
|
91
|
+
export const homeMetadata = (site, options = {}) => pageMetadata(site, {
|
|
92
|
+
path: "/",
|
|
93
|
+
title: options.title ?? site.title,
|
|
94
|
+
...(options.description === undefined ? {} : { description: options.description }),
|
|
95
|
+
});
|
|
96
|
+
/**
|
|
97
|
+
* @function notFoundMetadata
|
|
98
|
+
* @param site {Site} the site
|
|
99
|
+
* @param what {string} what's missing, like "Pack" (default "Page")
|
|
100
|
+
* @returns {Metadata} "Pack not found · host", noindex, and no canonical (audit T8: missing
|
|
101
|
+
* records returned {} and got the site's default title)
|
|
102
|
+
*/
|
|
103
|
+
export const notFoundMetadata = (site, what = "Page") => ({
|
|
104
|
+
title: { absolute: pageTitle(site, `${what} not found`) },
|
|
105
|
+
robots: { index: false, follow: false },
|
|
106
|
+
});
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/robots.ts
|
|
3
|
+
* @desc robots.txt with the AI stance written down (audit A1). A crawler that matches a named
|
|
4
|
+
* group ignores the `*` group (RFC 9309), so every allowed AI bot gets a group with the same
|
|
5
|
+
* rules as `*`, and a blocked one gets `Disallow: /`. Changing the stance is one word.
|
|
6
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
7
|
+
* @created Mon Sep 28, 2026
|
|
8
|
+
* @modified Mon Sep 28, 2026
|
|
9
|
+
*/
|
|
10
|
+
import type { MetadataRoute } from "next";
|
|
11
|
+
import { type Site } from "./site.js";
|
|
12
|
+
/** What a bot fetches for: model training, or search and answers (user fetches included). */
|
|
13
|
+
export type AiBotKind = "training" | "search";
|
|
14
|
+
/** One AI or search crawler, by its robots.txt user-agent token. */
|
|
15
|
+
export type AiBot = {
|
|
16
|
+
userAgent: string;
|
|
17
|
+
operator: string;
|
|
18
|
+
kind: AiBotKind;
|
|
19
|
+
/** A web search engine's own crawler: "block-all" keeps it, or the site leaves that engine. */
|
|
20
|
+
searchEngine?: true;
|
|
21
|
+
};
|
|
22
|
+
/** The crawlers robots() names. Order is the order their groups list them. */
|
|
23
|
+
export declare const AI_BOTS: readonly AiBot[];
|
|
24
|
+
/** The AI stance: allow every bot, block training only, or block every AI bot. */
|
|
25
|
+
export type AiBotsPolicy = "allow" | "block-training" | "block-all";
|
|
26
|
+
/** robots()'s options. */
|
|
27
|
+
export type RobotsOptions = {
|
|
28
|
+
/** Paths every crawler may fetch. Default ["/"]. */
|
|
29
|
+
allow?: readonly string[];
|
|
30
|
+
/** Private paths, like "/api/" or "/admin". Longest match wins over allow. */
|
|
31
|
+
disallow?: readonly string[];
|
|
32
|
+
/** Default "allow". */
|
|
33
|
+
aiBots?: AiBotsPolicy;
|
|
34
|
+
};
|
|
35
|
+
/**
|
|
36
|
+
* @function robots
|
|
37
|
+
* @param site {Site} the site (sitemap and host come from its origin)
|
|
38
|
+
* @param options {RobotsOptions} allow, disallow and the AI stance
|
|
39
|
+
* @returns {MetadataRoute.Robots} the `*` group, one group for the allowed AI bots with the same
|
|
40
|
+
* rules (Bingbot is always there), one `Disallow: /` group for the blocked ones (when
|
|
41
|
+
* any are), the sitemap URL and the host
|
|
42
|
+
*/
|
|
43
|
+
export declare const robots: (site: Site, options?: RobotsOptions) => MetadataRoute.Robots;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file src/seo/robots.ts
|
|
3
|
+
* @desc robots.txt with the AI stance written down (audit A1). A crawler that matches a named
|
|
4
|
+
* group ignores the `*` group (RFC 9309), so every allowed AI bot gets a group with the same
|
|
5
|
+
* rules as `*`, and a blocked one gets `Disallow: /`. Changing the stance is one word.
|
|
6
|
+
* @author David @dvhsh (https://dvh.sh)
|
|
7
|
+
* @created Mon Sep 28, 2026
|
|
8
|
+
* @modified Mon Sep 28, 2026
|
|
9
|
+
*/
|
|
10
|
+
import { absoluteUrl, origin } from "./site.js";
|
|
11
|
+
/** The crawlers robots() names. Order is the order their groups list them. */
|
|
12
|
+
export const AI_BOTS = [
|
|
13
|
+
{ userAgent: "GPTBot", operator: "OpenAI", kind: "training" },
|
|
14
|
+
{ userAgent: "OAI-SearchBot", operator: "OpenAI", kind: "search" },
|
|
15
|
+
{ userAgent: "ChatGPT-User", operator: "OpenAI", kind: "search" },
|
|
16
|
+
{ userAgent: "PerplexityBot", operator: "Perplexity", kind: "search" },
|
|
17
|
+
{ userAgent: "Perplexity-User", operator: "Perplexity", kind: "search" },
|
|
18
|
+
{ userAgent: "ClaudeBot", operator: "Anthropic", kind: "training" },
|
|
19
|
+
{ userAgent: "Claude-SearchBot", operator: "Anthropic", kind: "search" },
|
|
20
|
+
{ userAgent: "Claude-User", operator: "Anthropic", kind: "search" },
|
|
21
|
+
{ userAgent: "anthropic-ai", operator: "Anthropic", kind: "training" },
|
|
22
|
+
{ userAgent: "Google-Extended", operator: "Google", kind: "training" },
|
|
23
|
+
{ userAgent: "Applebot-Extended", operator: "Apple", kind: "training" },
|
|
24
|
+
{ userAgent: "Bingbot", operator: "Microsoft", kind: "search", searchEngine: true },
|
|
25
|
+
{ userAgent: "CCBot", operator: "Common Crawl", kind: "training" },
|
|
26
|
+
{ userAgent: "Bytespider", operator: "ByteDance", kind: "training" },
|
|
27
|
+
{ userAgent: "meta-externalagent", operator: "Meta", kind: "training" },
|
|
28
|
+
];
|
|
29
|
+
const blocked = (bot, policy) => policy === "block-all"
|
|
30
|
+
? !bot.searchEngine
|
|
31
|
+
: policy === "block-training" && bot.kind === "training";
|
|
32
|
+
/**
|
|
33
|
+
* @function robots
|
|
34
|
+
* @param site {Site} the site (sitemap and host come from its origin)
|
|
35
|
+
* @param options {RobotsOptions} allow, disallow and the AI stance
|
|
36
|
+
* @returns {MetadataRoute.Robots} the `*` group, one group for the allowed AI bots with the same
|
|
37
|
+
* rules (Bingbot is always there), one `Disallow: /` group for the blocked ones (when
|
|
38
|
+
* any are), the sitemap URL and the host
|
|
39
|
+
*/
|
|
40
|
+
export const robots = (site, options = {}) => {
|
|
41
|
+
const policy = options.aiBots ?? "allow";
|
|
42
|
+
const allow = [...(options.allow ?? ["/"])];
|
|
43
|
+
const rules = { allow, ...(options.disallow?.length ? { disallow: [...options.disallow] } : {}) };
|
|
44
|
+
const allowed = AI_BOTS.filter((bot) => !blocked(bot, policy)).map((bot) => bot.userAgent);
|
|
45
|
+
const denied = AI_BOTS.filter((bot) => blocked(bot, policy)).map((bot) => bot.userAgent);
|
|
46
|
+
return {
|
|
47
|
+
rules: [
|
|
48
|
+
{ userAgent: "*", ...rules },
|
|
49
|
+
{ userAgent: allowed, ...rules },
|
|
50
|
+
...(denied.length ? [{ userAgent: denied, disallow: "/" }] : []),
|
|
51
|
+
],
|
|
52
|
+
sitemap: absoluteUrl(site, "/sitemap.xml"),
|
|
53
|
+
host: origin(site.url),
|
|
54
|
+
};
|
|
55
|
+
};
|