@canmi/me 2026.1004.4 → 2026.1004.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -5
- package/dist/robots/src/index.d.ts +105 -0
- package/dist/robots/src/index.js +201 -0
- package/package.json +13 -2
package/README.md
CHANGED
|
@@ -1,7 +1,4 @@
|
|
|
1
1
|
# @canmi/me
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
(`@canmi/me/locales`). The `canmi` crate is the same addresses and identity for Rust.
|
|
6
|
-
|
|
7
|
-
Versioned by the day it was released, `YYYY.MDD.N`; see the repository's `spec/repository.md`.
|
|
3
|
+
My addresses, identity and languages, and what my hosts say in robots.txt.
|
|
4
|
+
Mostly for my own projects.
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
//#region robots/src/index.d.ts
|
|
2
|
+
/** The opening every policy shares. */
|
|
3
|
+
export declare const robotsTxtBase: readonly ["# https://www.robotstxt.org/robotstxt.html", 'User-agent: *'];
|
|
4
|
+
/**
|
|
5
|
+
* How a page's content may be used, said once and written in both spellings: Cloudflare's
|
|
6
|
+
* `Content-Signal` and the IETF AI Preferences draft's `Content-Usage`. Only a host that serves
|
|
7
|
+
* pages says it; a store of bytes or an API has rules and nothing more.
|
|
8
|
+
*/
|
|
9
|
+
export declare const SIGNALS: {
|
|
10
|
+
readonly search: true;
|
|
11
|
+
readonly aiInput: true;
|
|
12
|
+
readonly aiTrain: true;
|
|
13
|
+
};
|
|
14
|
+
/** Each spelling as a name and a value: a robots.txt line, and a header on a page or markdown. */
|
|
15
|
+
export declare const SIGNAL_HEADERS: readonly [readonly ['Content-Signal', `search=${string}, ai-input=${string}, ai-train=${string}`], readonly ['Content-Usage', `search=${string}, ai-use=${string}, train-ai=${string}`]];
|
|
16
|
+
export declare const signalLines: [string, string];
|
|
17
|
+
/**
|
|
18
|
+
* Cloudflare's terms for content signals, as the comment its own generator writes: the three
|
|
19
|
+
* meanings, and the reservation of rights a `no` makes. Worded and wrapped by Cloudflare, so kept
|
|
20
|
+
* as written.
|
|
21
|
+
*/
|
|
22
|
+
export declare const SIGNAL_TERMS: readonly ['As a condition of accessing this website, you agree to', 'abide by the following content signals:', '', '(a) If a content-signal = yes, you may collect content', 'for the corresponding use.', '(b) If a content-signal = no, you may not collect content', 'for the corresponding use.', '(c) If the website operator does not include a content', 'signal for a corresponding use, the website operator', 'neither grants nor restricts permission via content signal', 'with respect to the corresponding use.', '', 'The content signals and their meanings are:', '', 'search: building a search index and providing search', 'results (e.g., returning hyperlinks and short excerpts', "from your website's contents). Search does not include", 'providing AI-generated search summaries.', 'ai-input: inputting content into one or more AI models', '(e.g., retrieval augmented generation, grounding, or other', 'real-time taking of content for generative AI search', 'answers).', 'ai-train: training or fine-tuning AI models.', '', 'ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS', 'RESERVATIONS OF RIGHTS UNDER ARTICLE 4 OF THE EUROPEAN', 'UNION DIRECTIVE 2019/790 ON COPYRIGHT AND RELATED RIGHTS', 'IN THE DIGITAL SINGLE MARKET.'];
|
|
23
|
+
/** Where a finding is sent: the host's own security.txt, which names the contact. */
|
|
24
|
+
export declare const SECURITY_TXT_PATH = "/.well-known/security.txt";
|
|
25
|
+
/**
|
|
26
|
+
* A host's word to an agent sent to break in: its lines, said from the host's side, and the
|
|
27
|
+
* repository the host is built from, which is where the agent is sent instead.
|
|
28
|
+
*/
|
|
29
|
+
export type AgentNote = {
|
|
30
|
+
lines: readonly string[];
|
|
31
|
+
source: string;
|
|
32
|
+
};
|
|
33
|
+
/** The width a note's line keeps to, with its `# ` before it. */
|
|
34
|
+
export declare const NOTE_WIDTH = 72;
|
|
35
|
+
/**
|
|
36
|
+
* What is wrong with a note's lines, for a test to hold them to: a line past the width, or one
|
|
37
|
+
* holding a lone word. Breaking them is by hand; see spec/me/robots.md.
|
|
38
|
+
*/
|
|
39
|
+
export declare function noteProblems(lines: readonly string[]): string[];
|
|
40
|
+
/** A note as a file says it: the incident it nods to, the note, then where the code is. */
|
|
41
|
+
export declare function agentNote(note: AgentNote): string[];
|
|
42
|
+
export type RobotsTxtOptions = {
|
|
43
|
+
allow?: readonly string[];
|
|
44
|
+
disallow?: readonly string[];
|
|
45
|
+
/** Whether the host serves pages, and so says how their content may be used. */
|
|
46
|
+
signals?: boolean;
|
|
47
|
+
/** The host's word to an agent sent to break in, which the file ends with. */
|
|
48
|
+
note?: AgentNote;
|
|
49
|
+
sitemap?: string | readonly string[] | null;
|
|
50
|
+
};
|
|
51
|
+
export declare function robotsTxt(options?: RobotsTxtOptions): string;
|
|
52
|
+
/**
|
|
53
|
+
* A host that serves pages, as the list every sitemap follows names it: its origin, how often its
|
|
54
|
+
* root changes, and how much the host weighs in the whole of what the author runs -- the priority
|
|
55
|
+
* another host's sitemap gives its root.
|
|
56
|
+
*/
|
|
57
|
+
export type PageHost = {
|
|
58
|
+
name: string;
|
|
59
|
+
origin: string;
|
|
60
|
+
changefreq: string;
|
|
61
|
+
priority: string;
|
|
62
|
+
};
|
|
63
|
+
/** Every page host's sitemap, `name`'s first: what its robots.txt names. */
|
|
64
|
+
export declare function sitemapsFor(hosts: readonly PageHost[], name: string): string[];
|
|
65
|
+
/**
|
|
66
|
+
* A page host's root as another host's sitemap lists it: its weight in the whole, and no
|
|
67
|
+
* modification time -- nothing reaches across hosts to read another's.
|
|
68
|
+
*/
|
|
69
|
+
export declare function rootEntry(hosts: readonly PageHost[], name: string): SitemapEntry;
|
|
70
|
+
/**
|
|
71
|
+
* A page host's root in its own sitemap: how often it changes, as the list says, and the weight
|
|
72
|
+
* the host gives it among its own pages. The host adds its own modification time.
|
|
73
|
+
*/
|
|
74
|
+
export declare function ownRoot(hosts: readonly PageHost[], name: string, priority: string): SitemapEntry;
|
|
75
|
+
/**
|
|
76
|
+
* The other page hosts, by their roots alone, for `name`'s sitemap: each lists its own routes, so
|
|
77
|
+
* a host knows the others by name and needs nothing of theirs to build.
|
|
78
|
+
*/
|
|
79
|
+
export declare function peerEntries(hosts: readonly PageHost[], name: string): SitemapEntry[];
|
|
80
|
+
/**
|
|
81
|
+
* Where every sitemap's stylesheet is: a path on the sitemap's own origin, since a browser applies
|
|
82
|
+
* an XSL stylesheet to an XML document only from there.
|
|
83
|
+
*/
|
|
84
|
+
export declare const SITEMAP_STYLESHEET = "/sitemap.xsl";
|
|
85
|
+
export type SitemapEntry = {
|
|
86
|
+
loc: string;
|
|
87
|
+
lastmod?: string;
|
|
88
|
+
changefreq?: string;
|
|
89
|
+
priority?: string;
|
|
90
|
+
alternates?: readonly {
|
|
91
|
+
language_tag: string;
|
|
92
|
+
href: string;
|
|
93
|
+
}[];
|
|
94
|
+
};
|
|
95
|
+
/** A sitemap of `entries`, styled for a browser by `SITEMAP_STYLESHEET`. */
|
|
96
|
+
export declare function sitemapXml(entries: readonly SitemapEntry[]): string;
|
|
97
|
+
/**
|
|
98
|
+
* The security.txt `origin` answers with at `now`: RFC 9116's two required fields -- the author's
|
|
99
|
+
* contact and an expiry -- where the file lives, and the host's note. The expiry is a day boundary, so every answer on one day is the
|
|
100
|
+
* same text and caches as one.
|
|
101
|
+
*/
|
|
102
|
+
export declare function securityTxt(origin: string, now: Date, note: AgentNote): string;
|
|
103
|
+
/** The security.txt as a response for the host `request` arrived at. */
|
|
104
|
+
export declare function securityResponse(request: Request, note: AgentNote): Response;
|
|
105
|
+
//#endregion
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
import { CONTACT, EXTERNAL } from "../../urls/src/index.js";
|
|
2
|
+
//#region robots/src/index.ts
|
|
3
|
+
/**
|
|
4
|
+
* What every public host of the author's says to a crawler and to an agent, built here and
|
|
5
|
+
* declared by the repository that builds each host: the rules a robots.txt opens with, how page
|
|
6
|
+
* content may be used, the sitemaps, and the word to an agent sent to break in. Nothing here names
|
|
7
|
+
* a host. See spec/me/robots.md.
|
|
8
|
+
*/
|
|
9
|
+
/** The opening every policy shares. */
|
|
10
|
+
const robotsTxtBase = [`# ${EXTERNAL.robotstxt}`, "User-agent: *"];
|
|
11
|
+
/**
|
|
12
|
+
* How a page's content may be used, said once and written in both spellings: Cloudflare's
|
|
13
|
+
* `Content-Signal` and the IETF AI Preferences draft's `Content-Usage`. Only a host that serves
|
|
14
|
+
* pages says it; a store of bytes or an API has rules and nothing more.
|
|
15
|
+
*/
|
|
16
|
+
const SIGNALS = {
|
|
17
|
+
search: true,
|
|
18
|
+
aiInput: true,
|
|
19
|
+
aiTrain: true
|
|
20
|
+
};
|
|
21
|
+
const yes = (value, word, no) => value ? word : no;
|
|
22
|
+
/** Each spelling as a name and a value: a robots.txt line, and a header on a page or markdown. */
|
|
23
|
+
const SIGNAL_HEADERS = [["Content-Signal", `search=${yes(SIGNALS.search, "yes", "no")}, ai-input=${yes(SIGNALS.aiInput, "yes", "no")}, ai-train=${yes(SIGNALS.aiTrain, "yes", "no")}`], ["Content-Usage", `search=${yes(SIGNALS.search, "y", "n")}, ai-use=${yes(SIGNALS.aiInput, "y", "n")}, train-ai=${yes(SIGNALS.aiTrain, "y", "n")}`]];
|
|
24
|
+
const signalLines = SIGNAL_HEADERS.map(([name, value]) => `${name}: ${value}`);
|
|
25
|
+
/**
|
|
26
|
+
* Cloudflare's terms for content signals, as the comment its own generator writes: the three
|
|
27
|
+
* meanings, and the reservation of rights a `no` makes. Worded and wrapped by Cloudflare, so kept
|
|
28
|
+
* as written.
|
|
29
|
+
*/
|
|
30
|
+
const SIGNAL_TERMS = [
|
|
31
|
+
"As a condition of accessing this website, you agree to",
|
|
32
|
+
"abide by the following content signals:",
|
|
33
|
+
"",
|
|
34
|
+
"(a) If a content-signal = yes, you may collect content",
|
|
35
|
+
"for the corresponding use.",
|
|
36
|
+
"(b) If a content-signal = no, you may not collect content",
|
|
37
|
+
"for the corresponding use.",
|
|
38
|
+
"(c) If the website operator does not include a content",
|
|
39
|
+
"signal for a corresponding use, the website operator",
|
|
40
|
+
"neither grants nor restricts permission via content signal",
|
|
41
|
+
"with respect to the corresponding use.",
|
|
42
|
+
"",
|
|
43
|
+
"The content signals and their meanings are:",
|
|
44
|
+
"",
|
|
45
|
+
"search: building a search index and providing search",
|
|
46
|
+
"results (e.g., returning hyperlinks and short excerpts",
|
|
47
|
+
"from your website's contents). Search does not include",
|
|
48
|
+
"providing AI-generated search summaries.",
|
|
49
|
+
"ai-input: inputting content into one or more AI models",
|
|
50
|
+
"(e.g., retrieval augmented generation, grounding, or other",
|
|
51
|
+
"real-time taking of content for generative AI search",
|
|
52
|
+
"answers).",
|
|
53
|
+
"ai-train: training or fine-tuning AI models.",
|
|
54
|
+
"",
|
|
55
|
+
"ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS",
|
|
56
|
+
"RESERVATIONS OF RIGHTS UNDER ARTICLE 4 OF THE EUROPEAN",
|
|
57
|
+
"UNION DIRECTIVE 2019/790 ON COPYRIGHT AND RELATED RIGHTS",
|
|
58
|
+
"IN THE DIGITAL SINGLE MARKET."
|
|
59
|
+
];
|
|
60
|
+
/** Where a finding is sent: the host's own security.txt, which names the contact. */
|
|
61
|
+
const SECURITY_TXT_PATH = "/.well-known/security.txt";
|
|
62
|
+
/** The width a note's line keeps to, with its `# ` before it. */
|
|
63
|
+
const NOTE_WIDTH = 72;
|
|
64
|
+
/**
|
|
65
|
+
* What is wrong with a note's lines, for a test to hold them to: a line past the width, or one
|
|
66
|
+
* holding a lone word. Breaking them is by hand; see spec/me/robots.md.
|
|
67
|
+
*/
|
|
68
|
+
function noteProblems(lines) {
|
|
69
|
+
return lines.flatMap((line) => [...`# ${line}`.length > 72 ? [`past 72 columns: ${line}`] : [], ...line.split(" ").length < 2 ? [`a lone word: ${line}`] : []]);
|
|
70
|
+
}
|
|
71
|
+
/** A note as a file says it: the incident it nods to, the note, then where the code is. */
|
|
72
|
+
function agentNote(note) {
|
|
73
|
+
return [
|
|
74
|
+
`# ${EXTERNAL.agentIncident}`,
|
|
75
|
+
"",
|
|
76
|
+
...note.lines.map((line) => `# ${line}`),
|
|
77
|
+
"",
|
|
78
|
+
`# ${note.source}.git`
|
|
79
|
+
];
|
|
80
|
+
}
|
|
81
|
+
function robotsTxt(options = {}) {
|
|
82
|
+
const lines = [
|
|
83
|
+
robotsTxtBase[0],
|
|
84
|
+
"",
|
|
85
|
+
robotsTxtBase[1]
|
|
86
|
+
];
|
|
87
|
+
for (const path of options.allow ?? []) lines.push(`Allow: ${path}`);
|
|
88
|
+
for (const path of options.disallow ?? []) lines.push(`Disallow: ${path}`);
|
|
89
|
+
if (options.signals) {
|
|
90
|
+
lines.push("", ...SIGNAL_TERMS.map((line) => line ? `# ${line}` : ""));
|
|
91
|
+
lines.push("", `# ${EXTERNAL.contentSignals}`, "", signalLines[0]);
|
|
92
|
+
lines.push("", `# ${EXTERNAL.contentUsage}`, "", signalLines[1]);
|
|
93
|
+
}
|
|
94
|
+
if (options.note) lines.push("", ...agentNote(options.note));
|
|
95
|
+
const sitemaps = toList(options.sitemap);
|
|
96
|
+
if (sitemaps.length > 0) {
|
|
97
|
+
lines.push("");
|
|
98
|
+
for (const sitemap of sitemaps) lines.push(`Sitemap: ${sitemap}`);
|
|
99
|
+
}
|
|
100
|
+
return `${lines.join("\n")}\n`;
|
|
101
|
+
}
|
|
102
|
+
function toList(value) {
|
|
103
|
+
if (!value) return [];
|
|
104
|
+
return typeof value === "string" ? [value] : value;
|
|
105
|
+
}
|
|
106
|
+
/** `name` first, then every other page host in the list's order. */
|
|
107
|
+
function ownFirst(hosts, name) {
|
|
108
|
+
return [...hosts.filter((host) => host.name === name), ...hosts.filter((host) => host.name !== name)];
|
|
109
|
+
}
|
|
110
|
+
function hostOf(hosts, name) {
|
|
111
|
+
const host = hosts.find((candidate) => candidate.name === name);
|
|
112
|
+
if (!host) throw new Error(`${name} serves no pages`);
|
|
113
|
+
return host;
|
|
114
|
+
}
|
|
115
|
+
/** Every page host's sitemap, `name`'s first: what its robots.txt names. */
|
|
116
|
+
function sitemapsFor(hosts, name) {
|
|
117
|
+
return ownFirst(hosts, name).map((host) => `${host.origin}/sitemap.xml`);
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* A page host's root as another host's sitemap lists it: its weight in the whole, and no
|
|
121
|
+
* modification time -- nothing reaches across hosts to read another's.
|
|
122
|
+
*/
|
|
123
|
+
function rootEntry(hosts, name) {
|
|
124
|
+
const host = hostOf(hosts, name);
|
|
125
|
+
return {
|
|
126
|
+
loc: new URL("/", host.origin).href,
|
|
127
|
+
changefreq: host.changefreq,
|
|
128
|
+
priority: host.priority
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* A page host's root in its own sitemap: how often it changes, as the list says, and the weight
|
|
133
|
+
* the host gives it among its own pages. The host adds its own modification time.
|
|
134
|
+
*/
|
|
135
|
+
function ownRoot(hosts, name, priority) {
|
|
136
|
+
const host = hostOf(hosts, name);
|
|
137
|
+
return {
|
|
138
|
+
loc: new URL("/", host.origin).href,
|
|
139
|
+
changefreq: host.changefreq,
|
|
140
|
+
priority
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* The other page hosts, by their roots alone, for `name`'s sitemap: each lists its own routes, so
|
|
145
|
+
* a host knows the others by name and needs nothing of theirs to build.
|
|
146
|
+
*/
|
|
147
|
+
function peerEntries(hosts, name) {
|
|
148
|
+
return ownFirst(hosts, name).slice(1).map((host) => rootEntry(hosts, host.name));
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* Where every sitemap's stylesheet is: a path on the sitemap's own origin, since a browser applies
|
|
152
|
+
* an XSL stylesheet to an XML document only from there.
|
|
153
|
+
*/
|
|
154
|
+
const SITEMAP_STYLESHEET = "/sitemap.xsl";
|
|
155
|
+
/** A sitemap of `entries`, styled for a browser by `SITEMAP_STYLESHEET`. */
|
|
156
|
+
function sitemapXml(entries) {
|
|
157
|
+
const items = entries.map((entry) => {
|
|
158
|
+
return `\t<url>\n${[
|
|
159
|
+
`\t\t<loc>${entry.loc}</loc>`,
|
|
160
|
+
...(entry.alternates ?? []).map((alternate) => `\t\t<xhtml:link rel="alternate" hreflang="${alternate.language_tag}" href="${alternate.href}" />`),
|
|
161
|
+
...entry.lastmod ? [`\t\t<lastmod>${entry.lastmod}</lastmod>`] : [],
|
|
162
|
+
...entry.changefreq ? [`\t\t<changefreq>${entry.changefreq}</changefreq>`] : [],
|
|
163
|
+
...entry.priority ? [`\t\t<priority>${entry.priority}</priority>`] : []
|
|
164
|
+
].join("\n")}\n\t</url>`;
|
|
165
|
+
}).join("\n");
|
|
166
|
+
return `<?xml version="1.0" encoding="UTF-8"?>
|
|
167
|
+
<?xml-stylesheet type="text/xsl" href="${SITEMAP_STYLESHEET}"?>
|
|
168
|
+
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">
|
|
169
|
+
${items}
|
|
170
|
+
</urlset>
|
|
171
|
+
`;
|
|
172
|
+
}
|
|
173
|
+
/** RFC 9116 asks for an expiry under a year away; stated per request, so it never lapses. */
|
|
174
|
+
const VALID_DAYS = 180;
|
|
175
|
+
/**
|
|
176
|
+
* The security.txt `origin` answers with at `now`: RFC 9116's two required fields -- the author's
|
|
177
|
+
* contact and an expiry -- where the file lives, and the host's note. The expiry is a day boundary, so every answer on one day is the
|
|
178
|
+
* same text and caches as one.
|
|
179
|
+
*/
|
|
180
|
+
function securityTxt(origin, now, note) {
|
|
181
|
+
const expires = new Date(now);
|
|
182
|
+
expires.setUTCHours(0, 0, 0, 0);
|
|
183
|
+
expires.setUTCDate(expires.getUTCDate() + VALID_DAYS);
|
|
184
|
+
return [
|
|
185
|
+
`Contact: ${CONTACT.security}`,
|
|
186
|
+
`Expires: ${expires.toISOString()}`,
|
|
187
|
+
`Canonical: ${new URL(SECURITY_TXT_PATH, origin).href}`,
|
|
188
|
+
"",
|
|
189
|
+
...agentNote(note),
|
|
190
|
+
""
|
|
191
|
+
].join("\n");
|
|
192
|
+
}
|
|
193
|
+
/** The security.txt as a response for the host `request` arrived at. */
|
|
194
|
+
function securityResponse(request, note) {
|
|
195
|
+
return new Response(securityTxt(new URL(request.url).origin, /* @__PURE__ */ new Date(), note), { headers: {
|
|
196
|
+
"Content-Type": "text/plain; charset=utf-8",
|
|
197
|
+
"Cache-Control": "public, max-age=86400"
|
|
198
|
+
} });
|
|
199
|
+
}
|
|
200
|
+
//#endregion
|
|
201
|
+
export { NOTE_WIDTH, SECURITY_TXT_PATH, SIGNALS, SIGNAL_HEADERS, SIGNAL_TERMS, SITEMAP_STYLESHEET, agentNote, noteProblems, ownRoot, peerEntries, robotsTxt, robotsTxtBase, rootEntry, securityResponse, securityTxt, signalLines, sitemapXml, sitemapsFor };
|
package/package.json
CHANGED
|
@@ -1,7 +1,14 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@canmi/me",
|
|
3
|
-
"version": "2026.1004.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "2026.1004.6",
|
|
4
|
+
"description": "My addresses, identity and languages.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"canmi",
|
|
7
|
+
"identity",
|
|
8
|
+
"locales",
|
|
9
|
+
"robots",
|
|
10
|
+
"urls"
|
|
11
|
+
],
|
|
5
12
|
"license": "MIT",
|
|
6
13
|
"repository": {
|
|
7
14
|
"type": "git",
|
|
@@ -33,6 +40,10 @@
|
|
|
33
40
|
"./locales/format": {
|
|
34
41
|
"types": "./dist/locales/src/format.d.ts",
|
|
35
42
|
"default": "./dist/locales/src/format.js"
|
|
43
|
+
},
|
|
44
|
+
"./robots": {
|
|
45
|
+
"types": "./dist/robots/src/index.d.ts",
|
|
46
|
+
"default": "./dist/robots/src/index.js"
|
|
36
47
|
}
|
|
37
48
|
},
|
|
38
49
|
"scripts": {
|