ax-audit 3.6.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +138 -0
- package/LICENSE +1 -1
- package/README.md +58 -36
- package/dist/baseline.d.ts +2 -0
- package/dist/baseline.d.ts.map +1 -1
- package/dist/baseline.js +42 -4
- package/dist/baseline.js.map +1 -1
- package/dist/check-ids.d.ts +19 -0
- package/dist/check-ids.d.ts.map +1 -0
- package/dist/check-ids.js +53 -0
- package/dist/check-ids.js.map +1 -0
- package/dist/checks/agent-access.d.ts +23 -6
- package/dist/checks/agent-access.d.ts.map +1 -1
- package/dist/checks/agent-access.js +200 -54
- package/dist/checks/agent-access.js.map +1 -1
- package/dist/checks/agent-card.d.ts +37 -0
- package/dist/checks/agent-card.d.ts.map +1 -0
- package/dist/checks/agent-card.js +352 -0
- package/dist/checks/agent-card.js.map +1 -0
- package/dist/checks/agent-operability.d.ts +66 -0
- package/dist/checks/agent-operability.d.ts.map +1 -0
- package/dist/checks/agent-operability.js +383 -0
- package/dist/checks/agent-operability.js.map +1 -0
- package/dist/checks/agent-skills.d.ts +24 -0
- package/dist/checks/agent-skills.d.ts.map +1 -0
- package/dist/checks/agent-skills.js +316 -0
- package/dist/checks/agent-skills.js.map +1 -0
- package/dist/checks/ai-catalog.d.ts +28 -0
- package/dist/checks/ai-catalog.d.ts.map +1 -0
- package/dist/checks/ai-catalog.js +254 -0
- package/dist/checks/ai-catalog.js.map +1 -0
- package/dist/checks/ai-directives.d.ts +57 -0
- package/dist/checks/ai-directives.d.ts.map +1 -0
- package/dist/checks/ai-directives.js +263 -0
- package/dist/checks/ai-directives.js.map +1 -0
- package/dist/checks/api-discovery.d.ts +26 -0
- package/dist/checks/api-discovery.d.ts.map +1 -0
- package/dist/checks/api-discovery.js +432 -0
- package/dist/checks/api-discovery.js.map +1 -0
- package/dist/checks/auth-discovery.d.ts +28 -0
- package/dist/checks/auth-discovery.d.ts.map +1 -0
- package/dist/checks/auth-discovery.js +213 -0
- package/dist/checks/auth-discovery.js.map +1 -0
- package/dist/checks/commerce-discovery.d.ts +40 -0
- package/dist/checks/commerce-discovery.d.ts.map +1 -0
- package/dist/checks/commerce-discovery.js +295 -0
- package/dist/checks/commerce-discovery.js.map +1 -0
- package/dist/checks/content-negotiation.d.ts.map +1 -1
- package/dist/checks/content-negotiation.js +135 -20
- package/dist/checks/content-negotiation.js.map +1 -1
- package/dist/checks/crawl-efficiency.d.ts +13 -1
- package/dist/checks/crawl-efficiency.d.ts.map +1 -1
- package/dist/checks/crawl-efficiency.js +65 -1
- package/dist/checks/crawl-efficiency.js.map +1 -1
- package/dist/checks/frontmatter.d.ts +34 -0
- package/dist/checks/frontmatter.d.ts.map +1 -0
- package/dist/checks/frontmatter.js +100 -0
- package/dist/checks/frontmatter.js.map +1 -0
- package/dist/checks/html-rendering.d.ts.map +1 -1
- package/dist/checks/html-rendering.js +0 -1
- package/dist/checks/html-rendering.js.map +1 -1
- package/dist/checks/html-utils.d.ts +10 -0
- package/dist/checks/html-utils.d.ts.map +1 -1
- package/dist/checks/html-utils.js +19 -0
- package/dist/checks/html-utils.js.map +1 -1
- package/dist/checks/http-headers.d.ts.map +1 -1
- package/dist/checks/http-headers.js +82 -10
- package/dist/checks/http-headers.js.map +1 -1
- package/dist/checks/http-hygiene.d.ts +26 -0
- package/dist/checks/http-hygiene.d.ts.map +1 -0
- package/dist/checks/http-hygiene.js +257 -0
- package/dist/checks/http-hygiene.js.map +1 -0
- package/dist/checks/index.d.ts.map +1 -1
- package/dist/checks/index.js +24 -8
- package/dist/checks/index.js.map +1 -1
- package/dist/checks/llms-txt.d.ts +15 -0
- package/dist/checks/llms-txt.d.ts.map +1 -1
- package/dist/checks/llms-txt.js +162 -2
- package/dist/checks/llms-txt.js.map +1 -1
- package/dist/checks/mcp-discovery.d.ts +30 -0
- package/dist/checks/mcp-discovery.d.ts.map +1 -0
- package/dist/checks/mcp-discovery.js +523 -0
- package/dist/checks/mcp-discovery.js.map +1 -0
- package/dist/checks/meta-tags.d.ts.map +1 -1
- package/dist/checks/meta-tags.js +6 -5
- package/dist/checks/meta-tags.js.map +1 -1
- package/dist/checks/robots-parser.d.ts +110 -0
- package/dist/checks/robots-parser.d.ts.map +1 -0
- package/dist/checks/robots-parser.js +277 -0
- package/dist/checks/robots-parser.js.map +1 -0
- package/dist/checks/robots-txt.d.ts +2 -20
- package/dist/checks/robots-txt.d.ts.map +1 -1
- package/dist/checks/robots-txt.js +219 -120
- package/dist/checks/robots-txt.js.map +1 -1
- package/dist/checks/rsl.d.ts +0 -2
- package/dist/checks/rsl.d.ts.map +1 -1
- package/dist/checks/rsl.js +1 -11
- package/dist/checks/rsl.js.map +1 -1
- package/dist/checks/security-txt.d.ts.map +1 -1
- package/dist/checks/security-txt.js +0 -1
- package/dist/checks/security-txt.js.map +1 -1
- package/dist/checks/seo-basics.d.ts.map +1 -1
- package/dist/checks/seo-basics.js +0 -1
- package/dist/checks/seo-basics.js.map +1 -1
- package/dist/checks/sitemap.d.ts.map +1 -1
- package/dist/checks/sitemap.js +0 -1
- package/dist/checks/sitemap.js.map +1 -1
- package/dist/checks/structured-data.d.ts.map +1 -1
- package/dist/checks/structured-data.js +215 -4
- package/dist/checks/structured-data.js.map +1 -1
- package/dist/checks/structured-fields.d.ts +46 -0
- package/dist/checks/structured-fields.d.ts.map +1 -0
- package/dist/checks/structured-fields.js +112 -0
- package/dist/checks/structured-fields.js.map +1 -0
- package/dist/checks/surface.d.ts +59 -0
- package/dist/checks/surface.d.ts.map +1 -0
- package/dist/checks/surface.js +106 -0
- package/dist/checks/surface.js.map +1 -0
- package/dist/checks/tls-https.d.ts.map +1 -1
- package/dist/checks/tls-https.js +0 -1
- package/dist/checks/tls-https.js.map +1 -1
- package/dist/checks/usage-policy.d.ts +53 -0
- package/dist/checks/usage-policy.d.ts.map +1 -0
- package/dist/checks/usage-policy.js +339 -0
- package/dist/checks/usage-policy.js.map +1 -0
- package/dist/checks/utils.d.ts +25 -1
- package/dist/checks/utils.d.ts.map +1 -1
- package/dist/checks/utils.js +33 -1
- package/dist/checks/utils.js.map +1 -1
- package/dist/checks/waf.d.ts +75 -0
- package/dist/checks/waf.d.ts.map +1 -0
- package/dist/checks/waf.js +203 -0
- package/dist/checks/waf.js.map +1 -0
- package/dist/checks/webmcp.d.ts +55 -0
- package/dist/checks/webmcp.d.ts.map +1 -0
- package/dist/checks/webmcp.js +209 -0
- package/dist/checks/webmcp.js.map +1 -0
- package/dist/checks/well-known.d.ts +38 -0
- package/dist/checks/well-known.d.ts.map +1 -0
- package/dist/checks/well-known.js +202 -0
- package/dist/checks/well-known.js.map +1 -0
- package/dist/cli.d.ts +14 -0
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +124 -4
- package/dist/cli.js.map +1 -1
- package/dist/constants.d.ts +195 -14
- package/dist/constants.d.ts.map +1 -1
- package/dist/constants.js +598 -70
- package/dist/constants.js.map +1 -1
- package/dist/fetcher.d.ts.map +1 -1
- package/dist/fetcher.js +44 -14
- package/dist/fetcher.js.map +1 -1
- package/dist/guide-urls.js +1 -1
- package/dist/guide-urls.js.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/orchestrator.d.ts.map +1 -1
- package/dist/orchestrator.js +5 -1
- package/dist/orchestrator.js.map +1 -1
- package/dist/reporter/html.d.ts +9 -0
- package/dist/reporter/html.d.ts.map +1 -1
- package/dist/reporter/html.js +49 -11
- package/dist/reporter/html.js.map +1 -1
- package/dist/reporter/markdown.d.ts.map +1 -1
- package/dist/reporter/markdown.js +36 -6
- package/dist/reporter/markdown.js.map +1 -1
- package/dist/reporter/terminal.d.ts.map +1 -1
- package/dist/reporter/terminal.js +36 -1
- package/dist/reporter/terminal.js.map +1 -1
- package/dist/scorer.d.ts +10 -0
- package/dist/scorer.d.ts.map +1 -1
- package/dist/scorer.js +22 -5
- package/dist/scorer.js.map +1 -1
- package/dist/types.d.ts +90 -3
- package/dist/types.d.ts.map +1 -1
- package/docs/architecture.md +27 -11
- package/docs/checks.md +285 -52
- package/docs/cli.md +36 -0
- package/docs/concepts.md +27 -13
- package/docs/faq.md +18 -6
- package/docs/getting-started.md +20 -13
- package/docs/roadmap.md +367 -0
- package/package.json +13 -5
- package/dist/checks/agent-json.d.ts +0 -14
- package/dist/checks/agent-json.d.ts.map +0 -1
- package/dist/checks/agent-json.js +0 -167
- package/dist/checks/agent-json.js.map +0 -1
- package/dist/checks/mcp.d.ts +0 -4
- package/dist/checks/mcp.d.ts.map +0 -1
- package/dist/checks/mcp.js +0 -162
- package/dist/checks/mcp.js.map +0 -1
- package/dist/checks/openapi.d.ts +0 -4
- package/dist/checks/openapi.d.ts.map +0 -1
- package/dist/checks/openapi.js +0 -121
- package/dist/checks/openapi.js.map +0 -1
- package/dist/checks/well-known-ai.d.ts +0 -17
- package/dist/checks/well-known-ai.d.ts.map +0 -1
- package/dist/checks/well-known-ai.js +0 -123
- package/dist/checks/well-known-ai.js.map +0 -1
package/dist/constants.js
CHANGED
|
@@ -4,114 +4,531 @@ import { dirname, join } from 'node:path';
|
|
|
4
4
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
5
5
|
const pkg = JSON.parse(readFileSync(join(__dirname, '..', 'package.json'), 'utf-8'));
|
|
6
6
|
export const VERSION = pkg.version;
|
|
7
|
-
export const USER_AGENT = `ax-audit/${pkg.version} (https://github.com/
|
|
7
|
+
export const USER_AGENT = `ax-audit/${pkg.version} (https://github.com/duranitech/ax-audit)`;
|
|
8
8
|
/**
|
|
9
|
-
*
|
|
10
|
-
* `
|
|
11
|
-
*
|
|
9
|
+
* Per-token metadata for the crawlers worth explaining. Tokens in
|
|
10
|
+
* `AI_CRAWLERS` without an entry here are recognised for matching but carry no
|
|
11
|
+
* tailored advice — the long tail of data brokers and regional assistants.
|
|
12
12
|
*
|
|
13
|
-
*
|
|
14
|
-
|
|
15
|
-
|
|
13
|
+
* Every entry was verified against vendor documentation on 2026-09-04.
|
|
14
|
+
*/
|
|
15
|
+
export const CRAWLER_META = {
|
|
16
|
+
GPTBot: {
|
|
17
|
+
vendor: 'OpenAI',
|
|
18
|
+
purpose: 'training',
|
|
19
|
+
honorsRobots: true,
|
|
20
|
+
impact: 'Blocking it keeps your content out of OpenAI model training. It does not affect ChatGPT search citations.',
|
|
21
|
+
docUrl: 'https://developers.openai.com/api/docs/bots',
|
|
22
|
+
ipListUrl: 'https://openai.com/gptbot.json',
|
|
23
|
+
},
|
|
24
|
+
'OAI-SearchBot': {
|
|
25
|
+
vendor: 'OpenAI',
|
|
26
|
+
purpose: 'search',
|
|
27
|
+
honorsRobots: true,
|
|
28
|
+
impact: 'Blocking it removes your site from ChatGPT search answers and citations.',
|
|
29
|
+
docUrl: 'https://developers.openai.com/api/docs/bots',
|
|
30
|
+
ipListUrl: 'https://openai.com/searchbot.json',
|
|
31
|
+
note: 'OpenAI removed the "also used for training" language from this bot in December 2025.',
|
|
32
|
+
},
|
|
33
|
+
'ChatGPT-User': {
|
|
34
|
+
vendor: 'OpenAI',
|
|
35
|
+
purpose: 'user-fetch',
|
|
36
|
+
honorsRobots: 'partial',
|
|
37
|
+
impact: 'Fetches a page because a ChatGPT user asked for that URL. Blocking it breaks link-following in conversations.',
|
|
38
|
+
docUrl: 'https://developers.openai.com/api/docs/bots',
|
|
39
|
+
ipListUrl: 'https://openai.com/chatgpt-user.json',
|
|
40
|
+
note: 'Since December 2025 OpenAI documents that robots.txt rules may not apply to this user-triggered fetcher.',
|
|
41
|
+
},
|
|
42
|
+
'OAI-AdsBot': {
|
|
43
|
+
vendor: 'OpenAI',
|
|
44
|
+
purpose: 'search',
|
|
45
|
+
honorsRobots: true,
|
|
46
|
+
impact: 'Validates landing pages for ChatGPT ads. No training use.',
|
|
47
|
+
docUrl: 'https://developers.openai.com/api/docs/bots',
|
|
48
|
+
ipListUrl: 'https://openai.com/adsbot.json',
|
|
49
|
+
note: 'Introduced April 2026.',
|
|
50
|
+
},
|
|
51
|
+
ClaudeBot: {
|
|
52
|
+
vendor: 'Anthropic',
|
|
53
|
+
purpose: 'training',
|
|
54
|
+
honorsRobots: true,
|
|
55
|
+
impact: 'Blocking it keeps your content out of Claude model training.',
|
|
56
|
+
docUrl: 'https://support.claude.com/en/articles/8896518',
|
|
57
|
+
ipListUrl: 'https://claude.com/crawling/bots.json',
|
|
58
|
+
},
|
|
59
|
+
'Claude-SearchBot': {
|
|
60
|
+
vendor: 'Anthropic',
|
|
61
|
+
purpose: 'search',
|
|
62
|
+
honorsRobots: true,
|
|
63
|
+
impact: 'Blocking it removes your site from the index Claude cites when it searches the web.',
|
|
64
|
+
docUrl: 'https://support.claude.com/en/articles/8896518',
|
|
65
|
+
ipListUrl: 'https://claude.com/crawling/bots.json',
|
|
66
|
+
},
|
|
67
|
+
'Claude-User': {
|
|
68
|
+
vendor: 'Anthropic',
|
|
69
|
+
purpose: 'user-fetch',
|
|
70
|
+
honorsRobots: true,
|
|
71
|
+
impact: 'Fetches a page because a Claude user asked for that URL. Blocking it breaks link-following in conversations.',
|
|
72
|
+
docUrl: 'https://support.claude.com/en/articles/8896518',
|
|
73
|
+
ipListUrl: 'https://claude.com/crawling/bots.json',
|
|
74
|
+
note: 'Unusual among user-triggered fetchers: Anthropic documents that it does obey robots.txt.',
|
|
75
|
+
},
|
|
76
|
+
'Google-Extended': {
|
|
77
|
+
vendor: 'Google',
|
|
78
|
+
purpose: 'training',
|
|
79
|
+
honorsRobots: true,
|
|
80
|
+
impact: 'Controls Gemini training and grounding in Gemini Apps and Vertex AI. It does NOT remove your site from AI Overviews or AI Mode, which follow Googlebot and the snippet directives.',
|
|
81
|
+
docUrl: 'https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers',
|
|
82
|
+
tokenOnly: true,
|
|
83
|
+
note: 'A robots.txt token, not a crawler: no request ever carries this user agent.',
|
|
84
|
+
},
|
|
85
|
+
'Google-CloudVertexBot': {
|
|
86
|
+
vendor: 'Google',
|
|
87
|
+
purpose: 'training',
|
|
88
|
+
honorsRobots: true,
|
|
89
|
+
impact: 'Crawls sites at their owner\u2019s request to build Vertex AI Agents. Site-owner initiated.',
|
|
90
|
+
docUrl: 'https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers',
|
|
91
|
+
ipListUrl: 'https://developers.google.com/static/crawling/ipranges/common-crawlers.json',
|
|
92
|
+
},
|
|
93
|
+
'Google-Agent': {
|
|
94
|
+
vendor: 'Google',
|
|
95
|
+
purpose: 'agent',
|
|
96
|
+
honorsRobots: false,
|
|
97
|
+
impact: 'Google\u2019s user-triggered browsing agent. It generally ignores robots.txt, so rules for it are advisory.',
|
|
98
|
+
docUrl: 'https://developers.google.com/crawling/docs/crawlers-fetchers/google-agent',
|
|
99
|
+
ipListUrl: 'https://developers.google.com/static/search/apis/ipranges/user-triggered-agents.json',
|
|
100
|
+
signsRequests: true,
|
|
101
|
+
note: 'Introduced 2026-03-20, superseding Project Mariner. Web Bot Auth identity https://agent.bot.goog (experimental; not every request is signed).',
|
|
102
|
+
},
|
|
103
|
+
'Google-GeminiNotebook': {
|
|
104
|
+
vendor: 'Google',
|
|
105
|
+
purpose: 'user-fetch',
|
|
106
|
+
honorsRobots: false,
|
|
107
|
+
impact: 'Fetches sources a user added to Gemini Notebook. Ignores robots.txt.',
|
|
108
|
+
docUrl: 'https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers',
|
|
109
|
+
note: 'Renamed from Google-NotebookLM on 2026-07-17; the old token was supported until August 2026.',
|
|
110
|
+
},
|
|
111
|
+
bingbot: {
|
|
112
|
+
vendor: 'Microsoft',
|
|
113
|
+
purpose: 'search',
|
|
114
|
+
honorsRobots: true,
|
|
115
|
+
impact: 'Blocking it removes your site from Bing and from the index Copilot grounds its answers in.',
|
|
116
|
+
docUrl: 'https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0',
|
|
117
|
+
ipListUrl: 'https://www.bing.com/toolbox/bingbot.json',
|
|
118
|
+
note: 'Multi-purpose: page-level noarchive / nocache directives control the Copilot generative use separately.',
|
|
119
|
+
},
|
|
120
|
+
'Meta-ExternalAgent': {
|
|
121
|
+
vendor: 'Meta',
|
|
122
|
+
purpose: 'training',
|
|
123
|
+
honorsRobots: true,
|
|
124
|
+
impact: 'Blocking it keeps your content out of Meta AI training and indexing.',
|
|
125
|
+
docUrl: 'https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/',
|
|
126
|
+
note: 'Second-largest AI crawler by request share in mid-2026.',
|
|
127
|
+
},
|
|
128
|
+
'meta-webindexer': {
|
|
129
|
+
vendor: 'Meta',
|
|
130
|
+
purpose: 'search',
|
|
131
|
+
honorsRobots: true,
|
|
132
|
+
impact: 'Blocking it removes your site from Meta AI search results and citations.',
|
|
133
|
+
docUrl: 'https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/',
|
|
134
|
+
note: 'Newer than the widely copied robots.txt templates, so most sites have no rule for it.',
|
|
135
|
+
},
|
|
136
|
+
'meta-externalfetcher': {
|
|
137
|
+
vendor: 'Meta',
|
|
138
|
+
purpose: 'user-fetch',
|
|
139
|
+
honorsRobots: 'partial',
|
|
140
|
+
impact: 'Fetches a page for a user request or agentic task. Meta documents that it may bypass robots.txt.',
|
|
141
|
+
docUrl: 'https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/',
|
|
142
|
+
},
|
|
143
|
+
Applebot: {
|
|
144
|
+
vendor: 'Apple',
|
|
145
|
+
purpose: 'search',
|
|
146
|
+
honorsRobots: true,
|
|
147
|
+
impact: 'Powers Siri and Spotlight results, and since June 2026 also feeds Apple Intelligence answers.',
|
|
148
|
+
docUrl: 'https://support.apple.com/en-us/119829',
|
|
149
|
+
ipListUrl: 'https://search.developer.apple.com/applebot.json',
|
|
150
|
+
note: 'Falls back to Googlebot rules when no Applebot group exists.',
|
|
151
|
+
},
|
|
152
|
+
'Applebot-Extended': {
|
|
153
|
+
vendor: 'Apple',
|
|
154
|
+
purpose: 'training',
|
|
155
|
+
honorsRobots: true,
|
|
156
|
+
impact: 'Opts your content out of Apple foundation-model training without affecting Siri or Spotlight.',
|
|
157
|
+
docUrl: 'https://support.apple.com/en-us/119829',
|
|
158
|
+
tokenOnly: true,
|
|
159
|
+
note: 'A robots.txt token, not a crawler.',
|
|
160
|
+
},
|
|
161
|
+
Amazonbot: {
|
|
162
|
+
vendor: 'Amazon',
|
|
163
|
+
purpose: 'training',
|
|
164
|
+
honorsRobots: true,
|
|
165
|
+
impact: 'General crawler whose content may be used to train Amazon AI models.',
|
|
166
|
+
docUrl: 'https://developer.amazon.com/amazonbot',
|
|
167
|
+
ipListUrl: 'https://developer.amazon.com/amazonbot/ip-addresses/',
|
|
168
|
+
note: 'Managed through robots.txt only since 2026-06-15.',
|
|
169
|
+
},
|
|
170
|
+
'Amzn-SearchBot': {
|
|
171
|
+
vendor: 'Amazon',
|
|
172
|
+
purpose: 'search',
|
|
173
|
+
honorsRobots: true,
|
|
174
|
+
impact: 'Blocking it removes your site from Alexa and Rufus answers. No training use.',
|
|
175
|
+
docUrl: 'https://developer.amazon.com/amazonbot',
|
|
176
|
+
},
|
|
177
|
+
'Amzn-User': {
|
|
178
|
+
vendor: 'Amazon',
|
|
179
|
+
purpose: 'user-fetch',
|
|
180
|
+
honorsRobots: 'partial',
|
|
181
|
+
impact: 'Fetches a page for a live Amazon assistant request. No training use.',
|
|
182
|
+
docUrl: 'https://developer.amazon.com/amazonbot',
|
|
183
|
+
},
|
|
184
|
+
PerplexityBot: {
|
|
185
|
+
vendor: 'Perplexity',
|
|
186
|
+
purpose: 'search',
|
|
187
|
+
honorsRobots: true,
|
|
188
|
+
impact: 'Blocking it removes your site from Perplexity answers and citations. Not used for model training.',
|
|
189
|
+
docUrl: 'https://docs.perplexity.ai/docs/resources/perplexity-crawlers',
|
|
190
|
+
ipListUrl: 'https://www.perplexity.com/perplexitybot.json',
|
|
191
|
+
},
|
|
192
|
+
'Perplexity-User': {
|
|
193
|
+
vendor: 'Perplexity',
|
|
194
|
+
purpose: 'user-fetch',
|
|
195
|
+
honorsRobots: false,
|
|
196
|
+
impact: 'Fetches a page a Perplexity user opened. Perplexity documents that it generally ignores robots.txt.',
|
|
197
|
+
docUrl: 'https://docs.perplexity.ai/docs/resources/perplexity-crawlers',
|
|
198
|
+
ipListUrl: 'https://www.perplexity.com/perplexity-user.json',
|
|
199
|
+
},
|
|
200
|
+
CCBot: {
|
|
201
|
+
vendor: 'Common Crawl',
|
|
202
|
+
purpose: 'training',
|
|
203
|
+
honorsRobots: true,
|
|
204
|
+
impact: 'Builds the open Common Crawl corpus that many models train on. Blocking it is the single broadest training opt-out.',
|
|
205
|
+
docUrl: 'https://commoncrawl.org/ccbot',
|
|
206
|
+
ipListUrl: 'https://index.commoncrawl.org/ccbot.json',
|
|
207
|
+
},
|
|
208
|
+
Bytespider: {
|
|
209
|
+
vendor: 'ByteDance',
|
|
210
|
+
purpose: 'training',
|
|
211
|
+
honorsRobots: 'partial',
|
|
212
|
+
impact: 'Collects training data for ByteDance models. Compliance with robots.txt is disputed.',
|
|
213
|
+
docUrl: 'https://zhanzhang.toutiao.com/',
|
|
214
|
+
},
|
|
215
|
+
'MistralAI-User': {
|
|
216
|
+
vendor: 'Mistral',
|
|
217
|
+
purpose: 'user-fetch',
|
|
218
|
+
honorsRobots: true,
|
|
219
|
+
impact: 'Fetches a page for a Le Chat user request.',
|
|
220
|
+
docUrl: 'https://docs.mistral.ai/robots/',
|
|
221
|
+
ipListUrl: 'https://mistral.ai/mistralai-user-ips.json',
|
|
222
|
+
},
|
|
223
|
+
'MistralAI-Index': {
|
|
224
|
+
vendor: 'Mistral',
|
|
225
|
+
purpose: 'search',
|
|
226
|
+
honorsRobots: true,
|
|
227
|
+
impact: 'Blocking it removes your site from Le Chat search results. No training use.',
|
|
228
|
+
docUrl: 'https://docs.mistral.ai/robots/',
|
|
229
|
+
ipListUrl: 'https://mistral.ai/mistralai-index-ips.json',
|
|
230
|
+
},
|
|
231
|
+
'MistralAI-Training': {
|
|
232
|
+
vendor: 'Mistral',
|
|
233
|
+
purpose: 'training',
|
|
234
|
+
honorsRobots: true,
|
|
235
|
+
impact: 'Collects training data for Mistral models.',
|
|
236
|
+
docUrl: 'https://docs.mistral.ai/robots/',
|
|
237
|
+
},
|
|
238
|
+
DuckAssistBot: {
|
|
239
|
+
vendor: 'DuckDuckGo',
|
|
240
|
+
purpose: 'search',
|
|
241
|
+
honorsRobots: true,
|
|
242
|
+
impact: 'Fetches pages in real time for DuckDuckGo AI answers. No training use.',
|
|
243
|
+
docUrl: 'https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot/',
|
|
244
|
+
ipListUrl: 'https://duckduckgo.com/duckassistbot.json',
|
|
245
|
+
},
|
|
246
|
+
ExaSearchBot: {
|
|
247
|
+
vendor: 'Exa',
|
|
248
|
+
purpose: 'search',
|
|
249
|
+
honorsRobots: true,
|
|
250
|
+
impact: 'Builds the Exa search index used by AI agents and retrieval pipelines.',
|
|
251
|
+
docUrl: 'https://crawler.exa.ai/',
|
|
252
|
+
signsRequests: true,
|
|
253
|
+
note: 'Signs every request with Web Bot Auth, so an IP-verifying WAF can admit it precisely.',
|
|
254
|
+
},
|
|
255
|
+
Kagibot: {
|
|
256
|
+
vendor: 'Kagi',
|
|
257
|
+
purpose: 'search',
|
|
258
|
+
honorsRobots: true,
|
|
259
|
+
impact: 'Blocking it removes your site from Kagi search and its Assistant answers.',
|
|
260
|
+
docUrl: 'https://kagi.com/bot',
|
|
261
|
+
},
|
|
262
|
+
YouBot: {
|
|
263
|
+
vendor: 'You.com',
|
|
264
|
+
purpose: 'search',
|
|
265
|
+
honorsRobots: true,
|
|
266
|
+
impact: 'Indexes pages for You.com search and its LLM answers.',
|
|
267
|
+
docUrl: 'https://about.you.com/youbot/',
|
|
268
|
+
signsRequests: true,
|
|
269
|
+
},
|
|
270
|
+
AI2Bot: {
|
|
271
|
+
vendor: 'Allen Institute for AI',
|
|
272
|
+
purpose: 'training',
|
|
273
|
+
honorsRobots: true,
|
|
274
|
+
impact: 'Collects data for open research corpora (OLMo, Dolma).',
|
|
275
|
+
docUrl: 'https://allenai.org/crawler',
|
|
276
|
+
},
|
|
277
|
+
FirecrawlAgent: {
|
|
278
|
+
vendor: 'Firecrawl',
|
|
279
|
+
purpose: 'agent',
|
|
280
|
+
honorsRobots: true,
|
|
281
|
+
impact: 'Scraping-as-a-service used by agent builders to read your pages on demand.',
|
|
282
|
+
docUrl: 'https://docs.firecrawl.dev/',
|
|
283
|
+
},
|
|
284
|
+
};
|
|
285
|
+
/**
|
|
286
|
+
* Known AI clients grouped by what they do with a page. Matching is
|
|
287
|
+
* case-insensitive, per RFC 9309 §2.2.1, so the casing here is cosmetic and
|
|
288
|
+
* follows each vendor's own documentation.
|
|
289
|
+
*
|
|
290
|
+
* Verified against vendor documentation on 2026-09-04. Tokens that turned out
|
|
291
|
+
* never to have existed (`Gemini`, `GeminiBot`, `DeepSeek-AI`) or to belong to
|
|
292
|
+
* discontinued products (`NeevaBot`, `GoogleAgent-Mariner`) moved to
|
|
293
|
+
* `LEGACY_AI_CRAWLERS`.
|
|
16
294
|
*/
|
|
17
295
|
export const AI_CRAWLERS = {
|
|
18
296
|
training: [
|
|
19
297
|
'GPTBot',
|
|
20
298
|
'ClaudeBot',
|
|
21
|
-
'Claude-Web',
|
|
22
|
-
'Anthropic-AI',
|
|
23
|
-
'Google-Extended',
|
|
24
|
-
'CCBot',
|
|
25
|
-
'Bytespider',
|
|
26
299
|
'Meta-ExternalAgent',
|
|
27
|
-
'
|
|
28
|
-
'Cohere-AI',
|
|
29
|
-
'cohere-training-data-crawler',
|
|
300
|
+
'Google-Extended',
|
|
30
301
|
'Applebot-Extended',
|
|
31
302
|
'Amazonbot',
|
|
303
|
+
'CCBot',
|
|
304
|
+
'Bytespider',
|
|
305
|
+
'TikTokSpider',
|
|
306
|
+
'MistralAI-Training',
|
|
32
307
|
'AI2Bot',
|
|
33
|
-
'
|
|
34
|
-
'
|
|
308
|
+
'Ai2Bot-Dolma',
|
|
309
|
+
'DeepSeekBot',
|
|
35
310
|
'PanguBot',
|
|
36
|
-
'
|
|
37
|
-
'
|
|
38
|
-
'Kangaroo Bot',
|
|
311
|
+
'Google-CloudVertexBot',
|
|
312
|
+
'FacebookBot',
|
|
39
313
|
'Timpibot',
|
|
314
|
+
'Webzio-Extended',
|
|
40
315
|
'omgili',
|
|
41
316
|
'omgilibot',
|
|
42
317
|
'ImagesiftBot',
|
|
43
|
-
'
|
|
318
|
+
'Kangaroo Bot',
|
|
319
|
+
'Diffbot',
|
|
320
|
+
'YandexAdditional',
|
|
321
|
+
'YandexAdditionalBot',
|
|
44
322
|
],
|
|
45
323
|
search: [
|
|
46
324
|
'OAI-SearchBot',
|
|
47
|
-
'ChatGPT-User',
|
|
48
325
|
'Claude-SearchBot',
|
|
49
|
-
'Claude-User',
|
|
50
326
|
'PerplexityBot',
|
|
51
|
-
'
|
|
327
|
+
'meta-webindexer',
|
|
328
|
+
'Amzn-SearchBot',
|
|
329
|
+
'MistralAI-Index',
|
|
330
|
+
'Applebot',
|
|
331
|
+
'bingbot',
|
|
52
332
|
'DuckAssistBot',
|
|
53
333
|
'YouBot',
|
|
54
|
-
'
|
|
55
|
-
'
|
|
56
|
-
'
|
|
57
|
-
'GeminiBot',
|
|
58
|
-
'KagiBot',
|
|
59
|
-
'NeevaBot',
|
|
334
|
+
'Kagibot',
|
|
335
|
+
'PetalBot',
|
|
336
|
+
'ExaSearchBot',
|
|
60
337
|
'PhindBot',
|
|
338
|
+
'Yeti',
|
|
339
|
+
'OAI-AdsBot',
|
|
61
340
|
],
|
|
62
|
-
|
|
63
|
-
'
|
|
64
|
-
'
|
|
65
|
-
'
|
|
66
|
-
'
|
|
67
|
-
'
|
|
68
|
-
'
|
|
69
|
-
'
|
|
70
|
-
'
|
|
341
|
+
'user-fetch': [
|
|
342
|
+
'ChatGPT-User',
|
|
343
|
+
'Claude-User',
|
|
344
|
+
'Perplexity-User',
|
|
345
|
+
'MistralAI-User',
|
|
346
|
+
'meta-externalfetcher',
|
|
347
|
+
'Amzn-User',
|
|
348
|
+
'Google-GeminiNotebook',
|
|
349
|
+
'kagi-fetcher',
|
|
350
|
+
'Kimi-User',
|
|
351
|
+
'TongyiBot',
|
|
71
352
|
],
|
|
353
|
+
agent: ['Google-Agent', 'NovaAct', 'Manus-User', 'Devin', 'FirecrawlAgent', 'TavilyBot'],
|
|
72
354
|
};
|
|
73
|
-
export const ALL_AI_CRAWLERS = [...AI_CRAWLERS.training, ...AI_CRAWLERS.search, ...AI_CRAWLERS.fetching];
|
|
74
355
|
/**
|
|
75
|
-
*
|
|
76
|
-
*
|
|
77
|
-
*
|
|
356
|
+
* Tokens recognised for matching but never recommended: renamed, retired, or
|
|
357
|
+
* never real. A site that lists these is not wrong, but the rules are inert, so
|
|
358
|
+
* the audit says so rather than counting them as coverage.
|
|
359
|
+
*/
|
|
360
|
+
export const LEGACY_AI_CRAWLERS = {
|
|
361
|
+
'Claude-Web': 'Never documented by Anthropic and absent from its current crawler page. Use ClaudeBot.',
|
|
362
|
+
'Anthropic-AI': 'Never documented by Anthropic. Use ClaudeBot.',
|
|
363
|
+
'Google-NotebookLM': 'Renamed to Google-GeminiNotebook on 2026-07-17.',
|
|
364
|
+
'GoogleAgent-Mariner': 'Project Mariner was discontinued on 2026-05-04; superseded by Google-Agent.',
|
|
365
|
+
'Cohere-AI': 'Cohere states it operates no web crawlers.',
|
|
366
|
+
'cohere-training-data-crawler': 'Cohere states it operates no web crawlers.',
|
|
367
|
+
ExaBot: 'Superseded by ExaSearchBot.',
|
|
368
|
+
NeevaBot: 'Neeva was dissolved in 2023.',
|
|
369
|
+
Gemini: 'Not a real user-agent token. Gemini training and grounding are controlled by Google-Extended.',
|
|
370
|
+
GeminiBot: 'Not a real user-agent token. Gemini training and grounding are controlled by Google-Extended.',
|
|
371
|
+
'DeepSeek-AI': 'Not a documented token. The community-observed crawler identifies as DeepSeekBot.',
|
|
372
|
+
Goose: 'Block Goose is an agent framework that signs requests with Web Bot Auth; it has no robots.txt token.',
|
|
373
|
+
AwarioBot: 'Awario is a social-listening tool, not an AI crawler.',
|
|
374
|
+
AwarioRssBot: 'Awario is a social-listening tool, not an AI crawler.',
|
|
375
|
+
AwarioSmartBot: 'Awario is a social-listening tool, not an AI crawler.',
|
|
376
|
+
Operator: 'OpenAI Operator was discontinued on 2025-08-31 and never had a documented robots.txt token.',
|
|
377
|
+
};
|
|
378
|
+
export const ALL_AI_CRAWLERS = [
|
|
379
|
+
...AI_CRAWLERS.training,
|
|
380
|
+
...AI_CRAWLERS.search,
|
|
381
|
+
...AI_CRAWLERS['user-fetch'],
|
|
382
|
+
...AI_CRAWLERS.agent,
|
|
383
|
+
];
|
|
384
|
+
/**
|
|
385
|
+
* The crawlers that matter most as of September 2026, by request share
|
|
386
|
+
* (Cloudflare Radar) and by what a site loses when each is blocked. Used for
|
|
387
|
+
* reporting, access probes, and remediation advice.
|
|
388
|
+
*
|
|
389
|
+
* Composition: the six highest-volume clients (Googlebot's AI use is governed
|
|
390
|
+
* by Google-Extended, Applebot's by Applebot-Extended, hence the opt-out tokens
|
|
391
|
+
* standing in for them), the broadest training corpus (CCBot), the three search
|
|
392
|
+
* bots whose absence costs citations, and the one user-triggered fetcher with
|
|
393
|
+
* material volume.
|
|
78
394
|
*/
|
|
79
395
|
export const CORE_AI_CRAWLERS = [
|
|
80
396
|
'GPTBot',
|
|
81
397
|
'ClaudeBot',
|
|
82
|
-
'
|
|
83
|
-
'Claude-SearchBot',
|
|
398
|
+
'Meta-ExternalAgent',
|
|
84
399
|
'Google-Extended',
|
|
85
|
-
'
|
|
86
|
-
'
|
|
400
|
+
'Applebot-Extended',
|
|
401
|
+
'Amazonbot',
|
|
402
|
+
'Bytespider',
|
|
87
403
|
'CCBot',
|
|
404
|
+
'OAI-SearchBot',
|
|
405
|
+
'Claude-SearchBot',
|
|
406
|
+
'PerplexityBot',
|
|
407
|
+
'ChatGPT-User',
|
|
88
408
|
];
|
|
89
409
|
/**
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
410
|
+
* Core crawlers that actually issue HTTP requests.
|
|
411
|
+
*
|
|
412
|
+
* `Google-Extended` and `Applebot-Extended` are robots.txt control tokens: they
|
|
413
|
+
* govern how an already-crawled page may be used, and no request ever carries
|
|
414
|
+
* them as a user agent. Probing a site with those strings tests nothing, so
|
|
415
|
+
* access checks use this list instead of the full core set.
|
|
416
|
+
*/
|
|
417
|
+
export const PROBEABLE_CORE_CRAWLERS = CORE_AI_CRAWLERS.filter((token) => CRAWLER_META[token]?.tokenOnly !== true);
|
|
418
|
+
/** Look up a crawler's metadata case-insensitively. */
|
|
419
|
+
export function crawlerInfo(token) {
|
|
420
|
+
const key = Object.keys(CRAWLER_META).find((k) => k.toLowerCase() === token.toLowerCase());
|
|
421
|
+
return key === undefined ? undefined : CRAWLER_META[key];
|
|
422
|
+
}
|
|
423
|
+
/** Purpose bucket a token belongs to, or `undefined` when it is not a known AI client. */
|
|
424
|
+
export function crawlerPurpose(token) {
|
|
425
|
+
const lower = token.toLowerCase();
|
|
426
|
+
for (const [purpose, tokens] of Object.entries(AI_CRAWLERS)) {
|
|
427
|
+
if (tokens.some((t) => t.toLowerCase() === lower))
|
|
428
|
+
return purpose;
|
|
429
|
+
}
|
|
430
|
+
return undefined;
|
|
431
|
+
}
|
|
432
|
+
/** Explanation for a retired or fictional token, or `undefined` when the token is current. */
|
|
433
|
+
export function legacyCrawlerNote(token) {
|
|
434
|
+
const key = Object.keys(LEGACY_AI_CRAWLERS).find((k) => k.toLowerCase() === token.toLowerCase());
|
|
435
|
+
return key === undefined ? undefined : LEGACY_AI_CRAWLERS[key];
|
|
436
|
+
}
|
|
437
|
+
/**
|
|
438
|
+
* Weight per check, summing to 100. A check's own `meta.weight` overrides this
|
|
439
|
+
* map.
|
|
440
|
+
*
|
|
441
|
+
* The 4.0 distribution follows the evidence about what actually stops an agent,
|
|
442
|
+
* rather than what is easiest to check:
|
|
443
|
+
*
|
|
444
|
+
* - **Content (33)** leads, because the failure that breaks the most agents is
|
|
445
|
+
* a page with nothing in the HTML. Most crawlers do not run JavaScript, and a
|
|
446
|
+
* site whose content only appears after hydration is invisible to them no
|
|
447
|
+
* matter how many discovery files it publishes. `agent-operability` joins it
|
|
448
|
+
* at 7: browser agents read the accessibility tree, and a page of unnamed
|
|
449
|
+
* controls cannot be operated at all.
|
|
450
|
+
* - **Access (24)** is second, because a WAF rule or a `nosnippet` directive
|
|
451
|
+
* silently undoes everything else. These are also the failures operators are
|
|
452
|
+
* least likely to know about.
|
|
453
|
+
* - **Discovery (21)** carries robots.txt at 9 and llms.txt at 5. llms.txt was
|
|
454
|
+
* the highest-weighted check in 3.x at 11; it is demoted because the evidence
|
|
455
|
+
* is that most published files are never fetched by an AI search crawler, and
|
|
456
|
+
* the vendors that do read it are coding agents. It matters, but not twice as
|
|
457
|
+
* much as whether the page has content.
|
|
458
|
+
* - **Protocols (13)** are all conditional: a site without the surface reports
|
|
459
|
+
* N/A and the weight leaves the denominator, so a blog is never marked down
|
|
460
|
+
* for lacking an API description.
|
|
461
|
+
* - **Policy (9)** covers declared usage rights and security contact.
|
|
462
|
+
*
|
|
463
|
+
* Checks on draft specifications (`ai-catalog`, `webmcp`, `commerce-discovery`)
|
|
464
|
+
* stay at 0. Scoring a site against a specification that may be renamed next
|
|
465
|
+
* quarter would make the number less trustworthy, not more.
|
|
93
466
|
*/
|
|
94
467
|
export const CHECK_WEIGHTS = {
|
|
95
|
-
|
|
96
|
-
'
|
|
97
|
-
'
|
|
98
|
-
'structured-data':
|
|
99
|
-
'
|
|
100
|
-
'
|
|
101
|
-
|
|
102
|
-
'
|
|
103
|
-
'
|
|
104
|
-
'
|
|
105
|
-
|
|
106
|
-
'
|
|
107
|
-
|
|
108
|
-
'
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
'
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
'
|
|
468
|
+
/* Content — is there substance an agent can read? */
|
|
469
|
+
'html-rendering': 11,
|
|
470
|
+
'agent-operability': 7,
|
|
471
|
+
'structured-data': 6,
|
|
472
|
+
'seo-basics': 5,
|
|
473
|
+
'content-negotiation': 4,
|
|
474
|
+
/* Discovery — can an agent find the machine-readable entry points? */
|
|
475
|
+
'robots-txt': 9,
|
|
476
|
+
'llms-txt': 5,
|
|
477
|
+
'http-headers': 4,
|
|
478
|
+
sitemap: 2,
|
|
479
|
+
'meta-tags': 1,
|
|
480
|
+
/* Access — can an agent actually retrieve it? */
|
|
481
|
+
'agent-access': 9,
|
|
482
|
+
'ai-directives': 6,
|
|
483
|
+
'http-hygiene': 4,
|
|
484
|
+
'tls-https': 3,
|
|
485
|
+
'crawl-efficiency': 2,
|
|
486
|
+
/* Policy — what usage rights are declared, and do they agree? */
|
|
487
|
+
'usage-policy': 4,
|
|
488
|
+
'security-txt': 3,
|
|
489
|
+
rsl: 2,
|
|
490
|
+
/* Protocols — what can an agent call? All conditional: N/A when absent. */
|
|
491
|
+
'api-discovery': 4,
|
|
492
|
+
'agent-card': 3,
|
|
493
|
+
'mcp-discovery': 3,
|
|
494
|
+
'agent-skills': 2,
|
|
495
|
+
'auth-discovery': 1,
|
|
496
|
+
/* Draft specifications: reported, never scored. */
|
|
497
|
+
'ai-catalog': 0,
|
|
498
|
+
webmcp: 0,
|
|
499
|
+
'commerce-discovery': 0,
|
|
500
|
+
};
|
|
501
|
+
/**
|
|
502
|
+
* Report grouping per check. A check's own `meta.category` takes precedence.
|
|
503
|
+
* Mirrors the `CHECK_WEIGHTS` pattern: data lives here, overrides live on the check.
|
|
504
|
+
*/
|
|
505
|
+
export const CHECK_CATEGORIES = {
|
|
506
|
+
'html-rendering': 'content',
|
|
507
|
+
'agent-operability': 'content',
|
|
508
|
+
'structured-data': 'content',
|
|
509
|
+
'seo-basics': 'content',
|
|
510
|
+
'content-negotiation': 'content',
|
|
511
|
+
'llms-txt': 'discovery',
|
|
512
|
+
'robots-txt': 'discovery',
|
|
513
|
+
'http-headers': 'discovery',
|
|
514
|
+
'meta-tags': 'discovery',
|
|
515
|
+
sitemap: 'discovery',
|
|
516
|
+
'agent-access': 'access',
|
|
517
|
+
'ai-directives': 'access',
|
|
518
|
+
'http-hygiene': 'access',
|
|
519
|
+
'crawl-efficiency': 'access',
|
|
520
|
+
'tls-https': 'access',
|
|
521
|
+
'usage-policy': 'policy',
|
|
522
|
+
'security-txt': 'policy',
|
|
523
|
+
rsl: 'policy',
|
|
524
|
+
'agent-card': 'protocols',
|
|
525
|
+
'mcp-discovery': 'protocols',
|
|
526
|
+
'api-discovery': 'protocols',
|
|
527
|
+
'agent-skills': 'protocols',
|
|
528
|
+
'auth-discovery': 'protocols',
|
|
529
|
+
'ai-catalog': 'protocols',
|
|
530
|
+
webmcp: 'protocols',
|
|
531
|
+
'commerce-discovery': 'protocols',
|
|
115
532
|
};
|
|
116
533
|
export const GRADES = [
|
|
117
534
|
{ min: 90, label: 'Excellent', color: 'green' },
|
|
@@ -125,6 +542,28 @@ export const GRADES = [
|
|
|
125
542
|
* Absence of a signal is neutral — it neither grants nor restricts.
|
|
126
543
|
*/
|
|
127
544
|
export const CONTENT_SIGNALS = ['search', 'ai-input', 'ai-train'];
|
|
545
|
+
/**
|
|
546
|
+
* Values of the optional fourth Content Signals field `use`, introduced by
|
|
547
|
+
* Cloudflare on 2026-07-01 and now emitted by its managed robots.txt:
|
|
548
|
+
* `immediate` (interact, store nothing), `reference` (index, excerpt, link
|
|
549
|
+
* back — the stated default) and `full` (summarize and reproduce).
|
|
550
|
+
* https://blog.cloudflare.com/content-independence-day-ai-options/
|
|
551
|
+
*/
|
|
552
|
+
export const CONTENT_SIGNAL_USE_VALUES = ['immediate', 'reference', 'full'];
|
|
553
|
+
/**
|
|
554
|
+
* IETF AIPREF vocabulary (draft-ietf-aipref-vocab-07, 2026-08-19). Two
|
|
555
|
+
* categories only: `train-ai` (modifying model parameters) and `search`
|
|
556
|
+
* (retrieval that links back to the original). Values are `y` / `n`; an absent
|
|
557
|
+
* token means "unknown", never "allowed".
|
|
558
|
+
*
|
|
559
|
+
* The drafts are pre-working-group-last-call and carry the "DO NOT REFLECT
|
|
560
|
+
* CONSENSUS" boilerplate, so ax-audit reports these directives without ever
|
|
561
|
+
* requiring them. Note the token order is inverted against Content Signals
|
|
562
|
+
* (`train-ai` here, `ai-train` there).
|
|
563
|
+
* https://datatracker.ietf.org/wg/aipref/documents/
|
|
564
|
+
*/
|
|
565
|
+
export const AIPREF_TOKENS = ['train-ai', 'search'];
|
|
566
|
+
export const AIPREF_VALUES = ['y', 'n'];
|
|
128
567
|
/**
|
|
129
568
|
* Really Simple Licensing 1.0 (https://rslstandard.org/rsl): machine-readable licensing
|
|
130
569
|
* terms for content, discovered via robots.txt `License:`, HTTP `Link: rel="license"`,
|
|
@@ -145,7 +584,96 @@ export const RSL_PAYMENT_TYPES = [
|
|
|
145
584
|
'attribution',
|
|
146
585
|
'free',
|
|
147
586
|
];
|
|
587
|
+
/**
|
|
588
|
+
* A2A Agent Card (https://a2a-protocol.org).
|
|
589
|
+
*
|
|
590
|
+
* Two generations are in the wild. A2A 1.0 (2026-03-12) replaced the top-level
|
|
591
|
+
* `url`, `protocolVersion`, `preferredTransport` and `additionalInterfaces`
|
|
592
|
+
* fields with one `supportedInterfaces[]` array; 0.3-shaped cards are still the
|
|
593
|
+
* majority deployed. Required-field lists are taken from `specification/a2a.proto`
|
|
594
|
+
* (1.0) and `specification/json/a2a.json` at tag v0.3.0.
|
|
595
|
+
*/
|
|
596
|
+
export const AGENT_CARD_REQUIRED_V1 = [
|
|
597
|
+
'name',
|
|
598
|
+
'description',
|
|
599
|
+
'version',
|
|
600
|
+
'capabilities',
|
|
601
|
+
'supportedInterfaces',
|
|
602
|
+
'defaultInputModes',
|
|
603
|
+
'defaultOutputModes',
|
|
604
|
+
'skills',
|
|
605
|
+
];
|
|
606
|
+
export const AGENT_CARD_REQUIRED_V03 = [
|
|
607
|
+
'name',
|
|
608
|
+
'description',
|
|
609
|
+
'url',
|
|
610
|
+
'version',
|
|
611
|
+
'protocolVersion',
|
|
612
|
+
'capabilities',
|
|
613
|
+
'defaultInputModes',
|
|
614
|
+
'defaultOutputModes',
|
|
615
|
+
'skills',
|
|
616
|
+
];
|
|
617
|
+
/** Transport bindings a 1.0 interface may declare. */
|
|
618
|
+
export const A2A_PROTOCOL_BINDINGS = ['JSONRPC', 'GRPC', 'HTTP+JSON'];
|
|
619
|
+
/** Media type registered for A2A payloads (spec §14.1.1). */
|
|
620
|
+
export const A2A_MEDIA_TYPE = 'application/a2a+json';
|
|
621
|
+
/** @deprecated Superseded by AGENT_CARD_REQUIRED_V03 / _V1 in 3.7. Kept for one minor. */
|
|
148
622
|
export const AGENT_JSON_REQUIRED_FIELDS = ['name', 'description', 'url', 'skills'];
|
|
623
|
+
/**
|
|
624
|
+
* Model Context Protocol.
|
|
625
|
+
*
|
|
626
|
+
* `/.well-known/mcp.json` was never part of the specification; ax-audit
|
|
627
|
+
* recommended it before the ecosystem settled. The actual discovery work is
|
|
628
|
+
* SEP-2127 (open draft) and the `experimental-ext-server-card` repository,
|
|
629
|
+
* which recommend a **server card** at `<streamable-http-url>/server-card` and
|
|
630
|
+
* an entry in `/.well-known/ai-catalog.json`. Cloudflare and Mintlify serve one
|
|
631
|
+
* at `/.well-known/mcp/server-card.json`.
|
|
632
|
+
*
|
|
633
|
+
* Server cards deliberately carry no `tools[]`: tool lists come from a live
|
|
634
|
+
* `tools/list` call, not from a static document that would immediately drift.
|
|
635
|
+
*/
|
|
636
|
+
export const MCP_SERVER_CARD_MEDIA_TYPE = 'application/mcp-server-card+json';
|
|
637
|
+
export const MCP_SERVER_CARD_REQUIRED = ['$schema', 'name', 'version', 'description'];
|
|
638
|
+
export const MCP_REMOTE_TYPES = ['streamable-http', 'sse'];
|
|
639
|
+
/**
|
|
640
|
+
* Released MCP protocol versions, newest first. `2026-07-28` removed sessions
|
|
641
|
+
* and `initialize`, made `MCP-Protocol-Version`, `Mcp-Method` and `Mcp-Name`
|
|
642
|
+
* mandatory on every POST, and added the `server/discover` RPC.
|
|
643
|
+
* https://modelcontextprotocol.io/specification/versioning
|
|
644
|
+
*/
|
|
645
|
+
export const MCP_PROTOCOL_VERSIONS = ['2026-07-28', '2025-11-25', '2025-06-18', '2025-03-26', '2024-11-05'];
|
|
646
|
+
/** Versions old enough that a card advertising only these is worth flagging. */
|
|
647
|
+
export const MCP_STALE_PROTOCOL_VERSIONS = ['2024-11-05', '2025-03-26'];
|
|
648
|
+
/** Media types an ai-catalog entry may declare, per the Agent Card WG draft. */
|
|
649
|
+
export const AI_CATALOG_ENTRY_TYPES = {
|
|
650
|
+
'application/mcp-server-card+json': 'MCP server card',
|
|
651
|
+
'application/a2a-agent-card+json': 'A2A agent card',
|
|
652
|
+
'application/ai-catalog+json': 'nested AI catalog',
|
|
653
|
+
};
|
|
654
|
+
/**
|
|
655
|
+
* Where API descriptions live, in order of authority.
|
|
656
|
+
*
|
|
657
|
+
* `/.well-known/openapi.json` is a folk convention: it is not IANA-registered,
|
|
658
|
+
* and the OpenAPI specification recommends the file name `openapi.json` /
|
|
659
|
+
* `openapi.yaml` without prescribing a location. It stays first because
|
|
660
|
+
* ax-audit recommended it before 3.7 and sites followed that advice, but the
|
|
661
|
+
* common real-world locations are probed too.
|
|
662
|
+
*/
|
|
663
|
+
export const API_DESCRIPTION_PATHS = [
|
|
664
|
+
'/.well-known/openapi.json',
|
|
665
|
+
'/openapi.json',
|
|
666
|
+
'/openapi.yaml',
|
|
667
|
+
'/.well-known/openapi.yaml',
|
|
668
|
+
'/api/openapi.json',
|
|
669
|
+
'/v1/openapi.json',
|
|
670
|
+
'/swagger.json',
|
|
671
|
+
'/api-docs',
|
|
672
|
+
'/asyncapi.json',
|
|
673
|
+
'/arazzo.json',
|
|
674
|
+
];
|
|
675
|
+
/** RFC 9264 linkset media type, required by RFC 9727 for the API catalog. */
|
|
676
|
+
export const LINKSET_MEDIA_TYPE = 'application/linkset+json';
|
|
149
677
|
export const SECURITY_TXT_REQUIRED_FIELDS = ['Contact', 'Expires'];
|
|
150
678
|
export const SECURITY_HEADERS = [
|
|
151
679
|
{ name: 'strict-transport-security', label: 'Strict-Transport-Security', critical: true },
|