ax-audit 3.6.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (201) hide show
  1. package/CHANGELOG.md +138 -0
  2. package/LICENSE +1 -1
  3. package/README.md +58 -36
  4. package/dist/baseline.d.ts +2 -0
  5. package/dist/baseline.d.ts.map +1 -1
  6. package/dist/baseline.js +42 -4
  7. package/dist/baseline.js.map +1 -1
  8. package/dist/check-ids.d.ts +19 -0
  9. package/dist/check-ids.d.ts.map +1 -0
  10. package/dist/check-ids.js +53 -0
  11. package/dist/check-ids.js.map +1 -0
  12. package/dist/checks/agent-access.d.ts +23 -6
  13. package/dist/checks/agent-access.d.ts.map +1 -1
  14. package/dist/checks/agent-access.js +200 -54
  15. package/dist/checks/agent-access.js.map +1 -1
  16. package/dist/checks/agent-card.d.ts +37 -0
  17. package/dist/checks/agent-card.d.ts.map +1 -0
  18. package/dist/checks/agent-card.js +352 -0
  19. package/dist/checks/agent-card.js.map +1 -0
  20. package/dist/checks/agent-operability.d.ts +66 -0
  21. package/dist/checks/agent-operability.d.ts.map +1 -0
  22. package/dist/checks/agent-operability.js +383 -0
  23. package/dist/checks/agent-operability.js.map +1 -0
  24. package/dist/checks/agent-skills.d.ts +24 -0
  25. package/dist/checks/agent-skills.d.ts.map +1 -0
  26. package/dist/checks/agent-skills.js +316 -0
  27. package/dist/checks/agent-skills.js.map +1 -0
  28. package/dist/checks/ai-catalog.d.ts +28 -0
  29. package/dist/checks/ai-catalog.d.ts.map +1 -0
  30. package/dist/checks/ai-catalog.js +254 -0
  31. package/dist/checks/ai-catalog.js.map +1 -0
  32. package/dist/checks/ai-directives.d.ts +57 -0
  33. package/dist/checks/ai-directives.d.ts.map +1 -0
  34. package/dist/checks/ai-directives.js +263 -0
  35. package/dist/checks/ai-directives.js.map +1 -0
  36. package/dist/checks/api-discovery.d.ts +26 -0
  37. package/dist/checks/api-discovery.d.ts.map +1 -0
  38. package/dist/checks/api-discovery.js +432 -0
  39. package/dist/checks/api-discovery.js.map +1 -0
  40. package/dist/checks/auth-discovery.d.ts +28 -0
  41. package/dist/checks/auth-discovery.d.ts.map +1 -0
  42. package/dist/checks/auth-discovery.js +213 -0
  43. package/dist/checks/auth-discovery.js.map +1 -0
  44. package/dist/checks/commerce-discovery.d.ts +40 -0
  45. package/dist/checks/commerce-discovery.d.ts.map +1 -0
  46. package/dist/checks/commerce-discovery.js +295 -0
  47. package/dist/checks/commerce-discovery.js.map +1 -0
  48. package/dist/checks/content-negotiation.d.ts.map +1 -1
  49. package/dist/checks/content-negotiation.js +135 -20
  50. package/dist/checks/content-negotiation.js.map +1 -1
  51. package/dist/checks/crawl-efficiency.d.ts +13 -1
  52. package/dist/checks/crawl-efficiency.d.ts.map +1 -1
  53. package/dist/checks/crawl-efficiency.js +65 -1
  54. package/dist/checks/crawl-efficiency.js.map +1 -1
  55. package/dist/checks/frontmatter.d.ts +34 -0
  56. package/dist/checks/frontmatter.d.ts.map +1 -0
  57. package/dist/checks/frontmatter.js +100 -0
  58. package/dist/checks/frontmatter.js.map +1 -0
  59. package/dist/checks/html-rendering.d.ts.map +1 -1
  60. package/dist/checks/html-rendering.js +0 -1
  61. package/dist/checks/html-rendering.js.map +1 -1
  62. package/dist/checks/html-utils.d.ts +10 -0
  63. package/dist/checks/html-utils.d.ts.map +1 -1
  64. package/dist/checks/html-utils.js +19 -0
  65. package/dist/checks/html-utils.js.map +1 -1
  66. package/dist/checks/http-headers.d.ts.map +1 -1
  67. package/dist/checks/http-headers.js +82 -10
  68. package/dist/checks/http-headers.js.map +1 -1
  69. package/dist/checks/http-hygiene.d.ts +26 -0
  70. package/dist/checks/http-hygiene.d.ts.map +1 -0
  71. package/dist/checks/http-hygiene.js +257 -0
  72. package/dist/checks/http-hygiene.js.map +1 -0
  73. package/dist/checks/index.d.ts.map +1 -1
  74. package/dist/checks/index.js +24 -8
  75. package/dist/checks/index.js.map +1 -1
  76. package/dist/checks/llms-txt.d.ts +15 -0
  77. package/dist/checks/llms-txt.d.ts.map +1 -1
  78. package/dist/checks/llms-txt.js +162 -2
  79. package/dist/checks/llms-txt.js.map +1 -1
  80. package/dist/checks/mcp-discovery.d.ts +30 -0
  81. package/dist/checks/mcp-discovery.d.ts.map +1 -0
  82. package/dist/checks/mcp-discovery.js +523 -0
  83. package/dist/checks/mcp-discovery.js.map +1 -0
  84. package/dist/checks/meta-tags.d.ts.map +1 -1
  85. package/dist/checks/meta-tags.js +6 -5
  86. package/dist/checks/meta-tags.js.map +1 -1
  87. package/dist/checks/robots-parser.d.ts +110 -0
  88. package/dist/checks/robots-parser.d.ts.map +1 -0
  89. package/dist/checks/robots-parser.js +277 -0
  90. package/dist/checks/robots-parser.js.map +1 -0
  91. package/dist/checks/robots-txt.d.ts +2 -20
  92. package/dist/checks/robots-txt.d.ts.map +1 -1
  93. package/dist/checks/robots-txt.js +219 -120
  94. package/dist/checks/robots-txt.js.map +1 -1
  95. package/dist/checks/rsl.d.ts +0 -2
  96. package/dist/checks/rsl.d.ts.map +1 -1
  97. package/dist/checks/rsl.js +1 -11
  98. package/dist/checks/rsl.js.map +1 -1
  99. package/dist/checks/security-txt.d.ts.map +1 -1
  100. package/dist/checks/security-txt.js +0 -1
  101. package/dist/checks/security-txt.js.map +1 -1
  102. package/dist/checks/seo-basics.d.ts.map +1 -1
  103. package/dist/checks/seo-basics.js +0 -1
  104. package/dist/checks/seo-basics.js.map +1 -1
  105. package/dist/checks/sitemap.d.ts.map +1 -1
  106. package/dist/checks/sitemap.js +0 -1
  107. package/dist/checks/sitemap.js.map +1 -1
  108. package/dist/checks/structured-data.d.ts.map +1 -1
  109. package/dist/checks/structured-data.js +215 -4
  110. package/dist/checks/structured-data.js.map +1 -1
  111. package/dist/checks/structured-fields.d.ts +46 -0
  112. package/dist/checks/structured-fields.d.ts.map +1 -0
  113. package/dist/checks/structured-fields.js +112 -0
  114. package/dist/checks/structured-fields.js.map +1 -0
  115. package/dist/checks/surface.d.ts +59 -0
  116. package/dist/checks/surface.d.ts.map +1 -0
  117. package/dist/checks/surface.js +106 -0
  118. package/dist/checks/surface.js.map +1 -0
  119. package/dist/checks/tls-https.d.ts.map +1 -1
  120. package/dist/checks/tls-https.js +0 -1
  121. package/dist/checks/tls-https.js.map +1 -1
  122. package/dist/checks/usage-policy.d.ts +53 -0
  123. package/dist/checks/usage-policy.d.ts.map +1 -0
  124. package/dist/checks/usage-policy.js +339 -0
  125. package/dist/checks/usage-policy.js.map +1 -0
  126. package/dist/checks/utils.d.ts +25 -1
  127. package/dist/checks/utils.d.ts.map +1 -1
  128. package/dist/checks/utils.js +33 -1
  129. package/dist/checks/utils.js.map +1 -1
  130. package/dist/checks/waf.d.ts +75 -0
  131. package/dist/checks/waf.d.ts.map +1 -0
  132. package/dist/checks/waf.js +203 -0
  133. package/dist/checks/waf.js.map +1 -0
  134. package/dist/checks/webmcp.d.ts +55 -0
  135. package/dist/checks/webmcp.d.ts.map +1 -0
  136. package/dist/checks/webmcp.js +209 -0
  137. package/dist/checks/webmcp.js.map +1 -0
  138. package/dist/checks/well-known.d.ts +38 -0
  139. package/dist/checks/well-known.d.ts.map +1 -0
  140. package/dist/checks/well-known.js +202 -0
  141. package/dist/checks/well-known.js.map +1 -0
  142. package/dist/cli.d.ts +14 -0
  143. package/dist/cli.d.ts.map +1 -1
  144. package/dist/cli.js +124 -4
  145. package/dist/cli.js.map +1 -1
  146. package/dist/constants.d.ts +195 -14
  147. package/dist/constants.d.ts.map +1 -1
  148. package/dist/constants.js +598 -70
  149. package/dist/constants.js.map +1 -1
  150. package/dist/fetcher.d.ts.map +1 -1
  151. package/dist/fetcher.js +44 -14
  152. package/dist/fetcher.js.map +1 -1
  153. package/dist/guide-urls.js +1 -1
  154. package/dist/guide-urls.js.map +1 -1
  155. package/dist/index.d.ts +2 -1
  156. package/dist/index.d.ts.map +1 -1
  157. package/dist/index.js +1 -0
  158. package/dist/index.js.map +1 -1
  159. package/dist/orchestrator.d.ts.map +1 -1
  160. package/dist/orchestrator.js +5 -1
  161. package/dist/orchestrator.js.map +1 -1
  162. package/dist/reporter/html.d.ts +9 -0
  163. package/dist/reporter/html.d.ts.map +1 -1
  164. package/dist/reporter/html.js +49 -11
  165. package/dist/reporter/html.js.map +1 -1
  166. package/dist/reporter/markdown.d.ts.map +1 -1
  167. package/dist/reporter/markdown.js +36 -6
  168. package/dist/reporter/markdown.js.map +1 -1
  169. package/dist/reporter/terminal.d.ts.map +1 -1
  170. package/dist/reporter/terminal.js +36 -1
  171. package/dist/reporter/terminal.js.map +1 -1
  172. package/dist/scorer.d.ts +10 -0
  173. package/dist/scorer.d.ts.map +1 -1
  174. package/dist/scorer.js +22 -5
  175. package/dist/scorer.js.map +1 -1
  176. package/dist/types.d.ts +90 -3
  177. package/dist/types.d.ts.map +1 -1
  178. package/docs/architecture.md +27 -11
  179. package/docs/checks.md +285 -52
  180. package/docs/cli.md +36 -0
  181. package/docs/concepts.md +27 -13
  182. package/docs/faq.md +18 -6
  183. package/docs/getting-started.md +20 -13
  184. package/docs/roadmap.md +367 -0
  185. package/package.json +13 -5
  186. package/dist/checks/agent-json.d.ts +0 -14
  187. package/dist/checks/agent-json.d.ts.map +0 -1
  188. package/dist/checks/agent-json.js +0 -167
  189. package/dist/checks/agent-json.js.map +0 -1
  190. package/dist/checks/mcp.d.ts +0 -4
  191. package/dist/checks/mcp.d.ts.map +0 -1
  192. package/dist/checks/mcp.js +0 -162
  193. package/dist/checks/mcp.js.map +0 -1
  194. package/dist/checks/openapi.d.ts +0 -4
  195. package/dist/checks/openapi.d.ts.map +0 -1
  196. package/dist/checks/openapi.js +0 -121
  197. package/dist/checks/openapi.js.map +0 -1
  198. package/dist/checks/well-known-ai.d.ts +0 -17
  199. package/dist/checks/well-known-ai.d.ts.map +0 -1
  200. package/dist/checks/well-known-ai.js +0 -123
  201. package/dist/checks/well-known-ai.js.map +0 -1
package/docs/faq.md CHANGED
@@ -24,9 +24,21 @@ Weighted checks (14) sum to 100% and determine your overall score. Informational
24
24
 
25
25
  This is the most important caveat in the tool. ax-audit's probe sends a user-agent *containing* the crawler token (e.g. `...GPTBot/1.0`) but it is **not** the real, verified crawler. If your WAF verifies bots cryptographically ([Web Bot Auth](./concepts.md)) or by IP range, it will correctly pass the genuine GPTBot while rejecting ax-audit's unverified probe. **Before changing any WAF rule, confirm against your WAF logs** whether real crawler traffic is actually being served. If it is, this finding is a false positive for your setup.
26
26
 
27
- ### `well-known-ai` is low — should I worry?
27
+ ### Where did `well-known-ai` go?
28
28
 
29
- No. It's scored as *coverage bonus* over five emerging, partly-competing files (`ai.txt`, `genai.txt`, `ai-plugin.json`, `agents.json`, `nlweb.json`). None is universally adopted; a low score here is not a defect. Implement the ones relevant to your stack.
29
+ Removed in 4.0. Three of the five files it scored turned out to have no consumer at all: `/.well-known/nlweb.json` appears in no NLWeb release, document or commit; `genai.txt` has no specification; and `/ai-plugin.json` described ChatGPT plugins, shut down in April 2024. Marking a site down for omitting a file nobody specified is worse than not checking it.
30
+
31
+ ### A check says `n/a` — is that bad?
32
+
33
+ No, it is the point. A blog has no API to describe and no MCP server to advertise, so those checks report `n/a` and leave the score entirely rather than counting as failures. Everything counted against your site is something your site could have done.
34
+
35
+ If you are *planning* to build one of those surfaces and want it audited anyway, use `--profile api`, `--profile mcp`, or `--profile all`.
36
+
37
+ ### My score changed after upgrading to 4.0
38
+
39
+ Expected. 4.0 redistributed the weights and made protocol checks conditional. Sites with no API or MCP surface generally rise, because they are no longer marked down for lacking things they do not have.
40
+
41
+ Your saved baseline still works, but regression gating is suspended until you re-save it: comparing across a scoring change would report regressions your site did not cause. Run `ax-audit <url> --save-baseline .ax-baseline.json` to resume gating.
30
42
 
31
43
  ### `crawl-efficiency` says no compression, but my CDN compresses
32
44
 
@@ -34,13 +46,13 @@ The check reads the `Content-Encoding` header on the response it received. If a
34
46
 
35
47
  ### `content-negotiation` fails but I don't serve Markdown
36
48
 
37
- That's expected — most sites don't yet. It's informational (weight 0). Adopt it when you're ready; the [guide](https://lucioduran.com/projects/ax-audit/guides/content-negotiation) covers Cloudflare/Vercel zero-code options.
49
+ That's expected — most sites don't yet. It's informational (weight 0). Adopt it when you're ready; the [guide](https://axrush.com/guides/content-negotiation) covers Cloudflare/Vercel zero-code options.
38
50
 
39
51
  ## Running the tool
40
52
 
41
53
  ### My WAF is blocking ax-audit itself
42
54
 
43
- ax-audit sends a `User-Agent` of `ax-audit/<version> (+https://github.com/lucioduran/ax-audit)`. If your firewall challenges unknown agents, allowlist that UA (or the IP you run from) for the duration of the audit. Note that several checks deliberately send *other* user-agents (`agent-access`) and unusual `Accept` headers (`content-negotiation`) — a WAF rejecting those is itself a finding, not a tool bug.
55
+ ax-audit sends a `User-Agent` of `ax-audit/<version> (+https://github.com/duranitech/ax-audit)`. If your firewall challenges unknown agents, allowlist that UA (or the IP you run from) for the duration of the audit. Note that several checks deliberately send *other* user-agents (`agent-access`) and unusual `Accept` headers (`content-negotiation`) — a WAF rejecting those is itself a finding, not a tool bug.
44
56
 
45
57
  ### How do I audit a staging site behind auth?
46
58
 
@@ -70,8 +82,8 @@ Yes — `import { audit } from 'ax-audit'` returns a typed `AuditReport`. See [a
70
82
 
71
83
  ### How do I generate the files ax-audit checks for?
72
84
 
73
- Use [ax-init](https://github.com/lucioduran/ax-init) — it generates `llms.txt`, `robots.txt`, `agent.json`, `mcp.json`, `security.txt`, structured data, and header snippets, then you verify with `npx ax-audit`.
85
+ Use [ax-init](https://github.com/duranitech/ax-init) — it generates `llms.txt`, `robots.txt`, an Agent Card, `security.txt`, structured data, and header snippets, then you verify with `npx ax-audit`. Check its output against this tool: the standards moved in 2026, and a generator written against the older paths will produce files agents no longer look for.
74
86
 
75
87
  ## Still stuck?
76
88
 
77
- Open an issue at [github.com/lucioduran/ax-audit/issues](https://github.com/lucioduran/ax-audit/issues) with the output of `npx ax-audit <url> --verbose`.
89
+ Open an issue at [github.com/duranitech/ax-audit/issues](https://github.com/duranitech/ax-audit/issues) with the output of `npx ax-audit <url> --verbose`.
@@ -39,9 +39,13 @@ npx ax-audit https://your-site.com --only-failures
39
39
 
40
40
  AI agents interact with your site differently than browsers: most don't execute JavaScript, they look for machine-readable discovery files, and they respect (or at least read) your declared crawler policy. The audit measures three layers — if you're new to the standards involved (llms.txt, A2A, MCP, RSL, Content Signals), read [concepts.md](./concepts.md) first:
41
41
 
42
- 1. **Can agents find and read your content?** (`html-rendering`, `robots-txt`, `sitemap`, `tls-https`, `agent-access`)
43
- 2. **Did you publish the AI-specific surface?** (`llms-txt`, `agent-json`, `mcp`, `openapi`, `well-known-ai`, `meta-tags`, `structured-data`)
44
- 3. **Is the interaction efficient and well-governed?** (`content-negotiation`, `crawl-efficiency`, `rsl`, Content Signals, `http-headers`, `security-txt`, `seo-basics`)
42
+ 1. **Content** — is there substance an agent can read, and can a browser agent act on it? (`html-rendering`, `agent-operability`, `structured-data`, `seo-basics`, `content-negotiation`)
43
+ 2. **Access** — can an agent actually retrieve it? (`agent-access`, `ai-directives`, `http-hygiene`, `tls-https`, `crawl-efficiency`)
44
+ 3. **Discovery** — can an agent find your machine-readable files? (`robots-txt`, `llms-txt`, `http-headers`, `sitemap`, `meta-tags`)
45
+ 4. **Policy** — what usage rights do you declare, and do they agree? (`usage-policy`, `security-txt`, `rsl`)
46
+ 5. **Protocols** — what can an agent call? (`api-discovery`, `agent-card`, `mcp-discovery`, `agent-skills`, `auth-discovery`)
47
+
48
+ Protocol checks are conditional: if your site has no API to describe and no MCP server, they report **n/a** and leave the score alone rather than counting as failures.
45
49
 
46
50
  ## 3. Fix in impact order
47
51
 
@@ -49,14 +53,17 @@ The fastest path from Fair to Good, by weight and typical effort:
49
53
 
50
54
  | Step | Check | Weight | Typical effort |
51
55
  | --- | --- | --- | --- |
52
- | 1 | Create `/llms.txt` | 11% | 30 minutes — it's a Markdown file. `npx ax-init` generates it. |
53
- | 2 | Configure `robots.txt` for the 8 core AI crawlers | 11% | 15 minutes; `npx ax-init` generates this too |
54
- | 3 | Verify server-rendered content | 9% | Free if you SSR; significant if you ship an SPA shell |
55
- | 4 | Add JSON-LD structured data | 9% | 1–2 hours |
56
- | 5 | Security + discovery headers | 9% | 30 minutes of server config |
57
- | 6 | `agent.json` + `mcp.json` | 14% combined | An hour with the spec links in the guides |
56
+ | 1 | Verify server-rendered content | 11% | Free if you already render on the server; significant if you ship an SPA shell. Nothing else matters if an agent sees an empty page. |
57
+ | 2 | Confirm nothing blocks AI crawlers | 9% | 15 minutes in your WAF. Check `agent-access` first: your robots.txt may say one thing and your firewall another. |
58
+ | 3 | Configure `robots.txt` for the 12 core AI crawlers | 9% | 15 minutes; `npx ax-init` generates it |
59
+ | 4 | Name your buttons and label your inputs | 7% | Hours to days, and it is accessibility work you owed anyway |
60
+ | 5 | Check your page-level AI directives | 6% | 10 minutes. A stray `nosnippet` removes you from AI Overviews. |
61
+ | 6 | Add JSON-LD structured data | 6% | 1–2 hours |
62
+ | 7 | Create `/llms.txt` | 5% | 30 minutes — it is a Markdown file. Read the check's note on who actually fetches it first. |
63
+
64
+ Steps 1 and 2 are the ones worth doing today. They are also the two most likely to be silently broken: a hydration-only page and a firewall rule are both invisible from inside the site.
58
65
 
59
- The remaining weighted checks (`seo-basics`, `security-txt`, `meta-tags`, `openapi`, `tls-https`, `sitemap`, `well-known-ai`) are mostly configuration; the remediation guides give exact snippets for Nginx, Vercel, Netlify, and Express.
66
+ The remaining checks (`seo-basics`, `content-negotiation`, `usage-policy`, `http-hygiene`, `http-headers`, `security-txt`, `tls-https`, `sitemap`, `crawl-efficiency`, `rsl`, `meta-tags`) are mostly configuration; the remediation guides give exact snippets for Nginx, Vercel, Netlify, and Express.
60
67
 
61
68
  Re-run after each fix — all requests are cached per run, so audits are fast and cheap.
62
69
 
@@ -91,11 +98,11 @@ npx ax-audit https://your-site.com --checks agent-access
91
98
 
92
99
  - **"My score seems harsh."** The audit measures the AI-agent surface, not site quality. A beautiful SPA with no llms.txt, no structured data, and an empty `#root` div is genuinely poor AX — that's the point of the tool.
93
100
  - **"A check crashed / network error."** Transient failures retry automatically (`--retries`, default 2). For slow staging environments raise `--timeout`.
94
- - **"Which findings are safe to ignore?"** See the [FAQ](./faq.md) — notably the `agent-access` verified-bots caveat and `well-known-ai`, which is coverage bonus rather than baseline.
101
+ - **"Which findings are safe to ignore?"** See the [FAQ](./faq.md) — notably the `agent-access` verified-bots caveat, and the checks resting on draft specifications, which never affect your score.
95
102
 
96
103
  ## Next steps
97
104
 
98
- - [checks.md](./checks.md) — exact scoring of all 18 checks
105
+ - [checks.md](./checks.md) — exact scoring of all 26 checks, with the weight table
99
106
  - [concepts.md](./concepts.md) — the AX standards landscape explained
100
107
  - [cli.md](./cli.md) — every flag · [ci.md](./ci.md) — CI recipes · [api.md](./api.md) — programmatic use
101
- - [ax-init](https://github.com/lucioduran/ax-init) — generates most of the files this tool audits
108
+ - [ax-init](https://github.com/duranitech/ax-init) — generates most of the files this tool audits
@@ -0,0 +1,367 @@
1
+ # Roadmap: 3.7 → 4.0 — **completed 2026-09-04**
2
+
3
+ *Research snapshot and implementation plan. All four releases shipped; this document is kept as the record of what was verified and why each decision was made.*
4
+
5
+ ax-audit has not shipped since 3.6.0 (2026-06-09). This document records what changed in the agent-web ecosystem since then, which existing checks are now wrong or stale, which new checks are worth adding, and a phased implementation plan that respects the 3.x scoring policy (no downward score changes until 4.0).
6
+
7
+ Sources were verified against primary specs, vendor docs, IANA, IETF datatracker and GitHub on 2026-09-04. Items marked **(secondary)** rely on trade press or third-party write-ups only.
8
+
9
+ ---
10
+
11
+ ## 1. Executive summary
12
+
13
+ **Three existing checks probe paths that are no longer (or never were) the standard:**
14
+
15
+ | Check | Today | Reality (Sept 2026) |
16
+ | --- | --- | --- |
17
+ | `agent-json` | `/.well-known/agent.json`, hints `protocolVersion: "0.2.0"`, expects `authentication` | A2A moved to **`/.well-known/agent-card.json`** in v0.3.0 (2025-07-30), IANA-registered permanent. A2A **1.0.1** (2026-05-28) replaced `url`/`protocolVersion`/`preferredTransport` with `supportedInterfaces[]`; `authentication` → `securitySchemes`. |
18
+ | `mcp` | `/.well-known/mcp.json` with `tools[]`, hints `2024-11-05` | **Never a spec convention.** Current draft (SEP-2127 + `experimental-ext-server-card`) is `<mcp-endpoint>/server-card` (`application/mcp-server-card+json`), Cloudflare/Mintlify serve `/.well-known/mcp/server-card.json`; umbrella `/.well-known/ai-catalog.json`. Server cards carry **no** `tools[]`. Protocol version is **2026-07-28**. Auth discovery via RFC 9728 is mandatory for remote servers. |
19
+ | `well-known-ai` | ai.txt, genai.txt, ai-plugin.json, agents.json, nlweb.json | `nlweb.json` **does not exist** (NLWeb uses `/ask` + `/mcp`); `genai.txt` has no spec; `ai-plugin.json` dead since 2024-04-09; Wildcard `agents.json` dormant since 2025-08-21; Spawning `ai.txt` lives at root and has ~0 adoption. The whole bundle needs replacing. |
20
+
21
+ **The crawler list has fake, retired and misclassified tokens** (`Gemini`, `GeminiBot`, `DeepSeek-AI`, `NeevaBot`, `Goose`, `Awario*`; `ChatGPT-User`/`Claude-User`/`Perplexity-User`/`MistralAI-User`/`meta-externalfetcher` are user-triggered fetchers, not training bots) and is missing the bots that now dominate traffic (`meta-webindexer`, `Amzn-SearchBot`, `Amzn-User`, `MistralAI-Index`, `MistralAI-Training`, `Google-GeminiNotebook`, `Applebot`, `ExaSearchBot`).
22
+
23
+ **Two competitors now define the reference bar:** Google Lighthouse 13.3 shipped an "Agentic Browsing" category (2026-05-07) and Cloudflare launched an "Agent Readiness" score (2026-04-17). Both check things ax-audit does not: ARD / `ai-catalog.json`, WebMCP declarative forms, agent-skills index, RFC 9727 API catalog, OAuth discovery (RFC 8414/9728), Web Bot Auth key directories, Link-header discovery. Ora's AgentReady v1.0 (with Vercel and Mintlify, Aug 2026) adds HTTP-status honesty, `429 + Retry-After`, and conditional "N/A" scoring.
24
+
25
+ **New signals worth auditing:** robots meta AI directives (`nosnippet`, `max-snippet`, `noarchive`, `nocache` are the only page-level controls Google and Bing actually honor for AI answers), IETF AIPREF `Content-Usage`, Content Signals `use=` field, TDMRep, WAF challenge vs hard block vs 402 pay-per-crawl classification, browser-agent operability heuristics, llms.txt v2 (subpath files, `rel="describedby"`, `.md` mirrors), and Markdown-for-Agents token headers.
26
+
27
+ **Recommended shape:** three minor releases (3.7, 3.8, 3.9) that fix stale probes and add ~10 informational checks without lowering any score, then **4.0** that redistributes weights, introduces conditional (N/A) checks and report categories, and retires the legacy bundle.
28
+
29
+ ---
30
+
31
+ ## 2. Ecosystem changes since June 2026 (verified)
32
+
33
+ ### 2.1 Protocols
34
+
35
+ - **A2A 1.0.0** (2026-03-12, breaking) and **1.0.1** (2026-05-28). Card required fields: `name`, `description`, `version`, `capabilities`, `supportedInterfaces[]{url, protocolBinding ∈ JSONRPC|GRPC|HTTP+JSON, protocolVersion}`, `defaultInputModes[]`, `defaultOutputModes[]`, `skills[]`. Optional: `provider`, `documentationUrl`, `iconUrl`, `securitySchemes`, `security`/`securityRequirements`, `signatures[]`, `capabilities.extensions[]` (AP2 lives here). v0.3 cards (still the majority deployed) have top-level `url` + `protocolVersion`. No hosted 1.0 JSON schema; the v0.3.0 schema is at `raw.githubusercontent.com/a2aproject/A2A/v0.3.0/specification/json/a2a.json`. — https://github.com/a2aproject/A2A/releases, https://a2a-protocol.org/latest/specification/
36
+ - **MCP 2026-07-28** removed sessions and `initialize`; POSTs carry `MCP-Protocol-Version`, `Mcp-Method`, `Mcp-Name`; `GET /mcp` → 405; new `server/discover`. Discovery: SEP-2127 (open draft) → `experimental-ext-server-card`: `GET <endpoint>/server-card`, required `$schema`, `name` (reverse-DNS), `version`, `description`; optional `title`, `websiteUrl`, `repository`, `icons`, `remotes[]{type ∈ streamable-http|sse, url, supportedProtocolVersions[]}`; CORS `*` and `Cache-Control` recommended. Auth: RFC 9728 `/.well-known/oauth-protected-resource[/mcp]` → `authorization_servers[]` → RFC 8414 or OIDC discovery. — https://modelcontextprotocol.io/specification/2026-07-28/changelog, https://github.com/modelcontextprotocol/modelcontextprotocol/pull/2127, https://github.com/modelcontextprotocol/experimental-ext-server-card
37
+ - **ai-catalog.json / ARD**: Linux Foundation "Agent Card WG" `/.well-known/ai-catalog.json` (`specVersion`, `host{displayName, identifier}`, `entries[]{identifier, type, url|data, displayName}`; entry types `application/mcp-server-card+json`, `application/a2a-agent-card+json`). Agentic Resource Discovery (Google/Microsoft/HF listed) spec v0.91 (2026-08-26) at `/.well-known/ard.json`. Lighthouse's `ard-schema` audit discovers via robots.txt `Agentmap:`, `<link rel="ai-catalog">`, `Link: rel="ai-catalog"`, or the well-known path. — https://ai-catalog.io/, https://agenticresourcediscovery.org/spec/
38
+ - **WebMCP**: W3C WebML CG draft (2026-09-04). Imperative `document.modelContext.registerTool()` (`navigator.modelContext` deprecated). Declarative: `<form toolname tooldescription [toolautosubmit]>`, controls `toolparamdescription`. Chrome origin trial 149→156 (ends ~2026-11-16). Lighthouse audits `forms-missing-declarative-webmcp`. OpenAI enabled WebMCP in the ChatGPT desktop browser (2026-08-25) **(secondary)**. — https://webmachinelearning.github.io/webmcp/, https://developer.chrome.com/docs/ai/webmcp/declarative-api
39
+ - **Agent Skills discovery**: Cloudflare RFC v0.2.0 (2026-03-12) `/.well-known/agent-skills/index.json` (`$schema https://schemas.agentskills.io/discovery/0.2.0/schema.json`, `skills[]{name, type ∈ skill-md|archive, description ≤1024, url, digest "sha256:<64hex>"}`), SKILL.md at `/.well-known/agent-skills/{name}/SKILL.md`. Mintlify/Docus variant `/.well-known/skills/index.json`. SKILL.md frontmatter per agentskills.io: `name` (1–64, `[a-z0-9-]`), `description` (1–1024). — https://github.com/cloudflare/agent-skills-discovery-rfc, https://agentskills.io/specification
40
+ - **UCP** (Google/Shopify/Etsy/Walmart/Stripe): `/.well-known/ucp` (no extension, public), spec 2026-08-25: `ucp.version` (date string), `ucp.services` (reverse-DNS keys → transports rest/mcp/a2a with `schema` URL), `ucp.payment_handlers`, optional `ucp.capabilities`, `keys[]`. Shopify Agentic Storefronts GA March 2026. **ACP** (OpenAI/Stripe) has **no** discovery mechanism; AP2 is an A2A extension URI. — https://developers.google.com/merchant/ucp/guides/ucp-profile, https://developers.openai.com/commerce/specs/checkout
41
+ - **OpenAI Apps**: MCP-based; domain verification file `/.well-known/openai-apps-challenge`. — https://developers.openai.com/plugins/deploy/submission.md
42
+ - **NLWeb**: `/ask` (`query`, `site`, `mode ∈ list|summarize|generate`) and `/mcp`; repo active (2026-08-11). No manifest file.
43
+
44
+ ### 2.2 Content discovery and readability
45
+
46
+ - **llms.txt v2** (llmstxt.org, modified 2026-08-10): subpath files (`/docs/llms.txt`, most specific wins); `<link rel="describedby">` / `Link: rel="describedby"` → the covering llms.txt; per-page mirrors `page.md`, `page.html.md`, `index.html.md`, `index.md`. Google states Search ignores it (ai-optimization-guide, 2026-07-10). Adoption ~10% (SE Ranking, May 2026); Ahrefs (2026-06-15): 97% of published files never fetched, but Claude Code out-fetches every AI search bot. Lighthouse's `llms-txt` audit: 404 → N/A, 5xx → fail, present → fail on missing H1 / too short / no links.
47
+ - **Markdown for Agents**: Cloudflare (docs 2026-07-13) responds to `Accept: text/markdown` with `text/markdown`, `x-markdown-tokens`, `x-original-tokens`, `content-signal` header, `Vary: accept`; strips `ETag`/`Last-Modified`/`Content-Encoding`. Vercel (2026-09-03) negotiates on Accept **and** on agent UA without Accept, keeps `.md` suffix URLs, emits YAML frontmatter (`title`, `canonical_url`, `last_updated`…), `/sitemap.md`, and requires `Vary: Accept`. Senders of `Accept: text/markdown` (Checkly, Feb 2026): Claude Code, Cursor, OpenCode. No IETF standard; only `draft-consolidated-content` (individual). — https://developers.cloudflare.com/fundamentals/reference/markdown-for-agents/, https://vercel.com/docs/agent-resources/markdown-access
48
+ - **API discovery**: RFC 9727 `/.well-known/api-catalog` (`application/linkset+json`, `Link: rel="api-catalog"` on `/`, entries with `service-desc`/`service-doc`/`service-meta`/`status`); RFC 8631 relations. `/.well-known/openapi.json` is **not** IANA-registered. OpenAPI **3.2.0** (2025-09-19) recommends `openapi.json`/`openapi.yaml`; Arazzo 1.1.0 (2026-05-17).
49
+ - **Structured data**: Google: no special markup for AI features, but structured data must match visible text; FAQ rich results ended 2026-05-07; schema.org 30.0 (2026-03-19) adds no AI types. Bing "AI Performance" report (Feb 2026).
50
+ - **IANA well-known registry** (2026-08-19): registered and agent-relevant: `agent-card.json`, `api-catalog`, `oauth-protected-resource`, `oauth-authorization-server`, `tdmrep.json`, `gpc.json`, `security.txt`. **Not** registered: `openapi*`, `mcp*`, `ucp`, `ai.txt`, `llms*`, `agents.json`, `skills`, `ai-catalog.json`. Reports should label each probe as *registered* / *vendor convention* / *draft*.
51
+
52
+ ### 2.3 Usage-rights and access signals
53
+
54
+ - **IETF AIPREF** (`draft-ietf-aipref-vocab-07`, `-attach-05`, 2026-08-19; pre-WGLC, "does not reflect consensus"): tokens `train-ai`, `search`; values `y`/`n`; RFC 9651 dictionary. Carriers: HTTP `Content-Usage: train-ai=n` response header and robots.txt `Content-Usage: [/path ]train-ai=n` inside User-agent groups. Note the inversion vs Content Signals/RSL (`ai-train`). — https://datatracker.ietf.org/wg/aipref/documents/
55
+ - **Content Signals**: new optional 4th field `use=immediate|reference|full` (Cloudflare, 2026-07-01), emitted by managed robots.txt as `Content-signal: search=yes, ai-train=no, use=reference` (lower-case s, spaces after commas). Google publicly states no crawler honors `content-signal` **(secondary: Mueller, 2026-07-06)**. Cloudflare blocks Training + Agent categories by default on ad-bearing pages from **2026-09-15**.
56
+ - **RSL** still 1.0; 2026 errata add `<reporting profile= endpoint=>` (2026-06-12) and require ignoring unknown extension elements (2026-08-07). OLP: `401/402` + `WWW-Authenticate: License` + `Link rel="license"`.
57
+ - **Pay-per-crawl** (closed beta, doc 2026-07-28): `402` + `crawler-price: USD 0.01`; `200` + `crawler-charged`; `402` + `crawler-error`. Always-free paths: `/robots.txt`, `/sitemap.xml`, `/security.txt`, `/.well-known/security.txt`, `/crawlers.json`. AWS WAF x402 monetization (2026-06-15): `402` with `payment-signature`/`payment-response`. Cloudflare AI Crawl Control block = `403` **or** `402` with custom body.
58
+ - **Web Bot Auth**: WG adopted `draft-ietf-webbotauth-httpsig-protocol-00` (2026-09-01, Standards Track). Headers `Signature`, `Signature-Input`, `Signature-Agent`; `tag="web-bot-auth"`; key directory `/.well-known/http-message-signatures-directory` (`application/http-message-signatures-directory+json`, Ed25519 JWKS, response itself signed with `tag="http-message-signatures-directory"`). Origins may answer `403` + `Accept-Signature`. Signers: OpenAI (`https://chatgpt.com`), Google (`https://agent.bot.goog`, experimental), Exa, You.com, Amazon AgentCore. Verifiers: Cloudflare, AWS WAF, Vercel, Akamai.
59
+ - **WAF response signatures**: Cloudflare challenge → `cf-mitigated: challenge` (always `text/html`); Vercel → `x-vercel-mitigated: challenge` **(secondary)**; AWS WAF challenge → **`202`** + `x-amzn-waf-action: challenge`.
60
+ - **Robots meta**: Google `nosnippet` / `max-snippet:N` / `data-nosnippet` limit direct input to AI Overviews and AI Mode; `Google-Extended` governs Gemini training + grounding, **not** AI Overviews. Bing `noarchive` = excluded from Copilot grounding; `nocache` = URL/title/snippet only; new `data-snippet` attribute **(secondary)**. `noai`/`noimageai`: no operator commits to honoring. Google Search Console "Search generative AI control" (2026-08-31) is property-level and not machine-readable.
61
+ - **TDMRep** (W3C CG Final 2024-05-10, cited in the EU GPAI Code of Practice): `tdm-reservation: 0|1` / `tdm-policy` headers, `/.well-known/tdmrep.json` (`[{location, tdm-reservation, tdm-policy}]`), `<meta name="tdm-reservation">`; precedence meta > header > file.
62
+
63
+ ### 2.4 Crawler landscape (Cloudflare Radar, Aug 2026)
64
+
65
+ Googlebot 27%, Meta-ExternalAgent 12.7%, ClaudeBot 11.9%, Bingbot 8.5%, GPTBot 8.3%, Applebot 6.7%, Amazonbot 6.1%, Bytespider 4.7%, Claude-SearchBot 3.6%. Cloudflare's taxonomy is Search / Agent / Training. Only `Google-Agent` ships a documented UA for agentic browsing; ChatGPT agent, Claude in Chrome, Comet and Copilot use plain Chrome UAs and identify (if at all) through Web Bot Auth.
66
+
67
+ Vendor changes: OpenAI `ChatGPT-User` "robots.txt may not apply" (Dec 2025), new `OAI-AdsBot` (Apr 2026); Anthropic three-bot split documented (Feb 2026), IPs at `claude.com/crawling/bots.json`; Google `Google-Agent` (2026-03-20), `Google-NotebookLM` → `Google-GeminiNotebook` (2026-07-17), Mariner retired (2026-05-04); Meta `meta-webindexer` (search index, ~38% of tracked AI requests on peak days **(secondary)**); Amazon `Amzn-SearchBot`, `Amzn-User`, robots-only management from 2026-06-15; Mistral `MistralAI-Index`, `MistralAI-Training`; Apple's Applebot doc now states training use (2026-06-08); Cohere runs no crawlers; Exa `ExaSearchBot` signs every request.
68
+
69
+ ---
70
+
71
+ ## 3. Gap analysis against the current 18 checks
72
+
73
+ | Check | Status | Required change |
74
+ | --- | --- | --- |
75
+ | `llms-txt` | Stale (spec v2) | Add `rel="describedby"` detection (HTML + `Link`), subpath discovery, `.md` mirror probe, Lighthouse-aligned rules, link-liveness sampling, size heuristics. Reword copy: value is developer-agent tooling, not Google Search. |
76
+ | `robots-txt` | Stale list + partial Content Signals | Refresh `AI_CRAWLERS` (§4.1), tiered reporting (blocking search bots ≠ blocking training bots), `use=` field, case/space tolerance, Cloudflare-managed block detection, `Content-Usage` parsing, `Agentmap:` directive, `Google-Extended` semantics in hints. |
77
+ | `agent-json` | **Wrong path** | Probe `agent-card.json` first, `agent.json` as legacy fallback with warning. Detect card generation (1.0 vs 0.3) and validate accordingly. Drop `0.2.0` hint, flag `authentication`. |
78
+ | `mcp` | **Wrong convention** | Replace with server-card discovery chain (§4.4). Keep `mcp.json` as legacy fallback, never as a recommendation. |
79
+ | `openapi` | Folk path only | Become `api-discovery`: RFC 9727 first, then conventional paths, `Link`/`<link>` `service-desc`, OpenAPI 3.0–3.2. |
80
+ | `http-headers` | Narrow Link parsing, stale agent.json reference | Broaden Link relations (`describedby`, `api-catalog`, `ai-catalog`, `service-desc`, `service-doc`, `alternate text/markdown`), fix agent-card path, recognise `X-Llms-Txt`. |
81
+ | `well-known-ai` | **Mostly fictional bundle** | Freeze scoring in 3.x, add informational findings for real files; replace in 4.0 with `ai-catalog` + `agent-skills` and drop nlweb/genai/ai-plugin. |
82
+ | `meta-tags` | `ai:*` namespace has no known consumer | Keep, but demote in 4.0 weights; add `article:published_time`/`modified_time` reading for freshness. |
83
+ | `structured-data` | Type-presence only | Add `dateModified`/`datePublished`, `author` with `sameAs`, `Organization.sameAs`; consistency of `headline`/`name`/`description` vs visible text; neutralise FAQPage/HowTo. |
84
+ | `content-negotiation` | Good, extend | Realistic Accept string, UA-only probe (Vercel mode), `x-markdown-tokens`/`x-original-tokens`, `content-signal` header, `.md` suffix, frontmatter parse, `/sitemap.md`, `Link: rel="canonical"` on markdown. |
85
+ | `agent-access` | Good, extend | Classify responses (§4.9): challenge vs hard block vs 402 paywall vs `Accept-Signature` vs RSL OLP; separate "crawlers that honor robots" from "user fetchers that don't"; compare title/H1/JSON-LD hash, not only text length. |
86
+ | `crawl-efficiency` | Good | Add token estimate of extracted text, response time, `Retry-After` on any 429 seen. |
87
+ | `rsl` | Current | Accept `<reporting>`; ignore unknown extension elements; detect OLP responses. |
88
+ | `sitemap`, `seo-basics`, `tls-https`, `security-txt`, `html-rendering` | Current | Minor: `<html lang>` vs `Content-Language` vs `inLanguage` consistency; feeds `<link rel="alternate" type=rss/atom>`; `<img>`/`<iframe>` missing dimensions (CLS proxy). |
89
+
90
+ ---
91
+
92
+ ## 4. Specification of changes and new checks
93
+
94
+ Each new check follows the anatomy in [architecture.md](./architecture.md): `weight: 0` in 3.x, every warn/fail with `hint` + `learnMoreUrl`, network via `ctx.fetch`, regex primitives (no parser deps).
95
+
96
+ ### 4.1 `constants.ts` — crawler catalogue refresh
97
+
98
+ Replace the three buckets with four plus a legacy alias list. Matching stays case-insensitive (RFC 9309).
99
+
100
+ ```ts
101
+ export const AI_CRAWLERS = {
102
+ training: ['GPTBot', 'ClaudeBot', 'Meta-ExternalAgent', 'Google-Extended', 'Applebot-Extended',
103
+ 'Amazonbot', 'CCBot', 'Bytespider', 'TikTokSpider', 'MistralAI-Training', 'AI2Bot', 'Ai2Bot-Dolma',
104
+ 'DeepSeekBot', 'PanguBot', 'Google-CloudVertexBot', 'FacebookBot', 'Timpibot', 'Webzio-Extended',
105
+ 'omgili', 'omgilibot', 'ImagesiftBot', 'Kangaroo Bot', 'Diffbot', 'YandexAdditional', 'YandexAdditionalBot'],
106
+ search: ['OAI-SearchBot', 'Claude-SearchBot', 'PerplexityBot', 'meta-webindexer', 'Amzn-SearchBot',
107
+ 'MistralAI-Index', 'Applebot', 'bingbot', 'DuckAssistBot', 'YouBot', 'Kagibot', 'PetalBot',
108
+ 'ExaSearchBot', 'PhindBot', 'Yeti'],
109
+ userFetch: ['ChatGPT-User', 'Claude-User', 'Perplexity-User', 'MistralAI-User', 'meta-externalfetcher',
110
+ 'Amzn-User', 'Google-GeminiNotebook', 'kagi-fetcher', 'Kimi-User', 'TongyiBot'],
111
+ agentBrowsing: ['Google-Agent', 'NovaAct', 'Manus-User', 'Devin', 'FirecrawlAgent', 'TavilyBot'],
112
+ };
113
+ export const LEGACY_AI_CRAWLERS = ['Claude-Web', 'Anthropic-AI', 'Google-NotebookLM', 'GoogleAgent-Mariner',
114
+ 'cohere-ai', 'cohere-training-data-crawler', 'ExaBot', 'NeevaBot', 'Gemini', 'GeminiBot', 'DeepSeek-AI'];
115
+ export const CORE_AI_CRAWLERS = ['GPTBot', 'ClaudeBot', 'Meta-ExternalAgent', 'Google-Extended',
116
+ 'Applebot-Extended', 'Amazonbot', 'Bytespider', 'CCBot', 'OAI-SearchBot', 'Claude-SearchBot',
117
+ 'PerplexityBot', 'ChatGPT-User'];
118
+ ```
119
+
120
+ Add a `CRAWLER_META` map (`purpose`, `honorsRobots: boolean`, `docUrl`, `ipListUrl`) so findings can say *why* a block matters ("blocking OAI-SearchBot removes you from ChatGPT search answers; blocking GPTBot only affects training"). Bots documented as ignoring robots.txt (`Perplexity-User`, `Google-Agent`, `ChatGPT-User` partially) must not be scored as "missing" in robots.txt.
121
+
122
+ **Scoring impact in 3.x:** `robots-txt` deducts by missing core crawlers. Growing CORE from 8 to 12 would lower scores → keep the 8-token `CORE_AI_CRAWLERS_V3` for scoring until 4.0 and use the 12-token list for informational findings only.
123
+
124
+ ### 4.2 Shared infrastructure (prerequisites)
125
+
126
+ 1. **`src/checks/robots-parser.ts`** — extract `parseUserAgents`, `parseContentSignalDecls`, `parseRobotsLicenseDirectives` into one module, add `Content-Usage` (path-scoped, per group), `Agentmap:`, `Sitemap:`, Cloudflare managed-block markers. `robots-txt`, `rsl`, `agent-access`, `usage-policy`, `ai-catalog` all consume it; robots.txt is fetched once via the fetcher cache.
127
+ 2. **`src/checks/waf.ts`** — `classifyResponse(res): 'ok' | 'challenge-cloudflare' | 'challenge-vercel' | 'challenge-aws' | 'blocked' | 'paywall-ppc' | 'paywall-x402' | 'needs-signature' | 'license-required'` from status + headers (`cf-mitigated`, `x-vercel-mitigated`, `x-amzn-waf-action`, `crawler-price`, `payment-signature`, `Accept-Signature`, `WWW-Authenticate: License`). Used by `agent-access`, `http-hygiene`, `content-negotiation`.
128
+ 3. **`src/checks/structured-fields.ts`** — minimal RFC 9651 dictionary parser (`key=token`, `key=?1`, params) for `Content-Usage`, `Signature-Input`, `Content-Signal` header.
129
+ 4. **`src/checks/frontmatter.ts`** — flat `key: value` YAML frontmatter reader for SKILL.md and `.md` mirrors (no nested YAML).
130
+ 5. **`src/checks/tokens.ts`** — `estimateTokens(text)` (chars/4 heuristic, documented as approximate).
131
+ 6. **Fetcher**: add `method?: 'GET' | 'HEAD'` and `redirect?: 'follow' | 'manual'` to `FetchOptions`; expose `redirectCount` and `elapsedMs` on `FetchResponse`. HEAD is needed for llms.txt link sampling; manual redirects for hop counting and `.md` mirror checks.
132
+ 7. **Types**: `CheckResult.applicable?: boolean` (default `true`). Scorer excludes `applicable === false` from the denominator (4.0 behaviour; in 3.x reporters just display "N/A"). `CheckMeta.category: 'discovery' | 'content' | 'access' | 'policy' | 'protocols' | 'trust'` for grouped reports.
133
+ 8. **Well-known probe helper**: `probeWellKnown(ctx, paths[], { accept })` returning first hit with `registered: boolean` label from a small `WELL_KNOWN_REGISTRY` constant.
134
+
135
+ ### 4.3 `agent-card` (rename of `agent-json`, same id kept as alias in 3.x)
136
+
137
+ - Probe order: `/.well-known/agent-card.json` → `/.well-known/agent.json` (warn "legacy 0.2.x path; A2A ≥0.3 uses agent-card.json").
138
+ - Content-Type: accept `application/json` or `application/a2a+json`.
139
+ - Generation detection: `supportedInterfaces[]` → 1.0 rules; top-level `url` + `protocolVersion` → 0.3 rules; neither → fail "unrecognised card shape".
140
+ - 1.0 required: `name, description, version, capabilities, supportedInterfaces, defaultInputModes, defaultOutputModes, skills` (−15 each, as today). Each interface needs `url`, `protocolBinding`, `protocolVersion`.
141
+ - 0.3 required: `name, description, url, version, protocolVersion, capabilities, defaultInputModes, defaultOutputModes, skills`.
142
+ - `authentication` present → warn "removed in 0.2.x, use securitySchemes" (−5). `skills[]` entries need `id`, `name`, `description`, `tags`.
143
+ - Optional credit: `provider`, `documentationUrl`, `iconUrl`, `securitySchemes`, `signatures[]`, `capabilities.extensions[]` (report AP2/other extension URIs).
144
+ - Same-origin check on interface URLs (warn only).
145
+ - 3.x scoring: unchanged formulas; the new path can only raise scores.
146
+
147
+ ### 4.4 `mcp-discovery` (rename of `mcp`)
148
+
149
+ Discovery chain, first hit wins, each labelled with its standing:
150
+
151
+ 1. `/.well-known/ai-catalog.json` entries of type `application/mcp-server-card+json` (draft, LF).
152
+ 2. `/.well-known/mcp/server-card.json`, `/.well-known/mcp/server-cards.json` (Cloudflare / Mintlify convention).
153
+ 3. `/mcp/server-card` (experimental-ext-server-card recommendation) — also `<remote-url>/server-card` for any remote found in step 1–2.
154
+ 4. `/.well-known/mcp.json` (legacy, ax-audit's own past recommendation) → warn.
155
+
156
+ Validation of a server card: `$schema`, `name` (reverse-DNS `^[a-z0-9.-]+/[a-z0-9._-]+$` or dotted), `version`, `description` (−15 each); `remotes[]` with `type ∈ streamable-http|sse`, absolute `url`, `supportedProtocolVersions[]` containing a known version (`2026-07-28`, `2025-11-25`, `2025-06-18`, `2025-03-26`, `2024-11-05`; warn if only pre-2025 versions) (−10); `Content-Type: application/mcp-server-card+json` or `application/json` (−5); CORS `*` (−5); `Cache-Control` present (info). Do **not** expect `tools[]` — hint that tool lists come from `tools/list` at runtime.
157
+
158
+ Optional live probe (flag `--probe-mcp`, off by default): `GET <remote>` expecting `405` (2026-07-28 signal) or `POST` `server/discover` with `MCP-Protocol-Version`; any JSON-RPC response or a protocol-version error counts as "reachable".
159
+
160
+ Auth chain (informational, feeds `auth-discovery`): if any remote exists, probe `/.well-known/oauth-protected-resource` and `/.well-known/oauth-protected-resource<remote-path>`.
161
+
162
+ ### 4.5 `api-discovery` (rename of `openapi`)
163
+
164
+ 1. `HEAD /` (or reuse homepage headers) → `Link: rel="api-catalog"`; `GET /.well-known/api-catalog` with `Accept: application/linkset+json`. Validate `linkset[]`, each with `anchor` and at least one of `service-desc` / `service-doc`; resolve one `service-desc` and confirm it parses as OpenAPI/AsyncAPI/Arazzo.
165
+ 2. `<link rel="service-desc">`, `<link rel="service-doc">` in HTML and `Link` header (RFC 8631).
166
+ 3. Conventional paths: `/openapi.json`, `/openapi.yaml`, `/.well-known/openapi.json`, `/.well-known/openapi.yaml`, `/swagger.json`, `/api-docs`, `/v1/openapi.json`, `/arazzo.json`, `/asyncapi.json`.
167
+ 4. OpenAPI validation as today plus: version `3.0`–`3.2` (note `$self` in 3.2), `operationId` coverage (warn <100%), `servers[]`, `info.description`, `securitySchemes` presence, documented rate limits (`x-ratelimit*` extensions or `429` responses) (info).
168
+
169
+ Label: RFC 9727 = registered; folk paths = convention. Scoring in 3.x unchanged for sites that only have `/.well-known/openapi.json`.
170
+
171
+ ### 4.6 `ai-catalog` (new, replaces the discovery half of `well-known-ai`)
172
+
173
+ Discovery per Lighthouse: robots.txt `Agentmap:` → `<link rel="ai-catalog">`/`<link rel="ard">` → `Link: rel="ai-catalog"` → `/.well-known/ai-catalog.json` → `/.well-known/ard.json`. Validate `specVersion`, `host.identifier`, `entries[]` each with `identifier`, `type`, `displayName`, and `url` xor `data`; resolve each `url` (HEAD) and check its `Content-Type` matches `type`. Cross-reference: an `application/a2a-agent-card+json` entry should point where `agent-card` found the card; an MCP entry should match `mcp-discovery`.
174
+
175
+ ### 4.7 `agent-skills` (new)
176
+
177
+ Probe `/.well-known/agent-skills/index.json` (canonical RFC) then `/.well-known/skills/index.json` (Mintlify variant), then `/skill.md`. Validate index: `skills[]` non-empty; each `name` matches `^[a-z0-9-]{1,64}$`, `description` 1–1024, `type ∈ skill-md|archive`, `url` absolute or root-relative, `digest` matches `^sha256:[0-9a-f]{64}$` (warn if missing). Fetch up to 3 SKILL.md files: `text/markdown` Content-Type, frontmatter `name` equals index name and directory, `description` present, body ≤500 lines (info). Not applicable (`applicable: false`) when nothing is found **and** the site shows no docs signals (no `/docs` link, no llms.txt).
178
+
179
+ ### 4.8 `ai-directives` (new) — page-level AI controls
180
+
181
+ Parse `<meta name="robots|googlebot|bingbot">` and `X-Robots-Tag` (all values, comma-split, UA-prefixed forms). Report with vendor semantics:
182
+
183
+ | Directive | Finding |
184
+ | --- | --- |
185
+ | `noindex` on homepage | fail: invisible to every search-grounded assistant |
186
+ | `nosnippet` or `max-snippet:0` | warn: excluded as direct input to Google AI Overviews / AI Mode |
187
+ | `max-snippet:N` with N < 160 | info |
188
+ | `noarchive` | warn: excluded from Bing Copilot grounding |
189
+ | `nocache` | info: Copilot may use URL/title/snippet only |
190
+ | `noai`, `noimageai` | info: declared preference, no operator commits to honoring |
191
+ | `data-nosnippet` wrapping `<main>`, `<article>` or the H1 block | warn |
192
+ | `Content-Signal` says `ai-input=no` while page is not `nosnippet` | info (inconsistent intent) |
193
+ | Hint when robots.txt disallows `Google-Extended` | explain it does not remove the site from AI Overviews |
194
+
195
+ Score (4.0 proposal): start 100; `noindex` → 0; `nosnippet`/`noarchive` −30 each; contradictions −10. Weight 0 in 3.x.
196
+
197
+ ### 4.9 `agent-access` — response classification (informational check, free to change in 3.x)
198
+
199
+ Per probe, run `classifyResponse` and map to outcomes:
200
+
201
+ | Classification | Credit | Message |
202
+ | --- | --- | --- |
203
+ | `ok` | 1 | equivalent response |
204
+ | `blocked` consistent with robots intent | 1 | intentional |
205
+ | `challenge-*` | 0.5, tagged **inconclusive** | JS challenge served; real crawler may pass via IP/Web Bot Auth verification, but fetch-only agents cannot |
206
+ | `needs-signature` (`403` + `Accept-Signature`) | 0.75 | site requires Web Bot Auth; unsigned agents excluded |
207
+ | `paywall-ppc` / `paywall-x402` | 1, info | monetised for crawlers; report price header |
208
+ | `license-required` | 1, info | RSL OLP in effect |
209
+ | `blocked` while robots allows | 0 | as today |
210
+ | reduced content | 0.5 | as today, plus title/H1/JSON-LD hash comparison |
211
+
212
+ Probe set: the 12 CORE tokens split into "honors robots.txt" (scored) and "user-triggered fetchers" (reported, not scored). Also probe the always-free paths (`/robots.txt`, `/sitemap.xml`, `/llms.txt`) with the worst-treated UA and flag if they are blocked.
213
+
214
+ ### 4.10 `usage-policy` (new) — machine-readable rights signals and their consistency
215
+
216
+ Collect: robots.txt `Content-Signal` (incl. `use=`), robots.txt `Content-Usage`, HTTP `Content-Usage`, HTTP `content-signal`, RSL permits/prohibits (from `rsl`), TDMRep (`/.well-known/tdmrep.json`, `tdm-reservation`/`tdm-policy` headers, meta), `noai` meta. Validate syntax per spec (AIPREF dictionary `y/n`; warn on `yes/no` or `ai-train` under `Content-Usage`; Content Signals `yes/no` and `use ∈ immediate|reference|full`; TDMRep JSON array with `location` and `tdm-reservation ∈ 0|1`). Then compute a **consistency matrix** and flag contradictions: `ai-train=no` vs `Content-Usage: train-ai=y`; robots `Disallow` for all training bots vs `ai-train=yes`; RSL prohibits `ai-train` vs Content-Signal `ai-train=yes`; TDMRep reservation 1 with no other training signal (info). Copy must state that only robots.txt tokens are documented as honored by Google, OpenAI, Anthropic and Microsoft; the others are declarations with legal (EU AI Act) rather than technical weight.
217
+
218
+ ### 4.11 `http-hygiene` (new) — status honesty and fetch ergonomics
219
+
220
+ - Soft-404 probe: `GET /ax-audit-probe-<random>` must return `404`/`410` (warn on `200`, `302→/`, or `403`).
221
+ - Redirect hops on the homepage ≤1 (`redirect: 'manual'` follow-up); `http://` → `https://` counted separately (already in `tls-https`).
222
+ - Any `429` seen during the audit must carry `Retry-After`.
223
+ - Homepage `Content-Type` includes charset; `Content-Language` consistent with `<html lang>`.
224
+ - `HEAD /` supported (not `405`).
225
+ - Error body sanity: a `404` body should not be an empty 0-byte response.
226
+
227
+ ### 4.12 `webmcp` (new, static)
228
+
229
+ - Count `<form>`; count forms with both `toolname` and `tooldescription`; fail-level finding for `toolname` without `tooldescription` (mirrors Lighthouse schema validity); per tool form, coverage of `toolparamdescription` on named controls; note `toolautosubmit`.
230
+ - Inline/linked script text: `modelContext.registerTool` present → info "imperative WebMCP detected (cannot validate statically)"; `navigator.modelContext` → warn deprecated namespace.
231
+ - `<meta http-equiv="origin-trial">` presence → info.
232
+ - `applicable: false` when the page has zero forms and no script match. Weight stays 0 through 4.0 (origin trial only).
233
+
234
+ ### 4.13 `agent-operability` (new, static heuristics for browser agents)
235
+
236
+ Based on web.dev "AI agent site UX" (2026-04-01), Atlas/Claude browser-tool documentation and Lighthouse's agent accessibility subset:
237
+
238
+ - `<a>` without `href` or with `href="javascript:"`; `<div>`/`<span>` with `onclick` lacking `role` and `tabindex`.
239
+ - `<button>`/`[role=button]` without accessible name (text, `aria-label`, `aria-labelledby`, `title`, `<img alt>`).
240
+ - Form controls without `<label for>`, wrapping label, `aria-label` or `aria-labelledby`; missing `autocomplete` on common fields (info).
241
+ - `<iframe>` without `title`; `<table>` without `<th>`; `<time>` without `datetime`; heading-level skips.
242
+ - `<img>`/`<iframe>`/`<video>` without dimensions (CLS proxy).
243
+ - Entry-page blockers: reCAPTCHA/hCaptcha/Turnstile markup, `<meta http-equiv="refresh">`, cookie-consent frameworks when `<main>` text < 100 words.
244
+
245
+ Score proposal: proportional to the share of interactive elements that pass; informational tag "heuristic, static HTML only".
246
+
247
+ ### 4.14 Smaller extensions
248
+
249
+ - **`llms-txt`**: `describedby` links, subpath llms.txt when auditing a non-root URL, `.md` mirror of the homepage (`/index.md`, `/index.html.md`), HEAD-sample ≤20 links (report unreachable / redirected / disallowed-by-robots / `noindex` targets), duplicates, `## Optional` present, size warnings (>50 KB llms.txt, >1 MB llms-full.txt) as info, `/.well-known/llms.txt` mirror as info.
250
+ - **`content-negotiation`**: Accept `text/markdown, text/html;q=0.9, */*;q=0.1`; second probe with `User-Agent: Claude-Code/1.0` and default Accept; record `x-markdown-tokens`/`x-original-tokens` and compute savings; `content-signal` header capture; `.md` suffix probe; frontmatter `title`/`canonical_url`; `Link: rel="canonical"` on the markdown response; `/sitemap.md`.
251
+ - **`structured-data`**: `dateModified`/`datePublished` (bucketed freshness, future dates flagged), `author` → Person/Organization with `url`/`sameAs`, `Organization.sameAs` count, `headline`/`name` present in visible text, `isAccessibleForFree` vs body length sanity.
252
+ - **`http-headers`**: relations `describedby`, `api-catalog`, `ai-catalog`, `service-desc`, `service-doc`, `alternate` + `type="text/markdown"`, `c2pa-manifest`; `X-Llms-Txt`; fix agent-card path in hints; `Accept-Signature` and `Signature-Agent` presence (info).
253
+ - **`crawl-efficiency`**: `elapsedMs` TTFB proxy (warn >2 s), estimated tokens of extracted homepage text (warn >25k), `alt-svc` h3 (info).
254
+ - **`well-known-ai` (3.x only)**: keep the 5-file score frozen; rewrite hints (nlweb.json/genai.txt marked "no spec found", ai-plugin.json "retired 2024"); add informational probes for `/.well-known/http-message-signatures-directory` (validate JWKS shape if present), `/.well-known/openai-apps-challenge`, `/.well-known/tdmrep.json`, `/.well-known/gpc.json`, `/AGENTS.md`, `/ai.txt`. Retired in 4.0.
255
+ - **`commerce-discovery`** (new, conditional): applicable only when `Product`/`Offer` JSON-LD or `/cart|/checkout` links exist. `GET /.well-known/ucp` (fallback `/.well-known/ucp.json`): JSON, `ucp.version` date, `ucp.services` non-empty with resolvable `schema` URLs, `ucp.payment_handlers`, `keys[]`; public (no auth) required. Info only for ACP (`OPTIONS /checkout_sessions`).
256
+ - **`auth-discovery`** (new, conditional): applicable when `api-discovery`, `mcp-discovery` or `commerce-discovery` found something. Probe RFC 9728 `/.well-known/oauth-protected-resource`, RFC 8414 `/.well-known/oauth-authorization-server`, `/.well-known/openid-configuration`; require `authorization_servers[]` or `issuer` + `authorization_endpoint` + `token_endpoint`; `code_challenge_methods_supported` includes `S256`; `registration_endpoint` or `client_id_metadata_document_supported` (info).
257
+
258
+ Deferred (needs a headless browser or an LLM, out of scope for this package): rendered-vs-static diff, imperative WebMCP enumeration, CLS/INP, task-based agent runs (Ora/Netlify AXIS style), citation share-of-voice.
259
+
260
+ ---
261
+
262
+ ## 5. Release plan
263
+
264
+ ### 3.7.0 — "Correct the map" (no score can go down) — **shipped 2026-09-04**
265
+
266
+ 1. Shared infra: `robots-parser.ts`, `waf.ts`, fetcher `method`/`redirect`/`elapsedMs`, `CheckMeta.category`, `WELL_KNOWN_REGISTRY`.
267
+ 2. Crawler catalogue refresh (§4.1) with `CORE_AI_CRAWLERS_V3` frozen for scoring.
268
+ 3. `agent-card` probe chain (§4.3) and `mcp-discovery` chain (§4.4) with legacy fallbacks; ids `agent-json`/`mcp` kept as aliases for `--checks` and baselines.
269
+ 4. `api-discovery` probe chain (§4.5); id `openapi` aliased.
270
+ 5. `agent-access` classification (§4.9).
271
+ 6. `http-headers` relation broadening and hint fixes; `well-known-ai` hint rewrite + informational probes.
272
+ 7. `robots-txt`: `use=` field, case/space tolerance, tiered informational findings, `Content-Usage` and `Agentmap:` parsing (informational).
273
+ 8. Docs: checks.md, README table, CHANGELOG; remediation guides for every new anchor.
274
+ 9. Tests: 252 new (553 total).
275
+
276
+ Delivered in eleven commits. Two things were found by dogfooding rather than by planning: `agent-access` was probing sites with a user agent containing `Google-Extended`, a robots.txt control token no request ever carries; and a speculative probe against a single-page application returned the index shell, which the check reported as a malformed document. Both are fixed and regression-tested.
277
+
278
+ ### 3.8.0 — "New signals" (all weight 0) — **shipped**
279
+
280
+ `ai-directives`, `usage-policy`, `http-hygiene`, `ai-catalog`, `agent-skills`, `webmcp`, `commerce-discovery`, `auth-discovery` (with `applicable` flag displayed as N/A). `structured-fields.ts`, `frontmatter.ts`, `tokens.ts`. Reporters render categories and N/A. ~90 tests.
281
+
282
+ ### 3.9.0 — "Readability depth" — **shipped**
283
+
284
+ `agent-operability`; llms.txt v2 extensions and link sampling; content-negotiation extensions; structured-data freshness/author; crawl-efficiency tokens/TTFB; `--probe-mcp` flag. ~50 tests.
285
+
286
+ ### 4.0.0 — "Rescore" — **shipped**
287
+
288
+ - Weights redistributed (proposal below), `applicable: false` excluded from denominators, categories in every reporter, baseline files carry `schemaVersion: 2` with a migration path for 3.x baselines (missing checks are ignored, not regressions).
289
+ - Retire `well-known-ai`; ids `agent-json`, `mcp`, `openapi` removed (aliases dropped); `CORE_AI_CRAWLERS` = 12 tokens.
290
+ - llms.txt and `ai:*` meta demoted; access/policy and rendering promoted, following the evidence that fetch-only agents fail on JS-only content and WAF blocks far more often than on missing manifest files.
291
+ - New CLI: `--profile docs|api|commerce|default` (sets which conditional checks are forced applicable), `--category` filter, `--fail-on-category access:70`.
292
+
293
+ Proposed 4.0 weights (sum 100):
294
+
295
+ | Category | Subtotal | Checks and weights |
296
+ | --- | --- | --- |
297
+ | Content | 33 | html-rendering 10 · structured-data 7 · seo-basics 6 · content-negotiation 6 · sitemap 4 |
298
+ | Discovery | 25 | robots-txt 10 · llms-txt 7 · http-headers 5 · meta-tags 3 |
299
+ | Access | 23 | agent-access 8 · ai-directives 5 · http-hygiene 4 · crawl-efficiency 3 · tls-https 3 |
300
+ | Policy & trust | 9 | usage-policy 4 · security-txt 3 · rsl 2 |
301
+ | Protocols (conditional, N/A excluded) | 10 | agent-operability 3 · api-discovery 2 · agent-card 2 · mcp-discovery 2 · agent-skills 1 |
302
+ | Informational | 0 | ai-catalog · webmcp · commerce-discovery · auth-discovery |
303
+
304
+ ---
305
+
306
+ ## 5a. What actually shipped
307
+
308
+ | Release | Commits | Tests | Headline |
309
+ | --- | --- | --- | --- |
310
+ | 3.7.0 | 11 | 553 | Corrected three checks probing paths that were never the standard, and a crawler catalogue containing tokens that do not exist |
311
+ | 3.8.0 | 8 | 755 | Eight new checks, N/A reporting, category grouping |
312
+ | 3.9.0 | 4 | 828 | agent-operability, llms.txt v2, provenance and freshness, cost in tokens |
313
+ | 4.0.0 | 7 | 873 | Rescore, conditional protocol checks, profiles, per-area gates, versioned baselines |
314
+
315
+ Five things were found by running the tool rather than by planning it, and none would have surfaced any other way:
316
+
317
+ 1. `agent-access` probed sites with a user agent containing `Google-Extended` — a robots.txt control token no request ever carries, so the probe tested nothing.
318
+ 2. A speculative probe against a single-page application returned the index shell, and the check reported a malformed document on a site that had none.
319
+ 3. `commerce-discovery` told a SaaS site to build a commerce integration because it prices its plans with an `Offer`.
320
+ 4. The first draft of the 4.0 rescore produced a distribution nobody asked for, because per-check `meta.weight` values silently won over the central map.
321
+ 5. `structured-data` scored a well-marked-up site at 0 because its pattern required `type` to be the first attribute on the script tag. Next.js emits `id` first, so a whole class of sites was being told to add markup they already had.
322
+
323
+ Two items from the plan were **not** built, deliberately:
324
+
325
+ - **`--probe-mcp`** live handshake. A POST to a stranger's MCP endpoint is a side effect an audit should not have by default, and behind a flag it would be exercised too rarely to stay correct.
326
+ - **Dropping the renamed check ids.** The plan called for removing `agent-json`, `mcp` and `openapi` at 4.0. They cost one line each and live in CI configs, so they stay as permanent aliases.
327
+
328
+ ## 6. Cross-repo follow-ups
329
+
330
+ - **ax-init** generates `/.well-known/agent.json` and `/.well-known/mcp.json`: switch to `agent-card.json` (A2A 1.0 shape) and a server card at `/.well-known/mcp/server-card.json` plus `/.well-known/ai-catalog.json`; add `Content-Signal` with `use=`, `Link` headers (`describedby`, `api-catalog`), `.md` mirrors.
331
+ - **ax-skill** (`ax.md`) documents the old paths and the `ai:*` meta namespace as recommendations; update to the 3.7 reality and add the robots-meta AI directive semantics.
332
+ - **Remediation guides** at `axrush.com/guides/<check-id>`: one page per new check id, plus new anchors listed in each check's source. This is the largest non-code deliverable; ship guides with each minor release.
333
+
334
+ ---
335
+
336
+ ## 7. Risks and guardrails
337
+
338
+ - **Draft standards churn** (MCP server card path, ARD vs ai-catalog, WebMCP, AIPREF tokens). Guardrail: label every finding with standing (registered / convention / draft), keep these checks at weight 0 through 4.0, and centralise paths in constants so a rename is a one-line change.
339
+ - **False positives from spoofed-UA probes**: WAFs with IP or Web Bot Auth verification reject ax-audit while admitting the real bot. Guardrail: `challenge` and `needs-signature` outcomes are reported as inconclusive with the exact header seen, never as "blocks AI".
340
+ - **Penalising features a site does not offer** (APIs, commerce, skills). Guardrail: `applicable: false` + profiles; N/A never lowers the score.
341
+ - **Over-weighting files nobody fetches** (llms.txt, manifests). Guardrail: 4.0 weights favour rendering, access and policy; copy states which consumers actually read each file.
342
+ - **Heuristic checks** (`agent-operability`, extractability): informational, thresholds in constants, documented as static approximations.
343
+ - **Baseline compatibility**: renamed ids must map old → new in `diffBaseline` so `--fail-on-regression` does not fire on a rename.
344
+
345
+ ---
346
+
347
+ ## 8. Evidence index (primary sources checked 2026-09-04)
348
+
349
+ A2A releases and spec · https://github.com/a2aproject/A2A/releases · https://a2a-protocol.org/latest/specification/
350
+ MCP 2026-07-28 changelog and authorization · https://modelcontextprotocol.io/specification/2026-07-28/changelog · https://modelcontextprotocol.io/specification/2026-07-28/basic/authorization/authorization-server-discovery
351
+ MCP server card SEP-2127 and extension repo · https://github.com/modelcontextprotocol/modelcontextprotocol/pull/2127 · https://github.com/modelcontextprotocol/experimental-ext-server-card
352
+ ai-catalog / ARD · https://ai-catalog.io/ · https://agenticresourcediscovery.org/spec/
353
+ WebMCP · https://webmachinelearning.github.io/webmcp/ · https://developer.chrome.com/docs/ai/webmcp/declarative-api · https://developer.chrome.com/docs/lighthouse/agentic-browsing
354
+ Agent Skills · https://github.com/cloudflare/agent-skills-discovery-rfc · https://agentskills.io/specification · https://www.mintlify.com/docs/ai/skillmd
355
+ UCP / ACP · https://developers.google.com/merchant/ucp/guides/ucp-profile · https://ucp.dev/2026-08-25/specification/overview/ · https://developers.openai.com/commerce/specs/checkout
356
+ llms.txt v2 · https://llmstxt.org/ · Google AI optimization guide https://developers.google.com/search/docs/fundamentals/ai-optimization-guide
357
+ Markdown for Agents · https://developers.cloudflare.com/fundamentals/reference/markdown-for-agents/ · https://vercel.com/docs/agent-resources/markdown-access · https://vercel.com/kb/guide/agent-readability-spec
358
+ RFC 9727 / 8631 / 9728 / 8414 · https://www.rfc-editor.org/rfc/rfc9727.html · https://www.rfc-editor.org/rfc/rfc8631.html
359
+ IANA well-known registry · https://www.iana.org/assignments/well-known-uris/well-known-uris.xhtml
360
+ AIPREF · https://datatracker.ietf.org/wg/aipref/documents/ · Web Bot Auth · https://datatracker.ietf.org/group/webbotauth/documents/ · https://developers.cloudflare.com/bots/reference/bot-verification/web-bot-auth/
361
+ Content Signals `use=` · https://blog.cloudflare.com/content-independence-day-ai-options/ · https://developers.cloudflare.com/bots/additional-configurations/managed-robots-txt/
362
+ Pay-per-crawl · https://developers.cloudflare.com/ai-crawl-control/features/pay-per-crawl/ · AWS WAF monetization https://docs.aws.amazon.com/waf/latest/developerguide/waf-ai-traffic-monetization-how-it-works.html
363
+ WAF signatures · https://developers.cloudflare.com/cloudflare-challenges/challenge-types/challenge-pages/detect-response/ · https://docs.aws.amazon.com/waf/latest/developerguide/waf-captcha-and-challenge-actions.html
364
+ RSL errata · https://rslstandard.org/rsl/errata · TDMRep · https://www.w3.org/community/reports/tdmrep/CG-FINAL-tdmrep-20240510/
365
+ Robots meta semantics · https://developers.google.com/search/docs/crawling-indexing/robots-meta-tag · https://developers.google.com/search/docs/appearance/ai-features · https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat
366
+ Crawler docs · https://developers.openai.com/api/docs/bots · https://support.claude.com/en/articles/8896518 · https://developers.google.com/crawling/docs/crawlers-fetchers/google-agent · https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/ · https://developer.amazon.com/amazonbot · https://docs.mistral.ai/robots/ · https://docs.perplexity.ai/docs/resources/perplexity-crawlers · https://developers.cloudflare.com/ai-crawl-control/reference/bots/
367
+ Competitors · https://blog.cloudflare.com/agent-readiness/ · https://www.agentready.org/ · https://is-agentic.com · https://agent-ready.dev/ · https://github.com/addyosmani/agentic-seo · https://ahrefs.com/blog/llmstxt-study/
package/package.json CHANGED
@@ -1,27 +1,35 @@
1
1
  {
2
2
  "name": "ax-audit",
3
- "version": "3.6.0",
3
+ "version": "4.0.0",
4
4
  "description": "Audit websites for AI Agent Experience (AX) readiness. Lighthouse for AI Agents.",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
7
- "author": "Lucio Duran <email@lucioduran.com> (https://lucioduran.com)",
7
+ "author": "Durani Technologies (https://axrush.com)",
8
8
  "repository": {
9
9
  "type": "git",
10
- "url": "https://github.com/lucioduran/ax-audit.git"
10
+ "url": "https://github.com/duranitech/ax-audit.git"
11
11
  },
12
- "homepage": "https://github.com/lucioduran/ax-audit#readme",
12
+ "homepage": "https://axrush.com",
13
13
  "bugs": {
14
- "url": "https://github.com/lucioduran/ax-audit/issues"
14
+ "url": "https://github.com/duranitech/ax-audit/issues"
15
15
  },
16
16
  "keywords": [
17
17
  "ax",
18
18
  "ai",
19
19
  "agent-experience",
20
+ "agent-readiness",
20
21
  "audit",
21
22
  "lighthouse",
22
23
  "llms-txt",
23
24
  "a2a",
24
25
  "mcp",
26
+ "agent-card",
27
+ "webmcp",
28
+ "aeo",
29
+ "geo",
30
+ "ai-crawlers",
31
+ "robots-txt",
32
+ "content-signals",
25
33
  "seo",
26
34
  "cli",
27
35
  "devtools",
@@ -1,14 +0,0 @@
1
- import type { CheckContext, CheckResult, CheckMeta } from '../types.js';
2
- /**
3
- * "agent-json" — `/.well-known/agent.json` per the A2A (Agent-to-Agent) protocol.
4
- *
5
- * Validates the document on three axes:
6
- * 1. JSON well-formedness
7
- * 2. Required fields per the A2A spec (`name`, `description`, `url`, `skills`)
8
- * 3. Field semantics: `url` should resolve to the same origin as the audited site,
9
- * `skills[]` should each declare an `id` and `description`, and protocol/optional
10
- * fields are present where expected.
11
- */
12
- export declare const meta: CheckMeta;
13
- export default function check(ctx: CheckContext): Promise<CheckResult>;
14
- //# sourceMappingURL=agent-json.d.ts.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"agent-json.d.ts","sourceRoot":"","sources":["../../src/checks/agent-json.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,YAAY,EAAE,WAAW,EAAE,SAAS,EAAW,MAAM,aAAa,CAAC;AAGjF;;;;;;;;;GASG;AACH,eAAO,MAAM,IAAI,EAAE,SAKlB,CAAC;AAEF,wBAA8B,KAAK,CAAC,GAAG,EAAE,YAAY,GAAG,OAAO,CAAC,WAAW,CAAC,CA0I3E"}