ax-audit 4.2.1 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (188) hide show
  1. package/CHANGELOG.md +19 -0
  2. package/README.md +39 -4
  3. package/dist/baseline.js +1 -1
  4. package/dist/baseline.js.map +1 -1
  5. package/dist/check-ids.js +1 -1
  6. package/dist/check-ids.js.map +1 -1
  7. package/dist/cli.js +1 -1
  8. package/dist/cli.js.map +1 -1
  9. package/dist/constants.d.ts +1 -85
  10. package/dist/constants.d.ts.map +1 -1
  11. package/dist/constants.js +0 -135
  12. package/dist/constants.js.map +1 -1
  13. package/dist/index.d.ts +3 -1
  14. package/dist/index.d.ts.map +1 -1
  15. package/dist/index.js +2 -1
  16. package/dist/index.js.map +1 -1
  17. package/dist/license.d.ts +8 -0
  18. package/dist/license.d.ts.map +1 -0
  19. package/dist/license.js +18 -0
  20. package/dist/license.js.map +1 -0
  21. package/dist/metadata.d.ts +6 -0
  22. package/dist/metadata.d.ts.map +1 -0
  23. package/dist/metadata.js +205 -0
  24. package/dist/metadata.js.map +1 -0
  25. package/dist/orchestrator.d.ts +1 -0
  26. package/dist/orchestrator.d.ts.map +1 -1
  27. package/dist/orchestrator.js +134 -56
  28. package/dist/orchestrator.js.map +1 -1
  29. package/dist/scorer.js +2 -2
  30. package/dist/scorer.js.map +1 -1
  31. package/dist/types.d.ts +4 -0
  32. package/dist/types.d.ts.map +1 -1
  33. package/docs/api.md +14 -3
  34. package/docs/architecture.md +9 -98
  35. package/docs/ci.md +19 -0
  36. package/docs/cli.md +11 -0
  37. package/docs/faq.md +18 -0
  38. package/docs/getting-started.md +33 -1
  39. package/package.json +5 -5
  40. package/dist/checks/agent-access.d.ts +0 -33
  41. package/dist/checks/agent-access.d.ts.map +0 -1
  42. package/dist/checks/agent-access.js +0 -256
  43. package/dist/checks/agent-access.js.map +0 -1
  44. package/dist/checks/agent-card.d.ts +0 -37
  45. package/dist/checks/agent-card.d.ts.map +0 -1
  46. package/dist/checks/agent-card.js +0 -352
  47. package/dist/checks/agent-card.js.map +0 -1
  48. package/dist/checks/agent-operability.d.ts +0 -66
  49. package/dist/checks/agent-operability.d.ts.map +0 -1
  50. package/dist/checks/agent-operability.js +0 -383
  51. package/dist/checks/agent-operability.js.map +0 -1
  52. package/dist/checks/agent-skills.d.ts +0 -24
  53. package/dist/checks/agent-skills.d.ts.map +0 -1
  54. package/dist/checks/agent-skills.js +0 -316
  55. package/dist/checks/agent-skills.js.map +0 -1
  56. package/dist/checks/ai-catalog.d.ts +0 -28
  57. package/dist/checks/ai-catalog.d.ts.map +0 -1
  58. package/dist/checks/ai-catalog.js +0 -254
  59. package/dist/checks/ai-catalog.js.map +0 -1
  60. package/dist/checks/ai-directives.d.ts +0 -57
  61. package/dist/checks/ai-directives.d.ts.map +0 -1
  62. package/dist/checks/ai-directives.js +0 -263
  63. package/dist/checks/ai-directives.js.map +0 -1
  64. package/dist/checks/api-discovery.d.ts +0 -26
  65. package/dist/checks/api-discovery.d.ts.map +0 -1
  66. package/dist/checks/api-discovery.js +0 -432
  67. package/dist/checks/api-discovery.js.map +0 -1
  68. package/dist/checks/auth-discovery.d.ts +0 -28
  69. package/dist/checks/auth-discovery.d.ts.map +0 -1
  70. package/dist/checks/auth-discovery.js +0 -302
  71. package/dist/checks/auth-discovery.js.map +0 -1
  72. package/dist/checks/commerce-discovery.d.ts +0 -40
  73. package/dist/checks/commerce-discovery.d.ts.map +0 -1
  74. package/dist/checks/commerce-discovery.js +0 -295
  75. package/dist/checks/commerce-discovery.js.map +0 -1
  76. package/dist/checks/content-negotiation.d.ts +0 -4
  77. package/dist/checks/content-negotiation.d.ts.map +0 -1
  78. package/dist/checks/content-negotiation.js +0 -253
  79. package/dist/checks/content-negotiation.js.map +0 -1
  80. package/dist/checks/crawl-efficiency.d.ts +0 -16
  81. package/dist/checks/crawl-efficiency.d.ts.map +0 -1
  82. package/dist/checks/crawl-efficiency.js +0 -186
  83. package/dist/checks/crawl-efficiency.js.map +0 -1
  84. package/dist/checks/frontmatter.d.ts +0 -34
  85. package/dist/checks/frontmatter.d.ts.map +0 -1
  86. package/dist/checks/frontmatter.js +0 -100
  87. package/dist/checks/frontmatter.js.map +0 -1
  88. package/dist/checks/html-rendering.d.ts +0 -21
  89. package/dist/checks/html-rendering.d.ts.map +0 -1
  90. package/dist/checks/html-rendering.js +0 -221
  91. package/dist/checks/html-rendering.js.map +0 -1
  92. package/dist/checks/html-utils.d.ts +0 -66
  93. package/dist/checks/html-utils.d.ts.map +0 -1
  94. package/dist/checks/html-utils.js +0 -139
  95. package/dist/checks/html-utils.js.map +0 -1
  96. package/dist/checks/http-headers.d.ts +0 -11
  97. package/dist/checks/http-headers.d.ts.map +0 -1
  98. package/dist/checks/http-headers.js +0 -249
  99. package/dist/checks/http-headers.js.map +0 -1
  100. package/dist/checks/http-hygiene.d.ts +0 -26
  101. package/dist/checks/http-hygiene.d.ts.map +0 -1
  102. package/dist/checks/http-hygiene.js +0 -257
  103. package/dist/checks/http-hygiene.js.map +0 -1
  104. package/dist/checks/index.d.ts +0 -3
  105. package/dist/checks/index.d.ts.map +0 -1
  106. package/dist/checks/index.js +0 -55
  107. package/dist/checks/index.js.map +0 -1
  108. package/dist/checks/llms-txt.d.ts +0 -19
  109. package/dist/checks/llms-txt.d.ts.map +0 -1
  110. package/dist/checks/llms-txt.js +0 -281
  111. package/dist/checks/llms-txt.js.map +0 -1
  112. package/dist/checks/mcp-discovery.d.ts +0 -30
  113. package/dist/checks/mcp-discovery.d.ts.map +0 -1
  114. package/dist/checks/mcp-discovery.js +0 -523
  115. package/dist/checks/mcp-discovery.js.map +0 -1
  116. package/dist/checks/meta-tags.d.ts +0 -17
  117. package/dist/checks/meta-tags.d.ts.map +0 -1
  118. package/dist/checks/meta-tags.js +0 -188
  119. package/dist/checks/meta-tags.js.map +0 -1
  120. package/dist/checks/robots-parser.d.ts +0 -110
  121. package/dist/checks/robots-parser.d.ts.map +0 -1
  122. package/dist/checks/robots-parser.js +0 -277
  123. package/dist/checks/robots-parser.js.map +0 -1
  124. package/dist/checks/robots-txt.d.ts +0 -6
  125. package/dist/checks/robots-txt.d.ts.map +0 -1
  126. package/dist/checks/robots-txt.js +0 -367
  127. package/dist/checks/robots-txt.js.map +0 -1
  128. package/dist/checks/rsl.d.ts +0 -4
  129. package/dist/checks/rsl.d.ts.map +0 -1
  130. package/dist/checks/rsl.js +0 -242
  131. package/dist/checks/rsl.js.map +0 -1
  132. package/dist/checks/security-txt.d.ts +0 -4
  133. package/dist/checks/security-txt.d.ts.map +0 -1
  134. package/dist/checks/security-txt.js +0 -81
  135. package/dist/checks/security-txt.js.map +0 -1
  136. package/dist/checks/seo-basics.d.ts +0 -13
  137. package/dist/checks/seo-basics.d.ts.map +0 -1
  138. package/dist/checks/seo-basics.js +0 -221
  139. package/dist/checks/seo-basics.js.map +0 -1
  140. package/dist/checks/sitemap.d.ts +0 -12
  141. package/dist/checks/sitemap.d.ts.map +0 -1
  142. package/dist/checks/sitemap.js +0 -240
  143. package/dist/checks/sitemap.js.map +0 -1
  144. package/dist/checks/structured-data.d.ts +0 -4
  145. package/dist/checks/structured-data.d.ts.map +0 -1
  146. package/dist/checks/structured-data.js +0 -408
  147. package/dist/checks/structured-data.js.map +0 -1
  148. package/dist/checks/structured-fields.d.ts +0 -46
  149. package/dist/checks/structured-fields.d.ts.map +0 -1
  150. package/dist/checks/structured-fields.js +0 -112
  151. package/dist/checks/structured-fields.js.map +0 -1
  152. package/dist/checks/surface.d.ts +0 -59
  153. package/dist/checks/surface.d.ts.map +0 -1
  154. package/dist/checks/surface.js +0 -106
  155. package/dist/checks/surface.js.map +0 -1
  156. package/dist/checks/tls-https.d.ts +0 -13
  157. package/dist/checks/tls-https.d.ts.map +0 -1
  158. package/dist/checks/tls-https.js +0 -163
  159. package/dist/checks/tls-https.js.map +0 -1
  160. package/dist/checks/usage-policy.d.ts +0 -53
  161. package/dist/checks/usage-policy.d.ts.map +0 -1
  162. package/dist/checks/usage-policy.js +0 -339
  163. package/dist/checks/usage-policy.js.map +0 -1
  164. package/dist/checks/utils.d.ts +0 -40
  165. package/dist/checks/utils.d.ts.map +0 -1
  166. package/dist/checks/utils.js +0 -74
  167. package/dist/checks/utils.js.map +0 -1
  168. package/dist/checks/waf.d.ts +0 -75
  169. package/dist/checks/waf.d.ts.map +0 -1
  170. package/dist/checks/waf.js +0 -203
  171. package/dist/checks/waf.js.map +0 -1
  172. package/dist/checks/webmcp.d.ts +0 -55
  173. package/dist/checks/webmcp.d.ts.map +0 -1
  174. package/dist/checks/webmcp.js +0 -209
  175. package/dist/checks/webmcp.js.map +0 -1
  176. package/dist/checks/well-known.d.ts +0 -38
  177. package/dist/checks/well-known.d.ts.map +0 -1
  178. package/dist/checks/well-known.js +0 -202
  179. package/dist/checks/well-known.js.map +0 -1
  180. package/dist/fetcher.d.ts +0 -15
  181. package/dist/fetcher.d.ts.map +0 -1
  182. package/dist/fetcher.js +0 -127
  183. package/dist/fetcher.js.map +0 -1
  184. package/dist/guide-urls.d.ts +0 -2
  185. package/dist/guide-urls.d.ts.map +0 -1
  186. package/dist/guide-urls.js +0 -5
  187. package/dist/guide-urls.js.map +0 -1
  188. package/docs/roadmap.md +0 -367
package/docs/roadmap.md DELETED
@@ -1,367 +0,0 @@
1
- # Roadmap: 3.7 → 4.0 — **completed 2026-09-04**
2
-
3
- *Research snapshot and implementation plan. All four releases shipped; this document is kept as the record of what was verified and why each decision was made.*
4
-
5
- ax-audit has not shipped since 3.6.0 (2026-06-09). This document records what changed in the agent-web ecosystem since then, which existing checks are now wrong or stale, which new checks are worth adding, and a phased implementation plan that respects the 3.x scoring policy (no downward score changes until 4.0).
6
-
7
- Sources were verified against primary specs, vendor docs, IANA, IETF datatracker and GitHub on 2026-09-04. Items marked **(secondary)** rely on trade press or third-party write-ups only.
8
-
9
- ---
10
-
11
- ## 1. Executive summary
12
-
13
- **Three existing checks probe paths that are no longer (or never were) the standard:**
14
-
15
- | Check | Today | Reality (Sept 2026) |
16
- | --- | --- | --- |
17
- | `agent-json` | `/.well-known/agent.json`, hints `protocolVersion: "0.2.0"`, expects `authentication` | A2A moved to **`/.well-known/agent-card.json`** in v0.3.0 (2025-07-30), IANA-registered permanent. A2A **1.0.1** (2026-05-28) replaced `url`/`protocolVersion`/`preferredTransport` with `supportedInterfaces[]`; `authentication` → `securitySchemes`. |
18
- | `mcp` | `/.well-known/mcp.json` with `tools[]`, hints `2024-11-05` | **Never a spec convention.** Current draft (SEP-2127 + `experimental-ext-server-card`) is `<mcp-endpoint>/server-card` (`application/mcp-server-card+json`), Cloudflare/Mintlify serve `/.well-known/mcp/server-card.json`; umbrella `/.well-known/ai-catalog.json`. Server cards carry **no** `tools[]`. Protocol version is **2026-07-28**. Auth discovery via RFC 9728 is mandatory for remote servers. |
19
- | `well-known-ai` | ai.txt, genai.txt, ai-plugin.json, agents.json, nlweb.json | `nlweb.json` **does not exist** (NLWeb uses `/ask` + `/mcp`); `genai.txt` has no spec; `ai-plugin.json` dead since 2024-04-09; Wildcard `agents.json` dormant since 2025-08-21; Spawning `ai.txt` lives at root and has ~0 adoption. The whole bundle needs replacing. |
20
-
21
- **The crawler list has fake, retired and misclassified tokens** (`Gemini`, `GeminiBot`, `DeepSeek-AI`, `NeevaBot`, `Goose`, `Awario*`; `ChatGPT-User`/`Claude-User`/`Perplexity-User`/`MistralAI-User`/`meta-externalfetcher` are user-triggered fetchers, not training bots) and is missing the bots that now dominate traffic (`meta-webindexer`, `Amzn-SearchBot`, `Amzn-User`, `MistralAI-Index`, `MistralAI-Training`, `Google-GeminiNotebook`, `Applebot`, `ExaSearchBot`).
22
-
23
- **Two competitors now define the reference bar:** Google Lighthouse 13.3 shipped an "Agentic Browsing" category (2026-05-07) and Cloudflare launched an "Agent Readiness" score (2026-04-17). Both check things ax-audit does not: ARD / `ai-catalog.json`, WebMCP declarative forms, agent-skills index, RFC 9727 API catalog, OAuth discovery (RFC 8414/9728), Web Bot Auth key directories, Link-header discovery. Ora's AgentReady v1.0 (with Vercel and Mintlify, Aug 2026) adds HTTP-status honesty, `429 + Retry-After`, and conditional "N/A" scoring.
24
-
25
- **New signals worth auditing:** robots meta AI directives (`nosnippet`, `max-snippet`, `noarchive`, `nocache` are the only page-level controls Google and Bing actually honor for AI answers), IETF AIPREF `Content-Usage`, Content Signals `use=` field, TDMRep, WAF challenge vs hard block vs 402 pay-per-crawl classification, browser-agent operability heuristics, llms.txt v2 (subpath files, `rel="describedby"`, `.md` mirrors), and Markdown-for-Agents token headers.
26
-
27
- **Recommended shape:** three minor releases (3.7, 3.8, 3.9) that fix stale probes and add ~10 informational checks without lowering any score, then **4.0** that redistributes weights, introduces conditional (N/A) checks and report categories, and retires the legacy bundle.
28
-
29
- ---
30
-
31
- ## 2. Ecosystem changes since June 2026 (verified)
32
-
33
- ### 2.1 Protocols
34
-
35
- - **A2A 1.0.0** (2026-03-12, breaking) and **1.0.1** (2026-05-28). Card required fields: `name`, `description`, `version`, `capabilities`, `supportedInterfaces[]{url, protocolBinding ∈ JSONRPC|GRPC|HTTP+JSON, protocolVersion}`, `defaultInputModes[]`, `defaultOutputModes[]`, `skills[]`. Optional: `provider`, `documentationUrl`, `iconUrl`, `securitySchemes`, `security`/`securityRequirements`, `signatures[]`, `capabilities.extensions[]` (AP2 lives here). v0.3 cards (still the majority deployed) have top-level `url` + `protocolVersion`. No hosted 1.0 JSON schema; the v0.3.0 schema is at `raw.githubusercontent.com/a2aproject/A2A/v0.3.0/specification/json/a2a.json`. — https://github.com/a2aproject/A2A/releases, https://a2a-protocol.org/latest/specification/
36
- - **MCP 2026-07-28** removed sessions and `initialize`; POSTs carry `MCP-Protocol-Version`, `Mcp-Method`, `Mcp-Name`; `GET /mcp` → 405; new `server/discover`. Discovery: SEP-2127 (open draft) → `experimental-ext-server-card`: `GET <endpoint>/server-card`, required `$schema`, `name` (reverse-DNS), `version`, `description`; optional `title`, `websiteUrl`, `repository`, `icons`, `remotes[]{type ∈ streamable-http|sse, url, supportedProtocolVersions[]}`; CORS `*` and `Cache-Control` recommended. Auth: RFC 9728 `/.well-known/oauth-protected-resource[/mcp]` → `authorization_servers[]` → RFC 8414 or OIDC discovery. — https://modelcontextprotocol.io/specification/2026-07-28/changelog, https://github.com/modelcontextprotocol/modelcontextprotocol/pull/2127, https://github.com/modelcontextprotocol/experimental-ext-server-card
37
- - **ai-catalog.json / ARD**: Linux Foundation "Agent Card WG" `/.well-known/ai-catalog.json` (`specVersion`, `host{displayName, identifier}`, `entries[]{identifier, type, url|data, displayName}`; entry types `application/mcp-server-card+json`, `application/a2a-agent-card+json`). Agentic Resource Discovery (Google/Microsoft/HF listed) spec v0.91 (2026-08-26) at `/.well-known/ard.json`. Lighthouse's `ard-schema` audit discovers via robots.txt `Agentmap:`, `<link rel="ai-catalog">`, `Link: rel="ai-catalog"`, or the well-known path. — https://ai-catalog.io/, https://agenticresourcediscovery.org/spec/
38
- - **WebMCP**: W3C WebML CG draft (2026-09-04). Imperative `document.modelContext.registerTool()` (`navigator.modelContext` deprecated). Declarative: `<form toolname tooldescription [toolautosubmit]>`, controls `toolparamdescription`. Chrome origin trial 149→156 (ends ~2026-11-16). Lighthouse audits `forms-missing-declarative-webmcp`. OpenAI enabled WebMCP in the ChatGPT desktop browser (2026-08-25) **(secondary)**. — https://webmachinelearning.github.io/webmcp/, https://developer.chrome.com/docs/ai/webmcp/declarative-api
39
- - **Agent Skills discovery**: Cloudflare RFC v0.2.0 (2026-03-12) `/.well-known/agent-skills/index.json` (`$schema https://schemas.agentskills.io/discovery/0.2.0/schema.json`, `skills[]{name, type ∈ skill-md|archive, description ≤1024, url, digest "sha256:<64hex>"}`), SKILL.md at `/.well-known/agent-skills/{name}/SKILL.md`. Mintlify/Docus variant `/.well-known/skills/index.json`. SKILL.md frontmatter per agentskills.io: `name` (1–64, `[a-z0-9-]`), `description` (1–1024). — https://github.com/cloudflare/agent-skills-discovery-rfc, https://agentskills.io/specification
40
- - **UCP** (Google/Shopify/Etsy/Walmart/Stripe): `/.well-known/ucp` (no extension, public), spec 2026-08-25: `ucp.version` (date string), `ucp.services` (reverse-DNS keys → transports rest/mcp/a2a with `schema` URL), `ucp.payment_handlers`, optional `ucp.capabilities`, `keys[]`. Shopify Agentic Storefronts GA March 2026. **ACP** (OpenAI/Stripe) has **no** discovery mechanism; AP2 is an A2A extension URI. — https://developers.google.com/merchant/ucp/guides/ucp-profile, https://developers.openai.com/commerce/specs/checkout
41
- - **OpenAI Apps**: MCP-based; domain verification file `/.well-known/openai-apps-challenge`. — https://developers.openai.com/plugins/deploy/submission.md
42
- - **NLWeb**: `/ask` (`query`, `site`, `mode ∈ list|summarize|generate`) and `/mcp`; repo active (2026-08-11). No manifest file.
43
-
44
- ### 2.2 Content discovery and readability
45
-
46
- - **llms.txt v2** (llmstxt.org, modified 2026-08-10): subpath files (`/docs/llms.txt`, most specific wins); `<link rel="describedby">` / `Link: rel="describedby"` → the covering llms.txt; per-page mirrors `page.md`, `page.html.md`, `index.html.md`, `index.md`. Google states Search ignores it (ai-optimization-guide, 2026-07-10). Adoption ~10% (SE Ranking, May 2026); Ahrefs (2026-06-15): 97% of published files never fetched, but Claude Code out-fetches every AI search bot. Lighthouse's `llms-txt` audit: 404 → N/A, 5xx → fail, present → fail on missing H1 / too short / no links.
47
- - **Markdown for Agents**: Cloudflare (docs 2026-07-13) responds to `Accept: text/markdown` with `text/markdown`, `x-markdown-tokens`, `x-original-tokens`, `content-signal` header, `Vary: accept`; strips `ETag`/`Last-Modified`/`Content-Encoding`. Vercel (2026-09-03) negotiates on Accept **and** on agent UA without Accept, keeps `.md` suffix URLs, emits YAML frontmatter (`title`, `canonical_url`, `last_updated`…), `/sitemap.md`, and requires `Vary: Accept`. Senders of `Accept: text/markdown` (Checkly, Feb 2026): Claude Code, Cursor, OpenCode. No IETF standard; only `draft-consolidated-content` (individual). — https://developers.cloudflare.com/fundamentals/reference/markdown-for-agents/, https://vercel.com/docs/agent-resources/markdown-access
48
- - **API discovery**: RFC 9727 `/.well-known/api-catalog` (`application/linkset+json`, `Link: rel="api-catalog"` on `/`, entries with `service-desc`/`service-doc`/`service-meta`/`status`); RFC 8631 relations. `/.well-known/openapi.json` is **not** IANA-registered. OpenAPI **3.2.0** (2025-09-19) recommends `openapi.json`/`openapi.yaml`; Arazzo 1.1.0 (2026-05-17).
49
- - **Structured data**: Google: no special markup for AI features, but structured data must match visible text; FAQ rich results ended 2026-05-07; schema.org 30.0 (2026-03-19) adds no AI types. Bing "AI Performance" report (Feb 2026).
50
- - **IANA well-known registry** (2026-08-19): registered and agent-relevant: `agent-card.json`, `api-catalog`, `oauth-protected-resource`, `oauth-authorization-server`, `tdmrep.json`, `gpc.json`, `security.txt`. **Not** registered: `openapi*`, `mcp*`, `ucp`, `ai.txt`, `llms*`, `agents.json`, `skills`, `ai-catalog.json`. Reports should label each probe as *registered* / *vendor convention* / *draft*.
51
-
52
- ### 2.3 Usage-rights and access signals
53
-
54
- - **IETF AIPREF** (`draft-ietf-aipref-vocab-07`, `-attach-05`, 2026-08-19; pre-WGLC, "does not reflect consensus"): tokens `train-ai`, `search`; values `y`/`n`; RFC 9651 dictionary. Carriers: HTTP `Content-Usage: train-ai=n` response header and robots.txt `Content-Usage: [/path ]train-ai=n` inside User-agent groups. Note the inversion vs Content Signals/RSL (`ai-train`). — https://datatracker.ietf.org/wg/aipref/documents/
55
- - **Content Signals**: new optional 4th field `use=immediate|reference|full` (Cloudflare, 2026-07-01), emitted by managed robots.txt as `Content-signal: search=yes, ai-train=no, use=reference` (lower-case s, spaces after commas). Google publicly states no crawler honors `content-signal` **(secondary: Mueller, 2026-07-06)**. Cloudflare blocks Training + Agent categories by default on ad-bearing pages from **2026-09-15**.
56
- - **RSL** still 1.0; 2026 errata add `<reporting profile= endpoint=>` (2026-06-12) and require ignoring unknown extension elements (2026-08-07). OLP: `401/402` + `WWW-Authenticate: License` + `Link rel="license"`.
57
- - **Pay-per-crawl** (closed beta, doc 2026-07-28): `402` + `crawler-price: USD 0.01`; `200` + `crawler-charged`; `402` + `crawler-error`. Always-free paths: `/robots.txt`, `/sitemap.xml`, `/security.txt`, `/.well-known/security.txt`, `/crawlers.json`. AWS WAF x402 monetization (2026-06-15): `402` with `payment-signature`/`payment-response`. Cloudflare AI Crawl Control block = `403` **or** `402` with custom body.
58
- - **Web Bot Auth**: WG adopted `draft-ietf-webbotauth-httpsig-protocol-00` (2026-09-01, Standards Track). Headers `Signature`, `Signature-Input`, `Signature-Agent`; `tag="web-bot-auth"`; key directory `/.well-known/http-message-signatures-directory` (`application/http-message-signatures-directory+json`, Ed25519 JWKS, response itself signed with `tag="http-message-signatures-directory"`). Origins may answer `403` + `Accept-Signature`. Signers: OpenAI (`https://chatgpt.com`), Google (`https://agent.bot.goog`, experimental), Exa, You.com, Amazon AgentCore. Verifiers: Cloudflare, AWS WAF, Vercel, Akamai.
59
- - **WAF response signatures**: Cloudflare challenge → `cf-mitigated: challenge` (always `text/html`); Vercel → `x-vercel-mitigated: challenge` **(secondary)**; AWS WAF challenge → **`202`** + `x-amzn-waf-action: challenge`.
60
- - **Robots meta**: Google `nosnippet` / `max-snippet:N` / `data-nosnippet` limit direct input to AI Overviews and AI Mode; `Google-Extended` governs Gemini training + grounding, **not** AI Overviews. Bing `noarchive` = excluded from Copilot grounding; `nocache` = URL/title/snippet only; new `data-snippet` attribute **(secondary)**. `noai`/`noimageai`: no operator commits to honoring. Google Search Console "Search generative AI control" (2026-08-31) is property-level and not machine-readable.
61
- - **TDMRep** (W3C CG Final 2024-05-10, cited in the EU GPAI Code of Practice): `tdm-reservation: 0|1` / `tdm-policy` headers, `/.well-known/tdmrep.json` (`[{location, tdm-reservation, tdm-policy}]`), `<meta name="tdm-reservation">`; precedence meta > header > file.
62
-
63
- ### 2.4 Crawler landscape (Cloudflare Radar, Aug 2026)
64
-
65
- Googlebot 27%, Meta-ExternalAgent 12.7%, ClaudeBot 11.9%, Bingbot 8.5%, GPTBot 8.3%, Applebot 6.7%, Amazonbot 6.1%, Bytespider 4.7%, Claude-SearchBot 3.6%. Cloudflare's taxonomy is Search / Agent / Training. Only `Google-Agent` ships a documented UA for agentic browsing; ChatGPT agent, Claude in Chrome, Comet and Copilot use plain Chrome UAs and identify (if at all) through Web Bot Auth.
66
-
67
- Vendor changes: OpenAI `ChatGPT-User` "robots.txt may not apply" (Dec 2025), new `OAI-AdsBot` (Apr 2026); Anthropic three-bot split documented (Feb 2026), IPs at `claude.com/crawling/bots.json`; Google `Google-Agent` (2026-03-20), `Google-NotebookLM` → `Google-GeminiNotebook` (2026-07-17), Mariner retired (2026-05-04); Meta `meta-webindexer` (search index, ~38% of tracked AI requests on peak days **(secondary)**); Amazon `Amzn-SearchBot`, `Amzn-User`, robots-only management from 2026-06-15; Mistral `MistralAI-Index`, `MistralAI-Training`; Apple's Applebot doc now states training use (2026-06-08); Cohere runs no crawlers; Exa `ExaSearchBot` signs every request.
68
-
69
- ---
70
-
71
- ## 3. Gap analysis against the current 18 checks
72
-
73
- | Check | Status | Required change |
74
- | --- | --- | --- |
75
- | `llms-txt` | Stale (spec v2) | Add `rel="describedby"` detection (HTML + `Link`), subpath discovery, `.md` mirror probe, Lighthouse-aligned rules, link-liveness sampling, size heuristics. Reword copy: value is developer-agent tooling, not Google Search. |
76
- | `robots-txt` | Stale list + partial Content Signals | Refresh `AI_CRAWLERS` (§4.1), tiered reporting (blocking search bots ≠ blocking training bots), `use=` field, case/space tolerance, Cloudflare-managed block detection, `Content-Usage` parsing, `Agentmap:` directive, `Google-Extended` semantics in hints. |
77
- | `agent-json` | **Wrong path** | Probe `agent-card.json` first, `agent.json` as legacy fallback with warning. Detect card generation (1.0 vs 0.3) and validate accordingly. Drop `0.2.0` hint, flag `authentication`. |
78
- | `mcp` | **Wrong convention** | Replace with server-card discovery chain (§4.4). Keep `mcp.json` as legacy fallback, never as a recommendation. |
79
- | `openapi` | Folk path only | Become `api-discovery`: RFC 9727 first, then conventional paths, `Link`/`<link>` `service-desc`, OpenAPI 3.0–3.2. |
80
- | `http-headers` | Narrow Link parsing, stale agent.json reference | Broaden Link relations (`describedby`, `api-catalog`, `ai-catalog`, `service-desc`, `service-doc`, `alternate text/markdown`), fix agent-card path, recognise `X-Llms-Txt`. |
81
- | `well-known-ai` | **Mostly fictional bundle** | Freeze scoring in 3.x, add informational findings for real files; replace in 4.0 with `ai-catalog` + `agent-skills` and drop nlweb/genai/ai-plugin. |
82
- | `meta-tags` | `ai:*` namespace has no known consumer | Keep, but demote in 4.0 weights; add `article:published_time`/`modified_time` reading for freshness. |
83
- | `structured-data` | Type-presence only | Add `dateModified`/`datePublished`, `author` with `sameAs`, `Organization.sameAs`; consistency of `headline`/`name`/`description` vs visible text; neutralise FAQPage/HowTo. |
84
- | `content-negotiation` | Good, extend | Realistic Accept string, UA-only probe (Vercel mode), `x-markdown-tokens`/`x-original-tokens`, `content-signal` header, `.md` suffix, frontmatter parse, `/sitemap.md`, `Link: rel="canonical"` on markdown. |
85
- | `agent-access` | Good, extend | Classify responses (§4.9): challenge vs hard block vs 402 paywall vs `Accept-Signature` vs RSL OLP; separate "crawlers that honor robots" from "user fetchers that don't"; compare title/H1/JSON-LD hash, not only text length. |
86
- | `crawl-efficiency` | Good | Add token estimate of extracted text, response time, `Retry-After` on any 429 seen. |
87
- | `rsl` | Current | Accept `<reporting>`; ignore unknown extension elements; detect OLP responses. |
88
- | `sitemap`, `seo-basics`, `tls-https`, `security-txt`, `html-rendering` | Current | Minor: `<html lang>` vs `Content-Language` vs `inLanguage` consistency; feeds `<link rel="alternate" type=rss/atom>`; `<img>`/`<iframe>` missing dimensions (CLS proxy). |
89
-
90
- ---
91
-
92
- ## 4. Specification of changes and new checks
93
-
94
- Each new check follows the anatomy in [architecture.md](./architecture.md): `weight: 0` in 3.x, every warn/fail with `hint` + `learnMoreUrl`, network via `ctx.fetch`, regex primitives (no parser deps).
95
-
96
- ### 4.1 `constants.ts` — crawler catalogue refresh
97
-
98
- Replace the three buckets with four plus a legacy alias list. Matching stays case-insensitive (RFC 9309).
99
-
100
- ```ts
101
- export const AI_CRAWLERS = {
102
- training: ['GPTBot', 'ClaudeBot', 'Meta-ExternalAgent', 'Google-Extended', 'Applebot-Extended',
103
- 'Amazonbot', 'CCBot', 'Bytespider', 'TikTokSpider', 'MistralAI-Training', 'AI2Bot', 'Ai2Bot-Dolma',
104
- 'DeepSeekBot', 'PanguBot', 'Google-CloudVertexBot', 'FacebookBot', 'Timpibot', 'Webzio-Extended',
105
- 'omgili', 'omgilibot', 'ImagesiftBot', 'Kangaroo Bot', 'Diffbot', 'YandexAdditional', 'YandexAdditionalBot'],
106
- search: ['OAI-SearchBot', 'Claude-SearchBot', 'PerplexityBot', 'meta-webindexer', 'Amzn-SearchBot',
107
- 'MistralAI-Index', 'Applebot', 'bingbot', 'DuckAssistBot', 'YouBot', 'Kagibot', 'PetalBot',
108
- 'ExaSearchBot', 'PhindBot', 'Yeti'],
109
- userFetch: ['ChatGPT-User', 'Claude-User', 'Perplexity-User', 'MistralAI-User', 'meta-externalfetcher',
110
- 'Amzn-User', 'Google-GeminiNotebook', 'kagi-fetcher', 'Kimi-User', 'TongyiBot'],
111
- agentBrowsing: ['Google-Agent', 'NovaAct', 'Manus-User', 'Devin', 'FirecrawlAgent', 'TavilyBot'],
112
- };
113
- export const LEGACY_AI_CRAWLERS = ['Claude-Web', 'Anthropic-AI', 'Google-NotebookLM', 'GoogleAgent-Mariner',
114
- 'cohere-ai', 'cohere-training-data-crawler', 'ExaBot', 'NeevaBot', 'Gemini', 'GeminiBot', 'DeepSeek-AI'];
115
- export const CORE_AI_CRAWLERS = ['GPTBot', 'ClaudeBot', 'Meta-ExternalAgent', 'Google-Extended',
116
- 'Applebot-Extended', 'Amazonbot', 'Bytespider', 'CCBot', 'OAI-SearchBot', 'Claude-SearchBot',
117
- 'PerplexityBot', 'ChatGPT-User'];
118
- ```
119
-
120
- Add a `CRAWLER_META` map (`purpose`, `honorsRobots: boolean`, `docUrl`, `ipListUrl`) so findings can say *why* a block matters ("blocking OAI-SearchBot removes you from ChatGPT search answers; blocking GPTBot only affects training"). Bots documented as ignoring robots.txt (`Perplexity-User`, `Google-Agent`, `ChatGPT-User` partially) must not be scored as "missing" in robots.txt.
121
-
122
- **Scoring impact in 3.x:** `robots-txt` deducts by missing core crawlers. Growing CORE from 8 to 12 would lower scores → keep the 8-token `CORE_AI_CRAWLERS_V3` for scoring until 4.0 and use the 12-token list for informational findings only.
123
-
124
- ### 4.2 Shared infrastructure (prerequisites)
125
-
126
- 1. **`src/checks/robots-parser.ts`** — extract `parseUserAgents`, `parseContentSignalDecls`, `parseRobotsLicenseDirectives` into one module, add `Content-Usage` (path-scoped, per group), `Agentmap:`, `Sitemap:`, Cloudflare managed-block markers. `robots-txt`, `rsl`, `agent-access`, `usage-policy`, `ai-catalog` all consume it; robots.txt is fetched once via the fetcher cache.
127
- 2. **`src/checks/waf.ts`** — `classifyResponse(res): 'ok' | 'challenge-cloudflare' | 'challenge-vercel' | 'challenge-aws' | 'blocked' | 'paywall-ppc' | 'paywall-x402' | 'needs-signature' | 'license-required'` from status + headers (`cf-mitigated`, `x-vercel-mitigated`, `x-amzn-waf-action`, `crawler-price`, `payment-signature`, `Accept-Signature`, `WWW-Authenticate: License`). Used by `agent-access`, `http-hygiene`, `content-negotiation`.
128
- 3. **`src/checks/structured-fields.ts`** — minimal RFC 9651 dictionary parser (`key=token`, `key=?1`, params) for `Content-Usage`, `Signature-Input`, `Content-Signal` header.
129
- 4. **`src/checks/frontmatter.ts`** — flat `key: value` YAML frontmatter reader for SKILL.md and `.md` mirrors (no nested YAML).
130
- 5. **`src/checks/tokens.ts`** — `estimateTokens(text)` (chars/4 heuristic, documented as approximate).
131
- 6. **Fetcher**: add `method?: 'GET' | 'HEAD'` and `redirect?: 'follow' | 'manual'` to `FetchOptions`; expose `redirectCount` and `elapsedMs` on `FetchResponse`. HEAD is needed for llms.txt link sampling; manual redirects for hop counting and `.md` mirror checks.
132
- 7. **Types**: `CheckResult.applicable?: boolean` (default `true`). Scorer excludes `applicable === false` from the denominator (4.0 behaviour; in 3.x reporters just display "N/A"). `CheckMeta.category: 'discovery' | 'content' | 'access' | 'policy' | 'protocols' | 'trust'` for grouped reports.
133
- 8. **Well-known probe helper**: `probeWellKnown(ctx, paths[], { accept })` returning first hit with `registered: boolean` label from a small `WELL_KNOWN_REGISTRY` constant.
134
-
135
- ### 4.3 `agent-card` (rename of `agent-json`, same id kept as alias in 3.x)
136
-
137
- - Probe order: `/.well-known/agent-card.json` → `/.well-known/agent.json` (warn "legacy 0.2.x path; A2A ≥0.3 uses agent-card.json").
138
- - Content-Type: accept `application/json` or `application/a2a+json`.
139
- - Generation detection: `supportedInterfaces[]` → 1.0 rules; top-level `url` + `protocolVersion` → 0.3 rules; neither → fail "unrecognised card shape".
140
- - 1.0 required: `name, description, version, capabilities, supportedInterfaces, defaultInputModes, defaultOutputModes, skills` (−15 each, as today). Each interface needs `url`, `protocolBinding`, `protocolVersion`.
141
- - 0.3 required: `name, description, url, version, protocolVersion, capabilities, defaultInputModes, defaultOutputModes, skills`.
142
- - `authentication` present → warn "removed in 0.2.x, use securitySchemes" (−5). `skills[]` entries need `id`, `name`, `description`, `tags`.
143
- - Optional credit: `provider`, `documentationUrl`, `iconUrl`, `securitySchemes`, `signatures[]`, `capabilities.extensions[]` (report AP2/other extension URIs).
144
- - Same-origin check on interface URLs (warn only).
145
- - 3.x scoring: unchanged formulas; the new path can only raise scores.
146
-
147
- ### 4.4 `mcp-discovery` (rename of `mcp`)
148
-
149
- Discovery chain, first hit wins, each labelled with its standing:
150
-
151
- 1. `/.well-known/ai-catalog.json` entries of type `application/mcp-server-card+json` (draft, LF).
152
- 2. `/.well-known/mcp/server-card.json`, `/.well-known/mcp/server-cards.json` (Cloudflare / Mintlify convention).
153
- 3. `/mcp/server-card` (experimental-ext-server-card recommendation) — also `<remote-url>/server-card` for any remote found in step 1–2.
154
- 4. `/.well-known/mcp.json` (legacy, ax-audit's own past recommendation) → warn.
155
-
156
- Validation of a server card: `$schema`, `name` (reverse-DNS `^[a-z0-9.-]+/[a-z0-9._-]+$` or dotted), `version`, `description` (−15 each); `remotes[]` with `type ∈ streamable-http|sse`, absolute `url`, `supportedProtocolVersions[]` containing a known version (`2026-07-28`, `2025-11-25`, `2025-06-18`, `2025-03-26`, `2024-11-05`; warn if only pre-2025 versions) (−10); `Content-Type: application/mcp-server-card+json` or `application/json` (−5); CORS `*` (−5); `Cache-Control` present (info). Do **not** expect `tools[]` — hint that tool lists come from `tools/list` at runtime.
157
-
158
- Optional live probe (flag `--probe-mcp`, off by default): `GET <remote>` expecting `405` (2026-07-28 signal) or `POST` `server/discover` with `MCP-Protocol-Version`; any JSON-RPC response or a protocol-version error counts as "reachable".
159
-
160
- Auth chain (informational, feeds `auth-discovery`): if any remote exists, probe `/.well-known/oauth-protected-resource` and `/.well-known/oauth-protected-resource<remote-path>`.
161
-
162
- ### 4.5 `api-discovery` (rename of `openapi`)
163
-
164
- 1. `HEAD /` (or reuse homepage headers) → `Link: rel="api-catalog"`; `GET /.well-known/api-catalog` with `Accept: application/linkset+json`. Validate `linkset[]`, each with `anchor` and at least one of `service-desc` / `service-doc`; resolve one `service-desc` and confirm it parses as OpenAPI/AsyncAPI/Arazzo.
165
- 2. `<link rel="service-desc">`, `<link rel="service-doc">` in HTML and `Link` header (RFC 8631).
166
- 3. Conventional paths: `/openapi.json`, `/openapi.yaml`, `/.well-known/openapi.json`, `/.well-known/openapi.yaml`, `/swagger.json`, `/api-docs`, `/v1/openapi.json`, `/arazzo.json`, `/asyncapi.json`.
167
- 4. OpenAPI validation as today plus: version `3.0`–`3.2` (note `$self` in 3.2), `operationId` coverage (warn <100%), `servers[]`, `info.description`, `securitySchemes` presence, documented rate limits (`x-ratelimit*` extensions or `429` responses) (info).
168
-
169
- Label: RFC 9727 = registered; folk paths = convention. Scoring in 3.x unchanged for sites that only have `/.well-known/openapi.json`.
170
-
171
- ### 4.6 `ai-catalog` (new, replaces the discovery half of `well-known-ai`)
172
-
173
- Discovery per Lighthouse: robots.txt `Agentmap:` → `<link rel="ai-catalog">`/`<link rel="ard">` → `Link: rel="ai-catalog"` → `/.well-known/ai-catalog.json` → `/.well-known/ard.json`. Validate `specVersion`, `host.identifier`, `entries[]` each with `identifier`, `type`, `displayName`, and `url` xor `data`; resolve each `url` (HEAD) and check its `Content-Type` matches `type`. Cross-reference: an `application/a2a-agent-card+json` entry should point where `agent-card` found the card; an MCP entry should match `mcp-discovery`.
174
-
175
- ### 4.7 `agent-skills` (new)
176
-
177
- Probe `/.well-known/agent-skills/index.json` (canonical RFC) then `/.well-known/skills/index.json` (Mintlify variant), then `/skill.md`. Validate index: `skills[]` non-empty; each `name` matches `^[a-z0-9-]{1,64}$`, `description` 1–1024, `type ∈ skill-md|archive`, `url` absolute or root-relative, `digest` matches `^sha256:[0-9a-f]{64}$` (warn if missing). Fetch up to 3 SKILL.md files: `text/markdown` Content-Type, frontmatter `name` equals index name and directory, `description` present, body ≤500 lines (info). Not applicable (`applicable: false`) when nothing is found **and** the site shows no docs signals (no `/docs` link, no llms.txt).
178
-
179
- ### 4.8 `ai-directives` (new) — page-level AI controls
180
-
181
- Parse `<meta name="robots|googlebot|bingbot">` and `X-Robots-Tag` (all values, comma-split, UA-prefixed forms). Report with vendor semantics:
182
-
183
- | Directive | Finding |
184
- | --- | --- |
185
- | `noindex` on homepage | fail: invisible to every search-grounded assistant |
186
- | `nosnippet` or `max-snippet:0` | warn: excluded as direct input to Google AI Overviews / AI Mode |
187
- | `max-snippet:N` with N < 160 | info |
188
- | `noarchive` | warn: excluded from Bing Copilot grounding |
189
- | `nocache` | info: Copilot may use URL/title/snippet only |
190
- | `noai`, `noimageai` | info: declared preference, no operator commits to honoring |
191
- | `data-nosnippet` wrapping `<main>`, `<article>` or the H1 block | warn |
192
- | `Content-Signal` says `ai-input=no` while page is not `nosnippet` | info (inconsistent intent) |
193
- | Hint when robots.txt disallows `Google-Extended` | explain it does not remove the site from AI Overviews |
194
-
195
- Score (4.0 proposal): start 100; `noindex` → 0; `nosnippet`/`noarchive` −30 each; contradictions −10. Weight 0 in 3.x.
196
-
197
- ### 4.9 `agent-access` — response classification (informational check, free to change in 3.x)
198
-
199
- Per probe, run `classifyResponse` and map to outcomes:
200
-
201
- | Classification | Credit | Message |
202
- | --- | --- | --- |
203
- | `ok` | 1 | equivalent response |
204
- | `blocked` consistent with robots intent | 1 | intentional |
205
- | `challenge-*` | 0.5, tagged **inconclusive** | JS challenge served; real crawler may pass via IP/Web Bot Auth verification, but fetch-only agents cannot |
206
- | `needs-signature` (`403` + `Accept-Signature`) | 0.75 | site requires Web Bot Auth; unsigned agents excluded |
207
- | `paywall-ppc` / `paywall-x402` | 1, info | monetised for crawlers; report price header |
208
- | `license-required` | 1, info | RSL OLP in effect |
209
- | `blocked` while robots allows | 0 | as today |
210
- | reduced content | 0.5 | as today, plus title/H1/JSON-LD hash comparison |
211
-
212
- Probe set: the 12 CORE tokens split into "honors robots.txt" (scored) and "user-triggered fetchers" (reported, not scored). Also probe the always-free paths (`/robots.txt`, `/sitemap.xml`, `/llms.txt`) with the worst-treated UA and flag if they are blocked.
213
-
214
- ### 4.10 `usage-policy` (new) — machine-readable rights signals and their consistency
215
-
216
- Collect: robots.txt `Content-Signal` (incl. `use=`), robots.txt `Content-Usage`, HTTP `Content-Usage`, HTTP `content-signal`, RSL permits/prohibits (from `rsl`), TDMRep (`/.well-known/tdmrep.json`, `tdm-reservation`/`tdm-policy` headers, meta), `noai` meta. Validate syntax per spec (AIPREF dictionary `y/n`; warn on `yes/no` or `ai-train` under `Content-Usage`; Content Signals `yes/no` and `use ∈ immediate|reference|full`; TDMRep JSON array with `location` and `tdm-reservation ∈ 0|1`). Then compute a **consistency matrix** and flag contradictions: `ai-train=no` vs `Content-Usage: train-ai=y`; robots `Disallow` for all training bots vs `ai-train=yes`; RSL prohibits `ai-train` vs Content-Signal `ai-train=yes`; TDMRep reservation 1 with no other training signal (info). Copy must state that only robots.txt tokens are documented as honored by Google, OpenAI, Anthropic and Microsoft; the others are declarations with legal (EU AI Act) rather than technical weight.
217
-
218
- ### 4.11 `http-hygiene` (new) — status honesty and fetch ergonomics
219
-
220
- - Soft-404 probe: `GET /ax-audit-probe-<random>` must return `404`/`410` (warn on `200`, `302→/`, or `403`).
221
- - Redirect hops on the homepage ≤1 (`redirect: 'manual'` follow-up); `http://` → `https://` counted separately (already in `tls-https`).
222
- - Any `429` seen during the audit must carry `Retry-After`.
223
- - Homepage `Content-Type` includes charset; `Content-Language` consistent with `<html lang>`.
224
- - `HEAD /` supported (not `405`).
225
- - Error body sanity: a `404` body should not be an empty 0-byte response.
226
-
227
- ### 4.12 `webmcp` (new, static)
228
-
229
- - Count `<form>`; count forms with both `toolname` and `tooldescription`; fail-level finding for `toolname` without `tooldescription` (mirrors Lighthouse schema validity); per tool form, coverage of `toolparamdescription` on named controls; note `toolautosubmit`.
230
- - Inline/linked script text: `modelContext.registerTool` present → info "imperative WebMCP detected (cannot validate statically)"; `navigator.modelContext` → warn deprecated namespace.
231
- - `<meta http-equiv="origin-trial">` presence → info.
232
- - `applicable: false` when the page has zero forms and no script match. Weight stays 0 through 4.0 (origin trial only).
233
-
234
- ### 4.13 `agent-operability` (new, static heuristics for browser agents)
235
-
236
- Based on web.dev "AI agent site UX" (2026-04-01), Atlas/Claude browser-tool documentation and Lighthouse's agent accessibility subset:
237
-
238
- - `<a>` without `href` or with `href="javascript:"`; `<div>`/`<span>` with `onclick` lacking `role` and `tabindex`.
239
- - `<button>`/`[role=button]` without accessible name (text, `aria-label`, `aria-labelledby`, `title`, `<img alt>`).
240
- - Form controls without `<label for>`, wrapping label, `aria-label` or `aria-labelledby`; missing `autocomplete` on common fields (info).
241
- - `<iframe>` without `title`; `<table>` without `<th>`; `<time>` without `datetime`; heading-level skips.
242
- - `<img>`/`<iframe>`/`<video>` without dimensions (CLS proxy).
243
- - Entry-page blockers: reCAPTCHA/hCaptcha/Turnstile markup, `<meta http-equiv="refresh">`, cookie-consent frameworks when `<main>` text < 100 words.
244
-
245
- Score proposal: proportional to the share of interactive elements that pass; informational tag "heuristic, static HTML only".
246
-
247
- ### 4.14 Smaller extensions
248
-
249
- - **`llms-txt`**: `describedby` links, subpath llms.txt when auditing a non-root URL, `.md` mirror of the homepage (`/index.md`, `/index.html.md`), HEAD-sample ≤20 links (report unreachable / redirected / disallowed-by-robots / `noindex` targets), duplicates, `## Optional` present, size warnings (>50 KB llms.txt, >1 MB llms-full.txt) as info, `/.well-known/llms.txt` mirror as info.
250
- - **`content-negotiation`**: Accept `text/markdown, text/html;q=0.9, */*;q=0.1`; second probe with `User-Agent: Claude-Code/1.0` and default Accept; record `x-markdown-tokens`/`x-original-tokens` and compute savings; `content-signal` header capture; `.md` suffix probe; frontmatter `title`/`canonical_url`; `Link: rel="canonical"` on the markdown response; `/sitemap.md`.
251
- - **`structured-data`**: `dateModified`/`datePublished` (bucketed freshness, future dates flagged), `author` → Person/Organization with `url`/`sameAs`, `Organization.sameAs` count, `headline`/`name` present in visible text, `isAccessibleForFree` vs body length sanity.
252
- - **`http-headers`**: relations `describedby`, `api-catalog`, `ai-catalog`, `service-desc`, `service-doc`, `alternate` + `type="text/markdown"`, `c2pa-manifest`; `X-Llms-Txt`; fix agent-card path in hints; `Accept-Signature` and `Signature-Agent` presence (info).
253
- - **`crawl-efficiency`**: `elapsedMs` TTFB proxy (warn >2 s), estimated tokens of extracted homepage text (warn >25k), `alt-svc` h3 (info).
254
- - **`well-known-ai` (3.x only)**: keep the 5-file score frozen; rewrite hints (nlweb.json/genai.txt marked "no spec found", ai-plugin.json "retired 2024"); add informational probes for `/.well-known/http-message-signatures-directory` (validate JWKS shape if present), `/.well-known/openai-apps-challenge`, `/.well-known/tdmrep.json`, `/.well-known/gpc.json`, `/AGENTS.md`, `/ai.txt`. Retired in 4.0.
255
- - **`commerce-discovery`** (new, conditional): applicable only when `Product`/`Offer` JSON-LD or `/cart|/checkout` links exist. `GET /.well-known/ucp` (fallback `/.well-known/ucp.json`): JSON, `ucp.version` date, `ucp.services` non-empty with resolvable `schema` URLs, `ucp.payment_handlers`, `keys[]`; public (no auth) required. Info only for ACP (`OPTIONS /checkout_sessions`).
256
- - **`auth-discovery`** (new, conditional): applicable when `api-discovery`, `mcp-discovery` or `commerce-discovery` found something. Probe RFC 9728 `/.well-known/oauth-protected-resource`, RFC 8414 `/.well-known/oauth-authorization-server`, `/.well-known/openid-configuration`; require `authorization_servers[]` or `issuer` + `authorization_endpoint` + `token_endpoint`; `code_challenge_methods_supported` includes `S256`; `registration_endpoint` or `client_id_metadata_document_supported` (info).
257
-
258
- Deferred (needs a headless browser or an LLM, out of scope for this package): rendered-vs-static diff, imperative WebMCP enumeration, CLS/INP, task-based agent runs (Ora/Netlify AXIS style), citation share-of-voice.
259
-
260
- ---
261
-
262
- ## 5. Release plan
263
-
264
- ### 3.7.0 — "Correct the map" (no score can go down) — **shipped 2026-09-04**
265
-
266
- 1. Shared infra: `robots-parser.ts`, `waf.ts`, fetcher `method`/`redirect`/`elapsedMs`, `CheckMeta.category`, `WELL_KNOWN_REGISTRY`.
267
- 2. Crawler catalogue refresh (§4.1) with `CORE_AI_CRAWLERS_V3` frozen for scoring.
268
- 3. `agent-card` probe chain (§4.3) and `mcp-discovery` chain (§4.4) with legacy fallbacks; ids `agent-json`/`mcp` kept as aliases for `--checks` and baselines.
269
- 4. `api-discovery` probe chain (§4.5); id `openapi` aliased.
270
- 5. `agent-access` classification (§4.9).
271
- 6. `http-headers` relation broadening and hint fixes; `well-known-ai` hint rewrite + informational probes.
272
- 7. `robots-txt`: `use=` field, case/space tolerance, tiered informational findings, `Content-Usage` and `Agentmap:` parsing (informational).
273
- 8. Docs: checks.md, README table, CHANGELOG; remediation guides for every new anchor.
274
- 9. Tests: 252 new (553 total).
275
-
276
- Delivered in eleven commits. Two things were found by dogfooding rather than by planning: `agent-access` was probing sites with a user agent containing `Google-Extended`, a robots.txt control token no request ever carries; and a speculative probe against a single-page application returned the index shell, which the check reported as a malformed document. Both are fixed and regression-tested.
277
-
278
- ### 3.8.0 — "New signals" (all weight 0) — **shipped**
279
-
280
- `ai-directives`, `usage-policy`, `http-hygiene`, `ai-catalog`, `agent-skills`, `webmcp`, `commerce-discovery`, `auth-discovery` (with `applicable` flag displayed as N/A). `structured-fields.ts`, `frontmatter.ts`, `tokens.ts`. Reporters render categories and N/A. ~90 tests.
281
-
282
- ### 3.9.0 — "Readability depth" — **shipped**
283
-
284
- `agent-operability`; llms.txt v2 extensions and link sampling; content-negotiation extensions; structured-data freshness/author; crawl-efficiency tokens/TTFB; `--probe-mcp` flag. ~50 tests.
285
-
286
- ### 4.0.0 — "Rescore" — **shipped**
287
-
288
- - Weights redistributed (proposal below), `applicable: false` excluded from denominators, categories in every reporter, baseline files carry `schemaVersion: 2` with a migration path for 3.x baselines (missing checks are ignored, not regressions).
289
- - Retire `well-known-ai`; ids `agent-json`, `mcp`, `openapi` removed (aliases dropped); `CORE_AI_CRAWLERS` = 12 tokens.
290
- - llms.txt and `ai:*` meta demoted; access/policy and rendering promoted, following the evidence that fetch-only agents fail on JS-only content and WAF blocks far more often than on missing manifest files.
291
- - New CLI: `--profile docs|api|commerce|default` (sets which conditional checks are forced applicable), `--category` filter, `--fail-on-category access:70`.
292
-
293
- Proposed 4.0 weights (sum 100):
294
-
295
- | Category | Subtotal | Checks and weights |
296
- | --- | --- | --- |
297
- | Content | 33 | html-rendering 10 · structured-data 7 · seo-basics 6 · content-negotiation 6 · sitemap 4 |
298
- | Discovery | 25 | robots-txt 10 · llms-txt 7 · http-headers 5 · meta-tags 3 |
299
- | Access | 23 | agent-access 8 · ai-directives 5 · http-hygiene 4 · crawl-efficiency 3 · tls-https 3 |
300
- | Policy & trust | 9 | usage-policy 4 · security-txt 3 · rsl 2 |
301
- | Protocols (conditional, N/A excluded) | 10 | agent-operability 3 · api-discovery 2 · agent-card 2 · mcp-discovery 2 · agent-skills 1 |
302
- | Informational | 0 | ai-catalog · webmcp · commerce-discovery · auth-discovery |
303
-
304
- ---
305
-
306
- ## 5a. What actually shipped
307
-
308
- | Release | Commits | Tests | Headline |
309
- | --- | --- | --- | --- |
310
- | 3.7.0 | 11 | 553 | Corrected three checks probing paths that were never the standard, and a crawler catalogue containing tokens that do not exist |
311
- | 3.8.0 | 8 | 755 | Eight new checks, N/A reporting, category grouping |
312
- | 3.9.0 | 4 | 828 | agent-operability, llms.txt v2, provenance and freshness, cost in tokens |
313
- | 4.0.0 | 7 | 873 | Rescore, conditional protocol checks, profiles, per-area gates, versioned baselines |
314
-
315
- Five things were found by running the tool rather than by planning it, and none would have surfaced any other way:
316
-
317
- 1. `agent-access` probed sites with a user agent containing `Google-Extended` — a robots.txt control token no request ever carries, so the probe tested nothing.
318
- 2. A speculative probe against a single-page application returned the index shell, and the check reported a malformed document on a site that had none.
319
- 3. `commerce-discovery` told a SaaS site to build a commerce integration because it prices its plans with an `Offer`.
320
- 4. The first draft of the 4.0 rescore produced a distribution nobody asked for, because per-check `meta.weight` values silently won over the central map.
321
- 5. `structured-data` scored a well-marked-up site at 0 because its pattern required `type` to be the first attribute on the script tag. Next.js emits `id` first, so a whole class of sites was being told to add markup they already had.
322
-
323
- Two items from the plan were **not** built, deliberately:
324
-
325
- - **`--probe-mcp`** live handshake. A POST to a stranger's MCP endpoint is a side effect an audit should not have by default, and behind a flag it would be exercised too rarely to stay correct.
326
- - **Dropping the renamed check ids.** The plan called for removing `agent-json`, `mcp` and `openapi` at 4.0. They cost one line each and live in CI configs, so they stay as permanent aliases.
327
-
328
- ## 6. Cross-repo follow-ups
329
-
330
- - **ax-init** generates `/.well-known/agent.json` and `/.well-known/mcp.json`: switch to `agent-card.json` (A2A 1.0 shape) and a server card at `/.well-known/mcp/server-card.json` plus `/.well-known/ai-catalog.json`; add `Content-Signal` with `use=`, `Link` headers (`describedby`, `api-catalog`), `.md` mirrors.
331
- - **ax-skill** (`ax.md`) documents the old paths and the `ai:*` meta namespace as recommendations; update to the 3.7 reality and add the robots-meta AI directive semantics.
332
- - **Remediation guides** at `axrush.com/guides/<check-id>`: one page per new check id, plus new anchors listed in each check's source. This is the largest non-code deliverable; ship guides with each minor release.
333
-
334
- ---
335
-
336
- ## 7. Risks and guardrails
337
-
338
- - **Draft standards churn** (MCP server card path, ARD vs ai-catalog, WebMCP, AIPREF tokens). Guardrail: label every finding with standing (registered / convention / draft), keep these checks at weight 0 through 4.0, and centralise paths in constants so a rename is a one-line change.
339
- - **False positives from spoofed-UA probes**: WAFs with IP or Web Bot Auth verification reject ax-audit while admitting the real bot. Guardrail: `challenge` and `needs-signature` outcomes are reported as inconclusive with the exact header seen, never as "blocks AI".
340
- - **Penalising features a site does not offer** (APIs, commerce, skills). Guardrail: `applicable: false` + profiles; N/A never lowers the score.
341
- - **Over-weighting files nobody fetches** (llms.txt, manifests). Guardrail: 4.0 weights favour rendering, access and policy; copy states which consumers actually read each file.
342
- - **Heuristic checks** (`agent-operability`, extractability): informational, thresholds in constants, documented as static approximations.
343
- - **Baseline compatibility**: renamed ids must map old → new in `diffBaseline` so `--fail-on-regression` does not fire on a rename.
344
-
345
- ---
346
-
347
- ## 8. Evidence index (primary sources checked 2026-09-04)
348
-
349
- A2A releases and spec · https://github.com/a2aproject/A2A/releases · https://a2a-protocol.org/latest/specification/
350
- MCP 2026-07-28 changelog and authorization · https://modelcontextprotocol.io/specification/2026-07-28/changelog · https://modelcontextprotocol.io/specification/2026-07-28/basic/authorization/authorization-server-discovery
351
- MCP server card SEP-2127 and extension repo · https://github.com/modelcontextprotocol/modelcontextprotocol/pull/2127 · https://github.com/modelcontextprotocol/experimental-ext-server-card
352
- ai-catalog / ARD · https://ai-catalog.io/ · https://agenticresourcediscovery.org/spec/
353
- WebMCP · https://webmachinelearning.github.io/webmcp/ · https://developer.chrome.com/docs/ai/webmcp/declarative-api · https://developer.chrome.com/docs/lighthouse/agentic-browsing
354
- Agent Skills · https://github.com/cloudflare/agent-skills-discovery-rfc · https://agentskills.io/specification · https://www.mintlify.com/docs/ai/skillmd
355
- UCP / ACP · https://developers.google.com/merchant/ucp/guides/ucp-profile · https://ucp.dev/2026-08-25/specification/overview/ · https://developers.openai.com/commerce/specs/checkout
356
- llms.txt v2 · https://llmstxt.org/ · Google AI optimization guide https://developers.google.com/search/docs/fundamentals/ai-optimization-guide
357
- Markdown for Agents · https://developers.cloudflare.com/fundamentals/reference/markdown-for-agents/ · https://vercel.com/docs/agent-resources/markdown-access · https://vercel.com/kb/guide/agent-readability-spec
358
- RFC 9727 / 8631 / 9728 / 8414 · https://www.rfc-editor.org/rfc/rfc9727.html · https://www.rfc-editor.org/rfc/rfc8631.html
359
- IANA well-known registry · https://www.iana.org/assignments/well-known-uris/well-known-uris.xhtml
360
- AIPREF · https://datatracker.ietf.org/wg/aipref/documents/ · Web Bot Auth · https://datatracker.ietf.org/group/webbotauth/documents/ · https://developers.cloudflare.com/bots/reference/bot-verification/web-bot-auth/
361
- Content Signals `use=` · https://blog.cloudflare.com/content-independence-day-ai-options/ · https://developers.cloudflare.com/bots/additional-configurations/managed-robots-txt/
362
- Pay-per-crawl · https://developers.cloudflare.com/ai-crawl-control/features/pay-per-crawl/ · AWS WAF monetization https://docs.aws.amazon.com/waf/latest/developerguide/waf-ai-traffic-monetization-how-it-works.html
363
- WAF signatures · https://developers.cloudflare.com/cloudflare-challenges/challenge-types/challenge-pages/detect-response/ · https://docs.aws.amazon.com/waf/latest/developerguide/waf-captcha-and-challenge-actions.html
364
- RSL errata · https://rslstandard.org/rsl/errata · TDMRep · https://www.w3.org/community/reports/tdmrep/CG-FINAL-tdmrep-20240510/
365
- Robots meta semantics · https://developers.google.com/search/docs/crawling-indexing/robots-meta-tag · https://developers.google.com/search/docs/appearance/ai-features · https://blogs.bing.com/webmaster/september-2023/Announcing-new-options-for-webmasters-to-control-usage-of-their-content-in-Bing-Chat
366
- Crawler docs · https://developers.openai.com/api/docs/bots · https://support.claude.com/en/articles/8896518 · https://developers.google.com/crawling/docs/crawlers-fetchers/google-agent · https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/ · https://developer.amazon.com/amazonbot · https://docs.mistral.ai/robots/ · https://docs.perplexity.ai/docs/resources/perplexity-crawlers · https://developers.cloudflare.com/ai-crawl-control/reference/bots/
367
- Competitors · https://blog.cloudflare.com/agent-readiness/ · https://www.agentready.org/ · https://is-agentic.com · https://agent-ready.dev/ · https://github.com/addyosmani/agentic-seo · https://ahrefs.com/blog/llmstxt-study/