pagesight 0.12.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/package.json +1 -1
  2. package/src/tools/ai.ts +95 -3
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pagesight",
3
- "version": "0.12.2",
3
+ "version": "0.13.0",
4
4
  "description": "See your site the way search engines and AI see it.",
5
5
  "keywords": [
6
6
  "seo",
package/src/tools/ai.ts CHANGED
@@ -2,6 +2,85 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
2
  import { z } from "zod";
3
3
  import { auditAiCrawlers, type CrawlerStatus, fetchRobotsTxt, isAllowed, type RobotsTxt } from "../lib/robots.js";
4
4
 
5
+ // --- llms.txt detection ---
6
+
7
+ interface LlmsTxtResult {
8
+ exists: boolean;
9
+ size: number | null;
10
+ firstLine: string | null;
11
+ }
12
+
13
+ async function checkLlmsTxt(origin: string, path: string): Promise<LlmsTxtResult> {
14
+ try {
15
+ const res = await fetch(`${origin}${path}`, {
16
+ headers: { "User-Agent": "Pagesight/1.0" },
17
+ redirect: "follow",
18
+ });
19
+ if (!res.ok) return { exists: false, size: null, firstLine: null };
20
+ const text = await res.text();
21
+ const firstLine =
22
+ text
23
+ .split("\n")
24
+ .find((l) => l.trim().length > 0)
25
+ ?.trim() ?? null;
26
+ return { exists: true, size: text.length, firstLine };
27
+ } catch {
28
+ return { exists: false, size: null, firstLine: null };
29
+ }
30
+ }
31
+
32
+ function formatLlmsTxt(llmsTxt: LlmsTxtResult, llmsFullTxt: LlmsTxtResult): string[] {
33
+ const lines: string[] = ["--- LLM Visibility ---", ""];
34
+
35
+ if (llmsTxt.exists) {
36
+ const sizeKB = llmsTxt.size ? `${(llmsTxt.size / 1024).toFixed(1)} KB` : "unknown size";
37
+ lines.push(`llms.txt: FOUND (${sizeKB})`);
38
+ if (llmsTxt.firstLine) lines.push(` ${llmsTxt.firstLine}`);
39
+ } else {
40
+ lines.push("llms.txt: NOT FOUND — consider adding one for AI-friendly documentation");
41
+ }
42
+
43
+ if (llmsFullTxt.exists) {
44
+ const sizeKB = llmsFullTxt.size ? `${(llmsFullTxt.size / 1024).toFixed(1)} KB` : "unknown size";
45
+ lines.push(`llms-full.txt: FOUND (${sizeKB})`);
46
+ } else if (llmsTxt.exists) {
47
+ lines.push("llms-full.txt: NOT FOUND — consider adding full docs for AI context windows");
48
+ }
49
+
50
+ return lines;
51
+ }
52
+
53
+ // --- Category summary ---
54
+
55
+ function formatCategorySummary(crawlers: CrawlerStatus[]): string[] {
56
+ const categoryOrder = ["Training", "Search", "Assistant", "Agent", "Other"];
57
+ const summary = new Map<string, { blocked: number; allowed: number }>();
58
+
59
+ for (const bot of crawlers) {
60
+ const cat = normalizeCategory(bot.category);
61
+ const existing = summary.get(cat) ?? { blocked: 0, allowed: 0 };
62
+ if (bot.allowed) existing.allowed++;
63
+ else existing.blocked++;
64
+ summary.set(cat, existing);
65
+ }
66
+
67
+ const lines: string[] = ["", "--- AI Crawler Summary by Category ---", ""];
68
+ for (const cat of categoryOrder) {
69
+ const s = summary.get(cat);
70
+ if (!s) continue;
71
+ const total = s.blocked + s.allowed;
72
+ if (s.blocked === 0) {
73
+ lines.push(`${cat}: all ${total} allowed`);
74
+ } else if (s.allowed === 0) {
75
+ lines.push(`${cat}: all ${total} BLOCKED`);
76
+ } else {
77
+ lines.push(`${cat}: ${s.blocked} blocked, ${s.allowed} allowed (of ${total})`);
78
+ }
79
+ }
80
+
81
+ return lines;
82
+ }
83
+
5
84
  // Normalize the messy registry categories into clean buckets
6
85
  function normalizeCategory(raw: string): string {
7
86
  const lower = raw.toLowerCase();
@@ -120,7 +199,7 @@ function formatRobotsAudit(origin: string, robots: RobotsTxt, statusCode: number
120
199
  export function registerAiTool(server: McpServer): void {
121
200
  server.tool(
122
201
  "ai",
123
- "Analyze your site's AI visibility. Fetches robots.txt, audits which AI crawlers (training, search, assistant, agent) can access your content, validates syntax per RFC 9309, and checks specific paths. Shows how AI systems see your site.",
202
+ "Analyze your site's AI visibility. Audits AI crawler access (training, search, assistant, agent) via robots.txt, checks for llms.txt and llms-full.txt, validates syntax per RFC 9309, and tests specific path access. Shows how AI systems see your site.",
124
203
  {
125
204
  url: z
126
205
  .string()
@@ -159,8 +238,21 @@ export function registerAiTool(server: McpServer): void {
159
238
  return { content: [{ type: "text", text: lines.join("\n") }] };
160
239
  }
161
240
 
162
- const crawlers = await auditAiCrawlers(robotsTxt);
163
- return { content: [{ type: "text", text: formatRobotsAudit(origin, robotsTxt, statusCode, crawlers) }] };
241
+ const [crawlers, llmsTxt, llmsFullTxt] = await Promise.all([
242
+ auditAiCrawlers(robotsTxt),
243
+ checkLlmsTxt(origin, "/llms.txt"),
244
+ checkLlmsTxt(origin, "/llms-full.txt"),
245
+ ]);
246
+
247
+ const output: string[] = [formatRobotsAudit(origin, robotsTxt, statusCode, crawlers)];
248
+
249
+ if (crawlers.length > 0) {
250
+ output.push(...formatCategorySummary(crawlers));
251
+ }
252
+
253
+ output.push("", ...formatLlmsTxt(llmsTxt, llmsFullTxt));
254
+
255
+ return { content: [{ type: "text", text: output.join("\n") }] };
164
256
  } catch (err) {
165
257
  const msg = err instanceof Error ? err.message : String(err);
166
258
  return { content: [{ type: "text", text: `Error analyzing robots.txt: ${msg}` }] };