pagesight 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pagesight",
3
- "version": "0.12.1",
3
+ "version": "0.13.0",
4
4
  "description": "See your site the way search engines and AI see it.",
5
5
  "keywords": [
6
6
  "seo",
package/src/index.ts CHANGED
@@ -1,17 +1,12 @@
1
1
  #!/usr/bin/env bun
2
2
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
3
3
  import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
4
+ import { registerAiTool } from "./tools/ai.js";
4
5
  import { registerAuditTool } from "./tools/audit.js";
5
- import { registerCruxTool } from "./tools/crux.js";
6
- import { registerInspectTool } from "./tools/inspect.js";
7
- import { registerLinksTool } from "./tools/links.js";
8
- import { registerMetatagsTool } from "./tools/metatags.js";
9
- import { registerPagespeedTool } from "./tools/pagespeed.js";
10
- import { registerPerformanceTool } from "./tools/performance.js";
11
- import { registerRobotsTool } from "./tools/robots.js";
12
- import { registerSampleInspectTool } from "./tools/sample-inspect.js";
6
+ import { registerPageTool } from "./tools/page.js";
7
+ import { registerSearchTool } from "./tools/search.js";
13
8
  import { registerSetupTool } from "./tools/setup.js";
14
- import { registerSitemapsTool } from "./tools/sitemaps.js";
9
+ import { registerSpeedTool } from "./tools/speed.js";
15
10
 
16
11
  const pkg = await Bun.file(new URL("../package.json", import.meta.url)).json();
17
12
 
@@ -20,17 +15,12 @@ const server = new McpServer({
20
15
  version: pkg.version,
21
16
  });
22
17
 
18
+ registerAiTool(server);
23
19
  registerAuditTool(server);
24
- registerCruxTool(server);
25
- registerInspectTool(server);
26
- registerLinksTool(server);
27
- registerMetatagsTool(server);
28
- registerPagespeedTool(server);
29
- registerPerformanceTool(server);
30
- registerRobotsTool(server);
31
- registerSampleInspectTool(server);
32
- registerSitemapsTool(server);
20
+ registerPageTool(server);
21
+ registerSearchTool(server);
33
22
  registerSetupTool(server);
23
+ registerSpeedTool(server);
34
24
 
35
25
  const transport = new StdioServerTransport();
36
26
  await server.connect(transport);
@@ -0,0 +1,106 @@
1
+ import { inspectUrl } from "./gsc.js";
2
+
3
+ export interface SitemapParseResult {
4
+ urls: string[];
5
+ isSitemapIndex: boolean;
6
+ childSitemaps: string[];
7
+ }
8
+
9
+ export function parseSitemapXml(xml: string): SitemapParseResult {
10
+ const urls: string[] = [];
11
+ const childSitemaps: string[] = [];
12
+
13
+ const isSitemapIndex = /<sitemapindex/i.test(xml);
14
+
15
+ if (isSitemapIndex) {
16
+ for (const m of xml.matchAll(/<sitemap[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/sitemap>/gi)) {
17
+ childSitemaps.push(m[1].trim());
18
+ }
19
+ } else {
20
+ for (const m of xml.matchAll(/<url[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/url>/gi)) {
21
+ urls.push(m[1].trim());
22
+ }
23
+ }
24
+
25
+ return { urls, isSitemapIndex, childSitemaps };
26
+ }
27
+
28
+ export async function fetchSitemap(sitemapUrl: string): Promise<SitemapParseResult> {
29
+ const res = await fetch(sitemapUrl, {
30
+ headers: { "User-Agent": "Pagesight/1.0" },
31
+ });
32
+
33
+ if (!res.ok) {
34
+ throw new Error(`Failed to fetch sitemap ${sitemapUrl}: HTTP ${res.status}`);
35
+ }
36
+
37
+ const xml = await res.text();
38
+ return parseSitemapXml(xml);
39
+ }
40
+
41
+ export function sampleUrls(urls: string[], count: number, strategy: string): string[] {
42
+ if (urls.length <= count) return [...urls];
43
+
44
+ if (strategy === "first") {
45
+ return urls.slice(0, count);
46
+ }
47
+
48
+ if (strategy === "spread") {
49
+ const step = Math.floor(urls.length / count);
50
+ const sampled: string[] = [];
51
+ for (let i = 0; i < count; i++) {
52
+ sampled.push(urls[i * step]);
53
+ }
54
+ return sampled;
55
+ }
56
+
57
+ // random (default)
58
+ const shuffled = [...urls];
59
+ for (let i = shuffled.length - 1; i > 0; i--) {
60
+ const j = Math.floor(Math.random() * (i + 1));
61
+ [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]];
62
+ }
63
+ return shuffled.slice(0, count);
64
+ }
65
+
66
+ export interface InspectionSummary {
67
+ url: string;
68
+ verdict: string;
69
+ coverageState: string;
70
+ pageFetchState: string;
71
+ robotsTxtState: string;
72
+ indexingState: string;
73
+ lastCrawlTime: string | null;
74
+ googleCanonical: string | null;
75
+ error: string | null;
76
+ }
77
+
78
+ export async function inspectSingle(url: string, siteUrl: string): Promise<InspectionSummary> {
79
+ try {
80
+ const r = await inspectUrl(url, siteUrl);
81
+ const idx = r.indexStatusResult;
82
+ return {
83
+ url,
84
+ verdict: idx.verdict,
85
+ coverageState: idx.coverageState,
86
+ pageFetchState: idx.pageFetchState,
87
+ robotsTxtState: idx.robotsTxtState,
88
+ indexingState: idx.indexingState,
89
+ lastCrawlTime: idx.lastCrawlTime ?? null,
90
+ googleCanonical: idx.googleCanonical ?? null,
91
+ error: null,
92
+ };
93
+ } catch (err) {
94
+ return {
95
+ url,
96
+ verdict: "ERROR",
97
+ coverageState: "ERROR",
98
+ pageFetchState: "ERROR",
99
+ robotsTxtState: "ERROR",
100
+ indexingState: "ERROR",
101
+ lastCrawlTime: null,
102
+ googleCanonical: null,
103
+ error: err instanceof Error ? err.message : String(err),
104
+ };
105
+ }
106
+ }
@@ -2,6 +2,85 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
2
  import { z } from "zod";
3
3
  import { auditAiCrawlers, type CrawlerStatus, fetchRobotsTxt, isAllowed, type RobotsTxt } from "../lib/robots.js";
4
4
 
5
+ // --- llms.txt detection ---
6
+
7
+ interface LlmsTxtResult {
8
+ exists: boolean;
9
+ size: number | null;
10
+ firstLine: string | null;
11
+ }
12
+
13
+ async function checkLlmsTxt(origin: string, path: string): Promise<LlmsTxtResult> {
14
+ try {
15
+ const res = await fetch(`${origin}${path}`, {
16
+ headers: { "User-Agent": "Pagesight/1.0" },
17
+ redirect: "follow",
18
+ });
19
+ if (!res.ok) return { exists: false, size: null, firstLine: null };
20
+ const text = await res.text();
21
+ const firstLine =
22
+ text
23
+ .split("\n")
24
+ .find((l) => l.trim().length > 0)
25
+ ?.trim() ?? null;
26
+ return { exists: true, size: text.length, firstLine };
27
+ } catch {
28
+ return { exists: false, size: null, firstLine: null };
29
+ }
30
+ }
31
+
32
+ function formatLlmsTxt(llmsTxt: LlmsTxtResult, llmsFullTxt: LlmsTxtResult): string[] {
33
+ const lines: string[] = ["--- LLM Visibility ---", ""];
34
+
35
+ if (llmsTxt.exists) {
36
+ const sizeKB = llmsTxt.size ? `${(llmsTxt.size / 1024).toFixed(1)} KB` : "unknown size";
37
+ lines.push(`llms.txt: FOUND (${sizeKB})`);
38
+ if (llmsTxt.firstLine) lines.push(` ${llmsTxt.firstLine}`);
39
+ } else {
40
+ lines.push("llms.txt: NOT FOUND — consider adding one for AI-friendly documentation");
41
+ }
42
+
43
+ if (llmsFullTxt.exists) {
44
+ const sizeKB = llmsFullTxt.size ? `${(llmsFullTxt.size / 1024).toFixed(1)} KB` : "unknown size";
45
+ lines.push(`llms-full.txt: FOUND (${sizeKB})`);
46
+ } else if (llmsTxt.exists) {
47
+ lines.push("llms-full.txt: NOT FOUND — consider adding full docs for AI context windows");
48
+ }
49
+
50
+ return lines;
51
+ }
52
+
53
+ // --- Category summary ---
54
+
55
+ function formatCategorySummary(crawlers: CrawlerStatus[]): string[] {
56
+ const categoryOrder = ["Training", "Search", "Assistant", "Agent", "Other"];
57
+ const summary = new Map<string, { blocked: number; allowed: number }>();
58
+
59
+ for (const bot of crawlers) {
60
+ const cat = normalizeCategory(bot.category);
61
+ const existing = summary.get(cat) ?? { blocked: 0, allowed: 0 };
62
+ if (bot.allowed) existing.allowed++;
63
+ else existing.blocked++;
64
+ summary.set(cat, existing);
65
+ }
66
+
67
+ const lines: string[] = ["", "--- AI Crawler Summary by Category ---", ""];
68
+ for (const cat of categoryOrder) {
69
+ const s = summary.get(cat);
70
+ if (!s) continue;
71
+ const total = s.blocked + s.allowed;
72
+ if (s.blocked === 0) {
73
+ lines.push(`${cat}: all ${total} allowed`);
74
+ } else if (s.allowed === 0) {
75
+ lines.push(`${cat}: all ${total} BLOCKED`);
76
+ } else {
77
+ lines.push(`${cat}: ${s.blocked} blocked, ${s.allowed} allowed (of ${total})`);
78
+ }
79
+ }
80
+
81
+ return lines;
82
+ }
83
+
5
84
  // Normalize the messy registry categories into clean buckets
6
85
  function normalizeCategory(raw: string): string {
7
86
  const lower = raw.toLowerCase();
@@ -117,10 +196,10 @@ function formatRobotsAudit(origin: string, robots: RobotsTxt, statusCode: number
117
196
  return lines.join("\n");
118
197
  }
119
198
 
120
- export function registerRobotsTool(server: McpServer): void {
199
+ export function registerAiTool(server: McpServer): void {
121
200
  server.tool(
122
- "robots",
123
- "Fetch and analyze a site's robots.txt. Validates syntax per RFC 9309, audits AI crawler access (139+ bots), lists sitemaps. Shows blocked bots in detail, summarizes allowed.",
201
+ "ai",
202
+ "Analyze your site's AI visibility. Audits AI crawler access (training, search, assistant, agent) via robots.txt, checks for llms.txt and llms-full.txt, validates syntax per RFC 9309, and tests specific path access. Shows how AI systems see your site.",
124
203
  {
125
204
  url: z
126
205
  .string()
@@ -159,8 +238,21 @@ export function registerRobotsTool(server: McpServer): void {
159
238
  return { content: [{ type: "text", text: lines.join("\n") }] };
160
239
  }
161
240
 
162
- const crawlers = await auditAiCrawlers(robotsTxt);
163
- return { content: [{ type: "text", text: formatRobotsAudit(origin, robotsTxt, statusCode, crawlers) }] };
241
+ const [crawlers, llmsTxt, llmsFullTxt] = await Promise.all([
242
+ auditAiCrawlers(robotsTxt),
243
+ checkLlmsTxt(origin, "/llms.txt"),
244
+ checkLlmsTxt(origin, "/llms-full.txt"),
245
+ ]);
246
+
247
+ const output: string[] = [formatRobotsAudit(origin, robotsTxt, statusCode, crawlers)];
248
+
249
+ if (crawlers.length > 0) {
250
+ output.push(...formatCategorySummary(crawlers));
251
+ }
252
+
253
+ output.push("", ...formatLlmsTxt(llmsTxt, llmsFullTxt));
254
+
255
+ return { content: [{ type: "text", text: output.join("\n") }] };
164
256
  } catch (err) {
165
257
  const msg = err instanceof Error ? err.message : String(err);
166
258
  return { content: [{ type: "text", text: `Error analyzing robots.txt: ${msg}` }] };
@@ -3,7 +3,7 @@ import { z } from "zod";
3
3
  import { inspectUrl, listSitemaps } from "../lib/gsc.js";
4
4
  import { type PsiCategoryType, type PsiResult, runPagespeed } from "../lib/psi.js";
5
5
  import { auditAiCrawlers, fetchRobotsTxt, isAllowed } from "../lib/robots.js";
6
- import { fetchSitemap, inspectSingle, sampleUrls } from "./sample-inspect.js";
6
+ import { fetchSitemap, inspectSingle, sampleUrls } from "../lib/sitemap.js";
7
7
 
8
8
  type Severity = "HIGH" | "MEDIUM" | "LOW";
9
9