pagesight 0.12.1 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/index.ts +8 -18
- package/src/lib/sitemap.ts +106 -0
- package/src/tools/{robots.ts → ai.ts} +97 -5
- package/src/tools/audit.ts +1 -1
- package/src/tools/{metatags.ts → page.ts} +480 -131
- package/src/tools/search.ts +685 -0
- package/src/tools/{pagespeed.ts → speed.ts} +378 -57
- package/src/tools/crux.ts +0 -296
- package/src/tools/inspect.ts +0 -121
- package/src/tools/links.ts +0 -183
- package/src/tools/performance.ts +0 -325
- package/src/tools/sample-inspect.ts +0 -284
- package/src/tools/sitemaps.ts +0 -112
package/package.json
CHANGED
package/src/index.ts
CHANGED
|
@@ -1,17 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env bun
|
|
2
2
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
3
3
|
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
4
|
+
import { registerAiTool } from "./tools/ai.js";
|
|
4
5
|
import { registerAuditTool } from "./tools/audit.js";
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import { registerLinksTool } from "./tools/links.js";
|
|
8
|
-
import { registerMetatagsTool } from "./tools/metatags.js";
|
|
9
|
-
import { registerPagespeedTool } from "./tools/pagespeed.js";
|
|
10
|
-
import { registerPerformanceTool } from "./tools/performance.js";
|
|
11
|
-
import { registerRobotsTool } from "./tools/robots.js";
|
|
12
|
-
import { registerSampleInspectTool } from "./tools/sample-inspect.js";
|
|
6
|
+
import { registerPageTool } from "./tools/page.js";
|
|
7
|
+
import { registerSearchTool } from "./tools/search.js";
|
|
13
8
|
import { registerSetupTool } from "./tools/setup.js";
|
|
14
|
-
import {
|
|
9
|
+
import { registerSpeedTool } from "./tools/speed.js";
|
|
15
10
|
|
|
16
11
|
const pkg = await Bun.file(new URL("../package.json", import.meta.url)).json();
|
|
17
12
|
|
|
@@ -20,17 +15,12 @@ const server = new McpServer({
|
|
|
20
15
|
version: pkg.version,
|
|
21
16
|
});
|
|
22
17
|
|
|
18
|
+
registerAiTool(server);
|
|
23
19
|
registerAuditTool(server);
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
registerLinksTool(server);
|
|
27
|
-
registerMetatagsTool(server);
|
|
28
|
-
registerPagespeedTool(server);
|
|
29
|
-
registerPerformanceTool(server);
|
|
30
|
-
registerRobotsTool(server);
|
|
31
|
-
registerSampleInspectTool(server);
|
|
32
|
-
registerSitemapsTool(server);
|
|
20
|
+
registerPageTool(server);
|
|
21
|
+
registerSearchTool(server);
|
|
33
22
|
registerSetupTool(server);
|
|
23
|
+
registerSpeedTool(server);
|
|
34
24
|
|
|
35
25
|
const transport = new StdioServerTransport();
|
|
36
26
|
await server.connect(transport);
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { inspectUrl } from "./gsc.js";
|
|
2
|
+
|
|
3
|
+
export interface SitemapParseResult {
|
|
4
|
+
urls: string[];
|
|
5
|
+
isSitemapIndex: boolean;
|
|
6
|
+
childSitemaps: string[];
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export function parseSitemapXml(xml: string): SitemapParseResult {
|
|
10
|
+
const urls: string[] = [];
|
|
11
|
+
const childSitemaps: string[] = [];
|
|
12
|
+
|
|
13
|
+
const isSitemapIndex = /<sitemapindex/i.test(xml);
|
|
14
|
+
|
|
15
|
+
if (isSitemapIndex) {
|
|
16
|
+
for (const m of xml.matchAll(/<sitemap[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/sitemap>/gi)) {
|
|
17
|
+
childSitemaps.push(m[1].trim());
|
|
18
|
+
}
|
|
19
|
+
} else {
|
|
20
|
+
for (const m of xml.matchAll(/<url[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/url>/gi)) {
|
|
21
|
+
urls.push(m[1].trim());
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
return { urls, isSitemapIndex, childSitemaps };
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export async function fetchSitemap(sitemapUrl: string): Promise<SitemapParseResult> {
|
|
29
|
+
const res = await fetch(sitemapUrl, {
|
|
30
|
+
headers: { "User-Agent": "Pagesight/1.0" },
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
if (!res.ok) {
|
|
34
|
+
throw new Error(`Failed to fetch sitemap ${sitemapUrl}: HTTP ${res.status}`);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const xml = await res.text();
|
|
38
|
+
return parseSitemapXml(xml);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export function sampleUrls(urls: string[], count: number, strategy: string): string[] {
|
|
42
|
+
if (urls.length <= count) return [...urls];
|
|
43
|
+
|
|
44
|
+
if (strategy === "first") {
|
|
45
|
+
return urls.slice(0, count);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
if (strategy === "spread") {
|
|
49
|
+
const step = Math.floor(urls.length / count);
|
|
50
|
+
const sampled: string[] = [];
|
|
51
|
+
for (let i = 0; i < count; i++) {
|
|
52
|
+
sampled.push(urls[i * step]);
|
|
53
|
+
}
|
|
54
|
+
return sampled;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// random (default)
|
|
58
|
+
const shuffled = [...urls];
|
|
59
|
+
for (let i = shuffled.length - 1; i > 0; i--) {
|
|
60
|
+
const j = Math.floor(Math.random() * (i + 1));
|
|
61
|
+
[shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]];
|
|
62
|
+
}
|
|
63
|
+
return shuffled.slice(0, count);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export interface InspectionSummary {
|
|
67
|
+
url: string;
|
|
68
|
+
verdict: string;
|
|
69
|
+
coverageState: string;
|
|
70
|
+
pageFetchState: string;
|
|
71
|
+
robotsTxtState: string;
|
|
72
|
+
indexingState: string;
|
|
73
|
+
lastCrawlTime: string | null;
|
|
74
|
+
googleCanonical: string | null;
|
|
75
|
+
error: string | null;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export async function inspectSingle(url: string, siteUrl: string): Promise<InspectionSummary> {
|
|
79
|
+
try {
|
|
80
|
+
const r = await inspectUrl(url, siteUrl);
|
|
81
|
+
const idx = r.indexStatusResult;
|
|
82
|
+
return {
|
|
83
|
+
url,
|
|
84
|
+
verdict: idx.verdict,
|
|
85
|
+
coverageState: idx.coverageState,
|
|
86
|
+
pageFetchState: idx.pageFetchState,
|
|
87
|
+
robotsTxtState: idx.robotsTxtState,
|
|
88
|
+
indexingState: idx.indexingState,
|
|
89
|
+
lastCrawlTime: idx.lastCrawlTime ?? null,
|
|
90
|
+
googleCanonical: idx.googleCanonical ?? null,
|
|
91
|
+
error: null,
|
|
92
|
+
};
|
|
93
|
+
} catch (err) {
|
|
94
|
+
return {
|
|
95
|
+
url,
|
|
96
|
+
verdict: "ERROR",
|
|
97
|
+
coverageState: "ERROR",
|
|
98
|
+
pageFetchState: "ERROR",
|
|
99
|
+
robotsTxtState: "ERROR",
|
|
100
|
+
indexingState: "ERROR",
|
|
101
|
+
lastCrawlTime: null,
|
|
102
|
+
googleCanonical: null,
|
|
103
|
+
error: err instanceof Error ? err.message : String(err),
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
}
|
|
@@ -2,6 +2,85 @@ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
|
2
2
|
import { z } from "zod";
|
|
3
3
|
import { auditAiCrawlers, type CrawlerStatus, fetchRobotsTxt, isAllowed, type RobotsTxt } from "../lib/robots.js";
|
|
4
4
|
|
|
5
|
+
// --- llms.txt detection ---
|
|
6
|
+
|
|
7
|
+
interface LlmsTxtResult {
|
|
8
|
+
exists: boolean;
|
|
9
|
+
size: number | null;
|
|
10
|
+
firstLine: string | null;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
async function checkLlmsTxt(origin: string, path: string): Promise<LlmsTxtResult> {
|
|
14
|
+
try {
|
|
15
|
+
const res = await fetch(`${origin}${path}`, {
|
|
16
|
+
headers: { "User-Agent": "Pagesight/1.0" },
|
|
17
|
+
redirect: "follow",
|
|
18
|
+
});
|
|
19
|
+
if (!res.ok) return { exists: false, size: null, firstLine: null };
|
|
20
|
+
const text = await res.text();
|
|
21
|
+
const firstLine =
|
|
22
|
+
text
|
|
23
|
+
.split("\n")
|
|
24
|
+
.find((l) => l.trim().length > 0)
|
|
25
|
+
?.trim() ?? null;
|
|
26
|
+
return { exists: true, size: text.length, firstLine };
|
|
27
|
+
} catch {
|
|
28
|
+
return { exists: false, size: null, firstLine: null };
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function formatLlmsTxt(llmsTxt: LlmsTxtResult, llmsFullTxt: LlmsTxtResult): string[] {
|
|
33
|
+
const lines: string[] = ["--- LLM Visibility ---", ""];
|
|
34
|
+
|
|
35
|
+
if (llmsTxt.exists) {
|
|
36
|
+
const sizeKB = llmsTxt.size ? `${(llmsTxt.size / 1024).toFixed(1)} KB` : "unknown size";
|
|
37
|
+
lines.push(`llms.txt: FOUND (${sizeKB})`);
|
|
38
|
+
if (llmsTxt.firstLine) lines.push(` ${llmsTxt.firstLine}`);
|
|
39
|
+
} else {
|
|
40
|
+
lines.push("llms.txt: NOT FOUND — consider adding one for AI-friendly documentation");
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
if (llmsFullTxt.exists) {
|
|
44
|
+
const sizeKB = llmsFullTxt.size ? `${(llmsFullTxt.size / 1024).toFixed(1)} KB` : "unknown size";
|
|
45
|
+
lines.push(`llms-full.txt: FOUND (${sizeKB})`);
|
|
46
|
+
} else if (llmsTxt.exists) {
|
|
47
|
+
lines.push("llms-full.txt: NOT FOUND — consider adding full docs for AI context windows");
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
return lines;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// --- Category summary ---
|
|
54
|
+
|
|
55
|
+
function formatCategorySummary(crawlers: CrawlerStatus[]): string[] {
|
|
56
|
+
const categoryOrder = ["Training", "Search", "Assistant", "Agent", "Other"];
|
|
57
|
+
const summary = new Map<string, { blocked: number; allowed: number }>();
|
|
58
|
+
|
|
59
|
+
for (const bot of crawlers) {
|
|
60
|
+
const cat = normalizeCategory(bot.category);
|
|
61
|
+
const existing = summary.get(cat) ?? { blocked: 0, allowed: 0 };
|
|
62
|
+
if (bot.allowed) existing.allowed++;
|
|
63
|
+
else existing.blocked++;
|
|
64
|
+
summary.set(cat, existing);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const lines: string[] = ["", "--- AI Crawler Summary by Category ---", ""];
|
|
68
|
+
for (const cat of categoryOrder) {
|
|
69
|
+
const s = summary.get(cat);
|
|
70
|
+
if (!s) continue;
|
|
71
|
+
const total = s.blocked + s.allowed;
|
|
72
|
+
if (s.blocked === 0) {
|
|
73
|
+
lines.push(`${cat}: all ${total} allowed`);
|
|
74
|
+
} else if (s.allowed === 0) {
|
|
75
|
+
lines.push(`${cat}: all ${total} BLOCKED`);
|
|
76
|
+
} else {
|
|
77
|
+
lines.push(`${cat}: ${s.blocked} blocked, ${s.allowed} allowed (of ${total})`);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
return lines;
|
|
82
|
+
}
|
|
83
|
+
|
|
5
84
|
// Normalize the messy registry categories into clean buckets
|
|
6
85
|
function normalizeCategory(raw: string): string {
|
|
7
86
|
const lower = raw.toLowerCase();
|
|
@@ -117,10 +196,10 @@ function formatRobotsAudit(origin: string, robots: RobotsTxt, statusCode: number
|
|
|
117
196
|
return lines.join("\n");
|
|
118
197
|
}
|
|
119
198
|
|
|
120
|
-
export function
|
|
199
|
+
export function registerAiTool(server: McpServer): void {
|
|
121
200
|
server.tool(
|
|
122
|
-
"
|
|
123
|
-
"
|
|
201
|
+
"ai",
|
|
202
|
+
"Analyze your site's AI visibility. Audits AI crawler access (training, search, assistant, agent) via robots.txt, checks for llms.txt and llms-full.txt, validates syntax per RFC 9309, and tests specific path access. Shows how AI systems see your site.",
|
|
124
203
|
{
|
|
125
204
|
url: z
|
|
126
205
|
.string()
|
|
@@ -159,8 +238,21 @@ export function registerRobotsTool(server: McpServer): void {
|
|
|
159
238
|
return { content: [{ type: "text", text: lines.join("\n") }] };
|
|
160
239
|
}
|
|
161
240
|
|
|
162
|
-
const crawlers = await
|
|
163
|
-
|
|
241
|
+
const [crawlers, llmsTxt, llmsFullTxt] = await Promise.all([
|
|
242
|
+
auditAiCrawlers(robotsTxt),
|
|
243
|
+
checkLlmsTxt(origin, "/llms.txt"),
|
|
244
|
+
checkLlmsTxt(origin, "/llms-full.txt"),
|
|
245
|
+
]);
|
|
246
|
+
|
|
247
|
+
const output: string[] = [formatRobotsAudit(origin, robotsTxt, statusCode, crawlers)];
|
|
248
|
+
|
|
249
|
+
if (crawlers.length > 0) {
|
|
250
|
+
output.push(...formatCategorySummary(crawlers));
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
output.push("", ...formatLlmsTxt(llmsTxt, llmsFullTxt));
|
|
254
|
+
|
|
255
|
+
return { content: [{ type: "text", text: output.join("\n") }] };
|
|
164
256
|
} catch (err) {
|
|
165
257
|
const msg = err instanceof Error ? err.message : String(err);
|
|
166
258
|
return { content: [{ type: "text", text: `Error analyzing robots.txt: ${msg}` }] };
|
package/src/tools/audit.ts
CHANGED
|
@@ -3,7 +3,7 @@ import { z } from "zod";
|
|
|
3
3
|
import { inspectUrl, listSitemaps } from "../lib/gsc.js";
|
|
4
4
|
import { type PsiCategoryType, type PsiResult, runPagespeed } from "../lib/psi.js";
|
|
5
5
|
import { auditAiCrawlers, fetchRobotsTxt, isAllowed } from "../lib/robots.js";
|
|
6
|
-
import { fetchSitemap, inspectSingle, sampleUrls } from "
|
|
6
|
+
import { fetchSitemap, inspectSingle, sampleUrls } from "../lib/sitemap.js";
|
|
7
7
|
|
|
8
8
|
type Severity = "HIGH" | "MEDIUM" | "LOW";
|
|
9
9
|
|