pagesight 0.3.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pagesight",
3
- "version": "0.3.1",
3
+ "version": "0.5.0",
4
4
  "description": "See your site the way search engines and AI see it.",
5
5
  "keywords": [
6
6
  "seo",
package/src/index.ts CHANGED
@@ -7,6 +7,7 @@ import { registerMetatagsTool } from "./tools/metatags.js";
7
7
  import { registerPagespeedTool } from "./tools/pagespeed.js";
8
8
  import { registerPerformanceTool } from "./tools/performance.js";
9
9
  import { registerRobotsTool } from "./tools/robots.js";
10
+ import { registerSampleInspectTool } from "./tools/sample-inspect.js";
10
11
  import { registerSetupTool } from "./tools/setup.js";
11
12
  import { registerSitemapsTool } from "./tools/sitemaps.js";
12
13
 
@@ -23,6 +24,7 @@ registerMetatagsTool(server);
23
24
  registerPagespeedTool(server);
24
25
  registerPerformanceTool(server);
25
26
  registerRobotsTool(server);
27
+ registerSampleInspectTool(server);
26
28
  registerSitemapsTool(server);
27
29
  registerSetupTool(server);
28
30
 
package/src/tools/crux.ts CHANGED
@@ -193,14 +193,15 @@ export function registerCruxTool(server: McpServer): void {
193
193
  } catch (err) {
194
194
  const msg = err instanceof Error ? err.message : String(err);
195
195
  if (msg.includes("404")) {
196
- return {
197
- content: [
198
- {
199
- type: "text",
200
- text: `No CrUX data available for ${url ?? origin}. The page may not have enough Chrome user traffic.`,
201
- },
202
- ],
203
- };
196
+ const target = url ?? origin ?? "";
197
+ const lines = [`No CrUX data for ${target}.`, ""];
198
+ lines.push("CrUX requires sufficient Chrome user traffic (roughly 1,000+ monthly visits).");
199
+ if (url) {
200
+ const originUrl = new URL(url).origin;
201
+ lines.push(`Try origin-level data instead: origin "${originUrl}"`);
202
+ }
203
+ lines.push("For lab metrics without traffic requirements, use the pagespeed tool.");
204
+ return { content: [{ type: "text", text: lines.join("\n") }] };
204
205
  }
205
206
  if (msg.includes("SERVICE_DISABLED") || msg.includes("API_KEY_SERVICE_BLOCKED")) {
206
207
  return {
@@ -268,14 +269,15 @@ export function registerCruxTool(server: McpServer): void {
268
269
  } catch (err) {
269
270
  const msg = err instanceof Error ? err.message : String(err);
270
271
  if (msg.includes("404")) {
271
- return {
272
- content: [
273
- {
274
- type: "text",
275
- text: `No CrUX history data for ${url ?? origin}. The page may not have enough Chrome user traffic.`,
276
- },
277
- ],
278
- };
272
+ const target = url ?? origin ?? "";
273
+ const lines = [`No CrUX history data for ${target}.`, ""];
274
+ lines.push("CrUX requires sufficient Chrome user traffic (roughly 1,000+ monthly visits).");
275
+ if (url) {
276
+ const originUrl = new URL(url).origin;
277
+ lines.push(`Try origin-level data instead: origin "${originUrl}"`);
278
+ }
279
+ lines.push("For lab metrics without traffic requirements, use the pagespeed tool.");
280
+ return { content: [{ type: "text", text: lines.join("\n") }] };
279
281
  }
280
282
  if (msg.includes("SERVICE_DISABLED")) {
281
283
  return {
@@ -134,6 +134,77 @@ function formatJsonLd(data: unknown, indent = 0): string[] {
134
134
  return lines;
135
135
  }
136
136
 
137
+ interface RedirectHop {
138
+ url: string;
139
+ status: number;
140
+ }
141
+
142
+ interface ImageCheck {
143
+ url: string;
144
+ tag: string;
145
+ status: number | null;
146
+ contentType: string | null;
147
+ contentLength: number | null;
148
+ error: string | null;
149
+ }
150
+
151
+ async function followRedirects(
152
+ url: string,
153
+ ua: string,
154
+ maxHops = 10,
155
+ ): Promise<{ chain: RedirectHop[]; response: Response }> {
156
+ const chain: RedirectHop[] = [];
157
+ let current = url;
158
+
159
+ for (let i = 0; i < maxHops; i++) {
160
+ const res = await fetch(current, {
161
+ headers: { "User-Agent": ua, Accept: "text/html" },
162
+ redirect: "manual",
163
+ });
164
+
165
+ chain.push({ url: current, status: res.status });
166
+
167
+ if (res.status >= 300 && res.status < 400) {
168
+ const location = res.headers.get("location");
169
+ if (!location) break;
170
+ current = new URL(location, current).href;
171
+ continue;
172
+ }
173
+
174
+ return { chain, response: res };
175
+ }
176
+
177
+ // If we exhausted hops, do a final follow-redirect fetch
178
+ const res = await fetch(current, {
179
+ headers: { "User-Agent": ua, Accept: "text/html" },
180
+ redirect: "follow",
181
+ });
182
+ return { chain, response: res };
183
+ }
184
+
185
+ async function checkImage(imageUrl: string, tag: string): Promise<ImageCheck> {
186
+ try {
187
+ const res = await fetch(imageUrl, { method: "HEAD", redirect: "follow" });
188
+ return {
189
+ url: imageUrl,
190
+ tag,
191
+ status: res.status,
192
+ contentType: res.headers.get("content-type"),
193
+ contentLength: res.headers.has("content-length") ? Number(res.headers.get("content-length")) : null,
194
+ error: null,
195
+ };
196
+ } catch (err) {
197
+ return {
198
+ url: imageUrl,
199
+ tag,
200
+ status: null,
201
+ contentType: null,
202
+ contentLength: null,
203
+ error: err instanceof Error ? err.message : String(err),
204
+ };
205
+ }
206
+ }
207
+
137
208
  function formatMetatags(url: string, parsed: ParsedHead): string {
138
209
  const lines: string[] = [`=== Meta Tags: ${url} ===`, ""];
139
210
 
@@ -231,13 +302,7 @@ export function registerMetatagsTool(server: McpServer): void {
231
302
  async ({ url, user_agent }) => {
232
303
  try {
233
304
  const ua = user_agent ?? "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)";
234
- const res = await fetch(url, {
235
- headers: {
236
- "User-Agent": ua,
237
- Accept: "text/html",
238
- },
239
- redirect: "follow",
240
- });
305
+ const { chain, response: res } = await followRedirects(url, ua);
241
306
 
242
307
  if (!res.ok) {
243
308
  return {
@@ -252,15 +317,50 @@ export function registerMetatagsTool(server: McpServer): void {
252
317
 
253
318
  const html = await res.text();
254
319
  const parsed = parseHead(html);
255
- const finalUrl = res.url !== url ? `(redirected to ${res.url})\n\n` : "";
320
+ const output: string[] = [];
321
+
322
+ // Redirect chain (only if there were redirects)
323
+ if (chain.length > 1) {
324
+ output.push("--- Redirect Chain ---", "");
325
+ for (let i = 0; i < chain.length; i++) {
326
+ const hop = chain[i];
327
+ const prefix = i === chain.length - 1 ? "" : `${hop.status} → `;
328
+ output.push(`${i + 1}. ${prefix}${hop.url}`);
329
+ }
330
+ output.push("");
331
+ }
332
+
333
+ // Main meta tag report
334
+ const finalUrl = chain.length > 1 ? chain[chain.length - 1].url : url;
335
+ output.push(formatMetatags(finalUrl, parsed));
336
+
337
+ // OG/Twitter image validation
338
+ const ogImage = getMeta(parsed.meta, "og:image");
339
+ const twitterImage = getMeta(parsed.meta, "twitter:image");
340
+ const imagesToCheck: Array<{ url: string; tag: string }> = [];
341
+ if (ogImage) imagesToCheck.push({ url: ogImage, tag: "og:image" });
342
+ if (twitterImage && twitterImage !== ogImage) imagesToCheck.push({ url: twitterImage, tag: "twitter:image" });
343
+
344
+ if (imagesToCheck.length > 0) {
345
+ const checks = await Promise.all(imagesToCheck.map((img) => checkImage(img.url, img.tag)));
346
+ output.push("", "--- Image Validation ---", "");
347
+ for (const check of checks) {
348
+ if (check.error) {
349
+ output.push(`${check.tag}: FAILED — ${check.error}`);
350
+ output.push(` URL: ${check.url}`);
351
+ } else if (check.status && check.status >= 400) {
352
+ output.push(`${check.tag}: BROKEN — HTTP ${check.status}`);
353
+ output.push(` URL: ${check.url}`);
354
+ } else {
355
+ const size = check.contentLength ? ` (${Math.round(check.contentLength / 1024)} KB)` : "";
356
+ const type = check.contentType ? ` ${check.contentType}` : "";
357
+ output.push(`${check.tag}: OK —${type}${size}`);
358
+ }
359
+ }
360
+ }
256
361
 
257
362
  return {
258
- content: [
259
- {
260
- type: "text",
261
- text: `${finalUrl}${formatMetatags(res.url, parsed)}`,
262
- },
263
- ],
363
+ content: [{ type: "text", text: output.join("\n") }],
264
364
  };
265
365
  } catch (err) {
266
366
  const msg = err instanceof Error ? err.message : String(err);
@@ -0,0 +1,275 @@
1
+ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
+ import { z } from "zod";
3
+ import { inspectUrl, listSitemaps } from "../lib/gsc.js";
4
+
5
+ interface SitemapParseResult {
6
+ urls: string[];
7
+ isSitemapIndex: boolean;
8
+ childSitemaps: string[];
9
+ }
10
+
11
+ function parseSitemapXml(xml: string): SitemapParseResult {
12
+ const urls: string[] = [];
13
+ const childSitemaps: string[] = [];
14
+
15
+ // Check if it's a sitemap index
16
+ const isSitemapIndex = /<sitemapindex/i.test(xml);
17
+
18
+ if (isSitemapIndex) {
19
+ for (const m of xml.matchAll(/<sitemap[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/sitemap>/gi)) {
20
+ childSitemaps.push(m[1].trim());
21
+ }
22
+ } else {
23
+ for (const m of xml.matchAll(/<url[^>]*>[\s\S]*?<loc>\s*(.*?)\s*<\/loc>[\s\S]*?<\/url>/gi)) {
24
+ urls.push(m[1].trim());
25
+ }
26
+ }
27
+
28
+ return { urls, isSitemapIndex, childSitemaps };
29
+ }
30
+
31
+ async function fetchSitemap(sitemapUrl: string): Promise<SitemapParseResult> {
32
+ const res = await fetch(sitemapUrl, {
33
+ headers: { "User-Agent": "Pagesight/1.0" },
34
+ });
35
+
36
+ if (!res.ok) {
37
+ throw new Error(`Failed to fetch sitemap ${sitemapUrl}: HTTP ${res.status}`);
38
+ }
39
+
40
+ const xml = await res.text();
41
+ return parseSitemapXml(xml);
42
+ }
43
+
44
+ function sampleUrls(urls: string[], count: number, strategy: string): string[] {
45
+ if (urls.length <= count) return [...urls];
46
+
47
+ if (strategy === "first") {
48
+ return urls.slice(0, count);
49
+ }
50
+
51
+ if (strategy === "spread") {
52
+ const step = Math.floor(urls.length / count);
53
+ const sampled: string[] = [];
54
+ for (let i = 0; i < count; i++) {
55
+ sampled.push(urls[i * step]);
56
+ }
57
+ return sampled;
58
+ }
59
+
60
+ // random (default)
61
+ const shuffled = [...urls];
62
+ for (let i = shuffled.length - 1; i > 0; i--) {
63
+ const j = Math.floor(Math.random() * (i + 1));
64
+ [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]];
65
+ }
66
+ return shuffled.slice(0, count);
67
+ }
68
+
69
+ interface InspectionSummary {
70
+ url: string;
71
+ verdict: string;
72
+ coverageState: string;
73
+ pageFetchState: string;
74
+ robotsTxtState: string;
75
+ indexingState: string;
76
+ lastCrawlTime: string | null;
77
+ googleCanonical: string | null;
78
+ error: string | null;
79
+ }
80
+
81
+ async function inspectSingle(url: string, siteUrl: string): Promise<InspectionSummary> {
82
+ try {
83
+ const r = await inspectUrl(url, siteUrl);
84
+ const idx = r.indexStatusResult;
85
+ return {
86
+ url,
87
+ verdict: idx.verdict,
88
+ coverageState: idx.coverageState,
89
+ pageFetchState: idx.pageFetchState,
90
+ robotsTxtState: idx.robotsTxtState,
91
+ indexingState: idx.indexingState,
92
+ lastCrawlTime: idx.lastCrawlTime ?? null,
93
+ googleCanonical: idx.googleCanonical ?? null,
94
+ error: null,
95
+ };
96
+ } catch (err) {
97
+ return {
98
+ url,
99
+ verdict: "ERROR",
100
+ coverageState: "ERROR",
101
+ pageFetchState: "ERROR",
102
+ robotsTxtState: "ERROR",
103
+ indexingState: "ERROR",
104
+ lastCrawlTime: null,
105
+ googleCanonical: null,
106
+ error: err instanceof Error ? err.message : String(err),
107
+ };
108
+ }
109
+ }
110
+
111
+ function formatResults(siteUrl: string, sitemapUrl: string, totalUrls: number, results: InspectionSummary[]): string {
112
+ const lines: string[] = [
113
+ `=== Sample Inspection: ${siteUrl} ===`,
114
+ `Sitemap: ${sitemapUrl} (${totalUrls.toLocaleString()} URLs)`,
115
+ `Sampled: ${results.length}`,
116
+ "",
117
+ ];
118
+
119
+ // Summary
120
+ const verdictCounts: Record<string, number> = {};
121
+ const coverageCounts: Record<string, number> = {};
122
+ const fetchCounts: Record<string, number> = {};
123
+ let indexed = 0;
124
+ let errors = 0;
125
+
126
+ for (const r of results) {
127
+ if (r.error) {
128
+ errors++;
129
+ continue;
130
+ }
131
+ verdictCounts[r.verdict] = (verdictCounts[r.verdict] ?? 0) + 1;
132
+ coverageCounts[r.coverageState] = (coverageCounts[r.coverageState] ?? 0) + 1;
133
+ fetchCounts[r.pageFetchState] = (fetchCounts[r.pageFetchState] ?? 0) + 1;
134
+ if (r.verdict === "PASS") indexed++;
135
+ }
136
+
137
+ const inspected = results.length - errors;
138
+ lines.push("--- Summary ---", "");
139
+ lines.push(`Indexed: ${indexed}/${inspected}`);
140
+
141
+ if (indexed < inspected) {
142
+ lines.push(`Not indexed: ${inspected - indexed}/${inspected}`);
143
+ for (const [state, count] of Object.entries(coverageCounts)) {
144
+ if (state !== "Submitted and indexed" && state !== "Indexing allowed") {
145
+ lines.push(` ${state}: ${count}`);
146
+ }
147
+ }
148
+ }
149
+
150
+ // Page fetch issues
151
+ const fetchIssues = Object.entries(fetchCounts).filter(([s]) => s !== "SUCCESSFUL");
152
+ if (fetchIssues.length > 0) {
153
+ lines.push("");
154
+ lines.push("Page fetch issues:");
155
+ for (const [state, count] of fetchIssues) {
156
+ lines.push(` ${state}: ${count}`);
157
+ }
158
+ }
159
+
160
+ if (errors > 0) {
161
+ lines.push(`\nInspection errors: ${errors}`);
162
+ }
163
+
164
+ // Individual results
165
+ lines.push("", "--- Details ---", "");
166
+ for (let i = 0; i < results.length; i++) {
167
+ const r = results[i];
168
+ lines.push(`${i + 1}. ${r.url}`);
169
+ if (r.error) {
170
+ lines.push(` Error: ${r.error}`);
171
+ } else {
172
+ lines.push(` Verdict: ${r.verdict}`);
173
+ lines.push(` Coverage: ${r.coverageState}`);
174
+ lines.push(` Page fetch: ${r.pageFetchState}`);
175
+ if (r.robotsTxtState !== "ALLOWED") lines.push(` Robots.txt: ${r.robotsTxtState}`);
176
+ if (r.indexingState !== "INDEXING_ALLOWED") lines.push(` Indexing: ${r.indexingState}`);
177
+ if (r.lastCrawlTime) lines.push(` Last crawled: ${r.lastCrawlTime}`);
178
+ if (r.googleCanonical && r.googleCanonical !== r.url) {
179
+ lines.push(` Google canonical: ${r.googleCanonical}`);
180
+ }
181
+ }
182
+ lines.push("");
183
+ }
184
+
185
+ return lines.join("\n");
186
+ }
187
+
188
+ export function registerSampleInspectTool(server: McpServer): void {
189
+ server.tool(
190
+ "sample_inspect",
191
+ "Sample URLs from a sitemap and batch-inspect them via Google Search Console. Diagnoses indexing issues by revealing patterns across multiple URLs — why pages aren't indexed, common fetch errors, robots.txt blocks.",
192
+ {
193
+ site_url: z.string().describe("GSC property (e.g., 'https://example.com/' or 'sc-domain:example.com')"),
194
+ sitemap_url: z
195
+ .string()
196
+ .url()
197
+ .optional()
198
+ .describe("Sitemap URL to sample from. If omitted, discovers sitemaps from GSC."),
199
+ sample_size: z.number().min(1).max(10).optional().describe("Number of URLs to inspect. Default: 5. Max: 10."),
200
+ strategy: z
201
+ .enum(["random", "first", "spread"])
202
+ .optional()
203
+ .describe("Sampling strategy. 'random' (default), 'first' (first N), 'spread' (evenly spaced)."),
204
+ },
205
+ async ({ site_url, sitemap_url, sample_size, strategy }) => {
206
+ const count = sample_size ?? 5;
207
+ const sampleStrategy = strategy ?? "random";
208
+
209
+ try {
210
+ // Discover sitemap URL if not provided
211
+ let resolvedSitemapUrl = sitemap_url;
212
+ if (!resolvedSitemapUrl) {
213
+ const sitemaps = await listSitemaps(site_url);
214
+ if (sitemaps.length === 0) {
215
+ return {
216
+ content: [
217
+ {
218
+ type: "text",
219
+ text: `No sitemaps found for ${site_url} in GSC. Provide a sitemap_url directly.`,
220
+ },
221
+ ],
222
+ };
223
+ }
224
+ // Pick the first non-index sitemap, or the first one
225
+ const nonIndex = sitemaps.find((s) => !s.isSitemapsIndex);
226
+ resolvedSitemapUrl = (nonIndex ?? sitemaps[0]).path;
227
+ }
228
+
229
+ // Fetch and parse sitemap
230
+ let parsed = await fetchSitemap(resolvedSitemapUrl);
231
+
232
+ // If it's a sitemap index, fetch the first child
233
+ if (parsed.isSitemapIndex && parsed.childSitemaps.length > 0) {
234
+ const childUrl = parsed.childSitemaps[0];
235
+ parsed = await fetchSitemap(childUrl);
236
+ resolvedSitemapUrl = `${resolvedSitemapUrl} → ${childUrl}`;
237
+ }
238
+
239
+ if (parsed.urls.length === 0) {
240
+ return {
241
+ content: [
242
+ {
243
+ type: "text",
244
+ text: `Sitemap ${resolvedSitemapUrl} contains no URLs.`,
245
+ },
246
+ ],
247
+ };
248
+ }
249
+
250
+ // Sample URLs
251
+ const sampled = sampleUrls(parsed.urls, count, sampleStrategy);
252
+
253
+ // Inspect each URL sequentially (API rate limits)
254
+ const results: InspectionSummary[] = [];
255
+ for (const url of sampled) {
256
+ results.push(await inspectSingle(url, site_url));
257
+ }
258
+
259
+ return {
260
+ content: [
261
+ {
262
+ type: "text",
263
+ text: formatResults(site_url, resolvedSitemapUrl, parsed.urls.length, results),
264
+ },
265
+ ],
266
+ };
267
+ } catch (err) {
268
+ const msg = err instanceof Error ? err.message : String(err);
269
+ return {
270
+ content: [{ type: "text", text: `Error: ${msg}` }],
271
+ };
272
+ }
273
+ },
274
+ );
275
+ }