pagesight 0.13.1 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -8,34 +8,40 @@ See your site the way search engines and AI see it.
8
8
  npm install pagesight
9
9
  ```
10
10
 
11
- Your AI assistant can write your code. Now it can see your site. Index status, performance, real-user metrics, search traffic, meta tags, structured data, AI crawler access — one package, one call.
11
+ Your AI assistant can write your code. Now it can see your site. Index status, performance, real-user metrics, search traffic, meta tags, structured data, AI crawler access, link health — one package, one call.
12
12
 
13
13
  ```
14
14
  === Site Audit: https://example.com ===
15
15
 
16
+ 2 checks failed — results below are partial:
17
+
18
+ FAIL PageSpeed: quota exceeded
19
+ FAIL Sitemaps: permission denied
20
+
21
+ 5 findings:
22
+
16
23
  HIGH Missing canonical URL
17
- HIGH 7,772 sitemap URLs submitted, 0 indexed
24
+ HIGH 22 sitemap URLs submitted, 0 indexed
25
+ Auto-inspected 5 URLs:
26
+ - 2/5 indexed
27
+ - 3/5 Discovered - currently not indexed: /docs/, /pricing/, /about/
18
28
  MEDIUM Missing og:image — no social preview image
19
- MEDIUM Accessibility score: 89/100
20
29
  LOW Missing Twitter Card tags
21
- LOW No structured data (JSON-LD) found
30
+ LOW 6/139 AI crawlers blocked
22
31
  ```
23
32
 
24
33
  ## Tools
25
34
 
26
- | Tool | What it does |
27
- |------|-------------|
28
- | `audit` | One-call site audit. Runs all checks in parallel. Returns prioritized findings. |
29
- | `pagespeed` | Lighthouse scores, Core Web Vitals, opportunities, failing audits with fix links. |
30
- | `metatags` | OG, Twitter Card, canonical, JSON-LD with schema validation, redirect chain, image validation. |
31
- | `inspect` | Google index status, canonical choice, crawl state, rich results. |
32
- | `sample_inspect` | Sample URLs from a sitemap and batch-inspect. Diagnoses indexing patterns. |
33
- | `performance` | Search analytics — clicks, impressions, CTR, position. `compare: true` for period-over-period. |
34
- | `crux` | Real-user Core Web Vitals (p75, histograms). |
35
- | `crux_history` | CWV trends over time — up to 40 weekly data points. |
36
- | `robots` | robots.txt validation (RFC 9309) + AI crawler audit (139+ bots). |
37
- | `sitemaps` | Search Console properties and sitemaps with submitted/indexed counts. |
38
- | `setup` | Auth status and OAuth setup. |
35
+ 6 tools organized by intent:
36
+
37
+ | Tool | Intent | What it does |
38
+ |------|--------|-------------|
39
+ | `audit` | How's my site? | One-call site audit. Runs all checks in parallel. Prioritized findings with auto-drill-down on indexing issues. |
40
+ | `page` | What's on this URL? | Meta tags, OG, Twitter Card, JSON-LD validation (19 schema types), internal link health, redirect chains, WCAG contrast checker. Batch mode for multiple URLs. |
41
+ | `speed` | How fast is it? | PageSpeed single/batch/compare with Lighthouse scores and opportunities. CrUX real-user metrics (snapshot + history trends). |
42
+ | `search` | How's Google seeing me? | URL inspection, sample-inspect from sitemaps, sitemap management, search analytics with period-over-period comparison. |
43
+ | `ai` | How's AI seeing me? | AI crawler audit (139+ bots by category), robots.txt validation (RFC 9309), llms.txt detection, path access checks. |
44
+ | `setup` | Auth config | Auth status check and OAuth setup flow. |
39
45
 
40
46
  ## Setup
41
47
 
@@ -58,7 +64,7 @@ Add to your MCP config:
58
64
  }
59
65
  ```
60
66
 
61
- `robots`, `metatags`, and `pagespeed` work without credentials.
67
+ `page`, `speed`, and `ai` work without credentials. `search` and `audit` (for GSC checks) require OAuth or a service account.
62
68
 
63
69
  ### Full setup
64
70
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pagesight",
3
- "version": "0.13.1",
3
+ "version": "0.14.1",
4
4
  "description": "See your site the way search engines and AI see it.",
5
5
  "keywords": [
6
6
  "seo",
package/src/tools/ai.ts CHANGED
@@ -12,12 +12,24 @@ interface LlmsTxtResult {
12
12
 
13
13
  async function checkLlmsTxt(origin: string, path: string): Promise<LlmsTxtResult> {
14
14
  try {
15
+ const controller = new AbortController();
16
+ const timeout = setTimeout(() => controller.abort(), 10_000);
15
17
  const res = await fetch(`${origin}${path}`, {
16
18
  headers: { "User-Agent": "Pagesight/1.0" },
17
19
  redirect: "follow",
20
+ signal: controller.signal,
18
21
  });
22
+ clearTimeout(timeout);
19
23
  if (!res.ok) return { exists: false, size: null, firstLine: null };
24
+ // Cap at 1MB to avoid OOM on large responses
25
+ const contentLength = Number(res.headers.get("content-length") ?? 0);
26
+ if (contentLength > 1_048_576) {
27
+ return { exists: true, size: contentLength, firstLine: "(file too large to preview)" };
28
+ }
20
29
  const text = await res.text();
30
+ if (text.length > 1_048_576) {
31
+ return { exists: true, size: text.length, firstLine: "(file too large to preview)" };
32
+ }
21
33
  const firstLine =
22
34
  text
23
35
  .split("\n")
@@ -215,6 +227,7 @@ export function registerAiTool(server: McpServer): void {
215
227
  {
216
228
  url: z
217
229
  .string()
230
+ .url()
218
231
  .describe("Site URL or origin (e.g., 'https://example.com'). Fetches /robots.txt from this origin."),
219
232
  check_path: z
220
233
  .string()
package/src/tools/page.ts CHANGED
@@ -571,6 +571,17 @@ function formatMetatags(url: string, parsed: ParsedHead): string {
571
571
  lines.push(`${h.lang}: ${h.href}`);
572
572
  }
573
573
  lines.push("");
574
+ } else {
575
+ // Warn if URL suggests locale-specific content but no hreflang
576
+ const localePattern =
577
+ /\/(?:en|es|fr|de|ja|ko|pt|zh|ru|it|nl|sv|da|fi|nb|pl|tr|ar|hi|th|vi|id|ms|uk|cs|ro|hu|el|he|bg|hr|sk|sl|sr|lt|lv|et|ca|gl|eu|cy)(?:[-_][a-z]{2,4})?(?:\/|$)/i;
578
+ const checkUrl = parsed.canonical ?? url;
579
+ if (localePattern.test(checkUrl)) {
580
+ lines.push(
581
+ "WARN: URL appears locale-specific but no hreflang tags found — search engines may not discover alternate language versions",
582
+ );
583
+ lines.push("");
584
+ }
574
585
  }
575
586
 
576
587
  // Summary of missing tags relevant to search and social
@@ -604,7 +615,7 @@ async function checkLink(href: string): Promise<LinkResult> {
604
615
  try {
605
616
  for (let i = 0; i < 10; i++) {
606
617
  const res = await fetch(current, {
607
- method: "GET",
618
+ method: "HEAD",
608
619
  headers: { "User-Agent": "Mozilla/5.0 (compatible; Googlebot/2.1)", Accept: "text/html" },
609
620
  redirect: "manual",
610
621
  });
@@ -798,15 +809,19 @@ export function registerPageTool(server: McpServer): void {
798
809
  "page",
799
810
  "Analyze what's on a page — meta tags, Open Graph, Twitter Card, structured data (JSON-LD), internal links, and redirect chains. Also includes a WCAG contrast checker for accessibility fixes.",
800
811
  {
801
- url: z
802
- .string()
803
- .url()
812
+ url: z.string().url().optional().describe("Single URL to analyze. Use this OR urls, not both."),
813
+ urls: z
814
+ .array(z.string().url())
815
+ .min(2)
816
+ .max(10)
804
817
  .optional()
805
- .describe("Page URL to analyze. Returns meta tags, structured data, and optionally internal links."),
818
+ .describe(
819
+ "Multiple URLs (2-10) for batch analysis. Returns a summary table of meta tags, structured data, and link health per page.",
820
+ ),
806
821
  check_links: z
807
822
  .boolean()
808
823
  .optional()
809
- .describe("Also check all internal links for broken links and redirect chains. Default: false."),
824
+ .describe("Check internal links for broken links and redirect chains. Default: true. Set false to skip."),
810
825
  user_agent: z.string().optional().describe("Custom User-Agent for page fetch. Default: Googlebot-compatible."),
811
826
  foreground: z
812
827
  .string()
@@ -818,7 +833,129 @@ export function registerPageTool(server: McpServer): void {
818
833
  .optional()
819
834
  .describe("Whether text is large (≥18pt or ≥14pt bold). Lowers AA threshold to 3:1."),
820
835
  },
821
- async ({ url, check_links, user_agent, foreground, background, large_text }) => {
836
+ async ({ url, urls, check_links, user_agent, foreground, background, large_text }) => {
837
+ // --- Batch mode ---
838
+ if (urls) {
839
+ const ua = user_agent ?? "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)";
840
+ const results: Array<{
841
+ url: string;
842
+ title: string;
843
+ description: string;
844
+ canonical: string;
845
+ jsonLd: string;
846
+ links: string;
847
+ issues: string[];
848
+ error?: string;
849
+ }> = [];
850
+
851
+ for (const pageUrl of urls) {
852
+ try {
853
+ const { chain, response: res } = await followRedirects(pageUrl, ua);
854
+ if (!res.ok) {
855
+ results.push({
856
+ url: pageUrl,
857
+ title: "-",
858
+ description: "-",
859
+ canonical: "-",
860
+ jsonLd: "-",
861
+ links: "-",
862
+ issues: [`HTTP ${res.status}`],
863
+ });
864
+ continue;
865
+ }
866
+ const html = await res.text();
867
+ const parsed = parseHead(html);
868
+ const finalUrl = chain.length > 1 ? chain[chain.length - 1].url : pageUrl;
869
+
870
+ const issues: string[] = [];
871
+ if (!parsed.title) issues.push("no title");
872
+ if (!getMeta(parsed.meta, "description")) issues.push("no description");
873
+ else if ((getMeta(parsed.meta, "description") ?? "").length > 155) issues.push("description >155 chars");
874
+ if (!parsed.canonical) issues.push("no canonical");
875
+ if (parsed.jsonLd.length === 0) issues.push("no JSON-LD");
876
+ if (chain.length > 2) issues.push(`${chain.length - 1} redirects`);
877
+
878
+ // Quick link check
879
+ let linkSummary = "-";
880
+ if (check_links !== false) {
881
+ const origin = new URL(finalUrl).origin;
882
+ const links = extractInternalLinks(html, origin);
883
+ if (links.length > 0) {
884
+ const checked = await Promise.all(links.slice(0, 20).map((href) => checkLink(href)));
885
+ const broken = checked.filter((r) => r.error || (r.status && r.status >= 400)).length;
886
+ const redirected = checked.filter(
887
+ (r) => !r.error && r.redirectChain.length > 1 && r.status && r.status < 400,
888
+ ).length;
889
+ const parts: string[] = [`${links.length} found`];
890
+ if (broken > 0) parts.push(`${broken} broken`);
891
+ if (redirected > 0) parts.push(`${redirected} redirected`);
892
+ if (broken === 0 && redirected === 0) parts.push("all OK");
893
+ linkSummary = parts.join(", ");
894
+ if (broken > 0) issues.push(`${broken} broken links`);
895
+ if (redirected > 0) issues.push(`${redirected} redirect chains`);
896
+ } else {
897
+ linkSummary = "0 (SPA?)";
898
+ }
899
+ }
900
+
901
+ const jsonLdTypes = parsed.jsonLd
902
+ .map((block) => {
903
+ if (block && typeof block === "object" && "@type" in block)
904
+ return String((block as Record<string, unknown>)["@type"]);
905
+ return null;
906
+ })
907
+ .filter(Boolean);
908
+
909
+ results.push({
910
+ url: finalUrl,
911
+ title: parsed.title
912
+ ? parsed.title.length > 40
913
+ ? `${parsed.title.slice(0, 40)}...`
914
+ : parsed.title
915
+ : "(missing)",
916
+ description: getMeta(parsed.meta, "description") ? "yes" : "no",
917
+ canonical: parsed.canonical ? "yes" : "no",
918
+ jsonLd: jsonLdTypes.length > 0 ? jsonLdTypes.join(", ") : "none",
919
+ links: linkSummary,
920
+ issues,
921
+ });
922
+ } catch (err) {
923
+ results.push({
924
+ url: pageUrl,
925
+ title: "-",
926
+ description: "-",
927
+ canonical: "-",
928
+ jsonLd: "-",
929
+ links: "-",
930
+ issues: [err instanceof Error ? err.message : String(err)],
931
+ });
932
+ }
933
+ }
934
+
935
+ const lines: string[] = [`=== Batch Page Analysis (${results.length} URLs) ===`, ""];
936
+
937
+ for (const r of results) {
938
+ const u = new URL(r.url);
939
+ const allSameHost = results.every((x) => new URL(x.url).hostname === u.hostname);
940
+ const label = allSameHost ? u.pathname : `${u.hostname}${u.pathname}`;
941
+ lines.push(label);
942
+ lines.push(` Title: ${r.title}`);
943
+ lines.push(` Description: ${r.description} Canonical: ${r.canonical} JSON-LD: ${r.jsonLd}`);
944
+ if (check_links !== false) lines.push(` Links: ${r.links}`);
945
+ if (r.issues.length > 0) {
946
+ lines.push(` Issues: ${r.issues.join(", ")}`);
947
+ } else {
948
+ lines.push(" Issues: none");
949
+ }
950
+ lines.push("");
951
+ }
952
+
953
+ const issueCount = results.reduce((sum, r) => sum + r.issues.length, 0);
954
+ lines.push(`Total: ${results.length} pages, ${issueCount} issues`);
955
+
956
+ return { content: [{ type: "text", text: lines.join("\n") }] };
957
+ }
958
+
822
959
  // --- Contrast-only mode (no URL) ---
823
960
  if (foreground && background && !url) {
824
961
  const fg = parseHex(foreground);
@@ -971,8 +1108,34 @@ export function registerPageTool(server: McpServer): void {
971
1108
  }
972
1109
  }
973
1110
 
974
- // --- Link check (appended after metatags output) ---
975
- if (check_links) {
1111
+ // Contrast check (before links, after structured data)
1112
+ if (foreground && background) {
1113
+ const fg = parseHex(foreground);
1114
+ const bg = parseHex(background);
1115
+ if (fg && bg) {
1116
+ const fgLum = relativeLuminance(...fg);
1117
+ const bgLum = relativeLuminance(...bg);
1118
+ const ratio = contrastRatio(fgLum, bgLum);
1119
+ const isLarge = large_text ?? false;
1120
+ const level = wcagLevel(ratio, isLarge);
1121
+ const aaThreshold = isLarge ? 3 : 4.5;
1122
+ const aaaThreshold = isLarge ? 4.5 : 7;
1123
+ output.push(
1124
+ "",
1125
+ "--- Contrast Check ---",
1126
+ "",
1127
+ `Foreground: ${toHex(...fg)} Background: ${toHex(...bg)}`,
1128
+ `Ratio: ${ratio.toFixed(2)}:1 AA: ${ratio >= aaThreshold ? "PASS" : "FAIL"} AAA: ${ratio >= aaaThreshold ? "PASS" : "FAIL"} Result: ${level}`,
1129
+ );
1130
+ if (ratio < aaThreshold) {
1131
+ const suggested = findNearestPassing(fg, bg, aaThreshold);
1132
+ output.push(`Nearest AA-passing foreground: ${suggested}`);
1133
+ }
1134
+ }
1135
+ }
1136
+
1137
+ // --- Link check (last section) ---
1138
+ if (check_links !== false) {
976
1139
  const origin = new URL(url).origin;
977
1140
  const links = extractInternalLinks(html, origin);
978
1141
 
@@ -1001,32 +1164,6 @@ export function registerPageTool(server: McpServer): void {
1001
1164
  }
1002
1165
  }
1003
1166
 
1004
- // Append contrast check if colors were also provided
1005
- if (foreground && background) {
1006
- const fg = parseHex(foreground);
1007
- const bg = parseHex(background);
1008
- if (fg && bg) {
1009
- const fgLum = relativeLuminance(...fg);
1010
- const bgLum = relativeLuminance(...bg);
1011
- const ratio = contrastRatio(fgLum, bgLum);
1012
- const isLarge = large_text ?? false;
1013
- const level = wcagLevel(ratio, isLarge);
1014
- const aaThreshold = isLarge ? 3 : 4.5;
1015
- const aaaThreshold = isLarge ? 4.5 : 7;
1016
- output.push(
1017
- "",
1018
- "--- Contrast Check ---",
1019
- "",
1020
- `Foreground: ${toHex(...fg)} Background: ${toHex(...bg)}`,
1021
- `Ratio: ${ratio.toFixed(2)}:1 AA: ${ratio >= aaThreshold ? "PASS" : "FAIL"} AAA: ${ratio >= aaaThreshold ? "PASS" : "FAIL"} Result: ${level}`,
1022
- );
1023
- if (ratio < aaThreshold) {
1024
- const suggested = findNearestPassing(fg, bg, aaThreshold);
1025
- output.push(`Nearest AA-passing foreground: ${suggested}`);
1026
- }
1027
- }
1028
- }
1029
-
1030
1167
  return {
1031
1168
  content: [{ type: "text", text: output.join("\n") }],
1032
1169
  };
@@ -413,25 +413,33 @@ function formatComparison(
413
413
  lines.push("");
414
414
  }
415
415
 
416
+ lines.push("--- Regressed ---", "");
416
417
  if (regressed.length > 0) {
417
- lines.push("--- Regressed ---", "");
418
418
  for (const m of regressed) {
419
419
  const keys = m.keys.map((k, i) => `${dimensions[i] ?? "key"}=${k}`).join(" | ");
420
420
  const posChange = m.prevPos > 0 ? ` | Position: ${m.prevPos.toFixed(1)} → ${m.curPos.toFixed(1)}` : "";
421
421
  lines.push(`${keys}`);
422
422
  lines.push(` Clicks: ${m.prevClicks} → ${m.curClicks} (${pctChange(m.curClicks, m.prevClicks)})${posChange}`);
423
423
  }
424
- lines.push("");
424
+ } else {
425
+ lines.push("(none)");
425
426
  }
427
+ lines.push("");
426
428
 
427
429
  return lines.join("\n");
428
430
  }
429
431
 
430
432
  function daysAgo(n: number): string {
431
- // GSC dates are in PT (Pacific Time). Use UTC-8 as a stable approximation.
432
- const now = new Date(Date.now() - 8 * 60 * 60 * 1000);
433
- now.setDate(now.getDate() - n);
434
- return now.toISOString().split("T")[0];
433
+ // GSC dates are in Pacific Time. Use Intl to handle DST correctly.
434
+ const d = new Date();
435
+ d.setDate(d.getDate() - n);
436
+ const parts = new Intl.DateTimeFormat("en-CA", {
437
+ timeZone: "America/Los_Angeles",
438
+ year: "numeric",
439
+ month: "2-digit",
440
+ day: "2-digit",
441
+ }).format(d);
442
+ return parts; // en-CA formats as YYYY-MM-DD
435
443
  }
436
444
 
437
445
  // ── Tool registration ──
@@ -813,10 +813,10 @@ export function registerSpeedTool(server: McpServer): void {
813
813
  const cruxOrigin = origin;
814
814
 
815
815
  if (!cruxUrl && !cruxOrigin) {
816
- return { content: [{ type: "text" as const, text: "Error: provide either url or origin, not both." }] };
816
+ return { content: [{ type: "text" as const, text: "Error: provide url or origin for CrUX data." }] };
817
817
  }
818
818
  if (cruxUrl && cruxOrigin) {
819
- return { content: [{ type: "text" as const, text: "Error: provide either url or origin, not both." }] };
819
+ return { content: [{ type: "text" as const, text: "Error: provide url or origin, not both." }] };
820
820
  }
821
821
 
822
822
  try {
@@ -860,10 +860,10 @@ export function registerSpeedTool(server: McpServer): void {
860
860
  const histOrigin = origin;
861
861
 
862
862
  if (!histUrl && !histOrigin) {
863
- return { content: [{ type: "text" as const, text: "Error: provide either url or origin, not both." }] };
863
+ return { content: [{ type: "text" as const, text: "Error: provide url or origin for CrUX history." }] };
864
864
  }
865
865
  if (histUrl && histOrigin) {
866
- return { content: [{ type: "text" as const, text: "Error: provide either url or origin, not both." }] };
866
+ return { content: [{ type: "text" as const, text: "Error: provide url or origin, not both." }] };
867
867
  }
868
868
 
869
869
  try {