pagesight 0.16.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +17 -83
  2. package/docs/credentials.md +54 -0
  3. package/docs/diagnostics.md +85 -0
  4. package/docs/snapshots.md +123 -0
  5. package/docs/usage.md +130 -0
  6. package/package.json +26 -29
  7. package/src/api/bing.ts +83 -0
  8. package/src/api/compare-snapshots.ts +377 -0
  9. package/src/api/discover.ts +39 -0
  10. package/src/api/doctor.ts +35 -0
  11. package/src/api/evidence-schema.ts +78 -0
  12. package/src/api/evidence.ts +99 -0
  13. package/src/api/execute.ts +147 -0
  14. package/src/api/index.ts +5 -0
  15. package/src/api/reports.ts +90 -0
  16. package/src/api/schema.ts +207 -0
  17. package/src/api/snapshot.ts +183 -0
  18. package/src/api/ui-findings.ts +68 -0
  19. package/src/cli.ts +182 -0
  20. package/src/http.ts +39 -0
  21. package/src/index.ts +8 -25
  22. package/src/mcp-server.ts +27 -0
  23. package/src/mcp.ts +5 -0
  24. package/src/providers/bing.ts +48 -0
  25. package/src/{lib → providers}/crux.ts +4 -9
  26. package/src/providers/ga.ts +114 -0
  27. package/src/providers/google-tokens.ts +86 -0
  28. package/src/providers/gsc-auth.ts +93 -0
  29. package/src/{lib → providers}/gsc.ts +27 -24
  30. package/src/{lib/psi.ts → providers/pagespeed.ts} +3 -14
  31. package/src/shared/dates.ts +32 -0
  32. package/src/shared/http.ts +63 -0
  33. package/src/tools/ai.ts +19 -27
  34. package/src/tools/audit.ts +26 -97
  35. package/src/tools/observe.ts +28 -0
  36. package/src/tools/page/analyze.ts +194 -0
  37. package/src/tools/page/batch.ts +163 -0
  38. package/src/tools/page/contrast.ts +128 -0
  39. package/src/tools/page/links.ts +200 -0
  40. package/src/tools/page/metadata.ts +225 -0
  41. package/src/tools/page/structured-data.ts +288 -0
  42. package/src/tools/page/tool.ts +48 -0
  43. package/src/tools/search/actions.ts +56 -0
  44. package/src/tools/search/analytics.ts +266 -0
  45. package/src/tools/search/coverage.ts +247 -0
  46. package/src/tools/search/gaps.ts +129 -0
  47. package/src/tools/search/inspection.ts +160 -0
  48. package/src/tools/search/result.ts +3 -0
  49. package/src/tools/search/sample.ts +110 -0
  50. package/src/tools/search/schema.ts +62 -0
  51. package/src/{lib/sitemap.ts → tools/search/sitemap-sampling.ts} +1 -45
  52. package/src/tools/search/sites.ts +86 -0
  53. package/src/tools/search/tool.ts +11 -0
  54. package/src/tools/setup.ts +59 -20
  55. package/src/tools/speed/analyze.ts +191 -0
  56. package/src/tools/speed/batch.ts +265 -0
  57. package/src/tools/speed/crux.ts +176 -0
  58. package/src/tools/speed/pagespeed.ts +273 -0
  59. package/src/tools/speed/schema.ts +39 -0
  60. package/src/tools/speed/tool.ts +11 -0
  61. package/src/web/fetch.ts +31 -0
  62. package/src/web/images.ts +62 -0
  63. package/src/web/page-observation.ts +69 -0
  64. package/src/{lib → web}/robots.ts +20 -12
  65. package/src/web/sitemap-inventory.ts +64 -0
  66. package/src/web/sitemap-parser.ts +59 -0
  67. package/src/lib/auth.ts +0 -187
  68. package/src/tools/page.ts +0 -1241
  69. package/src/tools/search.ts +0 -852
  70. package/src/tools/speed.ts +0 -956
@@ -0,0 +1,273 @@
1
+ import { type PsiAudit, type PsiAuditDetailItem, type PsiResult } from "../../providers/pagespeed.js";
2
+ export const QUOTA_NOTE =
3
+ "\n\nNote: No GOOGLE_API_KEY configured — using shared quota (400 req/day). Set your own key to avoid rate limits.";
4
+
5
+ function scoreLabel(score: number | null): string {
6
+ if (score === null) return "N/A";
7
+ const pct = Math.round(score * 100);
8
+ if (pct >= 90) return `${pct} (good)`;
9
+ if (pct >= 50) return `${pct} (needs improvement)`;
10
+ return `${pct} (poor)`;
11
+ }
12
+
13
+ export function scorePct(score: number | null): number | null {
14
+ return score === null ? null : Math.round(score * 100);
15
+ }
16
+
17
+ function cwvRating(category: string): string {
18
+ if (category === "FAST") return "good";
19
+ if (category === "AVERAGE") return "needs improvement";
20
+ if (category === "SLOW") return "poor";
21
+ return category;
22
+ }
23
+
24
+ function formatLoadingExperience(label: string, exp: PsiResult["loadingExperience"]): string[] {
25
+ if (!exp?.metrics || Object.keys(exp.metrics).length === 0) return [];
26
+
27
+ const lines: string[] = [`--- ${label} (CrUX Field Data) ---`, ""];
28
+ lines.push(`Overall: ${cwvRating(exp.overall_category)}`, "");
29
+
30
+ const metricNames: Record<string, string> = {
31
+ CUMULATIVE_LAYOUT_SHIFT_SCORE: "CLS",
32
+ EXPERIMENTAL_TIME_TO_FIRST_BYTE: "TTFB",
33
+ FIRST_CONTENTFUL_PAINT_MS: "FCP",
34
+ FIRST_INPUT_DELAY_MS: "FID",
35
+ INTERACTION_TO_NEXT_PAINT: "INP",
36
+ LARGEST_CONTENTFUL_PAINT_MS: "LCP",
37
+ };
38
+
39
+ for (const [key, metric] of Object.entries(exp.metrics)) {
40
+ const name = metricNames[key] ?? key;
41
+ const unit = key.includes("LAYOUT_SHIFT") ? "" : "ms";
42
+ const value = key.includes("LAYOUT_SHIFT") ? (metric.percentile / 100).toFixed(2) : `${metric.percentile}${unit}`;
43
+ lines.push(`${name}: ${value} (${cwvRating(metric.category)})`);
44
+ }
45
+
46
+ return lines;
47
+ }
48
+
49
+ function formatOpportunities(audits: Record<string, PsiAudit>): string[] {
50
+ const opportunities: PsiAudit[] = [];
51
+
52
+ for (const audit of Object.values(audits)) {
53
+ if (audit.score === null || audit.score >= 1) continue;
54
+ const mode = audit.scoreDisplayMode;
55
+ const hasItems = (audit.details?.items?.length ?? 0) > 0;
56
+ const hasNumeric = audit.numericValue && audit.numericValue > 0;
57
+ if (mode === "metricSavings" || ((mode === "numeric" || mode === "binary") && hasNumeric)) {
58
+ if (hasNumeric || hasItems) {
59
+ opportunities.push(audit);
60
+ }
61
+ }
62
+ }
63
+
64
+ if (opportunities.length === 0) return [];
65
+
66
+ opportunities.sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
67
+
68
+ const lines: string[] = ["--- Opportunities ---", ""];
69
+ for (const audit of opportunities.slice(0, 10)) {
70
+ const severity = (audit.score ?? 0) < 0.5 ? "HIGH" : (audit.score ?? 0) < 0.9 ? "MEDIUM" : "LOW";
71
+ const unit = audit.numericUnit === "millisecond" ? "ms" : audit.numericUnit === "byte" ? " bytes" : "";
72
+ const savings = audit.displayValue ?? `${Math.round(audit.numericValue ?? 0)}${unit}`;
73
+ lines.push(`${severity} ${audit.title}`);
74
+ lines.push(` Potential savings: ${savings}`);
75
+
76
+ const items = audit.details?.items;
77
+ if (items && items.length > 0) {
78
+ for (const item of items.slice(0, 3)) {
79
+ lines.push(...formatDetailItem(item));
80
+ }
81
+ if (items.length > 3) {
82
+ lines.push(` ... and ${items.length - 3} more resources`);
83
+ }
84
+ }
85
+ lines.push("");
86
+ }
87
+
88
+ return lines;
89
+ }
90
+
91
+ function formatDiagnostics(audits: Record<string, PsiAudit>): string[] {
92
+ const failing: PsiAudit[] = [];
93
+
94
+ for (const audit of Object.values(audits)) {
95
+ if (
96
+ audit.score !== null &&
97
+ audit.score < 0.5 &&
98
+ (audit.scoreDisplayMode === "numeric" || audit.scoreDisplayMode === "metricSavings") &&
99
+ audit.displayValue
100
+ ) {
101
+ failing.push(audit);
102
+ }
103
+ }
104
+
105
+ if (failing.length === 0) return [];
106
+
107
+ failing.sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
108
+
109
+ const lines: string[] = ["--- Diagnostics ---", ""];
110
+ for (const audit of failing.slice(0, 10)) {
111
+ lines.push(`${audit.title}: ${audit.displayValue}`);
112
+
113
+ const linkMatch = audit.description?.match(/\[.*?\]\((https?:\/\/[^)]+)\)/);
114
+ if (linkMatch) lines.push(` Learn more: ${linkMatch[1]}`);
115
+
116
+ const items = audit.details?.items;
117
+ if (items && items.length > 0) {
118
+ for (const item of items.slice(0, 3)) {
119
+ lines.push(...formatDetailItem(item));
120
+ }
121
+ if (items.length > 3) {
122
+ lines.push(` ... and ${items.length - 3} more`);
123
+ }
124
+ }
125
+ lines.push("");
126
+ }
127
+
128
+ return lines;
129
+ }
130
+
131
+ function formatDetailItem(item: PsiAuditDetailItem): string[] {
132
+ const lines: string[] = [];
133
+
134
+ if (item.node) {
135
+ const n = item.node;
136
+ if (n.nodeLabel) lines.push(` Element: "${n.nodeLabel}"`);
137
+ if (n.selector) lines.push(` Selector: ${n.selector}`);
138
+ if (n.snippet) lines.push(` HTML: ${n.snippet}`);
139
+ if (n.explanation) lines.push(` Issue: ${n.explanation}`);
140
+ } else if (item.url) {
141
+ const parts = [` ${item.url}`];
142
+ if (item.wastedMs) parts.push(`wastedMs=${Math.round(item.wastedMs)}`);
143
+ if (item.wastedBytes) parts.push(`wastedBytes=${Math.round(item.wastedBytes)}`);
144
+ if (item.totalBytes) parts.push(`totalBytes=${Math.round(item.totalBytes)}`);
145
+ lines.push(parts.join(" "));
146
+ }
147
+
148
+ return lines;
149
+ }
150
+
151
+ function formatFailingAudits(audits: Record<string, PsiAudit>, categoryRefs: string[]): string[] {
152
+ const failing: PsiAudit[] = [];
153
+
154
+ for (const ref of categoryRefs) {
155
+ const audit = audits[ref];
156
+ if (audit && audit.score !== null && audit.score < 1) {
157
+ failing.push(audit);
158
+ }
159
+ }
160
+
161
+ if (failing.length === 0) return [];
162
+
163
+ failing.sort((a, b) => (a.score ?? 0) - (b.score ?? 0));
164
+
165
+ const lines: string[] = [];
166
+ for (const audit of failing) {
167
+ const severity = audit.score === 0 ? "FAIL" : (audit.score ?? 0) < 0.5 ? "WARN" : "INFO";
168
+ lines.push(`[${severity}] ${audit.title}`);
169
+ if (audit.displayValue) lines.push(` Value: ${audit.displayValue}`);
170
+
171
+ // Extract learn-more URL from description markdown
172
+ const linkMatch = audit.description?.match(/\[.*?\]\((https?:\/\/[^)]+)\)/);
173
+ if (linkMatch) lines.push(` Learn more: ${linkMatch[1]}`);
174
+
175
+ const items = audit.details?.items;
176
+ if (items && items.length > 0) {
177
+ const maxItems = 5;
178
+ for (const item of items.slice(0, maxItems)) {
179
+ lines.push(...formatDetailItem(item));
180
+ }
181
+ if (items.length > maxItems) {
182
+ lines.push(` ... and ${items.length - maxItems} more`);
183
+ }
184
+ }
185
+ lines.push("");
186
+ }
187
+
188
+ return lines;
189
+ }
190
+
191
+ export function formatPagespeed(url: string, result: PsiResult): string {
192
+ const lhr = result.lighthouseResult;
193
+ const lines: string[] = [
194
+ `=== PageSpeed: ${url} ===`,
195
+ `Strategy: ${lhr.configSettings.emulatedFormFactor}`,
196
+ `Lighthouse: ${lhr.lighthouseVersion}`,
197
+ `Analyzed: ${result.analysisUTCTimestamp}`,
198
+ "",
199
+ ];
200
+
201
+ // Runtime errors
202
+ if (lhr.runtimeError) {
203
+ lines.push(`ERROR: ${lhr.runtimeError.code} — ${lhr.runtimeError.message}`, "");
204
+ }
205
+
206
+ // Warnings
207
+ if (lhr.runWarnings && lhr.runWarnings.length > 0) {
208
+ for (const w of lhr.runWarnings) {
209
+ lines.push(`WARNING: ${w}`);
210
+ }
211
+ lines.push("");
212
+ }
213
+
214
+ // Category scores
215
+ lines.push("--- Scores ---", "");
216
+ for (const cat of Object.values(lhr.categories)) {
217
+ lines.push(`${cat.title}: ${scoreLabel(cat.score)}`);
218
+ }
219
+ lines.push("");
220
+
221
+ // Lab performance metrics
222
+ const cwvIds = [
223
+ "first-contentful-paint",
224
+ "largest-contentful-paint",
225
+ "total-blocking-time",
226
+ "cumulative-layout-shift",
227
+ "speed-index",
228
+ "interactive",
229
+ ];
230
+ const cwvLines: string[] = [];
231
+ for (const id of cwvIds) {
232
+ const audit = lhr.audits[id];
233
+ if (audit?.displayValue) {
234
+ cwvLines.push(`${audit.title}: ${audit.displayValue} ${scoreLabel(audit.score)}`);
235
+ }
236
+ }
237
+ if (cwvLines.length > 0) {
238
+ lines.push("--- Lab Performance Metrics ---", "", ...cwvLines, "");
239
+ }
240
+
241
+ // CrUX field data
242
+ const pageExp = formatLoadingExperience("Page", result.loadingExperience);
243
+ if (pageExp.length > 0) lines.push(...pageExp, "");
244
+
245
+ const originExp = formatLoadingExperience("Origin", result.originLoadingExperience);
246
+ if (originExp.length > 0) lines.push(...originExp, "");
247
+
248
+ // Opportunities
249
+ const opps = formatOpportunities(lhr.audits);
250
+ if (opps.length > 0) lines.push(...opps, "");
251
+
252
+ // Diagnostics
253
+ const diags = formatDiagnostics(lhr.audits);
254
+ if (diags.length > 0) lines.push(...diags, "");
255
+
256
+ // Failing audits per non-performance category (a11y, SEO, best practices)
257
+ for (const cat of Object.values(lhr.categories)) {
258
+ if (cat.id === "performance" || !cat.auditRefs) continue;
259
+ const score = cat.score !== null ? Math.round(cat.score * 100) : null;
260
+ if (score === null || score >= 100) continue;
261
+
262
+ const refs = cat.auditRefs.map((r) => r.id);
263
+ const details = formatFailingAudits(lhr.audits, refs);
264
+ if (details.length > 0) {
265
+ lines.push(`--- ${cat.title} Issues ---`, "", ...details);
266
+ }
267
+ }
268
+
269
+ // Timing
270
+ lines.push(`Analysis took ${(lhr.timing.total / 1000).toFixed(1)}s`);
271
+
272
+ return lines.join("\n");
273
+ }
@@ -0,0 +1,39 @@
1
+ import { z } from "zod";
2
+ export const speedSchema = {
3
+ action: z
4
+ .enum(["pagespeed", "crux", "crux_history"])
5
+ .optional()
6
+ .describe("Which analysis to run. Auto-detected: 'pagespeed' when url/urls provided, 'crux' when origin provided."),
7
+ url: z.string().url().optional().describe("URL to analyze (PageSpeed or CrUX)."),
8
+ urls: z
9
+ .array(z.string().url())
10
+ .min(2)
11
+ .max(10)
12
+ .optional()
13
+ .describe("Multiple URLs (2-10) for batch PageSpeed. 2 = compare, 3+ = summary table."),
14
+ strategy: z.enum(["mobile", "desktop"]).optional().describe("Device strategy for PageSpeed. Default: 'mobile'."),
15
+ categories: z
16
+ .array(z.enum(["performance", "accessibility", "best-practices", "seo"]))
17
+ .optional()
18
+ .describe("Lighthouse categories. Default: all four."),
19
+ locale: z.string().optional().describe("Locale for PageSpeed results."),
20
+ origin: z.string().optional().describe("Origin for CrUX data (e.g., 'https://example.com'). Triggers CrUX mode."),
21
+ form_factor: z.enum(["DESKTOP", "PHONE", "TABLET"]).optional().describe("CrUX device filter."),
22
+ metrics: z
23
+ .array(
24
+ z.enum([
25
+ "cumulative_layout_shift",
26
+ "first_contentful_paint",
27
+ "interaction_to_next_paint",
28
+ "largest_contentful_paint",
29
+ "experimental_time_to_first_byte",
30
+ "round_trip_time",
31
+ "navigation_types",
32
+ "form_factors",
33
+ ]),
34
+ )
35
+ .optional()
36
+ .describe("CrUX metrics to query."),
37
+ periods: z.number().min(1).max(40).optional().describe("CrUX history periods (1-40). Default: 25."),
38
+ };
39
+ export type SpeedOptions = z.infer<z.ZodObject<typeof speedSchema>>;
@@ -0,0 +1,11 @@
1
+ import type { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
+ import { speedSchema } from "./schema.js";
3
+ import { analyzeSpeed } from "./analyze.js";
4
+ export function registerSpeedTool(server: McpServer): void {
5
+ server.tool(
6
+ "speed",
7
+ "Analyze site performance. Run PageSpeed Insights (lab metrics, Lighthouse scores, opportunities) for single or multiple URLs, or query Chrome UX Report for real-world field data and historical trends.",
8
+ speedSchema,
9
+ analyzeSpeed,
10
+ );
11
+ }
@@ -0,0 +1,31 @@
1
+ import { readBounded, RequestError } from "../shared/http.js";
2
+
3
+ export async function fetchText(url: string, maxBytes: number, origin?: string) {
4
+ const redirects: Array<{ url: string; status: number; location: string }> = [];
5
+ let current = url;
6
+ const signal = AbortSignal.timeout(20_000);
7
+ for (let hop = 0; hop <= 5; hop++) {
8
+ const parsed = new URL(current);
9
+ if (
10
+ !["http:", "https:"].includes(parsed.protocol) ||
11
+ parsed.username ||
12
+ parsed.password ||
13
+ (origin && parsed.origin !== origin)
14
+ )
15
+ throw new RequestError("Redirect left the permitted origin or protocol", null, "invalid_redirect");
16
+ const response = await fetch(current, {
17
+ signal,
18
+ redirect: "manual",
19
+ headers: { "User-Agent": "Pagesight/0.17", Accept: "text/html,application/xml,text/plain" },
20
+ });
21
+ const location = response.headers.get("location");
22
+ if ([301, 302, 303, 307, 308].includes(response.status) && location) {
23
+ redirects.push({ url: current, status: response.status, location });
24
+ await response.body?.cancel();
25
+ current = new URL(location, current).href;
26
+ continue;
27
+ }
28
+ return { response, body: await readBounded(response, maxBytes), finalUrl: current, redirects };
29
+ }
30
+ throw new RequestError("Redirect limit exceeded", null, "redirect_limit");
31
+ }
@@ -0,0 +1,62 @@
1
+ /** Inspect fetched markup without executing scripts or requesting image URLs. */
2
+ export async function imageEvidence(html: string) {
3
+ const images: Array<{
4
+ src: string | null;
5
+ altPresent: boolean;
6
+ altText: string | null;
7
+ inNoscript: boolean;
8
+ width: string | null;
9
+ height: string | null;
10
+ role: string | null;
11
+ ariaHidden: string | null;
12
+ }> = [];
13
+ let omittedImages = 0;
14
+ const noscript: string[] = [];
15
+ let currentNoscript = "";
16
+ const handler = (inNoscript: boolean) => ({
17
+ element(el: HTMLRewriterTypes.Element) {
18
+ if (images.length >= 200) {
19
+ omittedImages++;
20
+ return;
21
+ }
22
+ const altPresent = el.hasAttribute("alt");
23
+ images.push({
24
+ src: el.getAttribute("src"),
25
+ altPresent,
26
+ altText: altPresent ? (el.getAttribute("alt") ?? "") : null,
27
+ inNoscript,
28
+ width: el.getAttribute("width"),
29
+ height: el.getAttribute("height"),
30
+ role: el.getAttribute("role"),
31
+ ariaHidden: el.getAttribute("aria-hidden"),
32
+ });
33
+ },
34
+ });
35
+ await new HTMLRewriter()
36
+ .on("img", handler(false))
37
+ .on("noscript", {
38
+ element(el) {
39
+ currentNoscript = "";
40
+ el.onEndTag(() => {
41
+ noscript.push(currentNoscript);
42
+ currentNoscript = "";
43
+ });
44
+ },
45
+ text(chunk) {
46
+ currentNoscript += chunk.text;
47
+ },
48
+ })
49
+ .transform(new Response(html))
50
+ .text();
51
+ if (currentNoscript) noscript.push(currentNoscript);
52
+ for (const fragment of noscript)
53
+ await new HTMLRewriter().on("img", handler(true)).transform(new Response(fragment)).text();
54
+ return {
55
+ images,
56
+ omittedImages,
57
+ truncated: omittedImages > 0,
58
+ warnings: [
59
+ "Images are fetched HTML evidence, including noscript fallback markup; scripts and image URLs were not loaded. Empty ALT may be appropriate for decorative images; meaning requires context.",
60
+ ],
61
+ };
62
+ }
@@ -0,0 +1,69 @@
1
+ import { imageEvidence } from "./images.js";
2
+ import { fetchText } from "./fetch.js";
3
+
4
+ export async function observePage(url: string) {
5
+ const { response, body, finalUrl, redirects } = await fetchText(url, 2_000_000);
6
+ const warnings = ["Fetched HTML only; JavaScript execution and crawler access were not tested."];
7
+ let title = "";
8
+ const metadata: { canonical: string | null; description: string | null } = { canonical: null, description: null };
9
+ const robots: string[] = [];
10
+ const data: string[] = [];
11
+ const rewriter = new HTMLRewriter()
12
+ .on("title", {
13
+ text(chunk) {
14
+ title += chunk.text;
15
+ },
16
+ })
17
+ .on('link[rel="canonical"]', {
18
+ element(el) {
19
+ const href = el.getAttribute("href");
20
+ if (href) {
21
+ try {
22
+ metadata.canonical = new URL(href, finalUrl).href;
23
+ } catch {
24
+ warnings.push("Malformed canonical href; canonical is unknown.");
25
+ }
26
+ }
27
+ },
28
+ })
29
+ .on("meta", {
30
+ element(el) {
31
+ const name = el.getAttribute("name")?.toLowerCase();
32
+ const value = el.getAttribute("content");
33
+ if (name === "description") metadata.description = value;
34
+ if ((name === "robots" || name === "googlebot") && value) robots.push(`${name}: ${value}`);
35
+ },
36
+ })
37
+ .on('script[type="application/ld+json"]', {
38
+ text(chunk) {
39
+ if (!data.length) data.push("");
40
+ data[data.length - 1] += chunk.text;
41
+ if (chunk.lastInTextNode) data.push("");
42
+ },
43
+ });
44
+ await rewriter.transform(new Response(body)).text();
45
+ return {
46
+ url,
47
+ finalUrl,
48
+ status: response.status,
49
+ redirects,
50
+ contentType: response.headers.get("content-type"),
51
+ xRobotsTag: response.headers.get("x-robots-tag"),
52
+ title: title.trim() || null,
53
+ description: metadata.description,
54
+ descriptionLength: metadata.description === null ? null : Array.from(metadata.description).length,
55
+ canonical: metadata.canonical,
56
+ robots,
57
+ sha256: Bun.CryptoHasher.hash("sha256", body, "hex"),
58
+ bytes: Buffer.byteLength(body),
59
+ structuredData: data.filter(Boolean).map((text) => {
60
+ try {
61
+ return { value: JSON.parse(text), validJson: true };
62
+ } catch {
63
+ return { value: null, validJson: false };
64
+ }
65
+ }),
66
+ imageEvidence: await imageEvidence(body),
67
+ warnings,
68
+ };
69
+ }
@@ -1,5 +1,5 @@
1
1
  /**
2
- * robots.txt parser per RFC 9309
2
+ * robots.txt parser based on RFC 9309
3
3
  * https://www.rfc-editor.org/rfc/rfc9309
4
4
  *
5
5
  * AI crawler registry sourced from:
@@ -152,17 +152,24 @@ export function parseRobotsTxt(raw: string): RobotsTxt {
152
152
  return { groups, sitemaps, raw, errors };
153
153
  }
154
154
 
155
- // --- Matching (per RFC 9309) ---
155
+ // --- Matching ---
156
+
157
+ // RFC 9309 §2.2.2: decode only unreserved ASCII; preserve reserved escapes on both sides.
158
+ function normalizePath(path: string): string {
159
+ return path.replace(/%[0-9a-f]{2}|\P{ASCII}/giu, (part) => {
160
+ if (part.startsWith("%")) {
161
+ const decoded = String.fromCharCode(Number.parseInt(part.slice(1), 16));
162
+ return /^[a-z0-9._~-]$/i.test(decoded) ? decoded : part.toUpperCase();
163
+ }
164
+ return encodeURIComponent(part.toWellFormed());
165
+ });
166
+ }
156
167
 
157
168
  function pathMatches(pattern: string, path: string): boolean {
158
169
  if (!pattern) return false;
159
170
 
160
- // RFC 9309 §2.2.2: decode percent-encoded characters for comparison
161
- try {
162
- path = decodeURIComponent(path);
163
- } catch {
164
- // malformed encoding, use as-is
165
- }
171
+ path = normalizePath(path);
172
+ pattern = normalizePath(pattern);
166
173
 
167
174
  let regex = "^";
168
175
  for (let i = 0; i < pattern.length; i++) {
@@ -231,7 +238,7 @@ export function isAllowed(
231
238
 
232
239
  for (const rule of matchedRules) {
233
240
  if (pathMatches(rule.path, path)) {
234
- const ruleLength = rule.path.length;
241
+ const ruleLength = normalizePath(rule.path).length;
235
242
  if (ruleLength > bestLength || (ruleLength === bestLength && rule.type === "allow")) {
236
243
  bestRule = rule;
237
244
  bestLength = ruleLength;
@@ -278,14 +285,15 @@ export async function fetchRobotsTxt(origin: string): Promise<{ robotsTxt: Robot
278
285
  redirect: "follow",
279
286
  });
280
287
 
281
- // RFC 9309: 4xx (except 429) = no restrictions (allow all)
282
- // 5xx and 429 = assume complete disallow
288
+ // RFC 9309 §§2.3.1.3–4 distinguish unavailable (4xx) from unreachable (5xx).
289
+ // Pagesight treats 429 conservatively; Google also excludes it from the allow-on-4xx rule.
290
+ // https://developers.google.com/search/docs/crawling-indexing/robots/robots_txt#handling-http-status-codes
283
291
  if (res.status >= 500 || res.status === 429) {
284
292
  const disallowAll: RobotsTxt = {
285
293
  groups: [{ userAgents: ["*"], rules: [{ type: "disallow", path: "/" }] }],
286
294
  sitemaps: [],
287
295
  raw: "",
288
- errors: [`Server returned ${res.status} — treating as full disallow per RFC 9309`],
296
+ errors: [`Server returned ${res.status} — treating as full disallow`],
289
297
  };
290
298
  return { robotsTxt: disallowAll, statusCode: res.status };
291
299
  }
@@ -0,0 +1,64 @@
1
+ import { RequestError } from "../shared/http.js";
2
+ import { parseInventorySitemap } from "./sitemap-parser.js";
3
+ import { fetchText } from "./fetch.js";
4
+
5
+ export async function observeSitemap(url: string) {
6
+ const origin = new URL(url).origin;
7
+ const queue = [url];
8
+ const scheduled = new Set(queue);
9
+ let omittedChildReferences = 0;
10
+ const visited = new Set<string>();
11
+ const urls = new Set<string>();
12
+ const documents: Array<{ url: string; status: number; sha256: string; bytes: number }> = [];
13
+ const errors: Array<{ url: string; code: string }> = [];
14
+ const futureLastmods = new Set<string>();
15
+ const today = new Date().toISOString().slice(0, 10);
16
+ let bytes = 0;
17
+ while (queue.length && visited.size < 5 && bytes < 8_000_000) {
18
+ const next = queue.shift();
19
+ if (!next || visited.has(next)) continue;
20
+ visited.add(next);
21
+ try {
22
+ if (new URL(next).origin !== origin)
23
+ throw new RequestError("Child sitemap is outside the site origin", null, "outside_origin");
24
+ const page = await fetchText(next, 8_000_000 - bytes, origin);
25
+ if (!page.response.ok) throw new RequestError("Sitemap fetch failed", page.response.status);
26
+ bytes += Buffer.byteLength(page.body);
27
+ const parsed = parseInventorySitemap(page.body);
28
+ documents.push({
29
+ url: page.finalUrl,
30
+ status: page.response.status,
31
+ bytes: Buffer.byteLength(page.body),
32
+ sha256: Bun.CryptoHasher.hash("sha256", page.body, "hex"),
33
+ });
34
+ for (const child of parsed.children) {
35
+ if (scheduled.has(child)) continue;
36
+ if (scheduled.size >= 5) {
37
+ omittedChildReferences++;
38
+ continue;
39
+ }
40
+ scheduled.add(child);
41
+ queue.push(child);
42
+ }
43
+ for (const item of parsed.urls) urls.add(item);
44
+ for (const lastmod of parsed.lastmods) if (lastmod.slice(0, 10) > today) futureLastmods.add(lastmod);
45
+ } catch (error) {
46
+ errors.push({ url: next, code: error instanceof RequestError ? error.code : "fetch_failed" });
47
+ }
48
+ }
49
+ return {
50
+ sitemap: url,
51
+ complete: queue.length === 0 && errors.length === 0 && omittedChildReferences === 0,
52
+ documents,
53
+ urls: [...urls],
54
+ urlCount: urls.size,
55
+ bytes,
56
+ unvisited: queue,
57
+ omittedChildReferences,
58
+ errors,
59
+ futureLastmods: [...futureLastmods],
60
+ warnings: [
61
+ "Sitemap membership is not indexing evidence. Inventory is limited to five same-origin XML documents and 8 MB.",
62
+ ],
63
+ };
64
+ }
@@ -0,0 +1,59 @@
1
+ import { SaxesParser, type SaxesTagNS } from "saxes";
2
+ import { RequestError } from "../shared/http.js";
3
+
4
+ export function parseInventorySitemap(xml: string) {
5
+ const parser = new SaxesParser({ xmlns: true });
6
+ const stack: SaxesTagNS[] = [];
7
+ const urls: string[] = [];
8
+ const children: string[] = [];
9
+ const lastmods: string[] = [];
10
+ let root = "";
11
+ let text = "";
12
+ let locations = 0;
13
+ const invalid = () => {
14
+ throw new RequestError("Expected well-formed sitemap XML", null, "invalid_sitemap");
15
+ };
16
+ parser.on("error", invalid);
17
+ parser.on("doctype", invalid);
18
+ parser.on("opentag", (tag) => {
19
+ stack.push(tag);
20
+ if (stack.length === 1) {
21
+ root = tag.local;
22
+ if (tag.prefix)
23
+ throw new RequestError(
24
+ "Prefixed sitemap XML is not supported by this inventory reader",
25
+ null,
26
+ "unsupported_sitemap",
27
+ );
28
+ if (
29
+ !["urlset", "sitemapindex"].includes(root) ||
30
+ (tag.uri && tag.uri !== "http://www.sitemaps.org/schemas/sitemap/0.9")
31
+ )
32
+ invalid();
33
+ }
34
+ if (stack.length === 2) {
35
+ if (tag.local !== (root === "urlset" ? "url" : "sitemap") || tag.uri !== stack[0].uri) invalid();
36
+ locations = 0;
37
+ }
38
+ if (stack.length === 3) text = "";
39
+ if (stack.length > 3 && ["loc", "lastmod"].includes(stack[2].local) && stack[2].uri === stack[0].uri) invalid();
40
+ });
41
+ const append = (value: string) => {
42
+ if (stack.length === 3) text += value;
43
+ };
44
+ parser.on("text", append);
45
+ parser.on("cdata", append);
46
+ parser.on("closetag", (tag) => {
47
+ if (stack.length === 3 && tag.uri === stack[0].uri) {
48
+ if (tag.local === "loc") {
49
+ if (++locations !== 1 || !text.trim()) invalid();
50
+ (root === "urlset" ? urls : children).push(text.trim());
51
+ }
52
+ if (tag.local === "lastmod") lastmods.push(text.trim());
53
+ }
54
+ if (stack.length === 2 && locations !== 1) invalid();
55
+ stack.pop();
56
+ });
57
+ parser.write(xml).close();
58
+ return { urls, children, lastmods };
59
+ }