@houtini/seo-audit-console 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/LICENSE +92 -0
  2. package/README.md +211 -0
  3. package/dist/audit/checks.d.ts +35 -0
  4. package/dist/audit/checks.d.ts.map +1 -0
  5. package/dist/audit/checks.js +1475 -0
  6. package/dist/audit/checks.js.map +1 -0
  7. package/dist/audit/drift.d.ts +40 -0
  8. package/dist/audit/drift.d.ts.map +1 -0
  9. package/dist/audit/drift.js +148 -0
  10. package/dist/audit/drift.js.map +1 -0
  11. package/dist/audit/engine.d.ts +33 -0
  12. package/dist/audit/engine.d.ts.map +1 -0
  13. package/dist/audit/engine.js +186 -0
  14. package/dist/audit/engine.js.map +1 -0
  15. package/dist/audit/opportunities.d.ts +23 -0
  16. package/dist/audit/opportunities.d.ts.map +1 -0
  17. package/dist/audit/opportunities.js +149 -0
  18. package/dist/audit/opportunities.js.map +1 -0
  19. package/dist/audit/report.d.ts +11 -0
  20. package/dist/audit/report.d.ts.map +1 -0
  21. package/dist/audit/report.js +59 -0
  22. package/dist/audit/report.js.map +1 -0
  23. package/dist/audit/schema-validate.d.ts +35 -0
  24. package/dist/audit/schema-validate.d.ts.map +1 -0
  25. package/dist/audit/schema-validate.js +293 -0
  26. package/dist/audit/schema-validate.js.map +1 -0
  27. package/dist/audit/templates.d.ts +43 -0
  28. package/dist/audit/templates.d.ts.map +1 -0
  29. package/dist/audit/templates.js +129 -0
  30. package/dist/audit/templates.js.map +1 -0
  31. package/dist/audit/topicGaps.d.ts +42 -0
  32. package/dist/audit/topicGaps.d.ts.map +1 -0
  33. package/dist/audit/topicGaps.js +181 -0
  34. package/dist/audit/topicGaps.js.map +1 -0
  35. package/dist/core/AuditDatabase.d.ts +33 -0
  36. package/dist/core/AuditDatabase.d.ts.map +1 -0
  37. package/dist/core/AuditDatabase.js +481 -0
  38. package/dist/core/AuditDatabase.js.map +1 -0
  39. package/dist/core/Backlinks.d.ts +33 -0
  40. package/dist/core/Backlinks.d.ts.map +1 -0
  41. package/dist/core/Backlinks.js +110 -0
  42. package/dist/core/Backlinks.js.map +1 -0
  43. package/dist/core/Crawler.d.ts +23 -0
  44. package/dist/core/Crawler.d.ts.map +1 -0
  45. package/dist/core/Crawler.js +588 -0
  46. package/dist/core/Crawler.js.map +1 -0
  47. package/dist/core/DataForSeoClient.d.ts +86 -0
  48. package/dist/core/DataForSeoClient.d.ts.map +1 -0
  49. package/dist/core/DataForSeoClient.js +232 -0
  50. package/dist/core/DataForSeoClient.js.map +1 -0
  51. package/dist/core/Entities.d.ts +23 -0
  52. package/dist/core/Entities.d.ts.map +1 -0
  53. package/dist/core/Entities.js +62 -0
  54. package/dist/core/Entities.js.map +1 -0
  55. package/dist/core/GscClient.d.ts +22 -0
  56. package/dist/core/GscClient.d.ts.map +1 -0
  57. package/dist/core/GscClient.js +93 -0
  58. package/dist/core/GscClient.js.map +1 -0
  59. package/dist/core/GscSync.d.ts +20 -0
  60. package/dist/core/GscSync.d.ts.map +1 -0
  61. package/dist/core/GscSync.js +133 -0
  62. package/dist/core/GscSync.js.map +1 -0
  63. package/dist/core/JobManager.d.ts +30 -0
  64. package/dist/core/JobManager.d.ts.map +1 -0
  65. package/dist/core/JobManager.js +68 -0
  66. package/dist/core/JobManager.js.map +1 -0
  67. package/dist/core/RankTracker.d.ts +25 -0
  68. package/dist/core/RankTracker.d.ts.map +1 -0
  69. package/dist/core/RankTracker.js +78 -0
  70. package/dist/core/RankTracker.js.map +1 -0
  71. package/dist/core/Refresh.d.ts +32 -0
  72. package/dist/core/Refresh.d.ts.map +1 -0
  73. package/dist/core/Refresh.js +73 -0
  74. package/dist/core/Refresh.js.map +1 -0
  75. package/dist/core/UrlInspector.d.ts +22 -0
  76. package/dist/core/UrlInspector.d.ts.map +1 -0
  77. package/dist/core/UrlInspector.js +92 -0
  78. package/dist/core/UrlInspector.js.map +1 -0
  79. package/dist/core/WikidataClient.d.ts +19 -0
  80. package/dist/core/WikidataClient.d.ts.map +1 -0
  81. package/dist/core/WikidataClient.js +59 -0
  82. package/dist/core/WikidataClient.js.map +1 -0
  83. package/dist/core/agentReadiness.d.ts +34 -0
  84. package/dist/core/agentReadiness.d.ts.map +1 -0
  85. package/dist/core/agentReadiness.js +119 -0
  86. package/dist/core/agentReadiness.js.map +1 -0
  87. package/dist/core/ctrModel.d.ts +2 -0
  88. package/dist/core/ctrModel.d.ts.map +1 -0
  89. package/dist/core/ctrModel.js +6 -0
  90. package/dist/core/ctrModel.js.map +1 -0
  91. package/dist/core/dashboardData.d.ts +312 -0
  92. package/dist/core/dashboardData.d.ts.map +1 -0
  93. package/dist/core/dashboardData.js +550 -0
  94. package/dist/core/dashboardData.js.map +1 -0
  95. package/dist/core/dataStorage.d.ts +42 -0
  96. package/dist/core/dataStorage.d.ts.map +1 -0
  97. package/dist/core/dataStorage.js +193 -0
  98. package/dist/core/dataStorage.js.map +1 -0
  99. package/dist/core/draftBrief.d.ts +27 -0
  100. package/dist/core/draftBrief.d.ts.map +1 -0
  101. package/dist/core/draftBrief.js +69 -0
  102. package/dist/core/draftBrief.js.map +1 -0
  103. package/dist/core/extract.d.ts +67 -0
  104. package/dist/core/extract.d.ts.map +1 -0
  105. package/dist/core/extract.js +262 -0
  106. package/dist/core/extract.js.map +1 -0
  107. package/dist/core/gscFreshness.d.ts +18 -0
  108. package/dist/core/gscFreshness.d.ts.map +1 -0
  109. package/dist/core/gscFreshness.js +32 -0
  110. package/dist/core/gscFreshness.js.map +1 -0
  111. package/dist/core/linkGraph.d.ts +19 -0
  112. package/dist/core/linkGraph.d.ts.map +1 -0
  113. package/dist/core/linkGraph.js +125 -0
  114. package/dist/core/linkGraph.js.map +1 -0
  115. package/dist/core/passageScore.d.ts +19 -0
  116. package/dist/core/passageScore.d.ts.map +1 -0
  117. package/dist/core/passageScore.js +59 -0
  118. package/dist/core/passageScore.js.map +1 -0
  119. package/dist/core/paths.d.ts +5 -0
  120. package/dist/core/paths.d.ts.map +1 -0
  121. package/dist/core/paths.js +15 -0
  122. package/dist/core/paths.js.map +1 -0
  123. package/dist/core/queryData.d.ts +35 -0
  124. package/dist/core/queryData.d.ts.map +1 -0
  125. package/dist/core/queryData.js +200 -0
  126. package/dist/core/queryData.js.map +1 -0
  127. package/dist/core/reranker.d.ts +8 -0
  128. package/dist/core/reranker.d.ts.map +1 -0
  129. package/dist/core/reranker.js +68 -0
  130. package/dist/core/reranker.js.map +1 -0
  131. package/dist/core/robots.d.ts +11 -0
  132. package/dist/core/robots.d.ts.map +1 -0
  133. package/dist/core/robots.js +76 -0
  134. package/dist/core/robots.js.map +1 -0
  135. package/dist/core/sitemap.d.ts +15 -0
  136. package/dist/core/sitemap.d.ts.map +1 -0
  137. package/dist/core/sitemap.js +100 -0
  138. package/dist/core/sitemap.js.map +1 -0
  139. package/dist/core/sql.d.ts +6 -0
  140. package/dist/core/sql.d.ts.map +1 -0
  141. package/dist/core/sql.js +6 -0
  142. package/dist/core/sql.js.map +1 -0
  143. package/dist/core/types.d.ts +22 -0
  144. package/dist/core/types.d.ts.map +1 -0
  145. package/dist/core/types.js +2 -0
  146. package/dist/core/types.js.map +1 -0
  147. package/dist/core/url-key.d.ts +39 -0
  148. package/dist/core/url-key.d.ts.map +1 -0
  149. package/dist/core/url-key.js +106 -0
  150. package/dist/core/url-key.js.map +1 -0
  151. package/dist/generators/index.d.ts +45 -0
  152. package/dist/generators/index.d.ts.map +1 -0
  153. package/dist/generators/index.js +184 -0
  154. package/dist/generators/index.js.map +1 -0
  155. package/dist/index.d.ts +3 -0
  156. package/dist/index.d.ts.map +1 -0
  157. package/dist/index.js +10 -0
  158. package/dist/index.js.map +1 -0
  159. package/dist/server.d.ts +7 -0
  160. package/dist/server.d.ts.map +1 -0
  161. package/dist/server.js +1387 -0
  162. package/dist/server.js.map +1 -0
  163. package/dist/src/ui/dashboard.html +347 -0
  164. package/dist/src/ui/sync-progress.html +104 -0
  165. package/package.json +101 -0
  166. package/server.json +57 -0
@@ -0,0 +1,23 @@
1
+ export declare const isHtmlContentType: (ct: string | null | undefined) => boolean;
2
+ export interface CrawlOptions {
3
+ maxPages?: number;
4
+ maxDepth?: number;
5
+ maxConcurrency?: number;
6
+ delayMs?: number;
7
+ userAgent?: string;
8
+ excludePatterns?: string[];
9
+ }
10
+ export interface CrawlResult {
11
+ crawlId: string;
12
+ siteUrl: string;
13
+ crawled: number;
14
+ failed: number;
15
+ skipped: number;
16
+ }
17
+ /** Self-contained, stdio-safe HTTP crawler writing into a property's AuditDatabase. */
18
+ export declare class Crawler {
19
+ private readonly dataDir;
20
+ constructor(dataDir: string);
21
+ run(siteUrl: string, opts: CrawlOptions, update: (p: Record<string, unknown>) => void, signal: AbortSignal): Promise<CrawlResult>;
22
+ }
23
+ //# sourceMappingURL=Crawler.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"Crawler.d.ts","sourceRoot":"","sources":["../../src/core/Crawler.ts"],"names":[],"mappings":"AAaA,eAAO,MAAM,iBAAiB,GAAI,IAAI,MAAM,GAAG,IAAI,GAAG,SAAS,KAAG,OACE,CAAC;AAqCrE,MAAM,WAAW,YAAY;IAC3B,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;CAC5B;AAED,MAAM,WAAW,WAAW;IAC1B,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,MAAM,CAAC;IAChB,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,EAAE,MAAM,CAAC;CACjB;AA4ID,uFAAuF;AACvF,qBAAa,OAAO;IACN,OAAO,CAAC,QAAQ,CAAC,OAAO;gBAAP,OAAO,EAAE,MAAM;IAEtC,GAAG,CACP,OAAO,EAAE,MAAM,EACf,IAAI,EAAE,YAAY,EAClB,MAAM,EAAE,CAAC,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,KAAK,IAAI,EAC5C,MAAM,EAAE,WAAW,GAClB,OAAO,CAAC,WAAW,CAAC;CAkWxB"}
@@ -0,0 +1,588 @@
1
+ import { randomUUID } from 'node:crypto';
2
+ import { Agent, fetch as ufetch } from 'undici';
3
+ import { AuditDatabase } from './AuditDatabase.js';
4
+ import { dbPathFor } from './paths.js';
5
+ import { urlKey, hostFormForProperty } from './url-key.js';
6
+ import { extractPage, isInternalHost } from './extract.js';
7
+ import { fetchRobots } from './robots.js';
8
+ import { finalizeLinkGraph } from './linkGraph.js';
9
+ import { fetchSitemapUrls } from './sitemap.js';
10
+ // Robust HTML detection from a Content-Type header: take the MIME type before any
11
+ // parameters (charset, boundary), normalised — text/html and XHTML count as HTML.
12
+ const HTML_MIME_TYPES = new Set(['text/html', 'application/xhtml+xml']);
13
+ export const isHtmlContentType = (ct) => HTML_MIME_TYPES.has((ct ?? '').split(';')[0].trim().toLowerCase());
14
+ // File types we never need the body of — images, media, fonts, archives, office docs, css/js,
15
+ // pdf. We HEAD these (status + content-type only, no download) — a big bandwidth/time win on
16
+ // asset-heavy sites. We still record them (status/type) so broken-link checks work.
17
+ const ASSET_EXT = /\.(jpe?g|png|gif|webp|avif|svg|svgz|ico|bmp|tiff?|heic|psd|pdf|zip|rar|7z|gz|tgz|tar|bz2|mp4|mpeg|mpg|m4v|webm|mov|avi|mkv|flv|wmv|mp3|m4a|aac|opus|weba|wav|ogg|oga|flac|wma|css|js|mjs|cjs|map|woff2?|ttf|otf|eot|dmg|exe|msi|apk|bin|doc|docx|xls|xlsx|ppt|pptx)$/i;
18
+ const isAssetUrl = (u) => { try {
19
+ return ASSET_EXT.test(new URL(u).pathname);
20
+ }
21
+ catch {
22
+ return false;
23
+ } };
24
+ // URL patterns never worth crawling — internal site-search, infinite param spaces, cart/builder
25
+ // states and CMS infrastructure. Skipped at enqueue (never fetched) so large sites stay light
26
+ // and the crawl doesn't drown in low-value URLs. Extend per-crawl via opts.excludePatterns.
27
+ const SKIP_PATTERNS = [
28
+ /[?&](sps_query|s|q|search|keyword|orderby|sort_by|add-to-cart|fl_builder|elementor-preview)=/i,
29
+ /[?&](pf_|filter[._]|dppref|variant=)/i, // Shopify/WooCommerce faceted filters + variant duplicates
30
+ /[?&](client_id|redirect_uri|response_type)=|\/authentication\/|\/oauth\/|\/login_with_shop\//i, // auth/login flows (no SEO value)
31
+ /\/search(-results)?\//i,
32
+ /[?&]replytocom=/i,
33
+ /\/(wp-json|wp-admin|wp-login\.php|xmlrpc\.php)(\/|$|\?)/i,
34
+ /\/(cart|checkout|my-account|basket)(\/|$)/i,
35
+ ];
36
+ const skipUrl = (u, extra) => SKIP_PATTERNS.some(re => re.test(u)) || extra.some(re => re.test(u));
37
+ // Classify a fetched URL's indexability + the REASON it's not indexable, so an audit can
38
+ // report *why* Google would skip a page (status, directives, canonical, type) — not just yes/no.
39
+ function classifyIndexability(isHtml, status, noindexMeta, xRobotsTag, canonicalKey, finalKey) {
40
+ if (status >= 400)
41
+ return { indexable: 0, reason: `http-${status}` }; // 404, 410, 500, 503, …
42
+ if (status >= 300)
43
+ return { indexable: 0, reason: 'redirect' }; // redirect chain exceeded maxHops
44
+ if (!isHtml)
45
+ return { indexable: 0, reason: 'non-html' };
46
+ if (/noindex/i.test(xRobotsTag ?? ''))
47
+ return { indexable: 0, reason: 'noindex-header' }; // X-Robots-Tag
48
+ if (noindexMeta)
49
+ return { indexable: 0, reason: 'noindex-meta' };
50
+ if (canonicalKey && canonicalKey !== finalKey)
51
+ return { indexable: 0, reason: 'canonicalised' };
52
+ return { indexable: 1, reason: null };
53
+ }
54
+ const DEFAULT_UA = 'Mozilla/5.0 (compatible; seo-audit-console/0.1; +https://github.com/houtini-ai/seo-audit-console)';
55
+ const FETCH_TIMEOUT_MS = 20000;
56
+ const MAX_TRANSIENT_RETRIES = 4; // 429/503 backoff attempts before recording the status
57
+ const FLUSH_EVERY = 50;
58
+ // Per-crawl connection pool: HTTP/2 when the origin supports it (one multiplexed connection
59
+ // instead of N TCP handshakes), keep-alive reuse, bounded per-origin connections. Compression
60
+ // is negotiated explicitly (gzip/br/deflate — undici decompresses transparently; the recorded
61
+ // content_encoding header still reflects what the server actually served).
62
+ const makeDispatcher = (connections) => new Agent({ allowH2: true, connections, keepAliveTimeout: 30_000, keepAliveMaxTimeout: 60_000 });
63
+ async function fetchWithRedirects(url, ua, dispatcher, maxHops = 5) {
64
+ const redirects = [];
65
+ let current = url;
66
+ let hops = 0;
67
+ let transientRetries = 0; // 429/503 backoff budget
68
+ let netRetries = 0; // network-error (timeout/reset) budget — separate so a 503 then a timeout doesn't starve it
69
+ const start = Date.now();
70
+ // Always GET — never HEAD. Some servers/CDNs return a different status for HEAD than for GET
71
+ // (e.g. a 404 on HEAD where GET is 200, or vice-versa), which would mis-record a URL's real
72
+ // status. We still avoid downloading asset bytes by cancelling the body stream for any non-HTML
73
+ // response below, so a GET stays as light as a HEAD for images/PDFs/etc.
74
+ const method = 'GET';
75
+ while (true) {
76
+ let res;
77
+ try {
78
+ res = (await ufetch(current, {
79
+ method,
80
+ redirect: 'manual',
81
+ dispatcher,
82
+ // pages change between crawls — ask upstream caches/CDNs for the current copy
83
+ headers: {
84
+ 'user-agent': ua,
85
+ accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
86
+ 'accept-encoding': 'gzip, deflate, br',
87
+ 'cache-control': 'no-cache',
88
+ pragma: 'no-cache',
89
+ },
90
+ signal: AbortSignal.timeout(FETCH_TIMEOUT_MS),
91
+ }));
92
+ }
93
+ catch (err) {
94
+ // The server slowed past the timeout or dropped the connection. Don't lose the page to a
95
+ // transient hiccup — wait (10s, 20s, 30s) and retry before giving up, so a slowdown becomes
96
+ // latency rather than a silent coverage gap. Uses its OWN budget (not the 429/503 one) so a
97
+ // page that 503s then times out still gets full network retries; trips `throttled` so the
98
+ // crawl paces itself down. Surfaces as a failure only after the budget is exhausted.
99
+ if (netRetries < MAX_TRANSIENT_RETRIES) {
100
+ netRetries++;
101
+ await sleep(Math.min(10000 * netRetries, 30000));
102
+ continue;
103
+ }
104
+ throw err;
105
+ }
106
+ // Transient throttling (429 Too Many Requests / 503 Service Unavailable) is NOT a broken
107
+ // page — back off (respecting Retry-After) and retry, so a fast crawl doesn't poison the
108
+ // data with rate-limit statuses. Only the final status after retries is recorded.
109
+ if ((res.status === 429 || res.status === 503) && transientRetries < MAX_TRANSIENT_RETRIES) {
110
+ const ra = Number(res.headers.get('retry-after'));
111
+ const waitMs = Number.isFinite(ra) && ra > 0 ? Math.min(ra * 1000, 20000) : Math.min(1000 * 2 ** transientRetries, 8000);
112
+ try {
113
+ await res.body?.cancel();
114
+ }
115
+ catch { /* no body */ }
116
+ transientRetries++;
117
+ await sleep(waitMs);
118
+ continue;
119
+ }
120
+ if (res.status >= 300 && res.status < 400 && res.headers.get('location') && hops < maxHops) {
121
+ const loc = new URL(res.headers.get('location'), current).toString();
122
+ // Drain the redirect response's body (some servers attach HTML to 301s) so the
123
+ // keep-alive connection is released — otherwise redirect-heavy crawls leak sockets.
124
+ try {
125
+ await res.body?.cancel();
126
+ }
127
+ catch { /* already closed */ }
128
+ redirects.push({ from: current, to: loc, status: res.status });
129
+ current = loc;
130
+ hops++;
131
+ continue;
132
+ }
133
+ const contentType = res.headers.get('content-type') ?? '';
134
+ const isHtml = isHtmlContentType(contentType);
135
+ // Read the body only for HTML; for any non-HTML response (image/PDF/asset) abort the transfer
136
+ // so we never pull the bytes — keeping the always-GET fetch as cheap as a HEAD would have been.
137
+ let body = '';
138
+ if (isHtml)
139
+ body = await res.text();
140
+ else {
141
+ try {
142
+ await res.body?.cancel();
143
+ }
144
+ catch { /* already closed */ }
145
+ }
146
+ const h = (name) => res.headers.get(name);
147
+ const sec = {};
148
+ for (const [k, name] of [
149
+ ['csp', 'content-security-policy'], ['hsts', 'strict-transport-security'],
150
+ ['xFrame', 'x-frame-options'], ['xContentType', 'x-content-type-options'],
151
+ ['referrerPolicy', 'referrer-policy'], ['permissionsPolicy', 'permissions-policy'],
152
+ ['server', 'server'],
153
+ ]) {
154
+ const v = h(name);
155
+ if (v)
156
+ sec[k] = v;
157
+ }
158
+ return {
159
+ finalUrl: current,
160
+ status: res.status,
161
+ contentType,
162
+ contentEncoding: h('content-encoding'),
163
+ cacheControl: h('cache-control'),
164
+ lastModified: h('last-modified'),
165
+ etag: h('etag'),
166
+ vary: h('vary'),
167
+ securityHeaders: Object.keys(sec).length ? JSON.stringify(sec) : null,
168
+ body,
169
+ redirects,
170
+ xRobotsTag: h('x-robots-tag'),
171
+ // For HTML (body read) always the DECODED size — content-length on a compressed
172
+ // response is the wire size, and downstream transfer estimates (×0.22) assume raw.
173
+ // Non-HTML: content-length when given, else null (body isn't read; 0 would mislead).
174
+ bytes: body ? Buffer.byteLength(body) : (Number(h('content-length')) || null),
175
+ timeMs: Date.now() - start,
176
+ throttled: transientRetries > 0 || netRetries > 0,
177
+ };
178
+ }
179
+ }
180
+ const sleep = (ms) => new Promise(r => setTimeout(r, ms));
181
+ /** Self-contained, stdio-safe HTTP crawler writing into a property's AuditDatabase. */
182
+ export class Crawler {
183
+ dataDir;
184
+ constructor(dataDir) {
185
+ this.dataDir = dataDir;
186
+ }
187
+ async run(siteUrl, opts, update, signal) {
188
+ const maxPages = opts.maxPages ?? 4000;
189
+ const maxDepth = opts.maxDepth ?? 10;
190
+ // Default 8 parallel fetches (was 4) — with HTTP/2 multiplexing this is one or two
191
+ // connections, not 8 sockets. Adaptive throttling still ramps the delay on any 429/503,
192
+ // so a struggling host slows the whole pool down.
193
+ const concurrency = Math.max(1, Math.min(opts.maxConcurrency ?? 8, 24));
194
+ const delayMs = opts.delayMs ?? 250;
195
+ const ua = opts.userAgent ?? DEFAULT_UA;
196
+ const excludeRes = (opts.excludePatterns ?? []).flatMap(p => { try {
197
+ return [new RegExp(p, 'i')];
198
+ }
199
+ catch {
200
+ return [];
201
+ } });
202
+ const seed = siteUrl.startsWith('sc-domain:') ? `https://${siteUrl.slice('sc-domain:'.length)}/` : siteUrl;
203
+ const baseHost = new URL(seed).hostname;
204
+ const keyOpts = { hostForm: hostFormForProperty(siteUrl) };
205
+ const origin = new URL(seed).origin;
206
+ const dispatcher = makeDispatcher(Math.min((opts.maxConcurrency ?? 8) * 2, 32));
207
+ const db = new AuditDatabase(dbPathFor(this.dataDir, siteUrl));
208
+ db.upsertProperty(siteUrl, keyOpts.hostForm ?? 'asis');
209
+ const crawlId = randomUUID().slice(0, 8);
210
+ const startedAt = new Date().toISOString();
211
+ db.db.prepare(`INSERT INTO crawl_metadata (crawl_id, base_url, base_domain, status, max_depth, max_pages, user_agent, started_at)
212
+ VALUES (?, ?, ?, 'running', ?, ?, ?, ?)`).run(crawlId, seed, baseHost, maxDepth, maxPages, ua, startedAt);
213
+ try {
214
+ // A crawl is a fresh snapshot of the current site — prior crawl data is cleared so
215
+ // re-crawls reflect changes (the "fix → re-crawl → verify" loop) instead of
216
+ // accumulating stale pages / duplicate links. GSC/inspection/findings are separate.
217
+ // The pages/links wipe is DEFERRED until the first successful page write: if the crawl
218
+ // dies before fetching anything (site down, seed unreachable), the previous good
219
+ // snapshot survives instead of leaving every downstream consumer an empty table.
220
+ // (It must still run before the first insert — old rows would otherwise absorb the
221
+ // new crawl's writes via ON CONFLICT(url_key) DO NOTHING.)
222
+ db.db.exec('DELETE FROM errors;'); // per-crawl diagnostics, not part of the snapshot
223
+ let wipedPrior = false;
224
+ const ensureWiped = () => {
225
+ if (wipedPrior)
226
+ return;
227
+ db.db.exec('DELETE FROM links; DELETE FROM pages;');
228
+ wipedPrior = true;
229
+ };
230
+ const robots = await fetchRobots(origin, ua);
231
+ const pageInsert = db.db.prepare(`INSERT INTO pages (crawl_id, url, url_key, status_code, content_type, content_encoding, cache_control,
232
+ last_modified, etag, vary, bytes, response_time_ms, depth,
233
+ is_internal, indexable, indexable_reason, noindex, title, title_length, meta_description, meta_description_length,
234
+ h1, h1_count, word_count, lang, charset, canonical_url, canonical_key, robots, x_robots_tag, viewport,
235
+ json_ld, og_tags, hreflang, redirects, internal_links, external_links,
236
+ image_count, images_without_alt, images_missing_dimensions, canonical_count, canonical_relative,
237
+ h2_count, heading_skips, rel_next, rel_prev, rel_amphtml, mixed_content_count, twitter_tags,
238
+ has_microdata, has_rdfa, has_favicon, has_analytics, body_chunks, security_headers)
239
+ VALUES (@crawl_id,@url,@url_key,@status_code,@content_type,@content_encoding,@cache_control,
240
+ @last_modified,@etag,@vary,@bytes,@response_time_ms,@depth,
241
+ @is_internal,@indexable,@indexable_reason,@noindex,@title,@title_length,@meta_description,@meta_description_length,
242
+ @h1,@h1_count,@word_count,@lang,@charset,@canonical_url,@canonical_key,@robots,@x_robots_tag,@viewport,
243
+ @json_ld,@og_tags,@hreflang,@redirects,@internal_links,@external_links,
244
+ @image_count,@images_without_alt,@images_missing_dimensions,@canonical_count,@canonical_relative,
245
+ @h2_count,@heading_skips,@rel_next,@rel_prev,@rel_amphtml,@mixed_content_count,@twitter_tags,
246
+ @has_microdata,@has_rdfa,@has_favicon,@has_analytics,@body_chunks,@security_headers)
247
+ ON CONFLICT(url_key) DO NOTHING`);
248
+ // Robots-disallowed URLs are NOT fetched (we respect robots), but we record them as
249
+ // known + not-indexable with the reason, so the audit can report what robots is blocking.
250
+ const robotsInsert = db.db.prepare(`INSERT INTO pages (crawl_id, url, url_key, status_code, is_internal, indexable, indexable_reason, depth)
251
+ VALUES (@crawl_id,@url,@url_key,NULL,1,0,'robots-disallowed',@depth)
252
+ ON CONFLICT(url_key) DO NOTHING`);
253
+ const linkInsert = db.db.prepare(`INSERT INTO links (crawl_id, source_url, source_key, target_url, target_key, anchor_text, is_internal, placement, rel)
254
+ VALUES (@crawl_id,@source_url,@source_key,@target_url,@target_key,@anchor_text,@is_internal,@placement,@rel)`);
255
+ const errorInsert = db.db.prepare(`INSERT INTO errors (crawl_id, url, error_type, error_message) VALUES (?, ?, ?, ?)`);
256
+ const flushPages = db.db.transaction((rows) => { for (const r of rows)
257
+ pageInsert.run(r); });
258
+ const flushLinks = db.db.transaction((rows) => { for (const r of rows)
259
+ linkInsert.run(r); });
260
+ const pageBuf = [];
261
+ const linkBuf = [];
262
+ const visited = new Set();
263
+ const emitted = new Set(); // final url_keys already extracted+written
264
+ const imageRefs = new Map(); // distinct <img> src → pages using it
265
+ const frontier = [];
266
+ let crawled = 0, failed = 0, skipped = 0, discovered = 0;
267
+ const enqueue = (url, depth) => {
268
+ const key = urlKey(url, keyOpts);
269
+ if (visited.has(key))
270
+ return;
271
+ visited.add(key);
272
+ discovered++;
273
+ frontier.push({ url, depth });
274
+ };
275
+ enqueue(seed, 0);
276
+ // Seed the frontier from the XML sitemap so unlinked/orphan pages get crawled — link-following
277
+ // alone misses them (e.g. directdrivewheels: 367 sitemap URLs, only ~100 reachable via links).
278
+ // Also stores the sitemap for crawl↔sitemap reconciliation. Off-host/junk/asset URLs filtered;
279
+ // bounded by maxPages.
280
+ if (!signal.aborted) {
281
+ try {
282
+ const sm = await fetchSitemapUrls(origin, ua, keyOpts);
283
+ const smInsert = db.db.prepare(`INSERT INTO sitemap_urls (url_key, url, lastmod, fetched_at) VALUES (?,?,?,datetime('now')) ON CONFLICT(url_key) DO NOTHING`);
284
+ db.db.transaction(() => { db.db.exec('DELETE FROM sitemap_urls'); for (const u of sm.urls)
285
+ smInsert.run(u.urlKey, u.url, u.lastmod); })();
286
+ for (const u of sm.urls) {
287
+ if (discovered >= maxPages)
288
+ break;
289
+ let host;
290
+ try {
291
+ host = new URL(u.url).hostname;
292
+ }
293
+ catch {
294
+ continue;
295
+ }
296
+ if (!isInternalHost(host, baseHost) || skipUrl(u.url, excludeRes) || isAssetUrl(u.url))
297
+ continue;
298
+ enqueue(u.url, 1); // sitemap URLs are top-level entry points
299
+ }
300
+ }
301
+ catch (err) {
302
+ console.error('[crawl] sitemap seed skipped:', err instanceof Error ? err.message : String(err));
303
+ }
304
+ }
305
+ // Also seed from GSC-known URLs (pages Google sends traffic to) so coverage doesn't depend on
306
+ // the site's sitemap being complete — e.g. simracing has 334 sitemap URLs but GSC knows 1,299.
307
+ // page_key is already a normalised URL; off-host/junk/asset filtered; bounded by maxPages.
308
+ try {
309
+ // Seed with the RAW GSC page URL, not the normalised page_key — the key strips
310
+ // trailing slashes / sorts params, so fetching it can add a phantom 301 hop (or 404)
311
+ // on sites where the canonical form matters. Fall back to page_key for old rows.
312
+ for (const row of db.db.prepare(`SELECT DISTINCT COALESCE(page, page_key) AS u FROM search_analytics WHERE COALESCE(page, page_key) IS NOT NULL`).all()) {
313
+ if (discovered >= maxPages)
314
+ break;
315
+ let host;
316
+ try {
317
+ host = new URL(row.u).hostname;
318
+ }
319
+ catch {
320
+ continue;
321
+ }
322
+ if (!isInternalHost(host, baseHost) || skipUrl(row.u, excludeRes) || isAssetUrl(row.u))
323
+ continue;
324
+ enqueue(row.u, 1);
325
+ }
326
+ }
327
+ catch (err) {
328
+ console.error('[crawl] GSC seed skipped:', err instanceof Error ? err.message : String(err));
329
+ }
330
+ const flush = (final = false) => {
331
+ if (pageBuf.length || linkBuf.length)
332
+ ensureWiped();
333
+ if (pageBuf.length && (final || pageBuf.length >= FLUSH_EVERY)) {
334
+ flushPages(pageBuf.splice(0));
335
+ }
336
+ if (linkBuf.length && (final || linkBuf.length >= FLUSH_EVERY)) {
337
+ flushLinks(linkBuf.splice(0));
338
+ }
339
+ };
340
+ const saveProgress = () => {
341
+ db.db.prepare(`UPDATE crawl_metadata SET urls_discovered=?, urls_crawled=?, urls_failed=?, urls_skipped=? WHERE crawl_id=?`)
342
+ .run(discovered, crawled, failed, skipped, crawlId);
343
+ update({ crawlId, crawled, discovered, failed, skipped });
344
+ };
345
+ // Adaptive pacing: start at the configured delay; ramp UP when the host throttles (429/503)
346
+ // and decay back toward base when it's clear. Same pages crawled (no data lost) — just paced
347
+ // so hard-throttled sites stop triggering retries (net faster + cleaner) and fast sites stay fast.
348
+ let dynamicDelay = delayMs;
349
+ const processOne = async (item) => {
350
+ if (dynamicDelay)
351
+ await sleep(dynamicDelay);
352
+ const path = (() => { try {
353
+ return new URL(item.url).pathname;
354
+ }
355
+ catch {
356
+ return '/';
357
+ } })();
358
+ if (!robots.isAllowed(path)) {
359
+ ensureWiped();
360
+ try {
361
+ robotsInsert.run({ crawl_id: crawlId, url: item.url, url_key: urlKey(item.url, keyOpts), depth: item.depth });
362
+ }
363
+ catch { /* dup */ }
364
+ skipped++;
365
+ return;
366
+ }
367
+ try {
368
+ const r = await fetchWithRedirects(item.url, ua, dispatcher);
369
+ // Self-tune the delay from the throttle signal.
370
+ if (r.throttled)
371
+ dynamicDelay = Math.min(Math.round(dynamicDelay * 1.5) + 200, 5000);
372
+ else if (dynamicDelay > delayMs)
373
+ dynamicDelay = Math.max(delayMs, Math.round(dynamicDelay * 0.9));
374
+ // Redirect that leaves the internal host or lands on a skip-pattern URL (e.g. an internal
375
+ // link 301-ing into Shopify's accounts.<domain> OAuth flow): record the ORIGINAL URL as a
376
+ // redirect-out and do NOT store the off-site target as a page or extract its links.
377
+ let offsite = false;
378
+ try {
379
+ offsite = r.redirects.length > 0 && (!isInternalHost(new URL(r.finalUrl).hostname, baseHost) || skipUrl(r.finalUrl, excludeRes));
380
+ }
381
+ catch { /* unparseable final URL */ }
382
+ const pageUrl = offsite ? item.url : r.finalUrl;
383
+ const finalKey = urlKey(pageUrl, keyOpts);
384
+ // Multiple enqueued URLs can redirect to the same final page. The pages table
385
+ // dedupes via ON CONFLICT(url_key), but re-extracting the same HTML would insert
386
+ // its outlinks AGAIN, inflating inlink_count/iPR weights — so skip re-processing.
387
+ if (emitted.has(finalKey)) {
388
+ crawled++;
389
+ flush();
390
+ return;
391
+ }
392
+ emitted.add(finalKey);
393
+ const status = offsite ? (r.redirects[0]?.status ?? r.status) : r.status;
394
+ const isHtml = !offsite && isHtmlContentType(r.contentType);
395
+ const ex = isHtml && r.status === 200 ? extractPage(r.body, r.finalUrl, baseHost, keyOpts, r.xRobotsTag) : null;
396
+ // Only HTML 200s are indexable; otherwise capture WHY not (status/type/directive/canonical).
397
+ const { indexable, reason } = offsite
398
+ ? { indexable: 0, reason: 'redirect' }
399
+ : classifyIndexability(isHtml, r.status, !!ex?.noindex, r.xRobotsTag, ex?.canonicalKey ?? null, finalKey);
400
+ pageBuf.push({
401
+ crawl_id: crawlId, url: pageUrl, url_key: finalKey, status_code: status,
402
+ content_type: r.contentType, content_encoding: r.contentEncoding, cache_control: r.cacheControl,
403
+ last_modified: r.lastModified, etag: r.etag, vary: r.vary,
404
+ bytes: r.bytes, response_time_ms: r.timeMs, depth: item.depth,
405
+ is_internal: 1, indexable, indexable_reason: reason, noindex: ex?.noindex ? 1 : 0,
406
+ title: ex?.title ?? null, title_length: ex?.titleLength ?? 0,
407
+ meta_description: ex?.metaDescription ?? null, meta_description_length: ex?.metaDescriptionLength ?? 0,
408
+ h1: ex?.h1 ?? null, h1_count: ex?.h1Count ?? 0, word_count: ex?.wordCount ?? 0,
409
+ lang: ex?.lang ?? null,
410
+ charset: ex?.charset ?? r.contentType.match(/charset=([\w-]+)/i)?.[1] ?? null, // meta, else Content-Type header
411
+ canonical_url: ex?.canonicalUrl ?? null, canonical_key: ex?.canonicalKey ?? null,
412
+ robots: ex?.robots ?? null, x_robots_tag: r.xRobotsTag ?? null, viewport: ex?.viewport ?? null,
413
+ json_ld: ex?.jsonLd ?? null, og_tags: ex?.ogTags ?? null, hreflang: ex?.hreflang ?? null,
414
+ redirects: r.redirects.length ? JSON.stringify(r.redirects) : null,
415
+ internal_links: ex?.internalLinks ?? 0, external_links: ex?.externalLinks ?? 0,
416
+ image_count: ex?.imageCount ?? 0, images_without_alt: ex?.imagesWithoutAlt ?? 0,
417
+ images_missing_dimensions: ex?.imagesMissingDimensions ?? 0,
418
+ canonical_count: ex?.canonicalCount ?? 0, canonical_relative: ex?.canonicalRelative ? 1 : 0,
419
+ h2_count: ex?.h2Count ?? 0, heading_skips: ex?.headingSkips ?? 0,
420
+ rel_next: ex?.relNext ? 1 : 0, rel_prev: ex?.relPrev ? 1 : 0, rel_amphtml: ex?.relAmphtml ?? null,
421
+ mixed_content_count: ex?.mixedContentCount ?? 0, twitter_tags: ex?.twitterTags ?? null,
422
+ has_microdata: ex?.hasMicrodata ? 1 : 0, has_rdfa: ex?.hasRdfa ? 1 : 0,
423
+ has_favicon: ex ? (ex.hasFavicon ? 1 : 0) : null, has_analytics: ex ? (ex.hasAnalytics ? 1 : 0) : null,
424
+ body_chunks: ex?.bodyChunks?.length ? JSON.stringify(ex.bodyChunks) : null,
425
+ security_headers: r.securityHeaders,
426
+ });
427
+ if (ex) {
428
+ for (const src of ex.imageSrcs)
429
+ imageRefs.set(src, (imageRefs.get(src) ?? 0) + 1);
430
+ for (const l of ex.links) {
431
+ linkBuf.push({
432
+ crawl_id: crawlId, source_url: r.finalUrl, source_key: finalKey,
433
+ target_url: l.targetUrl, target_key: l.targetKey, anchor_text: l.anchor,
434
+ is_internal: l.isInternal ? 1 : 0, placement: l.placement, rel: l.rel,
435
+ });
436
+ if (l.isInternal && !skipUrl(l.targetUrl, excludeRes) && item.depth + 1 <= maxDepth && (crawled + frontier.length) < maxPages) {
437
+ enqueue(l.targetUrl, item.depth + 1);
438
+ }
439
+ }
440
+ }
441
+ crawled++;
442
+ flush();
443
+ if (crawled % 10 === 0)
444
+ saveProgress();
445
+ }
446
+ catch (err) {
447
+ failed++;
448
+ errorInsert.run(crawlId, item.url, 'fetch', err instanceof Error ? err.message : String(err));
449
+ }
450
+ };
451
+ // Concurrency pool over the shared frontier. Only resolve once all in-flight
452
+ // work has drained — never while requests are still running (avoids late
453
+ // writes after the DB is closed when the page cap is hit mid-flight).
454
+ await new Promise((resolve) => {
455
+ let running = 0;
456
+ let finished = false;
457
+ const pump = () => {
458
+ if (finished)
459
+ return;
460
+ const capHit = crawled >= maxPages || signal.aborted;
461
+ if (running === 0 && (capHit || frontier.length === 0)) {
462
+ finished = true;
463
+ resolve();
464
+ return;
465
+ }
466
+ if (capHit)
467
+ return; // stop starting new work; let in-flight drain via finally → pump
468
+ while (running < concurrency && frontier.length > 0 && crawled < maxPages && !signal.aborted) {
469
+ const item = frontier.shift();
470
+ running++;
471
+ processOne(item).finally(() => { running--; pump(); });
472
+ }
473
+ };
474
+ pump();
475
+ });
476
+ flush(true);
477
+ // Post-crawl: sample image weights. One request per distinct <img> src, most-used first,
478
+ // reading ONLY the response headers (content-length + content-type) — the body is cancelled
479
+ // immediately, so no image bytes are ever downloaded. Bounded so a huge site can't stall
480
+ // the crawl; feeds the Site health image-weight report.
481
+ if (!signal.aborted && wipedPrior) { // wipedPrior = the crawl really wrote pages; keep old sample otherwise
482
+ db.db.exec('DELETE FROM image_assets');
483
+ }
484
+ // Shared header-only probe for the two post-crawl passes: fetch, read headers, cancel the
485
+ // body immediately — the bytes are never downloaded. Politeness lives in probePool: 2
486
+ // workers paced at (at least) the crawl delay, so the aggregate stays at a few rps.
487
+ const headProbe = async (url, extraHeaders, redirect) => {
488
+ try {
489
+ const res = await ufetch(url, { headers: { 'user-agent': ua, ...extraHeaders }, redirect, dispatcher, signal: AbortSignal.timeout(10000) });
490
+ try {
491
+ await res.body?.cancel();
492
+ }
493
+ catch { /* already closed */ }
494
+ return { status: res.status, header: (n) => res.headers.get(n) };
495
+ }
496
+ catch {
497
+ return null;
498
+ }
499
+ };
500
+ const probePool = async (items, run) => {
501
+ let idx = 0;
502
+ const worker = async () => {
503
+ while (idx < items.length && !signal.aborted) {
504
+ await run(items[idx++]);
505
+ if (delayMs)
506
+ await sleep(Math.max(delayMs, 250));
507
+ }
508
+ };
509
+ await Promise.all(Array.from({ length: 2 }, worker));
510
+ };
511
+ if (!signal.aborted && imageRefs.size) {
512
+ const MAX_IMAGES = 500;
513
+ // Same-host images obey the site's robots rules (same gate as the page crawl). Off-host
514
+ // (CDN) images are kept — the SEO question is what the page loads, not where it's hosted —
515
+ // but they go through the paced pool like everything else.
516
+ const sample = [...imageRefs.entries()]
517
+ .filter(([url]) => {
518
+ try {
519
+ const u = new URL(url);
520
+ return !isInternalHost(u.hostname, baseHost) || robots.isAllowed(u.pathname);
521
+ }
522
+ catch {
523
+ return false;
524
+ }
525
+ })
526
+ .sort((a, b) => b[1] - a[1]).slice(0, MAX_IMAGES);
527
+ const imgInsert = db.db.prepare(`INSERT INTO image_assets (url, bytes, status, content_type, used_on, fetched_at)
528
+ VALUES (?,?,?,?,?,datetime('now')) ON CONFLICT(url) DO NOTHING`);
529
+ const rows = [];
530
+ await probePool(sample, async ([url, usedOn]) => {
531
+ const r = await headProbe(url, {}, 'follow');
532
+ if (r) {
533
+ const len = Number(r.header('content-length'));
534
+ rows.push([url, Number.isFinite(len) && len > 0 ? len : null, r.status, (r.header('content-type') ?? '').split(';')[0].trim(), usedOn]);
535
+ }
536
+ else
537
+ rows.push([url, null, 0, '', usedOn]);
538
+ });
539
+ db.db.transaction(() => { for (const r of rows)
540
+ imgInsert.run(...r); })();
541
+ }
542
+ // Post-crawl: in-degree (orphans), internal PageRank (iPR) + body-only click depth.
543
+ finalizeLinkGraph(db.db, crawlId, urlKey(seed, keyOpts));
544
+ // Post-crawl: conditional-revalidation probe. Pages that advertised validators
545
+ // (Last-Modified / ETag) get ONE conditional re-request — a properly configured server
546
+ // answers 304 Not Modified; a 200 means it re-serves the full body to every revalidating
547
+ // crawler and cache, wasting crawl budget. Bounded sample, most-linked pages first.
548
+ if (!signal.aborted) {
549
+ const cands = db.db.prepare(`SELECT url, url_key, last_modified lm, etag FROM pages
550
+ WHERE status_code=200 AND is_internal=1 AND (last_modified IS NOT NULL OR etag IS NOT NULL)
551
+ ORDER BY inlink_count DESC LIMIT 30`).all();
552
+ const upd = db.db.prepare('UPDATE pages SET conditional_304=? WHERE url_key=?');
553
+ await probePool(cands, async (p) => {
554
+ const headers = {};
555
+ if (p.etag)
556
+ headers['if-none-match'] = p.etag;
557
+ if (p.lm)
558
+ headers['if-modified-since'] = p.lm;
559
+ const r = await headProbe(p.url, headers, 'manual');
560
+ if (r)
561
+ upd.run(r.status === 304 ? 1 : 0, p.url_key); // fetch error → leave NULL, never counted against the server
562
+ });
563
+ }
564
+ // (Sitemap is now fetched + stored + used as crawl seeds at the start of the crawl.)
565
+ const finishedAt = new Date().toISOString();
566
+ db.db.prepare(`UPDATE crawl_metadata SET status=?, urls_discovered=?, urls_crawled=?, urls_failed=?, urls_skipped=?,
567
+ finished_at=?, duration_ms=? WHERE crawl_id=?`).run(signal.aborted ? 'cancelled' : 'completed', discovered, crawled, failed, skipped, finishedAt, Date.parse(finishedAt) - Date.parse(startedAt), crawlId);
568
+ db.db.prepare(`UPDATE property_meta SET last_crawl_id=? WHERE site_url=?`).run(crawlId, siteUrl);
569
+ return { crawlId, siteUrl, crawled, failed, skipped };
570
+ }
571
+ catch (err) {
572
+ try {
573
+ db.db.prepare(`UPDATE crawl_metadata SET status='failed', error=?, finished_at=datetime('now') WHERE crawl_id=? AND status='running'`)
574
+ .run(err instanceof Error ? err.message : String(err), crawlId);
575
+ }
576
+ catch { /* best-effort */ }
577
+ throw err;
578
+ }
579
+ finally {
580
+ db.close();
581
+ try {
582
+ await dispatcher.close();
583
+ }
584
+ catch { /* pool already gone */ }
585
+ }
586
+ }
587
+ }
588
+ //# sourceMappingURL=Crawler.js.map