seo-gravity-mcp 1.3.2 → 1.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/adapters/adapterRegistry.d.ts +1 -0
- package/dist/adapters/adapterRegistry.js +6 -2
- package/dist/adapters/staticAdapter.d.ts +1 -1
- package/dist/adapters/staticAdapter.js +3 -2
- package/dist/adapters/unknownAdapter.d.ts +16 -0
- package/dist/adapters/unknownAdapter.js +49 -0
- package/dist/benchmark/methodology.js +1 -1
- package/dist/invariants/registry.js +26 -17
- package/dist/invariants/types.d.ts +9 -1
- package/dist/policy/loader.d.ts +126 -3
- package/dist/policy/loader.js +96 -109
- package/dist/policy/profiles.d.ts +2 -2
- package/dist/policy/types.d.ts +2 -1
- package/dist/test.js +90 -20
- package/dist/tools/orchestration.js +3 -7
- package/dist/types/evidence.d.ts +4 -4
- package/dist/types/seo.d.ts +2 -0
- package/dist/utils/astLocator.d.ts +2 -0
- package/dist/utils/astLocator.js +10 -1
- package/dist/utils/crawlGraph.js +41 -51
- package/dist/utils/gitDiffEngine.js +58 -24
- package/dist/utils/jsdomRenderer.d.ts +2 -1
- package/dist/utils/jsdomRenderer.js +15 -17
- package/dist/utils/projectScanner.d.ts +1 -1
- package/dist/utils/projectScanner.js +4 -2
- package/dist/utils/scraper.d.ts +6 -7
- package/dist/utils/scraper.js +13 -25
- package/dist/utils/snapshotEngine.d.ts +1 -1
- package/dist/utils/snapshotEngine.js +60 -36
- package/dist/utils/urlNormalizer.d.ts +2 -0
- package/dist/utils/urlNormalizer.js +60 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +2 -1
package/dist/utils/crawlGraph.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { fetchAndParsePage } from './scraper.js';
|
|
2
|
+
import { normalizeUrl } from './urlNormalizer.js';
|
|
2
3
|
export class CrawlGraphBuilder {
|
|
3
4
|
startUrl;
|
|
4
5
|
options;
|
|
@@ -13,22 +14,27 @@ export class CrawlGraphBuilder {
|
|
|
13
14
|
const maxDepth = this.options.maxDepth ?? 3;
|
|
14
15
|
const maxPages = this.options.maxPages ?? 50;
|
|
15
16
|
const queue = [{ url: this.startUrl, depth: 0 }];
|
|
16
|
-
let origin = '';
|
|
17
|
+
let origin = this.options.baseOrigin || '';
|
|
17
18
|
try {
|
|
18
19
|
if (this.startUrl.startsWith('http')) {
|
|
19
20
|
origin = new URL(this.startUrl).origin;
|
|
20
21
|
}
|
|
21
22
|
}
|
|
22
23
|
catch {
|
|
23
|
-
|
|
24
|
+
// Keep explicitly supplied baseOrigin, otherwise operate on relative paths.
|
|
24
25
|
}
|
|
25
|
-
// Initialize start node
|
|
26
26
|
this.ensureNode(this.startUrl, 0);
|
|
27
|
-
//
|
|
27
|
+
// Known routes are deliberately seeded as crawl candidates. Their depth is unknown
|
|
28
|
+
// until an actual internal link is observed, but we still crawl them so they can be
|
|
29
|
+
// evaluated rather than being mistaken for observed orphan nodes.
|
|
28
30
|
if (this.options.knownRoutePaths && origin) {
|
|
29
31
|
for (const route of this.options.knownRoutePaths) {
|
|
30
32
|
const fullUrl = `${origin}${route.startsWith('/') ? route : '/' + route}`;
|
|
31
|
-
|
|
33
|
+
const norm = normalizeUrl(fullUrl);
|
|
34
|
+
if (norm !== normalizeUrl(this.startUrl) && !this.visited.has(norm)) {
|
|
35
|
+
this.ensureNode(norm, -1);
|
|
36
|
+
queue.push({ url: norm, depth: 1 });
|
|
37
|
+
}
|
|
32
38
|
}
|
|
33
39
|
}
|
|
34
40
|
while (queue.length > 0 && this.visited.size < maxPages) {
|
|
@@ -38,27 +44,34 @@ export class CrawlGraphBuilder {
|
|
|
38
44
|
continue;
|
|
39
45
|
this.visited.add(normUrl);
|
|
40
46
|
const node = this.ensureNode(normUrl, current.depth);
|
|
41
|
-
|
|
47
|
+
if (node.clickDepth === -1 || current.depth < node.clickDepth) {
|
|
48
|
+
node.clickDepth = current.depth;
|
|
49
|
+
}
|
|
42
50
|
if (current.depth >= maxDepth)
|
|
43
51
|
continue;
|
|
44
52
|
try {
|
|
45
53
|
const parsed = await fetchAndParsePage(normUrl, origin);
|
|
46
54
|
node.title = parsed.title;
|
|
47
55
|
node.statusCode = parsed.statusCode;
|
|
48
|
-
const internalLinks = parsed.links.internal;
|
|
49
56
|
const pageDomain = origin || (normUrl.startsWith('http') ? new URL(normUrl).origin : '');
|
|
50
|
-
|
|
51
|
-
|
|
57
|
+
const internalDetails = parsed.links.internalDetails || parsed.links.internal.map(href => ({
|
|
58
|
+
href,
|
|
59
|
+
anchorText: href,
|
|
60
|
+
rel: []
|
|
61
|
+
}));
|
|
62
|
+
for (const detail of internalDetails) {
|
|
63
|
+
const targetUrl = resolveInternalUrl(detail.href, pageDomain, normUrl);
|
|
52
64
|
if (!targetUrl)
|
|
53
65
|
continue;
|
|
54
66
|
const normTarget = normalizeUrl(targetUrl);
|
|
55
|
-
const
|
|
67
|
+
const anchorText = detail.anchorText.trim();
|
|
68
|
+
const rel = detail.rel.map(value => value.toLowerCase());
|
|
56
69
|
this.edges.push({
|
|
57
70
|
sourceUrl: normUrl,
|
|
58
71
|
targetUrl: normTarget,
|
|
59
|
-
anchorText
|
|
60
|
-
isNofollow:
|
|
61
|
-
isGenericAnchor:
|
|
72
|
+
anchorText,
|
|
73
|
+
isNofollow: rel.includes('nofollow'),
|
|
74
|
+
isGenericAnchor: isGenericAnchorText(anchorText)
|
|
62
75
|
});
|
|
63
76
|
const targetNode = this.ensureNode(normTarget, current.depth + 1);
|
|
64
77
|
if (targetNode.clickDepth === -1 || targetNode.clickDepth > current.depth + 1) {
|
|
@@ -73,7 +86,6 @@ export class CrawlGraphBuilder {
|
|
|
73
86
|
node.statusCode = 500;
|
|
74
87
|
}
|
|
75
88
|
}
|
|
76
|
-
// Calculate In-degree and Out-degree
|
|
77
89
|
for (const edge of this.edges) {
|
|
78
90
|
const src = this.nodesMap.get(edge.sourceUrl);
|
|
79
91
|
const tgt = this.nodesMap.get(edge.targetUrl);
|
|
@@ -82,24 +94,21 @@ export class CrawlGraphBuilder {
|
|
|
82
94
|
if (tgt)
|
|
83
95
|
tgt.incomingLinksCount++;
|
|
84
96
|
}
|
|
85
|
-
// Compute PageRank Heuristic
|
|
86
97
|
this.computePageRank();
|
|
87
|
-
// Identify Orphans, Hubs, and Dead-Ends
|
|
88
98
|
const orphanPages = [];
|
|
89
99
|
const hubPages = [];
|
|
90
100
|
const deadEnds = [];
|
|
91
101
|
for (const [url, node] of this.nodesMap.entries()) {
|
|
92
|
-
//
|
|
93
|
-
|
|
102
|
+
// Only evaluated/visited nodes can be classified. Unobserved known routes are not
|
|
103
|
+
// treated as orphans until they have been crawled.
|
|
104
|
+
if (this.visited.has(url) && node.incomingLinksCount === 0 && url !== normalizeUrl(this.startUrl)) {
|
|
94
105
|
node.isOrphan = true;
|
|
95
106
|
orphanPages.push(url);
|
|
96
107
|
}
|
|
97
|
-
// Hub: 5 or more outgoing links
|
|
98
108
|
if (node.outgoingLinksCount >= 5) {
|
|
99
109
|
node.isHubPage = true;
|
|
100
110
|
hubPages.push(url);
|
|
101
111
|
}
|
|
102
|
-
// Dead end: visited page with 0 outgoing links
|
|
103
112
|
if (this.visited.has(url) && node.outgoingLinksCount === 0) {
|
|
104
113
|
node.isDeadEnd = true;
|
|
105
114
|
deadEnds.push(url);
|
|
@@ -145,15 +154,13 @@ export class CrawlGraphBuilder {
|
|
|
145
154
|
let pr = {};
|
|
146
155
|
for (const u of nodes)
|
|
147
156
|
pr[u] = 1 / N;
|
|
148
|
-
// Build incoming edge map
|
|
149
157
|
const incomingMap = new Map();
|
|
150
158
|
for (const u of nodes)
|
|
151
159
|
incomingMap.set(u, []);
|
|
152
160
|
for (const edge of this.edges) {
|
|
153
161
|
const arr = incomingMap.get(edge.targetUrl);
|
|
154
|
-
if (arr && !arr.includes(edge.sourceUrl))
|
|
162
|
+
if (arr && !arr.includes(edge.sourceUrl))
|
|
155
163
|
arr.push(edge.sourceUrl);
|
|
156
|
-
}
|
|
157
164
|
}
|
|
158
165
|
for (let iter = 0; iter < iterations; iter++) {
|
|
159
166
|
const nextPr = {};
|
|
@@ -169,13 +176,11 @@ export class CrawlGraphBuilder {
|
|
|
169
176
|
}
|
|
170
177
|
pr = nextPr;
|
|
171
178
|
}
|
|
172
|
-
// Normalize PageRank scores to 0.0 - 1.0
|
|
173
179
|
const maxPr = Math.max(...Object.values(pr), 0.0001);
|
|
174
180
|
for (const u of nodes) {
|
|
175
181
|
const node = this.nodesMap.get(u);
|
|
176
|
-
if (node)
|
|
182
|
+
if (node)
|
|
177
183
|
node.pageRankScore = Math.round(((pr[u] || 0) / maxPr) * 100) / 100;
|
|
178
|
-
}
|
|
179
184
|
}
|
|
180
185
|
}
|
|
181
186
|
findSimpleCycles() {
|
|
@@ -191,38 +196,23 @@ export class CrawlGraphBuilder {
|
|
|
191
196
|
return cycles.slice(0, 5);
|
|
192
197
|
}
|
|
193
198
|
}
|
|
194
|
-
function normalizeUrl(url) {
|
|
195
|
-
try {
|
|
196
|
-
if (url.startsWith('http')) {
|
|
197
|
-
const u = new URL(url);
|
|
198
|
-
return `${u.origin}${u.pathname.replace(/\/$/, '') || '/'}`;
|
|
199
|
-
}
|
|
200
|
-
}
|
|
201
|
-
catch {
|
|
202
|
-
// Ignored
|
|
203
|
-
}
|
|
204
|
-
return url.replace(/\/$/, '') || '/';
|
|
205
|
-
}
|
|
206
199
|
function resolveInternalUrl(href, origin, currentUrl) {
|
|
207
200
|
if (!href || href.startsWith('#') || href.startsWith('mailto:') || href.startsWith('tel:') || href.startsWith('javascript:')) {
|
|
208
201
|
return null;
|
|
209
202
|
}
|
|
210
|
-
if (href.startsWith('http://') || href.startsWith('https://')) {
|
|
211
|
-
if (origin && href.startsWith(origin))
|
|
212
|
-
return href;
|
|
213
|
-
return null;
|
|
214
|
-
}
|
|
215
|
-
if (href.startsWith('/')) {
|
|
216
|
-
return origin ? `${origin}${href}` : href;
|
|
217
|
-
}
|
|
218
|
-
// Relative path
|
|
219
203
|
try {
|
|
220
|
-
const base = currentUrl.startsWith('http')
|
|
221
|
-
|
|
222
|
-
|
|
204
|
+
const base = currentUrl.startsWith('http')
|
|
205
|
+
? currentUrl
|
|
206
|
+
: origin
|
|
207
|
+
? `${origin}${currentUrl.startsWith('/') ? currentUrl : '/' + currentUrl}`
|
|
208
|
+
: `http://localhost${currentUrl.startsWith('/') ? currentUrl : '/' + currentUrl}`;
|
|
209
|
+
const resolved = new URL(href, base);
|
|
210
|
+
if (origin && resolved.origin !== origin)
|
|
211
|
+
return null;
|
|
212
|
+
return resolved.toString();
|
|
223
213
|
}
|
|
224
214
|
catch {
|
|
225
|
-
return
|
|
215
|
+
return null;
|
|
226
216
|
}
|
|
227
217
|
}
|
|
228
218
|
function isGenericAnchorText(text) {
|
|
@@ -17,7 +17,11 @@ export function getChangedFilesSince(projectDir, baseRef = 'HEAD~1') {
|
|
|
17
17
|
});
|
|
18
18
|
return output.split('\n').map(l => l.trim().replace(/\\/g, '/')).filter(Boolean);
|
|
19
19
|
}
|
|
20
|
-
catch {
|
|
20
|
+
catch (err) {
|
|
21
|
+
// A failed requested diff is not equivalent to a dirty-worktree query. Fall back
|
|
22
|
+
// only for the conventional local-development baseline and make that behavior explicit.
|
|
23
|
+
if (baseRef !== 'HEAD~1')
|
|
24
|
+
throw new Error(`Unable to compute Git diff from '${baseRef}'.`);
|
|
21
25
|
try {
|
|
22
26
|
const status = execFileSync('git', ['status', '--porcelain'], {
|
|
23
27
|
cwd: projectDir,
|
|
@@ -30,24 +34,43 @@ export function getChangedFilesSince(projectDir, baseRef = 'HEAD~1') {
|
|
|
30
34
|
.filter(Boolean);
|
|
31
35
|
}
|
|
32
36
|
catch {
|
|
33
|
-
|
|
37
|
+
throw new Error(`Unable to determine Git changes in '${projectDir}'.`);
|
|
34
38
|
}
|
|
35
39
|
}
|
|
36
40
|
}
|
|
37
41
|
export function analyzeSemanticFileChange(filePath, projectDir) {
|
|
38
|
-
const
|
|
42
|
+
const normalizedPath = filePath.replace(/\\/g, '/');
|
|
43
|
+
const fullPath = path.resolve(projectDir, normalizedPath);
|
|
44
|
+
const projectRoot = path.resolve(projectDir);
|
|
45
|
+
if (fullPath !== projectRoot && !fullPath.startsWith(projectRoot + path.sep)) {
|
|
46
|
+
throw new Error(`Changed file '${filePath}' resolves outside the project root.`);
|
|
47
|
+
}
|
|
39
48
|
if (!fs.existsSync(fullPath)) {
|
|
49
|
+
// Deleted files need a conservative classification. We can reliably identify
|
|
50
|
+
// route/config categories from the path, but should not claim every invariant is affected.
|
|
51
|
+
const base = path.basename(normalizedPath).toLowerCase();
|
|
52
|
+
if (base.includes('sitemap') || base.includes('robots') || base.includes('llms')) {
|
|
53
|
+
return {
|
|
54
|
+
affectsMetadata: false,
|
|
55
|
+
affectsCanonical: false,
|
|
56
|
+
affectsSchema: false,
|
|
57
|
+
affectsLinks: false,
|
|
58
|
+
affectsRobotsOrSitemap: true,
|
|
59
|
+
likelyAffectedInvariants: ['INV-ROBOTS-ALLOWED', 'INV-SITEMAP-PRESENT', 'INV-LLMS-TXT'],
|
|
60
|
+
riskLevel: 'MEDIUM'
|
|
61
|
+
};
|
|
62
|
+
}
|
|
40
63
|
return {
|
|
41
|
-
affectsMetadata:
|
|
42
|
-
affectsCanonical:
|
|
43
|
-
affectsSchema:
|
|
44
|
-
affectsLinks:
|
|
45
|
-
affectsRobotsOrSitemap:
|
|
46
|
-
likelyAffectedInvariants: [
|
|
47
|
-
riskLevel: '
|
|
64
|
+
affectsMetadata: false,
|
|
65
|
+
affectsCanonical: false,
|
|
66
|
+
affectsSchema: false,
|
|
67
|
+
affectsLinks: false,
|
|
68
|
+
affectsRobotsOrSitemap: false,
|
|
69
|
+
likelyAffectedInvariants: [],
|
|
70
|
+
riskLevel: 'LOW'
|
|
48
71
|
};
|
|
49
72
|
}
|
|
50
|
-
const base = path.basename(
|
|
73
|
+
const base = path.basename(normalizedPath).toLowerCase();
|
|
51
74
|
if (base.includes('sitemap') || base.includes('robots') || base.includes('llms')) {
|
|
52
75
|
return {
|
|
53
76
|
affectsMetadata: false,
|
|
@@ -65,13 +88,18 @@ export function analyzeSemanticFileChange(filePath, projectDir) {
|
|
|
65
88
|
const affectsCanonical = ast.hasCanonicalDeclaration || /rel=["']canonical["']/i.test(content);
|
|
66
89
|
const affectsSchema = ast.hasSchemaMarkup || /application\/ld\+json/.test(content);
|
|
67
90
|
const affectsLinks = /<Link\b|<a\b|href=/i.test(content);
|
|
68
|
-
const affectedInvariants = [
|
|
91
|
+
const affectedInvariants = [];
|
|
69
92
|
if (affectsMetadata)
|
|
70
93
|
affectedInvariants.push('INV-TITLE-PRESENT');
|
|
71
94
|
if (affectsCanonical)
|
|
72
95
|
affectedInvariants.push('INV-CANONICAL-RESOLVES');
|
|
73
96
|
if (affectsLinks)
|
|
74
97
|
affectedInvariants.push('INV-LINK-ACCESSIBLE');
|
|
98
|
+
// A file can only affect HTTP status if it plausibly participates in routing/runtime
|
|
99
|
+
// behavior. Generic content/style files are not automatically treated as HTTP-risky.
|
|
100
|
+
if (affectsMetadata || affectsCanonical || affectsLinks || /route|page|server|middleware|controller/i.test(base)) {
|
|
101
|
+
affectedInvariants.unshift('INV-HTTP-200');
|
|
102
|
+
}
|
|
75
103
|
const riskLevel = (affectsCanonical || affectsMetadata) ? 'HIGH' : affectsLinks ? 'MEDIUM' : 'LOW';
|
|
76
104
|
return {
|
|
77
105
|
affectsMetadata,
|
|
@@ -79,21 +107,26 @@ export function analyzeSemanticFileChange(filePath, projectDir) {
|
|
|
79
107
|
affectsSchema,
|
|
80
108
|
affectsLinks,
|
|
81
109
|
affectsRobotsOrSitemap: false,
|
|
82
|
-
likelyAffectedInvariants: affectedInvariants,
|
|
110
|
+
likelyAffectedInvariants: [...new Set(affectedInvariants)],
|
|
83
111
|
riskLevel
|
|
84
112
|
};
|
|
85
113
|
}
|
|
86
114
|
export function mapChangedFilesToRoutes(changedFiles, routes) {
|
|
115
|
+
const normalizedChanged = changedFiles.map(f => f.replace(/\\/g, '/'));
|
|
87
116
|
const affected = [];
|
|
88
117
|
const unaffected = [];
|
|
89
|
-
const isGlobalFile =
|
|
90
|
-
f.
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
118
|
+
const isGlobalFile = normalizedChanged.some(f => {
|
|
119
|
+
const base = path.basename(f).toLowerCase();
|
|
120
|
+
return base.startsWith('layout.') ||
|
|
121
|
+
base.startsWith('sitemap.') ||
|
|
122
|
+
base.startsWith('robots.') ||
|
|
123
|
+
base === 'package.json' ||
|
|
124
|
+
base.startsWith('next.config') ||
|
|
125
|
+
base.startsWith('astro.config');
|
|
126
|
+
});
|
|
95
127
|
for (const r of routes) {
|
|
96
|
-
|
|
128
|
+
const routeSource = r.sourceFilePath.replace(/\\/g, '/');
|
|
129
|
+
if (isGlobalFile || normalizedChanged.includes(routeSource)) {
|
|
97
130
|
affected.push(r);
|
|
98
131
|
}
|
|
99
132
|
else {
|
|
@@ -113,16 +146,17 @@ export async function runDifferentialAudit(projectDir, baseRef = 'HEAD~1', baseU
|
|
|
113
146
|
for (const file of changedFiles) {
|
|
114
147
|
const sem = analyzeSemanticFileChange(file, resolved);
|
|
115
148
|
semanticImpacts[file] = sem;
|
|
116
|
-
const
|
|
149
|
+
const normalizedFile = file.replace(/\\/g, '/');
|
|
150
|
+
const matchingRoute = routes.find(r => r.sourceFilePath.replace(/\\/g, '/') === normalizedFile);
|
|
117
151
|
semanticDiffs.push({
|
|
118
|
-
changedFile:
|
|
152
|
+
changedFile: normalizedFile,
|
|
119
153
|
affectedRoute: matchingRoute?.routePath,
|
|
120
154
|
semanticCategory: sem,
|
|
121
|
-
impactDescription: `File change triggers risk [${sem.riskLevel}] on invariants: ${sem.likelyAffectedInvariants.join(', ')}`
|
|
155
|
+
impactDescription: `File change triggers risk [${sem.riskLevel}] on invariants: ${sem.likelyAffectedInvariants.join(', ') || 'none detected'}`
|
|
122
156
|
});
|
|
123
157
|
}
|
|
124
158
|
const snapshot = await createProjectSnapshot(resolved, { baseUrl });
|
|
125
|
-
const targetedFindings = snapshot.findings.filter(f => affected.some(r => r.routePath === f.affectedUrl || f.sourceLocation?.filePath === r.sourceFilePath));
|
|
159
|
+
const targetedFindings = snapshot.findings.filter(f => affected.some(r => r.routePath === f.affectedUrl || f.sourceLocation?.filePath?.replace(/\\/g, '/') === r.sourceFilePath.replace(/\\/g, '/')));
|
|
126
160
|
const hasCritical = targetedFindings.some(f => f.severity === 'critical' || f.severity === 'high');
|
|
127
161
|
return {
|
|
128
162
|
schemaVersion: 'seo.gravity/v1',
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { JsRenderingDiffReport } from '../types/seo.js';
|
|
2
2
|
/**
|
|
3
|
-
* Compares initial server HTML with
|
|
3
|
+
* Compares initial server HTML with a normalized JSDOM parse.
|
|
4
|
+
* This is a static DOM comparison; it does not execute application JavaScript.
|
|
4
5
|
*/
|
|
5
6
|
export declare function compareServerVsClientDom(url: string): Promise<JsRenderingDiffReport>;
|
|
@@ -2,7 +2,8 @@ import { JSDOM, VirtualConsole } from 'jsdom';
|
|
|
2
2
|
import axios from 'axios';
|
|
3
3
|
import { getRandomUserAgent } from './scraper.js';
|
|
4
4
|
/**
|
|
5
|
-
* Compares initial server HTML with
|
|
5
|
+
* Compares initial server HTML with a normalized JSDOM parse.
|
|
6
|
+
* This is a static DOM comparison; it does not execute application JavaScript.
|
|
6
7
|
*/
|
|
7
8
|
export async function compareServerVsClientDom(url) {
|
|
8
9
|
let serverHtml = '';
|
|
@@ -20,11 +21,10 @@ export async function compareServerVsClientDom(url) {
|
|
|
20
21
|
catch (err) {
|
|
21
22
|
throw new Error(`Failed to fetch server HTML from ${url}: ${err.message}`);
|
|
22
23
|
}
|
|
23
|
-
// Render in JSDOM with script execution enabled
|
|
24
24
|
const virtualConsole = new VirtualConsole();
|
|
25
|
-
virtualConsole.on('error', () => { });
|
|
25
|
+
virtualConsole.on('error', () => { });
|
|
26
26
|
virtualConsole.on('warn', () => { });
|
|
27
|
-
let
|
|
27
|
+
let normalizedDomHtml = serverHtml;
|
|
28
28
|
try {
|
|
29
29
|
const dom = new JSDOM(serverHtml, {
|
|
30
30
|
url,
|
|
@@ -32,21 +32,20 @@ export async function compareServerVsClientDom(url) {
|
|
|
32
32
|
resources: 'usable',
|
|
33
33
|
virtualConsole
|
|
34
34
|
});
|
|
35
|
-
|
|
35
|
+
normalizedDomHtml = dom.serialize();
|
|
36
36
|
}
|
|
37
37
|
catch {
|
|
38
|
-
|
|
38
|
+
normalizedDomHtml = serverHtml;
|
|
39
39
|
}
|
|
40
40
|
const serverLength = serverHtml.length;
|
|
41
|
-
const clientLength =
|
|
41
|
+
const clientLength = normalizedDomHtml.length;
|
|
42
42
|
const lengthDiff = Math.abs(clientLength - serverLength);
|
|
43
43
|
const percentDiff = Number(((lengthDiff / Math.max(serverLength, 1)) * 100).toFixed(1));
|
|
44
|
-
// Inspect link and heading differences
|
|
45
44
|
const serverLinks = Array.from(serverHtml.matchAll(/href=["'](https?:\/\/[^"']+|\/[^"']+)["']/gi)).map(m => m[1]);
|
|
46
|
-
const clientLinks = Array.from(
|
|
45
|
+
const clientLinks = Array.from(normalizedDomHtml.matchAll(/href=["'](https?:\/\/[^"']+|\/[^"']+)["']/gi)).map(m => m[1]);
|
|
47
46
|
const linksOnlyInClient = clientLinks.filter(l => !serverLinks.includes(l)).slice(0, 10);
|
|
48
47
|
const serverH1s = Array.from(serverHtml.matchAll(/<h1[^>]*>([\s\S]*?)<\/h1>/gi)).map(m => m[1].replace(/<[^>]+>/g, '').trim());
|
|
49
|
-
const clientH1s = Array.from(
|
|
48
|
+
const clientH1s = Array.from(normalizedDomHtml.matchAll(/<h1[^>]*>([\s\S]*?)<\/h1>/gi)).map(m => m[1].replace(/<[^>]+>/g, '').trim());
|
|
50
49
|
const headingsOnlyInClient = clientH1s.filter(h => !serverH1s.includes(h));
|
|
51
50
|
const crawlerRisk = linksOnlyInClient.length > 5 || percentDiff > 50
|
|
52
51
|
? 'High (Significant Hydration Dependence)'
|
|
@@ -54,17 +53,16 @@ export async function compareServerVsClientDom(url) {
|
|
|
54
53
|
? 'Medium'
|
|
55
54
|
: 'Low';
|
|
56
55
|
const recommendations = [];
|
|
57
|
-
if (linksOnlyInClient.length
|
|
58
|
-
recommendations.push(
|
|
56
|
+
if (percentDiff < 15 && linksOnlyInClient.length === 0 && headingsOnlyInClient.length === 0) {
|
|
57
|
+
recommendations.push('Static DOM normalization shows strong parity. This does not execute client JavaScript.');
|
|
59
58
|
}
|
|
60
|
-
|
|
61
|
-
recommendations.push('
|
|
62
|
-
}
|
|
63
|
-
if (percentDiff < 15 && linksOnlyInClient.length === 0) {
|
|
64
|
-
recommendations.push('Excellent hydration parity. Initial HTML matches rendered DOM cleanly for search engine bots.');
|
|
59
|
+
else {
|
|
60
|
+
recommendations.push('This result reflects static DOM normalization only; use a real browser runtime to test post-JavaScript hydration behavior.');
|
|
65
61
|
}
|
|
66
62
|
return {
|
|
67
63
|
url,
|
|
64
|
+
executionMode: 'static_dom_normalization',
|
|
65
|
+
domParser: 'jsdom_html_parser',
|
|
68
66
|
serverHtmlLength: serverLength,
|
|
69
67
|
hydratedDomLength: clientLength,
|
|
70
68
|
contentDifferencePercent: percentDiff,
|
|
@@ -2,5 +2,5 @@ import { ProjectFrameworkInfo, DiscoveredRoute, RouteSourceMapping } from '../ty
|
|
|
2
2
|
import { FrameworkAdapter } from '../adapters/types.js';
|
|
3
3
|
export declare function getProjectAdapter(projectDir: string): FrameworkAdapter;
|
|
4
4
|
export declare function detectFramework(projectDir: string): ProjectFrameworkInfo;
|
|
5
|
-
export declare function discoverRoutes(projectDir: string,
|
|
5
|
+
export declare function discoverRoutes(projectDir: string, frameworkInfo?: ProjectFrameworkInfo): DiscoveredRoute[];
|
|
6
6
|
export declare function mapUrlToRouteSource(targetUrl: string, discoveredRoutes: DiscoveredRoute[], projectDir?: string): RouteSourceMapping;
|
|
@@ -9,9 +9,11 @@ export function detectFramework(projectDir) {
|
|
|
9
9
|
const adapter = defaultAdapterRegistry.getAdapterForProject(resolvedDir);
|
|
10
10
|
return adapter.getProjectInfo(resolvedDir);
|
|
11
11
|
}
|
|
12
|
-
export function discoverRoutes(projectDir,
|
|
12
|
+
export function discoverRoutes(projectDir, frameworkInfo) {
|
|
13
13
|
const resolvedDir = path.resolve(projectDir);
|
|
14
|
-
const adapter =
|
|
14
|
+
const adapter = frameworkInfo?.framework
|
|
15
|
+
? (defaultAdapterRegistry.getAdapterById(frameworkInfo.framework) || defaultAdapterRegistry.getAdapterForProject(resolvedDir))
|
|
16
|
+
: defaultAdapterRegistry.getAdapterForProject(resolvedDir);
|
|
15
17
|
return adapter.discoverRoutes(resolvedDir);
|
|
16
18
|
}
|
|
17
19
|
export function mapUrlToRouteSource(targetUrl, discoveredRoutes, projectDir = '.') {
|
package/dist/utils/scraper.d.ts
CHANGED
|
@@ -1,5 +1,10 @@
|
|
|
1
1
|
import * as cheerio from 'cheerio';
|
|
2
2
|
export declare function getRandomUserAgent(): string;
|
|
3
|
+
export interface InternalLinkDetail {
|
|
4
|
+
href: string;
|
|
5
|
+
anchorText: string;
|
|
6
|
+
rel: string[];
|
|
7
|
+
}
|
|
3
8
|
export interface FetchedPageContent {
|
|
4
9
|
url: string;
|
|
5
10
|
statusCode: number;
|
|
@@ -19,6 +24,7 @@ export interface FetchedPageContent {
|
|
|
19
24
|
links: {
|
|
20
25
|
internal: string[];
|
|
21
26
|
external: string[];
|
|
27
|
+
internalDetails: InternalLinkDetail[];
|
|
22
28
|
};
|
|
23
29
|
images: Array<{
|
|
24
30
|
src: string;
|
|
@@ -26,11 +32,4 @@ export interface FetchedPageContent {
|
|
|
26
32
|
}>;
|
|
27
33
|
schemas: any[];
|
|
28
34
|
}
|
|
29
|
-
/**
|
|
30
|
-
* Robust fetcher that supports:
|
|
31
|
-
* 1. Live Web URLs (https://example.com)
|
|
32
|
-
* 2. Localhost Dev Servers (http://localhost:3000)
|
|
33
|
-
* 3. Local Workspace File Paths (d:/aide/index.html, ./public/test.html)
|
|
34
|
-
* 4. Raw HTML strings
|
|
35
|
-
*/
|
|
36
35
|
export declare function fetchAndParsePage(input: string, baseOrigin?: string): Promise<FetchedPageContent>;
|
package/dist/utils/scraper.js
CHANGED
|
@@ -14,29 +14,19 @@ export function getRandomUserAgent() {
|
|
|
14
14
|
return custom;
|
|
15
15
|
return USER_AGENTS[Math.floor(Math.random() * USER_AGENTS.length)];
|
|
16
16
|
}
|
|
17
|
-
/**
|
|
18
|
-
* Robust fetcher that supports:
|
|
19
|
-
* 1. Live Web URLs (https://example.com)
|
|
20
|
-
* 2. Localhost Dev Servers (http://localhost:3000)
|
|
21
|
-
* 3. Local Workspace File Paths (d:/aide/index.html, ./public/test.html)
|
|
22
|
-
* 4. Raw HTML strings
|
|
23
|
-
*/
|
|
24
17
|
export async function fetchAndParsePage(input, baseOrigin) {
|
|
25
18
|
let html = '';
|
|
26
19
|
let url = input;
|
|
27
20
|
let statusCode = 200;
|
|
28
21
|
let headers = {};
|
|
29
|
-
// Check if input is raw HTML
|
|
30
22
|
if (input.trim().startsWith('<') && input.includes('>')) {
|
|
31
23
|
html = input;
|
|
32
24
|
url = baseOrigin || 'raw-html-input';
|
|
33
25
|
}
|
|
34
|
-
// Check if input is a local file path
|
|
35
26
|
else if (fs.existsSync(input) && fs.statSync(input).isFile()) {
|
|
36
27
|
html = fs.readFileSync(input, 'utf-8');
|
|
37
28
|
url = `file://${path.resolve(input).replace(/\\/g, '/')}`;
|
|
38
29
|
}
|
|
39
|
-
// Otherwise treat as URL (local or remote)
|
|
40
30
|
else {
|
|
41
31
|
try {
|
|
42
32
|
const response = await axios.get(input, {
|
|
@@ -60,70 +50,68 @@ export async function fetchAndParsePage(input, baseOrigin) {
|
|
|
60
50
|
}
|
|
61
51
|
}
|
|
62
52
|
const $ = cheerio.load(html);
|
|
63
|
-
// Extract metadata
|
|
64
53
|
const title = $('title').first().text().trim() || $('meta[property="og:title"]').attr('content')?.trim() || '';
|
|
65
54
|
const metaDescription = $('meta[name="description"]').attr('content')?.trim() || $('meta[property="og:description"]').attr('content')?.trim() || '';
|
|
66
|
-
// Extract headings
|
|
67
55
|
const headings = {
|
|
68
56
|
h1: $('h1').map((_, el) => $(el).text().trim()).get().filter(Boolean),
|
|
69
57
|
h2: $('h2').map((_, el) => $(el).text().trim()).get().filter(Boolean),
|
|
70
58
|
h3: $('h3').map((_, el) => $(el).text().trim()).get().filter(Boolean),
|
|
71
59
|
h4: $('h4').map((_, el) => $(el).text().trim()).get().filter(Boolean),
|
|
72
60
|
};
|
|
73
|
-
// Clean body text
|
|
74
61
|
const cloneBody = $('body').clone();
|
|
75
62
|
cloneBody.find('script, style, noscript, nav, footer, iframe, svg').remove();
|
|
76
63
|
const cleanText = cloneBody.text().replace(/\s+/g, ' ').trim();
|
|
77
64
|
const words = cleanText.split(/\s+/).filter(w => w.length > 0);
|
|
78
65
|
const wordCount = words.length;
|
|
79
|
-
|
|
80
|
-
const domain = url.startsWith('http') ? new URL(url).hostname : '';
|
|
66
|
+
const domain = url.startsWith('http') ? new URL(url).hostname.toLowerCase() : '';
|
|
81
67
|
const internal = [];
|
|
82
68
|
const external = [];
|
|
69
|
+
const internalDetails = [];
|
|
83
70
|
$('a[href]').each((_, el) => {
|
|
84
71
|
const href = $(el).attr('href')?.trim() || '';
|
|
85
72
|
if (!href || href.startsWith('#') || href.startsWith('javascript:') || href.startsWith('mailto:') || href.startsWith('tel:'))
|
|
86
73
|
return;
|
|
74
|
+
const anchorText = $(el).text().replace(/\s+/g, ' ').trim();
|
|
75
|
+
const rel = ($(el).attr('rel') || '').split(/\s+/).map(v => v.trim()).filter(Boolean);
|
|
87
76
|
try {
|
|
88
77
|
if (href.startsWith('/')) {
|
|
89
78
|
internal.push(href);
|
|
79
|
+
internalDetails.push({ href, anchorText, rel });
|
|
90
80
|
}
|
|
91
|
-
else if (href.startsWith('http')) {
|
|
81
|
+
else if (href.startsWith('http://') || href.startsWith('https://')) {
|
|
92
82
|
const parsed = new URL(href);
|
|
93
|
-
if (domain && parsed.hostname === domain) {
|
|
83
|
+
if (domain && parsed.hostname.toLowerCase() === domain) {
|
|
94
84
|
internal.push(parsed.pathname + parsed.search);
|
|
85
|
+
internalDetails.push({ href, anchorText, rel });
|
|
95
86
|
}
|
|
96
87
|
else {
|
|
97
88
|
external.push(href);
|
|
98
89
|
}
|
|
99
90
|
}
|
|
100
91
|
else if (!href.includes(':')) {
|
|
101
|
-
// Relative link (e.g. "about", "../docs")
|
|
102
92
|
internal.push('/' + href.replace(/^\.?\//, ''));
|
|
93
|
+
internalDetails.push({ href, anchorText, rel });
|
|
103
94
|
}
|
|
104
95
|
}
|
|
105
96
|
catch {
|
|
106
|
-
// Ignore malformed URLs
|
|
97
|
+
// Ignore malformed URLs.
|
|
107
98
|
}
|
|
108
99
|
});
|
|
109
|
-
// Extract images
|
|
110
100
|
const images = [];
|
|
111
101
|
$('img').each((_, el) => {
|
|
112
102
|
const src = $(el).attr('src') || $(el).attr('data-src') || '';
|
|
113
103
|
const alt = $(el).attr('alt') || '';
|
|
114
104
|
images.push({ src, alt });
|
|
115
105
|
});
|
|
116
|
-
// Extract JSON-LD schemas
|
|
117
106
|
const schemas = [];
|
|
118
107
|
$('script[type="application/ld+json"]').each((_, el) => {
|
|
119
108
|
try {
|
|
120
109
|
const text = $(el).html();
|
|
121
|
-
if (text)
|
|
110
|
+
if (text)
|
|
122
111
|
schemas.push(JSON.parse(text));
|
|
123
|
-
}
|
|
124
112
|
}
|
|
125
113
|
catch {
|
|
126
|
-
//
|
|
114
|
+
// Preserve the page result; invalid JSON-LD is handled by schema analysis.
|
|
127
115
|
}
|
|
128
116
|
});
|
|
129
117
|
return {
|
|
@@ -137,7 +125,7 @@ export async function fetchAndParsePage(input, baseOrigin) {
|
|
|
137
125
|
headings,
|
|
138
126
|
cleanText,
|
|
139
127
|
wordCount,
|
|
140
|
-
links: { internal, external },
|
|
128
|
+
links: { internal, external, internalDetails },
|
|
141
129
|
images,
|
|
142
130
|
schemas
|
|
143
131
|
};
|
|
@@ -5,7 +5,7 @@ export interface CreateSnapshotOptions {
|
|
|
5
5
|
maxPagesToAudit?: number;
|
|
6
6
|
includeCrawlGraph?: boolean;
|
|
7
7
|
}
|
|
8
|
-
export declare function computeLogicalPageId(urlOrPath: string): string;
|
|
8
|
+
export declare function computeLogicalPageId(urlOrPath: string, baseUrl?: string): string;
|
|
9
9
|
export declare function extractGitMetadata(projectDir: string): GitMetadata;
|
|
10
10
|
export declare function createProjectSnapshot(projectPath: string, options?: CreateSnapshotOptions): Promise<ProjectSnapshot>;
|
|
11
11
|
export declare function compareSnapshots(baseline: ProjectSnapshot, current: ProjectSnapshot): RegressionReport;
|