mcp-h2-xpath-extractor 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/index.js +131 -0
- package/package.json +20 -0
package/index.js
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
import { Server } from "@modelcontextprotocol/sdk/server/index.js";
|
|
4
|
+
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
5
|
+
import {
|
|
6
|
+
CallToolRequestSchema,
|
|
7
|
+
ListToolsRequestSchema,
|
|
8
|
+
} from "@modelcontextprotocol/sdk/types.js";
|
|
9
|
+
import puppeteer from 'puppeteer';
|
|
10
|
+
|
|
11
|
+
const server = new Server(
|
|
12
|
+
{
|
|
13
|
+
name: "mcp-h2-xpath-extractor",
|
|
14
|
+
version: "1.0.0",
|
|
15
|
+
},
|
|
16
|
+
{
|
|
17
|
+
capabilities: {
|
|
18
|
+
tools: {},
|
|
19
|
+
},
|
|
20
|
+
}
|
|
21
|
+
);
|
|
22
|
+
|
|
23
|
+
server.setRequestHandler(ListToolsRequestSchema, async () => {
|
|
24
|
+
return {
|
|
25
|
+
tools: [
|
|
26
|
+
{
|
|
27
|
+
name: "extract_h2_xpaths",
|
|
28
|
+
description: "Opens a URL in Chrome and extracts the XPath of all h2 tags.",
|
|
29
|
+
inputSchema: {
|
|
30
|
+
type: "object",
|
|
31
|
+
properties: {
|
|
32
|
+
url: {
|
|
33
|
+
type: "string",
|
|
34
|
+
description: "The URL of the webpage to extract h2 XPaths from.",
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
required: ["url"],
|
|
38
|
+
},
|
|
39
|
+
},
|
|
40
|
+
],
|
|
41
|
+
};
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
45
|
+
if (request.params.name === "extract_h2_xpaths") {
|
|
46
|
+
const url = request.params.arguments?.url;
|
|
47
|
+
|
|
48
|
+
if (!url || typeof url !== 'string') {
|
|
49
|
+
throw new Error("URL is required and must be a string.");
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
let browser;
|
|
53
|
+
try {
|
|
54
|
+
browser = await puppeteer.launch({
|
|
55
|
+
headless: false,
|
|
56
|
+
args: ['--no-sandbox', '--disable-setuid-sandbox']
|
|
57
|
+
});
|
|
58
|
+
const page = await browser.newPage();
|
|
59
|
+
await page.goto(url, { waitUntil: 'networkidle2' });
|
|
60
|
+
|
|
61
|
+
const xpaths = await page.evaluate(() => {
|
|
62
|
+
function getXPath(element) {
|
|
63
|
+
if (element.id !== '') {
|
|
64
|
+
return 'id("' + element.id + '")';
|
|
65
|
+
}
|
|
66
|
+
if (element === document.body) {
|
|
67
|
+
return element.tagName.toLowerCase();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
let ix = 0;
|
|
71
|
+
let siblings = element.parentNode.childNodes;
|
|
72
|
+
for (let i = 0; i < siblings.length; i++) {
|
|
73
|
+
let sibling = siblings[i];
|
|
74
|
+
if (sibling === element) {
|
|
75
|
+
return getXPath(element.parentNode) + '/' + element.tagName.toLowerCase() + '[' + (ix + 1) + ']';
|
|
76
|
+
}
|
|
77
|
+
if (sibling.nodeType === 1 && sibling.tagName === element.tagName) {
|
|
78
|
+
ix++;
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const h2s = document.querySelectorAll('h2');
|
|
84
|
+
const results = [];
|
|
85
|
+
h2s.forEach(h2 => {
|
|
86
|
+
results.push({
|
|
87
|
+
text: h2.innerText.trim(),
|
|
88
|
+
xpath: getXPath(h2)
|
|
89
|
+
});
|
|
90
|
+
});
|
|
91
|
+
return results;
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
return {
|
|
95
|
+
content: [
|
|
96
|
+
{
|
|
97
|
+
type: "text",
|
|
98
|
+
text: JSON.stringify(xpaths, null, 2),
|
|
99
|
+
},
|
|
100
|
+
],
|
|
101
|
+
};
|
|
102
|
+
} catch (error) {
|
|
103
|
+
return {
|
|
104
|
+
isError: true,
|
|
105
|
+
content: [
|
|
106
|
+
{
|
|
107
|
+
type: "text",
|
|
108
|
+
text: `Error extracting XPaths: ${error.message}`,
|
|
109
|
+
},
|
|
110
|
+
],
|
|
111
|
+
};
|
|
112
|
+
} finally {
|
|
113
|
+
if (browser) {
|
|
114
|
+
await browser.close();
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
throw new Error(`Unknown tool: ${request.params.name}`);
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
async function run() {
|
|
123
|
+
const transport = new StdioServerTransport();
|
|
124
|
+
await server.connect(transport);
|
|
125
|
+
console.error("MCP H2 XPath Extractor Server running on stdio");
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
run().catch((error) => {
|
|
129
|
+
console.error("Fatal error running server:", error);
|
|
130
|
+
process.exit(1);
|
|
131
|
+
});
|
package/package.json
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "mcp-h2-xpath-extractor",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "An MCP server that opens a URL in Chrome and extracts the XPath of all h2 tags.",
|
|
5
|
+
"main": "index.js",
|
|
6
|
+
"type": "module",
|
|
7
|
+
"bin": {
|
|
8
|
+
"mcp-h2-xpath-extractor": "./index.js"
|
|
9
|
+
},
|
|
10
|
+
"scripts": {
|
|
11
|
+
"start": "node index.js"
|
|
12
|
+
},
|
|
13
|
+
"dependencies": {
|
|
14
|
+
"@modelcontextprotocol/sdk": "^1.0.1",
|
|
15
|
+
"puppeteer": "^23.4.1"
|
|
16
|
+
},
|
|
17
|
+
"keywords": ["mcp", "modelcontextprotocol", "xpath", "puppeteer"],
|
|
18
|
+
"author": "",
|
|
19
|
+
"license": "MIT"
|
|
20
|
+
}
|