playwright-e2e-mcp 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +497 -0
  3. package/dist/http.d.ts +27 -0
  4. package/dist/http.js +156 -0
  5. package/dist/http.js.map +1 -0
  6. package/dist/index.d.ts +12 -0
  7. package/dist/index.js +56 -0
  8. package/dist/index.js.map +1 -0
  9. package/dist/server.d.ts +23 -0
  10. package/dist/server.js +225 -0
  11. package/dist/server.js.map +1 -0
  12. package/dist/tools/compare-visual-state.d.ts +77 -0
  13. package/dist/tools/compare-visual-state.js +223 -0
  14. package/dist/tools/compare-visual-state.js.map +1 -0
  15. package/dist/tools/diagnose-flaky.d.ts +87 -0
  16. package/dist/tools/diagnose-flaky.js +258 -0
  17. package/dist/tools/diagnose-flaky.js.map +1 -0
  18. package/dist/tools/generate-e2e-test.d.ts +51 -0
  19. package/dist/tools/generate-e2e-test.js +274 -0
  20. package/dist/tools/generate-e2e-test.js.map +1 -0
  21. package/dist/tools/get-failure.d.ts +26 -0
  22. package/dist/tools/get-failure.js +220 -0
  23. package/dist/tools/get-failure.js.map +1 -0
  24. package/dist/tools/inspect-page.d.ts +57 -0
  25. package/dist/tools/inspect-page.js +124 -0
  26. package/dist/tools/inspect-page.js.map +1 -0
  27. package/dist/tools/list-tests.d.ts +35 -0
  28. package/dist/tools/list-tests.js +88 -0
  29. package/dist/tools/list-tests.js.map +1 -0
  30. package/dist/tools/run-test.d.ts +70 -0
  31. package/dist/tools/run-test.js +169 -0
  32. package/dist/tools/run-test.js.map +1 -0
  33. package/dist/tools/shared.d.ts +86 -0
  34. package/dist/tools/shared.js +490 -0
  35. package/dist/tools/shared.js.map +1 -0
  36. package/dist/tools/validate-selector.d.ts +32 -0
  37. package/dist/tools/validate-selector.js +93 -0
  38. package/dist/tools/validate-selector.js.map +1 -0
  39. package/dist/types/index.d.ts +344 -0
  40. package/dist/types/index.js +51 -0
  41. package/dist/types/index.js.map +1 -0
  42. package/dist/utils/change-analyzer.d.ts +46 -0
  43. package/dist/utils/change-analyzer.js +290 -0
  44. package/dist/utils/change-analyzer.js.map +1 -0
  45. package/dist/utils/image-diff.d.ts +61 -0
  46. package/dist/utils/image-diff.js +373 -0
  47. package/dist/utils/image-diff.js.map +1 -0
  48. package/dist/utils/logger.d.ts +21 -0
  49. package/dist/utils/logger.js +132 -0
  50. package/dist/utils/logger.js.map +1 -0
  51. package/dist/utils/path-utils.d.ts +55 -0
  52. package/dist/utils/path-utils.js +211 -0
  53. package/dist/utils/path-utils.js.map +1 -0
  54. package/dist/utils/playwright-runner.d.ts +63 -0
  55. package/dist/utils/playwright-runner.js +594 -0
  56. package/dist/utils/playwright-runner.js.map +1 -0
  57. package/dist/utils/project-detector.d.ts +43 -0
  58. package/dist/utils/project-detector.js +255 -0
  59. package/dist/utils/project-detector.js.map +1 -0
  60. package/dist/utils/report-parser.d.ts +60 -0
  61. package/dist/utils/report-parser.js +501 -0
  62. package/dist/utils/report-parser.js.map +1 -0
  63. package/dist/utils/trace-reader.d.ts +109 -0
  64. package/dist/utils/trace-reader.js +608 -0
  65. package/dist/utils/trace-reader.js.map +1 -0
  66. package/examples/sample-test.spec.ts +49 -0
  67. package/package.json +66 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 playwright-e2e-mcp contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,497 @@
1
+ # playwright-e2e-mcp
2
+
3
+ [![CI](https://github.com/trajectiq-ai/E2E/actions/workflows/ci.yml/badge.svg)](https://github.com/trajectiq-ai/E2E/actions/workflows/ci.yml) [![release](https://img.shields.io/github/v/release/trajectiq-ai/E2E)](https://github.com/trajectiq-ai/E2E/releases) [![MCP Registry](https://img.shields.io/badge/MCP%20Registry-io.github.trajectiq--ai%2FE2E-2563eb)](https://registry.modelcontextprotocol.io/)
4
+
5
+ An [MCP](https://modelcontextprotocol.io) server that lets AI agents **run, debug, and inspect Playwright end-to-end tests** — with structured results, actionable failure diagnostics, and live DOM inspection.
6
+
7
+ ```
8
+ run-test ──▶ get-failure ──▶ inspect-page ──▶ validate-selector ──▶ fix ──▶ re-run
9
+ ▲ │
10
+ └──────────────────────── list-tests ◀────────────────────────────────┘
11
+ ```
12
+
13
+ Instead of handing an agent raw Playwright output, this server turns every run into
14
+ machinable results: pass/fail stats, per-failure messages with `file:line`, a failure
15
+ kind (assertion, timeout, browser crash, syntax error, dead dev server, full disk…),
16
+ and a concrete "how to fix" hint. When a test fails because a selector no longer
17
+ matches, the agent can open the **live page** in a headless browser, see the real DOM
18
+ with unique CSS selectors, and validate the replacement selector before re-running.
19
+
20
+ ## Install
21
+
22
+ Works with Claude Desktop, Claude Code, Cursor, Windsurf, Codex, Gemini CLI, Freebuff
23
+ and every other MCP client — pick whichever route fits:
24
+
25
+ | Route | How |
26
+ | --- | --- |
27
+ | **MCP Registry** (registry-aware clients discover it automatically) | `io.github.trajectiq-ai/E2E` — [listing](https://registry.modelcontextprotocol.io/) |
28
+ | **Any client, no npm account needed** | `npx -y github:trajectiq-ai/E2E` |
29
+ | **Claude Desktop, zero Node setup** | double-click the [`.mcpb` extension](https://github.com/trajectiq-ai/E2E/releases) |
30
+ | **Remote-only clients (ChatGPT connectors)** | `https://playwright-e2e-mcp.vercel.app/api/mcp` |
31
+
32
+ Details and per-client config: [Installation](#installation) ·
33
+ [MCP client configuration](#mcp-client-configuration).
34
+
35
+ ---
36
+
37
+ ## Tools
38
+
39
+ | Tool | Purpose |
40
+ | --- | --- |
41
+ | `run-test` | Run Playwright tests and return stats, failures, diagnostics and hints |
42
+ | `get-failure` | Deep analysis of one failure: stack, expected/actual, **DOM snapshot at failure (from the Playwright trace)**, next steps |
43
+ | `inspect-page` | Open a URL headlessly and return the rendered DOM: selectors, visibility, boxes, text, console output, HTML |
44
+ | `list-tests` | List available tests (`file`, `line`, full title, projects) with filtering |
45
+ | `validate-selector` | Check a CSS selector against a live page: validity, match count, sample matches |
46
+ | `generate-e2e-test` | Scaffold a Playwright test from a description using the project's **real** selectors, discovered from recent file changes |
47
+ | `compare-visual-state` | Visual regression: screenshot before/after a change and report *what* moved and how colors shifted |
48
+ | `diagnose-flaky` | Run a failing test 2–10 times **with retries disabled** and return an evidence verdict: `CONSISTENTLY FAILING`, `FLAKY` or `NOT REPRODUCING` |
49
+
50
+ ### `run-test`
51
+
52
+ | Argument | Type | Description |
53
+ | --- | --- | --- |
54
+ | `projectRoot` | string | Project directory (default: server working directory) |
55
+ | `testFiles` | string[] | Files/directories relative to the root; `file:line` supported. Omit to run everything |
56
+ | `grep` | string | Only run tests whose title matches this regex |
57
+ | `browser` | `chromium` \| `firefox` \| `webkit` | Playwright project to run (matched against config project names) |
58
+ | `headed` | boolean | Visible browser window |
59
+ | `timeoutMs` | number | Hard wall-clock limit for the run (default `120000`); the whole process tree is killed past it and **partial results are returned** |
60
+ | `testTimeoutMs` | number | Per-test timeout passed to Playwright |
61
+ | `workers` / `retries` | number | Passed through to Playwright |
62
+ | `config` | string | `playwright.config` path **or 1-based index** when the project has several |
63
+ | `retryOnFailure` | boolean | Auto-retry failures **once** before reporting them (default `true`; ignored when `retries` is set) |
64
+ | `lastFailed` | boolean | Only re-run tests that failed in the previous run (Playwright `--last-failed`) — the fast fix → re-run loop |
65
+ | `args` | string[] | Extra CLI flags (shell metacharacters are rejected) |
66
+
67
+ Flakiness handling: by default the server injects `--retries=1` (unless the config
68
+ already sets `retries`), so a test that passes on the retry is reported as **flaky**,
69
+ not failed. Traces are captured automatically (`--trace=retain-on-failure`) so
70
+ `get-failure` can show the DOM at the moment of failure.
71
+
72
+ Example result:
73
+
74
+ ```markdown
75
+ ## Playwright run — ❌ FAILED
76
+
77
+ **Command:** `playwright test --config playwright.config.ts tests/checkout.spec.ts --reporter=json`
78
+ **duration 4.2s · exit 1 · config `playwright.config.ts`**
79
+
80
+ | passed | failed | flaky | skipped | duration |
81
+ | ---: | ---: | ---: | ---: | ---: |
82
+ | 0 | 1 | 0 | 0 | 1.1s |
83
+
84
+ ### ❌ 1 failing test(s)
85
+
86
+ ### 1 of 1. checkout.spec.ts › pays with card
87
+ **File:** `checkout.spec.ts:5` | **failed · server-unreachable**
88
+
89
+ ### ⚠️ SERVER_NOT_RUNNING
90
+ Your app (dev server) does not appear to be reachable. Start it in another terminal
91
+ (e.g. npm run dev / npm start), keep it running, then retry — or configure `webServer`
92
+ in playwright.config.* so Playwright starts it automatically.
93
+ ```
94
+
95
+ ### `get-failure`
96
+
97
+ | Argument | Type | Description |
98
+ | --- | --- | --- |
99
+ | `index` | number | 1-based failure index from the last run (default `1`) |
100
+ | `projectRoot` | string | Only used when re-reading the stored report |
101
+
102
+ Returns the message/code frame, expected vs actual, stack, failure kind with a
103
+ diagnosis, the test's console output, **the DOM snapshot from the Playwright trace
104
+ (plus the failed action, its selector, and the action log leading up to it)**,
105
+ **the network requests that failed** (4xx/5xx, dead endpoints, no-response — with
106
+ method, URL, status and resource type), **the console errors/warnings the page
107
+ logged before the failure**, and
108
+ numbered next steps (re-run this single test by `file:line`, headed/debug mode,
109
+ `validate-selector` when the message mentions a locator, …).
110
+
111
+ ### `inspect-page`
112
+
113
+ | Argument | Type | Description |
114
+ | --- | --- | --- |
115
+ | `url` | string | Full http(s) URL to open (required) |
116
+ | `projectRoot` | string | Project whose Playwright launches the browser |
117
+ | `selector` | string | Inspect matches of this CSS selector instead of the whole DOM |
118
+ | `waitFor` | string | Wait for a selector (CSS or `text=…`) before inspecting |
119
+ | `waitUntil` | `load` \| `domcontentloaded` \| `networkidle` | Navigation wait condition |
120
+ | `includeHtml` | boolean | Include the rendered HTML (capped) |
121
+ | `maxHtmlChars` | number | HTML cap, default `20000` |
122
+ | `timeoutMs` | number | Overall limit, default `45000` |
123
+
124
+ Returns each element's **unique CSS selector**, tag, visibility, bounding box, text and
125
+ attributes, plus captured console messages (errors first).
126
+
127
+ ### `list-tests`
128
+
129
+ | Argument | Type | Description |
130
+ | --- | --- | --- |
131
+ | `projectRoot` | string | Project directory |
132
+ | `config` | string | Config path or 1-based index |
133
+ | `testDir` | string | Restrict scanning to a directory (must stay inside the project) |
134
+ | `filter` | string | Case-insensitive substring filter on `file › title` |
135
+ | `limit` | number | Max tests returned, default `500` |
136
+
137
+ Uses `playwright test --list` when Playwright works, and **falls back to a source scan**
138
+ (keeping the reason) when the install or a spec file is broken.
139
+
140
+ ### `validate-selector`
141
+
142
+ | Argument | Type | Description |
143
+ | --- | --- | --- |
144
+ | `url` | string | Live page to test against (required) |
145
+ | `selector` | string | CSS selector to validate (required) |
146
+ | `projectRoot` | string | Project whose Playwright launches the browser |
147
+ | `timeoutMs` | number | Overall limit, default `45000` |
148
+
149
+ Verdicts: `✅ VALID — N matches` (with a sample of matches), `✅ VALID — 0 matches`
150
+ (with debugging advice), `❌ INVALID` (parse error + fix), or a warning when the input
151
+ uses a Playwright-only engine (`text=`, `xpath=`, `>>`, `:has-text()`), which is not
152
+ plain CSS.
153
+
154
+ ### `generate-e2e-test`
155
+
156
+ | Argument | Type | Description |
157
+ | --- | --- | --- |
158
+ | `description` | string | What the test should cover (required) |
159
+ | `pageUrl` | string | Page the test starts on (default: `baseURL` / `webServer.url` from config) |
160
+ | `testDir` / `file` | string | Where to write the spec (default: detected `testDir` + `generated/<slug>.spec.ts`) |
161
+ | `write` | boolean | Write the file to disk (default `true`) |
162
+ | `overwrite` | boolean | Replace an existing file at the target path |
163
+ | `liveInspect` | boolean | Cross-check selectors against the live page (default on when a URL is known) |
164
+ | `projectRoot` / `config` | string | As with the other tools |
165
+
166
+ Reads the agent's recent changes (`git status`, falling back to `git diff HEAD~1`,
167
+ then recent mtimes), extracts the locators those files actually declare
168
+ (`data-testid`, `getByRole`, `aria-label`, `placeholder`, `id`, `name`, element text),
169
+ ranks verified-live selectors first, writes a spec built from them, and reports each
170
+ selector with its source `file:line`.
171
+
172
+ ### `compare-visual-state`
173
+
174
+ | Argument | Type | Description |
175
+ | --- | --- | --- |
176
+ | `url` | string | Page to capture (required) |
177
+ | `name` | string | Baseline id, e.g. `checkout-page` (letters, digits, `. _ -`) |
178
+ | `action` | `compare` \| `baseline` | `compare` (default) diffs; `baseline` re-captures the reference |
179
+ | `selector` | string | Capture just this element |
180
+ | `fullPage` | boolean | Capture the full scrollable page |
181
+ | `tolerance` | number | Percent of pixels that may differ (default `0.1`) |
182
+ | `pixelThreshold` | number | Per-pixel channel delta considered different (default `60`) |
183
+ | `waitUntil` / `waitFor` / `timeoutMs` | — | As with `inspect-page` |
184
+
185
+ The first call saves a baseline under `.pw-mcp/visual/` (add that to `.gitignore`, or
186
+ commit it for CI comparisons). Later calls report changed-pixel counts, **merged
187
+ regions** (`(x, y) 120×40 — 1,200 px`), the **average color shift** ("blue → red"),
188
+ and write a red-highlighted diff image for review.
189
+
190
+ ### `diagnose-flaky`
191
+
192
+ | Argument | Type | Description |
193
+ | --- | --- | --- |
194
+ | `testFiles` | string[] | Tests to diagnose (`file:line` supported). Defaults to the tests that failed in the most recent run |
195
+ | `runs` | number | Times to run them, `2–10` (default `3`) |
196
+ | `browser` / `headed` / `workers` / `config` | — | As with `run-test` |
197
+ | `timeoutMs` | number | Hard wall-clock limit **per run** (default `120000`) |
198
+ | `projectRoot` | string | Project directory |
199
+
200
+ Every run executes with `--retries=0` and auto-retry disabled, so each result is
201
+ honest evidence. The response contains a per-run table (status, duration, first
202
+ failure), the count of **distinct normalized error signatures**, and one of:
203
+
204
+ - **❌ CONSISTENTLY FAILING** — failed every run (same error → reproducible bug,
205
+ different errors → still broken, just noisy). Fix it; it is not flaky.
206
+ - **⚠️ FLAKY** — some runs passed. Includes `N of M` counts and whether the
207
+ failures share one signature (real intermittent bug) or vary (timing/environment
208
+ instability).
209
+ - **✅ NOT REPRODUCING** — passed every re-run; the original failure was one-off.
210
+
211
+ The last run is stored, so `get-failure` can analyze it immediately afterwards.
212
+
213
+ ---
214
+
215
+ ## Installation
216
+
217
+ Requirements:
218
+
219
+ - Node.js **≥ 20** (the server is built on MCP SDK v2 — the `2026-07-28` spec line)
220
+ - A project with `@playwright/test` installed and browsers available
221
+ (`npx playwright install chromium`)
222
+
223
+ **No npm account needed** — install straight from GitHub (the `prepare` script
224
+ builds `dist/` automatically on install):
225
+
226
+ ```bash
227
+ npx -y github:trajectiq-ai/E2E
228
+ npm install -D github:trajectiq-ai/E2E @playwright/test # or as a project dependency
229
+ npx playwright install chromium
230
+ ```
231
+
232
+ Or grab the packaged tarball from the repo's **GitHub Releases** page and install
233
+ it locally:
234
+
235
+ ```bash
236
+ npm install -D https://github.com/trajectiq-ai/E2E/releases/download/v0.1.0/playwright-e2e-mcp-0.1.0.tgz
237
+ ```
238
+
239
+ Listed in the **official [MCP Registry](https://registry.modelcontextprotocol.io/)** as
240
+ `io.github.trajectiq-ai/E2E` — registry-aware clients discover it there, and every
241
+ `v*` release tag republishes the entry from CI via [`server.json`](server.json).
242
+
243
+ ### MCP client configuration
244
+
245
+ **Claude Code / generic (project-scoped):**
246
+
247
+ ```json
248
+ {
249
+ "mcpServers": {
250
+ "playwright-e2e": {
251
+ "command": "npx",
252
+ "args": ["-y", "github:trajectiq-ai/E2E"],
253
+ "env": { "PW_MCP_PROJECT_ROOT": "/absolute/path/to/your/project" }
254
+ }
255
+ }
256
+ }
257
+ ```
258
+
259
+ **Claude Desktop / Cursor / Windsurf:** add the same block to their MCP config file.
260
+ The server uses its working directory as the project root; set `PW_MCP_PROJECT_ROOT`
261
+ when the client launches it somewhere else (e.g. your home directory).
262
+
263
+ **Codex / VS Code / Copilot CLIs:**
264
+
265
+ ```bash
266
+ codex mcp add playwright-e2e -- npx -y github:trajectiq-ai/E2E
267
+ code --add-mcp '{"name":"playwright-e2e","command":"npx","args":["-y","github:trajectiq-ai/E2E"]}'
268
+ ```
269
+
270
+ Codex's defaults fight this server: the first launch clones the repo and runs
271
+ `tsc` (measured 30 s on a cold `npx` cache, against a 10 s
272
+ `startup_timeout_sec` default), and a Playwright run with retries beats the 60 s
273
+ `tool_timeout_sec` default. Raise both in `~/.codex/config.toml`:
274
+
275
+ ```toml
276
+ [mcp_servers.playwright-e2e]
277
+ command = "npx"
278
+ args = ["-y", "github:trajectiq-ai/E2E"]
279
+ startup_timeout_sec = 60
280
+ tool_timeout_sec = 600
281
+ ```
282
+
283
+ **Claude Desktop (one-click):** download and double-click the `.mcpb` Desktop
284
+ Extension attached to the [latest release](https://github.com/trajectiq-ai/E2E/releases) —
285
+ the bundle ships its own dependencies, so no Node setup is required. On install it
286
+ prompts once for your **project root** (defaults to your home folder) and wires it
287
+ into `PW_MCP_PROJECT_ROOT`, so the tools point at a real project from the first call.
288
+
289
+ **Claude Code:**
290
+
291
+ ```bash
292
+ claude mcp add playwright-e2e -- npx -y github:trajectiq-ai/E2E
293
+ ```
294
+
295
+ **Gemini CLI / Qwen Code:** paste the `mcpServers` block above into
296
+ `.gemini/settings.json` (Qwen Code: `.qwen/settings.json`) — both speak the same
297
+ MCP settings format.
298
+
299
+ **Freebuff / Codebuff (project-scoped):** this repo ships a committed
300
+ [`.agents/mcp.json`](.agents/mcp.json), so opening the checkout in Freebuff
301
+ attaches the server workspace-wide — no global config needed. Your own projects
302
+ can do the same: drop an `mcp.json` with the block above into their `.agents/`
303
+ directory. Freebuff asks you to trust a repository's `.agents/` on first run.
304
+
305
+ All tools ship MCP **tool annotations** (`readOnlyHint`, `destructiveHint`,
306
+ `idempotentHint`, `openWorldHint`), so clients can show accurate safety prompts
307
+ before running anything.
308
+
309
+ **From a local checkout:**
310
+
311
+ ```json
312
+ {
313
+ "mcpServers": {
314
+ "playwright-e2e": {
315
+ "command": "node",
316
+ "args": ["/path/to/playwright-e2e-mcp/dist/index.js"],
317
+ "env": { "PW_MCP_PROJECT_ROOT": "/path/to/your/project" }
318
+ }
319
+ }
320
+ }
321
+ ```
322
+
323
+ ## Hosted endpoint (ChatGPT & remote clients)
324
+
325
+ Some clients — ChatGPT custom connectors especially — only accept **remote
326
+ HTTPS** MCP servers and refuse to spawn a local `npx` process. This repo ships a
327
+ Streamable HTTP bridge for exactly that case:
328
+
329
+ | | |
330
+ | --- | --- |
331
+ | **Endpoint** | `https://playwright-e2e-mcp.vercel.app/api/mcp` |
332
+ | **Transport** | MCP Streamable HTTP (`POST` JSON in, JSON or SSE out) |
333
+ | **Auth** | none — the URL is public |
334
+ | **Source** | [`api/mcp.ts`](api/mcp.ts) → [`src/http.ts`](src/http.ts) |
335
+
336
+ The bridge runs the *same* `createServer()` as the stdio transport; the SDK
337
+ serves every request with a fresh server instance, which is what a serverless
338
+ function wants. `test/http-bridge.test.mjs` drives the real Node adapter over
339
+ `node:http` so a broken bridge fails in CI, not in ChatGPT.
340
+
341
+ **Add it to ChatGPT:** Settings → Connectors → turn on **Advanced → Developer
342
+ mode** → *Create custom connector* → paste the endpoint above → authentication
343
+ **None**.
344
+
345
+ Codex can also take the remote transport instead of spawning `npx`, if you'd
346
+ rather not ship Playwright to every machine:
347
+
348
+ ```bash
349
+ codex mcp add playwright-e2e-remote --url https://playwright-e2e-mcp.vercel.app/api/mcp
350
+ ```
351
+
352
+ **What to expect:** `list-tests` works and reports the specs bundled with the
353
+ deployment. Tools that spawn a browser (`run-test`, `inspect-page`,
354
+ `validate-selector`, `diagnose-flaky`, …) cannot download Chromium in a
355
+ serverless function, so they return their normal `NO_PLAYWRIGHT` hint. Use the
356
+ stdio install for real runs; the hosted endpoint is for discovery and for
357
+ clients that cannot run local processes.
358
+
359
+ ```bash
360
+ # verify the handshake without any client
361
+ curl -X POST https://playwright-e2e-mcp.vercel.app/api/mcp \
362
+ -H 'content-type: application/json' \
363
+ -H 'accept: application/json, text/event-stream' \
364
+ -d '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"curl","version":"1.0"}}}'
365
+ ```
366
+
367
+ Redeploy after a change:
368
+
369
+ ```bash
370
+ npx vercel deploy --yes --prod --token="$VERCEL_TOKEN"
371
+ ```
372
+
373
+ ## Configuration
374
+
375
+ | Environment variable | Default | Purpose |
376
+ | --- | --- | --- |
377
+ | `PW_MCP_PROJECT_ROOT` | server cwd | Default project root for every tool |
378
+ | `LOG_LEVEL` | `info` | `debug` \| `info` \| `warn` \| `error` \| `silent` |
379
+ | `LOG_FORMAT` | `text` | `text` or `json` (structured) |
380
+
381
+ Logs always go to **stderr** — stdout is reserved for the MCP protocol.
382
+
383
+ ## Typical workflow
384
+
385
+ 1. `generate-e2e-test` `{ "description": "checkout with a saved card" }` — scaffolds a
386
+ spec from your real selectors (skipped if you write the test yourself).
387
+ 2. `list-tests` — see what exists (`tests/checkout.spec.ts:5 checkout › pays with card`).
388
+ 3. `run-test` `{ "testFiles": ["tests/checkout.spec.ts"] }` — run it; get stats + failures
389
+ (flaky tests are auto-retried once before being called failures).
390
+ 4. `get-failure` `{ "index": 1 }` — read the code frame, expected/actual, **the DOM
391
+ snapshot at failure from the trace**, the failed network requests, the page's
392
+ console errors, and next steps.
393
+ 5. If it looks selector-related: `inspect-page` `{ "url": "http://localhost:3000/checkout" }`
394
+ to see the real DOM, then `validate-selector` to prove the replacement selector works.
395
+ 6. After changing CSS/components: `compare-visual-state` `{ "url": "…", "name": "checkout" }`
396
+ to catch unintended visual regressions.
397
+ 7. If a failure looks intermittent: `diagnose-flaky` `{ "runs": 3 }` — get the evidence
398
+ verdict (flaky vs consistently broken) before deciding what to fix.
399
+ 8. Fix the spec or the app, then re-run **only what failed**:
400
+ `run-test` `{ "lastFailed": true }`, and repeat until green.
401
+
402
+ ## Edge cases handled
403
+
404
+ | Situation | Behaviour |
405
+ | --- | --- |
406
+ | No Playwright installed | `NO_PLAYWRIGHT` error with the exact install commands for your package manager |
407
+ | Dev server not running | Failure classified `server-unreachable` / `SERVER_NOT_RUNNING` with a "start your dev server" hint (and `webServer` advice) |
408
+ | Test exceeds `timeoutMs` | Process **group** is killed (SIGINT→SIGKILL on POSIX, `taskkill /T /F` on Windows) and partial results are returned |
409
+ | Browser crashes | Classified `browser-crash` with retry / reinstall guidance |
410
+ | Windows backslashes | All paths normalized lexically (`C:\a\..\b` → `C:/b`); unit-tested on both platforms |
411
+ | Flaky tests | Failing tests are automatically retried once (`--retries=1`) before being reported; passes surface as **flaky** with a stability warning; `diagnose-flaky` decides flaky-vs-broken with multi-run evidence |
412
+ | Trace/DOM context | `--trace=retain-on-failure` is passed automatically, so `get-failure` can show the exact DOM at the moment of failure — plus the failed network requests (`*.network` logs) and console errors from the same trace |
413
+ | Slow re-runs after a fix | `run-test` with `lastFailed: true` re-runs only the tests that failed last time (`--last-failed`) |
414
+ | Several `playwright.config` files | Returns a numbered menu (`MULTIPLE_CONFIGS`); pick with `config: "2"` or a path |
415
+ | Syntax error in a spec | `SYNTAX_ERROR` with file:line; nothing crashes; `list-tests` falls back to a source scan |
416
+ | MCP client disconnects | Per-request `AbortSignal` kills the run; stdin end triggers shutdown, and every tracked child tree is force-killed (`killActiveChildren`) |
417
+ | Disk full | `ENOSPC` detected → `DISK_FULL` with a "free space" hint; logging never throws |
418
+ | Malicious paths | `../../etc/passwd`, absolute paths outside the root, URLs and null bytes are rejected with `INVALID_PATH`; CLI args are shell-metacharacter-checked |
419
+
420
+ ## Security notes
421
+
422
+ - **No shell**: Playwright is spawned as `node <playwright/cli.js> …` with an argument
423
+ array — no command interpolation.
424
+ - **Path sandbox**: user paths are resolved lexically and must stay inside the project root.
425
+ - **Cleanup**: temp report/script files are written to the OS temp dir and removed;
426
+ child processes are tracked and killed on shutdown.
427
+
428
+ ## Development
429
+
430
+ ```
431
+ src/
432
+ ├── index.ts # bin entry point (--version/--help, main-module guard)
433
+ ├── server.ts # McpServer setup, tool registration, shutdown handling
434
+ ├── tools/ # the eight tools + shared plumbing
435
+ ├── utils/ # playwright-runner, report-parser, project-detector, path-utils,
436
+ │ # logger, trace-reader (trace.zip → DOM/network/console),
437
+ │ # image-diff (PNG codec + pixel diff), change-analyzer
438
+ └── types/ # shared interfaces and the ErrorKind taxonomy
439
+ ```
440
+
441
+ Built on **`@modelcontextprotocol/server` v2** (the `2026-07-28` MCP spec line) with
442
+ Zod v4 standard schemas; every tool declares spec tool annotations.
443
+
444
+ ```bash
445
+ npm install
446
+ npm run build # tsc → dist/ (zero errors)
447
+ npm test # build + test/run-tests.mjs (70 unit tests, any Node ≥20)
448
+ npm run e2e # build + e2e/run.mjs: live MCP ↔ Playwright integration suite
449
+ ```
450
+
451
+ Tests cover the report parser (sample Playwright JSON, trace attachments), path utils
452
+ (Windows and macOS paths, sandboxing), the project detector (temp-dir fixtures: config
453
+ discovery, multiple configs, missing install, test-file scanning), the shared tool
454
+ helpers, the **trace reader** (synthetic trace.zip: error, failed action, DOM snapshot,
455
+ `*.network` failed-request parsing, console error/warning events), the **image diff**
456
+ (PNG round-trip, regions, color shift, dimension changes), the **change analyzer**
457
+ (selector extraction, git + mtime paths) and the **flaky verdict logic**
458
+ (failure signatures, CONSISTENTLY FAILING / FLAKY / NOT REPRODUCING / NO TESTS RAN).
459
+
460
+ ### Integration suite (`npm run e2e`)
461
+
462
+ Unit tests prove the logic; the integration suite proves the loop. It boots the real
463
+ server over stdio against a live fixture app and a Playwright project under
464
+ `e2e/fixture/`, then drives it exactly like an MCP client and asserts ~40 behaviours
465
+ that only appear end-to-end:
466
+
467
+ - initialize handshake, 8 tools, spec tool annotations and object input schemas,
468
+ - live DOM inspection, CSS selector validation (matches, zero matches, engine
469
+ syntax, parse errors), dead-server detection,
470
+ - visual regression: baseline → unchanged compare → `blue → red` diff detection,
471
+ - `run-test` pass/fail/`lastFailed` stats and meta lines,
472
+ - `get-failure` trace diagnostics: DOM at failure, the 404 network request,
473
+ the `console.error` message, diagnosis and next steps,
474
+ - auto-retry turning a first-run failure into `PASSED (1 flaky)`,
475
+ - `diagnose-flaky` verdict **FLAKY** (2 of 3 runs) with retries disabled,
476
+ - `generate-e2e-test` writing its scaffold, plus error paths
477
+ (missing test path, unknown tool, unreachable server).
478
+
479
+ First run needs the browser once: `npx playwright install chromium`.
480
+ CI runs the suite on Ubuntu and Windows (see `.github/workflows/ci.yml`).
481
+
482
+ ### Try the example
483
+
484
+ With Playwright installed in your project:
485
+
486
+ ```bash
487
+ npx playwright test examples/sample-test.spec.ts
488
+ ```
489
+
490
+ or ask your agent to call `run-test` with
491
+ `"testFiles": ["examples/sample-test.spec.ts"]` — it hits the public
492
+ [example.com](https://example.com) page, so it verifies browsers, network and the MCP
493
+ pipeline in one shot.
494
+
495
+ ## License
496
+
497
+ MIT
package/dist/http.d.ts ADDED
@@ -0,0 +1,27 @@
1
+ /**
2
+ * Streamable HTTP bridge for the MCP server.
3
+ *
4
+ * `createMcpHandler` (MCP SDK) is fetch-shaped: it answers the 2025
5
+ * streamable-HTTP transport *statelessly* — a fresh server per request, which
6
+ * is exactly what a serverless function wants — and the modern envelope
7
+ * transport on the same URL. This module adds the thin adapter for runtimes
8
+ * that hand you Node's `IncomingMessage`/`ServerResponse` pair (Vercel
9
+ * functions, a plain `node:http` server), so the very same code path can be
10
+ * verified locally before it is deployed.
11
+ *
12
+ * The stdio transport in server.ts is untouched; this is an additional way to
13
+ * reach the same eight tools.
14
+ */
15
+ import type { IncomingMessage, ServerResponse } from 'node:http';
16
+ import type { McpHttpHandler } from '@modelcontextprotocol/server';
17
+ /** Mirrors the SDK's default POST body bound, so oversized bodies die early. */
18
+ export declare const MAX_BODY_BYTES: number;
19
+ /** Build the fetch-shaped MCP handler backed by a fresh server per request. */
20
+ export declare function createMcpHttpHandler(): McpHttpHandler;
21
+ /**
22
+ * Serve one Node request through the MCP handler.
23
+ *
24
+ * Always settles: a transport failure becomes a JSON error response, and a
25
+ * mid-stream failure closes the socket so the client stops waiting.
26
+ */
27
+ export declare function handleNodeRequest(handler: McpHttpHandler, req: IncomingMessage, res: ServerResponse): Promise<void>;