mcpxray-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,552 @@
1
+ """Static TypeScript / JavaScript extractor.
2
+
3
+ Parses MCP server source **without executing it** (no TypeScript compiler, no
4
+ ``node``) and builds a :class:`~mcpxray.ir.McpServer`. Mirrors
5
+ :mod:`mcpxray.extract.python_static` in shape and philosophy: a stdlib-only
6
+ heuristic, no new dependency.
7
+
8
+ Recognises the two official ``@modelcontextprotocol/sdk`` shapes:
9
+
10
+ * **high-level** ``McpServer``::
11
+
12
+ server.tool("greet", "Greet a user", { name: z.string() }, async (args) => ({ ... }));
13
+ server.registerTool(
14
+ "greet",
15
+ { description: "...", inputSchema: z.object({ name: z.string() }) },
16
+ handler,
17
+ );
18
+
19
+ * **low-level** ``Server``::
20
+
21
+ server.setRequestHandler(
22
+ ListToolsRequestSchema,
23
+ async () => ({ tools: [{ name, description, inputSchema }] }),
24
+ );
25
+
26
+ Tool metadata is read with a small **bracket-matching span scanner** (balance
27
+ ``()`` / ``[]`` / ``{}`` while honouring ``'…'`` / ``"…"`` / `` `…` `` literals
28
+ and ``//`` + ``/* */`` comments) — precise enough for real servers without
29
+ pulling in a parser. Zod / Standard-Schema shapes (``z.string()`` …) are mapped
30
+ to JSON Schema best-effort.
31
+
32
+ Every rule is language-agnostic text scanning over
33
+ :attr:`McpServer.sources`, so populating ``sources`` with each file's text gives
34
+ a TS server secret (MCP102) / RCE (MCP103) / poisoning scanning for free; the
35
+ tool name + description + schema we extract here feed MCP101/104/105/106/107.
36
+ """
37
+
38
+ from __future__ import annotations
39
+
40
+ import json
41
+ import re
42
+ from pathlib import Path
43
+
44
+ from mcpxray.extract.base import Extractor, register_extractor
45
+ from mcpxray.extract.python_static import _iter_source_files
46
+ from mcpxray.ir import SOURCE_STATIC, McpServer, ServerMeta, Tool
47
+
48
+ _TS_EXTS = (".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".mts", ".cts")
49
+ # Declaration / generated bundles are not hand-written server code.
50
+ _SKIP_SUFFIXES = (".d.ts", ".d.mts", ".d.cts", ".min.js", ".min.mjs")
51
+
52
+ # A high-level tool registration: ``<obj>.tool(`` or ``<obj>.registerTool(``.
53
+ # The receiver name is unanchored (``server`` / ``mcp`` / ``app`` …).
54
+ _HL_RE = re.compile(r"\.\s*(?:tool|registerTool)\s*\(")
55
+ # Low-level tool-list handler. Anchored on the actual registration call so a bare
56
+ # ``import { ListToolsRequestSchema }`` doesn't re-extract the same tool array.
57
+ _LOWLEVEL_RE = re.compile(r"setRequestHandler\s*\(\s*ListToolsRequestSchema")
58
+ # ``name: z.<type>`` inside a Zod object / shape.
59
+ _ZOD_PROP_RE = re.compile(r"([A-Za-z_$][\w$]*)\s*:\s*z\s*\.\s*([A-Za-z]+)")
60
+ _ZOD_TYPE = {
61
+ "string": "string",
62
+ "str": "string",
63
+ "number": "number",
64
+ "num": "number",
65
+ "int": "integer",
66
+ "integer": "integer",
67
+ "boolean": "boolean",
68
+ "bool": "boolean",
69
+ "array": "array",
70
+ }
71
+ _NPM_LOCKFILES = {"package-lock.json", "npm-shrinkwrap.json", "pnpm-lock.yaml", "yarn.lock"}
72
+
73
+
74
+ def _iter_typescript_files(root: Path) -> list[Path]:
75
+ """TS/JS source files under ``root`` via the shared walker, dropping declarations."""
76
+ return [f for f in _iter_source_files(root, _TS_EXTS) if not f.name.endswith(_SKIP_SUFFIXES)]
77
+
78
+
79
+ # --- span scanner (bracket-matching, string/comment aware) --------------------
80
+
81
+
82
+ def _skip_string(text: str, i: int) -> int:
83
+ """Index just past the literal starting at ``i`` (a quote char).
84
+
85
+ Handles ``'`` / ``"`` / `` ` `` and backslash escapes. Template-literal
86
+ ``${…}`` is treated opaquely — its inner brackets don't count toward
87
+ balancing, which is exactly what span extraction needs.
88
+ """
89
+ quote = text[i]
90
+ i += 1
91
+ n = len(text)
92
+ while i < n:
93
+ c = text[i]
94
+ if c == "\\" and i + 1 < n:
95
+ i += 2
96
+ continue
97
+ if c == quote:
98
+ return i + 1
99
+ i += 1
100
+ return i
101
+
102
+
103
+ def _match_bracket(text: str, open_idx: int, open_ch: str, close_ch: str) -> int:
104
+ """Index of the bracket matching ``text[open_idx]`` (== ``open_ch``), or ``-1``."""
105
+ depth = 0
106
+ i = open_idx
107
+ n = len(text)
108
+ while i < n:
109
+ c = text[i]
110
+ if c == "/" and i + 1 < n and text[i + 1] == "/":
111
+ nl = text.find("\n", i)
112
+ i = n if nl == -1 else nl + 1
113
+ continue
114
+ if c == "/" and i + 1 < n and text[i + 1] == "*":
115
+ end = text.find("*/", i + 2)
116
+ i = n if end == -1 else end + 2
117
+ continue
118
+ if c in "'\"`":
119
+ i = _skip_string(text, i)
120
+ continue
121
+ if c == open_ch:
122
+ depth += 1
123
+ elif c == close_ch:
124
+ depth -= 1
125
+ if depth == 0:
126
+ return i
127
+ i += 1
128
+ return -1
129
+
130
+
131
+ def _split_top_level(inner: str) -> list[str]:
132
+ """Split a call body by top-level commas (depth-0 wrt ``()`` / ``[]`` / ``{}``)."""
133
+ args: list[str] = []
134
+ depth = 0
135
+ start = 0
136
+ i = 0
137
+ n = len(inner)
138
+ while i < n:
139
+ c = inner[i]
140
+ if c in "'\"`":
141
+ i = _skip_string(inner, i)
142
+ continue
143
+ if c in "([{":
144
+ depth += 1
145
+ elif c in ")]}":
146
+ depth -= 1
147
+ elif c == "," and depth == 0:
148
+ args.append(inner[start:i].strip())
149
+ start = i + 1
150
+ i += 1
151
+ tail = inner[start:].strip()
152
+ if tail:
153
+ args.append(tail)
154
+ return args
155
+
156
+
157
+ def _line_of(text: str, offset: int) -> int:
158
+ """1-indexed line number of ``offset`` in ``text``."""
159
+ return text.count("\n", 0, offset) + 1
160
+
161
+
162
+ # --- literal / field helpers -------------------------------------------------
163
+
164
+
165
+ _ESCAPES = {
166
+ "n": "\n",
167
+ "t": "\t",
168
+ "r": "\r",
169
+ "\\": "\\",
170
+ "'": "'",
171
+ '"': '"',
172
+ "`": "`",
173
+ "0": "\0",
174
+ "b": "\b",
175
+ "f": "\f",
176
+ }
177
+
178
+
179
+ def _unescape(s: str) -> str:
180
+ out: list[str] = []
181
+ i = 0
182
+ n = len(s)
183
+ while i < n:
184
+ c = s[i]
185
+ if c == "\\" and i + 1 < n:
186
+ out.append(_ESCAPES.get(s[i + 1], s[i + 1]))
187
+ i += 2
188
+ continue
189
+ out.append(c)
190
+ i += 1
191
+ return "".join(out)
192
+
193
+
194
+ def _literal_string(a: str) -> str | None:
195
+ """``"foo"`` / ``'foo'`` → ``foo``; anything else → ``None``."""
196
+ a = a.strip()
197
+ if len(a) >= 2 and a[0] in "'\"" and a[-1] == a[0]:
198
+ return _unescape(a[1:-1])
199
+ return None
200
+
201
+
202
+ def _literal_template(a: str) -> str | None:
203
+ """`` `foo` `` (no interpolation) → ``foo``; else ``None``."""
204
+ a = a.strip()
205
+ if len(a) >= 2 and a[0] == "`" and a[-1] == "`":
206
+ return _unescape(a[1:-1])
207
+ return None
208
+
209
+
210
+ def _maybe_string(a: str) -> str | None:
211
+ return _literal_string(a) or _literal_template(a)
212
+
213
+
214
+ def _field_str(text: str, field: str) -> str | None:
215
+ """Value of ``field: "…"`` inside an object literal (best-effort)."""
216
+ m = re.search(r"\b" + re.escape(field) + r"\s*:\s*(['\"])(.*?)\1", text, re.DOTALL)
217
+ return _unescape(m.group(2)) if m else None
218
+
219
+
220
+ def _has_field(text: str, field: str) -> bool:
221
+ return re.search(r"\b" + re.escape(field) + r"\s*:", text) is not None
222
+
223
+
224
+ # --- Zod / JSON-Schema → IR schema -------------------------------------------
225
+
226
+
227
+ def _zod_to_schema(inner: str) -> dict:
228
+ """Map a Zod object/shape body (``a: z.string(), b: z.number()``) to JSON Schema.
229
+
230
+ Unknown Zod types are **skipped** (not emitted as empty ``{}``) so MCP104
231
+ never fires "parameter has no type" on something we merely failed to map.
232
+ Discovered properties are all marked ``required`` — that is Zod's default
233
+ (``.optional()`` is the exception), and it keeps MCP104's "no required
234
+ parameters" rule from firing on every TS tool.
235
+ """
236
+ props: dict[str, dict] = {}
237
+ for m in _ZOD_PROP_RE.finditer(inner):
238
+ jt = _ZOD_TYPE.get(m.group(2).lower())
239
+ if jt:
240
+ props[m.group(1)] = {"type": jt}
241
+ schema: dict = {"type": "object", "properties": props}
242
+ if props:
243
+ schema["required"] = list(props.keys())
244
+ return schema
245
+
246
+
247
+ def _schema_from_json_literal(inner: str) -> dict:
248
+ """Best-effort schema from a low-level JSON-Schema object body."""
249
+ props: dict[str, dict] = {}
250
+ pm = re.search(r"\bproperties\s*:\s*", inner)
251
+ if pm:
252
+ brace = inner.find("{", pm.end())
253
+ if brace != -1:
254
+ end = _match_bracket(inner, brace, "{", "}")
255
+ block = inner[brace + 1 : end if end != -1 else len(inner)]
256
+ for m in re.finditer(r'("?)([A-Za-z_$][\w$]*)\1\s*:\s*\{([^{}]*)\}', block):
257
+ tm = re.search(r'type\s*:\s*["\'](\w+)["\']', m.group(3))
258
+ if tm:
259
+ props[m.group(2)] = {"type": tm.group(1)}
260
+ rm = re.search(r"\brequired\s*:\s*\[([^\]]*)\]", inner)
261
+ required = re.findall(r'["\']([^"\']+)["\']', rm.group(1)) if rm else []
262
+ schema: dict = {"type": "object", "properties": props}
263
+ if required:
264
+ schema["required"] = required
265
+ elif props:
266
+ schema["required"] = list(props.keys())
267
+ return schema
268
+
269
+
270
+ def _schema_from_value(val: str) -> dict:
271
+ """Interpret a schema-ish expression (Zod object, Zod shape, or JSON-Schema literal)."""
272
+ val = val.strip()
273
+ zo = re.match(r"z\s*\.\s*object\s*\(\s*", val)
274
+ if zo: # z.object({...})
275
+ brace = val.find("{", zo.end())
276
+ if brace != -1:
277
+ end = _match_bracket(val, brace, "{", "}")
278
+ if end != -1:
279
+ return _zod_to_schema(val[brace + 1 : end])
280
+ return {}
281
+ if val.startswith("{"):
282
+ end = _match_bracket(val, 0, "{", "}")
283
+ inner = val[1 : end if end != -1 else len(val)]
284
+ if re.search(r"\bz\s*\.", inner): # bare Zod shape { a: z.number() }
285
+ return _zod_to_schema(inner)
286
+ return _schema_from_json_literal(inner)
287
+ return {}
288
+
289
+
290
+ def _field_schema(text: str, field: str) -> dict:
291
+ """Schema following ``field:`` (e.g. ``inputSchema: z.object({...})``)."""
292
+ m = re.search(r"\b" + re.escape(field) + r"\s*:\s*", text)
293
+ if not m:
294
+ return {}
295
+ return _schema_from_value(text[m.end() :])
296
+
297
+
298
+ # --- tool extraction ---------------------------------------------------------
299
+
300
+
301
+ def _is_handler(a: str) -> bool:
302
+ """True if an argument looks like executable code (the trailing handler)."""
303
+ a = a.lstrip()
304
+ return a.startswith(("function", "async function")) or _arrow_index(a) != -1
305
+
306
+
307
+ def _handler_params(a: str) -> list[str] | None:
308
+ """Destructured parameter names a handler reads, for MCP105 drift checking.
309
+
310
+ * arrow ``(params) => …`` or ``async (params) => …`` — params are the last
311
+ ``(…)`` group before ``=>``;
312
+ * ``function (params) {…}`` — the first ``(…)`` group.
313
+
314
+ Only an *object-destructured* first param (``{a, b}``) is comparable: returns
315
+ the names (``[a, b]``). A bare identifier (``args``) returns ``None`` (we
316
+ can't know what it reads); an empty ``()`` returns ``[]`` (reads nothing).
317
+ Renames (``a: x``), defaults (``a = 1``) and rest (``...rest``) are handled.
318
+ """
319
+ a = a.strip()
320
+ arrow = _arrow_index(a)
321
+ if arrow != -1:
322
+ open_idx = a.rfind("(", 0, arrow)
323
+ else:
324
+ fm = re.match(r"(?:async\s+)?function\s*\*?\s*\(", a)
325
+ if not fm:
326
+ return None
327
+ open_idx = a.find("(", fm.start())
328
+ if open_idx == -1:
329
+ return None
330
+ close = _match_bracket(a, open_idx, "(", ")")
331
+ if close == -1:
332
+ return None
333
+ head = a[open_idx + 1 : close].strip()
334
+ if not head:
335
+ return [] # handler takes no params
336
+ if not head.startswith("{"):
337
+ return None # bare identifier — can't tell what it reads
338
+ end = _match_bracket(head, 0, "{", "}")
339
+ body = head[1 : end if end != -1 else len(head)]
340
+ names: list[str] = []
341
+ for part in _split_top_level(body):
342
+ token = part.strip()
343
+ if not token or token.startswith("..."):
344
+ continue
345
+ token = token.split("=", 1)[0].strip() # drop default value
346
+ token = token.split(":", 1)[0].strip() # drop rename / type annotation
347
+ m = re.match(r"[A-Za-z_$][\w$]*", token)
348
+ if m:
349
+ names.append(m.group(0))
350
+ return names
351
+
352
+
353
+ def _arrow_index(a: str) -> int:
354
+ """Index of the top-level ``=>`` in ``a`` (skipping those inside strings), or -1."""
355
+ i = 0
356
+ n = len(a)
357
+ while i < n:
358
+ c = a[i]
359
+ if c in "'\"`":
360
+ i = _skip_string(a, i)
361
+ continue
362
+ if c == "=" and a[i + 1 : i + 2] == ">":
363
+ return i
364
+ i += 1
365
+ return -1
366
+
367
+
368
+ def _tool_name(arg: str) -> str | None:
369
+ """The tool name from ``args[0]`` — a string literal or a ``{ name: "…" }`` config."""
370
+ s = _maybe_string(arg)
371
+ if s is not None:
372
+ return s
373
+ a = arg.strip()
374
+ if a.startswith("{"):
375
+ return _field_str(a, "name")
376
+ return None
377
+
378
+
379
+ def _extract_highlevel(text: str, posix: str, server: McpServer) -> None:
380
+ """``server.tool(...)`` / ``server.registerTool(...)`` registrations."""
381
+ for m in _HL_RE.finditer(text):
382
+ open_idx = m.end() - 1 # the '(' (regex ends right after it)
383
+ close = _match_bracket(text, open_idx, "(", ")")
384
+ if close == -1:
385
+ continue
386
+ args = _split_top_level(text[open_idx + 1 : close])
387
+ if not args:
388
+ continue
389
+ name = _tool_name(args[0])
390
+ if not name:
391
+ continue
392
+
393
+ desc: str | None = None
394
+ schema: dict = {}
395
+ handler: str | None = None
396
+ for arg in args[1:]:
397
+ a = arg.strip()
398
+ if not a:
399
+ continue
400
+ if _is_handler(a):
401
+ handler = a # captured for MCP105 schema/impl-drift checking
402
+ continue
403
+ # registerTool config object: { description, inputSchema, ... }
404
+ if a.startswith("{") and (_has_field(a, "description") or _has_field(a, "inputSchema")):
405
+ if desc is None:
406
+ desc = _field_str(a, "description")
407
+ if not schema:
408
+ schema = _field_schema(a, "inputSchema")
409
+ continue
410
+ s = _maybe_string(a) # bare description literal
411
+ if s is not None and desc is None:
412
+ desc = s
413
+ continue
414
+ if not schema: # z.object / Zod shape / JSON-Schema literal
415
+ sc = _schema_from_value(a)
416
+ if sc:
417
+ schema = sc
418
+
419
+ server.tools.append(
420
+ Tool(
421
+ name=name,
422
+ description=desc,
423
+ input_schema=schema,
424
+ source_path=posix,
425
+ line=_line_of(text, m.start()),
426
+ handler_params=_handler_params(handler) if handler else None,
427
+ )
428
+ )
429
+
430
+
431
+ def _extract_lowlevel(text: str, posix: str, server: McpServer) -> None:
432
+ """``setRequestHandler(ListToolsRequestSchema, () => ({ tools: [...] }))``."""
433
+ for m in _LOWLEVEL_RE.finditer(text):
434
+ tail_start = m.end()
435
+ tm = re.search(r"tools\s*:\s*\[", text[tail_start:])
436
+ if not tm:
437
+ continue
438
+ bracket = tail_start + tm.end() - 1 # the '['
439
+ end = _match_bracket(text, bracket, "[", "]")
440
+ if end == -1:
441
+ continue
442
+ arr = text[bracket + 1 : end]
443
+ i = 0
444
+ n = len(arr)
445
+ while i < n:
446
+ if arr[i] == "{":
447
+ obj_start = i # offset of this tool object's '{' within ``arr``
448
+ obj_end = _match_bracket(arr, obj_start, "{", "}")
449
+ if obj_end == -1:
450
+ break
451
+ obj = arr[obj_start + 1 : obj_end]
452
+ i = obj_end + 1
453
+ name = _field_str(obj, "name")
454
+ if not name:
455
+ continue
456
+ # Line of the object itself (not the handler anchor), so several
457
+ # anchors pointing at the same array de-duplicate to one tool.
458
+ server.tools.append(
459
+ Tool(
460
+ name=name,
461
+ description=_field_str(obj, "description"),
462
+ input_schema=_field_schema(obj, "inputSchema"),
463
+ source_path=posix,
464
+ line=_line_of(text, bracket + 1 + obj_start),
465
+ )
466
+ )
467
+ else:
468
+ i += 1
469
+
470
+
471
+ def _extract_file(path: Path, server: McpServer) -> None:
472
+ # ``errors="replace"`` so one oddly-encoded file can't abort the whole scan.
473
+ text = path.read_text(encoding="utf-8", errors="replace")
474
+ posix = path.as_posix()
475
+ server.sources[posix] = text
476
+ _extract_highlevel(text, posix, server)
477
+ _extract_lowlevel(text, posix, server)
478
+
479
+
480
+ # --- project metadata (dependencies, lockfiles) ------------------------------
481
+
482
+
483
+ def _load_package_json(root: Path, server: McpServer) -> None:
484
+ pkg = root / "package.json"
485
+ if not pkg.is_file():
486
+ return
487
+ try:
488
+ data = json.loads(pkg.read_text(encoding="utf-8"))
489
+ except (json.JSONDecodeError, OSError):
490
+ return
491
+ for section in ("dependencies", "devDependencies", "peerDependencies"):
492
+ deps = data.get(section)
493
+ if isinstance(deps, dict):
494
+ for name, spec in deps.items():
495
+ if isinstance(name, str):
496
+ server.dependencies[name] = str(spec)
497
+ if server.dependencies:
498
+ server.dep_file = str(pkg) # provenance for `--fix`
499
+
500
+
501
+ def _find_npm_lockfiles(root: Path, server: McpServer) -> None:
502
+ for name in _NPM_LOCKFILES:
503
+ if (root / name).is_file():
504
+ server.lockfiles.append(name)
505
+
506
+
507
+ @register_extractor
508
+ class TypescriptExtractor(Extractor):
509
+ """Extract tools from TypeScript/JavaScript MCP server source."""
510
+
511
+ language = "typescript"
512
+
513
+ def applies_to(self, path: Path) -> bool:
514
+ if path.is_file():
515
+ return path.suffix in _TS_EXTS and not path.name.endswith(_SKIP_SUFFIXES)
516
+ if path.is_dir():
517
+ # Parity with PythonExtractor: any TS/JS source present matches. A
518
+ # signal-free tree (e.g. a test dir holding only a fake secret) still
519
+ # scans its ``sources`` text so MCP102/103 can fire.
520
+ return any(True for _ in _iter_typescript_files(path))
521
+ return False
522
+
523
+ def extract(self, path: Path, *, root: Path | None = None) -> McpServer:
524
+ files = [path] if path.is_file() else _iter_typescript_files(path)
525
+ server = McpServer(
526
+ meta=ServerMeta(
527
+ name=path.stem if path.is_file() else path.name,
528
+ language=self.language,
529
+ path=str(path),
530
+ ),
531
+ source_mode=SOURCE_STATIC,
532
+ )
533
+ for f in files:
534
+ _extract_file(f, server)
535
+ # De-duplicate identical registrations (same name + site): a tool can be
536
+ # matched by more than one heuristic span, and the low-level array may be
537
+ # re-scanned across handler occurrences.
538
+ seen: set[tuple[str, str | None, int | None]] = set()
539
+ unique: list[Tool] = []
540
+ for t in server.tools:
541
+ key = (t.name, t.source_path, t.line)
542
+ if key in seen:
543
+ continue
544
+ seen.add(key)
545
+ unique.append(t)
546
+ server.tools = unique
547
+ # Metadata lives at the project root, which may be wider than the scan
548
+ # scope when the caller narrowed ``path`` to a subdirectory.
549
+ meta_root = root or (path if path.is_dir() else path.parent)
550
+ _load_package_json(meta_root, server)
551
+ _find_npm_lockfiles(meta_root, server)
552
+ return server