graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/manifest.py
ADDED
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
"""Deterministic package-manifest ingestion (#1377).
|
|
2
|
+
|
|
3
|
+
Package manifests (``apm.yml``, ``pyproject.toml``, ``Cargo.toml``, ``go.mod``,
|
|
4
|
+
``pom.xml``) declare a package and its dependencies. Left to the LLM document path, the same
|
|
5
|
+
package gets a different file-anchored node id from its own manifest than from
|
|
6
|
+
each dependent's dependency reference, so it splits into duplicate nodes. This
|
|
7
|
+
module parses manifests deterministically and emits ONE canonical package node
|
|
8
|
+
per package -- keyed by NAME via :func:`graphify.ids.make_id` -- plus
|
|
9
|
+
``depends_on`` edges, so a package referenced from N manifests collapses to a
|
|
10
|
+
single hub node (the dependency stub and the package's own definition node share
|
|
11
|
+
the canonical id and merge at build time).
|
|
12
|
+
|
|
13
|
+
Mirrors ``mcp_ingest``: recognized by filename, routed to the deterministic AST
|
|
14
|
+
path (never the LLM), so a manifest is extracted exactly once.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
import xml.etree.ElementTree as ET
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from graphify.ids import make_id
|
|
24
|
+
|
|
25
|
+
__all__ = ["is_package_manifest_path", "extract_package_manifest", "PACKAGE_MANIFEST_NAMES"]
|
|
26
|
+
|
|
27
|
+
# manifest filename (lowercased) -> ecosystem tag
|
|
28
|
+
PACKAGE_MANIFEST_NAMES: dict[str, str] = {
|
|
29
|
+
"apm.yml": "apm",
|
|
30
|
+
"apm.yaml": "apm",
|
|
31
|
+
"pyproject.toml": "python",
|
|
32
|
+
"cargo.toml": "cargo",
|
|
33
|
+
"go.mod": "go",
|
|
34
|
+
"pom.xml": "maven",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
_MAX_MANIFEST_BYTES = 2_000_000 # 2 MB cap — manifests are small; this rejects junk
|
|
38
|
+
|
|
39
|
+
_TOMLI_REQUIRED = (
|
|
40
|
+
"Package-manifest ingestion on Python < 3.11 needs tomli. "
|
|
41
|
+
"Install with: pip install 'tomli' "
|
|
42
|
+
"(or reinstall graphifyy, which declares tomli for python_version < '3.11')."
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _load_toml_module():
|
|
47
|
+
"""Return a tomllib-compatible module, or raise ImportError (#3283).
|
|
48
|
+
|
|
49
|
+
Returning ``None`` used to look identical to a virtual workspace root with
|
|
50
|
+
nothing to emit, so missing ``tomli`` on Python 3.10 silently dropped every
|
|
51
|
+
``Cargo.toml`` / ``pyproject.toml``. Raise instead: the caller
|
|
52
|
+
(``extract_package_manifest``) surfaces this as a visible per-manifest error
|
|
53
|
+
rather than dropping the file silently. In practice the runtime ``tomli``
|
|
54
|
+
dependency (python_version < '3.11') keeps this path unreachable for a
|
|
55
|
+
standard install.
|
|
56
|
+
"""
|
|
57
|
+
try:
|
|
58
|
+
import tomllib as _toml # type: ignore[import-not-found]
|
|
59
|
+
return _toml
|
|
60
|
+
except ImportError:
|
|
61
|
+
try:
|
|
62
|
+
import tomli as _toml # type: ignore[import-not-found,no-redef]
|
|
63
|
+
return _toml
|
|
64
|
+
except ImportError as exc:
|
|
65
|
+
raise ImportError(_TOMLI_REQUIRED) from exc
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def is_package_manifest_path(path: Path) -> bool:
|
|
69
|
+
"""True if ``path`` is a recognized package manifest (by filename)."""
|
|
70
|
+
return path.name.lower() in PACKAGE_MANIFEST_NAMES
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _pkg_id(name: str) -> str:
|
|
74
|
+
"""Canonical package node id, keyed by package NAME so every reference to the
|
|
75
|
+
same package -- its own manifest and any dependent's dependency line -- maps
|
|
76
|
+
to one node."""
|
|
77
|
+
return make_id("pkg", name)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def extract_package_manifest(path: Path) -> dict[str, Any]:
|
|
81
|
+
"""Parse a package manifest into a canonical package node + ``depends_on`` edges."""
|
|
82
|
+
try:
|
|
83
|
+
if path.stat().st_size > _MAX_MANIFEST_BYTES:
|
|
84
|
+
return {"nodes": [], "edges": [], "error": "manifest too large to index"}
|
|
85
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
86
|
+
except OSError as exc:
|
|
87
|
+
return {"nodes": [], "edges": [], "error": f"manifest read error: {exc}"}
|
|
88
|
+
|
|
89
|
+
eco = PACKAGE_MANIFEST_NAMES[path.name.lower()]
|
|
90
|
+
try:
|
|
91
|
+
info = _PARSERS[eco](text)
|
|
92
|
+
except Exception as exc: # noqa: BLE001 — a malformed manifest must not abort extraction
|
|
93
|
+
return {"nodes": [], "edges": [], "error": f"manifest parse error: {exc}"}
|
|
94
|
+
if not info or not info.get("name"):
|
|
95
|
+
return {"nodes": [], "edges": []}
|
|
96
|
+
|
|
97
|
+
name = info["name"]
|
|
98
|
+
str_path = str(path)
|
|
99
|
+
pkg_nid = _pkg_id(name)
|
|
100
|
+
node: dict[str, Any] = {
|
|
101
|
+
"id": pkg_nid,
|
|
102
|
+
"label": name,
|
|
103
|
+
"file_type": "code", # valid schema type; `type` distinguishes packages
|
|
104
|
+
"type": "package",
|
|
105
|
+
"ecosystem": eco,
|
|
106
|
+
"source_file": str_path,
|
|
107
|
+
"source_location": "L1",
|
|
108
|
+
}
|
|
109
|
+
if info.get("version"):
|
|
110
|
+
node["version"] = info["version"]
|
|
111
|
+
nodes: list[dict] = [node]
|
|
112
|
+
edges: list[dict] = []
|
|
113
|
+
|
|
114
|
+
seen: set[str] = set()
|
|
115
|
+
for dep in info.get("deps", []):
|
|
116
|
+
if not dep:
|
|
117
|
+
continue
|
|
118
|
+
dep_nid = _pkg_id(dep)
|
|
119
|
+
if dep_nid == pkg_nid or dep_nid in seen:
|
|
120
|
+
continue
|
|
121
|
+
seen.add(dep_nid)
|
|
122
|
+
# The edge targets the dependency's canonical package id. If that package's
|
|
123
|
+
# own manifest is in the corpus, the edge resolves to its (single) node; if
|
|
124
|
+
# the dependency is external, build_from_json prunes the dangling edge. We
|
|
125
|
+
# deliberately do NOT emit a stub node — a stub with an empty source_file
|
|
126
|
+
# would risk clobbering the real node's source_file under id-dedup.
|
|
127
|
+
edges.append({
|
|
128
|
+
"source": pkg_nid,
|
|
129
|
+
"target": dep_nid,
|
|
130
|
+
"relation": "depends_on",
|
|
131
|
+
"context": "dependency",
|
|
132
|
+
"confidence": "EXTRACTED",
|
|
133
|
+
"confidence_score": 1.0,
|
|
134
|
+
"source_file": str_path,
|
|
135
|
+
"source_location": "L1",
|
|
136
|
+
"weight": 1.0,
|
|
137
|
+
})
|
|
138
|
+
return {"nodes": nodes, "edges": edges}
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# ── per-ecosystem parsers: text -> {"name", "version"?, "deps": [str]} | None ──
|
|
142
|
+
|
|
143
|
+
def _coerce_deps(value: Any) -> list[str]:
|
|
144
|
+
"""A dependency block may be a list of names or a name->spec map."""
|
|
145
|
+
if isinstance(value, dict):
|
|
146
|
+
return [str(k) for k in value]
|
|
147
|
+
if isinstance(value, list):
|
|
148
|
+
out: list[str] = []
|
|
149
|
+
for item in value:
|
|
150
|
+
if isinstance(item, str):
|
|
151
|
+
out.append(item)
|
|
152
|
+
elif isinstance(item, dict) and item:
|
|
153
|
+
out.append(str(next(iter(item))))
|
|
154
|
+
return out
|
|
155
|
+
return []
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _parse_apm(text: str) -> dict | None:
|
|
159
|
+
try:
|
|
160
|
+
import yaml
|
|
161
|
+
except ImportError:
|
|
162
|
+
return _parse_apm_fallback(text)
|
|
163
|
+
data = yaml.safe_load(text)
|
|
164
|
+
if not isinstance(data, dict):
|
|
165
|
+
return None
|
|
166
|
+
return {
|
|
167
|
+
"name": data.get("name"),
|
|
168
|
+
"version": data.get("version"),
|
|
169
|
+
"deps": _coerce_deps(data.get("dependencies")),
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _parse_apm_fallback(text: str) -> dict | None:
|
|
174
|
+
"""Minimal line parser for apm.yml when PyYAML is unavailable: a top-level
|
|
175
|
+
``name:``/``version:`` plus a simple ``dependencies:`` block (list items or
|
|
176
|
+
a name map)."""
|
|
177
|
+
name = None
|
|
178
|
+
version = None
|
|
179
|
+
deps: list[str] = []
|
|
180
|
+
in_deps = False
|
|
181
|
+
for line in text.splitlines():
|
|
182
|
+
if not in_deps:
|
|
183
|
+
m = re.match(r'^name:\s*["\']?([^"\'\s#]+)', line)
|
|
184
|
+
if m:
|
|
185
|
+
name = m.group(1)
|
|
186
|
+
continue
|
|
187
|
+
# `version` is part of the manifest contract the YAML path already
|
|
188
|
+
# returns; dropping it here made a package node lose its version
|
|
189
|
+
# on every machine without PyYAML installed.
|
|
190
|
+
m = re.match(r'^version:\s*["\']?([^"\'\s#]+)', line)
|
|
191
|
+
if m:
|
|
192
|
+
version = m.group(1)
|
|
193
|
+
continue
|
|
194
|
+
if re.match(r'^dependencies:\s*$', line):
|
|
195
|
+
in_deps = True
|
|
196
|
+
continue
|
|
197
|
+
if in_deps:
|
|
198
|
+
dm = (re.match(r'^\s*-\s*["\']?([^"\'\s#:]+)', line)
|
|
199
|
+
or re.match(r'^\s{2,}([A-Za-z0-9._/@-]+)\s*:', line))
|
|
200
|
+
if dm:
|
|
201
|
+
deps.append(dm.group(1))
|
|
202
|
+
elif re.match(r'^\S', line): # next top-level key ends the block
|
|
203
|
+
in_deps = False
|
|
204
|
+
return {"name": name, "version": version, "deps": deps} if name else None
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _pep508_name(spec: str) -> str:
|
|
208
|
+
"""`requests>=2.0` -> `requests`; `pkg[extra]==1; python_version<'3.9'` -> `pkg`."""
|
|
209
|
+
return re.split(r'[\s<>=!~;\[\(]', spec.strip(), maxsplit=1)[0]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _parse_pyproject(text: str) -> dict | None:
|
|
213
|
+
_toml = _load_toml_module()
|
|
214
|
+
data = _toml.loads(text)
|
|
215
|
+
proj = data.get("project", {}) if isinstance(data.get("project"), dict) else {}
|
|
216
|
+
poetry = (data.get("tool", {}) or {}).get("poetry", {}) if isinstance(data.get("tool"), dict) else {}
|
|
217
|
+
name = proj.get("name") or (poetry.get("name") if isinstance(poetry, dict) else None)
|
|
218
|
+
if not name:
|
|
219
|
+
return None
|
|
220
|
+
deps: list[str] = [_pep508_name(s) for s in (proj.get("dependencies") or []) if isinstance(s, str)]
|
|
221
|
+
if isinstance(poetry, dict):
|
|
222
|
+
for dep in (poetry.get("dependencies") or {}):
|
|
223
|
+
if str(dep).lower() != "python":
|
|
224
|
+
deps.append(str(dep))
|
|
225
|
+
return {"name": name, "version": proj.get("version") or (poetry.get("version") if isinstance(poetry, dict) else None), "deps": deps}
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _parse_cargo(text: str) -> dict | None:
|
|
229
|
+
"""Cargo.toml: name/version from ``[package]``, runtime deps from
|
|
230
|
+
``[dependencies]`` plus every ``[target.<cfg>.dependencies]`` table (mirrors
|
|
231
|
+
``_parse_pyproject``'s runtime-only scope; dev-/build-dependencies excluded)."""
|
|
232
|
+
_toml = _load_toml_module()
|
|
233
|
+
data = _toml.loads(text)
|
|
234
|
+
pkg = data.get("package", {}) if isinstance(data.get("package"), dict) else {}
|
|
235
|
+
name = pkg.get("name")
|
|
236
|
+
# A virtual workspace root (``[workspace]``, no ``[package]``) declares no
|
|
237
|
+
# package of its own — emit nothing rather than a fabricated node. ``name`` is
|
|
238
|
+
# never workspace-inheritable in Cargo, but guard on the type anyway.
|
|
239
|
+
if not isinstance(name, str) or not name:
|
|
240
|
+
return None
|
|
241
|
+
# ``version`` may be workspace-inherited (``version.workspace = true``), which
|
|
242
|
+
# parses to a table; keep only a concrete string version.
|
|
243
|
+
version = pkg.get("version")
|
|
244
|
+
if not isinstance(version, str):
|
|
245
|
+
version = None
|
|
246
|
+
# A dependency value is a bare version string or an inline table; either way
|
|
247
|
+
# _coerce_deps keys it by the dependency NAME (the table/map key).
|
|
248
|
+
deps = _coerce_deps(data.get("dependencies"))
|
|
249
|
+
# Platform-conditional deps live under ``[target.<cfg>.dependencies]``; fold
|
|
250
|
+
# them in so a crate whose deps are entirely cfg-gated still emits its edges.
|
|
251
|
+
targets = data.get("target")
|
|
252
|
+
if isinstance(targets, dict):
|
|
253
|
+
for cfg in targets.values():
|
|
254
|
+
if isinstance(cfg, dict):
|
|
255
|
+
deps += _coerce_deps(cfg.get("dependencies"))
|
|
256
|
+
return {"name": name, "version": version, "deps": deps}
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _parse_gomod(text: str) -> dict | None:
|
|
260
|
+
name = None
|
|
261
|
+
deps: list[str] = []
|
|
262
|
+
in_block = False
|
|
263
|
+
for line in text.splitlines():
|
|
264
|
+
s = line.strip()
|
|
265
|
+
if name is None:
|
|
266
|
+
m = re.match(r'^module\s+(\S+)', s)
|
|
267
|
+
if m:
|
|
268
|
+
name = m.group(1)
|
|
269
|
+
continue
|
|
270
|
+
if re.match(r'^require\s*\(', s):
|
|
271
|
+
in_block = True
|
|
272
|
+
continue
|
|
273
|
+
if in_block:
|
|
274
|
+
if s.startswith(')'):
|
|
275
|
+
in_block = False
|
|
276
|
+
continue
|
|
277
|
+
dm = re.match(r'^(\S+)\s+v\S+', s)
|
|
278
|
+
if dm:
|
|
279
|
+
deps.append(dm.group(1))
|
|
280
|
+
else:
|
|
281
|
+
dm = re.match(r'^require\s+(\S+)\s+v\S+', s)
|
|
282
|
+
if dm:
|
|
283
|
+
deps.append(dm.group(1))
|
|
284
|
+
return {"name": name, "version": None, "deps": deps} if name else None
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _parse_pom(text: str) -> dict | None:
|
|
288
|
+
# Drop the default namespace so findtext/findall don't need the {uri} prefix.
|
|
289
|
+
text = re.sub(r'\sxmlns="[^"]*"', '', text, count=1)
|
|
290
|
+
root = ET.fromstring(text)
|
|
291
|
+
aid = root.findtext("artifactId")
|
|
292
|
+
gid = root.findtext("groupId")
|
|
293
|
+
if not aid:
|
|
294
|
+
return None
|
|
295
|
+
name = f"{gid}:{aid}" if gid else aid
|
|
296
|
+
deps: list[str] = []
|
|
297
|
+
for dep in root.findall(".//dependencies/dependency"):
|
|
298
|
+
da = dep.findtext("artifactId")
|
|
299
|
+
dg = dep.findtext("groupId")
|
|
300
|
+
if da:
|
|
301
|
+
deps.append(f"{dg}:{da}" if dg else da)
|
|
302
|
+
return {"name": name, "version": root.findtext("version"), "deps": deps}
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
_PARSERS = {
|
|
306
|
+
"apm": _parse_apm,
|
|
307
|
+
"python": _parse_pyproject,
|
|
308
|
+
"cargo": _parse_cargo,
|
|
309
|
+
"go": _parse_gomod,
|
|
310
|
+
"maven": _parse_pom,
|
|
311
|
+
}
|
graphify/mcp_ingest.py
ADDED
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
"""mcp_ingest.py — Extract MCP (Model Context Protocol) server configuration files.
|
|
2
|
+
|
|
3
|
+
Reads `.mcp.json` / `claude_desktop_config.json` / `mcp.json` / `mcp_servers.json`
|
|
4
|
+
and turns the `mcpServers` map into Graphify nodes and edges.
|
|
5
|
+
|
|
6
|
+
Symmetry with `serve.py`: Graphify exposes itself AS an MCP server. This module
|
|
7
|
+
indexes MCP servers AS a corpus type, completing the loop — an agent that runs
|
|
8
|
+
graphify with `--mcp` can now query its own configured MCP layer.
|
|
9
|
+
|
|
10
|
+
Entry point:
|
|
11
|
+
extract_mcp_config(path: Path) -> dict[str, list[dict]]
|
|
12
|
+
|
|
13
|
+
Returns `{"nodes": [...], "edges": [...]}` compatible with Graphify's
|
|
14
|
+
extraction-result format. Returns `{"nodes": [...], "edges": [...], "error": "..."}`
|
|
15
|
+
when the file is malformed, too large, or has no `mcpServers` map — the empty
|
|
16
|
+
result keeps it indistinguishable from "no MCP config here" for downstream
|
|
17
|
+
callers.
|
|
18
|
+
|
|
19
|
+
Detected filenames (case-sensitive, matched on basename):
|
|
20
|
+
- .mcp.json (Claude Code project config)
|
|
21
|
+
- claude_desktop_config.json (Claude Desktop)
|
|
22
|
+
- mcp.json (generic / per-tool)
|
|
23
|
+
- mcp_servers.json (alternate naming)
|
|
24
|
+
|
|
25
|
+
Schema emitted:
|
|
26
|
+
Node kinds:
|
|
27
|
+
- file the config file itself (label = filename)
|
|
28
|
+
- mcp_server one per entry under mcpServers
|
|
29
|
+
- mcp_command executable (npx, uvx, node, python, ...) — global ID
|
|
30
|
+
- mcp_package npm / pypi package id parsed from args — global ID
|
|
31
|
+
- env_var env variable NAME only — global ID. VALUES ARE NEVER READ.
|
|
32
|
+
|
|
33
|
+
Edge relations:
|
|
34
|
+
- contains file -> mcp_server
|
|
35
|
+
- references mcp_server -> mcp_command
|
|
36
|
+
- references mcp_server -> mcp_package
|
|
37
|
+
- requires_env mcp_server -> env_var (new relation; distinguishes
|
|
38
|
+
env dependencies from generic refs)
|
|
39
|
+
|
|
40
|
+
Security:
|
|
41
|
+
- Env var VALUES are never read, persisted, labelled, or surfaced. Only env
|
|
42
|
+
var NAMES become nodes. (`env: {"API_KEY": "sk-..."}` -> node "API_KEY" only.)
|
|
43
|
+
- File size capped at 1 MiB (matches extract_json).
|
|
44
|
+
- All labels go through `sanitize_label` (control characters stripped, length
|
|
45
|
+
capped) before emission.
|
|
46
|
+
- Args are NOT persisted as nodes/edges to avoid leaking paths or secrets that
|
|
47
|
+
some servers embed as positional args.
|
|
48
|
+
|
|
49
|
+
Cross-config emergent edges:
|
|
50
|
+
Because `mcp_command`, `mcp_package`, and `env_var` nodes use global IDs (no
|
|
51
|
+
per-file stem prefix), the same package or env var across two MCP configs
|
|
52
|
+
produces shared nodes — naturally surfacing "what configs depend on this
|
|
53
|
+
thing?" via graph traversal. Server nodes ARE stem-scoped so two configs
|
|
54
|
+
declaring different servers under the same key (e.g., both have "filesystem")
|
|
55
|
+
do not collide.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
from __future__ import annotations
|
|
59
|
+
|
|
60
|
+
import json
|
|
61
|
+
import re
|
|
62
|
+
import unicodedata
|
|
63
|
+
from pathlib import Path
|
|
64
|
+
from typing import Any
|
|
65
|
+
|
|
66
|
+
from graphify.ids import make_id as _shared_make_id
|
|
67
|
+
from graphify.security import sanitize_label
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
MCP_CONFIG_FILENAMES: frozenset[str] = frozenset({
|
|
71
|
+
".mcp.json",
|
|
72
|
+
"claude_desktop_config.json",
|
|
73
|
+
"mcp.json",
|
|
74
|
+
"mcp_servers.json",
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
_MAX_BYTES = 1_048_576 # 1 MiB — same cap as extract_json
|
|
78
|
+
_MAX_SERVERS_PER_FILE = 200 # generous; flags pathological configs
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def is_mcp_config_path(path: Path) -> bool:
|
|
82
|
+
"""Return True when ``path`` is a recognised MCP config filename."""
|
|
83
|
+
return path.name in MCP_CONFIG_FILENAMES
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def extract_mcp_config(path: Path) -> dict[str, Any]:
|
|
87
|
+
"""Parse an MCP config file into Graphify nodes and edges.
|
|
88
|
+
|
|
89
|
+
Behaviour matches other extractors in `extract.py`:
|
|
90
|
+
- returns ``{"nodes": [...], "edges": [...]}`` on success
|
|
91
|
+
- returns ``{"nodes": [], "edges": [], "error": "<reason>"}`` on parse
|
|
92
|
+
failure, oversize file, or missing ``mcpServers`` map
|
|
93
|
+
"""
|
|
94
|
+
try:
|
|
95
|
+
with path.open("rb") as fh:
|
|
96
|
+
raw = fh.read(_MAX_BYTES + 1)
|
|
97
|
+
except OSError as exc:
|
|
98
|
+
return {"nodes": [], "edges": [], "error": f"mcp_ingest read error: {exc}"}
|
|
99
|
+
|
|
100
|
+
if len(raw) > _MAX_BYTES:
|
|
101
|
+
return {"nodes": [], "edges": [], "error": "mcp config too large to index"}
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
text = raw.decode("utf-8")
|
|
105
|
+
except UnicodeDecodeError as exc:
|
|
106
|
+
return {"nodes": [], "edges": [], "error": f"mcp_ingest decode error: {exc}"}
|
|
107
|
+
|
|
108
|
+
try:
|
|
109
|
+
doc = json.loads(text)
|
|
110
|
+
except json.JSONDecodeError as exc:
|
|
111
|
+
return {"nodes": [], "edges": [], "error": f"mcp_ingest json error: {exc}"}
|
|
112
|
+
|
|
113
|
+
if not isinstance(doc, dict):
|
|
114
|
+
return {"nodes": [], "edges": [], "error": "mcp_ingest: root is not an object"}
|
|
115
|
+
|
|
116
|
+
servers = doc.get("mcpServers")
|
|
117
|
+
if not isinstance(servers, dict):
|
|
118
|
+
# Some tools nest the map (e.g., {"mcp": {"servers": {...}}}). Try one
|
|
119
|
+
# well-known alternate shape but do not search exhaustively.
|
|
120
|
+
nested = doc.get("mcp")
|
|
121
|
+
if isinstance(nested, dict):
|
|
122
|
+
servers = nested.get("servers")
|
|
123
|
+
if not isinstance(servers, dict):
|
|
124
|
+
return {"nodes": [], "edges": [], "error": "mcp_ingest: no mcpServers map"}
|
|
125
|
+
|
|
126
|
+
str_path = str(path)
|
|
127
|
+
file_nid = _make_id(str_path)
|
|
128
|
+
nodes: list[dict[str, Any]] = []
|
|
129
|
+
edges: list[dict[str, Any]] = []
|
|
130
|
+
seen_node_ids: set[str] = set()
|
|
131
|
+
seen_edge_keys: set[tuple[str, str, str]] = set()
|
|
132
|
+
|
|
133
|
+
_add_node(
|
|
134
|
+
nodes, seen_node_ids,
|
|
135
|
+
nid=file_nid,
|
|
136
|
+
label=path.name,
|
|
137
|
+
kind="mcp_config_file",
|
|
138
|
+
source_file=str_path,
|
|
139
|
+
line=1,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
file_stem = _file_stem(path)
|
|
143
|
+
server_count = 0
|
|
144
|
+
for server_name, spec in servers.items():
|
|
145
|
+
if not isinstance(server_name, str) or not server_name:
|
|
146
|
+
continue
|
|
147
|
+
if not isinstance(spec, dict):
|
|
148
|
+
# Skip non-object server entries silently — the broken entry is
|
|
149
|
+
# the user's, not ours.
|
|
150
|
+
continue
|
|
151
|
+
if server_count >= _MAX_SERVERS_PER_FILE:
|
|
152
|
+
break
|
|
153
|
+
server_count += 1
|
|
154
|
+
_emit_server(
|
|
155
|
+
server_name=server_name,
|
|
156
|
+
spec=spec,
|
|
157
|
+
file_nid=file_nid,
|
|
158
|
+
file_stem=file_stem,
|
|
159
|
+
source_file=str_path,
|
|
160
|
+
nodes=nodes,
|
|
161
|
+
edges=edges,
|
|
162
|
+
seen_node_ids=seen_node_ids,
|
|
163
|
+
seen_edge_keys=seen_edge_keys,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
return {"nodes": nodes, "edges": edges}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _emit_server(
|
|
170
|
+
*,
|
|
171
|
+
server_name: str,
|
|
172
|
+
spec: dict[str, Any],
|
|
173
|
+
file_nid: str,
|
|
174
|
+
file_stem: str,
|
|
175
|
+
source_file: str,
|
|
176
|
+
nodes: list[dict[str, Any]],
|
|
177
|
+
edges: list[dict[str, Any]],
|
|
178
|
+
seen_node_ids: set[str],
|
|
179
|
+
seen_edge_keys: set[tuple[str, str, str]],
|
|
180
|
+
) -> None:
|
|
181
|
+
"""Emit nodes/edges for one entry under ``mcpServers``."""
|
|
182
|
+
server_nid = _make_id(file_stem, "mcp_server", server_name)
|
|
183
|
+
_add_node(
|
|
184
|
+
nodes, seen_node_ids,
|
|
185
|
+
nid=server_nid,
|
|
186
|
+
label=server_name,
|
|
187
|
+
kind="mcp_server",
|
|
188
|
+
source_file=source_file,
|
|
189
|
+
line=1, # JSON doesn't expose line numbers without a parser pass
|
|
190
|
+
)
|
|
191
|
+
_add_edge(
|
|
192
|
+
edges, seen_edge_keys,
|
|
193
|
+
source=file_nid,
|
|
194
|
+
target=server_nid,
|
|
195
|
+
relation="contains",
|
|
196
|
+
source_file=source_file,
|
|
197
|
+
line=1,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
command = spec.get("command")
|
|
201
|
+
if isinstance(command, str) and command.strip():
|
|
202
|
+
cmd_label = command.strip()
|
|
203
|
+
cmd_nid = _make_id("mcp_command", cmd_label)
|
|
204
|
+
_add_node(
|
|
205
|
+
nodes, seen_node_ids,
|
|
206
|
+
nid=cmd_nid,
|
|
207
|
+
label=cmd_label,
|
|
208
|
+
kind="mcp_command",
|
|
209
|
+
source_file=source_file,
|
|
210
|
+
line=1,
|
|
211
|
+
)
|
|
212
|
+
_add_edge(
|
|
213
|
+
edges, seen_edge_keys,
|
|
214
|
+
source=server_nid,
|
|
215
|
+
target=cmd_nid,
|
|
216
|
+
relation="references",
|
|
217
|
+
source_file=source_file,
|
|
218
|
+
line=1,
|
|
219
|
+
context="command",
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
args = spec.get("args")
|
|
223
|
+
if isinstance(args, list):
|
|
224
|
+
package = _detect_package_from_args(args)
|
|
225
|
+
if package:
|
|
226
|
+
pkg_nid = _make_id("mcp_package", package)
|
|
227
|
+
_add_node(
|
|
228
|
+
nodes, seen_node_ids,
|
|
229
|
+
nid=pkg_nid,
|
|
230
|
+
label=package,
|
|
231
|
+
kind="mcp_package",
|
|
232
|
+
source_file=source_file,
|
|
233
|
+
line=1,
|
|
234
|
+
)
|
|
235
|
+
_add_edge(
|
|
236
|
+
edges, seen_edge_keys,
|
|
237
|
+
source=server_nid,
|
|
238
|
+
target=pkg_nid,
|
|
239
|
+
relation="references",
|
|
240
|
+
source_file=source_file,
|
|
241
|
+
line=1,
|
|
242
|
+
context="package",
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
env = spec.get("env")
|
|
246
|
+
if isinstance(env, dict):
|
|
247
|
+
# ONLY KEYS. Values may contain secrets and are never read here.
|
|
248
|
+
for env_name in env.keys():
|
|
249
|
+
if not isinstance(env_name, str) or not env_name:
|
|
250
|
+
continue
|
|
251
|
+
env_nid = _make_id("env_var", env_name)
|
|
252
|
+
_add_node(
|
|
253
|
+
nodes, seen_node_ids,
|
|
254
|
+
nid=env_nid,
|
|
255
|
+
label=env_name,
|
|
256
|
+
kind="env_var",
|
|
257
|
+
source_file=source_file,
|
|
258
|
+
line=1,
|
|
259
|
+
)
|
|
260
|
+
_add_edge(
|
|
261
|
+
edges, seen_edge_keys,
|
|
262
|
+
source=server_nid,
|
|
263
|
+
target=env_nid,
|
|
264
|
+
relation="requires_env",
|
|
265
|
+
source_file=source_file,
|
|
266
|
+
line=1,
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# ── Package detection from args ───────────────────────────────────────────────
|
|
271
|
+
|
|
272
|
+
# Patterns observed in real MCP server configs:
|
|
273
|
+
# ["-y", "@modelcontextprotocol/server-filesystem", "/data"] (npx)
|
|
274
|
+
# ["-y", "@org/pkg@1.2.3"]
|
|
275
|
+
# ["mcp-server-fetch"] (uvx / python)
|
|
276
|
+
# ["mcp-server-time", "--local-timezone=UTC"]
|
|
277
|
+
# ["@scoped/some-mcp"] (pnpx)
|
|
278
|
+
# ["mcp-server-fetch"] (uvx direct)
|
|
279
|
+
_NPM_PKG_RE = re.compile(r"^@[a-z0-9][a-z0-9._-]*/[a-z0-9][a-z0-9._-]*(?:@[\w.\-+]+)?$")
|
|
280
|
+
_PY_MCP_PKG_RE = re.compile(r"^[a-z0-9][a-z0-9._-]*-mcp(?:-[a-z0-9._-]+)?$|^mcp-[a-z0-9][a-z0-9._-]*$")
|
|
281
|
+
_ARG_FLAG_RE = re.compile(r"^-{1,2}\w")
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _detect_package_from_args(args: list[Any]) -> str | None:
|
|
285
|
+
"""Return the first arg that looks like an npm or pypi package id, else None.
|
|
286
|
+
|
|
287
|
+
Skips short flags (-y, --yes) and option arguments (--local-timezone=UTC).
|
|
288
|
+
"""
|
|
289
|
+
for raw in args:
|
|
290
|
+
if not isinstance(raw, str):
|
|
291
|
+
continue
|
|
292
|
+
arg = raw.strip()
|
|
293
|
+
if not arg or _ARG_FLAG_RE.match(arg):
|
|
294
|
+
continue
|
|
295
|
+
if _NPM_PKG_RE.match(arg):
|
|
296
|
+
return _strip_version(arg)
|
|
297
|
+
if _PY_MCP_PKG_RE.match(arg):
|
|
298
|
+
return arg
|
|
299
|
+
return None
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _strip_version(pkg: str) -> str:
|
|
303
|
+
"""Drop the ``@version`` suffix from an npm package id, preserving the scope.
|
|
304
|
+
|
|
305
|
+
Scoped: ``@scope/name`` or ``@scope/name@1.2.3`` — there are at most two
|
|
306
|
+
``@`` chars; the second is the version separator.
|
|
307
|
+
Unscoped: ``name`` or ``name@1.2.3``.
|
|
308
|
+
"""
|
|
309
|
+
if pkg.startswith("@"):
|
|
310
|
+
version_at = pkg.find("@", 1)
|
|
311
|
+
return pkg if version_at == -1 else pkg[:version_at]
|
|
312
|
+
version_at = pkg.find("@")
|
|
313
|
+
return pkg if version_at == -1 else pkg[:version_at]
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
# ── Node / edge construction (Graphify schema) ────────────────────────────────
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _add_node(
|
|
320
|
+
nodes: list[dict[str, Any]],
|
|
321
|
+
seen: set[str],
|
|
322
|
+
*,
|
|
323
|
+
nid: str,
|
|
324
|
+
label: str,
|
|
325
|
+
kind: str,
|
|
326
|
+
source_file: str,
|
|
327
|
+
line: int,
|
|
328
|
+
) -> None:
|
|
329
|
+
"""Append a node if not already present. ``kind`` is metadata, not file_type."""
|
|
330
|
+
if not nid or nid in seen:
|
|
331
|
+
return
|
|
332
|
+
seen.add(nid)
|
|
333
|
+
nodes.append({
|
|
334
|
+
"id": nid,
|
|
335
|
+
"label": sanitize_label(label),
|
|
336
|
+
"file_type": "code",
|
|
337
|
+
"source_file": source_file,
|
|
338
|
+
"source_location": f"L{line}",
|
|
339
|
+
"metadata": {"mcp_kind": kind},
|
|
340
|
+
})
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _add_edge(
|
|
344
|
+
edges: list[dict[str, Any]],
|
|
345
|
+
seen: set[tuple[str, str, str]],
|
|
346
|
+
*,
|
|
347
|
+
source: str,
|
|
348
|
+
target: str,
|
|
349
|
+
relation: str,
|
|
350
|
+
source_file: str,
|
|
351
|
+
line: int,
|
|
352
|
+
context: str | None = None,
|
|
353
|
+
) -> None:
|
|
354
|
+
"""Append an edge if (source, target, relation) is not already present."""
|
|
355
|
+
if not source or not target or source == target:
|
|
356
|
+
return
|
|
357
|
+
key = (source, target, relation)
|
|
358
|
+
if key in seen:
|
|
359
|
+
return
|
|
360
|
+
seen.add(key)
|
|
361
|
+
edge: dict[str, Any] = {
|
|
362
|
+
"source": source,
|
|
363
|
+
"target": target,
|
|
364
|
+
"relation": relation,
|
|
365
|
+
"confidence": "EXTRACTED",
|
|
366
|
+
"confidence_score": 1.0,
|
|
367
|
+
"source_file": source_file,
|
|
368
|
+
"source_location": f"L{line}",
|
|
369
|
+
"weight": 1.0,
|
|
370
|
+
}
|
|
371
|
+
if context:
|
|
372
|
+
edge["context"] = context
|
|
373
|
+
edges.append(edge)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
# ── ID helpers (kept local; mirror extract.py shape) ──────────────────────────
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _make_id(*parts: str) -> str:
|
|
380
|
+
"""Build a stable node ID via the single shared recipe (#1378)."""
|
|
381
|
+
return _shared_make_id(*parts)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
# Canonical recipe imported directly (no import cycle: extractors.base imports
|
|
385
|
+
# only graphify.ids), so this can no longer drift from extract._file_stem.
|
|
386
|
+
from graphify.extractors.base import _file_stem # noqa: E402
|