mcp-guardbench 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_guardbench-0.1.0/.dockerignore +25 -0
- mcp_guardbench-0.1.0/.env.example +41 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/bug_report.yml +38 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/config.yml +5 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/detection_result.yml +50 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/feature_request.yml +27 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/inspect_result.yml +41 -0
- mcp_guardbench-0.1.0/.github/ISSUE_TEMPLATE/test_case_proposal.yml +56 -0
- mcp_guardbench-0.1.0/.github/PULL_REQUEST_TEMPLATE.md +17 -0
- mcp_guardbench-0.1.0/.github/workflows/ci.yml +92 -0
- mcp_guardbench-0.1.0/.github/workflows/publish.yml +50 -0
- mcp_guardbench-0.1.0/.gitignore +38 -0
- mcp_guardbench-0.1.0/CHANGELOG.md +26 -0
- mcp_guardbench-0.1.0/CODE_OF_CONDUCT.md +55 -0
- mcp_guardbench-0.1.0/CONTRIBUTING.md +127 -0
- mcp_guardbench-0.1.0/Dockerfile +39 -0
- mcp_guardbench-0.1.0/LICENSE +21 -0
- mcp_guardbench-0.1.0/Makefile +172 -0
- mcp_guardbench-0.1.0/PKG-INFO +636 -0
- mcp_guardbench-0.1.0/README.md +586 -0
- mcp_guardbench-0.1.0/SECURITY.md +82 -0
- mcp_guardbench-0.1.0/alembic.ini +43 -0
- mcp_guardbench-0.1.0/docker-compose.yml +162 -0
- mcp_guardbench-0.1.0/docs/adapter-development.md +209 -0
- mcp_guardbench-0.1.0/docs/architecture.md +153 -0
- mcp_guardbench-0.1.0/docs/limitations.md +158 -0
- mcp_guardbench-0.1.0/docs/metrics.md +88 -0
- mcp_guardbench-0.1.0/docs/test-case-format.md +146 -0
- mcp_guardbench-0.1.0/docs/threat-model.md +89 -0
- mcp_guardbench-0.1.0/migrations/env.py +62 -0
- mcp_guardbench-0.1.0/migrations/script.py.mako +27 -0
- mcp_guardbench-0.1.0/migrations/versions/0001_initial_schema.py +320 -0
- mcp_guardbench-0.1.0/pyproject.toml +128 -0
- mcp_guardbench-0.1.0/reports/.gitkeep +0 -0
- mcp_guardbench-0.1.0/reports/sample/README.md +28 -0
- mcp_guardbench-0.1.0/reports/sample/report.json +6614 -0
- mcp_guardbench-0.1.0/reports/sample/report.md +498 -0
- mcp_guardbench-0.1.0/reports/sample/summary.csv +49 -0
- mcp_guardbench-0.1.0/scripts/bootstrap_env.py +68 -0
- mcp_guardbench-0.1.0/scripts/generate_report.py +27 -0
- mcp_guardbench-0.1.0/scripts/run_demo.py +26 -0
- mcp_guardbench-0.1.0/scripts/seed_demo.py +11 -0
- mcp_guardbench-0.1.0/scripts/test_postgres.py +41 -0
- mcp_guardbench-0.1.0/src/guardbench/__init__.py +7 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/capabilities.py +199 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/drift_detector.py +333 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/fingerprinting.py +131 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/metadata_analyzer.py +766 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/risk_rules.py +295 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/schema_analyzer.py +285 -0
- mcp_guardbench-0.1.0/src/guardbench/analysis/severity.py +65 -0
- mcp_guardbench-0.1.0/src/guardbench/api/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/api/dependencies.py +64 -0
- mcp_guardbench-0.1.0/src/guardbench/api/main.py +124 -0
- mcp_guardbench-0.1.0/src/guardbench/api/middleware.py +136 -0
- mcp_guardbench-0.1.0/src/guardbench/api/pagination.py +33 -0
- mcp_guardbench-0.1.0/src/guardbench/api/redaction.py +58 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/dashboard.py +53 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/findings.py +91 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/health.py +40 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/metrics.py +28 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/projects.py +47 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/runs.py +161 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/servers.py +122 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/test_cases.py +76 -0
- mcp_guardbench-0.1.0/src/guardbench/api/routes/tools.py +52 -0
- mcp_guardbench-0.1.0/src/guardbench/asyncio_utils.py +26 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/adapters.py +267 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/cisco_scanner.py +309 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/guards.py +458 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/metrics.py +305 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/orchestrator.py +291 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/report.py +607 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/result_normalizer.py +197 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/runner.py +429 -0
- mcp_guardbench-0.1.0/src/guardbench/benchmark/test_case_loader.py +99 -0
- mcp_guardbench-0.1.0/src/guardbench/cli/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/cli/main.py +796 -0
- mcp_guardbench-0.1.0/src/guardbench/config.py +66 -0
- mcp_guardbench-0.1.0/src/guardbench/dashboard/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/dashboard/api_client.py +208 -0
- mcp_guardbench-0.1.0/src/guardbench/dashboard/app.py +498 -0
- mcp_guardbench-0.1.0/src/guardbench/db/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/db/base.py +58 -0
- mcp_guardbench-0.1.0/src/guardbench/db/migrate.py +39 -0
- mcp_guardbench-0.1.0/src/guardbench/db/models.py +209 -0
- mcp_guardbench-0.1.0/src/guardbench/db/repositories.py +318 -0
- mcp_guardbench-0.1.0/src/guardbench/db/session.py +84 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/clock.py +34 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/enums.py +224 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/errors.py +53 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/markers.py +120 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/schemas.py +529 -0
- mcp_guardbench-0.1.0/src/guardbench/domain/testcase.py +176 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/__init__.py +8 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/client.py +196 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/configs.py +371 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/pins.py +96 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/render.py +152 -0
- mcp_guardbench-0.1.0/src/guardbench/inspection/service.py +182 -0
- mcp_guardbench-0.1.0/src/guardbench/logging_config.py +86 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/base.py +217 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/client_runner.py +83 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/common.py +49 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/fixtures.py +87 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/server_runner.py +48 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/README.md +51 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/benign_guidance_server.py +83 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/clean_server.py +160 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/drift_server.py +105 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/encoded_flow_server.py +74 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/excessive_permission_server.py +137 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/multilingual_poisoning_server.py +65 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/oversized_response_server.py +40 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/poisoned_description_server.py +68 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/poisoned_schema_server.py +88 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/response_injection_server.py +58 -0
- mcp_guardbench-0.1.0/src/guardbench/mcp_lab/servers/secret_flow_server.py +69 -0
- mcp_guardbench-0.1.0/src/guardbench/policy/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/policy/approval.py +162 -0
- mcp_guardbench-0.1.0/src/guardbench/policy/default_policy.yaml +60 -0
- mcp_guardbench-0.1.0/src/guardbench/policy/engine.py +297 -0
- mcp_guardbench-0.1.0/src/guardbench/policy/models.py +174 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/data_flow.py +211 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/event_bus.py +49 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/recorder.py +135 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/redaction.py +130 -0
- mcp_guardbench-0.1.0/src/guardbench/runtime/trace_context.py +102 -0
- mcp_guardbench-0.1.0/src/guardbench/safe_paths.py +38 -0
- mcp_guardbench-0.1.0/src/guardbench/services/__init__.py +0 -0
- mcp_guardbench-0.1.0/src/guardbench/services/demo.py +60 -0
- mcp_guardbench-0.1.0/src/guardbench/services/runs.py +235 -0
- mcp_guardbench-0.1.0/src/guardbench/services/scan.py +168 -0
- mcp_guardbench-0.1.0/src/guardbench/services/test_cases.py +66 -0
- mcp_guardbench-0.1.0/src/guardbench/services/trace.py +120 -0
- mcp_guardbench-0.1.0/test_cases/BN-001-clean-server-metadata.yaml +22 -0
- mcp_guardbench-0.1.0/test_cases/BN-002-clean-server-read-calls.yaml +30 -0
- mcp_guardbench-0.1.0/test_cases/BN-003-benign-cross-tool-guidance.yaml +30 -0
- mcp_guardbench-0.1.0/test_cases/DF-001-sensitive-marker-flow.yaml +42 -0
- mcp_guardbench-0.1.0/test_cases/DF-002-encoded-secret-flow.yaml +46 -0
- mcp_guardbench-0.1.0/test_cases/PA-001-excessive-permission.yaml +49 -0
- mcp_guardbench-0.1.0/test_cases/RD-001-tool-definition-drift.yaml +38 -0
- mcp_guardbench-0.1.0/test_cases/README.md +59 -0
- mcp_guardbench-0.1.0/test_cases/RI-001-response-injection.yaml +40 -0
- mcp_guardbench-0.1.0/test_cases/RS-001-oversized-response.yaml +30 -0
- mcp_guardbench-0.1.0/test_cases/TP-001-tool-description-injection.yaml +37 -0
- mcp_guardbench-0.1.0/test_cases/TP-002-schema-injection.yaml +37 -0
- mcp_guardbench-0.1.0/test_cases/TP-003-multilingual-tool-poisoning.yaml +42 -0
- mcp_guardbench-0.1.0/tests/conftest.py +122 -0
- mcp_guardbench-0.1.0/tests/integration/test_api.py +555 -0
- mcp_guardbench-0.1.0/tests/integration/test_benchmark.py +598 -0
- mcp_guardbench-0.1.0/tests/integration/test_cisco_scanner_adapter.py +229 -0
- mcp_guardbench-0.1.0/tests/integration/test_cli.py +512 -0
- mcp_guardbench-0.1.0/tests/integration/test_dashboard.py +328 -0
- mcp_guardbench-0.1.0/tests/integration/test_fixtures_mcp.py +139 -0
- mcp_guardbench-0.1.0/tests/integration/test_health_and_auth.py +72 -0
- mcp_guardbench-0.1.0/tests/integration/test_inspect_cli.py +164 -0
- mcp_guardbench-0.1.0/tests/integration/test_lab_fixtures.py +138 -0
- mcp_guardbench-0.1.0/tests/integration/test_migrations.py +81 -0
- mcp_guardbench-0.1.0/tests/integration/test_report.py +246 -0
- mcp_guardbench-0.1.0/tests/integration/test_scan_service.py +297 -0
- mcp_guardbench-0.1.0/tests/security/test_api_security.py +326 -0
- mcp_guardbench-0.1.0/tests/security/test_benchmark_safety.py +238 -0
- mcp_guardbench-0.1.0/tests/security/test_docker_config.py +201 -0
- mcp_guardbench-0.1.0/tests/security/test_fixture_isolation.py +185 -0
- mcp_guardbench-0.1.0/tests/security/test_inspection_readonly.py +71 -0
- mcp_guardbench-0.1.0/tests/security/test_nul_bytes.py +154 -0
- mcp_guardbench-0.1.0/tests/unit/test_bootstrap_env.py +100 -0
- mcp_guardbench-0.1.0/tests/unit/test_config_and_db.py +134 -0
- mcp_guardbench-0.1.0/tests/unit/test_dashboard_client.py +290 -0
- mcp_guardbench-0.1.0/tests/unit/test_domain_schemas.py +142 -0
- mcp_guardbench-0.1.0/tests/unit/test_drift_detector.py +277 -0
- mcp_guardbench-0.1.0/tests/unit/test_fingerprinting.py +185 -0
- mcp_guardbench-0.1.0/tests/unit/test_inspection_configs.py +200 -0
- mcp_guardbench-0.1.0/tests/unit/test_inspection_report.py +172 -0
- mcp_guardbench-0.1.0/tests/unit/test_metadata_analyzer.py +474 -0
- mcp_guardbench-0.1.0/tests/unit/test_metrics.py +312 -0
- mcp_guardbench-0.1.0/tests/unit/test_policy_engine.py +462 -0
- mcp_guardbench-0.1.0/tests/unit/test_recorder_and_flow.py +239 -0
- mcp_guardbench-0.1.0/tests/unit/test_redaction_and_logging.py +230 -0
- mcp_guardbench-0.1.0/tests/unit/test_schema_severity_capabilities.py +270 -0
- mcp_guardbench-0.1.0/tests/unit/test_test_case_loader.py +323 -0
- mcp_guardbench-0.1.0/tests/unit/test_trace_service.py +97 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
.git
|
|
2
|
+
.venv
|
|
3
|
+
.venv-*
|
|
4
|
+
venv
|
|
5
|
+
__pycache__
|
|
6
|
+
*.py[cod]
|
|
7
|
+
.mypy_cache
|
|
8
|
+
.ruff_cache
|
|
9
|
+
.pytest_cache
|
|
10
|
+
.coverage
|
|
11
|
+
htmlcov
|
|
12
|
+
data
|
|
13
|
+
*.db
|
|
14
|
+
*.sqlite
|
|
15
|
+
*.sqlite3
|
|
16
|
+
.env
|
|
17
|
+
.env.*
|
|
18
|
+
!.env.example
|
|
19
|
+
reports/*
|
|
20
|
+
!reports/.gitkeep
|
|
21
|
+
tests
|
|
22
|
+
docs
|
|
23
|
+
.idea
|
|
24
|
+
.vscode
|
|
25
|
+
.DS_Store
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# MCP-GuardBench configuration. Copy to `.env` and adjust.
|
|
2
|
+
# Every value below is a LOCAL LAB default. Nothing here is a real credential.
|
|
3
|
+
# The only "secrets" in this project are the synthetic markers
|
|
4
|
+
# (TEST_SECRET_123, TEST_PRIVATE_RECORD, SIMULATED_EXTERNAL_DESTINATION).
|
|
5
|
+
|
|
6
|
+
# --- Runtime mode -----------------------------------------------------------
|
|
7
|
+
# true = simplified local setup: API-key check disabled, tables auto-created.
|
|
8
|
+
# false = API requires GUARDBENCH_API_KEY on every non-health request.
|
|
9
|
+
GUARDBENCH_DEV_MODE=true
|
|
10
|
+
GUARDBENCH_LOG_LEVEL=INFO
|
|
11
|
+
GUARDBENCH_LOG_JSON=true
|
|
12
|
+
|
|
13
|
+
# --- Database ---------------------------------------------------------------
|
|
14
|
+
# Local default is SQLite. docker-compose overrides this with PostgreSQL.
|
|
15
|
+
GUARDBENCH_DATABASE_URL=sqlite:///./data/guardbench.db
|
|
16
|
+
|
|
17
|
+
# PostgreSQL container credentials (docker-compose only). Change for anything
|
|
18
|
+
# beyond a throwaway local lab.
|
|
19
|
+
POSTGRES_DB=guardbench
|
|
20
|
+
POSTGRES_USER=guardbench
|
|
21
|
+
POSTGRES_PASSWORD=change-me-local-lab-only
|
|
22
|
+
|
|
23
|
+
# --- API --------------------------------------------------------------------
|
|
24
|
+
GUARDBENCH_API_HOST=127.0.0.1
|
|
25
|
+
GUARDBENCH_API_PORT=8000
|
|
26
|
+
# Required when GUARDBENCH_DEV_MODE=false. Generate one, e.g.:
|
|
27
|
+
# python -c "import secrets; print(secrets.token_urlsafe(32))"
|
|
28
|
+
GUARDBENCH_API_KEY=
|
|
29
|
+
|
|
30
|
+
# --- Dashboard --------------------------------------------------------------
|
|
31
|
+
GUARDBENCH_API_URL=http://127.0.0.1:8000
|
|
32
|
+
GUARDBENCH_DASHBOARD_PORT=8501
|
|
33
|
+
|
|
34
|
+
# --- Lab safety limits ------------------------------------------------------
|
|
35
|
+
# Allowlisted directories. Test cases and fixtures outside these are rejected.
|
|
36
|
+
GUARDBENCH_TEST_CASES_DIR=test_cases
|
|
37
|
+
GUARDBENCH_REPORTS_DIR=reports
|
|
38
|
+
# Bounded model-context size for tool responses (bytes).
|
|
39
|
+
GUARDBENCH_MAX_RESPONSE_BYTES=4096
|
|
40
|
+
# Maximum tool calls allowed per trace (runaway-loop guard).
|
|
41
|
+
GUARDBENCH_MAX_CALLS_PER_TRACE=10
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
name: Bug report
|
|
2
|
+
description: Something in the lab itself is broken (not a missed/false detection; see the other template).
|
|
3
|
+
labels: [bug]
|
|
4
|
+
body:
|
|
5
|
+
- type: markdown
|
|
6
|
+
attributes:
|
|
7
|
+
value: |
|
|
8
|
+
Before filing: this is a **local security lab**. If a reference control missed an attack or
|
|
9
|
+
flagged a benign case, that's expected and tracked in [docs/limitations.md](../../docs/limitations.md),
|
|
10
|
+
not a bug; use the "Detection result" template instead.
|
|
11
|
+
- type: textarea
|
|
12
|
+
id: what-happened
|
|
13
|
+
attributes:
|
|
14
|
+
label: What happened
|
|
15
|
+
description: What you did, what you expected, what happened instead.
|
|
16
|
+
validations:
|
|
17
|
+
required: true
|
|
18
|
+
- type: textarea
|
|
19
|
+
id: repro
|
|
20
|
+
attributes:
|
|
21
|
+
label: Steps to reproduce
|
|
22
|
+
description: Exact commands (`guardbench ...`, `make ...`). A failing test is even better.
|
|
23
|
+
render: shell
|
|
24
|
+
validations:
|
|
25
|
+
required: true
|
|
26
|
+
- type: textarea
|
|
27
|
+
id: logs
|
|
28
|
+
attributes:
|
|
29
|
+
label: Relevant output
|
|
30
|
+
description: Paste error output here. Please double-check it doesn't contain a real secret or API key before posting. Synthetic markers like `TEST_SECRET_123` are fine.
|
|
31
|
+
render: shell
|
|
32
|
+
- type: input
|
|
33
|
+
id: version
|
|
34
|
+
attributes:
|
|
35
|
+
label: Version
|
|
36
|
+
description: Output of `guardbench --version`, plus your OS and Python version.
|
|
37
|
+
validations:
|
|
38
|
+
required: true
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
name: Detection result / false positive / false negative
|
|
2
|
+
description: A reference control (or a corpus test case) behaved in a way you think is wrong.
|
|
3
|
+
labels: [detection]
|
|
4
|
+
body:
|
|
5
|
+
- type: markdown
|
|
6
|
+
attributes:
|
|
7
|
+
value: |
|
|
8
|
+
This is the most useful kind of issue for this project. Please include enough for someone to
|
|
9
|
+
reproduce your exact run. See [docs/metrics.md](../../docs/metrics.md) for how detection and
|
|
10
|
+
prevention are defined and verified.
|
|
11
|
+
- type: dropdown
|
|
12
|
+
id: kind
|
|
13
|
+
attributes:
|
|
14
|
+
label: What kind of result?
|
|
15
|
+
options:
|
|
16
|
+
- False negative (an attack case was not detected/blocked when it should have been)
|
|
17
|
+
- False positive (a benign case triggered a detection/block/approval)
|
|
18
|
+
- Something else looks wrong in a report or the dashboard
|
|
19
|
+
validations:
|
|
20
|
+
required: true
|
|
21
|
+
- type: input
|
|
22
|
+
id: case
|
|
23
|
+
attributes:
|
|
24
|
+
label: Test case and adapter
|
|
25
|
+
description: e.g. `TP-001` under `reference-static`
|
|
26
|
+
validations:
|
|
27
|
+
required: true
|
|
28
|
+
- type: textarea
|
|
29
|
+
id: command
|
|
30
|
+
attributes:
|
|
31
|
+
label: Command used
|
|
32
|
+
render: shell
|
|
33
|
+
description: e.g. `guardbench benchmark run --adapter reference-static --case-id TP-001`
|
|
34
|
+
validations:
|
|
35
|
+
required: true
|
|
36
|
+
- type: textarea
|
|
37
|
+
id: report
|
|
38
|
+
attributes:
|
|
39
|
+
label: Relevant report excerpt or trace
|
|
40
|
+
description: Output of `guardbench trace --latest --adapter ... --case ...`, or the relevant report rows.
|
|
41
|
+
render: shell
|
|
42
|
+
validations:
|
|
43
|
+
required: true
|
|
44
|
+
- type: textarea
|
|
45
|
+
id: why
|
|
46
|
+
attributes:
|
|
47
|
+
label: Why you think this is wrong
|
|
48
|
+
description: What you expected the control (or the case's `expected:` block) to say, and why.
|
|
49
|
+
validations:
|
|
50
|
+
required: true
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
name: Feature request
|
|
2
|
+
description: A new fixture, adapter, metric, or capability you'd like to see.
|
|
3
|
+
labels: [enhancement]
|
|
4
|
+
body:
|
|
5
|
+
- type: markdown
|
|
6
|
+
attributes:
|
|
7
|
+
value: |
|
|
8
|
+
Read [SECURITY.md](../../SECURITY.md) first if this involves anything that touches real
|
|
9
|
+
servers, real credentials, or real network access: this project only ever attacks its own
|
|
10
|
+
local, synthetic fixtures, and any request that would weaken that boundary will be declined.
|
|
11
|
+
- type: textarea
|
|
12
|
+
id: problem
|
|
13
|
+
attributes:
|
|
14
|
+
label: What problem does this solve?
|
|
15
|
+
validations:
|
|
16
|
+
required: true
|
|
17
|
+
- type: textarea
|
|
18
|
+
id: proposal
|
|
19
|
+
attributes:
|
|
20
|
+
label: What would you like to see?
|
|
21
|
+
description: A new fixture (docs/limitations.md and mcp_lab/servers/README.md describe the constraints), a new adapter (see docs/adapter-development.md), a new metric, something else?
|
|
22
|
+
validations:
|
|
23
|
+
required: true
|
|
24
|
+
- type: textarea
|
|
25
|
+
id: alternatives
|
|
26
|
+
attributes:
|
|
27
|
+
label: Alternatives you've considered
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
name: guardbench inspect result on a real server
|
|
2
|
+
description: A false alarm, a miss, or a server that could not be inspected.
|
|
3
|
+
labels: [inspect]
|
|
4
|
+
body:
|
|
5
|
+
- type: markdown
|
|
6
|
+
attributes:
|
|
7
|
+
value: |
|
|
8
|
+
Reports from real servers are how the analyzer's rules get better. **Remove anything private
|
|
9
|
+
before posting:** tokens, internal URLs, file paths, and the `env`/`headers` of your MCP config.
|
|
10
|
+
`guardbench inspect --json` never prints configured env values or headers, but check anyway.
|
|
11
|
+
- type: dropdown
|
|
12
|
+
id: kind
|
|
13
|
+
attributes:
|
|
14
|
+
label: What happened?
|
|
15
|
+
options:
|
|
16
|
+
- False alarm (a finding on a server you believe is fine)
|
|
17
|
+
- Missed something (a hostile description or instruction that was not flagged)
|
|
18
|
+
- The server could not be inspected (error, timeout)
|
|
19
|
+
validations:
|
|
20
|
+
required: true
|
|
21
|
+
- type: input
|
|
22
|
+
id: server
|
|
23
|
+
attributes:
|
|
24
|
+
label: Server
|
|
25
|
+
description: Public package name and version, e.g. `@modelcontextprotocol/server-filesystem 2026.1.0`.
|
|
26
|
+
validations:
|
|
27
|
+
required: true
|
|
28
|
+
- type: textarea
|
|
29
|
+
id: output
|
|
30
|
+
attributes:
|
|
31
|
+
label: Relevant output
|
|
32
|
+
description: The finding lines (rule id, tool, evidence) or the error message.
|
|
33
|
+
render: shell
|
|
34
|
+
validations:
|
|
35
|
+
required: true
|
|
36
|
+
- type: textarea
|
|
37
|
+
id: why
|
|
38
|
+
attributes:
|
|
39
|
+
label: Why you think this is wrong
|
|
40
|
+
validations:
|
|
41
|
+
required: true
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
name: Propose a test case (new attack or benign pattern)
|
|
2
|
+
description: An MCP attack, or a benign real-world pattern, that the benchmark should cover.
|
|
3
|
+
labels: [test-case, good first issue]
|
|
4
|
+
body:
|
|
5
|
+
- type: markdown
|
|
6
|
+
attributes:
|
|
7
|
+
value: |
|
|
8
|
+
This is the contribution that makes the benchmark most meaningful. Cases written by people who
|
|
9
|
+
did **not** write the reference controls are the only way to learn how they really do: see the
|
|
10
|
+
"hard cases" in [test_cases/README.md](../../test_cases/README.md#-hard-cases).
|
|
11
|
+
|
|
12
|
+
Keep it inside the lab's safety rules ([SECURITY.md](../../SECURITY.md)): describe the attack
|
|
13
|
+
against a *synthetic* fixture, use only the synthetic markers (`TEST_SECRET_123`,
|
|
14
|
+
`TEST_PRIVATE_RECORD`, `SIMULATED_EXTERNAL_DESTINATION`), and never include a working exploit
|
|
15
|
+
against a real server.
|
|
16
|
+
- type: dropdown
|
|
17
|
+
id: kind
|
|
18
|
+
attributes:
|
|
19
|
+
label: What kind of case?
|
|
20
|
+
options:
|
|
21
|
+
- Attack the current controls probably miss
|
|
22
|
+
- Attack in a category not covered yet (resources, prompts, sampling, auth, multi-server, ...)
|
|
23
|
+
- Benign real-world pattern that might cause false alarms
|
|
24
|
+
validations:
|
|
25
|
+
required: true
|
|
26
|
+
- type: textarea
|
|
27
|
+
id: idea
|
|
28
|
+
attributes:
|
|
29
|
+
label: The attack (or pattern) in one paragraph
|
|
30
|
+
description: What does the server advertise or return, and what would a gullible agent do?
|
|
31
|
+
validations:
|
|
32
|
+
required: true
|
|
33
|
+
- type: textarea
|
|
34
|
+
id: example
|
|
35
|
+
attributes:
|
|
36
|
+
label: Example tool metadata or response (synthetic)
|
|
37
|
+
description: The hostile description, schema, or response text. Mark hostile text with GUARDBENCH-SYNTHETIC-INJECTION.
|
|
38
|
+
render: json
|
|
39
|
+
- type: textarea
|
|
40
|
+
id: expected
|
|
41
|
+
attributes:
|
|
42
|
+
label: Expected result
|
|
43
|
+
description: Which category should a good control detect it as, and should it be blocked?
|
|
44
|
+
validations:
|
|
45
|
+
required: true
|
|
46
|
+
- type: textarea
|
|
47
|
+
id: source
|
|
48
|
+
attributes:
|
|
49
|
+
label: Where this comes from (optional)
|
|
50
|
+
description: A paper, advisory, blog post, or real server this pattern was seen in.
|
|
51
|
+
- type: checkboxes
|
|
52
|
+
id: pr
|
|
53
|
+
attributes:
|
|
54
|
+
label: Contribution
|
|
55
|
+
options:
|
|
56
|
+
- label: I'd like to open a pull request with the fixture and YAML myself.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
## What this changes and why
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
## Checklist
|
|
6
|
+
|
|
7
|
+
- [ ] `make check` passes locally (lint + typecheck + tests)
|
|
8
|
+
- [ ] `make test-postgres` passes, if this touches the database layer
|
|
9
|
+
- [ ] New behavior has new tests (this project does not accept "tested locally, trust me" for
|
|
10
|
+
anything touching detection, prevention, redaction, or input validation)
|
|
11
|
+
- [ ] Docs updated in the same PR if this changes a metric definition, a policy rule, an API route,
|
|
12
|
+
a CLI command, or a documented limitation
|
|
13
|
+
- [ ] If this touches fixtures, adapters, or anything that could reach outside the local lab: I've
|
|
14
|
+
read [SECURITY.md](../SECURITY.md) and this stays inside the local-fixture safety boundary
|
|
15
|
+
|
|
16
|
+
## Anything reviewers should look at closely
|
|
17
|
+
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# CI for a local security lab: everything here runs against this repository's own fixtures
|
|
2
|
+
# and its own Docker Compose stack on the runner. Nothing in this workflow contacts an
|
|
3
|
+
# external server.
|
|
4
|
+
name: CI
|
|
5
|
+
|
|
6
|
+
on:
|
|
7
|
+
push:
|
|
8
|
+
branches: [main]
|
|
9
|
+
pull_request:
|
|
10
|
+
|
|
11
|
+
concurrency:
|
|
12
|
+
group: ${{ github.workflow }}-${{ github.ref }}
|
|
13
|
+
cancel-in-progress: true
|
|
14
|
+
|
|
15
|
+
permissions:
|
|
16
|
+
contents: read
|
|
17
|
+
|
|
18
|
+
jobs:
|
|
19
|
+
check:
|
|
20
|
+
name: lint, typecheck, test (SQLite, Python ${{ matrix.python-version }})
|
|
21
|
+
runs-on: ubuntu-latest
|
|
22
|
+
strategy:
|
|
23
|
+
fail-fast: false
|
|
24
|
+
matrix:
|
|
25
|
+
python-version: ["3.12", "3.13"]
|
|
26
|
+
steps:
|
|
27
|
+
- uses: actions/checkout@v7
|
|
28
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
29
|
+
- run: make check
|
|
30
|
+
env:
|
|
31
|
+
PYTHON_VERSION: ${{ matrix.python-version }}
|
|
32
|
+
- run: make coverage
|
|
33
|
+
if: matrix.python-version == '3.12'
|
|
34
|
+
env:
|
|
35
|
+
PYTHON_VERSION: ${{ matrix.python-version }}
|
|
36
|
+
|
|
37
|
+
postgres:
|
|
38
|
+
name: full suite against real PostgreSQL (embedded, no Docker)
|
|
39
|
+
runs-on: ubuntu-latest
|
|
40
|
+
steps:
|
|
41
|
+
- uses: actions/checkout@v7
|
|
42
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
43
|
+
- run: make install EXTRAS=dev,dashboard,pgtest
|
|
44
|
+
- run: .venv/bin/python scripts/test_postgres.py -q
|
|
45
|
+
|
|
46
|
+
cisco-scanner:
|
|
47
|
+
name: benchmark the open-source Cisco MCP Scanner (YARA analyzer, offline)
|
|
48
|
+
runs-on: ubuntu-latest
|
|
49
|
+
steps:
|
|
50
|
+
- uses: actions/checkout@v7
|
|
51
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
52
|
+
- run: make setup
|
|
53
|
+
- name: Install the pinned scanner in its own venv and run all four adapters
|
|
54
|
+
run: make cisco-demo
|
|
55
|
+
- name: The live adapter test must agree with the published results
|
|
56
|
+
run: .venv/bin/python -m pytest -q tests/integration/test_cisco_scanner_adapter.py
|
|
57
|
+
env:
|
|
58
|
+
GUARDBENCH_CISCO_MCP_SCANNER: .venv-scanners/bin/mcp-scanner
|
|
59
|
+
|
|
60
|
+
docker:
|
|
61
|
+
name: Docker image build + Compose smoke test
|
|
62
|
+
runs-on: ubuntu-latest
|
|
63
|
+
steps:
|
|
64
|
+
- uses: actions/checkout@v7
|
|
65
|
+
# docker-compose.yml refuses to render without a password and API key, so create .env first.
|
|
66
|
+
- name: Generate local-lab-only credentials for this CI run
|
|
67
|
+
run: python3 scripts/bootstrap_env.py
|
|
68
|
+
- name: Validate docker-compose.yml statically
|
|
69
|
+
run: docker compose config --quiet
|
|
70
|
+
- name: Build the image
|
|
71
|
+
run: docker compose build
|
|
72
|
+
- name: Start postgres, api, dashboard
|
|
73
|
+
run: docker compose up -d --wait --wait-timeout 180
|
|
74
|
+
- name: Health checks
|
|
75
|
+
run: |
|
|
76
|
+
curl -sf http://127.0.0.1:8000/health
|
|
77
|
+
curl -sf http://127.0.0.1:8501/_stcore/health
|
|
78
|
+
# Run as the runner's own user (like `make docker-demo`) so the container can write into the
|
|
79
|
+
# bind-mounted reports/ directory, which the checkout owns.
|
|
80
|
+
- name: Seed and run the demo inside the isolated test-runner container
|
|
81
|
+
run: |
|
|
82
|
+
docker compose --profile tools run --rm --user "$(id -u):$(id -g)" test-runner seed-demo
|
|
83
|
+
docker compose --profile tools run --rm --user "$(id -u):$(id -g)" test-runner benchmark run \
|
|
84
|
+
--project demo --cases test_cases/ \
|
|
85
|
+
--adapter no-defense-baseline --adapter reference-static --adapter reference-runtime \
|
|
86
|
+
--output reports/ci-demo
|
|
87
|
+
- name: Container logs (on failure)
|
|
88
|
+
if: failure()
|
|
89
|
+
run: docker compose logs
|
|
90
|
+
- name: Tear down
|
|
91
|
+
if: always()
|
|
92
|
+
run: docker compose down -v
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# Publishes to PyPI using Trusted Publishing (OIDC) -- no API token is stored in this repo.
|
|
2
|
+
# Triggers only when a GitHub Release is published, never on an ordinary push, so a release is
|
|
3
|
+
# always a deliberate, reviewable action, not a side effect of merging code.
|
|
4
|
+
#
|
|
5
|
+
# One-time setup required on pypi.org before this can succeed (see README/CONTRIBUTING):
|
|
6
|
+
# pypi.org -> your account -> Publishing -> "Add a pending publisher"
|
|
7
|
+
# PyPI project name : mcp-guardbench
|
|
8
|
+
# Owner : Manishmaurya89
|
|
9
|
+
# Repository name : mcp-guardbench
|
|
10
|
+
# Workflow name : publish.yml
|
|
11
|
+
# Environment (optional, recommended): pypi
|
|
12
|
+
name: Publish to PyPI
|
|
13
|
+
|
|
14
|
+
on:
|
|
15
|
+
release:
|
|
16
|
+
types: [published]
|
|
17
|
+
|
|
18
|
+
permissions:
|
|
19
|
+
contents: read
|
|
20
|
+
|
|
21
|
+
jobs:
|
|
22
|
+
build:
|
|
23
|
+
name: Build sdist + wheel
|
|
24
|
+
runs-on: ubuntu-latest
|
|
25
|
+
steps:
|
|
26
|
+
- uses: actions/checkout@v7
|
|
27
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
28
|
+
- run: uv build
|
|
29
|
+
- name: Verify the built distributions (same check PyPI runs)
|
|
30
|
+
run: uvx twine check dist/*
|
|
31
|
+
- uses: actions/upload-artifact@v5
|
|
32
|
+
with:
|
|
33
|
+
name: dist
|
|
34
|
+
path: dist/
|
|
35
|
+
|
|
36
|
+
publish:
|
|
37
|
+
name: Publish to PyPI
|
|
38
|
+
needs: build
|
|
39
|
+
runs-on: ubuntu-latest
|
|
40
|
+
environment:
|
|
41
|
+
name: pypi
|
|
42
|
+
url: https://pypi.org/p/mcp-guardbench
|
|
43
|
+
permissions:
|
|
44
|
+
id-token: write # required for Trusted Publishing; no secret token used or needed
|
|
45
|
+
steps:
|
|
46
|
+
- uses: actions/download-artifact@v6
|
|
47
|
+
with:
|
|
48
|
+
name: dist
|
|
49
|
+
path: dist/
|
|
50
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
.venv/
|
|
6
|
+
.venv-*/
|
|
7
|
+
venv/
|
|
8
|
+
build/
|
|
9
|
+
dist/
|
|
10
|
+
.mypy_cache/
|
|
11
|
+
.ruff_cache/
|
|
12
|
+
.pytest_cache/
|
|
13
|
+
.coverage
|
|
14
|
+
coverage.xml
|
|
15
|
+
htmlcov/
|
|
16
|
+
|
|
17
|
+
# Local environment (never commit real values)
|
|
18
|
+
.env
|
|
19
|
+
.env.*
|
|
20
|
+
!.env.example
|
|
21
|
+
|
|
22
|
+
# Local databases and generated output
|
|
23
|
+
*.db
|
|
24
|
+
*.sqlite
|
|
25
|
+
*.sqlite3
|
|
26
|
+
data/
|
|
27
|
+
reports/*
|
|
28
|
+
!reports/.gitkeep
|
|
29
|
+
!reports/sample/
|
|
30
|
+
!reports/sample/**
|
|
31
|
+
|
|
32
|
+
# Editors / OS
|
|
33
|
+
.idea/
|
|
34
|
+
.vscode/
|
|
35
|
+
.DS_Store
|
|
36
|
+
|
|
37
|
+
# Personal notes (audits, interview prep) - never meant for the public repo
|
|
38
|
+
AUDIT_REPORT.md
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/); the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/) and is pre-1.0, so minor versions may change behavior.
|
|
6
|
+
|
|
7
|
+
## [0.1.0] - 2026-10-02
|
|
8
|
+
|
|
9
|
+
First public release.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `guardbench inspect`: read-only check of the MCP servers configured in Claude Desktop, Claude Code,
|
|
14
|
+
Cursor, VS Code and Windsurf (stdio, Streamable HTTP, SSE; JSONC; `${VAR}` expansion). Requests
|
|
15
|
+
`tools/list` only, analyzes tool metadata and server instructions with the benchmark's rules, pins
|
|
16
|
+
fingerprints to catch rug pulls, masks configured secrets, escapes server output, and exits 1 for CI
|
|
17
|
+
when a finding reaches `--fail-on`. Validated against 7 official MCP reference servers.
|
|
18
|
+
- `cisco-mcp-scanner` adapter: benchmarks Cisco AI Defense's open-source MCP Scanner (YARA analyzer,
|
|
19
|
+
offline) as a separate program; `make cisco-demo` installs it into its own virtual environment.
|
|
20
|
+
- Hard test cases `TP-003` (tool poisoning written in Spanish) and `DF-002` (a secret that leaves
|
|
21
|
+
base64-encoded), and benign control `BN-003` (descriptions that legitimately reference sibling tools),
|
|
22
|
+
with three new fixtures. The reference controls were not changed to pass them.
|
|
23
|
+
- Fixture ground truth recognizes base64 and hex encodings of the synthetic markers.
|
|
24
|
+
- The benchmark lab: 11 local fixtures, 12 test cases, reference static and runtime controls, verified
|
|
25
|
+
detection and prevention metrics, reports (JSON, Markdown, CSV, HTML), REST API, read-only dashboard,
|
|
26
|
+
PostgreSQL and SQLite support, Docker Compose setup.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Contributor Covenant Code of Conduct
|
|
2
|
+
|
|
3
|
+
## Our pledge
|
|
4
|
+
|
|
5
|
+
We as members, contributors, and leaders pledge to make participation in our community a
|
|
6
|
+
harassment-free experience for everyone, regardless of age, body size, visible or invisible
|
|
7
|
+
disability, ethnicity, sex characteristics, gender identity and expression, level of experience,
|
|
8
|
+
education, socio-economic status, nationality, personal appearance, race, religion, or sexual
|
|
9
|
+
identity and orientation.
|
|
10
|
+
|
|
11
|
+
## Our standards
|
|
12
|
+
|
|
13
|
+
Examples of behavior that contributes to a positive environment:
|
|
14
|
+
|
|
15
|
+
* Demonstrating empathy and kindness toward other people
|
|
16
|
+
* Being respectful of differing opinions, viewpoints, and experiences
|
|
17
|
+
* Giving and gracefully accepting constructive feedback
|
|
18
|
+
* Accepting responsibility and apologizing to those affected by our mistakes, and learning from
|
|
19
|
+
the experience
|
|
20
|
+
* Focusing on what is best not just for us as individuals, but for the overall community
|
|
21
|
+
|
|
22
|
+
Examples of unacceptable behavior:
|
|
23
|
+
|
|
24
|
+
* The use of sexualized language or imagery, and sexual attention or advances of any kind
|
|
25
|
+
* Trolling, insulting or derogatory comments, and personal or political attacks
|
|
26
|
+
* Public or private harassment
|
|
27
|
+
* Publishing others' private information, such as a physical or email address, without their
|
|
28
|
+
explicit permission
|
|
29
|
+
* Using this project to test, scan, or attack systems you do not own or are not explicitly
|
|
30
|
+
authorized to test; see [SECURITY.md](SECURITY.md)
|
|
31
|
+
* Other conduct which could reasonably be considered inappropriate in a professional setting
|
|
32
|
+
|
|
33
|
+
## Enforcement responsibilities
|
|
34
|
+
|
|
35
|
+
Project maintainers are responsible for clarifying and enforcing our standards of acceptable
|
|
36
|
+
behavior and will take appropriate and fair corrective action in response to any behavior they
|
|
37
|
+
deem inappropriate, threatening, offensive, or harmful.
|
|
38
|
+
|
|
39
|
+
## Scope
|
|
40
|
+
|
|
41
|
+
This Code of Conduct applies within all community spaces (issues, pull requests, discussions), and
|
|
42
|
+
also applies when an individual is officially representing the community in public spaces.
|
|
43
|
+
|
|
44
|
+
## Enforcement
|
|
45
|
+
|
|
46
|
+
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to the project
|
|
47
|
+
maintainers via a private channel (a GitHub Security Advisory, or a direct message to a maintainer
|
|
48
|
+
if one is listed on the repository). All complaints will be reviewed and investigated promptly and
|
|
49
|
+
fairly.
|
|
50
|
+
|
|
51
|
+
## Attribution
|
|
52
|
+
|
|
53
|
+
This Code of Conduct is adapted from the [Contributor Covenant](https://www.contributor-covenant.org),
|
|
54
|
+
version 2.1, available at
|
|
55
|
+
https://www.contributor-covenant.org/version/2/1/code_of_conduct.html.
|