mcp-eval-gate 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_eval_gate-0.3.0/PKG-INFO +154 -0
- mcp_eval_gate-0.3.0/README.md +127 -0
- mcp_eval_gate-0.3.0/pyproject.toml +70 -0
- mcp_eval_gate-0.3.0/pyproject.toml.orig +56 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/__init__.py +2 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/baseline.py +61 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/cli.py +109 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/golden_set.py +66 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/judge.py +75 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/mcp_client.py +44 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/mcp_server.py +74 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/models.py +67 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/py.typed +0 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/report.py +67 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/runner.py +46 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/scoring.py +45 -0
- mcp_eval_gate-0.3.0/src/mcp_eval_gate/text_diff.py +35 -0
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: mcp-eval-gate
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: CI regression gate for MCP servers: run a golden set of tool calls, diff against a baseline, fail the build on regression
|
|
5
|
+
Keywords: mcp,model-context-protocol,regression-testing,testing,ci,golden-tests
|
|
6
|
+
Author: Umer Karachiwala
|
|
7
|
+
Author-email: Umer Karachiwala <karachiwalaumer2612@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Topic :: Software Development :: Testing
|
|
14
|
+
Classifier: Typing :: Typed
|
|
15
|
+
Requires-Dist: mcp>=2.0
|
|
16
|
+
Requires-Dist: pyyaml>=6.0
|
|
17
|
+
Requires-Dist: click>=8.1
|
|
18
|
+
Requires-Dist: rich>=13.7
|
|
19
|
+
Requires-Dist: anthropic>=0.40 ; extra == 'judge'
|
|
20
|
+
Requires-Python: >=3.12
|
|
21
|
+
Project-URL: Homepage, https://github.com/Umer-2612/mcp-eval-gate
|
|
22
|
+
Project-URL: Repository, https://github.com/Umer-2612/mcp-eval-gate
|
|
23
|
+
Project-URL: Issues, https://github.com/Umer-2612/mcp-eval-gate/issues
|
|
24
|
+
Project-URL: Changelog, https://github.com/Umer-2612/mcp-eval-gate/blob/main/CHANGELOG.md
|
|
25
|
+
Provides-Extra: judge
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# mcp-eval-gate
|
|
29
|
+
|
|
30
|
+
A CI regression gate for MCP servers. You write a golden set of tool calls with known-good
|
|
31
|
+
outputs. `mcp-eval-gate` calls those tools on your server, compares the results to a saved
|
|
32
|
+
baseline, and exits non-zero if anything got worse, even when the tool's schema did not
|
|
33
|
+
change.
|
|
34
|
+
|
|
35
|
+
## Why
|
|
36
|
+
|
|
37
|
+
An MCP server's tool names and input schemas can stay identical while what a tool returns
|
|
38
|
+
quietly gets worse: a refactor breaks a code path, a dependency bump changes behavior.
|
|
39
|
+
Nothing crashes, the agent using it just gets worse answers.
|
|
40
|
+
|
|
41
|
+
Other tools cover parts of this. The official Inspector's CLI runs scripted single-run
|
|
42
|
+
assertions. `mcp-server-diff` compares declared schemas and says it does not test output
|
|
43
|
+
correctness. A few other early tools record golden outputs and diff them too, for example
|
|
44
|
+
[vexyo](https://github.com/vexyohq/vexyo) and
|
|
45
|
+
[cisco-open/mcptoolkit-test](https://github.com/cisco-open/mcptoolkit-test).
|
|
46
|
+
`mcp-eval-gate` is another take on golden-output regression.
|
|
47
|
+
|
|
48
|
+
## Install
|
|
49
|
+
|
|
50
|
+
Requires Python 3.12+.
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install mcp-eval-gate
|
|
54
|
+
|
|
55
|
+
# only if you use match_type: judge
|
|
56
|
+
pip install "mcp-eval-gate[judge]"
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Quickstart
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
mcp-eval-gate init # scaffolds golden_set.yaml
|
|
63
|
+
# edit golden_set.yaml with how to reach your server and your test cases
|
|
64
|
+
mcp-eval-gate run --update-baseline # first run: record the baseline
|
|
65
|
+
mcp-eval-gate run # later runs: gate on regressions
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
`run` prints a table and a diff, and exits `1` if any case fails or regresses against the
|
|
69
|
+
baseline. A tool call that returns an error always fails its case.
|
|
70
|
+
|
|
71
|
+
## Golden set
|
|
72
|
+
|
|
73
|
+
```yaml
|
|
74
|
+
server:
|
|
75
|
+
command: node
|
|
76
|
+
args: ["dist/index.js"]
|
|
77
|
+
# or, for an HTTP server instead of stdio:
|
|
78
|
+
# url: http://localhost:3000/mcp
|
|
79
|
+
|
|
80
|
+
cases:
|
|
81
|
+
# substring match
|
|
82
|
+
- id: get-weather-nyc
|
|
83
|
+
tool_name: get_weather
|
|
84
|
+
tool_args:
|
|
85
|
+
city: "New York"
|
|
86
|
+
match_type: contains
|
|
87
|
+
expected_output: "New York"
|
|
88
|
+
|
|
89
|
+
# exact match
|
|
90
|
+
- id: cancel-subscription
|
|
91
|
+
tool_name: cancel_subscription
|
|
92
|
+
tool_args:
|
|
93
|
+
immediate: true
|
|
94
|
+
match_type: exact
|
|
95
|
+
expected_output: "cancelled"
|
|
96
|
+
|
|
97
|
+
# scored by an LLM against a rubric (needs ANTHROPIC_API_KEY)
|
|
98
|
+
- id: retention-policy-answer
|
|
99
|
+
tool_name: search_docs
|
|
100
|
+
tool_args:
|
|
101
|
+
query: "data retention policy"
|
|
102
|
+
match_type: judge
|
|
103
|
+
judge_criteria: "Answer must state data is retained for 90 days"
|
|
104
|
+
min_judge_score: 0.8
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
[`examples/golden_set.yaml`](https://github.com/Umer-2612/mcp-eval-gate/blob/main/examples/golden_set.yaml) is a copy you can edit.
|
|
108
|
+
|
|
109
|
+
## Run as an MCP tool
|
|
110
|
+
|
|
111
|
+
The package also installs `mcp-eval-gate-mcp`, an MCP server over stdio with one tool,
|
|
112
|
+
`run_eval_gate(config_path, baseline_path, update_baseline)`. Add it to an MCP client's config:
|
|
113
|
+
|
|
114
|
+
```json
|
|
115
|
+
{
|
|
116
|
+
"mcpServers": {
|
|
117
|
+
"mcp-eval-gate": { "command": "mcp-eval-gate-mcp" }
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Validation
|
|
123
|
+
|
|
124
|
+
[`validation/`](https://github.com/Umer-2612/mcp-eval-gate/tree/main/validation) has two runs against real servers, each with the actual command
|
|
125
|
+
output committed and steps to reproduce it. No paid API calls.
|
|
126
|
+
|
|
127
|
+
- A one-line regression planted in the official MCP reference server, caught with exit
|
|
128
|
+
code `1` and a real diff.
|
|
129
|
+
- [A real bug](https://github.com/Umer-2612/mcp-eval-gate/tree/main/validation/utf8-boundary) the official filesystem server shipped (garbled
|
|
130
|
+
text when a multi-byte character straddled a read boundary), caught by running the commit
|
|
131
|
+
before its upstream fix against a baseline from the fixed commit.
|
|
132
|
+
|
|
133
|
+
Both are small. They show the gate works end to end on real code, not how often this class
|
|
134
|
+
of bug occurs.
|
|
135
|
+
|
|
136
|
+
## Limitations
|
|
137
|
+
|
|
138
|
+
- Matching is exact, substring, or an LLM judge. There is no normalization for volatile
|
|
139
|
+
output such as timestamps or ids, so output that varies between runs can't be compared
|
|
140
|
+
reliably.
|
|
141
|
+
- It checks tool outputs only, not schemas or protocol conformance.
|
|
142
|
+
- `match_type: judge` has only been tested against a stub client, not the live Anthropic API.
|
|
143
|
+
|
|
144
|
+
## Development
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
uv sync --extra judge --dev
|
|
148
|
+
uv run pytest --cov=src --cov-report=term-missing
|
|
149
|
+
uv run ruff check src tests
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## License
|
|
153
|
+
|
|
154
|
+
MIT
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# mcp-eval-gate
|
|
2
|
+
|
|
3
|
+
A CI regression gate for MCP servers. You write a golden set of tool calls with known-good
|
|
4
|
+
outputs. `mcp-eval-gate` calls those tools on your server, compares the results to a saved
|
|
5
|
+
baseline, and exits non-zero if anything got worse, even when the tool's schema did not
|
|
6
|
+
change.
|
|
7
|
+
|
|
8
|
+
## Why
|
|
9
|
+
|
|
10
|
+
An MCP server's tool names and input schemas can stay identical while what a tool returns
|
|
11
|
+
quietly gets worse: a refactor breaks a code path, a dependency bump changes behavior.
|
|
12
|
+
Nothing crashes, the agent using it just gets worse answers.
|
|
13
|
+
|
|
14
|
+
Other tools cover parts of this. The official Inspector's CLI runs scripted single-run
|
|
15
|
+
assertions. `mcp-server-diff` compares declared schemas and says it does not test output
|
|
16
|
+
correctness. A few other early tools record golden outputs and diff them too, for example
|
|
17
|
+
[vexyo](https://github.com/vexyohq/vexyo) and
|
|
18
|
+
[cisco-open/mcptoolkit-test](https://github.com/cisco-open/mcptoolkit-test).
|
|
19
|
+
`mcp-eval-gate` is another take on golden-output regression.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
Requires Python 3.12+.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install mcp-eval-gate
|
|
27
|
+
|
|
28
|
+
# only if you use match_type: judge
|
|
29
|
+
pip install "mcp-eval-gate[judge]"
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Quickstart
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
mcp-eval-gate init # scaffolds golden_set.yaml
|
|
36
|
+
# edit golden_set.yaml with how to reach your server and your test cases
|
|
37
|
+
mcp-eval-gate run --update-baseline # first run: record the baseline
|
|
38
|
+
mcp-eval-gate run # later runs: gate on regressions
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
`run` prints a table and a diff, and exits `1` if any case fails or regresses against the
|
|
42
|
+
baseline. A tool call that returns an error always fails its case.
|
|
43
|
+
|
|
44
|
+
## Golden set
|
|
45
|
+
|
|
46
|
+
```yaml
|
|
47
|
+
server:
|
|
48
|
+
command: node
|
|
49
|
+
args: ["dist/index.js"]
|
|
50
|
+
# or, for an HTTP server instead of stdio:
|
|
51
|
+
# url: http://localhost:3000/mcp
|
|
52
|
+
|
|
53
|
+
cases:
|
|
54
|
+
# substring match
|
|
55
|
+
- id: get-weather-nyc
|
|
56
|
+
tool_name: get_weather
|
|
57
|
+
tool_args:
|
|
58
|
+
city: "New York"
|
|
59
|
+
match_type: contains
|
|
60
|
+
expected_output: "New York"
|
|
61
|
+
|
|
62
|
+
# exact match
|
|
63
|
+
- id: cancel-subscription
|
|
64
|
+
tool_name: cancel_subscription
|
|
65
|
+
tool_args:
|
|
66
|
+
immediate: true
|
|
67
|
+
match_type: exact
|
|
68
|
+
expected_output: "cancelled"
|
|
69
|
+
|
|
70
|
+
# scored by an LLM against a rubric (needs ANTHROPIC_API_KEY)
|
|
71
|
+
- id: retention-policy-answer
|
|
72
|
+
tool_name: search_docs
|
|
73
|
+
tool_args:
|
|
74
|
+
query: "data retention policy"
|
|
75
|
+
match_type: judge
|
|
76
|
+
judge_criteria: "Answer must state data is retained for 90 days"
|
|
77
|
+
min_judge_score: 0.8
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
[`examples/golden_set.yaml`](https://github.com/Umer-2612/mcp-eval-gate/blob/main/examples/golden_set.yaml) is a copy you can edit.
|
|
81
|
+
|
|
82
|
+
## Run as an MCP tool
|
|
83
|
+
|
|
84
|
+
The package also installs `mcp-eval-gate-mcp`, an MCP server over stdio with one tool,
|
|
85
|
+
`run_eval_gate(config_path, baseline_path, update_baseline)`. Add it to an MCP client's config:
|
|
86
|
+
|
|
87
|
+
```json
|
|
88
|
+
{
|
|
89
|
+
"mcpServers": {
|
|
90
|
+
"mcp-eval-gate": { "command": "mcp-eval-gate-mcp" }
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Validation
|
|
96
|
+
|
|
97
|
+
[`validation/`](https://github.com/Umer-2612/mcp-eval-gate/tree/main/validation) has two runs against real servers, each with the actual command
|
|
98
|
+
output committed and steps to reproduce it. No paid API calls.
|
|
99
|
+
|
|
100
|
+
- A one-line regression planted in the official MCP reference server, caught with exit
|
|
101
|
+
code `1` and a real diff.
|
|
102
|
+
- [A real bug](https://github.com/Umer-2612/mcp-eval-gate/tree/main/validation/utf8-boundary) the official filesystem server shipped (garbled
|
|
103
|
+
text when a multi-byte character straddled a read boundary), caught by running the commit
|
|
104
|
+
before its upstream fix against a baseline from the fixed commit.
|
|
105
|
+
|
|
106
|
+
Both are small. They show the gate works end to end on real code, not how often this class
|
|
107
|
+
of bug occurs.
|
|
108
|
+
|
|
109
|
+
## Limitations
|
|
110
|
+
|
|
111
|
+
- Matching is exact, substring, or an LLM judge. There is no normalization for volatile
|
|
112
|
+
output such as timestamps or ids, so output that varies between runs can't be compared
|
|
113
|
+
reliably.
|
|
114
|
+
- It checks tool outputs only, not schemas or protocol conformance.
|
|
115
|
+
- `match_type: judge` has only been tested against a stub client, not the live Anthropic API.
|
|
116
|
+
|
|
117
|
+
## Development
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
uv sync --extra judge --dev
|
|
121
|
+
uv run pytest --cov=src --cov-report=term-missing
|
|
122
|
+
uv run ruff check src tests
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## License
|
|
126
|
+
|
|
127
|
+
MIT
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mcp-eval-gate"
|
|
3
|
+
version = "0.3.0"
|
|
4
|
+
description = "CI regression gate for MCP servers: run a golden set of tool calls, diff against a baseline, fail the build on regression"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
requires-python = ">=3.12"
|
|
8
|
+
keywords = [
|
|
9
|
+
"mcp",
|
|
10
|
+
"model-context-protocol",
|
|
11
|
+
"regression-testing",
|
|
12
|
+
"testing",
|
|
13
|
+
"ci",
|
|
14
|
+
"golden-tests",
|
|
15
|
+
]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 3 - Alpha",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Topic :: Software Development :: Testing",
|
|
22
|
+
"Typing :: Typed",
|
|
23
|
+
]
|
|
24
|
+
dependencies = [
|
|
25
|
+
"mcp>=2.0",
|
|
26
|
+
"pyyaml>=6.0",
|
|
27
|
+
"click>=8.1",
|
|
28
|
+
"rich>=13.7",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[[project.authors]]
|
|
32
|
+
name = "Umer Karachiwala"
|
|
33
|
+
email = "karachiwalaumer2612@gmail.com"
|
|
34
|
+
|
|
35
|
+
[project.urls]
|
|
36
|
+
Homepage = "https://github.com/Umer-2612/mcp-eval-gate"
|
|
37
|
+
Repository = "https://github.com/Umer-2612/mcp-eval-gate"
|
|
38
|
+
Issues = "https://github.com/Umer-2612/mcp-eval-gate/issues"
|
|
39
|
+
Changelog = "https://github.com/Umer-2612/mcp-eval-gate/blob/main/CHANGELOG.md"
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
judge = ["anthropic>=0.40"]
|
|
43
|
+
|
|
44
|
+
[project.scripts]
|
|
45
|
+
mcp-eval-gate = "mcp_eval_gate.cli:main"
|
|
46
|
+
mcp-eval-gate-mcp = "mcp_eval_gate.mcp_server:main"
|
|
47
|
+
|
|
48
|
+
[dependency-groups]
|
|
49
|
+
dev = [
|
|
50
|
+
"pytest>=8.0",
|
|
51
|
+
"pytest-cov>=5.0",
|
|
52
|
+
"ruff>=0.7",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
[tool.ruff]
|
|
56
|
+
line-length = 115
|
|
57
|
+
target-version = "py312"
|
|
58
|
+
|
|
59
|
+
[tool.ruff.lint]
|
|
60
|
+
select = [
|
|
61
|
+
"E",
|
|
62
|
+
"F",
|
|
63
|
+
"I",
|
|
64
|
+
"UP",
|
|
65
|
+
"B",
|
|
66
|
+
]
|
|
67
|
+
|
|
68
|
+
[build-system]
|
|
69
|
+
requires = ["uv_build>=0.12.17,<0.13.0"]
|
|
70
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mcp-eval-gate"
|
|
3
|
+
version = "0.3.0"
|
|
4
|
+
description = "CI regression gate for MCP servers: run a golden set of tool calls, diff against a baseline, fail the build on regression"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
authors = [
|
|
8
|
+
{ name = "Umer Karachiwala", email = "karachiwalaumer2612@gmail.com" }
|
|
9
|
+
]
|
|
10
|
+
requires-python = ">=3.12"
|
|
11
|
+
keywords = ["mcp", "model-context-protocol", "regression-testing", "testing", "ci", "golden-tests"]
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 3 - Alpha",
|
|
14
|
+
"Intended Audience :: Developers",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.12",
|
|
17
|
+
"Topic :: Software Development :: Testing",
|
|
18
|
+
"Typing :: Typed",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"mcp>=2.0",
|
|
22
|
+
"pyyaml>=6.0",
|
|
23
|
+
"click>=8.1",
|
|
24
|
+
"rich>=13.7",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
Homepage = "https://github.com/Umer-2612/mcp-eval-gate"
|
|
29
|
+
Repository = "https://github.com/Umer-2612/mcp-eval-gate"
|
|
30
|
+
Issues = "https://github.com/Umer-2612/mcp-eval-gate/issues"
|
|
31
|
+
Changelog = "https://github.com/Umer-2612/mcp-eval-gate/blob/main/CHANGELOG.md"
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
judge = ["anthropic>=0.40"]
|
|
35
|
+
|
|
36
|
+
[project.scripts]
|
|
37
|
+
mcp-eval-gate = "mcp_eval_gate.cli:main"
|
|
38
|
+
mcp-eval-gate-mcp = "mcp_eval_gate.mcp_server:main"
|
|
39
|
+
|
|
40
|
+
[dependency-groups]
|
|
41
|
+
dev = [
|
|
42
|
+
"pytest>=8.0",
|
|
43
|
+
"pytest-cov>=5.0",
|
|
44
|
+
"ruff>=0.7",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
[tool.ruff]
|
|
48
|
+
line-length = 115
|
|
49
|
+
target-version = "py312"
|
|
50
|
+
|
|
51
|
+
[tool.ruff.lint]
|
|
52
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
53
|
+
|
|
54
|
+
[build-system]
|
|
55
|
+
requires = ["uv_build>=0.12.17,<0.13.0"]
|
|
56
|
+
build-backend = "uv_build"
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Load/save a committed baseline and diff a fresh run against it.
|
|
2
|
+
|
|
3
|
+
A baseline is a simple {case_id: score} snapshot, checked into the repo
|
|
4
|
+
alongside the golden set, so "did this change make things worse" is a
|
|
5
|
+
git-diffable question instead of a moving target.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from mcp_eval_gate.models import CaseResult, Regression
|
|
14
|
+
|
|
15
|
+
DEFAULT_THRESHOLD = 0.0
|
|
16
|
+
IMPLICIT_BASELINE_FOR_NEW_CASES = 1.0
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def load_baseline(path: Path) -> dict[str, float]:
|
|
20
|
+
if not path.exists():
|
|
21
|
+
return {}
|
|
22
|
+
return json.loads(path.read_text())
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def save_baseline(path: Path, results: list[CaseResult]) -> None:
|
|
26
|
+
baseline = {result.case_id: result.score for result in results}
|
|
27
|
+
path.write_text(json.dumps(baseline, indent=2, sort_keys=True) + "\n")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def diff_against_baseline(
|
|
31
|
+
results: list[CaseResult], baseline: dict[str, float], threshold: float = DEFAULT_THRESHOLD
|
|
32
|
+
) -> list[Regression]:
|
|
33
|
+
regressions: list[Regression] = []
|
|
34
|
+
seen_case_ids: set[str] = set()
|
|
35
|
+
|
|
36
|
+
for result in results:
|
|
37
|
+
seen_case_ids.add(result.case_id)
|
|
38
|
+
baseline_score = baseline.get(result.case_id, IMPLICIT_BASELINE_FOR_NEW_CASES)
|
|
39
|
+
drop = baseline_score - result.score
|
|
40
|
+
if drop > threshold:
|
|
41
|
+
regressions.append(
|
|
42
|
+
Regression(
|
|
43
|
+
case_id=result.case_id,
|
|
44
|
+
baseline_score=baseline_score,
|
|
45
|
+
current_score=result.score,
|
|
46
|
+
detail=result.detail,
|
|
47
|
+
)
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
for case_id, baseline_score in baseline.items():
|
|
51
|
+
if case_id not in seen_case_ids:
|
|
52
|
+
regressions.append(
|
|
53
|
+
Regression(
|
|
54
|
+
case_id=case_id,
|
|
55
|
+
baseline_score=baseline_score,
|
|
56
|
+
current_score=0.0,
|
|
57
|
+
detail="case present in baseline is missing from the current run",
|
|
58
|
+
)
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
return regressions
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""CLI: `mcp-eval-gate run` and `mcp-eval-gate init`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import anyio
|
|
9
|
+
import click
|
|
10
|
+
|
|
11
|
+
from mcp_eval_gate.baseline import diff_against_baseline, load_baseline, save_baseline
|
|
12
|
+
from mcp_eval_gate.golden_set import GoldenSetError, load_golden_set
|
|
13
|
+
from mcp_eval_gate.judge import build_default_anthropic_client
|
|
14
|
+
from mcp_eval_gate.report import exit_code_for, print_console_report, write_html_report
|
|
15
|
+
from mcp_eval_gate.runner import run_golden_set
|
|
16
|
+
|
|
17
|
+
SAMPLE_GOLDEN_SET = """\
|
|
18
|
+
server:
|
|
19
|
+
command: node
|
|
20
|
+
args: ["dist/index.js"]
|
|
21
|
+
# or, for an HTTP server instead of stdio:
|
|
22
|
+
# url: http://localhost:3000/mcp
|
|
23
|
+
|
|
24
|
+
judge_model: claude-sonnet-4-5
|
|
25
|
+
|
|
26
|
+
cases:
|
|
27
|
+
- id: example-contains-case
|
|
28
|
+
tool_name: get_weather
|
|
29
|
+
tool_args:
|
|
30
|
+
city: "New York"
|
|
31
|
+
match_type: contains
|
|
32
|
+
expected_output: "New York"
|
|
33
|
+
|
|
34
|
+
- id: example-exact-case
|
|
35
|
+
tool_name: cancel_subscription
|
|
36
|
+
tool_args:
|
|
37
|
+
immediate: true
|
|
38
|
+
match_type: exact
|
|
39
|
+
expected_output: "cancelled"
|
|
40
|
+
|
|
41
|
+
- id: example-judge-case
|
|
42
|
+
tool_name: search_docs
|
|
43
|
+
tool_args:
|
|
44
|
+
query: "data retention policy"
|
|
45
|
+
match_type: judge
|
|
46
|
+
judge_criteria: "Answer must state data is retained for 90 days"
|
|
47
|
+
min_judge_score: 0.8
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@click.group()
|
|
52
|
+
def main() -> None:
|
|
53
|
+
"""mcp-eval-gate: CI regression gate for MCP servers."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@main.command()
|
|
57
|
+
@click.option(
|
|
58
|
+
"--config", "config_path", type=click.Path(path_type=Path), default="golden_set.yaml", show_default=True
|
|
59
|
+
)
|
|
60
|
+
@click.option(
|
|
61
|
+
"--baseline", "baseline_path", type=click.Path(path_type=Path), default="baseline.json", show_default=True
|
|
62
|
+
)
|
|
63
|
+
@click.option(
|
|
64
|
+
"--update-baseline", is_flag=True, help="Overwrite the baseline with this run's scores instead of gating."
|
|
65
|
+
)
|
|
66
|
+
@click.option(
|
|
67
|
+
"--html-report", type=click.Path(path_type=Path), default=None, help="Write an HTML report to this path."
|
|
68
|
+
)
|
|
69
|
+
def run(config_path: Path, baseline_path: Path, update_baseline: bool, html_report: Path | None) -> None:
|
|
70
|
+
"""Run the golden set against a live MCP server and gate on regressions."""
|
|
71
|
+
try:
|
|
72
|
+
config = load_golden_set(config_path)
|
|
73
|
+
except GoldenSetError as exc:
|
|
74
|
+
click.secho(f"error: {exc}", fg="red", err=True)
|
|
75
|
+
sys.exit(2)
|
|
76
|
+
|
|
77
|
+
anthropic_client = build_default_anthropic_client()
|
|
78
|
+
results = anyio.run(lambda: run_golden_set(config, anthropic_client=anthropic_client))
|
|
79
|
+
|
|
80
|
+
if update_baseline:
|
|
81
|
+
save_baseline(baseline_path, results)
|
|
82
|
+
click.secho(f"baseline updated: {baseline_path}", fg="cyan")
|
|
83
|
+
print_console_report(results, regressions=[])
|
|
84
|
+
return
|
|
85
|
+
|
|
86
|
+
baseline = load_baseline(baseline_path)
|
|
87
|
+
regressions = diff_against_baseline(results, baseline)
|
|
88
|
+
print_console_report(results, regressions)
|
|
89
|
+
|
|
90
|
+
if html_report:
|
|
91
|
+
write_html_report(results, regressions, html_report)
|
|
92
|
+
click.echo(f"html report written to {html_report}")
|
|
93
|
+
|
|
94
|
+
sys.exit(exit_code_for(results, regressions))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@main.command()
|
|
98
|
+
@click.option("--out", type=click.Path(path_type=Path), default="golden_set.yaml", show_default=True)
|
|
99
|
+
def init(out: Path) -> None:
|
|
100
|
+
"""Scaffold a starter golden_set.yaml."""
|
|
101
|
+
if out.exists():
|
|
102
|
+
click.secho(f"refusing to overwrite existing file: {out}", fg="red", err=True)
|
|
103
|
+
sys.exit(1)
|
|
104
|
+
out.write_text(SAMPLE_GOLDEN_SET)
|
|
105
|
+
click.secho(f"wrote {out}", fg="green")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
if __name__ == "__main__":
|
|
109
|
+
main()
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Load a golden-set YAML file into a validated GoldenSetConfig."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import yaml
|
|
8
|
+
|
|
9
|
+
from mcp_eval_gate.models import GoldenCase, GoldenSetConfig, MatchType, ServerTarget
|
|
10
|
+
|
|
11
|
+
REQUIRED_CASE_FIELDS = ("id", "tool_name")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class GoldenSetError(ValueError):
|
|
15
|
+
"""Raised when a golden-set file is missing, malformed, or fails validation."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def load_golden_set(path: Path) -> GoldenSetConfig:
|
|
19
|
+
if not path.exists():
|
|
20
|
+
raise GoldenSetError(f"golden set file not found: {path}")
|
|
21
|
+
|
|
22
|
+
raw = yaml.safe_load(path.read_text()) or {}
|
|
23
|
+
|
|
24
|
+
if "server" not in raw:
|
|
25
|
+
raise GoldenSetError("golden set file is missing the required `server` block")
|
|
26
|
+
|
|
27
|
+
server = _parse_server(raw["server"])
|
|
28
|
+
cases = tuple(_parse_case(raw_case) for raw_case in raw.get("cases", []))
|
|
29
|
+
_reject_duplicate_ids(cases)
|
|
30
|
+
|
|
31
|
+
return GoldenSetConfig(
|
|
32
|
+
server=server,
|
|
33
|
+
cases=cases,
|
|
34
|
+
judge_model=raw.get("judge_model", "claude-sonnet-4-5"),
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _parse_server(raw_server: dict) -> ServerTarget:
|
|
39
|
+
try:
|
|
40
|
+
return ServerTarget(
|
|
41
|
+
command=raw_server.get("command"),
|
|
42
|
+
args=tuple(raw_server.get("args", ())),
|
|
43
|
+
env=raw_server.get("env"),
|
|
44
|
+
url=raw_server.get("url"),
|
|
45
|
+
)
|
|
46
|
+
except ValueError as exc:
|
|
47
|
+
raise GoldenSetError(str(exc)) from exc
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _parse_case(raw_case: dict) -> GoldenCase:
|
|
51
|
+
missing = [field for field in REQUIRED_CASE_FIELDS if field not in raw_case]
|
|
52
|
+
if missing:
|
|
53
|
+
case_id = raw_case.get("id", "<unknown>")
|
|
54
|
+
raise GoldenSetError(f"case '{case_id}' is missing required field(s): {', '.join(missing)}")
|
|
55
|
+
|
|
56
|
+
fields = dict(raw_case)
|
|
57
|
+
fields["match_type"] = MatchType(fields.get("match_type", MatchType.CONTAINS))
|
|
58
|
+
return GoldenCase(**fields)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _reject_duplicate_ids(cases: tuple[GoldenCase, ...]) -> None:
|
|
62
|
+
seen: set[str] = set()
|
|
63
|
+
for case in cases:
|
|
64
|
+
if case.id in seen:
|
|
65
|
+
raise GoldenSetError(f"duplicate case id: '{case.id}'")
|
|
66
|
+
seen.add(case.id)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""LLM-as-judge scoring for open-ended tool outputs, via the Anthropic API.
|
|
2
|
+
|
|
3
|
+
Used only for cases with match_type=judge — exact and contains cases are
|
|
4
|
+
scored deterministically in scoring.py and never need a model call.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
_JSON_OBJECT_PATTERN = re.compile(r"\{.*\}", re.DOTALL)
|
|
15
|
+
|
|
16
|
+
JUDGE_PROMPT_TEMPLATE = """You are grading an MCP tool's output against a rubric.
|
|
17
|
+
|
|
18
|
+
Rubric (must be satisfied):
|
|
19
|
+
{criteria}
|
|
20
|
+
|
|
21
|
+
Tool output:
|
|
22
|
+
{answer}
|
|
23
|
+
|
|
24
|
+
Respond with ONLY a JSON object of the form:
|
|
25
|
+
{{"score": <float between 0.0 and 1.0>, "reasoning": "<one sentence>"}}
|
|
26
|
+
A score of 1.0 means the rubric is fully satisfied; 0.0 means it is not satisfied at all."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class JudgeParseError(ValueError):
|
|
30
|
+
"""Raised when the judge model's response can't be parsed into a score."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def build_judge_prompt(*, criteria: str, answer: str) -> str:
|
|
34
|
+
return JUDGE_PROMPT_TEMPLATE.format(criteria=criteria, answer=answer)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def parse_judge_response(text: str) -> tuple[float, str]:
|
|
38
|
+
match = _JSON_OBJECT_PATTERN.search(text)
|
|
39
|
+
if not match:
|
|
40
|
+
raise JudgeParseError(f"no JSON object found in judge response: {text!r}")
|
|
41
|
+
|
|
42
|
+
try:
|
|
43
|
+
payload = json.loads(match.group(0))
|
|
44
|
+
except json.JSONDecodeError as exc:
|
|
45
|
+
raise JudgeParseError(f"judge response was not valid JSON: {text!r}") from exc
|
|
46
|
+
|
|
47
|
+
if "score" not in payload:
|
|
48
|
+
raise JudgeParseError(f"judge response missing 'score' field: {payload!r}")
|
|
49
|
+
|
|
50
|
+
score = max(0.0, min(1.0, float(payload["score"])))
|
|
51
|
+
reasoning = str(payload.get("reasoning", ""))
|
|
52
|
+
return score, reasoning
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def build_default_anthropic_client() -> Any | None:
|
|
56
|
+
"""Returns an Anthropic client if a key is configured and the extra is installed, else None."""
|
|
57
|
+
if not os.environ.get("ANTHROPIC_API_KEY"):
|
|
58
|
+
return None
|
|
59
|
+
try:
|
|
60
|
+
import anthropic
|
|
61
|
+
except ImportError:
|
|
62
|
+
return None
|
|
63
|
+
return anthropic.Anthropic()
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def judge_answer(client: Any, *, model: str, criteria: str, answer: str) -> tuple[float, str]:
|
|
67
|
+
prompt = build_judge_prompt(criteria=criteria, answer=answer)
|
|
68
|
+
response = client.messages.create(
|
|
69
|
+
model=model,
|
|
70
|
+
max_tokens=256,
|
|
71
|
+
temperature=0.0,
|
|
72
|
+
messages=[{"role": "user", "content": prompt}],
|
|
73
|
+
)
|
|
74
|
+
text = response.content[0].text
|
|
75
|
+
return parse_judge_response(text)
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Connect to an MCP server under test (stdio or HTTP) and call its tools.
|
|
2
|
+
|
|
3
|
+
This is what makes the tool protocol-generic: it speaks the MCP client side
|
|
4
|
+
directly, rather than a specific vendor's SDK, so it works against any
|
|
5
|
+
compliant MCP server regardless of the language or backend it's written in.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import AsyncIterator
|
|
11
|
+
from contextlib import asynccontextmanager
|
|
12
|
+
|
|
13
|
+
from mcp.client.session import ClientSession
|
|
14
|
+
from mcp.client.stdio import StdioServerParameters, stdio_client
|
|
15
|
+
from mcp.client.streamable_http import streamable_http_client
|
|
16
|
+
from mcp.types import CallToolResult, TextContent
|
|
17
|
+
|
|
18
|
+
from mcp_eval_gate.models import ServerTarget, ToolCallOutcome
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@asynccontextmanager
|
|
22
|
+
async def connect(target: ServerTarget) -> AsyncIterator[ClientSession]:
|
|
23
|
+
if target.command:
|
|
24
|
+
params = StdioServerParameters(command=target.command, args=list(target.args), env=target.env)
|
|
25
|
+
async with stdio_client(params) as (read, write):
|
|
26
|
+
async with ClientSession(read, write) as session:
|
|
27
|
+
await session.initialize()
|
|
28
|
+
yield session
|
|
29
|
+
return
|
|
30
|
+
|
|
31
|
+
async with streamable_http_client(target.url) as (read, write, _get_session_id):
|
|
32
|
+
async with ClientSession(read, write) as session:
|
|
33
|
+
await session.initialize()
|
|
34
|
+
yield session
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
async def call_tool(session: ClientSession, tool_name: str, arguments: dict[str, object]) -> ToolCallOutcome:
|
|
38
|
+
result = await session.call_tool(tool_name, arguments)
|
|
39
|
+
return parse_call_tool_result(result)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def parse_call_tool_result(result: CallToolResult) -> ToolCallOutcome:
|
|
43
|
+
text = "".join(block.text for block in result.content if isinstance(block, TextContent))
|
|
44
|
+
return ToolCallOutcome(text=text, structured=result.structured_content, is_error=bool(result.is_error))
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""MCP server exposing mcp-eval-gate as a tool callable from Claude Code, Cursor, etc.
|
|
2
|
+
|
|
3
|
+
Lets an engineer ask "did my last change break this MCP server's tools?" from
|
|
4
|
+
inside their agent harness, before committing, not just as a CI-only check.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from mcp.server.mcpserver import MCPServer
|
|
12
|
+
|
|
13
|
+
from mcp_eval_gate.baseline import diff_against_baseline, load_baseline, save_baseline
|
|
14
|
+
from mcp_eval_gate.golden_set import GoldenSetError, load_golden_set
|
|
15
|
+
from mcp_eval_gate.judge import build_default_anthropic_client
|
|
16
|
+
from mcp_eval_gate.runner import run_golden_set
|
|
17
|
+
|
|
18
|
+
mcp = MCPServer("mcp-eval-gate")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@mcp.tool()
|
|
22
|
+
async def run_eval_gate(
|
|
23
|
+
config_path: str,
|
|
24
|
+
baseline_path: str = "baseline.json",
|
|
25
|
+
update_baseline: bool = False,
|
|
26
|
+
) -> dict:
|
|
27
|
+
"""Run a golden-set eval against a live MCP server and report regressions vs. baseline.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
config_path: Path to the golden_set.yaml file.
|
|
31
|
+
baseline_path: Path to the committed baseline.json (created by a prior run).
|
|
32
|
+
update_baseline: If true, overwrite the baseline with this run's scores instead of gating.
|
|
33
|
+
"""
|
|
34
|
+
try:
|
|
35
|
+
config = load_golden_set(Path(config_path))
|
|
36
|
+
except GoldenSetError as exc:
|
|
37
|
+
return {"ok": False, "error": str(exc)}
|
|
38
|
+
|
|
39
|
+
anthropic_client = build_default_anthropic_client()
|
|
40
|
+
results = await run_golden_set(config, anthropic_client=anthropic_client)
|
|
41
|
+
|
|
42
|
+
if update_baseline:
|
|
43
|
+
save_baseline(Path(baseline_path), results)
|
|
44
|
+
return {"ok": True, "baseline_updated": True, "cases": _serialize_results(results)}
|
|
45
|
+
|
|
46
|
+
baseline = load_baseline(Path(baseline_path))
|
|
47
|
+
regressions = diff_against_baseline(results, baseline)
|
|
48
|
+
any_failed = any(not r.passed for r in results)
|
|
49
|
+
|
|
50
|
+
return {
|
|
51
|
+
"ok": not (any_failed or regressions),
|
|
52
|
+
"cases": _serialize_results(results),
|
|
53
|
+
"regressions": [
|
|
54
|
+
{
|
|
55
|
+
"case_id": r.case_id,
|
|
56
|
+
"baseline_score": r.baseline_score,
|
|
57
|
+
"current_score": r.current_score,
|
|
58
|
+
"detail": r.detail,
|
|
59
|
+
}
|
|
60
|
+
for r in regressions
|
|
61
|
+
],
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _serialize_results(results) -> list[dict]:
|
|
66
|
+
return [{"case_id": r.case_id, "score": r.score, "passed": r.passed, "detail": r.detail} for r in results]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def main() -> None:
|
|
70
|
+
mcp.run()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
if __name__ == "__main__":
|
|
74
|
+
main()
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Core data types shared across golden-set loading, MCP calls, scoring, and baseline diffing."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class MatchType(StrEnum):
|
|
10
|
+
EXACT = "exact"
|
|
11
|
+
CONTAINS = "contains"
|
|
12
|
+
JUDGE = "judge"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class ServerTarget:
|
|
17
|
+
"""How to reach the MCP server under test: a stdio subprocess or a URL."""
|
|
18
|
+
|
|
19
|
+
command: str | None = None
|
|
20
|
+
args: tuple[str, ...] = ()
|
|
21
|
+
env: dict[str, str] | None = None
|
|
22
|
+
url: str | None = None
|
|
23
|
+
|
|
24
|
+
def __post_init__(self) -> None:
|
|
25
|
+
if bool(self.command) == bool(self.url):
|
|
26
|
+
raise ValueError("ServerTarget needs exactly one of `command` (stdio) or `url` (HTTP)")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class GoldenCase:
|
|
31
|
+
id: str
|
|
32
|
+
tool_name: str
|
|
33
|
+
tool_args: dict[str, object] = field(default_factory=dict)
|
|
34
|
+
match_type: MatchType = MatchType.CONTAINS
|
|
35
|
+
expected_output: str | None = None
|
|
36
|
+
judge_criteria: str | None = None
|
|
37
|
+
min_judge_score: float = 0.8
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class ToolCallOutcome:
|
|
42
|
+
text: str
|
|
43
|
+
structured: dict | None
|
|
44
|
+
is_error: bool
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class CaseResult:
|
|
49
|
+
case_id: str
|
|
50
|
+
score: float
|
|
51
|
+
passed: bool
|
|
52
|
+
detail: str
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass(frozen=True)
|
|
56
|
+
class Regression:
|
|
57
|
+
case_id: str
|
|
58
|
+
baseline_score: float
|
|
59
|
+
current_score: float
|
|
60
|
+
detail: str
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass(frozen=True)
|
|
64
|
+
class GoldenSetConfig:
|
|
65
|
+
server: ServerTarget
|
|
66
|
+
cases: tuple[GoldenCase, ...]
|
|
67
|
+
judge_model: str = "claude-sonnet-4-5"
|
|
File without changes
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Console + HTML rendering for a run, and the pass/fail decision that gates CI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from rich.console import Console
|
|
8
|
+
from rich.table import Table
|
|
9
|
+
|
|
10
|
+
from mcp_eval_gate.models import CaseResult, Regression
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def exit_code_for(results: list[CaseResult], regressions: list[Regression]) -> int:
|
|
14
|
+
any_failed = any(not r.passed for r in results)
|
|
15
|
+
return 1 if any_failed or regressions else 0
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def print_console_report(results: list[CaseResult], regressions: list[Regression]) -> None:
|
|
19
|
+
console = Console()
|
|
20
|
+
table = Table(title="mcp-eval-gate results")
|
|
21
|
+
table.add_column("Case")
|
|
22
|
+
table.add_column("Score", justify="right")
|
|
23
|
+
table.add_column("Status")
|
|
24
|
+
table.add_column("Detail")
|
|
25
|
+
|
|
26
|
+
regressed_ids = {r.case_id for r in regressions}
|
|
27
|
+
for result in results:
|
|
28
|
+
status = "[green]PASS[/green]" if result.passed else "[red]FAIL[/red]"
|
|
29
|
+
if result.case_id in regressed_ids:
|
|
30
|
+
status += " [yellow](regression)[/yellow]"
|
|
31
|
+
table.add_row(result.case_id, f"{result.score:.2f}", status, result.detail)
|
|
32
|
+
|
|
33
|
+
console.print(table)
|
|
34
|
+
|
|
35
|
+
if regressions:
|
|
36
|
+
console.print(f"\n[bold red]{len(regressions)} regression(s) against baseline:[/bold red]")
|
|
37
|
+
for reg in regressions:
|
|
38
|
+
console.print(f" - {reg.case_id}: {reg.baseline_score:.2f} -> {reg.current_score:.2f} ({reg.detail})")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def write_html_report(results: list[CaseResult], regressions: list[Regression], path: Path) -> None:
|
|
42
|
+
regressed_ids = {r.case_id for r in regressions}
|
|
43
|
+
rows = "\n".join(
|
|
44
|
+
f'<tr class="{"fail" if not r.passed else "pass"}">'
|
|
45
|
+
f"<td>{r.case_id}</td><td>{r.score:.2f}</td>"
|
|
46
|
+
f"<td>{'PASS' if r.passed else 'FAIL'}{' (regression)' if r.case_id in regressed_ids else ''}</td>"
|
|
47
|
+
f"<td>{r.detail}</td></tr>"
|
|
48
|
+
for r in results
|
|
49
|
+
)
|
|
50
|
+
html = f"""<!doctype html>
|
|
51
|
+
<html><head><meta charset="utf-8"><title>mcp-eval-gate report</title>
|
|
52
|
+
<style>
|
|
53
|
+
body {{ font-family: -apple-system, sans-serif; margin: 2rem; }}
|
|
54
|
+
table {{ border-collapse: collapse; width: 100%; }}
|
|
55
|
+
td, th {{ border: 1px solid #ddd; padding: 8px; text-align: left; }}
|
|
56
|
+
tr.fail {{ background: #fdecea; }}
|
|
57
|
+
tr.pass {{ background: #eafaf1; }}
|
|
58
|
+
</style></head>
|
|
59
|
+
<body>
|
|
60
|
+
<h1>mcp-eval-gate report</h1>
|
|
61
|
+
<table>
|
|
62
|
+
<tr><th>Case</th><th>Score</th><th>Status</th><th>Detail</th></tr>
|
|
63
|
+
{rows}
|
|
64
|
+
</table>
|
|
65
|
+
</body></html>
|
|
66
|
+
"""
|
|
67
|
+
path.write_text(html)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Wires golden-set cases to a live MCP server and the scoring functions."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from mcp_eval_gate.judge import judge_answer
|
|
8
|
+
from mcp_eval_gate.mcp_client import call_tool, connect
|
|
9
|
+
from mcp_eval_gate.models import CaseResult, GoldenCase, GoldenSetConfig, MatchType
|
|
10
|
+
from mcp_eval_gate.scoring import score_case, score_judge_case
|
|
11
|
+
from mcp_eval_gate.text_diff import preview
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
async def run_golden_set(config: GoldenSetConfig, *, anthropic_client: Any | None = None) -> list[CaseResult]:
|
|
15
|
+
async with connect(config.server) as session:
|
|
16
|
+
results = []
|
|
17
|
+
for case in config.cases:
|
|
18
|
+
results.append(await run_case(case, session=session, config=config, anthropic_client=anthropic_client))
|
|
19
|
+
return results
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
async def run_case(
|
|
23
|
+
case: GoldenCase, *, session: Any, config: GoldenSetConfig, anthropic_client: Any | None
|
|
24
|
+
) -> CaseResult:
|
|
25
|
+
outcome = await call_tool(session, case.tool_name, case.tool_args)
|
|
26
|
+
|
|
27
|
+
if case.match_type != MatchType.JUDGE:
|
|
28
|
+
return score_case(case, outcome)
|
|
29
|
+
|
|
30
|
+
if outcome.is_error:
|
|
31
|
+
return CaseResult(
|
|
32
|
+
case.id, score=0.0, passed=False, detail=f"tool call returned an error: {preview(outcome.text)}"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
if anthropic_client is None:
|
|
36
|
+
return CaseResult(
|
|
37
|
+
case.id,
|
|
38
|
+
score=0.0,
|
|
39
|
+
passed=False,
|
|
40
|
+
detail="match_type=judge needs [judge] extras installed and ANTHROPIC_API_KEY set",
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
score, reasoning = judge_answer(
|
|
44
|
+
anthropic_client, model=config.judge_model, criteria=case.judge_criteria or "", answer=outcome.text
|
|
45
|
+
)
|
|
46
|
+
return score_judge_case(case, judge_score=score, reasoning=reasoning)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Deterministic scoring for MCP tool-call outcomes.
|
|
2
|
+
|
|
3
|
+
No MCP calls happen here, this module only compares an already-captured
|
|
4
|
+
ToolCallOutcome against the expectation declared in a golden-set case. Keeping
|
|
5
|
+
it pure is what makes it unit-testable without a live server.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from mcp_eval_gate.models import CaseResult, GoldenCase, MatchType, ToolCallOutcome
|
|
11
|
+
from mcp_eval_gate.text_diff import describe_difference, preview
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def score_case(case: GoldenCase, outcome: ToolCallOutcome) -> CaseResult:
|
|
15
|
+
if outcome.is_error:
|
|
16
|
+
return CaseResult(
|
|
17
|
+
case.id, score=0.0, passed=False, detail=f"tool call returned an error: {preview(outcome.text)}"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
if case.match_type == MatchType.EXACT:
|
|
21
|
+
return _score_exact(case, outcome)
|
|
22
|
+
if case.match_type == MatchType.CONTAINS:
|
|
23
|
+
return _score_contains(case, outcome)
|
|
24
|
+
|
|
25
|
+
raise ValueError(f"case '{case.id}' has match_type=judge, score it via score_judge_case after a judge call")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _score_exact(case: GoldenCase, outcome: ToolCallOutcome) -> CaseResult:
|
|
29
|
+
matched = outcome.text.strip() == (case.expected_output or "").strip()
|
|
30
|
+
expected = case.expected_output or ""
|
|
31
|
+
detail = "exact match" if matched else describe_difference(expected.strip(), outcome.text.strip())
|
|
32
|
+
return CaseResult(case.id, score=1.0 if matched else 0.0, passed=matched, detail=detail)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _score_contains(case: GoldenCase, outcome: ToolCallOutcome) -> CaseResult:
|
|
36
|
+
matched = (case.expected_output or "") in outcome.text
|
|
37
|
+
detail = (
|
|
38
|
+
"expected text found" if matched else f"expected output to contain {preview(case.expected_output or '')!r}"
|
|
39
|
+
)
|
|
40
|
+
return CaseResult(case.id, score=1.0 if matched else 0.0, passed=matched, detail=detail)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def score_judge_case(case: GoldenCase, *, judge_score: float, reasoning: str) -> CaseResult:
|
|
44
|
+
passed = judge_score >= case.min_judge_score
|
|
45
|
+
return CaseResult(case.id, score=judge_score, passed=passed, detail=reasoning)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Readable descriptions of how two strings differ, for failure details in reports."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
|
|
7
|
+
DEFAULT_CONTEXT = 20
|
|
8
|
+
DEFAULT_PREVIEW_LIMIT = 120
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def describe_difference(expected: str, actual: str, context: int = DEFAULT_CONTEXT) -> str:
|
|
12
|
+
index = _first_difference(expected, actual)
|
|
13
|
+
return (
|
|
14
|
+
f"first difference at char {index} "
|
|
15
|
+
f"(expected {len(expected)} chars, got {len(actual)}): "
|
|
16
|
+
f"expected {_snippet(expected, index, context)!r}, got {_snippet(actual, index, context)!r}"
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def preview(text: str, limit: int = DEFAULT_PREVIEW_LIMIT) -> str:
|
|
21
|
+
if len(text) <= limit:
|
|
22
|
+
return text
|
|
23
|
+
return f"{text[:limit]}… (+{len(text) - limit} more chars)"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _first_difference(expected: str, actual: str) -> int:
|
|
27
|
+
return len(os.path.commonprefix([expected, actual]))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _snippet(text: str, index: int, context: int) -> str:
|
|
31
|
+
start = max(0, index - context)
|
|
32
|
+
end = min(len(text), index + context)
|
|
33
|
+
prefix = "…" if start > 0 else ""
|
|
34
|
+
suffix = "…" if end < len(text) else ""
|
|
35
|
+
return f"{prefix}{text[start:end]}{suffix}"
|