evalrun 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrun-0.4.2/LICENSE +21 -0
- {evalrun-0.4.0/evalrun.egg-info → evalrun-0.4.2}/PKG-INFO +148 -10
- evalrun-0.4.0/PKG-INFO → evalrun-0.4.2/README.md +137 -23
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/agent.py +1 -1
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/agent.py +1 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/planner.py +1 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/support/triage_agent.py +1 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/agent.py +1 -2
- evalrun-0.4.2/cli/__init__.py +6 -0
- evalrun-0.4.2/cli/doctor.py +150 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/html_reporter.py +198 -61
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/main.py +273 -2
- evalrun-0.4.2/cli/validator.py +199 -0
- evalrun-0.4.0/README.md → evalrun-0.4.2/evalrun.egg-info/PKG-INFO +161 -7
- {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/SOURCES.txt +7 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/requires.txt +11 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/engine.py +1 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/base_llm.py +2 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/base.py +4 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/runner.py +2 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/openai_compatible.py +13 -2
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/__init__.py +4 -5
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/client.py +14 -2
- evalrun-0.4.2/framework/observability.py +26 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/sdk.py +48 -5
- {evalrun-0.4.0 → evalrun-0.4.2}/pyproject.toml +14 -3
- evalrun-0.4.2/tests/test_canonical_evidence.py +66 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_cli.py +70 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_custom_profiles.py +1 -1
- evalrun-0.4.2/tests/test_doctor.py +67 -0
- evalrun-0.4.2/tests/test_html_reporter.py +108 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_openai_compatible.py +1 -0
- evalrun-0.4.2/tests/test_sdk.py +168 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_ui.py +11 -2
- evalrun-0.4.2/tests/test_validator.py +205 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/ui/server.py +66 -21
- evalrun-0.4.0/LICENSE +0 -0
- evalrun-0.4.0/cli/__init__.py +0 -6
- evalrun-0.4.0/tests/test_sdk.py +0 -88
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/budget_auditor.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/parser.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/prompts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/schema.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/base.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/prompts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/prompts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/support/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/prompts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/session.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/demo.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/formatter.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/progress.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/cli/resolver.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/dependency_links.txt +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/entry_points.txt +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/top_level.txt +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/adapters.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/contracts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/suite.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/base.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/dimensions.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/adaptability.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/constraint.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/information_accuracy.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/personalization.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/planning.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/support.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/adaptability.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/constraint.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/information_accuracy.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/personalization.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/planning.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/testing.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/exceptions.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/base.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/factory.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/gemini.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/mock.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/openai.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/constraints.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/revision_summary.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/server.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/memory/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/memory/base.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/models.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parser.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/frontmatter.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/mapper.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/markdown.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/transformers.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/registry.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/support.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/travel.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/comparator.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/loader.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/utils.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/__init__.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/base.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/extractor.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/local.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/models.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/pipeline.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/prompts.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/utils.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/setup.cfg +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_adapters.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_agent_collaboration.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_config_workflow.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_engine.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_evaluators.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_generic_core.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_independent_auditor.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_mcp_agent_integration.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_mcp_constraints.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_parser.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_reflection_agent.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_regression.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_research_agent.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_revision_summary.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_runner.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_runner_v3.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_session_memory.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_smoke_cli.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_support_domain.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_usability_validation.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_utils.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_verification.py +0 -0
- {evalrun-0.4.0 → evalrun-0.4.2}/ui/__init__.py +0 -0
evalrun-0.4.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Shivam Bhardwaj
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: evalrun
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -10,8 +10,16 @@ Requires-Dist: requests>=2.28.0
|
|
|
10
10
|
Requires-Dist: pydantic>=2.0.0
|
|
11
11
|
Requires-Dist: python-frontmatter>=1.0.0
|
|
12
12
|
Requires-Dist: PyYAML>=6.0
|
|
13
|
-
Requires-Dist:
|
|
14
|
-
|
|
13
|
+
Requires-Dist: markdown-it-py>=3.0.0
|
|
14
|
+
Provides-Extra: observability
|
|
15
|
+
Requires-Dist: langfuse>=2.0.0; extra == "observability"
|
|
16
|
+
Provides-Extra: mcp
|
|
17
|
+
Requires-Dist: fastmcp>=0.1.0; extra == "mcp"
|
|
18
|
+
Requires-Dist: mcp<2,>=1.27; extra == "mcp"
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Requires-Dist: langfuse>=2.0.0; extra == "all"
|
|
21
|
+
Requires-Dist: fastmcp>=0.1.0; extra == "all"
|
|
22
|
+
Requires-Dist: mcp<2,>=1.27; extra == "all"
|
|
15
23
|
Dynamic: license-file
|
|
16
24
|
|
|
17
25
|
# evalrun
|
|
@@ -43,12 +51,30 @@ The toolkit works with hosted APIs and models running on your own computer. It d
|
|
|
43
51
|
|
|
44
52
|
### 1. Installation
|
|
45
53
|
|
|
54
|
+
Install the released package from PyPI:
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
python -m pip install evalrun
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
The package installs the `evalrun` command and all runtime dependencies. You do
|
|
61
|
+
not need to clone this repository to use the toolkit. To contribute or run the
|
|
62
|
+
latest unreleased source instead, clone the repository and use an editable
|
|
63
|
+
install:
|
|
64
|
+
|
|
46
65
|
```bash
|
|
47
66
|
git clone https://github.com/imshivamb/agent-eval-platform.git
|
|
48
67
|
cd agent-eval-platform
|
|
49
68
|
pip install -e .
|
|
50
69
|
```
|
|
51
70
|
|
|
71
|
+
Optional integrations are deliberately not required for the core CLI and SDK:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install "evalrun[observability]" # optional Langfuse tracing
|
|
75
|
+
pip install "evalrun[mcp]" # optional MCP validation tools
|
|
76
|
+
```
|
|
77
|
+
|
|
52
78
|
If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
|
|
53
79
|
|
|
54
80
|
```bash
|
|
@@ -66,7 +92,38 @@ evalrun demo
|
|
|
66
92
|
open results/demo/report.html # macOS
|
|
67
93
|
```
|
|
68
94
|
|
|
69
|
-
|
|
95
|
+
The offline demo is the safest way to confirm that installation works. It makes
|
|
96
|
+
no model requests, needs no API key, and executes only EvalRun's built-in demo
|
|
97
|
+
agent.
|
|
98
|
+
|
|
99
|
+
### 3. Create and check your own evaluation workspace
|
|
100
|
+
|
|
101
|
+
Create starter files in a new folder:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
evalrun init my-evaluation
|
|
105
|
+
cd my-evaluation
|
|
106
|
+
evalrun validate scenario.md
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Edit `scenario.md` to describe the user request, hard constraints, expected
|
|
110
|
+
behavior, scoring dimensions, pass criteria, and failure conditions. The
|
|
111
|
+
scenario is ordinary Markdown; you do not need to change EvalRun's source code.
|
|
112
|
+
Use `evalrun validate` before spending money on a model run. It reports the
|
|
113
|
+
exact missing heading or unsupported dimension and returns exit code `2` when
|
|
114
|
+
the file needs fixing.
|
|
115
|
+
|
|
116
|
+
Run environment diagnostics at any time with:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
evalrun doctor
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
`doctor` does not require an API key. Missing keys and unavailable model servers
|
|
123
|
+
are reported as information, while invalid Python or unwritable output paths are
|
|
124
|
+
reported as failures.
|
|
125
|
+
|
|
126
|
+
### 4. Single Scenario Run (Hosted Model)
|
|
70
127
|
|
|
71
128
|
```bash
|
|
72
129
|
export OPENAI_API_KEY="sk-proj-..."
|
|
@@ -92,6 +149,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
|
|
|
92
149
|
|
|
93
150
|
The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
|
|
94
151
|
|
|
152
|
+
For a target and judge at different providers, configure both explicitly:
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
evalrun run \
|
|
156
|
+
--scenario path/to/scenario.md \
|
|
157
|
+
--agent my_agent:MyAgent \
|
|
158
|
+
--model gemini-3.7-flash \
|
|
159
|
+
--base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
|
|
160
|
+
--api-key "$GEMINI_API_KEY" \
|
|
161
|
+
--judge-model stealth/ox-alpha \
|
|
162
|
+
--judge-base-url https://openrouter.ai/api/v1 \
|
|
163
|
+
--judge-api-key "$OPENROUTER_API_KEY" \
|
|
164
|
+
--output results/mixed-provider-run
|
|
165
|
+
```
|
|
166
|
+
|
|
95
167
|
For example, a Gemini run can be written as:
|
|
96
168
|
|
|
97
169
|
```bash
|
|
@@ -107,7 +179,7 @@ evalrun run \
|
|
|
107
179
|
--output results/gemini-run
|
|
108
180
|
```
|
|
109
181
|
|
|
110
|
-
###
|
|
182
|
+
### 5. Local Model Server Run (vLLM / Ollama)
|
|
111
183
|
|
|
112
184
|
```bash
|
|
113
185
|
# Users host their own local OpenAI-compatible server at http://localhost:8000/v1
|
|
@@ -121,7 +193,7 @@ evalrun run \
|
|
|
121
193
|
--output results/local-run
|
|
122
194
|
```
|
|
123
195
|
|
|
124
|
-
###
|
|
196
|
+
### 6. Guided Local Web UI
|
|
125
197
|
|
|
126
198
|
Launch the zero-dependency local web interface:
|
|
127
199
|
|
|
@@ -131,6 +203,17 @@ evalrun ui --port 8501
|
|
|
131
203
|
|
|
132
204
|
Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
|
|
133
205
|
|
|
206
|
+
The UI is a local convenience layer. It supports built-in agents and workspace
|
|
207
|
+
scenario paths. API keys are held in browser memory for the current request and
|
|
208
|
+
are not saved by EvalRun. Do not expose this server to the public internet and
|
|
209
|
+
do not paste untrusted Python code into an agent field. Complex or third-party
|
|
210
|
+
agents should be run through the CLI or SDK in the environment where their code
|
|
211
|
+
and dependencies are installed.
|
|
212
|
+
|
|
213
|
+
The UI intentionally permits only the built-in travel agent. Custom Python,
|
|
214
|
+
HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
|
|
215
|
+
agent specifications from a browser would allow local code execution.
|
|
216
|
+
|
|
134
217
|
---
|
|
135
218
|
|
|
136
219
|
## 🐍 Python SDK Usage
|
|
@@ -138,7 +221,7 @@ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scen
|
|
|
138
221
|
Integrate `evalrun` programmatically into Python automation pipelines:
|
|
139
222
|
|
|
140
223
|
```python
|
|
141
|
-
from framework
|
|
224
|
+
from framework import evaluate, compare
|
|
142
225
|
|
|
143
226
|
# 1. Execute Benchmark Evaluation
|
|
144
227
|
results = evaluate(
|
|
@@ -153,7 +236,7 @@ results = evaluate(
|
|
|
153
236
|
report = compare(
|
|
154
237
|
candidate_results=results,
|
|
155
238
|
baseline="results/run-001",
|
|
156
|
-
|
|
239
|
+
max_regression=5.0,
|
|
157
240
|
)
|
|
158
241
|
|
|
159
242
|
if report.release_blocked:
|
|
@@ -192,13 +275,26 @@ There is no central service to deploy for the current product. The recommended p
|
|
|
192
275
|
|
|
193
276
|
1. Publish the repository on GitHub.
|
|
194
277
|
2. Add tagged releases and a clear quick-start guide.
|
|
195
|
-
3. Users install
|
|
196
|
-
4.
|
|
278
|
+
3. Users install the released package with `pip install evalrun`.
|
|
279
|
+
4. Contributors install the repository with `pip install -e .` when working on source changes.
|
|
197
280
|
5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
|
|
198
281
|
6. Users provide their own hosted-model API keys or run their own local model server.
|
|
199
282
|
|
|
200
283
|
The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
|
|
201
284
|
|
|
285
|
+
### TestPyPI (maintainers only)
|
|
286
|
+
|
|
287
|
+
Before a release, maintainers may validate a candidate package from TestPyPI:
|
|
288
|
+
|
|
289
|
+
```bash
|
|
290
|
+
python -m pip install \
|
|
291
|
+
--index-url https://test.pypi.org/simple/ \
|
|
292
|
+
--extra-index-url https://pypi.org/simple/ \
|
|
293
|
+
evalrun
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
End users should use the normal PyPI install shown above.
|
|
297
|
+
|
|
202
298
|
### Credential promise
|
|
203
299
|
|
|
204
300
|
EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
|
|
@@ -250,6 +346,48 @@ evalrun run \
|
|
|
250
346
|
|
|
251
347
|
Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
|
|
252
348
|
|
|
349
|
+
Validate a suite before running it:
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
for scenario in evals/scenarios/my-domain/*.md; do
|
|
353
|
+
evalrun validate "$scenario" || exit 2
|
|
354
|
+
done
|
|
355
|
+
```
|
|
356
|
+
|
|
357
|
+
## Connect your own agent
|
|
358
|
+
|
|
359
|
+
EvalRun does not require agents to use a particular framework. The `--agent`
|
|
360
|
+
value selects one of three adapters:
|
|
361
|
+
|
|
362
|
+
| Agent form | Example | When to use |
|
|
363
|
+
| --- | --- | --- |
|
|
364
|
+
| Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
|
|
365
|
+
| HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
|
|
366
|
+
| CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
|
|
367
|
+
|
|
368
|
+
Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
|
|
369
|
+
must expose the adapter contract described in
|
|
370
|
+
[`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
|
|
371
|
+
complete syntax and examples.
|
|
372
|
+
|
|
373
|
+
For repeatable runs, put the same values in a JSON or TOML configuration file:
|
|
374
|
+
|
|
375
|
+
```bash
|
|
376
|
+
evalrun run --config evalrun.json
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
Command-line flags override values from the configuration file. Keep API keys
|
|
380
|
+
in environment variables and reference them from your shell; never commit them
|
|
381
|
+
to the config file.
|
|
382
|
+
|
|
383
|
+
## Troubleshooting
|
|
384
|
+
|
|
385
|
+
- `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
|
|
386
|
+
- `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
|
|
387
|
+
- Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
|
|
388
|
+
- A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
|
|
389
|
+
- `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
|
|
390
|
+
|
|
253
391
|
## 🏗️ Architecture & Documentation
|
|
254
392
|
|
|
255
393
|
For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
|
|
@@ -1,19 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: evalrun
|
|
3
|
-
Version: 0.4.0
|
|
4
|
-
Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
|
|
5
|
-
Requires-Python: >=3.10
|
|
6
|
-
Description-Content-Type: text/markdown
|
|
7
|
-
License-File: LICENSE
|
|
8
|
-
Requires-Dist: openai>=1.0.0
|
|
9
|
-
Requires-Dist: requests>=2.28.0
|
|
10
|
-
Requires-Dist: pydantic>=2.0.0
|
|
11
|
-
Requires-Dist: python-frontmatter>=1.0.0
|
|
12
|
-
Requires-Dist: PyYAML>=6.0
|
|
13
|
-
Requires-Dist: langfuse>=2.0.0
|
|
14
|
-
Requires-Dist: fastmcp>=0.1.0
|
|
15
|
-
Dynamic: license-file
|
|
16
|
-
|
|
17
1
|
# evalrun
|
|
18
2
|
|
|
19
3
|
> **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
|
|
@@ -43,12 +27,30 @@ The toolkit works with hosted APIs and models running on your own computer. It d
|
|
|
43
27
|
|
|
44
28
|
### 1. Installation
|
|
45
29
|
|
|
30
|
+
Install the released package from PyPI:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
python -m pip install evalrun
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The package installs the `evalrun` command and all runtime dependencies. You do
|
|
37
|
+
not need to clone this repository to use the toolkit. To contribute or run the
|
|
38
|
+
latest unreleased source instead, clone the repository and use an editable
|
|
39
|
+
install:
|
|
40
|
+
|
|
46
41
|
```bash
|
|
47
42
|
git clone https://github.com/imshivamb/agent-eval-platform.git
|
|
48
43
|
cd agent-eval-platform
|
|
49
44
|
pip install -e .
|
|
50
45
|
```
|
|
51
46
|
|
|
47
|
+
Optional integrations are deliberately not required for the core CLI and SDK:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install "evalrun[observability]" # optional Langfuse tracing
|
|
51
|
+
pip install "evalrun[mcp]" # optional MCP validation tools
|
|
52
|
+
```
|
|
53
|
+
|
|
52
54
|
If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
|
|
53
55
|
|
|
54
56
|
```bash
|
|
@@ -66,7 +68,38 @@ evalrun demo
|
|
|
66
68
|
open results/demo/report.html # macOS
|
|
67
69
|
```
|
|
68
70
|
|
|
69
|
-
|
|
71
|
+
The offline demo is the safest way to confirm that installation works. It makes
|
|
72
|
+
no model requests, needs no API key, and executes only EvalRun's built-in demo
|
|
73
|
+
agent.
|
|
74
|
+
|
|
75
|
+
### 3. Create and check your own evaluation workspace
|
|
76
|
+
|
|
77
|
+
Create starter files in a new folder:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
evalrun init my-evaluation
|
|
81
|
+
cd my-evaluation
|
|
82
|
+
evalrun validate scenario.md
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Edit `scenario.md` to describe the user request, hard constraints, expected
|
|
86
|
+
behavior, scoring dimensions, pass criteria, and failure conditions. The
|
|
87
|
+
scenario is ordinary Markdown; you do not need to change EvalRun's source code.
|
|
88
|
+
Use `evalrun validate` before spending money on a model run. It reports the
|
|
89
|
+
exact missing heading or unsupported dimension and returns exit code `2` when
|
|
90
|
+
the file needs fixing.
|
|
91
|
+
|
|
92
|
+
Run environment diagnostics at any time with:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
evalrun doctor
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`doctor` does not require an API key. Missing keys and unavailable model servers
|
|
99
|
+
are reported as information, while invalid Python or unwritable output paths are
|
|
100
|
+
reported as failures.
|
|
101
|
+
|
|
102
|
+
### 4. Single Scenario Run (Hosted Model)
|
|
70
103
|
|
|
71
104
|
```bash
|
|
72
105
|
export OPENAI_API_KEY="sk-proj-..."
|
|
@@ -92,6 +125,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
|
|
|
92
125
|
|
|
93
126
|
The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
|
|
94
127
|
|
|
128
|
+
For a target and judge at different providers, configure both explicitly:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
evalrun run \
|
|
132
|
+
--scenario path/to/scenario.md \
|
|
133
|
+
--agent my_agent:MyAgent \
|
|
134
|
+
--model gemini-3.7-flash \
|
|
135
|
+
--base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
|
|
136
|
+
--api-key "$GEMINI_API_KEY" \
|
|
137
|
+
--judge-model stealth/ox-alpha \
|
|
138
|
+
--judge-base-url https://openrouter.ai/api/v1 \
|
|
139
|
+
--judge-api-key "$OPENROUTER_API_KEY" \
|
|
140
|
+
--output results/mixed-provider-run
|
|
141
|
+
```
|
|
142
|
+
|
|
95
143
|
For example, a Gemini run can be written as:
|
|
96
144
|
|
|
97
145
|
```bash
|
|
@@ -107,7 +155,7 @@ evalrun run \
|
|
|
107
155
|
--output results/gemini-run
|
|
108
156
|
```
|
|
109
157
|
|
|
110
|
-
###
|
|
158
|
+
### 5. Local Model Server Run (vLLM / Ollama)
|
|
111
159
|
|
|
112
160
|
```bash
|
|
113
161
|
# Users host their own local OpenAI-compatible server at http://localhost:8000/v1
|
|
@@ -121,7 +169,7 @@ evalrun run \
|
|
|
121
169
|
--output results/local-run
|
|
122
170
|
```
|
|
123
171
|
|
|
124
|
-
###
|
|
172
|
+
### 6. Guided Local Web UI
|
|
125
173
|
|
|
126
174
|
Launch the zero-dependency local web interface:
|
|
127
175
|
|
|
@@ -131,6 +179,17 @@ evalrun ui --port 8501
|
|
|
131
179
|
|
|
132
180
|
Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
|
|
133
181
|
|
|
182
|
+
The UI is a local convenience layer. It supports built-in agents and workspace
|
|
183
|
+
scenario paths. API keys are held in browser memory for the current request and
|
|
184
|
+
are not saved by EvalRun. Do not expose this server to the public internet and
|
|
185
|
+
do not paste untrusted Python code into an agent field. Complex or third-party
|
|
186
|
+
agents should be run through the CLI or SDK in the environment where their code
|
|
187
|
+
and dependencies are installed.
|
|
188
|
+
|
|
189
|
+
The UI intentionally permits only the built-in travel agent. Custom Python,
|
|
190
|
+
HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
|
|
191
|
+
agent specifications from a browser would allow local code execution.
|
|
192
|
+
|
|
134
193
|
---
|
|
135
194
|
|
|
136
195
|
## 🐍 Python SDK Usage
|
|
@@ -138,7 +197,7 @@ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scen
|
|
|
138
197
|
Integrate `evalrun` programmatically into Python automation pipelines:
|
|
139
198
|
|
|
140
199
|
```python
|
|
141
|
-
from framework
|
|
200
|
+
from framework import evaluate, compare
|
|
142
201
|
|
|
143
202
|
# 1. Execute Benchmark Evaluation
|
|
144
203
|
results = evaluate(
|
|
@@ -153,7 +212,7 @@ results = evaluate(
|
|
|
153
212
|
report = compare(
|
|
154
213
|
candidate_results=results,
|
|
155
214
|
baseline="results/run-001",
|
|
156
|
-
|
|
215
|
+
max_regression=5.0,
|
|
157
216
|
)
|
|
158
217
|
|
|
159
218
|
if report.release_blocked:
|
|
@@ -192,13 +251,26 @@ There is no central service to deploy for the current product. The recommended p
|
|
|
192
251
|
|
|
193
252
|
1. Publish the repository on GitHub.
|
|
194
253
|
2. Add tagged releases and a clear quick-start guide.
|
|
195
|
-
3. Users install
|
|
196
|
-
4.
|
|
254
|
+
3. Users install the released package with `pip install evalrun`.
|
|
255
|
+
4. Contributors install the repository with `pip install -e .` when working on source changes.
|
|
197
256
|
5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
|
|
198
257
|
6. Users provide their own hosted-model API keys or run their own local model server.
|
|
199
258
|
|
|
200
259
|
The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
|
|
201
260
|
|
|
261
|
+
### TestPyPI (maintainers only)
|
|
262
|
+
|
|
263
|
+
Before a release, maintainers may validate a candidate package from TestPyPI:
|
|
264
|
+
|
|
265
|
+
```bash
|
|
266
|
+
python -m pip install \
|
|
267
|
+
--index-url https://test.pypi.org/simple/ \
|
|
268
|
+
--extra-index-url https://pypi.org/simple/ \
|
|
269
|
+
evalrun
|
|
270
|
+
```
|
|
271
|
+
|
|
272
|
+
End users should use the normal PyPI install shown above.
|
|
273
|
+
|
|
202
274
|
### Credential promise
|
|
203
275
|
|
|
204
276
|
EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
|
|
@@ -250,6 +322,48 @@ evalrun run \
|
|
|
250
322
|
|
|
251
323
|
Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
|
|
252
324
|
|
|
325
|
+
Validate a suite before running it:
|
|
326
|
+
|
|
327
|
+
```bash
|
|
328
|
+
for scenario in evals/scenarios/my-domain/*.md; do
|
|
329
|
+
evalrun validate "$scenario" || exit 2
|
|
330
|
+
done
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
## Connect your own agent
|
|
334
|
+
|
|
335
|
+
EvalRun does not require agents to use a particular framework. The `--agent`
|
|
336
|
+
value selects one of three adapters:
|
|
337
|
+
|
|
338
|
+
| Agent form | Example | When to use |
|
|
339
|
+
| --- | --- | --- |
|
|
340
|
+
| Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
|
|
341
|
+
| HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
|
|
342
|
+
| CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
|
|
343
|
+
|
|
344
|
+
Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
|
|
345
|
+
must expose the adapter contract described in
|
|
346
|
+
[`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
|
|
347
|
+
complete syntax and examples.
|
|
348
|
+
|
|
349
|
+
For repeatable runs, put the same values in a JSON or TOML configuration file:
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
evalrun run --config evalrun.json
|
|
353
|
+
```
|
|
354
|
+
|
|
355
|
+
Command-line flags override values from the configuration file. Keep API keys
|
|
356
|
+
in environment variables and reference them from your shell; never commit them
|
|
357
|
+
to the config file.
|
|
358
|
+
|
|
359
|
+
## Troubleshooting
|
|
360
|
+
|
|
361
|
+
- `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
|
|
362
|
+
- `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
|
|
363
|
+
- Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
|
|
364
|
+
- A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
|
|
365
|
+
- `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
|
|
366
|
+
|
|
253
367
|
## 🏗️ Architecture & Documentation
|
|
254
368
|
|
|
255
369
|
For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""Reflection agent implementation for auditing and critiquing planned itineraries."""
|
|
2
2
|
|
|
3
3
|
from typing import Optional
|
|
4
|
-
from langfuse import observe
|
|
5
4
|
from framework.llms import BaseLLM, Message
|
|
6
5
|
from framework.models import AgentOutput
|
|
7
6
|
from framework.memory import BaseSessionMemory
|
|
7
|
+
from framework.observability import observe
|
|
8
8
|
from agents.base import BaseAgent
|
|
9
9
|
from .prompts import REFLECTION_SYSTEM_PROMPT
|
|
10
10
|
|
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
"""Customer-support triage agent used by the second-domain benchmark."""
|
|
2
2
|
|
|
3
|
-
from langfuse import observe
|
|
4
|
-
|
|
5
3
|
from agents.base import BaseAgent
|
|
6
4
|
from framework.llms import BaseLLM, Message
|
|
7
5
|
from framework.models import AgentOutput
|
|
6
|
+
from framework.observability import observe
|
|
8
7
|
|
|
9
8
|
|
|
10
9
|
SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
|
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
from langfuse import observe
|
|
2
|
-
|
|
3
1
|
import asyncio
|
|
4
2
|
import json
|
|
5
3
|
from typing import Any, Dict, Optional
|
|
@@ -7,6 +5,7 @@ from framework.llms import BaseLLM, Message
|
|
|
7
5
|
from framework.models import AgentOutput
|
|
8
6
|
from framework.utils import parse_json_markdown
|
|
9
7
|
from framework.memory import BaseSessionMemory
|
|
8
|
+
from framework.observability import observe
|
|
10
9
|
from agents.base import BaseAgent
|
|
11
10
|
from agents.research.planner import ResearchPlanner
|
|
12
11
|
from framework.mcp.revision_summary import (
|