evalrun 0.4.1__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrun-0.4.2/LICENSE +21 -0
- {evalrun-0.4.1/evalrun.egg-info → evalrun-0.4.2}/PKG-INFO +121 -7
- evalrun-0.4.1/PKG-INFO → evalrun-0.4.2/README.md +110 -20
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/agent.py +1 -1
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/agent.py +1 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/planner.py +1 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/support/triage_agent.py +1 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/agent.py +1 -2
- evalrun-0.4.2/cli/__init__.py +6 -0
- evalrun-0.4.2/cli/doctor.py +150 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/html_reporter.py +198 -61
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/main.py +273 -2
- evalrun-0.4.2/cli/validator.py +199 -0
- evalrun-0.4.1/README.md → evalrun-0.4.2/evalrun.egg-info/PKG-INFO +134 -4
- {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/SOURCES.txt +7 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/requires.txt +11 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/engine.py +1 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/base_llm.py +2 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/base.py +4 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/runner.py +2 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/openai_compatible.py +13 -2
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/__init__.py +4 -5
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/client.py +14 -2
- evalrun-0.4.2/framework/observability.py +26 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/sdk.py +48 -5
- {evalrun-0.4.1 → evalrun-0.4.2}/pyproject.toml +14 -3
- evalrun-0.4.2/tests/test_canonical_evidence.py +66 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_cli.py +70 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_custom_profiles.py +1 -1
- evalrun-0.4.2/tests/test_doctor.py +67 -0
- evalrun-0.4.2/tests/test_html_reporter.py +108 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_openai_compatible.py +1 -0
- evalrun-0.4.2/tests/test_sdk.py +168 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_ui.py +11 -2
- evalrun-0.4.2/tests/test_validator.py +205 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/ui/server.py +66 -21
- evalrun-0.4.1/LICENSE +0 -0
- evalrun-0.4.1/cli/__init__.py +0 -6
- evalrun-0.4.1/tests/test_sdk.py +0 -88
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/budget_auditor.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/parser.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/prompts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/schema.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/base.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/prompts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/prompts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/support/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/prompts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/session.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/demo.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/formatter.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/progress.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/cli/resolver.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/dependency_links.txt +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/entry_points.txt +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/top_level.txt +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/adapters.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/contracts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/suite.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/base.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/dimensions.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/adaptability.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/constraint.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/information_accuracy.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/personalization.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/planning.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/support.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/adaptability.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/constraint.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/information_accuracy.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/personalization.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/planning.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/testing.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/exceptions.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/base.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/factory.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/gemini.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/mock.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/openai.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/constraints.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/revision_summary.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/server.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/memory/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/memory/base.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/models.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parser.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/frontmatter.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/mapper.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/markdown.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/transformers.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/registry.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/support.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/travel.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/comparator.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/loader.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/utils.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/__init__.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/base.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/extractor.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/local.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/models.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/pipeline.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/prompts.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/utils.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/setup.cfg +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_adapters.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_agent_collaboration.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_config_workflow.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_engine.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_evaluators.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_generic_core.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_independent_auditor.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_mcp_agent_integration.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_mcp_constraints.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_parser.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_reflection_agent.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_regression.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_research_agent.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_revision_summary.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_runner.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_runner_v3.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_session_memory.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_smoke_cli.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_support_domain.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_usability_validation.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_utils.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_verification.py +0 -0
- {evalrun-0.4.1 → evalrun-0.4.2}/ui/__init__.py +0 -0
evalrun-0.4.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Shivam Bhardwaj
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: evalrun
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -10,8 +10,16 @@ Requires-Dist: requests>=2.28.0
|
|
|
10
10
|
Requires-Dist: pydantic>=2.0.0
|
|
11
11
|
Requires-Dist: python-frontmatter>=1.0.0
|
|
12
12
|
Requires-Dist: PyYAML>=6.0
|
|
13
|
-
Requires-Dist:
|
|
14
|
-
|
|
13
|
+
Requires-Dist: markdown-it-py>=3.0.0
|
|
14
|
+
Provides-Extra: observability
|
|
15
|
+
Requires-Dist: langfuse>=2.0.0; extra == "observability"
|
|
16
|
+
Provides-Extra: mcp
|
|
17
|
+
Requires-Dist: fastmcp>=0.1.0; extra == "mcp"
|
|
18
|
+
Requires-Dist: mcp<2,>=1.27; extra == "mcp"
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Requires-Dist: langfuse>=2.0.0; extra == "all"
|
|
21
|
+
Requires-Dist: fastmcp>=0.1.0; extra == "all"
|
|
22
|
+
Requires-Dist: mcp<2,>=1.27; extra == "all"
|
|
15
23
|
Dynamic: license-file
|
|
16
24
|
|
|
17
25
|
# evalrun
|
|
@@ -60,6 +68,13 @@ cd agent-eval-platform
|
|
|
60
68
|
pip install -e .
|
|
61
69
|
```
|
|
62
70
|
|
|
71
|
+
Optional integrations are deliberately not required for the core CLI and SDK:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install "evalrun[observability]" # optional Langfuse tracing
|
|
75
|
+
pip install "evalrun[mcp]" # optional MCP validation tools
|
|
76
|
+
```
|
|
77
|
+
|
|
63
78
|
If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
|
|
64
79
|
|
|
65
80
|
```bash
|
|
@@ -77,7 +92,38 @@ evalrun demo
|
|
|
77
92
|
open results/demo/report.html # macOS
|
|
78
93
|
```
|
|
79
94
|
|
|
80
|
-
|
|
95
|
+
The offline demo is the safest way to confirm that installation works. It makes
|
|
96
|
+
no model requests, needs no API key, and executes only EvalRun's built-in demo
|
|
97
|
+
agent.
|
|
98
|
+
|
|
99
|
+
### 3. Create and check your own evaluation workspace
|
|
100
|
+
|
|
101
|
+
Create starter files in a new folder:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
evalrun init my-evaluation
|
|
105
|
+
cd my-evaluation
|
|
106
|
+
evalrun validate scenario.md
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Edit `scenario.md` to describe the user request, hard constraints, expected
|
|
110
|
+
behavior, scoring dimensions, pass criteria, and failure conditions. The
|
|
111
|
+
scenario is ordinary Markdown; you do not need to change EvalRun's source code.
|
|
112
|
+
Use `evalrun validate` before spending money on a model run. It reports the
|
|
113
|
+
exact missing heading or unsupported dimension and returns exit code `2` when
|
|
114
|
+
the file needs fixing.
|
|
115
|
+
|
|
116
|
+
Run environment diagnostics at any time with:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
evalrun doctor
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
`doctor` does not require an API key. Missing keys and unavailable model servers
|
|
123
|
+
are reported as information, while invalid Python or unwritable output paths are
|
|
124
|
+
reported as failures.
|
|
125
|
+
|
|
126
|
+
### 4. Single Scenario Run (Hosted Model)
|
|
81
127
|
|
|
82
128
|
```bash
|
|
83
129
|
export OPENAI_API_KEY="sk-proj-..."
|
|
@@ -103,6 +149,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
|
|
|
103
149
|
|
|
104
150
|
The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
|
|
105
151
|
|
|
152
|
+
For a target and judge at different providers, configure both explicitly:
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
evalrun run \
|
|
156
|
+
--scenario path/to/scenario.md \
|
|
157
|
+
--agent my_agent:MyAgent \
|
|
158
|
+
--model gemini-3.7-flash \
|
|
159
|
+
--base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
|
|
160
|
+
--api-key "$GEMINI_API_KEY" \
|
|
161
|
+
--judge-model stealth/ox-alpha \
|
|
162
|
+
--judge-base-url https://openrouter.ai/api/v1 \
|
|
163
|
+
--judge-api-key "$OPENROUTER_API_KEY" \
|
|
164
|
+
--output results/mixed-provider-run
|
|
165
|
+
```
|
|
166
|
+
|
|
106
167
|
For example, a Gemini run can be written as:
|
|
107
168
|
|
|
108
169
|
```bash
|
|
@@ -118,7 +179,7 @@ evalrun run \
|
|
|
118
179
|
--output results/gemini-run
|
|
119
180
|
```
|
|
120
181
|
|
|
121
|
-
###
|
|
182
|
+
### 5. Local Model Server Run (vLLM / Ollama)
|
|
122
183
|
|
|
123
184
|
```bash
|
|
124
185
|
# Users host their own local OpenAI-compatible server at http://localhost:8000/v1
|
|
@@ -132,7 +193,7 @@ evalrun run \
|
|
|
132
193
|
--output results/local-run
|
|
133
194
|
```
|
|
134
195
|
|
|
135
|
-
###
|
|
196
|
+
### 6. Guided Local Web UI
|
|
136
197
|
|
|
137
198
|
Launch the zero-dependency local web interface:
|
|
138
199
|
|
|
@@ -142,6 +203,17 @@ evalrun ui --port 8501
|
|
|
142
203
|
|
|
143
204
|
Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
|
|
144
205
|
|
|
206
|
+
The UI is a local convenience layer. It supports built-in agents and workspace
|
|
207
|
+
scenario paths. API keys are held in browser memory for the current request and
|
|
208
|
+
are not saved by EvalRun. Do not expose this server to the public internet and
|
|
209
|
+
do not paste untrusted Python code into an agent field. Complex or third-party
|
|
210
|
+
agents should be run through the CLI or SDK in the environment where their code
|
|
211
|
+
and dependencies are installed.
|
|
212
|
+
|
|
213
|
+
The UI intentionally permits only the built-in travel agent. Custom Python,
|
|
214
|
+
HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
|
|
215
|
+
agent specifications from a browser would allow local code execution.
|
|
216
|
+
|
|
145
217
|
---
|
|
146
218
|
|
|
147
219
|
## 🐍 Python SDK Usage
|
|
@@ -164,7 +236,7 @@ results = evaluate(
|
|
|
164
236
|
report = compare(
|
|
165
237
|
candidate_results=results,
|
|
166
238
|
baseline="results/run-001",
|
|
167
|
-
|
|
239
|
+
max_regression=5.0,
|
|
168
240
|
)
|
|
169
241
|
|
|
170
242
|
if report.release_blocked:
|
|
@@ -274,6 +346,48 @@ evalrun run \
|
|
|
274
346
|
|
|
275
347
|
Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
|
|
276
348
|
|
|
349
|
+
Validate a suite before running it:
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
for scenario in evals/scenarios/my-domain/*.md; do
|
|
353
|
+
evalrun validate "$scenario" || exit 2
|
|
354
|
+
done
|
|
355
|
+
```
|
|
356
|
+
|
|
357
|
+
## Connect your own agent
|
|
358
|
+
|
|
359
|
+
EvalRun does not require agents to use a particular framework. The `--agent`
|
|
360
|
+
value selects one of three adapters:
|
|
361
|
+
|
|
362
|
+
| Agent form | Example | When to use |
|
|
363
|
+
| --- | --- | --- |
|
|
364
|
+
| Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
|
|
365
|
+
| HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
|
|
366
|
+
| CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
|
|
367
|
+
|
|
368
|
+
Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
|
|
369
|
+
must expose the adapter contract described in
|
|
370
|
+
[`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
|
|
371
|
+
complete syntax and examples.
|
|
372
|
+
|
|
373
|
+
For repeatable runs, put the same values in a JSON or TOML configuration file:
|
|
374
|
+
|
|
375
|
+
```bash
|
|
376
|
+
evalrun run --config evalrun.json
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
Command-line flags override values from the configuration file. Keep API keys
|
|
380
|
+
in environment variables and reference them from your shell; never commit them
|
|
381
|
+
to the config file.
|
|
382
|
+
|
|
383
|
+
## Troubleshooting
|
|
384
|
+
|
|
385
|
+
- `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
|
|
386
|
+
- `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
|
|
387
|
+
- Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
|
|
388
|
+
- A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
|
|
389
|
+
- `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
|
|
390
|
+
|
|
277
391
|
## 🏗️ Architecture & Documentation
|
|
278
392
|
|
|
279
393
|
For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
|
|
@@ -1,19 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: evalrun
|
|
3
|
-
Version: 0.4.1
|
|
4
|
-
Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
|
|
5
|
-
Requires-Python: >=3.10
|
|
6
|
-
Description-Content-Type: text/markdown
|
|
7
|
-
License-File: LICENSE
|
|
8
|
-
Requires-Dist: openai>=1.0.0
|
|
9
|
-
Requires-Dist: requests>=2.28.0
|
|
10
|
-
Requires-Dist: pydantic>=2.0.0
|
|
11
|
-
Requires-Dist: python-frontmatter>=1.0.0
|
|
12
|
-
Requires-Dist: PyYAML>=6.0
|
|
13
|
-
Requires-Dist: langfuse>=2.0.0
|
|
14
|
-
Requires-Dist: fastmcp>=0.1.0
|
|
15
|
-
Dynamic: license-file
|
|
16
|
-
|
|
17
1
|
# evalrun
|
|
18
2
|
|
|
19
3
|
> **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
|
|
@@ -60,6 +44,13 @@ cd agent-eval-platform
|
|
|
60
44
|
pip install -e .
|
|
61
45
|
```
|
|
62
46
|
|
|
47
|
+
Optional integrations are deliberately not required for the core CLI and SDK:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install "evalrun[observability]" # optional Langfuse tracing
|
|
51
|
+
pip install "evalrun[mcp]" # optional MCP validation tools
|
|
52
|
+
```
|
|
53
|
+
|
|
63
54
|
If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
|
|
64
55
|
|
|
65
56
|
```bash
|
|
@@ -77,7 +68,38 @@ evalrun demo
|
|
|
77
68
|
open results/demo/report.html # macOS
|
|
78
69
|
```
|
|
79
70
|
|
|
80
|
-
|
|
71
|
+
The offline demo is the safest way to confirm that installation works. It makes
|
|
72
|
+
no model requests, needs no API key, and executes only EvalRun's built-in demo
|
|
73
|
+
agent.
|
|
74
|
+
|
|
75
|
+
### 3. Create and check your own evaluation workspace
|
|
76
|
+
|
|
77
|
+
Create starter files in a new folder:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
evalrun init my-evaluation
|
|
81
|
+
cd my-evaluation
|
|
82
|
+
evalrun validate scenario.md
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Edit `scenario.md` to describe the user request, hard constraints, expected
|
|
86
|
+
behavior, scoring dimensions, pass criteria, and failure conditions. The
|
|
87
|
+
scenario is ordinary Markdown; you do not need to change EvalRun's source code.
|
|
88
|
+
Use `evalrun validate` before spending money on a model run. It reports the
|
|
89
|
+
exact missing heading or unsupported dimension and returns exit code `2` when
|
|
90
|
+
the file needs fixing.
|
|
91
|
+
|
|
92
|
+
Run environment diagnostics at any time with:
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
evalrun doctor
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`doctor` does not require an API key. Missing keys and unavailable model servers
|
|
99
|
+
are reported as information, while invalid Python or unwritable output paths are
|
|
100
|
+
reported as failures.
|
|
101
|
+
|
|
102
|
+
### 4. Single Scenario Run (Hosted Model)
|
|
81
103
|
|
|
82
104
|
```bash
|
|
83
105
|
export OPENAI_API_KEY="sk-proj-..."
|
|
@@ -103,6 +125,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
|
|
|
103
125
|
|
|
104
126
|
The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
|
|
105
127
|
|
|
128
|
+
For a target and judge at different providers, configure both explicitly:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
evalrun run \
|
|
132
|
+
--scenario path/to/scenario.md \
|
|
133
|
+
--agent my_agent:MyAgent \
|
|
134
|
+
--model gemini-3.7-flash \
|
|
135
|
+
--base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
|
|
136
|
+
--api-key "$GEMINI_API_KEY" \
|
|
137
|
+
--judge-model stealth/ox-alpha \
|
|
138
|
+
--judge-base-url https://openrouter.ai/api/v1 \
|
|
139
|
+
--judge-api-key "$OPENROUTER_API_KEY" \
|
|
140
|
+
--output results/mixed-provider-run
|
|
141
|
+
```
|
|
142
|
+
|
|
106
143
|
For example, a Gemini run can be written as:
|
|
107
144
|
|
|
108
145
|
```bash
|
|
@@ -118,7 +155,7 @@ evalrun run \
|
|
|
118
155
|
--output results/gemini-run
|
|
119
156
|
```
|
|
120
157
|
|
|
121
|
-
###
|
|
158
|
+
### 5. Local Model Server Run (vLLM / Ollama)
|
|
122
159
|
|
|
123
160
|
```bash
|
|
124
161
|
# Users host their own local OpenAI-compatible server at http://localhost:8000/v1
|
|
@@ -132,7 +169,7 @@ evalrun run \
|
|
|
132
169
|
--output results/local-run
|
|
133
170
|
```
|
|
134
171
|
|
|
135
|
-
###
|
|
172
|
+
### 6. Guided Local Web UI
|
|
136
173
|
|
|
137
174
|
Launch the zero-dependency local web interface:
|
|
138
175
|
|
|
@@ -142,6 +179,17 @@ evalrun ui --port 8501
|
|
|
142
179
|
|
|
143
180
|
Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
|
|
144
181
|
|
|
182
|
+
The UI is a local convenience layer. It supports built-in agents and workspace
|
|
183
|
+
scenario paths. API keys are held in browser memory for the current request and
|
|
184
|
+
are not saved by EvalRun. Do not expose this server to the public internet and
|
|
185
|
+
do not paste untrusted Python code into an agent field. Complex or third-party
|
|
186
|
+
agents should be run through the CLI or SDK in the environment where their code
|
|
187
|
+
and dependencies are installed.
|
|
188
|
+
|
|
189
|
+
The UI intentionally permits only the built-in travel agent. Custom Python,
|
|
190
|
+
HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
|
|
191
|
+
agent specifications from a browser would allow local code execution.
|
|
192
|
+
|
|
145
193
|
---
|
|
146
194
|
|
|
147
195
|
## 🐍 Python SDK Usage
|
|
@@ -164,7 +212,7 @@ results = evaluate(
|
|
|
164
212
|
report = compare(
|
|
165
213
|
candidate_results=results,
|
|
166
214
|
baseline="results/run-001",
|
|
167
|
-
|
|
215
|
+
max_regression=5.0,
|
|
168
216
|
)
|
|
169
217
|
|
|
170
218
|
if report.release_blocked:
|
|
@@ -274,6 +322,48 @@ evalrun run \
|
|
|
274
322
|
|
|
275
323
|
Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
|
|
276
324
|
|
|
325
|
+
Validate a suite before running it:
|
|
326
|
+
|
|
327
|
+
```bash
|
|
328
|
+
for scenario in evals/scenarios/my-domain/*.md; do
|
|
329
|
+
evalrun validate "$scenario" || exit 2
|
|
330
|
+
done
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
## Connect your own agent
|
|
334
|
+
|
|
335
|
+
EvalRun does not require agents to use a particular framework. The `--agent`
|
|
336
|
+
value selects one of three adapters:
|
|
337
|
+
|
|
338
|
+
| Agent form | Example | When to use |
|
|
339
|
+
| --- | --- | --- |
|
|
340
|
+
| Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
|
|
341
|
+
| HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
|
|
342
|
+
| CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
|
|
343
|
+
|
|
344
|
+
Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
|
|
345
|
+
must expose the adapter contract described in
|
|
346
|
+
[`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
|
|
347
|
+
complete syntax and examples.
|
|
348
|
+
|
|
349
|
+
For repeatable runs, put the same values in a JSON or TOML configuration file:
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
evalrun run --config evalrun.json
|
|
353
|
+
```
|
|
354
|
+
|
|
355
|
+
Command-line flags override values from the configuration file. Keep API keys
|
|
356
|
+
in environment variables and reference them from your shell; never commit them
|
|
357
|
+
to the config file.
|
|
358
|
+
|
|
359
|
+
## Troubleshooting
|
|
360
|
+
|
|
361
|
+
- `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
|
|
362
|
+
- `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
|
|
363
|
+
- Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
|
|
364
|
+
- A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
|
|
365
|
+
- `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
|
|
366
|
+
|
|
277
367
|
## 🏗️ Architecture & Documentation
|
|
278
368
|
|
|
279
369
|
For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""Reflection agent implementation for auditing and critiquing planned itineraries."""
|
|
2
2
|
|
|
3
3
|
from typing import Optional
|
|
4
|
-
from langfuse import observe
|
|
5
4
|
from framework.llms import BaseLLM, Message
|
|
6
5
|
from framework.models import AgentOutput
|
|
7
6
|
from framework.memory import BaseSessionMemory
|
|
7
|
+
from framework.observability import observe
|
|
8
8
|
from agents.base import BaseAgent
|
|
9
9
|
from .prompts import REFLECTION_SYSTEM_PROMPT
|
|
10
10
|
|
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
"""Customer-support triage agent used by the second-domain benchmark."""
|
|
2
2
|
|
|
3
|
-
from langfuse import observe
|
|
4
|
-
|
|
5
3
|
from agents.base import BaseAgent
|
|
6
4
|
from framework.llms import BaseLLM, Message
|
|
7
5
|
from framework.models import AgentOutput
|
|
6
|
+
from framework.observability import observe
|
|
8
7
|
|
|
9
8
|
|
|
10
9
|
SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
|
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
from langfuse import observe
|
|
2
|
-
|
|
3
1
|
import asyncio
|
|
4
2
|
import json
|
|
5
3
|
from typing import Any, Dict, Optional
|
|
@@ -7,6 +5,7 @@ from framework.llms import BaseLLM, Message
|
|
|
7
5
|
from framework.models import AgentOutput
|
|
8
6
|
from framework.utils import parse_json_markdown
|
|
9
7
|
from framework.memory import BaseSessionMemory
|
|
8
|
+
from framework.observability import observe
|
|
10
9
|
from agents.base import BaseAgent
|
|
11
10
|
from agents.research.planner import ResearchPlanner
|
|
12
11
|
from framework.mcp.revision_summary import (
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Diagnostic doctor module for evalrun doctor command."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import os
|
|
5
|
+
import sys
|
|
6
|
+
import urllib.error
|
|
7
|
+
import urllib.request
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Dict, List, Tuple
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def run_doctor_checks() -> Tuple[bool, List[str]]:
|
|
13
|
+
"""Runs system diagnostics for EvalRun environment.
|
|
14
|
+
|
|
15
|
+
Returns:
|
|
16
|
+
Tuple of (all_ok: bool, lines: List[str])
|
|
17
|
+
"""
|
|
18
|
+
lines: List[str] = []
|
|
19
|
+
all_ok = True
|
|
20
|
+
|
|
21
|
+
lines.append("=====================================================================")
|
|
22
|
+
lines.append(" EVALRUN SYSTEM DOCTOR ")
|
|
23
|
+
lines.append("=====================================================================")
|
|
24
|
+
|
|
25
|
+
# 1. Python Version Check
|
|
26
|
+
py_ver = sys.version.split()[0]
|
|
27
|
+
py_ok = sys.version_info >= (3, 10)
|
|
28
|
+
if not py_ok:
|
|
29
|
+
all_ok = False
|
|
30
|
+
lines.append(f"[{'PASS' if py_ok else 'FAIL'}] Python Version: {py_ver} (Requirement: >= 3.10)")
|
|
31
|
+
|
|
32
|
+
# 2. EvalRun Version & Installation State
|
|
33
|
+
try:
|
|
34
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
35
|
+
pkg_version = version("evalrun")
|
|
36
|
+
install_type = "Installed Package"
|
|
37
|
+
except (PackageNotFoundError, ImportError):
|
|
38
|
+
pkg_version = "unknown"
|
|
39
|
+
pyproject = Path(__file__).resolve().parents[1] / "pyproject.toml"
|
|
40
|
+
try:
|
|
41
|
+
try:
|
|
42
|
+
import tomllib
|
|
43
|
+
with pyproject.open("rb") as f:
|
|
44
|
+
metadata = tomllib.load(f)
|
|
45
|
+
except ModuleNotFoundError:
|
|
46
|
+
import re
|
|
47
|
+
metadata = {}
|
|
48
|
+
match = re.search(r'(?m)^version\s*=\s*["\']([^"\']+)["\']', pyproject.read_text(encoding="utf-8"))
|
|
49
|
+
if match:
|
|
50
|
+
metadata = {"project": {"version": match.group(1)}}
|
|
51
|
+
pkg_version = str(metadata.get("project", {}).get("version", "unknown"))
|
|
52
|
+
except Exception:
|
|
53
|
+
pass
|
|
54
|
+
install_type = "Local Source Checkout"
|
|
55
|
+
|
|
56
|
+
lines.append(f"[PASS] EvalRun Version: {pkg_version} ({install_type})")
|
|
57
|
+
|
|
58
|
+
# 3. Model API Key Environment Variables
|
|
59
|
+
env_keys = {
|
|
60
|
+
"OPENAI_API_KEY": os.getenv("OPENAI_API_KEY"),
|
|
61
|
+
"GEMINI_API_KEY": os.getenv("GEMINI_API_KEY"),
|
|
62
|
+
"OPENROUTER_API_KEY": os.getenv("OPENROUTER_API_KEY"),
|
|
63
|
+
"NVIDIA_API_KEY": os.getenv("NVIDIA_API_KEY"),
|
|
64
|
+
}
|
|
65
|
+
set_keys = [k for k, v in env_keys.items() if v]
|
|
66
|
+
if set_keys:
|
|
67
|
+
lines.append(f"[PASS] Model API Keys Configured: {', '.join(set_keys)}")
|
|
68
|
+
else:
|
|
69
|
+
lines.append("[INFO] Model API Keys Configured: None (Offline demo works without keys)")
|
|
70
|
+
|
|
71
|
+
# 4. Agent Importability
|
|
72
|
+
try:
|
|
73
|
+
mod = importlib.import_module("agents.travel")
|
|
74
|
+
agent_cls = getattr(mod, "TravelPlanningAgent", None)
|
|
75
|
+
agent_ok = agent_cls is not None
|
|
76
|
+
except Exception:
|
|
77
|
+
agent_ok = False
|
|
78
|
+
|
|
79
|
+
if agent_ok:
|
|
80
|
+
lines.append("[PASS] Built-in Agent Import: agents.travel:TravelPlanningAgent")
|
|
81
|
+
else:
|
|
82
|
+
lines.append("[WARN] Built-in Agent Import: agents.travel not found in python path")
|
|
83
|
+
|
|
84
|
+
# 5. Endpoint Reachability (OpenAI API)
|
|
85
|
+
skip_network = os.getenv("EVALRUN_SKIP_NETWORK_CHECKS") == "1"
|
|
86
|
+
openai_reach = False
|
|
87
|
+
if not skip_network:
|
|
88
|
+
try:
|
|
89
|
+
req = urllib.request.Request("https://api.openai.com/v1/models", headers={"User-Agent": "evalrun-doctor"})
|
|
90
|
+
with urllib.request.urlopen(req, timeout=3) as resp:
|
|
91
|
+
openai_reach = resp.status in (200, 401)
|
|
92
|
+
except urllib.error.HTTPError as e:
|
|
93
|
+
openai_reach = e.code in (401, 403, 200)
|
|
94
|
+
except Exception:
|
|
95
|
+
openai_reach = False
|
|
96
|
+
|
|
97
|
+
if openai_reach:
|
|
98
|
+
lines.append("[PASS] Hosted Endpoint Reachability: https://api.openai.com/v1")
|
|
99
|
+
elif skip_network:
|
|
100
|
+
lines.append("[INFO] Hosted Endpoint Reachability: skipped (EVALRUN_SKIP_NETWORK_CHECKS=1)")
|
|
101
|
+
else:
|
|
102
|
+
lines.append("[INFO] Hosted Endpoint Reachability: https://api.openai.com/v1 (Offline or unreachable)")
|
|
103
|
+
|
|
104
|
+
# 6. Local Model Server Availability
|
|
105
|
+
local_reach = False
|
|
106
|
+
if not skip_network:
|
|
107
|
+
for local_url in ["http://localhost:8000/v1/models", "http://localhost:11434/api/tags"]:
|
|
108
|
+
try:
|
|
109
|
+
req = urllib.request.Request(local_url)
|
|
110
|
+
with urllib.request.urlopen(req, timeout=2) as resp:
|
|
111
|
+
if resp.status in (200, 401, 403):
|
|
112
|
+
local_reach = True
|
|
113
|
+
break
|
|
114
|
+
except urllib.error.HTTPError as e:
|
|
115
|
+
if e.code in (200, 401, 403):
|
|
116
|
+
local_reach = True
|
|
117
|
+
break
|
|
118
|
+
except Exception:
|
|
119
|
+
pass
|
|
120
|
+
|
|
121
|
+
if local_reach:
|
|
122
|
+
lines.append("[PASS] Local Model Server: Active local LLM server detected")
|
|
123
|
+
elif skip_network:
|
|
124
|
+
lines.append("[INFO] Local Model Server: skipped (EVALRUN_SKIP_NETWORK_CHECKS=1)")
|
|
125
|
+
else:
|
|
126
|
+
lines.append("[INFO] Local Model Server: No active local LLM server detected on 8000/11434")
|
|
127
|
+
|
|
128
|
+
# 7. Write Access to Output Folder
|
|
129
|
+
output_dir = Path("./eval_results")
|
|
130
|
+
try:
|
|
131
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
132
|
+
test_file = output_dir / ".doctor_temp"
|
|
133
|
+
test_file.write_text("write_test", encoding="utf-8")
|
|
134
|
+
test_file.unlink()
|
|
135
|
+
write_ok = True
|
|
136
|
+
except Exception:
|
|
137
|
+
write_ok = False
|
|
138
|
+
|
|
139
|
+
if not write_ok:
|
|
140
|
+
all_ok = False
|
|
141
|
+
lines.append(f"[{'PASS' if write_ok else 'FAIL'}] Output Directory Write Access: {output_dir.resolve()}")
|
|
142
|
+
|
|
143
|
+
lines.append("=====================================================================")
|
|
144
|
+
if all_ok:
|
|
145
|
+
lines.append(" Doctor Verdict: SYSTEM READY FOR EVALUATION RUNS")
|
|
146
|
+
else:
|
|
147
|
+
lines.append(" Doctor Verdict: ISSUES DETECTED - PLEASE REVIEW WARNINGS ABOVE")
|
|
148
|
+
lines.append("=====================================================================")
|
|
149
|
+
|
|
150
|
+
return all_ok, lines
|