evalrun 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. evalrun-0.4.2/LICENSE +21 -0
  2. {evalrun-0.4.0/evalrun.egg-info → evalrun-0.4.2}/PKG-INFO +148 -10
  3. evalrun-0.4.0/PKG-INFO → evalrun-0.4.2/README.md +137 -23
  4. {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/agent.py +1 -1
  5. {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/agent.py +1 -2
  6. {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/planner.py +1 -2
  7. {evalrun-0.4.0 → evalrun-0.4.2}/agents/support/triage_agent.py +1 -2
  8. {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/agent.py +1 -2
  9. evalrun-0.4.2/cli/__init__.py +6 -0
  10. evalrun-0.4.2/cli/doctor.py +150 -0
  11. {evalrun-0.4.0 → evalrun-0.4.2}/cli/html_reporter.py +198 -61
  12. {evalrun-0.4.0 → evalrun-0.4.2}/cli/main.py +273 -2
  13. evalrun-0.4.2/cli/validator.py +199 -0
  14. evalrun-0.4.0/README.md → evalrun-0.4.2/evalrun.egg-info/PKG-INFO +161 -7
  15. {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/SOURCES.txt +7 -0
  16. {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/requires.txt +11 -0
  17. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/engine.py +1 -2
  18. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/base_llm.py +2 -0
  19. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/base.py +4 -0
  20. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/runner.py +2 -2
  21. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/openai_compatible.py +13 -2
  22. {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/__init__.py +4 -5
  23. {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/client.py +14 -2
  24. evalrun-0.4.2/framework/observability.py +26 -0
  25. {evalrun-0.4.0 → evalrun-0.4.2}/framework/sdk.py +48 -5
  26. {evalrun-0.4.0 → evalrun-0.4.2}/pyproject.toml +14 -3
  27. evalrun-0.4.2/tests/test_canonical_evidence.py +66 -0
  28. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_cli.py +70 -0
  29. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_custom_profiles.py +1 -1
  30. evalrun-0.4.2/tests/test_doctor.py +67 -0
  31. evalrun-0.4.2/tests/test_html_reporter.py +108 -0
  32. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_openai_compatible.py +1 -0
  33. evalrun-0.4.2/tests/test_sdk.py +168 -0
  34. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_ui.py +11 -2
  35. evalrun-0.4.2/tests/test_validator.py +205 -0
  36. {evalrun-0.4.0 → evalrun-0.4.2}/ui/server.py +66 -21
  37. evalrun-0.4.0/LICENSE +0 -0
  38. evalrun-0.4.0/cli/__init__.py +0 -6
  39. evalrun-0.4.0/tests/test_sdk.py +0 -88
  40. {evalrun-0.4.0 → evalrun-0.4.2}/agents/__init__.py +0 -0
  41. {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/__init__.py +0 -0
  42. {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/budget_auditor.py +0 -0
  43. {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/parser.py +0 -0
  44. {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/prompts.py +0 -0
  45. {evalrun-0.4.0 → evalrun-0.4.2}/agents/auditor/schema.py +0 -0
  46. {evalrun-0.4.0 → evalrun-0.4.2}/agents/base.py +0 -0
  47. {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/__init__.py +0 -0
  48. {evalrun-0.4.0 → evalrun-0.4.2}/agents/reflection/prompts.py +0 -0
  49. {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/__init__.py +0 -0
  50. {evalrun-0.4.0 → evalrun-0.4.2}/agents/research/prompts.py +0 -0
  51. {evalrun-0.4.0 → evalrun-0.4.2}/agents/support/__init__.py +0 -0
  52. {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/__init__.py +0 -0
  53. {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/prompts.py +0 -0
  54. {evalrun-0.4.0 → evalrun-0.4.2}/agents/travel/session.py +0 -0
  55. {evalrun-0.4.0 → evalrun-0.4.2}/cli/demo.py +0 -0
  56. {evalrun-0.4.0 → evalrun-0.4.2}/cli/formatter.py +0 -0
  57. {evalrun-0.4.0 → evalrun-0.4.2}/cli/progress.py +0 -0
  58. {evalrun-0.4.0 → evalrun-0.4.2}/cli/resolver.py +0 -0
  59. {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/dependency_links.txt +0 -0
  60. {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/entry_points.txt +0 -0
  61. {evalrun-0.4.0 → evalrun-0.4.2}/evalrun.egg-info/top_level.txt +0 -0
  62. {evalrun-0.4.0 → evalrun-0.4.2}/framework/__init__.py +0 -0
  63. {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/__init__.py +0 -0
  64. {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/adapters.py +0 -0
  65. {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/contracts.py +0 -0
  66. {evalrun-0.4.0 → evalrun-0.4.2}/framework/core/suite.py +0 -0
  67. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/__init__.py +0 -0
  68. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/base.py +0 -0
  69. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/dimensions.py +0 -0
  70. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/__init__.py +0 -0
  71. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/adaptability.py +0 -0
  72. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/constraint.py +0 -0
  73. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/information_accuracy.py +0 -0
  74. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/personalization.py +0 -0
  75. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/planning.py +0 -0
  76. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/evaluators/support.py +0 -0
  77. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/__init__.py +0 -0
  78. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/adaptability.py +0 -0
  79. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/constraint.py +0 -0
  80. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/information_accuracy.py +0 -0
  81. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/personalization.py +0 -0
  82. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/prompts/planning.py +0 -0
  83. {evalrun-0.4.0 → evalrun-0.4.2}/framework/evaluation/testing.py +0 -0
  84. {evalrun-0.4.0 → evalrun-0.4.2}/framework/exceptions.py +0 -0
  85. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/__init__.py +0 -0
  86. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/base.py +0 -0
  87. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/factory.py +0 -0
  88. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/gemini.py +0 -0
  89. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/mock.py +0 -0
  90. {evalrun-0.4.0 → evalrun-0.4.2}/framework/llms/openai.py +0 -0
  91. {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/constraints.py +0 -0
  92. {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/revision_summary.py +0 -0
  93. {evalrun-0.4.0 → evalrun-0.4.2}/framework/mcp/server.py +0 -0
  94. {evalrun-0.4.0 → evalrun-0.4.2}/framework/memory/__init__.py +0 -0
  95. {evalrun-0.4.0 → evalrun-0.4.2}/framework/memory/base.py +0 -0
  96. {evalrun-0.4.0 → evalrun-0.4.2}/framework/models.py +0 -0
  97. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parser.py +0 -0
  98. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/__init__.py +0 -0
  99. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/frontmatter.py +0 -0
  100. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/mapper.py +0 -0
  101. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/markdown.py +0 -0
  102. {evalrun-0.4.0 → evalrun-0.4.2}/framework/parsers/transformers.py +0 -0
  103. {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/__init__.py +0 -0
  104. {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/registry.py +0 -0
  105. {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/support.py +0 -0
  106. {evalrun-0.4.0 → evalrun-0.4.2}/framework/profiles/travel.py +0 -0
  107. {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/__init__.py +0 -0
  108. {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/comparator.py +0 -0
  109. {evalrun-0.4.0 → evalrun-0.4.2}/framework/regression/loader.py +0 -0
  110. {evalrun-0.4.0 → evalrun-0.4.2}/framework/utils.py +0 -0
  111. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/__init__.py +0 -0
  112. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/base.py +0 -0
  113. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/extractor.py +0 -0
  114. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/local.py +0 -0
  115. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/models.py +0 -0
  116. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/pipeline.py +0 -0
  117. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/prompts.py +0 -0
  118. {evalrun-0.4.0 → evalrun-0.4.2}/framework/verification/utils.py +0 -0
  119. {evalrun-0.4.0 → evalrun-0.4.2}/setup.cfg +0 -0
  120. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_adapters.py +0 -0
  121. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_agent_collaboration.py +0 -0
  122. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_config_workflow.py +0 -0
  123. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_engine.py +0 -0
  124. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_evaluators.py +0 -0
  125. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_generic_core.py +0 -0
  126. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_independent_auditor.py +0 -0
  127. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_mcp_agent_integration.py +0 -0
  128. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_mcp_constraints.py +0 -0
  129. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_parser.py +0 -0
  130. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_reflection_agent.py +0 -0
  131. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_regression.py +0 -0
  132. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_research_agent.py +0 -0
  133. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_revision_summary.py +0 -0
  134. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_runner.py +0 -0
  135. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_runner_v3.py +0 -0
  136. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_session_memory.py +0 -0
  137. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_smoke_cli.py +0 -0
  138. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_support_domain.py +0 -0
  139. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_usability_validation.py +0 -0
  140. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_utils.py +0 -0
  141. {evalrun-0.4.0 → evalrun-0.4.2}/tests/test_verification.py +0 -0
  142. {evalrun-0.4.0 → evalrun-0.4.2}/ui/__init__.py +0 -0
evalrun-0.4.2/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Shivam Bhardwaj
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: evalrun
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -10,8 +10,16 @@ Requires-Dist: requests>=2.28.0
10
10
  Requires-Dist: pydantic>=2.0.0
11
11
  Requires-Dist: python-frontmatter>=1.0.0
12
12
  Requires-Dist: PyYAML>=6.0
13
- Requires-Dist: langfuse>=2.0.0
14
- Requires-Dist: fastmcp>=0.1.0
13
+ Requires-Dist: markdown-it-py>=3.0.0
14
+ Provides-Extra: observability
15
+ Requires-Dist: langfuse>=2.0.0; extra == "observability"
16
+ Provides-Extra: mcp
17
+ Requires-Dist: fastmcp>=0.1.0; extra == "mcp"
18
+ Requires-Dist: mcp<2,>=1.27; extra == "mcp"
19
+ Provides-Extra: all
20
+ Requires-Dist: langfuse>=2.0.0; extra == "all"
21
+ Requires-Dist: fastmcp>=0.1.0; extra == "all"
22
+ Requires-Dist: mcp<2,>=1.27; extra == "all"
15
23
  Dynamic: license-file
16
24
 
17
25
  # evalrun
@@ -43,12 +51,30 @@ The toolkit works with hosted APIs and models running on your own computer. It d
43
51
 
44
52
  ### 1. Installation
45
53
 
54
+ Install the released package from PyPI:
55
+
56
+ ```bash
57
+ python -m pip install evalrun
58
+ ```
59
+
60
+ The package installs the `evalrun` command and all runtime dependencies. You do
61
+ not need to clone this repository to use the toolkit. To contribute or run the
62
+ latest unreleased source instead, clone the repository and use an editable
63
+ install:
64
+
46
65
  ```bash
47
66
  git clone https://github.com/imshivamb/agent-eval-platform.git
48
67
  cd agent-eval-platform
49
68
  pip install -e .
50
69
  ```
51
70
 
71
+ Optional integrations are deliberately not required for the core CLI and SDK:
72
+
73
+ ```bash
74
+ pip install "evalrun[observability]" # optional Langfuse tracing
75
+ pip install "evalrun[mcp]" # optional MCP validation tools
76
+ ```
77
+
52
78
  If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
53
79
 
54
80
  ```bash
@@ -66,7 +92,38 @@ evalrun demo
66
92
  open results/demo/report.html # macOS
67
93
  ```
68
94
 
69
- ### 3. Single Scenario Run (Hosted Model)
95
+ The offline demo is the safest way to confirm that installation works. It makes
96
+ no model requests, needs no API key, and executes only EvalRun's built-in demo
97
+ agent.
98
+
99
+ ### 3. Create and check your own evaluation workspace
100
+
101
+ Create starter files in a new folder:
102
+
103
+ ```bash
104
+ evalrun init my-evaluation
105
+ cd my-evaluation
106
+ evalrun validate scenario.md
107
+ ```
108
+
109
+ Edit `scenario.md` to describe the user request, hard constraints, expected
110
+ behavior, scoring dimensions, pass criteria, and failure conditions. The
111
+ scenario is ordinary Markdown; you do not need to change EvalRun's source code.
112
+ Use `evalrun validate` before spending money on a model run. It reports the
113
+ exact missing heading or unsupported dimension and returns exit code `2` when
114
+ the file needs fixing.
115
+
116
+ Run environment diagnostics at any time with:
117
+
118
+ ```bash
119
+ evalrun doctor
120
+ ```
121
+
122
+ `doctor` does not require an API key. Missing keys and unavailable model servers
123
+ are reported as information, while invalid Python or unwritable output paths are
124
+ reported as failures.
125
+
126
+ ### 4. Single Scenario Run (Hosted Model)
70
127
 
71
128
  ```bash
72
129
  export OPENAI_API_KEY="sk-proj-..."
@@ -92,6 +149,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
92
149
 
93
150
  The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
94
151
 
152
+ For a target and judge at different providers, configure both explicitly:
153
+
154
+ ```bash
155
+ evalrun run \
156
+ --scenario path/to/scenario.md \
157
+ --agent my_agent:MyAgent \
158
+ --model gemini-3.7-flash \
159
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
160
+ --api-key "$GEMINI_API_KEY" \
161
+ --judge-model stealth/ox-alpha \
162
+ --judge-base-url https://openrouter.ai/api/v1 \
163
+ --judge-api-key "$OPENROUTER_API_KEY" \
164
+ --output results/mixed-provider-run
165
+ ```
166
+
95
167
  For example, a Gemini run can be written as:
96
168
 
97
169
  ```bash
@@ -107,7 +179,7 @@ evalrun run \
107
179
  --output results/gemini-run
108
180
  ```
109
181
 
110
- ### 4. Local Model Server Run (vLLM / Ollama)
182
+ ### 5. Local Model Server Run (vLLM / Ollama)
111
183
 
112
184
  ```bash
113
185
  # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
@@ -121,7 +193,7 @@ evalrun run \
121
193
  --output results/local-run
122
194
  ```
123
195
 
124
- ### 5. Guided Local Web UI
196
+ ### 6. Guided Local Web UI
125
197
 
126
198
  Launch the zero-dependency local web interface:
127
199
 
@@ -131,6 +203,17 @@ evalrun ui --port 8501
131
203
 
132
204
  Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
133
205
 
206
+ The UI is a local convenience layer. It supports built-in agents and workspace
207
+ scenario paths. API keys are held in browser memory for the current request and
208
+ are not saved by EvalRun. Do not expose this server to the public internet and
209
+ do not paste untrusted Python code into an agent field. Complex or third-party
210
+ agents should be run through the CLI or SDK in the environment where their code
211
+ and dependencies are installed.
212
+
213
+ The UI intentionally permits only the built-in travel agent. Custom Python,
214
+ HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
215
+ agent specifications from a browser would allow local code execution.
216
+
134
217
  ---
135
218
 
136
219
  ## 🐍 Python SDK Usage
@@ -138,7 +221,7 @@ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scen
138
221
  Integrate `evalrun` programmatically into Python automation pipelines:
139
222
 
140
223
  ```python
141
- from framework.sdk import evaluate, compare
224
+ from framework import evaluate, compare
142
225
 
143
226
  # 1. Execute Benchmark Evaluation
144
227
  results = evaluate(
@@ -153,7 +236,7 @@ results = evaluate(
153
236
  report = compare(
154
237
  candidate_results=results,
155
238
  baseline="results/run-001",
156
- max_overall_drop=5.0,
239
+ max_regression=5.0,
157
240
  )
158
241
 
159
242
  if report.release_blocked:
@@ -192,13 +275,26 @@ There is no central service to deploy for the current product. The recommended p
192
275
 
193
276
  1. Publish the repository on GitHub.
194
277
  2. Add tagged releases and a clear quick-start guide.
195
- 3. Users install it locally with `pip install -e .` from a clone.
196
- 4. Later publish the package to PyPI so users can run `pip install evalrun`.
278
+ 3. Users install the released package with `pip install evalrun`.
279
+ 4. Contributors install the repository with `pip install -e .` when working on source changes.
197
280
  5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
198
281
  6. Users provide their own hosted-model API keys or run their own local model server.
199
282
 
200
283
  The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
201
284
 
285
+ ### TestPyPI (maintainers only)
286
+
287
+ Before a release, maintainers may validate a candidate package from TestPyPI:
288
+
289
+ ```bash
290
+ python -m pip install \
291
+ --index-url https://test.pypi.org/simple/ \
292
+ --extra-index-url https://pypi.org/simple/ \
293
+ evalrun
294
+ ```
295
+
296
+ End users should use the normal PyPI install shown above.
297
+
202
298
  ### Credential promise
203
299
 
204
300
  EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
@@ -250,6 +346,48 @@ evalrun run \
250
346
 
251
347
  Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
252
348
 
349
+ Validate a suite before running it:
350
+
351
+ ```bash
352
+ for scenario in evals/scenarios/my-domain/*.md; do
353
+ evalrun validate "$scenario" || exit 2
354
+ done
355
+ ```
356
+
357
+ ## Connect your own agent
358
+
359
+ EvalRun does not require agents to use a particular framework. The `--agent`
360
+ value selects one of three adapters:
361
+
362
+ | Agent form | Example | When to use |
363
+ | --- | --- | --- |
364
+ | Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
365
+ | HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
366
+ | CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
367
+
368
+ Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
369
+ must expose the adapter contract described in
370
+ [`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
371
+ complete syntax and examples.
372
+
373
+ For repeatable runs, put the same values in a JSON or TOML configuration file:
374
+
375
+ ```bash
376
+ evalrun run --config evalrun.json
377
+ ```
378
+
379
+ Command-line flags override values from the configuration file. Keep API keys
380
+ in environment variables and reference them from your shell; never commit them
381
+ to the config file.
382
+
383
+ ## Troubleshooting
384
+
385
+ - `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
386
+ - `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
387
+ - Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
388
+ - A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
389
+ - `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
390
+
253
391
  ## 🏗️ Architecture & Documentation
254
392
 
255
393
  For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
@@ -1,19 +1,3 @@
1
- Metadata-Version: 2.4
2
- Name: evalrun
3
- Version: 0.4.0
4
- Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
5
- Requires-Python: >=3.10
6
- Description-Content-Type: text/markdown
7
- License-File: LICENSE
8
- Requires-Dist: openai>=1.0.0
9
- Requires-Dist: requests>=2.28.0
10
- Requires-Dist: pydantic>=2.0.0
11
- Requires-Dist: python-frontmatter>=1.0.0
12
- Requires-Dist: PyYAML>=6.0
13
- Requires-Dist: langfuse>=2.0.0
14
- Requires-Dist: fastmcp>=0.1.0
15
- Dynamic: license-file
16
-
17
1
  # evalrun
18
2
 
19
3
  > **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
@@ -43,12 +27,30 @@ The toolkit works with hosted APIs and models running on your own computer. It d
43
27
 
44
28
  ### 1. Installation
45
29
 
30
+ Install the released package from PyPI:
31
+
32
+ ```bash
33
+ python -m pip install evalrun
34
+ ```
35
+
36
+ The package installs the `evalrun` command and all runtime dependencies. You do
37
+ not need to clone this repository to use the toolkit. To contribute or run the
38
+ latest unreleased source instead, clone the repository and use an editable
39
+ install:
40
+
46
41
  ```bash
47
42
  git clone https://github.com/imshivamb/agent-eval-platform.git
48
43
  cd agent-eval-platform
49
44
  pip install -e .
50
45
  ```
51
46
 
47
+ Optional integrations are deliberately not required for the core CLI and SDK:
48
+
49
+ ```bash
50
+ pip install "evalrun[observability]" # optional Langfuse tracing
51
+ pip install "evalrun[mcp]" # optional MCP validation tools
52
+ ```
53
+
52
54
  If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
53
55
 
54
56
  ```bash
@@ -66,7 +68,38 @@ evalrun demo
66
68
  open results/demo/report.html # macOS
67
69
  ```
68
70
 
69
- ### 3. Single Scenario Run (Hosted Model)
71
+ The offline demo is the safest way to confirm that installation works. It makes
72
+ no model requests, needs no API key, and executes only EvalRun's built-in demo
73
+ agent.
74
+
75
+ ### 3. Create and check your own evaluation workspace
76
+
77
+ Create starter files in a new folder:
78
+
79
+ ```bash
80
+ evalrun init my-evaluation
81
+ cd my-evaluation
82
+ evalrun validate scenario.md
83
+ ```
84
+
85
+ Edit `scenario.md` to describe the user request, hard constraints, expected
86
+ behavior, scoring dimensions, pass criteria, and failure conditions. The
87
+ scenario is ordinary Markdown; you do not need to change EvalRun's source code.
88
+ Use `evalrun validate` before spending money on a model run. It reports the
89
+ exact missing heading or unsupported dimension and returns exit code `2` when
90
+ the file needs fixing.
91
+
92
+ Run environment diagnostics at any time with:
93
+
94
+ ```bash
95
+ evalrun doctor
96
+ ```
97
+
98
+ `doctor` does not require an API key. Missing keys and unavailable model servers
99
+ are reported as information, while invalid Python or unwritable output paths are
100
+ reported as failures.
101
+
102
+ ### 4. Single Scenario Run (Hosted Model)
70
103
 
71
104
  ```bash
72
105
  export OPENAI_API_KEY="sk-proj-..."
@@ -92,6 +125,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
92
125
 
93
126
  The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
94
127
 
128
+ For a target and judge at different providers, configure both explicitly:
129
+
130
+ ```bash
131
+ evalrun run \
132
+ --scenario path/to/scenario.md \
133
+ --agent my_agent:MyAgent \
134
+ --model gemini-3.7-flash \
135
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
136
+ --api-key "$GEMINI_API_KEY" \
137
+ --judge-model stealth/ox-alpha \
138
+ --judge-base-url https://openrouter.ai/api/v1 \
139
+ --judge-api-key "$OPENROUTER_API_KEY" \
140
+ --output results/mixed-provider-run
141
+ ```
142
+
95
143
  For example, a Gemini run can be written as:
96
144
 
97
145
  ```bash
@@ -107,7 +155,7 @@ evalrun run \
107
155
  --output results/gemini-run
108
156
  ```
109
157
 
110
- ### 4. Local Model Server Run (vLLM / Ollama)
158
+ ### 5. Local Model Server Run (vLLM / Ollama)
111
159
 
112
160
  ```bash
113
161
  # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
@@ -121,7 +169,7 @@ evalrun run \
121
169
  --output results/local-run
122
170
  ```
123
171
 
124
- ### 5. Guided Local Web UI
172
+ ### 6. Guided Local Web UI
125
173
 
126
174
  Launch the zero-dependency local web interface:
127
175
 
@@ -131,6 +179,17 @@ evalrun ui --port 8501
131
179
 
132
180
  Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
133
181
 
182
+ The UI is a local convenience layer. It supports built-in agents and workspace
183
+ scenario paths. API keys are held in browser memory for the current request and
184
+ are not saved by EvalRun. Do not expose this server to the public internet and
185
+ do not paste untrusted Python code into an agent field. Complex or third-party
186
+ agents should be run through the CLI or SDK in the environment where their code
187
+ and dependencies are installed.
188
+
189
+ The UI intentionally permits only the built-in travel agent. Custom Python,
190
+ HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
191
+ agent specifications from a browser would allow local code execution.
192
+
134
193
  ---
135
194
 
136
195
  ## 🐍 Python SDK Usage
@@ -138,7 +197,7 @@ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scen
138
197
  Integrate `evalrun` programmatically into Python automation pipelines:
139
198
 
140
199
  ```python
141
- from framework.sdk import evaluate, compare
200
+ from framework import evaluate, compare
142
201
 
143
202
  # 1. Execute Benchmark Evaluation
144
203
  results = evaluate(
@@ -153,7 +212,7 @@ results = evaluate(
153
212
  report = compare(
154
213
  candidate_results=results,
155
214
  baseline="results/run-001",
156
- max_overall_drop=5.0,
215
+ max_regression=5.0,
157
216
  )
158
217
 
159
218
  if report.release_blocked:
@@ -192,13 +251,26 @@ There is no central service to deploy for the current product. The recommended p
192
251
 
193
252
  1. Publish the repository on GitHub.
194
253
  2. Add tagged releases and a clear quick-start guide.
195
- 3. Users install it locally with `pip install -e .` from a clone.
196
- 4. Later publish the package to PyPI so users can run `pip install evalrun`.
254
+ 3. Users install the released package with `pip install evalrun`.
255
+ 4. Contributors install the repository with `pip install -e .` when working on source changes.
197
256
  5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
198
257
  6. Users provide their own hosted-model API keys or run their own local model server.
199
258
 
200
259
  The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
201
260
 
261
+ ### TestPyPI (maintainers only)
262
+
263
+ Before a release, maintainers may validate a candidate package from TestPyPI:
264
+
265
+ ```bash
266
+ python -m pip install \
267
+ --index-url https://test.pypi.org/simple/ \
268
+ --extra-index-url https://pypi.org/simple/ \
269
+ evalrun
270
+ ```
271
+
272
+ End users should use the normal PyPI install shown above.
273
+
202
274
  ### Credential promise
203
275
 
204
276
  EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
@@ -250,6 +322,48 @@ evalrun run \
250
322
 
251
323
  Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
252
324
 
325
+ Validate a suite before running it:
326
+
327
+ ```bash
328
+ for scenario in evals/scenarios/my-domain/*.md; do
329
+ evalrun validate "$scenario" || exit 2
330
+ done
331
+ ```
332
+
333
+ ## Connect your own agent
334
+
335
+ EvalRun does not require agents to use a particular framework. The `--agent`
336
+ value selects one of three adapters:
337
+
338
+ | Agent form | Example | When to use |
339
+ | --- | --- | --- |
340
+ | Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
341
+ | HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
342
+ | CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
343
+
344
+ Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
345
+ must expose the adapter contract described in
346
+ [`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
347
+ complete syntax and examples.
348
+
349
+ For repeatable runs, put the same values in a JSON or TOML configuration file:
350
+
351
+ ```bash
352
+ evalrun run --config evalrun.json
353
+ ```
354
+
355
+ Command-line flags override values from the configuration file. Keep API keys
356
+ in environment variables and reference them from your shell; never commit them
357
+ to the config file.
358
+
359
+ ## Troubleshooting
360
+
361
+ - `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
362
+ - `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
363
+ - Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
364
+ - A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
365
+ - `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
366
+
253
367
  ## 🏗️ Architecture & Documentation
254
368
 
255
369
  For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
@@ -1,10 +1,10 @@
1
1
  """Reflection agent implementation for auditing and critiquing planned itineraries."""
2
2
 
3
3
  from typing import Optional
4
- from langfuse import observe
5
4
  from framework.llms import BaseLLM, Message
6
5
  from framework.models import AgentOutput
7
6
  from framework.memory import BaseSessionMemory
7
+ from framework.observability import observe
8
8
  from agents.base import BaseAgent
9
9
  from .prompts import REFLECTION_SYSTEM_PROMPT
10
10
 
@@ -1,7 +1,6 @@
1
- from langfuse import observe
2
-
3
1
  from framework.llms import BaseLLM, Message
4
2
  from framework.models import AgentOutput
3
+ from framework.observability import observe
5
4
  from agents.base import BaseAgent
6
5
  from .prompts import RESEARCH_SYSTEM_PROMPT
7
6
 
@@ -2,8 +2,7 @@
2
2
 
3
3
  from framework.llms import BaseLLM, Message
4
4
  from framework.utils import parse_json_markdown
5
-
6
- from langfuse import observe
5
+ from framework.observability import observe
7
6
 
8
7
 
9
8
  class ResearchPlanner:
@@ -1,10 +1,9 @@
1
1
  """Customer-support triage agent used by the second-domain benchmark."""
2
2
 
3
- from langfuse import observe
4
-
5
3
  from agents.base import BaseAgent
6
4
  from framework.llms import BaseLLM, Message
7
5
  from framework.models import AgentOutput
6
+ from framework.observability import observe
8
7
 
9
8
 
10
9
  SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
@@ -1,5 +1,3 @@
1
- from langfuse import observe
2
-
3
1
  import asyncio
4
2
  import json
5
3
  from typing import Any, Dict, Optional
@@ -7,6 +5,7 @@ from framework.llms import BaseLLM, Message
7
5
  from framework.models import AgentOutput
8
6
  from framework.utils import parse_json_markdown
9
7
  from framework.memory import BaseSessionMemory
8
+ from framework.observability import observe
10
9
  from agents.base import BaseAgent
11
10
  from agents.research.planner import ResearchPlanner
12
11
  from framework.mcp.revision_summary import (
@@ -0,0 +1,6 @@
1
+ """evalrun CLI package exports."""
2
+
3
+ from cli.main import create_parser
4
+ from cli.resolver import resolve_agent
5
+
6
+ __all__ = ["create_parser", "resolve_agent"]