evalrun 0.4.1__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. evalrun-0.4.2/LICENSE +21 -0
  2. {evalrun-0.4.1/evalrun.egg-info → evalrun-0.4.2}/PKG-INFO +121 -7
  3. evalrun-0.4.1/PKG-INFO → evalrun-0.4.2/README.md +110 -20
  4. {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/agent.py +1 -1
  5. {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/agent.py +1 -2
  6. {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/planner.py +1 -2
  7. {evalrun-0.4.1 → evalrun-0.4.2}/agents/support/triage_agent.py +1 -2
  8. {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/agent.py +1 -2
  9. evalrun-0.4.2/cli/__init__.py +6 -0
  10. evalrun-0.4.2/cli/doctor.py +150 -0
  11. {evalrun-0.4.1 → evalrun-0.4.2}/cli/html_reporter.py +198 -61
  12. {evalrun-0.4.1 → evalrun-0.4.2}/cli/main.py +273 -2
  13. evalrun-0.4.2/cli/validator.py +199 -0
  14. evalrun-0.4.1/README.md → evalrun-0.4.2/evalrun.egg-info/PKG-INFO +134 -4
  15. {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/SOURCES.txt +7 -0
  16. {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/requires.txt +11 -0
  17. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/engine.py +1 -2
  18. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/base_llm.py +2 -0
  19. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/base.py +4 -0
  20. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/runner.py +2 -2
  21. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/openai_compatible.py +13 -2
  22. {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/__init__.py +4 -5
  23. {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/client.py +14 -2
  24. evalrun-0.4.2/framework/observability.py +26 -0
  25. {evalrun-0.4.1 → evalrun-0.4.2}/framework/sdk.py +48 -5
  26. {evalrun-0.4.1 → evalrun-0.4.2}/pyproject.toml +14 -3
  27. evalrun-0.4.2/tests/test_canonical_evidence.py +66 -0
  28. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_cli.py +70 -0
  29. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_custom_profiles.py +1 -1
  30. evalrun-0.4.2/tests/test_doctor.py +67 -0
  31. evalrun-0.4.2/tests/test_html_reporter.py +108 -0
  32. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_openai_compatible.py +1 -0
  33. evalrun-0.4.2/tests/test_sdk.py +168 -0
  34. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_ui.py +11 -2
  35. evalrun-0.4.2/tests/test_validator.py +205 -0
  36. {evalrun-0.4.1 → evalrun-0.4.2}/ui/server.py +66 -21
  37. evalrun-0.4.1/LICENSE +0 -0
  38. evalrun-0.4.1/cli/__init__.py +0 -6
  39. evalrun-0.4.1/tests/test_sdk.py +0 -88
  40. {evalrun-0.4.1 → evalrun-0.4.2}/agents/__init__.py +0 -0
  41. {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/__init__.py +0 -0
  42. {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/budget_auditor.py +0 -0
  43. {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/parser.py +0 -0
  44. {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/prompts.py +0 -0
  45. {evalrun-0.4.1 → evalrun-0.4.2}/agents/auditor/schema.py +0 -0
  46. {evalrun-0.4.1 → evalrun-0.4.2}/agents/base.py +0 -0
  47. {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/__init__.py +0 -0
  48. {evalrun-0.4.1 → evalrun-0.4.2}/agents/reflection/prompts.py +0 -0
  49. {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/__init__.py +0 -0
  50. {evalrun-0.4.1 → evalrun-0.4.2}/agents/research/prompts.py +0 -0
  51. {evalrun-0.4.1 → evalrun-0.4.2}/agents/support/__init__.py +0 -0
  52. {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/__init__.py +0 -0
  53. {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/prompts.py +0 -0
  54. {evalrun-0.4.1 → evalrun-0.4.2}/agents/travel/session.py +0 -0
  55. {evalrun-0.4.1 → evalrun-0.4.2}/cli/demo.py +0 -0
  56. {evalrun-0.4.1 → evalrun-0.4.2}/cli/formatter.py +0 -0
  57. {evalrun-0.4.1 → evalrun-0.4.2}/cli/progress.py +0 -0
  58. {evalrun-0.4.1 → evalrun-0.4.2}/cli/resolver.py +0 -0
  59. {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/dependency_links.txt +0 -0
  60. {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/entry_points.txt +0 -0
  61. {evalrun-0.4.1 → evalrun-0.4.2}/evalrun.egg-info/top_level.txt +0 -0
  62. {evalrun-0.4.1 → evalrun-0.4.2}/framework/__init__.py +0 -0
  63. {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/__init__.py +0 -0
  64. {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/adapters.py +0 -0
  65. {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/contracts.py +0 -0
  66. {evalrun-0.4.1 → evalrun-0.4.2}/framework/core/suite.py +0 -0
  67. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/__init__.py +0 -0
  68. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/base.py +0 -0
  69. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/dimensions.py +0 -0
  70. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/__init__.py +0 -0
  71. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/adaptability.py +0 -0
  72. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/constraint.py +0 -0
  73. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/information_accuracy.py +0 -0
  74. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/personalization.py +0 -0
  75. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/planning.py +0 -0
  76. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/evaluators/support.py +0 -0
  77. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/__init__.py +0 -0
  78. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/adaptability.py +0 -0
  79. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/constraint.py +0 -0
  80. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/information_accuracy.py +0 -0
  81. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/personalization.py +0 -0
  82. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/prompts/planning.py +0 -0
  83. {evalrun-0.4.1 → evalrun-0.4.2}/framework/evaluation/testing.py +0 -0
  84. {evalrun-0.4.1 → evalrun-0.4.2}/framework/exceptions.py +0 -0
  85. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/__init__.py +0 -0
  86. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/base.py +0 -0
  87. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/factory.py +0 -0
  88. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/gemini.py +0 -0
  89. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/mock.py +0 -0
  90. {evalrun-0.4.1 → evalrun-0.4.2}/framework/llms/openai.py +0 -0
  91. {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/constraints.py +0 -0
  92. {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/revision_summary.py +0 -0
  93. {evalrun-0.4.1 → evalrun-0.4.2}/framework/mcp/server.py +0 -0
  94. {evalrun-0.4.1 → evalrun-0.4.2}/framework/memory/__init__.py +0 -0
  95. {evalrun-0.4.1 → evalrun-0.4.2}/framework/memory/base.py +0 -0
  96. {evalrun-0.4.1 → evalrun-0.4.2}/framework/models.py +0 -0
  97. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parser.py +0 -0
  98. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/__init__.py +0 -0
  99. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/frontmatter.py +0 -0
  100. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/mapper.py +0 -0
  101. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/markdown.py +0 -0
  102. {evalrun-0.4.1 → evalrun-0.4.2}/framework/parsers/transformers.py +0 -0
  103. {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/__init__.py +0 -0
  104. {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/registry.py +0 -0
  105. {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/support.py +0 -0
  106. {evalrun-0.4.1 → evalrun-0.4.2}/framework/profiles/travel.py +0 -0
  107. {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/__init__.py +0 -0
  108. {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/comparator.py +0 -0
  109. {evalrun-0.4.1 → evalrun-0.4.2}/framework/regression/loader.py +0 -0
  110. {evalrun-0.4.1 → evalrun-0.4.2}/framework/utils.py +0 -0
  111. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/__init__.py +0 -0
  112. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/base.py +0 -0
  113. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/extractor.py +0 -0
  114. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/local.py +0 -0
  115. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/models.py +0 -0
  116. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/pipeline.py +0 -0
  117. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/prompts.py +0 -0
  118. {evalrun-0.4.1 → evalrun-0.4.2}/framework/verification/utils.py +0 -0
  119. {evalrun-0.4.1 → evalrun-0.4.2}/setup.cfg +0 -0
  120. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_adapters.py +0 -0
  121. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_agent_collaboration.py +0 -0
  122. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_config_workflow.py +0 -0
  123. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_engine.py +0 -0
  124. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_evaluators.py +0 -0
  125. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_generic_core.py +0 -0
  126. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_independent_auditor.py +0 -0
  127. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_mcp_agent_integration.py +0 -0
  128. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_mcp_constraints.py +0 -0
  129. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_parser.py +0 -0
  130. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_reflection_agent.py +0 -0
  131. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_regression.py +0 -0
  132. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_research_agent.py +0 -0
  133. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_revision_summary.py +0 -0
  134. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_runner.py +0 -0
  135. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_runner_v3.py +0 -0
  136. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_session_memory.py +0 -0
  137. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_smoke_cli.py +0 -0
  138. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_support_domain.py +0 -0
  139. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_usability_validation.py +0 -0
  140. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_utils.py +0 -0
  141. {evalrun-0.4.1 → evalrun-0.4.2}/tests/test_verification.py +0 -0
  142. {evalrun-0.4.1 → evalrun-0.4.2}/ui/__init__.py +0 -0
evalrun-0.4.2/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Shivam Bhardwaj
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: evalrun
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -10,8 +10,16 @@ Requires-Dist: requests>=2.28.0
10
10
  Requires-Dist: pydantic>=2.0.0
11
11
  Requires-Dist: python-frontmatter>=1.0.0
12
12
  Requires-Dist: PyYAML>=6.0
13
- Requires-Dist: langfuse>=2.0.0
14
- Requires-Dist: fastmcp>=0.1.0
13
+ Requires-Dist: markdown-it-py>=3.0.0
14
+ Provides-Extra: observability
15
+ Requires-Dist: langfuse>=2.0.0; extra == "observability"
16
+ Provides-Extra: mcp
17
+ Requires-Dist: fastmcp>=0.1.0; extra == "mcp"
18
+ Requires-Dist: mcp<2,>=1.27; extra == "mcp"
19
+ Provides-Extra: all
20
+ Requires-Dist: langfuse>=2.0.0; extra == "all"
21
+ Requires-Dist: fastmcp>=0.1.0; extra == "all"
22
+ Requires-Dist: mcp<2,>=1.27; extra == "all"
15
23
  Dynamic: license-file
16
24
 
17
25
  # evalrun
@@ -60,6 +68,13 @@ cd agent-eval-platform
60
68
  pip install -e .
61
69
  ```
62
70
 
71
+ Optional integrations are deliberately not required for the core CLI and SDK:
72
+
73
+ ```bash
74
+ pip install "evalrun[observability]" # optional Langfuse tracing
75
+ pip install "evalrun[mcp]" # optional MCP validation tools
76
+ ```
77
+
63
78
  If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
64
79
 
65
80
  ```bash
@@ -77,7 +92,38 @@ evalrun demo
77
92
  open results/demo/report.html # macOS
78
93
  ```
79
94
 
80
- ### 3. Single Scenario Run (Hosted Model)
95
+ The offline demo is the safest way to confirm that installation works. It makes
96
+ no model requests, needs no API key, and executes only EvalRun's built-in demo
97
+ agent.
98
+
99
+ ### 3. Create and check your own evaluation workspace
100
+
101
+ Create starter files in a new folder:
102
+
103
+ ```bash
104
+ evalrun init my-evaluation
105
+ cd my-evaluation
106
+ evalrun validate scenario.md
107
+ ```
108
+
109
+ Edit `scenario.md` to describe the user request, hard constraints, expected
110
+ behavior, scoring dimensions, pass criteria, and failure conditions. The
111
+ scenario is ordinary Markdown; you do not need to change EvalRun's source code.
112
+ Use `evalrun validate` before spending money on a model run. It reports the
113
+ exact missing heading or unsupported dimension and returns exit code `2` when
114
+ the file needs fixing.
115
+
116
+ Run environment diagnostics at any time with:
117
+
118
+ ```bash
119
+ evalrun doctor
120
+ ```
121
+
122
+ `doctor` does not require an API key. Missing keys and unavailable model servers
123
+ are reported as information, while invalid Python or unwritable output paths are
124
+ reported as failures.
125
+
126
+ ### 4. Single Scenario Run (Hosted Model)
81
127
 
82
128
  ```bash
83
129
  export OPENAI_API_KEY="sk-proj-..."
@@ -103,6 +149,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
103
149
 
104
150
  The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
105
151
 
152
+ For a target and judge at different providers, configure both explicitly:
153
+
154
+ ```bash
155
+ evalrun run \
156
+ --scenario path/to/scenario.md \
157
+ --agent my_agent:MyAgent \
158
+ --model gemini-3.7-flash \
159
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
160
+ --api-key "$GEMINI_API_KEY" \
161
+ --judge-model stealth/ox-alpha \
162
+ --judge-base-url https://openrouter.ai/api/v1 \
163
+ --judge-api-key "$OPENROUTER_API_KEY" \
164
+ --output results/mixed-provider-run
165
+ ```
166
+
106
167
  For example, a Gemini run can be written as:
107
168
 
108
169
  ```bash
@@ -118,7 +179,7 @@ evalrun run \
118
179
  --output results/gemini-run
119
180
  ```
120
181
 
121
- ### 4. Local Model Server Run (vLLM / Ollama)
182
+ ### 5. Local Model Server Run (vLLM / Ollama)
122
183
 
123
184
  ```bash
124
185
  # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
@@ -132,7 +193,7 @@ evalrun run \
132
193
  --output results/local-run
133
194
  ```
134
195
 
135
- ### 5. Guided Local Web UI
196
+ ### 6. Guided Local Web UI
136
197
 
137
198
  Launch the zero-dependency local web interface:
138
199
 
@@ -142,6 +203,17 @@ evalrun ui --port 8501
142
203
 
143
204
  Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
144
205
 
206
+ The UI is a local convenience layer. It supports built-in agents and workspace
207
+ scenario paths. API keys are held in browser memory for the current request and
208
+ are not saved by EvalRun. Do not expose this server to the public internet and
209
+ do not paste untrusted Python code into an agent field. Complex or third-party
210
+ agents should be run through the CLI or SDK in the environment where their code
211
+ and dependencies are installed.
212
+
213
+ The UI intentionally permits only the built-in travel agent. Custom Python,
214
+ HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
215
+ agent specifications from a browser would allow local code execution.
216
+
145
217
  ---
146
218
 
147
219
  ## 🐍 Python SDK Usage
@@ -164,7 +236,7 @@ results = evaluate(
164
236
  report = compare(
165
237
  candidate_results=results,
166
238
  baseline="results/run-001",
167
- max_overall_drop=5.0,
239
+ max_regression=5.0,
168
240
  )
169
241
 
170
242
  if report.release_blocked:
@@ -274,6 +346,48 @@ evalrun run \
274
346
 
275
347
  Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
276
348
 
349
+ Validate a suite before running it:
350
+
351
+ ```bash
352
+ for scenario in evals/scenarios/my-domain/*.md; do
353
+ evalrun validate "$scenario" || exit 2
354
+ done
355
+ ```
356
+
357
+ ## Connect your own agent
358
+
359
+ EvalRun does not require agents to use a particular framework. The `--agent`
360
+ value selects one of three adapters:
361
+
362
+ | Agent form | Example | When to use |
363
+ | --- | --- | --- |
364
+ | Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
365
+ | HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
366
+ | CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
367
+
368
+ Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
369
+ must expose the adapter contract described in
370
+ [`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
371
+ complete syntax and examples.
372
+
373
+ For repeatable runs, put the same values in a JSON or TOML configuration file:
374
+
375
+ ```bash
376
+ evalrun run --config evalrun.json
377
+ ```
378
+
379
+ Command-line flags override values from the configuration file. Keep API keys
380
+ in environment variables and reference them from your shell; never commit them
381
+ to the config file.
382
+
383
+ ## Troubleshooting
384
+
385
+ - `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
386
+ - `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
387
+ - Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
388
+ - A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
389
+ - `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
390
+
277
391
  ## 🏗️ Architecture & Documentation
278
392
 
279
393
  For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
@@ -1,19 +1,3 @@
1
- Metadata-Version: 2.4
2
- Name: evalrun
3
- Version: 0.4.1
4
- Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
5
- Requires-Python: >=3.10
6
- Description-Content-Type: text/markdown
7
- License-File: LICENSE
8
- Requires-Dist: openai>=1.0.0
9
- Requires-Dist: requests>=2.28.0
10
- Requires-Dist: pydantic>=2.0.0
11
- Requires-Dist: python-frontmatter>=1.0.0
12
- Requires-Dist: PyYAML>=6.0
13
- Requires-Dist: langfuse>=2.0.0
14
- Requires-Dist: fastmcp>=0.1.0
15
- Dynamic: license-file
16
-
17
1
  # evalrun
18
2
 
19
3
  > **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
@@ -60,6 +44,13 @@ cd agent-eval-platform
60
44
  pip install -e .
61
45
  ```
62
46
 
47
+ Optional integrations are deliberately not required for the core CLI and SDK:
48
+
49
+ ```bash
50
+ pip install "evalrun[observability]" # optional Langfuse tracing
51
+ pip install "evalrun[mcp]" # optional MCP validation tools
52
+ ```
53
+
63
54
  If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
64
55
 
65
56
  ```bash
@@ -77,7 +68,38 @@ evalrun demo
77
68
  open results/demo/report.html # macOS
78
69
  ```
79
70
 
80
- ### 3. Single Scenario Run (Hosted Model)
71
+ The offline demo is the safest way to confirm that installation works. It makes
72
+ no model requests, needs no API key, and executes only EvalRun's built-in demo
73
+ agent.
74
+
75
+ ### 3. Create and check your own evaluation workspace
76
+
77
+ Create starter files in a new folder:
78
+
79
+ ```bash
80
+ evalrun init my-evaluation
81
+ cd my-evaluation
82
+ evalrun validate scenario.md
83
+ ```
84
+
85
+ Edit `scenario.md` to describe the user request, hard constraints, expected
86
+ behavior, scoring dimensions, pass criteria, and failure conditions. The
87
+ scenario is ordinary Markdown; you do not need to change EvalRun's source code.
88
+ Use `evalrun validate` before spending money on a model run. It reports the
89
+ exact missing heading or unsupported dimension and returns exit code `2` when
90
+ the file needs fixing.
91
+
92
+ Run environment diagnostics at any time with:
93
+
94
+ ```bash
95
+ evalrun doctor
96
+ ```
97
+
98
+ `doctor` does not require an API key. Missing keys and unavailable model servers
99
+ are reported as information, while invalid Python or unwritable output paths are
100
+ reported as failures.
101
+
102
+ ### 4. Single Scenario Run (Hosted Model)
81
103
 
82
104
  ```bash
83
105
  export OPENAI_API_KEY="sk-proj-..."
@@ -103,6 +125,21 @@ EvalRun does not lock you to one model provider. The `--base-url` value is the a
103
125
 
104
126
  The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
105
127
 
128
+ For a target and judge at different providers, configure both explicitly:
129
+
130
+ ```bash
131
+ evalrun run \
132
+ --scenario path/to/scenario.md \
133
+ --agent my_agent:MyAgent \
134
+ --model gemini-3.7-flash \
135
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
136
+ --api-key "$GEMINI_API_KEY" \
137
+ --judge-model stealth/ox-alpha \
138
+ --judge-base-url https://openrouter.ai/api/v1 \
139
+ --judge-api-key "$OPENROUTER_API_KEY" \
140
+ --output results/mixed-provider-run
141
+ ```
142
+
106
143
  For example, a Gemini run can be written as:
107
144
 
108
145
  ```bash
@@ -118,7 +155,7 @@ evalrun run \
118
155
  --output results/gemini-run
119
156
  ```
120
157
 
121
- ### 4. Local Model Server Run (vLLM / Ollama)
158
+ ### 5. Local Model Server Run (vLLM / Ollama)
122
159
 
123
160
  ```bash
124
161
  # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
@@ -132,7 +169,7 @@ evalrun run \
132
169
  --output results/local-run
133
170
  ```
134
171
 
135
- ### 5. Guided Local Web UI
172
+ ### 6. Guided Local Web UI
136
173
 
137
174
  Launch the zero-dependency local web interface:
138
175
 
@@ -142,6 +179,17 @@ evalrun ui --port 8501
142
179
 
143
180
  Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
144
181
 
182
+ The UI is a local convenience layer. It supports built-in agents and workspace
183
+ scenario paths. API keys are held in browser memory for the current request and
184
+ are not saved by EvalRun. Do not expose this server to the public internet and
185
+ do not paste untrusted Python code into an agent field. Complex or third-party
186
+ agents should be run through the CLI or SDK in the environment where their code
187
+ and dependencies are installed.
188
+
189
+ The UI intentionally permits only the built-in travel agent. Custom Python,
190
+ HTTP, and CLI agent adapters remain CLI/SDK-only because accepting arbitrary
191
+ agent specifications from a browser would allow local code execution.
192
+
145
193
  ---
146
194
 
147
195
  ## 🐍 Python SDK Usage
@@ -164,7 +212,7 @@ results = evaluate(
164
212
  report = compare(
165
213
  candidate_results=results,
166
214
  baseline="results/run-001",
167
- max_overall_drop=5.0,
215
+ max_regression=5.0,
168
216
  )
169
217
 
170
218
  if report.release_blocked:
@@ -274,6 +322,48 @@ evalrun run \
274
322
 
275
323
  Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
276
324
 
325
+ Validate a suite before running it:
326
+
327
+ ```bash
328
+ for scenario in evals/scenarios/my-domain/*.md; do
329
+ evalrun validate "$scenario" || exit 2
330
+ done
331
+ ```
332
+
333
+ ## Connect your own agent
334
+
335
+ EvalRun does not require agents to use a particular framework. The `--agent`
336
+ value selects one of three adapters:
337
+
338
+ | Agent form | Example | When to use |
339
+ | --- | --- | --- |
340
+ | Python import | `my_agent:MyAgent` | Your agent is installed/importable in the current environment. |
341
+ | HTTP endpoint | `http://127.0.0.1:8080/predict` | Your agent is running as a local service. |
342
+ | CLI process | `cli:python my_agent.py` | Your agent reads the scenario request from stdin and writes its answer to stdout. |
343
+
344
+ Python agents receive the target model adapter from EvalRun. HTTP and CLI agents
345
+ must expose the adapter contract described in
346
+ [`docs/architecture.md`](docs/architecture.md). Run `evalrun --help` for the
347
+ complete syntax and examples.
348
+
349
+ For repeatable runs, put the same values in a JSON or TOML configuration file:
350
+
351
+ ```bash
352
+ evalrun run --config evalrun.json
353
+ ```
354
+
355
+ Command-line flags override values from the configuration file. Keep API keys
356
+ in environment variables and reference them from your shell; never commit them
357
+ to the config file.
358
+
359
+ ## Troubleshooting
360
+
361
+ - `evalrun: command not found`: activate the virtual environment and run `python -m pip install evalrun` (or `pip install -e .` from a checkout), then run `rehash` in zsh.
362
+ - `Missing required section`: run `evalrun validate path/to/scenario.md` and add the named Markdown heading.
363
+ - Model authentication or endpoint errors: run `evalrun doctor`, confirm the provider's base URL, and check the matching environment variable.
364
+ - A high evaluator score with `BLOCK`: inspect the independent-auditor findings in `report.html`; a quality score is not a release decision.
365
+ - `report.html` is not created after an error: inspect the terminal error and `manifest.json`; failed executions are recorded as failed results rather than silently treated as a passing empty suite.
366
+
277
367
  ## 🏗️ Architecture & Documentation
278
368
 
279
369
  For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
@@ -1,10 +1,10 @@
1
1
  """Reflection agent implementation for auditing and critiquing planned itineraries."""
2
2
 
3
3
  from typing import Optional
4
- from langfuse import observe
5
4
  from framework.llms import BaseLLM, Message
6
5
  from framework.models import AgentOutput
7
6
  from framework.memory import BaseSessionMemory
7
+ from framework.observability import observe
8
8
  from agents.base import BaseAgent
9
9
  from .prompts import REFLECTION_SYSTEM_PROMPT
10
10
 
@@ -1,7 +1,6 @@
1
- from langfuse import observe
2
-
3
1
  from framework.llms import BaseLLM, Message
4
2
  from framework.models import AgentOutput
3
+ from framework.observability import observe
5
4
  from agents.base import BaseAgent
6
5
  from .prompts import RESEARCH_SYSTEM_PROMPT
7
6
 
@@ -2,8 +2,7 @@
2
2
 
3
3
  from framework.llms import BaseLLM, Message
4
4
  from framework.utils import parse_json_markdown
5
-
6
- from langfuse import observe
5
+ from framework.observability import observe
7
6
 
8
7
 
9
8
  class ResearchPlanner:
@@ -1,10 +1,9 @@
1
1
  """Customer-support triage agent used by the second-domain benchmark."""
2
2
 
3
- from langfuse import observe
4
-
5
3
  from agents.base import BaseAgent
6
4
  from framework.llms import BaseLLM, Message
7
5
  from framework.models import AgentOutput
6
+ from framework.observability import observe
8
7
 
9
8
 
10
9
  SUPPORT_TRIAGE_SYSTEM_PROMPT = """You are a production customer-support triage agent.
@@ -1,5 +1,3 @@
1
- from langfuse import observe
2
-
3
1
  import asyncio
4
2
  import json
5
3
  from typing import Any, Dict, Optional
@@ -7,6 +5,7 @@ from framework.llms import BaseLLM, Message
7
5
  from framework.models import AgentOutput
8
6
  from framework.utils import parse_json_markdown
9
7
  from framework.memory import BaseSessionMemory
8
+ from framework.observability import observe
10
9
  from agents.base import BaseAgent
11
10
  from agents.research.planner import ResearchPlanner
12
11
  from framework.mcp.revision_summary import (
@@ -0,0 +1,6 @@
1
+ """evalrun CLI package exports."""
2
+
3
+ from cli.main import create_parser
4
+ from cli.resolver import resolve_agent
5
+
6
+ __all__ = ["create_parser", "resolve_agent"]
@@ -0,0 +1,150 @@
1
+ """Diagnostic doctor module for evalrun doctor command."""
2
+
3
+ import importlib
4
+ import os
5
+ import sys
6
+ import urllib.error
7
+ import urllib.request
8
+ from pathlib import Path
9
+ from typing import Dict, List, Tuple
10
+
11
+
12
+ def run_doctor_checks() -> Tuple[bool, List[str]]:
13
+ """Runs system diagnostics for EvalRun environment.
14
+
15
+ Returns:
16
+ Tuple of (all_ok: bool, lines: List[str])
17
+ """
18
+ lines: List[str] = []
19
+ all_ok = True
20
+
21
+ lines.append("=====================================================================")
22
+ lines.append(" EVALRUN SYSTEM DOCTOR ")
23
+ lines.append("=====================================================================")
24
+
25
+ # 1. Python Version Check
26
+ py_ver = sys.version.split()[0]
27
+ py_ok = sys.version_info >= (3, 10)
28
+ if not py_ok:
29
+ all_ok = False
30
+ lines.append(f"[{'PASS' if py_ok else 'FAIL'}] Python Version: {py_ver} (Requirement: >= 3.10)")
31
+
32
+ # 2. EvalRun Version & Installation State
33
+ try:
34
+ from importlib.metadata import PackageNotFoundError, version
35
+ pkg_version = version("evalrun")
36
+ install_type = "Installed Package"
37
+ except (PackageNotFoundError, ImportError):
38
+ pkg_version = "unknown"
39
+ pyproject = Path(__file__).resolve().parents[1] / "pyproject.toml"
40
+ try:
41
+ try:
42
+ import tomllib
43
+ with pyproject.open("rb") as f:
44
+ metadata = tomllib.load(f)
45
+ except ModuleNotFoundError:
46
+ import re
47
+ metadata = {}
48
+ match = re.search(r'(?m)^version\s*=\s*["\']([^"\']+)["\']', pyproject.read_text(encoding="utf-8"))
49
+ if match:
50
+ metadata = {"project": {"version": match.group(1)}}
51
+ pkg_version = str(metadata.get("project", {}).get("version", "unknown"))
52
+ except Exception:
53
+ pass
54
+ install_type = "Local Source Checkout"
55
+
56
+ lines.append(f"[PASS] EvalRun Version: {pkg_version} ({install_type})")
57
+
58
+ # 3. Model API Key Environment Variables
59
+ env_keys = {
60
+ "OPENAI_API_KEY": os.getenv("OPENAI_API_KEY"),
61
+ "GEMINI_API_KEY": os.getenv("GEMINI_API_KEY"),
62
+ "OPENROUTER_API_KEY": os.getenv("OPENROUTER_API_KEY"),
63
+ "NVIDIA_API_KEY": os.getenv("NVIDIA_API_KEY"),
64
+ }
65
+ set_keys = [k for k, v in env_keys.items() if v]
66
+ if set_keys:
67
+ lines.append(f"[PASS] Model API Keys Configured: {', '.join(set_keys)}")
68
+ else:
69
+ lines.append("[INFO] Model API Keys Configured: None (Offline demo works without keys)")
70
+
71
+ # 4. Agent Importability
72
+ try:
73
+ mod = importlib.import_module("agents.travel")
74
+ agent_cls = getattr(mod, "TravelPlanningAgent", None)
75
+ agent_ok = agent_cls is not None
76
+ except Exception:
77
+ agent_ok = False
78
+
79
+ if agent_ok:
80
+ lines.append("[PASS] Built-in Agent Import: agents.travel:TravelPlanningAgent")
81
+ else:
82
+ lines.append("[WARN] Built-in Agent Import: agents.travel not found in python path")
83
+
84
+ # 5. Endpoint Reachability (OpenAI API)
85
+ skip_network = os.getenv("EVALRUN_SKIP_NETWORK_CHECKS") == "1"
86
+ openai_reach = False
87
+ if not skip_network:
88
+ try:
89
+ req = urllib.request.Request("https://api.openai.com/v1/models", headers={"User-Agent": "evalrun-doctor"})
90
+ with urllib.request.urlopen(req, timeout=3) as resp:
91
+ openai_reach = resp.status in (200, 401)
92
+ except urllib.error.HTTPError as e:
93
+ openai_reach = e.code in (401, 403, 200)
94
+ except Exception:
95
+ openai_reach = False
96
+
97
+ if openai_reach:
98
+ lines.append("[PASS] Hosted Endpoint Reachability: https://api.openai.com/v1")
99
+ elif skip_network:
100
+ lines.append("[INFO] Hosted Endpoint Reachability: skipped (EVALRUN_SKIP_NETWORK_CHECKS=1)")
101
+ else:
102
+ lines.append("[INFO] Hosted Endpoint Reachability: https://api.openai.com/v1 (Offline or unreachable)")
103
+
104
+ # 6. Local Model Server Availability
105
+ local_reach = False
106
+ if not skip_network:
107
+ for local_url in ["http://localhost:8000/v1/models", "http://localhost:11434/api/tags"]:
108
+ try:
109
+ req = urllib.request.Request(local_url)
110
+ with urllib.request.urlopen(req, timeout=2) as resp:
111
+ if resp.status in (200, 401, 403):
112
+ local_reach = True
113
+ break
114
+ except urllib.error.HTTPError as e:
115
+ if e.code in (200, 401, 403):
116
+ local_reach = True
117
+ break
118
+ except Exception:
119
+ pass
120
+
121
+ if local_reach:
122
+ lines.append("[PASS] Local Model Server: Active local LLM server detected")
123
+ elif skip_network:
124
+ lines.append("[INFO] Local Model Server: skipped (EVALRUN_SKIP_NETWORK_CHECKS=1)")
125
+ else:
126
+ lines.append("[INFO] Local Model Server: No active local LLM server detected on 8000/11434")
127
+
128
+ # 7. Write Access to Output Folder
129
+ output_dir = Path("./eval_results")
130
+ try:
131
+ output_dir.mkdir(parents=True, exist_ok=True)
132
+ test_file = output_dir / ".doctor_temp"
133
+ test_file.write_text("write_test", encoding="utf-8")
134
+ test_file.unlink()
135
+ write_ok = True
136
+ except Exception:
137
+ write_ok = False
138
+
139
+ if not write_ok:
140
+ all_ok = False
141
+ lines.append(f"[{'PASS' if write_ok else 'FAIL'}] Output Directory Write Access: {output_dir.resolve()}")
142
+
143
+ lines.append("=====================================================================")
144
+ if all_ok:
145
+ lines.append(" Doctor Verdict: SYSTEM READY FOR EVALUATION RUNS")
146
+ else:
147
+ lines.append(" Doctor Verdict: ISSUES DETECTED - PLEASE REVIEW WARNINGS ABOVE")
148
+ lines.append("=====================================================================")
149
+
150
+ return all_ok, lines