evalrun 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. evalrun-0.4.0/LICENSE +0 -0
  2. evalrun-0.4.0/PKG-INFO +268 -0
  3. evalrun-0.4.0/README.md +252 -0
  4. evalrun-0.4.0/agents/__init__.py +6 -0
  5. evalrun-0.4.0/agents/auditor/__init__.py +13 -0
  6. evalrun-0.4.0/agents/auditor/budget_auditor.py +91 -0
  7. evalrun-0.4.0/agents/auditor/parser.py +139 -0
  8. evalrun-0.4.0/agents/auditor/prompts.py +64 -0
  9. evalrun-0.4.0/agents/auditor/schema.py +67 -0
  10. evalrun-0.4.0/agents/base.py +20 -0
  11. evalrun-0.4.0/agents/reflection/__init__.py +3 -0
  12. evalrun-0.4.0/agents/reflection/agent.py +77 -0
  13. evalrun-0.4.0/agents/reflection/prompts.py +18 -0
  14. evalrun-0.4.0/agents/research/__init__.py +4 -0
  15. evalrun-0.4.0/agents/research/agent.py +45 -0
  16. evalrun-0.4.0/agents/research/planner.py +52 -0
  17. evalrun-0.4.0/agents/research/prompts.py +14 -0
  18. evalrun-0.4.0/agents/support/__init__.py +5 -0
  19. evalrun-0.4.0/agents/support/triage_agent.py +45 -0
  20. evalrun-0.4.0/agents/travel/__init__.py +11 -0
  21. evalrun-0.4.0/agents/travel/agent.py +377 -0
  22. evalrun-0.4.0/agents/travel/prompts.py +30 -0
  23. evalrun-0.4.0/agents/travel/session.py +110 -0
  24. evalrun-0.4.0/cli/__init__.py +6 -0
  25. evalrun-0.4.0/cli/demo.py +47 -0
  26. evalrun-0.4.0/cli/formatter.py +93 -0
  27. evalrun-0.4.0/cli/html_reporter.py +647 -0
  28. evalrun-0.4.0/cli/main.py +423 -0
  29. evalrun-0.4.0/cli/progress.py +38 -0
  30. evalrun-0.4.0/cli/resolver.py +99 -0
  31. evalrun-0.4.0/evalrun.egg-info/PKG-INFO +268 -0
  32. evalrun-0.4.0/evalrun.egg-info/SOURCES.txt +130 -0
  33. evalrun-0.4.0/evalrun.egg-info/dependency_links.txt +1 -0
  34. evalrun-0.4.0/evalrun.egg-info/entry_points.txt +2 -0
  35. evalrun-0.4.0/evalrun.egg-info/requires.txt +7 -0
  36. evalrun-0.4.0/evalrun.egg-info/top_level.txt +4 -0
  37. evalrun-0.4.0/framework/__init__.py +70 -0
  38. evalrun-0.4.0/framework/core/__init__.py +17 -0
  39. evalrun-0.4.0/framework/core/adapters.py +118 -0
  40. evalrun-0.4.0/framework/core/contracts.py +88 -0
  41. evalrun-0.4.0/framework/core/suite.py +44 -0
  42. evalrun-0.4.0/framework/evaluation/__init__.py +22 -0
  43. evalrun-0.4.0/framework/evaluation/base.py +29 -0
  44. evalrun-0.4.0/framework/evaluation/dimensions.py +7 -0
  45. evalrun-0.4.0/framework/evaluation/engine.py +110 -0
  46. evalrun-0.4.0/framework/evaluation/evaluators/__init__.py +7 -0
  47. evalrun-0.4.0/framework/evaluation/evaluators/adaptability.py +27 -0
  48. evalrun-0.4.0/framework/evaluation/evaluators/base_llm.py +104 -0
  49. evalrun-0.4.0/framework/evaluation/evaluators/constraint.py +27 -0
  50. evalrun-0.4.0/framework/evaluation/evaluators/information_accuracy.py +41 -0
  51. evalrun-0.4.0/framework/evaluation/evaluators/personalization.py +27 -0
  52. evalrun-0.4.0/framework/evaluation/evaluators/planning.py +27 -0
  53. evalrun-0.4.0/framework/evaluation/evaluators/support.py +81 -0
  54. evalrun-0.4.0/framework/evaluation/prompts/__init__.py +11 -0
  55. evalrun-0.4.0/framework/evaluation/prompts/adaptability.py +57 -0
  56. evalrun-0.4.0/framework/evaluation/prompts/base.py +52 -0
  57. evalrun-0.4.0/framework/evaluation/prompts/constraint.py +41 -0
  58. evalrun-0.4.0/framework/evaluation/prompts/information_accuracy.py +79 -0
  59. evalrun-0.4.0/framework/evaluation/prompts/personalization.py +57 -0
  60. evalrun-0.4.0/framework/evaluation/prompts/planning.py +61 -0
  61. evalrun-0.4.0/framework/evaluation/runner.py +363 -0
  62. evalrun-0.4.0/framework/evaluation/testing.py +25 -0
  63. evalrun-0.4.0/framework/exceptions.py +49 -0
  64. evalrun-0.4.0/framework/llms/__init__.py +8 -0
  65. evalrun-0.4.0/framework/llms/base.py +37 -0
  66. evalrun-0.4.0/framework/llms/factory.py +38 -0
  67. evalrun-0.4.0/framework/llms/gemini.py +85 -0
  68. evalrun-0.4.0/framework/llms/mock.py +25 -0
  69. evalrun-0.4.0/framework/llms/openai.py +94 -0
  70. evalrun-0.4.0/framework/llms/openai_compatible.py +139 -0
  71. evalrun-0.4.0/framework/mcp/__init__.py +20 -0
  72. evalrun-0.4.0/framework/mcp/client.py +62 -0
  73. evalrun-0.4.0/framework/mcp/constraints.py +125 -0
  74. evalrun-0.4.0/framework/mcp/revision_summary.py +122 -0
  75. evalrun-0.4.0/framework/mcp/server.py +49 -0
  76. evalrun-0.4.0/framework/memory/__init__.py +3 -0
  77. evalrun-0.4.0/framework/memory/base.py +17 -0
  78. evalrun-0.4.0/framework/models.py +83 -0
  79. evalrun-0.4.0/framework/parser.py +45 -0
  80. evalrun-0.4.0/framework/parsers/__init__.py +12 -0
  81. evalrun-0.4.0/framework/parsers/frontmatter.py +28 -0
  82. evalrun-0.4.0/framework/parsers/mapper.py +66 -0
  83. evalrun-0.4.0/framework/parsers/markdown.py +122 -0
  84. evalrun-0.4.0/framework/parsers/transformers.py +112 -0
  85. evalrun-0.4.0/framework/profiles/__init__.py +28 -0
  86. evalrun-0.4.0/framework/profiles/registry.py +104 -0
  87. evalrun-0.4.0/framework/profiles/support.py +27 -0
  88. evalrun-0.4.0/framework/profiles/travel.py +78 -0
  89. evalrun-0.4.0/framework/regression/__init__.py +17 -0
  90. evalrun-0.4.0/framework/regression/comparator.py +273 -0
  91. evalrun-0.4.0/framework/regression/loader.py +145 -0
  92. evalrun-0.4.0/framework/sdk.py +151 -0
  93. evalrun-0.4.0/framework/utils.py +41 -0
  94. evalrun-0.4.0/framework/verification/__init__.py +13 -0
  95. evalrun-0.4.0/framework/verification/base.py +32 -0
  96. evalrun-0.4.0/framework/verification/extractor.py +105 -0
  97. evalrun-0.4.0/framework/verification/local.py +122 -0
  98. evalrun-0.4.0/framework/verification/models.py +112 -0
  99. evalrun-0.4.0/framework/verification/pipeline.py +36 -0
  100. evalrun-0.4.0/framework/verification/prompts.py +24 -0
  101. evalrun-0.4.0/framework/verification/utils.py +54 -0
  102. evalrun-0.4.0/pyproject.toml +29 -0
  103. evalrun-0.4.0/setup.cfg +4 -0
  104. evalrun-0.4.0/tests/test_adapters.py +153 -0
  105. evalrun-0.4.0/tests/test_agent_collaboration.py +121 -0
  106. evalrun-0.4.0/tests/test_cli.py +229 -0
  107. evalrun-0.4.0/tests/test_config_workflow.py +57 -0
  108. evalrun-0.4.0/tests/test_custom_profiles.py +128 -0
  109. evalrun-0.4.0/tests/test_engine.py +113 -0
  110. evalrun-0.4.0/tests/test_evaluators.py +144 -0
  111. evalrun-0.4.0/tests/test_generic_core.py +155 -0
  112. evalrun-0.4.0/tests/test_independent_auditor.py +277 -0
  113. evalrun-0.4.0/tests/test_mcp_agent_integration.py +192 -0
  114. evalrun-0.4.0/tests/test_mcp_constraints.py +111 -0
  115. evalrun-0.4.0/tests/test_openai_compatible.py +121 -0
  116. evalrun-0.4.0/tests/test_parser.py +82 -0
  117. evalrun-0.4.0/tests/test_reflection_agent.py +35 -0
  118. evalrun-0.4.0/tests/test_regression.py +254 -0
  119. evalrun-0.4.0/tests/test_research_agent.py +38 -0
  120. evalrun-0.4.0/tests/test_revision_summary.py +55 -0
  121. evalrun-0.4.0/tests/test_runner.py +177 -0
  122. evalrun-0.4.0/tests/test_runner_v3.py +98 -0
  123. evalrun-0.4.0/tests/test_sdk.py +88 -0
  124. evalrun-0.4.0/tests/test_session_memory.py +88 -0
  125. evalrun-0.4.0/tests/test_smoke_cli.py +85 -0
  126. evalrun-0.4.0/tests/test_support_domain.py +32 -0
  127. evalrun-0.4.0/tests/test_ui.py +147 -0
  128. evalrun-0.4.0/tests/test_usability_validation.py +134 -0
  129. evalrun-0.4.0/tests/test_utils.py +14 -0
  130. evalrun-0.4.0/tests/test_verification.py +117 -0
  131. evalrun-0.4.0/ui/__init__.py +1 -0
  132. evalrun-0.4.0/ui/server.py +255 -0
evalrun-0.4.0/LICENSE ADDED
File without changes
evalrun-0.4.0/PKG-INFO ADDED
@@ -0,0 +1,268 @@
1
+ Metadata-Version: 2.4
2
+ Name: evalrun
3
+ Version: 0.4.0
4
+ Summary: Provider-agnostic evaluation and regression-testing toolkit for tool-using AI agents.
5
+ Requires-Python: >=3.10
6
+ Description-Content-Type: text/markdown
7
+ License-File: LICENSE
8
+ Requires-Dist: openai>=1.0.0
9
+ Requires-Dist: requests>=2.28.0
10
+ Requires-Dist: pydantic>=2.0.0
11
+ Requires-Dist: python-frontmatter>=1.0.0
12
+ Requires-Dist: PyYAML>=6.0
13
+ Requires-Dist: langfuse>=2.0.0
14
+ Requires-Dist: fastmcp>=0.1.0
15
+ Dynamic: license-file
16
+
17
+ # evalrun
18
+
19
+ > **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
20
+
21
+ `evalrun` is a local-first toolkit for testing AI agents. You bring the agent, model, and API key; `evalrun` runs the checks on your machine, saves the evidence locally, and tells you whether the result should be released.
22
+
23
+ The toolkit works with hosted APIs and models running on your own computer. It does not host models, provide API credits, provision GPUs, or require a hosted account.
24
+
25
+ ---
26
+
27
+ ## 🌟 Key Capabilities
28
+
29
+ - **3-Tier Release Gatekeeping**:
30
+ - **Evaluator Thresholds**: Qualitative dimension scoring (0–100) via LLM judges.
31
+ - **Independent Auditor Gate**: Hard financial, policy, and math validation (blocks releases on unbudgeted items or currency hallucinations).
32
+ - **Baseline Regression Gate**: Automatically detects score drops ($\Delta \text{score}$) against stored baselines.
33
+ - **Local-First & Multi-Model**: Compatible with hosted APIs (OpenAI GPT-5.6, NVIDIA NIM, Gemini) and local model servers (vLLM, Ollama, LM Studio) on `http://localhost:8000/v1`.
34
+ - **Standalone HTML Review Reports**: Interactive local HTML report with failure quick-jump bars, auto-opened failing cards, search/filter controls, visual score progress bars, and raw model output inspection.
35
+ - **CI Exit Code Contract**:
36
+ - `0`: All scenarios passed evaluator, auditor gate passed, and no regression detected.
37
+ - `1`: Release blocked due to evaluator threshold failure, auditor violation, or baseline score regression.
38
+ - `2`: Runtime error, missing file, or invalid configuration.
39
+
40
+ ---
41
+
42
+ ## 🚀 Quick Start
43
+
44
+ ### 1. Installation
45
+
46
+ ```bash
47
+ git clone https://github.com/imshivamb/agent-eval-platform.git
48
+ cd agent-eval-platform
49
+ pip install -e .
50
+ ```
51
+
52
+ If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
53
+
54
+ ```bash
55
+ source .venv/bin/activate
56
+ python -m pip install -e .
57
+ rehash # zsh only, refreshes the command cache
58
+ ```
59
+
60
+ ### 2. Zero-Cost Offline Demo
61
+
62
+ Try EvalRun without an API key or network request:
63
+
64
+ ```bash
65
+ evalrun demo
66
+ open results/demo/report.html # macOS
67
+ ```
68
+
69
+ ### 3. Single Scenario Run (Hosted Model)
70
+
71
+ ```bash
72
+ export OPENAI_API_KEY="sk-proj-..."
73
+
74
+ evalrun run \
75
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
76
+ --agent agents.travel:TravelPlanningAgent \
77
+ --model gpt-5.6-terra \
78
+ --judge-model gpt-5.6-terra \
79
+ --output results/run-001
80
+ ```
81
+
82
+ ### Choosing a model endpoint
83
+
84
+ EvalRun does not lock you to one model provider. The `--base-url` value is the address where the model accepts OpenAI-compatible requests:
85
+
86
+ | Provider | Base URL example |
87
+ | --- | --- |
88
+ | OpenAI-compatible hosted service | `https://api.openai.com/v1` |
89
+ | Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` |
90
+ | OpenRouter | `https://openrouter.ai/api/v1` |
91
+ | Local vLLM, Ollama, or LM Studio | `http://localhost:8000/v1` |
92
+
93
+ The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
94
+
95
+ For example, a Gemini run can be written as:
96
+
97
+ ```bash
98
+ export GEMINI_API_KEY="your-key"
99
+
100
+ evalrun run \
101
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
102
+ --agent agents.travel:TravelPlanningAgent \
103
+ --model gemini-3.7-flash \
104
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
105
+ --api-key "$GEMINI_API_KEY" \
106
+ --judge-model gemini-3.7-flash \
107
+ --output results/gemini-run
108
+ ```
109
+
110
+ ### 4. Local Model Server Run (vLLM / Ollama)
111
+
112
+ ```bash
113
+ # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
114
+ evalrun run \
115
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
116
+ --agent agents.travel:TravelPlanningAgent \
117
+ --model qwen2.5-72b-instruct \
118
+ --base-url http://localhost:8000/v1 \
119
+ --api-key EMPTY \
120
+ --judge-model gpt-5.6-terra \
121
+ --output results/local-run
122
+ ```
123
+
124
+ ### 5. Guided Local Web UI
125
+
126
+ Launch the zero-dependency local web interface:
127
+
128
+ ```bash
129
+ evalrun ui --port 8501
130
+ ```
131
+
132
+ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
133
+
134
+ ---
135
+
136
+ ## 🐍 Python SDK Usage
137
+
138
+ Integrate `evalrun` programmatically into Python automation pipelines:
139
+
140
+ ```python
141
+ from framework.sdk import evaluate, compare
142
+
143
+ # 1. Execute Benchmark Evaluation
144
+ results = evaluate(
145
+ scenario="evals/scenarios/travel-agent/budget-constrained-itinerary.md",
146
+ agent="agents.travel:TravelPlanningAgent",
147
+ model="gpt-5.6-terra",
148
+ judge_model="gpt-5.6-terra",
149
+ base_url="http://localhost:8000/v1",
150
+ )
151
+
152
+ # 2. Compare Candidate Results against Baseline
153
+ report = compare(
154
+ candidate_results=results,
155
+ baseline="results/run-001",
156
+ max_overall_drop=5.0,
157
+ )
158
+
159
+ if report.release_blocked:
160
+ print(f"RELEASE BLOCKED: {report.summary['blocked_reason']}")
161
+ ```
162
+
163
+ ---
164
+
165
+ ## 📊 Baseline Regression Testing
166
+
167
+ Compare a candidate prompt, model version, or code change against a prior baseline run:
168
+
169
+ ```bash
170
+ evalrun run \
171
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
172
+ --agent agents.travel:TravelPlanningAgent \
173
+ --model gpt-5.6-terra \
174
+ --judge-model gpt-5.6-terra \
175
+ --baseline results/run-001 \
176
+ --max-regression 5.0 \
177
+ --max-dimension-regression 10.0 \
178
+ --output results/candidate-run
179
+ ```
180
+
181
+ ### Generated Artifacts
182
+
183
+ - `manifest.json`: Execution metadata with redacted API credentials.
184
+ - `regression_report.json`: Machine-readable score deltas ($\Delta \text{score}$) and gate decisions.
185
+ - `report.html`: Standalone interactive HTML report for human inspection.
186
+
187
+ ---
188
+
189
+ ## Distribution and deployment
190
+
191
+ There is no central service to deploy for the current product. The recommended path is:
192
+
193
+ 1. Publish the repository on GitHub.
194
+ 2. Add tagged releases and a clear quick-start guide.
195
+ 3. Users install it locally with `pip install -e .` from a clone.
196
+ 4. Later publish the package to PyPI so users can run `pip install evalrun`.
197
+ 5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
198
+ 6. Users provide their own hosted-model API keys or run their own local model server.
199
+
200
+ The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
201
+
202
+ ### Credential promise
203
+
204
+ EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
205
+
206
+ ## How a run works
207
+
208
+ ```text
209
+ Choose a scenario
210
+ ↓
211
+ Run your agent with your chosen model
212
+ ↓
213
+ Score the result against the criteria
214
+ ↓
215
+ Run the independent auditor
216
+ ↓
217
+ Compare with a previous run, if supplied
218
+ ↓
219
+ Write JSON and HTML evidence
220
+ ↓
221
+ Return PASS or BLOCK
222
+ ```
223
+
224
+ The command line is useful for repeatable runs, the Python SDK is useful inside scripts and pipelines, and the local browser interface is useful when you prefer a form. All three use the same evaluation engine.
225
+
226
+ ## Add your own evaluation scenario
227
+
228
+ Scenarios are ordinary Markdown files. Copy [`templates/scenario_template.md`](templates/scenario_template.md), edit the prompt and criteria, and save the file under a folder such as `evals/scenarios/my-domain/my-scenario.md`.
229
+
230
+ Each scenario defines:
231
+
232
+ - the request sent to the agent;
233
+ - hard constraints that must not be violated;
234
+ - the expected behavior;
235
+ - the dimensions to score;
236
+ - what counts as a pass or failure.
237
+
238
+ Run one custom scenario with `--scenario`:
239
+
240
+ ```bash
241
+ evalrun run \
242
+ --scenario evals/scenarios/my-domain/my-scenario.md \
243
+ --agent my_agent:MyAgent \
244
+ --model gemini-3.7-flash \
245
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
246
+ --api-key "$GEMINI_API_KEY" \
247
+ --judge-model gemini-3.7-flash \
248
+ --output results/my-scenario
249
+ ```
250
+
251
+ Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
252
+
253
+ ## 🏗️ Architecture & Documentation
254
+
255
+ For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
256
+
257
+ - Hosted Models Guide: [`docs/quickstart-hosted.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/quickstart-hosted.md)
258
+ - Local Models Guide: [`docs/quickstart-local.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/quickstart-local.md)
259
+ - CLI Specification: [`docs/phase4-local-cli-design.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/phase4-local-cli-design.md)
260
+ - Regression Engine Design: [`docs/phase5-regression-gates-design.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/phase5-regression-gates-design.md)
261
+
262
+ ---
263
+
264
+ ## ⚠️ Limitations & Reproducibility Guidelines
265
+
266
+ - **Judge Variance**: LLM judge evaluations can exhibit non-zero variance. For baseline regression testing, fix model versions and set deterministic sampling parameters where available.
267
+ - **Local Model Requirements**: Local evaluation throughput depends on server VRAM and concurrency settings. Ensure your local server handles parallel requests cleanly.
268
+ - **Credential Security**: Credentials in `manifest.json` and `regression_report.json` are automatically redacted into `"[REDACTED]"`. Never commit unredacted API keys.
@@ -0,0 +1,252 @@
1
+ # evalrun
2
+
3
+ > **Quality Score $\neq$ Release Decision.** An LLM evaluator can rate an itinerary 95/100 while an independent auditor blocks it for hard financial violations.
4
+
5
+ `evalrun` is a local-first toolkit for testing AI agents. You bring the agent, model, and API key; `evalrun` runs the checks on your machine, saves the evidence locally, and tells you whether the result should be released.
6
+
7
+ The toolkit works with hosted APIs and models running on your own computer. It does not host models, provide API credits, provision GPUs, or require a hosted account.
8
+
9
+ ---
10
+
11
+ ## 🌟 Key Capabilities
12
+
13
+ - **3-Tier Release Gatekeeping**:
14
+ - **Evaluator Thresholds**: Qualitative dimension scoring (0–100) via LLM judges.
15
+ - **Independent Auditor Gate**: Hard financial, policy, and math validation (blocks releases on unbudgeted items or currency hallucinations).
16
+ - **Baseline Regression Gate**: Automatically detects score drops ($\Delta \text{score}$) against stored baselines.
17
+ - **Local-First & Multi-Model**: Compatible with hosted APIs (OpenAI GPT-5.6, NVIDIA NIM, Gemini) and local model servers (vLLM, Ollama, LM Studio) on `http://localhost:8000/v1`.
18
+ - **Standalone HTML Review Reports**: Interactive local HTML report with failure quick-jump bars, auto-opened failing cards, search/filter controls, visual score progress bars, and raw model output inspection.
19
+ - **CI Exit Code Contract**:
20
+ - `0`: All scenarios passed evaluator, auditor gate passed, and no regression detected.
21
+ - `1`: Release blocked due to evaluator threshold failure, auditor violation, or baseline score regression.
22
+ - `2`: Runtime error, missing file, or invalid configuration.
23
+
24
+ ---
25
+
26
+ ## 🚀 Quick Start
27
+
28
+ ### 1. Installation
29
+
30
+ ```bash
31
+ git clone https://github.com/imshivamb/agent-eval-platform.git
32
+ cd agent-eval-platform
33
+ pip install -e .
34
+ ```
35
+
36
+ If your shell says `evalrun: command not found`, activate the virtual environment and install the repository first:
37
+
38
+ ```bash
39
+ source .venv/bin/activate
40
+ python -m pip install -e .
41
+ rehash # zsh only, refreshes the command cache
42
+ ```
43
+
44
+ ### 2. Zero-Cost Offline Demo
45
+
46
+ Try EvalRun without an API key or network request:
47
+
48
+ ```bash
49
+ evalrun demo
50
+ open results/demo/report.html # macOS
51
+ ```
52
+
53
+ ### 3. Single Scenario Run (Hosted Model)
54
+
55
+ ```bash
56
+ export OPENAI_API_KEY="sk-proj-..."
57
+
58
+ evalrun run \
59
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
60
+ --agent agents.travel:TravelPlanningAgent \
61
+ --model gpt-5.6-terra \
62
+ --judge-model gpt-5.6-terra \
63
+ --output results/run-001
64
+ ```
65
+
66
+ ### Choosing a model endpoint
67
+
68
+ EvalRun does not lock you to one model provider. The `--base-url` value is the address where the model accepts OpenAI-compatible requests:
69
+
70
+ | Provider | Base URL example |
71
+ | --- | --- |
72
+ | OpenAI-compatible hosted service | `https://api.openai.com/v1` |
73
+ | Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` |
74
+ | OpenRouter | `https://openrouter.ai/api/v1` |
75
+ | Local vLLM, Ollama, or LM Studio | `http://localhost:8000/v1` |
76
+
77
+ The judge normally uses the same endpoint and API key as the target model. You only need `--judge-base-url` or `--judge-api-key` when the judge is hosted somewhere different. The endpoint URL is not a credential; it simply tells EvalRun where to send the request.
78
+
79
+ For example, a Gemini run can be written as:
80
+
81
+ ```bash
82
+ export GEMINI_API_KEY="your-key"
83
+
84
+ evalrun run \
85
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
86
+ --agent agents.travel:TravelPlanningAgent \
87
+ --model gemini-3.7-flash \
88
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
89
+ --api-key "$GEMINI_API_KEY" \
90
+ --judge-model gemini-3.7-flash \
91
+ --output results/gemini-run
92
+ ```
93
+
94
+ ### 4. Local Model Server Run (vLLM / Ollama)
95
+
96
+ ```bash
97
+ # Users host their own local OpenAI-compatible server at http://localhost:8000/v1
98
+ evalrun run \
99
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
100
+ --agent agents.travel:TravelPlanningAgent \
101
+ --model qwen2.5-72b-instruct \
102
+ --base-url http://localhost:8000/v1 \
103
+ --api-key EMPTY \
104
+ --judge-model gpt-5.6-terra \
105
+ --output results/local-run
106
+ ```
107
+
108
+ ### 5. Guided Local Web UI
109
+
110
+ Launch the zero-dependency local web interface:
111
+
112
+ ```bash
113
+ evalrun ui --port 8501
114
+ ```
115
+
116
+ Open `http://127.0.0.1:8501` in your browser to configure endpoints, select scenarios, run evaluations, view pass/fail/block verdicts, and launch interactive HTML reports.
117
+
118
+ ---
119
+
120
+ ## 🐍 Python SDK Usage
121
+
122
+ Integrate `evalrun` programmatically into Python automation pipelines:
123
+
124
+ ```python
125
+ from framework.sdk import evaluate, compare
126
+
127
+ # 1. Execute Benchmark Evaluation
128
+ results = evaluate(
129
+ scenario="evals/scenarios/travel-agent/budget-constrained-itinerary.md",
130
+ agent="agents.travel:TravelPlanningAgent",
131
+ model="gpt-5.6-terra",
132
+ judge_model="gpt-5.6-terra",
133
+ base_url="http://localhost:8000/v1",
134
+ )
135
+
136
+ # 2. Compare Candidate Results against Baseline
137
+ report = compare(
138
+ candidate_results=results,
139
+ baseline="results/run-001",
140
+ max_overall_drop=5.0,
141
+ )
142
+
143
+ if report.release_blocked:
144
+ print(f"RELEASE BLOCKED: {report.summary['blocked_reason']}")
145
+ ```
146
+
147
+ ---
148
+
149
+ ## 📊 Baseline Regression Testing
150
+
151
+ Compare a candidate prompt, model version, or code change against a prior baseline run:
152
+
153
+ ```bash
154
+ evalrun run \
155
+ --scenario evals/scenarios/travel-agent/budget-constrained-itinerary.md \
156
+ --agent agents.travel:TravelPlanningAgent \
157
+ --model gpt-5.6-terra \
158
+ --judge-model gpt-5.6-terra \
159
+ --baseline results/run-001 \
160
+ --max-regression 5.0 \
161
+ --max-dimension-regression 10.0 \
162
+ --output results/candidate-run
163
+ ```
164
+
165
+ ### Generated Artifacts
166
+
167
+ - `manifest.json`: Execution metadata with redacted API credentials.
168
+ - `regression_report.json`: Machine-readable score deltas ($\Delta \text{score}$) and gate decisions.
169
+ - `report.html`: Standalone interactive HTML report for human inspection.
170
+
171
+ ---
172
+
173
+ ## Distribution and deployment
174
+
175
+ There is no central service to deploy for the current product. The recommended path is:
176
+
177
+ 1. Publish the repository on GitHub.
178
+ 2. Add tagged releases and a clear quick-start guide.
179
+ 3. Users install it locally with `pip install -e .` from a clone.
180
+ 4. Later publish the package to PyPI so users can run `pip install evalrun`.
181
+ 5. Host documentation on GitHub Pages if useful; the evaluation engine itself remains local.
182
+ 6. Users provide their own hosted-model API keys or run their own local model server.
183
+
184
+ The local UI is intended for local use, not public internet deployment. A shared hosted deployment would be a separate project requiring authentication, secret management, isolation, and hosted execution.
185
+
186
+ ### Credential promise
187
+
188
+ EvalRun does not provide, collect, or store model credentials. A key is read by the local process and passed to the selected model endpoint for that run. Keys are never written to `manifest.json`, `report.html`, or `regression_report.json`; reports contain only `[REDACTED]`. Prefer environment variables such as `OPENAI_API_KEY`, `GEMINI_API_KEY`, or `OPENROUTER_API_KEY`. Avoid putting real keys directly in shell commands because your terminal may save command history.
189
+
190
+ ## How a run works
191
+
192
+ ```text
193
+ Choose a scenario
194
+ ↓
195
+ Run your agent with your chosen model
196
+ ↓
197
+ Score the result against the criteria
198
+ ↓
199
+ Run the independent auditor
200
+ ↓
201
+ Compare with a previous run, if supplied
202
+ ↓
203
+ Write JSON and HTML evidence
204
+ ↓
205
+ Return PASS or BLOCK
206
+ ```
207
+
208
+ The command line is useful for repeatable runs, the Python SDK is useful inside scripts and pipelines, and the local browser interface is useful when you prefer a form. All three use the same evaluation engine.
209
+
210
+ ## Add your own evaluation scenario
211
+
212
+ Scenarios are ordinary Markdown files. Copy [`templates/scenario_template.md`](templates/scenario_template.md), edit the prompt and criteria, and save the file under a folder such as `evals/scenarios/my-domain/my-scenario.md`.
213
+
214
+ Each scenario defines:
215
+
216
+ - the request sent to the agent;
217
+ - hard constraints that must not be violated;
218
+ - the expected behavior;
219
+ - the dimensions to score;
220
+ - what counts as a pass or failure.
221
+
222
+ Run one custom scenario with `--scenario`:
223
+
224
+ ```bash
225
+ evalrun run \
226
+ --scenario evals/scenarios/my-domain/my-scenario.md \
227
+ --agent my_agent:MyAgent \
228
+ --model gemini-3.7-flash \
229
+ --base-url https://generativelanguage.googleapis.com/v1beta/openai/ \
230
+ --api-key "$GEMINI_API_KEY" \
231
+ --judge-model gemini-3.7-flash \
232
+ --output results/my-scenario
233
+ ```
234
+
235
+ Or place multiple `.md` files in a folder and use `--suite path/to/folder`.
236
+
237
+ ## 🏗️ Architecture & Documentation
238
+
239
+ For detailed system component diagrams, dual-pass auditor sequence flows, and release gate decision trees, see [`docs/architecture.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/architecture.md).
240
+
241
+ - Hosted Models Guide: [`docs/quickstart-hosted.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/quickstart-hosted.md)
242
+ - Local Models Guide: [`docs/quickstart-local.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/quickstart-local.md)
243
+ - CLI Specification: [`docs/phase4-local-cli-design.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/phase4-local-cli-design.md)
244
+ - Regression Engine Design: [`docs/phase5-regression-gates-design.md`](file:///Users/shivam/Projects/AI/agent-eval-platform/docs/phase5-regression-gates-design.md)
245
+
246
+ ---
247
+
248
+ ## ⚠️ Limitations & Reproducibility Guidelines
249
+
250
+ - **Judge Variance**: LLM judge evaluations can exhibit non-zero variance. For baseline regression testing, fix model versions and set deterministic sampling parameters where available.
251
+ - **Local Model Requirements**: Local evaluation throughput depends on server VRAM and concurrency settings. Ensure your local server handles parallel requests cleanly.
252
+ - **Credential Security**: Credentials in `manifest.json` and `regression_report.json` are automatically redacted into `"[REDACTED]"`. Never commit unredacted API keys.
@@ -0,0 +1,6 @@
1
+ """Agents package representing subjects under evaluation."""
2
+
3
+ from agents.base import BaseAgent
4
+ from agents.research.agent import ResearchAgent
5
+ from agents.research.planner import ResearchPlanner
6
+ from agents.reflection.agent import ReflectionAgent
@@ -0,0 +1,13 @@
1
+ """Independent Budget Auditor package exports."""
2
+
3
+ from agents.auditor.schema import AuditReport, BudgetViolation, FailureCode
4
+ from agents.auditor.parser import AuditParser
5
+ from agents.auditor.budget_auditor import IndependentBudgetAuditor
6
+
7
+ __all__ = [
8
+ "AuditReport",
9
+ "AuditParser",
10
+ "BudgetViolation",
11
+ "FailureCode",
12
+ "IndependentBudgetAuditor",
13
+ ]
@@ -0,0 +1,91 @@
1
+ """Independent Budget Auditor implementation for financial verification."""
2
+
3
+ from typing import Optional, Any
4
+
5
+ from agents.base import BaseAgent
6
+ from agents.auditor.schema import AuditReport, BudgetViolation, FailureCode
7
+ from agents.auditor.prompts import AUDITOR_SYSTEM_PROMPT, format_auditor_user_prompt
8
+ from agents.auditor.parser import AuditParser
9
+ from framework.llms.base import BaseLLM, Message
10
+
11
+
12
+ class IndependentBudgetAuditor(BaseAgent):
13
+ """An independent budget auditing agent with zero shared memory of reflection state."""
14
+
15
+ def __init__(self, llm: BaseLLM, max_retries: int = 2):
16
+ self.llm = llm
17
+ self.max_retries = max_retries
18
+
19
+ def audit(
20
+ self,
21
+ scenario_prompt: str,
22
+ itinerary_content: str,
23
+ total_budget_inr: Optional[float] = None,
24
+ daily_budget_jpy: Optional[float] = None,
25
+ ) -> AuditReport:
26
+ """Independently audits an itinerary against scenario budget constraints.
27
+
28
+ Args:
29
+ scenario_prompt: Original scenario specification prompt.
30
+ itinerary_content: Finalized itinerary content to audit.
31
+ total_budget_inr: Optional overall total budget limit in INR.
32
+ daily_budget_jpy: Optional daily spend allowance limit in JPY.
33
+
34
+ Returns:
35
+ An AuditReport instance detailing PASS/BLOCK status and violations.
36
+ """
37
+ user_content = format_auditor_user_prompt(
38
+ scenario_prompt=scenario_prompt,
39
+ itinerary_content=itinerary_content,
40
+ total_budget_inr=total_budget_inr,
41
+ daily_budget_jpy=daily_budget_jpy,
42
+ )
43
+
44
+ messages = [
45
+ Message(role="system", content=AUDITOR_SYSTEM_PROMPT),
46
+ Message(role="user", content=user_content),
47
+ ]
48
+
49
+ last_error = ""
50
+ last_raw_response = ""
51
+
52
+ for attempt in range(self.max_retries + 1):
53
+ response = self.llm.generate(messages)
54
+ last_raw_response = response.text
55
+ report, parse_err = AuditParser.parse_and_validate(response.text)
56
+ if report is not None:
57
+ report.retries_attempted = attempt
58
+ return report
59
+
60
+ last_error = parse_err or "Unknown validation error"
61
+ if attempt < self.max_retries:
62
+ messages.append(Message(role="assistant", content=response.text))
63
+ messages.append(
64
+ Message(
65
+ role="user",
66
+ content=f"Your previous response was rejected due to validation failure: '{last_error}'. Please correct your output and return ONLY a valid JSON object matching the exact schema inside ```json ... ```.",
67
+ )
68
+ )
69
+
70
+ # Fallback default report if all retries fail
71
+ return AuditReport(
72
+ status="BLOCK",
73
+ audit_score=0.0,
74
+ violations=[
75
+ BudgetViolation(
76
+ violation_type=FailureCode.MATH_HALLUCINATION,
77
+ description=f"Auditor failed output validation: {last_error}",
78
+ )
79
+ ],
80
+ reasoning_summary=f"Audit failed due to model output validation failure: {last_error}",
81
+ audit_confidence=0.0,
82
+ parse_error=last_error,
83
+ retries_attempted=self.max_retries + 1,
84
+ raw_model_response=last_raw_response,
85
+ )
86
+
87
+ def run(self, prompt: str, **kwargs) -> Any:
88
+ """Standard AgentOutput wrapper for pipeline runner compatibility."""
89
+ itinerary = kwargs.get("itinerary_content", prompt)
90
+ scenario = kwargs.get("scenario_prompt", prompt)
91
+ return self.audit(scenario_prompt=scenario, itinerary_content=itinerary)