evalrun 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agents/__init__.py +6 -0
- agents/auditor/__init__.py +13 -0
- agents/auditor/budget_auditor.py +91 -0
- agents/auditor/parser.py +139 -0
- agents/auditor/prompts.py +64 -0
- agents/auditor/schema.py +67 -0
- agents/base.py +20 -0
- agents/reflection/__init__.py +3 -0
- agents/reflection/agent.py +77 -0
- agents/reflection/prompts.py +18 -0
- agents/research/__init__.py +4 -0
- agents/research/agent.py +45 -0
- agents/research/planner.py +52 -0
- agents/research/prompts.py +14 -0
- agents/support/__init__.py +5 -0
- agents/support/triage_agent.py +45 -0
- agents/travel/__init__.py +11 -0
- agents/travel/agent.py +377 -0
- agents/travel/prompts.py +30 -0
- agents/travel/session.py +110 -0
- cli/__init__.py +6 -0
- cli/demo.py +47 -0
- cli/formatter.py +93 -0
- cli/html_reporter.py +647 -0
- cli/main.py +423 -0
- cli/progress.py +38 -0
- cli/resolver.py +99 -0
- evalrun-0.4.0.dist-info/METADATA +268 -0
- evalrun-0.4.0.dist-info/RECORD +100 -0
- evalrun-0.4.0.dist-info/WHEEL +5 -0
- evalrun-0.4.0.dist-info/entry_points.txt +2 -0
- evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
- evalrun-0.4.0.dist-info/top_level.txt +4 -0
- framework/__init__.py +70 -0
- framework/core/__init__.py +17 -0
- framework/core/adapters.py +118 -0
- framework/core/contracts.py +88 -0
- framework/core/suite.py +44 -0
- framework/evaluation/__init__.py +22 -0
- framework/evaluation/base.py +29 -0
- framework/evaluation/dimensions.py +7 -0
- framework/evaluation/engine.py +110 -0
- framework/evaluation/evaluators/__init__.py +7 -0
- framework/evaluation/evaluators/adaptability.py +27 -0
- framework/evaluation/evaluators/base_llm.py +104 -0
- framework/evaluation/evaluators/constraint.py +27 -0
- framework/evaluation/evaluators/information_accuracy.py +41 -0
- framework/evaluation/evaluators/personalization.py +27 -0
- framework/evaluation/evaluators/planning.py +27 -0
- framework/evaluation/evaluators/support.py +81 -0
- framework/evaluation/prompts/__init__.py +11 -0
- framework/evaluation/prompts/adaptability.py +57 -0
- framework/evaluation/prompts/base.py +52 -0
- framework/evaluation/prompts/constraint.py +41 -0
- framework/evaluation/prompts/information_accuracy.py +79 -0
- framework/evaluation/prompts/personalization.py +57 -0
- framework/evaluation/prompts/planning.py +61 -0
- framework/evaluation/runner.py +363 -0
- framework/evaluation/testing.py +25 -0
- framework/exceptions.py +49 -0
- framework/llms/__init__.py +8 -0
- framework/llms/base.py +37 -0
- framework/llms/factory.py +38 -0
- framework/llms/gemini.py +85 -0
- framework/llms/mock.py +25 -0
- framework/llms/openai.py +94 -0
- framework/llms/openai_compatible.py +139 -0
- framework/mcp/__init__.py +20 -0
- framework/mcp/client.py +62 -0
- framework/mcp/constraints.py +125 -0
- framework/mcp/revision_summary.py +122 -0
- framework/mcp/server.py +49 -0
- framework/memory/__init__.py +3 -0
- framework/memory/base.py +17 -0
- framework/models.py +83 -0
- framework/parser.py +45 -0
- framework/parsers/__init__.py +12 -0
- framework/parsers/frontmatter.py +28 -0
- framework/parsers/mapper.py +66 -0
- framework/parsers/markdown.py +122 -0
- framework/parsers/transformers.py +112 -0
- framework/profiles/__init__.py +28 -0
- framework/profiles/registry.py +104 -0
- framework/profiles/support.py +27 -0
- framework/profiles/travel.py +78 -0
- framework/regression/__init__.py +17 -0
- framework/regression/comparator.py +273 -0
- framework/regression/loader.py +145 -0
- framework/sdk.py +151 -0
- framework/utils.py +41 -0
- framework/verification/__init__.py +13 -0
- framework/verification/base.py +32 -0
- framework/verification/extractor.py +105 -0
- framework/verification/local.py +122 -0
- framework/verification/models.py +112 -0
- framework/verification/pipeline.py +36 -0
- framework/verification/prompts.py +24 -0
- framework/verification/utils.py +54 -0
- ui/__init__.py +1 -0
- ui/server.py +255 -0
cli/main.py
ADDED
|
@@ -0,0 +1,423 @@
|
|
|
1
|
+
"""Command-line interface entry point for evalrun."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
import uuid
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import List, Optional
|
|
12
|
+
|
|
13
|
+
# Suppress unconfigured Langfuse initialization warning logs
|
|
14
|
+
if not os.environ.get("LANGFUSE_PUBLIC_KEY"):
|
|
15
|
+
logging.getLogger("langfuse").setLevel(logging.ERROR)
|
|
16
|
+
|
|
17
|
+
from cli.formatter import format_terminal_summary, redact_credentials
|
|
18
|
+
from cli.resolver import resolve_agent
|
|
19
|
+
from framework.evaluation.runner import BenchmarkRunner
|
|
20
|
+
from framework.llms.openai_compatible import OpenAICompatibleLLM
|
|
21
|
+
from framework.models import EvaluationResult
|
|
22
|
+
from cli.progress import run_with_progress
|
|
23
|
+
from agents.auditor import IndependentBudgetAuditor
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def create_parser() -> argparse.ArgumentParser:
|
|
27
|
+
"""Creates the argparse parser for evalrun CLI."""
|
|
28
|
+
parser = argparse.ArgumentParser(
|
|
29
|
+
prog="evalrun",
|
|
30
|
+
description="Local evaluation & regression testing toolkit for tool-using AI agents.",
|
|
31
|
+
)
|
|
32
|
+
subparsers = parser.add_subparsers(dest="command", help="Command to execute")
|
|
33
|
+
|
|
34
|
+
run_parser = subparsers.add_parser("run", help="Run evaluation on a scenario or suite")
|
|
35
|
+
|
|
36
|
+
# Configuration file flag
|
|
37
|
+
run_parser.add_argument(
|
|
38
|
+
"--config",
|
|
39
|
+
"-c",
|
|
40
|
+
type=str,
|
|
41
|
+
default=None,
|
|
42
|
+
help="Path to a JSON or TOML run configuration file",
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
# Mutually exclusive input selection
|
|
46
|
+
input_group = run_parser.add_mutually_exclusive_group(required=False)
|
|
47
|
+
input_group.add_argument(
|
|
48
|
+
"--scenario",
|
|
49
|
+
"-s",
|
|
50
|
+
type=str,
|
|
51
|
+
help="Path to a single benchmark scenario markdown file (.md)",
|
|
52
|
+
)
|
|
53
|
+
input_group.add_argument(
|
|
54
|
+
"--suite",
|
|
55
|
+
type=str,
|
|
56
|
+
help="Path to a suite directory containing benchmark scenario files (.md)",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# Agent configuration
|
|
60
|
+
run_parser.add_argument(
|
|
61
|
+
"--agent",
|
|
62
|
+
"-a",
|
|
63
|
+
type=str,
|
|
64
|
+
default=None,
|
|
65
|
+
help="Python agent import specifier (e.g. 'agents.travel:TravelPlanningAgent')",
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
# Target model endpoint flags
|
|
69
|
+
run_parser.add_argument(
|
|
70
|
+
"--model",
|
|
71
|
+
"-m",
|
|
72
|
+
type=str,
|
|
73
|
+
default=None,
|
|
74
|
+
help="Target agent model identifier (e.g. 'qwen2.5-72b-instruct', 'gpt-5.6-terra')",
|
|
75
|
+
)
|
|
76
|
+
run_parser.add_argument(
|
|
77
|
+
"--base-url",
|
|
78
|
+
type=str,
|
|
79
|
+
default="https://api.openai.com/v1",
|
|
80
|
+
help="OpenAI-compatible endpoint URL for target agent (e.g. 'http://localhost:8000/v1')",
|
|
81
|
+
)
|
|
82
|
+
run_parser.add_argument(
|
|
83
|
+
"--api-key",
|
|
84
|
+
type=str,
|
|
85
|
+
default=None,
|
|
86
|
+
help="API key for target model (defaults to OPENAI_API_KEY env or 'EMPTY')",
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# Judge model endpoint flags
|
|
90
|
+
run_parser.add_argument(
|
|
91
|
+
"--judge-model",
|
|
92
|
+
type=str,
|
|
93
|
+
default=None,
|
|
94
|
+
help="Judge model identifier (defaults to target --model if unspecified)",
|
|
95
|
+
)
|
|
96
|
+
run_parser.add_argument(
|
|
97
|
+
"--judge-base-url",
|
|
98
|
+
type=str,
|
|
99
|
+
default=None,
|
|
100
|
+
help="Judge OpenAI-compatible base URL (defaults to target --base-url if unspecified)",
|
|
101
|
+
)
|
|
102
|
+
run_parser.add_argument(
|
|
103
|
+
"--judge-api-key",
|
|
104
|
+
type=str,
|
|
105
|
+
default=None,
|
|
106
|
+
help="Judge API key (defaults to target --api-key if unspecified)",
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Optional independent auditor endpoint flags
|
|
110
|
+
run_parser.add_argument(
|
|
111
|
+
"--auditor-model",
|
|
112
|
+
type=str,
|
|
113
|
+
default=None,
|
|
114
|
+
help="Independent auditor model; omitted to disable the auditor gate",
|
|
115
|
+
)
|
|
116
|
+
run_parser.add_argument(
|
|
117
|
+
"--auditor-base-url",
|
|
118
|
+
type=str,
|
|
119
|
+
default=None,
|
|
120
|
+
help="Auditor OpenAI-compatible endpoint (defaults to judge endpoint)",
|
|
121
|
+
)
|
|
122
|
+
run_parser.add_argument(
|
|
123
|
+
"--auditor-api-key",
|
|
124
|
+
type=str,
|
|
125
|
+
default=None,
|
|
126
|
+
help="Auditor API key (defaults to judge API key)",
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# Output, Baseline & Verification flags
|
|
130
|
+
run_parser.add_argument(
|
|
131
|
+
"--baseline",
|
|
132
|
+
type=str,
|
|
133
|
+
default=None,
|
|
134
|
+
help="Path to a baseline manifest.json file or directory from a prior run for regression comparison",
|
|
135
|
+
)
|
|
136
|
+
run_parser.add_argument(
|
|
137
|
+
"--max-regression",
|
|
138
|
+
type=float,
|
|
139
|
+
default=None,
|
|
140
|
+
help="Maximum allowed overall score drop before release is blocked (default: 5.0)",
|
|
141
|
+
)
|
|
142
|
+
run_parser.add_argument(
|
|
143
|
+
"--max-dimension-regression",
|
|
144
|
+
type=float,
|
|
145
|
+
default=None,
|
|
146
|
+
help="Maximum allowed per-dimension score drop before release is blocked (default: 10.0)",
|
|
147
|
+
)
|
|
148
|
+
run_parser.add_argument(
|
|
149
|
+
"--ground-truth",
|
|
150
|
+
type=str,
|
|
151
|
+
default=None,
|
|
152
|
+
help="Path to domain knowledge base JSON for factual claim verification",
|
|
153
|
+
)
|
|
154
|
+
run_parser.add_argument(
|
|
155
|
+
"--output",
|
|
156
|
+
"-o",
|
|
157
|
+
type=str,
|
|
158
|
+
default=None,
|
|
159
|
+
help="Output directory path for reports, traces, and manifest",
|
|
160
|
+
)
|
|
161
|
+
# UI Subcommand
|
|
162
|
+
ui_parser = subparsers.add_parser("ui", help="Start guided local UI web server")
|
|
163
|
+
ui_parser.add_argument("--host", type=str, default="127.0.0.1", help="Host address for local UI server (default: 127.0.0.1)")
|
|
164
|
+
ui_parser.add_argument("--port", type=int, default=8501, help="Port number for local UI server (default: 8501)")
|
|
165
|
+
|
|
166
|
+
demo_parser = subparsers.add_parser("demo", help="Run an offline demo without an API key")
|
|
167
|
+
demo_parser.add_argument("--output", type=str, default="results/demo", help="Directory for demo artifacts")
|
|
168
|
+
|
|
169
|
+
return parser
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def run_command(args: argparse.Namespace) -> int:
|
|
173
|
+
"""Executes the 'run' command. Returns CLI exit code (0 = all passed, 1 = eval failure, 2 = runtime error)."""
|
|
174
|
+
# Load config file if provided
|
|
175
|
+
if getattr(args, "config", None):
|
|
176
|
+
cfg_path = Path(args.config)
|
|
177
|
+
if not cfg_path.exists():
|
|
178
|
+
print(f"Error: Configuration file '{args.config}' not found.", file=sys.stderr)
|
|
179
|
+
return 2
|
|
180
|
+
try:
|
|
181
|
+
if cfg_path.suffix == ".json":
|
|
182
|
+
with open(cfg_path, "r", encoding="utf-8") as f:
|
|
183
|
+
cfg_data = json.load(f)
|
|
184
|
+
elif cfg_path.suffix == ".toml":
|
|
185
|
+
if sys.version_info >= (3, 11):
|
|
186
|
+
import tomllib
|
|
187
|
+
with open(cfg_path, "rb") as f:
|
|
188
|
+
cfg_data = tomllib.load(f)
|
|
189
|
+
else:
|
|
190
|
+
import toml
|
|
191
|
+
with open(cfg_path, "r", encoding="utf-8") as f:
|
|
192
|
+
cfg_data = toml.load(f)
|
|
193
|
+
else:
|
|
194
|
+
print(f"Error: Unsupported config format '{cfg_path.suffix}'. Use .json or .toml", file=sys.stderr)
|
|
195
|
+
return 2
|
|
196
|
+
|
|
197
|
+
for k, v in cfg_data.items():
|
|
198
|
+
if getattr(args, k, None) is None:
|
|
199
|
+
setattr(args, k, v)
|
|
200
|
+
except Exception as e:
|
|
201
|
+
print(f"Error loading configuration file '{args.config}': {e}", file=sys.stderr)
|
|
202
|
+
return 2
|
|
203
|
+
|
|
204
|
+
# Set fallback defaults for optional flags if still None
|
|
205
|
+
args.base_url = args.base_url or "https://api.openai.com/v1"
|
|
206
|
+
args.ground_truth = args.ground_truth or "ground_truth/japan_demo.json"
|
|
207
|
+
args.output = args.output or "./eval_results"
|
|
208
|
+
args.max_regression = 5.0 if args.max_regression is None else args.max_regression
|
|
209
|
+
args.max_dimension_regression = 10.0 if args.max_dimension_regression is None else args.max_dimension_regression
|
|
210
|
+
|
|
211
|
+
if not getattr(args, "scenario", None) and not getattr(args, "suite", None):
|
|
212
|
+
print("Error: Either --scenario or --suite or a valid config file specifying input is required.", file=sys.stderr)
|
|
213
|
+
return 2
|
|
214
|
+
|
|
215
|
+
if not getattr(args, "agent", None):
|
|
216
|
+
print("Error: --agent specifier is required.", file=sys.stderr)
|
|
217
|
+
return 2
|
|
218
|
+
|
|
219
|
+
if not getattr(args, "model", None):
|
|
220
|
+
print("Error: --model identifier is required.", file=sys.stderr)
|
|
221
|
+
return 2
|
|
222
|
+
output_dir = Path(args.output)
|
|
223
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
224
|
+
|
|
225
|
+
judge_model_name = args.judge_model or args.model
|
|
226
|
+
judge_base_url = args.judge_base_url or args.base_url
|
|
227
|
+
judge_api_key = args.judge_api_key or args.api_key
|
|
228
|
+
auditor_base_url = args.auditor_base_url or judge_base_url
|
|
229
|
+
auditor_api_key = args.auditor_api_key or judge_api_key
|
|
230
|
+
|
|
231
|
+
# Instantiate Model Endpoints
|
|
232
|
+
try:
|
|
233
|
+
target_llm = OpenAICompatibleLLM(
|
|
234
|
+
model_name=args.model,
|
|
235
|
+
base_url=args.base_url,
|
|
236
|
+
api_key=args.api_key,
|
|
237
|
+
)
|
|
238
|
+
judge_llm = OpenAICompatibleLLM(
|
|
239
|
+
model_name=judge_model_name,
|
|
240
|
+
base_url=judge_base_url,
|
|
241
|
+
api_key=judge_api_key,
|
|
242
|
+
)
|
|
243
|
+
auditor_llm = None
|
|
244
|
+
if args.auditor_model:
|
|
245
|
+
auditor_llm = OpenAICompatibleLLM(
|
|
246
|
+
model_name=args.auditor_model,
|
|
247
|
+
base_url=auditor_base_url,
|
|
248
|
+
api_key=auditor_api_key,
|
|
249
|
+
)
|
|
250
|
+
except Exception as e:
|
|
251
|
+
print(f"Error initializing model endpoint client: {e}", file=sys.stderr)
|
|
252
|
+
return 2
|
|
253
|
+
|
|
254
|
+
# Instantiate Target Agent
|
|
255
|
+
try:
|
|
256
|
+
agent_instance = resolve_agent(args.agent, target_llm)
|
|
257
|
+
except Exception as e:
|
|
258
|
+
print(f"Error resolving agent specifier '{args.agent}': {e}", file=sys.stderr)
|
|
259
|
+
return 2
|
|
260
|
+
|
|
261
|
+
# Instantiate BenchmarkRunner
|
|
262
|
+
auditor = IndependentBudgetAuditor(auditor_llm) if auditor_llm is not None else None
|
|
263
|
+
runner = BenchmarkRunner(
|
|
264
|
+
agent=agent_instance,
|
|
265
|
+
judge_llm=judge_llm,
|
|
266
|
+
local_verifier_path=args.ground_truth,
|
|
267
|
+
output_dir=str(output_dir),
|
|
268
|
+
auditor=auditor,
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
# Gather Scenario Files
|
|
272
|
+
scenario_files: List[Path] = []
|
|
273
|
+
if args.scenario:
|
|
274
|
+
scenario_file = Path(args.scenario)
|
|
275
|
+
if not scenario_file.exists():
|
|
276
|
+
print(f"Error: Scenario file '{args.scenario}' does not exist.", file=sys.stderr)
|
|
277
|
+
return 2
|
|
278
|
+
scenario_files.append(scenario_file)
|
|
279
|
+
elif args.suite:
|
|
280
|
+
suite_dir = Path(args.suite)
|
|
281
|
+
if not suite_dir.exists() or not suite_dir.is_dir():
|
|
282
|
+
print(f"Error: Suite directory '{args.suite}' does not exist or is not a directory.", file=sys.stderr)
|
|
283
|
+
return 2
|
|
284
|
+
scenario_files = sorted(list(suite_dir.glob("*.md")))
|
|
285
|
+
if not scenario_files:
|
|
286
|
+
print(f"Error: No benchmark scenario markdown files (.md) found in suite directory '{args.suite}'.", file=sys.stderr)
|
|
287
|
+
return 2
|
|
288
|
+
|
|
289
|
+
# Execute Evaluation Pipeline
|
|
290
|
+
results: List[EvaluationResult] = []
|
|
291
|
+
for s_file in scenario_files:
|
|
292
|
+
try:
|
|
293
|
+
print(f"[evalrun] Running scenario: {s_file.name} (model calls and evaluation in progress...)", flush=True)
|
|
294
|
+
res = run_with_progress(
|
|
295
|
+
s_file.name,
|
|
296
|
+
lambda: runner.run(str(s_file)),
|
|
297
|
+
)
|
|
298
|
+
results.append(res)
|
|
299
|
+
print(f"[evalrun] Finished {s_file.name}: score {res.overall_score:.2f} ({'PASS' if res.passed else 'FAIL'})", flush=True)
|
|
300
|
+
except Exception as e:
|
|
301
|
+
print(f"Error executing evaluation for scenario '{s_file}': {e}", file=sys.stderr)
|
|
302
|
+
return 2
|
|
303
|
+
|
|
304
|
+
# Determine Baseline Comparison & 3-Tier Release Gate Outcomes
|
|
305
|
+
evaluation_passed = all(r.passed for r in results)
|
|
306
|
+
regression_report_dict = None
|
|
307
|
+
|
|
308
|
+
if args.baseline:
|
|
309
|
+
try:
|
|
310
|
+
from framework.regression import load_baseline_manifest, compare_runs
|
|
311
|
+
baseline_data = load_baseline_manifest(args.baseline)
|
|
312
|
+
candidate_run_id = f"evalrun-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:6]}"
|
|
313
|
+
reg_report = compare_runs(
|
|
314
|
+
candidate_results=results,
|
|
315
|
+
baseline_data=baseline_data,
|
|
316
|
+
max_overall_drop=args.max_regression,
|
|
317
|
+
max_dim_drop=args.max_dimension_regression,
|
|
318
|
+
candidate_run_id=candidate_run_id,
|
|
319
|
+
)
|
|
320
|
+
regression_report_dict = reg_report.to_dict()
|
|
321
|
+
|
|
322
|
+
# Save standalone regression_report.json
|
|
323
|
+
reg_path = output_dir / "regression_report.json"
|
|
324
|
+
with open(reg_path, "w", encoding="utf-8") as f:
|
|
325
|
+
json.dump(redact_credentials(regression_report_dict), f, indent=2)
|
|
326
|
+
|
|
327
|
+
if reg_report.release_blocked:
|
|
328
|
+
evaluation_passed = False
|
|
329
|
+
except Exception as e:
|
|
330
|
+
print(f"Error performing baseline regression comparison: {e}", file=sys.stderr)
|
|
331
|
+
return 2
|
|
332
|
+
|
|
333
|
+
# Verify Independent Auditor Gate Decisions
|
|
334
|
+
for r in results:
|
|
335
|
+
gate_decision = getattr(r, "agent_metadata", {}).get("audit_gate_decision", "N/A")
|
|
336
|
+
if gate_decision == "BLOCK":
|
|
337
|
+
evaluation_passed = False
|
|
338
|
+
|
|
339
|
+
# Save Run Manifest
|
|
340
|
+
model_slug = str(args.model).replace("/", "_").replace(".", "_")
|
|
341
|
+
scenarios_summary = []
|
|
342
|
+
for r in results:
|
|
343
|
+
scenarios_summary.append({
|
|
344
|
+
"scenario_id": r.benchmark_id,
|
|
345
|
+
"scenario_name": r.benchmark_name,
|
|
346
|
+
"overall_score": r.overall_score,
|
|
347
|
+
"passed": r.passed,
|
|
348
|
+
"audit_gate_decision": getattr(r, "agent_metadata", {}).get("audit_gate_decision", "N/A"),
|
|
349
|
+
"report_path": f"{model_slug}_{r.benchmark_id}_report.json",
|
|
350
|
+
})
|
|
351
|
+
|
|
352
|
+
manifest = {
|
|
353
|
+
"run_id": f"evalrun-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:6]}",
|
|
354
|
+
"timestamp_utc": datetime.now(timezone.utc).isoformat(),
|
|
355
|
+
"target_agent_spec": args.agent,
|
|
356
|
+
"target_model": {
|
|
357
|
+
"model_name": args.model,
|
|
358
|
+
"base_url": args.base_url,
|
|
359
|
+
"api_key": "[REDACTED]" if args.api_key else "ENVIRONMENT_OR_EMPTY",
|
|
360
|
+
},
|
|
361
|
+
"judge_model": {
|
|
362
|
+
"model_name": judge_model_name,
|
|
363
|
+
"base_url": judge_base_url,
|
|
364
|
+
"api_key": "[REDACTED]" if judge_api_key else "ENVIRONMENT_OR_EMPTY",
|
|
365
|
+
},
|
|
366
|
+
"auditor_model": ({
|
|
367
|
+
"model_name": args.auditor_model,
|
|
368
|
+
"base_url": auditor_base_url,
|
|
369
|
+
"api_key": "[REDACTED]" if auditor_api_key else "ENVIRONMENT_OR_EMPTY",
|
|
370
|
+
} if args.auditor_model else None),
|
|
371
|
+
"baseline_path": args.baseline,
|
|
372
|
+
"ground_truth_path": args.ground_truth,
|
|
373
|
+
"output_dir": str(output_dir),
|
|
374
|
+
"total_scenarios": len(results),
|
|
375
|
+
"overall_passed": evaluation_passed,
|
|
376
|
+
"scenarios": scenarios_summary,
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
redacted_manifest = redact_credentials(manifest)
|
|
380
|
+
manifest_path = output_dir / "manifest.json"
|
|
381
|
+
with open(manifest_path, "w", encoding="utf-8") as f:
|
|
382
|
+
json.dump(redacted_manifest, f, indent=2)
|
|
383
|
+
|
|
384
|
+
# Generate Local HTML Report
|
|
385
|
+
try:
|
|
386
|
+
from cli.html_reporter import generate_html_report
|
|
387
|
+
generate_html_report(results, redacted_manifest, str(output_dir), regression_report=regression_report_dict)
|
|
388
|
+
except Exception as e:
|
|
389
|
+
print(f"Warning: Failed to generate HTML report: {e}", file=sys.stderr)
|
|
390
|
+
|
|
391
|
+
# Render Terminal Summary
|
|
392
|
+
terminal_report = format_terminal_summary(results, redacted_manifest, regression_report=regression_report_dict)
|
|
393
|
+
print(terminal_report)
|
|
394
|
+
|
|
395
|
+
return 0 if evaluation_passed else 1
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def main(argv: Optional[List[str]] = None) -> None:
|
|
399
|
+
parser = create_parser()
|
|
400
|
+
args = parser.parse_args(argv)
|
|
401
|
+
|
|
402
|
+
if args.command == "run":
|
|
403
|
+
exit_code = run_command(args)
|
|
404
|
+
sys.exit(exit_code)
|
|
405
|
+
elif args.command == "ui":
|
|
406
|
+
from ui.server import run_ui_server
|
|
407
|
+
server = run_ui_server(host=args.host, port=args.port)
|
|
408
|
+
try:
|
|
409
|
+
server.serve_forever()
|
|
410
|
+
except KeyboardInterrupt:
|
|
411
|
+
print("\nShutting down local UI server.")
|
|
412
|
+
server.server_close()
|
|
413
|
+
sys.exit(0)
|
|
414
|
+
elif args.command == "demo":
|
|
415
|
+
from cli.demo import run_demo
|
|
416
|
+
sys.exit(run_demo(output_dir=args.output))
|
|
417
|
+
else:
|
|
418
|
+
parser.print_help()
|
|
419
|
+
sys.exit(2)
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
if __name__ == "__main__":
|
|
423
|
+
main()
|
cli/progress.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Interactive terminal progress indicator for long model evaluations."""
|
|
2
|
+
|
|
3
|
+
import itertools
|
|
4
|
+
import sys
|
|
5
|
+
import threading
|
|
6
|
+
import time
|
|
7
|
+
from typing import Callable, TypeVar
|
|
8
|
+
|
|
9
|
+
T = TypeVar("T")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def run_with_progress(label: str, operation: Callable[[], T]) -> T:
|
|
13
|
+
"""Run an operation while showing a live spinner in an interactive terminal."""
|
|
14
|
+
interactive = bool(getattr(sys.stdout, "isatty", lambda: False)())
|
|
15
|
+
if not interactive:
|
|
16
|
+
return operation()
|
|
17
|
+
|
|
18
|
+
stop = threading.Event()
|
|
19
|
+
frames = itertools.cycle("⠋⠙⠹⠸⠼⠴⠦⠧⠇⠏")
|
|
20
|
+
statuses = itertools.cycle(("contacting model", "running agent", "scoring response", "writing report"))
|
|
21
|
+
started = time.monotonic()
|
|
22
|
+
|
|
23
|
+
def render() -> None:
|
|
24
|
+
while not stop.wait(0.35):
|
|
25
|
+
elapsed = int(time.monotonic() - started)
|
|
26
|
+
sys.stdout.write(f"\r[evalrun] {next(frames)} {label} — {next(statuses)} ({elapsed}s)")
|
|
27
|
+
sys.stdout.flush()
|
|
28
|
+
|
|
29
|
+
thread = threading.Thread(target=render, daemon=True)
|
|
30
|
+
thread.start()
|
|
31
|
+
try:
|
|
32
|
+
return operation()
|
|
33
|
+
finally:
|
|
34
|
+
stop.set()
|
|
35
|
+
thread.join(timeout=1.0)
|
|
36
|
+
elapsed = time.monotonic() - started
|
|
37
|
+
sys.stdout.write(f"\r[evalrun] ✓ {label} completed in {elapsed:.1f}s\033[K\n")
|
|
38
|
+
sys.stdout.flush()
|
cli/resolver.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Dynamic Python agent resolver for evalrun CLI."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import inspect
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
from framework.llms.base import BaseLLM
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def resolve_agent(agent_spec: str, llm: BaseLLM) -> Any:
|
|
12
|
+
"""Dynamically imports and constructs an agent instance from an import specifier.
|
|
13
|
+
|
|
14
|
+
Specifier format: 'package.module:ClassName' or 'package.module:factory_function'
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
agent_spec: Import specifier string (e.g. 'agents.travel:TravelPlanningAgent').
|
|
18
|
+
llm: Target LLM client instance to inject into the agent constructor/factory.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
An instantiated agent object.
|
|
22
|
+
|
|
23
|
+
Raises:
|
|
24
|
+
ValueError: If the specifier is malformed, module cannot be imported, or symbol cannot be constructed.
|
|
25
|
+
"""
|
|
26
|
+
# Terminals sometimes receive escaped underscores when a command is copied
|
|
27
|
+
# from rendered Markdown (``custom\_agent``). They are not meaningful in a
|
|
28
|
+
# Python import path, so normalize them at the CLI boundary.
|
|
29
|
+
agent_spec = agent_spec.replace("\\_", "_").strip()
|
|
30
|
+
|
|
31
|
+
# A console-script entry point has the virtualenv's ``bin`` directory at
|
|
32
|
+
# sys.path[0], not the user's working directory. Add the current project
|
|
33
|
+
# directory so ``evalrun --agent my_agent:Agent`` works without requiring
|
|
34
|
+
# users to set PYTHONPATH manually.
|
|
35
|
+
cwd = str(Path.cwd())
|
|
36
|
+
if cwd not in sys.path:
|
|
37
|
+
sys.path.insert(0, cwd)
|
|
38
|
+
|
|
39
|
+
# 1. HTTP Endpoint Agent Resolution
|
|
40
|
+
if agent_spec.startswith("http://") or agent_spec.startswith("https://"):
|
|
41
|
+
from framework.core.adapters import HttpAgentAdapter
|
|
42
|
+
return HttpAgentAdapter(endpoint_url=agent_spec)
|
|
43
|
+
|
|
44
|
+
# 2. CLI Subprocess Command Agent Resolution
|
|
45
|
+
if agent_spec.startswith("cli:"):
|
|
46
|
+
from framework.core.adapters import CliAgentAdapter
|
|
47
|
+
command = agent_spec[4:].strip()
|
|
48
|
+
return CliAgentAdapter(command=command)
|
|
49
|
+
|
|
50
|
+
if ":" not in agent_spec:
|
|
51
|
+
raise ValueError(
|
|
52
|
+
f"Invalid agent specification '{agent_spec}'. Expected format 'module:Class', 'http://...', or 'cli:command'."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
module_path, symbol_name = agent_spec.split(":", 1)
|
|
56
|
+
module_path = module_path.strip()
|
|
57
|
+
symbol_name = symbol_name.strip()
|
|
58
|
+
|
|
59
|
+
try:
|
|
60
|
+
module = importlib.import_module(module_path)
|
|
61
|
+
except ImportError as e:
|
|
62
|
+
raise ValueError(f"Failed to import agent module '{module_path}': {e}") from e
|
|
63
|
+
|
|
64
|
+
if not hasattr(module, symbol_name):
|
|
65
|
+
raise ValueError(f"Module '{module_path}' has no attribute or class '{symbol_name}'.")
|
|
66
|
+
|
|
67
|
+
symbol = getattr(module, symbol_name)
|
|
68
|
+
|
|
69
|
+
if inspect.isclass(symbol):
|
|
70
|
+
sig = inspect.signature(symbol.__init__)
|
|
71
|
+
params = sig.parameters
|
|
72
|
+
if "llm" in params:
|
|
73
|
+
return symbol(llm=llm)
|
|
74
|
+
elif len(params) <= 1: # Only self
|
|
75
|
+
return symbol()
|
|
76
|
+
else:
|
|
77
|
+
# Try passing llm as first positional argument
|
|
78
|
+
try:
|
|
79
|
+
return symbol(llm)
|
|
80
|
+
except Exception as e:
|
|
81
|
+
raise ValueError(
|
|
82
|
+
f"Could not instantiate agent class '{symbol_name}' with LLM parameter: {e}"
|
|
83
|
+
) from e
|
|
84
|
+
elif callable(symbol):
|
|
85
|
+
sig = inspect.signature(symbol)
|
|
86
|
+
params = sig.parameters
|
|
87
|
+
if "llm" in params:
|
|
88
|
+
return symbol(llm=llm)
|
|
89
|
+
elif len(params) == 0:
|
|
90
|
+
return symbol()
|
|
91
|
+
else:
|
|
92
|
+
try:
|
|
93
|
+
return symbol(llm)
|
|
94
|
+
except Exception as e:
|
|
95
|
+
raise ValueError(
|
|
96
|
+
f"Could not invoke agent factory '{symbol_name}' with LLM parameter: {e}"
|
|
97
|
+
) from e
|
|
98
|
+
else:
|
|
99
|
+
raise ValueError(f"Symbol '{symbol_name}' in module '{module_path}' is neither a class nor a callable factory.")
|