evalrun 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. agents/__init__.py +6 -0
  2. agents/auditor/__init__.py +13 -0
  3. agents/auditor/budget_auditor.py +91 -0
  4. agents/auditor/parser.py +139 -0
  5. agents/auditor/prompts.py +64 -0
  6. agents/auditor/schema.py +67 -0
  7. agents/base.py +20 -0
  8. agents/reflection/__init__.py +3 -0
  9. agents/reflection/agent.py +77 -0
  10. agents/reflection/prompts.py +18 -0
  11. agents/research/__init__.py +4 -0
  12. agents/research/agent.py +45 -0
  13. agents/research/planner.py +52 -0
  14. agents/research/prompts.py +14 -0
  15. agents/support/__init__.py +5 -0
  16. agents/support/triage_agent.py +45 -0
  17. agents/travel/__init__.py +11 -0
  18. agents/travel/agent.py +377 -0
  19. agents/travel/prompts.py +30 -0
  20. agents/travel/session.py +110 -0
  21. cli/__init__.py +6 -0
  22. cli/demo.py +47 -0
  23. cli/formatter.py +93 -0
  24. cli/html_reporter.py +647 -0
  25. cli/main.py +423 -0
  26. cli/progress.py +38 -0
  27. cli/resolver.py +99 -0
  28. evalrun-0.4.0.dist-info/METADATA +268 -0
  29. evalrun-0.4.0.dist-info/RECORD +100 -0
  30. evalrun-0.4.0.dist-info/WHEEL +5 -0
  31. evalrun-0.4.0.dist-info/entry_points.txt +2 -0
  32. evalrun-0.4.0.dist-info/licenses/LICENSE +0 -0
  33. evalrun-0.4.0.dist-info/top_level.txt +4 -0
  34. framework/__init__.py +70 -0
  35. framework/core/__init__.py +17 -0
  36. framework/core/adapters.py +118 -0
  37. framework/core/contracts.py +88 -0
  38. framework/core/suite.py +44 -0
  39. framework/evaluation/__init__.py +22 -0
  40. framework/evaluation/base.py +29 -0
  41. framework/evaluation/dimensions.py +7 -0
  42. framework/evaluation/engine.py +110 -0
  43. framework/evaluation/evaluators/__init__.py +7 -0
  44. framework/evaluation/evaluators/adaptability.py +27 -0
  45. framework/evaluation/evaluators/base_llm.py +104 -0
  46. framework/evaluation/evaluators/constraint.py +27 -0
  47. framework/evaluation/evaluators/information_accuracy.py +41 -0
  48. framework/evaluation/evaluators/personalization.py +27 -0
  49. framework/evaluation/evaluators/planning.py +27 -0
  50. framework/evaluation/evaluators/support.py +81 -0
  51. framework/evaluation/prompts/__init__.py +11 -0
  52. framework/evaluation/prompts/adaptability.py +57 -0
  53. framework/evaluation/prompts/base.py +52 -0
  54. framework/evaluation/prompts/constraint.py +41 -0
  55. framework/evaluation/prompts/information_accuracy.py +79 -0
  56. framework/evaluation/prompts/personalization.py +57 -0
  57. framework/evaluation/prompts/planning.py +61 -0
  58. framework/evaluation/runner.py +363 -0
  59. framework/evaluation/testing.py +25 -0
  60. framework/exceptions.py +49 -0
  61. framework/llms/__init__.py +8 -0
  62. framework/llms/base.py +37 -0
  63. framework/llms/factory.py +38 -0
  64. framework/llms/gemini.py +85 -0
  65. framework/llms/mock.py +25 -0
  66. framework/llms/openai.py +94 -0
  67. framework/llms/openai_compatible.py +139 -0
  68. framework/mcp/__init__.py +20 -0
  69. framework/mcp/client.py +62 -0
  70. framework/mcp/constraints.py +125 -0
  71. framework/mcp/revision_summary.py +122 -0
  72. framework/mcp/server.py +49 -0
  73. framework/memory/__init__.py +3 -0
  74. framework/memory/base.py +17 -0
  75. framework/models.py +83 -0
  76. framework/parser.py +45 -0
  77. framework/parsers/__init__.py +12 -0
  78. framework/parsers/frontmatter.py +28 -0
  79. framework/parsers/mapper.py +66 -0
  80. framework/parsers/markdown.py +122 -0
  81. framework/parsers/transformers.py +112 -0
  82. framework/profiles/__init__.py +28 -0
  83. framework/profiles/registry.py +104 -0
  84. framework/profiles/support.py +27 -0
  85. framework/profiles/travel.py +78 -0
  86. framework/regression/__init__.py +17 -0
  87. framework/regression/comparator.py +273 -0
  88. framework/regression/loader.py +145 -0
  89. framework/sdk.py +151 -0
  90. framework/utils.py +41 -0
  91. framework/verification/__init__.py +13 -0
  92. framework/verification/base.py +32 -0
  93. framework/verification/extractor.py +105 -0
  94. framework/verification/local.py +122 -0
  95. framework/verification/models.py +112 -0
  96. framework/verification/pipeline.py +36 -0
  97. framework/verification/prompts.py +24 -0
  98. framework/verification/utils.py +54 -0
  99. ui/__init__.py +1 -0
  100. ui/server.py +255 -0
cli/main.py ADDED
@@ -0,0 +1,423 @@
1
+ """Command-line interface entry point for evalrun."""
2
+
3
+ import argparse
4
+ import json
5
+ import logging
6
+ import os
7
+ import sys
8
+ import uuid
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+ from typing import List, Optional
12
+
13
+ # Suppress unconfigured Langfuse initialization warning logs
14
+ if not os.environ.get("LANGFUSE_PUBLIC_KEY"):
15
+ logging.getLogger("langfuse").setLevel(logging.ERROR)
16
+
17
+ from cli.formatter import format_terminal_summary, redact_credentials
18
+ from cli.resolver import resolve_agent
19
+ from framework.evaluation.runner import BenchmarkRunner
20
+ from framework.llms.openai_compatible import OpenAICompatibleLLM
21
+ from framework.models import EvaluationResult
22
+ from cli.progress import run_with_progress
23
+ from agents.auditor import IndependentBudgetAuditor
24
+
25
+
26
+ def create_parser() -> argparse.ArgumentParser:
27
+ """Creates the argparse parser for evalrun CLI."""
28
+ parser = argparse.ArgumentParser(
29
+ prog="evalrun",
30
+ description="Local evaluation & regression testing toolkit for tool-using AI agents.",
31
+ )
32
+ subparsers = parser.add_subparsers(dest="command", help="Command to execute")
33
+
34
+ run_parser = subparsers.add_parser("run", help="Run evaluation on a scenario or suite")
35
+
36
+ # Configuration file flag
37
+ run_parser.add_argument(
38
+ "--config",
39
+ "-c",
40
+ type=str,
41
+ default=None,
42
+ help="Path to a JSON or TOML run configuration file",
43
+ )
44
+
45
+ # Mutually exclusive input selection
46
+ input_group = run_parser.add_mutually_exclusive_group(required=False)
47
+ input_group.add_argument(
48
+ "--scenario",
49
+ "-s",
50
+ type=str,
51
+ help="Path to a single benchmark scenario markdown file (.md)",
52
+ )
53
+ input_group.add_argument(
54
+ "--suite",
55
+ type=str,
56
+ help="Path to a suite directory containing benchmark scenario files (.md)",
57
+ )
58
+
59
+ # Agent configuration
60
+ run_parser.add_argument(
61
+ "--agent",
62
+ "-a",
63
+ type=str,
64
+ default=None,
65
+ help="Python agent import specifier (e.g. 'agents.travel:TravelPlanningAgent')",
66
+ )
67
+
68
+ # Target model endpoint flags
69
+ run_parser.add_argument(
70
+ "--model",
71
+ "-m",
72
+ type=str,
73
+ default=None,
74
+ help="Target agent model identifier (e.g. 'qwen2.5-72b-instruct', 'gpt-5.6-terra')",
75
+ )
76
+ run_parser.add_argument(
77
+ "--base-url",
78
+ type=str,
79
+ default="https://api.openai.com/v1",
80
+ help="OpenAI-compatible endpoint URL for target agent (e.g. 'http://localhost:8000/v1')",
81
+ )
82
+ run_parser.add_argument(
83
+ "--api-key",
84
+ type=str,
85
+ default=None,
86
+ help="API key for target model (defaults to OPENAI_API_KEY env or 'EMPTY')",
87
+ )
88
+
89
+ # Judge model endpoint flags
90
+ run_parser.add_argument(
91
+ "--judge-model",
92
+ type=str,
93
+ default=None,
94
+ help="Judge model identifier (defaults to target --model if unspecified)",
95
+ )
96
+ run_parser.add_argument(
97
+ "--judge-base-url",
98
+ type=str,
99
+ default=None,
100
+ help="Judge OpenAI-compatible base URL (defaults to target --base-url if unspecified)",
101
+ )
102
+ run_parser.add_argument(
103
+ "--judge-api-key",
104
+ type=str,
105
+ default=None,
106
+ help="Judge API key (defaults to target --api-key if unspecified)",
107
+ )
108
+
109
+ # Optional independent auditor endpoint flags
110
+ run_parser.add_argument(
111
+ "--auditor-model",
112
+ type=str,
113
+ default=None,
114
+ help="Independent auditor model; omitted to disable the auditor gate",
115
+ )
116
+ run_parser.add_argument(
117
+ "--auditor-base-url",
118
+ type=str,
119
+ default=None,
120
+ help="Auditor OpenAI-compatible endpoint (defaults to judge endpoint)",
121
+ )
122
+ run_parser.add_argument(
123
+ "--auditor-api-key",
124
+ type=str,
125
+ default=None,
126
+ help="Auditor API key (defaults to judge API key)",
127
+ )
128
+
129
+ # Output, Baseline & Verification flags
130
+ run_parser.add_argument(
131
+ "--baseline",
132
+ type=str,
133
+ default=None,
134
+ help="Path to a baseline manifest.json file or directory from a prior run for regression comparison",
135
+ )
136
+ run_parser.add_argument(
137
+ "--max-regression",
138
+ type=float,
139
+ default=None,
140
+ help="Maximum allowed overall score drop before release is blocked (default: 5.0)",
141
+ )
142
+ run_parser.add_argument(
143
+ "--max-dimension-regression",
144
+ type=float,
145
+ default=None,
146
+ help="Maximum allowed per-dimension score drop before release is blocked (default: 10.0)",
147
+ )
148
+ run_parser.add_argument(
149
+ "--ground-truth",
150
+ type=str,
151
+ default=None,
152
+ help="Path to domain knowledge base JSON for factual claim verification",
153
+ )
154
+ run_parser.add_argument(
155
+ "--output",
156
+ "-o",
157
+ type=str,
158
+ default=None,
159
+ help="Output directory path for reports, traces, and manifest",
160
+ )
161
+ # UI Subcommand
162
+ ui_parser = subparsers.add_parser("ui", help="Start guided local UI web server")
163
+ ui_parser.add_argument("--host", type=str, default="127.0.0.1", help="Host address for local UI server (default: 127.0.0.1)")
164
+ ui_parser.add_argument("--port", type=int, default=8501, help="Port number for local UI server (default: 8501)")
165
+
166
+ demo_parser = subparsers.add_parser("demo", help="Run an offline demo without an API key")
167
+ demo_parser.add_argument("--output", type=str, default="results/demo", help="Directory for demo artifacts")
168
+
169
+ return parser
170
+
171
+
172
+ def run_command(args: argparse.Namespace) -> int:
173
+ """Executes the 'run' command. Returns CLI exit code (0 = all passed, 1 = eval failure, 2 = runtime error)."""
174
+ # Load config file if provided
175
+ if getattr(args, "config", None):
176
+ cfg_path = Path(args.config)
177
+ if not cfg_path.exists():
178
+ print(f"Error: Configuration file '{args.config}' not found.", file=sys.stderr)
179
+ return 2
180
+ try:
181
+ if cfg_path.suffix == ".json":
182
+ with open(cfg_path, "r", encoding="utf-8") as f:
183
+ cfg_data = json.load(f)
184
+ elif cfg_path.suffix == ".toml":
185
+ if sys.version_info >= (3, 11):
186
+ import tomllib
187
+ with open(cfg_path, "rb") as f:
188
+ cfg_data = tomllib.load(f)
189
+ else:
190
+ import toml
191
+ with open(cfg_path, "r", encoding="utf-8") as f:
192
+ cfg_data = toml.load(f)
193
+ else:
194
+ print(f"Error: Unsupported config format '{cfg_path.suffix}'. Use .json or .toml", file=sys.stderr)
195
+ return 2
196
+
197
+ for k, v in cfg_data.items():
198
+ if getattr(args, k, None) is None:
199
+ setattr(args, k, v)
200
+ except Exception as e:
201
+ print(f"Error loading configuration file '{args.config}': {e}", file=sys.stderr)
202
+ return 2
203
+
204
+ # Set fallback defaults for optional flags if still None
205
+ args.base_url = args.base_url or "https://api.openai.com/v1"
206
+ args.ground_truth = args.ground_truth or "ground_truth/japan_demo.json"
207
+ args.output = args.output or "./eval_results"
208
+ args.max_regression = 5.0 if args.max_regression is None else args.max_regression
209
+ args.max_dimension_regression = 10.0 if args.max_dimension_regression is None else args.max_dimension_regression
210
+
211
+ if not getattr(args, "scenario", None) and not getattr(args, "suite", None):
212
+ print("Error: Either --scenario or --suite or a valid config file specifying input is required.", file=sys.stderr)
213
+ return 2
214
+
215
+ if not getattr(args, "agent", None):
216
+ print("Error: --agent specifier is required.", file=sys.stderr)
217
+ return 2
218
+
219
+ if not getattr(args, "model", None):
220
+ print("Error: --model identifier is required.", file=sys.stderr)
221
+ return 2
222
+ output_dir = Path(args.output)
223
+ output_dir.mkdir(parents=True, exist_ok=True)
224
+
225
+ judge_model_name = args.judge_model or args.model
226
+ judge_base_url = args.judge_base_url or args.base_url
227
+ judge_api_key = args.judge_api_key or args.api_key
228
+ auditor_base_url = args.auditor_base_url or judge_base_url
229
+ auditor_api_key = args.auditor_api_key or judge_api_key
230
+
231
+ # Instantiate Model Endpoints
232
+ try:
233
+ target_llm = OpenAICompatibleLLM(
234
+ model_name=args.model,
235
+ base_url=args.base_url,
236
+ api_key=args.api_key,
237
+ )
238
+ judge_llm = OpenAICompatibleLLM(
239
+ model_name=judge_model_name,
240
+ base_url=judge_base_url,
241
+ api_key=judge_api_key,
242
+ )
243
+ auditor_llm = None
244
+ if args.auditor_model:
245
+ auditor_llm = OpenAICompatibleLLM(
246
+ model_name=args.auditor_model,
247
+ base_url=auditor_base_url,
248
+ api_key=auditor_api_key,
249
+ )
250
+ except Exception as e:
251
+ print(f"Error initializing model endpoint client: {e}", file=sys.stderr)
252
+ return 2
253
+
254
+ # Instantiate Target Agent
255
+ try:
256
+ agent_instance = resolve_agent(args.agent, target_llm)
257
+ except Exception as e:
258
+ print(f"Error resolving agent specifier '{args.agent}': {e}", file=sys.stderr)
259
+ return 2
260
+
261
+ # Instantiate BenchmarkRunner
262
+ auditor = IndependentBudgetAuditor(auditor_llm) if auditor_llm is not None else None
263
+ runner = BenchmarkRunner(
264
+ agent=agent_instance,
265
+ judge_llm=judge_llm,
266
+ local_verifier_path=args.ground_truth,
267
+ output_dir=str(output_dir),
268
+ auditor=auditor,
269
+ )
270
+
271
+ # Gather Scenario Files
272
+ scenario_files: List[Path] = []
273
+ if args.scenario:
274
+ scenario_file = Path(args.scenario)
275
+ if not scenario_file.exists():
276
+ print(f"Error: Scenario file '{args.scenario}' does not exist.", file=sys.stderr)
277
+ return 2
278
+ scenario_files.append(scenario_file)
279
+ elif args.suite:
280
+ suite_dir = Path(args.suite)
281
+ if not suite_dir.exists() or not suite_dir.is_dir():
282
+ print(f"Error: Suite directory '{args.suite}' does not exist or is not a directory.", file=sys.stderr)
283
+ return 2
284
+ scenario_files = sorted(list(suite_dir.glob("*.md")))
285
+ if not scenario_files:
286
+ print(f"Error: No benchmark scenario markdown files (.md) found in suite directory '{args.suite}'.", file=sys.stderr)
287
+ return 2
288
+
289
+ # Execute Evaluation Pipeline
290
+ results: List[EvaluationResult] = []
291
+ for s_file in scenario_files:
292
+ try:
293
+ print(f"[evalrun] Running scenario: {s_file.name} (model calls and evaluation in progress...)", flush=True)
294
+ res = run_with_progress(
295
+ s_file.name,
296
+ lambda: runner.run(str(s_file)),
297
+ )
298
+ results.append(res)
299
+ print(f"[evalrun] Finished {s_file.name}: score {res.overall_score:.2f} ({'PASS' if res.passed else 'FAIL'})", flush=True)
300
+ except Exception as e:
301
+ print(f"Error executing evaluation for scenario '{s_file}': {e}", file=sys.stderr)
302
+ return 2
303
+
304
+ # Determine Baseline Comparison & 3-Tier Release Gate Outcomes
305
+ evaluation_passed = all(r.passed for r in results)
306
+ regression_report_dict = None
307
+
308
+ if args.baseline:
309
+ try:
310
+ from framework.regression import load_baseline_manifest, compare_runs
311
+ baseline_data = load_baseline_manifest(args.baseline)
312
+ candidate_run_id = f"evalrun-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:6]}"
313
+ reg_report = compare_runs(
314
+ candidate_results=results,
315
+ baseline_data=baseline_data,
316
+ max_overall_drop=args.max_regression,
317
+ max_dim_drop=args.max_dimension_regression,
318
+ candidate_run_id=candidate_run_id,
319
+ )
320
+ regression_report_dict = reg_report.to_dict()
321
+
322
+ # Save standalone regression_report.json
323
+ reg_path = output_dir / "regression_report.json"
324
+ with open(reg_path, "w", encoding="utf-8") as f:
325
+ json.dump(redact_credentials(regression_report_dict), f, indent=2)
326
+
327
+ if reg_report.release_blocked:
328
+ evaluation_passed = False
329
+ except Exception as e:
330
+ print(f"Error performing baseline regression comparison: {e}", file=sys.stderr)
331
+ return 2
332
+
333
+ # Verify Independent Auditor Gate Decisions
334
+ for r in results:
335
+ gate_decision = getattr(r, "agent_metadata", {}).get("audit_gate_decision", "N/A")
336
+ if gate_decision == "BLOCK":
337
+ evaluation_passed = False
338
+
339
+ # Save Run Manifest
340
+ model_slug = str(args.model).replace("/", "_").replace(".", "_")
341
+ scenarios_summary = []
342
+ for r in results:
343
+ scenarios_summary.append({
344
+ "scenario_id": r.benchmark_id,
345
+ "scenario_name": r.benchmark_name,
346
+ "overall_score": r.overall_score,
347
+ "passed": r.passed,
348
+ "audit_gate_decision": getattr(r, "agent_metadata", {}).get("audit_gate_decision", "N/A"),
349
+ "report_path": f"{model_slug}_{r.benchmark_id}_report.json",
350
+ })
351
+
352
+ manifest = {
353
+ "run_id": f"evalrun-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:6]}",
354
+ "timestamp_utc": datetime.now(timezone.utc).isoformat(),
355
+ "target_agent_spec": args.agent,
356
+ "target_model": {
357
+ "model_name": args.model,
358
+ "base_url": args.base_url,
359
+ "api_key": "[REDACTED]" if args.api_key else "ENVIRONMENT_OR_EMPTY",
360
+ },
361
+ "judge_model": {
362
+ "model_name": judge_model_name,
363
+ "base_url": judge_base_url,
364
+ "api_key": "[REDACTED]" if judge_api_key else "ENVIRONMENT_OR_EMPTY",
365
+ },
366
+ "auditor_model": ({
367
+ "model_name": args.auditor_model,
368
+ "base_url": auditor_base_url,
369
+ "api_key": "[REDACTED]" if auditor_api_key else "ENVIRONMENT_OR_EMPTY",
370
+ } if args.auditor_model else None),
371
+ "baseline_path": args.baseline,
372
+ "ground_truth_path": args.ground_truth,
373
+ "output_dir": str(output_dir),
374
+ "total_scenarios": len(results),
375
+ "overall_passed": evaluation_passed,
376
+ "scenarios": scenarios_summary,
377
+ }
378
+
379
+ redacted_manifest = redact_credentials(manifest)
380
+ manifest_path = output_dir / "manifest.json"
381
+ with open(manifest_path, "w", encoding="utf-8") as f:
382
+ json.dump(redacted_manifest, f, indent=2)
383
+
384
+ # Generate Local HTML Report
385
+ try:
386
+ from cli.html_reporter import generate_html_report
387
+ generate_html_report(results, redacted_manifest, str(output_dir), regression_report=regression_report_dict)
388
+ except Exception as e:
389
+ print(f"Warning: Failed to generate HTML report: {e}", file=sys.stderr)
390
+
391
+ # Render Terminal Summary
392
+ terminal_report = format_terminal_summary(results, redacted_manifest, regression_report=regression_report_dict)
393
+ print(terminal_report)
394
+
395
+ return 0 if evaluation_passed else 1
396
+
397
+
398
+ def main(argv: Optional[List[str]] = None) -> None:
399
+ parser = create_parser()
400
+ args = parser.parse_args(argv)
401
+
402
+ if args.command == "run":
403
+ exit_code = run_command(args)
404
+ sys.exit(exit_code)
405
+ elif args.command == "ui":
406
+ from ui.server import run_ui_server
407
+ server = run_ui_server(host=args.host, port=args.port)
408
+ try:
409
+ server.serve_forever()
410
+ except KeyboardInterrupt:
411
+ print("\nShutting down local UI server.")
412
+ server.server_close()
413
+ sys.exit(0)
414
+ elif args.command == "demo":
415
+ from cli.demo import run_demo
416
+ sys.exit(run_demo(output_dir=args.output))
417
+ else:
418
+ parser.print_help()
419
+ sys.exit(2)
420
+
421
+
422
+ if __name__ == "__main__":
423
+ main()
cli/progress.py ADDED
@@ -0,0 +1,38 @@
1
+ """Interactive terminal progress indicator for long model evaluations."""
2
+
3
+ import itertools
4
+ import sys
5
+ import threading
6
+ import time
7
+ from typing import Callable, TypeVar
8
+
9
+ T = TypeVar("T")
10
+
11
+
12
+ def run_with_progress(label: str, operation: Callable[[], T]) -> T:
13
+ """Run an operation while showing a live spinner in an interactive terminal."""
14
+ interactive = bool(getattr(sys.stdout, "isatty", lambda: False)())
15
+ if not interactive:
16
+ return operation()
17
+
18
+ stop = threading.Event()
19
+ frames = itertools.cycle("⠋⠙⠹⠸⠼⠴⠦⠧⠇⠏")
20
+ statuses = itertools.cycle(("contacting model", "running agent", "scoring response", "writing report"))
21
+ started = time.monotonic()
22
+
23
+ def render() -> None:
24
+ while not stop.wait(0.35):
25
+ elapsed = int(time.monotonic() - started)
26
+ sys.stdout.write(f"\r[evalrun] {next(frames)} {label} — {next(statuses)} ({elapsed}s)")
27
+ sys.stdout.flush()
28
+
29
+ thread = threading.Thread(target=render, daemon=True)
30
+ thread.start()
31
+ try:
32
+ return operation()
33
+ finally:
34
+ stop.set()
35
+ thread.join(timeout=1.0)
36
+ elapsed = time.monotonic() - started
37
+ sys.stdout.write(f"\r[evalrun] ✓ {label} completed in {elapsed:.1f}s\033[K\n")
38
+ sys.stdout.flush()
cli/resolver.py ADDED
@@ -0,0 +1,99 @@
1
+ """Dynamic Python agent resolver for evalrun CLI."""
2
+
3
+ import importlib
4
+ import inspect
5
+ import sys
6
+ from pathlib import Path
7
+ from typing import Any
8
+ from framework.llms.base import BaseLLM
9
+
10
+
11
+ def resolve_agent(agent_spec: str, llm: BaseLLM) -> Any:
12
+ """Dynamically imports and constructs an agent instance from an import specifier.
13
+
14
+ Specifier format: 'package.module:ClassName' or 'package.module:factory_function'
15
+
16
+ Args:
17
+ agent_spec: Import specifier string (e.g. 'agents.travel:TravelPlanningAgent').
18
+ llm: Target LLM client instance to inject into the agent constructor/factory.
19
+
20
+ Returns:
21
+ An instantiated agent object.
22
+
23
+ Raises:
24
+ ValueError: If the specifier is malformed, module cannot be imported, or symbol cannot be constructed.
25
+ """
26
+ # Terminals sometimes receive escaped underscores when a command is copied
27
+ # from rendered Markdown (``custom\_agent``). They are not meaningful in a
28
+ # Python import path, so normalize them at the CLI boundary.
29
+ agent_spec = agent_spec.replace("\\_", "_").strip()
30
+
31
+ # A console-script entry point has the virtualenv's ``bin`` directory at
32
+ # sys.path[0], not the user's working directory. Add the current project
33
+ # directory so ``evalrun --agent my_agent:Agent`` works without requiring
34
+ # users to set PYTHONPATH manually.
35
+ cwd = str(Path.cwd())
36
+ if cwd not in sys.path:
37
+ sys.path.insert(0, cwd)
38
+
39
+ # 1. HTTP Endpoint Agent Resolution
40
+ if agent_spec.startswith("http://") or agent_spec.startswith("https://"):
41
+ from framework.core.adapters import HttpAgentAdapter
42
+ return HttpAgentAdapter(endpoint_url=agent_spec)
43
+
44
+ # 2. CLI Subprocess Command Agent Resolution
45
+ if agent_spec.startswith("cli:"):
46
+ from framework.core.adapters import CliAgentAdapter
47
+ command = agent_spec[4:].strip()
48
+ return CliAgentAdapter(command=command)
49
+
50
+ if ":" not in agent_spec:
51
+ raise ValueError(
52
+ f"Invalid agent specification '{agent_spec}'. Expected format 'module:Class', 'http://...', or 'cli:command'."
53
+ )
54
+
55
+ module_path, symbol_name = agent_spec.split(":", 1)
56
+ module_path = module_path.strip()
57
+ symbol_name = symbol_name.strip()
58
+
59
+ try:
60
+ module = importlib.import_module(module_path)
61
+ except ImportError as e:
62
+ raise ValueError(f"Failed to import agent module '{module_path}': {e}") from e
63
+
64
+ if not hasattr(module, symbol_name):
65
+ raise ValueError(f"Module '{module_path}' has no attribute or class '{symbol_name}'.")
66
+
67
+ symbol = getattr(module, symbol_name)
68
+
69
+ if inspect.isclass(symbol):
70
+ sig = inspect.signature(symbol.__init__)
71
+ params = sig.parameters
72
+ if "llm" in params:
73
+ return symbol(llm=llm)
74
+ elif len(params) <= 1: # Only self
75
+ return symbol()
76
+ else:
77
+ # Try passing llm as first positional argument
78
+ try:
79
+ return symbol(llm)
80
+ except Exception as e:
81
+ raise ValueError(
82
+ f"Could not instantiate agent class '{symbol_name}' with LLM parameter: {e}"
83
+ ) from e
84
+ elif callable(symbol):
85
+ sig = inspect.signature(symbol)
86
+ params = sig.parameters
87
+ if "llm" in params:
88
+ return symbol(llm=llm)
89
+ elif len(params) == 0:
90
+ return symbol()
91
+ else:
92
+ try:
93
+ return symbol(llm)
94
+ except Exception as e:
95
+ raise ValueError(
96
+ f"Could not invoke agent factory '{symbol_name}' with LLM parameter: {e}"
97
+ ) from e
98
+ else:
99
+ raise ValueError(f"Symbol '{symbol_name}' in module '{module_path}' is neither a class nor a callable factory.")