palace-eval 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. ___dashboard/node_modules/flatted/python/flatted.py +163 -0
  2. palace/__init__.py +25 -0
  3. palace/agents/__init__.py +19 -0
  4. palace/agents/api_agent.py +124 -0
  5. palace/agents/base_agent.py +66 -0
  6. palace/agents/mcp_agent.py +119 -0
  7. palace/agents/vivarium_agent.py +429 -0
  8. palace/analyzers/__init__.py +20 -0
  9. palace/analyzers/base.py +66 -0
  10. palace/analyzers/citation_verifier.py +426 -0
  11. palace/analyzers/fetch.py +198 -0
  12. palace/bundled_io_adapters.yaml +135 -0
  13. palace/data_utils/__init__.py +14 -0
  14. palace/data_utils/cybench_dataset/__init__.py +15 -0
  15. palace/data_utils/cybench_dataset/create_dataset.py +249 -0
  16. palace/data_utils/cybench_dataset/seed.py +31 -0
  17. palace/data_utils/cybench_dataset/verify.py +46 -0
  18. palace/data_utils/deepconsult_dataset/create_dataset.py +105 -0
  19. palace/data_utils/docretrieval_dataset/create_dataset.py +381 -0
  20. palace/data_utils/guardbench_dataset/create_dataset.py +113 -0
  21. palace/data_utils/healthbench_dataset/create_dataset.py +146 -0
  22. palace/data_utils/mmsafetybench_dataset/create_dataset.py +152 -0
  23. palace/data_utils/rtvlm_dataset/create_dataset.py +239 -0
  24. palace/data_utils/ruler_dataset/create_dataset.py +450 -0
  25. palace/data_utils/scale_multichallenge_dataset/create_dataset.py +95 -0
  26. palace/data_utils/scopus_dataset/create_dataset.py +138 -0
  27. palace/data_utils/swe_bench_dataset/create_dataset.py +172 -0
  28. palace/data_utils/swe_bench_dataset/seed.py +41 -0
  29. palace/data_utils/swe_bench_dataset/verify.py +276 -0
  30. palace/data_utils/swe_bench_pro_dataset/__init__.py +15 -0
  31. palace/data_utils/swe_bench_pro_dataset/create_dataset.py +193 -0
  32. palace/data_utils/swe_bench_pro_dataset/seed.py +34 -0
  33. palace/data_utils/swe_bench_pro_dataset/verify.py +89 -0
  34. palace/data_utils/sycophancy_dataset/create_dataset.py +209 -0
  35. palace/data_utils/values_dataset/create_dataset.py +112 -0
  36. palace/data_utils/vlguard_dataset/create_dataset.py +156 -0
  37. palace/entrypoints/__init__.py +15 -0
  38. palace/entrypoints/download/palace_download.py +900 -0
  39. palace/entrypoints/download/public_datasets_info.json +659 -0
  40. palace/entrypoints/palace_cli.py +274 -0
  41. palace/entrypoints/palace_run.py +165 -0
  42. palace/evaluation/__init__.py +19 -0
  43. palace/evaluation/dispatch.py +103 -0
  44. palace/evaluation/orchestrator.py +491 -0
  45. palace/evaluation/pipeline.py +283 -0
  46. palace/evaluation/renderers.py +389 -0
  47. palace/evaluation/types.py +89 -0
  48. palace/judge.py +183 -0
  49. palace/mcp_utils/__init__.py +14 -0
  50. palace/mcp_utils/mcp_client.py +67 -0
  51. palace/models/__init__.py +18 -0
  52. palace/models/api_model.py +399 -0
  53. palace/models/base_model.py +44 -0
  54. palace/prompts/fact_prompts.py +92 -0
  55. palace/task_types/__init__.py +42 -0
  56. palace/task_types/agentic.py +112 -0
  57. palace/task_types/base.py +225 -0
  58. palace/task_types/classification.py +159 -0
  59. palace/task_types/criteria_evaluation.py +503 -0
  60. palace/task_types/instruction_following.py +273 -0
  61. palace/task_types/qa.py +161 -0
  62. palace/utils/__init__.py +36 -0
  63. palace/utils/constants.py +30 -0
  64. palace/utils/exceptions.py +25 -0
  65. palace/utils/io_adapters.py +228 -0
  66. palace/utils/model_extra_params.py +94 -0
  67. palace/utils/multimodal.py +183 -0
  68. palace/utils/paths.py +39 -0
  69. palace/utils/printing.py +440 -0
  70. palace/utils/secrets.py +28 -0
  71. palace/utils/threading.py +149 -0
  72. palace_eval-1.0.0.dist-info/METADATA +236 -0
  73. palace_eval-1.0.0.dist-info/RECORD +78 -0
  74. palace_eval-1.0.0.dist-info/WHEEL +5 -0
  75. palace_eval-1.0.0.dist-info/entry_points.txt +4 -0
  76. palace_eval-1.0.0.dist-info/licenses/LICENSE +289 -0
  77. palace_eval-1.0.0.dist-info/licenses/NOTICE +143 -0
  78. palace_eval-1.0.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,163 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ # ISC License
16
+ #
17
+ # Copyright (c) 2018-2025, Andrea Giammarchi, @WebReflection
18
+ #
19
+ # Permission to use, copy, modify, and/or distribute this software for any
20
+ # purpose with or without fee is hereby granted, provided that the above
21
+ # copyright notice and this permission notice appear in all copies.
22
+ #
23
+ # THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
24
+ # REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
25
+ # AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
26
+ # INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
27
+ # LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
28
+ # OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
29
+ # PERFORMANCE OF THIS SOFTWARE.
30
+
31
+ import json as _json
32
+
33
+ class _Known:
34
+ def __init__(self):
35
+ self.key = []
36
+ self.value = []
37
+
38
+ class _String:
39
+ def __init__(self, value):
40
+ self.value = value
41
+
42
+
43
+ def _array_keys(value):
44
+ keys = []
45
+ i = 0
46
+ for _ in value:
47
+ keys.append(i)
48
+ i += 1
49
+ return keys
50
+
51
+ def _object_keys(value):
52
+ keys = []
53
+ for key in value:
54
+ keys.append(key)
55
+ return keys
56
+
57
+ def _is_array(value):
58
+ return isinstance(value, (list, tuple))
59
+
60
+ def _is_object(value):
61
+ return isinstance(value, dict)
62
+
63
+ def _is_string(value):
64
+ return isinstance(value, str)
65
+
66
+ def _index(known, input, value):
67
+ input.append(value)
68
+ index = str(len(input) - 1)
69
+ known.key.append(value)
70
+ known.value.append(index)
71
+ return index
72
+
73
+ def _loop(keys, input, known, output):
74
+ for key in keys:
75
+ value = output[key]
76
+ if isinstance(value, _String):
77
+ _ref(key, input[int(value.value)], input, known, output)
78
+
79
+ return output
80
+
81
+ def _ref(key, value, input, known, output):
82
+ if _is_array(value) and value not in known:
83
+ known.append(value)
84
+ value = _loop(_array_keys(value), input, known, value)
85
+ elif _is_object(value) and value not in known:
86
+ known.append(value)
87
+ value = _loop(_object_keys(value), input, known, value)
88
+
89
+ output[key] = value
90
+
91
+ def _relate(known, input, value):
92
+ if _is_string(value) or _is_array(value) or _is_object(value):
93
+ try:
94
+ return known.value[known.key.index(value)]
95
+ except:
96
+ return _index(known, input, value)
97
+
98
+ return value
99
+
100
+ def _transform(known, input, value):
101
+ if _is_array(value):
102
+ output = []
103
+ for val in value:
104
+ output.append(_relate(known, input, val))
105
+ return output
106
+
107
+ if _is_object(value):
108
+ obj = {}
109
+ for key in value:
110
+ obj[key] = _relate(known, input, value[key])
111
+ return obj
112
+
113
+ return value
114
+
115
+ def _wrap(value):
116
+ if _is_string(value):
117
+ return _String(value)
118
+
119
+ if _is_array(value):
120
+ i = 0
121
+ for val in value:
122
+ value[i] = _wrap(val)
123
+ i += 1
124
+
125
+ elif _is_object(value):
126
+ for key in value:
127
+ value[key] = _wrap(value[key])
128
+
129
+ return value
130
+
131
+ def parse(value, *args, **kwargs):
132
+ json = _json.loads(value, *args, **kwargs)
133
+ wrapped = []
134
+ for value in json:
135
+ wrapped.append(_wrap(value))
136
+
137
+ input = []
138
+ for value in wrapped:
139
+ if isinstance(value, _String):
140
+ input.append(value.value)
141
+ else:
142
+ input.append(value)
143
+
144
+ value = input[0]
145
+
146
+ if _is_array(value):
147
+ return _loop(_array_keys(value), input, [value], value)
148
+
149
+ if _is_object(value):
150
+ return _loop(_object_keys(value), input, [value], value)
151
+
152
+ return value
153
+
154
+
155
+ def stringify(value, *args, **kwargs):
156
+ known = _Known()
157
+ input = []
158
+ output = []
159
+ i = int(_index(known, input, value))
160
+ while i < len(input):
161
+ output.append(_transform(known, input, input[i]))
162
+ i += 1
163
+ return _json.dumps(output, *args, **kwargs)
palace/__init__.py ADDED
@@ -0,0 +1,25 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+
16
+ def __getattr__(name):
17
+ if name == "Evaluation":
18
+ from .evaluation.orchestrator import Evaluation
19
+
20
+ return Evaluation
21
+ if name == "evaluate":
22
+ from .evaluation.orchestrator import evaluate
23
+
24
+ return evaluate
25
+ raise AttributeError(f"module 'palace' has no attribute {name!r}")
@@ -0,0 +1,19 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ # ruff: noqa: I001
16
+ from palace.agents.base_agent import Agent as Agent
17
+ from palace.agents.mcp_agent import MCPAgent as MCPAgent
18
+ from palace.agents.api_agent import APIAgent as APIAgent
19
+ from palace.agents.vivarium_agent import VivariumAgent as VivariumAgent
@@ -0,0 +1,124 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ import logging
16
+ from typing import TYPE_CHECKING
17
+
18
+ from palace.agents import Agent
19
+ from palace.evaluation.types import AgentResult
20
+ from palace.models.api_model import create_api_model
21
+ from palace.utils.multimodal import build_multimodal_content
22
+ from palace.utils.printing import print
23
+
24
+ if TYPE_CHECKING:
25
+ from palace.evaluation.types import Attachment
26
+
27
+ _logger = logging.getLogger("palace.api_agent")
28
+
29
+ # Patterns that indicate a deterministic model capability limitation (not transient)
30
+ _UNSUPPORTED_PATTERNS = [
31
+ "context_length_exceeded",
32
+ "too many tokens",
33
+ "prompt is too long",
34
+ "maximum context length",
35
+ "input too long",
36
+ "exceeds the model's maximum",
37
+ "content_too_large",
38
+ "request too large",
39
+ "input tokens exceed",
40
+ ]
41
+
42
+
43
+ def _is_unsupported_error(e: Exception) -> bool:
44
+ """Detect deterministic capability limitations from API errors.
45
+
46
+ Returns True for errors that indicate the model cannot process the input
47
+ (e.g., context too long, unsupported modality). These are permanent for
48
+ the given input and should not be retried.
49
+ """
50
+ # OpenAI / vLLM: BadRequestError has a .code attribute
51
+ if hasattr(e, "code") and e.code == "context_length_exceeded":
52
+ return True
53
+ # Message-based detection (Anthropic, vLLM variants, other providers)
54
+ msg = str(e).lower()
55
+ return any(p in msg for p in _UNSUPPORTED_PATTERNS)
56
+
57
+
58
+ class APIAgent(Agent):
59
+ """Agent that calls an API endpoint (OpenAI-compatible, Azure OpenAI, or Anthropic).
60
+
61
+ Can be used for both agentic and non-agentic (black-box LLM) evaluation.
62
+ """
63
+
64
+ def __init__(
65
+ self,
66
+ /,
67
+ name: str,
68
+ url: str,
69
+ token: str | None = None,
70
+ api_type: str | None = None,
71
+ extra_params: dict | None = None,
72
+ ):
73
+ """Initialize an APIAgent.
74
+
75
+ Args:
76
+ name: The name of the agent, corresponding to a model ID on the API server.
77
+ url: The URL of the API server.
78
+ token: The API token for authentication. Defaults to None.
79
+ api_type: The API type to use ("openai", "anthropic", or "azure").
80
+ Auto-detected from model name if not specified.
81
+ extra_params: Extra kwargs to merge into every API call for this model.
82
+ """
83
+ if api_type is not None and api_type not in ["openai", "anthropic", "azure"]:
84
+ raise ValueError("api_type must be 'openai', 'anthropic', or 'azure'")
85
+ self._name = name
86
+ self.url = url
87
+ self.token = token
88
+ self.api_type = api_type
89
+ self._model = create_api_model(
90
+ model_id=name, url=url, token=token, api_type=api_type, extra_params=extra_params
91
+ )
92
+
93
+ @property
94
+ def name(self) -> str:
95
+ return self._name
96
+
97
+ async def run(
98
+ self, prompt: str, attachments: "list[Attachment] | None" = None, *, task_id: str | None = None
99
+ ) -> AgentResult:
100
+ self._model.quiet = not self.verbose
101
+ # Inline text attachments into prompt (non-agentic presentation)
102
+ if attachments:
103
+ prompt = self._inline_text_attachments(prompt, attachments)
104
+ content = build_multimodal_content(prompt, attachments)
105
+ try:
106
+ output = await self._model.generate([{"role": "user", "content": content}])
107
+ except Exception as e:
108
+ _logger.warning(f"Agent error: {e}")
109
+ if self.verbose:
110
+ print(f"[bold red]OpenAIAPI agent error: {e}[/]")
111
+ if _is_unsupported_error(e):
112
+ return AgentResult(outcome="unsupported", reason=f"unsupported: {e}")
113
+ return AgentResult(outcome="error", reason=f"agent_error: {e}")
114
+
115
+ return AgentResult(answer=output, metrics={})
116
+
117
+ @staticmethod
118
+ def _inline_text_attachments(prompt: str, attachments: "list[Attachment]") -> str:
119
+ """Inline text file content into the prompt for non-agentic evaluation."""
120
+ for att in attachments:
121
+ text = att.read_text()
122
+ if text:
123
+ prompt = f"Start of text attachment >>>\n{text}\n<<< End of text attachment\n\n{prompt}"
124
+ return prompt
@@ -0,0 +1,66 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ from abc import ABC, abstractmethod
16
+ from pathlib import Path
17
+ from typing import TYPE_CHECKING
18
+
19
+ if TYPE_CHECKING:
20
+ from palace.evaluation.types import AgentResult, Attachment
21
+ from palace.task_types.base import ExecutionEnvironment, Task
22
+
23
+
24
+ class Agent(ABC):
25
+ """Base class that defines the agent interface."""
26
+
27
+ verbose: bool = True
28
+ agentic: bool = False
29
+
30
+ @property
31
+ @abstractmethod
32
+ def name(self) -> str:
33
+ pass
34
+
35
+ @abstractmethod
36
+ async def run(
37
+ self, prompt: str, attachments: "list[Attachment] | None" = None, *, task_id: str | None = None
38
+ ) -> "AgentResult":
39
+ """Run the agent on the given prompt and return a result.
40
+
41
+ Args:
42
+ prompt: The text prompt for the agent
43
+ attachments: Optional list of Attachment objects for multimodal tasks
44
+ task_id: Optional task identifier for agents that need to correlate
45
+ with per-task state (e.g. VivariumAgent environments)
46
+
47
+ Returns:
48
+ AgentResult with answer, metrics, and skip status.
49
+ """
50
+ pass
51
+
52
+ async def on_tasklist_start(self, tasklist_path: Path, info: dict) -> None:
53
+ """Called before evaluating a tasklist. Override for setup."""
54
+ pass
55
+
56
+ async def on_tasklist_end(self) -> None:
57
+ """Called after evaluating a tasklist. Override for cleanup."""
58
+ pass
59
+
60
+ async def on_task_start(self, task: "Task") -> "ExecutionEnvironment | None":
61
+ """Called before each task. Return execution environment for agentic verification."""
62
+ return None
63
+
64
+ async def on_task_end(self, task: "Task") -> None:
65
+ """Called after each task (including verify). Override for per-task cleanup."""
66
+ pass
@@ -0,0 +1,119 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ import json
16
+ from typing import Any, Callable
17
+
18
+ from mcp.types import CallToolResult, TextContent
19
+ from tenacity import retry, retry_if_exception_type, stop_after_attempt, wait_exponential
20
+
21
+ from palace.agents import Agent
22
+ from palace.evaluation.types import AgentResult
23
+ from palace.mcp_utils.mcp_client import _call_tool, list_tools
24
+
25
+
26
+ class MCPAgent(Agent):
27
+ """Agent that connects to a remote MCP server."""
28
+
29
+ def __init__(
30
+ self,
31
+ url: str,
32
+ token: str | None = None,
33
+ name: str | None = None,
34
+ params: dict[str, Any] | None = None,
35
+ output_processor: Callable[[CallToolResult], str] | None = None,
36
+ ):
37
+ """Initialize the MCPAgent.
38
+
39
+ Args:
40
+ url: The URL of the MCP server.
41
+ token: Authentication token for the MCP server.
42
+ name: Name of the agent/tool to connect to.
43
+ params: Custom parameters to pass to the agent/tool.
44
+ output_processor: Function to process the agent/tool output.
45
+ """
46
+ self.url = url
47
+ self.token = token
48
+ self.params = params
49
+ self.output_processor = output_processor
50
+
51
+ available_agents = list_tools(url, token).tools
52
+
53
+ if len(available_agents) == 0:
54
+ raise ValueError(f"There is no agent or tool at {url}.")
55
+
56
+ if name is not None and name in [a.name for a in available_agents]:
57
+ self._name = name
58
+ elif name is not None and name not in [a.name for a in available_agents]:
59
+ raise ValueError(
60
+ f"There is no agent with the provided name {name} at {url}, only found: {available_agents}."
61
+ )
62
+ elif name is None and len(available_agents) > 1:
63
+ raise ValueError(
64
+ f"There is more than one agent at {url} but provided name is {name}. Specify one of {available_agents}."
65
+ )
66
+ else:
67
+ self._name = available_agents[0].name
68
+
69
+ if self.params is not None and "main" in self.params:
70
+ self._input_parameter = self.params["main"]
71
+ else:
72
+ try:
73
+ self._input_parameter = list(
74
+ [a for a in available_agents if a.name == self._name][0].inputSchema["properties"].keys()
75
+ )[0]
76
+ except Exception as e:
77
+ raise ValueError(f"Can't find the input parameter for the agent {self._name} at {url}.") from e
78
+
79
+ @property
80
+ def name(self) -> str:
81
+ return self._name
82
+
83
+ @retry(
84
+ stop=stop_after_attempt(5),
85
+ wait=wait_exponential(multiplier=5, max=60),
86
+ retry=retry_if_exception_type((ConnectionError, TimeoutError, OSError)),
87
+ )
88
+ async def _run_with_retry(self, task: str) -> AgentResult:
89
+ params = {self._input_parameter: task}
90
+ if self.params is not None and "custom" in self.params:
91
+ params |= self.params["custom"]
92
+
93
+ output: CallToolResult = await _call_tool(self.url, self.name, params, self.token)
94
+
95
+ if self.output_processor is not None:
96
+ answer = self.output_processor(output)
97
+ else:
98
+ if not isinstance(output.content[0], TextContent):
99
+ raise ValueError(f"MCPAgent expected TextContent, got: {type(output.content[0])}")
100
+ answer = output.content[0].text
101
+ if not isinstance(answer, str) or answer.strip() == "":
102
+ raise ValueError(f"MCPAgent answer not found in output content. Got: {output.content}")
103
+
104
+ try:
105
+ assert isinstance(output.content[1], TextContent)
106
+ metrics = json.loads(output.content[1].text)
107
+ assert isinstance(metrics, dict)
108
+ except Exception:
109
+ metrics = None
110
+
111
+ return AgentResult(answer=answer, metrics=metrics)
112
+
113
+ async def run(
114
+ self, prompt: str, attachments: "list[Any] | None" = None, *, task_id: str | None = None
115
+ ) -> AgentResult:
116
+ if attachments:
117
+ # TODO: pass image attachments as ImageContent when MCP SDK supports it in tool calls
118
+ return AgentResult(outcome="error", reason="unsupported_attachment")
119
+ return await self._run_with_retry(prompt)