palace-eval 1.0.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. palace/__init__.py +30 -0
  2. palace/agents/__init__.py +20 -0
  3. palace/agents/api_agent.py +148 -0
  4. palace/agents/base_agent.py +66 -0
  5. palace/agents/mcp_agent.py +119 -0
  6. palace/agents/vivarium_agent.py +438 -0
  7. palace/analyzers/__init__.py +20 -0
  8. palace/analyzers/base.py +66 -0
  9. palace/analyzers/citation_verifier.py +433 -0
  10. palace/analyzers/fetch.py +198 -0
  11. palace/bundled_io_adapters.yaml +135 -0
  12. palace/bundled_model_extra_params.yaml +29 -0
  13. palace/cli/__init__.py +104 -0
  14. palace/cli/config.py +472 -0
  15. palace/cli/discovery.py +440 -0
  16. palace/cli/download_cmd.py +310 -0
  17. palace/cli/git_adapters/__init__.py +167 -0
  18. palace/cli/git_adapters/base.py +132 -0
  19. palace/cli/git_adapters/github.py +264 -0
  20. palace/cli/git_adapters/gitlab.py +265 -0
  21. palace/cli/git_adapters/huggingface.py +160 -0
  22. palace/cli/git_adapters/local.py +184 -0
  23. palace/cli/init_cmd.py +218 -0
  24. palace/cli/local.py +154 -0
  25. palace/cli/publish.py +256 -0
  26. palace/cli/results.py +167 -0
  27. palace/cli/run.py +239 -0
  28. palace/cli/sources/__init__.py +20 -0
  29. palace/cli/sources/cache.py +130 -0
  30. palace/cli/sources/manager.py +378 -0
  31. palace/cli/sources_cmd.py +128 -0
  32. palace/cli/validate.py +97 -0
  33. palace/cli/validation/__init__.py +19 -0
  34. palace/cli/validation/validator.py +520 -0
  35. palace/cli/wizard.py +253 -0
  36. palace/download/__init__.py +217 -0
  37. palace/download/conversion_recipes.json +659 -0
  38. palace/entrypoints/__init__.py +15 -0
  39. palace/entrypoints/deprecated.py +96 -0
  40. palace/entrypoints/download/palace_download.py +897 -0
  41. palace/entrypoints/download/public_datasets_info.json +659 -0
  42. palace/entrypoints/palace_cli.py +252 -0
  43. palace/entrypoints/palace_run.py +165 -0
  44. palace/evaluation/__init__.py +19 -0
  45. palace/evaluation/dispatch.py +122 -0
  46. palace/evaluation/judge_config.py +20 -0
  47. palace/evaluation/orchestrator.py +498 -0
  48. palace/evaluation/pipeline.py +294 -0
  49. palace/evaluation/renderers.py +389 -0
  50. palace/evaluation/types.py +89 -0
  51. palace/judge.py +216 -0
  52. palace/mcp_utils/__init__.py +14 -0
  53. palace/mcp_utils/mcp_client.py +67 -0
  54. palace/models/__init__.py +18 -0
  55. palace/models/api_model.py +399 -0
  56. palace/models/base_model.py +44 -0
  57. palace/prompts/fact_prompts.py +92 -0
  58. palace/task_types/__init__.py +42 -0
  59. palace/task_types/agentic.py +112 -0
  60. palace/task_types/base.py +226 -0
  61. palace/task_types/classification.py +159 -0
  62. palace/task_types/criteria_evaluation.py +515 -0
  63. palace/task_types/instruction_following.py +273 -0
  64. palace/task_types/qa.py +165 -0
  65. palace/utils/__init__.py +36 -0
  66. palace/utils/config.py +174 -0
  67. palace/utils/constants.py +67 -0
  68. palace/utils/exceptions.py +58 -0
  69. palace/utils/io_adapters.py +228 -0
  70. palace/utils/model_extra_params.py +94 -0
  71. palace/utils/multimodal.py +183 -0
  72. palace/utils/paths.py +39 -0
  73. palace/utils/printing.py +440 -0
  74. palace/utils/secrets.py +51 -0
  75. palace/utils/threading.py +149 -0
  76. palace_eval-1.0.7.dist-info/METADATA +242 -0
  77. palace_eval-1.0.7.dist-info/RECORD +82 -0
  78. palace_eval-1.0.7.dist-info/WHEEL +5 -0
  79. palace_eval-1.0.7.dist-info/entry_points.txt +5 -0
  80. palace_eval-1.0.7.dist-info/licenses/LICENSE +289 -0
  81. palace_eval-1.0.7.dist-info/licenses/NOTICE +143 -0
  82. palace_eval-1.0.7.dist-info/top_level.txt +1 -0
palace/__init__.py ADDED
@@ -0,0 +1,30 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ from .evaluation.judge_config import JudgeConfig
16
+
17
+ __all__ = ["JudgeConfig"]
18
+
19
+
20
+ def __getattr__(name):
21
+ # Lazy imports for heavy modules
22
+ if name == "Evaluation":
23
+ from .evaluation.orchestrator import Evaluation
24
+
25
+ return Evaluation
26
+ if name == "evaluate":
27
+ from .evaluation.orchestrator import evaluate
28
+
29
+ return evaluate
30
+ raise AttributeError(f"module 'palace' has no attribute {name!r}")
@@ -0,0 +1,20 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ # ruff: noqa: I001
16
+ from palace.agents.base_agent import Agent as Agent
17
+ from palace.agents.mcp_agent import MCPAgent as MCPAgent
18
+ from palace.agents.api_agent import APIAgent as APIAgent
19
+ from palace.agents.vivarium_agent import VivariumAgent as VivariumAgent
20
+ from palace.utils.exceptions import ModelNotFoundError as ModelNotFoundError
@@ -0,0 +1,148 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ import logging
16
+ from typing import TYPE_CHECKING
17
+
18
+ from palace.agents import Agent
19
+ from palace.evaluation.types import AgentResult
20
+ from palace.models.api_model import create_api_model
21
+ from palace.utils.exceptions import ModelNotFoundError
22
+ from palace.utils.multimodal import build_multimodal_content
23
+ from palace.utils.printing import print
24
+
25
+ if TYPE_CHECKING:
26
+ from palace.evaluation.types import Attachment
27
+
28
+ _logger = logging.getLogger("palace.api_agent")
29
+
30
+ # Patterns that indicate a deterministic model capability limitation (not transient)
31
+ _UNSUPPORTED_PATTERNS = [
32
+ "context_length_exceeded",
33
+ "too many tokens",
34
+ "prompt is too long",
35
+ "maximum context length",
36
+ "input too long",
37
+ "exceeds the model's maximum",
38
+ "content_too_large",
39
+ "request too large",
40
+ "input tokens exceed",
41
+ ]
42
+
43
+ # Patterns indicating the model doesn't exist — should abort evaluation
44
+ _MODEL_NOT_FOUND_PATTERNS = [
45
+ "model not found",
46
+ "not available",
47
+ "does not exist",
48
+ "unknown model",
49
+ "invalid model",
50
+ "no such model",
51
+ ]
52
+
53
+
54
+ def _is_unsupported_error(e: Exception) -> bool:
55
+ """Detect deterministic capability limitations from API errors.
56
+
57
+ Returns True for errors that indicate the model cannot process the input
58
+ (e.g., context too long, unsupported modality). These are permanent for
59
+ the given input and should not be retried.
60
+ """
61
+ # OpenAI / vLLM: BadRequestError has a .code attribute
62
+ if hasattr(e, "code") and e.code == "context_length_exceeded":
63
+ return True
64
+ # Message-based detection (Anthropic, vLLM variants, other providers)
65
+ msg = str(e).lower()
66
+ return any(p in msg for p in _UNSUPPORTED_PATTERNS)
67
+
68
+
69
+ def _is_model_not_found(e: Exception) -> bool:
70
+ """Detect model-not-found errors that should abort the evaluation.
71
+
72
+ Returns True for errors that indicate the model doesn't exist on the endpoint.
73
+ These are configuration errors, not per-task issues.
74
+ """
75
+ msg = str(e).lower()
76
+ return any(p in msg for p in _MODEL_NOT_FOUND_PATTERNS)
77
+
78
+
79
+ class APIAgent(Agent):
80
+ """Agent that calls an API endpoint (OpenAI-compatible, Azure OpenAI, or Anthropic).
81
+
82
+ Can be used for both agentic and non-agentic (black-box LLM) evaluation.
83
+ """
84
+
85
+ def __init__(
86
+ self,
87
+ /,
88
+ name: str,
89
+ url: str,
90
+ token: str | None = None,
91
+ api_type: str | None = None,
92
+ extra_params: dict | None = None,
93
+ ):
94
+ """Initialize an APIAgent.
95
+
96
+ Args:
97
+ name: The name of the agent, corresponding to a model ID on the API server.
98
+ url: The URL of the API server.
99
+ token: The API token for authentication. Defaults to None.
100
+ api_type: The API type to use ("openai", "anthropic", or "azure").
101
+ Auto-detected from model name if not specified.
102
+ extra_params: Extra kwargs to merge into every API call for this model.
103
+ """
104
+ if api_type is not None and api_type not in ["openai", "anthropic", "azure"]:
105
+ raise ValueError("api_type must be 'openai', 'anthropic', or 'azure'")
106
+ self._name = name
107
+ self.url = url
108
+ self.token = token
109
+ self.api_type = api_type
110
+ self._model = create_api_model(
111
+ model_id=name, url=url, token=token, api_type=api_type, extra_params=extra_params
112
+ )
113
+
114
+ @property
115
+ def name(self) -> str:
116
+ return self._name
117
+
118
+ async def run(
119
+ self, prompt: str, attachments: "list[Attachment] | None" = None, *, task_id: str | None = None
120
+ ) -> AgentResult:
121
+ self._model.quiet = not self.verbose
122
+ # Inline text attachments into prompt (non-agentic presentation)
123
+ if attachments:
124
+ prompt = self._inline_text_attachments(prompt, attachments)
125
+ content = build_multimodal_content(prompt, attachments)
126
+ try:
127
+ output = await self._model.generate([{"role": "user", "content": content}])
128
+ except Exception as e:
129
+ _logger.warning(f"Agent error: {e}")
130
+ if self.verbose:
131
+ print(f"[bold red]OpenAIAPI agent error: {e}[/]")
132
+ # Model not found is a fatal configuration error — abort evaluation
133
+ if _is_model_not_found(e):
134
+ raise ModelNotFoundError(f"Model '{self._name}' not found on {self.url}: {e}") from e
135
+ if _is_unsupported_error(e):
136
+ return AgentResult(outcome="unsupported", reason=f"unsupported: {e}")
137
+ return AgentResult(outcome="error", reason=f"agent_error: {e}")
138
+
139
+ return AgentResult(answer=output, metrics={})
140
+
141
+ @staticmethod
142
+ def _inline_text_attachments(prompt: str, attachments: "list[Attachment]") -> str:
143
+ """Inline text file content into the prompt for non-agentic evaluation."""
144
+ for att in attachments:
145
+ text = att.read_text()
146
+ if text:
147
+ prompt = f"Start of text attachment >>>\n{text}\n<<< End of text attachment\n\n{prompt}"
148
+ return prompt
@@ -0,0 +1,66 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ from abc import ABC, abstractmethod
16
+ from pathlib import Path
17
+ from typing import TYPE_CHECKING
18
+
19
+ if TYPE_CHECKING:
20
+ from palace.evaluation.types import AgentResult, Attachment
21
+ from palace.task_types.base import ExecutionEnvironment, Task
22
+
23
+
24
+ class Agent(ABC):
25
+ """Base class that defines the agent interface."""
26
+
27
+ verbose: bool = True
28
+ agentic: bool = False
29
+
30
+ @property
31
+ @abstractmethod
32
+ def name(self) -> str:
33
+ pass
34
+
35
+ @abstractmethod
36
+ async def run(
37
+ self, prompt: str, attachments: "list[Attachment] | None" = None, *, task_id: str | None = None
38
+ ) -> "AgentResult":
39
+ """Run the agent on the given prompt and return a result.
40
+
41
+ Args:
42
+ prompt: The text prompt for the agent
43
+ attachments: Optional list of Attachment objects for multimodal tasks
44
+ task_id: Optional task identifier for agents that need to correlate
45
+ with per-task state (e.g. VivariumAgent environments)
46
+
47
+ Returns:
48
+ AgentResult with answer, metrics, and skip status.
49
+ """
50
+ pass
51
+
52
+ async def on_tasklist_start(self, tasklist_path: Path, info: dict) -> None:
53
+ """Called before evaluating a tasklist. Override for setup."""
54
+ pass
55
+
56
+ async def on_tasklist_end(self) -> None:
57
+ """Called after evaluating a tasklist. Override for cleanup."""
58
+ pass
59
+
60
+ async def on_task_start(self, task: "Task") -> "ExecutionEnvironment | None":
61
+ """Called before each task. Return execution environment for agentic verification."""
62
+ return None
63
+
64
+ async def on_task_end(self, task: "Task") -> None:
65
+ """Called after each task (including verify). Override for per-task cleanup."""
66
+ pass
@@ -0,0 +1,119 @@
1
+ # Copyright (C) 2025 European Union
2
+ #
3
+ # This program is free software: you can redistribute it and/or modify
4
+ # it under the terms of the European Union Public Licence (EUPL) v. 1.2
5
+ # as published by the European Union.
6
+ #
7
+ # This program is distributed in the hope that it will be useful,
8
+ # but WITHOUT ANY WARRANTY; without even the implied warranty of
9
+ # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
10
+ # European Union Public Licence for more details.
11
+ #
12
+ # You should have received a copy of the European Union Public Licence
13
+ # along with this program. If not, see <https://joinup.ec.europa.eu/collection/eupl/eupl-text-eupl-12>.
14
+
15
+ import json
16
+ from typing import Any, Callable
17
+
18
+ from mcp.types import CallToolResult, TextContent
19
+ from tenacity import retry, retry_if_exception_type, stop_after_attempt, wait_exponential
20
+
21
+ from palace.agents import Agent
22
+ from palace.evaluation.types import AgentResult
23
+ from palace.mcp_utils.mcp_client import _call_tool, list_tools
24
+
25
+
26
+ class MCPAgent(Agent):
27
+ """Agent that connects to a remote MCP server."""
28
+
29
+ def __init__(
30
+ self,
31
+ url: str,
32
+ token: str | None = None,
33
+ name: str | None = None,
34
+ params: dict[str, Any] | None = None,
35
+ output_processor: Callable[[CallToolResult], str] | None = None,
36
+ ):
37
+ """Initialize the MCPAgent.
38
+
39
+ Args:
40
+ url: The URL of the MCP server.
41
+ token: Authentication token for the MCP server.
42
+ name: Name of the agent/tool to connect to.
43
+ params: Custom parameters to pass to the agent/tool.
44
+ output_processor: Function to process the agent/tool output.
45
+ """
46
+ self.url = url
47
+ self.token = token
48
+ self.params = params
49
+ self.output_processor = output_processor
50
+
51
+ available_agents = list_tools(url, token).tools
52
+
53
+ if len(available_agents) == 0:
54
+ raise ValueError(f"There is no agent or tool at {url}.")
55
+
56
+ if name is not None and name in [a.name for a in available_agents]:
57
+ self._name = name
58
+ elif name is not None and name not in [a.name for a in available_agents]:
59
+ raise ValueError(
60
+ f"There is no agent with the provided name {name} at {url}, only found: {available_agents}."
61
+ )
62
+ elif name is None and len(available_agents) > 1:
63
+ raise ValueError(
64
+ f"There is more than one agent at {url} but provided name is {name}. Specify one of {available_agents}."
65
+ )
66
+ else:
67
+ self._name = available_agents[0].name
68
+
69
+ if self.params is not None and "main" in self.params:
70
+ self._input_parameter = self.params["main"]
71
+ else:
72
+ try:
73
+ self._input_parameter = list(
74
+ [a for a in available_agents if a.name == self._name][0].inputSchema["properties"].keys()
75
+ )[0]
76
+ except Exception as e:
77
+ raise ValueError(f"Can't find the input parameter for the agent {self._name} at {url}.") from e
78
+
79
+ @property
80
+ def name(self) -> str:
81
+ return self._name
82
+
83
+ @retry(
84
+ stop=stop_after_attempt(5),
85
+ wait=wait_exponential(multiplier=5, max=60),
86
+ retry=retry_if_exception_type((ConnectionError, TimeoutError, OSError)),
87
+ )
88
+ async def _run_with_retry(self, task: str) -> AgentResult:
89
+ params = {self._input_parameter: task}
90
+ if self.params is not None and "custom" in self.params:
91
+ params |= self.params["custom"]
92
+
93
+ output: CallToolResult = await _call_tool(self.url, self.name, params, self.token)
94
+
95
+ if self.output_processor is not None:
96
+ answer = self.output_processor(output)
97
+ else:
98
+ if not isinstance(output.content[0], TextContent):
99
+ raise ValueError(f"MCPAgent expected TextContent, got: {type(output.content[0])}")
100
+ answer = output.content[0].text
101
+ if not isinstance(answer, str) or answer.strip() == "":
102
+ raise ValueError(f"MCPAgent answer not found in output content. Got: {output.content}")
103
+
104
+ try:
105
+ assert isinstance(output.content[1], TextContent)
106
+ metrics = json.loads(output.content[1].text)
107
+ assert isinstance(metrics, dict)
108
+ except Exception:
109
+ metrics = None
110
+
111
+ return AgentResult(answer=answer, metrics=metrics)
112
+
113
+ async def run(
114
+ self, prompt: str, attachments: "list[Any] | None" = None, *, task_id: str | None = None
115
+ ) -> AgentResult:
116
+ if attachments:
117
+ # TODO: pass image attachments as ImageContent when MCP SDK supports it in tool calls
118
+ return AgentResult(outcome="error", reason="unsupported_attachment")
119
+ return await self._run_with_retry(prompt)