magma-bench 2.0.0b1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- magma_bench-2.0.0b1/LICENSE +24 -0
- magma_bench-2.0.0b1/PKG-INFO +108 -0
- magma_bench-2.0.0b1/README.md +84 -0
- magma_bench-2.0.0b1/pyproject.toml +44 -0
- magma_bench-2.0.0b1/setup.cfg +4 -0
- magma_bench-2.0.0b1/src/magma_bench/__init__.py +0 -0
- magma_bench-2.0.0b1/src/magma_bench/__main__.py +4 -0
- magma_bench-2.0.0b1/src/magma_bench/agents/__init__.py +3 -0
- magma_bench-2.0.0b1/src/magma_bench/agents/base/__init__.py +5 -0
- magma_bench-2.0.0b1/src/magma_bench/agents/base/benchmark_agent.py +202 -0
- magma_bench-2.0.0b1/src/magma_bench/artifacts/__init__.py +79 -0
- magma_bench-2.0.0b1/src/magma_bench/artifacts/models.py +424 -0
- magma_bench-2.0.0b1/src/magma_bench/artifacts/runtime.py +330 -0
- magma_bench-2.0.0b1/src/magma_bench/command_line.py +46 -0
- magma_bench-2.0.0b1/src/magma_bench/data_structures/__init__.py +16 -0
- magma_bench-2.0.0b1/src/magma_bench/data_structures/benchmark.py +47 -0
- magma_bench-2.0.0b1/src/magma_bench/data_structures/running_structure.py +103 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/__init__.py +2 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/_log_rule_compiler.py +47 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/_log_rule_registry.py +215 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/_predicate_compiler.py +91 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/_predicate_registry.py +321 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/log_rules.py +32 -0
- magma_bench-2.0.0b1/src/magma_bench/evalutations/predicate.py +63 -0
- magma_bench-2.0.0b1/src/magma_bench/executor/__init__.py +9 -0
- magma_bench-2.0.0b1/src/magma_bench/executor/context.py +80 -0
- magma_bench-2.0.0b1/src/magma_bench/executor/episode_runtime.py +24 -0
- magma_bench-2.0.0b1/src/magma_bench/executor/eval_executor.py +998 -0
- magma_bench-2.0.0b1/src/magma_bench/launch.py +205 -0
- magma_bench-2.0.0b1/src/magma_bench/loader/__init__.py +4 -0
- magma_bench-2.0.0b1/src/magma_bench/loader/group.py +66 -0
- magma_bench-2.0.0b1/src/magma_bench/loader/loader.py +314 -0
- magma_bench-2.0.0b1/src/magma_bench/results/__init__.py +47 -0
- magma_bench-2.0.0b1/src/magma_bench/results/manager.py +641 -0
- magma_bench-2.0.0b1/src/magma_bench/results/metrics.py +201 -0
- magma_bench-2.0.0b1/src/magma_bench/results/models.py +175 -0
- magma_bench-2.0.0b1/src/magma_bench/runner/__init__.py +3 -0
- magma_bench-2.0.0b1/src/magma_bench/runner/group_runner.py +466 -0
- magma_bench-2.0.0b1/src/magma_bench/runner/runner.py +364 -0
- magma_bench-2.0.0b1/src/magma_bench/video.py +523 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/PKG-INFO +108 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/SOURCES.txt +49 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/dependency_links.txt +1 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/entry_points.txt +2 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/requires.txt +13 -0
- magma_bench-2.0.0b1/src/magma_bench.egg-info/top_level.txt +1 -0
- magma_bench-2.0.0b1/tests/test_agent.py +275 -0
- magma_bench-2.0.0b1/tests/test_details.py +22 -0
- magma_bench-2.0.0b1/tests/test_results_metrics.py +629 -0
- magma_bench-2.0.0b1/tests/test_runner_answer_tools.py +832 -0
- magma_bench-2.0.0b1/tests/test_runtime_error_validation.py +14 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
BSD 2-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, MAGMA-s
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
16
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
17
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
18
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
19
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
20
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
21
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
22
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
23
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
24
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: magma_bench
|
|
3
|
+
Version: 2.0.0b1
|
|
4
|
+
Summary: Evaluate agents on interactive robotic tasks with MAGMA
|
|
5
|
+
Project-URL: Documentation, https://magma-rob.github.io/docs/intro
|
|
6
|
+
Project-URL: Repository, https://github.com/MAGMA-rob/magma-bench
|
|
7
|
+
Project-URL: Issues, https://github.com/MAGMA-rob/magma-bench/issues
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Requires-Python: <3.13,>=3.12
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: magma_core[simulation]<3.0.0,>=2.0.0b1
|
|
13
|
+
Requires-Dist: magma_scenarios<3.0.0,>=2.0.0
|
|
14
|
+
Requires-Dist: pydantic<3,>=2.4
|
|
15
|
+
Requires-Dist: requests<3,>=2.34.2
|
|
16
|
+
Requires-Dist: torch==2.13.0
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest; extra == "dev"
|
|
19
|
+
Provides-Extra: video
|
|
20
|
+
Requires-Dist: imageio-ffmpeg>=0.5; extra == "video"
|
|
21
|
+
Requires-Dist: numpy>=1.23; extra == "video"
|
|
22
|
+
Requires-Dist: Pillow>=10; extra == "video"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# MAGMA-BENCH
|
|
26
|
+
|
|
27
|
+
Evaluate agents on interactive robotic tasks with MAGMA. MAGMA-BENCH loads
|
|
28
|
+
compiled benchmark episodes, runs them in simulation, communicates with an agent
|
|
29
|
+
through the MAGMA HTTP protocol, and saves detailed results and metrics.
|
|
30
|
+
|
|
31
|
+
**Version 2.0.0b1 is the first beta of v2.** APIs, command-line options, and
|
|
32
|
+
result formats may change before the stable release. End-to-end validation is
|
|
33
|
+
still in progress.
|
|
34
|
+
|
|
35
|
+
[Official documentation](https://magma-rob.github.io/docs/intro) ·
|
|
36
|
+
[Report an issue](https://github.com/MAGMA-rob/magma-bench/issues)
|
|
37
|
+
|
|
38
|
+
## Installation
|
|
39
|
+
|
|
40
|
+
Python 3.12 is required. Simulation installation has been checked on Linux
|
|
41
|
+
x86_64. GPU simulation and rendering require compatible system drivers.
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
python -m pip install "magma_bench==2.0.0b1"
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
This installs `magma_core[simulation]>=2.0.0b1,<3.0.0` and
|
|
48
|
+
`magma_scenarios>=2.0.0,<3.0.0` with their Python dependencies. The benchmark
|
|
49
|
+
files and the agent server are separate inputs and are not bundled with this
|
|
50
|
+
package.
|
|
51
|
+
|
|
52
|
+
The version is pinned explicitly because this is a prerelease. To request the
|
|
53
|
+
latest version, including prereleases, use:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
python -m pip install --upgrade --pre magma_bench
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Usage
|
|
60
|
+
|
|
61
|
+
Start a compatible MAGMA agent server, then run a compiled benchmark directory:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
magma-bench run \
|
|
65
|
+
--benchmark-root /path/to/compiled-benchmark \
|
|
66
|
+
--agent-address http://127.0.0.1:8888 \
|
|
67
|
+
--run-name experiment-1
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
The agent must expose `/health`, `/v1/info`, and `/v1/responses` using protocol
|
|
71
|
+
version 2.0. Configuration can be supplied with `--config-path`; command-line
|
|
72
|
+
options override its benchmark settings.
|
|
73
|
+
|
|
74
|
+
Useful options include:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
magma-bench run --help
|
|
78
|
+
magma-bench run --benchmark-root /path/to/benchmark --scenarios coffee_comp
|
|
79
|
+
magma-bench run --benchmark-root /path/to/benchmark --skip-judge
|
|
80
|
+
magma-bench run --benchmark-root /path/to/benchmark --model-logs --videos
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Video output uses the optional dependencies:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
python -m pip install "magma_bench[video]==2.0.0b1"
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Version 2 uses result schema `2.0`. Resume a compatible run with
|
|
90
|
+
`--results-path`; older result schemas and agent adapters are unsupported.
|
|
91
|
+
|
|
92
|
+
See the [official documentation](https://magma-rob.github.io/docs/intro) for
|
|
93
|
+
benchmark preparation, configuration, agent setup, metrics, and result formats.
|
|
94
|
+
|
|
95
|
+
## Install from source
|
|
96
|
+
|
|
97
|
+
Install the tagged release from GitHub:
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
python -m pip install "magma_bench @ git+https://github.com/MAGMA-rob/magma-bench.git@v2.0.0b1"
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Dependencies are resolved from PyPI. For local development, clone the repository
|
|
104
|
+
and run `python -m pip install -e ".[dev,video]"`.
|
|
105
|
+
|
|
106
|
+
## License
|
|
107
|
+
|
|
108
|
+
[BSD 2-Clause](https://github.com/MAGMA-rob/magma-bench/blob/main/LICENSE).
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# MAGMA-BENCH
|
|
2
|
+
|
|
3
|
+
Evaluate agents on interactive robotic tasks with MAGMA. MAGMA-BENCH loads
|
|
4
|
+
compiled benchmark episodes, runs them in simulation, communicates with an agent
|
|
5
|
+
through the MAGMA HTTP protocol, and saves detailed results and metrics.
|
|
6
|
+
|
|
7
|
+
**Version 2.0.0b1 is the first beta of v2.** APIs, command-line options, and
|
|
8
|
+
result formats may change before the stable release. End-to-end validation is
|
|
9
|
+
still in progress.
|
|
10
|
+
|
|
11
|
+
[Official documentation](https://magma-rob.github.io/docs/intro) ·
|
|
12
|
+
[Report an issue](https://github.com/MAGMA-rob/magma-bench/issues)
|
|
13
|
+
|
|
14
|
+
## Installation
|
|
15
|
+
|
|
16
|
+
Python 3.12 is required. Simulation installation has been checked on Linux
|
|
17
|
+
x86_64. GPU simulation and rendering require compatible system drivers.
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
python -m pip install "magma_bench==2.0.0b1"
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
This installs `magma_core[simulation]>=2.0.0b1,<3.0.0` and
|
|
24
|
+
`magma_scenarios>=2.0.0,<3.0.0` with their Python dependencies. The benchmark
|
|
25
|
+
files and the agent server are separate inputs and are not bundled with this
|
|
26
|
+
package.
|
|
27
|
+
|
|
28
|
+
The version is pinned explicitly because this is a prerelease. To request the
|
|
29
|
+
latest version, including prereleases, use:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
python -m pip install --upgrade --pre magma_bench
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Usage
|
|
36
|
+
|
|
37
|
+
Start a compatible MAGMA agent server, then run a compiled benchmark directory:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
magma-bench run \
|
|
41
|
+
--benchmark-root /path/to/compiled-benchmark \
|
|
42
|
+
--agent-address http://127.0.0.1:8888 \
|
|
43
|
+
--run-name experiment-1
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The agent must expose `/health`, `/v1/info`, and `/v1/responses` using protocol
|
|
47
|
+
version 2.0. Configuration can be supplied with `--config-path`; command-line
|
|
48
|
+
options override its benchmark settings.
|
|
49
|
+
|
|
50
|
+
Useful options include:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
magma-bench run --help
|
|
54
|
+
magma-bench run --benchmark-root /path/to/benchmark --scenarios coffee_comp
|
|
55
|
+
magma-bench run --benchmark-root /path/to/benchmark --skip-judge
|
|
56
|
+
magma-bench run --benchmark-root /path/to/benchmark --model-logs --videos
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Video output uses the optional dependencies:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
python -m pip install "magma_bench[video]==2.0.0b1"
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Version 2 uses result schema `2.0`. Resume a compatible run with
|
|
66
|
+
`--results-path`; older result schemas and agent adapters are unsupported.
|
|
67
|
+
|
|
68
|
+
See the [official documentation](https://magma-rob.github.io/docs/intro) for
|
|
69
|
+
benchmark preparation, configuration, agent setup, metrics, and result formats.
|
|
70
|
+
|
|
71
|
+
## Install from source
|
|
72
|
+
|
|
73
|
+
Install the tagged release from GitHub:
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
python -m pip install "magma_bench @ git+https://github.com/MAGMA-rob/magma-bench.git@v2.0.0b1"
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Dependencies are resolved from PyPI. For local development, clone the repository
|
|
80
|
+
and run `python -m pip install -e ".[dev,video]"`.
|
|
81
|
+
|
|
82
|
+
## License
|
|
83
|
+
|
|
84
|
+
[BSD 2-Clause](https://github.com/MAGMA-rob/magma-bench/blob/main/LICENSE).
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# SPDX-License-Identifier: BSD-2-Clause
|
|
2
|
+
# Copyright (c) 2026, Loan Bernat
|
|
3
|
+
|
|
4
|
+
[project]
|
|
5
|
+
name = "magma_bench"
|
|
6
|
+
version = "2.0.0b1"
|
|
7
|
+
description = "Evaluate agents on interactive robotic tasks with MAGMA"
|
|
8
|
+
readme = "README.md"
|
|
9
|
+
requires-python = ">=3.12,<3.13"
|
|
10
|
+
classifiers = ["Development Status :: 4 - Beta"]
|
|
11
|
+
dependencies = [
|
|
12
|
+
"magma_core[simulation]>=2.0.0b1,<3.0.0",
|
|
13
|
+
"magma_scenarios>=2.0.0,<3.0.0",
|
|
14
|
+
"pydantic>=2.4,<3",
|
|
15
|
+
"requests>=2.34.2,<3",
|
|
16
|
+
"torch==2.13.0",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
[project.urls]
|
|
20
|
+
Documentation = "https://magma-rob.github.io/docs/intro"
|
|
21
|
+
Repository = "https://github.com/MAGMA-rob/magma-bench"
|
|
22
|
+
Issues = "https://github.com/MAGMA-rob/magma-bench/issues"
|
|
23
|
+
|
|
24
|
+
[project.optional-dependencies]
|
|
25
|
+
dev = ["pytest"]
|
|
26
|
+
video = [
|
|
27
|
+
"imageio-ffmpeg>=0.5",
|
|
28
|
+
"numpy>=1.23",
|
|
29
|
+
"Pillow>=10",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[build-system]
|
|
33
|
+
requires = ["setuptools>=61.0"]
|
|
34
|
+
build-backend = "setuptools.build_meta"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[project.scripts]
|
|
40
|
+
magma-bench = "magma_bench.command_line:main"
|
|
41
|
+
|
|
42
|
+
[tool.pytest.ini_options]
|
|
43
|
+
addopts = "-p no:launch_testing -p no:launch_ros"
|
|
44
|
+
testpaths = ["tests"]
|
|
File without changes
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from copy import deepcopy
|
|
4
|
+
import json
|
|
5
|
+
import threading
|
|
6
|
+
import time
|
|
7
|
+
from typing import Any, Dict, List, Optional
|
|
8
|
+
from uuid import uuid4
|
|
9
|
+
|
|
10
|
+
import requests
|
|
11
|
+
|
|
12
|
+
from magma_core.domain.agent_call import Call
|
|
13
|
+
from magma_core.protocol.agent import (
|
|
14
|
+
PROTOCOL_VERSION, AgentHealth, AgentInfo, AgentInput, AgentInstruction,
|
|
15
|
+
AgentOutput, AgentRequest, AgentResponse, JsonObject,
|
|
16
|
+
)
|
|
17
|
+
from magma_core.simulation.agents import AgentAnswer, BadAgentAnswer, ValidAgentAnswer
|
|
18
|
+
from magma_core.simulation.data_structures import EmptyInstruction
|
|
19
|
+
from magma_bench.data_structures import BenchmarkAgentResult, EpisodeSituation
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class BenchmarkAgent:
|
|
23
|
+
"""Batched HTTP client for a single protocol-v2 agent runtime."""
|
|
24
|
+
|
|
25
|
+
def __init__(
|
|
26
|
+
self,
|
|
27
|
+
agent_url: str,
|
|
28
|
+
agent_name: str | None = None,
|
|
29
|
+
extra_keys: JsonObject | None = None,
|
|
30
|
+
timeout: float = 360,
|
|
31
|
+
collect_model_logs: bool = False,
|
|
32
|
+
) -> None:
|
|
33
|
+
self.agent_url = agent_url.rstrip("/")
|
|
34
|
+
self.timeout = timeout
|
|
35
|
+
self.extra_keys = deepcopy(extra_keys) if extra_keys is not None else {}
|
|
36
|
+
self.collect_model_logs = collect_model_logs
|
|
37
|
+
health = requests.get(f"{self.agent_url}/health", timeout=timeout)
|
|
38
|
+
health.raise_for_status()
|
|
39
|
+
AgentHealth.model_validate(health.json())
|
|
40
|
+
info = requests.get(f"{self.agent_url}/v1/info", timeout=timeout)
|
|
41
|
+
info.raise_for_status()
|
|
42
|
+
self.info = AgentInfo.model_validate(info.json())
|
|
43
|
+
if self.info.protocol_version != PROTOCOL_VERSION:
|
|
44
|
+
raise ValueError(f"Unsupported agent protocol {self.info.protocol_version!r}")
|
|
45
|
+
if not self.info.capabilities.get("inference", False):
|
|
46
|
+
raise ValueError("Agent runtime does not support inference")
|
|
47
|
+
self.agent_name = agent_name or self.info.agent_id
|
|
48
|
+
self.lock = threading.Lock()
|
|
49
|
+
self.completed_results: List[BenchmarkAgentResult] = []
|
|
50
|
+
self.waiting_inputs: Dict[int, EpisodeSituation] = {}
|
|
51
|
+
self.thread_exception: Optional[Exception] = None
|
|
52
|
+
self.thread_running = True
|
|
53
|
+
self.thread = threading.Thread(
|
|
54
|
+
target=self._periodic_computing_of_answers, args=(4,), daemon=True,
|
|
55
|
+
)
|
|
56
|
+
self.thread.start()
|
|
57
|
+
|
|
58
|
+
def compute_agent_results(
|
|
59
|
+
self, batch_inputs: Dict[int, EpisodeSituation],
|
|
60
|
+
) -> List[BenchmarkAgentResult]:
|
|
61
|
+
if not batch_inputs:
|
|
62
|
+
return []
|
|
63
|
+
inputs: list[AgentInput] = []
|
|
64
|
+
for env_idx, situation in batch_inputs.items():
|
|
65
|
+
instruction = situation.current_instruction
|
|
66
|
+
if isinstance(instruction, EmptyInstruction):
|
|
67
|
+
raise ValueError("EmptyInstruction requires the suspended-skill resume path")
|
|
68
|
+
role = instruction.get_role()
|
|
69
|
+
if role not in {"USER", "SYSTEM"}:
|
|
70
|
+
raise ValueError(f"Unsupported input role {role!r}")
|
|
71
|
+
inputs.append(AgentInput(
|
|
72
|
+
id=env_idx,
|
|
73
|
+
instruction=AgentInstruction(
|
|
74
|
+
type="user" if role == "USER" else "env",
|
|
75
|
+
content=instruction.get_content(),
|
|
76
|
+
),
|
|
77
|
+
tools=deepcopy(situation.tools),
|
|
78
|
+
attributes=deepcopy(situation.attributes),
|
|
79
|
+
memory=deepcopy(situation.memory),
|
|
80
|
+
num_outputs=1,
|
|
81
|
+
extra_keys=deepcopy(self.extra_keys),
|
|
82
|
+
))
|
|
83
|
+
request = AgentRequest(request_id=uuid4().hex, inputs=inputs)
|
|
84
|
+
response = self.send_to_agent(request)
|
|
85
|
+
results: list[BenchmarkAgentResult] = []
|
|
86
|
+
for output in response.root:
|
|
87
|
+
updated = batch_inputs[output.source_id].snapshot()
|
|
88
|
+
updated.memory = deepcopy(output.memory)
|
|
89
|
+
diagnostics: list[dict[str, Any]] = []
|
|
90
|
+
if self.collect_model_logs:
|
|
91
|
+
diagnostics.append({
|
|
92
|
+
"internal_steps": deepcopy(output.internal_steps),
|
|
93
|
+
"error": (
|
|
94
|
+
output.error.model_dump(mode="json")
|
|
95
|
+
if output.error is not None
|
|
96
|
+
else None
|
|
97
|
+
),
|
|
98
|
+
})
|
|
99
|
+
results.append(BenchmarkAgentResult(
|
|
100
|
+
answer=self.normalize_model_response(output),
|
|
101
|
+
situation=updated,
|
|
102
|
+
model_diagnostics=diagnostics,
|
|
103
|
+
))
|
|
104
|
+
return results
|
|
105
|
+
|
|
106
|
+
def send_to_agent(self, request: AgentRequest) -> AgentResponse:
|
|
107
|
+
response = requests.post(
|
|
108
|
+
f"{self.agent_url}/v1/responses",
|
|
109
|
+
json=request.model_dump(mode="json"), timeout=self.timeout,
|
|
110
|
+
)
|
|
111
|
+
response.raise_for_status()
|
|
112
|
+
parsed = AgentResponse.model_validate(response.json())
|
|
113
|
+
parsed.validate_request(request)
|
|
114
|
+
return parsed
|
|
115
|
+
|
|
116
|
+
def normalize_model_response(
|
|
117
|
+
self, response: AgentOutput, agent_step_id: int = 0,
|
|
118
|
+
) -> AgentAnswer:
|
|
119
|
+
if response.status == "error":
|
|
120
|
+
assert response.error is not None
|
|
121
|
+
return BadAgentAnswer(
|
|
122
|
+
source_node_id=response.source_id, agent_step_id=agent_step_id, say="",
|
|
123
|
+
raw_action=json.dumps(response.model_dump(mode="json"), ensure_ascii=False),
|
|
124
|
+
reason=response.error.message,
|
|
125
|
+
)
|
|
126
|
+
assert response.output is not None
|
|
127
|
+
return ValidAgentAnswer(
|
|
128
|
+
source_node_id=response.source_id, agent_step_id=agent_step_id,
|
|
129
|
+
say=response.output.say,
|
|
130
|
+
calls=[
|
|
131
|
+
Call(
|
|
132
|
+
name=call.name,
|
|
133
|
+
arguments=deepcopy(call.arguments),
|
|
134
|
+
target_robot_name=call.target_robot_name,
|
|
135
|
+
)
|
|
136
|
+
for call in response.output.tool_calls
|
|
137
|
+
],
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
def get_agent_card(self) -> Dict[str, Any]:
|
|
141
|
+
return {
|
|
142
|
+
"agent": self.agent_name,
|
|
143
|
+
"agent_id": self.info.agent_id,
|
|
144
|
+
"agent_version": self.info.agent_version,
|
|
145
|
+
"protocol_version": self.info.protocol_version,
|
|
146
|
+
"extra_keys": deepcopy(self.extra_keys),
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
def get_pending_results(self) -> List[BenchmarkAgentResult]:
|
|
150
|
+
with self.lock:
|
|
151
|
+
if self.thread_exception is not None:
|
|
152
|
+
raise RuntimeError(
|
|
153
|
+
"The benchmark-agent background thread failed"
|
|
154
|
+
) from self.thread_exception
|
|
155
|
+
out = self.completed_results.copy()
|
|
156
|
+
self.completed_results.clear()
|
|
157
|
+
return out
|
|
158
|
+
|
|
159
|
+
def add_inputs(self, inputs: Dict[int, EpisodeSituation]) -> None:
|
|
160
|
+
with self.lock:
|
|
161
|
+
if self.thread_exception is not None:
|
|
162
|
+
raise RuntimeError(
|
|
163
|
+
"Cannot add inputs after the benchmark-agent thread failed"
|
|
164
|
+
) from self.thread_exception
|
|
165
|
+
if not self.thread_running:
|
|
166
|
+
raise RuntimeError("Cannot add inputs to a stopped benchmark agent")
|
|
167
|
+
for idx, situation in inputs.items():
|
|
168
|
+
if idx in self.waiting_inputs:
|
|
169
|
+
raise RuntimeError(
|
|
170
|
+
f"Environment {idx} already waits for an agent answer"
|
|
171
|
+
)
|
|
172
|
+
self.waiting_inputs[idx] = situation.snapshot()
|
|
173
|
+
|
|
174
|
+
def stop(self) -> None:
|
|
175
|
+
with self.lock:
|
|
176
|
+
self.thread_running = False
|
|
177
|
+
self.thread.join(timeout=5)
|
|
178
|
+
|
|
179
|
+
def _periodic_computing_of_answers(self, interval: float = 4.0) -> None:
|
|
180
|
+
try:
|
|
181
|
+
while self.thread_running:
|
|
182
|
+
start_time = time.time()
|
|
183
|
+
to_do: Dict[int, EpisodeSituation] = {}
|
|
184
|
+
with self.lock:
|
|
185
|
+
if self.waiting_inputs:
|
|
186
|
+
to_do = self.waiting_inputs.copy()
|
|
187
|
+
self.waiting_inputs.clear()
|
|
188
|
+
|
|
189
|
+
if to_do:
|
|
190
|
+
results = self.compute_agent_results(to_do)
|
|
191
|
+
with self.lock:
|
|
192
|
+
self.completed_results.extend(results)
|
|
193
|
+
|
|
194
|
+
elapsed = time.time() - start_time
|
|
195
|
+
sleep_time = max(0, interval - elapsed)
|
|
196
|
+
time.sleep(sleep_time)
|
|
197
|
+
|
|
198
|
+
except Exception as error:
|
|
199
|
+
with self.lock:
|
|
200
|
+
self.thread_exception = error
|
|
201
|
+
with self.lock:
|
|
202
|
+
self.thread_running = False
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
from .models import (
|
|
2
|
+
CONDITIONS,
|
|
3
|
+
Condition,
|
|
4
|
+
ActiveErrorSpec,
|
|
5
|
+
BenchmarkManifest,
|
|
6
|
+
CompiledEpisodeSpec,
|
|
7
|
+
DeclarativeStageSpec,
|
|
8
|
+
EpisodeIndexEntry,
|
|
9
|
+
EpisodeMetadata,
|
|
10
|
+
EpisodeSpec,
|
|
11
|
+
InstructionSpec,
|
|
12
|
+
InterventionSpec,
|
|
13
|
+
LogRuleSpec,
|
|
14
|
+
ObjectSpec,
|
|
15
|
+
ScenarioManifest,
|
|
16
|
+
SemanticManifest,
|
|
17
|
+
SerializedStageSpec,
|
|
18
|
+
SkeletonManifest,
|
|
19
|
+
Track,
|
|
20
|
+
VariantCondition,
|
|
21
|
+
StageInputSpec,
|
|
22
|
+
StagePresentationSpec,
|
|
23
|
+
StageSpec,
|
|
24
|
+
load_json_model,
|
|
25
|
+
)
|
|
26
|
+
from .runtime import (
|
|
27
|
+
DeclarativeStage,
|
|
28
|
+
apply_stage_presentation,
|
|
29
|
+
deserialize_error,
|
|
30
|
+
deserialize_goal,
|
|
31
|
+
deserialize_runtime_randomizer,
|
|
32
|
+
deserialize_serialized_stage,
|
|
33
|
+
deserialize_stage,
|
|
34
|
+
deserialize_task_stages,
|
|
35
|
+
serialize_error,
|
|
36
|
+
serialize_goal,
|
|
37
|
+
serialize_stage,
|
|
38
|
+
serialize_task_stages,
|
|
39
|
+
stage_presentation,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"CONDITIONS",
|
|
44
|
+
"Condition",
|
|
45
|
+
"ActiveErrorSpec",
|
|
46
|
+
"BenchmarkManifest",
|
|
47
|
+
"CompiledEpisodeSpec",
|
|
48
|
+
"DeclarativeStage",
|
|
49
|
+
"DeclarativeStageSpec",
|
|
50
|
+
"EpisodeIndexEntry",
|
|
51
|
+
"EpisodeMetadata",
|
|
52
|
+
"EpisodeSpec",
|
|
53
|
+
"InstructionSpec",
|
|
54
|
+
"InterventionSpec",
|
|
55
|
+
"LogRuleSpec",
|
|
56
|
+
"ObjectSpec",
|
|
57
|
+
"ScenarioManifest",
|
|
58
|
+
"SemanticManifest",
|
|
59
|
+
"SerializedStageSpec",
|
|
60
|
+
"SkeletonManifest",
|
|
61
|
+
"Track",
|
|
62
|
+
"VariantCondition",
|
|
63
|
+
"StageInputSpec",
|
|
64
|
+
"StagePresentationSpec",
|
|
65
|
+
"StageSpec",
|
|
66
|
+
"apply_stage_presentation",
|
|
67
|
+
"deserialize_error",
|
|
68
|
+
"deserialize_goal",
|
|
69
|
+
"deserialize_runtime_randomizer",
|
|
70
|
+
"deserialize_serialized_stage",
|
|
71
|
+
"deserialize_stage",
|
|
72
|
+
"deserialize_task_stages",
|
|
73
|
+
"load_json_model",
|
|
74
|
+
"serialize_error",
|
|
75
|
+
"serialize_goal",
|
|
76
|
+
"serialize_stage",
|
|
77
|
+
"serialize_task_stages",
|
|
78
|
+
"stage_presentation",
|
|
79
|
+
]
|