magma-bench 2.0.0b1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. magma_bench-2.0.0b1/LICENSE +24 -0
  2. magma_bench-2.0.0b1/PKG-INFO +108 -0
  3. magma_bench-2.0.0b1/README.md +84 -0
  4. magma_bench-2.0.0b1/pyproject.toml +44 -0
  5. magma_bench-2.0.0b1/setup.cfg +4 -0
  6. magma_bench-2.0.0b1/src/magma_bench/__init__.py +0 -0
  7. magma_bench-2.0.0b1/src/magma_bench/__main__.py +4 -0
  8. magma_bench-2.0.0b1/src/magma_bench/agents/__init__.py +3 -0
  9. magma_bench-2.0.0b1/src/magma_bench/agents/base/__init__.py +5 -0
  10. magma_bench-2.0.0b1/src/magma_bench/agents/base/benchmark_agent.py +202 -0
  11. magma_bench-2.0.0b1/src/magma_bench/artifacts/__init__.py +79 -0
  12. magma_bench-2.0.0b1/src/magma_bench/artifacts/models.py +424 -0
  13. magma_bench-2.0.0b1/src/magma_bench/artifacts/runtime.py +330 -0
  14. magma_bench-2.0.0b1/src/magma_bench/command_line.py +46 -0
  15. magma_bench-2.0.0b1/src/magma_bench/data_structures/__init__.py +16 -0
  16. magma_bench-2.0.0b1/src/magma_bench/data_structures/benchmark.py +47 -0
  17. magma_bench-2.0.0b1/src/magma_bench/data_structures/running_structure.py +103 -0
  18. magma_bench-2.0.0b1/src/magma_bench/evalutations/__init__.py +2 -0
  19. magma_bench-2.0.0b1/src/magma_bench/evalutations/_log_rule_compiler.py +47 -0
  20. magma_bench-2.0.0b1/src/magma_bench/evalutations/_log_rule_registry.py +215 -0
  21. magma_bench-2.0.0b1/src/magma_bench/evalutations/_predicate_compiler.py +91 -0
  22. magma_bench-2.0.0b1/src/magma_bench/evalutations/_predicate_registry.py +321 -0
  23. magma_bench-2.0.0b1/src/magma_bench/evalutations/log_rules.py +32 -0
  24. magma_bench-2.0.0b1/src/magma_bench/evalutations/predicate.py +63 -0
  25. magma_bench-2.0.0b1/src/magma_bench/executor/__init__.py +9 -0
  26. magma_bench-2.0.0b1/src/magma_bench/executor/context.py +80 -0
  27. magma_bench-2.0.0b1/src/magma_bench/executor/episode_runtime.py +24 -0
  28. magma_bench-2.0.0b1/src/magma_bench/executor/eval_executor.py +998 -0
  29. magma_bench-2.0.0b1/src/magma_bench/launch.py +205 -0
  30. magma_bench-2.0.0b1/src/magma_bench/loader/__init__.py +4 -0
  31. magma_bench-2.0.0b1/src/magma_bench/loader/group.py +66 -0
  32. magma_bench-2.0.0b1/src/magma_bench/loader/loader.py +314 -0
  33. magma_bench-2.0.0b1/src/magma_bench/results/__init__.py +47 -0
  34. magma_bench-2.0.0b1/src/magma_bench/results/manager.py +641 -0
  35. magma_bench-2.0.0b1/src/magma_bench/results/metrics.py +201 -0
  36. magma_bench-2.0.0b1/src/magma_bench/results/models.py +175 -0
  37. magma_bench-2.0.0b1/src/magma_bench/runner/__init__.py +3 -0
  38. magma_bench-2.0.0b1/src/magma_bench/runner/group_runner.py +466 -0
  39. magma_bench-2.0.0b1/src/magma_bench/runner/runner.py +364 -0
  40. magma_bench-2.0.0b1/src/magma_bench/video.py +523 -0
  41. magma_bench-2.0.0b1/src/magma_bench.egg-info/PKG-INFO +108 -0
  42. magma_bench-2.0.0b1/src/magma_bench.egg-info/SOURCES.txt +49 -0
  43. magma_bench-2.0.0b1/src/magma_bench.egg-info/dependency_links.txt +1 -0
  44. magma_bench-2.0.0b1/src/magma_bench.egg-info/entry_points.txt +2 -0
  45. magma_bench-2.0.0b1/src/magma_bench.egg-info/requires.txt +13 -0
  46. magma_bench-2.0.0b1/src/magma_bench.egg-info/top_level.txt +1 -0
  47. magma_bench-2.0.0b1/tests/test_agent.py +275 -0
  48. magma_bench-2.0.0b1/tests/test_details.py +22 -0
  49. magma_bench-2.0.0b1/tests/test_results_metrics.py +629 -0
  50. magma_bench-2.0.0b1/tests/test_runner_answer_tools.py +832 -0
  51. magma_bench-2.0.0b1/tests/test_runtime_error_validation.py +14 -0
@@ -0,0 +1,24 @@
1
+ BSD 2-Clause License
2
+
3
+ Copyright (c) 2026, MAGMA-s
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ 1. Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ 2. Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
16
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
17
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
18
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
19
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
20
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
21
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
22
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
23
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
24
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,108 @@
1
+ Metadata-Version: 2.4
2
+ Name: magma_bench
3
+ Version: 2.0.0b1
4
+ Summary: Evaluate agents on interactive robotic tasks with MAGMA
5
+ Project-URL: Documentation, https://magma-rob.github.io/docs/intro
6
+ Project-URL: Repository, https://github.com/MAGMA-rob/magma-bench
7
+ Project-URL: Issues, https://github.com/MAGMA-rob/magma-bench/issues
8
+ Classifier: Development Status :: 4 - Beta
9
+ Requires-Python: <3.13,>=3.12
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Requires-Dist: magma_core[simulation]<3.0.0,>=2.0.0b1
13
+ Requires-Dist: magma_scenarios<3.0.0,>=2.0.0
14
+ Requires-Dist: pydantic<3,>=2.4
15
+ Requires-Dist: requests<3,>=2.34.2
16
+ Requires-Dist: torch==2.13.0
17
+ Provides-Extra: dev
18
+ Requires-Dist: pytest; extra == "dev"
19
+ Provides-Extra: video
20
+ Requires-Dist: imageio-ffmpeg>=0.5; extra == "video"
21
+ Requires-Dist: numpy>=1.23; extra == "video"
22
+ Requires-Dist: Pillow>=10; extra == "video"
23
+ Dynamic: license-file
24
+
25
+ # MAGMA-BENCH
26
+
27
+ Evaluate agents on interactive robotic tasks with MAGMA. MAGMA-BENCH loads
28
+ compiled benchmark episodes, runs them in simulation, communicates with an agent
29
+ through the MAGMA HTTP protocol, and saves detailed results and metrics.
30
+
31
+ **Version 2.0.0b1 is the first beta of v2.** APIs, command-line options, and
32
+ result formats may change before the stable release. End-to-end validation is
33
+ still in progress.
34
+
35
+ [Official documentation](https://magma-rob.github.io/docs/intro) ·
36
+ [Report an issue](https://github.com/MAGMA-rob/magma-bench/issues)
37
+
38
+ ## Installation
39
+
40
+ Python 3.12 is required. Simulation installation has been checked on Linux
41
+ x86_64. GPU simulation and rendering require compatible system drivers.
42
+
43
+ ```bash
44
+ python -m pip install "magma_bench==2.0.0b1"
45
+ ```
46
+
47
+ This installs `magma_core[simulation]>=2.0.0b1,<3.0.0` and
48
+ `magma_scenarios>=2.0.0,<3.0.0` with their Python dependencies. The benchmark
49
+ files and the agent server are separate inputs and are not bundled with this
50
+ package.
51
+
52
+ The version is pinned explicitly because this is a prerelease. To request the
53
+ latest version, including prereleases, use:
54
+
55
+ ```bash
56
+ python -m pip install --upgrade --pre magma_bench
57
+ ```
58
+
59
+ ## Usage
60
+
61
+ Start a compatible MAGMA agent server, then run a compiled benchmark directory:
62
+
63
+ ```bash
64
+ magma-bench run \
65
+ --benchmark-root /path/to/compiled-benchmark \
66
+ --agent-address http://127.0.0.1:8888 \
67
+ --run-name experiment-1
68
+ ```
69
+
70
+ The agent must expose `/health`, `/v1/info`, and `/v1/responses` using protocol
71
+ version 2.0. Configuration can be supplied with `--config-path`; command-line
72
+ options override its benchmark settings.
73
+
74
+ Useful options include:
75
+
76
+ ```bash
77
+ magma-bench run --help
78
+ magma-bench run --benchmark-root /path/to/benchmark --scenarios coffee_comp
79
+ magma-bench run --benchmark-root /path/to/benchmark --skip-judge
80
+ magma-bench run --benchmark-root /path/to/benchmark --model-logs --videos
81
+ ```
82
+
83
+ Video output uses the optional dependencies:
84
+
85
+ ```bash
86
+ python -m pip install "magma_bench[video]==2.0.0b1"
87
+ ```
88
+
89
+ Version 2 uses result schema `2.0`. Resume a compatible run with
90
+ `--results-path`; older result schemas and agent adapters are unsupported.
91
+
92
+ See the [official documentation](https://magma-rob.github.io/docs/intro) for
93
+ benchmark preparation, configuration, agent setup, metrics, and result formats.
94
+
95
+ ## Install from source
96
+
97
+ Install the tagged release from GitHub:
98
+
99
+ ```bash
100
+ python -m pip install "magma_bench @ git+https://github.com/MAGMA-rob/magma-bench.git@v2.0.0b1"
101
+ ```
102
+
103
+ Dependencies are resolved from PyPI. For local development, clone the repository
104
+ and run `python -m pip install -e ".[dev,video]"`.
105
+
106
+ ## License
107
+
108
+ [BSD 2-Clause](https://github.com/MAGMA-rob/magma-bench/blob/main/LICENSE).
@@ -0,0 +1,84 @@
1
+ # MAGMA-BENCH
2
+
3
+ Evaluate agents on interactive robotic tasks with MAGMA. MAGMA-BENCH loads
4
+ compiled benchmark episodes, runs them in simulation, communicates with an agent
5
+ through the MAGMA HTTP protocol, and saves detailed results and metrics.
6
+
7
+ **Version 2.0.0b1 is the first beta of v2.** APIs, command-line options, and
8
+ result formats may change before the stable release. End-to-end validation is
9
+ still in progress.
10
+
11
+ [Official documentation](https://magma-rob.github.io/docs/intro) ·
12
+ [Report an issue](https://github.com/MAGMA-rob/magma-bench/issues)
13
+
14
+ ## Installation
15
+
16
+ Python 3.12 is required. Simulation installation has been checked on Linux
17
+ x86_64. GPU simulation and rendering require compatible system drivers.
18
+
19
+ ```bash
20
+ python -m pip install "magma_bench==2.0.0b1"
21
+ ```
22
+
23
+ This installs `magma_core[simulation]>=2.0.0b1,<3.0.0` and
24
+ `magma_scenarios>=2.0.0,<3.0.0` with their Python dependencies. The benchmark
25
+ files and the agent server are separate inputs and are not bundled with this
26
+ package.
27
+
28
+ The version is pinned explicitly because this is a prerelease. To request the
29
+ latest version, including prereleases, use:
30
+
31
+ ```bash
32
+ python -m pip install --upgrade --pre magma_bench
33
+ ```
34
+
35
+ ## Usage
36
+
37
+ Start a compatible MAGMA agent server, then run a compiled benchmark directory:
38
+
39
+ ```bash
40
+ magma-bench run \
41
+ --benchmark-root /path/to/compiled-benchmark \
42
+ --agent-address http://127.0.0.1:8888 \
43
+ --run-name experiment-1
44
+ ```
45
+
46
+ The agent must expose `/health`, `/v1/info`, and `/v1/responses` using protocol
47
+ version 2.0. Configuration can be supplied with `--config-path`; command-line
48
+ options override its benchmark settings.
49
+
50
+ Useful options include:
51
+
52
+ ```bash
53
+ magma-bench run --help
54
+ magma-bench run --benchmark-root /path/to/benchmark --scenarios coffee_comp
55
+ magma-bench run --benchmark-root /path/to/benchmark --skip-judge
56
+ magma-bench run --benchmark-root /path/to/benchmark --model-logs --videos
57
+ ```
58
+
59
+ Video output uses the optional dependencies:
60
+
61
+ ```bash
62
+ python -m pip install "magma_bench[video]==2.0.0b1"
63
+ ```
64
+
65
+ Version 2 uses result schema `2.0`. Resume a compatible run with
66
+ `--results-path`; older result schemas and agent adapters are unsupported.
67
+
68
+ See the [official documentation](https://magma-rob.github.io/docs/intro) for
69
+ benchmark preparation, configuration, agent setup, metrics, and result formats.
70
+
71
+ ## Install from source
72
+
73
+ Install the tagged release from GitHub:
74
+
75
+ ```bash
76
+ python -m pip install "magma_bench @ git+https://github.com/MAGMA-rob/magma-bench.git@v2.0.0b1"
77
+ ```
78
+
79
+ Dependencies are resolved from PyPI. For local development, clone the repository
80
+ and run `python -m pip install -e ".[dev,video]"`.
81
+
82
+ ## License
83
+
84
+ [BSD 2-Clause](https://github.com/MAGMA-rob/magma-bench/blob/main/LICENSE).
@@ -0,0 +1,44 @@
1
+ # SPDX-License-Identifier: BSD-2-Clause
2
+ # Copyright (c) 2026, Loan Bernat
3
+
4
+ [project]
5
+ name = "magma_bench"
6
+ version = "2.0.0b1"
7
+ description = "Evaluate agents on interactive robotic tasks with MAGMA"
8
+ readme = "README.md"
9
+ requires-python = ">=3.12,<3.13"
10
+ classifiers = ["Development Status :: 4 - Beta"]
11
+ dependencies = [
12
+ "magma_core[simulation]>=2.0.0b1,<3.0.0",
13
+ "magma_scenarios>=2.0.0,<3.0.0",
14
+ "pydantic>=2.4,<3",
15
+ "requests>=2.34.2,<3",
16
+ "torch==2.13.0",
17
+ ]
18
+
19
+ [project.urls]
20
+ Documentation = "https://magma-rob.github.io/docs/intro"
21
+ Repository = "https://github.com/MAGMA-rob/magma-bench"
22
+ Issues = "https://github.com/MAGMA-rob/magma-bench/issues"
23
+
24
+ [project.optional-dependencies]
25
+ dev = ["pytest"]
26
+ video = [
27
+ "imageio-ffmpeg>=0.5",
28
+ "numpy>=1.23",
29
+ "Pillow>=10",
30
+ ]
31
+
32
+ [build-system]
33
+ requires = ["setuptools>=61.0"]
34
+ build-backend = "setuptools.build_meta"
35
+
36
+ [tool.setuptools.packages.find]
37
+ where = ["src"]
38
+
39
+ [project.scripts]
40
+ magma-bench = "magma_bench.command_line:main"
41
+
42
+ [tool.pytest.ini_options]
43
+ addopts = "-p no:launch_testing -p no:launch_ros"
44
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
File without changes
@@ -0,0 +1,4 @@
1
+ from .command_line import main
2
+
3
+ if __name__ == "__main__":
4
+ raise SystemExit(main())
@@ -0,0 +1,3 @@
1
+ from .base import BenchmarkAgent
2
+
3
+ __all__ = ["BenchmarkAgent"]
@@ -0,0 +1,5 @@
1
+ from .benchmark_agent import BenchmarkAgent
2
+
3
+ __all__ = [
4
+ "BenchmarkAgent",
5
+ ]
@@ -0,0 +1,202 @@
1
+ from __future__ import annotations
2
+
3
+ from copy import deepcopy
4
+ import json
5
+ import threading
6
+ import time
7
+ from typing import Any, Dict, List, Optional
8
+ from uuid import uuid4
9
+
10
+ import requests
11
+
12
+ from magma_core.domain.agent_call import Call
13
+ from magma_core.protocol.agent import (
14
+ PROTOCOL_VERSION, AgentHealth, AgentInfo, AgentInput, AgentInstruction,
15
+ AgentOutput, AgentRequest, AgentResponse, JsonObject,
16
+ )
17
+ from magma_core.simulation.agents import AgentAnswer, BadAgentAnswer, ValidAgentAnswer
18
+ from magma_core.simulation.data_structures import EmptyInstruction
19
+ from magma_bench.data_structures import BenchmarkAgentResult, EpisodeSituation
20
+
21
+
22
+ class BenchmarkAgent:
23
+ """Batched HTTP client for a single protocol-v2 agent runtime."""
24
+
25
+ def __init__(
26
+ self,
27
+ agent_url: str,
28
+ agent_name: str | None = None,
29
+ extra_keys: JsonObject | None = None,
30
+ timeout: float = 360,
31
+ collect_model_logs: bool = False,
32
+ ) -> None:
33
+ self.agent_url = agent_url.rstrip("/")
34
+ self.timeout = timeout
35
+ self.extra_keys = deepcopy(extra_keys) if extra_keys is not None else {}
36
+ self.collect_model_logs = collect_model_logs
37
+ health = requests.get(f"{self.agent_url}/health", timeout=timeout)
38
+ health.raise_for_status()
39
+ AgentHealth.model_validate(health.json())
40
+ info = requests.get(f"{self.agent_url}/v1/info", timeout=timeout)
41
+ info.raise_for_status()
42
+ self.info = AgentInfo.model_validate(info.json())
43
+ if self.info.protocol_version != PROTOCOL_VERSION:
44
+ raise ValueError(f"Unsupported agent protocol {self.info.protocol_version!r}")
45
+ if not self.info.capabilities.get("inference", False):
46
+ raise ValueError("Agent runtime does not support inference")
47
+ self.agent_name = agent_name or self.info.agent_id
48
+ self.lock = threading.Lock()
49
+ self.completed_results: List[BenchmarkAgentResult] = []
50
+ self.waiting_inputs: Dict[int, EpisodeSituation] = {}
51
+ self.thread_exception: Optional[Exception] = None
52
+ self.thread_running = True
53
+ self.thread = threading.Thread(
54
+ target=self._periodic_computing_of_answers, args=(4,), daemon=True,
55
+ )
56
+ self.thread.start()
57
+
58
+ def compute_agent_results(
59
+ self, batch_inputs: Dict[int, EpisodeSituation],
60
+ ) -> List[BenchmarkAgentResult]:
61
+ if not batch_inputs:
62
+ return []
63
+ inputs: list[AgentInput] = []
64
+ for env_idx, situation in batch_inputs.items():
65
+ instruction = situation.current_instruction
66
+ if isinstance(instruction, EmptyInstruction):
67
+ raise ValueError("EmptyInstruction requires the suspended-skill resume path")
68
+ role = instruction.get_role()
69
+ if role not in {"USER", "SYSTEM"}:
70
+ raise ValueError(f"Unsupported input role {role!r}")
71
+ inputs.append(AgentInput(
72
+ id=env_idx,
73
+ instruction=AgentInstruction(
74
+ type="user" if role == "USER" else "env",
75
+ content=instruction.get_content(),
76
+ ),
77
+ tools=deepcopy(situation.tools),
78
+ attributes=deepcopy(situation.attributes),
79
+ memory=deepcopy(situation.memory),
80
+ num_outputs=1,
81
+ extra_keys=deepcopy(self.extra_keys),
82
+ ))
83
+ request = AgentRequest(request_id=uuid4().hex, inputs=inputs)
84
+ response = self.send_to_agent(request)
85
+ results: list[BenchmarkAgentResult] = []
86
+ for output in response.root:
87
+ updated = batch_inputs[output.source_id].snapshot()
88
+ updated.memory = deepcopy(output.memory)
89
+ diagnostics: list[dict[str, Any]] = []
90
+ if self.collect_model_logs:
91
+ diagnostics.append({
92
+ "internal_steps": deepcopy(output.internal_steps),
93
+ "error": (
94
+ output.error.model_dump(mode="json")
95
+ if output.error is not None
96
+ else None
97
+ ),
98
+ })
99
+ results.append(BenchmarkAgentResult(
100
+ answer=self.normalize_model_response(output),
101
+ situation=updated,
102
+ model_diagnostics=diagnostics,
103
+ ))
104
+ return results
105
+
106
+ def send_to_agent(self, request: AgentRequest) -> AgentResponse:
107
+ response = requests.post(
108
+ f"{self.agent_url}/v1/responses",
109
+ json=request.model_dump(mode="json"), timeout=self.timeout,
110
+ )
111
+ response.raise_for_status()
112
+ parsed = AgentResponse.model_validate(response.json())
113
+ parsed.validate_request(request)
114
+ return parsed
115
+
116
+ def normalize_model_response(
117
+ self, response: AgentOutput, agent_step_id: int = 0,
118
+ ) -> AgentAnswer:
119
+ if response.status == "error":
120
+ assert response.error is not None
121
+ return BadAgentAnswer(
122
+ source_node_id=response.source_id, agent_step_id=agent_step_id, say="",
123
+ raw_action=json.dumps(response.model_dump(mode="json"), ensure_ascii=False),
124
+ reason=response.error.message,
125
+ )
126
+ assert response.output is not None
127
+ return ValidAgentAnswer(
128
+ source_node_id=response.source_id, agent_step_id=agent_step_id,
129
+ say=response.output.say,
130
+ calls=[
131
+ Call(
132
+ name=call.name,
133
+ arguments=deepcopy(call.arguments),
134
+ target_robot_name=call.target_robot_name,
135
+ )
136
+ for call in response.output.tool_calls
137
+ ],
138
+ )
139
+
140
+ def get_agent_card(self) -> Dict[str, Any]:
141
+ return {
142
+ "agent": self.agent_name,
143
+ "agent_id": self.info.agent_id,
144
+ "agent_version": self.info.agent_version,
145
+ "protocol_version": self.info.protocol_version,
146
+ "extra_keys": deepcopy(self.extra_keys),
147
+ }
148
+
149
+ def get_pending_results(self) -> List[BenchmarkAgentResult]:
150
+ with self.lock:
151
+ if self.thread_exception is not None:
152
+ raise RuntimeError(
153
+ "The benchmark-agent background thread failed"
154
+ ) from self.thread_exception
155
+ out = self.completed_results.copy()
156
+ self.completed_results.clear()
157
+ return out
158
+
159
+ def add_inputs(self, inputs: Dict[int, EpisodeSituation]) -> None:
160
+ with self.lock:
161
+ if self.thread_exception is not None:
162
+ raise RuntimeError(
163
+ "Cannot add inputs after the benchmark-agent thread failed"
164
+ ) from self.thread_exception
165
+ if not self.thread_running:
166
+ raise RuntimeError("Cannot add inputs to a stopped benchmark agent")
167
+ for idx, situation in inputs.items():
168
+ if idx in self.waiting_inputs:
169
+ raise RuntimeError(
170
+ f"Environment {idx} already waits for an agent answer"
171
+ )
172
+ self.waiting_inputs[idx] = situation.snapshot()
173
+
174
+ def stop(self) -> None:
175
+ with self.lock:
176
+ self.thread_running = False
177
+ self.thread.join(timeout=5)
178
+
179
+ def _periodic_computing_of_answers(self, interval: float = 4.0) -> None:
180
+ try:
181
+ while self.thread_running:
182
+ start_time = time.time()
183
+ to_do: Dict[int, EpisodeSituation] = {}
184
+ with self.lock:
185
+ if self.waiting_inputs:
186
+ to_do = self.waiting_inputs.copy()
187
+ self.waiting_inputs.clear()
188
+
189
+ if to_do:
190
+ results = self.compute_agent_results(to_do)
191
+ with self.lock:
192
+ self.completed_results.extend(results)
193
+
194
+ elapsed = time.time() - start_time
195
+ sleep_time = max(0, interval - elapsed)
196
+ time.sleep(sleep_time)
197
+
198
+ except Exception as error:
199
+ with self.lock:
200
+ self.thread_exception = error
201
+ with self.lock:
202
+ self.thread_running = False
@@ -0,0 +1,79 @@
1
+ from .models import (
2
+ CONDITIONS,
3
+ Condition,
4
+ ActiveErrorSpec,
5
+ BenchmarkManifest,
6
+ CompiledEpisodeSpec,
7
+ DeclarativeStageSpec,
8
+ EpisodeIndexEntry,
9
+ EpisodeMetadata,
10
+ EpisodeSpec,
11
+ InstructionSpec,
12
+ InterventionSpec,
13
+ LogRuleSpec,
14
+ ObjectSpec,
15
+ ScenarioManifest,
16
+ SemanticManifest,
17
+ SerializedStageSpec,
18
+ SkeletonManifest,
19
+ Track,
20
+ VariantCondition,
21
+ StageInputSpec,
22
+ StagePresentationSpec,
23
+ StageSpec,
24
+ load_json_model,
25
+ )
26
+ from .runtime import (
27
+ DeclarativeStage,
28
+ apply_stage_presentation,
29
+ deserialize_error,
30
+ deserialize_goal,
31
+ deserialize_runtime_randomizer,
32
+ deserialize_serialized_stage,
33
+ deserialize_stage,
34
+ deserialize_task_stages,
35
+ serialize_error,
36
+ serialize_goal,
37
+ serialize_stage,
38
+ serialize_task_stages,
39
+ stage_presentation,
40
+ )
41
+
42
+ __all__ = [
43
+ "CONDITIONS",
44
+ "Condition",
45
+ "ActiveErrorSpec",
46
+ "BenchmarkManifest",
47
+ "CompiledEpisodeSpec",
48
+ "DeclarativeStage",
49
+ "DeclarativeStageSpec",
50
+ "EpisodeIndexEntry",
51
+ "EpisodeMetadata",
52
+ "EpisodeSpec",
53
+ "InstructionSpec",
54
+ "InterventionSpec",
55
+ "LogRuleSpec",
56
+ "ObjectSpec",
57
+ "ScenarioManifest",
58
+ "SemanticManifest",
59
+ "SerializedStageSpec",
60
+ "SkeletonManifest",
61
+ "Track",
62
+ "VariantCondition",
63
+ "StageInputSpec",
64
+ "StagePresentationSpec",
65
+ "StageSpec",
66
+ "apply_stage_presentation",
67
+ "deserialize_error",
68
+ "deserialize_goal",
69
+ "deserialize_runtime_randomizer",
70
+ "deserialize_serialized_stage",
71
+ "deserialize_stage",
72
+ "deserialize_task_stages",
73
+ "load_json_model",
74
+ "serialize_error",
75
+ "serialize_goal",
76
+ "serialize_stage",
77
+ "serialize_task_stages",
78
+ "stage_presentation",
79
+ ]