benchmax 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. {benchmax-0.2.2 → benchmax-0.2.3}/PKG-INFO +1 -1
  2. {benchmax-0.2.2 → benchmax-0.2.3}/pyproject.toml +1 -1
  3. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/__init__.py +2 -0
  4. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/environment.py +6 -0
  5. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/env.py +109 -0
  6. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/shared_types.py +11 -0
  7. {benchmax-0.2.2 → benchmax-0.2.3}/.gitignore +0 -0
  8. {benchmax-0.2.2 → benchmax-0.2.3}/README.md +0 -0
  9. {benchmax-0.2.2 → benchmax-0.2.3}/pytest.ini +0 -0
  10. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/auth.py +0 -0
  11. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/bundle.py +0 -0
  12. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/README.md +0 -0
  13. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/base/README.md +0 -0
  14. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/base/__init__.py +0 -0
  15. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/base/dataset.py +0 -0
  16. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/base/env.py +0 -0
  17. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/base/openai_types.py +0 -0
  18. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/dataset.py +0 -0
  19. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/README.md +0 -0
  20. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/__init__.py +0 -0
  21. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/bundled_agent.py +0 -0
  22. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/credentials.py +0 -0
  23. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/dataset.py +0 -0
  24. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/dep_check.py +0 -0
  25. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/harbor/types.py +0 -0
  26. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/identity.py +0 -0
  27. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/envs/logging.py +0 -0
  28. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rag/__init__.py +0 -0
  29. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rag/embed.py +0 -0
  30. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rag/env.py +0 -0
  31. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rag/search.py +0 -0
  32. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/README.md +0 -0
  33. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/__init__.py +0 -0
  34. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/adaptive.py +0 -0
  35. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/deterministic.py +0 -0
  36. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/diversity.py +0 -0
  37. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/judge.py +0 -0
  38. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/prompts.py +0 -0
  39. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/rubric.py +0 -0
  40. {benchmax-0.2.2 → benchmax-0.2.3}/src/benchmax/rewards/scoring.py +0 -0
  41. {benchmax-0.2.2 → benchmax-0.2.3}/tests/conftest.py +0 -0
  42. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/bundle/test_artifact.py +0 -0
  43. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/bundle/test_source_capture.py +0 -0
  44. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/envs/test_base_dataset.py +0 -0
  45. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/envs/test_base_env_group.py +0 -0
  46. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/envs/test_contract_types.py +0 -0
  47. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/envs/test_environment_group.py +0 -0
  48. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/envs/test_example_id.py +0 -0
  49. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/fakes/model_server.py +0 -0
  50. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/harbor/test_bundled_agent.py +0 -0
  51. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/harbor/test_harbor_dataset.py +0 -0
  52. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/harbor/test_harbor_env.py +0 -0
  53. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rag/corpus/test_embed.py +0 -0
  54. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rag/test_rag_env.py +0 -0
  55. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/conftest.py +0 -0
  56. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_adaptive.py +0 -0
  57. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_deterministic.py +0 -0
  58. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_diversity.py +0 -0
  59. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_diversity_env.py +0 -0
  60. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_judge.py +0 -0
  61. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_rubric.py +0 -0
  62. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/rewards/test_rubric_rewards.py +0 -0
  63. {benchmax-0.2.2 → benchmax-0.2.3}/tests/unit/test_auth.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: benchmax
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Platform-independent runtime for grouped LLM environments
5
5
  Author: benchmax Authors
6
6
  Classifier: Operating System :: OS Independent
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "benchmax"
3
- version = "0.2.2"
3
+ version = "0.2.3"
4
4
  description = "Platform-independent runtime for grouped LLM environments"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "benchmax Authors" }]
@@ -19,6 +19,7 @@ from benchmax.envs.shared_types import (
19
19
  RolloutFailure,
20
20
  RolloutOutcome,
21
21
  RolloutRequest,
22
+ ValidationDiagnostic,
22
23
  )
23
24
 
24
25
  __all__ = [
@@ -39,6 +40,7 @@ __all__ = [
39
40
  "RolloutFailure",
40
41
  "RolloutOutcome",
41
42
  "RolloutRequest",
43
+ "ValidationDiagnostic",
42
44
  "Tool",
43
45
  "StaticBearerAuth",
44
46
  "bind_model_auth",
@@ -15,6 +15,7 @@ from benchmax.envs.shared_types import (
15
15
  RolloutFailure,
16
16
  RolloutOutcome,
17
17
  RolloutRequest,
18
+ ValidationDiagnostic,
18
19
  )
19
20
 
20
21
  __all__ = ["Environment"]
@@ -42,6 +43,11 @@ class Environment[Payload, Attempt: RolloutAttempt](ABC):
42
43
 
43
44
  return False
44
45
 
46
+ def validation_diagnostics(self) -> Sequence[ValidationDiagnostic]:
47
+ """Return static warnings and errors for environment configuration."""
48
+
49
+ return ()
50
+
45
51
  @abstractmethod
46
52
  async def create_dataset(
47
53
  self,
@@ -26,6 +26,7 @@ from benchmax.envs.shared_types import (
26
26
  RolloutAttempt,
27
27
  RolloutOutcome,
28
28
  RolloutRequest,
29
+ ValidationDiagnostic,
29
30
  )
30
31
 
31
32
  if TYPE_CHECKING:
@@ -59,6 +60,72 @@ _TERMINATION_REASON_BY_EXCEPTION = {
59
60
  _HARNESS_REPORTED_TERMINATION_REASONS = frozenset(
60
61
  {"context_exceeded", "output_exceeded", "max_turns_exceeded", "tool_budget_exceeded"}
61
62
  )
63
+ _TRAINER_OWNED_MODEL_FIELDS = frozenset(
64
+ {
65
+ "temperature",
66
+ "top_p",
67
+ "top_k",
68
+ "presence_penalty",
69
+ "frequency_penalty",
70
+ "seed",
71
+ "stop",
72
+ }
73
+ )
74
+ _OUTPUT_CAP_FIELDS = frozenset({"max_tokens", "max_completion_tokens"})
75
+ _UNSUPPORTED_MODEL_FIELDS = frozenset(
76
+ {
77
+ "best_of",
78
+ "do_sample",
79
+ "function_call",
80
+ "functions",
81
+ "grammar",
82
+ "max_new_tokens",
83
+ "min_p",
84
+ "min_tokens",
85
+ "num_beams",
86
+ "prediction",
87
+ "reasoning_effort",
88
+ "repetition_penalty",
89
+ "typical_p",
90
+ "verbosity",
91
+ "web_search_options",
92
+ }
93
+ )
94
+
95
+
96
+ def _walk_model_controls(
97
+ value: Mapping[str, object],
98
+ prefix: str = "agent.kwargs",
99
+ ) -> Sequence[tuple[str, str, object]]:
100
+ controls: list[tuple[str, str, object]] = []
101
+ for field, child in value.items():
102
+ path = f"{prefix}.{field}"
103
+ controls.append((path, field, child))
104
+ if isinstance(child, Mapping):
105
+ controls.extend(_walk_model_controls(child, path))
106
+ return controls
107
+
108
+
109
+ def _unsupported_model_control(field: str, value: object) -> bool:
110
+ if field == "return_routed_experts":
111
+ return True
112
+ if field in _UNSUPPORTED_MODEL_FIELDS:
113
+ return value is not None
114
+ if field == "n":
115
+ return value not in (None, 1)
116
+ if field == "tool_choice":
117
+ return value not in (None, "auto")
118
+ if field == "logprobs":
119
+ return value not in (None, False)
120
+ if field == "top_logprobs":
121
+ return value is not None
122
+ if field == "parallel_tool_calls":
123
+ return value not in (None, True)
124
+ if field == "response_format":
125
+ return value not in (None, {}, {"type": "text"})
126
+ if field == "logit_bias":
127
+ return value not in (None, {})
128
+ return False
62
129
 
63
130
 
64
131
  class HarborEnv(Environment["TaskConfig", RolloutAttempt]):
@@ -74,6 +141,48 @@ class HarborEnv(Environment["TaskConfig", RolloutAttempt]):
74
141
 
75
142
  return self._requires_public_model_endpoint
76
143
 
144
+ def validation_diagnostics(self) -> Sequence[ValidationDiagnostic]:
145
+ """Flag Harbor harness controls that conflict with tracked training."""
146
+
147
+ agent = self._trial.agent
148
+ config = agent.config if isinstance(agent, BundledHarborAgent) else agent
149
+ kwargs = getattr(config, "kwargs", None)
150
+ if not isinstance(kwargs, Mapping):
151
+ return ()
152
+ diagnostics: list[ValidationDiagnostic] = []
153
+ for path, field, value in _walk_model_controls(kwargs):
154
+ if field in _OUTPUT_CAP_FIELDS:
155
+ diagnostics.append(
156
+ ValidationDiagnostic(
157
+ severity="warning",
158
+ code="harness_output_cap",
159
+ location=path,
160
+ message=(
161
+ f"{path} is a harness-requested output cap; Castform may clamp it "
162
+ "to the remaining trainer context budget"
163
+ ),
164
+ )
165
+ )
166
+ elif field in _TRAINER_OWNED_MODEL_FIELDS:
167
+ diagnostics.append(
168
+ ValidationDiagnostic(
169
+ severity="error",
170
+ code="trainer_owned_model_control",
171
+ location=path,
172
+ message=(f"{path} is trainer-owned and cannot be set by a Harbor harness"),
173
+ )
174
+ )
175
+ elif _unsupported_model_control(field, value):
176
+ diagnostics.append(
177
+ ValidationDiagnostic(
178
+ severity="error",
179
+ code="unsupported_model_control",
180
+ location=path,
181
+ message=f"{path}={value!r} is unsupported by tracked training sessions",
182
+ )
183
+ )
184
+ return tuple(diagnostics)
185
+
77
186
  def __init__(
78
187
  self,
79
188
  *,
@@ -13,12 +13,23 @@ __all__ = [
13
13
  "RolloutFailure",
14
14
  "RolloutOutcome",
15
15
  "RolloutRequest",
16
+ "ValidationDiagnostic",
16
17
  ]
17
18
 
18
19
  DatasetSplit = Literal["train", "eval"]
19
20
  type RewardMap = Mapping[str, float]
20
21
 
21
22
 
23
+ @dataclass(frozen=True, slots=True)
24
+ class ValidationDiagnostic:
25
+ """A configuration issue discoverable before executing a rollout."""
26
+
27
+ severity: Literal["warning", "error"]
28
+ code: str
29
+ message: str
30
+ location: str | None = None
31
+
32
+
22
33
  class RolloutFailure(RuntimeError): # noqa: N818 — public exported name
23
34
  """Operational rollout failure that should become a terminal outcome.
24
35
 
File without changes
File without changes
File without changes
File without changes
File without changes