benchmax 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. benchmax-0.2.2/PKG-INFO +111 -0
  2. benchmax-0.2.2/README.md +96 -0
  3. {benchmax-0.2.0 → benchmax-0.2.2}/pyproject.toml +2 -2
  4. benchmax-0.2.2/src/benchmax/auth.py +236 -0
  5. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/bundle.py +32 -55
  6. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/README.md +2 -3
  7. benchmax-0.2.2/src/benchmax/envs/base/README.md +80 -0
  8. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/__init__.py +1 -1
  9. benchmax-0.2.2/src/benchmax/envs/base/dataset.py +80 -0
  10. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/env.py +45 -55
  11. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/dataset.py +11 -1
  12. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/environment.py +63 -78
  13. benchmax-0.2.2/src/benchmax/envs/harbor/README.md +86 -0
  14. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/bundled_agent.py +9 -27
  15. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/credentials.py +3 -8
  16. benchmax-0.2.2/src/benchmax/envs/harbor/dataset.py +303 -0
  17. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/env.py +229 -84
  18. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/types.py +1 -3
  19. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/identity.py +2 -6
  20. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/logging.py +1 -1
  21. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/shared_types.py +24 -7
  22. benchmax-0.2.2/src/benchmax/rag/__init__.py +13 -0
  23. benchmax-0.2.2/src/benchmax/rag/embed.py +79 -0
  24. benchmax-0.2.2/src/benchmax/rag/env.py +699 -0
  25. benchmax-0.2.2/src/benchmax/rag/search.py +64 -0
  26. benchmax-0.2.2/src/benchmax/rewards/README.md +75 -0
  27. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/__init__.py +9 -9
  28. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/adaptive.py +3 -9
  29. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/deterministic.py +3 -3
  30. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/diversity.py +5 -4
  31. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/judge.py +10 -34
  32. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/prompts.py +5 -13
  33. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/rubric.py +15 -37
  34. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/scoring.py +11 -35
  35. {benchmax-0.2.0 → benchmax-0.2.2}/tests/conftest.py +1 -0
  36. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/bundle/test_artifact.py +3 -12
  37. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/bundle/test_source_capture.py +7 -26
  38. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_base_dataset.py +33 -2
  39. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_base_env_group.py +209 -64
  40. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_contract_types.py +7 -5
  41. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_environment_group.py +109 -44
  42. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_example_id.py +0 -1
  43. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/fakes/model_server.py +1 -3
  44. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_bundled_agent.py +12 -24
  45. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_harbor_dataset.py +125 -6
  46. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_harbor_env.py +200 -97
  47. benchmax-0.2.2/tests/unit/rag/corpus/test_embed.py +126 -0
  48. benchmax-0.2.2/tests/unit/rag/test_rag_env.py +790 -0
  49. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_adaptive.py +1 -4
  50. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_deterministic.py +7 -5
  51. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_diversity.py +0 -1
  52. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_diversity_env.py +9 -22
  53. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_judge.py +8 -4
  54. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_rubric.py +1 -4
  55. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_rubric_rewards.py +1 -4
  56. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/test_auth.py +27 -2
  57. benchmax-0.2.0/PKG-INFO +0 -193
  58. benchmax-0.2.0/README.md +0 -178
  59. benchmax-0.2.0/src/benchmax/auth.py +0 -109
  60. benchmax-0.2.0/src/benchmax/envs/base/README.md +0 -74
  61. benchmax-0.2.0/src/benchmax/envs/base/dataset.py +0 -75
  62. benchmax-0.2.0/src/benchmax/envs/harbor/README.md +0 -156
  63. benchmax-0.2.0/src/benchmax/envs/harbor/dataset.py +0 -158
  64. {benchmax-0.2.0 → benchmax-0.2.2}/.gitignore +0 -0
  65. {benchmax-0.2.0 → benchmax-0.2.2}/pytest.ini +0 -0
  66. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/__init__.py +2 -2
  67. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/openai_types.py +0 -0
  68. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/__init__.py +0 -0
  69. {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/dep_check.py +0 -0
  70. {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/conftest.py +2 -2
@@ -0,0 +1,111 @@
1
+ Metadata-Version: 2.4
2
+ Name: benchmax
3
+ Version: 0.2.2
4
+ Summary: Platform-independent runtime for grouped LLM environments
5
+ Author: benchmax Authors
6
+ Classifier: Operating System :: OS Independent
7
+ Classifier: Programming Language :: Python :: 3
8
+ Requires-Python: ==3.12.*
9
+ Requires-Dist: cloudpickle>=3.0.0
10
+ Requires-Dist: openai>=2.15.0
11
+ Requires-Dist: packaging>=24.0
12
+ Provides-Extra: harbor
13
+ Requires-Dist: harbor<0.19,>=0.18.0; extra == 'harbor'
14
+ Description-Content-Type: text/markdown
15
+
16
+ # benchmax
17
+
18
+ benchmax envs is where you define datasets, how to execute the rollout, and scoring each rollout.
19
+
20
+ for installation and project setup, start with the [main readme](../../README.md#get-started). working environments live in [`examples/`](../../examples/).
21
+
22
+ ## choose an environment
23
+
24
+ all benchmax environments implement the same dataset and rollout contracts. choose the adapter based on who owns the agent loop:
25
+
26
+ | environment | use it when | what it provides |
27
+ | --- | --- | --- |
28
+ | [`BaseEnv`](src/benchmax/envs/base/README.md) | default environment to extend - runs a simple loop with the option to make tool calls | chat completions, tool dispatch, turn limits, and reward hooks |
29
+ | [`HarborEnv`](src/benchmax/envs/harbor/README.md) | you already have a Harbor task or harness | Harbor agents, sandboxes, verifiers, and RewardKit integration |
30
+ | [`Environment`](src/benchmax/envs/README.md) | extend `Environment` if you need custom behavior not covered by `BaseEnv` and `HarborEnv` | the fundamental dataset, group execution, and outcome contracts |
31
+
32
+ most custom environments should extend `BaseEnv`. most Harbor users configure `HarborEnv` directly rather than subclassing it.
33
+
34
+ ## architecture
35
+
36
+ an environment defines its dataset and how a group of rollouts runs against each example.
37
+
38
+ ```text
39
+ Environment
40
+ ├── create_dataset(split, base_dir, max_examples)
41
+ │ └── Dataset
42
+ │ └── Example(id, payload)
43
+ └── run_group(requests)
44
+ ├── run_rollout(request) × group_size → RolloutAttempt × group_size
45
+ ├── adapter-specific scoring
46
+ ├── optional group-relative scoring
47
+ └── RolloutOutcome(rewards, termination_reason, error)
48
+ ```
49
+
50
+ every environment follows this shape. the environment decides what an example contains, how each attempt runs, which tools are available, and how the result is scored.
51
+
52
+ ## datasets
53
+
54
+ `create_dataset` receives a `train` or `eval` split and returns a fixed, ordered `Dataset` of `Example` objects.
55
+
56
+ each example contains:
57
+
58
+ - a stable id used to identify the datapoint across runs;
59
+ - an environment-owned payload consumed by its rollout implementation.
60
+
61
+ the optional `max_examples` argument limits how many examples are returned. when the data source supports it, the environment should stop loading once it reaches that limit.
62
+
63
+ `JsonlDataset`, Harbor datasets, and custom datasets all produce the same fundamental `Dataset` type. the trainer and validation flow do not depend on the source file format.
64
+
65
+ ## tools
66
+
67
+ `BaseEnv` exposes OpenAI-compatible function tools through `list_tools` and executes them through `run_tool`. an environment can provide no tools, one tool, or a collection of stateful tools.
68
+
69
+ with `HarborEnv`, the Harbor agent and harness define the available tools and how they interact with the sandbox. benchmax does not convert Harbor tools into `BaseEnv` tools.
70
+
71
+ ## execution and scoring
72
+
73
+ `run_group` receives multiple rollout requests for the same example, runs them concurrently, waits for all siblings, and returns one `RolloutOutcome` for each request.
74
+
75
+ successful scoring hooks return their named reward components. operational failures return no rewards and do not cancel successful siblings; the trainer treats absent components as zero. partial attempts that reach a context, output, turn, or tool limit can still be scored.
76
+
77
+ - `BaseEnv` runs the model and tool loop, then passes the transcript and example payload to `compute_reward`. `compute_group_rewards` can score the completed sibling group.
78
+ - `HarborEnv` runs the configured Harbor agent and sandbox, then preserves its verifier or RewardKit reward components.
79
+
80
+ ### helpers
81
+
82
+ `benchmax.rewards` provides deterministic text helpers, model judges, rubrics, ranking, adaptive rubrics, and diversity scoring for `BaseEnv` and direct `Environment` implementations. see the [rewards guide](src/benchmax/rewards/README.md).
83
+
84
+ Harbor environments normally use their harness verifier and RewardKit instead of benchmax reward helpers.
85
+
86
+ ## bundling
87
+
88
+ a bundle contains the environment class, its constructor arguments, the project-local source it needs, and its declared remote dependencies.
89
+
90
+ ```text
91
+ environment class + constructor arguments
92
+ + local source
93
+ + dependency metadata
94
+ │
95
+ ▼
96
+ portable bundle
97
+ ```
98
+
99
+ benchmax creates the portable artifact so the same environment can be loaded outside the author's checkout. `castform` handles uploading the bundle, validating it remotely, and using it for training.
100
+
101
+ Remote runtimes install `pip_dependencies` without enabling prereleases globally. If a dependency graph needs a prerelease, list that package explicitly even when it would normally be transitive. For example, use `pip_dependencies=["parent-package==1.0.0", "transitive-package==2.0.0rc1"]`.
102
+
103
+ ## further reading
104
+
105
+ - [base environment guide](src/benchmax/envs/base/README.md)
106
+ - [harbor environment guide](src/benchmax/envs/harbor/README.md)
107
+ - [reward helpers](src/benchmax/rewards/README.md)
108
+ - [examples](../../examples/)
109
+ - [development instructions](../../README.md#development)
110
+
111
+ apache 2.0 © 2026 CGFT Inc.
@@ -0,0 +1,96 @@
1
+ # benchmax
2
+
3
+ benchmax envs is where you define datasets, how to execute the rollout, and scoring each rollout.
4
+
5
+ for installation and project setup, start with the [main readme](../../README.md#get-started). working environments live in [`examples/`](../../examples/).
6
+
7
+ ## choose an environment
8
+
9
+ all benchmax environments implement the same dataset and rollout contracts. choose the adapter based on who owns the agent loop:
10
+
11
+ | environment | use it when | what it provides |
12
+ | --- | --- | --- |
13
+ | [`BaseEnv`](src/benchmax/envs/base/README.md) | default environment to extend - runs a simple loop with the option to make tool calls | chat completions, tool dispatch, turn limits, and reward hooks |
14
+ | [`HarborEnv`](src/benchmax/envs/harbor/README.md) | you already have a Harbor task or harness | Harbor agents, sandboxes, verifiers, and RewardKit integration |
15
+ | [`Environment`](src/benchmax/envs/README.md) | extend `Environment` if you need custom behavior not covered by `BaseEnv` and `HarborEnv` | the fundamental dataset, group execution, and outcome contracts |
16
+
17
+ most custom environments should extend `BaseEnv`. most Harbor users configure `HarborEnv` directly rather than subclassing it.
18
+
19
+ ## architecture
20
+
21
+ an environment defines its dataset and how a group of rollouts runs against each example.
22
+
23
+ ```text
24
+ Environment
25
+ ├── create_dataset(split, base_dir, max_examples)
26
+ │ └── Dataset
27
+ │ └── Example(id, payload)
28
+ └── run_group(requests)
29
+ ├── run_rollout(request) × group_size → RolloutAttempt × group_size
30
+ ├── adapter-specific scoring
31
+ ├── optional group-relative scoring
32
+ └── RolloutOutcome(rewards, termination_reason, error)
33
+ ```
34
+
35
+ every environment follows this shape. the environment decides what an example contains, how each attempt runs, which tools are available, and how the result is scored.
36
+
37
+ ## datasets
38
+
39
+ `create_dataset` receives a `train` or `eval` split and returns a fixed, ordered `Dataset` of `Example` objects.
40
+
41
+ each example contains:
42
+
43
+ - a stable id used to identify the datapoint across runs;
44
+ - an environment-owned payload consumed by its rollout implementation.
45
+
46
+ the optional `max_examples` argument limits how many examples are returned. when the data source supports it, the environment should stop loading once it reaches that limit.
47
+
48
+ `JsonlDataset`, Harbor datasets, and custom datasets all produce the same fundamental `Dataset` type. the trainer and validation flow do not depend on the source file format.
49
+
50
+ ## tools
51
+
52
+ `BaseEnv` exposes OpenAI-compatible function tools through `list_tools` and executes them through `run_tool`. an environment can provide no tools, one tool, or a collection of stateful tools.
53
+
54
+ with `HarborEnv`, the Harbor agent and harness define the available tools and how they interact with the sandbox. benchmax does not convert Harbor tools into `BaseEnv` tools.
55
+
56
+ ## execution and scoring
57
+
58
+ `run_group` receives multiple rollout requests for the same example, runs them concurrently, waits for all siblings, and returns one `RolloutOutcome` for each request.
59
+
60
+ successful scoring hooks return their named reward components. operational failures return no rewards and do not cancel successful siblings; the trainer treats absent components as zero. partial attempts that reach a context, output, turn, or tool limit can still be scored.
61
+
62
+ - `BaseEnv` runs the model and tool loop, then passes the transcript and example payload to `compute_reward`. `compute_group_rewards` can score the completed sibling group.
63
+ - `HarborEnv` runs the configured Harbor agent and sandbox, then preserves its verifier or RewardKit reward components.
64
+
65
+ ### helpers
66
+
67
+ `benchmax.rewards` provides deterministic text helpers, model judges, rubrics, ranking, adaptive rubrics, and diversity scoring for `BaseEnv` and direct `Environment` implementations. see the [rewards guide](src/benchmax/rewards/README.md).
68
+
69
+ Harbor environments normally use their harness verifier and RewardKit instead of benchmax reward helpers.
70
+
71
+ ## bundling
72
+
73
+ a bundle contains the environment class, its constructor arguments, the project-local source it needs, and its declared remote dependencies.
74
+
75
+ ```text
76
+ environment class + constructor arguments
77
+ + local source
78
+ + dependency metadata
79
+ │
80
+ ▼
81
+ portable bundle
82
+ ```
83
+
84
+ benchmax creates the portable artifact so the same environment can be loaded outside the author's checkout. `castform` handles uploading the bundle, validating it remotely, and using it for training.
85
+
86
+ Remote runtimes install `pip_dependencies` without enabling prereleases globally. If a dependency graph needs a prerelease, list that package explicitly even when it would normally be transitive. For example, use `pip_dependencies=["parent-package==1.0.0", "transitive-package==2.0.0rc1"]`.
87
+
88
+ ## further reading
89
+
90
+ - [base environment guide](src/benchmax/envs/base/README.md)
91
+ - [harbor environment guide](src/benchmax/envs/harbor/README.md)
92
+ - [reward helpers](src/benchmax/rewards/README.md)
93
+ - [examples](../../examples/)
94
+ - [development instructions](../../README.md#development)
95
+
96
+ apache 2.0 © 2026 CGFT Inc.
@@ -1,9 +1,9 @@
1
1
  [project]
2
2
  name = "benchmax"
3
- version = "0.2.0"
3
+ version = "0.2.2"
4
4
  description = "Platform-independent runtime for grouped LLM environments"
5
5
  readme = "README.md"
6
- authors = [{ name = "BenchMax Authors" }]
6
+ authors = [{ name = "benchmax Authors" }]
7
7
  requires-python = "==3.12.*"
8
8
  dependencies = [
9
9
  "cloudpickle>=3.0.0",
@@ -0,0 +1,236 @@
1
+ """Explicit, call-time authentication for model requests.
2
+
3
+ benchmax defines only the runtime contract. Platform packages and execution
4
+ runtimes provide concrete credential sources and bind injected credentials.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import asyncio
10
+ import contextvars
11
+ import threading
12
+ from collections.abc import AsyncIterator, Iterator, Mapping
13
+ from contextlib import contextmanager
14
+ from contextvars import ContextVar
15
+ from dataclasses import dataclass, field
16
+ from typing import Protocol, runtime_checkable
17
+
18
+ import httpx
19
+ from openai import AsyncOpenAI, OpenAI
20
+
21
+ __all__ = [
22
+ "InjectedAuth",
23
+ "ModelAuth",
24
+ "ModelRequestContext",
25
+ "RequestModelAuth",
26
+ "StaticBearerAuth",
27
+ "bind_model_auth",
28
+ "create_async_openai_client",
29
+ "create_openai_client",
30
+ ]
31
+
32
+
33
+ @dataclass(frozen=True, slots=True)
34
+ class ModelRequestContext:
35
+ """Identity of the model request about to be authorized."""
36
+
37
+ base_url: str
38
+ model: str
39
+ rollout_id: str
40
+
41
+
42
+ @runtime_checkable
43
+ class ModelAuth(Protocol):
44
+ """Return headers immediately before each model HTTP request."""
45
+
46
+ async def headers_for_request(
47
+ self,
48
+ context: ModelRequestContext,
49
+ ) -> Mapping[str, str]: ...
50
+
51
+
52
+ @dataclass(frozen=True, slots=True)
53
+ class StaticBearerAuth:
54
+ """Explicit bearer authentication for providers with a stable API key."""
55
+
56
+ token: str = field(repr=False)
57
+
58
+ def __post_init__(self) -> None:
59
+ if not isinstance(self.token, str) or not self.token:
60
+ raise ValueError("bearer token must be a non-empty string")
61
+
62
+ async def headers_for_request(
63
+ self,
64
+ context: ModelRequestContext,
65
+ ) -> Mapping[str, str]:
66
+ del context
67
+ return {"Authorization": f"Bearer {self.token}"}
68
+
69
+
70
+ class RequestModelAuth(httpx.Auth):
71
+ """Apply a :class:`ModelAuth` immediately before each HTTP request.
72
+
73
+ Both sync and async OpenAI-compatible clients use this adapter, so model
74
+ credential selection never falls back to an SDK environment variable or a
75
+ separate token resolver. When a sync client runs inside an active event
76
+ loop, auth is resolved in a context-preserving helper thread; custom
77
+ providers used there must not depend on primitives bound to another loop.
78
+ """
79
+
80
+ def __init__(self, auth: ModelAuth, context: ModelRequestContext) -> None:
81
+ if not isinstance(auth, ModelAuth):
82
+ raise TypeError("request auth must implement ModelAuth")
83
+ self._auth = auth
84
+ self._context = context
85
+
86
+ def sync_auth_flow(
87
+ self,
88
+ request: httpx.Request,
89
+ ) -> Iterator[httpx.Request]:
90
+ headers = _resolve_headers_sync(self._auth, self._context)
91
+ for name, value in headers.items():
92
+ request.headers[name] = value
93
+ yield request
94
+
95
+ async def async_auth_flow(
96
+ self,
97
+ request: httpx.Request,
98
+ ) -> AsyncIterator[httpx.Request]:
99
+ headers = await self._auth.headers_for_request(self._context)
100
+ for name, value in headers.items():
101
+ request.headers[name] = value
102
+ yield request
103
+
104
+
105
+ def create_openai_client(
106
+ *,
107
+ model: str,
108
+ base_url: str,
109
+ auth: ModelAuth,
110
+ request_id: str,
111
+ max_retries: int = 2,
112
+ ) -> OpenAI:
113
+ """Create a synchronous OpenAI-compatible client with explicit auth."""
114
+
115
+ context = ModelRequestContext(
116
+ base_url=base_url,
117
+ model=model,
118
+ rollout_id=request_id,
119
+ )
120
+ return OpenAI(
121
+ base_url=base_url,
122
+ api_key="benchmax-explicit-auth",
123
+ http_client=httpx.Client(auth=RequestModelAuth(auth, context)),
124
+ max_retries=max_retries,
125
+ )
126
+
127
+
128
+ def create_async_openai_client(
129
+ *,
130
+ model: str,
131
+ base_url: str,
132
+ auth: ModelAuth,
133
+ request_id: str,
134
+ max_retries: int = 2,
135
+ ) -> AsyncOpenAI:
136
+ """Create an asynchronous OpenAI-compatible client with explicit auth."""
137
+
138
+ context = ModelRequestContext(
139
+ base_url=base_url,
140
+ model=model,
141
+ rollout_id=request_id,
142
+ )
143
+ return AsyncOpenAI(
144
+ base_url=base_url,
145
+ api_key="benchmax-explicit-auth",
146
+ http_client=httpx.AsyncClient(auth=RequestModelAuth(auth, context)),
147
+ max_retries=max_retries,
148
+ )
149
+
150
+
151
+ def _resolve_headers_sync(
152
+ auth: ModelAuth,
153
+ context: ModelRequestContext,
154
+ ) -> Mapping[str, str]:
155
+ """Resolve async ``ModelAuth`` from a synchronous HTTP client.
156
+
157
+ RAG search backends expose synchronous embedding callables. When one is
158
+ invoked from an async environment tool, its event loop is already running;
159
+ resolve the auth coroutine in a context-preserving helper thread rather
160
+ than attempting a nested event loop.
161
+ """
162
+
163
+ async def resolve() -> Mapping[str, str]:
164
+ return await auth.headers_for_request(context)
165
+
166
+ try:
167
+ asyncio.get_running_loop()
168
+ except RuntimeError:
169
+ return asyncio.run(resolve())
170
+
171
+ copied_context = contextvars.copy_context()
172
+ result: list[Mapping[str, str]] = []
173
+ failure: list[BaseException] = []
174
+
175
+ def run() -> None:
176
+ try:
177
+ result.append(copied_context.run(lambda: asyncio.run(resolve())))
178
+ except BaseException as error: # propagate the original auth failure
179
+ failure.append(error)
180
+
181
+ thread = threading.Thread(target=run, daemon=True)
182
+ thread.start()
183
+ thread.join()
184
+ if failure:
185
+ raise failure[0]
186
+ if not result:
187
+ raise RuntimeError("model authentication did not return headers")
188
+ return result[0]
189
+
190
+
191
+ _BOUND_MODEL_AUTH: ContextVar[Mapping[str, ModelAuth] | None] = ContextVar(
192
+ "benchmax_bound_model_auth",
193
+ default=None,
194
+ )
195
+
196
+
197
+ @dataclass(frozen=True, slots=True)
198
+ class InjectedAuth:
199
+ """Serializable reference to authentication supplied by the runtime."""
200
+
201
+ name: str
202
+
203
+ def __post_init__(self) -> None:
204
+ if not isinstance(self.name, str) or not self.name.strip():
205
+ raise ValueError("injected auth name must be a non-empty string")
206
+
207
+ async def headers_for_request(
208
+ self,
209
+ context: ModelRequestContext,
210
+ ) -> Mapping[str, str]:
211
+ providers = _BOUND_MODEL_AUTH.get()
212
+ provider = providers.get(self.name) if providers is not None else None
213
+ if provider is None:
214
+ raise RuntimeError(f"No runtime model-auth provider was injected for {self.name!r}.")
215
+ if isinstance(provider, InjectedAuth):
216
+ raise RuntimeError(
217
+ f"Injected model-auth provider {self.name!r} cannot reference another InjectedAuth."
218
+ )
219
+ return await provider.headers_for_request(context)
220
+
221
+
222
+ @contextmanager
223
+ def bind_model_auth(providers: Mapping[str, ModelAuth]) -> Iterator[None]:
224
+ """Bind runtime providers for the current async execution context."""
225
+
226
+ normalized = dict(providers)
227
+ for name, provider in normalized.items():
228
+ if not isinstance(name, str) or not name.strip():
229
+ raise ValueError("model-auth provider names must be non-empty strings")
230
+ if not isinstance(provider, ModelAuth):
231
+ raise TypeError(f"model-auth provider {name!r} does not implement ModelAuth")
232
+ token = _BOUND_MODEL_AUTH.set(normalized)
233
+ try:
234
+ yield
235
+ finally:
236
+ _BOUND_MODEL_AUTH.reset(token)