benchmax 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- benchmax-0.2.2/PKG-INFO +111 -0
- benchmax-0.2.2/README.md +96 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/pyproject.toml +2 -2
- benchmax-0.2.2/src/benchmax/auth.py +236 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/bundle.py +32 -55
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/README.md +2 -3
- benchmax-0.2.2/src/benchmax/envs/base/README.md +80 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/__init__.py +1 -1
- benchmax-0.2.2/src/benchmax/envs/base/dataset.py +80 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/env.py +45 -55
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/dataset.py +11 -1
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/environment.py +63 -78
- benchmax-0.2.2/src/benchmax/envs/harbor/README.md +86 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/bundled_agent.py +9 -27
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/credentials.py +3 -8
- benchmax-0.2.2/src/benchmax/envs/harbor/dataset.py +303 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/env.py +229 -84
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/types.py +1 -3
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/identity.py +2 -6
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/logging.py +1 -1
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/shared_types.py +24 -7
- benchmax-0.2.2/src/benchmax/rag/__init__.py +13 -0
- benchmax-0.2.2/src/benchmax/rag/embed.py +79 -0
- benchmax-0.2.2/src/benchmax/rag/env.py +699 -0
- benchmax-0.2.2/src/benchmax/rag/search.py +64 -0
- benchmax-0.2.2/src/benchmax/rewards/README.md +75 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/__init__.py +9 -9
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/adaptive.py +3 -9
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/deterministic.py +3 -3
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/diversity.py +5 -4
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/judge.py +10 -34
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/prompts.py +5 -13
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/rubric.py +15 -37
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/rewards/scoring.py +11 -35
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/conftest.py +1 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/bundle/test_artifact.py +3 -12
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/bundle/test_source_capture.py +7 -26
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_base_dataset.py +33 -2
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_base_env_group.py +209 -64
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_contract_types.py +7 -5
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_environment_group.py +109 -44
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/envs/test_example_id.py +0 -1
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/fakes/model_server.py +1 -3
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_bundled_agent.py +12 -24
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_harbor_dataset.py +125 -6
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/harbor/test_harbor_env.py +200 -97
- benchmax-0.2.2/tests/unit/rag/corpus/test_embed.py +126 -0
- benchmax-0.2.2/tests/unit/rag/test_rag_env.py +790 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_adaptive.py +1 -4
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_deterministic.py +7 -5
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_diversity.py +0 -1
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_diversity_env.py +9 -22
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_judge.py +8 -4
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_rubric.py +1 -4
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/test_rubric_rewards.py +1 -4
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/test_auth.py +27 -2
- benchmax-0.2.0/PKG-INFO +0 -193
- benchmax-0.2.0/README.md +0 -178
- benchmax-0.2.0/src/benchmax/auth.py +0 -109
- benchmax-0.2.0/src/benchmax/envs/base/README.md +0 -74
- benchmax-0.2.0/src/benchmax/envs/base/dataset.py +0 -75
- benchmax-0.2.0/src/benchmax/envs/harbor/README.md +0 -156
- benchmax-0.2.0/src/benchmax/envs/harbor/dataset.py +0 -158
- {benchmax-0.2.0 → benchmax-0.2.2}/.gitignore +0 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/pytest.ini +0 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/__init__.py +2 -2
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/base/openai_types.py +0 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/__init__.py +0 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/src/benchmax/envs/harbor/dep_check.py +0 -0
- {benchmax-0.2.0 → benchmax-0.2.2}/tests/unit/rewards/conftest.py +2 -2
benchmax-0.2.2/PKG-INFO
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: benchmax
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: Platform-independent runtime for grouped LLM environments
|
|
5
|
+
Author: benchmax Authors
|
|
6
|
+
Classifier: Operating System :: OS Independent
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Requires-Python: ==3.12.*
|
|
9
|
+
Requires-Dist: cloudpickle>=3.0.0
|
|
10
|
+
Requires-Dist: openai>=2.15.0
|
|
11
|
+
Requires-Dist: packaging>=24.0
|
|
12
|
+
Provides-Extra: harbor
|
|
13
|
+
Requires-Dist: harbor<0.19,>=0.18.0; extra == 'harbor'
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# benchmax
|
|
17
|
+
|
|
18
|
+
benchmax envs is where you define datasets, how to execute the rollout, and scoring each rollout.
|
|
19
|
+
|
|
20
|
+
for installation and project setup, start with the [main readme](../../README.md#get-started). working environments live in [`examples/`](../../examples/).
|
|
21
|
+
|
|
22
|
+
## choose an environment
|
|
23
|
+
|
|
24
|
+
all benchmax environments implement the same dataset and rollout contracts. choose the adapter based on who owns the agent loop:
|
|
25
|
+
|
|
26
|
+
| environment | use it when | what it provides |
|
|
27
|
+
| --- | --- | --- |
|
|
28
|
+
| [`BaseEnv`](src/benchmax/envs/base/README.md) | default environment to extend - runs a simple loop with the option to make tool calls | chat completions, tool dispatch, turn limits, and reward hooks |
|
|
29
|
+
| [`HarborEnv`](src/benchmax/envs/harbor/README.md) | you already have a Harbor task or harness | Harbor agents, sandboxes, verifiers, and RewardKit integration |
|
|
30
|
+
| [`Environment`](src/benchmax/envs/README.md) | extend `Environment` if you need custom behavior not covered by `BaseEnv` and `HarborEnv` | the fundamental dataset, group execution, and outcome contracts |
|
|
31
|
+
|
|
32
|
+
most custom environments should extend `BaseEnv`. most Harbor users configure `HarborEnv` directly rather than subclassing it.
|
|
33
|
+
|
|
34
|
+
## architecture
|
|
35
|
+
|
|
36
|
+
an environment defines its dataset and how a group of rollouts runs against each example.
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
Environment
|
|
40
|
+
├── create_dataset(split, base_dir, max_examples)
|
|
41
|
+
│ └── Dataset
|
|
42
|
+
│ └── Example(id, payload)
|
|
43
|
+
└── run_group(requests)
|
|
44
|
+
├── run_rollout(request) × group_size → RolloutAttempt × group_size
|
|
45
|
+
├── adapter-specific scoring
|
|
46
|
+
├── optional group-relative scoring
|
|
47
|
+
└── RolloutOutcome(rewards, termination_reason, error)
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
every environment follows this shape. the environment decides what an example contains, how each attempt runs, which tools are available, and how the result is scored.
|
|
51
|
+
|
|
52
|
+
## datasets
|
|
53
|
+
|
|
54
|
+
`create_dataset` receives a `train` or `eval` split and returns a fixed, ordered `Dataset` of `Example` objects.
|
|
55
|
+
|
|
56
|
+
each example contains:
|
|
57
|
+
|
|
58
|
+
- a stable id used to identify the datapoint across runs;
|
|
59
|
+
- an environment-owned payload consumed by its rollout implementation.
|
|
60
|
+
|
|
61
|
+
the optional `max_examples` argument limits how many examples are returned. when the data source supports it, the environment should stop loading once it reaches that limit.
|
|
62
|
+
|
|
63
|
+
`JsonlDataset`, Harbor datasets, and custom datasets all produce the same fundamental `Dataset` type. the trainer and validation flow do not depend on the source file format.
|
|
64
|
+
|
|
65
|
+
## tools
|
|
66
|
+
|
|
67
|
+
`BaseEnv` exposes OpenAI-compatible function tools through `list_tools` and executes them through `run_tool`. an environment can provide no tools, one tool, or a collection of stateful tools.
|
|
68
|
+
|
|
69
|
+
with `HarborEnv`, the Harbor agent and harness define the available tools and how they interact with the sandbox. benchmax does not convert Harbor tools into `BaseEnv` tools.
|
|
70
|
+
|
|
71
|
+
## execution and scoring
|
|
72
|
+
|
|
73
|
+
`run_group` receives multiple rollout requests for the same example, runs them concurrently, waits for all siblings, and returns one `RolloutOutcome` for each request.
|
|
74
|
+
|
|
75
|
+
successful scoring hooks return their named reward components. operational failures return no rewards and do not cancel successful siblings; the trainer treats absent components as zero. partial attempts that reach a context, output, turn, or tool limit can still be scored.
|
|
76
|
+
|
|
77
|
+
- `BaseEnv` runs the model and tool loop, then passes the transcript and example payload to `compute_reward`. `compute_group_rewards` can score the completed sibling group.
|
|
78
|
+
- `HarborEnv` runs the configured Harbor agent and sandbox, then preserves its verifier or RewardKit reward components.
|
|
79
|
+
|
|
80
|
+
### helpers
|
|
81
|
+
|
|
82
|
+
`benchmax.rewards` provides deterministic text helpers, model judges, rubrics, ranking, adaptive rubrics, and diversity scoring for `BaseEnv` and direct `Environment` implementations. see the [rewards guide](src/benchmax/rewards/README.md).
|
|
83
|
+
|
|
84
|
+
Harbor environments normally use their harness verifier and RewardKit instead of benchmax reward helpers.
|
|
85
|
+
|
|
86
|
+
## bundling
|
|
87
|
+
|
|
88
|
+
a bundle contains the environment class, its constructor arguments, the project-local source it needs, and its declared remote dependencies.
|
|
89
|
+
|
|
90
|
+
```text
|
|
91
|
+
environment class + constructor arguments
|
|
92
|
+
+ local source
|
|
93
|
+
+ dependency metadata
|
|
94
|
+
│
|
|
95
|
+
▼
|
|
96
|
+
portable bundle
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
benchmax creates the portable artifact so the same environment can be loaded outside the author's checkout. `castform` handles uploading the bundle, validating it remotely, and using it for training.
|
|
100
|
+
|
|
101
|
+
Remote runtimes install `pip_dependencies` without enabling prereleases globally. If a dependency graph needs a prerelease, list that package explicitly even when it would normally be transitive. For example, use `pip_dependencies=["parent-package==1.0.0", "transitive-package==2.0.0rc1"]`.
|
|
102
|
+
|
|
103
|
+
## further reading
|
|
104
|
+
|
|
105
|
+
- [base environment guide](src/benchmax/envs/base/README.md)
|
|
106
|
+
- [harbor environment guide](src/benchmax/envs/harbor/README.md)
|
|
107
|
+
- [reward helpers](src/benchmax/rewards/README.md)
|
|
108
|
+
- [examples](../../examples/)
|
|
109
|
+
- [development instructions](../../README.md#development)
|
|
110
|
+
|
|
111
|
+
apache 2.0 © 2026 CGFT Inc.
|
benchmax-0.2.2/README.md
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# benchmax
|
|
2
|
+
|
|
3
|
+
benchmax envs is where you define datasets, how to execute the rollout, and scoring each rollout.
|
|
4
|
+
|
|
5
|
+
for installation and project setup, start with the [main readme](../../README.md#get-started). working environments live in [`examples/`](../../examples/).
|
|
6
|
+
|
|
7
|
+
## choose an environment
|
|
8
|
+
|
|
9
|
+
all benchmax environments implement the same dataset and rollout contracts. choose the adapter based on who owns the agent loop:
|
|
10
|
+
|
|
11
|
+
| environment | use it when | what it provides |
|
|
12
|
+
| --- | --- | --- |
|
|
13
|
+
| [`BaseEnv`](src/benchmax/envs/base/README.md) | default environment to extend - runs a simple loop with the option to make tool calls | chat completions, tool dispatch, turn limits, and reward hooks |
|
|
14
|
+
| [`HarborEnv`](src/benchmax/envs/harbor/README.md) | you already have a Harbor task or harness | Harbor agents, sandboxes, verifiers, and RewardKit integration |
|
|
15
|
+
| [`Environment`](src/benchmax/envs/README.md) | extend `Environment` if you need custom behavior not covered by `BaseEnv` and `HarborEnv` | the fundamental dataset, group execution, and outcome contracts |
|
|
16
|
+
|
|
17
|
+
most custom environments should extend `BaseEnv`. most Harbor users configure `HarborEnv` directly rather than subclassing it.
|
|
18
|
+
|
|
19
|
+
## architecture
|
|
20
|
+
|
|
21
|
+
an environment defines its dataset and how a group of rollouts runs against each example.
|
|
22
|
+
|
|
23
|
+
```text
|
|
24
|
+
Environment
|
|
25
|
+
├── create_dataset(split, base_dir, max_examples)
|
|
26
|
+
│ └── Dataset
|
|
27
|
+
│ └── Example(id, payload)
|
|
28
|
+
└── run_group(requests)
|
|
29
|
+
├── run_rollout(request) × group_size → RolloutAttempt × group_size
|
|
30
|
+
├── adapter-specific scoring
|
|
31
|
+
├── optional group-relative scoring
|
|
32
|
+
└── RolloutOutcome(rewards, termination_reason, error)
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
every environment follows this shape. the environment decides what an example contains, how each attempt runs, which tools are available, and how the result is scored.
|
|
36
|
+
|
|
37
|
+
## datasets
|
|
38
|
+
|
|
39
|
+
`create_dataset` receives a `train` or `eval` split and returns a fixed, ordered `Dataset` of `Example` objects.
|
|
40
|
+
|
|
41
|
+
each example contains:
|
|
42
|
+
|
|
43
|
+
- a stable id used to identify the datapoint across runs;
|
|
44
|
+
- an environment-owned payload consumed by its rollout implementation.
|
|
45
|
+
|
|
46
|
+
the optional `max_examples` argument limits how many examples are returned. when the data source supports it, the environment should stop loading once it reaches that limit.
|
|
47
|
+
|
|
48
|
+
`JsonlDataset`, Harbor datasets, and custom datasets all produce the same fundamental `Dataset` type. the trainer and validation flow do not depend on the source file format.
|
|
49
|
+
|
|
50
|
+
## tools
|
|
51
|
+
|
|
52
|
+
`BaseEnv` exposes OpenAI-compatible function tools through `list_tools` and executes them through `run_tool`. an environment can provide no tools, one tool, or a collection of stateful tools.
|
|
53
|
+
|
|
54
|
+
with `HarborEnv`, the Harbor agent and harness define the available tools and how they interact with the sandbox. benchmax does not convert Harbor tools into `BaseEnv` tools.
|
|
55
|
+
|
|
56
|
+
## execution and scoring
|
|
57
|
+
|
|
58
|
+
`run_group` receives multiple rollout requests for the same example, runs them concurrently, waits for all siblings, and returns one `RolloutOutcome` for each request.
|
|
59
|
+
|
|
60
|
+
successful scoring hooks return their named reward components. operational failures return no rewards and do not cancel successful siblings; the trainer treats absent components as zero. partial attempts that reach a context, output, turn, or tool limit can still be scored.
|
|
61
|
+
|
|
62
|
+
- `BaseEnv` runs the model and tool loop, then passes the transcript and example payload to `compute_reward`. `compute_group_rewards` can score the completed sibling group.
|
|
63
|
+
- `HarborEnv` runs the configured Harbor agent and sandbox, then preserves its verifier or RewardKit reward components.
|
|
64
|
+
|
|
65
|
+
### helpers
|
|
66
|
+
|
|
67
|
+
`benchmax.rewards` provides deterministic text helpers, model judges, rubrics, ranking, adaptive rubrics, and diversity scoring for `BaseEnv` and direct `Environment` implementations. see the [rewards guide](src/benchmax/rewards/README.md).
|
|
68
|
+
|
|
69
|
+
Harbor environments normally use their harness verifier and RewardKit instead of benchmax reward helpers.
|
|
70
|
+
|
|
71
|
+
## bundling
|
|
72
|
+
|
|
73
|
+
a bundle contains the environment class, its constructor arguments, the project-local source it needs, and its declared remote dependencies.
|
|
74
|
+
|
|
75
|
+
```text
|
|
76
|
+
environment class + constructor arguments
|
|
77
|
+
+ local source
|
|
78
|
+
+ dependency metadata
|
|
79
|
+
│
|
|
80
|
+
▼
|
|
81
|
+
portable bundle
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
benchmax creates the portable artifact so the same environment can be loaded outside the author's checkout. `castform` handles uploading the bundle, validating it remotely, and using it for training.
|
|
85
|
+
|
|
86
|
+
Remote runtimes install `pip_dependencies` without enabling prereleases globally. If a dependency graph needs a prerelease, list that package explicitly even when it would normally be transitive. For example, use `pip_dependencies=["parent-package==1.0.0", "transitive-package==2.0.0rc1"]`.
|
|
87
|
+
|
|
88
|
+
## further reading
|
|
89
|
+
|
|
90
|
+
- [base environment guide](src/benchmax/envs/base/README.md)
|
|
91
|
+
- [harbor environment guide](src/benchmax/envs/harbor/README.md)
|
|
92
|
+
- [reward helpers](src/benchmax/rewards/README.md)
|
|
93
|
+
- [examples](../../examples/)
|
|
94
|
+
- [development instructions](../../README.md#development)
|
|
95
|
+
|
|
96
|
+
apache 2.0 © 2026 CGFT Inc.
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "benchmax"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.2"
|
|
4
4
|
description = "Platform-independent runtime for grouped LLM environments"
|
|
5
5
|
readme = "README.md"
|
|
6
|
-
authors = [{ name = "
|
|
6
|
+
authors = [{ name = "benchmax Authors" }]
|
|
7
7
|
requires-python = "==3.12.*"
|
|
8
8
|
dependencies = [
|
|
9
9
|
"cloudpickle>=3.0.0",
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Explicit, call-time authentication for model requests.
|
|
2
|
+
|
|
3
|
+
benchmax defines only the runtime contract. Platform packages and execution
|
|
4
|
+
runtimes provide concrete credential sources and bind injected credentials.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import asyncio
|
|
10
|
+
import contextvars
|
|
11
|
+
import threading
|
|
12
|
+
from collections.abc import AsyncIterator, Iterator, Mapping
|
|
13
|
+
from contextlib import contextmanager
|
|
14
|
+
from contextvars import ContextVar
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Protocol, runtime_checkable
|
|
17
|
+
|
|
18
|
+
import httpx
|
|
19
|
+
from openai import AsyncOpenAI, OpenAI
|
|
20
|
+
|
|
21
|
+
__all__ = [
|
|
22
|
+
"InjectedAuth",
|
|
23
|
+
"ModelAuth",
|
|
24
|
+
"ModelRequestContext",
|
|
25
|
+
"RequestModelAuth",
|
|
26
|
+
"StaticBearerAuth",
|
|
27
|
+
"bind_model_auth",
|
|
28
|
+
"create_async_openai_client",
|
|
29
|
+
"create_openai_client",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class ModelRequestContext:
|
|
35
|
+
"""Identity of the model request about to be authorized."""
|
|
36
|
+
|
|
37
|
+
base_url: str
|
|
38
|
+
model: str
|
|
39
|
+
rollout_id: str
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@runtime_checkable
|
|
43
|
+
class ModelAuth(Protocol):
|
|
44
|
+
"""Return headers immediately before each model HTTP request."""
|
|
45
|
+
|
|
46
|
+
async def headers_for_request(
|
|
47
|
+
self,
|
|
48
|
+
context: ModelRequestContext,
|
|
49
|
+
) -> Mapping[str, str]: ...
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True, slots=True)
|
|
53
|
+
class StaticBearerAuth:
|
|
54
|
+
"""Explicit bearer authentication for providers with a stable API key."""
|
|
55
|
+
|
|
56
|
+
token: str = field(repr=False)
|
|
57
|
+
|
|
58
|
+
def __post_init__(self) -> None:
|
|
59
|
+
if not isinstance(self.token, str) or not self.token:
|
|
60
|
+
raise ValueError("bearer token must be a non-empty string")
|
|
61
|
+
|
|
62
|
+
async def headers_for_request(
|
|
63
|
+
self,
|
|
64
|
+
context: ModelRequestContext,
|
|
65
|
+
) -> Mapping[str, str]:
|
|
66
|
+
del context
|
|
67
|
+
return {"Authorization": f"Bearer {self.token}"}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class RequestModelAuth(httpx.Auth):
|
|
71
|
+
"""Apply a :class:`ModelAuth` immediately before each HTTP request.
|
|
72
|
+
|
|
73
|
+
Both sync and async OpenAI-compatible clients use this adapter, so model
|
|
74
|
+
credential selection never falls back to an SDK environment variable or a
|
|
75
|
+
separate token resolver. When a sync client runs inside an active event
|
|
76
|
+
loop, auth is resolved in a context-preserving helper thread; custom
|
|
77
|
+
providers used there must not depend on primitives bound to another loop.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
def __init__(self, auth: ModelAuth, context: ModelRequestContext) -> None:
|
|
81
|
+
if not isinstance(auth, ModelAuth):
|
|
82
|
+
raise TypeError("request auth must implement ModelAuth")
|
|
83
|
+
self._auth = auth
|
|
84
|
+
self._context = context
|
|
85
|
+
|
|
86
|
+
def sync_auth_flow(
|
|
87
|
+
self,
|
|
88
|
+
request: httpx.Request,
|
|
89
|
+
) -> Iterator[httpx.Request]:
|
|
90
|
+
headers = _resolve_headers_sync(self._auth, self._context)
|
|
91
|
+
for name, value in headers.items():
|
|
92
|
+
request.headers[name] = value
|
|
93
|
+
yield request
|
|
94
|
+
|
|
95
|
+
async def async_auth_flow(
|
|
96
|
+
self,
|
|
97
|
+
request: httpx.Request,
|
|
98
|
+
) -> AsyncIterator[httpx.Request]:
|
|
99
|
+
headers = await self._auth.headers_for_request(self._context)
|
|
100
|
+
for name, value in headers.items():
|
|
101
|
+
request.headers[name] = value
|
|
102
|
+
yield request
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def create_openai_client(
|
|
106
|
+
*,
|
|
107
|
+
model: str,
|
|
108
|
+
base_url: str,
|
|
109
|
+
auth: ModelAuth,
|
|
110
|
+
request_id: str,
|
|
111
|
+
max_retries: int = 2,
|
|
112
|
+
) -> OpenAI:
|
|
113
|
+
"""Create a synchronous OpenAI-compatible client with explicit auth."""
|
|
114
|
+
|
|
115
|
+
context = ModelRequestContext(
|
|
116
|
+
base_url=base_url,
|
|
117
|
+
model=model,
|
|
118
|
+
rollout_id=request_id,
|
|
119
|
+
)
|
|
120
|
+
return OpenAI(
|
|
121
|
+
base_url=base_url,
|
|
122
|
+
api_key="benchmax-explicit-auth",
|
|
123
|
+
http_client=httpx.Client(auth=RequestModelAuth(auth, context)),
|
|
124
|
+
max_retries=max_retries,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def create_async_openai_client(
|
|
129
|
+
*,
|
|
130
|
+
model: str,
|
|
131
|
+
base_url: str,
|
|
132
|
+
auth: ModelAuth,
|
|
133
|
+
request_id: str,
|
|
134
|
+
max_retries: int = 2,
|
|
135
|
+
) -> AsyncOpenAI:
|
|
136
|
+
"""Create an asynchronous OpenAI-compatible client with explicit auth."""
|
|
137
|
+
|
|
138
|
+
context = ModelRequestContext(
|
|
139
|
+
base_url=base_url,
|
|
140
|
+
model=model,
|
|
141
|
+
rollout_id=request_id,
|
|
142
|
+
)
|
|
143
|
+
return AsyncOpenAI(
|
|
144
|
+
base_url=base_url,
|
|
145
|
+
api_key="benchmax-explicit-auth",
|
|
146
|
+
http_client=httpx.AsyncClient(auth=RequestModelAuth(auth, context)),
|
|
147
|
+
max_retries=max_retries,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _resolve_headers_sync(
|
|
152
|
+
auth: ModelAuth,
|
|
153
|
+
context: ModelRequestContext,
|
|
154
|
+
) -> Mapping[str, str]:
|
|
155
|
+
"""Resolve async ``ModelAuth`` from a synchronous HTTP client.
|
|
156
|
+
|
|
157
|
+
RAG search backends expose synchronous embedding callables. When one is
|
|
158
|
+
invoked from an async environment tool, its event loop is already running;
|
|
159
|
+
resolve the auth coroutine in a context-preserving helper thread rather
|
|
160
|
+
than attempting a nested event loop.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
async def resolve() -> Mapping[str, str]:
|
|
164
|
+
return await auth.headers_for_request(context)
|
|
165
|
+
|
|
166
|
+
try:
|
|
167
|
+
asyncio.get_running_loop()
|
|
168
|
+
except RuntimeError:
|
|
169
|
+
return asyncio.run(resolve())
|
|
170
|
+
|
|
171
|
+
copied_context = contextvars.copy_context()
|
|
172
|
+
result: list[Mapping[str, str]] = []
|
|
173
|
+
failure: list[BaseException] = []
|
|
174
|
+
|
|
175
|
+
def run() -> None:
|
|
176
|
+
try:
|
|
177
|
+
result.append(copied_context.run(lambda: asyncio.run(resolve())))
|
|
178
|
+
except BaseException as error: # propagate the original auth failure
|
|
179
|
+
failure.append(error)
|
|
180
|
+
|
|
181
|
+
thread = threading.Thread(target=run, daemon=True)
|
|
182
|
+
thread.start()
|
|
183
|
+
thread.join()
|
|
184
|
+
if failure:
|
|
185
|
+
raise failure[0]
|
|
186
|
+
if not result:
|
|
187
|
+
raise RuntimeError("model authentication did not return headers")
|
|
188
|
+
return result[0]
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
_BOUND_MODEL_AUTH: ContextVar[Mapping[str, ModelAuth] | None] = ContextVar(
|
|
192
|
+
"benchmax_bound_model_auth",
|
|
193
|
+
default=None,
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
@dataclass(frozen=True, slots=True)
|
|
198
|
+
class InjectedAuth:
|
|
199
|
+
"""Serializable reference to authentication supplied by the runtime."""
|
|
200
|
+
|
|
201
|
+
name: str
|
|
202
|
+
|
|
203
|
+
def __post_init__(self) -> None:
|
|
204
|
+
if not isinstance(self.name, str) or not self.name.strip():
|
|
205
|
+
raise ValueError("injected auth name must be a non-empty string")
|
|
206
|
+
|
|
207
|
+
async def headers_for_request(
|
|
208
|
+
self,
|
|
209
|
+
context: ModelRequestContext,
|
|
210
|
+
) -> Mapping[str, str]:
|
|
211
|
+
providers = _BOUND_MODEL_AUTH.get()
|
|
212
|
+
provider = providers.get(self.name) if providers is not None else None
|
|
213
|
+
if provider is None:
|
|
214
|
+
raise RuntimeError(f"No runtime model-auth provider was injected for {self.name!r}.")
|
|
215
|
+
if isinstance(provider, InjectedAuth):
|
|
216
|
+
raise RuntimeError(
|
|
217
|
+
f"Injected model-auth provider {self.name!r} cannot reference another InjectedAuth."
|
|
218
|
+
)
|
|
219
|
+
return await provider.headers_for_request(context)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
@contextmanager
|
|
223
|
+
def bind_model_auth(providers: Mapping[str, ModelAuth]) -> Iterator[None]:
|
|
224
|
+
"""Bind runtime providers for the current async execution context."""
|
|
225
|
+
|
|
226
|
+
normalized = dict(providers)
|
|
227
|
+
for name, provider in normalized.items():
|
|
228
|
+
if not isinstance(name, str) or not name.strip():
|
|
229
|
+
raise ValueError("model-auth provider names must be non-empty strings")
|
|
230
|
+
if not isinstance(provider, ModelAuth):
|
|
231
|
+
raise TypeError(f"model-auth provider {name!r} does not implement ModelAuth")
|
|
232
|
+
token = _BOUND_MODEL_AUTH.set(normalized)
|
|
233
|
+
try:
|
|
234
|
+
yield
|
|
235
|
+
finally:
|
|
236
|
+
_BOUND_MODEL_AUTH.reset(token)
|