verifiers 0.2.2.dev42__py3-none-any.whl → 0.2.2.dev44__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +12 -7
- verifiers/v1/agent.py +9 -8
- verifiers/v1/cli/dashboard/eval.py +1 -1
- verifiers/v1/cli/eval/main.py +9 -7
- verifiers/v1/cli/eval/runner.py +12 -7
- verifiers/v1/cli/serve.py +17 -14
- verifiers/v1/configs/cli/env.py +16 -123
- verifiers/v1/configs/cli/eval.py +66 -6
- verifiers/v1/configs/cli/serve.py +53 -7
- verifiers/v1/configs/env.py +11 -4
- verifiers/v1/configs/legacy.py +24 -0
- verifiers/v1/configs/serve.py +63 -0
- verifiers/v1/env.py +24 -16
- verifiers/v1/gepa/config.py +3 -3
- verifiers/v1/legacy.py +12 -6
- verifiers/v1/push.py +1 -1
- verifiers/v1/serve/pool.py +8 -6
- verifiers/v1/serve/server.py +8 -6
- {verifiers-0.2.2.dev42.dist-info → verifiers-0.2.2.dev44.dist-info}/METADATA +2 -2
- {verifiers-0.2.2.dev42.dist-info → verifiers-0.2.2.dev44.dist-info}/RECORD +23 -21
- {verifiers-0.2.2.dev42.dist-info → verifiers-0.2.2.dev44.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev42.dist-info → verifiers-0.2.2.dev44.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev42.dist-info → verifiers-0.2.2.dev44.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -14,16 +14,18 @@ from verifiers.v1.clients import (
|
|
|
14
14
|
resolve_client,
|
|
15
15
|
)
|
|
16
16
|
from verifiers.v1.configs.agent import AgentConfig
|
|
17
|
-
from verifiers.v1.configs.cli.env import
|
|
18
|
-
ElasticPoolConfig,
|
|
19
|
-
EnvServerConfig,
|
|
20
|
-
StaticPoolConfig,
|
|
21
|
-
pool_serve_kwargs,
|
|
22
|
-
)
|
|
17
|
+
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
23
18
|
from verifiers.v1.configs.env import EnvConfig, default_agent_harness
|
|
24
19
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
25
20
|
from verifiers.v1.configs.judge import JudgeConfig, Judges
|
|
21
|
+
from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
26
22
|
from verifiers.v1.configs.retries import RetryConfig
|
|
23
|
+
from verifiers.v1.configs.serve import (
|
|
24
|
+
ElasticPoolConfig,
|
|
25
|
+
ServingConfig,
|
|
26
|
+
StaticPoolConfig,
|
|
27
|
+
pool_serve_kwargs,
|
|
28
|
+
)
|
|
27
29
|
from verifiers.v1.configs.task import TaskConfig
|
|
28
30
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
29
31
|
from verifiers.v1.decorators import metric, reward, stop, tool
|
|
@@ -248,7 +250,10 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
248
250
|
"Env",
|
|
249
251
|
"SingleAgentEnv",
|
|
250
252
|
"EnvConfig",
|
|
251
|
-
"
|
|
253
|
+
"ServingConfig",
|
|
254
|
+
"LegacyEnvConfig",
|
|
255
|
+
"resolve_env_field",
|
|
256
|
+
"narrowed_env_annotation",
|
|
252
257
|
"SingleAgentEnvConfig",
|
|
253
258
|
"AgentConfig",
|
|
254
259
|
"StaticPoolConfig",
|
verifiers/v1/agent.py
CHANGED
|
@@ -276,9 +276,9 @@ class Agent:
|
|
|
276
276
|
max_total_tokens=config.max_total_tokens,
|
|
277
277
|
)
|
|
278
278
|
self.timeout = config.timeout
|
|
279
|
-
# Env episode agents replace this with the
|
|
280
|
-
# Interactions acquire it only around active
|
|
281
|
-
# awaiting the caller between segments.
|
|
279
|
+
# Env episode agents replace this with the episode's agent semaphore
|
|
280
|
+
# (`--env.max-concurrent-agents`). Interactions acquire it only around active
|
|
281
|
+
# lifecycle work, never while awaiting the caller between segments.
|
|
282
282
|
self._gate: asyncio.Semaphore | None = None
|
|
283
283
|
# Env-owned standing, not config: `Env.setup` marks fixed agents
|
|
284
284
|
# untrainable and traces are stamped from here; inert outside an env.
|
|
@@ -578,9 +578,9 @@ class _EpisodeAgent(Agent):
|
|
|
578
578
|
bundle of references — expensive resources are env-owned and borrowed, so no
|
|
579
579
|
state spans concurrent episodes): traces get their agent standing the moment
|
|
580
580
|
they're created, finished ones land in `completed` (the episode's traces),
|
|
581
|
-
each run takes the
|
|
582
|
-
its own tasks — on an env-minted task they'd wrongly put MCP
|
|
583
|
-
(`tools=` overrides)."""
|
|
581
|
+
each run takes one of the episode's agent permits. The taskset's shared tool
|
|
582
|
+
servers ride only its own tasks — on an env-minted task they'd wrongly put MCP
|
|
583
|
+
in play (`tools=` overrides)."""
|
|
584
584
|
|
|
585
585
|
def __init__(
|
|
586
586
|
self,
|
|
@@ -664,8 +664,9 @@ class _EpisodeAgent(Agent):
|
|
|
664
664
|
"""The agent's `interaction`, with every trace stamped with its standing
|
|
665
665
|
at mint and captured in `completed` at close — an interaction driven from
|
|
666
666
|
`Env.run` stays crash-safe. Setup, each active segment, and close acquire
|
|
667
|
-
|
|
668
|
-
its caller, so peer interactions
|
|
667
|
+
an agent permit independently; the interaction holds no permit while awaiting
|
|
668
|
+
its caller, so peer interactions still interleave where an episode plays one
|
|
669
|
+
agent at a time."""
|
|
669
670
|
trace: Trace | None = None
|
|
670
671
|
|
|
671
672
|
def remember(current: Trace) -> None:
|
|
@@ -95,7 +95,7 @@ def _limits(config: EvalConfig) -> list[str]:
|
|
|
95
95
|
toks.append(f"{label}≤{v}")
|
|
96
96
|
turns = _seat_value(config, lambda spec: spec.max_turns)
|
|
97
97
|
return [
|
|
98
|
-
f"≤{config.max_concurrent}
|
|
98
|
+
f"≤{config.max_concurrent} episodes"
|
|
99
99
|
if config.max_concurrent
|
|
100
100
|
else "no concurrency cap",
|
|
101
101
|
"per-seat turn caps"
|
verifiers/v1/cli/eval/main.py
CHANGED
|
@@ -24,7 +24,7 @@ from verifiers.v1.utils.logging import setup_logging
|
|
|
24
24
|
logger = logging.getLogger(__name__)
|
|
25
25
|
|
|
26
26
|
USAGE = (
|
|
27
|
-
"usage: uv run eval [<taskset-id>] [--env.id <id>] [--id <env-id> (
|
|
27
|
+
"usage: uv run eval [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]\n"
|
|
28
28
|
" uv run eval --resume <output-dir> (re-run a previous run's missing/errored rollouts)"
|
|
29
29
|
)
|
|
30
30
|
|
|
@@ -49,12 +49,14 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
49
49
|
)
|
|
50
50
|
config = load_resume_config(resume_dir)
|
|
51
51
|
else:
|
|
52
|
-
legacy_id = any(
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
#
|
|
52
|
+
legacy_id = any(
|
|
53
|
+
a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv
|
|
54
|
+
)
|
|
55
|
+
# An env-block flag (or a since-moved flat axis) skips the usage gate so the
|
|
56
|
+
# typed parse renders its did-you-mean instead of a bare usage line.
|
|
56
57
|
typed_axis = any(
|
|
57
|
-
a.startswith(("--env.", "--taskset.", "--harness."))
|
|
58
|
+
a.startswith(("--env.", "--taskset.", "--harness.", "--serve."))
|
|
59
|
+
for a in argv
|
|
58
60
|
)
|
|
59
61
|
if (
|
|
60
62
|
not extract_id(argv, "env.taskset")
|
|
@@ -64,7 +66,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
64
66
|
):
|
|
65
67
|
raise SystemExit(
|
|
66
68
|
USAGE
|
|
67
|
-
) # need a taskset (positional / --env.taskset.id), a
|
|
69
|
+
) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or a @ file.toml
|
|
68
70
|
|
|
69
71
|
with plugin_errors():
|
|
70
72
|
config_type = narrow_config(EvalConfig, argv)
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -107,19 +107,24 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
107
107
|
import multiprocessing as mp
|
|
108
108
|
from functools import partial
|
|
109
109
|
|
|
110
|
-
from verifiers.v1.configs.
|
|
110
|
+
from verifiers.v1.configs.serve import pool_serve_kwargs
|
|
111
111
|
from verifiers.v1.serve import EnvClient, env_config_data, serve_env
|
|
112
112
|
from verifiers.v1.utils.logging import setup_logging
|
|
113
113
|
|
|
114
114
|
legacy = config.is_legacy
|
|
115
115
|
server_kwargs = (
|
|
116
116
|
{
|
|
117
|
-
"env_id": config.id,
|
|
118
|
-
"env_args": config.args,
|
|
119
|
-
"extra_env_kwargs": config.extra_env_kwargs,
|
|
117
|
+
"env_id": config.legacy.id,
|
|
118
|
+
"env_args": config.legacy.args,
|
|
119
|
+
"extra_env_kwargs": config.legacy.extra_env_kwargs,
|
|
120
120
|
}
|
|
121
121
|
if legacy
|
|
122
|
-
else {
|
|
122
|
+
else {
|
|
123
|
+
"config_data": env_config_data(config.env), # picklable across the spawn
|
|
124
|
+
# `-c` seeds each worker's episode bound unless `[serve]` pins one — so a
|
|
125
|
+
# pool carries `workers * bound` episodes, as `multiplex` implies.
|
|
126
|
+
"max_concurrent": config.worker_max_concurrent,
|
|
127
|
+
}
|
|
123
128
|
)
|
|
124
129
|
tasks = []
|
|
125
130
|
if not legacy:
|
|
@@ -142,7 +147,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
142
147
|
proc = mpctx.Process(
|
|
143
148
|
target=serve_env,
|
|
144
149
|
kwargs=dict(
|
|
145
|
-
**pool_serve_kwargs(config.pool),
|
|
150
|
+
**pool_serve_kwargs(config.serve.pool),
|
|
146
151
|
legacy=legacy,
|
|
147
152
|
address="tcp://127.0.0.1:0",
|
|
148
153
|
address_queue=address_queue,
|
|
@@ -210,7 +215,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
210
215
|
"running %dx%d rollouts via the env-server %s pool on %s",
|
|
211
216
|
len(plan),
|
|
212
217
|
config.num_rollouts,
|
|
213
|
-
config.pool.type,
|
|
218
|
+
config.serve.pool.type,
|
|
214
219
|
config.model,
|
|
215
220
|
)
|
|
216
221
|
logger.info("results: %s", out)
|
verifiers/v1/cli/serve.py
CHANGED
|
@@ -12,12 +12,12 @@ from verifiers.v1.cli.resolve import (
|
|
|
12
12
|
references_config_file,
|
|
13
13
|
with_positional_taskset,
|
|
14
14
|
)
|
|
15
|
-
from verifiers.v1.configs.cli.env import pool_serve_kwargs
|
|
16
15
|
from verifiers.v1.configs.cli.serve import ServeConfig
|
|
16
|
+
from verifiers.v1.configs.serve import pool_serve_kwargs
|
|
17
17
|
from verifiers.v1.serve import serve_env
|
|
18
18
|
from verifiers.v1.utils.logging import setup_logging
|
|
19
19
|
|
|
20
|
-
USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--id <env-id> (
|
|
20
|
+
USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]"
|
|
21
21
|
|
|
22
22
|
|
|
23
23
|
def main(argv: list[str] | None = None) -> None:
|
|
@@ -29,11 +29,12 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
29
29
|
with plugin_errors():
|
|
30
30
|
cli(narrow_config(ServeConfig, argv))
|
|
31
31
|
return
|
|
32
|
-
legacy_id = any(a == "--id" or a.startswith("--id=") for a in argv)
|
|
33
|
-
# An env-block flag (or a
|
|
34
|
-
# parse renders its did-you-mean
|
|
35
|
-
|
|
36
|
-
|
|
32
|
+
legacy_id = any(a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv)
|
|
33
|
+
# An env-block flag (or a since-moved flat axis) skips the usage gate so the typed
|
|
34
|
+
# parse renders its did-you-mean instead of a bare usage line.
|
|
35
|
+
typed_axis = any(
|
|
36
|
+
a.startswith(("--env.", "--taskset.", "--harness.", "--serve.")) for a in argv
|
|
37
|
+
)
|
|
37
38
|
if (
|
|
38
39
|
not extract_id(argv, "env.taskset")
|
|
39
40
|
and not legacy_id
|
|
@@ -42,7 +43,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
42
43
|
):
|
|
43
44
|
raise SystemExit(
|
|
44
45
|
USAGE
|
|
45
|
-
) # need a taskset (positional / --env.taskset.id), a
|
|
46
|
+
) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or @ file.toml
|
|
46
47
|
|
|
47
48
|
with plugin_errors():
|
|
48
49
|
config_type = narrow_config(ServeConfig, argv)
|
|
@@ -60,17 +61,19 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
60
61
|
# in each one.
|
|
61
62
|
server_kwargs = (
|
|
62
63
|
{
|
|
63
|
-
"env_id": config.id,
|
|
64
|
-
"env_args": config.args,
|
|
65
|
-
"extra_env_kwargs": config.extra_env_kwargs,
|
|
64
|
+
"env_id": config.legacy.id,
|
|
65
|
+
"env_args": config.legacy.args,
|
|
66
|
+
"extra_env_kwargs": config.legacy.extra_env_kwargs,
|
|
66
67
|
}
|
|
67
68
|
if config.is_legacy
|
|
68
|
-
|
|
69
|
+
# `--serve.max-concurrent` is each v1 worker's episode bound; the legacy
|
|
70
|
+
# bridge has never had one.
|
|
71
|
+
else {"config": config.env, "max_concurrent": config.serve.max_concurrent}
|
|
69
72
|
)
|
|
70
73
|
serve_env(
|
|
71
|
-
**pool_serve_kwargs(config.pool),
|
|
74
|
+
**pool_serve_kwargs(config.serve.pool),
|
|
72
75
|
legacy=config.is_legacy,
|
|
73
|
-
address=config.address,
|
|
76
|
+
address=config.serve.address,
|
|
74
77
|
log_setup=partial(setup_logging, level),
|
|
75
78
|
**server_kwargs,
|
|
76
79
|
)
|
verifiers/v1/configs/cli/env.py
CHANGED
|
@@ -1,35 +1,30 @@
|
|
|
1
|
-
"""Run-config plumbing around the `[env]` block: narrowing the `env` field of
|
|
2
|
-
|
|
1
|
+
"""Run-config plumbing around the `[env]` block: narrowing the `env` field of every
|
|
2
|
+
config that owns one, and the retired keys such a config refuses.
|
|
3
3
|
|
|
4
|
-
|
|
4
|
+
A run composes the blocks it needs — `[env]` (what runs, `configs/env.py`),
|
|
5
|
+
`[serve]` (how it's hosted, `configs/serve.py`), `[legacy]` (the v0 bridge,
|
|
6
|
+
`configs/legacy.py`) — plus its own fields. Nothing here is a base class: the eval
|
|
7
|
+
CLI, the `serve` CLI, GEPA and a trainer each declare their blocks and call these.
|
|
5
8
|
|
|
6
|
-
|
|
9
|
+
Declare the env field as `SerializeAsAny[EnvConfig] = Field(default_factory=
|
|
10
|
+
single_agent_env_config)`. The `SerializeAsAny` is load-bearing: pydantic serializes
|
|
11
|
+
by declared type, so a plain `EnvConfig` silently drops a narrowed subclass's agents
|
|
12
|
+
and knobs from `model_dump()` — the env-server wire's payload."""
|
|
13
|
+
|
|
14
|
+
from pydantic import ValidationError
|
|
7
15
|
from pydantic_config import BaseConfig
|
|
8
16
|
|
|
9
17
|
from verifiers.v1.configs.env import EnvConfig
|
|
10
|
-
from verifiers.v1.types import ID
|
|
11
18
|
from verifiers.v1.utils.generic import prefix_validation_error
|
|
12
19
|
|
|
13
20
|
|
|
14
21
|
def resolve_env_field(data: dict, narrowed: "type[EnvConfig] | None" = None) -> dict:
|
|
15
|
-
"""Shared `mode="before"` body for every run config owning an `env` field
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
authoritative, so validate against it directly."""
|
|
22
|
+
"""Shared `mode="before"` body for every run config owning an `env` field: narrow
|
|
23
|
+
`env` to the concrete env's config class. `narrowed` is the annotation the CLI
|
|
24
|
+
pre-resolved (`narrow_config`) — its id is authoritative, so validate against it
|
|
25
|
+
directly."""
|
|
20
26
|
if not isinstance(data, dict):
|
|
21
27
|
return data
|
|
22
|
-
if "taskset" in data:
|
|
23
|
-
raise ValueError(
|
|
24
|
-
"the taskset lives on the env now: --env.taskset.id <id> "
|
|
25
|
-
"(TOML: [env.taskset]), or the positional `eval <taskset-id>`"
|
|
26
|
-
)
|
|
27
|
-
if "harness" in data:
|
|
28
|
-
raise ValueError(
|
|
29
|
-
"a harness belongs to an agent now: --env.agent.harness.* on the "
|
|
30
|
-
"single-agent env, --env.<agent>.harness.* on a multi-agent one "
|
|
31
|
-
"(TOML: [env.agent.harness])"
|
|
32
|
-
)
|
|
33
28
|
raw = data.get("env")
|
|
34
29
|
if raw is None:
|
|
35
30
|
return data
|
|
@@ -63,105 +58,3 @@ def narrowed_env_annotation(cls) -> "type[EnvConfig] | None":
|
|
|
63
58
|
):
|
|
64
59
|
return annotation
|
|
65
60
|
return None
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
class StaticPoolConfig(BaseConfig):
|
|
69
|
-
"""Fixed env-server pool: pre-spawn `num_workers` workers up front."""
|
|
70
|
-
|
|
71
|
-
type: Literal["static"] = "static"
|
|
72
|
-
num_workers: int = Field(4, ge=1)
|
|
73
|
-
"""Worker processes to pre-spawn (1 = a single in-process server, no pool)."""
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
class ElasticPoolConfig(BaseConfig):
|
|
77
|
-
"""Elastic env-server pool: start at one worker and scale up on demand."""
|
|
78
|
-
|
|
79
|
-
type: Literal["elastic"] = "elastic"
|
|
80
|
-
max_workers: int | None = None
|
|
81
|
-
"""Upper bound on workers (None = unbounded)."""
|
|
82
|
-
multiplex: int = Field(128, ge=1)
|
|
83
|
-
"""Rollouts per worker for the scale-up trigger: add a worker once in-flight rollouts
|
|
84
|
-
reach 90% of `workers * multiplex`."""
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
# Discriminated on `type` so the CLI selects with `--pool.type static|elastic`.
|
|
88
|
-
PoolConfig = Annotated[
|
|
89
|
-
StaticPoolConfig | ElasticPoolConfig, Field(discriminator="type")
|
|
90
|
-
]
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
def pool_serve_kwargs(pool: StaticPoolConfig | ElasticPoolConfig) -> dict:
|
|
94
|
-
"""Unpack a pool config into `serve_env` kwargs (`max_workers` / `multiplex` / `elastic`)."""
|
|
95
|
-
if isinstance(pool, ElasticPoolConfig):
|
|
96
|
-
return {
|
|
97
|
-
"max_workers": pool.max_workers,
|
|
98
|
-
"multiplex": pool.multiplex,
|
|
99
|
-
"elastic": True,
|
|
100
|
-
}
|
|
101
|
-
return {"max_workers": pool.num_workers, "elastic": False}
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
def _single_agent_env_config() -> EnvConfig:
|
|
105
|
-
"""The default `env` block: the single-agent shape. Lazy — the concrete env
|
|
106
|
-
lives in `envs/`, which imports `env.py`."""
|
|
107
|
-
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
108
|
-
|
|
109
|
-
return SingleAgentEnvConfig()
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
class EnvServerConfig(BaseConfig):
|
|
113
|
-
"""A run's environment plus how it's *served*: the `env` block and the worker-pool
|
|
114
|
-
sizing. Shared by the `serve` CLI, server-backed eval, and prime-rl's orchestrator, so
|
|
115
|
-
they all configure the pool the same way (`--pool.type elastic|static`)."""
|
|
116
|
-
|
|
117
|
-
# SerializeAsAny: see EnvConfig.taskset — model_dump() must keep the subclass's
|
|
118
|
-
# agent fields and knobs.
|
|
119
|
-
env: SerializeAsAny[EnvConfig] = Field(default_factory=_single_agent_env_config)
|
|
120
|
-
"""The environment — the run's `[env]` block: which env, its seed taskset, each
|
|
121
|
-
agent, its knobs, and the run limits. Narrowed to the selected env's config
|
|
122
|
-
class by the env id, else the taskset id."""
|
|
123
|
-
pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
|
|
124
|
-
"""Worker-pool sizing for the env server. `elastic` (default) starts at one worker and
|
|
125
|
-
scales up on demand; `static` pre-spawns a fixed `num_workers`."""
|
|
126
|
-
# --- legacy (v0) backwards-compat -----------------------------------------
|
|
127
|
-
id: ID | None = None
|
|
128
|
-
"""Classic (v0) env id (`name`, `org/name`, or `org/name@version` — installed from the
|
|
129
|
-
hub on demand), loaded via `verifiers.load_environment` and run through the legacy
|
|
130
|
-
bridge. Set this *instead of* `env.taskset` to run a v0 environment."""
|
|
131
|
-
args: dict = Field(default_factory=dict)
|
|
132
|
-
"""Construction kwargs forwarded to `load_environment(id, **args)`."""
|
|
133
|
-
extra_env_kwargs: dict = Field(default_factory=dict)
|
|
134
|
-
"""Post-load kwargs applied to the v0 env via `env.set_kwargs(**extra_env_kwargs)` (e.g.
|
|
135
|
-
`max_total_completion_tokens`, `max_seq_len`, `timeout_seconds`) — typically
|
|
136
|
-
auto-populated by the orchestrator, distinct from the `args` passed at construction."""
|
|
137
|
-
|
|
138
|
-
@property
|
|
139
|
-
def is_legacy(self) -> bool:
|
|
140
|
-
"""A v0/legacy env (run via the bridge): a legacy `id` is set and no v1 taskset."""
|
|
141
|
-
return self.id is not None and not self.env.taskset.id
|
|
142
|
-
|
|
143
|
-
@property
|
|
144
|
-
def env_id(self) -> str:
|
|
145
|
-
"""The run's identifier: the v1 env's (`EnvConfig.env_id`), else the legacy
|
|
146
|
-
v0 env id."""
|
|
147
|
-
return self.env.env_id or self.id or ""
|
|
148
|
-
|
|
149
|
-
@model_validator(mode="after")
|
|
150
|
-
def _refuse_legacy_id_with_taskset(self):
|
|
151
|
-
"""A legacy `id` next to a v1 `env.taskset` would be silently inert
|
|
152
|
-
(`is_legacy` is False and the v0 env never loads); refuse the mix."""
|
|
153
|
-
if self.id is not None and self.env.taskset.id:
|
|
154
|
-
raise ValueError(
|
|
155
|
-
f"--id {self.id!r} is the legacy (v0) env id and can't combine with "
|
|
156
|
-
f"the v1 taskset {self.env.taskset.id!r}. Pairing an env with a "
|
|
157
|
-
f"taskset is --env.id {self.id!r} (TOML: id under [env]); to run the "
|
|
158
|
-
"v0 env instead, drop the taskset."
|
|
159
|
-
)
|
|
160
|
-
return self
|
|
161
|
-
|
|
162
|
-
# --- end legacy -----------------------------------------------------------
|
|
163
|
-
|
|
164
|
-
@model_validator(mode="before")
|
|
165
|
-
@classmethod
|
|
166
|
-
def _resolve_env(cls, data):
|
|
167
|
-
return resolve_env_field(data, narrowed_env_annotation(cls))
|
verifiers/v1/configs/cli/eval.py
CHANGED
|
@@ -3,14 +3,27 @@
|
|
|
3
3
|
from pathlib import Path
|
|
4
4
|
from uuid import uuid4
|
|
5
5
|
|
|
6
|
-
from pydantic import AliasChoices, Field, model_validator
|
|
6
|
+
from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
7
|
+
from pydantic_config import BaseConfig
|
|
7
8
|
|
|
8
9
|
from verifiers.v1.clients import ClientConfig, EvalClientConfig
|
|
9
|
-
from verifiers.v1.configs.cli.env import
|
|
10
|
+
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
11
|
+
from verifiers.v1.configs.env import EnvConfig
|
|
12
|
+
from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
13
|
+
from verifiers.v1.configs.serve import ServingConfig
|
|
14
|
+
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
10
15
|
from verifiers.v1.types import SamplingConfig
|
|
11
16
|
|
|
12
17
|
|
|
13
|
-
class EvalConfig(
|
|
18
|
+
class EvalConfig(BaseConfig):
|
|
19
|
+
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
|
20
|
+
"""The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
|
|
21
|
+
the selected env's config class by the env id, else the taskset id."""
|
|
22
|
+
serve: ServingConfig = ServingConfig()
|
|
23
|
+
"""How the env is hosted under `--server`: the worker pool, each worker's episode
|
|
24
|
+
bound. Ignored by an in-process run."""
|
|
25
|
+
legacy: LegacyEnvConfig = LegacyEnvConfig()
|
|
26
|
+
"""A classic (v0) environment to evaluate through the bridge instead of `[env]`."""
|
|
14
27
|
uuid: str = Field(default_factory=lambda: str(uuid4()), exclude=True)
|
|
15
28
|
"""Auto-generated run id — the leaf of the output dir, so runs never overwrite.
|
|
16
29
|
Excluded from the saved config so re-running `@ config.toml` lands in a fresh dir."""
|
|
@@ -37,9 +50,12 @@ class EvalConfig(EnvServerConfig):
|
|
|
37
50
|
shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
|
|
38
51
|
"""Shuffle tasks before taking the first `num_tasks`."""
|
|
39
52
|
max_concurrent: int | None = Field(
|
|
40
|
-
128, validation_alias=AliasChoices("max_concurrent", "c")
|
|
53
|
+
128, ge=1, validation_alias=AliasChoices("max_concurrent", "c")
|
|
41
54
|
)
|
|
42
|
-
"""
|
|
55
|
+
"""Episodes in flight at once, `None` for no limit. An episode plays its agents one
|
|
56
|
+
at a time, so this is the live agent runs too — until `--env.max-concurrent-agents`
|
|
57
|
+
says otherwise. Under `--server` it seeds each worker's bound, unless
|
|
58
|
+
`--serve.max-concurrent` pins one."""
|
|
43
59
|
verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
|
|
44
60
|
"""Log at debug level instead of the default info."""
|
|
45
61
|
dry_run: bool = Field(False, exclude=True)
|
|
@@ -49,7 +65,7 @@ class EvalConfig(EnvServerConfig):
|
|
|
49
65
|
"""Show a live dashboard instead of per-rollout logs (in-process only; an unset
|
|
50
66
|
`rich` defaults off under `--server`)."""
|
|
51
67
|
server: bool = False
|
|
52
|
-
"""Drive rollouts through the env-server worker pool (sized by
|
|
68
|
+
"""Drive rollouts through the env-server worker pool (sized by `[serve]`) instead of
|
|
53
69
|
in-process — the path prime-rl trains through. Incompatible with `--rich`."""
|
|
54
70
|
push: bool = True
|
|
55
71
|
"""Upload the finished run to the Prime Intellect platform (the private Evaluations
|
|
@@ -80,3 +96,47 @@ class EvalConfig(EnvServerConfig):
|
|
|
80
96
|
"`--server`; drop `--rich`."
|
|
81
97
|
)
|
|
82
98
|
return self
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def is_legacy(self) -> bool:
|
|
102
|
+
"""Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
|
|
103
|
+
return self.legacy.id is not None and not self.env.taskset.id
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def env_id(self) -> str:
|
|
107
|
+
"""The run's identifier: the v1 env's, else the v0 env id."""
|
|
108
|
+
return self.env.env_id or self.legacy.id or ""
|
|
109
|
+
|
|
110
|
+
@property
|
|
111
|
+
def worker_max_concurrent(self) -> int | None:
|
|
112
|
+
"""A served worker's episode bound: its own pin, else the run's `--max-concurrent`."""
|
|
113
|
+
return (
|
|
114
|
+
self.serve.max_concurrent
|
|
115
|
+
if self.serve.max_concurrent is not None
|
|
116
|
+
else self.max_concurrent
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
@model_validator(mode="after")
|
|
120
|
+
def _refuse_mixed_run(self):
|
|
121
|
+
# A v0 id next to any v1 env identity leaves one of the two going nowhere, and
|
|
122
|
+
# which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
|
|
123
|
+
# loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
|
|
124
|
+
if self.legacy.id is None or not self.env.env_id:
|
|
125
|
+
return self
|
|
126
|
+
if self.env.taskset.id:
|
|
127
|
+
raise ValueError(
|
|
128
|
+
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
|
|
129
|
+
f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
|
|
130
|
+
f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
|
|
131
|
+
"the v0 env instead, drop the taskset."
|
|
132
|
+
)
|
|
133
|
+
raise ValueError(
|
|
134
|
+
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
|
|
135
|
+
f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
|
|
136
|
+
"with the v1 env's name. Keep whichever one you meant to run."
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
@model_validator(mode="before")
|
|
140
|
+
@classmethod
|
|
141
|
+
def _resolve_env(cls, data):
|
|
142
|
+
return resolve_env_field(data, narrowed_env_annotation(cls))
|
|
@@ -1,16 +1,62 @@
|
|
|
1
1
|
"""Environment-server CLI configuration."""
|
|
2
2
|
|
|
3
|
-
from pydantic import AliasChoices, Field
|
|
3
|
+
from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
4
|
+
from pydantic_config import BaseConfig
|
|
4
5
|
|
|
5
|
-
from verifiers.v1.configs.cli.env import
|
|
6
|
+
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
7
|
+
from verifiers.v1.configs.env import EnvConfig
|
|
8
|
+
from verifiers.v1.configs.legacy import LegacyEnvConfig
|
|
9
|
+
from verifiers.v1.configs.serve import ServingConfig
|
|
10
|
+
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
6
11
|
|
|
7
12
|
|
|
8
|
-
class ServeConfig(
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
+
class ServeConfig(BaseConfig):
|
|
14
|
+
"""`uv run serve`: what to serve (`[env]`, or `[legacy]` for a classic v0 env) and
|
|
15
|
+
how it's hosted (`[serve]`)."""
|
|
16
|
+
|
|
17
|
+
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
|
18
|
+
"""The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
|
|
19
|
+
the selected env's config class by the env id, else the taskset id."""
|
|
20
|
+
serve: ServingConfig = ServingConfig()
|
|
21
|
+
"""How it's served: the worker pool, the bind address, each worker's episode bound."""
|
|
22
|
+
legacy: LegacyEnvConfig = LegacyEnvConfig()
|
|
23
|
+
"""A classic (v0) environment to serve through the bridge instead of `[env]`."""
|
|
13
24
|
verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
|
|
14
25
|
"""Log at debug level instead of info."""
|
|
15
26
|
dry_run: bool = False
|
|
16
27
|
"""Resolve + validate the config and dump it, then exit."""
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def is_legacy(self) -> bool:
|
|
31
|
+
"""Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
|
|
32
|
+
return self.legacy.id is not None and not self.env.taskset.id
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def env_id(self) -> str:
|
|
36
|
+
"""The run's identifier: the v1 env's, else the v0 env id."""
|
|
37
|
+
return self.env.env_id or self.legacy.id or ""
|
|
38
|
+
|
|
39
|
+
@model_validator(mode="after")
|
|
40
|
+
def _refuse_mixed_run(self):
|
|
41
|
+
# A v0 id next to any v1 env identity leaves one of the two going nowhere, and
|
|
42
|
+
# which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
|
|
43
|
+
# loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
|
|
44
|
+
if self.legacy.id is None or not self.env.env_id:
|
|
45
|
+
return self
|
|
46
|
+
if self.env.taskset.id:
|
|
47
|
+
raise ValueError(
|
|
48
|
+
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
|
|
49
|
+
f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
|
|
50
|
+
f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
|
|
51
|
+
"the v0 env instead, drop the taskset."
|
|
52
|
+
)
|
|
53
|
+
raise ValueError(
|
|
54
|
+
f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
|
|
55
|
+
f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
|
|
56
|
+
"with the v1 env's name. Keep whichever one you meant to run."
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
@model_validator(mode="before")
|
|
60
|
+
@classmethod
|
|
61
|
+
def _resolve_env(cls, data):
|
|
62
|
+
return resolve_env_field(data, narrowed_env_annotation(cls))
|
verifiers/v1/configs/env.py
CHANGED
|
@@ -3,7 +3,7 @@ each role as an `AgentConfig` field, and the env-level knobs."""
|
|
|
3
3
|
|
|
4
4
|
from typing import get_args
|
|
5
5
|
|
|
6
|
-
from pydantic import SerializeAsAny, model_validator
|
|
6
|
+
from pydantic import Field, SerializeAsAny, model_validator
|
|
7
7
|
from pydantic_config import BaseConfig
|
|
8
8
|
|
|
9
9
|
from verifiers.v1.configs.agent import AgentConfig
|
|
@@ -48,9 +48,16 @@ class EnvConfig(BaseConfig):
|
|
|
48
48
|
retries: RetryConfig = RetryConfig()
|
|
49
49
|
"""Whole-EPISODE retries — the coarse fallback for faults no agent owns; a
|
|
50
50
|
retried episode reruns whole (a half-played sibling context isn't reproducible)."""
|
|
51
|
-
|
|
52
|
-
"""
|
|
53
|
-
|
|
51
|
+
max_concurrent_agents: int | None = Field(1, ge=1)
|
|
52
|
+
"""How many of ONE episode's agent runs may be active at once (None = no limit).
|
|
53
|
+
One at a time by default, so the bound above — `--max-concurrent` episodes, a
|
|
54
|
+
dispatched training episode — is also one live agent run, whatever `run()` fans
|
|
55
|
+
out to internally. Raise it where per-episode latency matters and the
|
|
56
|
+
multiplication is wanted: `--env.max-concurrent-agents None` plays an episode's
|
|
57
|
+
`best-of-n` attempts together, so `-c` episodes carry `-c * n` live runs. Not
|
|
58
|
+
spelled `max_concurrent`: that key used to bound a served worker's agent runs
|
|
59
|
+
across episodes, and taking it as this one silently multiplies it by the episodes
|
|
60
|
+
in flight."""
|
|
54
61
|
interception: InterceptionConfig = ElasticInterceptionPoolConfig()
|
|
55
62
|
"""The interception shape: `elastic` (default), `server`, or `static`."""
|
|
56
63
|
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""The `[legacy]` block: running a classic (v0) environment through the bridge.
|
|
2
|
+
|
|
3
|
+
Quarantined in its own block, not mixed into `[env]` — a v0 env is a different
|
|
4
|
+
thing to run, not a variant of a v1 env's config."""
|
|
5
|
+
|
|
6
|
+
from pydantic import Field
|
|
7
|
+
from pydantic_config import BaseConfig
|
|
8
|
+
|
|
9
|
+
from verifiers.v1.types import ID
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class LegacyEnvConfig(BaseConfig):
|
|
13
|
+
"""A classic (v0) `verifiers` environment, loaded via `verifiers.load_environment`
|
|
14
|
+
and run through the legacy bridge. Set `id` *instead of* a v1 `env.taskset`."""
|
|
15
|
+
|
|
16
|
+
id: ID | None = None
|
|
17
|
+
"""v0 env id: `name`, `org/name`, or `org/name@version` — installed from the hub on
|
|
18
|
+
demand."""
|
|
19
|
+
args: dict = Field(default_factory=dict)
|
|
20
|
+
"""Construction kwargs forwarded to `load_environment(id, **args)`."""
|
|
21
|
+
extra_env_kwargs: dict = Field(default_factory=dict)
|
|
22
|
+
"""Post-load kwargs applied via `env.set_kwargs(**extra_env_kwargs)` (e.g.
|
|
23
|
+
`max_total_completion_tokens`, `max_seq_len`, `timeout_seconds`) — typically
|
|
24
|
+
auto-populated by a trainer, distinct from the `args` passed at construction."""
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""How an env is served: the worker pool, where it binds, and each worker's bound.
|
|
2
|
+
|
|
3
|
+
Serving is its own axis. `[env]` says what runs; this says how it's hosted, so an
|
|
4
|
+
eval, a `serve` process, and a trainer configure it identically instead of each
|
|
5
|
+
flattening pool knobs onto its own config."""
|
|
6
|
+
|
|
7
|
+
from typing import Annotated, Literal
|
|
8
|
+
|
|
9
|
+
from pydantic import Field
|
|
10
|
+
from pydantic_config import BaseConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class StaticPoolConfig(BaseConfig):
|
|
14
|
+
"""Fixed env-server pool: pre-spawn `num_workers` workers up front."""
|
|
15
|
+
|
|
16
|
+
type: Literal["static"] = "static"
|
|
17
|
+
num_workers: int = Field(4, ge=1)
|
|
18
|
+
"""Worker processes to pre-spawn (1 = a single in-process server, no pool)."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class ElasticPoolConfig(BaseConfig):
|
|
22
|
+
"""Elastic env-server pool: start at one worker and scale up on demand."""
|
|
23
|
+
|
|
24
|
+
type: Literal["elastic"] = "elastic"
|
|
25
|
+
max_workers: int | None = None
|
|
26
|
+
"""Upper bound on workers (None = unbounded)."""
|
|
27
|
+
multiplex: int = Field(128, ge=1)
|
|
28
|
+
"""Episodes per worker for the scale-up trigger: add a worker once in-flight episodes
|
|
29
|
+
reach 90% of `workers * multiplex`."""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# Discriminated on `type` so the CLI selects with `--serve.pool.type static|elastic`.
|
|
33
|
+
PoolConfig = Annotated[
|
|
34
|
+
StaticPoolConfig | ElasticPoolConfig, Field(discriminator="type")
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ServingConfig(BaseConfig):
|
|
39
|
+
"""The `[serve]` block: the worker pool, the ZMQ address, and each worker's
|
|
40
|
+
episode bound. Read by whoever hosts the env — the `serve` CLI, a server-backed
|
|
41
|
+
eval, a trainer's orchestrator."""
|
|
42
|
+
|
|
43
|
+
pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
|
|
44
|
+
"""Worker-pool sizing. `elastic` (default) starts at one worker and scales up on
|
|
45
|
+
demand; `static` pre-spawns a fixed `num_workers`."""
|
|
46
|
+
address: str = "tcp://127.0.0.1:5000"
|
|
47
|
+
"""ZMQ address the ROUTER binds (and clients connect to)."""
|
|
48
|
+
max_concurrent: int | None = Field(None, ge=1)
|
|
49
|
+
"""Episodes in flight per worker (None = take the run's own bound, e.g. an eval's
|
|
50
|
+
`--max-concurrent`). Pin it to hold a worker below what the run asks for; how many
|
|
51
|
+
agent runs one episode carries is the env's own
|
|
52
|
+
(`--env.max-concurrent-agents`, one at a time by default)."""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def pool_serve_kwargs(pool: StaticPoolConfig | ElasticPoolConfig) -> dict:
|
|
56
|
+
"""Unpack a pool config into `serve_env` kwargs (`max_workers` / `multiplex` / `elastic`)."""
|
|
57
|
+
if isinstance(pool, ElasticPoolConfig):
|
|
58
|
+
return {
|
|
59
|
+
"max_workers": pool.max_workers,
|
|
60
|
+
"multiplex": pool.multiplex,
|
|
61
|
+
"elastic": True,
|
|
62
|
+
}
|
|
63
|
+
return {"max_workers": pool.num_workers, "elastic": False}
|
verifiers/v1/env.py
CHANGED
|
@@ -154,7 +154,10 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
154
154
|
"""One episode: how the agents interact on `task`, returning nothing —
|
|
155
155
|
every finished run joins the episode automatically, stamped with its seat's
|
|
156
156
|
standing. An agent-run failure is data on its trace (this hook decides what
|
|
157
|
-
it means); an exception raised here is the episode itself failing.
|
|
157
|
+
it means); an exception raised here is the episode itself failing.
|
|
158
|
+
Independent agents are written as such (`asyncio.gather`); how many of them
|
|
159
|
+
actually run at once is the run's bound, not this hook's
|
|
160
|
+
(`--env.max-concurrent-agents`, one at a time by default)."""
|
|
158
161
|
|
|
159
162
|
async def finalize(self, task: Task, episode: Episode) -> None:
|
|
160
163
|
"""Cross-agent judgement — THE programmable judgement surface: plain
|
|
@@ -188,13 +191,16 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
188
191
|
def _episode_agents(
|
|
189
192
|
self,
|
|
190
193
|
ctx: ModelContext,
|
|
191
|
-
gate: "asyncio.Semaphore | None",
|
|
192
194
|
completed: list[Trace],
|
|
193
195
|
on_trace: Callable[[Trace], None] | None,
|
|
194
196
|
on_discard: Callable[[Trace], None] | None,
|
|
195
197
|
) -> Agents:
|
|
196
198
|
"""One episode's `Agents` — fresh value objects riding the live serving
|
|
197
|
-
resources (nothing shared across concurrent episodes)
|
|
199
|
+
resources (nothing shared across concurrent episodes), sharing one semaphore
|
|
200
|
+
so the episode plays `--env.max-concurrent-agents` of them at a time;
|
|
201
|
+
`setup()` sees them first."""
|
|
202
|
+
limit = self.config.max_concurrent_agents
|
|
203
|
+
gate = asyncio.Semaphore(limit) if limit else None
|
|
198
204
|
|
|
199
205
|
def make(name: str, spec: AgentConfig) -> Agent:
|
|
200
206
|
# Unpinned fields fall back to the run's ctx / the taskset's harness.
|
|
@@ -242,17 +248,17 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
242
248
|
*,
|
|
243
249
|
on_trace: Callable[[Trace], None] | None = None,
|
|
244
250
|
on_discard: Callable[[Trace], None] | None = None,
|
|
245
|
-
gate: asyncio.Semaphore | None = None,
|
|
246
251
|
) -> Episode:
|
|
247
252
|
"""One episode of `task`, minted as the wire atom: `setup()` then `run()`
|
|
248
|
-
over fresh agents, then `finalize()
|
|
249
|
-
|
|
253
|
+
over fresh agents, then `finalize()`. Its agents run one at a time
|
|
254
|
+
(`--env.max-concurrent-agents`), so one live episode is one live agent run
|
|
255
|
+
however `run()` fans out.
|
|
250
256
|
Traces join the episode as runs complete — a hook raising mid-way yields the
|
|
251
257
|
completed subset, its exception on the episode's `errors`. `on_trace` observes
|
|
252
258
|
each agent-run's trace at mint; `on_discard` its abandonment (a per-agent
|
|
253
259
|
retry mints a replacement)."""
|
|
254
260
|
episode = Episode(env=self.config.env_id)
|
|
255
|
-
agents = self._episode_agents(ctx,
|
|
261
|
+
agents = self._episode_agents(ctx, episode.traces, on_trace, on_discard)
|
|
256
262
|
try:
|
|
257
263
|
async with asyncio.timeout(self.config.timeout.episode):
|
|
258
264
|
async with boundary(EnvError, f"{type(self).__name__}.setup()"):
|
|
@@ -306,8 +312,10 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
306
312
|
on_complete: Callable[[Episode], Awaitable[None]] | None = None,
|
|
307
313
|
) -> Episode:
|
|
308
314
|
"""Run one planned episode to completion, with whole-episode
|
|
309
|
-
retries per `--env.retries`; `semaphore`
|
|
310
|
-
|
|
315
|
+
retries per `--env.retries`; `semaphore` bounds concurrent EPISODES — one
|
|
316
|
+
permit for the attempt in flight, held across the whole of it (its agents,
|
|
317
|
+
their boxes, `finalize()`) and released before a retry's backoff and before
|
|
318
|
+
`on_complete` (the runners' persistence hook, which fires when final)."""
|
|
311
319
|
|
|
312
320
|
async def attempt() -> Episode:
|
|
313
321
|
slot.traces = [] # a retry shows the fresh attempt's traces
|
|
@@ -318,13 +326,13 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
318
326
|
with contextlib.suppress(ValueError):
|
|
319
327
|
live.remove(trace)
|
|
320
328
|
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
329
|
+
async with semaphore or contextlib.nullcontext():
|
|
330
|
+
return await self.run_episode(
|
|
331
|
+
slot.task,
|
|
332
|
+
ctx,
|
|
333
|
+
on_trace=live.append,
|
|
334
|
+
on_discard=discard,
|
|
335
|
+
)
|
|
328
336
|
|
|
329
337
|
episode = await run_episode_with_retry(attempt, self.config.retries)
|
|
330
338
|
slot.traces = list(episode.traces)
|
verifiers/v1/gepa/config.py
CHANGED
|
@@ -4,9 +4,9 @@ GEPA optimizes one taskset's `Task.system_prompt` by alternating rollouts (`eval
|
|
|
4
4
|
teacher LM reflecting on the reflective dataset (`make_reflective_dataset`) — see
|
|
5
5
|
`verifiers.v1.gepa.adapter.GEPAAdapter`. Like `EvalConfig`, it owns an `env` field (the
|
|
6
6
|
environment: its taskset, seats, limits) and adds the optimization loop's own knobs (model,
|
|
7
|
-
reflection model, train/val split, budget). There is no
|
|
8
|
-
|
|
9
|
-
|
|
7
|
+
reflection model, train/val split, budget). There is no `[serve]` block here — GEPA
|
|
8
|
+
always runs in-process, since its adapter protocol is itself synchronous (see
|
|
9
|
+
`GEPAAdapter`)."""
|
|
10
10
|
|
|
11
11
|
from pathlib import Path
|
|
12
12
|
from uuid import uuid4
|
verifiers/v1/legacy.py
CHANGED
|
@@ -495,7 +495,7 @@ def _legacy_output_dir(config) -> Path:
|
|
|
495
495
|
|
|
496
496
|
if config.output_dir is not None:
|
|
497
497
|
return config.output_dir
|
|
498
|
-
name = f"{env_name(config.id)}--{config.model.replace('/', '--')}--legacy"
|
|
498
|
+
name = f"{env_name(config.legacy.id)}--{config.model.replace('/', '--')}--legacy"
|
|
499
499
|
return Path("outputs") / name / config.uuid
|
|
500
500
|
|
|
501
501
|
|
|
@@ -510,15 +510,21 @@ async def run_legacy_eval(config) -> list[Episode]:
|
|
|
510
510
|
|
|
511
511
|
# Install from the env hub on demand for an `org/name[@version]` id (a local id is
|
|
512
512
|
# already importable), then load by module name.
|
|
513
|
-
env = load_environment(
|
|
514
|
-
|
|
515
|
-
|
|
513
|
+
env = load_environment(
|
|
514
|
+
ensure_installed(config.legacy.id), **(config.legacy.args or {})
|
|
515
|
+
)
|
|
516
|
+
if (
|
|
517
|
+
config.legacy.extra_env_kwargs
|
|
518
|
+
): # post-load knobs (max_total_completion_tokens, …)
|
|
519
|
+
env.set_kwargs(**config.legacy.extra_env_kwargs)
|
|
516
520
|
dataset = env.get_eval_dataset() # the eval split (falls back to train when unset)
|
|
517
521
|
idxs = sample(list(range(len(dataset))), config.shuffle, config.num_tasks)
|
|
518
522
|
|
|
519
523
|
client = _eval_client(config.client, config.model)
|
|
520
524
|
sampling_args = config.sampling.model_dump(exclude_none=True)
|
|
521
|
-
taskset_id = env_name(
|
|
525
|
+
taskset_id = env_name(
|
|
526
|
+
config.legacy.id
|
|
527
|
+
) # the same identity the served bridge stamps
|
|
522
528
|
out_dir = _legacy_output_dir(config)
|
|
523
529
|
save_config(config, out_dir)
|
|
524
530
|
logger.info("results: %s", out_dir)
|
|
@@ -527,7 +533,7 @@ async def run_legacy_eval(config) -> list[Episode]:
|
|
|
527
533
|
len(idxs),
|
|
528
534
|
config.num_rollouts,
|
|
529
535
|
config.model,
|
|
530
|
-
config.id,
|
|
536
|
+
config.legacy.id,
|
|
531
537
|
)
|
|
532
538
|
|
|
533
539
|
sem = asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
|
verifiers/v1/push.py
CHANGED
|
@@ -186,7 +186,7 @@ def push_traces(
|
|
|
186
186
|
return finish(error="no PRIME_API_KEY (run `prime login`)")
|
|
187
187
|
|
|
188
188
|
traces = [trace for episode in episodes for trace in episode.traces]
|
|
189
|
-
env_name = (config.env.taskset.id) or config.id
|
|
189
|
+
env_name = (config.env.taskset.id) or config.legacy.id
|
|
190
190
|
metrics = _run_metrics(episodes, traces)
|
|
191
191
|
samples = _build_samples(episodes)
|
|
192
192
|
num_examples = len({t.task.data.idx for t in traces})
|
verifiers/v1/serve/pool.py
CHANGED
|
@@ -312,9 +312,10 @@ def serve_env(
|
|
|
312
312
|
if (
|
|
313
313
|
"config" in server_kwargs
|
|
314
314
|
): # dict-ify for the workers (config_data is picklable)
|
|
315
|
-
server_kwargs = {
|
|
316
|
-
|
|
317
|
-
|
|
315
|
+
server_kwargs = {**server_kwargs}
|
|
316
|
+
server_kwargs["config_data"] = env_config_data(
|
|
317
|
+
server_kwargs.pop("config")
|
|
318
|
+
)
|
|
318
319
|
pool = EnvServerPool(
|
|
319
320
|
server_kwargs,
|
|
320
321
|
max_workers,
|
|
@@ -335,9 +336,10 @@ def serve_env(
|
|
|
335
336
|
): # rebuild the env config for an in-process server
|
|
336
337
|
from verifiers.v1.loaders import resolve_env_config
|
|
337
338
|
|
|
338
|
-
server_kwargs = {
|
|
339
|
-
|
|
340
|
-
|
|
339
|
+
server_kwargs = {**server_kwargs}
|
|
340
|
+
server_kwargs["config"] = resolve_env_config(
|
|
341
|
+
server_kwargs.pop("config_data")
|
|
342
|
+
)
|
|
341
343
|
cls = LegacyEnvServer if legacy else EnvServer
|
|
342
344
|
cls.run_server(
|
|
343
345
|
address=address, address_queue=address_queue, **server_kwargs
|
verifiers/v1/serve/server.py
CHANGED
|
@@ -30,7 +30,10 @@ logger = logging.getLogger(__name__)
|
|
|
30
30
|
|
|
31
31
|
class EnvServer:
|
|
32
32
|
def __init__(
|
|
33
|
-
self,
|
|
33
|
+
self,
|
|
34
|
+
config: EnvConfig,
|
|
35
|
+
address: str = "tcp://127.0.0.1:5000",
|
|
36
|
+
max_concurrent: int | None = None,
|
|
34
37
|
) -> None:
|
|
35
38
|
self.address = address
|
|
36
39
|
self.taskset_id = config.taskset.id
|
|
@@ -52,9 +55,8 @@ class EnvServer:
|
|
|
52
55
|
# v1 envs never group-score (siblings score inside the env's own rollout);
|
|
53
56
|
# only the legacy (v0) bridge sets this.
|
|
54
57
|
self.requires_group_scoring = False
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
)
|
|
58
|
+
# This worker's episode bound (`--max-concurrent`), spanning requests.
|
|
59
|
+
self._gate = asyncio.Semaphore(max_concurrent) if max_concurrent else None
|
|
58
60
|
self._clients: dict[
|
|
59
61
|
tuple[str, str], Client
|
|
60
62
|
] = {} # (client_config, model) -> Client
|
|
@@ -122,8 +124,8 @@ class EnvServer:
|
|
|
122
124
|
async def _run(self, req: RunRequest) -> RunResponse:
|
|
123
125
|
ctx = self._context(req.client, req.model, req.sampling)
|
|
124
126
|
(slot,) = self.env.slots(self._build_task(req.task_data))
|
|
125
|
-
# The gate spans requests: `--
|
|
126
|
-
#
|
|
127
|
+
# The gate spans requests: `--max-concurrent` bounds this worker's episodes
|
|
128
|
+
# in flight the same way the in-process eval's semaphore does.
|
|
127
129
|
episode = await self.env.run_slot(slot, ctx, self._gate)
|
|
128
130
|
# Trust the env-minted episode; serialize it once before client-side re-typing.
|
|
129
131
|
return RunResponse.model_construct(episode=episode)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev44
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -39,7 +39,7 @@ Requires-Dist: prime-sandboxes>=0.2.33
|
|
|
39
39
|
Requires-Dist: prime-tunnel>=0.1.8
|
|
40
40
|
Requires-Dist: pydantic>=2.12.3
|
|
41
41
|
Requires-Dist: pyzmq>=27.1.0
|
|
42
|
-
Requires-Dist: renderers>=0.1.
|
|
42
|
+
Requires-Dist: renderers>=0.1.9.dev9
|
|
43
43
|
Requires-Dist: requests
|
|
44
44
|
Requires-Dist: rich>=11.0.0
|
|
45
45
|
Requires-Dist: setproctitle>=1.3.0
|
|
@@ -167,18 +167,18 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
|
|
|
167
167
|
verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
|
-
verifiers/v1/__init__.py,sha256=
|
|
171
|
-
verifiers/v1/agent.py,sha256=
|
|
170
|
+
verifiers/v1/__init__.py,sha256=xHKE-HZA8bN0JJovf4KbBI2mxBtZJ4hPNnzac4FtWwM,7332
|
|
171
|
+
verifiers/v1/agent.py,sha256=2MJBztogGCwDMt0DVZuuItADgZpTk3Xaf_x_cQswdds,31956
|
|
172
172
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
173
|
-
verifiers/v1/env.py,sha256=
|
|
173
|
+
verifiers/v1/env.py,sha256=Omfwspenzs7A_K-0oDVPH1lHS8YkzihmJEZ2DjK4Kqs,18445
|
|
174
174
|
verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
|
|
175
175
|
verifiers/v1/errors.py,sha256=Pj5Om8x1fP3TDPQQn6nUCoSPoZRsy2JeBz8pXhpPrDY,6883
|
|
176
176
|
verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
|
|
177
177
|
verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
|
|
178
178
|
verifiers/v1/judge.py,sha256=VnTi9iSfCJMCccYWc1qCTcT7iD3AcomS2cXfHwh1HVw,9423
|
|
179
|
-
verifiers/v1/legacy.py,sha256=
|
|
179
|
+
verifiers/v1/legacy.py,sha256=aPpHsiNyW3Ail3U8Nfx9XM6tS__Py_haEluiZWVQ50k,22398
|
|
180
180
|
verifiers/v1/loaders.py,sha256=TnNKY2msVo3p7m-5wjw376PYH-zTu3TLixXRst0VOiM,9985
|
|
181
|
-
verifiers/v1/push.py,sha256=
|
|
181
|
+
verifiers/v1/push.py,sha256=uq4ZQBYNPQhNsNORNmlWEwTL186E4deGXGqG6fZBQgM,11341
|
|
182
182
|
verifiers/v1/retries.py,sha256=GAUQB2k-SJGnKjLpI1Y_XOmhSBeqIquljkXQHXYfg88,5380
|
|
183
183
|
verifiers/v1/rollout.py,sha256=eU01fZRf87lQ5LQUVIDN38SlcUE0svXxfyS0gryNboE,20503
|
|
184
184
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
@@ -197,17 +197,17 @@ verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
|
|
|
197
197
|
verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
|
|
198
198
|
verifiers/v1/cli/replay.py,sha256=T4sg-MpvT9niymEJl0Y2TijgTy-tVQqjT1lQ3nwuOg8,10314
|
|
199
199
|
verifiers/v1/cli/resolve.py,sha256=1sV4z04v_HEs86Oi8ODJqfujXjrbShbu3u2kDpgpj94,4079
|
|
200
|
-
verifiers/v1/cli/serve.py,sha256=
|
|
200
|
+
verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
|
|
201
201
|
verifiers/v1/cli/validate.py,sha256=PMWRSZROiTh2KFzkA3Ez5-z09krVk2ZUTkFNMUqk2-4,9671
|
|
202
202
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
203
203
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
204
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
204
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=lPrP7Ikmz3r7v3ury2yAMLgYeweVGFdg0ZKSQ6lwwKU,34355
|
|
205
205
|
verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
|
|
206
206
|
verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
|
|
207
207
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
208
|
-
verifiers/v1/cli/eval/main.py,sha256=
|
|
208
|
+
verifiers/v1/cli/eval/main.py,sha256=zvbLy_ryn3S1fWoTmWL4QEReVO1Q_hIrgyK6Fp0q_ZM,5501
|
|
209
209
|
verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
|
|
210
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
210
|
+
verifiers/v1/cli/eval/runner.py,sha256=PwcIaVLGBJON9nC2u3z6aTGALmoa0vKAYrg0IYM-8Dw,11493
|
|
211
211
|
verifiers/v1/clients/__init__.py,sha256=bHGS5jfrAZ-VBUeXpasXkkKdHkFgjOSN9KbOJXrnFtU,511
|
|
212
212
|
verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
|
|
213
213
|
verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
|
|
@@ -215,19 +215,21 @@ verifiers/v1/clients/eval.py,sha256=QMpohNnQjDuo56BhOgHnAfAlcnQAxFpEFCySoIuexNg,
|
|
|
215
215
|
verifiers/v1/clients/train.py,sha256=6bw4TV_UskdjvDAxvHrgJWBqAq9r1DxzJGusGi7UoRM,13022
|
|
216
216
|
verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxVtvA,317
|
|
217
217
|
verifiers/v1/configs/agent.py,sha256=N2Lm0iZXFxfsSIiXUQ7f3RDZcoin3M7pJj6HVdCE5eo,3537
|
|
218
|
-
verifiers/v1/configs/env.py,sha256=
|
|
218
|
+
verifiers/v1/configs/env.py,sha256=4aUap3BJX3WCA2_fZ-jGPxF3khiVpORqmz_3eBT4j9w,7824
|
|
219
219
|
verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Vagg,1564
|
|
220
220
|
verifiers/v1/configs/judge.py,sha256=ZJlmWogHY8oDjvWvtNuVKvkzuheWalIjl1njJ4dE0Mk,2511
|
|
221
|
+
verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
|
|
221
222
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
223
|
+
verifiers/v1/configs/serve.py,sha256=cHHnFGQtqEwvaovdeoVMpCebTHgPXKUuIivN5IhxZR0,2596
|
|
222
224
|
verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
|
|
223
225
|
verifiers/v1/configs/taskset.py,sha256=IdZwsarn0_WvYdR_Kr4QYYdqbaoGYkbYyO5VA-RPYPc,657
|
|
224
226
|
verifiers/v1/configs/cli/__init__.py,sha256=3wBAONw5XkBPqbMjpHk5XBHlBVybR8JQ2JCPWijYBMU,444
|
|
225
227
|
verifiers/v1/configs/cli/debug.py,sha256=3lkKqmQFLJBcUZECChBHCrod5pbCKSmT47EhjILrxjI,2843
|
|
226
|
-
verifiers/v1/configs/cli/env.py,sha256=
|
|
227
|
-
verifiers/v1/configs/cli/eval.py,sha256=
|
|
228
|
+
verifiers/v1/configs/cli/env.py,sha256=XgjRXYpPjga0ZYIHEESY7VuVvtE6pWfAzf6JBLeuTIc,2695
|
|
229
|
+
verifiers/v1/configs/cli/eval.py,sha256=YWYMvPOIAG51e0JdmyA-U_P7eInMPZ_8DTFim6aWn6Y,6869
|
|
228
230
|
verifiers/v1/configs/cli/init.py,sha256=tMaTDF_lU1Af3JmQpksk2mwe8CAfpnHsquXhwepZhYo,1019
|
|
229
231
|
verifiers/v1/configs/cli/replay.py,sha256=57R-OZnVXPyrRVrWq2UeBBhnBgTdjGs1DJ_PWjNU3AY,2907
|
|
230
|
-
verifiers/v1/configs/cli/serve.py,sha256=
|
|
232
|
+
verifiers/v1/configs/cli/serve.py,sha256=1rAt0QOL-XT3DMmRUIA40t1E3YuoD1vOADfsiTwFBf0,2981
|
|
231
233
|
verifiers/v1/configs/cli/validate.py,sha256=xaYUjQ0MEAAbA_jp55zAtYxlTettZivUC0Tu1H1WZ50,2292
|
|
232
234
|
verifiers/v1/dialects/__init__.py,sha256=p72C7VYrFQzVgTcfmzpHGSwXlvThfc4pvjBxsqw7kHE,786
|
|
233
235
|
verifiers/v1/dialects/anthropic.py,sha256=nj9J7OOhexmPHa-sWiNP2xlI9_bGhvwM5VWzdh19pBc,13526
|
|
@@ -245,7 +247,7 @@ verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-
|
|
|
245
247
|
verifiers/v1/envs/user_sim/env.py,sha256=u376-UJEVdzgcit7FXjz5KEwlPxSU3qqoIrxs6foi2w,4567
|
|
246
248
|
verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
|
|
247
249
|
verifiers/v1/gepa/adapter.py,sha256=Wo3adNf1I3eT4L_03vxWrrg6HvZBz5KDQonCsGdhLgg,6185
|
|
248
|
-
verifiers/v1/gepa/config.py,sha256=
|
|
250
|
+
verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4626
|
|
249
251
|
verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
|
|
250
252
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
251
253
|
verifiers/v1/gepa/runner.py,sha256=o7Am2Ram-Ejbr_tVpN70zb0uQ5a1NIpXvlMhwWL22Xs,5406
|
|
@@ -301,8 +303,8 @@ verifiers/v1/runtimes/docker/__init__.py,sha256=eC5hjfyPwvwEKaeZUeP9EN8tJkrl0HUx
|
|
|
301
303
|
verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
|
|
302
304
|
verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
|
|
303
305
|
verifiers/v1/serve/client.py,sha256=amUhf4cFJ1xCb8kw_AoVviwN543Wknx8LzImq1ZpaiQ,7057
|
|
304
|
-
verifiers/v1/serve/pool.py,sha256=
|
|
305
|
-
verifiers/v1/serve/server.py,sha256=
|
|
306
|
+
verifiers/v1/serve/pool.py,sha256=e9Nc3nQKEyV9YqyKHl_RuqQ84zWVa8z2J1rc3rfCadM,14828
|
|
307
|
+
verifiers/v1/serve/server.py,sha256=VBOD3Gzq4p0ZD_8xgtG039i3JaefhOH0MmR6cxG56iY,9705
|
|
306
308
|
verifiers/v1/serve/types.py,sha256=AHFbt8C-Rzlee34RDPLQ3lPiYDLkUm5eFMnRQINiO6s,3016
|
|
307
309
|
verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
|
|
308
310
|
verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
|
|
@@ -327,8 +329,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
327
329
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
328
330
|
verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
|
|
329
331
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
330
|
-
verifiers-0.2.2.
|
|
331
|
-
verifiers-0.2.2.
|
|
332
|
-
verifiers-0.2.2.
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
332
|
+
verifiers-0.2.2.dev44.dist-info/METADATA,sha256=xAsUe3hwzJqF-nbNTppTvycNvcRTi_cnhIq8oFj9adQ,4545
|
|
333
|
+
verifiers-0.2.2.dev44.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
334
|
+
verifiers-0.2.2.dev44.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
335
|
+
verifiers-0.2.2.dev44.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
336
|
+
verifiers-0.2.2.dev44.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|