verifiers 0.2.2.dev42__py3-none-any.whl → 0.2.2.dev44__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/__init__.py CHANGED
@@ -14,16 +14,18 @@ from verifiers.v1.clients import (
14
14
  resolve_client,
15
15
  )
16
16
  from verifiers.v1.configs.agent import AgentConfig
17
- from verifiers.v1.configs.cli.env import (
18
- ElasticPoolConfig,
19
- EnvServerConfig,
20
- StaticPoolConfig,
21
- pool_serve_kwargs,
22
- )
17
+ from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
23
18
  from verifiers.v1.configs.env import EnvConfig, default_agent_harness
24
19
  from verifiers.v1.configs.harness import HarnessConfig
25
20
  from verifiers.v1.configs.judge import JudgeConfig, Judges
21
+ from verifiers.v1.configs.legacy import LegacyEnvConfig
26
22
  from verifiers.v1.configs.retries import RetryConfig
23
+ from verifiers.v1.configs.serve import (
24
+ ElasticPoolConfig,
25
+ ServingConfig,
26
+ StaticPoolConfig,
27
+ pool_serve_kwargs,
28
+ )
27
29
  from verifiers.v1.configs.task import TaskConfig
28
30
  from verifiers.v1.configs.taskset import TasksetConfig
29
31
  from verifiers.v1.decorators import metric, reward, stop, tool
@@ -248,7 +250,10 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
248
250
  "Env",
249
251
  "SingleAgentEnv",
250
252
  "EnvConfig",
251
- "EnvServerConfig",
253
+ "ServingConfig",
254
+ "LegacyEnvConfig",
255
+ "resolve_env_field",
256
+ "narrowed_env_annotation",
252
257
  "SingleAgentEnvConfig",
253
258
  "AgentConfig",
254
259
  "StaticPoolConfig",
verifiers/v1/agent.py CHANGED
@@ -276,9 +276,9 @@ class Agent:
276
276
  max_total_tokens=config.max_total_tokens,
277
277
  )
278
278
  self.timeout = config.timeout
279
- # Env episode agents replace this with the eval concurrency semaphore.
280
- # Interactions acquire it only around active lifecycle work, never while
281
- # awaiting the caller between segments.
279
+ # Env episode agents replace this with the episode's agent semaphore
280
+ # (`--env.max-concurrent-agents`). Interactions acquire it only around active
281
+ # lifecycle work, never while awaiting the caller between segments.
282
282
  self._gate: asyncio.Semaphore | None = None
283
283
  # Env-owned standing, not config: `Env.setup` marks fixed agents
284
284
  # untrainable and traces are stamped from here; inert outside an env.
@@ -578,9 +578,9 @@ class _EpisodeAgent(Agent):
578
578
  bundle of references — expensive resources are env-owned and borrowed, so no
579
579
  state spans concurrent episodes): traces get their agent standing the moment
580
580
  they're created, finished ones land in `completed` (the episode's traces),
581
- each run takes the eval's gate. The taskset's shared tool servers ride only
582
- its own tasks — on an env-minted task they'd wrongly put MCP in play
583
- (`tools=` overrides)."""
581
+ each run takes one of the episode's agent permits. The taskset's shared tool
582
+ servers ride only its own tasks — on an env-minted task they'd wrongly put MCP
583
+ in play (`tools=` overrides)."""
584
584
 
585
585
  def __init__(
586
586
  self,
@@ -664,8 +664,9 @@ class _EpisodeAgent(Agent):
664
664
  """The agent's `interaction`, with every trace stamped with its standing
665
665
  at mint and captured in `completed` at close — an interaction driven from
666
666
  `Env.run` stays crash-safe. Setup, each active segment, and close acquire
667
- the eval gate independently; the interaction holds no permit while awaiting
668
- its caller, so peer interactions can interleave even at concurrency one."""
667
+ an agent permit independently; the interaction holds no permit while awaiting
668
+ its caller, so peer interactions still interleave where an episode plays one
669
+ agent at a time."""
669
670
  trace: Trace | None = None
670
671
 
671
672
  def remember(current: Trace) -> None:
@@ -95,7 +95,7 @@ def _limits(config: EvalConfig) -> list[str]:
95
95
  toks.append(f"{label}≤{v}")
96
96
  turns = _seat_value(config, lambda spec: spec.max_turns)
97
97
  return [
98
- f"≤{config.max_concurrent} concurrent"
98
+ f"≤{config.max_concurrent} episodes"
99
99
  if config.max_concurrent
100
100
  else "no concurrency cap",
101
101
  "per-seat turn caps"
@@ -24,7 +24,7 @@ from verifiers.v1.utils.logging import setup_logging
24
24
  logger = logging.getLogger(__name__)
25
25
 
26
26
  USAGE = (
27
- "usage: uv run eval [<taskset-id>] [--env.id <id>] [--id <env-id> (legacy)] [options] [@ file.toml]\n"
27
+ "usage: uv run eval [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]\n"
28
28
  " uv run eval --resume <output-dir> (re-run a previous run's missing/errored rollouts)"
29
29
  )
30
30
 
@@ -49,12 +49,14 @@ def main(argv: list[str] | None = None) -> None:
49
49
  )
50
50
  config = load_resume_config(resume_dir)
51
51
  else:
52
- legacy_id = any(a == "--id" or a.startswith("--id=") for a in argv) # v0 env id
53
- # An env-block flag (or a retired flat axis) skips the usage gate so the
54
- # typed parse renders its did-you-mean / pointer to the new flags instead
55
- # of a bare usage line.
52
+ legacy_id = any(
53
+ a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv
54
+ )
55
+ # An env-block flag (or a since-moved flat axis) skips the usage gate so the
56
+ # typed parse renders its did-you-mean instead of a bare usage line.
56
57
  typed_axis = any(
57
- a.startswith(("--env.", "--taskset.", "--harness.")) for a in argv
58
+ a.startswith(("--env.", "--taskset.", "--harness.", "--serve."))
59
+ for a in argv
58
60
  )
59
61
  if (
60
62
  not extract_id(argv, "env.taskset")
@@ -64,7 +66,7 @@ def main(argv: list[str] | None = None) -> None:
64
66
  ):
65
67
  raise SystemExit(
66
68
  USAGE
67
- ) # need a taskset (positional / --env.taskset.id), a legacy --id, or a @ file.toml
69
+ ) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or a @ file.toml
68
70
 
69
71
  with plugin_errors():
70
72
  config_type = narrow_config(EvalConfig, argv)
@@ -107,19 +107,24 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
107
107
  import multiprocessing as mp
108
108
  from functools import partial
109
109
 
110
- from verifiers.v1.configs.cli.env import pool_serve_kwargs
110
+ from verifiers.v1.configs.serve import pool_serve_kwargs
111
111
  from verifiers.v1.serve import EnvClient, env_config_data, serve_env
112
112
  from verifiers.v1.utils.logging import setup_logging
113
113
 
114
114
  legacy = config.is_legacy
115
115
  server_kwargs = (
116
116
  {
117
- "env_id": config.id,
118
- "env_args": config.args,
119
- "extra_env_kwargs": config.extra_env_kwargs,
117
+ "env_id": config.legacy.id,
118
+ "env_args": config.legacy.args,
119
+ "extra_env_kwargs": config.legacy.extra_env_kwargs,
120
120
  }
121
121
  if legacy
122
- else {"config_data": env_config_data(config.env)} # picklable across the spawn
122
+ else {
123
+ "config_data": env_config_data(config.env), # picklable across the spawn
124
+ # `-c` seeds each worker's episode bound unless `[serve]` pins one — so a
125
+ # pool carries `workers * bound` episodes, as `multiplex` implies.
126
+ "max_concurrent": config.worker_max_concurrent,
127
+ }
123
128
  )
124
129
  tasks = []
125
130
  if not legacy:
@@ -142,7 +147,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
142
147
  proc = mpctx.Process(
143
148
  target=serve_env,
144
149
  kwargs=dict(
145
- **pool_serve_kwargs(config.pool),
150
+ **pool_serve_kwargs(config.serve.pool),
146
151
  legacy=legacy,
147
152
  address="tcp://127.0.0.1:0",
148
153
  address_queue=address_queue,
@@ -210,7 +215,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
210
215
  "running %dx%d rollouts via the env-server %s pool on %s",
211
216
  len(plan),
212
217
  config.num_rollouts,
213
- config.pool.type,
218
+ config.serve.pool.type,
214
219
  config.model,
215
220
  )
216
221
  logger.info("results: %s", out)
verifiers/v1/cli/serve.py CHANGED
@@ -12,12 +12,12 @@ from verifiers.v1.cli.resolve import (
12
12
  references_config_file,
13
13
  with_positional_taskset,
14
14
  )
15
- from verifiers.v1.configs.cli.env import pool_serve_kwargs
16
15
  from verifiers.v1.configs.cli.serve import ServeConfig
16
+ from verifiers.v1.configs.serve import pool_serve_kwargs
17
17
  from verifiers.v1.serve import serve_env
18
18
  from verifiers.v1.utils.logging import setup_logging
19
19
 
20
- USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--id <env-id> (legacy)] [options] [@ file.toml]"
20
+ USAGE = "usage: uv run serve [<taskset-id>] [--env.id <id>] [--legacy.id <env-id> (v0)] [options] [@ file.toml]"
21
21
 
22
22
 
23
23
  def main(argv: list[str] | None = None) -> None:
@@ -29,11 +29,12 @@ def main(argv: list[str] | None = None) -> None:
29
29
  with plugin_errors():
30
30
  cli(narrow_config(ServeConfig, argv))
31
31
  return
32
- legacy_id = any(a == "--id" or a.startswith("--id=") for a in argv) # v0 env id
33
- # An env-block flag (or a retired flat axis) skips the usage gate so the typed
34
- # parse renders its did-you-mean / pointer to the new flags instead of a bare
35
- # usage line.
36
- typed_axis = any(a.startswith(("--env.", "--taskset.", "--harness.")) for a in argv)
32
+ legacy_id = any(a == "--legacy.id" or a.startswith("--legacy.id=") for a in argv)
33
+ # An env-block flag (or a since-moved flat axis) skips the usage gate so the typed
34
+ # parse renders its did-you-mean instead of a bare usage line.
35
+ typed_axis = any(
36
+ a.startswith(("--env.", "--taskset.", "--harness.", "--serve.")) for a in argv
37
+ )
37
38
  if (
38
39
  not extract_id(argv, "env.taskset")
39
40
  and not legacy_id
@@ -42,7 +43,7 @@ def main(argv: list[str] | None = None) -> None:
42
43
  ):
43
44
  raise SystemExit(
44
45
  USAGE
45
- ) # need a taskset (positional / --env.taskset.id), a legacy --id, or @ file.toml
46
+ ) # need a taskset (positional / --env.taskset.id), a v0 --legacy.id, or @ file.toml
46
47
 
47
48
  with plugin_errors():
48
49
  config_type = narrow_config(ServeConfig, argv)
@@ -60,17 +61,19 @@ def main(argv: list[str] | None = None) -> None:
60
61
  # in each one.
61
62
  server_kwargs = (
62
63
  {
63
- "env_id": config.id,
64
- "env_args": config.args,
65
- "extra_env_kwargs": config.extra_env_kwargs,
64
+ "env_id": config.legacy.id,
65
+ "env_args": config.legacy.args,
66
+ "extra_env_kwargs": config.legacy.extra_env_kwargs,
66
67
  }
67
68
  if config.is_legacy
68
- else {"config": config.env}
69
+ # `--serve.max-concurrent` is each v1 worker's episode bound; the legacy
70
+ # bridge has never had one.
71
+ else {"config": config.env, "max_concurrent": config.serve.max_concurrent}
69
72
  )
70
73
  serve_env(
71
- **pool_serve_kwargs(config.pool),
74
+ **pool_serve_kwargs(config.serve.pool),
72
75
  legacy=config.is_legacy,
73
- address=config.address,
76
+ address=config.serve.address,
74
77
  log_setup=partial(setup_logging, level),
75
78
  **server_kwargs,
76
79
  )
@@ -1,35 +1,30 @@
1
- """Run-config plumbing around the `[env]` block: narrowing the `env` field of the
2
- configs that own one, and how a served env sizes its worker pool."""
1
+ """Run-config plumbing around the `[env]` block: narrowing the `env` field of every
2
+ config that owns one, and the retired keys such a config refuses.
3
3
 
4
- from typing import Annotated, Literal
4
+ A run composes the blocks it needs — `[env]` (what runs, `configs/env.py`),
5
+ `[serve]` (how it's hosted, `configs/serve.py`), `[legacy]` (the v0 bridge,
6
+ `configs/legacy.py`) — plus its own fields. Nothing here is a base class: the eval
7
+ CLI, the `serve` CLI, GEPA and a trainer each declare their blocks and call these.
5
8
 
6
- from pydantic import Field, SerializeAsAny, ValidationError, model_validator
9
+ Declare the env field as `SerializeAsAny[EnvConfig] = Field(default_factory=
10
+ single_agent_env_config)`. The `SerializeAsAny` is load-bearing: pydantic serializes
11
+ by declared type, so a plain `EnvConfig` silently drops a narrowed subclass's agents
12
+ and knobs from `model_dump()` — the env-server wire's payload."""
13
+
14
+ from pydantic import ValidationError
7
15
  from pydantic_config import BaseConfig
8
16
 
9
17
  from verifiers.v1.configs.env import EnvConfig
10
- from verifiers.v1.types import ID
11
18
  from verifiers.v1.utils.generic import prefix_validation_error
12
19
 
13
20
 
14
21
  def resolve_env_field(data: dict, narrowed: "type[EnvConfig] | None" = None) -> dict:
15
- """Shared `mode="before"` body for every run config owning an `env` field
16
- (`EnvServerConfig`, `GEPAConfig`): refuse the retired top-level axes with a
17
- pointer home, and narrow `env` to the concrete env's config class. `narrowed`
18
- is the annotation the CLI pre-resolved (`narrow_config`) — its id is
19
- authoritative, so validate against it directly."""
22
+ """Shared `mode="before"` body for every run config owning an `env` field: narrow
23
+ `env` to the concrete env's config class. `narrowed` is the annotation the CLI
24
+ pre-resolved (`narrow_config`) — its id is authoritative, so validate against it
25
+ directly."""
20
26
  if not isinstance(data, dict):
21
27
  return data
22
- if "taskset" in data:
23
- raise ValueError(
24
- "the taskset lives on the env now: --env.taskset.id <id> "
25
- "(TOML: [env.taskset]), or the positional `eval <taskset-id>`"
26
- )
27
- if "harness" in data:
28
- raise ValueError(
29
- "a harness belongs to an agent now: --env.agent.harness.* on the "
30
- "single-agent env, --env.<agent>.harness.* on a multi-agent one "
31
- "(TOML: [env.agent.harness])"
32
- )
33
28
  raw = data.get("env")
34
29
  if raw is None:
35
30
  return data
@@ -63,105 +58,3 @@ def narrowed_env_annotation(cls) -> "type[EnvConfig] | None":
63
58
  ):
64
59
  return annotation
65
60
  return None
66
-
67
-
68
- class StaticPoolConfig(BaseConfig):
69
- """Fixed env-server pool: pre-spawn `num_workers` workers up front."""
70
-
71
- type: Literal["static"] = "static"
72
- num_workers: int = Field(4, ge=1)
73
- """Worker processes to pre-spawn (1 = a single in-process server, no pool)."""
74
-
75
-
76
- class ElasticPoolConfig(BaseConfig):
77
- """Elastic env-server pool: start at one worker and scale up on demand."""
78
-
79
- type: Literal["elastic"] = "elastic"
80
- max_workers: int | None = None
81
- """Upper bound on workers (None = unbounded)."""
82
- multiplex: int = Field(128, ge=1)
83
- """Rollouts per worker for the scale-up trigger: add a worker once in-flight rollouts
84
- reach 90% of `workers * multiplex`."""
85
-
86
-
87
- # Discriminated on `type` so the CLI selects with `--pool.type static|elastic`.
88
- PoolConfig = Annotated[
89
- StaticPoolConfig | ElasticPoolConfig, Field(discriminator="type")
90
- ]
91
-
92
-
93
- def pool_serve_kwargs(pool: StaticPoolConfig | ElasticPoolConfig) -> dict:
94
- """Unpack a pool config into `serve_env` kwargs (`max_workers` / `multiplex` / `elastic`)."""
95
- if isinstance(pool, ElasticPoolConfig):
96
- return {
97
- "max_workers": pool.max_workers,
98
- "multiplex": pool.multiplex,
99
- "elastic": True,
100
- }
101
- return {"max_workers": pool.num_workers, "elastic": False}
102
-
103
-
104
- def _single_agent_env_config() -> EnvConfig:
105
- """The default `env` block: the single-agent shape. Lazy — the concrete env
106
- lives in `envs/`, which imports `env.py`."""
107
- from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
108
-
109
- return SingleAgentEnvConfig()
110
-
111
-
112
- class EnvServerConfig(BaseConfig):
113
- """A run's environment plus how it's *served*: the `env` block and the worker-pool
114
- sizing. Shared by the `serve` CLI, server-backed eval, and prime-rl's orchestrator, so
115
- they all configure the pool the same way (`--pool.type elastic|static`)."""
116
-
117
- # SerializeAsAny: see EnvConfig.taskset — model_dump() must keep the subclass's
118
- # agent fields and knobs.
119
- env: SerializeAsAny[EnvConfig] = Field(default_factory=_single_agent_env_config)
120
- """The environment — the run's `[env]` block: which env, its seed taskset, each
121
- agent, its knobs, and the run limits. Narrowed to the selected env's config
122
- class by the env id, else the taskset id."""
123
- pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
124
- """Worker-pool sizing for the env server. `elastic` (default) starts at one worker and
125
- scales up on demand; `static` pre-spawns a fixed `num_workers`."""
126
- # --- legacy (v0) backwards-compat -----------------------------------------
127
- id: ID | None = None
128
- """Classic (v0) env id (`name`, `org/name`, or `org/name@version` — installed from the
129
- hub on demand), loaded via `verifiers.load_environment` and run through the legacy
130
- bridge. Set this *instead of* `env.taskset` to run a v0 environment."""
131
- args: dict = Field(default_factory=dict)
132
- """Construction kwargs forwarded to `load_environment(id, **args)`."""
133
- extra_env_kwargs: dict = Field(default_factory=dict)
134
- """Post-load kwargs applied to the v0 env via `env.set_kwargs(**extra_env_kwargs)` (e.g.
135
- `max_total_completion_tokens`, `max_seq_len`, `timeout_seconds`) — typically
136
- auto-populated by the orchestrator, distinct from the `args` passed at construction."""
137
-
138
- @property
139
- def is_legacy(self) -> bool:
140
- """A v0/legacy env (run via the bridge): a legacy `id` is set and no v1 taskset."""
141
- return self.id is not None and not self.env.taskset.id
142
-
143
- @property
144
- def env_id(self) -> str:
145
- """The run's identifier: the v1 env's (`EnvConfig.env_id`), else the legacy
146
- v0 env id."""
147
- return self.env.env_id or self.id or ""
148
-
149
- @model_validator(mode="after")
150
- def _refuse_legacy_id_with_taskset(self):
151
- """A legacy `id` next to a v1 `env.taskset` would be silently inert
152
- (`is_legacy` is False and the v0 env never loads); refuse the mix."""
153
- if self.id is not None and self.env.taskset.id:
154
- raise ValueError(
155
- f"--id {self.id!r} is the legacy (v0) env id and can't combine with "
156
- f"the v1 taskset {self.env.taskset.id!r}. Pairing an env with a "
157
- f"taskset is --env.id {self.id!r} (TOML: id under [env]); to run the "
158
- "v0 env instead, drop the taskset."
159
- )
160
- return self
161
-
162
- # --- end legacy -----------------------------------------------------------
163
-
164
- @model_validator(mode="before")
165
- @classmethod
166
- def _resolve_env(cls, data):
167
- return resolve_env_field(data, narrowed_env_annotation(cls))
@@ -3,14 +3,27 @@
3
3
  from pathlib import Path
4
4
  from uuid import uuid4
5
5
 
6
- from pydantic import AliasChoices, Field, model_validator
6
+ from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
7
+ from pydantic_config import BaseConfig
7
8
 
8
9
  from verifiers.v1.clients import ClientConfig, EvalClientConfig
9
- from verifiers.v1.configs.cli.env import EnvServerConfig
10
+ from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
11
+ from verifiers.v1.configs.env import EnvConfig
12
+ from verifiers.v1.configs.legacy import LegacyEnvConfig
13
+ from verifiers.v1.configs.serve import ServingConfig
14
+ from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
10
15
  from verifiers.v1.types import SamplingConfig
11
16
 
12
17
 
13
- class EvalConfig(EnvServerConfig):
18
+ class EvalConfig(BaseConfig):
19
+ env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
20
+ """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
21
+ the selected env's config class by the env id, else the taskset id."""
22
+ serve: ServingConfig = ServingConfig()
23
+ """How the env is hosted under `--server`: the worker pool, each worker's episode
24
+ bound. Ignored by an in-process run."""
25
+ legacy: LegacyEnvConfig = LegacyEnvConfig()
26
+ """A classic (v0) environment to evaluate through the bridge instead of `[env]`."""
14
27
  uuid: str = Field(default_factory=lambda: str(uuid4()), exclude=True)
15
28
  """Auto-generated run id — the leaf of the output dir, so runs never overwrite.
16
29
  Excluded from the saved config so re-running `@ config.toml` lands in a fresh dir."""
@@ -37,9 +50,12 @@ class EvalConfig(EnvServerConfig):
37
50
  shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
38
51
  """Shuffle tasks before taking the first `num_tasks`."""
39
52
  max_concurrent: int | None = Field(
40
- 128, validation_alias=AliasChoices("max_concurrent", "c")
53
+ 128, ge=1, validation_alias=AliasChoices("max_concurrent", "c")
41
54
  )
42
- """Max rollouts in flight at once."""
55
+ """Episodes in flight at once, `None` for no limit. An episode plays its agents one
56
+ at a time, so this is the live agent runs too — until `--env.max-concurrent-agents`
57
+ says otherwise. Under `--server` it seeds each worker's bound, unless
58
+ `--serve.max-concurrent` pins one."""
43
59
  verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
44
60
  """Log at debug level instead of the default info."""
45
61
  dry_run: bool = Field(False, exclude=True)
@@ -49,7 +65,7 @@ class EvalConfig(EnvServerConfig):
49
65
  """Show a live dashboard instead of per-rollout logs (in-process only; an unset
50
66
  `rich` defaults off under `--server`)."""
51
67
  server: bool = False
52
- """Drive rollouts through the env-server worker pool (sized by `--pool.*`) instead of
68
+ """Drive rollouts through the env-server worker pool (sized by `[serve]`) instead of
53
69
  in-process — the path prime-rl trains through. Incompatible with `--rich`."""
54
70
  push: bool = True
55
71
  """Upload the finished run to the Prime Intellect platform (the private Evaluations
@@ -80,3 +96,47 @@ class EvalConfig(EnvServerConfig):
80
96
  "`--server`; drop `--rich`."
81
97
  )
82
98
  return self
99
+
100
+ @property
101
+ def is_legacy(self) -> bool:
102
+ """Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
103
+ return self.legacy.id is not None and not self.env.taskset.id
104
+
105
+ @property
106
+ def env_id(self) -> str:
107
+ """The run's identifier: the v1 env's, else the v0 env id."""
108
+ return self.env.env_id or self.legacy.id or ""
109
+
110
+ @property
111
+ def worker_max_concurrent(self) -> int | None:
112
+ """A served worker's episode bound: its own pin, else the run's `--max-concurrent`."""
113
+ return (
114
+ self.serve.max_concurrent
115
+ if self.serve.max_concurrent is not None
116
+ else self.max_concurrent
117
+ )
118
+
119
+ @model_validator(mode="after")
120
+ def _refuse_mixed_run(self):
121
+ # A v0 id next to any v1 env identity leaves one of the two going nowhere, and
122
+ # which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
123
+ # loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
124
+ if self.legacy.id is None or not self.env.env_id:
125
+ return self
126
+ if self.env.taskset.id:
127
+ raise ValueError(
128
+ f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
129
+ f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
130
+ f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
131
+ "the v0 env instead, drop the taskset."
132
+ )
133
+ raise ValueError(
134
+ f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
135
+ f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
136
+ "with the v1 env's name. Keep whichever one you meant to run."
137
+ )
138
+
139
+ @model_validator(mode="before")
140
+ @classmethod
141
+ def _resolve_env(cls, data):
142
+ return resolve_env_field(data, narrowed_env_annotation(cls))
@@ -1,16 +1,62 @@
1
1
  """Environment-server CLI configuration."""
2
2
 
3
- from pydantic import AliasChoices, Field
3
+ from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
4
+ from pydantic_config import BaseConfig
4
5
 
5
- from verifiers.v1.configs.cli.env import EnvServerConfig
6
+ from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
7
+ from verifiers.v1.configs.env import EnvConfig
8
+ from verifiers.v1.configs.legacy import LegacyEnvConfig
9
+ from verifiers.v1.configs.serve import ServingConfig
10
+ from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
6
11
 
7
12
 
8
- class ServeConfig(EnvServerConfig):
9
- address: str = Field(
10
- "tcp://127.0.0.1:5000", validation_alias=AliasChoices("address", "a")
11
- )
12
- """ZMQ address the ROUTER binds (and clients connect to)."""
13
+ class ServeConfig(BaseConfig):
14
+ """`uv run serve`: what to serve (`[env]`, or `[legacy]` for a classic v0 env) and
15
+ how it's hosted (`[serve]`)."""
16
+
17
+ env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
18
+ """The environment — which env, its seed taskset, each agent, its knobs. Narrowed to
19
+ the selected env's config class by the env id, else the taskset id."""
20
+ serve: ServingConfig = ServingConfig()
21
+ """How it's served: the worker pool, the bind address, each worker's episode bound."""
22
+ legacy: LegacyEnvConfig = LegacyEnvConfig()
23
+ """A classic (v0) environment to serve through the bridge instead of `[env]`."""
13
24
  verbose: bool = Field(False, validation_alias=AliasChoices("verbose", "v"))
14
25
  """Log at debug level instead of info."""
15
26
  dry_run: bool = False
16
27
  """Resolve + validate the config and dump it, then exit."""
28
+
29
+ @property
30
+ def is_legacy(self) -> bool:
31
+ """Whether this run goes through the v0 bridge: a legacy id and no v1 taskset."""
32
+ return self.legacy.id is not None and not self.env.taskset.id
33
+
34
+ @property
35
+ def env_id(self) -> str:
36
+ """The run's identifier: the v1 env's, else the v0 env id."""
37
+ return self.env.env_id or self.legacy.id or ""
38
+
39
+ @model_validator(mode="after")
40
+ def _refuse_mixed_run(self):
41
+ # A v0 id next to any v1 env identity leaves one of the two going nowhere, and
42
+ # which one depends on `is_legacy`: a taskset makes it False, so the v0 env never
43
+ # loads; a bare `--env.id` leaves it True, so the v0 env runs under the v1 name.
44
+ if self.legacy.id is None or not self.env.env_id:
45
+ return self
46
+ if self.env.taskset.id:
47
+ raise ValueError(
48
+ f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine "
49
+ f"with the v1 taskset {self.env.taskset.id!r}. Pairing a reusable env with "
50
+ f"a taskset is --env.id {self.legacy.id!r} (TOML: id under [env]); to run "
51
+ "the v0 env instead, drop the taskset."
52
+ )
53
+ raise ValueError(
54
+ f"--legacy.id {self.legacy.id!r} is a classic (v0) env and can't combine with "
55
+ f"the v1 env --env.id {self.env.id!r}: the v0 env is what would run, stamped "
56
+ "with the v1 env's name. Keep whichever one you meant to run."
57
+ )
58
+
59
+ @model_validator(mode="before")
60
+ @classmethod
61
+ def _resolve_env(cls, data):
62
+ return resolve_env_field(data, narrowed_env_annotation(cls))
@@ -3,7 +3,7 @@ each role as an `AgentConfig` field, and the env-level knobs."""
3
3
 
4
4
  from typing import get_args
5
5
 
6
- from pydantic import SerializeAsAny, model_validator
6
+ from pydantic import Field, SerializeAsAny, model_validator
7
7
  from pydantic_config import BaseConfig
8
8
 
9
9
  from verifiers.v1.configs.agent import AgentConfig
@@ -48,9 +48,16 @@ class EnvConfig(BaseConfig):
48
48
  retries: RetryConfig = RetryConfig()
49
49
  """Whole-EPISODE retries — the coarse fallback for faults no agent owns; a
50
50
  retried episode reruns whole (a half-played sibling context isn't reproducible)."""
51
- max_concurrent: int | None = None
52
- """Bounds concurrent agent runs on a SERVED env, per worker (None = no limit);
53
- the in-process eval CLI gates with its run-level `--max-concurrent` instead."""
51
+ max_concurrent_agents: int | None = Field(1, ge=1)
52
+ """How many of ONE episode's agent runs may be active at once (None = no limit).
53
+ One at a time by default, so the bound above — `--max-concurrent` episodes, a
54
+ dispatched training episode — is also one live agent run, whatever `run()` fans
55
+ out to internally. Raise it where per-episode latency matters and the
56
+ multiplication is wanted: `--env.max-concurrent-agents None` plays an episode's
57
+ `best-of-n` attempts together, so `-c` episodes carry `-c * n` live runs. Not
58
+ spelled `max_concurrent`: that key used to bound a served worker's agent runs
59
+ across episodes, and taking it as this one silently multiplies it by the episodes
60
+ in flight."""
54
61
  interception: InterceptionConfig = ElasticInterceptionPoolConfig()
55
62
  """The interception shape: `elastic` (default), `server`, or `static`."""
56
63
 
@@ -0,0 +1,24 @@
1
+ """The `[legacy]` block: running a classic (v0) environment through the bridge.
2
+
3
+ Quarantined in its own block, not mixed into `[env]` — a v0 env is a different
4
+ thing to run, not a variant of a v1 env's config."""
5
+
6
+ from pydantic import Field
7
+ from pydantic_config import BaseConfig
8
+
9
+ from verifiers.v1.types import ID
10
+
11
+
12
+ class LegacyEnvConfig(BaseConfig):
13
+ """A classic (v0) `verifiers` environment, loaded via `verifiers.load_environment`
14
+ and run through the legacy bridge. Set `id` *instead of* a v1 `env.taskset`."""
15
+
16
+ id: ID | None = None
17
+ """v0 env id: `name`, `org/name`, or `org/name@version` — installed from the hub on
18
+ demand."""
19
+ args: dict = Field(default_factory=dict)
20
+ """Construction kwargs forwarded to `load_environment(id, **args)`."""
21
+ extra_env_kwargs: dict = Field(default_factory=dict)
22
+ """Post-load kwargs applied via `env.set_kwargs(**extra_env_kwargs)` (e.g.
23
+ `max_total_completion_tokens`, `max_seq_len`, `timeout_seconds`) — typically
24
+ auto-populated by a trainer, distinct from the `args` passed at construction."""
@@ -0,0 +1,63 @@
1
+ """How an env is served: the worker pool, where it binds, and each worker's bound.
2
+
3
+ Serving is its own axis. `[env]` says what runs; this says how it's hosted, so an
4
+ eval, a `serve` process, and a trainer configure it identically instead of each
5
+ flattening pool knobs onto its own config."""
6
+
7
+ from typing import Annotated, Literal
8
+
9
+ from pydantic import Field
10
+ from pydantic_config import BaseConfig
11
+
12
+
13
+ class StaticPoolConfig(BaseConfig):
14
+ """Fixed env-server pool: pre-spawn `num_workers` workers up front."""
15
+
16
+ type: Literal["static"] = "static"
17
+ num_workers: int = Field(4, ge=1)
18
+ """Worker processes to pre-spawn (1 = a single in-process server, no pool)."""
19
+
20
+
21
+ class ElasticPoolConfig(BaseConfig):
22
+ """Elastic env-server pool: start at one worker and scale up on demand."""
23
+
24
+ type: Literal["elastic"] = "elastic"
25
+ max_workers: int | None = None
26
+ """Upper bound on workers (None = unbounded)."""
27
+ multiplex: int = Field(128, ge=1)
28
+ """Episodes per worker for the scale-up trigger: add a worker once in-flight episodes
29
+ reach 90% of `workers * multiplex`."""
30
+
31
+
32
+ # Discriminated on `type` so the CLI selects with `--serve.pool.type static|elastic`.
33
+ PoolConfig = Annotated[
34
+ StaticPoolConfig | ElasticPoolConfig, Field(discriminator="type")
35
+ ]
36
+
37
+
38
+ class ServingConfig(BaseConfig):
39
+ """The `[serve]` block: the worker pool, the ZMQ address, and each worker's
40
+ episode bound. Read by whoever hosts the env — the `serve` CLI, a server-backed
41
+ eval, a trainer's orchestrator."""
42
+
43
+ pool: PoolConfig = Field(default_factory=ElasticPoolConfig)
44
+ """Worker-pool sizing. `elastic` (default) starts at one worker and scales up on
45
+ demand; `static` pre-spawns a fixed `num_workers`."""
46
+ address: str = "tcp://127.0.0.1:5000"
47
+ """ZMQ address the ROUTER binds (and clients connect to)."""
48
+ max_concurrent: int | None = Field(None, ge=1)
49
+ """Episodes in flight per worker (None = take the run's own bound, e.g. an eval's
50
+ `--max-concurrent`). Pin it to hold a worker below what the run asks for; how many
51
+ agent runs one episode carries is the env's own
52
+ (`--env.max-concurrent-agents`, one at a time by default)."""
53
+
54
+
55
+ def pool_serve_kwargs(pool: StaticPoolConfig | ElasticPoolConfig) -> dict:
56
+ """Unpack a pool config into `serve_env` kwargs (`max_workers` / `multiplex` / `elastic`)."""
57
+ if isinstance(pool, ElasticPoolConfig):
58
+ return {
59
+ "max_workers": pool.max_workers,
60
+ "multiplex": pool.multiplex,
61
+ "elastic": True,
62
+ }
63
+ return {"max_workers": pool.num_workers, "elastic": False}
verifiers/v1/env.py CHANGED
@@ -154,7 +154,10 @@ class Env(ABC, Generic[ConfigT]):
154
154
  """One episode: how the agents interact on `task`, returning nothing —
155
155
  every finished run joins the episode automatically, stamped with its seat's
156
156
  standing. An agent-run failure is data on its trace (this hook decides what
157
- it means); an exception raised here is the episode itself failing."""
157
+ it means); an exception raised here is the episode itself failing.
158
+ Independent agents are written as such (`asyncio.gather`); how many of them
159
+ actually run at once is the run's bound, not this hook's
160
+ (`--env.max-concurrent-agents`, one at a time by default)."""
158
161
 
159
162
  async def finalize(self, task: Task, episode: Episode) -> None:
160
163
  """Cross-agent judgement — THE programmable judgement surface: plain
@@ -188,13 +191,16 @@ class Env(ABC, Generic[ConfigT]):
188
191
  def _episode_agents(
189
192
  self,
190
193
  ctx: ModelContext,
191
- gate: "asyncio.Semaphore | None",
192
194
  completed: list[Trace],
193
195
  on_trace: Callable[[Trace], None] | None,
194
196
  on_discard: Callable[[Trace], None] | None,
195
197
  ) -> Agents:
196
198
  """One episode's `Agents` — fresh value objects riding the live serving
197
- resources (nothing shared across concurrent episodes); `setup()` sees them first."""
199
+ resources (nothing shared across concurrent episodes), sharing one semaphore
200
+ so the episode plays `--env.max-concurrent-agents` of them at a time;
201
+ `setup()` sees them first."""
202
+ limit = self.config.max_concurrent_agents
203
+ gate = asyncio.Semaphore(limit) if limit else None
198
204
 
199
205
  def make(name: str, spec: AgentConfig) -> Agent:
200
206
  # Unpinned fields fall back to the run's ctx / the taskset's harness.
@@ -242,17 +248,17 @@ class Env(ABC, Generic[ConfigT]):
242
248
  *,
243
249
  on_trace: Callable[[Trace], None] | None = None,
244
250
  on_discard: Callable[[Trace], None] | None = None,
245
- gate: asyncio.Semaphore | None = None,
246
251
  ) -> Episode:
247
252
  """One episode of `task`, minted as the wire atom: `setup()` then `run()`
248
- over fresh agents, then `finalize()`; `gate` bounds
249
- the agent runs, so internal fan-out counts against `--max-concurrent` too.
253
+ over fresh agents, then `finalize()`. Its agents run one at a time
254
+ (`--env.max-concurrent-agents`), so one live episode is one live agent run
255
+ however `run()` fans out.
250
256
  Traces join the episode as runs complete — a hook raising mid-way yields the
251
257
  completed subset, its exception on the episode's `errors`. `on_trace` observes
252
258
  each agent-run's trace at mint; `on_discard` its abandonment (a per-agent
253
259
  retry mints a replacement)."""
254
260
  episode = Episode(env=self.config.env_id)
255
- agents = self._episode_agents(ctx, gate, episode.traces, on_trace, on_discard)
261
+ agents = self._episode_agents(ctx, episode.traces, on_trace, on_discard)
256
262
  try:
257
263
  async with asyncio.timeout(self.config.timeout.episode):
258
264
  async with boundary(EnvError, f"{type(self).__name__}.setup()"):
@@ -306,8 +312,10 @@ class Env(ABC, Generic[ConfigT]):
306
312
  on_complete: Callable[[Episode], Awaitable[None]] | None = None,
307
313
  ) -> Episode:
308
314
  """Run one planned episode to completion, with whole-episode
309
- retries per `--env.retries`; `semaphore` gates the agent RUNS, not the
310
- episode; `on_complete` (the runners' persistence hook) fires when final."""
315
+ retries per `--env.retries`; `semaphore` bounds concurrent EPISODES — one
316
+ permit for the attempt in flight, held across the whole of it (its agents,
317
+ their boxes, `finalize()`) and released before a retry's backoff and before
318
+ `on_complete` (the runners' persistence hook, which fires when final)."""
311
319
 
312
320
  async def attempt() -> Episode:
313
321
  slot.traces = [] # a retry shows the fresh attempt's traces
@@ -318,13 +326,13 @@ class Env(ABC, Generic[ConfigT]):
318
326
  with contextlib.suppress(ValueError):
319
327
  live.remove(trace)
320
328
 
321
- return await self.run_episode(
322
- slot.task,
323
- ctx,
324
- on_trace=live.append,
325
- on_discard=discard,
326
- gate=semaphore,
327
- )
329
+ async with semaphore or contextlib.nullcontext():
330
+ return await self.run_episode(
331
+ slot.task,
332
+ ctx,
333
+ on_trace=live.append,
334
+ on_discard=discard,
335
+ )
328
336
 
329
337
  episode = await run_episode_with_retry(attempt, self.config.retries)
330
338
  slot.traces = list(episode.traces)
@@ -4,9 +4,9 @@ GEPA optimizes one taskset's `Task.system_prompt` by alternating rollouts (`eval
4
4
  teacher LM reflecting on the reflective dataset (`make_reflective_dataset`) — see
5
5
  `verifiers.v1.gepa.adapter.GEPAAdapter`. Like `EvalConfig`, it owns an `env` field (the
6
6
  environment: its taskset, seats, limits) and adds the optimization loop's own knobs (model,
7
- reflection model, train/val split, budget). There is no worker pool here (`EnvServerConfig`
8
- is not a base) — GEPA always runs in-process, since its adapter protocol is itself
9
- synchronous (see `GEPAAdapter`)."""
7
+ reflection model, train/val split, budget). There is no `[serve]` block here — GEPA
8
+ always runs in-process, since its adapter protocol is itself synchronous (see
9
+ `GEPAAdapter`)."""
10
10
 
11
11
  from pathlib import Path
12
12
  from uuid import uuid4
verifiers/v1/legacy.py CHANGED
@@ -495,7 +495,7 @@ def _legacy_output_dir(config) -> Path:
495
495
 
496
496
  if config.output_dir is not None:
497
497
  return config.output_dir
498
- name = f"{env_name(config.id)}--{config.model.replace('/', '--')}--legacy"
498
+ name = f"{env_name(config.legacy.id)}--{config.model.replace('/', '--')}--legacy"
499
499
  return Path("outputs") / name / config.uuid
500
500
 
501
501
 
@@ -510,15 +510,21 @@ async def run_legacy_eval(config) -> list[Episode]:
510
510
 
511
511
  # Install from the env hub on demand for an `org/name[@version]` id (a local id is
512
512
  # already importable), then load by module name.
513
- env = load_environment(ensure_installed(config.id), **(config.args or {}))
514
- if config.extra_env_kwargs: # post-load knobs (max_total_completion_tokens, …)
515
- env.set_kwargs(**config.extra_env_kwargs)
513
+ env = load_environment(
514
+ ensure_installed(config.legacy.id), **(config.legacy.args or {})
515
+ )
516
+ if (
517
+ config.legacy.extra_env_kwargs
518
+ ): # post-load knobs (max_total_completion_tokens, …)
519
+ env.set_kwargs(**config.legacy.extra_env_kwargs)
516
520
  dataset = env.get_eval_dataset() # the eval split (falls back to train when unset)
517
521
  idxs = sample(list(range(len(dataset))), config.shuffle, config.num_tasks)
518
522
 
519
523
  client = _eval_client(config.client, config.model)
520
524
  sampling_args = config.sampling.model_dump(exclude_none=True)
521
- taskset_id = env_name(config.id) # the same identity the served bridge stamps
525
+ taskset_id = env_name(
526
+ config.legacy.id
527
+ ) # the same identity the served bridge stamps
522
528
  out_dir = _legacy_output_dir(config)
523
529
  save_config(config, out_dir)
524
530
  logger.info("results: %s", out_dir)
@@ -527,7 +533,7 @@ async def run_legacy_eval(config) -> list[Episode]:
527
533
  len(idxs),
528
534
  config.num_rollouts,
529
535
  config.model,
530
- config.id,
536
+ config.legacy.id,
531
537
  )
532
538
 
533
539
  sem = asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
verifiers/v1/push.py CHANGED
@@ -186,7 +186,7 @@ def push_traces(
186
186
  return finish(error="no PRIME_API_KEY (run `prime login`)")
187
187
 
188
188
  traces = [trace for episode in episodes for trace in episode.traces]
189
- env_name = (config.env.taskset.id) or config.id
189
+ env_name = (config.env.taskset.id) or config.legacy.id
190
190
  metrics = _run_metrics(episodes, traces)
191
191
  samples = _build_samples(episodes)
192
192
  num_examples = len({t.task.data.idx for t in traces})
@@ -312,9 +312,10 @@ def serve_env(
312
312
  if (
313
313
  "config" in server_kwargs
314
314
  ): # dict-ify for the workers (config_data is picklable)
315
- server_kwargs = {
316
- "config_data": env_config_data(server_kwargs["config"])
317
- }
315
+ server_kwargs = {**server_kwargs}
316
+ server_kwargs["config_data"] = env_config_data(
317
+ server_kwargs.pop("config")
318
+ )
318
319
  pool = EnvServerPool(
319
320
  server_kwargs,
320
321
  max_workers,
@@ -335,9 +336,10 @@ def serve_env(
335
336
  ): # rebuild the env config for an in-process server
336
337
  from verifiers.v1.loaders import resolve_env_config
337
338
 
338
- server_kwargs = {
339
- "config": resolve_env_config(server_kwargs["config_data"])
340
- }
339
+ server_kwargs = {**server_kwargs}
340
+ server_kwargs["config"] = resolve_env_config(
341
+ server_kwargs.pop("config_data")
342
+ )
341
343
  cls = LegacyEnvServer if legacy else EnvServer
342
344
  cls.run_server(
343
345
  address=address, address_queue=address_queue, **server_kwargs
@@ -30,7 +30,10 @@ logger = logging.getLogger(__name__)
30
30
 
31
31
  class EnvServer:
32
32
  def __init__(
33
- self, config: EnvConfig, address: str = "tcp://127.0.0.1:5000"
33
+ self,
34
+ config: EnvConfig,
35
+ address: str = "tcp://127.0.0.1:5000",
36
+ max_concurrent: int | None = None,
34
37
  ) -> None:
35
38
  self.address = address
36
39
  self.taskset_id = config.taskset.id
@@ -52,9 +55,8 @@ class EnvServer:
52
55
  # v1 envs never group-score (siblings score inside the env's own rollout);
53
56
  # only the legacy (v0) bridge sets this.
54
57
  self.requires_group_scoring = False
55
- self._gate = (
56
- asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
57
- )
58
+ # This worker's episode bound (`--max-concurrent`), spanning requests.
59
+ self._gate = asyncio.Semaphore(max_concurrent) if max_concurrent else None
58
60
  self._clients: dict[
59
61
  tuple[str, str], Client
60
62
  ] = {} # (client_config, model) -> Client
@@ -122,8 +124,8 @@ class EnvServer:
122
124
  async def _run(self, req: RunRequest) -> RunResponse:
123
125
  ctx = self._context(req.client, req.model, req.sampling)
124
126
  (slot,) = self.env.slots(self._build_task(req.task_data))
125
- # The gate spans requests: `--env.max-concurrent` bounds this worker's
126
- # agent runs the same way the in-process eval's semaphore does.
127
+ # The gate spans requests: `--max-concurrent` bounds this worker's episodes
128
+ # in flight the same way the in-process eval's semaphore does.
127
129
  episode = await self.env.run_slot(slot, ctx, self._gate)
128
130
  # Trust the env-minted episode; serialize it once before client-side re-typing.
129
131
  return RunResponse.model_construct(episode=episode)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev42
3
+ Version: 0.2.2.dev44
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -39,7 +39,7 @@ Requires-Dist: prime-sandboxes>=0.2.33
39
39
  Requires-Dist: prime-tunnel>=0.1.8
40
40
  Requires-Dist: pydantic>=2.12.3
41
41
  Requires-Dist: pyzmq>=27.1.0
42
- Requires-Dist: renderers>=0.1.8
42
+ Requires-Dist: renderers>=0.1.9.dev9
43
43
  Requires-Dist: requests
44
44
  Requires-Dist: rich>=11.0.0
45
45
  Requires-Dist: setproctitle>=1.3.0
@@ -167,18 +167,18 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
167
167
  verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
168
168
  verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
169
169
  verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
170
- verifiers/v1/__init__.py,sha256=QC3A5qz54RGO9WHAUg35zVt-Gde7JLf8ROCrwSLLy9U,7119
171
- verifiers/v1/agent.py,sha256=MAwqTuUcdBPlPwQ-DTYCwZI-duQH-h5PAS5lKxPpfeI,31875
170
+ verifiers/v1/__init__.py,sha256=xHKE-HZA8bN0JJovf4KbBI2mxBtZJ4hPNnzac4FtWwM,7332
171
+ verifiers/v1/agent.py,sha256=2MJBztogGCwDMt0DVZuuItADgZpTk3Xaf_x_cQswdds,31956
172
172
  verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
173
- verifiers/v1/env.py,sha256=QFwhD1NKbEUWGf0l6G2ttK4JLbhUt2MUCDciWaaJ_1s,17834
173
+ verifiers/v1/env.py,sha256=Omfwspenzs7A_K-0oDVPH1lHS8YkzihmJEZ2DjK4Kqs,18445
174
174
  verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
175
175
  verifiers/v1/errors.py,sha256=Pj5Om8x1fP3TDPQQn6nUCoSPoZRsy2JeBz8pXhpPrDY,6883
176
176
  verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
177
177
  verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
178
178
  verifiers/v1/judge.py,sha256=VnTi9iSfCJMCccYWc1qCTcT7iD3AcomS2cXfHwh1HVw,9423
179
- verifiers/v1/legacy.py,sha256=kGfx4CEmk17VEqwX6JfUBbjQRMqJ0tL4kkA2kqGypN8,22305
179
+ verifiers/v1/legacy.py,sha256=aPpHsiNyW3Ail3U8Nfx9XM6tS__Py_haEluiZWVQ50k,22398
180
180
  verifiers/v1/loaders.py,sha256=TnNKY2msVo3p7m-5wjw376PYH-zTu3TLixXRst0VOiM,9985
181
- verifiers/v1/push.py,sha256=LstxyShYGWl-tXRyuSoA_YIXytdydPVjhTJET_o3ZVI,11334
181
+ verifiers/v1/push.py,sha256=uq4ZQBYNPQhNsNORNmlWEwTL186E4deGXGqG6fZBQgM,11341
182
182
  verifiers/v1/retries.py,sha256=GAUQB2k-SJGnKjLpI1Y_XOmhSBeqIquljkXQHXYfg88,5380
183
183
  verifiers/v1/rollout.py,sha256=eU01fZRf87lQ5LQUVIDN38SlcUE0svXxfyS0gryNboE,20503
184
184
  verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
@@ -197,17 +197,17 @@ verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
197
197
  verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
198
198
  verifiers/v1/cli/replay.py,sha256=T4sg-MpvT9niymEJl0Y2TijgTy-tVQqjT1lQ3nwuOg8,10314
199
199
  verifiers/v1/cli/resolve.py,sha256=1sV4z04v_HEs86Oi8ODJqfujXjrbShbu3u2kDpgpj94,4079
200
- verifiers/v1/cli/serve.py,sha256=CIO6stXkIe-CozLpJMQZDRNAYyvVNyu3TrDtpiG3r-4,2667
200
+ verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
201
201
  verifiers/v1/cli/validate.py,sha256=PMWRSZROiTh2KFzkA3Ez5-z09krVk2ZUTkFNMUqk2-4,9671
202
202
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
203
203
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
204
- verifiers/v1/cli/dashboard/eval.py,sha256=HcdZKSkHVcA-lLMDgIZyI5coWOvp1l9w-id7Om38V7c,34357
204
+ verifiers/v1/cli/dashboard/eval.py,sha256=lPrP7Ikmz3r7v3ury2yAMLgYeweVGFdg0ZKSQ6lwwKU,34355
205
205
  verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
206
206
  verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
207
207
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
208
- verifiers/v1/cli/eval/main.py,sha256=bOAClP3qn_0JnyWu0Or5nLcl-cX-Y9AqKb2NrBFbA64,5481
208
+ verifiers/v1/cli/eval/main.py,sha256=zvbLy_ryn3S1fWoTmWL4QEReVO1Q_hIrgyK6Fp0q_ZM,5501
209
209
  verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
210
- verifiers/v1/cli/eval/runner.py,sha256=dKzeCTMRwUWwOLH15s3pCqZgRZmI03PYuRbzvw0IKDU,11212
210
+ verifiers/v1/cli/eval/runner.py,sha256=PwcIaVLGBJON9nC2u3z6aTGALmoa0vKAYrg0IYM-8Dw,11493
211
211
  verifiers/v1/clients/__init__.py,sha256=bHGS5jfrAZ-VBUeXpasXkkKdHkFgjOSN9KbOJXrnFtU,511
212
212
  verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
213
213
  verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
@@ -215,19 +215,21 @@ verifiers/v1/clients/eval.py,sha256=QMpohNnQjDuo56BhOgHnAfAlcnQAxFpEFCySoIuexNg,
215
215
  verifiers/v1/clients/train.py,sha256=6bw4TV_UskdjvDAxvHrgJWBqAq9r1DxzJGusGi7UoRM,13022
216
216
  verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxVtvA,317
217
217
  verifiers/v1/configs/agent.py,sha256=N2Lm0iZXFxfsSIiXUQ7f3RDZcoin3M7pJj6HVdCE5eo,3537
218
- verifiers/v1/configs/env.py,sha256=2mEU1n2xG-v5PNdB2EJioAYP6Witnlny9kYi0jwKFQk,7286
218
+ verifiers/v1/configs/env.py,sha256=4aUap3BJX3WCA2_fZ-jGPxF3khiVpORqmz_3eBT4j9w,7824
219
219
  verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Vagg,1564
220
220
  verifiers/v1/configs/judge.py,sha256=ZJlmWogHY8oDjvWvtNuVKvkzuheWalIjl1njJ4dE0Mk,2511
221
+ verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
221
222
  verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
223
+ verifiers/v1/configs/serve.py,sha256=cHHnFGQtqEwvaovdeoVMpCebTHgPXKUuIivN5IhxZR0,2596
222
224
  verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
223
225
  verifiers/v1/configs/taskset.py,sha256=IdZwsarn0_WvYdR_Kr4QYYdqbaoGYkbYyO5VA-RPYPc,657
224
226
  verifiers/v1/configs/cli/__init__.py,sha256=3wBAONw5XkBPqbMjpHk5XBHlBVybR8JQ2JCPWijYBMU,444
225
227
  verifiers/v1/configs/cli/debug.py,sha256=3lkKqmQFLJBcUZECChBHCrod5pbCKSmT47EhjILrxjI,2843
226
- verifiers/v1/configs/cli/env.py,sha256=Rw7KtAR-mKowlnn8Vc-dGpOWfknUyS6SyIu4jpGt1sk,7359
227
- verifiers/v1/configs/cli/eval.py,sha256=A5k7Iigd-mlv_2TMF-b3DNbneMhflO0fSWSOI9oRJtA,3812
228
+ verifiers/v1/configs/cli/env.py,sha256=XgjRXYpPjga0ZYIHEESY7VuVvtE6pWfAzf6JBLeuTIc,2695
229
+ verifiers/v1/configs/cli/eval.py,sha256=YWYMvPOIAG51e0JdmyA-U_P7eInMPZ_8DTFim6aWn6Y,6869
228
230
  verifiers/v1/configs/cli/init.py,sha256=tMaTDF_lU1Af3JmQpksk2mwe8CAfpnHsquXhwepZhYo,1019
229
231
  verifiers/v1/configs/cli/replay.py,sha256=57R-OZnVXPyrRVrWq2UeBBhnBgTdjGs1DJ_PWjNU3AY,2907
230
- verifiers/v1/configs/cli/serve.py,sha256=p5xLSVPBTX5DfjqWX9bsFbvFZqmOXYdmMOiII6Nntuk,573
232
+ verifiers/v1/configs/cli/serve.py,sha256=1rAt0QOL-XT3DMmRUIA40t1E3YuoD1vOADfsiTwFBf0,2981
231
233
  verifiers/v1/configs/cli/validate.py,sha256=xaYUjQ0MEAAbA_jp55zAtYxlTettZivUC0Tu1H1WZ50,2292
232
234
  verifiers/v1/dialects/__init__.py,sha256=p72C7VYrFQzVgTcfmzpHGSwXlvThfc4pvjBxsqw7kHE,786
233
235
  verifiers/v1/dialects/anthropic.py,sha256=nj9J7OOhexmPHa-sWiNP2xlI9_bGhvwM5VWzdh19pBc,13526
@@ -245,7 +247,7 @@ verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-
245
247
  verifiers/v1/envs/user_sim/env.py,sha256=u376-UJEVdzgcit7FXjz5KEwlPxSU3qqoIrxs6foi2w,4567
246
248
  verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
247
249
  verifiers/v1/gepa/adapter.py,sha256=Wo3adNf1I3eT4L_03vxWrrg6HvZBz5KDQonCsGdhLgg,6185
248
- verifiers/v1/gepa/config.py,sha256=MVaJfPMehoPq5kF1vsm3lEJrxTiryesxHrQDcxE6TAE,4656
250
+ verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4626
249
251
  verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
250
252
  verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
251
253
  verifiers/v1/gepa/runner.py,sha256=o7Am2Ram-Ejbr_tVpN70zb0uQ5a1NIpXvlMhwWL22Xs,5406
@@ -301,8 +303,8 @@ verifiers/v1/runtimes/docker/__init__.py,sha256=eC5hjfyPwvwEKaeZUeP9EN8tJkrl0HUx
301
303
  verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
302
304
  verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
303
305
  verifiers/v1/serve/client.py,sha256=amUhf4cFJ1xCb8kw_AoVviwN543Wknx8LzImq1ZpaiQ,7057
304
- verifiers/v1/serve/pool.py,sha256=cqSGUJvSavV4o73-PisacP5EyEXnQH_DCL9oFuBL5YU,14724
305
- verifiers/v1/serve/server.py,sha256=gTrSZYSG2DH-KpAhLMN2L1oaMC6oJrqtguNrDfgHQFU,9600
306
+ verifiers/v1/serve/pool.py,sha256=e9Nc3nQKEyV9YqyKHl_RuqQ84zWVa8z2J1rc3rfCadM,14828
307
+ verifiers/v1/serve/server.py,sha256=VBOD3Gzq4p0ZD_8xgtG039i3JaefhOH0MmR6cxG56iY,9705
306
308
  verifiers/v1/serve/types.py,sha256=AHFbt8C-Rzlee34RDPLQ3lPiYDLkUm5eFMnRQINiO6s,3016
307
309
  verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
308
310
  verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
@@ -327,8 +329,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
327
329
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
328
330
  verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
329
331
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
330
- verifiers-0.2.2.dev42.dist-info/METADATA,sha256=Sfv8yHjesqeUSatEPoWH_rvAosXXPvZ2xczF2O7_mzI,4540
331
- verifiers-0.2.2.dev42.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
332
- verifiers-0.2.2.dev42.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
333
- verifiers-0.2.2.dev42.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
334
- verifiers-0.2.2.dev42.dist-info/RECORD,,
332
+ verifiers-0.2.2.dev44.dist-info/METADATA,sha256=xAsUe3hwzJqF-nbNTppTvycNvcRTi_cnhIq8oFj9adQ,4545
333
+ verifiers-0.2.2.dev44.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
334
+ verifiers-0.2.2.dev44.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
335
+ verifiers-0.2.2.dev44.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
336
+ verifiers-0.2.2.dev44.dist-info/RECORD,,