verifiers 0.2.2.dev49__py3-none-any.whl → 0.2.2.dev51__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/cli/debug.py CHANGED
@@ -263,7 +263,14 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
263
263
 
264
264
  async def run_debug(config: DebugConfig) -> list[Trace]:
265
265
  taskset = vf.load_taskset(config.taskset)
266
- tasks = taskset.select(config.num_tasks, config.shuffle)
266
+ if config.num_tasks is None and taskset.INFINITE:
267
+ raise ValueError(
268
+ f"{type(taskset).__name__} is infinite - bound the run with -n"
269
+ )
270
+ selected = taskset.shuffle() if config.shuffle else taskset
271
+ if config.num_tasks is not None:
272
+ selected = selected.head(config.num_tasks)
273
+ tasks = list(selected)
267
274
  if isinstance(config.runtime, vf.SubprocessConfig) and any(
268
275
  type(t).NEEDS_CONTAINER or t.data.image for t in tasks
269
276
  ):
@@ -26,7 +26,15 @@ logger = logging.getLogger(__name__)
26
26
  async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
27
27
  logger.info("eval config:\n%s", config.model_dump_json(indent=2))
28
28
  client = resolve_client(config.client)
29
- tasks = env.taskset.select(config.num_tasks, config.shuffle)
29
+ taskset = env.taskset
30
+ if config.num_tasks is None and taskset.INFINITE:
31
+ raise ValueError(
32
+ f"{type(taskset).__name__} is infinite - bound the run with -n"
33
+ )
34
+ selected = taskset.shuffle() if config.shuffle else taskset
35
+ if config.num_tasks is not None:
36
+ selected = selected.head(config.num_tasks)
37
+ tasks = list(selected)
30
38
  ctx = ModelContext(client=client, model=config.model, sampling=config.sampling)
31
39
  semaphore = (
32
40
  asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
@@ -132,9 +140,15 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
132
140
 
133
141
  # The client owns the taskset: load it here, once — the server (and its pool
134
142
  # workers) never load data, they rebuild each dispatched task from its request.
135
- tasks = load_taskset(config.env.taskset).select(
136
- config.num_tasks, config.shuffle
137
- )
143
+ taskset = load_taskset(config.env.taskset)
144
+ if config.num_tasks is None and taskset.INFINITE:
145
+ raise ValueError(
146
+ f"{type(taskset).__name__} is infinite - bound the run with -n"
147
+ )
148
+ selected = taskset.shuffle() if config.shuffle else taskset
149
+ if config.num_tasks is not None:
150
+ selected = selected.head(config.num_tasks)
151
+ tasks = list(selected)
138
152
  # Spawned processes inherit no logging — hand them the main process's setup so
139
153
  # their rollout logs land in the output dir.
140
154
  level = "DEBUG" if config.verbose else "INFO"
@@ -203,7 +203,14 @@ async def _validate_task(task: Task, config: ValidateConfig) -> ResultRow:
203
203
 
204
204
  async def run_validate(config: ValidateConfig) -> list[dict]:
205
205
  taskset = vf.load_taskset(config.taskset)
206
- tasks = taskset.select(config.num_tasks, config.shuffle)
206
+ if config.num_tasks is None and taskset.INFINITE:
207
+ raise ValueError(
208
+ f"{type(taskset).__name__} is infinite - bound the run with -n"
209
+ )
210
+ selected = taskset.shuffle() if config.shuffle else taskset
211
+ if config.num_tasks is not None:
212
+ selected = selected.head(config.num_tasks)
213
+ tasks = list(selected)
207
214
  if isinstance(config.runtime, vf.SubprocessConfig) and any(
208
215
  type(t).NEEDS_CONTAINER or t.data.image for t in tasks
209
216
  ):
@@ -5,7 +5,7 @@ from collections.abc import Sequence
5
5
  from pathlib import Path
6
6
  from typing import Any
7
7
 
8
- from pydantic import BaseModel, SerializeAsAny, model_validator
8
+ from pydantic import BaseModel, SerializeAsAny
9
9
 
10
10
  from verifiers.v1.clients import BaseClientConfig
11
11
  from verifiers.v1.types import ID, SamplingConfig
@@ -20,15 +20,8 @@ class JudgeConfig(BaseClientConfig):
20
20
  weight: float = 1.0
21
21
  model: str = "openai/gpt-5.4-nano"
22
22
  sampling: SamplingConfig = SamplingConfig()
23
- prompt: str | None = None
24
- prompt_file: Path | None = None
25
- """Prompt file override, mutually exclusive with `prompt`."""
26
-
27
- @model_validator(mode="after")
28
- def check_prompt_source(self) -> "JudgeConfig":
29
- if self.prompt is not None and self.prompt_file is not None:
30
- raise ValueError("set `prompt` or `prompt_file`, not both")
31
- return self
23
+ prompt: Path | None = None
24
+ """File whose text overrides the judge's default prompt template."""
32
25
 
33
26
 
34
27
  Judges = list[SerializeAsAny[JudgeConfig]]
@@ -1,5 +1,7 @@
1
1
  """The taskset plugin's config: which rows load, under `--env.taskset.*`."""
2
2
 
3
+ from pathlib import Path
4
+
3
5
  from pydantic import SerializeAsAny
4
6
  from pydantic_config import BaseConfig
5
7
 
@@ -14,6 +16,9 @@ class TasksetConfig(BaseConfig):
14
16
  positional `eval <taskset-id>`)."""
15
17
  task: SerializeAsAny[TaskConfig] = TaskConfig()
16
18
  """Config passed to each task, under `--env.taskset.task.*`."""
19
+ system_prompt: Path | None = None
20
+ """File whose text overrides each task's `TaskData.system_prompt` on
21
+ iteration (e.g. a GEPA `best_system_prompt.txt`)."""
17
22
 
18
23
  @property
19
24
  def name(self) -> str:
verifiers/v1/env.py CHANGED
@@ -33,7 +33,7 @@ from verifiers.v1.retries import run_episode_with_retry
33
33
  from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
34
34
  from verifiers.v1.task import Task, resolve_server_config
35
35
  from verifiers.v1.trace import Error, Trace
36
- from verifiers.v1.utils.generic import generic_type
36
+ from verifiers.v1.utils.generic import concrete_type
37
37
  from verifiers.v1.utils.memory import trim_memory_periodically
38
38
 
39
39
  logger = logging.getLogger(__name__)
@@ -83,7 +83,7 @@ class Env(ABC, Generic[ConfigT]):
83
83
  def __init__(self, config: ConfigT) -> None:
84
84
  from verifiers.v1.loaders import load_harness, load_taskset
85
85
 
86
- config_cls = generic_type(type(self), EnvConfig, origin=Env) or EnvConfig
86
+ config_cls = concrete_type(type(self), EnvConfig, origin=Env) or EnvConfig
87
87
  if not isinstance(config, config_cls):
88
88
  raise TypeError(
89
89
  f"{type(self).__name__} declares Env[{config_cls.__name__}], "
@@ -180,41 +180,31 @@ class JudgeTask(vf.Task):
180
180
  class JudgeTaskConfig(vf.BaseConfig):
181
181
  """The judge's minted task: the grading policy and what lands in its box."""
182
182
 
183
- prompt: Path | str | None = None
184
- """Grading-policy override: inline text, or a policy file (a value ending in
185
- `.md`/`.txt` is read from disk). Replaces only the policy body — the verdict
186
- contract and workspace note are always appended, so a custom policy cannot
187
- break verdict scraping. May reference `{prompt}` (the solver task's prompt);
188
- if it doesn't, the task statement is appended after the policy."""
189
- hint: Path | str | None = None
190
- """Optional hints injected as their own section (inline text or a `.md`/
191
- `.txt` file): task-family pointers into the trace or box — e.g. for math,
192
- where the reference answer lives in the record; for SWE, to diff the repo
193
- or read `info.patch`."""
183
+ prompt: Path | None = None
184
+ """Grading-policy file. Replaces only the policy body — the verdict contract
185
+ and workspace note are always appended. May reference `{prompt}` (the solver
186
+ task's prompt); if it doesn't, the task statement is appended after."""
187
+ hint: Path | None = None
188
+ """Optional hints file injected as their own section: task-family pointers
189
+ into the trace or box — e.g. for math, where the reference answer lives in
190
+ the record; for SWE, to diff the repo or read `info.patch`."""
194
191
  rubric: Path | None = None
195
192
  """Criteria the judge grades against: a `.toml`/`.json` file with a
196
193
  `criteria` list — the plugged rubric judge's format, so the same rubric
197
194
  files work for both. None grades the single built-in `solved` criterion."""
198
195
 
199
- @staticmethod
200
- def _resolve(value: Path | str) -> str:
201
- path = Path(value)
202
- if isinstance(value, Path) or path.suffix in (".md", ".txt"):
203
- return path.read_text(encoding="utf-8")
204
- return str(value)
205
-
206
196
  def build_prompt(self) -> str:
207
197
  if self.prompt is None:
208
198
  return GRADE_PROMPT + "\n\n" + TASK_SECTION
209
- return self._resolve(self.prompt)
199
+ return self.prompt.read_text()
210
200
 
211
201
  def build_hint(self) -> str | None:
212
- return self._resolve(self.hint) if self.hint is not None else None
202
+ return self.hint.read_text() if self.hint is not None else None
213
203
 
214
204
  def criteria(self) -> list[Criterion]:
215
205
  if self.rubric is None:
216
206
  return [SOLVED]
217
- text = self.rubric.read_text(encoding="utf-8")
207
+ text = self.rubric.read_text()
218
208
  data = (
219
209
  tomllib.loads(text)
220
210
  if self.rubric.suffix.lower() == ".toml"
@@ -68,14 +68,10 @@ class GEPAAdapter:
68
68
  )
69
69
 
70
70
  async def _run_batch(self, batch: list[int], system_prompt: str) -> list[Episode]:
71
- # Inject the candidate by rebuilding each Task around a data row with the new
72
- # system_prompt (TaskData is frozen; behavior/config carry over unchanged).
73
- tasks = [
74
- type(t)(
75
- t.data.model_copy(update={"system_prompt": system_prompt}), t.config
76
- )
77
- for t in (self.tasks[idx] for idx in batch)
78
- ]
71
+ # Inject the candidate as a copy of each base task with its system_prompt overridden
72
+ # (`with_system_prompt` copies rather than reconstructs, so subclass state survives and
73
+ # the shared base task in `self.tasks` is left untouched for the next candidate).
74
+ tasks = [self.tasks[idx].with_system_prompt(system_prompt) for idx in batch]
79
75
  slots = [slot for task in tasks for slot in self.env.slots(task)]
80
76
  results = await asyncio.gather(
81
77
  *(
@@ -37,7 +37,10 @@ class _GEPALog:
37
37
 
38
38
  def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
39
39
  logger.info("gepa config:\n%s", config.model_dump_json(indent=2))
40
- all_tasks = env.taskset.select(config.num_train + config.num_val, config.shuffle)
40
+ # Global shuffle: an infinite taskset raises here — run it with
41
+ # shuffle=false (there is no whole set to sample from).
42
+ taskset = env.taskset.shuffle() if config.shuffle else env.taskset
43
+ all_tasks = list(taskset.head(config.num_train + config.num_val))
41
44
  train_tasks, val_tasks = split_tasks(all_tasks, config.num_train, config.num_val)
42
45
  selected_tasks = [*train_tasks, *val_tasks]
43
46
  # Seed from the tasks GEPA actually evaluates (train ∪ val), not the full pre-split pool —
@@ -105,7 +108,20 @@ def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
105
108
  "skip_perfect_score": False,
106
109
  "logger": _GEPALog(),
107
110
  }
108
- return optimize(**optimize_kwargs)
111
+ result = optimize(**optimize_kwargs)
112
+ if run_dir is not None:
113
+ # Persist the winning prompt as a plain file so it can be handed straight to
114
+ # eval/train via `--env.taskset.system-prompt` (see TasksetConfig).
115
+ candidate = result.best_candidate
116
+ best = (
117
+ candidate.get("system_prompt", "")
118
+ if isinstance(candidate, dict)
119
+ else str(candidate)
120
+ )
121
+ best_path = run_dir / "best_system_prompt.txt"
122
+ best_path.write_text(best, encoding="utf-8")
123
+ logger.info("best system prompt: %s", best_path)
124
+ return result
109
125
  finally:
110
126
  loop.run_until_complete(serving.__aexit__(None, None, None))
111
127
  finally:
verifiers/v1/judge.py CHANGED
@@ -63,7 +63,7 @@ from verifiers.v1.configs.judge import (
63
63
  from verifiers.v1.dialects.chat import message_to_wire
64
64
  from verifiers.v1.scoring import parse_judge_choice
65
65
  from verifiers.v1.types import Messages, StrictBaseModel, Usage
66
- from verifiers.v1.utils.generic import generic_type
66
+ from verifiers.v1.utils.generic import concrete_type
67
67
 
68
68
  if TYPE_CHECKING:
69
69
  from verifiers.v1.task import TaskData
@@ -111,18 +111,18 @@ ConfigT = TypeVar("ConfigT", bound=JudgeConfig, default=JudgeConfig)
111
111
 
112
112
  def judge_config_cls(cls: type) -> type[JudgeConfig]:
113
113
  """Resolve a judge's config specialization through its MRO, else `JudgeConfig`."""
114
- return generic_type(cls, JudgeConfig) or JudgeConfig
114
+ return concrete_type(cls, JudgeConfig) or JudgeConfig
115
115
 
116
116
 
117
117
  class Judge(Generic[ParsedT, ConfigT]):
118
118
  prompt: str | None = None
119
- """Default prompt template, overridden by config."""
119
+ """Default prompt template, overridden by a config `prompt` file."""
120
120
  schema: type[BaseModel] | None = None
121
121
 
122
122
  def __init__(self, config: ConfigT | None = None) -> None:
123
123
  self.config = cast(ConfigT, config or judge_config_cls(type(self))())
124
- if self.config.prompt_file is not None:
125
- self.prompt = self.config.prompt_file.read_text(encoding="utf-8")
124
+ if self.config.prompt is not None:
125
+ self.prompt = self.config.prompt.read_text()
126
126
 
127
127
  @property
128
128
  def reward_name(self) -> str:
@@ -132,7 +132,7 @@ class Judge(Generic[ParsedT, ConfigT]):
132
132
  return judge_key(self.config) or fallback or "judge"
133
133
 
134
134
  def build_messages(self, **fields: Any) -> str | Messages:
135
- template = self.config.prompt or self.prompt
135
+ template = self.prompt
136
136
  if template is None:
137
137
  raise ValueError(
138
138
  f"{type(self).__name__} has no `prompt`; set it or override build_messages"
verifiers/v1/loaders.py CHANGED
@@ -20,7 +20,7 @@ from verifiers.v1.harness import Harness
20
20
  from verifiers.v1.judge import Judge, judge_config_cls
21
21
  from verifiers.v1.task import Task
22
22
  from verifiers.v1.taskset import Taskset
23
- from verifiers.v1.utils.generic import generic_type, prefix_validation_error
23
+ from verifiers.v1.utils.generic import concrete_type, prefix_validation_error
24
24
  from verifiers.v1.utils.install import ensure_installed
25
25
 
26
26
 
@@ -208,7 +208,7 @@ def load_judge(config: JudgeConfig) -> Judge:
208
208
  def taskset_config_type(taskset_id: str) -> type[TasksetConfig]:
209
209
  """Resolve the taskset's config specialization through its MRO."""
210
210
  return (
211
- generic_type(taskset_class(taskset_id), TasksetConfig, origin=Taskset)
211
+ concrete_type(taskset_class(taskset_id), TasksetConfig, origin=Taskset)
212
212
  or TasksetConfig
213
213
  )
214
214
 
@@ -216,7 +216,7 @@ def taskset_config_type(taskset_id: str) -> type[TasksetConfig]:
216
216
  def harness_config_type(harness_id: str) -> type[HarnessConfig]:
217
217
  """Resolve the harness's config specialization through its MRO."""
218
218
  return (
219
- generic_type(harness_class(harness_id), HarnessConfig, origin=Harness)
219
+ concrete_type(harness_class(harness_id), HarnessConfig, origin=Harness)
220
220
  or HarnessConfig
221
221
  )
222
222
 
@@ -231,7 +231,7 @@ def env_config_type(taskset_id: str, env_id: str = "") -> type[EnvConfig]:
231
231
  its MRO — `SingleAgentEnvConfig` for a plain taskset. The run's `env` field
232
232
  narrows to this, which is what gives `--env.<role>.model` addressing."""
233
233
  return (
234
- generic_type(environment_class(taskset_id, env_id), EnvConfig, origin=Env)
234
+ concrete_type(environment_class(taskset_id, env_id), EnvConfig, origin=Env)
235
235
  or EnvConfig
236
236
  )
237
237
 
@@ -13,7 +13,7 @@ from pydantic import TypeAdapter, ValidationError
13
13
  from pydantic_config import BaseConfig
14
14
 
15
15
  from verifiers.v1.state import State, StateT, state_cls
16
- from verifiers.v1.utils.generic import generic_type
16
+ from verifiers.v1.utils.generic import concrete_type
17
17
 
18
18
  if TYPE_CHECKING:
19
19
  from httpx import AsyncClient, Response
@@ -286,7 +286,7 @@ class ServerBase(Generic[ConfigT, StateT]):
286
286
  @classmethod
287
287
  def _config_cls(cls) -> type[BaseConfig]:
288
288
  """Resolve the server's config specialization through its MRO."""
289
- if config_cls := generic_type(cls, BaseConfig):
289
+ if config_cls := concrete_type(cls, BaseConfig):
290
290
  return config_cls
291
291
  raise TypeError(
292
292
  f"{cls.__name__} must parameterize its config, e.g. Toolset[MyConfig]"
verifiers/v1/state.py CHANGED
@@ -8,7 +8,7 @@ from pydantic import ConfigDict
8
8
  from typing_extensions import TypeVar
9
9
 
10
10
  from verifiers.v1.types import StrictBaseModel
11
- from verifiers.v1.utils.generic import generic_type
11
+ from verifiers.v1.utils.generic import concrete_type
12
12
 
13
13
 
14
14
  class State(StrictBaseModel):
@@ -20,4 +20,4 @@ StateT = TypeVar("StateT", bound=State, default=State)
20
20
 
21
21
  def state_cls(cls: type) -> type[State]:
22
22
  """Resolve a class's `State` specialization through its MRO, else `State`."""
23
- return generic_type(cls, State) or State
23
+ return concrete_type(cls, State) or State
verifiers/v1/task.py CHANGED
@@ -24,10 +24,11 @@ wraps it in the declared `Task` — one task type per taskset.
24
24
 
25
25
  from __future__ import annotations
26
26
 
27
+ import copy
27
28
  import inspect
28
29
  import logging
29
30
  from collections.abc import Mapping
30
- from typing import TYPE_CHECKING, ClassVar, Generic
31
+ from typing import TYPE_CHECKING, ClassVar, Generic, Self
31
32
 
32
33
  from pydantic import ConfigDict, Field
33
34
  from pydantic_config import BaseConfig
@@ -38,7 +39,7 @@ from verifiers.v1.decorators import discover_decorated, invoke_all
38
39
  from verifiers.v1.errors import TaskError, boundary
39
40
  from verifiers.v1.state import StateT
40
41
  from verifiers.v1.types import Messages, StrictBaseModel, content_text
41
- from verifiers.v1.utils.generic import generic_type
42
+ from verifiers.v1.utils.generic import concrete_type
42
43
 
43
44
  if TYPE_CHECKING:
44
45
  from verifiers.v1.judge import Judge
@@ -49,27 +50,6 @@ if TYPE_CHECKING:
49
50
  logger = logging.getLogger(__name__)
50
51
 
51
52
 
52
- def _requires_runtime(fn) -> bool:
53
- param = inspect.signature(fn).parameters.get("runtime")
54
- # A defaulted runtime parameter can still be called offline with None.
55
- return param is not None and param.default is inspect.Parameter.empty
56
-
57
-
58
- def _record_result(
59
- trace: Trace,
60
- name: str,
61
- result,
62
- weight: float | None = None,
63
- ) -> None:
64
- """Record a scalar or keyed scoring result."""
65
- values = list(result.items()) if isinstance(result, Mapping) else [(name, result)]
66
- for key, value in values:
67
- if weight is None:
68
- trace.record_metric(key, value)
69
- else:
70
- trace.record_reward(key, value, weight)
71
-
72
-
73
53
  class TaskResources(StrictBaseModel):
74
54
  model_config = ConfigDict(frozen=True)
75
55
 
@@ -148,12 +128,12 @@ ConfigT = TypeVar("ConfigT", bound=TaskConfig, default=TaskConfig)
148
128
 
149
129
  def task_data_cls(cls: type) -> type[TaskData]:
150
130
  """Resolve a task's `TaskData` specialization through its MRO, else `TaskData`."""
151
- return generic_type(cls, TaskData) or TaskData
131
+ return concrete_type(cls, TaskData) or TaskData
152
132
 
153
133
 
154
134
  def task_config_cls(cls: type) -> type[TaskConfig]:
155
135
  """Resolve a task's `TaskConfig` specialization through its MRO, else `TaskConfig`."""
156
- return generic_type(cls, TaskConfig) or TaskConfig
136
+ return concrete_type(cls, TaskConfig) or TaskConfig
157
137
 
158
138
 
159
139
  def resolve_server_config(
@@ -197,6 +177,15 @@ class Task(Generic[DataT, StateT, ConfigT]):
197
177
  self.data = data
198
178
  self.config = config if config is not None else task_config_cls(type(self))()
199
179
 
180
+ def with_system_prompt(self, system_prompt: str) -> Self:
181
+ """A shallow copy of this task with `data.system_prompt` overridden. Copies the
182
+ instance instead of reconstructing via `type(self)(...)`, so a subclass with a
183
+ non-`(data, config)` constructor or extra load-time state keeps it. Used to apply the
184
+ config-layer / GEPA system prompt (see `TasksetConfig` and `verifiers.v1.gepa`)."""
185
+ clone = copy.copy(self)
186
+ clone.data = self.data.model_copy(update={"system_prompt": system_prompt})
187
+ return clone
188
+
200
189
  def plugged_judges(self) -> list[Judge]:
201
190
  from verifiers.v1.loaders import load_judge
202
191
 
@@ -229,6 +218,11 @@ class Task(Generic[DataT, StateT, ConfigT]):
229
218
  trace: Trace,
230
219
  runtime: Runtime | None = None,
231
220
  ) -> None:
221
+ def requires_runtime(fn) -> bool:
222
+ param = inspect.signature(fn).parameters.get("runtime")
223
+ # A defaulted runtime parameter can still be called offline with None.
224
+ return param is not None and param.default is inspect.Parameter.empty
225
+
232
226
  judges = self.plugged_judges()
233
227
  available = {"task": self.data, "trace": trace}
234
228
  if runtime is not None:
@@ -239,36 +233,50 @@ class Task(Generic[DataT, StateT, ConfigT]):
239
233
  rewards = discover_decorated(self, "reward")
240
234
  if runtime is None:
241
235
  skipped = [
242
- fn.__name__ for fn in (*metrics, *rewards) if _requires_runtime(fn)
236
+ fn.__name__ for fn in (*metrics, *rewards) if requires_runtime(fn)
243
237
  ] + [
244
238
  judge.reward_name
245
239
  for judge in judges
246
- if _requires_runtime(judge.score)
240
+ if requires_runtime(judge.score)
247
241
  ]
248
242
  if skipped:
249
243
  logger.info(
250
244
  "score: no runtime — skipped runtime-dependent signals: %s",
251
245
  skipped,
252
246
  )
253
- metrics = [fn for fn in metrics if not _requires_runtime(fn)]
254
- rewards = [fn for fn in rewards if not _requires_runtime(fn)]
247
+ metrics = [fn for fn in metrics if not requires_runtime(fn)]
248
+ rewards = [fn for fn in rewards if not requires_runtime(fn)]
255
249
  judges = [
256
- judge for judge in judges if not _requires_runtime(judge.score)
250
+ judge for judge in judges if not requires_runtime(judge.score)
257
251
  ]
258
252
 
259
253
  metric_results = await invoke_all(metrics, available)
260
254
  for fn, result in zip(metrics, metric_results):
261
- _record_result(trace, fn.__name__, result)
255
+ if isinstance(result, Mapping):
256
+ trace.record_metrics(result)
257
+ else:
258
+ trace.record_metric(fn.__name__, result)
262
259
  reward_results = await invoke_all(rewards, available)
263
260
  for fn, result in zip(rewards, reward_results):
264
- _record_result(
265
- trace, fn.__name__, result, getattr(fn, "_vf_weight", 1.0)
261
+ weight = getattr(fn, "_vf_weight", 1.0)
262
+ items = (
263
+ result.items()
264
+ if isinstance(result, Mapping)
265
+ else [(fn.__name__, result)]
266
266
  )
267
+ for key, value in items:
268
+ trace.record_reward(key, value, weight)
267
269
  judge_results = await invoke_all(
268
270
  [judge.score for judge in judges], available
269
271
  )
270
272
  for judge, result in zip(judges, judge_results):
271
- _record_result(trace, judge.reward_name, result, judge.config.weight)
273
+ items = (
274
+ result.items()
275
+ if isinstance(result, Mapping)
276
+ else [(judge.reward_name, result)]
277
+ )
278
+ for key, value in items:
279
+ trace.record_reward(key, value, judge.config.weight)
272
280
 
273
281
 
274
282
  TaskT = TypeVar("TaskT", bound=Task)
verifiers/v1/taskset.py CHANGED
@@ -1,88 +1,106 @@
1
1
  """The taskset: a thin loader that yields typed tasks.
2
2
 
3
- A `Taskset` is the data half of an environment: config in, tasks out. `load()` — the
4
- one subclass hook — builds each row's `TaskData` and wraps it in the task type with
5
- the config's task-facing subtree:
3
+ A `Taskset` is the data half of an environment: config in, tasks out. `load()` is
4
+ the main hook that builds each task:
6
5
 
7
6
  def load(self) -> Iterable[MyTask]:
8
- return [MyTask(MyData(idx=i, ...), self.config.task) for i in ...]
7
+ for i in ...:
8
+ yield MyTask(MyData(idx=i, ...), self.config.task)
9
9
 
10
- `load` may also be a generator, possibly infinite (declare `INFINITE = True`); runs
11
- materialize what they need through `select`, the env server pulls task by task.
12
-
13
- Load-time knobs live on the taskset config, task-facing knobs under its `task`
14
- subtree, shared tool servers on `tools`. All per-task behavior lives on the `Task`.
15
-
16
- The class stays generic (`Taskset[TaskT, TasksetConfigT]`) so the loaders can read
17
- the types: `taskset_config_type` narrows `--env.taskset.*` flags, `task_type` types
18
- the wire trace — one task type per taskset, so replay can rebuild saved rows.
10
+ `load` may also be a generator for infinite tasksets. There is a one-to-one
11
+ mapping between taskset and task type, i.e. a taskset may only yield one task
12
+ type.
19
13
  """
20
14
 
21
15
  from __future__ import annotations
22
16
 
17
+ import copy
23
18
  import itertools
24
- import logging
25
- from collections.abc import Iterable
26
- from typing import TYPE_CHECKING, ClassVar, Generic
19
+ import random
20
+ from abc import ABC, abstractmethod
21
+ from collections.abc import Callable, Iterable, Iterator
22
+ from typing import TYPE_CHECKING, ClassVar, Generic, Self
27
23
 
28
24
  from pydantic_config import BaseConfig
29
25
  from typing_extensions import TypeVar
30
26
 
31
27
  from verifiers.v1.configs.taskset import TasksetConfig
32
28
  from verifiers.v1.task import Task, TaskT, resolve_server_config
33
- from verifiers.v1.utils.generic import generic_type
34
- from verifiers.v1.utils.sampling import sample
29
+ from verifiers.v1.utils.generic import concrete_type
30
+ from verifiers.v1.utils.sampling import SEED
35
31
 
36
32
  if TYPE_CHECKING:
37
33
  from verifiers.v1.mcp import Toolset
38
34
 
39
- logger = logging.getLogger(__name__)
40
-
41
-
42
35
  TasksetConfigT = TypeVar("TasksetConfigT", bound=TasksetConfig, default=TasksetConfig)
43
36
 
44
37
 
45
- class Taskset(Generic[TaskT, TasksetConfigT]):
46
- INFINITE: ClassVar[bool] = False
47
- """Whether `load` yields tasks forever. Inherent to the taskset, not a config
48
- knob: runs bound themselves with `select(num_tasks)`, and shuffle is impossible."""
38
+ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
39
+ INFINITE: bool = False
40
+ """Whether the taskset is infinite (yields tasks forever). Class-declared;
41
+ a `head(n)` view shadows it per instance (bounded by construction)."""
49
42
 
50
43
  tools: ClassVar[tuple[type[Toolset], ...]] = ()
51
- """Tool server classes shared by one environment worker's rollouts."""
44
+ """Tool servers shared by all tasks in the taskset. The environment will
45
+ spawn a single, global instance, reused across tasks."""
52
46
 
53
47
  def __init__(self, config: TasksetConfigT) -> None:
54
48
  self.config = config
49
+ override = config.system_prompt
50
+ self.system_prompt = override.read_text() if override is not None else None
51
+ self.transform: Callable[[Iterator[TaskT]], Iterator[TaskT]] | None = None
52
+ """Iteration transform carried by `head`/`shuffle` views (see `view`)."""
53
+
54
+ @abstractmethod
55
+ def load(self) -> Iterable[TaskT]:
56
+ """Build and yield the taskset's tasks; may be a generator (see module doc)."""
57
+
58
+ def __iter__(self) -> Iterator[TaskT]:
59
+ """Lazily iterate `load()` with the config-layer system prompt applied and
60
+ any view transform on top — the read path; `load` is the subclass hook."""
61
+ prompt = self.system_prompt
62
+ tasks = (
63
+ task.with_system_prompt(prompt) if prompt is not None else task
64
+ for task in self.load()
65
+ )
66
+ yield from self.transform(tasks) if self.transform is not None else tasks
67
+
68
+ def view(self, transform: Callable[[Iterator[TaskT]], Iterator[TaskT]]) -> Self:
69
+ """A shallow copy of this taskset iterating through `transform`, composed
70
+ onto any transform this taskset already carries."""
71
+ clone = copy.copy(self)
72
+ prev = self.transform
73
+ clone.transform = (
74
+ transform if prev is None else lambda tasks: transform(prev(tasks))
75
+ )
76
+ return clone
77
+
78
+ def head(self, num_tasks: int) -> Self:
79
+ """A lazy, always-finite view of the first `num_tasks` tasks."""
80
+ view = self.view(lambda tasks: itertools.islice(tasks, num_tasks))
81
+ view.INFINITE = False
82
+ return view
83
+
84
+ def shuffle(self, seed: int | None = None) -> Self:
85
+ """A shuffled view under `seed` — the shared fixed seed when None, so runs
86
+ sample reproducibly (materializes the receiver on iteration); raises on an
87
+ infinite taskset — bound it first (`head(n).shuffle()`)."""
88
+ if self.INFINITE:
89
+ raise ValueError(
90
+ f"{type(self).__name__} is infinite - cannot shuffle; "
91
+ "bound it first with head(num_tasks)"
92
+ )
93
+
94
+ def shuffled(tasks: Iterator[TaskT]) -> Iterator[TaskT]:
95
+ materialized = list(tasks)
96
+ random.Random(SEED if seed is None else seed).shuffle(materialized)
97
+ return iter(materialized)
98
+
99
+ return self.view(shuffled)
55
100
 
56
101
  @classmethod
57
102
  def task_type(cls) -> type[Task]:
58
- return generic_type(cls, Task, origin=Taskset) or Task
59
-
60
- def load(self) -> Iterable[TaskT]:
61
- raise NotImplementedError
62
-
63
- def select(
64
- self, num_tasks: int | None = None, shuffle: bool = False
65
- ) -> list[TaskT]:
66
- """Materialize the first `num_tasks` off `load` (all when `None`), pulled
67
- lazily. `shuffle` samples from the whole taskset instead (fixed-seed), which
68
- materializes everything first; on an `INFINITE` taskset it's a warned no-op —
69
- the first `num_tasks` generated are already an arbitrary sample."""
70
- if type(self).INFINITE:
71
- if num_tasks is None:
72
- raise ValueError(
73
- f"{type(self).__name__} is infinite - select a bounded subset "
74
- "with num_tasks (-n on the CLI)"
75
- )
76
- if shuffle:
77
- logger.warning(
78
- "shuffle is a no-op on an infinite taskset - "
79
- "taking the first %d generated tasks",
80
- num_tasks,
81
- )
82
- return list(itertools.islice(self.load(), num_tasks))
83
- if shuffle:
84
- return sample(self.load(), shuffle=True, limit=num_tasks)
85
- return list(itertools.islice(self.load(), num_tasks))
103
+ return concrete_type(cls, Task, origin=Taskset) or Task
86
104
 
87
105
  def server_config(self, server_cls: type) -> BaseConfig:
88
106
  """The config a `tools` entry is built with, resolved off `self.config` (the
@@ -61,7 +61,6 @@ class LeanTaskConfig(TaskConfig):
61
61
  class LeanConfig(TasksetConfig):
62
62
  dataset: LeanDatasetConfig
63
63
  docker_image: str = DEFAULT_DOCKER_IMAGE
64
- system_prompt: str = DEFAULT_SYSTEM_PROMPT
65
64
  task: LeanTaskConfig = LeanTaskConfig()
66
65
 
67
66
 
@@ -187,7 +186,7 @@ class LeanTaskset(Taskset[LeanTask, LeanConfig]):
187
186
  idx=index,
188
187
  name=str(name) if name else f"task_{index:05d}",
189
188
  prompt=self._build_prompt(formal_statement, header),
190
- system_prompt=config.system_prompt,
189
+ system_prompt=DEFAULT_SYSTEM_PROMPT,
191
190
  image=config.docker_image,
192
191
  workdir=config.task.lean_project_path,
193
192
  resources=resources,
@@ -39,7 +39,7 @@ def deep_merge(base: dict, override: dict) -> dict:
39
39
  return merged
40
40
 
41
41
 
42
- def generic_type(
42
+ def concrete_type(
43
43
  cls: type, bound: type[T], *, origin: type | None = None
44
44
  ) -> type[T] | None:
45
45
  """Find a concrete bounded type through `cls`'s MRO, most-derived first."""
@@ -1,10 +1,9 @@
1
1
  """Shared sampling: an optional fixed-seed shuffle, then an optional head-slice.
2
2
 
3
- Every entrypoint narrows its items the same way: with `--shuffle`, a shuffle under a fixed
4
- seed so the sampled subset is the *same* every run (reproducible), then an optional slice to
5
- the first `limit`. Taskset entrypoints (eval, validate, debug, GEPA) go through
6
- `Taskset.select`, which shares this shuffle; the server eval path and the legacy bridge
7
- sample plain index lists here directly.
3
+ With `--shuffle`, a shuffle under the fixed `SEED` so the sampled subset is the *same*
4
+ every run (reproducible), then an optional slice to the first `limit`. Used by the paths
5
+ that sample plain index lists (the server eval path and the legacy bridge); tasksets
6
+ shuffle themselves (`Taskset.shuffle`, same default seed).
8
7
  """
9
8
 
10
9
  import random
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev49
3
+ Version: 0.2.2.dev51
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -170,35 +170,35 @@ verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2
170
170
  verifiers/v1/__init__.py,sha256=xHKE-HZA8bN0JJovf4KbBI2mxBtZJ4hPNnzac4FtWwM,7332
171
171
  verifiers/v1/agent.py,sha256=dj4GabQSs6Wzwbph5Rlx-efsVQsSuWV2YI5hfvq5mGA,31990
172
172
  verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
173
- verifiers/v1/env.py,sha256=3R3O_2dEk0feql_HKbvzlVqvRdwSjHfWML1aEJ0rZKY,18445
173
+ verifiers/v1/env.py,sha256=m-1TNzs9YyFbMkoW6c54vI5xG4jnW3LvLgvJhOgbg04,18447
174
174
  verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
175
175
  verifiers/v1/errors.py,sha256=slZnrtqMo_BgXQV46wJJhV6vvAAnVsuCbVGp2vBVWfs,6888
176
176
  verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
177
177
  verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
178
- verifiers/v1/judge.py,sha256=VnTi9iSfCJMCccYWc1qCTcT7iD3AcomS2cXfHwh1HVw,9423
178
+ verifiers/v1/judge.py,sha256=duZcWYTynzqShs8vfbExpFlgLS-JA9NFqdmCRLg3sbc,9393
179
179
  verifiers/v1/legacy.py,sha256=nkbdhFJZN1QzsjE42AB_A_c1zrk3_Y8zR07srZJS2ck,22584
180
- verifiers/v1/loaders.py,sha256=TnNKY2msVo3p7m-5wjw376PYH-zTu3TLixXRst0VOiM,9985
180
+ verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
181
181
  verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
182
182
  verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
183
183
  verifiers/v1/rollout.py,sha256=GJzHgJFlw8O1vDtWM__2JhSiG2GqbfjtH2ri1MRfw3g,20495
184
184
  verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
185
185
  verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
186
- verifiers/v1/state.py,sha256=R8tyQv8nsFV2pztrquDqCOa1Mk1fAp3w2GjqUugZwE0,689
187
- verifiers/v1/task.py,sha256=ad2YgS7ZeeBdezyWpuexLeGsCyIk5uuUJ061LXrYJ88,11074
188
- verifiers/v1/taskset.py,sha256=PX1-skAVhSpGF07qjxcG0Ctmq1LiMftMxxok-uc8qRY,3939
186
+ verifiers/v1/state.py,sha256=LhV7lpqcC19_ZQZ4KKO4dzTmZHJO4VEOpjPaLMBWh54,691
187
+ verifiers/v1/task.py,sha256=hRDIHAdBwP-aalLmhHeJ1VWW_9YJAkP21w7TkOU7VXU,11805
188
+ verifiers/v1/taskset.py,sha256=RPFkpRkeEyPX557_mK-JwraBuUKsguiCUKSFyotTclU,4636
189
189
  verifiers/v1/trace.py,sha256=9KqBeGaRUtWh6Gpdf6G1c7UZso0DB9Mjmtkzl6I_oYE,19458
190
190
  verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
191
191
  verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
192
192
  verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
193
193
  verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
194
- verifiers/v1/cli/debug.py,sha256=JZxsrUZyHpuEDiZQeIbHpMhhh6JtfOB8khPhzprQAYg,11223
194
+ verifiers/v1/cli/debug.py,sha256=mRBPlYjc93o-4sooVSJXl_H3zALC2vhAl3EBzN-DpoM,11507
195
195
  verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
196
196
  verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
197
197
  verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
198
198
  verifiers/v1/cli/replay.py,sha256=Jjyi61fl20xjGArdXTwZ3qKoSkC0iQ6WLSrvKkURdeI,10179
199
199
  verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
200
200
  verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
201
- verifiers/v1/cli/validate.py,sha256=s6M1RXqfnhoe1IfPW25I6aWjFpVK0kZ5YSMdDOtjcOo,9976
201
+ verifiers/v1/cli/validate.py,sha256=lZ_wWnLgIWH5rUlMtwzvF3uS_cFZA4ZqBowpzdI4gos,10260
202
202
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
203
203
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
204
204
  verifiers/v1/cli/dashboard/eval.py,sha256=i_Qn9enCU7aas-VTVAJocPNbTuHSUHThwQnls-3_kc4,34265
@@ -207,7 +207,7 @@ verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8m
207
207
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
208
208
  verifiers/v1/cli/eval/main.py,sha256=zvbLy_ryn3S1fWoTmWL4QEReVO1Q_hIrgyK6Fp0q_ZM,5501
209
209
  verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
210
- verifiers/v1/cli/eval/runner.py,sha256=wEOGgYh9l_BAXQKviCjmCvEofFkmFPf7T3lbZKj4Iiw,11508
210
+ verifiers/v1/cli/eval/runner.py,sha256=TKBnHSSs_knZcaHiH0ZjrF8DneEWlLp5b8O_SjQz_as,12130
211
211
  verifiers/v1/clients/__init__.py,sha256=bHGS5jfrAZ-VBUeXpasXkkKdHkFgjOSN9KbOJXrnFtU,511
212
212
  verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
213
213
  verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
@@ -217,12 +217,12 @@ verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxV
217
217
  verifiers/v1/configs/agent.py,sha256=N2Lm0iZXFxfsSIiXUQ7f3RDZcoin3M7pJj6HVdCE5eo,3537
218
218
  verifiers/v1/configs/env.py,sha256=4aUap3BJX3WCA2_fZ-jGPxF3khiVpORqmz_3eBT4j9w,7824
219
219
  verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Vagg,1564
220
- verifiers/v1/configs/judge.py,sha256=ZJlmWogHY8oDjvWvtNuVKvkzuheWalIjl1njJ4dE0Mk,2511
220
+ verifiers/v1/configs/judge.py,sha256=eEVtBGjAOmm1sSRd_UHVBpJEfUxMHxHDBFnvchmxs-w,2217
221
221
  verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
222
222
  verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
223
223
  verifiers/v1/configs/serve.py,sha256=cHHnFGQtqEwvaovdeoVMpCebTHgPXKUuIivN5IhxZR0,2596
224
224
  verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
225
- verifiers/v1/configs/taskset.py,sha256=IdZwsarn0_WvYdR_Kr4QYYdqbaoGYkbYyO5VA-RPYPc,657
225
+ verifiers/v1/configs/taskset.py,sha256=o-tr1xkmGmYG10_ySG-Du_djWS-dN4h7cv14WSvJ4eY,851
226
226
  verifiers/v1/configs/cli/__init__.py,sha256=3wBAONw5XkBPqbMjpHk5XBHlBVybR8JQ2JCPWijYBMU,444
227
227
  verifiers/v1/configs/cli/debug.py,sha256=3lkKqmQFLJBcUZECChBHCrod5pbCKSmT47EhjILrxjI,2843
228
228
  verifiers/v1/configs/cli/env.py,sha256=XgjRXYpPjga0ZYIHEESY7VuVvtE6pWfAzf6JBLeuTIc,2695
@@ -238,7 +238,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
238
238
  verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
239
239
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
240
240
  verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
241
- verifiers/v1/envs/agentic_judge/env.py,sha256=3XhGiMdKJBhpSyUdlfYYKPIR42Ll34WUvEFqbCtKdX8,15342
241
+ verifiers/v1/envs/agentic_judge/env.py,sha256=2g9yz3QfYUc_OZBkxzWOEKsGiT51siYMKB7PwvbCY2E,14878
242
242
  verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
243
243
  verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
244
244
  verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
@@ -246,11 +246,11 @@ verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54
246
246
  verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
247
247
  verifiers/v1/envs/user_sim/env.py,sha256=9L_KXjzfkTnDeQpn4ftQagZyPkJ3yjcQeZqZaq7kxjE,4567
248
248
  verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
249
- verifiers/v1/gepa/adapter.py,sha256=my7hM6GSzOw__F7Ot1E-XA_OhZyovhgbJ7fOQkW66l8,6149
249
+ verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,6139
250
250
  verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4626
251
251
  verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
252
252
  verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
253
- verifiers/v1/gepa/runner.py,sha256=o7Am2Ram-Ejbr_tVpN70zb0uQ5a1NIpXvlMhwWL22Xs,5406
253
+ verifiers/v1/gepa/runner.py,sha256=-LZEt0lod6NKfyxg2dvOXZjfzTv3AWwendexd1De26c,6260
254
254
  verifiers/v1/harnesses/__init__.py,sha256=3fgfAMzFrNgu7RIO_evZPxipPdiskiHun5EKsdS4y6k,1301
255
255
  verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
256
256
  verifiers/v1/harnesses/bash/harness.py,sha256=ZtjhsANuK3vXkf9mRCi8-sBMyPebJMBNHa58vTd2Fm8,5284
@@ -291,7 +291,7 @@ verifiers/v1/judges/rubric.py,sha256=kB0yWEtfC60zJBCVjOKBO00YtH8-AOofzPkuyccVTtQ
291
291
  verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
292
292
  verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
293
293
  verifiers/v1/mcp/launch.py,sha256=lRtcl98BO7imC44bsxO2yljwRQR_KCvJ6P-bd_hS5L8,18444
294
- verifiers/v1/mcp/server.py,sha256=uW8qddXQbtECxIPCU-lkOlFIv3VKWoFAeZFNhwPoyWo,10985
294
+ verifiers/v1/mcp/server.py,sha256=zS3-m-0IX3z8nVzUhQj391f8_OieE1nKr5S4Nj0dSSU,10987
295
295
  verifiers/v1/mcp/toolset.py,sha256=tSPXgTNREZyaeCnoFVnDfXxFAj_pStZMPUIgyPrY_Dk,996
296
296
  verifiers/v1/runtimes/__init__.py,sha256=TqNtJVahzQuLS4fgCVeT9xdOzTkp0hRAPOkl7eFVDUg,2011
297
297
  verifiers/v1/runtimes/base.py,sha256=sTCiLo51yAU_T-Voc3BTH-FuzyEM0cGY05VzFE4DmYc,14720
@@ -311,7 +311,7 @@ verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRH
311
311
  verifiers/v1/tasksets/harbor/taskset.py,sha256=2Iw1ZaPRhRjdbfClRLd0iqKWABjM7M525zFN4wuaeCk,13867
312
312
  verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
313
313
  verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
314
- verifiers/v1/tasksets/lean/taskset.py,sha256=0KujQgx2fTL_DyWSQidebKoD94U8oJp82PXoco6-3B4,9065
314
+ verifiers/v1/tasksets/lean/taskset.py,sha256=Fcn4UgTcdjMYnJ4CThaFZ28Rz_QDx0e21vJdW_vp12g,9019
315
315
  verifiers/v1/tasksets/openenv/__init__.py,sha256=G726meYYDplC3QE3AyjdSShq7FUcb4NkVqowuwPPuJk,303
316
316
  verifiers/v1/tasksets/openenv/taskset.py,sha256=1Wc9-5eN2HMKAwL34yTNex3zYOvXNXibF5wCROoMbFI,6075
317
317
  verifiers/v1/tasksets/textarena/__init__.py,sha256=Os2OlBY_pSH0B1DP32RitumgSLpq2LtcI7VDUHYPzRE,329
@@ -320,17 +320,17 @@ verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuF
320
320
  verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
321
321
  verifiers/v1/utils/compile.py,sha256=plpZYt-UD8q7Yx5pD7Rfy1w7Jer5TCBk2CJGHiNcPio,5696
322
322
  verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
323
- verifiers/v1/utils/generic.py,sha256=xwu8xNGdVOhM2K87C5cqhmnRf6apSPDASA15M0S2MaY,2054
323
+ verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
324
324
  verifiers/v1/utils/git.py,sha256=Mj2InIZlCAKWEWR9HPEMJX1OgkCVve5NjWgTCD4GmkM,4871
325
325
  verifiers/v1/utils/image.py,sha256=OFw_wdwVtbdwa8uJ3X3vYfhpUAi0CTH5HMBupjVsRRQ,282
326
326
  verifiers/v1/utils/install.py,sha256=fWNsyKrw_PyhC0Qhqv5Ri0adFm5Ddx4AVy_hbsra1GU,1390
327
327
  verifiers/v1/utils/interrupt.py,sha256=F-KKhc5ndPJJfhd3SuMqyqhXhA32FhCRy5KWFJpEoM4,1179
328
328
  verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y,2084
329
329
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
330
- verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
330
+ verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
331
331
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
332
- verifiers-0.2.2.dev49.dist-info/METADATA,sha256=58Q74xtCzA2vwqY2VjgOgFdSvyqzrV2vAyCCkgjOrCo,4545
333
- verifiers-0.2.2.dev49.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
334
- verifiers-0.2.2.dev49.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
335
- verifiers-0.2.2.dev49.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
336
- verifiers-0.2.2.dev49.dist-info/RECORD,,
332
+ verifiers-0.2.2.dev51.dist-info/METADATA,sha256=Ns-l7u4GlaZ2rco5UvfL-YtVj50LqCMzWiFX-zmXDEA,4545
333
+ verifiers-0.2.2.dev51.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
334
+ verifiers-0.2.2.dev51.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
335
+ verifiers-0.2.2.dev51.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
336
+ verifiers-0.2.2.dev51.dist-info/RECORD,,