verifiers 0.2.2.dev16__py3-none-any.whl → 0.2.2.dev18__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -232,7 +232,7 @@ def Overview(config: EvalConfig) -> Table:
232
232
  grid.add_column()
233
233
  seats = config.env.agent_harnesses()
234
234
  taskset = config.env.taskset
235
- env_label = taskset.name if taskset is not None else "no taskset"
235
+ env_label = taskset.name if taskset.id else "no taskset"
236
236
  if config.env.id:
237
237
  env_label = f"{env_name(config.env.id)}+{env_label}"
238
238
  # One seat story when every seat resolves the same way (the common case); one
@@ -254,9 +254,7 @@ def Overview(config: EvalConfig) -> Table:
254
254
  # value (or our `[...]`/`{...}` delimiters) can carry Rich markup that would otherwise be
255
255
  # parsed as styling and dropped. `id` is in the `env` row; harness `runtime.type` too (hidden
256
256
  # here), but only for the harness — `taskset.task.tools.runtime.type` has no other display.
257
- if taskset is not None and (
258
- taskset_over := overrides(taskset, skip=frozenset({"id"}))
259
- ):
257
+ if taskset_over := overrides(taskset, skip=frozenset({"id"})):
260
258
  grid.add_row("taskset", escape(" · ".join(taskset_over)))
261
259
  for role, h in seats.items():
262
260
  if harness_over := overrides(h, skip=frozenset({"id", "runtime.type"})):
@@ -4,13 +4,20 @@
4
4
  flags) and writes back into the same dir. `load` keeps the good saved rollouts and
5
5
  re-runs what's owed: missing rollouts (never written) and errored ones (dropped and
6
6
  redone).
7
+
8
+ A saved rollout is matched to a selected task by content: `task_key` hashes the
9
+ task's wire data. Tasks with identical data are interchangeable, a task whose data
10
+ changed since the interrupted run re-runs, and nothing depends on `data.idx`. The
11
+ legacy (v0) bridge still matches by row index (`key_of`).
7
12
  """
8
13
 
14
+ import hashlib
9
15
  import json
10
16
  import tomllib
11
- from collections import defaultdict
12
- from collections.abc import Callable
17
+ from collections import Counter, defaultdict
18
+ from collections.abc import Callable, Hashable, Mapping
13
19
  from pathlib import Path
20
+ from typing import TypeVar
14
21
 
15
22
  from pydantic_core import from_json
16
23
 
@@ -19,6 +26,30 @@ from verifiers.v1.configs.eval import EvalConfig
19
26
  from verifiers.v1.episode import Episode, WireEpisode
20
27
  from verifiers.v1.trace import WireTrace
21
28
 
29
+ K = TypeVar("K", bound=Hashable)
30
+
31
+
32
+ def task_key(data: Mapping) -> str:
33
+ """Content identity of one task's wire data — an `exclude_none` dump, the shape
34
+ saved rows already have on disk. `sort_keys` so field order can't split identity."""
35
+ return hashlib.sha256(json.dumps(data, sort_keys=True).encode()).hexdigest()
36
+
37
+
38
+ def distribute(
39
+ selected_keys: list[K], owed: dict[K, int], num_rollouts: int
40
+ ) -> list[int]:
41
+ """Spread each key's owed rollouts over its selection instances, in order —
42
+ content-identical tasks are interchangeable, so any instance can absorb the
43
+ debt (capped at `num_rollouts` each). Returns one count per selection."""
44
+ remaining = dict(owed)
45
+ counts: list[int] = []
46
+ for key in selected_keys:
47
+ take = min(num_rollouts, remaining.get(key, 0))
48
+ if take:
49
+ remaining[key] -= take
50
+ counts.append(take)
51
+ return counts
52
+
22
53
 
23
54
  def split_resume(argv: list[str]) -> tuple[Path | None, list[str]]:
24
55
  """Pull `--resume <dir>` / `--resume=<dir>` out of argv, returning (dir, the other args).
@@ -51,25 +82,27 @@ def load_resume_config(resume_dir: Path) -> EvalConfig:
51
82
 
52
83
  def load(
53
84
  resume_dir: Path,
54
- selected_idxs: list[int],
85
+ selected_keys: list[K],
55
86
  num_rollouts: int,
56
87
  complete: Callable[[Episode], bool] | None = None,
57
88
  *,
58
89
  whole_task: bool = False,
59
- ) -> tuple[list[Episode], dict[int, int]]:
60
- """Load the good saved rollouts as finished episodes and diff them against the
61
- run's target: returns (kept episodes, rollouts owed per task idx). A rollout is
62
- kept or redone as a unit — the episode — so a multi-trace rollout interrupted
63
- mid-write is simply owed again. `complete` is the environment's keep-verdict
64
- (`Env.complete`); without it (the server path) the default is
65
- `episode.ok`, so an errored rollout is dropped and re-run. `whole_task` redoes
66
- a partially-kept task as a unit — the legacy group-scored path, where
67
- `run_group` always serves the full n. Rewrites `traces.jsonl` to just the kept
68
- rows via a temp file + atomic rename, so an interrupted resume can't corrupt
69
- the prior results; the resumed rollouts then append. Pre-episode files (one
70
- bare trace per line) load each trace as a single-trace episode."""
90
+ key_of: Callable[[Mapping], K] | None = None,
91
+ ) -> tuple[list[Episode], dict[K, int]]:
92
+ """Load the good saved rollouts and diff them against the run's target: returns
93
+ (kept episodes, rollouts owed per task key). `selected_keys` is one key per
94
+ selected task (duplicates allowed — a key selected k times is owed up to
95
+ `k * num_rollouts`; spread back over the tasks with `distribute`). `key_of` maps
96
+ a saved row's task data to its key (default `task_key`; the legacy bridge uses
97
+ row indices). `complete` is the keep-verdict (default `episode.ok`); `whole_task`
98
+ redoes a partially-kept task whole (legacy group scoring). Rewrites
99
+ `traces.jsonl` to the kept rows via a temp file + atomic rename; a torn or
100
+ malformed row is owed again, never a crash."""
71
101
  path = resume_dir / TRACES_FILE
72
- selected = set(selected_idxs)
102
+ targets = {
103
+ key: count * num_rollouts for key, count in Counter(selected_keys).items()
104
+ }
105
+ keyed = key_of if key_of is not None else task_key
73
106
 
74
107
  def parse(row: dict) -> Episode:
75
108
  if sniff_episode(row):
@@ -77,7 +110,7 @@ def load(
77
110
  return Episode.of(WireTrace.model_validate(row))
78
111
 
79
112
  verdict = complete if complete is not None else (lambda episode: episode.ok)
80
- good: dict[int, list[tuple[bytes, Episode]]] = defaultdict(list)
113
+ good: dict[K, list[tuple[bytes, Episode]]] = defaultdict(list)
81
114
  if path.exists():
82
115
  with path.open("rb") as results:
83
116
  for line in results:
@@ -89,16 +122,16 @@ def load(
89
122
  except ValueError:
90
123
  row = json.loads(line)
91
124
  # The task rides each trace; a traceless record (a failure
92
- # before any trace minted) has no idx and is owed again.
125
+ # before any trace minted) has no task and is owed again.
93
126
  if sniff_episode(row):
94
- idx = row["traces"][0]["task"]["data"]["idx"]
127
+ key = keyed(row["traces"][0]["task"]["data"])
95
128
  else:
96
- idx = row["task"]["data"]["idx"]
129
+ key = keyed(row["task"]["data"])
97
130
  except (ValueError, KeyError, IndexError, TypeError):
98
131
  # A torn final line (the run died mid-write) or a foreign shape
99
132
  # is not a keepable rollout — it's owed again, never a crash.
100
133
  continue
101
- if idx not in selected or len(good[idx]) >= num_rollouts:
134
+ if key not in targets or len(good[key]) >= targets[key]:
102
135
  continue
103
136
  try:
104
137
  episode = parse(row)
@@ -106,20 +139,20 @@ def load(
106
139
  continue
107
140
  except Exception: # malformed row: redo it
108
141
  continue
109
- good[idx].append(
142
+ good[key].append(
110
143
  (line if line.endswith(b"\n") else line + b"\n", episode)
111
144
  )
112
145
  keep: list[bytes] = []
113
146
  episodes: list[Episode] = []
114
- owed: dict[int, int] = {}
115
- for idx in selected_idxs:
116
- rows = good.get(idx, [])
117
- if whole_task and len(rows) < num_rollouts:
147
+ owed: dict[K, int] = {}
148
+ for key, target in targets.items():
149
+ rows = good.get(key, [])
150
+ if whole_task and len(rows) < target:
118
151
  rows = [] # a partial unit redoes whole — its kept rows are dropped
119
152
  keep.extend(line for line, _ in rows)
120
153
  episodes.extend(episode for _, episode in rows)
121
- if missing := num_rollouts - len(rows):
122
- owed[idx] = missing
154
+ if missing := target - len(rows):
155
+ owed[key] = missing
123
156
  tmp = path.with_suffix(".jsonl.tmp")
124
157
  tmp.write_bytes(b"".join(keep))
125
158
  tmp.replace(path)
@@ -32,21 +32,25 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
32
32
  asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
33
33
  )
34
34
  out = output_path(config)
35
- owed: dict[str, int] | None = None
35
+ # One (task, rollouts-to-run) pair per selected task; resume shrinks the counts.
36
+ plan = [(task, config.num_rollouts) for task in tasks]
36
37
  # Kept on-disk rollouts rejoin the run as finished episodes; only owed ones re-run.
37
38
  finished: list[Episode] = []
38
39
  if config.resume is not None:
39
- finished, owed = resume.load(
40
- out, [t.data.idx for t in tasks], config.num_rollouts, env.complete
41
- )
40
+ keys = [
41
+ resume.task_key(t.data.model_dump(mode="json", exclude_none=True))
42
+ for t in tasks
43
+ ]
44
+ finished, owed = resume.load(out, keys, config.num_rollouts, env.complete)
42
45
  if not owed: # already complete - report it and exit successfully
43
46
  print(resume.nothing_to_resume_msg(out, len(tasks), config.num_rollouts))
44
47
  raise SystemExit(0)
45
- tasks = [task for task in tasks if owed.get(task.data.idx)]
48
+ counts = resume.distribute(keys, owed, config.num_rollouts)
49
+ plan = [(task, n) for task, n in zip(tasks, counts) if n]
46
50
  logger.info(
47
51
  "resuming %s: %d task(s), %d rollout(s) owed",
48
52
  out,
49
- len(tasks),
53
+ len(plan),
50
54
  sum(owed.values()),
51
55
  )
52
56
  else:
@@ -70,13 +74,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
70
74
  # Serving resources (shared tool servers, interception) come up once for the
71
75
  # run; plan slots inside so the env's agents borrow them.
72
76
  async with env.serving():
73
- planned = [
74
- slot
75
- for task in tasks
76
- for slot in env.slots(
77
- task, n=owed[task.data.idx] if owed else config.num_rollouts
78
- )
79
- ]
77
+ planned = [slot for task, n in plan for slot in env.slots(task, n=n)]
80
78
  slots = [RunSlot.finished(episode) for episode in finished] + planned
81
79
  push_state = None
82
80
  if config.push and config.rich:
@@ -123,6 +121,15 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
123
121
  if legacy
124
122
  else {"config_data": env_config_data(config.env)} # picklable across the spawn
125
123
  )
124
+ tasks = []
125
+ if not legacy:
126
+ from verifiers.v1.loaders import load_taskset
127
+
128
+ # The client owns the taskset: load it here, once — the server (and its pool
129
+ # workers) never load data, they rebuild each dispatched task from its request.
130
+ tasks = load_taskset(config.env.taskset).select(
131
+ config.num_tasks, config.shuffle
132
+ )
126
133
  # Spawned processes inherit no logging — hand them the main process's setup so
127
134
  # their rollout logs land in the output dir.
128
135
  level = "DEBUG" if config.verbose else "INFO"
@@ -151,48 +158,57 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
151
158
  address = await asyncio.to_thread(address_queue.get, timeout=600)
152
159
  client = EnvClient(address=address)
153
160
  await client.wait_for_server_startup(timeout=600)
154
- info = await client.info()
155
- # Only a legacy (v0) env group-scores; a v1 env scores siblings in its own
156
- # rollout.
157
- group_scored = info.requires_group_scoring
158
- if info.num_tasks is None: # infinite taskset - the run must be bounded
159
- if config.num_tasks is None:
160
- raise ValueError(
161
- f"{config.env_id} is infinite - bound the run with -n/--num-tasks"
162
- )
163
- if config.shuffle:
164
- logger.warning(
165
- "shuffle is a no-op on an infinite taskset - "
166
- "taking the first %d generated tasks",
167
- config.num_tasks,
168
- )
169
- idxs = list(range(config.num_tasks))
170
- else:
161
+ # A v1 run dispatches — and resumes — tasks by content: the client owns them,
162
+ # and `resume.task_key` is their identity. Only the legacy bridge is addressed
163
+ # by dataset row (its dataset lives server-side, reported via `info`), and
164
+ # only a legacy env group-scores; a v1 env scores siblings in its own rollout.
165
+ if legacy:
166
+ info = await client.info()
167
+ group_scored = info.requires_group_scoring
171
168
  idxs = sample(list(range(info.num_tasks)), config.shuffle, config.num_tasks)
169
+ plan = [({"task_idx": idx}, config.num_rollouts) for idx in idxs]
170
+ else:
171
+ group_scored = False
172
+ plan = [
173
+ ({"task_data": task.data.model_dump(mode="json")}, config.num_rollouts)
174
+ for task in tasks
175
+ ]
172
176
  out = output_path(config)
173
177
  finished: list[Episode] = []
174
178
  if config.resume is not None:
175
- # (legacy only) a group is served and scored together, so a partially-kept
176
- # task redoes as a whole group — whole_task drops its kept rows.
177
- finished, owed = resume.load(
178
- out, idxs, config.num_rollouts, whole_task=group_scored
179
- )
179
+ if legacy:
180
+ # A group is served and scored together, so a partially-kept task
181
+ # redoes as a whole group — whole_task drops its kept rows.
182
+ finished, owed = resume.load(
183
+ out,
184
+ idxs,
185
+ config.num_rollouts,
186
+ whole_task=group_scored,
187
+ key_of=lambda data: data.get("idx"),
188
+ )
189
+ counts = resume.distribute(idxs, owed, config.num_rollouts)
190
+ else:
191
+ keys = [
192
+ resume.task_key(t.data.model_dump(mode="json", exclude_none=True))
193
+ for t in tasks
194
+ ]
195
+ finished, owed = resume.load(out, keys, config.num_rollouts)
196
+ counts = resume.distribute(keys, owed, config.num_rollouts)
180
197
  if not owed: # already complete - report it and exit successfully
181
- print(resume.nothing_to_resume_msg(out, len(idxs), config.num_rollouts))
198
+ print(resume.nothing_to_resume_msg(out, len(plan), config.num_rollouts))
182
199
  raise SystemExit(0)
183
- idxs = [idx for idx in idxs if owed.get(idx)]
200
+ plan = [(payload, n) for (payload, _), n in zip(plan, counts) if n]
184
201
  logger.info(
185
202
  "resuming %s: %d task(s), %d rollout(s) owed",
186
203
  out,
187
- len(idxs),
204
+ len(plan),
188
205
  sum(owed.values()),
189
206
  )
190
207
  else:
191
- owed = {idx: config.num_rollouts for idx in idxs}
192
208
  save_config(config, out)
193
209
  logger.info(
194
210
  "running %dx%d rollouts via the env-server %s pool on %s",
195
- len(idxs),
211
+ len(plan),
196
212
  config.num_rollouts,
197
213
  config.pool.type,
198
214
  config.model,
@@ -224,13 +240,13 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
224
240
  records.append(Episode.of(trace))
225
241
  return records
226
242
 
227
- async def run_unit(idx: int) -> list[Episode]:
243
+ async def run_unit(payload: dict) -> list[Episode]:
228
244
  async with semaphore or contextlib.nullcontext():
229
245
  episode = await client.run(
230
- task_idx=idx,
231
246
  client=config.client,
232
247
  model=config.model,
233
248
  sampling=config.sampling,
249
+ **payload,
234
250
  )
235
251
  for trace in episode.traces:
236
252
  trace.stamp(EvalRunInfo(id=config.uuid))
@@ -238,12 +254,12 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
238
254
  return [episode]
239
255
 
240
256
  # A group-scored legacy task runs its rollouts together (one `run_group`
241
- # request, one worker); otherwise each rollout is its own `run_rollout`
242
- # request, dispatched least-busy across workers.
257
+ # request, one worker); otherwise each rollout is its own `run` request,
258
+ # dispatched least-busy across workers.
243
259
  units = (
244
- [run_group_unit(i) for i in idxs]
260
+ [run_group_unit(payload["task_idx"]) for payload, _ in plan]
245
261
  if group_scored
246
- else [run_unit(i) for i in idxs for _ in range(owed[i])]
262
+ else [run_unit(payload) for payload, n in plan for _ in range(n)]
247
263
  )
248
264
  results = await asyncio.gather(*units)
249
265
  await client.close()
verifiers/v1/cli/gepa.py CHANGED
@@ -66,7 +66,7 @@ def main(argv: list[str] | None = None) -> None:
66
66
  # Refuse multi-agent before the dry-run return, so --dry-run can't write a
67
67
  # config the real invocation would reject.
68
68
  env_cls = vf.environment_class(
69
- config.env.taskset.id if config.env.taskset is not None else "",
69
+ config.env.taskset.id,
70
70
  config.env.id,
71
71
  )
72
72
  if not issubclass(env_cls, vf.SingleAgentEnv):
@@ -36,8 +36,8 @@ def output_path(config: EvalConfig) -> Path:
36
36
  if config.output_dir is not None:
37
37
  return config.output_dir
38
38
  taskset = config.env.taskset
39
- env = taskset.name if taskset is not None else "no-taskset"
40
- if taskset is not None and taskset.id and config.env.id:
39
+ env = taskset.name if taskset.id else "no-taskset"
40
+ if taskset.id and config.env.id:
41
41
  # Same compounding as `EnvConfig.env_id`: a `best-of-n+gsm8k-v1` run must
42
42
  # not share a parent dir with a plain `gsm8k-v1` one.
43
43
  env = f"{env_name(config.env.id)}+{env}"
@@ -138,9 +138,7 @@ class EnvServerConfig(BaseConfig):
138
138
  @property
139
139
  def is_legacy(self) -> bool:
140
140
  """A v0/legacy env (run via the bridge): a legacy `id` is set and no v1 taskset."""
141
- return self.id is not None and (
142
- self.env.taskset is None or not self.env.taskset.id
143
- )
141
+ return self.id is not None and not self.env.taskset.id
144
142
 
145
143
  @property
146
144
  def env_id(self) -> str:
@@ -152,7 +150,7 @@ class EnvServerConfig(BaseConfig):
152
150
  def _refuse_legacy_id_with_taskset(self):
153
151
  """A legacy `id` next to a v1 `env.taskset` would be silently inert
154
152
  (`is_legacy` is False and the v0 env never loads); refuse the mix."""
155
- if self.id is not None and self.env.taskset is not None and self.env.taskset.id:
153
+ if self.id is not None and self.env.taskset.id:
156
154
  raise ValueError(
157
155
  f"--id {self.id!r} is the legacy (v0) env id and can't combine with "
158
156
  f"the v1 taskset {self.env.taskset.id!r}. Pairing an env with a "
verifiers/v1/env.py CHANGED
@@ -31,7 +31,7 @@ from verifiers.v1.retries import RetryConfig, run_episode_with_retry
31
31
  from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
32
32
  from verifiers.v1.errors import EnvError, boundary
33
33
  from verifiers.v1.task import Task, resolve_server_config
34
- from verifiers.v1.taskset import Taskset, TasksetConfig
34
+ from verifiers.v1.taskset import TasksetConfig
35
35
  from verifiers.v1.episode import Episode
36
36
  from verifiers.v1.trace import Error, Trace
37
37
  from verifiers.v1.utils.generic import deep_merge, generic_type
@@ -65,9 +65,9 @@ class EnvConfig(BaseConfig):
65
65
  """Which `Env` runs. Empty = the taskset's own, else `SingleAgentEnv`; set
66
66
  to pair a reusable env with any taskset (an explicit id wins over the bundled)."""
67
67
  # SerializeAsAny: the env-server wire needs the resolved subclass's fields.
68
- taskset: SerializeAsAny[TasksetConfig] | None = None
68
+ taskset: SerializeAsAny[TasksetConfig] = TasksetConfig()
69
69
  """The seed taskset — the rows every rollout starts from (`--env.taskset.id`).
70
- None only for an env that mints its tasks without a dataset."""
70
+ The id stays empty only for a legacy (v0) run, which sets the top-level `id`."""
71
71
  timeout: TimeoutConfig = TimeoutConfig()
72
72
  retries: RetryConfig = RetryConfig()
73
73
  """Whole-EPISODE retries — the coarse fallback for faults no agent owns; a
@@ -81,17 +81,14 @@ class EnvConfig(BaseConfig):
81
81
  @property
82
82
  def env_id(self) -> str:
83
83
  """The taskset id, prefixed by the paired env id (`best-of-n+gsm8k-v1`)."""
84
- taskset_id = self.taskset.id if self.taskset is not None else ""
85
- if taskset_id and self.id:
86
- return f"{self.id}+{taskset_id}"
87
- return taskset_id or self.id
84
+ if self.taskset.id and self.id:
85
+ return f"{self.id}+{self.taskset.id}"
86
+ return self.taskset.id or self.id
88
87
 
89
88
  def agent_harnesses(self) -> dict[str, HarnessConfig]:
90
89
  """Each declared role's resolved harness config (pin, else the taskset's
91
90
  default) — known without constructing the env."""
92
- default = default_agent_harness(
93
- self.taskset.id if self.taskset is not None else ""
94
- )
91
+ default = default_agent_harness(self.taskset.id)
95
92
  return {
96
93
  name: cfg.harness if cfg.harness is not None else default
97
94
  for name, cfg in _declared_agent_configs(self).items()
@@ -239,7 +236,7 @@ class Env(ABC, Generic[ConfigT]):
239
236
  f"{config_cls.__name__}(...) explicitly"
240
237
  )
241
238
  self.config: ConfigT = config
242
- if config.taskset is None:
239
+ if not config.taskset.id:
243
240
  raise ValueError(
244
241
  f"{type(self).__name__} needs a seed taskset — every rollout starts "
245
242
  "from one of its tasks: set --env.taskset.id (or the positional "
@@ -247,7 +244,7 @@ class Env(ABC, Generic[ConfigT]):
247
244
  )
248
245
  self.taskset = load_taskset(config.taskset)
249
246
  self._default_harness = default_agent_harness(config.taskset.id)
250
- task_cls = generic_type(type(self.taskset), Task, origin=Taskset) or Task
247
+ task_cls = type(self.taskset).task_type()
251
248
  self._task_cls: type[Task] = task_cls
252
249
  self._agent_specs: dict[str, AgentConfig] = _declared_agent_configs(self.config)
253
250
  if not self._agent_specs:
@@ -520,7 +517,7 @@ class Env(ABC, Generic[ConfigT]):
520
517
  """`requires_tunnel` over the consumers known before any rollout: role
521
518
  runtimes, live `shared` servers, and the task class's tool/user servers;
522
519
  a class overriding `server_config` conservatively counts as remote."""
523
- task_cls = generic_type(type(self.taskset), Task, origin=Taskset) or Task
520
+ task_cls = type(self.taskset).task_type()
524
521
  server_classes = [*task_cls.tools, *([task_cls.user] if task_cls.user else [])]
525
522
  if server_classes and task_cls.server_config is not Task.server_config:
526
523
  return True
verifiers/v1/harness.py CHANGED
@@ -4,6 +4,7 @@ import logging
4
4
  import os
5
5
  from abc import ABC, abstractmethod
6
6
  from collections.abc import Mapping
7
+ from pathlib import Path
7
8
  from typing import TYPE_CHECKING, ClassVar, Generic, TypeVar
8
9
 
9
10
  from pydantic import Field
@@ -41,6 +42,10 @@ class HarnessConfig(BaseConfig):
41
42
  forward_env: list[str] = Field(default_factory=list)
42
43
  """Host variables to forward without writing secrets into config; explicit `env` wins."""
43
44
  disabled_tools: list[str] | None = None
45
+ skills: list[Path] = Field(default_factory=list)
46
+ """Skill folders to upload into the program's skill discovery directory — each
47
+ lands at `<skills dir>/<folder name>`. Only harnesses whose program discovers
48
+ skills natively (`SUPPORTS_SKILLS`) accept them."""
44
49
 
45
50
  @property
46
51
  def name(self) -> str:
@@ -66,6 +71,15 @@ class Harness(ABC, Generic[ConfigT]):
66
71
  every real harness; the tool-less chat loops (`null`) override to False. Read
67
72
  where model-directed execution changes the rules: the subprocess-on-host
68
73
  warning, the judge env's sandbox requirement."""
74
+ SUPPORTS_SKILLS: ClassVar[bool] = False
75
+ """Whether the program discovers SKILL.md skills — its `setup` calls
76
+ `install_skills` with the program's fixed discovery location; configuring
77
+ `skills` on a harness without support is rejected up front."""
78
+ NEEDS_CONTAINER: ClassVar[bool] = True
79
+ """Whether the program must run in a container runtime: True for every harness
80
+ that installs and drives a third-party program — on the host (subprocess) it
81
+ leaks host state (auth, config, processes) both ways. Only the minimal
82
+ in-house loops (`bash`, `null`) override to False."""
69
83
 
70
84
  def __init__(self, config: ConfigT) -> None:
71
85
  self.config = config
@@ -102,6 +116,28 @@ class Harness(ABC, Generic[ConfigT]):
102
116
  async def setup(self, runtime: Runtime) -> None:
103
117
  """Provision this harness in `runtime` before its execution timeout starts."""
104
118
 
119
+ async def install_skills(self, runtime: Runtime, dest: str) -> None:
120
+ """Upload each `config.skills` folder into `runtime` at `dest/<folder name>` —
121
+ the program's fixed skill discovery location, which a supporting harness's
122
+ `setup` passes."""
123
+ for skill in self.config.skills:
124
+ # Resolve so `.`/`..` entries get their real folder name (and can't
125
+ # place files outside `dest`).
126
+ skill = skill.resolve()
127
+ if not skill.is_dir():
128
+ raise ValueError(f"skill {str(skill)!r} is not a folder")
129
+ executables = []
130
+ for file in sorted(skill.rglob("*")):
131
+ if not file.is_file():
132
+ continue
133
+ target = f"{dest}/{skill.name}/{file.relative_to(skill).as_posix()}"
134
+ await runtime.write(target, file.read_bytes())
135
+ if os.access(file, os.X_OK):
136
+ executables.append(target)
137
+ if executables:
138
+ # `write` moves bytes, not modes; restore the execute bits scripts need.
139
+ await runtime.run(["chmod", "+x", *executables], {})
140
+
105
141
  async def run(
106
142
  self,
107
143
  ctx: ModelContext,
@@ -41,6 +41,7 @@ class BashHarness(Harness[BashHarnessConfig]):
41
41
  SUPPORTS_MCP = True
42
42
  SUPPORTS_USER_SIM = True
43
43
  SUPPORTS_MESSAGE_PROMPT = True
44
+ NEEDS_CONTAINER = False
44
45
 
45
46
  async def setup(self, runtime: Runtime) -> None:
46
47
  await runtime.prepare_uv_script(PROGRAM_SOURCE, self.config.resolved_env)
@@ -13,6 +13,7 @@ from verifiers.v1.trace import Trace
13
13
  CLAUDE_HOME = "/tmp/vf-claude-code-{version}"
14
14
  CLAUDE_BIN = f"{CLAUDE_HOME}/.local/bin/claude"
15
15
  CLAUDE_CONFIG_DIR = ".vf-claude"
16
+ SKILLS_DIR = f"{CLAUDE_CONFIG_DIR}/skills"
16
17
  INSTALL = """
17
18
  set -e
18
19
  command -v curl >/dev/null || (apt-get update -qq && apt-get install -y -qq curl ca-certificates >/dev/null)
@@ -30,8 +31,10 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
30
31
  SUPPORTS_MCP = True
31
32
  # images would require streaming inputs
32
33
  SUPPORTS_MESSAGE_PROMPT = False
34
+ SUPPORTS_SKILLS = True
33
35
 
34
36
  async def setup(self, runtime: Runtime) -> None:
37
+ await self.install_skills(runtime, SKILLS_DIR)
35
38
  home = CLAUDE_HOME.format(version=self.config.version)
36
39
  binary = CLAUDE_BIN.format(version=self.config.version)
37
40
  script = INSTALL.format(version=self.config.version, home=home)
@@ -66,13 +69,13 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
66
69
  "ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
67
70
  "ANTHROPIC_API_KEY": secret,
68
71
  "CLAUDE_CONFIG_DIR": CLAUDE_CONFIG_DIR,
72
+ "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
69
73
  "DISABLE_AUTOUPDATER": "1",
70
74
  "IS_SANDBOX": "1",
71
75
  }
72
76
  argv = [
73
77
  CLAUDE_BIN.format(version=self.config.version),
74
78
  "--print",
75
- "--bare",
76
79
  "--dangerously-skip-permissions",
77
80
  "--no-session-persistence",
78
81
  "--model",
@@ -23,6 +23,7 @@ KEY_VAR = "CODEX_INTERCEPT_KEY"
23
23
 
24
24
  CODEX_DIR = "/tmp/vf-codex"
25
25
  CODEX_BIN = f"{CODEX_DIR}/bin/codex"
26
+ SKILLS_DIR = ".agents/skills"
26
27
  INSTALL = r"""
27
28
  set -e
28
29
  mkdir -p {dir}/bin
@@ -46,8 +47,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
46
47
  APPENDS_SYSTEM_PROMPT = False # TODO
47
48
  SUPPORTS_MCP = False # TODO
48
49
  SUPPORTS_MESSAGE_PROMPT = True
50
+ SUPPORTS_SKILLS = True
49
51
 
50
52
  async def setup(self, runtime: Runtime) -> None:
53
+ await self.install_skills(runtime, SKILLS_DIR)
51
54
  logger.info("codex: ensuring codex %s is installed", self.config.version)
52
55
  script = (
53
56
  INSTALL.replace("{version}", self.config.version)
@@ -13,6 +13,7 @@ logger = logging.getLogger(__name__)
13
13
 
14
14
  BINARY = "/tmp/vf-kimi-code/bin/kimi"
15
15
  KIMI_HOME = ".vf-kimi-code"
16
+ SKILLS_DIR = f"{KIMI_HOME}/skills"
16
17
 
17
18
  INSTALL = r"""
18
19
  set -e
@@ -39,8 +40,10 @@ class KimiCodeHarnessConfig(HarnessConfig):
39
40
  class KimiCodeHarness(Harness[KimiCodeHarnessConfig]):
40
41
  APPENDS_SYSTEM_PROMPT = False
41
42
  SUPPORTS_MCP = True
43
+ SUPPORTS_SKILLS = True
42
44
 
43
45
  async def setup(self, runtime: Runtime) -> None:
46
+ await self.install_skills(runtime, SKILLS_DIR)
44
47
  logger.info(
45
48
  "kimi-code: ensuring Kimi Code %s is installed", self.config.version
46
49
  )
@@ -20,6 +20,7 @@ class NullHarness(Harness[NullHarnessConfig]):
20
20
  SUPPORTS_USER_SIM = True
21
21
  SUPPORTS_MESSAGE_PROMPT = True
22
22
  EXECUTES_CODE = False
23
+ NEEDS_CONTAINER = False
23
24
 
24
25
  async def setup(self, runtime: Runtime) -> None:
25
26
  await runtime.prepare_uv_script(PROGRAM_SOURCE, self.config.resolved_env)
@@ -26,6 +26,7 @@ HOME_VAR = "VF_PI_ORIGINAL_HOME"
26
26
 
27
27
  PI_DIR = "/tmp/vf-pi"
28
28
  PI_BIN = f"{PI_DIR}/pi"
29
+ SKILLS_DIR = ".agents/skills"
29
30
  MCP_VERSION = "2.11.0"
30
31
  MCP_ADAPTER = f"{PI_DIR}/mcp/node_modules/pi-mcp-adapter/index.ts"
31
32
 
@@ -95,8 +96,12 @@ class PiHarness(Harness[PiHarnessConfig]):
95
96
  APPENDS_SYSTEM_PROMPT = True
96
97
  SUPPORTS_MCP = True
97
98
  SUPPORTS_MESSAGE_PROMPT = True
99
+ # Pi's project skill discovery is trust-gated (a prompt print mode can't answer),
100
+ # so the installed skills are passed explicitly via `--skill` at launch.
101
+ SUPPORTS_SKILLS = True
98
102
 
99
103
  async def setup(self, runtime: Runtime) -> None:
104
+ await self.install_skills(runtime, SKILLS_DIR)
100
105
  logger.info(
101
106
  "pi: ensuring Pi %s and pi-mcp-adapter %s are installed",
102
107
  self.config.version,
@@ -234,6 +239,12 @@ class PiHarness(Harness[PiHarnessConfig]):
234
239
  if self.config.disabled_tools
235
240
  else []
236
241
  )
242
+ skill_args = [
243
+ arg
244
+ for skill in self.config.skills
245
+ # Resolve like `install_skills` so the path matches what it wrote.
246
+ for arg in ("--skill", f"{SKILLS_DIR}/{skill.resolve().name}")
247
+ ]
237
248
  system_args = ["--append-system-prompt", system_prompt] if system_prompt else []
238
249
  argv = [
239
250
  "sh",
@@ -252,6 +263,7 @@ class PiHarness(Harness[PiHarnessConfig]):
252
263
  ctx.model,
253
264
  *mcp_args,
254
265
  *tool_args,
266
+ *skill_args,
255
267
  *system_args,
256
268
  *image_args,
257
269
  ]
@@ -13,6 +13,7 @@ from verifiers.v1.types import SystemMessage, TextContentPart, UserMessage
13
13
 
14
14
  POOL_DIR = "/tmp/vf-pool-{version}"
15
15
  SETTINGS_PATH = ".poolside/settings.local.yaml"
16
+ SKILLS_DIR = ".poolside/skills"
16
17
  INSTALL = r"""
17
18
  set -e
18
19
  command -v curl >/dev/null || (apt-get update -qq && apt-get install -y -qq curl ca-certificates >/dev/null)
@@ -35,8 +36,10 @@ class PoolHarness(Harness[PoolHarnessConfig]):
35
36
  APPENDS_SYSTEM_PROMPT = True
36
37
  SUPPORTS_MCP = True
37
38
  SUPPORTS_MESSAGE_PROMPT = True
39
+ SUPPORTS_SKILLS = True
38
40
 
39
41
  async def setup(self, runtime: Runtime) -> None:
42
+ await self.install_skills(runtime, SKILLS_DIR)
40
43
  directory = POOL_DIR.format(version=self.config.version)
41
44
  binary = f"{directory}/pool"
42
45
  script = INSTALL.replace("{version}", self.config.version).replace(
@@ -24,6 +24,7 @@ RLM_REPO = "github.com/PrimeIntellect-ai/rlm.git"
24
24
  RLM_HOME = ".rlm"
25
25
  RLM_DIR = "/tmp/vf-rlm"
26
26
  RLM_BIN = f"{RLM_DIR}/bin/rlm"
27
+ SKILLS_DIR = "/task/rlm-skills"
27
28
 
28
29
 
29
30
  class RLMHarnessConfig(HarnessConfig):
@@ -31,9 +32,9 @@ class RLMHarnessConfig(HarnessConfig):
31
32
  """Git ref (branch, tag, or commit) of rlm to install."""
32
33
  max_depth: int = 0
33
34
  """Recursion depth rlm may spawn sub-harnesses to (RLM_MAX_DEPTH)."""
34
- skills: list[BuiltinSkill] = []
35
+ builtin_skills: list[BuiltinSkill] = []
35
36
  """Built-in rlm skills to enable (RLM_SKILLS), e.g. `["edit"]`; empty enables none.
36
- The tool set is fixed (ipython); only built-in skills are selectable."""
37
+ The tool set is fixed (ipython); the base `skills` field takes SKILL.md paths."""
37
38
  summarize_at_tokens: int | tuple[int, int] | None = None
38
39
  """Auto-compaction threshold (RLM_SUMMARIZE_AT_TOKENS): compact the context once it grows
39
40
  past this many tokens. An int is a fixed threshold; a `(lo, hi)` pair draws a per-group
@@ -63,7 +64,7 @@ class RLMHarnessConfig(HarnessConfig):
63
64
  if self.disabled_tools:
64
65
  raise ValueError(
65
66
  "the rlm harness has a fixed tool set (ipython) and does not support "
66
- "`disabled_tools`; use `skills` to enable built-in skills instead."
67
+ "`disabled_tools`; use `builtin_skills` to enable built-in skills instead."
67
68
  )
68
69
  return self
69
70
 
@@ -71,8 +72,11 @@ class RLMHarnessConfig(HarnessConfig):
71
72
  class RLMHarness(Harness[RLMHarnessConfig]):
72
73
  APPENDS_SYSTEM_PROMPT = True
73
74
  SUPPORTS_MCP = True
75
+ SUPPORTS_SKILLS = True
74
76
 
75
77
  async def setup(self, runtime: Runtime) -> None:
78
+ # Before the installer: install.sh packages the skills it finds.
79
+ await self.install_skills(runtime, SKILLS_DIR)
76
80
  # install.sh fetches curl/uv itself; add git only when the image lacks it.
77
81
  install = (
78
82
  "command -v git >/dev/null 2>&1 || "
@@ -123,8 +127,8 @@ class RLMHarness(Harness[RLMHarnessConfig]):
123
127
  }
124
128
  if system_prompt is not None:
125
129
  env["RLM_APPEND_TO_SYSTEM_PROMPT"] = system_prompt
126
- if self.config.skills:
127
- env["RLM_SKILLS"] = ",".join(self.config.skills)
130
+ if self.config.builtin_skills:
131
+ env["RLM_SKILLS"] = ",".join(self.config.builtin_skills)
128
132
  if mcp_urls:
129
133
  env["RLM_MCP_CONFIG"] = json.dumps(
130
134
  {"mcpServers": {name: {"url": url} for name, url in mcp_urls.items()}}
@@ -18,18 +18,10 @@ class Terminus2HarnessConfig(HarnessConfig):
18
18
  class Terminus2Harness(Harness[Terminus2HarnessConfig]):
19
19
  APPENDS_SYSTEM_PROMPT = True
20
20
  SUPPORTS_MCP = False
21
+ # Beyond the usual host leaks, Terminus drives tmux: on the host its tmux server —
22
+ # and the `tmux kill-server` cleanup in `launch` — would share the user's own.
21
23
 
22
24
  async def setup(self, runtime: Runtime) -> None:
23
- # TODO: Terminus drives tmux; on the host (subprocess) runtime its tmux server — and the
24
- # `tmux kill-server` cleanup in `launch` — share the host's tmux, so a host run can kill
25
- # the user's own tmux session. Until tmux is isolated (a dedicated `tmux -L` socket + a
26
- # created, private TMUX_TMPDIR), refuse the host runtime; run Terminus 2 in a container.
27
- if runtime.type == "subprocess":
28
- raise RuntimeError(
29
- "Terminus 2 drives tmux and is unsafe on the subprocess (host) runtime — its tmux "
30
- "cleanup can kill the host's tmux server. Run it in a container runtime "
31
- "(--env.agent.harness.runtime.type docker|prime|modal)."
32
- )
33
25
  source = PROGRAM_SOURCE.replace("{version}", self.config.version)
34
26
  await runtime.prepare_uv_script(source, self.config.resolved_env)
35
27
 
verifiers/v1/legacy.py CHANGED
@@ -431,12 +431,23 @@ class LegacyEnvServer(EnvServer):
431
431
  state_columns=["trajectory"],
432
432
  )
433
433
 
434
+ @staticmethod
435
+ def _row(req: RunRequest) -> int:
436
+ """The dataset row a request addresses — the bridge's dataset lives
437
+ server-side, so requests must carry `task_idx` (v1 servers take `task_data`)."""
438
+ if req.task_idx is None:
439
+ raise ValueError(
440
+ "legacy env server requests address the dataset by task_idx"
441
+ )
442
+ return req.task_idx
443
+
434
444
  async def _run(self, req: RunRequest) -> RunResponse:
435
- out = await self._run_v0(req.task_idx, req.client, req.model, req.sampling)
445
+ task_idx = self._row(req)
446
+ out = await self._run_v0(task_idx, req.client, req.model, req.sampling)
436
447
  # Trust the bridge-minted record; serialize it once (mirrors `EnvServer`).
437
448
  return RunResponse.model_construct(
438
449
  episode=Episode.of(
439
- rollout_output_to_trace(out, req.task_idx), env=self.taskset_id
450
+ rollout_output_to_trace(out, task_idx), env=self.taskset_id
440
451
  )
441
452
  )
442
453
 
verifiers/v1/loaders.py CHANGED
@@ -187,8 +187,7 @@ def environment_class(taskset_id: str, env_id: str = "") -> type[Env]:
187
187
  def load_environment(config: EnvConfig) -> Env:
188
188
  """Construct the env for `config`. Every construction site (eval, serve, gepa)
189
189
  goes through here so subclass envs load everywhere."""
190
- taskset_id = config.taskset.id if config.taskset is not None else ""
191
- return environment_class(taskset_id, config.id)(config)
190
+ return environment_class(config.taskset.id, config.id)(config)
192
191
 
193
192
 
194
193
  def load_taskset(config: TasksetConfig) -> Taskset:
@@ -239,8 +238,7 @@ def resolve_env_config(data: dict | EnvConfig | None) -> EnvConfig:
239
238
  validate. The one entry every consumer takes (CLI, TOML, the env-server wire),
240
239
  so role fields always validate against the real config class."""
241
240
  if isinstance(data, EnvConfig):
242
- taskset_id = data.taskset.id if data.taskset is not None else ""
243
- cls = env_config_type(taskset_id, data.id)
241
+ cls = env_config_type(data.taskset.id, data.id)
244
242
  if isinstance(data, cls):
245
243
  return data # already at least as specifically typed — keep
246
244
  data = data.model_dump()
@@ -260,4 +258,4 @@ def resolve_env_config(data: dict | EnvConfig | None) -> EnvConfig:
260
258
  def task_type(taskset_id: str) -> type[Task]:
261
259
  """The taskset's `Task` subclass from its generic parameters — no data is
262
260
  loaded, so replay can cheaply recover the task type. Falls back to `Task`."""
263
- return generic_type(taskset_class(taskset_id), Task, origin=Taskset) or Task
261
+ return taskset_class(taskset_id).task_type()
verifiers/v1/push.py CHANGED
@@ -181,9 +181,7 @@ def push_traces(
181
181
  return finish(error="no PRIME_API_KEY (run `prime login`)")
182
182
 
183
183
  traces = [trace for episode in episodes for trace in episode.traces]
184
- env_name = (
185
- config.env.taskset.id if config.env.taskset is not None else ""
186
- ) or config.id
184
+ env_name = (config.env.taskset.id) or config.id
187
185
  metrics = _run_metrics(episodes, traces)
188
186
  samples = _build_samples(episodes)
189
187
  num_examples = len({t.task.data.idx for t in traces})
@@ -141,13 +141,25 @@ class EnvClient:
141
141
  return await self._request(InfoRequest(), InfoResponse)
142
142
 
143
143
  async def run(
144
- self, task_idx: int, client: ClientConfig, model: str, sampling: SamplingConfig
144
+ self,
145
+ client: ClientConfig,
146
+ model: str,
147
+ sampling: SamplingConfig,
148
+ task_data: dict | None = None,
149
+ # TODO: remove task_idx addressing once v0 (the legacy bridge) is deprecated.
150
+ task_idx: int | None = None,
145
151
  ) -> WireEpisode:
146
- """Run one rollout for `task_idx`; return its episode record — flat traces
147
- (typed `Trace[WireTaskData]`) plus the shared stamp."""
152
+ """Run one rollout; return its episode record — flat traces (typed
153
+ `Trace[WireTaskData]`) plus the shared stamp. A v1 server takes the task
154
+ itself (`task_data`, its dumped `TaskData`); the legacy bridge addresses
155
+ its server-side dataset by `task_idx`."""
148
156
  response = await self._request(
149
157
  RunRequest(
150
- task_idx=task_idx, client=client, model=model, sampling=sampling
158
+ task_data=task_data,
159
+ task_idx=task_idx,
160
+ client=client,
161
+ model=model,
162
+ sampling=sampling,
151
163
  ),
152
164
  RunResponse,
153
165
  )
@@ -22,29 +22,33 @@ from verifiers.v1.serve.types import (
22
22
  RunRequest,
23
23
  RunResponse,
24
24
  )
25
+ from verifiers.v1.task import Task, task_data_cls
25
26
  from verifiers.v1.types import SamplingConfig
26
27
 
27
28
  logger = logging.getLogger(__name__)
28
29
 
29
- MAX_LAZY_TASKS = 1_000_000
30
- """Most tasks an infinite taskset's generator is willing to build (and cache) per worker."""
31
-
32
30
 
33
31
  class EnvServer:
34
32
  def __init__(
35
33
  self, config: EnvConfig, address: str = "tcp://127.0.0.1:5000"
36
34
  ) -> None:
37
35
  self.address = address
38
- self.taskset_id = config.taskset.id if config.taskset is not None else ""
36
+ self.taskset_id = config.taskset.id
39
37
  self.env = load_environment(config)
40
- # A finite taskset materializes up front; an infinite one is pulled off its
41
- # generator on demand, so `num_tasks=None` on the wire means infinite.
42
- self._task_iter = iter(self.env.taskset.load())
43
- self._tasks: list = []
38
+ self.task_cls = type(self.env.taskset).task_type()
39
+ self.data_cls = task_data_cls(self.task_cls)
40
+ # A dispatched task is its client-side model_dump(): a field excluded from
41
+ # serialization would vanish on the wire and rebuild silently defaulted, so
42
+ # refuse to serve such a taskset.
43
+ excluded = [
44
+ name for name, field in self.data_cls.model_fields.items() if field.exclude
45
+ ]
46
+ if excluded:
47
+ raise ValueError(
48
+ f"{self.data_cls.__name__} excludes {excluded} from serialization — "
49
+ "a served task must survive the wire whole (drop exclude=True)"
50
+ )
44
51
  self.num_tasks: int | None = None
45
- if not type(self.env.taskset).INFINITE:
46
- self._tasks = list(self._task_iter)
47
- self.num_tasks = len(self._tasks)
48
52
  # v1 envs never group-score (siblings score inside the env's own rollout);
49
53
  # only the legacy (v0) bridge sets this.
50
54
  self.requires_group_scoring = False
@@ -69,8 +73,8 @@ class EnvServer:
69
73
  @classmethod
70
74
  def run_server(cls, address_queue=None, **kwargs) -> None:
71
75
  """Run a spawned server and report its concrete address when requested."""
72
- # Pin tqdm to a threading lock first, so the taskset load never leaks a
73
- # multiprocessing semaphore (resource_tracker warning at shutdown).
76
+ # Pin tqdm to a threading lock first, so a dataset pull (legacy bridge) never
77
+ # leaks a multiprocessing semaphore (resource_tracker warning at shutdown).
74
78
  use_threading_tqdm_lock()
75
79
  server = cls(**kwargs)
76
80
  if address_queue is not None:
@@ -83,24 +87,18 @@ class EnvServer:
83
87
  # of a spurious multiprocessing traceback, matching serve_env's own handling.
84
88
  pass
85
89
 
86
- def _task(self, idx: int):
87
- """The task at `idx`; an infinite taskset generates (and caches) up to `idx`
88
- on demand. Generation must be deterministic — every pool worker runs its own
89
- `load()`, so idx-addressing relies on all producing the same sequence. The
90
- `MAX_LAZY_TASKS` cap fails a runaway driver's request instead of hanging the
91
- worker generating toward it."""
92
- while len(self._tasks) <= idx:
93
- if idx >= MAX_LAZY_TASKS:
94
- raise IndexError(
95
- f"task_idx {idx} exceeds the lazy-generation cap ({MAX_LAZY_TASKS})"
96
- )
97
- try:
98
- self._tasks.append(next(self._task_iter))
99
- except StopIteration:
100
- raise IndexError(
101
- f"task_idx {idx} out of range ({len(self._tasks)} tasks)"
102
- ) from None
103
- return self._tasks[idx]
90
+ def _build_task(self, task_data: dict | None) -> Task:
91
+ """Rebuild a request's task from its wire data: validate into the taskset's
92
+ declared `TaskData` type and wrap it in the declared `Task` with the config's
93
+ task subtree — the same construction the taskset's own `load()` performs. The
94
+ client owns the taskset; this server never `load()`s data, so pool workers
95
+ don't each pull the dataset."""
96
+ if task_data is None:
97
+ raise ValueError(
98
+ "v1 env server requests carry task_data (task_idx addresses the legacy bridge)"
99
+ )
100
+ data = self.data_cls.model_validate(task_data)
101
+ return self.task_cls(data, self.env.config.taskset.task)
104
102
 
105
103
  def _client(self, client_config: ClientConfig, model: str) -> Client:
106
104
  """Cache clients because renderer initialization builds a tokenizer pool."""
@@ -123,7 +121,7 @@ class EnvServer:
123
121
 
124
122
  async def _run(self, req: RunRequest) -> RunResponse:
125
123
  ctx = self._context(req.client, req.model, req.sampling)
126
- (slot,) = self.env.slots(self._task(req.task_idx))
124
+ (slot,) = self.env.slots(self._build_task(req.task_data))
127
125
  # The gate spans requests: `--env.max-concurrent` bounds this worker's
128
126
  # agent runs the same way the in-process eval's semaphore does.
129
127
  episode = await self.env.run_slot(slot, ctx, self._gate)
@@ -182,7 +180,7 @@ class EnvServer:
182
180
  "EnvServer up: taskset=%s address=%s tasks=%s group_scoring=%s",
183
181
  self.taskset_id,
184
182
  self.address,
185
- self.num_tasks if self.num_tasks is not None else "infinite",
183
+ self.num_tasks if self.num_tasks is not None else "client-side",
186
184
  self.requires_group_scoring,
187
185
  )
188
186
  poller = zmq.asyncio.Poller()
@@ -1,6 +1,6 @@
1
1
  from typing import ClassVar
2
2
 
3
- from pydantic import BaseModel, Field, field_serializer
3
+ from pydantic import BaseModel, Field, field_serializer, model_validator
4
4
 
5
5
  from verifiers.v1.clients.config import ClientConfig
6
6
  from verifiers.v1.task import WireTaskData
@@ -34,7 +34,9 @@ class InfoRequest(BaseRequest):
34
34
 
35
35
  class InfoResponse(BaseResponse):
36
36
  num_tasks: int | None = None
37
- """Task count; `None` means the taskset is infinite (bound runs with `num_tasks`)."""
37
+ """Task count. Only the legacy bridge (whose dataset lives server-side) reports
38
+ one; a v1 server is stateless — its tasks live on the client — so this stays
39
+ `None`."""
38
40
  requires_group_scoring: bool = False
39
41
  """Whether tasks must be run as whole groups — legacy (v0) envs only; a v1
40
42
  server always reports False (sibling-dependent signals run inside the env's
@@ -42,12 +44,25 @@ class InfoResponse(BaseResponse):
42
44
 
43
45
 
44
46
  class RunRequest(BaseRequest):
47
+ """One env-rollout. v1 ships the task itself (`task_data`, the dumped `TaskData`
48
+ the server validates into the taskset's declared type); the legacy bridge
49
+ addresses its server-side dataset by row (`task_idx`)."""
50
+
45
51
  method: ClassVar[str] = "run"
46
- task_idx: int = Field(ge=0)
52
+ task_data: dict | None = None
53
+ task_idx: int | None = Field(None, ge=0)
47
54
  client: ClientConfig
48
55
  model: str
49
56
  sampling: SamplingConfig
50
57
 
58
+ @model_validator(mode="after")
59
+ def _exactly_one(self) -> "RunRequest":
60
+ if (self.task_data is None) == (self.task_idx is None):
61
+ raise ValueError(
62
+ "exactly one of task_data (v1) or task_idx (legacy) must be set"
63
+ )
64
+ return self
65
+
51
66
 
52
67
  class RunResponse(BaseResponse):
53
68
  episode: WireEpisode | None = None
verifiers/v1/taskset.py CHANGED
@@ -29,8 +29,9 @@ from pydantic import SerializeAsAny
29
29
  from pydantic_config import BaseConfig
30
30
  from typing_extensions import TypeVar
31
31
 
32
- from verifiers.v1.task import TaskConfig, TaskT, resolve_server_config
32
+ from verifiers.v1.task import Task, TaskConfig, TaskT, resolve_server_config
33
33
  from verifiers.v1.types import ID
34
+ from verifiers.v1.utils.generic import generic_type
34
35
  from verifiers.v1.utils.install import env_name
35
36
  from verifiers.v1.utils.sampling import sample
36
37
 
@@ -66,6 +67,10 @@ class Taskset(Generic[TaskT, TasksetConfigT]):
66
67
  def __init__(self, config: TasksetConfigT) -> None:
67
68
  self.config = config
68
69
 
70
+ @classmethod
71
+ def task_type(cls) -> type[Task]:
72
+ return generic_type(cls, Task, origin=Taskset) or Task
73
+
69
74
  def load(self) -> Iterable[TaskT]:
70
75
  raise NotImplementedError
71
76
 
@@ -87,8 +87,8 @@ class HarborData(TaskData):
87
87
  difficulty: str | None = None
88
88
  category: str | None = None
89
89
  tags: list[str] = []
90
- task_dir: str = Field("", exclude=True)
91
- """Host path to the task dir; used to stage tests/ to verify, not serialized."""
90
+ task_dir: str = ""
91
+ """Host path to the task dir; used to stage tests/ to verify."""
92
92
  verifier_env: dict[str, str] = {}
93
93
  """Raw [verifier.env] entries (literals or `${VAR}`/`${VAR:-default}` templates).
94
94
  Resolved against the host environment at scoring time, like `harbor run` — so a
@@ -95,6 +95,18 @@ def validate_pairing(
95
95
  f"{task_cls.__name__} exposes tool servers (MCP). Run it with a harness that "
96
96
  f"supports MCP (e.g. --env.agent.harness.id bash), or use tasks without tools."
97
97
  )
98
+ if not harness.SUPPORTS_SKILLS and harness.config.skills:
99
+ raise ValueError(
100
+ f"Harness {harness.config.id!r} has no native skill support, but "
101
+ "`skills` is set. Run them with a harness whose program discovers "
102
+ "skills (e.g. --env.agent.harness.id claude-code)."
103
+ )
104
+ if harness.NEEDS_CONTAINER and isinstance(runtime_config, SubprocessConfig):
105
+ raise ValueError(
106
+ f"Harness {harness.config.id!r} needs a container runtime "
107
+ "(NEEDS_CONTAINER), but this run resolves to the subprocess runtime; "
108
+ "use --env.agent.harness.runtime.type docker or prime."
109
+ )
98
110
  if not harness.SUPPORTS_USER_SIM and task_cls.user is not None:
99
111
  raise ValueError(
100
112
  f"Harness {harness.config.id!r} does not drive a user simulator, but "
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev16
3
+ Version: 0.2.2.dev18
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -170,42 +170,42 @@ verifiers/utils/version_utils.py,sha256=am3hZLnlaUWTFdllGZR1VMN91hEhiiE8FDmPf1cO
170
170
  verifiers/v1/__init__.py,sha256=r8khgi7_KSY4AfevEefSslyxytScmMzzCBA6kjqvd1I,6653
171
171
  verifiers/v1/agent.py,sha256=x_n4eV4iiUPLl8OWQULrDXeMzLTLA0_GXIYJED2nJyE,21261
172
172
  verifiers/v1/decorators.py,sha256=XRMkUQSyvXCYP5fOwzBYV5qEOlxLg6rzYhnhrHHHTIQ,3838
173
- verifiers/v1/env.py,sha256=XQHzcTCNLmAbupkFp3edvUJqtSxs6U3mJHVz3hAI6Y4,24734
173
+ verifiers/v1/env.py,sha256=79MMfFhQPJnLk5O96WqU_EuaVWMflz-XgiEszHvilCI,24565
174
174
  verifiers/v1/episode.py,sha256=a-8IwYecfaBAD0NPbPREQWUybx-KrgLev2JHC-shjgU,2301
175
175
  verifiers/v1/errors.py,sha256=pIn9sR12bcd-fBJhMsZjiZCWjH3Uf9Vwhfu1LakFhMY,7007
176
176
  verifiers/v1/graph.py,sha256=Mivq5jICUcyK-fhomFTwP4fQn688MXg6-hmdf03aC4Y,27703
177
- verifiers/v1/harness.py,sha256=dZUml8KKEPQZue7wIe8Vh7aXBCoKk_cUQeyLYyCiHdY,6541
177
+ verifiers/v1/harness.py,sha256=2BOeYLwREgpbZbCshpGkbxHtKEFV5izLO8eT8Ay3YE0,8603
178
178
  verifiers/v1/judge.py,sha256=u15e7YXPUKi-DWhqTrwsRy0M7UTj7O0066Rwm-RAXmU,11668
179
- verifiers/v1/legacy.py,sha256=QlkOI5hZ--54FB2umnM1xA_OmO5J6o7jv9cxeydSbdY,21848
180
- verifiers/v1/loaders.py,sha256=i-cusZiJNcYQm-E8-CcEhEvtjPC5pR5nyCtlUDoNN3E,9974
181
- verifiers/v1/push.py,sha256=Kjsk0Z-ylJMGA5JuQJK0G44IscKO1SsZ4CcO5_Sbsjw,9385
179
+ verifiers/v1/legacy.py,sha256=b53EDDOImu8rSaZR_ZE7jNaQTJ6aIw5T8hxlXU4fTh0,22280
180
+ verifiers/v1/loaders.py,sha256=ISI0UKk8UXUOJke8n2M3F0EpPmqqnTgey0QDq9Ryo8I,9808
181
+ verifiers/v1/push.py,sha256=_wRGO6UTQURSBFyPQzXSdkZGQEGZh6PyVzKNJgkyJVM,9329
182
182
  verifiers/v1/retries.py,sha256=5-LQZ2A7Tr1dpTtoJ-bX0_BrzecHm9Drf3EfQUIjVAw,6010
183
183
  verifiers/v1/rollout.py,sha256=cnVa7I98IEbIj-UmlRIe_SFZrpjJNVFnuOsQP4g4ask,16807
184
184
  verifiers/v1/scoring.py,sha256=I_mtqhTZ195hj2-CB5jzr3BY5GFKgNEvTf331f7Is4k,5740
185
185
  verifiers/v1/session.py,sha256=CyvKR1oN_UCP3SOm2vhGwn_wWlwB68GklJeFhwX5Bg8,7780
186
186
  verifiers/v1/state.py,sha256=hYo7DkOMNyh5pcKz7Xr7j-qQN2-LOKKDTlD-KdKmW40,698
187
187
  verifiers/v1/task.py,sha256=zzZsfDiulMPKorVrJHFCTYEQKagdVtAg_OlP-P0YRgk,11782
188
- verifiers/v1/taskset.py,sha256=L2wLzdX25ECyzspO8-TJcyGyvTjcuOXTduItOG42mPI,4209
188
+ verifiers/v1/taskset.py,sha256=0euaEA_Bm5hFdXifdmV3fi7p0yoHofvx7w-UOExxhBo,4386
189
189
  verifiers/v1/trace.py,sha256=CnVt4lOfyzEOoXIv_XSQqhYwxY77Tu_O__FLNVFTZxs,25146
190
190
  verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
191
191
  verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
192
192
  verifiers/v1/cli/debug.py,sha256=e-imOXOK1WkAkqEZqiGKsnsohDnFgm-U63sFQn_Htrg,10954
193
- verifiers/v1/cli/gepa.py,sha256=rYKxPiFDViH_Dcpr0ea8LHM0XQTc7itg2lH24EBH6YU,4068
193
+ verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
194
194
  verifiers/v1/cli/init.py,sha256=9rG0yjhZpKft1qhno6BI8GmpOzPlQ3r0ZBnujsevR_c,9785
195
- verifiers/v1/cli/output.py,sha256=sweTUDZeazNhe-DHyERyjFaW9m5Cs9XqNq2WK5zpay0,5919
195
+ verifiers/v1/cli/output.py,sha256=878YmRnx3aMCbd6LQoTTLU88k5EC53sX9iJJgOj4jzo,5886
196
196
  verifiers/v1/cli/replay.py,sha256=d5BBrTeaun8B9BXtkOu8VCL7gnnL6tFG_zNetqsoIuM,10224
197
197
  verifiers/v1/cli/resolve.py,sha256=1sV4z04v_HEs86Oi8ODJqfujXjrbShbu3u2kDpgpj94,4079
198
198
  verifiers/v1/cli/serve.py,sha256=6rGujVxcMBso-K3fTM8EdURzw2LHshyekJfbFY5KU7A,2659
199
199
  verifiers/v1/cli/validate.py,sha256=KvOGH6cacU8nT2hX_J230VLw2q-0HEPOdIu1Wsgr8TU,9543
200
200
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
201
201
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
202
- verifiers/v1/cli/dashboard/eval.py,sha256=dqLA-X4NEwQf6xcV-dEdswv_6xNqw4AxvcmTY4dTyuQ,33705
202
+ verifiers/v1/cli/dashboard/eval.py,sha256=wQIMksQoSxqlCqXp57exRaOap-kepHzA0_6inQiSFkY,33656
203
203
  verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
204
204
  verifiers/v1/cli/dashboard/validate.py,sha256=7olY6oq4-zY8tvrKTJQTBwqG5ZfO5FBJ6ZHHIlKoHxo,3646
205
205
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
206
206
  verifiers/v1/cli/eval/main.py,sha256=a1WzbkQP21pQJWI-AdgLCj6YB1bg4pDzVZF7OWT3WXM,5477
207
- verifiers/v1/cli/eval/resume.py,sha256=rUbte9oHfL9SljkDwuJGY682Uo_AI27-j6aJlha-Y2E,5822
208
- verifiers/v1/cli/eval/runner.py,sha256=nrAYAv18K3qUVbfb3K-8ZVioYv11lPBQnfm3352c2VY,10161
207
+ verifiers/v1/cli/eval/resume.py,sha256=can5G76NpUdEaBZF4-lClFB09XRZ82iRoJ-7oKxrcOs,7093
208
+ verifiers/v1/cli/eval/runner.py,sha256=IE5tiW3qjp6jtDbd92S6xMEW4UI_PRANN7oqJipuGr0,11204
209
209
  verifiers/v1/clients/__init__.py,sha256=wl1Qdx6lrPXnxvHaEip9NaUAad8i_1H6lwyWjGCCUbQ,511
210
210
  verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
211
211
  verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
@@ -213,7 +213,7 @@ verifiers/v1/clients/eval.py,sha256=7eyc3xB489E_SWH10h42NybL3MB22W5p9Tttza__P9Q,
213
213
  verifiers/v1/clients/train.py,sha256=3y2BtL7aYeSaiknJzKh-wLHsfi54nA4pt3BBWQ3ROMs,13043
214
214
  verifiers/v1/configs/__init__.py,sha256=PlbPK0iE3vw7dqqQ-3JMdzAGiNh6sXe4p9RPhQJKRR8,345
215
215
  verifiers/v1/configs/debug.py,sha256=Z-QVPheQLu_cph_ZrCqSZI0ufkF4n0DXfvpaFTNWbHY,2831
216
- verifiers/v1/configs/env.py,sha256=hNfMQYEPKInnfkNhd8m0LtuvPmolQjFdEVrTZzuBEHQ,7365
216
+ verifiers/v1/configs/env.py,sha256=GfV1_T26Fjh-XgomiGVeWmAIyIe-FjIpQ0uYKJGhBN0,7280
217
217
  verifiers/v1/configs/eval.py,sha256=nC5CmBmN3_ZoXK8QGoEiMecBRfCJxs-Qs-Nc8qZGJ_I,3808
218
218
  verifiers/v1/configs/init.py,sha256=u2Y0lKSdL9rVs2HW8qq3gIYfGNT8x8c9k355lt3Yq9A,1166
219
219
  verifiers/v1/configs/replay.py,sha256=ApH1FvuDeju-E2EHMmTsmqXu844bEPEHQbZQgAb76MM,2899
@@ -239,28 +239,28 @@ verifiers/v1/gepa/reflection.py,sha256=3wiMaXphu40i3nt38UhwNP06dMihqoVvaaBDO-52I
239
239
  verifiers/v1/gepa/runner.py,sha256=XYsXSvImrwkv6TuNtXBOEfUXfqRupHSj-VKnVYv7_3k,5374
240
240
  verifiers/v1/harnesses/__init__.py,sha256=BUIgBrFU9v8sqgPl5h6YtKHUYIpA-4xih9YhHV0zVRQ,1301
241
241
  verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
242
- verifiers/v1/harnesses/bash/harness.py,sha256=MNRhClpeZC1E_QxdDqSYmD4_E5EweeKyVtx8YLXlF8E,5201
242
+ verifiers/v1/harnesses/bash/harness.py,sha256=4ZWxx6dFlZEH4nLgb6NqfpSOn5LxFZsJ4U5J77clM_o,5229
243
243
  verifiers/v1/harnesses/bash/program.py,sha256=fu14BwvVEL3VJddpBC8zpZoapCWa83acAXR5XIe6SUs,15029
244
244
  verifiers/v1/harnesses/claude_code/__init__.py,sha256=JkZQylMSA3o1fvBx_NM41y_umuahmOSqFjTrJ2hryTw,171
245
- verifiers/v1/harnesses/claude_code/harness.py,sha256=N1tCfrCSn38zTFRHzE2psGYaM7-UxsXcH74ZveKArhY,3762
245
+ verifiers/v1/harnesses/claude_code/harness.py,sha256=d5fZ-pyP6RdJ7AHmyUa8B4tAlwm_427i0NFAOtsyLic,3926
246
246
  verifiers/v1/harnesses/codex/__init__.py,sha256=ocyBvlpSO8c3pT6D1ebrQohoXYESQcCyPT5kN3j1-N4,132
247
- verifiers/v1/harnesses/codex/harness.py,sha256=Zqo7dPdYHtQeFqp_tGLWNNxTzWB8GookQdNOvpP8mSQ,7262
247
+ verifiers/v1/harnesses/codex/harness.py,sha256=s3LA7beFu_Skp_lJsRPj2qW9VwX0uRc0sCcQcDgsfU4,7374
248
248
  verifiers/v1/harnesses/kimi_code/__init__.py,sha256=1BEotSfuzmVcE7UrxHeAREM5d7J20cGkonHuowlrzu4,161
249
- verifiers/v1/harnesses/kimi_code/harness.py,sha256=k0zXoEfMG4AS7yIDN6YAvuwqL8N2-RG4dSi7FG8BRs8,3527
249
+ verifiers/v1/harnesses/kimi_code/harness.py,sha256=G6QLWBH3JdNWD_UaiT5zEmT6hXV6STmPVAj7Ir2QTlU,3644
250
250
  verifiers/v1/harnesses/mini_swe_agent/__init__.py,sha256=1OBu2Y-XqL72-dGhFdgB6sr5uX_peEEXT9vEjPy1ovU,182
251
251
  verifiers/v1/harnesses/mini_swe_agent/harness.py,sha256=wypTAvzix-7B7jMNg3JKCf2ytKmAm6_jNa5p3yGO9aA,2351
252
252
  verifiers/v1/harnesses/mini_swe_agent/program.py,sha256=XPjDEsbJACB34tkUX9dJdIziWyKnMTvAJMLevI_FFxU,159
253
253
  verifiers/v1/harnesses/null/__init__.py,sha256=XDeKOPeoXUQi9US9_Z3_yH9YaZ4EiOMKJmUtNNdtD7s,127
254
- verifiers/v1/harnesses/null/harness.py,sha256=awHI7O4hTEfc-7XGuWhy9b5SMfLBCfhPqjmFzn5hse4,2448
254
+ verifiers/v1/harnesses/null/harness.py,sha256=AmfJkzHajfK3py8O5hbGT5lmsyJIMrA85NoTilCfvIk,2476
255
255
  verifiers/v1/harnesses/null/program.py,sha256=qEkzf45Oh-PlotpJxCD2nrWDf1Ocf_-b58vTerxrNzo,8357
256
256
  verifiers/v1/harnesses/pi/__init__.py,sha256=1mSnzOf-RdEchdeg6IHtcdobG5KIpAhuTdUIgsCIn74,117
257
- verifiers/v1/harnesses/pi/harness.py,sha256=Ns7CDiYT1w-BpKilUCBhdv85e1u-U7NwhkL2l3Y8QNk,9523
257
+ verifiers/v1/harnesses/pi/harness.py,sha256=YjXX-RfYmzERNfC4ju9oQHYpfZWJnq2M5pAM6DCZLcM,10070
258
258
  verifiers/v1/harnesses/pool/__init__.py,sha256=HTYsNiWGdiNEsSHj4xhPmshJSNb7SHjQ7d4yZQ4p3HM,127
259
- verifiers/v1/harnesses/pool/harness.py,sha256=5EGHK8HiLMvJuNKk5mRT949494-kGBjjA1hqk7Kipc4,5380
259
+ verifiers/v1/harnesses/pool/harness.py,sha256=Dui9Ov2r0qYEGAgs1lfJ-ExfLngAl3iKbmzghKmGLoo,5494
260
260
  verifiers/v1/harnesses/rlm/__init__.py,sha256=hwx51xTdEWzzYgMj-p4ETNcptoub3IPMDbEdJ4uB8Ns,122
261
- verifiers/v1/harnesses/rlm/harness.py,sha256=eouQp5DmH6Ni1_jPoCD0c4WQz71gyftOhbc3ydQKT94,6621
261
+ verifiers/v1/harnesses/rlm/harness.py,sha256=83xOPXmWAQwm-4xvTyEymWvxswfLyOLf-SDQIc_poGg,6849
262
262
  verifiers/v1/harnesses/terminus_2/__init__.py,sha256=l1pwfg1HyVDEqT9efdnFWcXqaZwtKwe7TrplkyxZSA0,166
263
- verifiers/v1/harnesses/terminus_2/harness.py,sha256=TBnw1rGfmmvhsHi38NXYkaySpPJPxvufRomZywscz3w,3556
263
+ verifiers/v1/harnesses/terminus_2/harness.py,sha256=JyBUrL50n0wz8--GJ_gSDiuVNTUQQdJj41w8pnGowOY,2987
264
264
  verifiers/v1/harnesses/terminus_2/program.py,sha256=DbkjUXTC_8-N05G9DJfzhGaMZQAfdTSNmf_THuKE1Zw,2606
265
265
  verifiers/v1/interception/__init__.py,sha256=SnQtCpmvc4ynIAiKDIobGVgRgss8BMTQTpZvKSygn24,2735
266
266
  verifiers/v1/interception/base.py,sha256=Riwf_qng-Gtfa4Phy2CH_HgHD4O23JaFkUpaMPc6sZw,2708
@@ -289,13 +289,13 @@ verifiers/v1/runtimes/subprocess.py,sha256=blhgfr_0YBzfxUNqiXCwMjdEUWwYW3ybebn70
289
289
  verifiers/v1/runtimes/docker/__init__.py,sha256=PoBwt7Ggd_W5TtNgUX41vPuut5_c70cQTmVWnSvSgIQ,15599
290
290
  verifiers/v1/runtimes/docker/egress.py,sha256=ELznqwFHpt1w6g4-AvoWRV_UkgvSMc5iq1DjCjvTM-k,14461
291
291
  verifiers/v1/serve/__init__.py,sha256=UDlVB8hIHIvsV4NsYUWX_icS4g67-MQI72tqMnGmZ5o,641
292
- verifiers/v1/serve/client.py,sha256=3zhHbAoCK6__2l8xCvZDENzSH0Ay6olHx3xtFW52eHI,6734
292
+ verifiers/v1/serve/client.py,sha256=4R-vkhlC_49x5ofl2J_uwITc9gSNjD2fBWJ5xK4WRu8,7132
293
293
  verifiers/v1/serve/pool.py,sha256=iOOg8PfYPY0prYlHDrrVVI6efElp-lwhqmbFdYYyKE8,14716
294
- verifiers/v1/serve/server.py,sha256=g388bicBa6pWYv-zdgbtuqRngMCE6jZJKJOeT1qL_qA,9576
295
- verifiers/v1/serve/types.py,sha256=Yub1EhibbCAmuvPUAg_0L4XgvTXxWDP9YuZdn7n3uv4,2336
294
+ verifiers/v1/serve/server.py,sha256=HoVf9xcbmD2IzIA9kgZHkxYXT8qVrWs9Uu9VKP0RmiM,9592
295
+ verifiers/v1/serve/types.py,sha256=wXKx1AUqEax7IAk4nYYWjS9uvTgTABQW-pdmKN-kqE4,3009
296
296
  verifiers/v1/tasksets/__init__.py,sha256=E1nVxnZjtvKsw-f9sW0zui5ndMMZnjMBAMdjfE5nKfI,686
297
297
  verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
298
- verifiers/v1/tasksets/harbor/taskset.py,sha256=keTUl54Y7syWVHziRynvT81G6lGIjpx-YBO6XBM8SkU,14593
298
+ verifiers/v1/tasksets/harbor/taskset.py,sha256=6dPdS4NhIFDvrqEiq7-XQOpbeGF0HTtNVat50d2qX8A,14556
299
299
  verifiers/v1/tasksets/lean/__init__.py,sha256=-13Mjoj6oGpiV0hxAF9fsw9-4xWpr5MPJl7oVkcQsBA,767
300
300
  verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
301
301
  verifiers/v1/tasksets/lean/taskset.py,sha256=UsE6R9O20dgUz_0oBtvZlzrWqiAHvhtp5Y1CJZv_GEY,8988
@@ -305,7 +305,7 @@ verifiers/v1/tasksets/textarena/__init__.py,sha256=_R_E3PFP3X0lQGoedu_J9lKlB1a8t
305
305
  verifiers/v1/tasksets/textarena/taskset.py,sha256=y2OD5zwtA_rCRqpC5I8r5fJgboFDkEX_2FIQ8OYcoh4,5252
306
306
  verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
307
307
  verifiers/v1/utils/aio.py,sha256=ZKTHeURNbWhTpHMyYuqMLsdPWVHyn5anfG1IJs5y-Zg,1481
308
- verifiers/v1/utils/compile.py,sha256=dFTXzqlG2534Jg2GwrM6GHyBVbROYJayAYntzLWkFMc,5586
308
+ verifiers/v1/utils/compile.py,sha256=uP2QUz_MiEhG4Kmb0r6LE2VqP9uH6mH-TBWXEV5Y4ok,6247
309
309
  verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
310
310
  verifiers/v1/utils/generic.py,sha256=xwu8xNGdVOhM2K87C5cqhmnRf6apSPDASA15M0S2MaY,2054
311
311
  verifiers/v1/utils/git.py,sha256=5VX5B3011yhclPex8Mc79ZmZ5ZT4hGfKaXoqL12QMdc,4872
@@ -316,8 +316,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
316
316
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
317
317
  verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
318
318
  verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
319
- verifiers-0.2.2.dev16.dist-info/METADATA,sha256=mGdtAAW0dufKathCBaYb0MM0Ir9kdzBDTJDRX0Pf8Ug,4540
320
- verifiers-0.2.2.dev16.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
321
- verifiers-0.2.2.dev16.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
322
- verifiers-0.2.2.dev16.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
323
- verifiers-0.2.2.dev16.dist-info/RECORD,,
319
+ verifiers-0.2.2.dev18.dist-info/METADATA,sha256=fW9wnVLClfpeL0usrYPdbc0cPhKZPhZp2J6oUHbbCsg,4540
320
+ verifiers-0.2.2.dev18.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
321
+ verifiers-0.2.2.dev18.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
322
+ verifiers-0.2.2.dev18.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
323
+ verifiers-0.2.2.dev18.dist-info/RECORD,,