verifiers 0.2.2.dev16__py3-none-any.whl → 0.2.2.dev18__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/dashboard/eval.py +2 -4
- verifiers/v1/cli/eval/resume.py +61 -28
- verifiers/v1/cli/eval/runner.py +62 -46
- verifiers/v1/cli/gepa.py +1 -1
- verifiers/v1/cli/output.py +2 -2
- verifiers/v1/configs/env.py +2 -4
- verifiers/v1/env.py +10 -13
- verifiers/v1/harness.py +36 -0
- verifiers/v1/harnesses/bash/harness.py +1 -0
- verifiers/v1/harnesses/claude_code/harness.py +4 -1
- verifiers/v1/harnesses/codex/harness.py +3 -0
- verifiers/v1/harnesses/kimi_code/harness.py +3 -0
- verifiers/v1/harnesses/null/harness.py +1 -0
- verifiers/v1/harnesses/pi/harness.py +12 -0
- verifiers/v1/harnesses/pool/harness.py +3 -0
- verifiers/v1/harnesses/rlm/harness.py +9 -5
- verifiers/v1/harnesses/terminus_2/harness.py +2 -10
- verifiers/v1/legacy.py +13 -2
- verifiers/v1/loaders.py +3 -5
- verifiers/v1/push.py +1 -3
- verifiers/v1/serve/client.py +16 -4
- verifiers/v1/serve/server.py +31 -33
- verifiers/v1/serve/types.py +18 -3
- verifiers/v1/taskset.py +6 -1
- verifiers/v1/tasksets/harbor/taskset.py +2 -2
- verifiers/v1/utils/compile.py +12 -0
- {verifiers-0.2.2.dev16.dist-info → verifiers-0.2.2.dev18.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev16.dist-info → verifiers-0.2.2.dev18.dist-info}/RECORD +31 -31
- {verifiers-0.2.2.dev16.dist-info → verifiers-0.2.2.dev18.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev16.dist-info → verifiers-0.2.2.dev18.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev16.dist-info → verifiers-0.2.2.dev18.dist-info}/licenses/LICENSE +0 -0
|
@@ -232,7 +232,7 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
232
232
|
grid.add_column()
|
|
233
233
|
seats = config.env.agent_harnesses()
|
|
234
234
|
taskset = config.env.taskset
|
|
235
|
-
env_label = taskset.name if taskset
|
|
235
|
+
env_label = taskset.name if taskset.id else "no taskset"
|
|
236
236
|
if config.env.id:
|
|
237
237
|
env_label = f"{env_name(config.env.id)}+{env_label}"
|
|
238
238
|
# One seat story when every seat resolves the same way (the common case); one
|
|
@@ -254,9 +254,7 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
254
254
|
# value (or our `[...]`/`{...}` delimiters) can carry Rich markup that would otherwise be
|
|
255
255
|
# parsed as styling and dropped. `id` is in the `env` row; harness `runtime.type` too (hidden
|
|
256
256
|
# here), but only for the harness — `taskset.task.tools.runtime.type` has no other display.
|
|
257
|
-
if
|
|
258
|
-
taskset_over := overrides(taskset, skip=frozenset({"id"}))
|
|
259
|
-
):
|
|
257
|
+
if taskset_over := overrides(taskset, skip=frozenset({"id"})):
|
|
260
258
|
grid.add_row("taskset", escape(" · ".join(taskset_over)))
|
|
261
259
|
for role, h in seats.items():
|
|
262
260
|
if harness_over := overrides(h, skip=frozenset({"id", "runtime.type"})):
|
verifiers/v1/cli/eval/resume.py
CHANGED
|
@@ -4,13 +4,20 @@
|
|
|
4
4
|
flags) and writes back into the same dir. `load` keeps the good saved rollouts and
|
|
5
5
|
re-runs what's owed: missing rollouts (never written) and errored ones (dropped and
|
|
6
6
|
redone).
|
|
7
|
+
|
|
8
|
+
A saved rollout is matched to a selected task by content: `task_key` hashes the
|
|
9
|
+
task's wire data. Tasks with identical data are interchangeable, a task whose data
|
|
10
|
+
changed since the interrupted run re-runs, and nothing depends on `data.idx`. The
|
|
11
|
+
legacy (v0) bridge still matches by row index (`key_of`).
|
|
7
12
|
"""
|
|
8
13
|
|
|
14
|
+
import hashlib
|
|
9
15
|
import json
|
|
10
16
|
import tomllib
|
|
11
|
-
from collections import defaultdict
|
|
12
|
-
from collections.abc import Callable
|
|
17
|
+
from collections import Counter, defaultdict
|
|
18
|
+
from collections.abc import Callable, Hashable, Mapping
|
|
13
19
|
from pathlib import Path
|
|
20
|
+
from typing import TypeVar
|
|
14
21
|
|
|
15
22
|
from pydantic_core import from_json
|
|
16
23
|
|
|
@@ -19,6 +26,30 @@ from verifiers.v1.configs.eval import EvalConfig
|
|
|
19
26
|
from verifiers.v1.episode import Episode, WireEpisode
|
|
20
27
|
from verifiers.v1.trace import WireTrace
|
|
21
28
|
|
|
29
|
+
K = TypeVar("K", bound=Hashable)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def task_key(data: Mapping) -> str:
|
|
33
|
+
"""Content identity of one task's wire data — an `exclude_none` dump, the shape
|
|
34
|
+
saved rows already have on disk. `sort_keys` so field order can't split identity."""
|
|
35
|
+
return hashlib.sha256(json.dumps(data, sort_keys=True).encode()).hexdigest()
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def distribute(
|
|
39
|
+
selected_keys: list[K], owed: dict[K, int], num_rollouts: int
|
|
40
|
+
) -> list[int]:
|
|
41
|
+
"""Spread each key's owed rollouts over its selection instances, in order —
|
|
42
|
+
content-identical tasks are interchangeable, so any instance can absorb the
|
|
43
|
+
debt (capped at `num_rollouts` each). Returns one count per selection."""
|
|
44
|
+
remaining = dict(owed)
|
|
45
|
+
counts: list[int] = []
|
|
46
|
+
for key in selected_keys:
|
|
47
|
+
take = min(num_rollouts, remaining.get(key, 0))
|
|
48
|
+
if take:
|
|
49
|
+
remaining[key] -= take
|
|
50
|
+
counts.append(take)
|
|
51
|
+
return counts
|
|
52
|
+
|
|
22
53
|
|
|
23
54
|
def split_resume(argv: list[str]) -> tuple[Path | None, list[str]]:
|
|
24
55
|
"""Pull `--resume <dir>` / `--resume=<dir>` out of argv, returning (dir, the other args).
|
|
@@ -51,25 +82,27 @@ def load_resume_config(resume_dir: Path) -> EvalConfig:
|
|
|
51
82
|
|
|
52
83
|
def load(
|
|
53
84
|
resume_dir: Path,
|
|
54
|
-
|
|
85
|
+
selected_keys: list[K],
|
|
55
86
|
num_rollouts: int,
|
|
56
87
|
complete: Callable[[Episode], bool] | None = None,
|
|
57
88
|
*,
|
|
58
89
|
whole_task: bool = False,
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
kept
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
rows via a temp file + atomic rename
|
|
69
|
-
|
|
70
|
-
bare trace per line) load each trace as a single-trace episode."""
|
|
90
|
+
key_of: Callable[[Mapping], K] | None = None,
|
|
91
|
+
) -> tuple[list[Episode], dict[K, int]]:
|
|
92
|
+
"""Load the good saved rollouts and diff them against the run's target: returns
|
|
93
|
+
(kept episodes, rollouts owed per task key). `selected_keys` is one key per
|
|
94
|
+
selected task (duplicates allowed — a key selected k times is owed up to
|
|
95
|
+
`k * num_rollouts`; spread back over the tasks with `distribute`). `key_of` maps
|
|
96
|
+
a saved row's task data to its key (default `task_key`; the legacy bridge uses
|
|
97
|
+
row indices). `complete` is the keep-verdict (default `episode.ok`); `whole_task`
|
|
98
|
+
redoes a partially-kept task whole (legacy group scoring). Rewrites
|
|
99
|
+
`traces.jsonl` to the kept rows via a temp file + atomic rename; a torn or
|
|
100
|
+
malformed row is owed again, never a crash."""
|
|
71
101
|
path = resume_dir / TRACES_FILE
|
|
72
|
-
|
|
102
|
+
targets = {
|
|
103
|
+
key: count * num_rollouts for key, count in Counter(selected_keys).items()
|
|
104
|
+
}
|
|
105
|
+
keyed = key_of if key_of is not None else task_key
|
|
73
106
|
|
|
74
107
|
def parse(row: dict) -> Episode:
|
|
75
108
|
if sniff_episode(row):
|
|
@@ -77,7 +110,7 @@ def load(
|
|
|
77
110
|
return Episode.of(WireTrace.model_validate(row))
|
|
78
111
|
|
|
79
112
|
verdict = complete if complete is not None else (lambda episode: episode.ok)
|
|
80
|
-
good: dict[
|
|
113
|
+
good: dict[K, list[tuple[bytes, Episode]]] = defaultdict(list)
|
|
81
114
|
if path.exists():
|
|
82
115
|
with path.open("rb") as results:
|
|
83
116
|
for line in results:
|
|
@@ -89,16 +122,16 @@ def load(
|
|
|
89
122
|
except ValueError:
|
|
90
123
|
row = json.loads(line)
|
|
91
124
|
# The task rides each trace; a traceless record (a failure
|
|
92
|
-
# before any trace minted) has no
|
|
125
|
+
# before any trace minted) has no task and is owed again.
|
|
93
126
|
if sniff_episode(row):
|
|
94
|
-
|
|
127
|
+
key = keyed(row["traces"][0]["task"]["data"])
|
|
95
128
|
else:
|
|
96
|
-
|
|
129
|
+
key = keyed(row["task"]["data"])
|
|
97
130
|
except (ValueError, KeyError, IndexError, TypeError):
|
|
98
131
|
# A torn final line (the run died mid-write) or a foreign shape
|
|
99
132
|
# is not a keepable rollout — it's owed again, never a crash.
|
|
100
133
|
continue
|
|
101
|
-
if
|
|
134
|
+
if key not in targets or len(good[key]) >= targets[key]:
|
|
102
135
|
continue
|
|
103
136
|
try:
|
|
104
137
|
episode = parse(row)
|
|
@@ -106,20 +139,20 @@ def load(
|
|
|
106
139
|
continue
|
|
107
140
|
except Exception: # malformed row: redo it
|
|
108
141
|
continue
|
|
109
|
-
good[
|
|
142
|
+
good[key].append(
|
|
110
143
|
(line if line.endswith(b"\n") else line + b"\n", episode)
|
|
111
144
|
)
|
|
112
145
|
keep: list[bytes] = []
|
|
113
146
|
episodes: list[Episode] = []
|
|
114
|
-
owed: dict[
|
|
115
|
-
for
|
|
116
|
-
rows = good.get(
|
|
117
|
-
if whole_task and len(rows) <
|
|
147
|
+
owed: dict[K, int] = {}
|
|
148
|
+
for key, target in targets.items():
|
|
149
|
+
rows = good.get(key, [])
|
|
150
|
+
if whole_task and len(rows) < target:
|
|
118
151
|
rows = [] # a partial unit redoes whole — its kept rows are dropped
|
|
119
152
|
keep.extend(line for line, _ in rows)
|
|
120
153
|
episodes.extend(episode for _, episode in rows)
|
|
121
|
-
if missing :=
|
|
122
|
-
owed[
|
|
154
|
+
if missing := target - len(rows):
|
|
155
|
+
owed[key] = missing
|
|
123
156
|
tmp = path.with_suffix(".jsonl.tmp")
|
|
124
157
|
tmp.write_bytes(b"".join(keep))
|
|
125
158
|
tmp.replace(path)
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -32,21 +32,25 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
|
32
32
|
asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
|
|
33
33
|
)
|
|
34
34
|
out = output_path(config)
|
|
35
|
-
|
|
35
|
+
# One (task, rollouts-to-run) pair per selected task; resume shrinks the counts.
|
|
36
|
+
plan = [(task, config.num_rollouts) for task in tasks]
|
|
36
37
|
# Kept on-disk rollouts rejoin the run as finished episodes; only owed ones re-run.
|
|
37
38
|
finished: list[Episode] = []
|
|
38
39
|
if config.resume is not None:
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
40
|
+
keys = [
|
|
41
|
+
resume.task_key(t.data.model_dump(mode="json", exclude_none=True))
|
|
42
|
+
for t in tasks
|
|
43
|
+
]
|
|
44
|
+
finished, owed = resume.load(out, keys, config.num_rollouts, env.complete)
|
|
42
45
|
if not owed: # already complete - report it and exit successfully
|
|
43
46
|
print(resume.nothing_to_resume_msg(out, len(tasks), config.num_rollouts))
|
|
44
47
|
raise SystemExit(0)
|
|
45
|
-
|
|
48
|
+
counts = resume.distribute(keys, owed, config.num_rollouts)
|
|
49
|
+
plan = [(task, n) for task, n in zip(tasks, counts) if n]
|
|
46
50
|
logger.info(
|
|
47
51
|
"resuming %s: %d task(s), %d rollout(s) owed",
|
|
48
52
|
out,
|
|
49
|
-
len(
|
|
53
|
+
len(plan),
|
|
50
54
|
sum(owed.values()),
|
|
51
55
|
)
|
|
52
56
|
else:
|
|
@@ -70,13 +74,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
|
70
74
|
# Serving resources (shared tool servers, interception) come up once for the
|
|
71
75
|
# run; plan slots inside so the env's agents borrow them.
|
|
72
76
|
async with env.serving():
|
|
73
|
-
planned = [
|
|
74
|
-
slot
|
|
75
|
-
for task in tasks
|
|
76
|
-
for slot in env.slots(
|
|
77
|
-
task, n=owed[task.data.idx] if owed else config.num_rollouts
|
|
78
|
-
)
|
|
79
|
-
]
|
|
77
|
+
planned = [slot for task, n in plan for slot in env.slots(task, n=n)]
|
|
80
78
|
slots = [RunSlot.finished(episode) for episode in finished] + planned
|
|
81
79
|
push_state = None
|
|
82
80
|
if config.push and config.rich:
|
|
@@ -123,6 +121,15 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
123
121
|
if legacy
|
|
124
122
|
else {"config_data": env_config_data(config.env)} # picklable across the spawn
|
|
125
123
|
)
|
|
124
|
+
tasks = []
|
|
125
|
+
if not legacy:
|
|
126
|
+
from verifiers.v1.loaders import load_taskset
|
|
127
|
+
|
|
128
|
+
# The client owns the taskset: load it here, once — the server (and its pool
|
|
129
|
+
# workers) never load data, they rebuild each dispatched task from its request.
|
|
130
|
+
tasks = load_taskset(config.env.taskset).select(
|
|
131
|
+
config.num_tasks, config.shuffle
|
|
132
|
+
)
|
|
126
133
|
# Spawned processes inherit no logging — hand them the main process's setup so
|
|
127
134
|
# their rollout logs land in the output dir.
|
|
128
135
|
level = "DEBUG" if config.verbose else "INFO"
|
|
@@ -151,48 +158,57 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
151
158
|
address = await asyncio.to_thread(address_queue.get, timeout=600)
|
|
152
159
|
client = EnvClient(address=address)
|
|
153
160
|
await client.wait_for_server_startup(timeout=600)
|
|
154
|
-
|
|
155
|
-
#
|
|
156
|
-
#
|
|
157
|
-
|
|
158
|
-
if
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
f"{config.env_id} is infinite - bound the run with -n/--num-tasks"
|
|
162
|
-
)
|
|
163
|
-
if config.shuffle:
|
|
164
|
-
logger.warning(
|
|
165
|
-
"shuffle is a no-op on an infinite taskset - "
|
|
166
|
-
"taking the first %d generated tasks",
|
|
167
|
-
config.num_tasks,
|
|
168
|
-
)
|
|
169
|
-
idxs = list(range(config.num_tasks))
|
|
170
|
-
else:
|
|
161
|
+
# A v1 run dispatches — and resumes — tasks by content: the client owns them,
|
|
162
|
+
# and `resume.task_key` is their identity. Only the legacy bridge is addressed
|
|
163
|
+
# by dataset row (its dataset lives server-side, reported via `info`), and
|
|
164
|
+
# only a legacy env group-scores; a v1 env scores siblings in its own rollout.
|
|
165
|
+
if legacy:
|
|
166
|
+
info = await client.info()
|
|
167
|
+
group_scored = info.requires_group_scoring
|
|
171
168
|
idxs = sample(list(range(info.num_tasks)), config.shuffle, config.num_tasks)
|
|
169
|
+
plan = [({"task_idx": idx}, config.num_rollouts) for idx in idxs]
|
|
170
|
+
else:
|
|
171
|
+
group_scored = False
|
|
172
|
+
plan = [
|
|
173
|
+
({"task_data": task.data.model_dump(mode="json")}, config.num_rollouts)
|
|
174
|
+
for task in tasks
|
|
175
|
+
]
|
|
172
176
|
out = output_path(config)
|
|
173
177
|
finished: list[Episode] = []
|
|
174
178
|
if config.resume is not None:
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
179
|
+
if legacy:
|
|
180
|
+
# A group is served and scored together, so a partially-kept task
|
|
181
|
+
# redoes as a whole group — whole_task drops its kept rows.
|
|
182
|
+
finished, owed = resume.load(
|
|
183
|
+
out,
|
|
184
|
+
idxs,
|
|
185
|
+
config.num_rollouts,
|
|
186
|
+
whole_task=group_scored,
|
|
187
|
+
key_of=lambda data: data.get("idx"),
|
|
188
|
+
)
|
|
189
|
+
counts = resume.distribute(idxs, owed, config.num_rollouts)
|
|
190
|
+
else:
|
|
191
|
+
keys = [
|
|
192
|
+
resume.task_key(t.data.model_dump(mode="json", exclude_none=True))
|
|
193
|
+
for t in tasks
|
|
194
|
+
]
|
|
195
|
+
finished, owed = resume.load(out, keys, config.num_rollouts)
|
|
196
|
+
counts = resume.distribute(keys, owed, config.num_rollouts)
|
|
180
197
|
if not owed: # already complete - report it and exit successfully
|
|
181
|
-
print(resume.nothing_to_resume_msg(out, len(
|
|
198
|
+
print(resume.nothing_to_resume_msg(out, len(plan), config.num_rollouts))
|
|
182
199
|
raise SystemExit(0)
|
|
183
|
-
|
|
200
|
+
plan = [(payload, n) for (payload, _), n in zip(plan, counts) if n]
|
|
184
201
|
logger.info(
|
|
185
202
|
"resuming %s: %d task(s), %d rollout(s) owed",
|
|
186
203
|
out,
|
|
187
|
-
len(
|
|
204
|
+
len(plan),
|
|
188
205
|
sum(owed.values()),
|
|
189
206
|
)
|
|
190
207
|
else:
|
|
191
|
-
owed = {idx: config.num_rollouts for idx in idxs}
|
|
192
208
|
save_config(config, out)
|
|
193
209
|
logger.info(
|
|
194
210
|
"running %dx%d rollouts via the env-server %s pool on %s",
|
|
195
|
-
len(
|
|
211
|
+
len(plan),
|
|
196
212
|
config.num_rollouts,
|
|
197
213
|
config.pool.type,
|
|
198
214
|
config.model,
|
|
@@ -224,13 +240,13 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
224
240
|
records.append(Episode.of(trace))
|
|
225
241
|
return records
|
|
226
242
|
|
|
227
|
-
async def run_unit(
|
|
243
|
+
async def run_unit(payload: dict) -> list[Episode]:
|
|
228
244
|
async with semaphore or contextlib.nullcontext():
|
|
229
245
|
episode = await client.run(
|
|
230
|
-
task_idx=idx,
|
|
231
246
|
client=config.client,
|
|
232
247
|
model=config.model,
|
|
233
248
|
sampling=config.sampling,
|
|
249
|
+
**payload,
|
|
234
250
|
)
|
|
235
251
|
for trace in episode.traces:
|
|
236
252
|
trace.stamp(EvalRunInfo(id=config.uuid))
|
|
@@ -238,12 +254,12 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
238
254
|
return [episode]
|
|
239
255
|
|
|
240
256
|
# A group-scored legacy task runs its rollouts together (one `run_group`
|
|
241
|
-
# request, one worker); otherwise each rollout is its own `
|
|
242
|
-
#
|
|
257
|
+
# request, one worker); otherwise each rollout is its own `run` request,
|
|
258
|
+
# dispatched least-busy across workers.
|
|
243
259
|
units = (
|
|
244
|
-
[run_group_unit(
|
|
260
|
+
[run_group_unit(payload["task_idx"]) for payload, _ in plan]
|
|
245
261
|
if group_scored
|
|
246
|
-
else [run_unit(
|
|
262
|
+
else [run_unit(payload) for payload, n in plan for _ in range(n)]
|
|
247
263
|
)
|
|
248
264
|
results = await asyncio.gather(*units)
|
|
249
265
|
await client.close()
|
verifiers/v1/cli/gepa.py
CHANGED
|
@@ -66,7 +66,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
66
66
|
# Refuse multi-agent before the dry-run return, so --dry-run can't write a
|
|
67
67
|
# config the real invocation would reject.
|
|
68
68
|
env_cls = vf.environment_class(
|
|
69
|
-
config.env.taskset.id
|
|
69
|
+
config.env.taskset.id,
|
|
70
70
|
config.env.id,
|
|
71
71
|
)
|
|
72
72
|
if not issubclass(env_cls, vf.SingleAgentEnv):
|
verifiers/v1/cli/output.py
CHANGED
|
@@ -36,8 +36,8 @@ def output_path(config: EvalConfig) -> Path:
|
|
|
36
36
|
if config.output_dir is not None:
|
|
37
37
|
return config.output_dir
|
|
38
38
|
taskset = config.env.taskset
|
|
39
|
-
env = taskset.name if taskset
|
|
40
|
-
if taskset
|
|
39
|
+
env = taskset.name if taskset.id else "no-taskset"
|
|
40
|
+
if taskset.id and config.env.id:
|
|
41
41
|
# Same compounding as `EnvConfig.env_id`: a `best-of-n+gsm8k-v1` run must
|
|
42
42
|
# not share a parent dir with a plain `gsm8k-v1` one.
|
|
43
43
|
env = f"{env_name(config.env.id)}+{env}"
|
verifiers/v1/configs/env.py
CHANGED
|
@@ -138,9 +138,7 @@ class EnvServerConfig(BaseConfig):
|
|
|
138
138
|
@property
|
|
139
139
|
def is_legacy(self) -> bool:
|
|
140
140
|
"""A v0/legacy env (run via the bridge): a legacy `id` is set and no v1 taskset."""
|
|
141
|
-
return self.id is not None and
|
|
142
|
-
self.env.taskset is None or not self.env.taskset.id
|
|
143
|
-
)
|
|
141
|
+
return self.id is not None and not self.env.taskset.id
|
|
144
142
|
|
|
145
143
|
@property
|
|
146
144
|
def env_id(self) -> str:
|
|
@@ -152,7 +150,7 @@ class EnvServerConfig(BaseConfig):
|
|
|
152
150
|
def _refuse_legacy_id_with_taskset(self):
|
|
153
151
|
"""A legacy `id` next to a v1 `env.taskset` would be silently inert
|
|
154
152
|
(`is_legacy` is False and the v0 env never loads); refuse the mix."""
|
|
155
|
-
if self.id is not None and self.env.taskset
|
|
153
|
+
if self.id is not None and self.env.taskset.id:
|
|
156
154
|
raise ValueError(
|
|
157
155
|
f"--id {self.id!r} is the legacy (v0) env id and can't combine with "
|
|
158
156
|
f"the v1 taskset {self.env.taskset.id!r}. Pairing an env with a "
|
verifiers/v1/env.py
CHANGED
|
@@ -31,7 +31,7 @@ from verifiers.v1.retries import RetryConfig, run_episode_with_retry
|
|
|
31
31
|
from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
|
|
32
32
|
from verifiers.v1.errors import EnvError, boundary
|
|
33
33
|
from verifiers.v1.task import Task, resolve_server_config
|
|
34
|
-
from verifiers.v1.taskset import
|
|
34
|
+
from verifiers.v1.taskset import TasksetConfig
|
|
35
35
|
from verifiers.v1.episode import Episode
|
|
36
36
|
from verifiers.v1.trace import Error, Trace
|
|
37
37
|
from verifiers.v1.utils.generic import deep_merge, generic_type
|
|
@@ -65,9 +65,9 @@ class EnvConfig(BaseConfig):
|
|
|
65
65
|
"""Which `Env` runs. Empty = the taskset's own, else `SingleAgentEnv`; set
|
|
66
66
|
to pair a reusable env with any taskset (an explicit id wins over the bundled)."""
|
|
67
67
|
# SerializeAsAny: the env-server wire needs the resolved subclass's fields.
|
|
68
|
-
taskset: SerializeAsAny[TasksetConfig]
|
|
68
|
+
taskset: SerializeAsAny[TasksetConfig] = TasksetConfig()
|
|
69
69
|
"""The seed taskset — the rows every rollout starts from (`--env.taskset.id`).
|
|
70
|
-
|
|
70
|
+
The id stays empty only for a legacy (v0) run, which sets the top-level `id`."""
|
|
71
71
|
timeout: TimeoutConfig = TimeoutConfig()
|
|
72
72
|
retries: RetryConfig = RetryConfig()
|
|
73
73
|
"""Whole-EPISODE retries — the coarse fallback for faults no agent owns; a
|
|
@@ -81,17 +81,14 @@ class EnvConfig(BaseConfig):
|
|
|
81
81
|
@property
|
|
82
82
|
def env_id(self) -> str:
|
|
83
83
|
"""The taskset id, prefixed by the paired env id (`best-of-n+gsm8k-v1`)."""
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
return taskset_id or self.id
|
|
84
|
+
if self.taskset.id and self.id:
|
|
85
|
+
return f"{self.id}+{self.taskset.id}"
|
|
86
|
+
return self.taskset.id or self.id
|
|
88
87
|
|
|
89
88
|
def agent_harnesses(self) -> dict[str, HarnessConfig]:
|
|
90
89
|
"""Each declared role's resolved harness config (pin, else the taskset's
|
|
91
90
|
default) — known without constructing the env."""
|
|
92
|
-
default = default_agent_harness(
|
|
93
|
-
self.taskset.id if self.taskset is not None else ""
|
|
94
|
-
)
|
|
91
|
+
default = default_agent_harness(self.taskset.id)
|
|
95
92
|
return {
|
|
96
93
|
name: cfg.harness if cfg.harness is not None else default
|
|
97
94
|
for name, cfg in _declared_agent_configs(self).items()
|
|
@@ -239,7 +236,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
239
236
|
f"{config_cls.__name__}(...) explicitly"
|
|
240
237
|
)
|
|
241
238
|
self.config: ConfigT = config
|
|
242
|
-
if config.taskset
|
|
239
|
+
if not config.taskset.id:
|
|
243
240
|
raise ValueError(
|
|
244
241
|
f"{type(self).__name__} needs a seed taskset — every rollout starts "
|
|
245
242
|
"from one of its tasks: set --env.taskset.id (or the positional "
|
|
@@ -247,7 +244,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
247
244
|
)
|
|
248
245
|
self.taskset = load_taskset(config.taskset)
|
|
249
246
|
self._default_harness = default_agent_harness(config.taskset.id)
|
|
250
|
-
task_cls =
|
|
247
|
+
task_cls = type(self.taskset).task_type()
|
|
251
248
|
self._task_cls: type[Task] = task_cls
|
|
252
249
|
self._agent_specs: dict[str, AgentConfig] = _declared_agent_configs(self.config)
|
|
253
250
|
if not self._agent_specs:
|
|
@@ -520,7 +517,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
520
517
|
"""`requires_tunnel` over the consumers known before any rollout: role
|
|
521
518
|
runtimes, live `shared` servers, and the task class's tool/user servers;
|
|
522
519
|
a class overriding `server_config` conservatively counts as remote."""
|
|
523
|
-
task_cls =
|
|
520
|
+
task_cls = type(self.taskset).task_type()
|
|
524
521
|
server_classes = [*task_cls.tools, *([task_cls.user] if task_cls.user else [])]
|
|
525
522
|
if server_classes and task_cls.server_config is not Task.server_config:
|
|
526
523
|
return True
|
verifiers/v1/harness.py
CHANGED
|
@@ -4,6 +4,7 @@ import logging
|
|
|
4
4
|
import os
|
|
5
5
|
from abc import ABC, abstractmethod
|
|
6
6
|
from collections.abc import Mapping
|
|
7
|
+
from pathlib import Path
|
|
7
8
|
from typing import TYPE_CHECKING, ClassVar, Generic, TypeVar
|
|
8
9
|
|
|
9
10
|
from pydantic import Field
|
|
@@ -41,6 +42,10 @@ class HarnessConfig(BaseConfig):
|
|
|
41
42
|
forward_env: list[str] = Field(default_factory=list)
|
|
42
43
|
"""Host variables to forward without writing secrets into config; explicit `env` wins."""
|
|
43
44
|
disabled_tools: list[str] | None = None
|
|
45
|
+
skills: list[Path] = Field(default_factory=list)
|
|
46
|
+
"""Skill folders to upload into the program's skill discovery directory — each
|
|
47
|
+
lands at `<skills dir>/<folder name>`. Only harnesses whose program discovers
|
|
48
|
+
skills natively (`SUPPORTS_SKILLS`) accept them."""
|
|
44
49
|
|
|
45
50
|
@property
|
|
46
51
|
def name(self) -> str:
|
|
@@ -66,6 +71,15 @@ class Harness(ABC, Generic[ConfigT]):
|
|
|
66
71
|
every real harness; the tool-less chat loops (`null`) override to False. Read
|
|
67
72
|
where model-directed execution changes the rules: the subprocess-on-host
|
|
68
73
|
warning, the judge env's sandbox requirement."""
|
|
74
|
+
SUPPORTS_SKILLS: ClassVar[bool] = False
|
|
75
|
+
"""Whether the program discovers SKILL.md skills — its `setup` calls
|
|
76
|
+
`install_skills` with the program's fixed discovery location; configuring
|
|
77
|
+
`skills` on a harness without support is rejected up front."""
|
|
78
|
+
NEEDS_CONTAINER: ClassVar[bool] = True
|
|
79
|
+
"""Whether the program must run in a container runtime: True for every harness
|
|
80
|
+
that installs and drives a third-party program — on the host (subprocess) it
|
|
81
|
+
leaks host state (auth, config, processes) both ways. Only the minimal
|
|
82
|
+
in-house loops (`bash`, `null`) override to False."""
|
|
69
83
|
|
|
70
84
|
def __init__(self, config: ConfigT) -> None:
|
|
71
85
|
self.config = config
|
|
@@ -102,6 +116,28 @@ class Harness(ABC, Generic[ConfigT]):
|
|
|
102
116
|
async def setup(self, runtime: Runtime) -> None:
|
|
103
117
|
"""Provision this harness in `runtime` before its execution timeout starts."""
|
|
104
118
|
|
|
119
|
+
async def install_skills(self, runtime: Runtime, dest: str) -> None:
|
|
120
|
+
"""Upload each `config.skills` folder into `runtime` at `dest/<folder name>` —
|
|
121
|
+
the program's fixed skill discovery location, which a supporting harness's
|
|
122
|
+
`setup` passes."""
|
|
123
|
+
for skill in self.config.skills:
|
|
124
|
+
# Resolve so `.`/`..` entries get their real folder name (and can't
|
|
125
|
+
# place files outside `dest`).
|
|
126
|
+
skill = skill.resolve()
|
|
127
|
+
if not skill.is_dir():
|
|
128
|
+
raise ValueError(f"skill {str(skill)!r} is not a folder")
|
|
129
|
+
executables = []
|
|
130
|
+
for file in sorted(skill.rglob("*")):
|
|
131
|
+
if not file.is_file():
|
|
132
|
+
continue
|
|
133
|
+
target = f"{dest}/{skill.name}/{file.relative_to(skill).as_posix()}"
|
|
134
|
+
await runtime.write(target, file.read_bytes())
|
|
135
|
+
if os.access(file, os.X_OK):
|
|
136
|
+
executables.append(target)
|
|
137
|
+
if executables:
|
|
138
|
+
# `write` moves bytes, not modes; restore the execute bits scripts need.
|
|
139
|
+
await runtime.run(["chmod", "+x", *executables], {})
|
|
140
|
+
|
|
105
141
|
async def run(
|
|
106
142
|
self,
|
|
107
143
|
ctx: ModelContext,
|
|
@@ -41,6 +41,7 @@ class BashHarness(Harness[BashHarnessConfig]):
|
|
|
41
41
|
SUPPORTS_MCP = True
|
|
42
42
|
SUPPORTS_USER_SIM = True
|
|
43
43
|
SUPPORTS_MESSAGE_PROMPT = True
|
|
44
|
+
NEEDS_CONTAINER = False
|
|
44
45
|
|
|
45
46
|
async def setup(self, runtime: Runtime) -> None:
|
|
46
47
|
await runtime.prepare_uv_script(PROGRAM_SOURCE, self.config.resolved_env)
|
|
@@ -13,6 +13,7 @@ from verifiers.v1.trace import Trace
|
|
|
13
13
|
CLAUDE_HOME = "/tmp/vf-claude-code-{version}"
|
|
14
14
|
CLAUDE_BIN = f"{CLAUDE_HOME}/.local/bin/claude"
|
|
15
15
|
CLAUDE_CONFIG_DIR = ".vf-claude"
|
|
16
|
+
SKILLS_DIR = f"{CLAUDE_CONFIG_DIR}/skills"
|
|
16
17
|
INSTALL = """
|
|
17
18
|
set -e
|
|
18
19
|
command -v curl >/dev/null || (apt-get update -qq && apt-get install -y -qq curl ca-certificates >/dev/null)
|
|
@@ -30,8 +31,10 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
30
31
|
SUPPORTS_MCP = True
|
|
31
32
|
# images would require streaming inputs
|
|
32
33
|
SUPPORTS_MESSAGE_PROMPT = False
|
|
34
|
+
SUPPORTS_SKILLS = True
|
|
33
35
|
|
|
34
36
|
async def setup(self, runtime: Runtime) -> None:
|
|
37
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
35
38
|
home = CLAUDE_HOME.format(version=self.config.version)
|
|
36
39
|
binary = CLAUDE_BIN.format(version=self.config.version)
|
|
37
40
|
script = INSTALL.format(version=self.config.version, home=home)
|
|
@@ -66,13 +69,13 @@ class ClaudeCodeHarness(Harness[ClaudeCodeHarnessConfig]):
|
|
|
66
69
|
"ANTHROPIC_BASE_URL": endpoint.removesuffix("/v1"),
|
|
67
70
|
"ANTHROPIC_API_KEY": secret,
|
|
68
71
|
"CLAUDE_CONFIG_DIR": CLAUDE_CONFIG_DIR,
|
|
72
|
+
"CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1",
|
|
69
73
|
"DISABLE_AUTOUPDATER": "1",
|
|
70
74
|
"IS_SANDBOX": "1",
|
|
71
75
|
}
|
|
72
76
|
argv = [
|
|
73
77
|
CLAUDE_BIN.format(version=self.config.version),
|
|
74
78
|
"--print",
|
|
75
|
-
"--bare",
|
|
76
79
|
"--dangerously-skip-permissions",
|
|
77
80
|
"--no-session-persistence",
|
|
78
81
|
"--model",
|
|
@@ -23,6 +23,7 @@ KEY_VAR = "CODEX_INTERCEPT_KEY"
|
|
|
23
23
|
|
|
24
24
|
CODEX_DIR = "/tmp/vf-codex"
|
|
25
25
|
CODEX_BIN = f"{CODEX_DIR}/bin/codex"
|
|
26
|
+
SKILLS_DIR = ".agents/skills"
|
|
26
27
|
INSTALL = r"""
|
|
27
28
|
set -e
|
|
28
29
|
mkdir -p {dir}/bin
|
|
@@ -46,8 +47,10 @@ class CodexHarness(Harness[CodexHarnessConfig]):
|
|
|
46
47
|
APPENDS_SYSTEM_PROMPT = False # TODO
|
|
47
48
|
SUPPORTS_MCP = False # TODO
|
|
48
49
|
SUPPORTS_MESSAGE_PROMPT = True
|
|
50
|
+
SUPPORTS_SKILLS = True
|
|
49
51
|
|
|
50
52
|
async def setup(self, runtime: Runtime) -> None:
|
|
53
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
51
54
|
logger.info("codex: ensuring codex %s is installed", self.config.version)
|
|
52
55
|
script = (
|
|
53
56
|
INSTALL.replace("{version}", self.config.version)
|
|
@@ -13,6 +13,7 @@ logger = logging.getLogger(__name__)
|
|
|
13
13
|
|
|
14
14
|
BINARY = "/tmp/vf-kimi-code/bin/kimi"
|
|
15
15
|
KIMI_HOME = ".vf-kimi-code"
|
|
16
|
+
SKILLS_DIR = f"{KIMI_HOME}/skills"
|
|
16
17
|
|
|
17
18
|
INSTALL = r"""
|
|
18
19
|
set -e
|
|
@@ -39,8 +40,10 @@ class KimiCodeHarnessConfig(HarnessConfig):
|
|
|
39
40
|
class KimiCodeHarness(Harness[KimiCodeHarnessConfig]):
|
|
40
41
|
APPENDS_SYSTEM_PROMPT = False
|
|
41
42
|
SUPPORTS_MCP = True
|
|
43
|
+
SUPPORTS_SKILLS = True
|
|
42
44
|
|
|
43
45
|
async def setup(self, runtime: Runtime) -> None:
|
|
46
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
44
47
|
logger.info(
|
|
45
48
|
"kimi-code: ensuring Kimi Code %s is installed", self.config.version
|
|
46
49
|
)
|
|
@@ -20,6 +20,7 @@ class NullHarness(Harness[NullHarnessConfig]):
|
|
|
20
20
|
SUPPORTS_USER_SIM = True
|
|
21
21
|
SUPPORTS_MESSAGE_PROMPT = True
|
|
22
22
|
EXECUTES_CODE = False
|
|
23
|
+
NEEDS_CONTAINER = False
|
|
23
24
|
|
|
24
25
|
async def setup(self, runtime: Runtime) -> None:
|
|
25
26
|
await runtime.prepare_uv_script(PROGRAM_SOURCE, self.config.resolved_env)
|
|
@@ -26,6 +26,7 @@ HOME_VAR = "VF_PI_ORIGINAL_HOME"
|
|
|
26
26
|
|
|
27
27
|
PI_DIR = "/tmp/vf-pi"
|
|
28
28
|
PI_BIN = f"{PI_DIR}/pi"
|
|
29
|
+
SKILLS_DIR = ".agents/skills"
|
|
29
30
|
MCP_VERSION = "2.11.0"
|
|
30
31
|
MCP_ADAPTER = f"{PI_DIR}/mcp/node_modules/pi-mcp-adapter/index.ts"
|
|
31
32
|
|
|
@@ -95,8 +96,12 @@ class PiHarness(Harness[PiHarnessConfig]):
|
|
|
95
96
|
APPENDS_SYSTEM_PROMPT = True
|
|
96
97
|
SUPPORTS_MCP = True
|
|
97
98
|
SUPPORTS_MESSAGE_PROMPT = True
|
|
99
|
+
# Pi's project skill discovery is trust-gated (a prompt print mode can't answer),
|
|
100
|
+
# so the installed skills are passed explicitly via `--skill` at launch.
|
|
101
|
+
SUPPORTS_SKILLS = True
|
|
98
102
|
|
|
99
103
|
async def setup(self, runtime: Runtime) -> None:
|
|
104
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
100
105
|
logger.info(
|
|
101
106
|
"pi: ensuring Pi %s and pi-mcp-adapter %s are installed",
|
|
102
107
|
self.config.version,
|
|
@@ -234,6 +239,12 @@ class PiHarness(Harness[PiHarnessConfig]):
|
|
|
234
239
|
if self.config.disabled_tools
|
|
235
240
|
else []
|
|
236
241
|
)
|
|
242
|
+
skill_args = [
|
|
243
|
+
arg
|
|
244
|
+
for skill in self.config.skills
|
|
245
|
+
# Resolve like `install_skills` so the path matches what it wrote.
|
|
246
|
+
for arg in ("--skill", f"{SKILLS_DIR}/{skill.resolve().name}")
|
|
247
|
+
]
|
|
237
248
|
system_args = ["--append-system-prompt", system_prompt] if system_prompt else []
|
|
238
249
|
argv = [
|
|
239
250
|
"sh",
|
|
@@ -252,6 +263,7 @@ class PiHarness(Harness[PiHarnessConfig]):
|
|
|
252
263
|
ctx.model,
|
|
253
264
|
*mcp_args,
|
|
254
265
|
*tool_args,
|
|
266
|
+
*skill_args,
|
|
255
267
|
*system_args,
|
|
256
268
|
*image_args,
|
|
257
269
|
]
|
|
@@ -13,6 +13,7 @@ from verifiers.v1.types import SystemMessage, TextContentPart, UserMessage
|
|
|
13
13
|
|
|
14
14
|
POOL_DIR = "/tmp/vf-pool-{version}"
|
|
15
15
|
SETTINGS_PATH = ".poolside/settings.local.yaml"
|
|
16
|
+
SKILLS_DIR = ".poolside/skills"
|
|
16
17
|
INSTALL = r"""
|
|
17
18
|
set -e
|
|
18
19
|
command -v curl >/dev/null || (apt-get update -qq && apt-get install -y -qq curl ca-certificates >/dev/null)
|
|
@@ -35,8 +36,10 @@ class PoolHarness(Harness[PoolHarnessConfig]):
|
|
|
35
36
|
APPENDS_SYSTEM_PROMPT = True
|
|
36
37
|
SUPPORTS_MCP = True
|
|
37
38
|
SUPPORTS_MESSAGE_PROMPT = True
|
|
39
|
+
SUPPORTS_SKILLS = True
|
|
38
40
|
|
|
39
41
|
async def setup(self, runtime: Runtime) -> None:
|
|
42
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
40
43
|
directory = POOL_DIR.format(version=self.config.version)
|
|
41
44
|
binary = f"{directory}/pool"
|
|
42
45
|
script = INSTALL.replace("{version}", self.config.version).replace(
|
|
@@ -24,6 +24,7 @@ RLM_REPO = "github.com/PrimeIntellect-ai/rlm.git"
|
|
|
24
24
|
RLM_HOME = ".rlm"
|
|
25
25
|
RLM_DIR = "/tmp/vf-rlm"
|
|
26
26
|
RLM_BIN = f"{RLM_DIR}/bin/rlm"
|
|
27
|
+
SKILLS_DIR = "/task/rlm-skills"
|
|
27
28
|
|
|
28
29
|
|
|
29
30
|
class RLMHarnessConfig(HarnessConfig):
|
|
@@ -31,9 +32,9 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
31
32
|
"""Git ref (branch, tag, or commit) of rlm to install."""
|
|
32
33
|
max_depth: int = 0
|
|
33
34
|
"""Recursion depth rlm may spawn sub-harnesses to (RLM_MAX_DEPTH)."""
|
|
34
|
-
|
|
35
|
+
builtin_skills: list[BuiltinSkill] = []
|
|
35
36
|
"""Built-in rlm skills to enable (RLM_SKILLS), e.g. `["edit"]`; empty enables none.
|
|
36
|
-
The tool set is fixed (ipython);
|
|
37
|
+
The tool set is fixed (ipython); the base `skills` field takes SKILL.md paths."""
|
|
37
38
|
summarize_at_tokens: int | tuple[int, int] | None = None
|
|
38
39
|
"""Auto-compaction threshold (RLM_SUMMARIZE_AT_TOKENS): compact the context once it grows
|
|
39
40
|
past this many tokens. An int is a fixed threshold; a `(lo, hi)` pair draws a per-group
|
|
@@ -63,7 +64,7 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
63
64
|
if self.disabled_tools:
|
|
64
65
|
raise ValueError(
|
|
65
66
|
"the rlm harness has a fixed tool set (ipython) and does not support "
|
|
66
|
-
"`disabled_tools`; use `
|
|
67
|
+
"`disabled_tools`; use `builtin_skills` to enable built-in skills instead."
|
|
67
68
|
)
|
|
68
69
|
return self
|
|
69
70
|
|
|
@@ -71,8 +72,11 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
71
72
|
class RLMHarness(Harness[RLMHarnessConfig]):
|
|
72
73
|
APPENDS_SYSTEM_PROMPT = True
|
|
73
74
|
SUPPORTS_MCP = True
|
|
75
|
+
SUPPORTS_SKILLS = True
|
|
74
76
|
|
|
75
77
|
async def setup(self, runtime: Runtime) -> None:
|
|
78
|
+
# Before the installer: install.sh packages the skills it finds.
|
|
79
|
+
await self.install_skills(runtime, SKILLS_DIR)
|
|
76
80
|
# install.sh fetches curl/uv itself; add git only when the image lacks it.
|
|
77
81
|
install = (
|
|
78
82
|
"command -v git >/dev/null 2>&1 || "
|
|
@@ -123,8 +127,8 @@ class RLMHarness(Harness[RLMHarnessConfig]):
|
|
|
123
127
|
}
|
|
124
128
|
if system_prompt is not None:
|
|
125
129
|
env["RLM_APPEND_TO_SYSTEM_PROMPT"] = system_prompt
|
|
126
|
-
if self.config.
|
|
127
|
-
env["RLM_SKILLS"] = ",".join(self.config.
|
|
130
|
+
if self.config.builtin_skills:
|
|
131
|
+
env["RLM_SKILLS"] = ",".join(self.config.builtin_skills)
|
|
128
132
|
if mcp_urls:
|
|
129
133
|
env["RLM_MCP_CONFIG"] = json.dumps(
|
|
130
134
|
{"mcpServers": {name: {"url": url} for name, url in mcp_urls.items()}}
|
|
@@ -18,18 +18,10 @@ class Terminus2HarnessConfig(HarnessConfig):
|
|
|
18
18
|
class Terminus2Harness(Harness[Terminus2HarnessConfig]):
|
|
19
19
|
APPENDS_SYSTEM_PROMPT = True
|
|
20
20
|
SUPPORTS_MCP = False
|
|
21
|
+
# Beyond the usual host leaks, Terminus drives tmux: on the host its tmux server —
|
|
22
|
+
# and the `tmux kill-server` cleanup in `launch` — would share the user's own.
|
|
21
23
|
|
|
22
24
|
async def setup(self, runtime: Runtime) -> None:
|
|
23
|
-
# TODO: Terminus drives tmux; on the host (subprocess) runtime its tmux server — and the
|
|
24
|
-
# `tmux kill-server` cleanup in `launch` — share the host's tmux, so a host run can kill
|
|
25
|
-
# the user's own tmux session. Until tmux is isolated (a dedicated `tmux -L` socket + a
|
|
26
|
-
# created, private TMUX_TMPDIR), refuse the host runtime; run Terminus 2 in a container.
|
|
27
|
-
if runtime.type == "subprocess":
|
|
28
|
-
raise RuntimeError(
|
|
29
|
-
"Terminus 2 drives tmux and is unsafe on the subprocess (host) runtime — its tmux "
|
|
30
|
-
"cleanup can kill the host's tmux server. Run it in a container runtime "
|
|
31
|
-
"(--env.agent.harness.runtime.type docker|prime|modal)."
|
|
32
|
-
)
|
|
33
25
|
source = PROGRAM_SOURCE.replace("{version}", self.config.version)
|
|
34
26
|
await runtime.prepare_uv_script(source, self.config.resolved_env)
|
|
35
27
|
|
verifiers/v1/legacy.py
CHANGED
|
@@ -431,12 +431,23 @@ class LegacyEnvServer(EnvServer):
|
|
|
431
431
|
state_columns=["trajectory"],
|
|
432
432
|
)
|
|
433
433
|
|
|
434
|
+
@staticmethod
|
|
435
|
+
def _row(req: RunRequest) -> int:
|
|
436
|
+
"""The dataset row a request addresses — the bridge's dataset lives
|
|
437
|
+
server-side, so requests must carry `task_idx` (v1 servers take `task_data`)."""
|
|
438
|
+
if req.task_idx is None:
|
|
439
|
+
raise ValueError(
|
|
440
|
+
"legacy env server requests address the dataset by task_idx"
|
|
441
|
+
)
|
|
442
|
+
return req.task_idx
|
|
443
|
+
|
|
434
444
|
async def _run(self, req: RunRequest) -> RunResponse:
|
|
435
|
-
|
|
445
|
+
task_idx = self._row(req)
|
|
446
|
+
out = await self._run_v0(task_idx, req.client, req.model, req.sampling)
|
|
436
447
|
# Trust the bridge-minted record; serialize it once (mirrors `EnvServer`).
|
|
437
448
|
return RunResponse.model_construct(
|
|
438
449
|
episode=Episode.of(
|
|
439
|
-
rollout_output_to_trace(out,
|
|
450
|
+
rollout_output_to_trace(out, task_idx), env=self.taskset_id
|
|
440
451
|
)
|
|
441
452
|
)
|
|
442
453
|
|
verifiers/v1/loaders.py
CHANGED
|
@@ -187,8 +187,7 @@ def environment_class(taskset_id: str, env_id: str = "") -> type[Env]:
|
|
|
187
187
|
def load_environment(config: EnvConfig) -> Env:
|
|
188
188
|
"""Construct the env for `config`. Every construction site (eval, serve, gepa)
|
|
189
189
|
goes through here so subclass envs load everywhere."""
|
|
190
|
-
|
|
191
|
-
return environment_class(taskset_id, config.id)(config)
|
|
190
|
+
return environment_class(config.taskset.id, config.id)(config)
|
|
192
191
|
|
|
193
192
|
|
|
194
193
|
def load_taskset(config: TasksetConfig) -> Taskset:
|
|
@@ -239,8 +238,7 @@ def resolve_env_config(data: dict | EnvConfig | None) -> EnvConfig:
|
|
|
239
238
|
validate. The one entry every consumer takes (CLI, TOML, the env-server wire),
|
|
240
239
|
so role fields always validate against the real config class."""
|
|
241
240
|
if isinstance(data, EnvConfig):
|
|
242
|
-
|
|
243
|
-
cls = env_config_type(taskset_id, data.id)
|
|
241
|
+
cls = env_config_type(data.taskset.id, data.id)
|
|
244
242
|
if isinstance(data, cls):
|
|
245
243
|
return data # already at least as specifically typed — keep
|
|
246
244
|
data = data.model_dump()
|
|
@@ -260,4 +258,4 @@ def resolve_env_config(data: dict | EnvConfig | None) -> EnvConfig:
|
|
|
260
258
|
def task_type(taskset_id: str) -> type[Task]:
|
|
261
259
|
"""The taskset's `Task` subclass from its generic parameters — no data is
|
|
262
260
|
loaded, so replay can cheaply recover the task type. Falls back to `Task`."""
|
|
263
|
-
return
|
|
261
|
+
return taskset_class(taskset_id).task_type()
|
verifiers/v1/push.py
CHANGED
|
@@ -181,9 +181,7 @@ def push_traces(
|
|
|
181
181
|
return finish(error="no PRIME_API_KEY (run `prime login`)")
|
|
182
182
|
|
|
183
183
|
traces = [trace for episode in episodes for trace in episode.traces]
|
|
184
|
-
env_name = (
|
|
185
|
-
config.env.taskset.id if config.env.taskset is not None else ""
|
|
186
|
-
) or config.id
|
|
184
|
+
env_name = (config.env.taskset.id) or config.id
|
|
187
185
|
metrics = _run_metrics(episodes, traces)
|
|
188
186
|
samples = _build_samples(episodes)
|
|
189
187
|
num_examples = len({t.task.data.idx for t in traces})
|
verifiers/v1/serve/client.py
CHANGED
|
@@ -141,13 +141,25 @@ class EnvClient:
|
|
|
141
141
|
return await self._request(InfoRequest(), InfoResponse)
|
|
142
142
|
|
|
143
143
|
async def run(
|
|
144
|
-
self,
|
|
144
|
+
self,
|
|
145
|
+
client: ClientConfig,
|
|
146
|
+
model: str,
|
|
147
|
+
sampling: SamplingConfig,
|
|
148
|
+
task_data: dict | None = None,
|
|
149
|
+
# TODO: remove task_idx addressing once v0 (the legacy bridge) is deprecated.
|
|
150
|
+
task_idx: int | None = None,
|
|
145
151
|
) -> WireEpisode:
|
|
146
|
-
"""Run one rollout
|
|
147
|
-
|
|
152
|
+
"""Run one rollout; return its episode record — flat traces (typed
|
|
153
|
+
`Trace[WireTaskData]`) plus the shared stamp. A v1 server takes the task
|
|
154
|
+
itself (`task_data`, its dumped `TaskData`); the legacy bridge addresses
|
|
155
|
+
its server-side dataset by `task_idx`."""
|
|
148
156
|
response = await self._request(
|
|
149
157
|
RunRequest(
|
|
150
|
-
|
|
158
|
+
task_data=task_data,
|
|
159
|
+
task_idx=task_idx,
|
|
160
|
+
client=client,
|
|
161
|
+
model=model,
|
|
162
|
+
sampling=sampling,
|
|
151
163
|
),
|
|
152
164
|
RunResponse,
|
|
153
165
|
)
|
verifiers/v1/serve/server.py
CHANGED
|
@@ -22,29 +22,33 @@ from verifiers.v1.serve.types import (
|
|
|
22
22
|
RunRequest,
|
|
23
23
|
RunResponse,
|
|
24
24
|
)
|
|
25
|
+
from verifiers.v1.task import Task, task_data_cls
|
|
25
26
|
from verifiers.v1.types import SamplingConfig
|
|
26
27
|
|
|
27
28
|
logger = logging.getLogger(__name__)
|
|
28
29
|
|
|
29
|
-
MAX_LAZY_TASKS = 1_000_000
|
|
30
|
-
"""Most tasks an infinite taskset's generator is willing to build (and cache) per worker."""
|
|
31
|
-
|
|
32
30
|
|
|
33
31
|
class EnvServer:
|
|
34
32
|
def __init__(
|
|
35
33
|
self, config: EnvConfig, address: str = "tcp://127.0.0.1:5000"
|
|
36
34
|
) -> None:
|
|
37
35
|
self.address = address
|
|
38
|
-
self.taskset_id = config.taskset.id
|
|
36
|
+
self.taskset_id = config.taskset.id
|
|
39
37
|
self.env = load_environment(config)
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
38
|
+
self.task_cls = type(self.env.taskset).task_type()
|
|
39
|
+
self.data_cls = task_data_cls(self.task_cls)
|
|
40
|
+
# A dispatched task is its client-side model_dump(): a field excluded from
|
|
41
|
+
# serialization would vanish on the wire and rebuild silently defaulted, so
|
|
42
|
+
# refuse to serve such a taskset.
|
|
43
|
+
excluded = [
|
|
44
|
+
name for name, field in self.data_cls.model_fields.items() if field.exclude
|
|
45
|
+
]
|
|
46
|
+
if excluded:
|
|
47
|
+
raise ValueError(
|
|
48
|
+
f"{self.data_cls.__name__} excludes {excluded} from serialization — "
|
|
49
|
+
"a served task must survive the wire whole (drop exclude=True)"
|
|
50
|
+
)
|
|
44
51
|
self.num_tasks: int | None = None
|
|
45
|
-
if not type(self.env.taskset).INFINITE:
|
|
46
|
-
self._tasks = list(self._task_iter)
|
|
47
|
-
self.num_tasks = len(self._tasks)
|
|
48
52
|
# v1 envs never group-score (siblings score inside the env's own rollout);
|
|
49
53
|
# only the legacy (v0) bridge sets this.
|
|
50
54
|
self.requires_group_scoring = False
|
|
@@ -69,8 +73,8 @@ class EnvServer:
|
|
|
69
73
|
@classmethod
|
|
70
74
|
def run_server(cls, address_queue=None, **kwargs) -> None:
|
|
71
75
|
"""Run a spawned server and report its concrete address when requested."""
|
|
72
|
-
# Pin tqdm to a threading lock first, so
|
|
73
|
-
# multiprocessing semaphore (resource_tracker warning at shutdown).
|
|
76
|
+
# Pin tqdm to a threading lock first, so a dataset pull (legacy bridge) never
|
|
77
|
+
# leaks a multiprocessing semaphore (resource_tracker warning at shutdown).
|
|
74
78
|
use_threading_tqdm_lock()
|
|
75
79
|
server = cls(**kwargs)
|
|
76
80
|
if address_queue is not None:
|
|
@@ -83,24 +87,18 @@ class EnvServer:
|
|
|
83
87
|
# of a spurious multiprocessing traceback, matching serve_env's own handling.
|
|
84
88
|
pass
|
|
85
89
|
|
|
86
|
-
def
|
|
87
|
-
"""
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
self._tasks.append(next(self._task_iter))
|
|
99
|
-
except StopIteration:
|
|
100
|
-
raise IndexError(
|
|
101
|
-
f"task_idx {idx} out of range ({len(self._tasks)} tasks)"
|
|
102
|
-
) from None
|
|
103
|
-
return self._tasks[idx]
|
|
90
|
+
def _build_task(self, task_data: dict | None) -> Task:
|
|
91
|
+
"""Rebuild a request's task from its wire data: validate into the taskset's
|
|
92
|
+
declared `TaskData` type and wrap it in the declared `Task` with the config's
|
|
93
|
+
task subtree — the same construction the taskset's own `load()` performs. The
|
|
94
|
+
client owns the taskset; this server never `load()`s data, so pool workers
|
|
95
|
+
don't each pull the dataset."""
|
|
96
|
+
if task_data is None:
|
|
97
|
+
raise ValueError(
|
|
98
|
+
"v1 env server requests carry task_data (task_idx addresses the legacy bridge)"
|
|
99
|
+
)
|
|
100
|
+
data = self.data_cls.model_validate(task_data)
|
|
101
|
+
return self.task_cls(data, self.env.config.taskset.task)
|
|
104
102
|
|
|
105
103
|
def _client(self, client_config: ClientConfig, model: str) -> Client:
|
|
106
104
|
"""Cache clients because renderer initialization builds a tokenizer pool."""
|
|
@@ -123,7 +121,7 @@ class EnvServer:
|
|
|
123
121
|
|
|
124
122
|
async def _run(self, req: RunRequest) -> RunResponse:
|
|
125
123
|
ctx = self._context(req.client, req.model, req.sampling)
|
|
126
|
-
(slot,) = self.env.slots(self.
|
|
124
|
+
(slot,) = self.env.slots(self._build_task(req.task_data))
|
|
127
125
|
# The gate spans requests: `--env.max-concurrent` bounds this worker's
|
|
128
126
|
# agent runs the same way the in-process eval's semaphore does.
|
|
129
127
|
episode = await self.env.run_slot(slot, ctx, self._gate)
|
|
@@ -182,7 +180,7 @@ class EnvServer:
|
|
|
182
180
|
"EnvServer up: taskset=%s address=%s tasks=%s group_scoring=%s",
|
|
183
181
|
self.taskset_id,
|
|
184
182
|
self.address,
|
|
185
|
-
self.num_tasks if self.num_tasks is not None else "
|
|
183
|
+
self.num_tasks if self.num_tasks is not None else "client-side",
|
|
186
184
|
self.requires_group_scoring,
|
|
187
185
|
)
|
|
188
186
|
poller = zmq.asyncio.Poller()
|
verifiers/v1/serve/types.py
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
from typing import ClassVar
|
|
2
2
|
|
|
3
|
-
from pydantic import BaseModel, Field, field_serializer
|
|
3
|
+
from pydantic import BaseModel, Field, field_serializer, model_validator
|
|
4
4
|
|
|
5
5
|
from verifiers.v1.clients.config import ClientConfig
|
|
6
6
|
from verifiers.v1.task import WireTaskData
|
|
@@ -34,7 +34,9 @@ class InfoRequest(BaseRequest):
|
|
|
34
34
|
|
|
35
35
|
class InfoResponse(BaseResponse):
|
|
36
36
|
num_tasks: int | None = None
|
|
37
|
-
"""Task count
|
|
37
|
+
"""Task count. Only the legacy bridge (whose dataset lives server-side) reports
|
|
38
|
+
one; a v1 server is stateless — its tasks live on the client — so this stays
|
|
39
|
+
`None`."""
|
|
38
40
|
requires_group_scoring: bool = False
|
|
39
41
|
"""Whether tasks must be run as whole groups — legacy (v0) envs only; a v1
|
|
40
42
|
server always reports False (sibling-dependent signals run inside the env's
|
|
@@ -42,12 +44,25 @@ class InfoResponse(BaseResponse):
|
|
|
42
44
|
|
|
43
45
|
|
|
44
46
|
class RunRequest(BaseRequest):
|
|
47
|
+
"""One env-rollout. v1 ships the task itself (`task_data`, the dumped `TaskData`
|
|
48
|
+
the server validates into the taskset's declared type); the legacy bridge
|
|
49
|
+
addresses its server-side dataset by row (`task_idx`)."""
|
|
50
|
+
|
|
45
51
|
method: ClassVar[str] = "run"
|
|
46
|
-
|
|
52
|
+
task_data: dict | None = None
|
|
53
|
+
task_idx: int | None = Field(None, ge=0)
|
|
47
54
|
client: ClientConfig
|
|
48
55
|
model: str
|
|
49
56
|
sampling: SamplingConfig
|
|
50
57
|
|
|
58
|
+
@model_validator(mode="after")
|
|
59
|
+
def _exactly_one(self) -> "RunRequest":
|
|
60
|
+
if (self.task_data is None) == (self.task_idx is None):
|
|
61
|
+
raise ValueError(
|
|
62
|
+
"exactly one of task_data (v1) or task_idx (legacy) must be set"
|
|
63
|
+
)
|
|
64
|
+
return self
|
|
65
|
+
|
|
51
66
|
|
|
52
67
|
class RunResponse(BaseResponse):
|
|
53
68
|
episode: WireEpisode | None = None
|
verifiers/v1/taskset.py
CHANGED
|
@@ -29,8 +29,9 @@ from pydantic import SerializeAsAny
|
|
|
29
29
|
from pydantic_config import BaseConfig
|
|
30
30
|
from typing_extensions import TypeVar
|
|
31
31
|
|
|
32
|
-
from verifiers.v1.task import TaskConfig, TaskT, resolve_server_config
|
|
32
|
+
from verifiers.v1.task import Task, TaskConfig, TaskT, resolve_server_config
|
|
33
33
|
from verifiers.v1.types import ID
|
|
34
|
+
from verifiers.v1.utils.generic import generic_type
|
|
34
35
|
from verifiers.v1.utils.install import env_name
|
|
35
36
|
from verifiers.v1.utils.sampling import sample
|
|
36
37
|
|
|
@@ -66,6 +67,10 @@ class Taskset(Generic[TaskT, TasksetConfigT]):
|
|
|
66
67
|
def __init__(self, config: TasksetConfigT) -> None:
|
|
67
68
|
self.config = config
|
|
68
69
|
|
|
70
|
+
@classmethod
|
|
71
|
+
def task_type(cls) -> type[Task]:
|
|
72
|
+
return generic_type(cls, Task, origin=Taskset) or Task
|
|
73
|
+
|
|
69
74
|
def load(self) -> Iterable[TaskT]:
|
|
70
75
|
raise NotImplementedError
|
|
71
76
|
|
|
@@ -87,8 +87,8 @@ class HarborData(TaskData):
|
|
|
87
87
|
difficulty: str | None = None
|
|
88
88
|
category: str | None = None
|
|
89
89
|
tags: list[str] = []
|
|
90
|
-
task_dir: str =
|
|
91
|
-
"""Host path to the task dir; used to stage tests/ to verify
|
|
90
|
+
task_dir: str = ""
|
|
91
|
+
"""Host path to the task dir; used to stage tests/ to verify."""
|
|
92
92
|
verifier_env: dict[str, str] = {}
|
|
93
93
|
"""Raw [verifier.env] entries (literals or `${VAR}`/`${VAR:-default}` templates).
|
|
94
94
|
Resolved against the host environment at scoring time, like `harbor run` — so a
|
verifiers/v1/utils/compile.py
CHANGED
|
@@ -95,6 +95,18 @@ def validate_pairing(
|
|
|
95
95
|
f"{task_cls.__name__} exposes tool servers (MCP). Run it with a harness that "
|
|
96
96
|
f"supports MCP (e.g. --env.agent.harness.id bash), or use tasks without tools."
|
|
97
97
|
)
|
|
98
|
+
if not harness.SUPPORTS_SKILLS and harness.config.skills:
|
|
99
|
+
raise ValueError(
|
|
100
|
+
f"Harness {harness.config.id!r} has no native skill support, but "
|
|
101
|
+
"`skills` is set. Run them with a harness whose program discovers "
|
|
102
|
+
"skills (e.g. --env.agent.harness.id claude-code)."
|
|
103
|
+
)
|
|
104
|
+
if harness.NEEDS_CONTAINER and isinstance(runtime_config, SubprocessConfig):
|
|
105
|
+
raise ValueError(
|
|
106
|
+
f"Harness {harness.config.id!r} needs a container runtime "
|
|
107
|
+
"(NEEDS_CONTAINER), but this run resolves to the subprocess runtime; "
|
|
108
|
+
"use --env.agent.harness.runtime.type docker or prime."
|
|
109
|
+
)
|
|
98
110
|
if not harness.SUPPORTS_USER_SIM and task_cls.user is not None:
|
|
99
111
|
raise ValueError(
|
|
100
112
|
f"Harness {harness.config.id!r} does not drive a user simulator, but "
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev18
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -170,42 +170,42 @@ verifiers/utils/version_utils.py,sha256=am3hZLnlaUWTFdllGZR1VMN91hEhiiE8FDmPf1cO
|
|
|
170
170
|
verifiers/v1/__init__.py,sha256=r8khgi7_KSY4AfevEefSslyxytScmMzzCBA6kjqvd1I,6653
|
|
171
171
|
verifiers/v1/agent.py,sha256=x_n4eV4iiUPLl8OWQULrDXeMzLTLA0_GXIYJED2nJyE,21261
|
|
172
172
|
verifiers/v1/decorators.py,sha256=XRMkUQSyvXCYP5fOwzBYV5qEOlxLg6rzYhnhrHHHTIQ,3838
|
|
173
|
-
verifiers/v1/env.py,sha256=
|
|
173
|
+
verifiers/v1/env.py,sha256=79MMfFhQPJnLk5O96WqU_EuaVWMflz-XgiEszHvilCI,24565
|
|
174
174
|
verifiers/v1/episode.py,sha256=a-8IwYecfaBAD0NPbPREQWUybx-KrgLev2JHC-shjgU,2301
|
|
175
175
|
verifiers/v1/errors.py,sha256=pIn9sR12bcd-fBJhMsZjiZCWjH3Uf9Vwhfu1LakFhMY,7007
|
|
176
176
|
verifiers/v1/graph.py,sha256=Mivq5jICUcyK-fhomFTwP4fQn688MXg6-hmdf03aC4Y,27703
|
|
177
|
-
verifiers/v1/harness.py,sha256=
|
|
177
|
+
verifiers/v1/harness.py,sha256=2BOeYLwREgpbZbCshpGkbxHtKEFV5izLO8eT8Ay3YE0,8603
|
|
178
178
|
verifiers/v1/judge.py,sha256=u15e7YXPUKi-DWhqTrwsRy0M7UTj7O0066Rwm-RAXmU,11668
|
|
179
|
-
verifiers/v1/legacy.py,sha256=
|
|
180
|
-
verifiers/v1/loaders.py,sha256=
|
|
181
|
-
verifiers/v1/push.py,sha256=
|
|
179
|
+
verifiers/v1/legacy.py,sha256=b53EDDOImu8rSaZR_ZE7jNaQTJ6aIw5T8hxlXU4fTh0,22280
|
|
180
|
+
verifiers/v1/loaders.py,sha256=ISI0UKk8UXUOJke8n2M3F0EpPmqqnTgey0QDq9Ryo8I,9808
|
|
181
|
+
verifiers/v1/push.py,sha256=_wRGO6UTQURSBFyPQzXSdkZGQEGZh6PyVzKNJgkyJVM,9329
|
|
182
182
|
verifiers/v1/retries.py,sha256=5-LQZ2A7Tr1dpTtoJ-bX0_BrzecHm9Drf3EfQUIjVAw,6010
|
|
183
183
|
verifiers/v1/rollout.py,sha256=cnVa7I98IEbIj-UmlRIe_SFZrpjJNVFnuOsQP4g4ask,16807
|
|
184
184
|
verifiers/v1/scoring.py,sha256=I_mtqhTZ195hj2-CB5jzr3BY5GFKgNEvTf331f7Is4k,5740
|
|
185
185
|
verifiers/v1/session.py,sha256=CyvKR1oN_UCP3SOm2vhGwn_wWlwB68GklJeFhwX5Bg8,7780
|
|
186
186
|
verifiers/v1/state.py,sha256=hYo7DkOMNyh5pcKz7Xr7j-qQN2-LOKKDTlD-KdKmW40,698
|
|
187
187
|
verifiers/v1/task.py,sha256=zzZsfDiulMPKorVrJHFCTYEQKagdVtAg_OlP-P0YRgk,11782
|
|
188
|
-
verifiers/v1/taskset.py,sha256=
|
|
188
|
+
verifiers/v1/taskset.py,sha256=0euaEA_Bm5hFdXifdmV3fi7p0yoHofvx7w-UOExxhBo,4386
|
|
189
189
|
verifiers/v1/trace.py,sha256=CnVt4lOfyzEOoXIv_XSQqhYwxY77Tu_O__FLNVFTZxs,25146
|
|
190
190
|
verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
|
|
191
191
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
192
192
|
verifiers/v1/cli/debug.py,sha256=e-imOXOK1WkAkqEZqiGKsnsohDnFgm-U63sFQn_Htrg,10954
|
|
193
|
-
verifiers/v1/cli/gepa.py,sha256=
|
|
193
|
+
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
194
194
|
verifiers/v1/cli/init.py,sha256=9rG0yjhZpKft1qhno6BI8GmpOzPlQ3r0ZBnujsevR_c,9785
|
|
195
|
-
verifiers/v1/cli/output.py,sha256=
|
|
195
|
+
verifiers/v1/cli/output.py,sha256=878YmRnx3aMCbd6LQoTTLU88k5EC53sX9iJJgOj4jzo,5886
|
|
196
196
|
verifiers/v1/cli/replay.py,sha256=d5BBrTeaun8B9BXtkOu8VCL7gnnL6tFG_zNetqsoIuM,10224
|
|
197
197
|
verifiers/v1/cli/resolve.py,sha256=1sV4z04v_HEs86Oi8ODJqfujXjrbShbu3u2kDpgpj94,4079
|
|
198
198
|
verifiers/v1/cli/serve.py,sha256=6rGujVxcMBso-K3fTM8EdURzw2LHshyekJfbFY5KU7A,2659
|
|
199
199
|
verifiers/v1/cli/validate.py,sha256=KvOGH6cacU8nT2hX_J230VLw2q-0HEPOdIu1Wsgr8TU,9543
|
|
200
200
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
201
201
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
202
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
202
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=wQIMksQoSxqlCqXp57exRaOap-kepHzA0_6inQiSFkY,33656
|
|
203
203
|
verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
|
|
204
204
|
verifiers/v1/cli/dashboard/validate.py,sha256=7olY6oq4-zY8tvrKTJQTBwqG5ZfO5FBJ6ZHHIlKoHxo,3646
|
|
205
205
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
206
206
|
verifiers/v1/cli/eval/main.py,sha256=a1WzbkQP21pQJWI-AdgLCj6YB1bg4pDzVZF7OWT3WXM,5477
|
|
207
|
-
verifiers/v1/cli/eval/resume.py,sha256=
|
|
208
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
207
|
+
verifiers/v1/cli/eval/resume.py,sha256=can5G76NpUdEaBZF4-lClFB09XRZ82iRoJ-7oKxrcOs,7093
|
|
208
|
+
verifiers/v1/cli/eval/runner.py,sha256=IE5tiW3qjp6jtDbd92S6xMEW4UI_PRANN7oqJipuGr0,11204
|
|
209
209
|
verifiers/v1/clients/__init__.py,sha256=wl1Qdx6lrPXnxvHaEip9NaUAad8i_1H6lwyWjGCCUbQ,511
|
|
210
210
|
verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
|
|
211
211
|
verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
|
|
@@ -213,7 +213,7 @@ verifiers/v1/clients/eval.py,sha256=7eyc3xB489E_SWH10h42NybL3MB22W5p9Tttza__P9Q,
|
|
|
213
213
|
verifiers/v1/clients/train.py,sha256=3y2BtL7aYeSaiknJzKh-wLHsfi54nA4pt3BBWQ3ROMs,13043
|
|
214
214
|
verifiers/v1/configs/__init__.py,sha256=PlbPK0iE3vw7dqqQ-3JMdzAGiNh6sXe4p9RPhQJKRR8,345
|
|
215
215
|
verifiers/v1/configs/debug.py,sha256=Z-QVPheQLu_cph_ZrCqSZI0ufkF4n0DXfvpaFTNWbHY,2831
|
|
216
|
-
verifiers/v1/configs/env.py,sha256=
|
|
216
|
+
verifiers/v1/configs/env.py,sha256=GfV1_T26Fjh-XgomiGVeWmAIyIe-FjIpQ0uYKJGhBN0,7280
|
|
217
217
|
verifiers/v1/configs/eval.py,sha256=nC5CmBmN3_ZoXK8QGoEiMecBRfCJxs-Qs-Nc8qZGJ_I,3808
|
|
218
218
|
verifiers/v1/configs/init.py,sha256=u2Y0lKSdL9rVs2HW8qq3gIYfGNT8x8c9k355lt3Yq9A,1166
|
|
219
219
|
verifiers/v1/configs/replay.py,sha256=ApH1FvuDeju-E2EHMmTsmqXu844bEPEHQbZQgAb76MM,2899
|
|
@@ -239,28 +239,28 @@ verifiers/v1/gepa/reflection.py,sha256=3wiMaXphu40i3nt38UhwNP06dMihqoVvaaBDO-52I
|
|
|
239
239
|
verifiers/v1/gepa/runner.py,sha256=XYsXSvImrwkv6TuNtXBOEfUXfqRupHSj-VKnVYv7_3k,5374
|
|
240
240
|
verifiers/v1/harnesses/__init__.py,sha256=BUIgBrFU9v8sqgPl5h6YtKHUYIpA-4xih9YhHV0zVRQ,1301
|
|
241
241
|
verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
|
|
242
|
-
verifiers/v1/harnesses/bash/harness.py,sha256=
|
|
242
|
+
verifiers/v1/harnesses/bash/harness.py,sha256=4ZWxx6dFlZEH4nLgb6NqfpSOn5LxFZsJ4U5J77clM_o,5229
|
|
243
243
|
verifiers/v1/harnesses/bash/program.py,sha256=fu14BwvVEL3VJddpBC8zpZoapCWa83acAXR5XIe6SUs,15029
|
|
244
244
|
verifiers/v1/harnesses/claude_code/__init__.py,sha256=JkZQylMSA3o1fvBx_NM41y_umuahmOSqFjTrJ2hryTw,171
|
|
245
|
-
verifiers/v1/harnesses/claude_code/harness.py,sha256=
|
|
245
|
+
verifiers/v1/harnesses/claude_code/harness.py,sha256=d5fZ-pyP6RdJ7AHmyUa8B4tAlwm_427i0NFAOtsyLic,3926
|
|
246
246
|
verifiers/v1/harnesses/codex/__init__.py,sha256=ocyBvlpSO8c3pT6D1ebrQohoXYESQcCyPT5kN3j1-N4,132
|
|
247
|
-
verifiers/v1/harnesses/codex/harness.py,sha256=
|
|
247
|
+
verifiers/v1/harnesses/codex/harness.py,sha256=s3LA7beFu_Skp_lJsRPj2qW9VwX0uRc0sCcQcDgsfU4,7374
|
|
248
248
|
verifiers/v1/harnesses/kimi_code/__init__.py,sha256=1BEotSfuzmVcE7UrxHeAREM5d7J20cGkonHuowlrzu4,161
|
|
249
|
-
verifiers/v1/harnesses/kimi_code/harness.py,sha256=
|
|
249
|
+
verifiers/v1/harnesses/kimi_code/harness.py,sha256=G6QLWBH3JdNWD_UaiT5zEmT6hXV6STmPVAj7Ir2QTlU,3644
|
|
250
250
|
verifiers/v1/harnesses/mini_swe_agent/__init__.py,sha256=1OBu2Y-XqL72-dGhFdgB6sr5uX_peEEXT9vEjPy1ovU,182
|
|
251
251
|
verifiers/v1/harnesses/mini_swe_agent/harness.py,sha256=wypTAvzix-7B7jMNg3JKCf2ytKmAm6_jNa5p3yGO9aA,2351
|
|
252
252
|
verifiers/v1/harnesses/mini_swe_agent/program.py,sha256=XPjDEsbJACB34tkUX9dJdIziWyKnMTvAJMLevI_FFxU,159
|
|
253
253
|
verifiers/v1/harnesses/null/__init__.py,sha256=XDeKOPeoXUQi9US9_Z3_yH9YaZ4EiOMKJmUtNNdtD7s,127
|
|
254
|
-
verifiers/v1/harnesses/null/harness.py,sha256=
|
|
254
|
+
verifiers/v1/harnesses/null/harness.py,sha256=AmfJkzHajfK3py8O5hbGT5lmsyJIMrA85NoTilCfvIk,2476
|
|
255
255
|
verifiers/v1/harnesses/null/program.py,sha256=qEkzf45Oh-PlotpJxCD2nrWDf1Ocf_-b58vTerxrNzo,8357
|
|
256
256
|
verifiers/v1/harnesses/pi/__init__.py,sha256=1mSnzOf-RdEchdeg6IHtcdobG5KIpAhuTdUIgsCIn74,117
|
|
257
|
-
verifiers/v1/harnesses/pi/harness.py,sha256=
|
|
257
|
+
verifiers/v1/harnesses/pi/harness.py,sha256=YjXX-RfYmzERNfC4ju9oQHYpfZWJnq2M5pAM6DCZLcM,10070
|
|
258
258
|
verifiers/v1/harnesses/pool/__init__.py,sha256=HTYsNiWGdiNEsSHj4xhPmshJSNb7SHjQ7d4yZQ4p3HM,127
|
|
259
|
-
verifiers/v1/harnesses/pool/harness.py,sha256=
|
|
259
|
+
verifiers/v1/harnesses/pool/harness.py,sha256=Dui9Ov2r0qYEGAgs1lfJ-ExfLngAl3iKbmzghKmGLoo,5494
|
|
260
260
|
verifiers/v1/harnesses/rlm/__init__.py,sha256=hwx51xTdEWzzYgMj-p4ETNcptoub3IPMDbEdJ4uB8Ns,122
|
|
261
|
-
verifiers/v1/harnesses/rlm/harness.py,sha256=
|
|
261
|
+
verifiers/v1/harnesses/rlm/harness.py,sha256=83xOPXmWAQwm-4xvTyEymWvxswfLyOLf-SDQIc_poGg,6849
|
|
262
262
|
verifiers/v1/harnesses/terminus_2/__init__.py,sha256=l1pwfg1HyVDEqT9efdnFWcXqaZwtKwe7TrplkyxZSA0,166
|
|
263
|
-
verifiers/v1/harnesses/terminus_2/harness.py,sha256=
|
|
263
|
+
verifiers/v1/harnesses/terminus_2/harness.py,sha256=JyBUrL50n0wz8--GJ_gSDiuVNTUQQdJj41w8pnGowOY,2987
|
|
264
264
|
verifiers/v1/harnesses/terminus_2/program.py,sha256=DbkjUXTC_8-N05G9DJfzhGaMZQAfdTSNmf_THuKE1Zw,2606
|
|
265
265
|
verifiers/v1/interception/__init__.py,sha256=SnQtCpmvc4ynIAiKDIobGVgRgss8BMTQTpZvKSygn24,2735
|
|
266
266
|
verifiers/v1/interception/base.py,sha256=Riwf_qng-Gtfa4Phy2CH_HgHD4O23JaFkUpaMPc6sZw,2708
|
|
@@ -289,13 +289,13 @@ verifiers/v1/runtimes/subprocess.py,sha256=blhgfr_0YBzfxUNqiXCwMjdEUWwYW3ybebn70
|
|
|
289
289
|
verifiers/v1/runtimes/docker/__init__.py,sha256=PoBwt7Ggd_W5TtNgUX41vPuut5_c70cQTmVWnSvSgIQ,15599
|
|
290
290
|
verifiers/v1/runtimes/docker/egress.py,sha256=ELznqwFHpt1w6g4-AvoWRV_UkgvSMc5iq1DjCjvTM-k,14461
|
|
291
291
|
verifiers/v1/serve/__init__.py,sha256=UDlVB8hIHIvsV4NsYUWX_icS4g67-MQI72tqMnGmZ5o,641
|
|
292
|
-
verifiers/v1/serve/client.py,sha256=
|
|
292
|
+
verifiers/v1/serve/client.py,sha256=4R-vkhlC_49x5ofl2J_uwITc9gSNjD2fBWJ5xK4WRu8,7132
|
|
293
293
|
verifiers/v1/serve/pool.py,sha256=iOOg8PfYPY0prYlHDrrVVI6efElp-lwhqmbFdYYyKE8,14716
|
|
294
|
-
verifiers/v1/serve/server.py,sha256=
|
|
295
|
-
verifiers/v1/serve/types.py,sha256=
|
|
294
|
+
verifiers/v1/serve/server.py,sha256=HoVf9xcbmD2IzIA9kgZHkxYXT8qVrWs9Uu9VKP0RmiM,9592
|
|
295
|
+
verifiers/v1/serve/types.py,sha256=wXKx1AUqEax7IAk4nYYWjS9uvTgTABQW-pdmKN-kqE4,3009
|
|
296
296
|
verifiers/v1/tasksets/__init__.py,sha256=E1nVxnZjtvKsw-f9sW0zui5ndMMZnjMBAMdjfE5nKfI,686
|
|
297
297
|
verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
|
|
298
|
-
verifiers/v1/tasksets/harbor/taskset.py,sha256=
|
|
298
|
+
verifiers/v1/tasksets/harbor/taskset.py,sha256=6dPdS4NhIFDvrqEiq7-XQOpbeGF0HTtNVat50d2qX8A,14556
|
|
299
299
|
verifiers/v1/tasksets/lean/__init__.py,sha256=-13Mjoj6oGpiV0hxAF9fsw9-4xWpr5MPJl7oVkcQsBA,767
|
|
300
300
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
301
301
|
verifiers/v1/tasksets/lean/taskset.py,sha256=UsE6R9O20dgUz_0oBtvZlzrWqiAHvhtp5Y1CJZv_GEY,8988
|
|
@@ -305,7 +305,7 @@ verifiers/v1/tasksets/textarena/__init__.py,sha256=_R_E3PFP3X0lQGoedu_J9lKlB1a8t
|
|
|
305
305
|
verifiers/v1/tasksets/textarena/taskset.py,sha256=y2OD5zwtA_rCRqpC5I8r5fJgboFDkEX_2FIQ8OYcoh4,5252
|
|
306
306
|
verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
307
307
|
verifiers/v1/utils/aio.py,sha256=ZKTHeURNbWhTpHMyYuqMLsdPWVHyn5anfG1IJs5y-Zg,1481
|
|
308
|
-
verifiers/v1/utils/compile.py,sha256=
|
|
308
|
+
verifiers/v1/utils/compile.py,sha256=uP2QUz_MiEhG4Kmb0r6LE2VqP9uH6mH-TBWXEV5Y4ok,6247
|
|
309
309
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
310
310
|
verifiers/v1/utils/generic.py,sha256=xwu8xNGdVOhM2K87C5cqhmnRf6apSPDASA15M0S2MaY,2054
|
|
311
311
|
verifiers/v1/utils/git.py,sha256=5VX5B3011yhclPex8Mc79ZmZ5ZT4hGfKaXoqL12QMdc,4872
|
|
@@ -316,8 +316,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
316
316
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
317
317
|
verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
|
|
318
318
|
verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
|
|
319
|
-
verifiers-0.2.2.
|
|
320
|
-
verifiers-0.2.2.
|
|
321
|
-
verifiers-0.2.2.
|
|
322
|
-
verifiers-0.2.2.
|
|
323
|
-
verifiers-0.2.2.
|
|
319
|
+
verifiers-0.2.2.dev18.dist-info/METADATA,sha256=fW9wnVLClfpeL0usrYPdbc0cPhKZPhZp2J6oUHbbCsg,4540
|
|
320
|
+
verifiers-0.2.2.dev18.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
321
|
+
verifiers-0.2.2.dev18.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
322
|
+
verifiers-0.2.2.dev18.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
323
|
+
verifiers-0.2.2.dev18.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|