verifiers 0.2.2.dev49__py3-none-any.whl → 0.2.2.dev51__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/debug.py +8 -1
- verifiers/v1/cli/eval/runner.py +18 -4
- verifiers/v1/cli/validate.py +8 -1
- verifiers/v1/configs/judge.py +3 -10
- verifiers/v1/configs/taskset.py +5 -0
- verifiers/v1/env.py +2 -2
- verifiers/v1/envs/agentic_judge/env.py +11 -21
- verifiers/v1/gepa/adapter.py +4 -8
- verifiers/v1/gepa/runner.py +18 -2
- verifiers/v1/judge.py +6 -6
- verifiers/v1/loaders.py +4 -4
- verifiers/v1/mcp/server.py +2 -2
- verifiers/v1/state.py +2 -2
- verifiers/v1/task.py +42 -34
- verifiers/v1/taskset.py +72 -54
- verifiers/v1/tasksets/lean/taskset.py +1 -2
- verifiers/v1/utils/generic.py +1 -1
- verifiers/v1/utils/sampling.py +4 -5
- {verifiers-0.2.2.dev49.dist-info → verifiers-0.2.2.dev51.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev49.dist-info → verifiers-0.2.2.dev51.dist-info}/RECORD +23 -23
- {verifiers-0.2.2.dev49.dist-info → verifiers-0.2.2.dev51.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev49.dist-info → verifiers-0.2.2.dev51.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev49.dist-info → verifiers-0.2.2.dev51.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/cli/debug.py
CHANGED
|
@@ -263,7 +263,14 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
263
263
|
|
|
264
264
|
async def run_debug(config: DebugConfig) -> list[Trace]:
|
|
265
265
|
taskset = vf.load_taskset(config.taskset)
|
|
266
|
-
|
|
266
|
+
if config.num_tasks is None and taskset.INFINITE:
|
|
267
|
+
raise ValueError(
|
|
268
|
+
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
269
|
+
)
|
|
270
|
+
selected = taskset.shuffle() if config.shuffle else taskset
|
|
271
|
+
if config.num_tasks is not None:
|
|
272
|
+
selected = selected.head(config.num_tasks)
|
|
273
|
+
tasks = list(selected)
|
|
267
274
|
if isinstance(config.runtime, vf.SubprocessConfig) and any(
|
|
268
275
|
type(t).NEEDS_CONTAINER or t.data.image for t in tasks
|
|
269
276
|
):
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -26,7 +26,15 @@ logger = logging.getLogger(__name__)
|
|
|
26
26
|
async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
27
27
|
logger.info("eval config:\n%s", config.model_dump_json(indent=2))
|
|
28
28
|
client = resolve_client(config.client)
|
|
29
|
-
|
|
29
|
+
taskset = env.taskset
|
|
30
|
+
if config.num_tasks is None and taskset.INFINITE:
|
|
31
|
+
raise ValueError(
|
|
32
|
+
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
33
|
+
)
|
|
34
|
+
selected = taskset.shuffle() if config.shuffle else taskset
|
|
35
|
+
if config.num_tasks is not None:
|
|
36
|
+
selected = selected.head(config.num_tasks)
|
|
37
|
+
tasks = list(selected)
|
|
30
38
|
ctx = ModelContext(client=client, model=config.model, sampling=config.sampling)
|
|
31
39
|
semaphore = (
|
|
32
40
|
asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
|
|
@@ -132,9 +140,15 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
132
140
|
|
|
133
141
|
# The client owns the taskset: load it here, once — the server (and its pool
|
|
134
142
|
# workers) never load data, they rebuild each dispatched task from its request.
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
143
|
+
taskset = load_taskset(config.env.taskset)
|
|
144
|
+
if config.num_tasks is None and taskset.INFINITE:
|
|
145
|
+
raise ValueError(
|
|
146
|
+
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
147
|
+
)
|
|
148
|
+
selected = taskset.shuffle() if config.shuffle else taskset
|
|
149
|
+
if config.num_tasks is not None:
|
|
150
|
+
selected = selected.head(config.num_tasks)
|
|
151
|
+
tasks = list(selected)
|
|
138
152
|
# Spawned processes inherit no logging — hand them the main process's setup so
|
|
139
153
|
# their rollout logs land in the output dir.
|
|
140
154
|
level = "DEBUG" if config.verbose else "INFO"
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -203,7 +203,14 @@ async def _validate_task(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
203
203
|
|
|
204
204
|
async def run_validate(config: ValidateConfig) -> list[dict]:
|
|
205
205
|
taskset = vf.load_taskset(config.taskset)
|
|
206
|
-
|
|
206
|
+
if config.num_tasks is None and taskset.INFINITE:
|
|
207
|
+
raise ValueError(
|
|
208
|
+
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
209
|
+
)
|
|
210
|
+
selected = taskset.shuffle() if config.shuffle else taskset
|
|
211
|
+
if config.num_tasks is not None:
|
|
212
|
+
selected = selected.head(config.num_tasks)
|
|
213
|
+
tasks = list(selected)
|
|
207
214
|
if isinstance(config.runtime, vf.SubprocessConfig) and any(
|
|
208
215
|
type(t).NEEDS_CONTAINER or t.data.image for t in tasks
|
|
209
216
|
):
|
verifiers/v1/configs/judge.py
CHANGED
|
@@ -5,7 +5,7 @@ from collections.abc import Sequence
|
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
|
-
from pydantic import BaseModel, SerializeAsAny
|
|
8
|
+
from pydantic import BaseModel, SerializeAsAny
|
|
9
9
|
|
|
10
10
|
from verifiers.v1.clients import BaseClientConfig
|
|
11
11
|
from verifiers.v1.types import ID, SamplingConfig
|
|
@@ -20,15 +20,8 @@ class JudgeConfig(BaseClientConfig):
|
|
|
20
20
|
weight: float = 1.0
|
|
21
21
|
model: str = "openai/gpt-5.4-nano"
|
|
22
22
|
sampling: SamplingConfig = SamplingConfig()
|
|
23
|
-
prompt:
|
|
24
|
-
|
|
25
|
-
"""Prompt file override, mutually exclusive with `prompt`."""
|
|
26
|
-
|
|
27
|
-
@model_validator(mode="after")
|
|
28
|
-
def check_prompt_source(self) -> "JudgeConfig":
|
|
29
|
-
if self.prompt is not None and self.prompt_file is not None:
|
|
30
|
-
raise ValueError("set `prompt` or `prompt_file`, not both")
|
|
31
|
-
return self
|
|
23
|
+
prompt: Path | None = None
|
|
24
|
+
"""File whose text overrides the judge's default prompt template."""
|
|
32
25
|
|
|
33
26
|
|
|
34
27
|
Judges = list[SerializeAsAny[JudgeConfig]]
|
verifiers/v1/configs/taskset.py
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
"""The taskset plugin's config: which rows load, under `--env.taskset.*`."""
|
|
2
2
|
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
3
5
|
from pydantic import SerializeAsAny
|
|
4
6
|
from pydantic_config import BaseConfig
|
|
5
7
|
|
|
@@ -14,6 +16,9 @@ class TasksetConfig(BaseConfig):
|
|
|
14
16
|
positional `eval <taskset-id>`)."""
|
|
15
17
|
task: SerializeAsAny[TaskConfig] = TaskConfig()
|
|
16
18
|
"""Config passed to each task, under `--env.taskset.task.*`."""
|
|
19
|
+
system_prompt: Path | None = None
|
|
20
|
+
"""File whose text overrides each task's `TaskData.system_prompt` on
|
|
21
|
+
iteration (e.g. a GEPA `best_system_prompt.txt`)."""
|
|
17
22
|
|
|
18
23
|
@property
|
|
19
24
|
def name(self) -> str:
|
verifiers/v1/env.py
CHANGED
|
@@ -33,7 +33,7 @@ from verifiers.v1.retries import run_episode_with_retry
|
|
|
33
33
|
from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
|
|
34
34
|
from verifiers.v1.task import Task, resolve_server_config
|
|
35
35
|
from verifiers.v1.trace import Error, Trace
|
|
36
|
-
from verifiers.v1.utils.generic import
|
|
36
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
37
37
|
from verifiers.v1.utils.memory import trim_memory_periodically
|
|
38
38
|
|
|
39
39
|
logger = logging.getLogger(__name__)
|
|
@@ -83,7 +83,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
83
83
|
def __init__(self, config: ConfigT) -> None:
|
|
84
84
|
from verifiers.v1.loaders import load_harness, load_taskset
|
|
85
85
|
|
|
86
|
-
config_cls =
|
|
86
|
+
config_cls = concrete_type(type(self), EnvConfig, origin=Env) or EnvConfig
|
|
87
87
|
if not isinstance(config, config_cls):
|
|
88
88
|
raise TypeError(
|
|
89
89
|
f"{type(self).__name__} declares Env[{config_cls.__name__}], "
|
|
@@ -180,41 +180,31 @@ class JudgeTask(vf.Task):
|
|
|
180
180
|
class JudgeTaskConfig(vf.BaseConfig):
|
|
181
181
|
"""The judge's minted task: the grading policy and what lands in its box."""
|
|
182
182
|
|
|
183
|
-
prompt: Path |
|
|
184
|
-
"""Grading-policy
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
`.txt` file): task-family pointers into the trace or box — e.g. for math,
|
|
192
|
-
where the reference answer lives in the record; for SWE, to diff the repo
|
|
193
|
-
or read `info.patch`."""
|
|
183
|
+
prompt: Path | None = None
|
|
184
|
+
"""Grading-policy file. Replaces only the policy body — the verdict contract
|
|
185
|
+
and workspace note are always appended. May reference `{prompt}` (the solver
|
|
186
|
+
task's prompt); if it doesn't, the task statement is appended after."""
|
|
187
|
+
hint: Path | None = None
|
|
188
|
+
"""Optional hints file injected as their own section: task-family pointers
|
|
189
|
+
into the trace or box — e.g. for math, where the reference answer lives in
|
|
190
|
+
the record; for SWE, to diff the repo or read `info.patch`."""
|
|
194
191
|
rubric: Path | None = None
|
|
195
192
|
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
196
193
|
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
197
194
|
files work for both. None grades the single built-in `solved` criterion."""
|
|
198
195
|
|
|
199
|
-
@staticmethod
|
|
200
|
-
def _resolve(value: Path | str) -> str:
|
|
201
|
-
path = Path(value)
|
|
202
|
-
if isinstance(value, Path) or path.suffix in (".md", ".txt"):
|
|
203
|
-
return path.read_text(encoding="utf-8")
|
|
204
|
-
return str(value)
|
|
205
|
-
|
|
206
196
|
def build_prompt(self) -> str:
|
|
207
197
|
if self.prompt is None:
|
|
208
198
|
return GRADE_PROMPT + "\n\n" + TASK_SECTION
|
|
209
|
-
return self.
|
|
199
|
+
return self.prompt.read_text()
|
|
210
200
|
|
|
211
201
|
def build_hint(self) -> str | None:
|
|
212
|
-
return self.
|
|
202
|
+
return self.hint.read_text() if self.hint is not None else None
|
|
213
203
|
|
|
214
204
|
def criteria(self) -> list[Criterion]:
|
|
215
205
|
if self.rubric is None:
|
|
216
206
|
return [SOLVED]
|
|
217
|
-
text = self.rubric.read_text(
|
|
207
|
+
text = self.rubric.read_text()
|
|
218
208
|
data = (
|
|
219
209
|
tomllib.loads(text)
|
|
220
210
|
if self.rubric.suffix.lower() == ".toml"
|
verifiers/v1/gepa/adapter.py
CHANGED
|
@@ -68,14 +68,10 @@ class GEPAAdapter:
|
|
|
68
68
|
)
|
|
69
69
|
|
|
70
70
|
async def _run_batch(self, batch: list[int], system_prompt: str) -> list[Episode]:
|
|
71
|
-
# Inject the candidate
|
|
72
|
-
#
|
|
73
|
-
tasks
|
|
74
|
-
|
|
75
|
-
t.data.model_copy(update={"system_prompt": system_prompt}), t.config
|
|
76
|
-
)
|
|
77
|
-
for t in (self.tasks[idx] for idx in batch)
|
|
78
|
-
]
|
|
71
|
+
# Inject the candidate as a copy of each base task with its system_prompt overridden
|
|
72
|
+
# (`with_system_prompt` copies rather than reconstructs, so subclass state survives and
|
|
73
|
+
# the shared base task in `self.tasks` is left untouched for the next candidate).
|
|
74
|
+
tasks = [self.tasks[idx].with_system_prompt(system_prompt) for idx in batch]
|
|
79
75
|
slots = [slot for task in tasks for slot in self.env.slots(task)]
|
|
80
76
|
results = await asyncio.gather(
|
|
81
77
|
*(
|
verifiers/v1/gepa/runner.py
CHANGED
|
@@ -37,7 +37,10 @@ class _GEPALog:
|
|
|
37
37
|
|
|
38
38
|
def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
|
|
39
39
|
logger.info("gepa config:\n%s", config.model_dump_json(indent=2))
|
|
40
|
-
|
|
40
|
+
# Global shuffle: an infinite taskset raises here — run it with
|
|
41
|
+
# shuffle=false (there is no whole set to sample from).
|
|
42
|
+
taskset = env.taskset.shuffle() if config.shuffle else env.taskset
|
|
43
|
+
all_tasks = list(taskset.head(config.num_train + config.num_val))
|
|
41
44
|
train_tasks, val_tasks = split_tasks(all_tasks, config.num_train, config.num_val)
|
|
42
45
|
selected_tasks = [*train_tasks, *val_tasks]
|
|
43
46
|
# Seed from the tasks GEPA actually evaluates (train ∪ val), not the full pre-split pool —
|
|
@@ -105,7 +108,20 @@ def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
|
|
|
105
108
|
"skip_perfect_score": False,
|
|
106
109
|
"logger": _GEPALog(),
|
|
107
110
|
}
|
|
108
|
-
|
|
111
|
+
result = optimize(**optimize_kwargs)
|
|
112
|
+
if run_dir is not None:
|
|
113
|
+
# Persist the winning prompt as a plain file so it can be handed straight to
|
|
114
|
+
# eval/train via `--env.taskset.system-prompt` (see TasksetConfig).
|
|
115
|
+
candidate = result.best_candidate
|
|
116
|
+
best = (
|
|
117
|
+
candidate.get("system_prompt", "")
|
|
118
|
+
if isinstance(candidate, dict)
|
|
119
|
+
else str(candidate)
|
|
120
|
+
)
|
|
121
|
+
best_path = run_dir / "best_system_prompt.txt"
|
|
122
|
+
best_path.write_text(best, encoding="utf-8")
|
|
123
|
+
logger.info("best system prompt: %s", best_path)
|
|
124
|
+
return result
|
|
109
125
|
finally:
|
|
110
126
|
loop.run_until_complete(serving.__aexit__(None, None, None))
|
|
111
127
|
finally:
|
verifiers/v1/judge.py
CHANGED
|
@@ -63,7 +63,7 @@ from verifiers.v1.configs.judge import (
|
|
|
63
63
|
from verifiers.v1.dialects.chat import message_to_wire
|
|
64
64
|
from verifiers.v1.scoring import parse_judge_choice
|
|
65
65
|
from verifiers.v1.types import Messages, StrictBaseModel, Usage
|
|
66
|
-
from verifiers.v1.utils.generic import
|
|
66
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
67
67
|
|
|
68
68
|
if TYPE_CHECKING:
|
|
69
69
|
from verifiers.v1.task import TaskData
|
|
@@ -111,18 +111,18 @@ ConfigT = TypeVar("ConfigT", bound=JudgeConfig, default=JudgeConfig)
|
|
|
111
111
|
|
|
112
112
|
def judge_config_cls(cls: type) -> type[JudgeConfig]:
|
|
113
113
|
"""Resolve a judge's config specialization through its MRO, else `JudgeConfig`."""
|
|
114
|
-
return
|
|
114
|
+
return concrete_type(cls, JudgeConfig) or JudgeConfig
|
|
115
115
|
|
|
116
116
|
|
|
117
117
|
class Judge(Generic[ParsedT, ConfigT]):
|
|
118
118
|
prompt: str | None = None
|
|
119
|
-
"""Default prompt template, overridden by config."""
|
|
119
|
+
"""Default prompt template, overridden by a config `prompt` file."""
|
|
120
120
|
schema: type[BaseModel] | None = None
|
|
121
121
|
|
|
122
122
|
def __init__(self, config: ConfigT | None = None) -> None:
|
|
123
123
|
self.config = cast(ConfigT, config or judge_config_cls(type(self))())
|
|
124
|
-
if self.config.
|
|
125
|
-
self.prompt = self.config.
|
|
124
|
+
if self.config.prompt is not None:
|
|
125
|
+
self.prompt = self.config.prompt.read_text()
|
|
126
126
|
|
|
127
127
|
@property
|
|
128
128
|
def reward_name(self) -> str:
|
|
@@ -132,7 +132,7 @@ class Judge(Generic[ParsedT, ConfigT]):
|
|
|
132
132
|
return judge_key(self.config) or fallback or "judge"
|
|
133
133
|
|
|
134
134
|
def build_messages(self, **fields: Any) -> str | Messages:
|
|
135
|
-
template = self.
|
|
135
|
+
template = self.prompt
|
|
136
136
|
if template is None:
|
|
137
137
|
raise ValueError(
|
|
138
138
|
f"{type(self).__name__} has no `prompt`; set it or override build_messages"
|
verifiers/v1/loaders.py
CHANGED
|
@@ -20,7 +20,7 @@ from verifiers.v1.harness import Harness
|
|
|
20
20
|
from verifiers.v1.judge import Judge, judge_config_cls
|
|
21
21
|
from verifiers.v1.task import Task
|
|
22
22
|
from verifiers.v1.taskset import Taskset
|
|
23
|
-
from verifiers.v1.utils.generic import
|
|
23
|
+
from verifiers.v1.utils.generic import concrete_type, prefix_validation_error
|
|
24
24
|
from verifiers.v1.utils.install import ensure_installed
|
|
25
25
|
|
|
26
26
|
|
|
@@ -208,7 +208,7 @@ def load_judge(config: JudgeConfig) -> Judge:
|
|
|
208
208
|
def taskset_config_type(taskset_id: str) -> type[TasksetConfig]:
|
|
209
209
|
"""Resolve the taskset's config specialization through its MRO."""
|
|
210
210
|
return (
|
|
211
|
-
|
|
211
|
+
concrete_type(taskset_class(taskset_id), TasksetConfig, origin=Taskset)
|
|
212
212
|
or TasksetConfig
|
|
213
213
|
)
|
|
214
214
|
|
|
@@ -216,7 +216,7 @@ def taskset_config_type(taskset_id: str) -> type[TasksetConfig]:
|
|
|
216
216
|
def harness_config_type(harness_id: str) -> type[HarnessConfig]:
|
|
217
217
|
"""Resolve the harness's config specialization through its MRO."""
|
|
218
218
|
return (
|
|
219
|
-
|
|
219
|
+
concrete_type(harness_class(harness_id), HarnessConfig, origin=Harness)
|
|
220
220
|
or HarnessConfig
|
|
221
221
|
)
|
|
222
222
|
|
|
@@ -231,7 +231,7 @@ def env_config_type(taskset_id: str, env_id: str = "") -> type[EnvConfig]:
|
|
|
231
231
|
its MRO — `SingleAgentEnvConfig` for a plain taskset. The run's `env` field
|
|
232
232
|
narrows to this, which is what gives `--env.<role>.model` addressing."""
|
|
233
233
|
return (
|
|
234
|
-
|
|
234
|
+
concrete_type(environment_class(taskset_id, env_id), EnvConfig, origin=Env)
|
|
235
235
|
or EnvConfig
|
|
236
236
|
)
|
|
237
237
|
|
verifiers/v1/mcp/server.py
CHANGED
|
@@ -13,7 +13,7 @@ from pydantic import TypeAdapter, ValidationError
|
|
|
13
13
|
from pydantic_config import BaseConfig
|
|
14
14
|
|
|
15
15
|
from verifiers.v1.state import State, StateT, state_cls
|
|
16
|
-
from verifiers.v1.utils.generic import
|
|
16
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
17
17
|
|
|
18
18
|
if TYPE_CHECKING:
|
|
19
19
|
from httpx import AsyncClient, Response
|
|
@@ -286,7 +286,7 @@ class ServerBase(Generic[ConfigT, StateT]):
|
|
|
286
286
|
@classmethod
|
|
287
287
|
def _config_cls(cls) -> type[BaseConfig]:
|
|
288
288
|
"""Resolve the server's config specialization through its MRO."""
|
|
289
|
-
if config_cls :=
|
|
289
|
+
if config_cls := concrete_type(cls, BaseConfig):
|
|
290
290
|
return config_cls
|
|
291
291
|
raise TypeError(
|
|
292
292
|
f"{cls.__name__} must parameterize its config, e.g. Toolset[MyConfig]"
|
verifiers/v1/state.py
CHANGED
|
@@ -8,7 +8,7 @@ from pydantic import ConfigDict
|
|
|
8
8
|
from typing_extensions import TypeVar
|
|
9
9
|
|
|
10
10
|
from verifiers.v1.types import StrictBaseModel
|
|
11
|
-
from verifiers.v1.utils.generic import
|
|
11
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
12
12
|
|
|
13
13
|
|
|
14
14
|
class State(StrictBaseModel):
|
|
@@ -20,4 +20,4 @@ StateT = TypeVar("StateT", bound=State, default=State)
|
|
|
20
20
|
|
|
21
21
|
def state_cls(cls: type) -> type[State]:
|
|
22
22
|
"""Resolve a class's `State` specialization through its MRO, else `State`."""
|
|
23
|
-
return
|
|
23
|
+
return concrete_type(cls, State) or State
|
verifiers/v1/task.py
CHANGED
|
@@ -24,10 +24,11 @@ wraps it in the declared `Task` — one task type per taskset.
|
|
|
24
24
|
|
|
25
25
|
from __future__ import annotations
|
|
26
26
|
|
|
27
|
+
import copy
|
|
27
28
|
import inspect
|
|
28
29
|
import logging
|
|
29
30
|
from collections.abc import Mapping
|
|
30
|
-
from typing import TYPE_CHECKING, ClassVar, Generic
|
|
31
|
+
from typing import TYPE_CHECKING, ClassVar, Generic, Self
|
|
31
32
|
|
|
32
33
|
from pydantic import ConfigDict, Field
|
|
33
34
|
from pydantic_config import BaseConfig
|
|
@@ -38,7 +39,7 @@ from verifiers.v1.decorators import discover_decorated, invoke_all
|
|
|
38
39
|
from verifiers.v1.errors import TaskError, boundary
|
|
39
40
|
from verifiers.v1.state import StateT
|
|
40
41
|
from verifiers.v1.types import Messages, StrictBaseModel, content_text
|
|
41
|
-
from verifiers.v1.utils.generic import
|
|
42
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
42
43
|
|
|
43
44
|
if TYPE_CHECKING:
|
|
44
45
|
from verifiers.v1.judge import Judge
|
|
@@ -49,27 +50,6 @@ if TYPE_CHECKING:
|
|
|
49
50
|
logger = logging.getLogger(__name__)
|
|
50
51
|
|
|
51
52
|
|
|
52
|
-
def _requires_runtime(fn) -> bool:
|
|
53
|
-
param = inspect.signature(fn).parameters.get("runtime")
|
|
54
|
-
# A defaulted runtime parameter can still be called offline with None.
|
|
55
|
-
return param is not None and param.default is inspect.Parameter.empty
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def _record_result(
|
|
59
|
-
trace: Trace,
|
|
60
|
-
name: str,
|
|
61
|
-
result,
|
|
62
|
-
weight: float | None = None,
|
|
63
|
-
) -> None:
|
|
64
|
-
"""Record a scalar or keyed scoring result."""
|
|
65
|
-
values = list(result.items()) if isinstance(result, Mapping) else [(name, result)]
|
|
66
|
-
for key, value in values:
|
|
67
|
-
if weight is None:
|
|
68
|
-
trace.record_metric(key, value)
|
|
69
|
-
else:
|
|
70
|
-
trace.record_reward(key, value, weight)
|
|
71
|
-
|
|
72
|
-
|
|
73
53
|
class TaskResources(StrictBaseModel):
|
|
74
54
|
model_config = ConfigDict(frozen=True)
|
|
75
55
|
|
|
@@ -148,12 +128,12 @@ ConfigT = TypeVar("ConfigT", bound=TaskConfig, default=TaskConfig)
|
|
|
148
128
|
|
|
149
129
|
def task_data_cls(cls: type) -> type[TaskData]:
|
|
150
130
|
"""Resolve a task's `TaskData` specialization through its MRO, else `TaskData`."""
|
|
151
|
-
return
|
|
131
|
+
return concrete_type(cls, TaskData) or TaskData
|
|
152
132
|
|
|
153
133
|
|
|
154
134
|
def task_config_cls(cls: type) -> type[TaskConfig]:
|
|
155
135
|
"""Resolve a task's `TaskConfig` specialization through its MRO, else `TaskConfig`."""
|
|
156
|
-
return
|
|
136
|
+
return concrete_type(cls, TaskConfig) or TaskConfig
|
|
157
137
|
|
|
158
138
|
|
|
159
139
|
def resolve_server_config(
|
|
@@ -197,6 +177,15 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
197
177
|
self.data = data
|
|
198
178
|
self.config = config if config is not None else task_config_cls(type(self))()
|
|
199
179
|
|
|
180
|
+
def with_system_prompt(self, system_prompt: str) -> Self:
|
|
181
|
+
"""A shallow copy of this task with `data.system_prompt` overridden. Copies the
|
|
182
|
+
instance instead of reconstructing via `type(self)(...)`, so a subclass with a
|
|
183
|
+
non-`(data, config)` constructor or extra load-time state keeps it. Used to apply the
|
|
184
|
+
config-layer / GEPA system prompt (see `TasksetConfig` and `verifiers.v1.gepa`)."""
|
|
185
|
+
clone = copy.copy(self)
|
|
186
|
+
clone.data = self.data.model_copy(update={"system_prompt": system_prompt})
|
|
187
|
+
return clone
|
|
188
|
+
|
|
200
189
|
def plugged_judges(self) -> list[Judge]:
|
|
201
190
|
from verifiers.v1.loaders import load_judge
|
|
202
191
|
|
|
@@ -229,6 +218,11 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
229
218
|
trace: Trace,
|
|
230
219
|
runtime: Runtime | None = None,
|
|
231
220
|
) -> None:
|
|
221
|
+
def requires_runtime(fn) -> bool:
|
|
222
|
+
param = inspect.signature(fn).parameters.get("runtime")
|
|
223
|
+
# A defaulted runtime parameter can still be called offline with None.
|
|
224
|
+
return param is not None and param.default is inspect.Parameter.empty
|
|
225
|
+
|
|
232
226
|
judges = self.plugged_judges()
|
|
233
227
|
available = {"task": self.data, "trace": trace}
|
|
234
228
|
if runtime is not None:
|
|
@@ -239,36 +233,50 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
239
233
|
rewards = discover_decorated(self, "reward")
|
|
240
234
|
if runtime is None:
|
|
241
235
|
skipped = [
|
|
242
|
-
fn.__name__ for fn in (*metrics, *rewards) if
|
|
236
|
+
fn.__name__ for fn in (*metrics, *rewards) if requires_runtime(fn)
|
|
243
237
|
] + [
|
|
244
238
|
judge.reward_name
|
|
245
239
|
for judge in judges
|
|
246
|
-
if
|
|
240
|
+
if requires_runtime(judge.score)
|
|
247
241
|
]
|
|
248
242
|
if skipped:
|
|
249
243
|
logger.info(
|
|
250
244
|
"score: no runtime — skipped runtime-dependent signals: %s",
|
|
251
245
|
skipped,
|
|
252
246
|
)
|
|
253
|
-
metrics = [fn for fn in metrics if not
|
|
254
|
-
rewards = [fn for fn in rewards if not
|
|
247
|
+
metrics = [fn for fn in metrics if not requires_runtime(fn)]
|
|
248
|
+
rewards = [fn for fn in rewards if not requires_runtime(fn)]
|
|
255
249
|
judges = [
|
|
256
|
-
judge for judge in judges if not
|
|
250
|
+
judge for judge in judges if not requires_runtime(judge.score)
|
|
257
251
|
]
|
|
258
252
|
|
|
259
253
|
metric_results = await invoke_all(metrics, available)
|
|
260
254
|
for fn, result in zip(metrics, metric_results):
|
|
261
|
-
|
|
255
|
+
if isinstance(result, Mapping):
|
|
256
|
+
trace.record_metrics(result)
|
|
257
|
+
else:
|
|
258
|
+
trace.record_metric(fn.__name__, result)
|
|
262
259
|
reward_results = await invoke_all(rewards, available)
|
|
263
260
|
for fn, result in zip(rewards, reward_results):
|
|
264
|
-
|
|
265
|
-
|
|
261
|
+
weight = getattr(fn, "_vf_weight", 1.0)
|
|
262
|
+
items = (
|
|
263
|
+
result.items()
|
|
264
|
+
if isinstance(result, Mapping)
|
|
265
|
+
else [(fn.__name__, result)]
|
|
266
266
|
)
|
|
267
|
+
for key, value in items:
|
|
268
|
+
trace.record_reward(key, value, weight)
|
|
267
269
|
judge_results = await invoke_all(
|
|
268
270
|
[judge.score for judge in judges], available
|
|
269
271
|
)
|
|
270
272
|
for judge, result in zip(judges, judge_results):
|
|
271
|
-
|
|
273
|
+
items = (
|
|
274
|
+
result.items()
|
|
275
|
+
if isinstance(result, Mapping)
|
|
276
|
+
else [(judge.reward_name, result)]
|
|
277
|
+
)
|
|
278
|
+
for key, value in items:
|
|
279
|
+
trace.record_reward(key, value, judge.config.weight)
|
|
272
280
|
|
|
273
281
|
|
|
274
282
|
TaskT = TypeVar("TaskT", bound=Task)
|
verifiers/v1/taskset.py
CHANGED
|
@@ -1,88 +1,106 @@
|
|
|
1
1
|
"""The taskset: a thin loader that yields typed tasks.
|
|
2
2
|
|
|
3
|
-
A `Taskset` is the data half of an environment: config in, tasks out. `load()`
|
|
4
|
-
|
|
5
|
-
the config's task-facing subtree:
|
|
3
|
+
A `Taskset` is the data half of an environment: config in, tasks out. `load()` is
|
|
4
|
+
the main hook that builds each task:
|
|
6
5
|
|
|
7
6
|
def load(self) -> Iterable[MyTask]:
|
|
8
|
-
|
|
7
|
+
for i in ...:
|
|
8
|
+
yield MyTask(MyData(idx=i, ...), self.config.task)
|
|
9
9
|
|
|
10
|
-
`load` may also be a generator
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
Load-time knobs live on the taskset config, task-facing knobs under its `task`
|
|
14
|
-
subtree, shared tool servers on `tools`. All per-task behavior lives on the `Task`.
|
|
15
|
-
|
|
16
|
-
The class stays generic (`Taskset[TaskT, TasksetConfigT]`) so the loaders can read
|
|
17
|
-
the types: `taskset_config_type` narrows `--env.taskset.*` flags, `task_type` types
|
|
18
|
-
the wire trace — one task type per taskset, so replay can rebuild saved rows.
|
|
10
|
+
`load` may also be a generator for infinite tasksets. There is a one-to-one
|
|
11
|
+
mapping between taskset and task type, i.e. a taskset may only yield one task
|
|
12
|
+
type.
|
|
19
13
|
"""
|
|
20
14
|
|
|
21
15
|
from __future__ import annotations
|
|
22
16
|
|
|
17
|
+
import copy
|
|
23
18
|
import itertools
|
|
24
|
-
import
|
|
25
|
-
from
|
|
26
|
-
from
|
|
19
|
+
import random
|
|
20
|
+
from abc import ABC, abstractmethod
|
|
21
|
+
from collections.abc import Callable, Iterable, Iterator
|
|
22
|
+
from typing import TYPE_CHECKING, ClassVar, Generic, Self
|
|
27
23
|
|
|
28
24
|
from pydantic_config import BaseConfig
|
|
29
25
|
from typing_extensions import TypeVar
|
|
30
26
|
|
|
31
27
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
32
28
|
from verifiers.v1.task import Task, TaskT, resolve_server_config
|
|
33
|
-
from verifiers.v1.utils.generic import
|
|
34
|
-
from verifiers.v1.utils.sampling import
|
|
29
|
+
from verifiers.v1.utils.generic import concrete_type
|
|
30
|
+
from verifiers.v1.utils.sampling import SEED
|
|
35
31
|
|
|
36
32
|
if TYPE_CHECKING:
|
|
37
33
|
from verifiers.v1.mcp import Toolset
|
|
38
34
|
|
|
39
|
-
logger = logging.getLogger(__name__)
|
|
40
|
-
|
|
41
|
-
|
|
42
35
|
TasksetConfigT = TypeVar("TasksetConfigT", bound=TasksetConfig, default=TasksetConfig)
|
|
43
36
|
|
|
44
37
|
|
|
45
|
-
class Taskset(Generic[TaskT, TasksetConfigT]):
|
|
46
|
-
INFINITE:
|
|
47
|
-
"""Whether
|
|
48
|
-
|
|
38
|
+
class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
39
|
+
INFINITE: bool = False
|
|
40
|
+
"""Whether the taskset is infinite (yields tasks forever). Class-declared;
|
|
41
|
+
a `head(n)` view shadows it per instance (bounded by construction)."""
|
|
49
42
|
|
|
50
43
|
tools: ClassVar[tuple[type[Toolset], ...]] = ()
|
|
51
|
-
"""Tool
|
|
44
|
+
"""Tool servers shared by all tasks in the taskset. The environment will
|
|
45
|
+
spawn a single, global instance, reused across tasks."""
|
|
52
46
|
|
|
53
47
|
def __init__(self, config: TasksetConfigT) -> None:
|
|
54
48
|
self.config = config
|
|
49
|
+
override = config.system_prompt
|
|
50
|
+
self.system_prompt = override.read_text() if override is not None else None
|
|
51
|
+
self.transform: Callable[[Iterator[TaskT]], Iterator[TaskT]] | None = None
|
|
52
|
+
"""Iteration transform carried by `head`/`shuffle` views (see `view`)."""
|
|
53
|
+
|
|
54
|
+
@abstractmethod
|
|
55
|
+
def load(self) -> Iterable[TaskT]:
|
|
56
|
+
"""Build and yield the taskset's tasks; may be a generator (see module doc)."""
|
|
57
|
+
|
|
58
|
+
def __iter__(self) -> Iterator[TaskT]:
|
|
59
|
+
"""Lazily iterate `load()` with the config-layer system prompt applied and
|
|
60
|
+
any view transform on top — the read path; `load` is the subclass hook."""
|
|
61
|
+
prompt = self.system_prompt
|
|
62
|
+
tasks = (
|
|
63
|
+
task.with_system_prompt(prompt) if prompt is not None else task
|
|
64
|
+
for task in self.load()
|
|
65
|
+
)
|
|
66
|
+
yield from self.transform(tasks) if self.transform is not None else tasks
|
|
67
|
+
|
|
68
|
+
def view(self, transform: Callable[[Iterator[TaskT]], Iterator[TaskT]]) -> Self:
|
|
69
|
+
"""A shallow copy of this taskset iterating through `transform`, composed
|
|
70
|
+
onto any transform this taskset already carries."""
|
|
71
|
+
clone = copy.copy(self)
|
|
72
|
+
prev = self.transform
|
|
73
|
+
clone.transform = (
|
|
74
|
+
transform if prev is None else lambda tasks: transform(prev(tasks))
|
|
75
|
+
)
|
|
76
|
+
return clone
|
|
77
|
+
|
|
78
|
+
def head(self, num_tasks: int) -> Self:
|
|
79
|
+
"""A lazy, always-finite view of the first `num_tasks` tasks."""
|
|
80
|
+
view = self.view(lambda tasks: itertools.islice(tasks, num_tasks))
|
|
81
|
+
view.INFINITE = False
|
|
82
|
+
return view
|
|
83
|
+
|
|
84
|
+
def shuffle(self, seed: int | None = None) -> Self:
|
|
85
|
+
"""A shuffled view under `seed` — the shared fixed seed when None, so runs
|
|
86
|
+
sample reproducibly (materializes the receiver on iteration); raises on an
|
|
87
|
+
infinite taskset — bound it first (`head(n).shuffle()`)."""
|
|
88
|
+
if self.INFINITE:
|
|
89
|
+
raise ValueError(
|
|
90
|
+
f"{type(self).__name__} is infinite - cannot shuffle; "
|
|
91
|
+
"bound it first with head(num_tasks)"
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
def shuffled(tasks: Iterator[TaskT]) -> Iterator[TaskT]:
|
|
95
|
+
materialized = list(tasks)
|
|
96
|
+
random.Random(SEED if seed is None else seed).shuffle(materialized)
|
|
97
|
+
return iter(materialized)
|
|
98
|
+
|
|
99
|
+
return self.view(shuffled)
|
|
55
100
|
|
|
56
101
|
@classmethod
|
|
57
102
|
def task_type(cls) -> type[Task]:
|
|
58
|
-
return
|
|
59
|
-
|
|
60
|
-
def load(self) -> Iterable[TaskT]:
|
|
61
|
-
raise NotImplementedError
|
|
62
|
-
|
|
63
|
-
def select(
|
|
64
|
-
self, num_tasks: int | None = None, shuffle: bool = False
|
|
65
|
-
) -> list[TaskT]:
|
|
66
|
-
"""Materialize the first `num_tasks` off `load` (all when `None`), pulled
|
|
67
|
-
lazily. `shuffle` samples from the whole taskset instead (fixed-seed), which
|
|
68
|
-
materializes everything first; on an `INFINITE` taskset it's a warned no-op —
|
|
69
|
-
the first `num_tasks` generated are already an arbitrary sample."""
|
|
70
|
-
if type(self).INFINITE:
|
|
71
|
-
if num_tasks is None:
|
|
72
|
-
raise ValueError(
|
|
73
|
-
f"{type(self).__name__} is infinite - select a bounded subset "
|
|
74
|
-
"with num_tasks (-n on the CLI)"
|
|
75
|
-
)
|
|
76
|
-
if shuffle:
|
|
77
|
-
logger.warning(
|
|
78
|
-
"shuffle is a no-op on an infinite taskset - "
|
|
79
|
-
"taking the first %d generated tasks",
|
|
80
|
-
num_tasks,
|
|
81
|
-
)
|
|
82
|
-
return list(itertools.islice(self.load(), num_tasks))
|
|
83
|
-
if shuffle:
|
|
84
|
-
return sample(self.load(), shuffle=True, limit=num_tasks)
|
|
85
|
-
return list(itertools.islice(self.load(), num_tasks))
|
|
103
|
+
return concrete_type(cls, Task, origin=Taskset) or Task
|
|
86
104
|
|
|
87
105
|
def server_config(self, server_cls: type) -> BaseConfig:
|
|
88
106
|
"""The config a `tools` entry is built with, resolved off `self.config` (the
|
|
@@ -61,7 +61,6 @@ class LeanTaskConfig(TaskConfig):
|
|
|
61
61
|
class LeanConfig(TasksetConfig):
|
|
62
62
|
dataset: LeanDatasetConfig
|
|
63
63
|
docker_image: str = DEFAULT_DOCKER_IMAGE
|
|
64
|
-
system_prompt: str = DEFAULT_SYSTEM_PROMPT
|
|
65
64
|
task: LeanTaskConfig = LeanTaskConfig()
|
|
66
65
|
|
|
67
66
|
|
|
@@ -187,7 +186,7 @@ class LeanTaskset(Taskset[LeanTask, LeanConfig]):
|
|
|
187
186
|
idx=index,
|
|
188
187
|
name=str(name) if name else f"task_{index:05d}",
|
|
189
188
|
prompt=self._build_prompt(formal_statement, header),
|
|
190
|
-
system_prompt=
|
|
189
|
+
system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
191
190
|
image=config.docker_image,
|
|
192
191
|
workdir=config.task.lean_project_path,
|
|
193
192
|
resources=resources,
|
verifiers/v1/utils/generic.py
CHANGED
|
@@ -39,7 +39,7 @@ def deep_merge(base: dict, override: dict) -> dict:
|
|
|
39
39
|
return merged
|
|
40
40
|
|
|
41
41
|
|
|
42
|
-
def
|
|
42
|
+
def concrete_type(
|
|
43
43
|
cls: type, bound: type[T], *, origin: type | None = None
|
|
44
44
|
) -> type[T] | None:
|
|
45
45
|
"""Find a concrete bounded type through `cls`'s MRO, most-derived first."""
|
verifiers/v1/utils/sampling.py
CHANGED
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
"""Shared sampling: an optional fixed-seed shuffle, then an optional head-slice.
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
`Taskset.
|
|
7
|
-
sample plain index lists here directly.
|
|
3
|
+
With `--shuffle`, a shuffle under the fixed `SEED` so the sampled subset is the *same*
|
|
4
|
+
every run (reproducible), then an optional slice to the first `limit`. Used by the paths
|
|
5
|
+
that sample plain index lists (the server eval path and the legacy bridge); tasksets
|
|
6
|
+
shuffle themselves (`Taskset.shuffle`, same default seed).
|
|
8
7
|
"""
|
|
9
8
|
|
|
10
9
|
import random
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev51
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -170,35 +170,35 @@ verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2
|
|
|
170
170
|
verifiers/v1/__init__.py,sha256=xHKE-HZA8bN0JJovf4KbBI2mxBtZJ4hPNnzac4FtWwM,7332
|
|
171
171
|
verifiers/v1/agent.py,sha256=dj4GabQSs6Wzwbph5Rlx-efsVQsSuWV2YI5hfvq5mGA,31990
|
|
172
172
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
173
|
-
verifiers/v1/env.py,sha256=
|
|
173
|
+
verifiers/v1/env.py,sha256=m-1TNzs9YyFbMkoW6c54vI5xG4jnW3LvLgvJhOgbg04,18447
|
|
174
174
|
verifiers/v1/episode.py,sha256=KWyM9ovTphEIo2260BYu2HvtnOOTxjLRDPrF_jTCiuA,2497
|
|
175
175
|
verifiers/v1/errors.py,sha256=slZnrtqMo_BgXQV46wJJhV6vvAAnVsuCbVGp2vBVWfs,6888
|
|
176
176
|
verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
|
|
177
177
|
verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
|
|
178
|
-
verifiers/v1/judge.py,sha256=
|
|
178
|
+
verifiers/v1/judge.py,sha256=duZcWYTynzqShs8vfbExpFlgLS-JA9NFqdmCRLg3sbc,9393
|
|
179
179
|
verifiers/v1/legacy.py,sha256=nkbdhFJZN1QzsjE42AB_A_c1zrk3_Y8zR07srZJS2ck,22584
|
|
180
|
-
verifiers/v1/loaders.py,sha256=
|
|
180
|
+
verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
|
|
181
181
|
verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
|
|
182
182
|
verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
|
|
183
183
|
verifiers/v1/rollout.py,sha256=GJzHgJFlw8O1vDtWM__2JhSiG2GqbfjtH2ri1MRfw3g,20495
|
|
184
184
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
185
185
|
verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
186
|
-
verifiers/v1/state.py,sha256=
|
|
187
|
-
verifiers/v1/task.py,sha256=
|
|
188
|
-
verifiers/v1/taskset.py,sha256=
|
|
186
|
+
verifiers/v1/state.py,sha256=LhV7lpqcC19_ZQZ4KKO4dzTmZHJO4VEOpjPaLMBWh54,691
|
|
187
|
+
verifiers/v1/task.py,sha256=hRDIHAdBwP-aalLmhHeJ1VWW_9YJAkP21w7TkOU7VXU,11805
|
|
188
|
+
verifiers/v1/taskset.py,sha256=RPFkpRkeEyPX557_mK-JwraBuUKsguiCUKSFyotTclU,4636
|
|
189
189
|
verifiers/v1/trace.py,sha256=9KqBeGaRUtWh6Gpdf6G1c7UZso0DB9Mjmtkzl6I_oYE,19458
|
|
190
190
|
verifiers/v1/types.py,sha256=5tZyG4r4bLJ9a14oybHy7bSCpt5X5A17eRwmIszuRo8,9118
|
|
191
191
|
verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
|
|
192
192
|
verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
|
|
193
193
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
194
|
-
verifiers/v1/cli/debug.py,sha256=
|
|
194
|
+
verifiers/v1/cli/debug.py,sha256=mRBPlYjc93o-4sooVSJXl_H3zALC2vhAl3EBzN-DpoM,11507
|
|
195
195
|
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
196
196
|
verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
|
|
197
197
|
verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
|
|
198
198
|
verifiers/v1/cli/replay.py,sha256=Jjyi61fl20xjGArdXTwZ3qKoSkC0iQ6WLSrvKkURdeI,10179
|
|
199
199
|
verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
|
|
200
200
|
verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
|
|
201
|
-
verifiers/v1/cli/validate.py,sha256=
|
|
201
|
+
verifiers/v1/cli/validate.py,sha256=lZ_wWnLgIWH5rUlMtwzvF3uS_cFZA4ZqBowpzdI4gos,10260
|
|
202
202
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
203
203
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
204
204
|
verifiers/v1/cli/dashboard/eval.py,sha256=i_Qn9enCU7aas-VTVAJocPNbTuHSUHThwQnls-3_kc4,34265
|
|
@@ -207,7 +207,7 @@ verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8m
|
|
|
207
207
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
208
208
|
verifiers/v1/cli/eval/main.py,sha256=zvbLy_ryn3S1fWoTmWL4QEReVO1Q_hIrgyK6Fp0q_ZM,5501
|
|
209
209
|
verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
|
|
210
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
210
|
+
verifiers/v1/cli/eval/runner.py,sha256=TKBnHSSs_knZcaHiH0ZjrF8DneEWlLp5b8O_SjQz_as,12130
|
|
211
211
|
verifiers/v1/clients/__init__.py,sha256=bHGS5jfrAZ-VBUeXpasXkkKdHkFgjOSN9KbOJXrnFtU,511
|
|
212
212
|
verifiers/v1/clients/client.py,sha256=9gZScQM6CVQq9RZ_3slVBEsik6tr56uOFSVNvQfMdFQ,3288
|
|
213
213
|
verifiers/v1/clients/config.py,sha256=QMR0PkpFLcLi_Zw6bVimigNjNiWeJ-4YvJ_JsSS9yGw,5228
|
|
@@ -217,12 +217,12 @@ verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxV
|
|
|
217
217
|
verifiers/v1/configs/agent.py,sha256=N2Lm0iZXFxfsSIiXUQ7f3RDZcoin3M7pJj6HVdCE5eo,3537
|
|
218
218
|
verifiers/v1/configs/env.py,sha256=4aUap3BJX3WCA2_fZ-jGPxF3khiVpORqmz_3eBT4j9w,7824
|
|
219
219
|
verifiers/v1/configs/harness.py,sha256=CG8hGbHl3rr6yawH3YYMAFl01M7Piiwf4w5oiw2Vagg,1564
|
|
220
|
-
verifiers/v1/configs/judge.py,sha256=
|
|
220
|
+
verifiers/v1/configs/judge.py,sha256=eEVtBGjAOmm1sSRd_UHVBpJEfUxMHxHDBFnvchmxs-w,2217
|
|
221
221
|
verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
|
|
222
222
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
223
223
|
verifiers/v1/configs/serve.py,sha256=cHHnFGQtqEwvaovdeoVMpCebTHgPXKUuIivN5IhxZR0,2596
|
|
224
224
|
verifiers/v1/configs/task.py,sha256=1ozZjfpahin7I_TqzC7f3xr7CHSthAdydegFPXW4d9k,1101
|
|
225
|
-
verifiers/v1/configs/taskset.py,sha256=
|
|
225
|
+
verifiers/v1/configs/taskset.py,sha256=o-tr1xkmGmYG10_ySG-Du_djWS-dN4h7cv14WSvJ4eY,851
|
|
226
226
|
verifiers/v1/configs/cli/__init__.py,sha256=3wBAONw5XkBPqbMjpHk5XBHlBVybR8JQ2JCPWijYBMU,444
|
|
227
227
|
verifiers/v1/configs/cli/debug.py,sha256=3lkKqmQFLJBcUZECChBHCrod5pbCKSmT47EhjILrxjI,2843
|
|
228
228
|
verifiers/v1/configs/cli/env.py,sha256=XgjRXYpPjga0ZYIHEESY7VuVvtE6pWfAzf6JBLeuTIc,2695
|
|
@@ -238,7 +238,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
|
|
|
238
238
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
239
239
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
240
240
|
verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
|
|
241
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
241
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=2g9yz3QfYUc_OZBkxzWOEKsGiT51siYMKB7PwvbCY2E,14878
|
|
242
242
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
243
243
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
244
244
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -246,11 +246,11 @@ verifiers/v1/envs/single_agent/env.py,sha256=lPs_a6VA7KoH1xqA3T8S32Jzb_d4zrx0_54
|
|
|
246
246
|
verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-EYgxOVx0s,118
|
|
247
247
|
verifiers/v1/envs/user_sim/env.py,sha256=9L_KXjzfkTnDeQpn4ftQagZyPkJ3yjcQeZqZaq7kxjE,4567
|
|
248
248
|
verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
|
|
249
|
-
verifiers/v1/gepa/adapter.py,sha256=
|
|
249
|
+
verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,6139
|
|
250
250
|
verifiers/v1/gepa/config.py,sha256=gFn8Re4GoeHJs2PZ-iw4jCEZ72AiM1lNgwaA0uvJlYw,4626
|
|
251
251
|
verifiers/v1/gepa/dataset.py,sha256=N2QYqUljHVuAdCbUuydEIlarvAWa6LFdGKu9qsOvjro,2186
|
|
252
252
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
253
|
-
verifiers/v1/gepa/runner.py,sha256
|
|
253
|
+
verifiers/v1/gepa/runner.py,sha256=-LZEt0lod6NKfyxg2dvOXZjfzTv3AWwendexd1De26c,6260
|
|
254
254
|
verifiers/v1/harnesses/__init__.py,sha256=3fgfAMzFrNgu7RIO_evZPxipPdiskiHun5EKsdS4y6k,1301
|
|
255
255
|
verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
|
|
256
256
|
verifiers/v1/harnesses/bash/harness.py,sha256=ZtjhsANuK3vXkf9mRCi8-sBMyPebJMBNHa58vTd2Fm8,5284
|
|
@@ -291,7 +291,7 @@ verifiers/v1/judges/rubric.py,sha256=kB0yWEtfC60zJBCVjOKBO00YtH8-AOofzPkuyccVTtQ
|
|
|
291
291
|
verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
|
|
292
292
|
verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
|
|
293
293
|
verifiers/v1/mcp/launch.py,sha256=lRtcl98BO7imC44bsxO2yljwRQR_KCvJ6P-bd_hS5L8,18444
|
|
294
|
-
verifiers/v1/mcp/server.py,sha256=
|
|
294
|
+
verifiers/v1/mcp/server.py,sha256=zS3-m-0IX3z8nVzUhQj391f8_OieE1nKr5S4Nj0dSSU,10987
|
|
295
295
|
verifiers/v1/mcp/toolset.py,sha256=tSPXgTNREZyaeCnoFVnDfXxFAj_pStZMPUIgyPrY_Dk,996
|
|
296
296
|
verifiers/v1/runtimes/__init__.py,sha256=TqNtJVahzQuLS4fgCVeT9xdOzTkp0hRAPOkl7eFVDUg,2011
|
|
297
297
|
verifiers/v1/runtimes/base.py,sha256=sTCiLo51yAU_T-Voc3BTH-FuzyEM0cGY05VzFE4DmYc,14720
|
|
@@ -311,7 +311,7 @@ verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRH
|
|
|
311
311
|
verifiers/v1/tasksets/harbor/taskset.py,sha256=2Iw1ZaPRhRjdbfClRLd0iqKWABjM7M525zFN4wuaeCk,13867
|
|
312
312
|
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
313
313
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
314
|
-
verifiers/v1/tasksets/lean/taskset.py,sha256=
|
|
314
|
+
verifiers/v1/tasksets/lean/taskset.py,sha256=Fcn4UgTcdjMYnJ4CThaFZ28Rz_QDx0e21vJdW_vp12g,9019
|
|
315
315
|
verifiers/v1/tasksets/openenv/__init__.py,sha256=G726meYYDplC3QE3AyjdSShq7FUcb4NkVqowuwPPuJk,303
|
|
316
316
|
verifiers/v1/tasksets/openenv/taskset.py,sha256=1Wc9-5eN2HMKAwL34yTNex3zYOvXNXibF5wCROoMbFI,6075
|
|
317
317
|
verifiers/v1/tasksets/textarena/__init__.py,sha256=Os2OlBY_pSH0B1DP32RitumgSLpq2LtcI7VDUHYPzRE,329
|
|
@@ -320,17 +320,17 @@ verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuF
|
|
|
320
320
|
verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
|
|
321
321
|
verifiers/v1/utils/compile.py,sha256=plpZYt-UD8q7Yx5pD7Rfy1w7Jer5TCBk2CJGHiNcPio,5696
|
|
322
322
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
323
|
-
verifiers/v1/utils/generic.py,sha256=
|
|
323
|
+
verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
|
|
324
324
|
verifiers/v1/utils/git.py,sha256=Mj2InIZlCAKWEWR9HPEMJX1OgkCVve5NjWgTCD4GmkM,4871
|
|
325
325
|
verifiers/v1/utils/image.py,sha256=OFw_wdwVtbdwa8uJ3X3vYfhpUAi0CTH5HMBupjVsRRQ,282
|
|
326
326
|
verifiers/v1/utils/install.py,sha256=fWNsyKrw_PyhC0Qhqv5Ri0adFm5Ddx4AVy_hbsra1GU,1390
|
|
327
327
|
verifiers/v1/utils/interrupt.py,sha256=F-KKhc5ndPJJfhd3SuMqyqhXhA32FhCRy5KWFJpEoM4,1179
|
|
328
328
|
verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y,2084
|
|
329
329
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
330
|
-
verifiers/v1/utils/sampling.py,sha256=
|
|
330
|
+
verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
|
|
331
331
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
332
|
-
verifiers-0.2.2.
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
335
|
-
verifiers-0.2.2.
|
|
336
|
-
verifiers-0.2.2.
|
|
332
|
+
verifiers-0.2.2.dev51.dist-info/METADATA,sha256=Ns-l7u4GlaZ2rco5UvfL-YtVj50LqCMzWiFX-zmXDEA,4545
|
|
333
|
+
verifiers-0.2.2.dev51.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
334
|
+
verifiers-0.2.2.dev51.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
335
|
+
verifiers-0.2.2.dev51.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
336
|
+
verifiers-0.2.2.dev51.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|