verifiers 0.3.2.dev151__py3-none-any.whl → 0.3.2.dev152__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +4 -0
- verifiers/v1/cli/debug.py +3 -6
- verifiers/v1/cli/eval/runner.py +3 -6
- verifiers/v1/cli/init.py +1 -1
- verifiers/v1/cli/validate.py +3 -6
- verifiers/v1/configs/cli/debug.py +4 -8
- verifiers/v1/configs/cli/eval.py +4 -8
- verifiers/v1/configs/cli/validate.py +4 -8
- verifiers/v1/configs/select.py +118 -0
- verifiers/v1/gepa/config.py +7 -6
- verifiers/v1/gepa/dataset.py +2 -2
- verifiers/v1/gepa/runner.py +2 -4
- verifiers/v1/task.py +9 -3
- verifiers/v1/taskset.py +158 -39
- verifiers/v1/tasksets/nemo_gym/taskset.py +0 -1
- verifiers/v1/tasksets/openenv/taskset.py +0 -1
- verifiers/v1/tasksets/textarena/taskset.py +0 -1
- {verifiers-0.3.2.dev151.dist-info → verifiers-0.3.2.dev152.dist-info}/METADATA +1 -1
- {verifiers-0.3.2.dev151.dist-info → verifiers-0.3.2.dev152.dist-info}/RECORD +22 -21
- {verifiers-0.3.2.dev151.dist-info → verifiers-0.3.2.dev152.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev151.dist-info → verifiers-0.3.2.dev152.dist-info}/entry_points.txt +0 -0
- {verifiers-0.3.2.dev151.dist-info → verifiers-0.3.2.dev152.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -23,6 +23,7 @@ from verifiers.v1.configs.env import EnvConfig, SharedEnvConfig, default_agent_h
|
|
|
23
23
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
24
24
|
from verifiers.v1.configs.judge import JudgeConfig, Judges
|
|
25
25
|
from verifiers.v1.configs.retries import RetryConfig
|
|
26
|
+
from verifiers.v1.configs.select import SelectCLIConfig, SelectConfig, TaskMatchConfig
|
|
26
27
|
from verifiers.v1.configs.serve import (
|
|
27
28
|
ElasticPoolConfig,
|
|
28
29
|
ServeConfig,
|
|
@@ -289,6 +290,9 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
289
290
|
# taskset / harness / runtime / environment
|
|
290
291
|
"Taskset",
|
|
291
292
|
"TaskConfig",
|
|
293
|
+
"SelectConfig",
|
|
294
|
+
"SelectCLIConfig",
|
|
295
|
+
"TaskMatchConfig",
|
|
292
296
|
"TasksetConfig",
|
|
293
297
|
"SharedTasksetConfig",
|
|
294
298
|
"DecoratedFunctionConfig",
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -269,15 +269,12 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
269
269
|
|
|
270
270
|
|
|
271
271
|
async def run_debug(config: DebugConfig) -> list[Trace]:
|
|
272
|
-
taskset = vf.load_taskset(config.taskset)
|
|
273
|
-
if
|
|
272
|
+
taskset = vf.load_taskset(config.taskset).select(config.select)
|
|
273
|
+
if not taskset.bounded:
|
|
274
274
|
raise ValueError(
|
|
275
275
|
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
276
276
|
)
|
|
277
|
-
|
|
278
|
-
if config.num_tasks is not None:
|
|
279
|
-
selected = selected.head(config.num_tasks)
|
|
280
|
-
tasks = list(selected)
|
|
277
|
+
tasks = list(taskset)
|
|
281
278
|
if isinstance(config.runtime, vf.SubprocessConfig) and any(
|
|
282
279
|
type(t).NEEDS_CONTAINER or t.data.image for t in tasks
|
|
283
280
|
):
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -82,15 +82,12 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
82
82
|
from verifiers.v1.utils.loaders import load_environment
|
|
83
83
|
|
|
84
84
|
env = load_environment(config.env)
|
|
85
|
-
taskset = env.taskset
|
|
86
|
-
if
|
|
85
|
+
taskset = env.taskset.select(config.select)
|
|
86
|
+
if not taskset.bounded:
|
|
87
87
|
raise ValueError(
|
|
88
88
|
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
89
89
|
)
|
|
90
|
-
|
|
91
|
-
if config.num_tasks is not None:
|
|
92
|
-
selected = selected.head(config.num_tasks)
|
|
93
|
-
tasks = list(selected)
|
|
90
|
+
tasks = list(taskset)
|
|
94
91
|
out = output_path(config)
|
|
95
92
|
# One (task, rollouts-to-run) pair per selected task; resume shrinks the counts.
|
|
96
93
|
plan = [(task, config.num_rollouts) for task in tasks]
|
verifiers/v1/cli/init.py
CHANGED
|
@@ -109,7 +109,7 @@ class {prefix}Taskset(vf.Taskset[{prefix}Task, {prefix}Config]):
|
|
|
109
109
|
def load(self) -> list[{prefix}Task]:
|
|
110
110
|
raise NotImplementedError(
|
|
111
111
|
"Return this taskset's tasks, e.g. "
|
|
112
|
-
"[{prefix}Task({prefix}Data(
|
|
112
|
+
"[{prefix}Task({prefix}Data(prompt=...), self.config.task) "
|
|
113
113
|
"for i in range(self.config.num_tasks)]."
|
|
114
114
|
)
|
|
115
115
|
'''
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -325,15 +325,12 @@ async def _validate_task(task: Task, config: ValidateConfig) -> ResultRow:
|
|
|
325
325
|
|
|
326
326
|
|
|
327
327
|
async def run_validate(config: ValidateConfig) -> list[dict]:
|
|
328
|
-
taskset = vf.load_taskset(config.taskset)
|
|
329
|
-
if
|
|
328
|
+
taskset = vf.load_taskset(config.taskset).select(config.select)
|
|
329
|
+
if not taskset.bounded:
|
|
330
330
|
raise ValueError(
|
|
331
331
|
f"{type(taskset).__name__} is infinite - bound the run with -n"
|
|
332
332
|
)
|
|
333
|
-
|
|
334
|
-
if config.num_tasks is not None:
|
|
335
|
-
selected = selected.head(config.num_tasks)
|
|
336
|
-
tasks = list(selected)
|
|
333
|
+
tasks = list(taskset)
|
|
337
334
|
if isinstance(config.runtime, vf.SubprocessConfig) and any(
|
|
338
335
|
type(t).NEEDS_CONTAINER or t.data.image for t in tasks
|
|
339
336
|
):
|
|
@@ -7,6 +7,7 @@ from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
|
7
7
|
from pydantic_config import BaseConfig
|
|
8
8
|
|
|
9
9
|
from verifiers.v1.configs.cli.validate import CheckTimeoutConfig
|
|
10
|
+
from verifiers.v1.configs.select import SelectCLIConfig
|
|
10
11
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
11
12
|
from verifiers.v1.runtimes import PrimeConfig, RuntimeConfig
|
|
12
13
|
|
|
@@ -15,6 +16,9 @@ class DebugConfig(BaseConfig):
|
|
|
15
16
|
uuid: str = Field(default_factory=lambda: str(uuid4()), exclude=True)
|
|
16
17
|
"""Auto-generated run id, used as the default output directory leaf."""
|
|
17
18
|
taskset: SerializeAsAny[TasksetConfig] = TasksetConfig()
|
|
19
|
+
select: SelectCLIConfig = SelectCLIConfig()
|
|
20
|
+
"""Which of the taskset's tasks to debug, under `--select.*` (`-n` sets
|
|
21
|
+
`select.limit`, `-s` sets `select.shuffle`)."""
|
|
18
22
|
runtime: RuntimeConfig = PrimeConfig()
|
|
19
23
|
"""Where each task's setup hook and debug action run."""
|
|
20
24
|
command: str | None = None
|
|
@@ -26,14 +30,6 @@ class DebugConfig(BaseConfig):
|
|
|
26
30
|
timeout: CheckTimeoutConfig = CheckTimeoutConfig()
|
|
27
31
|
"""Per-task stage timeouts: `--timeout.setup` for the `setup` hook, `--timeout.total`
|
|
28
32
|
for the debug command/script."""
|
|
29
|
-
num_tasks: int | None = Field(
|
|
30
|
-
None,
|
|
31
|
-
ge=1,
|
|
32
|
-
validation_alias=AliasChoices("num_tasks", "n", "num_examples", "batch_size"),
|
|
33
|
-
)
|
|
34
|
-
"""How many tasks to debug (None = all)."""
|
|
35
|
-
shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
|
|
36
|
-
"""Shuffle tasks before taking the first `num_tasks`."""
|
|
37
33
|
max_concurrent: int | None = Field(
|
|
38
34
|
128, validation_alias=AliasChoices("max_concurrent", "c")
|
|
39
35
|
)
|
verifiers/v1/configs/cli/eval.py
CHANGED
|
@@ -9,6 +9,7 @@ from pydantic_config import BaseConfig
|
|
|
9
9
|
from verifiers.v1.clients import ClientConfig, EvalClientConfig
|
|
10
10
|
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
11
11
|
from verifiers.v1.configs.env import EnvConfig
|
|
12
|
+
from verifiers.v1.configs.select import SelectCLIConfig
|
|
12
13
|
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
13
14
|
from verifiers.v1.types import SamplingConfig
|
|
14
15
|
|
|
@@ -84,12 +85,9 @@ class EvalConfig(BaseConfig):
|
|
|
84
85
|
"""Model id."""
|
|
85
86
|
client: ClientConfig = EvalClientConfig()
|
|
86
87
|
sampling: SamplingConfig = SamplingConfig()
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
validation_alias=AliasChoices("batch_size", "num_examples", "num_tasks", "n"),
|
|
91
|
-
)
|
|
92
|
-
"""How many tasks to evaluate (None = all)."""
|
|
88
|
+
select: SelectCLIConfig = SelectCLIConfig()
|
|
89
|
+
"""Which of the taskset's tasks to evaluate, under `--select.*` (`-n` sets
|
|
90
|
+
`select.limit`, `-s` sets `select.shuffle`)."""
|
|
93
91
|
num_rollouts: int = Field(
|
|
94
92
|
1,
|
|
95
93
|
ge=1,
|
|
@@ -98,8 +96,6 @@ class EvalConfig(BaseConfig):
|
|
|
98
96
|
),
|
|
99
97
|
)
|
|
100
98
|
"""Independent episodes per task — the trainer's group size."""
|
|
101
|
-
shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
|
|
102
|
-
"""Shuffle tasks before taking the first `num_tasks`."""
|
|
103
99
|
max_concurrent: int | None = Field(
|
|
104
100
|
128, ge=1, validation_alias=AliasChoices("max_concurrent", "c")
|
|
105
101
|
)
|
|
@@ -7,6 +7,7 @@ from pydantic import AliasChoices, Field, SerializeAsAny, model_validator
|
|
|
7
7
|
from pydantic_config import BaseConfig
|
|
8
8
|
|
|
9
9
|
from verifiers.v1.configs.cli.eval import RunConfig
|
|
10
|
+
from verifiers.v1.configs.select import SelectCLIConfig
|
|
10
11
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
11
12
|
from verifiers.v1.runtimes import PrimeConfig, RuntimeConfig
|
|
12
13
|
|
|
@@ -24,6 +25,9 @@ class ValidateConfig(BaseConfig):
|
|
|
24
25
|
"""Run identity: `run.name` auto-generates as `<taskset>--validate--<short-id>` and
|
|
25
26
|
names the run directory under `output_dir`."""
|
|
26
27
|
taskset: SerializeAsAny[TasksetConfig] = TasksetConfig()
|
|
28
|
+
select: SelectCLIConfig = SelectCLIConfig()
|
|
29
|
+
"""Which of the taskset's tasks to validate, under `--select.*` (`-n` sets
|
|
30
|
+
`select.limit`, `-s` sets `select.shuffle`)."""
|
|
27
31
|
runtime: RuntimeConfig = PrimeConfig()
|
|
28
32
|
"""Where each task's validation hooks run."""
|
|
29
33
|
timeout: CheckTimeoutConfig = CheckTimeoutConfig()
|
|
@@ -31,14 +35,6 @@ class ValidateConfig(BaseConfig):
|
|
|
31
35
|
"""Run only `Task.setup`."""
|
|
32
36
|
only_gold: bool = False
|
|
33
37
|
"""Run only `Task.setup` and `Task.validate`."""
|
|
34
|
-
num_tasks: int | None = Field(
|
|
35
|
-
None,
|
|
36
|
-
ge=1,
|
|
37
|
-
validation_alias=AliasChoices("num_tasks", "n", "num_examples", "batch_size"),
|
|
38
|
-
)
|
|
39
|
-
"""How many tasks to validate (None = all)."""
|
|
40
|
-
shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
|
|
41
|
-
"""Shuffle tasks before taking the first `num_tasks`."""
|
|
42
38
|
max_concurrent: int | None = Field(
|
|
43
39
|
128, validation_alias=AliasChoices("max_concurrent", "c")
|
|
44
40
|
)
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Which of a taskset's tasks a run uses: `Taskset.select(SelectConfig)`, under
|
|
2
|
+
`--select.*` on the eval, debug, validate and GEPA CLIs (`SelectCLIConfig`)."""
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
import sys
|
|
6
|
+
|
|
7
|
+
from pydantic import AliasChoices, Field, field_validator
|
|
8
|
+
from pydantic_config import BaseConfig
|
|
9
|
+
|
|
10
|
+
IDX_RANGE = re.compile(r"(\d*):(\d*)(?::(\d*))?")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class TaskMatchConfig(BaseConfig):
|
|
14
|
+
"""Tasks named by position or identity; a task matches when any list names it."""
|
|
15
|
+
|
|
16
|
+
idx: list[int | str] = Field(default_factory=list)
|
|
17
|
+
"""Positions in the taskset's `load()` stream (`TaskData.idx`): ints and Python
|
|
18
|
+
slices `start:stop:step`, each part optional (`"100:"`, `":50"`, `"::2"`).
|
|
19
|
+
Negative positions are not supported. A comma-separated string also works
|
|
20
|
+
(`"0:10,17"`)."""
|
|
21
|
+
ids: list[str] = Field(default_factory=list)
|
|
22
|
+
"""`TaskData.id` values."""
|
|
23
|
+
keys: list[str] = Field(default_factory=list)
|
|
24
|
+
"""`Task.key` values, as recorded on each trace (`task.key`)."""
|
|
25
|
+
names: list[str] = Field(default_factory=list)
|
|
26
|
+
"""`TaskData.name` values."""
|
|
27
|
+
|
|
28
|
+
@field_validator("idx", mode="before")
|
|
29
|
+
@classmethod
|
|
30
|
+
def _parse_idx(cls, value):
|
|
31
|
+
if isinstance(value, (int, str)):
|
|
32
|
+
value = [value]
|
|
33
|
+
items: list[int | str] = []
|
|
34
|
+
for item in value:
|
|
35
|
+
parts = item.split(",") if isinstance(item, str) else [item]
|
|
36
|
+
items.extend(_parse_idx_item(part) for part in parts)
|
|
37
|
+
return items
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def empty(self) -> bool:
|
|
41
|
+
return not (self.idx or self.ids or self.keys or self.names)
|
|
42
|
+
|
|
43
|
+
def idx_ranges(self) -> list[range]:
|
|
44
|
+
"""`idx` as ranges; an open-ended slice stops at `sys.maxsize`."""
|
|
45
|
+
ranges: list[range] = []
|
|
46
|
+
for item in self.idx:
|
|
47
|
+
if isinstance(item, int):
|
|
48
|
+
ranges.append(range(item, item + 1))
|
|
49
|
+
continue
|
|
50
|
+
start, stop, step = (item.split(":") + [""])[:3]
|
|
51
|
+
ranges.append(
|
|
52
|
+
range(int(start or 0), int(stop or sys.maxsize), int(step or 1))
|
|
53
|
+
)
|
|
54
|
+
return ranges
|
|
55
|
+
|
|
56
|
+
@property
|
|
57
|
+
def idx_tail(self) -> bool:
|
|
58
|
+
"""Whether an `idx` slice names every position from some start on."""
|
|
59
|
+
return any(r.stop == sys.maxsize and r.step == 1 for r in self.idx_ranges())
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def idx_stop(self) -> int | None:
|
|
63
|
+
"""One past the last position this can match, when it names only closed `idx`
|
|
64
|
+
ranges; None when a match could lie anywhere in the stream."""
|
|
65
|
+
ranges = self.idx_ranges()
|
|
66
|
+
if not ranges or self.ids or self.keys or self.names:
|
|
67
|
+
return None
|
|
68
|
+
stop = max(r.stop for r in ranges)
|
|
69
|
+
return None if stop == sys.maxsize else stop
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse_idx_item(item: int | str) -> int | str:
|
|
73
|
+
if isinstance(item, int):
|
|
74
|
+
if item < 0:
|
|
75
|
+
raise ValueError(f"idx {item} is negative")
|
|
76
|
+
return item
|
|
77
|
+
text = item.strip()
|
|
78
|
+
if text.isdigit():
|
|
79
|
+
return int(text)
|
|
80
|
+
match = IDX_RANGE.fullmatch(text)
|
|
81
|
+
if match is None:
|
|
82
|
+
raise ValueError(f"idx {item!r} is not an int or a 'start:stop:step' slice")
|
|
83
|
+
start, stop, step = match.groups()
|
|
84
|
+
if step and int(step) == 0:
|
|
85
|
+
raise ValueError(f"idx slice {item!r} has step 0")
|
|
86
|
+
if stop and int(start or 0) >= int(stop):
|
|
87
|
+
raise ValueError(f"idx range {item!r} is empty")
|
|
88
|
+
return text
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class SelectConfig(BaseConfig):
|
|
92
|
+
"""The steps apply in a fixed order: `include` keeps tasks, `exclude` drops them,
|
|
93
|
+
then `shuffle`, `skip` and `limit` pick from what is left."""
|
|
94
|
+
|
|
95
|
+
include: TaskMatchConfig = TaskMatchConfig()
|
|
96
|
+
"""Keep only the tasks this names. Empty keeps all."""
|
|
97
|
+
exclude: TaskMatchConfig = TaskMatchConfig()
|
|
98
|
+
"""Drop the tasks this names."""
|
|
99
|
+
shuffle: bool = False
|
|
100
|
+
"""Shuffle the kept tasks under `seed`. Needs a finite stream: an infinite taskset
|
|
101
|
+
must be bounded by closed `include.idx` ranges first."""
|
|
102
|
+
seed: int = 0
|
|
103
|
+
"""Seed for `shuffle`, fixed so runs select the same tasks."""
|
|
104
|
+
skip: int = Field(0, ge=0)
|
|
105
|
+
"""Drop this many tasks after the shuffle."""
|
|
106
|
+
limit: int | None = Field(None, ge=1)
|
|
107
|
+
"""Take at most this many tasks after `skip` (None = all)."""
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
class SelectCLIConfig(SelectConfig):
|
|
111
|
+
"""`SelectConfig` with CLI short flags, for the one `select` block of an entrypoint:
|
|
112
|
+
`-n` sets `limit` and `-s` sets `shuffle`."""
|
|
113
|
+
|
|
114
|
+
shuffle: bool = Field(False, validation_alias=AliasChoices("shuffle", "s"))
|
|
115
|
+
"""Shuffle the kept tasks under `seed` (`-s`). Needs a finite stream: an infinite
|
|
116
|
+
taskset must be bounded by closed `include.idx` ranges first."""
|
|
117
|
+
limit: int | None = Field(None, ge=1, validation_alias=AliasChoices("limit", "n"))
|
|
118
|
+
"""Take at most this many tasks after `skip` (`-n`; None = all)."""
|
verifiers/v1/gepa/config.py
CHANGED
|
@@ -17,6 +17,7 @@ from verifiers.v1.clients import EvalClientConfig
|
|
|
17
17
|
from verifiers.v1.configs.cli.env import narrowed_env_annotation, resolve_env_field
|
|
18
18
|
from verifiers.v1.configs.cli.eval import RunConfig, default_run_name
|
|
19
19
|
from verifiers.v1.configs.env import EnvConfig
|
|
20
|
+
from verifiers.v1.configs.select import SelectCLIConfig
|
|
20
21
|
from verifiers.v1.envs.single_agent import SingleAgentEnvConfig
|
|
21
22
|
from verifiers.v1.types import SamplingConfig
|
|
22
23
|
|
|
@@ -49,17 +50,17 @@ class GEPAConfig(BaseConfig):
|
|
|
49
50
|
reflection_client: EvalClientConfig | None = None
|
|
50
51
|
"""Endpoint for `reflection_model`. None = reuse `client`."""
|
|
51
52
|
|
|
53
|
+
select: SelectCLIConfig = SelectCLIConfig()
|
|
54
|
+
"""Which of the taskset's tasks GEPA splits into train/val, under `--select.*`
|
|
55
|
+
(`-n` sets `select.limit`, `-s` sets `select.shuffle`)."""
|
|
52
56
|
num_train: int = Field(100, ge=1)
|
|
53
57
|
"""Tasks reserved for reflection minibatches (GEPA never scores the full trainset at once)."""
|
|
54
58
|
num_val: int = Field(50, ge=1)
|
|
55
59
|
"""Tasks held out to score each candidate system prompt for the pareto frontier."""
|
|
56
|
-
shuffle: bool = Field(True, validation_alias=AliasChoices("shuffle", "s"))
|
|
57
|
-
"""Shuffle tasks before splitting into train/val — v1 tasksets have no generic train/val
|
|
58
|
-
split, so GEPA carves one out of `Taskset.select` the way `run_eval` samples (fixed
|
|
59
|
-
seed, so the split is reproducible across runs)."""
|
|
60
60
|
seed: int = 0
|
|
61
|
-
"""Seed for GEPA's optimizer (candidate selection / minibatch sampling).
|
|
62
|
-
|
|
61
|
+
"""Seed for GEPA's optimizer (candidate selection / minibatch sampling). The train/val
|
|
62
|
+
split comes from `select` (`-s` shuffles it under `select.seed`), so this doesn't
|
|
63
|
+
change it."""
|
|
63
64
|
|
|
64
65
|
max_total_rollouts: int = Field(500)
|
|
65
66
|
"""Total rollouts GEPA may spend across the whole optimization run."""
|
verifiers/v1/gepa/dataset.py
CHANGED
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
v1 tasksets have no generic train/val split concept (`TasksetConfig` has no `split` field;
|
|
4
4
|
individual tasksets define ad hoc ones inconsistently), so GEPA carves one out of the tasks
|
|
5
|
-
`
|
|
6
|
-
|
|
5
|
+
`select` yields (reproducible across runs like every other entrypoint): two
|
|
6
|
+
disjoint slices.
|
|
7
7
|
"""
|
|
8
8
|
|
|
9
9
|
from verifiers.v1.task import Task
|
verifiers/v1/gepa/runner.py
CHANGED
|
@@ -38,10 +38,8 @@ class _GEPALog:
|
|
|
38
38
|
|
|
39
39
|
def run_gepa(env: Env, config: GEPAConfig) -> GEPAResult:
|
|
40
40
|
logger.info("gepa config:\n%s", config.model_dump_json(indent=2))
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
taskset = env.taskset.shuffle() if config.shuffle else env.taskset
|
|
44
|
-
all_tasks = list(taskset.head(config.num_train + config.num_val))
|
|
41
|
+
taskset = env.taskset.select(config.select)
|
|
42
|
+
all_tasks = list(taskset.take(config.num_train + config.num_val))
|
|
45
43
|
train_tasks, val_tasks = split_tasks(all_tasks, config.num_train, config.num_val)
|
|
46
44
|
selected_tasks = [*train_tasks, *val_tasks]
|
|
47
45
|
# Seed from the tasks GEPA actually evaluates (train ∪ val), not the full pre-split pool —
|
verifiers/v1/task.py
CHANGED
|
@@ -82,7 +82,9 @@ class TaskData(BaseModel):
|
|
|
82
82
|
model_config = ConfigDict(frozen=True)
|
|
83
83
|
|
|
84
84
|
idx: int | None = None
|
|
85
|
-
"""
|
|
85
|
+
"""Task index in the taskset's `load()` stream - automatically set."""
|
|
86
|
+
id: str | None = None
|
|
87
|
+
"""Optional durable id from the task's source, e.g. a dataset's instance id."""
|
|
86
88
|
name: str | None = None
|
|
87
89
|
"""Optional human-readable task name."""
|
|
88
90
|
description: str | None = None
|
|
@@ -163,11 +165,15 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
163
165
|
"""
|
|
164
166
|
return self.hash
|
|
165
167
|
|
|
166
|
-
def
|
|
168
|
+
def with_data(self, **update: Any) -> Self:
|
|
169
|
+
"""A shallow copy whose data has `update` applied."""
|
|
167
170
|
clone = copy.copy(self)
|
|
168
|
-
clone.data = self.data.model_copy(update=
|
|
171
|
+
clone.data = self.data.model_copy(update=update)
|
|
169
172
|
return clone
|
|
170
173
|
|
|
174
|
+
def with_system_prompt(self, system_prompt: str) -> Self:
|
|
175
|
+
return self.with_data(system_prompt=system_prompt)
|
|
176
|
+
|
|
171
177
|
def runtime_env(self) -> dict[str, str]:
|
|
172
178
|
"""Live-only process environment; unlike TaskData, it is not traced."""
|
|
173
179
|
return {}
|
verifiers/v1/taskset.py
CHANGED
|
@@ -4,25 +4,34 @@ A `Taskset` is the data half of an environment: config in, tasks out. `load()` i
|
|
|
4
4
|
the main hook that builds each task:
|
|
5
5
|
|
|
6
6
|
def load(self) -> Iterable[MyTask]:
|
|
7
|
-
for
|
|
8
|
-
yield MyTask(MyData(
|
|
7
|
+
for row in ...:
|
|
8
|
+
yield MyTask(MyData(prompt=..., ...), self.config.task)
|
|
9
9
|
|
|
10
10
|
`load` may also be a generator for infinite tasksets. There is a one-to-one
|
|
11
11
|
mapping between taskset and task type, i.e. a taskset may only yield one task
|
|
12
|
-
type.
|
|
12
|
+
type. Iterating the taskset sets each task's `idx` to its position in `load()`.
|
|
13
|
+
|
|
14
|
+
Lazy views pick which tasks an iteration yields, chainable in any order:
|
|
15
|
+
|
|
16
|
+
taskset.include(idx=["0:100"]).exclude(names=["broken"]).shuffle(seed=0).take(5)
|
|
17
|
+
|
|
18
|
+
`select(SelectConfig)` chains them in the config's fixed order; the eval, debug,
|
|
19
|
+
validate and GEPA entrypoints and prime-rl all select through it.
|
|
13
20
|
"""
|
|
14
21
|
|
|
15
22
|
from __future__ import annotations
|
|
16
23
|
|
|
17
24
|
import copy
|
|
18
25
|
import itertools
|
|
26
|
+
import logging
|
|
19
27
|
import random
|
|
20
28
|
from abc import ABC, abstractmethod
|
|
21
29
|
from collections.abc import Callable, Iterable, Iterator
|
|
22
|
-
from typing import TYPE_CHECKING, Generic, Self
|
|
30
|
+
from typing import TYPE_CHECKING, Any, Generic, Self
|
|
23
31
|
|
|
24
32
|
from typing_extensions import TypeVar
|
|
25
33
|
|
|
34
|
+
from verifiers.v1.configs.select import SelectConfig, TaskMatchConfig
|
|
26
35
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
27
36
|
from verifiers.v1.task import Task, TaskT
|
|
28
37
|
from verifiers.v1.utils.generic import concrete_type
|
|
@@ -30,69 +39,138 @@ from verifiers.v1.utils.generic import concrete_type
|
|
|
30
39
|
if TYPE_CHECKING:
|
|
31
40
|
from verifiers.v1.mcp import Toolset
|
|
32
41
|
|
|
33
|
-
|
|
42
|
+
logger = logging.getLogger(__name__)
|
|
34
43
|
|
|
35
44
|
TasksetConfigT = TypeVar("TasksetConfigT", bound=TasksetConfig, default=TasksetConfig)
|
|
36
45
|
|
|
46
|
+
Transform = Callable[[Iterator[Any]], Iterator[Any]]
|
|
47
|
+
|
|
37
48
|
|
|
38
49
|
class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
39
50
|
INFINITE: bool = False
|
|
40
|
-
"""Whether
|
|
41
|
-
|
|
51
|
+
"""Whether `load()` yields tasks forever. A view can still bound the iteration
|
|
52
|
+
(see `bounded`)."""
|
|
53
|
+
|
|
54
|
+
transform: Transform | None = None
|
|
55
|
+
"""Iteration transform carried by the views (see `_view`)."""
|
|
56
|
+
_bounded: bool | None = None
|
|
57
|
+
_ordered: bool = True
|
|
58
|
+
"""Whether the view still yields tasks in `load()` order (no shuffle yet)."""
|
|
42
59
|
|
|
43
60
|
def __init__(self, config: TasksetConfigT) -> None:
|
|
44
61
|
self.config = config
|
|
45
62
|
override = config.system_prompt
|
|
46
63
|
self.system_prompt = override.read_text() if override is not None else None
|
|
47
|
-
self.transform: Callable[[Iterator[TaskT]], Iterator[TaskT]] | None = None
|
|
48
|
-
"""Iteration transform carried by `head`/`shuffle` views (see `view`)."""
|
|
49
64
|
|
|
50
65
|
@abstractmethod
|
|
51
66
|
def load(self) -> Iterable[TaskT]:
|
|
52
67
|
"""Build and yield the taskset's tasks; may be a generator (see module doc)."""
|
|
53
68
|
|
|
69
|
+
@property
|
|
70
|
+
def bounded(self) -> bool:
|
|
71
|
+
"""Whether iterating ends: `load()` is finite, or a view caps it (`take`, or
|
|
72
|
+
`include` with only closed `idx` ranges)."""
|
|
73
|
+
return not self.INFINITE if self._bounded is None else self._bounded
|
|
74
|
+
|
|
54
75
|
def __iter__(self) -> Iterator[TaskT]:
|
|
55
|
-
"""Lazily iterate `load()` with
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
76
|
+
"""Lazily iterate `load()` with each task's `idx` set to its position and the
|
|
77
|
+
config-layer system prompt applied, then the views' transform. The views see
|
|
78
|
+
the final task data, so a `keys` match compares the keys that traces record.
|
|
79
|
+
This is the read path; `load` is the subclass hook."""
|
|
80
|
+
update = (
|
|
81
|
+
{} if self.system_prompt is None else {"system_prompt": self.system_prompt}
|
|
61
82
|
)
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
def view(self, transform: Callable[[Iterator[TaskT]], Iterator[TaskT]]) -> Self:
|
|
65
|
-
"""A shallow copy of this taskset iterating through `transform`, composed
|
|
66
|
-
onto any transform this taskset already carries."""
|
|
67
|
-
clone = copy.copy(self)
|
|
68
|
-
prev = self.transform
|
|
69
|
-
clone.transform = (
|
|
70
|
-
transform if prev is None else lambda tasks: transform(prev(tasks))
|
|
83
|
+
tasks: Iterator[TaskT] = (
|
|
84
|
+
task.with_data(idx=idx, **update) for idx, task in enumerate(self.load())
|
|
71
85
|
)
|
|
72
|
-
return
|
|
73
|
-
|
|
74
|
-
def
|
|
75
|
-
"""A
|
|
76
|
-
|
|
77
|
-
|
|
86
|
+
return tasks if self.transform is None else self.transform(tasks)
|
|
87
|
+
|
|
88
|
+
def include(self, **match: Any) -> Self:
|
|
89
|
+
"""A view keeping only the tasks `match` names by `idx`, `ids`, `keys` or
|
|
90
|
+
`names` (see `TaskMatchConfig`). An identity match on an unbounded view could
|
|
91
|
+
read forever waiting for a match, so it raises there."""
|
|
92
|
+
config = TaskMatchConfig(**match)
|
|
93
|
+
if not self.bounded and (config.ids or config.keys or config.names):
|
|
94
|
+
raise ValueError(
|
|
95
|
+
f"{type(self).__name__} is infinite - include by ids, keys or names "
|
|
96
|
+
"may never end; bound it first with closed include idx ranges"
|
|
97
|
+
)
|
|
98
|
+
# Positions arrive in increasing order until a shuffle, so reading can stop
|
|
99
|
+
# past the last closed idx range.
|
|
100
|
+
stop = config.idx_stop if self._ordered else None
|
|
101
|
+
view = self._view(lambda tasks: _match(tasks, config, True, "include", stop))
|
|
102
|
+
if config.idx_stop is not None:
|
|
103
|
+
view._bounded = True
|
|
78
104
|
return view
|
|
79
105
|
|
|
80
|
-
def
|
|
81
|
-
"""A
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
106
|
+
def exclude(self, **match: Any) -> Self:
|
|
107
|
+
"""A view dropping the tasks `match` names (see `include`). Dropping every
|
|
108
|
+
position from some start on leaves an unbounded view nothing to yield past
|
|
109
|
+
it, so that raises there."""
|
|
110
|
+
config = TaskMatchConfig(**match)
|
|
111
|
+
if not self.bounded and config.idx_tail:
|
|
112
|
+
raise ValueError(
|
|
113
|
+
f"{type(self).__name__} is infinite - excluding an open idx range "
|
|
114
|
+
"drops every task after its start; exclude a closed range instead"
|
|
115
|
+
)
|
|
116
|
+
return self._view(lambda tasks: _match(tasks, config, False, "exclude", None))
|
|
117
|
+
|
|
118
|
+
def shuffle(self, seed: int = 0) -> Self:
|
|
119
|
+
"""A shuffled view under `seed` (materializes on iteration); raises on an
|
|
120
|
+
unbounded view — bound it first with `take` or closed `include` idx ranges.
|
|
121
|
+
`select` applies `limit` after the shuffle, so there only closed
|
|
122
|
+
`include.idx` ranges can bound it."""
|
|
123
|
+
if not self.bounded:
|
|
85
124
|
raise ValueError(
|
|
86
|
-
f"{type(self).__name__} is infinite - cannot shuffle; "
|
|
87
|
-
"
|
|
125
|
+
f"{type(self).__name__} is infinite - cannot shuffle; bound it first "
|
|
126
|
+
"with closed include idx ranges (e.g. --select.include.idx 0:1000), "
|
|
127
|
+
"or with take(n) before shuffle() in Python"
|
|
88
128
|
)
|
|
89
129
|
|
|
90
130
|
def shuffled(tasks: Iterator[TaskT]) -> Iterator[TaskT]:
|
|
91
131
|
materialized = list(tasks)
|
|
92
|
-
random.Random(
|
|
132
|
+
random.Random(seed).shuffle(materialized)
|
|
93
133
|
return iter(materialized)
|
|
94
134
|
|
|
95
|
-
|
|
135
|
+
view = self._view(shuffled)
|
|
136
|
+
view._ordered = False
|
|
137
|
+
return view
|
|
138
|
+
|
|
139
|
+
def skip(self, num_tasks: int) -> Self:
|
|
140
|
+
"""A view without the first `num_tasks` tasks."""
|
|
141
|
+
return self._view(lambda tasks: itertools.islice(tasks, num_tasks, None))
|
|
142
|
+
|
|
143
|
+
def take(self, num_tasks: int) -> Self:
|
|
144
|
+
"""A lazy, always-finite view of the first `num_tasks` tasks."""
|
|
145
|
+
view = self._view(lambda tasks: itertools.islice(tasks, num_tasks))
|
|
146
|
+
view._bounded = True
|
|
147
|
+
return view
|
|
148
|
+
|
|
149
|
+
def select(self, config: SelectConfig) -> Self:
|
|
150
|
+
"""The views `config` describes, chained in its fixed order: `include`,
|
|
151
|
+
`exclude`, `shuffle`, `skip`, `take`."""
|
|
152
|
+
view = self
|
|
153
|
+
if not config.include.empty:
|
|
154
|
+
view = view.include(**config.include.model_dump())
|
|
155
|
+
if not config.exclude.empty:
|
|
156
|
+
view = view.exclude(**config.exclude.model_dump())
|
|
157
|
+
if config.shuffle:
|
|
158
|
+
view = view.shuffle(config.seed)
|
|
159
|
+
if config.skip:
|
|
160
|
+
view = view.skip(config.skip)
|
|
161
|
+
if config.limit is not None:
|
|
162
|
+
view = view.take(config.limit)
|
|
163
|
+
return view
|
|
164
|
+
|
|
165
|
+
def _view(self, transform: Transform) -> Self:
|
|
166
|
+
"""A shallow copy of this taskset iterating through `transform`, composed
|
|
167
|
+
onto any transform this taskset already carries."""
|
|
168
|
+
clone = copy.copy(self)
|
|
169
|
+
prev = self.transform
|
|
170
|
+
clone.transform = (
|
|
171
|
+
transform if prev is None else lambda tasks: transform(prev(tasks))
|
|
172
|
+
)
|
|
173
|
+
return clone
|
|
96
174
|
|
|
97
175
|
@classmethod
|
|
98
176
|
def task_type(cls) -> type[Task]:
|
|
@@ -109,3 +187,44 @@ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
|
109
187
|
return [SearchToolset(config.tools)]
|
|
110
188
|
"""
|
|
111
189
|
return []
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _match(
|
|
193
|
+
tasks: Iterator[TaskT],
|
|
194
|
+
match: TaskMatchConfig,
|
|
195
|
+
keep: bool,
|
|
196
|
+
label: str,
|
|
197
|
+
stop: int | None,
|
|
198
|
+
) -> Iterator[TaskT]:
|
|
199
|
+
"""The tasks `match` names (`keep`) or does not name, reading no further than
|
|
200
|
+
position `stop`. Once reading ends, warns about entries that matched no task, as
|
|
201
|
+
these are usually typos."""
|
|
202
|
+
ranges = match.idx_ranges()
|
|
203
|
+
wanted = {"ids": set(match.ids), "keys": set(match.keys), "names": set(match.names)}
|
|
204
|
+
found: dict[str, set[str]] = {field: set() for field in wanted}
|
|
205
|
+
hit_ranges: set[int] = set()
|
|
206
|
+
for task in tasks:
|
|
207
|
+
idx = task.data.idx
|
|
208
|
+
if stop is not None and idx >= stop:
|
|
209
|
+
break
|
|
210
|
+
in_ranges = {i for i, r in enumerate(ranges) if idx in r}
|
|
211
|
+
hit_ranges |= in_ranges
|
|
212
|
+
values = {
|
|
213
|
+
"ids": task.data.id,
|
|
214
|
+
"keys": task.key if match.keys else None,
|
|
215
|
+
"names": task.data.name,
|
|
216
|
+
}
|
|
217
|
+
hits = [f for f, value in values.items() if value in wanted[f]]
|
|
218
|
+
for field in hits:
|
|
219
|
+
found[field].add(values[field])
|
|
220
|
+
if bool(hits or in_ranges) == keep:
|
|
221
|
+
yield task
|
|
222
|
+
if stop is not None and idx + 1 >= stop:
|
|
223
|
+
break
|
|
224
|
+
missing: dict[str, list] = {
|
|
225
|
+
f: sorted(wanted[f] - found[f]) for f in wanted if wanted[f] - found[f]
|
|
226
|
+
}
|
|
227
|
+
if unmatched := [match.idx[i] for i in range(len(ranges)) if i not in hit_ranges]:
|
|
228
|
+
missing["idx"] = unmatched
|
|
229
|
+
if missing:
|
|
230
|
+
logger.warning("%s matched no task for %s", label, missing)
|
|
@@ -107,7 +107,6 @@ class TextArenaTaskset(vf.Taskset[TextArenaTask, TextArenaConfig]):
|
|
|
107
107
|
for i in itertools.count():
|
|
108
108
|
yield TextArenaTask(
|
|
109
109
|
TextArenaData(
|
|
110
|
-
idx=i,
|
|
111
110
|
name=f"{self.config.game}#{i}",
|
|
112
111
|
prompt=observation(i) if seed_specific else first,
|
|
113
112
|
system_prompt=SYSTEM_PROMPT,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.3.2.
|
|
3
|
+
Version: 0.3.2.dev152
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
verifiers/__init__.py,sha256=F5IVMw6mhGaPBMCw7QKk0la_r5mMwydP2hdTTdQucik,352
|
|
2
|
-
verifiers/v1/__init__.py,sha256=
|
|
2
|
+
verifiers/v1/__init__.py,sha256=6t2FsosAjmyfJz6qS8tDQ2gWCvQfyRzqk1mYwv5_-EQ,8738
|
|
3
3
|
verifiers/v1/agent.py,sha256=yBCCBe5tVTeQ8OZDlVSZLw_IDvCmEQSnF3wQ1BZcT14,32342
|
|
4
4
|
verifiers/v1/env.py,sha256=GqrVxxrvv-9WitXhD8Rre0tKQgnQWeD8mTM2gDHMVBs,18040
|
|
5
5
|
verifiers/v1/episode.py,sha256=UPtA02aGWA-fY8cJKEuYCMtQVnPN1C0BSN13LggNT8U,5986
|
|
@@ -11,21 +11,21 @@ verifiers/v1/rollout.py,sha256=WiMCJyAypsR1OaXvzJQx7G-T67v__AnnDNoXotuI_VE,25444
|
|
|
11
11
|
verifiers/v1/semantic.py,sha256=VQ_buw6Ya6QtRoh_fiYI5Avs5RmIQCfWH0H0bQc4aoc,4704
|
|
12
12
|
verifiers/v1/session.py,sha256=qVMgKmqkItemR3uYUDAFuDz7huP593kHQbge8FlRba8,28420
|
|
13
13
|
verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
|
|
14
|
-
verifiers/v1/task.py,sha256=
|
|
15
|
-
verifiers/v1/taskset.py,sha256=
|
|
14
|
+
verifiers/v1/task.py,sha256=6-Tw0iBLsc4hinvVPl_b2EUX8unAAnxtHMXNSIHXsnU,12886
|
|
15
|
+
verifiers/v1/taskset.py,sha256=LE_3ogL7itU2Amk-N20cHkZ7Ev6MznJ0HpC8B1xb4w0,9482
|
|
16
16
|
verifiers/v1/trace.py,sha256=tgzkft4nKF4QGeGXwoqkfXjaUbG65Nn1oBFzMUFp-u0,32234
|
|
17
17
|
verifiers/v1/types.py,sha256=-In5plgB3mQLqtRKHC3QDCPDvuruAqVd8WIjKrsGB_I,10578
|
|
18
18
|
verifiers/v1/acp/__init__.py,sha256=9RySmxFeEMT2XSJy5wYevqEhQdul2jHl9f5XAribG3A,13631
|
|
19
19
|
verifiers/v1/acp/runner.py,sha256=zPo-2ZXmFMmQchhD7nNzlZUrabG09qiBvpB67KVGm3M,14095
|
|
20
20
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
21
|
-
verifiers/v1/cli/debug.py,sha256=
|
|
21
|
+
verifiers/v1/cli/debug.py,sha256=OgzXCW9HaTNruZil4f70WP5tR6toKliKaVvV98C5qzE,11665
|
|
22
22
|
verifiers/v1/cli/gepa.py,sha256=IyS3YZCCHVydb-OyPeGdtCYsz6OYqXXddXqYq8RLucM,4374
|
|
23
|
-
verifiers/v1/cli/init.py,sha256=
|
|
23
|
+
verifiers/v1/cli/init.py,sha256=k1wawdQBquiiepPrb0enBT-vdKvfSievgs4CiwSdFUc,7601
|
|
24
24
|
verifiers/v1/cli/output.py,sha256=4FRuQ6UIZCigMfUUO2nCFFJKxls3DtWDXswMCVTxjIA,5010
|
|
25
25
|
verifiers/v1/cli/replay.py,sha256=LkeaV8Sxa7kHue9JKHDg0V_N6KXOcjN_r6RmeQVcxfk,10396
|
|
26
26
|
verifiers/v1/cli/resolve.py,sha256=Q1EHM7wWQo0YkwJA898qtZbYLaIkjIVq3Q7kUp2YIxs,4346
|
|
27
27
|
verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,470
|
|
28
|
-
verifiers/v1/cli/validate.py,sha256=
|
|
28
|
+
verifiers/v1/cli/validate.py,sha256=rrEHbGy1RkHt2dnxBqg85VXyB27f3rGGZxlPzZzgmzI,17168
|
|
29
29
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
30
30
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
31
31
|
verifiers/v1/cli/dashboard/eval.py,sha256=69vKmLaEY1ll1wahjRGvO1_7QsUQ1URtZOG7ZawvRCw,38347
|
|
@@ -35,7 +35,7 @@ verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3h
|
|
|
35
35
|
verifiers/v1/cli/eval/hint.py,sha256=NXP2_JYrQHguqr68LxOJG3lYAsue_VNoepyTVPn-N4E,256
|
|
36
36
|
verifiers/v1/cli/eval/main.py,sha256=46aHpha6yhFzbqUrRogC5wC2XAwdZcpp-4siqLpUExY,5984
|
|
37
37
|
verifiers/v1/cli/eval/resume.py,sha256=cmFdxLwyU66lyAuXb7U8mNBB-8NLX41JdCg1EuU4w9g,3818
|
|
38
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
38
|
+
verifiers/v1/cli/eval/runner.py,sha256=eGnGZUNqt9gG0PK8icesOXbIcIvXuN5157emOM57cd8,6830
|
|
39
39
|
verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
|
|
40
40
|
verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
|
|
41
41
|
verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
|
|
@@ -49,16 +49,17 @@ verifiers/v1/configs/harness.py,sha256=ZrR-Egzfft95oMoqQ78bAq9TqP882Q68M_47Ajw9l
|
|
|
49
49
|
verifiers/v1/configs/judge.py,sha256=WEwA8D4X5JSaIP3c-E2wPnvgByWeT6JRAn1d8gfiQ40,2142
|
|
50
50
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
51
51
|
verifiers/v1/configs/runtime.py,sha256=ZEGx5JEgkMOLBQhiKAg6nzqKMhUu4W9tCE3AR--X8HY,5412
|
|
52
|
+
verifiers/v1/configs/select.py,sha256=_HdcepQaJ1sDCYcs7Vqx72npDwHGOgUjsLDZmcgnonU,4743
|
|
52
53
|
verifiers/v1/configs/serve.py,sha256=oa_0BrAVEKboU344w0hYkSaXj8lf26t0H87gvncAo0Q,2716
|
|
53
54
|
verifiers/v1/configs/task.py,sha256=57OnEGAa6jTiG12Lx_elREU02koYh8UkM0t0Vvxahys,2614
|
|
54
55
|
verifiers/v1/configs/taskset.py,sha256=LLbKdV2LW0ZjB9-fsKErIVKjchOAPcDLWGLVjgXKG_0,935
|
|
55
56
|
verifiers/v1/configs/cli/__init__.py,sha256=YMDUPdbRlDJ77dhj3QqzmyRhqNgUvuQdpj_bdBqzZPM,78
|
|
56
|
-
verifiers/v1/configs/cli/debug.py,sha256=
|
|
57
|
+
verifiers/v1/configs/cli/debug.py,sha256=CcUZ9MEcv0iAL9w2BkTfnZg20yWJ-Vn8P30-e5tuCic,2741
|
|
57
58
|
verifiers/v1/configs/cli/env.py,sha256=FcapEAtgeHzRuVaqk8BU36ZxmMOfdJ1XlioPbdwwspo,3990
|
|
58
|
-
verifiers/v1/configs/cli/eval.py,sha256=
|
|
59
|
+
verifiers/v1/configs/cli/eval.py,sha256=9Zw7mAUipMhoIcwRa_lgo5gII4rMmMEBAgzcPxPWPjw,6895
|
|
59
60
|
verifiers/v1/configs/cli/init.py,sha256=JvIUOdhfZfNOXg2wm2i06jLLHpb97tNMzZqmcK_9wVw,916
|
|
60
61
|
verifiers/v1/configs/cli/replay.py,sha256=vUCNs-9j1CXcAZhGozN1EaztcXK_dYVK3iN_fRmbSXw,2929
|
|
61
|
-
verifiers/v1/configs/cli/validate.py,sha256=
|
|
62
|
+
verifiers/v1/configs/cli/validate.py,sha256=oc-SSot6-5a0vXC-yFEQxYbb1O4r8k0Or8d_ijmizkU,3581
|
|
62
63
|
verifiers/v1/dialects/__init__.py,sha256=MNOYM2NdTlSxGK4VtoEjMtD5zya2MpjPDX5jMvZUWEo,736
|
|
63
64
|
verifiers/v1/dialects/anthropic.py,sha256=rkxS8K4xvSTZwS86Mz_a4MfHzrkYf9Vvu8wCerzVNIw,26111
|
|
64
65
|
verifiers/v1/dialects/base.py,sha256=7peaHJGdVT93G3erTnlAcRJi7tGvJ9BUkd_W5O4HGms,16140
|
|
@@ -78,10 +79,10 @@ verifiers/v1/envs/user_sim/__init__.py,sha256=nnT695HdqjAEouYL5wtVu6MhIp2wf5yuK-
|
|
|
78
79
|
verifiers/v1/envs/user_sim/env.py,sha256=deI4r-Utn_v4E_Vb9P_X3LzFh2GRkCO4OlQt8BOCMGY,4942
|
|
79
80
|
verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
|
|
80
81
|
verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,6139
|
|
81
|
-
verifiers/v1/gepa/config.py,sha256=
|
|
82
|
-
verifiers/v1/gepa/dataset.py,sha256=
|
|
82
|
+
verifiers/v1/gepa/config.py,sha256=zRjrFkiyMWzD1LXvjuUV_rJtXVO5fWNiraVruHoIADA,4830
|
|
83
|
+
verifiers/v1/gepa/dataset.py,sha256=cS4MtQVoBOnAHKs8REQm5lKGeqUpdXPU1hROU67aBG4,2130
|
|
83
84
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
84
|
-
verifiers/v1/gepa/runner.py,sha256
|
|
85
|
+
verifiers/v1/gepa/runner.py,sha256=-pMy3u5sffEQ4-aDexQviHeQ9NXdNv5rBBnNmhyx_0U,5943
|
|
85
86
|
verifiers/v1/harnesses/__init__.py,sha256=2JPwrRoYoigH-HfBtiFnatRdFEVHLv4rwlKfsFEibyE,1803
|
|
86
87
|
verifiers/v1/harnesses/node.py,sha256=Cci-zCNiVRCtLkJuNOt9iKr7ZIb3ik_YugefXtRyt78,4349
|
|
87
88
|
verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
|
|
@@ -165,12 +166,12 @@ verifiers/v1/tasksets/harbor/toolset.py,sha256=_7HXu-uuvqzhmJ1EX6c3KWLg0g0eRrLjC
|
|
|
165
166
|
verifiers/v1/tasksets/nemo_gym/__init__.py,sha256=654gAvG3Wd_Z_nh6mypeKn4s_IIQ0Ge6vcZvxTHC2_E,306
|
|
166
167
|
verifiers/v1/tasksets/nemo_gym/response.py,sha256=EhrsquEP8g0bjzAThGfE6Ypgy86_nVX235NiFYtVDVM,4463
|
|
167
168
|
verifiers/v1/tasksets/nemo_gym/server.py,sha256=33C-VDZxPtMV2TFxJVIKijmtBLNlVeRs92ctPE3qAsM,2036
|
|
168
|
-
verifiers/v1/tasksets/nemo_gym/taskset.py,sha256=
|
|
169
|
+
verifiers/v1/tasksets/nemo_gym/taskset.py,sha256=ZBL73Tv-YsmY_XA49T6Vx7xRfJlfsk8QtM8WZ723ZPo,6332
|
|
169
170
|
verifiers/v1/tasksets/nemo_gym/toolset.py,sha256=7iaRooFIfCN16wdSJbYFAOsc9q-haTivINO4S65UY6U,3715
|
|
170
171
|
verifiers/v1/tasksets/openenv/__init__.py,sha256=G726meYYDplC3QE3AyjdSShq7FUcb4NkVqowuwPPuJk,303
|
|
171
|
-
verifiers/v1/tasksets/openenv/taskset.py,sha256=
|
|
172
|
+
verifiers/v1/tasksets/openenv/taskset.py,sha256=PzoKgoaN5DpwY1LoWoS6p1yZdC08BP2q60Cqqbl6QSs,6046
|
|
172
173
|
verifiers/v1/tasksets/textarena/__init__.py,sha256=Os2OlBY_pSH0B1DP32RitumgSLpq2LtcI7VDUHYPzRE,329
|
|
173
|
-
verifiers/v1/tasksets/textarena/taskset.py,sha256=
|
|
174
|
+
verifiers/v1/tasksets/textarena/taskset.py,sha256=zGslerxmH2ZT7FJdP-72R6a9p71kr2pNS3r5_xnq1lQ,4406
|
|
174
175
|
verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
175
176
|
verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
|
|
176
177
|
verifiers/v1/utils/artifacts.py,sha256=i9DRhxViMAf-aBPjs3_9Q8RWVRC5pD-_v97VbNPPNcY,9638
|
|
@@ -192,8 +193,8 @@ verifiers/v1/utils/scope.py,sha256=bzUEiWDOVdbj7dXOwGOlzsQ_-Xu0j9NR5TalQB-8TS8,8
|
|
|
192
193
|
verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
|
|
193
194
|
verifiers/v1/utils/trace_store.py,sha256=RsCDonGR-bs-tvEstcppiJm3jFmAdIiG83fA5KtCbAw,2801
|
|
194
195
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
195
|
-
verifiers-0.3.2.
|
|
196
|
-
verifiers-0.3.2.
|
|
197
|
-
verifiers-0.3.2.
|
|
198
|
-
verifiers-0.3.2.
|
|
199
|
-
verifiers-0.3.2.
|
|
196
|
+
verifiers-0.3.2.dev152.dist-info/METADATA,sha256=GPTXqNo6x_F7kc4NNULLjsI0ZcqMJjqU-l9xJqqxm48,4161
|
|
197
|
+
verifiers-0.3.2.dev152.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
198
|
+
verifiers-0.3.2.dev152.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
|
|
199
|
+
verifiers-0.3.2.dev152.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
200
|
+
verifiers-0.3.2.dev152.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|