verifiers 0.2.2.dev53__py3-none-any.whl → 0.2.2.dev55__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/__init__.py +2 -4
- verifiers/v1/artifacts.py +2 -4
- verifiers/v1/cli/dashboard/eval.py +29 -42
- verifiers/v1/cli/debug.py +6 -6
- verifiers/v1/cli/replay.py +3 -3
- verifiers/v1/env.py +2 -2
- verifiers/v1/envs/agentic_judge/env.py +2 -3
- verifiers/v1/episode.py +52 -24
- verifiers/v1/graph.py +3 -4
- verifiers/v1/judge.py +3 -4
- verifiers/v1/judges/rubric.py +5 -5
- verifiers/v1/legacy.py +4 -3
- verifiers/v1/rollout.py +5 -5
- verifiers/v1/serve/server.py +2 -2
- verifiers/v1/state.py +2 -3
- verifiers/v1/task.py +56 -80
- verifiers/v1/taskset.py +0 -2
- verifiers/v1/tasksets/harbor/taskset.py +3 -4
- verifiers/v1/trace.py +22 -23
- verifiers/v1/types.py +13 -17
- {verifiers-0.2.2.dev53.dist-info → verifiers-0.2.2.dev55.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev53.dist-info → verifiers-0.2.2.dev55.dist-info}/RECORD +25 -25
- {verifiers-0.2.2.dev53.dist-info → verifiers-0.2.2.dev55.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev53.dist-info → verifiers-0.2.2.dev55.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev53.dist-info → verifiers-0.2.2.dev55.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/__init__.py
CHANGED
|
@@ -115,10 +115,10 @@ from verifiers.v1.taskset import Taskset
|
|
|
115
115
|
from verifiers.v1.trace import (
|
|
116
116
|
TRACE_VERSION,
|
|
117
117
|
AgentInfo,
|
|
118
|
+
AgentSpan,
|
|
118
119
|
Branch,
|
|
119
120
|
Error,
|
|
120
121
|
EvalRunInfo,
|
|
121
|
-
GenerationSpan,
|
|
122
122
|
ModelCall,
|
|
123
123
|
Reward,
|
|
124
124
|
RunInfo,
|
|
@@ -144,7 +144,6 @@ from verifiers.v1.types import (
|
|
|
144
144
|
Response,
|
|
145
145
|
Sampling,
|
|
146
146
|
SamplingConfig,
|
|
147
|
-
StrictBaseModel,
|
|
148
147
|
SystemMessage,
|
|
149
148
|
TextContentPart,
|
|
150
149
|
Tool,
|
|
@@ -177,7 +176,6 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
177
176
|
"Response",
|
|
178
177
|
"Sampling",
|
|
179
178
|
"SamplingConfig",
|
|
180
|
-
"StrictBaseModel",
|
|
181
179
|
"SystemMessage",
|
|
182
180
|
"TextContentPart",
|
|
183
181
|
"Tool",
|
|
@@ -213,7 +211,7 @@ __all__ = [ # noqa: RUF022 - grouped by public API area
|
|
|
213
211
|
"Timing",
|
|
214
212
|
"TimeSpan",
|
|
215
213
|
"TimeSplit",
|
|
216
|
-
"
|
|
214
|
+
"AgentSpan",
|
|
217
215
|
"Error",
|
|
218
216
|
# decorators
|
|
219
217
|
"stop",
|
verifiers/v1/artifacts.py
CHANGED
|
@@ -8,9 +8,7 @@ import uuid
|
|
|
8
8
|
from pathlib import PurePosixPath
|
|
9
9
|
from typing import TYPE_CHECKING
|
|
10
10
|
|
|
11
|
-
from pydantic import Field
|
|
12
|
-
|
|
13
|
-
from verifiers.v1.types import StrictBaseModel
|
|
11
|
+
from pydantic import BaseModel, Field
|
|
14
12
|
|
|
15
13
|
if TYPE_CHECKING:
|
|
16
14
|
from verifiers.v1.runtimes import Runtime
|
|
@@ -25,7 +23,7 @@ MAX_ARTIFACT_BYTES = 32 * 1024 * 1024
|
|
|
25
23
|
agent's image, so the repo is already there and only its output has to travel."""
|
|
26
24
|
|
|
27
25
|
|
|
28
|
-
class Artifact(
|
|
26
|
+
class Artifact(BaseModel):
|
|
29
27
|
"""One path to restore at the same location in another runtime."""
|
|
30
28
|
|
|
31
29
|
source: str
|
|
@@ -393,18 +393,19 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
393
393
|
phase_count: dict[str, int] = {}
|
|
394
394
|
model_secs = harness_secs = 0.0
|
|
395
395
|
for trace in done:
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
if
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
396
|
+
total_in += trace.num_input_tokens
|
|
397
|
+
total_out += trace.num_output_tokens
|
|
398
|
+
usage = trace.usage
|
|
399
|
+
if usage is not None:
|
|
400
|
+
if usage.cached_input_tokens is not None:
|
|
401
|
+
total_cached += usage.cached_input_tokens
|
|
402
|
+
have_cached = True
|
|
403
|
+
if usage.reasoning_tokens is not None:
|
|
404
|
+
total_reasoning += usage.reasoning_tokens
|
|
405
|
+
have_reasoning = True
|
|
406
|
+
if usage.cost is not None:
|
|
407
|
+
total_cost += usage.cost
|
|
408
|
+
have_cost = True
|
|
408
409
|
# Judge / auxiliary scoring calls (off the message graph) shown separately from the agent's.
|
|
409
410
|
judge = Usage.aggregate(trace.extra_usage)
|
|
410
411
|
if judge is not None:
|
|
@@ -413,13 +414,13 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
413
414
|
if judge.cost is not None:
|
|
414
415
|
total_judge_cost += judge.cost
|
|
415
416
|
have_judge = True
|
|
416
|
-
for phase in ("boot", "setup", "
|
|
417
|
+
for phase in ("boot", "setup", "agent", "finalize", "scoring"):
|
|
417
418
|
span = getattr(trace.timing, phase)
|
|
418
419
|
if span.end: # phase was timed for this rollout
|
|
419
420
|
phase_secs[phase] = phase_secs.get(phase, 0.0) + span.duration
|
|
420
421
|
phase_count[phase] = phase_count.get(phase, 0) + 1
|
|
421
|
-
model_secs += trace.timing.
|
|
422
|
-
harness_secs += trace.timing.
|
|
422
|
+
model_secs += trace.timing.agent.model.duration
|
|
423
|
+
harness_secs += trace.timing.agent.harness.duration
|
|
423
424
|
if (
|
|
424
425
|
total_in
|
|
425
426
|
or total_out
|
|
@@ -448,12 +449,12 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
448
449
|
usage.append(cost)
|
|
449
450
|
grid.add_row("usage", " · ".join(usage))
|
|
450
451
|
time_segments = []
|
|
451
|
-
for phase in ("boot", "setup", "
|
|
452
|
+
for phase in ("boot", "setup", "agent", "finalize", "scoring"):
|
|
452
453
|
count = phase_count.get(phase)
|
|
453
454
|
if not count:
|
|
454
455
|
continue
|
|
455
456
|
segment = f"{phase} {format_time(phase_secs[phase] / count)}"
|
|
456
|
-
if phase == "
|
|
457
|
+
if phase == "agent":
|
|
457
458
|
segment += (
|
|
458
459
|
f" (model {format_time(model_secs / count)}"
|
|
459
460
|
f" + harness {format_time(harness_secs / count)})"
|
|
@@ -464,26 +465,6 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
464
465
|
return grid if grid.row_count else None
|
|
465
466
|
|
|
466
467
|
|
|
467
|
-
def _tokens(trace: Trace) -> tuple[int, int, int | None, int | None, int]:
|
|
468
|
-
"""Input/output tokens summed across all branches: per branch, output is every assistant
|
|
469
|
-
(completion) token generated across its turns and input is the fed-in tokens counted once
|
|
470
|
-
(system + user + tool) — the final sequence minus everything the model generated. A rollout
|
|
471
|
-
yields one training sample per branch (a linear trace is a single branch; compaction and
|
|
472
|
-
subagents add more), so the totals sum them — matching `Trace.num_input_tokens` /
|
|
473
|
-
`Trace.num_output_tokens`, whose sum is `num_total_tokens`.
|
|
474
|
-
|
|
475
|
-
Both counts come from provider-reported usage. Returns the branch count from the same derived
|
|
476
|
-
view so each dashboard tick materializes it once."""
|
|
477
|
-
usage = trace.usage
|
|
478
|
-
cached = usage.cached_input_tokens if usage else None
|
|
479
|
-
reasoning = usage.reasoning_tokens if usage else None
|
|
480
|
-
branches = trace.branches
|
|
481
|
-
nbranches = len(branches)
|
|
482
|
-
prompt = sum(b.num_input_tokens for b in branches)
|
|
483
|
-
completion = sum(b.num_output_tokens for b in branches)
|
|
484
|
-
return prompt, completion, cached, reasoning, nbranches
|
|
485
|
-
|
|
486
|
-
|
|
487
468
|
def _stage(trace: Trace) -> str:
|
|
488
469
|
"""The stage a live (not-yet-done) rollout is in, derived from its trace's timing
|
|
489
470
|
spans — the engine opens and closes each span exactly at the stage transitions, so
|
|
@@ -495,7 +476,7 @@ def _stage(trace: Trace) -> str:
|
|
|
495
476
|
for stage, span in (
|
|
496
477
|
("scoring", trace.timing.scoring),
|
|
497
478
|
("finalize", trace.timing.finalize),
|
|
498
|
-
("running", trace.timing.
|
|
479
|
+
("running", trace.timing.agent),
|
|
499
480
|
("setup", trace.timing.setup),
|
|
500
481
|
("boot", trace.timing.boot),
|
|
501
482
|
):
|
|
@@ -551,7 +532,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
551
532
|
base = f"name={task.name[:32]}" if task.name else f"idx={task.idx}"
|
|
552
533
|
if not slot.traces:
|
|
553
534
|
if slot.done: # the env's rollout() itself failed before any trace
|
|
554
|
-
error =
|
|
535
|
+
error = (
|
|
536
|
+
slot.episode.last_error if slot.episode is not None else None
|
|
537
|
+
)
|
|
555
538
|
group_rows.append(
|
|
556
539
|
(
|
|
557
540
|
"error",
|
|
@@ -602,7 +585,7 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
602
585
|
end = (
|
|
603
586
|
t.timing.scoring.end
|
|
604
587
|
or t.timing.finalize.end
|
|
605
|
-
or t.timing.
|
|
588
|
+
or t.timing.agent.end
|
|
606
589
|
# a rollout that errored in boot/setup has only that span's end — freeze there
|
|
607
590
|
# once done, else (still running) the timer would grow off `now` forever
|
|
608
591
|
or (
|
|
@@ -612,8 +595,12 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
612
595
|
)
|
|
613
596
|
or now
|
|
614
597
|
)
|
|
615
|
-
prompt, completion
|
|
616
|
-
|
|
598
|
+
prompt, completion = t.num_input_tokens, t.num_output_tokens
|
|
599
|
+
nbranches = t.num_branches
|
|
600
|
+
usage = t.usage
|
|
601
|
+
cached = usage.cached_input_tokens if usage else None
|
|
602
|
+
reasoning = usage.reasoning_tokens if usage else None
|
|
603
|
+
cost = usage.cost if usage else None
|
|
617
604
|
tokens = ""
|
|
618
605
|
if prompt or completion:
|
|
619
606
|
tokens = f"{format_count(prompt)}/{format_count(completion)} tokens"
|
verifiers/v1/cli/debug.py
CHANGED
|
@@ -123,16 +123,16 @@ def record_debug_error(
|
|
|
123
123
|
action_timeout: float | None,
|
|
124
124
|
) -> None:
|
|
125
125
|
now = time.time()
|
|
126
|
-
for span in (trace.timing.boot, trace.timing.setup, trace.timing.
|
|
126
|
+
for span in (trace.timing.boot, trace.timing.setup, trace.timing.agent):
|
|
127
127
|
if span.start and not span.end:
|
|
128
128
|
span.end = now
|
|
129
|
-
in_action = bool(trace.timing.
|
|
129
|
+
in_action = bool(trace.timing.agent.start)
|
|
130
130
|
stage = (
|
|
131
131
|
"debug action" if in_action else "setup" if trace.timing.setup.start else "boot"
|
|
132
132
|
)
|
|
133
133
|
timeout = action_timeout if in_action else setup_timeout
|
|
134
134
|
error_start = (
|
|
135
|
-
trace.timing.
|
|
135
|
+
trace.timing.agent.start
|
|
136
136
|
if in_action
|
|
137
137
|
else trace.timing.setup.start or trace.timing.boot.start
|
|
138
138
|
)
|
|
@@ -234,9 +234,9 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
234
234
|
await runtime.prepare_execution([])
|
|
235
235
|
trace.timing.setup.end = time.time()
|
|
236
236
|
|
|
237
|
-
trace.timing.
|
|
237
|
+
trace.timing.agent.start = time.time()
|
|
238
238
|
debug.update(await run_action(runtime, config))
|
|
239
|
-
trace.timing.
|
|
239
|
+
trace.timing.agent.end = time.time()
|
|
240
240
|
if not debug.get("ok"):
|
|
241
241
|
record_action_failure(trace, debug)
|
|
242
242
|
trace.stop(str(debug["reason"]))
|
|
@@ -246,7 +246,7 @@ async def debug_task(task: Task, config: DebugConfig) -> tuple[Trace, bool]:
|
|
|
246
246
|
except Exception as e: # noqa: BLE001 - persist any framework failure on the trace
|
|
247
247
|
record_debug_error(trace, debug, e, setup_timeout, config.timeout.total)
|
|
248
248
|
finally:
|
|
249
|
-
trace.
|
|
249
|
+
trace.split_agent_time()
|
|
250
250
|
trace.info["debug"] = debug
|
|
251
251
|
try:
|
|
252
252
|
await runtime.stop()
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -31,7 +31,7 @@ from verifiers.v1.cli.resolve import narrow_taskset_config
|
|
|
31
31
|
from verifiers.v1.configs.agent import WireAgentConfig
|
|
32
32
|
from verifiers.v1.configs.cli.replay import ReplayConfig
|
|
33
33
|
from verifiers.v1.state import state_cls
|
|
34
|
-
from verifiers.v1.task import Task, WireTaskData
|
|
34
|
+
from verifiers.v1.task import Task, WireTaskData
|
|
35
35
|
from verifiers.v1.trace import Trace
|
|
36
36
|
from verifiers.v1.utils.interrupt import install_interrupt
|
|
37
37
|
from verifiers.v1.utils.logging import setup_logging
|
|
@@ -76,7 +76,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
76
76
|
"score() — multi-agent runs don't support replay"
|
|
77
77
|
)
|
|
78
78
|
task_cls = vf.task_type(config.taskset.id)
|
|
79
|
-
data_cls =
|
|
79
|
+
data_cls = task_cls.data_type()
|
|
80
80
|
# `WireTaskData` reads any taskset's saved task without importing its Task type.
|
|
81
81
|
# An episode may hold no traces (its env hooks failed before any agent ran);
|
|
82
82
|
# there's nothing to re-score, so it drops out in the flatten. Each kept trace
|
|
@@ -84,7 +84,7 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
84
84
|
episodes = read_episodes(
|
|
85
85
|
source, Trace[WireTaskData, state_cls(task_cls), WireAgentConfig]
|
|
86
86
|
)
|
|
87
|
-
sourced = [(trace, e.env) for e in episodes for trace in e.traces]
|
|
87
|
+
sourced = [(trace, e.env.id) for e in episodes for trace in e.traces]
|
|
88
88
|
if config.num_traces is not None:
|
|
89
89
|
sourced = sourced[: config.num_traces]
|
|
90
90
|
traces = [trace for trace, _ in sourced]
|
verifiers/v1/env.py
CHANGED
|
@@ -20,7 +20,7 @@ from verifiers.v1.configs.env import (
|
|
|
20
20
|
_declared_agent_configs,
|
|
21
21
|
default_agent_harness,
|
|
22
22
|
)
|
|
23
|
-
from verifiers.v1.episode import Episode
|
|
23
|
+
from verifiers.v1.episode import EnvInfo, Episode
|
|
24
24
|
from verifiers.v1.errors import EnvError, boundary
|
|
25
25
|
from verifiers.v1.harness import Harness, HarnessConfig
|
|
26
26
|
from verifiers.v1.interception import (
|
|
@@ -257,7 +257,7 @@ class Env(ABC, Generic[ConfigT]):
|
|
|
257
257
|
completed subset, its exception on the episode's `errors`. `on_trace` observes
|
|
258
258
|
each agent-run's trace at mint; `on_discard` its abandonment (a per-agent
|
|
259
259
|
retry mints a replacement)."""
|
|
260
|
-
episode = Episode(env=self.config.env_id)
|
|
260
|
+
episode = Episode(env=EnvInfo(id=self.config.env_id))
|
|
261
261
|
agents = self._episode_agents(ctx, episode.traces, on_trace, on_discard)
|
|
262
262
|
try:
|
|
263
263
|
async with asyncio.timeout(self.config.timeout.episode):
|
|
@@ -19,10 +19,9 @@ import re
|
|
|
19
19
|
import tomllib
|
|
20
20
|
from pathlib import Path
|
|
21
21
|
|
|
22
|
-
from pydantic import Field, field_validator
|
|
22
|
+
from pydantic import BaseModel, Field, field_validator
|
|
23
23
|
|
|
24
24
|
import verifiers.v1 as vf
|
|
25
|
-
from verifiers.v1.types import StrictBaseModel
|
|
26
25
|
from verifiers.v1.utils.compile import validate_pairing
|
|
27
26
|
|
|
28
27
|
VERDICT_FILE = "/tmp/verdict.json"
|
|
@@ -39,7 +38,7 @@ TASK_SECTION = """\
|
|
|
39
38
|
{prompt}"""
|
|
40
39
|
|
|
41
40
|
|
|
42
|
-
class Criterion(
|
|
41
|
+
class Criterion(BaseModel):
|
|
43
42
|
"""One rubric criterion — the plugged rubric judge's format, mirrored so the
|
|
44
43
|
same `criteria` files grade both judges."""
|
|
45
44
|
|
verifiers/v1/episode.py
CHANGED
|
@@ -3,51 +3,79 @@
|
|
|
3
3
|
import uuid
|
|
4
4
|
from typing import Generic
|
|
5
5
|
|
|
6
|
-
from pydantic import Field
|
|
6
|
+
from pydantic import BaseModel, Field
|
|
7
7
|
|
|
8
8
|
from verifiers.v1.configs.agent import WireAgentConfig
|
|
9
9
|
from verifiers.v1.state import State, StateT
|
|
10
10
|
from verifiers.v1.task import DataT, WireTaskData
|
|
11
11
|
from verifiers.v1.trace import AgentConfigT, Error, Trace
|
|
12
|
-
from verifiers.v1.types import
|
|
12
|
+
from verifiers.v1.types import Usage
|
|
13
13
|
|
|
14
14
|
|
|
15
|
-
class
|
|
16
|
-
"""
|
|
17
|
-
next to its flat `traces` — the object `finalize()` receives, the engine
|
|
18
|
-
returns, and the durability envelope: one episode is one `traces.jsonl` line
|
|
19
|
-
and one serve reply, so it persists and arrives whole or not at all — a torn
|
|
20
|
-
line is the whole episode owed again, and a failure before any trace minted
|
|
21
|
-
still leaves its errors here. Episode standing lives ONLY here (zero
|
|
22
|
-
redundancy on the traces); per-trace facts (`agent`, per-trace errors) stay
|
|
23
|
-
on the traces, which remain the atomic unit.
|
|
15
|
+
class EnvInfo(BaseModel):
|
|
16
|
+
"""The env that ran the episode, self-describing without the run's config."""
|
|
24
17
|
|
|
25
|
-
|
|
26
|
-
`
|
|
18
|
+
id: str = ""
|
|
19
|
+
"""`EnvConfig.env_id`, e.g. `agentic-judge+gsm8k-v1`."""
|
|
27
20
|
|
|
28
|
-
|
|
29
|
-
|
|
21
|
+
|
|
22
|
+
class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
23
|
+
"""The artifact Env.run produces. Contains multiple agents' traces."""
|
|
30
24
|
|
|
31
25
|
id: str = Field(default_factory=lambda: uuid.uuid4().hex)
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
26
|
+
|
|
27
|
+
env: EnvInfo = Field(default_factory=EnvInfo)
|
|
28
|
+
"""The env that produced this episode."""
|
|
35
29
|
ok: bool = False
|
|
36
|
-
"""
|
|
37
|
-
engine when the final attempt's hooks and every trace concluded clean.
|
|
38
|
-
Distinct from `errors` emptiness: a retried-and-recovered episode is `ok`
|
|
39
|
-
and still keeps its earlier attempts' errors."""
|
|
30
|
+
"""Whether the episode completed successfully."""
|
|
40
31
|
errors: list[Error] = Field(default_factory=list)
|
|
32
|
+
"""Every error captured across attempts, oldest to newest."""
|
|
41
33
|
traces: list[Trace[DataT, StateT, AgentConfigT]] = Field(default_factory=list)
|
|
34
|
+
"""Every agent's trace, in completion order."""
|
|
42
35
|
|
|
43
36
|
@property
|
|
44
|
-
def
|
|
37
|
+
def last_error(self) -> Error | None:
|
|
38
|
+
"""The last episode-level error captured across attempts."""
|
|
45
39
|
return self.errors[-1] if self.errors else None
|
|
46
40
|
|
|
41
|
+
@property
|
|
42
|
+
def usage(self) -> Usage | None:
|
|
43
|
+
"""Provider-reported usage summed across every trace's model calls;
|
|
44
|
+
judge/off-graph usage stays on the traces (`Trace.extra_usage`)."""
|
|
45
|
+
return Usage.aggregate(u for t in self.traces if (u := t.usage) is not None)
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def num_input_tokens(self) -> int:
|
|
49
|
+
"""Fed-in tokens (system + user + tool), summed across traces."""
|
|
50
|
+
return sum(t.num_input_tokens for t in self.traces)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def num_output_tokens(self) -> int:
|
|
54
|
+
"""Model-generated tokens across all turns, summed across traces."""
|
|
55
|
+
return sum(t.num_output_tokens for t in self.traces)
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def num_total_tokens(self) -> int:
|
|
59
|
+
"""Final sequence lengths per branch, summed across traces."""
|
|
60
|
+
return sum(t.num_total_tokens for t in self.traces)
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def num_turns(self) -> int:
|
|
64
|
+
"""Sampled turns, summed across traces."""
|
|
65
|
+
return sum(t.num_turns for t in self.traces)
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def by_agent(self) -> dict[str, list[Trace[DataT, StateT, AgentConfigT]]]:
|
|
69
|
+
"""Traces grouped by agent name (e.g. n solvers), in completion order."""
|
|
70
|
+
grouped: dict[str, list[Trace[DataT, StateT, AgentConfigT]]] = {}
|
|
71
|
+
for trace in self.traces:
|
|
72
|
+
grouped.setdefault(trace.agent.name, []).append(trace)
|
|
73
|
+
return grouped
|
|
74
|
+
|
|
47
75
|
@classmethod
|
|
48
76
|
def of(cls, trace: Trace, env: str = "") -> "Episode":
|
|
49
77
|
"""The single-agent record: one trace as its own episode."""
|
|
50
|
-
return cls(env=env, traces=[trace], ok=trace.ok)
|
|
78
|
+
return cls(env=EnvInfo(id=env), traces=[trace], ok=trace.ok)
|
|
51
79
|
|
|
52
80
|
|
|
53
81
|
WireEpisode = Episode[WireTaskData, State, WireAgentConfig]
|
verifiers/v1/graph.py
CHANGED
|
@@ -25,7 +25,7 @@ from dataclasses import dataclass
|
|
|
25
25
|
from typing import TYPE_CHECKING, Any
|
|
26
26
|
|
|
27
27
|
import numpy as np
|
|
28
|
-
from pydantic import ConfigDict, Field, field_serializer, field_validator
|
|
28
|
+
from pydantic import BaseModel, ConfigDict, Field, field_serializer, field_validator
|
|
29
29
|
from pydantic.json_schema import SkipJsonSchema
|
|
30
30
|
from renderers.base import MultiModalData, PlaceholderRange, RenderedTokens
|
|
31
31
|
|
|
@@ -34,7 +34,6 @@ from verifiers.v1.types import (
|
|
|
34
34
|
KeptTokens,
|
|
35
35
|
Message,
|
|
36
36
|
Response,
|
|
37
|
-
StrictBaseModel,
|
|
38
37
|
TextContentPart,
|
|
39
38
|
Tool,
|
|
40
39
|
ToolMessage,
|
|
@@ -62,7 +61,7 @@ def _decode_ndarray(d: dict) -> np.ndarray:
|
|
|
62
61
|
return np.frombuffer(d["data"], dtype=np.dtype(d["dtype"])).reshape(d["shape"])
|
|
63
62
|
|
|
64
63
|
|
|
65
|
-
class MessageNode(
|
|
64
|
+
class MessageNode(BaseModel):
|
|
66
65
|
"""One message in the graph: a message plus the tokens it adds to the cumulative
|
|
67
66
|
sequence. Concatenating a root→leaf path's nodes reconstructs that branch's full token
|
|
68
67
|
sequence; the mask/logprobs make it a training sample."""
|
|
@@ -121,7 +120,7 @@ class MessageNode(StrictBaseModel):
|
|
|
121
120
|
sampling-replay training. Rides the wire as raw-bytes `__nd__` dicts; kept off disk
|
|
122
121
|
by the dump-site `exclude` in prime-rl."""
|
|
123
122
|
|
|
124
|
-
model_config = ConfigDict(
|
|
123
|
+
model_config = ConfigDict(arbitrary_types_allowed=True)
|
|
125
124
|
|
|
126
125
|
@field_serializer("multi_modal_data")
|
|
127
126
|
def serialize_multi_modal_data(self, mmd: MultiModalData | None) -> dict | None:
|
verifiers/v1/judge.py
CHANGED
|
@@ -62,7 +62,7 @@ from verifiers.v1.configs.judge import (
|
|
|
62
62
|
)
|
|
63
63
|
from verifiers.v1.dialects.chat import message_to_wire
|
|
64
64
|
from verifiers.v1.scoring import parse_judge_choice
|
|
65
|
-
from verifiers.v1.types import Messages,
|
|
65
|
+
from verifiers.v1.types import Messages, Usage
|
|
66
66
|
from verifiers.v1.utils.generic import concrete_type
|
|
67
67
|
|
|
68
68
|
if TYPE_CHECKING:
|
|
@@ -72,7 +72,7 @@ if TYPE_CHECKING:
|
|
|
72
72
|
ParsedT = TypeVar("ParsedT")
|
|
73
73
|
|
|
74
74
|
|
|
75
|
-
class JudgeResponse(
|
|
75
|
+
class JudgeResponse(BaseModel, Generic[ParsedT]):
|
|
76
76
|
text: str
|
|
77
77
|
parsed: ParsedT | None = None
|
|
78
78
|
usage: Usage | None = None
|
|
@@ -197,8 +197,7 @@ class Judge(Generic[ParsedT, ConfigT]):
|
|
|
197
197
|
)
|
|
198
198
|
if response.parsed is None:
|
|
199
199
|
raise RuntimeError(
|
|
200
|
-
f"judge returned no parseable structured output "
|
|
201
|
-
f"(finish_reason={choice.finish_reason})"
|
|
200
|
+
f"judge returned no parseable structured output (finish_reason={choice.finish_reason})"
|
|
202
201
|
)
|
|
203
202
|
else:
|
|
204
203
|
completion = await client.chat.completions.create(**kwargs)
|
verifiers/v1/judges/rubric.py
CHANGED
|
@@ -9,13 +9,13 @@ from functools import cached_property
|
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
from typing import cast
|
|
11
11
|
|
|
12
|
-
from pydantic import Field, field_validator
|
|
12
|
+
from pydantic import BaseModel, Field, field_validator
|
|
13
13
|
|
|
14
14
|
from verifiers.v1.configs.judge import JudgeConfig
|
|
15
15
|
from verifiers.v1.judge import Judge, JudgeView, judge_question, judge_response
|
|
16
16
|
from verifiers.v1.task import TaskData
|
|
17
17
|
from verifiers.v1.trace import Trace
|
|
18
|
-
from verifiers.v1.types import ID
|
|
18
|
+
from verifiers.v1.types import ID
|
|
19
19
|
|
|
20
20
|
RUBRIC_PROMPT = (Path(__file__).resolve().parent / "rubric.txt").read_text(
|
|
21
21
|
encoding="utf-8"
|
|
@@ -56,7 +56,7 @@ def first_verdicts_object(text: str) -> dict | None:
|
|
|
56
56
|
return None
|
|
57
57
|
|
|
58
58
|
|
|
59
|
-
class Criterion(
|
|
59
|
+
class Criterion(BaseModel):
|
|
60
60
|
name: str
|
|
61
61
|
"""Key for the criterion's metric (`<judge name>/<name>`) and its `weights` override."""
|
|
62
62
|
text: str
|
|
@@ -109,13 +109,13 @@ class RubricJudgeConfig(JudgeConfig):
|
|
|
109
109
|
handle either. Transient HTTP failures are already retried by the OpenAI client."""
|
|
110
110
|
|
|
111
111
|
|
|
112
|
-
class CriterionVerdict(
|
|
112
|
+
class CriterionVerdict(BaseModel):
|
|
113
113
|
name: str
|
|
114
114
|
reason: str
|
|
115
115
|
verdict: str
|
|
116
116
|
|
|
117
117
|
|
|
118
|
-
class RubricVerdicts(
|
|
118
|
+
class RubricVerdicts(BaseModel):
|
|
119
119
|
verdicts: list[CriterionVerdict]
|
|
120
120
|
|
|
121
121
|
|
verifiers/v1/legacy.py
CHANGED
|
@@ -36,8 +36,8 @@ from verifiers.v1.serve.types import (
|
|
|
36
36
|
from verifiers.v1.task import WireTaskData
|
|
37
37
|
from verifiers.v1.trace import (
|
|
38
38
|
AgentInfo,
|
|
39
|
+
AgentSpan,
|
|
39
40
|
Error,
|
|
40
|
-
GenerationSpan,
|
|
41
41
|
ModelCall,
|
|
42
42
|
Reward,
|
|
43
43
|
TimeSpan,
|
|
@@ -199,7 +199,8 @@ def _to_v1_tokens(raw: Any) -> TurnTokens | None:
|
|
|
199
199
|
def _timing(raw: Any) -> Timing:
|
|
200
200
|
"""Map the v0 timing record's generation/scoring durations onto a v1 ``Timing``
|
|
201
201
|
(we only have durations, so each span is encoded as start=0, end=duration).
|
|
202
|
-
v0's per-turn ``model``/``env``
|
|
202
|
+
v0's ``generation`` duration becomes the agent span; its per-turn ``model``/``env``
|
|
203
|
+
span collections carry the model/harness split."""
|
|
203
204
|
|
|
204
205
|
def _dur(node: Any) -> float:
|
|
205
206
|
if isinstance(node, dict):
|
|
@@ -212,7 +213,7 @@ def _timing(raw: Any) -> Timing:
|
|
|
212
213
|
|
|
213
214
|
raw = raw or {}
|
|
214
215
|
return Timing(
|
|
215
|
-
|
|
216
|
+
agent=AgentSpan(
|
|
216
217
|
start=0.0,
|
|
217
218
|
end=_dur(raw.get("generation")),
|
|
218
219
|
model=TimeSplit(duration=_dur(raw.get("model"))),
|
verifiers/v1/rollout.py
CHANGED
|
@@ -297,7 +297,7 @@ class RolloutRun:
|
|
|
297
297
|
raise
|
|
298
298
|
now = time.time()
|
|
299
299
|
self.trace.timing.setup.end = now
|
|
300
|
-
self.trace.timing.
|
|
300
|
+
self.trace.timing.agent.start = now
|
|
301
301
|
return True
|
|
302
302
|
|
|
303
303
|
async def step(self, messages: Messages | None = None) -> bool:
|
|
@@ -397,8 +397,8 @@ class RolloutRun:
|
|
|
397
397
|
try:
|
|
398
398
|
await self._stack.aclose()
|
|
399
399
|
finally:
|
|
400
|
-
if trace.timing.
|
|
401
|
-
trace.timing.
|
|
400
|
+
if trace.timing.agent.start and not trace.timing.agent.end:
|
|
401
|
+
trace.timing.agent.end = time.time()
|
|
402
402
|
if not self._failed and self._opened:
|
|
403
403
|
trace.timing.finalize.start = time.time()
|
|
404
404
|
async with boundary(TaskError, "task finalize"):
|
|
@@ -430,13 +430,13 @@ class RolloutRun:
|
|
|
430
430
|
for span in (
|
|
431
431
|
trace.timing.boot,
|
|
432
432
|
trace.timing.setup,
|
|
433
|
-
trace.timing.
|
|
433
|
+
trace.timing.agent,
|
|
434
434
|
trace.timing.finalize,
|
|
435
435
|
trace.timing.scoring,
|
|
436
436
|
):
|
|
437
437
|
if span.start and not span.end:
|
|
438
438
|
span.end = now
|
|
439
|
-
trace.
|
|
439
|
+
trace.split_agent_time()
|
|
440
440
|
if runtime is not None:
|
|
441
441
|
try:
|
|
442
442
|
await self.harness.cleanup(trace, runtime)
|
verifiers/v1/serve/server.py
CHANGED
|
@@ -22,7 +22,7 @@ from verifiers.v1.serve.types import (
|
|
|
22
22
|
RunRequest,
|
|
23
23
|
RunResponse,
|
|
24
24
|
)
|
|
25
|
-
from verifiers.v1.task import Task
|
|
25
|
+
from verifiers.v1.task import Task
|
|
26
26
|
from verifiers.v1.types import SamplingConfig
|
|
27
27
|
|
|
28
28
|
logger = logging.getLogger(__name__)
|
|
@@ -39,7 +39,7 @@ class EnvServer:
|
|
|
39
39
|
self.taskset_id = config.taskset.id
|
|
40
40
|
self.env = load_environment(config)
|
|
41
41
|
self.task_cls = type(self.env.taskset).task_type()
|
|
42
|
-
self.data_cls =
|
|
42
|
+
self.data_cls = self.task_cls.data_type()
|
|
43
43
|
# A dispatched task is its client-side model_dump(): a field excluded from
|
|
44
44
|
# serialization would vanish on the wire and rebuild silently defaulted, so
|
|
45
45
|
# refuse to serve such a taskset.
|
verifiers/v1/state.py
CHANGED
|
@@ -4,14 +4,13 @@ Tool servers synchronize it through the interception state channel. It is exclud
|
|
|
4
4
|
from serialized traces.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
from pydantic import ConfigDict, Field
|
|
7
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
8
8
|
from typing_extensions import TypeVar
|
|
9
9
|
|
|
10
|
-
from verifiers.v1.types import StrictBaseModel
|
|
11
10
|
from verifiers.v1.utils.generic import concrete_type
|
|
12
11
|
|
|
13
12
|
|
|
14
|
-
class State(
|
|
13
|
+
class State(BaseModel):
|
|
15
14
|
model_config = ConfigDict(ser_json_inf_nan="constants")
|
|
16
15
|
artifacts: dict[str, bytes] = Field(default_factory=dict)
|
|
17
16
|
|
verifiers/v1/task.py
CHANGED
|
@@ -1,25 +1,10 @@
|
|
|
1
|
-
"""Task data
|
|
2
|
-
|
|
3
|
-
`TaskData` is the wire half: a frozen pydantic model carrying
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
`Task` is the behavior half: runtime prep (`setup`/`finalize`), server declarations
|
|
9
|
-
(`tools`), well-formedness (`validate`), and per-trace judgement
|
|
10
|
-
(`@reward`/`@metric` methods plus the plugged judges from `config.judges`, run by
|
|
11
|
-
`score`). Subclass per dataset and parameterize `Task[MyData, MyState, MyConfig]`
|
|
12
|
-
(all three default); judgement that compares sibling traces lives on
|
|
13
|
-
`Env.finalize` instead.
|
|
14
|
-
|
|
15
|
-
A Task instance is shared across its rollouts (`-r n` runs hold the same instance),
|
|
16
|
-
so hooks must not stash per-rollout state on `self` — that lives on the trace
|
|
17
|
-
(`trace.state`).
|
|
18
|
-
|
|
19
|
-
On the wire only the data travels (plus the producing class's name,
|
|
20
|
-
`trace.task.type`): a saved row reads back as `WireTaskData` without importing the
|
|
21
|
-
taskset; a re-scoring consumer (`replay`) rebuilds the declared `TaskData` type and
|
|
22
|
-
wraps it in the declared `Task` — one task type per taskset.
|
|
1
|
+
"""Task data + behavior.
|
|
2
|
+
|
|
3
|
+
`TaskData` is the wire half: a frozen pydantic model carrying the data which
|
|
4
|
+
initializes a task instance. Rides on `trace.task.data` in `traces.jsonl`.
|
|
5
|
+
|
|
6
|
+
`Task` is the behavior half: runtime prep (`setup`/`finalize`), tool declarations
|
|
7
|
+
(`tools`), and scoring (`@reward`/`@metric`) methods.
|
|
23
8
|
"""
|
|
24
9
|
|
|
25
10
|
from __future__ import annotations
|
|
@@ -30,7 +15,7 @@ import logging
|
|
|
30
15
|
from collections.abc import Mapping
|
|
31
16
|
from typing import TYPE_CHECKING, ClassVar, Generic, Self
|
|
32
17
|
|
|
33
|
-
from pydantic import ConfigDict, Field
|
|
18
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
34
19
|
from pydantic_config import BaseConfig
|
|
35
20
|
from typing_extensions import TypeVar
|
|
36
21
|
|
|
@@ -39,7 +24,7 @@ from verifiers.v1.configs.task import TaskConfig
|
|
|
39
24
|
from verifiers.v1.decorators import discover_decorated, invoke_all
|
|
40
25
|
from verifiers.v1.errors import TaskError, boundary
|
|
41
26
|
from verifiers.v1.state import StateT
|
|
42
|
-
from verifiers.v1.types import Messages,
|
|
27
|
+
from verifiers.v1.types import Messages, content_text
|
|
43
28
|
from verifiers.v1.utils.generic import concrete_type
|
|
44
29
|
|
|
45
30
|
if TYPE_CHECKING:
|
|
@@ -51,7 +36,9 @@ if TYPE_CHECKING:
|
|
|
51
36
|
logger = logging.getLogger(__name__)
|
|
52
37
|
|
|
53
38
|
|
|
54
|
-
class TaskResources(
|
|
39
|
+
class TaskResources(BaseModel):
|
|
40
|
+
"""Optional resource limits for the task."""
|
|
41
|
+
|
|
55
42
|
model_config = ConfigDict(frozen=True)
|
|
56
43
|
|
|
57
44
|
cpu: float | None = None
|
|
@@ -64,37 +51,41 @@ class TaskResources(StrictBaseModel):
|
|
|
64
51
|
"""Disk in GB (enforced by prime; advisory on docker/modal)."""
|
|
65
52
|
|
|
66
53
|
|
|
67
|
-
class TaskTimeout(
|
|
54
|
+
class TaskTimeout(BaseModel):
|
|
68
55
|
"""Optional per-task timeout overrides, in seconds."""
|
|
69
56
|
|
|
70
57
|
model_config = ConfigDict(frozen=True)
|
|
71
58
|
|
|
72
59
|
setup: float | None = None
|
|
60
|
+
"""Timeout (in seconds) for the task's setup hook."""
|
|
73
61
|
agent: float | None = None
|
|
62
|
+
"""Timeout (in seconds) for the agent's solve attempt."""
|
|
74
63
|
finalize: float | None = None
|
|
64
|
+
"""Timeout (in seconds) for the task's finalize hook."""
|
|
75
65
|
scoring: float | None = None
|
|
66
|
+
"""Timeout (in seconds) for the task's scoring."""
|
|
76
67
|
|
|
77
68
|
|
|
78
|
-
class TaskData(
|
|
79
|
-
"""The task's wire half: one row's pure data, a frozen pydantic model. Subclass
|
|
80
|
-
per dataset to add typed task-specific fields next to the base fields; behavior
|
|
81
|
-
lives on `Task`, which wraps this (`self.data`)."""
|
|
82
|
-
|
|
69
|
+
class TaskData(BaseModel):
|
|
83
70
|
model_config = ConfigDict(frozen=True)
|
|
84
71
|
|
|
85
72
|
idx: int | None = None
|
|
86
|
-
"""
|
|
87
|
-
on the eval path); `None` for ad-hoc tasks minted in scripts."""
|
|
73
|
+
"""Taskset-autogenerated index."""
|
|
88
74
|
name: str | None = None
|
|
75
|
+
"""Optional human-readable task name."""
|
|
89
76
|
description: str | None = None
|
|
77
|
+
"""Optional human-readable task description."""
|
|
78
|
+
|
|
90
79
|
prompt: str | Messages | None = None
|
|
91
|
-
"""Initial user prompt;
|
|
92
|
-
task through `agent.interaction()`, whose first `turn(message)` speaks first. (A
|
|
93
|
-
default, not just optional: the wire drops `None`s — `traces.jsonl` rows for
|
|
94
|
-
prompt-less tasks must read back.)"""
|
|
80
|
+
"""Initial user prompt; unset if the user opens the conversation."""
|
|
95
81
|
system_prompt: str | None = None
|
|
82
|
+
"""Optional system prompt to prepend to the user prompt."""
|
|
83
|
+
|
|
96
84
|
image: str | None = None
|
|
85
|
+
"""Optional Docker image to use for the task. Only relevant for tasks that run in a container."""
|
|
97
86
|
workdir: str | None = None
|
|
87
|
+
"""Optional working directory to use for the task. Only relevant for tasks that run in a container."""
|
|
88
|
+
|
|
98
89
|
network_allow: list[str] = Field(default_factory=lambda: ["*"])
|
|
99
90
|
"""Execution-time destinations requested by this task. `*` leaves the runtime
|
|
100
91
|
allowlist unchanged; a concrete list replaces a wildcard or combines with existing
|
|
@@ -103,11 +94,13 @@ class TaskData(StrictBaseModel):
|
|
|
103
94
|
"""Execution-time destinations denied by this task and combined with runtime
|
|
104
95
|
blocks. Non-empty concrete allowlists cannot be combined with blocklists. Docker
|
|
105
96
|
framework routes take precedence; ordinary Prime deny rules pass through unchanged."""
|
|
97
|
+
|
|
106
98
|
artifacts: list[Artifact] = Field(default_factory=list)
|
|
107
99
|
"""Paths collected from one runtime and restored at the same locations in another,
|
|
108
100
|
on top of the implicitly collected `/logs/artifacts/` convention dir. Declare
|
|
109
101
|
runtime outputs that must cross that boundary. A declared path that is missing at
|
|
110
102
|
collection time fails the rollout."""
|
|
103
|
+
|
|
111
104
|
timeout: TaskTimeout = TaskTimeout()
|
|
112
105
|
resources: TaskResources = TaskResources()
|
|
113
106
|
|
|
@@ -125,23 +118,10 @@ class WireTaskData(TaskData):
|
|
|
125
118
|
model_config = ConfigDict(extra="allow")
|
|
126
119
|
|
|
127
120
|
|
|
128
|
-
# No `default=`: an unparameterized `Trace`'s `task` field must serialize duck-typed
|
|
129
|
-
# (a defaulted TypeVar narrows pydantic's serialization to the base `TaskData`, silently
|
|
130
|
-
# dropping subclass fields from the wire).
|
|
131
121
|
DataT = TypeVar("DataT", bound=TaskData)
|
|
132
122
|
ConfigT = TypeVar("ConfigT", bound=TaskConfig, default=TaskConfig)
|
|
133
123
|
|
|
134
124
|
|
|
135
|
-
def task_data_cls(cls: type) -> type[TaskData]:
|
|
136
|
-
"""Resolve a task's `TaskData` specialization through its MRO, else `TaskData`."""
|
|
137
|
-
return concrete_type(cls, TaskData) or TaskData
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
def task_config_cls(cls: type) -> type[TaskConfig]:
|
|
141
|
-
"""Resolve a task's `TaskConfig` specialization through its MRO, else `TaskConfig`."""
|
|
142
|
-
return concrete_type(cls, TaskConfig) or TaskConfig
|
|
143
|
-
|
|
144
|
-
|
|
145
125
|
def resolve_server_config(
|
|
146
126
|
owner: str, config: BaseConfig, server_cls: type, *, sole: bool = True
|
|
147
127
|
) -> BaseConfig:
|
|
@@ -168,48 +148,20 @@ def resolve_server_config(
|
|
|
168
148
|
|
|
169
149
|
|
|
170
150
|
class Task(Generic[DataT, StateT, ConfigT]):
|
|
171
|
-
"""Behavior, lifecycle, servers, and scoring for one `TaskData` row.
|
|
172
|
-
|
|
173
|
-
Parameterize as `Task[MyData, MyState, MyConfig]`. Construction accepts the row
|
|
174
|
-
and an optional config; omitting config builds the declared config type. One task
|
|
175
|
-
instance is shared across a rollout group, so per-rollout state belongs on the trace.
|
|
176
|
-
"""
|
|
177
|
-
|
|
178
151
|
NEEDS_CONTAINER: ClassVar[bool] = False
|
|
152
|
+
"""Whether the task needs a containerized environment (isolated filesystem, ...)."""
|
|
179
153
|
|
|
180
154
|
tools: ClassVar[tuple[type[Toolset], ...]] = ()
|
|
181
155
|
|
|
182
156
|
def __init__(self, data: DataT, config: ConfigT | None = None) -> None:
|
|
183
157
|
self.data = data
|
|
184
|
-
self.config = config if config is not None else
|
|
158
|
+
self.config = config if config is not None else self.config_type()()
|
|
185
159
|
|
|
186
160
|
def with_system_prompt(self, system_prompt: str) -> Self:
|
|
187
|
-
"""A shallow copy of this task with `data.system_prompt` overridden. Copies the
|
|
188
|
-
instance instead of reconstructing via `type(self)(...)`, so a subclass with a
|
|
189
|
-
non-`(data, config)` constructor or extra load-time state keeps it. Used to apply the
|
|
190
|
-
config-layer / GEPA system prompt (see `TasksetConfig` and `verifiers.v1.gepa`)."""
|
|
191
161
|
clone = copy.copy(self)
|
|
192
162
|
clone.data = self.data.model_copy(update={"system_prompt": system_prompt})
|
|
193
163
|
return clone
|
|
194
164
|
|
|
195
|
-
def plugged_judges(self) -> list[Judge]:
|
|
196
|
-
from verifiers.v1.loaders import load_judge
|
|
197
|
-
|
|
198
|
-
return [load_judge(config) for config in self.config.judges]
|
|
199
|
-
|
|
200
|
-
def server_config(self, server_cls: type) -> BaseConfig:
|
|
201
|
-
"""The config a declared server class (`tools`) is built with (see
|
|
202
|
-
`resolve_server_config`). Override to pair explicitly."""
|
|
203
|
-
return resolve_server_config(
|
|
204
|
-
type(self).__name__,
|
|
205
|
-
self.config,
|
|
206
|
-
server_cls,
|
|
207
|
-
sole=len(set(type(self).tools)) == 1,
|
|
208
|
-
)
|
|
209
|
-
|
|
210
|
-
def tool_servers(self) -> list[Toolset]:
|
|
211
|
-
return [cls(self.server_config(cls)) for cls in type(self).tools]
|
|
212
|
-
|
|
213
165
|
async def setup(self, trace: Trace, runtime: Runtime) -> None:
|
|
214
166
|
return None
|
|
215
167
|
|
|
@@ -284,5 +236,29 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
284
236
|
for key, value in items:
|
|
285
237
|
trace.record_reward(key, value, judge.config.weight)
|
|
286
238
|
|
|
239
|
+
@classmethod
|
|
240
|
+
def data_type(cls) -> type[TaskData]:
|
|
241
|
+
return concrete_type(cls, TaskData) or TaskData
|
|
242
|
+
|
|
243
|
+
@classmethod
|
|
244
|
+
def config_type(cls) -> type[TaskConfig]:
|
|
245
|
+
return concrete_type(cls, TaskConfig) or TaskConfig
|
|
246
|
+
|
|
247
|
+
def plugged_judges(self) -> list[Judge]:
|
|
248
|
+
from verifiers.v1.loaders import load_judge
|
|
249
|
+
|
|
250
|
+
return [load_judge(config) for config in self.config.judges]
|
|
251
|
+
|
|
252
|
+
def server_config(self, server_cls: type) -> BaseConfig:
|
|
253
|
+
return resolve_server_config(
|
|
254
|
+
type(self).__name__,
|
|
255
|
+
self.config,
|
|
256
|
+
server_cls,
|
|
257
|
+
sole=len(set(type(self).tools)) == 1,
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
def tool_servers(self) -> list[Toolset]:
|
|
261
|
+
return [cls(self.server_config(cls)) for cls in type(self).tools]
|
|
262
|
+
|
|
287
263
|
|
|
288
264
|
TaskT = TypeVar("TaskT", bound=Task)
|
verifiers/v1/taskset.py
CHANGED
|
@@ -103,8 +103,6 @@ class Taskset(ABC, Generic[TaskT, TasksetConfigT]):
|
|
|
103
103
|
return concrete_type(cls, Task, origin=Taskset) or Task
|
|
104
104
|
|
|
105
105
|
def server_config(self, server_cls: type) -> BaseConfig:
|
|
106
|
-
"""The config a `tools` entry is built with, resolved off `self.config` (the
|
|
107
|
-
taskset config; see `resolve_server_config`). Override to pair explicitly."""
|
|
108
106
|
return resolve_server_config(
|
|
109
107
|
type(self).__name__,
|
|
110
108
|
self.config,
|
|
@@ -22,7 +22,7 @@ from collections.abc import Iterator
|
|
|
22
22
|
from functools import lru_cache
|
|
23
23
|
from pathlib import Path
|
|
24
24
|
|
|
25
|
-
from pydantic import Field
|
|
25
|
+
from pydantic import BaseModel, Field
|
|
26
26
|
|
|
27
27
|
from verifiers.v1.artifacts import Artifact, collect
|
|
28
28
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
@@ -32,7 +32,6 @@ from verifiers.v1.runtimes import Runtime
|
|
|
32
32
|
from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
|
|
33
33
|
from verifiers.v1.taskset import Taskset
|
|
34
34
|
from verifiers.v1.trace import Trace
|
|
35
|
-
from verifiers.v1.types import StrictBaseModel
|
|
36
35
|
|
|
37
36
|
CACHE = Path.home() / ".cache" / "harbor"
|
|
38
37
|
HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
|
|
@@ -73,12 +72,12 @@ class HarborConfig(TasksetConfig):
|
|
|
73
72
|
has what the task needs (e.g. you've pointed the runtime at the right image)."""
|
|
74
73
|
|
|
75
74
|
|
|
76
|
-
class Author(
|
|
75
|
+
class Author(BaseModel):
|
|
77
76
|
name: str | None = None
|
|
78
77
|
email: str | None = None
|
|
79
78
|
|
|
80
79
|
|
|
81
|
-
class CollectHook(
|
|
80
|
+
class CollectHook(BaseModel):
|
|
82
81
|
"""One `[[verifier.collect]]` command, run in the agent's box by `finalize`."""
|
|
83
82
|
|
|
84
83
|
command: str
|
verifiers/v1/trace.py
CHANGED
|
@@ -7,7 +7,7 @@ from collections.abc import Mapping
|
|
|
7
7
|
from typing import TYPE_CHECKING, Annotated, Any, Generic, Literal
|
|
8
8
|
|
|
9
9
|
import numpy as np
|
|
10
|
-
from pydantic import Field, PrivateAttr
|
|
10
|
+
from pydantic import BaseModel, Field, PrivateAttr
|
|
11
11
|
from renderers.base import MultiModalData
|
|
12
12
|
from typing_extensions import TypeVar
|
|
13
13
|
|
|
@@ -27,7 +27,6 @@ from verifiers.v1.types import (
|
|
|
27
27
|
KeptTokens,
|
|
28
28
|
Messages,
|
|
29
29
|
Sampling,
|
|
30
|
-
StrictBaseModel,
|
|
31
30
|
Tool,
|
|
32
31
|
ToolMessage,
|
|
33
32
|
Usage,
|
|
@@ -50,7 +49,7 @@ EXCLUDE_FIELDS: dict = {
|
|
|
50
49
|
"""Raw tensor fields kept on the msgpack wire but excluded from disk serialization."""
|
|
51
50
|
|
|
52
51
|
|
|
53
|
-
class TimeSpan(
|
|
52
|
+
class TimeSpan(BaseModel):
|
|
54
53
|
"""Wall-clock timestamps with a derived, non-serialized duration in seconds."""
|
|
55
54
|
|
|
56
55
|
start: float = 0.0
|
|
@@ -61,34 +60,34 @@ class TimeSpan(StrictBaseModel):
|
|
|
61
60
|
return max(0.0, self.end - self.start) if self.end else 0.0
|
|
62
61
|
|
|
63
62
|
|
|
64
|
-
class TimeSplit(
|
|
63
|
+
class TimeSplit(BaseModel):
|
|
65
64
|
"""Records a measured duration in seconds."""
|
|
66
65
|
|
|
67
66
|
duration: float = 0.0
|
|
68
67
|
|
|
69
68
|
|
|
70
|
-
class
|
|
69
|
+
class AgentSpan(TimeSpan):
|
|
71
70
|
model: TimeSplit = Field(default_factory=TimeSplit)
|
|
72
71
|
harness: TimeSplit = Field(default_factory=TimeSplit)
|
|
73
72
|
|
|
74
73
|
|
|
75
|
-
class Timing(
|
|
74
|
+
class Timing(BaseModel):
|
|
76
75
|
start: float = Field(default_factory=time.time)
|
|
77
76
|
boot: TimeSpan = Field(default_factory=TimeSpan)
|
|
78
77
|
setup: TimeSpan = Field(default_factory=TimeSpan)
|
|
79
|
-
|
|
78
|
+
agent: AgentSpan = Field(default_factory=AgentSpan)
|
|
80
79
|
finalize: TimeSpan = Field(default_factory=TimeSpan)
|
|
81
80
|
scoring: TimeSpan = Field(default_factory=TimeSpan)
|
|
82
81
|
|
|
83
82
|
|
|
84
|
-
class Error(
|
|
83
|
+
class Error(BaseModel):
|
|
85
84
|
type: str
|
|
86
85
|
message: str
|
|
87
86
|
status_code: int | None = None
|
|
88
87
|
traceback: str | None = None
|
|
89
88
|
|
|
90
89
|
|
|
91
|
-
class VersionInfo(
|
|
90
|
+
class VersionInfo(BaseModel):
|
|
92
91
|
version: str
|
|
93
92
|
commit: str | None = None
|
|
94
93
|
|
|
@@ -104,7 +103,7 @@ def _current_build() -> VersionInfo:
|
|
|
104
103
|
AgentConfigT = TypeVar("AgentConfigT", bound=AgentConfig, default=AgentConfig)
|
|
105
104
|
|
|
106
105
|
|
|
107
|
-
class AgentInfo(
|
|
106
|
+
class AgentInfo(BaseModel, Generic[AgentConfigT]):
|
|
108
107
|
config: AgentConfigT
|
|
109
108
|
"""The resolved config that rebuilds the agent (`Agent(trace.agent.config)`)."""
|
|
110
109
|
runtime: RuntimeInfo | None = None
|
|
@@ -115,7 +114,7 @@ class AgentInfo(StrictBaseModel, Generic[AgentConfigT]):
|
|
|
115
114
|
"""Whether this trace's tokens train the run's policy."""
|
|
116
115
|
|
|
117
116
|
|
|
118
|
-
class TraceTask(
|
|
117
|
+
class TraceTask(BaseModel, Generic[DataT]):
|
|
119
118
|
"""The task as recorded on the trace, self-describing without the run's config."""
|
|
120
119
|
|
|
121
120
|
type: str
|
|
@@ -124,7 +123,7 @@ class TraceTask(StrictBaseModel, Generic[DataT]):
|
|
|
124
123
|
"""The (immutable) row being solved."""
|
|
125
124
|
|
|
126
125
|
|
|
127
|
-
class Reward(
|
|
126
|
+
class Reward(BaseModel):
|
|
128
127
|
score: float
|
|
129
128
|
weight: float = 1.0
|
|
130
129
|
|
|
@@ -133,14 +132,14 @@ class Reward(StrictBaseModel):
|
|
|
133
132
|
return self.score * self.weight
|
|
134
133
|
|
|
135
134
|
|
|
136
|
-
class EvalRunInfo(
|
|
135
|
+
class EvalRunInfo(BaseModel):
|
|
137
136
|
type: Literal["eval"] = "eval"
|
|
138
137
|
|
|
139
138
|
id: str
|
|
140
139
|
step: int | None = None
|
|
141
140
|
|
|
142
141
|
|
|
143
|
-
class TrainRunInfo(
|
|
142
|
+
class TrainRunInfo(BaseModel):
|
|
144
143
|
type: Literal["train"] = "train"
|
|
145
144
|
|
|
146
145
|
id: str
|
|
@@ -151,7 +150,7 @@ RunInfo = Annotated[EvalRunInfo | TrainRunInfo, Field(discriminator="type")]
|
|
|
151
150
|
"""The run a trace belongs to, discriminated on `type`."""
|
|
152
151
|
|
|
153
152
|
|
|
154
|
-
class ModelCall(
|
|
153
|
+
class ModelCall(BaseModel):
|
|
155
154
|
"""A model call, automatically recorded at intercept time."""
|
|
156
155
|
|
|
157
156
|
node: int | None = None
|
|
@@ -172,7 +171,7 @@ class ModelCall(StrictBaseModel):
|
|
|
172
171
|
"""The failure that ended this call, coupled to the exchange that caused it."""
|
|
173
172
|
|
|
174
173
|
|
|
175
|
-
class Branch(
|
|
174
|
+
class Branch(BaseModel):
|
|
176
175
|
"""A root-to-leaf graph path; each branch becomes one training sample."""
|
|
177
176
|
|
|
178
177
|
index: int
|
|
@@ -298,7 +297,7 @@ class Branch(StrictBaseModel):
|
|
|
298
297
|
return self.num_total_tokens - self.num_output_tokens
|
|
299
298
|
|
|
300
299
|
|
|
301
|
-
class Trace(
|
|
300
|
+
class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
302
301
|
version: int = TRACE_VERSION
|
|
303
302
|
"""The trace schema this trace serializes as."""
|
|
304
303
|
id: str = Field(default_factory=lambda: uuid.uuid4().hex)
|
|
@@ -492,14 +491,14 @@ class Trace(StrictBaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
492
491
|
if self.stop_condition is None:
|
|
493
492
|
self.stop_condition = condition
|
|
494
493
|
|
|
495
|
-
def
|
|
496
|
-
"""Split the
|
|
497
|
-
|
|
498
|
-
if not
|
|
494
|
+
def split_agent_time(self) -> None:
|
|
495
|
+
"""Split the agent span into model and harness time."""
|
|
496
|
+
span = self.timing.agent
|
|
497
|
+
if not span.end:
|
|
499
498
|
return
|
|
500
499
|
model = sum(call.time.duration for call in self.calls)
|
|
501
|
-
|
|
502
|
-
|
|
500
|
+
span.model.duration = min(model, span.duration)
|
|
501
|
+
span.harness.duration = span.duration - span.model.duration
|
|
503
502
|
|
|
504
503
|
def record_error(self, error: Exception) -> None:
|
|
505
504
|
"""Record an error, and stop the trace as failed."""
|
verifiers/v1/types.py
CHANGED
|
@@ -7,20 +7,16 @@ from renderers.base import MultiModalData
|
|
|
7
7
|
from typing_extensions import TypedDict
|
|
8
8
|
|
|
9
9
|
|
|
10
|
-
class
|
|
11
|
-
model_config = ConfigDict(extra="forbid")
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
class TextContentPart(StrictBaseModel):
|
|
10
|
+
class TextContentPart(BaseModel):
|
|
15
11
|
type: Literal["text"] = "text"
|
|
16
12
|
text: str
|
|
17
13
|
|
|
18
14
|
|
|
19
|
-
class ImageUrlSource(
|
|
15
|
+
class ImageUrlSource(BaseModel):
|
|
20
16
|
url: str
|
|
21
17
|
|
|
22
18
|
|
|
23
|
-
class ImageUrlContentPart(
|
|
19
|
+
class ImageUrlContentPart(BaseModel):
|
|
24
20
|
type: Literal["image_url"] = "image_url"
|
|
25
21
|
image_url: ImageUrlSource
|
|
26
22
|
|
|
@@ -57,24 +53,24 @@ def content_text(content: "MessageContent | None") -> str:
|
|
|
57
53
|
)
|
|
58
54
|
|
|
59
55
|
|
|
60
|
-
class SystemMessage(
|
|
56
|
+
class SystemMessage(BaseModel):
|
|
61
57
|
role: Literal["system"] = "system"
|
|
62
58
|
content: MessageContent
|
|
63
59
|
|
|
64
60
|
|
|
65
|
-
class UserMessage(
|
|
61
|
+
class UserMessage(BaseModel):
|
|
66
62
|
role: Literal["user"] = "user"
|
|
67
63
|
content: MessageContent
|
|
68
64
|
|
|
69
65
|
|
|
70
|
-
class ToolCall(
|
|
66
|
+
class ToolCall(BaseModel):
|
|
71
67
|
id: str
|
|
72
68
|
name: str
|
|
73
69
|
arguments: str
|
|
74
70
|
"""Raw JSON string of arguments, exactly as the model emitted it."""
|
|
75
71
|
|
|
76
72
|
|
|
77
|
-
class AssistantMessage(
|
|
73
|
+
class AssistantMessage(BaseModel):
|
|
78
74
|
role: Literal["assistant"] = "assistant"
|
|
79
75
|
content: str | None = None
|
|
80
76
|
reasoning_content: str | None = None
|
|
@@ -83,7 +79,7 @@ class AssistantMessage(StrictBaseModel):
|
|
|
83
79
|
"""Opaque native items replayed to preserve signed or encrypted reasoning state."""
|
|
84
80
|
|
|
85
81
|
|
|
86
|
-
class ToolMessage(
|
|
82
|
+
class ToolMessage(BaseModel):
|
|
87
83
|
role: Literal["tool"] = "tool"
|
|
88
84
|
tool_call_id: str
|
|
89
85
|
content: MessageContent
|
|
@@ -98,7 +94,7 @@ Message = Annotated[
|
|
|
98
94
|
Messages = list[Message]
|
|
99
95
|
|
|
100
96
|
|
|
101
|
-
class Tool(
|
|
97
|
+
class Tool(BaseModel):
|
|
102
98
|
name: str
|
|
103
99
|
description: str
|
|
104
100
|
parameters: dict[str, Any]
|
|
@@ -108,7 +104,7 @@ class Tool(StrictBaseModel):
|
|
|
108
104
|
FinishReason = Literal["stop", "length", "tool_calls"] | None
|
|
109
105
|
|
|
110
106
|
|
|
111
|
-
class Usage(
|
|
107
|
+
class Usage(BaseModel):
|
|
112
108
|
"""Provider token accounting.
|
|
113
109
|
|
|
114
110
|
`prompt_tokens` excludes cache reads; `input_tokens` adds them back. Reasoning tokens
|
|
@@ -191,10 +187,10 @@ class KeptTokens:
|
|
|
191
187
|
counts: Any
|
|
192
188
|
|
|
193
189
|
|
|
194
|
-
class TurnTokens(
|
|
190
|
+
class TurnTokens(BaseModel):
|
|
195
191
|
"""Training tokens from renderer tokenization or provider-returned token IDs."""
|
|
196
192
|
|
|
197
|
-
model_config = ConfigDict(
|
|
193
|
+
model_config = ConfigDict(arbitrary_types_allowed=True)
|
|
198
194
|
|
|
199
195
|
prompt_ids: list[int] = Field(default_factory=list)
|
|
200
196
|
completion_ids: list[int] = Field(default_factory=list)
|
|
@@ -220,7 +216,7 @@ class TurnTokens(StrictBaseModel):
|
|
|
220
216
|
kept_tokens: KeptTokens | None = Field(default=None, exclude=True)
|
|
221
217
|
|
|
222
218
|
|
|
223
|
-
class Response(
|
|
219
|
+
class Response(BaseModel):
|
|
224
220
|
id: str
|
|
225
221
|
created: int
|
|
226
222
|
model: str
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev55
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -167,42 +167,42 @@ verifiers/utils/threaded_sandbox_client.py,sha256=XRzYhWBpMev7kw_aNMWAy6lfa-9uAn
|
|
|
167
167
|
verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs,1020
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
|
-
verifiers/v1/__init__.py,sha256=
|
|
170
|
+
verifiers/v1/__init__.py,sha256=Im5qTZy9mH6PRTUi-1V6W7EWKZzJyTWLGKMYALbWz_w,7467
|
|
171
171
|
verifiers/v1/agent.py,sha256=dj4GabQSs6Wzwbph5Rlx-efsVQsSuWV2YI5hfvq5mGA,31990
|
|
172
|
-
verifiers/v1/artifacts.py,sha256=
|
|
172
|
+
verifiers/v1/artifacts.py,sha256=5eKOjDSUWwj7GrhxILMHdz2ngEONWMdcTGlBcuGgrLU,5834
|
|
173
173
|
verifiers/v1/decorators.py,sha256=9CcCwgZlRjVGPqT5G1CFJUHP4OIMuBGkUJZkcI6dnbk,3865
|
|
174
|
-
verifiers/v1/env.py,sha256=
|
|
175
|
-
verifiers/v1/episode.py,sha256=
|
|
174
|
+
verifiers/v1/env.py,sha256=a7cKeecsm4d1lZPhRJy-m7FFWHrHF2msOdqma7o5I6M,18468
|
|
175
|
+
verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
|
|
176
176
|
verifiers/v1/errors.py,sha256=slZnrtqMo_BgXQV46wJJhV6vvAAnVsuCbVGp2vBVWfs,6888
|
|
177
|
-
verifiers/v1/graph.py,sha256
|
|
177
|
+
verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
|
|
178
178
|
verifiers/v1/harness.py,sha256=cDiDuhp29aqL4CxiqSZz-IT0QlOcVlye_vTTHZblWo8,10948
|
|
179
|
-
verifiers/v1/judge.py,sha256=
|
|
180
|
-
verifiers/v1/legacy.py,sha256=
|
|
179
|
+
verifiers/v1/judge.py,sha256=hISmuXidKCMROni9_4q4I-EeoO_uEieafGd2WBjRNKM,9338
|
|
180
|
+
verifiers/v1/legacy.py,sha256=ISk3Bix9Hy5EjZtyV9QO0LNyDNDklBHVjFokOrw_3U8,22628
|
|
181
181
|
verifiers/v1/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
|
|
182
182
|
verifiers/v1/push.py,sha256=MWZmvyj0iyzNIO0xRhlkCXtFF1plq6EJi8rWlvrFtvI,11363
|
|
183
183
|
verifiers/v1/retries.py,sha256=quUGVruUkZx-9WQcFiocMfL_M056voU8gwDqsuLR3f4,5412
|
|
184
|
-
verifiers/v1/rollout.py,sha256=
|
|
184
|
+
verifiers/v1/rollout.py,sha256=pmF4sgyHDRUUTnOxOWgBsPP8x6L_KFGaeR_dDKziANs,20470
|
|
185
185
|
verifiers/v1/scoring.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
186
186
|
verifiers/v1/session.py,sha256=J6ZPuvsg6vMTJAs9dMUMtfdjCuvp6Hom877DVdKFKlY,6912
|
|
187
|
-
verifiers/v1/state.py,sha256=
|
|
188
|
-
verifiers/v1/task.py,sha256=
|
|
189
|
-
verifiers/v1/taskset.py,sha256=
|
|
190
|
-
verifiers/v1/trace.py,sha256=
|
|
191
|
-
verifiers/v1/types.py,sha256=
|
|
187
|
+
verifiers/v1/state.py,sha256=pcpN2V6tX7rIO5XdtdUYay1j0ktEwrMtq7w1ApfsPdE,675
|
|
188
|
+
verifiers/v1/task.py,sha256=pM2S4YU56jTuvvwHe00r1pBupQ42pvMUzpQhUXqpAPk,10232
|
|
189
|
+
verifiers/v1/taskset.py,sha256=5ffTlnmiN2GDePhonGSJ1a7iTseLPOC0DkwaVwf7uY4,4465
|
|
190
|
+
verifiers/v1/trace.py,sha256=_bGHOrE6fRLGm0JYc0PrrrO3lyr-qqOF5Y1bHfViwR0,19347
|
|
191
|
+
verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
|
|
192
192
|
verifiers/v1/acp/__init__.py,sha256=9dwH6fLopNndRmcpRSYC8ghzzMquaLfgh645QM-Dwic,2189
|
|
193
193
|
verifiers/v1/acp/_runner.py,sha256=FqEnHR0Q_YrCt_EuaJqiqTpd2_kFa41FqcEunwOhvTI,6856
|
|
194
194
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
195
|
-
verifiers/v1/cli/debug.py,sha256
|
|
195
|
+
verifiers/v1/cli/debug.py,sha256=-jeEV_Q5tV8SOkWc1GdPZcXZWON_jU5gjeO4YoZm79o,11482
|
|
196
196
|
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
197
197
|
verifiers/v1/cli/init.py,sha256=dVpX1YMz7DFmHgskAsPjP-HmVTGoZfC0ZUyE-T4I6l0,8322
|
|
198
198
|
verifiers/v1/cli/output.py,sha256=pMS5Z0EBgyAVShNCAk9jVURNeLOaTfeRWpt0J8hHjq8,5890
|
|
199
|
-
verifiers/v1/cli/replay.py,sha256=
|
|
199
|
+
verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
|
|
200
200
|
verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
|
|
201
201
|
verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,2866
|
|
202
202
|
verifiers/v1/cli/validate.py,sha256=lZ_wWnLgIWH5rUlMtwzvF3uS_cFZA4ZqBowpzdI4gos,10260
|
|
203
203
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
204
204
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
205
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
205
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=QJlBKRvEaxBYWZ-WQQfJqaBZoVDIO3MnJnzeD3VusEI,33436
|
|
206
206
|
verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
|
|
207
207
|
verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
|
|
208
208
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
@@ -239,7 +239,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
|
|
|
239
239
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
240
240
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
241
241
|
verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
|
|
242
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
242
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=QRYxs8M5ivk7jnEwPgYfCLVZTBQOuSc64xydoBixXDs,16961
|
|
243
243
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
244
244
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
245
245
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -288,7 +288,7 @@ verifiers/v1/interception/tunnel/prime.py,sha256=K2n2daP_iNoO-087owAYHBkR7z0PBNn
|
|
|
288
288
|
verifiers/v1/judges/__init__.py,sha256=MUIBWcx6c70BykrTDHhB0ITIyJPxXe6xDBou4VM7jDQ,286
|
|
289
289
|
verifiers/v1/judges/reference.py,sha256=iEVHw-iJ6dbwLOOqrvWrEMJOcUEYJ0YspT04rZz7Sd4,3868
|
|
290
290
|
verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93V6A8,353
|
|
291
|
-
verifiers/v1/judges/rubric.py,sha256=
|
|
291
|
+
verifiers/v1/judges/rubric.py,sha256=PglrJhMQR6scFyfz20jVlLQEzeqXKV2P6r0K0K7jjBE,11916
|
|
292
292
|
verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
|
|
293
293
|
verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
|
|
294
294
|
verifiers/v1/mcp/launch.py,sha256=lRtcl98BO7imC44bsxO2yljwRQR_KCvJ6P-bd_hS5L8,18444
|
|
@@ -305,11 +305,11 @@ verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22
|
|
|
305
305
|
verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
|
|
306
306
|
verifiers/v1/serve/client.py,sha256=amUhf4cFJ1xCb8kw_AoVviwN543Wknx8LzImq1ZpaiQ,7057
|
|
307
307
|
verifiers/v1/serve/pool.py,sha256=e9Nc3nQKEyV9YqyKHl_RuqQ84zWVa8z2J1rc3rfCadM,14828
|
|
308
|
-
verifiers/v1/serve/server.py,sha256=
|
|
308
|
+
verifiers/v1/serve/server.py,sha256=IfJunpYX3MmNDToEcY-QHxDGfbZnK2i3C1-sgft6xfw,9687
|
|
309
309
|
verifiers/v1/serve/types.py,sha256=y8Cf9wKOZRKDXcThFWVygTFHITJofOinGiIzwjQGCUE,2674
|
|
310
310
|
verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
|
|
311
311
|
verifiers/v1/tasksets/harbor/__init__.py,sha256=wwVGA8N7E2Q2qA5WGHuLRrn4DgG5oaRHrCS2RW3NXVY,195
|
|
312
|
-
verifiers/v1/tasksets/harbor/taskset.py,sha256=
|
|
312
|
+
verifiers/v1/tasksets/harbor/taskset.py,sha256=w7o7yGM-FmvLEW4zNvwutvOX4n9cq8eQWwb7WxUYtSA,18235
|
|
313
313
|
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
314
314
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
315
315
|
verifiers/v1/tasksets/lean/taskset.py,sha256=Fcn4UgTcdjMYnJ4CThaFZ28Rz_QDx0e21vJdW_vp12g,9019
|
|
@@ -330,8 +330,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
|
|
|
330
330
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
331
331
|
verifiers/v1/utils/sampling.py,sha256=52TJ5-Hmpvp4YSgKNR-LmAPVwm4RDcH-o2-BnbVTmYQ,965
|
|
332
332
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
333
|
-
verifiers-0.2.2.
|
|
334
|
-
verifiers-0.2.2.
|
|
335
|
-
verifiers-0.2.2.
|
|
336
|
-
verifiers-0.2.2.
|
|
337
|
-
verifiers-0.2.2.
|
|
333
|
+
verifiers-0.2.2.dev55.dist-info/METADATA,sha256=Jcl9dMFMROAPjPXzDCMNCz_r43jDq4Xp2QjEZZ0Z9b4,4545
|
|
334
|
+
verifiers-0.2.2.dev55.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
335
|
+
verifiers-0.2.2.dev55.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
336
|
+
verifiers-0.2.2.dev55.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
337
|
+
verifiers-0.2.2.dev55.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|