verifiers 0.2.2.dev79__py3-none-any.whl → 0.2.2.dev81__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/eval/runner.py +6 -8
- verifiers/v1/cli/output.py +6 -2
- verifiers/v1/configs/judge.py +2 -2
- verifiers/v1/envs/agentic_judge/env.py +13 -82
- verifiers/v1/episode.py +16 -2
- verifiers/v1/harnesses/rlm/harness.py +5 -13
- verifiers/v1/interception/server.py +3 -3
- verifiers/v1/judges/rubric.py +76 -62
- verifiers/v1/legacy.py +13 -13
- verifiers/v1/session.py +8 -0
- verifiers/v1/trace.py +0 -9
- {verifiers-0.2.2.dev79.dist-info → verifiers-0.2.2.dev81.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev79.dist-info → verifiers-0.2.2.dev81.dist-info}/RECORD +16 -16
- {verifiers-0.2.2.dev79.dist-info → verifiers-0.2.2.dev81.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev79.dist-info → verifiers-0.2.2.dev81.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev79.dist-info → verifiers-0.2.2.dev81.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -10,7 +10,6 @@ from verifiers.v1.cli.dashboard import dashboard
|
|
|
10
10
|
from verifiers.v1.cli.eval import resume
|
|
11
11
|
from verifiers.v1.cli.output import (
|
|
12
12
|
append_episode,
|
|
13
|
-
append_trace,
|
|
14
13
|
output_path,
|
|
15
14
|
save_config,
|
|
16
15
|
)
|
|
@@ -77,8 +76,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
|
|
|
77
76
|
write_lock = asyncio.Lock()
|
|
78
77
|
|
|
79
78
|
async def on_complete(episode: Episode) -> None:
|
|
80
|
-
|
|
81
|
-
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
79
|
+
episode.record_run(EvalRunInfo(id=config.uuid))
|
|
82
80
|
await append_episode(out, episode, write_lock)
|
|
83
81
|
|
|
84
82
|
# Serving resources (shared tool servers, interception) come up once for the
|
|
@@ -258,9 +256,10 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
258
256
|
)
|
|
259
257
|
records = []
|
|
260
258
|
for trace in traces:
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
259
|
+
record = Episode.of(trace, env=config.env_id)
|
|
260
|
+
record.record_run(EvalRunInfo(id=config.uuid))
|
|
261
|
+
await append_episode(out, record, write_lock)
|
|
262
|
+
records.append(record)
|
|
264
263
|
return records
|
|
265
264
|
|
|
266
265
|
async def run_unit(payload: dict) -> list[Episode]:
|
|
@@ -271,8 +270,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
|
|
|
271
270
|
sampling=config.sampling,
|
|
272
271
|
**payload,
|
|
273
272
|
)
|
|
274
|
-
|
|
275
|
-
trace.record_run(EvalRunInfo(id=config.uuid))
|
|
273
|
+
episode.record_run(EvalRunInfo(id=config.uuid))
|
|
276
274
|
await append_episode(out, episode, write_lock)
|
|
277
275
|
return [episode]
|
|
278
276
|
|
verifiers/v1/cli/output.py
CHANGED
|
@@ -12,6 +12,7 @@ before the episode atom (one bare trace per line) are still readable:
|
|
|
12
12
|
|
|
13
13
|
import asyncio
|
|
14
14
|
import json
|
|
15
|
+
from functools import cache
|
|
15
16
|
from pathlib import Path
|
|
16
17
|
|
|
17
18
|
import tomli_w
|
|
@@ -29,6 +30,9 @@ TRACES_FILE = "traces.jsonl"
|
|
|
29
30
|
CONFIG_FILE = "config.toml"
|
|
30
31
|
"""Filename a run's resolved config is written to (re-runnable via `@ config.toml`)."""
|
|
31
32
|
|
|
33
|
+
# Compiling an adapter is the expensive part; run output reuses only a few model classes.
|
|
34
|
+
type_adapter = cache(TypeAdapter)
|
|
35
|
+
|
|
32
36
|
|
|
33
37
|
def output_path(config: EvalConfig) -> Path:
|
|
34
38
|
"""Where this run writes: `outputs/<env>--<model>--<harness>/<uuid>` (or the explicit
|
|
@@ -72,7 +76,7 @@ def save_config(config: BaseModel, results_dir: Path) -> None:
|
|
|
72
76
|
def write_episode(results_dir: Path, episode: Episode) -> None:
|
|
73
77
|
"""Serialize and append one rollout episode in the worker thread."""
|
|
74
78
|
# Preserve fields declared by typed Trace subclasses nested in the episode.
|
|
75
|
-
data =
|
|
79
|
+
data = type_adapter(type(episode)).dump_json(episode, exclude_none=True)
|
|
76
80
|
with (results_dir / TRACES_FILE).open("ab") as f:
|
|
77
81
|
f.write(data + b"\n")
|
|
78
82
|
|
|
@@ -88,7 +92,7 @@ def read_episodes(results_dir: Path, trace_type: type) -> list[Episode]:
|
|
|
88
92
|
`trace_type` (`Trace[WireTaskData, ...]` reads any taskset's file without
|
|
89
93
|
importing it). A pre-episode line (one bare trace) is wrapped as a single-trace
|
|
90
94
|
record, so both file generations read uniformly."""
|
|
91
|
-
trace_adapter =
|
|
95
|
+
trace_adapter = type_adapter(trace_type)
|
|
92
96
|
episodes: list[Episode] = []
|
|
93
97
|
with (results_dir / TRACES_FILE).open(encoding="utf-8") as f:
|
|
94
98
|
for line in f:
|
verifiers/v1/configs/judge.py
CHANGED
|
@@ -5,7 +5,7 @@ from collections.abc import Sequence
|
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
|
-
from pydantic import BaseModel, SerializeAsAny
|
|
8
|
+
from pydantic import BaseModel, FiniteFloat, SerializeAsAny
|
|
9
9
|
|
|
10
10
|
from verifiers.v1.clients import BaseClientConfig
|
|
11
11
|
from verifiers.v1.types import ID, SamplingConfig
|
|
@@ -17,7 +17,7 @@ class JudgeConfig(BaseClientConfig):
|
|
|
17
17
|
"""Plugin id; empty for a judge called directly by task code."""
|
|
18
18
|
name: str = ""
|
|
19
19
|
"""Reward key override for a plugged judge."""
|
|
20
|
-
weight:
|
|
20
|
+
weight: FiniteFloat = 1.0
|
|
21
21
|
model: str = "openai/gpt-5.4-nano"
|
|
22
22
|
sampling: SamplingConfig = SamplingConfig()
|
|
23
23
|
prompt: Path | None = None
|
|
@@ -14,14 +14,18 @@ task's collected artifacts.
|
|
|
14
14
|
"""
|
|
15
15
|
|
|
16
16
|
import json
|
|
17
|
-
import math
|
|
18
17
|
import re
|
|
19
|
-
import tomllib
|
|
20
18
|
from pathlib import Path
|
|
21
19
|
|
|
22
|
-
from pydantic import
|
|
20
|
+
from pydantic import FiniteFloat
|
|
23
21
|
|
|
24
22
|
import verifiers.v1 as vf
|
|
23
|
+
from verifiers.v1.judges.rubric import (
|
|
24
|
+
Criterion,
|
|
25
|
+
RubricVerdicts,
|
|
26
|
+
load_criteria,
|
|
27
|
+
score_verdicts,
|
|
28
|
+
)
|
|
25
29
|
from verifiers.v1.utils.compile import validate_pairing
|
|
26
30
|
|
|
27
31
|
VERDICT_FILE = "/tmp/verdict.json"
|
|
@@ -38,29 +42,6 @@ TASK_SECTION = """\
|
|
|
38
42
|
{prompt}"""
|
|
39
43
|
|
|
40
44
|
|
|
41
|
-
class Criterion(BaseModel):
|
|
42
|
-
"""One rubric criterion — the plugged rubric judge's format, mirrored so the
|
|
43
|
-
same `criteria` files grade both judges."""
|
|
44
|
-
|
|
45
|
-
name: str
|
|
46
|
-
"""Key for the criterion's metric (`judge/<name>`)."""
|
|
47
|
-
text: str
|
|
48
|
-
weight: float = 1.0
|
|
49
|
-
"""The criterion's share of the reward."""
|
|
50
|
-
choices: list[str] = Field(default_factory=lambda: ["no", "yes"])
|
|
51
|
-
"""Allowed answers, ordered **worst → best**: the first scores 0.0, the last 1.0, the rest
|
|
52
|
-
evenly spaced by rank. Default `["no", "yes"]` is a binary check. Needs >= 2, no duplicates."""
|
|
53
|
-
|
|
54
|
-
@field_validator("choices")
|
|
55
|
-
@classmethod
|
|
56
|
-
def _check_choices(cls, v: list[str]) -> list[str]:
|
|
57
|
-
if len(v) < 2:
|
|
58
|
-
raise ValueError(f"`choices` needs at least two options, got {v}")
|
|
59
|
-
if len(set(v)) != len(v):
|
|
60
|
-
raise ValueError(f"`choices` has duplicate options: {v}")
|
|
61
|
-
return v
|
|
62
|
-
|
|
63
|
-
|
|
64
45
|
SOLVED = Criterion(
|
|
65
46
|
name="solved",
|
|
66
47
|
text="The task is fully solved: what the task asked for is achieved, and "
|
|
@@ -210,7 +191,7 @@ class JudgeTask(vf.Task):
|
|
|
210
191
|
f"the judge wrote no verdict to {VERDICT_FILE}; its final act must "
|
|
211
192
|
'be writing {"verdicts": [{"name", "reason", "verdict"}, ...]} there'
|
|
212
193
|
) from e
|
|
213
|
-
trace.info["verdict"] =
|
|
194
|
+
trace.info["verdict"] = RubricVerdicts.model_validate_json(raw).model_dump()
|
|
214
195
|
|
|
215
196
|
|
|
216
197
|
class TextFile(vf.BaseConfig):
|
|
@@ -259,38 +240,16 @@ class JudgeTaskConfig(vf.BaseConfig):
|
|
|
259
240
|
def criteria(self) -> list[Criterion]:
|
|
260
241
|
if self.rubric is None:
|
|
261
242
|
return [SOLVED]
|
|
262
|
-
|
|
263
|
-
data = (
|
|
264
|
-
tomllib.loads(text)
|
|
265
|
-
if self.rubric.suffix.lower() == ".toml"
|
|
266
|
-
else json.loads(text)
|
|
267
|
-
)
|
|
268
|
-
items = data.get("criteria", []) if isinstance(data, dict) else data
|
|
269
|
-
criteria = [Criterion.model_validate(item) for item in items]
|
|
270
|
-
if not criteria:
|
|
271
|
-
raise ValueError(f"rubric file '{self.rubric}' lists no criteria")
|
|
272
|
-
names = [criterion.name for criterion in criteria]
|
|
273
|
-
if len(set(names)) != len(names):
|
|
274
|
-
raise ValueError(
|
|
275
|
-
f"rubric file '{self.rubric}' has duplicate criterion names"
|
|
276
|
-
)
|
|
277
|
-
if bad := [c.name for c in criteria if not 0 <= c.weight < math.inf]:
|
|
278
|
-
raise ValueError(
|
|
279
|
-
f"rubric '{self.rubric}' has negative or non-finite criterion "
|
|
280
|
-
f"weights: {bad}"
|
|
281
|
-
)
|
|
282
|
-
if sum(criterion.weight for criterion in criteria) <= 0:
|
|
283
|
-
raise ValueError(f"rubric '{self.rubric}' has no positive criterion weight")
|
|
284
|
-
return criteria
|
|
243
|
+
return load_criteria(self.rubric)
|
|
285
244
|
|
|
286
245
|
|
|
287
246
|
class ScoreConfig(vf.BaseConfig):
|
|
288
247
|
"""How the judge's verdict composes with the taskset's own rewards on the
|
|
289
248
|
solver's trace. Judge-only by default."""
|
|
290
249
|
|
|
291
|
-
task_weight:
|
|
250
|
+
task_weight: FiniteFloat = 0.0
|
|
292
251
|
"""Scale applied to the taskset's own rewards; 1 keeps them next to the verdict."""
|
|
293
|
-
judge_weight:
|
|
252
|
+
judge_weight: FiniteFloat = 1.0
|
|
294
253
|
"""Weight of the judge's verdict in the solver's reward."""
|
|
295
254
|
|
|
296
255
|
|
|
@@ -363,37 +322,9 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
363
322
|
if "judge" not in by_agent:
|
|
364
323
|
return
|
|
365
324
|
solution, verdict = by_agent["solver"], by_agent["judge"]
|
|
366
|
-
|
|
367
|
-
if not isinstance(data, dict) or not isinstance(data.get("verdicts"), list):
|
|
368
|
-
raise TypeError(
|
|
369
|
-
f"no verdicts on the judge's trace (expected {VERDICT_FILE} with a "
|
|
370
|
-
'"verdicts" list)'
|
|
371
|
-
)
|
|
325
|
+
verdicts = RubricVerdicts.model_validate(verdict.info.get("verdict")).verdicts
|
|
372
326
|
criteria = self.config.task.criteria()
|
|
373
|
-
|
|
374
|
-
answers: dict[str, str] = {}
|
|
375
|
-
for entry in data["verdicts"]:
|
|
376
|
-
if not isinstance(entry, dict):
|
|
377
|
-
raise TypeError(f"verdict entry {entry!r} is not an object")
|
|
378
|
-
name = str(entry.get("name"))
|
|
379
|
-
if name in answers:
|
|
380
|
-
# Contradictory duplicates must not collapse to whichever came last.
|
|
381
|
-
raise ValueError(f"judge answered criterion {name!r} more than once")
|
|
382
|
-
answers[name] = str(entry.get("verdict"))
|
|
383
|
-
if sorted(answers) != sorted(by_criterion):
|
|
384
|
-
raise ValueError(
|
|
385
|
-
f"judge verdicts name {sorted(answers)}, expected the rubric's "
|
|
386
|
-
f"{sorted(by_criterion)}"
|
|
387
|
-
)
|
|
388
|
-
scores: dict[str, float] = {}
|
|
389
|
-
for name, answer in answers.items():
|
|
390
|
-
choices = by_criterion[name].choices
|
|
391
|
-
# An off-menu answer is a judge failure, not a zero score.
|
|
392
|
-
if answer not in choices:
|
|
393
|
-
raise ValueError(
|
|
394
|
-
f"judge answered {answer!r} for '{name}', expected one of {choices}"
|
|
395
|
-
)
|
|
396
|
-
scores[name] = choices.index(answer) / (len(choices) - 1)
|
|
327
|
+
scores = score_verdicts(verdicts, criteria, "the rubric's")
|
|
397
328
|
for criterion in criteria:
|
|
398
329
|
solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
|
|
399
330
|
if self.config.score.task_weight != 1.0:
|
verifiers/v1/episode.py
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
"""The episode — one run's traces plus their shared standing, whole."""
|
|
2
2
|
|
|
3
3
|
import uuid
|
|
4
|
-
from typing import Generic
|
|
4
|
+
from typing import Any, Generic
|
|
5
5
|
|
|
6
6
|
from pydantic import BaseModel, Field
|
|
7
7
|
|
|
8
8
|
from verifiers.v1.configs.agent import WireAgentConfig
|
|
9
9
|
from verifiers.v1.state import State, StateT
|
|
10
10
|
from verifiers.v1.task import DataT, WireTaskData
|
|
11
|
-
from verifiers.v1.trace import AgentConfigT, Error, Trace
|
|
11
|
+
from verifiers.v1.trace import AgentConfigT, Error, RunInfo, Trace
|
|
12
12
|
from verifiers.v1.types import Usage
|
|
13
13
|
|
|
14
14
|
|
|
@@ -26,12 +26,19 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
26
26
|
|
|
27
27
|
env: EnvInfo = Field(default_factory=EnvInfo)
|
|
28
28
|
"""The env that produced this episode."""
|
|
29
|
+
run: RunInfo | None = None
|
|
30
|
+
"""The run this episode belongs to (eval or train), consumer-stamped. It lives here rather than
|
|
31
|
+
on each trace because the episode is what a consumer dispatches, and an episode that produced
|
|
32
|
+
no traces would otherwise have nowhere to say which run it was."""
|
|
29
33
|
ok: bool = False
|
|
30
34
|
"""Whether the episode completed successfully."""
|
|
31
35
|
errors: list[Error] = Field(default_factory=list)
|
|
32
36
|
"""Every error captured across attempts, oldest to newest."""
|
|
33
37
|
traces: list[Trace[DataT, StateT, AgentConfigT]] = Field(default_factory=list)
|
|
34
38
|
"""Every agent's trace, in completion order."""
|
|
39
|
+
info: dict[str, Any] = Field(default_factory=dict)
|
|
40
|
+
"""Scratch space for episode-level metadata, the counterpart to `Trace.info`. What describes
|
|
41
|
+
the whole episode belongs here rather than repeated on each of its traces."""
|
|
35
42
|
|
|
36
43
|
@property
|
|
37
44
|
def last_error(self) -> Error | None:
|
|
@@ -72,6 +79,13 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
72
79
|
grouped.setdefault(trace.agent.name, []).append(trace)
|
|
73
80
|
return grouped
|
|
74
81
|
|
|
82
|
+
def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
|
|
83
|
+
"""Record the run identity and any extra metadata about this episode. Both describe the
|
|
84
|
+
episode as a whole, so they are recorded once here rather than repeated on every trace."""
|
|
85
|
+
if run is not None:
|
|
86
|
+
self.run = run
|
|
87
|
+
self.info.update(info)
|
|
88
|
+
|
|
75
89
|
@classmethod
|
|
76
90
|
def of(cls, trace: Trace, env: str = "") -> "Episode":
|
|
77
91
|
"""The single-agent record: one trace as its own episode."""
|
|
@@ -6,7 +6,7 @@ import random
|
|
|
6
6
|
import shlex
|
|
7
7
|
from typing import Literal
|
|
8
8
|
|
|
9
|
-
from pydantic import Field, model_validator
|
|
9
|
+
from pydantic import Field, PositiveInt, model_validator
|
|
10
10
|
|
|
11
11
|
from verifiers.v1.clients import ModelContext
|
|
12
12
|
from verifiers.v1.configs.harness import HarnessConfig
|
|
@@ -38,26 +38,18 @@ class RLMHarnessConfig(HarnessConfig):
|
|
|
38
38
|
builtin_skills: list[BuiltinSkill] = Field(default_factory=list)
|
|
39
39
|
"""Built-in rlm skills to enable (RLM_SKILLS), e.g. `["edit"]`; empty enables none.
|
|
40
40
|
The tool set is fixed (ipython); the base `skills` field takes SKILL.md paths."""
|
|
41
|
-
summarize_at_tokens:
|
|
41
|
+
summarize_at_tokens: PositiveInt | tuple[PositiveInt, PositiveInt] | None = None
|
|
42
42
|
"""Auto-compaction threshold (RLM_SUMMARIZE_AT_TOKENS): compact the context once it grows
|
|
43
43
|
past this many tokens. An int is a fixed threshold; a `(lo, hi)` pair draws a per-group
|
|
44
44
|
threshold (seeded by the task index, so a task's rollouts share one draw and tasks vary).
|
|
45
45
|
`None` disables auto-compaction; ints must be positive."""
|
|
46
46
|
|
|
47
47
|
@model_validator(mode="after")
|
|
48
|
-
def
|
|
48
|
+
def validate_range(self) -> "RLMHarnessConfig":
|
|
49
49
|
value = self.summarize_at_tokens
|
|
50
|
-
if isinstance(value, tuple):
|
|
51
|
-
lo, hi = value
|
|
52
|
-
if lo <= 0 or hi <= 0:
|
|
53
|
-
raise ValueError("`summarize_at_tokens` range bounds must be positive.")
|
|
54
|
-
if lo > hi:
|
|
55
|
-
raise ValueError(
|
|
56
|
-
"`summarize_at_tokens` range must be (lo, hi) with lo <= hi."
|
|
57
|
-
)
|
|
58
|
-
elif value is not None and value <= 0:
|
|
50
|
+
if isinstance(value, tuple) and value[0] > value[1]:
|
|
59
51
|
raise ValueError(
|
|
60
|
-
"`summarize_at_tokens` must be
|
|
52
|
+
"`summarize_at_tokens` range must be (lo, hi) with lo <= hi."
|
|
61
53
|
)
|
|
62
54
|
return self
|
|
63
55
|
|
|
@@ -30,7 +30,7 @@ from contextlib import asynccontextmanager
|
|
|
30
30
|
from typing import Literal
|
|
31
31
|
|
|
32
32
|
from aiohttp import web
|
|
33
|
-
from pydantic import
|
|
33
|
+
from pydantic import ValidationError
|
|
34
34
|
from pydantic_core import PydanticSerializationError, from_json, to_json
|
|
35
35
|
|
|
36
36
|
from verifiers.v1 import graph
|
|
@@ -737,7 +737,7 @@ class InterceptionServer(Interception):
|
|
|
737
737
|
state = session.trace.state
|
|
738
738
|
return web.Response(
|
|
739
739
|
# TypeAdapter emits UTF-8 bytes directly, avoiding a JSON str copy in aiohttp.
|
|
740
|
-
body=
|
|
740
|
+
body=session.state_adapter.dump_json(state),
|
|
741
741
|
content_type="application/json",
|
|
742
742
|
charset="utf-8",
|
|
743
743
|
)
|
|
@@ -768,7 +768,7 @@ class InterceptionServer(Interception):
|
|
|
768
768
|
state_cls = type(session.trace.state)
|
|
769
769
|
raw = await request.read()
|
|
770
770
|
try:
|
|
771
|
-
new_state =
|
|
771
|
+
new_state = session.state_adapter.validate_json(raw)
|
|
772
772
|
except ValidationError as e:
|
|
773
773
|
# Reject malformed, over-nested, or mismatched state before it enters the shared channel.
|
|
774
774
|
logger.warning("state PUT rejected: id=%s %s", session.trace.id, e)
|
verifiers/v1/judges/rubric.py
CHANGED
|
@@ -2,14 +2,13 @@
|
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import json
|
|
5
|
-
import math
|
|
6
5
|
import re
|
|
7
6
|
import tomllib
|
|
8
7
|
from functools import cached_property
|
|
9
8
|
from pathlib import Path
|
|
10
|
-
from typing import cast
|
|
9
|
+
from typing import Annotated, cast
|
|
11
10
|
|
|
12
|
-
from pydantic import BaseModel, Field, field_validator
|
|
11
|
+
from pydantic import BaseModel, Field, TypeAdapter, field_validator
|
|
13
12
|
|
|
14
13
|
from verifiers.v1.configs.judge import JudgeConfig
|
|
15
14
|
from verifiers.v1.judge import Judge, JudgeView, judge_question, judge_response
|
|
@@ -17,6 +16,8 @@ from verifiers.v1.task import TaskData
|
|
|
17
16
|
from verifiers.v1.trace import Trace
|
|
18
17
|
from verifiers.v1.types import ID
|
|
19
18
|
|
|
19
|
+
CriterionWeight = Annotated[float, Field(ge=0, allow_inf_nan=False)]
|
|
20
|
+
|
|
20
21
|
RUBRIC_PROMPT = (Path(__file__).resolve().parent / "rubric.txt").read_text(
|
|
21
22
|
encoding="utf-8"
|
|
22
23
|
)
|
|
@@ -60,29 +61,62 @@ class Criterion(BaseModel):
|
|
|
60
61
|
name: str
|
|
61
62
|
"""Key for the criterion's metric (`<judge name>/<name>`) and its `weights` override."""
|
|
62
63
|
text: str
|
|
63
|
-
weight:
|
|
64
|
+
weight: CriterionWeight = 1.0
|
|
64
65
|
"""The criterion's share of the reward (overridable per name via `weights` in config)."""
|
|
65
|
-
choices: list[str] = Field(default_factory=lambda: ["no", "yes"])
|
|
66
|
+
choices: list[str] = Field(default_factory=lambda: ["no", "yes"], min_length=2)
|
|
66
67
|
"""Allowed answers, ordered **worst → best**: the first scores 0.0, the last 1.0, the rest
|
|
67
68
|
evenly spaced by rank. Default `["no", "yes"]` is a binary check. Needs >= 2, no duplicates."""
|
|
68
69
|
|
|
69
70
|
@field_validator("choices")
|
|
70
71
|
@classmethod
|
|
71
72
|
def _check_choices(cls, v: list[str]) -> list[str]:
|
|
72
|
-
if len(v) < 2:
|
|
73
|
-
raise ValueError(f"`choices` needs at least two options, got {v}")
|
|
74
73
|
if len(set(v)) != len(v):
|
|
75
74
|
raise ValueError(f"`choices` has duplicate options: {v}")
|
|
76
75
|
return v
|
|
77
76
|
|
|
78
77
|
|
|
78
|
+
CRITERIA_ADAPTER = TypeAdapter(list[Criterion])
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def load_criteria(
|
|
82
|
+
path: Path, weights: dict[str, CriterionWeight] | None = None
|
|
83
|
+
) -> list[Criterion]:
|
|
84
|
+
"""Load a JSON or TOML rubric and apply validated config overrides."""
|
|
85
|
+
text = path.read_text(encoding="utf-8")
|
|
86
|
+
data = tomllib.loads(text) if path.suffix.lower() == ".toml" else json.loads(text)
|
|
87
|
+
items = data.get("criteria", []) if isinstance(data, dict) else data
|
|
88
|
+
criteria = CRITERIA_ADAPTER.validate_python(items)
|
|
89
|
+
if not criteria:
|
|
90
|
+
raise ValueError(f"rubric file '{path}' lists no criteria")
|
|
91
|
+
names = [criterion.name for criterion in criteria]
|
|
92
|
+
if len(set(names)) != len(names):
|
|
93
|
+
raise ValueError(f"rubric file '{path}' has duplicate criterion names")
|
|
94
|
+
overrides = weights or {}
|
|
95
|
+
if unknown := set(overrides) - set(names):
|
|
96
|
+
raise ValueError(
|
|
97
|
+
f"`weights` overrides name no criterion in '{path}': {sorted(unknown)}"
|
|
98
|
+
)
|
|
99
|
+
criteria = [
|
|
100
|
+
criterion.model_copy(
|
|
101
|
+
update={"weight": overrides.get(criterion.name, criterion.weight)}
|
|
102
|
+
)
|
|
103
|
+
for criterion in criteria
|
|
104
|
+
]
|
|
105
|
+
total = sum(criterion.weight for criterion in criteria)
|
|
106
|
+
if total == float("inf"):
|
|
107
|
+
raise ValueError(f"rubric '{path}' has a non-finite total criterion weight")
|
|
108
|
+
if total <= 0:
|
|
109
|
+
raise ValueError(f"rubric '{path}' has no positive criterion weight")
|
|
110
|
+
return criteria
|
|
111
|
+
|
|
112
|
+
|
|
79
113
|
class RubricJudgeConfig(JudgeConfig):
|
|
80
114
|
id: ID = "rubric"
|
|
81
115
|
"""Pinned to the built-in, so a code-level default entry needs no explicit id."""
|
|
82
116
|
path: Path
|
|
83
117
|
"""A `.toml` or `.json` file containing a `criteria` list. Relative paths resolve
|
|
84
118
|
against the evaluation's working directory."""
|
|
85
|
-
weights: dict[str,
|
|
119
|
+
weights: dict[str, CriterionWeight] = Field(default_factory=dict)
|
|
86
120
|
"""Per-criterion weight overrides by criterion name (config wins over the file)."""
|
|
87
121
|
question_field: str = ""
|
|
88
122
|
"""Task field to fill the prompt's `{question}`; empty = the task's prompt rendered as
|
|
@@ -96,7 +130,7 @@ class RubricJudgeConfig(JudgeConfig):
|
|
|
96
130
|
"""How much of the rollout fills `{response}` (see `JudgeView`). Defaults to the whole
|
|
97
131
|
transcript — rubric criteria typically grade the process (tool use, citations,
|
|
98
132
|
intermediate steps), not just the final answer."""
|
|
99
|
-
max_criteria: int | None = None
|
|
133
|
+
max_criteria: int | None = Field(default=None, ge=1)
|
|
100
134
|
"""How many criteria to grade per judge call. `None` (default) grades all criteria in one
|
|
101
135
|
call. `1` sends one call per criterion (n independent judges); `k` batches them k-at-a-time.
|
|
102
136
|
Batches are graded concurrently and merged. Smaller batches trade more calls for focus/
|
|
@@ -119,45 +153,43 @@ class RubricVerdicts(BaseModel):
|
|
|
119
153
|
verdicts: list[CriterionVerdict]
|
|
120
154
|
|
|
121
155
|
|
|
156
|
+
def score_verdicts(
|
|
157
|
+
verdicts: list[CriterionVerdict],
|
|
158
|
+
criteria: list[Criterion],
|
|
159
|
+
expected: str,
|
|
160
|
+
) -> dict[str, float]:
|
|
161
|
+
"""Validate a complete named verdict set and normalize its ordered choices."""
|
|
162
|
+
by_criterion = {criterion.name: criterion for criterion in criteria}
|
|
163
|
+
answers: dict[str, str] = {}
|
|
164
|
+
for verdict in verdicts:
|
|
165
|
+
if verdict.name in answers:
|
|
166
|
+
raise ValueError(
|
|
167
|
+
f"judge answered criterion {verdict.name!r} more than once"
|
|
168
|
+
)
|
|
169
|
+
answers[verdict.name] = verdict.verdict
|
|
170
|
+
if sorted(answers) != sorted(by_criterion):
|
|
171
|
+
raise ValueError(
|
|
172
|
+
f"judge verdicts name {sorted(answers)}, expected {expected} "
|
|
173
|
+
f"{sorted(by_criterion)}"
|
|
174
|
+
)
|
|
175
|
+
scores: dict[str, float] = {}
|
|
176
|
+
for name, answer in answers.items():
|
|
177
|
+
choices = by_criterion[name].choices
|
|
178
|
+
if answer not in choices:
|
|
179
|
+
raise ValueError(
|
|
180
|
+
f"judge answered {answer!r} for '{name}', expected one of {choices}"
|
|
181
|
+
)
|
|
182
|
+
scores[name] = normalize_choice(answer, choices)
|
|
183
|
+
return scores
|
|
184
|
+
|
|
185
|
+
|
|
122
186
|
class RubricJudge(Judge[RubricVerdicts, RubricJudgeConfig]):
|
|
123
187
|
prompt = RUBRIC_PROMPT
|
|
124
188
|
schema = RubricVerdicts
|
|
125
189
|
|
|
126
190
|
@cached_property
|
|
127
191
|
def criteria(self) -> list[Criterion]:
|
|
128
|
-
path
|
|
129
|
-
text = path.read_text(encoding="utf-8")
|
|
130
|
-
data = (
|
|
131
|
-
tomllib.loads(text) if path.suffix.lower() == ".toml" else json.loads(text)
|
|
132
|
-
)
|
|
133
|
-
items = data.get("criteria", []) if isinstance(data, dict) else data
|
|
134
|
-
criteria = [Criterion.model_validate(item) for item in items]
|
|
135
|
-
if not criteria:
|
|
136
|
-
raise ValueError(f"rubric file '{path}' lists no criteria")
|
|
137
|
-
names = [criterion.name for criterion in criteria]
|
|
138
|
-
if len(set(names)) != len(names):
|
|
139
|
-
raise ValueError(f"rubric file '{path}' has duplicate criterion names")
|
|
140
|
-
if unknown := set(self.config.weights) - set(names):
|
|
141
|
-
raise ValueError(
|
|
142
|
-
f"`weights` overrides name no criterion in '{path}': {sorted(unknown)}"
|
|
143
|
-
)
|
|
144
|
-
criteria = [
|
|
145
|
-
criterion.model_copy(
|
|
146
|
-
update={
|
|
147
|
-
"weight": self.config.weights.get(criterion.name, criterion.weight)
|
|
148
|
-
}
|
|
149
|
-
)
|
|
150
|
-
for criterion in criteria
|
|
151
|
-
]
|
|
152
|
-
if bad := [c.name for c in criteria if not 0 <= c.weight < math.inf]:
|
|
153
|
-
# A negative weight would invert a criterion (pushing the reward out of [0, 1]);
|
|
154
|
-
# NaN/inf (which json.loads accepts) would corrupt the weighted mean.
|
|
155
|
-
raise ValueError(
|
|
156
|
-
f"rubric '{path}' has negative or non-finite criterion weights: {bad}"
|
|
157
|
-
)
|
|
158
|
-
if sum(criterion.weight for criterion in criteria) <= 0:
|
|
159
|
-
raise ValueError(f"rubric '{path}' has no positive criterion weight")
|
|
160
|
-
return criteria
|
|
192
|
+
return load_criteria(self.config.path, self.config.weights)
|
|
161
193
|
|
|
162
194
|
async def grade_batch(
|
|
163
195
|
self, task: TaskData, trace: Trace, batch: list[Criterion]
|
|
@@ -205,30 +237,12 @@ class RubricJudge(Judge[RubricVerdicts, RubricJudgeConfig]):
|
|
|
205
237
|
f"judge returned no verdicts JSON object: {result.text!r}"
|
|
206
238
|
)
|
|
207
239
|
verdicts = RubricVerdicts.model_validate(obj).verdicts
|
|
208
|
-
#
|
|
209
|
-
|
|
210
|
-
by_criterion = {c.name: c for c in batch}
|
|
211
|
-
if sorted(v.name for v in verdicts) != sorted(by_criterion):
|
|
212
|
-
raise ValueError(
|
|
213
|
-
f"judge verdicts name {sorted(v.name for v in verdicts)}, expected the "
|
|
214
|
-
f"batch's {sorted(by_criterion)}"
|
|
215
|
-
)
|
|
216
|
-
scores: dict[str, float] = {}
|
|
217
|
-
for v in verdicts:
|
|
218
|
-
choices = by_criterion[v.name].choices
|
|
219
|
-
# An off-menu answer is a judge failure, not a zero score.
|
|
220
|
-
if v.verdict not in choices:
|
|
221
|
-
raise ValueError(
|
|
222
|
-
f"judge answered {v.verdict!r} for '{v.name}', expected one of {choices}"
|
|
223
|
-
)
|
|
224
|
-
scores[v.name] = normalize_choice(v.verdict, choices)
|
|
225
|
-
return scores
|
|
240
|
+
# A malformed verdict is a judge failure and must error the rollout, not score the model.
|
|
241
|
+
return score_verdicts(verdicts, batch, "the batch's")
|
|
226
242
|
|
|
227
243
|
async def score(self, task: TaskData, trace: Trace) -> float:
|
|
228
244
|
criteria = self.criteria
|
|
229
245
|
k = self.config.max_criteria
|
|
230
|
-
if k is not None and k < 1:
|
|
231
|
-
raise ValueError(f"`max_criteria` must be >= 1 or None, got {k}")
|
|
232
246
|
batches = (
|
|
233
247
|
[criteria]
|
|
234
248
|
if k is None
|
verifiers/v1/legacy.py
CHANGED
|
@@ -16,11 +16,11 @@ v1 stays importable without the v0 package present.
|
|
|
16
16
|
import contextlib
|
|
17
17
|
import logging
|
|
18
18
|
from pathlib import Path
|
|
19
|
-
from typing import Any
|
|
19
|
+
from typing import Annotated, Any
|
|
20
20
|
|
|
21
21
|
import zmq
|
|
22
22
|
import zmq.asyncio
|
|
23
|
-
from pydantic import ValidationError
|
|
23
|
+
from pydantic import BeforeValidator, OnErrorOmit, TypeAdapter, ValidationError
|
|
24
24
|
|
|
25
25
|
from verifiers.v1 import graph
|
|
26
26
|
from verifiers.v1.configs.agent import AgentConfig
|
|
@@ -76,19 +76,19 @@ def _as_dict(obj: Any) -> Any:
|
|
|
76
76
|
return obj
|
|
77
77
|
|
|
78
78
|
|
|
79
|
+
TOOLS_ADAPTER = TypeAdapter(
|
|
80
|
+
list[OnErrorOmit[Annotated[Tool, BeforeValidator(_as_dict)]]]
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
79
84
|
def _to_v1_tools(raw: Any) -> list[Tool] | None:
|
|
80
85
|
"""Map v0 ``RolloutOutput.tool_defs`` onto ``Trace.tools``. The v0 and v1 ``Tool``
|
|
81
86
|
shapes are identical (name/description/parameters/strict), so this is a re-validation;
|
|
82
87
|
malformed entries are dropped rather than failing the whole trace mapping."""
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
continue
|
|
88
|
-
try:
|
|
89
|
-
defs.append(Tool.model_validate(t))
|
|
90
|
-
except ValidationError:
|
|
91
|
-
continue
|
|
88
|
+
try:
|
|
89
|
+
defs = TOOLS_ADAPTER.validate_python(raw or [])
|
|
90
|
+
except ValidationError:
|
|
91
|
+
return None
|
|
92
92
|
return defs or None
|
|
93
93
|
|
|
94
94
|
|
|
@@ -276,8 +276,8 @@ def rollout_output_to_trace(out: dict, task_idx: int) -> Trace:
|
|
|
276
276
|
# base task type above.
|
|
277
277
|
agent=AgentInfo(config=AgentConfig()),
|
|
278
278
|
tools=_to_v1_tools(out.get("tool_defs")) or [],
|
|
279
|
-
rewards={"reward": Reward(score=
|
|
280
|
-
metrics=
|
|
279
|
+
rewards={"reward": Reward(score=out.get("reward") or 0.0)},
|
|
280
|
+
metrics=out.get("metrics") or {},
|
|
281
281
|
info=dict(out.get("info") or {}),
|
|
282
282
|
is_completed=bool(out.get("is_completed", True)),
|
|
283
283
|
# Bridged rollouts are complete by construction; the sentinel mirrors
|
verifiers/v1/session.py
CHANGED
|
@@ -11,8 +11,11 @@ import asyncio
|
|
|
11
11
|
import logging
|
|
12
12
|
from collections.abc import Awaitable, Callable
|
|
13
13
|
from dataclasses import dataclass, field
|
|
14
|
+
from functools import cached_property
|
|
14
15
|
from typing import TYPE_CHECKING
|
|
15
16
|
|
|
17
|
+
from pydantic import TypeAdapter
|
|
18
|
+
|
|
16
19
|
from verifiers.v1.clients import Client, ModelContext
|
|
17
20
|
from verifiers.v1.trace import Trace
|
|
18
21
|
|
|
@@ -96,6 +99,11 @@ class RolloutSession:
|
|
|
96
99
|
its client disconnects, so a request whose program died at teardown would keep driving
|
|
97
100
|
the exchange (upstream call, simulator turn) — unregistering cancels these instead."""
|
|
98
101
|
|
|
102
|
+
@cached_property
|
|
103
|
+
def state_adapter(self) -> TypeAdapter:
|
|
104
|
+
"""The rollout's state codec, built only when a state channel is used."""
|
|
105
|
+
return TypeAdapter(type(self.trace.state))
|
|
106
|
+
|
|
99
107
|
def adopt(self, task: "asyncio.Task | None") -> None:
|
|
100
108
|
"""Track a handler task serving this session, for cancellation at release.
|
|
101
109
|
Callers adopt in the same synchronous stretch that fetched the session, so
|
verifiers/v1/trace.py
CHANGED
|
@@ -304,9 +304,6 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
304
304
|
"""Unique ID for this trace, auto-generated."""
|
|
305
305
|
verifiers: VersionInfo = Field(default_factory=_current_build)
|
|
306
306
|
"""The verifiers version that produced this trace."""
|
|
307
|
-
run: RunInfo | None = None
|
|
308
|
-
"""The run this trace belongs to (eval or train), consumer-stamped."""
|
|
309
|
-
|
|
310
307
|
task: TraceTask[DataT]
|
|
311
308
|
"""The task data that seeded this trace."""
|
|
312
309
|
agent: AgentInfo[AgentConfigT]
|
|
@@ -480,12 +477,6 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
480
477
|
if response.usage is not None:
|
|
481
478
|
self.extra_usage.append(response.usage)
|
|
482
479
|
|
|
483
|
-
def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
|
|
484
|
-
"""Record the run identity (eval / train), and optional extra info."""
|
|
485
|
-
if run is not None:
|
|
486
|
-
self.run = run
|
|
487
|
-
self.info.update(info)
|
|
488
|
-
|
|
489
480
|
def stop(self, condition: str) -> None:
|
|
490
481
|
"""Stop the trace, optionally with a stop condition."""
|
|
491
482
|
self.is_completed = True
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev81
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -170,18 +170,18 @@ verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2
|
|
|
170
170
|
verifiers/v1/__init__.py,sha256=v6qrJQFeNL2b4wd0PsDWiituw2Y-f5rMaFmdbdP6SdM,7505
|
|
171
171
|
verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
|
|
172
172
|
verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
|
|
173
|
-
verifiers/v1/episode.py,sha256=
|
|
173
|
+
verifiers/v1/episode.py,sha256=8G5OaGOku9S_gavQdpfUZgRZfrf1zmpjFrBL-TsppIk,4123
|
|
174
174
|
verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
|
|
175
175
|
verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
|
|
176
176
|
verifiers/v1/harness.py,sha256=eKuqvYbQjnCMavkiYWZQDHJgH08m6SJ4PuevPuShjS0,11097
|
|
177
177
|
verifiers/v1/judge.py,sha256=Pvr0C41ah1qkNSZ0WXdkDvBMfv5YSfY1REHncwlJ8cg,9346
|
|
178
|
-
verifiers/v1/legacy.py,sha256=
|
|
178
|
+
verifiers/v1/legacy.py,sha256=YvcWF6d8xvkeqUPM36oif0ZrjmB0mj7nNrfwha0Nr2w,22850
|
|
179
179
|
verifiers/v1/rollout.py,sha256=s1f1VZAAAI89j3y9ceq_E4mvX0F7cFQAnyx27SZsjvA,17881
|
|
180
|
-
verifiers/v1/session.py,sha256=
|
|
180
|
+
verifiers/v1/session.py,sha256=p8vz89DJUmb9r8dkq80zn2WBEqo8Vwh8ot1ZXm_WVAM,7210
|
|
181
181
|
verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
|
|
182
182
|
verifiers/v1/task.py,sha256=jwMKiKlMtksTd8j2dcDlFtb6LC8BRE-N5jkNXlMp2jc,9457
|
|
183
183
|
verifiers/v1/taskset.py,sha256=fp2E0IEhL_Ybj9cegZwljfTmW27p_30zTWHFKw_kXE4,4374
|
|
184
|
-
verifiers/v1/trace.py,sha256=
|
|
184
|
+
verifiers/v1/trace.py,sha256=esJE1aT9jRCbBad-BVR9xjX5Emk4mHdYmC7Zct-3zK8,19129
|
|
185
185
|
verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
|
|
186
186
|
verifiers/v1/acp/__init__.py,sha256=9SYCFtzGUM_wH98ldpVLcFhoq2M999jbG0tU5ODY70U,2297
|
|
187
187
|
verifiers/v1/acp/_runner.py,sha256=BcYaNZHuawzCgxOVdhiF6PY_B1HMxIA1MwHy0mOzhSk,7740
|
|
@@ -189,7 +189,7 @@ verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,
|
|
|
189
189
|
verifiers/v1/cli/debug.py,sha256=WGCiapH2CCwyPfESca5623-_csQv93PoglR8yX5-D3g,11488
|
|
190
190
|
verifiers/v1/cli/gepa.py,sha256=hLIbS_4jBx0vvy4PXoePR5riFXcNvzQCMixZ-g2OR_U,4026
|
|
191
191
|
verifiers/v1/cli/init.py,sha256=Qj96Kwzfhaejhrc0yYqHNRl3ncoID028RGlHvS1C98Q,8508
|
|
192
|
-
verifiers/v1/cli/output.py,sha256=
|
|
192
|
+
verifiers/v1/cli/output.py,sha256=JiNjMO2kwM4p3zMPg3k4spuJbbE7-jOAtYBttPap7qI,6045
|
|
193
193
|
verifiers/v1/cli/replay.py,sha256=DAh9kB5PB-ppZLGhxdJwrCPch_Yky5HKhplwGFhKEz0,10164
|
|
194
194
|
verifiers/v1/cli/resolve.py,sha256=OsNw4C3r9xOVFa-mminJFcex_bAsJ6EQ8K7A_H8u2kQ,4370
|
|
195
195
|
verifiers/v1/cli/validate.py,sha256=7ax6FqIzNBSYjd3JXK4v7Vbq34Rozyh1OvGpYnGqlSg,10266
|
|
@@ -201,7 +201,7 @@ verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8m
|
|
|
201
201
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
202
202
|
verifiers/v1/cli/eval/main.py,sha256=xiMMIIbsy77SEOryHXOQyswS9WyrPKTyBwW65W9WTjI,5511
|
|
203
203
|
verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
|
|
204
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
204
|
+
verifiers/v1/cli/eval/runner.py,sha256=oAMWf6uwuS_tEyE6OY_Q0-9_TTVCduKC-RJu797QFEc,12117
|
|
205
205
|
verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
|
|
206
206
|
verifiers/v1/clients/base.py,sha256=hhtTumKC7lgI6cj7Hks91DCjqNTwAh40DD_j8CWGd6c,1560
|
|
207
207
|
verifiers/v1/clients/client.py,sha256=L4axOl-dF9_MSYagB_xAO5qbeE53jNZJnB39VLdVBwo,3156
|
|
@@ -212,7 +212,7 @@ verifiers/v1/configs/agent.py,sha256=V_2FDvPHRB3_bHN9QnNnDYOjlmTP7NtG22SaopU-X64
|
|
|
212
212
|
verifiers/v1/configs/client.py,sha256=dCi2aC_wVHCrM0xRi9foFcbqp9pN2GR7yxEs1z_ChJM,4371
|
|
213
213
|
verifiers/v1/configs/env.py,sha256=7ESsWZRWJpKqR5EEV1rmj0R1Dkn2Pn8MVSXPJOzRiF8,7885
|
|
214
214
|
verifiers/v1/configs/harness.py,sha256=uDBb3DoX3RGbH2fOAvVVMaluP3VivtMfurGiBbbpuAs,1695
|
|
215
|
-
verifiers/v1/configs/judge.py,sha256=
|
|
215
|
+
verifiers/v1/configs/judge.py,sha256=jc9w8eYsqSEyMMY7n12EWyYcwqQl5s-47xV3Me7WA0c,2242
|
|
216
216
|
verifiers/v1/configs/legacy.py,sha256=YOJM6L5u6-jYpZ1CLpPInufKdmz1lEsMNKIRdT2JaaY,1080
|
|
217
217
|
verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
|
|
218
218
|
verifiers/v1/configs/serve.py,sha256=nSoJcTX2S0Vy9EUeNqW0Yo-f9_MUqpBiY-Q9FF_4cRo,2576
|
|
@@ -232,7 +232,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
|
|
|
232
232
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
233
233
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
234
234
|
verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
|
|
235
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
235
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=0KXuTEg4nd0jprPqFgQYFxf2JLdPVbUmljbgPOT-5Bg,13810
|
|
236
236
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
237
237
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
238
238
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -274,14 +274,14 @@ verifiers/v1/harnesses/pi/harness.py,sha256=_O-pyg2RM23-RzQosKHKf8uG8KflS4OrWSsE
|
|
|
274
274
|
verifiers/v1/harnesses/pool/__init__.py,sha256=HTYsNiWGdiNEsSHj4xhPmshJSNb7SHjQ7d4yZQ4p3HM,127
|
|
275
275
|
verifiers/v1/harnesses/pool/harness.py,sha256=6DlOA2W3s-NrxdQ9GaMmZg7mSNJGz2GC21kH2Pmoy9w,4025
|
|
276
276
|
verifiers/v1/harnesses/rlm/__init__.py,sha256=hwx51xTdEWzzYgMj-p4ETNcptoub3IPMDbEdJ4uB8Ns,122
|
|
277
|
-
verifiers/v1/harnesses/rlm/harness.py,sha256=
|
|
277
|
+
verifiers/v1/harnesses/rlm/harness.py,sha256=icZ91FrylW5U8S0sQ178aGJSYyf6eH2JYZvmIKy0L4c,6918
|
|
278
278
|
verifiers/v1/harnesses/terminus_2/__init__.py,sha256=l1pwfg1HyVDEqT9efdnFWcXqaZwtKwe7TrplkyxZSA0,166
|
|
279
279
|
verifiers/v1/harnesses/terminus_2/harness.py,sha256=h-W2Yc4juKcg9YuGp7sZtJghTdu-S6pq4Wf5cP0VBeE,3027
|
|
280
280
|
verifiers/v1/harnesses/terminus_2/program.py,sha256=XWflwULPfVUyrwAldXLRVOoUWoWW9bYNsaygWef0Sfg,2668
|
|
281
281
|
verifiers/v1/interception/__init__.py,sha256=5Kvo7tcBUVaa65s9A1NiOH3CPpzw-P5btXwxB4Nbbvs,4251
|
|
282
282
|
verifiers/v1/interception/base.py,sha256=Mg16Y104lyZilF9shdivdqYoqz4OYp978i_myyrulu8,2529
|
|
283
283
|
verifiers/v1/interception/pool.py,sha256=sk_zcdTV8EVRuzqlYMXxon_42tSDAlFvwcyOO2EF7hw,6473
|
|
284
|
-
verifiers/v1/interception/server.py,sha256=
|
|
284
|
+
verifiers/v1/interception/server.py,sha256=siy2WizKHsTyM2LiigQVcrTR2oACQsW4nclmiyF3m2M,36748
|
|
285
285
|
verifiers/v1/interception/tunnel/__init__.py,sha256=eVNZJszj6myrKhx9jmJz7VRd9qlfj0AUvUZwI2bXEj8,893
|
|
286
286
|
verifiers/v1/interception/tunnel/base.py,sha256=DZB6uPLwM4Qy7n7m0vKeyg-Px87MhsmpYTpe2vpPEng,1998
|
|
287
287
|
verifiers/v1/interception/tunnel/custom.py,sha256=yL4UbGf4bAuMsxk42mEP-t1wITQCT8BE9FHQIz0qRMA,1708
|
|
@@ -289,7 +289,7 @@ verifiers/v1/interception/tunnel/prime.py,sha256=PP3Be3PfbfsLlhF4FfpsZt7IkiYiC1D
|
|
|
289
289
|
verifiers/v1/judges/__init__.py,sha256=MUIBWcx6c70BykrTDHhB0ITIyJPxXe6xDBou4VM7jDQ,286
|
|
290
290
|
verifiers/v1/judges/reference.py,sha256=iEVHw-iJ6dbwLOOqrvWrEMJOcUEYJ0YspT04rZz7Sd4,3868
|
|
291
291
|
verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93V6A8,353
|
|
292
|
-
verifiers/v1/judges/rubric.py,sha256=
|
|
292
|
+
verifiers/v1/judges/rubric.py,sha256=qwBjJCWfaUUcQA9A4EXFQK0NumGjoevSFT2HjgBZut4,12019
|
|
293
293
|
verifiers/v1/judges/rubric.txt,sha256=3KA2ZeYYdcDweJhUGopxu9EFgARhu2aWft5S5cj8Bc8,323
|
|
294
294
|
verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,408
|
|
295
295
|
verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20171
|
|
@@ -337,8 +337,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
|
|
|
337
337
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
338
338
|
verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
339
339
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
340
|
-
verifiers-0.2.2.
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
344
|
-
verifiers-0.2.2.
|
|
340
|
+
verifiers-0.2.2.dev81.dist-info/METADATA,sha256=RsTZSMxzSbJsukFkrvjA45a6IZtUOBxWSpfT6cs1LnU,4545
|
|
341
|
+
verifiers-0.2.2.dev81.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
342
|
+
verifiers-0.2.2.dev81.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
|
|
343
|
+
verifiers-0.2.2.dev81.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
344
|
+
verifiers-0.2.2.dev81.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|