verifiers 0.3.2.dev71__py3-none-any.whl → 0.3.2.dev73__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/dashboard/eval.py +15 -7
- verifiers/v1/cli/eval/main.py +0 -4
- verifiers/v1/cli/eval/runner.py +69 -34
- verifiers/v1/configs/cli/eval.py +12 -2
- verifiers/v1/runtimes/prime.py +9 -6
- verifiers/v1/utils/platform.py +139 -292
- {verifiers-0.3.2.dev71.dist-info → verifiers-0.3.2.dev73.dist-info}/METADATA +2 -1
- {verifiers-0.3.2.dev71.dist-info → verifiers-0.3.2.dev73.dist-info}/RECORD +11 -11
- {verifiers-0.3.2.dev71.dist-info → verifiers-0.3.2.dev73.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev71.dist-info → verifiers-0.3.2.dev73.dist-info}/entry_points.txt +0 -0
- {verifiers-0.3.2.dev71.dist-info → verifiers-0.3.2.dev73.dist-info}/licenses/LICENSE +0 -0
|
@@ -281,17 +281,25 @@ def Overview(config: EvalConfig) -> Table:
|
|
|
281
281
|
|
|
282
282
|
|
|
283
283
|
def _push_footer(push: "PushState | None") -> Group | None:
|
|
284
|
-
"""The `--push`
|
|
285
|
-
|
|
286
|
-
|
|
284
|
+
"""The `--push` line under the rollouts: dim with the run's URL while it streams,
|
|
285
|
+
white once pushed, red when it failed. `None` when `--push` is off or the run stayed
|
|
286
|
+
local."""
|
|
287
287
|
if push is None or not push.started:
|
|
288
288
|
return None
|
|
289
|
-
if
|
|
290
|
-
|
|
291
|
-
elif push.url:
|
|
289
|
+
if push.error and push.url:
|
|
290
|
+
# The run exists and holds what streamed up; only closing it out failed.
|
|
292
291
|
line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
|
|
293
|
-
|
|
292
|
+
line.append(f" not closed out: {push.error}", style="red")
|
|
293
|
+
elif push.error:
|
|
294
294
|
line = Text(f"Trace push failed ({push.error})", style="red", overflow="fold")
|
|
295
|
+
elif push.incomplete and push.url:
|
|
296
|
+
# Closed out, but the uploader lost records: what landed is there, say what didn't.
|
|
297
|
+
line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
|
|
298
|
+
line.append(f" incomplete: {push.incomplete}", style="red")
|
|
299
|
+
elif not push.finished:
|
|
300
|
+
line = Text(f"Pushing traces ({push.url})", style="dim", overflow="fold")
|
|
301
|
+
else:
|
|
302
|
+
line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
|
|
295
303
|
return Group(Rule(style="dim"), line)
|
|
296
304
|
|
|
297
305
|
|
verifiers/v1/cli/eval/main.py
CHANGED
|
@@ -134,10 +134,6 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
134
134
|
# Graceful cleanup has already run (each rollout's `finally`); partial results are on
|
|
135
135
|
# disk. Exit on the conventional Ctrl-C code without a traceback.
|
|
136
136
|
raise SystemExit(130)
|
|
137
|
-
if config.push and config.rich is None:
|
|
138
|
-
from verifiers.v1.utils.platform import push_traces
|
|
139
|
-
|
|
140
|
-
push_traces(episodes, config)
|
|
141
137
|
if (
|
|
142
138
|
config.rich is None
|
|
143
139
|
): # --rich is the whole output; otherwise dump each trace as JSON
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -13,8 +13,8 @@ import asyncio
|
|
|
13
13
|
import contextlib
|
|
14
14
|
import logging
|
|
15
15
|
import time
|
|
16
|
-
from collections.abc import AsyncIterator, Awaitable, Callable
|
|
17
|
-
from typing import cast
|
|
16
|
+
from collections.abc import AsyncIterator, Awaitable, Callable, Iterable
|
|
17
|
+
from typing import TypeVar, cast
|
|
18
18
|
|
|
19
19
|
from verifiers.v1.cli.dashboard import dashboard
|
|
20
20
|
from verifiers.v1.cli.eval import resume
|
|
@@ -30,9 +30,36 @@ from verifiers.v1.configs.cli.eval import EvalConfig
|
|
|
30
30
|
from verifiers.v1.configs.serve import ServeConfig
|
|
31
31
|
from verifiers.v1.env import Env, RunSlot
|
|
32
32
|
from verifiers.v1.episode import Episode, EvalRunInfo
|
|
33
|
+
from verifiers.v1.utils.aio import run_shielded
|
|
34
|
+
from verifiers.v1.utils.platform import (
|
|
35
|
+
PushState,
|
|
36
|
+
abort_run,
|
|
37
|
+
finish_run,
|
|
38
|
+
log_episodes,
|
|
39
|
+
open_run,
|
|
40
|
+
)
|
|
33
41
|
|
|
34
42
|
logger = logging.getLogger(__name__)
|
|
35
43
|
|
|
44
|
+
T = TypeVar("T")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
async def gather_rollouts(rollouts: Iterable[Awaitable[T]]) -> list[T]:
|
|
48
|
+
"""`asyncio.gather`, but one rollout failing cancels the rest and waits for them
|
|
49
|
+
to unwind, so nothing keeps uploading into a run the caller is already closing.
|
|
50
|
+
Not a `TaskGroup`: that wraps errors in an `ExceptionGroup`, and `main` would no
|
|
51
|
+
longer see a `KeyboardInterrupt` as Ctrl-C."""
|
|
52
|
+
tasks = [asyncio.ensure_future(rollout) for rollout in rollouts]
|
|
53
|
+
try:
|
|
54
|
+
return await asyncio.gather(*tasks)
|
|
55
|
+
except BaseException:
|
|
56
|
+
for task in tasks:
|
|
57
|
+
task.cancel()
|
|
58
|
+
# return_exceptions: wait for every task, not just the first to cancel.
|
|
59
|
+
await asyncio.gather(*tasks, return_exceptions=True)
|
|
60
|
+
raise
|
|
61
|
+
|
|
62
|
+
|
|
36
63
|
RunSlotFn = Callable[[RunSlot], Awaitable[Episode]]
|
|
37
64
|
OnComplete = Callable[[Episode], Awaitable[None]]
|
|
38
65
|
|
|
@@ -203,47 +230,55 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
203
230
|
asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
|
|
204
231
|
)
|
|
205
232
|
write_lock = asyncio.Lock()
|
|
233
|
+
push_state = PushState()
|
|
234
|
+
|
|
235
|
+
# Opened before the first rollout so every episode streams as it lands.
|
|
236
|
+
run = open_run(config, push_state, num_examples=len(tasks))
|
|
237
|
+
# Resumed rollouts are part of this run too.
|
|
238
|
+
log_episodes(run, finished)
|
|
206
239
|
|
|
207
240
|
async def on_complete(episode: Episode) -> None:
|
|
208
241
|
episode.record_run(EvalRunInfo(id=config.run.id, name=config.run.name))
|
|
209
242
|
await append_episode(out, episode, write_lock)
|
|
243
|
+
await asyncio.to_thread(log_episodes, run, [episode])
|
|
210
244
|
|
|
211
245
|
backend = (
|
|
212
246
|
_in_process(env, config, semaphore, on_complete)
|
|
213
247
|
if env is not None
|
|
214
248
|
else _server(config, config.serve, semaphore, on_complete)
|
|
215
249
|
)
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
250
|
+
# The run is closed out whatever breaks, backend setup and teardown included.
|
|
251
|
+
try:
|
|
252
|
+
async with backend as run_slot:
|
|
253
|
+
# The display slots: in-process ones are the env's own (it fills their live
|
|
254
|
+
# traces); a served rollout's is a client-side stand-in its worker never sees.
|
|
255
|
+
planned = [
|
|
256
|
+
slot
|
|
257
|
+
for task, n in plan
|
|
258
|
+
for slot in (
|
|
259
|
+
env.slots(task, n)
|
|
260
|
+
if env is not None
|
|
261
|
+
else [RunSlot(task) for _ in range(n)]
|
|
262
|
+
)
|
|
263
|
+
]
|
|
264
|
+
slots = [RunSlot.finished(episode) for episode in finished] + planned
|
|
265
|
+
display = (
|
|
266
|
+
dashboard(slots, config, start, push=push_state)
|
|
267
|
+
if config.rich is not None
|
|
268
|
+
else contextlib.nullcontext()
|
|
226
269
|
)
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
episodes = finished + list(results)
|
|
242
|
-
if (
|
|
243
|
-
push_state is not None
|
|
244
|
-
): # upload off the event loop so the view keeps refreshing
|
|
245
|
-
from verifiers.v1.utils.platform import push_traces
|
|
246
|
-
|
|
247
|
-
push_state.started = True
|
|
248
|
-
await asyncio.to_thread(push_traces, episodes, config, push_state)
|
|
270
|
+
async with display:
|
|
271
|
+
results = await gather_rollouts(run_slot(slot) for slot in planned)
|
|
272
|
+
episodes = finished + list(results)
|
|
273
|
+
# Drain and close out off the event loop so the view keeps refreshing.
|
|
274
|
+
# Shielded: a Ctrl-C here must not cancel the close-out before the
|
|
275
|
+
# worker picks it up (a cancelled executor item never runs), so it
|
|
276
|
+
# runs to completion first and the interrupt is re-raised after —
|
|
277
|
+
# by which point the run is finished and `abort_run` has nothing to do.
|
|
278
|
+
await run_shielded(
|
|
279
|
+
asyncio.to_thread(finish_run, run, episodes, push_state)
|
|
280
|
+
)
|
|
281
|
+
except BaseException as e:
|
|
282
|
+
await asyncio.to_thread(abort_run, run, e, push_state)
|
|
283
|
+
raise
|
|
249
284
|
return episodes
|
verifiers/v1/configs/cli/eval.py
CHANGED
|
@@ -48,13 +48,23 @@ class RunConfig(BaseConfig):
|
|
|
48
48
|
"""Run directory name — the run writes to `output_dir / dir`. Defaults to `run.name`;
|
|
49
49
|
set it only when the directory should differ from the display name."""
|
|
50
50
|
|
|
51
|
-
|
|
52
|
-
_id: str = PrivateAttr(default_factory=lambda: str(uuid4()))
|
|
51
|
+
_id: str | None = PrivateAttr(default=None)
|
|
53
52
|
|
|
54
53
|
@property
|
|
55
54
|
def id(self) -> str:
|
|
55
|
+
"""The run's one id, assigned by `open_run` from the prime-runs handle: the
|
|
56
|
+
platform's evaluation id online, the SDK's local id otherwise. The run mints
|
|
57
|
+
none of its own, so the run dir, every trace and the dashboard agree."""
|
|
58
|
+
if self._id is None:
|
|
59
|
+
raise RuntimeError("the run has no id until `open_run` has opened it")
|
|
56
60
|
return self._id
|
|
57
61
|
|
|
62
|
+
def assign_id(self, run_id: str) -> None:
|
|
63
|
+
"""Called once by `open_run`, before the first rollout."""
|
|
64
|
+
if self._id is not None and self._id != run_id:
|
|
65
|
+
raise RuntimeError(f"the run already has id {self._id!r}")
|
|
66
|
+
self._id = run_id
|
|
67
|
+
|
|
58
68
|
|
|
59
69
|
class EvalConfig(BaseConfig):
|
|
60
70
|
env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
|
verifiers/v1/runtimes/prime.py
CHANGED
|
@@ -295,17 +295,20 @@ class PrimeRuntime(Runtime):
|
|
|
295
295
|
|
|
296
296
|
async def run(self, argv: list[str], env: dict[str, str]) -> ProgramResult:
|
|
297
297
|
try:
|
|
298
|
-
#
|
|
299
|
-
|
|
300
|
-
# long SDK deadline is only a final safety bound.
|
|
301
|
-
result = await self._client.run_background_job(
|
|
298
|
+
# Poll directly so rollout cancellation owns the execution timeout.
|
|
299
|
+
job = await self._client.start_background_job(
|
|
302
300
|
self.info.id,
|
|
303
301
|
shlex.join(argv),
|
|
304
|
-
timeout=EFFECTIVELY_UNBOUNDED_SECONDS,
|
|
305
302
|
working_dir=self.config.workdir,
|
|
306
303
|
env=self.process_env(env),
|
|
307
|
-
poll_interval=1,
|
|
308
304
|
)
|
|
305
|
+
delay = 0.1
|
|
306
|
+
while True:
|
|
307
|
+
result = await self._client.get_background_job(self.info.id, job)
|
|
308
|
+
if result.completed:
|
|
309
|
+
break
|
|
310
|
+
await asyncio.sleep(delay)
|
|
311
|
+
delay = min(delay * 2, 3)
|
|
309
312
|
except (
|
|
310
313
|
Exception
|
|
311
314
|
) as e: # a sandbox/API failure is one rollout's problem, not the eval's
|
verifiers/v1/utils/platform.py
CHANGED
|
@@ -1,317 +1,164 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""The eval's run on the Prime Intellect platform (`--no-push` to keep it local)."""
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
samples -> finalize). Each sample keeps the complete native Episode as its source
|
|
5
|
-
of truth and includes a flat summary for older Platform consumers. Auth + base URL
|
|
6
|
-
come from `$PRIME_API_KEY` / `~/.prime/config.json`.
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
import json
|
|
3
|
+
import asyncio
|
|
10
4
|
import logging
|
|
11
5
|
import os
|
|
6
|
+
from collections.abc import Mapping
|
|
12
7
|
from dataclasses import dataclass
|
|
13
8
|
from typing import Any
|
|
14
9
|
|
|
15
|
-
import
|
|
10
|
+
import prime_runs as pr
|
|
11
|
+
from prime_runs.projection import (
|
|
12
|
+
build_samples, # noqa: F401 - prime-rl imports it from here
|
|
13
|
+
)
|
|
16
14
|
|
|
17
15
|
from verifiers.v1.configs.cli.eval import EvalConfig
|
|
18
16
|
from verifiers.v1.episode import Episode
|
|
19
|
-
from verifiers.v1.trace import Trace
|
|
20
|
-
from verifiers.v1.utils.prime import load_prime_config
|
|
21
17
|
|
|
22
18
|
logger = logging.getLogger(__name__)
|
|
23
19
|
|
|
24
|
-
DEFAULT_API_URL = "https://api.primeintellect.ai"
|
|
25
|
-
DEFAULT_FRONTEND_URL = "https://app.primeintellect.ai"
|
|
26
|
-
# Repeated /samples posts append; match the Prime Evals client's request ceiling.
|
|
27
|
-
_MAX_SAMPLES_PAYLOAD_BYTES = 25 * 1024 * 1024
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def json_bytes(value: Any) -> int:
|
|
31
|
-
return len(
|
|
32
|
-
json.dumps(
|
|
33
|
-
value,
|
|
34
|
-
ensure_ascii=False,
|
|
35
|
-
separators=(",", ":"),
|
|
36
|
-
allow_nan=False,
|
|
37
|
-
).encode("utf-8")
|
|
38
|
-
)
|
|
39
|
-
|
|
40
20
|
|
|
41
21
|
@dataclass
|
|
42
22
|
class PushState:
|
|
43
|
-
"""
|
|
23
|
+
"""The dashboard's view of the run: reads through to it, owns no I/O."""
|
|
44
24
|
|
|
45
|
-
|
|
46
|
-
done: bool = False
|
|
47
|
-
url: str | None = None
|
|
25
|
+
run: pr.Run | None = None
|
|
48
26
|
error: str | None = None
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
def
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
"
|
|
73
|
-
|
|
74
|
-
"
|
|
75
|
-
"
|
|
76
|
-
"
|
|
77
|
-
|
|
78
|
-
#
|
|
79
|
-
"
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
"is_completed": trace.is_completed,
|
|
85
|
-
"is_truncated": trace.is_truncated,
|
|
86
|
-
"metrics": trace.metrics,
|
|
87
|
-
"error": trace.last_error.model_dump(mode="json", exclude_none=True)
|
|
88
|
-
if trace.last_error
|
|
89
|
-
else None,
|
|
90
|
-
"stop_condition": trace.stop_condition,
|
|
91
|
-
"trajectory": [
|
|
92
|
-
{
|
|
93
|
-
"messages": dump(branch.messages),
|
|
94
|
-
"num_input_tokens": branch.num_input_tokens,
|
|
95
|
-
"num_output_tokens": branch.num_output_tokens,
|
|
96
|
-
}
|
|
97
|
-
for branch in branches
|
|
98
|
-
],
|
|
99
|
-
"token_usage": trace.usage.model_dump(mode="json", exclude_none=True)
|
|
100
|
-
if trace.usage
|
|
101
|
-
else None,
|
|
102
|
-
"info": dict(trace.info) or None,
|
|
103
|
-
}
|
|
104
|
-
# Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
|
|
105
|
-
# per-function outputs were); env metrics stay nested.
|
|
106
|
-
for name, reward in trace.rewards.items():
|
|
107
|
-
if reward is not None:
|
|
108
|
-
sample.setdefault(name, reward.score)
|
|
109
|
-
return sample
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
def credentials() -> tuple[str | None, str, str, str | None]:
|
|
113
|
-
"""(api_key, api_base, frontend_url, team_id) from env vars / `~/.prime/config.json`."""
|
|
114
|
-
cfg = load_prime_config()
|
|
115
|
-
api_key = os.getenv("PRIME_API_KEY") or cfg.get("api_key")
|
|
116
|
-
base = (
|
|
117
|
-
os.getenv("PRIME_API_BASE_URL")
|
|
118
|
-
or os.getenv("PRIME_BASE_URL")
|
|
119
|
-
or cfg.get("base_url")
|
|
120
|
-
or DEFAULT_API_URL
|
|
121
|
-
)
|
|
122
|
-
base = base.rstrip("/").removesuffix("/api/v1")
|
|
123
|
-
frontend = (
|
|
124
|
-
os.getenv("PRIME_FRONTEND_URL")
|
|
125
|
-
or cfg.get("frontend_url")
|
|
126
|
-
or DEFAULT_FRONTEND_URL
|
|
127
|
-
)
|
|
128
|
-
team_id = os.getenv("PRIME_TEAM_ID") or cfg.get("team_id")
|
|
129
|
-
return api_key, base, frontend, team_id
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
def run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]:
|
|
133
|
-
"""Run-level aggregates as v0's `GenerateMetadata`. Rewards/metrics aggregate
|
|
134
|
-
over the trainable traces only — fixed agents (a judge, a modeled user) often
|
|
135
|
-
carry no rewards and would dilute every mean with structural zeros — falling
|
|
136
|
-
back to all traces when none are trainable (same rule as the dashboard).
|
|
137
|
-
`avg_error` is the share of EPISODES that aren't ok: a hook failure counts
|
|
138
|
-
even when its traces are clean or it left none."""
|
|
139
|
-
scored = [t for t in traces if t.agent.trainable] or traces
|
|
140
|
-
sums: dict[str, float] = {}
|
|
141
|
-
counts: dict[str, int] = {}
|
|
142
|
-
for trace in scored:
|
|
143
|
-
scores = {
|
|
144
|
-
name: reward.score
|
|
145
|
-
for name, reward in trace.rewards.items()
|
|
146
|
-
if reward is not None
|
|
147
|
-
}
|
|
148
|
-
metrics = {
|
|
149
|
-
name: value for name, value in trace.metrics.items() if value is not None
|
|
150
|
-
}
|
|
151
|
-
for name, value in {**scores, **metrics}.items():
|
|
152
|
-
sums[name] = sums.get(name, 0.0) + value
|
|
153
|
-
counts[name] = counts.get(name, 0) + 1
|
|
154
|
-
n = len(scored)
|
|
155
|
-
avg_error = sum(not e.ok for e in episodes) / len(episodes) if episodes else 0.0
|
|
156
|
-
return {
|
|
157
|
-
"avg_reward": sum(t.reward for t in scored) / n if n else 0.0,
|
|
158
|
-
"avg_metrics": {name: sums[name] / counts[name] for name in sums},
|
|
159
|
-
"avg_error": avg_error,
|
|
27
|
+
incomplete: str | None = None
|
|
28
|
+
"""Set when the run closed out but the uploader lost records on the way: the run
|
|
29
|
+
exists and holds what did land, so the footer says so rather than "failed"."""
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def url(self) -> str | None:
|
|
33
|
+
return self.run.url if self.run is not None else None
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def finished(self) -> bool:
|
|
37
|
+
return self.run is not None and self.run.finished
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def started(self) -> bool:
|
|
41
|
+
"""A live run, or a reason there isn't one."""
|
|
42
|
+
return self.error is not None or self.url is not None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def open_run(config: EvalConfig, state: PushState, *, num_examples: int) -> pr.Run:
|
|
46
|
+
"""Open the run this eval streams into, before the first rollout, and give the
|
|
47
|
+
config the run's id. A run that cannot be opened is logged and replaced by a
|
|
48
|
+
disabled one; the eval goes on."""
|
|
49
|
+
identity: dict[str, Any] = {
|
|
50
|
+
"name": config.run.name,
|
|
51
|
+
# Resolved by name via the hub's get-or-create; no taskset, nothing to attach to.
|
|
52
|
+
"environments": [config.env.taskset.id] if config.env.taskset.id else [],
|
|
53
|
+
"model": config.model,
|
|
54
|
+
"framework": "verifiers",
|
|
55
|
+
# The v0 keys the dashboard's lists read. The config itself is a follow-up:
|
|
56
|
+
# a dump or the launched file can carry credentials and needs masking first.
|
|
57
|
+
"config": {
|
|
58
|
+
"model": config.model,
|
|
59
|
+
"num_examples": num_examples,
|
|
60
|
+
"rollouts_per_example": config.num_rollouts,
|
|
61
|
+
},
|
|
160
62
|
}
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
"native_trace_index": summary_trace_index,
|
|
192
|
-
}
|
|
193
|
-
if len(b'{"samples":[]}') + json_bytes(sample) <= _MAX_SAMPLES_PAYLOAD_BYTES:
|
|
194
|
-
samples.append(sample)
|
|
195
|
-
continue
|
|
63
|
+
if config.push and os.getenv(pr.MODE_ENV, "").strip().lower() == "disabled":
|
|
64
|
+
# The SDK's own kill switch; the explicit `mode="online"` below would override it.
|
|
65
|
+
logger.info("--push: %s=disabled; running without a platform run", pr.MODE_ENV)
|
|
66
|
+
elif config.push:
|
|
67
|
+
try:
|
|
68
|
+
state.run = pr.init(mode="online", **identity)
|
|
69
|
+
except Exception as e: # noqa: BLE001 - a failed upload must not fail the eval
|
|
70
|
+
logger.warning(
|
|
71
|
+
"--push: could not open the run (%s: %s); running without it",
|
|
72
|
+
type(e).__name__,
|
|
73
|
+
e,
|
|
74
|
+
)
|
|
75
|
+
state.error = f"{type(e).__name__}: {e}"
|
|
76
|
+
if state.run is None:
|
|
77
|
+
state.run = pr.init(mode="disabled", **identity)
|
|
78
|
+
# The run's one id: the platform's when online, the SDK's local one otherwise.
|
|
79
|
+
# The SDK keys every upload to it regardless; this is for the local records.
|
|
80
|
+
config.run.assign_id(state.run.id)
|
|
81
|
+
return state.run
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def log_episodes(run: pr.Run, episodes: list[Episode]) -> None:
|
|
85
|
+
"""Hand finished episodes to the run, best effort: the SDK already keeps upload
|
|
86
|
+
failures on its own thread, so this only guards the hand-off itself. A problem
|
|
87
|
+
here is the platform's, never the eval's."""
|
|
88
|
+
if not episodes:
|
|
89
|
+
return
|
|
90
|
+
try:
|
|
91
|
+
run.log_episodes(episodes)
|
|
92
|
+
except Exception as e: # noqa: BLE001 - the rollouts are on disk; report, don't raise
|
|
196
93
|
logger.warning(
|
|
197
|
-
"
|
|
198
|
-
|
|
94
|
+
"--push: could not queue %d episode(s) (%s: %s)",
|
|
95
|
+
len(episodes),
|
|
96
|
+
type(e).__name__,
|
|
97
|
+
e,
|
|
199
98
|
)
|
|
200
|
-
samples.extend(
|
|
201
|
-
trace_to_sample(candidate, number, episode.id)
|
|
202
|
-
for candidate in episode.traces
|
|
203
|
-
)
|
|
204
|
-
return samples
|
|
205
|
-
|
|
206
99
|
|
|
207
|
-
def push_traces(
|
|
208
|
-
episodes: list[Episode],
|
|
209
|
-
config: EvalConfig,
|
|
210
|
-
state: "PushState | None" = None,
|
|
211
|
-
) -> str | None:
|
|
212
|
-
"""Upload a finished run to the platform; return the viewer URL (None if
|
|
213
|
-
skipped/failed). Resolves the env by name (get-or-create, so a local run
|
|
214
|
-
uploads without a prior `prime env push`); when `state` is given, records the
|
|
215
|
-
outcome on it so the dashboard's status line resolves."""
|
|
216
100
|
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
return url
|
|
223
|
-
|
|
224
|
-
api_key, base, frontend, team_id = credentials()
|
|
225
|
-
if not api_key:
|
|
101
|
+
def finish_run(run: pr.Run, episodes: list[Episode], state: PushState) -> None:
|
|
102
|
+
"""Drain, write the run's aggregates, close it out. Blocking: call it off the loop."""
|
|
103
|
+
try:
|
|
104
|
+
summary = pr.metrics.from_episodes(episodes)
|
|
105
|
+
except Exception as e: # noqa: BLE001 - close the run even without its headline
|
|
226
106
|
logger.warning(
|
|
227
|
-
"--push:
|
|
107
|
+
"--push: could not aggregate the run's metrics (%s: %s)",
|
|
108
|
+
type(e).__name__,
|
|
109
|
+
e,
|
|
228
110
|
)
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
111
|
+
summary = None
|
|
112
|
+
_close(run, state, summary=summary)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def abort_run(run: pr.Run, error: BaseException, state: PushState) -> None:
|
|
116
|
+
"""Close the run out after the eval broke, so it doesn't sit at running. Not
|
|
117
|
+
for a break during `finish_run`: that close-out completes on its own thread
|
|
118
|
+
and the SDK lets the first `finish()` decide the status — it sets `run.status`
|
|
119
|
+
as it starts, so an in-flight one is visible here before it is `finished`."""
|
|
120
|
+
if run.finished or run.status is not pr.RunStatus.RUNNING:
|
|
121
|
+
return
|
|
122
|
+
if isinstance(error, (KeyboardInterrupt, asyncio.CancelledError)):
|
|
123
|
+
status, message = pr.RunStatus.CANCELLED, "interrupted"
|
|
124
|
+
else:
|
|
125
|
+
status, message = pr.RunStatus.FAILED, f"{type(error).__name__}: {error}"
|
|
126
|
+
_close(run, state, status=status, error=message)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _close(
|
|
130
|
+
run: pr.Run,
|
|
131
|
+
state: PushState,
|
|
132
|
+
summary: Mapping[str, Any] | None = None,
|
|
133
|
+
status: pr.RunStatus = pr.RunStatus.COMPLETED,
|
|
134
|
+
error: str | None = None,
|
|
135
|
+
) -> None:
|
|
136
|
+
"""`run.finish()`, best effort: the results are on disk, so nothing here may raise."""
|
|
249
137
|
try:
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
resp = client.post(f"{api}{path}", json=body)
|
|
278
|
-
resp.raise_for_status()
|
|
279
|
-
return resp.json()
|
|
280
|
-
|
|
281
|
-
env_id = post("/environmentshub/resolve", {"name": env_name, **team})[
|
|
282
|
-
"data"
|
|
283
|
-
]["id"]
|
|
284
|
-
eval_id = post(
|
|
285
|
-
"/evaluations/",
|
|
286
|
-
{
|
|
287
|
-
"name": config.run.name,
|
|
288
|
-
"environments": [{"id": env_id}],
|
|
289
|
-
"model_name": config.model,
|
|
290
|
-
"dataset": env_name,
|
|
291
|
-
"framework": "verifiers",
|
|
292
|
-
"metadata": metadata,
|
|
293
|
-
"metrics": metrics,
|
|
294
|
-
"tags": [],
|
|
295
|
-
**team,
|
|
296
|
-
},
|
|
297
|
-
)["evaluation_id"]
|
|
298
|
-
for batch in batches:
|
|
299
|
-
body = json.dumps(
|
|
300
|
-
{"samples": batch},
|
|
301
|
-
ensure_ascii=False,
|
|
302
|
-
separators=(",", ":"),
|
|
303
|
-
allow_nan=False,
|
|
304
|
-
).encode("utf-8")
|
|
305
|
-
resp = client.post(
|
|
306
|
-
f"{api}/evaluations/{eval_id}/samples",
|
|
307
|
-
content=body,
|
|
308
|
-
)
|
|
309
|
-
resp.raise_for_status()
|
|
310
|
-
post(f"/evaluations/{eval_id}/finalize", {"metrics": metrics})
|
|
311
|
-
except Exception as e: # noqa: BLE001 - push is best-effort across the full upload
|
|
312
|
-
logger.warning("--push: upload failed (%s: %s); skipping", type(e).__name__, e)
|
|
313
|
-
return finish(error=f"{type(e).__name__}: {e}")
|
|
314
|
-
|
|
315
|
-
url = f"{frontend}/dashboard/evaluations/{eval_id}"
|
|
316
|
-
logger.info("--push: uploaded %d samples -> %s", len(samples), url)
|
|
317
|
-
return finish(url=url)
|
|
138
|
+
run.finish(summary, status=status, error=error)
|
|
139
|
+
except Exception as e: # noqa: BLE001 - the run is over; report, don't raise
|
|
140
|
+
logger.warning(
|
|
141
|
+
"--push: could not close out the run (%s: %s)", type(e).__name__, e
|
|
142
|
+
)
|
|
143
|
+
if state.error is None:
|
|
144
|
+
state.error = f"{type(e).__name__}: {e}"
|
|
145
|
+
else:
|
|
146
|
+
state.incomplete = _losses(run)
|
|
147
|
+
if state.incomplete:
|
|
148
|
+
logger.warning("--push: %s, but %s", status.value, state.incomplete)
|
|
149
|
+
if run.url:
|
|
150
|
+
# The run's own status: `finish()` is a no-op once another caller closed it.
|
|
151
|
+
logger.info("--push: %s -> %s", run.status.value, run.url)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _losses(run: pr.Run) -> str | None:
|
|
155
|
+
"""What the uploader could not store, or `None`. A sink that switched itself off
|
|
156
|
+
quietly (Prime Traces outside the beta) is not a loss and is not counted here."""
|
|
157
|
+
parts = [
|
|
158
|
+
f"{count} record(s) not stored by the {sink} sink"
|
|
159
|
+
for sink, count in sorted(run.failed_records.items())
|
|
160
|
+
if count
|
|
161
|
+
]
|
|
162
|
+
if run.dropped_records:
|
|
163
|
+
parts.append(f"{run.dropped_records} record(s) never queued (uploader overrun)")
|
|
164
|
+
return "; ".join(parts) or None
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.3.2.
|
|
3
|
+
Version: 0.3.2.dev73
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -35,6 +35,7 @@ Requires-Dist: msgpack>=1.1.2
|
|
|
35
35
|
Requires-Dist: numpy>=2.1.0
|
|
36
36
|
Requires-Dist: openai<3.0.0,>=2.54.0
|
|
37
37
|
Requires-Dist: prime-pydantic-config[toml]>=0.4.3
|
|
38
|
+
Requires-Dist: prime-runs>=0.1.1
|
|
38
39
|
Requires-Dist: prime-sandboxes>=0.2.39
|
|
39
40
|
Requires-Dist: prime-tunnel>=0.1.8
|
|
40
41
|
Requires-Dist: pydantic>=2.12.3
|
|
@@ -29,13 +29,13 @@ verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,47
|
|
|
29
29
|
verifiers/v1/cli/validate.py,sha256=G3YXZTwxVDQIyR_ASapJceixOxRSJl69NnFPBqkF3N0,17173
|
|
30
30
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
31
31
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
32
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
32
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=wfEUkpJxE0liMv_wqBLQU5555z4FIcTIFZ38EO8Puh4,38303
|
|
33
33
|
verifiers/v1/cli/dashboard/replay.py,sha256=_X5up9MJsRbd_sWe4LgNTPNT8k1ip4xNpo50hxI3Ljk,2697
|
|
34
34
|
verifiers/v1/cli/dashboard/validate.py,sha256=xrpK3Y90JsoDCQYLLvvsnJ4CPRMvSSJIghfHK62E6FQ,3687
|
|
35
35
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
36
|
-
verifiers/v1/cli/eval/main.py,sha256=
|
|
36
|
+
verifiers/v1/cli/eval/main.py,sha256=FqNjQuQkz-VXnpJ4sKZh2LQp_rKwtT3_xXxYG_YJ_3E,6107
|
|
37
37
|
verifiers/v1/cli/eval/resume.py,sha256=fYkWa0Fudq--IWfk25j_McLm67bXYWvksWmJR4u4QAU,3811
|
|
38
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
38
|
+
verifiers/v1/cli/eval/runner.py,sha256=BNxbA81GmvyzyGOq-UBGoOUyCHjlSoIP4sZlrgi1wb8,11607
|
|
39
39
|
verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
|
|
40
40
|
verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
|
|
41
41
|
verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
|
|
@@ -55,7 +55,7 @@ verifiers/v1/configs/taskset.py,sha256=N5sFYjzjcmnUEiNiHH69vaOfl8fsVwmqXmWmPixqO
|
|
|
55
55
|
verifiers/v1/configs/cli/__init__.py,sha256=YMDUPdbRlDJ77dhj3QqzmyRhqNgUvuQdpj_bdBqzZPM,78
|
|
56
56
|
verifiers/v1/configs/cli/debug.py,sha256=6ohG1AB0FBH_s8dVOTre86NlIf_fClz-3FiCBWgR-IM,2854
|
|
57
57
|
verifiers/v1/configs/cli/env.py,sha256=YdIo0RW7VBhq_uCoa7RfnkkzGU5PW4kTa6Bn20lnd5c,2592
|
|
58
|
-
verifiers/v1/configs/cli/eval.py,sha256=
|
|
58
|
+
verifiers/v1/configs/cli/eval.py,sha256=DwDk7KMosgtXNx6qYixOvgkCb83JbVUjL2utq2n3Spo,6877
|
|
59
59
|
verifiers/v1/configs/cli/init.py,sha256=JvIUOdhfZfNOXg2wm2i06jLLHpb97tNMzZqmcK_9wVw,916
|
|
60
60
|
verifiers/v1/configs/cli/replay.py,sha256=vUCNs-9j1CXcAZhGozN1EaztcXK_dYVK3iN_fRmbSXw,2929
|
|
61
61
|
verifiers/v1/configs/cli/validate.py,sha256=NXQ9KvhGLODLrg1fs3Jn-SOq1hhPxauVWiPSEp6wje0,3691
|
|
@@ -144,7 +144,7 @@ verifiers/v1/runtimes/__init__.py,sha256=ppc1R_sx-aMcuJSx4t99rTSjloUXI0M6EUj676-
|
|
|
144
144
|
verifiers/v1/runtimes/base.py,sha256=VMxYMnGE_zj4k1VseTxEtKhzOZroahRnZSfvruDenEw,17468
|
|
145
145
|
verifiers/v1/runtimes/limiters.py,sha256=C6cWDD4pwyFwUY4QFke3oiMXwydSNIfSbJSIglUC2kE,3062
|
|
146
146
|
verifiers/v1/runtimes/modal.py,sha256=wUNik2Wul1YISausdABIMAoSaTA0MNDwKUayKxVLwAA,13342
|
|
147
|
-
verifiers/v1/runtimes/prime.py,sha256=
|
|
147
|
+
verifiers/v1/runtimes/prime.py,sha256=JUj_jl1UVXHKDif41UcLXnnwX5Sdt9lQTqw14FtIl9c,19346
|
|
148
148
|
verifiers/v1/runtimes/subprocess.py,sha256=QmpQ235_xrX6gowrr8557VgbydOXpLbfFq0KKGfJThk,8774
|
|
149
149
|
verifiers/v1/runtimes/docker/__init__.py,sha256=JN0mI10a_BKQpRRJKX_NqStAoODAEk2dAJF9mrz69ug,20972
|
|
150
150
|
verifiers/v1/runtimes/docker/egress.py,sha256=iFH_1XotlEM-OCaW7GNu3TWVEq1sqesKYweyVBX8g90,14360
|
|
@@ -184,13 +184,13 @@ verifiers/v1/utils/loaders.py,sha256=2TrgZK3MzKub1GnfaTVIQhlwF6q4qATXZCjicH6nT9c
|
|
|
184
184
|
verifiers/v1/utils/logging.py,sha256=-zgaM-HZsgxumZ4wYNlRMZepp6JanCSjeL2wSVulwJM,2296
|
|
185
185
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
186
186
|
verifiers/v1/utils/paths.py,sha256=it_4JWf_8bPZe4TRcyvL7_KJ4fsbEsxhhIY_xA8Ikus,308
|
|
187
|
-
verifiers/v1/utils/platform.py,sha256=
|
|
187
|
+
verifiers/v1/utils/platform.py,sha256=V3n0_d7urcYlz6-N5FjEB_ghwq2juw20MccJvQOh8G8,6544
|
|
188
188
|
verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,970
|
|
189
189
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
190
190
|
verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
|
|
191
191
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
192
|
-
verifiers-0.3.2.
|
|
193
|
-
verifiers-0.3.2.
|
|
194
|
-
verifiers-0.3.2.
|
|
195
|
-
verifiers-0.3.2.
|
|
196
|
-
verifiers-0.3.2.
|
|
192
|
+
verifiers-0.3.2.dev73.dist-info/METADATA,sha256=XOpxs0ftFNLCrFqlitTNnKxkm1RJDID8j88_OpRODEY,4237
|
|
193
|
+
verifiers-0.3.2.dev73.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
194
|
+
verifiers-0.3.2.dev73.dist-info/entry_points.txt,sha256=uqQje0TMsr7k6nXlWxlPs13Si0evb3pVCEDF6c-YL80,241
|
|
195
|
+
verifiers-0.3.2.dev73.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
196
|
+
verifiers-0.3.2.dev73.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|