verifiers 0.3.2.dev71__py3-none-any.whl → 0.3.2.dev73__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -281,17 +281,25 @@ def Overview(config: EvalConfig) -> Table:
281
281
 
282
282
 
283
283
  def _push_footer(push: "PushState | None") -> Group | None:
284
- """The `--push` status line under the rollouts, shown once the run finishes and the upload
285
- begins: dim `Pushing traces...` while it runs, then white `Traces pushed (<url>)` or red
286
- `Trace push failed (<err>)`. `None` (no line) until the upload starts and when `--push` is off."""
284
+ """The `--push` line under the rollouts: dim with the run's URL while it streams,
285
+ white once pushed, red when it failed. `None` when `--push` is off or the run stayed
286
+ local."""
287
287
  if push is None or not push.started:
288
288
  return None
289
- if not push.done:
290
- line = Text("Pushing traces...", style="dim")
291
- elif push.url:
289
+ if push.error and push.url:
290
+ # The run exists and holds what streamed up; only closing it out failed.
292
291
  line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
293
- else:
292
+ line.append(f" not closed out: {push.error}", style="red")
293
+ elif push.error:
294
294
  line = Text(f"Trace push failed ({push.error})", style="red", overflow="fold")
295
+ elif push.incomplete and push.url:
296
+ # Closed out, but the uploader lost records: what landed is there, say what didn't.
297
+ line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
298
+ line.append(f" incomplete: {push.incomplete}", style="red")
299
+ elif not push.finished:
300
+ line = Text(f"Pushing traces ({push.url})", style="dim", overflow="fold")
301
+ else:
302
+ line = Text(f"Traces pushed ({push.url})", style="white", overflow="fold")
295
303
  return Group(Rule(style="dim"), line)
296
304
 
297
305
 
@@ -134,10 +134,6 @@ def main(argv: list[str] | None = None) -> None:
134
134
  # Graceful cleanup has already run (each rollout's `finally`); partial results are on
135
135
  # disk. Exit on the conventional Ctrl-C code without a traceback.
136
136
  raise SystemExit(130)
137
- if config.push and config.rich is None:
138
- from verifiers.v1.utils.platform import push_traces
139
-
140
- push_traces(episodes, config)
141
137
  if (
142
138
  config.rich is None
143
139
  ): # --rich is the whole output; otherwise dump each trace as JSON
@@ -13,8 +13,8 @@ import asyncio
13
13
  import contextlib
14
14
  import logging
15
15
  import time
16
- from collections.abc import AsyncIterator, Awaitable, Callable
17
- from typing import cast
16
+ from collections.abc import AsyncIterator, Awaitable, Callable, Iterable
17
+ from typing import TypeVar, cast
18
18
 
19
19
  from verifiers.v1.cli.dashboard import dashboard
20
20
  from verifiers.v1.cli.eval import resume
@@ -30,9 +30,36 @@ from verifiers.v1.configs.cli.eval import EvalConfig
30
30
  from verifiers.v1.configs.serve import ServeConfig
31
31
  from verifiers.v1.env import Env, RunSlot
32
32
  from verifiers.v1.episode import Episode, EvalRunInfo
33
+ from verifiers.v1.utils.aio import run_shielded
34
+ from verifiers.v1.utils.platform import (
35
+ PushState,
36
+ abort_run,
37
+ finish_run,
38
+ log_episodes,
39
+ open_run,
40
+ )
33
41
 
34
42
  logger = logging.getLogger(__name__)
35
43
 
44
+ T = TypeVar("T")
45
+
46
+
47
+ async def gather_rollouts(rollouts: Iterable[Awaitable[T]]) -> list[T]:
48
+ """`asyncio.gather`, but one rollout failing cancels the rest and waits for them
49
+ to unwind, so nothing keeps uploading into a run the caller is already closing.
50
+ Not a `TaskGroup`: that wraps errors in an `ExceptionGroup`, and `main` would no
51
+ longer see a `KeyboardInterrupt` as Ctrl-C."""
52
+ tasks = [asyncio.ensure_future(rollout) for rollout in rollouts]
53
+ try:
54
+ return await asyncio.gather(*tasks)
55
+ except BaseException:
56
+ for task in tasks:
57
+ task.cancel()
58
+ # return_exceptions: wait for every task, not just the first to cancel.
59
+ await asyncio.gather(*tasks, return_exceptions=True)
60
+ raise
61
+
62
+
36
63
  RunSlotFn = Callable[[RunSlot], Awaitable[Episode]]
37
64
  OnComplete = Callable[[Episode], Awaitable[None]]
38
65
 
@@ -203,47 +230,55 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
203
230
  asyncio.Semaphore(config.max_concurrent) if config.max_concurrent else None
204
231
  )
205
232
  write_lock = asyncio.Lock()
233
+ push_state = PushState()
234
+
235
+ # Opened before the first rollout so every episode streams as it lands.
236
+ run = open_run(config, push_state, num_examples=len(tasks))
237
+ # Resumed rollouts are part of this run too.
238
+ log_episodes(run, finished)
206
239
 
207
240
  async def on_complete(episode: Episode) -> None:
208
241
  episode.record_run(EvalRunInfo(id=config.run.id, name=config.run.name))
209
242
  await append_episode(out, episode, write_lock)
243
+ await asyncio.to_thread(log_episodes, run, [episode])
210
244
 
211
245
  backend = (
212
246
  _in_process(env, config, semaphore, on_complete)
213
247
  if env is not None
214
248
  else _server(config, config.serve, semaphore, on_complete)
215
249
  )
216
- async with backend as run_slot:
217
- # The display slots: in-process ones are the env's own (it fills their live
218
- # traces); a served rollout's is a client-side stand-in its worker never sees.
219
- planned = [
220
- slot
221
- for task, n in plan
222
- for slot in (
223
- env.slots(task, n)
224
- if env is not None
225
- else [RunSlot(task) for _ in range(n)]
250
+ # The run is closed out whatever breaks, backend setup and teardown included.
251
+ try:
252
+ async with backend as run_slot:
253
+ # The display slots: in-process ones are the env's own (it fills their live
254
+ # traces); a served rollout's is a client-side stand-in its worker never sees.
255
+ planned = [
256
+ slot
257
+ for task, n in plan
258
+ for slot in (
259
+ env.slots(task, n)
260
+ if env is not None
261
+ else [RunSlot(task) for _ in range(n)]
262
+ )
263
+ ]
264
+ slots = [RunSlot.finished(episode) for episode in finished] + planned
265
+ display = (
266
+ dashboard(slots, config, start, push=push_state)
267
+ if config.rich is not None
268
+ else contextlib.nullcontext()
226
269
  )
227
- ]
228
- slots = [RunSlot.finished(episode) for episode in finished] + planned
229
- push_state = None
230
- if config.push and config.rich is not None:
231
- from verifiers.v1.utils.platform import PushState
232
-
233
- push_state = PushState()
234
- display = (
235
- dashboard(slots, config, start, push=push_state)
236
- if config.rich is not None
237
- else contextlib.nullcontext()
238
- )
239
- async with display:
240
- results = await asyncio.gather(*(run_slot(slot) for slot in planned))
241
- episodes = finished + list(results)
242
- if (
243
- push_state is not None
244
- ): # upload off the event loop so the view keeps refreshing
245
- from verifiers.v1.utils.platform import push_traces
246
-
247
- push_state.started = True
248
- await asyncio.to_thread(push_traces, episodes, config, push_state)
270
+ async with display:
271
+ results = await gather_rollouts(run_slot(slot) for slot in planned)
272
+ episodes = finished + list(results)
273
+ # Drain and close out off the event loop so the view keeps refreshing.
274
+ # Shielded: a Ctrl-C here must not cancel the close-out before the
275
+ # worker picks it up (a cancelled executor item never runs), so it
276
+ # runs to completion first and the interrupt is re-raised after —
277
+ # by which point the run is finished and `abort_run` has nothing to do.
278
+ await run_shielded(
279
+ asyncio.to_thread(finish_run, run, episodes, push_state)
280
+ )
281
+ except BaseException as e:
282
+ await asyncio.to_thread(abort_run, run, e, push_state)
283
+ raise
249
284
  return episodes
@@ -48,13 +48,23 @@ class RunConfig(BaseConfig):
48
48
  """Run directory name — the run writes to `output_dir / dir`. Defaults to `run.name`;
49
49
  set it only when the directory should differ from the display name."""
50
50
 
51
- # TODO: fetch the id from the Prime SDK once runs are registered there.
52
- _id: str = PrivateAttr(default_factory=lambda: str(uuid4()))
51
+ _id: str | None = PrivateAttr(default=None)
53
52
 
54
53
  @property
55
54
  def id(self) -> str:
55
+ """The run's one id, assigned by `open_run` from the prime-runs handle: the
56
+ platform's evaluation id online, the SDK's local id otherwise. The run mints
57
+ none of its own, so the run dir, every trace and the dashboard agree."""
58
+ if self._id is None:
59
+ raise RuntimeError("the run has no id until `open_run` has opened it")
56
60
  return self._id
57
61
 
62
+ def assign_id(self, run_id: str) -> None:
63
+ """Called once by `open_run`, before the first rollout."""
64
+ if self._id is not None and self._id != run_id:
65
+ raise RuntimeError(f"the run already has id {self._id!r}")
66
+ self._id = run_id
67
+
58
68
 
59
69
  class EvalConfig(BaseConfig):
60
70
  env: SerializeAsAny[EnvConfig] = SingleAgentEnvConfig()
@@ -295,17 +295,20 @@ class PrimeRuntime(Runtime):
295
295
 
296
296
  async def run(self, argv: list[str], env: dict[str, str]) -> ProgramResult:
297
297
  try:
298
- # The shared SDK client coalesces concurrent VM job polls into batches.
299
- # Rollout cancellation remains the practical execution timeout; this
300
- # long SDK deadline is only a final safety bound.
301
- result = await self._client.run_background_job(
298
+ # Poll directly so rollout cancellation owns the execution timeout.
299
+ job = await self._client.start_background_job(
302
300
  self.info.id,
303
301
  shlex.join(argv),
304
- timeout=EFFECTIVELY_UNBOUNDED_SECONDS,
305
302
  working_dir=self.config.workdir,
306
303
  env=self.process_env(env),
307
- poll_interval=1,
308
304
  )
305
+ delay = 0.1
306
+ while True:
307
+ result = await self._client.get_background_job(self.info.id, job)
308
+ if result.completed:
309
+ break
310
+ await asyncio.sleep(delay)
311
+ delay = min(delay * 2, 3)
309
312
  except (
310
313
  Exception
311
314
  ) as e: # a sandbox/API failure is one rollout's problem, not the eval's
@@ -1,317 +1,164 @@
1
- """Push a finished eval run to the Prime Intellect platform (`--no-push` to skip).
1
+ """The eval's run on the Prime Intellect platform (`--no-push` to keep it local)."""
2
2
 
3
- Uploads one sample per v1 `Episode` over the `/evaluations/` API (create -> push
4
- samples -> finalize). Each sample keeps the complete native Episode as its source
5
- of truth and includes a flat summary for older Platform consumers. Auth + base URL
6
- come from `$PRIME_API_KEY` / `~/.prime/config.json`.
7
- """
8
-
9
- import json
3
+ import asyncio
10
4
  import logging
11
5
  import os
6
+ from collections.abc import Mapping
12
7
  from dataclasses import dataclass
13
8
  from typing import Any
14
9
 
15
- import httpx
10
+ import prime_runs as pr
11
+ from prime_runs.projection import (
12
+ build_samples, # noqa: F401 - prime-rl imports it from here
13
+ )
16
14
 
17
15
  from verifiers.v1.configs.cli.eval import EvalConfig
18
16
  from verifiers.v1.episode import Episode
19
- from verifiers.v1.trace import Trace
20
- from verifiers.v1.utils.prime import load_prime_config
21
17
 
22
18
  logger = logging.getLogger(__name__)
23
19
 
24
- DEFAULT_API_URL = "https://api.primeintellect.ai"
25
- DEFAULT_FRONTEND_URL = "https://app.primeintellect.ai"
26
- # Repeated /samples posts append; match the Prime Evals client's request ceiling.
27
- _MAX_SAMPLES_PAYLOAD_BYTES = 25 * 1024 * 1024
28
-
29
-
30
- def json_bytes(value: Any) -> int:
31
- return len(
32
- json.dumps(
33
- value,
34
- ensure_ascii=False,
35
- separators=(",", ":"),
36
- allow_nan=False,
37
- ).encode("utf-8")
38
- )
39
-
40
20
 
41
21
  @dataclass
42
22
  class PushState:
43
- """Mutable upload status shared with the dashboard."""
23
+ """The dashboard's view of the run: reads through to it, owns no I/O."""
44
24
 
45
- started: bool = False
46
- done: bool = False
47
- url: str | None = None
25
+ run: pr.Run | None = None
48
26
  error: str | None = None
49
-
50
-
51
- def trace_to_sample(
52
- trace: Trace, rollout_number: int = 1, episode_id: str | None = None
53
- ) -> dict[str, Any]:
54
- """One trace -> the platform's sample dict (the v0 eval-sample format).
55
-
56
- The hub table stays flat — one row per trace; its episode is denormalized onto
57
- the row (`episode_id` from the envelope, plus the trace's own `agent`/`trainable`),
58
- so a multi-trace rollout's grouping travels with each row without a nested
59
- schema. No prompt/completion split (meaningless mid-branch): `completion` is the
60
- final branch's messages, `trajectory` one message list per branch."""
61
-
62
- def dump(messages):
63
- return [m.model_dump(mode="json", exclude_none=True) for m in messages]
64
-
65
- task = trace.task.data.model_dump(mode="json", exclude_none=True)
66
- branches = trace.branches
67
- sample = {
68
- "sample_id": trace.id,
69
- "example_id": trace.task.data.idx,
70
- "rollout_number": rollout_number,
71
- "episode_id": episode_id,
72
- "agent": trace.agent.name,
73
- "trainable": trace.agent.trainable,
74
- "task": task,
75
- "prompt": [],
76
- "completion": dump(branches[-1].messages) if branches else [],
77
- "answer": task.get("answer"),
78
- # Keyed `tool_defs` because the v0 sample format already carries it there.
79
- "tool_defs": [t.model_dump(mode="json", exclude_none=True) for t in trace.tools]
80
- if trace.tools
81
- else None,
82
- "reward": trace.reward,
83
- "timing": trace.timing.model_dump(mode="json", exclude_none=True),
84
- "is_completed": trace.is_completed,
85
- "is_truncated": trace.is_truncated,
86
- "metrics": trace.metrics,
87
- "error": trace.last_error.model_dump(mode="json", exclude_none=True)
88
- if trace.last_error
89
- else None,
90
- "stop_condition": trace.stop_condition,
91
- "trajectory": [
92
- {
93
- "messages": dump(branch.messages),
94
- "num_input_tokens": branch.num_input_tokens,
95
- "num_output_tokens": branch.num_output_tokens,
96
- }
97
- for branch in branches
98
- ],
99
- "token_usage": trace.usage.model_dump(mode="json", exclude_none=True)
100
- if trace.usage
101
- else None,
102
- "info": dict(trace.info) or None,
103
- }
104
- # Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
105
- # per-function outputs were); env metrics stay nested.
106
- for name, reward in trace.rewards.items():
107
- if reward is not None:
108
- sample.setdefault(name, reward.score)
109
- return sample
110
-
111
-
112
- def credentials() -> tuple[str | None, str, str, str | None]:
113
- """(api_key, api_base, frontend_url, team_id) from env vars / `~/.prime/config.json`."""
114
- cfg = load_prime_config()
115
- api_key = os.getenv("PRIME_API_KEY") or cfg.get("api_key")
116
- base = (
117
- os.getenv("PRIME_API_BASE_URL")
118
- or os.getenv("PRIME_BASE_URL")
119
- or cfg.get("base_url")
120
- or DEFAULT_API_URL
121
- )
122
- base = base.rstrip("/").removesuffix("/api/v1")
123
- frontend = (
124
- os.getenv("PRIME_FRONTEND_URL")
125
- or cfg.get("frontend_url")
126
- or DEFAULT_FRONTEND_URL
127
- )
128
- team_id = os.getenv("PRIME_TEAM_ID") or cfg.get("team_id")
129
- return api_key, base, frontend, team_id
130
-
131
-
132
- def run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]:
133
- """Run-level aggregates as v0's `GenerateMetadata`. Rewards/metrics aggregate
134
- over the trainable traces only — fixed agents (a judge, a modeled user) often
135
- carry no rewards and would dilute every mean with structural zeros — falling
136
- back to all traces when none are trainable (same rule as the dashboard).
137
- `avg_error` is the share of EPISODES that aren't ok: a hook failure counts
138
- even when its traces are clean or it left none."""
139
- scored = [t for t in traces if t.agent.trainable] or traces
140
- sums: dict[str, float] = {}
141
- counts: dict[str, int] = {}
142
- for trace in scored:
143
- scores = {
144
- name: reward.score
145
- for name, reward in trace.rewards.items()
146
- if reward is not None
147
- }
148
- metrics = {
149
- name: value for name, value in trace.metrics.items() if value is not None
150
- }
151
- for name, value in {**scores, **metrics}.items():
152
- sums[name] = sums.get(name, 0.0) + value
153
- counts[name] = counts.get(name, 0) + 1
154
- n = len(scored)
155
- avg_error = sum(not e.ok for e in episodes) / len(episodes) if episodes else 0.0
156
- return {
157
- "avg_reward": sum(t.reward for t in scored) / n if n else 0.0,
158
- "avg_metrics": {name: sums[name] / counts[name] for name in sums},
159
- "avg_error": avg_error,
27
+ incomplete: str | None = None
28
+ """Set when the run closed out but the uploader lost records on the way: the run
29
+ exists and holds what did land, so the footer says so rather than "failed"."""
30
+
31
+ @property
32
+ def url(self) -> str | None:
33
+ return self.run.url if self.run is not None else None
34
+
35
+ @property
36
+ def finished(self) -> bool:
37
+ return self.run is not None and self.run.finished
38
+
39
+ @property
40
+ def started(self) -> bool:
41
+ """A live run, or a reason there isn't one."""
42
+ return self.error is not None or self.url is not None
43
+
44
+
45
+ def open_run(config: EvalConfig, state: PushState, *, num_examples: int) -> pr.Run:
46
+ """Open the run this eval streams into, before the first rollout, and give the
47
+ config the run's id. A run that cannot be opened is logged and replaced by a
48
+ disabled one; the eval goes on."""
49
+ identity: dict[str, Any] = {
50
+ "name": config.run.name,
51
+ # Resolved by name via the hub's get-or-create; no taskset, nothing to attach to.
52
+ "environments": [config.env.taskset.id] if config.env.taskset.id else [],
53
+ "model": config.model,
54
+ "framework": "verifiers",
55
+ # The v0 keys the dashboard's lists read. The config itself is a follow-up:
56
+ # a dump or the launched file can carry credentials and needs masking first.
57
+ "config": {
58
+ "model": config.model,
59
+ "num_examples": num_examples,
60
+ "rollouts_per_example": config.num_rollouts,
61
+ },
160
62
  }
161
-
162
-
163
- def build_samples(episodes: list[Episode]) -> list[dict[str, Any]]:
164
- """One Platform sample per Episode, with a legacy-compatible trace summary.
165
-
166
- The native Episode in `info.native_wrapper` is authoritative and contains every
167
- trace. One trainable trace (or the first trace) supplies only the flat summary
168
- used by older consumers. `native_trace_index` identifies that summary trace.
169
- """
170
- counts: dict[int, int] = {}
171
- samples = []
172
- for episode in episodes:
173
- if not episode.traces:
174
- continue
175
- summary_trace_index = next(
176
- (
177
- index
178
- for index, candidate in enumerate(episode.traces)
179
- if candidate.agent.trainable
180
- ),
181
- 0,
182
- )
183
- summary_trace = episode.traces[summary_trace_index]
184
- idx = summary_trace.task.data.idx
185
- counts[idx] = number = counts.get(idx, 0) + 1
186
- sample = trace_to_sample(summary_trace, number, episode.id)
187
- sample["sample_id"] = episode.id
188
- sample["info"] = {
189
- **(sample["info"] or {}),
190
- "native_wrapper": episode.to_record(),
191
- "native_trace_index": summary_trace_index,
192
- }
193
- if len(b'{"samples":[]}') + json_bytes(sample) <= _MAX_SAMPLES_PAYLOAD_BYTES:
194
- samples.append(sample)
195
- continue
63
+ if config.push and os.getenv(pr.MODE_ENV, "").strip().lower() == "disabled":
64
+ # The SDK's own kill switch; the explicit `mode="online"` below would override it.
65
+ logger.info("--push: %s=disabled; running without a platform run", pr.MODE_ENV)
66
+ elif config.push:
67
+ try:
68
+ state.run = pr.init(mode="online", **identity)
69
+ except Exception as e: # noqa: BLE001 - a failed upload must not fail the eval
70
+ logger.warning(
71
+ "--push: could not open the run (%s: %s); running without it",
72
+ type(e).__name__,
73
+ e,
74
+ )
75
+ state.error = f"{type(e).__name__}: {e}"
76
+ if state.run is None:
77
+ state.run = pr.init(mode="disabled", **identity)
78
+ # The run's one id: the platform's when online, the SDK's local one otherwise.
79
+ # The SDK keys every upload to it regardless; this is for the local records.
80
+ config.run.assign_id(state.run.id)
81
+ return state.run
82
+
83
+
84
+ def log_episodes(run: pr.Run, episodes: list[Episode]) -> None:
85
+ """Hand finished episodes to the run, best effort: the SDK already keeps upload
86
+ failures on its own thread, so this only guards the hand-off itself. A problem
87
+ here is the platform's, never the eval's."""
88
+ if not episodes:
89
+ return
90
+ try:
91
+ run.log_episodes(episodes)
92
+ except Exception as e: # noqa: BLE001 - the rollouts are on disk; report, don't raise
196
93
  logger.warning(
197
- "Episode %s exceeds the Platform sample limit; uploading projected traces",
198
- episode.id,
94
+ "--push: could not queue %d episode(s) (%s: %s)",
95
+ len(episodes),
96
+ type(e).__name__,
97
+ e,
199
98
  )
200
- samples.extend(
201
- trace_to_sample(candidate, number, episode.id)
202
- for candidate in episode.traces
203
- )
204
- return samples
205
-
206
99
 
207
- def push_traces(
208
- episodes: list[Episode],
209
- config: EvalConfig,
210
- state: "PushState | None" = None,
211
- ) -> str | None:
212
- """Upload a finished run to the platform; return the viewer URL (None if
213
- skipped/failed). Resolves the env by name (get-or-create, so a local run
214
- uploads without a prior `prime env push`); when `state` is given, records the
215
- outcome on it so the dashboard's status line resolves."""
216
100
 
217
- def finish(url: str | None = None, error: str | None = None) -> str | None:
218
- if state is not None:
219
- state.url = url
220
- state.error = error
221
- state.done = True
222
- return url
223
-
224
- api_key, base, frontend, team_id = credentials()
225
- if not api_key:
101
+ def finish_run(run: pr.Run, episodes: list[Episode], state: PushState) -> None:
102
+ """Drain, write the run's aggregates, close it out. Blocking: call it off the loop."""
103
+ try:
104
+ summary = pr.metrics.from_episodes(episodes)
105
+ except Exception as e: # noqa: BLE001 - close the run even without its headline
226
106
  logger.warning(
227
- "--push: no PRIME_API_KEY (set it or run `prime login`); skipping upload"
107
+ "--push: could not aggregate the run's metrics (%s: %s)",
108
+ type(e).__name__,
109
+ e,
228
110
  )
229
- return finish(error="no PRIME_API_KEY (run `prime login`)")
230
-
231
- traces = [trace for episode in episodes for trace in episode.traces]
232
- env_name = config.env.taskset.id
233
- metrics = run_metrics(episodes, traces)
234
- num_examples = len({t.task.data.idx for t in traces})
235
- metadata = {
236
- "framework": "verifiers",
237
- "run_id": config.run.id,
238
- "model": config.model,
239
- "num_examples": num_examples,
240
- "rollouts_per_example": config.num_rollouts,
241
- **metrics,
242
- }
243
-
244
- team = {"team_id": team_id} if team_id else {}
245
- api = f"{base}/api/v1"
246
- headers = {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"}
247
- # The run is done and its results saved; a network blip here must not crash it
248
- # — log and skip the upload instead.
111
+ summary = None
112
+ _close(run, state, summary=summary)
113
+
114
+
115
+ def abort_run(run: pr.Run, error: BaseException, state: PushState) -> None:
116
+ """Close the run out after the eval broke, so it doesn't sit at running. Not
117
+ for a break during `finish_run`: that close-out completes on its own thread
118
+ and the SDK lets the first `finish()` decide the status — it sets `run.status`
119
+ as it starts, so an in-flight one is visible here before it is `finished`."""
120
+ if run.finished or run.status is not pr.RunStatus.RUNNING:
121
+ return
122
+ if isinstance(error, (KeyboardInterrupt, asyncio.CancelledError)):
123
+ status, message = pr.RunStatus.CANCELLED, "interrupted"
124
+ else:
125
+ status, message = pr.RunStatus.FAILED, f"{type(error).__name__}: {error}"
126
+ _close(run, state, status=status, error=message)
127
+
128
+
129
+ def _close(
130
+ run: pr.Run,
131
+ state: PushState,
132
+ summary: Mapping[str, Any] | None = None,
133
+ status: pr.RunStatus = pr.RunStatus.COMPLETED,
134
+ error: str | None = None,
135
+ ) -> None:
136
+ """`run.finish()`, best effort: the results are on disk, so nothing here may raise."""
249
137
  try:
250
- samples = build_samples(episodes)
251
- batches: list[list[dict[str, Any]]] = []
252
- batch: list[dict[str, Any]] = []
253
- payload_bytes = len(b'{"samples":[]}')
254
- for i, sample in enumerate(samples):
255
- sample_bytes = json_bytes(sample)
256
- sample_payload_bytes = len(b'{"samples":[]}') + sample_bytes
257
- if sample_payload_bytes > _MAX_SAMPLES_PAYLOAD_BYTES:
258
- raise ValueError(
259
- f"sample {i} is too large to upload "
260
- f"({sample_payload_bytes} > "
261
- f"{_MAX_SAMPLES_PAYLOAD_BYTES} bytes)"
262
- )
263
- next_payload_bytes = payload_bytes + (1 if batch else 0) + sample_bytes
264
- if batch and next_payload_bytes > _MAX_SAMPLES_PAYLOAD_BYTES:
265
- batches.append(batch)
266
- batch = []
267
- payload_bytes = len(b'{"samples":[]}')
268
- next_payload_bytes = payload_bytes + sample_bytes
269
- batch.append(sample)
270
- payload_bytes = next_payload_bytes
271
- if batch or not samples:
272
- batches.append(batch)
273
-
274
- with httpx.Client(headers=headers, timeout=300.0) as client:
275
-
276
- def post(path: str, body: dict) -> dict:
277
- resp = client.post(f"{api}{path}", json=body)
278
- resp.raise_for_status()
279
- return resp.json()
280
-
281
- env_id = post("/environmentshub/resolve", {"name": env_name, **team})[
282
- "data"
283
- ]["id"]
284
- eval_id = post(
285
- "/evaluations/",
286
- {
287
- "name": config.run.name,
288
- "environments": [{"id": env_id}],
289
- "model_name": config.model,
290
- "dataset": env_name,
291
- "framework": "verifiers",
292
- "metadata": metadata,
293
- "metrics": metrics,
294
- "tags": [],
295
- **team,
296
- },
297
- )["evaluation_id"]
298
- for batch in batches:
299
- body = json.dumps(
300
- {"samples": batch},
301
- ensure_ascii=False,
302
- separators=(",", ":"),
303
- allow_nan=False,
304
- ).encode("utf-8")
305
- resp = client.post(
306
- f"{api}/evaluations/{eval_id}/samples",
307
- content=body,
308
- )
309
- resp.raise_for_status()
310
- post(f"/evaluations/{eval_id}/finalize", {"metrics": metrics})
311
- except Exception as e: # noqa: BLE001 - push is best-effort across the full upload
312
- logger.warning("--push: upload failed (%s: %s); skipping", type(e).__name__, e)
313
- return finish(error=f"{type(e).__name__}: {e}")
314
-
315
- url = f"{frontend}/dashboard/evaluations/{eval_id}"
316
- logger.info("--push: uploaded %d samples -> %s", len(samples), url)
317
- return finish(url=url)
138
+ run.finish(summary, status=status, error=error)
139
+ except Exception as e: # noqa: BLE001 - the run is over; report, don't raise
140
+ logger.warning(
141
+ "--push: could not close out the run (%s: %s)", type(e).__name__, e
142
+ )
143
+ if state.error is None:
144
+ state.error = f"{type(e).__name__}: {e}"
145
+ else:
146
+ state.incomplete = _losses(run)
147
+ if state.incomplete:
148
+ logger.warning("--push: %s, but %s", status.value, state.incomplete)
149
+ if run.url:
150
+ # The run's own status: `finish()` is a no-op once another caller closed it.
151
+ logger.info("--push: %s -> %s", run.status.value, run.url)
152
+
153
+
154
+ def _losses(run: pr.Run) -> str | None:
155
+ """What the uploader could not store, or `None`. A sink that switched itself off
156
+ quietly (Prime Traces outside the beta) is not a loss and is not counted here."""
157
+ parts = [
158
+ f"{count} record(s) not stored by the {sink} sink"
159
+ for sink, count in sorted(run.failed_records.items())
160
+ if count
161
+ ]
162
+ if run.dropped_records:
163
+ parts.append(f"{run.dropped_records} record(s) never queued (uploader overrun)")
164
+ return "; ".join(parts) or None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: verifiers
3
- Version: 0.3.2.dev71
3
+ Version: 0.3.2.dev73
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -35,6 +35,7 @@ Requires-Dist: msgpack>=1.1.2
35
35
  Requires-Dist: numpy>=2.1.0
36
36
  Requires-Dist: openai<3.0.0,>=2.54.0
37
37
  Requires-Dist: prime-pydantic-config[toml]>=0.4.3
38
+ Requires-Dist: prime-runs>=0.1.1
38
39
  Requires-Dist: prime-sandboxes>=0.2.39
39
40
  Requires-Dist: prime-tunnel>=0.1.8
40
41
  Requires-Dist: pydantic>=2.12.3
@@ -29,13 +29,13 @@ verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,47
29
29
  verifiers/v1/cli/validate.py,sha256=G3YXZTwxVDQIyR_ASapJceixOxRSJl69NnFPBqkF3N0,17173
30
30
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
31
31
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
32
- verifiers/v1/cli/dashboard/eval.py,sha256=hr2r54wvcMYw-CEjDsVxrBQgZGp9cGrgrsRtjaaSNPs,37822
32
+ verifiers/v1/cli/dashboard/eval.py,sha256=wfEUkpJxE0liMv_wqBLQU5555z4FIcTIFZ38EO8Puh4,38303
33
33
  verifiers/v1/cli/dashboard/replay.py,sha256=_X5up9MJsRbd_sWe4LgNTPNT8k1ip4xNpo50hxI3Ljk,2697
34
34
  verifiers/v1/cli/dashboard/validate.py,sha256=xrpK3Y90JsoDCQYLLvvsnJ4CPRMvSSJIghfHK62E6FQ,3687
35
35
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
36
- verifiers/v1/cli/eval/main.py,sha256=m-7ZqBx4J2asueMR4oA43qzifOvaUnrGMKrTHi8fEZQ,6250
36
+ verifiers/v1/cli/eval/main.py,sha256=FqNjQuQkz-VXnpJ4sKZh2LQp_rKwtT3_xXxYG_YJ_3E,6107
37
37
  verifiers/v1/cli/eval/resume.py,sha256=fYkWa0Fudq--IWfk25j_McLm67bXYWvksWmJR4u4QAU,3811
38
- verifiers/v1/cli/eval/runner.py,sha256=tZDPWzd6U7-5N43HGRlV7p-RnUiDleYwMUBVHzWH7mk,10015
38
+ verifiers/v1/cli/eval/runner.py,sha256=BNxbA81GmvyzyGOq-UBGoOUyCHjlSoIP4sZlrgi1wb8,11607
39
39
  verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
40
40
  verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
41
41
  verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
@@ -55,7 +55,7 @@ verifiers/v1/configs/taskset.py,sha256=N5sFYjzjcmnUEiNiHH69vaOfl8fsVwmqXmWmPixqO
55
55
  verifiers/v1/configs/cli/__init__.py,sha256=YMDUPdbRlDJ77dhj3QqzmyRhqNgUvuQdpj_bdBqzZPM,78
56
56
  verifiers/v1/configs/cli/debug.py,sha256=6ohG1AB0FBH_s8dVOTre86NlIf_fClz-3FiCBWgR-IM,2854
57
57
  verifiers/v1/configs/cli/env.py,sha256=YdIo0RW7VBhq_uCoa7RfnkkzGU5PW4kTa6Bn20lnd5c,2592
58
- verifiers/v1/configs/cli/eval.py,sha256=AkT-wY_pm4WJ8zNABvfC3GoYOjrNCkyVXtUGquBDmUI,6341
58
+ verifiers/v1/configs/cli/eval.py,sha256=DwDk7KMosgtXNx6qYixOvgkCb83JbVUjL2utq2n3Spo,6877
59
59
  verifiers/v1/configs/cli/init.py,sha256=JvIUOdhfZfNOXg2wm2i06jLLHpb97tNMzZqmcK_9wVw,916
60
60
  verifiers/v1/configs/cli/replay.py,sha256=vUCNs-9j1CXcAZhGozN1EaztcXK_dYVK3iN_fRmbSXw,2929
61
61
  verifiers/v1/configs/cli/validate.py,sha256=NXQ9KvhGLODLrg1fs3Jn-SOq1hhPxauVWiPSEp6wje0,3691
@@ -144,7 +144,7 @@ verifiers/v1/runtimes/__init__.py,sha256=ppc1R_sx-aMcuJSx4t99rTSjloUXI0M6EUj676-
144
144
  verifiers/v1/runtimes/base.py,sha256=VMxYMnGE_zj4k1VseTxEtKhzOZroahRnZSfvruDenEw,17468
145
145
  verifiers/v1/runtimes/limiters.py,sha256=C6cWDD4pwyFwUY4QFke3oiMXwydSNIfSbJSIglUC2kE,3062
146
146
  verifiers/v1/runtimes/modal.py,sha256=wUNik2Wul1YISausdABIMAoSaTA0MNDwKUayKxVLwAA,13342
147
- verifiers/v1/runtimes/prime.py,sha256=7LIu-ciN_VT8GU6xD3W29k4ZJN1TpcL_5libDUN0VU0,19304
147
+ verifiers/v1/runtimes/prime.py,sha256=JUj_jl1UVXHKDif41UcLXnnwX5Sdt9lQTqw14FtIl9c,19346
148
148
  verifiers/v1/runtimes/subprocess.py,sha256=QmpQ235_xrX6gowrr8557VgbydOXpLbfFq0KKGfJThk,8774
149
149
  verifiers/v1/runtimes/docker/__init__.py,sha256=JN0mI10a_BKQpRRJKX_NqStAoODAEk2dAJF9mrz69ug,20972
150
150
  verifiers/v1/runtimes/docker/egress.py,sha256=iFH_1XotlEM-OCaW7GNu3TWVEq1sqesKYweyVBX8g90,14360
@@ -184,13 +184,13 @@ verifiers/v1/utils/loaders.py,sha256=2TrgZK3MzKub1GnfaTVIQhlwF6q4qATXZCjicH6nT9c
184
184
  verifiers/v1/utils/logging.py,sha256=-zgaM-HZsgxumZ4wYNlRMZepp6JanCSjeL2wSVulwJM,2296
185
185
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
186
186
  verifiers/v1/utils/paths.py,sha256=it_4JWf_8bPZe4TRcyvL7_KJ4fsbEsxhhIY_xA8Ikus,308
187
- verifiers/v1/utils/platform.py,sha256=3P4ryQTHIA8r3hdi-cTGOzKJSsaJEZqsQ8HPK9Rtjxg,12269
187
+ verifiers/v1/utils/platform.py,sha256=V3n0_d7urcYlz6-N5FjEB_ghwq2juw20MccJvQOh8G8,6544
188
188
  verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,970
189
189
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
190
190
  verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
191
191
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
192
- verifiers-0.3.2.dev71.dist-info/METADATA,sha256=FeO8Ciqzot_uCkoMeTBll1GESqgEg7_1YHMybm994oQ,4204
193
- verifiers-0.3.2.dev71.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
194
- verifiers-0.3.2.dev71.dist-info/entry_points.txt,sha256=uqQje0TMsr7k6nXlWxlPs13Si0evb3pVCEDF6c-YL80,241
195
- verifiers-0.3.2.dev71.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
196
- verifiers-0.3.2.dev71.dist-info/RECORD,,
192
+ verifiers-0.3.2.dev73.dist-info/METADATA,sha256=XOpxs0ftFNLCrFqlitTNnKxkm1RJDID8j88_OpRODEY,4237
193
+ verifiers-0.3.2.dev73.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
194
+ verifiers-0.3.2.dev73.dist-info/entry_points.txt,sha256=uqQje0TMsr7k6nXlWxlPs13Si0evb3pVCEDF6c-YL80,241
195
+ verifiers-0.3.2.dev73.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
196
+ verifiers-0.3.2.dev73.dist-info/RECORD,,