verifiers 0.2.2.dev80__py3-none-any.whl → 0.2.2.dev81__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,7 +10,6 @@ from verifiers.v1.cli.dashboard import dashboard
10
10
  from verifiers.v1.cli.eval import resume
11
11
  from verifiers.v1.cli.output import (
12
12
  append_episode,
13
- append_trace,
14
13
  output_path,
15
14
  save_config,
16
15
  )
@@ -77,8 +76,7 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
77
76
  write_lock = asyncio.Lock()
78
77
 
79
78
  async def on_complete(episode: Episode) -> None:
80
- for trace in episode.traces:
81
- trace.record_run(EvalRunInfo(id=config.uuid))
79
+ episode.record_run(EvalRunInfo(id=config.uuid))
82
80
  await append_episode(out, episode, write_lock)
83
81
 
84
82
  # Serving resources (shared tool servers, interception) come up once for the
@@ -258,9 +256,10 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
258
256
  )
259
257
  records = []
260
258
  for trace in traces:
261
- trace.record_run(EvalRunInfo(id=config.uuid))
262
- await append_trace(out, trace, write_lock, env=config.env_id)
263
- records.append(Episode.of(trace))
259
+ record = Episode.of(trace, env=config.env_id)
260
+ record.record_run(EvalRunInfo(id=config.uuid))
261
+ await append_episode(out, record, write_lock)
262
+ records.append(record)
264
263
  return records
265
264
 
266
265
  async def run_unit(payload: dict) -> list[Episode]:
@@ -271,8 +270,7 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
271
270
  sampling=config.sampling,
272
271
  **payload,
273
272
  )
274
- for trace in episode.traces:
275
- trace.record_run(EvalRunInfo(id=config.uuid))
273
+ episode.record_run(EvalRunInfo(id=config.uuid))
276
274
  await append_episode(out, episode, write_lock)
277
275
  return [episode]
278
276
 
verifiers/v1/episode.py CHANGED
@@ -1,14 +1,14 @@
1
1
  """The episode — one run's traces plus their shared standing, whole."""
2
2
 
3
3
  import uuid
4
- from typing import Generic
4
+ from typing import Any, Generic
5
5
 
6
6
  from pydantic import BaseModel, Field
7
7
 
8
8
  from verifiers.v1.configs.agent import WireAgentConfig
9
9
  from verifiers.v1.state import State, StateT
10
10
  from verifiers.v1.task import DataT, WireTaskData
11
- from verifiers.v1.trace import AgentConfigT, Error, Trace
11
+ from verifiers.v1.trace import AgentConfigT, Error, RunInfo, Trace
12
12
  from verifiers.v1.types import Usage
13
13
 
14
14
 
@@ -26,12 +26,19 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
26
26
 
27
27
  env: EnvInfo = Field(default_factory=EnvInfo)
28
28
  """The env that produced this episode."""
29
+ run: RunInfo | None = None
30
+ """The run this episode belongs to (eval or train), consumer-stamped. It lives here rather than
31
+ on each trace because the episode is what a consumer dispatches, and an episode that produced
32
+ no traces would otherwise have nowhere to say which run it was."""
29
33
  ok: bool = False
30
34
  """Whether the episode completed successfully."""
31
35
  errors: list[Error] = Field(default_factory=list)
32
36
  """Every error captured across attempts, oldest to newest."""
33
37
  traces: list[Trace[DataT, StateT, AgentConfigT]] = Field(default_factory=list)
34
38
  """Every agent's trace, in completion order."""
39
+ info: dict[str, Any] = Field(default_factory=dict)
40
+ """Scratch space for episode-level metadata, the counterpart to `Trace.info`. What describes
41
+ the whole episode belongs here rather than repeated on each of its traces."""
35
42
 
36
43
  @property
37
44
  def last_error(self) -> Error | None:
@@ -72,6 +79,13 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
72
79
  grouped.setdefault(trace.agent.name, []).append(trace)
73
80
  return grouped
74
81
 
82
+ def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
83
+ """Record the run identity and any extra metadata about this episode. Both describe the
84
+ episode as a whole, so they are recorded once here rather than repeated on every trace."""
85
+ if run is not None:
86
+ self.run = run
87
+ self.info.update(info)
88
+
75
89
  @classmethod
76
90
  def of(cls, trace: Trace, env: str = "") -> "Episode":
77
91
  """The single-agent record: one trace as its own episode."""
verifiers/v1/trace.py CHANGED
@@ -304,9 +304,6 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
304
304
  """Unique ID for this trace, auto-generated."""
305
305
  verifiers: VersionInfo = Field(default_factory=_current_build)
306
306
  """The verifiers version that produced this trace."""
307
- run: RunInfo | None = None
308
- """The run this trace belongs to (eval or train), consumer-stamped."""
309
-
310
307
  task: TraceTask[DataT]
311
308
  """The task data that seeded this trace."""
312
309
  agent: AgentInfo[AgentConfigT]
@@ -480,12 +477,6 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
480
477
  if response.usage is not None:
481
478
  self.extra_usage.append(response.usage)
482
479
 
483
- def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
484
- """Record the run identity (eval / train), and optional extra info."""
485
- if run is not None:
486
- self.run = run
487
- self.info.update(info)
488
-
489
480
  def stop(self, condition: str) -> None:
490
481
  """Stop the trace, optionally with a stop condition."""
491
482
  self.is_completed = True
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev80
3
+ Version: 0.2.2.dev81
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -170,7 +170,7 @@ verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2
170
170
  verifiers/v1/__init__.py,sha256=v6qrJQFeNL2b4wd0PsDWiituw2Y-f5rMaFmdbdP6SdM,7505
171
171
  verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
172
172
  verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
173
- verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
173
+ verifiers/v1/episode.py,sha256=8G5OaGOku9S_gavQdpfUZgRZfrf1zmpjFrBL-TsppIk,4123
174
174
  verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
175
175
  verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
176
176
  verifiers/v1/harness.py,sha256=eKuqvYbQjnCMavkiYWZQDHJgH08m6SJ4PuevPuShjS0,11097
@@ -181,7 +181,7 @@ verifiers/v1/session.py,sha256=p8vz89DJUmb9r8dkq80zn2WBEqo8Vwh8ot1ZXm_WVAM,7210
181
181
  verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
182
182
  verifiers/v1/task.py,sha256=jwMKiKlMtksTd8j2dcDlFtb6LC8BRE-N5jkNXlMp2jc,9457
183
183
  verifiers/v1/taskset.py,sha256=fp2E0IEhL_Ybj9cegZwljfTmW27p_30zTWHFKw_kXE4,4374
184
- verifiers/v1/trace.py,sha256=Z4uK-lYAbl-6CHFhTo-gadHnbi63LC0sJtUERC_m5ss,19477
184
+ verifiers/v1/trace.py,sha256=esJE1aT9jRCbBad-BVR9xjX5Emk4mHdYmC7Zct-3zK8,19129
185
185
  verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
186
186
  verifiers/v1/acp/__init__.py,sha256=9SYCFtzGUM_wH98ldpVLcFhoq2M999jbG0tU5ODY70U,2297
187
187
  verifiers/v1/acp/_runner.py,sha256=BcYaNZHuawzCgxOVdhiF6PY_B1HMxIA1MwHy0mOzhSk,7740
@@ -201,7 +201,7 @@ verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8m
201
201
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
202
202
  verifiers/v1/cli/eval/main.py,sha256=xiMMIIbsy77SEOryHXOQyswS9WyrPKTyBwW65W9WTjI,5511
203
203
  verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
204
- verifiers/v1/cli/eval/runner.py,sha256=iR_Bapy5r0IC43GpzlfKIO9RAzFmKcY6EMcwYbEJupY,12181
204
+ verifiers/v1/cli/eval/runner.py,sha256=oAMWf6uwuS_tEyE6OY_Q0-9_TTVCduKC-RJu797QFEc,12117
205
205
  verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
206
206
  verifiers/v1/clients/base.py,sha256=hhtTumKC7lgI6cj7Hks91DCjqNTwAh40DD_j8CWGd6c,1560
207
207
  verifiers/v1/clients/client.py,sha256=L4axOl-dF9_MSYagB_xAO5qbeE53jNZJnB39VLdVBwo,3156
@@ -337,8 +337,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
337
337
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
338
338
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
339
339
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
340
- verifiers-0.2.2.dev80.dist-info/METADATA,sha256=3cfpp8Jk-lCp-IOUdH240_oLbMkUMqv2JO_lPPzUMB8,4545
341
- verifiers-0.2.2.dev80.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
342
- verifiers-0.2.2.dev80.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
343
- verifiers-0.2.2.dev80.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
344
- verifiers-0.2.2.dev80.dist-info/RECORD,,
340
+ verifiers-0.2.2.dev81.dist-info/METADATA,sha256=RsTZSMxzSbJsukFkrvjA45a6IZtUOBxWSpfT6cs1LnU,4545
341
+ verifiers-0.2.2.dev81.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
342
+ verifiers-0.2.2.dev81.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
343
+ verifiers-0.2.2.dev81.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
344
+ verifiers-0.2.2.dev81.dist-info/RECORD,,