verifiers 0.2.2.dev90__py3-none-any.whl → 0.2.2.dev92__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,6 +10,7 @@ from verifiers.v1.cli.dashboard import dashboard
10
10
  from verifiers.v1.cli.eval import resume
11
11
  from verifiers.v1.cli.output import (
12
12
  append_episode,
13
+ append_trace,
13
14
  output_path,
14
15
  save_config,
15
16
  )
@@ -76,7 +77,8 @@ async def run_eval(env: Env, config: EvalConfig) -> list[Episode]:
76
77
  write_lock = asyncio.Lock()
77
78
 
78
79
  async def on_complete(episode: Episode) -> None:
79
- episode.record_run(EvalRunInfo(id=config.uuid))
80
+ for trace in episode.traces:
81
+ trace.record_run(EvalRunInfo(id=config.uuid))
80
82
  await append_episode(out, episode, write_lock)
81
83
 
82
84
  # Serving resources (shared tool servers, interception) come up once for the
@@ -256,10 +258,9 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
256
258
  )
257
259
  records = []
258
260
  for trace in traces:
259
- record = Episode.of(trace, env=config.env_id)
260
- record.record_run(EvalRunInfo(id=config.uuid))
261
- await append_episode(out, record, write_lock)
262
- records.append(record)
261
+ trace.record_run(EvalRunInfo(id=config.uuid))
262
+ await append_trace(out, trace, write_lock, env=config.env_id)
263
+ records.append(Episode.of(trace))
263
264
  return records
264
265
 
265
266
  async def run_unit(payload: dict) -> list[Episode]:
@@ -270,7 +271,8 @@ async def run_eval_server(config: EvalConfig) -> list[Episode]:
270
271
  sampling=config.sampling,
271
272
  **payload,
272
273
  )
273
- episode.record_run(EvalRunInfo(id=config.uuid))
274
+ for trace in episode.traces:
275
+ trace.record_run(EvalRunInfo(id=config.uuid))
274
276
  await append_episode(out, episode, write_lock)
275
277
  return [episode]
276
278
 
verifiers/v1/episode.py CHANGED
@@ -1,14 +1,14 @@
1
1
  """The episode — one run's traces plus their shared standing, whole."""
2
2
 
3
3
  import uuid
4
- from typing import Any, Generic
4
+ from typing import Generic
5
5
 
6
6
  from pydantic import BaseModel, Field
7
7
 
8
8
  from verifiers.v1.configs.agent import WireAgentConfig
9
9
  from verifiers.v1.state import State, StateT
10
10
  from verifiers.v1.task import DataT, WireTaskData
11
- from verifiers.v1.trace import AgentConfigT, Error, RunInfo, Trace
11
+ from verifiers.v1.trace import AgentConfigT, Error, Trace
12
12
  from verifiers.v1.types import Usage
13
13
 
14
14
 
@@ -26,19 +26,12 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
26
26
 
27
27
  env: EnvInfo = Field(default_factory=EnvInfo)
28
28
  """The env that produced this episode."""
29
- run: RunInfo | None = None
30
- """The run this episode belongs to (eval or train), consumer-stamped. It lives here rather than
31
- on each trace because the episode is what a consumer dispatches, and an episode that produced
32
- no traces would otherwise have nowhere to say which run it was."""
33
29
  ok: bool = False
34
30
  """Whether the episode completed successfully."""
35
31
  errors: list[Error] = Field(default_factory=list)
36
32
  """Every error captured across attempts, oldest to newest."""
37
33
  traces: list[Trace[DataT, StateT, AgentConfigT]] = Field(default_factory=list)
38
34
  """Every agent's trace, in completion order."""
39
- info: dict[str, Any] = Field(default_factory=dict)
40
- """Scratch space for episode-level metadata, the counterpart to `Trace.info`. What describes
41
- the whole episode belongs here rather than repeated on each of its traces."""
42
35
 
43
36
  @property
44
37
  def last_error(self) -> Error | None:
@@ -79,13 +72,6 @@ class Episode(BaseModel, Generic[DataT, StateT, AgentConfigT]):
79
72
  grouped.setdefault(trace.agent.name, []).append(trace)
80
73
  return grouped
81
74
 
82
- def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
83
- """Record the run identity and any extra metadata about this episode. Both describe the
84
- episode as a whole, so they are recorded once here rather than repeated on every trace."""
85
- if run is not None:
86
- self.run = run
87
- self.info.update(info)
88
-
89
75
  @classmethod
90
76
  def of(cls, trace: Trace, env: str = "") -> "Episode":
91
77
  """The single-agent record: one trace as its own episode."""
verifiers/v1/trace.py CHANGED
@@ -321,6 +321,9 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
321
321
  """Unique ID for this trace, auto-generated."""
322
322
  verifiers: VersionInfo = Field(default_factory=_current_build)
323
323
  """The verifiers version that produced this trace."""
324
+ run: RunInfo | None = None
325
+ """The run this trace belongs to (eval or train), consumer-stamped."""
326
+
324
327
  task: TraceTask[DataT]
325
328
  """The task data that seeded this trace."""
326
329
  agent: AgentInfo[AgentConfigT]
@@ -494,6 +497,12 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
494
497
  if response.usage is not None:
495
498
  self.extra_usage.append(response.usage)
496
499
 
500
+ def record_run(self, run: RunInfo | None = None, **info: Any) -> None:
501
+ """Record the run identity (eval / train), and optional extra info."""
502
+ if run is not None:
503
+ self.run = run
504
+ self.info.update(info)
505
+
497
506
  def stop(self, condition: str) -> None:
498
507
  """Stop the trace, optionally with a stop condition."""
499
508
  self.is_completed = True
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev90
3
+ Version: 0.2.2.dev92
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -170,7 +170,7 @@ verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2
170
170
  verifiers/v1/__init__.py,sha256=y-2SJbNv_3Hbg-a6eTpmygk8w2L1qtcJtkaJOsiCf1o,7585
171
171
  verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
172
172
  verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
173
- verifiers/v1/episode.py,sha256=8G5OaGOku9S_gavQdpfUZgRZfrf1zmpjFrBL-TsppIk,4123
173
+ verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
174
174
  verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
175
175
  verifiers/v1/graph.py,sha256=psE4gww42TuSy_B4rJfGauycGTzPes6Z2SKFAzPYg7M,29427
176
176
  verifiers/v1/harness.py,sha256=45mocvRaVAgk_wfpf0QCcHVAKemB1WyIO96BhDJif70,13923
@@ -181,7 +181,7 @@ verifiers/v1/session.py,sha256=_XSgVaBjdbN1sBUB5kLRh0cb0rbQ6S86CVlA-j6MRh0,7218
181
181
  verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
182
182
  verifiers/v1/task.py,sha256=jwMKiKlMtksTd8j2dcDlFtb6LC8BRE-N5jkNXlMp2jc,9457
183
183
  verifiers/v1/taskset.py,sha256=fp2E0IEhL_Ybj9cegZwljfTmW27p_30zTWHFKw_kXE4,4374
184
- verifiers/v1/trace.py,sha256=Mjc9NaqmUVKXhn6RKaTaGVxPRldCGqgHxb74cqtp5_k,20026
184
+ verifiers/v1/trace.py,sha256=zhNlt_DLsQAydDnq6o_aQSKujqndNwG0qfgBwp-ez_g,20374
185
185
  verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
186
186
  verifiers/v1/acp/__init__.py,sha256=RusAr7bFDHOjQc3mgqnADfhgen9bJlyLs8wQke2fUUs,11675
187
187
  verifiers/v1/acp/runner.py,sha256=4VWM-KWS7jD2tuU_CxO4wwnYNrii6hTDawX_4q4qoLc,14730
@@ -201,7 +201,7 @@ verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8m
201
201
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
202
202
  verifiers/v1/cli/eval/main.py,sha256=xiMMIIbsy77SEOryHXOQyswS9WyrPKTyBwW65W9WTjI,5511
203
203
  verifiers/v1/cli/eval/resume.py,sha256=QwXuLPs2lN-aMZk1CyaT12egRPzeI6vfY39DAd0nYLU,7171
204
- verifiers/v1/cli/eval/runner.py,sha256=oAMWf6uwuS_tEyE6OY_Q0-9_TTVCduKC-RJu797QFEc,12117
204
+ verifiers/v1/cli/eval/runner.py,sha256=iR_Bapy5r0IC43GpzlfKIO9RAzFmKcY6EMcwYbEJupY,12181
205
205
  verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
206
206
  verifiers/v1/clients/base.py,sha256=hhtTumKC7lgI6cj7Hks91DCjqNTwAh40DD_j8CWGd6c,1560
207
207
  verifiers/v1/clients/client.py,sha256=L4axOl-dF9_MSYagB_xAO5qbeE53jNZJnB39VLdVBwo,3156
@@ -339,8 +339,8 @@ verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XV
339
339
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
340
340
  verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
341
341
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
342
- verifiers-0.2.2.dev90.dist-info/METADATA,sha256=_973kReUuLinz4-6GkmuL45btXhe4JiqdVl3qrzBP30,4545
343
- verifiers-0.2.2.dev90.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
344
- verifiers-0.2.2.dev90.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
345
- verifiers-0.2.2.dev90.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
346
- verifiers-0.2.2.dev90.dist-info/RECORD,,
342
+ verifiers-0.2.2.dev92.dist-info/METADATA,sha256=jIbEzEPviNzI_K-XM6UF8XU9hQlDUSzDIsoOQmGzikM,4545
343
+ verifiers-0.2.2.dev92.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
344
+ verifiers-0.2.2.dev92.dist-info/entry_points.txt,sha256=v6v0QT9vVExnfn4br42MostI_o3f3ZbWYlTormz-U3g,515
345
+ verifiers-0.2.2.dev92.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
346
+ verifiers-0.2.2.dev92.dist-info/RECORD,,