verifiers 0.3.2.dev136__py3-none-any.whl → 0.3.2.dev138__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/agent.py CHANGED
@@ -12,13 +12,15 @@ import logging
12
12
  from collections.abc import AsyncIterator, Callable, Mapping
13
13
  from contextlib import asynccontextmanager, nullcontext
14
14
  from dataclasses import dataclass, replace
15
- from typing import Self
15
+ from typing import Generic, Self, cast
16
+
17
+ from typing_extensions import TypeVar
16
18
 
17
19
  from verifiers.v1.clients import (
18
20
  EvalClientConfig,
19
21
  ModelContext,
20
22
  )
21
- from verifiers.v1.configs.agent import AgentConfig, TimeoutConfig
23
+ from verifiers.v1.configs.agent import AgentConfig, TimeoutConfig, agent_config_fields
22
24
  from verifiers.v1.configs.runtime import NetworkPolicyConfig
23
25
  from verifiers.v1.dialects import parse_message
24
26
  from verifiers.v1.harness import Harness
@@ -42,6 +44,7 @@ from verifiers.v1.types import (
42
44
  ToolMessage,
43
45
  UserMessage,
44
46
  )
47
+ from verifiers.v1.utils.aio import run_shielded
45
48
  from verifiers.v1.utils.compile import (
46
49
  cap_remote_agent_timeout,
47
50
  resolve_runtime_config,
@@ -462,9 +465,8 @@ class Agent:
462
465
  run.trace.stop("agent_completed")
463
466
  trace = await run.close()
464
467
  except BaseException:
465
- # A cancellation mid-run (or a lifetime bug raised to the caller) means
466
- # close() never runs — free the run's servers and owned runtime first.
467
- await run.abort()
468
+ # Finish cleanup even if another cancellation arrives during abort().
469
+ await run_shielded(run.abort())
468
470
  raise
469
471
  if trace.agent.runtime is not None:
470
472
  trace.agent.runtime.borrowed = runtime is not None
@@ -720,27 +722,22 @@ def make_agent(
720
722
  return Agent(config, interception=interception)
721
723
 
722
724
 
723
- MakeAgent = Callable[[str, AgentConfig], Agent]
725
+ AgentT = TypeVar("AgentT", bound=Agent, default=Agent)
726
+ MakeAgent = Callable[[str, AgentConfig], AgentT]
724
727
  """An agent factory keyed by name — what `Agents` calls per scraped config field."""
725
728
 
726
729
 
727
- def agent_config_fields(config) -> dict[str, AgentConfig]:
728
- """The top-level `AgentConfig` fields declared on a config, in declaration
729
- order — the env's agents, keyed by field name (the only naming site)."""
730
- return {name: value for name, value in config if isinstance(value, AgentConfig)}
731
-
732
-
733
- class Agents:
730
+ class Agents(Generic[AgentT]):
734
731
  """A config's agents, addressed by attribute: every top-level `AgentConfig`
735
732
  field becomes an `Agent` under the field's name (`agents.solver`)."""
736
733
 
737
- def __init__(self, config, make: MakeAgent | None = None) -> None:
738
- self._agents: dict[str, Agent] = {
739
- name: make_agent(value) if make is None else make(name, value)
734
+ def __init__(self, config, make: MakeAgent[AgentT] | None = None) -> None:
735
+ self._agents: dict[str, AgentT] = {
736
+ name: cast(AgentT, make_agent(value)) if make is None else make(name, value)
740
737
  for name, value in agent_config_fields(config).items()
741
738
  }
742
739
 
743
- def __getattr__(self, name: str) -> Agent:
740
+ def __getattr__(self, name: str) -> AgentT:
744
741
  # self.__dict__ directly: attribute lookup re-entering __getattr__ before
745
742
  # __init__ ran (copy/unpickle) must raise, not recurse.
746
743
  agents = self.__dict__.get("_agents")
@@ -81,9 +81,9 @@ def format_cost_usd(cost_usd: float) -> str:
81
81
  def _seat_value(config: EvalConfig, read):
82
82
  """A cap as the overview shows it: the declared seats' shared value, the
83
83
  string 'per-seat' when they disagree (caps live on the seats)."""
84
- from verifiers.v1.configs.env import _declared_agent_configs
84
+ from verifiers.v1.configs.agent import agent_config_fields
85
85
 
86
- values = {read(spec) for spec in _declared_agent_configs(config.env).values()}
86
+ values = {read(spec) for spec in agent_config_fields(config.env).values()}
87
87
  return values.pop() if len(values) == 1 else "per-seat"
88
88
 
89
89
 
verifiers/v1/cli/debug.py CHANGED
@@ -17,7 +17,7 @@ from uuid import uuid4
17
17
  from pydantic_config import cli
18
18
 
19
19
  import verifiers.v1 as vf
20
- from verifiers.v1.cli.output import append_trace, save_config
20
+ from verifiers.v1.cli.output import save_config
21
21
  from verifiers.v1.cli.resolve import (
22
22
  extract_id,
23
23
  narrow_taskset_config,
@@ -34,6 +34,7 @@ from verifiers.v1.utils.compile import resolve_runtime_config
34
34
  from verifiers.v1.utils.decorators import invoke
35
35
  from verifiers.v1.utils.interrupt import install_interrupt
36
36
  from verifiers.v1.utils.logging import setup_logging
37
+ from verifiers.v1.utils.trace_store import append_trace
37
38
 
38
39
  logger = logging.getLogger(__name__)
39
40
 
@@ -10,7 +10,6 @@ from pydantic_config import cli
10
10
 
11
11
  from verifiers.v1.cli.eval.runner import run_eval
12
12
  from verifiers.v1.cli.output import (
13
- TRACES_FILE,
14
13
  create_attempt_log_dir,
15
14
  output_path,
16
15
  saved_config_path,
@@ -26,6 +25,7 @@ from verifiers.v1.cli.resolve import (
26
25
  from verifiers.v1.configs.cli.eval import EvalConfig
27
26
  from verifiers.v1.utils.interrupt import install_interrupt
28
27
  from verifiers.v1.utils.logging import setup_logging
28
+ from verifiers.v1.utils.trace_store import TRACES_FILE
29
29
 
30
30
  logger = logging.getLogger(__name__)
31
31
 
@@ -16,9 +16,9 @@ from pathlib import Path
16
16
 
17
17
  from pydantic_core import from_json
18
18
 
19
- from verifiers.v1.cli.output import TRACES_FILE
20
19
  from verifiers.v1.episode import WireEpisode
21
20
  from verifiers.v1.task import task_key
21
+ from verifiers.v1.utils.trace_store import TRACES_FILE
22
22
 
23
23
 
24
24
  def load(
@@ -17,11 +17,7 @@ from typing import TypeVar, cast
17
17
  from verifiers.v1.cli.dashboard import dashboard
18
18
  from verifiers.v1.cli.eval import resume
19
19
  from verifiers.v1.cli.eval.hint import PRIME_RL_HINT
20
- from verifiers.v1.cli.output import (
21
- append_episode,
22
- output_path,
23
- save_config,
24
- )
20
+ from verifiers.v1.cli.output import output_path, save_config
25
21
  from verifiers.v1.cli.resume import distribute
26
22
  from verifiers.v1.clients import ModelContext
27
23
  from verifiers.v1.configs.cli.eval import EvalConfig
@@ -35,6 +31,7 @@ from verifiers.v1.utils.platform import (
35
31
  log_episodes,
36
32
  open_run,
37
33
  )
34
+ from verifiers.v1.utils.trace_store import append_episode
38
35
 
39
36
  logger = logging.getLogger(__name__)
40
37
 
verifiers/v1/cli/gepa.py CHANGED
@@ -16,7 +16,7 @@ import uuid
16
16
  from pydantic_config import cli
17
17
 
18
18
  import verifiers.v1 as vf
19
- from verifiers.v1.cli.output import TRACES_FILE, output_path, write_config
19
+ from verifiers.v1.cli.output import output_path, write_config
20
20
  from verifiers.v1.cli.resolve import (
21
21
  extract_id,
22
22
  narrow_config,
@@ -27,6 +27,7 @@ from verifiers.v1.cli.resolve import (
27
27
  from verifiers.v1.gepa import GEPAConfig, run_gepa
28
28
  from verifiers.v1.utils.interrupt import install_interrupt
29
29
  from verifiers.v1.utils.logging import setup_logging
30
+ from verifiers.v1.utils.trace_store import TRACES_FILE
30
31
 
31
32
  logger = logging.getLogger(__name__)
32
33
 
@@ -1,31 +1,13 @@
1
- """On-disk output: traces.jsonl (one rollout episode per line) + configs/<cli>.json.
2
-
3
- Each line is an `Episode` — the episode's standing (`id`/`env`/`errors`) inlined
4
- next to its flat, self-contained traces — so an episode persists whole or not at all: a torn line is the
5
- whole episode owed on resume, and a failure before any trace minted still leaves
6
- its errors on disk. The JSON file is the run's resolved config in the format the
7
- CLI reads (`@ configs/<cli>.json`), so a run is re-runnable from its own output. Lines
8
- append as episodes complete, so results are durable mid-run. Files written
9
- by this surface contain episodes only.
10
- """
11
-
12
- import asyncio
1
+ """CLI output directories, saved launch/config files, and per-attempt log paths."""
2
+
13
3
  import json
14
4
  import os
15
- from functools import cache
16
5
  from pathlib import Path
17
6
 
18
- from pydantic import BaseModel, TypeAdapter
7
+ from pydantic import BaseModel
19
8
 
20
9
  from verifiers.v1.configs.cli.eval import EvalConfig
21
- from verifiers.v1.episode import EnvInfo, Episode, WireEpisode
22
- from verifiers.v1.state import StateT
23
- from verifiers.v1.task import DataT
24
- from verifiers.v1.trace import AgentConfigT, Trace
25
- from verifiers.v1.utils.aio import run_shielded
26
-
27
- TRACES_FILE = "traces.jsonl"
28
- """Filename a run's rollout episodes are written to (one JSON episode per line)."""
10
+ from verifiers.v1.utils.trace_store import TRACES_FILE
29
11
 
30
12
  CONFIG_DIR = "configs"
31
13
  """Directory inside a run dir holding its configs: the launch TOML copied verbatim to
@@ -78,10 +60,6 @@ def saved_config_path(run_dir: Path) -> Path | None:
78
60
  return candidates[0] if candidates else None
79
61
 
80
62
 
81
- # Compiling an adapter is the expensive part; run output reuses only a few model classes.
82
- type_adapter = cache(TypeAdapter)
83
-
84
-
85
63
  def output_path(config: EvalConfig) -> Path:
86
64
  """Where this run writes: `output_dir / run.dir` — the same grouping convention as
87
65
  training. The run directory defaults to the auto-generated run name
@@ -142,64 +120,3 @@ def save_config(
142
120
  (results_dir / TRACES_FILE).write_text(
143
121
  ""
144
122
  ) # fresh; appended to as rollouts complete
145
-
146
-
147
- def write_episode(
148
- results_dir: Path, episode: Episode[DataT, StateT, AgentConfigT]
149
- ) -> None:
150
- """Serialize and append one rollout episode in the worker thread."""
151
- # Preserve fields declared by typed Trace subclasses nested in the episode.
152
- data = type_adapter(type(episode)).dump_json(episode, exclude_none=True)
153
- with (results_dir / TRACES_FILE).open("ab") as f:
154
- f.write(data + b"\n")
155
-
156
-
157
- def read_episodes(results_dir: Path, trace_type: type) -> list[WireEpisode]:
158
- """Load a run's saved rollouts from `traces.jsonl` with traces typed as
159
- `trace_type` (`Trace[WireTaskData, ...]` reads any taskset's file without
160
- importing it)."""
161
- trace_adapter = type_adapter(trace_type)
162
- episodes: list[WireEpisode] = []
163
- with (results_dir / TRACES_FILE).open(encoding="utf-8") as f:
164
- for line in f:
165
- if not line.strip():
166
- continue
167
- row = json.loads(line)
168
- record = WireEpisode.model_validate({**row, "traces": []})
169
- record.traces = [
170
- trace_adapter.validate_python(trace) for trace in row["traces"]
171
- ]
172
- episodes.append(record)
173
- return episodes
174
-
175
-
176
- async def append_episode(
177
- results_dir: Path,
178
- episode: Episode[DataT, StateT, AgentConfigT],
179
- lock: asyncio.Lock,
180
- ) -> None:
181
- """Append one finished rollout episode without blocking the event loop. The run's
182
- shared lock preserves whole-line ordering, and awaiting the worker preserves
183
- per-episode durability."""
184
-
185
- async def persist() -> None:
186
- async with lock:
187
- await asyncio.to_thread(write_episode, results_dir, episode)
188
-
189
- # Run lock acquisition and the worker to completion even under cancellation, so
190
- # finalized episodes are never lost mid-write (`run_shielded` re-raises the cancellation).
191
- await run_shielded(persist())
192
-
193
-
194
- async def append_trace(
195
- results_dir: Path, trace: Trace, lock: asyncio.Lock, env: str = ""
196
- ) -> None:
197
- """Append one finished trace as a single-agent rollout episode — debug and replay,
198
- which complete trace-at-a-time, both go through here."""
199
- episode = Episode(
200
- env=EnvInfo(id=env),
201
- task=trace.task,
202
- traces=[trace],
203
- ok=trace.ok,
204
- )
205
- await append_episode(results_dir, episode, lock)
@@ -22,13 +22,7 @@ from pydantic_config import cli
22
22
 
23
23
  import verifiers.v1 as vf
24
24
  from verifiers.v1.cli.dashboard.replay import ReplayProgress, replay_dashboard
25
- from verifiers.v1.cli.output import (
26
- append_trace,
27
- read_episodes,
28
- save_config,
29
- saved_config_path,
30
- write_config,
31
- )
25
+ from verifiers.v1.cli.output import save_config, saved_config_path, write_config
32
26
  from verifiers.v1.cli.resolve import narrow_taskset_config
33
27
  from verifiers.v1.configs.agent import WireAgentConfig
34
28
  from verifiers.v1.configs.cli.replay import ReplayConfig
@@ -37,6 +31,7 @@ from verifiers.v1.task import Task, WireTaskData
37
31
  from verifiers.v1.trace import Trace
38
32
  from verifiers.v1.utils.interrupt import install_interrupt
39
33
  from verifiers.v1.utils.logging import setup_logging
34
+ from verifiers.v1.utils.trace_store import append_trace, read_episodes
40
35
 
41
36
  logger = logging.getLogger(__name__)
42
37
 
@@ -1,6 +1,8 @@
1
1
  """One env agent's config: who plays the seat, and its per-run caps."""
2
2
 
3
- from pydantic import SerializeAsAny, model_validator
3
+ from typing import Any
4
+
5
+ from pydantic import BaseModel, SerializeAsAny, model_validator
4
6
  from pydantic_config import BaseConfig
5
7
 
6
8
  from verifiers.v1.clients import ClientConfig
@@ -8,6 +10,7 @@ from verifiers.v1.configs.harness import HarnessConfig, WireHarnessConfig
8
10
  from verifiers.v1.configs.retries import RetryConfig
9
11
  from verifiers.v1.runtimes import PrimeConfig, RuntimeConfig
10
12
  from verifiers.v1.types import SamplingConfig
13
+ from verifiers.v1.utils.generic import deep_merge
11
14
 
12
15
 
13
16
  class TimeoutConfig(BaseConfig):
@@ -78,3 +81,50 @@ class WireAgentConfig(AgentConfig):
78
81
  def _resolve_harness(cls, data):
79
82
  """Override: a record read resolves no plugins."""
80
83
  return data
84
+
85
+
86
+ def agent_config_fields(config: BaseModel) -> dict[str, AgentConfig]:
87
+ """Top-level agent configs, in declaration order, keyed by their field names."""
88
+ return {name: value for name, value in config if isinstance(value, AgentConfig)}
89
+
90
+
91
+ def merge_agent_defaults(config: type[BaseModel], data: Any) -> Any:
92
+ """Merge partial agent overrides onto their declared defaults."""
93
+ if isinstance(data, dict):
94
+ for name, field in config.model_fields.items():
95
+ if isinstance(field.default, AgentConfig) and isinstance(
96
+ data.get(name), dict
97
+ ):
98
+ data[name] = deep_merge(
99
+ field.default.model_dump(exclude_none=True), data[name]
100
+ )
101
+ return data
102
+
103
+
104
+ def resolve_agent(
105
+ spec: AgentConfig,
106
+ *,
107
+ model: str | None = None,
108
+ client: ClientConfig | None = None,
109
+ sampling: SamplingConfig | None = None,
110
+ harness: HarnessConfig | None = None,
111
+ ) -> AgentConfig:
112
+ """`spec` with what it leaves unset filled from the run's defaults; its own
113
+ sampling values merge over the run's. The one place a seat's identity resolves,
114
+ for configured agent roles."""
115
+ merged = spec.sampling if sampling is None else sampling
116
+ if sampling is not None and spec.sampling is not None:
117
+ merged = sampling.model_copy(
118
+ update=deep_merge(
119
+ sampling.model_dump(exclude_unset=True),
120
+ spec.sampling.model_dump(exclude_unset=True),
121
+ )
122
+ )
123
+ return spec.model_copy(
124
+ update={
125
+ "harness": spec.harness if spec.harness is not None else harness,
126
+ "model": spec.model if spec.model is not None else model,
127
+ "client": spec.client if spec.client is not None else client,
128
+ "sampling": merged,
129
+ }
130
+ )
@@ -6,13 +6,16 @@ from typing import get_args
6
6
  from pydantic import Field, SerializeAsAny, model_validator
7
7
  from pydantic_config import BaseConfig
8
8
 
9
- from verifiers.v1.configs.agent import AgentConfig
9
+ from verifiers.v1.configs.agent import (
10
+ AgentConfig,
11
+ agent_config_fields,
12
+ merge_agent_defaults,
13
+ )
10
14
  from verifiers.v1.configs.harness import HarnessConfig
11
15
  from verifiers.v1.configs.retries import RetryConfig
12
16
  from verifiers.v1.configs.taskset import TasksetConfig
13
17
  from verifiers.v1.interception import ElasticInterceptionPoolConfig, InterceptionConfig
14
18
  from verifiers.v1.types import ID
15
- from verifiers.v1.utils.generic import deep_merge
16
19
 
17
20
 
18
21
  class TimeoutConfig(BaseConfig):
@@ -73,7 +76,7 @@ class EnvConfig(BaseConfig):
73
76
  default = default_agent_harness(self.taskset.id)
74
77
  return {
75
78
  name: cfg.harness if cfg.harness is not None else default
76
- for name, cfg in _declared_agent_configs(self).items()
79
+ for name, cfg in agent_config_fields(self).items()
77
80
  }
78
81
 
79
82
  @model_validator(mode="before")
@@ -110,17 +113,7 @@ class EnvConfig(BaseConfig):
110
113
  @model_validator(mode="before")
111
114
  @classmethod
112
115
  def _merge_role_defaults(cls, data):
113
- """Deep-merge partial role data over the field's declared default — plain
114
- validation would replace the instance wholesale, resetting its other pins."""
115
- if isinstance(data, dict):
116
- for name, field in cls.model_fields.items():
117
- if isinstance(field.default, AgentConfig) and isinstance(
118
- data.get(name), dict
119
- ):
120
- data[name] = deep_merge(
121
- field.default.model_dump(exclude_none=True), data[name]
122
- )
123
- return data
116
+ return merge_agent_defaults(cls, data)
124
117
 
125
118
  @classmethod
126
119
  def __pydantic_init_subclass__(cls, **kwargs):
@@ -146,16 +139,6 @@ class EnvConfig(BaseConfig):
146
139
  )
147
140
 
148
141
 
149
- def _declared_agent_configs(config: EnvConfig) -> dict[str, AgentConfig]:
150
- """The `AgentConfig` fields declared on an env's config, in declaration order —
151
- the env's roles, each seat keyed by its field name (the only naming site)."""
152
- return {
153
- name: getattr(config, name)
154
- for name, field in type(config).model_fields.items()
155
- if isinstance(field.default, AgentConfig)
156
- }
157
-
158
-
159
142
  def default_agent_harness(taskset_id: str) -> HarnessConfig:
160
143
  """What an unpinned role's `harness=None` resolves to: the taskset's bundled
161
144
  harness when it ships one, else the built-in `bash`."""
verifiers/v1/env.py CHANGED
@@ -14,12 +14,12 @@ from typing import (
14
14
 
15
15
  from verifiers.v1.agent import Agent, Agents, _EpisodeAgent
16
16
  from verifiers.v1.clients import ModelContext
17
- from verifiers.v1.configs.agent import AgentConfig
18
- from verifiers.v1.configs.env import (
19
- EnvConfig,
20
- _declared_agent_configs,
21
- default_agent_harness,
17
+ from verifiers.v1.configs.agent import (
18
+ AgentConfig,
19
+ agent_config_fields,
20
+ resolve_agent,
22
21
  )
22
+ from verifiers.v1.configs.env import EnvConfig, default_agent_harness
23
23
  from verifiers.v1.episode import EnvInfo, Episode
24
24
  from verifiers.v1.errors import EnvError, boundary
25
25
  from verifiers.v1.harness import Harness, HarnessConfig
@@ -32,7 +32,7 @@ from verifiers.v1.mcp import SharedToolServer, serve_shared
32
32
  from verifiers.v1.runtimes import SubprocessConfig, runtime_is_local
33
33
  from verifiers.v1.task import Task
34
34
  from verifiers.v1.trace import Error, Trace, TraceTask
35
- from verifiers.v1.utils.generic import concrete_type, deep_merge
35
+ from verifiers.v1.utils.generic import concrete_type
36
36
  from verifiers.v1.utils.memory import trim_memory_periodically
37
37
  from verifiers.v1.utils.retries import run_episode_with_retry
38
38
 
@@ -105,7 +105,7 @@ class Env(ABC, Generic[ConfigT]):
105
105
  self._default_harness = default_agent_harness(config.taskset.id)
106
106
  task_cls = type(self.taskset).task_type()
107
107
  self._task_cls: type[Task] = task_cls
108
- self._agent_specs: dict[str, AgentConfig] = _declared_agent_configs(self.config)
108
+ self._agent_specs: dict[str, AgentConfig] = agent_config_fields(self.config)
109
109
  if not self._agent_specs:
110
110
  raise ValueError(
111
111
  f"{type(self).__name__} declares no agents; declare each as an "
@@ -205,23 +205,12 @@ class Env(ABC, Generic[ConfigT]):
205
205
 
206
206
  def make(name: str, spec: AgentConfig) -> Agent:
207
207
  # Unpinned fields fall back to the run's ctx / the taskset's harness.
208
- sampling = ctx.sampling
209
- if spec.sampling is not None:
210
- sampling = sampling.model_copy(
211
- update=deep_merge(
212
- sampling.model_dump(exclude_unset=True),
213
- spec.sampling.model_dump(exclude_unset=True),
214
- )
215
- )
216
- resolved = spec.model_copy(
217
- update={
218
- "harness": spec.harness
219
- if spec.harness is not None
220
- else self._default_harness,
221
- "model": spec.model if spec.model is not None else ctx.model,
222
- "sampling": sampling,
223
- "client": spec.client if spec.client is not None else ctx.client,
224
- }
208
+ resolved = resolve_agent(
209
+ spec,
210
+ model=ctx.model,
211
+ client=ctx.client,
212
+ sampling=ctx.sampling,
213
+ harness=self._default_harness,
225
214
  )
226
215
  return _EpisodeAgent(
227
216
  resolved,
@@ -11,7 +11,7 @@ import logging
11
11
  from gepa.api import optimize
12
12
  from gepa.core.result import GEPAResult
13
13
 
14
- from verifiers.v1.cli.output import append_episode, output_path, save_config
14
+ from verifiers.v1.cli.output import output_path, save_config
15
15
  from verifiers.v1.clients import ModelContext
16
16
  from verifiers.v1.env import Env
17
17
  from verifiers.v1.episode import Episode
@@ -22,6 +22,7 @@ from verifiers.v1.gepa.dataset import (
22
22
  split_tasks,
23
23
  )
24
24
  from verifiers.v1.gepa.reflection import build_reflection_lm
25
+ from verifiers.v1.utils.trace_store import append_episode
25
26
 
26
27
  logger = logging.getLogger(__name__)
27
28
 
@@ -4,7 +4,6 @@ import argparse
4
4
  import asyncio
5
5
  import json
6
6
  import logging
7
- import random
8
7
  import subprocess
9
8
  from contextlib import AsyncExitStack
10
9
  from pathlib import Path
@@ -13,6 +12,13 @@ from typing import TYPE_CHECKING
13
12
  import httpx
14
13
  from openai import APIConnectionError, APIStatusError, AsyncOpenAI, omit
15
14
  from openai.lib.streaming.chat import AsyncChatCompletionStream
15
+ from tenacity import (
16
+ AsyncRetrying,
17
+ before_sleep_log,
18
+ retry_if_exception_type,
19
+ stop_after_attempt,
20
+ wait_random_exponential,
21
+ )
16
22
 
17
23
  if TYPE_CHECKING:
18
24
  # The harness bundles this module into the generated script before execution.
@@ -247,9 +253,16 @@ async def chat(
247
253
  kwargs["tools"] = tools
248
254
  if tools and tool_choice is not None:
249
255
  kwargs["tool_choice"] = tool_choice
250
- for attempt in range(client.max_retries + 1):
256
+ async for attempt in AsyncRetrying(
257
+ retry=retry_if_exception_type((APIConnectionError, httpx.TransportError)),
258
+ stop=stop_after_attempt(client.max_retries + 1),
259
+ wait=wait_random_exponential(multiplier=0.5, max=8.0),
260
+ before_sleep=before_sleep_log(logging.getLogger(__name__), logging.WARNING),
261
+ reraise=True,
262
+ ):
251
263
  # Reuse the interception server's body-digest replay guard on stream retries.
252
- headers = {"x-stainless-retry-count": str(attempt)} if attempt else omit
264
+ retry_count = attempt.retry_state.attempt_number - 1
265
+ headers = {"x-stainless-retry-count": str(retry_count)} if retry_count else omit
253
266
  raw_stream = await client.chat.completions.create(
254
267
  **kwargs,
255
268
  stream=True,
@@ -257,19 +270,8 @@ async def chat(
257
270
  extra_headers=headers,
258
271
  )
259
272
  # The SDK retries request setup; only stream consumption is retried here.
260
- try:
273
+ with attempt:
261
274
  return await _read_chat_completion(raw_stream)
262
- except (APIConnectionError, httpx.TransportError) as error:
263
- cause = error.__cause__ or error
264
- if attempt == client.max_retries:
265
- raise
266
- logging.getLogger(__name__).warning(
267
- "Retrying interrupted model stream (%s/%s): %r",
268
- attempt + 1,
269
- client.max_retries,
270
- cause,
271
- )
272
- await asyncio.sleep(min(0.5 * 2**attempt, 8.0) * random.uniform(0.75, 1.0))
273
275
 
274
276
 
275
277
  async def _read_chat_completion(raw_stream):
@@ -17,13 +17,18 @@ from urllib.parse import urlsplit
17
17
 
18
18
  from verifiers.v1.configs.runtime import NetworkPolicyConfig
19
19
  from verifiers.v1.errors import SandboxError
20
- from verifiers.v1.runtimes.base import SERVICE_PORT, BaseRuntimeInfo, parse_gpu
20
+ from verifiers.v1.runtimes.base import (
21
+ SERVICE_PORT,
22
+ BaseRuntimeInfo,
23
+ parse_gpu,
24
+ )
21
25
  from verifiers.v1.runtimes.container import ContainerConfig, ContainerRuntime, cli
22
26
  from verifiers.v1.runtimes.docker.egress import (
23
27
  EgressProxy,
24
28
  NetworkPolicy,
25
29
  is_loopback_host,
26
30
  )
31
+ from verifiers.v1.utils.scope import run_scope
27
32
 
28
33
  logger = logging.getLogger(__name__)
29
34
 
@@ -157,9 +162,11 @@ class DockerRuntime(ContainerRuntime):
157
162
  for key, value in self.env.items()
158
163
  for arg in ("--env", f"{key}={value}")
159
164
  ]
165
+ self._label_args = ["--label", f"verifiers.run={run_scope()}"]
160
166
  run = await cli(
161
167
  self.engine,
162
168
  "run",
169
+ *self._label_args,
163
170
  "--detach",
164
171
  *options,
165
172
  *env_args,
@@ -296,6 +303,7 @@ class DockerRuntime(ContainerRuntime):
296
303
  helper = await cli(
297
304
  self.engine,
298
305
  "run",
306
+ *self._label_args,
299
307
  "--rm",
300
308
  "--user",
301
309
  "0",
@@ -385,6 +393,7 @@ class DockerRuntime(ContainerRuntime):
385
393
  cut = await cli(
386
394
  self.engine,
387
395
  "run",
396
+ *self._label_args,
388
397
  "--rm",
389
398
  "--network",
390
399
  f"container:{self._container}",
@@ -185,12 +185,14 @@ class PrimeRuntime(Runtime):
185
185
  "gpu_type": gpu_type,
186
186
  "region": self.config.region,
187
187
  }
188
+ scope = run_scope()
189
+ labels = [*BASE_LABELS, *self.config.labels, scope]
188
190
  try:
189
191
  async with (
190
192
  creation_limiter(
191
193
  (self.config.creates_per_min or 0) / 60,
192
194
  "prime-sandbox",
193
- run_scope(),
195
+ scope,
194
196
  )
195
197
  or contextlib.nullcontext()
196
198
  ):
@@ -201,9 +203,7 @@ class PrimeRuntime(Runtime):
201
203
  sandbox = await self._client.create(
202
204
  CreateSandboxRequest(
203
205
  name=self.name,
204
- labels=list(
205
- dict.fromkeys([*BASE_LABELS, *self.config.labels])
206
- ),
206
+ labels=list(dict.fromkeys(labels)),
207
207
  docker_image=self.config.image,
208
208
  environment_vars=self.env,
209
209
  **{k: v for k, v in options.items() if v is not None},
@@ -1,4 +1,4 @@
1
- """Turn-by-turn episode deltas over the env-serve wire.
1
+ """Incremental trace deltas for episode streams and file writers.
2
2
 
3
3
  The worker streams a served episode as it grows. Each trace announces its own changes
4
4
  (`Trace.notify`, fired by the rollout at every phase change and by the interception proxy
@@ -25,7 +25,7 @@ from __future__ import annotations
25
25
  import asyncio
26
26
  import copy
27
27
  import logging
28
- from collections.abc import Awaitable, Callable
28
+ from collections.abc import Awaitable, Callable, Iterable
29
29
  from typing import TYPE_CHECKING, Any, Self
30
30
 
31
31
  import msgpack
@@ -36,7 +36,6 @@ from verifiers.v1.graph import _decode_ndarray, _encode_ndarray
36
36
  from verifiers.v1.serve.encoding import msgpack_encoder
37
37
 
38
38
  if TYPE_CHECKING:
39
- from verifiers.v1.env import RunSlot
40
39
  from verifiers.v1.trace import Trace
41
40
 
42
41
  logger = logging.getLogger(__name__)
@@ -102,16 +101,19 @@ class TraceCursor:
102
101
 
103
102
 
104
103
  class DeltaStreamer:
105
- """Sends a `RunSlot`'s trace changes as deltas, one flush per burst of changes.
104
+ """Sends the current traces' changes as deltas, one flush per burst of changes.
106
105
 
107
- `watch` subscribes a minted trace (pass it as `run_slot`'s `on_trace`); each
106
+ `watch` subscribes a minted trace (pass it as the runner's `on_trace`); each
108
107
  `Trace.notify` schedules a flush, and changes landing in the same loop iteration
109
108
  ride one flush. Use as an async context manager around the rollout: a clean exit
110
- flushes the final state (the slot then holds the finished episode's traces), a
111
- cancelled rollout flushes nothing — its client has already gone."""
109
+ flushes the final state; a cancelled rollout settles pending sends without a final flush."""
112
110
 
113
- def __init__(self, slot: RunSlot, send: Callable[[bytes], Awaitable[None]]) -> None:
114
- self.slot = slot
111
+ def __init__(
112
+ self,
113
+ traces: Callable[[], Iterable[Trace]],
114
+ send: Callable[[dict], Awaitable[None]],
115
+ ) -> None:
116
+ self.traces = traces
115
117
  self.send = send
116
118
  self.cursors: dict[str, TraceCursor] = {}
117
119
  self._scheduled: asyncio.Handle | None = None
@@ -132,6 +134,7 @@ class DeltaStreamer:
132
134
  else:
133
135
  for task in self._flushes:
134
136
  task.cancel()
137
+ await asyncio.gather(*self._flushes, return_exceptions=True)
135
138
 
136
139
  def watch(self, trace: Trace) -> None:
137
140
  trace.watch(self._changed)
@@ -153,7 +156,7 @@ class DeltaStreamer:
153
156
  async with self._lock:
154
157
  for trace_id, delta, cursor in self.diff():
155
158
  try:
156
- await self.send(pack(delta))
159
+ await self.send(delta)
157
160
  except Exception: # the cursor stays put: the next flush diffs it again
158
161
  logger.warning(
159
162
  "failed to send delta for %s", trace_id, exc_info=True
@@ -193,7 +196,7 @@ class DeltaStreamer:
193
196
  """Each trace's delta against its sent cursor, with the cursor as it stands once
194
197
  that delta is sent (None for a discard). Nothing here is committed: `flush`
195
198
  stores a cursor only after its delta left."""
196
- traces = list(self.slot.traces)
199
+ traces = list(self.traces())
197
200
  live = {trace.id for trace in traces}
198
201
  deltas: list[tuple[str, dict, TraceCursor | None]] = []
199
202
  for trace_id in [trace_id for trace_id in self.cursors if trace_id not in live]:
@@ -8,7 +8,7 @@ import zmq.asyncio
8
8
  from verifiers.v1.clients import ModelContext
9
9
  from verifiers.v1.configs.client import ClientConfig
10
10
  from verifiers.v1.configs.env import EnvConfig
11
- from verifiers.v1.serve.delta import DeltaStreamer, TraceSummary, dump
11
+ from verifiers.v1.serve.delta import DeltaStreamer, TraceSummary, dump, pack
12
12
  from verifiers.v1.serve.encoding import msgpack_encoder
13
13
  from verifiers.v1.serve.types import (
14
14
  BaseResponse,
@@ -100,15 +100,15 @@ class EnvServer:
100
100
  ctx = self._context(req.client, req.model, req.sampling)
101
101
  (slot,) = self.env.slots(self._build_task(req.task_data))
102
102
 
103
- async def send_delta(data: bytes) -> None:
103
+ async def send_delta(delta: dict) -> None:
104
104
  await self.frontend.send_multipart(
105
- [client_id, request_id, b"delta", data], copy=False
105
+ [client_id, request_id, b"delta", pack(delta)], copy=False
106
106
  )
107
107
 
108
108
  # The gate spans requests: `--max-concurrent` bounds this worker's episodes
109
109
  # in flight the same way the in-process eval's semaphore does. The streamer
110
110
  # ships each trace as it changes; the reply below carries only the rest.
111
- async with DeltaStreamer(slot, send_delta) as streamer:
111
+ async with DeltaStreamer(lambda: slot.traces, send_delta) as streamer:
112
112
  episode = await self.env.run_slot(
113
113
  slot, ctx, self._gate, on_trace=streamer.watch
114
114
  )
@@ -0,0 +1,80 @@
1
+ """Read and write standard saved traces."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ from functools import cache
8
+ from pathlib import Path
9
+
10
+ from pydantic import TypeAdapter
11
+
12
+ from verifiers.v1.episode import EnvInfo, Episode, WireEpisode
13
+ from verifiers.v1.state import StateT
14
+ from verifiers.v1.task import DataT
15
+ from verifiers.v1.trace import AgentConfigT, Trace
16
+ from verifiers.v1.utils.aio import run_shielded
17
+
18
+ TRACES_FILE = "traces.jsonl"
19
+ type_adapter = cache(TypeAdapter)
20
+
21
+
22
+ def write_episode(
23
+ results_dir: Path, episode: Episode[DataT, StateT, AgentConfigT]
24
+ ) -> None:
25
+ """Serialize and append one rollout episode in the worker thread."""
26
+ # Preserve fields declared by typed Trace subclasses nested in the episode.
27
+ data = type_adapter(type(episode)).dump_json(episode, exclude_none=True)
28
+ with (results_dir / TRACES_FILE).open("ab") as f:
29
+ f.write(data + b"\n")
30
+
31
+
32
+ def read_episodes(results_dir: Path, trace_type: type) -> list[WireEpisode]:
33
+ """Load a run's saved rollouts from `traces.jsonl` with traces typed as
34
+ `trace_type` (`Trace[WireTaskData, ...]` reads any taskset's file without
35
+ importing it)."""
36
+ trace_adapter = type_adapter(trace_type)
37
+ episodes: list[WireEpisode] = []
38
+ with (results_dir / TRACES_FILE).open(encoding="utf-8") as f:
39
+ for line in f:
40
+ if not line.strip():
41
+ continue
42
+ row = json.loads(line)
43
+ record = WireEpisode.model_validate({**row, "traces": []})
44
+ record.traces = [
45
+ trace_adapter.validate_python(trace) for trace in row["traces"]
46
+ ]
47
+ episodes.append(record)
48
+ return episodes
49
+
50
+
51
+ async def append_episode(
52
+ results_dir: Path,
53
+ episode: Episode[DataT, StateT, AgentConfigT],
54
+ lock: asyncio.Lock,
55
+ ) -> None:
56
+ """Append one finished rollout episode without blocking the event loop. The run's
57
+ shared lock preserves whole-line ordering, and awaiting the worker preserves
58
+ per-episode durability."""
59
+
60
+ async def persist() -> None:
61
+ async with lock:
62
+ await asyncio.to_thread(write_episode, results_dir, episode)
63
+
64
+ # Run lock acquisition and the worker to completion even under cancellation, so
65
+ # finalized episodes are never lost mid-write (`run_shielded` re-raises the cancellation).
66
+ await run_shielded(persist())
67
+
68
+
69
+ async def append_trace(
70
+ results_dir: Path, trace: Trace, lock: asyncio.Lock, env: str = ""
71
+ ) -> None:
72
+ """Append one finished trace as a single-agent rollout episode — debug and replay,
73
+ which complete trace-at-a-time, both go through here."""
74
+ episode = Episode(
75
+ env=EnvInfo(id=env),
76
+ task=trace.task,
77
+ traces=[trace],
78
+ ok=trace.ok,
79
+ )
80
+ await append_episode(results_dir, episode, lock)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: verifiers
3
- Version: 0.3.2.dev136
3
+ Version: 0.3.2.dev138
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -1,7 +1,7 @@
1
1
  verifiers/__init__.py,sha256=F5IVMw6mhGaPBMCw7QKk0la_r5mMwydP2hdTTdQucik,352
2
2
  verifiers/v1/__init__.py,sha256=zzXyzUkCJCXfPRajYNNQlgIAz6_-uu9oYNb-XsbIjeU,8434
3
- verifiers/v1/agent.py,sha256=Lc4ySCWBCZRqED-M2LwpPcewAH9k4AuS_ubq-E6-wDQ,32808
4
- verifiers/v1/env.py,sha256=Hon7gcepciMYgki_jN4-l4XCcKRZRYuYm7c17nfeiFE,18520
3
+ verifiers/v1/agent.py,sha256=10hXkda0sC5FxT2a32t9En5QlExqcxgnBYtGYk7oins,32651
4
+ verifiers/v1/env.py,sha256=5xB51-IwVEMoEzQ44vV2HD1O2VFcX3H3-vPBCPl_49o,17956
5
5
  verifiers/v1/episode.py,sha256=DKUkG9YWBRbaqAP7VCg9DcOB95WSH_EM_LY4jkJ63hA,5817
6
6
  verifiers/v1/errors.py,sha256=uQySmCUrjDYFkBIii1vBFPiEcPFPQZ9EGpas_6HvcQI,5279
7
7
  verifiers/v1/graph.py,sha256=3HtCYL36MDAoiBk7u8yi6FxUBkIqQ7VYn9_IL0NEOog,41559
@@ -18,33 +18,33 @@ verifiers/v1/types.py,sha256=xPetvJxksKPYKic389RFnVvy2vN-gD2bPZFZoEyzmp4,10191
18
18
  verifiers/v1/acp/__init__.py,sha256=9RySmxFeEMT2XSJy5wYevqEhQdul2jHl9f5XAribG3A,13631
19
19
  verifiers/v1/acp/runner.py,sha256=zPo-2ZXmFMmQchhD7nNzlZUrabG09qiBvpB67KVGm3M,14095
20
20
  verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
21
- verifiers/v1/cli/debug.py,sha256=sgXi2y17qr2yd5kbkL_IzwWN1nIFBOa1adu7bcFzaVk,11780
22
- verifiers/v1/cli/gepa.py,sha256=F0Q1ApfUL_mInFnmaJ5-yBDRYBtBuH72OkWelPoGCp4,4332
21
+ verifiers/v1/cli/debug.py,sha256=G05-k9mqsMNIf7L-Vlxgdmw3QFwFF94-FodKXV7GYbQ,11822
22
+ verifiers/v1/cli/gepa.py,sha256=IyS3YZCCHVydb-OyPeGdtCYsz6OYqXXddXqYq8RLucM,4374
23
23
  verifiers/v1/cli/init.py,sha256=6wr14011_Dygv10R6_YWg5A3oSrz187E2WspGRHIfZw,7608
24
- verifiers/v1/cli/output.py,sha256=lAl6vwKf9mVuNZb8PzOexv6Q2h-H6t0oUPcapffY3uY,8351
25
- verifiers/v1/cli/replay.py,sha256=JjhaGczQ41zN6wz1P1nbGGvv0xdfQlX4OmZItGPDpZc,10395
24
+ verifiers/v1/cli/output.py,sha256=4FRuQ6UIZCigMfUUO2nCFFJKxls3DtWDXswMCVTxjIA,5010
25
+ verifiers/v1/cli/replay.py,sha256=ukCmu4-SGYZa2kB9TpNERUYT6mAVEjNJxRNqwIH5eYk,10412
26
26
  verifiers/v1/cli/resolve.py,sha256=Q1EHM7wWQo0YkwJA898qtZbYLaIkjIVq3Q7kUp2YIxs,4346
27
27
  verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,470
28
28
  verifiers/v1/cli/validate.py,sha256=Z_rUldl8jzXz52oy3FBIRGr4HYhWFXrq43WmU0sJ-4M,17325
29
29
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
30
30
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
31
- verifiers/v1/cli/dashboard/eval.py,sha256=HEXPEegNXdgU9zAIuHzsWsoSIHyiOcDZh4tCGFNxKAk,38353
31
+ verifiers/v1/cli/dashboard/eval.py,sha256=69vKmLaEY1ll1wahjRGvO1_7QsUQ1URtZOG7ZawvRCw,38347
32
32
  verifiers/v1/cli/dashboard/replay.py,sha256=_X5up9MJsRbd_sWe4LgNTPNT8k1ip4xNpo50hxI3Ljk,2697
33
33
  verifiers/v1/cli/dashboard/validate.py,sha256=xrpK3Y90JsoDCQYLLvvsnJ4CPRMvSSJIghfHK62E6FQ,3687
34
34
  verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
35
35
  verifiers/v1/cli/eval/hint.py,sha256=NXP2_JYrQHguqr68LxOJG3lYAsue_VNoepyTVPn-N4E,256
36
- verifiers/v1/cli/eval/main.py,sha256=L_vKnMJEbpo1IKjbm0TXjXv_1x9we7mFv-jH_xY211Q,5946
37
- verifiers/v1/cli/eval/resume.py,sha256=fYkWa0Fudq--IWfk25j_McLm67bXYWvksWmJR4u4QAU,3811
38
- verifiers/v1/cli/eval/runner.py,sha256=z1yU2_rAA6tHz3PL7NQuzm9v3-4Ri18DXD5SRBKXqVY,6962
36
+ verifiers/v1/cli/eval/main.py,sha256=46aHpha6yhFzbqUrRogC5wC2XAwdZcpp-4siqLpUExY,5984
37
+ verifiers/v1/cli/eval/resume.py,sha256=cmFdxLwyU66lyAuXb7U8mNBB-8NLX41JdCg1EuU4w9g,3818
38
+ verifiers/v1/cli/eval/runner.py,sha256=JLitCJajj6h34L0Kq5fWklS7FAsiGOrdGp_c0qVN7Xc,6987
39
39
  verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
40
40
  verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
41
41
  verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
42
42
  verifiers/v1/clients/eval.py,sha256=cJucyUFsbF1JlG8nF0Tlw2EBMSoAF0YMklgYiOSoNpg,8322
43
43
  verifiers/v1/clients/train.py,sha256=jm_z_Gkwqo4IKEVN8WrDnEyJykXxmswFHzlsnFop7_I,18945
44
44
  verifiers/v1/configs/__init__.py,sha256=X7u6X7B3ieD1HZl79Tv1kg-yPHbT0QLvAebaGqxVtvA,317
45
- verifiers/v1/configs/agent.py,sha256=PiWMNueum67vnyLLpIgxXZDb3FTQEUFqo2Z1siOOJqs,3342
45
+ verifiers/v1/configs/agent.py,sha256=76ZvO0EE_JCEep96vOYfvAmC4fsLHtJn_cKbJhve6gU,5241
46
46
  verifiers/v1/configs/client.py,sha256=fPWtBJv5lPkqttwk0AF2ubfLdzxX2XCO0Zr-A9V4mtk,4440
47
- verifiers/v1/configs/env.py,sha256=-Bcdd4MEPfekVc_fnpiwsb9UDwCNdNCKrAHGoQuuXu8,7784
47
+ verifiers/v1/configs/env.py,sha256=xDcfoBcor6loLrK79YdwEdui5OGVLM_InVA8HnLcjPQ,6864
48
48
  verifiers/v1/configs/harness.py,sha256=ZrR-Egzfft95oMoqQ78bAq9TqP882Q68M_47Ajw9lE8,2918
49
49
  verifiers/v1/configs/judge.py,sha256=WEwA8D4X5JSaIP3c-E2wPnvgByWeT6JRAn1d8gfiQ40,2142
50
50
  verifiers/v1/configs/retries.py,sha256=-nPlgmz7J_NZ4GOYG6oCc3thXo4La2SufBmvQGvvPGo,821
@@ -81,7 +81,7 @@ verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,
81
81
  verifiers/v1/gepa/config.py,sha256=PhZRLkhaPciNm_IhwVX3e9DM8Y2IdCe4Fbi5WjH5aro,4883
82
82
  verifiers/v1/gepa/dataset.py,sha256=SPGG55BVbJ9HtCEvm-xyl52QSbldFwGqWvRIcaLelsM,2171
83
83
  verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
84
- verifiers/v1/gepa/runner.py,sha256=fGEyFzejdQKBzBbvdiTDHV3JtRhXMXMGSBU5y8C8i3w,6054
84
+ verifiers/v1/gepa/runner.py,sha256=s2eUV0T0mUGy-7XxupoX757z9kZoKAR73NN5M6rUi6E,6096
85
85
  verifiers/v1/harnesses/__init__.py,sha256=2JPwrRoYoigH-HfBtiFnatRdFEVHLv4rwlKfsFEibyE,1803
86
86
  verifiers/v1/harnesses/node.py,sha256=-lDPh30p65IETPVLkKn4yhEK49rtEGVQf9TNs7MVQYE,2133
87
87
  verifiers/v1/harnesses/bash/__init__.py,sha256=IAV4eMMqDHSl-9ANb0Q3kxmBZ22SeHZFBM448X4gQio,140
@@ -118,7 +118,7 @@ verifiers/v1/harnesses/terminus_2/harness.py,sha256=cKQJluW01mkxEccTLFwSWsVMavyi
118
118
  verifiers/v1/harnesses/terminus_2/program.py,sha256=X8MZdyItU3gbXGXnGOlJ9LWcUTnqIDaxlhkZvoRiqKY,2703
119
119
  verifiers/v1/harnesses/utils/__init__.py,sha256=eTS-Ft2HwoLiYn5u1t_KueLmaZB_W12cdYmcVsE-G5A,364
120
120
  verifiers/v1/harnesses/utils/compaction.py,sha256=DeY6CYLEsv7-vJm26sT2mRaPHu3W2ywiZhvFr_pi6Hg,10495
121
- verifiers/v1/harnesses/utils/core.py,sha256=Kqp0npmiJlGtT5vt7r2sX70R29W9d_ODR7gN-ufKrug,19642
121
+ verifiers/v1/harnesses/utils/core.py,sha256=A4B_TbZBQVd6IYQyytrO8gbmrO-aVqjRl0-w8GClWF4,19676
122
122
  verifiers/v1/harnesses/utils/install.py,sha256=BDkkUVZ1brSzCrbQOvMO5mVIE2DsV2pE7rTC5yROYUw,2992
123
123
  verifiers/v1/harnesses/utils/launch.py,sha256=RqPuUY-kT2XVuLSGtXW7kRQEaP-o2KzkiJ7vcoo2neM,2883
124
124
  verifiers/v1/harnesses/utils/mcp.py,sha256=oNhKfLjAW-wScJcDIsW_HfuvKY-JkaGBBZrRLqgRK1g,7355
@@ -145,16 +145,16 @@ verifiers/v1/runtimes/base.py,sha256=IKmrJSJltWX9M1nPUZExKluPJM4N82VsnqQGFz0ECV8
145
145
  verifiers/v1/runtimes/container.py,sha256=puQD8X9acP6GefoCNEnq40POi-nXPD-729bUjTVShfM,9554
146
146
  verifiers/v1/runtimes/limiters.py,sha256=rOfQJcMdYcLSKq8AL9O09BfGOLqpYmNgrCxXXNcHLmU,2849
147
147
  verifiers/v1/runtimes/modal.py,sha256=6ccitC0HVsAP_l8soJ_RQYNIVKrJjDsz2vAJBtxUZjw,17476
148
- verifiers/v1/runtimes/prime.py,sha256=Ly7eTEJY06-kjBBZhpJVALtsW23M-ZLJ1mlq27uSS9U,19158
148
+ verifiers/v1/runtimes/prime.py,sha256=vuRlBxuYLMLLy6J4cL2GhifnNhsLnzaN0e9FUN-yinc,19149
149
149
  verifiers/v1/runtimes/subprocess.py,sha256=QmpQ235_xrX6gowrr8557VgbydOXpLbfFq0KKGfJThk,8774
150
- verifiers/v1/runtimes/docker/__init__.py,sha256=2wceP-8V-w0dQBEeLtoJFfCZ135DxP2rcdcUBmSWiNU,18543
150
+ verifiers/v1/runtimes/docker/__init__.py,sha256=KaUbd87exTIT5JOoyNmEYFJ-JmnXf-p2MriBn2BdQtQ,18775
151
151
  verifiers/v1/runtimes/docker/egress.py,sha256=wtKWTL2_zFidxMR7lDRmfevQlOsl2Iyk88ebdO4Kqqw,24566
152
152
  verifiers/v1/serve/__init__.py,sha256=wOKuwfugzmDT4yIniwzCXlf4G6tv6Cg8Kc--WMa38Bs,553
153
153
  verifiers/v1/serve/client.py,sha256=E2y2x4qqYveWRWC9VZYQEu19aMEryPWTdXnYzbq5uTA,9829
154
- verifiers/v1/serve/delta.py,sha256=vbwFy9vGcSIL1gJad5gvLHr2vEBiPtl3-sO_HoHeqSM,13574
154
+ verifiers/v1/serve/delta.py,sha256=I3sJxS39rksOCeX9EO5FNWAR33Z-9tccxeEkU1txgLw,13615
155
155
  verifiers/v1/serve/encoding.py,sha256=hBZFucAZK9riXOV3DaHcskq9zHTT8gVGkoFVOPfnr30,2499
156
156
  verifiers/v1/serve/pool.py,sha256=bT-FOlIuOiBtOasuPB-p9887tkpEqvQUJBgn1Q45ph0,15912
157
- verifiers/v1/serve/server.py,sha256=i6XrY_PMAapkrP54tYX8-K6iGmOmrR7BbQ-3Qhn3w4s,9885
157
+ verifiers/v1/serve/server.py,sha256=kaQFh_tVRcmRaQoB5o6CU4y8Fh9X-zFg-cmcvU2GExg,9913
158
158
  verifiers/v1/serve/types.py,sha256=Qw5IjiJhsZz4giSXUBi9dfInBnQZ5_xiUmgH2qp8jPI,1741
159
159
  verifiers/v1/tasksets/__init__.py,sha256=2ijE3qIlLoUA02g6iysOwWQylyBBQL1GhSpjv8_bp0w,521
160
160
  verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
@@ -189,9 +189,10 @@ verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,9
189
189
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
190
190
  verifiers/v1/utils/scope.py,sha256=bzUEiWDOVdbj7dXOwGOlzsQ_-Xu0j9NR5TalQB-8TS8,838
191
191
  verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
192
+ verifiers/v1/utils/trace_store.py,sha256=RsCDonGR-bs-tvEstcppiJm3jFmAdIiG83fA5KtCbAw,2801
192
193
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
193
- verifiers-0.3.2.dev136.dist-info/METADATA,sha256=Dody0KdHsobWBNx2Mc1KbvuGAH152EZaPxEonMGcoJ4,4161
194
- verifiers-0.3.2.dev136.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
195
- verifiers-0.3.2.dev136.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
196
- verifiers-0.3.2.dev136.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
197
- verifiers-0.3.2.dev136.dist-info/RECORD,,
194
+ verifiers-0.3.2.dev138.dist-info/METADATA,sha256=5cCtrF7ceNh2ee4zKBCZH_VkAuhIRxR2TOO6Nev91jM,4161
195
+ verifiers-0.3.2.dev138.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
196
+ verifiers-0.3.2.dev138.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
197
+ verifiers-0.3.2.dev138.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
198
+ verifiers-0.3.2.dev138.dist-info/RECORD,,
@@ -1,4 +1,4 @@
1
1
  Wheel-Version: 1.0
2
- Generator: hatchling 1.32.0
2
+ Generator: hatchling 1.32.3
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any