verifiers 0.3.2.dev127__py3-none-any.whl → 0.3.2.dev129__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
verifiers/v1/cli/debug.py CHANGED
@@ -3,10 +3,12 @@
3
3
  import asyncio
4
4
  import contextlib
5
5
  import logging
6
+ import os
6
7
  import shlex
7
8
  import sys
8
9
  import time
9
10
  import traceback
11
+ import uuid
10
12
  from collections.abc import Awaitable
11
13
  from pathlib import Path
12
14
  from typing import Any
@@ -316,6 +318,8 @@ async def run_debug(config: DebugConfig) -> list[Trace]:
316
318
 
317
319
 
318
320
  def main(argv: list[str] | None = None) -> None:
321
+ # The run identity: every process this run spawns inherits it.
322
+ os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
319
323
  argv = with_positional_taskset(
320
324
  list(sys.argv[1:]) if argv is None else list(argv), flag="--taskset.id"
321
325
  )
@@ -9,6 +9,7 @@ runs env servers); this CLI is the quick local path.
9
9
  import asyncio
10
10
  import contextlib
11
11
  import logging
12
+ import os
12
13
  import time
13
14
  from collections.abc import AsyncIterator, Awaitable, Callable, Iterable
14
15
  from typing import TypeVar, cast
@@ -139,6 +140,8 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
139
140
 
140
141
  # Opened before the first rollout so every episode streams as it lands.
141
142
  run = open_run(config, push_state, num_examples=len(tasks))
143
+ # The run identity: every process this run spawns inherits it.
144
+ os.environ.setdefault("VF_RUN_ID", config.run.id)
142
145
  # Resumed rollouts are part of this run too.
143
146
  log_episodes(run, finished)
144
147
 
verifiers/v1/cli/gepa.py CHANGED
@@ -9,7 +9,9 @@ and the actual parse is `pydantic_config.cli`.
9
9
  """
10
10
 
11
11
  import logging
12
+ import os
12
13
  import sys
14
+ import uuid
13
15
 
14
16
  from pydantic_config import cli
15
17
 
@@ -32,6 +34,8 @@ USAGE = "usage: uv run vf-gepa [<taskset-id>] [--env.id <id>] --model <model> [o
32
34
 
33
35
 
34
36
  def main(argv: list[str] | None = None) -> None:
37
+ # The run identity: every process this run spawns inherits it.
38
+ os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
35
39
  argv = with_positional_taskset(list(sys.argv[1:]) if argv is None else list(argv))
36
40
 
37
41
  if not argv or any(arg in ("-h", "--help") for arg in argv):
@@ -12,8 +12,10 @@ import asyncio
12
12
  import contextlib
13
13
  import json
14
14
  import logging
15
+ import os
15
16
  import sys
16
17
  import time
18
+ import uuid
17
19
  from pathlib import Path
18
20
 
19
21
  from pydantic_config import cli
@@ -197,6 +199,8 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
197
199
 
198
200
 
199
201
  def main(argv: list[str] | None = None) -> None:
202
+ # The run identity: every process this run spawns inherits it.
203
+ os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
200
204
  argv = list(sys.argv[1:]) if argv is None else list(argv)
201
205
  if not argv or any(a in ("-h", "--help") for a in argv):
202
206
  print(USAGE)
@@ -4,9 +4,11 @@ import asyncio
4
4
  import contextlib
5
5
  import json
6
6
  import logging
7
+ import os
7
8
  import shutil
8
9
  import sys
9
10
  import time
11
+ import uuid
10
12
  from collections import Counter, defaultdict
11
13
  from collections.abc import Mapping, Sequence
12
14
  from pathlib import Path
@@ -425,6 +427,8 @@ async def run_validate(config: ValidateConfig) -> list[dict]:
425
427
 
426
428
 
427
429
  def main(argv: list[str] | None = None) -> None:
430
+ # The run identity: every process this run spawns inherits it.
431
+ os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
428
432
  argv = with_positional_taskset(
429
433
  list(sys.argv[1:]) if argv is None else list(argv), flag="--taskset.id"
430
434
  )
@@ -40,5 +40,5 @@ def resolve_gepa_seed_prompt(tasks: list[Task], initial_prompt: str | None) -> s
40
40
  "no task in this taskset sets Task.system_prompt — some tasksets bake instructions "
41
41
  "directly into `prompt` instead (e.g. gsm8k) and can't be optimized this way. Pass "
42
42
  "--initial-prompt to seed one explicitly, or pick a taskset whose load() sets "
43
- "system_prompt on its task data (e.g. reverse-text, lean, textarena)."
43
+ "system_prompt on its task data (e.g. reverse-text, textarena)."
44
44
  )
@@ -9,15 +9,19 @@ from collections.abc import AsyncIterator
9
9
  from typing import Literal
10
10
 
11
11
  from verifiers.v1.interception.tunnel.base import BaseTunnelConfig, Tunnel
12
- from verifiers.v1.runtimes.limiters import creation_limiter
12
+ from verifiers.v1.runtimes.limiters import CreationLimiter
13
13
  from verifiers.v1.utils.aio import run_shielded
14
14
  from verifiers.v1.utils.prime import ensure_prime_auth
15
+ from verifiers.v1.utils.scope import run_scope
15
16
 
16
17
  # The prime_tunnel service caps tunnel starts at 512/min per API token — a property of the
17
- # tunnel service, shared by every process for the user that opens one. One user-global
18
+ # tunnel service, shared by every process of a run that opens one. One run-scoped
18
19
  # limiter, not a per-runtime config knob.
19
20
  _TUNNELS_PER_MIN = 512
20
- TUNNEL_LIMITER = creation_limiter(_TUNNELS_PER_MIN / 60, "prime-tunnel")
21
+
22
+
23
+ def tunnel_limiter() -> CreationLimiter:
24
+ return CreationLimiter("prime-tunnel", run_scope(), _TUNNELS_PER_MIN / 60)
21
25
 
22
26
 
23
27
  class PrimeTunnelConfig(BaseTunnelConfig):
@@ -35,8 +39,8 @@ class PrimeTunnel(Tunnel[PrimeTunnelConfig]):
35
39
  @contextlib.asynccontextmanager
36
40
  async def expose(self, port: int) -> AsyncIterator[str]:
37
41
  """Bridge the host `port` to a public URL via prime_tunnel (frpc). Tunnel creation
38
- is network-bound and globally rate-capped (512/min, user-wide via the shared
39
- `TUNNEL_LIMITER`), so transient failures are retried; a terminal one raises
42
+ is network-bound and rate-capped (512/min, run-wide via the shared
43
+ `tunnel_limiter`), so transient failures are retried; a terminal one raises
40
44
  `TunnelError`. The tunnel is torn down on exit."""
41
45
  from prime_tunnel import Tunnel as TunnelClient
42
46
 
@@ -48,7 +52,7 @@ class PrimeTunnel(Tunnel[PrimeTunnelConfig]):
48
52
  async for attempt in retrying(retries=3, label=label):
49
53
  with attempt:
50
54
  client = TunnelClient(local_port=port)
51
- async with TUNNEL_LIMITER:
55
+ async with tunnel_limiter():
52
56
  url = str(await client.start()).rstrip("/")
53
57
  except Exception as e:
54
58
  raise TunnelError(f"{label} failed: {e}") from e
@@ -1,11 +1,12 @@
1
- """User-global creation-rate limiters for the remote runtimes.
1
+ """Creation-rate limiters for the remote runtimes.
2
2
 
3
3
  A leaky bucket backed by a lock file under the user cache (``~/.cache/verifiers``, falling
4
- back to the temp dir when no home is resolvable), so a provider's per-account creation rate
5
- (Modal sandboxes, Prime tunnels) is enforced across EVERY process for the user —
6
- the single-process eval and all the elastically-spawned env-server worker processes alike — not
7
- just within one process. Keyed by name: one bucket file per name, shared by every process (and
8
- run) for the user.
4
+ back to the temp dir when no home is resolvable), so a provider's creation rate (Modal
5
+ sandboxes, Prime sandboxes and tunnels) is enforced across EVERY process that shares the
6
+ bucket — the eval process and all the elastically-spawned env-server worker processes alike —
7
+ not just within one process. A bucket is named by the limiter's name and a scope the caller
8
+ picks; the runtimes pass the run id, so one run's backlog (or the reservations a killed run
9
+ left behind) never delays another run.
9
10
  """
10
11
 
11
12
  import asyncio
@@ -22,17 +23,18 @@ LIMITER_DIR = CACHE_DIR / "limiter"
22
23
  class CreationLimiter:
23
24
  """An async leaky bucket shared across processes via a lock file: each `async with`
24
25
  reserves the next `1/per_sec`-spaced slot (advancing the on-disk cursor under an exclusive
25
- flock) and sleeps until it, so the aggregate creation rate across all of the user's
26
- processes stays at `per_sec`. The reservation runs off the event loop; the wait does not
27
- hold the lock. Backlogs over five minutes fail rather than silently stalling creation."""
26
+ flock) and sleeps until it, so the aggregate creation rate across every process sharing
27
+ the bucket stays at `per_sec`. The reservation runs off the event loop; the wait does not
28
+ hold the lock. Reservations are never released, so a cancelled waiter still holds its
29
+ slot; the backlog drains at `per_sec` regardless."""
28
30
 
29
- def __init__(self, name: str, per_sec: float) -> None:
31
+ def __init__(self, name: str, scope: str, per_sec: float) -> None:
30
32
  self._interval = 1 / per_sec
31
- self._path = LIMITER_DIR / f"{name}.bucket"
33
+ self._path = LIMITER_DIR / f"{name}-{scope.replace('/', '--')}.bucket"
32
34
 
33
35
  def _reserve(self) -> float:
34
36
  os.makedirs(LIMITER_DIR, exist_ok=True)
35
- # Shared buckets require a clock comparable across hosts and boots.
37
+ # Shared buckets require a clock comparable across the run's hosts.
36
38
  with open(self._path, "a+") as f:
37
39
  fcntl.flock(f.fileno(), fcntl.LOCK_EX)
38
40
  try:
@@ -40,19 +42,13 @@ class CreationLimiter:
40
42
  data = f.read().strip()
41
43
  now = time.time()
42
44
  slot = max(now, float(data) if data else 0.0)
43
- wait = slot - now
44
- if wait > 5 * 60:
45
- raise TimeoutError(
46
- f"{self._path.stem} creation limiter backlog of {wait:.1f}s "
47
- f"exceeds 300s ({self._path})"
48
- )
49
45
  f.seek(0)
50
46
  f.truncate()
51
47
  f.write(repr(slot + self._interval))
52
48
  f.flush()
53
- return wait
54
49
  finally:
55
50
  fcntl.flock(f.fileno(), fcntl.LOCK_UN)
51
+ return slot - now
56
52
 
57
53
  async def __aenter__(self) -> Self:
58
54
  wait = await asyncio.to_thread(self._reserve)
@@ -64,16 +60,12 @@ class CreationLimiter:
64
60
  return False
65
61
 
66
62
 
67
- _creation_limiters: dict[str, CreationLimiter] = {}
68
-
69
-
70
- def creation_limiter(per_sec: float | None, name: str) -> CreationLimiter | None:
71
- """A user-global limiter pacing `name`'s creation to `per_sec`/s (None/<= 0 disables).
72
-
73
- All callers (and processes) sharing a `name` share one bucket, so use one rate per name."""
63
+ def creation_limiter(
64
+ per_sec: float | None, name: str, scope: str
65
+ ) -> CreationLimiter | None:
66
+ """A limiter pacing `name`'s creation to `per_sec`/s within `scope` (None/<= 0
67
+ disables). All callers (and processes) sharing a name and scope share one bucket, so
68
+ use one rate per name."""
74
69
  if not per_sec or per_sec <= 0:
75
70
  return None
76
- limiter = _creation_limiters.get(name)
77
- if limiter is None:
78
- limiter = _creation_limiters[name] = CreationLimiter(name, per_sec)
79
- return limiter
71
+ return CreationLimiter(name, scope, per_sec)
@@ -31,6 +31,7 @@ from verifiers.v1.runtimes.base import (
31
31
  )
32
32
  from verifiers.v1.runtimes.limiters import creation_limiter
33
33
  from verifiers.v1.utils.aio import run_shielded
34
+ from verifiers.v1.utils.scope import run_scope
34
35
 
35
36
  logger = logging.getLogger(__name__)
36
37
 
@@ -93,7 +94,7 @@ class ModalConfig(NetworkPolicyConfig):
93
94
  """Disk in GB. Modal sandboxes have no disk knob, so this is accepted (so a task can
94
95
  declare it without a warning) but not enforced."""
95
96
  creates_per_sec: float | None = 40.0
96
- """Pace sandbox creation to this many per second, enforced user-wide across every
97
+ """Pace sandbox creation to this many per second, enforced run-wide across every
97
98
  env-server worker process (None/<= 0 disables it)."""
98
99
 
99
100
  @model_validator(mode="after")
@@ -181,7 +182,9 @@ class ModalRuntime(Runtime):
181
182
  try:
182
183
  app = await modal.App.lookup.aio(_APP_NAME, create_if_missing=True)
183
184
  async with (
184
- creation_limiter(self.config.creates_per_sec, "modal-sandbox")
185
+ creation_limiter(
186
+ self.config.creates_per_sec, "modal-sandbox", run_scope()
187
+ )
185
188
  or contextlib.nullcontext()
186
189
  ):
187
190
  await run_shielded(self._create_sandbox(app))
@@ -28,6 +28,7 @@ from verifiers.v1.runtimes.base import (
28
28
  from verifiers.v1.runtimes.limiters import creation_limiter
29
29
  from verifiers.v1.utils.aio import run_shielded
30
30
  from verifiers.v1.utils.prime import ensure_prime_auth
31
+ from verifiers.v1.utils.scope import run_scope
31
32
 
32
33
  logger = logging.getLogger(__name__)
33
34
 
@@ -87,9 +88,9 @@ class PrimeConfig(NetworkPolicyConfig):
87
88
  idle_timeout: float | None = 3600
88
89
  """Seconds of inactivity before the sandbox self-deletes (None disables)."""
89
90
  creates_per_min: int | None = None
90
- """Pace sandbox creation to this many per minute, enforced user-wide across every
91
+ """Pace sandbox creation to this many per minute, enforced run-wide across every
91
92
  env-server worker process (None/<= 0 disables it). (Tunnel creation is limited separately
92
- and globally — see interception.tunnel.prime.TUNNEL_LIMITER.)"""
93
+ — see interception.tunnel.prime.tunnel_limiter.)"""
93
94
 
94
95
  @model_validator(mode="after")
95
96
  def _validate_egress(self) -> "PrimeConfig":
@@ -175,7 +176,9 @@ class PrimeRuntime(Runtime):
175
176
  try:
176
177
  async with (
177
178
  creation_limiter(
178
- (self.config.creates_per_min or 0) / 60, "prime-sandbox"
179
+ (self.config.creates_per_min or 0) / 60,
180
+ "prime-sandbox",
181
+ run_scope(),
179
182
  )
180
183
  or contextlib.nullcontext()
181
184
  ):
@@ -1,10 +1,4 @@
1
1
  from verifiers.v1.tasksets.harbor import HarborConfig, HarborTaskset
2
- from verifiers.v1.tasksets.lean import (
3
- LeanConfig,
4
- LeanDatasetConfig,
5
- LeanTask,
6
- LeanTaskset,
7
- )
8
2
  from verifiers.v1.tasksets.nemo_gym import NeMoGymConfig, NeMoGymTaskset
9
3
  from verifiers.v1.tasksets.openenv import (
10
4
  OpenEnvConfig,
@@ -18,10 +12,6 @@ from verifiers.v1.tasksets.openenv import (
18
12
  __all__ = [
19
13
  "HarborConfig",
20
14
  "HarborTaskset",
21
- "LeanConfig",
22
- "LeanDatasetConfig",
23
- "LeanTask",
24
- "LeanTaskset",
25
15
  "NeMoGymConfig",
26
16
  "NeMoGymTaskset",
27
17
  "OpenEnvConfig",
@@ -0,0 +1,27 @@
1
+ """The scope that state shared by a run's processes is keyed by."""
2
+
3
+ import logging
4
+ import os
5
+ import uuid
6
+
7
+ logger = logging.getLogger(__name__)
8
+
9
+ _process_scope: str | None = None
10
+
11
+
12
+ def run_scope() -> str:
13
+ """``$VF_RUN_ID``, set by every entrypoint (the `vf-*` CLIs and the prime-rl launchers)
14
+ and inherited by spawned env servers and pool workers. A process started without one
15
+ gets a scope of its own, minted once per process."""
16
+ global _process_scope
17
+ run_id = os.environ.get("VF_RUN_ID")
18
+ if run_id:
19
+ return run_id
20
+ if _process_scope is None:
21
+ _process_scope = uuid.uuid4().hex
22
+ logger.warning(
23
+ "VF_RUN_ID is unset - run-scoped state such as creation limiters covers only "
24
+ "this process (scope %s)",
25
+ _process_scope,
26
+ )
27
+ return _process_scope
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: verifiers
3
- Version: 0.3.2.dev127
3
+ Version: 0.3.2.dev129
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -48,8 +48,6 @@ Requires-Dist: typing-extensions>=4.12.2
48
48
  Requires-Dist: uvicorn>=0.52.0
49
49
  Provides-Extra: harbor
50
50
  Requires-Dist: harbor==0.21.0; (python_full_version >= '3.12') and extra == 'harbor'
51
- Provides-Extra: lean
52
- Requires-Dist: datasets<6.0.0,>=3.3.0; extra == 'lean'
53
51
  Provides-Extra: modal
54
52
  Requires-Dist: modal>=1.5.4; extra == 'modal'
55
53
  Provides-Extra: openenv
@@ -18,14 +18,14 @@ verifiers/v1/types.py,sha256=xPetvJxksKPYKic389RFnVvy2vN-gD2bPZFZoEyzmp4,10191
18
18
  verifiers/v1/acp/__init__.py,sha256=9RySmxFeEMT2XSJy5wYevqEhQdul2jHl9f5XAribG3A,13631
19
19
  verifiers/v1/acp/runner.py,sha256=zPo-2ZXmFMmQchhD7nNzlZUrabG09qiBvpB67KVGm3M,14095
20
20
  verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
21
- verifiers/v1/cli/debug.py,sha256=3hPBoP8iceAUyx3lAjZ0z-QhiC5_z-jaRJfCTdxUFrk,11634
22
- verifiers/v1/cli/gepa.py,sha256=3-_GckCUttJRgKlsB9q_Ikhf4DtmTOIhyXOo6gNUiWc,4186
21
+ verifiers/v1/cli/debug.py,sha256=sgXi2y17qr2yd5kbkL_IzwWN1nIFBOa1adu7bcFzaVk,11780
22
+ verifiers/v1/cli/gepa.py,sha256=F0Q1ApfUL_mInFnmaJ5-yBDRYBtBuH72OkWelPoGCp4,4332
23
23
  verifiers/v1/cli/init.py,sha256=6wr14011_Dygv10R6_YWg5A3oSrz187E2WspGRHIfZw,7608
24
24
  verifiers/v1/cli/output.py,sha256=lAl6vwKf9mVuNZb8PzOexv6Q2h-H6t0oUPcapffY3uY,8351
25
- verifiers/v1/cli/replay.py,sha256=8EMROnzIWJ_2Mxs_eqZ19yProml9PZnxaIxHkqSwxXE,10249
25
+ verifiers/v1/cli/replay.py,sha256=JjhaGczQ41zN6wz1P1nbGGvv0xdfQlX4OmZItGPDpZc,10395
26
26
  verifiers/v1/cli/resolve.py,sha256=Q1EHM7wWQo0YkwJA898qtZbYLaIkjIVq3Q7kUp2YIxs,4346
27
27
  verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,470
28
- verifiers/v1/cli/validate.py,sha256=K7mCz-ZI7sq69MQPg6QvreHOuEj-bgMln69a28kDTvA,17179
28
+ verifiers/v1/cli/validate.py,sha256=Z_rUldl8jzXz52oy3FBIRGr4HYhWFXrq43WmU0sJ-4M,17325
29
29
  verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
30
30
  verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
31
31
  verifiers/v1/cli/dashboard/eval.py,sha256=HEXPEegNXdgU9zAIuHzsWsoSIHyiOcDZh4tCGFNxKAk,38353
@@ -35,7 +35,7 @@ verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3h
35
35
  verifiers/v1/cli/eval/hint.py,sha256=NXP2_JYrQHguqr68LxOJG3lYAsue_VNoepyTVPn-N4E,256
36
36
  verifiers/v1/cli/eval/main.py,sha256=L_vKnMJEbpo1IKjbm0TXjXv_1x9we7mFv-jH_xY211Q,5946
37
37
  verifiers/v1/cli/eval/resume.py,sha256=fYkWa0Fudq--IWfk25j_McLm67bXYWvksWmJR4u4QAU,3811
38
- verifiers/v1/cli/eval/runner.py,sha256=gvwR3O6FM-yZMvS5uMgST3NGv5Egi7AG4cfYhP7bRC0,6831
38
+ verifiers/v1/cli/eval/runner.py,sha256=z1yU2_rAA6tHz3PL7NQuzm9v3-4Ri18DXD5SRBKXqVY,6962
39
39
  verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
40
40
  verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
41
41
  verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
@@ -79,7 +79,7 @@ verifiers/v1/envs/user_sim/env.py,sha256=deI4r-Utn_v4E_Vb9P_X3LzFh2GRkCO4OlQt8BO
79
79
  verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
80
80
  verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,6139
81
81
  verifiers/v1/gepa/config.py,sha256=PhZRLkhaPciNm_IhwVX3e9DM8Y2IdCe4Fbi5WjH5aro,4883
82
- verifiers/v1/gepa/dataset.py,sha256=hFuR7dou4VSKa0ujb8sFmGm9a19mjrsmTa1MMW1Kv7I,2177
82
+ verifiers/v1/gepa/dataset.py,sha256=SPGG55BVbJ9HtCEvm-xyl52QSbldFwGqWvRIcaLelsM,2171
83
83
  verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
84
84
  verifiers/v1/gepa/runner.py,sha256=fGEyFzejdQKBzBbvdiTDHV3JtRhXMXMGSBU5y8C8i3w,6054
85
85
  verifiers/v1/harnesses/__init__.py,sha256=2JPwrRoYoigH-HfBtiFnatRdFEVHLv4rwlKfsFEibyE,1803
@@ -129,7 +129,7 @@ verifiers/v1/interception/server.py,sha256=wxw67OQkLgzvt1HMJyjYmdcoIhn4vuJBe5oXZ
129
129
  verifiers/v1/interception/tunnel/__init__.py,sha256=eVNZJszj6myrKhx9jmJz7VRd9qlfj0AUvUZwI2bXEj8,893
130
130
  verifiers/v1/interception/tunnel/base.py,sha256=DZB6uPLwM4Qy7n7m0vKeyg-Px87MhsmpYTpe2vpPEng,1998
131
131
  verifiers/v1/interception/tunnel/custom.py,sha256=yL4UbGf4bAuMsxk42mEP-t1wITQCT8BE9FHQIz0qRMA,1708
132
- verifiers/v1/interception/tunnel/prime.py,sha256=fIhU4Sa53f2FQWrwIgpDjh8XO9E9RdF9k1kh0TwDLTo,2721
132
+ verifiers/v1/interception/tunnel/prime.py,sha256=jVk_6xyrxXiHtsVFErcRNd4MoENdFbqzfC0i20nlizU,2803
133
133
  verifiers/v1/judges/__init__.py,sha256=MUIBWcx6c70BykrTDHhB0ITIyJPxXe6xDBou4VM7jDQ,286
134
134
  verifiers/v1/judges/reference.py,sha256=iEVHw-iJ6dbwLOOqrvWrEMJOcUEYJ0YspT04rZz7Sd4,3868
135
135
  verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93V6A8,353
@@ -143,9 +143,9 @@ verifiers/v1/runtimes/__init__.py,sha256=JQlQ029J14asfYb-UxpENshMyPztWbSL0-ZqSyD
143
143
  verifiers/v1/runtimes/apptainer.py,sha256=Ru9QJna_l0goIzg71G-4RU_D-7wZUNz139S5TEvbzog,8211
144
144
  verifiers/v1/runtimes/base.py,sha256=IKmrJSJltWX9M1nPUZExKluPJM4N82VsnqQGFz0ECV8,17777
145
145
  verifiers/v1/runtimes/container.py,sha256=puQD8X9acP6GefoCNEnq40POi-nXPD-729bUjTVShfM,9554
146
- verifiers/v1/runtimes/limiters.py,sha256=C6cWDD4pwyFwUY4QFke3oiMXwydSNIfSbJSIglUC2kE,3062
147
- verifiers/v1/runtimes/modal.py,sha256=M1zBIMI6zJAq6io_t8uKGtXIxSZeEuAMuG3nR4L5pAU,17379
148
- verifiers/v1/runtimes/prime.py,sha256=BXwDK-eurEzOKLY1ezuC0UYZdEoI0Y9Cfcd0IhKjMPc,17586
146
+ verifiers/v1/runtimes/limiters.py,sha256=rOfQJcMdYcLSKq8AL9O09BfGOLqpYmNgrCxXXNcHLmU,2849
147
+ verifiers/v1/runtimes/modal.py,sha256=6ccitC0HVsAP_l8soJ_RQYNIVKrJjDsz2vAJBtxUZjw,17476
148
+ verifiers/v1/runtimes/prime.py,sha256=nTtu-uBjc39UxfMPEyndshfAiIdk1V4L4aE6zSVVSn0,17673
149
149
  verifiers/v1/runtimes/subprocess.py,sha256=QmpQ235_xrX6gowrr8557VgbydOXpLbfFq0KKGfJThk,8774
150
150
  verifiers/v1/runtimes/docker/__init__.py,sha256=2wceP-8V-w0dQBEeLtoJFfCZ135DxP2rcdcUBmSWiNU,18543
151
151
  verifiers/v1/runtimes/docker/egress.py,sha256=wtKWTL2_zFidxMR7lDRmfevQlOsl2Iyk88ebdO4Kqqw,24566
@@ -156,14 +156,11 @@ verifiers/v1/serve/encoding.py,sha256=hBZFucAZK9riXOV3DaHcskq9zHTT8gVGkoFVOPfnr3
156
156
  verifiers/v1/serve/pool.py,sha256=bT-FOlIuOiBtOasuPB-p9887tkpEqvQUJBgn1Q45ph0,15912
157
157
  verifiers/v1/serve/server.py,sha256=i6XrY_PMAapkrP54tYX8-K6iGmOmrR7BbQ-3Qhn3w4s,9885
158
158
  verifiers/v1/serve/types.py,sha256=Qw5IjiJhsZz4giSXUBi9dfInBnQZ5_xiUmgH2qp8jPI,1741
159
- verifiers/v1/tasksets/__init__.py,sha256=5hTKSPLXSZTgKBbeKQI7FEp06c0ZsaQfuwvrvVFLgb8,712
159
+ verifiers/v1/tasksets/__init__.py,sha256=2ijE3qIlLoUA02g6iysOwWQylyBBQL1GhSpjv8_bp0w,521
160
160
  verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
161
161
  verifiers/v1/tasksets/harbor/env.py,sha256=ATMkfwmrE_2jpOOffbjLQ91feG7TOtefBGlpwWbRq8c,3604
162
162
  verifiers/v1/tasksets/harbor/taskset.py,sha256=35ljhNapn1EOlaYFTuVZxAMVU0KDfmFxQOGOcUk2Ow0,34082
163
163
  verifiers/v1/tasksets/harbor/toolset.py,sha256=_7HXu-uuvqzhmJ1EX6c3KWLg0g0eRrLjCq1kBsZhIwo,3159
164
- verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
165
- verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
166
- verifiers/v1/tasksets/lean/taskset.py,sha256=Jax76i22G9dOAswQHiWujVbFIM_PUz5yGJI_ShR4n8Y,9238
167
164
  verifiers/v1/tasksets/nemo_gym/__init__.py,sha256=654gAvG3Wd_Z_nh6mypeKn4s_IIQ0Ge6vcZvxTHC2_E,306
168
165
  verifiers/v1/tasksets/nemo_gym/response.py,sha256=EhrsquEP8g0bjzAThGfE6Ypgy86_nVX235NiFYtVDVM,4463
169
166
  verifiers/v1/tasksets/nemo_gym/server.py,sha256=33C-VDZxPtMV2TFxJVIKijmtBLNlVeRs92ctPE3qAsM,2036
@@ -190,10 +187,11 @@ verifiers/v1/utils/paths.py,sha256=it_4JWf_8bPZe4TRcyvL7_KJ4fsbEsxhhIY_xA8Ikus,3
190
187
  verifiers/v1/utils/platform.py,sha256=YwTHVA1Onn74lG6S3sdHus9OlP-rAbwkk9Po7h-BjpE,7159
191
188
  verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,970
192
189
  verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
190
+ verifiers/v1/utils/scope.py,sha256=bzUEiWDOVdbj7dXOwGOlzsQ_-Xu0j9NR5TalQB-8TS8,838
193
191
  verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
194
192
  verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
195
- verifiers-0.3.2.dev127.dist-info/METADATA,sha256=F8kPNYPLJWPcdQ0mwZw2mQ7hKpVaOQ9ap_ASg6SmLyo,4237
196
- verifiers-0.3.2.dev127.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
197
- verifiers-0.3.2.dev127.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
198
- verifiers-0.3.2.dev127.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
199
- verifiers-0.3.2.dev127.dist-info/RECORD,,
193
+ verifiers-0.3.2.dev129.dist-info/METADATA,sha256=ckfiJIStyeQf1e_fIyVWL-7jtc5yiEqtyEWdvxHCNKQ,4161
194
+ verifiers-0.3.2.dev129.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
195
+ verifiers-0.3.2.dev129.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
196
+ verifiers-0.3.2.dev129.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
197
+ verifiers-0.3.2.dev129.dist-info/RECORD,,
@@ -1,33 +0,0 @@
1
- from verifiers.v1.tasksets.lean.scoring import (
2
- build_starter_file,
3
- expected_protected_signature,
4
- parse_compile_output,
5
- protected_signature_substring_present,
6
- strip_lean_comments,
7
- )
8
- from verifiers.v1.tasksets.lean.taskset import (
9
- DEFAULT_DOCKER_IMAGE,
10
- LEAN_PROJECT_PATH,
11
- PROOF_FILE_PATH,
12
- LeanConfig,
13
- LeanDatasetConfig,
14
- LeanTask,
15
- LeanTaskConfig,
16
- LeanTaskset,
17
- )
18
-
19
- __all__ = [
20
- "DEFAULT_DOCKER_IMAGE",
21
- "LEAN_PROJECT_PATH",
22
- "PROOF_FILE_PATH",
23
- "LeanConfig",
24
- "LeanDatasetConfig",
25
- "LeanTask",
26
- "LeanTaskConfig",
27
- "LeanTaskset",
28
- "build_starter_file",
29
- "expected_protected_signature",
30
- "parse_compile_output",
31
- "protected_signature_substring_present",
32
- "strip_lean_comments",
33
- ]
@@ -1,262 +0,0 @@
1
- """Pure helpers for the Lean taskset: starter-file construction, theorem-signature
2
- canonicalization, the reward-hacking signature guard, and compile-output parsing.
3
-
4
- Everything here is a pure function over plain strings — no runtime, no sandbox, no
5
- verifiers types — so it's trivially unit-testable and shared between ``setup``
6
- (building the starter file), the reward (the signature guard + compile parsing),
7
- and ``validate`` (planting the gold proof).
8
- """
9
-
10
- from __future__ import annotations
11
-
12
- import re
13
-
14
- PROTECTED_HEADER_COMMENT = (
15
- "-- DO NOT MODIFY the theorem statement below. The grader checks\n"
16
- "-- that the original `theorem ... := by` text still appears in\n"
17
- "-- this file. Only edit the proof body (currently `sorry`) and\n"
18
- "-- lines after it."
19
- )
20
-
21
-
22
- # ── Imports / preamble normalization ─────────────────────────────────────────
23
-
24
-
25
- def normalize_imports(text: str) -> str:
26
- """Collapse every ``import Mathlib*`` line to a single ``import Mathlib``.
27
-
28
- Some datasets (miniF2F) ship fine-grained Mathlib imports that don't resolve
29
- against the monolithic ``import Mathlib`` the sandbox image provides.
30
- """
31
- lines = text.split("\n")
32
- out: list[str] = []
33
- inserted = False
34
- for line in lines:
35
- if line.strip().startswith("import Mathlib"):
36
- if not inserted:
37
- out.append("import Mathlib")
38
- inserted = True
39
- else:
40
- out.append(line)
41
- return "\n".join(out)
42
-
43
-
44
- def _build_preamble(imports_str: str, header: str, normalize: bool) -> str:
45
- if header and header.strip().startswith("import"):
46
- preamble = header.strip()
47
- return normalize_imports(preamble) if normalize else preamble
48
- parts = [imports_str.strip()]
49
- if header and header.strip():
50
- parts.append(header.strip())
51
- preamble = "\n\n".join(parts)
52
- return normalize_imports(preamble) if normalize else preamble
53
-
54
-
55
- # ── Theorem signature canonicalization ───────────────────────────────────────
56
-
57
-
58
- def _normalize_signature(stmt: str) -> str:
59
- """Canonicalize a Lean theorem statement to end with ``:= by``.
60
-
61
- Strips trailing ``sorry``/``admit`` placeholders and any trailing ``by`` /
62
- ``:=`` tokens, then re-appends `` := by``. Places the appended token on a new
63
- indented line when the last line of the stripped signature already contains a
64
- ``--`` comment (otherwise the ``:= by`` would land inside the line comment and
65
- Lean would silently ignore it).
66
- """
67
- s = stmt.rstrip()
68
- s = re.sub(r"\s*\b(?:sorry|admit)\b\s*$", "", s)
69
- s = re.sub(r"\s*\bby\b\s*$", "", s)
70
- s = re.sub(r"\s*:=\s*$", "", s)
71
- s = s.rstrip()
72
- last_newline = s.rfind("\n")
73
- last_line = s[last_newline + 1 :] if last_newline != -1 else s
74
- sep = "\n " if "--" in last_line else " "
75
- return s + sep + ":= by"
76
-
77
-
78
- def _split_imports_and_signature(stmt: str) -> tuple[str, str]:
79
- decl_match = re.search(r"^(?:theorem|lemma|example)\s", stmt, flags=re.MULTILINE)
80
- if not decl_match:
81
- return "", stmt
82
- return stmt[: decl_match.start()].rstrip(), stmt[decl_match.start() :]
83
-
84
-
85
- def expected_protected_signature(formal_statement: str) -> str:
86
- """Return the canonical ``theorem ... := by`` block the reward pins as ground truth.
87
-
88
- Pure function over ``formal_statement``; the imports/header portion (if the
89
- statement carries one inline) is stripped first.
90
- """
91
- stmt = formal_statement or ""
92
- if stmt.strip().startswith("import "):
93
- _, signature_raw = _split_imports_and_signature(stmt)
94
- else:
95
- signature_raw = stmt
96
- return _normalize_signature(signature_raw).strip()
97
-
98
-
99
- # ── Comment / string stripping for the signature guard ───────────────────────
100
-
101
-
102
- def strip_lean_comments(text: str) -> str:
103
- """Remove Lean line/block comments **and string literals** from ``text``.
104
-
105
- Lean comments come in two forms: ``-- ...`` line comments (to end of line) and
106
- ``/- ... -/`` block comments (nestable; ``/-- ... -/`` doc comments are a
107
- special case). String literals must also be stripped: a Lean
108
- ``"theorem ... := by"`` constant or doc-string would otherwise let a model hide
109
- the pinned signature inside a string while rewriting the live declaration to a
110
- trivial one, defeating the substring guard. We handle both regular
111
- double-quoted strings (with backslash escapes) and triple-quoted raw strings.
112
-
113
- Block comments nest, so we count depth; line comments end at the next newline.
114
- Outside comments and strings, newlines are preserved so the result keeps
115
- roughly the right shape for substring matching downstream.
116
- """
117
- out: list[str] = []
118
- i = 0
119
- n = len(text)
120
- block_depth = 0
121
- in_line_comment = False
122
- while i < n:
123
- ch = text[i]
124
- if in_line_comment:
125
- if ch == "\n":
126
- in_line_comment = False
127
- out.append(ch)
128
- i += 1
129
- continue
130
- if block_depth > 0:
131
- if i + 1 < n and text[i : i + 2] == "-/":
132
- block_depth -= 1
133
- i += 2
134
- continue
135
- if i + 1 < n and text[i : i + 2] == "/-":
136
- block_depth += 1
137
- i += 2
138
- continue
139
- if ch == "\n":
140
- out.append(ch)
141
- i += 1
142
- continue
143
- if i + 1 < n and text[i : i + 2] == "/-":
144
- block_depth = 1
145
- i += 2
146
- continue
147
- if i + 1 < n and text[i : i + 2] == "--":
148
- in_line_comment = True
149
- i += 2
150
- continue
151
- # Triple-quoted raw string ``"""..."""`` — skip until the closing
152
- # triple-quote, preserving newlines.
153
- if i + 2 < n and text[i : i + 3] == '"""':
154
- i += 3
155
- while i < n:
156
- if i + 2 < n and text[i : i + 3] == '"""':
157
- i += 3
158
- break
159
- if text[i] == "\n":
160
- out.append("\n")
161
- i += 1
162
- continue
163
- # Regular string ``"..."`` — handle ``\\`` escapes, stop at the next
164
- # unescaped quote or newline (Lean strings are single-line).
165
- if ch == '"':
166
- i += 1
167
- while i < n and text[i] != '"' and text[i] != "\n":
168
- if text[i] == "\\" and i + 1 < n:
169
- i += 2
170
- else:
171
- i += 1
172
- if i < n and text[i] == '"':
173
- i += 1
174
- continue
175
- out.append(ch)
176
- i += 1
177
- return "".join(out)
178
-
179
-
180
- def protected_signature_substring_present(
181
- content: str, expected_signature: str
182
- ) -> bool:
183
- """True when the locked signature text still appears in the file.
184
-
185
- Strips Lean comments from BOTH sides — without that a model could paste the
186
- pinned signature into a ``--`` or ``/- ... -/`` block while rewriting the live
187
- declaration to something trivial, and the asymmetric variant would also misfire
188
- if ``expected_signature`` itself contains a comment. Tries an exact substring
189
- match first, then a whitespace-flexible match (each side collapsed to
190
- single-spaced tokens) so the model can re-indent or reflow whitespace freely.
191
- """
192
- if not expected_signature:
193
- return True
194
- decommented_content = strip_lean_comments(content)
195
- decommented_expected = strip_lean_comments(expected_signature)
196
- if not decommented_expected.strip():
197
- return True
198
- if decommented_expected in decommented_content:
199
- return True
200
- flat_signature = " ".join(decommented_expected.split())
201
- flat_content = " ".join(decommented_content.split())
202
- return flat_signature in flat_content
203
-
204
-
205
- # ── Starter-file construction ────────────────────────────────────────────────
206
-
207
-
208
- def build_starter_file(
209
- formal_statement: str,
210
- header: str = "",
211
- imports: str = "import Mathlib",
212
- normalize: bool = False,
213
- proof_body: str | None = None,
214
- ) -> str:
215
- """Construct the starter proof file.
216
-
217
- Layout: preamble (imports / header) + a brief ``-- DO NOT MODIFY`` comment
218
- block + the normalized theorem signature + the proof body. If ``proof_body`` is
219
- None (the default) the body is the placeholder `` sorry`` — what's planted at
220
- rollout start. A supplied gold ``proof_body`` (e.g. by ``validate``) replaces
221
- the placeholder so the file is the full reference solution.
222
- """
223
- stmt = formal_statement or ""
224
- if stmt.strip().startswith("import "):
225
- imports_block, signature_raw = _split_imports_and_signature(stmt)
226
- preamble = normalize_imports(imports_block) if normalize else imports_block
227
- else:
228
- preamble = _build_preamble(imports or "import Mathlib", header, normalize)
229
- signature_raw = stmt
230
-
231
- signature = _normalize_signature(signature_raw)
232
- body = " sorry" if proof_body is None else proof_body.rstrip()
233
- wrapped = f"{PROTECTED_HEADER_COMMENT}\n{signature}\n{body}\n"
234
- if preamble:
235
- return preamble.rstrip() + "\n\n" + wrapped
236
- return wrapped
237
-
238
-
239
- # ── Compile-output parsing ───────────────────────────────────────────────────
240
-
241
-
242
- def parse_compile_output(output: str) -> tuple[bool, str, int]:
243
- """Parse a ``lake env lean ...; echo EXIT_CODE:$?`` transcript.
244
-
245
- Returns ``(compiled, cleaned_output, exit_code)`` where ``compiled`` is True iff
246
- the compiler exited 0 with no ``declaration uses 'sorry'`` diagnostic.
247
-
248
- Matches the LAST ``EXIT_CODE:N`` — that's the one our shell appends at the end
249
- of the command. Matching the first occurrence would let a model inject
250
- ``#eval IO.println "EXIT_CODE:0"`` into the proof file to bypass the
251
- sorry/exit-code checks: the regex would hit the injected marker, truncate
252
- everything after it (hiding the real ``declaration uses 'sorry'`` diagnostic and
253
- the real EXIT_CODE), and report success.
254
- """
255
- exit_code = 1
256
- matches = list(re.finditer(r"EXIT_CODE:(\d+)", output))
257
- if matches:
258
- last = matches[-1]
259
- exit_code = int(last.group(1))
260
- output = output[: last.start()].strip()
261
- has_sorry = bool(re.search(r"declaration uses 'sorry'", output))
262
- return (exit_code == 0 and not has_sorry), output, exit_code
@@ -1,229 +0,0 @@
1
- """Lean 4 theorem-proving tasks backed by Hugging Face datasets.
2
-
3
- Each task plants a ``sorry``-based starter file, lets any container-capable
4
- harness edit it, and rewards a clean ``lake env lean`` compile. The original
5
- theorem signature must remain present, which prevents replacing the assigned
6
- statement with an easier theorem. Dataset-specific packages can subclass this
7
- taskset and only supply column mappings.
8
- """
9
-
10
- from __future__ import annotations
11
-
12
- import shlex
13
- from collections.abc import Iterator
14
-
15
- from pydantic_config import BaseConfig
16
-
17
- from verifiers.v1.configs.task import TaskConfig
18
- from verifiers.v1.configs.taskset import TasksetConfig
19
- from verifiers.v1.runtimes import Runtime
20
- from verifiers.v1.state import State
21
- from verifiers.v1.task import Task, TaskData, TaskResources
22
- from verifiers.v1.taskset import Taskset
23
- from verifiers.v1.tasksets.lean.scoring import (
24
- build_starter_file,
25
- expected_protected_signature,
26
- parse_compile_output,
27
- protected_signature_substring_present,
28
- )
29
- from verifiers.v1.trace import Trace
30
- from verifiers.v1.utils.decorators import reward
31
-
32
- # Lean v4.27 with Mathlib v4.27.
33
- DEFAULT_DOCKER_IMAGE = "team-clyvldofb0000gg1kx39rgzjq/lean-tactic:mathlib-v4.27.0-v3"
34
- LEAN_PROJECT_PATH = "/workspace/mathlib4"
35
- PROOF_FILE_PATH = "/tmp/proof.lean"
36
-
37
- DEFAULT_SYSTEM_PROMPT = "You are an expert Lean 4 theorem prover working with Mathlib."
38
-
39
-
40
- class LeanDatasetConfig(BaseConfig):
41
- name: str
42
- """HuggingFace dataset id (required; each per-dataset package sets it)."""
43
- split: str = "train"
44
- subset: str | None = None
45
- statement_column: str = "formal_statement"
46
- header_column: str | None = None
47
- imports_column: str | None = None
48
- name_column: str | None = None
49
- proof_column: str | None = None
50
- """Column holding the gold proof body (used by ``validate``); None = no gold."""
51
- normalize_mathlib_imports: bool = False
52
-
53
-
54
- class LeanTaskConfig(TaskConfig):
55
- lean_project_path: str = LEAN_PROJECT_PATH
56
- proof_file_path: str = PROOF_FILE_PATH
57
- compile_timeout: int = 300
58
- """Per-compile ``timeout`` wrapper (seconds), bounding each ``lake env lean``."""
59
-
60
-
61
- class LeanConfig(TasksetConfig):
62
- dataset: LeanDatasetConfig
63
- docker_image: str = DEFAULT_DOCKER_IMAGE
64
- task: LeanTaskConfig = LeanTaskConfig()
65
-
66
-
67
- class LeanData(TaskData):
68
- formal_statement: str
69
- header: str = ""
70
- imports: str = "import Mathlib"
71
- normalize_mathlib_imports: bool = False
72
- # Canonical ``theorem ... := by`` text pinned at load; the reward checks it
73
- # still appears in the final file (the only edit the reward cares about).
74
- protected_signature: str = ""
75
- # Gold proof body (replaces `` sorry``); "" when the dataset ships no gold.
76
- formal_proof: str = ""
77
-
78
-
79
- class LeanTask(Task[LeanData, State, LeanTaskConfig]):
80
- NEEDS_CONTAINER = True
81
-
82
- async def _compile(self, runtime: Runtime) -> tuple[bool, str, int]:
83
- cmd = (
84
- f"cd {shlex.quote(self.config.lean_project_path)} && "
85
- f"timeout {self.config.compile_timeout} lake env lean "
86
- f"{shlex.quote(self.config.proof_file_path)} 2>&1; "
87
- "echo EXIT_CODE:$?"
88
- )
89
- result = await runtime.run(["bash", "-lc", cmd], {})
90
- return parse_compile_output((result.stdout or "") + (result.stderr or ""))
91
-
92
- async def setup(self, runtime: Runtime) -> None:
93
- content = build_starter_file(
94
- self.data.formal_statement,
95
- header=self.data.header,
96
- imports=self.data.imports,
97
- normalize=self.data.normalize_mathlib_imports,
98
- )
99
- await runtime.write(self.config.proof_file_path, content.encode())
100
-
101
- @reward(weight=1.0)
102
- async def lean_compiled(self, trace: Trace, runtime: Runtime) -> float:
103
- """Require both the assigned signature and a clean Lean compile."""
104
- if trace.has_error:
105
- return 0.0
106
-
107
- # Setup created this file, so a read failure is an infrastructure error.
108
- current = (await runtime.read(self.config.proof_file_path)).decode(
109
- "utf-8", "replace"
110
- )
111
-
112
- expected_sig = self.data.protected_signature or expected_protected_signature(
113
- self.data.formal_statement
114
- )
115
- if expected_sig and not protected_signature_substring_present(
116
- current, expected_sig
117
- ):
118
- trace.info["lean_tampered"] = True
119
- trace.info["compile_output"] = "signature rewritten or hidden in a comment"
120
- return 0.0
121
- trace.info["lean_tampered"] = False
122
-
123
- compiled, output, exit_code = await self._compile(runtime)
124
- trace.info["lean_compiled"] = compiled
125
- trace.info["compile_exit_code"] = exit_code
126
- trace.info["compile_output"] = output[-4000:]
127
- return 1.0 if compiled else 0.0
128
-
129
- async def validate(self, runtime: Runtime) -> bool | None:
130
- """Compile the gold proof; rows without one have nothing to preflight."""
131
- gold = (self.data.formal_proof or "").rstrip()
132
- if not gold:
133
- return None
134
- content = build_starter_file(
135
- self.data.formal_statement,
136
- header=self.data.header,
137
- imports=self.data.imports,
138
- normalize=self.data.normalize_mathlib_imports,
139
- proof_body=gold,
140
- )
141
- await runtime.write(self.config.proof_file_path, content.encode())
142
- compiled, _, _ = await self._compile(runtime)
143
- return compiled
144
-
145
-
146
- class LeanTaskset(Taskset[LeanTask, LeanConfig]):
147
- def load(self) -> Iterator[LeanTask]:
148
- try:
149
- from datasets import load_dataset
150
- except ModuleNotFoundError as e:
151
- raise ModuleNotFoundError(
152
- "the Lean taskset requires the `lean` extra; install `verifiers[lean]`"
153
- ) from e
154
-
155
- config = self.config
156
- ds = config.dataset
157
- raw = load_dataset(
158
- ds.name,
159
- ds.subset,
160
- split=ds.split,
161
- num_proc=8,
162
- )
163
-
164
- # Validate optional columns before empty-value fallbacks hide a typo.
165
- for label, col in (
166
- ("statement_column", ds.statement_column),
167
- ("header_column", ds.header_column),
168
- ("imports_column", ds.imports_column),
169
- ("name_column", ds.name_column),
170
- ("proof_column", ds.proof_column),
171
- ):
172
- if col is not None and col not in raw.column_names:
173
- raise ValueError(
174
- f"dataset.{label}={col!r} not found in {ds.name!r}; columns={raw.column_names}"
175
- )
176
-
177
- resources = TaskResources(cpu=4, memory=4, disk=10)
178
- for index, row in enumerate(raw):
179
- # An empty statement would reduce the signature guard to `:= by`.
180
- formal_statement = row[ds.statement_column]
181
- if not isinstance(formal_statement, str) or not formal_statement.strip():
182
- continue
183
- # A disabled optional column is None; row.get(None) yields the same
184
- # empty fallback as a present-but-empty column.
185
- header = row.get(ds.header_column) or ""
186
- imports = row.get(ds.imports_column) or "import Mathlib"
187
- gold = row.get(ds.proof_column) or ""
188
- name = row.get(ds.name_column)
189
- yield LeanTask(
190
- LeanData(
191
- idx=index,
192
- name=str(name) if name else f"task_{index:05d}",
193
- prompt=self._build_prompt(formal_statement, header),
194
- system_prompt=DEFAULT_SYSTEM_PROMPT,
195
- image=config.docker_image,
196
- workdir=config.task.lean_project_path,
197
- resources=resources,
198
- formal_statement=formal_statement,
199
- header=header,
200
- imports=imports,
201
- normalize_mathlib_imports=ds.normalize_mathlib_imports,
202
- protected_signature=expected_protected_signature(formal_statement),
203
- formal_proof=gold,
204
- ),
205
- self.config.task,
206
- )
207
-
208
- def _build_prompt(self, formal_statement: str, header: str) -> str:
209
- cfg = self.config
210
- block = (
211
- "Prove the following Lean 4 theorem. A starter proof file is at "
212
- f"`{cfg.task.proof_file_path}` with the theorem statement and a `sorry` "
213
- "placeholder already in place. Edit it and compile with "
214
- f"`cd {cfg.task.lean_project_path} && lake env lean {cfg.task.proof_file_path}`.\n\n"
215
- f"```lean\n{formal_statement}\n```"
216
- )
217
- if header:
218
- block += f"\n\nThe file header (imports/namespaces) is already set up:\n```lean\n{header}\n```"
219
- block += (
220
- "\n\nDo NOT modify the theorem statement (the lines from `theorem ...` "
221
- "through `:= by`) — the grader checks the original statement still "
222
- "appears and gives zero reward if you rewrote it. Write your proof "
223
- "tactics in place of `sorry`; the final proof must not contain `sorry` "
224
- "or `admit`. A clean compile prints nothing and exits 0."
225
- )
226
- return block
227
-
228
-
229
- __all__ = ["LeanConfig", "LeanTask", "LeanTaskset"]