verifiers 0.3.2.dev127__py3-none-any.whl → 0.3.2.dev129__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/debug.py +4 -0
- verifiers/v1/cli/eval/runner.py +3 -0
- verifiers/v1/cli/gepa.py +4 -0
- verifiers/v1/cli/replay.py +4 -0
- verifiers/v1/cli/validate.py +4 -0
- verifiers/v1/gepa/dataset.py +1 -1
- verifiers/v1/interception/tunnel/prime.py +10 -6
- verifiers/v1/runtimes/limiters.py +22 -30
- verifiers/v1/runtimes/modal.py +5 -2
- verifiers/v1/runtimes/prime.py +6 -3
- verifiers/v1/tasksets/__init__.py +0 -10
- verifiers/v1/utils/scope.py +27 -0
- {verifiers-0.3.2.dev127.dist-info → verifiers-0.3.2.dev129.dist-info}/METADATA +1 -3
- {verifiers-0.3.2.dev127.dist-info → verifiers-0.3.2.dev129.dist-info}/RECORD +17 -19
- verifiers/v1/tasksets/lean/__init__.py +0 -33
- verifiers/v1/tasksets/lean/scoring.py +0 -262
- verifiers/v1/tasksets/lean/taskset.py +0 -229
- {verifiers-0.3.2.dev127.dist-info → verifiers-0.3.2.dev129.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev127.dist-info → verifiers-0.3.2.dev129.dist-info}/entry_points.txt +0 -0
- {verifiers-0.3.2.dev127.dist-info → verifiers-0.3.2.dev129.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/cli/debug.py
CHANGED
|
@@ -3,10 +3,12 @@
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import contextlib
|
|
5
5
|
import logging
|
|
6
|
+
import os
|
|
6
7
|
import shlex
|
|
7
8
|
import sys
|
|
8
9
|
import time
|
|
9
10
|
import traceback
|
|
11
|
+
import uuid
|
|
10
12
|
from collections.abc import Awaitable
|
|
11
13
|
from pathlib import Path
|
|
12
14
|
from typing import Any
|
|
@@ -316,6 +318,8 @@ async def run_debug(config: DebugConfig) -> list[Trace]:
|
|
|
316
318
|
|
|
317
319
|
|
|
318
320
|
def main(argv: list[str] | None = None) -> None:
|
|
321
|
+
# The run identity: every process this run spawns inherits it.
|
|
322
|
+
os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
|
|
319
323
|
argv = with_positional_taskset(
|
|
320
324
|
list(sys.argv[1:]) if argv is None else list(argv), flag="--taskset.id"
|
|
321
325
|
)
|
verifiers/v1/cli/eval/runner.py
CHANGED
|
@@ -9,6 +9,7 @@ runs env servers); this CLI is the quick local path.
|
|
|
9
9
|
import asyncio
|
|
10
10
|
import contextlib
|
|
11
11
|
import logging
|
|
12
|
+
import os
|
|
12
13
|
import time
|
|
13
14
|
from collections.abc import AsyncIterator, Awaitable, Callable, Iterable
|
|
14
15
|
from typing import TypeVar, cast
|
|
@@ -139,6 +140,8 @@ async def run_eval(config: EvalConfig) -> list[Episode]:
|
|
|
139
140
|
|
|
140
141
|
# Opened before the first rollout so every episode streams as it lands.
|
|
141
142
|
run = open_run(config, push_state, num_examples=len(tasks))
|
|
143
|
+
# The run identity: every process this run spawns inherits it.
|
|
144
|
+
os.environ.setdefault("VF_RUN_ID", config.run.id)
|
|
142
145
|
# Resumed rollouts are part of this run too.
|
|
143
146
|
log_episodes(run, finished)
|
|
144
147
|
|
verifiers/v1/cli/gepa.py
CHANGED
|
@@ -9,7 +9,9 @@ and the actual parse is `pydantic_config.cli`.
|
|
|
9
9
|
"""
|
|
10
10
|
|
|
11
11
|
import logging
|
|
12
|
+
import os
|
|
12
13
|
import sys
|
|
14
|
+
import uuid
|
|
13
15
|
|
|
14
16
|
from pydantic_config import cli
|
|
15
17
|
|
|
@@ -32,6 +34,8 @@ USAGE = "usage: uv run vf-gepa [<taskset-id>] [--env.id <id>] --model <model> [o
|
|
|
32
34
|
|
|
33
35
|
|
|
34
36
|
def main(argv: list[str] | None = None) -> None:
|
|
37
|
+
# The run identity: every process this run spawns inherits it.
|
|
38
|
+
os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
|
|
35
39
|
argv = with_positional_taskset(list(sys.argv[1:]) if argv is None else list(argv))
|
|
36
40
|
|
|
37
41
|
if not argv or any(arg in ("-h", "--help") for arg in argv):
|
verifiers/v1/cli/replay.py
CHANGED
|
@@ -12,8 +12,10 @@ import asyncio
|
|
|
12
12
|
import contextlib
|
|
13
13
|
import json
|
|
14
14
|
import logging
|
|
15
|
+
import os
|
|
15
16
|
import sys
|
|
16
17
|
import time
|
|
18
|
+
import uuid
|
|
17
19
|
from pathlib import Path
|
|
18
20
|
|
|
19
21
|
from pydantic_config import cli
|
|
@@ -197,6 +199,8 @@ async def run_replay(config: ReplayConfig, source: Path, out: Path) -> list[Trac
|
|
|
197
199
|
|
|
198
200
|
|
|
199
201
|
def main(argv: list[str] | None = None) -> None:
|
|
202
|
+
# The run identity: every process this run spawns inherits it.
|
|
203
|
+
os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
|
|
200
204
|
argv = list(sys.argv[1:]) if argv is None else list(argv)
|
|
201
205
|
if not argv or any(a in ("-h", "--help") for a in argv):
|
|
202
206
|
print(USAGE)
|
verifiers/v1/cli/validate.py
CHANGED
|
@@ -4,9 +4,11 @@ import asyncio
|
|
|
4
4
|
import contextlib
|
|
5
5
|
import json
|
|
6
6
|
import logging
|
|
7
|
+
import os
|
|
7
8
|
import shutil
|
|
8
9
|
import sys
|
|
9
10
|
import time
|
|
11
|
+
import uuid
|
|
10
12
|
from collections import Counter, defaultdict
|
|
11
13
|
from collections.abc import Mapping, Sequence
|
|
12
14
|
from pathlib import Path
|
|
@@ -425,6 +427,8 @@ async def run_validate(config: ValidateConfig) -> list[dict]:
|
|
|
425
427
|
|
|
426
428
|
|
|
427
429
|
def main(argv: list[str] | None = None) -> None:
|
|
430
|
+
# The run identity: every process this run spawns inherits it.
|
|
431
|
+
os.environ.setdefault("VF_RUN_ID", uuid.uuid4().hex)
|
|
428
432
|
argv = with_positional_taskset(
|
|
429
433
|
list(sys.argv[1:]) if argv is None else list(argv), flag="--taskset.id"
|
|
430
434
|
)
|
verifiers/v1/gepa/dataset.py
CHANGED
|
@@ -40,5 +40,5 @@ def resolve_gepa_seed_prompt(tasks: list[Task], initial_prompt: str | None) -> s
|
|
|
40
40
|
"no task in this taskset sets Task.system_prompt — some tasksets bake instructions "
|
|
41
41
|
"directly into `prompt` instead (e.g. gsm8k) and can't be optimized this way. Pass "
|
|
42
42
|
"--initial-prompt to seed one explicitly, or pick a taskset whose load() sets "
|
|
43
|
-
"system_prompt on its task data (e.g. reverse-text,
|
|
43
|
+
"system_prompt on its task data (e.g. reverse-text, textarena)."
|
|
44
44
|
)
|
|
@@ -9,15 +9,19 @@ from collections.abc import AsyncIterator
|
|
|
9
9
|
from typing import Literal
|
|
10
10
|
|
|
11
11
|
from verifiers.v1.interception.tunnel.base import BaseTunnelConfig, Tunnel
|
|
12
|
-
from verifiers.v1.runtimes.limiters import
|
|
12
|
+
from verifiers.v1.runtimes.limiters import CreationLimiter
|
|
13
13
|
from verifiers.v1.utils.aio import run_shielded
|
|
14
14
|
from verifiers.v1.utils.prime import ensure_prime_auth
|
|
15
|
+
from verifiers.v1.utils.scope import run_scope
|
|
15
16
|
|
|
16
17
|
# The prime_tunnel service caps tunnel starts at 512/min per API token — a property of the
|
|
17
|
-
# tunnel service, shared by every process
|
|
18
|
+
# tunnel service, shared by every process of a run that opens one. One run-scoped
|
|
18
19
|
# limiter, not a per-runtime config knob.
|
|
19
20
|
_TUNNELS_PER_MIN = 512
|
|
20
|
-
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def tunnel_limiter() -> CreationLimiter:
|
|
24
|
+
return CreationLimiter("prime-tunnel", run_scope(), _TUNNELS_PER_MIN / 60)
|
|
21
25
|
|
|
22
26
|
|
|
23
27
|
class PrimeTunnelConfig(BaseTunnelConfig):
|
|
@@ -35,8 +39,8 @@ class PrimeTunnel(Tunnel[PrimeTunnelConfig]):
|
|
|
35
39
|
@contextlib.asynccontextmanager
|
|
36
40
|
async def expose(self, port: int) -> AsyncIterator[str]:
|
|
37
41
|
"""Bridge the host `port` to a public URL via prime_tunnel (frpc). Tunnel creation
|
|
38
|
-
is network-bound and
|
|
39
|
-
`
|
|
42
|
+
is network-bound and rate-capped (512/min, run-wide via the shared
|
|
43
|
+
`tunnel_limiter`), so transient failures are retried; a terminal one raises
|
|
40
44
|
`TunnelError`. The tunnel is torn down on exit."""
|
|
41
45
|
from prime_tunnel import Tunnel as TunnelClient
|
|
42
46
|
|
|
@@ -48,7 +52,7 @@ class PrimeTunnel(Tunnel[PrimeTunnelConfig]):
|
|
|
48
52
|
async for attempt in retrying(retries=3, label=label):
|
|
49
53
|
with attempt:
|
|
50
54
|
client = TunnelClient(local_port=port)
|
|
51
|
-
async with
|
|
55
|
+
async with tunnel_limiter():
|
|
52
56
|
url = str(await client.start()).rstrip("/")
|
|
53
57
|
except Exception as e:
|
|
54
58
|
raise TunnelError(f"{label} failed: {e}") from e
|
|
@@ -1,11 +1,12 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""Creation-rate limiters for the remote runtimes.
|
|
2
2
|
|
|
3
3
|
A leaky bucket backed by a lock file under the user cache (``~/.cache/verifiers``, falling
|
|
4
|
-
back to the temp dir when no home is resolvable), so a provider's
|
|
5
|
-
|
|
6
|
-
the
|
|
7
|
-
just within one process.
|
|
8
|
-
run
|
|
4
|
+
back to the temp dir when no home is resolvable), so a provider's creation rate (Modal
|
|
5
|
+
sandboxes, Prime sandboxes and tunnels) is enforced across EVERY process that shares the
|
|
6
|
+
bucket — the eval process and all the elastically-spawned env-server worker processes alike —
|
|
7
|
+
not just within one process. A bucket is named by the limiter's name and a scope the caller
|
|
8
|
+
picks; the runtimes pass the run id, so one run's backlog (or the reservations a killed run
|
|
9
|
+
left behind) never delays another run.
|
|
9
10
|
"""
|
|
10
11
|
|
|
11
12
|
import asyncio
|
|
@@ -22,17 +23,18 @@ LIMITER_DIR = CACHE_DIR / "limiter"
|
|
|
22
23
|
class CreationLimiter:
|
|
23
24
|
"""An async leaky bucket shared across processes via a lock file: each `async with`
|
|
24
25
|
reserves the next `1/per_sec`-spaced slot (advancing the on-disk cursor under an exclusive
|
|
25
|
-
flock) and sleeps until it, so the aggregate creation rate across
|
|
26
|
-
|
|
27
|
-
hold the lock.
|
|
26
|
+
flock) and sleeps until it, so the aggregate creation rate across every process sharing
|
|
27
|
+
the bucket stays at `per_sec`. The reservation runs off the event loop; the wait does not
|
|
28
|
+
hold the lock. Reservations are never released, so a cancelled waiter still holds its
|
|
29
|
+
slot; the backlog drains at `per_sec` regardless."""
|
|
28
30
|
|
|
29
|
-
def __init__(self, name: str, per_sec: float) -> None:
|
|
31
|
+
def __init__(self, name: str, scope: str, per_sec: float) -> None:
|
|
30
32
|
self._interval = 1 / per_sec
|
|
31
|
-
self._path = LIMITER_DIR / f"{name}.bucket"
|
|
33
|
+
self._path = LIMITER_DIR / f"{name}-{scope.replace('/', '--')}.bucket"
|
|
32
34
|
|
|
33
35
|
def _reserve(self) -> float:
|
|
34
36
|
os.makedirs(LIMITER_DIR, exist_ok=True)
|
|
35
|
-
# Shared buckets require a clock comparable across
|
|
37
|
+
# Shared buckets require a clock comparable across the run's hosts.
|
|
36
38
|
with open(self._path, "a+") as f:
|
|
37
39
|
fcntl.flock(f.fileno(), fcntl.LOCK_EX)
|
|
38
40
|
try:
|
|
@@ -40,19 +42,13 @@ class CreationLimiter:
|
|
|
40
42
|
data = f.read().strip()
|
|
41
43
|
now = time.time()
|
|
42
44
|
slot = max(now, float(data) if data else 0.0)
|
|
43
|
-
wait = slot - now
|
|
44
|
-
if wait > 5 * 60:
|
|
45
|
-
raise TimeoutError(
|
|
46
|
-
f"{self._path.stem} creation limiter backlog of {wait:.1f}s "
|
|
47
|
-
f"exceeds 300s ({self._path})"
|
|
48
|
-
)
|
|
49
45
|
f.seek(0)
|
|
50
46
|
f.truncate()
|
|
51
47
|
f.write(repr(slot + self._interval))
|
|
52
48
|
f.flush()
|
|
53
|
-
return wait
|
|
54
49
|
finally:
|
|
55
50
|
fcntl.flock(f.fileno(), fcntl.LOCK_UN)
|
|
51
|
+
return slot - now
|
|
56
52
|
|
|
57
53
|
async def __aenter__(self) -> Self:
|
|
58
54
|
wait = await asyncio.to_thread(self._reserve)
|
|
@@ -64,16 +60,12 @@ class CreationLimiter:
|
|
|
64
60
|
return False
|
|
65
61
|
|
|
66
62
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
All callers (and processes) sharing a `name` share one bucket, so use one rate per name."""
|
|
63
|
+
def creation_limiter(
|
|
64
|
+
per_sec: float | None, name: str, scope: str
|
|
65
|
+
) -> CreationLimiter | None:
|
|
66
|
+
"""A limiter pacing `name`'s creation to `per_sec`/s within `scope` (None/<= 0
|
|
67
|
+
disables). All callers (and processes) sharing a name and scope share one bucket, so
|
|
68
|
+
use one rate per name."""
|
|
74
69
|
if not per_sec or per_sec <= 0:
|
|
75
70
|
return None
|
|
76
|
-
|
|
77
|
-
if limiter is None:
|
|
78
|
-
limiter = _creation_limiters[name] = CreationLimiter(name, per_sec)
|
|
79
|
-
return limiter
|
|
71
|
+
return CreationLimiter(name, scope, per_sec)
|
verifiers/v1/runtimes/modal.py
CHANGED
|
@@ -31,6 +31,7 @@ from verifiers.v1.runtimes.base import (
|
|
|
31
31
|
)
|
|
32
32
|
from verifiers.v1.runtimes.limiters import creation_limiter
|
|
33
33
|
from verifiers.v1.utils.aio import run_shielded
|
|
34
|
+
from verifiers.v1.utils.scope import run_scope
|
|
34
35
|
|
|
35
36
|
logger = logging.getLogger(__name__)
|
|
36
37
|
|
|
@@ -93,7 +94,7 @@ class ModalConfig(NetworkPolicyConfig):
|
|
|
93
94
|
"""Disk in GB. Modal sandboxes have no disk knob, so this is accepted (so a task can
|
|
94
95
|
declare it without a warning) but not enforced."""
|
|
95
96
|
creates_per_sec: float | None = 40.0
|
|
96
|
-
"""Pace sandbox creation to this many per second, enforced
|
|
97
|
+
"""Pace sandbox creation to this many per second, enforced run-wide across every
|
|
97
98
|
env-server worker process (None/<= 0 disables it)."""
|
|
98
99
|
|
|
99
100
|
@model_validator(mode="after")
|
|
@@ -181,7 +182,9 @@ class ModalRuntime(Runtime):
|
|
|
181
182
|
try:
|
|
182
183
|
app = await modal.App.lookup.aio(_APP_NAME, create_if_missing=True)
|
|
183
184
|
async with (
|
|
184
|
-
creation_limiter(
|
|
185
|
+
creation_limiter(
|
|
186
|
+
self.config.creates_per_sec, "modal-sandbox", run_scope()
|
|
187
|
+
)
|
|
185
188
|
or contextlib.nullcontext()
|
|
186
189
|
):
|
|
187
190
|
await run_shielded(self._create_sandbox(app))
|
verifiers/v1/runtimes/prime.py
CHANGED
|
@@ -28,6 +28,7 @@ from verifiers.v1.runtimes.base import (
|
|
|
28
28
|
from verifiers.v1.runtimes.limiters import creation_limiter
|
|
29
29
|
from verifiers.v1.utils.aio import run_shielded
|
|
30
30
|
from verifiers.v1.utils.prime import ensure_prime_auth
|
|
31
|
+
from verifiers.v1.utils.scope import run_scope
|
|
31
32
|
|
|
32
33
|
logger = logging.getLogger(__name__)
|
|
33
34
|
|
|
@@ -87,9 +88,9 @@ class PrimeConfig(NetworkPolicyConfig):
|
|
|
87
88
|
idle_timeout: float | None = 3600
|
|
88
89
|
"""Seconds of inactivity before the sandbox self-deletes (None disables)."""
|
|
89
90
|
creates_per_min: int | None = None
|
|
90
|
-
"""Pace sandbox creation to this many per minute, enforced
|
|
91
|
+
"""Pace sandbox creation to this many per minute, enforced run-wide across every
|
|
91
92
|
env-server worker process (None/<= 0 disables it). (Tunnel creation is limited separately
|
|
92
|
-
|
|
93
|
+
— see interception.tunnel.prime.tunnel_limiter.)"""
|
|
93
94
|
|
|
94
95
|
@model_validator(mode="after")
|
|
95
96
|
def _validate_egress(self) -> "PrimeConfig":
|
|
@@ -175,7 +176,9 @@ class PrimeRuntime(Runtime):
|
|
|
175
176
|
try:
|
|
176
177
|
async with (
|
|
177
178
|
creation_limiter(
|
|
178
|
-
(self.config.creates_per_min or 0) / 60,
|
|
179
|
+
(self.config.creates_per_min or 0) / 60,
|
|
180
|
+
"prime-sandbox",
|
|
181
|
+
run_scope(),
|
|
179
182
|
)
|
|
180
183
|
or contextlib.nullcontext()
|
|
181
184
|
):
|
|
@@ -1,10 +1,4 @@
|
|
|
1
1
|
from verifiers.v1.tasksets.harbor import HarborConfig, HarborTaskset
|
|
2
|
-
from verifiers.v1.tasksets.lean import (
|
|
3
|
-
LeanConfig,
|
|
4
|
-
LeanDatasetConfig,
|
|
5
|
-
LeanTask,
|
|
6
|
-
LeanTaskset,
|
|
7
|
-
)
|
|
8
2
|
from verifiers.v1.tasksets.nemo_gym import NeMoGymConfig, NeMoGymTaskset
|
|
9
3
|
from verifiers.v1.tasksets.openenv import (
|
|
10
4
|
OpenEnvConfig,
|
|
@@ -18,10 +12,6 @@ from verifiers.v1.tasksets.openenv import (
|
|
|
18
12
|
__all__ = [
|
|
19
13
|
"HarborConfig",
|
|
20
14
|
"HarborTaskset",
|
|
21
|
-
"LeanConfig",
|
|
22
|
-
"LeanDatasetConfig",
|
|
23
|
-
"LeanTask",
|
|
24
|
-
"LeanTaskset",
|
|
25
15
|
"NeMoGymConfig",
|
|
26
16
|
"NeMoGymTaskset",
|
|
27
17
|
"OpenEnvConfig",
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""The scope that state shared by a run's processes is keyed by."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
import uuid
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger(__name__)
|
|
8
|
+
|
|
9
|
+
_process_scope: str | None = None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def run_scope() -> str:
|
|
13
|
+
"""``$VF_RUN_ID``, set by every entrypoint (the `vf-*` CLIs and the prime-rl launchers)
|
|
14
|
+
and inherited by spawned env servers and pool workers. A process started without one
|
|
15
|
+
gets a scope of its own, minted once per process."""
|
|
16
|
+
global _process_scope
|
|
17
|
+
run_id = os.environ.get("VF_RUN_ID")
|
|
18
|
+
if run_id:
|
|
19
|
+
return run_id
|
|
20
|
+
if _process_scope is None:
|
|
21
|
+
_process_scope = uuid.uuid4().hex
|
|
22
|
+
logger.warning(
|
|
23
|
+
"VF_RUN_ID is unset - run-scoped state such as creation limiters covers only "
|
|
24
|
+
"this process (scope %s)",
|
|
25
|
+
_process_scope,
|
|
26
|
+
)
|
|
27
|
+
return _process_scope
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.3.2.
|
|
3
|
+
Version: 0.3.2.dev129
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -48,8 +48,6 @@ Requires-Dist: typing-extensions>=4.12.2
|
|
|
48
48
|
Requires-Dist: uvicorn>=0.52.0
|
|
49
49
|
Provides-Extra: harbor
|
|
50
50
|
Requires-Dist: harbor==0.21.0; (python_full_version >= '3.12') and extra == 'harbor'
|
|
51
|
-
Provides-Extra: lean
|
|
52
|
-
Requires-Dist: datasets<6.0.0,>=3.3.0; extra == 'lean'
|
|
53
51
|
Provides-Extra: modal
|
|
54
52
|
Requires-Dist: modal>=1.5.4; extra == 'modal'
|
|
55
53
|
Provides-Extra: openenv
|
|
@@ -18,14 +18,14 @@ verifiers/v1/types.py,sha256=xPetvJxksKPYKic389RFnVvy2vN-gD2bPZFZoEyzmp4,10191
|
|
|
18
18
|
verifiers/v1/acp/__init__.py,sha256=9RySmxFeEMT2XSJy5wYevqEhQdul2jHl9f5XAribG3A,13631
|
|
19
19
|
verifiers/v1/acp/runner.py,sha256=zPo-2ZXmFMmQchhD7nNzlZUrabG09qiBvpB67KVGm3M,14095
|
|
20
20
|
verifiers/v1/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
21
|
-
verifiers/v1/cli/debug.py,sha256=
|
|
22
|
-
verifiers/v1/cli/gepa.py,sha256=
|
|
21
|
+
verifiers/v1/cli/debug.py,sha256=sgXi2y17qr2yd5kbkL_IzwWN1nIFBOa1adu7bcFzaVk,11780
|
|
22
|
+
verifiers/v1/cli/gepa.py,sha256=F0Q1ApfUL_mInFnmaJ5-yBDRYBtBuH72OkWelPoGCp4,4332
|
|
23
23
|
verifiers/v1/cli/init.py,sha256=6wr14011_Dygv10R6_YWg5A3oSrz187E2WspGRHIfZw,7608
|
|
24
24
|
verifiers/v1/cli/output.py,sha256=lAl6vwKf9mVuNZb8PzOexv6Q2h-H6t0oUPcapffY3uY,8351
|
|
25
|
-
verifiers/v1/cli/replay.py,sha256=
|
|
25
|
+
verifiers/v1/cli/replay.py,sha256=JjhaGczQ41zN6wz1P1nbGGvv0xdfQlX4OmZItGPDpZc,10395
|
|
26
26
|
verifiers/v1/cli/resolve.py,sha256=Q1EHM7wWQo0YkwJA898qtZbYLaIkjIVq3Q7kUp2YIxs,4346
|
|
27
27
|
verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,470
|
|
28
|
-
verifiers/v1/cli/validate.py,sha256=
|
|
28
|
+
verifiers/v1/cli/validate.py,sha256=Z_rUldl8jzXz52oy3FBIRGr4HYhWFXrq43WmU0sJ-4M,17325
|
|
29
29
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
30
30
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
31
31
|
verifiers/v1/cli/dashboard/eval.py,sha256=HEXPEegNXdgU9zAIuHzsWsoSIHyiOcDZh4tCGFNxKAk,38353
|
|
@@ -35,7 +35,7 @@ verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3h
|
|
|
35
35
|
verifiers/v1/cli/eval/hint.py,sha256=NXP2_JYrQHguqr68LxOJG3lYAsue_VNoepyTVPn-N4E,256
|
|
36
36
|
verifiers/v1/cli/eval/main.py,sha256=L_vKnMJEbpo1IKjbm0TXjXv_1x9we7mFv-jH_xY211Q,5946
|
|
37
37
|
verifiers/v1/cli/eval/resume.py,sha256=fYkWa0Fudq--IWfk25j_McLm67bXYWvksWmJR4u4QAU,3811
|
|
38
|
-
verifiers/v1/cli/eval/runner.py,sha256=
|
|
38
|
+
verifiers/v1/cli/eval/runner.py,sha256=z1yU2_rAA6tHz3PL7NQuzm9v3-4Ri18DXD5SRBKXqVY,6962
|
|
39
39
|
verifiers/v1/clients/__init__.py,sha256=Ysig0tE_0E4Jsfgfes1XHN-fK1s_RCXdqZD6E7lK4PU,507
|
|
40
40
|
verifiers/v1/clients/base.py,sha256=PoDw4GMqrfuPTFVK6n2jDmlD0_lJ5zTrYcNio5yq6K8,1691
|
|
41
41
|
verifiers/v1/clients/client.py,sha256=zqC_AkiD9pl0kxxIbupNiehbS-YnVdp3EMZ0LOHTd8U,3125
|
|
@@ -79,7 +79,7 @@ verifiers/v1/envs/user_sim/env.py,sha256=deI4r-Utn_v4E_Vb9P_X3LzFh2GRkCO4OlQt8BO
|
|
|
79
79
|
verifiers/v1/gepa/__init__.py,sha256=6nmdRE0-34AKioPHBjTxUg5Jo_2z7tMX-OU3zMNpAJI,197
|
|
80
80
|
verifiers/v1/gepa/adapter.py,sha256=YNvHMR2L5Utl-vtPfjZxmDCd2tTcoAPWG6aVGnV1icM,6139
|
|
81
81
|
verifiers/v1/gepa/config.py,sha256=PhZRLkhaPciNm_IhwVX3e9DM8Y2IdCe4Fbi5WjH5aro,4883
|
|
82
|
-
verifiers/v1/gepa/dataset.py,sha256=
|
|
82
|
+
verifiers/v1/gepa/dataset.py,sha256=SPGG55BVbJ9HtCEvm-xyl52QSbldFwGqWvRIcaLelsM,2171
|
|
83
83
|
verifiers/v1/gepa/reflection.py,sha256=ptHhx0lcDLcWhWhXuslh9T8gPVNpnX5XLPtXoneW1yA,1054
|
|
84
84
|
verifiers/v1/gepa/runner.py,sha256=fGEyFzejdQKBzBbvdiTDHV3JtRhXMXMGSBU5y8C8i3w,6054
|
|
85
85
|
verifiers/v1/harnesses/__init__.py,sha256=2JPwrRoYoigH-HfBtiFnatRdFEVHLv4rwlKfsFEibyE,1803
|
|
@@ -129,7 +129,7 @@ verifiers/v1/interception/server.py,sha256=wxw67OQkLgzvt1HMJyjYmdcoIhn4vuJBe5oXZ
|
|
|
129
129
|
verifiers/v1/interception/tunnel/__init__.py,sha256=eVNZJszj6myrKhx9jmJz7VRd9qlfj0AUvUZwI2bXEj8,893
|
|
130
130
|
verifiers/v1/interception/tunnel/base.py,sha256=DZB6uPLwM4Qy7n7m0vKeyg-Px87MhsmpYTpe2vpPEng,1998
|
|
131
131
|
verifiers/v1/interception/tunnel/custom.py,sha256=yL4UbGf4bAuMsxk42mEP-t1wITQCT8BE9FHQIz0qRMA,1708
|
|
132
|
-
verifiers/v1/interception/tunnel/prime.py,sha256=
|
|
132
|
+
verifiers/v1/interception/tunnel/prime.py,sha256=jVk_6xyrxXiHtsVFErcRNd4MoENdFbqzfC0i20nlizU,2803
|
|
133
133
|
verifiers/v1/judges/__init__.py,sha256=MUIBWcx6c70BykrTDHhB0ITIyJPxXe6xDBou4VM7jDQ,286
|
|
134
134
|
verifiers/v1/judges/reference.py,sha256=iEVHw-iJ6dbwLOOqrvWrEMJOcUEYJ0YspT04rZz7Sd4,3868
|
|
135
135
|
verifiers/v1/judges/reference.txt,sha256=Ej35kGXiT2uJ0LcHIeez49rCS_9uHmCQKoVdn93V6A8,353
|
|
@@ -143,9 +143,9 @@ verifiers/v1/runtimes/__init__.py,sha256=JQlQ029J14asfYb-UxpENshMyPztWbSL0-ZqSyD
|
|
|
143
143
|
verifiers/v1/runtimes/apptainer.py,sha256=Ru9QJna_l0goIzg71G-4RU_D-7wZUNz139S5TEvbzog,8211
|
|
144
144
|
verifiers/v1/runtimes/base.py,sha256=IKmrJSJltWX9M1nPUZExKluPJM4N82VsnqQGFz0ECV8,17777
|
|
145
145
|
verifiers/v1/runtimes/container.py,sha256=puQD8X9acP6GefoCNEnq40POi-nXPD-729bUjTVShfM,9554
|
|
146
|
-
verifiers/v1/runtimes/limiters.py,sha256=
|
|
147
|
-
verifiers/v1/runtimes/modal.py,sha256=
|
|
148
|
-
verifiers/v1/runtimes/prime.py,sha256=
|
|
146
|
+
verifiers/v1/runtimes/limiters.py,sha256=rOfQJcMdYcLSKq8AL9O09BfGOLqpYmNgrCxXXNcHLmU,2849
|
|
147
|
+
verifiers/v1/runtimes/modal.py,sha256=6ccitC0HVsAP_l8soJ_RQYNIVKrJjDsz2vAJBtxUZjw,17476
|
|
148
|
+
verifiers/v1/runtimes/prime.py,sha256=nTtu-uBjc39UxfMPEyndshfAiIdk1V4L4aE6zSVVSn0,17673
|
|
149
149
|
verifiers/v1/runtimes/subprocess.py,sha256=QmpQ235_xrX6gowrr8557VgbydOXpLbfFq0KKGfJThk,8774
|
|
150
150
|
verifiers/v1/runtimes/docker/__init__.py,sha256=2wceP-8V-w0dQBEeLtoJFfCZ135DxP2rcdcUBmSWiNU,18543
|
|
151
151
|
verifiers/v1/runtimes/docker/egress.py,sha256=wtKWTL2_zFidxMR7lDRmfevQlOsl2Iyk88ebdO4Kqqw,24566
|
|
@@ -156,14 +156,11 @@ verifiers/v1/serve/encoding.py,sha256=hBZFucAZK9riXOV3DaHcskq9zHTT8gVGkoFVOPfnr3
|
|
|
156
156
|
verifiers/v1/serve/pool.py,sha256=bT-FOlIuOiBtOasuPB-p9887tkpEqvQUJBgn1Q45ph0,15912
|
|
157
157
|
verifiers/v1/serve/server.py,sha256=i6XrY_PMAapkrP54tYX8-K6iGmOmrR7BbQ-3Qhn3w4s,9885
|
|
158
158
|
verifiers/v1/serve/types.py,sha256=Qw5IjiJhsZz4giSXUBi9dfInBnQZ5_xiUmgH2qp8jPI,1741
|
|
159
|
-
verifiers/v1/tasksets/__init__.py,sha256=
|
|
159
|
+
verifiers/v1/tasksets/__init__.py,sha256=2ijE3qIlLoUA02g6iysOwWQylyBBQL1GhSpjv8_bp0w,521
|
|
160
160
|
verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
|
|
161
161
|
verifiers/v1/tasksets/harbor/env.py,sha256=ATMkfwmrE_2jpOOffbjLQ91feG7TOtefBGlpwWbRq8c,3604
|
|
162
162
|
verifiers/v1/tasksets/harbor/taskset.py,sha256=35ljhNapn1EOlaYFTuVZxAMVU0KDfmFxQOGOcUk2Ow0,34082
|
|
163
163
|
verifiers/v1/tasksets/harbor/toolset.py,sha256=_7HXu-uuvqzhmJ1EX6c3KWLg0g0eRrLjCq1kBsZhIwo,3159
|
|
164
|
-
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
165
|
-
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
166
|
-
verifiers/v1/tasksets/lean/taskset.py,sha256=Jax76i22G9dOAswQHiWujVbFIM_PUz5yGJI_ShR4n8Y,9238
|
|
167
164
|
verifiers/v1/tasksets/nemo_gym/__init__.py,sha256=654gAvG3Wd_Z_nh6mypeKn4s_IIQ0Ge6vcZvxTHC2_E,306
|
|
168
165
|
verifiers/v1/tasksets/nemo_gym/response.py,sha256=EhrsquEP8g0bjzAThGfE6Ypgy86_nVX235NiFYtVDVM,4463
|
|
169
166
|
verifiers/v1/tasksets/nemo_gym/server.py,sha256=33C-VDZxPtMV2TFxJVIKijmtBLNlVeRs92ctPE3qAsM,2036
|
|
@@ -190,10 +187,11 @@ verifiers/v1/utils/paths.py,sha256=it_4JWf_8bPZe4TRcyvL7_KJ4fsbEsxhhIY_xA8Ikus,3
|
|
|
190
187
|
verifiers/v1/utils/platform.py,sha256=YwTHVA1Onn74lG6S3sdHus9OlP-rAbwkk9Po7h-BjpE,7159
|
|
191
188
|
verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,970
|
|
192
189
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
190
|
+
verifiers/v1/utils/scope.py,sha256=bzUEiWDOVdbj7dXOwGOlzsQ_-Xu0j9NR5TalQB-8TS8,838
|
|
193
191
|
verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
|
|
194
192
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
195
|
-
verifiers-0.3.2.
|
|
196
|
-
verifiers-0.3.2.
|
|
197
|
-
verifiers-0.3.2.
|
|
198
|
-
verifiers-0.3.2.
|
|
199
|
-
verifiers-0.3.2.
|
|
193
|
+
verifiers-0.3.2.dev129.dist-info/METADATA,sha256=ckfiJIStyeQf1e_fIyVWL-7jtc5yiEqtyEWdvxHCNKQ,4161
|
|
194
|
+
verifiers-0.3.2.dev129.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
195
|
+
verifiers-0.3.2.dev129.dist-info/entry_points.txt,sha256=iugElcdWPKbQM7uFF0lZ8iUpHsNr17-BwEAAjJWxV3U,259
|
|
196
|
+
verifiers-0.3.2.dev129.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
197
|
+
verifiers-0.3.2.dev129.dist-info/RECORD,,
|
|
@@ -1,33 +0,0 @@
|
|
|
1
|
-
from verifiers.v1.tasksets.lean.scoring import (
|
|
2
|
-
build_starter_file,
|
|
3
|
-
expected_protected_signature,
|
|
4
|
-
parse_compile_output,
|
|
5
|
-
protected_signature_substring_present,
|
|
6
|
-
strip_lean_comments,
|
|
7
|
-
)
|
|
8
|
-
from verifiers.v1.tasksets.lean.taskset import (
|
|
9
|
-
DEFAULT_DOCKER_IMAGE,
|
|
10
|
-
LEAN_PROJECT_PATH,
|
|
11
|
-
PROOF_FILE_PATH,
|
|
12
|
-
LeanConfig,
|
|
13
|
-
LeanDatasetConfig,
|
|
14
|
-
LeanTask,
|
|
15
|
-
LeanTaskConfig,
|
|
16
|
-
LeanTaskset,
|
|
17
|
-
)
|
|
18
|
-
|
|
19
|
-
__all__ = [
|
|
20
|
-
"DEFAULT_DOCKER_IMAGE",
|
|
21
|
-
"LEAN_PROJECT_PATH",
|
|
22
|
-
"PROOF_FILE_PATH",
|
|
23
|
-
"LeanConfig",
|
|
24
|
-
"LeanDatasetConfig",
|
|
25
|
-
"LeanTask",
|
|
26
|
-
"LeanTaskConfig",
|
|
27
|
-
"LeanTaskset",
|
|
28
|
-
"build_starter_file",
|
|
29
|
-
"expected_protected_signature",
|
|
30
|
-
"parse_compile_output",
|
|
31
|
-
"protected_signature_substring_present",
|
|
32
|
-
"strip_lean_comments",
|
|
33
|
-
]
|
|
@@ -1,262 +0,0 @@
|
|
|
1
|
-
"""Pure helpers for the Lean taskset: starter-file construction, theorem-signature
|
|
2
|
-
canonicalization, the reward-hacking signature guard, and compile-output parsing.
|
|
3
|
-
|
|
4
|
-
Everything here is a pure function over plain strings — no runtime, no sandbox, no
|
|
5
|
-
verifiers types — so it's trivially unit-testable and shared between ``setup``
|
|
6
|
-
(building the starter file), the reward (the signature guard + compile parsing),
|
|
7
|
-
and ``validate`` (planting the gold proof).
|
|
8
|
-
"""
|
|
9
|
-
|
|
10
|
-
from __future__ import annotations
|
|
11
|
-
|
|
12
|
-
import re
|
|
13
|
-
|
|
14
|
-
PROTECTED_HEADER_COMMENT = (
|
|
15
|
-
"-- DO NOT MODIFY the theorem statement below. The grader checks\n"
|
|
16
|
-
"-- that the original `theorem ... := by` text still appears in\n"
|
|
17
|
-
"-- this file. Only edit the proof body (currently `sorry`) and\n"
|
|
18
|
-
"-- lines after it."
|
|
19
|
-
)
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
# ── Imports / preamble normalization ─────────────────────────────────────────
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
def normalize_imports(text: str) -> str:
|
|
26
|
-
"""Collapse every ``import Mathlib*`` line to a single ``import Mathlib``.
|
|
27
|
-
|
|
28
|
-
Some datasets (miniF2F) ship fine-grained Mathlib imports that don't resolve
|
|
29
|
-
against the monolithic ``import Mathlib`` the sandbox image provides.
|
|
30
|
-
"""
|
|
31
|
-
lines = text.split("\n")
|
|
32
|
-
out: list[str] = []
|
|
33
|
-
inserted = False
|
|
34
|
-
for line in lines:
|
|
35
|
-
if line.strip().startswith("import Mathlib"):
|
|
36
|
-
if not inserted:
|
|
37
|
-
out.append("import Mathlib")
|
|
38
|
-
inserted = True
|
|
39
|
-
else:
|
|
40
|
-
out.append(line)
|
|
41
|
-
return "\n".join(out)
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
def _build_preamble(imports_str: str, header: str, normalize: bool) -> str:
|
|
45
|
-
if header and header.strip().startswith("import"):
|
|
46
|
-
preamble = header.strip()
|
|
47
|
-
return normalize_imports(preamble) if normalize else preamble
|
|
48
|
-
parts = [imports_str.strip()]
|
|
49
|
-
if header and header.strip():
|
|
50
|
-
parts.append(header.strip())
|
|
51
|
-
preamble = "\n\n".join(parts)
|
|
52
|
-
return normalize_imports(preamble) if normalize else preamble
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
# ── Theorem signature canonicalization ───────────────────────────────────────
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def _normalize_signature(stmt: str) -> str:
|
|
59
|
-
"""Canonicalize a Lean theorem statement to end with ``:= by``.
|
|
60
|
-
|
|
61
|
-
Strips trailing ``sorry``/``admit`` placeholders and any trailing ``by`` /
|
|
62
|
-
``:=`` tokens, then re-appends `` := by``. Places the appended token on a new
|
|
63
|
-
indented line when the last line of the stripped signature already contains a
|
|
64
|
-
``--`` comment (otherwise the ``:= by`` would land inside the line comment and
|
|
65
|
-
Lean would silently ignore it).
|
|
66
|
-
"""
|
|
67
|
-
s = stmt.rstrip()
|
|
68
|
-
s = re.sub(r"\s*\b(?:sorry|admit)\b\s*$", "", s)
|
|
69
|
-
s = re.sub(r"\s*\bby\b\s*$", "", s)
|
|
70
|
-
s = re.sub(r"\s*:=\s*$", "", s)
|
|
71
|
-
s = s.rstrip()
|
|
72
|
-
last_newline = s.rfind("\n")
|
|
73
|
-
last_line = s[last_newline + 1 :] if last_newline != -1 else s
|
|
74
|
-
sep = "\n " if "--" in last_line else " "
|
|
75
|
-
return s + sep + ":= by"
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
def _split_imports_and_signature(stmt: str) -> tuple[str, str]:
|
|
79
|
-
decl_match = re.search(r"^(?:theorem|lemma|example)\s", stmt, flags=re.MULTILINE)
|
|
80
|
-
if not decl_match:
|
|
81
|
-
return "", stmt
|
|
82
|
-
return stmt[: decl_match.start()].rstrip(), stmt[decl_match.start() :]
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
def expected_protected_signature(formal_statement: str) -> str:
|
|
86
|
-
"""Return the canonical ``theorem ... := by`` block the reward pins as ground truth.
|
|
87
|
-
|
|
88
|
-
Pure function over ``formal_statement``; the imports/header portion (if the
|
|
89
|
-
statement carries one inline) is stripped first.
|
|
90
|
-
"""
|
|
91
|
-
stmt = formal_statement or ""
|
|
92
|
-
if stmt.strip().startswith("import "):
|
|
93
|
-
_, signature_raw = _split_imports_and_signature(stmt)
|
|
94
|
-
else:
|
|
95
|
-
signature_raw = stmt
|
|
96
|
-
return _normalize_signature(signature_raw).strip()
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
# ── Comment / string stripping for the signature guard ───────────────────────
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
def strip_lean_comments(text: str) -> str:
|
|
103
|
-
"""Remove Lean line/block comments **and string literals** from ``text``.
|
|
104
|
-
|
|
105
|
-
Lean comments come in two forms: ``-- ...`` line comments (to end of line) and
|
|
106
|
-
``/- ... -/`` block comments (nestable; ``/-- ... -/`` doc comments are a
|
|
107
|
-
special case). String literals must also be stripped: a Lean
|
|
108
|
-
``"theorem ... := by"`` constant or doc-string would otherwise let a model hide
|
|
109
|
-
the pinned signature inside a string while rewriting the live declaration to a
|
|
110
|
-
trivial one, defeating the substring guard. We handle both regular
|
|
111
|
-
double-quoted strings (with backslash escapes) and triple-quoted raw strings.
|
|
112
|
-
|
|
113
|
-
Block comments nest, so we count depth; line comments end at the next newline.
|
|
114
|
-
Outside comments and strings, newlines are preserved so the result keeps
|
|
115
|
-
roughly the right shape for substring matching downstream.
|
|
116
|
-
"""
|
|
117
|
-
out: list[str] = []
|
|
118
|
-
i = 0
|
|
119
|
-
n = len(text)
|
|
120
|
-
block_depth = 0
|
|
121
|
-
in_line_comment = False
|
|
122
|
-
while i < n:
|
|
123
|
-
ch = text[i]
|
|
124
|
-
if in_line_comment:
|
|
125
|
-
if ch == "\n":
|
|
126
|
-
in_line_comment = False
|
|
127
|
-
out.append(ch)
|
|
128
|
-
i += 1
|
|
129
|
-
continue
|
|
130
|
-
if block_depth > 0:
|
|
131
|
-
if i + 1 < n and text[i : i + 2] == "-/":
|
|
132
|
-
block_depth -= 1
|
|
133
|
-
i += 2
|
|
134
|
-
continue
|
|
135
|
-
if i + 1 < n and text[i : i + 2] == "/-":
|
|
136
|
-
block_depth += 1
|
|
137
|
-
i += 2
|
|
138
|
-
continue
|
|
139
|
-
if ch == "\n":
|
|
140
|
-
out.append(ch)
|
|
141
|
-
i += 1
|
|
142
|
-
continue
|
|
143
|
-
if i + 1 < n and text[i : i + 2] == "/-":
|
|
144
|
-
block_depth = 1
|
|
145
|
-
i += 2
|
|
146
|
-
continue
|
|
147
|
-
if i + 1 < n and text[i : i + 2] == "--":
|
|
148
|
-
in_line_comment = True
|
|
149
|
-
i += 2
|
|
150
|
-
continue
|
|
151
|
-
# Triple-quoted raw string ``"""..."""`` — skip until the closing
|
|
152
|
-
# triple-quote, preserving newlines.
|
|
153
|
-
if i + 2 < n and text[i : i + 3] == '"""':
|
|
154
|
-
i += 3
|
|
155
|
-
while i < n:
|
|
156
|
-
if i + 2 < n and text[i : i + 3] == '"""':
|
|
157
|
-
i += 3
|
|
158
|
-
break
|
|
159
|
-
if text[i] == "\n":
|
|
160
|
-
out.append("\n")
|
|
161
|
-
i += 1
|
|
162
|
-
continue
|
|
163
|
-
# Regular string ``"..."`` — handle ``\\`` escapes, stop at the next
|
|
164
|
-
# unescaped quote or newline (Lean strings are single-line).
|
|
165
|
-
if ch == '"':
|
|
166
|
-
i += 1
|
|
167
|
-
while i < n and text[i] != '"' and text[i] != "\n":
|
|
168
|
-
if text[i] == "\\" and i + 1 < n:
|
|
169
|
-
i += 2
|
|
170
|
-
else:
|
|
171
|
-
i += 1
|
|
172
|
-
if i < n and text[i] == '"':
|
|
173
|
-
i += 1
|
|
174
|
-
continue
|
|
175
|
-
out.append(ch)
|
|
176
|
-
i += 1
|
|
177
|
-
return "".join(out)
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
def protected_signature_substring_present(
|
|
181
|
-
content: str, expected_signature: str
|
|
182
|
-
) -> bool:
|
|
183
|
-
"""True when the locked signature text still appears in the file.
|
|
184
|
-
|
|
185
|
-
Strips Lean comments from BOTH sides — without that a model could paste the
|
|
186
|
-
pinned signature into a ``--`` or ``/- ... -/`` block while rewriting the live
|
|
187
|
-
declaration to something trivial, and the asymmetric variant would also misfire
|
|
188
|
-
if ``expected_signature`` itself contains a comment. Tries an exact substring
|
|
189
|
-
match first, then a whitespace-flexible match (each side collapsed to
|
|
190
|
-
single-spaced tokens) so the model can re-indent or reflow whitespace freely.
|
|
191
|
-
"""
|
|
192
|
-
if not expected_signature:
|
|
193
|
-
return True
|
|
194
|
-
decommented_content = strip_lean_comments(content)
|
|
195
|
-
decommented_expected = strip_lean_comments(expected_signature)
|
|
196
|
-
if not decommented_expected.strip():
|
|
197
|
-
return True
|
|
198
|
-
if decommented_expected in decommented_content:
|
|
199
|
-
return True
|
|
200
|
-
flat_signature = " ".join(decommented_expected.split())
|
|
201
|
-
flat_content = " ".join(decommented_content.split())
|
|
202
|
-
return flat_signature in flat_content
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
# ── Starter-file construction ────────────────────────────────────────────────
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
def build_starter_file(
|
|
209
|
-
formal_statement: str,
|
|
210
|
-
header: str = "",
|
|
211
|
-
imports: str = "import Mathlib",
|
|
212
|
-
normalize: bool = False,
|
|
213
|
-
proof_body: str | None = None,
|
|
214
|
-
) -> str:
|
|
215
|
-
"""Construct the starter proof file.
|
|
216
|
-
|
|
217
|
-
Layout: preamble (imports / header) + a brief ``-- DO NOT MODIFY`` comment
|
|
218
|
-
block + the normalized theorem signature + the proof body. If ``proof_body`` is
|
|
219
|
-
None (the default) the body is the placeholder `` sorry`` — what's planted at
|
|
220
|
-
rollout start. A supplied gold ``proof_body`` (e.g. by ``validate``) replaces
|
|
221
|
-
the placeholder so the file is the full reference solution.
|
|
222
|
-
"""
|
|
223
|
-
stmt = formal_statement or ""
|
|
224
|
-
if stmt.strip().startswith("import "):
|
|
225
|
-
imports_block, signature_raw = _split_imports_and_signature(stmt)
|
|
226
|
-
preamble = normalize_imports(imports_block) if normalize else imports_block
|
|
227
|
-
else:
|
|
228
|
-
preamble = _build_preamble(imports or "import Mathlib", header, normalize)
|
|
229
|
-
signature_raw = stmt
|
|
230
|
-
|
|
231
|
-
signature = _normalize_signature(signature_raw)
|
|
232
|
-
body = " sorry" if proof_body is None else proof_body.rstrip()
|
|
233
|
-
wrapped = f"{PROTECTED_HEADER_COMMENT}\n{signature}\n{body}\n"
|
|
234
|
-
if preamble:
|
|
235
|
-
return preamble.rstrip() + "\n\n" + wrapped
|
|
236
|
-
return wrapped
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
# ── Compile-output parsing ───────────────────────────────────────────────────
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
def parse_compile_output(output: str) -> tuple[bool, str, int]:
|
|
243
|
-
"""Parse a ``lake env lean ...; echo EXIT_CODE:$?`` transcript.
|
|
244
|
-
|
|
245
|
-
Returns ``(compiled, cleaned_output, exit_code)`` where ``compiled`` is True iff
|
|
246
|
-
the compiler exited 0 with no ``declaration uses 'sorry'`` diagnostic.
|
|
247
|
-
|
|
248
|
-
Matches the LAST ``EXIT_CODE:N`` — that's the one our shell appends at the end
|
|
249
|
-
of the command. Matching the first occurrence would let a model inject
|
|
250
|
-
``#eval IO.println "EXIT_CODE:0"`` into the proof file to bypass the
|
|
251
|
-
sorry/exit-code checks: the regex would hit the injected marker, truncate
|
|
252
|
-
everything after it (hiding the real ``declaration uses 'sorry'`` diagnostic and
|
|
253
|
-
the real EXIT_CODE), and report success.
|
|
254
|
-
"""
|
|
255
|
-
exit_code = 1
|
|
256
|
-
matches = list(re.finditer(r"EXIT_CODE:(\d+)", output))
|
|
257
|
-
if matches:
|
|
258
|
-
last = matches[-1]
|
|
259
|
-
exit_code = int(last.group(1))
|
|
260
|
-
output = output[: last.start()].strip()
|
|
261
|
-
has_sorry = bool(re.search(r"declaration uses 'sorry'", output))
|
|
262
|
-
return (exit_code == 0 and not has_sorry), output, exit_code
|
|
@@ -1,229 +0,0 @@
|
|
|
1
|
-
"""Lean 4 theorem-proving tasks backed by Hugging Face datasets.
|
|
2
|
-
|
|
3
|
-
Each task plants a ``sorry``-based starter file, lets any container-capable
|
|
4
|
-
harness edit it, and rewards a clean ``lake env lean`` compile. The original
|
|
5
|
-
theorem signature must remain present, which prevents replacing the assigned
|
|
6
|
-
statement with an easier theorem. Dataset-specific packages can subclass this
|
|
7
|
-
taskset and only supply column mappings.
|
|
8
|
-
"""
|
|
9
|
-
|
|
10
|
-
from __future__ import annotations
|
|
11
|
-
|
|
12
|
-
import shlex
|
|
13
|
-
from collections.abc import Iterator
|
|
14
|
-
|
|
15
|
-
from pydantic_config import BaseConfig
|
|
16
|
-
|
|
17
|
-
from verifiers.v1.configs.task import TaskConfig
|
|
18
|
-
from verifiers.v1.configs.taskset import TasksetConfig
|
|
19
|
-
from verifiers.v1.runtimes import Runtime
|
|
20
|
-
from verifiers.v1.state import State
|
|
21
|
-
from verifiers.v1.task import Task, TaskData, TaskResources
|
|
22
|
-
from verifiers.v1.taskset import Taskset
|
|
23
|
-
from verifiers.v1.tasksets.lean.scoring import (
|
|
24
|
-
build_starter_file,
|
|
25
|
-
expected_protected_signature,
|
|
26
|
-
parse_compile_output,
|
|
27
|
-
protected_signature_substring_present,
|
|
28
|
-
)
|
|
29
|
-
from verifiers.v1.trace import Trace
|
|
30
|
-
from verifiers.v1.utils.decorators import reward
|
|
31
|
-
|
|
32
|
-
# Lean v4.27 with Mathlib v4.27.
|
|
33
|
-
DEFAULT_DOCKER_IMAGE = "team-clyvldofb0000gg1kx39rgzjq/lean-tactic:mathlib-v4.27.0-v3"
|
|
34
|
-
LEAN_PROJECT_PATH = "/workspace/mathlib4"
|
|
35
|
-
PROOF_FILE_PATH = "/tmp/proof.lean"
|
|
36
|
-
|
|
37
|
-
DEFAULT_SYSTEM_PROMPT = "You are an expert Lean 4 theorem prover working with Mathlib."
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
class LeanDatasetConfig(BaseConfig):
|
|
41
|
-
name: str
|
|
42
|
-
"""HuggingFace dataset id (required; each per-dataset package sets it)."""
|
|
43
|
-
split: str = "train"
|
|
44
|
-
subset: str | None = None
|
|
45
|
-
statement_column: str = "formal_statement"
|
|
46
|
-
header_column: str | None = None
|
|
47
|
-
imports_column: str | None = None
|
|
48
|
-
name_column: str | None = None
|
|
49
|
-
proof_column: str | None = None
|
|
50
|
-
"""Column holding the gold proof body (used by ``validate``); None = no gold."""
|
|
51
|
-
normalize_mathlib_imports: bool = False
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
class LeanTaskConfig(TaskConfig):
|
|
55
|
-
lean_project_path: str = LEAN_PROJECT_PATH
|
|
56
|
-
proof_file_path: str = PROOF_FILE_PATH
|
|
57
|
-
compile_timeout: int = 300
|
|
58
|
-
"""Per-compile ``timeout`` wrapper (seconds), bounding each ``lake env lean``."""
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
class LeanConfig(TasksetConfig):
|
|
62
|
-
dataset: LeanDatasetConfig
|
|
63
|
-
docker_image: str = DEFAULT_DOCKER_IMAGE
|
|
64
|
-
task: LeanTaskConfig = LeanTaskConfig()
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
class LeanData(TaskData):
|
|
68
|
-
formal_statement: str
|
|
69
|
-
header: str = ""
|
|
70
|
-
imports: str = "import Mathlib"
|
|
71
|
-
normalize_mathlib_imports: bool = False
|
|
72
|
-
# Canonical ``theorem ... := by`` text pinned at load; the reward checks it
|
|
73
|
-
# still appears in the final file (the only edit the reward cares about).
|
|
74
|
-
protected_signature: str = ""
|
|
75
|
-
# Gold proof body (replaces `` sorry``); "" when the dataset ships no gold.
|
|
76
|
-
formal_proof: str = ""
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
class LeanTask(Task[LeanData, State, LeanTaskConfig]):
|
|
80
|
-
NEEDS_CONTAINER = True
|
|
81
|
-
|
|
82
|
-
async def _compile(self, runtime: Runtime) -> tuple[bool, str, int]:
|
|
83
|
-
cmd = (
|
|
84
|
-
f"cd {shlex.quote(self.config.lean_project_path)} && "
|
|
85
|
-
f"timeout {self.config.compile_timeout} lake env lean "
|
|
86
|
-
f"{shlex.quote(self.config.proof_file_path)} 2>&1; "
|
|
87
|
-
"echo EXIT_CODE:$?"
|
|
88
|
-
)
|
|
89
|
-
result = await runtime.run(["bash", "-lc", cmd], {})
|
|
90
|
-
return parse_compile_output((result.stdout or "") + (result.stderr or ""))
|
|
91
|
-
|
|
92
|
-
async def setup(self, runtime: Runtime) -> None:
|
|
93
|
-
content = build_starter_file(
|
|
94
|
-
self.data.formal_statement,
|
|
95
|
-
header=self.data.header,
|
|
96
|
-
imports=self.data.imports,
|
|
97
|
-
normalize=self.data.normalize_mathlib_imports,
|
|
98
|
-
)
|
|
99
|
-
await runtime.write(self.config.proof_file_path, content.encode())
|
|
100
|
-
|
|
101
|
-
@reward(weight=1.0)
|
|
102
|
-
async def lean_compiled(self, trace: Trace, runtime: Runtime) -> float:
|
|
103
|
-
"""Require both the assigned signature and a clean Lean compile."""
|
|
104
|
-
if trace.has_error:
|
|
105
|
-
return 0.0
|
|
106
|
-
|
|
107
|
-
# Setup created this file, so a read failure is an infrastructure error.
|
|
108
|
-
current = (await runtime.read(self.config.proof_file_path)).decode(
|
|
109
|
-
"utf-8", "replace"
|
|
110
|
-
)
|
|
111
|
-
|
|
112
|
-
expected_sig = self.data.protected_signature or expected_protected_signature(
|
|
113
|
-
self.data.formal_statement
|
|
114
|
-
)
|
|
115
|
-
if expected_sig and not protected_signature_substring_present(
|
|
116
|
-
current, expected_sig
|
|
117
|
-
):
|
|
118
|
-
trace.info["lean_tampered"] = True
|
|
119
|
-
trace.info["compile_output"] = "signature rewritten or hidden in a comment"
|
|
120
|
-
return 0.0
|
|
121
|
-
trace.info["lean_tampered"] = False
|
|
122
|
-
|
|
123
|
-
compiled, output, exit_code = await self._compile(runtime)
|
|
124
|
-
trace.info["lean_compiled"] = compiled
|
|
125
|
-
trace.info["compile_exit_code"] = exit_code
|
|
126
|
-
trace.info["compile_output"] = output[-4000:]
|
|
127
|
-
return 1.0 if compiled else 0.0
|
|
128
|
-
|
|
129
|
-
async def validate(self, runtime: Runtime) -> bool | None:
|
|
130
|
-
"""Compile the gold proof; rows without one have nothing to preflight."""
|
|
131
|
-
gold = (self.data.formal_proof or "").rstrip()
|
|
132
|
-
if not gold:
|
|
133
|
-
return None
|
|
134
|
-
content = build_starter_file(
|
|
135
|
-
self.data.formal_statement,
|
|
136
|
-
header=self.data.header,
|
|
137
|
-
imports=self.data.imports,
|
|
138
|
-
normalize=self.data.normalize_mathlib_imports,
|
|
139
|
-
proof_body=gold,
|
|
140
|
-
)
|
|
141
|
-
await runtime.write(self.config.proof_file_path, content.encode())
|
|
142
|
-
compiled, _, _ = await self._compile(runtime)
|
|
143
|
-
return compiled
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
class LeanTaskset(Taskset[LeanTask, LeanConfig]):
|
|
147
|
-
def load(self) -> Iterator[LeanTask]:
|
|
148
|
-
try:
|
|
149
|
-
from datasets import load_dataset
|
|
150
|
-
except ModuleNotFoundError as e:
|
|
151
|
-
raise ModuleNotFoundError(
|
|
152
|
-
"the Lean taskset requires the `lean` extra; install `verifiers[lean]`"
|
|
153
|
-
) from e
|
|
154
|
-
|
|
155
|
-
config = self.config
|
|
156
|
-
ds = config.dataset
|
|
157
|
-
raw = load_dataset(
|
|
158
|
-
ds.name,
|
|
159
|
-
ds.subset,
|
|
160
|
-
split=ds.split,
|
|
161
|
-
num_proc=8,
|
|
162
|
-
)
|
|
163
|
-
|
|
164
|
-
# Validate optional columns before empty-value fallbacks hide a typo.
|
|
165
|
-
for label, col in (
|
|
166
|
-
("statement_column", ds.statement_column),
|
|
167
|
-
("header_column", ds.header_column),
|
|
168
|
-
("imports_column", ds.imports_column),
|
|
169
|
-
("name_column", ds.name_column),
|
|
170
|
-
("proof_column", ds.proof_column),
|
|
171
|
-
):
|
|
172
|
-
if col is not None and col not in raw.column_names:
|
|
173
|
-
raise ValueError(
|
|
174
|
-
f"dataset.{label}={col!r} not found in {ds.name!r}; columns={raw.column_names}"
|
|
175
|
-
)
|
|
176
|
-
|
|
177
|
-
resources = TaskResources(cpu=4, memory=4, disk=10)
|
|
178
|
-
for index, row in enumerate(raw):
|
|
179
|
-
# An empty statement would reduce the signature guard to `:= by`.
|
|
180
|
-
formal_statement = row[ds.statement_column]
|
|
181
|
-
if not isinstance(formal_statement, str) or not formal_statement.strip():
|
|
182
|
-
continue
|
|
183
|
-
# A disabled optional column is None; row.get(None) yields the same
|
|
184
|
-
# empty fallback as a present-but-empty column.
|
|
185
|
-
header = row.get(ds.header_column) or ""
|
|
186
|
-
imports = row.get(ds.imports_column) or "import Mathlib"
|
|
187
|
-
gold = row.get(ds.proof_column) or ""
|
|
188
|
-
name = row.get(ds.name_column)
|
|
189
|
-
yield LeanTask(
|
|
190
|
-
LeanData(
|
|
191
|
-
idx=index,
|
|
192
|
-
name=str(name) if name else f"task_{index:05d}",
|
|
193
|
-
prompt=self._build_prompt(formal_statement, header),
|
|
194
|
-
system_prompt=DEFAULT_SYSTEM_PROMPT,
|
|
195
|
-
image=config.docker_image,
|
|
196
|
-
workdir=config.task.lean_project_path,
|
|
197
|
-
resources=resources,
|
|
198
|
-
formal_statement=formal_statement,
|
|
199
|
-
header=header,
|
|
200
|
-
imports=imports,
|
|
201
|
-
normalize_mathlib_imports=ds.normalize_mathlib_imports,
|
|
202
|
-
protected_signature=expected_protected_signature(formal_statement),
|
|
203
|
-
formal_proof=gold,
|
|
204
|
-
),
|
|
205
|
-
self.config.task,
|
|
206
|
-
)
|
|
207
|
-
|
|
208
|
-
def _build_prompt(self, formal_statement: str, header: str) -> str:
|
|
209
|
-
cfg = self.config
|
|
210
|
-
block = (
|
|
211
|
-
"Prove the following Lean 4 theorem. A starter proof file is at "
|
|
212
|
-
f"`{cfg.task.proof_file_path}` with the theorem statement and a `sorry` "
|
|
213
|
-
"placeholder already in place. Edit it and compile with "
|
|
214
|
-
f"`cd {cfg.task.lean_project_path} && lake env lean {cfg.task.proof_file_path}`.\n\n"
|
|
215
|
-
f"```lean\n{formal_statement}\n```"
|
|
216
|
-
)
|
|
217
|
-
if header:
|
|
218
|
-
block += f"\n\nThe file header (imports/namespaces) is already set up:\n```lean\n{header}\n```"
|
|
219
|
-
block += (
|
|
220
|
-
"\n\nDo NOT modify the theorem statement (the lines from `theorem ...` "
|
|
221
|
-
"through `:= by`) — the grader checks the original statement still "
|
|
222
|
-
"appears and gives zero reward if you rewrote it. Write your proof "
|
|
223
|
-
"tactics in place of `sorry`; the final proof must not contain `sorry` "
|
|
224
|
-
"or `admit`. A clean compile prints nothing and exits 0."
|
|
225
|
-
)
|
|
226
|
-
return block
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
__all__ = ["LeanConfig", "LeanTask", "LeanTaskset"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|