verifiers 0.2.2.dev73__py3-none-any.whl → 0.2.2.dev75__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/agent.py +2 -8
- verifiers/v1/cli/dashboard/eval.py +6 -4
- verifiers/v1/envs/agentic_judge/env.py +2 -1
- verifiers/v1/harness.py +8 -1
- verifiers/v1/runtimes/__init__.py +19 -0
- verifiers/v1/runtimes/base.py +36 -2
- verifiers/v1/runtimes/docker/__init__.py +1 -1
- verifiers/v1/runtimes/modal.py +1 -1
- verifiers/v1/runtimes/prime.py +1 -1
- verifiers/v1/runtimes/subprocess.py +1 -1
- verifiers/v1/task.py +21 -11
- verifiers/v1/tasksets/harbor/__init__.py +9 -1
- verifiers/v1/tasksets/harbor/env.py +124 -0
- verifiers/v1/tasksets/harbor/taskset.py +247 -51
- verifiers/v1/trace.py +6 -5
- verifiers/v1/utils/decorators.py +17 -1
- verifiers/v1/utils/platform.py +11 -3
- {verifiers-0.2.2.dev73.dist-info → verifiers-0.2.2.dev75.dist-info}/METADATA +1 -1
- {verifiers-0.2.2.dev73.dist-info → verifiers-0.2.2.dev75.dist-info}/RECORD +22 -21
- {verifiers-0.2.2.dev73.dist-info → verifiers-0.2.2.dev75.dist-info}/WHEEL +0 -0
- {verifiers-0.2.2.dev73.dist-info → verifiers-0.2.2.dev75.dist-info}/entry_points.txt +0 -0
- {verifiers-0.2.2.dev73.dist-info → verifiers-0.2.2.dev75.dist-info}/licenses/LICENSE +0 -0
verifiers/v1/agent.py
CHANGED
|
@@ -29,7 +29,7 @@ from verifiers.v1.runtimes import (
|
|
|
29
29
|
Runtime,
|
|
30
30
|
RuntimeConfig,
|
|
31
31
|
SubprocessConfig,
|
|
32
|
-
|
|
32
|
+
provision_runtime,
|
|
33
33
|
runtime_is_local,
|
|
34
34
|
)
|
|
35
35
|
from verifiers.v1.session import RolloutLimits
|
|
@@ -547,14 +547,8 @@ class Agent:
|
|
|
547
547
|
if task is not None
|
|
548
548
|
else self.runtime_config
|
|
549
549
|
)
|
|
550
|
-
|
|
551
|
-
try:
|
|
552
|
-
# start() inside the try: a failed start may already hold a remote
|
|
553
|
-
# sandbox, so it must reach stop() (safe on a partially-started runtime).
|
|
554
|
-
await runtime.start()
|
|
550
|
+
async with provision_runtime(config) as runtime:
|
|
555
551
|
yield runtime
|
|
556
|
-
finally:
|
|
557
|
-
await runtime.stop()
|
|
558
552
|
|
|
559
553
|
|
|
560
554
|
class _EpisodeAgent(Agent):
|
|
@@ -332,8 +332,8 @@ def Progress(
|
|
|
332
332
|
|
|
333
333
|
def _score_segments(traces: list[Trace], source: str) -> str | None:
|
|
334
334
|
"""`name mean` segments for every reward/metric key seen across `traces`,
|
|
335
|
-
first-seen order (a trace
|
|
336
|
-
can vary); None when no trace recorded anything."""
|
|
335
|
+
first-seen order (a trace carries only its own task's signals — unscored ones
|
|
336
|
+
seeded as `None` — so keys can vary); None when no trace recorded anything."""
|
|
337
337
|
names: list[str] = []
|
|
338
338
|
for trace in traces:
|
|
339
339
|
names.extend(n for n in getattr(trace, source) if n not in names)
|
|
@@ -347,11 +347,13 @@ def _score_segments(traces: list[Trace], source: str) -> str | None:
|
|
|
347
347
|
|
|
348
348
|
|
|
349
349
|
def _score(trace: Trace, source: str, name: str) -> float:
|
|
350
|
-
"""Rewards carry raw score + weight; the breakdown shows the raw score.
|
|
350
|
+
"""Rewards carry raw score + weight; the breakdown shows the raw score. A missing
|
|
351
|
+
or unscored (seeded `None`) entry reads as 0.0."""
|
|
351
352
|
if source == "rewards":
|
|
352
353
|
reward = trace.rewards.get(name)
|
|
353
354
|
return reward.score if reward is not None else 0.0
|
|
354
|
-
|
|
355
|
+
value = trace.metrics.get(name)
|
|
356
|
+
return value if value is not None else 0.0
|
|
355
357
|
|
|
356
358
|
|
|
357
359
|
def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
@@ -398,7 +398,8 @@ class AgenticJudgeEnv(vf.Env[AgenticJudgeEnvConfig]):
|
|
|
398
398
|
solution.record_metric(f"judge/{criterion.name}", scores[criterion.name])
|
|
399
399
|
if self.config.score.task_weight != 1.0:
|
|
400
400
|
for reward in solution.rewards.values():
|
|
401
|
-
reward
|
|
401
|
+
if reward is not None:
|
|
402
|
+
reward.weight *= self.config.score.task_weight
|
|
402
403
|
total = sum(criterion.weight for criterion in criteria)
|
|
403
404
|
reward = sum(c.weight * scores[c.name] for c in criteria) / total
|
|
404
405
|
solution.record_reward("judge", reward, weight=self.config.score.judge_weight)
|
verifiers/v1/harness.py
CHANGED
|
@@ -12,7 +12,12 @@ from verifiers.v1.errors import HarnessError, SandboxError, boundary
|
|
|
12
12
|
from verifiers.v1.runtimes import ProgramResult, Runtime
|
|
13
13
|
from verifiers.v1.task import TaskData
|
|
14
14
|
from verifiers.v1.types import Messages
|
|
15
|
-
from verifiers.v1.utils.decorators import
|
|
15
|
+
from verifiers.v1.utils.decorators import (
|
|
16
|
+
discover_decorated,
|
|
17
|
+
invoke_all,
|
|
18
|
+
seed,
|
|
19
|
+
unseed,
|
|
20
|
+
)
|
|
16
21
|
|
|
17
22
|
if TYPE_CHECKING:
|
|
18
23
|
# Annotation-only: `Trace` appears in signatures only, so this module stays
|
|
@@ -155,10 +160,12 @@ class Harness(ABC, Generic[ConfigT]):
|
|
|
155
160
|
`runtime`) and can read what the harness left behind in the runtime."""
|
|
156
161
|
available = {"task": trace.task.data, "trace": trace, "runtime": runtime}
|
|
157
162
|
fns = discover_decorated(self, "metric")
|
|
163
|
+
seed(trace.metrics, (fn.__name__ for fn in fns))
|
|
158
164
|
async with boundary(HarnessError, f"harness {self.config.id!r} metric"):
|
|
159
165
|
results = await invoke_all(fns, available)
|
|
160
166
|
for fn, result in zip(fns, results):
|
|
161
167
|
if isinstance(result, Mapping):
|
|
168
|
+
unseed(trace.metrics, fn.__name__)
|
|
162
169
|
trace.record_metrics(result)
|
|
163
170
|
else:
|
|
164
171
|
trace.record_metric(fn.__name__, result)
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
from collections.abc import AsyncIterator
|
|
2
|
+
from contextlib import asynccontextmanager
|
|
1
3
|
from typing import Annotated
|
|
2
4
|
|
|
3
5
|
from pydantic import Field
|
|
@@ -45,6 +47,22 @@ def make_runtime(config: RuntimeConfig, name: str | None = None) -> Runtime:
|
|
|
45
47
|
return runtime
|
|
46
48
|
|
|
47
49
|
|
|
50
|
+
@asynccontextmanager
|
|
51
|
+
async def provision_runtime(
|
|
52
|
+
config: RuntimeConfig, name: str | None = None
|
|
53
|
+
) -> AsyncIterator[Runtime]:
|
|
54
|
+
"""Provision a box from `config` and tear it down on exit.
|
|
55
|
+
|
|
56
|
+
`start()` sits inside the `try`: a failed start may already hold a paid sandbox, so
|
|
57
|
+
it has to reach `stop()` (which is safe on a partially-started runtime)."""
|
|
58
|
+
runtime = make_runtime(config, name)
|
|
59
|
+
try:
|
|
60
|
+
await runtime.start()
|
|
61
|
+
yield runtime
|
|
62
|
+
finally:
|
|
63
|
+
await runtime.stop()
|
|
64
|
+
|
|
65
|
+
|
|
48
66
|
def runtime_is_local(config: RuntimeConfig) -> bool:
|
|
49
67
|
"""Whether a runtime of this config exchanges host-local URLs without a public
|
|
50
68
|
tunnel, read off the runtime class without provisioning one."""
|
|
@@ -71,5 +89,6 @@ __all__ = [
|
|
|
71
89
|
"SubprocessRuntime",
|
|
72
90
|
"SubprocessRuntimeInfo",
|
|
73
91
|
"make_runtime",
|
|
92
|
+
"provision_runtime",
|
|
74
93
|
"runtime_is_local",
|
|
75
94
|
]
|
verifiers/v1/runtimes/base.py
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import atexit
|
|
5
|
+
import base64
|
|
5
6
|
import contextlib
|
|
6
7
|
import hashlib
|
|
7
8
|
import logging
|
|
@@ -285,9 +286,42 @@ class Runtime(ABC):
|
|
|
285
286
|
argv = await self.prepare_uv_script(script, env)
|
|
286
287
|
return await self.run([*argv, *(args or [])], env or {})
|
|
287
288
|
|
|
289
|
+
async def read(self, path: str, max_bytes: int | None = None) -> bytes:
|
|
290
|
+
"""Read `path` into host memory. `max_bytes` caps the transfer, raising past
|
|
291
|
+
the cap — for a file written by something we don't control, whose size we
|
|
292
|
+
can't assume. The cap is enforced inside the box rather than after the
|
|
293
|
+
transfer, and base64 because `run` returns decoded text. Framework method —
|
|
294
|
+
override `_read`, not this."""
|
|
295
|
+
if max_bytes is None:
|
|
296
|
+
return await self._read(path)
|
|
297
|
+
# Through a temp file, not a pipe: `head | base64` exits with base64's 0
|
|
298
|
+
# even when the path is missing, and a missing file must raise here just
|
|
299
|
+
# as it does from `_read`.
|
|
300
|
+
result = await self.run(
|
|
301
|
+
[
|
|
302
|
+
"sh",
|
|
303
|
+
"-c",
|
|
304
|
+
(
|
|
305
|
+
"t=$(mktemp) || exit 1; "
|
|
306
|
+
'head -c "$1" -- "$2" > "$t" || { rm -f "$t"; exit 1; }; '
|
|
307
|
+
'base64 < "$t"; rc=$?; rm -f "$t"; exit $rc'
|
|
308
|
+
),
|
|
309
|
+
"sh",
|
|
310
|
+
str(max_bytes + 1),
|
|
311
|
+
path,
|
|
312
|
+
],
|
|
313
|
+
{},
|
|
314
|
+
)
|
|
315
|
+
if result.exit_code:
|
|
316
|
+
raise SandboxError(f"read {path!r}: {result.stderr.strip()[-500:]}")
|
|
317
|
+
data = base64.b64decode(result.stdout)
|
|
318
|
+
if len(data) > max_bytes:
|
|
319
|
+
raise SandboxError(f"read {path!r}: over the {max_bytes} byte limit")
|
|
320
|
+
return data
|
|
321
|
+
|
|
288
322
|
@abstractmethod
|
|
289
|
-
async def
|
|
290
|
-
|
|
323
|
+
async def _read(self, path: str) -> bytes:
|
|
324
|
+
"""Read the whole file at `path`; `read` adds the optional transfer cap."""
|
|
291
325
|
|
|
292
326
|
@abstractmethod
|
|
293
327
|
async def write(self, path: str, data: bytes) -> None:
|
|
@@ -334,7 +334,7 @@ class DockerRuntime(Runtime):
|
|
|
334
334
|
if run.exit_code != 0:
|
|
335
335
|
raise SandboxError(f"docker exec -d failed: {run.stderr.strip()}")
|
|
336
336
|
|
|
337
|
-
async def
|
|
337
|
+
async def _read(self, path: str) -> bytes:
|
|
338
338
|
proc = await asyncio.create_subprocess_exec(
|
|
339
339
|
"docker",
|
|
340
340
|
"exec",
|
verifiers/v1/runtimes/modal.py
CHANGED
|
@@ -165,7 +165,7 @@ class ModalRuntime(Runtime):
|
|
|
165
165
|
return path
|
|
166
166
|
return f"{self.config.workdir.rstrip('/')}/{path}"
|
|
167
167
|
|
|
168
|
-
async def
|
|
168
|
+
async def _read(self, path: str) -> bytes:
|
|
169
169
|
try:
|
|
170
170
|
return await self._sandbox.filesystem.read_bytes.aio(self._abs(path))
|
|
171
171
|
except Exception as e:
|
verifiers/v1/runtimes/prime.py
CHANGED
|
@@ -270,7 +270,7 @@ class PrimeRuntime(Runtime):
|
|
|
270
270
|
f"prime background launch failed: {result.stderr.strip()}"
|
|
271
271
|
)
|
|
272
272
|
|
|
273
|
-
async def
|
|
273
|
+
async def _read(self, path: str) -> bytes:
|
|
274
274
|
# Avoid background-job log limits and base64 overhead by downloading binary data directly.
|
|
275
275
|
# The temporary file is removed on every exit, and its byte read stays off the event loop.
|
|
276
276
|
target = (
|
|
@@ -96,7 +96,7 @@ class SubprocessRuntime(Runtime):
|
|
|
96
96
|
proc
|
|
97
97
|
) # killed in stop() — a host process won't die on its own
|
|
98
98
|
|
|
99
|
-
async def
|
|
99
|
+
async def _read(self, path: str) -> bytes:
|
|
100
100
|
return await asyncio.to_thread((self.workdir / path).read_bytes)
|
|
101
101
|
|
|
102
102
|
async def write(self, path: str, data: bytes) -> None:
|
verifiers/v1/task.py
CHANGED
|
@@ -23,7 +23,12 @@ from verifiers.v1.errors import TaskError, boundary
|
|
|
23
23
|
from verifiers.v1.state import StateT
|
|
24
24
|
from verifiers.v1.types import Messages, content_text
|
|
25
25
|
from verifiers.v1.utils.artifacts import Artifact
|
|
26
|
-
from verifiers.v1.utils.decorators import
|
|
26
|
+
from verifiers.v1.utils.decorators import (
|
|
27
|
+
discover_decorated,
|
|
28
|
+
invoke_all,
|
|
29
|
+
seed,
|
|
30
|
+
unseed,
|
|
31
|
+
)
|
|
27
32
|
from verifiers.v1.utils.generic import concrete_type
|
|
28
33
|
|
|
29
34
|
if TYPE_CHECKING:
|
|
@@ -180,31 +185,36 @@ class Task(Generic[DataT, StateT, ConfigT]):
|
|
|
180
185
|
judge for judge in judges if not requires_runtime(judge.score)
|
|
181
186
|
]
|
|
182
187
|
|
|
188
|
+
seed(trace.metrics, (fn.__name__ for fn in metrics))
|
|
189
|
+
seed(trace.rewards, (fn.__name__ for fn in rewards))
|
|
190
|
+
seed(trace.rewards, (judge.reward_name for judge in judges))
|
|
191
|
+
|
|
183
192
|
metric_results = await invoke_all(metrics, available)
|
|
184
193
|
for fn, result in zip(metrics, metric_results):
|
|
185
194
|
if isinstance(result, Mapping):
|
|
195
|
+
unseed(trace.metrics, fn.__name__)
|
|
186
196
|
trace.record_metrics(result)
|
|
187
197
|
else:
|
|
188
198
|
trace.record_metric(fn.__name__, result)
|
|
189
199
|
reward_results = await invoke_all(rewards, available)
|
|
190
200
|
for fn, result in zip(rewards, reward_results):
|
|
191
201
|
weight = getattr(fn, "_vf_weight", 1.0)
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
202
|
+
if isinstance(result, Mapping):
|
|
203
|
+
unseed(trace.rewards, fn.__name__)
|
|
204
|
+
items = result.items()
|
|
205
|
+
else:
|
|
206
|
+
items = [(fn.__name__, result)]
|
|
197
207
|
for key, value in items:
|
|
198
208
|
trace.record_reward(key, value, weight)
|
|
199
209
|
judge_results = await invoke_all(
|
|
200
210
|
[judge.score for judge in judges], available
|
|
201
211
|
)
|
|
202
212
|
for judge, result in zip(judges, judge_results):
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
213
|
+
if isinstance(result, Mapping):
|
|
214
|
+
unseed(trace.rewards, judge.reward_name)
|
|
215
|
+
items = result.items()
|
|
216
|
+
else:
|
|
217
|
+
items = [(judge.reward_name, result)]
|
|
208
218
|
for key, value in items:
|
|
209
219
|
trace.record_reward(key, value, judge.config.weight)
|
|
210
220
|
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
from verifiers.v1.tasksets.harbor.env import HarborEnv, HarborEnvConfig
|
|
1
2
|
from verifiers.v1.tasksets.harbor.taskset import (
|
|
2
3
|
HarborConfig,
|
|
3
4
|
HarborData,
|
|
@@ -5,4 +6,11 @@ from verifiers.v1.tasksets.harbor.taskset import (
|
|
|
5
6
|
HarborTaskset,
|
|
6
7
|
)
|
|
7
8
|
|
|
8
|
-
__all__ = [
|
|
9
|
+
__all__ = [
|
|
10
|
+
"HarborConfig",
|
|
11
|
+
"HarborData",
|
|
12
|
+
"HarborEnv",
|
|
13
|
+
"HarborEnvConfig",
|
|
14
|
+
"HarborTask",
|
|
15
|
+
"HarborTaskset",
|
|
16
|
+
]
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""The harbor taskset's own env: the single solver seat, plus separate-verifier
|
|
2
|
+
grading for tasks that declare ``[verifier].environment_mode = "separate"``.
|
|
3
|
+
|
|
4
|
+
The default env for harbor runs (the taskset package exports it). A shared-verifier
|
|
5
|
+
task runs exactly as under the single-agent env: one `agent` trace, graded in the
|
|
6
|
+
box it worked in. A separate-verifier task is graded by `finalize` instead: the
|
|
7
|
+
solver's declared artifacts travel (collected by its task `finalize` while its box
|
|
8
|
+
is alive), a fresh box is provisioned from the task's verifier declaration,
|
|
9
|
+
`tests/` is staged there, and the verifier's rewards land on the solver's trace.
|
|
10
|
+
No second agent is involved — the verifier is the task's own `tests/test.sh`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import logging
|
|
15
|
+
from contextlib import AsyncExitStack
|
|
16
|
+
|
|
17
|
+
from pydantic import Field
|
|
18
|
+
|
|
19
|
+
import verifiers.v1 as vf
|
|
20
|
+
from verifiers.v1.runtimes import RuntimeConfig, provision_runtime
|
|
21
|
+
from verifiers.v1.tasksets.harbor.taskset import (
|
|
22
|
+
HarborTask,
|
|
23
|
+
verifier_box_data,
|
|
24
|
+
)
|
|
25
|
+
from verifiers.v1.utils.artifacts import restore
|
|
26
|
+
from verifiers.v1.utils.compile import resolve_runtime_config
|
|
27
|
+
from verifiers.v1.utils.retries import backoff
|
|
28
|
+
|
|
29
|
+
logger = logging.getLogger(__name__)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class HarborEnvConfig(vf.EnvConfig):
|
|
33
|
+
agent: vf.AgentConfig = vf.AgentConfig()
|
|
34
|
+
"""The one seat — the policy under evaluation/training; pin
|
|
35
|
+
`--env.agent.harness.*` to choose its program or runtime."""
|
|
36
|
+
verifier_runtime: RuntimeConfig | None = None
|
|
37
|
+
"""Where a separate-verifier task grades. None derives the grading box from
|
|
38
|
+
the solver's runtime policy; set it (e.g. `--env.verifier-runtime.type prime
|
|
39
|
+
--env.verifier-runtime.vm true`) when the verifier needs different placement
|
|
40
|
+
than the agent."""
|
|
41
|
+
verifier_retries: int = Field(2, ge=0)
|
|
42
|
+
"""Extra attempts at provisioning-and-grading the separate box before the
|
|
43
|
+
episode fails. Grading is deterministic; what these retry is the
|
|
44
|
+
infrastructure around it (image pulls, provisioning)."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class HarborEnv(vf.Env[HarborEnvConfig]):
|
|
48
|
+
async def run(self, task: vf.Task, agents: vf.Agents) -> None:
|
|
49
|
+
if not isinstance(task, HarborTask):
|
|
50
|
+
raise TypeError(
|
|
51
|
+
f"the harbor env runs harbor tasks; got {type(task).__name__}"
|
|
52
|
+
)
|
|
53
|
+
if task.data.verifier is None:
|
|
54
|
+
await agents.agent.run(task)
|
|
55
|
+
return
|
|
56
|
+
# Resolve the verifier's box before the solve, so an impossible pairing
|
|
57
|
+
# (e.g. a restricted Prime verifier without vm=true) costs nothing
|
|
58
|
+
# rather than a full agent run.
|
|
59
|
+
self._verifier_config(task)
|
|
60
|
+
await agents.agent.run(task.graded_elsewhere())
|
|
61
|
+
|
|
62
|
+
def _verifier_config(self, task: HarborTask) -> RuntimeConfig:
|
|
63
|
+
base = (
|
|
64
|
+
self.config.verifier_runtime
|
|
65
|
+
if self.config.verifier_runtime is not None
|
|
66
|
+
else self.config.agent.runtime
|
|
67
|
+
)
|
|
68
|
+
return resolve_runtime_config(base, HarborTask(verifier_box_data(task.data)))
|
|
69
|
+
|
|
70
|
+
async def finalize(self, task: vf.Task, episode: vf.Episode) -> None:
|
|
71
|
+
"""Grade a separate-verifier task in its own box, onto the solver's trace.
|
|
72
|
+
|
|
73
|
+
Provision a fresh box from the task's verifier declaration, restore the
|
|
74
|
+
solver's collected artifacts, stage `tests/`, run the verifier, and record
|
|
75
|
+
its rewards (and any extra reward.json keys as metrics) on the solver's
|
|
76
|
+
trace. Infrastructure failures retry per `verifier_retries`; the last one
|
|
77
|
+
fails the episode — a grading box that can't be reached must never read
|
|
78
|
+
as reward 0."""
|
|
79
|
+
if not isinstance(task, HarborTask) or task.data.verifier is None:
|
|
80
|
+
return
|
|
81
|
+
solution = episode.traces[0]
|
|
82
|
+
if not solution.ok:
|
|
83
|
+
return
|
|
84
|
+
grader = HarborTask(verifier_box_data(task.data))
|
|
85
|
+
scores = await self._grade(self._verifier_config(task), grader, solution)
|
|
86
|
+
items = scores.items() if isinstance(scores, dict) else [("solved", scores)]
|
|
87
|
+
for name, value in items:
|
|
88
|
+
solution.record_reward(name, value)
|
|
89
|
+
|
|
90
|
+
async def _grade(
|
|
91
|
+
self, config: RuntimeConfig, grader: HarborTask, solution: vf.Trace
|
|
92
|
+
) -> float | dict[str, float]:
|
|
93
|
+
last: Exception | None = None
|
|
94
|
+
for attempt in range(self.config.verifier_retries + 1):
|
|
95
|
+
if attempt:
|
|
96
|
+
delay = backoff(attempt - 1)
|
|
97
|
+
logger.warning(
|
|
98
|
+
"harbor verifier attempt %d/%d failed (%s); retrying in %.1fs",
|
|
99
|
+
attempt,
|
|
100
|
+
self.config.verifier_retries + 1,
|
|
101
|
+
last,
|
|
102
|
+
delay,
|
|
103
|
+
)
|
|
104
|
+
await asyncio.sleep(delay)
|
|
105
|
+
try:
|
|
106
|
+
# The scoring deadline covers provisioning and grading, but not the
|
|
107
|
+
# box's teardown: a score already in hand must not be discarded
|
|
108
|
+
# because the teardown ran out the clock.
|
|
109
|
+
async with AsyncExitStack() as boxes:
|
|
110
|
+
async with asyncio.timeout(grader.data.timeout.scoring):
|
|
111
|
+
box = await boxes.enter_async_context(provision_runtime(config))
|
|
112
|
+
await box.prepare_setup()
|
|
113
|
+
# Artifacts first, tests second: an artifact entry pointing
|
|
114
|
+
# into /tests must not survive staging, which wipes and
|
|
115
|
+
# rebuilds that directory.
|
|
116
|
+
await restore(box, solution.state.artifacts)
|
|
117
|
+
await grader._stage_tests(box, wipe=True)
|
|
118
|
+
await box.prepare_execution([])
|
|
119
|
+
scores = await grader._graded(box, solution)
|
|
120
|
+
return scores
|
|
121
|
+
except Exception as e: # noqa: BLE001 - each attempt's failure is retried
|
|
122
|
+
last = e
|
|
123
|
+
assert last is not None
|
|
124
|
+
raise last
|
|
@@ -1,18 +1,25 @@
|
|
|
1
1
|
"""Harbor tasksets backed by Harbor Hub packages.
|
|
2
2
|
|
|
3
3
|
The Harbor CLI downloads and caches each task directory. Its verifier runs in the
|
|
4
|
-
|
|
4
|
+
runtime the harness edited — or, when the task asks for it with
|
|
5
|
+
``[verifier].environment_mode = "separate"``, in a second box the agent never
|
|
6
|
+
touched, carrying only what the task declared — the harbor env provisions and
|
|
7
|
+
grades that box (see ``env.py``). Either way the score lands in
|
|
5
8
|
``/logs/verifier/reward.json`` or the legacy ``reward.txt``.
|
|
6
9
|
|
|
7
10
|
A pullable ``[environment].docker_image`` becomes ``TaskData.image``. Verifiers does
|
|
8
11
|
not build Dockerfile-only environments, so those are rejected unless ``ignore_dockerfile``
|
|
9
12
|
deliberately uses the harness runtime image. Tasks without an environment also use that
|
|
10
|
-
image unless ``require_image`` is set.
|
|
13
|
+
image unless ``require_image`` is set. The same rule applies to a declared
|
|
14
|
+
``[verifier.environment]``: it needs a pullable ``docker_image``, since Harbor would
|
|
15
|
+
otherwise build the verifier image from ``tests/Dockerfile``.
|
|
11
16
|
"""
|
|
12
17
|
|
|
13
18
|
import asyncio
|
|
19
|
+
import copy
|
|
14
20
|
import hashlib
|
|
15
21
|
import io
|
|
22
|
+
import logging
|
|
16
23
|
import shutil
|
|
17
24
|
import subprocess
|
|
18
25
|
import sys
|
|
@@ -26,7 +33,7 @@ from typing import Annotated
|
|
|
26
33
|
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
|
|
27
34
|
|
|
28
35
|
from verifiers.v1.configs.taskset import TasksetConfig
|
|
29
|
-
from verifiers.v1.errors import SandboxError
|
|
36
|
+
from verifiers.v1.errors import SandboxError, TaskError
|
|
30
37
|
from verifiers.v1.runtimes import Runtime
|
|
31
38
|
from verifiers.v1.task import Task, TaskData, TaskResources, TaskTimeout
|
|
32
39
|
from verifiers.v1.taskset import Taskset
|
|
@@ -34,9 +41,12 @@ from verifiers.v1.trace import Trace
|
|
|
34
41
|
from verifiers.v1.utils.artifacts import Artifact, collect
|
|
35
42
|
from verifiers.v1.utils.decorators import reward
|
|
36
43
|
|
|
44
|
+
logger = logging.getLogger(__name__)
|
|
45
|
+
|
|
37
46
|
CACHE = Path.home() / ".cache" / "harbor"
|
|
38
47
|
HARBOR_INSTALL_HINT = "uv sync --python 3.12 --extra harbor"
|
|
39
48
|
REWARD_JSON = "/logs/verifier/reward.json"
|
|
49
|
+
MAX_REWARD_BYTES = 1024 * 1024
|
|
40
50
|
REWARD_JSON_ADAPTER = TypeAdapter(
|
|
41
51
|
float | Annotated[dict[str, float], Field(min_length=1)],
|
|
42
52
|
config=ConfigDict(strict=True, allow_inf_nan=False),
|
|
@@ -76,6 +86,11 @@ class HarborConfig(TasksetConfig):
|
|
|
76
86
|
instead of rejecting it. The Dockerfile is NOT built, so the task scores against the
|
|
77
87
|
harness image rather than its declared environment — only correct when that image already
|
|
78
88
|
has what the task needs (e.g. you've pointed the runtime at the right image)."""
|
|
89
|
+
ignore_separate_verifier: bool = False
|
|
90
|
+
"""Grade every task in the agent's own box, even one whose `[verifier]` asks for a
|
|
91
|
+
separate one. Trades the isolation for a sandbox per task; useful when provisioning
|
|
92
|
+
is the bottleneck. Note what it gives up: the grader becomes reachable by the agent
|
|
93
|
+
that just ran."""
|
|
79
94
|
|
|
80
95
|
|
|
81
96
|
class Author(BaseModel):
|
|
@@ -90,6 +105,27 @@ class CollectHook(BaseModel):
|
|
|
90
105
|
timeout_sec: float = 600.0
|
|
91
106
|
|
|
92
107
|
|
|
108
|
+
class VerifierConfig(BaseModel):
|
|
109
|
+
"""The box this task's verifier wants, when it wants one of its own.
|
|
110
|
+
|
|
111
|
+
`None` on `HarborData` means shared — grade where the agent worked, which is still
|
|
112
|
+
Harbor's default and every task that says nothing."""
|
|
113
|
+
|
|
114
|
+
image: str | None = None
|
|
115
|
+
"""Pullable ref from `[verifier.environment].docker_image`. None keeps the task's
|
|
116
|
+
own image, which is what Harbor's fresh copy of `[environment]` resolves to."""
|
|
117
|
+
resources: TaskResources = TaskResources()
|
|
118
|
+
workdir: str | None = None
|
|
119
|
+
fresh_copy: bool = False
|
|
120
|
+
"""Whether this came from Harbor's fresh copy of `[environment]` rather than a
|
|
121
|
+
declared `[verifier.environment]`. A fresh copy inherits the agent box's resolved
|
|
122
|
+
resources; a declared environment states its own, and what it omits falls back to
|
|
123
|
+
the run's rather than to the agent's task-derived values."""
|
|
124
|
+
network_allow: list[str] = Field(default_factory=lambda: ["*"])
|
|
125
|
+
"""Destinations the verifier may reach, from the verifier's network mode. `["*"]`
|
|
126
|
+
is unrestricted; `[]` is Harbor's `no-network` / `allow_internet = false`."""
|
|
127
|
+
|
|
128
|
+
|
|
93
129
|
class HarborData(TaskData):
|
|
94
130
|
"""Parsed ``task.toml`` metadata plus the host-side verifier directory.
|
|
95
131
|
|
|
@@ -111,6 +147,9 @@ class HarborData(TaskData):
|
|
|
111
147
|
collect: list[CollectHook] = Field(default_factory=list)
|
|
112
148
|
"""`[[verifier.collect]]` blocks: commands that snapshot runtime state into files
|
|
113
149
|
after the agent stops, so the files can travel to a grading box as artifacts."""
|
|
150
|
+
verifier: VerifierConfig | None = None
|
|
151
|
+
"""The verifier's own box, when `[verifier].environment_mode` asks for one. None
|
|
152
|
+
grades in the agent's box."""
|
|
114
153
|
|
|
115
154
|
|
|
116
155
|
class HarborTask(Task[HarborData]):
|
|
@@ -145,30 +184,59 @@ class HarborTask(Task[HarborData]):
|
|
|
145
184
|
)
|
|
146
185
|
trace.state.artifacts = await collect(runtime, self.data.artifacts)
|
|
147
186
|
|
|
148
|
-
|
|
149
|
-
|
|
187
|
+
def graded_elsewhere(self) -> "HarborTask":
|
|
188
|
+
"""A copy whose `solved` records nothing here: the harbor env grades this
|
|
189
|
+
task's finished work in a separate box of the task's choosing."""
|
|
190
|
+
clone = copy.copy(self)
|
|
191
|
+
clone._graded_elsewhere = True
|
|
192
|
+
return clone
|
|
193
|
+
|
|
194
|
+
_graded_elsewhere: bool = False
|
|
195
|
+
|
|
196
|
+
async def _stage_tests(self, runtime: Runtime, wipe: bool = False) -> None:
|
|
197
|
+
"""Put the task package's `tests/` in `/tests`, where `test.sh` expects it.
|
|
198
|
+
|
|
199
|
+
Raises rather than scoring stale state: a leftover reward file — planted by
|
|
200
|
+
the agent or shipped in the image — must be gone before `test.sh` runs, so a
|
|
201
|
+
removal that fails must not fall through to reading it.
|
|
202
|
+
|
|
203
|
+
`wipe` for a box we did not watch being built: a fresh container of the task's
|
|
204
|
+
image can ship its own `/tests`, and a leftover file there would be graded as
|
|
205
|
+
though it came from the package.
|
|
206
|
+
"""
|
|
150
207
|
await runtime.write(
|
|
151
208
|
"/tmp/tests.tgz", make_tar(Path(self.data.task_dir) / "tests")
|
|
152
209
|
)
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
"mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests",
|
|
158
|
-
],
|
|
159
|
-
{},
|
|
160
|
-
)
|
|
161
|
-
await runtime.run(
|
|
162
|
-
[
|
|
163
|
-
"sh",
|
|
164
|
-
"-c",
|
|
165
|
-
(
|
|
166
|
-
"rm -f /logs/verifier/reward.json /logs/verifier/reward.txt"
|
|
167
|
-
" && cd /tests && bash test.sh"
|
|
168
|
-
),
|
|
169
|
-
],
|
|
170
|
-
verifier_env(self.data),
|
|
210
|
+
stage = (
|
|
211
|
+
f"{'rm -rf /tests && ' if wipe else ''}"
|
|
212
|
+
"rm -f /logs/verifier/reward.json /logs/verifier/reward.txt && "
|
|
213
|
+
"mkdir -p /logs/verifier /tests && tar -xzf /tmp/tests.tgz -C /tests"
|
|
171
214
|
)
|
|
215
|
+
result = await runtime.run(["sh", "-c", stage], {})
|
|
216
|
+
if result.exit_code:
|
|
217
|
+
raise TaskError(
|
|
218
|
+
f"staging tests failed (exit {result.exit_code}): "
|
|
219
|
+
f"{(result.stderr or result.stdout).strip()[-500:]}"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
@reward(weight=1.0)
|
|
223
|
+
async def solved(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
|
|
224
|
+
if self.data.verifier is not None:
|
|
225
|
+
if not self._graded_elsewhere:
|
|
226
|
+
raise TaskError(
|
|
227
|
+
f"task {self.data.name!r} declares a separate verifier "
|
|
228
|
+
'([verifier].environment_mode = "separate"); grade it through '
|
|
229
|
+
"the harbor env (this taskset's default), or force shared "
|
|
230
|
+
"grading with --taskset.ignore-separate-verifier"
|
|
231
|
+
)
|
|
232
|
+
return {}
|
|
233
|
+
await self._stage_tests(runtime)
|
|
234
|
+
return await self._graded(runtime, trace)
|
|
235
|
+
|
|
236
|
+
async def _graded(self, runtime: Runtime, trace: Trace) -> float | dict[str, float]:
|
|
237
|
+
# By absolute path, in the runtime's configured workdir: Harbor execs the
|
|
238
|
+
# script the same way, and scripts do grade the agent's work at `$PWD`.
|
|
239
|
+
await runtime.run(["bash", "/tests/test.sh"], verifier_env(self.data))
|
|
172
240
|
scores = await self._reward_json(runtime)
|
|
173
241
|
if scores is not None:
|
|
174
242
|
if isinstance(scores, dict) and "reward" in scores:
|
|
@@ -178,19 +246,75 @@ class HarborTask(Task[HarborData]):
|
|
|
178
246
|
return {"reward": scores["reward"]}
|
|
179
247
|
return scores
|
|
180
248
|
try:
|
|
181
|
-
reward = (
|
|
249
|
+
reward = (
|
|
250
|
+
(
|
|
251
|
+
await runtime.read(
|
|
252
|
+
"/logs/verifier/reward.txt", max_bytes=MAX_REWARD_BYTES
|
|
253
|
+
)
|
|
254
|
+
)
|
|
255
|
+
.decode()
|
|
256
|
+
.strip()
|
|
257
|
+
)
|
|
182
258
|
return float(reward or 0)
|
|
183
259
|
except (SandboxError, OSError, ValueError):
|
|
184
260
|
return 0.0
|
|
185
261
|
|
|
186
262
|
async def _reward_json(self, runtime: Runtime) -> float | dict[str, float] | None:
|
|
187
|
-
"""Read Harbor's scalar or keyed JSON reward, if it is valid.
|
|
263
|
+
"""Read Harbor's scalar or keyed JSON reward, if it is valid.
|
|
264
|
+
|
|
265
|
+
Bounded: this is a grading input, and nothing guarantees its size.
|
|
266
|
+
"""
|
|
188
267
|
try:
|
|
189
|
-
return REWARD_JSON_ADAPTER.validate_json(
|
|
268
|
+
return REWARD_JSON_ADAPTER.validate_json(
|
|
269
|
+
await runtime.read(REWARD_JSON, max_bytes=MAX_REWARD_BYTES)
|
|
270
|
+
)
|
|
190
271
|
except (SandboxError, OSError, ValidationError):
|
|
191
272
|
return None
|
|
192
273
|
|
|
193
274
|
|
|
275
|
+
def verifier_box_data(data: HarborData) -> HarborData:
|
|
276
|
+
"""The verifier's box, declared as task data — the harbor env resolves the
|
|
277
|
+
grading runtime from it (image, workdir, resources, network policy), exactly
|
|
278
|
+
as the solver's box resolves from the solver task's.
|
|
279
|
+
|
|
280
|
+
Which box follows Harbor: a declared `[verifier.environment]` states its own
|
|
281
|
+
image, workdir, and resources, and what it omits is the run's default; a
|
|
282
|
+
fresh copy of `[environment]` keeps the task's own. The verifier's network
|
|
283
|
+
policy applies either way."""
|
|
284
|
+
verifier = data.verifier
|
|
285
|
+
if verifier is None:
|
|
286
|
+
raise TaskError(f"task {data.name!r} declares no separate verifier")
|
|
287
|
+
fresh = verifier.fresh_copy
|
|
288
|
+
return data.model_copy(
|
|
289
|
+
update={
|
|
290
|
+
"name": f"{data.name} (verifier)",
|
|
291
|
+
"image": verifier.image if verifier.image is not None else data.image,
|
|
292
|
+
"workdir": data.workdir if fresh else verifier.workdir,
|
|
293
|
+
"resources": data.resources if fresh else verifier.resources,
|
|
294
|
+
"network_allow": list(verifier.network_allow),
|
|
295
|
+
"network_block": [],
|
|
296
|
+
}
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def task_resources(environment, multiplier: float) -> TaskResources:
|
|
301
|
+
"""Harbor environment resource requests, scaled, as `TaskResources`.
|
|
302
|
+
|
|
303
|
+
Harbor declares CPU counts and MB sizes; `TaskResources` wants counts and GB.
|
|
304
|
+
GPU requests are never scaled.
|
|
305
|
+
"""
|
|
306
|
+
return TaskResources(
|
|
307
|
+
cpu=environment.cpus * multiplier if environment.cpus else None,
|
|
308
|
+
memory=environment.memory_mb / 1024 * multiplier
|
|
309
|
+
if environment.memory_mb
|
|
310
|
+
else None,
|
|
311
|
+
gpu=str(environment.gpus) if environment.gpus else None,
|
|
312
|
+
disk=environment.storage_mb / 1024 * multiplier
|
|
313
|
+
if environment.storage_mb
|
|
314
|
+
else None,
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
|
|
194
318
|
def harbor_cli() -> str:
|
|
195
319
|
scripts_dir = Path(sys.executable).parent
|
|
196
320
|
harbor_bin = shutil.which("harbor", path=str(scripts_dir))
|
|
@@ -318,7 +442,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
318
442
|
|
|
319
443
|
harbor_task = HarborModelTask(task_dir)
|
|
320
444
|
parsed = harbor_task.config
|
|
321
|
-
artifacts,
|
|
445
|
+
artifacts, hooks, verifier = parse_verifier_extras(task_dir, parsed, harbor_config)
|
|
322
446
|
environment = parsed.environment
|
|
323
447
|
network = parsed.agent.explicit_phase_policy() or environment.resolve_baseline()
|
|
324
448
|
task, meta = parsed.task, parsed.metadata
|
|
@@ -368,18 +492,7 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
368
492
|
if scoring_timeout is not None
|
|
369
493
|
else None,
|
|
370
494
|
),
|
|
371
|
-
resources=
|
|
372
|
-
cpu=environment.cpus * harbor_config.resource_multiplier
|
|
373
|
-
if environment.cpus
|
|
374
|
-
else None,
|
|
375
|
-
memory=environment.memory_mb / 1024 * harbor_config.resource_multiplier
|
|
376
|
-
if environment.memory_mb
|
|
377
|
-
else None,
|
|
378
|
-
gpu=str(environment.gpus) if environment.gpus else None,
|
|
379
|
-
disk=environment.storage_mb / 1024 * harbor_config.resource_multiplier
|
|
380
|
-
if environment.storage_mb
|
|
381
|
-
else None,
|
|
382
|
-
),
|
|
495
|
+
resources=task_resources(environment, harbor_config.resource_multiplier),
|
|
383
496
|
keywords=task.keywords if task else [],
|
|
384
497
|
authors=authors,
|
|
385
498
|
difficulty=meta.get("difficulty"),
|
|
@@ -388,14 +501,22 @@ def parse_task(task_dir: Path, idx: int, harbor_config: HarborConfig) -> HarborD
|
|
|
388
501
|
task_dir=str(task_dir),
|
|
389
502
|
verifier_env=parsed.verifier.env,
|
|
390
503
|
artifacts=artifacts,
|
|
391
|
-
collect=
|
|
504
|
+
collect=hooks,
|
|
505
|
+
verifier=verifier,
|
|
392
506
|
)
|
|
393
507
|
|
|
394
508
|
|
|
395
509
|
def parse_verifier_extras(
|
|
396
|
-
task_dir: Path, parsed
|
|
397
|
-
) -> tuple[list[Artifact], list[CollectHook]]:
|
|
398
|
-
"""
|
|
510
|
+
task_dir: Path, parsed, harbor_config: HarborConfig
|
|
511
|
+
) -> tuple[list[Artifact], list[CollectHook], VerifierConfig | None]:
|
|
512
|
+
"""Harbor's `artifacts`, `[[verifier.collect]]` blocks, and verifier environment,
|
|
513
|
+
narrowed to what verifiers' verifier-runtime integration can honor.
|
|
514
|
+
|
|
515
|
+
The convention dir is deliberately not prepended here (Harbor's
|
|
516
|
+
`with_convention_entry` would): collection injects it itself, as an optional sweep.
|
|
517
|
+
Prepending it would make it an explicitly declared entry, and declared entries are
|
|
518
|
+
required — which would fail every task that never writes there.
|
|
519
|
+
"""
|
|
399
520
|
from harbor.constants import MAIN_SERVICE_NAME
|
|
400
521
|
from harbor.models.task.artifacts import (
|
|
401
522
|
effective_artifact_service,
|
|
@@ -403,13 +524,6 @@ def parse_verifier_extras(
|
|
|
403
524
|
)
|
|
404
525
|
|
|
405
526
|
verifier = parsed.verifier
|
|
406
|
-
if verifier.environment is not None:
|
|
407
|
-
raise ValueError(
|
|
408
|
-
f"{task_dir.name}: [verifier.environment] declares a separate verifier "
|
|
409
|
-
"image. Grading runs in a fresh box built from the task's own image, so "
|
|
410
|
-
"only the agent's delta has to travel; a different verifier image needs "
|
|
411
|
-
"the full working tree copied over and isn't supported yet."
|
|
412
|
-
)
|
|
413
527
|
if verifier.user is not None:
|
|
414
528
|
raise ValueError(f"{task_dir.name}: [verifier].user is not supported")
|
|
415
529
|
|
|
@@ -447,7 +561,89 @@ def parse_verifier_extras(
|
|
|
447
561
|
)
|
|
448
562
|
hooks.append(CollectHook(command=hook.command, timeout_sec=hook.timeout_sec))
|
|
449
563
|
|
|
450
|
-
return artifacts, hooks
|
|
564
|
+
return artifacts, hooks, parse_verifier_environment(task_dir, parsed, harbor_config)
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def parse_verifier_environment(
|
|
568
|
+
task_dir: Path, parsed, harbor_config: HarborConfig
|
|
569
|
+
) -> VerifierConfig | None:
|
|
570
|
+
"""The box Harbor wants this task's verifier in, or None to grade in the agent's.
|
|
571
|
+
|
|
572
|
+
Harbor resolves `[verifier.environment]` if declared, else a deep copy of
|
|
573
|
+
`[environment]` — so a mode-only `separate` lands on the task's own image and needs
|
|
574
|
+
nothing but a second box. A declared environment is the case that can name a
|
|
575
|
+
different image, and the case that can name none at all: there Harbor builds
|
|
576
|
+
`tests/Dockerfile`, which verifiers never does.
|
|
577
|
+
"""
|
|
578
|
+
from harbor.models.task.config import NetworkMode, TaskOS
|
|
579
|
+
from harbor.models.task.verifier_mode import (
|
|
580
|
+
VerifierEnvironmentMode,
|
|
581
|
+
resolve_effective_verifier_env_config,
|
|
582
|
+
resolve_task_verifier_mode,
|
|
583
|
+
)
|
|
584
|
+
|
|
585
|
+
if resolve_task_verifier_mode(parsed) != VerifierEnvironmentMode.SEPARATE:
|
|
586
|
+
return None
|
|
587
|
+
if harbor_config.ignore_separate_verifier:
|
|
588
|
+
logger.warning(
|
|
589
|
+
"%s: asks for a separate verifier; grading in the agent's box anyway "
|
|
590
|
+
"(--taskset.ignore-separate-verifier)",
|
|
591
|
+
task_dir.name,
|
|
592
|
+
)
|
|
593
|
+
return None
|
|
594
|
+
|
|
595
|
+
environment = resolve_effective_verifier_env_config(parsed, None)
|
|
596
|
+
if environment is None: # unreachable while the mode is SEPARATE
|
|
597
|
+
raise ValueError(f"{task_dir.name}: separate verifier resolved no environment")
|
|
598
|
+
declared = parsed.verifier.environment is not None
|
|
599
|
+
|
|
600
|
+
if declared and environment.docker_image is None:
|
|
601
|
+
if not harbor_config.ignore_dockerfile:
|
|
602
|
+
raise ValueError(
|
|
603
|
+
f"{task_dir.name}: [verifier.environment] names no docker_image, so "
|
|
604
|
+
"Harbor would build the verifier image from tests/Dockerfile. Verifiers "
|
|
605
|
+
"pulls images and never builds them: build and push it yourself (e.g. "
|
|
606
|
+
"`prime images push`) and set [verifier.environment].docker_image to the "
|
|
607
|
+
"resulting ref, or pass --taskset.ignore-dockerfile to grade in the "
|
|
608
|
+
"agent's image instead."
|
|
609
|
+
)
|
|
610
|
+
logger.warning(
|
|
611
|
+
"%s: [verifier.environment] names no docker_image — grading in the agent's "
|
|
612
|
+
"image rather than building tests/Dockerfile, so the verifier runs somewhere "
|
|
613
|
+
"the task never declared",
|
|
614
|
+
task_dir.name,
|
|
615
|
+
)
|
|
616
|
+
unsupported = [
|
|
617
|
+
field
|
|
618
|
+
for field in ("healthcheck", "mcp_servers", "skills_dir", "gpu_types", "tpu")
|
|
619
|
+
if getattr(environment, field, None)
|
|
620
|
+
]
|
|
621
|
+
if environment.os != TaskOS.LINUX or unsupported:
|
|
622
|
+
raise ValueError(
|
|
623
|
+
f"{task_dir.name}: verifier environment declares "
|
|
624
|
+
f"{unsupported or environment.os}, which verifiers' verifier-runtime "
|
|
625
|
+
"integration cannot honor"
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
network = parsed.verifier.explicit_phase_policy() or environment.resolve_baseline()
|
|
629
|
+
return VerifierConfig(
|
|
630
|
+
image=environment.docker_image if declared else None,
|
|
631
|
+
# A declared environment states its own resources; what it leaves out is the
|
|
632
|
+
# run's default, not the agent task's. A fresh copy is the task's environment,
|
|
633
|
+
# so it keeps whatever the agent box resolved to.
|
|
634
|
+
resources=(
|
|
635
|
+
task_resources(environment, harbor_config.resource_multiplier)
|
|
636
|
+
if declared
|
|
637
|
+
else TaskResources()
|
|
638
|
+
),
|
|
639
|
+
workdir=environment.workdir if declared else None,
|
|
640
|
+
fresh_copy=not declared,
|
|
641
|
+
network_allow=(
|
|
642
|
+
["*"]
|
|
643
|
+
if network.network_mode == NetworkMode.PUBLIC
|
|
644
|
+
else list(network.allowed_hosts)
|
|
645
|
+
),
|
|
646
|
+
)
|
|
451
647
|
|
|
452
648
|
|
|
453
649
|
def verifier_env(task: HarborData) -> dict[str, str]:
|
verifiers/v1/trace.py
CHANGED
|
@@ -319,10 +319,11 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
319
319
|
calls: list[ModelCall] = Field(default_factory=list)
|
|
320
320
|
"""Every model call; automatically recorded at intercept time + linked into `nodes`."""
|
|
321
321
|
|
|
322
|
-
rewards: dict[str, Reward] = Field(default_factory=dict)
|
|
323
|
-
"""Named, weighted rewards
|
|
324
|
-
|
|
325
|
-
|
|
322
|
+
rewards: dict[str, Reward | None] = Field(default_factory=dict)
|
|
323
|
+
"""Named, weighted rewards; `None` means scoring didn't run (e.g. because of a
|
|
324
|
+
preceding error)."""
|
|
325
|
+
metrics: dict[str, float | None] = Field(default_factory=dict)
|
|
326
|
+
"""Unweighted, named metrics; `None` as in `rewards`."""
|
|
326
327
|
info: dict[str, Any] = Field(default_factory=dict)
|
|
327
328
|
"""Scratch space for task-specific metadata."""
|
|
328
329
|
state: StateT = Field(default_factory=State, exclude=True)
|
|
@@ -346,7 +347,7 @@ class Trace(BaseModel, Generic[DataT, StateT, AgentConfigT]):
|
|
|
346
347
|
|
|
347
348
|
@property
|
|
348
349
|
def reward(self) -> float:
|
|
349
|
-
return sum(r.value for r in self.rewards.values())
|
|
350
|
+
return sum(r.value for r in self.rewards.values() if r is not None)
|
|
350
351
|
|
|
351
352
|
@property
|
|
352
353
|
def has_error(self) -> bool:
|
verifiers/v1/utils/decorators.py
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import inspect
|
|
5
|
-
from collections.abc import Callable
|
|
5
|
+
from collections.abc import Callable, Iterable
|
|
6
6
|
from typing import Any, TypeVar, overload
|
|
7
7
|
|
|
8
8
|
F = TypeVar("F", bound=Callable[..., Any])
|
|
@@ -38,6 +38,22 @@ async def invoke_all(
|
|
|
38
38
|
return await asyncio.gather(*(invoke(fn, available) for fn in fns))
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
def seed(scores: dict[str, Any], names: Iterable[str]) -> None:
|
|
42
|
+
"""Mark every expected scoring key as unscored (`None`) before invocation, so a
|
|
43
|
+
trace records which signals should have run even when scoring fails midway.
|
|
44
|
+
Overwrites: a re-scoring attempt resets its own names, so stale values from a
|
|
45
|
+
previous attempt never read as fresh."""
|
|
46
|
+
for name in names:
|
|
47
|
+
scores[name] = None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def unseed(scores: dict[str, Any], name: str) -> None:
|
|
51
|
+
"""Drop a still-unscored seed — a handler returning keyed scores records under
|
|
52
|
+
its result's keys, not its function name."""
|
|
53
|
+
if scores.get(name) is None:
|
|
54
|
+
scores.pop(name, None)
|
|
55
|
+
|
|
56
|
+
|
|
41
57
|
def mark(attr: str, **extra: Any) -> Callable[[F], F]:
|
|
42
58
|
def decorator(f: F) -> F:
|
|
43
59
|
setattr(f, attr, True)
|
verifiers/v1/utils/platform.py
CHANGED
|
@@ -93,7 +93,8 @@ def trace_to_sample(
|
|
|
93
93
|
# Flatten sub-rewards to top-level keys the way v0 does (raw scores, as v0's
|
|
94
94
|
# per-function outputs were); env metrics stay nested.
|
|
95
95
|
for name, reward in trace.rewards.items():
|
|
96
|
-
|
|
96
|
+
if reward is not None:
|
|
97
|
+
sample.setdefault(name, reward.score)
|
|
97
98
|
return sample
|
|
98
99
|
|
|
99
100
|
|
|
@@ -128,8 +129,15 @@ def _run_metrics(episodes: list[Episode], traces: list[Trace]) -> dict[str, Any]
|
|
|
128
129
|
sums: dict[str, float] = {}
|
|
129
130
|
counts: dict[str, int] = {}
|
|
130
131
|
for trace in scored:
|
|
131
|
-
scores = {
|
|
132
|
-
|
|
132
|
+
scores = {
|
|
133
|
+
name: reward.score
|
|
134
|
+
for name, reward in trace.rewards.items()
|
|
135
|
+
if reward is not None
|
|
136
|
+
}
|
|
137
|
+
metrics = {
|
|
138
|
+
name: value for name, value in trace.metrics.items() if value is not None
|
|
139
|
+
}
|
|
140
|
+
for name, value in {**scores, **metrics}.items():
|
|
133
141
|
sums[name] = sums.get(name, 0.0) + value
|
|
134
142
|
counts[name] = counts.get(name, 0) + 1
|
|
135
143
|
n = len(scored)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev75
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -168,20 +168,20 @@ verifiers/utils/tool_utils.py,sha256=gInWZODWQUZN4TyEMuuITyOUm9qlHJxtcvr1zZl_zJs
|
|
|
168
168
|
verifiers/utils/usage_utils.py,sha256=Mw6WC78r6EsLyRFdP9ICveqptkjizbSkAC86ZH0geVU,3918
|
|
169
169
|
verifiers/utils/version_utils.py,sha256=39gV8-mNF9GspAgKloLo8oy7_MVluMrwGPrntQv2hx4,2656
|
|
170
170
|
verifiers/v1/__init__.py,sha256=kacTRsGsRHfOruv16ZyIOesfvCH_nW98ARKPYSgcNIc,7509
|
|
171
|
-
verifiers/v1/agent.py,sha256=
|
|
171
|
+
verifiers/v1/agent.py,sha256=MQVKGjUhrP8uzRXAq94uJHkEsdHFziPxQHZ7TOD324I,31306
|
|
172
172
|
verifiers/v1/env.py,sha256=g6fpoG-Z9lM3vTKrrU-QRJY3S88_y9zsPHwBvfs5vdY,17865
|
|
173
173
|
verifiers/v1/episode.py,sha256=YtuRTV_PpwDglXFeN6UNg8IswpP1HseFLCYtascGMNI,3218
|
|
174
174
|
verifiers/v1/errors.py,sha256=kQeEPX06TwAIuiz7MlWFYirrpKbx5STi6A5b_dtQQoA,6885
|
|
175
175
|
verifiers/v1/graph.py,sha256=aBAzh1Ibp3klKNYjvzNyJSbjV8Oi5TJ791dN7o6ex1E,29070
|
|
176
|
-
verifiers/v1/harness.py,sha256=
|
|
176
|
+
verifiers/v1/harness.py,sha256=eKuqvYbQjnCMavkiYWZQDHJgH08m6SJ4PuevPuShjS0,11097
|
|
177
177
|
verifiers/v1/judge.py,sha256=Pvr0C41ah1qkNSZ0WXdkDvBMfv5YSfY1REHncwlJ8cg,9346
|
|
178
178
|
verifiers/v1/legacy.py,sha256=eLsFiwihOMh4zS_u1V7DXF22r9beoG4k2JGwNEcSPfg,22872
|
|
179
179
|
verifiers/v1/rollout.py,sha256=s1f1VZAAAI89j3y9ceq_E4mvX0F7cFQAnyx27SZsjvA,17881
|
|
180
180
|
verifiers/v1/session.py,sha256=tNAraQfiJKpWqgBFNLyPhfHiw0JhzzkyOfRcKWGnR2Q,6939
|
|
181
181
|
verifiers/v1/state.py,sha256=EckF2bWp-vV4b1jYJ9sLI5xrfGuI5spIgYYwW926toI,595
|
|
182
|
-
verifiers/v1/task.py,sha256=
|
|
182
|
+
verifiers/v1/task.py,sha256=jwMKiKlMtksTd8j2dcDlFtb6LC8BRE-N5jkNXlMp2jc,9457
|
|
183
183
|
verifiers/v1/taskset.py,sha256=fp2E0IEhL_Ybj9cegZwljfTmW27p_30zTWHFKw_kXE4,4374
|
|
184
|
-
verifiers/v1/trace.py,sha256=
|
|
184
|
+
verifiers/v1/trace.py,sha256=Z4uK-lYAbl-6CHFhTo-gadHnbi63LC0sJtUERC_m5ss,19477
|
|
185
185
|
verifiers/v1/types.py,sha256=1PxamJspmoTc3OlFZAwH6o_2i_Tg9ZNvHxttIHWIxq8,8948
|
|
186
186
|
verifiers/v1/acp/__init__.py,sha256=9SYCFtzGUM_wH98ldpVLcFhoq2M999jbG0tU5ODY70U,2297
|
|
187
187
|
verifiers/v1/acp/_runner.py,sha256=BcYaNZHuawzCgxOVdhiF6PY_B1HMxIA1MwHy0mOzhSk,7740
|
|
@@ -196,7 +196,7 @@ verifiers/v1/cli/serve.py,sha256=_mdJaHVVe4pXzcpAot0ugcmiN3ra2gS4GrGmLghnX5M,286
|
|
|
196
196
|
verifiers/v1/cli/validate.py,sha256=7ax6FqIzNBSYjd3JXK4v7Vbq34Rozyh1OvGpYnGqlSg,10266
|
|
197
197
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
198
198
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
199
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
199
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=4j2YCPyzNHDT7d8MwqYR3C906lzwDVAd3FxgFYFgsw0,33588
|
|
200
200
|
verifiers/v1/cli/dashboard/replay.py,sha256=CvRVaf0dUum5v0IzPQbFPPikBQmTrkMeaunGfVMTWHc,2755
|
|
201
201
|
verifiers/v1/cli/dashboard/validate.py,sha256=rLsQ_31DjTIZpC_VO7zSG41ncUEBxqCx8mL9nTnGiMc,3650
|
|
202
202
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
@@ -234,7 +234,7 @@ verifiers/v1/dialects/chat.py,sha256=R_otqi5N0snB1QVKIYloNeLHR9kc7OmI06HYg1ir_Lw
|
|
|
234
234
|
verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
|
|
235
235
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
236
236
|
verifiers/v1/envs/agentic_judge/__init__.py,sha256=6vwtMCQ_cuvuubCF5-nrE5W7gGgzE-LIPKorkM5DXdw,309
|
|
237
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
237
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=6QrfuIlg-9MkcqIn5_6F3qJ6ddLJNaeYsZRJC9EzGkw,17011
|
|
238
238
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
239
239
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
240
240
|
verifiers/v1/envs/single_agent/__init__.py,sha256=r0edAc_6gtHBhPzeBGVcdHZZqSdCPZnLMje6eVgyz5U,138
|
|
@@ -297,13 +297,13 @@ verifiers/v1/mcp/__init__.py,sha256=pm8ZWBiL1q1tY4wHvJzM_QrZ9oR8aGm-bIRWiRJsgHI,
|
|
|
297
297
|
verifiers/v1/mcp/launch.py,sha256=wBpyeiOO86l1H2lVMbLVSV9Lbp90XeyXqauuK8snfEU,20171
|
|
298
298
|
verifiers/v1/mcp/server.py,sha256=dRBxytFgS6rv1BS6t5EgBd3xeF6U1nTyt0ArDCLk3VE,12301
|
|
299
299
|
verifiers/v1/mcp/toolset.py,sha256=T8Ioiq4wphxVOTHUDf28ZoQBM2ZYEevGnGgw8TFEMiA,1002
|
|
300
|
-
verifiers/v1/runtimes/__init__.py,sha256=
|
|
301
|
-
verifiers/v1/runtimes/base.py,sha256=
|
|
300
|
+
verifiers/v1/runtimes/__init__.py,sha256=Kwe612RjtGzrQhc19QdXc7CmZAnQLH-EdK_xqkKP3Tk,2629
|
|
301
|
+
verifiers/v1/runtimes/base.py,sha256=tZRELSfZmi3l_1BsjabnNC3mEPfqNvUNevdnRvLPv6w,16257
|
|
302
302
|
verifiers/v1/runtimes/limiters.py,sha256=ZjjYUFEt20yH8ggL5uU9w9U500AOB44qSUTQ7bYyh0c,2730
|
|
303
|
-
verifiers/v1/runtimes/modal.py,sha256=
|
|
304
|
-
verifiers/v1/runtimes/prime.py,sha256=
|
|
305
|
-
verifiers/v1/runtimes/subprocess.py,sha256=
|
|
306
|
-
verifiers/v1/runtimes/docker/__init__.py,sha256=
|
|
303
|
+
verifiers/v1/runtimes/modal.py,sha256=eFiT-Ovf_RwQZbSw3Dwv2EXeHV0d6uGvPswqKYKCw5k,8826
|
|
304
|
+
verifiers/v1/runtimes/prime.py,sha256=kwvPqUI8D-GvcFwYRdpTHE1hXS0vfJGPlFOLgzm4Mvs,14756
|
|
305
|
+
verifiers/v1/runtimes/subprocess.py,sha256=ZrHRlTuTyCrcSKpiPD2ElcPpe97EvVsXJ8BfwUaNXiI,5988
|
|
306
|
+
verifiers/v1/runtimes/docker/__init__.py,sha256=vyEgf6XgAwjGgBicJchJvNgcDgYXrRvyQC4W52-23GU,14556
|
|
307
307
|
verifiers/v1/runtimes/docker/egress.py,sha256=37UXgNX9gnoLxjMg4KZaJIGoApCAM6MK22eJGNoMi38,14514
|
|
308
308
|
verifiers/v1/serve/__init__.py,sha256=sr_zkDzAHnOcQm-QRm8xwJBmCMFD-xhKIFPaDZvMbUs,641
|
|
309
309
|
verifiers/v1/serve/client.py,sha256=CxpOwCn4ehuCXfqDJeV9zj7O0XC7MPcI-OZ1L_CSDlo,7057
|
|
@@ -311,8 +311,9 @@ verifiers/v1/serve/pool.py,sha256=AyOMjDYi9PLGA8Z-bnr6Fd5cnubuRavEBuudbVWtq2E,14
|
|
|
311
311
|
verifiers/v1/serve/server.py,sha256=eFWJsTUDbMywKQXKDuk57QvS7Y4IZwW2Ok3LwJHMFLI,9200
|
|
312
312
|
verifiers/v1/serve/types.py,sha256=3F4IIGIQgYITlEFxeX3uNJYrJlKugFrPTozbDNwDfBY,2674
|
|
313
313
|
verifiers/v1/tasksets/__init__.py,sha256=b62O4WDNbGfJH_4NjYmznn5otWovdfvUgJZyoIZyLh8,596
|
|
314
|
-
verifiers/v1/tasksets/harbor/__init__.py,sha256=
|
|
315
|
-
verifiers/v1/tasksets/harbor/
|
|
314
|
+
verifiers/v1/tasksets/harbor/__init__.py,sha256=JxThzHFYEkuSfQbVf3pEnjordgdfe4MU-QOsSSbyovU,326
|
|
315
|
+
verifiers/v1/tasksets/harbor/env.py,sha256=bNF5dfbhcsdrd_8JmW5ved8Yk-mHhrOgKkVEA563qEo,5847
|
|
316
|
+
verifiers/v1/tasksets/harbor/taskset.py,sha256=vmDjP7dhxchZ5-J9IUeiATwwchxUvAeciCevD6W_N4c,29032
|
|
316
317
|
verifiers/v1/tasksets/lean/__init__.py,sha256=oyv-GbJQBvCyGwo7Ysxm2NjZW-aX4j1uDoHU0f-E7xQ,767
|
|
317
318
|
verifiers/v1/tasksets/lean/scoring.py,sha256=sfzsT6MUK0zdMgAQ57_P_0CBaPHy3M7go71QMr5hDrs,10599
|
|
318
319
|
verifiers/v1/tasksets/lean/taskset.py,sha256=XgUqX0uJvHC8sKFhbK92iu6nUGcPuXCUpI3ZBhDcb7c,9025
|
|
@@ -324,7 +325,7 @@ verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuF
|
|
|
324
325
|
verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
|
|
325
326
|
verifiers/v1/utils/artifacts.py,sha256=vOzbPSuMirUecyylrhvLZ_vAcTd6JTJBztGPDQ0ANNk,6706
|
|
326
327
|
verifiers/v1/utils/compile.py,sha256=Oh4hDVRNsql_BSOMh4kDEuze9_ZF17Q7uzBhhU70UZA,5663
|
|
327
|
-
verifiers/v1/utils/decorators.py,sha256=
|
|
328
|
+
verifiers/v1/utils/decorators.py,sha256=DG_K3HMiWqRHsY1bA7FJD3uTHUVuLw6SUJV-PnNRifA,4538
|
|
328
329
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
329
330
|
verifiers/v1/utils/generic.py,sha256=2VIN9RapMO9pVkFHB1UKUHck1vlTJ1pjfad2NzxuuXA,2055
|
|
330
331
|
verifiers/v1/utils/git.py,sha256=MTRinOvG4EddfjdahHkFZUihHmhJ37n5ROtgjzjE1NA,8366
|
|
@@ -334,12 +335,12 @@ verifiers/v1/utils/interrupt.py,sha256=F-KKhc5ndPJJfhd3SuMqyqhXhA32FhCRy5KWFJpEo
|
|
|
334
335
|
verifiers/v1/utils/loaders.py,sha256=NSIBdNU3JVeyddL-dM4dJt6khCHyjCgzJuFtLrgiwD0,9989
|
|
335
336
|
verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y,2084
|
|
336
337
|
verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
|
|
337
|
-
verifiers/v1/utils/platform.py,sha256=
|
|
338
|
+
verifiers/v1/utils/platform.py,sha256=56Ixmk1SER6q5LvzyYcA0hgzGpzhxKZL8Hiqp4c_XVc,11576
|
|
338
339
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
339
340
|
verifiers/v1/utils/score.py,sha256=-R5Cog14r_tJ6Q_oOfGr5VPAm2CCUhofYZB6B9lm4wM,5787
|
|
340
341
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
341
|
-
verifiers-0.2.2.
|
|
342
|
-
verifiers-0.2.2.
|
|
343
|
-
verifiers-0.2.2.
|
|
344
|
-
verifiers-0.2.2.
|
|
345
|
-
verifiers-0.2.2.
|
|
342
|
+
verifiers-0.2.2.dev75.dist-info/METADATA,sha256=8ZNGDY1kLfAvLaXC5voMOQeGRtvHeflfC3eV0v34rWg,4545
|
|
343
|
+
verifiers-0.2.2.dev75.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
344
|
+
verifiers-0.2.2.dev75.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
|
|
345
|
+
verifiers-0.2.2.dev75.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
346
|
+
verifiers-0.2.2.dev75.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|