verifiers 0.3.2.dev55__py3-none-any.whl → 0.3.2.dev57__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/envs/agentic_judge/env.py +20 -8
- verifiers/v1/harnesses/codex/harness.py +0 -6
- {verifiers-0.3.2.dev55.dist-info → verifiers-0.3.2.dev57.dist-info}/METADATA +1 -1
- {verifiers-0.3.2.dev55.dist-info → verifiers-0.3.2.dev57.dist-info}/RECORD +7 -7
- {verifiers-0.3.2.dev55.dist-info → verifiers-0.3.2.dev57.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev55.dist-info → verifiers-0.3.2.dev57.dist-info}/entry_points.txt +0 -0
- {verifiers-0.3.2.dev55.dist-info → verifiers-0.3.2.dev57.dist-info}/licenses/LICENSE +0 -0
|
@@ -5,7 +5,9 @@ a fresh box from the solver's runtime policy and restores only the task's collec
|
|
|
5
5
|
artifacts; `--env.id shared-agentic-judge` explicitly runs the judge in the
|
|
6
6
|
solver's box. The judge grades rubric criteria (`[env.task]`: policy prompt,
|
|
7
7
|
criteria file) and writes its verdicts to `/tmp/verdict.json`, with the solver's
|
|
8
|
-
|
|
8
|
+
observable trace record uploaded at `/tmp/trace.json`. Hidden reasoning and opaque
|
|
9
|
+
provider state are omitted by default and may be explicitly included through the
|
|
10
|
+
judge task config. `finalize()` validates the verdicts
|
|
9
11
|
strictly onto the solver's trace — `judge/<name>` metrics plus a weighted-mean
|
|
10
12
|
`judge` reward, composed with the taskset's own rewards via `[env.score]`
|
|
11
13
|
(judge-only by default).
|
|
@@ -82,15 +84,16 @@ def _render(template: str, **fields: str) -> str:
|
|
|
82
84
|
|
|
83
85
|
|
|
84
86
|
_RECORD_NOTE = f"""\
|
|
85
|
-
The agent's
|
|
87
|
+
The agent's observable trace record (JSON: messages, tool calls, and its `info`
|
|
86
88
|
artifacts) is written by the harness — not the agent — at `{TRACE_FILE}`. The
|
|
87
89
|
record can be very large — never dump it whole; peek selectively (list its
|
|
88
90
|
keys, then slice out specific fields with python or jq) and pull only what you
|
|
89
|
-
need.
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
91
|
+
need. Hidden reasoning and opaque provider state are not included unless the run
|
|
92
|
+
explicitly opts in. The record may also carry the task's own scores/metrics and
|
|
93
|
+
reference material (a gold answer, a reference solution, held-out tests). Those are
|
|
94
|
+
context, not your standard: recorded scores can be wrong and references can be
|
|
95
|
+
narrower than the task; do not over-index on how a reference solves it. Your verdict
|
|
96
|
+
is what YOU verified by execution."""
|
|
94
97
|
|
|
95
98
|
SHARED_WORKSPACE_NOTE = f"""\
|
|
96
99
|
## Your workspace
|
|
@@ -146,7 +149,13 @@ class JudgeTask(vf.Task):
|
|
|
146
149
|
paths they had.
|
|
147
150
|
"""
|
|
148
151
|
solved = solution.task.data
|
|
149
|
-
|
|
152
|
+
record = solution.to_record()
|
|
153
|
+
if not config.include_hidden_reasoning:
|
|
154
|
+
for node in record["nodes"]:
|
|
155
|
+
message = node["message"]
|
|
156
|
+
message.pop("reasoning_content", None)
|
|
157
|
+
message.pop("provider_state", None)
|
|
158
|
+
files = {TRACE_FILE: json.dumps(record).encode()}
|
|
150
159
|
template = config.build_prompt()
|
|
151
160
|
body = _render(template, prompt=solved.prompt_text)
|
|
152
161
|
if "{prompt}" not in template:
|
|
@@ -224,6 +233,9 @@ class JudgeTaskConfig(vf.BaseConfig):
|
|
|
224
233
|
"""Criteria the judge grades against: a `.toml`/`.json` file with a
|
|
225
234
|
`criteria` list — the plugged rubric judge's format, so the same rubric
|
|
226
235
|
files work for both. None grades the single built-in `solved` criterion."""
|
|
236
|
+
include_hidden_reasoning: bool = False
|
|
237
|
+
"""Include assistant reasoning and opaque provider state in `/tmp/trace.json`.
|
|
238
|
+
Disabled by default so the judge grades only observable agent behavior."""
|
|
227
239
|
|
|
228
240
|
@staticmethod
|
|
229
241
|
def _resolve(value: TextSource) -> str:
|
|
@@ -120,12 +120,6 @@ class CodexHarness(ACPHarness[CodexHarnessConfig]):
|
|
|
120
120
|
mcp_urls: dict[str, str],
|
|
121
121
|
) -> dict[str, str]:
|
|
122
122
|
home = self.trace_home(trace)
|
|
123
|
-
created = await runtime.run(["mkdir", "-p", home], {})
|
|
124
|
-
if created.exit_code != 0:
|
|
125
|
-
raise RuntimeError(
|
|
126
|
-
f"failed to create Codex home: {created.stderr.strip()[-500:]}"
|
|
127
|
-
)
|
|
128
|
-
|
|
129
123
|
mcp_config = "features={mcp_2026_07_28=true}\n" + (
|
|
130
124
|
"mcp_servers={"
|
|
131
125
|
+ ",".join(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.3.2.
|
|
3
|
+
Version: 0.3.2.dev57
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -66,7 +66,7 @@ verifiers/v1/dialects/chat.py,sha256=_pMsNpbffiXiK0_0tboAKtuj7-smUnkS9H0LPK4j1mU
|
|
|
66
66
|
verifiers/v1/dialects/responses.py,sha256=KRuS9TKw2uoNz4NsG7E-H6eyhm_EJQwVtZ5_XW9iB3c,29090
|
|
67
67
|
verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
68
68
|
verifiers/v1/envs/agentic_judge/__init__.py,sha256=LfEcRHLG_siKzhzWmODDXWSLpLgfum3UFdQDK9XG0gk,325
|
|
69
|
-
verifiers/v1/envs/agentic_judge/env.py,sha256=
|
|
69
|
+
verifiers/v1/envs/agentic_judge/env.py,sha256=GovnHqgRm8nQLAM7wMx57-A_wJBRYopsfiy2Bgf32nA,15339
|
|
70
70
|
verifiers/v1/envs/best_of_n/__init__.py,sha256=mubASiwXMNIhnQFDccNAc7yZ0OFsOVK8hH8WIx9b2A4,119
|
|
71
71
|
verifiers/v1/envs/best_of_n/env.py,sha256=KEUJBCVqblnXP_lEwm3mjGdaikBMkecz8UjOFPBhxc0,1987
|
|
72
72
|
verifiers/v1/envs/isolated_verifier/__init__.py,sha256=HZxc65f87pSIPRslwTZzs0I17wGa-UMM-c-CHliwPss,214
|
|
@@ -92,7 +92,7 @@ verifiers/v1/harnesses/browser_use/program.py,sha256=3dSFHSDIW3R5ZejV23sXReEKnGQ
|
|
|
92
92
|
verifiers/v1/harnesses/claude_code/__init__.py,sha256=JkZQylMSA3o1fvBx_NM41y_umuahmOSqFjTrJ2hryTw,171
|
|
93
93
|
verifiers/v1/harnesses/claude_code/harness.py,sha256=Racxbs1EAH3gRwSeUptAIa3kB-EUm_nreUnHMwRUHuE,4119
|
|
94
94
|
verifiers/v1/harnesses/codex/__init__.py,sha256=ocyBvlpSO8c3pT6D1ebrQohoXYESQcCyPT5kN3j1-N4,132
|
|
95
|
-
verifiers/v1/harnesses/codex/harness.py,sha256=
|
|
95
|
+
verifiers/v1/harnesses/codex/harness.py,sha256=trrOj9qMLkPWxnh-Iz87uGYuwEg97cwrt67PxHrayrk,6825
|
|
96
96
|
verifiers/v1/harnesses/hermes_agent/__init__.py,sha256=k84Y8UTTf1KOw5LGqPb2dk-hYFagPevDJBOF7qsR_vo,176
|
|
97
97
|
verifiers/v1/harnesses/hermes_agent/harness.py,sha256=dSsE2VZ8jaM-U14QXO3J7P-U6NMpSuOz3DUMtDWpWk0,3622
|
|
98
98
|
verifiers/v1/harnesses/hermes_agent/program.py,sha256=RaIYm6NFD8KLXERGXp6rOE4QrQjlcTnm1Fu33xYsiHw,570
|
|
@@ -189,8 +189,8 @@ verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,9
|
|
|
189
189
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
190
190
|
verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
|
|
191
191
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
192
|
-
verifiers-0.3.2.
|
|
193
|
-
verifiers-0.3.2.
|
|
194
|
-
verifiers-0.3.2.
|
|
195
|
-
verifiers-0.3.2.
|
|
196
|
-
verifiers-0.3.2.
|
|
192
|
+
verifiers-0.3.2.dev57.dist-info/METADATA,sha256=0TbLIm4QKGNQoSm54MoeFlzZV4esh0OxCatJga_9d74,4204
|
|
193
|
+
verifiers-0.3.2.dev57.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
194
|
+
verifiers-0.3.2.dev57.dist-info/entry_points.txt,sha256=uqQje0TMsr7k6nXlWxlPs13Si0evb3pVCEDF6c-YL80,241
|
|
195
|
+
verifiers-0.3.2.dev57.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
196
|
+
verifiers-0.3.2.dev57.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|