verifiers 0.2.2.dev30__py3-none-any.whl → 0.2.2.dev32__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,9 +1,9 @@
1
1
  """The OpenAI Responses dialect (codex and friends).
2
2
 
3
3
  Request parsing walks the `input` items, folding each run of assistant-side items (reasoning /
4
- assistant message / function_call) into one typed assistant message; response parsing reads the
5
- `output` items. Relay-only: the eval client forwards the program's bytes to a `/responses`
6
- endpoint and this dialect parses a copy for the trace. Server-side statefulness
4
+ assistant message / function or custom tool call) into one typed assistant message; response
5
+ parsing reads the `output` items. Relay-only: the eval client forwards the program's bytes to a
6
+ `/responses` endpoint and this dialect parses a copy for the trace. Server-side statefulness
7
7
  (`previous_response_id`) is not emulated — the endpoint owns it.
8
8
  """
9
9
 
@@ -97,8 +97,7 @@ def parse_content(content) -> str | list[ContentPart]:
97
97
 
98
98
 
99
99
  def fold_assistant(items: list[dict]) -> AssistantMessage:
100
- """One run of assistant-side items (reasoning / message / function_call) -> one typed
101
- assistant message."""
100
+ """One run of assistant-side items -> one typed assistant message."""
102
101
  content = ""
103
102
  reasoning: list[str] = []
104
103
  calls: list[ToolCall] = []
@@ -106,12 +105,12 @@ def fold_assistant(items: list[dict]) -> AssistantMessage:
106
105
  if item.get("type") == "reasoning":
107
106
  reasoning += [s.get("text", "") for s in item.get("summary") or []]
108
107
  reasoning += [c.get("text", "") for c in item.get("content") or []]
109
- elif item.get("type") == "function_call":
108
+ elif item.get("type") in ("function_call", "custom_tool_call"):
110
109
  calls.append(
111
110
  ToolCall(
112
111
  id=item.get("call_id", ""),
113
112
  name=item.get("name", ""),
114
- arguments=item.get("arguments", ""),
113
+ arguments=item.get("arguments", item.get("input", "")),
115
114
  )
116
115
  )
117
116
  else: # an assistant message item
@@ -151,12 +150,12 @@ def response_from_wire(response: OpenAIResponse) -> Response:
151
150
  elif kind == "reasoning":
152
151
  reasoning += [s.get("text", "") for s in item.get("summary") or []]
153
152
  reasoning += [c.get("text", "") for c in item.get("content") or []]
154
- elif kind == "function_call":
153
+ elif kind in ("function_call", "custom_tool_call"):
155
154
  calls.append(
156
155
  ToolCall(
157
156
  id=item.get("call_id", ""),
158
157
  name=item.get("name", ""),
159
- arguments=item.get("arguments", ""),
158
+ arguments=item.get("arguments", item.get("input", "")),
160
159
  )
161
160
  )
162
161
  tool_calls = calls or None
@@ -279,7 +278,10 @@ class ResponsesDialect(Dialect[dict, OpenAIResponse]):
279
278
  run = []
280
279
  if assistant:
281
280
  run.append(item)
282
- elif item.get("type") == "function_call_output":
281
+ elif item.get("type") in (
282
+ "function_call_output",
283
+ "custom_tool_call_output",
284
+ ):
283
285
  output = item.get("output")
284
286
  content = (
285
287
  parse_content(output)
verifiers/v1/graph.py CHANGED
@@ -212,12 +212,17 @@ def _canonical_tool_arguments(arguments: str) -> str:
212
212
  return arguments
213
213
 
214
214
 
215
+ # Provider-specific fields not represented by typed messages but required on replay.
216
+ _PROVIDER_STATE_FIELDS = frozenset({"encrypted_content", "signature", "data", "phase"})
217
+
218
+
215
219
  def message_hash(message: Message) -> str:
216
220
  """Stable content hash on the fields that round-trip through a prompt — role, content
217
221
  (None and "" equal), assistant reasoning content when present, assistant tool calls,
218
- tool call id. Two messages hash equal iff they're the same conversational message, so a
219
- re-stated prefix message dedups to one node. The dedup key for sharing a prefix across
220
- turns/branches; salt-free so it is identical across processes and after deserialization."""
222
+ opaque continuation state, tool call id. Two messages hash equal iff they're the same
223
+ conversational message, so a re-stated prefix message dedups to one node. The dedup key
224
+ for sharing a prefix across turns/branches; salt-free so it is identical across processes
225
+ and after deserialization."""
221
226
  digest = hashlib.blake2b(digest_size=16)
222
227
 
223
228
  def add(value: str) -> None:
@@ -241,10 +246,37 @@ def message_hash(message: Message) -> str:
241
246
  if message.reasoning_content is not None:
242
247
  add("reasoning_content")
243
248
  add(message.reasoning_content)
244
- if message.provider_state:
245
- # Signed/encrypted continuation state distinguishes otherwise equal turns.
249
+ for item in message.provider_state or []:
250
+ kind = item.get("type") or (
251
+ "message" if item.get("role") == "assistant" else ""
252
+ )
253
+ hashed_state = {
254
+ key: item[key]
255
+ for key in _PROVIDER_STATE_FIELDS
256
+ if item.get(key) is not None
257
+ }
258
+ if kind == "message" and isinstance(item.get("content"), list):
259
+ # Keep content parts the typed message does not expose, such as refusals.
260
+ unparsed_content = [
261
+ part
262
+ for part in item.get("content") or []
263
+ if part.get("type") not in ("input_text", "output_text")
264
+ ]
265
+ if unparsed_content:
266
+ hashed_state["content"] = unparsed_content
267
+ represented = kind in ("message", "reasoning") or (
268
+ kind in ("function_call", "custom_tool_call")
269
+ and any(
270
+ call.id == item.get("call_id") for call in message.tool_calls or []
271
+ )
272
+ )
273
+ if represented and not hashed_state:
274
+ continue
275
+ # Unknown provider items still distinguish built-in calls and actions.
276
+ state = hashed_state if represented else item
246
277
  add("provider_state")
247
- add(json.dumps(message.provider_state, sort_keys=True))
278
+ add(kind)
279
+ add(json.dumps(state, sort_keys=True))
248
280
  for tc in message.tool_calls or []:
249
281
  add("tool_call")
250
282
  add(tc.id)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: verifiers
3
- Version: 0.2.2.dev30
3
+ Version: 0.2.2.dev32
4
4
  Summary: Verifiers: Environments for LLM Reinforcement Learning
5
5
  Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
6
6
  Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
@@ -173,7 +173,7 @@ verifiers/v1/decorators.py,sha256=XRMkUQSyvXCYP5fOwzBYV5qEOlxLg6rzYhnhrHHHTIQ,38
173
173
  verifiers/v1/env.py,sha256=w6sHWdWijLRZzhwVyJjadnGtziNqJAPCeMUoNdzHoLg,17712
174
174
  verifiers/v1/episode.py,sha256=tfmFMjOy0aZJ2Q_tdSYMXhFlrhqNqnZKPbRH-CDm3oo,2497
175
175
  verifiers/v1/errors.py,sha256=Pj5Om8x1fP3TDPQQn6nUCoSPoZRsy2JeBz8pXhpPrDY,6883
176
- verifiers/v1/graph.py,sha256=Mivq5jICUcyK-fhomFTwP4fQn688MXg6-hmdf03aC4Y,27703
176
+ verifiers/v1/graph.py,sha256=-j8MNdIv1r5ifajv5ggnGPsfBNB74IO3WKLcrpYh9hg,29102
177
177
  verifiers/v1/harness.py,sha256=Ukzhk7wxSUSVnN4tRAUMfniN9R9MBSNzs2SiT5YcRzQ,10711
178
178
  verifiers/v1/judge.py,sha256=ZWvr6uCniSwqzzjKrlyh-lzfCh5q-VyB16W42x0aWlg,9449
179
179
  verifiers/v1/legacy.py,sha256=8eVGhutQEgJG4qabhph4Xs1VzTWvmMxMn3-Zk-6ViWM,22306
@@ -233,7 +233,7 @@ verifiers/v1/dialects/__init__.py,sha256=PR3CJFI5siVJBXcF7H2Wi_UCygotHhL0Oj6DIOF
233
233
  verifiers/v1/dialects/anthropic.py,sha256=nj9J7OOhexmPHa-sWiNP2xlI9_bGhvwM5VWzdh19pBc,13526
234
234
  verifiers/v1/dialects/base.py,sha256=YZQDnasQseDXK-GHLfJLZgyHIIygS80YhD_sTjmUbfA,8126
235
235
  verifiers/v1/dialects/chat.py,sha256=DFvjmIW86jZUMWuz-hOqzxC6tiiFibiAzNEJ3JkpvjU,13998
236
- verifiers/v1/dialects/responses.py,sha256=vVgeT7-aWjtfED2IKlDwTATKn_AQZruHNL6mkFTTZTU,13548
236
+ verifiers/v1/dialects/responses.py,sha256=ismnxUikPcPUIm80FUTPBbj-Cn1e0yGAsISMEseAWRc,13679
237
237
  verifiers/v1/envs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
238
238
  verifiers/v1/envs/agentic_judge/__init__.py,sha256=X7vQbbbfmJ_j-mJEfBkuB4a8QUPkUYkpG7yp0xYOEmg,279
239
239
  verifiers/v1/envs/agentic_judge/env.py,sha256=_zEKG9i3LsOVIQWULXYAHpbmrQNPWBRayJx2yjvtPbQ,15307
@@ -327,8 +327,8 @@ verifiers/v1/utils/logging.py,sha256=OcMHA6NsYux3oIzjPuI95rDWmFBHNcHDjZeNIhXTX-Y
327
327
  verifiers/v1/utils/memory.py,sha256=ZkIvGk6uITAH5sKon65LifKPbvZr8mJ__PVc23FZpOQ,1835
328
328
  verifiers/v1/utils/sampling.py,sha256=JczGzBn6s3wsIrSY2Hy3hE8m6NreNgsNzhSjDikQqX0,1037
329
329
  verifiers/v1/utils/version.py,sha256=75ZtI2NHBmlb52KpcKLUpiSXp8q4bASr7uKeXoCKlT8,1582
330
- verifiers-0.2.2.dev30.dist-info/METADATA,sha256=uDNFFjXlNyB_GcznwsBBZ80q9-wE4XeZE-whq3WzSDg,4540
331
- verifiers-0.2.2.dev30.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
332
- verifiers-0.2.2.dev30.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
333
- verifiers-0.2.2.dev30.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
334
- verifiers-0.2.2.dev30.dist-info/RECORD,,
330
+ verifiers-0.2.2.dev32.dist-info/METADATA,sha256=LpUuAntVwZmy9G8YvVcwEOfT-3yTokGn3i3eyi4rjgk,4540
331
+ verifiers-0.2.2.dev32.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
332
+ verifiers-0.2.2.dev32.dist-info/entry_points.txt,sha256=dF82JUYEFslR1AOW8LZfuBWeZeQoyX4wCRrduklg8-I,551
333
+ verifiers-0.2.2.dev32.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
334
+ verifiers-0.2.2.dev32.dist-info/RECORD,,