verifiers 0.3.2.dev62__py3-none-any.whl → 0.3.2.dev64__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifiers/v1/cli/dashboard/eval.py +50 -15
- verifiers/v1/utils/artifacts.py +26 -4
- {verifiers-0.3.2.dev62.dist-info → verifiers-0.3.2.dev64.dist-info}/METADATA +1 -1
- {verifiers-0.3.2.dev62.dist-info → verifiers-0.3.2.dev64.dist-info}/RECORD +7 -7
- {verifiers-0.3.2.dev62.dist-info → verifiers-0.3.2.dev64.dist-info}/WHEEL +0 -0
- {verifiers-0.3.2.dev62.dist-info → verifiers-0.3.2.dev64.dist-info}/entry_points.txt +0 -0
- {verifiers-0.3.2.dev62.dist-info → verifiers-0.3.2.dev64.dist-info}/licenses/LICENSE +0 -0
|
@@ -295,9 +295,22 @@ def _push_footer(push: "PushState | None") -> Group | None:
|
|
|
295
295
|
return Group(Rule(style="dim"), line)
|
|
296
296
|
|
|
297
297
|
|
|
298
|
+
class TraceStats:
|
|
299
|
+
"""Resource totals for a dashboard frame, retained once the episode is final."""
|
|
300
|
+
|
|
301
|
+
def __init__(self, trace: Trace) -> None:
|
|
302
|
+
self.num_input_tokens = trace.num_input_tokens
|
|
303
|
+
self.num_output_tokens = trace.num_output_tokens
|
|
304
|
+
self.num_turns = trace.num_turns
|
|
305
|
+
self.num_branches = trace.num_branches
|
|
306
|
+
self.usage = trace.usage
|
|
307
|
+
self.judge_usage = Usage.aggregate(trace.extra_usage)
|
|
308
|
+
|
|
309
|
+
|
|
298
310
|
def Progress(
|
|
299
311
|
slots: list[RunSlot],
|
|
300
312
|
start: float,
|
|
313
|
+
completed: dict[str, TraceStats],
|
|
301
314
|
page: tuple[int, int] | None = None,
|
|
302
315
|
) -> Group:
|
|
303
316
|
# On resume, `slots` includes the previous session's kept rollouts (as finished slots), so
|
|
@@ -332,7 +345,7 @@ def Progress(
|
|
|
332
345
|
ProgressBar(total=total or 1, completed=len(done)),
|
|
333
346
|
Text(stats),
|
|
334
347
|
)
|
|
335
|
-
breakdown = _breakdown(scored, done_traces)
|
|
348
|
+
breakdown = _breakdown(scored, done_traces, completed)
|
|
336
349
|
return Group(row, breakdown) if breakdown is not None else Group(row)
|
|
337
350
|
|
|
338
351
|
|
|
@@ -362,7 +375,9 @@ def _score(trace: Trace, source: str, name: str) -> float:
|
|
|
362
375
|
return value if value is not None else 0.0
|
|
363
376
|
|
|
364
377
|
|
|
365
|
-
def _breakdown(
|
|
378
|
+
def _breakdown(
|
|
379
|
+
scored: list[Trace], done: list[Trace], completed: dict[str, TraceStats]
|
|
380
|
+
) -> Table | None:
|
|
366
381
|
"""Score rows read the policy view (`scored` — trainable traces); with several
|
|
367
382
|
roles in play they split per role, each role averaging over its OWN traces (no
|
|
368
383
|
dilution, so the split covers every role — an untrainable seat's received
|
|
@@ -401,9 +416,10 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
401
416
|
phase_count: dict[str, int] = {}
|
|
402
417
|
model_secs = harness_secs = 0.0
|
|
403
418
|
for trace in done:
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
419
|
+
stats = completed[trace.id]
|
|
420
|
+
total_in += stats.num_input_tokens
|
|
421
|
+
total_out += stats.num_output_tokens
|
|
422
|
+
usage = stats.usage
|
|
407
423
|
if usage is not None:
|
|
408
424
|
if usage.cached_input_tokens is not None:
|
|
409
425
|
total_cached += usage.cached_input_tokens
|
|
@@ -415,7 +431,7 @@ def _breakdown(scored: list[Trace], done: list[Trace]) -> Table | None:
|
|
|
415
431
|
total_cost += usage.cost
|
|
416
432
|
have_cost = True
|
|
417
433
|
# Judge / auxiliary scoring calls (off the message graph) shown separately from the agent's.
|
|
418
|
-
judge =
|
|
434
|
+
judge = stats.judge_usage
|
|
419
435
|
if judge is not None:
|
|
420
436
|
total_judge_in += judge.input_tokens
|
|
421
437
|
total_judge_out += judge.completion_tokens
|
|
@@ -530,7 +546,12 @@ def _brace(i: int, size: int) -> str:
|
|
|
530
546
|
return "╭" if i == 0 else "╰" if i == size - 1 else "│"
|
|
531
547
|
|
|
532
548
|
|
|
533
|
-
def Rows(
|
|
549
|
+
def Rows(
|
|
550
|
+
groups: list[list[RunSlot]],
|
|
551
|
+
now: float,
|
|
552
|
+
runtime_type: str,
|
|
553
|
+
completed: dict[str, TraceStats],
|
|
554
|
+
) -> Table:
|
|
534
555
|
# (brace, state, left sections, result, time); a slot contributes one row per live
|
|
535
556
|
# trace (a multi-agent episode shows each role's trace), braced per task.
|
|
536
557
|
rows: list[tuple[str, str, list[str], str, str]] = []
|
|
@@ -596,7 +617,8 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
596
617
|
runtime = f"{rt.type}({rt.id})" if rt.id else rt.type
|
|
597
618
|
else:
|
|
598
619
|
runtime = runtime_type
|
|
599
|
-
|
|
620
|
+
stats = completed[t.id] if slot.done else TraceStats(t)
|
|
621
|
+
turns = stats.num_turns
|
|
600
622
|
start = t.timing.boot.start or t.timing.setup.start
|
|
601
623
|
end = (
|
|
602
624
|
t.timing.scoring.end
|
|
@@ -611,9 +633,9 @@ def Rows(groups: list[list[RunSlot]], now: float, runtime_type: str) -> Table:
|
|
|
611
633
|
)
|
|
612
634
|
or now
|
|
613
635
|
)
|
|
614
|
-
prompt, completion =
|
|
615
|
-
nbranches =
|
|
616
|
-
usage =
|
|
636
|
+
prompt, completion = stats.num_input_tokens, stats.num_output_tokens
|
|
637
|
+
nbranches = stats.num_branches
|
|
638
|
+
usage = stats.usage
|
|
617
639
|
cached = usage.cached_input_tokens if usage else None
|
|
618
640
|
reasoning = usage.reasoning_tokens if usage else None
|
|
619
641
|
cost = usage.cost if usage else None
|
|
@@ -786,16 +808,26 @@ def _render(
|
|
|
786
808
|
pager: Pager,
|
|
787
809
|
header: Group | Table,
|
|
788
810
|
runtime_type: str,
|
|
811
|
+
completed: dict[str, TraceStats],
|
|
789
812
|
push: "PushState | None" = None,
|
|
790
813
|
tail: LogTail | None = None,
|
|
791
814
|
) -> Group:
|
|
792
815
|
now = time.time()
|
|
816
|
+
# A trace can finish before other agents and env scoring update the episode.
|
|
817
|
+
# Only done slots are final; resumed episodes enter the view already done.
|
|
818
|
+
completed.update(
|
|
819
|
+
(trace.id, TraceStats(trace))
|
|
820
|
+
for slot in slots
|
|
821
|
+
if slot.done
|
|
822
|
+
for trace in slot.traces
|
|
823
|
+
if trace.id not in completed
|
|
824
|
+
)
|
|
793
825
|
# The --push status line (and, on Ctrl-C, the cleanup notice) appear under the rollouts. Measure
|
|
794
826
|
# the fixed top (header + progress + rule) and the footer so the rollout rows fill what's left;
|
|
795
827
|
# page through them (timer / arrows) when they'd overflow (else rich truncates).
|
|
796
828
|
footers = [f for f in (_push_footer(push), _interrupt_footer()) if f is not None]
|
|
797
829
|
footer = Group(*footers) if footers else None
|
|
798
|
-
progress = Progress(slots, start)
|
|
830
|
+
progress = Progress(slots, start, completed)
|
|
799
831
|
top = Group(header, progress, Rule(style="dim"))
|
|
800
832
|
reserved = len(_CONSOLE.render_lines(top))
|
|
801
833
|
if footer is not None:
|
|
@@ -813,12 +845,12 @@ def _render(
|
|
|
813
845
|
return Group(*parts)
|
|
814
846
|
page_groups, index, count = _paginate(_groups(slots), rows_per_page, pager, now)
|
|
815
847
|
if count > 1:
|
|
816
|
-
progress = Progress(slots, start, page=(index + 1, count))
|
|
848
|
+
progress = Progress(slots, start, completed, page=(index + 1, count))
|
|
817
849
|
parts = [
|
|
818
850
|
header,
|
|
819
851
|
progress,
|
|
820
852
|
Rule(style="dim"),
|
|
821
|
-
Rows(page_groups, now, runtime_type),
|
|
853
|
+
Rows(page_groups, now, runtime_type, completed),
|
|
822
854
|
]
|
|
823
855
|
if footer is not None:
|
|
824
856
|
parts.append(footer)
|
|
@@ -833,6 +865,7 @@ async def dashboard(
|
|
|
833
865
|
push: "PushState | None" = None,
|
|
834
866
|
):
|
|
835
867
|
pager = Pager()
|
|
868
|
+
completed: dict[str, TraceStats] = {}
|
|
836
869
|
warning = _warning(config)
|
|
837
870
|
header = Group(warning, Text(""), Overview(config)) if warning else Overview(config)
|
|
838
871
|
runtime_type = (
|
|
@@ -850,7 +883,9 @@ async def dashboard(
|
|
|
850
883
|
else None
|
|
851
884
|
)
|
|
852
885
|
async with live_view(
|
|
853
|
-
lambda: _render(
|
|
886
|
+
lambda: _render(
|
|
887
|
+
slots, start, pager, header, runtime_type, completed, push, tail
|
|
888
|
+
),
|
|
854
889
|
on_key=None if tail is not None else pager.on_key,
|
|
855
890
|
):
|
|
856
891
|
yield
|
verifiers/v1/utils/artifacts.py
CHANGED
|
@@ -91,12 +91,33 @@ async def collect(
|
|
|
91
91
|
raise RuntimeError(f"artifact {artifact.source!r} declared more than once")
|
|
92
92
|
seen.add(artifact.source)
|
|
93
93
|
|
|
94
|
+
# Batch roots to save remote round trips. Leave room below the shell argument
|
|
95
|
+
# limit for the command itself and further quoting by runtime transports.
|
|
96
|
+
batches: list[list[str]] = [[]]
|
|
97
|
+
batch_bytes = 0
|
|
98
|
+
for artifact in entries:
|
|
99
|
+
source_bytes = len(shlex.quote(artifact.source).encode()) + 1
|
|
100
|
+
if batches[-1] and batch_bytes + source_bytes > 8 * 1024:
|
|
101
|
+
batches.append([])
|
|
102
|
+
batch_bytes = 0
|
|
103
|
+
batches[-1].append(artifact.source)
|
|
104
|
+
batch_bytes += source_bytes
|
|
105
|
+
|
|
106
|
+
existence: list[str] = []
|
|
107
|
+
for sources in batches:
|
|
108
|
+
output = await _run(
|
|
109
|
+
runtime,
|
|
110
|
+
f"for source in {shlex.join(sources)}; do "
|
|
111
|
+
'if test -e "$source" || test -L "$source"; then echo 1; else echo 0; fi; '
|
|
112
|
+
"done",
|
|
113
|
+
"check artifact roots",
|
|
114
|
+
)
|
|
115
|
+
existence.extend(output.splitlines())
|
|
94
116
|
collected: dict[str, bytes | None] = {}
|
|
95
117
|
budget = MAX_ARTIFACT_BYTES
|
|
96
|
-
for artifact in entries:
|
|
118
|
+
for artifact, exists in zip(entries, existence, strict=True):
|
|
97
119
|
source = artifact.source
|
|
98
|
-
exists
|
|
99
|
-
if (await runtime.run(["sh", "-c", exists], {})).exit_code != 0:
|
|
120
|
+
if exists != "1":
|
|
100
121
|
if not artifact.required:
|
|
101
122
|
collected[source] = None
|
|
102
123
|
continue
|
|
@@ -205,8 +226,9 @@ def _validate_restore(root: str, archive: bytes | None) -> None:
|
|
|
205
226
|
raise RuntimeError(f"unreadable artifact archive for {root!r}: {exc}") from exc
|
|
206
227
|
|
|
207
228
|
|
|
208
|
-
async def _run(runtime: Runtime, command: str, action: str) ->
|
|
229
|
+
async def _run(runtime: Runtime, command: str, action: str) -> str:
|
|
209
230
|
result = await runtime.run(["sh", "-c", command], {})
|
|
210
231
|
if result.exit_code:
|
|
211
232
|
detail = (result.stderr or result.stdout).strip()[-500:]
|
|
212
233
|
raise RuntimeError(f"failed to {action}: {detail}")
|
|
234
|
+
return result.stdout
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: verifiers
|
|
3
|
-
Version: 0.3.2.
|
|
3
|
+
Version: 0.3.2.dev64
|
|
4
4
|
Summary: Verifiers: Environments for LLM Reinforcement Learning
|
|
5
5
|
Project-URL: Homepage, https://github.com/primeintellect-ai/verifiers
|
|
6
6
|
Project-URL: Documentation, https://github.com/primeintellect-ai/verifiers
|
|
@@ -29,7 +29,7 @@ verifiers/v1/cli/resume.py,sha256=hqV5AxxqeIInJeEZQerQznuijhMNijFEFcHZIltfLdE,47
|
|
|
29
29
|
verifiers/v1/cli/validate.py,sha256=G3YXZTwxVDQIyR_ASapJceixOxRSJl69NnFPBqkF3N0,17173
|
|
30
30
|
verifiers/v1/cli/dashboard/__init__.py,sha256=v-baMxQuWxOCsbU7-p_jj2Q9BUnTN-TWi1q2hK6rU2s,198
|
|
31
31
|
verifiers/v1/cli/dashboard/base.py,sha256=kUP93zJSIVLptSbnWX6MOy8kGg63fpeAzPqakCpPyDc,3547
|
|
32
|
-
verifiers/v1/cli/dashboard/eval.py,sha256=
|
|
32
|
+
verifiers/v1/cli/dashboard/eval.py,sha256=hr2r54wvcMYw-CEjDsVxrBQgZGp9cGrgrsRtjaaSNPs,37822
|
|
33
33
|
verifiers/v1/cli/dashboard/replay.py,sha256=_X5up9MJsRbd_sWe4LgNTPNT8k1ip4xNpo50hxI3Ljk,2697
|
|
34
34
|
verifiers/v1/cli/dashboard/validate.py,sha256=xrpK3Y90JsoDCQYLLvvsnJ4CPRMvSSJIghfHK62E6FQ,3687
|
|
35
35
|
verifiers/v1/cli/eval/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
@@ -172,7 +172,7 @@ verifiers/v1/tasksets/textarena/__init__.py,sha256=Os2OlBY_pSH0B1DP32RitumgSLpq2
|
|
|
172
172
|
verifiers/v1/tasksets/textarena/taskset.py,sha256=9fa_unlFiFZyuN4I02rW3x4qyRnFWXqwTUZTloQefu0,4433
|
|
173
173
|
verifiers/v1/utils/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
174
174
|
verifiers/v1/utils/aio.py,sha256=0yNFOqB2oO6Qktc6NI_ZSAGuo1aHsIK_wuT1BtmoZ3I,1532
|
|
175
|
-
verifiers/v1/utils/artifacts.py,sha256=
|
|
175
|
+
verifiers/v1/utils/artifacts.py,sha256=PFrSkwAyvzHyT6JjlFeGOPIXDmS-uRajaN3am4HZ0TI,9263
|
|
176
176
|
verifiers/v1/utils/compile.py,sha256=5pxy31wx2Q-3pQjOO5B9SH6OT-wwb8r4F1LAl6HMeQA,5835
|
|
177
177
|
verifiers/v1/utils/decorators.py,sha256=vhoN6YoKb15-Fkqmc9RaNlUIbt9WRvNK4zAn2bAxfS4,5367
|
|
178
178
|
verifiers/v1/utils/format.py,sha256=NfQo9M5KMzWCZpQaet5YtsJx56vCKAZPfbjZZJLJhhM,2187
|
|
@@ -189,8 +189,8 @@ verifiers/v1/utils/prime.py,sha256=UTYRjp9cbjNb6CVmBHda-1wWZAIIfxNOmTyNuT7_wL4,9
|
|
|
189
189
|
verifiers/v1/utils/retries.py,sha256=Y2ZgrAjn-qNkRKZP_RVNL_7EK0iaRcZeCa404lpGYi4,5417
|
|
190
190
|
verifiers/v1/utils/score.py,sha256=493yJVMw8teCu9JxapxMFPFzI0hNUqdo0Y2nGW4kckk,6200
|
|
191
191
|
verifiers/v1/utils/version.py,sha256=-obEo_-l9-D8FLef4hYxncOe-uJpxrM1g2Hig_37Sgs,1607
|
|
192
|
-
verifiers-0.3.2.
|
|
193
|
-
verifiers-0.3.2.
|
|
194
|
-
verifiers-0.3.2.
|
|
195
|
-
verifiers-0.3.2.
|
|
196
|
-
verifiers-0.3.2.
|
|
192
|
+
verifiers-0.3.2.dev64.dist-info/METADATA,sha256=uYX4LJNV3yeutj3Pwcw8IU3Fbz-CYNSzBwOs5vX50nY,4204
|
|
193
|
+
verifiers-0.3.2.dev64.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
194
|
+
verifiers-0.3.2.dev64.dist-info/entry_points.txt,sha256=uqQje0TMsr7k6nXlWxlPs13Si0evb3pVCEDF6c-YL80,241
|
|
195
|
+
verifiers-0.3.2.dev64.dist-info/licenses/LICENSE,sha256=v0RrUsdV3IDoZhrRce297IXS3xMHNJ-_LdLpFAUWb9k,1072
|
|
196
|
+
verifiers-0.3.2.dev64.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|