whileai 0.56__tar.gz → 0.57__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {whileai-0.56 → whileai-0.57}/PKG-INFO +34 -5
- whileai-0.56/whileai.egg-info/PKG-INFO → whileai-0.57/README.md +32 -39
- {whileai-0.56 → whileai-0.57}/pyproject.toml +2 -1
- {whileai-0.56 → whileai-0.57}/whileai/simulations/__init__.py +4 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/data.py +1 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/run/engine.py +21 -0
- whileai-0.57/whileai/simulations/score/audit.py +296 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/hygiene.py +70 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/judging.py +36 -1
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/optimize.py +18 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/simulation.py +54 -0
- whileai-0.56/README.md → whileai-0.57/whileai.egg-info/PKG-INFO +68 -3
- {whileai-0.56 → whileai-0.57}/whileai.egg-info/SOURCES.txt +1 -0
- {whileai-0.56 → whileai-0.57}/LICENSE +0 -0
- {whileai-0.56 → whileai-0.57}/MANIFEST.in +0 -0
- {whileai-0.56 → whileai-0.57}/setup.cfg +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/_env.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/auth.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/cli.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/ingest.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/py.typed +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/__main__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/environment.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/export.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/actionspace.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/adapters.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/agents.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/coverage.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/diversity.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/embeddings.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/explore.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/generator.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/offline_agent.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/scenarios.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/generate/usage_meter.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/ingest/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/ingest/otel.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/ingest/platform.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/ingest/traces.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/monitor.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/py.typed +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/run/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/run/config.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/run/rows.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/run/spec.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/schema.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/schemas/row-v1.json +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/agreement.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/checklist.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/curriculum.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/delta.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/grade_llm.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/grading.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/grounding.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/hack_scan.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/judge_trust.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/labels.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/llm_judge.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/logprobs.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/markers.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/pairwise.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/passat.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/preflight.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/privileged.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/publish_gate.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/quality.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/reference.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/rubric.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/spec.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/stage.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/stats.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/style.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/score/trace.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/training.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/base.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/code.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/math.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/structured.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/verify/text.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/world/__init__.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai/simulations/world/sandbox.py +0 -0
- {whileai-0.56 → whileai-0.57}/whileai.egg-info/dependency_links.txt +0 -0
- {whileai-0.56 → whileai-0.57}/whileai.egg-info/entry_points.txt +0 -0
- {whileai-0.56 → whileai-0.57}/whileai.egg-info/requires.txt +0 -0
- {whileai-0.56 → whileai-0.57}/whileai.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: whileai
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.57
|
|
4
4
|
Summary: While Python SDK: trace ingestion and verification client, plus agent simulations grounded in your tools and system prompt.
|
|
5
5
|
Author: While
|
|
6
6
|
License: Apache-2.0
|
|
@@ -8,7 +8,7 @@ Project-URL: Homepage, https://github.com/whilehq/whileai-sdk
|
|
|
8
8
|
Project-URL: Repository, https://github.com/whilehq/whileai-sdk
|
|
9
9
|
Project-URL: Issues, https://github.com/whilehq/whileai-sdk/issues
|
|
10
10
|
Project-URL: Datasets, https://huggingface.co/datasets/zero-proof-ai/agent-simulations
|
|
11
|
-
Keywords: rl,reinforcement-learning,grpo,synthetic-data,agents,tool-use,evaluation,simulation,llm
|
|
11
|
+
Keywords: zp,zeroproof,while,evals,agent-evals,rl,reinforcement-learning,grpo,synthetic-data,agents,tool-use,evaluation,simulation,llm
|
|
12
12
|
Classifier: Development Status :: 3 - Alpha
|
|
13
13
|
Classifier: Intended Audience :: Developers
|
|
14
14
|
Classifier: Intended Audience :: Science/Research
|
|
@@ -41,7 +41,7 @@ The While Python SDK. One package, two importable modules:
|
|
|
41
41
|
- `whileai`: the platform client. OTLP trace ingest and trace-dataset listing against the token gate.
|
|
42
42
|
- `whileai.simulations`: post-training data for an agent. Give it the agent's traces, or its tools and system prompt; it simulates the situations, the people, and the world, plays the agent through multi-turn tool-calling conversations, and returns rows for your grader.
|
|
43
43
|
|
|
44
|
-
**Renamed.** This SDK was `zeroproof` (ZeroProof is now While). `pip install zeroproof` still works: it installs `whileai`, and `import zeroproof` (or the older `zeroproof_simulations`) resolves to the same modules with a deprecation warning. `ZEROPROOF_*` environment variables and a saved `~/.zeroproof/credentials.json` are still read. Change the import when you can; new releases land under `whileai`.
|
|
44
|
+
**Renamed.** This SDK was `zeroproof` (ZeroProof is now While). `pip install zeroproof` still works: it installs `whileai`, and `import zeroproof` (or the older `zeroproof_simulations`) resolves to the same modules with a deprecation warning. `ZEROPROOF_*` environment variables and a saved `~/.zeroproof/credentials.json` are still read. Change the import when you can; new releases land under `whileai`. zp, ZeroProof and While all name this one product: the package is `whileai`, the import is `whileai.simulations`, keys start with `zp_`. A machine with the old package still picks up `~/.zeroproof/credentials.json`; set `WHILEAI_HOME` to a fresh directory to isolate a new account from it.
|
|
45
45
|
|
|
46
46
|
Releases of `whileai` before 0.3 were an unrelated encrypted agent-to-agent messaging client. That code was removed in 0.04; pin `whileai<0.3` if you still depend on it.
|
|
47
47
|
|
|
@@ -187,6 +187,34 @@ and runs in seconds, but the situations are less varied than a model writes,
|
|
|
187
187
|
so it is for wiring up your agent and grader, not for a training set — for
|
|
188
188
|
that, bring a model below.
|
|
189
189
|
|
|
190
|
+
### Evals for the agent you already have
|
|
191
|
+
|
|
192
|
+
Not training anything yet? The shortest path is an eval: wrap your agent
|
|
193
|
+
as `agent(message) -> {steps, final_text}`, write the policy as a judge
|
|
194
|
+
that reads the trajectory, run the asks `k` times each, and read pass@1
|
|
195
|
+
with its interval. Offline first, then the hosted writer. The how-to is
|
|
196
|
+
[docs/evals.md](docs/evals.md); the runnable version is
|
|
197
|
+
[`recipes/02-measure/eval-your-agent`](recipes/02-measure/eval-your-agent),
|
|
198
|
+
which ends at a CI gate, not a push.
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
data = wai.simulate(
|
|
202
|
+
agent,
|
|
203
|
+
tools=TOOLS,
|
|
204
|
+
system_prompt=POLICY,
|
|
205
|
+
seeds=SEEDS,
|
|
206
|
+
simulator=False,
|
|
207
|
+
mode="rl",
|
|
208
|
+
repeats=4,
|
|
209
|
+
repeat_policy="fixed",
|
|
210
|
+
)
|
|
211
|
+
scored = wai.evaluate(data, judge) # eval lineage: never the reward
|
|
212
|
+
print(wai.pass_at(scored.rows), *scored.warnings) # a hollow run says so here
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
`scored.warnings` is new: no rollout called a tool, a declared tool no
|
|
216
|
+
rollout touched, a marker that fired on no row. A 1.00 on a run like that is not a result; the note names the fix.
|
|
217
|
+
|
|
190
218
|
### Bring a model
|
|
191
219
|
|
|
192
220
|
Bring your own model. Any OpenAI-compatible chat endpoint that returns tool
|
|
@@ -421,8 +449,8 @@ When the rows carry none of that metadata the export warns: the reward
|
|
|
421
449
|
reduces to `conduct_grade`, a process reward, and a policy trained on it
|
|
422
450
|
alone learns to call nothing (`recipes/03-select/prime-intellect-rl`). `wai.load_environment(spec)`
|
|
423
451
|
builds the environment in a process that has `verifiers` (`pip install
|
|
424
|
-
'whileai[rl]'`);
|
|
425
|
-
hand over an executable world with a hidden test suite.
|
|
452
|
+
'whileai[rl]'`); the [tool-call-efficiency](https://huggingface.co/datasets/zero-proof-ai/tool-call-efficiency)
|
|
453
|
+
dataset is the same shape built by hand over an executable world with a hidden test suite.
|
|
426
454
|
|
|
427
455
|
Training notes, each with the chapter of rlhfbook.com behind it. Calibrate
|
|
428
456
|
difficulty with 8 to 16 rollouts per task before exporting so the band is
|
|
@@ -626,6 +654,7 @@ network.
|
|
|
626
654
|
| Simulate and grade | [`recipes/01-simulate/agent-behavior`](recipes/01-simulate/agent-behavior) | Start here if the platform is new to you. Runs a coding agent with bad habits against real tests, streams every turn to While as OTLP spans plus a judge verdict, and fills a dashboard with behaviour worth looking at. Needs a key and a model endpoint; stdlib only. |
|
|
627
655
|
| Simulate and grade | [`recipes/01-simulate/verifiers`](recipes/01-simulate/verifiers) | Verifiable rewards: math (`MathEqual`), an answer-and-format gate (`All`), code run against hidden tests (`CodeExec`), and a JSON-schema check, each feeding `grade`/`optimize`. Offline. |
|
|
628
656
|
| Measure | [`recipes/02-measure/pass-at-k`](recipes/02-measure/pass-at-k) | pass@1, pass^k and pass@k with their intervals for one agent, the per-ask histogram the mean hides, and what each number tells you to do next. Offline. |
|
|
657
|
+
| Measure | [`recipes/02-measure/eval-your-agent`](recipes/02-measure/eval-your-agent) | Evals for the agent you already have: the callable wrapper, the policy as a judge that reads the trajectory, pass@1 with an interval and pass^k per policy branch, the coverage warnings that catch a hollow run, and a CI gate. Two scripted refund bots, one careful and one eager, so the eval visibly separates them. Offline, seconds. How-to: [docs/evals.md](docs/evals.md). |
|
|
629
658
|
| Measure | [`recipes/02-measure/reward-hacking`](recipes/02-measure/reward-hacking) | Reward hacking caught before, during and after training: the within-ask scan, the judge probes, the trajectory flags, and the proxy-vs-target verdict on a scripted agent and two judges. Offline, seconds, no key. How-to: [docs/reward-hacking.md](docs/reward-hacking.md). |
|
|
630
659
|
| Measure | [`recipes/02-measure/safety-evals`](recipes/02-measure/safety-evals) | Safety evals for a tool-using agent: prompt injection (direct, and planted in a tool result), data exfiltration, secret leakage, unauthorized writes, plus the benign controls that catch over-refusal. Trajectory markers as the judge, pass^k per attack class, the judge checked against hand labels, and a before/after that fails the fix which got safe by refusing. Offline, seconds. How-to: [docs/safety-evals.md](docs/safety-evals.md). |
|
|
631
660
|
| Measure | [`recipes/02-measure/safety-evals-marketplace`](recipes/02-measure/safety-evals-marketplace) | The same safety eval for a marketplace agent: the injection is planted in user-generated reviews, the private data is per tenant (a competitor's buyer-intent list), one of the writes is a public post, and a flag needs a moderation ticket. Six trajectory markers, pass^k per attack class, the guarded before/after, and `live.py` to run the suite on a real model through Ollama with no key. Offline, seconds. |
|
|
@@ -1,39 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: whileai
|
|
3
|
-
Version: 0.56
|
|
4
|
-
Summary: While Python SDK: trace ingestion and verification client, plus agent simulations grounded in your tools and system prompt.
|
|
5
|
-
Author: While
|
|
6
|
-
License: Apache-2.0
|
|
7
|
-
Project-URL: Homepage, https://github.com/whilehq/whileai-sdk
|
|
8
|
-
Project-URL: Repository, https://github.com/whilehq/whileai-sdk
|
|
9
|
-
Project-URL: Issues, https://github.com/whilehq/whileai-sdk/issues
|
|
10
|
-
Project-URL: Datasets, https://huggingface.co/datasets/zero-proof-ai/agent-simulations
|
|
11
|
-
Keywords: rl,reinforcement-learning,grpo,synthetic-data,agents,tool-use,evaluation,simulation,llm
|
|
12
|
-
Classifier: Development Status :: 3 - Alpha
|
|
13
|
-
Classifier: Intended Audience :: Developers
|
|
14
|
-
Classifier: Intended Audience :: Science/Research
|
|
15
|
-
Classifier: License :: OSI Approved :: Apache Software License
|
|
16
|
-
Classifier: Programming Language :: Python :: 3
|
|
17
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
-
Classifier: Typing :: Typed
|
|
23
|
-
Requires-Python: >=3.10
|
|
24
|
-
Description-Content-Type: text/markdown
|
|
25
|
-
License-File: LICENSE
|
|
26
|
-
Requires-Dist: requests>=2.25.0
|
|
27
|
-
Provides-Extra: dev
|
|
28
|
-
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
29
|
-
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
30
|
-
Requires-Dist: ruff>=0.12; extra == "dev"
|
|
31
|
-
Requires-Dist: mypy>=1.16; extra == "dev"
|
|
32
|
-
Requires-Dist: types-requests; extra == "dev"
|
|
33
|
-
Provides-Extra: rl
|
|
34
|
-
Requires-Dist: verifiers>=0.3; (python_version >= "3.11" and python_version < "3.14") and extra == "rl"
|
|
35
|
-
Dynamic: license-file
|
|
36
|
-
|
|
37
1
|
# whileai
|
|
38
2
|
|
|
39
3
|
The While Python SDK. One package, two importable modules:
|
|
@@ -41,7 +5,7 @@ The While Python SDK. One package, two importable modules:
|
|
|
41
5
|
- `whileai`: the platform client. OTLP trace ingest and trace-dataset listing against the token gate.
|
|
42
6
|
- `whileai.simulations`: post-training data for an agent. Give it the agent's traces, or its tools and system prompt; it simulates the situations, the people, and the world, plays the agent through multi-turn tool-calling conversations, and returns rows for your grader.
|
|
43
7
|
|
|
44
|
-
**Renamed.** This SDK was `zeroproof` (ZeroProof is now While). `pip install zeroproof` still works: it installs `whileai`, and `import zeroproof` (or the older `zeroproof_simulations`) resolves to the same modules with a deprecation warning. `ZEROPROOF_*` environment variables and a saved `~/.zeroproof/credentials.json` are still read. Change the import when you can; new releases land under `whileai`.
|
|
8
|
+
**Renamed.** This SDK was `zeroproof` (ZeroProof is now While). `pip install zeroproof` still works: it installs `whileai`, and `import zeroproof` (or the older `zeroproof_simulations`) resolves to the same modules with a deprecation warning. `ZEROPROOF_*` environment variables and a saved `~/.zeroproof/credentials.json` are still read. Change the import when you can; new releases land under `whileai`. zp, ZeroProof and While all name this one product: the package is `whileai`, the import is `whileai.simulations`, keys start with `zp_`. A machine with the old package still picks up `~/.zeroproof/credentials.json`; set `WHILEAI_HOME` to a fresh directory to isolate a new account from it.
|
|
45
9
|
|
|
46
10
|
Releases of `whileai` before 0.3 were an unrelated encrypted agent-to-agent messaging client. That code was removed in 0.04; pin `whileai<0.3` if you still depend on it.
|
|
47
11
|
|
|
@@ -187,6 +151,34 @@ and runs in seconds, but the situations are less varied than a model writes,
|
|
|
187
151
|
so it is for wiring up your agent and grader, not for a training set — for
|
|
188
152
|
that, bring a model below.
|
|
189
153
|
|
|
154
|
+
### Evals for the agent you already have
|
|
155
|
+
|
|
156
|
+
Not training anything yet? The shortest path is an eval: wrap your agent
|
|
157
|
+
as `agent(message) -> {steps, final_text}`, write the policy as a judge
|
|
158
|
+
that reads the trajectory, run the asks `k` times each, and read pass@1
|
|
159
|
+
with its interval. Offline first, then the hosted writer. The how-to is
|
|
160
|
+
[docs/evals.md](docs/evals.md); the runnable version is
|
|
161
|
+
[`recipes/02-measure/eval-your-agent`](recipes/02-measure/eval-your-agent),
|
|
162
|
+
which ends at a CI gate, not a push.
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
data = wai.simulate(
|
|
166
|
+
agent,
|
|
167
|
+
tools=TOOLS,
|
|
168
|
+
system_prompt=POLICY,
|
|
169
|
+
seeds=SEEDS,
|
|
170
|
+
simulator=False,
|
|
171
|
+
mode="rl",
|
|
172
|
+
repeats=4,
|
|
173
|
+
repeat_policy="fixed",
|
|
174
|
+
)
|
|
175
|
+
scored = wai.evaluate(data, judge) # eval lineage: never the reward
|
|
176
|
+
print(wai.pass_at(scored.rows), *scored.warnings) # a hollow run says so here
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
`scored.warnings` is new: no rollout called a tool, a declared tool no
|
|
180
|
+
rollout touched, a marker that fired on no row. A 1.00 on a run like that is not a result; the note names the fix.
|
|
181
|
+
|
|
190
182
|
### Bring a model
|
|
191
183
|
|
|
192
184
|
Bring your own model. Any OpenAI-compatible chat endpoint that returns tool
|
|
@@ -421,8 +413,8 @@ When the rows carry none of that metadata the export warns: the reward
|
|
|
421
413
|
reduces to `conduct_grade`, a process reward, and a policy trained on it
|
|
422
414
|
alone learns to call nothing (`recipes/03-select/prime-intellect-rl`). `wai.load_environment(spec)`
|
|
423
415
|
builds the environment in a process that has `verifiers` (`pip install
|
|
424
|
-
'whileai[rl]'`);
|
|
425
|
-
hand over an executable world with a hidden test suite.
|
|
416
|
+
'whileai[rl]'`); the [tool-call-efficiency](https://huggingface.co/datasets/zero-proof-ai/tool-call-efficiency)
|
|
417
|
+
dataset is the same shape built by hand over an executable world with a hidden test suite.
|
|
426
418
|
|
|
427
419
|
Training notes, each with the chapter of rlhfbook.com behind it. Calibrate
|
|
428
420
|
difficulty with 8 to 16 rollouts per task before exporting so the band is
|
|
@@ -626,6 +618,7 @@ network.
|
|
|
626
618
|
| Simulate and grade | [`recipes/01-simulate/agent-behavior`](recipes/01-simulate/agent-behavior) | Start here if the platform is new to you. Runs a coding agent with bad habits against real tests, streams every turn to While as OTLP spans plus a judge verdict, and fills a dashboard with behaviour worth looking at. Needs a key and a model endpoint; stdlib only. |
|
|
627
619
|
| Simulate and grade | [`recipes/01-simulate/verifiers`](recipes/01-simulate/verifiers) | Verifiable rewards: math (`MathEqual`), an answer-and-format gate (`All`), code run against hidden tests (`CodeExec`), and a JSON-schema check, each feeding `grade`/`optimize`. Offline. |
|
|
628
620
|
| Measure | [`recipes/02-measure/pass-at-k`](recipes/02-measure/pass-at-k) | pass@1, pass^k and pass@k with their intervals for one agent, the per-ask histogram the mean hides, and what each number tells you to do next. Offline. |
|
|
621
|
+
| Measure | [`recipes/02-measure/eval-your-agent`](recipes/02-measure/eval-your-agent) | Evals for the agent you already have: the callable wrapper, the policy as a judge that reads the trajectory, pass@1 with an interval and pass^k per policy branch, the coverage warnings that catch a hollow run, and a CI gate. Two scripted refund bots, one careful and one eager, so the eval visibly separates them. Offline, seconds. How-to: [docs/evals.md](docs/evals.md). |
|
|
629
622
|
| Measure | [`recipes/02-measure/reward-hacking`](recipes/02-measure/reward-hacking) | Reward hacking caught before, during and after training: the within-ask scan, the judge probes, the trajectory flags, and the proxy-vs-target verdict on a scripted agent and two judges. Offline, seconds, no key. How-to: [docs/reward-hacking.md](docs/reward-hacking.md). |
|
|
630
623
|
| Measure | [`recipes/02-measure/safety-evals`](recipes/02-measure/safety-evals) | Safety evals for a tool-using agent: prompt injection (direct, and planted in a tool result), data exfiltration, secret leakage, unauthorized writes, plus the benign controls that catch over-refusal. Trajectory markers as the judge, pass^k per attack class, the judge checked against hand labels, and a before/after that fails the fix which got safe by refusing. Offline, seconds. How-to: [docs/safety-evals.md](docs/safety-evals.md). |
|
|
631
624
|
| Measure | [`recipes/02-measure/safety-evals-marketplace`](recipes/02-measure/safety-evals-marketplace) | The same safety eval for a marketplace agent: the injection is planted in user-generated reviews, the private data is per tenant (a competitor's buyer-intent list), one of the writes is a public post, and a flag needs a moderation ticket. Six trajectory markers, pass^k per attack class, the guarded before/after, and `live.py` to run the suite on a real model through Ollama with no key. Offline, seconds. |
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "whileai"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.57"
|
|
8
8
|
description = "While Python SDK: trace ingestion and verification client, plus agent simulations grounded in your tools and system prompt."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "Apache-2.0"}
|
|
@@ -12,6 +12,7 @@ requires-python = ">=3.10"
|
|
|
12
12
|
dependencies = ["requests>=2.25.0"]
|
|
13
13
|
authors = [{name = "While"}]
|
|
14
14
|
keywords = [
|
|
15
|
+
"zp", "zeroproof", "while", "evals", "agent-evals",
|
|
15
16
|
"rl", "reinforcement-learning", "grpo", "synthetic-data", "agents",
|
|
16
17
|
"tool-use", "evaluation", "simulation", "llm",
|
|
17
18
|
]
|
|
@@ -102,6 +102,7 @@ from .schema import (
|
|
|
102
102
|
validate,
|
|
103
103
|
)
|
|
104
104
|
from .score.agreement import judge_agreement
|
|
105
|
+
from .score.audit import audit_grades, format_audit
|
|
105
106
|
from .score.checklist import expected_outcome, outcome_check, privileged_context, task_checklist
|
|
106
107
|
from .score.curriculum import curriculum, format_curriculum, retire_solved
|
|
107
108
|
from .score.delta import delta_report, format_delta_report
|
|
@@ -115,6 +116,7 @@ from .score.grounding import (
|
|
|
115
116
|
from .score.hack_scan import format_hack_scan, format_hack_scan_diff, hack_scan, hack_scan_diff
|
|
116
117
|
from .score.hygiene import (
|
|
117
118
|
HACK_THRESHOLD,
|
|
119
|
+
coverage_warnings,
|
|
118
120
|
dedupe_groups,
|
|
119
121
|
length_report,
|
|
120
122
|
near_duplicate_prompts,
|
|
@@ -235,6 +237,7 @@ __all__ = [
|
|
|
235
237
|
"attach_holdout",
|
|
236
238
|
"attach_labels",
|
|
237
239
|
"attach_rubric",
|
|
240
|
+
"audit_grades",
|
|
238
241
|
"behavior_signature",
|
|
239
242
|
"behavioral_markers",
|
|
240
243
|
"build_dimensions",
|
|
@@ -273,6 +276,7 @@ __all__ = [
|
|
|
273
276
|
"export_training",
|
|
274
277
|
"filter_rl_rows",
|
|
275
278
|
"flaw_rows",
|
|
279
|
+
"format_audit",
|
|
276
280
|
"format_curriculum",
|
|
277
281
|
"format_delta_report",
|
|
278
282
|
"format_hack_monitor",
|
|
@@ -2851,6 +2851,27 @@ class Run:
|
|
|
2851
2851
|
data.degraded.append("same_model")
|
|
2852
2852
|
data.warnings.append(note)
|
|
2853
2853
|
log.warning(note)
|
|
2854
|
+
# A run whose rollouts never called a tool is hollow: the writer
|
|
2855
|
+
# asked about things the world does not have, or the wrapper did
|
|
2856
|
+
# not record steps. Grading it gives a number that means nothing.
|
|
2857
|
+
rows = data.trajectories
|
|
2858
|
+
if rows and data.declared_tools:
|
|
2859
|
+
with_calls = sum(
|
|
2860
|
+
1
|
|
2861
|
+
for r in rows
|
|
2862
|
+
if any(isinstance(s, dict) and s.get("tool") for s in (r.get("steps") or []))
|
|
2863
|
+
)
|
|
2864
|
+
if with_calls == 0:
|
|
2865
|
+
note = (
|
|
2866
|
+
f"0 of {len(rows)} rollouts called a tool, so this run says nothing "
|
|
2867
|
+
"about tool use. Put the ids your world has (order numbers, account "
|
|
2868
|
+
"names) in the tool descriptions or in seeds=, and check the agent "
|
|
2869
|
+
"wrapper records its steps, before grading it."
|
|
2870
|
+
)
|
|
2871
|
+
if "no_tool_calls" not in data.degraded:
|
|
2872
|
+
data.degraded.append("no_tool_calls")
|
|
2873
|
+
data.warnings.append(note)
|
|
2874
|
+
log.warning(note)
|
|
2854
2875
|
if c.out_path is not None and data.trajectories:
|
|
2855
2876
|
data.save(str(c.out_path), meta=True)
|
|
2856
2877
|
return data
|
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
"""Is the verifier failing answers that are right?
|
|
2
|
+
|
|
3
|
+
A verifier is a rule: execution match, exact match, a regex on the final
|
|
4
|
+
number. Rules fail correct answers for reasons that have nothing to do
|
|
5
|
+
with correctness (rounding, ordering, a column named differently, an
|
|
6
|
+
equivalent row set), and every such false negative halves the
|
|
7
|
+
separation on its task: the policy is told that a right answer was
|
|
8
|
+
wrong. On 570 SEC XBRL tasks, fixing the grader moved mean reward from
|
|
9
|
+
0.20 to 0.48 with no training at all. The RLHF book treats verifiers as
|
|
10
|
+
solved (ch. 14: "a scoring function that returns a positive reward when
|
|
11
|
+
the answer is correct and 0 otherwise"); this module is our own
|
|
12
|
+
measurement of how often that function is wrong.
|
|
13
|
+
|
|
14
|
+
``audit_grades`` samples failed rows, asks a judge whether each reply is
|
|
15
|
+
in fact correct given the reference, and reports the false-negative rate
|
|
16
|
+
with a Wilson interval and the verifier reasons that dominate. The judge
|
|
17
|
+
is a second opinion, not ground truth: the book's judge-prompt rules
|
|
18
|
+
apply (ch. 7: length must not sway it, no position bias, temperature 0
|
|
19
|
+
for stable ratings), and a rate near the threshold deserves a check of
|
|
20
|
+
the judge itself on human labels (``judge_agreement``, ``judge_trust``).
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import random
|
|
26
|
+
import re
|
|
27
|
+
from collections.abc import Callable, Sequence
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
from .judging import run_judge
|
|
31
|
+
from .stats import wilson_interval
|
|
32
|
+
|
|
33
|
+
#: above this share of audited failures, the verifier is the problem to fix first
|
|
34
|
+
FN_WARN = 0.10
|
|
35
|
+
|
|
36
|
+
#: what the judge is asked when the row has no rubric of its own
|
|
37
|
+
AUDIT_QUESTION = (
|
|
38
|
+
"The reference answer is the answer key for this task. Is the reply's final answer "
|
|
39
|
+
"correct, meaning it gives the same result as the reference? Differences in "
|
|
40
|
+
"formatting, rounding, ordering (unless the task asks for an order), column or "
|
|
41
|
+
"field naming, or an equivalent way of expressing the same result do not make it "
|
|
42
|
+
"wrong. A missing, different, or partial result does."
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
_DIGITS = re.compile(r"\d+(\.\d+)?")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _reason_key(reason: Any, verifier: str | None) -> str:
|
|
49
|
+
"""The kind of failure a verifier reason names, without its numbers:
|
|
50
|
+
``result differs: got 3 rows x 2 cols`` and ``result differs: got 10
|
|
51
|
+
rows x 1 col`` are one reason."""
|
|
52
|
+
text = str(reason or "").strip()
|
|
53
|
+
if verifier and text.lower().startswith(verifier.lower() + ":"):
|
|
54
|
+
text = text[len(verifier) + 1 :].strip()
|
|
55
|
+
head = re.split(r"[:(]", text, maxsplit=1)[0].strip()
|
|
56
|
+
head = _DIGITS.sub("#", head).lower()
|
|
57
|
+
return head or "(no reason)"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _binary(row: dict) -> int | None:
|
|
61
|
+
value = row.get("reward")
|
|
62
|
+
if value is None or isinstance(value, bool):
|
|
63
|
+
return None
|
|
64
|
+
try:
|
|
65
|
+
f = float(value)
|
|
66
|
+
except (TypeError, ValueError):
|
|
67
|
+
return None
|
|
68
|
+
if f in (0.0, 1.0):
|
|
69
|
+
return int(f)
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _audit_row(row: dict, question: str) -> dict:
|
|
74
|
+
"""The row as the judge sees it: the reply and the reference, the
|
|
75
|
+
audit question as the rubric when the row has none, and the
|
|
76
|
+
verifier's own verdict alongside so a judge that reads it can say
|
|
77
|
+
why the rule was wrong."""
|
|
78
|
+
out = dict(row)
|
|
79
|
+
priv = dict(row.get("privileged") or {})
|
|
80
|
+
if not priv.get("rubric"):
|
|
81
|
+
priv["rubric"] = question
|
|
82
|
+
out["privileged"] = priv
|
|
83
|
+
out["audit"] = {
|
|
84
|
+
"question": question,
|
|
85
|
+
"verifier": row.get("judge_name"),
|
|
86
|
+
"verifier_reason": row.get("reason"),
|
|
87
|
+
"verifier_reward": row.get("reward"),
|
|
88
|
+
}
|
|
89
|
+
for key in ("reward", "reason", "judge_status", "judge_meta", "judge_name", "failure_class"):
|
|
90
|
+
out.pop(key, None)
|
|
91
|
+
return out
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _rate(hits: int, n: int) -> tuple[float | None, tuple[float, float] | None]:
|
|
95
|
+
if not n:
|
|
96
|
+
return None, None
|
|
97
|
+
return round(hits / n, 4), wilson_interval(hits, n)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def audit_grades(
|
|
101
|
+
rows: Sequence[dict],
|
|
102
|
+
*,
|
|
103
|
+
judge: Callable[[dict], Any],
|
|
104
|
+
sample: int = 40,
|
|
105
|
+
passes: int = 0,
|
|
106
|
+
question: str = AUDIT_QUESTION,
|
|
107
|
+
seed: int = 0,
|
|
108
|
+
concurrency: int = 4,
|
|
109
|
+
timeout: float = 120,
|
|
110
|
+
judge_name: str | None = None,
|
|
111
|
+
) -> dict[str, Any]:
|
|
112
|
+
"""Estimate the verifier's false-negative rate from a judged sample.
|
|
113
|
+
|
|
114
|
+
``rows`` are graded by the verifier (``run_judge(rows, verifier)`` or
|
|
115
|
+
``data.grade(judge=verifier)``): ``reward`` 0/1, ``reason`` from the
|
|
116
|
+
rule. ``sample`` failed rows (reward 0, judge ok) are drawn with
|
|
117
|
+
``seed`` and each is put to ``judge`` (any judge in the ``run_judge``
|
|
118
|
+
contract: ``rubric_judge()``, ``grade_llm``, your own callable) with
|
|
119
|
+
the reference in place and ``question`` as the rubric when the row
|
|
120
|
+
carries none. A judge reward at or above 0.5 on a failed row is a
|
|
121
|
+
false negative. ``passes`` samples passed rows the same way for the
|
|
122
|
+
false-positive side.
|
|
123
|
+
|
|
124
|
+
Returns ``fn_rate`` with ``fn_ci95`` (Wilson), ``estimated_wrong_fails``
|
|
125
|
+
(the rate over every failed row), ``reasons`` (the verifier's failure
|
|
126
|
+
kinds in the sample, each with how many the judge overturned), a few
|
|
127
|
+
``examples``, ``fp_rate`` when ``passes`` > 0, and ``warnings``. Above
|
|
128
|
+
``FN_WARN`` the summary says to fix the verifier before training;
|
|
129
|
+
``select_for_rl(audit=report)`` and ``optimize(audit=)`` carry the
|
|
130
|
+
same warning into the selection.
|
|
131
|
+
"""
|
|
132
|
+
if sample < 1:
|
|
133
|
+
raise ValueError("sample is how many failed rows to put to the judge; at least 1")
|
|
134
|
+
graded = [r for r in rows if isinstance(r, dict) and _binary(r) is not None]
|
|
135
|
+
failed = [r for r in graded if _binary(r) == 0 and r.get("judge_status", "ok") == "ok"]
|
|
136
|
+
passed = [r for r in graded if _binary(r) == 1 and r.get("judge_status", "ok") == "ok"]
|
|
137
|
+
verifier = next((str(r.get("judge_name")) for r in graded if r.get("judge_name")), None)
|
|
138
|
+
rng = random.Random(seed)
|
|
139
|
+
take_f = failed if len(failed) <= sample else rng.sample(failed, sample)
|
|
140
|
+
take_p = (
|
|
141
|
+
[] if passes <= 0 else (passed if len(passed) <= passes else rng.sample(passed, passes))
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
def _judge(batch: list[dict]) -> list[dict]:
|
|
145
|
+
if not batch:
|
|
146
|
+
return []
|
|
147
|
+
audit_rows = [_audit_row(r, question) for r in batch]
|
|
148
|
+
scored = run_judge(
|
|
149
|
+
audit_rows,
|
|
150
|
+
judge,
|
|
151
|
+
judge_name=judge_name,
|
|
152
|
+
concurrency=concurrency,
|
|
153
|
+
timeout=timeout,
|
|
154
|
+
source="audit",
|
|
155
|
+
)
|
|
156
|
+
return list(scored.rows)
|
|
157
|
+
|
|
158
|
+
judged_f = _judge(list(take_f))
|
|
159
|
+
judged_p = _judge(list(take_p))
|
|
160
|
+
judge_label = next((str(r.get("judge_name")) for r in judged_f + judged_p), judge_name)
|
|
161
|
+
|
|
162
|
+
checked_f = [r for r in judged_f if r.get("judge_status") == "ok" and _num(r) is not None]
|
|
163
|
+
overturned = [r for r in checked_f if float(_num(r) or 0.0) >= 0.5]
|
|
164
|
+
checked_p = [r for r in judged_p if r.get("judge_status") == "ok" and _num(r) is not None]
|
|
165
|
+
fp_rows = [r for r in checked_p if float(_num(r) or 0.0) < 0.5]
|
|
166
|
+
errors = sum(1 for r in judged_f + judged_p if r.get("judge_status") != "ok")
|
|
167
|
+
|
|
168
|
+
fn_rate, fn_ci = _rate(len(overturned), len(checked_f))
|
|
169
|
+
fp_rate, fp_ci = _rate(len(fp_rows), len(checked_p))
|
|
170
|
+
|
|
171
|
+
reasons: dict[str, dict[str, int]] = {}
|
|
172
|
+
for r in checked_f:
|
|
173
|
+
key = _reason_key((r.get("audit") or {}).get("verifier_reason"), verifier)
|
|
174
|
+
slot = reasons.setdefault(key, {"n": 0, "fn": 0})
|
|
175
|
+
slot["n"] += 1
|
|
176
|
+
if float(_num(r) or 0.0) >= 0.5:
|
|
177
|
+
slot["fn"] += 1
|
|
178
|
+
reasons = dict(sorted(reasons.items(), key=lambda kv: (-kv[1]["fn"], -kv[1]["n"], kv[0])))
|
|
179
|
+
|
|
180
|
+
examples = [
|
|
181
|
+
{
|
|
182
|
+
"scenario_id": r.get("scenario_id"),
|
|
183
|
+
"rollout_index": r.get("rollout_index"),
|
|
184
|
+
"verifier_reason": (r.get("audit") or {}).get("verifier_reason"),
|
|
185
|
+
"judge_reason": r.get("reason"),
|
|
186
|
+
}
|
|
187
|
+
for r in overturned[:5]
|
|
188
|
+
]
|
|
189
|
+
estimated = round(float(fn_rate) * len(failed)) if fn_rate is not None else None
|
|
190
|
+
|
|
191
|
+
warnings: list[str] = []
|
|
192
|
+
if fn_rate is None:
|
|
193
|
+
summary = (
|
|
194
|
+
f"no failed row could be audited ({len(failed)} failed, {errors} judge errors); "
|
|
195
|
+
"nothing to say about the verifier"
|
|
196
|
+
)
|
|
197
|
+
warnings.append(summary)
|
|
198
|
+
else:
|
|
199
|
+
lo, hi = fn_ci or (0.0, 0.0)
|
|
200
|
+
summary = (
|
|
201
|
+
f"the judge overturned {len(overturned)} of {len(checked_f)} audited failures: "
|
|
202
|
+
f"false-negative rate {fn_rate:.0%} (95% {lo:.0%}..{hi:.0%}), about {estimated} of "
|
|
203
|
+
f"{len(failed)} failed rows are right answers the verifier rejected"
|
|
204
|
+
)
|
|
205
|
+
if fn_rate > FN_WARN:
|
|
206
|
+
top = next(iter(reasons), None)
|
|
207
|
+
warnings.append(
|
|
208
|
+
f"VERIFIER: {summary}. Fix the verifier before training: each false negative "
|
|
209
|
+
"halves the separation on its task, and a grader fix has moved mean reward "
|
|
210
|
+
"0.20 -> 0.48 with no training"
|
|
211
|
+
+ (f". The reason the judge overturns most: {top!r}" if top else "")
|
|
212
|
+
+ "."
|
|
213
|
+
)
|
|
214
|
+
if len(checked_f) < 20:
|
|
215
|
+
warnings.append(
|
|
216
|
+
f"{len(checked_f)} audited failures is a small sample; the interval is wide. "
|
|
217
|
+
"Raise sample= for a rate worth acting on."
|
|
218
|
+
)
|
|
219
|
+
if fp_rate is not None and fp_rate > FN_WARN:
|
|
220
|
+
warnings.append(
|
|
221
|
+
f"the judge also disagreed with {len(fp_rows)} of {len(checked_p)} audited passes "
|
|
222
|
+
f"(false-positive rate {fp_rate:.0%}); the rule may be too loose as well"
|
|
223
|
+
)
|
|
224
|
+
if errors:
|
|
225
|
+
warnings.append(f"{errors} audit rows were not judged (judge error or timeout)")
|
|
226
|
+
if fn_rate is not None:
|
|
227
|
+
warnings.append(
|
|
228
|
+
"The judge is a second opinion, not ground truth; check it on human labels "
|
|
229
|
+
"(judge_agreement, judge_trust) before acting on a rate near the threshold."
|
|
230
|
+
)
|
|
231
|
+
return {
|
|
232
|
+
"n_rows": len(graded),
|
|
233
|
+
"n_failed": len(failed),
|
|
234
|
+
"n_passed": len(passed),
|
|
235
|
+
"n_sampled": len(take_f),
|
|
236
|
+
"n_checked": len(checked_f),
|
|
237
|
+
"false_negatives": len(overturned),
|
|
238
|
+
"fn_rate": fn_rate,
|
|
239
|
+
"fn_ci95": fn_ci,
|
|
240
|
+
"estimated_wrong_fails": estimated,
|
|
241
|
+
"n_passes_sampled": len(take_p),
|
|
242
|
+
"n_passes_checked": len(checked_p),
|
|
243
|
+
"false_positives": len(fp_rows),
|
|
244
|
+
"fp_rate": fp_rate,
|
|
245
|
+
"fp_ci95": fp_ci,
|
|
246
|
+
"judge_errors": errors,
|
|
247
|
+
"reasons": reasons,
|
|
248
|
+
"examples": examples,
|
|
249
|
+
"verifier": verifier,
|
|
250
|
+
"judge": judge_label,
|
|
251
|
+
"question": question,
|
|
252
|
+
"warnings": warnings,
|
|
253
|
+
"summary": summary,
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _num(row: dict) -> float | None:
|
|
258
|
+
value = row.get("reward")
|
|
259
|
+
if value is None or isinstance(value, bool):
|
|
260
|
+
return None
|
|
261
|
+
try:
|
|
262
|
+
return float(value)
|
|
263
|
+
except (TypeError, ValueError):
|
|
264
|
+
return None
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def audit_warning(audit: dict[str, Any] | None) -> str | None:
|
|
268
|
+
"""The one line a selection report carries when an audit found the
|
|
269
|
+
verifier wrong too often; ``None`` otherwise."""
|
|
270
|
+
if not isinstance(audit, dict):
|
|
271
|
+
return None
|
|
272
|
+
rate = audit.get("fn_rate")
|
|
273
|
+
if rate is None or float(rate) <= FN_WARN:
|
|
274
|
+
return None
|
|
275
|
+
ci = audit.get("fn_ci95") or (0.0, 0.0)
|
|
276
|
+
return (
|
|
277
|
+
f"verifier false-negative rate {float(rate):.0%} (95% {ci[0]:.0%}..{ci[1]:.0%}, "
|
|
278
|
+
f"{audit.get('n_checked')} audited): right answers are being scored 0. Fix the "
|
|
279
|
+
"verifier before training on this selection (audit_grades)."
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def format_audit(report: dict[str, Any]) -> str:
|
|
284
|
+
"""The block a person reads: the summary, then the reasons."""
|
|
285
|
+
lines = [str(report.get("summary") or "")]
|
|
286
|
+
for key, slot in (report.get("reasons") or {}).items():
|
|
287
|
+
lines.append(f" {key}: {slot['fn']} of {slot['n']} overturned")
|
|
288
|
+
for ex in report.get("examples") or []:
|
|
289
|
+
lines.append(
|
|
290
|
+
f" {ex.get('scenario_id')} r{ex.get('rollout_index')}: verifier said "
|
|
291
|
+
f"{ex.get('verifier_reason')!r}; judge: {ex.get('judge_reason')!r}"
|
|
292
|
+
)
|
|
293
|
+
return "\n".join(lines)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
__all__ = ["AUDIT_QUESTION", "FN_WARN", "audit_grades", "audit_warning", "format_audit"]
|
|
@@ -314,11 +314,81 @@ def hygiene_warnings(
|
|
|
314
314
|
return out
|
|
315
315
|
|
|
316
316
|
|
|
317
|
+
def _tool_name(tool: Any) -> str:
|
|
318
|
+
if isinstance(tool, str):
|
|
319
|
+
return tool
|
|
320
|
+
if not isinstance(tool, dict):
|
|
321
|
+
return ""
|
|
322
|
+
fn = tool.get("function")
|
|
323
|
+
if isinstance(fn, dict) and fn.get("name"):
|
|
324
|
+
return str(fn["name"])
|
|
325
|
+
return str(tool.get("name") or "")
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def coverage_warnings(
|
|
329
|
+
rows: Sequence[dict], *, tools: Sequence[dict] | Sequence[str] | None = None
|
|
330
|
+
) -> list[str]:
|
|
331
|
+
"""Plain-words notes on whether a graded run can mean anything.
|
|
332
|
+
|
|
333
|
+
A pass@1 of 1.00 over rows where the agent never called a tool, or a
|
|
334
|
+
marker that fired on no row, is the most expensive eval failure there
|
|
335
|
+
is, because it reads as a result. ``run_judge`` (so ``evaluate`` and
|
|
336
|
+
``data.grade``) attaches these to ``ScoredData.warnings`` and logs
|
|
337
|
+
them once. ``tools=`` is the declared tool list (OpenAI or bare shape,
|
|
338
|
+
or just names); ``evaluate(data, judge)`` and ``data.grade`` read it
|
|
339
|
+
off the run. The tool checks need it, so an agent with no tools is
|
|
340
|
+
never called hollow for calling none. Each note names the fix.
|
|
341
|
+
"""
|
|
342
|
+
row_list = [r for r in rows if isinstance(r, dict)]
|
|
343
|
+
out: list[str] = []
|
|
344
|
+
n = len(row_list)
|
|
345
|
+
if n == 0:
|
|
346
|
+
return out
|
|
347
|
+
fix = (
|
|
348
|
+
"Put the ids your world has (order numbers, account names) in the tool "
|
|
349
|
+
"descriptions or in seeds=, and check the agent wrapper records its steps."
|
|
350
|
+
)
|
|
351
|
+
names = [nm for nm in (_tool_name(t) for t in tools or []) if nm]
|
|
352
|
+
with_calls = sum(1 for r in row_list if tool_calls(r) > 0)
|
|
353
|
+
if names and with_calls == 0:
|
|
354
|
+
out.append(f"0 of {n} rows called a tool, so this score says nothing about tool use. {fix}")
|
|
355
|
+
elif names:
|
|
356
|
+
called: set[str] = set()
|
|
357
|
+
for r in row_list:
|
|
358
|
+
for s in r.get("steps") or r.get("tool_trace") or []:
|
|
359
|
+
if isinstance(s, dict) and s.get("tool"):
|
|
360
|
+
called.add(str(s["tool"]))
|
|
361
|
+
never = [nm for nm in names if nm not in called]
|
|
362
|
+
if never:
|
|
363
|
+
out.append(
|
|
364
|
+
f"{len(never)} of {len(names)} declared tools were never called "
|
|
365
|
+
f"({', '.join(never)}); the policy branches behind them are untested. "
|
|
366
|
+
"Add a seed ask for each."
|
|
367
|
+
)
|
|
368
|
+
populated: dict[str, int] = {}
|
|
369
|
+
for r in row_list:
|
|
370
|
+
marks = r.get("markers")
|
|
371
|
+
if not isinstance(marks, dict):
|
|
372
|
+
continue
|
|
373
|
+
for name, value in marks.items():
|
|
374
|
+
populated.setdefault(str(name), 0)
|
|
375
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
376
|
+
populated[str(name)] += 1
|
|
377
|
+
for name in sorted(populated):
|
|
378
|
+
if populated[name] == 0:
|
|
379
|
+
out.append(
|
|
380
|
+
f"marker {name!r} fired on 0 of {n} rows; its rate is not a measurement. "
|
|
381
|
+
"Add a seed ask that exercises it, or drop it from the report."
|
|
382
|
+
)
|
|
383
|
+
return out
|
|
384
|
+
|
|
385
|
+
|
|
317
386
|
__all__ = [
|
|
318
387
|
"DEFAULT_MAX_SPREAD",
|
|
319
388
|
"HACK_THRESHOLD",
|
|
320
389
|
"NEAR_DUP_JACCARD",
|
|
321
390
|
"assistant_turns",
|
|
391
|
+
"coverage_warnings",
|
|
322
392
|
"dedupe_groups",
|
|
323
393
|
"drop_truncated",
|
|
324
394
|
"hygiene_warnings",
|