pytest-inspect-evals 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytest_inspect_evals-0.1.0/LICENSE +21 -0
- pytest_inspect_evals-0.1.0/PKG-INFO +148 -0
- pytest_inspect_evals-0.1.0/README.md +124 -0
- pytest_inspect_evals-0.1.0/pyproject.toml +121 -0
- pytest_inspect_evals-0.1.0/pyproject.toml.orig +127 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/__init__.py +10 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/_hf.py +69 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/_windows.py +25 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/assertions.py +145 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/gates.py +97 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/huggingface.py +459 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/metrics.py +257 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/plugin.py +112 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/py.typed +0 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/sandbox.py +170 -0
- pytest_inspect_evals-0.1.0/src/pytest_inspect_evals/solvers.py +84 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Matt Fisher
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pytest-inspect-evals
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations
|
|
5
|
+
Author: Matt Fisher
|
|
6
|
+
Author-email: Matt Fisher <m@ttfisher.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Classifier: Framework :: Pytest
|
|
10
|
+
Requires-Dist: inspect-ai>=0.3.262
|
|
11
|
+
Requires-Dist: numpy>=1.26.0
|
|
12
|
+
Requires-Dist: pydantic>=2.10.0
|
|
13
|
+
Requires-Dist: pytest>=8.0
|
|
14
|
+
Requires-Dist: pytest-rerunfailures>=15
|
|
15
|
+
Requires-Dist: requests>=2.32.0
|
|
16
|
+
Requires-Dist: urllib3>=2.0
|
|
17
|
+
Requires-Dist: datasets>=4.8.5 ; extra == 'huggingface'
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Project-URL: Homepage, https://github.com/Generality-Labs/pytest-inspect-evals
|
|
20
|
+
Project-URL: Repository, https://github.com/Generality-Labs/pytest-inspect-evals
|
|
21
|
+
Project-URL: Issues, https://github.com/Generality-Labs/pytest-inspect-evals/issues
|
|
22
|
+
Provides-Extra: huggingface
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# pytest-inspect-evals
|
|
26
|
+
|
|
27
|
+
A pytest plugin with the test gates, fixtures and helpers shared by [Inspect AI](https://inspect.aisi.org.uk/) evaluation repos. It started as the shared test code in [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). Installing it replaces a copied `conftest.py` and `tests/utils/` with one pinned dependency.
|
|
28
|
+
|
|
29
|
+
## Install
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
uv add --dev pytest-inspect-evals
|
|
33
|
+
# or, to use the Hugging Face dataset helpers:
|
|
34
|
+
uv add --dev "pytest-inspect-evals[huggingface]"
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
The plugin loads automatically. There is nothing to add to `conftest.py`.
|
|
38
|
+
|
|
39
|
+
## Test gates
|
|
40
|
+
|
|
41
|
+
Four markers mark tests that are skipped unless you switch them on.
|
|
42
|
+
|
|
43
|
+
| Marker | CLI flag | Env var | Pytest config option | Default |
|
|
44
|
+
| ------------------ | -------------------- | ---------------------------- | -------------------------------- | ------- |
|
|
45
|
+
| `slow` | `--runslow` | `RUN_SLOW_TESTS` | `inspect_evals_slow` | off |
|
|
46
|
+
| `dataset_download` | `--dataset-download` | `RUN_DATASET_DOWNLOAD_TESTS` | `inspect_evals_dataset_download` | off |
|
|
47
|
+
| `k8s` | `--runk8s` | `RUN_K8S_TESTS` | `inspect_evals_k8s` | off |
|
|
48
|
+
| `gpu` | `--rungpu` | `RUN_GPU_TESTS` | `inspect_evals_gpu` | off |
|
|
49
|
+
|
|
50
|
+
The env var wins when it is set to a non-empty value: `1`, `true`, `yes` or `on` switch the gate on, anything else switches it off. Otherwise the CLI flag switches it on. Otherwise the config option decides, and the default is off. An empty env var counts as unset.
|
|
51
|
+
|
|
52
|
+
To run a gate by default in your repo, set its option in `pyproject.toml`:
|
|
53
|
+
|
|
54
|
+
```toml
|
|
55
|
+
[tool.pytest.ini_options]
|
|
56
|
+
inspect_evals_dataset_download = true
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Three more markers have fixed behaviour:
|
|
60
|
+
|
|
61
|
+
- `huggingface`: skipped when `HF_TOKEN` is unset or blank. Otherwise the test is retried up to twice, 60 seconds apart, and a gated-dataset error (`GatedRepoError`, or `DatasetNotFoundError` mentioning a gated dataset) is reported as a skip.
|
|
62
|
+
- `docker` and `posix_only`: skipped on Windows.
|
|
63
|
+
|
|
64
|
+
The plugin registers all seven markers, so they work under `--strict-markers` without any config.
|
|
65
|
+
|
|
66
|
+
## Adding your own gate
|
|
67
|
+
|
|
68
|
+
`skip_if_marker_present` is the function the plugin uses. Call it from your `conftest.py` for repo-specific gates:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
from pytest_inspect_evals.gates import skip_if_marker_present
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def pytest_addoption(parser):
|
|
75
|
+
parser.addoption("--custom-smoke", action="store_true", default=False)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def pytest_configure(config):
|
|
79
|
+
config.addinivalue_line("markers", "custom_smoke: tests that call a live service")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def pytest_collection_modifyitems(config, items):
|
|
83
|
+
skip_if_marker_present(
|
|
84
|
+
config,
|
|
85
|
+
items,
|
|
86
|
+
marker="custom_smoke",
|
|
87
|
+
cli_flag="--custom-smoke",
|
|
88
|
+
env_var="RUN_CUSTOM_SMOKE_TESTS",
|
|
89
|
+
reason="custom_smoke tests disabled (set RUN_CUSTOM_SMOKE_TESTS=1 or pass --custom-smoke)",
|
|
90
|
+
)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Fixtures
|
|
94
|
+
|
|
95
|
+
`set_model_roles` returns a context manager that sets model roles, for testing scorers that call a grader model:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
def test_judge_scorer(set_model_roles):
|
|
99
|
+
with set_model_roles(grader=get_model("mockllm/model")):
|
|
100
|
+
...
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`mock_docker_sandbox` stubs the Docker sandbox lifecycle, so an eval with `sandbox="docker"` runs without Docker. Combine it with mocked `sandbox().exec` results to test a scorer end to end:
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
def test_e2e(mock_docker_sandbox):
|
|
107
|
+
[log] = eval(my_task(), model="mockllm/model")
|
|
108
|
+
assert log.status == "success"
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## Helpers
|
|
112
|
+
|
|
113
|
+
- `pytest_inspect_evals.assertions`: `assert_task_structure`, `assert_eval_success`, `run_single_sample_eval`, `get_metric_value`.
|
|
114
|
+
- `pytest_inspect_evals.solvers`: `mock_solver_with_output`, `run_command`.
|
|
115
|
+
- `pytest_inspect_evals.sandbox`: `MockExecResult`, `create_sandbox_tool_task`, `assert_sandbox_test_passed`.
|
|
116
|
+
- `pytest_inspect_evals.metrics`: `run_metrics`, `assert_agreeing_epochs_change_nothing`, `MetricEvalError`. These run custom metrics through the real epoch reducer, the way an eval does.
|
|
117
|
+
- `pytest_inspect_evals.huggingface` (needs the `huggingface` extra): `assert_huggingface_dataset_structure`, `get_dataset_infos_dict`, `assert_huggingface_dataset_is_valid`, `assert_dataset_contains_subsets`, `assert_dataset_has_columns`. These check a dataset's schema through the Hugging Face dataset viewer API without downloading it.
|
|
118
|
+
|
|
119
|
+
## Migrating from a copied conftest
|
|
120
|
+
|
|
121
|
+
If your `conftest.py` came from inspect_evals or inspect-evals-template, remove what the plugin now provides:
|
|
122
|
+
|
|
123
|
+
- the `pytest_addoption` lines for `--runslow`, `--dataset-download`, `--runk8s` and `--rungpu`;
|
|
124
|
+
- the matching skip logic in `pytest_collection_modifyitems`;
|
|
125
|
+
- the seven markers above from your `markers =` config (keeping them is harmless).
|
|
126
|
+
|
|
127
|
+
If you keep the options, pytest stops at startup with `argparse.ArgumentError: argument --runslow: conflicting option string: --runslow`.
|
|
128
|
+
|
|
129
|
+
## Private inspect_ai APIs
|
|
130
|
+
|
|
131
|
+
Two fixtures use inspect_ai internals that have no public equivalent: `set_model_roles` calls `inspect_ai.model._model.init_model_roles`, and `mock_docker_sandbox` patches `inspect_ai.util._sandbox.docker.docker.DockerSandboxEnvironment`. A future inspect_ai release could break them.
|
|
132
|
+
|
|
133
|
+
## Development
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
uv sync
|
|
137
|
+
uv run pre-commit install # optional: run the lint stack on every commit
|
|
138
|
+
uv run pytest
|
|
139
|
+
uv run basedpyright src
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Each pull request adds a changelog fragment rather than editing `CHANGELOG.md`, so concurrent PRs don't conflict. Run `uv run scriv create`, uncomment the sections that apply in the new file under `changelog.d/`, and commit it with the change, or delete it if the change needs no entry. To release, run the **Prepare release** workflow from the Actions tab. It bumps `version` in `pyproject.toml`, collects the fragments into `CHANGELOG.md`, and opens a release pull request, whose CI starts once you click **Approve workflows to run** on it; merging it tags the release. It needs _Settings → Actions → General_ → **Allow GitHub Actions to create and approve pull requests**. By hand, the same is `uv version --bump minor` and `uv run scriv collect`.
|
|
143
|
+
|
|
144
|
+
Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
|
|
145
|
+
|
|
146
|
+
## Releasing
|
|
147
|
+
|
|
148
|
+
See [RELEASING.md](RELEASING.md).
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
# pytest-inspect-evals
|
|
2
|
+
|
|
3
|
+
A pytest plugin with the test gates, fixtures and helpers shared by [Inspect AI](https://inspect.aisi.org.uk/) evaluation repos. It started as the shared test code in [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). Installing it replaces a copied `conftest.py` and `tests/utils/` with one pinned dependency.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
uv add --dev pytest-inspect-evals
|
|
9
|
+
# or, to use the Hugging Face dataset helpers:
|
|
10
|
+
uv add --dev "pytest-inspect-evals[huggingface]"
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
The plugin loads automatically. There is nothing to add to `conftest.py`.
|
|
14
|
+
|
|
15
|
+
## Test gates
|
|
16
|
+
|
|
17
|
+
Four markers mark tests that are skipped unless you switch them on.
|
|
18
|
+
|
|
19
|
+
| Marker | CLI flag | Env var | Pytest config option | Default |
|
|
20
|
+
| ------------------ | -------------------- | ---------------------------- | -------------------------------- | ------- |
|
|
21
|
+
| `slow` | `--runslow` | `RUN_SLOW_TESTS` | `inspect_evals_slow` | off |
|
|
22
|
+
| `dataset_download` | `--dataset-download` | `RUN_DATASET_DOWNLOAD_TESTS` | `inspect_evals_dataset_download` | off |
|
|
23
|
+
| `k8s` | `--runk8s` | `RUN_K8S_TESTS` | `inspect_evals_k8s` | off |
|
|
24
|
+
| `gpu` | `--rungpu` | `RUN_GPU_TESTS` | `inspect_evals_gpu` | off |
|
|
25
|
+
|
|
26
|
+
The env var wins when it is set to a non-empty value: `1`, `true`, `yes` or `on` switch the gate on, anything else switches it off. Otherwise the CLI flag switches it on. Otherwise the config option decides, and the default is off. An empty env var counts as unset.
|
|
27
|
+
|
|
28
|
+
To run a gate by default in your repo, set its option in `pyproject.toml`:
|
|
29
|
+
|
|
30
|
+
```toml
|
|
31
|
+
[tool.pytest.ini_options]
|
|
32
|
+
inspect_evals_dataset_download = true
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
Three more markers have fixed behaviour:
|
|
36
|
+
|
|
37
|
+
- `huggingface`: skipped when `HF_TOKEN` is unset or blank. Otherwise the test is retried up to twice, 60 seconds apart, and a gated-dataset error (`GatedRepoError`, or `DatasetNotFoundError` mentioning a gated dataset) is reported as a skip.
|
|
38
|
+
- `docker` and `posix_only`: skipped on Windows.
|
|
39
|
+
|
|
40
|
+
The plugin registers all seven markers, so they work under `--strict-markers` without any config.
|
|
41
|
+
|
|
42
|
+
## Adding your own gate
|
|
43
|
+
|
|
44
|
+
`skip_if_marker_present` is the function the plugin uses. Call it from your `conftest.py` for repo-specific gates:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
from pytest_inspect_evals.gates import skip_if_marker_present
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def pytest_addoption(parser):
|
|
51
|
+
parser.addoption("--custom-smoke", action="store_true", default=False)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def pytest_configure(config):
|
|
55
|
+
config.addinivalue_line("markers", "custom_smoke: tests that call a live service")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def pytest_collection_modifyitems(config, items):
|
|
59
|
+
skip_if_marker_present(
|
|
60
|
+
config,
|
|
61
|
+
items,
|
|
62
|
+
marker="custom_smoke",
|
|
63
|
+
cli_flag="--custom-smoke",
|
|
64
|
+
env_var="RUN_CUSTOM_SMOKE_TESTS",
|
|
65
|
+
reason="custom_smoke tests disabled (set RUN_CUSTOM_SMOKE_TESTS=1 or pass --custom-smoke)",
|
|
66
|
+
)
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
## Fixtures
|
|
70
|
+
|
|
71
|
+
`set_model_roles` returns a context manager that sets model roles, for testing scorers that call a grader model:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
def test_judge_scorer(set_model_roles):
|
|
75
|
+
with set_model_roles(grader=get_model("mockllm/model")):
|
|
76
|
+
...
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
`mock_docker_sandbox` stubs the Docker sandbox lifecycle, so an eval with `sandbox="docker"` runs without Docker. Combine it with mocked `sandbox().exec` results to test a scorer end to end:
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
def test_e2e(mock_docker_sandbox):
|
|
83
|
+
[log] = eval(my_task(), model="mockllm/model")
|
|
84
|
+
assert log.status == "success"
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Helpers
|
|
88
|
+
|
|
89
|
+
- `pytest_inspect_evals.assertions`: `assert_task_structure`, `assert_eval_success`, `run_single_sample_eval`, `get_metric_value`.
|
|
90
|
+
- `pytest_inspect_evals.solvers`: `mock_solver_with_output`, `run_command`.
|
|
91
|
+
- `pytest_inspect_evals.sandbox`: `MockExecResult`, `create_sandbox_tool_task`, `assert_sandbox_test_passed`.
|
|
92
|
+
- `pytest_inspect_evals.metrics`: `run_metrics`, `assert_agreeing_epochs_change_nothing`, `MetricEvalError`. These run custom metrics through the real epoch reducer, the way an eval does.
|
|
93
|
+
- `pytest_inspect_evals.huggingface` (needs the `huggingface` extra): `assert_huggingface_dataset_structure`, `get_dataset_infos_dict`, `assert_huggingface_dataset_is_valid`, `assert_dataset_contains_subsets`, `assert_dataset_has_columns`. These check a dataset's schema through the Hugging Face dataset viewer API without downloading it.
|
|
94
|
+
|
|
95
|
+
## Migrating from a copied conftest
|
|
96
|
+
|
|
97
|
+
If your `conftest.py` came from inspect_evals or inspect-evals-template, remove what the plugin now provides:
|
|
98
|
+
|
|
99
|
+
- the `pytest_addoption` lines for `--runslow`, `--dataset-download`, `--runk8s` and `--rungpu`;
|
|
100
|
+
- the matching skip logic in `pytest_collection_modifyitems`;
|
|
101
|
+
- the seven markers above from your `markers =` config (keeping them is harmless).
|
|
102
|
+
|
|
103
|
+
If you keep the options, pytest stops at startup with `argparse.ArgumentError: argument --runslow: conflicting option string: --runslow`.
|
|
104
|
+
|
|
105
|
+
## Private inspect_ai APIs
|
|
106
|
+
|
|
107
|
+
Two fixtures use inspect_ai internals that have no public equivalent: `set_model_roles` calls `inspect_ai.model._model.init_model_roles`, and `mock_docker_sandbox` patches `inspect_ai.util._sandbox.docker.docker.DockerSandboxEnvironment`. A future inspect_ai release could break them.
|
|
108
|
+
|
|
109
|
+
## Development
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
uv sync
|
|
113
|
+
uv run pre-commit install # optional: run the lint stack on every commit
|
|
114
|
+
uv run pytest
|
|
115
|
+
uv run basedpyright src
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Each pull request adds a changelog fragment rather than editing `CHANGELOG.md`, so concurrent PRs don't conflict. Run `uv run scriv create`, uncomment the sections that apply in the new file under `changelog.d/`, and commit it with the change, or delete it if the change needs no entry. To release, run the **Prepare release** workflow from the Actions tab. It bumps `version` in `pyproject.toml`, collects the fragments into `CHANGELOG.md`, and opens a release pull request, whose CI starts once you click **Approve workflows to run** on it; merging it tags the release. It needs _Settings → Actions → General_ → **Allow GitHub Actions to create and approve pull requests**. By hand, the same is `uv version --bump minor` and `uv run scriv collect`.
|
|
119
|
+
|
|
120
|
+
Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
|
|
121
|
+
|
|
122
|
+
## Releasing
|
|
123
|
+
|
|
124
|
+
See [RELEASING.md](RELEASING.md).
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv_build>=0.12.22,<0.13.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pytest-inspect-evals"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.11"
|
|
13
|
+
classifiers = ["Framework :: Pytest"]
|
|
14
|
+
dependencies = [
|
|
15
|
+
"inspect_ai>=0.3.262",
|
|
16
|
+
"numpy>=1.26.0",
|
|
17
|
+
"pydantic>=2.10.0",
|
|
18
|
+
"pytest>=8.0",
|
|
19
|
+
"pytest-rerunfailures>=15",
|
|
20
|
+
"requests>=2.32.0",
|
|
21
|
+
"urllib3>=2.0",
|
|
22
|
+
]
|
|
23
|
+
|
|
24
|
+
[[project.authors]]
|
|
25
|
+
name = "Matt Fisher"
|
|
26
|
+
email = "m@ttfisher.com"
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
huggingface = ["datasets>=4.8.5"]
|
|
30
|
+
|
|
31
|
+
[project.entry-points.pytest11]
|
|
32
|
+
pytest_inspect_evals = "pytest_inspect_evals.plugin"
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/Generality-Labs/pytest-inspect-evals"
|
|
36
|
+
Repository = "https://github.com/Generality-Labs/pytest-inspect-evals"
|
|
37
|
+
Issues = "https://github.com/Generality-Labs/pytest-inspect-evals/issues"
|
|
38
|
+
|
|
39
|
+
[dependency-groups]
|
|
40
|
+
dev = [
|
|
41
|
+
"pytest>=8.0.0",
|
|
42
|
+
"pytest-cov>=7.0.0",
|
|
43
|
+
"basedpyright>=1.39",
|
|
44
|
+
"pre-commit>=4.0.0",
|
|
45
|
+
"scriv>=1.8",
|
|
46
|
+
"datasets>=4.8.5",
|
|
47
|
+
"pytest-asyncio>=1.0",
|
|
48
|
+
"mdformat>=1.0.0",
|
|
49
|
+
"mdformat-gfm>=1.0.0",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
[tool.scriv]
|
|
53
|
+
format = "md"
|
|
54
|
+
fragment_directory = "changelog.d"
|
|
55
|
+
categories = [
|
|
56
|
+
"Added",
|
|
57
|
+
"Changed",
|
|
58
|
+
"Deprecated",
|
|
59
|
+
"Removed",
|
|
60
|
+
"Fixed",
|
|
61
|
+
"Security",
|
|
62
|
+
]
|
|
63
|
+
skip_fragments = "[A-Z]*"
|
|
64
|
+
new_fragment_template = "file: TEMPLATE.md"
|
|
65
|
+
entry_title_template = "[{{ version }}] - {{ date.strftime('%Y-%m-%d') }}"
|
|
66
|
+
md_header_level = "2"
|
|
67
|
+
md_html_anchors = false
|
|
68
|
+
compact_fragments = true
|
|
69
|
+
version = "literal: pyproject.toml: project.version"
|
|
70
|
+
|
|
71
|
+
[tool.ruff]
|
|
72
|
+
line-length = 100
|
|
73
|
+
src = ["src"]
|
|
74
|
+
target-version = "py311"
|
|
75
|
+
|
|
76
|
+
[tool.ruff.lint]
|
|
77
|
+
select = [
|
|
78
|
+
"E",
|
|
79
|
+
"W",
|
|
80
|
+
"F",
|
|
81
|
+
"I",
|
|
82
|
+
"UP",
|
|
83
|
+
"B",
|
|
84
|
+
"SIM",
|
|
85
|
+
"D",
|
|
86
|
+
"C4",
|
|
87
|
+
"PT",
|
|
88
|
+
"PIE",
|
|
89
|
+
"DTZ",
|
|
90
|
+
"ISC",
|
|
91
|
+
"ASYNC",
|
|
92
|
+
"N",
|
|
93
|
+
"FURB",
|
|
94
|
+
"RUF",
|
|
95
|
+
"PLE",
|
|
96
|
+
"PLW",
|
|
97
|
+
]
|
|
98
|
+
ignore = [
|
|
99
|
+
"E501",
|
|
100
|
+
"D10",
|
|
101
|
+
"D415",
|
|
102
|
+
"ISC001",
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
[tool.ruff.lint.pydocstyle]
|
|
106
|
+
convention = "google"
|
|
107
|
+
|
|
108
|
+
[tool.ruff.lint.per-file-ignores]
|
|
109
|
+
"tests/**" = ["D"]
|
|
110
|
+
|
|
111
|
+
[tool.basedpyright]
|
|
112
|
+
pythonVersion = "3.11"
|
|
113
|
+
typeCheckingMode = "standard"
|
|
114
|
+
reportMissingTypeStubs = false
|
|
115
|
+
|
|
116
|
+
[tool.pytest.ini_options]
|
|
117
|
+
testpaths = ["tests"]
|
|
118
|
+
addopts = "-rA --durations=10 --color=yes --cov=src --cov-report=term-missing"
|
|
119
|
+
asyncio_mode = "auto"
|
|
120
|
+
asyncio_default_fixture_loop_scope = "function"
|
|
121
|
+
filterwarnings = ['ignore:The configuration option "asyncio_default_fixture_loop_scope" is unset:pytest.PytestDeprecationWarning']
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
# uv's own backend. It builds src/pytest_inspect_evals/ with no further config.
|
|
3
|
+
# Keep the upper bound at the next uv minor, as `uv init` writes it, and raise
|
|
4
|
+
# both bounds when moving to a new uv minor.
|
|
5
|
+
requires = ["uv_build>=0.12.22,<0.13.0"]
|
|
6
|
+
build-backend = "uv_build"
|
|
7
|
+
|
|
8
|
+
[project]
|
|
9
|
+
name = "pytest-inspect-evals"
|
|
10
|
+
version = "0.1.0"
|
|
11
|
+
description = "pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
license = "MIT"
|
|
14
|
+
license-files = ["LICENSE"] # uv_build ships only the license files named here
|
|
15
|
+
authors = [{ name = "Matt Fisher", email = "m@ttfisher.com" }]
|
|
16
|
+
requires-python = ">=3.11"
|
|
17
|
+
classifiers = ["Framework :: Pytest"]
|
|
18
|
+
dependencies = [
|
|
19
|
+
"inspect_ai>=0.3.262",
|
|
20
|
+
"numpy>=1.26.0",
|
|
21
|
+
"pydantic>=2.10.0",
|
|
22
|
+
"pytest>=8.0",
|
|
23
|
+
"pytest-rerunfailures>=15",
|
|
24
|
+
"requests>=2.32.0",
|
|
25
|
+
"urllib3>=2.0",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
# Only pytest_inspect_evals.huggingface needs this.
|
|
30
|
+
huggingface = ["datasets>=4.8.5"]
|
|
31
|
+
|
|
32
|
+
[project.entry-points.pytest11]
|
|
33
|
+
pytest_inspect_evals = "pytest_inspect_evals.plugin"
|
|
34
|
+
|
|
35
|
+
[project.urls]
|
|
36
|
+
Homepage = "https://github.com/Generality-Labs/pytest-inspect-evals"
|
|
37
|
+
Repository = "https://github.com/Generality-Labs/pytest-inspect-evals"
|
|
38
|
+
Issues = "https://github.com/Generality-Labs/pytest-inspect-evals/issues"
|
|
39
|
+
[dependency-groups]
|
|
40
|
+
dev = [
|
|
41
|
+
"pytest>=8.0.0",
|
|
42
|
+
"pytest-cov>=7.0.0",
|
|
43
|
+
"basedpyright>=1.39",
|
|
44
|
+
"pre-commit>=4.0.0",
|
|
45
|
+
"scriv>=1.8",
|
|
46
|
+
"datasets>=4.8.5",
|
|
47
|
+
"pytest-asyncio>=1.0",
|
|
48
|
+
"mdformat>=1.0.0",
|
|
49
|
+
"mdformat-gfm>=1.0.0",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
[tool.scriv]
|
|
54
|
+
# One fragment per pull request in changelog.d/, collected into CHANGELOG.md
|
|
55
|
+
# at release time, so concurrent PRs never edit the same lines of the changelog.
|
|
56
|
+
format = "md"
|
|
57
|
+
fragment_directory = "changelog.d"
|
|
58
|
+
categories = ["Added", "Changed", "Deprecated", "Removed", "Fixed", "Security"]
|
|
59
|
+
skip_fragments = "[A-Z]*" # TEMPLATE.md
|
|
60
|
+
new_fragment_template = "file: TEMPLATE.md"
|
|
61
|
+
entry_title_template = "[{{ version }}] - {{ date.strftime('%Y-%m-%d') }}"
|
|
62
|
+
md_header_level = "2"
|
|
63
|
+
md_html_anchors = false
|
|
64
|
+
# Without this, two fragments in one category leave a blank line between them,
|
|
65
|
+
# which makes the list loose and fails mdformat on the release commit.
|
|
66
|
+
compact_fragments = true
|
|
67
|
+
# The version's single source is [project] version above, which
|
|
68
|
+
# `uv version --bump` (and the Prepare release workflow) rewrites.
|
|
69
|
+
version = "literal: pyproject.toml: project.version"
|
|
70
|
+
|
|
71
|
+
[tool.ruff]
|
|
72
|
+
line-length = 100
|
|
73
|
+
src = ["src"]
|
|
74
|
+
target-version = "py311"
|
|
75
|
+
|
|
76
|
+
[tool.ruff.lint]
|
|
77
|
+
select = [
|
|
78
|
+
"E", # pycodestyle errors
|
|
79
|
+
"W", # pycodestyle warnings
|
|
80
|
+
"F", # pyflakes
|
|
81
|
+
"I", # isort
|
|
82
|
+
"UP", # pyupgrade
|
|
83
|
+
"B", # flake8-bugbear
|
|
84
|
+
"SIM", # flake8-simplify
|
|
85
|
+
"D", # pydocstyle
|
|
86
|
+
"C4", # flake8-comprehensions
|
|
87
|
+
"PT", # flake8-pytest-style
|
|
88
|
+
"PIE", # flake8-pie
|
|
89
|
+
"DTZ", # flake8-datetimez (timezone-aware datetimes)
|
|
90
|
+
"ISC", # implicit-str-concat (catches missing commas)
|
|
91
|
+
"ASYNC", # flake8-async
|
|
92
|
+
"N", # pep8-naming
|
|
93
|
+
"FURB", # refurb (modernization)
|
|
94
|
+
"RUF", # ruff-specific rules
|
|
95
|
+
"PLE", # pylint errors
|
|
96
|
+
"PLW", # pylint warnings
|
|
97
|
+
]
|
|
98
|
+
ignore = [
|
|
99
|
+
"E501", # line-too-long — the formatter owns line length
|
|
100
|
+
"D10", # missing docstrings — don't force them
|
|
101
|
+
"D415", # first-line punctuation — the module docstring is the project
|
|
102
|
+
# description, whose punctuation varies
|
|
103
|
+
"ISC001", # single-line implicit concat — conflicts with the formatter
|
|
104
|
+
]
|
|
105
|
+
|
|
106
|
+
[tool.ruff.lint.pydocstyle]
|
|
107
|
+
convention = "google"
|
|
108
|
+
|
|
109
|
+
[tool.ruff.lint.per-file-ignores]
|
|
110
|
+
"tests/**" = ["D"] # tests don't need docstring linting
|
|
111
|
+
|
|
112
|
+
[tool.basedpyright]
|
|
113
|
+
pythonVersion = "3.11"
|
|
114
|
+
# "standard", not "strict": v0.1 moves code verbatim from inspect_evals, which
|
|
115
|
+
# was checked with mypy. Tighten once the code is ours.
|
|
116
|
+
typeCheckingMode = "standard"
|
|
117
|
+
reportMissingTypeStubs = false # third-party libs are often untyped
|
|
118
|
+
|
|
119
|
+
[tool.pytest.ini_options]
|
|
120
|
+
testpaths = ["tests"]
|
|
121
|
+
addopts = "-rA --durations=10 --color=yes --cov=src --cov-report=term-missing"
|
|
122
|
+
asyncio_mode = "auto"
|
|
123
|
+
asyncio_default_fixture_loop_scope = "function"
|
|
124
|
+
filterwarnings = [
|
|
125
|
+
# pytester's throwaway test dirs have no pytest-asyncio config of their own.
|
|
126
|
+
'ignore:The configuration option "asyncio_default_fixture_loop_scope" is unset:pytest.PytestDeprecationWarning',
|
|
127
|
+
]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
# The version is set once, in pyproject.toml; this reads it back from the
|
|
6
|
+
# installed package metadata.
|
|
7
|
+
try:
|
|
8
|
+
__version__ = version("pytest-inspect-evals")
|
|
9
|
+
except PackageNotFoundError: # a source tree that was never installed
|
|
10
|
+
__version__ = "unknown"
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Hugging Face handling for tests marked huggingface."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from importlib.util import find_spec
|
|
5
|
+
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def hf_disable_tokenizer_parallelism() -> None:
|
|
10
|
+
"""Disable HF tokenizers parallelism to avoid segfaults."""
|
|
11
|
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def hf_configure_logging() -> None:
|
|
15
|
+
"""Raise datasets and huggingface_hub logging to INFO when datasets is installed."""
|
|
16
|
+
if find_spec("datasets") is None:
|
|
17
|
+
return
|
|
18
|
+
import datasets
|
|
19
|
+
import huggingface_hub
|
|
20
|
+
|
|
21
|
+
datasets.logging.set_verbosity_info()
|
|
22
|
+
huggingface_hub.logging.set_verbosity_info()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def hf_apply_collection_markers(items: list[pytest.Item]) -> None:
|
|
26
|
+
hf_items = [item for item in items if item.get_closest_marker("huggingface") is not None]
|
|
27
|
+
|
|
28
|
+
if not hf_items:
|
|
29
|
+
return
|
|
30
|
+
|
|
31
|
+
if not os.environ.get("HF_TOKEN", "").strip():
|
|
32
|
+
for item in hf_items:
|
|
33
|
+
item.add_marker(pytest.mark.skip(reason=f"HF_TOKEN not set ({item.name})"))
|
|
34
|
+
|
|
35
|
+
else:
|
|
36
|
+
for item in hf_items:
|
|
37
|
+
item.add_marker(pytest.mark.flaky(reruns=2, reruns_delay=60.0))
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _is_hf_gated_dataset_failure(
|
|
41
|
+
item: pytest.Item, call: pytest.CallInfo, report: pytest.TestReport
|
|
42
|
+
) -> bool:
|
|
43
|
+
return (
|
|
44
|
+
report.when == "call"
|
|
45
|
+
and report.failed
|
|
46
|
+
and item.get_closest_marker("huggingface") is not None
|
|
47
|
+
and bool(call.excinfo)
|
|
48
|
+
and is_gated_dataset_exception(call.excinfo.value) # type: ignore
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def hf_convert_gated_failure_to_skip(
|
|
53
|
+
item: pytest.Item, call: pytest.CallInfo, report: pytest.TestReport
|
|
54
|
+
) -> None:
|
|
55
|
+
if _is_hf_gated_dataset_failure(item, call, report):
|
|
56
|
+
report.outcome = "skipped"
|
|
57
|
+
line_number = item.reportinfo()[1] or 0
|
|
58
|
+
report.longrepr = (
|
|
59
|
+
str(item.fspath),
|
|
60
|
+
line_number,
|
|
61
|
+
f"Skipped: Gated dataset ({item.name})",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def is_gated_dataset_exception(exc: BaseException) -> bool:
|
|
66
|
+
exc_name = type(exc).__name__
|
|
67
|
+
return exc_name == "GatedRepoError" or (
|
|
68
|
+
exc_name == "DatasetNotFoundError" and "gated dataset" in str(exc).lower()
|
|
69
|
+
)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Skip tests that can't run on Windows."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def windows_skip_unsupported_tests(items: list[pytest.Item], platform: str | None = None) -> None:
|
|
9
|
+
"""Skip POSIX-only and Docker tests on Windows."""
|
|
10
|
+
if (platform or sys.platform) != "win32":
|
|
11
|
+
return
|
|
12
|
+
|
|
13
|
+
for item in items:
|
|
14
|
+
if item.get_closest_marker("posix_only") is not None:
|
|
15
|
+
item.add_marker(
|
|
16
|
+
pytest.mark.skip(
|
|
17
|
+
reason=f"Skipping {item.name}: test requires POSIX system (not supported on Windows)"
|
|
18
|
+
)
|
|
19
|
+
)
|
|
20
|
+
if item.get_closest_marker("docker") is not None:
|
|
21
|
+
item.add_marker(
|
|
22
|
+
pytest.mark.skip(
|
|
23
|
+
reason=f"Skipping {item.name}: Docker tests are not supported on Windows CI"
|
|
24
|
+
)
|
|
25
|
+
)
|