pytest-inspect-evals 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Matt Fisher
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,148 @@
1
+ Metadata-Version: 2.4
2
+ Name: pytest-inspect-evals
3
+ Version: 0.1.0
4
+ Summary: pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations
5
+ Author: Matt Fisher
6
+ Author-email: Matt Fisher <m@ttfisher.com>
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Classifier: Framework :: Pytest
10
+ Requires-Dist: inspect-ai>=0.3.262
11
+ Requires-Dist: numpy>=1.26.0
12
+ Requires-Dist: pydantic>=2.10.0
13
+ Requires-Dist: pytest>=8.0
14
+ Requires-Dist: pytest-rerunfailures>=15
15
+ Requires-Dist: requests>=2.32.0
16
+ Requires-Dist: urllib3>=2.0
17
+ Requires-Dist: datasets>=4.8.5 ; extra == 'huggingface'
18
+ Requires-Python: >=3.11
19
+ Project-URL: Homepage, https://github.com/Generality-Labs/pytest-inspect-evals
20
+ Project-URL: Repository, https://github.com/Generality-Labs/pytest-inspect-evals
21
+ Project-URL: Issues, https://github.com/Generality-Labs/pytest-inspect-evals/issues
22
+ Provides-Extra: huggingface
23
+ Description-Content-Type: text/markdown
24
+
25
+ # pytest-inspect-evals
26
+
27
+ A pytest plugin with the test gates, fixtures and helpers shared by [Inspect AI](https://inspect.aisi.org.uk/) evaluation repos. It started as the shared test code in [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). Installing it replaces a copied `conftest.py` and `tests/utils/` with one pinned dependency.
28
+
29
+ ## Install
30
+
31
+ ```bash
32
+ uv add --dev pytest-inspect-evals
33
+ # or, to use the Hugging Face dataset helpers:
34
+ uv add --dev "pytest-inspect-evals[huggingface]"
35
+ ```
36
+
37
+ The plugin loads automatically. There is nothing to add to `conftest.py`.
38
+
39
+ ## Test gates
40
+
41
+ Four markers mark tests that are skipped unless you switch them on.
42
+
43
+ | Marker | CLI flag | Env var | Pytest config option | Default |
44
+ | ------------------ | -------------------- | ---------------------------- | -------------------------------- | ------- |
45
+ | `slow` | `--runslow` | `RUN_SLOW_TESTS` | `inspect_evals_slow` | off |
46
+ | `dataset_download` | `--dataset-download` | `RUN_DATASET_DOWNLOAD_TESTS` | `inspect_evals_dataset_download` | off |
47
+ | `k8s` | `--runk8s` | `RUN_K8S_TESTS` | `inspect_evals_k8s` | off |
48
+ | `gpu` | `--rungpu` | `RUN_GPU_TESTS` | `inspect_evals_gpu` | off |
49
+
50
+ The env var wins when it is set to a non-empty value: `1`, `true`, `yes` or `on` switch the gate on, anything else switches it off. Otherwise the CLI flag switches it on. Otherwise the config option decides, and the default is off. An empty env var counts as unset.
51
+
52
+ To run a gate by default in your repo, set its option in `pyproject.toml`:
53
+
54
+ ```toml
55
+ [tool.pytest.ini_options]
56
+ inspect_evals_dataset_download = true
57
+ ```
58
+
59
+ Three more markers have fixed behaviour:
60
+
61
+ - `huggingface`: skipped when `HF_TOKEN` is unset or blank. Otherwise the test is retried up to twice, 60 seconds apart, and a gated-dataset error (`GatedRepoError`, or `DatasetNotFoundError` mentioning a gated dataset) is reported as a skip.
62
+ - `docker` and `posix_only`: skipped on Windows.
63
+
64
+ The plugin registers all seven markers, so they work under `--strict-markers` without any config.
65
+
66
+ ## Adding your own gate
67
+
68
+ `skip_if_marker_present` is the function the plugin uses. Call it from your `conftest.py` for repo-specific gates:
69
+
70
+ ```python
71
+ from pytest_inspect_evals.gates import skip_if_marker_present
72
+
73
+
74
+ def pytest_addoption(parser):
75
+ parser.addoption("--custom-smoke", action="store_true", default=False)
76
+
77
+
78
+ def pytest_configure(config):
79
+ config.addinivalue_line("markers", "custom_smoke: tests that call a live service")
80
+
81
+
82
+ def pytest_collection_modifyitems(config, items):
83
+ skip_if_marker_present(
84
+ config,
85
+ items,
86
+ marker="custom_smoke",
87
+ cli_flag="--custom-smoke",
88
+ env_var="RUN_CUSTOM_SMOKE_TESTS",
89
+ reason="custom_smoke tests disabled (set RUN_CUSTOM_SMOKE_TESTS=1 or pass --custom-smoke)",
90
+ )
91
+ ```
92
+
93
+ ## Fixtures
94
+
95
+ `set_model_roles` returns a context manager that sets model roles, for testing scorers that call a grader model:
96
+
97
+ ```python
98
+ def test_judge_scorer(set_model_roles):
99
+ with set_model_roles(grader=get_model("mockllm/model")):
100
+ ...
101
+ ```
102
+
103
+ `mock_docker_sandbox` stubs the Docker sandbox lifecycle, so an eval with `sandbox="docker"` runs without Docker. Combine it with mocked `sandbox().exec` results to test a scorer end to end:
104
+
105
+ ```python
106
+ def test_e2e(mock_docker_sandbox):
107
+ [log] = eval(my_task(), model="mockllm/model")
108
+ assert log.status == "success"
109
+ ```
110
+
111
+ ## Helpers
112
+
113
+ - `pytest_inspect_evals.assertions`: `assert_task_structure`, `assert_eval_success`, `run_single_sample_eval`, `get_metric_value`.
114
+ - `pytest_inspect_evals.solvers`: `mock_solver_with_output`, `run_command`.
115
+ - `pytest_inspect_evals.sandbox`: `MockExecResult`, `create_sandbox_tool_task`, `assert_sandbox_test_passed`.
116
+ - `pytest_inspect_evals.metrics`: `run_metrics`, `assert_agreeing_epochs_change_nothing`, `MetricEvalError`. These run custom metrics through the real epoch reducer, the way an eval does.
117
+ - `pytest_inspect_evals.huggingface` (needs the `huggingface` extra): `assert_huggingface_dataset_structure`, `get_dataset_infos_dict`, `assert_huggingface_dataset_is_valid`, `assert_dataset_contains_subsets`, `assert_dataset_has_columns`. These check a dataset's schema through the Hugging Face dataset viewer API without downloading it.
118
+
119
+ ## Migrating from a copied conftest
120
+
121
+ If your `conftest.py` came from inspect_evals or inspect-evals-template, remove what the plugin now provides:
122
+
123
+ - the `pytest_addoption` lines for `--runslow`, `--dataset-download`, `--runk8s` and `--rungpu`;
124
+ - the matching skip logic in `pytest_collection_modifyitems`;
125
+ - the seven markers above from your `markers =` config (keeping them is harmless).
126
+
127
+ If you keep the options, pytest stops at startup with `argparse.ArgumentError: argument --runslow: conflicting option string: --runslow`.
128
+
129
+ ## Private inspect_ai APIs
130
+
131
+ Two fixtures use inspect_ai internals that have no public equivalent: `set_model_roles` calls `inspect_ai.model._model.init_model_roles`, and `mock_docker_sandbox` patches `inspect_ai.util._sandbox.docker.docker.DockerSandboxEnvironment`. A future inspect_ai release could break them.
132
+
133
+ ## Development
134
+
135
+ ```bash
136
+ uv sync
137
+ uv run pre-commit install # optional: run the lint stack on every commit
138
+ uv run pytest
139
+ uv run basedpyright src
140
+ ```
141
+
142
+ Each pull request adds a changelog fragment rather than editing `CHANGELOG.md`, so concurrent PRs don't conflict. Run `uv run scriv create`, uncomment the sections that apply in the new file under `changelog.d/`, and commit it with the change, or delete it if the change needs no entry. To release, run the **Prepare release** workflow from the Actions tab. It bumps `version` in `pyproject.toml`, collects the fragments into `CHANGELOG.md`, and opens a release pull request, whose CI starts once you click **Approve workflows to run** on it; merging it tags the release. It needs _Settings → Actions → General_ → **Allow GitHub Actions to create and approve pull requests**. By hand, the same is `uv version --bump minor` and `uv run scriv collect`.
143
+
144
+ Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
145
+
146
+ ## Releasing
147
+
148
+ See [RELEASING.md](RELEASING.md).
@@ -0,0 +1,124 @@
1
+ # pytest-inspect-evals
2
+
3
+ A pytest plugin with the test gates, fixtures and helpers shared by [Inspect AI](https://inspect.aisi.org.uk/) evaluation repos. It started as the shared test code in [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). Installing it replaces a copied `conftest.py` and `tests/utils/` with one pinned dependency.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ uv add --dev pytest-inspect-evals
9
+ # or, to use the Hugging Face dataset helpers:
10
+ uv add --dev "pytest-inspect-evals[huggingface]"
11
+ ```
12
+
13
+ The plugin loads automatically. There is nothing to add to `conftest.py`.
14
+
15
+ ## Test gates
16
+
17
+ Four markers mark tests that are skipped unless you switch them on.
18
+
19
+ | Marker | CLI flag | Env var | Pytest config option | Default |
20
+ | ------------------ | -------------------- | ---------------------------- | -------------------------------- | ------- |
21
+ | `slow` | `--runslow` | `RUN_SLOW_TESTS` | `inspect_evals_slow` | off |
22
+ | `dataset_download` | `--dataset-download` | `RUN_DATASET_DOWNLOAD_TESTS` | `inspect_evals_dataset_download` | off |
23
+ | `k8s` | `--runk8s` | `RUN_K8S_TESTS` | `inspect_evals_k8s` | off |
24
+ | `gpu` | `--rungpu` | `RUN_GPU_TESTS` | `inspect_evals_gpu` | off |
25
+
26
+ The env var wins when it is set to a non-empty value: `1`, `true`, `yes` or `on` switch the gate on, anything else switches it off. Otherwise the CLI flag switches it on. Otherwise the config option decides, and the default is off. An empty env var counts as unset.
27
+
28
+ To run a gate by default in your repo, set its option in `pyproject.toml`:
29
+
30
+ ```toml
31
+ [tool.pytest.ini_options]
32
+ inspect_evals_dataset_download = true
33
+ ```
34
+
35
+ Three more markers have fixed behaviour:
36
+
37
+ - `huggingface`: skipped when `HF_TOKEN` is unset or blank. Otherwise the test is retried up to twice, 60 seconds apart, and a gated-dataset error (`GatedRepoError`, or `DatasetNotFoundError` mentioning a gated dataset) is reported as a skip.
38
+ - `docker` and `posix_only`: skipped on Windows.
39
+
40
+ The plugin registers all seven markers, so they work under `--strict-markers` without any config.
41
+
42
+ ## Adding your own gate
43
+
44
+ `skip_if_marker_present` is the function the plugin uses. Call it from your `conftest.py` for repo-specific gates:
45
+
46
+ ```python
47
+ from pytest_inspect_evals.gates import skip_if_marker_present
48
+
49
+
50
+ def pytest_addoption(parser):
51
+ parser.addoption("--custom-smoke", action="store_true", default=False)
52
+
53
+
54
+ def pytest_configure(config):
55
+ config.addinivalue_line("markers", "custom_smoke: tests that call a live service")
56
+
57
+
58
+ def pytest_collection_modifyitems(config, items):
59
+ skip_if_marker_present(
60
+ config,
61
+ items,
62
+ marker="custom_smoke",
63
+ cli_flag="--custom-smoke",
64
+ env_var="RUN_CUSTOM_SMOKE_TESTS",
65
+ reason="custom_smoke tests disabled (set RUN_CUSTOM_SMOKE_TESTS=1 or pass --custom-smoke)",
66
+ )
67
+ ```
68
+
69
+ ## Fixtures
70
+
71
+ `set_model_roles` returns a context manager that sets model roles, for testing scorers that call a grader model:
72
+
73
+ ```python
74
+ def test_judge_scorer(set_model_roles):
75
+ with set_model_roles(grader=get_model("mockllm/model")):
76
+ ...
77
+ ```
78
+
79
+ `mock_docker_sandbox` stubs the Docker sandbox lifecycle, so an eval with `sandbox="docker"` runs without Docker. Combine it with mocked `sandbox().exec` results to test a scorer end to end:
80
+
81
+ ```python
82
+ def test_e2e(mock_docker_sandbox):
83
+ [log] = eval(my_task(), model="mockllm/model")
84
+ assert log.status == "success"
85
+ ```
86
+
87
+ ## Helpers
88
+
89
+ - `pytest_inspect_evals.assertions`: `assert_task_structure`, `assert_eval_success`, `run_single_sample_eval`, `get_metric_value`.
90
+ - `pytest_inspect_evals.solvers`: `mock_solver_with_output`, `run_command`.
91
+ - `pytest_inspect_evals.sandbox`: `MockExecResult`, `create_sandbox_tool_task`, `assert_sandbox_test_passed`.
92
+ - `pytest_inspect_evals.metrics`: `run_metrics`, `assert_agreeing_epochs_change_nothing`, `MetricEvalError`. These run custom metrics through the real epoch reducer, the way an eval does.
93
+ - `pytest_inspect_evals.huggingface` (needs the `huggingface` extra): `assert_huggingface_dataset_structure`, `get_dataset_infos_dict`, `assert_huggingface_dataset_is_valid`, `assert_dataset_contains_subsets`, `assert_dataset_has_columns`. These check a dataset's schema through the Hugging Face dataset viewer API without downloading it.
94
+
95
+ ## Migrating from a copied conftest
96
+
97
+ If your `conftest.py` came from inspect_evals or inspect-evals-template, remove what the plugin now provides:
98
+
99
+ - the `pytest_addoption` lines for `--runslow`, `--dataset-download`, `--runk8s` and `--rungpu`;
100
+ - the matching skip logic in `pytest_collection_modifyitems`;
101
+ - the seven markers above from your `markers =` config (keeping them is harmless).
102
+
103
+ If you keep the options, pytest stops at startup with `argparse.ArgumentError: argument --runslow: conflicting option string: --runslow`.
104
+
105
+ ## Private inspect_ai APIs
106
+
107
+ Two fixtures use inspect_ai internals that have no public equivalent: `set_model_roles` calls `inspect_ai.model._model.init_model_roles`, and `mock_docker_sandbox` patches `inspect_ai.util._sandbox.docker.docker.DockerSandboxEnvironment`. A future inspect_ai release could break them.
108
+
109
+ ## Development
110
+
111
+ ```bash
112
+ uv sync
113
+ uv run pre-commit install # optional: run the lint stack on every commit
114
+ uv run pytest
115
+ uv run basedpyright src
116
+ ```
117
+
118
+ Each pull request adds a changelog fragment rather than editing `CHANGELOG.md`, so concurrent PRs don't conflict. Run `uv run scriv create`, uncomment the sections that apply in the new file under `changelog.d/`, and commit it with the change, or delete it if the change needs no entry. To release, run the **Prepare release** workflow from the Actions tab. It bumps `version` in `pyproject.toml`, collects the fragments into `CHANGELOG.md`, and opens a release pull request, whose CI starts once you click **Approve workflows to run** on it; merging it tags the release. It needs _Settings → Actions → General_ → **Allow GitHub Actions to create and approve pull requests**. By hand, the same is `uv version --bump minor` and `uv run scriv collect`.
119
+
120
+ Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
121
+
122
+ ## Releasing
123
+
124
+ See [RELEASING.md](RELEASING.md).
@@ -0,0 +1,121 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.12.22,<0.13.0"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "pytest-inspect-evals"
7
+ version = "0.1.0"
8
+ description = "pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.11"
13
+ classifiers = ["Framework :: Pytest"]
14
+ dependencies = [
15
+ "inspect_ai>=0.3.262",
16
+ "numpy>=1.26.0",
17
+ "pydantic>=2.10.0",
18
+ "pytest>=8.0",
19
+ "pytest-rerunfailures>=15",
20
+ "requests>=2.32.0",
21
+ "urllib3>=2.0",
22
+ ]
23
+
24
+ [[project.authors]]
25
+ name = "Matt Fisher"
26
+ email = "m@ttfisher.com"
27
+
28
+ [project.optional-dependencies]
29
+ huggingface = ["datasets>=4.8.5"]
30
+
31
+ [project.entry-points.pytest11]
32
+ pytest_inspect_evals = "pytest_inspect_evals.plugin"
33
+
34
+ [project.urls]
35
+ Homepage = "https://github.com/Generality-Labs/pytest-inspect-evals"
36
+ Repository = "https://github.com/Generality-Labs/pytest-inspect-evals"
37
+ Issues = "https://github.com/Generality-Labs/pytest-inspect-evals/issues"
38
+
39
+ [dependency-groups]
40
+ dev = [
41
+ "pytest>=8.0.0",
42
+ "pytest-cov>=7.0.0",
43
+ "basedpyright>=1.39",
44
+ "pre-commit>=4.0.0",
45
+ "scriv>=1.8",
46
+ "datasets>=4.8.5",
47
+ "pytest-asyncio>=1.0",
48
+ "mdformat>=1.0.0",
49
+ "mdformat-gfm>=1.0.0",
50
+ ]
51
+
52
+ [tool.scriv]
53
+ format = "md"
54
+ fragment_directory = "changelog.d"
55
+ categories = [
56
+ "Added",
57
+ "Changed",
58
+ "Deprecated",
59
+ "Removed",
60
+ "Fixed",
61
+ "Security",
62
+ ]
63
+ skip_fragments = "[A-Z]*"
64
+ new_fragment_template = "file: TEMPLATE.md"
65
+ entry_title_template = "[{{ version }}] - {{ date.strftime('%Y-%m-%d') }}"
66
+ md_header_level = "2"
67
+ md_html_anchors = false
68
+ compact_fragments = true
69
+ version = "literal: pyproject.toml: project.version"
70
+
71
+ [tool.ruff]
72
+ line-length = 100
73
+ src = ["src"]
74
+ target-version = "py311"
75
+
76
+ [tool.ruff.lint]
77
+ select = [
78
+ "E",
79
+ "W",
80
+ "F",
81
+ "I",
82
+ "UP",
83
+ "B",
84
+ "SIM",
85
+ "D",
86
+ "C4",
87
+ "PT",
88
+ "PIE",
89
+ "DTZ",
90
+ "ISC",
91
+ "ASYNC",
92
+ "N",
93
+ "FURB",
94
+ "RUF",
95
+ "PLE",
96
+ "PLW",
97
+ ]
98
+ ignore = [
99
+ "E501",
100
+ "D10",
101
+ "D415",
102
+ "ISC001",
103
+ ]
104
+
105
+ [tool.ruff.lint.pydocstyle]
106
+ convention = "google"
107
+
108
+ [tool.ruff.lint.per-file-ignores]
109
+ "tests/**" = ["D"]
110
+
111
+ [tool.basedpyright]
112
+ pythonVersion = "3.11"
113
+ typeCheckingMode = "standard"
114
+ reportMissingTypeStubs = false
115
+
116
+ [tool.pytest.ini_options]
117
+ testpaths = ["tests"]
118
+ addopts = "-rA --durations=10 --color=yes --cov=src --cov-report=term-missing"
119
+ asyncio_mode = "auto"
120
+ asyncio_default_fixture_loop_scope = "function"
121
+ filterwarnings = ['ignore:The configuration option "asyncio_default_fixture_loop_scope" is unset:pytest.PytestDeprecationWarning']
@@ -0,0 +1,127 @@
1
+ [build-system]
2
+ # uv's own backend. It builds src/pytest_inspect_evals/ with no further config.
3
+ # Keep the upper bound at the next uv minor, as `uv init` writes it, and raise
4
+ # both bounds when moving to a new uv minor.
5
+ requires = ["uv_build>=0.12.22,<0.13.0"]
6
+ build-backend = "uv_build"
7
+
8
+ [project]
9
+ name = "pytest-inspect-evals"
10
+ version = "0.1.0"
11
+ description = "pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"
12
+ readme = "README.md"
13
+ license = "MIT"
14
+ license-files = ["LICENSE"] # uv_build ships only the license files named here
15
+ authors = [{ name = "Matt Fisher", email = "m@ttfisher.com" }]
16
+ requires-python = ">=3.11"
17
+ classifiers = ["Framework :: Pytest"]
18
+ dependencies = [
19
+ "inspect_ai>=0.3.262",
20
+ "numpy>=1.26.0",
21
+ "pydantic>=2.10.0",
22
+ "pytest>=8.0",
23
+ "pytest-rerunfailures>=15",
24
+ "requests>=2.32.0",
25
+ "urllib3>=2.0",
26
+ ]
27
+
28
+ [project.optional-dependencies]
29
+ # Only pytest_inspect_evals.huggingface needs this.
30
+ huggingface = ["datasets>=4.8.5"]
31
+
32
+ [project.entry-points.pytest11]
33
+ pytest_inspect_evals = "pytest_inspect_evals.plugin"
34
+
35
+ [project.urls]
36
+ Homepage = "https://github.com/Generality-Labs/pytest-inspect-evals"
37
+ Repository = "https://github.com/Generality-Labs/pytest-inspect-evals"
38
+ Issues = "https://github.com/Generality-Labs/pytest-inspect-evals/issues"
39
+ [dependency-groups]
40
+ dev = [
41
+ "pytest>=8.0.0",
42
+ "pytest-cov>=7.0.0",
43
+ "basedpyright>=1.39",
44
+ "pre-commit>=4.0.0",
45
+ "scriv>=1.8",
46
+ "datasets>=4.8.5",
47
+ "pytest-asyncio>=1.0",
48
+ "mdformat>=1.0.0",
49
+ "mdformat-gfm>=1.0.0",
50
+ ]
51
+
52
+
53
+ [tool.scriv]
54
+ # One fragment per pull request in changelog.d/, collected into CHANGELOG.md
55
+ # at release time, so concurrent PRs never edit the same lines of the changelog.
56
+ format = "md"
57
+ fragment_directory = "changelog.d"
58
+ categories = ["Added", "Changed", "Deprecated", "Removed", "Fixed", "Security"]
59
+ skip_fragments = "[A-Z]*" # TEMPLATE.md
60
+ new_fragment_template = "file: TEMPLATE.md"
61
+ entry_title_template = "[{{ version }}] - {{ date.strftime('%Y-%m-%d') }}"
62
+ md_header_level = "2"
63
+ md_html_anchors = false
64
+ # Without this, two fragments in one category leave a blank line between them,
65
+ # which makes the list loose and fails mdformat on the release commit.
66
+ compact_fragments = true
67
+ # The version's single source is [project] version above, which
68
+ # `uv version --bump` (and the Prepare release workflow) rewrites.
69
+ version = "literal: pyproject.toml: project.version"
70
+
71
+ [tool.ruff]
72
+ line-length = 100
73
+ src = ["src"]
74
+ target-version = "py311"
75
+
76
+ [tool.ruff.lint]
77
+ select = [
78
+ "E", # pycodestyle errors
79
+ "W", # pycodestyle warnings
80
+ "F", # pyflakes
81
+ "I", # isort
82
+ "UP", # pyupgrade
83
+ "B", # flake8-bugbear
84
+ "SIM", # flake8-simplify
85
+ "D", # pydocstyle
86
+ "C4", # flake8-comprehensions
87
+ "PT", # flake8-pytest-style
88
+ "PIE", # flake8-pie
89
+ "DTZ", # flake8-datetimez (timezone-aware datetimes)
90
+ "ISC", # implicit-str-concat (catches missing commas)
91
+ "ASYNC", # flake8-async
92
+ "N", # pep8-naming
93
+ "FURB", # refurb (modernization)
94
+ "RUF", # ruff-specific rules
95
+ "PLE", # pylint errors
96
+ "PLW", # pylint warnings
97
+ ]
98
+ ignore = [
99
+ "E501", # line-too-long — the formatter owns line length
100
+ "D10", # missing docstrings — don't force them
101
+ "D415", # first-line punctuation — the module docstring is the project
102
+ # description, whose punctuation varies
103
+ "ISC001", # single-line implicit concat — conflicts with the formatter
104
+ ]
105
+
106
+ [tool.ruff.lint.pydocstyle]
107
+ convention = "google"
108
+
109
+ [tool.ruff.lint.per-file-ignores]
110
+ "tests/**" = ["D"] # tests don't need docstring linting
111
+
112
+ [tool.basedpyright]
113
+ pythonVersion = "3.11"
114
+ # "standard", not "strict": v0.1 moves code verbatim from inspect_evals, which
115
+ # was checked with mypy. Tighten once the code is ours.
116
+ typeCheckingMode = "standard"
117
+ reportMissingTypeStubs = false # third-party libs are often untyped
118
+
119
+ [tool.pytest.ini_options]
120
+ testpaths = ["tests"]
121
+ addopts = "-rA --durations=10 --color=yes --cov=src --cov-report=term-missing"
122
+ asyncio_mode = "auto"
123
+ asyncio_default_fixture_loop_scope = "function"
124
+ filterwarnings = [
125
+ # pytester's throwaway test dirs have no pytest-asyncio config of their own.
126
+ 'ignore:The configuration option "asyncio_default_fixture_loop_scope" is unset:pytest.PytestDeprecationWarning',
127
+ ]
@@ -0,0 +1,10 @@
1
+ """pytest plugin with shared test gates, fixtures and helpers for Inspect AI evaluations"""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ # The version is set once, in pyproject.toml; this reads it back from the
6
+ # installed package metadata.
7
+ try:
8
+ __version__ = version("pytest-inspect-evals")
9
+ except PackageNotFoundError: # a source tree that was never installed
10
+ __version__ = "unknown"
@@ -0,0 +1,69 @@
1
+ """Hugging Face handling for tests marked huggingface."""
2
+
3
+ import os
4
+ from importlib.util import find_spec
5
+
6
+ import pytest
7
+
8
+
9
+ def hf_disable_tokenizer_parallelism() -> None:
10
+ """Disable HF tokenizers parallelism to avoid segfaults."""
11
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
12
+
13
+
14
+ def hf_configure_logging() -> None:
15
+ """Raise datasets and huggingface_hub logging to INFO when datasets is installed."""
16
+ if find_spec("datasets") is None:
17
+ return
18
+ import datasets
19
+ import huggingface_hub
20
+
21
+ datasets.logging.set_verbosity_info()
22
+ huggingface_hub.logging.set_verbosity_info()
23
+
24
+
25
+ def hf_apply_collection_markers(items: list[pytest.Item]) -> None:
26
+ hf_items = [item for item in items if item.get_closest_marker("huggingface") is not None]
27
+
28
+ if not hf_items:
29
+ return
30
+
31
+ if not os.environ.get("HF_TOKEN", "").strip():
32
+ for item in hf_items:
33
+ item.add_marker(pytest.mark.skip(reason=f"HF_TOKEN not set ({item.name})"))
34
+
35
+ else:
36
+ for item in hf_items:
37
+ item.add_marker(pytest.mark.flaky(reruns=2, reruns_delay=60.0))
38
+
39
+
40
+ def _is_hf_gated_dataset_failure(
41
+ item: pytest.Item, call: pytest.CallInfo, report: pytest.TestReport
42
+ ) -> bool:
43
+ return (
44
+ report.when == "call"
45
+ and report.failed
46
+ and item.get_closest_marker("huggingface") is not None
47
+ and bool(call.excinfo)
48
+ and is_gated_dataset_exception(call.excinfo.value) # type: ignore
49
+ )
50
+
51
+
52
+ def hf_convert_gated_failure_to_skip(
53
+ item: pytest.Item, call: pytest.CallInfo, report: pytest.TestReport
54
+ ) -> None:
55
+ if _is_hf_gated_dataset_failure(item, call, report):
56
+ report.outcome = "skipped"
57
+ line_number = item.reportinfo()[1] or 0
58
+ report.longrepr = (
59
+ str(item.fspath),
60
+ line_number,
61
+ f"Skipped: Gated dataset ({item.name})",
62
+ )
63
+
64
+
65
+ def is_gated_dataset_exception(exc: BaseException) -> bool:
66
+ exc_name = type(exc).__name__
67
+ return exc_name == "GatedRepoError" or (
68
+ exc_name == "DatasetNotFoundError" and "gated dataset" in str(exc).lower()
69
+ )
@@ -0,0 +1,25 @@
1
+ """Skip tests that can't run on Windows."""
2
+
3
+ import sys
4
+
5
+ import pytest
6
+
7
+
8
+ def windows_skip_unsupported_tests(items: list[pytest.Item], platform: str | None = None) -> None:
9
+ """Skip POSIX-only and Docker tests on Windows."""
10
+ if (platform or sys.platform) != "win32":
11
+ return
12
+
13
+ for item in items:
14
+ if item.get_closest_marker("posix_only") is not None:
15
+ item.add_marker(
16
+ pytest.mark.skip(
17
+ reason=f"Skipping {item.name}: test requires POSIX system (not supported on Windows)"
18
+ )
19
+ )
20
+ if item.get_closest_marker("docker") is not None:
21
+ item.add_marker(
22
+ pytest.mark.skip(
23
+ reason=f"Skipping {item.name}: Docker tests are not supported on Windows CI"
24
+ )
25
+ )