inspect-evals-lint 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- inspect_evals_lint-0.1.0/.claude/hooks/session-start.sh +18 -0
- inspect_evals_lint-0.1.0/.claude/settings.json +14 -0
- inspect_evals_lint-0.1.0/.copier-answers.yml +18 -0
- inspect_evals_lint-0.1.0/.github/dependabot.yml +30 -0
- inspect_evals_lint-0.1.0/.github/repo-settings.json +10 -0
- inspect_evals_lint-0.1.0/.github/rulesets/main.json +40 -0
- inspect_evals_lint-0.1.0/.github/workflows/ci.yml +20 -0
- inspect_evals_lint-0.1.0/.github/workflows/publish.yml +36 -0
- inspect_evals_lint-0.1.0/.github/workflows/template-update.yml +29 -0
- inspect_evals_lint-0.1.0/.github/zizmor.yml +9 -0
- inspect_evals_lint-0.1.0/.gitignore +43 -0
- inspect_evals_lint-0.1.0/.mdformat.toml +3 -0
- inspect_evals_lint-0.1.0/.pre-commit-config.yaml +91 -0
- inspect_evals_lint-0.1.0/.python-version +1 -0
- inspect_evals_lint-0.1.0/CHANGELOG.md +24 -0
- inspect_evals_lint-0.1.0/LICENSE +21 -0
- inspect_evals_lint-0.1.0/PKG-INFO +116 -0
- inspect_evals_lint-0.1.0/README.md +101 -0
- inspect_evals_lint-0.1.0/RELEASING.md +35 -0
- inspect_evals_lint-0.1.0/docs/CHECKS.md +42 -0
- inspect_evals_lint-0.1.0/pyproject.toml +93 -0
- inspect_evals_lint-0.1.0/scripts/setup-repo.sh +93 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/__init__.py +31 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/__main__.py +6 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/__init__.py +57 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/best_practices.py +233 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/code_quality.py +91 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/dependencies.py +255 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/file_structure.py +456 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/sandbox.py +139 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/tests.py +303 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/checks/utils.py +127 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/cli.py +151 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/config.py +237 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/models.py +81 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/output.py +325 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/runner.py +210 -0
- inspect_evals_lint-0.1.0/src/inspect_evals_lint/suppressions.py +68 -0
- inspect_evals_lint-0.1.0/tests/__init__.py +0 -0
- inspect_evals_lint-0.1.0/tests/conftest.py +149 -0
- inspect_evals_lint-0.1.0/tests/test_best_practices.py +388 -0
- inspect_evals_lint-0.1.0/tests/test_checks.py +497 -0
- inspect_evals_lint-0.1.0/tests/test_cli.py +127 -0
- inspect_evals_lint-0.1.0/tests/test_config.py +145 -0
- inspect_evals_lint-0.1.0/tests/test_output.py +172 -0
- inspect_evals_lint-0.1.0/tests/test_runner.py +330 -0
- inspect_evals_lint-0.1.0/tests/test_suppressions.py +66 -0
- inspect_evals_lint-0.1.0/uv.lock +506 -0
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
set -euo pipefail
|
|
3
|
+
|
|
4
|
+
if [ "${CLAUDE_CODE_REMOTE:-}" != "true" ]; then
|
|
5
|
+
exit 0
|
|
6
|
+
fi
|
|
7
|
+
|
|
8
|
+
cd "$CLAUDE_PROJECT_DIR"
|
|
9
|
+
|
|
10
|
+
# Install the project + dev tooling (pytest, basedpyright, pre-commit) into the uv-managed venv.
|
|
11
|
+
uv sync
|
|
12
|
+
|
|
13
|
+
# Installs the git hook (so `git commit` runs it automatically) and
|
|
14
|
+
# pre-warms hook environments/cache. Pre-existing findings shouldn't block
|
|
15
|
+
# session start, so this step is non-blocking - failures surface as normal
|
|
16
|
+
# output for the agent to see and address, not a broken session.
|
|
17
|
+
uv run pre-commit install
|
|
18
|
+
uv run pre-commit run --all-files || true
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# This file is auto-generated and updated by Copier; do not edit by hand.
|
|
2
|
+
# Run `copier update` to pull in template changes.
|
|
3
|
+
_commit: v1.8.1
|
|
4
|
+
_src_path: gh:Generality-Labs/python-project-template
|
|
5
|
+
author_email: m@ttfisher.com
|
|
6
|
+
author_name: Matt Fisher
|
|
7
|
+
github_owner: Generality-Labs
|
|
8
|
+
license: MIT
|
|
9
|
+
project_description: 'Static checks for Inspect AI evaluations: structure, tests,
|
|
10
|
+
best practices and sandbox pinning'
|
|
11
|
+
project_kind: library
|
|
12
|
+
project_name: inspect-evals-lint
|
|
13
|
+
publish_to_pypi: true
|
|
14
|
+
python_version: '3.11'
|
|
15
|
+
use_coverage_gate: false
|
|
16
|
+
use_frontend: false
|
|
17
|
+
use_template_update: true
|
|
18
|
+
use_typos: true
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
version: 2
|
|
2
|
+
updates:
|
|
3
|
+
# Bump the SHA-pinned GitHub Actions (opens PRs with changelogs + the new pin).
|
|
4
|
+
- package-ecosystem: github-actions
|
|
5
|
+
directory: /
|
|
6
|
+
schedule:
|
|
7
|
+
interval: weekly
|
|
8
|
+
cooldown:
|
|
9
|
+
default-days: 7 # don't adopt a release until it's had time to be vetted/yanked
|
|
10
|
+
groups:
|
|
11
|
+
actions:
|
|
12
|
+
patterns: ["*"]
|
|
13
|
+
ignore:
|
|
14
|
+
# Our own reusable workflows are referenced by moving major tag on
|
|
15
|
+
# purpose (see .github/zizmor.yml), so template CI fixes flow in
|
|
16
|
+
# without a bump PR per release. Left un-ignored, Dependabot rewrites
|
|
17
|
+
# `@v1` to a fixed `@v1.x.y` and re-bumps it on every release — and
|
|
18
|
+
# `copier update` restores the moving tag, so the two fight forever.
|
|
19
|
+
- dependency-name: "Generality-Labs/*"
|
|
20
|
+
|
|
21
|
+
# Keep Python dependencies (pyproject.toml + uv.lock) current.
|
|
22
|
+
- package-ecosystem: uv
|
|
23
|
+
directory: /
|
|
24
|
+
schedule:
|
|
25
|
+
interval: weekly
|
|
26
|
+
cooldown:
|
|
27
|
+
default-days: 7
|
|
28
|
+
groups:
|
|
29
|
+
python:
|
|
30
|
+
patterns: ["*"]
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "protect-main",
|
|
3
|
+
"target": "branch",
|
|
4
|
+
"enforcement": "active",
|
|
5
|
+
"conditions": {
|
|
6
|
+
"ref_name": {
|
|
7
|
+
"include": ["~DEFAULT_BRANCH"],
|
|
8
|
+
"exclude": []
|
|
9
|
+
}
|
|
10
|
+
},
|
|
11
|
+
"bypass_actors": [
|
|
12
|
+
{
|
|
13
|
+
"actor_id": 5,
|
|
14
|
+
"actor_type": "RepositoryRole",
|
|
15
|
+
"bypass_mode": "always"
|
|
16
|
+
}
|
|
17
|
+
],
|
|
18
|
+
"rules": [
|
|
19
|
+
{ "type": "deletion" },
|
|
20
|
+
{ "type": "non_fast_forward" },
|
|
21
|
+
{
|
|
22
|
+
"type": "pull_request",
|
|
23
|
+
"parameters": {
|
|
24
|
+
"required_approving_review_count": 0,
|
|
25
|
+
"dismiss_stale_reviews_on_push": false,
|
|
26
|
+
"require_code_owner_review": false,
|
|
27
|
+
"require_last_push_approval": false,
|
|
28
|
+
"required_review_thread_resolution": false
|
|
29
|
+
}
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
"type": "required_status_checks",
|
|
33
|
+
"parameters": {
|
|
34
|
+
"strict_required_status_checks_policy": false,
|
|
35
|
+
"required_status_checks": [
|
|
36
|
+
{ "context": "ci / Lint, type-check, and test" } ]
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
]
|
|
40
|
+
}
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches:
|
|
6
|
+
- main
|
|
7
|
+
pull_request:
|
|
8
|
+
|
|
9
|
+
permissions:
|
|
10
|
+
contents: read
|
|
11
|
+
|
|
12
|
+
jobs:
|
|
13
|
+
ci:
|
|
14
|
+
# Moving major tag on purpose (see .github/zizmor.yml): we own this repo,
|
|
15
|
+
# so CI fixes flow in without a bump PR per template release. Dependabot
|
|
16
|
+
# is told to leave it alone in .github/dependabot.yml — don't "helpfully"
|
|
17
|
+
# pin this to an exact version, or copier update will just revert it.
|
|
18
|
+
uses: Generality-Labs/python-project-template/.github/workflows/python-ci.yml@v1
|
|
19
|
+
with:
|
|
20
|
+
python-version: "3.11"
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
concurrency:
|
|
12
|
+
group: publish
|
|
13
|
+
cancel-in-progress: false
|
|
14
|
+
|
|
15
|
+
jobs:
|
|
16
|
+
publish:
|
|
17
|
+
name: Build and publish to PyPI
|
|
18
|
+
runs-on: ubuntu-latest
|
|
19
|
+
environment: pypi
|
|
20
|
+
permissions:
|
|
21
|
+
id-token: write # required for PyPI trusted publishing (OIDC, no API token)
|
|
22
|
+
steps:
|
|
23
|
+
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
24
|
+
with:
|
|
25
|
+
persist-credentials: false
|
|
26
|
+
|
|
27
|
+
- uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
|
28
|
+
with:
|
|
29
|
+
python-version: "3.11"
|
|
30
|
+
enable-cache: false # no caching in the privileged publish job (cache-poisoning surface)
|
|
31
|
+
|
|
32
|
+
- name: Build sdist + wheel
|
|
33
|
+
run: uv build
|
|
34
|
+
|
|
35
|
+
- name: Publish to PyPI
|
|
36
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
name: Template Update
|
|
2
|
+
|
|
3
|
+
# Pulls scaffolded-file changes from the template via `copier update`, opening
|
|
4
|
+
# a PR only when something actually changed. Changes to the shared reusable
|
|
5
|
+
# workflows need no run here — those are pinned `@v1` and propagate on their
|
|
6
|
+
# own when the tag moves.
|
|
7
|
+
#
|
|
8
|
+
# To update to something other than `v1`, run `uvx copier update` locally with
|
|
9
|
+
# `--vcs-ref`.
|
|
10
|
+
on:
|
|
11
|
+
schedule:
|
|
12
|
+
- cron: "0 6 * * 1" # Mondays, 06:00 UTC
|
|
13
|
+
workflow_dispatch:
|
|
14
|
+
|
|
15
|
+
permissions:
|
|
16
|
+
contents: read
|
|
17
|
+
|
|
18
|
+
jobs:
|
|
19
|
+
template-update:
|
|
20
|
+
uses: Generality-Labs/python-project-template/.github/workflows/template-update.yml@v1
|
|
21
|
+
permissions:
|
|
22
|
+
contents: write # push the branch holding the template update
|
|
23
|
+
pull-requests: write # open the PR for it
|
|
24
|
+
with:
|
|
25
|
+
# Set true, and pass the secret below, to have Claude attempt the merge
|
|
26
|
+
# conflicts copier can't resolve on its own.
|
|
27
|
+
resolve-conflicts-with-claude: false
|
|
28
|
+
# secrets:
|
|
29
|
+
# anthropic-api-key: ${{ secrets.ANTHROPIC_API_KEY }}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Allow tag/branch refs for our own reusable workflows and actions (we control
|
|
2
|
+
# those repos and pin them to a moving major tag on purpose); still require full
|
|
3
|
+
# commit-SHA pins for third-party actions.
|
|
4
|
+
rules:
|
|
5
|
+
unpinned-uses:
|
|
6
|
+
config:
|
|
7
|
+
policies:
|
|
8
|
+
"Generality-Labs/*": ref-pin
|
|
9
|
+
"*": hash-pin
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Byte-compiled
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# Virtual environments
|
|
7
|
+
.venv/
|
|
8
|
+
venv/
|
|
9
|
+
|
|
10
|
+
# Packaging / build artifacts
|
|
11
|
+
build/
|
|
12
|
+
dist/
|
|
13
|
+
*.egg-info/
|
|
14
|
+
*.egg
|
|
15
|
+
|
|
16
|
+
# Tool caches
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.ruff_cache/
|
|
19
|
+
|
|
20
|
+
# Coverage
|
|
21
|
+
.coverage
|
|
22
|
+
.coverage.*
|
|
23
|
+
coverage.xml
|
|
24
|
+
htmlcov/
|
|
25
|
+
|
|
26
|
+
# Local env / secrets — never commit API keys
|
|
27
|
+
.env
|
|
28
|
+
.env.*
|
|
29
|
+
!.env.example
|
|
30
|
+
|
|
31
|
+
# Claude Code per-user overrides (the shared .claude/settings.json is committed)
|
|
32
|
+
.claude/settings.local.json
|
|
33
|
+
|
|
34
|
+
# Eval / run output
|
|
35
|
+
logs/
|
|
36
|
+
|
|
37
|
+
# Editors
|
|
38
|
+
.idea/
|
|
39
|
+
*.swp
|
|
40
|
+
|
|
41
|
+
# OS
|
|
42
|
+
.DS_Store
|
|
43
|
+
Thumbs.db
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Latest Python version we support — keeps hook envs consistent.
|
|
2
|
+
default_language_version:
|
|
3
|
+
python: python3.12
|
|
4
|
+
|
|
5
|
+
repos:
|
|
6
|
+
- repo: https://github.com/pre-commit/pre-commit-hooks
|
|
7
|
+
rev: 3e8a8703264a2f4a69428a0aa4dcb512790b2c8c # v6.0.0
|
|
8
|
+
hooks:
|
|
9
|
+
- id: check-added-large-files
|
|
10
|
+
- id: check-json
|
|
11
|
+
- id: check-yaml
|
|
12
|
+
- id: debug-statements
|
|
13
|
+
- id: detect-private-key
|
|
14
|
+
- id: end-of-file-fixer
|
|
15
|
+
|
|
16
|
+
- repo: https://github.com/astral-sh/ruff-pre-commit
|
|
17
|
+
rev: c59bba8fb259db0fec2bbb77ad8ba51ea7341b56 # v0.15.20
|
|
18
|
+
hooks:
|
|
19
|
+
- id: ruff-check
|
|
20
|
+
args: [--fix]
|
|
21
|
+
- id: ruff-format
|
|
22
|
+
|
|
23
|
+
- repo: https://github.com/astral-sh/uv-pre-commit
|
|
24
|
+
rev: b7359babf83a050dcd7bf892538f3df09c3d4ba6 # 0.11.29
|
|
25
|
+
hooks:
|
|
26
|
+
- id: uv-lock # keep uv.lock in sync with pyproject.toml
|
|
27
|
+
|
|
28
|
+
- repo: https://github.com/rhysd/actionlint
|
|
29
|
+
rev: 914e7df21a07ef503a81201c76d2b11c789d3fca # v1.7.12
|
|
30
|
+
hooks:
|
|
31
|
+
- id: actionlint # GitHub Actions correctness (complements zizmor's security audits)
|
|
32
|
+
|
|
33
|
+
- repo: https://github.com/shellcheck-py/shellcheck-py
|
|
34
|
+
rev: 745eface02aef23e168a8afb6b5737818efbea95 # v0.11.0.1
|
|
35
|
+
hooks:
|
|
36
|
+
- id: shellcheck # lints shell scripts (e.g. .claude/hooks/session-start.sh)
|
|
37
|
+
# Pinned independently of default_language_version above, and it must
|
|
38
|
+
# stay below 3.13 when that default moves up. shellcheck-py ships no
|
|
39
|
+
# wheel in its git repo, so pre-commit builds it from source, and that
|
|
40
|
+
# build downloads the binary with a bare urllib.request.urlopen. Python
|
|
41
|
+
# 3.13 turns on ssl.VERIFY_X509_STRICT by default, which rejects a CA
|
|
42
|
+
# certificate carrying no keyUsage extension -- as the MITM proxies in
|
|
43
|
+
# front of some sandboxes do. The hook environment then can't be built at
|
|
44
|
+
# all, so `git commit` fails there, not merely `pre-commit run`. SKIP= is
|
|
45
|
+
# no help: it applies to `pre-commit run`, not to installation.
|
|
46
|
+
#
|
|
47
|
+
# The interpreter builds this hook's environment and nothing else, so it
|
|
48
|
+
# has no bearing on what shellcheck checks. The durable fix is a local
|
|
49
|
+
# hook using the prebuilt wheel from PyPI -- no build, no download -- at
|
|
50
|
+
# the cost of `pre-commit autoupdate` no longer tracking it.
|
|
51
|
+
language_version: python3.12
|
|
52
|
+
|
|
53
|
+
- repo: https://github.com/crate-ci/typos
|
|
54
|
+
rev: bee27e3a4fd1ea2111cf90ab89cd076c870fce14 # v1.48.0
|
|
55
|
+
hooks:
|
|
56
|
+
# Report only. The hook's own default args are
|
|
57
|
+
# [--write-changes, --force-exclude], which rewrite source rather than
|
|
58
|
+
# reporting it: anything the checker mistakes for a misspelling gets
|
|
59
|
+
# renamed silently, including identifiers and domain vocabulary. A
|
|
60
|
+
# correction needs a human to approve it, so --write-changes is dropped.
|
|
61
|
+
- id: typos # spell-check code, comments, and docs
|
|
62
|
+
args: [--force-exclude]
|
|
63
|
+
|
|
64
|
+
- repo: https://github.com/zizmorcore/zizmor-pre-commit
|
|
65
|
+
rev: e3eebf65325ccc992422292cb7a4baee967cf815 # v1.26.1
|
|
66
|
+
hooks:
|
|
67
|
+
# Several of zizmor's audits (artipacked, impostor-commit,
|
|
68
|
+
# known-vulnerable-actions, stale-action-refs, ref-version-mismatch, ...)
|
|
69
|
+
# need to query the GitHub API for upstream Actions metadata, which isn't
|
|
70
|
+
# reliably available in every environment this runs in (sandboxes,
|
|
71
|
+
# restricted networks). Split into two hooks so local runs always get
|
|
72
|
+
# zizmor's offline coverage (most of its ruleset) without depending on
|
|
73
|
+
# network access, while CI - which has real network + a real GH_TOKEN -
|
|
74
|
+
# additionally runs the online-only audits via `--hook-stage manual`.
|
|
75
|
+
# `--persona=auditor` is zizmor's most thorough mode; findings at any
|
|
76
|
+
# severity fail the run.
|
|
77
|
+
- id: zizmor
|
|
78
|
+
name: zizmor (offline checks)
|
|
79
|
+
args: [--no-online-audits, --persona=auditor]
|
|
80
|
+
- id: zizmor
|
|
81
|
+
name: zizmor (online checks - GitHub API, CI only)
|
|
82
|
+
args: [--persona=auditor]
|
|
83
|
+
stages: [manual]
|
|
84
|
+
|
|
85
|
+
- repo: https://github.com/executablebooks/mdformat
|
|
86
|
+
rev: 82912cdaea4fb830f751504486a7879c70526547 # 1.0.0
|
|
87
|
+
hooks:
|
|
88
|
+
- id: mdformat
|
|
89
|
+
# Formatting config (wrap, number, ...) lives in .mdformat.toml.
|
|
90
|
+
additional_dependencies:
|
|
91
|
+
- mdformat-gfm # padded/aligned tables + GFM extensions (task lists, strikethrough)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.11
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.1.0] - 2026-09-16
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- Initial extraction of the `autolint` checks from [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals) `tools/run_autolint/` as an installable package with an `inspect-evals-lint` CLI.
|
|
14
|
+
- `[tool.inspect-evals-lint]` configuration with `monorepo` and `template` layout presets covering source root, tests root, import prefix, registry mode (`module`, `entry-points`, `none`), non-eval directories, required `eval.yaml` fields, isolated package directory, disabled checks and the sandbox image allowlist.
|
|
15
|
+
- Repository root discovery from the nearest configured `pyproject.toml`, with `--root` and `--preset` overrides.
|
|
16
|
+
- Malformed `eval.yaml`, unparsable main files and unreadable test files now produce `fail` results instead of crashing the run.
|
|
17
|
+
- `register` layout preset for single-evaluation upstream repositories, built from three new configuration keys: `tests-layout = "flat"` accepts test files directly under the tests root when `<tests-root>/<eval>/` is absent (and skips `tests_init` there), `readme-location = "repo-root"` accepts the repository's top-level `README.md`, and `eval-yaml-required = false` turns a missing `eval.yaml` into a skip while still validating one that is present.
|
|
18
|
+
- `main_file` and `init_exports` accept `tasks.py` as the module holding the `@task` functions, as an alternative to `<eval_name>.py`.
|
|
19
|
+
- `--json` writes results as a single JSON document to stdout (progress goes to stderr), with paths relative to the repository root, for badges and other tooling. Each result carries its `category` (the CHECKS.md section), also available as `runner.CHECK_CATEGORIES`. `render_json`, `report_to_dict` and `reports_to_dict` expose the same in the Python API.
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- `external_dependencies` compares distribution names in PEP 503 normalised form, so `import inspect_ai` is satisfied by `dependencies = ["inspect-ai"]`. Before, the check failed on that spelling whenever the linter ran in an environment without the package installed. Requirement strings using `!=`, `~=`, `@` or `,` are also parsed correctly.
|
|
24
|
+
- The notice about a missing `[tool.inspect-evals-lint]` table no longer has its brackets swallowed as console markup.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Matt Fisher
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: inspect-evals-lint
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Static checks for Inspect AI evaluations: structure, tests, best practices and sandbox pinning
|
|
5
|
+
Project-URL: Homepage, https://github.com/Generality-Labs/inspect-evals-lint
|
|
6
|
+
Project-URL: Repository, https://github.com/Generality-Labs/inspect-evals-lint
|
|
7
|
+
Project-URL: Issues, https://github.com/Generality-Labs/inspect-evals-lint/issues
|
|
8
|
+
Author-email: Matt Fisher <m@ttfisher.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Python: >=3.11
|
|
12
|
+
Requires-Dist: pyyaml>=6.0
|
|
13
|
+
Requires-Dist: rich>=13.0
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# inspect-evals-lint
|
|
17
|
+
|
|
18
|
+
Static checks for [Inspect AI](https://inspect.aisi.org.uk/) evaluations: file structure, test coverage conventions, best practices and sandbox image pinning.
|
|
19
|
+
|
|
20
|
+
These checks began life as the `autolint` tool inside [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). They are packaged here so any repository of Inspect evaluations can run the same checks, including standalone repos built from the [inspect-evals-template](https://github.com/Generality-Labs/inspect-evals-template) and submitted to the inspect_evals register.
|
|
21
|
+
|
|
22
|
+
Nothing is imported or executed from the evaluation being checked. Every check is static analysis over Python source (via `ast`), `eval.yaml`, compose files and `pyproject.toml`.
|
|
23
|
+
|
|
24
|
+
## Install
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
uv add --dev inspect-evals-lint
|
|
28
|
+
# or
|
|
29
|
+
pip install inspect-evals-lint
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Usage
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
inspect-evals-lint <eval_name> # one evaluation
|
|
36
|
+
inspect-evals-lint --all-evals # every evaluation in the repo
|
|
37
|
+
inspect-evals-lint --all-evals --summary-only
|
|
38
|
+
inspect-evals-lint --check-summary # per-check compliance across evals
|
|
39
|
+
inspect-evals-lint <eval_name> --check registry
|
|
40
|
+
inspect-evals-lint --all-evals --json > lint.json
|
|
41
|
+
inspect-evals-lint --list-checks
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
`--json` writes one document to stdout and sends progress to stderr, so the output can be piped straight into other tooling. It carries `passed`, run-wide `summary` counts and, per evaluation, every check's `status`, `category` (the [CHECKS.md](docs/CHECKS.md) section: `file_structure`, `code_quality`, `tests` or `best_practices`), `message`, `file` (relative to the repository root when possible) and `line`.
|
|
45
|
+
|
|
46
|
+
The repository root is the nearest `pyproject.toml` carrying a `[tool.inspect-evals-lint]` table (falling back to the nearest `pyproject.toml`, then the current directory). Pass `--root` to override.
|
|
47
|
+
|
|
48
|
+
Exit codes: `0` all checks passed (warnings, skips and suppressions count as passing), `1` at least one check failed, `2` usage or configuration error.
|
|
49
|
+
|
|
50
|
+
## Configuration
|
|
51
|
+
|
|
52
|
+
Configuration lives in `pyproject.toml`. Pick a layout preset and override any field:
|
|
53
|
+
|
|
54
|
+
```toml
|
|
55
|
+
[tool.inspect-evals-lint]
|
|
56
|
+
preset = "template" # or "monorepo" / "register"
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
| Key | `template` preset | `monorepo` preset | `register` preset | Meaning |
|
|
60
|
+
| --------------------------- | ------------------------------------------------ | -------------------------------- | ----------------- | -------------------------------------------------------------------------------------------------------- |
|
|
61
|
+
| `source-root` | `src` | `src/inspect_evals` | `src` | Directory with one sub-directory per evaluation |
|
|
62
|
+
| `tests-root` | `tests` | `tests` | `tests` | Directory holding `<tests-root>/<eval>/` |
|
|
63
|
+
| `tests-layout` | `per-eval` | `per-eval` | `flat` | `flat` also accepts test files directly under `tests-root` when `<tests-root>/<eval>/` is absent |
|
|
64
|
+
| `readme-location` | `eval-dir` | `eval-dir` | `repo-root` | `repo-root` also accepts the repository's top-level `README.md` |
|
|
65
|
+
| `eval-yaml-required` | `true` | `true` | `false` | Whether a missing `eval.yaml` fails (a present one is always validated) |
|
|
66
|
+
| `import-prefix` | `""` | `inspect_evals` | `""` | Dotted prefix evaluations import under |
|
|
67
|
+
| `registry` | `entry-points` | `module` | `entry-points` | `entry-points` reads `[project.entry-points.inspect_ai]`; `module` greps a registry module; `none` skips |
|
|
68
|
+
| `registry-module` | unset | `src/inspect_evals/_registry.py` | unset | Required when `registry = "module"` |
|
|
69
|
+
| `non-eval-dirs` | `["utils", "examples"]` | `["utils"]` | same as template | Sub-directories of `source-root` that are not evaluations |
|
|
70
|
+
| `eval-yaml-required-fields` | `title, description, group, contributors, tasks` | same | same | Keys every `eval.yaml` must define |
|
|
71
|
+
| `isolated-packages-dir` | unset | `packages` | unset | Per-eval `pyproject.toml` directory for isolated dependency sets |
|
|
72
|
+
| `disabled-checks` | `[]` | `[]` | `[]` | Checks that never run |
|
|
73
|
+
| `sandbox-image-allowlist` | `{}` | `{}` | `{}` | `{ eval = ["image/ref"] }` pairs allowed to stay unpinned (warn, not fail) |
|
|
74
|
+
|
|
75
|
+
Without a `[tool.inspect-evals-lint]` table the `template` preset is used. `--preset` overrides the table for one run.
|
|
76
|
+
|
|
77
|
+
The `register` preset is for an upstream repository listed in the [inspect_evals register](https://github.com/UKGovernmentBEIS/inspect_evals/blob/main/register/README.md): one evaluation, tests directly under `tests/`, the README at the repository root, and metadata held by the register entry rather than an `eval.yaml` in the repo. Run it from outside the repo with `inspect-evals-lint --root <clone> --preset register --all-evals --json`.
|
|
78
|
+
|
|
79
|
+
## Suppressing a check
|
|
80
|
+
|
|
81
|
+
- Line: `# noautolint: <check_name>` on the offending line (checks that report per-site results: `private_api_imports`, `get_model_location`).
|
|
82
|
+
- File: `# noautolint-file: <check_name>` within the first ten lines of a file.
|
|
83
|
+
- Directory: a `.noautolint` file in a sub-directory listing check names, one per line. Files under that directory are also excluded from AST-based checks.
|
|
84
|
+
- Evaluation: a `.noautolint` file in the evaluation directory listing check names.
|
|
85
|
+
|
|
86
|
+
## Checks
|
|
87
|
+
|
|
88
|
+
See [docs/CHECKS.md](docs/CHECKS.md) for the full list with the reasoning behind each check.
|
|
89
|
+
|
|
90
|
+
## Python API
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
from pathlib import Path
|
|
94
|
+
from inspect_evals_lint import lint_evaluation, load_config, get_all_eval_names
|
|
95
|
+
|
|
96
|
+
root = Path(".")
|
|
97
|
+
config = load_config(root)
|
|
98
|
+
for name in get_all_eval_names(root, config):
|
|
99
|
+
report = lint_evaluation(root, name, config)
|
|
100
|
+
print(name, report.passed(), report.summary())
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
## Development
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
uv sync
|
|
107
|
+
uv run pre-commit install # optional: run the lint stack on every commit
|
|
108
|
+
uv run pytest
|
|
109
|
+
uv run basedpyright src
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
|
|
113
|
+
|
|
114
|
+
## Releasing
|
|
115
|
+
|
|
116
|
+
See [RELEASING.md](RELEASING.md).
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# inspect-evals-lint
|
|
2
|
+
|
|
3
|
+
Static checks for [Inspect AI](https://inspect.aisi.org.uk/) evaluations: file structure, test coverage conventions, best practices and sandbox image pinning.
|
|
4
|
+
|
|
5
|
+
These checks began life as the `autolint` tool inside [inspect_evals](https://github.com/UKGovernmentBEIS/inspect_evals). They are packaged here so any repository of Inspect evaluations can run the same checks, including standalone repos built from the [inspect-evals-template](https://github.com/Generality-Labs/inspect-evals-template) and submitted to the inspect_evals register.
|
|
6
|
+
|
|
7
|
+
Nothing is imported or executed from the evaluation being checked. Every check is static analysis over Python source (via `ast`), `eval.yaml`, compose files and `pyproject.toml`.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
uv add --dev inspect-evals-lint
|
|
13
|
+
# or
|
|
14
|
+
pip install inspect-evals-lint
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Usage
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
inspect-evals-lint <eval_name> # one evaluation
|
|
21
|
+
inspect-evals-lint --all-evals # every evaluation in the repo
|
|
22
|
+
inspect-evals-lint --all-evals --summary-only
|
|
23
|
+
inspect-evals-lint --check-summary # per-check compliance across evals
|
|
24
|
+
inspect-evals-lint <eval_name> --check registry
|
|
25
|
+
inspect-evals-lint --all-evals --json > lint.json
|
|
26
|
+
inspect-evals-lint --list-checks
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
`--json` writes one document to stdout and sends progress to stderr, so the output can be piped straight into other tooling. It carries `passed`, run-wide `summary` counts and, per evaluation, every check's `status`, `category` (the [CHECKS.md](docs/CHECKS.md) section: `file_structure`, `code_quality`, `tests` or `best_practices`), `message`, `file` (relative to the repository root when possible) and `line`.
|
|
30
|
+
|
|
31
|
+
The repository root is the nearest `pyproject.toml` carrying a `[tool.inspect-evals-lint]` table (falling back to the nearest `pyproject.toml`, then the current directory). Pass `--root` to override.
|
|
32
|
+
|
|
33
|
+
Exit codes: `0` all checks passed (warnings, skips and suppressions count as passing), `1` at least one check failed, `2` usage or configuration error.
|
|
34
|
+
|
|
35
|
+
## Configuration
|
|
36
|
+
|
|
37
|
+
Configuration lives in `pyproject.toml`. Pick a layout preset and override any field:
|
|
38
|
+
|
|
39
|
+
```toml
|
|
40
|
+
[tool.inspect-evals-lint]
|
|
41
|
+
preset = "template" # or "monorepo" / "register"
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
| Key | `template` preset | `monorepo` preset | `register` preset | Meaning |
|
|
45
|
+
| --------------------------- | ------------------------------------------------ | -------------------------------- | ----------------- | -------------------------------------------------------------------------------------------------------- |
|
|
46
|
+
| `source-root` | `src` | `src/inspect_evals` | `src` | Directory with one sub-directory per evaluation |
|
|
47
|
+
| `tests-root` | `tests` | `tests` | `tests` | Directory holding `<tests-root>/<eval>/` |
|
|
48
|
+
| `tests-layout` | `per-eval` | `per-eval` | `flat` | `flat` also accepts test files directly under `tests-root` when `<tests-root>/<eval>/` is absent |
|
|
49
|
+
| `readme-location` | `eval-dir` | `eval-dir` | `repo-root` | `repo-root` also accepts the repository's top-level `README.md` |
|
|
50
|
+
| `eval-yaml-required` | `true` | `true` | `false` | Whether a missing `eval.yaml` fails (a present one is always validated) |
|
|
51
|
+
| `import-prefix` | `""` | `inspect_evals` | `""` | Dotted prefix evaluations import under |
|
|
52
|
+
| `registry` | `entry-points` | `module` | `entry-points` | `entry-points` reads `[project.entry-points.inspect_ai]`; `module` greps a registry module; `none` skips |
|
|
53
|
+
| `registry-module` | unset | `src/inspect_evals/_registry.py` | unset | Required when `registry = "module"` |
|
|
54
|
+
| `non-eval-dirs` | `["utils", "examples"]` | `["utils"]` | same as template | Sub-directories of `source-root` that are not evaluations |
|
|
55
|
+
| `eval-yaml-required-fields` | `title, description, group, contributors, tasks` | same | same | Keys every `eval.yaml` must define |
|
|
56
|
+
| `isolated-packages-dir` | unset | `packages` | unset | Per-eval `pyproject.toml` directory for isolated dependency sets |
|
|
57
|
+
| `disabled-checks` | `[]` | `[]` | `[]` | Checks that never run |
|
|
58
|
+
| `sandbox-image-allowlist` | `{}` | `{}` | `{}` | `{ eval = ["image/ref"] }` pairs allowed to stay unpinned (warn, not fail) |
|
|
59
|
+
|
|
60
|
+
Without a `[tool.inspect-evals-lint]` table the `template` preset is used. `--preset` overrides the table for one run.
|
|
61
|
+
|
|
62
|
+
The `register` preset is for an upstream repository listed in the [inspect_evals register](https://github.com/UKGovernmentBEIS/inspect_evals/blob/main/register/README.md): one evaluation, tests directly under `tests/`, the README at the repository root, and metadata held by the register entry rather than an `eval.yaml` in the repo. Run it from outside the repo with `inspect-evals-lint --root <clone> --preset register --all-evals --json`.
|
|
63
|
+
|
|
64
|
+
## Suppressing a check
|
|
65
|
+
|
|
66
|
+
- Line: `# noautolint: <check_name>` on the offending line (checks that report per-site results: `private_api_imports`, `get_model_location`).
|
|
67
|
+
- File: `# noautolint-file: <check_name>` within the first ten lines of a file.
|
|
68
|
+
- Directory: a `.noautolint` file in a sub-directory listing check names, one per line. Files under that directory are also excluded from AST-based checks.
|
|
69
|
+
- Evaluation: a `.noautolint` file in the evaluation directory listing check names.
|
|
70
|
+
|
|
71
|
+
## Checks
|
|
72
|
+
|
|
73
|
+
See [docs/CHECKS.md](docs/CHECKS.md) for the full list with the reasoning behind each check.
|
|
74
|
+
|
|
75
|
+
## Python API
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from pathlib import Path
|
|
79
|
+
from inspect_evals_lint import lint_evaluation, load_config, get_all_eval_names
|
|
80
|
+
|
|
81
|
+
root = Path(".")
|
|
82
|
+
config = load_config(root)
|
|
83
|
+
for name in get_all_eval_names(root, config):
|
|
84
|
+
report = lint_evaluation(root, name, config)
|
|
85
|
+
print(name, report.passed(), report.summary())
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Development
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
uv sync
|
|
92
|
+
uv run pre-commit install # optional: run the lint stack on every commit
|
|
93
|
+
uv run pytest
|
|
94
|
+
uv run basedpyright src
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Linting (ruff, [zizmor](https://docs.zizmor.sh/), mdformat) runs via [pre-commit](https://pre-commit.com); CI runs the same stack plus basedpyright and pytest via the shared [`python-ci`](https://github.com/Generality-Labs/python-project-template) reusable workflow.
|
|
98
|
+
|
|
99
|
+
## Releasing
|
|
100
|
+
|
|
101
|
+
See [RELEASING.md](RELEASING.md).
|