skilltest-pytest 0.3.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/PKG-INFO +33 -5
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/README.md +31 -3
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/pyproject.toml +2 -2
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/skilltest_pytest/__init__.py +36 -0
- skilltest_pytest-0.5.0/tests/collected/deploy.skilltest.yaml +33 -0
- skilltest_pytest-0.5.0/tests/collected/deployer/SKILL.md +13 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/tests/test_plugin.py +31 -1
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/uv.lock +2 -2
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/.gitignore +0 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/project.json +0 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/skilltest_pytest/plugin.py +0 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/tests/collected/greet.skilltest.yaml +0 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/tests/collected/greeter/SKILL.md +0 -0
- {skilltest_pytest-0.3.0 → skilltest_pytest-0.5.0}/tests/conftest.py +0 -0
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: skilltest-pytest
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
|
|
5
5
|
Author: Nick DeRobertis
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Requires-Python: >=3.12
|
|
8
8
|
Requires-Dist: pytest>=8
|
|
9
|
-
Requires-Dist: skilltest-sdk==0.
|
|
9
|
+
Requires-Dist: skilltest-sdk==0.5.0
|
|
10
10
|
Description-Content-Type: text/markdown
|
|
11
11
|
|
|
12
12
|
# skilltest-pytest
|
|
@@ -34,12 +34,40 @@ evals:
|
|
|
34
34
|
**As code**, for matrices and deterministic mix-ins:
|
|
35
35
|
|
|
36
36
|
```python
|
|
37
|
-
from skilltest_pytest import run_skill
|
|
37
|
+
from skilltest_pytest import run_skill, describe_failures, assistant_text
|
|
38
38
|
|
|
39
39
|
def test_greeter():
|
|
40
40
|
report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
41
|
-
assert report.passed,
|
|
42
|
-
assert "Dr. Smith" in report.runs[0].transcript
|
|
41
|
+
assert report.passed, describe_failures(report)
|
|
42
|
+
assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Assert on tool use, and stream
|
|
46
|
+
|
|
47
|
+
The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
|
|
48
|
+
returns the normalized `tool_call` events a run took (each a `ToolEvent` with
|
|
49
|
+
`kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
|
|
50
|
+
test can **short-circuit** on bad behavior:
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from skilltest_pytest import run_skill, tool_calls
|
|
54
|
+
|
|
55
|
+
def test_commits_but_never_deletes():
|
|
56
|
+
report = run_skill("cases/edit.skilltest.yaml")
|
|
57
|
+
calls = tool_calls(report.runs[0].transcript)
|
|
58
|
+
assert any("git commit" in str(c.input) for c in calls)
|
|
59
|
+
assert not any("rm -rf" in str(c.input) for c in calls)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
import asyncio
|
|
64
|
+
from skilltest_pytest import stream_skill
|
|
65
|
+
|
|
66
|
+
def test_makes_no_network_call():
|
|
67
|
+
async def go():
|
|
68
|
+
async for ev in stream_skill("cases/edit.skilltest.yaml"):
|
|
69
|
+
assert ev.event.name != "curl", "skill made a network call"
|
|
70
|
+
asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
|
|
43
71
|
```
|
|
44
72
|
|
|
45
73
|
## Configuration
|
|
@@ -23,12 +23,40 @@ evals:
|
|
|
23
23
|
**As code**, for matrices and deterministic mix-ins:
|
|
24
24
|
|
|
25
25
|
```python
|
|
26
|
-
from skilltest_pytest import run_skill
|
|
26
|
+
from skilltest_pytest import run_skill, describe_failures, assistant_text
|
|
27
27
|
|
|
28
28
|
def test_greeter():
|
|
29
29
|
report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
30
|
-
assert report.passed,
|
|
31
|
-
assert "Dr. Smith" in report.runs[0].transcript
|
|
30
|
+
assert report.passed, describe_failures(report)
|
|
31
|
+
assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Assert on tool use, and stream
|
|
35
|
+
|
|
36
|
+
The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
|
|
37
|
+
returns the normalized `tool_call` events a run took (each a `ToolEvent` with
|
|
38
|
+
`kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
|
|
39
|
+
test can **short-circuit** on bad behavior:
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from skilltest_pytest import run_skill, tool_calls
|
|
43
|
+
|
|
44
|
+
def test_commits_but_never_deletes():
|
|
45
|
+
report = run_skill("cases/edit.skilltest.yaml")
|
|
46
|
+
calls = tool_calls(report.runs[0].transcript)
|
|
47
|
+
assert any("git commit" in str(c.input) for c in calls)
|
|
48
|
+
assert not any("rm -rf" in str(c.input) for c in calls)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import asyncio
|
|
53
|
+
from skilltest_pytest import stream_skill
|
|
54
|
+
|
|
55
|
+
def test_makes_no_network_call():
|
|
56
|
+
async def go():
|
|
57
|
+
async for ev in stream_skill("cases/edit.skilltest.yaml"):
|
|
58
|
+
assert ev.event.name != "curl", "skill made a network call"
|
|
59
|
+
asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
|
|
32
60
|
```
|
|
33
61
|
|
|
34
62
|
## Configuration
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "skilltest-pytest"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.5.0"
|
|
4
4
|
description = "pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12"
|
|
7
7
|
license = "MIT"
|
|
8
8
|
authors = [{ name = "Nick DeRobertis" }]
|
|
9
9
|
dependencies = [
|
|
10
|
-
"skilltest-sdk==0.
|
|
10
|
+
"skilltest-sdk==0.5.0",
|
|
11
11
|
"pytest>=8",
|
|
12
12
|
]
|
|
13
13
|
|
|
@@ -20,24 +20,42 @@ from skilltest_sdk import (
|
|
|
20
20
|
ENV_BIN,
|
|
21
21
|
ENV_PROVIDER,
|
|
22
22
|
BooleanDetail,
|
|
23
|
+
CallsDetail,
|
|
23
24
|
CaseRun,
|
|
24
25
|
EvalOutcome,
|
|
26
|
+
Matcher,
|
|
25
27
|
Message,
|
|
28
|
+
MockCall,
|
|
26
29
|
NumericDetail,
|
|
27
30
|
Report,
|
|
31
|
+
SkillStream,
|
|
28
32
|
SkilltestError,
|
|
29
33
|
SkilltestProviderError,
|
|
30
34
|
SkilltestUsageError,
|
|
35
|
+
StreamEvent,
|
|
31
36
|
Summary,
|
|
37
|
+
ToolCall,
|
|
38
|
+
ToolEvent,
|
|
39
|
+
ToolMock,
|
|
40
|
+
ToolSpy,
|
|
32
41
|
Transcript,
|
|
33
42
|
Usage,
|
|
34
43
|
ValidationFinding,
|
|
35
44
|
ValidationReport,
|
|
45
|
+
anything,
|
|
36
46
|
assistant_text,
|
|
47
|
+
contains,
|
|
48
|
+
deny,
|
|
37
49
|
describe_failures,
|
|
38
50
|
failed_evals,
|
|
39
51
|
failed_runs,
|
|
52
|
+
matching,
|
|
53
|
+
rewrite,
|
|
40
54
|
run_skill,
|
|
55
|
+
spy,
|
|
56
|
+
stream_skill,
|
|
57
|
+
stub,
|
|
58
|
+
tool_calls,
|
|
41
59
|
validate_skill,
|
|
42
60
|
)
|
|
43
61
|
|
|
@@ -47,24 +65,42 @@ __all__ = [
|
|
|
47
65
|
"ENV_BIN",
|
|
48
66
|
"ENV_PROVIDER",
|
|
49
67
|
"BooleanDetail",
|
|
68
|
+
"CallsDetail",
|
|
50
69
|
"CaseRun",
|
|
51
70
|
"EvalOutcome",
|
|
71
|
+
"Matcher",
|
|
52
72
|
"Message",
|
|
73
|
+
"MockCall",
|
|
53
74
|
"NumericDetail",
|
|
54
75
|
"Report",
|
|
76
|
+
"SkillStream",
|
|
55
77
|
"SkilltestError",
|
|
56
78
|
"SkilltestFailure",
|
|
57
79
|
"SkilltestProviderError",
|
|
58
80
|
"SkilltestUsageError",
|
|
81
|
+
"StreamEvent",
|
|
59
82
|
"Summary",
|
|
83
|
+
"ToolCall",
|
|
84
|
+
"ToolEvent",
|
|
85
|
+
"ToolMock",
|
|
86
|
+
"ToolSpy",
|
|
60
87
|
"Transcript",
|
|
61
88
|
"Usage",
|
|
62
89
|
"ValidationFinding",
|
|
63
90
|
"ValidationReport",
|
|
91
|
+
"anything",
|
|
64
92
|
"assistant_text",
|
|
93
|
+
"contains",
|
|
94
|
+
"deny",
|
|
65
95
|
"describe_failures",
|
|
66
96
|
"failed_evals",
|
|
67
97
|
"failed_runs",
|
|
98
|
+
"matching",
|
|
99
|
+
"rewrite",
|
|
68
100
|
"run_skill",
|
|
101
|
+
"spy",
|
|
102
|
+
"stream_skill",
|
|
103
|
+
"stub",
|
|
104
|
+
"tool_calls",
|
|
69
105
|
"validate_skill",
|
|
70
106
|
]
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Auto-collected mocked case: the declarative `mocks:` block plus deterministic
|
|
2
|
+
# `called`/`not_called` evals — no plugin code involved, proving the whole
|
|
3
|
+
# mock/spy surface rides YAML collection.
|
|
4
|
+
name: collected_deploy
|
|
5
|
+
skill: ./deployer
|
|
6
|
+
input: "Deploy the app"
|
|
7
|
+
mocks:
|
|
8
|
+
- name: push
|
|
9
|
+
match: { tool: bash, pattern: "git push( --force)?\\b" }
|
|
10
|
+
stub: Everything up-to-date
|
|
11
|
+
- name: danger
|
|
12
|
+
match: { contains: "rm -rf" }
|
|
13
|
+
deny: destructive commands are blocked
|
|
14
|
+
- name: git
|
|
15
|
+
match: { tool: bash, pattern: "\\bgit\\b" }
|
|
16
|
+
evals:
|
|
17
|
+
- type: boolean
|
|
18
|
+
name: canned-output-surfaced
|
|
19
|
+
criterion: "the reply mentions `Everything up-to-date`"
|
|
20
|
+
- type: called
|
|
21
|
+
mock: push
|
|
22
|
+
times: 1
|
|
23
|
+
- type: called
|
|
24
|
+
name: both-git-calls-observed
|
|
25
|
+
mock: git
|
|
26
|
+
times: 2
|
|
27
|
+
- type: called
|
|
28
|
+
name: the-deny-intercepted
|
|
29
|
+
mock: danger
|
|
30
|
+
times: 1
|
|
31
|
+
- type: not_called
|
|
32
|
+
mock: push
|
|
33
|
+
where: { command: { contains: "--force" } }
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: deployer
|
|
3
|
+
description: Deploys the app by pushing to git and reporting the outcome.
|
|
4
|
+
---
|
|
5
|
+
# Deployer
|
|
6
|
+
|
|
7
|
+
Deploy the current app: push the branch, check the working tree, clean the
|
|
8
|
+
build directory, and report the outcome in one sentence.
|
|
9
|
+
|
|
10
|
+
<!-- fake-reply: Deployment finished. -->
|
|
11
|
+
<!-- fake-tool: bash git push origin main -->
|
|
12
|
+
<!-- fake-tool: bash git status -->
|
|
13
|
+
<!-- fake-tool: bash rm -rf /tmp/build -->
|
|
@@ -8,11 +8,12 @@ whose eval fails — can be asserted on too.
|
|
|
8
8
|
|
|
9
9
|
from __future__ import annotations
|
|
10
10
|
|
|
11
|
+
import asyncio
|
|
11
12
|
from pathlib import Path
|
|
12
13
|
|
|
13
14
|
import pytest
|
|
14
15
|
|
|
15
|
-
from skilltest_pytest import describe_failures, run_skill
|
|
16
|
+
from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
|
|
16
17
|
|
|
17
18
|
SKILL_MD = """\
|
|
18
19
|
---
|
|
@@ -40,6 +41,20 @@ def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
|
|
|
40
41
|
assert report.passed, describe_failures(report)
|
|
41
42
|
|
|
42
43
|
|
|
44
|
+
def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
|
|
45
|
+
# The tool-event and streaming surfaces are re-exported from the plugin too.
|
|
46
|
+
report = run_skill(cases / "tool_events.yaml")
|
|
47
|
+
assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
|
|
48
|
+
|
|
49
|
+
async def go() -> list[str | None]:
|
|
50
|
+
names: list[str | None] = []
|
|
51
|
+
async for ev in stream_skill(cases / "tool_events.yaml"):
|
|
52
|
+
names.append(ev.event.name)
|
|
53
|
+
return names
|
|
54
|
+
|
|
55
|
+
assert asyncio.run(go()) == ["edit_file", "bash"]
|
|
56
|
+
|
|
57
|
+
|
|
43
58
|
def test_collected_case_passes(pytester: pytest.Pytester) -> None:
|
|
44
59
|
write_skill(pytester.path)
|
|
45
60
|
pytester.makefile(
|
|
@@ -75,3 +90,18 @@ def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester)
|
|
|
75
90
|
result = pytester.runpytest_subprocess()
|
|
76
91
|
result.assert_outcomes(failed=1)
|
|
77
92
|
result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_mock_api_is_reexported_and_binds(cases: Path) -> None:
|
|
96
|
+
# The mock/spy API rides the one-dependency re-export; code-level mocks
|
|
97
|
+
# intercept and bind through the plugin's SDK exactly as through the SDK.
|
|
98
|
+
from skilltest_pytest import contains, matching, run_skill, spy, stub
|
|
99
|
+
|
|
100
|
+
push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date")
|
|
101
|
+
git = spy(tool="bash", pattern=r"\bgit\b")
|
|
102
|
+
report = run_skill(cases / "deploy_plain.yaml", mocks=[push, git])
|
|
103
|
+
assert report.passed
|
|
104
|
+
push.assert_called_once()
|
|
105
|
+
assert push.calls[0].command == "git push origin main"
|
|
106
|
+
git.assert_called_with(command=contains("git status"))
|
|
107
|
+
git.where(command=matching(r"\bsudo\b")).assert_not_called()
|
|
@@ -189,7 +189,7 @@ wheels = [
|
|
|
189
189
|
|
|
190
190
|
[[package]]
|
|
191
191
|
name = "skilltest-pytest"
|
|
192
|
-
version = "0.
|
|
192
|
+
version = "0.5.0"
|
|
193
193
|
source = { editable = "." }
|
|
194
194
|
dependencies = [
|
|
195
195
|
{ name = "pytest" },
|
|
@@ -216,7 +216,7 @@ dev = [
|
|
|
216
216
|
|
|
217
217
|
[[package]]
|
|
218
218
|
name = "skilltest-sdk"
|
|
219
|
-
version = "0.
|
|
219
|
+
version = "0.5.0"
|
|
220
220
|
source = { editable = "../../sdks/python" }
|
|
221
221
|
dependencies = [
|
|
222
222
|
{ name = "pydantic" },
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|