skilltest-pytest 0.4.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- skilltest_pytest-0.6.0/PKG-INFO +111 -0
- skilltest_pytest-0.6.0/README.md +100 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/pyproject.toml +2 -2
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/skilltest_pytest/__init__.py +56 -8
- skilltest_pytest-0.6.0/tests/collected/deploy.skilltest.yaml +33 -0
- skilltest_pytest-0.6.0/tests/collected/deployer/SKILL.md +13 -0
- skilltest_pytest-0.6.0/tests/test_plugin.py +171 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/uv.lock +2 -2
- skilltest_pytest-0.4.0/PKG-INFO +0 -58
- skilltest_pytest-0.4.0/README.md +0 -47
- skilltest_pytest-0.4.0/tests/test_plugin.py +0 -92
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/.gitignore +0 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/project.json +0 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/skilltest_pytest/plugin.py +0 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/tests/collected/greet.skilltest.yaml +0 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/tests/collected/greeter/SKILL.md +0 -0
- {skilltest_pytest-0.4.0 → skilltest_pytest-0.6.0}/tests/conftest.py +0 -0
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: skilltest-pytest
|
|
3
|
+
Version: 0.6.0
|
|
4
|
+
Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
|
|
5
|
+
Author: Nick DeRobertis
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Requires-Python: >=3.12
|
|
8
|
+
Requires-Dist: pytest>=8
|
|
9
|
+
Requires-Dist: skilltest-sdk==0.6.0
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# skilltest-pytest
|
|
13
|
+
|
|
14
|
+
A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
|
|
15
|
+
AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
|
|
16
|
+
your own deterministic checks. Built on
|
|
17
|
+
[`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
|
|
18
|
+
re-exported here, so a pytest suite needs only this one dependency.
|
|
19
|
+
|
|
20
|
+
## Define the whole case in code (recommended)
|
|
21
|
+
|
|
22
|
+
Build the case — skill, input, evals, an optional simulated user, mocks — right
|
|
23
|
+
in the test. Everything the YAML carries has a typed builder, so the case, its
|
|
24
|
+
mocks, and any deterministic transcript checks live in one place:
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
from skilltest_pytest import TestCase, run_skill, boolean, numeric, describe_failures
|
|
28
|
+
|
|
29
|
+
def test_greeter():
|
|
30
|
+
case = TestCase(
|
|
31
|
+
skill="skills/greeter", # resolved relative to the working dir
|
|
32
|
+
input="Greet Dr. Smith, who has an appointment today.",
|
|
33
|
+
evals=[
|
|
34
|
+
boolean("the reply greets Dr. Smith by name"),
|
|
35
|
+
numeric("how warm is the tone", min=0, max=10, threshold=7),
|
|
36
|
+
],
|
|
37
|
+
)
|
|
38
|
+
report = run_skill(case, platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
39
|
+
assert report.passed, describe_failures(report)
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Multi-turn cases add `user(...)`; deterministic call-count checks use `called` /
|
|
43
|
+
`not_called` referencing a named `stub`/`spy` (or the mock objects' own
|
|
44
|
+
assertions — see below). `run_skill` also takes `platforms=`/`models=` to fan a
|
|
45
|
+
case across a matrix.
|
|
46
|
+
|
|
47
|
+
## Or point at a YAML file
|
|
48
|
+
|
|
49
|
+
`run_skill` accepts a path just as well (`run_skill("cases/greet.yaml")`), and
|
|
50
|
+
**auto-collection** still works: name a case `something.skilltest.yaml` and
|
|
51
|
+
pytest runs it with no test function at all —
|
|
52
|
+
|
|
53
|
+
```yaml
|
|
54
|
+
# greet.skilltest.yaml
|
|
55
|
+
skill: ./skills/greeter
|
|
56
|
+
input: "Greet Dr. Smith."
|
|
57
|
+
evals:
|
|
58
|
+
- type: boolean
|
|
59
|
+
criterion: "the reply greets Dr. Smith by name"
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
The full field reference for both forms is [`docs/schema.md`](../../docs/schema.md).
|
|
63
|
+
|
|
64
|
+
## Assert on tool use, and stream
|
|
65
|
+
|
|
66
|
+
The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
|
|
67
|
+
returns the normalized `tool_call` events a run took (each a `ToolEvent` with
|
|
68
|
+
`kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
|
|
69
|
+
test can **short-circuit** on bad behavior:
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from skilltest_pytest import TestCase, run_skill, tool_calls, boolean
|
|
73
|
+
|
|
74
|
+
EDIT_CASE = TestCase(
|
|
75
|
+
skill="skills/editor",
|
|
76
|
+
input="Update the config and commit it.",
|
|
77
|
+
evals=[boolean("the change was committed")],
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
def test_commits_but_never_deletes():
|
|
81
|
+
report = run_skill(EDIT_CASE)
|
|
82
|
+
calls = tool_calls(report.runs[0].transcript)
|
|
83
|
+
assert any("git commit" in str(c.input) for c in calls)
|
|
84
|
+
assert not any("rm -rf" in str(c.input) for c in calls)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
import asyncio
|
|
89
|
+
from skilltest_pytest import stream_skill
|
|
90
|
+
|
|
91
|
+
def test_makes_no_network_call():
|
|
92
|
+
async def go():
|
|
93
|
+
async for ev in stream_skill(EDIT_CASE):
|
|
94
|
+
assert ev.event.name != "curl", "skill made a network call"
|
|
95
|
+
asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## Configuration
|
|
99
|
+
|
|
100
|
+
The plugin shells out to the `skilltest` binary. Point it at one with the
|
|
101
|
+
`SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
|
|
102
|
+
`provider=`), and set defaults in `pyproject.toml`:
|
|
103
|
+
|
|
104
|
+
```toml
|
|
105
|
+
[tool.pytest.ini_options]
|
|
106
|
+
skilltest_provider = "oneharness"
|
|
107
|
+
skilltest_platforms = ["claude-code"]
|
|
108
|
+
skilltest_models = ["claude-opus-4-8"]
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
See the repository root for the provider protocol and the full schema.
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# skilltest-pytest
|
|
2
|
+
|
|
3
|
+
A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
|
|
4
|
+
AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
|
|
5
|
+
your own deterministic checks. Built on
|
|
6
|
+
[`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
|
|
7
|
+
re-exported here, so a pytest suite needs only this one dependency.
|
|
8
|
+
|
|
9
|
+
## Define the whole case in code (recommended)
|
|
10
|
+
|
|
11
|
+
Build the case — skill, input, evals, an optional simulated user, mocks — right
|
|
12
|
+
in the test. Everything the YAML carries has a typed builder, so the case, its
|
|
13
|
+
mocks, and any deterministic transcript checks live in one place:
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from skilltest_pytest import TestCase, run_skill, boolean, numeric, describe_failures
|
|
17
|
+
|
|
18
|
+
def test_greeter():
|
|
19
|
+
case = TestCase(
|
|
20
|
+
skill="skills/greeter", # resolved relative to the working dir
|
|
21
|
+
input="Greet Dr. Smith, who has an appointment today.",
|
|
22
|
+
evals=[
|
|
23
|
+
boolean("the reply greets Dr. Smith by name"),
|
|
24
|
+
numeric("how warm is the tone", min=0, max=10, threshold=7),
|
|
25
|
+
],
|
|
26
|
+
)
|
|
27
|
+
report = run_skill(case, platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
28
|
+
assert report.passed, describe_failures(report)
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Multi-turn cases add `user(...)`; deterministic call-count checks use `called` /
|
|
32
|
+
`not_called` referencing a named `stub`/`spy` (or the mock objects' own
|
|
33
|
+
assertions — see below). `run_skill` also takes `platforms=`/`models=` to fan a
|
|
34
|
+
case across a matrix.
|
|
35
|
+
|
|
36
|
+
## Or point at a YAML file
|
|
37
|
+
|
|
38
|
+
`run_skill` accepts a path just as well (`run_skill("cases/greet.yaml")`), and
|
|
39
|
+
**auto-collection** still works: name a case `something.skilltest.yaml` and
|
|
40
|
+
pytest runs it with no test function at all —
|
|
41
|
+
|
|
42
|
+
```yaml
|
|
43
|
+
# greet.skilltest.yaml
|
|
44
|
+
skill: ./skills/greeter
|
|
45
|
+
input: "Greet Dr. Smith."
|
|
46
|
+
evals:
|
|
47
|
+
- type: boolean
|
|
48
|
+
criterion: "the reply greets Dr. Smith by name"
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
The full field reference for both forms is [`docs/schema.md`](../../docs/schema.md).
|
|
52
|
+
|
|
53
|
+
## Assert on tool use, and stream
|
|
54
|
+
|
|
55
|
+
The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
|
|
56
|
+
returns the normalized `tool_call` events a run took (each a `ToolEvent` with
|
|
57
|
+
`kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
|
|
58
|
+
test can **short-circuit** on bad behavior:
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from skilltest_pytest import TestCase, run_skill, tool_calls, boolean
|
|
62
|
+
|
|
63
|
+
EDIT_CASE = TestCase(
|
|
64
|
+
skill="skills/editor",
|
|
65
|
+
input="Update the config and commit it.",
|
|
66
|
+
evals=[boolean("the change was committed")],
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
def test_commits_but_never_deletes():
|
|
70
|
+
report = run_skill(EDIT_CASE)
|
|
71
|
+
calls = tool_calls(report.runs[0].transcript)
|
|
72
|
+
assert any("git commit" in str(c.input) for c in calls)
|
|
73
|
+
assert not any("rm -rf" in str(c.input) for c in calls)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import asyncio
|
|
78
|
+
from skilltest_pytest import stream_skill
|
|
79
|
+
|
|
80
|
+
def test_makes_no_network_call():
|
|
81
|
+
async def go():
|
|
82
|
+
async for ev in stream_skill(EDIT_CASE):
|
|
83
|
+
assert ev.event.name != "curl", "skill made a network call"
|
|
84
|
+
asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Configuration
|
|
88
|
+
|
|
89
|
+
The plugin shells out to the `skilltest` binary. Point it at one with the
|
|
90
|
+
`SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
|
|
91
|
+
`provider=`), and set defaults in `pyproject.toml`:
|
|
92
|
+
|
|
93
|
+
```toml
|
|
94
|
+
[tool.pytest.ini_options]
|
|
95
|
+
skilltest_provider = "oneharness"
|
|
96
|
+
skilltest_platforms = ["claude-code"]
|
|
97
|
+
skilltest_models = ["claude-opus-4-8"]
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
See the repository root for the provider protocol and the full schema.
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "skilltest-pytest"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.6.0"
|
|
4
4
|
description = "pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12"
|
|
7
7
|
license = "MIT"
|
|
8
8
|
authors = [{ name = "Nick DeRobertis" }]
|
|
9
9
|
dependencies = [
|
|
10
|
-
"skilltest-sdk==0.
|
|
10
|
+
"skilltest-sdk==0.6.0",
|
|
11
11
|
"pytest>=8",
|
|
12
12
|
]
|
|
13
13
|
|
|
@@ -1,17 +1,23 @@
|
|
|
1
1
|
"""skilltest-pytest: run AI-skill tests and natural-language evals as pytest.
|
|
2
2
|
|
|
3
|
-
The pytest integration on top of [`skilltest-sdk`][skilltest_sdk]
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
suite only needs one dependency:
|
|
3
|
+
The pytest integration on top of [`skilltest-sdk`][skilltest_sdk], whose API is
|
|
4
|
+
re-exported here so a pytest suite needs only this one dependency. Define the
|
|
5
|
+
whole case in code (the recommended form):
|
|
7
6
|
|
|
8
|
-
from skilltest_pytest import run_skill,
|
|
7
|
+
from skilltest_pytest import TestCase, run_skill, boolean, describe_failures
|
|
9
8
|
|
|
10
9
|
def test_greeter():
|
|
11
|
-
|
|
10
|
+
case = TestCase(
|
|
11
|
+
skill="skills/greeter",
|
|
12
|
+
input="Greet Dr. Smith.",
|
|
13
|
+
evals=[boolean("the reply greets Dr. Smith by name")],
|
|
14
|
+
)
|
|
15
|
+
report = run_skill(case)
|
|
12
16
|
assert report.passed, describe_failures(report)
|
|
13
|
-
|
|
14
|
-
|
|
17
|
+
|
|
18
|
+
Existing YAML cases stay first-class: ``run_skill("cases/greet.yaml")`` runs a
|
|
19
|
+
file, and any ``*.skilltest.yaml`` dropped next to your tests is auto-collected
|
|
20
|
+
as a test item with no code at all.
|
|
15
21
|
"""
|
|
16
22
|
|
|
17
23
|
from __future__ import annotations
|
|
@@ -20,29 +26,50 @@ from skilltest_sdk import (
|
|
|
20
26
|
ENV_BIN,
|
|
21
27
|
ENV_PROVIDER,
|
|
22
28
|
BooleanDetail,
|
|
29
|
+
CallsDetail,
|
|
23
30
|
CaseRun,
|
|
31
|
+
Eval,
|
|
24
32
|
EvalOutcome,
|
|
33
|
+
Matcher,
|
|
25
34
|
Message,
|
|
35
|
+
MockCall,
|
|
26
36
|
NumericDetail,
|
|
27
37
|
Report,
|
|
38
|
+
SimulatedUser,
|
|
28
39
|
SkillStream,
|
|
29
40
|
SkilltestError,
|
|
30
41
|
SkilltestProviderError,
|
|
31
42
|
SkilltestUsageError,
|
|
32
43
|
StreamEvent,
|
|
33
44
|
Summary,
|
|
45
|
+
TestCase,
|
|
46
|
+
ToolCall,
|
|
34
47
|
ToolEvent,
|
|
48
|
+
ToolMock,
|
|
49
|
+
ToolSpy,
|
|
35
50
|
Transcript,
|
|
36
51
|
Usage,
|
|
37
52
|
ValidationFinding,
|
|
38
53
|
ValidationReport,
|
|
54
|
+
anything,
|
|
39
55
|
assistant_text,
|
|
56
|
+
boolean,
|
|
57
|
+
called,
|
|
58
|
+
contains,
|
|
59
|
+
deny,
|
|
40
60
|
describe_failures,
|
|
41
61
|
failed_evals,
|
|
42
62
|
failed_runs,
|
|
63
|
+
matching,
|
|
64
|
+
not_called,
|
|
65
|
+
numeric,
|
|
66
|
+
rewrite,
|
|
43
67
|
run_skill,
|
|
68
|
+
spy,
|
|
44
69
|
stream_skill,
|
|
70
|
+
stub,
|
|
45
71
|
tool_calls,
|
|
72
|
+
user,
|
|
46
73
|
validate_skill,
|
|
47
74
|
)
|
|
48
75
|
|
|
@@ -52,11 +79,16 @@ __all__ = [
|
|
|
52
79
|
"ENV_BIN",
|
|
53
80
|
"ENV_PROVIDER",
|
|
54
81
|
"BooleanDetail",
|
|
82
|
+
"CallsDetail",
|
|
55
83
|
"CaseRun",
|
|
84
|
+
"Eval",
|
|
56
85
|
"EvalOutcome",
|
|
86
|
+
"Matcher",
|
|
57
87
|
"Message",
|
|
88
|
+
"MockCall",
|
|
58
89
|
"NumericDetail",
|
|
59
90
|
"Report",
|
|
91
|
+
"SimulatedUser",
|
|
60
92
|
"SkillStream",
|
|
61
93
|
"SkilltestError",
|
|
62
94
|
"SkilltestFailure",
|
|
@@ -64,17 +96,33 @@ __all__ = [
|
|
|
64
96
|
"SkilltestUsageError",
|
|
65
97
|
"StreamEvent",
|
|
66
98
|
"Summary",
|
|
99
|
+
"TestCase",
|
|
100
|
+
"ToolCall",
|
|
67
101
|
"ToolEvent",
|
|
102
|
+
"ToolMock",
|
|
103
|
+
"ToolSpy",
|
|
68
104
|
"Transcript",
|
|
69
105
|
"Usage",
|
|
70
106
|
"ValidationFinding",
|
|
71
107
|
"ValidationReport",
|
|
108
|
+
"anything",
|
|
72
109
|
"assistant_text",
|
|
110
|
+
"boolean",
|
|
111
|
+
"called",
|
|
112
|
+
"contains",
|
|
113
|
+
"deny",
|
|
73
114
|
"describe_failures",
|
|
74
115
|
"failed_evals",
|
|
75
116
|
"failed_runs",
|
|
117
|
+
"matching",
|
|
118
|
+
"not_called",
|
|
119
|
+
"numeric",
|
|
120
|
+
"rewrite",
|
|
76
121
|
"run_skill",
|
|
122
|
+
"spy",
|
|
77
123
|
"stream_skill",
|
|
124
|
+
"stub",
|
|
78
125
|
"tool_calls",
|
|
126
|
+
"user",
|
|
79
127
|
"validate_skill",
|
|
80
128
|
]
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Auto-collected mocked case: the declarative `mocks:` block plus deterministic
|
|
2
|
+
# `called`/`not_called` evals — no plugin code involved, proving the whole
|
|
3
|
+
# mock/spy surface rides YAML collection.
|
|
4
|
+
name: collected_deploy
|
|
5
|
+
skill: ./deployer
|
|
6
|
+
input: "Deploy the app"
|
|
7
|
+
mocks:
|
|
8
|
+
- name: push
|
|
9
|
+
match: { tool: bash, pattern: "git push( --force)?\\b" }
|
|
10
|
+
stub: Everything up-to-date
|
|
11
|
+
- name: danger
|
|
12
|
+
match: { contains: "rm -rf" }
|
|
13
|
+
deny: destructive commands are blocked
|
|
14
|
+
- name: git
|
|
15
|
+
match: { tool: bash, pattern: "\\bgit\\b" }
|
|
16
|
+
evals:
|
|
17
|
+
- type: boolean
|
|
18
|
+
name: canned-output-surfaced
|
|
19
|
+
criterion: "the reply mentions `Everything up-to-date`"
|
|
20
|
+
- type: called
|
|
21
|
+
mock: push
|
|
22
|
+
times: 1
|
|
23
|
+
- type: called
|
|
24
|
+
name: both-git-calls-observed
|
|
25
|
+
mock: git
|
|
26
|
+
times: 2
|
|
27
|
+
- type: called
|
|
28
|
+
name: the-deny-intercepted
|
|
29
|
+
mock: danger
|
|
30
|
+
times: 1
|
|
31
|
+
- type: not_called
|
|
32
|
+
mock: push
|
|
33
|
+
where: { command: { contains: "--force" } }
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: deployer
|
|
3
|
+
description: Deploys the app by pushing to git and reporting the outcome.
|
|
4
|
+
---
|
|
5
|
+
# Deployer
|
|
6
|
+
|
|
7
|
+
Deploy the current app: push the branch, check the working tree, clean the
|
|
8
|
+
build directory, and report the outcome in one sentence.
|
|
9
|
+
|
|
10
|
+
<!-- fake-reply: Deployment finished. -->
|
|
11
|
+
<!-- fake-tool: bash git push origin main -->
|
|
12
|
+
<!-- fake-tool: bash git status -->
|
|
13
|
+
<!-- fake-tool: bash rm -rf /tmp/build -->
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""E2e tests for the pytest integration.
|
|
2
|
+
|
|
3
|
+
The happy path of auto-collection is also exercised by `collected/`
|
|
4
|
+
(`greet.skilltest.yaml` runs as part of this very suite); the `pytester` tests
|
|
5
|
+
here drive a *child* pytest end-to-end so the failure path — a collected case
|
|
6
|
+
whose eval fails — can be asserted on too.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import asyncio
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import pytest
|
|
15
|
+
|
|
16
|
+
from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
|
|
17
|
+
|
|
18
|
+
SKILL_MD = """\
|
|
19
|
+
---
|
|
20
|
+
name: greeter
|
|
21
|
+
description: A local greeter skill used to exercise pytest auto-collection.
|
|
22
|
+
---
|
|
23
|
+
# Greeter
|
|
24
|
+
|
|
25
|
+
Greet the user by name.
|
|
26
|
+
|
|
27
|
+
<!-- fake-reply: Hello, Dr. Smith! Welcome to the clinic. -->
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def write_skill(root: Path) -> None:
|
|
32
|
+
skill = root / "greeter"
|
|
33
|
+
skill.mkdir()
|
|
34
|
+
(skill / "SKILL.md").write_text(SKILL_MD)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
|
|
38
|
+
# One dependency is enough for a pytest suite: the SDK's code-level API is
|
|
39
|
+
# available straight from skilltest_pytest.
|
|
40
|
+
report = run_skill(cases / "greet_pass.yaml")
|
|
41
|
+
assert report.passed, describe_failures(report)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
|
|
45
|
+
# The tool-event and streaming surfaces are re-exported from the plugin too.
|
|
46
|
+
report = run_skill(cases / "tool_events.yaml")
|
|
47
|
+
assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
|
|
48
|
+
|
|
49
|
+
async def go() -> list[str | None]:
|
|
50
|
+
names: list[str | None] = []
|
|
51
|
+
async for ev in stream_skill(cases / "tool_events.yaml"):
|
|
52
|
+
names.append(ev.event.name)
|
|
53
|
+
return names
|
|
54
|
+
|
|
55
|
+
assert asyncio.run(go()) == ["edit_file", "bash"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_collected_case_passes(pytester: pytest.Pytester) -> None:
|
|
59
|
+
write_skill(pytester.path)
|
|
60
|
+
pytester.makefile(
|
|
61
|
+
".skilltest.yaml",
|
|
62
|
+
greet="""
|
|
63
|
+
name: collected_greet
|
|
64
|
+
skill: ./greeter
|
|
65
|
+
input: "Greet Dr. Smith."
|
|
66
|
+
evals:
|
|
67
|
+
- type: boolean
|
|
68
|
+
name: names-the-patient
|
|
69
|
+
criterion: "the reply greets `Dr. Smith` by name"
|
|
70
|
+
""",
|
|
71
|
+
)
|
|
72
|
+
result = pytester.runpytest_subprocess()
|
|
73
|
+
result.assert_outcomes(passed=1)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester) -> None:
|
|
77
|
+
write_skill(pytester.path)
|
|
78
|
+
pytester.makefile(
|
|
79
|
+
".skilltest.yaml",
|
|
80
|
+
farewell="""
|
|
81
|
+
name: collected_farewell
|
|
82
|
+
skill: ./greeter
|
|
83
|
+
input: "Greet Dr. Smith."
|
|
84
|
+
evals:
|
|
85
|
+
- type: boolean
|
|
86
|
+
name: says-goodbye
|
|
87
|
+
criterion: "the reply contains a `goodbye`"
|
|
88
|
+
""",
|
|
89
|
+
)
|
|
90
|
+
result = pytester.runpytest_subprocess()
|
|
91
|
+
result.assert_outcomes(failed=1)
|
|
92
|
+
result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_code_defined_case_is_reexported_and_runs(fixtures: Path) -> None:
|
|
96
|
+
# The recommended form: build the whole case in code and hand it to
|
|
97
|
+
# run_skill — the case API rides the same one-dependency re-export.
|
|
98
|
+
from skilltest_pytest import TestCase, boolean, run_skill
|
|
99
|
+
|
|
100
|
+
case = TestCase(
|
|
101
|
+
skill=fixtures / "skills" / "greeter",
|
|
102
|
+
input="Greet Dr. Smith, who has an appointment today.",
|
|
103
|
+
evals=[boolean("the reply greets `Dr. Smith` by name")],
|
|
104
|
+
)
|
|
105
|
+
report = run_skill(case)
|
|
106
|
+
assert report.passed, describe_failures(report)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_full_code_defined_case_surface_is_reexported(fixtures: Path) -> None:
|
|
110
|
+
# Every case builder rides the one-dependency re-export: a multi-turn case
|
|
111
|
+
# with judge evals, and a mocked case with a deterministic call eval —
|
|
112
|
+
# streamed and buffered — defined entirely in code through the plugin.
|
|
113
|
+
from skilltest_pytest import (
|
|
114
|
+
TestCase,
|
|
115
|
+
boolean,
|
|
116
|
+
called,
|
|
117
|
+
numeric,
|
|
118
|
+
run_skill,
|
|
119
|
+
stream_skill,
|
|
120
|
+
stub,
|
|
121
|
+
user,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
multi = TestCase(
|
|
125
|
+
skill=fixtures / "skills" / "greeter",
|
|
126
|
+
input="I'd like to confirm my appointment, please.",
|
|
127
|
+
user=user(
|
|
128
|
+
"You are a terse patient.\nsay: Yes, please go ahead.",
|
|
129
|
+
done_when="the conversation has reached turns>=2",
|
|
130
|
+
max_turns=4,
|
|
131
|
+
),
|
|
132
|
+
evals=[
|
|
133
|
+
boolean("the assistant confirmed the appointment (`confirmed`)"),
|
|
134
|
+
numeric("mentions `confirmed`", min=0, max=10, threshold=5, comparator=">"),
|
|
135
|
+
],
|
|
136
|
+
)
|
|
137
|
+
report = run_skill(multi)
|
|
138
|
+
assert report.passed, describe_failures(report)
|
|
139
|
+
assert report.runs[0].turns == 2
|
|
140
|
+
|
|
141
|
+
push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date", name="push")
|
|
142
|
+
mocked = TestCase(
|
|
143
|
+
skill=fixtures / "skills" / "deployer",
|
|
144
|
+
input="Deploy the app",
|
|
145
|
+
mocks=[push],
|
|
146
|
+
evals=[called("push", times=1)],
|
|
147
|
+
)
|
|
148
|
+
stream = stream_skill(mocked)
|
|
149
|
+
|
|
150
|
+
async def drain() -> None:
|
|
151
|
+
async for _ in stream:
|
|
152
|
+
pass
|
|
153
|
+
|
|
154
|
+
asyncio.run(drain())
|
|
155
|
+
assert stream.report is not None and stream.report.passed
|
|
156
|
+
push.assert_called_once()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def test_mock_api_is_reexported_and_binds(cases: Path) -> None:
|
|
160
|
+
# The mock/spy API rides the one-dependency re-export; code-level mocks
|
|
161
|
+
# intercept and bind through the plugin's SDK exactly as through the SDK.
|
|
162
|
+
from skilltest_pytest import contains, matching, run_skill, spy, stub
|
|
163
|
+
|
|
164
|
+
push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date")
|
|
165
|
+
git = spy(tool="bash", pattern=r"\bgit\b")
|
|
166
|
+
report = run_skill(cases / "deploy_plain.yaml", mocks=[push, git])
|
|
167
|
+
assert report.passed
|
|
168
|
+
push.assert_called_once()
|
|
169
|
+
assert push.calls[0].command == "git push origin main"
|
|
170
|
+
git.assert_called_with(command=contains("git status"))
|
|
171
|
+
git.where(command=matching(r"\bsudo\b")).assert_not_called()
|
|
@@ -189,7 +189,7 @@ wheels = [
|
|
|
189
189
|
|
|
190
190
|
[[package]]
|
|
191
191
|
name = "skilltest-pytest"
|
|
192
|
-
version = "0.
|
|
192
|
+
version = "0.6.0"
|
|
193
193
|
source = { editable = "." }
|
|
194
194
|
dependencies = [
|
|
195
195
|
{ name = "pytest" },
|
|
@@ -216,7 +216,7 @@ dev = [
|
|
|
216
216
|
|
|
217
217
|
[[package]]
|
|
218
218
|
name = "skilltest-sdk"
|
|
219
|
-
version = "0.
|
|
219
|
+
version = "0.6.0"
|
|
220
220
|
source = { editable = "../../sdks/python" }
|
|
221
221
|
dependencies = [
|
|
222
222
|
{ name = "pydantic" },
|
skilltest_pytest-0.4.0/PKG-INFO
DELETED
|
@@ -1,58 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: skilltest-pytest
|
|
3
|
-
Version: 0.4.0
|
|
4
|
-
Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
|
|
5
|
-
Author: Nick DeRobertis
|
|
6
|
-
License-Expression: MIT
|
|
7
|
-
Requires-Python: >=3.12
|
|
8
|
-
Requires-Dist: pytest>=8
|
|
9
|
-
Requires-Dist: skilltest-sdk==0.4.0
|
|
10
|
-
Description-Content-Type: text/markdown
|
|
11
|
-
|
|
12
|
-
# skilltest-pytest
|
|
13
|
-
|
|
14
|
-
A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
|
|
15
|
-
AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
|
|
16
|
-
your own deterministic checks. Built on
|
|
17
|
-
[`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
|
|
18
|
-
re-exported here, so a pytest suite needs only this one dependency.
|
|
19
|
-
|
|
20
|
-
## Two ways to use it
|
|
21
|
-
|
|
22
|
-
**Auto-collected case files.** Name a case `something.skilltest.yaml` and pytest
|
|
23
|
-
runs it:
|
|
24
|
-
|
|
25
|
-
```yaml
|
|
26
|
-
# greet.skilltest.yaml
|
|
27
|
-
skill: ./skills/greeter
|
|
28
|
-
input: "Greet Dr. Smith."
|
|
29
|
-
evals:
|
|
30
|
-
- type: boolean
|
|
31
|
-
criterion: "the reply greets Dr. Smith by name"
|
|
32
|
-
```
|
|
33
|
-
|
|
34
|
-
**As code**, for matrices and deterministic mix-ins:
|
|
35
|
-
|
|
36
|
-
```python
|
|
37
|
-
from skilltest_pytest import run_skill
|
|
38
|
-
|
|
39
|
-
def test_greeter():
|
|
40
|
-
report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
41
|
-
assert report.passed, report.describe_failures()
|
|
42
|
-
assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
|
|
43
|
-
```
|
|
44
|
-
|
|
45
|
-
## Configuration
|
|
46
|
-
|
|
47
|
-
The plugin shells out to the `skilltest` binary. Point it at one with the
|
|
48
|
-
`SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
|
|
49
|
-
`provider=`), and set defaults in `pyproject.toml`:
|
|
50
|
-
|
|
51
|
-
```toml
|
|
52
|
-
[tool.pytest.ini_options]
|
|
53
|
-
skilltest_provider = "oneharness"
|
|
54
|
-
skilltest_platforms = ["claude-code"]
|
|
55
|
-
skilltest_models = ["claude-opus-4-8"]
|
|
56
|
-
```
|
|
57
|
-
|
|
58
|
-
See the repository root for the provider protocol and the full schema.
|
skilltest_pytest-0.4.0/README.md
DELETED
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
# skilltest-pytest
|
|
2
|
-
|
|
3
|
-
A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
|
|
4
|
-
AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
|
|
5
|
-
your own deterministic checks. Built on
|
|
6
|
-
[`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
|
|
7
|
-
re-exported here, so a pytest suite needs only this one dependency.
|
|
8
|
-
|
|
9
|
-
## Two ways to use it
|
|
10
|
-
|
|
11
|
-
**Auto-collected case files.** Name a case `something.skilltest.yaml` and pytest
|
|
12
|
-
runs it:
|
|
13
|
-
|
|
14
|
-
```yaml
|
|
15
|
-
# greet.skilltest.yaml
|
|
16
|
-
skill: ./skills/greeter
|
|
17
|
-
input: "Greet Dr. Smith."
|
|
18
|
-
evals:
|
|
19
|
-
- type: boolean
|
|
20
|
-
criterion: "the reply greets Dr. Smith by name"
|
|
21
|
-
```
|
|
22
|
-
|
|
23
|
-
**As code**, for matrices and deterministic mix-ins:
|
|
24
|
-
|
|
25
|
-
```python
|
|
26
|
-
from skilltest_pytest import run_skill
|
|
27
|
-
|
|
28
|
-
def test_greeter():
|
|
29
|
-
report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
|
|
30
|
-
assert report.passed, report.describe_failures()
|
|
31
|
-
assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
|
|
32
|
-
```
|
|
33
|
-
|
|
34
|
-
## Configuration
|
|
35
|
-
|
|
36
|
-
The plugin shells out to the `skilltest` binary. Point it at one with the
|
|
37
|
-
`SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
|
|
38
|
-
`provider=`), and set defaults in `pyproject.toml`:
|
|
39
|
-
|
|
40
|
-
```toml
|
|
41
|
-
[tool.pytest.ini_options]
|
|
42
|
-
skilltest_provider = "oneharness"
|
|
43
|
-
skilltest_platforms = ["claude-code"]
|
|
44
|
-
skilltest_models = ["claude-opus-4-8"]
|
|
45
|
-
```
|
|
46
|
-
|
|
47
|
-
See the repository root for the provider protocol and the full schema.
|
|
@@ -1,92 +0,0 @@
|
|
|
1
|
-
"""E2e tests for the pytest integration.
|
|
2
|
-
|
|
3
|
-
The happy path of auto-collection is also exercised by `collected/`
|
|
4
|
-
(`greet.skilltest.yaml` runs as part of this very suite); the `pytester` tests
|
|
5
|
-
here drive a *child* pytest end-to-end so the failure path — a collected case
|
|
6
|
-
whose eval fails — can be asserted on too.
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
from __future__ import annotations
|
|
10
|
-
|
|
11
|
-
import asyncio
|
|
12
|
-
from pathlib import Path
|
|
13
|
-
|
|
14
|
-
import pytest
|
|
15
|
-
|
|
16
|
-
from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
|
|
17
|
-
|
|
18
|
-
SKILL_MD = """\
|
|
19
|
-
---
|
|
20
|
-
name: greeter
|
|
21
|
-
description: A local greeter skill used to exercise pytest auto-collection.
|
|
22
|
-
---
|
|
23
|
-
# Greeter
|
|
24
|
-
|
|
25
|
-
Greet the user by name.
|
|
26
|
-
|
|
27
|
-
<!-- fake-reply: Hello, Dr. Smith! Welcome to the clinic. -->
|
|
28
|
-
"""
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
def write_skill(root: Path) -> None:
|
|
32
|
-
skill = root / "greeter"
|
|
33
|
-
skill.mkdir()
|
|
34
|
-
(skill / "SKILL.md").write_text(SKILL_MD)
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
|
|
38
|
-
# One dependency is enough for a pytest suite: the SDK's code-level API is
|
|
39
|
-
# available straight from skilltest_pytest.
|
|
40
|
-
report = run_skill(cases / "greet_pass.yaml")
|
|
41
|
-
assert report.passed, describe_failures(report)
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
|
|
45
|
-
# The tool-event and streaming surfaces are re-exported from the plugin too.
|
|
46
|
-
report = run_skill(cases / "tool_events.yaml")
|
|
47
|
-
assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
|
|
48
|
-
|
|
49
|
-
async def go() -> list[str | None]:
|
|
50
|
-
names: list[str | None] = []
|
|
51
|
-
async for ev in stream_skill(cases / "tool_events.yaml"):
|
|
52
|
-
names.append(ev.event.name)
|
|
53
|
-
return names
|
|
54
|
-
|
|
55
|
-
assert asyncio.run(go()) == ["edit_file", "bash"]
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def test_collected_case_passes(pytester: pytest.Pytester) -> None:
|
|
59
|
-
write_skill(pytester.path)
|
|
60
|
-
pytester.makefile(
|
|
61
|
-
".skilltest.yaml",
|
|
62
|
-
greet="""
|
|
63
|
-
name: collected_greet
|
|
64
|
-
skill: ./greeter
|
|
65
|
-
input: "Greet Dr. Smith."
|
|
66
|
-
evals:
|
|
67
|
-
- type: boolean
|
|
68
|
-
name: names-the-patient
|
|
69
|
-
criterion: "the reply greets `Dr. Smith` by name"
|
|
70
|
-
""",
|
|
71
|
-
)
|
|
72
|
-
result = pytester.runpytest_subprocess()
|
|
73
|
-
result.assert_outcomes(passed=1)
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester) -> None:
|
|
77
|
-
write_skill(pytester.path)
|
|
78
|
-
pytester.makefile(
|
|
79
|
-
".skilltest.yaml",
|
|
80
|
-
farewell="""
|
|
81
|
-
name: collected_farewell
|
|
82
|
-
skill: ./greeter
|
|
83
|
-
input: "Greet Dr. Smith."
|
|
84
|
-
evals:
|
|
85
|
-
- type: boolean
|
|
86
|
-
name: says-goodbye
|
|
87
|
-
criterion: "the reply contains a `goodbye`"
|
|
88
|
-
""",
|
|
89
|
-
)
|
|
90
|
-
result = pytester.runpytest_subprocess()
|
|
91
|
-
result.assert_outcomes(failed=1)
|
|
92
|
-
result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|