skilltest-pytest 0.4.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,111 @@
1
+ Metadata-Version: 2.4
2
+ Name: skilltest-pytest
3
+ Version: 0.6.0
4
+ Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
5
+ Author: Nick DeRobertis
6
+ License-Expression: MIT
7
+ Requires-Python: >=3.12
8
+ Requires-Dist: pytest>=8
9
+ Requires-Dist: skilltest-sdk==0.6.0
10
+ Description-Content-Type: text/markdown
11
+
12
+ # skilltest-pytest
13
+
14
+ A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
15
+ AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
16
+ your own deterministic checks. Built on
17
+ [`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
18
+ re-exported here, so a pytest suite needs only this one dependency.
19
+
20
+ ## Define the whole case in code (recommended)
21
+
22
+ Build the case — skill, input, evals, an optional simulated user, mocks — right
23
+ in the test. Everything the YAML carries has a typed builder, so the case, its
24
+ mocks, and any deterministic transcript checks live in one place:
25
+
26
+ ```python
27
+ from skilltest_pytest import TestCase, run_skill, boolean, numeric, describe_failures
28
+
29
+ def test_greeter():
30
+ case = TestCase(
31
+ skill="skills/greeter", # resolved relative to the working dir
32
+ input="Greet Dr. Smith, who has an appointment today.",
33
+ evals=[
34
+ boolean("the reply greets Dr. Smith by name"),
35
+ numeric("how warm is the tone", min=0, max=10, threshold=7),
36
+ ],
37
+ )
38
+ report = run_skill(case, platforms=["claude-code"], models=["claude-opus-4-8"])
39
+ assert report.passed, describe_failures(report)
40
+ ```
41
+
42
+ Multi-turn cases add `user(...)`; deterministic call-count checks use `called` /
43
+ `not_called` referencing a named `stub`/`spy` (or the mock objects' own
44
+ assertions — see below). `run_skill` also takes `platforms=`/`models=` to fan a
45
+ case across a matrix.
46
+
47
+ ## Or point at a YAML file
48
+
49
+ `run_skill` accepts a path just as well (`run_skill("cases/greet.yaml")`), and
50
+ **auto-collection** still works: name a case `something.skilltest.yaml` and
51
+ pytest runs it with no test function at all —
52
+
53
+ ```yaml
54
+ # greet.skilltest.yaml
55
+ skill: ./skills/greeter
56
+ input: "Greet Dr. Smith."
57
+ evals:
58
+ - type: boolean
59
+ criterion: "the reply greets Dr. Smith by name"
60
+ ```
61
+
62
+ The full field reference for both forms is [`docs/schema.md`](../../docs/schema.md).
63
+
64
+ ## Assert on tool use, and stream
65
+
66
+ The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
67
+ returns the normalized `tool_call` events a run took (each a `ToolEvent` with
68
+ `kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
69
+ test can **short-circuit** on bad behavior:
70
+
71
+ ```python
72
+ from skilltest_pytest import TestCase, run_skill, tool_calls, boolean
73
+
74
+ EDIT_CASE = TestCase(
75
+ skill="skills/editor",
76
+ input="Update the config and commit it.",
77
+ evals=[boolean("the change was committed")],
78
+ )
79
+
80
+ def test_commits_but_never_deletes():
81
+ report = run_skill(EDIT_CASE)
82
+ calls = tool_calls(report.runs[0].transcript)
83
+ assert any("git commit" in str(c.input) for c in calls)
84
+ assert not any("rm -rf" in str(c.input) for c in calls)
85
+ ```
86
+
87
+ ```python
88
+ import asyncio
89
+ from skilltest_pytest import stream_skill
90
+
91
+ def test_makes_no_network_call():
92
+ async def go():
93
+ async for ev in stream_skill(EDIT_CASE):
94
+ assert ev.event.name != "curl", "skill made a network call"
95
+ asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
96
+ ```
97
+
98
+ ## Configuration
99
+
100
+ The plugin shells out to the `skilltest` binary. Point it at one with the
101
+ `SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
102
+ `provider=`), and set defaults in `pyproject.toml`:
103
+
104
+ ```toml
105
+ [tool.pytest.ini_options]
106
+ skilltest_provider = "oneharness"
107
+ skilltest_platforms = ["claude-code"]
108
+ skilltest_models = ["claude-opus-4-8"]
109
+ ```
110
+
111
+ See the repository root for the provider protocol and the full schema.
@@ -0,0 +1,100 @@
1
+ # skilltest-pytest
2
+
3
+ A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
4
+ AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
5
+ your own deterministic checks. Built on
6
+ [`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
7
+ re-exported here, so a pytest suite needs only this one dependency.
8
+
9
+ ## Define the whole case in code (recommended)
10
+
11
+ Build the case — skill, input, evals, an optional simulated user, mocks — right
12
+ in the test. Everything the YAML carries has a typed builder, so the case, its
13
+ mocks, and any deterministic transcript checks live in one place:
14
+
15
+ ```python
16
+ from skilltest_pytest import TestCase, run_skill, boolean, numeric, describe_failures
17
+
18
+ def test_greeter():
19
+ case = TestCase(
20
+ skill="skills/greeter", # resolved relative to the working dir
21
+ input="Greet Dr. Smith, who has an appointment today.",
22
+ evals=[
23
+ boolean("the reply greets Dr. Smith by name"),
24
+ numeric("how warm is the tone", min=0, max=10, threshold=7),
25
+ ],
26
+ )
27
+ report = run_skill(case, platforms=["claude-code"], models=["claude-opus-4-8"])
28
+ assert report.passed, describe_failures(report)
29
+ ```
30
+
31
+ Multi-turn cases add `user(...)`; deterministic call-count checks use `called` /
32
+ `not_called` referencing a named `stub`/`spy` (or the mock objects' own
33
+ assertions — see below). `run_skill` also takes `platforms=`/`models=` to fan a
34
+ case across a matrix.
35
+
36
+ ## Or point at a YAML file
37
+
38
+ `run_skill` accepts a path just as well (`run_skill("cases/greet.yaml")`), and
39
+ **auto-collection** still works: name a case `something.skilltest.yaml` and
40
+ pytest runs it with no test function at all —
41
+
42
+ ```yaml
43
+ # greet.skilltest.yaml
44
+ skill: ./skills/greeter
45
+ input: "Greet Dr. Smith."
46
+ evals:
47
+ - type: boolean
48
+ criterion: "the reply greets Dr. Smith by name"
49
+ ```
50
+
51
+ The full field reference for both forms is [`docs/schema.md`](../../docs/schema.md).
52
+
53
+ ## Assert on tool use, and stream
54
+
55
+ The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
56
+ returns the normalized `tool_call` events a run took (each a `ToolEvent` with
57
+ `kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
58
+ test can **short-circuit** on bad behavior:
59
+
60
+ ```python
61
+ from skilltest_pytest import TestCase, run_skill, tool_calls, boolean
62
+
63
+ EDIT_CASE = TestCase(
64
+ skill="skills/editor",
65
+ input="Update the config and commit it.",
66
+ evals=[boolean("the change was committed")],
67
+ )
68
+
69
+ def test_commits_but_never_deletes():
70
+ report = run_skill(EDIT_CASE)
71
+ calls = tool_calls(report.runs[0].transcript)
72
+ assert any("git commit" in str(c.input) for c in calls)
73
+ assert not any("rm -rf" in str(c.input) for c in calls)
74
+ ```
75
+
76
+ ```python
77
+ import asyncio
78
+ from skilltest_pytest import stream_skill
79
+
80
+ def test_makes_no_network_call():
81
+ async def go():
82
+ async for ev in stream_skill(EDIT_CASE):
83
+ assert ev.event.name != "curl", "skill made a network call"
84
+ asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
85
+ ```
86
+
87
+ ## Configuration
88
+
89
+ The plugin shells out to the `skilltest` binary. Point it at one with the
90
+ `SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
91
+ `provider=`), and set defaults in `pyproject.toml`:
92
+
93
+ ```toml
94
+ [tool.pytest.ini_options]
95
+ skilltest_provider = "oneharness"
96
+ skilltest_platforms = ["claude-code"]
97
+ skilltest_models = ["claude-opus-4-8"]
98
+ ```
99
+
100
+ See the repository root for the provider protocol and the full schema.
@@ -1,13 +1,13 @@
1
1
  [project]
2
2
  name = "skilltest-pytest"
3
- version = "0.4.0"
3
+ version = "0.6.0"
4
4
  description = "pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
7
7
  license = "MIT"
8
8
  authors = [{ name = "Nick DeRobertis" }]
9
9
  dependencies = [
10
- "skilltest-sdk==0.4.0",
10
+ "skilltest-sdk==0.6.0",
11
11
  "pytest>=8",
12
12
  ]
13
13
 
@@ -1,17 +1,23 @@
1
1
  """skilltest-pytest: run AI-skill tests and natural-language evals as pytest.
2
2
 
3
- The pytest integration on top of [`skilltest-sdk`][skilltest_sdk]: drop a
4
- ``*.skilltest.yaml`` next to your other tests and pytest collects it as a test
5
- item. The SDK's code-level API is re-exported here for convenience, so a pytest
6
- suite only needs one dependency:
3
+ The pytest integration on top of [`skilltest-sdk`][skilltest_sdk], whose API is
4
+ re-exported here so a pytest suite needs only this one dependency. Define the
5
+ whole case in code (the recommended form):
7
6
 
8
- from skilltest_pytest import run_skill, validate_skill
7
+ from skilltest_pytest import TestCase, run_skill, boolean, describe_failures
9
8
 
10
9
  def test_greeter():
11
- report = run_skill("cases/greet.yaml")
10
+ case = TestCase(
11
+ skill="skills/greeter",
12
+ input="Greet Dr. Smith.",
13
+ evals=[boolean("the reply greets Dr. Smith by name")],
14
+ )
15
+ report = run_skill(case)
12
16
  assert report.passed, describe_failures(report)
13
- # Mix in a deterministic check on the transcript:
14
- assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
17
+
18
+ Existing YAML cases stay first-class: ``run_skill("cases/greet.yaml")`` runs a
19
+ file, and any ``*.skilltest.yaml`` dropped next to your tests is auto-collected
20
+ as a test item with no code at all.
15
21
  """
16
22
 
17
23
  from __future__ import annotations
@@ -20,29 +26,50 @@ from skilltest_sdk import (
20
26
  ENV_BIN,
21
27
  ENV_PROVIDER,
22
28
  BooleanDetail,
29
+ CallsDetail,
23
30
  CaseRun,
31
+ Eval,
24
32
  EvalOutcome,
33
+ Matcher,
25
34
  Message,
35
+ MockCall,
26
36
  NumericDetail,
27
37
  Report,
38
+ SimulatedUser,
28
39
  SkillStream,
29
40
  SkilltestError,
30
41
  SkilltestProviderError,
31
42
  SkilltestUsageError,
32
43
  StreamEvent,
33
44
  Summary,
45
+ TestCase,
46
+ ToolCall,
34
47
  ToolEvent,
48
+ ToolMock,
49
+ ToolSpy,
35
50
  Transcript,
36
51
  Usage,
37
52
  ValidationFinding,
38
53
  ValidationReport,
54
+ anything,
39
55
  assistant_text,
56
+ boolean,
57
+ called,
58
+ contains,
59
+ deny,
40
60
  describe_failures,
41
61
  failed_evals,
42
62
  failed_runs,
63
+ matching,
64
+ not_called,
65
+ numeric,
66
+ rewrite,
43
67
  run_skill,
68
+ spy,
44
69
  stream_skill,
70
+ stub,
45
71
  tool_calls,
72
+ user,
46
73
  validate_skill,
47
74
  )
48
75
 
@@ -52,11 +79,16 @@ __all__ = [
52
79
  "ENV_BIN",
53
80
  "ENV_PROVIDER",
54
81
  "BooleanDetail",
82
+ "CallsDetail",
55
83
  "CaseRun",
84
+ "Eval",
56
85
  "EvalOutcome",
86
+ "Matcher",
57
87
  "Message",
88
+ "MockCall",
58
89
  "NumericDetail",
59
90
  "Report",
91
+ "SimulatedUser",
60
92
  "SkillStream",
61
93
  "SkilltestError",
62
94
  "SkilltestFailure",
@@ -64,17 +96,33 @@ __all__ = [
64
96
  "SkilltestUsageError",
65
97
  "StreamEvent",
66
98
  "Summary",
99
+ "TestCase",
100
+ "ToolCall",
67
101
  "ToolEvent",
102
+ "ToolMock",
103
+ "ToolSpy",
68
104
  "Transcript",
69
105
  "Usage",
70
106
  "ValidationFinding",
71
107
  "ValidationReport",
108
+ "anything",
72
109
  "assistant_text",
110
+ "boolean",
111
+ "called",
112
+ "contains",
113
+ "deny",
73
114
  "describe_failures",
74
115
  "failed_evals",
75
116
  "failed_runs",
117
+ "matching",
118
+ "not_called",
119
+ "numeric",
120
+ "rewrite",
76
121
  "run_skill",
122
+ "spy",
77
123
  "stream_skill",
124
+ "stub",
78
125
  "tool_calls",
126
+ "user",
79
127
  "validate_skill",
80
128
  ]
@@ -0,0 +1,33 @@
1
+ # Auto-collected mocked case: the declarative `mocks:` block plus deterministic
2
+ # `called`/`not_called` evals — no plugin code involved, proving the whole
3
+ # mock/spy surface rides YAML collection.
4
+ name: collected_deploy
5
+ skill: ./deployer
6
+ input: "Deploy the app"
7
+ mocks:
8
+ - name: push
9
+ match: { tool: bash, pattern: "git push( --force)?\\b" }
10
+ stub: Everything up-to-date
11
+ - name: danger
12
+ match: { contains: "rm -rf" }
13
+ deny: destructive commands are blocked
14
+ - name: git
15
+ match: { tool: bash, pattern: "\\bgit\\b" }
16
+ evals:
17
+ - type: boolean
18
+ name: canned-output-surfaced
19
+ criterion: "the reply mentions `Everything up-to-date`"
20
+ - type: called
21
+ mock: push
22
+ times: 1
23
+ - type: called
24
+ name: both-git-calls-observed
25
+ mock: git
26
+ times: 2
27
+ - type: called
28
+ name: the-deny-intercepted
29
+ mock: danger
30
+ times: 1
31
+ - type: not_called
32
+ mock: push
33
+ where: { command: { contains: "--force" } }
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: deployer
3
+ description: Deploys the app by pushing to git and reporting the outcome.
4
+ ---
5
+ # Deployer
6
+
7
+ Deploy the current app: push the branch, check the working tree, clean the
8
+ build directory, and report the outcome in one sentence.
9
+
10
+ <!-- fake-reply: Deployment finished. -->
11
+ <!-- fake-tool: bash git push origin main -->
12
+ <!-- fake-tool: bash git status -->
13
+ <!-- fake-tool: bash rm -rf /tmp/build -->
@@ -0,0 +1,171 @@
1
+ """E2e tests for the pytest integration.
2
+
3
+ The happy path of auto-collection is also exercised by `collected/`
4
+ (`greet.skilltest.yaml` runs as part of this very suite); the `pytester` tests
5
+ here drive a *child* pytest end-to-end so the failure path — a collected case
6
+ whose eval fails — can be asserted on too.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import asyncio
12
+ from pathlib import Path
13
+
14
+ import pytest
15
+
16
+ from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
17
+
18
+ SKILL_MD = """\
19
+ ---
20
+ name: greeter
21
+ description: A local greeter skill used to exercise pytest auto-collection.
22
+ ---
23
+ # Greeter
24
+
25
+ Greet the user by name.
26
+
27
+ <!-- fake-reply: Hello, Dr. Smith! Welcome to the clinic. -->
28
+ """
29
+
30
+
31
+ def write_skill(root: Path) -> None:
32
+ skill = root / "greeter"
33
+ skill.mkdir()
34
+ (skill / "SKILL.md").write_text(SKILL_MD)
35
+
36
+
37
+ def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
38
+ # One dependency is enough for a pytest suite: the SDK's code-level API is
39
+ # available straight from skilltest_pytest.
40
+ report = run_skill(cases / "greet_pass.yaml")
41
+ assert report.passed, describe_failures(report)
42
+
43
+
44
+ def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
45
+ # The tool-event and streaming surfaces are re-exported from the plugin too.
46
+ report = run_skill(cases / "tool_events.yaml")
47
+ assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
48
+
49
+ async def go() -> list[str | None]:
50
+ names: list[str | None] = []
51
+ async for ev in stream_skill(cases / "tool_events.yaml"):
52
+ names.append(ev.event.name)
53
+ return names
54
+
55
+ assert asyncio.run(go()) == ["edit_file", "bash"]
56
+
57
+
58
+ def test_collected_case_passes(pytester: pytest.Pytester) -> None:
59
+ write_skill(pytester.path)
60
+ pytester.makefile(
61
+ ".skilltest.yaml",
62
+ greet="""
63
+ name: collected_greet
64
+ skill: ./greeter
65
+ input: "Greet Dr. Smith."
66
+ evals:
67
+ - type: boolean
68
+ name: names-the-patient
69
+ criterion: "the reply greets `Dr. Smith` by name"
70
+ """,
71
+ )
72
+ result = pytester.runpytest_subprocess()
73
+ result.assert_outcomes(passed=1)
74
+
75
+
76
+ def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester) -> None:
77
+ write_skill(pytester.path)
78
+ pytester.makefile(
79
+ ".skilltest.yaml",
80
+ farewell="""
81
+ name: collected_farewell
82
+ skill: ./greeter
83
+ input: "Greet Dr. Smith."
84
+ evals:
85
+ - type: boolean
86
+ name: says-goodbye
87
+ criterion: "the reply contains a `goodbye`"
88
+ """,
89
+ )
90
+ result = pytester.runpytest_subprocess()
91
+ result.assert_outcomes(failed=1)
92
+ result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])
93
+
94
+
95
+ def test_code_defined_case_is_reexported_and_runs(fixtures: Path) -> None:
96
+ # The recommended form: build the whole case in code and hand it to
97
+ # run_skill — the case API rides the same one-dependency re-export.
98
+ from skilltest_pytest import TestCase, boolean, run_skill
99
+
100
+ case = TestCase(
101
+ skill=fixtures / "skills" / "greeter",
102
+ input="Greet Dr. Smith, who has an appointment today.",
103
+ evals=[boolean("the reply greets `Dr. Smith` by name")],
104
+ )
105
+ report = run_skill(case)
106
+ assert report.passed, describe_failures(report)
107
+
108
+
109
+ def test_full_code_defined_case_surface_is_reexported(fixtures: Path) -> None:
110
+ # Every case builder rides the one-dependency re-export: a multi-turn case
111
+ # with judge evals, and a mocked case with a deterministic call eval —
112
+ # streamed and buffered — defined entirely in code through the plugin.
113
+ from skilltest_pytest import (
114
+ TestCase,
115
+ boolean,
116
+ called,
117
+ numeric,
118
+ run_skill,
119
+ stream_skill,
120
+ stub,
121
+ user,
122
+ )
123
+
124
+ multi = TestCase(
125
+ skill=fixtures / "skills" / "greeter",
126
+ input="I'd like to confirm my appointment, please.",
127
+ user=user(
128
+ "You are a terse patient.\nsay: Yes, please go ahead.",
129
+ done_when="the conversation has reached turns>=2",
130
+ max_turns=4,
131
+ ),
132
+ evals=[
133
+ boolean("the assistant confirmed the appointment (`confirmed`)"),
134
+ numeric("mentions `confirmed`", min=0, max=10, threshold=5, comparator=">"),
135
+ ],
136
+ )
137
+ report = run_skill(multi)
138
+ assert report.passed, describe_failures(report)
139
+ assert report.runs[0].turns == 2
140
+
141
+ push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date", name="push")
142
+ mocked = TestCase(
143
+ skill=fixtures / "skills" / "deployer",
144
+ input="Deploy the app",
145
+ mocks=[push],
146
+ evals=[called("push", times=1)],
147
+ )
148
+ stream = stream_skill(mocked)
149
+
150
+ async def drain() -> None:
151
+ async for _ in stream:
152
+ pass
153
+
154
+ asyncio.run(drain())
155
+ assert stream.report is not None and stream.report.passed
156
+ push.assert_called_once()
157
+
158
+
159
+ def test_mock_api_is_reexported_and_binds(cases: Path) -> None:
160
+ # The mock/spy API rides the one-dependency re-export; code-level mocks
161
+ # intercept and bind through the plugin's SDK exactly as through the SDK.
162
+ from skilltest_pytest import contains, matching, run_skill, spy, stub
163
+
164
+ push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date")
165
+ git = spy(tool="bash", pattern=r"\bgit\b")
166
+ report = run_skill(cases / "deploy_plain.yaml", mocks=[push, git])
167
+ assert report.passed
168
+ push.assert_called_once()
169
+ assert push.calls[0].command == "git push origin main"
170
+ git.assert_called_with(command=contains("git status"))
171
+ git.where(command=matching(r"\bsudo\b")).assert_not_called()
@@ -189,7 +189,7 @@ wheels = [
189
189
 
190
190
  [[package]]
191
191
  name = "skilltest-pytest"
192
- version = "0.4.0"
192
+ version = "0.6.0"
193
193
  source = { editable = "." }
194
194
  dependencies = [
195
195
  { name = "pytest" },
@@ -216,7 +216,7 @@ dev = [
216
216
 
217
217
  [[package]]
218
218
  name = "skilltest-sdk"
219
- version = "0.4.0"
219
+ version = "0.6.0"
220
220
  source = { editable = "../../sdks/python" }
221
221
  dependencies = [
222
222
  { name = "pydantic" },
@@ -1,58 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: skilltest-pytest
3
- Version: 0.4.0
4
- Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
5
- Author: Nick DeRobertis
6
- License-Expression: MIT
7
- Requires-Python: >=3.12
8
- Requires-Dist: pytest>=8
9
- Requires-Dist: skilltest-sdk==0.4.0
10
- Description-Content-Type: text/markdown
11
-
12
- # skilltest-pytest
13
-
14
- A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
15
- AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
16
- your own deterministic checks. Built on
17
- [`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
18
- re-exported here, so a pytest suite needs only this one dependency.
19
-
20
- ## Two ways to use it
21
-
22
- **Auto-collected case files.** Name a case `something.skilltest.yaml` and pytest
23
- runs it:
24
-
25
- ```yaml
26
- # greet.skilltest.yaml
27
- skill: ./skills/greeter
28
- input: "Greet Dr. Smith."
29
- evals:
30
- - type: boolean
31
- criterion: "the reply greets Dr. Smith by name"
32
- ```
33
-
34
- **As code**, for matrices and deterministic mix-ins:
35
-
36
- ```python
37
- from skilltest_pytest import run_skill
38
-
39
- def test_greeter():
40
- report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
41
- assert report.passed, report.describe_failures()
42
- assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
43
- ```
44
-
45
- ## Configuration
46
-
47
- The plugin shells out to the `skilltest` binary. Point it at one with the
48
- `SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
49
- `provider=`), and set defaults in `pyproject.toml`:
50
-
51
- ```toml
52
- [tool.pytest.ini_options]
53
- skilltest_provider = "oneharness"
54
- skilltest_platforms = ["claude-code"]
55
- skilltest_models = ["claude-opus-4-8"]
56
- ```
57
-
58
- See the repository root for the provider protocol and the full schema.
@@ -1,47 +0,0 @@
1
- # skilltest-pytest
2
-
3
- A [pytest](https://pytest.org) plugin for [skilltest](../../README.md): run
4
- AI-skill tests and natural-language evals as ordinary pytest tests, and mix in
5
- your own deterministic checks. Built on
6
- [`skilltest-sdk`](../../sdks/python/README.md) — the SDK's code API is
7
- re-exported here, so a pytest suite needs only this one dependency.
8
-
9
- ## Two ways to use it
10
-
11
- **Auto-collected case files.** Name a case `something.skilltest.yaml` and pytest
12
- runs it:
13
-
14
- ```yaml
15
- # greet.skilltest.yaml
16
- skill: ./skills/greeter
17
- input: "Greet Dr. Smith."
18
- evals:
19
- - type: boolean
20
- criterion: "the reply greets Dr. Smith by name"
21
- ```
22
-
23
- **As code**, for matrices and deterministic mix-ins:
24
-
25
- ```python
26
- from skilltest_pytest import run_skill
27
-
28
- def test_greeter():
29
- report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
30
- assert report.passed, report.describe_failures()
31
- assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
32
- ```
33
-
34
- ## Configuration
35
-
36
- The plugin shells out to the `skilltest` binary. Point it at one with the
37
- `SKILLTEST_BIN` env var (or `bin=`), the provider with `SKILLTEST_PROVIDER` (or
38
- `provider=`), and set defaults in `pyproject.toml`:
39
-
40
- ```toml
41
- [tool.pytest.ini_options]
42
- skilltest_provider = "oneharness"
43
- skilltest_platforms = ["claude-code"]
44
- skilltest_models = ["claude-opus-4-8"]
45
- ```
46
-
47
- See the repository root for the provider protocol and the full schema.
@@ -1,92 +0,0 @@
1
- """E2e tests for the pytest integration.
2
-
3
- The happy path of auto-collection is also exercised by `collected/`
4
- (`greet.skilltest.yaml` runs as part of this very suite); the `pytester` tests
5
- here drive a *child* pytest end-to-end so the failure path — a collected case
6
- whose eval fails — can be asserted on too.
7
- """
8
-
9
- from __future__ import annotations
10
-
11
- import asyncio
12
- from pathlib import Path
13
-
14
- import pytest
15
-
16
- from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
17
-
18
- SKILL_MD = """\
19
- ---
20
- name: greeter
21
- description: A local greeter skill used to exercise pytest auto-collection.
22
- ---
23
- # Greeter
24
-
25
- Greet the user by name.
26
-
27
- <!-- fake-reply: Hello, Dr. Smith! Welcome to the clinic. -->
28
- """
29
-
30
-
31
- def write_skill(root: Path) -> None:
32
- skill = root / "greeter"
33
- skill.mkdir()
34
- (skill / "SKILL.md").write_text(SKILL_MD)
35
-
36
-
37
- def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
38
- # One dependency is enough for a pytest suite: the SDK's code-level API is
39
- # available straight from skilltest_pytest.
40
- report = run_skill(cases / "greet_pass.yaml")
41
- assert report.passed, describe_failures(report)
42
-
43
-
44
- def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
45
- # The tool-event and streaming surfaces are re-exported from the plugin too.
46
- report = run_skill(cases / "tool_events.yaml")
47
- assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
48
-
49
- async def go() -> list[str | None]:
50
- names: list[str | None] = []
51
- async for ev in stream_skill(cases / "tool_events.yaml"):
52
- names.append(ev.event.name)
53
- return names
54
-
55
- assert asyncio.run(go()) == ["edit_file", "bash"]
56
-
57
-
58
- def test_collected_case_passes(pytester: pytest.Pytester) -> None:
59
- write_skill(pytester.path)
60
- pytester.makefile(
61
- ".skilltest.yaml",
62
- greet="""
63
- name: collected_greet
64
- skill: ./greeter
65
- input: "Greet Dr. Smith."
66
- evals:
67
- - type: boolean
68
- name: names-the-patient
69
- criterion: "the reply greets `Dr. Smith` by name"
70
- """,
71
- )
72
- result = pytester.runpytest_subprocess()
73
- result.assert_outcomes(passed=1)
74
-
75
-
76
- def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester) -> None:
77
- write_skill(pytester.path)
78
- pytester.makefile(
79
- ".skilltest.yaml",
80
- farewell="""
81
- name: collected_farewell
82
- skill: ./greeter
83
- input: "Greet Dr. Smith."
84
- evals:
85
- - type: boolean
86
- name: says-goodbye
87
- criterion: "the reply contains a `goodbye`"
88
- """,
89
- )
90
- result = pytester.runpytest_subprocess()
91
- result.assert_outcomes(failed=1)
92
- result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])