skilltest-pytest 0.3.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: skilltest-pytest
3
- Version: 0.3.0
3
+ Version: 0.5.0
4
4
  Summary: pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk.
5
5
  Author: Nick DeRobertis
6
6
  License-Expression: MIT
7
7
  Requires-Python: >=3.12
8
8
  Requires-Dist: pytest>=8
9
- Requires-Dist: skilltest-sdk==0.3.0
9
+ Requires-Dist: skilltest-sdk==0.5.0
10
10
  Description-Content-Type: text/markdown
11
11
 
12
12
  # skilltest-pytest
@@ -34,12 +34,40 @@ evals:
34
34
  **As code**, for matrices and deterministic mix-ins:
35
35
 
36
36
  ```python
37
- from skilltest_pytest import run_skill
37
+ from skilltest_pytest import run_skill, describe_failures, assistant_text
38
38
 
39
39
  def test_greeter():
40
40
  report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
41
- assert report.passed, report.describe_failures()
42
- assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
41
+ assert report.passed, describe_failures(report)
42
+ assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
43
+ ```
44
+
45
+ ## Assert on tool use, and stream
46
+
47
+ The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
48
+ returns the normalized `tool_call` events a run took (each a `ToolEvent` with
49
+ `kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
50
+ test can **short-circuit** on bad behavior:
51
+
52
+ ```python
53
+ from skilltest_pytest import run_skill, tool_calls
54
+
55
+ def test_commits_but_never_deletes():
56
+ report = run_skill("cases/edit.skilltest.yaml")
57
+ calls = tool_calls(report.runs[0].transcript)
58
+ assert any("git commit" in str(c.input) for c in calls)
59
+ assert not any("rm -rf" in str(c.input) for c in calls)
60
+ ```
61
+
62
+ ```python
63
+ import asyncio
64
+ from skilltest_pytest import stream_skill
65
+
66
+ def test_makes_no_network_call():
67
+ async def go():
68
+ async for ev in stream_skill("cases/edit.skilltest.yaml"):
69
+ assert ev.event.name != "curl", "skill made a network call"
70
+ asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
43
71
  ```
44
72
 
45
73
  ## Configuration
@@ -23,12 +23,40 @@ evals:
23
23
  **As code**, for matrices and deterministic mix-ins:
24
24
 
25
25
  ```python
26
- from skilltest_pytest import run_skill
26
+ from skilltest_pytest import run_skill, describe_failures, assistant_text
27
27
 
28
28
  def test_greeter():
29
29
  report = run_skill("cases/greet.yaml", platforms=["claude-code"], models=["claude-opus-4-8"])
30
- assert report.passed, report.describe_failures()
31
- assert "Dr. Smith" in report.runs[0].transcript.assistant_text()
30
+ assert report.passed, describe_failures(report)
31
+ assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
32
+ ```
33
+
34
+ ## Assert on tool use, and stream
35
+
36
+ The SDK's tool-event and streaming surfaces are re-exported too. `tool_calls`
37
+ returns the normalized `tool_call` events a run took (each a `ToolEvent` with
38
+ `kind`/`name`/`input`/`output`/`index`), and `stream_skill` yields them live so a
39
+ test can **short-circuit** on bad behavior:
40
+
41
+ ```python
42
+ from skilltest_pytest import run_skill, tool_calls
43
+
44
+ def test_commits_but_never_deletes():
45
+ report = run_skill("cases/edit.skilltest.yaml")
46
+ calls = tool_calls(report.runs[0].transcript)
47
+ assert any("git commit" in str(c.input) for c in calls)
48
+ assert not any("rm -rf" in str(c.input) for c in calls)
49
+ ```
50
+
51
+ ```python
52
+ import asyncio
53
+ from skilltest_pytest import stream_skill
54
+
55
+ def test_makes_no_network_call():
56
+ async def go():
57
+ async for ev in stream_skill("cases/edit.skilltest.yaml"):
58
+ assert ev.event.name != "curl", "skill made a network call"
59
+ asyncio.run(go()) # or use pytest-asyncio and `async def test_...`
32
60
  ```
33
61
 
34
62
  ## Configuration
@@ -1,13 +1,13 @@
1
1
  [project]
2
2
  name = "skilltest-pytest"
3
- version = "0.3.0"
3
+ version = "0.5.0"
4
4
  description = "pytest integration for skilltest: auto-collect *.skilltest.yaml cases as pytest tests, built on skilltest-sdk."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
7
7
  license = "MIT"
8
8
  authors = [{ name = "Nick DeRobertis" }]
9
9
  dependencies = [
10
- "skilltest-sdk==0.3.0",
10
+ "skilltest-sdk==0.5.0",
11
11
  "pytest>=8",
12
12
  ]
13
13
 
@@ -20,24 +20,42 @@ from skilltest_sdk import (
20
20
  ENV_BIN,
21
21
  ENV_PROVIDER,
22
22
  BooleanDetail,
23
+ CallsDetail,
23
24
  CaseRun,
24
25
  EvalOutcome,
26
+ Matcher,
25
27
  Message,
28
+ MockCall,
26
29
  NumericDetail,
27
30
  Report,
31
+ SkillStream,
28
32
  SkilltestError,
29
33
  SkilltestProviderError,
30
34
  SkilltestUsageError,
35
+ StreamEvent,
31
36
  Summary,
37
+ ToolCall,
38
+ ToolEvent,
39
+ ToolMock,
40
+ ToolSpy,
32
41
  Transcript,
33
42
  Usage,
34
43
  ValidationFinding,
35
44
  ValidationReport,
45
+ anything,
36
46
  assistant_text,
47
+ contains,
48
+ deny,
37
49
  describe_failures,
38
50
  failed_evals,
39
51
  failed_runs,
52
+ matching,
53
+ rewrite,
40
54
  run_skill,
55
+ spy,
56
+ stream_skill,
57
+ stub,
58
+ tool_calls,
41
59
  validate_skill,
42
60
  )
43
61
 
@@ -47,24 +65,42 @@ __all__ = [
47
65
  "ENV_BIN",
48
66
  "ENV_PROVIDER",
49
67
  "BooleanDetail",
68
+ "CallsDetail",
50
69
  "CaseRun",
51
70
  "EvalOutcome",
71
+ "Matcher",
52
72
  "Message",
73
+ "MockCall",
53
74
  "NumericDetail",
54
75
  "Report",
76
+ "SkillStream",
55
77
  "SkilltestError",
56
78
  "SkilltestFailure",
57
79
  "SkilltestProviderError",
58
80
  "SkilltestUsageError",
81
+ "StreamEvent",
59
82
  "Summary",
83
+ "ToolCall",
84
+ "ToolEvent",
85
+ "ToolMock",
86
+ "ToolSpy",
60
87
  "Transcript",
61
88
  "Usage",
62
89
  "ValidationFinding",
63
90
  "ValidationReport",
91
+ "anything",
64
92
  "assistant_text",
93
+ "contains",
94
+ "deny",
65
95
  "describe_failures",
66
96
  "failed_evals",
67
97
  "failed_runs",
98
+ "matching",
99
+ "rewrite",
68
100
  "run_skill",
101
+ "spy",
102
+ "stream_skill",
103
+ "stub",
104
+ "tool_calls",
69
105
  "validate_skill",
70
106
  ]
@@ -0,0 +1,33 @@
1
+ # Auto-collected mocked case: the declarative `mocks:` block plus deterministic
2
+ # `called`/`not_called` evals — no plugin code involved, proving the whole
3
+ # mock/spy surface rides YAML collection.
4
+ name: collected_deploy
5
+ skill: ./deployer
6
+ input: "Deploy the app"
7
+ mocks:
8
+ - name: push
9
+ match: { tool: bash, pattern: "git push( --force)?\\b" }
10
+ stub: Everything up-to-date
11
+ - name: danger
12
+ match: { contains: "rm -rf" }
13
+ deny: destructive commands are blocked
14
+ - name: git
15
+ match: { tool: bash, pattern: "\\bgit\\b" }
16
+ evals:
17
+ - type: boolean
18
+ name: canned-output-surfaced
19
+ criterion: "the reply mentions `Everything up-to-date`"
20
+ - type: called
21
+ mock: push
22
+ times: 1
23
+ - type: called
24
+ name: both-git-calls-observed
25
+ mock: git
26
+ times: 2
27
+ - type: called
28
+ name: the-deny-intercepted
29
+ mock: danger
30
+ times: 1
31
+ - type: not_called
32
+ mock: push
33
+ where: { command: { contains: "--force" } }
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: deployer
3
+ description: Deploys the app by pushing to git and reporting the outcome.
4
+ ---
5
+ # Deployer
6
+
7
+ Deploy the current app: push the branch, check the working tree, clean the
8
+ build directory, and report the outcome in one sentence.
9
+
10
+ <!-- fake-reply: Deployment finished. -->
11
+ <!-- fake-tool: bash git push origin main -->
12
+ <!-- fake-tool: bash git status -->
13
+ <!-- fake-tool: bash rm -rf /tmp/build -->
@@ -8,11 +8,12 @@ whose eval fails — can be asserted on too.
8
8
 
9
9
  from __future__ import annotations
10
10
 
11
+ import asyncio
11
12
  from pathlib import Path
12
13
 
13
14
  import pytest
14
15
 
15
- from skilltest_pytest import describe_failures, run_skill
16
+ from skilltest_pytest import describe_failures, run_skill, stream_skill, tool_calls
16
17
 
17
18
  SKILL_MD = """\
18
19
  ---
@@ -40,6 +41,20 @@ def test_sdk_api_is_reexported_and_works(cases: Path) -> None:
40
41
  assert report.passed, describe_failures(report)
41
42
 
42
43
 
44
+ def test_tool_events_and_streaming_reexported_and_work(cases: Path) -> None:
45
+ # The tool-event and streaming surfaces are re-exported from the plugin too.
46
+ report = run_skill(cases / "tool_events.yaml")
47
+ assert [c.name for c in tool_calls(report.runs[0].transcript)] == ["edit_file", "bash"]
48
+
49
+ async def go() -> list[str | None]:
50
+ names: list[str | None] = []
51
+ async for ev in stream_skill(cases / "tool_events.yaml"):
52
+ names.append(ev.event.name)
53
+ return names
54
+
55
+ assert asyncio.run(go()) == ["edit_file", "bash"]
56
+
57
+
43
58
  def test_collected_case_passes(pytester: pytest.Pytester) -> None:
44
59
  write_skill(pytester.path)
45
60
  pytester.makefile(
@@ -75,3 +90,18 @@ def test_collected_case_failure_reports_judge_reason(pytester: pytest.Pytester)
75
90
  result = pytester.runpytest_subprocess()
76
91
  result.assert_outcomes(failed=1)
77
92
  result.stdout.fnmatch_lines(["*skilltest case failed:*", "*says-goodbye*"])
93
+
94
+
95
+ def test_mock_api_is_reexported_and_binds(cases: Path) -> None:
96
+ # The mock/spy API rides the one-dependency re-export; code-level mocks
97
+ # intercept and bind through the plugin's SDK exactly as through the SDK.
98
+ from skilltest_pytest import contains, matching, run_skill, spy, stub
99
+
100
+ push = stub(pattern=r"git push( --force)?\b", output="Everything up-to-date")
101
+ git = spy(tool="bash", pattern=r"\bgit\b")
102
+ report = run_skill(cases / "deploy_plain.yaml", mocks=[push, git])
103
+ assert report.passed
104
+ push.assert_called_once()
105
+ assert push.calls[0].command == "git push origin main"
106
+ git.assert_called_with(command=contains("git status"))
107
+ git.where(command=matching(r"\bsudo\b")).assert_not_called()
@@ -189,7 +189,7 @@ wheels = [
189
189
 
190
190
  [[package]]
191
191
  name = "skilltest-pytest"
192
- version = "0.3.0"
192
+ version = "0.5.0"
193
193
  source = { editable = "." }
194
194
  dependencies = [
195
195
  { name = "pytest" },
@@ -216,7 +216,7 @@ dev = [
216
216
 
217
217
  [[package]]
218
218
  name = "skilltest-sdk"
219
- version = "0.3.0"
219
+ version = "0.5.0"
220
220
  source = { editable = "../../sdks/python" }
221
221
  dependencies = [
222
222
  { name = "pydantic" },