skilltest-sdk 0.2.2__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: skilltest-sdk
3
- Version: 0.2.2
3
+ Version: 0.4.0
4
4
  Summary: Python SDK for the skilltest CLI: run AI-skill tests and natural-language evals from Python, with a typed report contract.
5
5
  Author: Nick DeRobertis
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "skilltest-sdk"
3
- version = "0.2.2"
3
+ version = "0.4.0"
4
4
  description = "Python SDK for the skilltest CLI: run AI-skill tests and natural-language evals from Python, with a typed report contract."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
@@ -24,6 +24,7 @@ from .models import (
24
24
  NumericDetail,
25
25
  Report,
26
26
  Summary,
27
+ ToolEvent,
27
28
  Transcript,
28
29
  Usage,
29
30
  ValidationFinding,
@@ -32,8 +33,10 @@ from .models import (
32
33
  describe_failures,
33
34
  failed_evals,
34
35
  failed_runs,
36
+ tool_calls,
35
37
  )
36
38
  from .runner import ENV_BIN, ENV_PROVIDER, run_skill, validate_skill
39
+ from .stream import SkillStream, StreamEvent, stream_skill
37
40
 
38
41
  __all__ = [
39
42
  "ENV_BIN",
@@ -44,10 +47,13 @@ __all__ = [
44
47
  "Message",
45
48
  "NumericDetail",
46
49
  "Report",
50
+ "SkillStream",
47
51
  "SkilltestError",
48
52
  "SkilltestProviderError",
49
53
  "SkilltestUsageError",
54
+ "StreamEvent",
50
55
  "Summary",
56
+ "ToolEvent",
51
57
  "Transcript",
52
58
  "Usage",
53
59
  "ValidationFinding",
@@ -57,5 +63,7 @@ __all__ = [
57
63
  "failed_evals",
58
64
  "failed_runs",
59
65
  "run_skill",
66
+ "stream_skill",
67
+ "tool_calls",
60
68
  "validate_skill",
61
69
  ]
@@ -2,7 +2,7 @@
2
2
  # filename: report.schema.json
3
3
 
4
4
  from __future__ import annotations
5
- from typing import Literal
5
+ from typing import Any, Literal
6
6
  from pydantic import BaseModel, Field
7
7
 
8
8
 
@@ -46,6 +46,41 @@ class EvalOutcome(BaseModel):
46
46
  reason: str = Field(..., description="The judge's stated reason.")
47
47
 
48
48
 
49
+ class ToolEvent(BaseModel):
50
+ """
51
+ One normalized tool-call / action event the skill took during a turn, lifted
52
+ from oneharness's `events` array (its `--events` output). Harness-agnostic, so
53
+ a consumer can inspect *what the skill did* — shell commands, file edits, tool
54
+ uses — across any harness, not just the final text. Mirrors the oneharness
55
+ action-event shape; `input` is the structured, tool-shaped args so a consumer
56
+ can match on the command string or file path without re-parsing.
57
+
58
+ `input` is a free-form JSON value, so `Message`/`Transcript` are `PartialEq`
59
+ but not `Eq`.
60
+ """
61
+
62
+ index: int | None = Field(
63
+ 0,
64
+ description='Position within the turn, so ordering ("did X before Y") is expressible.',
65
+ ge=0,
66
+ )
67
+ input: Any | None = Field(
68
+ None,
69
+ description="Structured tool arguments (the command, the file path); `null` when none.",
70
+ )
71
+ kind: str = Field(
72
+ ...,
73
+ description="`tool_call` (the skill invoked a tool) or `tool_result` (the observation).",
74
+ )
75
+ name: str | None = Field(
76
+ None,
77
+ description="Normalized tool name where knowable (e.g. `bash`, `edit_file`); `null` for\na `tool_result` or when the harness did not name it.",
78
+ )
79
+ output: str | None = Field(
80
+ None, description="The result/observation text, when the transcript exposed it."
81
+ )
82
+
83
+
49
84
  class Usage(BaseModel):
50
85
  """
51
86
  Token / cost usage for one provider call.
@@ -67,6 +102,10 @@ class Message(BaseModel):
67
102
  """
68
103
 
69
104
  content: str
105
+ events: list[ToolEvent] | None = Field(
106
+ None,
107
+ description="The normalized tool events the skill took producing this turn (assistant\nturns only, and only when the harness exposed a tool transcript via\noneharness `--events`). Empty otherwise. Surfaced for post-hoc analysis\nand streamed live for short-circuiting.",
108
+ )
70
109
  role: Literal["user", "assistant", "system"] = Field(..., description="Who produced a message.")
71
110
 
72
111
 
@@ -19,6 +19,7 @@ from ._report import (
19
19
  NumericDetail,
20
20
  Report,
21
21
  Summary,
22
+ ToolEvent,
22
23
  Transcript,
23
24
  Usage,
24
25
  )
@@ -32,6 +33,7 @@ __all__ = [
32
33
  "NumericDetail",
33
34
  "Report",
34
35
  "Summary",
36
+ "ToolEvent",
35
37
  "Transcript",
36
38
  "Usage",
37
39
  "ValidationFinding",
@@ -40,6 +42,7 @@ __all__ = [
40
42
  "describe_failures",
41
43
  "failed_evals",
42
44
  "failed_runs",
45
+ "tool_calls",
43
46
  ]
44
47
 
45
48
 
@@ -48,6 +51,21 @@ def assistant_text(transcript: Transcript) -> str:
48
51
  return "\n".join(m.content for m in transcript.messages if m.role == "assistant")
49
52
 
50
53
 
54
+ def tool_calls(transcript: Transcript) -> list[ToolEvent]:
55
+ """Every ``tool_call`` event across the transcript's assistant turns, in order.
56
+
57
+ The normalized tool events the skill took (shell commands, file edits, tool
58
+ uses), lifted from oneharness ``--events`` — for asserting on *what the skill
59
+ did*, not just what it said. Empty for harnesses that expose no transcript.
60
+ """
61
+ return [
62
+ event
63
+ for message in transcript.messages
64
+ for event in (message.events or [])
65
+ if event.kind == "tool_call"
66
+ ]
67
+
68
+
51
69
  def failed_evals(run: CaseRun) -> list[EvalOutcome]:
52
70
  """The evals of a run that did not pass."""
53
71
  return [e for e in run.evals if not e.passed]
@@ -104,10 +104,41 @@ def run_skill(
104
104
  the caller can assert and inspect. Only bad input ([`SkilltestUsageError`])
105
105
  and provider failures ([`SkilltestProviderError`]) raise.
106
106
  """
107
+ argv = build_run_argv(
108
+ case,
109
+ bin=bin,
110
+ provider=provider,
111
+ platforms=platforms,
112
+ models=models,
113
+ judge_model=judge_model,
114
+ max_turns=max_turns,
115
+ config=config,
116
+ fmt="json",
117
+ )
118
+ proc = _run(argv, cwd)
119
+ _raise_for_status(proc)
120
+ return _parse(Report, proc.stdout)
121
+
122
+
123
+ def build_run_argv(
124
+ case: str | Path,
125
+ *,
126
+ bin: str | Path | None,
127
+ provider: str | Sequence[str] | None,
128
+ platforms: Sequence[str],
129
+ models: Sequence[str],
130
+ judge_model: str | None,
131
+ max_turns: int | None,
132
+ config: str | Path | None,
133
+ fmt: str,
134
+ ) -> list[str]:
135
+ """Build the ``skilltest run`` argv for output format ``fmt`` (``json`` for the
136
+ buffered API, ``json-stream`` for the streaming API). Internal, shared by
137
+ ``run_skill`` and the streaming API."""
107
138
  argv = [_resolve_bin(bin)]
108
139
  if config is not None:
109
140
  argv += ["--config", str(config)]
110
- argv += ["run", str(case), "--format", "json"]
141
+ argv += ["run", str(case), "--format", fmt]
111
142
 
112
143
  resolved_provider = _resolve_provider(provider)
113
144
  if resolved_provider is not None:
@@ -120,10 +151,7 @@ def run_skill(
120
151
  argv += ["--judge-model", judge_model]
121
152
  if max_turns is not None:
122
153
  argv += ["--max-turns", str(max_turns)]
123
-
124
- proc = _run(argv, cwd)
125
- _raise_for_status(proc)
126
- return _parse(Report, proc.stdout)
154
+ return argv
127
155
 
128
156
 
129
157
  def validate_skill(
@@ -140,14 +168,19 @@ def validate_skill(
140
168
 
141
169
 
142
170
  def _raise_for_status(proc: subprocess.CompletedProcess[str]) -> None:
143
- if proc.returncode in _REPORTING_CODES:
171
+ raise_for_code(proc.returncode, proc.stderr.strip() or proc.stdout.strip())
172
+
173
+
174
+ def raise_for_code(code: int | None, detail: str) -> None:
175
+ """Map a skilltest exit code to an exception (shared by the buffered and
176
+ streaming APIs). Codes 0/1 produce a report and never raise."""
177
+ if code in _REPORTING_CODES:
144
178
  return
145
- detail = proc.stderr.strip() or proc.stdout.strip()
146
- if proc.returncode == 2:
179
+ if code == 2:
147
180
  raise SkilltestUsageError(detail)
148
- if proc.returncode == 3:
181
+ if code == 3:
149
182
  raise SkilltestProviderError(detail)
150
- raise SkilltestError(f"skilltest exited {proc.returncode}: {detail}")
183
+ raise SkilltestError(f"skilltest exited {code}: {detail}")
151
184
 
152
185
 
153
186
  def _parse[T: BaseModel](model: type[T], stdout: str) -> T:
@@ -0,0 +1,128 @@
1
+ """Stream a running skilltest case as an async iterator of tool events.
2
+
3
+ Opt-in on top of the buffered [`run_skill`][skilltest_sdk.runner.run_skill]:
4
+ iterate a [`SkillStream`][skilltest_sdk.stream.SkillStream] with ``async for`` to
5
+ receive each normalized tool event the instant the skill takes it, and ``break``
6
+ to **short-circuit** — the CLI subprocess is killed, which closes oneharness's
7
+ stream and tears the harness down, so a bad turn is cut off instead of paid for
8
+ in full. After the stream completes normally, ``.report`` holds the final
9
+ [`Report`].
10
+
11
+ ```python
12
+ stream = stream_skill("cases/edit.skilltest.yaml")
13
+ async for ev in stream:
14
+ if ev.event.name == "bash" and "rm -rf" in str(ev.event.input):
15
+ break # disallowed action — abort the run now
16
+ report = stream.report # the full Report, when the stream ran to completion
17
+ ```
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import asyncio
23
+ import contextlib
24
+ import json
25
+ from collections.abc import AsyncIterator, Sequence
26
+ from pathlib import Path
27
+
28
+ from pydantic import BaseModel
29
+
30
+ from ._report import ToolEvent
31
+ from .errors import SkilltestProviderError
32
+ from .models import Report
33
+ from .runner import ENV_BIN, build_run_argv, raise_for_code
34
+
35
+
36
+ class StreamEvent(BaseModel):
37
+ """One streamed tool event, tagged with the run it belongs to."""
38
+
39
+ case: str
40
+ platform: str
41
+ model: str
42
+ #: 1-based assistant-turn index within the run.
43
+ turn: int
44
+ event: ToolEvent
45
+
46
+
47
+ class SkillStream:
48
+ """An async stream of [`StreamEvent`][skilltest_sdk.stream.StreamEvent]s from
49
+ a running case. Iterate with ``async for``; ``break`` to short-circuit. When
50
+ the stream runs to completion, ``report`` holds the final [`Report`]."""
51
+
52
+ def __init__(self, argv: list[str], cwd: str | None) -> None:
53
+ self._argv = argv
54
+ self._cwd = cwd
55
+ #: The final report, populated once the stream completes normally.
56
+ self.report: Report | None = None
57
+
58
+ def __aiter__(self) -> AsyncIterator[StreamEvent]:
59
+ return self._iterate()
60
+
61
+ async def _iterate(self) -> AsyncIterator[StreamEvent]:
62
+ try:
63
+ proc = await asyncio.create_subprocess_exec(
64
+ *self._argv,
65
+ stdout=asyncio.subprocess.PIPE,
66
+ stderr=asyncio.subprocess.PIPE,
67
+ cwd=self._cwd,
68
+ )
69
+ except FileNotFoundError as exc:
70
+ raise SkilltestProviderError(
71
+ f"could not run skilltest binary `{self._argv[0]}`: {exc}. "
72
+ f"Set {ENV_BIN} or pass bin=..."
73
+ ) from exc
74
+
75
+ assert proc.stdout is not None
76
+ try:
77
+ async for raw in proc.stdout:
78
+ line = raw.decode().strip()
79
+ if not line:
80
+ continue
81
+ obj = json.loads(line)
82
+ kind = obj.get("type")
83
+ if kind == "event":
84
+ yield StreamEvent.model_validate(obj)
85
+ elif kind == "result":
86
+ self.report = Report.model_validate(obj["report"])
87
+ await proc.wait()
88
+ # A hard failure (bad input / provider error) once the stream ends.
89
+ detail = ""
90
+ if proc.stderr is not None:
91
+ detail = (await proc.stderr.read()).decode().strip()
92
+ raise_for_code(proc.returncode, detail)
93
+ finally:
94
+ # The consumer stopped early (break): kill the CLI so oneharness's
95
+ # stream closes and the harness is torn down.
96
+ if proc.returncode is None:
97
+ proc.kill()
98
+ with contextlib.suppress(ProcessLookupError):
99
+ await proc.wait()
100
+
101
+
102
+ def stream_skill(
103
+ case: str | Path,
104
+ *,
105
+ bin: str | Path | None = None,
106
+ provider: str | Sequence[str] | None = None,
107
+ platforms: Sequence[str] = (),
108
+ models: Sequence[str] = (),
109
+ judge_model: str | None = None,
110
+ max_turns: int | None = None,
111
+ config: str | Path | None = None,
112
+ cwd: str | Path | None = None,
113
+ ) -> SkillStream:
114
+ """Start a streaming run and return a [`SkillStream`] to iterate. Same
115
+ arguments as [`run_skill`][skilltest_sdk.runner.run_skill]; the run does not
116
+ begin until iteration starts."""
117
+ argv = build_run_argv(
118
+ case,
119
+ bin=bin,
120
+ provider=provider,
121
+ platforms=platforms,
122
+ models=models,
123
+ judge_model=judge_model,
124
+ max_turns=max_turns,
125
+ config=config,
126
+ fmt="json-stream",
127
+ )
128
+ return SkillStream(argv, str(cwd) if cwd is not None else None)
@@ -14,6 +14,7 @@ from skilltest_sdk import (
14
14
  describe_failures,
15
15
  failed_evals,
16
16
  run_skill,
17
+ tool_calls,
17
18
  validate_skill,
18
19
  )
19
20
 
@@ -26,6 +27,13 @@ def test_happy_path_passes_and_exposes_transcript(cases: Path) -> None:
26
27
  assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
27
28
 
28
29
 
30
+ def test_tool_calls_are_exposed_for_analysis(cases: Path) -> None:
31
+ report = run_skill(cases / "tool_events.yaml")
32
+ calls = tool_calls(report.runs[0].transcript)
33
+ assert [c.name for c in calls] == ["edit_file", "bash"]
34
+ assert calls[1].input == {"command": 'git commit -m "update config"'}
35
+
36
+
29
37
  def test_numeric_eval_detail_is_typed(cases: Path) -> None:
30
38
  report = run_skill(cases / "greet_numeric.yaml")
31
39
  assert report.passed
@@ -0,0 +1,37 @@
1
+ """Streaming-API e2e tests against the built binary + fake provider."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ from pathlib import Path
7
+
8
+ from skilltest_sdk import Report, StreamEvent, stream_skill
9
+
10
+
11
+ def test_stream_yields_events_then_exposes_report(cases: Path) -> None:
12
+ async def go() -> tuple[list[str | None], Report | None]:
13
+ stream = stream_skill(cases / "tool_events.yaml")
14
+ names: list[str | None] = []
15
+ async for ev in stream:
16
+ assert isinstance(ev, StreamEvent)
17
+ assert ev.case == "tool_events"
18
+ assert ev.turn == 1
19
+ names.append(ev.event.name)
20
+ return names, stream.report
21
+
22
+ names, report = asyncio.run(go())
23
+ assert names == ["edit_file", "bash"]
24
+ assert report is not None
25
+ assert report.passed
26
+
27
+
28
+ def test_stream_short_circuits_on_break(cases: Path) -> None:
29
+ async def go() -> int:
30
+ stream = stream_skill(cases / "tool_events.yaml")
31
+ seen = 0
32
+ async for _ in stream:
33
+ seen += 1
34
+ break # abort after the first event
35
+ return seen
36
+
37
+ assert asyncio.run(go()) == 1
@@ -482,7 +482,7 @@ wheels = [
482
482
 
483
483
  [[package]]
484
484
  name = "skilltest-sdk"
485
- version = "0.2.2"
485
+ version = "0.4.0"
486
486
  source = { editable = "." }
487
487
  dependencies = [
488
488
  { name = "pydantic" },
File without changes
File without changes