skilltest-sdk 0.2.2__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/PKG-INFO +1 -1
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/pyproject.toml +1 -1
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/__init__.py +8 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/_report.py +40 -1
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/models.py +18 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/runner.py +43 -10
- skilltest_sdk-0.4.0/skilltest_sdk/stream.py +128 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/tests/test_api.py +8 -0
- skilltest_sdk-0.4.0/tests/test_stream.py +37 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/uv.lock +1 -1
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/.gitignore +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/README.md +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/project.json +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/_validation.py +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/skilltest_sdk/errors.py +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/tests/conftest.py +0 -0
- {skilltest_sdk-0.2.2 → skilltest_sdk-0.4.0}/tests/test_resolve.py +0 -0
|
@@ -24,6 +24,7 @@ from .models import (
|
|
|
24
24
|
NumericDetail,
|
|
25
25
|
Report,
|
|
26
26
|
Summary,
|
|
27
|
+
ToolEvent,
|
|
27
28
|
Transcript,
|
|
28
29
|
Usage,
|
|
29
30
|
ValidationFinding,
|
|
@@ -32,8 +33,10 @@ from .models import (
|
|
|
32
33
|
describe_failures,
|
|
33
34
|
failed_evals,
|
|
34
35
|
failed_runs,
|
|
36
|
+
tool_calls,
|
|
35
37
|
)
|
|
36
38
|
from .runner import ENV_BIN, ENV_PROVIDER, run_skill, validate_skill
|
|
39
|
+
from .stream import SkillStream, StreamEvent, stream_skill
|
|
37
40
|
|
|
38
41
|
__all__ = [
|
|
39
42
|
"ENV_BIN",
|
|
@@ -44,10 +47,13 @@ __all__ = [
|
|
|
44
47
|
"Message",
|
|
45
48
|
"NumericDetail",
|
|
46
49
|
"Report",
|
|
50
|
+
"SkillStream",
|
|
47
51
|
"SkilltestError",
|
|
48
52
|
"SkilltestProviderError",
|
|
49
53
|
"SkilltestUsageError",
|
|
54
|
+
"StreamEvent",
|
|
50
55
|
"Summary",
|
|
56
|
+
"ToolEvent",
|
|
51
57
|
"Transcript",
|
|
52
58
|
"Usage",
|
|
53
59
|
"ValidationFinding",
|
|
@@ -57,5 +63,7 @@ __all__ = [
|
|
|
57
63
|
"failed_evals",
|
|
58
64
|
"failed_runs",
|
|
59
65
|
"run_skill",
|
|
66
|
+
"stream_skill",
|
|
67
|
+
"tool_calls",
|
|
60
68
|
"validate_skill",
|
|
61
69
|
]
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# filename: report.schema.json
|
|
3
3
|
|
|
4
4
|
from __future__ import annotations
|
|
5
|
-
from typing import Literal
|
|
5
|
+
from typing import Any, Literal
|
|
6
6
|
from pydantic import BaseModel, Field
|
|
7
7
|
|
|
8
8
|
|
|
@@ -46,6 +46,41 @@ class EvalOutcome(BaseModel):
|
|
|
46
46
|
reason: str = Field(..., description="The judge's stated reason.")
|
|
47
47
|
|
|
48
48
|
|
|
49
|
+
class ToolEvent(BaseModel):
|
|
50
|
+
"""
|
|
51
|
+
One normalized tool-call / action event the skill took during a turn, lifted
|
|
52
|
+
from oneharness's `events` array (its `--events` output). Harness-agnostic, so
|
|
53
|
+
a consumer can inspect *what the skill did* — shell commands, file edits, tool
|
|
54
|
+
uses — across any harness, not just the final text. Mirrors the oneharness
|
|
55
|
+
action-event shape; `input` is the structured, tool-shaped args so a consumer
|
|
56
|
+
can match on the command string or file path without re-parsing.
|
|
57
|
+
|
|
58
|
+
`input` is a free-form JSON value, so `Message`/`Transcript` are `PartialEq`
|
|
59
|
+
but not `Eq`.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
index: int | None = Field(
|
|
63
|
+
0,
|
|
64
|
+
description='Position within the turn, so ordering ("did X before Y") is expressible.',
|
|
65
|
+
ge=0,
|
|
66
|
+
)
|
|
67
|
+
input: Any | None = Field(
|
|
68
|
+
None,
|
|
69
|
+
description="Structured tool arguments (the command, the file path); `null` when none.",
|
|
70
|
+
)
|
|
71
|
+
kind: str = Field(
|
|
72
|
+
...,
|
|
73
|
+
description="`tool_call` (the skill invoked a tool) or `tool_result` (the observation).",
|
|
74
|
+
)
|
|
75
|
+
name: str | None = Field(
|
|
76
|
+
None,
|
|
77
|
+
description="Normalized tool name where knowable (e.g. `bash`, `edit_file`); `null` for\na `tool_result` or when the harness did not name it.",
|
|
78
|
+
)
|
|
79
|
+
output: str | None = Field(
|
|
80
|
+
None, description="The result/observation text, when the transcript exposed it."
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
49
84
|
class Usage(BaseModel):
|
|
50
85
|
"""
|
|
51
86
|
Token / cost usage for one provider call.
|
|
@@ -67,6 +102,10 @@ class Message(BaseModel):
|
|
|
67
102
|
"""
|
|
68
103
|
|
|
69
104
|
content: str
|
|
105
|
+
events: list[ToolEvent] | None = Field(
|
|
106
|
+
None,
|
|
107
|
+
description="The normalized tool events the skill took producing this turn (assistant\nturns only, and only when the harness exposed a tool transcript via\noneharness `--events`). Empty otherwise. Surfaced for post-hoc analysis\nand streamed live for short-circuiting.",
|
|
108
|
+
)
|
|
70
109
|
role: Literal["user", "assistant", "system"] = Field(..., description="Who produced a message.")
|
|
71
110
|
|
|
72
111
|
|
|
@@ -19,6 +19,7 @@ from ._report import (
|
|
|
19
19
|
NumericDetail,
|
|
20
20
|
Report,
|
|
21
21
|
Summary,
|
|
22
|
+
ToolEvent,
|
|
22
23
|
Transcript,
|
|
23
24
|
Usage,
|
|
24
25
|
)
|
|
@@ -32,6 +33,7 @@ __all__ = [
|
|
|
32
33
|
"NumericDetail",
|
|
33
34
|
"Report",
|
|
34
35
|
"Summary",
|
|
36
|
+
"ToolEvent",
|
|
35
37
|
"Transcript",
|
|
36
38
|
"Usage",
|
|
37
39
|
"ValidationFinding",
|
|
@@ -40,6 +42,7 @@ __all__ = [
|
|
|
40
42
|
"describe_failures",
|
|
41
43
|
"failed_evals",
|
|
42
44
|
"failed_runs",
|
|
45
|
+
"tool_calls",
|
|
43
46
|
]
|
|
44
47
|
|
|
45
48
|
|
|
@@ -48,6 +51,21 @@ def assistant_text(transcript: Transcript) -> str:
|
|
|
48
51
|
return "\n".join(m.content for m in transcript.messages if m.role == "assistant")
|
|
49
52
|
|
|
50
53
|
|
|
54
|
+
def tool_calls(transcript: Transcript) -> list[ToolEvent]:
|
|
55
|
+
"""Every ``tool_call`` event across the transcript's assistant turns, in order.
|
|
56
|
+
|
|
57
|
+
The normalized tool events the skill took (shell commands, file edits, tool
|
|
58
|
+
uses), lifted from oneharness ``--events`` — for asserting on *what the skill
|
|
59
|
+
did*, not just what it said. Empty for harnesses that expose no transcript.
|
|
60
|
+
"""
|
|
61
|
+
return [
|
|
62
|
+
event
|
|
63
|
+
for message in transcript.messages
|
|
64
|
+
for event in (message.events or [])
|
|
65
|
+
if event.kind == "tool_call"
|
|
66
|
+
]
|
|
67
|
+
|
|
68
|
+
|
|
51
69
|
def failed_evals(run: CaseRun) -> list[EvalOutcome]:
|
|
52
70
|
"""The evals of a run that did not pass."""
|
|
53
71
|
return [e for e in run.evals if not e.passed]
|
|
@@ -104,10 +104,41 @@ def run_skill(
|
|
|
104
104
|
the caller can assert and inspect. Only bad input ([`SkilltestUsageError`])
|
|
105
105
|
and provider failures ([`SkilltestProviderError`]) raise.
|
|
106
106
|
"""
|
|
107
|
+
argv = build_run_argv(
|
|
108
|
+
case,
|
|
109
|
+
bin=bin,
|
|
110
|
+
provider=provider,
|
|
111
|
+
platforms=platforms,
|
|
112
|
+
models=models,
|
|
113
|
+
judge_model=judge_model,
|
|
114
|
+
max_turns=max_turns,
|
|
115
|
+
config=config,
|
|
116
|
+
fmt="json",
|
|
117
|
+
)
|
|
118
|
+
proc = _run(argv, cwd)
|
|
119
|
+
_raise_for_status(proc)
|
|
120
|
+
return _parse(Report, proc.stdout)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def build_run_argv(
|
|
124
|
+
case: str | Path,
|
|
125
|
+
*,
|
|
126
|
+
bin: str | Path | None,
|
|
127
|
+
provider: str | Sequence[str] | None,
|
|
128
|
+
platforms: Sequence[str],
|
|
129
|
+
models: Sequence[str],
|
|
130
|
+
judge_model: str | None,
|
|
131
|
+
max_turns: int | None,
|
|
132
|
+
config: str | Path | None,
|
|
133
|
+
fmt: str,
|
|
134
|
+
) -> list[str]:
|
|
135
|
+
"""Build the ``skilltest run`` argv for output format ``fmt`` (``json`` for the
|
|
136
|
+
buffered API, ``json-stream`` for the streaming API). Internal, shared by
|
|
137
|
+
``run_skill`` and the streaming API."""
|
|
107
138
|
argv = [_resolve_bin(bin)]
|
|
108
139
|
if config is not None:
|
|
109
140
|
argv += ["--config", str(config)]
|
|
110
|
-
argv += ["run", str(case), "--format",
|
|
141
|
+
argv += ["run", str(case), "--format", fmt]
|
|
111
142
|
|
|
112
143
|
resolved_provider = _resolve_provider(provider)
|
|
113
144
|
if resolved_provider is not None:
|
|
@@ -120,10 +151,7 @@ def run_skill(
|
|
|
120
151
|
argv += ["--judge-model", judge_model]
|
|
121
152
|
if max_turns is not None:
|
|
122
153
|
argv += ["--max-turns", str(max_turns)]
|
|
123
|
-
|
|
124
|
-
proc = _run(argv, cwd)
|
|
125
|
-
_raise_for_status(proc)
|
|
126
|
-
return _parse(Report, proc.stdout)
|
|
154
|
+
return argv
|
|
127
155
|
|
|
128
156
|
|
|
129
157
|
def validate_skill(
|
|
@@ -140,14 +168,19 @@ def validate_skill(
|
|
|
140
168
|
|
|
141
169
|
|
|
142
170
|
def _raise_for_status(proc: subprocess.CompletedProcess[str]) -> None:
|
|
143
|
-
|
|
171
|
+
raise_for_code(proc.returncode, proc.stderr.strip() or proc.stdout.strip())
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def raise_for_code(code: int | None, detail: str) -> None:
|
|
175
|
+
"""Map a skilltest exit code to an exception (shared by the buffered and
|
|
176
|
+
streaming APIs). Codes 0/1 produce a report and never raise."""
|
|
177
|
+
if code in _REPORTING_CODES:
|
|
144
178
|
return
|
|
145
|
-
|
|
146
|
-
if proc.returncode == 2:
|
|
179
|
+
if code == 2:
|
|
147
180
|
raise SkilltestUsageError(detail)
|
|
148
|
-
if
|
|
181
|
+
if code == 3:
|
|
149
182
|
raise SkilltestProviderError(detail)
|
|
150
|
-
raise SkilltestError(f"skilltest exited {
|
|
183
|
+
raise SkilltestError(f"skilltest exited {code}: {detail}")
|
|
151
184
|
|
|
152
185
|
|
|
153
186
|
def _parse[T: BaseModel](model: type[T], stdout: str) -> T:
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Stream a running skilltest case as an async iterator of tool events.
|
|
2
|
+
|
|
3
|
+
Opt-in on top of the buffered [`run_skill`][skilltest_sdk.runner.run_skill]:
|
|
4
|
+
iterate a [`SkillStream`][skilltest_sdk.stream.SkillStream] with ``async for`` to
|
|
5
|
+
receive each normalized tool event the instant the skill takes it, and ``break``
|
|
6
|
+
to **short-circuit** — the CLI subprocess is killed, which closes oneharness's
|
|
7
|
+
stream and tears the harness down, so a bad turn is cut off instead of paid for
|
|
8
|
+
in full. After the stream completes normally, ``.report`` holds the final
|
|
9
|
+
[`Report`].
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
stream = stream_skill("cases/edit.skilltest.yaml")
|
|
13
|
+
async for ev in stream:
|
|
14
|
+
if ev.event.name == "bash" and "rm -rf" in str(ev.event.input):
|
|
15
|
+
break # disallowed action — abort the run now
|
|
16
|
+
report = stream.report # the full Report, when the stream ran to completion
|
|
17
|
+
```
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import asyncio
|
|
23
|
+
import contextlib
|
|
24
|
+
import json
|
|
25
|
+
from collections.abc import AsyncIterator, Sequence
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
from pydantic import BaseModel
|
|
29
|
+
|
|
30
|
+
from ._report import ToolEvent
|
|
31
|
+
from .errors import SkilltestProviderError
|
|
32
|
+
from .models import Report
|
|
33
|
+
from .runner import ENV_BIN, build_run_argv, raise_for_code
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class StreamEvent(BaseModel):
|
|
37
|
+
"""One streamed tool event, tagged with the run it belongs to."""
|
|
38
|
+
|
|
39
|
+
case: str
|
|
40
|
+
platform: str
|
|
41
|
+
model: str
|
|
42
|
+
#: 1-based assistant-turn index within the run.
|
|
43
|
+
turn: int
|
|
44
|
+
event: ToolEvent
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class SkillStream:
|
|
48
|
+
"""An async stream of [`StreamEvent`][skilltest_sdk.stream.StreamEvent]s from
|
|
49
|
+
a running case. Iterate with ``async for``; ``break`` to short-circuit. When
|
|
50
|
+
the stream runs to completion, ``report`` holds the final [`Report`]."""
|
|
51
|
+
|
|
52
|
+
def __init__(self, argv: list[str], cwd: str | None) -> None:
|
|
53
|
+
self._argv = argv
|
|
54
|
+
self._cwd = cwd
|
|
55
|
+
#: The final report, populated once the stream completes normally.
|
|
56
|
+
self.report: Report | None = None
|
|
57
|
+
|
|
58
|
+
def __aiter__(self) -> AsyncIterator[StreamEvent]:
|
|
59
|
+
return self._iterate()
|
|
60
|
+
|
|
61
|
+
async def _iterate(self) -> AsyncIterator[StreamEvent]:
|
|
62
|
+
try:
|
|
63
|
+
proc = await asyncio.create_subprocess_exec(
|
|
64
|
+
*self._argv,
|
|
65
|
+
stdout=asyncio.subprocess.PIPE,
|
|
66
|
+
stderr=asyncio.subprocess.PIPE,
|
|
67
|
+
cwd=self._cwd,
|
|
68
|
+
)
|
|
69
|
+
except FileNotFoundError as exc:
|
|
70
|
+
raise SkilltestProviderError(
|
|
71
|
+
f"could not run skilltest binary `{self._argv[0]}`: {exc}. "
|
|
72
|
+
f"Set {ENV_BIN} or pass bin=..."
|
|
73
|
+
) from exc
|
|
74
|
+
|
|
75
|
+
assert proc.stdout is not None
|
|
76
|
+
try:
|
|
77
|
+
async for raw in proc.stdout:
|
|
78
|
+
line = raw.decode().strip()
|
|
79
|
+
if not line:
|
|
80
|
+
continue
|
|
81
|
+
obj = json.loads(line)
|
|
82
|
+
kind = obj.get("type")
|
|
83
|
+
if kind == "event":
|
|
84
|
+
yield StreamEvent.model_validate(obj)
|
|
85
|
+
elif kind == "result":
|
|
86
|
+
self.report = Report.model_validate(obj["report"])
|
|
87
|
+
await proc.wait()
|
|
88
|
+
# A hard failure (bad input / provider error) once the stream ends.
|
|
89
|
+
detail = ""
|
|
90
|
+
if proc.stderr is not None:
|
|
91
|
+
detail = (await proc.stderr.read()).decode().strip()
|
|
92
|
+
raise_for_code(proc.returncode, detail)
|
|
93
|
+
finally:
|
|
94
|
+
# The consumer stopped early (break): kill the CLI so oneharness's
|
|
95
|
+
# stream closes and the harness is torn down.
|
|
96
|
+
if proc.returncode is None:
|
|
97
|
+
proc.kill()
|
|
98
|
+
with contextlib.suppress(ProcessLookupError):
|
|
99
|
+
await proc.wait()
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def stream_skill(
|
|
103
|
+
case: str | Path,
|
|
104
|
+
*,
|
|
105
|
+
bin: str | Path | None = None,
|
|
106
|
+
provider: str | Sequence[str] | None = None,
|
|
107
|
+
platforms: Sequence[str] = (),
|
|
108
|
+
models: Sequence[str] = (),
|
|
109
|
+
judge_model: str | None = None,
|
|
110
|
+
max_turns: int | None = None,
|
|
111
|
+
config: str | Path | None = None,
|
|
112
|
+
cwd: str | Path | None = None,
|
|
113
|
+
) -> SkillStream:
|
|
114
|
+
"""Start a streaming run and return a [`SkillStream`] to iterate. Same
|
|
115
|
+
arguments as [`run_skill`][skilltest_sdk.runner.run_skill]; the run does not
|
|
116
|
+
begin until iteration starts."""
|
|
117
|
+
argv = build_run_argv(
|
|
118
|
+
case,
|
|
119
|
+
bin=bin,
|
|
120
|
+
provider=provider,
|
|
121
|
+
platforms=platforms,
|
|
122
|
+
models=models,
|
|
123
|
+
judge_model=judge_model,
|
|
124
|
+
max_turns=max_turns,
|
|
125
|
+
config=config,
|
|
126
|
+
fmt="json-stream",
|
|
127
|
+
)
|
|
128
|
+
return SkillStream(argv, str(cwd) if cwd is not None else None)
|
|
@@ -14,6 +14,7 @@ from skilltest_sdk import (
|
|
|
14
14
|
describe_failures,
|
|
15
15
|
failed_evals,
|
|
16
16
|
run_skill,
|
|
17
|
+
tool_calls,
|
|
17
18
|
validate_skill,
|
|
18
19
|
)
|
|
19
20
|
|
|
@@ -26,6 +27,13 @@ def test_happy_path_passes_and_exposes_transcript(cases: Path) -> None:
|
|
|
26
27
|
assert "Dr. Smith" in assistant_text(report.runs[0].transcript)
|
|
27
28
|
|
|
28
29
|
|
|
30
|
+
def test_tool_calls_are_exposed_for_analysis(cases: Path) -> None:
|
|
31
|
+
report = run_skill(cases / "tool_events.yaml")
|
|
32
|
+
calls = tool_calls(report.runs[0].transcript)
|
|
33
|
+
assert [c.name for c in calls] == ["edit_file", "bash"]
|
|
34
|
+
assert calls[1].input == {"command": 'git commit -m "update config"'}
|
|
35
|
+
|
|
36
|
+
|
|
29
37
|
def test_numeric_eval_detail_is_typed(cases: Path) -> None:
|
|
30
38
|
report = run_skill(cases / "greet_numeric.yaml")
|
|
31
39
|
assert report.passed
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Streaming-API e2e tests against the built binary + fake provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from skilltest_sdk import Report, StreamEvent, stream_skill
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_stream_yields_events_then_exposes_report(cases: Path) -> None:
|
|
12
|
+
async def go() -> tuple[list[str | None], Report | None]:
|
|
13
|
+
stream = stream_skill(cases / "tool_events.yaml")
|
|
14
|
+
names: list[str | None] = []
|
|
15
|
+
async for ev in stream:
|
|
16
|
+
assert isinstance(ev, StreamEvent)
|
|
17
|
+
assert ev.case == "tool_events"
|
|
18
|
+
assert ev.turn == 1
|
|
19
|
+
names.append(ev.event.name)
|
|
20
|
+
return names, stream.report
|
|
21
|
+
|
|
22
|
+
names, report = asyncio.run(go())
|
|
23
|
+
assert names == ["edit_file", "bash"]
|
|
24
|
+
assert report is not None
|
|
25
|
+
assert report.passed
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_stream_short_circuits_on_break(cases: Path) -> None:
|
|
29
|
+
async def go() -> int:
|
|
30
|
+
stream = stream_skill(cases / "tool_events.yaml")
|
|
31
|
+
seen = 0
|
|
32
|
+
async for _ in stream:
|
|
33
|
+
seen += 1
|
|
34
|
+
break # abort after the first event
|
|
35
|
+
return seen
|
|
36
|
+
|
|
37
|
+
assert asyncio.run(go()) == 1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|