skilltest-sdk 0.12.1__py3-none-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,150 @@
1
+ """skilltest-sdk: the Python SDK for the ``skilltest`` CLI.
2
+
3
+ A thin, typed wrapper around the CLI and nothing else: run test cases, validate
4
+ skills, and get back models mirroring the ``--format json`` contract. The
5
+ models are generated from the CLI's own JSON Schemas (``just gen-contract``),
6
+ so they cannot drift from the binary. Test frameworks build on this —
7
+ ``skilltest-pytest`` adds pytest collection on top.
8
+
9
+ Define the whole case in code (the recommended form):
10
+
11
+ from skilltest_sdk import TestCase, run_skill, boolean, describe_failures
12
+
13
+ case = TestCase(
14
+ skill="skills/greeter",
15
+ input="Greet Dr. Smith, who has an appointment today.",
16
+ evals=[boolean("the reply greets Dr. Smith by name")],
17
+ )
18
+ report = run_skill(case)
19
+ assert report.passed, describe_failures(report)
20
+
21
+ ``run_skill`` also takes a path to an existing test-case YAML file (or a
22
+ directory of them), and the pytest plugin auto-discovers ``*.skilltest.yaml``.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from .case import (
28
+ Eval,
29
+ MockRefEval,
30
+ SimulatedUser,
31
+ TestCase,
32
+ boolean,
33
+ called,
34
+ not_called,
35
+ numeric,
36
+ user,
37
+ )
38
+ from .errors import (
39
+ ProviderErrorKind,
40
+ SkilltestAuthError,
41
+ SkilltestError,
42
+ SkilltestModelNotFoundError,
43
+ SkilltestOverloadedError,
44
+ SkilltestProtocolError,
45
+ SkilltestProviderError,
46
+ SkilltestQuotaError,
47
+ SkilltestRateLimitError,
48
+ SkilltestSpawnError,
49
+ SkilltestTimeoutError,
50
+ SkilltestUsageError,
51
+ )
52
+ from .mock import (
53
+ Matcher,
54
+ ToolCall,
55
+ ToolMock,
56
+ ToolSpy,
57
+ anything,
58
+ contains,
59
+ deny,
60
+ matching,
61
+ rewrite,
62
+ spy,
63
+ stub,
64
+ )
65
+ from .models import (
66
+ BooleanDetail,
67
+ CallsDetail,
68
+ CaseRun,
69
+ EvalOutcome,
70
+ Message,
71
+ MockCall,
72
+ NumericDetail,
73
+ Report,
74
+ ReportError,
75
+ Summary,
76
+ ToolEvent,
77
+ Transcript,
78
+ Usage,
79
+ ValidationFinding,
80
+ ValidationReport,
81
+ assistant_text,
82
+ describe_failures,
83
+ failed_evals,
84
+ failed_runs,
85
+ tool_calls,
86
+ )
87
+ from .runner import ENV_BIN, ENV_PROVIDER, run_skill, validate_skill
88
+ from .stream import SkillStream, StreamEvent, stream_skill
89
+
90
+ __all__ = [
91
+ "ENV_BIN",
92
+ "ENV_PROVIDER",
93
+ "BooleanDetail",
94
+ "CallsDetail",
95
+ "CaseRun",
96
+ "Eval",
97
+ "EvalOutcome",
98
+ "Matcher",
99
+ "Message",
100
+ "MockCall",
101
+ "MockRefEval",
102
+ "NumericDetail",
103
+ "ProviderErrorKind",
104
+ "Report",
105
+ "ReportError",
106
+ "SimulatedUser",
107
+ "SkillStream",
108
+ "SkilltestAuthError",
109
+ "SkilltestError",
110
+ "SkilltestModelNotFoundError",
111
+ "SkilltestOverloadedError",
112
+ "SkilltestProtocolError",
113
+ "SkilltestProviderError",
114
+ "SkilltestQuotaError",
115
+ "SkilltestRateLimitError",
116
+ "SkilltestSpawnError",
117
+ "SkilltestTimeoutError",
118
+ "SkilltestUsageError",
119
+ "StreamEvent",
120
+ "Summary",
121
+ "TestCase",
122
+ "ToolCall",
123
+ "ToolEvent",
124
+ "ToolMock",
125
+ "ToolSpy",
126
+ "Transcript",
127
+ "Usage",
128
+ "ValidationFinding",
129
+ "ValidationReport",
130
+ "anything",
131
+ "assistant_text",
132
+ "boolean",
133
+ "called",
134
+ "contains",
135
+ "deny",
136
+ "describe_failures",
137
+ "failed_evals",
138
+ "failed_runs",
139
+ "matching",
140
+ "not_called",
141
+ "numeric",
142
+ "rewrite",
143
+ "run_skill",
144
+ "spy",
145
+ "stream_skill",
146
+ "stub",
147
+ "tool_calls",
148
+ "user",
149
+ "validate_skill",
150
+ ]
Binary file
skilltest_sdk/_case.py ADDED
@@ -0,0 +1,251 @@
1
+ # generated by datamodel-codegen:
2
+ # filename: case.schema.json
3
+
4
+ from __future__ import annotations
5
+ from typing import Any, Literal
6
+ from pydantic import BaseModel, ConfigDict, Field, RootModel
7
+
8
+
9
+ class DenyMessage(BaseModel):
10
+ """
11
+ The explicit form of a [`DenySpec`]: the model-visible message.
12
+ """
13
+
14
+ model_config = ConfigDict(
15
+ extra="forbid",
16
+ )
17
+ message: str
18
+
19
+
20
+ class BooleanEval(BaseModel):
21
+ """
22
+ Assert a plain-English criterion holds (or, with `expected: false`, that
23
+ it does not).
24
+ """
25
+
26
+ model_config = ConfigDict(
27
+ extra="forbid",
28
+ )
29
+ criterion: str = Field(
30
+ ..., description="The criterion the judge evaluates against the transcript."
31
+ )
32
+ expected: bool | None = Field(
33
+ True, description="What the judge's verdict must equal to pass. Defaults to `true`."
34
+ )
35
+ name: str | None = Field(None, description="Optional human label for reports.")
36
+ type: Literal["boolean"]
37
+
38
+
39
+ class NumericEval(BaseModel):
40
+ """
41
+ Score a plain-English criterion on a numeric scale and compare it to a
42
+ threshold.
43
+ """
44
+
45
+ model_config = ConfigDict(
46
+ extra="forbid",
47
+ )
48
+ comparator: Literal["gte", "gt", "lte", "lt"] | None = Field(
49
+ "gte", description="How the score is compared to `threshold`. Defaults to `>=`."
50
+ )
51
+ criterion: str = Field(..., description="The criterion the judge scores.")
52
+ max: float = Field(..., description="Inclusive upper bound of the scale.")
53
+ min: float = Field(..., description="Inclusive lower bound of the scale.")
54
+ name: str | None = Field(None, description="Optional human label for reports.")
55
+ threshold: float = Field(..., description="The passing threshold.")
56
+ type: Literal["numeric"]
57
+
58
+
59
+ class FieldPredicateSpec(BaseModel):
60
+ """
61
+ The explicit form of a [`FieldPredicate`]; every given key must hold.
62
+ """
63
+
64
+ model_config = ConfigDict(
65
+ extra="forbid",
66
+ )
67
+ contains: str | None = None
68
+ equals: str | None = None
69
+ pattern: str | None = Field(
70
+ None, description="Unanchored regex (linear-time; same engine oneharness uses)."
71
+ )
72
+
73
+
74
+ class SimulatedUser(BaseModel):
75
+ """
76
+ The simulated-user block that turns a single-turn case into a multi-turn one.
77
+ When present, after each assistant turn the runner asks the provider to play
78
+ the user (guided by `persona`) until `done_when` holds or `max_turns` is hit.
79
+ """
80
+
81
+ model_config = ConfigDict(
82
+ extra="forbid",
83
+ )
84
+ done_when: str | None = Field(
85
+ None,
86
+ description="A plain-English condition; when the judge decides it holds, the\nconversation ends. Optional — without it the run ends at `max_turns` or\nwhen the skill reports itself done.",
87
+ )
88
+ max_turns: int | None = Field(
89
+ None, description="Per-case override of the global assistant-turn cap.", ge=0
90
+ )
91
+ persona: str = Field(
92
+ ..., description="Instructions describing how the simulated user should behave."
93
+ )
94
+
95
+
96
+ class StubOutput(BaseModel):
97
+ """
98
+ The explicit form of a [`StubSpec`]: the canned output plus an exit code.
99
+ """
100
+
101
+ model_config = ConfigDict(
102
+ extra="forbid",
103
+ )
104
+ exit_code: int | None = Field(0, description="Non-zero fakes a failing command. Default 0.")
105
+ output: str
106
+
107
+
108
+ class StubSpec1(RootModel[list[str | StubOutput]]):
109
+ root: list[str | StubOutput] = Field(
110
+ ...,
111
+ description="Ordered responses for successive intercepted calls; the last repeats.",
112
+ min_length=1,
113
+ )
114
+
115
+
116
+ class MockMatch(BaseModel):
117
+ """
118
+ What a mock/spy declaration matches on. At least one criterion is required;
119
+ every given criterion must hold (AND).
120
+ """
121
+
122
+ model_config = ConfigDict(
123
+ extra="forbid",
124
+ )
125
+ contains: str | None = Field(
126
+ None,
127
+ description="Substring match over the raw hook event JSON (the tool name and its\ninput always serialize into it). Note the haystack is JSON: quotes and\nbackslashes inside tool input appear escaped.",
128
+ )
129
+ input: dict[str, str | FieldPredicateSpec] | None = Field(
130
+ None,
131
+ description="Per-field predicates on the tool's input: the key is the argument name\n(`command`, `file_path`, …), and every listed field must match (a field\nabsent from the call fails the matcher). Non-string fields are compared\nagainst their compact JSON form.",
132
+ )
133
+ pattern: str | None = Field(
134
+ None,
135
+ description="Unanchored regex over the same haystack as `contains`, for non-exact\nneedles like `git push( --force)?`. Linear-time engine (no lookarounds).",
136
+ )
137
+ tool: str | None = Field(
138
+ None,
139
+ description="Case-insensitive exact match on the tool name. Tool names are\nper-harness (`Bash` on claude-code, `bash` on opencode/crush), so\ncross-harness matchers usually prefer `contains`/`pattern`/`input`.",
140
+ )
141
+
142
+
143
+ class CalledEval(BaseModel):
144
+ """
145
+ Deterministic: assert the referenced mock/spy observed at least one
146
+ matching call (or exactly `times`). Evaluated against the mock channel's
147
+ records, not by a judge.
148
+ """
149
+
150
+ model_config = ConfigDict(
151
+ extra="forbid",
152
+ )
153
+ mock: str = Field(
154
+ ..., description="The `mocks:` declaration (mock or spy) this asserts on, by name."
155
+ )
156
+ name: str | None = Field(None, description="Optional human label for reports.")
157
+ times: int | None = Field(
158
+ None, description='Exact required call count; absent means "at least once".', ge=0
159
+ )
160
+ type: Literal["called"]
161
+ where: dict[str, str | FieldPredicateSpec] | None = Field(
162
+ None, description="Optional per-field input predicates narrowing which calls count."
163
+ )
164
+
165
+
166
+ class NotCalledEval(BaseModel):
167
+ """
168
+ Deterministic: assert the referenced mock/spy observed **no** matching
169
+ call.
170
+ """
171
+
172
+ model_config = ConfigDict(
173
+ extra="forbid",
174
+ )
175
+ mock: str = Field(
176
+ ..., description="The `mocks:` declaration (mock or spy) this asserts on, by name."
177
+ )
178
+ name: str | None = Field(None, description="Optional human label for reports.")
179
+ type: Literal["not_called"]
180
+ where: dict[str, str | FieldPredicateSpec] | None = Field(
181
+ None, description="Optional per-field input predicates narrowing which calls count."
182
+ )
183
+
184
+
185
+ class MockDecl(BaseModel):
186
+ """
187
+ One mock or spy declaration — an entry of a test case's `mocks:` block, of
188
+ the CLI's `--mocks` file, or synthesized by an SDK. Exactly one of `stub` /
189
+ `deny` / `rewrite` makes it a **mock** (the call is intercepted); none makes
190
+ it a **spy** (observed only, matched locally against the returned records).
191
+ """
192
+
193
+ model_config = ConfigDict(
194
+ extra="forbid",
195
+ )
196
+ deny: str | DenyMessage | None = Field(
197
+ None, description="Block the call; the model reads the message as the tool's feedback."
198
+ )
199
+ match: MockMatch
200
+ name: str | None = Field(
201
+ None,
202
+ description="Name evals (`type: called` / `not_called`) reference this declaration\nby. Optional; unnamed declarations get a positional fallback\n(`mock_<i>` / `spy_<i>`) used in reports and error messages.",
203
+ )
204
+ rewrite: Any | None = Field(
205
+ None,
206
+ description="Substitute raw input fields (a JSON object) — the low-level escape\nhatch, and the way to mock file reads (rewrite `file_path` to a\nfixture).",
207
+ )
208
+ stub: str | StubOutput | StubSpec1 | None = Field(
209
+ None,
210
+ description="Fake a shell call's result: the real command never runs and the model\nreceives this output as the tool's genuine result.",
211
+ )
212
+
213
+
214
+ class TestCase(BaseModel):
215
+ """
216
+ One test case.
217
+
218
+ This type is the source of truth for the **input contract**
219
+ (`schemas/case.schema.json`): the shape a `--case-json` payload — and the
220
+ SDKs' generated case models — must have. Serialization skips
221
+ absent/default fields so the canonical JSON form is minimal; the SDK case
222
+ builders emit that same form (pinned by the kitchen-sink golden in
223
+ `tests/fixtures/contract/`).
224
+ """
225
+
226
+ model_config = ConfigDict(
227
+ extra="forbid",
228
+ )
229
+ evals: list[BooleanEval | NumericEval | CalledEval | NotCalledEval] = Field(
230
+ ..., description="The evals that decide whether this case passes. Must be non-empty."
231
+ )
232
+ input: str = Field(
233
+ ..., description="The initial data/prompt handed to the skill as the first user message."
234
+ )
235
+ mocks: list[MockDecl] | None = Field(
236
+ None,
237
+ description="Mock/spy declarations for this case: a declaration with a `stub`/`deny`/\n`rewrite` action intercepts matching tool calls; one without observes\nonly. `called`/`not_called` evals reference these by `name`.",
238
+ )
239
+ name: str | None = Field(
240
+ None, description="Human-readable name (defaults to the file stem when loaded from a file)."
241
+ )
242
+ skill: str = Field(
243
+ ..., description="Path to the skill directory under test, relative to the test-case file."
244
+ )
245
+ spy: bool | None = Field(
246
+ None,
247
+ description="Record every tool call through the mock/spy channel even with no\n`mocks` declared, so code-level consumers (the SDKs' spies) get records.\nImplied whenever `mocks` is non-empty.",
248
+ )
249
+ user: SimulatedUser | None = Field(
250
+ None, description="Present for multi-turn cases; absent for single-turn."
251
+ )
@@ -0,0 +1,50 @@
1
+ # generated by datamodel-codegen:
2
+ # filename: error.schema.json
3
+
4
+ from __future__ import annotations
5
+ from typing import Literal
6
+ from pydantic import BaseModel, Field
7
+
8
+
9
+ class ReportError(BaseModel):
10
+ """
11
+ A structured error, emitted as the `--format json` / `json-stream` output
12
+ when a `skilltest run` cannot produce a [`Report`].
13
+
14
+ This is the machine-readable counterpart to the human hint the CLI prints on
15
+ stderr: it rides on stdout so SDK/plugin consumers get the [`ProviderErrorKind`]
16
+ (and the `code`/`context`) for targeted handling — retry on
17
+ [`ProviderErrorKind::Timeout`], fail fast on [`ProviderErrorKind::Auth`] —
18
+ instead of matching substrings in the message. For `json` the object is
19
+ emitted bare; for `json-stream` it is the terminal
20
+ `{"type":"error","error":{…}}` line.
21
+ """
22
+
23
+ code: Literal["usage", "provider"] = Field(
24
+ ..., description="The coarse failure class, matching the process exit code."
25
+ )
26
+ context: str | None = Field(
27
+ None,
28
+ description="The provider context the failure came from (e.g. `oneharness:claude-code`,\n`api-judge`). Absent for usage errors.",
29
+ )
30
+ kind: (
31
+ Literal[
32
+ "auth",
33
+ "rate_limit",
34
+ "model_not_found",
35
+ "quota",
36
+ "overloaded",
37
+ "timeout",
38
+ "spawn",
39
+ "protocol",
40
+ "other",
41
+ ]
42
+ | None
43
+ ) = Field(
44
+ None,
45
+ description="The structured provider-failure category, when skilltest could classify\nit. Absent for usage errors and for unclassified provider failures.",
46
+ )
47
+ message: str = Field(
48
+ ...,
49
+ description="A human-readable description of what went wrong (the same text printed on\nstderr, minus the suggested-action hint).",
50
+ )
@@ -0,0 +1,220 @@
1
+ # generated by datamodel-codegen:
2
+ # filename: report.schema.json
3
+
4
+ from __future__ import annotations
5
+ from typing import Any, Literal
6
+ from pydantic import BaseModel, Field
7
+
8
+
9
+ class BooleanDetail(BaseModel):
10
+ """
11
+ The kind-specific detail of an eval outcome, for reporting.
12
+
13
+ The variant titles name the generated SDK model for each union arm, so keep
14
+ them stable: they are part of the SDK API surface.
15
+ """
16
+
17
+ expected: bool
18
+ kind: Literal["boolean"]
19
+ value: bool
20
+
21
+
22
+ class NumericDetail(BaseModel):
23
+ """
24
+ The kind-specific detail of an eval outcome, for reporting.
25
+
26
+ The variant titles name the generated SDK model for each union arm, so keep
27
+ them stable: they are part of the SDK API surface.
28
+ """
29
+
30
+ comparator: Literal["gte", "gt", "lte", "lt"] = Field(
31
+ ..., description="How a numeric score is compared to its threshold."
32
+ )
33
+ kind: Literal["numeric"]
34
+ threshold: float
35
+ value: float
36
+
37
+
38
+ class CallsDetail(BaseModel):
39
+ """
40
+ The deterministic `called`/`not_called` verdict: how many observed calls
41
+ matched, against what expectation.
42
+ """
43
+
44
+ count: int = Field(..., description="Matching calls observed.", ge=0)
45
+ kind: Literal["calls"]
46
+ negated: bool | None = Field(
47
+ False, description="True for `not_called` (the eval required absence)."
48
+ )
49
+ times: int | None = Field(
50
+ None,
51
+ description='The exact count required (`times`); `null` means "at least one"\n(or, with `negated`, "none").',
52
+ ge=0,
53
+ )
54
+
55
+
56
+ class EvalOutcome(BaseModel):
57
+ """
58
+ The result of running one eval against a transcript.
59
+ """
60
+
61
+ detail: BooleanDetail | NumericDetail | CallsDetail = Field(
62
+ ..., description="Kind-specific verdict detail."
63
+ )
64
+ label: str = Field(..., description="The eval's label (name or criterion).")
65
+ passed: bool = Field(..., description="Whether the eval passed.")
66
+ reason: str = Field(..., description="The judge's stated reason.")
67
+
68
+
69
+ class MockCall(BaseModel):
70
+ """
71
+ One observed tool call, as recorded by the mock/spy channel: the harness
72
+ hook's spy log for real runs, or the provider's `mock_calls` response for
73
+ the command protocol. Carries the **original, pre-rewrite** input — the
74
+ transcript's `events` show post-rewrite reality (the stub that actually
75
+ ran); this shows what the skill *attempted*.
76
+ """
77
+
78
+ action: str = Field(
79
+ ...,
80
+ description="The verdict applied: `allow` (fell through every rule), `deny`,\n`rewrite`, or `stub`.",
81
+ )
82
+ input: Any | None = Field(
83
+ None, description="The tool's original input arguments; `null` when the event carried none."
84
+ )
85
+ mock: str | None = Field(
86
+ None,
87
+ description="Name of the mock declaration the intercepting rule came from; `null`\nfor `allow`. Filled by the runner, so SDKs bind records to mock objects\nby name instead of re-deriving rule indices.",
88
+ )
89
+ rule: int | None = Field(
90
+ None, description="Index of the compiled rule that intercepted; `null` for `allow`.", ge=0
91
+ )
92
+ tool: str | None = Field(
93
+ None, description="Tool name as the harness reported it; `null` when the event named none."
94
+ )
95
+
96
+
97
+ class ToolEvent(BaseModel):
98
+ """
99
+ One normalized tool-call / action event the skill took during a turn, lifted
100
+ from oneharness's `events` array (its `--events` output). Harness-agnostic, so
101
+ a consumer can inspect *what the skill did* — shell commands, file edits, tool
102
+ uses — across any harness, not just the final text. Mirrors the oneharness
103
+ action-event shape; `input` is the structured, tool-shaped args so a consumer
104
+ can match on the command string or file path without re-parsing.
105
+
106
+ `input` is a free-form JSON value, so `Message`/`Transcript` are `PartialEq`
107
+ but not `Eq`.
108
+ """
109
+
110
+ index: int | None = Field(
111
+ 0,
112
+ description='Position within the turn, so ordering ("did X before Y") is expressible.',
113
+ ge=0,
114
+ )
115
+ input: Any | None = Field(
116
+ None,
117
+ description="Structured tool arguments (the command, the file path); `null` when none.",
118
+ )
119
+ kind: str = Field(
120
+ ...,
121
+ description="`tool_call` (the skill invoked a tool) or `tool_result` (the observation).",
122
+ )
123
+ name: str | None = Field(
124
+ None,
125
+ description="Normalized tool name where knowable (e.g. `bash`, `edit_file`); `null` for\na `tool_result` or when the harness did not name it.",
126
+ )
127
+ output: str | None = Field(
128
+ None, description="The result/observation text, when the transcript exposed it."
129
+ )
130
+
131
+
132
+ class Usage(BaseModel):
133
+ """
134
+ Token / cost usage for one provider call.
135
+
136
+ Each field is independently optional because not every harness reports every
137
+ signal (cost is commonly absent on subscription auth; some harnesses report
138
+ no usage at all). The whole struct is `Option<Usage>` on a turn — `None`
139
+ means "no signal," not "zero."
140
+ """
141
+
142
+ cost_usd: float | None = None
143
+ input_tokens: int | None = Field(None, ge=0)
144
+ output_tokens: int | None = Field(None, ge=0)
145
+
146
+
147
+ class Message(BaseModel):
148
+ """
149
+ A single turn in the conversation.
150
+ """
151
+
152
+ content: str
153
+ events: list[ToolEvent] | None = Field(
154
+ None,
155
+ description="The normalized tool events the skill took producing this turn (assistant\nturns only, and only when the harness exposed a tool transcript via\noneharness `--events`). Empty otherwise. Surfaced for post-hoc analysis\nand streamed live for short-circuiting.",
156
+ )
157
+ role: Literal["user", "assistant", "system"] = Field(..., description="Who produced a message.")
158
+
159
+
160
+ class Summary(BaseModel):
161
+ """
162
+ Aggregate pass/fail counts for a report.
163
+ """
164
+
165
+ cases: int = Field(..., description="Distinct test cases represented.", ge=0)
166
+ failed: int = Field(..., description="Runs that failed.", ge=0)
167
+ passed: int = Field(..., description="Runs that passed.", ge=0)
168
+ runs: int = Field(..., description="Total (case × platform × model) runs.", ge=0)
169
+ usage: Usage | None = Field(
170
+ None,
171
+ description="Aggregated token/cost usage across every run in the report. Omitted\nwhen no run reported usage.",
172
+ )
173
+
174
+
175
+ class Transcript(BaseModel):
176
+ """
177
+ An ordered list of messages. Thin wrapper so the type reads clearly at call
178
+ sites and so we can grow conversation-level helpers without churn.
179
+ """
180
+
181
+ messages: list[Message]
182
+
183
+
184
+ class CaseRun(BaseModel):
185
+ """
186
+ The result of running one test case on one (platform, model) pair.
187
+ """
188
+
189
+ case: str = Field(..., description="The test case name.")
190
+ evals: list[EvalOutcome] = Field(..., description="Per-eval outcomes, in declaration order.")
191
+ history_command: str | None = Field(
192
+ None,
193
+ description="A ready-to-run command that replays this run's recorded transcript — e.g.\n`oneharness history show <name> --history-dir <dir>` — so a past run can\nbe reviewed after the fact. Present only when the run was recorded (the\noneharness provider with history enabled); `null` for providers/configs\nthat record no history.",
194
+ )
195
+ mock_calls: list[MockCall] | None = Field(
196
+ None,
197
+ description='Every tool call the mock/spy channel observed, in order, with the\noriginal (pre-rewrite) input and the verdict applied. `null` when the\nchannel was off for this run (no `mocks`, no `spy`); an empty array\nmeans the channel was on and the skill made no tool calls — SDKs use\nthat distinction so a spy on a channel-less run errs instead of reading\nas "zero calls".',
198
+ )
199
+ model: str = Field(..., description="The model this run used.")
200
+ passed: bool = Field(..., description="True iff every eval in this run passed.")
201
+ platform: str = Field(..., description="The harness platform this run used.")
202
+ skill: str = Field(..., description="Absolute-ish path to the skill that was exercised.")
203
+ transcript: Transcript = Field(
204
+ ..., description="The full conversation, for debugging and deterministic mix-in checks."
205
+ )
206
+ turns: int = Field(..., description="Number of assistant turns produced.", ge=0)
207
+ usage: Usage | None = Field(
208
+ None,
209
+ description="Aggregated token/cost usage across every provider call in this run\n(skill turns + simulated-user turns + judge calls). Omitted when no\nusage was reported (e.g. the fake provider or a harness that doesn't\nsurface usage).",
210
+ )
211
+
212
+
213
+ class Report(BaseModel):
214
+ """
215
+ The top-level report for a `skilltest run` invocation.
216
+ """
217
+
218
+ passed: bool = Field(..., description="True iff every run passed.")
219
+ runs: list[CaseRun] = Field(..., description="Every individual run.")
220
+ summary: Summary = Field(..., description="Aggregate counts.")
@@ -0,0 +1,24 @@
1
+ # generated by datamodel-codegen:
2
+ # filename: validation.schema.json
3
+
4
+ from __future__ import annotations
5
+ from pydantic import BaseModel, Field
6
+
7
+
8
+ class ValidationFinding(BaseModel):
9
+ """
10
+ One problem found while validating a skill, as serialized in the
11
+ `skilltest validate --format json` output.
12
+ """
13
+
14
+ message: str = Field(..., description="What is wrong and how to fix it.")
15
+ skill: str = Field(..., description="The skill directory the finding is about.")
16
+
17
+
18
+ class ValidationReport(BaseModel):
19
+ """
20
+ The top-level report for a `skilltest validate` invocation.
21
+ """
22
+
23
+ findings: list[ValidationFinding] = Field(..., description="Every finding, in discovery order.")
24
+ valid: bool = Field(..., description="True iff no findings were produced.")