@rudderhq/agent-runtime-opencode-local 0.5.1 → 0.5.2-canary.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/server/browser-capability.test.js +12 -2
- package/dist/server/browser-capability.test.js.map +1 -1
- package/dist/server/execute.d.ts +1 -0
- package/dist/server/execute.d.ts.map +1 -1
- package/dist/server/execute.js +466 -419
- package/dist/server/execute.js.map +1 -1
- package/package.json +2 -2
- package/skills/browser/SKILL.md +172 -41
- package/skills/browser/agents/openai.yaml +12 -2
- package/skills/browser/evals/evals.json +41 -0
- package/skills/browser/references/interaction-guide.md +171 -0
- package/skills/browser/references/tool-contract.md +192 -68
- package/skills/rudder-docs/references/cli-reference.md +18 -1
- package/skills/skill-creator/CHANGELOG.md +26 -0
- package/skills/skill-creator/LICENSE.txt +202 -0
- package/skills/skill-creator/SKILL.md +544 -3
- package/skills/skill-creator/agents/analyzer.md +274 -0
- package/skills/skill-creator/agents/comparator.md +202 -0
- package/skills/skill-creator/agents/grader.md +223 -0
- package/skills/skill-creator/assets/eval_review.html +146 -0
- package/skills/skill-creator/eval-viewer/generate_review.py +471 -0
- package/skills/skill-creator/eval-viewer/viewer.html +1325 -0
- package/skills/skill-creator/references/compatibility/claudecode.md +29 -0
- package/skills/skill-creator/references/compatibility/codex.md +69 -0
- package/skills/skill-creator/references/compatibility/other.md +41 -0
- package/skills/skill-creator/references/rudder.md +96 -0
- package/skills/skill-creator/references/schemas.md +430 -0
- package/skills/skill-creator/scripts/__init__.py +0 -0
- package/skills/skill-creator/scripts/aggregate_benchmark.py +401 -0
- package/skills/skill-creator/scripts/generate_report.py +335 -0
- package/skills/skill-creator/scripts/improve_description.py +197 -0
- package/skills/skill-creator/scripts/model_backends.py +115 -0
- package/skills/skill-creator/scripts/package_skill.py +136 -0
- package/skills/skill-creator/scripts/quick_validate.py +103 -0
- package/skills/skill-creator/scripts/run_eval.py +363 -0
- package/skills/skill-creator/scripts/run_loop.py +319 -0
- package/skills/skill-creator/scripts/utils.py +223 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Claude Code Compatibility
|
|
2
|
+
|
|
3
|
+
Use this guide when the current host is Claude Code, or when evaluating a skill's trigger behavior with the Claude Code CLI.
|
|
4
|
+
|
|
5
|
+
## Capabilities
|
|
6
|
+
|
|
7
|
+
| Capability | Support | Notes |
|
|
8
|
+
|------------|---------|-------|
|
|
9
|
+
| Draft or edit skills | Yes | Normal filesystem edits work. |
|
|
10
|
+
| Run execution evals | Yes | Use realistic prompts and preserve run artifacts. |
|
|
11
|
+
| Parallel baseline runs | Yes | Run with-skill and baseline runs in the same turn when subagents are available. |
|
|
12
|
+
| Description tuning | Yes | Prefer the Claude backend for trigger measurement. |
|
|
13
|
+
| Packaging | Yes | Use `scripts/package_skill.py` when packaging is requested. |
|
|
14
|
+
|
|
15
|
+
## Trigger Evaluation
|
|
16
|
+
|
|
17
|
+
Use `scripts/run_eval.py --backend claude` or `scripts/run_loop --backend claude` when available. Claude Code can observe real skill routing behavior through `claude -p`, so this is the highest-fidelity backend for description tuning.
|
|
18
|
+
|
|
19
|
+
Treat these measurements as stronger evidence than judged proxy routing. Still include near-miss prompts and held-out prompts so the description does not overfit obvious positive cases.
|
|
20
|
+
|
|
21
|
+
## Baselines
|
|
22
|
+
|
|
23
|
+
For new skills, compare against a no-skill baseline. For existing skills, snapshot the old skill before editing and use that snapshot as the baseline.
|
|
24
|
+
|
|
25
|
+
Launch with-skill and baseline runs together when possible so duration and model conditions are comparable. Save outputs, grading, and timing data in the same workspace structure described in `SKILL.md`.
|
|
26
|
+
|
|
27
|
+
## Host Metadata
|
|
28
|
+
|
|
29
|
+
Claude Code does not require `agents/openai.yaml`. Do not create that file solely for Claude Code. If the same skill will also be distributed to Codex/OpenAI, also read `codex.md` and add the Codex/OpenAI metadata there.
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# Codex/OpenAI Compatibility
|
|
2
|
+
|
|
3
|
+
Use this guide when the target host is Codex or an OpenAI product surface that supports skills.
|
|
4
|
+
|
|
5
|
+
## Capabilities
|
|
6
|
+
|
|
7
|
+
| Capability | Support | Notes |
|
|
8
|
+
|------------|---------|-------|
|
|
9
|
+
| Draft or edit skills | Yes | Use normal skill structure with `SKILL.md` as the source of agent instructions. |
|
|
10
|
+
| Run execution evals | Yes | Use available subagents or `codex exec`-based flows when supported. |
|
|
11
|
+
| Parallel baseline runs | Usually | Depends on the current Codex host and subagent availability. |
|
|
12
|
+
| Description tuning | Yes | Codex backend results are judged routing proxies, not native invocation telemetry. |
|
|
13
|
+
| Packaging | Yes | Use `scripts/package_skill.py` when packaging is requested. |
|
|
14
|
+
|
|
15
|
+
## Routing
|
|
16
|
+
|
|
17
|
+
Codex skill routing depends primarily on the `name` and `description` fields in `SKILL.md` frontmatter. Keep all "when to use this skill" information in `description`; the body is loaded only after the skill triggers.
|
|
18
|
+
|
|
19
|
+
Use `scripts/run_eval.py --backend codex` or `scripts/run_loop --backend codex` to test whether the description makes the intended routing obvious. Treat this as a judged proxy, not proof of native skill invocation behavior.
|
|
20
|
+
|
|
21
|
+
## `agents/openai.yaml`
|
|
22
|
+
|
|
23
|
+
For Codex/OpenAI distribution, create or update `agents/openai.yaml` when the skill should have product-facing metadata, a default prompt, declared MCP dependencies, or explicit invocation policy.
|
|
24
|
+
|
|
25
|
+
`agents/openai.yaml` is machine/product metadata. It does not replace `SKILL.md`, and it should not contain the skill's execution workflow.
|
|
26
|
+
|
|
27
|
+
Recommended minimal shape:
|
|
28
|
+
|
|
29
|
+
```yaml
|
|
30
|
+
interface:
|
|
31
|
+
display_name: "Human Readable Name"
|
|
32
|
+
short_description: "Short UI summary"
|
|
33
|
+
default_prompt: "Use $skill-name to complete the task."
|
|
34
|
+
|
|
35
|
+
policy:
|
|
36
|
+
allow_implicit_invocation: true
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Field guidance:
|
|
40
|
+
|
|
41
|
+
- Quote string values.
|
|
42
|
+
- Keep YAML keys unquoted.
|
|
43
|
+
- Make `interface.display_name` a human-facing name for UI lists and chips.
|
|
44
|
+
- Keep `interface.short_description` short enough for UI scanning, usually 25-64 characters.
|
|
45
|
+
- Make `interface.default_prompt` a short starter prompt that explicitly includes `$skill-name`.
|
|
46
|
+
- Add `interface.icon_small`, `interface.icon_large`, or `interface.brand_color` only when the user provided assets or branding.
|
|
47
|
+
- Add `dependencies.tools` only for real required tools, such as an MCP server the skill expects.
|
|
48
|
+
- Use `policy.allow_implicit_invocation: false` only when the skill should be invoked explicitly via `$skill-name` rather than routed automatically.
|
|
49
|
+
|
|
50
|
+
Example dependency block:
|
|
51
|
+
|
|
52
|
+
```yaml
|
|
53
|
+
dependencies:
|
|
54
|
+
tools:
|
|
55
|
+
- type: "mcp"
|
|
56
|
+
value: "github"
|
|
57
|
+
description: "GitHub MCP server"
|
|
58
|
+
transport: "streamable_http"
|
|
59
|
+
url: "https://api.githubcopilot.com/mcp/"
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Validation
|
|
63
|
+
|
|
64
|
+
Before reporting a Codex/OpenAI-targeted skill as ready:
|
|
65
|
+
|
|
66
|
+
1. Parse `SKILL.md` frontmatter and confirm `name` and `description` are valid.
|
|
67
|
+
2. Parse `agents/openai.yaml` if it exists.
|
|
68
|
+
3. Run this skill's `scripts/quick_validate.py` against the target skill.
|
|
69
|
+
4. If an OpenAI-provided validator is available in the host, run that as an additional compatibility check.
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Other Host Compatibility
|
|
2
|
+
|
|
3
|
+
Use this guide for generic shell agents, chat-only hosts, headless worker hosts, or any environment that is not clearly Claude Code or Codex/OpenAI.
|
|
4
|
+
|
|
5
|
+
## Capabilities
|
|
6
|
+
|
|
7
|
+
| Capability | Support | Notes |
|
|
8
|
+
|------------|---------|-------|
|
|
9
|
+
| Draft or edit skills | Yes | Requires normal filesystem access. |
|
|
10
|
+
| Run execution evals | Depends | Requires a usable agent CLI or manual serial runs. |
|
|
11
|
+
| Parallel baseline runs | Depends | Skip baselines when subagents are unavailable. |
|
|
12
|
+
| Description tuning | Depends | Requires a shell-accessible backend such as `claude` or `codex`. |
|
|
13
|
+
| Packaging | Yes | `scripts/package_skill.py` works with Python and filesystem access. |
|
|
14
|
+
|
|
15
|
+
## Chat-Only Hosts
|
|
16
|
+
|
|
17
|
+
When there are no subagents, run test prompts serially yourself. Read the skill, follow it on each test prompt, save any generated files, and ask for human feedback inline.
|
|
18
|
+
|
|
19
|
+
Skip quantitative baseline benchmarking when there is no independent baseline runner. Qualitative review is more honest than pretending a self-run comparison is independent.
|
|
20
|
+
|
|
21
|
+
## Headless Worker Hosts
|
|
22
|
+
|
|
23
|
+
When a browser cannot open, generate a static review artifact:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
python <skill-creator-path>/eval-viewer/generate_review.py \
|
|
27
|
+
<workspace>/iteration-N \
|
|
28
|
+
--skill-name "my-skill" \
|
|
29
|
+
--benchmark <workspace>/iteration-N/benchmark.json \
|
|
30
|
+
--static <workspace>/iteration-N/review.html
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Make sure each run directory has an `outputs/` subdirectory before launching the viewer. If a lightweight benchmark only produced grading and timing files, create a small `outputs/summary.md` so the viewer has something to render.
|
|
34
|
+
|
|
35
|
+
## Metadata
|
|
36
|
+
|
|
37
|
+
Do not create `agents/openai.yaml` by default for generic or chat-only hosts. Create it only when the skill will also be distributed to Codex/OpenAI, then follow `codex.md`.
|
|
38
|
+
|
|
39
|
+
## Description Optimization
|
|
40
|
+
|
|
41
|
+
Run description optimization only when the current environment exposes a compatible shell backend. If no backend exists, write realistic trigger and near-miss prompts for later testing, then continue with manual qualitative evaluation.
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Rudder Compatibility
|
|
2
|
+
|
|
3
|
+
Read this reference whenever the run exposes Rudder context such as
|
|
4
|
+
`RUDDER_AGENT_ID`, `RUDDER_ORG_ID`, `AGENT_HOME`, or
|
|
5
|
+
`RUDDER_ORG_SKILLS_DIR`. Rudder decides which skills are enabled for a run, so
|
|
6
|
+
filesystem discovery by Codex, Claude, or another provider is not sufficient.
|
|
7
|
+
|
|
8
|
+
Inside Rudder, this is the default compatibility guide. Do not also load a
|
|
9
|
+
guide from `references/compatibility/` unless the user explicitly requests
|
|
10
|
+
provider-native compatibility, packaging, evaluation, or a cross-host
|
|
11
|
+
comparison.
|
|
12
|
+
|
|
13
|
+
## Choose Ownership Before Writing
|
|
14
|
+
|
|
15
|
+
Use the narrowest durable owner that matches the user's intent:
|
|
16
|
+
|
|
17
|
+
- **Current agent only:** install the complete package under
|
|
18
|
+
`$AGENT_HOME/skills/<slug>`.
|
|
19
|
+
- **Organization or team:** install the complete package under
|
|
20
|
+
`$RUDDER_ORG_SKILLS_DIR/<slug>` and import it into the organization Skill
|
|
21
|
+
Library.
|
|
22
|
+
|
|
23
|
+
Do not use `~/.agents/skills`, a provider-native skill directory, or a temporary
|
|
24
|
+
runtime mount as Rudder's source of truth. Those locations may be discoverable,
|
|
25
|
+
but Rudder loads only the skills resolved for the agent and invocation.
|
|
26
|
+
|
|
27
|
+
## Agent-Private Skills
|
|
28
|
+
|
|
29
|
+
For a `SKILL.md`-only package, the first-party creation command can create and
|
|
30
|
+
enable it in one operation:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
rudder agent skills create "$RUDDER_AGENT_ID" \
|
|
34
|
+
--name "<name>" \
|
|
35
|
+
--slug "<slug>" \
|
|
36
|
+
--markdown-file <path-to-SKILL.md> \
|
|
37
|
+
--enable \
|
|
38
|
+
--json
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
For a package with scripts, references, assets, or agent metadata, write the
|
|
42
|
+
whole directory to `$AGENT_HOME/skills/<slug>`, validate it there, and then
|
|
43
|
+
enable it additively:
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
rudder agent skills enable "$RUDDER_AGENT_ID" "agent:<slug>" --json
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Do not claim the skill will load merely because its files exist. Report the
|
|
50
|
+
installed and enabled states separately, and remember that enablement affects
|
|
51
|
+
future runs rather than rewriting the current run's loaded context.
|
|
52
|
+
|
|
53
|
+
## Organization Skills
|
|
54
|
+
|
|
55
|
+
Importing or replacing an organization skill is a governed mutation. Proceed
|
|
56
|
+
only when the user explicitly requested organization sharing and the current
|
|
57
|
+
actor has organization Skill management permission. Inspect the current
|
|
58
|
+
library first so an existing same-name package is not overwritten by surprise.
|
|
59
|
+
|
|
60
|
+
After writing and validating the complete package at
|
|
61
|
+
`$RUDDER_ORG_SKILLS_DIR/<slug>`, register its full file inventory:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
rudder skill import \
|
|
65
|
+
--org-id "$RUDDER_ORG_ID" \
|
|
66
|
+
--source "$RUDDER_ORG_SKILLS_DIR/<slug>" \
|
|
67
|
+
--json
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Use the returned key or selection reference when enabling the skill for an
|
|
71
|
+
agent:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
rudder agent skills enable "<agent-id>" "<returned-selection-ref>" --json
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Importing and enabling are separate operations. `skills enable` is additive;
|
|
78
|
+
do not use `skills sync` unless the user explicitly wants to replace the full
|
|
79
|
+
optional skill set and every desired existing selection has been preserved.
|
|
80
|
+
|
|
81
|
+
## Eval And Review Workspace
|
|
82
|
+
|
|
83
|
+
Bundled Rudder skills and provider materializations are read-only inputs. Keep
|
|
84
|
+
evaluation outputs elsewhere:
|
|
85
|
+
|
|
86
|
+
- With project context, prefer
|
|
87
|
+
`$RUDDER_PROJECT_LIBRARY_ROOT/skill-evals/<slug>-workspace/`.
|
|
88
|
+
- Without project context, use
|
|
89
|
+
`$RUDDER_ORG_WORKSPACE_ROOT/artifacts/YYYY-MM-DD/skill-evals/<slug>-workspace/`.
|
|
90
|
+
|
|
91
|
+
Keep the normal `iteration-N/eval-name/{with_skill,old_skill}/` layout inside
|
|
92
|
+
that directory. Generate the review viewer there and return a Rudder-visible
|
|
93
|
+
Library link when the environment provides the Library reference command.
|
|
94
|
+
|
|
95
|
+
Never print or request `RUDDER_API_KEY`. Use the injected Rudder CLI context and
|
|
96
|
+
the typed Rudder tools when available.
|
|
@@ -0,0 +1,430 @@
|
|
|
1
|
+
# JSON Schemas
|
|
2
|
+
|
|
3
|
+
This document defines the JSON schemas used by skill-creator.
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## evals.json
|
|
8
|
+
|
|
9
|
+
Defines the evals for a skill. Located at `evals/evals.json` within the skill directory.
|
|
10
|
+
|
|
11
|
+
```json
|
|
12
|
+
{
|
|
13
|
+
"skill_name": "example-skill",
|
|
14
|
+
"evals": [
|
|
15
|
+
{
|
|
16
|
+
"id": 1,
|
|
17
|
+
"prompt": "User's example prompt",
|
|
18
|
+
"expected_output": "Description of expected result",
|
|
19
|
+
"files": ["evals/files/sample1.pdf"],
|
|
20
|
+
"expectations": [
|
|
21
|
+
"The output includes X",
|
|
22
|
+
"The skill used script Y"
|
|
23
|
+
]
|
|
24
|
+
}
|
|
25
|
+
]
|
|
26
|
+
}
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
**Fields:**
|
|
30
|
+
- `skill_name`: Name matching the skill's frontmatter
|
|
31
|
+
- `evals[].id`: Unique integer identifier
|
|
32
|
+
- `evals[].prompt`: The task to execute
|
|
33
|
+
- `evals[].expected_output`: Human-readable description of success
|
|
34
|
+
- `evals[].files`: Optional list of input file paths (relative to skill root)
|
|
35
|
+
- `evals[].expectations`: List of verifiable statements
|
|
36
|
+
|
|
37
|
+
---
|
|
38
|
+
|
|
39
|
+
## history.json
|
|
40
|
+
|
|
41
|
+
Tracks version progression in Improve mode. Located at workspace root.
|
|
42
|
+
|
|
43
|
+
```json
|
|
44
|
+
{
|
|
45
|
+
"started_at": "2026-01-15T10:30:00Z",
|
|
46
|
+
"skill_name": "pdf",
|
|
47
|
+
"current_best": "v2",
|
|
48
|
+
"iterations": [
|
|
49
|
+
{
|
|
50
|
+
"version": "v0",
|
|
51
|
+
"parent": null,
|
|
52
|
+
"expectation_pass_rate": 0.65,
|
|
53
|
+
"grading_result": "baseline",
|
|
54
|
+
"is_current_best": false
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
"version": "v1",
|
|
58
|
+
"parent": "v0",
|
|
59
|
+
"expectation_pass_rate": 0.75,
|
|
60
|
+
"grading_result": "won",
|
|
61
|
+
"is_current_best": false
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"version": "v2",
|
|
65
|
+
"parent": "v1",
|
|
66
|
+
"expectation_pass_rate": 0.85,
|
|
67
|
+
"grading_result": "won",
|
|
68
|
+
"is_current_best": true
|
|
69
|
+
}
|
|
70
|
+
]
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
**Fields:**
|
|
75
|
+
- `started_at`: ISO timestamp of when improvement started
|
|
76
|
+
- `skill_name`: Name of the skill being improved
|
|
77
|
+
- `current_best`: Version identifier of the best performer
|
|
78
|
+
- `iterations[].version`: Version identifier (v0, v1, ...)
|
|
79
|
+
- `iterations[].parent`: Parent version this was derived from
|
|
80
|
+
- `iterations[].expectation_pass_rate`: Pass rate from grading
|
|
81
|
+
- `iterations[].grading_result`: "baseline", "won", "lost", or "tie"
|
|
82
|
+
- `iterations[].is_current_best`: Whether this is the current best version
|
|
83
|
+
|
|
84
|
+
---
|
|
85
|
+
|
|
86
|
+
## grading.json
|
|
87
|
+
|
|
88
|
+
Output from the grader agent. Located at `<run-dir>/grading.json`.
|
|
89
|
+
|
|
90
|
+
```json
|
|
91
|
+
{
|
|
92
|
+
"expectations": [
|
|
93
|
+
{
|
|
94
|
+
"text": "The output includes the name 'John Smith'",
|
|
95
|
+
"passed": true,
|
|
96
|
+
"evidence": "Found in transcript Step 3: 'Extracted names: John Smith, Sarah Johnson'"
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
"text": "The spreadsheet has a SUM formula in cell B10",
|
|
100
|
+
"passed": false,
|
|
101
|
+
"evidence": "No spreadsheet was created. The output was a text file."
|
|
102
|
+
}
|
|
103
|
+
],
|
|
104
|
+
"summary": {
|
|
105
|
+
"passed": 2,
|
|
106
|
+
"failed": 1,
|
|
107
|
+
"total": 3,
|
|
108
|
+
"pass_rate": 0.67
|
|
109
|
+
},
|
|
110
|
+
"execution_metrics": {
|
|
111
|
+
"tool_calls": {
|
|
112
|
+
"Read": 5,
|
|
113
|
+
"Write": 2,
|
|
114
|
+
"Bash": 8
|
|
115
|
+
},
|
|
116
|
+
"total_tool_calls": 15,
|
|
117
|
+
"total_steps": 6,
|
|
118
|
+
"errors_encountered": 0,
|
|
119
|
+
"output_chars": 12450,
|
|
120
|
+
"transcript_chars": 3200
|
|
121
|
+
},
|
|
122
|
+
"timing": {
|
|
123
|
+
"executor_duration_seconds": 165.0,
|
|
124
|
+
"grader_duration_seconds": 26.0,
|
|
125
|
+
"total_duration_seconds": 191.0
|
|
126
|
+
},
|
|
127
|
+
"claims": [
|
|
128
|
+
{
|
|
129
|
+
"claim": "The form has 12 fillable fields",
|
|
130
|
+
"type": "factual",
|
|
131
|
+
"verified": true,
|
|
132
|
+
"evidence": "Counted 12 fields in field_info.json"
|
|
133
|
+
}
|
|
134
|
+
],
|
|
135
|
+
"user_notes_summary": {
|
|
136
|
+
"uncertainties": ["Used 2023 data, may be stale"],
|
|
137
|
+
"needs_review": [],
|
|
138
|
+
"workarounds": ["Fell back to text overlay for non-fillable fields"]
|
|
139
|
+
},
|
|
140
|
+
"eval_feedback": {
|
|
141
|
+
"suggestions": [
|
|
142
|
+
{
|
|
143
|
+
"assertion": "The output includes the name 'John Smith'",
|
|
144
|
+
"reason": "A hallucinated document that mentions the name would also pass"
|
|
145
|
+
}
|
|
146
|
+
],
|
|
147
|
+
"overall": "Assertions check presence but not correctness."
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
**Fields:**
|
|
153
|
+
- `expectations[]`: Graded expectations with evidence
|
|
154
|
+
- `summary`: Aggregate pass/fail counts
|
|
155
|
+
- `execution_metrics`: Tool usage and output size (from executor's metrics.json)
|
|
156
|
+
- `timing`: Wall clock timing (from timing.json)
|
|
157
|
+
- `claims`: Extracted and verified claims from the output
|
|
158
|
+
- `user_notes_summary`: Issues flagged by the executor
|
|
159
|
+
- `eval_feedback`: (optional) Improvement suggestions for the evals, only present when the grader identifies issues worth raising
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
## metrics.json
|
|
164
|
+
|
|
165
|
+
Output from the executor agent. Located at `<run-dir>/outputs/metrics.json`.
|
|
166
|
+
|
|
167
|
+
```json
|
|
168
|
+
{
|
|
169
|
+
"tool_calls": {
|
|
170
|
+
"Read": 5,
|
|
171
|
+
"Write": 2,
|
|
172
|
+
"Bash": 8,
|
|
173
|
+
"Edit": 1,
|
|
174
|
+
"Glob": 2,
|
|
175
|
+
"Grep": 0
|
|
176
|
+
},
|
|
177
|
+
"total_tool_calls": 18,
|
|
178
|
+
"total_steps": 6,
|
|
179
|
+
"files_created": ["filled_form.pdf", "field_values.json"],
|
|
180
|
+
"errors_encountered": 0,
|
|
181
|
+
"output_chars": 12450,
|
|
182
|
+
"transcript_chars": 3200
|
|
183
|
+
}
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
**Fields:**
|
|
187
|
+
- `tool_calls`: Count per tool type
|
|
188
|
+
- `total_tool_calls`: Sum of all tool calls
|
|
189
|
+
- `total_steps`: Number of major execution steps
|
|
190
|
+
- `files_created`: List of output files created
|
|
191
|
+
- `errors_encountered`: Number of errors during execution
|
|
192
|
+
- `output_chars`: Total character count of output files
|
|
193
|
+
- `transcript_chars`: Character count of transcript
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
## timing.json
|
|
198
|
+
|
|
199
|
+
Wall clock timing for a run. Located at `<run-dir>/timing.json`.
|
|
200
|
+
|
|
201
|
+
**How to capture:** When a subagent task completes, the task notification includes `total_tokens` and `duration_ms`. Save these immediately — they are not persisted anywhere else and cannot be recovered after the fact.
|
|
202
|
+
|
|
203
|
+
```json
|
|
204
|
+
{
|
|
205
|
+
"total_tokens": 84852,
|
|
206
|
+
"duration_ms": 23332,
|
|
207
|
+
"total_duration_seconds": 23.3,
|
|
208
|
+
"executor_start": "2026-01-15T10:30:00Z",
|
|
209
|
+
"executor_end": "2026-01-15T10:32:45Z",
|
|
210
|
+
"executor_duration_seconds": 165.0,
|
|
211
|
+
"grader_start": "2026-01-15T10:32:46Z",
|
|
212
|
+
"grader_end": "2026-01-15T10:33:12Z",
|
|
213
|
+
"grader_duration_seconds": 26.0
|
|
214
|
+
}
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
---
|
|
218
|
+
|
|
219
|
+
## benchmark.json
|
|
220
|
+
|
|
221
|
+
Output from Benchmark mode. Located at `benchmarks/<timestamp>/benchmark.json`.
|
|
222
|
+
|
|
223
|
+
```json
|
|
224
|
+
{
|
|
225
|
+
"metadata": {
|
|
226
|
+
"skill_name": "pdf",
|
|
227
|
+
"skill_path": "/path/to/pdf",
|
|
228
|
+
"executor_model": "claude-sonnet-4-20250514",
|
|
229
|
+
"analyzer_model": "most-capable-model",
|
|
230
|
+
"timestamp": "2026-01-15T10:30:00Z",
|
|
231
|
+
"evals_run": [1, 2, 3],
|
|
232
|
+
"runs_per_configuration": 3
|
|
233
|
+
},
|
|
234
|
+
|
|
235
|
+
"runs": [
|
|
236
|
+
{
|
|
237
|
+
"eval_id": 1,
|
|
238
|
+
"eval_name": "Ocean",
|
|
239
|
+
"configuration": "with_skill",
|
|
240
|
+
"run_number": 1,
|
|
241
|
+
"result": {
|
|
242
|
+
"pass_rate": 0.85,
|
|
243
|
+
"passed": 6,
|
|
244
|
+
"failed": 1,
|
|
245
|
+
"total": 7,
|
|
246
|
+
"time_seconds": 42.5,
|
|
247
|
+
"tokens": 3800,
|
|
248
|
+
"tool_calls": 18,
|
|
249
|
+
"errors": 0
|
|
250
|
+
},
|
|
251
|
+
"expectations": [
|
|
252
|
+
{"text": "...", "passed": true, "evidence": "..."}
|
|
253
|
+
],
|
|
254
|
+
"notes": [
|
|
255
|
+
"Used 2023 data, may be stale",
|
|
256
|
+
"Fell back to text overlay for non-fillable fields"
|
|
257
|
+
]
|
|
258
|
+
}
|
|
259
|
+
],
|
|
260
|
+
|
|
261
|
+
"run_summary": {
|
|
262
|
+
"with_skill": {
|
|
263
|
+
"pass_rate": {"mean": 0.85, "stddev": 0.05, "min": 0.80, "max": 0.90},
|
|
264
|
+
"time_seconds": {"mean": 45.0, "stddev": 12.0, "min": 32.0, "max": 58.0},
|
|
265
|
+
"tokens": {"mean": 3800, "stddev": 400, "min": 3200, "max": 4100}
|
|
266
|
+
},
|
|
267
|
+
"without_skill": {
|
|
268
|
+
"pass_rate": {"mean": 0.35, "stddev": 0.08, "min": 0.28, "max": 0.45},
|
|
269
|
+
"time_seconds": {"mean": 32.0, "stddev": 8.0, "min": 24.0, "max": 42.0},
|
|
270
|
+
"tokens": {"mean": 2100, "stddev": 300, "min": 1800, "max": 2500}
|
|
271
|
+
},
|
|
272
|
+
"delta": {
|
|
273
|
+
"pass_rate": "+0.50",
|
|
274
|
+
"time_seconds": "+13.0",
|
|
275
|
+
"tokens": "+1700"
|
|
276
|
+
}
|
|
277
|
+
},
|
|
278
|
+
|
|
279
|
+
"notes": [
|
|
280
|
+
"Assertion 'Output is a PDF file' passes 100% in both configurations - may not differentiate skill value",
|
|
281
|
+
"Eval 3 shows high variance (50% ± 40%) - may be flaky or model-dependent",
|
|
282
|
+
"Without-skill runs consistently fail on table extraction expectations",
|
|
283
|
+
"Skill adds 13s average execution time but improves pass rate by 50%"
|
|
284
|
+
]
|
|
285
|
+
}
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
**Fields:**
|
|
289
|
+
- `metadata`: Information about the benchmark run
|
|
290
|
+
- `skill_name`: Name of the skill
|
|
291
|
+
- `timestamp`: When the benchmark was run
|
|
292
|
+
- `evals_run`: List of eval names or IDs
|
|
293
|
+
- `runs_per_configuration`: Number of runs per config (e.g. 3)
|
|
294
|
+
- `runs[]`: Individual run results
|
|
295
|
+
- `eval_id`: Numeric eval identifier
|
|
296
|
+
- `eval_name`: Human-readable eval name (used as section header in the viewer)
|
|
297
|
+
- `configuration`: Must be `"with_skill"` or `"without_skill"` (the viewer uses this exact string for grouping and color coding)
|
|
298
|
+
- `run_number`: Integer run number (1, 2, 3...)
|
|
299
|
+
- `result`: Nested object with `pass_rate`, `passed`, `total`, `time_seconds`, `tokens`, `errors`
|
|
300
|
+
- `run_summary`: Statistical aggregates per configuration
|
|
301
|
+
- `with_skill` / `without_skill`: Each contains `pass_rate`, `time_seconds`, `tokens` objects with `mean` and `stddev` fields
|
|
302
|
+
- `delta`: Difference strings like `"+0.50"`, `"+13.0"`, `"+1700"`
|
|
303
|
+
- `notes`: Freeform observations from the analyzer
|
|
304
|
+
|
|
305
|
+
**Important:** The viewer reads these field names exactly. Using `config` instead of `configuration`, or putting `pass_rate` at the top level of a run instead of nested under `result`, will cause the viewer to show empty/zero values. Always reference this schema when generating benchmark.json manually.
|
|
306
|
+
|
|
307
|
+
---
|
|
308
|
+
|
|
309
|
+
## comparison.json
|
|
310
|
+
|
|
311
|
+
Output from blind comparator. Located at `<grading-dir>/comparison-N.json`.
|
|
312
|
+
|
|
313
|
+
```json
|
|
314
|
+
{
|
|
315
|
+
"winner": "A",
|
|
316
|
+
"reasoning": "Output A provides a complete solution with proper formatting and all required fields. Output B is missing the date field and has formatting inconsistencies.",
|
|
317
|
+
"rubric": {
|
|
318
|
+
"A": {
|
|
319
|
+
"content": {
|
|
320
|
+
"correctness": 5,
|
|
321
|
+
"completeness": 5,
|
|
322
|
+
"accuracy": 4
|
|
323
|
+
},
|
|
324
|
+
"structure": {
|
|
325
|
+
"organization": 4,
|
|
326
|
+
"formatting": 5,
|
|
327
|
+
"usability": 4
|
|
328
|
+
},
|
|
329
|
+
"content_score": 4.7,
|
|
330
|
+
"structure_score": 4.3,
|
|
331
|
+
"overall_score": 9.0
|
|
332
|
+
},
|
|
333
|
+
"B": {
|
|
334
|
+
"content": {
|
|
335
|
+
"correctness": 3,
|
|
336
|
+
"completeness": 2,
|
|
337
|
+
"accuracy": 3
|
|
338
|
+
},
|
|
339
|
+
"structure": {
|
|
340
|
+
"organization": 3,
|
|
341
|
+
"formatting": 2,
|
|
342
|
+
"usability": 3
|
|
343
|
+
},
|
|
344
|
+
"content_score": 2.7,
|
|
345
|
+
"structure_score": 2.7,
|
|
346
|
+
"overall_score": 5.4
|
|
347
|
+
}
|
|
348
|
+
},
|
|
349
|
+
"output_quality": {
|
|
350
|
+
"A": {
|
|
351
|
+
"score": 9,
|
|
352
|
+
"strengths": ["Complete solution", "Well-formatted", "All fields present"],
|
|
353
|
+
"weaknesses": ["Minor style inconsistency in header"]
|
|
354
|
+
},
|
|
355
|
+
"B": {
|
|
356
|
+
"score": 5,
|
|
357
|
+
"strengths": ["Readable output", "Correct basic structure"],
|
|
358
|
+
"weaknesses": ["Missing date field", "Formatting inconsistencies", "Partial data extraction"]
|
|
359
|
+
}
|
|
360
|
+
},
|
|
361
|
+
"expectation_results": {
|
|
362
|
+
"A": {
|
|
363
|
+
"passed": 4,
|
|
364
|
+
"total": 5,
|
|
365
|
+
"pass_rate": 0.80,
|
|
366
|
+
"details": [
|
|
367
|
+
{"text": "Output includes name", "passed": true}
|
|
368
|
+
]
|
|
369
|
+
},
|
|
370
|
+
"B": {
|
|
371
|
+
"passed": 3,
|
|
372
|
+
"total": 5,
|
|
373
|
+
"pass_rate": 0.60,
|
|
374
|
+
"details": [
|
|
375
|
+
{"text": "Output includes name", "passed": true}
|
|
376
|
+
]
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
---
|
|
383
|
+
|
|
384
|
+
## analysis.json
|
|
385
|
+
|
|
386
|
+
Output from post-hoc analyzer. Located at `<grading-dir>/analysis.json`.
|
|
387
|
+
|
|
388
|
+
```json
|
|
389
|
+
{
|
|
390
|
+
"comparison_summary": {
|
|
391
|
+
"winner": "A",
|
|
392
|
+
"winner_skill": "path/to/winner/skill",
|
|
393
|
+
"loser_skill": "path/to/loser/skill",
|
|
394
|
+
"comparator_reasoning": "Brief summary of why comparator chose winner"
|
|
395
|
+
},
|
|
396
|
+
"winner_strengths": [
|
|
397
|
+
"Clear step-by-step instructions for handling multi-page documents",
|
|
398
|
+
"Included validation script that caught formatting errors"
|
|
399
|
+
],
|
|
400
|
+
"loser_weaknesses": [
|
|
401
|
+
"Vague instruction 'process the document appropriately' led to inconsistent behavior",
|
|
402
|
+
"No script for validation, agent had to improvise"
|
|
403
|
+
],
|
|
404
|
+
"instruction_following": {
|
|
405
|
+
"winner": {
|
|
406
|
+
"score": 9,
|
|
407
|
+
"issues": ["Minor: skipped optional logging step"]
|
|
408
|
+
},
|
|
409
|
+
"loser": {
|
|
410
|
+
"score": 6,
|
|
411
|
+
"issues": [
|
|
412
|
+
"Did not use the skill's formatting template",
|
|
413
|
+
"Invented own approach instead of following step 3"
|
|
414
|
+
]
|
|
415
|
+
}
|
|
416
|
+
},
|
|
417
|
+
"improvement_suggestions": [
|
|
418
|
+
{
|
|
419
|
+
"priority": "high",
|
|
420
|
+
"category": "instructions",
|
|
421
|
+
"suggestion": "Replace 'process the document appropriately' with explicit steps",
|
|
422
|
+
"expected_impact": "Would eliminate ambiguity that caused inconsistent behavior"
|
|
423
|
+
}
|
|
424
|
+
],
|
|
425
|
+
"transcript_insights": {
|
|
426
|
+
"winner_execution_pattern": "Read skill -> Followed 5-step process -> Used validation script",
|
|
427
|
+
"loser_execution_pattern": "Read skill -> Unclear on approach -> Tried 3 different methods"
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
```
|
|
File without changes
|