@uinaf/skillcheck 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -9
- package/dist/cli.js +8 -6
- package/dist/cursor-provider.js +119 -0
- package/dist/scenario.js +19 -8
- package/docs/authoring.md +70 -0
- package/docs/releasing.md +4 -4
- package/docs/scenarios.md +12 -8
- package/docs/usage.md +11 -2
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|

|
|
2
2
|
|
|
3
|
-
#
|
|
3
|
+
# uinaf/skillcheck
|
|
4
4
|
|
|
5
5
|
lint and eval harness for agent skills. one CLI with two halves: a keyless
|
|
6
6
|
structural lint any repo can run in CI, and a promptfoo-driven eval loop that
|
|
@@ -49,14 +49,15 @@ CLI it documents.
|
|
|
49
49
|
|
|
50
50
|
## docs
|
|
51
51
|
|
|
52
|
-
| doc
|
|
53
|
-
|
|
|
54
|
-
| [usage](docs/usage.md)
|
|
55
|
-
| [scenarios](docs/scenarios.md)
|
|
56
|
-
| [
|
|
57
|
-
| [
|
|
58
|
-
| [
|
|
59
|
-
| [
|
|
52
|
+
| doc | when |
|
|
53
|
+
| --------------------------------------------------------------- | ------------------------------------- |
|
|
54
|
+
| [usage](docs/usage.md) | every subcommand, flag, and auth path |
|
|
55
|
+
| [scenarios](docs/scenarios.md) | writing an eval scenario |
|
|
56
|
+
| [authoring](docs/authoring.md) | writing and auditing the skill itself |
|
|
57
|
+
| [adoption](docs/adoption.md) | wiring the lint into another repo |
|
|
58
|
+
| [releasing](docs/releasing.md) | the npm pipeline |
|
|
59
|
+
| [contributing](CONTRIBUTING.md) | local setup and the verify gate |
|
|
60
|
+
| [security](https://github.com/uinaf/skillcheck/security/policy) | reporting a vulnerability |
|
|
60
61
|
|
|
61
62
|
## license
|
|
62
63
|
|
package/dist/cli.js
CHANGED
|
@@ -59,7 +59,7 @@ function stateDirs(root) {
|
|
|
59
59
|
}
|
|
60
60
|
function runOptions(flags) {
|
|
61
61
|
const harness = flags.get("--harness") ?? "claude";
|
|
62
|
-
if (harness !== "claude" && harness !== "codex") fail(`--harness must be claude or
|
|
62
|
+
if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
|
|
63
63
|
return {
|
|
64
64
|
harness,
|
|
65
65
|
agentModel: flags.get("--agent"),
|
|
@@ -94,7 +94,8 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
94
94
|
const dirs = stateDirs(root);
|
|
95
95
|
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
96
96
|
scratchDir: dirs.scratch,
|
|
97
|
-
transformPath: path.join(here, `transform${selfExt}`)
|
|
97
|
+
transformPath: path.join(here, `transform${selfExt}`),
|
|
98
|
+
cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
|
|
98
99
|
});
|
|
99
100
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
100
101
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
@@ -165,7 +166,7 @@ function discoverScenarios(root) {
|
|
|
165
166
|
}
|
|
166
167
|
function cmdRun(argv) {
|
|
167
168
|
const { positional, flags } = parseArgs(argv);
|
|
168
|
-
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex]");
|
|
169
|
+
if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--harness claude|codex|cursor]");
|
|
169
170
|
const o = runScenario(positional[0], runOptions(flags), resolveRoot(flags));
|
|
170
171
|
if (o.score === void 0) {
|
|
171
172
|
console.error(`ERROR ${o.name}: ${o.error ?? "no usable result"} (promptfoo rc=${o.rc})`);
|
|
@@ -225,8 +226,9 @@ function reduceResults(dir, allowMixed) {
|
|
|
225
226
|
const provider = raw.config?.providers?.[0];
|
|
226
227
|
const judge = raw.config?.defaultTest?.options?.provider;
|
|
227
228
|
const base = f.replace(/\.json$/, "");
|
|
228
|
-
const
|
|
229
|
-
const
|
|
229
|
+
const suffix = base.match(/--(codex|cursor)$/);
|
|
230
|
+
const harness = suffix === null ? "claude" : suffix[1];
|
|
231
|
+
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
230
232
|
let sha = "unattested";
|
|
231
233
|
try {
|
|
232
234
|
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
@@ -239,7 +241,7 @@ function reduceResults(dir, allowMixed) {
|
|
|
239
241
|
skills_tree_sha: sha,
|
|
240
242
|
score: res.score,
|
|
241
243
|
pass: res.success,
|
|
242
|
-
agent_model: provider?.config?.model ??
|
|
244
|
+
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
243
245
|
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? "unknown",
|
|
244
246
|
latency_ms: res.latencyMs,
|
|
245
247
|
tokens: (res.tokenUsage?.total ?? 0) + (res.tokenUsage?.assertions?.total ?? 0)
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
//#region src/cursor-provider.ts
|
|
3
|
+
function newStreamState() {
|
|
4
|
+
return {
|
|
5
|
+
isError: false,
|
|
6
|
+
skillCalls: []
|
|
7
|
+
};
|
|
8
|
+
}
|
|
9
|
+
const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
|
|
10
|
+
function foldLine(state, line) {
|
|
11
|
+
let event;
|
|
12
|
+
try {
|
|
13
|
+
event = JSON.parse(line);
|
|
14
|
+
} catch {
|
|
15
|
+
return;
|
|
16
|
+
}
|
|
17
|
+
if (typeof event !== "object" || event === null) return;
|
|
18
|
+
const e = event;
|
|
19
|
+
if (e.type === "tool_call" && e.subtype === "started") {
|
|
20
|
+
const p = e.tool_call?.readToolCall?.args?.path;
|
|
21
|
+
const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
|
|
22
|
+
if (m) state.skillCalls.push({
|
|
23
|
+
name: m[1],
|
|
24
|
+
source: "project",
|
|
25
|
+
path: p
|
|
26
|
+
});
|
|
27
|
+
return;
|
|
28
|
+
}
|
|
29
|
+
if (e.type === "result") {
|
|
30
|
+
state.isError = e.is_error === true || e.subtype !== "success";
|
|
31
|
+
state.result = typeof e.result === "string" ? e.result : "";
|
|
32
|
+
if (e.usage) {
|
|
33
|
+
const prompt = e.usage.inputTokens ?? 0;
|
|
34
|
+
const completion = e.usage.outputTokens ?? 0;
|
|
35
|
+
state.tokenUsage = {
|
|
36
|
+
prompt,
|
|
37
|
+
completion,
|
|
38
|
+
cached: e.usage.cacheReadTokens ?? 0,
|
|
39
|
+
total: prompt + completion
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
var CursorAgentProvider = class {
|
|
45
|
+
providerId;
|
|
46
|
+
config;
|
|
47
|
+
constructor(options) {
|
|
48
|
+
this.providerId = options.id ?? "cursor-agent";
|
|
49
|
+
if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
|
|
50
|
+
this.config = options.config;
|
|
51
|
+
}
|
|
52
|
+
id() {
|
|
53
|
+
return this.providerId;
|
|
54
|
+
}
|
|
55
|
+
async callApi(prompt) {
|
|
56
|
+
const command = this.config.command ?? "cursor-agent";
|
|
57
|
+
const args = [
|
|
58
|
+
"-p",
|
|
59
|
+
"--trust",
|
|
60
|
+
"--output-format",
|
|
61
|
+
"stream-json"
|
|
62
|
+
];
|
|
63
|
+
if (this.config.model !== void 0) args.push("--model", this.config.model);
|
|
64
|
+
const state = newStreamState();
|
|
65
|
+
const stderr = [];
|
|
66
|
+
let stdoutBuf = "";
|
|
67
|
+
return new Promise((resolve) => {
|
|
68
|
+
const child = spawn(command, args, {
|
|
69
|
+
cwd: this.config.working_dir,
|
|
70
|
+
env: process.env,
|
|
71
|
+
stdio: [
|
|
72
|
+
"pipe",
|
|
73
|
+
"pipe",
|
|
74
|
+
"pipe"
|
|
75
|
+
]
|
|
76
|
+
});
|
|
77
|
+
const timeoutMs = this.config.timeout_ms ?? 9e5;
|
|
78
|
+
const timer = setTimeout(() => {
|
|
79
|
+
child.kill("SIGKILL");
|
|
80
|
+
resolve({ error: `cursor-agent timed out after ${timeoutMs}ms` });
|
|
81
|
+
}, timeoutMs);
|
|
82
|
+
child.on("error", (err) => {
|
|
83
|
+
clearTimeout(timer);
|
|
84
|
+
resolve({ error: `failed to spawn ${command}: ${err.message}` });
|
|
85
|
+
});
|
|
86
|
+
child.stdout.on("data", (chunk) => {
|
|
87
|
+
stdoutBuf += chunk.toString("utf8");
|
|
88
|
+
const lines = stdoutBuf.split("\n");
|
|
89
|
+
stdoutBuf = lines.pop() ?? "";
|
|
90
|
+
for (const line of lines) foldLine(state, line);
|
|
91
|
+
});
|
|
92
|
+
child.stderr.on("data", (chunk) => {
|
|
93
|
+
stderr.push(chunk.toString("utf8"));
|
|
94
|
+
});
|
|
95
|
+
child.on("close", (code) => {
|
|
96
|
+
clearTimeout(timer);
|
|
97
|
+
if (stdoutBuf !== "") foldLine(state, stdoutBuf);
|
|
98
|
+
if (state.result === void 0) {
|
|
99
|
+
const tail = stderr.join("").trim().slice(-2e3);
|
|
100
|
+
resolve({ error: `cursor-agent exited ${code ?? "by signal"} without a result event${tail === "" ? "" : `: ${tail}`}` });
|
|
101
|
+
return;
|
|
102
|
+
}
|
|
103
|
+
if (state.isError) {
|
|
104
|
+
resolve({ error: state.result || "cursor-agent reported an error result" });
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
resolve({
|
|
108
|
+
output: state.result,
|
|
109
|
+
tokenUsage: state.tokenUsage,
|
|
110
|
+
metadata: { skillCalls: state.skillCalls }
|
|
111
|
+
});
|
|
112
|
+
});
|
|
113
|
+
child.stdin.write(prompt);
|
|
114
|
+
child.stdin.end();
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
};
|
|
118
|
+
//#endregion
|
|
119
|
+
export { CursorAgentProvider as default, foldLine, newStreamState };
|
package/dist/scenario.js
CHANGED
|
@@ -35,7 +35,7 @@ function loadScenario(scenarioDir) {
|
|
|
35
35
|
function runNameFor(scenarioDir, harness) {
|
|
36
36
|
const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
|
|
37
37
|
if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
|
|
38
|
-
return harness === "
|
|
38
|
+
return harness === "claude" ? `${m[1]}--${m[2]}` : `${m[1]}--${m[2]}--${harness}`;
|
|
39
39
|
}
|
|
40
40
|
function frontmatterRange(text) {
|
|
41
41
|
const lines = text.split("\n");
|
|
@@ -53,7 +53,11 @@ function stripHiddenFlag(skillMd) {
|
|
|
53
53
|
if (!range) return skillMd;
|
|
54
54
|
return skillMd.split("\n").filter((l, i) => !(i >= range[0] && i < range[1] && l.startsWith("disable-model-invocation:"))).join("\n");
|
|
55
55
|
}
|
|
56
|
-
const RESERVED = /* @__PURE__ */ new Set([
|
|
56
|
+
const RESERVED = /* @__PURE__ */ new Set([
|
|
57
|
+
".claude",
|
|
58
|
+
".agents",
|
|
59
|
+
".cursor"
|
|
60
|
+
]);
|
|
57
61
|
function materialize(s, runDir, harness) {
|
|
58
62
|
const workdir = path.join(runDir, "workdir");
|
|
59
63
|
fs.rmSync(runDir, {
|
|
@@ -78,7 +82,7 @@ function materialize(s, runDir, harness) {
|
|
|
78
82
|
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
79
83
|
fs.writeFileSync(dest, content);
|
|
80
84
|
}
|
|
81
|
-
const roots = harness === "codex" ? [".claude", ".agents"] : [".claude"];
|
|
85
|
+
const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
|
|
82
86
|
for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
|
|
83
87
|
recursive: true,
|
|
84
88
|
filter: (src) => path.basename(src) !== "evals"
|
|
@@ -106,7 +110,14 @@ function materialize(s, runDir, harness) {
|
|
|
106
110
|
manifestPath
|
|
107
111
|
};
|
|
108
112
|
}
|
|
109
|
-
function agentProvider(opts, workdir, skill) {
|
|
113
|
+
function agentProvider(opts, workdir, skill, paths) {
|
|
114
|
+
if (opts.harness === "cursor") return {
|
|
115
|
+
id: `file://${paths.cursorProviderPath}`,
|
|
116
|
+
config: {
|
|
117
|
+
...opts.agentModel ? { model: opts.agentModel } : {},
|
|
118
|
+
working_dir: workdir
|
|
119
|
+
}
|
|
120
|
+
};
|
|
110
121
|
if (opts.harness === "codex") return {
|
|
111
122
|
id: "openai:codex-sdk",
|
|
112
123
|
config: {
|
|
@@ -138,11 +149,11 @@ function agentProvider(opts, workdir, skill) {
|
|
|
138
149
|
}
|
|
139
150
|
};
|
|
140
151
|
}
|
|
141
|
-
function buildConfig(s, workdir, manifestPath, opts,
|
|
152
|
+
function buildConfig(s, workdir, manifestPath, opts, paths) {
|
|
142
153
|
return {
|
|
143
154
|
description: `${s.skill}/${s.scenario}`,
|
|
144
155
|
prompts: ["{{task}}"],
|
|
145
|
-
providers: [agentProvider(opts, workdir, s.skill)],
|
|
156
|
+
providers: [agentProvider(opts, workdir, s.skill, paths)],
|
|
146
157
|
defaultTest: { options: {
|
|
147
158
|
provider: process.env.ANTHROPIC_API_KEY ? `anthropic:messages:${opts.judgeModel}` : {
|
|
148
159
|
id: "anthropic:claude-agent-sdk",
|
|
@@ -173,7 +184,7 @@ function buildConfig(s, workdir, manifestPath, opts, transformPath) {
|
|
|
173
184
|
}
|
|
174
185
|
}
|
|
175
186
|
},
|
|
176
|
-
transform: `file://${transformPath}`
|
|
187
|
+
transform: `file://${paths.transformPath}`
|
|
177
188
|
} },
|
|
178
189
|
tests: [{
|
|
179
190
|
description: s.criteria.context,
|
|
@@ -216,7 +227,7 @@ function generateRun(scenarioDir, opts, paths) {
|
|
|
216
227
|
});
|
|
217
228
|
fs.symlinkSync(sdkDir, link, "dir");
|
|
218
229
|
}
|
|
219
|
-
const config = buildConfig(s, workdir, manifestPath, opts, paths
|
|
230
|
+
const config = buildConfig(s, workdir, manifestPath, opts, paths);
|
|
220
231
|
const configPath = path.join(runDir, "promptfooconfig.json");
|
|
221
232
|
fs.writeFileSync(configPath, JSON.stringify(config, null, 2));
|
|
222
233
|
return {
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# authoring and auditing skills
|
|
2
|
+
|
|
3
|
+
what `lint` and evals cannot judge: whether a skill is worth routing to and
|
|
4
|
+
cheap to load. use this when writing a skill or auditing one. evidence beats
|
|
5
|
+
stylistic preference; run `skillcheck lint` first and let this cover the rest.
|
|
6
|
+
|
|
7
|
+
## metadata and discovery
|
|
8
|
+
|
|
9
|
+
- `name` is concrete and easy to say out loud. `helper`, `tools`, `utils` are
|
|
10
|
+
discovery smells.
|
|
11
|
+
- `description` is third person and says both what the skill does and when to
|
|
12
|
+
use it. it is an always-loaded retrieval pointer: front-load the concrete
|
|
13
|
+
action or domain that should activate it.
|
|
14
|
+
- one trigger per materially distinct request branch. collapse synonyms that
|
|
15
|
+
rename the same branch.
|
|
16
|
+
- state the main overlap boundary without naming another skill.
|
|
17
|
+
|
|
18
|
+
## body shape
|
|
19
|
+
|
|
20
|
+
- keep `SKILL.md` on workflow, principles, boundaries, and routing. lead with
|
|
21
|
+
the task, not a bibliography.
|
|
22
|
+
- assume the model is smart; spend tokens on repo-specific judgment. delete any
|
|
23
|
+
instruction that would not change a capable model's behavior.
|
|
24
|
+
- match freedom to risk: high for contextual judgment, medium when a preferred
|
|
25
|
+
pattern exists, low for fragile operations.
|
|
26
|
+
- say what evidence to gather and what a complete result includes. end each
|
|
27
|
+
step with an observable completion condition, not "understood" or "handled".
|
|
28
|
+
|
|
29
|
+
## progressive disclosure
|
|
30
|
+
|
|
31
|
+
- durable detail, rubrics, and long examples go in `references/`, one hop from
|
|
32
|
+
`SKILL.md`, each with a task-shaped retrieval job. material every path needs
|
|
33
|
+
stays inline.
|
|
34
|
+
- for repeated deterministic work, route to the target's existing framework,
|
|
35
|
+
schema, task graph, or library; otherwise add a tested module in the
|
|
36
|
+
project's primary language, not ad-hoc shell rendered as prose.
|
|
37
|
+
- when executable code belongs to another maintained project, link the exact
|
|
38
|
+
public artifact and state the contract it demonstrates; do not fork it into
|
|
39
|
+
prose.
|
|
40
|
+
- a package stays independently usable: state prerequisites and out-of-scope
|
|
41
|
+
next steps locally. never invoke, import, or assume a sibling skill.
|
|
42
|
+
|
|
43
|
+
## audit
|
|
44
|
+
|
|
45
|
+
grade each dimension strong, mixed, or weak:
|
|
46
|
+
|
|
47
|
+
| dimension | question |
|
|
48
|
+
| ---------------------- | --------------------------------------------------------------------- |
|
|
49
|
+
| discovery | does metadata alone route a realistic request here |
|
|
50
|
+
| workflow | does the body say how to begin, what evidence to gather, when to stop |
|
|
51
|
+
| progressive disclosure | is detail in the right file |
|
|
52
|
+
| repo fit | are links, commands, and conventions current |
|
|
53
|
+
| verification | is the strongest mechanical check named, plus a real evidence loop |
|
|
54
|
+
| boundaries | are limits and next steps stated without leaning on a sibling skill |
|
|
55
|
+
|
|
56
|
+
blockers, must-fix: invalid frontmatter; a description that fails discovery;
|
|
57
|
+
stale commands, paths, or links; a workflow with no start, evidence loop, or
|
|
58
|
+
completion; conflicts with the repo's guidance; sibling-skill dependencies.
|
|
59
|
+
|
|
60
|
+
major findings: vague name; synonym-stuffed description; bloated `SKILL.md`;
|
|
61
|
+
missing or muddy boundaries; prose re-inventing a deterministic tool; abstract
|
|
62
|
+
examples.
|
|
63
|
+
|
|
64
|
+
## improve
|
|
65
|
+
|
|
66
|
+
fix blockers first, then the highest-leverage majors. prefer the smallest
|
|
67
|
+
change that improves activation, decision quality, or proof. when pruning,
|
|
68
|
+
measure common-path context for representative requests; line count alone does
|
|
69
|
+
not reveal retrieval cost. after edits, rerun `skillcheck lint` and the repo's
|
|
70
|
+
gate, and rerun evals when behavior was the thing changed.
|
package/docs/releasing.md
CHANGED
|
@@ -7,12 +7,12 @@ a push to `main` runs one workflow, `.github/workflows/release.yml`:
|
|
|
7
7
|
```text
|
|
8
8
|
verify ──┐
|
|
9
9
|
├──> release npm publish, OIDC + uinaf-releaser (release environment)
|
|
10
|
-
|
|
10
|
+
scan ────┘
|
|
11
11
|
```
|
|
12
12
|
|
|
13
|
-
`verify` and `
|
|
14
|
-
`
|
|
15
|
-
|
|
13
|
+
`verify` and `scan` are the shared gate: `verify` is called from `verify.yml`,
|
|
14
|
+
and `scan` calls the shared scan in `uinaf/.github`, the same one `scan.yml`
|
|
15
|
+
runs for pull requests. keep it that way: a second copy of the gate on a push-to-`main` workflow
|
|
16
16
|
races this one over the same commit.
|
|
17
17
|
|
|
18
18
|
the file name `release.yml` is load-bearing. see below.
|
package/docs/scenarios.md
CHANGED
|
@@ -8,9 +8,9 @@ a scenario is two files in a frozen location:
|
|
|
8
8
|
```
|
|
9
9
|
|
|
10
10
|
the path is the identity: `<skill>--<scenario>` names the run, the result file,
|
|
11
|
-
and the scorecard entry. on the codex
|
|
12
|
-
so
|
|
13
|
-
is not discovered.
|
|
11
|
+
and the scorecard entry. on the codex and cursor harnesses the name gains a
|
|
12
|
+
`--codex` or `--cursor` suffix, so every harness can hold results side by side.
|
|
13
|
+
a directory missing either file is not discovered.
|
|
14
14
|
|
|
15
15
|
## task.md
|
|
16
16
|
|
|
@@ -27,9 +27,9 @@ Fix the failing check in the config below.
|
|
|
27
27
|
|
|
28
28
|
each block is replaced in the prompt with a pointer ("Input file `config.json`
|
|
29
29
|
is available in your working directory.") and written to disk. destinations
|
|
30
|
-
must stay under the workdir, must not collide, and must not target `.claude
|
|
31
|
-
`.agents/`, since a fixture that writes agent config would be
|
|
32
|
-
own examiner.
|
|
30
|
+
must stay under the workdir, must not collide, and must not target `.claude/`,
|
|
31
|
+
`.agents/`, or `.cursor/`, since a fixture that writes agent config would be
|
|
32
|
+
configuring its own examiner.
|
|
33
33
|
|
|
34
34
|
write the task the way a user would write it. do not name the skill, describe
|
|
35
35
|
its steps, or hint at the checklist: routing is part of what is being measured.
|
|
@@ -81,6 +81,10 @@ frontmatter block; body text mentioning the key does not count.
|
|
|
81
81
|
## the workdir
|
|
82
82
|
|
|
83
83
|
per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
|
|
84
|
-
time. the skill under test is installed
|
|
85
|
-
`.agents/skills/<skill>/` on
|
|
84
|
+
time. the skill under test is installed where the harness discovers skills —
|
|
85
|
+
`.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
|
|
86
|
+
`.cursor/skills/<skill>/` alone on cursor — with its `evals/` directory
|
|
86
87
|
excluded, so criteria never leak into the agent's context.
|
|
88
|
+
|
|
89
|
+
scenario quality is behavioral proof; [authoring](authoring.md) covers the
|
|
90
|
+
judgment layer lint and evals cannot grade.
|
package/docs/usage.md
CHANGED
|
@@ -45,8 +45,16 @@ message, and writes no provenance sidecar. it is never reported as
|
|
|
45
45
|
`FAIL score=0.0000`; only a real judged verdict can fail a run.
|
|
46
46
|
|
|
47
47
|
defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
|
|
48
|
-
`--max-turns 50`. on the codex
|
|
49
|
-
the
|
|
48
|
+
`--max-turns 50`. on the codex and cursor harnesses, omitting `--agent` leaves
|
|
49
|
+
the model to that CLI's own default.
|
|
50
|
+
|
|
51
|
+
`--harness cursor` drives the scenario through the Cursor Agent CLI
|
|
52
|
+
(`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
|
|
53
|
+
`--agent` names a Cursor model id, e.g. `composer-2.5`. there is no promptfoo
|
|
54
|
+
cursor provider, so the run uses this package's own provider module, which
|
|
55
|
+
replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
56
|
+
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
57
|
+
evidence. the judge leg is unchanged.
|
|
50
58
|
|
|
51
59
|
## sweep
|
|
52
60
|
|
|
@@ -121,5 +129,6 @@ written inside the installed package.
|
|
|
121
129
|
| `ANTHROPIC_API_KEY` | judge grades over `anthropic:messages:<model>` instead of the SDK |
|
|
122
130
|
| `CODEX_HOME` (default `~/.codex`) | where the codex harness finds the local `codex` CLI login |
|
|
123
131
|
| `OPENAI_API_KEY` | codex agent auth when there is no local login |
|
|
132
|
+
| `CURSOR_API_KEY` | cursor agent auth; a logged-in `cursor-agent` also works |
|
|
124
133
|
|
|
125
134
|
the judge stays on the Anthropic selection regardless of the agent harness.
|