@uinaf/skillcheck 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +48 -19
- package/dist/cursor-process.js +71 -0
- package/dist/cursor-provider.js +95 -33
- package/docs/adoption.md +1 -1
- package/docs/usage.md +24 -9
- package/package.json +5 -4
package/dist/cli.js
CHANGED
|
@@ -107,9 +107,9 @@ function promptfooEntry() {
|
|
|
107
107
|
function classifyResult(raw) {
|
|
108
108
|
const root = raw;
|
|
109
109
|
const res = root?.results?.results?.[0];
|
|
110
|
-
if (res === void 0) return { error: "promptfoo output carried no result" };
|
|
110
|
+
if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
|
|
111
111
|
const message = typeof res.error === "string" ? res.error.trim() : "";
|
|
112
|
-
const stats = root?.results?.stats;
|
|
112
|
+
const stats = root?.results?.stats ?? void 0;
|
|
113
113
|
const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
|
|
114
114
|
if (message !== "" && !gradedByStats) return { error: message };
|
|
115
115
|
if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
|
|
@@ -128,6 +128,9 @@ function gitHead(root) {
|
|
|
128
128
|
function metaPath(resultPath) {
|
|
129
129
|
return resultPath.replace(/\.json$/, ".meta.json");
|
|
130
130
|
}
|
|
131
|
+
function attemptPath(resultPath) {
|
|
132
|
+
return `${resultPath}.attempt`;
|
|
133
|
+
}
|
|
131
134
|
function runScenario(scenarioDir, opts, root) {
|
|
132
135
|
const dirs = stateDirs(root);
|
|
133
136
|
const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
|
|
@@ -137,6 +140,7 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
137
140
|
});
|
|
138
141
|
fs.mkdirSync(dirs.results, { recursive: true });
|
|
139
142
|
const resultPath = path.join(dirs.results, `${name}.json`);
|
|
143
|
+
fs.writeFileSync(attemptPath(resultPath), "{}\n");
|
|
140
144
|
fs.rmSync(resultPath, { force: true });
|
|
141
145
|
fs.rmSync(metaPath(resultPath), { force: true });
|
|
142
146
|
const sha = gitHead(root);
|
|
@@ -174,12 +178,15 @@ function runScenario(scenarioDir, opts, root) {
|
|
|
174
178
|
outcome.score = verdict.score;
|
|
175
179
|
outcome.pass = verdict.pass;
|
|
176
180
|
outcome.error = verdict.error;
|
|
177
|
-
if (verdict.score !== void 0)
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
181
|
+
if (verdict.score !== void 0) {
|
|
182
|
+
fs.writeFileSync(metaPath(resultPath), JSON.stringify({
|
|
183
|
+
skills_tree_sha: sha,
|
|
184
|
+
harness: opts.harness,
|
|
185
|
+
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
186
|
+
tool_version: toolVersion()
|
|
187
|
+
}, null, 2) + "\n");
|
|
188
|
+
fs.rmSync(attemptPath(resultPath), { force: true });
|
|
189
|
+
}
|
|
183
190
|
return outcome;
|
|
184
191
|
}
|
|
185
192
|
function discoverScenarios(root) {
|
|
@@ -247,11 +254,30 @@ function cmdSweep(argv) {
|
|
|
247
254
|
console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
|
|
248
255
|
process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
|
|
249
256
|
}
|
|
257
|
+
function resultIdentity(file) {
|
|
258
|
+
const base = file.replace(/\.json$/, "");
|
|
259
|
+
const suffix = base.match(/--(codex|cursor)$/);
|
|
260
|
+
const harness = suffix === null ? "claude" : suffix[1];
|
|
261
|
+
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
262
|
+
return {
|
|
263
|
+
skill,
|
|
264
|
+
scenario: rest.join("--"),
|
|
265
|
+
harness
|
|
266
|
+
};
|
|
267
|
+
}
|
|
250
268
|
function reduceResults(dir, allowMixed) {
|
|
251
269
|
const entries = [];
|
|
252
270
|
const skipped = [];
|
|
253
271
|
const shas = /* @__PURE__ */ new Set();
|
|
254
|
-
|
|
272
|
+
const files = fs.readdirSync(dir);
|
|
273
|
+
const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
|
|
274
|
+
const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
|
|
275
|
+
for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
|
|
276
|
+
if (incomplete.has(f)) {
|
|
277
|
+
console.error(`skipping ${f}: attempt did not complete with a graded result`);
|
|
278
|
+
skipped.push(f);
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
255
281
|
let raw;
|
|
256
282
|
try {
|
|
257
283
|
raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
|
|
@@ -259,17 +285,16 @@ function reduceResults(dir, allowMixed) {
|
|
|
259
285
|
raw = void 0;
|
|
260
286
|
}
|
|
261
287
|
const res = raw?.results?.results?.[0];
|
|
262
|
-
|
|
263
|
-
|
|
288
|
+
const verdict = classifyResult(raw);
|
|
289
|
+
if (verdict.score === void 0 || verdict.pass === void 0) {
|
|
290
|
+
console.error(`skipping ${f}: ${verdict.error}`);
|
|
264
291
|
skipped.push(f);
|
|
265
292
|
continue;
|
|
266
293
|
}
|
|
267
294
|
const provider = raw.config?.providers?.[0];
|
|
268
295
|
const judge = raw.config?.defaultTest?.options?.provider;
|
|
269
296
|
const base = f.replace(/\.json$/, "");
|
|
270
|
-
const
|
|
271
|
-
const harness = suffix === null ? "claude" : suffix[1];
|
|
272
|
-
const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
|
|
297
|
+
const { skill, scenario, harness } = resultIdentity(f);
|
|
273
298
|
let sha = "unattested";
|
|
274
299
|
try {
|
|
275
300
|
sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
|
|
@@ -277,11 +302,11 @@ function reduceResults(dir, allowMixed) {
|
|
|
277
302
|
shas.add(sha);
|
|
278
303
|
entries.push({
|
|
279
304
|
skill,
|
|
280
|
-
scenario
|
|
305
|
+
scenario,
|
|
281
306
|
harness,
|
|
282
307
|
skills_tree_sha: sha,
|
|
283
|
-
score:
|
|
284
|
-
pass:
|
|
308
|
+
score: verdict.score,
|
|
309
|
+
pass: verdict.pass,
|
|
285
310
|
agent_model: provider?.config?.model ?? `${harness}-default`,
|
|
286
311
|
judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
|
|
287
312
|
latency_ms: res.latencyMs,
|
|
@@ -335,15 +360,19 @@ function cmdSummarize(argv) {
|
|
|
335
360
|
const { positional, flags } = parseArgs(argv);
|
|
336
361
|
if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
|
|
337
362
|
const dirs = stateDirs(resolveRoot(flags));
|
|
338
|
-
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}
|
|
363
|
+
if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
|
|
339
364
|
const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
|
|
340
365
|
fs.mkdirSync(dirs.scorecards, { recursive: true });
|
|
341
366
|
const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
|
|
342
367
|
const existing = readExistingScorecard(out);
|
|
368
|
+
const skippedKeys = new Set(skipped.map((file) => entryKey(resultIdentity(file))));
|
|
369
|
+
if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
|
|
343
370
|
const merged = mergeScorecard(existing, entries);
|
|
371
|
+
const treeSha = treeShaOf(merged.entries);
|
|
372
|
+
if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
|
|
344
373
|
const scorecard = {
|
|
345
374
|
ran_at: (/* @__PURE__ */ new Date()).toISOString(),
|
|
346
|
-
skills_tree_sha:
|
|
375
|
+
skills_tree_sha: treeSha,
|
|
347
376
|
scenarios: merged.entries
|
|
348
377
|
};
|
|
349
378
|
fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
//#region src/cursor-process.ts
|
|
3
|
+
let child;
|
|
4
|
+
let terminal = false;
|
|
5
|
+
let cleaning = false;
|
|
6
|
+
let watchdog = setTimeout(cleanup, 5e3);
|
|
7
|
+
function cleanup() {
|
|
8
|
+
if (cleaning) return;
|
|
9
|
+
cleaning = true;
|
|
10
|
+
clearTimeout(watchdog);
|
|
11
|
+
if (process.platform !== "win32") process.kill(-process.pid, "SIGKILL");
|
|
12
|
+
child?.kill("SIGKILL");
|
|
13
|
+
process.exit(1);
|
|
14
|
+
}
|
|
15
|
+
function fail(error) {
|
|
16
|
+
if (terminal) return;
|
|
17
|
+
terminal = true;
|
|
18
|
+
process.send?.({
|
|
19
|
+
type: "failure",
|
|
20
|
+
error
|
|
21
|
+
}, cleanup);
|
|
22
|
+
clearTimeout(watchdog);
|
|
23
|
+
watchdog = setTimeout(cleanup, 1e3);
|
|
24
|
+
}
|
|
25
|
+
if (!process.send) process.exit(1);
|
|
26
|
+
process.on("disconnect", cleanup);
|
|
27
|
+
process.on("message", (message) => {
|
|
28
|
+
if (message === "stop") return cleanup();
|
|
29
|
+
if (message === "cleanup" && terminal) {
|
|
30
|
+
process.send?.({ type: "cleanup" }, cleanup);
|
|
31
|
+
return;
|
|
32
|
+
}
|
|
33
|
+
if (child !== void 0 || terminal) return;
|
|
34
|
+
if (typeof message !== "object" || message === null || !("command" in message) || typeof message.command !== "string" || !("cwd" in message) || typeof message.cwd !== "string" || !("args" in message) || !Array.isArray(message.args) || !message.args.every((arg) => typeof arg === "string") || !("timeoutMs" in message) || typeof message.timeoutMs !== "number" || !Number.isFinite(message.timeoutMs) || message.timeoutMs <= 0) return fail("cursor-agent supervisor received invalid configuration");
|
|
35
|
+
clearTimeout(watchdog);
|
|
36
|
+
watchdog = setTimeout(() => fail("cursor-agent supervisor watchdog timed out"), message.timeoutMs + 1e3);
|
|
37
|
+
const command = message.command;
|
|
38
|
+
try {
|
|
39
|
+
child = spawn(command, message.args, {
|
|
40
|
+
cwd: message.cwd,
|
|
41
|
+
stdio: [
|
|
42
|
+
"pipe",
|
|
43
|
+
"inherit",
|
|
44
|
+
"inherit"
|
|
45
|
+
]
|
|
46
|
+
});
|
|
47
|
+
} catch (error) {
|
|
48
|
+
return fail(`failed to spawn ${command}: ${error instanceof Error ? error.message : String(error)}`);
|
|
49
|
+
}
|
|
50
|
+
child.on("error", (error) => fail(`failed to spawn ${command}: ${error.message}`));
|
|
51
|
+
child.stdin?.on("error", (error) => fail(`cursor-agent stdin failed: ${error.message}`));
|
|
52
|
+
process.stdin.on("error", (error) => fail(`cursor-agent prompt stream failed: ${error.message}`));
|
|
53
|
+
if (child.stdin) process.stdin.pipe(child.stdin);
|
|
54
|
+
child.on("exit", (code, signal) => {
|
|
55
|
+
if (terminal) return;
|
|
56
|
+
if (!child?.stdin?.writableFinished) return fail("cursor-agent stdin closed before prompt delivery");
|
|
57
|
+
terminal = true;
|
|
58
|
+
process.stdin.unpipe();
|
|
59
|
+
clearTimeout(watchdog);
|
|
60
|
+
watchdog = setTimeout(cleanup, 1e3);
|
|
61
|
+
process.send?.({
|
|
62
|
+
type: "terminal",
|
|
63
|
+
code,
|
|
64
|
+
signal
|
|
65
|
+
}, (error) => {
|
|
66
|
+
if (error) cleanup();
|
|
67
|
+
});
|
|
68
|
+
});
|
|
69
|
+
});
|
|
70
|
+
//#endregion
|
|
71
|
+
export {};
|
package/dist/cursor-provider.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { fork } from "node:child_process";
|
|
2
|
+
import path from "node:path";
|
|
2
3
|
//#region src/cursor-provider.ts
|
|
3
4
|
function newStreamState() {
|
|
4
5
|
return {
|
|
@@ -65,53 +66,114 @@ var CursorAgentProvider = class {
|
|
|
65
66
|
const stderr = [];
|
|
66
67
|
let stdoutBuf = "";
|
|
67
68
|
return new Promise((resolve) => {
|
|
68
|
-
const
|
|
69
|
-
|
|
70
|
-
|
|
69
|
+
const timeoutMs = this.config.timeout_ms ?? 9e5;
|
|
70
|
+
const supervisor = fork(new URL(`./cursor-process${path.extname(import.meta.url)}`, import.meta.url), {
|
|
71
|
+
execArgv: [],
|
|
72
|
+
detached: process.platform !== "win32",
|
|
71
73
|
stdio: [
|
|
72
74
|
"pipe",
|
|
73
75
|
"pipe",
|
|
74
|
-
"pipe"
|
|
76
|
+
"pipe",
|
|
77
|
+
"ipc"
|
|
75
78
|
]
|
|
76
79
|
});
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
});
|
|
86
|
-
child.stdout.on("data", (chunk) => {
|
|
87
|
-
stdoutBuf += chunk.toString("utf8");
|
|
88
|
-
const lines = stdoutBuf.split("\n");
|
|
89
|
-
stdoutBuf = lines.pop() ?? "";
|
|
90
|
-
for (const line of lines) foldLine(state, line);
|
|
91
|
-
});
|
|
92
|
-
child.stderr.on("data", (chunk) => {
|
|
93
|
-
stderr.push(chunk.toString("utf8"));
|
|
94
|
-
});
|
|
95
|
-
child.on("close", (code) => {
|
|
80
|
+
let failure;
|
|
81
|
+
let terminal;
|
|
82
|
+
let cleanupConfirmed = false;
|
|
83
|
+
let settled = false;
|
|
84
|
+
let cleanupTimer;
|
|
85
|
+
const finish = (closed, signal = null) => {
|
|
86
|
+
if (settled) return;
|
|
87
|
+
settled = true;
|
|
96
88
|
clearTimeout(timer);
|
|
89
|
+
clearTimeout(cleanupTimer);
|
|
90
|
+
supervisor.stdin?.destroy();
|
|
91
|
+
supervisor.stdout?.destroy();
|
|
92
|
+
supervisor.stderr?.destroy();
|
|
93
|
+
if (!closed) {
|
|
94
|
+
supervisor.kill("SIGKILL");
|
|
95
|
+
supervisor.unref();
|
|
96
|
+
if (supervisor.connected) supervisor.disconnect();
|
|
97
|
+
}
|
|
97
98
|
if (stdoutBuf !== "") foldLine(state, stdoutBuf);
|
|
98
|
-
|
|
99
|
+
const metadata = { skillCalls: state.skillCalls };
|
|
100
|
+
const fail = (error) => resolve({
|
|
101
|
+
error,
|
|
102
|
+
tokenUsage: state.tokenUsage,
|
|
103
|
+
metadata
|
|
104
|
+
});
|
|
105
|
+
if (failure !== void 0) {
|
|
99
106
|
const tail = stderr.join("").trim().slice(-2e3);
|
|
100
|
-
|
|
101
|
-
return;
|
|
107
|
+
const missingResult = terminal !== void 0 && state.result === void 0 ? " without a result event" : "";
|
|
108
|
+
return fail(`${failure}${missingResult}${tail === "" ? "" : `: ${tail}`}`);
|
|
102
109
|
}
|
|
103
|
-
if (
|
|
104
|
-
|
|
105
|
-
|
|
110
|
+
if (!closed || !cleanupConfirmed || terminal === void 0 || process.platform !== "win32" && signal !== "SIGKILL") return fail("cursor-agent supervisor ended without confirmed cleanup");
|
|
111
|
+
if (terminal.code !== 0 || terminal.signal !== null || state.result === void 0) {
|
|
112
|
+
const tail = stderr.join("").trim().slice(-2e3);
|
|
113
|
+
return fail(`cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}${state.result === void 0 ? " without a result event" : ""}${tail === "" ? "" : `: ${tail}`}`);
|
|
106
114
|
}
|
|
115
|
+
if (state.isError) return fail(state.result || "cursor-agent reported an error result");
|
|
107
116
|
resolve({
|
|
108
117
|
output: state.result,
|
|
109
118
|
tokenUsage: state.tokenUsage,
|
|
110
|
-
metadata
|
|
119
|
+
metadata
|
|
111
120
|
});
|
|
121
|
+
};
|
|
122
|
+
const boundCleanup = () => {
|
|
123
|
+
cleanupTimer ??= setTimeout(() => {
|
|
124
|
+
failure = failure === void 0 ? "cursor-agent cleanup or pipe draining timed out" : `${failure}; cleanup or pipe draining timed out`;
|
|
125
|
+
finish(false);
|
|
126
|
+
}, 2e3);
|
|
127
|
+
};
|
|
128
|
+
const stop = (error) => {
|
|
129
|
+
if (settled) return;
|
|
130
|
+
failure ??= error;
|
|
131
|
+
boundCleanup();
|
|
132
|
+
if (supervisor.connected) supervisor.send("stop", () => {});
|
|
133
|
+
};
|
|
134
|
+
const timer = setTimeout(() => stop(`cursor-agent timed out after ${timeoutMs}ms`), timeoutMs);
|
|
135
|
+
supervisor.on("error", (err) => stop(`cursor-agent supervisor failed: ${err.message}`));
|
|
136
|
+
supervisor.on("disconnect", () => {
|
|
137
|
+
if (!cleanupConfirmed && !settled) stop("cursor-agent supervisor disconnected before cleanup");
|
|
138
|
+
});
|
|
139
|
+
supervisor.on("message", (message) => {
|
|
140
|
+
if (settled) return;
|
|
141
|
+
if (typeof message !== "object" || message === null) return;
|
|
142
|
+
if (!("type" in message)) return;
|
|
143
|
+
if (message.type === "terminal" && "code" in message && "signal" in message && (message.code === null || typeof message.code === "number") && (message.signal === null || typeof message.signal === "string")) {
|
|
144
|
+
terminal = {
|
|
145
|
+
code: message.code,
|
|
146
|
+
signal: message.signal
|
|
147
|
+
};
|
|
148
|
+
clearTimeout(timer);
|
|
149
|
+
if (terminal.code !== 0 || terminal.signal !== null) failure ??= `cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}`;
|
|
150
|
+
boundCleanup();
|
|
151
|
+
supervisor.send("cleanup", (err) => {
|
|
152
|
+
if (err) stop(`cursor-agent cleanup request failed: ${err.message}`);
|
|
153
|
+
});
|
|
154
|
+
} else if (message.type === "failure" && "error" in message && typeof message.error === "string") stop(message.error);
|
|
155
|
+
else if (message.type === "cleanup" && terminal !== void 0) cleanupConfirmed = true;
|
|
156
|
+
});
|
|
157
|
+
supervisor.stdout?.on("data", (chunk) => {
|
|
158
|
+
stdoutBuf += chunk.toString("utf8");
|
|
159
|
+
const lines = stdoutBuf.split("\n");
|
|
160
|
+
stdoutBuf = lines.pop() ?? "";
|
|
161
|
+
for (const line of lines) foldLine(state, line);
|
|
162
|
+
});
|
|
163
|
+
supervisor.stderr?.on("data", (chunk) => stderr.push(chunk.toString("utf8")));
|
|
164
|
+
supervisor.stdout?.on("error", (err) => stop(`cursor-agent stdout failed: ${err.message}`));
|
|
165
|
+
supervisor.stderr?.on("error", (err) => stop(`cursor-agent stderr failed: ${err.message}`));
|
|
166
|
+
supervisor.stdin?.on("error", (err) => stop(`cursor-agent stdin failed: ${err.message}`));
|
|
167
|
+
supervisor.on("close", (_code, signal) => finish(true, signal));
|
|
168
|
+
supervisor.send({
|
|
169
|
+
command,
|
|
170
|
+
args,
|
|
171
|
+
cwd: this.config.working_dir,
|
|
172
|
+
timeoutMs
|
|
173
|
+
}, (err) => {
|
|
174
|
+
if (err) stop(`cursor-agent supervisor initialization failed: ${err.message}`);
|
|
112
175
|
});
|
|
113
|
-
|
|
114
|
-
child.stdin.end();
|
|
176
|
+
supervisor.stdin?.end(prompt);
|
|
115
177
|
});
|
|
116
178
|
}
|
|
117
179
|
};
|
package/docs/adoption.md
CHANGED
|
@@ -44,7 +44,7 @@ the registry.
|
|
|
44
44
|
Sweeps need model credentials, so they stay off consumer CI and run from an
|
|
45
45
|
operator machine or a job that already holds gateway auth. They also need the
|
|
46
46
|
eval engine, which is an optional peer precisely so the lint-only install
|
|
47
|
-
above stays small
|
|
47
|
+
above stays small. Install it next to the package on the operator machine:
|
|
48
48
|
|
|
49
49
|
```sh
|
|
50
50
|
pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
|
package/docs/usage.md
CHANGED
|
@@ -37,10 +37,10 @@ skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-
|
|
|
37
37
|
|
|
38
38
|
Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
|
|
39
39
|
installs the skill under test into that workdir, drives the agent, and grades
|
|
40
|
-
the files it wrote. Exit 0 pass, 1 graded fail, 2 error
|
|
41
|
-
usable
|
|
42
|
-
carries the exact `pnpm add` command; see
|
|
43
|
-
[adoption](adoption.md#evals)
|
|
40
|
+
the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
|
|
41
|
+
Exit 2 covers missing usable promptfoo output or optional eval peers. The
|
|
42
|
+
message carries the exact `pnpm add` command; see
|
|
43
|
+
[adoption](adoption.md#evals).
|
|
44
44
|
|
|
45
45
|
A test that errored was never graded, so it exits 2, prints the provider's
|
|
46
46
|
message, and writes no provenance sidecar. It is never reported as
|
|
@@ -56,7 +56,13 @@ the model to that CLI's own default.
|
|
|
56
56
|
cursor provider, so the run uses this package's own provider module, which
|
|
57
57
|
replays the CLI's `stream-json` output: the `result` event becomes the graded
|
|
58
58
|
output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
|
|
59
|
-
evidence.
|
|
59
|
+
evidence. Grading requires both a successful result event and harness exit code
|
|
60
|
+
zero. On macOS and Linux, a supervisor owns the process group and kills remaining
|
|
61
|
+
helpers when the harness exits, times out, or the provider disconnects. Output
|
|
62
|
+
pipes have a separate two-second cleanup/drain deadline; incomplete cleanup is
|
|
63
|
+
an error. Helpers that detach into another process group are outside this cleanup
|
|
64
|
+
boundary; retained output pipes still cause a bounded error. Windows retains
|
|
65
|
+
only direct-child cleanup. The judge leg is unchanged.
|
|
60
66
|
|
|
61
67
|
`--judge` takes either a bare Claude model (graded through the Anthropic
|
|
62
68
|
selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
|
|
@@ -87,7 +93,7 @@ discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
|
|
|
87
93
|
within one scenario, not across them.
|
|
88
94
|
|
|
89
95
|
One known failure mode: judge calls through a gateway can drop at the transport
|
|
90
|
-
layer ([uinaf/
|
|
96
|
+
layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
|
|
91
97
|
That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
|
|
92
98
|
mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
|
|
93
99
|
what is missing.
|
|
@@ -109,8 +115,16 @@ six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
|
|
|
109
115
|
six. A same-date file that cannot be parsed stops the write instead of being
|
|
110
116
|
overwritten.
|
|
111
117
|
|
|
112
|
-
Files that are not promptfoo results
|
|
113
|
-
failing the reduction.
|
|
118
|
+
Files that are not promptfoo results and ungraded transport errors are skipped
|
|
119
|
+
with a warning rather than failing the reduction. Graded assertion failures
|
|
120
|
+
remain scored results. If a skipped file matches an existing scorecard row,
|
|
121
|
+
summary generation fails and leaves the scorecard unchanged, so an errored rerun
|
|
122
|
+
cannot carry forward its old score. This also applies with `--allow-mixed`.
|
|
123
|
+
Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
|
|
124
|
+
are written. An outstanding marker makes `summarize` skip that identity even
|
|
125
|
+
when the child produced no result file or left partial output. The marker does
|
|
126
|
+
not count as a result for the sweep's existence check, so no-output failures
|
|
127
|
+
remain eligible for retry.
|
|
114
128
|
|
|
115
129
|
## Provenance
|
|
116
130
|
|
|
@@ -126,7 +140,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
|
|
|
126
140
|
```
|
|
127
141
|
|
|
128
142
|
`summarize` reads those sidecars and refuses to mix skills-tree revisions in one
|
|
129
|
-
scorecard
|
|
143
|
+
scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
|
|
144
|
+
Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
|
|
130
145
|
becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
|
|
131
146
|
`unattested`.
|
|
132
147
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uinaf/skillcheck",
|
|
3
|
-
"version": "0.5.
|
|
3
|
+
"version": "0.5.2",
|
|
4
4
|
"description": "Lint and eval harness for agent skills",
|
|
5
5
|
"homepage": "https://github.com/uinaf/skillcheck#readme",
|
|
6
6
|
"bugs": {
|
|
@@ -27,9 +27,10 @@
|
|
|
27
27
|
"registry": "https://registry.npmjs.org/"
|
|
28
28
|
},
|
|
29
29
|
"scripts": {
|
|
30
|
-
"verify": "vp
|
|
30
|
+
"verify": "vp run ready",
|
|
31
|
+
"verify:full": "vp run --no-cache ready",
|
|
31
32
|
"prepare": "vp config --no-agent",
|
|
32
|
-
"prepublishOnly": "pnpm run verify"
|
|
33
|
+
"prepublishOnly": "pnpm run verify:full"
|
|
33
34
|
},
|
|
34
35
|
"devDependencies": {
|
|
35
36
|
"@anthropic-ai/claude-agent-sdk": "^0.3.233",
|
|
@@ -58,5 +59,5 @@
|
|
|
58
59
|
"engines": {
|
|
59
60
|
"node": ">=24"
|
|
60
61
|
},
|
|
61
|
-
"packageManager": "pnpm@
|
|
62
|
+
"packageManager": "pnpm@12.0.0"
|
|
62
63
|
}
|