@uinaf/skillcheck 0.5.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -107,9 +107,9 @@ function promptfooEntry() {
107
107
  function classifyResult(raw) {
108
108
  const root = raw;
109
109
  const res = root?.results?.results?.[0];
110
- if (res === void 0) return { error: "promptfoo output carried no result" };
110
+ if (res === void 0 || res === null) return { error: "promptfoo output carried no result" };
111
111
  const message = typeof res.error === "string" ? res.error.trim() : "";
112
- const stats = root?.results?.stats;
112
+ const stats = root?.results?.stats ?? void 0;
113
113
  const gradedByStats = stats !== void 0 && (stats.errors ?? 0) === 0 && ((stats.failures ?? 0) > 0 || (stats.successes ?? 0) > 0);
114
114
  if (message !== "" && !gradedByStats) return { error: message };
115
115
  if (stats !== void 0 && (stats.errors ?? 0) > 0 && (stats.successes ?? 0) === 0 && (stats.failures ?? 0) === 0) return { error: "promptfoo reported an errored test with nothing graded" };
@@ -128,6 +128,9 @@ function gitHead(root) {
128
128
  function metaPath(resultPath) {
129
129
  return resultPath.replace(/\.json$/, ".meta.json");
130
130
  }
131
+ function attemptPath(resultPath) {
132
+ return `${resultPath}.attempt`;
133
+ }
131
134
  function runScenario(scenarioDir, opts, root) {
132
135
  const dirs = stateDirs(root);
133
136
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
@@ -137,6 +140,7 @@ function runScenario(scenarioDir, opts, root) {
137
140
  });
138
141
  fs.mkdirSync(dirs.results, { recursive: true });
139
142
  const resultPath = path.join(dirs.results, `${name}.json`);
143
+ fs.writeFileSync(attemptPath(resultPath), "{}\n");
140
144
  fs.rmSync(resultPath, { force: true });
141
145
  fs.rmSync(metaPath(resultPath), { force: true });
142
146
  const sha = gitHead(root);
@@ -174,12 +178,15 @@ function runScenario(scenarioDir, opts, root) {
174
178
  outcome.score = verdict.score;
175
179
  outcome.pass = verdict.pass;
176
180
  outcome.error = verdict.error;
177
- if (verdict.score !== void 0) fs.writeFileSync(metaPath(resultPath), JSON.stringify({
178
- skills_tree_sha: sha,
179
- harness: opts.harness,
180
- ran_at: (/* @__PURE__ */ new Date()).toISOString(),
181
- tool_version: toolVersion()
182
- }, null, 2) + "\n");
181
+ if (verdict.score !== void 0) {
182
+ fs.writeFileSync(metaPath(resultPath), JSON.stringify({
183
+ skills_tree_sha: sha,
184
+ harness: opts.harness,
185
+ ran_at: (/* @__PURE__ */ new Date()).toISOString(),
186
+ tool_version: toolVersion()
187
+ }, null, 2) + "\n");
188
+ fs.rmSync(attemptPath(resultPath), { force: true });
189
+ }
183
190
  return outcome;
184
191
  }
185
192
  function discoverScenarios(root) {
@@ -247,11 +254,30 @@ function cmdSweep(argv) {
247
254
  console.log(`\nsweep: ${passed} passed, ${failed} failed, ${errored} errored, ${skipped} skipped`);
248
255
  process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
249
256
  }
257
+ function resultIdentity(file) {
258
+ const base = file.replace(/\.json$/, "");
259
+ const suffix = base.match(/--(codex|cursor)$/);
260
+ const harness = suffix === null ? "claude" : suffix[1];
261
+ const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
262
+ return {
263
+ skill,
264
+ scenario: rest.join("--"),
265
+ harness
266
+ };
267
+ }
250
268
  function reduceResults(dir, allowMixed) {
251
269
  const entries = [];
252
270
  const skipped = [];
253
271
  const shas = /* @__PURE__ */ new Set();
254
- for (const f of fs.readdirSync(dir).filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json")).sort()) {
272
+ const files = fs.readdirSync(dir);
273
+ const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
274
+ const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
275
+ for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
276
+ if (incomplete.has(f)) {
277
+ console.error(`skipping ${f}: attempt did not complete with a graded result`);
278
+ skipped.push(f);
279
+ continue;
280
+ }
255
281
  let raw;
256
282
  try {
257
283
  raw = JSON.parse(fs.readFileSync(path.join(dir, f), "utf8"));
@@ -259,17 +285,16 @@ function reduceResults(dir, allowMixed) {
259
285
  raw = void 0;
260
286
  }
261
287
  const res = raw?.results?.results?.[0];
262
- if (typeof res?.score !== "number" || typeof res?.success !== "boolean") {
263
- console.error(`skipping ${f}: not a promptfoo result`);
288
+ const verdict = classifyResult(raw);
289
+ if (verdict.score === void 0 || verdict.pass === void 0) {
290
+ console.error(`skipping ${f}: ${verdict.error}`);
264
291
  skipped.push(f);
265
292
  continue;
266
293
  }
267
294
  const provider = raw.config?.providers?.[0];
268
295
  const judge = raw.config?.defaultTest?.options?.provider;
269
296
  const base = f.replace(/\.json$/, "");
270
- const suffix = base.match(/--(codex|cursor)$/);
271
- const harness = suffix === null ? "claude" : suffix[1];
272
- const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
297
+ const { skill, scenario, harness } = resultIdentity(f);
273
298
  let sha = "unattested";
274
299
  try {
275
300
  sha = JSON.parse(fs.readFileSync(path.join(dir, `${base}.meta.json`), "utf8")).skills_tree_sha ?? "unattested";
@@ -277,11 +302,11 @@ function reduceResults(dir, allowMixed) {
277
302
  shas.add(sha);
278
303
  entries.push({
279
304
  skill,
280
- scenario: rest.join("--"),
305
+ scenario,
281
306
  harness,
282
307
  skills_tree_sha: sha,
283
- score: res.score,
284
- pass: res.success,
308
+ score: verdict.score,
309
+ pass: verdict.pass,
285
310
  agent_model: provider?.config?.model ?? `${harness}-default`,
286
311
  judge_model: typeof judge === "string" ? judge.replace(/^anthropic:messages:/, "") : judge?.config?.model ?? judge?.id ?? "unknown",
287
312
  latency_ms: res.latencyMs,
@@ -335,15 +360,19 @@ function cmdSummarize(argv) {
335
360
  const { positional, flags } = parseArgs(argv);
336
361
  if (positional.length > 0) fail("usage: skillcheck summarize [--root DIR] [--allow-mixed]");
337
362
  const dirs = stateDirs(resolveRoot(flags));
338
- if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results} — run some evals first`);
363
+ if (!fs.existsSync(dirs.results)) fail(`no results directory at ${dirs.results}; run some evals first`);
339
364
  const { entries, skipped } = reduceResults(dirs.results, flags.get("--allow-mixed") === true);
340
365
  fs.mkdirSync(dirs.scorecards, { recursive: true });
341
366
  const out = path.join(dirs.scorecards, `${(/* @__PURE__ */ new Date()).toISOString().slice(0, 10)}.json`);
342
367
  const existing = readExistingScorecard(out);
368
+ const skippedKeys = new Set(skipped.map((file) => entryKey(resultIdentity(file))));
369
+ if (existing.some((entry) => skippedKeys.has(entryKey(entry)))) throw new Error("skipped rerun matches an existing score; refusing to carry it or overwrite the scorecard");
343
370
  const merged = mergeScorecard(existing, entries);
371
+ const treeSha = treeShaOf(merged.entries);
372
+ if (treeSha === "mixed" && flags.get("--allow-mixed") !== true) throw new Error("scorecard spans multiple skills-tree revisions; rerun stale ones or pass --allow-mixed");
344
373
  const scorecard = {
345
374
  ran_at: (/* @__PURE__ */ new Date()).toISOString(),
346
- skills_tree_sha: treeShaOf(merged.entries),
375
+ skills_tree_sha: treeSha,
347
376
  scenarios: merged.entries
348
377
  };
349
378
  fs.writeFileSync(out, JSON.stringify(scorecard, null, 2) + "\n");
package/docs/adoption.md CHANGED
@@ -44,7 +44,7 @@ the registry.
44
44
  Sweeps need model credentials, so they stay off consumer CI and run from an
45
45
  operator machine or a job that already holds gateway auth. They also need the
46
46
  eval engine, which is an optional peer precisely so the lint-only install
47
- above stays small — install it next to the package on the operator machine:
47
+ above stays small. Install it next to the package on the operator machine:
48
48
 
49
49
  ```sh
50
50
  pnpm add -D promptfoo @anthropic-ai/claude-agent-sdk @openai/codex-sdk
package/docs/usage.md CHANGED
@@ -37,10 +37,10 @@ skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-
37
37
 
38
38
  Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
39
39
  installs the skill under test into that workdir, drives the agent, and grades
40
- the files it wrote. Exit 0 pass, 1 graded fail, 2 error (promptfoo produced no
41
- usable result, or the optional eval peers are not installed — the message
42
- carries the exact `pnpm add` command; see
43
- [adoption](adoption.md#evals)).
40
+ the files it wrote. Exit 0 means pass, 1 means graded fail, and 2 means error.
41
+ Exit 2 covers missing usable promptfoo output or optional eval peers. The
42
+ message carries the exact `pnpm add` command; see
43
+ [adoption](adoption.md#evals).
44
44
 
45
45
  A test that errored was never graded, so it exits 2, prints the provider's
46
46
  message, and writes no provenance sidecar. It is never reported as
@@ -87,7 +87,7 @@ discovered. Exit 2 if anything errored, 1 if anything failed, else 0.
87
87
  within one scenario, not across them.
88
88
 
89
89
  One known failure mode: judge calls through a gateway can drop at the transport
90
- layer ([uinaf/agent-platform#28](https://github.com/uinaf/agent-platform/issues/28)).
90
+ layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
91
91
  That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
92
92
  mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
93
93
  what is missing.
@@ -109,8 +109,16 @@ six of twenty-nine scenarios therefore leaves twenty-nine rows in the file, not
109
109
  six. A same-date file that cannot be parsed stops the write instead of being
110
110
  overwritten.
111
111
 
112
- Files that are not promptfoo results are skipped with a warning rather than
113
- failing the reduction.
112
+ Files that are not promptfoo results and ungraded transport errors are skipped
113
+ with a warning rather than failing the reduction. Graded assertion failures
114
+ remain scored results. If a skipped file matches an existing scorecard row,
115
+ summary generation fails and leaves the scorecard unchanged, so an errored rerun
116
+ cannot carry forward its old score. This also applies with `--allow-mixed`.
117
+ Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
118
+ are written. An outstanding marker makes `summarize` skip that identity even
119
+ when the child produced no result file or left partial output. The marker does
120
+ not count as a result for the sweep's existence check, so no-output failures
121
+ remain eligible for retry.
114
122
 
115
123
  ## Provenance
116
124
 
@@ -126,7 +134,8 @@ Each successful run writes a `<name>.meta.json` sidecar next to its result:
126
134
  ```
127
135
 
128
136
  `summarize` reads those sidecars and refuses to mix skills-tree revisions in one
129
- scorecard unless `--allow-mixed`, in which case the top-level `skills_tree_sha`
137
+ scorecard, including retained rows from partial reruns, unless `--allow-mixed`.
138
+ Rejection leaves the existing scorecard unchanged. With the override, the top-level `skills_tree_sha`
130
139
  becomes `mixed` and per-entry shas remain. A result with no sidecar reduces as
131
140
  `unattested`.
132
141
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.5.0",
3
+ "version": "0.5.1",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -27,9 +27,10 @@
27
27
  "registry": "https://registry.npmjs.org/"
28
28
  },
29
29
  "scripts": {
30
- "verify": "vp check && vp pack && vp test run",
30
+ "verify": "vp run ready",
31
+ "verify:full": "vp run --no-cache ready",
31
32
  "prepare": "vp config --no-agent",
32
- "prepublishOnly": "pnpm run verify"
33
+ "prepublishOnly": "pnpm run verify:full"
33
34
  },
34
35
  "devDependencies": {
35
36
  "@anthropic-ai/claude-agent-sdk": "^0.3.233",
@@ -58,5 +59,5 @@
58
59
  "engines": {
59
60
  "node": ">=24"
60
61
  },
61
- "packageManager": "pnpm@11.22.0"
62
+ "packageManager": "pnpm@12.0.0"
62
63
  }