@uinaf/skillcheck 1.7.0 → 1.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -191,6 +191,11 @@ function classifyRow(raw, stats) {
191
191
  const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
192
192
  const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
193
193
  if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
194
+ const judgeError = (Array.isArray(checklist.componentResults) ? checklist.componentResults : []).find((c) => c?.metadata?.graderError === true);
195
+ if (judgeError) {
196
+ const reason = typeof judgeError.reason === "string" ? judgeError.reason.trim() : "";
197
+ return { error: `judge call failed${reason ? `: ${reason.slice(0, 300)}` : ""}` };
198
+ }
194
199
  return {
195
200
  score: checklist.score,
196
201
  pass: res.success,
@@ -36,6 +36,18 @@ function skillEvidence(context) {
36
36
  })) return `read of the installed ${skill}/SKILL.md`;
37
37
  const opening = [...installed].map((p) => fs.readFileSync(p, "utf8").slice(0, 300).trim()).filter((t) => t.length >= 40);
38
38
  if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.output === "string" && opening.some((t) => c.output.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
39
+ if (codexCommandOutputs(context.providerResponse?.raw).some((out) => opening.some((t) => out.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
40
+ }
41
+ function codexCommandOutputs(raw) {
42
+ let parsed = raw;
43
+ if (typeof raw === "string") try {
44
+ parsed = JSON.parse(raw);
45
+ } catch {
46
+ return [];
47
+ }
48
+ const items = parsed?.items;
49
+ if (!Array.isArray(items)) return [];
50
+ return items.flatMap((item) => item?.type === "command_execution" && typeof item.aggregated_output === "string" ? [item.aggregated_output] : []);
39
51
  }
40
52
  function assertSkillUsed(_output, context) {
41
53
  const evidence = skillEvidence(context);
package/docs/usage.md CHANGED
@@ -228,7 +228,8 @@ and keeps every row.
228
228
 
229
229
  Files that are not promptfoo results and ungraded transport errors are skipped
230
230
  with a warning rather than failing the reduction. Graded assertion failures
231
- remain scored results. If a skipped file matches an existing scorecard row,
231
+ remain scored results. A rubric item whose judge call failed (promptfoo tags it
232
+ `graderError`) errors the trial instead, since it judged nothing. If a skipped file matches an existing scorecard row,
232
233
  summary generation fails and leaves the scorecard unchanged, so an errored rerun
233
234
  cannot carry forward its old score. A graded result for the same identity
234
235
  supersedes a skipped attempt only when the result file is newer. This also
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "1.7.0",
3
+ "version": "1.7.2",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {