@uinaf/skillcheck 1.7.0 → 1.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +5 -0
- package/dist/skill-evidence.js +12 -0
- package/docs/usage.md +2 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -191,6 +191,11 @@ function classifyRow(raw, stats) {
|
|
|
191
191
|
const checklist = components.find((c) => c?.metadata?.assertionSet?.type === "assert-set");
|
|
192
192
|
const skillUsed = components.find((c) => c?.assertion?.type === "skill-used" || c?.assertion?.metric === "skill-used");
|
|
193
193
|
if (typeof checklist?.score !== "number" || typeof skillUsed?.score !== "number") return { error: "promptfoo result carried no checklist or skill-used verdict" };
|
|
194
|
+
const judgeError = (Array.isArray(checklist.componentResults) ? checklist.componentResults : []).find((c) => c?.metadata?.graderError === true);
|
|
195
|
+
if (judgeError) {
|
|
196
|
+
const reason = typeof judgeError.reason === "string" ? judgeError.reason.trim() : "";
|
|
197
|
+
return { error: `judge call failed${reason ? `: ${reason.slice(0, 300)}` : ""}` };
|
|
198
|
+
}
|
|
194
199
|
return {
|
|
195
200
|
score: checklist.score,
|
|
196
201
|
pass: res.success,
|
package/dist/skill-evidence.js
CHANGED
|
@@ -36,6 +36,18 @@ function skillEvidence(context) {
|
|
|
36
36
|
})) return `read of the installed ${skill}/SKILL.md`;
|
|
37
37
|
const opening = [...installed].map((p) => fs.readFileSync(p, "utf8").slice(0, 300).trim()).filter((t) => t.length >= 40);
|
|
38
38
|
if (calls(metadata?.toolCalls).find((c) => c.name === "Bash" && c.is_error === false && typeof c.output === "string" && opening.some((t) => c.output.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
|
|
39
|
+
if (codexCommandOutputs(context.providerResponse?.raw).some((out) => opening.some((t) => out.includes(t)))) return `shell read of the installed ${skill}/SKILL.md`;
|
|
40
|
+
}
|
|
41
|
+
function codexCommandOutputs(raw) {
|
|
42
|
+
let parsed = raw;
|
|
43
|
+
if (typeof raw === "string") try {
|
|
44
|
+
parsed = JSON.parse(raw);
|
|
45
|
+
} catch {
|
|
46
|
+
return [];
|
|
47
|
+
}
|
|
48
|
+
const items = parsed?.items;
|
|
49
|
+
if (!Array.isArray(items)) return [];
|
|
50
|
+
return items.flatMap((item) => item?.type === "command_execution" && typeof item.aggregated_output === "string" ? [item.aggregated_output] : []);
|
|
39
51
|
}
|
|
40
52
|
function assertSkillUsed(_output, context) {
|
|
41
53
|
const evidence = skillEvidence(context);
|
package/docs/usage.md
CHANGED
|
@@ -228,7 +228,8 @@ and keeps every row.
|
|
|
228
228
|
|
|
229
229
|
Files that are not promptfoo results and ungraded transport errors are skipped
|
|
230
230
|
with a warning rather than failing the reduction. Graded assertion failures
|
|
231
|
-
remain scored results.
|
|
231
|
+
remain scored results. A rubric item whose judge call failed (promptfoo tags it
|
|
232
|
+
`graderError`) errors the trial instead, since it judged nothing. If a skipped file matches an existing scorecard row,
|
|
232
233
|
summary generation fails and leaves the scorecard unchanged, so an errored rerun
|
|
233
234
|
cannot carry forward its old score. A graded result for the same identity
|
|
234
235
|
supersedes a skipped attempt only when the result file is newer. This also
|