@icntswm/skillcheck 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +149 -0
- package/dist/agents/claude.js +384 -0
- package/dist/agents/types.js +1 -0
- package/dist/batch.js +111 -0
- package/dist/budget.js +29 -0
- package/dist/cases.js +249 -0
- package/dist/cli.js +673 -0
- package/dist/confusion.js +45 -0
- package/dist/describe.js +282 -0
- package/dist/judge.js +53 -0
- package/dist/junit.js +84 -0
- package/dist/lexical.js +117 -0
- package/dist/lint.js +51 -0
- package/dist/pool.js +35 -0
- package/dist/report.js +93 -0
- package/dist/results.js +48 -0
- package/dist/version.js +7 -0
- package/package.json +47 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Confusion pairs over failed routing runs: which expected skill was missed
|
|
3
|
+
* and what the agent loaded instead. Error runs are skipped (they say nothing
|
|
4
|
+
* about routing); so are runs whose only failures concern forbid/none, since
|
|
5
|
+
* the expected-label rules below produce no label for them.
|
|
6
|
+
*/
|
|
7
|
+
export function confusion(results) {
|
|
8
|
+
const counts = new Map();
|
|
9
|
+
for (const res of results) {
|
|
10
|
+
for (const run of res.runs) {
|
|
11
|
+
if (run.ok || run.error)
|
|
12
|
+
continue;
|
|
13
|
+
for (const expected of expectedLabels(res.case, run.loaded)) {
|
|
14
|
+
for (const got of gotLabels(res.case, run.loaded)) {
|
|
15
|
+
const key = `${expected}\u0000${got}`;
|
|
16
|
+
const pair = counts.get(key);
|
|
17
|
+
if (pair)
|
|
18
|
+
pair.count++;
|
|
19
|
+
else
|
|
20
|
+
counts.set(key, { expected, got, count: 1 });
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
return [...counts.values()].sort((a, b) => b.count - a.count || a.expected.localeCompare(b.expected) || a.got.localeCompare(b.got));
|
|
26
|
+
}
|
|
27
|
+
/** A missed expectation; expect_any collapses to one "a|b" label. */
|
|
28
|
+
function expectedLabels(c, loaded) {
|
|
29
|
+
const labels = [];
|
|
30
|
+
for (const name of c.expect)
|
|
31
|
+
if (!loaded.includes(name))
|
|
32
|
+
labels.push(name);
|
|
33
|
+
if (c.expect_any.length > 0 && !c.expect_any.some((name) => loaded.includes(name))) {
|
|
34
|
+
labels.push(c.expect_any.join("|"));
|
|
35
|
+
}
|
|
36
|
+
if (c.first && loaded[0] !== c.first && !labels.includes(c.first))
|
|
37
|
+
labels.push(c.first);
|
|
38
|
+
return labels;
|
|
39
|
+
}
|
|
40
|
+
/** Loaded skills nobody asked for; "(nothing)" when the agent loaded none. */
|
|
41
|
+
function gotLabels(c, loaded) {
|
|
42
|
+
const allowed = new Set([...c.expect, ...c.expect_any, ...(c.first ? [c.first] : [])]);
|
|
43
|
+
const unexpected = loaded.filter((name) => !allowed.has(name));
|
|
44
|
+
return unexpected.length > 0 ? unexpected : ["(nothing)"];
|
|
45
|
+
}
|
package/dist/describe.js
ADDED
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
import * as fs from "node:fs";
|
|
2
|
+
import * as os from "node:os";
|
|
3
|
+
import * as path from "node:path";
|
|
4
|
+
import { parse as parseYaml } from "yaml";
|
|
5
|
+
/** Config roots to scan, earlier first: user config, then project (project overrides).
|
|
6
|
+
* An explicit configDir beats CLAUDE_CONFIG_DIR. */
|
|
7
|
+
export function skillRoots(opts) {
|
|
8
|
+
const home = opts?.home ?? os.homedir();
|
|
9
|
+
const cwd = opts?.cwd ?? process.cwd();
|
|
10
|
+
const configRoot = opts?.configDir || process.env.CLAUDE_CONFIG_DIR || path.join(home, ".claude");
|
|
11
|
+
return [configRoot, path.join(cwd, ".claude")];
|
|
12
|
+
}
|
|
13
|
+
/** Skill dir names under dir: subdirs with a SKILL.md, symlinks followed. */
|
|
14
|
+
export function skillDirNames(dir) {
|
|
15
|
+
const out = [];
|
|
16
|
+
let entries;
|
|
17
|
+
try {
|
|
18
|
+
entries = fs.readdirSync(dir);
|
|
19
|
+
}
|
|
20
|
+
catch {
|
|
21
|
+
return out;
|
|
22
|
+
}
|
|
23
|
+
for (const entry of entries) {
|
|
24
|
+
// statSync follows symlinks, so linked skill dirs count too
|
|
25
|
+
try {
|
|
26
|
+
if (fs.statSync(path.join(dir, entry)).isDirectory()
|
|
27
|
+
&& fs.statSync(path.join(dir, entry, "SKILL.md")).isFile()) {
|
|
28
|
+
out.push(entry);
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
catch {
|
|
32
|
+
// not a skill dir
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
return out;
|
|
36
|
+
}
|
|
37
|
+
/** Command names under dir: top-level .md files. */
|
|
38
|
+
export function commandNames(dir) {
|
|
39
|
+
const out = [];
|
|
40
|
+
let entries;
|
|
41
|
+
try {
|
|
42
|
+
entries = fs.readdirSync(dir);
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return out;
|
|
46
|
+
}
|
|
47
|
+
for (const entry of entries) {
|
|
48
|
+
if (!entry.endsWith(".md"))
|
|
49
|
+
continue;
|
|
50
|
+
try {
|
|
51
|
+
if (fs.statSync(path.join(dir, entry)).isFile())
|
|
52
|
+
out.push(entry.slice(0, -3));
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
// not a regular file
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
return out;
|
|
59
|
+
}
|
|
60
|
+
export function loadSkillDocs(opts) {
|
|
61
|
+
const cwd = opts?.cwd ?? process.cwd();
|
|
62
|
+
const byName = new Map();
|
|
63
|
+
const roots = skillRoots(opts);
|
|
64
|
+
for (const root of roots) {
|
|
65
|
+
const inRoot = new Map();
|
|
66
|
+
for (const name of commandNames(path.join(root, "commands"))) {
|
|
67
|
+
const file = path.join(root, "commands", `${name}.md`);
|
|
68
|
+
inRoot.set(name, { name, kind: "command", file, description: readDescription(file), plugin: null });
|
|
69
|
+
}
|
|
70
|
+
// within one root a skill beats a command with the same name
|
|
71
|
+
for (const name of skillDirNames(path.join(root, "skills"))) {
|
|
72
|
+
const file = path.join(root, "skills", name, "SKILL.md");
|
|
73
|
+
inRoot.set(name, { name, kind: "skill", file, description: readDescription(file), plugin: null });
|
|
74
|
+
}
|
|
75
|
+
// a later root overrides an earlier one
|
|
76
|
+
for (const [name, doc] of inRoot)
|
|
77
|
+
byName.set(name, doc);
|
|
78
|
+
}
|
|
79
|
+
// plugin docs carry a `plugin:` prefix, so they never collide with the above
|
|
80
|
+
for (const doc of pluginDocs(roots[0], cwd)) {
|
|
81
|
+
if (!byName.has(doc.name))
|
|
82
|
+
byName.set(doc.name, doc);
|
|
83
|
+
}
|
|
84
|
+
return [...byName.values()].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
|
|
85
|
+
}
|
|
86
|
+
/** Description from frontmatter plus when_to_use; never throws, "" when absent. */
|
|
87
|
+
function readDescription(file) {
|
|
88
|
+
return describeFrom(readFrontmatter(file));
|
|
89
|
+
}
|
|
90
|
+
function describeFrom(fm) {
|
|
91
|
+
const description = fm.description;
|
|
92
|
+
if (description === undefined)
|
|
93
|
+
return "";
|
|
94
|
+
const when = fm.whenToUse;
|
|
95
|
+
const joined = when !== undefined && when.length > 0 ? `${description} ${when}` : description;
|
|
96
|
+
return joined.replace(/\s+/g, " ").trim();
|
|
97
|
+
}
|
|
98
|
+
function readFrontmatter(file) {
|
|
99
|
+
let text;
|
|
100
|
+
try {
|
|
101
|
+
text = fs.readFileSync(file, "utf8");
|
|
102
|
+
}
|
|
103
|
+
catch {
|
|
104
|
+
return {};
|
|
105
|
+
}
|
|
106
|
+
const lines = text.split("\n");
|
|
107
|
+
if (lines[0]?.trim() !== "---")
|
|
108
|
+
return {};
|
|
109
|
+
const end = lines.findIndex((line, i) => i > 0 && line.trim() === "---");
|
|
110
|
+
if (end < 0)
|
|
111
|
+
return {};
|
|
112
|
+
return frontmatter(lines.slice(1, end));
|
|
113
|
+
}
|
|
114
|
+
function frontmatter(body) {
|
|
115
|
+
try {
|
|
116
|
+
// descriptions are often multi-line `>-` scalars, so this needs a real yaml parser
|
|
117
|
+
const parsed = parseYaml(body.join("\n"));
|
|
118
|
+
if (typeof parsed === "object" && parsed !== null && !Array.isArray(parsed)) {
|
|
119
|
+
const obj = parsed;
|
|
120
|
+
const out = {};
|
|
121
|
+
if (typeof obj.name === "string")
|
|
122
|
+
out.name = obj.name;
|
|
123
|
+
if (typeof obj.description === "string")
|
|
124
|
+
out.description = obj.description;
|
|
125
|
+
if (typeof obj.when_to_use === "string")
|
|
126
|
+
out.whenToUse = obj.when_to_use;
|
|
127
|
+
return out;
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
catch {
|
|
131
|
+
// fall through: Claude Code accepts files yaml does not
|
|
132
|
+
}
|
|
133
|
+
// real command files carry non-YAML lines like `argument-hint: [PROJ-1234] [repo]`
|
|
134
|
+
// that break the whole document; read the fields we need line by line instead
|
|
135
|
+
return {
|
|
136
|
+
name: readField(body, "name"),
|
|
137
|
+
description: readField(body, "description"),
|
|
138
|
+
whenToUse: readField(body, "when_to_use"),
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
const BLOCK_MARKERS = new Set(["", ">", ">-", "|", "|-"]);
|
|
142
|
+
/** Line-based value of a top-level frontmatter key; folded scalars continue below. */
|
|
143
|
+
function readField(lines, key) {
|
|
144
|
+
const prefix = `${key}:`;
|
|
145
|
+
for (let i = 0; i < lines.length; i++) {
|
|
146
|
+
const line = lines[i];
|
|
147
|
+
if (!line.startsWith(prefix))
|
|
148
|
+
continue;
|
|
149
|
+
const rest = line.slice(prefix.length).trim();
|
|
150
|
+
if (!BLOCK_MARKERS.has(rest))
|
|
151
|
+
return unquote(rest);
|
|
152
|
+
const parts = [];
|
|
153
|
+
for (let j = i + 1; j < lines.length; j++) {
|
|
154
|
+
const cont = lines[j];
|
|
155
|
+
if (cont.trim() === "")
|
|
156
|
+
continue;
|
|
157
|
+
if (cont !== cont.trimStart())
|
|
158
|
+
parts.push(cont.trim()); // indented deeper than the key
|
|
159
|
+
else
|
|
160
|
+
break;
|
|
161
|
+
}
|
|
162
|
+
return parts.length > 0 ? unquote(parts.join(" ")) : undefined;
|
|
163
|
+
}
|
|
164
|
+
return undefined;
|
|
165
|
+
}
|
|
166
|
+
function unquote(value) {
|
|
167
|
+
const first = value[0];
|
|
168
|
+
if ((first === '"' || first === "'") && value.length >= 2 && value[value.length - 1] === first) {
|
|
169
|
+
return value.slice(1, -1);
|
|
170
|
+
}
|
|
171
|
+
return value;
|
|
172
|
+
}
|
|
173
|
+
/** Plugin skills and commands registered under the user config root. */
|
|
174
|
+
function pluginDocs(configRoot, cwd) {
|
|
175
|
+
const registry = readJson(path.join(configRoot, "plugins", "installed_plugins.json"));
|
|
176
|
+
const plugins = registry?.plugins;
|
|
177
|
+
if (typeof plugins !== "object" || plugins === null)
|
|
178
|
+
return [];
|
|
179
|
+
const disabled = disabledPlugins(configRoot);
|
|
180
|
+
const out = [];
|
|
181
|
+
for (const [key, entries] of Object.entries(plugins)) {
|
|
182
|
+
if (disabled.has(key))
|
|
183
|
+
continue;
|
|
184
|
+
if (!Array.isArray(entries))
|
|
185
|
+
continue;
|
|
186
|
+
const plugin = key.split("@")[0]; // name is the part before the marketplace
|
|
187
|
+
for (const raw of entries) {
|
|
188
|
+
if (typeof raw !== "object" || raw === null)
|
|
189
|
+
continue;
|
|
190
|
+
const entry = raw;
|
|
191
|
+
const scoped = entry.scope === "project" || entry.scope === "local";
|
|
192
|
+
if (entry.scope !== "user" && !(scoped && entry.projectPath === cwd))
|
|
193
|
+
continue;
|
|
194
|
+
const installPath = entry.installPath;
|
|
195
|
+
if (typeof installPath !== "string" || installPath === "")
|
|
196
|
+
continue;
|
|
197
|
+
out.push(...pluginSkillDocs(plugin, installPath), ...pluginCommandDocs(plugin, installPath));
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
function disabledPlugins(configRoot) {
|
|
203
|
+
const enabled = readJson(path.join(configRoot, "settings.json"))?.enabledPlugins;
|
|
204
|
+
const out = new Set();
|
|
205
|
+
if (typeof enabled !== "object" || enabled === null)
|
|
206
|
+
return out;
|
|
207
|
+
for (const [key, value] of Object.entries(enabled)) {
|
|
208
|
+
if (value === false)
|
|
209
|
+
out.add(key);
|
|
210
|
+
}
|
|
211
|
+
return out;
|
|
212
|
+
}
|
|
213
|
+
function pluginSkillDocs(plugin, installPath) {
|
|
214
|
+
const declared = pluginManifest(installPath)?.skills;
|
|
215
|
+
const dirs = Array.isArray(declared)
|
|
216
|
+
? declared.filter((s) => typeof s === "string").map((rel) => path.resolve(installPath, rel))
|
|
217
|
+
: skillDirNames(path.join(installPath, "skills")).map((name) => path.join(installPath, "skills", name));
|
|
218
|
+
const out = [];
|
|
219
|
+
for (const dir of dirs) {
|
|
220
|
+
const file = path.join(dir, "SKILL.md");
|
|
221
|
+
if (!isFile(file))
|
|
222
|
+
continue;
|
|
223
|
+
const fm = readFrontmatter(file);
|
|
224
|
+
const name = fm.name !== undefined && fm.name.trim() !== "" ? fm.name.trim() : path.basename(dir);
|
|
225
|
+
out.push({ name: `${plugin}:${name}`, kind: "skill", file, description: describeFrom(fm), plugin });
|
|
226
|
+
}
|
|
227
|
+
return out;
|
|
228
|
+
}
|
|
229
|
+
function pluginCommandDocs(plugin, installPath) {
|
|
230
|
+
const declared = pluginManifest(installPath)?.commands;
|
|
231
|
+
let files;
|
|
232
|
+
if (Array.isArray(declared)) {
|
|
233
|
+
files = [];
|
|
234
|
+
for (const entry of declared) {
|
|
235
|
+
if (typeof entry !== "string")
|
|
236
|
+
continue;
|
|
237
|
+
const p = path.resolve(installPath, entry);
|
|
238
|
+
if (isFile(p)) {
|
|
239
|
+
if (p.endsWith(".md"))
|
|
240
|
+
files.push(p);
|
|
241
|
+
}
|
|
242
|
+
else {
|
|
243
|
+
files.push(...commandFiles(p));
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
else {
|
|
248
|
+
files = commandFiles(path.join(installPath, "commands"));
|
|
249
|
+
}
|
|
250
|
+
return files.map((file) => ({
|
|
251
|
+
name: `${plugin}:${path.basename(file).replace(/\.md$/, "")}`,
|
|
252
|
+
kind: "command",
|
|
253
|
+
file,
|
|
254
|
+
description: readDescription(file),
|
|
255
|
+
plugin,
|
|
256
|
+
}));
|
|
257
|
+
}
|
|
258
|
+
function commandFiles(dir) {
|
|
259
|
+
return commandNames(dir).map((name) => path.join(dir, `${name}.md`));
|
|
260
|
+
}
|
|
261
|
+
function pluginManifest(installPath) {
|
|
262
|
+
return readJson(path.join(installPath, ".claude-plugin", "plugin.json"));
|
|
263
|
+
}
|
|
264
|
+
function readJson(file) {
|
|
265
|
+
try {
|
|
266
|
+
const parsed = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
267
|
+
if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed))
|
|
268
|
+
return null;
|
|
269
|
+
return parsed;
|
|
270
|
+
}
|
|
271
|
+
catch {
|
|
272
|
+
return null; // missing or broken registry just means no plugins
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
function isFile(p) {
|
|
276
|
+
try {
|
|
277
|
+
return fs.statSync(p).isFile();
|
|
278
|
+
}
|
|
279
|
+
catch {
|
|
280
|
+
return false;
|
|
281
|
+
}
|
|
282
|
+
}
|
package/dist/judge.js
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
export const DIAGNOSIS = "skill named in text but not invoked — a model or directive limit, not routing";
|
|
2
|
+
export const DIAGNOSIS_LIMIT = "reproduce on a stronger model before editing descriptions";
|
|
3
|
+
export function judge(c, r) {
|
|
4
|
+
if (r.error) {
|
|
5
|
+
return {
|
|
6
|
+
ok: false, reasons: [r.error], reason: r.error, loaded: r.loaded,
|
|
7
|
+
diagnosis: null, costUsd: r.costUsd, error: r.error,
|
|
8
|
+
stoppedEarly: r.stoppedEarly, durationMs: r.durationMs,
|
|
9
|
+
};
|
|
10
|
+
}
|
|
11
|
+
const got = r.loaded;
|
|
12
|
+
const reasons = [];
|
|
13
|
+
for (const name of c.expect) {
|
|
14
|
+
if (!got.includes(name))
|
|
15
|
+
reasons.push(`not loaded ${name}`);
|
|
16
|
+
}
|
|
17
|
+
if (c.expect_any.length > 0 && !c.expect_any.some((name) => got.includes(name))) {
|
|
18
|
+
reasons.push(`not loaded any of [${c.expect_any.join(", ")}]`);
|
|
19
|
+
}
|
|
20
|
+
for (const name of c.forbid) {
|
|
21
|
+
if (got.includes(name))
|
|
22
|
+
reasons.push(`forbidden ${name}`);
|
|
23
|
+
}
|
|
24
|
+
if (c.first) {
|
|
25
|
+
if (got.length === 0)
|
|
26
|
+
reasons.push(`nothing loaded, expected ${c.first} first`);
|
|
27
|
+
else if (got[0] !== c.first)
|
|
28
|
+
reasons.push(`loaded ${got[0]} first, expected ${c.first}`);
|
|
29
|
+
}
|
|
30
|
+
if (c.none && got.length > 0) {
|
|
31
|
+
reasons.push(`expected nothing, loaded [${got.join(", ")}]`);
|
|
32
|
+
}
|
|
33
|
+
const unique = [...new Set(reasons)];
|
|
34
|
+
const ok = unique.length === 0;
|
|
35
|
+
let diagnosis = null;
|
|
36
|
+
if (!ok && got.length === 0) {
|
|
37
|
+
const text = r.text.toLowerCase();
|
|
38
|
+
const named = [...c.expect, ...c.expect_any].some((name) => text.includes(name.toLowerCase()));
|
|
39
|
+
if (named)
|
|
40
|
+
diagnosis = DIAGNOSIS;
|
|
41
|
+
}
|
|
42
|
+
return {
|
|
43
|
+
ok, reasons: unique, reason: unique.slice(0, 2).join("; "), loaded: got,
|
|
44
|
+
diagnosis, costUsd: r.costUsd, error: null,
|
|
45
|
+
stoppedEarly: r.stoppedEarly, durationMs: r.durationMs,
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
export function aggregate(c, runs, threshold) {
|
|
49
|
+
const passed = runs.filter((r) => r.ok).length;
|
|
50
|
+
// epsilon: 1/3 * 3 >= 1.0 must hold despite float noise
|
|
51
|
+
const ok = runs.length > 0 && passed / runs.length >= threshold - 1e-9;
|
|
52
|
+
return { case: c, runs, passed, ok, threshold };
|
|
53
|
+
}
|
package/dist/junit.js
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/** Codepoints that are not representable in XML 1.0 at all. */
|
|
2
|
+
const INVALID_XML = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\uFFFE\uFFFF]/g;
|
|
3
|
+
/**
|
|
4
|
+
* JUnit XML as GitLab and GitHub test reporters understand it. A run that
|
|
5
|
+
* errored is an <error> (the model never answered), a wrong routing is a
|
|
6
|
+
* <failure>.
|
|
7
|
+
*/
|
|
8
|
+
export function toJunit(report) {
|
|
9
|
+
const suiteName = `skillcheck.${report.agent}`;
|
|
10
|
+
const rows = report.cases.map((c) => ({ c, kind: classify(c) }));
|
|
11
|
+
const count = (kind) => rows.filter((r) => r.kind === kind).length;
|
|
12
|
+
const totals = `tests="${report.cases.length}" failures="${count("failed")}" errors="${count("errored")}"`;
|
|
13
|
+
const properties = [" <properties>", ` <property name="agent" value="${esc(report.agent)}"/>`];
|
|
14
|
+
if (report.model !== null)
|
|
15
|
+
properties.push(` <property name="model" value="${esc(report.model)}"/>`);
|
|
16
|
+
properties.push(` <property name="costUsd" value="${report.summary.costUsd.toFixed(2)}"/>`);
|
|
17
|
+
// early-stopped runs report no cost, so costUsd alone understates the spend
|
|
18
|
+
if (report.summary.unknownCostRuns > 0) {
|
|
19
|
+
properties.push(` <property name="unknownCostRuns" value="${report.summary.unknownCostRuns}"/>`);
|
|
20
|
+
}
|
|
21
|
+
properties.push(" </properties>");
|
|
22
|
+
const cases = rows.map(({ c, kind }) => testcase(c, kind, suiteName));
|
|
23
|
+
return [
|
|
24
|
+
'<?xml version="1.0" encoding="UTF-8"?>',
|
|
25
|
+
// Jenkins' schema allows `skipped` on <testsuite> only, not on <testsuites>
|
|
26
|
+
`<testsuites name="skillcheck" ${totals} time="${sec(report.durationMs)}">`,
|
|
27
|
+
` <testsuite name="${esc(suiteName)}" ${totals} skipped="${count("skipped")}" time="${sec(report.durationMs)}" timestamp="${esc(report.startedAt)}">`,
|
|
28
|
+
...properties,
|
|
29
|
+
...cases,
|
|
30
|
+
" </testsuite>",
|
|
31
|
+
"</testsuites>",
|
|
32
|
+
"",
|
|
33
|
+
].join("\n");
|
|
34
|
+
}
|
|
35
|
+
function classify(c) {
|
|
36
|
+
if (c.status !== "failed")
|
|
37
|
+
return c.status === "skipped" ? "skipped" : "passed";
|
|
38
|
+
const failed = c.runs.filter((r) => !r.ok);
|
|
39
|
+
return failed.length > 0 && failed.every((r) => r.error) ? "errored" : "failed";
|
|
40
|
+
}
|
|
41
|
+
function testcase(c, kind, suiteName) {
|
|
42
|
+
const time = sec(c.runs.reduce((ms, r) => ms + r.durationMs, 0));
|
|
43
|
+
const name = `#${c.id ?? c.index} ${collapse(c.query)}`;
|
|
44
|
+
const open = ` <testcase name="${esc(name)}" classname="${esc(suiteName)}" time="${time}">`;
|
|
45
|
+
if (kind === "passed")
|
|
46
|
+
return ` <testcase name="${esc(name)}" classname="${esc(suiteName)}" time="${time}"/>`;
|
|
47
|
+
if (kind === "skipped")
|
|
48
|
+
return `${open}\n <skipped message="budget reached"/>\n </testcase>`;
|
|
49
|
+
const error = kind === "errored";
|
|
50
|
+
const culprit = c.runs.find((r) => !r.ok && (error ? r.error : !r.error));
|
|
51
|
+
const tag = error ? "error" : "failure";
|
|
52
|
+
const type = error ? "run" : "routing";
|
|
53
|
+
const message = error ? culprit?.error ?? "" : culprit?.reason ?? "";
|
|
54
|
+
return `${open}\n <${tag} message="${esc(message)}" type="${type}">${esc(detail(c))}</${tag}>\n </testcase>`;
|
|
55
|
+
}
|
|
56
|
+
/** One line per run, then the diagnoses, then the case note. */
|
|
57
|
+
function detail(c) {
|
|
58
|
+
const lines = c.runs.map((r, i) => {
|
|
59
|
+
const head = `run ${i + 1}: ${r.ok ? "ok" : "FAIL"} loaded [${r.loaded.join(", ")}]`;
|
|
60
|
+
return r.ok ? head : `${head} · ${r.reason}`;
|
|
61
|
+
});
|
|
62
|
+
const diagnoses = [...new Set(c.runs.map((r) => r.diagnosis).filter((d) => d !== null))];
|
|
63
|
+
lines.push(...diagnoses.map((d) => `diagnosis: ${d}`));
|
|
64
|
+
if (c.note !== null)
|
|
65
|
+
lines.push(`note: ${c.note}`);
|
|
66
|
+
return lines.join("\n");
|
|
67
|
+
}
|
|
68
|
+
/** Milliseconds to seconds, at most three decimals, trailing zeros trimmed. */
|
|
69
|
+
function sec(ms) {
|
|
70
|
+
return String(Number((ms / 1000).toFixed(3)));
|
|
71
|
+
}
|
|
72
|
+
/** Collapse to one line without truncating; the full query must stay in the name. */
|
|
73
|
+
function collapse(text) {
|
|
74
|
+
return text.replace(/\s+/g, " ").trim();
|
|
75
|
+
}
|
|
76
|
+
function esc(text) {
|
|
77
|
+
return text
|
|
78
|
+
.replace(INVALID_XML, "")
|
|
79
|
+
.replace(/&/g, "&")
|
|
80
|
+
.replace(/</g, "<")
|
|
81
|
+
.replace(/>/g, ">")
|
|
82
|
+
.replace(/"/g, """)
|
|
83
|
+
.replace(/'/g, "'");
|
|
84
|
+
}
|
package/dist/lexical.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
// A tiny TF-IDF index: enough lexical signal to compare descriptions without
|
|
2
|
+
// pulling in a stemming or embeddings dependency.
|
|
3
|
+
const STOPWORDS = new Set([
|
|
4
|
+
// Russian and English function words carry no routing signal. MR, CI, DB, PR
|
|
5
|
+
// are two-letter but real, so tokenize keeps words of length >= 2 and the
|
|
6
|
+
// common two-letter function words are listed here. ё folds to е before the
|
|
7
|
+
// check, so the е spellings are listed.
|
|
8
|
+
"и", "в", "на", "не", "что", "это", "как", "для", "по", "из", "или", "при", "об", "но", "же",
|
|
9
|
+
"the", "and", "for", "with", "this", "that", "from", "into", "when", "use", "not", "are", "was",
|
|
10
|
+
"за", "до", "от", "то", "ли", "бы", "мы", "вы", "он", "она", "оно", "они", "мой", "мои", "моих",
|
|
11
|
+
"мое", "его", "ее", "их", "там", "тут", "вот", "уже", "еще", "так",
|
|
12
|
+
"to", "of", "in", "on", "is", "it", "an", "as", "at", "be", "by", "or", "if", "do", "we",
|
|
13
|
+
"you", "my", "me", "your", "our", "its",
|
|
14
|
+
]);
|
|
15
|
+
export function tokenize(text) {
|
|
16
|
+
// camelCase boundaries become spaces, and ё is treated as е
|
|
17
|
+
const spaced = text.replace(/[ёЁ]/g, (c) => (c === "ё" ? "е" : "Е")).replace(/([a-z0-9])([A-Z])/g, "$1 $2");
|
|
18
|
+
const words = spaced.toLowerCase().match(/[\p{L}\p{N}]+/gu) ?? [];
|
|
19
|
+
return words
|
|
20
|
+
.filter((w) => w.length >= 2 && !STOPWORDS.has(w))
|
|
21
|
+
// crude stem: 5 chars is enough to tell inflections of one word from
|
|
22
|
+
// different words, and a real stemmer is not worth the dependency
|
|
23
|
+
.map((w) => (w.length > 5 ? w.slice(0, 5) : w));
|
|
24
|
+
}
|
|
25
|
+
/** Doc text = description plus the name itself, split on the usual separators. */
|
|
26
|
+
function docText(name, text) {
|
|
27
|
+
return `${text} ${name.split(/[-_:]/).join(" ")}`;
|
|
28
|
+
}
|
|
29
|
+
function termFreq(tokens) {
|
|
30
|
+
const tf = new Map();
|
|
31
|
+
for (const t of tokens)
|
|
32
|
+
tf.set(t, (tf.get(t) ?? 0) + 1);
|
|
33
|
+
return tf;
|
|
34
|
+
}
|
|
35
|
+
function normalize(vec) {
|
|
36
|
+
let sum = 0;
|
|
37
|
+
for (const w of vec.values())
|
|
38
|
+
sum += w * w;
|
|
39
|
+
const norm = Math.sqrt(sum);
|
|
40
|
+
if (norm === 0)
|
|
41
|
+
return; // empty vector stays all-zero, cosine gives 0
|
|
42
|
+
for (const [t, w] of vec)
|
|
43
|
+
vec.set(t, w / norm);
|
|
44
|
+
}
|
|
45
|
+
function cosine(a, b) {
|
|
46
|
+
const [small, big] = a.size <= b.size ? [a, b] : [b, a];
|
|
47
|
+
let dot = 0;
|
|
48
|
+
for (const [t, w] of small) {
|
|
49
|
+
const wb = big.get(t);
|
|
50
|
+
if (wb !== undefined)
|
|
51
|
+
dot += w * wb;
|
|
52
|
+
}
|
|
53
|
+
return dot;
|
|
54
|
+
}
|
|
55
|
+
export class LexicalIndex {
|
|
56
|
+
vectors = new Map();
|
|
57
|
+
idf = new Map();
|
|
58
|
+
constructor(docs) {
|
|
59
|
+
const df = new Map();
|
|
60
|
+
const tfs = docs.map((d) => {
|
|
61
|
+
const tf = termFreq(tokenize(docText(d.name, d.text)));
|
|
62
|
+
for (const t of tf.keys())
|
|
63
|
+
df.set(t, (df.get(t) ?? 0) + 1);
|
|
64
|
+
return { name: d.name, tf };
|
|
65
|
+
});
|
|
66
|
+
const n = docs.length;
|
|
67
|
+
for (const [t, count] of df)
|
|
68
|
+
this.idf.set(t, Math.log(1 + n / count));
|
|
69
|
+
for (const { name, tf } of tfs) {
|
|
70
|
+
const vec = new Map();
|
|
71
|
+
for (const [t, count] of tf)
|
|
72
|
+
vec.set(t, count * (this.idf.get(t) ?? 0));
|
|
73
|
+
normalize(vec);
|
|
74
|
+
this.vectors.set(name, vec);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
rank(query) {
|
|
78
|
+
const q = this.vectorize(tokenize(query));
|
|
79
|
+
const out = [...this.vectors.entries()].map(([name, vec]) => ({ name, score: cosine(q, vec) }));
|
|
80
|
+
out.sort((a, b) => b.score - a.score || (a.name < b.name ? -1 : 1));
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
83
|
+
similarity(a, b) {
|
|
84
|
+
const va = this.vectors.get(a);
|
|
85
|
+
const vb = this.vectors.get(b);
|
|
86
|
+
if (!va || !vb)
|
|
87
|
+
return 0;
|
|
88
|
+
return cosine(va, vb);
|
|
89
|
+
}
|
|
90
|
+
pairs(minScore) {
|
|
91
|
+
const names = [...this.vectors.keys()];
|
|
92
|
+
const out = [];
|
|
93
|
+
for (let i = 0; i < names.length; i++) {
|
|
94
|
+
for (let j = i + 1; j < names.length; j++) {
|
|
95
|
+
let a = names[i];
|
|
96
|
+
let b = names[j];
|
|
97
|
+
if (a > b)
|
|
98
|
+
[a, b] = [b, a];
|
|
99
|
+
const score = this.similarity(a, b);
|
|
100
|
+
if (score >= minScore)
|
|
101
|
+
out.push({ a, b, score });
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
out.sort((x, y) => y.score - x.score || (x.a < y.a ? -1 : x.a > y.a ? 1 : x.b < y.b ? -1 : x.b > y.b ? 1 : 0));
|
|
105
|
+
return out;
|
|
106
|
+
}
|
|
107
|
+
vectorize(tokens) {
|
|
108
|
+
const vec = new Map();
|
|
109
|
+
for (const [t, count] of termFreq(tokens)) {
|
|
110
|
+
const idf = this.idf.get(t);
|
|
111
|
+
if (idf !== undefined)
|
|
112
|
+
vec.set(t, count * idf); // unknown words carry no signal
|
|
113
|
+
}
|
|
114
|
+
normalize(vec);
|
|
115
|
+
return vec;
|
|
116
|
+
}
|
|
117
|
+
}
|
package/dist/lint.js
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { LexicalIndex } from "./lexical.js";
|
|
2
|
+
// measured on a real config: a true near-duplicate pair scored 0.42, a known
|
|
3
|
+
// confusion pair 0.20, so 0.3 separates them
|
|
4
|
+
export const LINT_DEFAULTS = { top: 5, overlap: 0.3, minLength: 40 };
|
|
5
|
+
export function lint(docs, suite, opts) {
|
|
6
|
+
const index = new LexicalIndex(docs.map((d) => ({ name: d.name, text: d.description })));
|
|
7
|
+
const short = docs
|
|
8
|
+
.filter((d) => d.description.length < opts.minLength)
|
|
9
|
+
.map((d) => ({ name: d.name, kind: d.kind, length: d.description.length }));
|
|
10
|
+
const similar = index.pairs(opts.overlap);
|
|
11
|
+
const far = [];
|
|
12
|
+
const covered = new Set();
|
|
13
|
+
if (suite) {
|
|
14
|
+
for (const c of suite.cases) {
|
|
15
|
+
for (const name of [...c.expect, ...c.expect_any, ...c.forbid, ...(c.first ? [c.first] : [])])
|
|
16
|
+
covered.add(name);
|
|
17
|
+
if (c.none)
|
|
18
|
+
continue; // a none case expects nothing lexical
|
|
19
|
+
far.push(...farForCase(c, index, opts.top));
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
// with no suite there is no evidence either way, so uncovered stays empty;
|
|
23
|
+
// plugin skills are someone else's, they are not required to have cases
|
|
24
|
+
const uncovered = suite
|
|
25
|
+
? docs.filter((d) => d.kind === "skill" && d.plugin === null && !covered.has(d.name)).map((d) => d.name).sort()
|
|
26
|
+
: [];
|
|
27
|
+
return { docs: docs.length, cases: suite?.cases.length ?? 0, short, similar, far, uncovered };
|
|
28
|
+
}
|
|
29
|
+
function farForCase(c, index, top) {
|
|
30
|
+
const ranking = index.rank(c.query);
|
|
31
|
+
const byName = new Map(ranking.map((r, i) => [r.name, i + 1]));
|
|
32
|
+
const targets = c.expect.map((name) => ({ label: name, members: [name] }));
|
|
33
|
+
// the whole expect_any is one target, ranked by its best member
|
|
34
|
+
if (c.expect_any.length > 0)
|
|
35
|
+
targets.push({ label: c.expect_any.join("|"), members: c.expect_any });
|
|
36
|
+
if (c.first !== undefined && !c.expect.includes(c.first))
|
|
37
|
+
targets.push({ label: c.first, members: [c.first] });
|
|
38
|
+
const out = [];
|
|
39
|
+
for (const t of targets) {
|
|
40
|
+
// names without a doc (built-ins, typos) cannot be ranked; plugin `x:y`
|
|
41
|
+
// names have docs once plugins are loaded and are ranked like any other
|
|
42
|
+
const ranks = t.members.map((m) => byName.get(m)).filter((r) => r !== undefined);
|
|
43
|
+
if (ranks.length === 0)
|
|
44
|
+
continue;
|
|
45
|
+
const rank = Math.min(...ranks);
|
|
46
|
+
if (rank > top) {
|
|
47
|
+
out.push({ index: c.index, id: c.id ?? null, query: c.query, expected: t.label, rank, top: ranking.slice(0, 3).map((r) => r.name) });
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
package/dist/pool.js
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
export async function runPool(items, concurrency, worker, opts) {
|
|
2
|
+
const cap = Math.max(1, Math.floor(concurrency) || 1);
|
|
3
|
+
const results = new Array(items.length).fill(undefined);
|
|
4
|
+
const inflight = new Set();
|
|
5
|
+
let next = 0;
|
|
6
|
+
let failed = false;
|
|
7
|
+
let firstError;
|
|
8
|
+
const canStart = () => !failed && next < items.length && (!opts?.shouldStart || opts.shouldStart());
|
|
9
|
+
const start = () => {
|
|
10
|
+
const i = next++;
|
|
11
|
+
const task = (async () => {
|
|
12
|
+
try {
|
|
13
|
+
results[i] = await worker(items[i], i);
|
|
14
|
+
}
|
|
15
|
+
catch (e) {
|
|
16
|
+
if (!failed) {
|
|
17
|
+
failed = true;
|
|
18
|
+
firstError = e;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
})();
|
|
22
|
+
const slot = task.then(() => { inflight.delete(slot); });
|
|
23
|
+
inflight.add(slot);
|
|
24
|
+
};
|
|
25
|
+
while (canStart() && inflight.size < cap)
|
|
26
|
+
start();
|
|
27
|
+
while (inflight.size > 0) {
|
|
28
|
+
await Promise.race(inflight);
|
|
29
|
+
while (canStart() && inflight.size < cap)
|
|
30
|
+
start();
|
|
31
|
+
}
|
|
32
|
+
if (failed)
|
|
33
|
+
throw firstError;
|
|
34
|
+
return results;
|
|
35
|
+
}
|