skilltrigger 0.0.0-stage → 0.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +86 -0
- package/LICENSE +21 -0
- package/README.md +109 -2
- package/bin/lib/args.mjs +37 -0
- package/bin/lib/changelog.mjs +37 -0
- package/bin/lib/check.mjs +106 -0
- package/bin/lib/claude.mjs +106 -0
- package/bin/lib/compare.mjs +102 -0
- package/bin/lib/environment.mjs +78 -0
- package/bin/lib/gates.mjs +186 -0
- package/bin/lib/plugins.mjs +106 -0
- package/bin/lib/report.mjs +160 -0
- package/bin/lib/runner.mjs +79 -0
- package/bin/lib/skill.mjs +87 -0
- package/bin/lib/stream.mjs +150 -0
- package/bin/lib/toggles.mjs +62 -0
- package/bin/lib/traps.mjs +12 -0
- package/bin/skilltrigger.mjs +196 -0
- package/package.json +84 -4
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
// The two inputs: a skill's SKILL.md (its name and description) and an eval set in
|
|
2
|
+
// skill-creator's format — a JSON array of { query, should_trigger }, any other field
|
|
3
|
+
// carried through untouched.
|
|
4
|
+
import { existsSync, readFileSync, statSync } from 'node:fs'
|
|
5
|
+
import { join } from 'node:path'
|
|
6
|
+
|
|
7
|
+
// Enough YAML for skill frontmatter: top-level `key: value`, single- and double-quoted
|
|
8
|
+
// scalars, and `|` / `>` block scalars (with their -/+ chomping indicators accepted).
|
|
9
|
+
// Anything nested is skipped — skilltrigger needs two keys.
|
|
10
|
+
export function parseFrontmatter(text) {
|
|
11
|
+
const m = String(text).match(/^---\r?\n([\s\S]*?)\r?\n---\s*(?:\r?\n|$)/)
|
|
12
|
+
if (!m) return null
|
|
13
|
+
const lines = m[1].split(/\r?\n/)
|
|
14
|
+
const out = {}
|
|
15
|
+
for (let i = 0; i < lines.length; i++) {
|
|
16
|
+
const kv = lines[i].match(/^([A-Za-z_][\w-]*):\s*(.*)$/)
|
|
17
|
+
if (!kv) continue
|
|
18
|
+
const [, key, rest] = kv
|
|
19
|
+
const block = rest.match(/^([|>])[-+]?\s*$/)
|
|
20
|
+
if (block) {
|
|
21
|
+
const body = []
|
|
22
|
+
while (i + 1 < lines.length && (/^\s/.test(lines[i + 1]) || lines[i + 1] === '')) body.push(lines[++i])
|
|
23
|
+
const indent = Math.min(...body.filter((l) => l.trim()).map((l) => l.match(/^\s*/)[0].length))
|
|
24
|
+
const stripped = body.map((l) => l.slice(indent))
|
|
25
|
+
out[key] = block[1] === '|' ? stripped.join('\n').trim() : fold(stripped)
|
|
26
|
+
continue
|
|
27
|
+
}
|
|
28
|
+
out[key] = scalar(rest.trim())
|
|
29
|
+
}
|
|
30
|
+
return out
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// Folded: lines join with a space, a blank line becomes a newline.
|
|
34
|
+
function fold(lines) {
|
|
35
|
+
const paras = []
|
|
36
|
+
let cur = []
|
|
37
|
+
for (const l of lines) {
|
|
38
|
+
if (l.trim() === '') {
|
|
39
|
+
if (cur.length) paras.push(cur.join(' '))
|
|
40
|
+
cur = []
|
|
41
|
+
} else cur.push(l.trim())
|
|
42
|
+
}
|
|
43
|
+
if (cur.length) paras.push(cur.join(' '))
|
|
44
|
+
return paras.join('\n')
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function scalar(s) {
|
|
48
|
+
if (s.startsWith('"') && s.endsWith('"') && s.length >= 2) return JSON.parse(s.replace(/\\'/g, "'"))
|
|
49
|
+
if (s.startsWith("'") && s.endsWith("'") && s.length >= 2) return s.slice(1, -1).replace(/''/g, "'")
|
|
50
|
+
return s
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export function loadSkill(path) {
|
|
54
|
+
const file = existsSync(path) && statSync(path).isDirectory() ? join(path, 'SKILL.md') : path
|
|
55
|
+
if (!existsSync(file)) throw new Error(`no SKILL.md at ${path}`)
|
|
56
|
+
const fm = parseFrontmatter(readFileSync(file, 'utf8'))
|
|
57
|
+
if (!fm) throw new Error(`${file} has no frontmatter`)
|
|
58
|
+
if (!fm.name) throw new Error(`${file} has no name in its frontmatter`)
|
|
59
|
+
if (!/^[A-Za-z0-9][\w.-]*$/.test(fm.name)) throw new Error(`${file}: the name "${fm.name}" is not a plain skill name`)
|
|
60
|
+
if (!fm.description) throw new Error(`${file} has no description in its frontmatter`)
|
|
61
|
+
return { name: fm.name, description: fm.description }
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export function parseEvalSet(text) {
|
|
65
|
+
let data
|
|
66
|
+
try {
|
|
67
|
+
data = JSON.parse(text)
|
|
68
|
+
} catch (e) {
|
|
69
|
+
throw new Error(`the eval set is not valid JSON: ${e.message}`)
|
|
70
|
+
}
|
|
71
|
+
if (!Array.isArray(data)) throw new Error('the eval set must be a JSON array of { "query": string, "should_trigger": boolean }')
|
|
72
|
+
if (!data.length) throw new Error('the eval set is empty')
|
|
73
|
+
const seen = new Set()
|
|
74
|
+
data.forEach((it, i) => {
|
|
75
|
+
if (!it || typeof it !== 'object') throw new Error(`item ${i}: not an object`)
|
|
76
|
+
if (typeof it.query !== 'string' || !it.query.trim()) throw new Error(`item ${i}: query must be a non-empty string`)
|
|
77
|
+
if (typeof it.should_trigger !== 'boolean') throw new Error(`item ${i}: should_trigger must be true or false`)
|
|
78
|
+
if (seen.has(it.query)) throw new Error(`item ${i}: duplicate query — compare matches reports by query text`)
|
|
79
|
+
seen.add(it.query)
|
|
80
|
+
})
|
|
81
|
+
return data
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function loadEvalSet(file) {
|
|
85
|
+
if (!existsSync(file)) throw new Error(`no eval set at ${file}`)
|
|
86
|
+
return parseEvalSet(readFileSync(file, 'utf8'))
|
|
87
|
+
}
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
// Reading one run's stream. `claude -p --output-format stream-json --verbose
|
|
2
|
+
// --include-partial-messages` prints one JSON event per line: a `system`/`init` event
|
|
3
|
+
// with the roster, `stream_event`s wrapping the API's own streaming events, complete
|
|
4
|
+
// `assistant` and `user` messages, and a final `result`.
|
|
5
|
+
//
|
|
6
|
+
// The definition of a trigger, which the README states in the same words: the stub is
|
|
7
|
+
// loaded when the model's FIRST decisive message carries a tool call that names it —
|
|
8
|
+
// `Skill` or `SlashCommand` with the stub's name in its input, or `Read` of the stub
|
|
9
|
+
// file. A first message that only fetches tool schemas (`ToolSearch`) is not decisive;
|
|
10
|
+
// the next one is. Anything else in the first decisive message — a text answer,
|
|
11
|
+
// another tool, another skill — is `not-triggered`. That is skill-creator's rule too,
|
|
12
|
+
// minus its one shortcut: it stopped at the first tool of any other kind, so a Bash
|
|
13
|
+
// call listed before the Skill call in the same message counted as a miss.
|
|
14
|
+
//
|
|
15
|
+
// An authentication failure or an API error is an `error`, never `not-triggered`:
|
|
16
|
+
// that confusion is how an expired login used to read as a clean 0/40.
|
|
17
|
+
|
|
18
|
+
const LOADERS = new Set(['Skill', 'SlashCommand', 'Read'])
|
|
19
|
+
const NEUTRAL = new Set(['ToolSearch'])
|
|
20
|
+
const ERROR_TEXT = /^\s*(?:Failed to authenticate|API Error\b|Invalid API key|OAuth token has expired|Please run \/login|Credit balance is too low)/i
|
|
21
|
+
|
|
22
|
+
export function createDetector({ stubName, stubPath }) {
|
|
23
|
+
const names = (tool, input) => {
|
|
24
|
+
if (!LOADERS.has(tool)) return false
|
|
25
|
+
const s = typeof input === 'string' ? input : JSON.stringify(input ?? {})
|
|
26
|
+
if (tool === 'Read') return s.includes(stubPath) || s.includes(`${stubName}.md`)
|
|
27
|
+
return s.includes(stubName)
|
|
28
|
+
}
|
|
29
|
+
const state = {
|
|
30
|
+
init: null,
|
|
31
|
+
// Per message: the tool calls seen so far, and whether any of them was decisive.
|
|
32
|
+
blocks: new Map(),
|
|
33
|
+
decisive: false,
|
|
34
|
+
resultSeen: false,
|
|
35
|
+
}
|
|
36
|
+
const done = (outcome, reason) => ({ outcome, reason })
|
|
37
|
+
|
|
38
|
+
const endOfMessage = () => {
|
|
39
|
+
// Only a message that did something decides: neutral tools alone keep listening.
|
|
40
|
+
const decisive = state.decisive
|
|
41
|
+
state.blocks.clear()
|
|
42
|
+
state.decisive = false
|
|
43
|
+
return decisive ? done('not-triggered', 'the first decisive message did not load the stub') : null
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function feed(ev) {
|
|
47
|
+
if (!ev || typeof ev !== 'object') return null
|
|
48
|
+
if (ev.type === 'system' && ev.subtype === 'init') {
|
|
49
|
+
state.init = ev
|
|
50
|
+
if (Array.isArray(ev.slash_commands) && !ev.slash_commands.some((c) => String(c).replace(/^\//, '').split(':').pop() === stubName)) {
|
|
51
|
+
return done('error', `the stub ${stubName} is not in the roster the init event lists — the model never saw it`)
|
|
52
|
+
}
|
|
53
|
+
return null
|
|
54
|
+
}
|
|
55
|
+
if (ev.type === 'stream_event') {
|
|
56
|
+
const se = ev.event ?? {}
|
|
57
|
+
if (se.type === 'content_block_start') {
|
|
58
|
+
const cb = se.content_block ?? {}
|
|
59
|
+
if (cb.type === 'tool_use') {
|
|
60
|
+
state.blocks.set(se.index ?? 0, { tool: cb.name, json: '' })
|
|
61
|
+
if (!NEUTRAL.has(cb.name)) state.decisive = true
|
|
62
|
+
} else if (cb.type === 'text') state.decisive = true
|
|
63
|
+
} else if (se.type === 'content_block_delta') {
|
|
64
|
+
const b = state.blocks.get(se.index ?? 0)
|
|
65
|
+
if (b && se.delta?.type === 'input_json_delta') {
|
|
66
|
+
b.json += se.delta.partial_json ?? ''
|
|
67
|
+
if (names(b.tool, b.json)) return done('triggered', `${b.tool} named the stub`)
|
|
68
|
+
}
|
|
69
|
+
} else if (se.type === 'content_block_stop') {
|
|
70
|
+
const b = state.blocks.get(se.index ?? 0)
|
|
71
|
+
if (b && names(b.tool, b.json)) return done('triggered', `${b.tool} named the stub`)
|
|
72
|
+
} else if (se.type === 'message_stop') {
|
|
73
|
+
return endOfMessage()
|
|
74
|
+
}
|
|
75
|
+
return null
|
|
76
|
+
}
|
|
77
|
+
if (ev.type === 'assistant') {
|
|
78
|
+
const msg = ev.message ?? {}
|
|
79
|
+
const content = Array.isArray(msg.content) ? msg.content : []
|
|
80
|
+
const text = content.filter((c) => c.type === 'text').map((c) => c.text ?? '').join('\n')
|
|
81
|
+
if (ev.error || msg.model === '<synthetic>' || ERROR_TEXT.test(text)) return done('error', firstLine(text) || `assistant error: ${ev.error}`)
|
|
82
|
+
for (const c of content) {
|
|
83
|
+
if (c.type !== 'tool_use') continue
|
|
84
|
+
if (names(c.name, c.input)) return done('triggered', `${c.name} named the stub`)
|
|
85
|
+
if (!NEUTRAL.has(c.name)) state.decisive = true
|
|
86
|
+
}
|
|
87
|
+
if (text) state.decisive = true
|
|
88
|
+
return null
|
|
89
|
+
}
|
|
90
|
+
if (ev.type === 'user') {
|
|
91
|
+
// A tool result means the message that asked for it is over, whether or not a
|
|
92
|
+
// message_stop was streamed for it.
|
|
93
|
+
return endOfMessage()
|
|
94
|
+
}
|
|
95
|
+
if (ev.type === 'result') {
|
|
96
|
+
state.resultSeen = true
|
|
97
|
+
const text = typeof ev.result === 'string' ? ev.result : ''
|
|
98
|
+
if (ev.is_error || (ev.subtype && ev.subtype !== 'success') || ERROR_TEXT.test(text)) return done('error', firstLine(text) || `result: ${ev.subtype}`)
|
|
99
|
+
return done('not-triggered', 'the session ended without loading the stub')
|
|
100
|
+
}
|
|
101
|
+
return null
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function end({ code, stderr = '' } = {}) {
|
|
105
|
+
const why = firstLine(stderr)
|
|
106
|
+
return done('error', `the stream ended without a decision (exit ${code ?? '?'})${why ? `: ${why}` : ''}`)
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return {
|
|
110
|
+
feed,
|
|
111
|
+
end,
|
|
112
|
+
get init() {
|
|
113
|
+
return state.init
|
|
114
|
+
},
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const firstLine = (s) => String(s ?? '').trim().split('\n')[0].slice(0, 200)
|
|
119
|
+
|
|
120
|
+
// Newline-delimited JSON from a byte stream: chunks arrive split anywhere, and a line
|
|
121
|
+
// that is not JSON (a warning the CLI prints to stdout) is skipped, not fatal.
|
|
122
|
+
export function lineSplitter(onEvent) {
|
|
123
|
+
let buf = ''
|
|
124
|
+
const emit = (line) => {
|
|
125
|
+
const t = line.trim()
|
|
126
|
+
if (!t) return
|
|
127
|
+
let ev
|
|
128
|
+
try {
|
|
129
|
+
ev = JSON.parse(t)
|
|
130
|
+
} catch {
|
|
131
|
+
return
|
|
132
|
+
}
|
|
133
|
+
onEvent(ev)
|
|
134
|
+
}
|
|
135
|
+
return {
|
|
136
|
+
push(chunk) {
|
|
137
|
+
buf += chunk
|
|
138
|
+
let i
|
|
139
|
+
while ((i = buf.indexOf('\n')) >= 0) {
|
|
140
|
+
const line = buf.slice(0, i)
|
|
141
|
+
buf = buf.slice(i + 1)
|
|
142
|
+
emit(line)
|
|
143
|
+
}
|
|
144
|
+
},
|
|
145
|
+
end() {
|
|
146
|
+
if (buf) emit(buf)
|
|
147
|
+
buf = ''
|
|
148
|
+
},
|
|
149
|
+
}
|
|
150
|
+
}
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
// The re-enable reminder. The conflict gate prints `claude plugin disable` and `enable`
|
|
2
|
+
// and runs neither; once the user disables the plugin it is no longer a conflict, so
|
|
3
|
+
// nothing would mention it again (ST-21). The toggles the gate printed are remembered in
|
|
4
|
+
// one small file under the run's --out directory — never under ~/.claude — and the
|
|
5
|
+
// enable command is printed by every preflight that sees the plugin still disabled and at
|
|
6
|
+
// the end of every run, until a preflight sees it enabled again or no longer installed.
|
|
7
|
+
//
|
|
8
|
+
// The file holds plugin ids, scopes and the two commands; it is not a report and none of
|
|
9
|
+
// it enters one.
|
|
10
|
+
import { mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
|
11
|
+
import { join } from 'node:path'
|
|
12
|
+
import { toggleCommands } from './plugins.mjs'
|
|
13
|
+
|
|
14
|
+
export const TOGGLES_FILE = '.skilltrigger-toggles'
|
|
15
|
+
|
|
16
|
+
export function readToggles(outDir) {
|
|
17
|
+
try {
|
|
18
|
+
const list = JSON.parse(readFileSync(join(outDir, TOGGLES_FILE), 'utf8'))?.plugins
|
|
19
|
+
return Array.isArray(list) ? list.filter((t) => typeof t?.id === 'string' && typeof t.enable === 'string') : []
|
|
20
|
+
} catch {
|
|
21
|
+
return []
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export function writeToggles(outDir, list) {
|
|
26
|
+
const file = join(outDir, TOGGLES_FILE)
|
|
27
|
+
if (!list.length) return rmSync(file, { force: true })
|
|
28
|
+
mkdirSync(outDir, { recursive: true })
|
|
29
|
+
writeFileSync(file, JSON.stringify({ plugins: list }, null, 2) + '\n')
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// remembered: what the file holds; conflicts: what the gate found now (ids may be null);
|
|
33
|
+
// plugins: the parsed plugin list, or null when it was not read.
|
|
34
|
+
// → { keep, reminders } — reminders are the kept entries whose plugin is disabled now.
|
|
35
|
+
export function reconcile({ remembered, conflicts = [], plugins = null, date }) {
|
|
36
|
+
const byId = new Map(remembered.map((t) => [t.id, t]))
|
|
37
|
+
for (const c of conflicts.filter((x) => x.id)) {
|
|
38
|
+
const { disable, enable } = toggleCommands([c])
|
|
39
|
+
byId.set(c.id, { id: c.id, scope: c.scope ?? null, disable: disable[0], enable: enable[0], since: byId.get(c.id)?.since ?? date })
|
|
40
|
+
}
|
|
41
|
+
if (!plugins) return { keep: [...byId.values()], reminders: [] }
|
|
42
|
+
const now = new Map(plugins.map((p) => [p.id, p]))
|
|
43
|
+
const conflicting = new Set(conflicts.map((c) => c.id).filter(Boolean))
|
|
44
|
+
const keep = []
|
|
45
|
+
const reminders = []
|
|
46
|
+
for (const t of byId.values()) {
|
|
47
|
+
const p = now.get(t.id)
|
|
48
|
+
if (!p) continue // uninstalled: nothing to re-enable
|
|
49
|
+
if (p.enabled && !conflicting.has(t.id)) continue // enabled again: done
|
|
50
|
+
keep.push(t)
|
|
51
|
+
if (!p.enabled) reminders.push(t)
|
|
52
|
+
}
|
|
53
|
+
return { keep, reminders }
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export function formatReminders(reminders, { after = false } = {}) {
|
|
57
|
+
if (!reminders.length) return null
|
|
58
|
+
const head = after
|
|
59
|
+
? 're-enable what the conflict gate asked you to disable, once the measurements are done:'
|
|
60
|
+
: `reminder: ${reminders.map((t) => t.id).join(', ')} ${reminders.length === 1 ? 'is' : 'are'} disabled, as the conflict gate asked — re-enable after the measurement:`
|
|
61
|
+
return [head, ...reminders.map((t) => ` ${t.enable}`)].join('\n')
|
|
62
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
// The six ways the old measurement returned a clean-looking zero that meant nothing,
|
|
2
|
+
// each with the phrase the README must use for it (`check` holds it to that) and what
|
|
3
|
+
// skilltrigger does about it. The first five were hit in practice between 2026-09-09
|
|
4
|
+
// and 2026-09-18; the sixth is the drift note — the roster changing the number.
|
|
5
|
+
export const TRAPS = [
|
|
6
|
+
{ id: 'parallel', phrase: 'Parallel workers', handled: 'strictly serial runner; no workers option' },
|
|
7
|
+
{ id: 'shadowing', phrase: 'An installed plugin shadowing the stub', handled: 'conflict gate: refuses, prints disable/enable commands' },
|
|
8
|
+
{ id: 'outdated', phrase: 'An outdated CLI', handled: 'round-trip gate with the chosen model; version recorded' },
|
|
9
|
+
{ id: 'login', phrase: 'An expired login', handled: 'auth gate, round-trip gate, auth errors are errors' },
|
|
10
|
+
{ id: 'sleep', phrase: 'A machine that sleeps', handled: 'sleep gate: caffeinate on macOS, warning elsewhere; timeouts counted apart' },
|
|
11
|
+
{ id: 'roster', phrase: 'The skill roster', handled: 'roster gate records it; compare warns when it differs' },
|
|
12
|
+
]
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// skilltrigger — how often a Claude Code skill's description makes the model load it,
|
|
3
|
+
// measured behind gates that refuse to report a number they cannot trust.
|
|
4
|
+
//
|
|
5
|
+
// skilltrigger preflight [--model M] [--skill <dir>] [--out <dir>] [--allow-conflict]
|
|
6
|
+
// skilltrigger run --skill <dir> --eval <file> [--runs 2] [--model M] [--timeout 30]
|
|
7
|
+
// [--description "<override>"] [--out <dir>] [--allow-conflict]
|
|
8
|
+
// skilltrigger compare <a.json> <b.json>
|
|
9
|
+
// skilltrigger check
|
|
10
|
+
//
|
|
11
|
+
// Exit codes: 0 done · 1 usage or input error · 2 a gate failed · 3 no verdict.
|
|
12
|
+
import { readFileSync } from 'node:fs'
|
|
13
|
+
import { dirname, join, resolve } from 'node:path'
|
|
14
|
+
import { fileURLToPath } from 'node:url'
|
|
15
|
+
import { UsageError, number, parseArgs } from './lib/args.mjs'
|
|
16
|
+
import { checkRepo } from './lib/check.mjs'
|
|
17
|
+
import { compare } from './lib/compare.mjs'
|
|
18
|
+
import { formatGates, preflight } from './lib/gates.mjs'
|
|
19
|
+
import { NO_VERDICT_SHARE, headline, localDate, partialLine, summarise, writeReport } from './lib/report.mjs'
|
|
20
|
+
import { runAll } from './lib/runner.mjs'
|
|
21
|
+
import { loadEvalSet, loadSkill } from './lib/skill.mjs'
|
|
22
|
+
import { formatReminders, readToggles, reconcile, writeToggles } from './lib/toggles.mjs'
|
|
23
|
+
|
|
24
|
+
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..')
|
|
25
|
+
const pkg = JSON.parse(readFileSync(join(ROOT, 'package.json'), 'utf8'))
|
|
26
|
+
|
|
27
|
+
const HELP = readFileSync(fileURLToPath(import.meta.url), 'utf8')
|
|
28
|
+
.split('\n')
|
|
29
|
+
.slice(1, 12)
|
|
30
|
+
.map((l) => l.replace(/^\/\/ ?/, ''))
|
|
31
|
+
.join('\n')
|
|
32
|
+
|
|
33
|
+
const DEFAULT_OUT = 'skilltrigger-results'
|
|
34
|
+
|
|
35
|
+
// The gates, then the re-enable reminder: the toggles the conflict gate printed are kept
|
|
36
|
+
// under --out, and any of those plugins still disabled is named with its enable command.
|
|
37
|
+
async function gates({ model, skillName, allowConflict, outDir }) {
|
|
38
|
+
console.log(`skilltrigger ${pkg.version} — preflight`)
|
|
39
|
+
const pf = await preflight({ model, skillName, allowConflict })
|
|
40
|
+
console.log(formatGates(pf.gates))
|
|
41
|
+
const conflict = pf.gates.find((x) => x.id === 'conflict')
|
|
42
|
+
const { keep, reminders } = reconcile({ remembered: readToggles(outDir), conflicts: conflict?.conflicts ?? [], plugins: conflict?.plugins ?? null, date: localDate() })
|
|
43
|
+
writeToggles(outDir, keep)
|
|
44
|
+
const note = formatReminders(reminders)
|
|
45
|
+
if (note) console.log(`\n${note}`)
|
|
46
|
+
return { ...pf, reminders }
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async function cmdPreflight(argv) {
|
|
50
|
+
const { opts, positional } = parseArgs(argv, { model: 'string', skill: 'string', out: 'string', 'allow-conflict': 'boolean' })
|
|
51
|
+
if (positional.length) throw new UsageError(`unexpected argument: ${positional[0]}`)
|
|
52
|
+
const skillName = opts.skill ? loadSkill(opts.skill).name : null
|
|
53
|
+
const pf = await gates({ model: opts.model ?? null, skillName, allowConflict: Boolean(opts['allow-conflict']), outDir: resolve(opts.out ?? DEFAULT_OUT) })
|
|
54
|
+
if (!pf.ok) {
|
|
55
|
+
console.log('\npreflight failed — a run now would measure the failure, not the description')
|
|
56
|
+
return 2
|
|
57
|
+
}
|
|
58
|
+
console.log('\npreflight ok')
|
|
59
|
+
return 0
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
async function cmdRun(argv) {
|
|
63
|
+
const { opts, positional } = parseArgs(argv, {
|
|
64
|
+
skill: 'string',
|
|
65
|
+
eval: 'string',
|
|
66
|
+
runs: 'string',
|
|
67
|
+
model: 'string',
|
|
68
|
+
timeout: 'string',
|
|
69
|
+
description: 'string',
|
|
70
|
+
out: 'string',
|
|
71
|
+
'allow-conflict': 'boolean',
|
|
72
|
+
})
|
|
73
|
+
if (positional.length) throw new UsageError(`unexpected argument: ${positional[0]}`)
|
|
74
|
+
if (!opts.skill) throw new UsageError('run needs --skill <dir>')
|
|
75
|
+
if (!opts.eval) throw new UsageError('run needs --eval <file>')
|
|
76
|
+
const runs = number(opts, 'runs', { fallback: 2, min: 1, integer: true })
|
|
77
|
+
const timeout = number(opts, 'timeout', { fallback: 30, min: 0.1 })
|
|
78
|
+
const skill = loadSkill(opts.skill)
|
|
79
|
+
const items = loadEvalSet(opts.eval)
|
|
80
|
+
const overridden = opts.description !== undefined
|
|
81
|
+
const description = overridden ? opts.description : skill.description
|
|
82
|
+
if (!description.trim()) throw new UsageError('--description is empty')
|
|
83
|
+
const outDir = resolve(opts.out ?? DEFAULT_OUT)
|
|
84
|
+
const model = opts.model ?? null
|
|
85
|
+
|
|
86
|
+
const pf = await gates({ model, skillName: skill.name, allowConflict: Boolean(opts['allow-conflict']), outDir })
|
|
87
|
+
if (!pf.ok) {
|
|
88
|
+
console.log('\nrefusing to run: a failed gate makes the number meaningless. Fix it and run again.')
|
|
89
|
+
return 2
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const planned = items.length * runs
|
|
93
|
+
console.log(`\n${skill.name}: ${items.length} queries × ${runs} runs = ${planned}, serial, ${timeout} s timeout each`)
|
|
94
|
+
const width = String(planned).length
|
|
95
|
+
const mark = { triggered: '+', 'not-triggered': '.', timeout: 'T', error: 'E' }
|
|
96
|
+
const res = await runAll({
|
|
97
|
+
items,
|
|
98
|
+
runs,
|
|
99
|
+
skillName: skill.name,
|
|
100
|
+
description,
|
|
101
|
+
model,
|
|
102
|
+
timeoutMs: timeout * 1000,
|
|
103
|
+
env: process.env,
|
|
104
|
+
noVerdictShare: NO_VERDICT_SHARE,
|
|
105
|
+
onRun: ({ item, done, outcome, reason, ms }) => {
|
|
106
|
+
const q = item.query.length > 70 ? `${item.query.slice(0, 69)}…` : item.query
|
|
107
|
+
const why = outcome === 'timeout' || outcome === 'error' ? ` — ${reason}` : ''
|
|
108
|
+
console.log(` [${String(done).padStart(width)}/${planned}] ${mark[outcome]} ${outcome.padEnd(13)} ${(ms / 1000).toFixed(1).padStart(5)} s ${item.should_trigger ? 'pos' : 'neg'} ${q}${why}`)
|
|
109
|
+
},
|
|
110
|
+
})
|
|
111
|
+
if (res.aborted) console.log(` stopped: more than ${NO_VERDICT_SHARE * 100}% of the planned runs timed out or failed — no verdict is possible`)
|
|
112
|
+
|
|
113
|
+
const rep = summarise({
|
|
114
|
+
items,
|
|
115
|
+
outcomes: res.outcomes,
|
|
116
|
+
models: res.models,
|
|
117
|
+
skillName: skill.name,
|
|
118
|
+
description,
|
|
119
|
+
overridden,
|
|
120
|
+
facts: pf.facts,
|
|
121
|
+
runsPerQuery: runs,
|
|
122
|
+
timeoutSeconds: timeout,
|
|
123
|
+
planned,
|
|
124
|
+
aborted: res.aborted,
|
|
125
|
+
toolVersion: pkg.version,
|
|
126
|
+
})
|
|
127
|
+
const files = writeReport(rep, outDir)
|
|
128
|
+
console.log(`\n${headline(rep)}`)
|
|
129
|
+
if (partialLine(rep)) console.log(partialLine(rep))
|
|
130
|
+
if (rep.verdict === 'ok') console.log(`per query: ${rep.totals.passed}/${rep.totals.queries} pass at a trigger rate threshold of ${rep.triggerThreshold}; no-verdict threshold ${rep.noVerdictThreshold * 100}% of runs`)
|
|
131
|
+
console.log(`report: ${files.json}\n ${files.md}`)
|
|
132
|
+
const after = formatReminders(pf.reminders, { after: true })
|
|
133
|
+
if (after) console.log(`\n${after}`)
|
|
134
|
+
return rep.verdict === 'ok' ? 0 : 3
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
function cmdCompare(argv) {
|
|
138
|
+
const { positional } = parseArgs(argv, {})
|
|
139
|
+
if (positional.length !== 2) throw new UsageError('compare needs two report files: compare <a.json> <b.json>')
|
|
140
|
+
const [a, b] = positional.map((f) => {
|
|
141
|
+
let r
|
|
142
|
+
try {
|
|
143
|
+
r = JSON.parse(readFileSync(f, 'utf8'))
|
|
144
|
+
} catch (e) {
|
|
145
|
+
throw new UsageError(`${f}: not a readable report (${e.message})`)
|
|
146
|
+
}
|
|
147
|
+
if (r?.tool !== 'skilltrigger' || !Array.isArray(r.queries)) throw new UsageError(`${f}: not a skilltrigger report`)
|
|
148
|
+
return r
|
|
149
|
+
})
|
|
150
|
+
console.log(compare(a, b).text)
|
|
151
|
+
return 0
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function cmdCheck() {
|
|
155
|
+
const errors = checkRepo(ROOT)
|
|
156
|
+
for (const e of errors) console.error(`check: ${e}`)
|
|
157
|
+
if (errors.length) return 1
|
|
158
|
+
console.log(`ok — skilltrigger ${pkg.version}: CHANGELOG, README traps, no dependencies, no private strings`)
|
|
159
|
+
return 0
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function main() {
|
|
163
|
+
const [cmd, ...rest] = process.argv.slice(2)
|
|
164
|
+
try {
|
|
165
|
+
switch (cmd) {
|
|
166
|
+
case 'preflight':
|
|
167
|
+
return await cmdPreflight(rest)
|
|
168
|
+
case 'run':
|
|
169
|
+
return await cmdRun(rest)
|
|
170
|
+
case 'compare':
|
|
171
|
+
return cmdCompare(rest)
|
|
172
|
+
case 'check':
|
|
173
|
+
return cmdCheck()
|
|
174
|
+
case '--version':
|
|
175
|
+
case '-v':
|
|
176
|
+
console.log(pkg.version)
|
|
177
|
+
return 0
|
|
178
|
+
case undefined:
|
|
179
|
+
case 'help':
|
|
180
|
+
case '--help':
|
|
181
|
+
case '-h':
|
|
182
|
+
console.log(HELP)
|
|
183
|
+
return 0
|
|
184
|
+
default:
|
|
185
|
+
throw new UsageError(`unknown command: ${cmd}`)
|
|
186
|
+
}
|
|
187
|
+
} catch (e) {
|
|
188
|
+
// A bad skill directory or eval file is the user's input, not a crash: one line.
|
|
189
|
+
console.error(`skilltrigger: ${e.message}`)
|
|
190
|
+
if (e instanceof UsageError) console.error('run `skilltrigger help` for usage')
|
|
191
|
+
if (process.env.SKILLTRIGGER_DEBUG) console.error(e.stack)
|
|
192
|
+
return 1
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
process.exitCode = await main()
|
package/package.json
CHANGED
|
@@ -1,6 +1,86 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "skilltrigger",
|
|
3
|
-
"version": "0.0.
|
|
4
|
-
"
|
|
5
|
-
"
|
|
6
|
-
|
|
3
|
+
"version": "0.0.2",
|
|
4
|
+
"description": "How often a Claude Code skill's description makes the model load it, on prompts that should trigger it and near-misses that should not, measured behind gates that refuse to report a number they cannot trust.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": {
|
|
7
|
+
"skilltrigger": "bin/skilltrigger.mjs"
|
|
8
|
+
},
|
|
9
|
+
"files": [
|
|
10
|
+
"bin",
|
|
11
|
+
"README.md",
|
|
12
|
+
"CHANGELOG.md",
|
|
13
|
+
"LICENSE"
|
|
14
|
+
],
|
|
15
|
+
"engines": {
|
|
16
|
+
"node": ">=18"
|
|
17
|
+
},
|
|
18
|
+
"scripts": {
|
|
19
|
+
"test": "node bin/skilltrigger.mjs check && node --test",
|
|
20
|
+
"backlog": "npx --yes https://codeload.github.com/Allan-Nava/backlogsync/tar.gz/974f2bd96a7030af8f30eab093b05b7b46482ede check",
|
|
21
|
+
"build:site": "node site/build.mjs",
|
|
22
|
+
"roadmap": "npx --yes https://codeload.github.com/Allan-Nava/backlogsync/tar.gz/974f2bd96a7030af8f30eab093b05b7b46482ede roadmap"
|
|
23
|
+
},
|
|
24
|
+
"backlogsync": {
|
|
25
|
+
"prefix": "ST",
|
|
26
|
+
"name": "skilltrigger",
|
|
27
|
+
"regenerate": "npm run roadmap",
|
|
28
|
+
"labels": {
|
|
29
|
+
"gates": [
|
|
30
|
+
"3b4bb8",
|
|
31
|
+
"Preflight gates: CLI, auth, round trip, conflict, sleep, roster"
|
|
32
|
+
],
|
|
33
|
+
"runner": [
|
|
34
|
+
"0e8a16",
|
|
35
|
+
"The serial runner and the stream detector"
|
|
36
|
+
],
|
|
37
|
+
"report": [
|
|
38
|
+
"fbca04",
|
|
39
|
+
"Reports, verdict rule, compare"
|
|
40
|
+
],
|
|
41
|
+
"release": [
|
|
42
|
+
"5319e7",
|
|
43
|
+
"Publishing and versioning"
|
|
44
|
+
],
|
|
45
|
+
"docs": [
|
|
46
|
+
"0075ca",
|
|
47
|
+
"README, CONTRIBUTING, site"
|
|
48
|
+
],
|
|
49
|
+
"project": [
|
|
50
|
+
"6a737d",
|
|
51
|
+
"Backlog, roadmap, repo hygiene"
|
|
52
|
+
],
|
|
53
|
+
"tests": [
|
|
54
|
+
"d4c5f9",
|
|
55
|
+
"Test coverage and the fake claude"
|
|
56
|
+
],
|
|
57
|
+
"enhancement": [
|
|
58
|
+
"a2eeef",
|
|
59
|
+
"New capability"
|
|
60
|
+
],
|
|
61
|
+
"prio-med": [
|
|
62
|
+
"e4b429",
|
|
63
|
+
"Medium priority in BACKLOG.md"
|
|
64
|
+
]
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"repository": {
|
|
68
|
+
"type": "git",
|
|
69
|
+
"url": "https://github.com/Allan-Nava/skilltrigger"
|
|
70
|
+
},
|
|
71
|
+
"homepage": "https://allan-nava.github.io/skilltrigger/",
|
|
72
|
+
"bugs": "https://github.com/Allan-Nava/skilltrigger/issues",
|
|
73
|
+
"keywords": [
|
|
74
|
+
"claude-code",
|
|
75
|
+
"skills",
|
|
76
|
+
"evals",
|
|
77
|
+
"trigger-rate",
|
|
78
|
+
"skill-description",
|
|
79
|
+
"measurement"
|
|
80
|
+
],
|
|
81
|
+
"author": "Allan Nava (https://github.com/Allan-Nava)",
|
|
82
|
+
"license": "MIT",
|
|
83
|
+
"devDependencies": {
|
|
84
|
+
"marked": "^17.0.0"
|
|
85
|
+
}
|
|
86
|
+
}
|