skilltrigger 0.0.0-stage → 0.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,78 @@
1
+ // What the inherited environment contributes to every run, counted. Each `claude -p`
2
+ // inherits the user's whole configuration minus CLAUDECODE: memory files, hooks and MCP
3
+ // servers all reach the model, and a memory file naming the skill moves the rate without
4
+ // changing any other recorded field. So the report records how many of each there were
5
+ // — counts only: never a file's contents, a path, a hook command or a server name (ST-20).
6
+ //
7
+ // Reading only. Nothing under the configuration directory is ever written.
8
+ //
9
+ // memory files project: CLAUDE.md, CLAUDE.local.md and .claude/CLAUDE.md in the run's
10
+ // temporary project and each of its ancestors, plus .claude/rules/**/*.md
11
+ // in the project itself; user: CLAUDE.md and rules/**/*.md in the
12
+ // configuration directory ($CLAUDE_CONFIG_DIR, else ~/.claude)
13
+ // hooks hook commands in the user and project settings files and in each
14
+ // enabled plugin's hooks/hooks.json (managed settings are not read)
15
+ // mcpServers the length of the init event's mcp_servers, null when it has none
16
+ import { existsSync, readFileSync, readdirSync, statSync } from 'node:fs'
17
+ import { homedir } from 'node:os'
18
+ import { dirname, join } from 'node:path'
19
+
20
+ const isFile = (f) => {
21
+ try {
22
+ return statSync(f).isFile()
23
+ } catch {
24
+ return false
25
+ }
26
+ }
27
+
28
+ function markdownUnder(dir) {
29
+ if (!existsSync(dir)) return 0
30
+ let n = 0
31
+ for (const d of readdirSync(dir, { withFileTypes: true })) {
32
+ if (d.isDirectory()) n += markdownUnder(join(dir, d.name))
33
+ else if (d.isFile() && d.name.endsWith('.md')) n++
34
+ }
35
+ return n
36
+ }
37
+
38
+ export const configDirOf = (env = process.env) => env.CLAUDE_CONFIG_DIR || join(env.HOME || homedir(), '.claude')
39
+
40
+ export function memoryFiles({ projectDir, configDir }) {
41
+ let project = markdownUnder(join(projectDir, '.claude', 'rules'))
42
+ for (let dir = projectDir; ; dir = dirname(dir)) {
43
+ for (const f of ['CLAUDE.md', 'CLAUDE.local.md', join('.claude', 'CLAUDE.md')]) if (isFile(join(dir, f))) project++
44
+ if (dirname(dir) === dir) break
45
+ }
46
+ const user = (isFile(join(configDir, 'CLAUDE.md')) ? 1 : 0) + markdownUnder(join(configDir, 'rules'))
47
+ return { project, user }
48
+ }
49
+
50
+ function hooksIn(file) {
51
+ if (!isFile(file)) return 0
52
+ let hooks
53
+ try {
54
+ hooks = JSON.parse(readFileSync(file, 'utf8'))?.hooks
55
+ } catch {
56
+ return 0
57
+ }
58
+ if (!hooks || typeof hooks !== 'object') return 0
59
+ let n = 0
60
+ for (const groups of Object.values(hooks)) if (Array.isArray(groups)) for (const grp of groups) n += Array.isArray(grp?.hooks) ? grp.hooks.length : 0
61
+ return n
62
+ }
63
+
64
+ export function countHooks({ configDir, projectDir = null, plugins = [] }) {
65
+ const files = [join(configDir, 'settings.json'), ...(projectDir ? [join(projectDir, '.claude', 'settings.json'), join(projectDir, '.claude', 'settings.local.json')] : [])]
66
+ for (const p of plugins) if (p.enabled && p.installPath) files.push(join(p.installPath, 'hooks', 'hooks.json'))
67
+ return files.reduce((n, f) => n + hooksIn(f), 0)
68
+ }
69
+
70
+ // `memory` is passed when it was counted earlier, from a directory since removed.
71
+ export function environmentSummary({ env = process.env, projectDir = null, memory = null, plugins = [], init = null }) {
72
+ const configDir = configDirOf(env)
73
+ return {
74
+ memoryFiles: memory ?? memoryFiles({ projectDir, configDir }),
75
+ hooks: countHooks({ configDir, projectDir, plugins }),
76
+ mcpServers: Array.isArray(init?.mcp_servers) ? init.mcp_servers.length : null,
77
+ }
78
+ }
@@ -0,0 +1,186 @@
1
+ // The gates. Each one is a way the measurement used to return a clean-looking zero
2
+ // that meant nothing; each returns { id, status, detail, fix } where status is ok,
3
+ // warn, fail or skip. Any fail refuses the run.
4
+ //
5
+ // cli claude is on PATH and answers --version (the version is recorded)
6
+ // auth `claude auth status` says logged in
7
+ // round-trip `claude -p "Reply with exactly: pong"` answers pong, with the chosen
8
+ // model — catches the expired OAuth that auth status can miss, and the
9
+ // 400 an outdated CLI answers for a model it does not know
10
+ // conflict no enabled plugin (or other visible skill) shadows the stub
11
+ // sleep the machine cannot sleep mid-run (macOS: caffeinate), else a warning
12
+ // roster how many slash commands and skills the model sees, from the init event
13
+ import { spawn } from 'node:child_process'
14
+ import { createHash } from 'node:crypto'
15
+ import { mkdtempSync, rmSync } from 'node:fs'
16
+ import { tmpdir } from 'node:os'
17
+ import { join } from 'node:path'
18
+ import { execClaude, streamClaude } from './claude.mjs'
19
+ import { configDirOf, environmentSummary, memoryFiles } from './environment.mjs'
20
+ import { findConflicts, parsePluginList, toggleCommands } from './plugins.mjs'
21
+
22
+ export const PONG_PROMPT = 'Reply with exactly: pong'
23
+ const g = (id, status, detail, fix) => ({ id, status, detail, ...(fix ? { fix } : {}) })
24
+ const firstLine = (s) => String(s ?? '').trim().split('\n')[0].slice(0, 200)
25
+ const truncate = (s, n) => (s.length > n ? `${s.slice(0, n - 1)}…` : s)
26
+
27
+ async function cliGate(env) {
28
+ const r = await execClaude(['--version'], { env, timeoutMs: 15000 })
29
+ if (r.notFound) return g('cli', 'fail', `claude not found on PATH`, 'install Claude Code (https://code.claude.com/docs) or put `claude` on PATH; SKILLTRIGGER_CLAUDE names another binary')
30
+ const version = r.stdout.trim().split('\n')[0]
31
+ if (r.code !== 0 || !/\d+\.\d+\.\d+/.test(version)) return g('cli', 'fail', `claude --version did not answer a version (exit ${r.code}${r.timedOut ? ', timed out' : ''}): ${firstLine(r.stderr || r.stdout)}`, 'reinstall or update Claude Code: `claude update`')
32
+ return { ...g('cli', 'ok', `claude ${version}`), version }
33
+ }
34
+
35
+ async function authGate(env) {
36
+ const r = await execClaude(['auth', 'status'], { env, timeoutMs: 20000 })
37
+ let loggedIn = null
38
+ try {
39
+ loggedIn = JSON.parse(r.stdout).loggedIn === true
40
+ } catch {
41
+ if (/not logged in|logged out/i.test(r.stdout)) loggedIn = false
42
+ else if (/logged in/i.test(r.stdout)) loggedIn = true
43
+ }
44
+ if (r.code === 0 && loggedIn) return g('auth', 'ok', 'claude auth status: logged in')
45
+ return g('auth', 'fail', `claude auth status: not logged in (exit ${r.code}${r.timedOut ? ', timed out' : ''})`, 'run `claude auth login` (or `/login` in an interactive session), then preflight again')
46
+ }
47
+
48
+ // One stream-json round trip serves two gates: the answer is the round trip, the
49
+ // init event is the roster. It runs in an empty temporary project so the roster is
50
+ // what every run will see, minus the stub.
51
+ async function roundTrip({ env, model }) {
52
+ const dir = mkdtempSync(join(tmpdir(), 'skilltrigger-preflight-'))
53
+ try {
54
+ let init = null
55
+ let text = ''
56
+ let result = null
57
+ const args = ['-p', PONG_PROMPT, '--output-format', 'stream-json', '--verbose', '--no-session-persistence', ...(model ? ['--model', model] : [])]
58
+ const r = await streamClaude(args, {
59
+ cwd: dir,
60
+ env,
61
+ timeoutMs: 120000,
62
+ onEvent: (ev) => {
63
+ if (ev.type === 'system' && ev.subtype === 'init') init = ev
64
+ if (ev.type === 'assistant') for (const c of ev.message?.content ?? []) if (c.type === 'text') text += c.text ?? ''
65
+ if (ev.type === 'result') {
66
+ result = ev
67
+ return true
68
+ }
69
+ return false
70
+ },
71
+ })
72
+ const answer = (typeof result?.result === 'string' ? result.result : text).trim()
73
+ let gate
74
+ if (r.timedOut) gate = g('round-trip', 'fail', 'claude -p did not answer within 120 s', 'check the network and the CLI by hand: `claude -p "Reply with exactly: pong"`')
75
+ else if (result?.is_error || /Failed to authenticate|API Error|Invalid API key|OAuth token has expired/i.test(answer)) {
76
+ gate = g('round-trip', 'fail', `claude -p answered an error: ${firstLine(answer)}`, /authenticat|oauth|api key|401/i.test(answer) ? 'log in again: `claude auth login` — auth status can say logged in while the token is expired' : 'update the CLI (`claude update`) or pass a --model this CLI knows')
77
+ } else if (!/^\W*pong\W*$/i.test(answer)) gate = g('round-trip', 'fail', `claude -p answered "${firstLine(answer || r.stderr) || '(nothing)'}", not pong (exit ${r.code})`, 'run `claude -p "Reply with exactly: pong"` by hand and fix what it prints')
78
+ else gate = g('round-trip', 'ok', `claude -p answered pong${model ? ` (model ${model})` : ''}`)
79
+ // The memory files are counted from this directory, before it is removed: the runs'
80
+ // projects are made the same way, beside it, so they share its ancestors.
81
+ const memory = memoryFiles({ projectDir: dir, configDir: configDirOf(env) })
82
+ return { gate, init, memory }
83
+ } finally {
84
+ rmSync(dir, { recursive: true, force: true })
85
+ }
86
+ }
87
+
88
+ function rosterGate(init) {
89
+ if (!init) return g('roster', 'fail', 'no init event in the stream — the roster cannot be recorded', 'update the CLI: `claude update`; the init event of `claude -p --output-format stream-json --verbose` lists the roster')
90
+ if (!Array.isArray(init.slash_commands)) return g('roster', 'fail', 'the init event lists no slash_commands — the roster cannot be recorded', 'update the CLI: `claude update`')
91
+ // Members beside the counts: two rosters of one size can differ (ST-20). Hashed, not
92
+ // named — the first 12 hex digits of each name's SHA-256, sorted — because the roster
93
+ // is every command and skill on the machine, private ones included, and a report is
94
+ // meant to be pasted. The preflight project holds no stub, so no stub is among them.
95
+ const tag = (name) => createHash('sha256').update(name).digest('hex').slice(0, 12)
96
+ const hashes = (list) => [...new Set(list.map((s) => (typeof s === 'string' ? s : s?.name)).filter(Boolean).map(tag))].sort()
97
+ const skillList = Array.isArray(init.skills) ? init.skills : null
98
+ const roster = { slashCommands: init.slash_commands.length, skills: skillList ? skillList.length : null, commandHashes: hashes(init.slash_commands), skillHashes: skillList ? hashes(skillList) : null }
99
+ const skills = roster.skills === null ? '' : `, ${roster.skills} skills`
100
+ return { ...g('roster', 'ok', `${roster.slashCommands} slash commands${skills} visible to the model`), roster, model: init.model ?? null, entries: [...init.slash_commands, ...(Array.isArray(init.skills) ? init.skills.map((s) => (typeof s === 'string' ? s : s?.name)) : [])].filter(Boolean) }
101
+ }
102
+
103
+ async function conflictGate({ env, skillName, allowConflict, rosterEntries }) {
104
+ const r = await execClaude(['plugin', 'list', '--json'], { env, timeoutMs: 30000 })
105
+ let plugins = parsePluginList(r.stdout)
106
+ let asked = { cmd: 'claude plugin list --json', r }
107
+ if (r.code !== 0 && !plugins?.length) {
108
+ // An older CLI without --json: the human form.
109
+ const plain = await execClaude(['plugin', 'list'], { env, timeoutMs: 30000 })
110
+ plugins = parsePluginList(plain.stdout)
111
+ asked = { cmd: 'claude plugin list', r: plain }
112
+ }
113
+ if (plugins === null) {
114
+ // Unreadable is not empty: a list in a new shape would otherwise read as "no plugins"
115
+ // and the gate would pass with nothing checked (ST-18).
116
+ const got = truncate(firstLine(asked.r.stdout || asked.r.stderr), 80) || '(nothing)'
117
+ const detail = `cannot read the plugin list: \`${asked.cmd}\` exited ${asked.r.code}${asked.r.timedOut ? ' (timed out)' : ''} and printed "${got}" — which plugins are enabled is unknown`
118
+ const fix = ['run `claude plugin list --json` by hand; if the CLI changed its output, update skilltrigger or report the shape', '--allow-conflict measures anyway and records that it did']
119
+ return { ...g('conflict', allowConflict ? 'warn' : 'fail', allowConflict ? `${detail} (allowed by --allow-conflict, recorded in the report)` : detail, fix.join('\n')), unreadable: true }
120
+ }
121
+ const enabled = plugins.filter((p) => p.enabled)
122
+ if (!skillName) return { ...g('conflict', 'warn', `no --skill given, so nothing to compare against ${enabled.length} enabled plugin(s)${enabled.length ? `: ${enabled.map((p) => p.id).join(', ')}` : ''}`, 'pass --skill <dir> to check this gate'), plugins }
123
+ const conflicts = findConflicts({ plugins, skillName, roster: rosterEntries })
124
+ if (!conflicts.length) return { ...g('conflict', 'ok', `no enabled plugin or visible skill named ${skillName} (${enabled.length} enabled plugin(s) checked)`), plugins }
125
+ const { disable, enable } = toggleCommands(conflicts)
126
+ const unaccounted = conflicts.filter((c) => !c.id).map((c) => c.via)
127
+ const lines = [
128
+ ...(disable.length ? ['disable before the run:', ...disable.map((c) => ` ${c}`), 'and re-enable after it:', ...enable.map((c) => ` ${c}`)] : []),
129
+ ...(unaccounted.length ? [`move aside the skill the roster shows (${unaccounted.join(', ')}) — a user or project skill of the same name`] : []),
130
+ 'skilltrigger runs none of these itself; --allow-conflict measures anyway and records that it did',
131
+ ]
132
+ const detail = `${conflicts.map((c) => c.via).join('; ')} — the model would load it, not the stub`
133
+ return { ...g('conflict', allowConflict ? 'warn' : 'fail', allowConflict ? `${detail} (allowed by --allow-conflict, recorded in the report)` : detail, lines.join('\n')), conflicts, plugins }
134
+ }
135
+
136
+ // macOS: `caffeinate -i -s -w <pid>` holds off idle and system sleep for as long as
137
+ // this process lives and exits with it. Elsewhere there is no one portable switch.
138
+ export function sleepGate({ platform = process.platform, caffeinate = '/usr/bin/caffeinate', pid = process.pid } = {}) {
139
+ if (platform !== 'darwin') {
140
+ return Promise.resolve(g('sleep', 'warn', `cannot hold the machine awake on ${platform}: a sleep mid-run turns runs into timeouts`, 'keep the machine awake for the run, e.g. `systemd-inhibit --what=idle:sleep skilltrigger run …` on Linux'))
141
+ }
142
+ return new Promise((resolve) => {
143
+ const child = spawn(caffeinate, ['-i', '-s', '-w', String(pid)], { stdio: 'ignore', detached: false })
144
+ child.on('error', (e) => resolve(g('sleep', 'fail', `could not start caffeinate: ${e.code ?? e.message}`, 'run under `caffeinate -i -s skilltrigger run …` by hand')))
145
+ child.on('spawn', () => {
146
+ child.unref()
147
+ resolve(g('sleep', 'ok', `caffeinate -i -s -w ${pid} holds the machine awake until this process exits`))
148
+ })
149
+ })
150
+ }
151
+
152
+ // → { gates, ok, facts: { cliVersion, model, roster, conflictAllowed, environment } }
153
+ export async function preflight({ env = process.env, model = null, skillName = null, allowConflict = false, platform = process.platform } = {}) {
154
+ const gates = []
155
+ const cli = await cliGate(env)
156
+ gates.push(cli)
157
+ const facts = { cliVersion: cli.version ?? null, model, roster: null, conflictAllowed: false, environment: null }
158
+ if (cli.status !== 'ok') {
159
+ for (const id of ['auth', 'round-trip', 'conflict']) gates.push(g(id, 'skip', 'not run: the cli gate failed'))
160
+ gates.push(await sleepGate({ platform }))
161
+ gates.push(g('roster', 'skip', 'not run: the cli gate failed'))
162
+ } else {
163
+ gates.push(await authGate(env))
164
+ const { gate, init, memory } = await roundTrip({ env, model })
165
+ gates.push(gate)
166
+ const roster = rosterGate(init)
167
+ const conflict = await conflictGate({ env, skillName, allowConflict, rosterEntries: roster.entries ?? [] })
168
+ gates.push(conflict)
169
+ gates.push(await sleepGate({ platform }))
170
+ gates.push(roster)
171
+ facts.roster = roster.roster ?? null
172
+ facts.model = model ?? roster.model ?? null
173
+ facts.conflictAllowed = conflict.status === 'warn' && Boolean(conflict.conflicts?.length || conflict.unreadable)
174
+ facts.environment = environmentSummary({ env, memory, plugins: conflict.plugins ?? [], init })
175
+ }
176
+ return { gates, ok: !gates.some((x) => x.status === 'fail'), facts }
177
+ }
178
+
179
+ export function formatGates(gates) {
180
+ const out = []
181
+ for (const x of gates) {
182
+ out.push(` ${x.status.padEnd(5)} ${x.id.padEnd(11)} ${x.detail}`)
183
+ if (x.fix && x.status !== 'ok') for (const [i, l] of x.fix.split('\n').entries()) out.push(`${' '.repeat(20)}${i === 0 ? 'fix: ' : ' '}${l}`)
184
+ }
185
+ return out.join('\n')
186
+ }
@@ -0,0 +1,106 @@
1
+ // The shadowing trap: an installed plugin that carries a skill of the same name. The
2
+ // model loads the real one, the stub is never invoked, and the rate reads as zero.
3
+ //
4
+ // Two sources, because neither sees everything:
5
+ // - `claude plugin list --json` names every plugin, whether it is enabled and where
6
+ // it is installed; the skills are read off <installPath>/skills/*/SKILL.md (and
7
+ // commands/*.md). Reading only — nothing under the plugin cache is ever written.
8
+ // - the roster in the `init` event of a stub-less `claude -p`, which lists what the
9
+ // model actually sees: it also catches a user-level skill or a plugin the list
10
+ // printed in a form this parser could not read.
11
+ import { existsSync, readFileSync, readdirSync } from 'node:fs'
12
+ import { join } from 'node:path'
13
+ import { parseFrontmatter } from './skill.mjs'
14
+
15
+ const split = (id) => {
16
+ const at = id.lastIndexOf('@')
17
+ return at > 0 ? { name: id.slice(0, at), marketplace: id.slice(at + 1) } : { name: id, marketplace: null }
18
+ }
19
+
20
+ // → the plugins, [] when the output says there are none, or null when it is in no shape
21
+ // this parser knows. The two answers are kept apart on purpose: "zero plugins" lets the
22
+ // conflict gate pass, "could not read" must not (ST-18).
23
+ export function parsePluginList(stdout) {
24
+ const text = String(stdout ?? '').trim()
25
+ if (!text) return null
26
+ if (text.startsWith('[')) {
27
+ let list
28
+ try {
29
+ list = JSON.parse(text)
30
+ } catch {
31
+ return null
32
+ }
33
+ if (!Array.isArray(list) || !list.every((p) => typeof p?.id === 'string' && p.id)) return null
34
+ return list.map((p) => ({ id: p.id, ...split(p.id), enabled: p.enabled !== false, scope: p.scope ?? null, installPath: p.installPath ?? null }))
35
+ }
36
+ // The human form: ❯ name@marketplace / Version: / Scope: / Status: ✔ enabled
37
+ const out = []
38
+ let cur = null
39
+ for (const line of text.split('\n')) {
40
+ const head = line.match(/^\s*(?:❯|>|\*|-)?\s*([\w.-]+@[\w.-]+)\s*$/)
41
+ if (head) {
42
+ cur = { id: head[1], ...split(head[1]), enabled: true, scope: null, installPath: null }
43
+ out.push(cur)
44
+ continue
45
+ }
46
+ if (!cur) continue
47
+ const scope = line.match(/^\s*Scope:\s*(\S+)/i)
48
+ if (scope) cur.scope = scope[1]
49
+ const status = line.match(/^\s*Status:\s*(.*)$/i)
50
+ if (status) cur.enabled = !/disabled/i.test(status[1])
51
+ }
52
+ if (out.length) return out
53
+ // A human form that says so: "No plugins installed.", or a heading over "(none)".
54
+ if (/\bno plugins\b/i.test(text) || /^installed plugins:\s*(\(none\))?\s*$/i.test(text.replace(/\s+/g, ' ').trim())) return []
55
+ return null
56
+ }
57
+
58
+ export function pluginSkills(installPath) {
59
+ if (!installPath || !existsSync(installPath)) return []
60
+ const names = new Set()
61
+ const skills = join(installPath, 'skills')
62
+ if (existsSync(skills)) {
63
+ for (const d of readdirSync(skills, { withFileTypes: true })) {
64
+ if (!d.isDirectory()) continue
65
+ const f = join(skills, d.name, 'SKILL.md')
66
+ if (!existsSync(f)) continue
67
+ let name = d.name
68
+ try {
69
+ name = parseFrontmatter(readFileSync(f, 'utf8'))?.name || d.name
70
+ } catch {}
71
+ names.add(name)
72
+ }
73
+ }
74
+ const commands = join(installPath, 'commands')
75
+ if (existsSync(commands)) for (const f of readdirSync(commands)) if (f.endsWith('.md')) names.add(f.slice(0, -3))
76
+ return [...names].sort()
77
+ }
78
+
79
+ // → [{ id, scope, via }] — id null when the roster shows a same-named skill no plugin
80
+ // accounts for (a user or project skill).
81
+ export function findConflicts({ plugins, skillName, roster = [] }) {
82
+ const found = new Map()
83
+ for (const p of plugins) {
84
+ if (!p.enabled) continue
85
+ if (pluginSkills(p.installPath).includes(skillName)) found.set(p.id, { id: p.id, scope: p.scope, via: `${p.id} carries a skill named ${skillName}` })
86
+ }
87
+ for (const entry of roster) {
88
+ const e = String(entry).replace(/^\//, '')
89
+ const [prefix, rest] = e.includes(':') ? [e.slice(0, e.lastIndexOf(':')), e.slice(e.lastIndexOf(':') + 1)] : [null, e]
90
+ if (rest !== skillName) continue
91
+ if (prefix) {
92
+ const p = plugins.find((x) => x.enabled && x.name === prefix)
93
+ if (p && !found.has(p.id)) found.set(p.id, { id: p.id, scope: p.scope, via: `roster: ${e}` })
94
+ else if (!p && ![...found.values()].some((f) => f.via === `roster: ${e}`)) found.set(`roster:${e}`, { id: null, scope: null, via: `roster: ${e}` })
95
+ } else if (!found.has(`roster:${e}`)) found.set(`roster:${e}`, { id: null, scope: null, via: `roster: ${e}` })
96
+ }
97
+ return [...found.values()]
98
+ }
99
+
100
+ // The commands are printed, never run: disabling a plugin is the user's decision, and so
101
+ // is re-enabling it — bin/lib/toggles.mjs only reminds them of it (ST-21).
102
+ export function toggleCommands(conflicts) {
103
+ const withScope = (verb, c) => `claude plugin ${verb} ${c.id}${c.scope ? ` --scope ${c.scope}` : ''}`
104
+ const plugins = conflicts.filter((c) => c.id)
105
+ return { disable: plugins.map((c) => withScope('disable', c)), enable: plugins.map((c) => withScope('enable', c)) }
106
+ }
@@ -0,0 +1,160 @@
1
+ // The report: what was measured, against what, and whether it may be read at all.
2
+ //
3
+ // It carries the eval queries (and any extra fields they came with), the counts, the
4
+ // versions, the roster's member names and the model each run reported — nothing else.
5
+ // No description text (a hash and a length stand in for it, so compare can tell two
6
+ // texts apart), no paths, no stub names, no stderr; what the inherited environment
7
+ // contributed is counted, never quoted.
8
+ import { createHash } from 'node:crypto'
9
+ import { existsSync, mkdirSync, writeFileSync } from 'node:fs'
10
+ import { join } from 'node:path'
11
+
12
+ export const TRIGGER_THRESHOLD = 0.5
13
+ export const NO_VERDICT_SHARE = 0.1
14
+
15
+ export function localDate(d = new Date()) {
16
+ const p = (n) => String(n).padStart(2, '0')
17
+ return `${d.getFullYear()}-${p(d.getMonth() + 1)}-${p(d.getDate())}`
18
+ }
19
+
20
+ export function summarise({ items, outcomes, models = [], skillName, description, overridden, facts, runsPerQuery, timeoutSeconds, planned, aborted, toolVersion, date = localDate() }) {
21
+ const queries = items.map((item, i) => {
22
+ const o = outcomes[i] ?? []
23
+ const hits = o.filter((x) => x === 'triggered').length
24
+ const runs = o.filter((x) => x === 'triggered' || x === 'not-triggered').length
25
+ const rate = runs ? hits / runs : null
26
+ const pass = rate === null ? null : item.should_trigger ? rate >= TRIGGER_THRESHOLD : rate < TRIGGER_THRESHOLD
27
+ // partial: some of its executed runs measured nothing, but not all of them.
28
+ return { ...item, outcomes: o, models: models[i] ?? [], hits, runs, timeouts: o.filter((x) => x === 'timeout').length, errors: o.filter((x) => x === 'error').length, pass, partial: runs > 0 && runs < o.length }
29
+ })
30
+ const sum = (list, k) => list.reduce((a, q) => a + q[k], 0)
31
+ const pos = queries.filter((q) => q.should_trigger)
32
+ const neg = queries.filter((q) => !q.should_trigger)
33
+ const executed = queries.reduce((a, q) => a + q.outcomes.length, 0)
34
+ const timeouts = sum(queries, 'timeouts')
35
+ const errors = sum(queries, 'errors')
36
+ const badShare = executed ? (timeouts + errors) / executed : 1
37
+ // The share counts runs, not queries: two positives can lose every run inside 10% and
38
+ // leave a positives total that silently omits them. A query that was run and measured
39
+ // nothing makes the verdict impossible to read query by query, so it is no verdict (ST-19).
40
+ const lostQueries = queries.filter((q) => q.outcomes.length > 0 && q.runs === 0).map((q) => q.query)
41
+ const partialQueries = queries.filter((q) => q.partial).map((q) => q.query)
42
+ const runModels = {}
43
+ for (const q of queries) for (const m of q.models) if (m) runModels[m] = (runModels[m] ?? 0) + 1
44
+ const verdict = !aborted && executed > 0 && badShare <= NO_VERDICT_SHARE && !lostQueries.length ? 'ok' : 'no-verdict'
45
+ return {
46
+ tool: 'skilltrigger',
47
+ toolVersion,
48
+ date,
49
+ skill: skillName,
50
+ cliVersion: facts.cliVersion,
51
+ model: facts.model ?? 'default',
52
+ roster: facts.roster,
53
+ runModels,
54
+ environment: facts.environment ?? null,
55
+ runsPerQuery,
56
+ timeoutSeconds,
57
+ triggerThreshold: TRIGGER_THRESHOLD,
58
+ noVerdictThreshold: NO_VERDICT_SHARE,
59
+ conflictAllowed: Boolean(facts.conflictAllowed),
60
+ description: { bytes: Buffer.byteLength(description), sha256: createHash('sha256').update(description).digest('hex'), overridden: Boolean(overridden) },
61
+ verdict,
62
+ aborted: Boolean(aborted),
63
+ lostQueries,
64
+ partialQueries,
65
+ totals: {
66
+ planned,
67
+ executed,
68
+ positives: { triggered: sum(pos, 'hits'), runs: sum(pos, 'runs') },
69
+ negatives: { fired: sum(neg, 'hits'), runs: sum(neg, 'runs') },
70
+ timeouts,
71
+ errors,
72
+ passed: queries.filter((q) => q.pass === true).length,
73
+ queries: queries.length,
74
+ },
75
+ queries,
76
+ }
77
+ }
78
+
79
+ const pct = (n) => `${Math.round(n * 100)}%`
80
+ const plural = (n, one, many = `${one}s`) => `${n} ${n === 1 ? one : many}`
81
+
82
+ export function formatEnvironment(e) {
83
+ if (!e) return 'unknown'
84
+ const mem = e.memoryFiles ? `${plural(e.memoryFiles.project, 'project memory file')}, ${plural(e.memoryFiles.user, 'user memory file')}` : 'memory files unknown'
85
+ return `${mem} · ${plural(e.hooks ?? 0, 'hook')} · ${e.mcpServers === null || e.mcpServers === undefined ? 'MCP servers not listed' : plural(e.mcpServers, 'MCP server')}`
86
+ }
87
+ export const formatRunModels = (m) => (m && Object.keys(m).length ? Object.entries(m).map(([k, v]) => `${k} ×${v}`).join(', ') : 'none reported')
88
+ const cell = (s) => String(s).replace(/\|/g, '\\|').replace(/\n/g, ' ')
89
+
90
+ const quoted = (list) => list.map((q) => `"${q.length > 60 ? `${q.slice(0, 59)}…` : q}"`).join(', ')
91
+ const queriesWord = (n) => `${n} ${n === 1 ? 'query' : 'queries'}`
92
+
93
+ export function headline(rep) {
94
+ const t = rep.totals
95
+ if (rep.verdict !== 'ok') {
96
+ const share = (t.timeouts + t.errors) / (t.executed || 1) > rep.noVerdictThreshold || !t.executed || rep.aborted
97
+ const lost = rep.lostQueries ?? []
98
+ const parts = []
99
+ if (share) parts.push(`${t.timeouts} timeout(s) and ${t.errors} error(s) in ${t.executed} run(s)${rep.aborted ? ` (stopped after ${t.executed} of ${t.planned})` : ''} — more than ${pct(rep.noVerdictThreshold)} of the runs did not measure anything`)
100
+ if (lost.length) parts.push(`${queriesWord(lost.length)} lost every run to timeouts or errors — ${quoted(lost)} — and the totals would leave ${lost.length === 1 ? 'it' : 'them'} out`)
101
+ if (!parts.length) parts.push(`${t.timeouts} timeout(s) and ${t.errors} error(s) in ${t.executed} run(s)`)
102
+ return `no verdict: ${parts.join('; ')}`
103
+ }
104
+ return `positives triggered ${t.positives.triggered}/${t.positives.runs} · negatives fired ${t.negatives.fired}/${t.negatives.runs} · ${t.timeouts} timeout(s), ${t.errors} error(s) counted apart`
105
+ }
106
+
107
+ // A line naming the queries that kept the verdict on fewer measured runs than planned.
108
+ export function partialLine(rep) {
109
+ const p = rep.partialQueries ?? []
110
+ return p.length ? `measured on fewer runs than planned: ${queriesWord(p.length)} — ${quoted(p)}` : null
111
+ }
112
+
113
+ export function toMarkdown(rep) {
114
+ const t = rep.totals
115
+ const roster = rep.roster ? `${rep.roster.slashCommands} slash commands${rep.roster.skills === null ? '' : `, ${rep.roster.skills} skills`}` : 'unknown'
116
+ const out = [
117
+ `# skilltrigger — ${rep.skill}, ${rep.date}`,
118
+ '',
119
+ rep.verdict === 'ok' ? `**${headline(rep)}**` : `**No verdict.** ${headline(rep).replace(/^no verdict: /, '')}. The per-query counts below are not a measurement.`,
120
+ '',
121
+ '| | |',
122
+ '|---|---|',
123
+ `| CLI | ${cell(rep.cliVersion)} |`,
124
+ `| Model | ${cell(rep.model)} |`,
125
+ `| Models the runs reported | ${cell(formatRunModels(rep.runModels))} |`,
126
+ `| Roster | ${roster} |`,
127
+ `| Inherited environment | ${formatEnvironment(rep.environment)} |`,
128
+ `| Runs per query | ${rep.runsPerQuery} |`,
129
+ `| Timeout | ${rep.timeoutSeconds} s |`,
130
+ `| Pass threshold | trigger rate ≥ ${rep.triggerThreshold} for a positive, < ${rep.triggerThreshold} for a negative |`,
131
+ `| No verdict when | timeouts + errors > ${pct(rep.noVerdictThreshold)} of runs, or any query lost every run |`,
132
+ `| Description | ${rep.description.bytes} bytes, sha256 ${rep.description.sha256.slice(0, 12)}${rep.description.overridden ? ', overridden with --description' : ''} |`,
133
+ ...(rep.conflictAllowed ? ['| Plugin conflict | **allowed with --allow-conflict** — a same-named skill was visible during the runs |'] : []),
134
+ `| Version | skilltrigger ${rep.toolVersion} |`,
135
+ '',
136
+ '| Query | Should trigger | Hits/runs | Timeouts | Errors | Pass |',
137
+ '|---|---|---:|---:|---:|---|',
138
+ ...rep.queries.map((q) => {
139
+ const pass = q.pass === null ? (q.outcomes.length ? 'lost every run' : '—') : q.pass ? 'pass' : 'fail'
140
+ return `| ${cell(q.query)} | ${q.should_trigger ? 'yes' : 'no'} | ${q.hits}/${q.runs} | ${q.timeouts} | ${q.errors} | ${pass}${q.partial ? ` (${q.runs} of ${q.outcomes.length} runs)` : ''} |`
141
+ }),
142
+ '',
143
+ ...(partialLine(rep) ? [`${cell(partialLine(rep).replace(/^m/, 'M'))}.`, ''] : []),
144
+ `${t.passed} of ${t.queries} queries pass. Two runs per query resolve to ±1 per query: read a difference of one hit as noise.`,
145
+ '',
146
+ ]
147
+ return out.join('\n')
148
+ }
149
+
150
+ // <out>/<date>-<skill>.json and .md; a second report the same day gets -2, -3…
151
+ export function writeReport(rep, outDir) {
152
+ mkdirSync(outDir, { recursive: true })
153
+ let base = `${rep.date}-${rep.skill}`
154
+ for (let n = 2; existsSync(join(outDir, `${base}.json`)) || existsSync(join(outDir, `${base}.md`)); n++) base = `${rep.date}-${rep.skill}-${n}`
155
+ const json = join(outDir, `${base}.json`)
156
+ const md = join(outDir, `${base}.md`)
157
+ writeFileSync(json, JSON.stringify(rep, null, 2) + '\n')
158
+ writeFileSync(md, toMarkdown(rep))
159
+ return { json, md }
160
+ }
@@ -0,0 +1,79 @@
1
+ // The runner. Strictly serial — there is no workers option, on purpose: with N
2
+ // parallel runs the model sees N stubs with the same description, invokes whichever
3
+ // it likes, and only the run whose stub was picked counts the hit, so the measured
4
+ // rate is about 1/N of the true one. skill-creator's default of ten workers measured
5
+ // a tenth of the truth.
6
+ //
7
+ // One run: a fresh temporary project holding one stub command whose frontmatter
8
+ // description is the text under test, a `claude -p` in it, the stream read until the
9
+ // first decisive message, the process killed, the directory removed.
10
+ import { randomBytes } from 'node:crypto'
11
+ import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
12
+ import { tmpdir } from 'node:os'
13
+ import { join } from 'node:path'
14
+ import { streamClaude } from './claude.mjs'
15
+ import { createDetector } from './stream.mjs'
16
+
17
+ export const OUTCOMES = ['triggered', 'not-triggered', 'timeout', 'error']
18
+
19
+ // A YAML block scalar, so quotes and colons in the description cannot break it.
20
+ export function stubText(skillName, description) {
21
+ const body = String(description).split('\n').map((l) => (l ? ` ${l}` : '')).join('\n')
22
+ return `---\ndescription: |\n${body}\n---\n\n# ${skillName}\n\nThis command stands in for the ${skillName} skill while its description is measured.\n`
23
+ }
24
+
25
+ export function makeProject(skillName, description) {
26
+ const dir = mkdtempSync(join(tmpdir(), 'skilltrigger-run-'))
27
+ const stubName = `${skillName}-stub-${randomBytes(4).toString('hex')}`
28
+ const commands = join(dir, '.claude', 'commands')
29
+ mkdirSync(commands, { recursive: true })
30
+ const stubPath = join(commands, `${stubName}.md`)
31
+ writeFileSync(stubPath, stubText(skillName, description))
32
+ return { dir, stubName, stubPath, remove: () => rmSync(dir, { recursive: true, force: true }) }
33
+ }
34
+
35
+ // → { outcome, reason, ms, model }
36
+ export async function runOnce({ query, skillName, description, model, timeoutMs, env }) {
37
+ const project = makeProject(skillName, description)
38
+ try {
39
+ const detector = createDetector({ stubName: project.stubName, stubPath: project.stubPath })
40
+ const args = ['-p', query, '--output-format', 'stream-json', '--verbose', '--include-partial-messages', '--no-session-persistence', ...(model ? ['--model', model] : [])]
41
+ const r = await streamClaude(args, { cwd: project.dir, env, timeoutMs, onEvent: (ev) => detector.feed(ev) })
42
+ const used = detector.init?.model ?? null
43
+ if (r.stopped) return { ...r.stopped, ms: r.ms, model: used }
44
+ if (r.timedOut) return { outcome: 'timeout', reason: `no decision within ${timeoutMs / 1000} s`, ms: r.ms, model: used }
45
+ if (r.notFound) return { outcome: 'error', reason: 'claude not found on PATH', ms: r.ms, model: used }
46
+ return { ...detector.end({ code: r.code, stderr: r.stderr }), ms: r.ms, model: used }
47
+ } finally {
48
+ project.remove()
49
+ }
50
+ }
51
+
52
+ // Serial, query by query, run by run. Stops as soon as the timeouts and errors pass
53
+ // the no-verdict share of the whole plan: from there no verdict is possible, and every
54
+ // further run is spent on a number that will not be reported.
55
+ export async function runAll({ items, runs, skillName, description, model, timeoutMs, env, noVerdictShare, onRun = () => {}, runOnceFn = runOnce }) {
56
+ const planned = items.length * runs
57
+ const outcomes = items.map(() => [])
58
+ // The model each run's init event reported, beside its outcome: an alias can resolve to
59
+ // a different model than the preflight saw (ST-20).
60
+ const models = items.map(() => [])
61
+ let bad = 0
62
+ let done = 0
63
+ let aborted = false
64
+ outer: for (let r = 0; r < runs; r++) {
65
+ for (const [i, item] of items.entries()) {
66
+ const res = await runOnceFn({ query: item.query, skillName, description, model, timeoutMs, env })
67
+ outcomes[i].push(res.outcome)
68
+ models[i].push(res.model ?? null)
69
+ done++
70
+ if (res.outcome === 'timeout' || res.outcome === 'error') bad++
71
+ onRun({ item, run: r + 1, done, planned, ...res })
72
+ if (bad > noVerdictShare * planned) {
73
+ aborted = done < planned
74
+ break outer
75
+ }
76
+ }
77
+ }
78
+ return { outcomes, models, planned, done, aborted }
79
+ }