@uinaf/skillcheck 0.5.4 → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -60,7 +60,7 @@ function stateDirs(root) {
60
60
  }
61
61
  function runOptions(flags) {
62
62
  const harness = flags.get("--harness") ?? "claude";
63
- if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
63
+ if (harness !== "claude" && harness !== "codex" && harness !== "grok") fail(`--harness must be claude, codex, or grok, got ${harness}`);
64
64
  if (flags.has("--max-turns") && harness !== "claude") fail("--max-turns is only supported with --harness claude");
65
65
  const agent = flags.get("--agent");
66
66
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
@@ -137,7 +137,7 @@ function runScenario(scenarioDir, opts, root) {
137
137
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
138
138
  scratchDir: dirs.scratch,
139
139
  transformPath: path.join(here, `transform${selfExt}`),
140
- cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
140
+ grokProviderPath: path.join(here, `grok-provider${selfExt}`)
141
141
  });
142
142
  fs.mkdirSync(dirs.results, { recursive: true });
143
143
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -212,7 +212,7 @@ function discoverScenarios(root) {
212
212
  }
213
213
  function cmdRun(argv) {
214
214
  const { positional, flags } = parseArgs(argv);
215
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
215
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|grok]");
216
216
  const opts = runOptions(flags);
217
217
  ensureEvalPackages(opts);
218
218
  const o = runScenario(positional[0], opts, resolveRoot(flags));
@@ -235,7 +235,7 @@ function cmdSweep(argv) {
235
235
  for (const dir of discoverScenarios(root)) {
236
236
  const name = runNameFor(dir, opts.harness);
237
237
  const resultPath = path.join(resultsDir, `${name}.json`);
238
- if (!all && fs.existsSync(resultPath)) {
238
+ if (!all && fs.existsSync(resultPath) && !fs.existsSync(attemptPath(resultPath))) {
239
239
  skipped++;
240
240
  console.log(`SKIP ${name} (results exist; use --all to rerun)`);
241
241
  continue;
@@ -256,13 +256,21 @@ function cmdSweep(argv) {
256
256
  process.exit(errored > 0 ? 2 : failed > 0 ? 1 : 0);
257
257
  }
258
258
  function resultIdentity(file) {
259
+ const decode = (part) => {
260
+ if (!part.startsWith("~v2~")) return part;
261
+ try {
262
+ return decodeURIComponent(part.slice(4));
263
+ } catch {
264
+ return part;
265
+ }
266
+ };
259
267
  const base = file.replace(/\.json$/, "");
260
- const suffix = base.match(/--(codex|cursor)$/);
268
+ const suffix = base.match(/--(codex|grok|cursor)$/);
261
269
  const harness = suffix === null ? "claude" : suffix[1];
262
- const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
270
+ const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
263
271
  return {
264
- skill,
265
- scenario: rest.join("--"),
272
+ skill: decode(skill),
273
+ scenario: decode(rest.join("--")),
266
274
  harness
267
275
  };
268
276
  }
@@ -274,6 +282,11 @@ function reduceResults(dir, allowMixed) {
274
282
  const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
275
283
  const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
276
284
  for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
285
+ if (f.endsWith("--cursor.json")) {
286
+ console.error(`skipping ${f}: Cursor harness is retired`);
287
+ skipped.push(f);
288
+ continue;
289
+ }
277
290
  if (incomplete.has(f)) {
278
291
  console.error(`skipping ${f}: attempt did not complete with a graded result`);
279
292
  skipped.push(f);
@@ -0,0 +1,144 @@
1
+ import { spawn } from "node:child_process";
2
+ import path from "node:path";
3
+ //#region src/grok-provider.ts
4
+ const OUTPUT_CAP = 1e6;
5
+ function within(root, file) {
6
+ const rel = path.relative(root, file);
7
+ return rel !== "" && rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
8
+ }
9
+ var GrokProvider = class {
10
+ config;
11
+ constructor(options) {
12
+ if (!options.config?.working_dir || !options.config.skill) throw new Error("grok provider requires working_dir and skill");
13
+ this.config = options.config;
14
+ }
15
+ id() {
16
+ return "grok:cli";
17
+ }
18
+ async callApi(prompt, _context, callOptions) {
19
+ const { working_dir: workdir, skill, model, command = "grok", timeout_ms = 6e5 } = this.config;
20
+ const skillFile = path.join(workdir, ".grok", "skills", skill, "SKILL.md");
21
+ const args = [
22
+ "--trust",
23
+ "--no-auto-update",
24
+ "--cwd",
25
+ workdir,
26
+ "--output-format",
27
+ "streaming-json",
28
+ "--permission-mode",
29
+ "acceptEdits",
30
+ "--disable-web-search",
31
+ "--no-subagents",
32
+ ...model ? ["--model", model] : [],
33
+ "-p",
34
+ prompt
35
+ ];
36
+ return new Promise((resolve) => {
37
+ const child = spawn(command, args, {
38
+ cwd: workdir,
39
+ detached: process.platform !== "win32",
40
+ stdio: [
41
+ "ignore",
42
+ "pipe",
43
+ "pipe"
44
+ ]
45
+ });
46
+ let stdout = "";
47
+ let stderr = "";
48
+ let output = "";
49
+ let stopReason;
50
+ let usage;
51
+ let cost;
52
+ let error;
53
+ const reads = /* @__PURE__ */ new Map();
54
+ const skillCalls = [];
55
+ const stop = (reason) => {
56
+ error ??= reason;
57
+ if (child.pid !== void 0) try {
58
+ if (process.platform === "win32") child.kill();
59
+ else process.kill(-child.pid, "SIGKILL");
60
+ } catch {}
61
+ };
62
+ const consume = (line) => {
63
+ if (!line.trim()) return;
64
+ let event;
65
+ try {
66
+ event = JSON.parse(line);
67
+ } catch {
68
+ stop("grok emitted invalid streaming JSON");
69
+ return;
70
+ }
71
+ if (event.type === "text" && typeof event.data === "string") {
72
+ output += event.data;
73
+ if (output.length > OUTPUT_CAP) stop("grok output exceeded limit");
74
+ } else if (event.type === "tool_call" && event.toolName === "read_file") {
75
+ const target = event.rawInput?.target_file;
76
+ if (event.toolCallId && typeof target === "string") {
77
+ const resolved = path.resolve(workdir, target);
78
+ if (within(workdir, resolved)) reads.set(event.toolCallId, resolved);
79
+ }
80
+ } else if (event.type === "tool_call_update" && event.status === "completed") {
81
+ const target = event.toolCallId && reads.get(event.toolCallId);
82
+ if (target === skillFile && !skillCalls.some((call) => call.path === target)) skillCalls.push({
83
+ name: skill,
84
+ source: "project",
85
+ path: target
86
+ });
87
+ } else if (event.type === "end") {
88
+ stopReason = event.stopReason;
89
+ usage = event.usage;
90
+ cost = event.total_cost_usd;
91
+ }
92
+ };
93
+ child.stdout?.on("data", (chunk) => {
94
+ stdout += chunk.toString("utf8");
95
+ if (stdout.length > OUTPUT_CAP) {
96
+ stop("grok event stream exceeded limit");
97
+ return;
98
+ }
99
+ let newline = stdout.indexOf("\n");
100
+ while (newline !== -1) {
101
+ consume(stdout.slice(0, newline));
102
+ stdout = stdout.slice(newline + 1);
103
+ newline = stdout.indexOf("\n");
104
+ }
105
+ });
106
+ child.stderr?.on("data", (chunk) => {
107
+ stderr = (stderr + chunk.toString("utf8")).slice(-4e3);
108
+ });
109
+ child.on("error", (cause) => {
110
+ error ??= `grok could not start: ${cause.message}`;
111
+ });
112
+ const timer = setTimeout(() => stop(`grok timed out after ${timeout_ms}ms`), timeout_ms);
113
+ const abort = () => stop("grok run aborted");
114
+ callOptions?.abortSignal?.addEventListener("abort", abort, { once: true });
115
+ if (callOptions?.abortSignal?.aborted) abort();
116
+ child.on("exit", () => {
117
+ if (process.platform !== "win32" && child.pid !== void 0) try {
118
+ process.kill(-child.pid, "SIGKILL");
119
+ } catch {}
120
+ });
121
+ child.on("close", (code, signal) => {
122
+ clearTimeout(timer);
123
+ callOptions?.abortSignal?.removeEventListener("abort", abort);
124
+ if (stdout) consume(stdout);
125
+ if (error) resolve({ error });
126
+ else if (code !== 0) resolve({ error: `grok exited ${signal ?? code}: ${stderr.trim() || "no diagnostics"}` });
127
+ else if (stopReason !== "end_turn") resolve({ error: `grok stopped with ${stopReason ?? "no terminal event"}` });
128
+ else resolve({
129
+ output,
130
+ metadata: { skillCalls },
131
+ ...usage ? { tokenUsage: {
132
+ prompt: usage.input_tokens ?? 0,
133
+ completion: usage.output_tokens ?? 0,
134
+ total: usage.total_tokens ?? 0,
135
+ cached: usage.cache_read_input_tokens ?? 0
136
+ } } : {},
137
+ ...typeof cost === "number" ? { cost } : {}
138
+ });
139
+ });
140
+ });
141
+ }
142
+ };
143
+ //#endregion
144
+ export { GrokProvider as default };
package/dist/scenario.js CHANGED
@@ -32,10 +32,15 @@ function loadScenario(scenarioDir) {
32
32
  criteria
33
33
  };
34
34
  }
35
+ function encodeRunNamePart(part) {
36
+ if (!part.includes("--") && !part.startsWith("-") && !part.endsWith("-") && !part.startsWith("~v2~")) return part;
37
+ return `~v2~${encodeURIComponent(part).replaceAll("-", "%2D").replaceAll("~", "%7E")}`;
38
+ }
35
39
  function runNameFor(scenarioDir, harness) {
36
40
  const m = path.resolve(scenarioDir).match(/skills\/([^/]+)\/evals\/([^/]+)$/);
37
41
  if (!m) throw new Error(`not a scenario dir (want .../skills/<skill>/evals/<scenario>): ${scenarioDir}`);
38
- return harness === "claude" ? `${m[1]}--${m[2]}` : `${m[1]}--${m[2]}--${harness}`;
42
+ const name = `${encodeRunNamePart(m[1])}--${encodeRunNamePart(m[2])}`;
43
+ return harness === "claude" ? name : `${name}--${harness}`;
39
44
  }
40
45
  function frontmatterRange(text) {
41
46
  const lines = text.split("\n");
@@ -56,7 +61,8 @@ function stripHiddenFlag(skillMd) {
56
61
  const RESERVED = /* @__PURE__ */ new Set([
57
62
  ".claude",
58
63
  ".agents",
59
- ".cursor"
64
+ ".grok",
65
+ "node_modules"
60
66
  ]);
61
67
  function materialize(s, runDir, harness) {
62
68
  const workdir = path.join(runDir, "workdir");
@@ -82,7 +88,7 @@ function materialize(s, runDir, harness) {
82
88
  fs.mkdirSync(path.dirname(dest), { recursive: true });
83
89
  fs.writeFileSync(dest, content);
84
90
  }
85
- const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
91
+ const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
86
92
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
87
93
  recursive: true,
88
94
  filter: (src) => path.basename(src) !== "evals"
@@ -98,7 +104,7 @@ function materialize(s, runDir, harness) {
98
104
  for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
99
105
  const p = path.join(dir, e.name);
100
106
  if (e.isDirectory()) {
101
- if (e.name !== ".claude") walk(p);
107
+ if (e.name !== ".claude" && e.name !== ".grok" && e.name !== "node_modules") walk(p);
102
108
  } else manifest[path.relative(workdir, p)] = createHash("sha256").update(fs.readFileSync(p)).digest("hex");
103
109
  }
104
110
  };
@@ -111,11 +117,12 @@ function materialize(s, runDir, harness) {
111
117
  };
112
118
  }
113
119
  function agentProvider(opts, workdir, skill, paths) {
114
- if (opts.harness === "cursor") return {
115
- id: `file://${paths.cursorProviderPath}`,
120
+ if (opts.harness === "grok") return {
121
+ id: `file://${paths.grokProviderPath}`,
116
122
  config: {
117
- ...opts.agentModel ? { model: opts.agentModel } : {},
118
- working_dir: workdir
123
+ working_dir: workdir,
124
+ skill,
125
+ ...opts.agentModel ? { model: opts.agentModel } : {}
119
126
  }
120
127
  };
121
128
  if (opts.harness === "codex") return {
@@ -256,4 +263,4 @@ function generateRun(scenarioDir, opts, paths) {
256
263
  };
257
264
  }
258
265
  //#endregion
259
- export { buildConfig, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
266
+ export { buildConfig, encodeRunNamePart, generateRun, isHiddenSkill, loadScenario, materialize, requiredEvalPackages, resolvePackageDir, runNameFor, sdkNodeModulesDir, stripHiddenFlag };
package/dist/transform.js CHANGED
@@ -23,7 +23,7 @@ function transform(output, context) {
23
23
  for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
24
24
  const p = path.join(dir, e.name);
25
25
  if (e.isDirectory()) {
26
- if (e.name !== ".claude") walk(p);
26
+ if (e.name !== ".claude" && e.name !== ".grok" && e.name !== "node_modules") walk(p);
27
27
  } else if (e.isFile()) {
28
28
  const rel = path.relative(workdir, p);
29
29
  visited.add(rel);
package/docs/scenarios.md CHANGED
@@ -8,8 +8,10 @@ A scenario is two files in a frozen location:
8
8
  ```
9
9
 
10
10
  The path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. On the codex and cursor harnesses the name gains a
12
- `--codex` or `--cursor` suffix, so every harness can hold results side by side.
11
+ and the scorecard entry. Codex and Grok results gain `--codex` and `--grok`
12
+ suffixes, so harnesses can hold results side by side. Double hyphens and
13
+ leading or trailing hyphens in a skill or scenario name are escaped in the run
14
+ and result filename; the scorecard keeps the original name.
13
15
  A directory missing either file is not discovered.
14
16
 
15
17
  ## task.md
@@ -28,8 +30,8 @@ Fix the failing check in the config below.
28
30
  Each block is replaced in the prompt with a pointer ("Input file `config.json`
29
31
  is available in your working directory.") and written to disk. Destinations
30
32
  must stay under the workdir, must not collide, and must not target `.claude/`,
31
- `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
- configuring its own examiner.
33
+ `.agents/`, `.grok/`, or `node_modules/`, since a fixture must not configure its
34
+ own examiner or embed generated dependencies.
33
35
 
34
36
  Write the task the way a user would write it. Do not name the skill, describe
35
37
  its steps, or hint at the checklist: routing is part of what is being measured.
@@ -83,8 +85,8 @@ frontmatter block; body text mentioning the key does not count.
83
85
  Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
86
  time. The skill under test is installed where the harness discovers skills:
85
87
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
- `.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
87
- excluded, so criteria never leak into the agent's context.
88
+ `.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
89
+ criteria never leak into the agent's context.
88
90
 
89
91
  Scenario quality is behavioral proof; [authoring](authoring.md) covers the
90
92
  judgment layer lint and evals cannot grade.
package/docs/usage.md CHANGED
@@ -34,6 +34,7 @@ clean, 1 with findings.
34
34
  skillcheck run skills/<skill>/evals/<scenario>
35
35
  skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
36
36
  skillcheck run <scenario-dir> --harness claude --max-turns 80
37
+ skillcheck run <scenario-dir> --harness grok
37
38
  ```
38
39
 
39
40
  Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
@@ -49,22 +50,15 @@ message, and writes no provenance sidecar. It is never reported as
49
50
 
50
51
  Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
51
52
  and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
52
- Claude; passing it with `codex` or `cursor` fails before the eval starts. On
53
+ Claude; passing it with `codex` or `grok` fails before the eval starts. On
53
54
  those harnesses, omitting `--agent` leaves the model to that CLI's own default.
54
55
 
55
- `--harness cursor` drives the scenario through the Cursor Agent CLI
56
- (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
57
- `--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
58
- cursor provider, so the run uses this package's own provider module, which
59
- replays the CLI's `stream-json` output: the `result` event becomes the graded
60
- output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
61
- evidence. Grading requires both a successful result event and harness exit code
62
- zero. On macOS and Linux, a supervisor owns the process group and kills remaining
63
- helpers when the harness exits, times out, or the provider disconnects. Output
64
- pipes have a separate two-second cleanup/drain deadline; incomplete cleanup is
65
- an error. Helpers that detach into another process group are outside this cleanup
66
- boundary; retained output pipes still cause a bounded error. Windows retains
67
- only direct-child cleanup. The judge leg is unchanged.
56
+ `--harness grok` runs the locally installed Grok Build CLI in the disposable
57
+ workdir with the skill under `.grok/skills/`. It uses native streaming events
58
+ to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
59
+ Grok must be logged in locally or have its supported credentials configured.
60
+ The run disables web search and subagents and grants edit permission in the
61
+ workdir. `--agent` selects a Grok model ID.
68
62
 
69
63
  `--judge` takes either a bare Claude model (graded through the Anthropic
70
64
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -83,7 +77,7 @@ provider-qualified judge; the Anthropic judge does not take one.
83
77
  ## Sweep
84
78
 
85
79
  ```sh
86
- skillcheck sweep # only scenarios without results
80
+ skillcheck sweep # scenarios without completed results
87
81
  skillcheck sweep --all # rerun everything
88
82
  ```
89
83
 
@@ -97,8 +91,8 @@ within one scenario, not across them.
97
91
  One known failure mode: judge calls through a gateway can drop at the transport
98
92
  layer ([uinaf/zebroid-infra#44](https://github.com/uinaf/zebroid-infra/issues/44)).
99
93
  That surfaces as an ERROR with no usable result, not as a graded FAIL, and the
100
- mitigation is a rerun. `sweep` without `--all` resumes, so a rerun only picks up
101
- what is missing.
94
+ mitigation is a rerun. `sweep` without `--all` resumes, so a rerun picks up
95
+ missing or ungraded results.
102
96
 
103
97
  ## Summarize
104
98
 
@@ -122,6 +116,8 @@ with a warning rather than failing the reduction. Graded assertion failures
122
116
  remain scored results. If a skipped file matches an existing scorecard row,
123
117
  summary generation fails and leaves the scorecard unchanged, so an errored rerun
124
118
  cannot carry forward its old score. This also applies with `--allow-mixed`.
119
+ Results from the retired Cursor harness are skipped with their original identity,
120
+ so they cannot become Claude scores or silently carry an old Cursor row.
125
121
  Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
126
122
  are written. An outstanding marker makes `summarize` skip that identity even
127
123
  when the child produced no result file or left partial output. The marker does
@@ -162,7 +158,6 @@ written inside the installed package.
162
158
  | `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
163
159
  | `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
164
160
  | `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
165
- | `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
166
161
  | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
167
162
 
168
163
  A bare `--judge` model stays on the Anthropic selection regardless of the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.5.4",
3
+ "version": "1.0.1",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -1,71 +0,0 @@
1
- import { spawn } from "node:child_process";
2
- //#region src/cursor-process.ts
3
- let child;
4
- let terminal = false;
5
- let cleaning = false;
6
- let watchdog = setTimeout(cleanup, 5e3);
7
- function cleanup() {
8
- if (cleaning) return;
9
- cleaning = true;
10
- clearTimeout(watchdog);
11
- if (process.platform !== "win32") process.kill(-process.pid, "SIGKILL");
12
- child?.kill("SIGKILL");
13
- process.exit(1);
14
- }
15
- function fail(error) {
16
- if (terminal) return;
17
- terminal = true;
18
- process.send?.({
19
- type: "failure",
20
- error
21
- }, cleanup);
22
- clearTimeout(watchdog);
23
- watchdog = setTimeout(cleanup, 1e3);
24
- }
25
- if (!process.send) process.exit(1);
26
- process.on("disconnect", cleanup);
27
- process.on("message", (message) => {
28
- if (message === "stop") return cleanup();
29
- if (message === "cleanup" && terminal) {
30
- process.send?.({ type: "cleanup" }, cleanup);
31
- return;
32
- }
33
- if (child !== void 0 || terminal) return;
34
- if (typeof message !== "object" || message === null || !("command" in message) || typeof message.command !== "string" || !("cwd" in message) || typeof message.cwd !== "string" || !("args" in message) || !Array.isArray(message.args) || !message.args.every((arg) => typeof arg === "string") || !("timeoutMs" in message) || typeof message.timeoutMs !== "number" || !Number.isFinite(message.timeoutMs) || message.timeoutMs <= 0) return fail("cursor-agent supervisor received invalid configuration");
35
- clearTimeout(watchdog);
36
- watchdog = setTimeout(() => fail("cursor-agent supervisor watchdog timed out"), message.timeoutMs + 1e3);
37
- const command = message.command;
38
- try {
39
- child = spawn(command, message.args, {
40
- cwd: message.cwd,
41
- stdio: [
42
- "pipe",
43
- "inherit",
44
- "inherit"
45
- ]
46
- });
47
- } catch (error) {
48
- return fail(`failed to spawn ${command}: ${error instanceof Error ? error.message : String(error)}`);
49
- }
50
- child.on("error", (error) => fail(`failed to spawn ${command}: ${error.message}`));
51
- child.stdin?.on("error", (error) => fail(`cursor-agent stdin failed: ${error.message}`));
52
- process.stdin.on("error", (error) => fail(`cursor-agent prompt stream failed: ${error.message}`));
53
- if (child.stdin) process.stdin.pipe(child.stdin);
54
- child.on("exit", (code, signal) => {
55
- if (terminal) return;
56
- if (!child?.stdin?.writableFinished) return fail("cursor-agent stdin closed before prompt delivery");
57
- terminal = true;
58
- process.stdin.unpipe();
59
- clearTimeout(watchdog);
60
- watchdog = setTimeout(cleanup, 1e3);
61
- process.send?.({
62
- type: "terminal",
63
- code,
64
- signal
65
- }, (error) => {
66
- if (error) cleanup();
67
- });
68
- });
69
- });
70
- //#endregion
71
- export {};
@@ -1,181 +0,0 @@
1
- import { fork } from "node:child_process";
2
- import path from "node:path";
3
- //#region src/cursor-provider.ts
4
- function newStreamState() {
5
- return {
6
- isError: false,
7
- skillCalls: []
8
- };
9
- }
10
- const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
11
- function foldLine(state, line) {
12
- let event;
13
- try {
14
- event = JSON.parse(line);
15
- } catch {
16
- return;
17
- }
18
- if (typeof event !== "object" || event === null) return;
19
- const e = event;
20
- if (e.type === "tool_call" && e.subtype === "started") {
21
- const p = e.tool_call?.readToolCall?.args?.path;
22
- const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
23
- if (m) state.skillCalls.push({
24
- name: m[1],
25
- source: "project",
26
- path: p
27
- });
28
- return;
29
- }
30
- if (e.type === "result") {
31
- state.isError = e.is_error === true || e.subtype !== "success";
32
- state.result = typeof e.result === "string" ? e.result : "";
33
- if (e.usage) {
34
- const prompt = e.usage.inputTokens ?? 0;
35
- const completion = e.usage.outputTokens ?? 0;
36
- state.tokenUsage = {
37
- prompt,
38
- completion,
39
- cached: e.usage.cacheReadTokens ?? 0,
40
- total: prompt + completion
41
- };
42
- }
43
- }
44
- }
45
- var CursorAgentProvider = class {
46
- providerId;
47
- config;
48
- constructor(options) {
49
- this.providerId = options.id ?? "cursor-agent";
50
- if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
51
- this.config = options.config;
52
- }
53
- id() {
54
- return this.providerId;
55
- }
56
- async callApi(prompt) {
57
- const command = this.config.command ?? "cursor-agent";
58
- const args = [
59
- "-p",
60
- "--trust",
61
- "--output-format",
62
- "stream-json"
63
- ];
64
- if (this.config.model !== void 0) args.push("--model", this.config.model);
65
- const state = newStreamState();
66
- const stderr = [];
67
- let stdoutBuf = "";
68
- return new Promise((resolve) => {
69
- const timeoutMs = this.config.timeout_ms ?? 9e5;
70
- const supervisor = fork(new URL(`./cursor-process${path.extname(import.meta.url)}`, import.meta.url), {
71
- execArgv: [],
72
- detached: process.platform !== "win32",
73
- stdio: [
74
- "pipe",
75
- "pipe",
76
- "pipe",
77
- "ipc"
78
- ]
79
- });
80
- let failure;
81
- let terminal;
82
- let cleanupConfirmed = false;
83
- let settled = false;
84
- let cleanupTimer;
85
- const finish = (closed, signal = null) => {
86
- if (settled) return;
87
- settled = true;
88
- clearTimeout(timer);
89
- clearTimeout(cleanupTimer);
90
- supervisor.stdin?.destroy();
91
- supervisor.stdout?.destroy();
92
- supervisor.stderr?.destroy();
93
- if (!closed) {
94
- supervisor.kill("SIGKILL");
95
- supervisor.unref();
96
- if (supervisor.connected) supervisor.disconnect();
97
- }
98
- if (stdoutBuf !== "") foldLine(state, stdoutBuf);
99
- const metadata = { skillCalls: state.skillCalls };
100
- const fail = (error) => resolve({
101
- error,
102
- tokenUsage: state.tokenUsage,
103
- metadata
104
- });
105
- if (failure !== void 0) {
106
- const tail = stderr.join("").trim().slice(-2e3);
107
- const missingResult = terminal !== void 0 && state.result === void 0 ? " without a result event" : "";
108
- return fail(`${failure}${missingResult}${tail === "" ? "" : `: ${tail}`}`);
109
- }
110
- if (!closed || !cleanupConfirmed || terminal === void 0 || process.platform !== "win32" && signal !== "SIGKILL") return fail("cursor-agent supervisor ended without confirmed cleanup");
111
- if (terminal.code !== 0 || terminal.signal !== null || state.result === void 0) {
112
- const tail = stderr.join("").trim().slice(-2e3);
113
- return fail(`cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}${state.result === void 0 ? " without a result event" : ""}${tail === "" ? "" : `: ${tail}`}`);
114
- }
115
- if (state.isError) return fail(state.result || "cursor-agent reported an error result");
116
- resolve({
117
- output: state.result,
118
- tokenUsage: state.tokenUsage,
119
- metadata
120
- });
121
- };
122
- const boundCleanup = () => {
123
- cleanupTimer ??= setTimeout(() => {
124
- failure = failure === void 0 ? "cursor-agent cleanup or pipe draining timed out" : `${failure}; cleanup or pipe draining timed out`;
125
- finish(false);
126
- }, 2e3);
127
- };
128
- const stop = (error) => {
129
- if (settled) return;
130
- failure ??= error;
131
- boundCleanup();
132
- if (supervisor.connected) supervisor.send("stop", () => {});
133
- };
134
- const timer = setTimeout(() => stop(`cursor-agent timed out after ${timeoutMs}ms`), timeoutMs);
135
- supervisor.on("error", (err) => stop(`cursor-agent supervisor failed: ${err.message}`));
136
- supervisor.on("disconnect", () => {
137
- if (!cleanupConfirmed && !settled) stop("cursor-agent supervisor disconnected before cleanup");
138
- });
139
- supervisor.on("message", (message) => {
140
- if (settled) return;
141
- if (typeof message !== "object" || message === null) return;
142
- if (!("type" in message)) return;
143
- if (message.type === "terminal" && "code" in message && "signal" in message && (message.code === null || typeof message.code === "number") && (message.signal === null || typeof message.signal === "string")) {
144
- terminal = {
145
- code: message.code,
146
- signal: message.signal
147
- };
148
- clearTimeout(timer);
149
- if (terminal.code !== 0 || terminal.signal !== null) failure ??= `cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}`;
150
- boundCleanup();
151
- supervisor.send("cleanup", (err) => {
152
- if (err) stop(`cursor-agent cleanup request failed: ${err.message}`);
153
- });
154
- } else if (message.type === "failure" && "error" in message && typeof message.error === "string") stop(message.error);
155
- else if (message.type === "cleanup" && terminal !== void 0) cleanupConfirmed = true;
156
- });
157
- supervisor.stdout?.on("data", (chunk) => {
158
- stdoutBuf += chunk.toString("utf8");
159
- const lines = stdoutBuf.split("\n");
160
- stdoutBuf = lines.pop() ?? "";
161
- for (const line of lines) foldLine(state, line);
162
- });
163
- supervisor.stderr?.on("data", (chunk) => stderr.push(chunk.toString("utf8")));
164
- supervisor.stdout?.on("error", (err) => stop(`cursor-agent stdout failed: ${err.message}`));
165
- supervisor.stderr?.on("error", (err) => stop(`cursor-agent stderr failed: ${err.message}`));
166
- supervisor.stdin?.on("error", (err) => stop(`cursor-agent stdin failed: ${err.message}`));
167
- supervisor.on("close", (_code, signal) => finish(true, signal));
168
- supervisor.send({
169
- command,
170
- args,
171
- cwd: this.config.working_dir,
172
- timeoutMs
173
- }, (err) => {
174
- if (err) stop(`cursor-agent supervisor initialization failed: ${err.message}`);
175
- });
176
- supervisor.stdin?.end(prompt);
177
- });
178
- }
179
- };
180
- //#endregion
181
- export { CursorAgentProvider as default, foldLine, newStreamState };