@uinaf/skillcheck 0.5.3 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -60,7 +60,8 @@ function stateDirs(root) {
60
60
  }
61
61
  function runOptions(flags) {
62
62
  const harness = flags.get("--harness") ?? "claude";
63
- if (harness !== "claude" && harness !== "codex" && harness !== "cursor") fail(`--harness must be claude, codex, or cursor, got ${harness}`);
63
+ if (harness !== "claude" && harness !== "codex" && harness !== "grok") fail(`--harness must be claude, codex, or grok, got ${harness}`);
64
+ if (flags.has("--max-turns") && harness !== "claude") fail("--max-turns is only supported with --harness claude");
64
65
  const agent = flags.get("--agent");
65
66
  const judgeModel = flags.get("--judge") ?? "claude-opus-5";
66
67
  const judgeEffort = flags.get("--judge-effort");
@@ -136,7 +137,7 @@ function runScenario(scenarioDir, opts, root) {
136
137
  const { name, configPath } = generateRun(path.resolve(scenarioDir), opts, {
137
138
  scratchDir: dirs.scratch,
138
139
  transformPath: path.join(here, `transform${selfExt}`),
139
- cursorProviderPath: path.join(here, `cursor-provider${selfExt}`)
140
+ grokProviderPath: path.join(here, `grok-provider${selfExt}`)
140
141
  });
141
142
  fs.mkdirSync(dirs.results, { recursive: true });
142
143
  const resultPath = path.join(dirs.results, `${name}.json`);
@@ -211,7 +212,7 @@ function discoverScenarios(root) {
211
212
  }
212
213
  function cmdRun(argv) {
213
214
  const { positional, flags } = parseArgs(argv);
214
- if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|cursor]");
215
+ if (positional.length !== 1) fail("usage: skillcheck run <scenario-dir> [--root DIR] [--agent MODEL] [--judge MODEL] [--judge-effort EFFORT] [--harness claude|codex|grok]");
215
216
  const opts = runOptions(flags);
216
217
  ensureEvalPackages(opts);
217
218
  const o = runScenario(positional[0], opts, resolveRoot(flags));
@@ -256,9 +257,9 @@ function cmdSweep(argv) {
256
257
  }
257
258
  function resultIdentity(file) {
258
259
  const base = file.replace(/\.json$/, "");
259
- const suffix = base.match(/--(codex|cursor)$/);
260
+ const suffix = base.match(/--(codex|grok|cursor)$/);
260
261
  const harness = suffix === null ? "claude" : suffix[1];
261
- const [skill, ...rest] = base.replace(/--(codex|cursor)$/, "").split("--");
262
+ const [skill, ...rest] = base.replace(/--(codex|grok|cursor)$/, "").split("--");
262
263
  return {
263
264
  skill,
264
265
  scenario: rest.join("--"),
@@ -273,6 +274,11 @@ function reduceResults(dir, allowMixed) {
273
274
  const incomplete = new Set(files.filter((f) => f.endsWith(".json.attempt")).map((f) => f.replace(/\.attempt$/, "")));
274
275
  const results = files.filter((f) => f.endsWith(".json") && !f.endsWith(".meta.json"));
275
276
  for (const f of [.../* @__PURE__ */ new Set([...results, ...incomplete])].sort()) {
277
+ if (f.endsWith("--cursor.json")) {
278
+ console.error(`skipping ${f}: Cursor harness is retired`);
279
+ skipped.push(f);
280
+ continue;
281
+ }
276
282
  if (incomplete.has(f)) {
277
283
  console.error(`skipping ${f}: attempt did not complete with a graded result`);
278
284
  skipped.push(f);
@@ -0,0 +1,139 @@
1
+ import { spawn } from "node:child_process";
2
+ import path from "node:path";
3
+ //#region src/grok-provider.ts
4
+ const OUTPUT_CAP = 1e6;
5
+ function within(root, file) {
6
+ const rel = path.relative(root, file);
7
+ return rel !== "" && rel !== ".." && !rel.startsWith(`..${path.sep}`) && !path.isAbsolute(rel);
8
+ }
9
+ var GrokProvider = class {
10
+ config;
11
+ constructor(options) {
12
+ if (!options.config?.working_dir || !options.config.skill) throw new Error("grok provider requires working_dir and skill");
13
+ this.config = options.config;
14
+ }
15
+ id() {
16
+ return "grok:cli";
17
+ }
18
+ async callApi(prompt, _context, callOptions) {
19
+ const { working_dir: workdir, skill, model, command = "grok", timeout_ms = 6e5 } = this.config;
20
+ const skillFile = path.join(workdir, ".grok", "skills", skill, "SKILL.md");
21
+ const args = [
22
+ "--trust",
23
+ "--no-auto-update",
24
+ "--cwd",
25
+ workdir,
26
+ "--output-format",
27
+ "streaming-json",
28
+ "--permission-mode",
29
+ "acceptEdits",
30
+ "--disable-web-search",
31
+ "--no-subagents",
32
+ ...model ? ["--model", model] : [],
33
+ "-p",
34
+ prompt
35
+ ];
36
+ return new Promise((resolve) => {
37
+ const child = spawn(command, args, {
38
+ cwd: workdir,
39
+ detached: process.platform !== "win32",
40
+ stdio: [
41
+ "ignore",
42
+ "pipe",
43
+ "pipe"
44
+ ]
45
+ });
46
+ let stdout = "";
47
+ let stderr = "";
48
+ let output = "";
49
+ let stopReason;
50
+ let usage;
51
+ let cost;
52
+ let error;
53
+ const reads = /* @__PURE__ */ new Map();
54
+ const skillCalls = [];
55
+ const stop = (reason) => {
56
+ error ??= reason;
57
+ if (child.pid !== void 0) try {
58
+ if (process.platform === "win32") child.kill();
59
+ else process.kill(-child.pid, "SIGKILL");
60
+ } catch {}
61
+ };
62
+ const consume = (line) => {
63
+ if (!line.trim()) return;
64
+ let event;
65
+ try {
66
+ event = JSON.parse(line);
67
+ } catch {
68
+ stop("grok emitted invalid streaming JSON");
69
+ return;
70
+ }
71
+ if (event.type === "text" && typeof event.data === "string") {
72
+ output += event.data;
73
+ if (output.length > OUTPUT_CAP) stop("grok output exceeded limit");
74
+ } else if (event.type === "tool_call" && event.toolName === "read_file") {
75
+ const target = event.rawInput?.target_file;
76
+ if (event.toolCallId && typeof target === "string") {
77
+ const resolved = path.resolve(workdir, target);
78
+ if (within(workdir, resolved)) reads.set(event.toolCallId, resolved);
79
+ }
80
+ } else if (event.type === "tool_call_update" && event.status === "completed") {
81
+ const target = event.toolCallId && reads.get(event.toolCallId);
82
+ if (target === skillFile && !skillCalls.some((call) => call.path === target)) skillCalls.push({
83
+ name: skill,
84
+ source: "project",
85
+ path: target
86
+ });
87
+ } else if (event.type === "end") {
88
+ stopReason = event.stopReason;
89
+ usage = event.usage;
90
+ cost = event.total_cost_usd;
91
+ }
92
+ };
93
+ child.stdout?.on("data", (chunk) => {
94
+ stdout += chunk.toString("utf8");
95
+ if (stdout.length > OUTPUT_CAP) {
96
+ stop("grok event stream exceeded limit");
97
+ return;
98
+ }
99
+ let newline = stdout.indexOf("\n");
100
+ while (newline !== -1) {
101
+ consume(stdout.slice(0, newline));
102
+ stdout = stdout.slice(newline + 1);
103
+ newline = stdout.indexOf("\n");
104
+ }
105
+ });
106
+ child.stderr?.on("data", (chunk) => {
107
+ stderr = (stderr + chunk.toString("utf8")).slice(-4e3);
108
+ });
109
+ child.on("error", (cause) => {
110
+ error ??= `grok could not start: ${cause.message}`;
111
+ });
112
+ const timer = setTimeout(() => stop(`grok timed out after ${timeout_ms}ms`), timeout_ms);
113
+ const abort = () => stop("grok run aborted");
114
+ callOptions?.abortSignal?.addEventListener("abort", abort, { once: true });
115
+ if (callOptions?.abortSignal?.aborted) abort();
116
+ child.on("close", (code, signal) => {
117
+ clearTimeout(timer);
118
+ callOptions?.abortSignal?.removeEventListener("abort", abort);
119
+ if (stdout) consume(stdout);
120
+ if (error) resolve({ error });
121
+ else if (code !== 0) resolve({ error: `grok exited ${signal ?? code}: ${stderr.trim() || "no diagnostics"}` });
122
+ else if (stopReason !== "end_turn") resolve({ error: `grok stopped with ${stopReason ?? "no terminal event"}` });
123
+ else resolve({
124
+ output,
125
+ metadata: { skillCalls },
126
+ ...usage ? { tokenUsage: {
127
+ prompt: usage.input_tokens ?? 0,
128
+ completion: usage.output_tokens ?? 0,
129
+ total: usage.total_tokens ?? 0,
130
+ cached: usage.cache_read_input_tokens ?? 0
131
+ } } : {},
132
+ ...typeof cost === "number" ? { cost } : {}
133
+ });
134
+ });
135
+ });
136
+ }
137
+ };
138
+ //#endregion
139
+ export { GrokProvider as default };
package/dist/scenario.js CHANGED
@@ -56,7 +56,8 @@ function stripHiddenFlag(skillMd) {
56
56
  const RESERVED = /* @__PURE__ */ new Set([
57
57
  ".claude",
58
58
  ".agents",
59
- ".cursor"
59
+ ".grok",
60
+ "node_modules"
60
61
  ]);
61
62
  function materialize(s, runDir, harness) {
62
63
  const workdir = path.join(runDir, "workdir");
@@ -82,7 +83,7 @@ function materialize(s, runDir, harness) {
82
83
  fs.mkdirSync(path.dirname(dest), { recursive: true });
83
84
  fs.writeFileSync(dest, content);
84
85
  }
85
- const roots = harness === "codex" ? [".claude", ".agents"] : harness === "cursor" ? [".cursor"] : [".claude"];
86
+ const roots = harness === "codex" ? [".claude", ".agents"] : harness === "grok" ? [".grok"] : [".claude"];
86
87
  for (const root of roots) fs.cpSync(s.skillDir, path.join(workdir, root, "skills", s.skill), {
87
88
  recursive: true,
88
89
  filter: (src) => path.basename(src) !== "evals"
@@ -98,7 +99,7 @@ function materialize(s, runDir, harness) {
98
99
  for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
99
100
  const p = path.join(dir, e.name);
100
101
  if (e.isDirectory()) {
101
- if (e.name !== ".claude") walk(p);
102
+ if (e.name !== ".claude" && e.name !== ".grok" && e.name !== "node_modules") walk(p);
102
103
  } else manifest[path.relative(workdir, p)] = createHash("sha256").update(fs.readFileSync(p)).digest("hex");
103
104
  }
104
105
  };
@@ -111,11 +112,12 @@ function materialize(s, runDir, harness) {
111
112
  };
112
113
  }
113
114
  function agentProvider(opts, workdir, skill, paths) {
114
- if (opts.harness === "cursor") return {
115
- id: `file://${paths.cursorProviderPath}`,
115
+ if (opts.harness === "grok") return {
116
+ id: `file://${paths.grokProviderPath}`,
116
117
  config: {
117
- ...opts.agentModel ? { model: opts.agentModel } : {},
118
- working_dir: workdir
118
+ working_dir: workdir,
119
+ skill,
120
+ ...opts.agentModel ? { model: opts.agentModel } : {}
119
121
  }
120
122
  };
121
123
  if (opts.harness === "codex") return {
package/dist/transform.js CHANGED
@@ -23,7 +23,7 @@ function transform(output, context) {
23
23
  for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
24
24
  const p = path.join(dir, e.name);
25
25
  if (e.isDirectory()) {
26
- if (e.name !== ".claude") walk(p);
26
+ if (e.name !== ".claude" && e.name !== ".grok" && e.name !== "node_modules") walk(p);
27
27
  } else if (e.isFile()) {
28
28
  const rel = path.relative(workdir, p);
29
29
  visited.add(rel);
package/docs/releasing.md CHANGED
@@ -6,7 +6,7 @@ A push to `main` runs one workflow, `.github/workflows/release.yml`:
6
6
 
7
7
  ```text
8
8
  verify ──┐
9
- ├──> release npm publish, OIDC + uinaf-releaser (release environment)
9
+ ├──> release npm publish, OIDC + uinaf-ci (release environment)
10
10
  scan ────┘
11
11
  ```
12
12
 
@@ -20,15 +20,15 @@ The file name `release.yml` is load-bearing. See below.
20
20
  ## npm
21
21
 
22
22
  `@uinaf/skillcheck` publishes from `.github/workflows/release.yml` via npm
23
- Trusted Publishing (OIDC) and the `uinaf-releaser` GitHub App. There is no npm
23
+ Trusted Publishing (OIDC) and the `uinaf-ci` GitHub App. There is no npm
24
24
  token in this repository, in its environments, or in the organization.
25
25
 
26
26
  Required on the `release` GitHub Environment:
27
27
 
28
- | Name | Kind | Purpose |
29
- | ------------------------------- | ------ | ------------------------------------------- |
30
- | `UINAF_RELEASE_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
31
- | `UINAF_RELEASE_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
28
+ | Name | Kind | Purpose |
29
+ | -------------------------- | ------ | ------------------------------------------- |
30
+ | `UINAF_CI_APP_CLIENT_ID` | var | GitHub App client id for the releaser bot |
31
+ | `UINAF_CI_APP_PRIVATE_KEY` | secret | GitHub App private key for the releaser bot |
32
32
 
33
33
  The trusted publisher on npmjs.com is registered by **file path**, so
34
34
  `.github/workflows/release.yml` cannot be renamed or moved without editing that
@@ -40,6 +40,7 @@ Deleting the `release` environment deletes both rows above with it, and there is
40
40
  no repo-level fallback: `create-github-app-token` then runs with empty inputs
41
41
  and the job fails at that step. The private key cannot be read back from
42
42
  GitHub; recreating it means generating a new one in the App settings.
43
+ The App private key lives on the `release` Environment; the shared release-npm job binds that Environment, so GitHub substitutes the Environment value for the secret named in `release.yml`.
43
44
 
44
45
  ## Version history
45
46
 
package/docs/scenarios.md CHANGED
@@ -8,8 +8,8 @@ A scenario is two files in a frozen location:
8
8
  ```
9
9
 
10
10
  The path is the identity: `<skill>--<scenario>` names the run, the result file,
11
- and the scorecard entry. On the codex and cursor harnesses the name gains a
12
- `--codex` or `--cursor` suffix, so every harness can hold results side by side.
11
+ and the scorecard entry. Codex and Grok results gain `--codex` and `--grok`
12
+ suffixes, so harnesses can hold results side by side.
13
13
  A directory missing either file is not discovered.
14
14
 
15
15
  ## task.md
@@ -28,8 +28,8 @@ Fix the failing check in the config below.
28
28
  Each block is replaced in the prompt with a pointer ("Input file `config.json`
29
29
  is available in your working directory.") and written to disk. Destinations
30
30
  must stay under the workdir, must not collide, and must not target `.claude/`,
31
- `.agents/`, or `.cursor/`, since a fixture that writes agent config would be
32
- configuring its own examiner.
31
+ `.agents/`, `.grok/`, or `node_modules/`, since a fixture must not configure its
32
+ own examiner or embed generated dependencies.
33
33
 
34
34
  Write the task the way a user would write it. Do not name the skill, describe
35
35
  its steps, or hint at the checklist: routing is part of what is being measured.
@@ -83,8 +83,8 @@ frontmatter block; body text mentioning the key does not count.
83
83
  Per run, under `<root>/.skillcheck/scratch/<name>/`, rebuilt from scratch each
84
84
  time. The skill under test is installed where the harness discovers skills:
85
85
  `.claude/skills/<skill>/`, plus `.agents/skills/<skill>/` on codex, or
86
- `.cursor/skills/<skill>/` alone on cursor, with its `evals/` directory
87
- excluded, so criteria never leak into the agent's context.
86
+ `.grok/skills/<skill>/` on Grok, with its `evals/` directory excluded, so
87
+ criteria never leak into the agent's context.
88
88
 
89
89
  Scenario quality is behavioral proof; [authoring](authoring.md) covers the
90
90
  judgment layer lint and evals cannot grade.
package/docs/usage.md CHANGED
@@ -32,7 +32,9 @@ clean, 1 with findings.
32
32
 
33
33
  ```sh
34
34
  skillcheck run skills/<skill>/evals/<scenario>
35
- skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex --max-turns 80
35
+ skillcheck run <scenario-dir> --agent MODEL --judge MODEL --harness codex
36
+ skillcheck run <scenario-dir> --harness claude --max-turns 80
37
+ skillcheck run <scenario-dir> --harness grok
36
38
  ```
37
39
 
38
40
  Materializes the scenario into `<root>/.skillcheck/scratch/<name>/workdir`,
@@ -47,22 +49,16 @@ message, and writes no provenance sidecar. It is never reported as
47
49
  `FAIL score=0.0000`; only a real judged verdict can fail a run.
48
50
 
49
51
  Defaults: `--harness claude`, agent `claude-opus-5`, judge `claude-opus-5`,
50
- `--max-turns 50`. On the codex and cursor harnesses, omitting `--agent` leaves
51
- the model to that CLI's own default.
52
-
53
- `--harness cursor` drives the scenario through the Cursor Agent CLI
54
- (`cursor-agent` on PATH) with the skill installed under `.cursor/skills/`;
55
- `--agent` names a Cursor model id, e.g. `composer-2.5`. There is no promptfoo
56
- cursor provider, so the run uses this package's own provider module, which
57
- replays the CLI's `stream-json` output: the `result` event becomes the graded
58
- output and `SKILL.md` reads under `.cursor/skills/` become the `skill-used`
59
- evidence. Grading requires both a successful result event and harness exit code
60
- zero. On macOS and Linux, a supervisor owns the process group and kills remaining
61
- helpers when the harness exits, times out, or the provider disconnects. Output
62
- pipes have a separate two-second cleanup/drain deadline; incomplete cleanup is
63
- an error. Helpers that detach into another process group are outside this cleanup
64
- boundary; retained output pipes still cause a bounded error. Windows retains
65
- only direct-child cleanup. The judge leg is unchanged.
52
+ and a Claude agent limit of 50 turns. `--max-turns` changes that limit only for
53
+ Claude; passing it with `codex` or `grok` fails before the eval starts. On
54
+ those harnesses, omitting `--agent` leaves the model to that CLI's own default.
55
+
56
+ `--harness grok` runs the locally installed Grok Build CLI in the disposable
57
+ workdir with the skill under `.grok/skills/`. It uses native streaming events
58
+ to count a completed read of that skill's `SKILL.md` as `skill-used` evidence.
59
+ Grok must be logged in locally or have its supported credentials configured.
60
+ The run disables web search and subagents and grants edit permission in the
61
+ workdir. `--agent` selects a Grok model ID.
66
62
 
67
63
  `--judge` takes either a bare Claude model (graded through the Anthropic
68
64
  selection in [auth](#auth)) or a provider-qualified promptfoo id, passed
@@ -120,6 +116,8 @@ with a warning rather than failing the reduction. Graded assertion failures
120
116
  remain scored results. If a skipped file matches an existing scorecard row,
121
117
  summary generation fails and leaves the scorecard unchanged, so an errored rerun
122
118
  cannot carry forward its old score. This also applies with `--allow-mixed`.
119
+ Results from the retired Cursor harness are skipped with their original identity,
120
+ so they cannot become Claude scores or silently carry an old Cursor row.
123
121
  Runs keep a `<name>.json.attempt` marker until a graded result and its provenance
124
122
  are written. An outstanding marker makes `summarize` skip that identity even
125
123
  when the child produced no result file or left partial output. The marker does
@@ -160,7 +158,6 @@ written inside the installed package.
160
158
  | `ANTHROPIC_API_KEY` | Judge grades over `anthropic:messages:<model>` instead of the SDK |
161
159
  | `CODEX_HOME` (default `~/.codex`) | Where the codex harness finds the local `codex` CLI login |
162
160
  | `OPENAI_API_KEY` | Agent auth for codex when there is no local login |
163
- | `CURSOR_API_KEY` | Agent auth for cursor; a logged-in `cursor-agent` also works |
164
161
  | `OPENAI_API_KEY` + `OPENAI_BASE_URL` | A provider-qualified `--judge openai:…`, optionally via a gateway |
165
162
 
166
163
  A bare `--judge` model stays on the Anthropic selection regardless of the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uinaf/skillcheck",
3
- "version": "0.5.3",
3
+ "version": "1.0.0",
4
4
  "description": "Lint and eval harness for agent skills",
5
5
  "homepage": "https://github.com/uinaf/skillcheck#readme",
6
6
  "bugs": {
@@ -1,71 +0,0 @@
1
- import { spawn } from "node:child_process";
2
- //#region src/cursor-process.ts
3
- let child;
4
- let terminal = false;
5
- let cleaning = false;
6
- let watchdog = setTimeout(cleanup, 5e3);
7
- function cleanup() {
8
- if (cleaning) return;
9
- cleaning = true;
10
- clearTimeout(watchdog);
11
- if (process.platform !== "win32") process.kill(-process.pid, "SIGKILL");
12
- child?.kill("SIGKILL");
13
- process.exit(1);
14
- }
15
- function fail(error) {
16
- if (terminal) return;
17
- terminal = true;
18
- process.send?.({
19
- type: "failure",
20
- error
21
- }, cleanup);
22
- clearTimeout(watchdog);
23
- watchdog = setTimeout(cleanup, 1e3);
24
- }
25
- if (!process.send) process.exit(1);
26
- process.on("disconnect", cleanup);
27
- process.on("message", (message) => {
28
- if (message === "stop") return cleanup();
29
- if (message === "cleanup" && terminal) {
30
- process.send?.({ type: "cleanup" }, cleanup);
31
- return;
32
- }
33
- if (child !== void 0 || terminal) return;
34
- if (typeof message !== "object" || message === null || !("command" in message) || typeof message.command !== "string" || !("cwd" in message) || typeof message.cwd !== "string" || !("args" in message) || !Array.isArray(message.args) || !message.args.every((arg) => typeof arg === "string") || !("timeoutMs" in message) || typeof message.timeoutMs !== "number" || !Number.isFinite(message.timeoutMs) || message.timeoutMs <= 0) return fail("cursor-agent supervisor received invalid configuration");
35
- clearTimeout(watchdog);
36
- watchdog = setTimeout(() => fail("cursor-agent supervisor watchdog timed out"), message.timeoutMs + 1e3);
37
- const command = message.command;
38
- try {
39
- child = spawn(command, message.args, {
40
- cwd: message.cwd,
41
- stdio: [
42
- "pipe",
43
- "inherit",
44
- "inherit"
45
- ]
46
- });
47
- } catch (error) {
48
- return fail(`failed to spawn ${command}: ${error instanceof Error ? error.message : String(error)}`);
49
- }
50
- child.on("error", (error) => fail(`failed to spawn ${command}: ${error.message}`));
51
- child.stdin?.on("error", (error) => fail(`cursor-agent stdin failed: ${error.message}`));
52
- process.stdin.on("error", (error) => fail(`cursor-agent prompt stream failed: ${error.message}`));
53
- if (child.stdin) process.stdin.pipe(child.stdin);
54
- child.on("exit", (code, signal) => {
55
- if (terminal) return;
56
- if (!child?.stdin?.writableFinished) return fail("cursor-agent stdin closed before prompt delivery");
57
- terminal = true;
58
- process.stdin.unpipe();
59
- clearTimeout(watchdog);
60
- watchdog = setTimeout(cleanup, 1e3);
61
- process.send?.({
62
- type: "terminal",
63
- code,
64
- signal
65
- }, (error) => {
66
- if (error) cleanup();
67
- });
68
- });
69
- });
70
- //#endregion
71
- export {};
@@ -1,181 +0,0 @@
1
- import { fork } from "node:child_process";
2
- import path from "node:path";
3
- //#region src/cursor-provider.ts
4
- function newStreamState() {
5
- return {
6
- isError: false,
7
- skillCalls: []
8
- };
9
- }
10
- const SKILL_MD = /(?:^|\/)\.cursor\/skills\/([^/]+)\/SKILL\.md$/;
11
- function foldLine(state, line) {
12
- let event;
13
- try {
14
- event = JSON.parse(line);
15
- } catch {
16
- return;
17
- }
18
- if (typeof event !== "object" || event === null) return;
19
- const e = event;
20
- if (e.type === "tool_call" && e.subtype === "started") {
21
- const p = e.tool_call?.readToolCall?.args?.path;
22
- const m = typeof p === "string" ? SKILL_MD.exec(p) : null;
23
- if (m) state.skillCalls.push({
24
- name: m[1],
25
- source: "project",
26
- path: p
27
- });
28
- return;
29
- }
30
- if (e.type === "result") {
31
- state.isError = e.is_error === true || e.subtype !== "success";
32
- state.result = typeof e.result === "string" ? e.result : "";
33
- if (e.usage) {
34
- const prompt = e.usage.inputTokens ?? 0;
35
- const completion = e.usage.outputTokens ?? 0;
36
- state.tokenUsage = {
37
- prompt,
38
- completion,
39
- cached: e.usage.cacheReadTokens ?? 0,
40
- total: prompt + completion
41
- };
42
- }
43
- }
44
- }
45
- var CursorAgentProvider = class {
46
- providerId;
47
- config;
48
- constructor(options) {
49
- this.providerId = options.id ?? "cursor-agent";
50
- if (options.config?.working_dir === void 0) throw new Error("cursor-agent provider requires config.working_dir");
51
- this.config = options.config;
52
- }
53
- id() {
54
- return this.providerId;
55
- }
56
- async callApi(prompt) {
57
- const command = this.config.command ?? "cursor-agent";
58
- const args = [
59
- "-p",
60
- "--trust",
61
- "--output-format",
62
- "stream-json"
63
- ];
64
- if (this.config.model !== void 0) args.push("--model", this.config.model);
65
- const state = newStreamState();
66
- const stderr = [];
67
- let stdoutBuf = "";
68
- return new Promise((resolve) => {
69
- const timeoutMs = this.config.timeout_ms ?? 9e5;
70
- const supervisor = fork(new URL(`./cursor-process${path.extname(import.meta.url)}`, import.meta.url), {
71
- execArgv: [],
72
- detached: process.platform !== "win32",
73
- stdio: [
74
- "pipe",
75
- "pipe",
76
- "pipe",
77
- "ipc"
78
- ]
79
- });
80
- let failure;
81
- let terminal;
82
- let cleanupConfirmed = false;
83
- let settled = false;
84
- let cleanupTimer;
85
- const finish = (closed, signal = null) => {
86
- if (settled) return;
87
- settled = true;
88
- clearTimeout(timer);
89
- clearTimeout(cleanupTimer);
90
- supervisor.stdin?.destroy();
91
- supervisor.stdout?.destroy();
92
- supervisor.stderr?.destroy();
93
- if (!closed) {
94
- supervisor.kill("SIGKILL");
95
- supervisor.unref();
96
- if (supervisor.connected) supervisor.disconnect();
97
- }
98
- if (stdoutBuf !== "") foldLine(state, stdoutBuf);
99
- const metadata = { skillCalls: state.skillCalls };
100
- const fail = (error) => resolve({
101
- error,
102
- tokenUsage: state.tokenUsage,
103
- metadata
104
- });
105
- if (failure !== void 0) {
106
- const tail = stderr.join("").trim().slice(-2e3);
107
- const missingResult = terminal !== void 0 && state.result === void 0 ? " without a result event" : "";
108
- return fail(`${failure}${missingResult}${tail === "" ? "" : `: ${tail}`}`);
109
- }
110
- if (!closed || !cleanupConfirmed || terminal === void 0 || process.platform !== "win32" && signal !== "SIGKILL") return fail("cursor-agent supervisor ended without confirmed cleanup");
111
- if (terminal.code !== 0 || terminal.signal !== null || state.result === void 0) {
112
- const tail = stderr.join("").trim().slice(-2e3);
113
- return fail(`cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}${state.result === void 0 ? " without a result event" : ""}${tail === "" ? "" : `: ${tail}`}`);
114
- }
115
- if (state.isError) return fail(state.result || "cursor-agent reported an error result");
116
- resolve({
117
- output: state.result,
118
- tokenUsage: state.tokenUsage,
119
- metadata
120
- });
121
- };
122
- const boundCleanup = () => {
123
- cleanupTimer ??= setTimeout(() => {
124
- failure = failure === void 0 ? "cursor-agent cleanup or pipe draining timed out" : `${failure}; cleanup or pipe draining timed out`;
125
- finish(false);
126
- }, 2e3);
127
- };
128
- const stop = (error) => {
129
- if (settled) return;
130
- failure ??= error;
131
- boundCleanup();
132
- if (supervisor.connected) supervisor.send("stop", () => {});
133
- };
134
- const timer = setTimeout(() => stop(`cursor-agent timed out after ${timeoutMs}ms`), timeoutMs);
135
- supervisor.on("error", (err) => stop(`cursor-agent supervisor failed: ${err.message}`));
136
- supervisor.on("disconnect", () => {
137
- if (!cleanupConfirmed && !settled) stop("cursor-agent supervisor disconnected before cleanup");
138
- });
139
- supervisor.on("message", (message) => {
140
- if (settled) return;
141
- if (typeof message !== "object" || message === null) return;
142
- if (!("type" in message)) return;
143
- if (message.type === "terminal" && "code" in message && "signal" in message && (message.code === null || typeof message.code === "number") && (message.signal === null || typeof message.signal === "string")) {
144
- terminal = {
145
- code: message.code,
146
- signal: message.signal
147
- };
148
- clearTimeout(timer);
149
- if (terminal.code !== 0 || terminal.signal !== null) failure ??= `cursor-agent exited ${terminal.signal ?? terminal.code ?? "without status"}`;
150
- boundCleanup();
151
- supervisor.send("cleanup", (err) => {
152
- if (err) stop(`cursor-agent cleanup request failed: ${err.message}`);
153
- });
154
- } else if (message.type === "failure" && "error" in message && typeof message.error === "string") stop(message.error);
155
- else if (message.type === "cleanup" && terminal !== void 0) cleanupConfirmed = true;
156
- });
157
- supervisor.stdout?.on("data", (chunk) => {
158
- stdoutBuf += chunk.toString("utf8");
159
- const lines = stdoutBuf.split("\n");
160
- stdoutBuf = lines.pop() ?? "";
161
- for (const line of lines) foldLine(state, line);
162
- });
163
- supervisor.stderr?.on("data", (chunk) => stderr.push(chunk.toString("utf8")));
164
- supervisor.stdout?.on("error", (err) => stop(`cursor-agent stdout failed: ${err.message}`));
165
- supervisor.stderr?.on("error", (err) => stop(`cursor-agent stderr failed: ${err.message}`));
166
- supervisor.stdin?.on("error", (err) => stop(`cursor-agent stdin failed: ${err.message}`));
167
- supervisor.on("close", (_code, signal) => finish(true, signal));
168
- supervisor.send({
169
- command,
170
- args,
171
- cwd: this.config.working_dir,
172
- timeoutMs
173
- }, (err) => {
174
- if (err) stop(`cursor-agent supervisor initialization failed: ${err.message}`);
175
- });
176
- supervisor.stdin?.end(prompt);
177
- });
178
- }
179
- };
180
- //#endregion
181
- export { CursorAgentProvider as default, foldLine, newStreamState };