@bugmole/cli 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. package/.bugmole.env.example +20 -0
  2. package/LICENSE +7 -0
  3. package/README.md +293 -0
  4. package/TESTING.md +117 -0
  5. package/bugmole.config.yaml +39 -0
  6. package/package.json +85 -0
  7. package/scripts/billing/paypal-setup.mjs +121 -0
  8. package/scripts/bugmole-continue.ts +318 -0
  9. package/scripts/bugmole-init.ts +188 -0
  10. package/scripts/bugmole.cjs +17 -0
  11. package/scripts/bugmole.test.ts +344 -0
  12. package/scripts/bugmole.ts +657 -0
  13. package/scripts/ensure-maestro.cjs +79 -0
  14. package/scripts/ios-tunnel-keeper.sh +45 -0
  15. package/scripts/ios-wda-keeper.sh +66 -0
  16. package/scripts/sync-plan-catalog.d.mts +3 -0
  17. package/scripts/sync-plan-catalog.mjs +16 -0
  18. package/scripts/ui-parity-diff.py +65 -0
  19. package/scripts/ui-parity-requirements.txt +1 -0
  20. package/scripts/verify-manage-to-plans.mts +194 -0
  21. package/spec/app-ui-audit.schema.json +176 -0
  22. package/spec/blockers.yaml +79 -0
  23. package/spec/bugs.index.json +42 -0
  24. package/spec/design-dna.schema.json +38 -0
  25. package/spec/domain_rules.yaml +24 -0
  26. package/spec/flows/flow_manage-to-plans-64c3c61b.yaml +8 -0
  27. package/spec/flows/flow_screen-to-evidence-6c3ab309.yaml +7 -0
  28. package/spec/journey_graph.yaml +126 -0
  29. package/spec/journeys.graph.json +2618 -0
  30. package/spec/plans/dashboard-smoke.flow.yaml +20 -0
  31. package/spec/plans/example.flow.yaml +99 -0
  32. package/spec/plans/flow_api-keys-to-logout-7396cd5d.flow.yaml +33 -0
  33. package/spec/plans/flow_api-keys-to-screen-157e10a6.flow.yaml +33 -0
  34. package/spec/plans/flow_manage-to-audits-176c1eb8.flow.yaml +33 -0
  35. package/spec/plans/flow_manage-to-blockers-633282c8.flow.yaml +112 -0
  36. package/spec/plans/flow_manage-to-devices-ad57e202.flow.yaml +33 -0
  37. package/spec/plans/flow_manage-to-logout-86c54656.flow.yaml +33 -0
  38. package/spec/plans/flow_manage-to-manage-resource-action-6325f99a.flow.yaml +185 -0
  39. package/spec/plans/flow_manage-to-parity-issues-01fa2579.flow.yaml +34 -0
  40. package/spec/plans/flow_manage-to-plans-64c3c61b.flow.yaml +123 -0
  41. package/spec/plans/flow_manage-to-screen-6857f915.flow.yaml +106 -0
  42. package/spec/plans/flow_manage-to-tasks-c5791c95.flow.yaml +33 -0
  43. package/spec/plans/flow_profile-to-screen-6045aa3d.flow.yaml +33 -0
  44. package/spec/plans/flow_register-to-api-auth-login-460a9cc8.flow.yaml +34 -0
  45. package/spec/plans/flow_screen-to-audits-07666356.flow.yaml +33 -0
  46. package/spec/plans/flow_screen-to-blockers-091d258b.flow.yaml +33 -0
  47. package/spec/plans/flow_screen-to-devices-cad72f75.flow.yaml +33 -0
  48. package/spec/plans/flow_screen-to-evidence-6c3ab309.flow.yaml +126 -0
  49. package/spec/plans/flow_screen-to-parity-issues-2218e3ad.flow.yaml +34 -0
  50. package/spec/plans/flow_screen-to-plans-8232a94f.flow.yaml +33 -0
  51. package/spec/plans/flow_screen-to-tasks-540671d1.flow.yaml +33 -0
  52. package/spec/plans/login.flow.yaml +22 -0
  53. package/spec/plans/owner-operations.flow.yaml +20 -0
  54. package/spec/project_config.yaml +55 -0
  55. package/spec/roles.yaml +30 -0
  56. package/spec/schema.md +394 -0
  57. package/spec/test-case-results.schema.json +62 -0
  58. package/spec/test-cases.schema.json +85 -0
  59. package/spec/ui-parity-audit.schema.json +194 -0
  60. package/spec/ui-reverse-engineering.schema.json +94 -0
  61. package/src/billing/plan-catalog.test.ts +46 -0
  62. package/src/billing/plan-catalog.ts +199 -0
  63. package/src/integrations/aws-sigv4.test.ts +42 -0
  64. package/src/integrations/aws-sigv4.ts +72 -0
  65. package/src/integrations/device-farm.ts +155 -0
  66. package/src/integrations/github-app.test.ts +57 -0
  67. package/src/integrations/github-app.ts +143 -0
  68. package/src/integrations/gitlab.ts +81 -0
  69. package/src/integrations/temp-email.test.ts +123 -0
  70. package/src/integrations/temp-email.ts +175 -0
  71. package/src/integrations/testflight-feedback.test.ts +51 -0
  72. package/src/integrations/testflight-feedback.ts +173 -0
  73. package/src/integrations/webdriver-client.ts +131 -0
  74. package/src/mcp/server.test.ts +1220 -0
  75. package/src/mcp/server.ts +3064 -0
  76. package/src/mcp/write-test-cases.test.ts +287 -0
  77. package/src/registry/api-key-client.ts +39 -0
  78. package/src/registry/control-plane-client.ts +212 -0
  79. package/src/registry/migrations/0001_registry.sql +47 -0
  80. package/src/registry/migrations/0002_device_authorizations.sql +23 -0
  81. package/src/registry/migrations/0003_project_environments.sql +25 -0
  82. package/src/registry/migrations/0004_device_authorization_email.sql +1 -0
  83. package/src/registry/migrations/0005_testing_control_plane.sql +55 -0
  84. package/src/registry/migrations/0006_workspaces.sql +36 -0
  85. package/src/registry/migrations/0007_project_apps.sql +26 -0
  86. package/src/registry/migrations/0008_agent_tasks.sql +30 -0
  87. package/src/registry/migrations/0009_journey_revisions.sql +17 -0
  88. package/src/registry/migrations/0010_agent_task_journey.sql +2 -0
  89. package/src/registry/migrations/0011_run_journey_revision.sql +1 -0
  90. package/src/registry/migrations/0012_canonical_flow_execution.sql +10 -0
  91. package/src/registry/migrations/0013_device_sessions.sql +22 -0
  92. package/src/registry/migrations/0014_agent_task_step.sql +1 -0
  93. package/src/registry/migrations/0015_agent_task_attempts.sql +6 -0
  94. package/src/registry/migrations/0016_github_issue_tracker.sql +54 -0
  95. package/src/registry/migrations/0017_run_targets.sql +7 -0
  96. package/src/registry/migrations/0018_run_fix_from.sql +4 -0
  97. package/src/registry/migrations/0019_workspace_flags.sql +9 -0
  98. package/src/registry/migrations/0020_orgs.sql +40 -0
  99. package/src/registry/migrations/0021_billing_core.sql +58 -0
  100. package/src/registry/migrations/0022_cloud_runners.sql +19 -0
  101. package/src/registry/migrations/0023_signup.sql +4 -0
  102. package/src/registry/migrations/0024_billing.sql +67 -0
  103. package/src/registry/migrations/0025_notifications.sql +47 -0
  104. package/src/registry/migrations/0026_repo_bindings.sql +28 -0
  105. package/src/registry/migrations/0027_feedback.sql +29 -0
  106. package/src/registry/migrations/0028_devices.sql +48 -0
  107. package/src/registry/migrations/0029_sso.sql +31 -0
  108. package/src/registry/migrations/0030_workspace_domains.sql +18 -0
  109. package/src/registry/migrations/0031_gitlab_and_teams.sql +45 -0
  110. package/src/registry/migrations/0032_personas.sql +15 -0
  111. package/src/registry/migrations/0033_bugmole_rename.sql +11 -0
  112. package/src/registry/task-scheduling.test.ts +100 -0
  113. package/src/registry/task-scheduling.ts +80 -0
  114. package/src/registry-worker/ai/platform-model.ts +77 -0
  115. package/src/registry-worker/ai/routes.test.ts +88 -0
  116. package/src/registry-worker/ai/routes.ts +90 -0
  117. package/src/registry-worker/artifacts.test.ts +98 -0
  118. package/src/registry-worker/artifacts.ts +85 -0
  119. package/src/registry-worker/billing/billing-core.test.ts +175 -0
  120. package/src/registry-worker/billing/checkout-routes.ts +209 -0
  121. package/src/registry-worker/billing/enforcement.ts +69 -0
  122. package/src/registry-worker/billing/entitlements.ts +108 -0
  123. package/src/registry-worker/billing/ledger.ts +186 -0
  124. package/src/registry-worker/billing/paypal/api.ts +259 -0
  125. package/src/registry-worker/billing/paypal/client.ts +91 -0
  126. package/src/registry-worker/billing/paypal/provider.ts +143 -0
  127. package/src/registry-worker/billing/paypal.test.ts +466 -0
  128. package/src/registry-worker/billing/provider.ts +114 -0
  129. package/src/registry-worker/billing/routes.ts +66 -0
  130. package/src/registry-worker/billing/subscriptions.ts +780 -0
  131. package/src/registry-worker/billing/thresholds.ts +107 -0
  132. package/src/registry-worker/core.ts +308 -0
  133. package/src/registry-worker/devices/devices.test.ts +185 -0
  134. package/src/registry-worker/devices/policy.ts +71 -0
  135. package/src/registry-worker/devices/routes.ts +453 -0
  136. package/src/registry-worker/domains/domains.test.ts +210 -0
  137. package/src/registry-worker/domains/routes.ts +139 -0
  138. package/src/registry-worker/domains.ts +88 -0
  139. package/src/registry-worker/email/sender.ts +75 -0
  140. package/src/registry-worker/env.d.ts +14716 -0
  141. package/src/registry-worker/features.ts +20 -0
  142. package/src/registry-worker/feedback/feedback.test.ts +230 -0
  143. package/src/registry-worker/feedback/format.ts +148 -0
  144. package/src/registry-worker/feedback/routes.ts +386 -0
  145. package/src/registry-worker/flags.ts +39 -0
  146. package/src/registry-worker/github/checks.test.ts +177 -0
  147. package/src/registry-worker/github/checks.ts +374 -0
  148. package/src/registry-worker/gitlab/checks.test.ts +141 -0
  149. package/src/registry-worker/gitlab/checks.ts +349 -0
  150. package/src/registry-worker/hooks.ts +54 -0
  151. package/src/registry-worker/index.ts +2077 -0
  152. package/src/registry-worker/jobs/index.ts +29 -0
  153. package/src/registry-worker/jobs/retention.ts +68 -0
  154. package/src/registry-worker/mcp/mcp.test.ts +355 -0
  155. package/src/registry-worker/mcp/routes.ts +215 -0
  156. package/src/registry-worker/mcp/token.ts +126 -0
  157. package/src/registry-worker/mcp/tools.ts +563 -0
  158. package/src/registry-worker/notifications/alerts.ts +212 -0
  159. package/src/registry-worker/notifications/notifications.test.ts +298 -0
  160. package/src/registry-worker/notifications/outbox.ts +83 -0
  161. package/src/registry-worker/notifications/routes.ts +280 -0
  162. package/src/registry-worker/notifications/secrets.ts +49 -0
  163. package/src/registry-worker/notifications/slack.ts +96 -0
  164. package/src/registry-worker/notifications/teams.ts +46 -0
  165. package/src/registry-worker/org/audit.ts +116 -0
  166. package/src/registry-worker/org/routes.test.ts +163 -0
  167. package/src/registry-worker/org/routes.ts +302 -0
  168. package/src/registry-worker/personas/personas.test.ts +78 -0
  169. package/src/registry-worker/personas/routes.ts +100 -0
  170. package/src/registry-worker/repo-triggers.ts +20 -0
  171. package/src/registry-worker/routes/index.ts +74 -0
  172. package/src/registry-worker/run-events.ts +24 -0
  173. package/src/registry-worker/runner/dispatch.ts +219 -0
  174. package/src/registry-worker/runner/jobs.ts +43 -0
  175. package/src/registry-worker/runner/metering.ts +82 -0
  176. package/src/registry-worker/runner/policy.ts +59 -0
  177. package/src/registry-worker/runner/routes.ts +171 -0
  178. package/src/registry-worker/runner/runner.test.ts +358 -0
  179. package/src/registry-worker/runner/tokens.ts +93 -0
  180. package/src/registry-worker/runs.test.ts +60 -0
  181. package/src/registry-worker/signup/policy.ts +57 -0
  182. package/src/registry-worker/signup/routes.ts +106 -0
  183. package/src/registry-worker/signup/signup.test.ts +81 -0
  184. package/src/registry-worker/sso/aegis.ts +141 -0
  185. package/src/registry-worker/sso/membership.ts +157 -0
  186. package/src/registry-worker/sso/routes.ts +458 -0
  187. package/src/registry-worker/sso/sso.test.ts +344 -0
  188. package/src/registry-worker/testing/d1-shim.ts +180 -0
  189. package/src/registry-worker/testing/harness.ts +137 -0
  190. package/src/runner-worker/index.ts +108 -0
  191. package/src/runtime/ai-analysis.ts +97 -0
  192. package/src/runtime/ai-exploration.test.ts +32 -0
  193. package/src/runtime/ai-exploration.ts +69 -0
  194. package/src/runtime/ai-repair.ts +74 -0
  195. package/src/runtime/ai-work.test.ts +99 -0
  196. package/src/runtime/android-screen-record.test.ts +75 -0
  197. package/src/runtime/android-screen-record.ts +192 -0
  198. package/src/runtime/app-understanding.test.ts +123 -0
  199. package/src/runtime/app-understanding.ts +201 -0
  200. package/src/runtime/appium-driver.test.ts +179 -0
  201. package/src/runtime/appium-driver.ts +295 -0
  202. package/src/runtime/blocker-resolution.test.ts +113 -0
  203. package/src/runtime/blocker-resolution.ts +111 -0
  204. package/src/runtime/browser-matrix.integration.test.ts +212 -0
  205. package/src/runtime/browser-matrix.test.ts +143 -0
  206. package/src/runtime/browser-matrix.ts +200 -0
  207. package/src/runtime/canonical-flow.test.ts +52 -0
  208. package/src/runtime/config-validate.ts +185 -0
  209. package/src/runtime/continuous-execution.ts +291 -0
  210. package/src/runtime/cursor-applescript.ts +573 -0
  211. package/src/runtime/cursor-cli-driver.test.ts +78 -0
  212. package/src/runtime/cursor-cli-driver.ts +156 -0
  213. package/src/runtime/cursor-driver-example.ts +117 -0
  214. package/src/runtime/cursor-driver-index.ts +65 -0
  215. package/src/runtime/cursor-driver-init.ts +277 -0
  216. package/src/runtime/cursor-driver-run.test.ts +15 -0
  217. package/src/runtime/cursor-driver-run.ts +323 -0
  218. package/src/runtime/cursor-driver.ts +332 -0
  219. package/src/runtime/cursor-llm-example.ts +90 -0
  220. package/src/runtime/cursor-llm.ts +206 -0
  221. package/src/runtime/cursor-mcp-monitor.ts +386 -0
  222. package/src/runtime/device-clouds/browserstack.ts +73 -0
  223. package/src/runtime/device-clouds/device-farm.ts +52 -0
  224. package/src/runtime/device-clouds/index.ts +92 -0
  225. package/src/runtime/device-clouds/kobiton.ts +70 -0
  226. package/src/runtime/device-clouds/targets.ts +44 -0
  227. package/src/runtime/device-clouds/types.ts +62 -0
  228. package/src/runtime/diff-proposal.ts +84 -0
  229. package/src/runtime/discovery-task.test.ts +29 -0
  230. package/src/runtime/discovery-task.ts +284 -0
  231. package/src/runtime/driver-recovery.ts +69 -0
  232. package/src/runtime/driver.ts +79 -0
  233. package/src/runtime/environment.test.ts +104 -0
  234. package/src/runtime/environment.ts +137 -0
  235. package/src/runtime/executor.test.ts +509 -0
  236. package/src/runtime/executor.ts +921 -0
  237. package/src/runtime/explorer.test.ts +101 -0
  238. package/src/runtime/explorer.ts +1013 -0
  239. package/src/runtime/failure-analysis.test.ts +111 -0
  240. package/src/runtime/failure-analysis.ts +272 -0
  241. package/src/runtime/fixtures/fake-maestro.sh +36 -0
  242. package/src/runtime/flow-language.test.ts +268 -0
  243. package/src/runtime/flow-language.ts +414 -0
  244. package/src/runtime/init-wizard.ts +354 -0
  245. package/src/runtime/ios-screen-record.test.ts +68 -0
  246. package/src/runtime/ios-screen-record.ts +155 -0
  247. package/src/runtime/journey-editor.ts +452 -0
  248. package/src/runtime/journey-evidence.test.ts +161 -0
  249. package/src/runtime/journey-evidence.ts +180 -0
  250. package/src/runtime/journey-graph.test.ts +257 -0
  251. package/src/runtime/journey-graph.ts +170 -0
  252. package/src/runtime/legacy-names.ts +32 -0
  253. package/src/runtime/llm-example.ts +105 -0
  254. package/src/runtime/llm.ts +527 -0
  255. package/src/runtime/local-browser.test.ts +45 -0
  256. package/src/runtime/local-browser.ts +48 -0
  257. package/src/runtime/local-registry-stub.test.ts +325 -0
  258. package/src/runtime/local-registry-stub.ts +803 -0
  259. package/src/runtime/maestro-driver.test.ts +84 -0
  260. package/src/runtime/maestro-driver.ts +209 -0
  261. package/src/runtime/mole-voice.ts +21 -0
  262. package/src/runtime/nav-crawl.test.ts +100 -0
  263. package/src/runtime/nav-crawl.ts +153 -0
  264. package/src/runtime/pipeline.test.ts +405 -0
  265. package/src/runtime/pipeline.ts +833 -0
  266. package/src/runtime/planner.test.ts +37 -0
  267. package/src/runtime/planner.ts +274 -0
  268. package/src/runtime/platform-ai.ts +76 -0
  269. package/src/runtime/playwright-driver.test.ts +93 -0
  270. package/src/runtime/playwright-driver.ts +620 -0
  271. package/src/runtime/project-spec.ts +140 -0
  272. package/src/runtime/record-run-verdicts.ts +68 -0
  273. package/src/runtime/reporter.test.ts +56 -0
  274. package/src/runtime/reporter.ts +158 -0
  275. package/src/runtime/reset.test.ts +44 -0
  276. package/src/runtime/reset.ts +61 -0
  277. package/src/runtime/reviewer.test.ts +73 -0
  278. package/src/runtime/reviewer.ts +158 -0
  279. package/src/runtime/run-job.ts +136 -0
  280. package/src/runtime/run-once.test.ts +207 -0
  281. package/src/runtime/run-once.ts +168 -0
  282. package/src/runtime/run.ts +132 -0
  283. package/src/runtime/screen-recording.ts +34 -0
  284. package/src/runtime/serve-gateway.test.ts +74 -0
  285. package/src/runtime/serve-gateway.ts +164 -0
  286. package/src/runtime/serve-worker.test.ts +23 -0
  287. package/src/runtime/serve-worker.ts +278 -0
  288. package/src/runtime/site-discovery.test.ts +168 -0
  289. package/src/runtime/site-discovery.ts +308 -0
  290. package/src/runtime/target-runner.ts +144 -0
  291. package/src/runtime/test-case-verdicts.test.ts +94 -0
  292. package/src/runtime/test-case-verdicts.ts +120 -0
  293. package/src/runtime/ui-reverse-engineering/coordinator.test.ts +59 -0
  294. package/src/runtime/ui-reverse-engineering/coordinator.ts +391 -0
  295. package/src/runtime/ui-reverse-engineering/types.ts +197 -0
  296. package/src/runtime/web-suite.test.ts +97 -0
  297. package/src/runtime/web-suite.ts +176 -0
  298. package/src/storage/create-object-store.ts +144 -0
  299. package/src/storage/keys.ts +34 -0
  300. package/src/storage/local-artifact-server.test.ts +314 -0
  301. package/src/storage/local-artifact-server.ts +357 -0
  302. package/src/storage/object-store.test.ts +28 -0
  303. package/src/storage/object-store.ts +101 -0
  304. package/src/storage/registry-object-store.ts +88 -0
  305. package/src/storage/remote-object-store.ts +104 -0
  306. package/src/storage/storage-directory.test.ts +43 -0
  307. package/src/storage/storage-directory.ts +24 -0
  308. package/tsconfig.json +24 -0
@@ -0,0 +1,156 @@
1
+ import { execFile, type ExecFileException } from "node:child_process";
2
+ import path from "node:path";
3
+ import { promisify } from "node:util";
4
+ import {
5
+ generateCursorDriverPrompt,
6
+ validateCommitPayload,
7
+ type CursorDriverOptions,
8
+ type CursorDriverResponse,
9
+ } from "./cursor-driver.js";
10
+
11
+ const execFileAsync = promisify(execFile);
12
+
13
+ /**
14
+ * Long-running stages (a full source exploration with screenshot capture)
15
+ * routinely outlive a short timeout: the agent gets SIGTERM'd mid-flight and
16
+ * the pipeline stalls with "cursor-agent run timed out" even though it was
17
+ * making progress. Investigation work therefore gets a much longer budget
18
+ * than a small code edit.
19
+ */
20
+ export const DEFAULT_CURSOR_TIMEOUT_MS = 300_000;
21
+ export const INVESTIGATION_CURSOR_TIMEOUT_MS = 1_800_000;
22
+
23
+ export function resolveCursorTimeoutMs(
24
+ taskType?: string,
25
+ override?: number,
26
+ env: NodeJS.ProcessEnv = process.env,
27
+ ): number {
28
+ if (Number.isFinite(override) && (override as number) > 0) return override as number;
29
+ const configured = Number(env.BUGMOLE_CURSOR_TIMEOUT_MS);
30
+ if (Number.isFinite(configured) && configured > 0) return configured;
31
+ return taskType === "investigation" ? INVESTIGATION_CURSOR_TIMEOUT_MS : DEFAULT_CURSOR_TIMEOUT_MS;
32
+ }
33
+
34
+ export type CursorCliDriverOptions = CursorDriverOptions & {
35
+ /** @default "cursor-agent" */
36
+ command?: string;
37
+ /** @default 300000, or 1800000 for investigation tasks */
38
+ timeoutMs?: number;
39
+ /** Drives the default timeout when timeoutMs is not given. */
40
+ taskType?: string;
41
+ model?: string;
42
+ /** Workspace directory the agent should operate in (defaults to cwd). */
43
+ workspaceDirectory?: string;
44
+ };
45
+
46
+ /**
47
+ * cursor-agent resolves `--workspace` against its own working directory, and
48
+ * this driver already spawns it with that same directory as `cwd`. A relative
49
+ * value was therefore applied twice: `apps/qa` from a repo-root run became
50
+ * `<repo>/apps/qa/apps/qa`, and cursor-agent exited immediately with
51
+ * "Workspace directory does not exist".
52
+ *
53
+ * That surfaced as `cursor-agent run timed out.`, so it read like a slow agent
54
+ * rather than a bad path — and every task routed through this driver failed
55
+ * the same way, which is why no test case ever reached a verdict.
56
+ *
57
+ * Resolving once keeps the two uses in agreement, and is a no-op for a path
58
+ * that is already absolute.
59
+ */
60
+ export function resolveWorkspaceDirectory(workspaceDirectory?: string): string | undefined {
61
+ return workspaceDirectory ? path.resolve(workspaceDirectory) : undefined;
62
+ }
63
+
64
+ export function buildCursorCliArgs(prompt: string, options: CursorCliDriverOptions = {}): string[] {
65
+ const args = ["-p", prompt, "--output-format", "json", "--force"];
66
+ if (options.model) args.push("--model", options.model);
67
+ const workspace = resolveWorkspaceDirectory(options.workspaceDirectory);
68
+ if (workspace) args.push("--workspace", workspace);
69
+ return args;
70
+ }
71
+
72
+ /**
73
+ * Parses `cursor-agent -p --output-format json` stdout into a commit-style
74
+ * payload, reusing the same {status, summary, ...} shape the AppleScript
75
+ * path's `bugmole_commit_response` expects, so downstream consumers do not
76
+ * need to know which delivery mechanism produced a response.
77
+ *
78
+ * The exact `--output-format json` envelope was not exercised against a live
79
+ * `cursor-agent` run while writing this (doing so would mean actually
80
+ * spawning an autonomous coding-agent session as a side effect of writing
81
+ * this file, which needs its own explicit go-ahead). This tries the
82
+ * reasonable shapes generously and always falls back to treating raw stdout
83
+ * as the summary rather than throwing, so an unexpected envelope degrades to
84
+ * "unstructured but present" instead of a hard failure — confirm the real
85
+ * shape against one real run before relying on this in production.
86
+ */
87
+ export function parseCursorCliOutput(stdout: string): CursorDriverResponse["payload"] | undefined {
88
+ const trimmed = stdout.trim();
89
+ if (!trimmed) return undefined;
90
+ try {
91
+ const parsed = JSON.parse(trimmed);
92
+ if (parsed && typeof parsed === "object") {
93
+ const { valid } = validateCommitPayload(parsed);
94
+ if (valid) return parsed;
95
+ const text =
96
+ typeof parsed.result === "string" ? parsed.result
97
+ : typeof parsed.text === "string" ? parsed.text
98
+ : typeof parsed.message === "string" ? parsed.message
99
+ : typeof parsed.response === "string" ? parsed.response
100
+ : undefined;
101
+ if (text) return { status: "ok", summary: text.slice(0, 400), answer_markdown: text };
102
+ }
103
+ } catch {
104
+ // Not JSON; fall through to the raw-text fallback below.
105
+ }
106
+ return { status: "ok", summary: trimmed.slice(0, 400), answer_markdown: trimmed };
107
+ }
108
+
109
+ /**
110
+ * Runs a prompt through the real, headless `cursor-agent` CLI instead of
111
+ * AppleScript-injecting it into the Cursor GUI. `-p` mode blocks until the
112
+ * agent finishes and prints its final response, so there is no polling or
113
+ * "nudge" loop here the way the AppleScript path needs — the agent still has
114
+ * MCP access (via `cursor-agent mcp`) and can call the same `bugmole_*`
115
+ * tools; this driver only replaces *delivery* of the prompt and *receipt* of
116
+ * the final answer.
117
+ */
118
+ export async function runCursorCliDriver(
119
+ task: string,
120
+ options: CursorCliDriverOptions,
121
+ ): Promise<CursorDriverResponse> {
122
+ const command = options.command?.trim() || "cursor-agent";
123
+ const prompt = generateCursorDriverPrompt(task, options);
124
+ const args = buildCursorCliArgs(prompt, options);
125
+ try {
126
+ const { stdout } = await execFileAsync(command, args, {
127
+ // Same resolved path the --workspace arg carries; see
128
+ // resolveWorkspaceDirectory for why these must not disagree.
129
+ cwd: resolveWorkspaceDirectory(options.workspaceDirectory),
130
+ timeout: resolveCursorTimeoutMs(options.taskType, options.timeoutMs),
131
+ maxBuffer: 16 * 1024 * 1024,
132
+ });
133
+ const payload = parseCursorCliOutput(stdout);
134
+ if (!payload) {
135
+ return { success: false, error: "cursor-agent produced no output.", errorCode: "TIMEOUT" };
136
+ }
137
+ const validation = validateCommitPayload(payload);
138
+ if (!validation.valid) {
139
+ return {
140
+ success: false,
141
+ error: `Payload validation failed: ${validation.errors.join(", ")}`,
142
+ errorCode: "VALIDATION_FAILED",
143
+ };
144
+ }
145
+ return { success: true, payload };
146
+ } catch (error) {
147
+ const failure = error as ExecFileException;
148
+ if (failure.code === "ENOENT") {
149
+ return { success: false, error: `cursor-agent executable was not found: ${command}`, errorCode: "MCP_UNAVAILABLE" };
150
+ }
151
+ if (failure.killed || failure.signal === "SIGTERM") {
152
+ return { success: false, error: "cursor-agent run timed out.", errorCode: "TIMEOUT" };
153
+ }
154
+ return { success: false, error: failure.message || "cursor-agent run failed." };
155
+ }
156
+ }
@@ -0,0 +1,117 @@
1
+ /**
2
+ * Example usage of CURSOR_DRIVER as an LLM API adapter
3
+ *
4
+ * This demonstrates how to use CURSOR_DRIVER in place of traditional LLM calls.
5
+ */
6
+
7
+ import { runCursorDriver, cursorComplete, cursorCompleteStructured } from "./cursor-driver-run.js";
8
+
9
+ /**
10
+ * Example 1: Simple text completion (like OpenAI API)
11
+ */
12
+ export async function exampleSimpleCompletion(cfg: any) {
13
+ const task = "Analyze the codebase and suggest improvements to the error handling patterns.";
14
+
15
+ const result = await cursorComplete(task, {
16
+ artifactsDir: cfg.runtime.artifacts_dir || "./artifacts",
17
+ mcpHost: cfg.mcp?.host || "127.0.0.1",
18
+ mcpPort: cfg.mcp?.port || 3187,
19
+ });
20
+
21
+ console.log("Completion result:", result);
22
+ return result;
23
+ }
24
+
25
+ /**
26
+ * Example 2: Structured completion with full payload
27
+ */
28
+ export async function exampleStructuredCompletion(cfg: any) {
29
+ const task = `
30
+ Review the test failures in the execution results and propose fixes.
31
+ Focus on selector mismatches and timing issues.
32
+ `;
33
+
34
+ const result = await cursorCompleteStructured(task, {
35
+ artifactsDir: cfg.runtime.artifacts_dir || "./artifacts",
36
+ mcpHost: cfg.mcp?.host || "127.0.0.1",
37
+ mcpPort: cfg.mcp?.port || 3187,
38
+ requestId: `review_${Date.now()}`,
39
+ customInstructions: "Pay special attention to Playwright selectors and wait conditions.",
40
+ });
41
+
42
+ if (result.success && result.payload) {
43
+ console.log("Status:", result.payload.status);
44
+ console.log("Summary:", result.payload.summary);
45
+ console.log("Actions:", result.payload.actions);
46
+ console.log("Artifacts:", result.payload.artifacts);
47
+ } else {
48
+ console.error("Error:", result.error);
49
+ }
50
+
51
+ return result;
52
+ }
53
+
54
+ /**
55
+ * Example 3: Using in a diff proposal workflow
56
+ */
57
+ export async function exampleDiffProposal(cfg: any, evidence: any, failures: any[]) {
58
+ const task = `
59
+ Based on the execution evidence and failures, propose minimal changes to the spec file.
60
+
61
+ Evidence: ${JSON.stringify(evidence, null, 2)}
62
+ Failures: ${JSON.stringify(failures, null, 2)}
63
+
64
+ Generate a diff that addresses these issues while maintaining backward compatibility.
65
+ `;
66
+
67
+ const result = await runCursorDriver(task, {
68
+ artifactsDir: cfg.runtime.artifacts_dir || "./artifacts",
69
+ mcpHost: cfg.mcp?.host || "127.0.0.1",
70
+ mcpPort: cfg.mcp?.port || 3187,
71
+ timeout: 600000, // 10 minutes for complex tasks
72
+ requestId: `diff_proposal_${Date.now()}`,
73
+ });
74
+
75
+ return result;
76
+ }
77
+
78
+ /**
79
+ * Example 4: Error handling
80
+ */
81
+ export async function exampleWithErrorHandling(cfg: any, task: string) {
82
+ try {
83
+ const result = await runCursorDriver(task, {
84
+ artifactsDir: cfg.runtime.artifacts_dir || "./artifacts",
85
+ mcpHost: cfg.mcp?.host || "127.0.0.1",
86
+ mcpPort: cfg.mcp?.port || 3187,
87
+ timeout: 300000,
88
+ });
89
+
90
+ if (!result.success) {
91
+ switch (result.errorCode) {
92
+ case "TIMEOUT":
93
+ console.error("Task timed out. Consider increasing timeout or simplifying the task.");
94
+ break;
95
+ case "MCP_UNAVAILABLE":
96
+ console.error("MCP server is not available. Check if the server is running.");
97
+ break;
98
+ case "INJECTION_FAILED":
99
+ console.error(
100
+ "Failed to inject prompt into Cursor. Check if Cursor is running and accessible.",
101
+ );
102
+ break;
103
+ case "VALIDATION_FAILED":
104
+ console.error("Response validation failed:", result.error);
105
+ break;
106
+ default:
107
+ console.error("Unknown error:", result.error);
108
+ }
109
+ return null;
110
+ }
111
+
112
+ return result.payload;
113
+ } catch (error: any) {
114
+ console.error("Unexpected error:", error);
115
+ return null;
116
+ }
117
+ }
@@ -0,0 +1,65 @@
1
+ /**
2
+ * CURSOR_DRIVER - Main exports
3
+ *
4
+ * This module provides a complete LLM API adapter using Cursor as the reasoning engine.
5
+ * Use this instead of traditional LLM API calls for tasks that benefit from Cursor's
6
+ * codebase context and tool access.
7
+ */
8
+
9
+ // Continuous execution (automatically follows next_steps)
10
+ export {
11
+ executeContinuously,
12
+ type ContinuousExecutionOptions,
13
+ type ContinuousExecutionResult,
14
+ } from "./continuous-execution.js";
15
+
16
+ // Unified LLM interface (works with OpenAI, Anthropic, Cursor, Ollama)
17
+ export {
18
+ llm,
19
+ llmStructured,
20
+ resetLLMConfigCache,
21
+ type LLMProvider,
22
+ type LLMOptions,
23
+ type LLMResponse,
24
+ } from "./llm.js";
25
+
26
+ // Simple LLM-like interface (recommended for Cursor use cases)
27
+ export {
28
+ cursorLLM,
29
+ cursorLLMStructured,
30
+ resetConfigCache,
31
+ type CursorLLMOptions,
32
+ } from "./cursor-llm.js";
33
+
34
+ // Main driver function (advanced use)
35
+ export { runCursorDriver, cursorComplete, cursorCompleteStructured } from "./cursor-driver-run.js";
36
+
37
+ // Prompt generation utilities
38
+ export {
39
+ generateFinalizationProtocol,
40
+ appendFinalizationProtocol,
41
+ generateCursorDriverPrompt,
42
+ validateCommitPayload,
43
+ type CursorDriverOptions,
44
+ } from "./cursor-driver.js";
45
+
46
+ // AppleScript utilities (for advanced use cases)
47
+ export {
48
+ injectPrompt,
49
+ activateCursor,
50
+ ensureChatPanelOpen,
51
+ createNewChat,
52
+ isCursorRunning,
53
+ type CursorInjectionOptions,
54
+ } from "./cursor-applescript.js";
55
+
56
+ // MCP monitoring utilities (for advanced use cases)
57
+ export {
58
+ waitForCommitResponse,
59
+ getLatestCommit,
60
+ type CommitResponse,
61
+ type MonitorOptions,
62
+ } from "./cursor-mcp-monitor.js";
63
+
64
+ // Response types
65
+ export type { CursorDriverResponse, CursorDriverRunOptions } from "./cursor-driver.js";
@@ -0,0 +1,277 @@
1
+ /**
2
+ * CURSOR_DRIVER initialization task generator
3
+ *
4
+ * Generates a comprehensive task prompt that instructs Cursor to:
5
+ * 1. Explore the application using bugmole_explore MCP tool
6
+ * 2. Read existing spec files using bugmole_read_spec
7
+ * 3. Create/update all necessary spec files using bugmole_write_spec
8
+ * 4. Follow templates and schemas from the docs
9
+ */
10
+
11
+ import fs from "node:fs";
12
+ import path from "node:path";
13
+
14
+ export interface InitOptions {
15
+ /**
16
+ * Base URL of the application to explore
17
+ */
18
+ baseUrl?: string;
19
+
20
+ /**
21
+ * Maximum exploration depth
22
+ * @default 5
23
+ */
24
+ maxDepth?: number;
25
+
26
+ /**
27
+ * Project directory path (for context)
28
+ */
29
+ projectDirectory?: string;
30
+
31
+ /**
32
+ * Whether to force re-initialization even if files exist
33
+ * @default false
34
+ */
35
+ force?: boolean;
36
+ }
37
+
38
+ /**
39
+ * Generates the initialization task prompt for Cursor
40
+ */
41
+ export function generateInitTask(options: InitOptions = {}): string {
42
+ const { baseUrl, maxDepth = 5, projectDirectory, force = false } = options;
43
+
44
+ let task = `# Bugmole MCP Initialization Task
45
+
46
+ You are tasked with initializing the Bugmole MCP framework for this project. This involves exploring the application, understanding its structure, and creating all necessary spec files.
47
+
48
+ ## Your Mission
49
+
50
+ Use the available Bugmole MCP tools to:
51
+ 1. Explore the application to discover user journeys
52
+ 2. Read existing spec files (if any) to understand current state
53
+ 3. Create or update all required spec files following the framework's schemas and templates
54
+
55
+ ## Available MCP Tools
56
+
57
+ You have access to these Bugmole MCP tools:
58
+ - \`bugmole_explore\` - Explore the application and discover user journeys
59
+ - \`bugmole_journey_create\`, \`bugmole_journey_add_step\`, \`bugmole_journey_add_transition\` - Build flows incrementally
60
+ - \`bugmole_journey_update\`, \`bugmole_journey_remove\` - Edit flows and their steps
61
+ - \`bugmole_journey_next\` - Check the flow chart for the next exploration task
62
+ - \`bugmole_read_spec\` - Read existing spec files
63
+ - \`bugmole_write_spec\` - Write or update spec files
64
+ - \`bugmole_plan\` - Generate journey plans (optional, for later)
65
+
66
+ ## Step-by-Step Instructions
67
+
68
+ ### Step 1: Explore the Application
69
+
70
+ First, explore the application to discover screens and user journeys:
71
+
72
+ 1. Call \`bugmole_explore\` with:
73
+ - \`baseUrl\`: ${
74
+ baseUrl ||
75
+ "The application base URL (check environment variables, .env file, or bugmole.config.yaml for PLAYWRIGHT_BASE_URL)"
76
+ }
77
+ - \`maxDepth\`: ${maxDepth}
78
+
79
+ ${
80
+ baseUrl
81
+ ? ""
82
+ : "**Note**: If the base URL is not available, you can skip exploration and focus on creating spec files based on codebase analysis. However, exploration is recommended for accurate journey discovery."
83
+ }
84
+
85
+ This will discover screens and actions, building the journey graph. It is only the first exploration pass.
86
+
87
+ After every exploration or discovery pass, call \`bugmole_journey_next\` for the selected flow. Perform its returned task, record the result, and repeat until it returns \`exhausted\` or \`blocked\`. Do not call \`bugmole_plan\` while it still returns an actionable task.
88
+
89
+ ### Step 2: Check Existing Spec Files
90
+
91
+ Read existing spec files to understand what's already in place:
92
+
93
+ 1. Check if these files exist using \`bugmole_read_spec\`:
94
+ - \`roles.yaml\` - Role and capability definitions
95
+ - \`domain_rules.yaml\` - Business rules and invariants
96
+ - \`blockers.yaml\` - UI blocker handling rules
97
+ - \`journeys.graph.json\` - Journey graph (created by explorer)
98
+ - \`bugs.index.json\` - Bug backlog structure
99
+
100
+ ### Step 3: Initialize Required Spec Files
101
+
102
+ Create or update each spec file following the schemas. Reference these files for guidance:
103
+ - \`spec/schema.md\` - Complete schema reference
104
+ - \`docs/file-authoring-template.md\` - Template for creating spec files
105
+ - \`spec/roles.yaml\` - Example roles file (if exists, use as reference)
106
+ - \`spec/domain_rules.yaml\` - Example domain rules (if exists, use as reference)
107
+ - \`spec/blockers.yaml\` - Example blockers (if exists, use as reference)
108
+
109
+ #### 3.1: roles.yaml
110
+
111
+ Create \`roles.yaml\` defining:
112
+ - All user roles/actors in the application
113
+ - Capabilities for each role
114
+ - Capability groups for organization
115
+
116
+ **Requirements:**
117
+ - Follow \`spec/schema.md\` section "2) spec/roles.yaml"
118
+ - Include at least 2-3 roles (e.g., user, admin, viewer)
119
+ - Define capabilities that match the application's features
120
+ - Use the existing template structure if available
121
+
122
+ #### 3.2: domain_rules.yaml
123
+
124
+ Create \`domain_rules.yaml\` defining:
125
+ - Business prerequisites
126
+ - Policies (what's allowed/blocked)
127
+ - Invariants (what must always be true)
128
+
129
+ **Requirements:**
130
+ - Follow \`spec/schema.md\` section "3) spec/domain_rules.yaml"
131
+ - Include at least one entity with prerequisites
132
+ - Include at least one policy
133
+ - Include at least one invariant
134
+
135
+ #### 3.3: blockers.yaml
136
+
137
+ Create \`blockers.yaml\` defining:
138
+ - UI interruptions (modals, dialogs, overlays)
139
+ - How to handle each blocker type
140
+ - Hard-stop conditions
141
+
142
+ **Requirements:**
143
+ - Follow \`spec/schema.md\` section "4) spec/blockers.yaml"
144
+ - Include common UI blockers (modals, toasts, loading states)
145
+ - Define resolution strategies
146
+ - Include at least one hard-stop condition
147
+
148
+ #### 3.4: bugs.index.json
149
+
150
+ Create \`bugs.index.json\` for bug backlog structure:
151
+
152
+ **Requirements:**
153
+ - Follow \`spec/schema.md\` section "6) spec/bugs.index.json"
154
+ - Initialize with empty structure if no bugs exist yet
155
+
156
+ ### Step 4: Verify Journey Graph
157
+
158
+ After exploration, verify that \`journeys.graph.json\` was created/updated:
159
+ - Check if it contains discovered screens (nodes)
160
+ - Check if it contains discovered actions (edges)
161
+ - If empty or minimal, you may need to manually enhance it based on application understanding
162
+ - Check the selected flow with \`bugmole_journey_get\` and continue its \`bugmole_journey_next\` loop
163
+
164
+ ## Important Guidelines
165
+
166
+ 1. **Follow Schemas Strictly**: Always reference \`spec/schema.md\` for exact field names and structure
167
+ 2. **Use Templates**: Reference \`docs/file-authoring-template.md\` for file authoring guidance
168
+ 3. **Be Domain-Agnostic**: The framework works for any application - adapt the examples to this project
169
+ 4. **Quality Over Speed**: Create complete, well-structured files rather than minimal placeholders
170
+ 5. **Preserve Existing Content**: If files exist and are well-formed, enhance them rather than replacing
171
+ ${
172
+ force
173
+ ? "6. **Force Mode**: You are instructed to re-initialize even if files exist"
174
+ : "6. **Preserve Mode**: Only create missing files, enhance existing ones"
175
+ }
176
+
177
+ ## Project Context
178
+
179
+ ${projectDirectory ? `- Project directory: ${projectDirectory}` : "- Current working directory"}
180
+ - Application type: Analyze the codebase to determine (web app, API, etc.)
181
+ - Framework: Analyze to determine (React, Next.js, Vue, etc.)
182
+
183
+ ## Expected Outcome
184
+
185
+ After completing this task, the \`spec/\` directory should contain:
186
+ - ✅ \`roles.yaml\` - Complete role definitions
187
+ - ✅ \`domain_rules.yaml\` - Business rules and invariants
188
+ - ✅ \`blockers.yaml\` - UI blocker handling
189
+ - ✅ \`journeys.graph.json\` - Discovered journey graph
190
+ - ✅ \`bugs.index.json\` - Bug backlog structure (initialized)
191
+
192
+ All files should:
193
+ - Follow the schema exactly
194
+ - Be ready for use by the framework
195
+ - Contain realistic, application-specific content (not just examples)
196
+
197
+ ## Important: Next Steps
198
+
199
+ If you cannot complete all steps in a single response, you MUST include a \`next_steps\` field in your commit payload with the remaining steps. For example:
200
+
201
+ \`\`\`json
202
+ {
203
+ "status": "ok",
204
+ "summary": "Completed initial investigation, need to continue with exploration",
205
+ "next_steps": [
206
+ "Start the application using the discovered command",
207
+ "Use bugmole_explore with the base URL to discover journeys",
208
+ "Create test plans for discovered journeys"
209
+ ]
210
+ }
211
+ \`\`\`
212
+
213
+ The framework will automatically continue with these next steps. Only omit \`next_steps\` when the task is fully complete.
214
+
215
+ ## Notes
216
+
217
+ - If the application URL is not available, focus on creating the spec files based on codebase analysis
218
+ - If exploration fails, you can still create spec files based on understanding the codebase structure
219
+ - Use \`bugmole_read_spec\` to verify files after writing them
220
+ - All file paths are relative to \`spec/\` directory
221
+
222
+ Begin by exploring the application, then proceed with file creation.`;
223
+
224
+ return task;
225
+ }
226
+
227
+ /**
228
+ * Reads the project structure to infer application details
229
+ */
230
+ export function inferApplicationDetails(projectDirectory?: string): {
231
+ baseUrl?: string;
232
+ appType?: string;
233
+ framework?: string;
234
+ } {
235
+ const dir = projectDirectory || process.cwd();
236
+ const packageJsonPath = path.join(dir, "package.json");
237
+
238
+ const details: {
239
+ baseUrl?: string;
240
+ appType?: string;
241
+ framework?: string;
242
+ } = {};
243
+
244
+ // Try to read package.json
245
+ if (fs.existsSync(packageJsonPath)) {
246
+ try {
247
+ const pkg = JSON.parse(fs.readFileSync(packageJsonPath, "utf-8"));
248
+
249
+ // Infer framework from dependencies
250
+ if (pkg.dependencies || pkg.devDependencies) {
251
+ const deps = { ...pkg.dependencies, ...pkg.devDependencies };
252
+ if (deps.next) details.framework = "Next.js";
253
+ else if (deps.react) details.framework = "React";
254
+ else if (deps.vue) details.framework = "Vue";
255
+ else if (deps["@angular/core"]) details.framework = "Angular";
256
+ }
257
+ } catch (e) {
258
+ // Ignore parse errors
259
+ }
260
+ }
261
+
262
+ // Try to infer base URL from environment or config
263
+ const envPath = path.join(dir, ".env");
264
+ if (fs.existsSync(envPath)) {
265
+ try {
266
+ const envContent = fs.readFileSync(envPath, "utf-8");
267
+ const urlMatch = envContent.match(/PLAYWRIGHT_BASE_URL=(.+)/);
268
+ if (urlMatch) {
269
+ details.baseUrl = urlMatch[1].trim();
270
+ }
271
+ } catch (e) {
272
+ // Ignore errors
273
+ }
274
+ }
275
+
276
+ return details;
277
+ }
@@ -0,0 +1,15 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { shouldFallBackToCursorGui } from "./cursor-driver-run.js";
4
+
5
+ const missingCli = { success: false, error: "cursor-agent executable was not found: cursor-agent", errorCode: "MCP_UNAVAILABLE" as const };
6
+
7
+ test("falls back to the Cursor GUI only on macOS when cursor-agent is missing", () => {
8
+ assert.equal(shouldFallBackToCursorGui(missingCli, "darwin"), true);
9
+ assert.equal(shouldFallBackToCursorGui(missingCli, "linux"), false);
10
+ });
11
+
12
+ test("never falls back after the CLI ran, whether it succeeded or failed for another reason", () => {
13
+ assert.equal(shouldFallBackToCursorGui({ success: true }, "darwin"), false);
14
+ assert.equal(shouldFallBackToCursorGui({ success: false, error: "cursor-agent run timed out.", errorCode: "TIMEOUT" }, "darwin"), false);
15
+ });