@bugmole/cli 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. package/.bugmole.env.example +20 -0
  2. package/LICENSE +7 -0
  3. package/README.md +293 -0
  4. package/TESTING.md +117 -0
  5. package/bugmole.config.yaml +39 -0
  6. package/package.json +85 -0
  7. package/scripts/billing/paypal-setup.mjs +121 -0
  8. package/scripts/bugmole-continue.ts +318 -0
  9. package/scripts/bugmole-init.ts +188 -0
  10. package/scripts/bugmole.cjs +17 -0
  11. package/scripts/bugmole.test.ts +344 -0
  12. package/scripts/bugmole.ts +657 -0
  13. package/scripts/ensure-maestro.cjs +79 -0
  14. package/scripts/ios-tunnel-keeper.sh +45 -0
  15. package/scripts/ios-wda-keeper.sh +66 -0
  16. package/scripts/sync-plan-catalog.d.mts +3 -0
  17. package/scripts/sync-plan-catalog.mjs +16 -0
  18. package/scripts/ui-parity-diff.py +65 -0
  19. package/scripts/ui-parity-requirements.txt +1 -0
  20. package/scripts/verify-manage-to-plans.mts +194 -0
  21. package/spec/app-ui-audit.schema.json +176 -0
  22. package/spec/blockers.yaml +79 -0
  23. package/spec/bugs.index.json +42 -0
  24. package/spec/design-dna.schema.json +38 -0
  25. package/spec/domain_rules.yaml +24 -0
  26. package/spec/flows/flow_manage-to-plans-64c3c61b.yaml +8 -0
  27. package/spec/flows/flow_screen-to-evidence-6c3ab309.yaml +7 -0
  28. package/spec/journey_graph.yaml +126 -0
  29. package/spec/journeys.graph.json +2618 -0
  30. package/spec/plans/dashboard-smoke.flow.yaml +20 -0
  31. package/spec/plans/example.flow.yaml +99 -0
  32. package/spec/plans/flow_api-keys-to-logout-7396cd5d.flow.yaml +33 -0
  33. package/spec/plans/flow_api-keys-to-screen-157e10a6.flow.yaml +33 -0
  34. package/spec/plans/flow_manage-to-audits-176c1eb8.flow.yaml +33 -0
  35. package/spec/plans/flow_manage-to-blockers-633282c8.flow.yaml +112 -0
  36. package/spec/plans/flow_manage-to-devices-ad57e202.flow.yaml +33 -0
  37. package/spec/plans/flow_manage-to-logout-86c54656.flow.yaml +33 -0
  38. package/spec/plans/flow_manage-to-manage-resource-action-6325f99a.flow.yaml +185 -0
  39. package/spec/plans/flow_manage-to-parity-issues-01fa2579.flow.yaml +34 -0
  40. package/spec/plans/flow_manage-to-plans-64c3c61b.flow.yaml +123 -0
  41. package/spec/plans/flow_manage-to-screen-6857f915.flow.yaml +106 -0
  42. package/spec/plans/flow_manage-to-tasks-c5791c95.flow.yaml +33 -0
  43. package/spec/plans/flow_profile-to-screen-6045aa3d.flow.yaml +33 -0
  44. package/spec/plans/flow_register-to-api-auth-login-460a9cc8.flow.yaml +34 -0
  45. package/spec/plans/flow_screen-to-audits-07666356.flow.yaml +33 -0
  46. package/spec/plans/flow_screen-to-blockers-091d258b.flow.yaml +33 -0
  47. package/spec/plans/flow_screen-to-devices-cad72f75.flow.yaml +33 -0
  48. package/spec/plans/flow_screen-to-evidence-6c3ab309.flow.yaml +126 -0
  49. package/spec/plans/flow_screen-to-parity-issues-2218e3ad.flow.yaml +34 -0
  50. package/spec/plans/flow_screen-to-plans-8232a94f.flow.yaml +33 -0
  51. package/spec/plans/flow_screen-to-tasks-540671d1.flow.yaml +33 -0
  52. package/spec/plans/login.flow.yaml +22 -0
  53. package/spec/plans/owner-operations.flow.yaml +20 -0
  54. package/spec/project_config.yaml +55 -0
  55. package/spec/roles.yaml +30 -0
  56. package/spec/schema.md +394 -0
  57. package/spec/test-case-results.schema.json +62 -0
  58. package/spec/test-cases.schema.json +85 -0
  59. package/spec/ui-parity-audit.schema.json +194 -0
  60. package/spec/ui-reverse-engineering.schema.json +94 -0
  61. package/src/billing/plan-catalog.test.ts +46 -0
  62. package/src/billing/plan-catalog.ts +199 -0
  63. package/src/integrations/aws-sigv4.test.ts +42 -0
  64. package/src/integrations/aws-sigv4.ts +72 -0
  65. package/src/integrations/device-farm.ts +155 -0
  66. package/src/integrations/github-app.test.ts +57 -0
  67. package/src/integrations/github-app.ts +143 -0
  68. package/src/integrations/gitlab.ts +81 -0
  69. package/src/integrations/temp-email.test.ts +123 -0
  70. package/src/integrations/temp-email.ts +175 -0
  71. package/src/integrations/testflight-feedback.test.ts +51 -0
  72. package/src/integrations/testflight-feedback.ts +173 -0
  73. package/src/integrations/webdriver-client.ts +131 -0
  74. package/src/mcp/server.test.ts +1220 -0
  75. package/src/mcp/server.ts +3064 -0
  76. package/src/mcp/write-test-cases.test.ts +287 -0
  77. package/src/registry/api-key-client.ts +39 -0
  78. package/src/registry/control-plane-client.ts +212 -0
  79. package/src/registry/migrations/0001_registry.sql +47 -0
  80. package/src/registry/migrations/0002_device_authorizations.sql +23 -0
  81. package/src/registry/migrations/0003_project_environments.sql +25 -0
  82. package/src/registry/migrations/0004_device_authorization_email.sql +1 -0
  83. package/src/registry/migrations/0005_testing_control_plane.sql +55 -0
  84. package/src/registry/migrations/0006_workspaces.sql +36 -0
  85. package/src/registry/migrations/0007_project_apps.sql +26 -0
  86. package/src/registry/migrations/0008_agent_tasks.sql +30 -0
  87. package/src/registry/migrations/0009_journey_revisions.sql +17 -0
  88. package/src/registry/migrations/0010_agent_task_journey.sql +2 -0
  89. package/src/registry/migrations/0011_run_journey_revision.sql +1 -0
  90. package/src/registry/migrations/0012_canonical_flow_execution.sql +10 -0
  91. package/src/registry/migrations/0013_device_sessions.sql +22 -0
  92. package/src/registry/migrations/0014_agent_task_step.sql +1 -0
  93. package/src/registry/migrations/0015_agent_task_attempts.sql +6 -0
  94. package/src/registry/migrations/0016_github_issue_tracker.sql +54 -0
  95. package/src/registry/migrations/0017_run_targets.sql +7 -0
  96. package/src/registry/migrations/0018_run_fix_from.sql +4 -0
  97. package/src/registry/migrations/0019_workspace_flags.sql +9 -0
  98. package/src/registry/migrations/0020_orgs.sql +40 -0
  99. package/src/registry/migrations/0021_billing_core.sql +58 -0
  100. package/src/registry/migrations/0022_cloud_runners.sql +19 -0
  101. package/src/registry/migrations/0023_signup.sql +4 -0
  102. package/src/registry/migrations/0024_billing.sql +67 -0
  103. package/src/registry/migrations/0025_notifications.sql +47 -0
  104. package/src/registry/migrations/0026_repo_bindings.sql +28 -0
  105. package/src/registry/migrations/0027_feedback.sql +29 -0
  106. package/src/registry/migrations/0028_devices.sql +48 -0
  107. package/src/registry/migrations/0029_sso.sql +31 -0
  108. package/src/registry/migrations/0030_workspace_domains.sql +18 -0
  109. package/src/registry/migrations/0031_gitlab_and_teams.sql +45 -0
  110. package/src/registry/migrations/0032_personas.sql +15 -0
  111. package/src/registry/migrations/0033_bugmole_rename.sql +11 -0
  112. package/src/registry/task-scheduling.test.ts +100 -0
  113. package/src/registry/task-scheduling.ts +80 -0
  114. package/src/registry-worker/ai/platform-model.ts +77 -0
  115. package/src/registry-worker/ai/routes.test.ts +88 -0
  116. package/src/registry-worker/ai/routes.ts +90 -0
  117. package/src/registry-worker/artifacts.test.ts +98 -0
  118. package/src/registry-worker/artifacts.ts +85 -0
  119. package/src/registry-worker/billing/billing-core.test.ts +175 -0
  120. package/src/registry-worker/billing/checkout-routes.ts +209 -0
  121. package/src/registry-worker/billing/enforcement.ts +69 -0
  122. package/src/registry-worker/billing/entitlements.ts +108 -0
  123. package/src/registry-worker/billing/ledger.ts +186 -0
  124. package/src/registry-worker/billing/paypal/api.ts +259 -0
  125. package/src/registry-worker/billing/paypal/client.ts +91 -0
  126. package/src/registry-worker/billing/paypal/provider.ts +143 -0
  127. package/src/registry-worker/billing/paypal.test.ts +466 -0
  128. package/src/registry-worker/billing/provider.ts +114 -0
  129. package/src/registry-worker/billing/routes.ts +66 -0
  130. package/src/registry-worker/billing/subscriptions.ts +780 -0
  131. package/src/registry-worker/billing/thresholds.ts +107 -0
  132. package/src/registry-worker/core.ts +308 -0
  133. package/src/registry-worker/devices/devices.test.ts +185 -0
  134. package/src/registry-worker/devices/policy.ts +71 -0
  135. package/src/registry-worker/devices/routes.ts +453 -0
  136. package/src/registry-worker/domains/domains.test.ts +210 -0
  137. package/src/registry-worker/domains/routes.ts +139 -0
  138. package/src/registry-worker/domains.ts +88 -0
  139. package/src/registry-worker/email/sender.ts +75 -0
  140. package/src/registry-worker/env.d.ts +14716 -0
  141. package/src/registry-worker/features.ts +20 -0
  142. package/src/registry-worker/feedback/feedback.test.ts +230 -0
  143. package/src/registry-worker/feedback/format.ts +148 -0
  144. package/src/registry-worker/feedback/routes.ts +386 -0
  145. package/src/registry-worker/flags.ts +39 -0
  146. package/src/registry-worker/github/checks.test.ts +177 -0
  147. package/src/registry-worker/github/checks.ts +374 -0
  148. package/src/registry-worker/gitlab/checks.test.ts +141 -0
  149. package/src/registry-worker/gitlab/checks.ts +349 -0
  150. package/src/registry-worker/hooks.ts +54 -0
  151. package/src/registry-worker/index.ts +2077 -0
  152. package/src/registry-worker/jobs/index.ts +29 -0
  153. package/src/registry-worker/jobs/retention.ts +68 -0
  154. package/src/registry-worker/mcp/mcp.test.ts +355 -0
  155. package/src/registry-worker/mcp/routes.ts +215 -0
  156. package/src/registry-worker/mcp/token.ts +126 -0
  157. package/src/registry-worker/mcp/tools.ts +563 -0
  158. package/src/registry-worker/notifications/alerts.ts +212 -0
  159. package/src/registry-worker/notifications/notifications.test.ts +298 -0
  160. package/src/registry-worker/notifications/outbox.ts +83 -0
  161. package/src/registry-worker/notifications/routes.ts +280 -0
  162. package/src/registry-worker/notifications/secrets.ts +49 -0
  163. package/src/registry-worker/notifications/slack.ts +96 -0
  164. package/src/registry-worker/notifications/teams.ts +46 -0
  165. package/src/registry-worker/org/audit.ts +116 -0
  166. package/src/registry-worker/org/routes.test.ts +163 -0
  167. package/src/registry-worker/org/routes.ts +302 -0
  168. package/src/registry-worker/personas/personas.test.ts +78 -0
  169. package/src/registry-worker/personas/routes.ts +100 -0
  170. package/src/registry-worker/repo-triggers.ts +20 -0
  171. package/src/registry-worker/routes/index.ts +74 -0
  172. package/src/registry-worker/run-events.ts +24 -0
  173. package/src/registry-worker/runner/dispatch.ts +219 -0
  174. package/src/registry-worker/runner/jobs.ts +43 -0
  175. package/src/registry-worker/runner/metering.ts +82 -0
  176. package/src/registry-worker/runner/policy.ts +59 -0
  177. package/src/registry-worker/runner/routes.ts +171 -0
  178. package/src/registry-worker/runner/runner.test.ts +358 -0
  179. package/src/registry-worker/runner/tokens.ts +93 -0
  180. package/src/registry-worker/runs.test.ts +60 -0
  181. package/src/registry-worker/signup/policy.ts +57 -0
  182. package/src/registry-worker/signup/routes.ts +106 -0
  183. package/src/registry-worker/signup/signup.test.ts +81 -0
  184. package/src/registry-worker/sso/aegis.ts +141 -0
  185. package/src/registry-worker/sso/membership.ts +157 -0
  186. package/src/registry-worker/sso/routes.ts +458 -0
  187. package/src/registry-worker/sso/sso.test.ts +344 -0
  188. package/src/registry-worker/testing/d1-shim.ts +180 -0
  189. package/src/registry-worker/testing/harness.ts +137 -0
  190. package/src/runner-worker/index.ts +108 -0
  191. package/src/runtime/ai-analysis.ts +97 -0
  192. package/src/runtime/ai-exploration.test.ts +32 -0
  193. package/src/runtime/ai-exploration.ts +69 -0
  194. package/src/runtime/ai-repair.ts +74 -0
  195. package/src/runtime/ai-work.test.ts +99 -0
  196. package/src/runtime/android-screen-record.test.ts +75 -0
  197. package/src/runtime/android-screen-record.ts +192 -0
  198. package/src/runtime/app-understanding.test.ts +123 -0
  199. package/src/runtime/app-understanding.ts +201 -0
  200. package/src/runtime/appium-driver.test.ts +179 -0
  201. package/src/runtime/appium-driver.ts +295 -0
  202. package/src/runtime/blocker-resolution.test.ts +113 -0
  203. package/src/runtime/blocker-resolution.ts +111 -0
  204. package/src/runtime/browser-matrix.integration.test.ts +212 -0
  205. package/src/runtime/browser-matrix.test.ts +143 -0
  206. package/src/runtime/browser-matrix.ts +200 -0
  207. package/src/runtime/canonical-flow.test.ts +52 -0
  208. package/src/runtime/config-validate.ts +185 -0
  209. package/src/runtime/continuous-execution.ts +291 -0
  210. package/src/runtime/cursor-applescript.ts +573 -0
  211. package/src/runtime/cursor-cli-driver.test.ts +78 -0
  212. package/src/runtime/cursor-cli-driver.ts +156 -0
  213. package/src/runtime/cursor-driver-example.ts +117 -0
  214. package/src/runtime/cursor-driver-index.ts +65 -0
  215. package/src/runtime/cursor-driver-init.ts +277 -0
  216. package/src/runtime/cursor-driver-run.test.ts +15 -0
  217. package/src/runtime/cursor-driver-run.ts +323 -0
  218. package/src/runtime/cursor-driver.ts +332 -0
  219. package/src/runtime/cursor-llm-example.ts +90 -0
  220. package/src/runtime/cursor-llm.ts +206 -0
  221. package/src/runtime/cursor-mcp-monitor.ts +386 -0
  222. package/src/runtime/device-clouds/browserstack.ts +73 -0
  223. package/src/runtime/device-clouds/device-farm.ts +52 -0
  224. package/src/runtime/device-clouds/index.ts +92 -0
  225. package/src/runtime/device-clouds/kobiton.ts +70 -0
  226. package/src/runtime/device-clouds/targets.ts +44 -0
  227. package/src/runtime/device-clouds/types.ts +62 -0
  228. package/src/runtime/diff-proposal.ts +84 -0
  229. package/src/runtime/discovery-task.test.ts +29 -0
  230. package/src/runtime/discovery-task.ts +284 -0
  231. package/src/runtime/driver-recovery.ts +69 -0
  232. package/src/runtime/driver.ts +79 -0
  233. package/src/runtime/environment.test.ts +104 -0
  234. package/src/runtime/environment.ts +137 -0
  235. package/src/runtime/executor.test.ts +509 -0
  236. package/src/runtime/executor.ts +921 -0
  237. package/src/runtime/explorer.test.ts +101 -0
  238. package/src/runtime/explorer.ts +1013 -0
  239. package/src/runtime/failure-analysis.test.ts +111 -0
  240. package/src/runtime/failure-analysis.ts +272 -0
  241. package/src/runtime/fixtures/fake-maestro.sh +36 -0
  242. package/src/runtime/flow-language.test.ts +268 -0
  243. package/src/runtime/flow-language.ts +414 -0
  244. package/src/runtime/init-wizard.ts +354 -0
  245. package/src/runtime/ios-screen-record.test.ts +68 -0
  246. package/src/runtime/ios-screen-record.ts +155 -0
  247. package/src/runtime/journey-editor.ts +452 -0
  248. package/src/runtime/journey-evidence.test.ts +161 -0
  249. package/src/runtime/journey-evidence.ts +180 -0
  250. package/src/runtime/journey-graph.test.ts +257 -0
  251. package/src/runtime/journey-graph.ts +170 -0
  252. package/src/runtime/legacy-names.ts +32 -0
  253. package/src/runtime/llm-example.ts +105 -0
  254. package/src/runtime/llm.ts +527 -0
  255. package/src/runtime/local-browser.test.ts +45 -0
  256. package/src/runtime/local-browser.ts +48 -0
  257. package/src/runtime/local-registry-stub.test.ts +325 -0
  258. package/src/runtime/local-registry-stub.ts +803 -0
  259. package/src/runtime/maestro-driver.test.ts +84 -0
  260. package/src/runtime/maestro-driver.ts +209 -0
  261. package/src/runtime/mole-voice.ts +21 -0
  262. package/src/runtime/nav-crawl.test.ts +100 -0
  263. package/src/runtime/nav-crawl.ts +153 -0
  264. package/src/runtime/pipeline.test.ts +405 -0
  265. package/src/runtime/pipeline.ts +833 -0
  266. package/src/runtime/planner.test.ts +37 -0
  267. package/src/runtime/planner.ts +274 -0
  268. package/src/runtime/platform-ai.ts +76 -0
  269. package/src/runtime/playwright-driver.test.ts +93 -0
  270. package/src/runtime/playwright-driver.ts +620 -0
  271. package/src/runtime/project-spec.ts +140 -0
  272. package/src/runtime/record-run-verdicts.ts +68 -0
  273. package/src/runtime/reporter.test.ts +56 -0
  274. package/src/runtime/reporter.ts +158 -0
  275. package/src/runtime/reset.test.ts +44 -0
  276. package/src/runtime/reset.ts +61 -0
  277. package/src/runtime/reviewer.test.ts +73 -0
  278. package/src/runtime/reviewer.ts +158 -0
  279. package/src/runtime/run-job.ts +136 -0
  280. package/src/runtime/run-once.test.ts +207 -0
  281. package/src/runtime/run-once.ts +168 -0
  282. package/src/runtime/run.ts +132 -0
  283. package/src/runtime/screen-recording.ts +34 -0
  284. package/src/runtime/serve-gateway.test.ts +74 -0
  285. package/src/runtime/serve-gateway.ts +164 -0
  286. package/src/runtime/serve-worker.test.ts +23 -0
  287. package/src/runtime/serve-worker.ts +278 -0
  288. package/src/runtime/site-discovery.test.ts +168 -0
  289. package/src/runtime/site-discovery.ts +308 -0
  290. package/src/runtime/target-runner.ts +144 -0
  291. package/src/runtime/test-case-verdicts.test.ts +94 -0
  292. package/src/runtime/test-case-verdicts.ts +120 -0
  293. package/src/runtime/ui-reverse-engineering/coordinator.test.ts +59 -0
  294. package/src/runtime/ui-reverse-engineering/coordinator.ts +391 -0
  295. package/src/runtime/ui-reverse-engineering/types.ts +197 -0
  296. package/src/runtime/web-suite.test.ts +97 -0
  297. package/src/runtime/web-suite.ts +176 -0
  298. package/src/storage/create-object-store.ts +144 -0
  299. package/src/storage/keys.ts +34 -0
  300. package/src/storage/local-artifact-server.test.ts +314 -0
  301. package/src/storage/local-artifact-server.ts +357 -0
  302. package/src/storage/object-store.test.ts +28 -0
  303. package/src/storage/object-store.ts +101 -0
  304. package/src/storage/registry-object-store.ts +88 -0
  305. package/src/storage/remote-object-store.ts +104 -0
  306. package/src/storage/storage-directory.test.ts +43 -0
  307. package/src/storage/storage-directory.ts +24 -0
  308. package/tsconfig.json +24 -0
@@ -0,0 +1,921 @@
1
+ import fs from "node:fs";
2
+ import path from "node:path";
3
+ import YAML from "yaml";
4
+ import { createProjectObjectStore } from "../storage/create-object-store.js";
5
+ import { AppiumDriver } from "./appium-driver.js";
6
+ import type { DeviceCloudAccess } from "./device-clouds/index.js";
7
+ import { json, type ObjectStore } from "../storage/object-store.js";
8
+ import { runKey, type RunManifest, projectPrefix, runPrefix } from "../storage/keys.js";
9
+ import type { JourneyScreenshotEvidence } from "./journey-graph.js";
10
+ import { journeyScreenshotKey } from "./journey-evidence.js";
11
+ import type { AppDriver, DriverPlatform } from "./driver.js";
12
+ import { MaestroDriver } from "./maestro-driver.js";
13
+ import { PlaywrightDriver } from "./playwright-driver.js";
14
+ import { recoverDriverFailure } from "./driver-recovery.js";
15
+ import { startAndroidScreenRecording } from "./android-screen-record.js";
16
+ import { startIosScreenRecording } from "./ios-screen-record.js";
17
+ import { defaultTargets, targetLabel, type RunTarget } from "./browser-matrix.js";
18
+ import { applyFlowChanges, type FailureAnalysis, type FailureContext, type FlowChange } from "./failure-analysis.js";
19
+ import { describeStepLabel, parseMaestroStyleFlow, type FlowStep } from "./flow-language.js";
20
+ import { explainWith } from "./ai-analysis.js";
21
+ import { parseRepair, repairPrompt } from "./ai-repair.js";
22
+ import { modelCaller, type ModelCallIds } from "./platform-ai.js";
23
+ import {
24
+ aggregateTargetResults,
25
+ queuedTargetResult,
26
+ resolveTargetConcurrency,
27
+ runPool,
28
+ stepCounts,
29
+ type TargetResult,
30
+ } from "./target-runner.js";
31
+
32
+ export interface RunOptions {
33
+ planId: string;
34
+ journeyId?: string;
35
+ runId?: string;
36
+ /**
37
+ * The selected environment's base URL. Overrides the plan target, and moves
38
+ * absolute flow URLs onto this origin so a flow runs where it was sent.
39
+ */
40
+ baseUrl?: string;
41
+ /** Browsers/devices to fan a web run out across. Defaults to execution.playwright config. */
42
+ targets?: RunTarget[];
43
+ onProgress?: (progress: {
44
+ currentStep: string;
45
+ stepsExecuted: number;
46
+ stepsPassed: number;
47
+ stepsFailed: number;
48
+ node?: NodeExecutionResult;
49
+ targetResults?: TargetResult[];
50
+ }) => void | Promise<void>;
51
+ captureScreenshot?: (nodeId: string, action: string) => Promise<Uint8Array>;
52
+ /**
53
+ * A fix to apply to the flow before running it. "suggested" applies the
54
+ * earlier run's checked suggestion (the dashboard's "Apply fix"); "ai" asks
55
+ * a model for the edit first, from the earlier failure (AI test repair).
56
+ */
57
+ applyFix?: {
58
+ changes: FlowChange[];
59
+ sourceRunId?: string;
60
+ targetId?: string;
61
+ mode?: "suggested" | "ai";
62
+ failureContext?: FailureContext;
63
+ analysis?: FailureAnalysis;
64
+ };
65
+ /** Test seam: builds the driver for a driver id instead of the registered adapters. */
66
+ driverFactory?: (cfg: any, driverId: string) => AppDriver;
67
+ /** Credentials and app builds for device-cloud targets, loaded from the registry. */
68
+ deviceClouds?: DeviceCloudAccess;
69
+ }
70
+
71
+ export interface NodeExecutionResult {
72
+ nodeId: string;
73
+ action: string;
74
+ status: "success" | "blocked" | "failed";
75
+ reason: { code: string; message: string; details?: unknown };
76
+ proofs: any[];
77
+ startedAt: string;
78
+ completedAt: string;
79
+ screenshot?: JourneyScreenshotEvidence;
80
+ }
81
+
82
+ export interface ExecutionResult {
83
+ status: "success" | "error" | "partial";
84
+ runId: string;
85
+ journeyId: string;
86
+ stepsExecuted: number;
87
+ stepsPassed: number;
88
+ stepsFailed: number;
89
+ driver?: string;
90
+ platform?: string;
91
+ recoveryAttempts?: number;
92
+ failureCode?: string;
93
+ evidencePath?: string;
94
+ /** Object-store key of the run's manifest.json. */
95
+ manifestKey?: string;
96
+ message?: string;
97
+ error?: string;
98
+ /** Per browser/device outcome when the run fanned out across a matrix. */
99
+ targetResults?: TargetResult[];
100
+ durationMs?: number;
101
+ /** Set when a suggested fix was written to the flow before this run. */
102
+ fixApplied?: { sourceRunId?: string; targetId?: string; flow: string };
103
+ results?: Array<{
104
+ step: string;
105
+ nodeId: string;
106
+ action: string;
107
+ outcome: "PASS" | "FAIL" | "UNKNOWN";
108
+ proofs: any[];
109
+ status: NodeExecutionResult["status"];
110
+ reason: NodeExecutionResult["reason"];
111
+ startedAt: string;
112
+ completedAt: string;
113
+ screenshot?: JourneyScreenshotEvidence;
114
+ }>;
115
+ }
116
+
117
+ function configuredDriver(cfg: any, driverId: string): AppDriver {
118
+ if (driverId === "maestro") {
119
+ return new MaestroDriver({
120
+ command: cfg.execution?.maestro?.command,
121
+ extraArgs: Array.isArray(cfg.execution?.maestro?.args) ? cfg.execution.maestro.args : undefined,
122
+ });
123
+ }
124
+ if (driverId === "playwright") {
125
+ return new PlaywrightDriver({
126
+ headless: cfg.execution?.playwright?.headless,
127
+ });
128
+ }
129
+ throw new Error(`No app driver adapter is registered for "${driverId}"`);
130
+ }
131
+
132
+ /**
133
+ * Executor: Executes plan steps with Playwright, produces Proof Tokens
134
+ */
135
+ /**
136
+ * A cloud worker starts from a clean container, so the plan may exist only in the object store,
137
+ * where the planner publishes it (projects/<id>/plans/<planId>.flow.yaml).
138
+ */
139
+ export async function readStoredPlan(store: ObjectStore, projectId: string, planId: string): Promise<string | undefined> {
140
+ if (!/^[a-z0-9._-]+$/i.test(planId)) return undefined;
141
+ const base = planId.replace(/(\.flow)?\.ya?ml$/i, "");
142
+ for (const name of [`${base}.flow.yaml`, `${base}.yaml`]) {
143
+ const key = `${projectPrefix(projectId)}/plans/${name}`;
144
+ try {
145
+ if (await store.has(key)) return await store.getText(key);
146
+ } catch {
147
+ // Try the next candidate; a missing plan is reported by the caller.
148
+ }
149
+ }
150
+ return undefined;
151
+ }
152
+
153
+ /**
154
+ * Resolves a plan's flow file, downloading it from the object store (projects/<id>/spec/<flow>)
155
+ * into the run's artifacts directory when it isn't on the worker's disk.
156
+ */
157
+ export async function resolveFlowPath(
158
+ store: ObjectStore,
159
+ projectId: string,
160
+ runId: string,
161
+ artifactsDir: string,
162
+ flowReference: string,
163
+ localFlowPath: string,
164
+ ): Promise<string> {
165
+ if (path.isAbsolute(flowReference) || fs.existsSync(localFlowPath)) return localFlowPath;
166
+ const normalized = flowReference.replaceAll("\\", "/").replace(/^\/+/, "");
167
+ if (normalized.split("/").includes("..")) return localFlowPath;
168
+ const key = `${projectPrefix(projectId)}/spec/${normalized}`;
169
+ try {
170
+ if (!(await store.has(key))) return localFlowPath;
171
+ const destination = path.join(artifactsDir, projectId, runId, "spec", normalized);
172
+ await fs.promises.mkdir(path.dirname(destination), { recursive: true });
173
+ await fs.promises.writeFile(destination, await store.get(key));
174
+ return destination;
175
+ } catch {
176
+ return localFlowPath;
177
+ }
178
+ }
179
+
180
+ /** The plan's step nodes after a fix: inserted flow steps get nodes of their own. */
181
+ function syncPlanSteps(planSteps: any[], changes: readonly FlowChange[], fixedSteps: FlowStep[]): any[] {
182
+ const next = [...planSteps];
183
+ const ordered = [...changes].sort((left, right) => right.index - left.index);
184
+ let added = 0;
185
+ for (const change of ordered) {
186
+ if (change.op === "insert") {
187
+ const nodes = change.steps.map((_step, offset) => {
188
+ added += 1;
189
+ return { node_id: `fix-${added}`, action: describeStepLabel(fixedSteps[change.index + offset]) };
190
+ });
191
+ next.splice(change.index, 0, ...nodes);
192
+ } else {
193
+ next[change.index] = { ...next[change.index], action: describeStepLabel(fixedSteps[change.index]) };
194
+ }
195
+ }
196
+ return next;
197
+ }
198
+
199
+ /**
200
+ * Lets the project's AI model explain a failed target in plain words, and
201
+ * rewrites that target's analysis.json in place before it is uploaded.
202
+ */
203
+ async function refineAnalysis(cfg: any, result: Awaited<ReturnType<AppDriver["run"]>>, ids: ModelCallIds): Promise<FailureAnalysis | undefined> {
204
+ const analysisFile = result.artifacts.find((artifact) => artifact.name === "analysis.json");
205
+ const contextFile = result.artifacts.find((artifact) => artifact.name === "failure-context.json");
206
+ if (!analysisFile) return undefined;
207
+ try {
208
+ const analysis = JSON.parse(await fs.promises.readFile(analysisFile.path, "utf8")) as FailureAnalysis;
209
+ if (!contextFile) return analysis;
210
+ const context = JSON.parse(await fs.promises.readFile(contextFile.path, "utf8")) as FailureContext;
211
+ // The project's own model, else Bugmole AI (1 credit); without either the
212
+ // rules' explanation stands.
213
+ const caller = modelCaller(cfg, ids);
214
+ if (!caller) return analysis;
215
+ const explained = await explainWith(async (prompt) => caller(prompt, "analysis"), analysis, context);
216
+ if (explained !== analysis) await fs.promises.writeFile(analysisFile.path, `${JSON.stringify(explained, null, 2)}\n`, "utf8");
217
+ return explained;
218
+ } catch {
219
+ return undefined;
220
+ }
221
+ }
222
+
223
+ export async function execute(cfg: any, runOptions: RunOptions): Promise<ExecutionResult> {
224
+ // Progress is reporting, not execution: a failed update (a registry blip)
225
+ // must not fail the run it describes. The next update resends the state.
226
+ const reportProgress = runOptions.onProgress;
227
+ const options: RunOptions = reportProgress
228
+ ? {
229
+ ...runOptions,
230
+ onProgress: async (progress) => {
231
+ try {
232
+ await reportProgress(progress);
233
+ } catch (error) {
234
+ console.warn(`[Bugmole] Progress update failed: ${error instanceof Error ? error.message : error}`);
235
+ }
236
+ },
237
+ }
238
+ : runOptions;
239
+ try {
240
+ const { planId, journeyId } = options;
241
+ const specDir = cfg.runtime.spec_dir || "./spec";
242
+ const artifactsDir = cfg.runtime.artifacts_dir || "./runs";
243
+ const store: ObjectStore = createProjectObjectStore(cfg, artifactsDir);
244
+ const plansDir = path.join(specDir, "plans");
245
+
246
+ // Generate run ID
247
+ const runId = options.runId ?? `${cfg.runtime.run_id_prefix || "run"}_${Date.now()}`;
248
+ const startedAt = new Date().toISOString();
249
+
250
+ // Load plan - handle both journey ID and full paths
251
+ let planPath: string;
252
+
253
+ // If planId is already a full path, use it directly
254
+ if (path.isAbsolute(planId) || planId.includes("/") || planId.includes("\\")) {
255
+ // It's a path - try to resolve it
256
+ if (path.isAbsolute(planId)) {
257
+ planPath = planId;
258
+ } else {
259
+ // Relative path - try multiple locations
260
+ const possiblePaths = [
261
+ path.join(specDir, planId),
262
+ path.join(plansDir, planId),
263
+ planId, // Try as-is
264
+ ];
265
+
266
+ // Also try with .flow.yaml extension if not present
267
+ if (!planId.endsWith(".yaml") && !planId.endsWith(".yml")) {
268
+ possiblePaths.push(
269
+ path.join(plansDir, `${planId}.flow.yaml`),
270
+ path.join(specDir, `${planId}.flow.yaml`),
271
+ path.join(plansDir, `${planId}.yaml`), // Also try without .flow prefix
272
+ path.join(specDir, `${planId}.yaml`),
273
+ );
274
+ } else {
275
+ // If it already has .yaml extension, also try with .flow.yaml
276
+ const baseName = planId.replace(/\.(yaml|yml)$/, "");
277
+ possiblePaths.push(
278
+ path.join(plansDir, `${baseName}.flow.yaml`),
279
+ path.join(specDir, `${baseName}.flow.yaml`),
280
+ );
281
+ }
282
+
283
+ // Try project directory from environment
284
+ const projectDirs = [
285
+ process.env.CURSOR_PROJECT_DIRECTORY,
286
+ process.env.BUGMOLE_PROJECT_DIR,
287
+ ].filter(Boolean);
288
+
289
+ for (const projectDir of projectDirs) {
290
+ if (projectDir) {
291
+ const projectSpecDir = path.join(projectDir, "spec");
292
+ const projectPlansDir = path.join(projectSpecDir, "plans");
293
+ possiblePaths.push(
294
+ path.join(projectSpecDir, planId),
295
+ path.join(projectPlansDir, planId),
296
+ );
297
+ if (!planId.endsWith(".yaml") && !planId.endsWith(".yml")) {
298
+ possiblePaths.push(
299
+ path.join(projectPlansDir, `${planId}.flow.yaml`),
300
+ path.join(projectPlansDir, `${planId}.yaml`), // Also try without .flow
301
+ );
302
+ } else {
303
+ // If it already has .yaml extension, also try with .flow.yaml
304
+ const baseName = planId.replace(/\.(yaml|yml)$/, "");
305
+ possiblePaths.push(path.join(projectPlansDir, `${baseName}.flow.yaml`));
306
+ }
307
+ }
308
+ }
309
+
310
+ // Additional fallback: if we're in ai-qa-framework, try to find neko-coffee in nested paths
311
+ if (process.cwd().includes("ai-qa-framework")) {
312
+ const parentDir = path.dirname(process.cwd());
313
+ const commonProjectNames = ["neko-coffee", "kive", "neko-coffee-pos"];
314
+
315
+ // Try nested paths like Projects/kive/neko-coffee
316
+ for (const projectName of commonProjectNames) {
317
+ const kiveDir = path.join(parentDir, "kive");
318
+ if (fs.existsSync(kiveDir)) {
319
+ const potentialProjectDir = path.join(kiveDir, projectName);
320
+ if (fs.existsSync(potentialProjectDir)) {
321
+ const projectPlansDir = path.join(potentialProjectDir, "spec", "plans");
322
+ possiblePaths.push(path.join(projectPlansDir, planId));
323
+ if (!planId.endsWith(".yaml") && !planId.endsWith(".yml")) {
324
+ possiblePaths.push(
325
+ path.join(projectPlansDir, `${planId}.yaml`),
326
+ path.join(projectPlansDir, `${planId}.flow.yaml`),
327
+ );
328
+ } else {
329
+ const baseName = planId.replace(/\.(yaml|yml)$/, "");
330
+ possiblePaths.push(
331
+ path.join(projectPlansDir, `${baseName}.yaml`),
332
+ path.join(projectPlansDir, `${baseName}.flow.yaml`),
333
+ );
334
+ }
335
+ }
336
+ }
337
+ }
338
+ }
339
+
340
+ // Find first existing path
341
+ planPath = possiblePaths.find((p) => fs.existsSync(p)) || possiblePaths[0];
342
+ }
343
+ } else {
344
+ // It's just a journey ID - try both .yaml and .flow.yaml
345
+ const possiblePaths = [
346
+ path.join(plansDir, `${planId}.yaml`),
347
+ path.join(plansDir, `${planId}.flow.yaml`),
348
+ path.join(specDir, "plans", `${planId}.yaml`),
349
+ path.join(specDir, "plans", `${planId}.flow.yaml`),
350
+ ];
351
+
352
+ // Try project directory from environment
353
+ const projectDirs = [
354
+ process.env.CURSOR_PROJECT_DIRECTORY,
355
+ process.env.BUGMOLE_PROJECT_DIR,
356
+ ].filter(Boolean);
357
+
358
+ for (const projectDir of projectDirs) {
359
+ if (projectDir) {
360
+ const projectPlansDir = path.join(projectDir, "spec", "plans");
361
+ possiblePaths.push(
362
+ path.join(projectPlansDir, `${planId}.yaml`),
363
+ path.join(projectPlansDir, `${planId}.flow.yaml`),
364
+ );
365
+ }
366
+ }
367
+
368
+ // Additional fallback: if we're in ai-qa-framework, try to find neko-coffee in nested paths
369
+ if (process.cwd().includes("ai-qa-framework")) {
370
+ const parentDir = path.dirname(process.cwd());
371
+ const commonProjectNames = ["neko-coffee", "kive", "neko-coffee-pos"];
372
+
373
+ // Try nested paths like Projects/kive/neko-coffee
374
+ for (const projectName of commonProjectNames) {
375
+ const kiveDir = path.join(parentDir, "kive");
376
+ if (fs.existsSync(kiveDir)) {
377
+ const potentialProjectDir = path.join(kiveDir, projectName);
378
+ if (fs.existsSync(potentialProjectDir)) {
379
+ const projectPlansDir = path.join(potentialProjectDir, "spec", "plans");
380
+ possiblePaths.push(
381
+ path.join(projectPlansDir, `${planId}.yaml`),
382
+ path.join(projectPlansDir, `${planId}.flow.yaml`),
383
+ );
384
+ }
385
+ }
386
+ }
387
+ }
388
+
389
+ // Find first existing path
390
+ planPath = possiblePaths.find((p) => fs.existsSync(p)) || possiblePaths[0];
391
+ }
392
+
393
+ const storedPlan = fs.existsSync(planPath) ? undefined : await readStoredPlan(store, cfg.project?.id ?? "default", planId);
394
+ if (!fs.existsSync(planPath) && storedPlan === undefined) {
395
+ return {
396
+ status: "error",
397
+ runId,
398
+ journeyId: journeyId || "unknown",
399
+ stepsExecuted: 0,
400
+ stepsPassed: 0,
401
+ stepsFailed: 0,
402
+ error:
403
+ `Plan not found: ${planPath}\n` +
404
+ `Plan ID provided: ${planId}\n` +
405
+ `Spec dir: ${specDir}\n` +
406
+ `Plans dir: ${plansDir}\n` +
407
+ `Current working directory: ${process.cwd()}\n` +
408
+ `Environment variables: CURSOR_PROJECT_DIRECTORY=${process.env.CURSOR_PROJECT_DIRECTORY || "not set"}, BUGMOLE_PROJECT_DIR=${process.env.BUGMOLE_PROJECT_DIR || "not set"}`,
409
+ };
410
+ }
411
+
412
+ const planContent = storedPlan ?? fs.readFileSync(planPath, "utf-8");
413
+ const plan = YAML.parse(planContent);
414
+
415
+ // Find journey to execute
416
+ let journey: any = null;
417
+ for (const suite of plan.suites || []) {
418
+ for (const j of suite.journeys || []) {
419
+ if (!journeyId || j.id === journeyId) {
420
+ journey = j;
421
+ break;
422
+ }
423
+ }
424
+ if (journey) break;
425
+ }
426
+
427
+ if (!journey) {
428
+ return {
429
+ status: "error",
430
+ runId,
431
+ journeyId: journeyId || "unknown",
432
+ stepsExecuted: 0,
433
+ stepsPassed: 0,
434
+ stepsFailed: 0,
435
+ error: `Journey not found: ${journeyId || "in plan"}`,
436
+ };
437
+ }
438
+
439
+ const driverId = journey.driver || plan.driver || cfg.execution?.driver || "maestro";
440
+ const platform = (journey.platform || plan.platform || cfg.project?.platform || "web") as DriverPlatform;
441
+ const flowReference = journey.flow || plan.flow;
442
+ let driverResult: Awaited<ReturnType<AppDriver["run"]>>;
443
+ let driver: AppDriver | undefined;
444
+ let recoveryAttempts = 0;
445
+ let targetResults: TargetResult[] | undefined;
446
+ let fixApplied: ExecutionResult["fixApplied"];
447
+ let fixError: string | undefined;
448
+ // Artifacts to upload, with the per-target folder they belong under.
449
+ let targetArtifacts: Array<{ targetId?: string; artifact: Awaited<ReturnType<AppDriver["run"]>>["artifacts"][number] }> = [];
450
+ if (driverId === "simulation") {
451
+ driverResult = {
452
+ status: "blocked",
453
+ message: "Simulation is available only as an explicit test driver.",
454
+ artifacts: [],
455
+ failureCode: "simulation_driver",
456
+ };
457
+ } else if (!flowReference) {
458
+ driverResult = {
459
+ status: "blocked",
460
+ message: `Driver "${driverId}" requires a flow reference.`,
461
+ artifacts: [],
462
+ failureCode: "missing_flow",
463
+ };
464
+ } else {
465
+ driver = (options.driverFactory ?? configuredDriver)(cfg, driverId);
466
+ const target = journey.target || plan.target || {};
467
+ const localFlowPath = path.isAbsolute(flowReference)
468
+ ? flowReference
469
+ : path.resolve(specDir, flowReference);
470
+ const flowPath = await resolveFlowPath(store, cfg.project?.id ?? "default", runId, artifactsDir, flowReference, localFlowPath);
471
+ if (options.applyFix) {
472
+ try {
473
+ const original = await fs.promises.readFile(flowPath, "utf8");
474
+ if (options.applyFix.mode === "ai") {
475
+ if (!options.applyFix.failureContext) throw new Error("the earlier run has no failure details to repair from.");
476
+ const caller = modelCaller(cfg, { projectId: cfg.project?.id ?? "default", runId, targetId: options.applyFix.targetId });
477
+ if (!caller) throw new Error("no AI model is available: connect a model key or Bugmole AI.");
478
+ await options.onProgress?.({ currentStep: "Asking AI for a fix", stepsExecuted: 0, stepsPassed: 0, stepsFailed: 0 });
479
+ const reply = await caller(repairPrompt(original, options.applyFix.failureContext, options.applyFix.analysis), "repair");
480
+ const proposed = parseRepair(reply.text, original);
481
+ if (!proposed) throw new Error("AI couldn't propose a change that passes the flow checks; the app may be at fault.");
482
+ options.applyFix = { ...options.applyFix, changes: proposed.changes };
483
+ }
484
+ if (options.applyFix.changes.length === 0) throw new Error("the earlier run has no suggested fix to apply.");
485
+ const updated = applyFlowChanges(original, options.applyFix.changes);
486
+ await fs.promises.writeFile(flowPath, updated, "utf8");
487
+ // When the plan lists one node per flow step, keep that list in step
488
+ // with the fixed flow, so results still line up step by step.
489
+ const stepsBefore = parseMaestroStyleFlow(original, "http://localhost").steps.length;
490
+ const fixedSteps = parseMaestroStyleFlow(updated, "http://localhost").steps;
491
+ if (Array.isArray(journey.steps) && journey.steps.length === stepsBefore) {
492
+ journey.steps = syncPlanSteps(journey.steps, options.applyFix.changes, fixedSteps);
493
+ const planText = YAML.stringify(plan);
494
+ if (storedPlan === undefined) await fs.promises.writeFile(planPath, planText, "utf8");
495
+ else await store.put(`${projectPrefix(cfg.project?.id ?? "default")}/plans/${planId}.flow.yaml`, planText, "text/yaml");
496
+ }
497
+ // Keep the stored copy in step, so every worker runs the fixed flow.
498
+ if (!path.isAbsolute(flowReference)) {
499
+ const storedKey = `${projectPrefix(cfg.project?.id ?? "default")}/spec/${flowReference.replaceAll("\\", "/").replace(/^\/+/, "")}`;
500
+ if (flowPath !== localFlowPath || await store.has(storedKey).catch(() => false)) {
501
+ await store.put(storedKey, updated, "text/yaml");
502
+ }
503
+ }
504
+ fixApplied = { sourceRunId: options.applyFix.sourceRunId, targetId: options.applyFix.targetId, flow: flowReference };
505
+ await options.onProgress?.({
506
+ currentStep: options.applyFix.mode === "ai" ? "Applied the AI fix" : "Applied the suggested fix",
507
+ stepsExecuted: 0,
508
+ stepsPassed: 0,
509
+ stepsFailed: 0,
510
+ });
511
+ } catch (error) {
512
+ fixError = error instanceof Error ? error.message : String(error);
513
+ }
514
+ }
515
+ const driverContext = {
516
+ runId,
517
+ projectId: cfg.project?.id ?? "default",
518
+ journeyId: journey.id,
519
+ flowPath,
520
+ platform,
521
+ target: {
522
+ baseUrl: options.baseUrl || target.base_url || target.baseUrl,
523
+ appId: target.app_id || target.appId,
524
+ deviceId: target.device_id || target.deviceId,
525
+ serial: target.serial,
526
+ },
527
+ timeoutMs: cfg.execution?.maestro?.timeout_ms ?? 300_000,
528
+ artifactsDir,
529
+ rebaseOrigin: Boolean(options.baseUrl),
530
+ };
531
+ // Bracket the whole run (including recovery retries) with a real
532
+ // device screen recording, so the video reflects the actual attempt
533
+ // that was graded rather than a separately-timed replay.
534
+ const videoBasePath = path.join(artifactsDir, cfg.project?.id ?? "default", runId, "run.mp4");
535
+ const recordingOptions = {
536
+ chunkSeconds: cfg.execution?.recording?.chunk_seconds,
537
+ // Off unless asked for: dropping frames makes the video shorter than
538
+ // the run it records, which is only what you want when you already
539
+ // know you are looking for the moments something moved.
540
+ skipIdle: cfg.execution?.recording?.skip_idle === true,
541
+ };
542
+ // iOS has no screenrecord, so it is recorded from WebDriverAgent's MJPEG
543
+ // server instead — the same runner already driving input. It records
544
+ // only where WDA is actually running, which is the same condition that
545
+ // makes an iOS run drivable at all.
546
+ // Device clouds record their own sessions; local recording only applies to local devices.
547
+ const onlyCloudTargets = Boolean(options.targets?.length) && options.targets!.every((target) => target.cloud);
548
+ const androidRecording = platform === "android" && driverContext.target.serial && !onlyCloudTargets
549
+ ? await startAndroidScreenRecording(driverContext.target.serial, "adb", recordingOptions)
550
+ : undefined;
551
+ const iosRecording = platform === "ios" && !onlyCloudTargets
552
+ ? await startIosScreenRecording(videoBasePath, recordingOptions)
553
+ : undefined;
554
+ // Collapsed to one shape here rather than shared through the recorders:
555
+ // Android pulls chunks off the device at stop and needs the destination
556
+ // then, while iOS has been writing to it all along.
557
+ const stopRecording: (() => Promise<string[]>) | undefined =
558
+ androidRecording ? () => androidRecording.stop(videoBasePath)
559
+ : iosRecording ? () => iosRecording.stop()
560
+ : undefined;
561
+ const recoveryPolicy = journey.recovery || plan.recovery || cfg.execution?.recovery;
562
+ const matrixConfigured = Array.isArray(cfg.execution?.playwright?.browsers) || Array.isArray(cfg.execution?.playwright?.devices);
563
+ // Device-cloud targets run any flow on real devices; browser targets only apply to web flows.
564
+ const cloudTargets = options.targets?.filter((target) => target.cloud) ?? [];
565
+ const runTargets = driverId === "playwright" && platform === "web"
566
+ ? (options.targets?.length ? options.targets : matrixConfigured ? defaultTargets(cfg) : [])
567
+ : cloudTargets;
568
+ const deviceDriver = runTargets.some((target) => target.cloud)
569
+ ? (options.driverFactory ? options.driverFactory(cfg, "appium") : new AppiumDriver({ access: options.deviceClouds }))
570
+ : undefined;
571
+ if (fixError) {
572
+ driverResult = {
573
+ status: "blocked",
574
+ message: `The suggested fix could not be applied: ${fixError}`,
575
+ artifacts: [],
576
+ failureCode: "fix_not_applicable",
577
+ };
578
+ } else if (runTargets.length > 0) {
579
+ const activeDriver = driver;
580
+ const live = runTargets.map(queuedTargetResult);
581
+ // Lanes report concurrently; updates are chained so they reach the
582
+ // registry in order.
583
+ let reporting: Promise<void> = Promise.resolve();
584
+ const report = (currentStep: string) => {
585
+ const snapshot = live.map((entry) => ({ ...entry }));
586
+ reporting = reporting.then(() => options.onProgress?.({
587
+ currentStep,
588
+ stepsExecuted: 0,
589
+ stepsPassed: 0,
590
+ stepsFailed: 0,
591
+ targetResults: snapshot,
592
+ }));
593
+ return reporting;
594
+ };
595
+ await report(`Starting ${runTargets.length} ${runTargets.length === 1 ? "target" : "targets"}`);
596
+ const outcomes = await runPool(
597
+ runTargets,
598
+ resolveTargetConcurrency(cfg.execution?.playwright?.parallel, runTargets.length),
599
+ async (runTarget, index) => {
600
+ const startedAtMs = Date.now();
601
+ live[index] = { ...live[index], status: "running", startedAt: new Date(startedAtMs).toISOString() };
602
+ await report(`Running on ${targetLabel(runTarget)}`);
603
+ const context = { ...driverContext, runTarget };
604
+ const laneDriver = runTarget.cloud && deviceDriver ? deviceDriver : activeDriver;
605
+ let result: Awaited<ReturnType<AppDriver["run"]>>;
606
+ let attempts = 0;
607
+ try {
608
+ result = await laneDriver.run(context);
609
+ const recovered = await recoverDriverFailure(laneDriver, context, result, recoveryPolicy);
610
+ result = recovered.result;
611
+ attempts = recovered.attempts;
612
+ } catch (error) {
613
+ result = {
614
+ status: "failed",
615
+ message: error instanceof Error ? error.message : String(error),
616
+ artifacts: [],
617
+ failureCode: "driver_error",
618
+ };
619
+ }
620
+ const analysis = result.status === "failed"
621
+ ? await refineAnalysis(cfg, result, { projectId: cfg.project?.id ?? "default", runId, targetId: runTarget.id })
622
+ : undefined;
623
+ const counts = stepCounts(result);
624
+ live[index] = {
625
+ ...live[index],
626
+ status: result.status,
627
+ completedAt: new Date().toISOString(),
628
+ durationMs: result.durationMs ?? Date.now() - startedAtMs,
629
+ stepsPassed: counts.stepsPassed,
630
+ stepsFailed: counts.stepsFailed,
631
+ ...(result.failureCode ? { failureCode: result.failureCode } : {}),
632
+ message: result.message,
633
+ recoveryAttempts: attempts,
634
+ ...(result.steps ? {
635
+ steps: result.steps.map((step) => ({
636
+ index: step.sequenceNumber,
637
+ status: step.status,
638
+ ...(step.label ? { label: step.label } : {}),
639
+ ...(step.durationMs !== undefined ? { durationMs: step.durationMs } : {}),
640
+ ...(step.status === "failed" && step.message ? { message: step.message } : {}),
641
+ })),
642
+ } : {}),
643
+ ...(analysis ? {
644
+ analysis: {
645
+ title: analysis.title,
646
+ category: analysis.category,
647
+ source: analysis.source,
648
+ ...(analysis.suggestedFix ? { fixSummary: analysis.suggestedFix.summary } : {}),
649
+ },
650
+ } : {}),
651
+ };
652
+ await report(`${targetLabel(runTarget)} ${result.status}`);
653
+ return { target: runTarget, result, attempts };
654
+ },
655
+ );
656
+ driverResult = aggregateTargetResults(outcomes);
657
+ recoveryAttempts = outcomes.reduce((total, outcome) => total + outcome.attempts, 0);
658
+ targetArtifacts = outcomes.flatMap(({ target: runTarget, result }) =>
659
+ result.artifacts.map((artifact) => ({ targetId: runTarget.id, artifact })));
660
+ targetResults = live;
661
+ } else {
662
+ driverResult = await driver.run(driverContext);
663
+ const recovery = await recoverDriverFailure(driver, driverContext, driverResult, recoveryPolicy);
664
+ driverResult = recovery.result;
665
+ recoveryAttempts = recovery.attempts;
666
+ }
667
+ if (stopRecording) {
668
+ // One artifact per chunk: screenrecord stops itself at three minutes,
669
+ // so a longer run is several files rather than a truncated one.
670
+ const videoPaths = await stopRecording();
671
+ if (videoPaths.length > 0) {
672
+ driverResult = {
673
+ ...driverResult,
674
+ artifacts: [
675
+ ...driverResult.artifacts,
676
+ ...videoPaths.map((videoPath) => ({
677
+ path: videoPath,
678
+ contentType: "video/mp4",
679
+ name: path.basename(videoPath),
680
+ })),
681
+ ],
682
+ };
683
+ }
684
+ }
685
+ }
686
+
687
+ if (!targetResults) targetArtifacts = driverResult.artifacts.map((artifact) => ({ artifact }));
688
+ const results: ExecutionResult["results"] = [];
689
+ const driverArtifactKeys: RunManifest["artifacts"] = [];
690
+ let stepsPassed = 0;
691
+ let stepsFailed = 0;
692
+
693
+ // Real per-step data only stands in for the aggregate driver verdict when
694
+ // its cardinality matches the plan's declared steps: journey.steps is an
695
+ // authored abstraction (node_id/action/proofs) while driverResult.steps
696
+ // is Maestro's literal flow-file command sequence, and the two do not
697
+ // share a vocabulary. A length match is the only case where a positional
698
+ // correspondence is defensible; otherwise every step would silently
699
+ // inherit a status attributed to the wrong action.
700
+ const journeySteps = journey.steps || [];
701
+ const perStepResults = driverResult.steps;
702
+ const hasReliableStepCorrespondence =
703
+ !!perStepResults && perStepResults.length === journeySteps.length;
704
+
705
+ for (const step of journeySteps) {
706
+ const nodeStartedAt = new Date().toISOString();
707
+ const nodeId = step.node_id || `step:${results.length + 1}`;
708
+ await options.onProgress?.({
709
+ currentStep: step.action || "unknown",
710
+ stepsExecuted: results.length,
711
+ stepsPassed,
712
+ stepsFailed,
713
+ });
714
+ let screenshot: JourneyScreenshotEvidence | undefined;
715
+ if (options.captureScreenshot) {
716
+ await options.onProgress?.({
717
+ currentStep: `Capturing screenshot: ${step.action || nodeId}`,
718
+ stepsExecuted: results.length,
719
+ stepsPassed,
720
+ stepsFailed,
721
+ });
722
+ try {
723
+ const bytes = await options.captureScreenshot(nodeId, step.action || "unknown");
724
+ const artifactKey = journeyScreenshotKey(cfg.project?.id ?? "default", journey.id, nodeId);
725
+ await store.put(artifactKey, bytes, "image/png");
726
+ screenshot = {
727
+ source: "runtime",
728
+ status: "captured",
729
+ artifactKey,
730
+ contentType: "image/png",
731
+ capturedAt: new Date().toISOString(),
732
+ };
733
+ } catch (error) {
734
+ screenshot = {
735
+ source: "runtime",
736
+ status: "blocked",
737
+ reason: error instanceof Error ? error.message : "Runtime screenshot capture failed",
738
+ };
739
+ }
740
+ }
741
+ const stepIndex = results.length;
742
+ const observedStep = hasReliableStepCorrespondence ? perStepResults![stepIndex] : undefined;
743
+ const outcome = observedStep
744
+ ? observedStep.status === "passed"
745
+ ? "PASS"
746
+ : observedStep.status === "failed"
747
+ ? "FAIL"
748
+ : "UNKNOWN"
749
+ : driverResult.status === "passed"
750
+ ? "PASS"
751
+ : driverResult.status === "failed"
752
+ ? "FAIL"
753
+ : "UNKNOWN";
754
+ const nodeStatus: NodeExecutionResult["status"] =
755
+ outcome === "PASS" ? "success" : outcome === "FAIL" ? "failed" : "blocked";
756
+ const reason: NodeExecutionResult["reason"] = observedStep
757
+ ? {
758
+ code: outcome === "PASS" ? "verified" : outcome === "FAIL" ? "execution_failed" : "verification_blocked",
759
+ message: observedStep.message ?? `Observed driver step status: ${observedStep.status}.`,
760
+ details: { driver: driverId, platform, durationMs: observedStep.durationMs },
761
+ }
762
+ : outcome === "PASS"
763
+ ? {
764
+ code: "aggregate_only",
765
+ message: "The overall driver run passed; this step was not independently verified.",
766
+ }
767
+ : outcome === "FAIL"
768
+ ? {
769
+ code: "aggregate_only",
770
+ message: "The overall driver run failed; this step was not independently verified.",
771
+ }
772
+ : {
773
+ code: driverResult.failureCode || "verification_blocked",
774
+ message: driverResult.message,
775
+ details: { driver: driverId, platform, exitCode: driverResult.exitCode },
776
+ };
777
+ const node: NodeExecutionResult = {
778
+ nodeId,
779
+ action: step.action || "unknown",
780
+ status: nodeStatus,
781
+ reason,
782
+ proofs: step.proofs || [],
783
+ startedAt: observedStep?.timestamp ?? nodeStartedAt,
784
+ completedAt: observedStep?.timestamp ?? new Date().toISOString(),
785
+ screenshot,
786
+ };
787
+ results.push({
788
+ nodeId,
789
+ step: step.action || "unknown",
790
+ action: step.action || "unknown",
791
+ outcome,
792
+ proofs: step.proofs || [],
793
+ status: node.status,
794
+ reason: node.reason,
795
+ startedAt: node.startedAt,
796
+ completedAt: node.completedAt,
797
+ screenshot: node.screenshot,
798
+ });
799
+ await options.onProgress?.({
800
+ currentStep: step.action || "unknown",
801
+ stepsExecuted: results.length,
802
+ stepsPassed,
803
+ stepsFailed,
804
+ node,
805
+ });
806
+
807
+ if (outcome === "PASS") stepsPassed++;
808
+ else if (outcome === "FAIL") stepsFailed++;
809
+ }
810
+
811
+ // Without a positional match every plan step inherits the aggregate
812
+ // verdict, so one failed command would read as "11 failed". The driver's
813
+ // own step list is the honest count whenever it exists.
814
+ if (!hasReliableStepCorrespondence && perStepResults?.length) {
815
+ ({ stepsPassed, stepsFailed } = stepCounts(driverResult));
816
+ }
817
+
818
+ for (const { targetId, artifact } of targetArtifacts) {
819
+ try {
820
+ const contents = await fs.promises.readFile(artifact.path);
821
+ const name = artifact.name || path.basename(artifact.path);
822
+ const artifactKey = runKey(
823
+ cfg.project?.id ?? "default",
824
+ runId,
825
+ targetId ? `artifacts/${targetId}/${name}` : `artifacts/${name}`,
826
+ );
827
+ await store.put(
828
+ artifactKey,
829
+ contents,
830
+ artifact.contentType || "application/octet-stream",
831
+ );
832
+ driverArtifactKeys.push({ key: artifactKey, contentType: artifact.contentType });
833
+ const entry = targetId ? targetResults?.find((candidate) => candidate.targetId === targetId) : undefined;
834
+ if (entry) {
835
+ entry.artifactKeys.push(artifactKey);
836
+ if (name === "playwright-final.png" || name === "playwright-failure.png") entry.finalScreenshotKey = artifactKey;
837
+ if (name === "playwright.log") entry.logKey = artifactKey;
838
+ if (name === "playwright-trace.zip") entry.traceKey = artifactKey;
839
+ if (name === "analysis.json") entry.analysisKey = artifactKey;
840
+ // Playwright's web recording always uses this name; a device's screen
841
+ // recording (iOS/Android) is named after its own chunked video file
842
+ // instead, so match by content type for that case. First video wins:
843
+ // a device recording that screenrecord split into several chunks
844
+ // keeps its later parts reachable via artifactKeys, but the run
845
+ // detail view plays the first one.
846
+ if ((name === "playwright-run.webm" || artifact.contentType?.startsWith("video/")) && !entry.videoKey) entry.videoKey = artifactKey;
847
+ }
848
+ } catch {
849
+ // Driver output remains diagnostic even if an optional artifact cannot be uploaded.
850
+ }
851
+ }
852
+
853
+ // Save execution results
854
+ const completedAt = new Date().toISOString();
855
+ const resultsKey = runKey(cfg.project?.id ?? "default", runId, "results.json");
856
+ await store.put(resultsKey, json({
857
+ runId,
858
+ journeyId: journey.id,
859
+ driver: driverId,
860
+ platform,
861
+ failureCode: driverResult.failureCode,
862
+ recoveryAttempts,
863
+ planId,
864
+ timestamp: completedAt,
865
+ ...(targetResults ? { baseUrl: options.baseUrl, targets: targetResults } : {}),
866
+ results,
867
+ }), "application/json");
868
+ const manifest: RunManifest = {
869
+ version: 1,
870
+ projectId: cfg.project?.id ?? "default",
871
+ tenantId: cfg.project?.tenant,
872
+ environmentId: cfg.project?.environment ?? cfg.runtime.env,
873
+ runId,
874
+ planId,
875
+ journeyId: journey.id,
876
+ status: stepsFailed === 0 && !results.some((result) => result.status === "blocked") ? "success" : "partial",
877
+ startedAt,
878
+ completedAt,
879
+ stepsExecuted: results.length,
880
+ stepsPassed,
881
+ stepsFailed,
882
+ artifacts: [{ key: resultsKey, contentType: "application/json" }, ...driverArtifactKeys],
883
+ ...(targetResults ? { targets: targetResults } : {}),
884
+ };
885
+ const manifestKey = runKey(manifest.projectId, runId, "manifest.json");
886
+ await store.put(manifestKey, json(manifest), "application/json");
887
+
888
+ return {
889
+ status: stepsFailed === 0 && !results.some((result) => result.status === "blocked") ? "success" : "partial",
890
+ runId,
891
+ journeyId: journey.id,
892
+ stepsExecuted: results.length,
893
+ stepsPassed,
894
+ stepsFailed,
895
+ driver: driverId,
896
+ platform,
897
+ recoveryAttempts,
898
+ failureCode: driverResult.failureCode,
899
+ evidencePath: runPrefix(manifest.projectId, runId),
900
+ manifestKey,
901
+ message: targetResults
902
+ ? `Execution completed on ${targetResults.length} target(s). ${driverResult.message}`
903
+ : `Execution completed. ${stepsPassed} passed, ${stepsFailed} failed.`,
904
+ ...(targetResults ? { targetResults } : {}),
905
+ durationMs: Date.parse(completedAt) - Date.parse(startedAt),
906
+ ...(fixApplied ? { fixApplied } : {}),
907
+ results,
908
+ };
909
+ } catch (error: any) {
910
+ return {
911
+ status: "error",
912
+ runId: `error_${Date.now()}`,
913
+ journeyId: options.journeyId || "unknown",
914
+ stepsExecuted: 0,
915
+ stepsPassed: 0,
916
+ stepsFailed: 0,
917
+ error: error.message || "Execution failed",
918
+ };
919
+ }
920
+ }
921
+