@bugmole/cli 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. package/.bugmole.env.example +20 -0
  2. package/LICENSE +7 -0
  3. package/README.md +293 -0
  4. package/TESTING.md +117 -0
  5. package/bugmole.config.yaml +39 -0
  6. package/package.json +85 -0
  7. package/scripts/billing/paypal-setup.mjs +121 -0
  8. package/scripts/bugmole-continue.ts +318 -0
  9. package/scripts/bugmole-init.ts +188 -0
  10. package/scripts/bugmole.cjs +17 -0
  11. package/scripts/bugmole.test.ts +344 -0
  12. package/scripts/bugmole.ts +657 -0
  13. package/scripts/ensure-maestro.cjs +79 -0
  14. package/scripts/ios-tunnel-keeper.sh +45 -0
  15. package/scripts/ios-wda-keeper.sh +66 -0
  16. package/scripts/sync-plan-catalog.d.mts +3 -0
  17. package/scripts/sync-plan-catalog.mjs +16 -0
  18. package/scripts/ui-parity-diff.py +65 -0
  19. package/scripts/ui-parity-requirements.txt +1 -0
  20. package/scripts/verify-manage-to-plans.mts +194 -0
  21. package/spec/app-ui-audit.schema.json +176 -0
  22. package/spec/blockers.yaml +79 -0
  23. package/spec/bugs.index.json +42 -0
  24. package/spec/design-dna.schema.json +38 -0
  25. package/spec/domain_rules.yaml +24 -0
  26. package/spec/flows/flow_manage-to-plans-64c3c61b.yaml +8 -0
  27. package/spec/flows/flow_screen-to-evidence-6c3ab309.yaml +7 -0
  28. package/spec/journey_graph.yaml +126 -0
  29. package/spec/journeys.graph.json +2618 -0
  30. package/spec/plans/dashboard-smoke.flow.yaml +20 -0
  31. package/spec/plans/example.flow.yaml +99 -0
  32. package/spec/plans/flow_api-keys-to-logout-7396cd5d.flow.yaml +33 -0
  33. package/spec/plans/flow_api-keys-to-screen-157e10a6.flow.yaml +33 -0
  34. package/spec/plans/flow_manage-to-audits-176c1eb8.flow.yaml +33 -0
  35. package/spec/plans/flow_manage-to-blockers-633282c8.flow.yaml +112 -0
  36. package/spec/plans/flow_manage-to-devices-ad57e202.flow.yaml +33 -0
  37. package/spec/plans/flow_manage-to-logout-86c54656.flow.yaml +33 -0
  38. package/spec/plans/flow_manage-to-manage-resource-action-6325f99a.flow.yaml +185 -0
  39. package/spec/plans/flow_manage-to-parity-issues-01fa2579.flow.yaml +34 -0
  40. package/spec/plans/flow_manage-to-plans-64c3c61b.flow.yaml +123 -0
  41. package/spec/plans/flow_manage-to-screen-6857f915.flow.yaml +106 -0
  42. package/spec/plans/flow_manage-to-tasks-c5791c95.flow.yaml +33 -0
  43. package/spec/plans/flow_profile-to-screen-6045aa3d.flow.yaml +33 -0
  44. package/spec/plans/flow_register-to-api-auth-login-460a9cc8.flow.yaml +34 -0
  45. package/spec/plans/flow_screen-to-audits-07666356.flow.yaml +33 -0
  46. package/spec/plans/flow_screen-to-blockers-091d258b.flow.yaml +33 -0
  47. package/spec/plans/flow_screen-to-devices-cad72f75.flow.yaml +33 -0
  48. package/spec/plans/flow_screen-to-evidence-6c3ab309.flow.yaml +126 -0
  49. package/spec/plans/flow_screen-to-parity-issues-2218e3ad.flow.yaml +34 -0
  50. package/spec/plans/flow_screen-to-plans-8232a94f.flow.yaml +33 -0
  51. package/spec/plans/flow_screen-to-tasks-540671d1.flow.yaml +33 -0
  52. package/spec/plans/login.flow.yaml +22 -0
  53. package/spec/plans/owner-operations.flow.yaml +20 -0
  54. package/spec/project_config.yaml +55 -0
  55. package/spec/roles.yaml +30 -0
  56. package/spec/schema.md +394 -0
  57. package/spec/test-case-results.schema.json +62 -0
  58. package/spec/test-cases.schema.json +85 -0
  59. package/spec/ui-parity-audit.schema.json +194 -0
  60. package/spec/ui-reverse-engineering.schema.json +94 -0
  61. package/src/billing/plan-catalog.test.ts +46 -0
  62. package/src/billing/plan-catalog.ts +199 -0
  63. package/src/integrations/aws-sigv4.test.ts +42 -0
  64. package/src/integrations/aws-sigv4.ts +72 -0
  65. package/src/integrations/device-farm.ts +155 -0
  66. package/src/integrations/github-app.test.ts +57 -0
  67. package/src/integrations/github-app.ts +143 -0
  68. package/src/integrations/gitlab.ts +81 -0
  69. package/src/integrations/temp-email.test.ts +123 -0
  70. package/src/integrations/temp-email.ts +175 -0
  71. package/src/integrations/testflight-feedback.test.ts +51 -0
  72. package/src/integrations/testflight-feedback.ts +173 -0
  73. package/src/integrations/webdriver-client.ts +131 -0
  74. package/src/mcp/server.test.ts +1220 -0
  75. package/src/mcp/server.ts +3064 -0
  76. package/src/mcp/write-test-cases.test.ts +287 -0
  77. package/src/registry/api-key-client.ts +39 -0
  78. package/src/registry/control-plane-client.ts +212 -0
  79. package/src/registry/migrations/0001_registry.sql +47 -0
  80. package/src/registry/migrations/0002_device_authorizations.sql +23 -0
  81. package/src/registry/migrations/0003_project_environments.sql +25 -0
  82. package/src/registry/migrations/0004_device_authorization_email.sql +1 -0
  83. package/src/registry/migrations/0005_testing_control_plane.sql +55 -0
  84. package/src/registry/migrations/0006_workspaces.sql +36 -0
  85. package/src/registry/migrations/0007_project_apps.sql +26 -0
  86. package/src/registry/migrations/0008_agent_tasks.sql +30 -0
  87. package/src/registry/migrations/0009_journey_revisions.sql +17 -0
  88. package/src/registry/migrations/0010_agent_task_journey.sql +2 -0
  89. package/src/registry/migrations/0011_run_journey_revision.sql +1 -0
  90. package/src/registry/migrations/0012_canonical_flow_execution.sql +10 -0
  91. package/src/registry/migrations/0013_device_sessions.sql +22 -0
  92. package/src/registry/migrations/0014_agent_task_step.sql +1 -0
  93. package/src/registry/migrations/0015_agent_task_attempts.sql +6 -0
  94. package/src/registry/migrations/0016_github_issue_tracker.sql +54 -0
  95. package/src/registry/migrations/0017_run_targets.sql +7 -0
  96. package/src/registry/migrations/0018_run_fix_from.sql +4 -0
  97. package/src/registry/migrations/0019_workspace_flags.sql +9 -0
  98. package/src/registry/migrations/0020_orgs.sql +40 -0
  99. package/src/registry/migrations/0021_billing_core.sql +58 -0
  100. package/src/registry/migrations/0022_cloud_runners.sql +19 -0
  101. package/src/registry/migrations/0023_signup.sql +4 -0
  102. package/src/registry/migrations/0024_billing.sql +67 -0
  103. package/src/registry/migrations/0025_notifications.sql +47 -0
  104. package/src/registry/migrations/0026_repo_bindings.sql +28 -0
  105. package/src/registry/migrations/0027_feedback.sql +29 -0
  106. package/src/registry/migrations/0028_devices.sql +48 -0
  107. package/src/registry/migrations/0029_sso.sql +31 -0
  108. package/src/registry/migrations/0030_workspace_domains.sql +18 -0
  109. package/src/registry/migrations/0031_gitlab_and_teams.sql +45 -0
  110. package/src/registry/migrations/0032_personas.sql +15 -0
  111. package/src/registry/migrations/0033_bugmole_rename.sql +11 -0
  112. package/src/registry/task-scheduling.test.ts +100 -0
  113. package/src/registry/task-scheduling.ts +80 -0
  114. package/src/registry-worker/ai/platform-model.ts +77 -0
  115. package/src/registry-worker/ai/routes.test.ts +88 -0
  116. package/src/registry-worker/ai/routes.ts +90 -0
  117. package/src/registry-worker/artifacts.test.ts +98 -0
  118. package/src/registry-worker/artifacts.ts +85 -0
  119. package/src/registry-worker/billing/billing-core.test.ts +175 -0
  120. package/src/registry-worker/billing/checkout-routes.ts +209 -0
  121. package/src/registry-worker/billing/enforcement.ts +69 -0
  122. package/src/registry-worker/billing/entitlements.ts +108 -0
  123. package/src/registry-worker/billing/ledger.ts +186 -0
  124. package/src/registry-worker/billing/paypal/api.ts +259 -0
  125. package/src/registry-worker/billing/paypal/client.ts +91 -0
  126. package/src/registry-worker/billing/paypal/provider.ts +143 -0
  127. package/src/registry-worker/billing/paypal.test.ts +466 -0
  128. package/src/registry-worker/billing/provider.ts +114 -0
  129. package/src/registry-worker/billing/routes.ts +66 -0
  130. package/src/registry-worker/billing/subscriptions.ts +780 -0
  131. package/src/registry-worker/billing/thresholds.ts +107 -0
  132. package/src/registry-worker/core.ts +308 -0
  133. package/src/registry-worker/devices/devices.test.ts +185 -0
  134. package/src/registry-worker/devices/policy.ts +71 -0
  135. package/src/registry-worker/devices/routes.ts +453 -0
  136. package/src/registry-worker/domains/domains.test.ts +210 -0
  137. package/src/registry-worker/domains/routes.ts +139 -0
  138. package/src/registry-worker/domains.ts +88 -0
  139. package/src/registry-worker/email/sender.ts +75 -0
  140. package/src/registry-worker/env.d.ts +14716 -0
  141. package/src/registry-worker/features.ts +20 -0
  142. package/src/registry-worker/feedback/feedback.test.ts +230 -0
  143. package/src/registry-worker/feedback/format.ts +148 -0
  144. package/src/registry-worker/feedback/routes.ts +386 -0
  145. package/src/registry-worker/flags.ts +39 -0
  146. package/src/registry-worker/github/checks.test.ts +177 -0
  147. package/src/registry-worker/github/checks.ts +374 -0
  148. package/src/registry-worker/gitlab/checks.test.ts +141 -0
  149. package/src/registry-worker/gitlab/checks.ts +349 -0
  150. package/src/registry-worker/hooks.ts +54 -0
  151. package/src/registry-worker/index.ts +2077 -0
  152. package/src/registry-worker/jobs/index.ts +29 -0
  153. package/src/registry-worker/jobs/retention.ts +68 -0
  154. package/src/registry-worker/mcp/mcp.test.ts +355 -0
  155. package/src/registry-worker/mcp/routes.ts +215 -0
  156. package/src/registry-worker/mcp/token.ts +126 -0
  157. package/src/registry-worker/mcp/tools.ts +563 -0
  158. package/src/registry-worker/notifications/alerts.ts +212 -0
  159. package/src/registry-worker/notifications/notifications.test.ts +298 -0
  160. package/src/registry-worker/notifications/outbox.ts +83 -0
  161. package/src/registry-worker/notifications/routes.ts +280 -0
  162. package/src/registry-worker/notifications/secrets.ts +49 -0
  163. package/src/registry-worker/notifications/slack.ts +96 -0
  164. package/src/registry-worker/notifications/teams.ts +46 -0
  165. package/src/registry-worker/org/audit.ts +116 -0
  166. package/src/registry-worker/org/routes.test.ts +163 -0
  167. package/src/registry-worker/org/routes.ts +302 -0
  168. package/src/registry-worker/personas/personas.test.ts +78 -0
  169. package/src/registry-worker/personas/routes.ts +100 -0
  170. package/src/registry-worker/repo-triggers.ts +20 -0
  171. package/src/registry-worker/routes/index.ts +74 -0
  172. package/src/registry-worker/run-events.ts +24 -0
  173. package/src/registry-worker/runner/dispatch.ts +219 -0
  174. package/src/registry-worker/runner/jobs.ts +43 -0
  175. package/src/registry-worker/runner/metering.ts +82 -0
  176. package/src/registry-worker/runner/policy.ts +59 -0
  177. package/src/registry-worker/runner/routes.ts +171 -0
  178. package/src/registry-worker/runner/runner.test.ts +358 -0
  179. package/src/registry-worker/runner/tokens.ts +93 -0
  180. package/src/registry-worker/runs.test.ts +60 -0
  181. package/src/registry-worker/signup/policy.ts +57 -0
  182. package/src/registry-worker/signup/routes.ts +106 -0
  183. package/src/registry-worker/signup/signup.test.ts +81 -0
  184. package/src/registry-worker/sso/aegis.ts +141 -0
  185. package/src/registry-worker/sso/membership.ts +157 -0
  186. package/src/registry-worker/sso/routes.ts +458 -0
  187. package/src/registry-worker/sso/sso.test.ts +344 -0
  188. package/src/registry-worker/testing/d1-shim.ts +180 -0
  189. package/src/registry-worker/testing/harness.ts +137 -0
  190. package/src/runner-worker/index.ts +108 -0
  191. package/src/runtime/ai-analysis.ts +97 -0
  192. package/src/runtime/ai-exploration.test.ts +32 -0
  193. package/src/runtime/ai-exploration.ts +69 -0
  194. package/src/runtime/ai-repair.ts +74 -0
  195. package/src/runtime/ai-work.test.ts +99 -0
  196. package/src/runtime/android-screen-record.test.ts +75 -0
  197. package/src/runtime/android-screen-record.ts +192 -0
  198. package/src/runtime/app-understanding.test.ts +123 -0
  199. package/src/runtime/app-understanding.ts +201 -0
  200. package/src/runtime/appium-driver.test.ts +179 -0
  201. package/src/runtime/appium-driver.ts +295 -0
  202. package/src/runtime/blocker-resolution.test.ts +113 -0
  203. package/src/runtime/blocker-resolution.ts +111 -0
  204. package/src/runtime/browser-matrix.integration.test.ts +212 -0
  205. package/src/runtime/browser-matrix.test.ts +143 -0
  206. package/src/runtime/browser-matrix.ts +200 -0
  207. package/src/runtime/canonical-flow.test.ts +52 -0
  208. package/src/runtime/config-validate.ts +185 -0
  209. package/src/runtime/continuous-execution.ts +291 -0
  210. package/src/runtime/cursor-applescript.ts +573 -0
  211. package/src/runtime/cursor-cli-driver.test.ts +78 -0
  212. package/src/runtime/cursor-cli-driver.ts +156 -0
  213. package/src/runtime/cursor-driver-example.ts +117 -0
  214. package/src/runtime/cursor-driver-index.ts +65 -0
  215. package/src/runtime/cursor-driver-init.ts +277 -0
  216. package/src/runtime/cursor-driver-run.test.ts +15 -0
  217. package/src/runtime/cursor-driver-run.ts +323 -0
  218. package/src/runtime/cursor-driver.ts +332 -0
  219. package/src/runtime/cursor-llm-example.ts +90 -0
  220. package/src/runtime/cursor-llm.ts +206 -0
  221. package/src/runtime/cursor-mcp-monitor.ts +386 -0
  222. package/src/runtime/device-clouds/browserstack.ts +73 -0
  223. package/src/runtime/device-clouds/device-farm.ts +52 -0
  224. package/src/runtime/device-clouds/index.ts +92 -0
  225. package/src/runtime/device-clouds/kobiton.ts +70 -0
  226. package/src/runtime/device-clouds/targets.ts +44 -0
  227. package/src/runtime/device-clouds/types.ts +62 -0
  228. package/src/runtime/diff-proposal.ts +84 -0
  229. package/src/runtime/discovery-task.test.ts +29 -0
  230. package/src/runtime/discovery-task.ts +284 -0
  231. package/src/runtime/driver-recovery.ts +69 -0
  232. package/src/runtime/driver.ts +79 -0
  233. package/src/runtime/environment.test.ts +104 -0
  234. package/src/runtime/environment.ts +137 -0
  235. package/src/runtime/executor.test.ts +509 -0
  236. package/src/runtime/executor.ts +921 -0
  237. package/src/runtime/explorer.test.ts +101 -0
  238. package/src/runtime/explorer.ts +1013 -0
  239. package/src/runtime/failure-analysis.test.ts +111 -0
  240. package/src/runtime/failure-analysis.ts +272 -0
  241. package/src/runtime/fixtures/fake-maestro.sh +36 -0
  242. package/src/runtime/flow-language.test.ts +268 -0
  243. package/src/runtime/flow-language.ts +414 -0
  244. package/src/runtime/init-wizard.ts +354 -0
  245. package/src/runtime/ios-screen-record.test.ts +68 -0
  246. package/src/runtime/ios-screen-record.ts +155 -0
  247. package/src/runtime/journey-editor.ts +452 -0
  248. package/src/runtime/journey-evidence.test.ts +161 -0
  249. package/src/runtime/journey-evidence.ts +180 -0
  250. package/src/runtime/journey-graph.test.ts +257 -0
  251. package/src/runtime/journey-graph.ts +170 -0
  252. package/src/runtime/legacy-names.ts +32 -0
  253. package/src/runtime/llm-example.ts +105 -0
  254. package/src/runtime/llm.ts +527 -0
  255. package/src/runtime/local-browser.test.ts +45 -0
  256. package/src/runtime/local-browser.ts +48 -0
  257. package/src/runtime/local-registry-stub.test.ts +325 -0
  258. package/src/runtime/local-registry-stub.ts +803 -0
  259. package/src/runtime/maestro-driver.test.ts +84 -0
  260. package/src/runtime/maestro-driver.ts +209 -0
  261. package/src/runtime/mole-voice.ts +21 -0
  262. package/src/runtime/nav-crawl.test.ts +100 -0
  263. package/src/runtime/nav-crawl.ts +153 -0
  264. package/src/runtime/pipeline.test.ts +405 -0
  265. package/src/runtime/pipeline.ts +833 -0
  266. package/src/runtime/planner.test.ts +37 -0
  267. package/src/runtime/planner.ts +274 -0
  268. package/src/runtime/platform-ai.ts +76 -0
  269. package/src/runtime/playwright-driver.test.ts +93 -0
  270. package/src/runtime/playwright-driver.ts +620 -0
  271. package/src/runtime/project-spec.ts +140 -0
  272. package/src/runtime/record-run-verdicts.ts +68 -0
  273. package/src/runtime/reporter.test.ts +56 -0
  274. package/src/runtime/reporter.ts +158 -0
  275. package/src/runtime/reset.test.ts +44 -0
  276. package/src/runtime/reset.ts +61 -0
  277. package/src/runtime/reviewer.test.ts +73 -0
  278. package/src/runtime/reviewer.ts +158 -0
  279. package/src/runtime/run-job.ts +136 -0
  280. package/src/runtime/run-once.test.ts +207 -0
  281. package/src/runtime/run-once.ts +168 -0
  282. package/src/runtime/run.ts +132 -0
  283. package/src/runtime/screen-recording.ts +34 -0
  284. package/src/runtime/serve-gateway.test.ts +74 -0
  285. package/src/runtime/serve-gateway.ts +164 -0
  286. package/src/runtime/serve-worker.test.ts +23 -0
  287. package/src/runtime/serve-worker.ts +278 -0
  288. package/src/runtime/site-discovery.test.ts +168 -0
  289. package/src/runtime/site-discovery.ts +308 -0
  290. package/src/runtime/target-runner.ts +144 -0
  291. package/src/runtime/test-case-verdicts.test.ts +94 -0
  292. package/src/runtime/test-case-verdicts.ts +120 -0
  293. package/src/runtime/ui-reverse-engineering/coordinator.test.ts +59 -0
  294. package/src/runtime/ui-reverse-engineering/coordinator.ts +391 -0
  295. package/src/runtime/ui-reverse-engineering/types.ts +197 -0
  296. package/src/runtime/web-suite.test.ts +97 -0
  297. package/src/runtime/web-suite.ts +176 -0
  298. package/src/storage/create-object-store.ts +144 -0
  299. package/src/storage/keys.ts +34 -0
  300. package/src/storage/local-artifact-server.test.ts +314 -0
  301. package/src/storage/local-artifact-server.ts +357 -0
  302. package/src/storage/object-store.test.ts +28 -0
  303. package/src/storage/object-store.ts +101 -0
  304. package/src/storage/registry-object-store.ts +88 -0
  305. package/src/storage/remote-object-store.ts +104 -0
  306. package/src/storage/storage-directory.test.ts +43 -0
  307. package/src/storage/storage-directory.ts +24 -0
  308. package/tsconfig.json +24 -0
@@ -0,0 +1,833 @@
1
+ /**
2
+ * Autonomous QA pipeline.
3
+ *
4
+ * Pulling "Explore this app" is meant to start the platform working on its
5
+ * own — explore, then assume flows, then assume test cases, then plan —
6
+ * and keep going until it is done or something says stop. Previously the
7
+ * worker ran exactly one task and stopped: nothing enqueued follow-on
8
+ * work, and the agent guidance only ever said "do this one task and
9
+ * report". A real run finished exploration and then sat idle forever.
10
+ *
11
+ * The agent may enqueue its own next stage, but progress cannot depend on
12
+ * that: the agent demonstrably died mid-pipeline (its CLI was killed by a
13
+ * timeout) and nothing recovered. So the worker also reconciles. Stage
14
+ * state is DERIVED from real artifacts on every pass rather than stored as
15
+ * a separate state machine, which means the two paths converge instead of
16
+ * fighting, and a half-finished pipeline repairs itself rather than
17
+ * drifting out of sync with what is actually on disk.
18
+ */
19
+
20
+ import fs from "node:fs";
21
+ import path from "node:path";
22
+ import YAML from "yaml";
23
+ import { createProjectObjectStore } from "../storage/create-object-store.js";
24
+ import { projectPrefix } from "../storage/keys.js";
25
+ import { MAX_TASK_ATTEMPTS } from "../registry/task-scheduling.js";
26
+ import { driverForPlatform } from "./planner.js";
27
+
28
+ export type PipelineStage = "explore" | "roles" | "audit" | "testcases" | "plan" | "run";
29
+
30
+ export type PipelineStatus = "running" | "paused" | "stopped" | "complete";
31
+
32
+ export type PipelineJourney = {
33
+ id: string;
34
+ actor?: string;
35
+ revision?: number;
36
+ nodeIds?: string[];
37
+ stability?: string;
38
+ linkedToGraph?: boolean;
39
+ };
40
+
41
+ export type PipelineTask = {
42
+ status: string;
43
+ idempotencyKey?: string;
44
+ completionSummary?: string | null;
45
+ title?: string;
46
+ attempts?: number;
47
+ errorMessage?: string | null;
48
+ };
49
+
50
+ export type PipelineInputs = {
51
+ projectId: string;
52
+ journeys: PipelineJourney[];
53
+ /** Actor names defined in roles.yaml. */
54
+ knownActors: string[];
55
+ /** Journey ids that already have a written test-case set. */
56
+ caseSetJourneyIds: string[];
57
+ /** Journey ids that already have an App Audit artifact. */
58
+ auditedJourneyIds?: string[];
59
+ /** Journey ids that already have a published plan. */
60
+ plannedJourneyIds: string[];
61
+ /** Journey ids that already have a run (of any status). */
62
+ ranJourneyIds?: string[];
63
+ /**
64
+ * Every recorded run with its status.
65
+ *
66
+ * Needed because "has a run" is not enough to decide what to do next: a run
67
+ * still in flight must not be duplicated, while a run that already finished
68
+ * must not block a fresh attempt. Keyed only on ids, those two cases look
69
+ * identical — which is how a finished run ended up being handed back
70
+ * forever under a stable idempotency key.
71
+ */
72
+ runs?: Array<{ journeyId?: string; status?: string }>;
73
+ /**
74
+ * Journey ids where every written test case carries a recorded verdict.
75
+ *
76
+ * A run artifact only proves a run happened. This proves the run actually
77
+ * judged the cases it was supposed to, which is the difference between
78
+ * "executed" and "covered".
79
+ */
80
+ verdictJourneyIds?: string[];
81
+ /** Published plans, used to start a run for a planned journey. */
82
+ publishedPlans?: Array<{ planId: string; journeyId?: string }>;
83
+ /** Environment to run against, when one is configured. */
84
+ environmentId?: string;
85
+ /** Current agent tasks for the project. */
86
+ tasks: PipelineTask[];
87
+ /** Blocker categories declared hard_stop in spec/blockers.yaml. */
88
+ hardStopCategories: string[];
89
+ /** Blocker categories actually encountered and recorded. */
90
+ recordedBlockerCategories: string[];
91
+ /** Set when a stage is waiting on a human answer. */
92
+ pendingClarification?: string;
93
+ };
94
+
95
+ export type PipelineNextTask = {
96
+ stage: PipelineStage;
97
+ journeyId?: string;
98
+ title: string;
99
+ description: string;
100
+ taskType: "code" | "ui" | "investigation";
101
+ idempotencyKey: string;
102
+ };
103
+
104
+ /**
105
+ * The run stage produces a run, not an agent task: executing a published
106
+ * plan is the worker's own job (processRun), and it is what turns a test
107
+ * case assumption into evidence. Without this the pipeline stopped at
108
+ * "planned" and every case stayed unproven.
109
+ */
110
+ export type PipelineNextRun = {
111
+ stage: "run";
112
+ journeyId: string;
113
+ planId: string;
114
+ idempotencyKey: string;
115
+ environmentId?: string;
116
+ };
117
+
118
+ export type PipelineState = {
119
+ status: PipelineStatus;
120
+ reason: string;
121
+ next?: PipelineNextTask;
122
+ nextRun?: PipelineNextRun;
123
+ stages: Array<{ stage: PipelineStage; satisfied: boolean; detail: string }>;
124
+ /**
125
+ * Artifacts written against flows that no longer exist. Surfaced rather
126
+ * than quietly ignored: they are the visible trace of the graph having
127
+ * been regenerated, and they are why a stage can look further along than
128
+ * it is.
129
+ */
130
+ orphanedArtifacts: Array<{ kind: string; journeyId: string }>;
131
+ /** Stages that gave up and need a human answer before work can continue. */
132
+ blockedTasks?: Array<{ title: string; idempotencyKey?: string; attempts: number; reason?: string }>;
133
+ };
134
+
135
+ export const PIPELINE_KEY_PREFIX = "pipeline";
136
+
137
+ /** Deterministic per-stage key, so the agent's own enqueue and the
138
+ * reconciler's converge on one task instead of creating duplicates. */
139
+ export function pipelineIdempotencyKey(
140
+ projectId: string,
141
+ stage: PipelineStage,
142
+ journeyId?: string,
143
+ ): string {
144
+ return journeyId
145
+ ? `${PIPELINE_KEY_PREFIX}:${projectId}:${stage}:${journeyId}`
146
+ : `${PIPELINE_KEY_PREFIX}:${projectId}:${stage}`;
147
+ }
148
+
149
+ export function isPipelineTask(task: PipelineTask): boolean {
150
+ return typeof task.idempotencyKey === "string"
151
+ && task.idempotencyKey.startsWith(`${PIPELINE_KEY_PREFIX}:`);
152
+ }
153
+
154
+ /** A journey the graph considers real enough to build on. */
155
+ function isCanonical(journey: PipelineJourney): boolean {
156
+ return journey.stability === "stable" && journey.linkedToGraph === true;
157
+ }
158
+
159
+ function activeTask(tasks: PipelineTask[]): PipelineTask | undefined {
160
+ return tasks.find((task) => task.status === "queued" || task.status === "running");
161
+ }
162
+
163
+ /**
164
+ * Decide what the pipeline should do next.
165
+ *
166
+ * Pure on purpose: every stop rule is a branch here, so each can be proven
167
+ * in isolation rather than only observable by running a live agent.
168
+ */
169
+ export function derivePipelineState(inputs: PipelineInputs): PipelineState {
170
+ const {
171
+ projectId, journeys, knownActors, caseSetJourneyIds,
172
+ plannedJourneyIds, tasks, hardStopCategories, recordedBlockerCategories,
173
+ } = inputs;
174
+
175
+ const withNodes = journeys.filter((journey) => (journey.nodeIds?.length ?? 0) > 0);
176
+ const canonical = withNodes.filter(isCanonical);
177
+ // Artifacts outlive the flows they describe: regenerating the graph
178
+ // gives journeys new ids, orphaning every case set, audit and plan
179
+ // written against the old ones. Counting those raw made the pipeline
180
+ // report "4/7 flows have test cases" when none of the live flows had
181
+ // any, and mark planning satisfied while nothing was planned — so it
182
+ // would have skipped the stage entirely. Only artifacts belonging to a
183
+ // flow that still exists count.
184
+ const canonicalIds = new Set(canonical.map((journey) => journey.id));
185
+ const belongsToLiveFlow = (id: string) => canonicalIds.has(id);
186
+ const caseSets = new Set(caseSetJourneyIds.filter(belongsToLiveFlow));
187
+ const planned = new Set(plannedJourneyIds.filter(belongsToLiveFlow));
188
+ const orphanedArtifacts = [
189
+ ...caseSetJourneyIds.filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "test cases", journeyId: id })),
190
+ ...(inputs.auditedJourneyIds ?? []).filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "audit", journeyId: id })),
191
+ ...plannedJourneyIds.filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "plan", journeyId: id })),
192
+ ];
193
+ const missingActors = [...new Set(
194
+ canonical
195
+ .map((journey) => journey.actor)
196
+ .filter((actor): actor is string => Boolean(actor) && !knownActors.includes(actor!)),
197
+ )];
198
+ const audited = new Set((inputs.auditedJourneyIds ?? []).filter(belongsToLiveFlow));
199
+ const needsAudit = canonical.filter((journey) => !audited.has(journey.id));
200
+ const needsCases = canonical.filter((journey) => !caseSets.has(journey.id));
201
+ const needsPlan = canonical.filter((journey) => caseSets.has(journey.id) && !planned.has(journey.id));
202
+ const ran = new Set((inputs.ranJourneyIds ?? []).filter(belongsToLiveFlow));
203
+ const judged = new Set((inputs.verdictJourneyIds ?? []).filter(belongsToLiveFlow));
204
+ const ranButUnjudged = [...ran].filter((id) => !judged.has(id));
205
+ // The run stage is about planned flows, so only verdicts for flows that
206
+ // are currently planned count toward it. Without the intersection a flow
207
+ // judged under a plan that has since been discarded still counted, and the
208
+ // stage read "20/2 planned flows judged" — more judged than exist.
209
+ const judgedAndPlanned = [...judged].filter((id) => planned.has(id));
210
+ // Runs still in flight, and how many have already finished per flow. A
211
+ // finished run must not satisfy a fresh attempt, and an in-flight one must
212
+ // not be duplicated.
213
+ const TERMINAL_RUN = ["success", "partial", "error", "cancelled"];
214
+ const runsInFlight = new Set(
215
+ (inputs.runs ?? [])
216
+ .filter((run) => run.journeyId && !TERMINAL_RUN.includes(run.status ?? ""))
217
+ .map((run) => run.journeyId!),
218
+ );
219
+ const finishedRunCounts = new Map<string, number>();
220
+ for (const run of inputs.runs ?? []) {
221
+ if (!run.journeyId || !TERMINAL_RUN.includes(run.status ?? "")) continue;
222
+ finishedRunCounts.set(run.journeyId, (finishedRunCounts.get(run.journeyId) ?? 0) + 1);
223
+ }
224
+ // A flow that ran but recorded no verdicts still needs running: the run
225
+ // produced artifacts without judging anything, so re-running it is the only
226
+ // way the cases get a status. Keyed off judged rather than ran so the
227
+ // pipeline cannot declare itself finished over unjudged cases.
228
+ const needsRun = canonical.filter((journey) => planned.has(journey.id) && !judged.has(journey.id));
229
+
230
+ const stages: PipelineState["stages"] = [
231
+ {
232
+ stage: "explore",
233
+ satisfied: withNodes.length > 0,
234
+ detail: `${withNodes.length} discovered flow${withNodes.length === 1 ? "" : "s"}`,
235
+ },
236
+ {
237
+ stage: "roles",
238
+ satisfied: missingActors.length === 0,
239
+ detail: missingActors.length
240
+ ? `undefined actor${missingActors.length === 1 ? "" : "s"}: ${missingActors.join(", ")}`
241
+ : "every flow actor is defined",
242
+ },
243
+ {
244
+ // The audit inventories each screen's controls and runs the component
245
+ // scan that populates the library, so it comes before test cases:
246
+ // knowing what a screen is built from is what lets a case be written
247
+ // against something real.
248
+ stage: "audit",
249
+ satisfied: canonical.length > 0 && needsAudit.length === 0,
250
+ detail: `${audited.size}/${canonical.length} flows audited`,
251
+ },
252
+ {
253
+ stage: "testcases",
254
+ satisfied: canonical.length > 0 && needsCases.length === 0,
255
+ detail: `${caseSets.size}/${canonical.length} flows have test case assumptions`,
256
+ },
257
+ {
258
+ stage: "plan",
259
+ satisfied: canonical.length > 0 && planned.size === canonical.length,
260
+ detail: `${planned.size}/${canonical.length} flows planned`,
261
+ },
262
+ {
263
+ // Until a plan actually runs, every test case assumption is still
264
+ // just an assumption — this is the stage that produces evidence.
265
+ //
266
+ // "Executed" is not the bar. A run artifact only proves a run happened;
267
+ // this stage is satisfied when the run recorded a verdict for every
268
+ // case it was meant to judge. Counting runs alone reported 20/20 here
269
+ // while all 118 cases still had no status, which is exactly the kind of
270
+ // green that hides an untested app.
271
+ stage: "run",
272
+ satisfied: canonical.length > 0 && planned.size > 0 && judgedAndPlanned.length === planned.size,
273
+ // Counted as a set difference, not ran minus judged: a flow can be
274
+ // judged without a registry run record, which made that subtraction
275
+ // report a negative number of flows.
276
+ detail: ranButUnjudged.length === 0
277
+ ? `${judgedAndPlanned.length}/${planned.size} planned flows executed and judged`
278
+ : `${judgedAndPlanned.length}/${planned.size} planned flows judged `
279
+ + `(${ranButUnjudged.length} ran but recorded no verdicts)`,
280
+ },
281
+ ];
282
+
283
+ const stopped = (reason: string): PipelineState => ({ status: "stopped", reason, stages, orphanedArtifacts });
284
+
285
+ // Cancelling is the user's kill switch, so it must halt the whole
286
+ // pipeline, not just the one task — otherwise the reconciler would
287
+ // cheerfully queue the next stage a few seconds later. Re-triggering
288
+ // recycles that task back to queued, which clears the stop.
289
+ const cancelled = tasks.find((task) =>
290
+ isPipelineTask(task) && task.status === "error" && task.completionSummary === "Cancelled");
291
+ if (cancelled) {
292
+ return stopped("A pipeline task was cancelled. Re-run the stage to resume.");
293
+ }
294
+
295
+ const hitHardStop = recordedBlockerCategories.find((category) => hardStopCategories.includes(category));
296
+ if (hitHardStop) {
297
+ return stopped(`Hard stop: the "${hitHardStop}" blocker halts the run by policy.`);
298
+ }
299
+
300
+ if (inputs.pendingClarification) {
301
+ // Paused, not stopped: the work is fine, it just needs an answer that
302
+ // the agent must not invent.
303
+ return { status: "paused", reason: inputs.pendingClarification, stages, orphanedArtifacts };
304
+ }
305
+
306
+ // A pipeline task that has burned its attempts is not going to succeed
307
+ // on its own. It is neither in flight nor terminal, so the reconciler
308
+ // would keep "enqueueing" it — getting the same blocked task back every
309
+ // few seconds, making no progress and saying nothing. Surface it as
310
+ // needing an answer instead, which is the one thing that can unblock it.
311
+ const exhausted = tasks.filter((task) =>
312
+ isPipelineTask(task)
313
+ && task.status === "blocked"
314
+ && (task.attempts ?? 0) >= MAX_TASK_ATTEMPTS);
315
+ if (exhausted.length) {
316
+ const first = exhausted[0];
317
+ const detail = first.errorMessage?.trim() || first.completionSummary?.trim();
318
+ return {
319
+ status: "paused",
320
+ reason: `${exhausted.length} stage${exhausted.length === 1 ? "" : "s"} need your input — `
321
+ + `"${first.title ?? first.idempotencyKey}" stopped after ${MAX_TASK_ATTEMPTS} attempts`
322
+ + (detail ? `: ${detail}` : "."),
323
+ stages,
324
+ orphanedArtifacts,
325
+ blockedTasks: exhausted.map((task) => ({
326
+ title: task.title ?? task.idempotencyKey ?? "pipeline task",
327
+ idempotencyKey: task.idempotencyKey,
328
+ attempts: task.attempts ?? 0,
329
+ reason: task.errorMessage?.trim() || task.completionSummary?.trim(),
330
+ })),
331
+ };
332
+ }
333
+
334
+ // Only one stage runs at a time; the worker already serialises task
335
+ // execution, and queueing ahead would just create work that races.
336
+ const inFlight = activeTask(tasks);
337
+ if (inFlight) {
338
+ return { status: "running", reason: "A task is already in flight.", stages, orphanedArtifacts };
339
+ }
340
+
341
+ const nextStage = stages.find((stage) => !stage.satisfied);
342
+ if (!nextStage) {
343
+ return { status: "complete", reason: "Every stage is satisfied.", stages, orphanedArtifacts };
344
+ }
345
+
346
+ if (nextStage.stage === "run") {
347
+ const journey = needsRun[0];
348
+ if (journey && runsInFlight.has(journey.id)) {
349
+ // Its run is already executing. Starting another would either duplicate
350
+ // the work or, under a stable idempotency key, hand back the same run
351
+ // and look like progress that is not happening.
352
+ return {
353
+ status: "running",
354
+ reason: `A run of "${journey.id}" is already in flight.`,
355
+ stages,
356
+ orphanedArtifacts,
357
+ };
358
+ }
359
+ // A flow that has burned this many finished runs and still has no
360
+ // verdicts is not going to get them by being run again. Stop and say so,
361
+ // rather than starting run number four and calling it progress.
362
+ const attempts = journey ? finishedRunCounts.get(journey.id) ?? 0 : 0;
363
+ if (journey && attempts >= MAX_TASK_ATTEMPTS) {
364
+ return {
365
+ status: "paused",
366
+ reason: `Flow "${journey.id}" ran ${attempts} times without recording any verdict — it needs your input.`,
367
+ stages,
368
+ orphanedArtifacts,
369
+ };
370
+ }
371
+ const plan = (inputs.publishedPlans ?? []).find((candidate) => candidate.journeyId === journey?.id);
372
+ if (!journey || !plan) {
373
+ // A plan exists on disk but was never published to the registry, so
374
+ // there is nothing runnable yet. Say so instead of silently idling.
375
+ return {
376
+ status: "paused",
377
+ reason: `Flow "${journey?.id ?? "unknown"}" is planned but has no published plan to run.`,
378
+ stages,
379
+ orphanedArtifacts,
380
+ };
381
+ }
382
+ return {
383
+ status: "running",
384
+ reason: "Next stage: run",
385
+ nextRun: {
386
+ stage: "run",
387
+ journeyId: journey.id,
388
+ planId: plan.planId,
389
+ // Discriminated by how many runs of this flow have already finished.
390
+ // The key has to stay stable while a run is in flight, so repeated
391
+ // reconciles do not duplicate it — but it must change once that run
392
+ // finishes, or the registry keeps handing back the completed run and
393
+ // the flow can never be run again. That is exactly what happened: a
394
+ // finished run from hours earlier was returned on every poll, so the
395
+ // pipeline announced "started run" forever and nothing executed.
396
+ idempotencyKey: `${pipelineIdempotencyKey(projectId, "run", journey.id)}:${attempts}`,
397
+ environmentId: inputs.environmentId,
398
+ },
399
+ stages,
400
+ orphanedArtifacts,
401
+ };
402
+ }
403
+
404
+ const next = buildNextTask(projectId, nextStage.stage, {
405
+ missingActors,
406
+ needsAudit,
407
+ needsCases,
408
+ needsPlan,
409
+ });
410
+ return { status: "running", reason: `Next stage: ${nextStage.stage}`, next, stages, orphanedArtifacts };
411
+ }
412
+
413
+ function buildNextTask(
414
+ projectId: string,
415
+ stage: PipelineStage,
416
+ context: {
417
+ missingActors: string[];
418
+ needsAudit: PipelineJourney[];
419
+ needsCases: PipelineJourney[];
420
+ needsPlan: PipelineJourney[];
421
+ },
422
+ ): PipelineNextTask {
423
+ if (stage === "explore") {
424
+ return {
425
+ stage,
426
+ taskType: "investigation",
427
+ title: "Explore the app and build the journey graph",
428
+ description: [
429
+ "Read bugmole://guidance/agent-testing and follow it end to end.",
430
+ "Run bugmole_explore against the configured environment, then record what you actually",
431
+ "observe with bugmole_record_discovery. After every exploration call bugmole_journey_next",
432
+ "and perform the returned task until the flow is exhausted or explicitly blocked.",
433
+ ].join(" "),
434
+ idempotencyKey: pipelineIdempotencyKey(projectId, stage),
435
+ };
436
+ }
437
+ if (stage === "roles") {
438
+ return {
439
+ stage,
440
+ taskType: "code",
441
+ title: "Define the actors the discovered flows use",
442
+ description: [
443
+ `The discovered flows use actors that roles.yaml does not define: ${context.missingActors.join(", ")}.`,
444
+ "Planning refuses an undefined actor, so the pipeline cannot proceed until this is reconciled.",
445
+ "Inspect the application's real roles and either add the missing actor to spec/roles.yaml with",
446
+ "accurate permissions, or update the flow to use an existing role that genuinely matches.",
447
+ "Do not invent permissions to make the check pass.",
448
+ ].join(" "),
449
+ idempotencyKey: pipelineIdempotencyKey(projectId, stage),
450
+ };
451
+ }
452
+ if (stage === "audit") {
453
+ const journey = context.needsAudit[0];
454
+ return {
455
+ stage,
456
+ journeyId: journey?.id,
457
+ taskType: "investigation",
458
+ title: `Audit the controls and components of ${journey?.id ?? "the discovered flow"}`,
459
+ description: [
460
+ "Read bugmole://guidance/app-ui-audit and follow it for this flow.",
461
+ "",
462
+ "First populate the component library, because nothing else does:",
463
+ "call bugmole_ui_re_start with kind \"existing_app\" and the app's base URL, then",
464
+ "bugmole_ui_re_stage with stage \"atoms\". That runs the Storybook/CSF source scan and",
465
+ "records the catalogue the dashboard's component library reads. If the project has no",
466
+ "Storybook, the scan reports that as a blocked reason — report it as-is rather than",
467
+ "inventing components.",
468
+ "",
469
+ `Then audit "${journey?.id}": inventory every interactive control on each screen with its`,
470
+ "handler and evidence, and call bugmole_write_ui_audit with the completed object.",
471
+ ].join("\n"),
472
+ idempotencyKey: pipelineIdempotencyKey(projectId, stage, journey?.id),
473
+ };
474
+ }
475
+ if (stage === "testcases") {
476
+ const journey = context.needsCases[0];
477
+ return {
478
+ stage,
479
+ journeyId: journey?.id,
480
+ taskType: "investigation",
481
+ title: `Write test case assumptions for ${journey?.id ?? "the discovered flow"}`,
482
+ description: [
483
+ `Read bugmole://spec/test-cases.schema.json and write the test case assumptions for "${journey?.id}".`,
484
+ "Every case must name a real nodeId from that flow, say what it is trying to prove, and list",
485
+ "observable expectations. Cite what each assumption rests on in `basis`; a case you cannot ground",
486
+ "must be confidence \"low\" rather than a confident guess. Then call bugmole_write_test_cases.",
487
+ ].join(" "),
488
+ idempotencyKey: pipelineIdempotencyKey(projectId, stage, journey?.id),
489
+ };
490
+ }
491
+ const journey = context.needsPlan[0];
492
+ return {
493
+ stage: "plan",
494
+ journeyId: journey?.id,
495
+ taskType: "investigation",
496
+ title: `Plan ${journey?.id ?? "the discovered flow"}`,
497
+ description: [
498
+ `Call bugmole_plan for "${journey?.id}" using the actor defined for that flow.`,
499
+ "Planning is blocked while the flow still has an actionable exploration task, so check",
500
+ "bugmole_journey_next first and finish any remaining verification before planning.",
501
+ ].join(" "),
502
+ idempotencyKey: pipelineIdempotencyKey(projectId, "plan", journey?.id),
503
+ };
504
+ }
505
+
506
+ // --- Loading real state -----------------------------------------------------
507
+ //
508
+ // Everything below reads the same artifacts the dashboard reads, so the
509
+ // pipeline's view of "what is done" is the same view a human gets rather
510
+ // than a private ledger that can disagree with reality.
511
+
512
+ function store(cfg: any) {
513
+ return createProjectObjectStore(cfg);
514
+ }
515
+
516
+ function readYaml(filePath: string): any {
517
+ try {
518
+ return YAML.parse(fs.readFileSync(filePath, "utf8"));
519
+ } catch {
520
+ return null;
521
+ }
522
+ }
523
+
524
+ /** Actor names defined in roles.yaml, which the planner validates against. */
525
+ export function knownActors(specDir: string): string[] {
526
+ const roles = readYaml(path.join(specDir, "roles.yaml"));
527
+ return Array.isArray(roles?.roles)
528
+ ? roles.roles.map((role: any) => role?.name).filter((name: unknown): name is string => typeof name === "string")
529
+ : [];
530
+ }
531
+
532
+ /**
533
+ * Blocker categories that halt the run by policy.
534
+ *
535
+ * spec/blockers.yaml has declared these since it was written, but nothing
536
+ * in the runtime ever read them — they were documentation the agent might
537
+ * choose to honour. The pipeline enforces them.
538
+ */
539
+ export function hardStopCategories(specDir: string): string[] {
540
+ const doc = readYaml(path.join(specDir, "blockers.yaml"));
541
+ const fromBlockers = Array.isArray(doc?.blockers)
542
+ ? doc.blockers.filter((entry: any) => entry?.hard_stop === true).map((entry: any) => entry?.name)
543
+ : [];
544
+ const fromHardStops = Array.isArray(doc?.hard_stops)
545
+ ? doc.hard_stops.map((entry: any) => entry?.name)
546
+ : [];
547
+ return [...new Set([...fromBlockers, ...fromHardStops])]
548
+ .filter((name: unknown): name is string => typeof name === "string");
549
+ }
550
+
551
+ /**
552
+ * Journeys whose every written case carries a recorded verdict.
553
+ *
554
+ * Partial credit would defeat the purpose: a flow with one judged case and
555
+ * six unjudged ones has not been covered, and counting it would put the
556
+ * pipeline back to reporting green over an untested app. A "blocked" verdict
557
+ * still counts — the run looked and said why it could not check.
558
+ */
559
+ async function listJourneyIdsWithVerdicts(cfg: any, projectId: string): Promise<string[]> {
560
+ try {
561
+ const objects = await store(cfg).list(`${projectPrefix(projectId)}/test-cases/`);
562
+ const ids = await Promise.all(objects
563
+ .filter((object) => object.key.endsWith(".json"))
564
+ .map(async (object) => {
565
+ try {
566
+ const caseSet = JSON.parse(await store(cfg).getText(object.key));
567
+ const journeyId = typeof caseSet?.journeyId === "string" ? caseSet.journeyId : null;
568
+ const caseIds = Array.isArray(caseSet?.cases)
569
+ ? caseSet.cases.map((item: any) => item?.caseId).filter((id: unknown) => typeof id === "string")
570
+ : [];
571
+ if (!journeyId || caseIds.length === 0) return null;
572
+ const resultKey = `${projectPrefix(projectId)}/test-case-results/${caseSet.caseSetId}.json`;
573
+ const results = JSON.parse(await store(cfg).getText(resultKey));
574
+ // A verdict recorded against a superseded revision is not evidence
575
+ // about the cases as they stand now, matching how the dashboard
576
+ // discards it rather than showing it as proof.
577
+ if (Number.isInteger(results?.journeyRevision)
578
+ && Number.isInteger(caseSet?.journeyRevision)
579
+ && results.journeyRevision !== caseSet.journeyRevision) {
580
+ return null;
581
+ }
582
+ const judged = new Set(
583
+ (Array.isArray(results?.results) ? results.results : [])
584
+ .map((item: any) => item?.caseId)
585
+ .filter((id: unknown) => typeof id === "string"),
586
+ );
587
+ return caseIds.every((id: string) => judged.has(id)) ? journeyId : null;
588
+ } catch {
589
+ // No results file for this case set, or it is unreadable: not judged.
590
+ return null;
591
+ }
592
+ }));
593
+ return ids.filter((id): id is string => Boolean(id));
594
+ } catch {
595
+ return [];
596
+ }
597
+ }
598
+
599
+ async function listJourneyIdsWithCaseSets(cfg: any, projectId: string): Promise<string[]> {
600
+ try {
601
+ const objects = await store(cfg).list(`${projectPrefix(projectId)}/test-cases/`);
602
+ const ids = await Promise.all(objects
603
+ .filter((object) => object.key.endsWith(".json"))
604
+ .map(async (object) => {
605
+ try {
606
+ const parsed = JSON.parse(await store(cfg).getText(object.key));
607
+ return typeof parsed?.journeyId === "string" ? parsed.journeyId : null;
608
+ } catch {
609
+ return null;
610
+ }
611
+ }));
612
+ return ids.filter((id): id is string => Boolean(id));
613
+ } catch {
614
+ return [];
615
+ }
616
+ }
617
+
618
+ async function listAuditedJourneyIds(cfg: any, projectId: string): Promise<string[]> {
619
+ try {
620
+ const objects = await store(cfg).list(`${projectPrefix(projectId)}/ui-audits/`);
621
+ const ids = await Promise.all(objects
622
+ .filter((object) => object.key.endsWith(".json"))
623
+ .map(async (object) => {
624
+ try {
625
+ const parsed = JSON.parse(await store(cfg).getText(object.key));
626
+ return typeof parsed?.journeyId === "string" ? parsed.journeyId : null;
627
+ } catch {
628
+ return null;
629
+ }
630
+ }));
631
+ return [...new Set(ids.filter((id): id is string => Boolean(id)))];
632
+ } catch {
633
+ return [];
634
+ }
635
+ }
636
+
637
+ async function listRecordedBlockerCategories(cfg: any, projectId: string): Promise<string[]> {
638
+ try {
639
+ const objects = await store(cfg).list(`${projectPrefix(projectId)}/blocker-resolutions/`);
640
+ const categories = await Promise.all(objects
641
+ .filter((object) => object.key.endsWith(".json"))
642
+ .map(async (object) => {
643
+ try {
644
+ const parsed = JSON.parse(await store(cfg).getText(object.key));
645
+ return typeof parsed?.category === "string" ? parsed.category : null;
646
+ } catch {
647
+ return null;
648
+ }
649
+ }));
650
+ return [...new Set(categories.filter((category): category is string => Boolean(category)))];
651
+ } catch {
652
+ return [];
653
+ }
654
+ }
655
+
656
+ /** Journey ids that already have a published plan on disk. */
657
+ /**
658
+ * Whether a written plan could actually execute.
659
+ *
660
+ * Only the driver/platform pairing is checked, because that is the one that
661
+ * fails silently: the run starts, the driver rejects the plan outright, and
662
+ * every case it was meant to prove is recorded blocked without anything
663
+ * being tested. A plan unreadable for any other reason is left alone rather
664
+ * than quietly discarded.
665
+ */
666
+ function isRunnablePlan(planPath: string): boolean {
667
+ let plan: any;
668
+ try {
669
+ plan = YAML.parse(fs.readFileSync(planPath, "utf8"));
670
+ } catch {
671
+ return true;
672
+ }
673
+ const driver = typeof plan?.driver === "string" ? plan.driver : undefined;
674
+ const platform = typeof plan?.platform === "string" ? plan.platform : "web";
675
+ if (!driver) return true;
676
+ return driverForPlatform(driver, platform) === driver;
677
+ }
678
+
679
+ export function plannedJourneyIds(specDir: string, journeyIds: string[]): string[] {
680
+ const plansDir = path.join(specDir, "plans");
681
+ let files: string[];
682
+ try {
683
+ files = fs.readdirSync(plansDir);
684
+ } catch {
685
+ return [];
686
+ }
687
+ // The planner writes <journeyId sanitised>.flow.yaml, so match by that
688
+ // same transform rather than guessing from the file's contents.
689
+ //
690
+ // A plan whose driver cannot drive its own platform does not count as
691
+ // planned. Eighteen web plans were written pinned to maestro, a mobile
692
+ // runner, and every run of them failed identically without checking
693
+ // anything. Treating those as planned left the pipeline satisfied with
694
+ // work that could never succeed; treating them as missing makes it
695
+ // re-plan them with a driver that can actually run.
696
+ const planIds = new Set(files
697
+ .filter((file) => file.endsWith(".flow.yaml"))
698
+ .filter((file) => isRunnablePlan(path.join(plansDir, file)))
699
+ .map((file) => file.replace(/\.flow\.yaml$/, "")));
700
+ return journeyIds.filter((id) => planIds.has(id.replace(/[^a-z0-9._-]/gi, "_").toLowerCase()));
701
+ }
702
+
703
+ export async function loadPipelineInputs(
704
+ cfg: any,
705
+ tasks: PipelineTask[],
706
+ registry?: {
707
+ runs?: Array<{ journeyId?: string; status?: string }>;
708
+ plans?: Array<{ planId: string; journeyId?: string }>;
709
+ },
710
+ ): Promise<PipelineInputs> {
711
+ const projectId = cfg.project?.id ?? "default";
712
+ const specDir = cfg.runtime?.spec_dir || "./spec";
713
+ const graphPath = path.join(specDir, "journeys.graph.json");
714
+ let journeys: PipelineJourney[] = [];
715
+ try {
716
+ const graph = JSON.parse(fs.readFileSync(graphPath, "utf8"));
717
+ journeys = Array.isArray(graph?.journeys) ? graph.journeys : [];
718
+ } catch {
719
+ journeys = [];
720
+ }
721
+ const [caseSetJourneyIds, recordedBlockerCategories, auditedJourneyIds, verdictJourneyIds] = await Promise.all([
722
+ listJourneyIdsWithCaseSets(cfg, projectId),
723
+ listRecordedBlockerCategories(cfg, projectId),
724
+ listAuditedJourneyIds(cfg, projectId),
725
+ listJourneyIdsWithVerdicts(cfg, projectId),
726
+ ]);
727
+ return {
728
+ projectId,
729
+ journeys,
730
+ knownActors: knownActors(specDir),
731
+ caseSetJourneyIds,
732
+ auditedJourneyIds,
733
+ plannedJourneyIds: plannedJourneyIds(specDir, journeys.map((journey) => journey.id)),
734
+ tasks,
735
+ hardStopCategories: hardStopCategories(specDir),
736
+ recordedBlockerCategories,
737
+ ranJourneyIds: [...new Set(
738
+ (registry?.runs ?? [])
739
+ .map((run) => run.journeyId)
740
+ .filter((id): id is string => Boolean(id)),
741
+ )],
742
+ verdictJourneyIds,
743
+ runs: (registry?.runs ?? []).map((run: any) => ({ journeyId: run.journeyId, status: run.status })),
744
+ publishedPlans: registry?.plans ?? [],
745
+ environmentId: cfg.project?.environment_id ?? cfg.project?.environmentId ?? cfg.runtime?.environment,
746
+ };
747
+ }
748
+
749
+ type ReconcileClient = {
750
+ listTasks(): Promise<{ tasks: any[] }>;
751
+ createTask(input: {
752
+ title: string;
753
+ description: string;
754
+ taskType?: "code" | "ui" | "investigation";
755
+ priority?: number;
756
+ idempotencyKey: string;
757
+ journeyId?: string;
758
+ }): Promise<unknown>;
759
+ listRuns?(): Promise<{ runs: any[] }>;
760
+ listPlans?(): Promise<{ plans: any[] }>;
761
+ createRun?(input: {
762
+ planId: string;
763
+ idempotencyKey: string;
764
+ environmentId?: string;
765
+ }): Promise<unknown>;
766
+ };
767
+
768
+ export type ReconcileResult = {
769
+ state: PipelineState;
770
+ enqueued?: PipelineNextTask;
771
+ startedRun?: PipelineNextRun;
772
+ };
773
+
774
+ /**
775
+ * Advance the pipeline by at most one stage.
776
+ *
777
+ * Safe to call on every worker poll: the stage is derived from artifacts
778
+ * and the task carries a deterministic idempotency key, so repeating this
779
+ * on an unchanged project is a no-op rather than a queue of duplicates.
780
+ */
781
+ export function pipelineStateKey(projectId: string): string {
782
+ return `${projectPrefix(projectId)}/pipeline/state.json`;
783
+ }
784
+
785
+ /**
786
+ * Publish what the pipeline is doing, so it is visible in the dashboard.
787
+ *
788
+ * The pipeline was already advancing on its own, but the only way to see
789
+ * where a project had got to was to run a script against the artifact
790
+ * store — which is not a usable answer to "where do we go from here".
791
+ */
792
+ export async function publishPipelineState(cfg: any, state: PipelineState): Promise<void> {
793
+ const projectId = cfg.project?.id ?? "default";
794
+ await store(cfg).put(
795
+ pipelineStateKey(projectId),
796
+ JSON.stringify({ ...state, projectId, updatedAt: new Date().toISOString() }, null, 2) + "\n",
797
+ "application/json",
798
+ );
799
+ }
800
+
801
+ export async function reconcilePipeline(cfg: any, client: ReconcileClient): Promise<ReconcileResult> {
802
+ const [listed, runs, plans] = await Promise.all([
803
+ client.listTasks(),
804
+ client.listRuns?.().catch(() => ({ runs: [] })) ?? Promise.resolve({ runs: [] }),
805
+ client.listPlans?.().catch(() => ({ plans: [] })) ?? Promise.resolve({ plans: [] }),
806
+ ]);
807
+ const inputs = await loadPipelineInputs(cfg, listed.tasks ?? [], {
808
+ runs: runs.runs ?? [],
809
+ plans: plans.plans ?? [],
810
+ });
811
+ const state = derivePipelineState(inputs);
812
+ // Published on every pass so the dashboard reflects the live position,
813
+ // including when nothing is queued.
814
+ await publishPipelineState(cfg, state).catch(() => undefined);
815
+
816
+ if (state.nextRun && client.createRun) {
817
+ await client.createRun({
818
+ planId: state.nextRun.planId,
819
+ idempotencyKey: state.nextRun.idempotencyKey,
820
+ environmentId: state.nextRun.environmentId,
821
+ });
822
+ return { state, startedRun: state.nextRun };
823
+ }
824
+ if (!state.next) return { state };
825
+ await client.createTask({
826
+ title: state.next.title,
827
+ description: state.next.description,
828
+ taskType: state.next.taskType,
829
+ idempotencyKey: state.next.idempotencyKey,
830
+ journeyId: state.next.journeyId,
831
+ });
832
+ return { state, enqueued: state.next };
833
+ }