@bugmole/cli 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. package/LICENSE +18 -4
  2. package/package.json +7 -4
  3. package/scripts/bugmole-admin.mjs +194 -0
  4. package/scripts/bugmole-admin.test.mjs +62 -0
  5. package/scripts/bugmole.test.ts +94 -2
  6. package/scripts/bugmole.ts +290 -43
  7. package/scripts/ensure-playwright.cjs +87 -0
  8. package/scripts/ensure-playwright.test.ts +36 -0
  9. package/scripts/sync-byok.d.mts +2 -0
  10. package/scripts/sync-byok.mjs +18 -0
  11. package/scripts/sync-launcher-version.cjs +24 -0
  12. package/scripts/sync-launcher-version.test.ts +28 -0
  13. package/scripts/sync-plan-catalog.d.mts +3 -0
  14. package/scripts/sync-plan-catalog.mjs +16 -6
  15. package/spec/domain_rules.yaml +139 -0
  16. package/spec/roles.yaml +20 -0
  17. package/spec/test-case-results.schema.json +20 -4
  18. package/spec/test-cases.schema.json +131 -16
  19. package/src/billing/plan-catalog.test.ts +82 -2
  20. package/src/billing/plan-catalog.ts +212 -9
  21. package/src/integrations/jira.ts +264 -0
  22. package/src/mcp/roles-and-review.test.ts +111 -0
  23. package/src/mcp/server.ts +358 -22
  24. package/src/mcp/write-test-cases.test.ts +68 -0
  25. package/src/registry/control-plane-client.ts +121 -6
  26. package/src/registry/migrations/0034_signup_attribution.sql +19 -0
  27. package/src/registry/migrations/0035_domain_verify_file.sql +6 -0
  28. package/src/registry/migrations/0036_task_approval.sql +11 -0
  29. package/src/registry/migrations/0037_spec_proposals.sql +27 -0
  30. package/src/registry/migrations/0038_local_worker_seen.sql +5 -0
  31. package/src/registry/migrations/0039_project_secrets.sql +37 -0
  32. package/src/registry/migrations/0040_roles_and_task_review.sql +31 -0
  33. package/src/registry/migrations/0041_jira_integration.sql +75 -0
  34. package/src/registry/migrations/0042_subscription_gaps.sql +10 -0
  35. package/src/registry/migrations/0043_workspace_feature_overrides.sql +17 -0
  36. package/src/registry/migrations/0044_project_identity.sql +31 -0
  37. package/src/registry/project-identity.test.ts +99 -0
  38. package/src/registry/project-identity.ts +206 -0
  39. package/src/registry/roles.test.ts +66 -0
  40. package/src/registry/roles.ts +201 -0
  41. package/src/registry/task-scheduling.test.ts +94 -0
  42. package/src/registry/task-scheduling.ts +199 -2
  43. package/src/registry/test-case-revisions.test.ts +57 -0
  44. package/src/registry/test-case-revisions.ts +132 -0
  45. package/src/registry-worker/ai/routes.ts +3 -3
  46. package/src/registry-worker/artifacts.ts +23 -4
  47. package/src/registry-worker/billing/billing-core.test.ts +1 -1
  48. package/src/registry-worker/billing/checkout-routes.ts +53 -4
  49. package/src/registry-worker/billing/enforcement.ts +67 -12
  50. package/src/registry-worker/billing/paypal/api.ts +14 -0
  51. package/src/registry-worker/billing/paypal/client.ts +9 -0
  52. package/src/registry-worker/billing/paypal/provider.ts +10 -1
  53. package/src/registry-worker/billing/paypal.test.ts +108 -1
  54. package/src/registry-worker/billing/plan-gaps.test.ts +216 -0
  55. package/src/registry-worker/billing/provider.ts +8 -0
  56. package/src/registry-worker/billing/routes.ts +2 -1
  57. package/src/registry-worker/billing/subscriptions.ts +114 -4
  58. package/src/registry-worker/core.ts +24 -0
  59. package/src/registry-worker/devices/policy.ts +4 -5
  60. package/src/registry-worker/devices/routes.ts +8 -8
  61. package/src/registry-worker/domains/domains.test.ts +64 -1
  62. package/src/registry-worker/domains/routes.ts +32 -15
  63. package/src/registry-worker/domains.ts +40 -2
  64. package/src/registry-worker/feature-access.ts +165 -0
  65. package/src/registry-worker/feature-flags/admin.ts +300 -0
  66. package/src/registry-worker/feature-flags/feature-flags.test.ts +270 -0
  67. package/src/registry-worker/feature-flags/routes.ts +102 -0
  68. package/src/registry-worker/features.ts +11 -0
  69. package/src/registry-worker/feedback/feedback.test.ts +47 -0
  70. package/src/registry-worker/feedback/routes.ts +2 -2
  71. package/src/registry-worker/feedback/website.ts +77 -0
  72. package/src/registry-worker/flags.ts +80 -18
  73. package/src/registry-worker/github/checks.ts +4 -4
  74. package/src/registry-worker/hooks.ts +9 -0
  75. package/src/registry-worker/identity/oidc.test.ts +103 -0
  76. package/src/registry-worker/identity/oidc.ts +176 -0
  77. package/src/registry-worker/index.ts +494 -173
  78. package/src/registry-worker/jira/connection.ts +111 -0
  79. package/src/registry-worker/jira/jira.test.ts +404 -0
  80. package/src/registry-worker/jira/routes.ts +432 -0
  81. package/src/registry-worker/jira/workflow.ts +577 -0
  82. package/src/registry-worker/jobs/retention.ts +21 -5
  83. package/src/registry-worker/mcp/tools.ts +2 -1
  84. package/src/registry-worker/notifications/alerts.ts +4 -4
  85. package/src/registry-worker/notifications/notifications.test.ts +11 -0
  86. package/src/registry-worker/notifications/routes.ts +9 -10
  87. package/src/registry-worker/notifications/teams.ts +6 -2
  88. package/src/registry-worker/org/routes.test.ts +24 -1
  89. package/src/registry-worker/org/routes.ts +9 -8
  90. package/src/registry-worker/projects/identity.ts +194 -0
  91. package/src/registry-worker/projects/inactivity.ts +162 -0
  92. package/src/registry-worker/projects/projects.test.ts +283 -0
  93. package/src/registry-worker/proposals/proposals.test.ts +80 -0
  94. package/src/registry-worker/proposals/routes.ts +183 -0
  95. package/src/registry-worker/roles/roles.test.ts +84 -0
  96. package/src/registry-worker/roles/routes.ts +141 -0
  97. package/src/registry-worker/runner/dispatch.ts +1 -0
  98. package/src/registry-worker/runner/routes.ts +22 -5
  99. package/src/registry-worker/runner/runner.test.ts +59 -0
  100. package/src/registry-worker/runner/tokens.ts +12 -1
  101. package/src/registry-worker/secrets/crypto.ts +135 -0
  102. package/src/registry-worker/secrets/routes.ts +296 -0
  103. package/src/registry-worker/secrets/secrets.test.ts +237 -0
  104. package/src/registry-worker/signup/attribution.test.ts +147 -0
  105. package/src/registry-worker/signup/attribution.ts +194 -0
  106. package/src/registry-worker/signup/policy.ts +2 -2
  107. package/src/registry-worker/signup/routes.ts +12 -3
  108. package/src/registry-worker/sso/membership.ts +4 -2
  109. package/src/registry-worker/sso/routes.ts +24 -8
  110. package/src/registry-worker/sso/sso.test.ts +3 -2
  111. package/src/registry-worker/task-approval.test.ts +67 -0
  112. package/src/registry-worker/task-resume.test.ts +196 -0
  113. package/src/registry-worker/task-review.test.ts +121 -0
  114. package/src/runtime/ai-exploration.test.ts +34 -1
  115. package/src/runtime/ai-exploration.ts +97 -3
  116. package/src/runtime/appium-driver.ts +7 -1
  117. package/src/runtime/apply-proposals.test.ts +74 -0
  118. package/src/runtime/apply-proposals.ts +41 -0
  119. package/src/runtime/cloud-secrets.test.ts +201 -0
  120. package/src/runtime/cloud-secrets.ts +210 -0
  121. package/src/runtime/config-validate.ts +12 -9
  122. package/src/runtime/cursor-driver-run.ts +14 -3
  123. package/src/runtime/discovery-task.test.ts +26 -1
  124. package/src/runtime/discovery-task.ts +87 -7
  125. package/src/runtime/executor.ts +18 -10
  126. package/src/runtime/explorer.test.ts +32 -0
  127. package/src/runtime/explorer.ts +48 -0
  128. package/src/runtime/flow-language.ts +43 -8
  129. package/src/runtime/init-wizard.ts +112 -6
  130. package/src/runtime/journey-editor.ts +78 -2
  131. package/src/runtime/journey-graph.test.ts +60 -0
  132. package/src/runtime/local-registry-stub.test.ts +64 -0
  133. package/src/runtime/local-registry-stub.ts +255 -7
  134. package/src/runtime/local-vault.ts +65 -0
  135. package/src/runtime/pipeline.test.ts +23 -0
  136. package/src/runtime/pipeline.ts +61 -20
  137. package/src/runtime/planner.test.ts +24 -1
  138. package/src/runtime/planner.ts +88 -23
  139. package/src/runtime/playwright-driver.test.ts +164 -0
  140. package/src/runtime/playwright-driver.ts +286 -15
  141. package/src/runtime/project-identity.test.ts +161 -0
  142. package/src/runtime/project-identity.ts +232 -0
  143. package/src/runtime/project-roles.test.ts +107 -0
  144. package/src/runtime/project-roles.ts +158 -0
  145. package/src/runtime/propose-cli.ts +66 -0
  146. package/src/runtime/record-run-verdicts.test.ts +58 -0
  147. package/src/runtime/record-run-verdicts.ts +38 -4
  148. package/src/runtime/reporter.ts +1 -1
  149. package/src/runtime/reset.ts +2 -0
  150. package/src/runtime/roles-cli.test.ts +47 -0
  151. package/src/runtime/roles-cli.ts +60 -0
  152. package/src/runtime/run-job.test.ts +21 -0
  153. package/src/runtime/run-job.ts +8 -0
  154. package/src/runtime/run-once.ts +68 -5
  155. package/src/runtime/scenario-matrix.test.ts +41 -0
  156. package/src/runtime/scenario-matrix.ts +98 -0
  157. package/src/runtime/secret-driver.ts +77 -0
  158. package/src/runtime/secret-redaction.ts +69 -0
  159. package/src/runtime/secret-sources.test.ts +442 -0
  160. package/src/runtime/secret-sources.ts +176 -0
  161. package/src/runtime/secrets-cli.ts +146 -0
  162. package/src/runtime/serve-worker.ts +64 -6
  163. package/src/runtime/site-discovery.test.ts +181 -4
  164. package/src/runtime/site-discovery.ts +248 -29
  165. package/src/runtime/vault-federation.test.ts +37 -0
  166. package/src/runtime/vault-federation.ts +128 -0
  167. package/src/runtime/web-suite.test.ts +65 -0
  168. package/src/runtime/web-suite.ts +8 -2
  169. package/src/storage/create-object-store.ts +5 -1
  170. package/src/storage/object-store.ts +12 -1
  171. package/src/storage/test-case-results.test.ts +64 -0
  172. package/src/storage/test-case-results.ts +49 -0
  173. package/src/vendor/byok.ts +371 -0
package/src/mcp/server.ts CHANGED
@@ -1,13 +1,19 @@
1
1
  import http from "node:http";
2
+ import { activeRevision, caseProblem, isObsolete, revisionOf } from "../registry/test-case-revisions.js";
3
+ import { SCENARIO_CATEGORIES, SCENARIO_MEANING, isScenarioCategory, normalizeScenarioExclusions, scenarioMatrix } from "../runtime/scenario-matrix.js";
4
+ import { applyAcceptedProposals } from "../runtime/apply-proposals.js";
2
5
  import fs from "node:fs";
3
6
  import path from "node:path";
4
7
  import { URL } from "node:url";
5
8
  import * as readline from "node:readline";
6
9
  import { createProjectObjectStore } from "../storage/create-object-store.js";
7
10
  import { json, type ObjectStore } from "../storage/object-store.js";
8
- import { projectPrefix } from "../storage/keys.js";
9
- import { ControlPlaneClient, isClaimableTask } from "../registry/control-plane-client.js";
11
+ import { projectPrefix, runKey } from "../storage/keys.js";
12
+ import { putTestCaseResults } from "../storage/test-case-results.js";
13
+ import { ControlPlaneClient, isClaimableTask, workerApprovalMode } from "../registry/control-plane-client.js";
10
14
  import { getEnvironmentContract, setEnvironmentContract } from "../runtime/environment.js";
15
+ import { collectRoleSuggestions, holdAgentRoleWrites, projectActorProblem, projectRoles, proposeRoleSuggestions, rolesClient } from "../runtime/project-roles.js";
16
+ import { ROLE_SOURCES, normalizeRoleName } from "../registry/roles.js";
11
17
  import {
12
18
  completeCoordinatorJob,
13
19
  extractAtomicCandidates,
@@ -103,9 +109,9 @@ interface Prompt {
103
109
  function taskClient(cfg: any): ControlPlaneClient {
104
110
  const registryUrl = process.env.BUGMOLE_REGISTRY_URL ?? cfg.registry?.url ?? "https://api.bugmole.com";
105
111
  const apiKey = process.env.BUGMOLE_API_KEY ?? cfg.registry?.api_key;
106
- if (!apiKey) throw new Error("BUGMOLE_API_KEY is required for agent task queue operations");
112
+ if (!apiKey) throw new Error("This needs your Bugmole account: set BUGMOLE_API_KEY in .bugmole.env (create a key on the dashboard's API keys page).");
107
113
  const projectId = cfg.project?.id;
108
- if (!projectId) throw new Error("A project id is required for agent task queue operations");
114
+ if (!projectId) throw new Error("This needs a project ID: set project.id in .bugmole/config.yaml.");
109
115
  return new ControlPlaneClient({ registryUrl, apiKey, projectId });
110
116
  }
111
117
 
@@ -959,7 +965,7 @@ export function getTools(cfg: any): Tool[] {
959
965
  {
960
966
  name: "bugmole_write_test_cases",
961
967
  description:
962
- "Validate and write test case assumptions for a discovered journey. Every case must reference a node (and optionally an edge) that exists in that journey at that revision — assumptions are grounded in the real graph, never invented.",
968
+ "Validate and write test case assumptions for a discovered journey. Every case must reference a node (and optionally an edge) that exists in that journey at that revision — assumptions are grounded in the real graph, never invented. Give each case a category (happy_path, validation, error_state, permission, edge_case): every step needs a happy_path case, and each other category needs a case somewhere in the journey or a scenarioExclusions entry with the reason it doesn't apply. The result lists what is still missing.",
963
969
  inputSchema: {
964
970
  type: "object",
965
971
  properties: {
@@ -971,10 +977,26 @@ export function getTools(cfg: any): Tool[] {
971
977
  required: ["caseSet"],
972
978
  },
973
979
  },
980
+ {
981
+ name: "bugmole_propose_test_case",
982
+ description:
983
+ "Propose adding, changing or retiring one test case. A change never edits a case in place: once accepted it becomes a new revision and the old one is marked obsolete. The project's approval mode decides whether a manager must accept it (manual, auto) or it's accepted on arrival (extreme). Say why in summary.",
984
+ inputSchema: {
985
+ type: "object",
986
+ properties: {
987
+ caseSetId: { type: "string" },
988
+ caseId: { type: "string" },
989
+ action: { type: "string", enum: ["add", "change", "retire"] },
990
+ case: { type: "object", description: "The full case for add/change (title, nodeId, intent, steps, expected, confidence, basis), as in bugmole://spec/test-cases.schema.json" },
991
+ summary: { type: "string", description: "Why this change: what the app or the understanding of it changed" },
992
+ },
993
+ required: ["caseSetId", "caseId", "action", "summary"],
994
+ },
995
+ },
974
996
  {
975
997
  name: "bugmole_write_test_case_results",
976
998
  description:
977
- "Record what a run observed for test cases already written by bugmole_write_test_cases. Every result must name a caseId that exists in that case set at that revision. This is the only thing that moves a case off \"planned\" — assumptions never count as evidence.",
999
+ "Record what a run observed for test cases already written by bugmole_write_test_cases. Every result must name a caseId that exists in that case set at that revision. This is the only thing that moves a case off \"planned\" — assumptions never count as evidence. Each write replaces the latest results and keeps its own copy in the case's history, so pass runId when there is one.",
978
1000
  inputSchema: {
979
1001
  type: "object",
980
1002
  properties: {
@@ -1103,9 +1125,50 @@ export function getTools(cfg: any): Tool[] {
1103
1125
  required: ["title", "description", "idempotencyKey"],
1104
1126
  },
1105
1127
  },
1128
+ {
1129
+ name: "bugmole_secrets_status",
1130
+ description:
1131
+ "List the secrets the project's flows type as {{secret.NAME}} (passwords, tokens), where each one comes from (Google Secret Manager, AWS Secrets Manager, this machine's secure store or keychain, or an environment variable) and whether this machine can supply it. Never returns a value. Never ask a person to paste a password into the chat: when one is missing, tell them to run `bugmole secrets set NAME` in their own terminal, or to map NAME to their vault under secrets: in the Bugmole config.",
1132
+ inputSchema: { type: "object", properties: {} },
1133
+ },
1134
+ {
1135
+ name: "bugmole_sign_in",
1136
+ description:
1137
+ "Sign in to the app by running a sign-in flow that types {{secret.NAME}} values, on this machine, and save the signed-in browser session (cookies and local storage). Returns the session file to explore or test as a signed-in user (Playwright storageState; bugmole --mode discover --sign-in uses the same flow). Secrets are read from the project's vault and never reach you.",
1138
+ inputSchema: {
1139
+ type: "object",
1140
+ properties: {
1141
+ flow: { type: "string", description: "Path of the sign-in flow, relative to the workspace (e.g. spec/flows/sign-in.yaml)" },
1142
+ baseUrl: { type: "string", description: "The app's address; the flow's own URL is moved onto it" },
1143
+ },
1144
+ required: ["flow"],
1145
+ },
1146
+ },
1147
+ {
1148
+ name: "bugmole_roles_list",
1149
+ description:
1150
+ "List the project's roles (who a journey is run as: guest, member, admin…) with their status. Only confirmed roles may be used as a plan's actor or a test case set's actor; suggested ones wait for a person, rejected ones are never used.",
1151
+ inputSchema: { type: "object", properties: {} },
1152
+ },
1153
+ {
1154
+ name: "bugmole_role_propose",
1155
+ description:
1156
+ "Suggest a role for a person to confirm, rename or reject on the dashboard's Roles page. You cannot confirm a role yourself. Give the evidence that points at it (a sign-in wall, a role-gated route or menu, a spec entry). With fromEvidence: true, Bugmole derives suggestions from the journey graph, blockers and roles.yaml instead. Suggesting a role a person already decided on changes nothing.",
1157
+ inputSchema: {
1158
+ type: "object",
1159
+ properties: {
1160
+ name: { type: "string", description: "Role name, used as the actor id (e.g. member, admin)" },
1161
+ description: { type: "string", description: "Who this is and what they can do" },
1162
+ source: { type: "string", enum: ROLE_SOURCES.filter((source) => source !== "person") },
1163
+ evidence: { type: "string", description: "What points at this role, e.g. \"sign-in screen at /login\" or \"route /admin\"" },
1164
+ fromEvidence: { type: "boolean", description: "Derive and suggest roles from the project's graph, blockers and roles.yaml" },
1165
+ },
1166
+ },
1167
+ },
1106
1168
  {
1107
1169
  name: "bugmole_task_update",
1108
- description: "Report progress, continuation steps, parity evidence, or a blocked/completed status for the claimed task.",
1170
+ description:
1171
+ "Report progress, continuation steps, parity evidence, or a blocked/completed status for the claimed task. Depending on the project's approval mode, a task you report completed may go to review (status in_review) instead: a person then accepts it or sends it back with a note. The response says so with inReview: true; do not treat the work as accepted.",
1109
1172
  inputSchema: {
1110
1173
  type: "object",
1111
1174
  properties: {
@@ -1126,7 +1189,7 @@ export function getTools(cfg: any): Tool[] {
1126
1189
  {
1127
1190
  name: "bugmole_commit_response",
1128
1191
  description:
1129
- "Finalize a Cursor task by committing structured response payload. This is the mandatory final step for all CURSOR_DRIVER tasks. Use the SIMPLEST possible JSON - only status and summary are required. Everything else is optional.",
1192
+ "Finish a task Bugmole handed to this agent by committing its structured response. It is the required last step of every such task. Keep the JSON simple: only status and summary are required.",
1130
1193
  inputSchema: {
1131
1194
  type: "object",
1132
1195
  properties: {
@@ -1496,7 +1559,7 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1496
1559
 
1497
1560
  case "bugmole_explore": {
1498
1561
  const { explore } = await import("../runtime/explorer.js");
1499
- return await explore(cfg, {
1562
+ const explored = await explore(cfg, {
1500
1563
  baseUrl: args.baseUrl,
1501
1564
  maxDepth: args.maxDepth,
1502
1565
  projectId: args.projectId,
@@ -1506,6 +1569,128 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1506
1569
  journeyId: args.journeyId,
1507
1570
  actor: args.actor,
1508
1571
  });
1572
+ // Discovery is where roles show up (a sign-in wall, an /admin route),
1573
+ // so suggest them now for a person to decide on. Best effort: a
1574
+ // project without a registry key still gets the suggestions listed.
1575
+ if (explored.status !== "success") return explored;
1576
+ const suggestions = collectRoleSuggestions(cfg);
1577
+ if (!suggestions.length) return explored;
1578
+ const client = rolesClient(cfg);
1579
+ const roles = await projectRoles(cfg, client);
1580
+ const pending = suggestions.filter((suggestion) => !roles.usable.includes(suggestion.name));
1581
+ if (!pending.length) return explored;
1582
+ const proposed = client ? await proposeRoleSuggestions(pending, client, roles.usable) : [];
1583
+ return {
1584
+ ...explored,
1585
+ roleSuggestions: client ? proposed : pending.map((item) => ({ name: item.name, source: item.source, status: "suggested", evidence: item.evidence })),
1586
+ nextSteps: [
1587
+ ...((explored as { nextSteps?: string[] }).nextSteps ?? []),
1588
+ client
1589
+ ? `Bugmole suggested roles (${pending.map((item) => item.name).join(", ")}). A person confirms, renames or rejects them on the Roles page; only confirmed roles are used for test cases and runs.`
1590
+ : `Roles this exploration points at: ${pending.map((item) => item.name).join(", ")}. Suggest them with bugmole_role_propose (needs BUGMOLE_API_KEY) or add them to roles.yaml with status: suggested for a person to confirm.`,
1591
+ ],
1592
+ };
1593
+ }
1594
+
1595
+ case "bugmole_secrets_status": {
1596
+ const { projectSecretStatuses } = await import("../runtime/secrets-cli.js");
1597
+ const { statuses, problems } = await projectSecretStatuses(cfg, process.cwd());
1598
+ const missing = statuses.filter((status) => !status.ready);
1599
+ return {
1600
+ secrets: statuses,
1601
+ problems,
1602
+ nextSteps: missing.length
1603
+ ? [`Ask the person to supply ${missing.map((status) => status.name).join(", ")} themselves: \`bugmole secrets set NAME\` in their own terminal, or a vault reference under secrets: in the config. Don't ask for the value in chat.`]
1604
+ : [],
1605
+ };
1606
+ }
1607
+ case "bugmole_sign_in": {
1608
+ if (typeof args.flow !== "string" || !args.flow.trim()) throw new Error("flow is required");
1609
+ const flowPath = path.resolve(process.cwd(), args.flow);
1610
+ if (!flowPath.startsWith(path.resolve(process.cwd()) + path.sep)) throw new Error("flow must be inside the workspace");
1611
+ if (!fs.existsSync(flowPath)) throw new Error(`No flow at ${args.flow}`);
1612
+ const projectId = cfg.project?.id ?? "default";
1613
+ const { secretMap } = await import("../runtime/secret-sources.js");
1614
+ const { withFlowSecrets } = await import("../runtime/secret-driver.js");
1615
+ const { PlaywrightDriver } = await import("../runtime/playwright-driver.js");
1616
+ const runId = `sign-in-${Date.now()}`;
1617
+ const directory = path.resolve(cfg.runtime?.artifacts_dir || "./artifacts", projectId, "sessions", runId);
1618
+ // Session files hold the app's cookies: only this user may read them.
1619
+ fs.mkdirSync(directory, { recursive: true, mode: 0o700 });
1620
+ fs.chmodSync(path.dirname(directory), 0o700);
1621
+ const sessionStatePath = path.join(directory, "session-state.json");
1622
+ const driver = withFlowSecrets(new PlaywrightDriver({ headless: true }), { map: secretMap(cfg).map, projectId });
1623
+ const result = await driver.run({
1624
+ runId,
1625
+ projectId,
1626
+ journeyId: "sign-in",
1627
+ flowPath,
1628
+ platform: "web",
1629
+ target: typeof args.baseUrl === "string" && args.baseUrl ? { baseUrl: args.baseUrl } : {},
1630
+ rebaseOrigin: typeof args.baseUrl === "string" && Boolean(args.baseUrl),
1631
+ timeoutMs: 60_000,
1632
+ artifactsDir: path.join(directory, "artifacts"),
1633
+ sessionStatePath,
1634
+ });
1635
+ if (result.status !== "passed") {
1636
+ return {
1637
+ status: result.status,
1638
+ message: result.message,
1639
+ nextSteps: result.failureCode === "secret_unavailable"
1640
+ ? ["Call bugmole_secrets_status, then ask the person to supply the missing secret in their own terminal (bugmole secrets set NAME) or in their vault."]
1641
+ : ["Check the sign-in flow against the app, then try again."],
1642
+ };
1643
+ }
1644
+ return {
1645
+ status: "passed",
1646
+ sessionStatePath,
1647
+ message: "Signed in. The session file holds cookies for the app: use it as Playwright storageState, and don't paste its contents anywhere.",
1648
+ };
1649
+ }
1650
+ case "bugmole_roles_list": {
1651
+ const client = rolesClient(cfg);
1652
+ const roles = await projectRoles(cfg, client);
1653
+ const registry = client && roles.registryRead ? (await client.listRoles().catch(() => ({ roles: [] }))).roles : [];
1654
+ return {
1655
+ confirmed: roles.usable,
1656
+ suggested: roles.suggested,
1657
+ roles: registry.length ? registry : roles.spec,
1658
+ source: registry.length ? "registry" : "spec/roles.yaml",
1659
+ ...(client && !roles.registryRead ? { warning: "Bugmole couldn't reach the account, so only spec/roles.yaml was read." } : {}),
1660
+ nextSteps: roles.suggested.length
1661
+ ? [`Ask a person to confirm or reject the suggested roles on the dashboard's Roles page: ${roles.suggested.join(", ")}`]
1662
+ : [],
1663
+ };
1664
+ }
1665
+
1666
+ case "bugmole_role_propose": {
1667
+ const client = rolesClient(cfg);
1668
+ if (!client) throw new Error("BUGMOLE_API_KEY and a project id are required to suggest roles; or add the role to roles.yaml with status: suggested");
1669
+ if (args.fromEvidence === true) {
1670
+ const roles = await projectRoles(cfg, client);
1671
+ const proposed = await proposeRoleSuggestions(collectRoleSuggestions(cfg), client, roles.usable);
1672
+ return {
1673
+ proposed,
1674
+ message: proposed.length
1675
+ ? "Suggested. A person confirms, renames or rejects each one on the Roles page; only confirmed roles are used."
1676
+ : "Nothing new to suggest: the evidence points at no role that isn't already confirmed.",
1677
+ };
1678
+ }
1679
+ const name = normalizeRoleName(args.name);
1680
+ if (!name) throw new Error("name is required: letters, numbers and underscores, up to 60 characters");
1681
+ if (typeof args.evidence !== "string" || !args.evidence.trim()) throw new Error("Say what points at this role in evidence");
1682
+ const source = ROLE_SOURCES.includes(args.source) && args.source !== "person" ? args.source : "agent";
1683
+ const { role, alreadyDecided } = await client.proposeRole({
1684
+ name, description: typeof args.description === "string" ? args.description : "", source, evidence: args.evidence,
1685
+ });
1686
+ return {
1687
+ role,
1688
+ message: alreadyDecided
1689
+ ? `A person already ${role.status} "${role.name}"; suggesting it again changes nothing.`
1690
+ : role.status === "confirmed"
1691
+ ? `"${role.name}" is already a confirmed role.`
1692
+ : `Suggested "${role.name}". It is used for test cases and runs only after a person confirms it on the Roles page.`,
1693
+ };
1509
1694
  }
1510
1695
 
1511
1696
  case "bugmole_journey_list": {
@@ -1597,6 +1782,26 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1597
1782
  }
1598
1783
 
1599
1784
  case "bugmole_journey_stabilize": {
1785
+ // "Stable" means a run proved the journey: check that run exists,
1786
+ // passed, and ran this journey, instead of trusting any ID.
1787
+ const runId = String(args.verificationRunId ?? "").trim();
1788
+ const store = createProjectObjectStore(cfg, cfg.runtime.artifacts_dir || "./artifacts");
1789
+ const manifestKey = runKey(cfg.project?.id ?? "default", runId, "manifest.json");
1790
+ let manifest: { status?: string; journeyId?: string } | null = null;
1791
+ try {
1792
+ manifest = runId ? JSON.parse(await store.getText(manifestKey)) : null;
1793
+ } catch {
1794
+ manifest = null;
1795
+ }
1796
+ if (!manifest) {
1797
+ return { status: "error", error: `No run ${runId || "(empty)"} in this project. Run the journey with bugmole_run, then pass that run's ID.` };
1798
+ }
1799
+ if (manifest.status !== "success") {
1800
+ return { status: "error", error: `Run ${runId} did not pass (${manifest.status}). Fix the journey and run it again before marking it stable.` };
1801
+ }
1802
+ if (manifest.journeyId && manifest.journeyId !== args.journeyId) {
1803
+ return { status: "error", error: `Run ${runId} ran journey ${manifest.journeyId}, not ${args.journeyId}.` };
1804
+ }
1600
1805
  const result = await stabilizeJourney(cfg, {
1601
1806
  journeyId: args.journeyId,
1602
1807
  verificationRunId: args.verificationRunId,
@@ -1781,6 +1986,7 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1781
1986
  }
1782
1987
 
1783
1988
  case "bugmole_plan": {
1989
+ // The planner accepts only confirmed roles as the actor (src/registry/roles.ts).
1784
1990
  const { plan } = await import("../runtime/planner.js");
1785
1991
  return await plan(cfg, {
1786
1992
  journeyId: args.journeyId,
@@ -1852,6 +2058,13 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1852
2058
  `caseSet.journeyRevision must match the selected journey revision (${journey.revision})`,
1853
2059
  );
1854
2060
  }
2061
+ // Test cases are planned for confirmed roles only: a case set that
2062
+ // names its actor must name one a person confirmed.
2063
+ if (caseSet.actor !== undefined) {
2064
+ if (typeof caseSet.actor !== "string" || !caseSet.actor.trim()) throw new Error("caseSet.actor must be a role name");
2065
+ const actorIssue = projectActorProblem(await projectRoles(cfg), caseSet.actor.trim());
2066
+ if (actorIssue) throw new Error(actorIssue);
2067
+ }
1855
2068
  const allowedNodes = new Set(journey.nodeIds);
1856
2069
  const allowedEdges = new Set(journey.edgeIds);
1857
2070
  const seenCaseIds = new Set<string>();
@@ -1865,8 +2078,16 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1865
2078
  if (!/^[a-zA-Z0-9._-]+$/.test(testCase.caseId)) {
1866
2079
  throw new Error(`case ${testCase.caseId}: caseId may contain only letters, numbers, ., _, and -`);
1867
2080
  }
1868
- if (seenCaseIds.has(testCase.caseId)) throw new Error(`duplicate caseId: ${testCase.caseId}`);
1869
- seenCaseIds.add(testCase.caseId);
2081
+ // A set keeps every revision of a case; only one may be current.
2082
+ const revisionKey = `${testCase.caseId}@${revisionOf(testCase)}`;
2083
+ if (seenCaseIds.has(revisionKey)) throw new Error(`duplicate caseId: ${testCase.caseId} (revision ${revisionOf(testCase)})`);
2084
+ seenCaseIds.add(revisionKey);
2085
+ if (!isObsolete(testCase)) {
2086
+ if (seenCaseIds.has(`${testCase.caseId}@active`)) throw new Error(`case ${testCase.caseId} has more than one current revision`);
2087
+ seenCaseIds.add(`${testCase.caseId}@active`);
2088
+ }
2089
+ // An obsolete revision may name steps the journey no longer has.
2090
+ if (isObsolete(testCase)) continue;
1870
2091
  if (!allowedNodes.has(testCase.nodeId)) {
1871
2092
  throw new Error(
1872
2093
  `case ${testCase.caseId}: nodeId "${testCase.nodeId}" is not a step of journey ${caseSet.journeyId}`,
@@ -1895,7 +2116,30 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1895
2116
  `case ${testCase.caseId}: a case with no basis must be confidence "low"`,
1896
2117
  );
1897
2118
  }
2119
+ if (testCase.category !== undefined && !isScenarioCategory(testCase.category)) {
2120
+ throw new Error(`case ${testCase.caseId}: category must be one of ${SCENARIO_CATEGORIES.join(", ")}`);
2121
+ }
1898
2122
  }
2123
+ if (caseSet.scenarioExclusions !== undefined) {
2124
+ if (!Array.isArray(caseSet.scenarioExclusions)) throw new Error("caseSet.scenarioExclusions must be an array");
2125
+ for (const exclusion of caseSet.scenarioExclusions) {
2126
+ if (!exclusion || typeof exclusion !== "object" || !isScenarioCategory(exclusion.category)) {
2127
+ throw new Error(`each scenario exclusion needs a category: one of ${SCENARIO_CATEGORIES.join(", ")}`);
2128
+ }
2129
+ // Saying a scenario doesn't apply is a claim like any other: it needs its reason.
2130
+ if (typeof exclusion.reason !== "string" || !exclusion.reason.trim()) {
2131
+ throw new Error(`scenario exclusion ${exclusion.category}: reason is required`);
2132
+ }
2133
+ if (exclusion.nodeId !== undefined && !allowedNodes.has(exclusion.nodeId)) {
2134
+ throw new Error(`scenario exclusion ${exclusion.category}: nodeId "${exclusion.nodeId}" is not a step of journey ${caseSet.journeyId}`);
2135
+ }
2136
+ }
2137
+ }
2138
+ const matrix = scenarioMatrix(
2139
+ journey.nodeIds,
2140
+ caseSet.cases.filter((testCase: any) => !isObsolete(testCase)),
2141
+ normalizeScenarioExclusions(caseSet.scenarioExclusions),
2142
+ );
1899
2143
  const artifactsDir = cfg.runtime.artifacts_dir || "./artifacts";
1900
2144
  const store: ObjectStore = createProjectObjectStore(cfg, artifactsDir);
1901
2145
  const projectId = cfg.project?.id ?? "default";
@@ -1912,12 +2156,63 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1912
2156
  journeyId: caseSet.journeyId,
1913
2157
  caseCount: caseSet.cases.length,
1914
2158
  objectKey,
2159
+ scenarios: {
2160
+ missing: matrix.missing,
2161
+ uncategorized: matrix.uncategorized,
2162
+ },
1915
2163
  nextSteps: [
2164
+ ...matrix.missing.map((gap) => gap.nodeId
2165
+ ? `Step ${gap.nodeId} has no ${gap.category} case: add one (${SCENARIO_MEANING[gap.category]})`
2166
+ : `No step covers ${gap.category}: add a case where it belongs (${SCENARIO_MEANING[gap.category]}), or a scenarioExclusions entry saying why it doesn't apply`),
2167
+ ...(matrix.uncategorized.length
2168
+ ? [`Give these cases a category so the scenario matrix can place them: ${matrix.uncategorized.join(", ")}`]
2169
+ : []),
1916
2170
  `Plan the journey with bugmole_plan for "${caseSet.journeyId}" so these cases can be executed`,
1917
2171
  ],
1918
2172
  };
1919
2173
  }
1920
2174
 
2175
+ case "bugmole_propose_test_case": {
2176
+ const action = args.action;
2177
+ if (!["add", "change", "retire"].includes(action)) throw new Error("action must be add, change or retire");
2178
+ const projectId = cfg.project?.id ?? "default";
2179
+ const store: ObjectStore = createProjectObjectStore(cfg, cfg.runtime.artifacts_dir || "./artifacts");
2180
+ const key = `${projectPrefix(projectId)}/test-cases/${args.caseSetId}.json`;
2181
+ if (!(await store.has(key))) throw new Error(`No case set ${args.caseSetId}; write it with bugmole_write_test_cases first`);
2182
+ const caseSet = JSON.parse(await store.getText(key));
2183
+ const current = activeRevision(caseSet, args.caseId);
2184
+ if (action !== "add" && !current) throw new Error(`${args.caseId} has no current revision in ${args.caseSetId}`);
2185
+ if (action === "add" && current) throw new Error(`${args.caseId} already exists at revision ${revisionOf(current)}; propose a change instead`);
2186
+ if (action !== "retire") {
2187
+ const problem = caseProblem({ ...(args.case ?? {}), caseId: args.caseId });
2188
+ if (problem) throw new Error(`case ${args.caseId}: ${problem}`);
2189
+ // Grounded like bugmole_write_test_cases: the step must exist in the journey.
2190
+ const journey = readJourneyGraph(cfg).journeys.find((item) => item.id === caseSet.journeyId);
2191
+ if (journey && !journey.nodeIds.includes(args.case.nodeId)) {
2192
+ throw new Error(`case ${args.caseId}: nodeId "${args.case.nodeId}" is not a step of journey ${caseSet.journeyId}`);
2193
+ }
2194
+ }
2195
+ const client = taskClient(cfg);
2196
+ const { proposal } = await client.proposeTestCaseChange({
2197
+ caseSetId: args.caseSetId,
2198
+ caseId: args.caseId,
2199
+ action,
2200
+ ...(current ? { baseRevision: revisionOf(current) } : {}),
2201
+ ...(action !== "retire" ? { case: args.case } : {}),
2202
+ summary: args.summary,
2203
+ source: "mcp",
2204
+ });
2205
+ // Accepted on arrival (extreme mode) and this project keeps its own files: apply it now.
2206
+ const applied = proposal.status === "accepted" && !proposal.appliedAt ? await applyAcceptedProposals(cfg, client) : [];
2207
+ return {
2208
+ status: "success",
2209
+ proposal: applied.length ? { ...proposal, applied: applied.find((item) => item.id === proposal.id) } : proposal,
2210
+ nextSteps: proposal.status === "pending"
2211
+ ? ["A project manager reviews this proposal on the Tests page; it becomes a new revision once accepted"]
2212
+ : [],
2213
+ };
2214
+ }
2215
+
1921
2216
  case "bugmole_write_test_case_results": {
1922
2217
  const resultSet = args.resultSet;
1923
2218
  if (!resultSet || typeof resultSet !== "object" || Array.isArray(resultSet)) {
@@ -1955,7 +2250,11 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1955
2250
  + `(${storedCaseSet.journeyRevision}) — re-run against the current assumptions`,
1956
2251
  );
1957
2252
  }
1958
- const knownCaseIds = new Set((storedCaseSet.cases ?? []).map((item: any) => item.caseId));
2253
+ // Only the current revision of a case can be judged; an obsolete one
2254
+ // was replaced before this run. Each verdict records which revision it
2255
+ // judged, so a later revision starts untested.
2256
+ const activeCases = new Map<string, any>((storedCaseSet.cases ?? []).filter((item: any) => !isObsolete(item)).map((item: any) => [item.caseId, item]));
2257
+ const knownCaseIds = new Set(activeCases.keys());
1959
2258
  const allowedStatuses = ["passed", "failed", "blocked", "needs_setup"];
1960
2259
  const seen = new Set<string>();
1961
2260
  for (const result of resultSet.results) {
@@ -1973,6 +2272,7 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1973
2272
  if (typeof result.detail !== "string" || !result.detail.trim()) {
1974
2273
  throw new Error(`result ${result.caseId}: detail is required — say what was observed`);
1975
2274
  }
2275
+ result.caseRevision = revisionOf(activeCases.get(result.caseId));
1976
2276
  }
1977
2277
  const record = {
1978
2278
  ...resultSet,
@@ -1981,8 +2281,9 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1981
2281
  journeyRevision: storedCaseSet.journeyRevision,
1982
2282
  recordedAt: resultSet.recordedAt ?? new Date().toISOString(),
1983
2283
  };
1984
- const objectKey = `${projectPrefix(projectId)}/test-case-results/${resultSet.caseSetId}.json`;
1985
- await store.put(objectKey, JSON.stringify(record, null, 2) + "\n", "application/json");
2284
+ // Latest verdicts plus one history entry per write, so earlier runs'
2285
+ // verdicts survive for the case's pass/fail trail.
2286
+ const { objectKey, historyKey } = await putTestCaseResults(store, projectId, record);
1986
2287
  const counts = resultSet.results.reduce((acc: Record<string, number>, item: any) => {
1987
2288
  acc[item.status] = (acc[item.status] ?? 0) + 1;
1988
2289
  return acc;
@@ -1995,6 +2296,7 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
1995
2296
  of: knownCaseIds.size,
1996
2297
  counts,
1997
2298
  objectKey,
2299
+ historyKey,
1998
2300
  };
1999
2301
  }
2000
2302
 
@@ -2273,9 +2575,14 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
2273
2575
  case "bugmole_task_next": {
2274
2576
  const client = taskClient(cfg);
2275
2577
  const listed = await client.listTasks();
2276
- const next = listed.tasks.find((task) => isClaimableTask(task));
2578
+ const next = listed.tasks.find((task) => isClaimableTask(task, Date.now(), workerApprovalMode()));
2277
2579
  if (!next) return { task: null, message: "No queued agent tasks are available." };
2278
- return client.claimTask(next.id, args.workerId);
2580
+ const claimed = await client.claimTask(next.id, args.workerId);
2581
+ // A task a person sent back from review carries what they want changed.
2582
+ if (claimed.task?.reviewNote && next.reviewNote) {
2583
+ return { ...claimed, message: `A reviewer sent this task back: ${claimed.task.reviewNote}. Address it before completing again.` };
2584
+ }
2585
+ return claimed;
2279
2586
  }
2280
2587
 
2281
2588
  case "bugmole_task_create": {
@@ -2297,7 +2604,20 @@ export async function callToolInternal(name: string, args: any, cfg: any): Promi
2297
2604
  const client = taskClient(cfg);
2298
2605
  const update = { ...args };
2299
2606
  delete update.taskId;
2300
- return client.updateTask(args.taskId, update);
2607
+ const result = await client.updateTask(args.taskId, update) as { task: { status: string }; inReview?: boolean; message?: string };
2608
+ // The project's approval mode may hold a completed task for a person.
2609
+ if (result.inReview || result.task?.status === "in_review") {
2610
+ return {
2611
+ ...result,
2612
+ inReview: true,
2613
+ message: result.message ?? "This task is now in review: a person accepts it or sends it back with a note.",
2614
+ nextSteps: [
2615
+ "Do not treat this work as accepted yet; a person reviews it on the Agent Tasks board.",
2616
+ "If it comes back, bugmole_task_next returns it with the reviewer's note in reviewNote.",
2617
+ ],
2618
+ };
2619
+ }
2620
+ return result;
2301
2621
  }
2302
2622
 
2303
2623
  case "bugmole_commit_response": {
@@ -2552,13 +2872,14 @@ export async function readResource(params: any, cfg: any): Promise<any> {
2552
2872
  "3. Call `bugmole_environment_get` for the selected environment. If it is missing or incomplete, call `bugmole_environment_set` with the safe base URL, mock/reference URL, device transport, device label, and reproducible start command before exploring.",
2553
2873
  "4. For an ADB environment, verify the named device is reachable and follow the recorded start command; do not guess a generic browser URL. For Lumis android-local, the expected transport is Nokia via ADB and the mock/reference URL is `https://gcore.otlrs.com/lumis-consumer/`.",
2554
2874
  "5. Call `bugmole_explore` with the environment ID; omit baseUrl when the environment contract provides it.",
2875
+ "5a. If the app needs sign-in, never ask the person for a password in chat. Call `bugmole_secrets_status` to see which secrets the flows type and where each comes from; if one is missing, ask the person to run `bugmole secrets set NAME` in their own terminal or map it to their vault (Google Secret Manager, AWS Secrets Manager) under secrets: in the Bugmole config. Then call `bugmole_sign_in` with the sign-in flow to get a signed-in session to explore with.",
2555
2876
  "6. If an authenticated browser or device session is available, explore one concrete task and call `bugmole_record_discovery` with screens, transitions, proof tokens, `verifiedControlIds` for every button/link/action checked, and `completedNodeIds` after the terminal action actually succeeds.",
2556
2877
  "7. After every exploration or discovery result, call `bugmole_journey_get` and `bugmole_journey_next` for the selected flow.",
2557
2878
  "8. Perform the returned nextTask, add/edit the flow with `bugmole_journey_add_step` or `bugmole_journey_add_transition`, then call `bugmole_journey_next` again.",
2558
2879
  "9. If the chart returns blocked, inspect its clarification question and check the remembered resolution. Ask the user when it is missing or stale; record their answer with `bugmole_resolve_blocker`, then retry the chart.",
2559
2880
  "10. Continue this chart-check loop until every transition is runtime-verified, the terminal goal is completed, and `bugmole_journey_next` returns exhausted or an explicit blocked reason that cannot be resolved.",
2560
- "11. Once the flow is exhausted, write its test case assumptions: read `bugmole://spec/test-cases.schema.json` and call `bugmole_write_test_cases` with one case per step worth proving. Each case must name a real `nodeId` from that journey, state what it is trying to prove, and list observable expectations. Cite what each assumption is based on in `basis`; a case with nothing behind it must be `confidence: \"low\"` rather than a confident guess.",
2561
- "12. Confirm the selected actor exists in `roles.yaml`, then call `bugmole_plan` only after the flow is exhausted or documented as blocked.",
2881
+ "11. Once the flow is exhausted, write its test case assumptions: read `bugmole://spec/test-cases.schema.json` and call `bugmole_write_test_cases` with one case per step worth proving. Each case must name a real `nodeId` from that journey, state what it is trying to prove, and list observable expectations. Cite what each assumption is based on in `basis`; a case with nothing behind it must be `confidence: \"low\"` rather than a confident guess. Give every case a `category`: each step needs a `happy_path` case, and the journey needs `validation`, `error_state`, `permission` and `edge_case` cases on the steps they belong to. When one truly doesn't apply (no input to validate, a single role), add a `scenarioExclusions` entry with the reason instead of leaving a silent gap. Follow the returned `nextSteps` until nothing is missing.",
2882
+ "12. Confirm the selected actor is a confirmed role (`bugmole_roles_list`), then call `bugmole_plan` only after the flow is exhausted or documented as blocked. Bugmole suggests roles and a person decides: if the flow needs a role that isn't confirmed, suggest it with `bugmole_role_propose` (with its evidence) and stop there — you cannot confirm a role yourself, and a suggested or rejected role is never planned for.",
2562
2883
  "13. After a run actually exercises those cases, record what it observed: read `bugmole://spec/test-case-results.schema.json` and call `bugmole_write_test_case_results` with one result per case you checked. This is the only thing that moves a case off \"planned\" — writing assumptions never counts as proof, so a flow whose cases were verified but never recorded reads as 0% covered. Say what you saw in `detail` (status code, element present or missing, assertion that failed), use \"failed\" when a check ran and the expectation did not hold, and \"blocked\" only when it could not be run at all. Do not record a result for a case you did not actually exercise.",
2563
2884
  "14. Commit progress/results with `bugmole_commit_response`, always including the returned `nextSteps` until the workflow is complete.",
2564
2885
  "",
@@ -2674,8 +2995,9 @@ export async function readResource(params: any, cfg: any): Promise<any> {
2674
2995
  "For taskType ui, read the parity guidance and schema, rerun affected flows, generate a fresh parity artifact, and include parityAuditId and parityMatchScore.",
2675
2996
  "A UI task cannot be completed without fresh parity evidence. If parity is blocked, leave the task blocked with concrete next steps.",
2676
2997
  "When finished, call bugmole_task_update with status completed, completionSummary, changedFiles, and required parity evidence.",
2998
+ "Depending on the project's approval mode the task then goes to review (inReview: true, status in_review) instead of completed: manual reviews every task, auto reviews code and ui tasks, extreme none. A person accepts it or sends it back; a task sent back returns through bugmole_task_next with the reviewer's note in reviewNote — address that note before completing again.",
2677
2999
  "",
2678
- "Finishing a task is not the end of the work. This queue drives an autonomous pipeline: explore, then define any missing actors in roles.yaml, then write test case assumptions with bugmole_write_test_cases, then bugmole_plan. If your task leaves an obvious next stage undone, queue it with bugmole_task_create before you complete, using a stable idempotencyKey of the form `pipeline:<projectId>:<stage>:<journeyId>` so it cannot duplicate an existing one.",
3000
+ "Finishing a task is not the end of the work. This queue drives an autonomous pipeline: explore, then suggest any missing roles with bugmole_role_propose (a person confirms them), then write test case assumptions with bugmole_write_test_cases, then bugmole_plan. If your task leaves an obvious next stage undone, queue it with bugmole_task_create before you complete, using a stable idempotencyKey of the form `pipeline:<projectId>:<stage>:<journeyId>` so it cannot duplicate an existing one.",
2679
3001
  "The worker independently reconciles the same pipeline from the artifacts on disk, so a follow-up you skip is still picked up. Never invent a stage that the evidence does not support just to keep the pipeline moving, and never mark work completed that you did not actually finish — a blocked task with concrete next steps is more useful than a false completion.",
2680
3002
  ].join("\n"),
2681
3003
  }],
@@ -2728,7 +3050,7 @@ export async function getPrompt(params: any, cfg: any): Promise<any> {
2728
3050
  "Use bugmole_explore for source/application discovery when a runnable base URL exists. If you have an authenticated browser session, inspect the owner dashboard there and persist concrete observations with bugmole_record_discovery.",
2729
3051
  "After every meaningful phase, call bugmole_report_progress, read the current journey chart, and call bugmole_journey_next.",
2730
3052
  "Perform the returned flow task, record its concrete screen/transition evidence, and repeat the chart check until it returns exhausted or blocked.",
2731
- "Before bugmole_plan, verify that the journey graph has real nodes/actions, no actionable next task remains, and the actor exists in roles.yaml.",
3053
+ "Before bugmole_plan, verify that the journey graph has real nodes/actions, no actionable next task remains, and the actor is a confirmed role (bugmole_roles_list). Suggest missing roles with bugmole_role_propose; only a person confirms them.",
2732
3054
  "If discovery is incomplete, return a blocked status with concrete next_steps instead of creating a generic login-only plan.",
2733
3055
  "After planning, read the generated plan resource, validate its steps against the discovered evidence, and only then hand it to the dashboard for execution.",
2734
3056
  ].join("\n"),
@@ -2812,7 +3134,7 @@ export async function getPrompt(params: any, cfg: any): Promise<any> {
2812
3134
  "You are continuing a dashboard-created Bugmole agent task.",
2813
3135
  "Read bugmole://guidance/agent-task-queue, then call bugmole_task_next with a stable workerId.",
2814
3136
  "Implement the claimed task and report progress. For UI tasks, rerun parity after the fix and do not complete without a fresh parity audit id and match score.",
2815
- "Call bugmole_task_update with completed or blocked status and concrete evidence.",
3137
+ "Call bugmole_task_update with completed or blocked status and concrete evidence. If the response says inReview, a person reviews the work before it counts as completed.",
2816
3138
  ].join("\n"),
2817
3139
  },
2818
3140
  }],
@@ -3053,6 +3375,16 @@ export function writeSpecFile(filePath: string, content: string, cfg: any): any
3053
3375
  const dir = path.dirname(fullPath);
3054
3376
  fs.mkdirSync(dir, { recursive: true });
3055
3377
 
3378
+ // An agent can suggest roles in roles.yaml but not confirm them: new or
3379
+ // promoted roles are written `status: suggested` for a person to decide.
3380
+ let heldBack: string[] = [];
3381
+ if (path.normalize(filePath) === "roles.yaml") {
3382
+ const previous = fs.existsSync(fullPath) ? fs.readFileSync(fullPath, "utf-8") : null;
3383
+ const held = holdAgentRoleWrites(previous, content);
3384
+ content = held.content;
3385
+ heldBack = held.heldBack;
3386
+ }
3387
+
3056
3388
  // Write file
3057
3389
  fs.writeFileSync(fullPath, content, "utf-8");
3058
3390
 
@@ -3060,5 +3392,9 @@ export function writeSpecFile(filePath: string, content: string, cfg: any): any
3060
3392
  success: true,
3061
3393
  path: filePath,
3062
3394
  size: content.length,
3395
+ ...(heldBack.length ? {
3396
+ suggestedRoles: heldBack,
3397
+ message: `Written as suggestions: ${heldBack.join(", ")}. A person confirms or rejects them on the Roles page before they are used for test cases or runs.`,
3398
+ } : {}),
3063
3399
  };
3064
3400
  }