@bugmole/cli 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.bugmole.env.example +20 -0
- package/LICENSE +7 -0
- package/README.md +293 -0
- package/TESTING.md +117 -0
- package/bugmole.config.yaml +39 -0
- package/package.json +85 -0
- package/scripts/billing/paypal-setup.mjs +121 -0
- package/scripts/bugmole-continue.ts +318 -0
- package/scripts/bugmole-init.ts +188 -0
- package/scripts/bugmole.cjs +17 -0
- package/scripts/bugmole.test.ts +344 -0
- package/scripts/bugmole.ts +657 -0
- package/scripts/ensure-maestro.cjs +79 -0
- package/scripts/ios-tunnel-keeper.sh +45 -0
- package/scripts/ios-wda-keeper.sh +66 -0
- package/scripts/sync-plan-catalog.d.mts +3 -0
- package/scripts/sync-plan-catalog.mjs +16 -0
- package/scripts/ui-parity-diff.py +65 -0
- package/scripts/ui-parity-requirements.txt +1 -0
- package/scripts/verify-manage-to-plans.mts +194 -0
- package/spec/app-ui-audit.schema.json +176 -0
- package/spec/blockers.yaml +79 -0
- package/spec/bugs.index.json +42 -0
- package/spec/design-dna.schema.json +38 -0
- package/spec/domain_rules.yaml +24 -0
- package/spec/flows/flow_manage-to-plans-64c3c61b.yaml +8 -0
- package/spec/flows/flow_screen-to-evidence-6c3ab309.yaml +7 -0
- package/spec/journey_graph.yaml +126 -0
- package/spec/journeys.graph.json +2618 -0
- package/spec/plans/dashboard-smoke.flow.yaml +20 -0
- package/spec/plans/example.flow.yaml +99 -0
- package/spec/plans/flow_api-keys-to-logout-7396cd5d.flow.yaml +33 -0
- package/spec/plans/flow_api-keys-to-screen-157e10a6.flow.yaml +33 -0
- package/spec/plans/flow_manage-to-audits-176c1eb8.flow.yaml +33 -0
- package/spec/plans/flow_manage-to-blockers-633282c8.flow.yaml +112 -0
- package/spec/plans/flow_manage-to-devices-ad57e202.flow.yaml +33 -0
- package/spec/plans/flow_manage-to-logout-86c54656.flow.yaml +33 -0
- package/spec/plans/flow_manage-to-manage-resource-action-6325f99a.flow.yaml +185 -0
- package/spec/plans/flow_manage-to-parity-issues-01fa2579.flow.yaml +34 -0
- package/spec/plans/flow_manage-to-plans-64c3c61b.flow.yaml +123 -0
- package/spec/plans/flow_manage-to-screen-6857f915.flow.yaml +106 -0
- package/spec/plans/flow_manage-to-tasks-c5791c95.flow.yaml +33 -0
- package/spec/plans/flow_profile-to-screen-6045aa3d.flow.yaml +33 -0
- package/spec/plans/flow_register-to-api-auth-login-460a9cc8.flow.yaml +34 -0
- package/spec/plans/flow_screen-to-audits-07666356.flow.yaml +33 -0
- package/spec/plans/flow_screen-to-blockers-091d258b.flow.yaml +33 -0
- package/spec/plans/flow_screen-to-devices-cad72f75.flow.yaml +33 -0
- package/spec/plans/flow_screen-to-evidence-6c3ab309.flow.yaml +126 -0
- package/spec/plans/flow_screen-to-parity-issues-2218e3ad.flow.yaml +34 -0
- package/spec/plans/flow_screen-to-plans-8232a94f.flow.yaml +33 -0
- package/spec/plans/flow_screen-to-tasks-540671d1.flow.yaml +33 -0
- package/spec/plans/login.flow.yaml +22 -0
- package/spec/plans/owner-operations.flow.yaml +20 -0
- package/spec/project_config.yaml +55 -0
- package/spec/roles.yaml +30 -0
- package/spec/schema.md +394 -0
- package/spec/test-case-results.schema.json +62 -0
- package/spec/test-cases.schema.json +85 -0
- package/spec/ui-parity-audit.schema.json +194 -0
- package/spec/ui-reverse-engineering.schema.json +94 -0
- package/src/billing/plan-catalog.test.ts +46 -0
- package/src/billing/plan-catalog.ts +199 -0
- package/src/integrations/aws-sigv4.test.ts +42 -0
- package/src/integrations/aws-sigv4.ts +72 -0
- package/src/integrations/device-farm.ts +155 -0
- package/src/integrations/github-app.test.ts +57 -0
- package/src/integrations/github-app.ts +143 -0
- package/src/integrations/gitlab.ts +81 -0
- package/src/integrations/temp-email.test.ts +123 -0
- package/src/integrations/temp-email.ts +175 -0
- package/src/integrations/testflight-feedback.test.ts +51 -0
- package/src/integrations/testflight-feedback.ts +173 -0
- package/src/integrations/webdriver-client.ts +131 -0
- package/src/mcp/server.test.ts +1220 -0
- package/src/mcp/server.ts +3064 -0
- package/src/mcp/write-test-cases.test.ts +287 -0
- package/src/registry/api-key-client.ts +39 -0
- package/src/registry/control-plane-client.ts +212 -0
- package/src/registry/migrations/0001_registry.sql +47 -0
- package/src/registry/migrations/0002_device_authorizations.sql +23 -0
- package/src/registry/migrations/0003_project_environments.sql +25 -0
- package/src/registry/migrations/0004_device_authorization_email.sql +1 -0
- package/src/registry/migrations/0005_testing_control_plane.sql +55 -0
- package/src/registry/migrations/0006_workspaces.sql +36 -0
- package/src/registry/migrations/0007_project_apps.sql +26 -0
- package/src/registry/migrations/0008_agent_tasks.sql +30 -0
- package/src/registry/migrations/0009_journey_revisions.sql +17 -0
- package/src/registry/migrations/0010_agent_task_journey.sql +2 -0
- package/src/registry/migrations/0011_run_journey_revision.sql +1 -0
- package/src/registry/migrations/0012_canonical_flow_execution.sql +10 -0
- package/src/registry/migrations/0013_device_sessions.sql +22 -0
- package/src/registry/migrations/0014_agent_task_step.sql +1 -0
- package/src/registry/migrations/0015_agent_task_attempts.sql +6 -0
- package/src/registry/migrations/0016_github_issue_tracker.sql +54 -0
- package/src/registry/migrations/0017_run_targets.sql +7 -0
- package/src/registry/migrations/0018_run_fix_from.sql +4 -0
- package/src/registry/migrations/0019_workspace_flags.sql +9 -0
- package/src/registry/migrations/0020_orgs.sql +40 -0
- package/src/registry/migrations/0021_billing_core.sql +58 -0
- package/src/registry/migrations/0022_cloud_runners.sql +19 -0
- package/src/registry/migrations/0023_signup.sql +4 -0
- package/src/registry/migrations/0024_billing.sql +67 -0
- package/src/registry/migrations/0025_notifications.sql +47 -0
- package/src/registry/migrations/0026_repo_bindings.sql +28 -0
- package/src/registry/migrations/0027_feedback.sql +29 -0
- package/src/registry/migrations/0028_devices.sql +48 -0
- package/src/registry/migrations/0029_sso.sql +31 -0
- package/src/registry/migrations/0030_workspace_domains.sql +18 -0
- package/src/registry/migrations/0031_gitlab_and_teams.sql +45 -0
- package/src/registry/migrations/0032_personas.sql +15 -0
- package/src/registry/migrations/0033_bugmole_rename.sql +11 -0
- package/src/registry/task-scheduling.test.ts +100 -0
- package/src/registry/task-scheduling.ts +80 -0
- package/src/registry-worker/ai/platform-model.ts +77 -0
- package/src/registry-worker/ai/routes.test.ts +88 -0
- package/src/registry-worker/ai/routes.ts +90 -0
- package/src/registry-worker/artifacts.test.ts +98 -0
- package/src/registry-worker/artifacts.ts +85 -0
- package/src/registry-worker/billing/billing-core.test.ts +175 -0
- package/src/registry-worker/billing/checkout-routes.ts +209 -0
- package/src/registry-worker/billing/enforcement.ts +69 -0
- package/src/registry-worker/billing/entitlements.ts +108 -0
- package/src/registry-worker/billing/ledger.ts +186 -0
- package/src/registry-worker/billing/paypal/api.ts +259 -0
- package/src/registry-worker/billing/paypal/client.ts +91 -0
- package/src/registry-worker/billing/paypal/provider.ts +143 -0
- package/src/registry-worker/billing/paypal.test.ts +466 -0
- package/src/registry-worker/billing/provider.ts +114 -0
- package/src/registry-worker/billing/routes.ts +66 -0
- package/src/registry-worker/billing/subscriptions.ts +780 -0
- package/src/registry-worker/billing/thresholds.ts +107 -0
- package/src/registry-worker/core.ts +308 -0
- package/src/registry-worker/devices/devices.test.ts +185 -0
- package/src/registry-worker/devices/policy.ts +71 -0
- package/src/registry-worker/devices/routes.ts +453 -0
- package/src/registry-worker/domains/domains.test.ts +210 -0
- package/src/registry-worker/domains/routes.ts +139 -0
- package/src/registry-worker/domains.ts +88 -0
- package/src/registry-worker/email/sender.ts +75 -0
- package/src/registry-worker/env.d.ts +14716 -0
- package/src/registry-worker/features.ts +20 -0
- package/src/registry-worker/feedback/feedback.test.ts +230 -0
- package/src/registry-worker/feedback/format.ts +148 -0
- package/src/registry-worker/feedback/routes.ts +386 -0
- package/src/registry-worker/flags.ts +39 -0
- package/src/registry-worker/github/checks.test.ts +177 -0
- package/src/registry-worker/github/checks.ts +374 -0
- package/src/registry-worker/gitlab/checks.test.ts +141 -0
- package/src/registry-worker/gitlab/checks.ts +349 -0
- package/src/registry-worker/hooks.ts +54 -0
- package/src/registry-worker/index.ts +2077 -0
- package/src/registry-worker/jobs/index.ts +29 -0
- package/src/registry-worker/jobs/retention.ts +68 -0
- package/src/registry-worker/mcp/mcp.test.ts +355 -0
- package/src/registry-worker/mcp/routes.ts +215 -0
- package/src/registry-worker/mcp/token.ts +126 -0
- package/src/registry-worker/mcp/tools.ts +563 -0
- package/src/registry-worker/notifications/alerts.ts +212 -0
- package/src/registry-worker/notifications/notifications.test.ts +298 -0
- package/src/registry-worker/notifications/outbox.ts +83 -0
- package/src/registry-worker/notifications/routes.ts +280 -0
- package/src/registry-worker/notifications/secrets.ts +49 -0
- package/src/registry-worker/notifications/slack.ts +96 -0
- package/src/registry-worker/notifications/teams.ts +46 -0
- package/src/registry-worker/org/audit.ts +116 -0
- package/src/registry-worker/org/routes.test.ts +163 -0
- package/src/registry-worker/org/routes.ts +302 -0
- package/src/registry-worker/personas/personas.test.ts +78 -0
- package/src/registry-worker/personas/routes.ts +100 -0
- package/src/registry-worker/repo-triggers.ts +20 -0
- package/src/registry-worker/routes/index.ts +74 -0
- package/src/registry-worker/run-events.ts +24 -0
- package/src/registry-worker/runner/dispatch.ts +219 -0
- package/src/registry-worker/runner/jobs.ts +43 -0
- package/src/registry-worker/runner/metering.ts +82 -0
- package/src/registry-worker/runner/policy.ts +59 -0
- package/src/registry-worker/runner/routes.ts +171 -0
- package/src/registry-worker/runner/runner.test.ts +358 -0
- package/src/registry-worker/runner/tokens.ts +93 -0
- package/src/registry-worker/runs.test.ts +60 -0
- package/src/registry-worker/signup/policy.ts +57 -0
- package/src/registry-worker/signup/routes.ts +106 -0
- package/src/registry-worker/signup/signup.test.ts +81 -0
- package/src/registry-worker/sso/aegis.ts +141 -0
- package/src/registry-worker/sso/membership.ts +157 -0
- package/src/registry-worker/sso/routes.ts +458 -0
- package/src/registry-worker/sso/sso.test.ts +344 -0
- package/src/registry-worker/testing/d1-shim.ts +180 -0
- package/src/registry-worker/testing/harness.ts +137 -0
- package/src/runner-worker/index.ts +108 -0
- package/src/runtime/ai-analysis.ts +97 -0
- package/src/runtime/ai-exploration.test.ts +32 -0
- package/src/runtime/ai-exploration.ts +69 -0
- package/src/runtime/ai-repair.ts +74 -0
- package/src/runtime/ai-work.test.ts +99 -0
- package/src/runtime/android-screen-record.test.ts +75 -0
- package/src/runtime/android-screen-record.ts +192 -0
- package/src/runtime/app-understanding.test.ts +123 -0
- package/src/runtime/app-understanding.ts +201 -0
- package/src/runtime/appium-driver.test.ts +179 -0
- package/src/runtime/appium-driver.ts +295 -0
- package/src/runtime/blocker-resolution.test.ts +113 -0
- package/src/runtime/blocker-resolution.ts +111 -0
- package/src/runtime/browser-matrix.integration.test.ts +212 -0
- package/src/runtime/browser-matrix.test.ts +143 -0
- package/src/runtime/browser-matrix.ts +200 -0
- package/src/runtime/canonical-flow.test.ts +52 -0
- package/src/runtime/config-validate.ts +185 -0
- package/src/runtime/continuous-execution.ts +291 -0
- package/src/runtime/cursor-applescript.ts +573 -0
- package/src/runtime/cursor-cli-driver.test.ts +78 -0
- package/src/runtime/cursor-cli-driver.ts +156 -0
- package/src/runtime/cursor-driver-example.ts +117 -0
- package/src/runtime/cursor-driver-index.ts +65 -0
- package/src/runtime/cursor-driver-init.ts +277 -0
- package/src/runtime/cursor-driver-run.test.ts +15 -0
- package/src/runtime/cursor-driver-run.ts +323 -0
- package/src/runtime/cursor-driver.ts +332 -0
- package/src/runtime/cursor-llm-example.ts +90 -0
- package/src/runtime/cursor-llm.ts +206 -0
- package/src/runtime/cursor-mcp-monitor.ts +386 -0
- package/src/runtime/device-clouds/browserstack.ts +73 -0
- package/src/runtime/device-clouds/device-farm.ts +52 -0
- package/src/runtime/device-clouds/index.ts +92 -0
- package/src/runtime/device-clouds/kobiton.ts +70 -0
- package/src/runtime/device-clouds/targets.ts +44 -0
- package/src/runtime/device-clouds/types.ts +62 -0
- package/src/runtime/diff-proposal.ts +84 -0
- package/src/runtime/discovery-task.test.ts +29 -0
- package/src/runtime/discovery-task.ts +284 -0
- package/src/runtime/driver-recovery.ts +69 -0
- package/src/runtime/driver.ts +79 -0
- package/src/runtime/environment.test.ts +104 -0
- package/src/runtime/environment.ts +137 -0
- package/src/runtime/executor.test.ts +509 -0
- package/src/runtime/executor.ts +921 -0
- package/src/runtime/explorer.test.ts +101 -0
- package/src/runtime/explorer.ts +1013 -0
- package/src/runtime/failure-analysis.test.ts +111 -0
- package/src/runtime/failure-analysis.ts +272 -0
- package/src/runtime/fixtures/fake-maestro.sh +36 -0
- package/src/runtime/flow-language.test.ts +268 -0
- package/src/runtime/flow-language.ts +414 -0
- package/src/runtime/init-wizard.ts +354 -0
- package/src/runtime/ios-screen-record.test.ts +68 -0
- package/src/runtime/ios-screen-record.ts +155 -0
- package/src/runtime/journey-editor.ts +452 -0
- package/src/runtime/journey-evidence.test.ts +161 -0
- package/src/runtime/journey-evidence.ts +180 -0
- package/src/runtime/journey-graph.test.ts +257 -0
- package/src/runtime/journey-graph.ts +170 -0
- package/src/runtime/legacy-names.ts +32 -0
- package/src/runtime/llm-example.ts +105 -0
- package/src/runtime/llm.ts +527 -0
- package/src/runtime/local-browser.test.ts +45 -0
- package/src/runtime/local-browser.ts +48 -0
- package/src/runtime/local-registry-stub.test.ts +325 -0
- package/src/runtime/local-registry-stub.ts +803 -0
- package/src/runtime/maestro-driver.test.ts +84 -0
- package/src/runtime/maestro-driver.ts +209 -0
- package/src/runtime/mole-voice.ts +21 -0
- package/src/runtime/nav-crawl.test.ts +100 -0
- package/src/runtime/nav-crawl.ts +153 -0
- package/src/runtime/pipeline.test.ts +405 -0
- package/src/runtime/pipeline.ts +833 -0
- package/src/runtime/planner.test.ts +37 -0
- package/src/runtime/planner.ts +274 -0
- package/src/runtime/platform-ai.ts +76 -0
- package/src/runtime/playwright-driver.test.ts +93 -0
- package/src/runtime/playwright-driver.ts +620 -0
- package/src/runtime/project-spec.ts +140 -0
- package/src/runtime/record-run-verdicts.ts +68 -0
- package/src/runtime/reporter.test.ts +56 -0
- package/src/runtime/reporter.ts +158 -0
- package/src/runtime/reset.test.ts +44 -0
- package/src/runtime/reset.ts +61 -0
- package/src/runtime/reviewer.test.ts +73 -0
- package/src/runtime/reviewer.ts +158 -0
- package/src/runtime/run-job.ts +136 -0
- package/src/runtime/run-once.test.ts +207 -0
- package/src/runtime/run-once.ts +168 -0
- package/src/runtime/run.ts +132 -0
- package/src/runtime/screen-recording.ts +34 -0
- package/src/runtime/serve-gateway.test.ts +74 -0
- package/src/runtime/serve-gateway.ts +164 -0
- package/src/runtime/serve-worker.test.ts +23 -0
- package/src/runtime/serve-worker.ts +278 -0
- package/src/runtime/site-discovery.test.ts +168 -0
- package/src/runtime/site-discovery.ts +308 -0
- package/src/runtime/target-runner.ts +144 -0
- package/src/runtime/test-case-verdicts.test.ts +94 -0
- package/src/runtime/test-case-verdicts.ts +120 -0
- package/src/runtime/ui-reverse-engineering/coordinator.test.ts +59 -0
- package/src/runtime/ui-reverse-engineering/coordinator.ts +391 -0
- package/src/runtime/ui-reverse-engineering/types.ts +197 -0
- package/src/runtime/web-suite.test.ts +97 -0
- package/src/runtime/web-suite.ts +176 -0
- package/src/storage/create-object-store.ts +144 -0
- package/src/storage/keys.ts +34 -0
- package/src/storage/local-artifact-server.test.ts +314 -0
- package/src/storage/local-artifact-server.ts +357 -0
- package/src/storage/object-store.test.ts +28 -0
- package/src/storage/object-store.ts +101 -0
- package/src/storage/registry-object-store.ts +88 -0
- package/src/storage/remote-object-store.ts +104 -0
- package/src/storage/storage-directory.test.ts +43 -0
- package/src/storage/storage-directory.ts +24 -0
- package/tsconfig.json +24 -0
|
@@ -0,0 +1,833 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Autonomous QA pipeline.
|
|
3
|
+
*
|
|
4
|
+
* Pulling "Explore this app" is meant to start the platform working on its
|
|
5
|
+
* own — explore, then assume flows, then assume test cases, then plan —
|
|
6
|
+
* and keep going until it is done or something says stop. Previously the
|
|
7
|
+
* worker ran exactly one task and stopped: nothing enqueued follow-on
|
|
8
|
+
* work, and the agent guidance only ever said "do this one task and
|
|
9
|
+
* report". A real run finished exploration and then sat idle forever.
|
|
10
|
+
*
|
|
11
|
+
* The agent may enqueue its own next stage, but progress cannot depend on
|
|
12
|
+
* that: the agent demonstrably died mid-pipeline (its CLI was killed by a
|
|
13
|
+
* timeout) and nothing recovered. So the worker also reconciles. Stage
|
|
14
|
+
* state is DERIVED from real artifacts on every pass rather than stored as
|
|
15
|
+
* a separate state machine, which means the two paths converge instead of
|
|
16
|
+
* fighting, and a half-finished pipeline repairs itself rather than
|
|
17
|
+
* drifting out of sync with what is actually on disk.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import fs from "node:fs";
|
|
21
|
+
import path from "node:path";
|
|
22
|
+
import YAML from "yaml";
|
|
23
|
+
import { createProjectObjectStore } from "../storage/create-object-store.js";
|
|
24
|
+
import { projectPrefix } from "../storage/keys.js";
|
|
25
|
+
import { MAX_TASK_ATTEMPTS } from "../registry/task-scheduling.js";
|
|
26
|
+
import { driverForPlatform } from "./planner.js";
|
|
27
|
+
|
|
28
|
+
export type PipelineStage = "explore" | "roles" | "audit" | "testcases" | "plan" | "run";
|
|
29
|
+
|
|
30
|
+
export type PipelineStatus = "running" | "paused" | "stopped" | "complete";
|
|
31
|
+
|
|
32
|
+
export type PipelineJourney = {
|
|
33
|
+
id: string;
|
|
34
|
+
actor?: string;
|
|
35
|
+
revision?: number;
|
|
36
|
+
nodeIds?: string[];
|
|
37
|
+
stability?: string;
|
|
38
|
+
linkedToGraph?: boolean;
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
export type PipelineTask = {
|
|
42
|
+
status: string;
|
|
43
|
+
idempotencyKey?: string;
|
|
44
|
+
completionSummary?: string | null;
|
|
45
|
+
title?: string;
|
|
46
|
+
attempts?: number;
|
|
47
|
+
errorMessage?: string | null;
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
export type PipelineInputs = {
|
|
51
|
+
projectId: string;
|
|
52
|
+
journeys: PipelineJourney[];
|
|
53
|
+
/** Actor names defined in roles.yaml. */
|
|
54
|
+
knownActors: string[];
|
|
55
|
+
/** Journey ids that already have a written test-case set. */
|
|
56
|
+
caseSetJourneyIds: string[];
|
|
57
|
+
/** Journey ids that already have an App Audit artifact. */
|
|
58
|
+
auditedJourneyIds?: string[];
|
|
59
|
+
/** Journey ids that already have a published plan. */
|
|
60
|
+
plannedJourneyIds: string[];
|
|
61
|
+
/** Journey ids that already have a run (of any status). */
|
|
62
|
+
ranJourneyIds?: string[];
|
|
63
|
+
/**
|
|
64
|
+
* Every recorded run with its status.
|
|
65
|
+
*
|
|
66
|
+
* Needed because "has a run" is not enough to decide what to do next: a run
|
|
67
|
+
* still in flight must not be duplicated, while a run that already finished
|
|
68
|
+
* must not block a fresh attempt. Keyed only on ids, those two cases look
|
|
69
|
+
* identical — which is how a finished run ended up being handed back
|
|
70
|
+
* forever under a stable idempotency key.
|
|
71
|
+
*/
|
|
72
|
+
runs?: Array<{ journeyId?: string; status?: string }>;
|
|
73
|
+
/**
|
|
74
|
+
* Journey ids where every written test case carries a recorded verdict.
|
|
75
|
+
*
|
|
76
|
+
* A run artifact only proves a run happened. This proves the run actually
|
|
77
|
+
* judged the cases it was supposed to, which is the difference between
|
|
78
|
+
* "executed" and "covered".
|
|
79
|
+
*/
|
|
80
|
+
verdictJourneyIds?: string[];
|
|
81
|
+
/** Published plans, used to start a run for a planned journey. */
|
|
82
|
+
publishedPlans?: Array<{ planId: string; journeyId?: string }>;
|
|
83
|
+
/** Environment to run against, when one is configured. */
|
|
84
|
+
environmentId?: string;
|
|
85
|
+
/** Current agent tasks for the project. */
|
|
86
|
+
tasks: PipelineTask[];
|
|
87
|
+
/** Blocker categories declared hard_stop in spec/blockers.yaml. */
|
|
88
|
+
hardStopCategories: string[];
|
|
89
|
+
/** Blocker categories actually encountered and recorded. */
|
|
90
|
+
recordedBlockerCategories: string[];
|
|
91
|
+
/** Set when a stage is waiting on a human answer. */
|
|
92
|
+
pendingClarification?: string;
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
export type PipelineNextTask = {
|
|
96
|
+
stage: PipelineStage;
|
|
97
|
+
journeyId?: string;
|
|
98
|
+
title: string;
|
|
99
|
+
description: string;
|
|
100
|
+
taskType: "code" | "ui" | "investigation";
|
|
101
|
+
idempotencyKey: string;
|
|
102
|
+
};
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* The run stage produces a run, not an agent task: executing a published
|
|
106
|
+
* plan is the worker's own job (processRun), and it is what turns a test
|
|
107
|
+
* case assumption into evidence. Without this the pipeline stopped at
|
|
108
|
+
* "planned" and every case stayed unproven.
|
|
109
|
+
*/
|
|
110
|
+
export type PipelineNextRun = {
|
|
111
|
+
stage: "run";
|
|
112
|
+
journeyId: string;
|
|
113
|
+
planId: string;
|
|
114
|
+
idempotencyKey: string;
|
|
115
|
+
environmentId?: string;
|
|
116
|
+
};
|
|
117
|
+
|
|
118
|
+
export type PipelineState = {
|
|
119
|
+
status: PipelineStatus;
|
|
120
|
+
reason: string;
|
|
121
|
+
next?: PipelineNextTask;
|
|
122
|
+
nextRun?: PipelineNextRun;
|
|
123
|
+
stages: Array<{ stage: PipelineStage; satisfied: boolean; detail: string }>;
|
|
124
|
+
/**
|
|
125
|
+
* Artifacts written against flows that no longer exist. Surfaced rather
|
|
126
|
+
* than quietly ignored: they are the visible trace of the graph having
|
|
127
|
+
* been regenerated, and they are why a stage can look further along than
|
|
128
|
+
* it is.
|
|
129
|
+
*/
|
|
130
|
+
orphanedArtifacts: Array<{ kind: string; journeyId: string }>;
|
|
131
|
+
/** Stages that gave up and need a human answer before work can continue. */
|
|
132
|
+
blockedTasks?: Array<{ title: string; idempotencyKey?: string; attempts: number; reason?: string }>;
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
export const PIPELINE_KEY_PREFIX = "pipeline";
|
|
136
|
+
|
|
137
|
+
/** Deterministic per-stage key, so the agent's own enqueue and the
|
|
138
|
+
* reconciler's converge on one task instead of creating duplicates. */
|
|
139
|
+
export function pipelineIdempotencyKey(
|
|
140
|
+
projectId: string,
|
|
141
|
+
stage: PipelineStage,
|
|
142
|
+
journeyId?: string,
|
|
143
|
+
): string {
|
|
144
|
+
return journeyId
|
|
145
|
+
? `${PIPELINE_KEY_PREFIX}:${projectId}:${stage}:${journeyId}`
|
|
146
|
+
: `${PIPELINE_KEY_PREFIX}:${projectId}:${stage}`;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
export function isPipelineTask(task: PipelineTask): boolean {
|
|
150
|
+
return typeof task.idempotencyKey === "string"
|
|
151
|
+
&& task.idempotencyKey.startsWith(`${PIPELINE_KEY_PREFIX}:`);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** A journey the graph considers real enough to build on. */
|
|
155
|
+
function isCanonical(journey: PipelineJourney): boolean {
|
|
156
|
+
return journey.stability === "stable" && journey.linkedToGraph === true;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function activeTask(tasks: PipelineTask[]): PipelineTask | undefined {
|
|
160
|
+
return tasks.find((task) => task.status === "queued" || task.status === "running");
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Decide what the pipeline should do next.
|
|
165
|
+
*
|
|
166
|
+
* Pure on purpose: every stop rule is a branch here, so each can be proven
|
|
167
|
+
* in isolation rather than only observable by running a live agent.
|
|
168
|
+
*/
|
|
169
|
+
export function derivePipelineState(inputs: PipelineInputs): PipelineState {
|
|
170
|
+
const {
|
|
171
|
+
projectId, journeys, knownActors, caseSetJourneyIds,
|
|
172
|
+
plannedJourneyIds, tasks, hardStopCategories, recordedBlockerCategories,
|
|
173
|
+
} = inputs;
|
|
174
|
+
|
|
175
|
+
const withNodes = journeys.filter((journey) => (journey.nodeIds?.length ?? 0) > 0);
|
|
176
|
+
const canonical = withNodes.filter(isCanonical);
|
|
177
|
+
// Artifacts outlive the flows they describe: regenerating the graph
|
|
178
|
+
// gives journeys new ids, orphaning every case set, audit and plan
|
|
179
|
+
// written against the old ones. Counting those raw made the pipeline
|
|
180
|
+
// report "4/7 flows have test cases" when none of the live flows had
|
|
181
|
+
// any, and mark planning satisfied while nothing was planned — so it
|
|
182
|
+
// would have skipped the stage entirely. Only artifacts belonging to a
|
|
183
|
+
// flow that still exists count.
|
|
184
|
+
const canonicalIds = new Set(canonical.map((journey) => journey.id));
|
|
185
|
+
const belongsToLiveFlow = (id: string) => canonicalIds.has(id);
|
|
186
|
+
const caseSets = new Set(caseSetJourneyIds.filter(belongsToLiveFlow));
|
|
187
|
+
const planned = new Set(plannedJourneyIds.filter(belongsToLiveFlow));
|
|
188
|
+
const orphanedArtifacts = [
|
|
189
|
+
...caseSetJourneyIds.filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "test cases", journeyId: id })),
|
|
190
|
+
...(inputs.auditedJourneyIds ?? []).filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "audit", journeyId: id })),
|
|
191
|
+
...plannedJourneyIds.filter((id) => !belongsToLiveFlow(id)).map((id) => ({ kind: "plan", journeyId: id })),
|
|
192
|
+
];
|
|
193
|
+
const missingActors = [...new Set(
|
|
194
|
+
canonical
|
|
195
|
+
.map((journey) => journey.actor)
|
|
196
|
+
.filter((actor): actor is string => Boolean(actor) && !knownActors.includes(actor!)),
|
|
197
|
+
)];
|
|
198
|
+
const audited = new Set((inputs.auditedJourneyIds ?? []).filter(belongsToLiveFlow));
|
|
199
|
+
const needsAudit = canonical.filter((journey) => !audited.has(journey.id));
|
|
200
|
+
const needsCases = canonical.filter((journey) => !caseSets.has(journey.id));
|
|
201
|
+
const needsPlan = canonical.filter((journey) => caseSets.has(journey.id) && !planned.has(journey.id));
|
|
202
|
+
const ran = new Set((inputs.ranJourneyIds ?? []).filter(belongsToLiveFlow));
|
|
203
|
+
const judged = new Set((inputs.verdictJourneyIds ?? []).filter(belongsToLiveFlow));
|
|
204
|
+
const ranButUnjudged = [...ran].filter((id) => !judged.has(id));
|
|
205
|
+
// The run stage is about planned flows, so only verdicts for flows that
|
|
206
|
+
// are currently planned count toward it. Without the intersection a flow
|
|
207
|
+
// judged under a plan that has since been discarded still counted, and the
|
|
208
|
+
// stage read "20/2 planned flows judged" — more judged than exist.
|
|
209
|
+
const judgedAndPlanned = [...judged].filter((id) => planned.has(id));
|
|
210
|
+
// Runs still in flight, and how many have already finished per flow. A
|
|
211
|
+
// finished run must not satisfy a fresh attempt, and an in-flight one must
|
|
212
|
+
// not be duplicated.
|
|
213
|
+
const TERMINAL_RUN = ["success", "partial", "error", "cancelled"];
|
|
214
|
+
const runsInFlight = new Set(
|
|
215
|
+
(inputs.runs ?? [])
|
|
216
|
+
.filter((run) => run.journeyId && !TERMINAL_RUN.includes(run.status ?? ""))
|
|
217
|
+
.map((run) => run.journeyId!),
|
|
218
|
+
);
|
|
219
|
+
const finishedRunCounts = new Map<string, number>();
|
|
220
|
+
for (const run of inputs.runs ?? []) {
|
|
221
|
+
if (!run.journeyId || !TERMINAL_RUN.includes(run.status ?? "")) continue;
|
|
222
|
+
finishedRunCounts.set(run.journeyId, (finishedRunCounts.get(run.journeyId) ?? 0) + 1);
|
|
223
|
+
}
|
|
224
|
+
// A flow that ran but recorded no verdicts still needs running: the run
|
|
225
|
+
// produced artifacts without judging anything, so re-running it is the only
|
|
226
|
+
// way the cases get a status. Keyed off judged rather than ran so the
|
|
227
|
+
// pipeline cannot declare itself finished over unjudged cases.
|
|
228
|
+
const needsRun = canonical.filter((journey) => planned.has(journey.id) && !judged.has(journey.id));
|
|
229
|
+
|
|
230
|
+
const stages: PipelineState["stages"] = [
|
|
231
|
+
{
|
|
232
|
+
stage: "explore",
|
|
233
|
+
satisfied: withNodes.length > 0,
|
|
234
|
+
detail: `${withNodes.length} discovered flow${withNodes.length === 1 ? "" : "s"}`,
|
|
235
|
+
},
|
|
236
|
+
{
|
|
237
|
+
stage: "roles",
|
|
238
|
+
satisfied: missingActors.length === 0,
|
|
239
|
+
detail: missingActors.length
|
|
240
|
+
? `undefined actor${missingActors.length === 1 ? "" : "s"}: ${missingActors.join(", ")}`
|
|
241
|
+
: "every flow actor is defined",
|
|
242
|
+
},
|
|
243
|
+
{
|
|
244
|
+
// The audit inventories each screen's controls and runs the component
|
|
245
|
+
// scan that populates the library, so it comes before test cases:
|
|
246
|
+
// knowing what a screen is built from is what lets a case be written
|
|
247
|
+
// against something real.
|
|
248
|
+
stage: "audit",
|
|
249
|
+
satisfied: canonical.length > 0 && needsAudit.length === 0,
|
|
250
|
+
detail: `${audited.size}/${canonical.length} flows audited`,
|
|
251
|
+
},
|
|
252
|
+
{
|
|
253
|
+
stage: "testcases",
|
|
254
|
+
satisfied: canonical.length > 0 && needsCases.length === 0,
|
|
255
|
+
detail: `${caseSets.size}/${canonical.length} flows have test case assumptions`,
|
|
256
|
+
},
|
|
257
|
+
{
|
|
258
|
+
stage: "plan",
|
|
259
|
+
satisfied: canonical.length > 0 && planned.size === canonical.length,
|
|
260
|
+
detail: `${planned.size}/${canonical.length} flows planned`,
|
|
261
|
+
},
|
|
262
|
+
{
|
|
263
|
+
// Until a plan actually runs, every test case assumption is still
|
|
264
|
+
// just an assumption — this is the stage that produces evidence.
|
|
265
|
+
//
|
|
266
|
+
// "Executed" is not the bar. A run artifact only proves a run happened;
|
|
267
|
+
// this stage is satisfied when the run recorded a verdict for every
|
|
268
|
+
// case it was meant to judge. Counting runs alone reported 20/20 here
|
|
269
|
+
// while all 118 cases still had no status, which is exactly the kind of
|
|
270
|
+
// green that hides an untested app.
|
|
271
|
+
stage: "run",
|
|
272
|
+
satisfied: canonical.length > 0 && planned.size > 0 && judgedAndPlanned.length === planned.size,
|
|
273
|
+
// Counted as a set difference, not ran minus judged: a flow can be
|
|
274
|
+
// judged without a registry run record, which made that subtraction
|
|
275
|
+
// report a negative number of flows.
|
|
276
|
+
detail: ranButUnjudged.length === 0
|
|
277
|
+
? `${judgedAndPlanned.length}/${planned.size} planned flows executed and judged`
|
|
278
|
+
: `${judgedAndPlanned.length}/${planned.size} planned flows judged `
|
|
279
|
+
+ `(${ranButUnjudged.length} ran but recorded no verdicts)`,
|
|
280
|
+
},
|
|
281
|
+
];
|
|
282
|
+
|
|
283
|
+
const stopped = (reason: string): PipelineState => ({ status: "stopped", reason, stages, orphanedArtifacts });
|
|
284
|
+
|
|
285
|
+
// Cancelling is the user's kill switch, so it must halt the whole
|
|
286
|
+
// pipeline, not just the one task — otherwise the reconciler would
|
|
287
|
+
// cheerfully queue the next stage a few seconds later. Re-triggering
|
|
288
|
+
// recycles that task back to queued, which clears the stop.
|
|
289
|
+
const cancelled = tasks.find((task) =>
|
|
290
|
+
isPipelineTask(task) && task.status === "error" && task.completionSummary === "Cancelled");
|
|
291
|
+
if (cancelled) {
|
|
292
|
+
return stopped("A pipeline task was cancelled. Re-run the stage to resume.");
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
const hitHardStop = recordedBlockerCategories.find((category) => hardStopCategories.includes(category));
|
|
296
|
+
if (hitHardStop) {
|
|
297
|
+
return stopped(`Hard stop: the "${hitHardStop}" blocker halts the run by policy.`);
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
if (inputs.pendingClarification) {
|
|
301
|
+
// Paused, not stopped: the work is fine, it just needs an answer that
|
|
302
|
+
// the agent must not invent.
|
|
303
|
+
return { status: "paused", reason: inputs.pendingClarification, stages, orphanedArtifacts };
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// A pipeline task that has burned its attempts is not going to succeed
|
|
307
|
+
// on its own. It is neither in flight nor terminal, so the reconciler
|
|
308
|
+
// would keep "enqueueing" it — getting the same blocked task back every
|
|
309
|
+
// few seconds, making no progress and saying nothing. Surface it as
|
|
310
|
+
// needing an answer instead, which is the one thing that can unblock it.
|
|
311
|
+
const exhausted = tasks.filter((task) =>
|
|
312
|
+
isPipelineTask(task)
|
|
313
|
+
&& task.status === "blocked"
|
|
314
|
+
&& (task.attempts ?? 0) >= MAX_TASK_ATTEMPTS);
|
|
315
|
+
if (exhausted.length) {
|
|
316
|
+
const first = exhausted[0];
|
|
317
|
+
const detail = first.errorMessage?.trim() || first.completionSummary?.trim();
|
|
318
|
+
return {
|
|
319
|
+
status: "paused",
|
|
320
|
+
reason: `${exhausted.length} stage${exhausted.length === 1 ? "" : "s"} need your input — `
|
|
321
|
+
+ `"${first.title ?? first.idempotencyKey}" stopped after ${MAX_TASK_ATTEMPTS} attempts`
|
|
322
|
+
+ (detail ? `: ${detail}` : "."),
|
|
323
|
+
stages,
|
|
324
|
+
orphanedArtifacts,
|
|
325
|
+
blockedTasks: exhausted.map((task) => ({
|
|
326
|
+
title: task.title ?? task.idempotencyKey ?? "pipeline task",
|
|
327
|
+
idempotencyKey: task.idempotencyKey,
|
|
328
|
+
attempts: task.attempts ?? 0,
|
|
329
|
+
reason: task.errorMessage?.trim() || task.completionSummary?.trim(),
|
|
330
|
+
})),
|
|
331
|
+
};
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
// Only one stage runs at a time; the worker already serialises task
|
|
335
|
+
// execution, and queueing ahead would just create work that races.
|
|
336
|
+
const inFlight = activeTask(tasks);
|
|
337
|
+
if (inFlight) {
|
|
338
|
+
return { status: "running", reason: "A task is already in flight.", stages, orphanedArtifacts };
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
const nextStage = stages.find((stage) => !stage.satisfied);
|
|
342
|
+
if (!nextStage) {
|
|
343
|
+
return { status: "complete", reason: "Every stage is satisfied.", stages, orphanedArtifacts };
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
if (nextStage.stage === "run") {
|
|
347
|
+
const journey = needsRun[0];
|
|
348
|
+
if (journey && runsInFlight.has(journey.id)) {
|
|
349
|
+
// Its run is already executing. Starting another would either duplicate
|
|
350
|
+
// the work or, under a stable idempotency key, hand back the same run
|
|
351
|
+
// and look like progress that is not happening.
|
|
352
|
+
return {
|
|
353
|
+
status: "running",
|
|
354
|
+
reason: `A run of "${journey.id}" is already in flight.`,
|
|
355
|
+
stages,
|
|
356
|
+
orphanedArtifacts,
|
|
357
|
+
};
|
|
358
|
+
}
|
|
359
|
+
// A flow that has burned this many finished runs and still has no
|
|
360
|
+
// verdicts is not going to get them by being run again. Stop and say so,
|
|
361
|
+
// rather than starting run number four and calling it progress.
|
|
362
|
+
const attempts = journey ? finishedRunCounts.get(journey.id) ?? 0 : 0;
|
|
363
|
+
if (journey && attempts >= MAX_TASK_ATTEMPTS) {
|
|
364
|
+
return {
|
|
365
|
+
status: "paused",
|
|
366
|
+
reason: `Flow "${journey.id}" ran ${attempts} times without recording any verdict — it needs your input.`,
|
|
367
|
+
stages,
|
|
368
|
+
orphanedArtifacts,
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
const plan = (inputs.publishedPlans ?? []).find((candidate) => candidate.journeyId === journey?.id);
|
|
372
|
+
if (!journey || !plan) {
|
|
373
|
+
// A plan exists on disk but was never published to the registry, so
|
|
374
|
+
// there is nothing runnable yet. Say so instead of silently idling.
|
|
375
|
+
return {
|
|
376
|
+
status: "paused",
|
|
377
|
+
reason: `Flow "${journey?.id ?? "unknown"}" is planned but has no published plan to run.`,
|
|
378
|
+
stages,
|
|
379
|
+
orphanedArtifacts,
|
|
380
|
+
};
|
|
381
|
+
}
|
|
382
|
+
return {
|
|
383
|
+
status: "running",
|
|
384
|
+
reason: "Next stage: run",
|
|
385
|
+
nextRun: {
|
|
386
|
+
stage: "run",
|
|
387
|
+
journeyId: journey.id,
|
|
388
|
+
planId: plan.planId,
|
|
389
|
+
// Discriminated by how many runs of this flow have already finished.
|
|
390
|
+
// The key has to stay stable while a run is in flight, so repeated
|
|
391
|
+
// reconciles do not duplicate it — but it must change once that run
|
|
392
|
+
// finishes, or the registry keeps handing back the completed run and
|
|
393
|
+
// the flow can never be run again. That is exactly what happened: a
|
|
394
|
+
// finished run from hours earlier was returned on every poll, so the
|
|
395
|
+
// pipeline announced "started run" forever and nothing executed.
|
|
396
|
+
idempotencyKey: `${pipelineIdempotencyKey(projectId, "run", journey.id)}:${attempts}`,
|
|
397
|
+
environmentId: inputs.environmentId,
|
|
398
|
+
},
|
|
399
|
+
stages,
|
|
400
|
+
orphanedArtifacts,
|
|
401
|
+
};
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
const next = buildNextTask(projectId, nextStage.stage, {
|
|
405
|
+
missingActors,
|
|
406
|
+
needsAudit,
|
|
407
|
+
needsCases,
|
|
408
|
+
needsPlan,
|
|
409
|
+
});
|
|
410
|
+
return { status: "running", reason: `Next stage: ${nextStage.stage}`, next, stages, orphanedArtifacts };
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
function buildNextTask(
|
|
414
|
+
projectId: string,
|
|
415
|
+
stage: PipelineStage,
|
|
416
|
+
context: {
|
|
417
|
+
missingActors: string[];
|
|
418
|
+
needsAudit: PipelineJourney[];
|
|
419
|
+
needsCases: PipelineJourney[];
|
|
420
|
+
needsPlan: PipelineJourney[];
|
|
421
|
+
},
|
|
422
|
+
): PipelineNextTask {
|
|
423
|
+
if (stage === "explore") {
|
|
424
|
+
return {
|
|
425
|
+
stage,
|
|
426
|
+
taskType: "investigation",
|
|
427
|
+
title: "Explore the app and build the journey graph",
|
|
428
|
+
description: [
|
|
429
|
+
"Read bugmole://guidance/agent-testing and follow it end to end.",
|
|
430
|
+
"Run bugmole_explore against the configured environment, then record what you actually",
|
|
431
|
+
"observe with bugmole_record_discovery. After every exploration call bugmole_journey_next",
|
|
432
|
+
"and perform the returned task until the flow is exhausted or explicitly blocked.",
|
|
433
|
+
].join(" "),
|
|
434
|
+
idempotencyKey: pipelineIdempotencyKey(projectId, stage),
|
|
435
|
+
};
|
|
436
|
+
}
|
|
437
|
+
if (stage === "roles") {
|
|
438
|
+
return {
|
|
439
|
+
stage,
|
|
440
|
+
taskType: "code",
|
|
441
|
+
title: "Define the actors the discovered flows use",
|
|
442
|
+
description: [
|
|
443
|
+
`The discovered flows use actors that roles.yaml does not define: ${context.missingActors.join(", ")}.`,
|
|
444
|
+
"Planning refuses an undefined actor, so the pipeline cannot proceed until this is reconciled.",
|
|
445
|
+
"Inspect the application's real roles and either add the missing actor to spec/roles.yaml with",
|
|
446
|
+
"accurate permissions, or update the flow to use an existing role that genuinely matches.",
|
|
447
|
+
"Do not invent permissions to make the check pass.",
|
|
448
|
+
].join(" "),
|
|
449
|
+
idempotencyKey: pipelineIdempotencyKey(projectId, stage),
|
|
450
|
+
};
|
|
451
|
+
}
|
|
452
|
+
if (stage === "audit") {
|
|
453
|
+
const journey = context.needsAudit[0];
|
|
454
|
+
return {
|
|
455
|
+
stage,
|
|
456
|
+
journeyId: journey?.id,
|
|
457
|
+
taskType: "investigation",
|
|
458
|
+
title: `Audit the controls and components of ${journey?.id ?? "the discovered flow"}`,
|
|
459
|
+
description: [
|
|
460
|
+
"Read bugmole://guidance/app-ui-audit and follow it for this flow.",
|
|
461
|
+
"",
|
|
462
|
+
"First populate the component library, because nothing else does:",
|
|
463
|
+
"call bugmole_ui_re_start with kind \"existing_app\" and the app's base URL, then",
|
|
464
|
+
"bugmole_ui_re_stage with stage \"atoms\". That runs the Storybook/CSF source scan and",
|
|
465
|
+
"records the catalogue the dashboard's component library reads. If the project has no",
|
|
466
|
+
"Storybook, the scan reports that as a blocked reason — report it as-is rather than",
|
|
467
|
+
"inventing components.",
|
|
468
|
+
"",
|
|
469
|
+
`Then audit "${journey?.id}": inventory every interactive control on each screen with its`,
|
|
470
|
+
"handler and evidence, and call bugmole_write_ui_audit with the completed object.",
|
|
471
|
+
].join("\n"),
|
|
472
|
+
idempotencyKey: pipelineIdempotencyKey(projectId, stage, journey?.id),
|
|
473
|
+
};
|
|
474
|
+
}
|
|
475
|
+
if (stage === "testcases") {
|
|
476
|
+
const journey = context.needsCases[0];
|
|
477
|
+
return {
|
|
478
|
+
stage,
|
|
479
|
+
journeyId: journey?.id,
|
|
480
|
+
taskType: "investigation",
|
|
481
|
+
title: `Write test case assumptions for ${journey?.id ?? "the discovered flow"}`,
|
|
482
|
+
description: [
|
|
483
|
+
`Read bugmole://spec/test-cases.schema.json and write the test case assumptions for "${journey?.id}".`,
|
|
484
|
+
"Every case must name a real nodeId from that flow, say what it is trying to prove, and list",
|
|
485
|
+
"observable expectations. Cite what each assumption rests on in `basis`; a case you cannot ground",
|
|
486
|
+
"must be confidence \"low\" rather than a confident guess. Then call bugmole_write_test_cases.",
|
|
487
|
+
].join(" "),
|
|
488
|
+
idempotencyKey: pipelineIdempotencyKey(projectId, stage, journey?.id),
|
|
489
|
+
};
|
|
490
|
+
}
|
|
491
|
+
const journey = context.needsPlan[0];
|
|
492
|
+
return {
|
|
493
|
+
stage: "plan",
|
|
494
|
+
journeyId: journey?.id,
|
|
495
|
+
taskType: "investigation",
|
|
496
|
+
title: `Plan ${journey?.id ?? "the discovered flow"}`,
|
|
497
|
+
description: [
|
|
498
|
+
`Call bugmole_plan for "${journey?.id}" using the actor defined for that flow.`,
|
|
499
|
+
"Planning is blocked while the flow still has an actionable exploration task, so check",
|
|
500
|
+
"bugmole_journey_next first and finish any remaining verification before planning.",
|
|
501
|
+
].join(" "),
|
|
502
|
+
idempotencyKey: pipelineIdempotencyKey(projectId, "plan", journey?.id),
|
|
503
|
+
};
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
// --- Loading real state -----------------------------------------------------
|
|
507
|
+
//
|
|
508
|
+
// Everything below reads the same artifacts the dashboard reads, so the
|
|
509
|
+
// pipeline's view of "what is done" is the same view a human gets rather
|
|
510
|
+
// than a private ledger that can disagree with reality.
|
|
511
|
+
|
|
512
|
+
function store(cfg: any) {
|
|
513
|
+
return createProjectObjectStore(cfg);
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
function readYaml(filePath: string): any {
|
|
517
|
+
try {
|
|
518
|
+
return YAML.parse(fs.readFileSync(filePath, "utf8"));
|
|
519
|
+
} catch {
|
|
520
|
+
return null;
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
/** Actor names defined in roles.yaml, which the planner validates against. */
|
|
525
|
+
export function knownActors(specDir: string): string[] {
|
|
526
|
+
const roles = readYaml(path.join(specDir, "roles.yaml"));
|
|
527
|
+
return Array.isArray(roles?.roles)
|
|
528
|
+
? roles.roles.map((role: any) => role?.name).filter((name: unknown): name is string => typeof name === "string")
|
|
529
|
+
: [];
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
/**
|
|
533
|
+
* Blocker categories that halt the run by policy.
|
|
534
|
+
*
|
|
535
|
+
* spec/blockers.yaml has declared these since it was written, but nothing
|
|
536
|
+
* in the runtime ever read them — they were documentation the agent might
|
|
537
|
+
* choose to honour. The pipeline enforces them.
|
|
538
|
+
*/
|
|
539
|
+
export function hardStopCategories(specDir: string): string[] {
|
|
540
|
+
const doc = readYaml(path.join(specDir, "blockers.yaml"));
|
|
541
|
+
const fromBlockers = Array.isArray(doc?.blockers)
|
|
542
|
+
? doc.blockers.filter((entry: any) => entry?.hard_stop === true).map((entry: any) => entry?.name)
|
|
543
|
+
: [];
|
|
544
|
+
const fromHardStops = Array.isArray(doc?.hard_stops)
|
|
545
|
+
? doc.hard_stops.map((entry: any) => entry?.name)
|
|
546
|
+
: [];
|
|
547
|
+
return [...new Set([...fromBlockers, ...fromHardStops])]
|
|
548
|
+
.filter((name: unknown): name is string => typeof name === "string");
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
/**
|
|
552
|
+
* Journeys whose every written case carries a recorded verdict.
|
|
553
|
+
*
|
|
554
|
+
* Partial credit would defeat the purpose: a flow with one judged case and
|
|
555
|
+
* six unjudged ones has not been covered, and counting it would put the
|
|
556
|
+
* pipeline back to reporting green over an untested app. A "blocked" verdict
|
|
557
|
+
* still counts — the run looked and said why it could not check.
|
|
558
|
+
*/
|
|
559
|
+
async function listJourneyIdsWithVerdicts(cfg: any, projectId: string): Promise<string[]> {
|
|
560
|
+
try {
|
|
561
|
+
const objects = await store(cfg).list(`${projectPrefix(projectId)}/test-cases/`);
|
|
562
|
+
const ids = await Promise.all(objects
|
|
563
|
+
.filter((object) => object.key.endsWith(".json"))
|
|
564
|
+
.map(async (object) => {
|
|
565
|
+
try {
|
|
566
|
+
const caseSet = JSON.parse(await store(cfg).getText(object.key));
|
|
567
|
+
const journeyId = typeof caseSet?.journeyId === "string" ? caseSet.journeyId : null;
|
|
568
|
+
const caseIds = Array.isArray(caseSet?.cases)
|
|
569
|
+
? caseSet.cases.map((item: any) => item?.caseId).filter((id: unknown) => typeof id === "string")
|
|
570
|
+
: [];
|
|
571
|
+
if (!journeyId || caseIds.length === 0) return null;
|
|
572
|
+
const resultKey = `${projectPrefix(projectId)}/test-case-results/${caseSet.caseSetId}.json`;
|
|
573
|
+
const results = JSON.parse(await store(cfg).getText(resultKey));
|
|
574
|
+
// A verdict recorded against a superseded revision is not evidence
|
|
575
|
+
// about the cases as they stand now, matching how the dashboard
|
|
576
|
+
// discards it rather than showing it as proof.
|
|
577
|
+
if (Number.isInteger(results?.journeyRevision)
|
|
578
|
+
&& Number.isInteger(caseSet?.journeyRevision)
|
|
579
|
+
&& results.journeyRevision !== caseSet.journeyRevision) {
|
|
580
|
+
return null;
|
|
581
|
+
}
|
|
582
|
+
const judged = new Set(
|
|
583
|
+
(Array.isArray(results?.results) ? results.results : [])
|
|
584
|
+
.map((item: any) => item?.caseId)
|
|
585
|
+
.filter((id: unknown) => typeof id === "string"),
|
|
586
|
+
);
|
|
587
|
+
return caseIds.every((id: string) => judged.has(id)) ? journeyId : null;
|
|
588
|
+
} catch {
|
|
589
|
+
// No results file for this case set, or it is unreadable: not judged.
|
|
590
|
+
return null;
|
|
591
|
+
}
|
|
592
|
+
}));
|
|
593
|
+
return ids.filter((id): id is string => Boolean(id));
|
|
594
|
+
} catch {
|
|
595
|
+
return [];
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
async function listJourneyIdsWithCaseSets(cfg: any, projectId: string): Promise<string[]> {
|
|
600
|
+
try {
|
|
601
|
+
const objects = await store(cfg).list(`${projectPrefix(projectId)}/test-cases/`);
|
|
602
|
+
const ids = await Promise.all(objects
|
|
603
|
+
.filter((object) => object.key.endsWith(".json"))
|
|
604
|
+
.map(async (object) => {
|
|
605
|
+
try {
|
|
606
|
+
const parsed = JSON.parse(await store(cfg).getText(object.key));
|
|
607
|
+
return typeof parsed?.journeyId === "string" ? parsed.journeyId : null;
|
|
608
|
+
} catch {
|
|
609
|
+
return null;
|
|
610
|
+
}
|
|
611
|
+
}));
|
|
612
|
+
return ids.filter((id): id is string => Boolean(id));
|
|
613
|
+
} catch {
|
|
614
|
+
return [];
|
|
615
|
+
}
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
async function listAuditedJourneyIds(cfg: any, projectId: string): Promise<string[]> {
|
|
619
|
+
try {
|
|
620
|
+
const objects = await store(cfg).list(`${projectPrefix(projectId)}/ui-audits/`);
|
|
621
|
+
const ids = await Promise.all(objects
|
|
622
|
+
.filter((object) => object.key.endsWith(".json"))
|
|
623
|
+
.map(async (object) => {
|
|
624
|
+
try {
|
|
625
|
+
const parsed = JSON.parse(await store(cfg).getText(object.key));
|
|
626
|
+
return typeof parsed?.journeyId === "string" ? parsed.journeyId : null;
|
|
627
|
+
} catch {
|
|
628
|
+
return null;
|
|
629
|
+
}
|
|
630
|
+
}));
|
|
631
|
+
return [...new Set(ids.filter((id): id is string => Boolean(id)))];
|
|
632
|
+
} catch {
|
|
633
|
+
return [];
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
async function listRecordedBlockerCategories(cfg: any, projectId: string): Promise<string[]> {
|
|
638
|
+
try {
|
|
639
|
+
const objects = await store(cfg).list(`${projectPrefix(projectId)}/blocker-resolutions/`);
|
|
640
|
+
const categories = await Promise.all(objects
|
|
641
|
+
.filter((object) => object.key.endsWith(".json"))
|
|
642
|
+
.map(async (object) => {
|
|
643
|
+
try {
|
|
644
|
+
const parsed = JSON.parse(await store(cfg).getText(object.key));
|
|
645
|
+
return typeof parsed?.category === "string" ? parsed.category : null;
|
|
646
|
+
} catch {
|
|
647
|
+
return null;
|
|
648
|
+
}
|
|
649
|
+
}));
|
|
650
|
+
return [...new Set(categories.filter((category): category is string => Boolean(category)))];
|
|
651
|
+
} catch {
|
|
652
|
+
return [];
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
/** Journey ids that already have a published plan on disk. */
|
|
657
|
+
/**
|
|
658
|
+
* Whether a written plan could actually execute.
|
|
659
|
+
*
|
|
660
|
+
* Only the driver/platform pairing is checked, because that is the one that
|
|
661
|
+
* fails silently: the run starts, the driver rejects the plan outright, and
|
|
662
|
+
* every case it was meant to prove is recorded blocked without anything
|
|
663
|
+
* being tested. A plan unreadable for any other reason is left alone rather
|
|
664
|
+
* than quietly discarded.
|
|
665
|
+
*/
|
|
666
|
+
function isRunnablePlan(planPath: string): boolean {
|
|
667
|
+
let plan: any;
|
|
668
|
+
try {
|
|
669
|
+
plan = YAML.parse(fs.readFileSync(planPath, "utf8"));
|
|
670
|
+
} catch {
|
|
671
|
+
return true;
|
|
672
|
+
}
|
|
673
|
+
const driver = typeof plan?.driver === "string" ? plan.driver : undefined;
|
|
674
|
+
const platform = typeof plan?.platform === "string" ? plan.platform : "web";
|
|
675
|
+
if (!driver) return true;
|
|
676
|
+
return driverForPlatform(driver, platform) === driver;
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
export function plannedJourneyIds(specDir: string, journeyIds: string[]): string[] {
|
|
680
|
+
const plansDir = path.join(specDir, "plans");
|
|
681
|
+
let files: string[];
|
|
682
|
+
try {
|
|
683
|
+
files = fs.readdirSync(plansDir);
|
|
684
|
+
} catch {
|
|
685
|
+
return [];
|
|
686
|
+
}
|
|
687
|
+
// The planner writes <journeyId sanitised>.flow.yaml, so match by that
|
|
688
|
+
// same transform rather than guessing from the file's contents.
|
|
689
|
+
//
|
|
690
|
+
// A plan whose driver cannot drive its own platform does not count as
|
|
691
|
+
// planned. Eighteen web plans were written pinned to maestro, a mobile
|
|
692
|
+
// runner, and every run of them failed identically without checking
|
|
693
|
+
// anything. Treating those as planned left the pipeline satisfied with
|
|
694
|
+
// work that could never succeed; treating them as missing makes it
|
|
695
|
+
// re-plan them with a driver that can actually run.
|
|
696
|
+
const planIds = new Set(files
|
|
697
|
+
.filter((file) => file.endsWith(".flow.yaml"))
|
|
698
|
+
.filter((file) => isRunnablePlan(path.join(plansDir, file)))
|
|
699
|
+
.map((file) => file.replace(/\.flow\.yaml$/, "")));
|
|
700
|
+
return journeyIds.filter((id) => planIds.has(id.replace(/[^a-z0-9._-]/gi, "_").toLowerCase()));
|
|
701
|
+
}
|
|
702
|
+
|
|
703
|
+
export async function loadPipelineInputs(
|
|
704
|
+
cfg: any,
|
|
705
|
+
tasks: PipelineTask[],
|
|
706
|
+
registry?: {
|
|
707
|
+
runs?: Array<{ journeyId?: string; status?: string }>;
|
|
708
|
+
plans?: Array<{ planId: string; journeyId?: string }>;
|
|
709
|
+
},
|
|
710
|
+
): Promise<PipelineInputs> {
|
|
711
|
+
const projectId = cfg.project?.id ?? "default";
|
|
712
|
+
const specDir = cfg.runtime?.spec_dir || "./spec";
|
|
713
|
+
const graphPath = path.join(specDir, "journeys.graph.json");
|
|
714
|
+
let journeys: PipelineJourney[] = [];
|
|
715
|
+
try {
|
|
716
|
+
const graph = JSON.parse(fs.readFileSync(graphPath, "utf8"));
|
|
717
|
+
journeys = Array.isArray(graph?.journeys) ? graph.journeys : [];
|
|
718
|
+
} catch {
|
|
719
|
+
journeys = [];
|
|
720
|
+
}
|
|
721
|
+
const [caseSetJourneyIds, recordedBlockerCategories, auditedJourneyIds, verdictJourneyIds] = await Promise.all([
|
|
722
|
+
listJourneyIdsWithCaseSets(cfg, projectId),
|
|
723
|
+
listRecordedBlockerCategories(cfg, projectId),
|
|
724
|
+
listAuditedJourneyIds(cfg, projectId),
|
|
725
|
+
listJourneyIdsWithVerdicts(cfg, projectId),
|
|
726
|
+
]);
|
|
727
|
+
return {
|
|
728
|
+
projectId,
|
|
729
|
+
journeys,
|
|
730
|
+
knownActors: knownActors(specDir),
|
|
731
|
+
caseSetJourneyIds,
|
|
732
|
+
auditedJourneyIds,
|
|
733
|
+
plannedJourneyIds: plannedJourneyIds(specDir, journeys.map((journey) => journey.id)),
|
|
734
|
+
tasks,
|
|
735
|
+
hardStopCategories: hardStopCategories(specDir),
|
|
736
|
+
recordedBlockerCategories,
|
|
737
|
+
ranJourneyIds: [...new Set(
|
|
738
|
+
(registry?.runs ?? [])
|
|
739
|
+
.map((run) => run.journeyId)
|
|
740
|
+
.filter((id): id is string => Boolean(id)),
|
|
741
|
+
)],
|
|
742
|
+
verdictJourneyIds,
|
|
743
|
+
runs: (registry?.runs ?? []).map((run: any) => ({ journeyId: run.journeyId, status: run.status })),
|
|
744
|
+
publishedPlans: registry?.plans ?? [],
|
|
745
|
+
environmentId: cfg.project?.environment_id ?? cfg.project?.environmentId ?? cfg.runtime?.environment,
|
|
746
|
+
};
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
type ReconcileClient = {
|
|
750
|
+
listTasks(): Promise<{ tasks: any[] }>;
|
|
751
|
+
createTask(input: {
|
|
752
|
+
title: string;
|
|
753
|
+
description: string;
|
|
754
|
+
taskType?: "code" | "ui" | "investigation";
|
|
755
|
+
priority?: number;
|
|
756
|
+
idempotencyKey: string;
|
|
757
|
+
journeyId?: string;
|
|
758
|
+
}): Promise<unknown>;
|
|
759
|
+
listRuns?(): Promise<{ runs: any[] }>;
|
|
760
|
+
listPlans?(): Promise<{ plans: any[] }>;
|
|
761
|
+
createRun?(input: {
|
|
762
|
+
planId: string;
|
|
763
|
+
idempotencyKey: string;
|
|
764
|
+
environmentId?: string;
|
|
765
|
+
}): Promise<unknown>;
|
|
766
|
+
};
|
|
767
|
+
|
|
768
|
+
export type ReconcileResult = {
|
|
769
|
+
state: PipelineState;
|
|
770
|
+
enqueued?: PipelineNextTask;
|
|
771
|
+
startedRun?: PipelineNextRun;
|
|
772
|
+
};
|
|
773
|
+
|
|
774
|
+
/**
|
|
775
|
+
* Advance the pipeline by at most one stage.
|
|
776
|
+
*
|
|
777
|
+
* Safe to call on every worker poll: the stage is derived from artifacts
|
|
778
|
+
* and the task carries a deterministic idempotency key, so repeating this
|
|
779
|
+
* on an unchanged project is a no-op rather than a queue of duplicates.
|
|
780
|
+
*/
|
|
781
|
+
export function pipelineStateKey(projectId: string): string {
|
|
782
|
+
return `${projectPrefix(projectId)}/pipeline/state.json`;
|
|
783
|
+
}
|
|
784
|
+
|
|
785
|
+
/**
|
|
786
|
+
* Publish what the pipeline is doing, so it is visible in the dashboard.
|
|
787
|
+
*
|
|
788
|
+
* The pipeline was already advancing on its own, but the only way to see
|
|
789
|
+
* where a project had got to was to run a script against the artifact
|
|
790
|
+
* store — which is not a usable answer to "where do we go from here".
|
|
791
|
+
*/
|
|
792
|
+
export async function publishPipelineState(cfg: any, state: PipelineState): Promise<void> {
|
|
793
|
+
const projectId = cfg.project?.id ?? "default";
|
|
794
|
+
await store(cfg).put(
|
|
795
|
+
pipelineStateKey(projectId),
|
|
796
|
+
JSON.stringify({ ...state, projectId, updatedAt: new Date().toISOString() }, null, 2) + "\n",
|
|
797
|
+
"application/json",
|
|
798
|
+
);
|
|
799
|
+
}
|
|
800
|
+
|
|
801
|
+
export async function reconcilePipeline(cfg: any, client: ReconcileClient): Promise<ReconcileResult> {
|
|
802
|
+
const [listed, runs, plans] = await Promise.all([
|
|
803
|
+
client.listTasks(),
|
|
804
|
+
client.listRuns?.().catch(() => ({ runs: [] })) ?? Promise.resolve({ runs: [] }),
|
|
805
|
+
client.listPlans?.().catch(() => ({ plans: [] })) ?? Promise.resolve({ plans: [] }),
|
|
806
|
+
]);
|
|
807
|
+
const inputs = await loadPipelineInputs(cfg, listed.tasks ?? [], {
|
|
808
|
+
runs: runs.runs ?? [],
|
|
809
|
+
plans: plans.plans ?? [],
|
|
810
|
+
});
|
|
811
|
+
const state = derivePipelineState(inputs);
|
|
812
|
+
// Published on every pass so the dashboard reflects the live position,
|
|
813
|
+
// including when nothing is queued.
|
|
814
|
+
await publishPipelineState(cfg, state).catch(() => undefined);
|
|
815
|
+
|
|
816
|
+
if (state.nextRun && client.createRun) {
|
|
817
|
+
await client.createRun({
|
|
818
|
+
planId: state.nextRun.planId,
|
|
819
|
+
idempotencyKey: state.nextRun.idempotencyKey,
|
|
820
|
+
environmentId: state.nextRun.environmentId,
|
|
821
|
+
});
|
|
822
|
+
return { state, startedRun: state.nextRun };
|
|
823
|
+
}
|
|
824
|
+
if (!state.next) return { state };
|
|
825
|
+
await client.createTask({
|
|
826
|
+
title: state.next.title,
|
|
827
|
+
description: state.next.description,
|
|
828
|
+
taskType: state.next.taskType,
|
|
829
|
+
idempotencyKey: state.next.idempotencyKey,
|
|
830
|
+
journeyId: state.next.journeyId,
|
|
831
|
+
});
|
|
832
|
+
return { state, enqueued: state.next };
|
|
833
|
+
}
|