@muggleai/works 4.12.2 → 4.12.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-I4VLYJ7M.js → chunk-CPF6AR2I.js} +499 -147
- package/dist/{chunk-2DVZ2LYO.js → chunk-JNI7INIO.js} +2 -2
- package/dist/cli.js +2 -2
- package/dist/index.js +2 -2
- package/dist/plugin/.claude-plugin/plugin.json +1 -1
- package/dist/plugin/.cursor-plugin/plugin.json +1 -1
- package/dist/plugin/skills/_shared/dev-server-readiness.md +39 -0
- package/dist/plugin/skills/_shared/failure-mode-handling.md +19 -4
- package/dist/plugin/skills/_shared/github-cli-recipes/submitted-reviews.md +1 -0
- package/dist/plugin/skills/_shared/pr-branch-worktree.md +31 -0
- package/dist/plugin/skills/_shared/pr-followup-helpers/allow-list.md +1 -1
- package/dist/plugin/skills/muggle-feedback/ops/submit.md +1 -1
- package/dist/plugin/skills/muggle-pr-followup/contract.md +3 -2
- package/dist/plugin/skills/muggle-test/SKILL.md +7 -3
- package/dist/plugin/skills/muggle-test-feature-local/SKILL.md +6 -1
- package/dist/plugin/skills/muggle-test-prepare/SKILL.md +40 -256
- package/dist/plugin/skills/muggle-test-prepare/steps/check-running.md +31 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/env-file.md +18 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/fresh-install.md +20 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/identify-services.md +28 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/readiness-report.md +26 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/rebase-check.md +3 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/scope.md +11 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/smoke-test.md +26 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/start-commands.md +29 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/start-services.md +21 -0
- package/dist/plugin/skills/muggle-test-prepare/steps/viability-check.md +21 -0
- package/dist/release-manifest.json +4 -4
- package/dist/{src-ARTTHWNP.js → src-YR5UKLPC.js} +1 -1
- package/package.json +6 -6
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.cursor-plugin/plugin.json +1 -1
- package/plugin/skills/_shared/dev-server-readiness.md +39 -0
- package/plugin/skills/_shared/failure-mode-handling.md +19 -4
- package/plugin/skills/_shared/github-cli-recipes/submitted-reviews.md +1 -0
- package/plugin/skills/_shared/pr-branch-worktree.md +31 -0
- package/plugin/skills/_shared/pr-followup-helpers/allow-list.md +1 -1
- package/plugin/skills/muggle-feedback/ops/submit.md +1 -1
- package/plugin/skills/muggle-pr-followup/contract.md +3 -2
- package/plugin/skills/muggle-test/SKILL.md +7 -3
- package/plugin/skills/muggle-test-feature-local/SKILL.md +6 -1
- package/plugin/skills/muggle-test-prepare/SKILL.md +40 -256
- package/plugin/skills/muggle-test-prepare/steps/check-running.md +31 -0
- package/plugin/skills/muggle-test-prepare/steps/env-file.md +18 -0
- package/plugin/skills/muggle-test-prepare/steps/fresh-install.md +20 -0
- package/plugin/skills/muggle-test-prepare/steps/identify-services.md +28 -0
- package/plugin/skills/muggle-test-prepare/steps/readiness-report.md +26 -0
- package/plugin/skills/muggle-test-prepare/steps/rebase-check.md +3 -0
- package/plugin/skills/muggle-test-prepare/steps/scope.md +11 -0
- package/plugin/skills/muggle-test-prepare/steps/smoke-test.md +26 -0
- package/plugin/skills/muggle-test-prepare/steps/start-commands.md +29 -0
- package/plugin/skills/muggle-test-prepare/steps/start-services.md +21 -0
- package/plugin/skills/muggle-test-prepare/steps/viability-check.md +21 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { __export, getLogger, getConfig, createChildLogger, buildElectronAppReleaseAssetUrl, getAuthService, hasApiKey, getElectronAppVersion, getElectronAppDir, getPlatformKey, isFirstRun, writePreferences, DEFAULT_PREFERENCES, getDataDir, PREFERENCES_FILE_NAME, isElectronAppInstalled, getElectronAppChecksums, getChecksumForPlatform, verifyFileChecksum, calculateFileChecksum, initTelemetry, Surface, ServiceName, track, EventName, getQaTools, getLocalQaTools, performLogout, performLogin, toolRequiresAuth, getCallerCredentials, hasShownDisclosure, getDisclosureCopy, markDisclosureShown, getBundledElectronAppVersion, getElectronAppVersionSource, getCredentialsFilePath, buildElectronAppChecksumsUrl, __require } from './chunk-
|
|
1
|
+
import { __export, getLogger, getConfig, createChildLogger, buildElectronAppReleaseAssetUrl, getAuthService, hasApiKey, getElectronAppVersion, getElectronAppDir, getPlatformKey, isFirstRun, writePreferences, DEFAULT_PREFERENCES, getDataDir, PREFERENCES_FILE_NAME, isElectronAppInstalled, getElectronAppChecksums, getChecksumForPlatform, verifyFileChecksum, calculateFileChecksum, initTelemetry, Surface, ServiceName, track, EventName, getQaTools, getLocalQaTools, performLogout, performLogin, toolRequiresAuth, getCallerCredentials, hasShownDisclosure, getDisclosureCopy, markDisclosureShown, getBundledElectronAppVersion, getElectronAppVersionSource, getCredentialsFilePath, buildElectronAppChecksumsUrl, __require } from './chunk-CPF6AR2I.js';
|
|
2
2
|
import { Server } from '@modelcontextprotocol/sdk/server/index.js';
|
|
3
3
|
import { ListToolsRequestSchema, CallToolRequestSchema, ListResourcesRequestSchema, ReadResourceRequestSchema } from '@modelcontextprotocol/sdk/types.js';
|
|
4
4
|
import { v4 } from 'uuid';
|
|
@@ -736,7 +736,7 @@ async function resolveGsScreenshotUrls(report, opts) {
|
|
|
736
736
|
if (gsUrls.length === 0) {
|
|
737
737
|
return report;
|
|
738
738
|
}
|
|
739
|
-
const mcps = await import('./src-
|
|
739
|
+
const mcps = await import('./src-YR5UKLPC.js');
|
|
740
740
|
const credentials = await mcps.getCallerCredentialsAsync();
|
|
741
741
|
if (!credentials.bearerToken && !credentials.apiKey) {
|
|
742
742
|
stderrWrite(
|
package/dist/cli.js
CHANGED
package/dist/index.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
export { src_exports as commands, createUnifiedMcpServer, server_exports as server } from './chunk-
|
|
2
|
-
export { createChildLogger, e2e_exports as e2e, getConfig, getLocalQaTools, getLogger, getQaTools, local_exports as localQa, mcp_exports as mcp, e2e_exports as qa, src_exports as shared } from './chunk-
|
|
1
|
+
export { src_exports as commands, createUnifiedMcpServer, server_exports as server } from './chunk-JNI7INIO.js';
|
|
2
|
+
export { createChildLogger, e2e_exports as e2e, getConfig, getLocalQaTools, getLogger, getQaTools, local_exports as localQa, mcp_exports as mcp, e2e_exports as qa, src_exports as shared } from './chunk-CPF6AR2I.js';
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "muggle",
|
|
3
3
|
"description": "Run real-browser end-to-end (E2E) acceptance tests on your web app from any AI coding agent. Generate test scripts from plain English, replay them on localhost, capture screenshots, and validate user flows like signup, checkout, and dashboards. Works across Claude Code, Cursor, Codex, and Windsurf.",
|
|
4
|
-
"version": "4.12.
|
|
4
|
+
"version": "4.12.4",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Muggle AI",
|
|
7
7
|
"email": "support@muggle-ai.com"
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "muggle",
|
|
3
3
|
"displayName": "Muggle AI",
|
|
4
4
|
"description": "Ship quality products with AI-powered end-to-end (E2E) acceptance testing that validates your web app like a real user — from Claude Code and Cursor to PR.",
|
|
5
|
-
"version": "4.12.
|
|
5
|
+
"version": "4.12.4",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Muggle AI",
|
|
8
8
|
"email": "support@muggle-ai.com"
|
|
@@ -49,6 +49,45 @@ netstat -ano | findstr /R /C:":3000 " /C:":3001 " /C:":4200 " /C:":5173 " /C:":8
|
|
|
49
49
|
|
|
50
50
|
If the app declares a backend URL in its env file, probe the backend's health endpoint before treating the dev server as usable. 5xx or unreachable → halt; the frontend may render but its data layer is dead, so any query against it is meaningless.
|
|
51
51
|
|
|
52
|
+
## Body sniff patterns
|
|
53
|
+
|
|
54
|
+
A `200 OK` can still be a build-error overlay or stack trace. Search the response body (case-insensitive) for broken-build markers — a match means unhealthy regardless of status.
|
|
55
|
+
|
|
56
|
+
| Stack | Pattern (regex) |
|
|
57
|
+
|:------|:----------------|
|
|
58
|
+
| Next.js | `__next_error__\|Failed to compile\|webpack-internal://` |
|
|
59
|
+
| Vite | `vite-error-overlay\|Internal server error\|\[plugin:` |
|
|
60
|
+
| Node / Express | `MODULE_NOT_FOUND\|Cannot find module\|npm ERR!\|Cannot GET /\|Cannot POST /\|Error: ENOENT\|EACCES\|EADDRINUSE` |
|
|
61
|
+
| Django | `TemplateSyntaxError\|ProgrammingError at /\|<h1>Server Error \(500\)</h1>` |
|
|
62
|
+
| Flask | `Werkzeug Debugger\|werkzeug-debug` |
|
|
63
|
+
| FastAPI / Python | `Traceback \(most recent call last\)\|ModuleNotFoundError\|ImportError` |
|
|
64
|
+
| Rails | `Better Errors\|ActionController::RoutingError\|<title>Action Controller:` |
|
|
65
|
+
| Spring Boot | `Whitelabel Error Page` |
|
|
66
|
+
| Tomcat | `HTTP Status 500.*Apache Tomcat` |
|
|
67
|
+
| Laravel / PHP | `Whoops\\\\|<b>Fatal error</b>\|Parse error:\|Stack trace:` |
|
|
68
|
+
| JS stack frame | `at .*\(.*\.[jt]sx?:\d+:\d+\)` |
|
|
69
|
+
| Java stack frame | `at \w+(\.\w+)+\(\w+\.java:\d+\)` |
|
|
70
|
+
| Python stack frame | `File ".*", line \d+, in ` |
|
|
71
|
+
| Ruby stack frame | `\.rb:\d+:in ` |
|
|
72
|
+
|
|
73
|
+
The bash/PowerShell snippets below use the union of all patterns above. Trim per-stack when you know the target.
|
|
74
|
+
|
|
75
|
+
#### bash/zsh
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
BODY=$(curl -sS -L --max-redirs 1 --max-time 3 "$URL")
|
|
79
|
+
PATTERN='__next_error__|Failed to compile|webpack-internal://|vite-error-overlay|Internal server error|MODULE_NOT_FOUND|Cannot find module|Cannot GET /|npm ERR!|Error: ENOENT|EACCES|EADDRINUSE|TemplateSyntaxError|Werkzeug Debugger|Traceback \(most recent call last\)|ModuleNotFoundError|ImportError|Better Errors|ActionController::RoutingError|Whitelabel Error Page|Apache Tomcat|Whoops|<b>Fatal error</b>|Parse error:|Stack trace:|at .*\(.*\.[jt]sx?:[0-9]+:[0-9]+\)|at \w+(\.\w+)+\(\w+\.java:[0-9]+\)|File ".*", line [0-9]+, in |\.rb:[0-9]+:in '
|
|
80
|
+
echo "$BODY" | grep -qiE "$PATTERN" && { echo "BODY-SNIFF FAIL"; echo "$BODY" | grep -iE "$PATTERN" | head -3; exit 1; }
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
#### PowerShell
|
|
84
|
+
|
|
85
|
+
```powershell
|
|
86
|
+
$body = (Invoke-WebRequest -Uri $url -TimeoutSec 3 -MaximumRedirection 1 -ErrorAction Stop).Content
|
|
87
|
+
$pattern = '__next_error__|Failed to compile|webpack-internal://|vite-error-overlay|Internal server error|MODULE_NOT_FOUND|Cannot find module|Cannot GET /|npm ERR!|Error: ENOENT|EACCES|EADDRINUSE|TemplateSyntaxError|Werkzeug Debugger|Traceback \(most recent call last\)|ModuleNotFoundError|ImportError|Better Errors|ActionController::RoutingError|Whitelabel Error Page|Apache Tomcat|Whoops|<b>Fatal error</b>|Parse error:|Stack trace:|at .*\(.*\.[jt]sx?:\d+:\d+\)|at \w+(\.\w+)+\(\w+\.java:\d+\)|File ".*", line \d+, in |\.rb:\d+:in '
|
|
88
|
+
if ($body -imatch $pattern) { Write-Host "BODY-SNIFF FAIL"; [regex]::Matches($body, $pattern, 'IgnoreCase') | Select-Object -First 3 | ForEach-Object { $_.Value }; exit 1 }
|
|
89
|
+
```
|
|
90
|
+
|
|
52
91
|
## Two-stage readiness — after starting a dev server
|
|
53
92
|
|
|
54
93
|
Network reachability is necessary but not sufficient. Many dev servers bind to a port before build/startup work is complete. Wait for **both** network readiness and application readiness before issuing requests.
|
|
@@ -109,9 +109,15 @@ Triggered when `muggle-local-execute-replay` returns `status: "failed"` (or non-
|
|
|
109
109
|
| **stale-script** | The test script no longer matches the live UI (selectors moved, label paths changed, page renamed). The product still works; the script is out of date. |
|
|
110
110
|
| **product-defect** | The script and infra are fine; the user's app actually misbehaved (assertion failure on previously-passing step, unexpected error, wrong page after action). This is the failure mode acceptance testing exists to catch. |
|
|
111
111
|
|
|
112
|
-
###
|
|
112
|
+
### Where to read signals
|
|
113
|
+
|
|
114
|
+
Call `muggle-local-run-result-get` (local) or the remote equivalent and read **structured fields**, not `execute`'s response stdout tail (it's a truncated display excerpt and routinely cuts off mid-sentence). Order:
|
|
113
115
|
|
|
114
|
-
|
|
116
|
+
1. `Status` + `Error` — the verdict and the one-line cause.
|
|
117
|
+
2. `Artifacts` section, when present — opens `artifactsDir`. Read `results.md` (step-by-step + screenshot links) for the per-step verdict, then `action-script.json` for what the agent attempted.
|
|
118
|
+
3. `stdout.log` / `stderr.log` only when the Artifacts section is absent or `results.md` doesn't exist (e.g. early Electron failure).
|
|
119
|
+
|
|
120
|
+
### Initial signal heuristics
|
|
115
121
|
|
|
116
122
|
- **infra** signals: `electron-crash`, `chromium-error`, `click-no-effect-on-clickable-element`, `timeout-on-trivial-wait`, `internal-error-in-mcp-output`.
|
|
117
123
|
- **stale-script** signals: `element-not-found`, `selector-timeout`, `label-path-mismatch`, `nav-target-404`, `aria-label-changed`.
|
|
@@ -189,9 +195,18 @@ Triggered when `muggle-local-execute-test-generation` (or the remote equivalent)
|
|
|
189
195
|
| **agent-course** | The generation agent went down a wrong path (chose the wrong button, misread the goal, looped on a blocking modal). The product is fine and the test case is fine — the agent's *course* needs steering. |
|
|
190
196
|
| **product-uxux** | The product itself blocks the test (broken page, missing element, server error). Agent can't proceed because the feature doesn't actually work. |
|
|
191
197
|
|
|
192
|
-
###
|
|
198
|
+
### Where to read signals
|
|
193
199
|
|
|
194
|
-
|
|
200
|
+
Same rule as section B: read **structured fields** from `muggle-local-run-result-get`, not `execute`'s response stdout tail. The `Artifacts` section is present on failed regen too — `action-script.json` is included when generation reached the step-emission stage (typical for `goal_not_achievable`: the agent's attempted steps + halt summary). `results.md` and per-step screenshots are absent on failure (electron-app emits those only on the successful completion path).
|
|
201
|
+
|
|
202
|
+
Order:
|
|
203
|
+
|
|
204
|
+
1. `Status` + `Error` — the verdict and one-line cause. `Error: Electron exited with code 26` typically means `goal_not_achievable`.
|
|
205
|
+
2. `action-script.json` in `artifactsDir` when present — read the steps the agent attempted and the `summaryStep` (halt reason, goal-not-achievable verdict).
|
|
206
|
+
3. `stdout.log` / `stderr.log` at `artifactsDir/` — last 100 lines is usually enough; look for the final structured summary the generation agent emitted (it appears near the end as a JSON-ish block, not in the truncated execute tail).
|
|
207
|
+
4. Remote regen — fetch the workflow run with `muggle-remote-wf-get-ts-gen-latest-run`; signals live in `summaryStep` and the per-step list there.
|
|
208
|
+
|
|
209
|
+
### Initial signal heuristics
|
|
195
210
|
|
|
196
211
|
- **transient**: `network-error`, `llm-rate-limit`, `single-tool-call-error`, run had partial progress then died.
|
|
197
212
|
- **infra**: `electron-mcp-handler-crash`, `internal-validation-error`, `pipeline-stuck`, identical failure repeated more than twice.
|
|
@@ -13,3 +13,4 @@ Filter client-side:
|
|
|
13
13
|
- `id` not in `last_seen.escalated_review_ids`
|
|
14
14
|
- `user.login` in the resolved allow-list
|
|
15
15
|
- `state` in `{CHANGES_REQUESTED, COMMENTED}`, OR `APPROVED` with a non-empty body or at least one line comment
|
|
16
|
+
- **Not a reply-wrapper.** `POST /pulls/<n>/comments/<id>/replies` creates an implicit review whose comments all have `in_reply_to_id` set. Fetch each candidate review's comments via `gh api repos/<owner>/<repo>/pulls/<n>/reviews/<id>/comments` and drop the review if every comment has a non-null `in_reply_to_id` (no new top-level critique). Without this clause, the loop's own threaded replies — submitted under the PR author's identity in single-account workflows — pass the allow-list and re-dispatch `/muggle-do` on a no-op cycle.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# PR-Branch Worktree — Shared Reference
|
|
2
|
+
|
|
3
|
+
> Source of truth for materializing a PR's branch in an isolated worktree so the user's main checkout is never disturbed. Used by `muggle-test` (and any future skill that takes a GitHub PR URL). Skills MUST link here rather than restate the steps.
|
|
4
|
+
|
|
5
|
+
## When this applies
|
|
6
|
+
|
|
7
|
+
A skill receives a GitHub PR URL of the form `github.com/<org>/<repo>/pull/<n>` and needs the PR's branch checked out to test against it locally.
|
|
8
|
+
|
|
9
|
+
## Steps
|
|
10
|
+
|
|
11
|
+
1. **Resolve the PR's head branch:**
|
|
12
|
+
`gh pr view <n> --repo <org>/<repo> --json headRefName -q .headRefName`
|
|
13
|
+
2. **Sanitize the branch name** for filesystem use — replace `/` and other path separators with `-`. Example: `claude/regen-test-replay-flow-ZSScQ` → `claude-regen-test-replay-flow-ZSScQ`.
|
|
14
|
+
3. **Build the target worktree path:** `<repo>/.claude/worktrees/<sanitized-branch>`.
|
|
15
|
+
4. **Materialize the worktree:**
|
|
16
|
+
- If the target path does NOT exist:
|
|
17
|
+
- `git -C <repo> fetch origin <branch>`
|
|
18
|
+
- `git -C <repo> worktree add <target-path> <branch>`
|
|
19
|
+
- If the target path EXISTS (reused from a prior run):
|
|
20
|
+
- `git -C <target-path> fetch`
|
|
21
|
+
- `git -C <target-path> reset --hard origin/<branch>` — picks up new pushes, drops any local cruft.
|
|
22
|
+
5. **Use the worktree path as the working directory** for the rest of the run, including:
|
|
23
|
+
- Passing it as the **`cwd` parameter** to `muggle-local-execute-test-generation` and `muggle-local-execute-replay`. This is required, not optional — see `_shared/failure-mode-handling.md` and the lock identity discussion in those tools' MCP source.
|
|
24
|
+
- Resolving any `npm install` / dev-server start commands inside the worktree (it has its own `node_modules/` and `.env*` files).
|
|
25
|
+
6. **Tell the user** where the worktree lives so they can clean it up later with `git -C <repo> worktree remove <target-path>`.
|
|
26
|
+
|
|
27
|
+
## Invariants
|
|
28
|
+
|
|
29
|
+
- **Never switch the user's main checkout.** The whole point of this flow is isolation; `git checkout <branch>` on the main checkout is forbidden.
|
|
30
|
+
- **Never share `node_modules/` via symlink** across worktrees. Each worktree runs its own `npm install` (or `pnpm install`) — webpack's `resolve.symlinks: true` rewrites paths and breaks asset-identity tracking.
|
|
31
|
+
- **`.env*` files do not propagate.** A freshly created worktree has no env files unless the repo commits them. If the parent skill's dev server fails to boot, check whether `.env.local` (or framework equivalent) needs to be copied from the main checkout before launching.
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
The address-reviews flow only acts on reviews submitted by users in the **allow-list** = (requested reviewers ∪ CODEOWNERS ∪ {PR author}) − bots. Re-resolve every invocation — never cache across cycles.
|
|
4
4
|
|
|
5
|
-
The PR author is implicitly a valid reviewer: in single-account workflows, the human running the agent and the PR's author are the same identity, and the agent must honor their reviews.
|
|
5
|
+
The PR author is implicitly a valid reviewer: in single-account workflows, the human running the agent and the PR's author are the same identity, and the agent must honor their reviews. Self-loop is prevented at the watcher's filter layer rather than here: `POST /pulls/<n>/comments/<id>/replies` does create an implicit review under the loop user's identity, and the watcher drops it via the reply-wrapper clause in [`../github-cli-recipes/submitted-reviews.md`](../github-cli-recipes/submitted-reviews.md). Including the PR author in this allow-list is therefore safe.
|
|
6
6
|
|
|
7
7
|
## Step 1: requested reviewers
|
|
8
8
|
|
|
@@ -8,7 +8,7 @@ Pick the first applicable path. Stop at the first that yields an `actionScriptId
|
|
|
8
8
|
|
|
9
9
|
### 1a. Dashboard URL in user's prompt
|
|
10
10
|
|
|
11
|
-
A Muggle dashboard URL looks like `https://
|
|
11
|
+
A Muggle dashboard URL looks like `https://www.muggle-ai.com/muggleTestV0/dashboard/projects/<projectId>/...`. Scan the user's recent message for any `https://www.muggle-ai.com/...` URL.
|
|
12
12
|
|
|
13
13
|
- Extract any UUID-shaped path segments. If `/projects/<uuid>` is present, capture as `projectId`. If `/test-scripts/<uuid>` is present, capture as `testScriptId`.
|
|
14
14
|
- If a `testScriptId` was captured: call `muggle-remote-test-script-get` to get the script and read `actionScriptId` off it. Done — proceed to step 2.
|
|
@@ -37,7 +37,8 @@ If `state` is `MERGED` or `CLOSED`:
|
|
|
37
37
|
2. Write `result.md` per [`state-schemas.md`](state-schemas.md#resultmd).
|
|
38
38
|
3. Append a terminal line to `followup.log` per [`output-templates/watcher-log.md`](output-templates/watcher-log.md).
|
|
39
39
|
4. Emit a `tick` event with `terminal: true` per [`../_shared/telemetry-events/pr-followup-tick.md`](../_shared/telemetry-events/pr-followup-tick.md).
|
|
40
|
-
5.
|
|
40
|
+
5. **Cancel the cron schedule that fires this watcher.** `/loop 1m ...` from bootstrap was registered via `CronCreate`; a fixed-interval cron keeps firing regardless of whether the skill re-dispatches. Call `CronList`, find any job whose command ends with `/muggle:muggle-pr-followup <slug> <pr-number>` (exact two-arg match), and `CronDelete` it. No-op if none matches — the tick may have been invoked manually rather than via `/loop`.
|
|
41
|
+
6. Exit. The watcher has now unscheduled itself; no future ticks will fire for this PR.
|
|
41
42
|
|
|
42
43
|
### Step 3 — Fetch new submitted reviews
|
|
43
44
|
|
|
@@ -66,7 +67,7 @@ The watcher does **not** classify. Classification, batching, replying, escalatio
|
|
|
66
67
|
```
|
|
67
68
|
3. Append a dispatching line to `followup.log` per [`output-templates/watcher-log.md`](output-templates/watcher-log.md).
|
|
68
69
|
4. Emit a `tick` event with `reviews_seen: <count>`, `dispatched_review_ids: [<id>, ...]`.
|
|
69
|
-
5. Exit.
|
|
70
|
+
5. Exit. The cron schedule from bootstrap keeps firing the watcher every minute, so the next tick still arrives even though this turn dispatched `/muggle-do`. The watcher only self-unschedules in Step 2 (terminal).
|
|
70
71
|
|
|
71
72
|
## Output
|
|
72
73
|
|
|
@@ -96,7 +96,7 @@ Analyze the changes to understand what's impacted. Two sources, picked by what t
|
|
|
96
96
|
**PR URL** (user passed `github.com/<org>/<repo>/pull/<n>`):
|
|
97
97
|
1. `gh pr diff <n> --repo <org>/<repo> --name-only` for the changed file list
|
|
98
98
|
2. `gh pr diff <n> --repo <org>/<repo>` for the actual diff
|
|
99
|
-
3.
|
|
99
|
+
3. Materialize the PR branch in a dedicated worktree per [`_shared/pr-branch-worktree.md`](../_shared/pr-branch-worktree.md). Use that worktree path as the `cwd` for the rest of the run (including the `cwd` parameter on local execute tools).
|
|
100
100
|
|
|
101
101
|
Either way:
|
|
102
102
|
1. Identify impacted feature areas:
|
|
@@ -273,6 +273,7 @@ Execution itself **must** be sequential because there is only one local Electron
|
|
|
273
273
|
1. Call `muggle-local-execute-test-generation`:
|
|
274
274
|
- `testCase`: Full test case object from the parallel fetch above
|
|
275
275
|
- `localUrl`: User's local URL from the pre-flight question
|
|
276
|
+
- `cwd`: Absolute path of the active working directory — the PR-branch worktree if one was created in Step 2, otherwise the user's repo root. Drives the cross-worktree single-flight lock so concurrent muggle-test runs from different branches serialize.
|
|
276
277
|
- `showUi`: from the `showElectronBrowser` resolution — omit (default visible) for `always`, pass `false` for `never`
|
|
277
278
|
- `freshSession`: `true` if the test case requires a clean browser state (see above), omit otherwise
|
|
278
279
|
2. Store the returned `runId` and tag the result `mode: "regen"`.
|
|
@@ -282,14 +283,16 @@ Execution itself **must** be sequential because there is only one local Electron
|
|
|
282
283
|
2. Call `muggle-local-execute-replay`:
|
|
283
284
|
- `testScript`: from `muggle-remote-test-script-get`
|
|
284
285
|
- `actionScript`: from `muggle-remote-action-script-get`
|
|
285
|
-
- `localUrl`, `showUi`, `freshSession`: same resolution as regen
|
|
286
|
+
- `localUrl`, `cwd`, `showUi`, `freshSession`: same resolution as regen
|
|
286
287
|
3. Store the returned `runId` and tag the result `mode: "replay"`.
|
|
287
288
|
|
|
288
289
|
If a run fails, log it and continue to the next — do not abort the batch. Failures are routed through Step 7C's post-failure handler after the batch completes.
|
|
289
290
|
|
|
290
291
|
### Collect results (in parallel)
|
|
291
292
|
|
|
292
|
-
For every `runId`, issue all `muggle-local-run-result-get` calls in parallel. Extract
|
|
293
|
+
For every `runId`, issue all `muggle-local-run-result-get` calls in parallel. Extract from the **structured response only** (not from `execute`'s stdout tail, which is a truncated display excerpt): `Status`, `Error`, `Duration`, and the `Artifacts` section (always present after a run completes — names `artifactsDir` and lists the files actually on disk).
|
|
294
|
+
|
|
295
|
+
For passed runs, `results.md` inside `artifactsDir` is the step-by-step verdict — read it before summarizing. For failed runs, `stdout.log` + `stderr.log` are always present and `action-script.json` is present when generation reached the step-emission stage (typical for `goal_not_achievable`); use `Error` as the headline verdict and route through Step 7C.
|
|
293
296
|
|
|
294
297
|
### Publish each run to cloud (gated by `autoPublishLocalResults`)
|
|
295
298
|
|
|
@@ -459,6 +462,7 @@ This is a suggestion, not automatic invocation. Skip silently if every test pass
|
|
|
459
462
|
## Guardrails
|
|
460
463
|
|
|
461
464
|
- **Always confirm intent first** — never assume local vs remote without asking
|
|
465
|
+
- **PR URLs always run in a dedicated worktree** — never switch the user's main checkout. Create or reuse `<repo>/.claude/worktrees/<sanitized-branch>` and pass that path as the `cwd` parameter to local execute tools. The cross-worktree single-flight lock relies on this to serialize concurrent runs from different branches.
|
|
462
466
|
- **User MUST select project** — present clickable options via `AskUserQuestion`, wait for explicit choice, never auto-select
|
|
463
467
|
- **Best-effort shortlist use cases** — use the change summary to narrow the list to the most relevant 1–5 use cases and pre-check them; never dump every use case in the project on the user. Always leave an escape hatch to reveal the full list.
|
|
464
468
|
- **Best-effort shortlist test cases** — same idea: pre-check the test cases most relevant to the change summary; never enumerate every test case attached to a use case. Always leave an escape hatch to reveal the full list.
|
|
@@ -195,8 +195,12 @@ If publish rejects with `has no generated actionScript steps to publish` (true z
|
|
|
195
195
|
|
|
196
196
|
### 9. Report
|
|
197
197
|
|
|
198
|
+
**Do not diagnose from `execute`'s response stdout tail.** That tail is a truncated excerpt for human display and routinely cuts off mid-sentence. The only ground truth is the run record.
|
|
199
|
+
|
|
198
200
|
- `muggle-local-run-result-get` with the run id from execute.
|
|
199
|
-
-
|
|
201
|
+
- **Read in this order:** `Status` → `Error` → **`Artifacts` section** (always present after a run completes; names `artifactsDir` and lists the files actually on disk: `action-script.json`, `results.md`, `screenshots/`, `stdout.log`, `stderr.log`). On a `passed` run, `results.md` is the step-by-step verdict with screenshot links — read it before summarizing.
|
|
202
|
+
- **On failure**, the `Artifacts` section is still present. `stdout.log` + `stderr.log` are always there. `action-script.json` is there when generation got far enough to emit it (typical for `goal_not_achievable` / mid-progress crashes — the file holds the agent's attempted steps + halt summary). `results.md` and per-step screenshots are absent on failure (electron-app only emits those on the successful completion path) — don't hunt elsewhere on disk for them.
|
|
203
|
+
- Include in the report: status, duration, pass/fail summary, per-step summary (passed runs), artifact paths, errors if failed, and script view URL when publishing ran.
|
|
200
204
|
|
|
201
205
|
### 9a. Route failures through the failure-mode handler
|
|
202
206
|
|
|
@@ -242,6 +246,7 @@ After reporting results:
|
|
|
242
246
|
|
|
243
247
|
- No silent auth skip.
|
|
244
248
|
- **Never prompt for Electron launch approval** before execution — invoking this skill is the approval. Just run.
|
|
249
|
+
- **Never diagnose a failed run from `execute`'s response stdout tail.** Always call `muggle-local-run-result-get` first; classify only from its structured fields and (when present) the artifacts it names. The execute tail is an excerpt and routinely truncates the failure cause.
|
|
245
250
|
- If replayable scripts exist, do not default to generation without user choice.
|
|
246
251
|
- No hiding failures: surface errors and artifact paths.
|
|
247
252
|
- **Always offer the agent-guidance reminder after every Electron run** (Step 9b) — pass or fail — unless 9a already routed the user into `muggle-feedback`. Never silently end a run without giving the user a one-click path to flag what was wrong.
|