swipium 2.0.1 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. package/CHANGELOG.md +64 -0
  2. package/README.md +32 -20
  3. package/THREAT_MODEL.md +109 -21
  4. package/dist/automationGen/run.js.map +1 -1
  5. package/dist/cli/init.js +101 -11
  6. package/dist/cli/init.js.map +1 -1
  7. package/dist/cli/verify.js +2 -2
  8. package/dist/cli/verify.js.map +1 -1
  9. package/dist/consent/consent.js +365 -17
  10. package/dist/consent/consent.js.map +1 -1
  11. package/dist/context/projectRoot.js +9 -4
  12. package/dist/context/projectRoot.js.map +1 -1
  13. package/dist/context/protocolEra.js +15 -0
  14. package/dist/context/protocolEra.js.map +1 -0
  15. package/dist/drivers/DirectDriver.js +5 -2
  16. package/dist/drivers/DirectDriver.js.map +1 -1
  17. package/dist/featureTesting/executionBootstrap.js +85 -77
  18. package/dist/featureTesting/executionBootstrap.js.map +1 -1
  19. package/dist/flows/run.js +9 -5
  20. package/dist/flows/run.js.map +1 -1
  21. package/dist/lib/abortScope.js +36 -3
  22. package/dist/lib/abortScope.js.map +1 -1
  23. package/dist/lib/android.js +15 -6
  24. package/dist/lib/android.js.map +1 -1
  25. package/dist/lib/codexEnv.js +112 -0
  26. package/dist/lib/codexEnv.js.map +1 -0
  27. package/dist/lib/logger.js +18 -0
  28. package/dist/lib/logger.js.map +1 -1
  29. package/dist/lib/result.js +188 -10
  30. package/dist/lib/result.js.map +1 -1
  31. package/dist/lib/schemaHash.js +20 -31
  32. package/dist/lib/schemaHash.js.map +1 -1
  33. package/dist/lib/simctl.js +102 -3
  34. package/dist/lib/simctl.js.map +1 -1
  35. package/dist/lib/toolSchema.js +143 -0
  36. package/dist/lib/toolSchema.js.map +1 -0
  37. package/dist/lib/wda.js +5 -2
  38. package/dist/lib/wda.js.map +1 -1
  39. package/dist/mobileAudit/runner.js +12 -4
  40. package/dist/mobileAudit/runner.js.map +1 -1
  41. package/dist/oracle/failures.js +2 -2
  42. package/dist/oracle/failures.js.map +1 -1
  43. package/dist/orchestration/testThis/execute.js +21 -4
  44. package/dist/orchestration/testThis/execute.js.map +1 -1
  45. package/dist/orchestration/testThis/pipeline.js +2 -0
  46. package/dist/orchestration/testThis/pipeline.js.map +1 -1
  47. package/dist/orchestration/testThis/plan.js.map +1 -1
  48. package/dist/server.js +564 -139
  49. package/dist/server.js.map +1 -1
  50. package/dist/services/prepareAndroid.js +3 -2
  51. package/dist/services/prepareAndroid.js.map +1 -1
  52. package/dist/services/prepareIos.js +31 -4
  53. package/dist/services/prepareIos.js.map +1 -1
  54. package/dist/services/smoke.js +38 -4
  55. package/dist/services/smoke.js.map +1 -1
  56. package/dist/session/processRegistry.js +12 -1
  57. package/dist/session/processRegistry.js.map +1 -1
  58. package/dist/snapshot/parse.js +64 -9
  59. package/dist/snapshot/parse.js.map +1 -1
  60. package/dist/snapshot/present.js +14 -4
  61. package/dist/snapshot/present.js.map +1 -1
  62. package/dist/snapshot/settle.js +14 -3
  63. package/dist/snapshot/settle.js.map +1 -1
  64. package/dist/tools/act.js +694 -670
  65. package/dist/tools/act.js.map +1 -1
  66. package/dist/tools/agent.js +15 -16
  67. package/dist/tools/agent.js.map +1 -1
  68. package/dist/tools/appControl.js +4 -4
  69. package/dist/tools/appControl.js.map +1 -1
  70. package/dist/tools/appMap.js +11 -13
  71. package/dist/tools/appMap.js.map +1 -1
  72. package/dist/tools/build.js +4 -6
  73. package/dist/tools/build.js.map +1 -1
  74. package/dist/tools/bundletool.js +3 -3
  75. package/dist/tools/bundletool.js.map +1 -1
  76. package/dist/tools/clearOverlay.js +2 -3
  77. package/dist/tools/clearOverlay.js.map +1 -1
  78. package/dist/tools/device.js +4 -3
  79. package/dist/tools/device.js.map +1 -1
  80. package/dist/tools/doctor.js +23 -11
  81. package/dist/tools/doctor.js.map +1 -1
  82. package/dist/tools/explore.js +15 -21
  83. package/dist/tools/explore.js.map +1 -1
  84. package/dist/tools/featureTesting.js +22 -6
  85. package/dist/tools/featureTesting.js.map +1 -1
  86. package/dist/tools/firstRun.js +4 -5
  87. package/dist/tools/firstRun.js.map +1 -1
  88. package/dist/tools/flow.js +13 -17
  89. package/dist/tools/flow.js.map +1 -1
  90. package/dist/tools/flowRepair.js +3 -5
  91. package/dist/tools/flowRepair.js.map +1 -1
  92. package/dist/tools/generate.js +10 -11
  93. package/dist/tools/generate.js.map +1 -1
  94. package/dist/tools/getArtifact.js +114 -10
  95. package/dist/tools/getArtifact.js.map +1 -1
  96. package/dist/tools/health.js +2 -1
  97. package/dist/tools/health.js.map +1 -1
  98. package/dist/tools/ios.js +35 -13
  99. package/dist/tools/ios.js.map +1 -1
  100. package/dist/tools/issues.js +8 -8
  101. package/dist/tools/issues.js.map +1 -1
  102. package/dist/tools/jobs.js +26 -10
  103. package/dist/tools/jobs.js.map +1 -1
  104. package/dist/tools/metro.js +4 -5
  105. package/dist/tools/metro.js.map +1 -1
  106. package/dist/tools/mobileAudit.js +11 -10
  107. package/dist/tools/mobileAudit.js.map +1 -1
  108. package/dist/tools/note.js +2 -3
  109. package/dist/tools/note.js.map +1 -1
  110. package/dist/tools/prepareIosTarget.js +102 -31
  111. package/dist/tools/prepareIosTarget.js.map +1 -1
  112. package/dist/tools/prepareTarget.js +3 -5
  113. package/dist/tools/prepareTarget.js.map +1 -1
  114. package/dist/tools/report.js +4 -5
  115. package/dist/tools/report.js.map +1 -1
  116. package/dist/tools/resolveArtifact.js +2 -3
  117. package/dist/tools/resolveArtifact.js.map +1 -1
  118. package/dist/tools/resolveTarget.js +3 -6
  119. package/dist/tools/resolveTarget.js.map +1 -1
  120. package/dist/tools/screenRecord.js +2 -3
  121. package/dist/tools/screenRecord.js.map +1 -1
  122. package/dist/tools/screenshot.js +2 -2
  123. package/dist/tools/screenshot.js.map +1 -1
  124. package/dist/tools/smoke.js +4 -2
  125. package/dist/tools/smoke.js.map +1 -1
  126. package/dist/tools/snapshot.js +6 -7
  127. package/dist/tools/snapshot.js.map +1 -1
  128. package/dist/tools/startSession.js +12 -16
  129. package/dist/tools/startSession.js.map +1 -1
  130. package/dist/tools/suite.js +2 -4
  131. package/dist/tools/suite.js.map +1 -1
  132. package/dist/tools/testSuite.js +13 -19
  133. package/dist/tools/testSuite.js.map +1 -1
  134. package/dist/tools/testThis.js +28 -13
  135. package/dist/tools/testThis.js.map +1 -1
  136. package/dist/tools/visual.js +7 -9
  137. package/dist/tools/visual.js.map +1 -1
  138. package/dist/tools/wait.js +130 -21
  139. package/dist/tools/wait.js.map +1 -1
  140. package/dist/tools/wda.js +199 -60
  141. package/dist/tools/wda.js.map +1 -1
  142. package/dist/version.js +1 -1
  143. package/docs/README.md +4 -4
  144. package/docs/ci-reports.md +39 -7
  145. package/docs/concepts.md +37 -25
  146. package/docs/flows.md +1 -1
  147. package/docs/mcp-server.md +281 -76
  148. package/docs/physical-devices.md +4 -4
  149. package/docs/tools.md +59 -30
  150. package/package.json +4 -3
package/docs/tools.md CHANGED
@@ -51,19 +51,26 @@ Annotations describe the worst case of a tool. [Consent](concepts.md#consent) ga
51
51
 
52
52
  ### Response modes
53
53
 
54
- `responseMode` controls the **text** channel only. `structuredContent` always carries the complete payload in every mode.
54
+ `responseMode` controls the **text** channel, plus one thing in `structuredContent`: element lists. `structuredContent` always carries every field in every mode, but outside `verbose` the `elements` of `qa_snapshot` and `qa_act` are one-line `@eN` strings (see [Element lines](#element-lines)) instead of one JSON object per element. Claude Code and Codex show the model `structuredContent`, so the lines are what the model reads; they are about 40% smaller.
55
55
 
56
56
  | Mode | Text channel |
57
57
  | --- | --- |
58
58
  | `compact` | The summary line plus any `swipium://` URIs (`artifactUri`, `artifactUris`, `screenshotUri`, `reportUri`). No JSON. |
59
59
  | `normal` (default) | The summary plus one block of JSON. Fields the summary already rendered are left out of that JSON, and a `renderedAbove` key lists them: `elements` and `diff` for `qa_snapshot`; `elements`, `removed`, `hint`, and `stateChanged` for `qa_act`. `renderedAbove` exists only in the text; it is never in `structuredContent`. |
60
- | `verbose` | The summary plus the full payload as JSON, including the fields `normal` leaves out (so there is no `renderedAbove`). |
60
+ | `verbose` | The summary plus the full payload as JSON, including the fields `normal` leaves out (so there is no `renderedAbove`). `elements` in `structuredContent` (and in that JSON) are full objects: `ref`, `role`, `label`, `id`, `text`, `bounds`, `clickable`, `focused`, `secure`. |
61
61
 
62
62
  The mode is a session setting, set by `qa_start_session` or `qa_test_this` (`responseMode`). Calls without a `sessionId` use the `responseMode` argument when the tool has one, else `normal`. The mode is chosen before a call runs, so `qa_test_this {sessionId, responseMode}` on an existing session takes effect from the next call. Errors always show their heading lines; compact mode drops only the JSON.
63
63
 
64
64
  ### Result envelope
65
65
 
66
- A success is `{ok:true, …payload}`. A budget stop is also a success: `{ok:true, stopped:true, reason}` (for example `action budget reached (20/20)`). Some results add `notes[]` (non-fatal remarks, such as parameters ignored for the chosen mode) or `warnings[]`.
66
+ A success is `{ok:true, …payload}`. A budget stop is also a success: `{ok:true, stopped:true, reason}` (for example `action budget reached (20/20)`). Some results add `notes[]` (non-fatal remarks, such as parameters ignored for the chosen mode, a clamped timeout, or a job still driving the device) or `warnings[]`.
67
+
68
+ Because Claude Code and Codex show the model `structuredContent` rather than the text block, a success's `structuredContent` also carries two copies from the text:
69
+
70
+ - **`summary`** (first key): the summary's first line, capped at 300 characters. `qa_job_status`, `qa_metro`, `qa_build` plans, and a finished `qa_test_this {waitForCompletion:true}` copy the whole summary instead (capped at 4000 characters), because their later lines say things the payload does not. Rendered `@eN` lines are never copied; they are already in `elements`. Left out when the payload has its own `summary`.
71
+ - **`next`** (last key): the next calls the summary suggests, as strings that start with the tool name (for example `qa_report to summarize what was verified`). Left out when there is none, or when the payload already has `next`, `nextSteps`, `nextBestAction`, `nextAction`, or `nextRecommendedAction`.
72
+
73
+ Errors carry neither; their guidance is `nextSteps`.
67
74
 
68
75
  An error has `isError:true` and this `structuredContent`:
69
76
 
@@ -81,14 +88,21 @@ Triage fields (`bucket`, `owner`, `canSwipiumFix`) are not part of the error con
81
88
 
82
89
  The text channel renders the same error as `❌ <what>`, then `changedState=… retrySafe=…`, `next: …`, and `hint: …` lines.
83
90
 
91
+ Errors are size-capped, because they often echo caller input: `what` and each `nextSteps` entry keep the head and tail of anything over 2000 characters, at most 20 `nextSteps` are kept, and an error still over 64 KB drops its extra fields (listed in `extraDropped`).
92
+
84
93
  ### Unknown arguments and stale clients
85
94
 
86
- - **Unknown arguments**: every tool rejects top-level arguments its input schema does not declare, before anything runs. The call returns `INVALID_ARGUMENT` with `unknownArguments` and `acceptedParameters`. For example, `qa_app_control {action:"force_stop", appId:"…"}` is refused: `appId` is not a parameter, and the action always targets the session's app. Deprecated aliases that are still declared (`qa_wda udid`, `qa_suite_generate creativityLevel`, `qa_issue_log until`) are accepted. Nested objects are not checked this way.
87
- - **Stale clients**: a client started before an upgrade may still send 1.5-era calls. Removed tool names, `qa_ios` with `action:"screenshot"` or `action:"wda_*"`, and `qa_wait` with `for:"job_done"` return `failureCode:"STALE_CLIENT"` with `removedCall`, `replacement` (the call to use), and `clientHint` (restart the client so it reloads the tool list). See [Migrating from 1.5.0](#migrating-from-150). `qa_doctor` with `expectedVersion`, `expectedToolCount`, or `expectedSchemaHash` detects the same condition.
95
+ Every call is checked against the tool's input schema before anything runs, in this order: stale-client shapes, then unknown arguments, then per-field validation. Each refusal has `changedState:false` and `retrySafe:true`.
96
+
97
+ - **Unknown arguments**: every tool rejects top-level arguments its input schema does not declare. The call returns `INVALID_ARGUMENT` with `unknownArguments` (up to 20 names; `unknownArgumentCount` gives the total when there are more) and `acceptedParameters`. For example, `qa_app_control {action:"force_stop", appId:"…"}` is refused: `appId` is not a parameter, and the action always targets the session's app. Deprecated aliases that are still declared (`qa_wda udid`, `qa_suite_generate creativityLevel`, `qa_issue_log until`) are accepted. Inside nested objects (such as `qa_act target`), undeclared keys are dropped, not rejected.
98
+ - **Invalid values**: a missing required argument, a wrong type, a value outside an enum, or a number outside its bounds returns `INVALID_ARGUMENT` with `invalidArguments` (up to 20 `{path, message}` entries, such as `{path:"target.text", message:"Expected string, received number"}`) and `acceptedParameters`. `what` joins them: `qa_act: invalid arguments: action: Required. Nothing was run.`
99
+ - **Stale clients**: a client started before an upgrade may still send 1.5-era calls. Removed tool names, `qa_ios` with `action:"screenshot"` or `action:"wda_*"`, and `qa_wait` with `for:"job_done"` return `failureCode:"STALE_CLIENT"` with `removedCall`, `replacement` (the call to use), and `clientHint` (restart the client so it reloads the tool list). See [Migrating from 1.5.0](#migrating-from-150). A tool name that never existed gets the plain MCP `Tool <name> not found` error instead. `qa_doctor` with `expectedVersion`, `expectedToolCount`, or `expectedSchemaHash` detects the same condition.
88
100
 
89
101
  ### Jobs and cancellation
90
102
 
91
- Long operations return a `jobId` to poll with [qa_job_status](#qa_job_status); cancelled work returns `CANCELLED`. The job lifecycle, status versus `result.state`, long-polling, and cancellation rules are in [Sessions and jobs](concepts.md#sessions-and-jobs).
103
+ Long operations return a `jobId` to poll with [qa_job_status](#qa_job_status); cancelled work returns `CANCELLED`. The job lifecycle, status versus `result.state`, long-polling, and cancellation rules are in [Sessions and jobs](concepts.md#sessions-and-jobs). What each tool does when its call is cancelled is in its section.
104
+
105
+ A tool that drives the device (taps, installs, launches, boots, toggles device settings, records, or starts a job that does) still runs when the session has a job running, but while that job drives the same device the result gets a note: `job <jobId> is still driving this device; actions may interleave. Poll qa_job_status or qa_job_cancel first.` `qa_build` and `qa_bundletool` jobs run on the host and never trigger it, and neither do observation tools such as `qa_snapshot` or `qa_screenshot`.
92
106
 
93
107
  ## Tool index
94
108
 
@@ -128,7 +142,7 @@ Hints: **RO** read-only, **D** destructive, **I** idempotent write, blank for ot
128
142
  | `qa_screenshot` | drive | | | Screenshot artifact with coordinate-space metadata. |
129
143
  | `qa_note` | drive | | | Record a workflow outcome for the report. |
130
144
  | `qa_visual` | drive | | yes | Screenshot checks: assert, baseline, diff, OCR find, image find. |
131
- | `qa_wait` | drive | RO | | Wait for `device_online` or `metro_ready`. |
145
+ | `qa_wait` | drive | RO | | Wait for `device_online`, `metro_ready`, `wda_ready`, or `simulator_booted`. |
132
146
  | `qa_smoke` | run | | | Launch, baseline health, evidence, and every saved flow. |
133
147
  | `qa_explore` | run | | yes | Bounded, safe-by-default exploration job; builds a screen graph. |
134
148
  | `qa_report` | run | | | Session report, plus optional CI exports. |
@@ -160,7 +174,7 @@ Autopilot, orientation, job polling, blockers, and artifacts.
160
174
 
161
175
  Autopilot for a low-context request such as "test this app". It resolves the project, finds or builds an artifact, picks a simulator, then plans or executes prepare > smoke > (explore) > report > (suite).
162
176
 
163
- - **`mode`**: `plan` (default) has no side effects and returns the plan, preconditions, and any consent it will need. `execute` returns `state:"running"` and a `jobId` at once. `interactive` asks the credentials question up front (when the project likely has a login and no credentials are available) and then runs as a job like `execute`. `waitForCompletion:true` blocks up to `timeoutMs` (default 120000) and returns the terminal result directly.
177
+ - **`mode`**: `plan` (default) has no side effects and returns the plan, preconditions, and any consent it will need. `execute` returns `state:"running"` and a `jobId` at once. `interactive` asks the credentials question up front (when the project likely has a login and no credentials are available) and then runs as a job like `execute`. `waitForCompletion:true` blocks up to `timeoutMs` (default 45000, max 50000; larger values are clamped to 50000 with a note in `notes`, never rejected) and returns the terminal result directly, or `state:"running"` with `timedOutWaiting:true` and the `jobId` to poll. Cancelling that call ends the wait early with the same `state:"running"` result; the job keeps running.
164
178
  - **`goal`** sets default flags; explicit `explore`, `generateSuite`, and `stopOnNeedsInput` win.
165
179
 
166
180
  | goal | Explore | Suite | Stops for input | Notes |
@@ -173,7 +187,7 @@ Autopilot for a low-context request such as "test this app". It resolves the pro
173
187
  | `test_login` | no | no | yes | Stops for credentials when none are available. |
174
188
  | `reproduce_bug` | yes | no | no | Focus with `goalText`. |
175
189
 
176
- - **Other parameters**: `platform` (`android` or `ios`, default inferred), `device`, `buildIfNeeded` (default true), `allowOutsideRoot`, `fastSmoke`, `responseMode`, `consentId`/`approve`. `preferRealDevice:true` always returns `PHYSICAL_DEVICE_UNSUPPORTED`, even when no phone is connected (see [Devices](concepts.md#devices)).
190
+ - **Other parameters**: `platform` (`android` or `ios`, default inferred), `device`, `buildIfNeeded` (default true), `allowOutsideRoot`, `fastSmoke`, `responseMode` (`compact` returns the summary + URIs only; it stays the session's default for later calls), `consentId`/`approve`. `preferRealDevice:true` always returns `PHYSICAL_DEVICE_UNSUPPORTED`, even when no phone is connected (see [Devices](concepts.md#devices)).
177
191
  - **Consent**: one combined `test_this_plan` consent covers build, boot, and install; its risk is the highest of its steps. Its envelope carries `sessionId`, so the approving re-call reuses the session.
178
192
  - **Terminal states** (in the `qa_job_status` result): `completed`, `blocked`, `unsafe`, or `needs_input`. Every terminal state writes a report. The result keeps the report compact (`reportSummary`, `reportUri`, suite and app-map counts) and adds `attempted`, `workaroundsAttempted`, `artifactChoice`, `targetChoice`, `blockers[]`, and `nextRecommendedAction`.
179
193
  - **needs_input**: the run stopped on one question it was asked to stop for (`stopOnNeedsInput`, `goal:"test_login"`, or `interactive`), such as a login form that needs credentials. The result carries `needsInput` (the question, its fields, and a `resume` call), and `nextRecommendedAction` is that call. When the question can be asked before any work starts, `qa_test_this` returns `state:"needs_input"` directly, with no `jobId`. Without those flags, the run completes with pre-login coverage and returns the question as `optionalQuestion`. Answering "test pre-login only" sets `loginOutOfScope:true` for the session.
@@ -199,7 +213,7 @@ Autopilot for a low-context request such as "test this app". It resolves the pro
199
213
 
200
214
  ### qa_job_status
201
215
 
202
- Polls a job. Parameters: `sessionId`, `jobId`, `waitMs` (0 to 120000; long-polls until the job leaves `running`). Returns `{jobId, kind, status, progress, progressDetail, error, result, artifactUris}`, plus `waited:{waitedMs, timedOut}` when `waitMs` is set. See [Sessions and jobs](concepts.md#jobs).
216
+ Polls a job. Parameters: `sessionId`, `jobId`, `waitMs` (default 0, return at once; otherwise long-polls until the job leaves `running`; use 45000, values above 50000 are clamped to 50000 so one call ends before a 60 s client tool timeout). An unknown `jobId` is `INVALID_ARGUMENT`. Cancelling the call ends the wait with `CANCELLED` and leaves the job running (use `qa_job_cancel` to stop it). Returns `{jobId, kind, status, progress, progressDetail, error, result, artifactUris}`, plus `waited:{waitedMs, timedOut}` when `waitMs` is set. See [Sessions and jobs](concepts.md#jobs).
203
217
 
204
218
  ### qa_job_cancel
205
219
 
@@ -221,7 +235,7 @@ Answers a `needs_input` question. Parameters: `sessionId`, `kind` (for example `
221
235
 
222
236
  ### qa_get_artifact
223
237
 
224
- Reads a `swipium://session/<id>/<kind>/<name>` artifact, for clients without MCP resources. `mode` defaults to `metadata` for images and `inline` for text. A text artifact whose redaction was `partial` reports `redaction:"partial"` plus `redactionNote` (see [Secrets and redaction](concepts.md#secrets-and-redaction)).
238
+ Reads a `swipium://session/<id>/<kind>/<name>` artifact, for clients without MCP resources. `mode` defaults to `inline` for text and `metadata` for images and every other binary (screen recordings, archives). `metadata` returns `{uri, mime, kind, bytes, path, redaction, hint}`; `inline` returns an image as image content and any other binary as a base64 blob resource. Text over 1 MB returns the first 1 MB (the last 1 MB for logs, including `*.log` files) with a `[swipium: truncated ...]` marker naming the local file; binaries over 8 MB are not inlined (the result names the local file instead). MCP `resources/read` applies the same caps. A text artifact whose redaction was `partial` reports `redaction:"partial"` plus `redactionNote` in metadata, and a separate warning block when inlined (see [Secrets and redaction](concepts.md#secrets-and-redaction)). An unknown URI returns `INVALID_ARGUMENT`.
225
239
 
226
240
  ## Setup
227
241
 
@@ -229,7 +243,13 @@ Check the toolchain, open a session, and prepare a simulator.
229
243
 
230
244
  ### qa_doctor
231
245
 
232
- Checks Node, the Android SDK and emulator, Xcode and `simctl`, WDA, and client freshness. `platform` is `android`, `ios`, or `both` (default `both` on macOS, where it is ready if either platform is; `android` elsewhere). `client` (`claude`, `gemini`, `codex`, `cursor`, or `vscode`) tailors registration hints. `expectedVersion`, `expectedToolCount`, and `expectedSchemaHash` add a `client-freshness` check that reports a stale client.
246
+ Checks Node, the Android SDK and emulator, Xcode and `simctl`, WDA, and client freshness. `platform` is `android`, `ios`, or `both` (default `both` on macOS, where it is ready if either platform is; `android` elsewhere). `client` (`claude`, `gemini`, `codex`, `cursor`, or `vscode`) adds a `clientHint` with registration advice. When the connected client is Codex (or `client:"codex"`), the result adds two optional rows and a `codex` field:
247
+
248
+ - **`codex-env`**: Codex passes MCP servers only a fixed env whitelist plus `env_vars` and the `env` table, so shell exports never arrive otherwise. The row lists which Swipium env names are visible (names only, never values) and warns when no Android SDK is found or `java -version` fails without `JAVA_HOME`: forward `ANDROID_HOME` / `JAVA_HOME` if you installed them in a custom location, or install them first. Approval grants (`SWIPIUM_CONSENT_PREAPPROVE`, `SWIPIUM_ALLOW_REMOTE_WDA`) are never in the default list; set them literally in `env = { ... }`.
249
+ - **`codex-tool-timeout`**: a reminder to keep `tool_timeout_sec` at 600 or more and `startup_timeout_sec` at 30 (the server cannot read them; `swipium init codex` writes both).
250
+ - **`codex`**: `{envVarsLine, toolTimeoutSec}`, the exact `env_vars = [...]` line for `[mcp_servers.swipium]`; the text output prints it too.
251
+
252
+ `expectedVersion`, `expectedToolCount`, and `expectedSchemaHash` add a `client-freshness` check that reports a stale client.
233
253
 
234
254
  ### qa_start_session
235
255
 
@@ -282,6 +302,7 @@ Prepares an iOS Simulator: picks and boots one, installs a simulator `.app`, lau
282
302
  - **Parameters**: `sessionId`, `app` (absolute or project-relative), `bundleId`, `device` (UDID or name substring), `launch`, `attachWda`, `consentId`/`approve`.
283
303
  - **`attachWda`**: `auto` (default) probes WDA and stays visual-only, with a recorded workaround, when it is unreachable, non-loopback, or session creation fails. `required` fails instead (`WDA_UNREACHABLE`, `WDA_SESSION_FAILED`, or `DESTRUCTIVE_REFUSED` for a non-loopback URL). `skip` does not probe.
284
304
  - **Consent**: `install_app`, risk low for an app inside the project root, medium outside it.
305
+ - **Cold boot**: the call waits at most 30 s for the simulator to boot. If it is still booting, the simulator is bound to the session and the call returns `ok` with `status:"booting"`, `udid`, `name`, `elapsedMs`, and a `jobId`: the rest (boot, install, launch, WDA check) runs as a `prepare_ios` job with the consent already given. Poll `qa_job_status {jobId, waitMs:45000}`; the finished job's `result` has the same fields as a direct answer.
285
306
  - **Failure codes**: `IPA_NEEDS_REAL_DEVICE` (a `.ipa` is refused), `IOS_SIMULATOR_APP_MISSING`, `IOS_APP_WRONG_ARCH`, `SIMULATOR_RUNTIME_MISSING`, `SIMULATOR_BOOT_FAILED`, `SIMULATOR_BOOT_TIMEOUT`, `BUNDLE_ID_NOT_FOUND`.
286
307
 
287
308
  ### qa_ios
@@ -291,23 +312,28 @@ Direct iOS Simulator control (macOS only). `action` is one of:
291
312
  | action | Parameters | Consent |
292
313
  | --- | --- | --- |
293
314
  | `list` | | |
294
- | `boot` | `device` (UDID or name substring). Binds the simulator to the session. | none (low-risk, reversible) |
315
+ | `boot` | `device` (UDID or name substring; default: a booted simulator, else the first iPhone). Binds the simulator to the session. | none (low-risk, reversible) |
295
316
  | `install` | `app` (a `.app`, absolute or project-relative) | `install_app`, medium |
296
317
  | `launch`, `terminate` | `bundleId` | |
297
318
  | `openurl` | `url` (deep link) | |
298
319
  | `logs` | `last` (default `5m`) | |
299
- | `privacy_reset` | `bundleId`, `service` (for example `location`, `photos`, `camera`, `all`) | low |
300
- | `erase` | `device`. Wipes the simulator. | `erase_device`, high |
320
+ | `privacy_reset` | `service` (for example `location`, `photos`, `camera`, `all`), `bundleId` (default: the session's app) | `privacy_reset`, low |
321
+ | `erase` | `device` (default: the bound simulator). Wipes the simulator. | `erase_device`, high |
301
322
 
302
- Screenshots go through `qa_screenshot`, and WebDriverAgent through `qa_wda`. The old `wda_*` and `screenshot` actions return `STALE_CLIENT`.
323
+ `boot` waits at most 40 s. A booted simulator returns `status:"booted"`. A cold boot that takes longer returns `ok` with `status:"booting"`, `udid`, `name`, `elapsedMs`, and `next`; the simulator is already bound and keeps booting in the background, so poll `qa_wait {for:"simulator_booted"}` before `install` or `launch`. Cancelling the call returns `CANCELLED` (`changedState:true`) and leaves the boot running.
324
+
325
+ Every action except `list` and `boot` needs a simulator bound to the session (by `boot`, `qa_prepare_ios_target`, or `qa_test_this`), else `NO_DEVICE`. Screenshots go through `qa_screenshot`, and WebDriverAgent through `qa_wda`. The old `wda_*` and `screenshot` actions return `STALE_CLIENT`.
303
326
 
304
327
  ### qa_wda
305
328
 
306
329
  Diagnoses, attaches, or manages WebDriverAgent for structured iOS automation. Without WDA, iOS stays visual-only and `qa_visual` does the checking (see [iOS modes](concepts.md#ios-modes)).
307
330
 
308
331
  - **`action`**: `status`, `doctor`, `diagnose`, `logs`, and `tune` inspect an existing setup. `attach` connects to an external WDA at `webDriverAgentUrl` (default `http://127.0.0.1:8100`). `build` and `start` manage one (consent `wda_build` / `wda_start`, medium) from `wdaProjectPath` (default: an installed Appium WebDriverAgent when one is found), with `derivedDataPath` and `scheme` (default `WebDriverAgentRunner`); build and start output is captured as artifacts. `stop` terminates it.
332
+ - **Long-running actions** (one call stays under a 60 s client tool timeout):
333
+ - `build` runs `xcodebuild build-for-testing` as a background job (kill timer 10 min) and returns `{jobId, status:"running"}` at once. Poll `qa_job_status` (with `waitMs`); the job result carries `built`, `logUri`, `wdaBuildProduct`, and on failure `failureCode` (`WDA_BUILD_FAILED` or `WDA_SIGNING_FAILED`) and `nextSteps`. `qa_job_cancel` stops the build.
334
+ - `start` launches WDA and waits for `/status` for at most 45 s (or `ios.wda.startupTimeoutMs`, default 120000, when smaller). If WDA is still starting, it returns `ok` with `status:"starting"`, `pid`, `logUri`, and `remainingStartupMs`; poll `qa_wait {for:"wda_ready"}` until satisfied, then `attach`. `WDA_START_FAILED` is returned when the xcodebuild process exits early or the startup timeout is already spent. Cancelling the call during that wait returns `CANCELLED` (`changedState:true`) and leaves the managed WDA running, so `attach` can still use it. With `ios.wda.reuse` (default true), a WDA that already answers ready at the URL is reused (`reused:true`, `started:false`) instead of starting a second one.
309
335
  - **`device`**: the simulator UDID behind this WDA (default: the session device). `udid` is a deprecated alias. `bundleId` defaults to the session's app. A non-loopback URL needs `allowNonLoopback:true` plus consent (see [iOS modes](concepts.md#ios-modes)).
310
- - **Failure codes**:
336
+ - **Failure codes** (for `build`, the build failures arrive in the job result):
311
337
  - `attach`: `MULTIPLE_DEVICES` whenever no `device` is given and none is bound to the session (it never guesses), `WDA_UNREACHABLE`, `WDA_SESSION_FAILED`, `STALE_WDA_DEVICE`, `DESTRUCTIVE_REFUSED` (non-loopback URL without approval).
312
338
  - `build` and `start`: `NO_DEVICE` (no UDID given or bound), `BACKEND_UNSUPPORTED` (no Xcode command line tools), `NO_ARTIFACT` (no WebDriverAgent project found), `WDA_BUILD_FAILED`, `WDA_SIGNING_FAILED`, `WDA_START_FAILED` (with `managedPid` while a managed WDA is still running: attach to it or `stop` it first).
313
339
 
@@ -396,8 +422,9 @@ Observe, act, assert, and collect evidence.
396
422
 
397
423
  ### qa_snapshot
398
424
 
399
- Captures the screen as compact, addressable elements (`@e1`, `@e2`, …) with a `snapshotQuality` verdict. Interactive elements only, with no screenshot. Busy screens are capped; `filter` (a substring of text, label, id, or role) finds capped elements, and `diff:true` returns only what changed since the previous snapshot. Refs are invalid after navigation.
425
+ Captures the screen as compact, addressable elements (`@e1`, `@e2`, …) with a snapshot-quality verdict (`quality`, plus `qualityReasons`). Interactive elements only, with no screenshot. Busy screens are capped at 60 elements (the most interaction-relevant ones, focused, clickable, fields, and scrollables first, kept in screen order, with `elementsOmitted` counting the rest); `filter` (a substring of text, label, id, or role) finds capped elements, and `diff:true` returns only what changed since the previous snapshot. Refs are invalid after navigation.
400
426
 
427
+ - **Element lines**<a id="element-lines"></a>: each entry of `elements` is one line, the same line the text block renders: `@e3 [button] "Log in" #login_btn text="Sign in" [40,200][1040,245] (focused,secure,non-clickable)`. In order: the ref, the role, the name (the label, else the text; `""` when neither), `#id` when there is one (quoted when it is not a plain token), `text="..."` only when it differs from the label, the bounds as `[x1,y1][x2,y2]` (left out when unknown), and flags in parentheses when any apply. Names are JSON-quoted (quotes and newlines escaped) and cut at 80 characters with `...`. Known secrets are redacted and secure values are `«secure»` before the line is built. `responseMode:"verbose"` returns the full objects instead; `qa_inspect` returns every attribute of one ref.
401
428
  - **iOS with WDA**: an element's `id` is its accessibility identifier (WDA's `name` when it differs from the label). `TextField`, `SecureTextField`, `SearchField`, and `TextView` are `text-field` elements that show their typed value; secure fields stay masked.
402
429
  - **Overlays**: banners and snackbars are reported only with an overlay signal (an overlay-like class or id, a dismiss control, or banner wording). Navigation-bar titles, text fields, and list rows are never reported as overlays.
403
430
  - **Visual-fallback**: after `maxSnapshotFailures` consecutive failed dumps (default 3), the session switches to visual-fallback (`VISUAL_ONLY_SCREEN`) for that screen. Each later `qa_snapshot` and `qa_act` still tries one bounded structured dump; the first success switches back (`modeRecovered:true`) and resets the count.
@@ -421,12 +448,12 @@ Performs one action, waits for the screen to settle, and observes: `changed`, `s
421
448
  | `scroll` | `direction` | `untilVisible` (a target), `maxScrolls` (default 8) |
422
449
  | `press` | `key` (`back`, `home`, `enter`) | |
423
450
  | `open_url` | `url` | |
424
- | `wait` | | `for` (`{settled:true}` by default, or an element), `timeoutMs` (default 8000) |
451
+ | `wait` | | `for` (`{settled:true}` by default, or an element: `ref`, `text`, `id`, or a WDA `selector`), `timeoutMs` (default 8000) |
425
452
 
426
- Every action also takes `observe` and `timeoutMs` (the settle-wait cap).
453
+ Every action also takes `observe` and `timeoutMs` (the settle-wait cap, default 8000). `timeoutMs` is at most 50000: larger values are clamped to 50000 with a note in `notes` (not rejected); negative values are `INVALID_ARGUMENT`. A `wait` for an element that never shows up returns `ELEMENT_NOT_FOUND`; a `selector` wait off WDA is `BACKEND_UNSUPPORTED`. Cancelling a `wait` stops polling at once and returns `CANCELLED`.
427
454
 
428
455
  - **Targets**: an `@eN` ref, `text`, `id`, a native `selector` on WDA (`accessibility id`, `name`, `predicate string`, or `class chain`), or `x`/`y` coordinates.
429
- - **Observe**: `diff` (the default once a snapshot exists) returns added and removed elements; `full` returns the capped list; `none` returns verdicts only. When more than half of the post-action elements are new (a navigation), `diff` returns the full capped list with `diffAsFull:true`, `addedCount`, and `removedCount`.
456
+ - **Observe**: `elements` use the `qa_snapshot` [element lines](#element-lines) (objects in `verbose`). `diff` (the default once a snapshot exists) returns added and removed elements; `full` returns the capped list; `none` returns verdicts only. When more than half of the post-action elements are new (a navigation), `diff` returns the full capped list with `diffAsFull:true`, `addedCount`, and `removedCount`.
430
457
  - **`changed`**: true when elements appeared or disappeared, when positions moved (so a scroll that only shifts content counts), or when a checked, selected, or value state changed. State changes are listed in `stateChanged`, so a toggle is never retried as a press (which would toggle it back).
431
458
  - **Keyboard**: if a tap target's center is inside the soft keyboard's frame, Swipium hides the keyboard (never a blind BACK), waits, re-resolves the target, and taps only once it is uncovered; success adds `keyboardHidden:true`. If the keyboard cannot be hidden, the result is `KEYBOARD_OBSTRUCTION` with `changedState:false`. If it was hidden but the target is gone or still covered, `KEYBOARD_OBSTRUCTION` with `changedState:true` and `keyboardHidden:true`. Nothing is tapped in either case. Targets above the keyboard (an accessory toolbar, suggestion chips) are tapped without hiding it. When the keyboard's area is unknown (no frame, or a frame taller than 55% of the screen), Swipium taps without hiding it and warns `keyboard is up; could not determine its area`.
432
459
  - **Other overlays**: an element drawn over the target returns `OVERLAY_OBSTRUCTION` with `blockedByOverlay` instead of a blind tap. `ignoreOverlay:true` skips both checks. Coordinate taps are always treated as deliberate.
@@ -458,7 +485,7 @@ Records a structured outcome for one workflow, so the report is honest about wha
458
485
  - `workflow` and `outcome` (`pass`, `fail`, `blocked`, `skipped`, `not_applicable`) are required.
459
486
  - `category` (`app_bug`, `mcp_limitation`, `missing_test_data`, `intentionally_skipped`, `destructive_refused`, `other`) says why and is independent of the outcome. A failing note without a category is recorded as `app_bug`, which also lands in the issue ledger as an app-owned `app_bug` with medium severity. Pass `category:"mcp_limitation"` for tool problems.
460
487
  - Use `outcome:"blocked"` with `missingPrecondition`, `requiredState`, and `recommendedSetup` instead of a false failure.
461
- - Attach evidence in `artifactUris`. For a screenshot-verified check, use `qa_visual mode:"assert"`.
488
+ - `reason` explains the outcome. Attach evidence in `artifactUris`; `verifiedVisually:true` says the pass was confirmed from an attached screenshot. For a screenshot-verified check, `qa_visual mode:"assert"` does both in one call.
462
489
 
463
490
  ### qa_visual
464
491
 
@@ -485,7 +512,7 @@ A mode called without its required argument returns `INVALID_ARGUMENT`.
485
512
 
486
513
  ### qa_wait
487
514
 
488
- Blocks (bounded) until a setup condition holds, instead of a shell `sleep`. `for`: `device_online` (default timeout 180000 ms) or `metro_ready` (60000 ms). Returns `satisfied`, `timedOut`, and the current state. To wait for a job, use `qa_job_status` with `waitMs`.
515
+ Blocks (bounded) until a setup condition holds, instead of a shell `sleep`. `for`: `device_online` (an adb device; Android only), `metro_ready`, `wda_ready` (the session's WebDriverAgent `/status` reports ready: the attached WDA, else the URL of the last `qa_wda start`, else the configured `ios.wda.url`; a non-loopback configured URL is refused with `DESTRUCTIVE_REFUSED` until it is attached with consent), or `simulator_booted` (the iOS Simulator bound to the session finished booting, for example after `qa_ios boot` returned `status:"booting"`; no bound simulator is `NO_DEVICE`, and a boot that failed in the background is `SIMULATOR_BOOT_FAILED` or `SIMULATOR_BOOT_TIMEOUT` instead of a timeout). `timeoutMs`: integer of at least 0, default 45000; larger values (older docs used 60000 or 180000) are accepted and clamped to 50000 with a note, so one call stays under a 60 s client tool timeout. On `timedOut`, call again. Returns `satisfied`, `timedOut`, and the current state. Cancelling the call stops polling at once and returns `CANCELLED`. To wait for a job, use `qa_job_status` with `waitMs`.
489
516
 
490
517
  ## Run
491
518
 
@@ -497,6 +524,8 @@ Server-side smoke on a prepared device: optionally launches (`launch`, default t
497
524
 
498
525
  Repository flows are untrusted, so `qa_smoke` never runs a flow with mutating steps or an external OCR or visual provider implicitly. Such a flow is recorded as `blocked` (category `destructive_refused`) with a pointer to `qa_flow_run` and its consent.
499
526
 
527
+ Cancelling the call skips the rest of the baseline and flows and returns `CANCELLED` (`changedState:true`). The interrupted workflow is noted as `skipped`, never as a failure or a health finding.
528
+
500
529
  ### qa_explore
501
530
 
502
531
  Bounded, safe-by-default exploration of the launched app, as a job. It observes screens, taps ranked safe actions, checks health after each, and builds a screen graph (JSON and Markdown, `graphUri` in the job result). Taps are recorded for `qa_generate`, and the app map is updated.
@@ -552,7 +581,7 @@ Searches the feature index, static topology, runtime graph, and tests for a natu
552
581
 
553
582
  ### qa_app_map_feature_scope
554
583
 
555
- Resolves a feature (`featureId`, or a free-text `query`) into a focused test scope: code symbols, static and runtime screens, existing tests, objective, coverage gaps, strategy, and ranked candidates. It asks one disambiguation question only on a genuine tie. Works without a map (it falls back to a code scan). `includeCode` (default true) and `limit` (default 8 per list) shape query mode; `sessionId` adds runtime evidence.
584
+ Resolves a feature (`featureId`, or a free-text `query`) into a focused test scope: code symbols, static and runtime screens, existing tests, objective, coverage gaps, strategy, and ranked candidates. It asks one disambiguation question only on a genuine tie. Works without a map (it falls back to a code scan). `includeCode` (default true) and `limit` (default 8 per list) shape query mode; `platform` (`android` or `ios`) only names the platform in the objective; `sessionId` adds runtime evidence.
556
585
 
557
586
  ### qa_app_map_update
558
587
 
@@ -566,7 +595,7 @@ A focused test of one named `feature` (natural language).
566
595
 
567
596
  - **`mode:"plan"`** (default, read-only): scope, objective, generated cases, required fixtures, and an ordered plan.
568
597
  - **`mode:"execute"`**: a job that explores toward the feature, records pass, fail, or blocked per case, updates the app map, and writes a report (see the `qa_job_status` result). `interactive` runs until the first question.
569
- - **Without `sessionId`**, `execute` bootstraps a device from `projectRoot` (optionally `platform` and `device`) with one consent for boot, install, and launch.
598
+ - **Without `sessionId`**, `execute` bootstraps a device from `projectRoot` (optionally `platform` and `device`) with one consent for boot, install, and launch. The boot, install, and launch run inside the job, so the call returns at once; a preparation failure fails the job with the same blocker the call used to return.
570
599
  - A feature behind auth, a paywall, a permission, or a missing fixture is **blocked** with setup guidance, not failed.
571
600
  - Other parameters: `creativity` (`conservative`, `standard`, `creative`, `adversarial`) with `allowAdversarial`, `maxScreens` (default 8), `maxActions` (default 20), `timeoutMs`, `generateCases` (default true), `includeCode`, `limit`.
572
601
 
@@ -599,7 +628,7 @@ Compiles an existing POM suite on disk (`suite`, relative to `.swipium/`, defaul
599
628
 
600
629
  ### qa_flow_repair
601
630
 
602
- Given a failed step (`failedStep`, zero-based, from `qa_flow_run`) and the current screen, suggests a stronger locator plus app code changes (such as adding `accessibilityIdentifier` or `testID`). An exact id, label, or text match on the current screen is high confidence. Otherwise candidates are restricted to the failed target's role (a tap stays on a button, an `inputText` stays on a text field) and ranked by text similarity (medium for a contained match, low for a similar one), so a button renamed from "Sign in" to "Log in" is never repaired to the "Email" field. `apply:true` patches simple YAML selector steps in a flow file only at high or medium confidence, and records the patch in the mutation ledger; at low confidence it returns the proposal with `applied:false` and a note, and it never patches inline `flowYaml`. The flow must resolve inside the project root, otherwise `UNSAFE_ACTION_REFUSED`.
631
+ Given a flow (`flow`, a name or path under `.swipium/flows`, or inline `flowYaml`), a failed step (`failedStep`, zero-based, from `qa_flow_run`'s `failedAtStep`), and the current screen, suggests a stronger locator plus app code changes (such as adding `accessibilityIdentifier` or `testID`). An exact id, label, or text match on the current screen is high confidence. Otherwise candidates are restricted to the failed target's role (a tap stays on a button, an `inputText` stays on a text field) and ranked by text similarity (medium for a contained match, low for a similar one), so a button renamed from "Sign in" to "Log in" is never repaired to the "Email" field. `apply:true` patches simple YAML selector steps in a flow file only at high or medium confidence, and records the patch in the mutation ledger; at low confidence it returns the proposal with `applied:false` and a note, and it never patches inline `flowYaml`. The flow must resolve inside the project root, otherwise `UNSAFE_ACTION_REFUSED`.
603
632
 
604
633
  ## Generate
605
634
 
@@ -679,9 +708,9 @@ Plans or executes a named release audit. `profile` (required):
679
708
  | `resilience` | Offline, relaunch, and rotation. |
680
709
  | `release_gate` | All of the above, plus locator readiness and issue recurrence. |
681
710
 
682
- `mode:"plan"` (default) returns the checklist and safety contract without a device. `mode:"execute"` needs a prepared session, runs every check, logs failed and blocked checks to the issue ledger with evidence, and returns the release impact. It always runs to completion. No check passes without evidence. `allowTestAccountDeletion` permits deleting disposable test accounts (never a real account). `offlineMode` hints that resilience checks should drive offline state; `targetApp` and `sourceRevision` (`{commit, buildVersion, branch}`) identify what was audited.
711
+ `mode:"plan"` (default) returns the checklist and safety contract without a device. `mode:"execute"` needs a prepared session, runs every check, logs failed and blocked checks to the issue ledger with evidence, and returns the release impact. It runs inside the call (no job). No check passes without evidence. Cancelling the call skips the remaining checks and returns `CANCELLED` (`changedState:true`): an interrupted check is not evidence, so nothing is logged to the issue ledger for it. `allowTestAccountDeletion` permits deleting disposable test accounts (never a real account). `offlineMode` hints that resilience checks should drive offline state; `targetApp` and `sourceRevision` (`{commit, buildVersion, branch}`) identify what was audited.
683
712
 
684
- Executing `resilience` or `release_gate` (which runs all four other profiles) needs the `network_change` consent (medium), the same gate as `qa_network`, because it toggles airplane mode. The tool returns the consent request before running anything. With consent, the original airplane state is recorded first and restored afterwards, even if a check fails.
713
+ Executing `resilience` or `release_gate` (which runs all four other profiles) needs the `network_change` consent (medium), the same gate as `qa_network`, because it toggles airplane mode. The tool returns the consent request before running anything. With consent, the original airplane state is recorded first and restored afterwards, even if a check fails or the call is cancelled.
685
714
 
686
715
  ## First run
687
716
 
@@ -732,7 +761,7 @@ These sections used to live on this page:
732
761
 
733
762
  Every code a tool can return is in the catalog, and `qa_explain_blocker` explains any of them. Each code has a **bucket** (how to triage it: `app_bug`, `environment`, `missing_data`, `mcp_limitation`, or `unsafe_refused`), an **owner** (who fixes it: `app`, `environment`, `swipium`, or `user`), a severity, and a default retry safety. The tables below are grouped by bucket; owner is per code.
734
763
 
735
- Codes marked **reserved** are defined for classifying evidence, reports, and policy rules (so `blockOn`, `warnOn`, `ignoreKnown`, and report consumers can name them stably), but no tool returns them in 2.0.0. A reserved code may start being returned in a minor release. Codes marked **finding** appear as health findings in reports rather than as tool errors.
764
+ Codes marked **reserved** are defined for classifying evidence, reports, and policy rules (so `blockOn`, `warnOn`, `ignoreKnown`, and report consumers can name them stably), but no tool returns them yet. A reserved code may start being returned in a minor release. Codes marked **finding** appear as health findings in reports rather than as tool errors.
736
765
 
737
766
  ### Bucket: app_bug
738
767
 
@@ -839,7 +868,7 @@ Expected guardrails, not bugs.
839
868
  | --- | --- | --- |
840
869
  | `INVALID_ARGUMENT` | user | A malformed, missing, or undeclared argument, or an unknown `sessionId` or `jobId`. Nothing ran. |
841
870
  | `CANCELLED` | user | The call or job was cancelled. Not a failure. |
842
- | `CONSENT_DECLINED`, `CONSENT_CANCELLED`, `CONSENT_REFUSED` | user | See [Consent](concepts.md#consent). |
871
+ | `CONSENT_DECLINED`, `CONSENT_CANCELLED`, `CONSENT_REFUSED` | user | Nothing ran. When the client's consent prompt was declined or cancelled, the error carries `action`, `answeredInMs`, `likelyAutomatic`, and, when the prompt itself failed, `elicitationFailure`. See [Consent](concepts.md#consent). |
843
872
  | `DESTRUCTIVE_REFUSED` | user | A destructive action without approval (including a remote WDA URL). |
844
873
  | `UNSAFE_ACTION_REFUSED` | user | An unsafe action or a path outside the project root. |
845
874
  | `BUNDLE_LOSS_REFUSED` | user | A wipe that would remove a debug build's JS bundle, without `acknowledgeBundleRisk`. |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "swipium",
3
- "version": "2.0.1",
3
+ "version": "2.2.0",
4
4
  "private": false,
5
5
  "description": "Swipium MCP server for simulator-based mobile QA workflows, evidence capture, app maps, and generated test suites.",
6
6
  "keywords": [
@@ -64,10 +64,11 @@
64
64
  "inspector": "npm run build && npx @modelcontextprotocol/inspector node dist/index.js"
65
65
  },
66
66
  "dependencies": {
67
- "@modelcontextprotocol/sdk": "^1.19.1",
67
+ "@modelcontextprotocol/client": "^2.3.1",
68
+ "@modelcontextprotocol/server": "^2.3.1",
68
69
  "fast-xml-parser": "^5.8.0",
69
70
  "yaml": "^2.9.0",
70
- "zod": "^3.25.76"
71
+ "zod": "^4.2.0"
71
72
  },
72
73
  "devDependencies": {
73
74
  "@eslint/js": "^10.0.1",