swipium 2.0.1 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +64 -0
- package/README.md +32 -20
- package/THREAT_MODEL.md +109 -21
- package/dist/automationGen/run.js.map +1 -1
- package/dist/cli/init.js +101 -11
- package/dist/cli/init.js.map +1 -1
- package/dist/cli/verify.js +2 -2
- package/dist/cli/verify.js.map +1 -1
- package/dist/consent/consent.js +365 -17
- package/dist/consent/consent.js.map +1 -1
- package/dist/context/projectRoot.js +9 -4
- package/dist/context/projectRoot.js.map +1 -1
- package/dist/context/protocolEra.js +15 -0
- package/dist/context/protocolEra.js.map +1 -0
- package/dist/drivers/DirectDriver.js +5 -2
- package/dist/drivers/DirectDriver.js.map +1 -1
- package/dist/featureTesting/executionBootstrap.js +85 -77
- package/dist/featureTesting/executionBootstrap.js.map +1 -1
- package/dist/flows/run.js +9 -5
- package/dist/flows/run.js.map +1 -1
- package/dist/lib/abortScope.js +36 -3
- package/dist/lib/abortScope.js.map +1 -1
- package/dist/lib/android.js +15 -6
- package/dist/lib/android.js.map +1 -1
- package/dist/lib/codexEnv.js +112 -0
- package/dist/lib/codexEnv.js.map +1 -0
- package/dist/lib/logger.js +18 -0
- package/dist/lib/logger.js.map +1 -1
- package/dist/lib/result.js +188 -10
- package/dist/lib/result.js.map +1 -1
- package/dist/lib/schemaHash.js +20 -31
- package/dist/lib/schemaHash.js.map +1 -1
- package/dist/lib/simctl.js +102 -3
- package/dist/lib/simctl.js.map +1 -1
- package/dist/lib/toolSchema.js +143 -0
- package/dist/lib/toolSchema.js.map +1 -0
- package/dist/lib/wda.js +5 -2
- package/dist/lib/wda.js.map +1 -1
- package/dist/mobileAudit/runner.js +12 -4
- package/dist/mobileAudit/runner.js.map +1 -1
- package/dist/oracle/failures.js +2 -2
- package/dist/oracle/failures.js.map +1 -1
- package/dist/orchestration/testThis/execute.js +21 -4
- package/dist/orchestration/testThis/execute.js.map +1 -1
- package/dist/orchestration/testThis/pipeline.js +2 -0
- package/dist/orchestration/testThis/pipeline.js.map +1 -1
- package/dist/orchestration/testThis/plan.js.map +1 -1
- package/dist/server.js +564 -139
- package/dist/server.js.map +1 -1
- package/dist/services/prepareAndroid.js +3 -2
- package/dist/services/prepareAndroid.js.map +1 -1
- package/dist/services/prepareIos.js +31 -4
- package/dist/services/prepareIos.js.map +1 -1
- package/dist/services/smoke.js +38 -4
- package/dist/services/smoke.js.map +1 -1
- package/dist/session/processRegistry.js +12 -1
- package/dist/session/processRegistry.js.map +1 -1
- package/dist/snapshot/parse.js +64 -9
- package/dist/snapshot/parse.js.map +1 -1
- package/dist/snapshot/present.js +14 -4
- package/dist/snapshot/present.js.map +1 -1
- package/dist/snapshot/settle.js +14 -3
- package/dist/snapshot/settle.js.map +1 -1
- package/dist/tools/act.js +694 -670
- package/dist/tools/act.js.map +1 -1
- package/dist/tools/agent.js +15 -16
- package/dist/tools/agent.js.map +1 -1
- package/dist/tools/appControl.js +4 -4
- package/dist/tools/appControl.js.map +1 -1
- package/dist/tools/appMap.js +11 -13
- package/dist/tools/appMap.js.map +1 -1
- package/dist/tools/build.js +4 -6
- package/dist/tools/build.js.map +1 -1
- package/dist/tools/bundletool.js +3 -3
- package/dist/tools/bundletool.js.map +1 -1
- package/dist/tools/clearOverlay.js +2 -3
- package/dist/tools/clearOverlay.js.map +1 -1
- package/dist/tools/device.js +4 -3
- package/dist/tools/device.js.map +1 -1
- package/dist/tools/doctor.js +23 -11
- package/dist/tools/doctor.js.map +1 -1
- package/dist/tools/explore.js +15 -21
- package/dist/tools/explore.js.map +1 -1
- package/dist/tools/featureTesting.js +22 -6
- package/dist/tools/featureTesting.js.map +1 -1
- package/dist/tools/firstRun.js +4 -5
- package/dist/tools/firstRun.js.map +1 -1
- package/dist/tools/flow.js +13 -17
- package/dist/tools/flow.js.map +1 -1
- package/dist/tools/flowRepair.js +3 -5
- package/dist/tools/flowRepair.js.map +1 -1
- package/dist/tools/generate.js +10 -11
- package/dist/tools/generate.js.map +1 -1
- package/dist/tools/getArtifact.js +114 -10
- package/dist/tools/getArtifact.js.map +1 -1
- package/dist/tools/health.js +2 -1
- package/dist/tools/health.js.map +1 -1
- package/dist/tools/ios.js +35 -13
- package/dist/tools/ios.js.map +1 -1
- package/dist/tools/issues.js +8 -8
- package/dist/tools/issues.js.map +1 -1
- package/dist/tools/jobs.js +26 -10
- package/dist/tools/jobs.js.map +1 -1
- package/dist/tools/metro.js +4 -5
- package/dist/tools/metro.js.map +1 -1
- package/dist/tools/mobileAudit.js +11 -10
- package/dist/tools/mobileAudit.js.map +1 -1
- package/dist/tools/note.js +2 -3
- package/dist/tools/note.js.map +1 -1
- package/dist/tools/prepareIosTarget.js +102 -31
- package/dist/tools/prepareIosTarget.js.map +1 -1
- package/dist/tools/prepareTarget.js +3 -5
- package/dist/tools/prepareTarget.js.map +1 -1
- package/dist/tools/report.js +4 -5
- package/dist/tools/report.js.map +1 -1
- package/dist/tools/resolveArtifact.js +2 -3
- package/dist/tools/resolveArtifact.js.map +1 -1
- package/dist/tools/resolveTarget.js +3 -6
- package/dist/tools/resolveTarget.js.map +1 -1
- package/dist/tools/screenRecord.js +2 -3
- package/dist/tools/screenRecord.js.map +1 -1
- package/dist/tools/screenshot.js +2 -2
- package/dist/tools/screenshot.js.map +1 -1
- package/dist/tools/smoke.js +4 -2
- package/dist/tools/smoke.js.map +1 -1
- package/dist/tools/snapshot.js +6 -7
- package/dist/tools/snapshot.js.map +1 -1
- package/dist/tools/startSession.js +12 -16
- package/dist/tools/startSession.js.map +1 -1
- package/dist/tools/suite.js +2 -4
- package/dist/tools/suite.js.map +1 -1
- package/dist/tools/testSuite.js +13 -19
- package/dist/tools/testSuite.js.map +1 -1
- package/dist/tools/testThis.js +28 -13
- package/dist/tools/testThis.js.map +1 -1
- package/dist/tools/visual.js +7 -9
- package/dist/tools/visual.js.map +1 -1
- package/dist/tools/wait.js +130 -21
- package/dist/tools/wait.js.map +1 -1
- package/dist/tools/wda.js +199 -60
- package/dist/tools/wda.js.map +1 -1
- package/dist/version.js +1 -1
- package/docs/README.md +4 -4
- package/docs/ci-reports.md +39 -7
- package/docs/concepts.md +37 -25
- package/docs/flows.md +1 -1
- package/docs/mcp-server.md +281 -76
- package/docs/physical-devices.md +4 -4
- package/docs/tools.md +59 -30
- package/package.json +4 -3
package/dist/server.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { statSync } from 'node:fs';
|
|
2
3
|
import { basename } from 'node:path';
|
|
3
4
|
import { fileURLToPath } from 'node:url';
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
5
|
+
import { StdioServerTransport, serveStdio } from '@modelcontextprotocol/server/stdio';
|
|
6
|
+
import { CLIENT_CAPABILITIES_META_KEY, McpServer, ProtocolError, ProtocolErrorCode, ResourceNotFoundError, ResourceTemplate, inputRequired, inputResponse, isInputRequiredResult, } from '@modelcontextprotocol/server';
|
|
6
7
|
import { SessionStore, decodeUriSegment, encodeUriSegment, isWithinRoot } from './session/store.js';
|
|
7
|
-
import { burnConsent, peekConsent, requestConsentDecision, runWithConsentScope, setElicitationProvider, } from './consent/consent.js';
|
|
8
|
+
import { AUTO_ANSWER_MS, burnConsent, issueConsentPrompt, noPromptDecision, operatorPolicyDecision, peekConsent, preapproveHint, redeemConsentPrompt, requestConsentDecision, runWithConsentScope, setElicitationProvider, settleConsentAnswer, warnIgnoredPreapprovals, } from './consent/consent.js';
|
|
8
9
|
import { registerDoctor } from './tools/doctor.js';
|
|
9
10
|
import { registerStartSession } from './tools/startSession.js';
|
|
10
11
|
import { registerPrepareTarget } from './tools/prepareTarget.js';
|
|
@@ -21,7 +22,7 @@ import { registerIos } from './tools/ios.js';
|
|
|
21
22
|
import { registerWda } from './tools/wda.js';
|
|
22
23
|
import { registerClearOverlay } from './tools/clearOverlay.js';
|
|
23
24
|
import { registerJobs } from './tools/jobs.js';
|
|
24
|
-
import { registerGetArtifact } from './tools/getArtifact.js';
|
|
25
|
+
import { registerGetArtifact, readArtifactResource } from './tools/getArtifact.js';
|
|
25
26
|
import { registerNote } from './tools/note.js';
|
|
26
27
|
import { registerVisual } from './tools/visual.js';
|
|
27
28
|
import { registerFlow } from './tools/flow.js';
|
|
@@ -47,15 +48,17 @@ import { registerResolveTarget } from './tools/resolveTarget.js';
|
|
|
47
48
|
import { registerBuild } from './tools/build.js';
|
|
48
49
|
import { registerBundletool } from './tools/bundletool.js';
|
|
49
50
|
import { registerPrompts } from './prompts/index.js';
|
|
50
|
-
import { log } from './lib/logger.js';
|
|
51
|
-
import { qaError, runWithResponseMode } from './lib/result.js';
|
|
51
|
+
import { log, logEnabled } from './lib/logger.js';
|
|
52
|
+
import { qaAnnotate, qaError, runWithResponseMode } from './lib/result.js';
|
|
52
53
|
import { recordToolErrorFromResult } from './report/toolHealth.js';
|
|
53
54
|
import { computeSchemaHash, describeZodField, setSchemaHash } from './lib/schemaHash.js';
|
|
54
55
|
import { REMOVED_TOOLS, STALE_CLIENT_HINT, SWIPIUM_VERSION, TOOL_COUNT, TOOL_NAMES, TOOL_NAME_SET } from './version.js';
|
|
55
56
|
import { toolAnnotations } from './lib/toolAnnotations.js';
|
|
57
|
+
import { MAX_ECHOED_KEYS, checkToolArgs, echoKey, strictToolSchema, toolInputJsonSchema, } from './lib/toolSchema.js';
|
|
56
58
|
import { CAPABILITY_GROUPS } from './core/capabilityGroups.js';
|
|
57
59
|
import { ensureAndroidToolsOnPath } from './lib/android.js';
|
|
58
60
|
import { annotateRootSource, withRootResolutionRecording } from './context/projectRoot.js';
|
|
61
|
+
import { markModernServer, servesModernEra } from './context/protocolEra.js';
|
|
59
62
|
import { reapOrphanedProcesses } from './session/processRegistry.js';
|
|
60
63
|
import { runWithSignal } from './lib/abortScope.js';
|
|
61
64
|
/** How long a consent prompt may stay open before it counts as unanswered (> refusal).
|
|
@@ -100,6 +103,15 @@ export function buildConsentPromptMessage(req) {
|
|
|
100
103
|
const msg = lines.join('\n');
|
|
101
104
|
return msg.length > 2000 ? `${msg.slice(0, 1999)}…` : msg;
|
|
102
105
|
}
|
|
106
|
+
/** The consent form: ONE flat boolean field (MCP elicitation allows flat primitive schemas only).
|
|
107
|
+
* Shared by both eras: `elicitation/create` (2025) and the InputRequiredResult form (2026-07-28). */
|
|
108
|
+
const CONSENT_FORM_SCHEMA = {
|
|
109
|
+
type: 'object',
|
|
110
|
+
properties: {
|
|
111
|
+
approve: { type: 'boolean', title: 'Approve', description: 'Allow Swipium to perform this action.' },
|
|
112
|
+
},
|
|
113
|
+
required: ['approve'],
|
|
114
|
+
};
|
|
103
115
|
function makeElicitationProvider(server) {
|
|
104
116
|
return async (req, ctx) => {
|
|
105
117
|
// The SDK normalises a bare `elicitation: {}` capability to `{ form: {} }`.
|
|
@@ -107,13 +119,7 @@ function makeElicitationProvider(server) {
|
|
|
107
119
|
return 'unavailable';
|
|
108
120
|
const answer = await server.server.elicitInput({
|
|
109
121
|
message: buildConsentPromptMessage(req),
|
|
110
|
-
requestedSchema:
|
|
111
|
-
type: 'object',
|
|
112
|
-
properties: {
|
|
113
|
-
approve: { type: 'boolean', title: 'Approve', description: 'Allow Swipium to perform this action.' },
|
|
114
|
-
},
|
|
115
|
-
required: ['approve'],
|
|
116
|
-
},
|
|
122
|
+
requestedSchema: CONSENT_FORM_SCHEMA,
|
|
117
123
|
}, {
|
|
118
124
|
timeout: ELICITATION_TIMEOUT_MS,
|
|
119
125
|
...(ctx?.signal ? { signal: ctx.signal } : {}),
|
|
@@ -153,6 +159,61 @@ function pendingConsentId(result) {
|
|
|
153
159
|
const sc = result?.structuredContent;
|
|
154
160
|
return sc?.requiresConsent === true && typeof sc.consentId === 'string' ? sc.consentId : undefined;
|
|
155
161
|
}
|
|
162
|
+
/** The InputRequiredResult key of the consent prompt (protocol 2026-07-28). */
|
|
163
|
+
export const CONSENT_INPUT_KEY = 'swipium_consent';
|
|
164
|
+
/** sha256 over the canonical JSON (sorted keys) of a tool call's validated arguments: binds a
|
|
165
|
+
* 2026 consent prompt to the exact call it was issued for. Exported for tests. */
|
|
166
|
+
export function argsDigest(args) {
|
|
167
|
+
const canon = (v) => {
|
|
168
|
+
if (Array.isArray(v))
|
|
169
|
+
return v.map(canon);
|
|
170
|
+
if (v && typeof v === 'object') {
|
|
171
|
+
const out = {};
|
|
172
|
+
for (const k of Object.keys(v).sort())
|
|
173
|
+
out[k] = canon(v[k]);
|
|
174
|
+
return out;
|
|
175
|
+
}
|
|
176
|
+
return v;
|
|
177
|
+
};
|
|
178
|
+
return createHash('sha256')
|
|
179
|
+
.update(JSON.stringify(canon(args ?? {})) ?? '')
|
|
180
|
+
.digest('hex');
|
|
181
|
+
}
|
|
182
|
+
/** Does this protocol 2026-07-28 request declare form elicitation in its per-request
|
|
183
|
+
* `_meta` client capabilities? A bare `elicitation: {}` means form (spec + SDK reading); a client
|
|
184
|
+
* that declares only `url` cannot show our form. */
|
|
185
|
+
export function requestSupportsFormElicitation(ctx) {
|
|
186
|
+
const caps = ctx?.mcpReq?.envelope?.[CLIENT_CAPABILITIES_META_KEY];
|
|
187
|
+
const el = caps?.elicitation;
|
|
188
|
+
if (!el || typeof el !== 'object' || Array.isArray(el))
|
|
189
|
+
return false;
|
|
190
|
+
const modes = el;
|
|
191
|
+
return modes.form !== undefined || modes.url === undefined;
|
|
192
|
+
}
|
|
193
|
+
/** requestState errors leave the tool wrapper as JSON-RPC errors, never as tool results. */
|
|
194
|
+
const requestStateErrors = new WeakSet();
|
|
195
|
+
/** The JSON-RPC rejection of a requestState that does not redeem: the SDK's own frozen shape
|
|
196
|
+
* (-32602, "Invalid or expired requestState", data.reason invalid_request_state). The detailed
|
|
197
|
+
* reason goes to stderr only. */
|
|
198
|
+
function invalidRequestState(tool, why) {
|
|
199
|
+
log('warn', 'rejected a tools/call requestState', { tool, reason: why });
|
|
200
|
+
const err = new ProtocolError(ProtocolErrorCode.InvalidParams, 'Invalid or expired requestState', { reason: 'invalid_request_state' });
|
|
201
|
+
requestStateErrors.add(err);
|
|
202
|
+
return err;
|
|
203
|
+
}
|
|
204
|
+
/** The InputRequiredResult that asks the client's user to decide `consentId` (2026-07-28). */
|
|
205
|
+
function consentInputRequired(req, token) {
|
|
206
|
+
return inputRequired({
|
|
207
|
+
inputRequests: {
|
|
208
|
+
[CONSENT_INPUT_KEY]: inputRequired.elicit({
|
|
209
|
+
mode: 'form',
|
|
210
|
+
message: buildConsentPromptMessage(req),
|
|
211
|
+
requestedSchema: CONSENT_FORM_SCHEMA,
|
|
212
|
+
}),
|
|
213
|
+
},
|
|
214
|
+
requestState: token,
|
|
215
|
+
});
|
|
216
|
+
}
|
|
156
217
|
/**
|
|
157
218
|
* Consent routing (consent.ts header): when a tool handler returns a requiresConsent envelope,
|
|
158
219
|
* route the pending consent through a REAL out-of-band user prompt before the envelope ever
|
|
@@ -162,20 +223,119 @@ function pendingConsentId(result) {
|
|
|
162
223
|
* - elicitation declined > CONSENT_DECLINED; cancelled/timed out/transport error >
|
|
163
224
|
* CONSENT_CANCELLED (retry-safe: a re-call issues a fresh prompt). Either way the challenge
|
|
164
225
|
* is burned (no approve:true self-approval afterwards) and a `refused` ledger row is written;
|
|
226
|
+
* - operator-policy (action listed in SWIPIUM_CONSENT_PREAPPROVE and allowed by its tier, see
|
|
227
|
+
* consent.ts operatorPolicyCovers) > re-invoke exactly like an elicitation approval, without
|
|
228
|
+
* prompting (consumeConsent tags it 'operator-policy');
|
|
165
229
|
* - refused (SWIPIUM_REQUIRE_ELICITATION=1 and no elicitation support) > CONSENT_REFUSED;
|
|
166
230
|
* - client-assertion (client does not advertise elicitation) > the portable envelope unchanged.
|
|
167
231
|
* The re-invocation's result is never routed again, so this cannot loop; if it asks for a NEW
|
|
168
232
|
* consent (the target changed under the prompt) that challenge is burned and CONSENT_CANCELLED
|
|
169
233
|
* returned, so the model never receives a self-approvable envelope on an elicitation client.
|
|
234
|
+
*
|
|
235
|
+
* Protocol 2026-07-28 (routing.era 'modern'): there is no elicitation/create request to await.
|
|
236
|
+
* After the operator pre-approval check, a request whose `_meta` client capabilities declare form
|
|
237
|
+
* elicitation gets an InputRequiredResult carrying the same form and a single-use requestState
|
|
238
|
+
* handle (consent.ts issueConsentPrompt); the client's user answers and the client retries the
|
|
239
|
+
* call, which resumeConsentPrompt settles into the very same outcomes. A request without that
|
|
240
|
+
* capability gets the portable envelope (or CONSENT_REFUSED under SWIPIUM_REQUIRE_ELICITATION=1).
|
|
170
241
|
*/
|
|
171
|
-
async function routePendingConsent(result, args, reinvoke, ledger) {
|
|
242
|
+
async function routePendingConsent(result, args, reinvoke, ledger, routing) {
|
|
172
243
|
const consentId = pendingConsentId(result);
|
|
173
244
|
if (!consentId)
|
|
174
245
|
return result;
|
|
175
246
|
const first = (args[0] ?? {});
|
|
176
|
-
const
|
|
177
|
-
const
|
|
178
|
-
const
|
|
247
|
+
const ctx = args[1];
|
|
248
|
+
const mcpReq = ctx?.mcpReq;
|
|
249
|
+
const req = peekConsent(consentId); // read BEFORE deciding: a refusal burns the challenge
|
|
250
|
+
let decision;
|
|
251
|
+
if (routing.era === 'modern') {
|
|
252
|
+
if (!req)
|
|
253
|
+
return result;
|
|
254
|
+
const byPolicy = operatorPolicyDecision(consentId);
|
|
255
|
+
if (byPolicy)
|
|
256
|
+
decision = byPolicy;
|
|
257
|
+
else if (!requestSupportsFormElicitation(ctx))
|
|
258
|
+
decision = noPromptDecision(consentId);
|
|
259
|
+
else {
|
|
260
|
+
const sessionId = typeof first.sessionId === 'string' ? first.sessionId : undefined;
|
|
261
|
+
const token = issueConsentPrompt(consentId, {
|
|
262
|
+
tool: ledger.tool,
|
|
263
|
+
argsDigest: argsDigest(first),
|
|
264
|
+
...(sessionId ? { sessionId } : {}),
|
|
265
|
+
});
|
|
266
|
+
if (!token)
|
|
267
|
+
return result;
|
|
268
|
+
log('debug', 'consent prompt sent as InputRequiredResult', { tool: ledger.tool, action: req.action, consentId });
|
|
269
|
+
return consentInputRequired(req, token);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
else {
|
|
273
|
+
decision = await requestConsentDecision(consentId, {
|
|
274
|
+
signal: mcpReq?.signal,
|
|
275
|
+
relatedRequestId: mcpReq?.id,
|
|
276
|
+
provider: routing.provider,
|
|
277
|
+
});
|
|
278
|
+
}
|
|
279
|
+
if (decision.mechanism === 'client-assertion')
|
|
280
|
+
return result; // portable path (the model relays the consent envelope)
|
|
281
|
+
return applyConsentDecision(decision, consentId, req, first, args, reinvoke, ledger);
|
|
282
|
+
}
|
|
283
|
+
/**
|
|
284
|
+
* Protocol 2026-07-28: the client's retry of a tool call that answered our InputRequiredResult.
|
|
285
|
+
* The requestState must redeem a live, single-use handle (consent.ts redeemConsentPrompt) issued
|
|
286
|
+
* for THIS tool with THESE arguments (digest, sessionId included); anything else is rejected as
|
|
287
|
+
* -32602 and never runs the tool. The binding is server-side, so a client or model cannot point an
|
|
288
|
+
* answer at another consentId, session or action, and a replayed handle is already gone.
|
|
289
|
+
* - accept + approve:true > approved (re-invoke with the consent attached, ledgered 'elicitation');
|
|
290
|
+
* - accept + approve:false, or decline > CONSENT_DECLINED; cancel > CONSENT_CANCELLED;
|
|
291
|
+
* - answered after ELICITATION_TIMEOUT_MS > CONSENT_CANCELLED (the prompt expired);
|
|
292
|
+
* - no answer for our key > a fresh InputRequiredResult for the same consent (spec: ask again).
|
|
293
|
+
* likelyAutomatic is measured from the InputRequiredResult to the retry.
|
|
294
|
+
*/
|
|
295
|
+
async function resumeConsentPrompt(token, args, reinvoke, ledger) {
|
|
296
|
+
const first = (args[0] ?? {});
|
|
297
|
+
const ctx = args[1];
|
|
298
|
+
const handle = redeemConsentPrompt(token);
|
|
299
|
+
if (!handle)
|
|
300
|
+
throw invalidRequestState(ledger.tool, 'unknown, expired or already used');
|
|
301
|
+
const sessionId = typeof first.sessionId === 'string' ? first.sessionId : undefined;
|
|
302
|
+
if (handle.tool !== ledger.tool || handle.argsDigest !== argsDigest(first) || handle.sessionId !== sessionId) {
|
|
303
|
+
// Bound to another call: burn the consent too (only the holder of the handle can get here).
|
|
304
|
+
burnConsent(handle.consentId);
|
|
305
|
+
throw invalidRequestState(ledger.tool, 'issued for a different tool call');
|
|
306
|
+
}
|
|
307
|
+
const req = peekConsent(handle.consentId);
|
|
308
|
+
if (!req)
|
|
309
|
+
throw invalidRequestState(ledger.tool, 'the consent is no longer pending');
|
|
310
|
+
const elapsedMs = Date.now() - handle.issuedAt;
|
|
311
|
+
let decision;
|
|
312
|
+
if (elapsedMs > ELICITATION_TIMEOUT_MS) {
|
|
313
|
+
decision = settleConsentAnswer(handle.consentId, 'cancelled', elapsedMs, `the consent prompt expired after ${ELICITATION_TIMEOUT_MS} ms`);
|
|
314
|
+
}
|
|
315
|
+
else {
|
|
316
|
+
const answer = inputResponse(ctx?.mcpReq?.inputResponses, CONSENT_INPUT_KEY);
|
|
317
|
+
if (answer.kind === 'missing') {
|
|
318
|
+
// Retried without our answer (or with an unparseable one): ask again for the same consent.
|
|
319
|
+
const again = issueConsentPrompt(handle.consentId, handle);
|
|
320
|
+
if (!again)
|
|
321
|
+
throw invalidRequestState(ledger.tool, 'the consent is no longer pending');
|
|
322
|
+
return consentInputRequired(req, again);
|
|
323
|
+
}
|
|
324
|
+
if (answer.kind !== 'elicit') {
|
|
325
|
+
decision = settleConsentAnswer(handle.consentId, 'cancelled', elapsedMs, 'the client answered the consent prompt with a non-elicitation result');
|
|
326
|
+
}
|
|
327
|
+
else if (answer.action === 'accept') {
|
|
328
|
+
decision = settleConsentAnswer(handle.consentId, answer.content?.approve === true ? 'approved' : 'declined', elapsedMs);
|
|
329
|
+
}
|
|
330
|
+
else {
|
|
331
|
+
decision = settleConsentAnswer(handle.consentId, answer.action === 'decline' ? 'declined' : 'cancelled', elapsedMs);
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
return applyConsentDecision(decision, handle.consentId, req, first, args, reinvoke, ledger);
|
|
335
|
+
}
|
|
336
|
+
/** The outcome of a consent decision (shared by both eras): an envelope for a refusal, or the
|
|
337
|
+
* tool re-invoked once with the consent attached for an approval. */
|
|
338
|
+
async function applyConsentDecision(decision, consentId, req, first, args, reinvoke, ledger) {
|
|
179
339
|
if (decision.mechanism === 'refused') {
|
|
180
340
|
recordConsentRefusal(ledger.sessions, ledger.tool, first.sessionId, consentId, req, 'policy', decision.reason);
|
|
181
341
|
return qaError({
|
|
@@ -186,51 +346,143 @@ async function routePendingConsent(result, args, reinvoke, ledger) {
|
|
|
186
346
|
nextSteps: ['Connect with an MCP client that supports elicitation, or unset SWIPIUM_REQUIRE_ELICITATION.'],
|
|
187
347
|
});
|
|
188
348
|
}
|
|
189
|
-
if (decision.
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
}
|
|
349
|
+
if (!decision.approved) {
|
|
350
|
+
// A failed elicitation (timeout, transport error, aborted call) is not an answer: ledger it
|
|
351
|
+
// as 'transport/abort' and never flag it as likely automatic.
|
|
352
|
+
const failed = decision.failure !== undefined;
|
|
353
|
+
recordConsentRefusal(ledger.sessions, ledger.tool, first.sessionId, consentId, req, 'elicitation', failed ? `transport/abort: ${decision.reason}` : decision.reason);
|
|
354
|
+
// Headless clients (codex exec, claude -p) answer prompts automatically: a near-instant
|
|
355
|
+
// answer is flagged as likely automatic and only then carries the pre-approve hint (a real
|
|
356
|
+
// human decline must not be nudged toward disabling consent).
|
|
357
|
+
const action = req?.action ?? 'unknown';
|
|
358
|
+
const likelyAutomatic = !failed && decision.elapsedMs < AUTO_ANSWER_MS;
|
|
359
|
+
const extra = failed
|
|
360
|
+
? { action, answeredInMs: decision.elapsedMs, likelyAutomatic: false, elicitationFailure: decision.failure }
|
|
361
|
+
: { action, answeredInMs: decision.elapsedMs, likelyAutomatic };
|
|
362
|
+
const hint = likelyAutomatic ? [preapproveHint(action, req)] : [];
|
|
363
|
+
if (decision.outcome === 'cancelled') {
|
|
201
364
|
return qaError({
|
|
202
|
-
what:
|
|
203
|
-
changedState: false,
|
|
204
|
-
retrySafe: false,
|
|
205
|
-
failureCode: 'CONSENT_DECLINED',
|
|
206
|
-
nextSteps: ['Do not retry this action; ask the user before attempting it again.'],
|
|
207
|
-
});
|
|
208
|
-
}
|
|
209
|
-
const out = await reinvoke([{ ...first, consentId, approve: true }, ...args.slice(1)]);
|
|
210
|
-
const again = pendingConsentId(out);
|
|
211
|
-
if (again) {
|
|
212
|
-
burnConsent(again);
|
|
213
|
-
burnConsent(consentId);
|
|
214
|
-
return qaError({
|
|
215
|
-
what: 'The action changed while the user was deciding, so the approval no longer matches it. Nothing ran.',
|
|
365
|
+
what: decision.reason,
|
|
216
366
|
changedState: false,
|
|
217
367
|
retrySafe: true,
|
|
218
368
|
failureCode: 'CONSENT_CANCELLED',
|
|
219
|
-
nextSteps: [
|
|
220
|
-
|
|
369
|
+
nextSteps: [
|
|
370
|
+
failed
|
|
371
|
+
? 'Nothing ran. The consent prompt failed (timeout, transport error or cancelled call) before anyone answered: re-call the tool (without consentId) to prompt again, or ask the user first.'
|
|
372
|
+
: likelyAutomatic
|
|
373
|
+
? 'Nothing ran. The prompt was answered too fast for a human: do not re-call in a loop, ask the user first.'
|
|
374
|
+
: 'Nothing ran. Re-call the tool (without consentId) to show the user a fresh consent prompt, or ask them first.',
|
|
375
|
+
...hint,
|
|
376
|
+
],
|
|
377
|
+
}, extra);
|
|
221
378
|
}
|
|
222
|
-
return
|
|
379
|
+
return qaError({
|
|
380
|
+
what: likelyAutomatic
|
|
381
|
+
? `The client declined "${action}" in ${decision.elapsedMs} ms, likely automatically without showing the user`
|
|
382
|
+
: 'User declined via elicitation prompt',
|
|
383
|
+
changedState: false,
|
|
384
|
+
retrySafe: false,
|
|
385
|
+
failureCode: 'CONSENT_DECLINED',
|
|
386
|
+
nextSteps: ['Do not retry this action; ask the user before attempting it again.', ...hint],
|
|
387
|
+
}, extra);
|
|
388
|
+
}
|
|
389
|
+
const out = await reinvoke([{ ...first, consentId, approve: true }, ...args.slice(1)]);
|
|
390
|
+
const again = pendingConsentId(out);
|
|
391
|
+
if (again) {
|
|
392
|
+
burnConsent(again);
|
|
393
|
+
burnConsent(consentId);
|
|
394
|
+
return qaError({
|
|
395
|
+
what: 'The action changed while the user was deciding, so the approval no longer matches it. Nothing ran.',
|
|
396
|
+
changedState: false,
|
|
397
|
+
retrySafe: true,
|
|
398
|
+
failureCode: 'CONSENT_CANCELLED',
|
|
399
|
+
nextSteps: ['Re-call the tool (without consentId) to prompt the user for the current action.'],
|
|
400
|
+
});
|
|
223
401
|
}
|
|
224
|
-
return
|
|
402
|
+
return out;
|
|
225
403
|
}
|
|
226
404
|
/**
|
|
227
405
|
* Wrap every tool handler so it runs inside the calling session's response mode.
|
|
228
406
|
* Resolved once, centrally. Individual tools stay mode-agnostic.
|
|
229
|
-
* `compact` shrinks the text channel; `structuredContent`
|
|
407
|
+
* `compact` shrinks the text channel; `structuredContent` always carries every field (element
|
|
408
|
+
* lists are one-line @eN strings outside `verbose`, see snapshot/present.ts).
|
|
230
409
|
* The same wrapper also routes requiresConsent envelopes through out-of-band elicitation
|
|
231
410
|
* (routePendingConsent), so every consent-gated tool inherits it with zero per-tool changes.
|
|
232
411
|
*/
|
|
233
|
-
|
|
412
|
+
// Job kinds that never touch the device (host-side build / conversion); everything else a job
|
|
413
|
+
// runs (test_this, explore, test_feature, boot+install) drives the session's device.
|
|
414
|
+
const HOST_ONLY_JOB_PREFIXES = ['build:', 'bundletool:'];
|
|
415
|
+
/** Tools that DRIVE the device or the app on it (tap/type/swipe, install/launch/stop, boot/erase,
|
|
416
|
+
* orientation/location/network, recorder, WDA/Metro wiring, or start a job that does). Explicit
|
|
417
|
+
* allowlist: the readOnlyHint test also warned for tools that never touch the device (qa_note,
|
|
418
|
+
* qa_generate, qa_suite_*, qa_app_map_*, qa_issue_log, qa_build...). Observation-only device
|
|
419
|
+
* tools (qa_snapshot, qa_screenshot, qa_device_info, qa_check_health, qa_flow_repair) are left
|
|
420
|
+
* out: they do not change what the job sees. Exported for tests. */
|
|
421
|
+
export const DEVICE_DRIVING_TOOLS = new Set([
|
|
422
|
+
'qa_test_this',
|
|
423
|
+
'qa_continue_from_blocker',
|
|
424
|
+
'qa_prepare_target',
|
|
425
|
+
'qa_prepare_ios_target',
|
|
426
|
+
'qa_ios',
|
|
427
|
+
'qa_wda',
|
|
428
|
+
'qa_orientation',
|
|
429
|
+
'qa_geolocation',
|
|
430
|
+
'qa_network',
|
|
431
|
+
'qa_metro',
|
|
432
|
+
'qa_app_control',
|
|
433
|
+
'qa_screen_record',
|
|
434
|
+
'qa_act',
|
|
435
|
+
'qa_clear_overlay',
|
|
436
|
+
'qa_visual',
|
|
437
|
+
'qa_smoke',
|
|
438
|
+
'qa_explore',
|
|
439
|
+
'qa_test_feature',
|
|
440
|
+
'qa_flow_run',
|
|
441
|
+
'qa_first_run',
|
|
442
|
+
'qa_mobile_audit',
|
|
443
|
+
]);
|
|
444
|
+
/** Advisory note (never a block) when a device-driving tool is called on a session whose
|
|
445
|
+
* background job is still driving the same device: the two can interleave taps/installs.
|
|
446
|
+
* Exported for tests. */
|
|
447
|
+
export function runningJobNote(sessions, tool, sessionId) {
|
|
448
|
+
if (!sessionId || !DEVICE_DRIVING_TOOLS.has(tool))
|
|
449
|
+
return undefined;
|
|
450
|
+
const s = sessions.get(sessionId);
|
|
451
|
+
if (!s)
|
|
452
|
+
return undefined;
|
|
453
|
+
for (const j of s.jobs.values()) {
|
|
454
|
+
if (j.status !== 'running' || HOST_ONLY_JOB_PREFIXES.some((p) => j.kind.startsWith(p)))
|
|
455
|
+
continue;
|
|
456
|
+
return `job ${j.jobId} is still driving this device; actions may interleave. Poll qa_job_status or qa_job_cancel first.`;
|
|
457
|
+
}
|
|
458
|
+
return undefined;
|
|
459
|
+
}
|
|
460
|
+
/** qaAnnotate, but keeps any notes the tool already put in structuredContent.notes. */
|
|
461
|
+
function annotateKeepingNotes(result, note) {
|
|
462
|
+
const prior = result.structuredContent?.notes;
|
|
463
|
+
const out = qaAnnotate(result, [note]);
|
|
464
|
+
if (Array.isArray(prior))
|
|
465
|
+
out.structuredContent = { ...out.structuredContent, notes: [...prior, note] };
|
|
466
|
+
return out;
|
|
467
|
+
}
|
|
468
|
+
/** One debug line per tool call (SWIPIUM_LOG_LEVEL=debug). Metadata only: argument values can
|
|
469
|
+
* carry secrets and are never logged. */
|
|
470
|
+
function logToolCall(tool, sessionId, startedAt, out, signal) {
|
|
471
|
+
if (!logEnabled('debug'))
|
|
472
|
+
return;
|
|
473
|
+
const sc = out?.structuredContent;
|
|
474
|
+
const failureCode = typeof sc?.failureCode === 'string' ? sc.failureCode : undefined;
|
|
475
|
+
log('debug', 'tool call', {
|
|
476
|
+
tool,
|
|
477
|
+
...(sessionId ? { sessionId } : {}),
|
|
478
|
+
durationMs: Date.now() - startedAt,
|
|
479
|
+
isError: out ? out.isError === true : true,
|
|
480
|
+
...(out ? {} : { threw: true }),
|
|
481
|
+
...(failureCode ? { failureCode } : {}),
|
|
482
|
+
cancelled: signal?.aborted === true || failureCode === 'CANCELLED',
|
|
483
|
+
});
|
|
484
|
+
}
|
|
485
|
+
function installResponseModeWrapper(server, sessions, surface, attempted, tools, routing) {
|
|
234
486
|
const orig = server.registerTool.bind(server);
|
|
235
487
|
const valid = (m) => m === 'compact' || m === 'normal' || m === 'verbose';
|
|
236
488
|
server.registerTool = (name, config, handler) => {
|
|
@@ -242,10 +494,10 @@ function installResponseModeWrapper(server, sessions, surface, attempted, paramN
|
|
|
242
494
|
const cfg = config;
|
|
243
495
|
const inputKeys = Object.entries(cfg?.inputSchema ?? {}).map(([k, v]) => `${k}:${describeZodField(v)}`);
|
|
244
496
|
surface.push({ name, description: cfg?.description ?? '', inputKeys });
|
|
245
|
-
paramNames.set(name, Object.keys(cfg?.inputSchema ?? {}));
|
|
246
497
|
// MCP annotations for every tool, from one reviewed table (src/lib/toolAnnotations.ts).
|
|
247
|
-
const
|
|
248
|
-
|
|
498
|
+
const annotations = toolAnnotations(name);
|
|
499
|
+
const annotated = { ...config, annotations };
|
|
500
|
+
const wrapped = async (...a) => {
|
|
249
501
|
const first = a[0];
|
|
250
502
|
// Prefer the existing session's mode; fall back to a directly-passed responseMode so the
|
|
251
503
|
// session-CREATING call (qa_start_session, no sessionId yet) also honors compact.
|
|
@@ -254,20 +506,63 @@ function installResponseModeWrapper(server, sessions, surface, attempted, paramN
|
|
|
254
506
|
// Every call records the project root it resolves (if any) so the result carries
|
|
255
507
|
// `rootSource` (+ a note when the root was only guessed from the server cwd).
|
|
256
508
|
// Consents minted/consumed during this call are bound to its sessionId (consent.ts).
|
|
257
|
-
// Cancellation: the call's MCP signal (
|
|
258
|
-
// to THIS call (abortScope): driver adb/WDA calls made by the tool abort with it, and a
|
|
509
|
+
// Cancellation: the call's MCP signal (ctx.mcpReq.signal, the handler's last argument) is
|
|
510
|
+
// scoped to THIS call (abortScope): driver adb/WDA calls made by the tool abort with it, and a
|
|
259
511
|
// background job's signal (bound by the job itself) never leaks into or out of it.
|
|
260
|
-
const
|
|
261
|
-
const
|
|
512
|
+
const callCtx = a[a.length - 1];
|
|
513
|
+
const signal = callCtx?.mcpReq?.signal;
|
|
514
|
+
const callSignal = signal instanceof AbortSignal ? signal : undefined;
|
|
515
|
+
// Protocol 2026-07-28: a retry answering our consent InputRequiredResult echoes requestState.
|
|
516
|
+
const state = routing.era === 'modern' && typeof callCtx?.mcpReq?.requestState === 'function' ? callCtx.mcpReq.requestState() : undefined;
|
|
262
517
|
const run = async (callArgs) => {
|
|
263
518
|
const { value, resolved } = await runWithSignal(callSignal, () => withRootResolutionRecording(async () => runWithConsentScope(first?.sessionId, () => runWithResponseMode(mode, () => handler(...callArgs)))));
|
|
264
519
|
return annotateRootSource(value, resolved);
|
|
265
520
|
};
|
|
266
|
-
|
|
267
|
-
const
|
|
268
|
-
|
|
269
|
-
|
|
521
|
+
// Checked BEFORE the call so a tool that starts its own job never warns about itself.
|
|
522
|
+
const jobNote = runningJobNote(sessions, name, first?.sessionId);
|
|
523
|
+
const startedAt = Date.now();
|
|
524
|
+
let out;
|
|
525
|
+
let interim = false;
|
|
526
|
+
try {
|
|
527
|
+
let routed;
|
|
528
|
+
if (state !== undefined) {
|
|
529
|
+
if (typeof state !== 'string')
|
|
530
|
+
throw invalidRequestState(name, 'not a string');
|
|
531
|
+
routed = await resumeConsentPrompt(state, a, run, { sessions, tool: name });
|
|
532
|
+
}
|
|
533
|
+
else {
|
|
534
|
+
routed = await routePendingConsent(await run(a), a, run, { sessions, tool: name }, routing);
|
|
535
|
+
}
|
|
536
|
+
// An InputRequiredResult is an interim answer (the client's user is being asked): no tool
|
|
537
|
+
// status, no notes, it goes to the client as is.
|
|
538
|
+
if (isInputRequiredResult(routed)) {
|
|
539
|
+
interim = true;
|
|
540
|
+
return routed;
|
|
541
|
+
}
|
|
542
|
+
out = routed;
|
|
543
|
+
recordToolErrorFromResult(sessions, name, a[0], out, callSignal); // qa_report tool status (report/toolHealth.ts)
|
|
544
|
+
return jobNote ? annotateKeepingNotes(out, jobNote) : out;
|
|
545
|
+
}
|
|
546
|
+
finally {
|
|
547
|
+
if (interim)
|
|
548
|
+
log('debug', 'tool call', { tool: name, durationMs: Date.now() - startedAt, inputRequired: true });
|
|
549
|
+
else
|
|
550
|
+
logToolCall(name, first?.sessionId, startedAt, out, callSignal);
|
|
551
|
+
}
|
|
552
|
+
};
|
|
553
|
+
const schema = strictToolSchema(cfg?.inputSchema);
|
|
554
|
+
tools.set(name, {
|
|
555
|
+
name,
|
|
556
|
+
...(cfg?.title !== undefined ? { title: cfg.title } : {}),
|
|
557
|
+
description: cfg?.description ?? '',
|
|
558
|
+
annotations,
|
|
559
|
+
schema,
|
|
560
|
+
accepted: Object.keys(cfg?.inputSchema ?? {}),
|
|
561
|
+
handler: (args, extra) => wrapped(args, extra),
|
|
270
562
|
});
|
|
563
|
+
// Also registered with the SDK so McpServer declares the tools capability; its own
|
|
564
|
+
// tools/list + tools/call handlers are replaced by installToolHandlers once all tools exist.
|
|
565
|
+
return orig(name, annotated, wrapped);
|
|
271
566
|
};
|
|
272
567
|
}
|
|
273
568
|
/** Legacy (≤ 1.5) call shapes that a client spawned before the upgrade may still send, mapped to
|
|
@@ -297,36 +592,11 @@ function staleClientError(name, replacement) {
|
|
|
297
592
|
clientHint: STALE_CLIENT_HINT,
|
|
298
593
|
}, { removedCall: name, replacement });
|
|
299
594
|
}
|
|
300
|
-
/** The SDK stores handlers per method; wrapping the stored (already SDK-wrapped) function keeps
|
|
301
|
-
* the SDK's own request/result validation intact. */
|
|
302
|
-
function wrapRequestHandler(server, method, wrap) {
|
|
303
|
-
const map = server.server._requestHandlers;
|
|
304
|
-
const orig = map?.get(method);
|
|
305
|
-
if (!map || !orig) {
|
|
306
|
-
log('warn', 'could not wrap MCP request handler (SDK internals changed?)', { method });
|
|
307
|
-
return;
|
|
308
|
-
}
|
|
309
|
-
map.set(method, wrap(orig));
|
|
310
|
-
}
|
|
311
|
-
/** Drop the per-schema `$schema` dialect key (~2.8 KB of repetition across the tool list). */
|
|
312
|
-
export function stripSchemaDialect(result) {
|
|
313
|
-
const tools = result?.tools;
|
|
314
|
-
if (!Array.isArray(tools))
|
|
315
|
-
return result;
|
|
316
|
-
for (const t of tools) {
|
|
317
|
-
for (const k of ['inputSchema', 'outputSchema']) {
|
|
318
|
-
const sch = t[k];
|
|
319
|
-
if (sch && typeof sch === 'object' && '$schema' in sch)
|
|
320
|
-
delete sch.$schema;
|
|
321
|
-
}
|
|
322
|
-
}
|
|
323
|
-
return result;
|
|
324
|
-
}
|
|
325
595
|
/** Top-level argument keys a tool's input schema does not declare. The advertised JSON schema
|
|
326
|
-
* says additionalProperties:false,
|
|
327
|
-
*
|
|
328
|
-
*
|
|
329
|
-
*
|
|
596
|
+
* says additionalProperties:false, and the tool's strict zod object rejects them too: a call like
|
|
597
|
+
* qa_app_control { action:"force_stop", appId:"other.app" } must never run against the session's
|
|
598
|
+
* app while the caller believes it targeted another. Deprecated aliases that are still declared in
|
|
599
|
+
* the schema are accepted (they are schema properties). Exported for tests. */
|
|
330
600
|
export function unknownArgumentKeys(args, accepted) {
|
|
331
601
|
if (!args || typeof args !== 'object' || Array.isArray(args))
|
|
332
602
|
return [];
|
|
@@ -334,41 +604,128 @@ export function unknownArgumentKeys(args, accepted) {
|
|
|
334
604
|
return Object.keys(args).filter((k) => !known.has(k));
|
|
335
605
|
}
|
|
336
606
|
export function unknownArgumentsError(name, unknown, accepted) {
|
|
607
|
+
const shown = unknown.slice(0, MAX_ECHOED_KEYS).map(echoKey);
|
|
608
|
+
const more = unknown.length > shown.length ? ` (+${unknown.length - shown.length} more)` : '';
|
|
337
609
|
const list = (keys) => keys.map((k) => JSON.stringify(k)).join(', ');
|
|
338
610
|
return qaError({
|
|
339
|
-
what: `${name} does not accept the argument${unknown.length === 1 ? '' : 's'} ${list(
|
|
611
|
+
what: `${name} does not accept the argument${unknown.length === 1 ? '' : 's'} ${list(shown)}${more}. Nothing was run.`,
|
|
340
612
|
changedState: false,
|
|
341
613
|
retrySafe: true,
|
|
342
614
|
failureCode: 'INVALID_ARGUMENT',
|
|
343
615
|
nextSteps: [
|
|
344
|
-
`Remove ${list(
|
|
616
|
+
`Remove ${list(shown)}${more} and re-call. Accepted parameters: ${accepted.length ? list(accepted) : '(none)'}.`,
|
|
345
617
|
'If the tool list looks outdated, restart the MCP client so it reloads the current schemas.',
|
|
346
618
|
],
|
|
347
|
-
}, { unknownArguments: unknown, acceptedParameters: [...accepted] });
|
|
348
|
-
}
|
|
349
|
-
/**
|
|
350
|
-
*
|
|
351
|
-
*
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
619
|
+
}, { unknownArguments: shown, ...(more ? { unknownArgumentCount: unknown.length } : {}), acceptedParameters: [...accepted] });
|
|
620
|
+
}
|
|
621
|
+
/** Typed INVALID_ARGUMENT envelope for schema-validation failures (missing / wrong-typed / bad enum
|
|
622
|
+
* arguments), one `path: message` entry per issue ("sessionId: Required; target.text: Expected
|
|
623
|
+
* string, received number"). `issues` are already capped (checkToolArgs); `total` is the uncapped
|
|
624
|
+
* count. Exported for tests. */
|
|
625
|
+
export function invalidArgumentsError(name, issues, total, accepted) {
|
|
626
|
+
const more = total > issues.length ? `; (+${total - issues.length} more)` : '';
|
|
627
|
+
const what = issues.map((i) => `${i.path}: ${i.message}`).join('; ') + more;
|
|
628
|
+
const list = (keys) => keys.map((k) => JSON.stringify(k)).join(', ');
|
|
629
|
+
return qaError({
|
|
630
|
+
what: `${name}: invalid arguments: ${what}. Nothing was run.`,
|
|
631
|
+
changedState: false,
|
|
632
|
+
retrySafe: true,
|
|
633
|
+
failureCode: 'INVALID_ARGUMENT',
|
|
634
|
+
nextSteps: [
|
|
635
|
+
`Fix the listed argument(s) and re-call. Accepted parameters: ${accepted.length ? list(accepted) : '(none)'}.`,
|
|
636
|
+
'If the tool list looks outdated, restart the MCP client so it reloads the current schemas.',
|
|
637
|
+
],
|
|
638
|
+
}, { invalidArguments: [...issues], acceptedParameters: [...accepted] });
|
|
639
|
+
}
|
|
640
|
+
/** JSON-RPC code for "resource not found" on the wire: -32602 (Invalid params) with the URI in
|
|
641
|
+
* `data`. MCP 2025-11-25 suggested -32002 (what Swipium sent up to 2.1); SDK v2 rewrites -32002 to
|
|
642
|
+
* -32602 at its encode seam on every protocol revision, and 2026-07-28 requires -32602. The spec
|
|
643
|
+
* asks clients to accept both. */
|
|
644
|
+
export const RESOURCE_NOT_FOUND = ProtocolErrorCode.InvalidParams;
|
|
645
|
+
/** resources/read for a URI that matches a template but names nothing that exists: the SDK's
|
|
646
|
+
* typed ResourceNotFoundError (code RESOURCE_NOT_FOUND, `data.uri`) with a reason, instead of the
|
|
647
|
+
* generic -32603 a plain Error becomes. */
|
|
648
|
+
export function resourceNotFound(uri, why) {
|
|
649
|
+
return new ResourceNotFoundError(uri, `Resource not found: ${uri}: ${why}`);
|
|
650
|
+
}
|
|
651
|
+
/** tools/call result for a tool name nobody registered: the isError text result SDK 1.x has always
|
|
652
|
+
* produced (kept verbatim so clients see the same answer whatever SDK serves them). */
|
|
653
|
+
function unknownToolResult(name) {
|
|
654
|
+
return { content: [{ type: 'text', text: `MCP error -32602: Tool ${name} not found` }], isError: true };
|
|
655
|
+
}
|
|
656
|
+
/**
|
|
657
|
+
* tools/list and tools/call, answered from the ToolEntry table through the SDK's public low-level
|
|
658
|
+
* `Server.setRequestHandler` (replacing the handlers McpServer installed at registration):
|
|
659
|
+
* - tools/list: every tool with its strict-object JSON schema (no `$schema` key), computed once.
|
|
660
|
+
* 2025-era instances keep registration order (unchanged since 2.0); protocol 2026-07-28
|
|
661
|
+
* instances list tools sorted by name (deterministic order, cache-friendly). The schema hash is
|
|
662
|
+
* order-independent (it sorts by name itself), so it is the same for both;
|
|
663
|
+
* - tools/call, in order: removed tool names and legacy call shapes > STALE_CLIENT (with the
|
|
664
|
+
* replacement + stale-client hint); unknown tool names > the SDK 1.x "Tool not found" isError
|
|
665
|
+
* result; undeclared top-level arguments > INVALID_ARGUMENT (unknownArguments); schema failures
|
|
666
|
+
* > INVALID_ARGUMENT (invalidArguments), or STALE_CLIENT when the call is a legacy shape; then
|
|
667
|
+
* the wrapped handler runs with the validated arguments and the SDK request context. A handler
|
|
668
|
+
* that throws becomes an isError text result, as McpServer does.
|
|
669
|
+
*/
|
|
670
|
+
function installToolHandlers(server, tools, era) {
|
|
671
|
+
let listed;
|
|
672
|
+
server.server.setRequestHandler('tools/list', () => {
|
|
673
|
+
const ordered = [...tools.values()];
|
|
674
|
+
if (era === 'modern')
|
|
675
|
+
ordered.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
|
|
676
|
+
listed ??= ordered.map((t) => ({
|
|
677
|
+
name: t.name,
|
|
678
|
+
...(t.title !== undefined ? { title: t.title } : {}),
|
|
679
|
+
description: t.description,
|
|
680
|
+
inputSchema: toolInputJsonSchema(t.schema),
|
|
681
|
+
annotations: t.annotations,
|
|
682
|
+
execution: { taskSupport: 'forbidden' },
|
|
683
|
+
}));
|
|
684
|
+
return { tools: listed };
|
|
685
|
+
});
|
|
686
|
+
server.server.setRequestHandler('tools/call', async (request, ctx) => {
|
|
687
|
+
const name = String(request.params.name);
|
|
688
|
+
const args = request.params.arguments;
|
|
689
|
+
const startedAt = Date.now();
|
|
690
|
+
// Envelopes built here never reach the tool wrapper (the handler did not run): still log the
|
|
691
|
+
// debug tool-call line for them. Session ids are caller input here: only a sane string is logged.
|
|
692
|
+
const rejected = (out) => {
|
|
693
|
+
const sid = typeof args?.sessionId === 'string' && args.sessionId.length <= 64 ? args.sessionId : undefined;
|
|
694
|
+
logToolCall(name.slice(0, 100), sid, startedAt, out, undefined);
|
|
695
|
+
return out;
|
|
696
|
+
};
|
|
356
697
|
const replacement = staleClientReplacement(name, args);
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
if (
|
|
361
|
-
const unknown = unknownArgumentKeys(args, accepted);
|
|
698
|
+
const tool = tools.get(name);
|
|
699
|
+
if (!tool)
|
|
700
|
+
return replacement ? rejected(staleClientError(name, replacement)) : unknownToolResult(name);
|
|
701
|
+
if (!replacement) {
|
|
702
|
+
const unknown = unknownArgumentKeys(args, tool.accepted);
|
|
362
703
|
if (unknown.length)
|
|
363
|
-
return unknownArgumentsError(name, unknown, accepted);
|
|
704
|
+
return rejected(unknownArgumentsError(name, unknown, tool.accepted));
|
|
705
|
+
}
|
|
706
|
+
const checked = checkToolArgs(tool.schema, args);
|
|
707
|
+
if (!checked.ok) {
|
|
708
|
+
// Legacy enum values fail the CURRENT schema: say what replaced them instead.
|
|
709
|
+
if (replacement)
|
|
710
|
+
return rejected(staleClientError(`${name} ${JSON.stringify(args?.action ?? args?.for)}`, replacement));
|
|
711
|
+
return rejected(invalidArgumentsError(name, checked.issues, checked.total, tool.accepted));
|
|
712
|
+
}
|
|
713
|
+
try {
|
|
714
|
+
const result = await tool.handler(checked.data, ctx);
|
|
715
|
+
// Protocol 2026-07-28 consent prompt: the SDK seam checks it against the request's
|
|
716
|
+
// capabilities and the client retries the call with the answer.
|
|
717
|
+
if (isInputRequiredResult(result))
|
|
718
|
+
return result;
|
|
719
|
+
// projectCallToolResult: the SDK's per-era result projection, which low-level tools/call
|
|
720
|
+
// handlers apply themselves (identity for Swipium's object-shaped structuredContent).
|
|
721
|
+
return server.server.projectCallToolResult(result, undefined);
|
|
722
|
+
}
|
|
723
|
+
catch (error) {
|
|
724
|
+
if (error && typeof error === 'object' && requestStateErrors.has(error))
|
|
725
|
+
throw error; // -32602 on the wire
|
|
726
|
+
return { content: [{ type: 'text', text: error instanceof Error ? error.message : String(error) }], isError: true };
|
|
364
727
|
}
|
|
365
|
-
const result = (await orig(request, extra));
|
|
366
|
-
// Legacy enum values fail the CURRENT schema's validation > rewrite that raw error only.
|
|
367
|
-
if (replacement && result?.isError && !result.structuredContent)
|
|
368
|
-
return staleClientError(`${name} ${JSON.stringify(args?.action ?? args?.for)}`, replacement);
|
|
369
|
-
return result;
|
|
370
728
|
});
|
|
371
|
-
wrapRequestHandler(server, 'tools/list', (orig) => async (request, extra) => stripSchemaDialect(await orig(request, extra)));
|
|
372
729
|
}
|
|
373
730
|
/** Startup assertion: every registerTool() call must be
|
|
374
731
|
* allowlisted in TOOL_NAMES, and every TOOL_NAMES entry must actually get registered.
|
|
@@ -403,13 +760,14 @@ function capResourceListing(all, cap = RESOURCE_LIST_CAP) {
|
|
|
403
760
|
last.description = `${last.description ? `${last.description} ` : ''}[listing capped: showing ${cap} of ${all.length}; the rest remain readable by URI]`;
|
|
404
761
|
return shown;
|
|
405
762
|
}
|
|
406
|
-
/** Project roots the CURRENT client works in: its MCP roots (
|
|
407
|
-
* plus the roots of sessions created or used in this server process. resources/list is scoped to
|
|
763
|
+
/** Project roots the CURRENT client works in: its MCP roots (2025-era clients that advertise the
|
|
764
|
+
* capability) plus the roots of sessions created or used in this server process. resources/list is scoped to
|
|
408
765
|
* these so one client never browses another project's artifacts from the machine-wide registry. */
|
|
409
766
|
async function currentProjectRoots(server, sessions) {
|
|
410
767
|
const roots = new Set(sessions.activeRoots());
|
|
411
768
|
try {
|
|
412
|
-
|
|
769
|
+
// Never on a 2026-07-28 instance: roots/list is a server-to-client request it cannot send.
|
|
770
|
+
if (!servesModernEra(server) && server.server.getClientCapabilities()?.roots) {
|
|
413
771
|
const res = await server.server.listRoots(undefined, { timeout: 5_000 });
|
|
414
772
|
for (const r of res.roots ?? [])
|
|
415
773
|
if (typeof r.uri === 'string' && r.uri.startsWith('file://'))
|
|
@@ -463,19 +821,43 @@ function makeAppMapLister(sessions) {
|
|
|
463
821
|
return value;
|
|
464
822
|
};
|
|
465
823
|
}
|
|
824
|
+
/** Protocol 2026-07-28 cache hints (`ttlMs` / `cacheScope`, SEP-2549). The SDK carries them on a
|
|
825
|
+
* symbol-keyed field that only its 2026 encoder reads, so 2025-era responses are byte-identical.
|
|
826
|
+
* - tools, prompts, resource templates and server/discover never change while the process runs
|
|
827
|
+
* (a new Swipium version is a new process): fresh for an hour, identical for every user;
|
|
828
|
+
* - resources/list (session artifacts, app maps of the current project) and resources/read change
|
|
829
|
+
* as the run goes and name local paths: always stale, never shared. */
|
|
830
|
+
export const SERVER_CACHE_HINTS = {
|
|
831
|
+
'server/discover': { ttlMs: 3_600_000, cacheScope: 'public' },
|
|
832
|
+
'tools/list': { ttlMs: 3_600_000, cacheScope: 'public' },
|
|
833
|
+
'prompts/list': { ttlMs: 3_600_000, cacheScope: 'public' },
|
|
834
|
+
'resources/templates/list': { ttlMs: 3_600_000, cacheScope: 'public' },
|
|
835
|
+
'resources/list': { ttlMs: 0, cacheScope: 'private' },
|
|
836
|
+
'resources/read': { ttlMs: 0, cacheScope: 'private' },
|
|
837
|
+
};
|
|
466
838
|
/** Construct the server and register all tools + the artifact resource. Exported for tests. */
|
|
467
|
-
export function createServer() {
|
|
468
|
-
const
|
|
839
|
+
export function createServer(options = {}) {
|
|
840
|
+
const era = options.era ?? 'legacy';
|
|
841
|
+
const server = new McpServer({ name: 'swipium', version: SWIPIUM_VERSION }, { instructions: SERVER_INSTRUCTIONS, cacheHints: SERVER_CACHE_HINTS });
|
|
469
842
|
// Out-of-band consent (consent.ts header): when the connected client supports MCP
|
|
470
843
|
// elicitation, pending consents are decided by a real user prompt instead of a
|
|
471
|
-
// model-mediated re-call.
|
|
472
|
-
// since they are only known after `initialize` (long after tool registration).
|
|
473
|
-
|
|
474
|
-
|
|
844
|
+
// model-mediated re-call. 2025 era: the provider checks client capabilities lazily per call,
|
|
845
|
+
// since they are only known after `initialize` (long after tool registration). 2026-07-28: the
|
|
846
|
+
// prompt rides in an InputRequiredResult (routePendingConsent), gated on the per-request
|
|
847
|
+
// capabilities; the instance is marked so nothing tries a server-to-client request.
|
|
848
|
+
let provider;
|
|
849
|
+
if (era === 'modern')
|
|
850
|
+
markModernServer(server);
|
|
851
|
+
else {
|
|
852
|
+
provider = makeElicitationProvider(server);
|
|
853
|
+
setElicitationProvider(provider);
|
|
854
|
+
}
|
|
855
|
+
warnIgnoredPreapprovals(); // SWIPIUM_CONSENT_PREAPPROVE: one stderr line for ignored names
|
|
856
|
+
const sessions = options.sessions ?? new SessionStore();
|
|
475
857
|
const surface = [];
|
|
476
858
|
const attemptedToolNames = new Set();
|
|
477
|
-
const
|
|
478
|
-
installResponseModeWrapper(server, sessions, surface, attemptedToolNames,
|
|
859
|
+
const tools = new Map();
|
|
860
|
+
installResponseModeWrapper(server, sessions, surface, attemptedToolNames, tools, { era, ...(provider ? { provider } : {}) });
|
|
479
861
|
// Setup / context
|
|
480
862
|
registerDoctor(server);
|
|
481
863
|
registerStartSession(server, sessions);
|
|
@@ -527,7 +909,7 @@ export function createServer() {
|
|
|
527
909
|
// freeze the surface's content fingerprint.
|
|
528
910
|
assertToolSurface(attemptedToolNames);
|
|
529
911
|
setSchemaHash(computeSchemaHash(surface));
|
|
530
|
-
|
|
912
|
+
installToolHandlers(server, tools, era);
|
|
531
913
|
// Reusable workflow templates (MCP prompts capability): thin orchestration of the tools above.
|
|
532
914
|
registerPrompts(server);
|
|
533
915
|
// Artifacts as MCP resources (clients that support them); qa_get_artifact is the fallback.
|
|
@@ -555,12 +937,16 @@ export function createServer() {
|
|
|
555
937
|
}), { title: 'QA artifact', description: 'Session artifacts: screenshots, dumps, reports, logs.' }, async (uri) => {
|
|
556
938
|
const found = sessions.findArtifact(uri.href);
|
|
557
939
|
if (!found)
|
|
558
|
-
throw
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
return
|
|
940
|
+
throw resourceNotFound(uri.href, 'unknown artifact (check the URI, or list artifacts via resources/list or qa_report)');
|
|
941
|
+
try {
|
|
942
|
+
// Size-capped: big text returns head/tail + marker, big binaries are not inlined.
|
|
943
|
+
return readArtifactResource(uri.href, found.rec);
|
|
944
|
+
}
|
|
945
|
+
catch (e) {
|
|
946
|
+
if (e.code === 'ENOENT')
|
|
947
|
+
throw resourceNotFound(uri.href, 'the artifact file is gone (cleaned up?)');
|
|
948
|
+
throw e;
|
|
562
949
|
}
|
|
563
|
-
return { contents: [{ uri: uri.href, mimeType: rec.mime, text: readFileSync(rec.path, 'utf8') }] };
|
|
564
950
|
});
|
|
565
951
|
// App Knowledge Map as MCP resources: full map + per-feature / per-screen /
|
|
566
952
|
// test-suite sections, so large map data is read by URI instead of flooding a tool's text result.
|
|
@@ -576,11 +962,11 @@ export function createServer() {
|
|
|
576
962
|
const projectId = decodeUriSegment(String(vars.projectId));
|
|
577
963
|
const root = resolveAppMapRoot(projectId, sessions);
|
|
578
964
|
if (!root)
|
|
579
|
-
throw
|
|
965
|
+
throw resourceNotFound(uri.href, `unknown project ${projectId} (build the map first with qa_app_map_build)`);
|
|
580
966
|
const kind = vars.kind ? decodeUriSegment(String(vars.kind)) : undefined;
|
|
581
967
|
const res = readAppMapResource(root, { kind, id: vars.id ? decodeUriSegment(String(vars.id)) : undefined });
|
|
582
968
|
if (!res)
|
|
583
|
-
throw
|
|
969
|
+
throw resourceNotFound(uri.href, 'no such app-map section (list ids with qa_app_map_read)');
|
|
584
970
|
return { contents: [{ uri: uri.href, mimeType: res.mimeType, text: res.text }] };
|
|
585
971
|
});
|
|
586
972
|
// Bare full-map URI: swipium://project/{projectId}/app-map (no trailing section).
|
|
@@ -590,10 +976,10 @@ export function createServer() {
|
|
|
590
976
|
const projectId = decodeUriSegment(String(vars.projectId));
|
|
591
977
|
const root = resolveAppMapRoot(projectId, sessions);
|
|
592
978
|
if (!root)
|
|
593
|
-
throw
|
|
979
|
+
throw resourceNotFound(uri.href, `unknown project ${projectId} (build the map first with qa_app_map_build)`);
|
|
594
980
|
const res = readAppMapResource(root, {});
|
|
595
981
|
if (!res)
|
|
596
|
-
throw
|
|
982
|
+
throw resourceNotFound(uri.href, 'no app map on disk for this project (build it with qa_app_map_build)');
|
|
597
983
|
return { contents: [{ uri: uri.href, mimeType: res.mimeType, text: res.text }] };
|
|
598
984
|
});
|
|
599
985
|
return { server, sessions };
|
|
@@ -609,7 +995,10 @@ export async function startServer() {
|
|
|
609
995
|
catch (e) {
|
|
610
996
|
log('warn', 'android sdk path setup failed', { err: String(e) });
|
|
611
997
|
}
|
|
612
|
-
|
|
998
|
+
// One SessionStore for the process: serveStdio builds the McpServer instance lazily, when the
|
|
999
|
+
// client's opening message picks the era, and may build (then discard) a `server/discover`
|
|
1000
|
+
// probe instance before a 2025 client falls back to `initialize`.
|
|
1001
|
+
const sessions = new SessionStore();
|
|
613
1002
|
const transport = new StdioServerTransport();
|
|
614
1003
|
// Persistence is debounced (SessionStore.persist), so make sure a graceful exit never
|
|
615
1004
|
// loses the trailing write. 'exit' handlers must be synchronous; flushAll is.
|
|
@@ -622,8 +1011,13 @@ export async function startServer() {
|
|
|
622
1011
|
return;
|
|
623
1012
|
restoring = true;
|
|
624
1013
|
const changed = sessions.list().filter((s) => s.network?.changed).length;
|
|
625
|
-
|
|
1014
|
+
// Cancel every running job FIRST: a job left running could flip device network state (or
|
|
1015
|
+
// anything else) back after the restore below.
|
|
1016
|
+
const cancelledJobs = cancelAllRunningJobs(sessions);
|
|
1017
|
+
log('info', 'shutdown: restoring network', { why, changed, cancelledJobs });
|
|
626
1018
|
try {
|
|
1019
|
+
if (cancelledJobs)
|
|
1020
|
+
await new Promise((r) => setImmediate(r)); // let aborted workers unwind
|
|
627
1021
|
await restoreAllNetwork(sessions);
|
|
628
1022
|
await stopAllRecordings(); // don't leave a device screen-recording after we exit
|
|
629
1023
|
await stopAllMetro(sessions); // don't leave a node bundler holding :8081 after we exit
|
|
@@ -641,14 +1035,26 @@ export async function startServer() {
|
|
|
641
1035
|
process.once('SIGTERM', () => void restoreThenExit(143, 'SIGTERM'));
|
|
642
1036
|
// Startup banner (P1.8): version + tool count on stderr so a stale build is obvious in logs.
|
|
643
1037
|
log('info', 'swipium starting', { version: SWIPIUM_VERSION, tools: TOOL_COUNT });
|
|
644
|
-
|
|
645
|
-
//
|
|
646
|
-
//
|
|
1038
|
+
// Dual-era stdio (spec 2026-07-28 basic/versioning): serveStdio answers `server/discover` for
|
|
1039
|
+
// protocol 2026-07-28 clients (per-request `_meta`, no handshake) and `initialize` for
|
|
1040
|
+
// 2025-06-18 / 2025-11-25 clients, pinning ONE instance per connection to the era the client
|
|
1041
|
+
// opened with. Every instance comes from createServer, so both eras serve the same tools.
|
|
1042
|
+
serveStdio(({ era }) => {
|
|
1043
|
+
log('debug', 'mcp connection era', { era });
|
|
1044
|
+
return createServer({ era, sessions }).server;
|
|
1045
|
+
}, { transport, onerror: (e) => log('debug', 'mcp stdio', { err: String(e) }) });
|
|
1046
|
+
// Chain AFTER serveStdio: it sets the transport's onclose itself, so wrap rather than assign
|
|
1047
|
+
// before (which gets overwritten). stdin EOF = client disconnected.
|
|
647
1048
|
const sdkOnClose = transport.onclose;
|
|
648
1049
|
transport.onclose = () => {
|
|
649
1050
|
sdkOnClose?.();
|
|
650
1051
|
void restoreThenExit(0, 'transport-close');
|
|
651
1052
|
};
|
|
1053
|
+
// stdin EOF = the client is gone. SDK v2's StdioServerTransport closes itself on EOF (onclose
|
|
1054
|
+
// above); SDK 1.x never did, and an in-flight call (a 30 s qa_wait, a long job) kept the process
|
|
1055
|
+
// alive. Listen ourselves too, so the cleanup never depends on the SDK version (restoreThenExit
|
|
1056
|
+
// runs once).
|
|
1057
|
+
process.stdin.once('end', () => void restoreThenExit(0, 'stdin-end'));
|
|
652
1058
|
log('info', 'swipium connected over stdio');
|
|
653
1059
|
// Reap long-lived children (Metro, managed WDA, recorders) left behind by a crashed previous
|
|
654
1060
|
// server run. Runs AFTER connect and in the background, so a slow sweep (lock wait, WDA /status
|
|
@@ -658,6 +1064,25 @@ export async function startServer() {
|
|
|
658
1064
|
// resumed iOS session keeps its WDA (shutdown intentionally leaves managed WDA running).
|
|
659
1065
|
void startOrphanSweep();
|
|
660
1066
|
}
|
|
1067
|
+
/** Shutdown: cancel (abort) every running job in every session. Returns how many were cancelled.
|
|
1068
|
+
* Exported for tests. */
|
|
1069
|
+
export function cancelAllRunningJobs(sessions) {
|
|
1070
|
+
let n = 0;
|
|
1071
|
+
for (const s of sessions.list()) {
|
|
1072
|
+
for (const j of [...(s.jobs?.values() ?? [])]) {
|
|
1073
|
+
if (j.status !== 'running')
|
|
1074
|
+
continue;
|
|
1075
|
+
try {
|
|
1076
|
+
if (sessions.cancelJob(s, j.jobId))
|
|
1077
|
+
n++;
|
|
1078
|
+
}
|
|
1079
|
+
catch (e) {
|
|
1080
|
+
log('warn', 'shutdown: cancel job failed', { sessionId: s.id, jobId: j.jobId, err: String(e) });
|
|
1081
|
+
}
|
|
1082
|
+
}
|
|
1083
|
+
}
|
|
1084
|
+
return n;
|
|
1085
|
+
}
|
|
661
1086
|
/** Background orphan sweep; never rejects. Exported for tests. */
|
|
662
1087
|
export async function startOrphanSweep(reap = reapOrphanedProcesses) {
|
|
663
1088
|
try {
|