simframe 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/mcp.js CHANGED
@@ -13,11 +13,63 @@ import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
13
13
  import * as actions from './actions.js';
14
14
  import * as api from './index.js';
15
15
  import * as input from './input.js';
16
+ import * as metrics from './metrics.js';
16
17
  import * as navigate from './navigate.js';
17
18
  import { bootedDevices, permissionServices } from './platform/index.js';
18
19
  import * as store from './store.js';
19
20
  import * as view from './view.js';
20
21
 
22
+ /**
23
+ * Per-call overrides for the two experiment knobs.
24
+ *
25
+ * Both are read from the call first and the environment second, so an A/B is an
26
+ * argument rather than a server restart — and a run in the wrong mode stops
27
+ * being a thing that can silently happen.
28
+ */
29
+ export function modesFor(base = {}, args = {}) {
30
+ const out = { ...base };
31
+ if (args.sensor) out.sensor = String(args.sensor);
32
+ if (args.planner) out.planner = String(args.planner);
33
+ if (args.supervisor) out.supervisor = String(args.supervisor);
34
+ return out;
35
+ }
36
+
37
+ /** Offered on every tool that reads or acts, because either can be compared. */
38
+ /**
39
+ * Which declared-required argument is absent, phrased as advice.
40
+ *
41
+ * Reads the tool's own `required` array, so a tool that gains a required
42
+ * argument gains this for free and cannot drift out of step with it.
43
+ */
44
+ function missingRequired(name, args) {
45
+ const tool = TOOLS.find((t) => t.name === name);
46
+ const need = tool?.inputSchema?.required ?? [];
47
+ const absent = need.filter((k) => args?.[k] === undefined || args?.[k] === null || args?.[k] === '');
48
+ if (!absent.length) return null;
49
+ const known = Object.keys(args ?? {}).filter((k) => k !== 'device');
50
+ return `${name}: missing required ${absent.map((k) => `"${k}"`).join(', ')}.`
51
+ + (known.length ? ` You passed: ${known.map((k) => `"${k}"`).join(', ')}.` : '')
52
+ + ` ${absent.length === 1 ? 'That argument is' : 'Those arguments are'} the one${absent.length === 1 ? '' : 's'} this tool acts on — pass ${absent.map((k) => `"${k}"`).join(' and ')} and retry.`;
53
+ }
54
+
55
+ const modeProps = {
56
+ sensor: {
57
+ type: 'string',
58
+ enum: ['full', 'ax-first'],
59
+ description: 'Perception for this call. "full" fuses accessibility and OCR (~164ms). "ax-first" reads the tree alone (~50ms) and pays for OCR only when something fails to resolve. Omit to keep the server default.',
60
+ },
61
+ planner: {
62
+ type: 'string',
63
+ enum: ['none', 'apple'],
64
+ description: 'Local model for this call. Orders the containers seek opens; it cannot choose an action. Omit to keep the server default.',
65
+ },
66
+ supervisor: {
67
+ type: 'string',
68
+ enum: ['none', 'apple'],
69
+ description: 'Local supervisor for this call. When a step fails it answers wait, retry or stop — nothing else — before the batch is abandoned. Omit to keep the server default.',
70
+ },
71
+ };
72
+
21
73
  const deviceProp = {
22
74
  device: {
23
75
  type: 'string',
@@ -42,11 +94,12 @@ const TOOLS = [
42
94
  {
43
95
  name: 'sim_ui',
44
96
  description:
45
- 'READ THE SCREEN. Returns a compact text map: every element with a number, its region (nav-bar / content / tab-bar), type, label, state and tap point, plus which screen this is and how much simframe already knows about it. Roughly a tenth the cost of a screenshot and strictly more useful, because it says what is tappable and where — no measuring pixels by eye. The numbers are selectors: whatever this returns as #3, you can tap as "#3". Start here, not with sim_look.',
97
+ 'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Whatever it calls #3, you can tap as "#3". Start here, never with sim_look.',
46
98
  inputSchema: {
47
99
  type: 'object',
48
100
  properties: {
49
101
  ...deviceProp,
102
+ ...modeProps,
50
103
  filter: { type: 'string', description: 'Only elements whose label or read text contains this.' },
51
104
  interactive: { type: 'boolean', description: 'Only elements that look tappable.' },
52
105
  all: { type: 'boolean', description: 'Include the status bar and every collapsed region (default false).' },
@@ -57,20 +110,25 @@ const TOOLS = [
57
110
  {
58
111
  name: 'sim_do',
59
112
  description:
60
- 'THE MAIN TOOL. Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each step waits for the screen to settle against a baseline captured before it, and is verified against what that action did here last time — so a step that navigated somewhere unintended stops the flow instead of tapping on into the wrong screen. A twelve-step flow costs one round trip instead of twelve. Prefer this over the single-action tools whenever you know more than one step ahead, and put asserts in the flow rather than checking between calls. Returns the compact screen map of where the flow ended; no image.',
113
+ 'THE MAIN TOOL, and the cheapest path. Plan the WHOLE flow and run it in one call — tap, type, scroll, wait, assert — asserting after each step that matters. Every step settles and is checked against what that action did here before, so a wrong turn halts the flow instead of tapping on. Single-action tools are for recovery.',
61
114
  inputSchema: {
62
115
  type: 'object',
63
116
  properties: {
64
117
  ...deviceProp,
118
+ ...modeProps,
119
+ supervise: {
120
+ type: 'string',
121
+ description: 'What the local supervisor should know about this app while the batch runs — how lists load, what makes a control stay disabled, what a benign failure looks like here. It has no knowledge of the app; you do. Ignored when no supervisor is enabled.',
122
+ },
65
123
  steps: {
66
124
  type: 'array',
67
125
  description:
68
- 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}.',
126
+ 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
69
127
  items: { type: 'object' },
70
128
  },
71
129
  autoSettle: {
72
130
  type: 'boolean',
73
- description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
131
+ description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input — with it off the trailing map is read before the last gesture has finished, so it can describe the screen you were on rather than the one you are on.',
74
132
  },
75
133
  stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
76
134
  timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
@@ -92,7 +150,7 @@ const TOOLS = [
92
150
  description: 'Tap one thing. For more than one step, use sim_do — it batches the verification and costs one round trip. Returns the screen map afterwards.',
93
151
  inputSchema: {
94
152
  type: 'object',
95
- properties: { ...deviceProp, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
153
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
96
154
  required: ['sel'],
97
155
  },
98
156
  },
@@ -101,7 +159,7 @@ const TOOLS = [
101
159
  description: 'Focus a field and type into it. Prefer a sim_do step when this is part of a sequence.',
102
160
  inputSchema: {
103
161
  type: 'object',
104
- properties: { ...deviceProp, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
162
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
105
163
  required: ['sel', 'text'],
106
164
  },
107
165
  },
@@ -112,6 +170,7 @@ const TOOLS = [
112
170
  type: 'object',
113
171
  properties: {
114
172
  ...deviceProp,
173
+ ...modeProps,
115
174
  ...selectorProp('What to bring into view'),
116
175
  direction: { type: 'string', enum: ['down', 'up', 'left', 'right'] },
117
176
  maxScrolls: { type: 'number', description: 'Give up after this many screens (default 6).' },
@@ -124,7 +183,7 @@ const TOOLS = [
124
183
  description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
125
184
  inputSchema: {
126
185
  type: 'object',
127
- properties: { ...deviceProp, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
186
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
128
187
  required: ['sel'],
129
188
  },
130
189
  },
@@ -135,6 +194,7 @@ const TOOLS = [
135
194
  type: 'object',
136
195
  properties: {
137
196
  ...deviceProp,
197
+ ...modeProps,
138
198
  ...selectorProp('What to check'),
139
199
  is: {
140
200
  type: 'string',
@@ -149,10 +209,10 @@ const TOOLS = [
149
209
  {
150
210
  name: 'sim_goto',
151
211
  description:
152
- 'Walk to a screen simframe has already been to, by name, planning the route through remembered transitions and verifying every step. Zero reasoning and zero images: the graph knows which taps lead where. Refuses rather than guesses — if the destination is unknown, ambiguous, or unreachable through known transitions, it says which and lists what it does know. Call it with no target to see the known screens.',
212
+ 'Walk to a screen simframe already knows, over edges it has already verified, with no model call per step. Names come from sim_recall or a previous map. Refuses rather than guesses when the route is unknown or the name is ambiguous — a refusal is cheap and a wrong walk is not.',
153
213
  inputSchema: {
154
214
  type: 'object',
155
- properties: { ...deviceProp, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
215
+ properties: { ...deviceProp, ...modeProps, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
156
216
  },
157
217
  },
158
218
  {
@@ -160,7 +220,7 @@ const TOOLS = [
160
220
  description: 'Replay a saved flow by name, verifying each step. Omit `name` to list the saved flows. Save one with sim_do\'s `saveAs`.',
161
221
  inputSchema: {
162
222
  type: 'object',
163
- properties: { ...deviceProp, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
223
+ properties: { ...deviceProp, ...modeProps, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
164
224
  },
165
225
  },
166
226
  {
@@ -170,6 +230,7 @@ const TOOLS = [
170
230
  type: 'object',
171
231
  properties: {
172
232
  ...deviceProp,
233
+ ...modeProps,
173
234
  bundleId: { type: 'string' },
174
235
  relaunch: { type: 'boolean', description: 'Terminate first. Without this, launching an already-running app silently does nothing and you test the screen you were already on.' },
175
236
  args: { type: 'array', items: { type: 'string' }, description: 'Launch arguments passed to the app.' },
@@ -183,7 +244,7 @@ const TOOLS = [
183
244
  description: 'Open a URL or deep link on the device — the fastest way to reach a screen when the app has a link for it.',
184
245
  inputSchema: {
185
246
  type: 'object',
186
- properties: { ...deviceProp, url: { type: 'string' } },
247
+ properties: { ...deviceProp, ...modeProps, url: { type: 'string' } },
187
248
  required: ['url'],
188
249
  },
189
250
  },
@@ -194,6 +255,7 @@ const TOOLS = [
194
255
  type: 'object',
195
256
  properties: {
196
257
  ...deviceProp,
258
+ ...modeProps,
197
259
  action: { type: 'string', enum: ['grant', 'revoke', 'reset'] },
198
260
  service: { type: 'string' },
199
261
  bundleId: { type: 'string' },
@@ -204,10 +266,10 @@ const TOOLS = [
204
266
  {
205
267
  name: 'sim_find',
206
268
  description:
207
- 'Resolve an intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, where on screen you meant, and icon-only controls by their common name. When two things answer equally well it says so and lists them rather than guessing — a wrong tap is worse than a question, because it can do something and leave you believing it did the right thing. Use it when you are unsure a selector will resolve; otherwise just tap.',
269
+ 'Resolve one intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, and where on screen you meant. When two things answer equally well it says so and lists them rather than guessing. Use it when you doubt a selector will resolve; otherwise just tap. Prefer a label or a #ref over coordinates.',
208
270
  inputSchema: {
209
271
  type: 'object',
210
- properties: { ...deviceProp, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
272
+ properties: { ...deviceProp, ...modeProps, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
211
273
  required: ['intent'],
212
274
  },
213
275
  },
@@ -219,6 +281,7 @@ const TOOLS = [
219
281
  type: 'object',
220
282
  properties: {
221
283
  ...deviceProp,
284
+ ...modeProps,
222
285
  since: { type: 'string', description: 'Compare against this frame hash instead of your last look.' },
223
286
  },
224
287
  },
@@ -226,11 +289,12 @@ const TOOLS = [
226
289
  {
227
290
  name: 'sim_wait',
228
291
  description:
229
- 'Block until the screen finishes reacting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly in the moment before an animation starts. Returns the compact screen map by default, not an image. Inside a flow you rarely need this: sim_do settles after every step already.',
292
+ 'Wait for the screen to change, settle, or both. sim_do already settles after every step, so you rarely need this inside a flow — reach for it when something moves without you acting, like a push or a background load. Pass `since` from a hash captured before the thing you are waiting on.',
230
293
  inputSchema: {
231
294
  type: 'object',
232
295
  properties: {
233
296
  ...deviceProp,
297
+ ...modeProps,
234
298
  mode: {
235
299
  type: 'string',
236
300
  enum: ['settle', 'change', 'stable'],
@@ -246,16 +310,24 @@ const TOOLS = [
246
310
  {
247
311
  name: 'sim_look',
248
312
  description:
249
- 'THE ONLY TOOL THAT RETURNS AN IMAGE, and the most expensive one. Returns the newest buffered frame immediately — no screenshot wait. Call it only when the text map is genuinely not enough: checking visual layout, colour, spacing, an animation, or something the accessibility tree and OCR both cannot see. For "what is on screen and what can I tap", sim_ui answers better and costs a tenth as much.',
313
+ 'A screenshot: ~1600 tokens, the most expensive call here. Only for what text cannot answer — layout, colour, spacing, a control the map omits. NOT for what a field contains, whether a button is enabled, or whether an action worked: sim_ui reports the first two and the flow\'s own verdict already answered the third.',
250
314
  inputSchema: {
251
315
  type: 'object',
252
316
  properties: {
253
317
  ...deviceProp,
318
+ ...modeProps,
254
319
  detail: {
255
320
  type: 'string',
256
321
  enum: ['low', 'normal', 'high'],
257
322
  description: 'Image size: low (~420px, cheapest), normal (~700px, default), high (1024px, readable small text). Capped at 1024px on the long edge.',
258
323
  },
324
+ region: {
325
+ type: 'object',
326
+ description: 'Crop to part of the screen and enlarge it, in POINTS — the same coordinates the element map prints: {"x":18,"y":260,"width":366,"height":80}. Use it when a whole screen cannot answer the question at 1024px: selected versus unselected, a chevron, a validation mark. Pair it with detail:"high".',
327
+ properties: {
328
+ x: { type: 'number' }, y: { type: 'number' }, width: { type: 'number' }, height: { type: 'number' },
329
+ },
330
+ },
259
331
  maxAgeMs: { type: 'number', description: 'If the buffered frame is older than this, wait for a fresher one (default 900).' },
260
332
  },
261
333
  },
@@ -268,6 +340,7 @@ const TOOLS = [
268
340
  type: 'object',
269
341
  properties: {
270
342
  ...deviceProp,
343
+ ...modeProps,
271
344
  count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
272
345
  spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
273
346
  thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
@@ -277,11 +350,12 @@ const TOOLS = [
277
350
  {
278
351
  name: 'sim_recall',
279
352
  description:
280
- 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago, how long it took, how much of the screen moved. action "at" returns the buffered frame from a past moment — an image. Use this when you look up and find the screen already different, instead of re-running the action.',
353
+ 'What happened recently, as text: the screens visited, the actions taken, and what each one did. Use it to re-orient after a failure instead of taking a screenshot, and to learn the screen names sim_goto accepts.',
281
354
  inputSchema: {
282
355
  type: 'object',
283
356
  properties: {
284
357
  ...deviceProp,
358
+ ...modeProps,
285
359
  action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
286
360
  spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
287
361
  msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
@@ -297,6 +371,7 @@ const TOOLS = [
297
371
  properties: {
298
372
  action: { type: 'string', enum: ['status', 'start', 'stop'] },
299
373
  ...deviceProp,
374
+ ...modeProps,
300
375
  fps: { type: 'number', description: 'Capture rate while the screen is moving (default 4).' },
301
376
  },
302
377
  required: ['action'],
@@ -338,9 +413,38 @@ function baselineFor(udid, explicit) {
338
413
  const text = (s) => ({ type: 'text', text: s });
339
414
  const image = (png) => ({ type: 'image', data: png.toString('base64'), mimeType: 'image/png' });
340
415
 
416
+ /**
417
+ * The device's name, and its UDID when the name cannot identify it.
418
+ *
419
+ * Two booted simulators can share a name — measured on this machine: two called
420
+ * "iPhone 17 Pro" at once, which is the default state after creating a second
421
+ * device of the same model. A caller reading a header that says only
422
+ * "iPhone 17 Pro" has no way to tell which one answered, and a reporter spent a
423
+ * session unsure whether they were looking at their own app. `xcrun simctl list
424
+ * devices booted` makes the collision trivial to see, so the header says which.
425
+ */
426
+ function deviceLabel(device) {
427
+ if (!device?.udid) return device?.name ?? 'unknown device';
428
+ const clash = (lastBooted ?? []).filter((d) => d.name === device.name).length > 1;
429
+ return clash ? `${device.name} (${device.udid.slice(0, 8)})` : device.name;
430
+ }
431
+
432
+ /** What the last device listing saw, so a name collision can be noticed at all. */
433
+ let lastBooted = null;
434
+ export function noteBooted(devices) {
435
+ lastBooted = Array.isArray(devices) ? devices.map((d) => ({ name: d.name, udid: d.udid })) : null;
436
+ const names = new Map();
437
+ for (const d of lastBooted ?? []) names.set(d.name, (names.get(d.name) ?? 0) + 1);
438
+ const shared = [...names].filter(([, n]) => n > 1).map(([name]) => name);
439
+ return shared.length
440
+ ? `WARNING: ${shared.map((n) => JSON.stringify(n)).join(', ')} names more than one booted device`
441
+ + ' — pass "device" with a UDID, because a name cannot identify which one you mean'
442
+ : null;
443
+ }
444
+
341
445
  function header(device, state, ageMs, extra = '') {
342
446
  return (
343
- `${device.name} · ${device.runtime} · frame #${state.seq} · ${ageMs}ms old · ` +
447
+ `${deviceLabel(device)} · ${device.runtime} · frame #${state.seq} · ${ageMs}ms old · ` +
344
448
  `${state.width}x${state.height} · still for ${state.stableForMs}ms${extra ? ` · ${extra}` : ''}`
345
449
  );
346
450
  }
@@ -364,7 +468,9 @@ function sinceLine(since) {
364
468
  : `unchanged since your last look ${since.ageMs}ms ago`;
365
469
  }
366
470
 
367
- export async function serve({ device: defaultDevice, options = {} } = {}) {
471
+ export async function serve({ device: defaultDevice, options: baseOptions = {} } = {}) {
472
+ // The last device a caller named, for the life of this server.
473
+ let lastDevice = null;
368
474
  const server = new Server(
369
475
  { name: 'simframe', version: packageVersion() },
370
476
  { capabilities: { tools: {} } },
@@ -374,7 +480,34 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
374
480
 
375
481
  server.setRequestHandler(CallToolRequestSchema, async (req) => {
376
482
  const args = req.params.arguments || {};
377
- const target = args.device || defaultDevice;
483
+ // Sticky, and stickiness is not guessing.
484
+ //
485
+ // Reported: `sim_launch` accepted `device`, and the very next `sim_ui`
486
+ // refused with "2 simulators are booted and none was named" — so a UDID had
487
+ // to ride on all ~15 subsequent calls. Refusing to *choose* between two
488
+ // booted devices is right; forgetting which one the caller already named is
489
+ // not. This remembers only what was explicitly passed, so nothing is ever
490
+ // inferred from a boot list.
491
+ const target = args.device || lastDevice || defaultDevice;
492
+ if (args.device) lastDevice = String(args.device);
493
+ // An MCP server's environment is fixed when it spawns, so a tester asked to
494
+ // compare two sensor modes inside one session could not: round 6 ran its
495
+ // baseline and could not run either variant. The suggested workaround was
496
+ // three server entries with three env blocks, which is worse — three servers
497
+ // on one device is three writers, against the one-writer-per-device rule.
498
+ //
499
+ // So the modes are arguments. Absent, the environment still decides, so
500
+ // nothing that was working changes.
501
+ const options = modesFor(baseOptions, args);
502
+ // Name the missing argument, using the tool's own schema.
503
+ //
504
+ // Reported: `sim_scroll_to` called with `target:` instead of `sel:` answered
505
+ // `simframe: empty step` — which names neither the tool, nor the parameter,
506
+ // nor even that an argument was absent, and cost a schema lookup to decode.
507
+ // The declared `required` list is right there, so the check is generic
508
+ // rather than one guard per tool, and it says what to pass.
509
+ const missing = missingRequired(req.params.name, args);
510
+ if (missing) return { content: [text(missing)], isError: true };
378
511
  try {
379
512
  switch (req.params.name) {
380
513
  case 'sim_look':
@@ -450,6 +583,81 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
450
583
  await server.connect(new StdioServerTransport());
451
584
  }
452
585
 
586
+ /**
587
+ * Crop a frame to a region of the screen and enlarge it.
588
+ *
589
+ * A whole screen at 1024px cannot answer a question about one control: a
590
+ * selected filter chip and an unselected one look identical at that size, and an
591
+ * agent shelled out to `simctl io` and PIL to crop and upscale a chip row for
592
+ * every check it made. The region arrives in **points** — the same coordinates
593
+ * the element map prints — because that is what a caller has in hand.
594
+ */
595
+ /**
596
+ * Read a region argument that may not have arrived as an object.
597
+ *
598
+ * A client is free to hand a declared-object property over as a JSON string,
599
+ * and one did: `{"x":0,"y":60,...}` arrived as text, every field read as
600
+ * undefined, and the crop silently became the whole screen — reported back as
601
+ * `cropped to 402x874pt at 0,0`, which is a crop that did not happen described
602
+ * as one that did. An array is accepted too, because [x, y, width, height] is
603
+ * the shape anyone would try first, and it used to fail exactly as quietly.
604
+ */
605
+ export function readRegion(region) {
606
+ let r = region;
607
+ if (typeof r === 'string') {
608
+ try { r = JSON.parse(r); } catch { return null; }
609
+ }
610
+ if (Array.isArray(r)) {
611
+ const [x, y, width, height] = r.map(Number);
612
+ return [x, y, width, height].every(Number.isFinite) ? { x, y, width, height } : null;
613
+ }
614
+ if (!r || typeof r !== 'object') return null;
615
+ const num = (...keys) => {
616
+ for (const k of keys) if (Number.isFinite(Number(r[k]))) return Number(r[k]);
617
+ return null;
618
+ };
619
+ const width = num('width', 'w');
620
+ const height = num('height', 'h');
621
+ // A region with no size is not a region. Saying so beats returning the whole
622
+ // screen under a caption that claims otherwise.
623
+ if (width == null && height == null) return null;
624
+ return { x: num('x') ?? 0, y: num('y') ?? 0, width, height };
625
+ }
626
+
627
+ async function cropRegion(png, region, points) {
628
+ try {
629
+ const { decodePng, encodePng, cropBitmap, scaleBitmap } = await import('./png.js');
630
+ const bmp = decodePng(png);
631
+ const pw = points?.width;
632
+ const ph = points?.height;
633
+ if (!Number.isFinite(pw) || !Number.isFinite(ph) || !pw || !ph) return { note: 'the screen point size is unknown' };
634
+ // The frame we hold is already downscaled for the model, so map points onto
635
+ // *this* bitmap rather than onto the device's native pixels.
636
+ const sx = bmp.width / pw;
637
+ const sy = bmp.height / ph;
638
+ const x = Number(region.x ?? 0) * sx;
639
+ const y = Number(region.y ?? 0) * sy;
640
+ const w = Number(region.width ?? region.w ?? pw) * sx;
641
+ const h = Number(region.height ?? region.h ?? ph) * sy;
642
+ if (!(w >= 1) || !(h >= 1)) return { note: 'the region has no size' };
643
+ const cut = cropBitmap(bmp, x, y, w, h);
644
+ // Enlarge to the same budget the whole screen gets, so the detail per point
645
+ // is the whole reason to ask for a region.
646
+ const long = Math.max(cut.width, cut.height);
647
+ const factor = Math.min(6, Math.max(1, Math.round(1024 / long)));
648
+ const big = factor > 1 ? scaleBitmap(cut, cut.width * factor, cut.height * factor) : cut;
649
+ return {
650
+ png: encodePng(big),
651
+ note: `cropped to ${Math.round(Number(region.width ?? region.w ?? pw))}x`
652
+ + `${Math.round(Number(region.height ?? region.h ?? ph))}pt at `
653
+ + `${Math.round(Number(region.x ?? 0))},${Math.round(Number(region.y ?? 0))}`
654
+ + `, enlarged ${factor}x`,
655
+ };
656
+ } catch (err) {
657
+ return { note: String(err.message).split('\n')[0] };
658
+ }
659
+ }
660
+
453
661
  async function look(target, args, options) {
454
662
  const maxAge = args.maxAgeMs ?? 900;
455
663
  // Never native resolution. An image is the expensive path by a factor of ten
@@ -472,9 +680,37 @@ async function look(target, args, options) {
472
680
  const st = await api.getState(target, { since: prior, options });
473
681
  remember(res.device.udid, res.state);
474
682
  const lines = [header(res.device, res.state, res.ageMs), sinceLine(st.since)];
683
+ // Louder than a trailing note, because it invalidates the image itself rather
684
+ // than qualifying it.
685
+ if (res.frameBehindMs) {
686
+ lines.unshift(`WARNING: this image is ${Math.round(res.frameBehindMs / 100) / 10}s older than the screen state`
687
+ + ' — it is very likely NOT what is on the device now. Read sim_ui, which is read live,'
688
+ + ' or look again in a moment.');
689
+ }
475
690
  const warn = livenessLine(st.live);
476
691
  if (warn) lines.unshift(warn);
477
- return { content: [text(lines.filter(Boolean).join('\n')), image(res.png)] };
692
+ let png = res.png;
693
+ if (args.region) {
694
+ const parsed = readRegion(args.region);
695
+ if (!parsed) {
696
+ lines.push('that region could not be read (expected {"x":0,"y":260,"width":402,"height":80} in points,'
697
+ + ' or [x, y, width, height]) — this is the whole screen');
698
+ }
699
+ // The point size, which the frame does not carry: `state.width/height` are
700
+ // the *captured frame's* pixels (322x700 here), not the screen's points
701
+ // (402x874). Scaling by them gave a 1:1 ratio, so a crop at y=760 clamped
702
+ // to a single pixel row and returned a 119-byte image — which looked like
703
+ // it had worked. `screenIdentity` answers it in ~17ms warm.
704
+ const geo = await api.screenIdentity(target, { options, confirmNovel: false }).catch(() => null);
705
+ const cropped = parsed ? await cropRegion(png, parsed, geo?.points) : { note: null };
706
+ if (cropped.png) {
707
+ png = cropped.png;
708
+ lines.push(cropped.note);
709
+ } else if (cropped.note) {
710
+ lines.push(`could not crop that region (${cropped.note}) — this is the whole screen`);
711
+ }
712
+ }
713
+ return { content: [text(lines.filter(Boolean).join('\n')), image(png)] };
478
714
  }
479
715
 
480
716
  async function state(target, args, options) {
@@ -635,10 +871,38 @@ async function recall(target, args, options) {
635
871
  * is free. Reading the screen a second time to render it was the whole cost of
636
872
  * a text-first surface, and it does not have to be paid.
637
873
  */
874
+ /**
875
+ * The map at the end of an action, and why it is re-read rather than recalled.
876
+ *
877
+ * It used to be rendered from whatever `screenIdentity` had in hand during
878
+ * verification, which is memory-first by design — so the trailing map could
879
+ * describe the screen as it was seconds earlier. Reported from a real session:
880
+ * *"do's trailing dump is still stale, so I still ran `ui --refresh` after
881
+ * nearly every call — that remains the biggest speed tax."*
882
+ *
883
+ * Saying how old it was (the header does) turned out not to be enough: an agent
884
+ * that cannot trust the map spends a turn re-reading it, and a turn is the
885
+ * expensive unit here. So an action pays one perception pass — a few hundred
886
+ * milliseconds, locally, once — to save a model round trip. That is the whole
887
+ * trade this phase is about, and it is the right way round.
888
+ *
889
+ * `refresh: false` is still available for the read-only tools, where the caller
890
+ * asked for a map and can ask again.
891
+ */
638
892
  async function mapFrom(target, options, identity, extra = {}) {
639
893
  try {
640
- const m = await view.screenMap(target, { options, identity: identity?.entry ? identity : undefined });
641
- return view.render({ ...m, ...extra });
894
+ const fresh = extra.refresh !== false;
895
+ const m = await view.screenMap(target, {
896
+ options,
897
+ refresh: fresh,
898
+ identity: fresh ? undefined : (identity?.entry ? identity : undefined),
899
+ });
900
+ const rendered = view.render({ ...m, ...extra });
901
+ // The line that decides whether the model stops to think. Everything it
902
+ // needs is already computed for the map above it, so this costs nothing —
903
+ // and "nothing here needs you" is a thing only the daemon can say.
904
+ if (extra.hint === false) return rendered;
905
+ return `${rendered}\n${view.hintFor(m, { flowOk: extra.flowOk, escalated: extra.escalated })}`;
642
906
  } catch (err) {
643
907
  return `(could not read the screen: ${err.message})`;
644
908
  }
@@ -679,10 +943,29 @@ async function doScript(target, args, options) {
679
943
  stableMs: args.stableMs,
680
944
  timeoutMs: args.timeoutMs,
681
945
  continueOnError: args.continueOnError,
946
+ // The plan's briefing for its own first responder. Ignored when no
947
+ // supervisor is enabled, so passing it is always safe.
948
+ supervise: args.supervise,
682
949
  options,
683
950
  });
684
951
 
685
952
  const lines = stepLines(res);
953
+ // A map read without settling is not a reading of the screen you are now on,
954
+ // and it used to arrive looking exactly like one. Reported: two taps under
955
+ // `autoSettle:false` returned a map showing nothing had happened, so the
956
+ // agent moved on — the page had in fact zoomed all the way out, and they only
957
+ // found out two calls later when an unrelated failure printed a real map.
958
+ if (args.autoSettle === false) {
959
+ lines.push('autoSettle was off, so the map below was read without waiting for the last action to finish'
960
+ + ' — it may describe the screen before that action landed. Re-read (sim_ui) before acting on it.');
961
+ }
962
+ // Every local ruling is reported, because a wrong one has to be correctable
963
+ // rather than mysterious — and the model's own stated reason is shown as its
964
+ // claim, not as the ground for what happened.
965
+ for (const s_ of res.supervisions ?? []) {
966
+ lines.push(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
967
+ + (s_.reason ? ` (it said: "${s_.reason}")` : ''));
968
+ }
686
969
  if (args.saveAs) {
687
970
  const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
688
971
  lines.push(
@@ -691,7 +974,12 @@ async function doScript(target, args, options) {
691
974
  : `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
692
975
  );
693
976
  }
694
- lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
977
+ const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
978
+ lines.push('', await mapFrom(target, options, res.endScreen, {
979
+ verdictLine: verdictLineFor(res.results),
980
+ flowOk: res.ok,
981
+ escalated,
982
+ }));
695
983
 
696
984
  const content = [text(lines.join('\n'))];
697
985
  // Images only when explicitly asked for. A frame attached to every flow was
@@ -785,7 +1073,9 @@ async function ui(target, args, options) {
785
1073
  refresh: args.refresh,
786
1074
  });
787
1075
  if (m.identity.state) remember(m.device.udid, m.identity.state);
788
- return { content: [text(m.text)] };
1076
+ // The hint belongs here too: the agent has just looked, so "you do not need
1077
+ // to look again" is exactly the thing worth saying at this moment.
1078
+ return { content: [text(`${m.text}\n${view.hintFor(m)}`)] };
789
1079
  }
790
1080
 
791
1081
  async function find(target, args, options) {
@@ -840,7 +1130,9 @@ function listStateDirs() {
840
1130
  async function devices() {
841
1131
  const booted = await bootedDevices();
842
1132
  if (!booted.length) return { content: [text('no booted devices')] };
843
- return {
844
- content: [text(booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n'))],
845
- };
1133
+ // Noticing a name collision here is what lets every later header disambiguate
1134
+ // itself, and it costs nothing: this listing is already being made.
1135
+ const clash = noteBooted(booted);
1136
+ const list = booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1137
+ return { content: [text(clash ? `${clash}\n\n${list}` : list)] };
846
1138
  }