simframe 0.5.0 → 0.6.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/mcp.js CHANGED
@@ -13,8 +13,10 @@ import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
13
13
  import * as actions from './actions.js';
14
14
  import * as api from './index.js';
15
15
  import * as input from './input.js';
16
- import { bootedDevices } from './simctl.js';
16
+ import * as navigate from './navigate.js';
17
+ import { bootedDevices, PERMISSION_SERVICES } from './simctl.js';
17
18
  import * as store from './store.js';
19
+ import * as view from './view.js';
18
20
 
19
21
  const deviceProp = {
20
22
  device: {
@@ -23,149 +25,267 @@ const deviceProp = {
23
25
  },
24
26
  };
25
27
 
28
+ /**
29
+ * One selector grammar everywhere.
30
+ *
31
+ * `#3` is the cheapest thing a caller can say and the least ambiguous, because
32
+ * simframe numbered it; a bare phrase is resolved by intent, which is more
33
+ * forgiving and occasionally has to ask which one was meant.
34
+ */
35
+ const SELECTOR = 'Selector: "#3" (a number from the last screen map — cheapest and unambiguous), a label or phrase like "Save" or "the Assets tab" (resolved by intent), or "@120,400" for raw point coordinates.';
36
+
37
+ const selectorProp = (what = 'What to act on') => ({
38
+ sel: { type: 'string', description: `${what}. ${SELECTOR}` },
39
+ });
40
+
26
41
  const TOOLS = [
27
42
  {
28
- name: 'sim_look',
43
+ name: 'sim_ui',
29
44
  description:
30
- 'Look at the iOS Simulator screen right now. Returns the newest buffered frame immediately — a background capture loop keeps it warm, so there is no screenshot wait. Use this instead of taking a screenshot. Prefer detail "low" for layout checks and "high" only when you must read small text.',
45
+ 'READ THE SCREEN. Returns a compact text map: every element with a number, its region (nav-bar / content / tab-bar), type, label, state and tap point, plus which screen this is and how much simframe already knows about it. Roughly a tenth the cost of a screenshot and strictly more useful, because it says what is tappable and where — no measuring pixels by eye. The numbers are selectors: whatever this returns as #3, you can tap as "#3". Start here, not with sim_look.',
31
46
  inputSchema: {
32
47
  type: 'object',
33
48
  properties: {
34
49
  ...deviceProp,
35
- detail: {
36
- type: 'string',
37
- enum: ['low', 'normal', 'high', 'full'],
38
- description:
39
- 'Image size: low (~420px, cheapest), normal (~700px, default), high (~1100px, readable small text), full (native resolution).',
40
- },
41
- maxAgeMs: {
42
- type: 'number',
43
- description:
44
- 'If the buffered frame is older than this, wait for a fresher one (default 900).',
45
- },
50
+ filter: { type: 'string', description: 'Only elements whose label or read text contains this.' },
51
+ interactive: { type: 'boolean', description: 'Only elements that look tappable.' },
52
+ all: { type: 'boolean', description: 'Include the status bar and every collapsed region (default false).' },
53
+ refresh: { type: 'boolean', description: 'Re-read the screen instead of using remembered layout. Use when you believe memory is stale.' },
46
54
  },
47
55
  },
48
56
  },
49
57
  {
50
- name: 'sim_state',
58
+ name: 'sim_do',
51
59
  description:
52
- 'Cheap TEXT-ONLY check of the simulator screen: a stable screen hash, whether anything has changed SINCE YOUR LAST LOOK in this session, how long the screen has been still, and an ASCII map of which regions moved. Costs a tiny fraction of an image. Use it to poll ("has it finished loading?", "did my tap register?") and call sim_look only when you need to see pixels. The comparison is against the last frame you observed through any simframe tool, so calling this before and after an action is the reliable way to tell whether the action did anything.',
60
+ 'THE MAIN TOOL. Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each step waits for the screen to settle against a baseline captured before it, and is verified against what that action did here last time — so a step that navigated somewhere unintended stops the flow instead of tapping on into the wrong screen. A twelve-step flow costs one round trip instead of twelve. Prefer this over the single-action tools whenever you know more than one step ahead, and put asserts in the flow rather than checking between calls. Returns the compact screen map of where the flow ended; no image.',
53
61
  inputSchema: {
54
62
  type: 'object',
55
63
  properties: {
56
64
  ...deviceProp,
57
- since: {
58
- type: 'string',
65
+ steps: {
66
+ type: 'array',
59
67
  description:
60
- 'Compare against this specific frame hash instead of your last look. Defaults to your previous observation in this session.',
68
+ 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}.',
69
+ items: { type: 'object' },
70
+ },
71
+ autoSettle: {
72
+ type: 'boolean',
73
+ description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
74
+ },
75
+ stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
76
+ timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
77
+ continueOnError: { type: 'boolean', description: 'Keep going after a failed step (default false).' },
78
+ saveAs: {
79
+ type: 'string',
80
+ description: 'If every step verifies, save the flow under this name so it can be replayed with sim_flow_run.',
81
+ },
82
+ images: {
83
+ type: 'boolean',
84
+ description: 'Attach a frame for every {"look"} step (default false — the text map is normally what you want).',
61
85
  },
62
86
  },
87
+ required: ['steps'],
63
88
  },
64
89
  },
65
90
  {
66
- name: 'sim_wait',
67
- description:
68
- 'Block until the screen finishes reacting, then return the frame. Use after a tap, launch or navigation instead of sleeping and screenshotting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly if you call it in the moment before an animation starts. The baseline is whatever you last observed in this session, so the normal pattern is: call sim_state or sim_look, act, then call sim_wait. If the change already completed before you call, that is detected rather than waited out.',
91
+ name: 'sim_tap',
92
+ description: 'Tap one thing. For more than one step, use sim_do — it batches the verification and costs one round trip. Returns the screen map afterwards.',
93
+ inputSchema: {
94
+ type: 'object',
95
+ properties: { ...deviceProp, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
96
+ required: ['sel'],
97
+ },
98
+ },
99
+ {
100
+ name: 'sim_type_into',
101
+ description: 'Focus a field and type into it. Prefer a sim_do step when this is part of a sequence.',
102
+ inputSchema: {
103
+ type: 'object',
104
+ properties: { ...deviceProp, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
105
+ required: ['sel', 'text'],
106
+ },
107
+ },
108
+ {
109
+ name: 'sim_scroll_to',
110
+ description: 'Scroll until something is in view. A control that scrolled off the bottom of a list is not missing, and this is the difference.',
69
111
  inputSchema: {
70
112
  type: 'object',
71
113
  properties: {
72
114
  ...deviceProp,
73
- mode: {
74
- type: 'string',
75
- enum: ['settle', 'change', 'stable'],
76
- description:
77
- 'settle (default): wait for a change, then for it to hold still. change: return as soon as it differs from the baseline. stable: return once it is still, even if nothing ever changed.',
78
- },
79
- since: {
115
+ ...selectorProp('What to bring into view'),
116
+ direction: { type: 'string', enum: ['down', 'up', 'left', 'right'] },
117
+ maxScrolls: { type: 'number', description: 'Give up after this many screens (default 6).' },
118
+ },
119
+ required: ['sel'],
120
+ },
121
+ },
122
+ {
123
+ name: 'sim_wait_for',
124
+ description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
125
+ inputSchema: {
126
+ type: 'object',
127
+ properties: { ...deviceProp, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
128
+ required: ['sel'],
129
+ },
130
+ },
131
+ {
132
+ name: 'sim_assert',
133
+ description: 'Check one thing about the screen and fail loudly if it is not so. Cheaper inside a sim_do flow, where a failed assert stops the remaining steps.',
134
+ inputSchema: {
135
+ type: 'object',
136
+ properties: {
137
+ ...deviceProp,
138
+ ...selectorProp('What to check'),
139
+ is: {
80
140
  type: 'string',
81
- description:
82
- 'Frame hash to treat as the "before" state. Defaults to your last observation in this session. Pass this when you captured a hash before acting.',
141
+ enum: ['visible', 'gone', 'enabled', 'disabled', 'value'],
142
+ description: 'Default visible. "value" compares against `equals`.',
83
143
  },
84
- stableMs: { type: 'number', description: 'How long the screen must hold still for mode "stable" (default 600).' },
85
- timeoutMs: { type: 'number', description: 'Give up after this long (default 8000).' },
86
- includeImage: { type: 'boolean', description: 'Attach the resulting frame as an image (default true).' },
87
- detail: { type: 'string', enum: ['low', 'normal', 'high', 'full'] },
144
+ equals: { type: 'string', description: 'For is="value": the text the element should read.' },
88
145
  },
146
+ required: ['sel'],
89
147
  },
90
148
  },
91
149
  {
92
- name: 'sim_strip',
150
+ name: 'sim_goto',
93
151
  description:
94
- 'Return the last few buffered frames tiled into ONE image, left to right, oldest first. Lets you understand a transition, animation or flicker in a single call instead of a burst of screenshots. Frames are already buffered, so this looks backwards in time — it does not wait.',
152
+ 'Walk to a screen simframe has already been to, by name, planning the route through remembered transitions and verifying every step. Zero reasoning and zero images: the graph knows which taps lead where. Refuses rather than guesses — if the destination is unknown, ambiguous, or unreachable through known transitions, it says which and lists what it does know. Call it with no target to see the known screens.',
153
+ inputSchema: {
154
+ type: 'object',
155
+ properties: { ...deviceProp, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
156
+ },
157
+ },
158
+ {
159
+ name: 'sim_flow_run',
160
+ description: 'Replay a saved flow by name, verifying each step. Omit `name` to list the saved flows. Save one with sim_do\'s `saveAs`.',
161
+ inputSchema: {
162
+ type: 'object',
163
+ properties: { ...deviceProp, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
164
+ },
165
+ },
166
+ {
167
+ name: 'sim_launch',
168
+ description: 'Launch or relaunch an app, optionally with launch arguments and environment variables — the way to put an app into a test mode without touching its UI.',
95
169
  inputSchema: {
96
170
  type: 'object',
97
171
  properties: {
98
172
  ...deviceProp,
99
- count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
100
- spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
101
- thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
173
+ bundleId: { type: 'string' },
174
+ relaunch: { type: 'boolean', description: 'Terminate first. Without this, launching an already-running app silently does nothing and you test the screen you were already on.' },
175
+ args: { type: 'array', items: { type: 'string' }, description: 'Launch arguments passed to the app.' },
176
+ env: { type: 'object', description: 'Environment variables for the app process.' },
102
177
  },
178
+ required: ['bundleId'],
103
179
  },
104
180
  },
105
181
  {
106
- name: 'sim_recall',
182
+ name: 'sim_open_url',
183
+ description: 'Open a URL or deep link on the simulator — the fastest way to reach a screen when the app has a link for it.',
184
+ inputSchema: {
185
+ type: 'object',
186
+ properties: { ...deviceProp, url: { type: 'string' } },
187
+ required: ['url'],
188
+ },
189
+ },
190
+ {
191
+ name: 'sim_permission',
192
+ description: `Grant, revoke or reset a privacy permission for an app. Do this instead of tapping the system alert: the alert is not part of the app under test, and its buttons move between iOS versions. Services: ${PERMISSION_SERVICES.join(', ')}.`,
193
+ inputSchema: {
194
+ type: 'object',
195
+ properties: {
196
+ ...deviceProp,
197
+ action: { type: 'string', enum: ['grant', 'revoke', 'reset'] },
198
+ service: { type: 'string' },
199
+ bundleId: { type: 'string' },
200
+ },
201
+ required: ['action', 'service'],
202
+ },
203
+ },
204
+ {
205
+ name: 'sim_find',
206
+ description:
207
+ 'Resolve an intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, where on screen you meant, and icon-only controls by their common name. When two things answer equally well it says so and lists them rather than guessing — a wrong tap is worse than a question, because it can do something and leave you believing it did the right thing. Use it when you are unsure a selector will resolve; otherwise just tap.',
208
+ inputSchema: {
209
+ type: 'object',
210
+ properties: { ...deviceProp, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
211
+ required: ['intent'],
212
+ },
213
+ },
214
+ {
215
+ name: 'sim_state',
107
216
  description:
108
- 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen — every frame for the last 10s, thinned to about 2fps before that. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago it started, how long it took, how much of the screen it moved. action "at" returns the buffered frame from a moment in the past. Use this when you look up and find the screen already different, or when something flashed by and you need to know what it was — instead of guessing or re-running the action.',
217
+ 'Cheapest possible check: a stable screen hash, whether anything changed SINCE YOUR LAST LOOK in this session, how long the screen has been still, and an ASCII map of which regions moved. A fraction of the cost of even the text screen map. Use it to poll ("has it finished loading?"). To find out WHAT is on screen, use sim_ui.',
109
218
  inputSchema: {
110
219
  type: 'object',
111
220
  properties: {
112
221
  ...deviceProp,
113
- action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
114
- spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
115
- msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
222
+ since: { type: 'string', description: 'Compare against this frame hash instead of your last look.' },
116
223
  },
117
224
  },
118
225
  },
119
226
  {
120
- name: 'sim_do',
227
+ name: 'sim_wait',
121
228
  description:
122
- 'Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each action automatically waits for the screen to settle before the next step, using a baseline captured before that action, so steps do not race the UI. This is the fastest way to drive the simulator — a twelve-step flow costs one round trip instead of twelve. Prefer it over single taps whenever you know more than one step ahead. Steps stop at the first failure and the result says exactly which step failed and why. Requires idb for input; observation-only steps work without it.',
229
+ 'Block until the screen finishes reacting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly in the moment before an animation starts. Returns the compact screen map by default, not an image. Inside a flow you rarely need this: sim_do settles after every step already.',
123
230
  inputSchema: {
124
231
  type: 'object',
125
232
  properties: {
126
233
  ...deviceProp,
127
- steps: {
128
- type: 'array',
129
- description:
130
- 'Ordered steps. Shorthand forms: {"tap":"Save"} (by accessibility label; add "index" if ambiguous), {"tapAt":{"x":100,"y":200,"space":"points"|"image"}}, {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":"com.example.app"}, {"openUrl":"myapp://x"}, {"waitText":"Saved","timeoutMs":5000}, {"assertText":"Saved"}, {"assertGone":"Spinner"}, {"settle":{"stableMs":600}}, {"look":{"detail":"low"}}, {"pause":300}.',
131
- items: { type: 'object' },
234
+ mode: {
235
+ type: 'string',
236
+ enum: ['settle', 'change', 'stable'],
237
+ description: 'settle (default): wait for a change, then for it to hold still. change: return as soon as it differs. stable: return once it is still, even if nothing moved.',
132
238
  },
133
- autoSettle: {
134
- type: 'boolean',
135
- description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
239
+ since: { type: 'string', description: 'Frame hash to treat as the "before" state. Defaults to your last observation.' },
240
+ stableMs: { type: 'number', description: 'How long the screen must hold still for mode "stable" (default 600).' },
241
+ timeoutMs: { type: 'number', description: 'Give up after this long (default 8000).' },
242
+ map: { type: 'boolean', description: 'Return the screen map afterwards (default true).' },
243
+ },
244
+ },
245
+ },
246
+ {
247
+ name: 'sim_look',
248
+ description:
249
+ 'THE ONLY TOOL THAT RETURNS AN IMAGE, and the most expensive one. Returns the newest buffered frame immediately — no screenshot wait. Call it only when the text map is genuinely not enough: checking visual layout, colour, spacing, an animation, or something the accessibility tree and OCR both cannot see. For "what is on screen and what can I tap", sim_ui answers better and costs a tenth as much.',
250
+ inputSchema: {
251
+ type: 'object',
252
+ properties: {
253
+ ...deviceProp,
254
+ detail: {
255
+ type: 'string',
256
+ enum: ['low', 'normal', 'high'],
257
+ description: 'Image size: low (~420px, cheapest), normal (~700px, default), high (1024px, readable small text). Capped at 1024px on the long edge.',
136
258
  },
137
- stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
138
- timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
139
- continueOnError: { type: 'boolean', description: 'Keep going after a failed step (default false).' },
140
- finalLook: { type: 'boolean', description: 'Attach a frame of the end state (default true).' },
259
+ maxAgeMs: { type: 'number', description: 'If the buffered frame is older than this, wait for a fresher one (default 900).' },
141
260
  },
142
- required: ['steps'],
143
261
  },
144
262
  },
145
263
  {
146
- name: 'sim_ui',
264
+ name: 'sim_strip',
147
265
  description:
148
- 'Read the screen as an accessibility tree instead of an image: every element with its label, value, type and position in points. Often cheaper AND more useful than a screenshot, because it tells you what is actually tappable and gives exact coordinates — no measuring pixels by eye. Use it before tapping something you are unsure about. Requires idb.',
266
+ 'Return the last few buffered frames tiled into ONE image, oldest first — an image, so not cheap. Lets you understand a transition, animation or flicker in a single call instead of a burst of screenshots. Looks backwards in time; it does not wait.',
149
267
  inputSchema: {
150
268
  type: 'object',
151
269
  properties: {
152
270
  ...deviceProp,
153
- filter: { type: 'string', description: 'Only elements whose label, value or identifier contains this text.' },
154
- interactive: { type: 'boolean', description: 'Only elements that look tappable (buttons, fields, cells).' },
271
+ count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
272
+ spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
273
+ thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
155
274
  },
156
275
  },
157
276
  },
158
277
  {
159
- name: 'sim_find',
278
+ name: 'sim_recall',
160
279
  description:
161
- 'Resolve an intent to one control on screen: "tap Save", "the Assets tab", "back". Understands verbs, typos, where on screen you meant, and icon-only controls by their common name. Returns the element with its tap point and the reasons it was chosen. When two things answer equally well it says so and lists them rather than guessing — a wrong tap is worse than a question, because it can do something and leave you believing it did the right thing. Cheaper and more reliable than reading a screenshot to find a control.',
280
+ 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago, how long it took, how much of the screen moved. action "at" returns the buffered frame from a past moment — an image. Use this when you look up and find the screen already different, instead of re-running the action.',
162
281
  inputSchema: {
163
282
  type: 'object',
164
283
  properties: {
165
284
  ...deviceProp,
166
- intent: { type: 'string', description: 'What you want to act on, in your own words.' },
285
+ action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
286
+ spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
287
+ msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
167
288
  },
168
- required: ['intent'],
169
289
  },
170
290
  },
171
291
  {
@@ -273,6 +393,48 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
273
393
  return await doScript(target, args, options);
274
394
  case 'sim_ui':
275
395
  return await ui(target, args, options);
396
+ case 'sim_tap':
397
+ return await oneStep(target, { tap: args.sel, index: args.index }, args, options);
398
+ case 'sim_type_into':
399
+ return await oneStep(
400
+ target,
401
+ args.paste
402
+ ? { paste: { into: args.sel, text: args.text } }
403
+ : { type: { into: args.sel, text: args.text } },
404
+ args,
405
+ options,
406
+ );
407
+ case 'sim_scroll_to':
408
+ return await oneStep(
409
+ target,
410
+ { scrollTo: args.sel, direction: args.direction, maxScrolls: args.maxScrolls },
411
+ args,
412
+ options,
413
+ );
414
+ case 'sim_wait_for':
415
+ return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs }, args, options);
416
+ case 'sim_assert':
417
+ return await oneStep(target, { assert: args.sel, is: args.is, equals: args.equals }, args, options);
418
+ case 'sim_launch':
419
+ return await oneStep(
420
+ target,
421
+ { launch: { value: args.bundleId, relaunch: args.relaunch, args: args.args, env: args.env } },
422
+ args,
423
+ options,
424
+ );
425
+ case 'sim_open_url':
426
+ return await oneStep(target, { openUrl: args.url }, args, options);
427
+ case 'sim_permission':
428
+ return await oneStep(
429
+ target,
430
+ { permission: { value: args.service, grant: args.action, bundleId: args.bundleId } },
431
+ args,
432
+ options,
433
+ );
434
+ case 'sim_goto':
435
+ return await goto(target, args, options);
436
+ case 'sim_flow_run':
437
+ return await flowRun(target, args, options);
276
438
  case 'sim_capture':
277
439
  return await capture(target, args, options);
278
440
  case 'sim_devices':
@@ -290,7 +452,11 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
290
452
 
291
453
  async function look(target, args, options) {
292
454
  const maxAge = args.maxAgeMs ?? 900;
293
- let res = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
455
+ // Never native resolution. An image is the expensive path by a factor of ten
456
+ // even when Claude Code handles it correctly; a 1024px frame says everything
457
+ // a 393-point screen has to say.
458
+ const detail = api.modelDetail(args.detail ?? 'normal');
459
+ let res = await api.getFrame(target, { detail, options });
294
460
  if (res.ageMs > maxAge) {
295
461
  // Freshness was requested; the loop is already running, so just let it tick.
296
462
  const deadline = Date.now() + Math.min(2000, maxAge * 3);
@@ -300,7 +466,7 @@ async function look(target, args, options) {
300
466
  if (next && Date.now() - next.capturedAt <= maxAge) break;
301
467
  await new Promise((r) => setTimeout(r, 50));
302
468
  }
303
- res = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
469
+ res = await api.getFrame(target, { detail, options });
304
470
  }
305
471
  const prior = baselineFor(res.device.udid, undefined);
306
472
  const st = await api.getState(target, { since: prior, options });
@@ -367,19 +533,24 @@ async function wait(target, args, options) {
367
533
  }
368
534
  }
369
535
  remember(device.udid, res.state);
370
- const content = [text(`${note}\n${header(res.device, res.state, Date.now() - res.state.capturedAt)}`)];
371
- if (args.includeImage !== false) {
372
- const frame = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
373
- content.push(image(frame.png));
536
+ const lines = [note, header(res.device, res.state, Date.now() - res.state.capturedAt)];
537
+ // The screen map, not a frame. This call is almost always "what happened?",
538
+ // and the answer to that is a list of elements, not 1,600 tokens of pixels.
539
+ if (args.map !== false) {
540
+ try {
541
+ lines.push('', (await view.screenMap(target, { options })).text);
542
+ } catch (err) {
543
+ lines.push(`(could not read the screen: ${err.message})`);
544
+ }
374
545
  }
375
- return { content };
546
+ return { content: [text(lines.join('\n'))] };
376
547
  }
377
548
 
378
549
  async function strip(target, args, options) {
379
550
  const res = await api.getStrip(target, {
380
551
  count: Math.min(12, args.count ?? 5),
381
552
  spanMs: args.spanMs,
382
- thumbMaxDim: args.thumbMaxDim ?? 240,
553
+ thumbMaxDim: Math.min(api.MODEL_MAX_IMAGE_DIM, args.thumbMaxDim ?? 240),
383
554
  options,
384
555
  });
385
556
  const offsets = res.frames.map((f) => `+${f.offsetMs}ms`).join(' ');
@@ -403,6 +574,8 @@ function ago(ms) {
403
574
 
404
575
  async function recall(target, args, options) {
405
576
  if (args.action === 'at') {
577
+ // No cap needed: buffered ring frames are captured at 700px, already under
578
+ // the model image ceiling.
406
579
  const res = await api.getFrameAt(target, { msAgo: args.msAgo ?? 0, options });
407
580
  return {
408
581
  content: [
@@ -441,16 +614,34 @@ async function recall(target, args, options) {
441
614
  return { content: [text(lines.join('\n'))] };
442
615
  }
443
616
 
444
- async function doScript(target, args, options) {
445
- const res = await actions.runScript(target, {
446
- steps: args.steps,
447
- autoSettle: args.autoSettle,
448
- stableMs: args.stableMs,
449
- timeoutMs: args.timeoutMs,
450
- continueOnError: args.continueOnError,
451
- options,
452
- });
617
+ /**
618
+ * The compact map, from an identity reading that has already been taken.
619
+ *
620
+ * A flow verifies where it ended up in order to record the edge, and that
621
+ * reading carries the elements with it — so describing the end state to Claude
622
+ * is free. Reading the screen a second time to render it was the whole cost of
623
+ * a text-first surface, and it does not have to be paid.
624
+ */
625
+ async function mapFrom(target, options, identity, extra = {}) {
626
+ try {
627
+ const m = await view.screenMap(target, { options, identity: identity?.entry ? identity : undefined });
628
+ return view.render({ ...m, ...extra });
629
+ } catch (err) {
630
+ return `(could not read the screen: ${err.message})`;
631
+ }
632
+ }
453
633
 
634
+ function verdictLineFor(results) {
635
+ const last = [...(results ?? [])].reverse().find((r) => r.verification);
636
+ if (!last) return null;
637
+ const v = last.verification;
638
+ // `detail` already spells out the kind mismatch and the times-seen count, so
639
+ // adding the prediction and the kind again made this line say everything
640
+ // three times — in the one place where brevity is the entire feature.
641
+ return `last action: [${last.index}] ${last.action} — ${v.verdict}: ${v.detail}`;
642
+ }
643
+
644
+ function stepLines(res) {
454
645
  const lines = [
455
646
  `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps in ${res.totalMs}ms`,
456
647
  ];
@@ -465,45 +656,123 @@ async function doScript(target, args, options) {
465
656
  );
466
657
  }
467
658
  if (!res.ok) lines.push('later steps were not run; the screen is left wherever the failing step stopped');
659
+ return lines;
660
+ }
468
661
 
469
- const content = [text(lines.join('\n'))];
470
- for (const f of res.frames) content.push(image(f.png));
471
- if (args.finalLook !== false) {
472
- const frame = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
473
- remember(res.device.udid, frame.state);
474
- content.push(text(`end state — ${header(frame.device, frame.state, frame.ageMs)}`), image(frame.png));
662
+ async function doScript(target, args, options) {
663
+ const res = await actions.runScript(target, {
664
+ steps: args.steps,
665
+ autoSettle: args.autoSettle,
666
+ stableMs: args.stableMs,
667
+ timeoutMs: args.timeoutMs,
668
+ continueOnError: args.continueOnError,
669
+ options,
670
+ });
671
+
672
+ const lines = stepLines(res);
673
+ if (args.saveAs) {
674
+ const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
675
+ lines.push(
676
+ saved.ok
677
+ ? `saved as flow "${saved.name}" (${saved.steps} steps) — replay with sim_flow_run`
678
+ : `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
679
+ );
475
680
  }
681
+ lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
682
+
683
+ const content = [text(lines.join('\n'))];
684
+ // Images only when explicitly asked for. A frame attached to every flow was
685
+ // the largest single cost in this server, and it answered a question the map
686
+ // answers better.
687
+ if (args.images) for (const f of res.frames) content.push(image(f.png));
476
688
  return { content, isError: !res.ok };
477
689
  }
478
690
 
479
- async function ui(target, args, options) {
691
+ /** Every single-action tool is one sim_do step, so verification works the same way. */
692
+ async function oneStep(target, step, args, options) {
693
+ const clean = Object.fromEntries(Object.entries(step).filter(([, v]) => v !== undefined));
694
+ return await doScript(target, { ...args, steps: [clean], saveAs: undefined }, options);
695
+ }
696
+
697
+ async function goto(target, args, options) {
480
698
  const { device } = await api.ensureDaemon(target, options);
481
- const driver = await input.detectDriver();
482
- if (!driver.available) return { isError: true, content: [text(driver.reason)] };
483
-
484
- let nodes = await input.describeAll(device.udid);
485
- if (args.filter) {
486
- const q = String(args.filter).toLowerCase();
487
- nodes = nodes.filter((n) =>
488
- [n.label, n.value, n.identifier].filter(Boolean).join(' ').toLowerCase().includes(q),
489
- );
699
+ if (!args.screen) {
700
+ const known = navigate.knownScreens(device.udid);
701
+ if (!known.length) {
702
+ return {
703
+ content: [text('no screens remembered on this device yet — drive a flow with sim_do and simframe learns them as it goes')],
704
+ };
705
+ }
706
+ return {
707
+ content: [
708
+ text(
709
+ `${known.length} screen(s) simframe can navigate to:\n` +
710
+ known.map((k) => ` ${k.hash} ${k.edges} exit(s) ${k.name}`).join('\n'),
711
+ ),
712
+ ],
713
+ };
490
714
  }
491
- if (args.interactive) {
492
- nodes = nodes.filter((n) => /button|field|cell|link|switch|slider|tab|menu/i.test(n.type || ''));
715
+
716
+ const res = await navigate.goto(target, args.screen, { options });
717
+ if (!res.ok && res.reason) {
718
+ const why = {
719
+ 'unknown-screen': `no remembered screen matches "${args.screen}". Known: ${(res.known ?? []).map((k) => k.name).join(', ') || 'none'}`,
720
+ ambiguous: `"${args.screen}" fits ${(res.candidates ?? []).length} screens equally: ${(res.candidates ?? []).map((c) => c.name).join(', ')} — say which`,
721
+ 'no-route': `"${res.to}" is known, but no remembered path reaches it from where you are (${res.from})`,
722
+ 'unreplayable-edge': `the route to "${res.to}" includes a step simframe cannot replay exactly`,
723
+ }[res.reason] ?? res.reason;
724
+ return { isError: true, content: [text(`sim_goto refused rather than guess: ${why}`)] };
493
725
  }
494
- if (!nodes.length) return { content: [text('no matching elements on screen')] };
495
-
496
- const rows = nodes.slice(0, 200).map((n) => {
497
- const c = input.centerOf(n);
498
- const name = [n.label, n.value && `= ${n.value}`, n.identifier && `#${n.identifier}`]
499
- .filter(Boolean)
500
- .join(' ');
501
- const fallback = n.rawLabel ? '(icon-only — tap by these coordinates)' : '(unlabelled)';
502
- return ` ${(n.type || '?').padEnd(14)} ${String(`${c.x},${c.y}`).padEnd(10)} ${name || fallback}`;
726
+
727
+ const lines = res.already
728
+ ? [`already on "${res.screen}"`]
729
+ : [
730
+ `${res.ok ? 'arrived at' : 'DID NOT REACH'} "${res.screen}" in ${res.ranSteps}/${res.steps.length} remembered steps`,
731
+ ...(res.results ?? []).map((r) => ` ${r.ok ? 'ok ' : 'FAIL'} [${r.index}] ${r.action}: ${r.ok ? r.detail : r.error}`),
732
+ ];
733
+ lines.push('', await mapFrom(target, options, null));
734
+ return { content: [text(lines.join('\n'))], isError: !res.ok };
735
+ }
736
+
737
+ async function flowRun(target, args, options) {
738
+ const { device } = await api.ensureDaemon(target, options);
739
+ if (!args.name) {
740
+ const flows = navigate.listFlows(device.udid);
741
+ return {
742
+ content: [
743
+ text(
744
+ flows.length
745
+ ? `saved flows:\n${flows.map((f) => ` ${f.name} (${f.steps} steps)`).join('\n')}`
746
+ : 'no saved flows — run one with sim_do and pass saveAs',
747
+ ),
748
+ ],
749
+ };
750
+ }
751
+ const res = await navigate.runFlow(target, args.name, { options });
752
+ if (res.reason === 'unknown-flow') {
753
+ return {
754
+ isError: true,
755
+ content: [text(`no flow called "${args.name}". Saved: ${(res.known ?? []).join(', ') || 'none'}`)],
756
+ };
757
+ }
758
+ const lines = [`flow "${args.name}"`, ...stepLines(res)];
759
+ lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
760
+ return { content: [text(lines.join('\n'))], isError: !res.ok };
761
+ }
762
+
763
+ async function ui(target, args, options) {
764
+ // No accessibility gate here any more. The old sim_ui asked the tree
765
+ // directly, so a machine without idb could capture and tap perfectly well
766
+ // and still not read the screen — even though OCR alone answers most of them.
767
+ const m = await view.screenMap(target, {
768
+ options,
769
+ filter: args.filter,
770
+ interactive: args.interactive,
771
+ all: args.all,
772
+ refresh: args.refresh,
503
773
  });
504
- const head = `${device.name} — ${nodes.length} element${nodes.length === 1 ? '' : 's'} (type, tap point in points, label)`;
505
- const tail = nodes.length > 200 ? `\n ... ${nodes.length - 200} more; use filter to narrow` : '';
506
- return { content: [text(`${head}\n${rows.join('\n')}${tail}`)] };
774
+ if (m.identity.state) remember(m.device.udid, m.identity.state);
775
+ return { content: [text(m.text)] };
507
776
  }
508
777
 
509
778
  async function find(target, args, options) {