simframe 0.4.2 → 0.6.0-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +334 -85
  2. package/native/simframed/Package.swift +16 -0
  3. package/native/simframed/Sources/PrivateAPI/AccessibilityBridge.swift +379 -0
  4. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +523 -0
  5. package/native/simframed/Sources/PrivateAPI/HIDKeyboard.swift +70 -0
  6. package/native/simframed/Sources/PrivateAPI/IndigoHID.swift +121 -0
  7. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +149 -0
  8. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +112 -0
  9. package/native/simframed/Sources/SimframeCore/Bitmap.swift +61 -0
  10. package/native/simframed/Sources/SimframeCore/ControlSocket.swift +122 -0
  11. package/native/simframed/Sources/SimframeCore/CoreGraphicsScaler.swift +70 -0
  12. package/native/simframed/Sources/SimframeCore/Element.swift +148 -0
  13. package/native/simframed/Sources/SimframeCore/FrameStore.swift +303 -0
  14. package/native/simframed/Sources/SimframeCore/Hashing.swift +119 -0
  15. package/native/simframed/Sources/SimframeCore/Motion.swift +431 -0
  16. package/native/simframed/Sources/SimframeCore/PNGWriter.swift +40 -0
  17. package/native/simframed/Sources/SimframeCore/VisionOCR.swift +75 -0
  18. package/native/simframed/Sources/simframed/main.swift +485 -0
  19. package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +270 -0
  20. package/package.json +12 -4
  21. package/scripts/bench-flow.mjs +54 -0
  22. package/scripts/bench.sh +98 -0
  23. package/scripts/check-package.mjs +99 -0
  24. package/scripts/ci-memory.mjs +416 -0
  25. package/scripts/eval-fingerprint.mjs +192 -0
  26. package/scripts/smoke.mjs +76 -0
  27. package/scripts/sync-server-version.mjs +39 -0
  28. package/scripts/verify-baseline.mjs +65 -0
  29. package/skills/simframe/SKILL.md +173 -0
  30. package/src/actions.js +264 -18
  31. package/src/cli.js +561 -89
  32. package/src/control.js +77 -0
  33. package/src/daemon.js +8 -1
  34. package/src/engine.js +99 -0
  35. package/src/fingerprint.js +183 -0
  36. package/src/graph.js +411 -0
  37. package/src/index.js +351 -24
  38. package/src/input.js +179 -2
  39. package/src/matching.js +265 -0
  40. package/src/mcp.js +425 -112
  41. package/src/navigate.js +120 -0
  42. package/src/refs.js +141 -0
  43. package/src/regions.js +267 -0
  44. package/src/screenmap.js +119 -22
  45. package/src/simctl.js +74 -5
  46. package/src/store.js +8 -0
  47. package/src/view.js +342 -0
package/src/mcp.js CHANGED
@@ -7,12 +7,16 @@ import {
7
7
  ListToolsRequestSchema,
8
8
  } from '@modelcontextprotocol/sdk/types.js';
9
9
  import fs from 'node:fs';
10
+ import path from 'node:path';
11
+ import { fileURLToPath } from 'node:url';
10
12
  import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
11
13
  import * as actions from './actions.js';
12
14
  import * as api from './index.js';
13
15
  import * as input from './input.js';
14
- import { bootedDevices } from './simctl.js';
16
+ import * as navigate from './navigate.js';
17
+ import { bootedDevices, PERMISSION_SERVICES } from './simctl.js';
15
18
  import * as store from './store.js';
19
+ import * as view from './view.js';
16
20
 
17
21
  const deviceProp = {
18
22
  device: {
@@ -21,135 +25,266 @@ const deviceProp = {
21
25
  },
22
26
  };
23
27
 
28
+ /**
29
+ * One selector grammar everywhere.
30
+ *
31
+ * `#3` is the cheapest thing a caller can say and the least ambiguous, because
32
+ * simframe numbered it; a bare phrase is resolved by intent, which is more
33
+ * forgiving and occasionally has to ask which one was meant.
34
+ */
35
+ const SELECTOR = 'Selector: "#3" (a number from the last screen map — cheapest and unambiguous), a label or phrase like "Save" or "the Assets tab" (resolved by intent), or "@120,400" for raw point coordinates.';
36
+
37
+ const selectorProp = (what = 'What to act on') => ({
38
+ sel: { type: 'string', description: `${what}. ${SELECTOR}` },
39
+ });
40
+
24
41
  const TOOLS = [
25
42
  {
26
- name: 'sim_look',
43
+ name: 'sim_ui',
27
44
  description:
28
- 'Look at the iOS Simulator screen right now. Returns the newest buffered frame immediately — a background capture loop keeps it warm, so there is no screenshot wait. Use this instead of taking a screenshot. Prefer detail "low" for layout checks and "high" only when you must read small text.',
45
+ 'READ THE SCREEN. Returns a compact text map: every element with a number, its region (nav-bar / content / tab-bar), type, label, state and tap point, plus which screen this is and how much simframe already knows about it. Roughly a tenth the cost of a screenshot and strictly more useful, because it says what is tappable and where — no measuring pixels by eye. The numbers are selectors: whatever this returns as #3, you can tap as "#3". Start here, not with sim_look.',
29
46
  inputSchema: {
30
47
  type: 'object',
31
48
  properties: {
32
49
  ...deviceProp,
33
- detail: {
34
- type: 'string',
35
- enum: ['low', 'normal', 'high', 'full'],
36
- description:
37
- 'Image size: low (~420px, cheapest), normal (~700px, default), high (~1100px, readable small text), full (native resolution).',
38
- },
39
- maxAgeMs: {
40
- type: 'number',
41
- description:
42
- 'If the buffered frame is older than this, wait for a fresher one (default 900).',
43
- },
50
+ filter: { type: 'string', description: 'Only elements whose label or read text contains this.' },
51
+ interactive: { type: 'boolean', description: 'Only elements that look tappable.' },
52
+ all: { type: 'boolean', description: 'Include the status bar and every collapsed region (default false).' },
53
+ refresh: { type: 'boolean', description: 'Re-read the screen instead of using remembered layout. Use when you believe memory is stale.' },
44
54
  },
45
55
  },
46
56
  },
47
57
  {
48
- name: 'sim_state',
58
+ name: 'sim_do',
49
59
  description:
50
- 'Cheap TEXT-ONLY check of the simulator screen: a stable screen hash, whether anything has changed SINCE YOUR LAST LOOK in this session, how long the screen has been still, and an ASCII map of which regions moved. Costs a tiny fraction of an image. Use it to poll ("has it finished loading?", "did my tap register?") and call sim_look only when you need to see pixels. The comparison is against the last frame you observed through any simframe tool, so calling this before and after an action is the reliable way to tell whether the action did anything.',
60
+ 'THE MAIN TOOL. Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each step waits for the screen to settle against a baseline captured before it, and is verified against what that action did here last time — so a step that navigated somewhere unintended stops the flow instead of tapping on into the wrong screen. A twelve-step flow costs one round trip instead of twelve. Prefer this over the single-action tools whenever you know more than one step ahead, and put asserts in the flow rather than checking between calls. Returns the compact screen map of where the flow ended; no image.',
51
61
  inputSchema: {
52
62
  type: 'object',
53
63
  properties: {
54
64
  ...deviceProp,
55
- since: {
56
- type: 'string',
65
+ steps: {
66
+ type: 'array',
57
67
  description:
58
- 'Compare against this specific frame hash instead of your last look. Defaults to your previous observation in this session.',
68
+ 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}.',
69
+ items: { type: 'object' },
70
+ },
71
+ autoSettle: {
72
+ type: 'boolean',
73
+ description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
74
+ },
75
+ stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
76
+ timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
77
+ continueOnError: { type: 'boolean', description: 'Keep going after a failed step (default false).' },
78
+ saveAs: {
79
+ type: 'string',
80
+ description: 'If every step verifies, save the flow under this name so it can be replayed with sim_flow_run.',
81
+ },
82
+ images: {
83
+ type: 'boolean',
84
+ description: 'Attach a frame for every {"look"} step (default false — the text map is normally what you want).',
59
85
  },
60
86
  },
87
+ required: ['steps'],
61
88
  },
62
89
  },
63
90
  {
64
- name: 'sim_wait',
65
- description:
66
- 'Block until the screen finishes reacting, then return the frame. Use after a tap, launch or navigation instead of sleeping and screenshotting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly if you call it in the moment before an animation starts. The baseline is whatever you last observed in this session, so the normal pattern is: call sim_state or sim_look, act, then call sim_wait. If the change already completed before you call, that is detected rather than waited out.',
91
+ name: 'sim_tap',
92
+ description: 'Tap one thing. For more than one step, use sim_do — it batches the verification and costs one round trip. Returns the screen map afterwards.',
93
+ inputSchema: {
94
+ type: 'object',
95
+ properties: { ...deviceProp, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
96
+ required: ['sel'],
97
+ },
98
+ },
99
+ {
100
+ name: 'sim_type_into',
101
+ description: 'Focus a field and type into it. Prefer a sim_do step when this is part of a sequence.',
102
+ inputSchema: {
103
+ type: 'object',
104
+ properties: { ...deviceProp, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
105
+ required: ['sel', 'text'],
106
+ },
107
+ },
108
+ {
109
+ name: 'sim_scroll_to',
110
+ description: 'Scroll until something is in view. A control that scrolled off the bottom of a list is not missing, and this is the difference.',
67
111
  inputSchema: {
68
112
  type: 'object',
69
113
  properties: {
70
114
  ...deviceProp,
71
- mode: {
72
- type: 'string',
73
- enum: ['settle', 'change', 'stable'],
74
- description:
75
- 'settle (default): wait for a change, then for it to hold still. change: return as soon as it differs from the baseline. stable: return once it is still, even if nothing ever changed.',
76
- },
77
- since: {
115
+ ...selectorProp('What to bring into view'),
116
+ direction: { type: 'string', enum: ['down', 'up', 'left', 'right'] },
117
+ maxScrolls: { type: 'number', description: 'Give up after this many screens (default 6).' },
118
+ },
119
+ required: ['sel'],
120
+ },
121
+ },
122
+ {
123
+ name: 'sim_wait_for',
124
+ description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
125
+ inputSchema: {
126
+ type: 'object',
127
+ properties: { ...deviceProp, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
128
+ required: ['sel'],
129
+ },
130
+ },
131
+ {
132
+ name: 'sim_assert',
133
+ description: 'Check one thing about the screen and fail loudly if it is not so. Cheaper inside a sim_do flow, where a failed assert stops the remaining steps.',
134
+ inputSchema: {
135
+ type: 'object',
136
+ properties: {
137
+ ...deviceProp,
138
+ ...selectorProp('What to check'),
139
+ is: {
78
140
  type: 'string',
79
- description:
80
- 'Frame hash to treat as the "before" state. Defaults to your last observation in this session. Pass this when you captured a hash before acting.',
141
+ enum: ['visible', 'gone', 'enabled', 'disabled', 'value'],
142
+ description: 'Default visible. "value" compares against `equals`.',
81
143
  },
82
- stableMs: { type: 'number', description: 'How long the screen must hold still for mode "stable" (default 600).' },
83
- timeoutMs: { type: 'number', description: 'Give up after this long (default 8000).' },
84
- includeImage: { type: 'boolean', description: 'Attach the resulting frame as an image (default true).' },
85
- detail: { type: 'string', enum: ['low', 'normal', 'high', 'full'] },
144
+ equals: { type: 'string', description: 'For is="value": the text the element should read.' },
86
145
  },
146
+ required: ['sel'],
87
147
  },
88
148
  },
89
149
  {
90
- name: 'sim_strip',
150
+ name: 'sim_goto',
91
151
  description:
92
- 'Return the last few buffered frames tiled into ONE image, left to right, oldest first. Lets you understand a transition, animation or flicker in a single call instead of a burst of screenshots. Frames are already buffered, so this looks backwards in time — it does not wait.',
152
+ 'Walk to a screen simframe has already been to, by name, planning the route through remembered transitions and verifying every step. Zero reasoning and zero images: the graph knows which taps lead where. Refuses rather than guesses — if the destination is unknown, ambiguous, or unreachable through known transitions, it says which and lists what it does know. Call it with no target to see the known screens.',
153
+ inputSchema: {
154
+ type: 'object',
155
+ properties: { ...deviceProp, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
156
+ },
157
+ },
158
+ {
159
+ name: 'sim_flow_run',
160
+ description: 'Replay a saved flow by name, verifying each step. Omit `name` to list the saved flows. Save one with sim_do\'s `saveAs`.',
161
+ inputSchema: {
162
+ type: 'object',
163
+ properties: { ...deviceProp, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
164
+ },
165
+ },
166
+ {
167
+ name: 'sim_launch',
168
+ description: 'Launch or relaunch an app, optionally with launch arguments and environment variables — the way to put an app into a test mode without touching its UI.',
93
169
  inputSchema: {
94
170
  type: 'object',
95
171
  properties: {
96
172
  ...deviceProp,
97
- count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
98
- spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
99
- thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
173
+ bundleId: { type: 'string' },
174
+ relaunch: { type: 'boolean', description: 'Terminate first. Without this, launching an already-running app silently does nothing and you test the screen you were already on.' },
175
+ args: { type: 'array', items: { type: 'string' }, description: 'Launch arguments passed to the app.' },
176
+ env: { type: 'object', description: 'Environment variables for the app process.' },
100
177
  },
178
+ required: ['bundleId'],
101
179
  },
102
180
  },
103
181
  {
104
- name: 'sim_recall',
182
+ name: 'sim_open_url',
183
+ description: 'Open a URL or deep link on the simulator — the fastest way to reach a screen when the app has a link for it.',
184
+ inputSchema: {
185
+ type: 'object',
186
+ properties: { ...deviceProp, url: { type: 'string' } },
187
+ required: ['url'],
188
+ },
189
+ },
190
+ {
191
+ name: 'sim_permission',
192
+ description: `Grant, revoke or reset a privacy permission for an app. Do this instead of tapping the system alert: the alert is not part of the app under test, and its buttons move between iOS versions. Services: ${PERMISSION_SERVICES.join(', ')}.`,
193
+ inputSchema: {
194
+ type: 'object',
195
+ properties: {
196
+ ...deviceProp,
197
+ action: { type: 'string', enum: ['grant', 'revoke', 'reset'] },
198
+ service: { type: 'string' },
199
+ bundleId: { type: 'string' },
200
+ },
201
+ required: ['action', 'service'],
202
+ },
203
+ },
204
+ {
205
+ name: 'sim_find',
206
+ description:
207
+ 'Resolve an intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, where on screen you meant, and icon-only controls by their common name. When two things answer equally well it says so and lists them rather than guessing — a wrong tap is worse than a question, because it can do something and leave you believing it did the right thing. Use it when you are unsure a selector will resolve; otherwise just tap.',
208
+ inputSchema: {
209
+ type: 'object',
210
+ properties: { ...deviceProp, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
211
+ required: ['intent'],
212
+ },
213
+ },
214
+ {
215
+ name: 'sim_state',
105
216
  description:
106
- 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen — every frame for the last 10s, thinned to about 2fps before that. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago it started, how long it took, how much of the screen it moved. action "at" returns the buffered frame from a moment in the past. Use this when you look up and find the screen already different, or when something flashed by and you need to know what it was — instead of guessing or re-running the action.',
217
+ 'Cheapest possible check: a stable screen hash, whether anything changed SINCE YOUR LAST LOOK in this session, how long the screen has been still, and an ASCII map of which regions moved. A fraction of the cost of even the text screen map. Use it to poll ("has it finished loading?"). To find out WHAT is on screen, use sim_ui.',
107
218
  inputSchema: {
108
219
  type: 'object',
109
220
  properties: {
110
221
  ...deviceProp,
111
- action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
112
- spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
113
- msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
222
+ since: { type: 'string', description: 'Compare against this frame hash instead of your last look.' },
114
223
  },
115
224
  },
116
225
  },
117
226
  {
118
- name: 'sim_do',
227
+ name: 'sim_wait',
119
228
  description:
120
- 'Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each action automatically waits for the screen to settle before the next step, using a baseline captured before that action, so steps do not race the UI. This is the fastest way to drive the simulator — a twelve-step flow costs one round trip instead of twelve. Prefer it over single taps whenever you know more than one step ahead. Steps stop at the first failure and the result says exactly which step failed and why. Requires idb for input; observation-only steps work without it.',
229
+ 'Block until the screen finishes reacting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly in the moment before an animation starts. Returns the compact screen map by default, not an image. Inside a flow you rarely need this: sim_do settles after every step already.',
121
230
  inputSchema: {
122
231
  type: 'object',
123
232
  properties: {
124
233
  ...deviceProp,
125
- steps: {
126
- type: 'array',
127
- description:
128
- 'Ordered steps. Shorthand forms: {"tap":"Save"} (by accessibility label; add "index" if ambiguous), {"tapAt":{"x":100,"y":200,"space":"points"|"image"}}, {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":"com.example.app"}, {"openUrl":"myapp://x"}, {"waitText":"Saved","timeoutMs":5000}, {"assertText":"Saved"}, {"assertGone":"Spinner"}, {"settle":{"stableMs":600}}, {"look":{"detail":"low"}}, {"pause":300}.',
129
- items: { type: 'object' },
234
+ mode: {
235
+ type: 'string',
236
+ enum: ['settle', 'change', 'stable'],
237
+ description: 'settle (default): wait for a change, then for it to hold still. change: return as soon as it differs. stable: return once it is still, even if nothing moved.',
130
238
  },
131
- autoSettle: {
132
- type: 'boolean',
133
- description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
239
+ since: { type: 'string', description: 'Frame hash to treat as the "before" state. Defaults to your last observation.' },
240
+ stableMs: { type: 'number', description: 'How long the screen must hold still for mode "stable" (default 600).' },
241
+ timeoutMs: { type: 'number', description: 'Give up after this long (default 8000).' },
242
+ map: { type: 'boolean', description: 'Return the screen map afterwards (default true).' },
243
+ },
244
+ },
245
+ },
246
+ {
247
+ name: 'sim_look',
248
+ description:
249
+ 'THE ONLY TOOL THAT RETURNS AN IMAGE, and the most expensive one. Returns the newest buffered frame immediately — no screenshot wait. Call it only when the text map is genuinely not enough: checking visual layout, colour, spacing, an animation, or something the accessibility tree and OCR both cannot see. For "what is on screen and what can I tap", sim_ui answers better and costs a tenth as much.',
250
+ inputSchema: {
251
+ type: 'object',
252
+ properties: {
253
+ ...deviceProp,
254
+ detail: {
255
+ type: 'string',
256
+ enum: ['low', 'normal', 'high'],
257
+ description: 'Image size: low (~420px, cheapest), normal (~700px, default), high (1024px, readable small text). Capped at 1024px on the long edge.',
134
258
  },
135
- stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
136
- timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
137
- continueOnError: { type: 'boolean', description: 'Keep going after a failed step (default false).' },
138
- finalLook: { type: 'boolean', description: 'Attach a frame of the end state (default true).' },
259
+ maxAgeMs: { type: 'number', description: 'If the buffered frame is older than this, wait for a fresher one (default 900).' },
139
260
  },
140
- required: ['steps'],
141
261
  },
142
262
  },
143
263
  {
144
- name: 'sim_ui',
264
+ name: 'sim_strip',
145
265
  description:
146
- 'Read the screen as an accessibility tree instead of an image: every element with its label, value, type and position in points. Often cheaper AND more useful than a screenshot, because it tells you what is actually tappable and gives exact coordinates — no measuring pixels by eye. Use it before tapping something you are unsure about. Requires idb.',
266
+ 'Return the last few buffered frames tiled into ONE image, oldest first — an image, so not cheap. Lets you understand a transition, animation or flicker in a single call instead of a burst of screenshots. Looks backwards in time; it does not wait.',
147
267
  inputSchema: {
148
268
  type: 'object',
149
269
  properties: {
150
270
  ...deviceProp,
151
- filter: { type: 'string', description: 'Only elements whose label, value or identifier contains this text.' },
152
- interactive: { type: 'boolean', description: 'Only elements that look tappable (buttons, fields, cells).' },
271
+ count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
272
+ spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
273
+ thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
274
+ },
275
+ },
276
+ },
277
+ {
278
+ name: 'sim_recall',
279
+ description:
280
+ 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago, how long it took, how much of the screen moved. action "at" returns the buffered frame from a past moment — an image. Use this when you look up and find the screen already different, instead of re-running the action.',
281
+ inputSchema: {
282
+ type: 'object',
283
+ properties: {
284
+ ...deviceProp,
285
+ action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
286
+ spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
287
+ msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
153
288
  },
154
289
  },
155
290
  },
@@ -180,6 +315,16 @@ const TOOLS = [
180
315
  // frame-to-frame delta look broken in practice.
181
316
  const lastSeen = new Map();
182
317
 
318
+ /** Report the real version: a hardcoded one silently drifts every release. */
319
+ function packageVersion() {
320
+ try {
321
+ const here = path.dirname(fileURLToPath(import.meta.url));
322
+ return JSON.parse(fs.readFileSync(path.join(here, '..', 'package.json'), 'utf8')).version;
323
+ } catch {
324
+ return '0.0.0';
325
+ }
326
+ }
327
+
183
328
  function remember(udid, state) {
184
329
  lastSeen.set(udid, { hash: state.hash, seq: state.seq, at: state.capturedAt });
185
330
  }
@@ -221,7 +366,7 @@ function sinceLine(since) {
221
366
 
222
367
  export async function serve({ device: defaultDevice, options = {} } = {}) {
223
368
  const server = new Server(
224
- { name: 'simframe', version: '0.1.0' },
369
+ { name: 'simframe', version: packageVersion() },
225
370
  { capabilities: { tools: {} } },
226
371
  );
227
372
 
@@ -242,10 +387,54 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
242
387
  return await strip(target, args, options);
243
388
  case 'sim_recall':
244
389
  return await recall(target, args, options);
390
+ case 'sim_find':
391
+ return await find(target, args, options);
245
392
  case 'sim_do':
246
393
  return await doScript(target, args, options);
247
394
  case 'sim_ui':
248
395
  return await ui(target, args, options);
396
+ case 'sim_tap':
397
+ return await oneStep(target, { tap: args.sel, index: args.index }, args, options);
398
+ case 'sim_type_into':
399
+ return await oneStep(
400
+ target,
401
+ args.paste
402
+ ? { paste: { into: args.sel, text: args.text } }
403
+ : { type: { into: args.sel, text: args.text } },
404
+ args,
405
+ options,
406
+ );
407
+ case 'sim_scroll_to':
408
+ return await oneStep(
409
+ target,
410
+ { scrollTo: args.sel, direction: args.direction, maxScrolls: args.maxScrolls },
411
+ args,
412
+ options,
413
+ );
414
+ case 'sim_wait_for':
415
+ return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs }, args, options);
416
+ case 'sim_assert':
417
+ return await oneStep(target, { assert: args.sel, is: args.is, equals: args.equals }, args, options);
418
+ case 'sim_launch':
419
+ return await oneStep(
420
+ target,
421
+ { launch: { value: args.bundleId, relaunch: args.relaunch, args: args.args, env: args.env } },
422
+ args,
423
+ options,
424
+ );
425
+ case 'sim_open_url':
426
+ return await oneStep(target, { openUrl: args.url }, args, options);
427
+ case 'sim_permission':
428
+ return await oneStep(
429
+ target,
430
+ { permission: { value: args.service, grant: args.action, bundleId: args.bundleId } },
431
+ args,
432
+ options,
433
+ );
434
+ case 'sim_goto':
435
+ return await goto(target, args, options);
436
+ case 'sim_flow_run':
437
+ return await flowRun(target, args, options);
249
438
  case 'sim_capture':
250
439
  return await capture(target, args, options);
251
440
  case 'sim_devices':
@@ -263,7 +452,11 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
263
452
 
264
453
  async function look(target, args, options) {
265
454
  const maxAge = args.maxAgeMs ?? 900;
266
- let res = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
455
+ // Never native resolution. An image is the expensive path by a factor of ten
456
+ // even when Claude Code handles it correctly; a 1024px frame says everything
457
+ // a 393-point screen has to say.
458
+ const detail = api.modelDetail(args.detail ?? 'normal');
459
+ let res = await api.getFrame(target, { detail, options });
267
460
  if (res.ageMs > maxAge) {
268
461
  // Freshness was requested; the loop is already running, so just let it tick.
269
462
  const deadline = Date.now() + Math.min(2000, maxAge * 3);
@@ -273,7 +466,7 @@ async function look(target, args, options) {
273
466
  if (next && Date.now() - next.capturedAt <= maxAge) break;
274
467
  await new Promise((r) => setTimeout(r, 50));
275
468
  }
276
- res = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
469
+ res = await api.getFrame(target, { detail, options });
277
470
  }
278
471
  const prior = baselineFor(res.device.udid, undefined);
279
472
  const st = await api.getState(target, { since: prior, options });
@@ -340,19 +533,24 @@ async function wait(target, args, options) {
340
533
  }
341
534
  }
342
535
  remember(device.udid, res.state);
343
- const content = [text(`${note}\n${header(res.device, res.state, Date.now() - res.state.capturedAt)}`)];
344
- if (args.includeImage !== false) {
345
- const frame = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
346
- content.push(image(frame.png));
536
+ const lines = [note, header(res.device, res.state, Date.now() - res.state.capturedAt)];
537
+ // The screen map, not a frame. This call is almost always "what happened?",
538
+ // and the answer to that is a list of elements, not 1,600 tokens of pixels.
539
+ if (args.map !== false) {
540
+ try {
541
+ lines.push('', (await view.screenMap(target, { options })).text);
542
+ } catch (err) {
543
+ lines.push(`(could not read the screen: ${err.message})`);
544
+ }
347
545
  }
348
- return { content };
546
+ return { content: [text(lines.join('\n'))] };
349
547
  }
350
548
 
351
549
  async function strip(target, args, options) {
352
550
  const res = await api.getStrip(target, {
353
551
  count: Math.min(12, args.count ?? 5),
354
552
  spanMs: args.spanMs,
355
- thumbMaxDim: args.thumbMaxDim ?? 240,
553
+ thumbMaxDim: Math.min(api.MODEL_MAX_IMAGE_DIM, args.thumbMaxDim ?? 240),
356
554
  options,
357
555
  });
358
556
  const offsets = res.frames.map((f) => `+${f.offsetMs}ms`).join(' ');
@@ -376,6 +574,8 @@ function ago(ms) {
376
574
 
377
575
  async function recall(target, args, options) {
378
576
  if (args.action === 'at') {
577
+ // No cap needed: buffered ring frames are captured at 700px, already under
578
+ // the model image ceiling.
379
579
  const res = await api.getFrameAt(target, { msAgo: args.msAgo ?? 0, options });
380
580
  return {
381
581
  content: [
@@ -414,16 +614,34 @@ async function recall(target, args, options) {
414
614
  return { content: [text(lines.join('\n'))] };
415
615
  }
416
616
 
417
- async function doScript(target, args, options) {
418
- const res = await actions.runScript(target, {
419
- steps: args.steps,
420
- autoSettle: args.autoSettle,
421
- stableMs: args.stableMs,
422
- timeoutMs: args.timeoutMs,
423
- continueOnError: args.continueOnError,
424
- options,
425
- });
617
+ /**
618
+ * The compact map, from an identity reading that has already been taken.
619
+ *
620
+ * A flow verifies where it ended up in order to record the edge, and that
621
+ * reading carries the elements with it — so describing the end state to Claude
622
+ * is free. Reading the screen a second time to render it was the whole cost of
623
+ * a text-first surface, and it does not have to be paid.
624
+ */
625
+ async function mapFrom(target, options, identity, extra = {}) {
626
+ try {
627
+ const m = await view.screenMap(target, { options, identity: identity?.entry ? identity : undefined });
628
+ return view.render({ ...m, ...extra });
629
+ } catch (err) {
630
+ return `(could not read the screen: ${err.message})`;
631
+ }
632
+ }
633
+
634
+ function verdictLineFor(results) {
635
+ const last = [...(results ?? [])].reverse().find((r) => r.verification);
636
+ if (!last) return null;
637
+ const v = last.verification;
638
+ // `detail` already spells out the kind mismatch and the times-seen count, so
639
+ // adding the prediction and the kind again made this line say everything
640
+ // three times — in the one place where brevity is the entire feature.
641
+ return `last action: [${last.index}] ${last.action} — ${v.verdict}: ${v.detail}`;
642
+ }
426
643
 
644
+ function stepLines(res) {
427
645
  const lines = [
428
646
  `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps in ${res.totalMs}ms`,
429
647
  ];
@@ -438,45 +656,140 @@ async function doScript(target, args, options) {
438
656
  );
439
657
  }
440
658
  if (!res.ok) lines.push('later steps were not run; the screen is left wherever the failing step stopped');
659
+ return lines;
660
+ }
441
661
 
442
- const content = [text(lines.join('\n'))];
443
- for (const f of res.frames) content.push(image(f.png));
444
- if (args.finalLook !== false) {
445
- const frame = await api.getFrame(target, { detail: args.detail ?? 'normal', options });
446
- remember(res.device.udid, frame.state);
447
- content.push(text(`end state — ${header(frame.device, frame.state, frame.ageMs)}`), image(frame.png));
662
+ async function doScript(target, args, options) {
663
+ const res = await actions.runScript(target, {
664
+ steps: args.steps,
665
+ autoSettle: args.autoSettle,
666
+ stableMs: args.stableMs,
667
+ timeoutMs: args.timeoutMs,
668
+ continueOnError: args.continueOnError,
669
+ options,
670
+ });
671
+
672
+ const lines = stepLines(res);
673
+ if (args.saveAs) {
674
+ const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
675
+ lines.push(
676
+ saved.ok
677
+ ? `saved as flow "${saved.name}" (${saved.steps} steps) — replay with sim_flow_run`
678
+ : `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
679
+ );
448
680
  }
681
+ lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
682
+
683
+ const content = [text(lines.join('\n'))];
684
+ // Images only when explicitly asked for. A frame attached to every flow was
685
+ // the largest single cost in this server, and it answered a question the map
686
+ // answers better.
687
+ if (args.images) for (const f of res.frames) content.push(image(f.png));
449
688
  return { content, isError: !res.ok };
450
689
  }
451
690
 
452
- async function ui(target, args, options) {
691
+ /** Every single-action tool is one sim_do step, so verification works the same way. */
692
+ async function oneStep(target, step, args, options) {
693
+ const clean = Object.fromEntries(Object.entries(step).filter(([, v]) => v !== undefined));
694
+ return await doScript(target, { ...args, steps: [clean], saveAs: undefined }, options);
695
+ }
696
+
697
+ async function goto(target, args, options) {
453
698
  const { device } = await api.ensureDaemon(target, options);
454
- const driver = await input.detectDriver();
455
- if (!driver.available) return { isError: true, content: [text(driver.reason)] };
456
-
457
- let nodes = await input.describeAll(device.udid);
458
- if (args.filter) {
459
- const q = String(args.filter).toLowerCase();
460
- nodes = nodes.filter((n) =>
461
- [n.label, n.value, n.identifier].filter(Boolean).join(' ').toLowerCase().includes(q),
462
- );
699
+ if (!args.screen) {
700
+ const known = navigate.knownScreens(device.udid);
701
+ if (!known.length) {
702
+ return {
703
+ content: [text('no screens remembered on this device yet — drive a flow with sim_do and simframe learns them as it goes')],
704
+ };
705
+ }
706
+ return {
707
+ content: [
708
+ text(
709
+ `${known.length} screen(s) simframe can navigate to:\n` +
710
+ known.map((k) => ` ${k.hash} ${k.edges} exit(s) ${k.name}`).join('\n'),
711
+ ),
712
+ ],
713
+ };
463
714
  }
464
- if (args.interactive) {
465
- nodes = nodes.filter((n) => /button|field|cell|link|switch|slider|tab|menu/i.test(n.type || ''));
715
+
716
+ const res = await navigate.goto(target, args.screen, { options });
717
+ if (!res.ok && res.reason) {
718
+ const why = {
719
+ 'unknown-screen': `no remembered screen matches "${args.screen}". Known: ${(res.known ?? []).map((k) => k.name).join(', ') || 'none'}`,
720
+ ambiguous: `"${args.screen}" fits ${(res.candidates ?? []).length} screens equally: ${(res.candidates ?? []).map((c) => c.name).join(', ')} — say which`,
721
+ 'no-route': `"${res.to}" is known, but no remembered path reaches it from where you are (${res.from})`,
722
+ 'unreplayable-edge': `the route to "${res.to}" includes a step simframe cannot replay exactly`,
723
+ }[res.reason] ?? res.reason;
724
+ return { isError: true, content: [text(`sim_goto refused rather than guess: ${why}`)] };
466
725
  }
467
- if (!nodes.length) return { content: [text('no matching elements on screen')] };
468
-
469
- const rows = nodes.slice(0, 200).map((n) => {
470
- const c = input.centerOf(n);
471
- const name = [n.label, n.value && `= ${n.value}`, n.identifier && `#${n.identifier}`]
472
- .filter(Boolean)
473
- .join(' ');
474
- const fallback = n.rawLabel ? '(icon-only — tap by these coordinates)' : '(unlabelled)';
475
- return ` ${(n.type || '?').padEnd(14)} ${String(`${c.x},${c.y}`).padEnd(10)} ${name || fallback}`;
726
+
727
+ const lines = res.already
728
+ ? [`already on "${res.screen}"`]
729
+ : [
730
+ `${res.ok ? 'arrived at' : 'DID NOT REACH'} "${res.screen}" in ${res.ranSteps}/${res.steps.length} remembered steps`,
731
+ ...(res.results ?? []).map((r) => ` ${r.ok ? 'ok ' : 'FAIL'} [${r.index}] ${r.action}: ${r.ok ? r.detail : r.error}`),
732
+ ];
733
+ lines.push('', await mapFrom(target, options, null));
734
+ return { content: [text(lines.join('\n'))], isError: !res.ok };
735
+ }
736
+
737
+ async function flowRun(target, args, options) {
738
+ const { device } = await api.ensureDaemon(target, options);
739
+ if (!args.name) {
740
+ const flows = navigate.listFlows(device.udid);
741
+ return {
742
+ content: [
743
+ text(
744
+ flows.length
745
+ ? `saved flows:\n${flows.map((f) => ` ${f.name} (${f.steps} steps)`).join('\n')}`
746
+ : 'no saved flows — run one with sim_do and pass saveAs',
747
+ ),
748
+ ],
749
+ };
750
+ }
751
+ const res = await navigate.runFlow(target, args.name, { options });
752
+ if (res.reason === 'unknown-flow') {
753
+ return {
754
+ isError: true,
755
+ content: [text(`no flow called "${args.name}". Saved: ${(res.known ?? []).join(', ') || 'none'}`)],
756
+ };
757
+ }
758
+ const lines = [`flow "${args.name}"`, ...stepLines(res)];
759
+ lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
760
+ return { content: [text(lines.join('\n'))], isError: !res.ok };
761
+ }
762
+
763
+ async function ui(target, args, options) {
764
+ // No accessibility gate here any more. The old sim_ui asked the tree
765
+ // directly, so a machine without idb could capture and tap perfectly well
766
+ // and still not read the screen — even though OCR alone answers most of them.
767
+ const m = await view.screenMap(target, {
768
+ options,
769
+ filter: args.filter,
770
+ interactive: args.interactive,
771
+ all: args.all,
772
+ refresh: args.refresh,
476
773
  });
477
- const head = `${device.name} — ${nodes.length} element${nodes.length === 1 ? '' : 's'} (type, tap point in points, label)`;
478
- const tail = nodes.length > 200 ? `\n ... ${nodes.length - 200} more; use filter to narrow` : '';
479
- return { content: [text(`${head}\n${rows.join('\n')}${tail}`)] };
774
+ if (m.identity.state) remember(m.device.udid, m.identity.state);
775
+ return { content: [text(m.text)] };
776
+ }
777
+
778
+ async function find(target, args, options) {
779
+ try {
780
+ const r = await api.locate(target, String(args.intent ?? ''), { options });
781
+ const lines = [
782
+ `${r.target.label ?? '(icon-only)'} — tap at (${r.target.x}, ${r.target.y})`,
783
+ `${r.target.region ?? 'content'} · ${r.target.type ?? '?'} · seen by ${r.target.source} · score ${r.score ?? '-'}`,
784
+ ];
785
+ if (r.reasons?.length) lines.push(`chosen because: ${r.reasons.join(', ')}`);
786
+ if (r.alternatives?.length) {
787
+ lines.push(`also considered: ${r.alternatives.map((a) => `"${a.label}" (${a.score})`).join(', ')}`);
788
+ }
789
+ return { content: [text(lines.join('\n'))] };
790
+ } catch (err) {
791
+ return { isError: true, content: [text(err.message)] };
792
+ }
480
793
  }
481
794
 
482
795
  async function capture(target, args, options) {