simframe 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +181 -2
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
  9. package/native/simframed/Sources/simframed/main.swift +33 -3
  10. package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
  11. package/native/supervise.swift +216 -0
  12. package/package.json +4 -1
  13. package/scripts/check-package.mjs +22 -2
  14. package/scripts/check-private.mjs +9 -0
  15. package/scripts/ci-integration-local.sh +79 -0
  16. package/scripts/ci-memory.mjs +104 -20
  17. package/scripts/collect-rulings.mjs +312 -0
  18. package/scripts/eval-fingerprint.mjs +100 -23
  19. package/scripts/eval-perception.mjs +33 -0
  20. package/scripts/phase17-corpus.mjs +176 -0
  21. package/scripts/probe-network.mjs +118 -0
  22. package/scripts/soak-capture.mjs +72 -0
  23. package/skills/simframe/SKILL.md +257 -7
  24. package/src/actions.js +1791 -44
  25. package/src/cli.js +216 -14
  26. package/src/control.js +1 -0
  27. package/src/fingerprint.js +43 -1
  28. package/src/graph.js +136 -7
  29. package/src/index.js +252 -14
  30. package/src/input.js +66 -3
  31. package/src/localhelper.js +161 -0
  32. package/src/matching.js +64 -1
  33. package/src/mcp.js +333 -32
  34. package/src/metrics.js +148 -3
  35. package/src/ocr.js +18 -1
  36. package/src/planner.js +195 -0
  37. package/src/platform/android.js +25 -1
  38. package/src/platform/index.js +11 -1
  39. package/src/platform/ios.js +25 -1
  40. package/src/png.js +26 -0
  41. package/src/refs.js +51 -8
  42. package/src/regions.js +215 -1
  43. package/src/screenmap.js +89 -9
  44. package/src/supervisor.js +161 -0
  45. package/src/view.js +375 -11
  46. package/src/vocabulary.js +134 -0
  47. package/src/wrote.js +136 -0
package/src/mcp.js CHANGED
@@ -13,11 +13,63 @@ import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
13
13
  import * as actions from './actions.js';
14
14
  import * as api from './index.js';
15
15
  import * as input from './input.js';
16
+ import * as metrics from './metrics.js';
16
17
  import * as navigate from './navigate.js';
17
18
  import { bootedDevices, permissionServices } from './platform/index.js';
18
19
  import * as store from './store.js';
19
20
  import * as view from './view.js';
20
21
 
22
+ /**
23
+ * Per-call overrides for the two experiment knobs.
24
+ *
25
+ * Both are read from the call first and the environment second, so an A/B is an
26
+ * argument rather than a server restart — and a run in the wrong mode stops
27
+ * being a thing that can silently happen.
28
+ */
29
+ export function modesFor(base = {}, args = {}) {
30
+ const out = { ...base };
31
+ if (args.sensor) out.sensor = String(args.sensor);
32
+ if (args.planner) out.planner = String(args.planner);
33
+ if (args.supervisor) out.supervisor = String(args.supervisor);
34
+ return out;
35
+ }
36
+
37
+ /** Offered on every tool that reads or acts, because either can be compared. */
38
+ /**
39
+ * Which declared-required argument is absent, phrased as advice.
40
+ *
41
+ * Reads the tool's own `required` array, so a tool that gains a required
42
+ * argument gains this for free and cannot drift out of step with it.
43
+ */
44
+ function missingRequired(name, args) {
45
+ const tool = TOOLS.find((t) => t.name === name);
46
+ const need = tool?.inputSchema?.required ?? [];
47
+ const absent = need.filter((k) => args?.[k] === undefined || args?.[k] === null || args?.[k] === '');
48
+ if (!absent.length) return null;
49
+ const known = Object.keys(args ?? {}).filter((k) => k !== 'device');
50
+ return `${name}: missing required ${absent.map((k) => `"${k}"`).join(', ')}.`
51
+ + (known.length ? ` You passed: ${known.map((k) => `"${k}"`).join(', ')}.` : '')
52
+ + ` ${absent.length === 1 ? 'That argument is' : 'Those arguments are'} the one${absent.length === 1 ? '' : 's'} this tool acts on — pass ${absent.map((k) => `"${k}"`).join(' and ')} and retry.`;
53
+ }
54
+
55
+ const modeProps = {
56
+ sensor: {
57
+ type: 'string',
58
+ enum: ['full', 'ax-first'],
59
+ description: 'Perception for this call. "full" fuses accessibility and OCR (~164ms). "ax-first" reads the tree alone (~50ms) and pays for OCR only when something fails to resolve. Omit to keep the server default.',
60
+ },
61
+ planner: {
62
+ type: 'string',
63
+ enum: ['none', 'apple'],
64
+ description: 'Local model for this call. Orders the containers seek opens; it cannot choose an action. Omit to keep the server default.',
65
+ },
66
+ supervisor: {
67
+ type: 'string',
68
+ enum: ['none', 'apple'],
69
+ description: 'Local supervisor for this call. When a step fails it answers wait, retry or stop — nothing else — before the batch is abandoned. Omit to keep the server default.',
70
+ },
71
+ };
72
+
21
73
  const deviceProp = {
22
74
  device: {
23
75
  type: 'string',
@@ -26,13 +78,22 @@ const deviceProp = {
26
78
  };
27
79
 
28
80
  /**
29
- * One selector grammar everywhere.
81
+ * One selector grammar everywhere, and the order is the recommendation.
30
82
  *
31
- * `#3` is the cheapest thing a caller can say and the least ambiguous, because
32
- * simframe numbered it; a bare phrase is resolved by intent, which is more
33
- * forgiving and occasionally has to ask which one was meant.
83
+ * It used to lead with `#3` and call it "cheapest and unambiguous". Four peer
84
+ * rounds running reported the opposite: intent resolution worked every time,
85
+ * while refs renumbered underneath them and were only safe inside the round
86
+ * trip that issued them. The README was corrected and these descriptions were
87
+ * not, which is the half a caller actually reads.
88
+ *
89
+ * "Unambiguous" was also the wrong word for it. A ref is exact about which
90
+ * element simframe meant and says nothing about whether that element is still
91
+ * there — the failure mode that needed `staleKind` to tell a moved layout from
92
+ * a different screen, and then a score floor and a region check on top of that
93
+ * before a relabelled ref could be trusted. A label carries its own evidence;
94
+ * a number carries none.
34
95
  */
35
- const SELECTOR = 'Selector: "#3" (a number from the last screen map — cheapest and unambiguous), a label or phrase like "Save" or "the Assets tab" (resolved by intent), or "@120,400" for raw point coordinates.';
96
+ const SELECTOR = 'Selector: a label or phrase like "Save" or "the Assets tab" or "back" (resolved by intent — verbs, typos, synonyms, icon-only controls: START HERE), "#3" (a number from the last screen map — exact, but only inside the round trip that numbered it), or "@120,400" for raw point coordinates (last resort: it cannot tell you it missed).';
36
97
 
37
98
  const selectorProp = (what = 'What to act on') => ({
38
99
  sel: { type: 'string', description: `${what}. ${SELECTOR}` },
@@ -42,11 +103,12 @@ const TOOLS = [
42
103
  {
43
104
  name: 'sim_ui',
44
105
  description:
45
- 'READ THE SCREEN. Returns a compact text map: every element with a number, its region (nav-bar / content / tab-bar), type, label, state and tap point, plus which screen this is and how much simframe already knows about it. Roughly a tenth the cost of a screenshot and strictly more useful, because it says what is tappable and where — no measuring pixels by eye. The numbers are selectors: whatever this returns as #3, you can tap as "#3". Start here, not with sim_look.',
106
+ 'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Act on what it shows by NAME — whatever it calls "General", you can tap as "General"; the #3 numbers are exact but only until the screen moves. Start here, never with sim_look.',
46
107
  inputSchema: {
47
108
  type: 'object',
48
109
  properties: {
49
110
  ...deviceProp,
111
+ ...modeProps,
50
112
  filter: { type: 'string', description: 'Only elements whose label or read text contains this.' },
51
113
  interactive: { type: 'boolean', description: 'Only elements that look tappable.' },
52
114
  all: { type: 'boolean', description: 'Include the status bar and every collapsed region (default false).' },
@@ -57,20 +119,25 @@ const TOOLS = [
57
119
  {
58
120
  name: 'sim_do',
59
121
  description:
60
- 'THE MAIN TOOL. Run a whole flow in ONE call: tap, type, scroll, wait and assert, in order. Each step waits for the screen to settle against a baseline captured before it, and is verified against what that action did here last time — so a step that navigated somewhere unintended stops the flow instead of tapping on into the wrong screen. A twelve-step flow costs one round trip instead of twelve. Prefer this over the single-action tools whenever you know more than one step ahead, and put asserts in the flow rather than checking between calls. Returns the compact screen map of where the flow ended; no image.',
122
+ 'THE MAIN TOOL, and the cheapest path. Plan the WHOLE flow and run it in one call — tap, type, scroll, wait, assert — asserting after each step that matters. Every step settles and is checked against what that action did here before, so a wrong turn halts the flow instead of tapping on. Single-action tools are for recovery.',
61
123
  inputSchema: {
62
124
  type: 'object',
63
125
  properties: {
64
126
  ...deviceProp,
127
+ ...modeProps,
128
+ supervise: {
129
+ type: 'string',
130
+ description: 'What the local supervisor should know about this app while the batch runs — how lists load, what makes a control stay disabled, what a benign failure looks like here. It has no knowledge of the app; you do. Ignored when no supervisor is enabled.',
131
+ },
65
132
  steps: {
66
133
  type: 'array',
67
134
  description:
68
- 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}.',
135
+ 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
69
136
  items: { type: 'object' },
70
137
  },
71
138
  autoSettle: {
72
139
  type: 'boolean',
73
- description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
140
+ description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input — with it off the trailing map is read before the last gesture has finished, so it can describe the screen you were on rather than the one you are on.',
74
141
  },
75
142
  stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
76
143
  timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
@@ -92,7 +159,7 @@ const TOOLS = [
92
159
  description: 'Tap one thing. For more than one step, use sim_do — it batches the verification and costs one round trip. Returns the screen map afterwards.',
93
160
  inputSchema: {
94
161
  type: 'object',
95
- properties: { ...deviceProp, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
162
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
96
163
  required: ['sel'],
97
164
  },
98
165
  },
@@ -101,7 +168,7 @@ const TOOLS = [
101
168
  description: 'Focus a field and type into it. Prefer a sim_do step when this is part of a sequence.',
102
169
  inputSchema: {
103
170
  type: 'object',
104
- properties: { ...deviceProp, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
171
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
105
172
  required: ['sel', 'text'],
106
173
  },
107
174
  },
@@ -112,6 +179,7 @@ const TOOLS = [
112
179
  type: 'object',
113
180
  properties: {
114
181
  ...deviceProp,
182
+ ...modeProps,
115
183
  ...selectorProp('What to bring into view'),
116
184
  direction: { type: 'string', enum: ['down', 'up', 'left', 'right'] },
117
185
  maxScrolls: { type: 'number', description: 'Give up after this many screens (default 6).' },
@@ -124,7 +192,7 @@ const TOOLS = [
124
192
  description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
125
193
  inputSchema: {
126
194
  type: 'object',
127
- properties: { ...deviceProp, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
195
+ properties: { ...deviceProp, ...modeProps, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
128
196
  required: ['sel'],
129
197
  },
130
198
  },
@@ -135,6 +203,7 @@ const TOOLS = [
135
203
  type: 'object',
136
204
  properties: {
137
205
  ...deviceProp,
206
+ ...modeProps,
138
207
  ...selectorProp('What to check'),
139
208
  is: {
140
209
  type: 'string',
@@ -149,10 +218,10 @@ const TOOLS = [
149
218
  {
150
219
  name: 'sim_goto',
151
220
  description:
152
- 'Walk to a screen simframe has already been to, by name, planning the route through remembered transitions and verifying every step. Zero reasoning and zero images: the graph knows which taps lead where. Refuses rather than guesses — if the destination is unknown, ambiguous, or unreachable through known transitions, it says which and lists what it does know. Call it with no target to see the known screens.',
221
+ 'Walk to a screen simframe already knows, over edges it has already verified, with no model call per step. Names come from sim_recall or a previous map. Refuses rather than guesses when the route is unknown or the name is ambiguous — a refusal is cheap and a wrong walk is not.',
153
222
  inputSchema: {
154
223
  type: 'object',
155
- properties: { ...deviceProp, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
224
+ properties: { ...deviceProp, ...modeProps, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
156
225
  },
157
226
  },
158
227
  {
@@ -160,7 +229,7 @@ const TOOLS = [
160
229
  description: 'Replay a saved flow by name, verifying each step. Omit `name` to list the saved flows. Save one with sim_do\'s `saveAs`.',
161
230
  inputSchema: {
162
231
  type: 'object',
163
- properties: { ...deviceProp, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
232
+ properties: { ...deviceProp, ...modeProps, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
164
233
  },
165
234
  },
166
235
  {
@@ -170,6 +239,7 @@ const TOOLS = [
170
239
  type: 'object',
171
240
  properties: {
172
241
  ...deviceProp,
242
+ ...modeProps,
173
243
  bundleId: { type: 'string' },
174
244
  relaunch: { type: 'boolean', description: 'Terminate first. Without this, launching an already-running app silently does nothing and you test the screen you were already on.' },
175
245
  args: { type: 'array', items: { type: 'string' }, description: 'Launch arguments passed to the app.' },
@@ -183,7 +253,7 @@ const TOOLS = [
183
253
  description: 'Open a URL or deep link on the device — the fastest way to reach a screen when the app has a link for it.',
184
254
  inputSchema: {
185
255
  type: 'object',
186
- properties: { ...deviceProp, url: { type: 'string' } },
256
+ properties: { ...deviceProp, ...modeProps, url: { type: 'string' } },
187
257
  required: ['url'],
188
258
  },
189
259
  },
@@ -194,6 +264,7 @@ const TOOLS = [
194
264
  type: 'object',
195
265
  properties: {
196
266
  ...deviceProp,
267
+ ...modeProps,
197
268
  action: { type: 'string', enum: ['grant', 'revoke', 'reset'] },
198
269
  service: { type: 'string' },
199
270
  bundleId: { type: 'string' },
@@ -204,10 +275,10 @@ const TOOLS = [
204
275
  {
205
276
  name: 'sim_find',
206
277
  description:
207
- 'Resolve an intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, where on screen you meant, and icon-only controls by their common name. When two things answer equally well it says so and lists them rather than guessing — a wrong tap is worse than a question, because it can do something and leave you believing it did the right thing. Use it when you are unsure a selector will resolve; otherwise just tap.',
278
+ 'Resolve one intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, and where on screen you meant. When two things answer equally well it says so and lists them rather than guessing. Use it when you doubt a selector will resolve; otherwise just tap. Prefer a label or a #ref over coordinates.',
208
279
  inputSchema: {
209
280
  type: 'object',
210
- properties: { ...deviceProp, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
281
+ properties: { ...deviceProp, ...modeProps, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
211
282
  required: ['intent'],
212
283
  },
213
284
  },
@@ -219,6 +290,7 @@ const TOOLS = [
219
290
  type: 'object',
220
291
  properties: {
221
292
  ...deviceProp,
293
+ ...modeProps,
222
294
  since: { type: 'string', description: 'Compare against this frame hash instead of your last look.' },
223
295
  },
224
296
  },
@@ -226,11 +298,12 @@ const TOOLS = [
226
298
  {
227
299
  name: 'sim_wait',
228
300
  description:
229
- 'Block until the screen finishes reacting. Default mode "settle" waits for the screen to CHANGE and then hold still, which is what you want after acting — plain "stable" can return instantly in the moment before an animation starts. Returns the compact screen map by default, not an image. Inside a flow you rarely need this: sim_do settles after every step already.',
301
+ 'Wait for the screen to change, settle, or both. sim_do already settles after every step, so you rarely need this inside a flow — reach for it when something moves without you acting, like a push or a background load. Pass `since` from a hash captured before the thing you are waiting on.',
230
302
  inputSchema: {
231
303
  type: 'object',
232
304
  properties: {
233
305
  ...deviceProp,
306
+ ...modeProps,
234
307
  mode: {
235
308
  type: 'string',
236
309
  enum: ['settle', 'change', 'stable'],
@@ -246,16 +319,24 @@ const TOOLS = [
246
319
  {
247
320
  name: 'sim_look',
248
321
  description:
249
- 'THE ONLY TOOL THAT RETURNS AN IMAGE, and the most expensive one. Returns the newest buffered frame immediately — no screenshot wait. Call it only when the text map is genuinely not enough: checking visual layout, colour, spacing, an animation, or something the accessibility tree and OCR both cannot see. For "what is on screen and what can I tap", sim_ui answers better and costs a tenth as much.',
322
+ 'A screenshot: ~1600 tokens, the most expensive call here. Only for what text cannot answer — layout, colour, spacing, a control the map omits. NOT for what a field contains, whether a button is enabled, or whether an action worked: sim_ui reports the first two and the flow\'s own verdict already answered the third.',
250
323
  inputSchema: {
251
324
  type: 'object',
252
325
  properties: {
253
326
  ...deviceProp,
327
+ ...modeProps,
254
328
  detail: {
255
329
  type: 'string',
256
330
  enum: ['low', 'normal', 'high'],
257
331
  description: 'Image size: low (~420px, cheapest), normal (~700px, default), high (1024px, readable small text). Capped at 1024px on the long edge.',
258
332
  },
333
+ region: {
334
+ type: 'object',
335
+ description: 'Crop to part of the screen and enlarge it, in POINTS — the same coordinates the element map prints: {"x":18,"y":260,"width":366,"height":80}. Use it when a whole screen cannot answer the question at 1024px: selected versus unselected, a chevron, a validation mark. Pair it with detail:"high".',
336
+ properties: {
337
+ x: { type: 'number' }, y: { type: 'number' }, width: { type: 'number' }, height: { type: 'number' },
338
+ },
339
+ },
259
340
  maxAgeMs: { type: 'number', description: 'If the buffered frame is older than this, wait for a fresher one (default 900).' },
260
341
  },
261
342
  },
@@ -268,6 +349,7 @@ const TOOLS = [
268
349
  type: 'object',
269
350
  properties: {
270
351
  ...deviceProp,
352
+ ...modeProps,
271
353
  count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
272
354
  spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
273
355
  thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
@@ -277,11 +359,12 @@ const TOOLS = [
277
359
  {
278
360
  name: 'sim_recall',
279
361
  description:
280
- 'Look BACKWARDS in time. simframe remembers roughly the last 60 seconds of the screen. action "timeline" (default) returns a TEXT-ONLY summary of what happened and when: each change, how long ago, how long it took, how much of the screen moved. action "at" returns the buffered frame from a past moment — an image. Use this when you look up and find the screen already different, instead of re-running the action.',
362
+ 'What happened recently, as text: the screens visited, the actions taken, and what each one did. Use it to re-orient after a failure instead of taking a screenshot, and to learn the screen names sim_goto accepts.',
281
363
  inputSchema: {
282
364
  type: 'object',
283
365
  properties: {
284
366
  ...deviceProp,
367
+ ...modeProps,
285
368
  action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
286
369
  spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
287
370
  msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
@@ -297,6 +380,7 @@ const TOOLS = [
297
380
  properties: {
298
381
  action: { type: 'string', enum: ['status', 'start', 'stop'] },
299
382
  ...deviceProp,
383
+ ...modeProps,
300
384
  fps: { type: 'number', description: 'Capture rate while the screen is moving (default 4).' },
301
385
  },
302
386
  required: ['action'],
@@ -338,9 +422,38 @@ function baselineFor(udid, explicit) {
338
422
  const text = (s) => ({ type: 'text', text: s });
339
423
  const image = (png) => ({ type: 'image', data: png.toString('base64'), mimeType: 'image/png' });
340
424
 
425
+ /**
426
+ * The device's name, and its UDID when the name cannot identify it.
427
+ *
428
+ * Two booted simulators can share a name — measured on this machine: two called
429
+ * "iPhone 17 Pro" at once, which is the default state after creating a second
430
+ * device of the same model. A caller reading a header that says only
431
+ * "iPhone 17 Pro" has no way to tell which one answered, and a reporter spent a
432
+ * session unsure whether they were looking at their own app. `xcrun simctl list
433
+ * devices booted` makes the collision trivial to see, so the header says which.
434
+ */
435
+ function deviceLabel(device) {
436
+ if (!device?.udid) return device?.name ?? 'unknown device';
437
+ const clash = (lastBooted ?? []).filter((d) => d.name === device.name).length > 1;
438
+ return clash ? `${device.name} (${device.udid.slice(0, 8)})` : device.name;
439
+ }
440
+
441
+ /** What the last device listing saw, so a name collision can be noticed at all. */
442
+ let lastBooted = null;
443
+ export function noteBooted(devices) {
444
+ lastBooted = Array.isArray(devices) ? devices.map((d) => ({ name: d.name, udid: d.udid })) : null;
445
+ const names = new Map();
446
+ for (const d of lastBooted ?? []) names.set(d.name, (names.get(d.name) ?? 0) + 1);
447
+ const shared = [...names].filter(([, n]) => n > 1).map(([name]) => name);
448
+ return shared.length
449
+ ? `WARNING: ${shared.map((n) => JSON.stringify(n)).join(', ')} names more than one booted device`
450
+ + ' — pass "device" with a UDID, because a name cannot identify which one you mean'
451
+ : null;
452
+ }
453
+
341
454
  function header(device, state, ageMs, extra = '') {
342
455
  return (
343
- `${device.name} · ${device.runtime} · frame #${state.seq} · ${ageMs}ms old · ` +
456
+ `${deviceLabel(device)} · ${device.runtime} · frame #${state.seq} · ${ageMs}ms old · ` +
344
457
  `${state.width}x${state.height} · still for ${state.stableForMs}ms${extra ? ` · ${extra}` : ''}`
345
458
  );
346
459
  }
@@ -364,7 +477,9 @@ function sinceLine(since) {
364
477
  : `unchanged since your last look ${since.ageMs}ms ago`;
365
478
  }
366
479
 
367
- export async function serve({ device: defaultDevice, options = {} } = {}) {
480
+ export async function serve({ device: defaultDevice, options: baseOptions = {} } = {}) {
481
+ // The last device a caller named, for the life of this server.
482
+ let lastDevice = null;
368
483
  const server = new Server(
369
484
  { name: 'simframe', version: packageVersion() },
370
485
  { capabilities: { tools: {} } },
@@ -374,7 +489,34 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
374
489
 
375
490
  server.setRequestHandler(CallToolRequestSchema, async (req) => {
376
491
  const args = req.params.arguments || {};
377
- const target = args.device || defaultDevice;
492
+ // Sticky, and stickiness is not guessing.
493
+ //
494
+ // Reported: `sim_launch` accepted `device`, and the very next `sim_ui`
495
+ // refused with "2 simulators are booted and none was named" — so a UDID had
496
+ // to ride on all ~15 subsequent calls. Refusing to *choose* between two
497
+ // booted devices is right; forgetting which one the caller already named is
498
+ // not. This remembers only what was explicitly passed, so nothing is ever
499
+ // inferred from a boot list.
500
+ const target = args.device || lastDevice || defaultDevice;
501
+ if (args.device) lastDevice = String(args.device);
502
+ // An MCP server's environment is fixed when it spawns, so a tester asked to
503
+ // compare two sensor modes inside one session could not: round 6 ran its
504
+ // baseline and could not run either variant. The suggested workaround was
505
+ // three server entries with three env blocks, which is worse — three servers
506
+ // on one device is three writers, against the one-writer-per-device rule.
507
+ //
508
+ // So the modes are arguments. Absent, the environment still decides, so
509
+ // nothing that was working changes.
510
+ const options = modesFor(baseOptions, args);
511
+ // Name the missing argument, using the tool's own schema.
512
+ //
513
+ // Reported: `sim_scroll_to` called with `target:` instead of `sel:` answered
514
+ // `simframe: empty step` — which names neither the tool, nor the parameter,
515
+ // nor even that an argument was absent, and cost a schema lookup to decode.
516
+ // The declared `required` list is right there, so the check is generic
517
+ // rather than one guard per tool, and it says what to pass.
518
+ const missing = missingRequired(req.params.name, args);
519
+ if (missing) return { content: [text(missing)], isError: true };
378
520
  try {
379
521
  switch (req.params.name) {
380
522
  case 'sim_look':
@@ -450,6 +592,81 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
450
592
  await server.connect(new StdioServerTransport());
451
593
  }
452
594
 
595
+ /**
596
+ * Crop a frame to a region of the screen and enlarge it.
597
+ *
598
+ * A whole screen at 1024px cannot answer a question about one control: a
599
+ * selected filter chip and an unselected one look identical at that size, and an
600
+ * agent shelled out to `simctl io` and PIL to crop and upscale a chip row for
601
+ * every check it made. The region arrives in **points** — the same coordinates
602
+ * the element map prints — because that is what a caller has in hand.
603
+ */
604
+ /**
605
+ * Read a region argument that may not have arrived as an object.
606
+ *
607
+ * A client is free to hand a declared-object property over as a JSON string,
608
+ * and one did: `{"x":0,"y":60,...}` arrived as text, every field read as
609
+ * undefined, and the crop silently became the whole screen — reported back as
610
+ * `cropped to 402x874pt at 0,0`, which is a crop that did not happen described
611
+ * as one that did. An array is accepted too, because [x, y, width, height] is
612
+ * the shape anyone would try first, and it used to fail exactly as quietly.
613
+ */
614
+ export function readRegion(region) {
615
+ let r = region;
616
+ if (typeof r === 'string') {
617
+ try { r = JSON.parse(r); } catch { return null; }
618
+ }
619
+ if (Array.isArray(r)) {
620
+ const [x, y, width, height] = r.map(Number);
621
+ return [x, y, width, height].every(Number.isFinite) ? { x, y, width, height } : null;
622
+ }
623
+ if (!r || typeof r !== 'object') return null;
624
+ const num = (...keys) => {
625
+ for (const k of keys) if (Number.isFinite(Number(r[k]))) return Number(r[k]);
626
+ return null;
627
+ };
628
+ const width = num('width', 'w');
629
+ const height = num('height', 'h');
630
+ // A region with no size is not a region. Saying so beats returning the whole
631
+ // screen under a caption that claims otherwise.
632
+ if (width == null && height == null) return null;
633
+ return { x: num('x') ?? 0, y: num('y') ?? 0, width, height };
634
+ }
635
+
636
+ async function cropRegion(png, region, points) {
637
+ try {
638
+ const { decodePng, encodePng, cropBitmap, scaleBitmap } = await import('./png.js');
639
+ const bmp = decodePng(png);
640
+ const pw = points?.width;
641
+ const ph = points?.height;
642
+ if (!Number.isFinite(pw) || !Number.isFinite(ph) || !pw || !ph) return { note: 'the screen point size is unknown' };
643
+ // The frame we hold is already downscaled for the model, so map points onto
644
+ // *this* bitmap rather than onto the device's native pixels.
645
+ const sx = bmp.width / pw;
646
+ const sy = bmp.height / ph;
647
+ const x = Number(region.x ?? 0) * sx;
648
+ const y = Number(region.y ?? 0) * sy;
649
+ const w = Number(region.width ?? region.w ?? pw) * sx;
650
+ const h = Number(region.height ?? region.h ?? ph) * sy;
651
+ if (!(w >= 1) || !(h >= 1)) return { note: 'the region has no size' };
652
+ const cut = cropBitmap(bmp, x, y, w, h);
653
+ // Enlarge to the same budget the whole screen gets, so the detail per point
654
+ // is the whole reason to ask for a region.
655
+ const long = Math.max(cut.width, cut.height);
656
+ const factor = Math.min(6, Math.max(1, Math.round(1024 / long)));
657
+ const big = factor > 1 ? scaleBitmap(cut, cut.width * factor, cut.height * factor) : cut;
658
+ return {
659
+ png: encodePng(big),
660
+ note: `cropped to ${Math.round(Number(region.width ?? region.w ?? pw))}x`
661
+ + `${Math.round(Number(region.height ?? region.h ?? ph))}pt at `
662
+ + `${Math.round(Number(region.x ?? 0))},${Math.round(Number(region.y ?? 0))}`
663
+ + `, enlarged ${factor}x`,
664
+ };
665
+ } catch (err) {
666
+ return { note: String(err.message).split('\n')[0] };
667
+ }
668
+ }
669
+
453
670
  async function look(target, args, options) {
454
671
  const maxAge = args.maxAgeMs ?? 900;
455
672
  // Never native resolution. An image is the expensive path by a factor of ten
@@ -472,9 +689,37 @@ async function look(target, args, options) {
472
689
  const st = await api.getState(target, { since: prior, options });
473
690
  remember(res.device.udid, res.state);
474
691
  const lines = [header(res.device, res.state, res.ageMs), sinceLine(st.since)];
692
+ // Louder than a trailing note, because it invalidates the image itself rather
693
+ // than qualifying it.
694
+ if (res.frameBehindMs) {
695
+ lines.unshift(`WARNING: this image is ${Math.round(res.frameBehindMs / 100) / 10}s older than the screen state`
696
+ + ' — it is very likely NOT what is on the device now. Read sim_ui, which is read live,'
697
+ + ' or look again in a moment.');
698
+ }
475
699
  const warn = livenessLine(st.live);
476
700
  if (warn) lines.unshift(warn);
477
- return { content: [text(lines.filter(Boolean).join('\n')), image(res.png)] };
701
+ let png = res.png;
702
+ if (args.region) {
703
+ const parsed = readRegion(args.region);
704
+ if (!parsed) {
705
+ lines.push('that region could not be read (expected {"x":0,"y":260,"width":402,"height":80} in points,'
706
+ + ' or [x, y, width, height]) — this is the whole screen');
707
+ }
708
+ // The point size, which the frame does not carry: `state.width/height` are
709
+ // the *captured frame's* pixels (322x700 here), not the screen's points
710
+ // (402x874). Scaling by them gave a 1:1 ratio, so a crop at y=760 clamped
711
+ // to a single pixel row and returned a 119-byte image — which looked like
712
+ // it had worked. `screenIdentity` answers it in ~17ms warm.
713
+ const geo = await api.screenIdentity(target, { options, confirmNovel: false }).catch(() => null);
714
+ const cropped = parsed ? await cropRegion(png, parsed, geo?.points) : { note: null };
715
+ if (cropped.png) {
716
+ png = cropped.png;
717
+ lines.push(cropped.note);
718
+ } else if (cropped.note) {
719
+ lines.push(`could not crop that region (${cropped.note}) — this is the whole screen`);
720
+ }
721
+ }
722
+ return { content: [text(lines.filter(Boolean).join('\n')), image(png)] };
478
723
  }
479
724
 
480
725
  async function state(target, args, options) {
@@ -635,10 +880,38 @@ async function recall(target, args, options) {
635
880
  * is free. Reading the screen a second time to render it was the whole cost of
636
881
  * a text-first surface, and it does not have to be paid.
637
882
  */
883
+ /**
884
+ * The map at the end of an action, and why it is re-read rather than recalled.
885
+ *
886
+ * It used to be rendered from whatever `screenIdentity` had in hand during
887
+ * verification, which is memory-first by design — so the trailing map could
888
+ * describe the screen as it was seconds earlier. Reported from a real session:
889
+ * *"do's trailing dump is still stale, so I still ran `ui --refresh` after
890
+ * nearly every call — that remains the biggest speed tax."*
891
+ *
892
+ * Saying how old it was (the header does) turned out not to be enough: an agent
893
+ * that cannot trust the map spends a turn re-reading it, and a turn is the
894
+ * expensive unit here. So an action pays one perception pass — a few hundred
895
+ * milliseconds, locally, once — to save a model round trip. That is the whole
896
+ * trade this phase is about, and it is the right way round.
897
+ *
898
+ * `refresh: false` is still available for the read-only tools, where the caller
899
+ * asked for a map and can ask again.
900
+ */
638
901
  async function mapFrom(target, options, identity, extra = {}) {
639
902
  try {
640
- const m = await view.screenMap(target, { options, identity: identity?.entry ? identity : undefined });
641
- return view.render({ ...m, ...extra });
903
+ const fresh = extra.refresh !== false;
904
+ const m = await view.screenMap(target, {
905
+ options,
906
+ refresh: fresh,
907
+ identity: fresh ? undefined : (identity?.entry ? identity : undefined),
908
+ });
909
+ const rendered = view.render({ ...m, ...extra });
910
+ // The line that decides whether the model stops to think. Everything it
911
+ // needs is already computed for the map above it, so this costs nothing —
912
+ // and "nothing here needs you" is a thing only the daemon can say.
913
+ if (extra.hint === false) return rendered;
914
+ return `${rendered}\n${view.hintFor(m, { flowOk: extra.flowOk, escalated: extra.escalated })}`;
642
915
  } catch (err) {
643
916
  return `(could not read the screen: ${err.message})`;
644
917
  }
@@ -679,10 +952,29 @@ async function doScript(target, args, options) {
679
952
  stableMs: args.stableMs,
680
953
  timeoutMs: args.timeoutMs,
681
954
  continueOnError: args.continueOnError,
955
+ // The plan's briefing for its own first responder. Ignored when no
956
+ // supervisor is enabled, so passing it is always safe.
957
+ supervise: args.supervise,
682
958
  options,
683
959
  });
684
960
 
685
961
  const lines = stepLines(res);
962
+ // A map read without settling is not a reading of the screen you are now on,
963
+ // and it used to arrive looking exactly like one. Reported: two taps under
964
+ // `autoSettle:false` returned a map showing nothing had happened, so the
965
+ // agent moved on — the page had in fact zoomed all the way out, and they only
966
+ // found out two calls later when an unrelated failure printed a real map.
967
+ if (args.autoSettle === false) {
968
+ lines.push('autoSettle was off, so the map below was read without waiting for the last action to finish'
969
+ + ' — it may describe the screen before that action landed. Re-read (sim_ui) before acting on it.');
970
+ }
971
+ // Every local ruling is reported, because a wrong one has to be correctable
972
+ // rather than mysterious — and the model's own stated reason is shown as its
973
+ // claim, not as the ground for what happened.
974
+ for (const s_ of res.supervisions ?? []) {
975
+ lines.push(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
976
+ + (s_.reason ? ` (it said: "${s_.reason}")` : ''));
977
+ }
686
978
  if (args.saveAs) {
687
979
  const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
688
980
  lines.push(
@@ -691,7 +983,12 @@ async function doScript(target, args, options) {
691
983
  : `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
692
984
  );
693
985
  }
694
- lines.push('', await mapFrom(target, options, res.endScreen, { verdictLine: verdictLineFor(res.results) }));
986
+ const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
987
+ lines.push('', await mapFrom(target, options, res.endScreen, {
988
+ verdictLine: verdictLineFor(res.results),
989
+ flowOk: res.ok,
990
+ escalated,
991
+ }));
695
992
 
696
993
  const content = [text(lines.join('\n'))];
697
994
  // Images only when explicitly asked for. A frame attached to every flow was
@@ -785,7 +1082,9 @@ async function ui(target, args, options) {
785
1082
  refresh: args.refresh,
786
1083
  });
787
1084
  if (m.identity.state) remember(m.device.udid, m.identity.state);
788
- return { content: [text(m.text)] };
1085
+ // The hint belongs here too: the agent has just looked, so "you do not need
1086
+ // to look again" is exactly the thing worth saying at this moment.
1087
+ return { content: [text(`${m.text}\n${view.hintFor(m)}`)] };
789
1088
  }
790
1089
 
791
1090
  async function find(target, args, options) {
@@ -840,7 +1139,9 @@ function listStateDirs() {
840
1139
  async function devices() {
841
1140
  const booted = await bootedDevices();
842
1141
  if (!booted.length) return { content: [text('no booted devices')] };
843
- return {
844
- content: [text(booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n'))],
845
- };
1142
+ // Noticing a name collision here is what lets every later header disambiguate
1143
+ // itself, and it costs nothing: this listing is already being made.
1144
+ const clash = noteBooted(booted);
1145
+ const list = booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1146
+ return { content: [text(clash ? `${clash}\n\n${list}` : list)] };
846
1147
  }