simframe 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +181 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +33 -3
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +216 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +257 -7
- package/src/actions.js +1791 -44
- package/src/cli.js +216 -14
- package/src/control.js +1 -0
- package/src/fingerprint.js +43 -1
- package/src/graph.js +136 -7
- package/src/index.js +252 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +161 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +333 -32
- package/src/metrics.js +148 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +25 -1
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +25 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +215 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +161 -0
- package/src/view.js +375 -11
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/mcp.js
CHANGED
|
@@ -13,11 +13,63 @@ import { REGION_COLS, REGION_ROWS, regionMap } from './analyze.js';
|
|
|
13
13
|
import * as actions from './actions.js';
|
|
14
14
|
import * as api from './index.js';
|
|
15
15
|
import * as input from './input.js';
|
|
16
|
+
import * as metrics from './metrics.js';
|
|
16
17
|
import * as navigate from './navigate.js';
|
|
17
18
|
import { bootedDevices, permissionServices } from './platform/index.js';
|
|
18
19
|
import * as store from './store.js';
|
|
19
20
|
import * as view from './view.js';
|
|
20
21
|
|
|
22
|
+
/**
|
|
23
|
+
* Per-call overrides for the two experiment knobs.
|
|
24
|
+
*
|
|
25
|
+
* Both are read from the call first and the environment second, so an A/B is an
|
|
26
|
+
* argument rather than a server restart — and a run in the wrong mode stops
|
|
27
|
+
* being a thing that can silently happen.
|
|
28
|
+
*/
|
|
29
|
+
export function modesFor(base = {}, args = {}) {
|
|
30
|
+
const out = { ...base };
|
|
31
|
+
if (args.sensor) out.sensor = String(args.sensor);
|
|
32
|
+
if (args.planner) out.planner = String(args.planner);
|
|
33
|
+
if (args.supervisor) out.supervisor = String(args.supervisor);
|
|
34
|
+
return out;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** Offered on every tool that reads or acts, because either can be compared. */
|
|
38
|
+
/**
|
|
39
|
+
* Which declared-required argument is absent, phrased as advice.
|
|
40
|
+
*
|
|
41
|
+
* Reads the tool's own `required` array, so a tool that gains a required
|
|
42
|
+
* argument gains this for free and cannot drift out of step with it.
|
|
43
|
+
*/
|
|
44
|
+
function missingRequired(name, args) {
|
|
45
|
+
const tool = TOOLS.find((t) => t.name === name);
|
|
46
|
+
const need = tool?.inputSchema?.required ?? [];
|
|
47
|
+
const absent = need.filter((k) => args?.[k] === undefined || args?.[k] === null || args?.[k] === '');
|
|
48
|
+
if (!absent.length) return null;
|
|
49
|
+
const known = Object.keys(args ?? {}).filter((k) => k !== 'device');
|
|
50
|
+
return `${name}: missing required ${absent.map((k) => `"${k}"`).join(', ')}.`
|
|
51
|
+
+ (known.length ? ` You passed: ${known.map((k) => `"${k}"`).join(', ')}.` : '')
|
|
52
|
+
+ ` ${absent.length === 1 ? 'That argument is' : 'Those arguments are'} the one${absent.length === 1 ? '' : 's'} this tool acts on — pass ${absent.map((k) => `"${k}"`).join(' and ')} and retry.`;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const modeProps = {
|
|
56
|
+
sensor: {
|
|
57
|
+
type: 'string',
|
|
58
|
+
enum: ['full', 'ax-first'],
|
|
59
|
+
description: 'Perception for this call. "full" fuses accessibility and OCR (~164ms). "ax-first" reads the tree alone (~50ms) and pays for OCR only when something fails to resolve. Omit to keep the server default.',
|
|
60
|
+
},
|
|
61
|
+
planner: {
|
|
62
|
+
type: 'string',
|
|
63
|
+
enum: ['none', 'apple'],
|
|
64
|
+
description: 'Local model for this call. Orders the containers seek opens; it cannot choose an action. Omit to keep the server default.',
|
|
65
|
+
},
|
|
66
|
+
supervisor: {
|
|
67
|
+
type: 'string',
|
|
68
|
+
enum: ['none', 'apple'],
|
|
69
|
+
description: 'Local supervisor for this call. When a step fails it answers wait, retry or stop — nothing else — before the batch is abandoned. Omit to keep the server default.',
|
|
70
|
+
},
|
|
71
|
+
};
|
|
72
|
+
|
|
21
73
|
const deviceProp = {
|
|
22
74
|
device: {
|
|
23
75
|
type: 'string',
|
|
@@ -26,13 +78,22 @@ const deviceProp = {
|
|
|
26
78
|
};
|
|
27
79
|
|
|
28
80
|
/**
|
|
29
|
-
* One selector grammar everywhere.
|
|
81
|
+
* One selector grammar everywhere, and the order is the recommendation.
|
|
30
82
|
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
83
|
+
* It used to lead with `#3` and call it "cheapest and unambiguous". Four peer
|
|
84
|
+
* rounds running reported the opposite: intent resolution worked every time,
|
|
85
|
+
* while refs renumbered underneath them and were only safe inside the round
|
|
86
|
+
* trip that issued them. The README was corrected and these descriptions were
|
|
87
|
+
* not, which is the half a caller actually reads.
|
|
88
|
+
*
|
|
89
|
+
* "Unambiguous" was also the wrong word for it. A ref is exact about which
|
|
90
|
+
* element simframe meant and says nothing about whether that element is still
|
|
91
|
+
* there — the failure mode that needed `staleKind` to tell a moved layout from
|
|
92
|
+
* a different screen, and then a score floor and a region check on top of that
|
|
93
|
+
* before a relabelled ref could be trusted. A label carries its own evidence;
|
|
94
|
+
* a number carries none.
|
|
34
95
|
*/
|
|
35
|
-
const SELECTOR = 'Selector: "#3" (a number from the last screen map —
|
|
96
|
+
const SELECTOR = 'Selector: a label or phrase like "Save" or "the Assets tab" or "back" (resolved by intent — verbs, typos, synonyms, icon-only controls: START HERE), "#3" (a number from the last screen map — exact, but only inside the round trip that numbered it), or "@120,400" for raw point coordinates (last resort: it cannot tell you it missed).';
|
|
36
97
|
|
|
37
98
|
const selectorProp = (what = 'What to act on') => ({
|
|
38
99
|
sel: { type: 'string', description: `${what}. ${SELECTOR}` },
|
|
@@ -42,11 +103,12 @@ const TOOLS = [
|
|
|
42
103
|
{
|
|
43
104
|
name: 'sim_ui',
|
|
44
105
|
description:
|
|
45
|
-
'READ THE SCREEN
|
|
106
|
+
'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Act on what it shows by NAME — whatever it calls "General", you can tap as "General"; the #3 numbers are exact but only until the screen moves. Start here, never with sim_look.',
|
|
46
107
|
inputSchema: {
|
|
47
108
|
type: 'object',
|
|
48
109
|
properties: {
|
|
49
110
|
...deviceProp,
|
|
111
|
+
...modeProps,
|
|
50
112
|
filter: { type: 'string', description: 'Only elements whose label or read text contains this.' },
|
|
51
113
|
interactive: { type: 'boolean', description: 'Only elements that look tappable.' },
|
|
52
114
|
all: { type: 'boolean', description: 'Include the status bar and every collapsed region (default false).' },
|
|
@@ -57,20 +119,25 @@ const TOOLS = [
|
|
|
57
119
|
{
|
|
58
120
|
name: 'sim_do',
|
|
59
121
|
description:
|
|
60
|
-
'THE MAIN TOOL.
|
|
122
|
+
'THE MAIN TOOL, and the cheapest path. Plan the WHOLE flow and run it in one call — tap, type, scroll, wait, assert — asserting after each step that matters. Every step settles and is checked against what that action did here before, so a wrong turn halts the flow instead of tapping on. Single-action tools are for recovery.',
|
|
61
123
|
inputSchema: {
|
|
62
124
|
type: 'object',
|
|
63
125
|
properties: {
|
|
64
126
|
...deviceProp,
|
|
127
|
+
...modeProps,
|
|
128
|
+
supervise: {
|
|
129
|
+
type: 'string',
|
|
130
|
+
description: 'What the local supervisor should know about this app while the batch runs — how lists load, what makes a control stay disabled, what a benign failure looks like here. It has no knowledge of the app; you do. Ignored when no supervisor is enabled.',
|
|
131
|
+
},
|
|
65
132
|
steps: {
|
|
66
133
|
type: 'array',
|
|
67
134
|
description:
|
|
68
|
-
'Ordered steps. Every selector below accepts "
|
|
135
|
+
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
69
136
|
items: { type: 'object' },
|
|
70
137
|
},
|
|
71
138
|
autoSettle: {
|
|
72
139
|
type: 'boolean',
|
|
73
|
-
description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input.',
|
|
140
|
+
description: 'Wait for the screen to settle after each action (default true). Turn off only for deliberate rapid input — with it off the trailing map is read before the last gesture has finished, so it can describe the screen you were on rather than the one you are on.',
|
|
74
141
|
},
|
|
75
142
|
stableMs: { type: 'number', description: 'How still the screen must be to count as settled (default 500).' },
|
|
76
143
|
timeoutMs: { type: 'number', description: 'Per-step settle timeout (default 8000).' },
|
|
@@ -92,7 +159,7 @@ const TOOLS = [
|
|
|
92
159
|
description: 'Tap one thing. For more than one step, use sim_do — it batches the verification and costs one round trip. Returns the screen map afterwards.',
|
|
93
160
|
inputSchema: {
|
|
94
161
|
type: 'object',
|
|
95
|
-
properties: { ...deviceProp, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
|
|
162
|
+
properties: { ...deviceProp, ...modeProps, ...selectorProp('What to tap'), index: { type: 'number', description: 'Which match, when the selector fits several.' } },
|
|
96
163
|
required: ['sel'],
|
|
97
164
|
},
|
|
98
165
|
},
|
|
@@ -101,7 +168,7 @@ const TOOLS = [
|
|
|
101
168
|
description: 'Focus a field and type into it. Prefer a sim_do step when this is part of a sequence.',
|
|
102
169
|
inputSchema: {
|
|
103
170
|
type: 'object',
|
|
104
|
-
properties: { ...deviceProp, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
|
|
171
|
+
properties: { ...deviceProp, ...modeProps, ...selectorProp('The field'), text: { type: 'string' }, paste: { type: 'boolean', description: 'Use the pasteboard instead of the keyboard — much faster for long strings.' } },
|
|
105
172
|
required: ['sel', 'text'],
|
|
106
173
|
},
|
|
107
174
|
},
|
|
@@ -112,6 +179,7 @@ const TOOLS = [
|
|
|
112
179
|
type: 'object',
|
|
113
180
|
properties: {
|
|
114
181
|
...deviceProp,
|
|
182
|
+
...modeProps,
|
|
115
183
|
...selectorProp('What to bring into view'),
|
|
116
184
|
direction: { type: 'string', enum: ['down', 'up', 'left', 'right'] },
|
|
117
185
|
maxScrolls: { type: 'number', description: 'Give up after this many screens (default 6).' },
|
|
@@ -124,7 +192,7 @@ const TOOLS = [
|
|
|
124
192
|
description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
|
|
125
193
|
inputSchema: {
|
|
126
194
|
type: 'object',
|
|
127
|
-
properties: { ...deviceProp, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
|
|
195
|
+
properties: { ...deviceProp, ...modeProps, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
|
|
128
196
|
required: ['sel'],
|
|
129
197
|
},
|
|
130
198
|
},
|
|
@@ -135,6 +203,7 @@ const TOOLS = [
|
|
|
135
203
|
type: 'object',
|
|
136
204
|
properties: {
|
|
137
205
|
...deviceProp,
|
|
206
|
+
...modeProps,
|
|
138
207
|
...selectorProp('What to check'),
|
|
139
208
|
is: {
|
|
140
209
|
type: 'string',
|
|
@@ -149,10 +218,10 @@ const TOOLS = [
|
|
|
149
218
|
{
|
|
150
219
|
name: 'sim_goto',
|
|
151
220
|
description:
|
|
152
|
-
'Walk to a screen simframe
|
|
221
|
+
'Walk to a screen simframe already knows, over edges it has already verified, with no model call per step. Names come from sim_recall or a previous map. Refuses rather than guesses when the route is unknown or the name is ambiguous — a refusal is cheap and a wrong walk is not.',
|
|
153
222
|
inputSchema: {
|
|
154
223
|
type: 'object',
|
|
155
|
-
properties: { ...deviceProp, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
|
|
224
|
+
properties: { ...deviceProp, ...modeProps, screen: { type: 'string', description: 'What the screen is called, e.g. "Settings" or an 8-character screen hash. Omit to list what is known.' } },
|
|
156
225
|
},
|
|
157
226
|
},
|
|
158
227
|
{
|
|
@@ -160,7 +229,7 @@ const TOOLS = [
|
|
|
160
229
|
description: 'Replay a saved flow by name, verifying each step. Omit `name` to list the saved flows. Save one with sim_do\'s `saveAs`.',
|
|
161
230
|
inputSchema: {
|
|
162
231
|
type: 'object',
|
|
163
|
-
properties: { ...deviceProp, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
|
|
232
|
+
properties: { ...deviceProp, ...modeProps, name: { type: 'string', description: 'Flow to run. Omit to list.' } },
|
|
164
233
|
},
|
|
165
234
|
},
|
|
166
235
|
{
|
|
@@ -170,6 +239,7 @@ const TOOLS = [
|
|
|
170
239
|
type: 'object',
|
|
171
240
|
properties: {
|
|
172
241
|
...deviceProp,
|
|
242
|
+
...modeProps,
|
|
173
243
|
bundleId: { type: 'string' },
|
|
174
244
|
relaunch: { type: 'boolean', description: 'Terminate first. Without this, launching an already-running app silently does nothing and you test the screen you were already on.' },
|
|
175
245
|
args: { type: 'array', items: { type: 'string' }, description: 'Launch arguments passed to the app.' },
|
|
@@ -183,7 +253,7 @@ const TOOLS = [
|
|
|
183
253
|
description: 'Open a URL or deep link on the device — the fastest way to reach a screen when the app has a link for it.',
|
|
184
254
|
inputSchema: {
|
|
185
255
|
type: 'object',
|
|
186
|
-
properties: { ...deviceProp, url: { type: 'string' } },
|
|
256
|
+
properties: { ...deviceProp, ...modeProps, url: { type: 'string' } },
|
|
187
257
|
required: ['url'],
|
|
188
258
|
},
|
|
189
259
|
},
|
|
@@ -194,6 +264,7 @@ const TOOLS = [
|
|
|
194
264
|
type: 'object',
|
|
195
265
|
properties: {
|
|
196
266
|
...deviceProp,
|
|
267
|
+
...modeProps,
|
|
197
268
|
action: { type: 'string', enum: ['grant', 'revoke', 'reset'] },
|
|
198
269
|
service: { type: 'string' },
|
|
199
270
|
bundleId: { type: 'string' },
|
|
@@ -204,10 +275,10 @@ const TOOLS = [
|
|
|
204
275
|
{
|
|
205
276
|
name: 'sim_find',
|
|
206
277
|
description:
|
|
207
|
-
'Resolve
|
|
278
|
+
'Resolve one intent to one control: "tap Save", "the Assets tab", "back". Understands verbs, typos, and where on screen you meant. When two things answer equally well it says so and lists them rather than guessing. Use it when you doubt a selector will resolve; otherwise just tap. Prefer a label or a #ref over coordinates.',
|
|
208
279
|
inputSchema: {
|
|
209
280
|
type: 'object',
|
|
210
|
-
properties: { ...deviceProp, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
|
|
281
|
+
properties: { ...deviceProp, ...modeProps, intent: { type: 'string', description: 'What you want to act on, in your own words.' } },
|
|
211
282
|
required: ['intent'],
|
|
212
283
|
},
|
|
213
284
|
},
|
|
@@ -219,6 +290,7 @@ const TOOLS = [
|
|
|
219
290
|
type: 'object',
|
|
220
291
|
properties: {
|
|
221
292
|
...deviceProp,
|
|
293
|
+
...modeProps,
|
|
222
294
|
since: { type: 'string', description: 'Compare against this frame hash instead of your last look.' },
|
|
223
295
|
},
|
|
224
296
|
},
|
|
@@ -226,11 +298,12 @@ const TOOLS = [
|
|
|
226
298
|
{
|
|
227
299
|
name: 'sim_wait',
|
|
228
300
|
description:
|
|
229
|
-
'
|
|
301
|
+
'Wait for the screen to change, settle, or both. sim_do already settles after every step, so you rarely need this inside a flow — reach for it when something moves without you acting, like a push or a background load. Pass `since` from a hash captured before the thing you are waiting on.',
|
|
230
302
|
inputSchema: {
|
|
231
303
|
type: 'object',
|
|
232
304
|
properties: {
|
|
233
305
|
...deviceProp,
|
|
306
|
+
...modeProps,
|
|
234
307
|
mode: {
|
|
235
308
|
type: 'string',
|
|
236
309
|
enum: ['settle', 'change', 'stable'],
|
|
@@ -246,16 +319,24 @@ const TOOLS = [
|
|
|
246
319
|
{
|
|
247
320
|
name: 'sim_look',
|
|
248
321
|
description:
|
|
249
|
-
'
|
|
322
|
+
'A screenshot: ~1600 tokens, the most expensive call here. Only for what text cannot answer — layout, colour, spacing, a control the map omits. NOT for what a field contains, whether a button is enabled, or whether an action worked: sim_ui reports the first two and the flow\'s own verdict already answered the third.',
|
|
250
323
|
inputSchema: {
|
|
251
324
|
type: 'object',
|
|
252
325
|
properties: {
|
|
253
326
|
...deviceProp,
|
|
327
|
+
...modeProps,
|
|
254
328
|
detail: {
|
|
255
329
|
type: 'string',
|
|
256
330
|
enum: ['low', 'normal', 'high'],
|
|
257
331
|
description: 'Image size: low (~420px, cheapest), normal (~700px, default), high (1024px, readable small text). Capped at 1024px on the long edge.',
|
|
258
332
|
},
|
|
333
|
+
region: {
|
|
334
|
+
type: 'object',
|
|
335
|
+
description: 'Crop to part of the screen and enlarge it, in POINTS — the same coordinates the element map prints: {"x":18,"y":260,"width":366,"height":80}. Use it when a whole screen cannot answer the question at 1024px: selected versus unselected, a chevron, a validation mark. Pair it with detail:"high".',
|
|
336
|
+
properties: {
|
|
337
|
+
x: { type: 'number' }, y: { type: 'number' }, width: { type: 'number' }, height: { type: 'number' },
|
|
338
|
+
},
|
|
339
|
+
},
|
|
259
340
|
maxAgeMs: { type: 'number', description: 'If the buffered frame is older than this, wait for a fresher one (default 900).' },
|
|
260
341
|
},
|
|
261
342
|
},
|
|
@@ -268,6 +349,7 @@ const TOOLS = [
|
|
|
268
349
|
type: 'object',
|
|
269
350
|
properties: {
|
|
270
351
|
...deviceProp,
|
|
352
|
+
...modeProps,
|
|
271
353
|
count: { type: 'number', description: 'How many frames to tile (default 5, max 12).' },
|
|
272
354
|
spanMs: { type: 'number', description: 'Only include frames from the last N milliseconds.' },
|
|
273
355
|
thumbMaxDim: { type: 'number', description: 'Height budget per frame in pixels (default 240).' },
|
|
@@ -277,11 +359,12 @@ const TOOLS = [
|
|
|
277
359
|
{
|
|
278
360
|
name: 'sim_recall',
|
|
279
361
|
description:
|
|
280
|
-
'
|
|
362
|
+
'What happened recently, as text: the screens visited, the actions taken, and what each one did. Use it to re-orient after a failure instead of taking a screenshot, and to learn the screen names sim_goto accepts.',
|
|
281
363
|
inputSchema: {
|
|
282
364
|
type: 'object',
|
|
283
365
|
properties: {
|
|
284
366
|
...deviceProp,
|
|
367
|
+
...modeProps,
|
|
285
368
|
action: { type: 'string', enum: ['timeline', 'at'], description: 'timeline (default) or at.' },
|
|
286
369
|
spanMs: { type: 'number', description: 'For timeline: how far back to summarise (default 60000).' },
|
|
287
370
|
msAgo: { type: 'number', description: 'For at: how long ago the moment of interest was, in milliseconds.' },
|
|
@@ -297,6 +380,7 @@ const TOOLS = [
|
|
|
297
380
|
properties: {
|
|
298
381
|
action: { type: 'string', enum: ['status', 'start', 'stop'] },
|
|
299
382
|
...deviceProp,
|
|
383
|
+
...modeProps,
|
|
300
384
|
fps: { type: 'number', description: 'Capture rate while the screen is moving (default 4).' },
|
|
301
385
|
},
|
|
302
386
|
required: ['action'],
|
|
@@ -338,9 +422,38 @@ function baselineFor(udid, explicit) {
|
|
|
338
422
|
const text = (s) => ({ type: 'text', text: s });
|
|
339
423
|
const image = (png) => ({ type: 'image', data: png.toString('base64'), mimeType: 'image/png' });
|
|
340
424
|
|
|
425
|
+
/**
|
|
426
|
+
* The device's name, and its UDID when the name cannot identify it.
|
|
427
|
+
*
|
|
428
|
+
* Two booted simulators can share a name — measured on this machine: two called
|
|
429
|
+
* "iPhone 17 Pro" at once, which is the default state after creating a second
|
|
430
|
+
* device of the same model. A caller reading a header that says only
|
|
431
|
+
* "iPhone 17 Pro" has no way to tell which one answered, and a reporter spent a
|
|
432
|
+
* session unsure whether they were looking at their own app. `xcrun simctl list
|
|
433
|
+
* devices booted` makes the collision trivial to see, so the header says which.
|
|
434
|
+
*/
|
|
435
|
+
function deviceLabel(device) {
|
|
436
|
+
if (!device?.udid) return device?.name ?? 'unknown device';
|
|
437
|
+
const clash = (lastBooted ?? []).filter((d) => d.name === device.name).length > 1;
|
|
438
|
+
return clash ? `${device.name} (${device.udid.slice(0, 8)})` : device.name;
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
/** What the last device listing saw, so a name collision can be noticed at all. */
|
|
442
|
+
let lastBooted = null;
|
|
443
|
+
export function noteBooted(devices) {
|
|
444
|
+
lastBooted = Array.isArray(devices) ? devices.map((d) => ({ name: d.name, udid: d.udid })) : null;
|
|
445
|
+
const names = new Map();
|
|
446
|
+
for (const d of lastBooted ?? []) names.set(d.name, (names.get(d.name) ?? 0) + 1);
|
|
447
|
+
const shared = [...names].filter(([, n]) => n > 1).map(([name]) => name);
|
|
448
|
+
return shared.length
|
|
449
|
+
? `WARNING: ${shared.map((n) => JSON.stringify(n)).join(', ')} names more than one booted device`
|
|
450
|
+
+ ' — pass "device" with a UDID, because a name cannot identify which one you mean'
|
|
451
|
+
: null;
|
|
452
|
+
}
|
|
453
|
+
|
|
341
454
|
function header(device, state, ageMs, extra = '') {
|
|
342
455
|
return (
|
|
343
|
-
`${device
|
|
456
|
+
`${deviceLabel(device)} · ${device.runtime} · frame #${state.seq} · ${ageMs}ms old · ` +
|
|
344
457
|
`${state.width}x${state.height} · still for ${state.stableForMs}ms${extra ? ` · ${extra}` : ''}`
|
|
345
458
|
);
|
|
346
459
|
}
|
|
@@ -364,7 +477,9 @@ function sinceLine(since) {
|
|
|
364
477
|
: `unchanged since your last look ${since.ageMs}ms ago`;
|
|
365
478
|
}
|
|
366
479
|
|
|
367
|
-
export async function serve({ device: defaultDevice, options = {} } = {}) {
|
|
480
|
+
export async function serve({ device: defaultDevice, options: baseOptions = {} } = {}) {
|
|
481
|
+
// The last device a caller named, for the life of this server.
|
|
482
|
+
let lastDevice = null;
|
|
368
483
|
const server = new Server(
|
|
369
484
|
{ name: 'simframe', version: packageVersion() },
|
|
370
485
|
{ capabilities: { tools: {} } },
|
|
@@ -374,7 +489,34 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
|
|
|
374
489
|
|
|
375
490
|
server.setRequestHandler(CallToolRequestSchema, async (req) => {
|
|
376
491
|
const args = req.params.arguments || {};
|
|
377
|
-
|
|
492
|
+
// Sticky, and stickiness is not guessing.
|
|
493
|
+
//
|
|
494
|
+
// Reported: `sim_launch` accepted `device`, and the very next `sim_ui`
|
|
495
|
+
// refused with "2 simulators are booted and none was named" — so a UDID had
|
|
496
|
+
// to ride on all ~15 subsequent calls. Refusing to *choose* between two
|
|
497
|
+
// booted devices is right; forgetting which one the caller already named is
|
|
498
|
+
// not. This remembers only what was explicitly passed, so nothing is ever
|
|
499
|
+
// inferred from a boot list.
|
|
500
|
+
const target = args.device || lastDevice || defaultDevice;
|
|
501
|
+
if (args.device) lastDevice = String(args.device);
|
|
502
|
+
// An MCP server's environment is fixed when it spawns, so a tester asked to
|
|
503
|
+
// compare two sensor modes inside one session could not: round 6 ran its
|
|
504
|
+
// baseline and could not run either variant. The suggested workaround was
|
|
505
|
+
// three server entries with three env blocks, which is worse — three servers
|
|
506
|
+
// on one device is three writers, against the one-writer-per-device rule.
|
|
507
|
+
//
|
|
508
|
+
// So the modes are arguments. Absent, the environment still decides, so
|
|
509
|
+
// nothing that was working changes.
|
|
510
|
+
const options = modesFor(baseOptions, args);
|
|
511
|
+
// Name the missing argument, using the tool's own schema.
|
|
512
|
+
//
|
|
513
|
+
// Reported: `sim_scroll_to` called with `target:` instead of `sel:` answered
|
|
514
|
+
// `simframe: empty step` — which names neither the tool, nor the parameter,
|
|
515
|
+
// nor even that an argument was absent, and cost a schema lookup to decode.
|
|
516
|
+
// The declared `required` list is right there, so the check is generic
|
|
517
|
+
// rather than one guard per tool, and it says what to pass.
|
|
518
|
+
const missing = missingRequired(req.params.name, args);
|
|
519
|
+
if (missing) return { content: [text(missing)], isError: true };
|
|
378
520
|
try {
|
|
379
521
|
switch (req.params.name) {
|
|
380
522
|
case 'sim_look':
|
|
@@ -450,6 +592,81 @@ export async function serve({ device: defaultDevice, options = {} } = {}) {
|
|
|
450
592
|
await server.connect(new StdioServerTransport());
|
|
451
593
|
}
|
|
452
594
|
|
|
595
|
+
/**
|
|
596
|
+
* Crop a frame to a region of the screen and enlarge it.
|
|
597
|
+
*
|
|
598
|
+
* A whole screen at 1024px cannot answer a question about one control: a
|
|
599
|
+
* selected filter chip and an unselected one look identical at that size, and an
|
|
600
|
+
* agent shelled out to `simctl io` and PIL to crop and upscale a chip row for
|
|
601
|
+
* every check it made. The region arrives in **points** — the same coordinates
|
|
602
|
+
* the element map prints — because that is what a caller has in hand.
|
|
603
|
+
*/
|
|
604
|
+
/**
|
|
605
|
+
* Read a region argument that may not have arrived as an object.
|
|
606
|
+
*
|
|
607
|
+
* A client is free to hand a declared-object property over as a JSON string,
|
|
608
|
+
* and one did: `{"x":0,"y":60,...}` arrived as text, every field read as
|
|
609
|
+
* undefined, and the crop silently became the whole screen — reported back as
|
|
610
|
+
* `cropped to 402x874pt at 0,0`, which is a crop that did not happen described
|
|
611
|
+
* as one that did. An array is accepted too, because [x, y, width, height] is
|
|
612
|
+
* the shape anyone would try first, and it used to fail exactly as quietly.
|
|
613
|
+
*/
|
|
614
|
+
export function readRegion(region) {
|
|
615
|
+
let r = region;
|
|
616
|
+
if (typeof r === 'string') {
|
|
617
|
+
try { r = JSON.parse(r); } catch { return null; }
|
|
618
|
+
}
|
|
619
|
+
if (Array.isArray(r)) {
|
|
620
|
+
const [x, y, width, height] = r.map(Number);
|
|
621
|
+
return [x, y, width, height].every(Number.isFinite) ? { x, y, width, height } : null;
|
|
622
|
+
}
|
|
623
|
+
if (!r || typeof r !== 'object') return null;
|
|
624
|
+
const num = (...keys) => {
|
|
625
|
+
for (const k of keys) if (Number.isFinite(Number(r[k]))) return Number(r[k]);
|
|
626
|
+
return null;
|
|
627
|
+
};
|
|
628
|
+
const width = num('width', 'w');
|
|
629
|
+
const height = num('height', 'h');
|
|
630
|
+
// A region with no size is not a region. Saying so beats returning the whole
|
|
631
|
+
// screen under a caption that claims otherwise.
|
|
632
|
+
if (width == null && height == null) return null;
|
|
633
|
+
return { x: num('x') ?? 0, y: num('y') ?? 0, width, height };
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
async function cropRegion(png, region, points) {
|
|
637
|
+
try {
|
|
638
|
+
const { decodePng, encodePng, cropBitmap, scaleBitmap } = await import('./png.js');
|
|
639
|
+
const bmp = decodePng(png);
|
|
640
|
+
const pw = points?.width;
|
|
641
|
+
const ph = points?.height;
|
|
642
|
+
if (!Number.isFinite(pw) || !Number.isFinite(ph) || !pw || !ph) return { note: 'the screen point size is unknown' };
|
|
643
|
+
// The frame we hold is already downscaled for the model, so map points onto
|
|
644
|
+
// *this* bitmap rather than onto the device's native pixels.
|
|
645
|
+
const sx = bmp.width / pw;
|
|
646
|
+
const sy = bmp.height / ph;
|
|
647
|
+
const x = Number(region.x ?? 0) * sx;
|
|
648
|
+
const y = Number(region.y ?? 0) * sy;
|
|
649
|
+
const w = Number(region.width ?? region.w ?? pw) * sx;
|
|
650
|
+
const h = Number(region.height ?? region.h ?? ph) * sy;
|
|
651
|
+
if (!(w >= 1) || !(h >= 1)) return { note: 'the region has no size' };
|
|
652
|
+
const cut = cropBitmap(bmp, x, y, w, h);
|
|
653
|
+
// Enlarge to the same budget the whole screen gets, so the detail per point
|
|
654
|
+
// is the whole reason to ask for a region.
|
|
655
|
+
const long = Math.max(cut.width, cut.height);
|
|
656
|
+
const factor = Math.min(6, Math.max(1, Math.round(1024 / long)));
|
|
657
|
+
const big = factor > 1 ? scaleBitmap(cut, cut.width * factor, cut.height * factor) : cut;
|
|
658
|
+
return {
|
|
659
|
+
png: encodePng(big),
|
|
660
|
+
note: `cropped to ${Math.round(Number(region.width ?? region.w ?? pw))}x`
|
|
661
|
+
+ `${Math.round(Number(region.height ?? region.h ?? ph))}pt at `
|
|
662
|
+
+ `${Math.round(Number(region.x ?? 0))},${Math.round(Number(region.y ?? 0))}`
|
|
663
|
+
+ `, enlarged ${factor}x`,
|
|
664
|
+
};
|
|
665
|
+
} catch (err) {
|
|
666
|
+
return { note: String(err.message).split('\n')[0] };
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
|
|
453
670
|
async function look(target, args, options) {
|
|
454
671
|
const maxAge = args.maxAgeMs ?? 900;
|
|
455
672
|
// Never native resolution. An image is the expensive path by a factor of ten
|
|
@@ -472,9 +689,37 @@ async function look(target, args, options) {
|
|
|
472
689
|
const st = await api.getState(target, { since: prior, options });
|
|
473
690
|
remember(res.device.udid, res.state);
|
|
474
691
|
const lines = [header(res.device, res.state, res.ageMs), sinceLine(st.since)];
|
|
692
|
+
// Louder than a trailing note, because it invalidates the image itself rather
|
|
693
|
+
// than qualifying it.
|
|
694
|
+
if (res.frameBehindMs) {
|
|
695
|
+
lines.unshift(`WARNING: this image is ${Math.round(res.frameBehindMs / 100) / 10}s older than the screen state`
|
|
696
|
+
+ ' — it is very likely NOT what is on the device now. Read sim_ui, which is read live,'
|
|
697
|
+
+ ' or look again in a moment.');
|
|
698
|
+
}
|
|
475
699
|
const warn = livenessLine(st.live);
|
|
476
700
|
if (warn) lines.unshift(warn);
|
|
477
|
-
|
|
701
|
+
let png = res.png;
|
|
702
|
+
if (args.region) {
|
|
703
|
+
const parsed = readRegion(args.region);
|
|
704
|
+
if (!parsed) {
|
|
705
|
+
lines.push('that region could not be read (expected {"x":0,"y":260,"width":402,"height":80} in points,'
|
|
706
|
+
+ ' or [x, y, width, height]) — this is the whole screen');
|
|
707
|
+
}
|
|
708
|
+
// The point size, which the frame does not carry: `state.width/height` are
|
|
709
|
+
// the *captured frame's* pixels (322x700 here), not the screen's points
|
|
710
|
+
// (402x874). Scaling by them gave a 1:1 ratio, so a crop at y=760 clamped
|
|
711
|
+
// to a single pixel row and returned a 119-byte image — which looked like
|
|
712
|
+
// it had worked. `screenIdentity` answers it in ~17ms warm.
|
|
713
|
+
const geo = await api.screenIdentity(target, { options, confirmNovel: false }).catch(() => null);
|
|
714
|
+
const cropped = parsed ? await cropRegion(png, parsed, geo?.points) : { note: null };
|
|
715
|
+
if (cropped.png) {
|
|
716
|
+
png = cropped.png;
|
|
717
|
+
lines.push(cropped.note);
|
|
718
|
+
} else if (cropped.note) {
|
|
719
|
+
lines.push(`could not crop that region (${cropped.note}) — this is the whole screen`);
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
return { content: [text(lines.filter(Boolean).join('\n')), image(png)] };
|
|
478
723
|
}
|
|
479
724
|
|
|
480
725
|
async function state(target, args, options) {
|
|
@@ -635,10 +880,38 @@ async function recall(target, args, options) {
|
|
|
635
880
|
* is free. Reading the screen a second time to render it was the whole cost of
|
|
636
881
|
* a text-first surface, and it does not have to be paid.
|
|
637
882
|
*/
|
|
883
|
+
/**
|
|
884
|
+
* The map at the end of an action, and why it is re-read rather than recalled.
|
|
885
|
+
*
|
|
886
|
+
* It used to be rendered from whatever `screenIdentity` had in hand during
|
|
887
|
+
* verification, which is memory-first by design — so the trailing map could
|
|
888
|
+
* describe the screen as it was seconds earlier. Reported from a real session:
|
|
889
|
+
* *"do's trailing dump is still stale, so I still ran `ui --refresh` after
|
|
890
|
+
* nearly every call — that remains the biggest speed tax."*
|
|
891
|
+
*
|
|
892
|
+
* Saying how old it was (the header does) turned out not to be enough: an agent
|
|
893
|
+
* that cannot trust the map spends a turn re-reading it, and a turn is the
|
|
894
|
+
* expensive unit here. So an action pays one perception pass — a few hundred
|
|
895
|
+
* milliseconds, locally, once — to save a model round trip. That is the whole
|
|
896
|
+
* trade this phase is about, and it is the right way round.
|
|
897
|
+
*
|
|
898
|
+
* `refresh: false` is still available for the read-only tools, where the caller
|
|
899
|
+
* asked for a map and can ask again.
|
|
900
|
+
*/
|
|
638
901
|
async function mapFrom(target, options, identity, extra = {}) {
|
|
639
902
|
try {
|
|
640
|
-
const
|
|
641
|
-
|
|
903
|
+
const fresh = extra.refresh !== false;
|
|
904
|
+
const m = await view.screenMap(target, {
|
|
905
|
+
options,
|
|
906
|
+
refresh: fresh,
|
|
907
|
+
identity: fresh ? undefined : (identity?.entry ? identity : undefined),
|
|
908
|
+
});
|
|
909
|
+
const rendered = view.render({ ...m, ...extra });
|
|
910
|
+
// The line that decides whether the model stops to think. Everything it
|
|
911
|
+
// needs is already computed for the map above it, so this costs nothing —
|
|
912
|
+
// and "nothing here needs you" is a thing only the daemon can say.
|
|
913
|
+
if (extra.hint === false) return rendered;
|
|
914
|
+
return `${rendered}\n${view.hintFor(m, { flowOk: extra.flowOk, escalated: extra.escalated })}`;
|
|
642
915
|
} catch (err) {
|
|
643
916
|
return `(could not read the screen: ${err.message})`;
|
|
644
917
|
}
|
|
@@ -679,10 +952,29 @@ async function doScript(target, args, options) {
|
|
|
679
952
|
stableMs: args.stableMs,
|
|
680
953
|
timeoutMs: args.timeoutMs,
|
|
681
954
|
continueOnError: args.continueOnError,
|
|
955
|
+
// The plan's briefing for its own first responder. Ignored when no
|
|
956
|
+
// supervisor is enabled, so passing it is always safe.
|
|
957
|
+
supervise: args.supervise,
|
|
682
958
|
options,
|
|
683
959
|
});
|
|
684
960
|
|
|
685
961
|
const lines = stepLines(res);
|
|
962
|
+
// A map read without settling is not a reading of the screen you are now on,
|
|
963
|
+
// and it used to arrive looking exactly like one. Reported: two taps under
|
|
964
|
+
// `autoSettle:false` returned a map showing nothing had happened, so the
|
|
965
|
+
// agent moved on — the page had in fact zoomed all the way out, and they only
|
|
966
|
+
// found out two calls later when an unrelated failure printed a real map.
|
|
967
|
+
if (args.autoSettle === false) {
|
|
968
|
+
lines.push('autoSettle was off, so the map below was read without waiting for the last action to finish'
|
|
969
|
+
+ ' — it may describe the screen before that action landed. Re-read (sim_ui) before acting on it.');
|
|
970
|
+
}
|
|
971
|
+
// Every local ruling is reported, because a wrong one has to be correctable
|
|
972
|
+
// rather than mysterious — and the model's own stated reason is shown as its
|
|
973
|
+
// claim, not as the ground for what happened.
|
|
974
|
+
for (const s_ of res.supervisions ?? []) {
|
|
975
|
+
lines.push(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
|
|
976
|
+
+ (s_.reason ? ` (it said: "${s_.reason}")` : ''));
|
|
977
|
+
}
|
|
686
978
|
if (args.saveAs) {
|
|
687
979
|
const saved = navigate.saveFlow(res.device.udid, args.saveAs, res);
|
|
688
980
|
lines.push(
|
|
@@ -691,7 +983,12 @@ async function doScript(target, args, options) {
|
|
|
691
983
|
: `NOT saved as "${args.saveAs}": ${saved.reason}${saved.verdicts ? ` (${saved.verdicts.join(', ')})` : ''}`,
|
|
692
984
|
);
|
|
693
985
|
}
|
|
694
|
-
|
|
986
|
+
const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
|
|
987
|
+
lines.push('', await mapFrom(target, options, res.endScreen, {
|
|
988
|
+
verdictLine: verdictLineFor(res.results),
|
|
989
|
+
flowOk: res.ok,
|
|
990
|
+
escalated,
|
|
991
|
+
}));
|
|
695
992
|
|
|
696
993
|
const content = [text(lines.join('\n'))];
|
|
697
994
|
// Images only when explicitly asked for. A frame attached to every flow was
|
|
@@ -785,7 +1082,9 @@ async function ui(target, args, options) {
|
|
|
785
1082
|
refresh: args.refresh,
|
|
786
1083
|
});
|
|
787
1084
|
if (m.identity.state) remember(m.device.udid, m.identity.state);
|
|
788
|
-
|
|
1085
|
+
// The hint belongs here too: the agent has just looked, so "you do not need
|
|
1086
|
+
// to look again" is exactly the thing worth saying at this moment.
|
|
1087
|
+
return { content: [text(`${m.text}\n${view.hintFor(m)}`)] };
|
|
789
1088
|
}
|
|
790
1089
|
|
|
791
1090
|
async function find(target, args, options) {
|
|
@@ -840,7 +1139,9 @@ function listStateDirs() {
|
|
|
840
1139
|
async function devices() {
|
|
841
1140
|
const booted = await bootedDevices();
|
|
842
1141
|
if (!booted.length) return { content: [text('no booted devices')] };
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
1142
|
+
// Noticing a name collision here is what lets every later header disambiguate
|
|
1143
|
+
// itself, and it costs nothing: this listing is already being made.
|
|
1144
|
+
const clash = noteBooted(booted);
|
|
1145
|
+
const list = booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
|
|
1146
|
+
return { content: [text(clash ? `${clash}\n\n${list}` : list)] };
|
|
846
1147
|
}
|