simframe 0.11.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.js CHANGED
@@ -303,6 +303,18 @@ export const STALE_FRAME_MS = 2500;
303
303
  */
304
304
  export const MEMORY_SETTLE_MS = 250;
305
305
 
306
+ /**
307
+ * How well a relabelled ref must match before it is acted on.
308
+ *
309
+ * Above `matching.MINIMUM_SCORE` (0.45) on purpose: that floor is for a label
310
+ * the caller wrote, and this is a label simframe substituted after refusing
311
+ * their `#n`. A fuzzy name match returns `similarity * 0.72` and a prefix match
312
+ * is scaled by its coverage, so neither can reach 0.8 on the name alone — which
313
+ * makes this "the label matched nearly exactly", not a tuned constant. The
314
+ * recovery that made this necessary scored 0.64.
315
+ */
316
+ export const RELABEL_MIN_SCORE = 0.8;
317
+
306
318
  /** Below this a "change" is a clock digit or a caret, not a new screen. */
307
319
  export const MINOR_CHANGE = 0.004;
308
320
  export const MAJOR_CHANGE = 0.03;
@@ -1251,6 +1263,35 @@ async function locateWith(
1251
1263
  second.staleLabel = err.staleLabel;
1252
1264
  throw second;
1253
1265
  }
1266
+ // A recovery is held to a higher bar than the lookup it stands in for,
1267
+ // and to one the caller never has to think about.
1268
+ //
1269
+ // `MINIMUM_SCORE` (0.45) is the bar for a label the caller actually
1270
+ // wrote. This label is one *we substituted on their behalf* after
1271
+ // refusing their `#n`, so a weak match here is not "close enough" — it is
1272
+ // us choosing a target nobody named. The reported wrong answer scored
1273
+ // **0.64** and cleared the ordinary floor comfortably.
1274
+ //
1275
+ // 0.8 is structural rather than fitted to that incident: a fuzzy name
1276
+ // match returns `similarity * 0.72` and a prefix match is scaled by its
1277
+ // coverage, so neither reaches 0.8 on the name alone. Only a near-exact
1278
+ // name does. And the region check needs no tuned number at all — if the
1279
+ // map would not offer this target, a recovery may not silently pick it.
1280
+ const region = again.target?.region ?? 'content';
1281
+ const weak = Number.isFinite(again.score) && again.score < RELABEL_MIN_SCORE;
1282
+ if (weak || !regions.offerable(region)) {
1283
+ err.message = `#${selector.ref} cannot be trusted here, and "${err.staleLabel}" was not`
1284
+ + ' safely re-findable either:'
1285
+ + (weak ? ` the best match scored ${again.score.toFixed(2)}, below the ${RELABEL_MIN_SCORE} a`
1286
+ + ' relabel needs (a number you did not ask for may not become a tap on a guess).' : '')
1287
+ + (!regions.offerable(region) ? ` the best match sits in the ${region}, which sim_ui does not`
1288
+ + ' offer as something to act on.' : '')
1289
+ + ' Read the screen again (sim_ui) and name the target.';
1290
+ err.staleRef = true;
1291
+ err.staleLabel = err.staleLabel;
1292
+ err.relabelRefused = { score: again.score ?? null, region };
1293
+ throw err;
1294
+ }
1254
1295
  return {
1255
1296
  ...again,
1256
1297
  from: 'ref-relabelled',
@@ -94,7 +94,11 @@ export function lineServer({ ensureBinary, what }) {
94
94
  try { msg = JSON.parse(line); } catch { continue; }
95
95
  if (!settled) {
96
96
  settled = true;
97
- if (msg.ready) resolve({ ok: true, child, waiters });
97
+ // The ready line may carry facts about the helper worth keeping —
98
+ // the model's context window, for one, which used to be a constant
99
+ // we repeated in comments. Passed through rather than parsed here,
100
+ // because this file knows about lines and not about models.
101
+ if (msg.ready) resolve({ ok: true, child, waiters, hello: msg });
98
102
  else resolve({ ok: false, reason: msg.unavailable ?? `the ${what} did not become ready` });
99
103
  continue;
100
104
  }
@@ -145,7 +149,9 @@ export function lineServer({ ensureBinary, what }) {
145
149
  },
146
150
  async status() {
147
151
  const live = await open();
148
- return live?.ok ? { ok: true } : { ok: false, reason: live?.reason ?? 'unavailable' };
152
+ return live?.ok
153
+ ? { ok: true, hello: live.hello ?? {} }
154
+ : { ok: false, reason: live?.reason ?? 'unavailable' };
149
155
  },
150
156
  close() {
151
157
  try { session?.child?.kill(); } catch { /* already gone */ }
package/src/mcp.js CHANGED
@@ -78,13 +78,22 @@ const deviceProp = {
78
78
  };
79
79
 
80
80
  /**
81
- * One selector grammar everywhere.
81
+ * One selector grammar everywhere, and the order is the recommendation.
82
82
  *
83
- * `#3` is the cheapest thing a caller can say and the least ambiguous, because
84
- * simframe numbered it; a bare phrase is resolved by intent, which is more
85
- * forgiving and occasionally has to ask which one was meant.
83
+ * It used to lead with `#3` and call it "cheapest and unambiguous". Four peer
84
+ * rounds running reported the opposite: intent resolution worked every time,
85
+ * while refs renumbered underneath them and were only safe inside the round
86
+ * trip that issued them. The README was corrected and these descriptions were
87
+ * not, which is the half a caller actually reads.
88
+ *
89
+ * "Unambiguous" was also the wrong word for it. A ref is exact about which
90
+ * element simframe meant and says nothing about whether that element is still
91
+ * there — the failure mode that needed `staleKind` to tell a moved layout from
92
+ * a different screen, and then a score floor and a region check on top of that
93
+ * before a relabelled ref could be trusted. A label carries its own evidence;
94
+ * a number carries none.
86
95
  */
87
- const SELECTOR = 'Selector: "#3" (a number from the last screen map — cheapest and unambiguous), a label or phrase like "Save" or "the Assets tab" (resolved by intent), or "@120,400" for raw point coordinates.';
96
+ const SELECTOR = 'Selector: a label or phrase like "Save" or "the Assets tab" or "back" (resolved by intent — verbs, typos, synonyms, icon-only controls: START HERE), "#3" (a number from the last screen map — exact, but only inside the round trip that numbered it), or "@120,400" for raw point coordinates (last resort: it cannot tell you it missed).';
88
97
 
89
98
  const selectorProp = (what = 'What to act on') => ({
90
99
  sel: { type: 'string', description: `${what}. ${SELECTOR}` },
@@ -94,7 +103,7 @@ const TOOLS = [
94
103
  {
95
104
  name: 'sim_ui',
96
105
  description:
97
- 'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Whatever it calls #3, you can tap as "#3". Start here, never with sim_look.',
106
+ 'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Act on what it shows by NAME — whatever it calls "General", you can tap as "General"; the #3 numbers are exact but only until the screen moves. Start here, never with sim_look.',
98
107
  inputSchema: {
99
108
  type: 'object',
100
109
  properties: {
@@ -123,7 +132,7 @@ const TOOLS = [
123
132
  steps: {
124
133
  type: 'array',
125
134
  description:
126
- 'Ordered steps. Every selector below accepts "#3" | "Save" | "@120,400". Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
135
+ 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
127
136
  items: { type: 'object' },
128
137
  },
129
138
  autoSettle: {
package/src/metrics.js CHANGED
@@ -24,6 +24,21 @@ export const REASONS = [
24
24
 
25
25
  export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
26
26
 
27
+ /**
28
+ * What the executor observed after a supervisor ruling — the field that makes a
29
+ * ruling scoreable rather than merely recorded.
30
+ *
31
+ * A ruling on its own says what the supervisor thought. Three items on the
32
+ * deferred list (101's p95 replay, 106's cost of the steps a `stop` skipped,
33
+ * 96's response variable) need what happened *next*, and until now nothing
34
+ * persisted a ruling at all: they went into a `supervisions` array on the
35
+ * result and died with the process. Three rulings had ever existed anywhere.
36
+ *
37
+ * Closed, and mandatory, for the same reason `REASONS` is: a vocabulary that
38
+ * admits "other" collects a pile of "other".
39
+ */
40
+ export const RULING_OUTCOMES = ['recovered', 'still_failed', 'stopped', 'no_ruling'];
41
+
27
42
  /**
28
43
  * Which faculty would have removed this escalation.
29
44
  *
@@ -57,6 +72,7 @@ function metricPaths(udid) {
57
72
  dir,
58
73
  escalations: path.join(dir, 'escalations.jsonl'),
59
74
  flows: path.join(dir, 'flows.jsonl'),
75
+ supervisions: path.join(dir, 'supervisions.jsonl'),
60
76
  baselines: path.join(dir, 'baselines'),
61
77
  };
62
78
  }
@@ -109,6 +125,106 @@ export function readJsonl(file, { limit } = {}) {
109
125
 
110
126
  export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
111
127
  export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
128
+ export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supervisions, opts);
129
+
130
+ /**
131
+ * Write down one supervisor ruling and what came of it.
132
+ *
133
+ * The fields are chosen so the three items waiting on this can be answered
134
+ * **offline, from the log**, rather than by another device run:
135
+ *
136
+ * - `screen`/`edge` — which `(screen_hash, action)` the ruling was about, so
137
+ * rulings can be grouped by edge the way the graph groups timings.
138
+ * - `p95`/`samples` — what the graph knew about that edge **at the moment of
139
+ * the ruling**. Recording it now rather than looking it up later is the
140
+ * difference between 101 being arithmetic and 101 being archaeology: the
141
+ * graph keeps learning, so a p95 read next week is not the p95 the
142
+ * supervisor was implicitly competing with.
143
+ * - `stillMs` — the same input the supervisor got, so a replay sees what it saw.
144
+ * - `expect` — the plan's own note, because a ruling made with a briefing and
145
+ * one made without are not the same measurement (96's critical arm).
146
+ * - `outcome` — what the executor observed afterwards. Mandatory.
147
+ *
148
+ * `from` separates a deterministic rule from a model answer. Both are rulings
149
+ * and both have outcomes, but a comparison that mixed them would credit the
150
+ * model for what a two-line rule decided.
151
+ */
152
+ export function recordSupervision(udid, {
153
+ session, index, step, edge, screen, decision, from, reason, ms,
154
+ stillMs, p95, samples, expect, failure, outcome,
155
+ }) {
156
+ // Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
157
+ // difference is deliberate: this is called from inside a flow's failure
158
+ // handler, where a throw would turn a recoverable step failure into a crash.
159
+ // It still lands in `lastWriteError`, which `simframe escalations` prints.
160
+ if (!RULING_OUTCOMES.includes(outcome)) {
161
+ lastWriteError = `supervision outcome must be one of ${RULING_OUTCOMES.join('/')}, got ${JSON.stringify(outcome)}`;
162
+ return false;
163
+ }
164
+ return appendJsonl(metricPaths(udid).supervisions, {
165
+ timestamp: new Date().toISOString(),
166
+ session_id: session ?? SESSION_ID,
167
+ client: CLIENT,
168
+ step_index: index ?? null,
169
+ step: step ?? null,
170
+ edge: edge ?? null,
171
+ screen_fingerprint: screen ?? null,
172
+ decision,
173
+ from: from ?? 'model',
174
+ // Recorded, never presented as the ground for what happened: the supervisor
175
+ // has returned a correct decision with a reason citing a rule that did not
176
+ // apply. Keeping it is how that stays measurable instead of anecdotal.
177
+ reason: reason ? String(reason).slice(0, 200) : null,
178
+ latency_ms: Number.isFinite(ms) ? Math.round(ms) : null,
179
+ still_ms: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
180
+ edge_p95_ms: Number.isFinite(p95) ? Math.round(p95) : null,
181
+ edge_samples: Number.isFinite(samples) ? samples : null,
182
+ expect: expect ? String(expect).slice(0, 200) : null,
183
+ failure: failure ? String(failure).slice(0, 300) : null,
184
+ outcome,
185
+ });
186
+ }
187
+
188
+ /**
189
+ * What the ruling log currently says, and whether it can yet answer 101.
190
+ *
191
+ * Deliberately counts rather than scores. Item 101 asks which `wait`/`retry`
192
+ * rulings a p95-per-edge lookup would have got right, and that is a separate
193
+ * piece of work; what this answers is the question that comes first and was
194
+ * embarrassing to get wrong once already — **is there a population to measure
195
+ * at all, and do its rows carry the fields the measurement needs.** `p95_known`
196
+ * is that readiness check: a ruling recorded on an edge the graph had never
197
+ * timed cannot take part in the comparison, however many of them there are.
198
+ */
199
+ export function supervisionBreakdown(records) {
200
+ const by = (key) => {
201
+ const out = {};
202
+ for (const r of records) {
203
+ const k = r[key] ?? 'unknown';
204
+ out[k] = (out[k] ?? 0) + 1;
205
+ }
206
+ return out;
207
+ };
208
+ const matrix = {};
209
+ for (const r of records) {
210
+ const k = `${r.decision ?? 'unknown'} -> ${r.outcome ?? 'unknown'}`;
211
+ matrix[k] = (matrix[k] ?? 0) + 1;
212
+ }
213
+ const timed = records.filter((r) => Number.isFinite(r.edge_p95_ms));
214
+ const latencies = records.map((r) => r.latency_ms).filter((n) => Number.isFinite(n)).sort((a, b) => a - b);
215
+ return {
216
+ total: records.length,
217
+ by_decision: by('decision'),
218
+ by_outcome: by('outcome'),
219
+ by_from: by('from'),
220
+ decision_to_outcome: matrix,
221
+ // Readiness for 101, not a result for it.
222
+ p95_known: timed.length,
223
+ p95_unknown: records.length - timed.length,
224
+ sessions: [...new Set(records.map((r) => r.session_id).filter(Boolean))],
225
+ median_latency_ms: latencies.length ? latencies[latencies.length >> 1] : null,
226
+ };
227
+ }
112
228
 
113
229
  /**
114
230
  * Mark an error as an escalation with a reason, at the site that knows why.
@@ -996,6 +996,28 @@ function capabilities() {
996
996
  };
997
997
  }
998
998
 
999
+ /**
1000
+ * Not implemented here, and it says so in its own vocabulary.
1001
+ *
1002
+ * `revive` exists for a display that has stopped rendering — a CoreSimulator
1003
+ * pathology. An emulator's failure modes are its own (the console port going
1004
+ * away, adb losing the device) and the remedy is not the same sequence, so
1005
+ * borrowing the other platform's answer would be the mistake `doctor` made when
1006
+ * it reported "input driver: idb" for an emulator: a claim about a tool that has
1007
+ * never spoken to an Android device.
1008
+ *
1009
+ * When an emulator wedge is characterised rather than guessed at, this becomes
1010
+ * a real implementation. Until then the honest answer is that this backend does
1011
+ * not have the layer.
1012
+ */
1013
+ async function restartDevice(serial) {
1014
+ throw new Error(
1015
+ `simframe cannot yet restart an emulator (${serial}) — that remedy is written for a`
1016
+ + ' simulator display that stopped rendering, and an emulator fails differently.'
1017
+ + ' Restart it from Android Studio, or `adb -s <serial> emu kill` and relaunch.',
1018
+ );
1019
+ }
1020
+
999
1021
  /** @type {import('./index.js').Platform} */
1000
1022
  export const platform = {
1001
1023
  id: 'android',
@@ -1011,6 +1033,7 @@ export const platform = {
1011
1033
  launchApp,
1012
1034
  terminateApp,
1013
1035
  openUrl,
1036
+ restartDevice,
1014
1037
  setPermission,
1015
1038
  setPasteboard,
1016
1039
  getPasteboard,
@@ -61,7 +61,7 @@ export const PLATFORM_SURFACE = Object.freeze([
61
61
  'id', 'deviceNoun',
62
62
  'listDevices', 'bootedDevices', 'resolveDevice', 'isBootedSync', 'ownsUdid',
63
63
  'geometry', 'inputDriver',
64
- 'screenshot', 'launchApp', 'terminateApp', 'openUrl',
64
+ 'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
66
  'bootedAt',
67
67
  ]);
@@ -260,6 +260,16 @@ export const inputDriverFor = (udid) => platformFor(udid).inputDriver(udid);
260
260
  */
261
261
  export const capabilitiesFor = (udid) => platformFor(udid).capabilities();
262
262
 
263
+ /**
264
+ * Power-cycle a device, on the backend that owns it.
265
+ *
266
+ * Reached only from `simframe revive`, never from the capture loop: the loop
267
+ * detects a stalled display and reports it, and restarting is the operator's
268
+ * call. A backend that does not have this remedy throws in its own terms rather
269
+ * than borrowing the other's.
270
+ */
271
+ export const restartDevice = (udid) => platformFor(udid).restartDevice(udid);
272
+
263
273
  /** What `simframe doctor` should check: each registered backend's own toolchain. */
264
274
  export function toolchainChecks() {
265
275
  return backends().flatMap((backend) => backend.toolchain());
@@ -66,6 +66,19 @@ async function bootedDevices(opts) {
66
66
  */
67
67
  async function resolveDevice(query, opts) {
68
68
  const all = await listDevices(opts);
69
+ return pickDevice(query, all);
70
+ }
71
+
72
+ /**
73
+ * Which device a query means, given the whole list.
74
+ *
75
+ * Separated from the listing so the decision can be **tested** rather than
76
+ * reasoned about, because two peer rounds in a row reported being handed the
77
+ * wrong device and both were decided here. The same move as `decisionOf` in the
78
+ * supervisor: the one line where a wrong answer is expensive should be a
79
+ * function somebody can call with adversarial input.
80
+ */
81
+ export function pickDevice(query, all) {
69
82
  const booted = all.filter((d) => d.state === 'Booted');
70
83
  if (!query) {
71
84
  if (booted.length === 0) throw new Error('no booted simulator (open Simulator.app or run `xcrun simctl boot <udid>`)');
@@ -98,8 +111,40 @@ async function resolveDevice(query, opts) {
98
111
  const q = query.toLowerCase();
99
112
  const pools = [booted, all];
100
113
  for (const pool of pools) {
101
- const exact = pool.find((d) => d.udid.toLowerCase() === q || d.name.toLowerCase() === q);
102
- if (exact) return exact;
114
+ // A UDID is unique, so an exact UDID match needs no further thought.
115
+ const byUdid = pool.find((d) => d.udid.toLowerCase() === q);
116
+ if (byUdid) return byUdid;
117
+ // A **name is not unique**, and this branch used to treat it as though it
118
+ // were: one `find` over both fields returned whichever device the list
119
+ // happened to put first, and short-circuited past the ambiguity guard
120
+ // below. Item 83 recorded that two booted devices on this machine are both
121
+ // called "iPhone 17 Pro" and answered it by warning in `sim_devices` and
122
+ // printing a UDID prefix in headers — leaving the resolver, which is where
123
+ // the choice is actually made, untouched.
124
+ //
125
+ // What that cost, reported from a three-hour session on a real app: a
126
+ // caller passed the shared name, read a screen that was "38ms old" and an
127
+ // hour wrong, concluded the app had signed itself out, and abandoned a
128
+ // verification run that was fine. The frame was fresh — it was the *other*
129
+ // device's, idling on a login screen. `refresh: true` returned the matching
130
+ // tree because it refreshed the same wrong device. Two independent-looking
131
+ // sources agreeing with each other and both wrong.
132
+ //
133
+ // So a name that names two devices is an ambiguity, exactly like a partial
134
+ // match that hits two, and it refuses for the same reason: the cost of
135
+ // guessing wrong is reading somebody else's screen and believing it.
136
+ const byName = pool.filter((d) => d.name.toLowerCase() === q);
137
+ if (byName.length === 1) return byName[0];
138
+ if (byName.length > 1) {
139
+ throw Object.assign(
140
+ new Error(
141
+ `"${query}" is the name of ${byName.length} devices: `
142
+ + `${byName.map((d) => d.udid).join(', ')} — a name cannot say which one you mean, `
143
+ + 'so pass the UDID',
144
+ ),
145
+ { ambiguous: true },
146
+ );
147
+ }
103
148
  const partial = pool.filter((d) => d.name.toLowerCase().includes(q));
104
149
  if (partial.length === 1) return partial[0];
105
150
  if (partial.length > 1) {
@@ -189,6 +234,28 @@ async function openUrl(udid, url) {
189
234
  await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
190
235
  }
191
236
 
237
+ /**
238
+ * Shut a device down and bring it back, waiting for the boot to finish.
239
+ *
240
+ * The remedy for a display that has stopped rendering, which the capture loop
241
+ * can detect and must not perform: it reports `stalled` and stops, because a
242
+ * capture loop that rebooted the device it was watching would be a tool
243
+ * reaching for the mains when a reading looks wrong. This is the operator's
244
+ * decision, reached by `simframe revive`.
245
+ *
246
+ * `bootstatus -b` and not `boot`, for the reason it is used in CI: `boot`
247
+ * returns before the device is usable, and everything downstream then races the
248
+ * boot. Timeboxed generously — a cold boot on a busy machine is slow, and a
249
+ * boot that never finishes should fail here with a reason rather than as a
250
+ * puzzle further down.
251
+ */
252
+ async function restartDevice(udid) {
253
+ // Tolerated: a device that is already off cannot be shut down, and that is
254
+ // the state this command is most often reached from.
255
+ await run('xcrun', ['simctl', 'shutdown', udid], { timeout: 60_000 }).catch(() => null);
256
+ await run('xcrun', ['simctl', 'bootstatus', udid, '-b'], { timeout: 240_000 });
257
+ }
258
+
192
259
  const PERMISSION_SERVICES = [
193
260
  'all', 'calendar', 'contacts-limited', 'contacts', 'location', 'location-always',
194
261
  'photos-add', 'photos', 'media-library', 'microphone', 'motion', 'reminders', 'siri',
@@ -347,6 +414,7 @@ export const platform = {
347
414
  launchApp,
348
415
  terminateApp,
349
416
  openUrl,
417
+ restartDevice,
350
418
  setPermission,
351
419
  setPasteboard,
352
420
  permissionServices: () => PERMISSION_SERVICES,
package/src/regions.js CHANGED
@@ -24,6 +24,27 @@ export const REGIONS = [
24
24
  'content',
25
25
  ];
26
26
 
27
+ /**
28
+ * Regions the screen map will not offer as something to act on.
29
+ *
30
+ * The status bar says the time and the battery level. It is on every screen, it
31
+ * is never what anybody wants to tap, and it costs a row every time — so
32
+ * `sim_ui` hides it.
33
+ *
34
+ * It lives here rather than in `view.js`, where it started, because it is not
35
+ * only a presentation rule. A target the map refuses to *show* must also be a
36
+ * target nothing may resolve onto *behind the caller's back*, and the case that
37
+ * proved it was exactly that: a stale `#1` numbered "Reminders" in Reminders,
38
+ * re-resolved in Contacts onto the status-bar back-to-app breadcrumb
39
+ * "• Reminders", scored 0.64 and handed back a tap point at (47,40) — a place
40
+ * the map would never have put in front of anybody. One rule, one home, both
41
+ * readers.
42
+ */
43
+ const UNOFFERED_REGIONS = new Set(['status-bar']);
44
+
45
+ /** Would the screen map offer a target in this region? */
46
+ export const offerable = (region) => !UNOFFERED_REGIONS.has(region);
47
+
27
48
  /**
28
49
  * The status bar stays positional, and deliberately.
29
50
  *
@@ -63,6 +84,46 @@ const BOTTOM_CHROME_LIMIT = 0.82;
63
84
  * screen sets its own scale.
64
85
  */
65
86
  const MIN_BOUNDARY_GAP_PT = 10;
87
+ /**
88
+ * How wide a lone row may be and still be a screen's title rather than its
89
+ * first paragraph.
90
+ *
91
+ * 0.6 of the screen, measured: the Settings root's large title is 133 pt of
92
+ * 402 (0.33), while Reminders' empty-state heading "Welcome to Reminders" is
93
+ * 326 pt (0.81) and is content — it describes the screen instead of naming it.
94
+ * A title is a name and names are short.
95
+ */
96
+ const LARGE_TITLE_MAX_WIDTH_FRACTION = 0.6;
97
+ /**
98
+ * How far below the status bar a large title can start.
99
+ *
100
+ * iOS draws one at a **system** offset, not an app-chosen one, so this is a
101
+ * bound on a platform constant rather than a tuned threshold. Measured on the
102
+ * bench device: the Settings root's title starts 63-79 pt below the status bar
103
+ * depending on which sensor reports its box, while example.com's `<h1>` — page
104
+ * *content* that merely happens to be the first row, because Safari on iOS puts
105
+ * its chrome at the bottom — starts **122 pt** down.
106
+ *
107
+ * Without this bound the rule promoted that `<h1>` to chrome and "example
108
+ * domain" entered the screen's identity. Pulling page content into identity is
109
+ * the exact failure this module has been bitten by twice (a phantom keyboard,
110
+ * and content that merely fell into a band), so the bound is not optional.
111
+ */
112
+ const LARGE_TITLE_MAX_INSET_PT = 96;
113
+ /**
114
+ * And the least it can be, before it is just the next row.
115
+ *
116
+ * Absolute, like the maximum, and for the same reason: the inset is drawn by
117
+ * the system, so it is not a function of what the screen contains. The first
118
+ * version tested it against the screen's *median row gap* — and the testbed
119
+ * caught that on its first day, with two screens of the same app. A list of 24
120
+ * rows has a median gap of 0 and the rule fired; a list of 4 rows above a tab
121
+ * bar has a median gap of **414**, because the empty area counts as a gap, and
122
+ * the rule did not. Same title, same inset of 62.9pt, opposite answers — so one
123
+ * screen had a name in its identity and the other did not, and the graph then
124
+ * called them the same screen at 0.50 similarity.
125
+ */
126
+ const LARGE_TITLE_MIN_INSET_PT = 24;
66
127
  const BOUNDARY_GAP_FACTOR = 1.9;
67
128
 
68
129
  /** A tab bar is several things spread across the width, not one thing at the bottom. */
@@ -177,6 +238,50 @@ export function bands(elements, screen) {
177
238
  }
178
239
  }
179
240
 
241
+ // --- a large title, which has no gap under it to be found by.
242
+ //
243
+ // The loop above identifies top chrome by the whitespace *beneath* it, and an
244
+ // iOS large title is drawn tight against the content it heads: measured on
245
+ // the Settings root, 79 pt of inset above it and **5.3 pt** below, against a
246
+ // bar of `max(10, typical * 1.9)` = 66.5. No threshold reaches that, so the
247
+ // title fell into `content` and was discarded as content — leaving the screen
248
+ // with **no name at all** in its fingerprint, in either sensor mode.
249
+ //
250
+ // That is not cosmetic. Chrome labels are the only text identity keeps, and
251
+ // `fingerprint.js` names the consequence: two list screens with identical
252
+ // structure differ by their title and nothing else says so. A nameless screen
253
+ // is pure geometry, and on a hosted runner two sparse nameless readings
254
+ // matched exactly — one screen's hash for two screens.
255
+ //
256
+ // So it is found by the inset *above* it instead, which is the half iOS does
257
+ // provide. A large title sits alone, narrow, high, under a generous gap; a
258
+ // compact bar is the mirror image of that (20-33 pt above, 118-134 below) and
259
+ // is already caught by the loop. Deliberately not keyed on the `Heading`
260
+ // role: OCR has no roles, and the reading that actually collided was
261
+ // OCR-only, so a role test would work only in the case that does not fail.
262
+ //
263
+ // Measured across all 17 perception fixtures before being written here: it
264
+ // changes exactly one of them, the Settings root.
265
+ if (!navBarBottom) {
266
+ const first = rows.findIndex((r) => r.top >= statusBarBottom - 1);
267
+ const row = first >= 0 ? rows[first] : null;
268
+ if (
269
+ row
270
+ && first + 1 < rows.length
271
+ && row.items.length === 1
272
+ && (row.items[0].frame?.width ?? 0) <= screen.width * LARGE_TITLE_MAX_WIDTH_FRACTION
273
+ && row.bottom <= screen.height * TOP_CHROME_LIMIT
274
+ // A real separation from the status bar, but a system-sized one: far
275
+ // enough to be an inset, near enough to still be the app's own title.
276
+ // Both bounds absolute — see LARGE_TITLE_MIN_INSET_PT for what keying the
277
+ // lower one on the screen's own row rhythm cost.
278
+ && row.top - statusBarBottom >= LARGE_TITLE_MIN_INSET_PT
279
+ && row.top - statusBarBottom <= LARGE_TITLE_MAX_INSET_PT
280
+ ) {
281
+ navBarBottom = row.bottom;
282
+ }
283
+ }
284
+
180
285
  // --- bottom chrome. One row: a tab bar is one row by construction, and the
181
286
  // gap above it is what separates it from the list it floats over.
182
287
  let tabBarTop = Infinity;
package/src/supervisor.js CHANGED
@@ -45,6 +45,30 @@ const helper = lineServer({
45
45
 
46
46
  export const DECISIONS = new Set(['wait', 'retry', 'stop']);
47
47
 
48
+ /**
49
+ * The vocabulary gate, as a function so it can be *tested* rather than grepped.
50
+ *
51
+ * This one line is the safety property: the supervisor cannot invent a step,
52
+ * skip one, substitute a target or continue past an unexpected screen, because
53
+ * those are not words it can say. Anything outside the three is not a decision
54
+ * and becomes `null`, which means "behave as if there is no supervisor".
55
+ *
56
+ * It was previously inline, and the test that guarded it matched the source
57
+ * text — so it broke when the branch grew an else, with nothing actually wrong.
58
+ * A property this important deserves an assertion that runs it.
59
+ */
60
+ export function decisionOf(answer) {
61
+ // A string, checked rather than coerced. `String(["wait"])` is `"wait"`, so a
62
+ // `String(...)` coercion here let `{decision: ["wait"]}` through the one gate
63
+ // that defines this component's answer space. Found the first time this
64
+ // property was *run* instead of grepped for in the source — the old test
65
+ // matched the source text of the branch and could never have caught it.
66
+ const raw = answer?.decision;
67
+ if (typeof raw !== 'string') return null;
68
+ const decision = raw.toLowerCase();
69
+ return DECISIONS.has(decision) ? decision : null;
70
+ }
71
+
48
72
  /** Which backend the caller asked for, per call first and environment second. */
49
73
  export function requested(options) {
50
74
  const raw = String(options?.supervisor ?? process.env.SIMFRAME_SUPERVISOR ?? '').trim().toLowerCase();
@@ -62,6 +86,7 @@ export function requested(options) {
62
86
  */
63
87
  export async function judge({
64
88
  goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
89
+ detail,
65
90
  } = {}) {
66
91
  if (!requested(options)) return null;
67
92
  if (!step || !failure) return null;
@@ -74,10 +99,24 @@ export async function judge({
74
99
  stillMs: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
75
100
  note: note ? String(note).slice(0, 200) : null,
76
101
  }, timeoutMs);
77
- const decision = String(answer?.decision ?? '').toLowerCase();
78
- // An answer outside the vocabulary is not a decision. Refusing it here is
79
- // what makes the three-word constraint real rather than merely documented.
80
- if (!DECISIONS.has(decision)) return null;
102
+ const decision = decisionOf(answer);
103
+ if (decision == null) {
104
+ // Why it did not answer, for the caller's log — through an out-parameter
105
+ // rather than the return value, because returning anything truthy here
106
+ // would change what the executor does. `null` means "behave as if there is
107
+ // no supervisor" and that safety property is the one thing in this file
108
+ // that must not become conditional.
109
+ //
110
+ // Before this, every failure reached the supervision log as "the
111
+ // supervisor did not answer": a timeout, a guardrail refusal and a model
112
+ // that was never installed were one indistinguishable line.
113
+ if (detail && typeof detail === 'object') {
114
+ detail.kind = answer?.kind
115
+ ?? (answer == null ? 'no answer' : answer.decision ? 'outside the vocabulary' : 'unparseable');
116
+ if (answer?.error) detail.error = String(answer.error).slice(0, 200);
117
+ }
118
+ return null;
119
+ }
81
120
  return { decision, reason: String(answer.reason ?? '').slice(0, 120), ms: answer.ms ?? null };
82
121
  }
83
122
 
@@ -101,16 +140,21 @@ export async function status(options) {
101
140
  screen: ['Probe'],
102
141
  stillMs: 5000,
103
142
  }, 6000);
104
- const decision = String(probe?.decision ?? '').toLowerCase();
105
- if (!DECISIONS.has(decision)) {
143
+ if (decisionOf(probe) == null) {
106
144
  return {
107
145
  supervisor: 'none',
108
146
  detail: 'the model loaded but did not answer a probe — it is present and not working',
109
147
  };
110
148
  }
149
+ // The window, read from the model rather than repeated from documentation.
150
+ // Worth printing: the worst case our own clipping allows measures 1,918
151
+ // tokens against it, and that ratio is the reason there is no per-call token
152
+ // budget check — see DEFERRED 99.
153
+ const ctx = live.hello?.contextSize;
111
154
  return {
112
155
  supervisor: 'apple',
113
- detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms; may only answer wait/retry/stop`,
156
+ detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms;`
157
+ + `${ctx ? ` ${ctx}-token window;` : ''} may only answer wait/retry/stop`,
114
158
  };
115
159
  }
116
160