simframe 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/cli.js CHANGED
@@ -47,6 +47,10 @@ const USAGE = `simframe — always-warm iOS Simulator frames
47
47
  simframe baseline list recorded runs per flow, and what is committed
48
48
  simframe hpi [device] Human Parity Index, per flow and overall
49
49
  simframe escalations [device] why simframe handed decisions back, by reason
50
+ (--session=<id> narrows to one agent; the
51
+ ids are listed in the output. SIMFRAME_SESSION
52
+ names one, but only at process start — an
53
+ already-running MCP server cannot pick it up)
50
54
  simframe devices list simulators
51
55
  simframe doctor check that this machine can capture
52
56
  (--strict, or SIMFRAME_STRICT=1, makes any
@@ -162,6 +166,25 @@ const num = (v, fallback) => (v == null ? fallback : Number(v));
162
166
  * script that has to parse one command's prose and another's JSON will parse
163
167
  * the prose wrong exactly once and then be trusted anyway.
164
168
  */
169
+ /**
170
+ * The machine-readable half of a failure.
171
+ *
172
+ * A refusal recognisable only by reading its prose is a refusal nobody can
173
+ * depend on. Our own CI asserted the stale-ref guard by matching three
174
+ * phrasings and went red when a fourth arrived — a *better* one, naming the
175
+ * label the number stood for. The sentence is for a person; these fields are
176
+ * the contract, and they live in one function because `find` reports its own
177
+ * failures and the top-level handler reports the rest.
178
+ */
179
+ function failureJson(err) {
180
+ return {
181
+ ok: false,
182
+ error: err.message,
183
+ reason: metrics.escalationOf(err)?.reason ?? null,
184
+ ...(err.staleRef ? { staleRef: true, staleKind: err.staleKind ?? null, staleLabel: err.staleLabel ?? null } : {}),
185
+ };
186
+ }
187
+
165
188
  function emit(flags, json, lines) {
166
189
  if (flags.json) {
167
190
  console.log(JSON.stringify(json, null, 2));
@@ -171,11 +194,26 @@ function emit(flags, json, lines) {
171
194
  if (body != null) console.log(Array.isArray(body) ? body.filter((l) => l != null).join('\n') : body);
172
195
  }
173
196
 
174
- /** The end-state screen map, rendered from a reading the flow already took. */
175
- async function mapText(device, options, identity) {
197
+ /**
198
+ * The end-state screen map, re-read rather than recalled, with its hint.
199
+ *
200
+ * Both halves were reported against Phase 11.5 and both were right. The map was
201
+ * rendered from whatever reading the flow already had, which is memory-first —
202
+ * so a trailing map could describe the screen as it was seconds ago, and the
203
+ * remedy in practice was a `ui --refresh` after nearly every call, which is a
204
+ * whole extra turn to save a few hundred milliseconds. Wrong way round.
205
+ *
206
+ * And the hint was only ever printed by the MCP server, so no CLI user could
207
+ * see it and no CLI run could test it.
208
+ */
209
+ async function mapText(device, options, identity, { flowOk = true, escalated = false, refresh = true } = {}) {
176
210
  try {
177
- const m = await view.screenMap(device, { options, identity: identity?.entry ? identity : undefined });
178
- return m.text;
211
+ const m = await view.screenMap(device, {
212
+ options,
213
+ refresh,
214
+ identity: refresh ? undefined : (identity?.entry ? identity : undefined),
215
+ });
216
+ return `${m.text}\n${view.hintFor(m, { flowOk, escalated })}`;
179
217
  } catch (err) {
180
218
  return `(could not read the screen: ${err.message})`;
181
219
  }
@@ -231,6 +269,11 @@ async function main() {
231
269
  if (flags.maxDim) options.maxDim = num(flags.maxDim);
232
270
  if (flags.ringSize) options.ringSize = num(flags.ringSize);
233
271
  if (flags.engine) options.engine = String(flags.engine);
272
+ // Per-command overrides for the two experiment knobs, so an A/B is an
273
+ // argument rather than a restart. `--sensor=ax-first`, `--planner=apple`.
274
+ if (flags.sensor) options.sensor = String(flags.sensor);
275
+ if (flags.planner) options.planner = String(flags.planner);
276
+ if (flags.supervisor) options.supervisor = String(flags.supervisor);
234
277
 
235
278
  switch (command) {
236
279
  case undefined:
@@ -533,6 +576,7 @@ async function main() {
533
576
  all: Boolean(flags.all),
534
577
  refresh: Boolean(flags.refresh),
535
578
  });
579
+ m.text = `${m.text}\n${view.hintFor(m)}`;
536
580
  emit(
537
581
  flags,
538
582
  {
@@ -585,6 +629,7 @@ async function main() {
585
629
  stableMs: num(flags.stableMs, 500),
586
630
  timeoutMs: num(flags.timeoutMs, 8000),
587
631
  continueOnError: Boolean(flags.continueOnError),
632
+ supervise: flags.supervise ? String(flags.supervise) : undefined,
588
633
  options,
589
634
  });
590
635
  const saved = flags.save
@@ -592,7 +637,16 @@ async function main() {
592
637
  : null;
593
638
  // `--map=false` arrives as the string "false"; `--no-map` as true.
594
639
  const wantMap = !flags.json && flags.noMap !== true && String(flags.map ?? 'true') !== 'false';
595
- const map = wantMap ? await mapText(flags.device, options, res.endScreen) : null;
640
+ // Every local ruling, so a wrong one is correctable rather than
641
+ // mysterious — and the model's stated reason is shown as its claim.
642
+ for (const s_ of res.supervisions ?? []) {
643
+ console.log(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
644
+ + (s_.reason ? ` (it said: "${s_.reason}")` : ''));
645
+ }
646
+ const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
647
+ const map = wantMap
648
+ ? await mapText(flags.device, options, res.endScreen, { flowOk: res.ok, escalated })
649
+ : null;
596
650
  emit(
597
651
  flags,
598
652
  {
@@ -812,7 +866,7 @@ async function main() {
812
866
  ],
813
867
  );
814
868
  } catch (err) {
815
- emit(flags, { ok: false, error: err.message }, err.message);
869
+ emit(flags, failureJson(err), err.message);
816
870
  process.exitCode = 1;
817
871
  }
818
872
  return;
@@ -1047,6 +1101,10 @@ async function main() {
1047
1101
  json: Boolean(flags.json),
1048
1102
  strict: Boolean(flags.strict) || process.env.SIMFRAME_STRICT === '1',
1049
1103
  device: flags.device,
1104
+ // So `doctor --sensor=ax-first --planner=apple` reports the mode the
1105
+ // caller is about to use, not the one the environment happens to hold.
1106
+ // Confirming the mode before a run is the whole reason to read this.
1107
+ options,
1050
1108
  });
1051
1109
  return;
1052
1110
  }
@@ -1123,7 +1181,7 @@ async function blackScreenProbe(udid) {
1123
1181
  }
1124
1182
  }
1125
1183
 
1126
- async function doctor({ json = false, strict = false, device } = {}) {
1184
+ async function doctor({ json = false, strict = false, device, options = {} } = {}) {
1127
1185
  const checks = [];
1128
1186
  // `level` is 'ok' | 'warn' | 'fail'. A warn means it works but not the way it
1129
1187
  // should — the exact state that used to be invisible.
@@ -1172,6 +1230,55 @@ async function doctor({ json = false, strict = false, device } = {}) {
1172
1230
  add('on-device OCR', 'warn', err.message, { key: 'ocr.available', value: false });
1173
1231
  }
1174
1232
 
1233
+ // Which sensors a read asks for, and what the recognition-level flag can and
1234
+ // cannot reach. Stated because `SIMFRAME_OCR` looks like it configures the
1235
+ // OCR everyone uses and does not: the daemon reads text in-process off the
1236
+ // framebuffer and owns its own recognition level, so the flag only reaches
1237
+ // the no-daemon fallback in `native/ocr.swift`.
1238
+ try {
1239
+ const api = await import('./index.js');
1240
+ const ocrMod = await import('./ocr.js');
1241
+ const mode = api.sensorMode(options);
1242
+ add('sensor mode', 'ok',
1243
+ mode === 'ax-first'
1244
+ ? 'ax-first — the tree alone (~50ms), paying for OCR only when a resolve fails'
1245
+ : 'full — accessibility and OCR fused on every read (~164ms)',
1246
+ { key: 'sensor.mode', value: mode });
1247
+ add('OCR level', 'ok', `${ocrMod.level()} (fallback helper only; the daemon owns its own)`, {
1248
+ key: 'ocr.level',
1249
+ value: ocrMod.level(),
1250
+ });
1251
+ } catch { /* reported by the layers above */ }
1252
+
1253
+ // The local supervisor. Behind the hands and in front of the reasoner, and
1254
+ // able to say only wait/retry/stop.
1255
+ try {
1256
+ const supervisor = await import('./supervisor.js');
1257
+ const st = await supervisor.status(options);
1258
+ add('local supervisor', 'ok', `${st.supervisor} — ${st.detail}`, {
1259
+ key: 'supervisor.backend',
1260
+ value: st.supervisor,
1261
+ });
1262
+ supervisor.close();
1263
+ } catch (err) {
1264
+ add('local supervisor', 'ok', `none — ${err.message}`, { key: 'supervisor.backend', value: 'none' });
1265
+ }
1266
+
1267
+ // The local planner tier. `none` is the normal answer and not a fault: it is
1268
+ // off unless SIMFRAME_PLANNER asks for it, and it only ever reorders
1269
+ // candidates that exploration was going to try anyway.
1270
+ try {
1271
+ const planner = await import('./planner.js');
1272
+ const st = await planner.status(options);
1273
+ add('local planner', 'ok', `${st.planner} — ${st.detail}`, {
1274
+ key: 'planner.backend',
1275
+ value: st.planner,
1276
+ });
1277
+ planner.close();
1278
+ } catch (err) {
1279
+ add('local planner', 'ok', `none — ${err.message}`, { key: 'planner.backend', value: 'none' });
1280
+ }
1281
+
1175
1282
  try {
1176
1283
  let booted = await bootedDevices();
1177
1284
  // Respect --device. Without this, doctor reports on every booted simulator,
@@ -1398,13 +1505,28 @@ async function doctor({ json = false, strict = false, device } = {}) {
1398
1505
  process.exitCode = failed.length || (strict && warned.length) ? 1 : 0;
1399
1506
  }
1400
1507
 
1401
- main().catch((err) => {
1508
+ // A long-lived local helper must not decide when the CLI exits. It is closed
1509
+ // after every command, whether or not one was ever started — `close()` on an
1510
+ // unopened planner is a no-op, and leaving it open made a finished flow hang.
1511
+ const closeHelpers = async () => {
1512
+ try {
1513
+ const planner = await import('./planner.js');
1514
+ planner.close();
1515
+ const supervisor = await import('./supervisor.js');
1516
+ supervisor.close();
1517
+ } catch { /* nothing to close */ }
1518
+ };
1519
+
1520
+ main().then(closeHelpers, async (err) => {
1521
+ await closeHelpers();
1522
+ throw err;
1523
+ }).catch((err) => {
1402
1524
  // A caller that asked for JSON gets JSON, failures included. Printing prose
1403
1525
  // here handed `JSON.parse` a SyntaxError instead of a reason, so a script
1404
1526
  // could not tell "the daemon lost the display" from "simframe is broken" —
1405
1527
  // which is the whole point of a machine-readable interface.
1406
1528
  if (process.argv.includes('--json')) {
1407
- process.stdout.write(`${JSON.stringify({ ok: false, error: err.message }, null, 2)}\n`);
1529
+ process.stdout.write(`${JSON.stringify(failureJson(err), null, 2)}\n`);
1408
1530
  } else {
1409
1531
  process.stderr.write(`simframe: ${err.message}\n`);
1410
1532
  }
package/src/control.js CHANGED
@@ -65,6 +65,7 @@ export const swipe = (udid, from, to, opts = {}) =>
65
65
  export const type = (udid, text) => request(udid, { action: 'type', text });
66
66
  export const paste = (udid, text) => request(udid, { action: 'paste', text });
67
67
  export const press = (udid, button) => request(udid, { action: 'press', button });
68
+ export const key = (udid, usage, modifiers = []) => request(udid, { action: 'key', usage, modifiers });
68
69
  export const status = (udid) => request(udid, { action: 'status' });
69
70
  export const resetInput = (udid) => request(udid, { action: 'resetInput' });
70
71
  export const longPress = (udid, x, y, opts = {}) => request(udid, { action: 'longPress', x, y, ...opts });
@@ -30,8 +30,26 @@ import * as regions from './regions.js';
30
30
  * list screen. Same rules, different input, therefore different hashes —
31
31
  * and a stored hash that can never match again is the quietest kind of
32
32
  * wrong, which is what this counter exists to prevent.
33
+ * 5 — a phantom keyboard was deleting screens' content from their identity. A
34
+ * dozen short text rows of uniform height stacked low on a read-only
35
+ * summary satisfied every size-and-uniformity test for a keyboard, and
36
+ * `tokens` discards everything below `keyboardTop` — so two screens of one
37
+ * wizard, sharing a nav title and a step indicator, collapsed onto a single
38
+ * hash. `detectKeyboardTop` now requires the small uniform boxes to be
39
+ * key-shaped. Every screen with content in its lower half hashes
40
+ * differently, so the stored graph and maps must go.
41
+ * 6 — the opposite half of the same bug, and it took a recorded screen to see.
42
+ * `KEYBOARD_MIN_FRACTION` is a *detection window*, not a keyboard's height,
43
+ * and its edge was being used as the boundary — so on an iPhone 17 Pro with
44
+ * the software keyboard up, the window starts at y=629 while the `q`–`p`
45
+ * row's frame top is **590**, and that whole row fell outside it. Ten
46
+ * keyboard keys were reported as page content and counted into the screen's
47
+ * identity, in the same map that said `keyboard up`. The boundary now
48
+ * extends upward while the rows above keep being key-shaped, which page
49
+ * content is not. Any screen fingerprinted with a keyboard up hashes
50
+ * differently, so the stored graph and maps must go again.
33
51
  */
34
- export const TOKEN_RULES_VERSION = 4;
52
+ export const TOKEN_RULES_VERSION = 6;
35
53
 
36
54
  /** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
37
55
  export const GRID = 24;
package/src/graph.js CHANGED
@@ -295,6 +295,38 @@ function replayable(step) {
295
295
  return step.launch != null || step.openUrl != null ? null : step;
296
296
  }
297
297
 
298
+ /**
299
+ * What has worked from this screen before, in the caller's own vocabulary.
300
+ *
301
+ * The graph has always known this and never said it. A map reported
302
+ * `(known, 3 known exits)` — the *count* — so an agent on a screen simframe had
303
+ * driven successfully six times still had to read it to learn what was tappable.
304
+ * Measured across two peer rounds: a flow whose steps were known in advance ran
305
+ * 16 steps in **one** call, and the same agent on a screen the graph also knew
306
+ * but whose labels it did not spent 25 calls on 31 steps. The difference was not
307
+ * perception. It was whether a plan existed before execution started.
308
+ *
309
+ * Ordered by how often each has worked, because that is the order an agent
310
+ * should try them in.
311
+ */
312
+ export function exitsOf(node, { limit = 8 } = {}) {
313
+ return (node?.edges ?? [])
314
+ .map((e) => ({
315
+ action: e.step?.action ?? (e.action ?? '').split(':')[0] ?? 'tap',
316
+ label: e.step?.value ?? e.step?.target ?? e.step?.label ?? e.step?.into ?? null,
317
+ to: e.to ?? null,
318
+ count: e.count ?? 0,
319
+ kind: e.kind ?? null,
320
+ }))
321
+ // A `#13` was a ref on the screen it was typed on and means nothing on the
322
+ // next visit; a raw coordinate is not a name either. Neither is reusable
323
+ // vocabulary, which is the whole point of this list.
324
+ .filter((e) => e.label != null && !/^#\d+$/.test(String(e.label).trim())
325
+ && !/^@?-?\d+\s*,\s*-?\d+$/.test(String(e.label).trim()))
326
+ .sort((a, b) => b.count - a.count)
327
+ .slice(0, limit);
328
+ }
329
+
298
330
  /**
299
331
  * What to call this screen, for a human typing `goto`.
300
332
  *
@@ -756,12 +788,36 @@ export const VERDICTS = ['ok', 'no-visible-change', 'unexpected-screen', 'unveri
756
788
  * app and the verdict still said `unexpected-screen`, because the verdict never
757
789
  * asked the graph.
758
790
  */
791
+ /**
792
+ * Are these two readings the same screen?
793
+ *
794
+ * `nearestScreen` has always had a token-similarity tolerance, precisely so
795
+ * that a screen whose *content* differs — a list with different rows, a form
796
+ * showing a different record — still resolves to the screen it is. The
797
+ * verification path threw that away: it passed bare hash strings, and a string
798
+ * carries no tokens, so only an exact hash could ever match.
799
+ *
800
+ * The cost was measured. `unexpected-screen` fired three times in one reported
801
+ * run and was wrong all three; two were this — the tester picked a different
802
+ * asset than earlier runs had, so the content differed, so the hash differed,
803
+ * so a correct navigation was called a wrong turn. Their conclusion: *"this
804
+ * will fire on every run that varies its test data — i.e. every useful run."*
805
+ * And because a failed step abandons the rest of its batch, each false alarm
806
+ * costs a round trip, which is the thing the whole design is trying to buy.
807
+ *
808
+ * So pass the reading, not just its name: `{hash, tokens}` lets the tolerance
809
+ * that already exists do its job. A string still works and still means "exact
810
+ * match only", which is right for a stored prediction that has no tokens.
811
+ */
759
812
  function sameScreen(udid, a, b) {
760
- if (!a || !b) return false;
761
- if (a === b) return true;
813
+ const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
814
+ const ha = hashOf(a);
815
+ const hb = hashOf(b);
816
+ if (!ha || !hb) return false;
817
+ if (ha === hb) return true;
762
818
  if (!udid) return false;
763
- const nodeA = nearestScreen(udid, a)?.node;
764
- const nodeB = nearestScreen(udid, b)?.node;
819
+ const nodeA = nearestScreen(udid, typeof a === 'string' ? a : { hash: ha, tokens: a?.tokens })?.node;
820
+ const nodeB = nearestScreen(udid, typeof b === 'string' ? b : { hash: hb, tokens: b?.tokens })?.node;
765
821
  return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
766
822
  }
767
823
 
@@ -777,9 +833,35 @@ function sameScreen(udid, a, b) {
777
833
  */
778
834
  export const CONFIDENT_OBSERVATIONS = 2;
779
835
 
780
- export function verdict({ udid, prediction, before, after, kind }) {
781
- if (!before || !after) return { verdict: 'unverified', detail: 'no state to compare' };
782
- const moved = before !== after;
836
+ /**
837
+ * Actions whose correct outcome is that the screen stays where it is.
838
+ *
839
+ * Typing into a field does not navigate, so screen-identity movement cannot
840
+ * say whether it worked — and answering with `no-visible-change` was actively
841
+ * harmful three ways. It printed a verdict contradicting the wait's own
842
+ * observation on the same line (`[a small change, in one region only] …
843
+ * [no-visible-change]`, both true, of different questions). It is an escalating
844
+ * verdict, so a clean flow that typed anything told the caller to stop and
845
+ * think. And it invited a re-type, which doubles a field that cannot be
846
+ * cleared.
847
+ *
848
+ * These steps are verified by reading the field back instead — see
849
+ * `fieldContents` in actions.js.
850
+ */
851
+ export const STAYS_ON_SCREEN = new Set(['type', 'paste', 'key']);
852
+
853
+ export function verdict({ udid, prediction, before, after, kind, action }) {
854
+ // `before`/`after` may be a hash or a whole reading. A reading carries its
855
+ // tokens, which is what lets a content-varied screen still be recognised as
856
+ // the screen it is.
857
+ const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
858
+ const beforeHash = hashOf(before);
859
+ const afterHash = hashOf(after);
860
+ if (!beforeHash || !afterHash) return { verdict: 'unverified', detail: 'no state to compare' };
861
+ const moved = beforeHash !== afterHash;
862
+ if (STAYS_ON_SCREEN.has(action) && !moved) {
863
+ return { verdict: 'ok', detail: 'the screen was not expected to change, and did not' };
864
+ }
783
865
  if (!prediction) {
784
866
  if (!moved) return { verdict: 'no-visible-change', detail: 'the screen did not change, and nothing predicted it would' };
785
867
  return { verdict: 'unverified', detail: 'this action has not been seen on this screen before' };