simframe 0.13.0 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/actions.js CHANGED
@@ -2555,7 +2555,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
2555
2555
  for (let attempt = 0; ; attempt += 1) {
2556
2556
  for (const one of alternatives) {
2557
2557
  try {
2558
- const found = await api.locate(deviceQuery, one, { refresh: attempt > 0 });
2558
+ const found = await api.locate(deviceQuery, one, { refresh: step.refresh !== false, options: ctx.options });
2559
2559
  return `${JSON.stringify(one)} appeared at ${found.target.x},${found.target.y}`
2560
2560
  + ` (first of ${alternatives.length} awaited)`;
2561
2561
  } catch (err) {
@@ -2563,10 +2563,12 @@ async function runStep(deviceQuery, udid, step, ctx) {
2563
2563
  }
2564
2564
  }
2565
2565
  if (Date.now() >= limit) {
2566
+ const stillAny = await stillnessNote(deviceQuery, ctx);
2566
2567
  throw new Error(
2567
2568
  `none of ${alternatives.length} awaited strings appeared`
2568
2569
  + ` (${alternatives.map((a) => JSON.stringify(a)).join(', ')}) in ${step.timeoutMs ?? 8000}ms.`
2569
- + ` Last: ${lastError}`,
2570
+ + ` Last: ${lastError}`
2571
+ + (stillAny ? ` (${stillAny.note})` : ''),
2570
2572
  );
2571
2573
  }
2572
2574
  await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 200, timeoutMs: 700, options: ctx.options })
@@ -2575,7 +2577,26 @@ async function runStep(deviceQuery, udid, step, ctx) {
2575
2577
  }
2576
2578
  for (let attempt = 0; ; attempt += 1) {
2577
2579
  try {
2578
- const found = await api.locate(deviceQuery, query, { index: step.index, refresh: attempt > 0 });
2580
+ // Fresh, like `assert` — and for the same reason, found the same way.
2581
+ //
2582
+ // This read `refresh: attempt > 0`, so the FIRST look resolved against
2583
+ // the remembered map. A wait that can be satisfied by memory is not a
2584
+ // wait: if recall hands back the wrong screen's map — which it does
2585
+ // when two screens collide on a layout hash — the target "appears"
2586
+ // without ever having been on screen, instantly, and the caller
2587
+ // proceeds against a screen it is not on.
2588
+ //
2589
+ // Caught by the fingerprint eval on CI, which is the only place it
2590
+ // could show: `settings-general` read the **Settings root** and its
2591
+ // `waitFor "About"` had passed. The signature is in the round numbers
2592
+ // — r2 and r3, never r1 — because memory has to be warm before it can
2593
+ // lie, and round 1 is always cold.
2594
+ //
2595
+ // `assert` was made fresh by default after this exact defect cost a
2596
+ // reported session; `waitFor` was left as it was. One perception pass
2597
+ // is the price, and it is the same trade: a read is cheaper than the
2598
+ // round trip a wrong verdict causes.
2599
+ const found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
2579
2600
  return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
2580
2601
  } catch (err) {
2581
2602
  lastError = err.message;
@@ -2609,9 +2630,23 @@ async function runStep(deviceQuery, udid, step, ctx) {
2609
2630
  }
2610
2631
  }
2611
2632
  if (Date.now() >= limit) break;
2633
+ // Opt-in: stop early on a screen that has plainly stopped changing.
2634
+ if (Number.isFinite(step.failIfStillFor)) {
2635
+ const still = await stillnessNote(deviceQuery, ctx);
2636
+ if (still && still.ms >= step.failIfStillFor) {
2637
+ throw new Error(`gave up on ${query} after ${Date.now() - (limit - (step.timeoutMs ?? 8000))}ms:`
2638
+ + ` ${still.note}, which is past the ${step.failIfStillFor}ms you said to stop at.`
2639
+ + ` Last: ${lastError}`);
2640
+ }
2641
+ }
2612
2642
  await sleep(POLL_MS);
2613
2643
  }
2614
- throw new Error(`waited ${step.timeoutMs ?? 8000}ms for ${query}: ${lastError}`);
2644
+ // Say how long it had been still. A wait that burned three minutes on a
2645
+ // screen static for the last twelve seconds should not make the reader
2646
+ // work that out from a second command.
2647
+ const still = await stillnessNote(deviceQuery, ctx);
2648
+ throw new Error(`waited ${step.timeoutMs ?? 8000}ms for ${query}: ${lastError}`
2649
+ + (still ? ` (${still.note} — pass failIfStillFor to stop early next time)` : ''));
2615
2650
  }
2616
2651
 
2617
2652
  // One assert step for every condition, because `assertText` could only ask
@@ -2639,6 +2674,32 @@ async function runStep(deviceQuery, udid, step, ctx) {
2639
2674
  // is the opposite of what you want learned."*
2640
2675
  found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
2641
2676
  } catch (err) {
2677
+ // Ambiguity answers "is it there", and it answers it *yes*.
2678
+ //
2679
+ // `locate` refuses a query that matches several elements, which is right
2680
+ // for `tap` — picking the wrong one of two taps the wrong thing — and
2681
+ // wrong for a question about presence. Reported from the field on
2682
+ // 0.13.0: `assert Administrator is visible` failed, and aborted the rest
2683
+ // of the batch, against a screen with **two** Administrators on it. The
2684
+ // assertion's own semantics were satisfied twice over. The message was
2685
+ // praised in the same breath — candidates, coordinates and scores — so
2686
+ // what was wrong was the verdict, not the diagnosis.
2687
+ //
2688
+ // The `gone` direction had the mirror-image bug and nobody had hit it
2689
+ // yet: any error at all returned "is gone", so a query matching two
2690
+ // visible elements would have reported them absent. Ambiguity is the one
2691
+ // error that is positive evidence of presence, and it was being read as
2692
+ // proof of absence.
2693
+ //
2694
+ // Strict single-match resolution stays for `enabled`, `disabled` and
2695
+ // `value`, where the question is about a *particular* element and
2696
+ // answering it from the wrong one is how a confident wrong answer gets
2697
+ // made.
2698
+ if (err.ambiguous && err.candidates?.length) {
2699
+ const n = err.candidates.length;
2700
+ if (want === 'visible') return `${query} is visible (${n} things match it here — a presence check does not have to choose)`;
2701
+ if (want === 'gone') throw new Error(`${query}: still here — ${n} things on this screen match it`);
2702
+ }
2642
2703
  if (want === 'gone') return `${query} is gone`;
2643
2704
  throw new Error(`${query}: ${err.message}`);
2644
2705
  }
@@ -2744,3 +2805,55 @@ export function settleEvidence(w) {
2744
2805
  return (parts.length ? parts.join('; ') : 'nothing observed')
2745
2806
  + (m ? `\n${m.map}` : '');
2746
2807
  }
2808
+
2809
+ /**
2810
+ * The one-line flow summary, so `ok` and `FAIL` each mean exactly one thing.
2811
+ *
2812
+ * `FLOW FAILED — 3/3 steps` was reported from the field as reading like a
2813
+ * success, and the reporter had it exactly: the denominator means *attempted*
2814
+ * on failure and *succeeded* on success, so the same shape carries opposite
2815
+ * meanings. `flow completed — 5/5 steps` and `FLOW FAILED — 3/3 steps` differ
2816
+ * only in a word, and the numbers argue against the word.
2817
+ *
2818
+ * On failure it says how many worked and how many did not, which is the thing a
2819
+ * caller has to know to decide whether to resume or re-plan.
2820
+ */
2821
+ export function flowSummary(res, { withTime = true } = {}) {
2822
+ const time = withTime && Number.isFinite(res.totalMs) ? ` in ${res.totalMs}ms` : '';
2823
+ if (res.ok) return `flow completed — ${res.ranSteps}/${res.totalSteps} steps${time}`;
2824
+ const failed = (res.results ?? []).filter((r) => r.ok === false).length || 1;
2825
+ const worked = Math.max(0, res.ranSteps - failed);
2826
+ const unattempted = Math.max(0, res.totalSteps - res.ranSteps);
2827
+ return `FLOW FAILED — ${worked} ok, ${failed} failed`
2828
+ + (unattempted ? `, ${unattempted} not attempted` : '')
2829
+ + ` (of ${res.totalSteps})${time}`;
2830
+ }
2831
+
2832
+ /**
2833
+ * How long the screen has been still, for a wait that is about to give up.
2834
+ *
2835
+ * A field report: a 180-second wait for a control that never appeared, on a
2836
+ * screen that had been static for about twelve of those seconds — the app had
2837
+ * logged itself out and was sitting on a login form. The failure message was
2838
+ * praised for listing what *was* on screen; it arrived three minutes late.
2839
+ *
2840
+ * Stillness is already tracked and already printed by other commands, so the
2841
+ * information existed and this wait simply never asked for it. Reported on
2842
+ * every timeout, and — only when the caller opts in — allowed to end the wait
2843
+ * early. Opt-in and not default, because a still screen is exactly what a
2844
+ * pending network call looks like: the target may yet arrive, and a wait that
2845
+ * gave up on stillness alone would break the case waits exist for.
2846
+ *
2847
+ * This is only trustworthy because of 134. Before the animation threshold was
2848
+ * measured, a completely static screen claimed something was animating on 54%
2849
+ * of its frames, and "nothing has moved" could not be said with a straight face.
2850
+ */
2851
+ async function stillnessNote(deviceQuery, ctx) {
2852
+ try {
2853
+ const { state } = await api.getState(deviceQuery, { options: ctx.options });
2854
+ const ms = state?.stableForMs;
2855
+ return Number.isFinite(ms) ? { ms, note: `the screen has not moved for ${Math.round(ms)}ms` } : null;
2856
+ } catch {
2857
+ return null;
2858
+ }
2859
+ }
package/src/cli.js CHANGED
@@ -12,6 +12,7 @@ import * as baseline from './baseline.js';
12
12
  import * as metrics from './metrics.js';
13
13
  import * as navigate from './navigate.js';
14
14
  import { decodePng } from './png.js';
15
+ import * as storage from './storage.js';
15
16
  import * as store from './store.js';
16
17
  import * as view from './view.js';
17
18
 
@@ -33,6 +34,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
33
34
  simframe tap <selector> tap #3, "Save", or @120,400
34
35
  simframe do <script.json> run a scripted flow (see below)
35
36
  simframe screens [device] list screens this device has learned
37
+ simframe storage [bundle-id] what the app saved (works on a shut-down device)
36
38
  simframe goto <screen> walk to a known screen through known steps
37
39
  simframe flow save <name> <script.json> run a flow and save it if every step verifies
38
40
  simframe flow run <name> replay a saved flow
@@ -801,7 +803,7 @@ async function main() {
801
803
  },
802
804
  [
803
805
  ...res.results.map(stepLine),
804
- `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps in ${res.totalMs}ms`,
806
+ actions.flowSummary(res),
805
807
  saved && (saved.ok ? `saved flow "${saved.name}" — ${saved.steps} steps` : `not saved: ${saved.reason}`),
806
808
  map && `\n${map}`,
807
809
  ],
@@ -860,6 +862,25 @@ async function main() {
860
862
  return;
861
863
  }
862
864
 
865
+ case 'storage': {
866
+ // A file read, like `screens`, and deliberately not a daemon call. The
867
+ // whole value of this command is that it answers on a device that is not
868
+ // running — measured: simctl itself cannot, on this Xcode. Routing it
869
+ // through the daemon would throw that away for no gain.
870
+ const device = await resolveDevice(flags.device);
871
+ const [bundleId] = positional;
872
+ if (!bundleId) {
873
+ const list = await storage.apps(device.udid);
874
+ const match = flags.match ? String(flags.match).toLowerCase() : null;
875
+ const shown = match ? list.filter((a) => a.bundleId.toLowerCase().includes(match)) : list;
876
+ emit(flags, shown, storage.formatApps(shown));
877
+ return;
878
+ }
879
+ const result = await storage.read(device.udid, bundleId);
880
+ emit(flags, result, storage.format(result));
881
+ return;
882
+ }
883
+
863
884
  case 'screens': {
864
885
  // Reading what this device has learned is a file read. It used to go
865
886
  // through ensureDaemon, so a device whose capture had stopped could not
@@ -920,7 +941,7 @@ async function main() {
920
941
  }
921
942
  emit(flags, res, [
922
943
  ...(res.results ?? []).map(stepLine),
923
- `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps`,
944
+ actions.flowSummary(res, { withTime: false }),
924
945
  ]);
925
946
  process.exitCode = res.ok ? 0 : 1;
926
947
  return;
package/src/graph.js CHANGED
@@ -396,7 +396,18 @@ export function describe(node) {
396
396
  if (tabs.length) return tabs.slice(0, 3).join(' / ');
397
397
  const anyChrome = labels(/:(nav-bar|tab-bar):/);
398
398
  if (anyChrome.length) return anyChrome.slice(0, 3).join(' ');
399
- return node.hash.slice(0, 8);
399
+ // No name. Say so, rather than handing back a hash dressed as one.
400
+ //
401
+ // This fell back to `node.hash.slice(0, 8)`, and the map prints the name in
402
+ // quotes after the identity hash — so an unnamed screen read
403
+ // `screen 299dd147 "a9505378"`: two hashes, one of them looking like a title,
404
+ // beside the `screen 089bec77 "time sheets"` a named screen produces.
405
+ // Reported from the field, with the right fix attached: omit the quoted part
406
+ // rather than echo a second hash.
407
+ //
408
+ // Callers that need *something* to print in a list supply their own fallback,
409
+ // which is a decision about presentation and belongs at the point of display.
410
+ return null;
400
411
  }
401
412
 
402
413
  /** Find a known screen by what a human would call it. */
@@ -404,7 +415,9 @@ export function findScreen(udid, query) {
404
415
  const wanted = String(query ?? '').trim();
405
416
  if (!wanted) return null;
406
417
  const scored = allNodes(udid)
407
- .map((node) => ({ node, name: describe(node) }))
418
+ // The short hash stays searchable: `goto 089bec77` worked before `describe`
419
+ // stopped inventing names and must keep working.
420
+ .map((node) => ({ node, name: describe(node) ?? node.hash.slice(0, 8) }))
408
421
  .map((c) => ({ ...c, score: matching.nameScore(c.name, wanted) }))
409
422
  .filter((c) => c.score > 0)
410
423
  .sort((a, b) => b.score - a.score);
package/src/matching.js CHANGED
@@ -170,7 +170,23 @@ export function rank(targets, intent, { screen } = {}) {
170
170
  // An icon-only control has no readable name, so a synonym is the only way
171
171
  // to reach it — this is how "back" finds a bare chevron.
172
172
  if (group && base < 0.5 && !t.label && t.rawLabel) base = 0.55;
173
- if (group && base < 0.5 && names.some((n) => group.words.includes(norm(n)))) base = 0.9;
173
+ // Genuine synonymy only: the name must be a *different* word in the group.
174
+ //
175
+ // This branch fires only when `base < 0.5`, which means its whole job is to
176
+ // overrule the coverage scaling the two branches in `nameScore` were taught
177
+ // — and when the query already contains the name literally, that scaling was
178
+ // the right answer and this flat 0.9 throws it away. Reported (138): a long
179
+ // descriptive phrase ending "…under Settings" scored a *heading* labelled
180
+ // "Settings" at 0.9 and the row the caller meant at 0.265, tapped the
181
+ // heading, and returned `ok [no visible change]` with an `or` list untried.
182
+ //
183
+ // A name the query spells out has already been scored on how much of the
184
+ // query it covers. What this branch is for is the case that scoring cannot
185
+ // see at all: "back" reaching a control labelled "Previous", "settings"
186
+ // reaching "Preferences". That is synonymy, and it is unaffected.
187
+ const spelledOut = (n) => bare.includes(norm(n)) || norm(intent).includes(norm(n));
188
+ if (group && base < 0.5
189
+ && names.some((n) => group.words.includes(norm(n)) && !spelledOut(n))) base = 0.9;
174
190
  if (base <= 0) continue;
175
191
 
176
192
  const reasons = [matched ? `label "${matched}"` : 'icon-only'];
package/src/mcp.js CHANGED
@@ -15,7 +15,8 @@ import * as api from './index.js';
15
15
  import * as input from './input.js';
16
16
  import * as metrics from './metrics.js';
17
17
  import * as navigate from './navigate.js';
18
- import { bootedDevices, permissionServices } from './platform/index.js';
18
+ import { bootedDevices, listDevices, permissionServices, resolveDevice } from './platform/index.js';
19
+ import * as storage from './storage.js';
19
20
  import * as store from './store.js';
20
21
  import * as view from './view.js';
21
22
 
@@ -132,7 +133,7 @@ const TOOLS = [
132
133
  steps: {
133
134
  type: 'array',
134
135
  description:
135
- 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
136
+ 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}} (add "failIfStillFor":15000 to stop early once the screen has plainly stopped changing — a 180s wait once burned three minutes on an app that had logged itself out; without it a timeout still reports how long the screen had been still), {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
136
137
  items: { type: 'object' },
137
138
  },
138
139
  autoSettle: {
@@ -192,7 +193,19 @@ const TOOLS = [
192
193
  description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
193
194
  inputSchema: {
194
195
  type: 'object',
195
- properties: { ...deviceProp, ...modeProps, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
196
+ properties: {
197
+ ...deviceProp,
198
+ ...modeProps,
199
+ ...selectorProp('What to wait for'),
200
+ timeoutMs: { type: 'number', description: 'Default 8000.' },
201
+ failIfStillFor: {
202
+ type: 'number',
203
+ description: 'Give up early once the screen has not moved for this long and the target is still absent.'
204
+ + ' Off by default, because a still screen is also what a pending network call looks like —'
205
+ + ' use it when the thing you await would arrive with a visible change or not at all.'
206
+ + ' A timeout reports the stillness either way.',
207
+ },
208
+ },
196
209
  required: ['sel'],
197
210
  },
198
211
  },
@@ -386,10 +399,42 @@ const TOOLS = [
386
399
  required: ['action'],
387
400
  },
388
401
  },
402
+ {
403
+ name: 'sim_storage',
404
+ description:
405
+ 'What the app saved, as text: its UserDefaults and (for React Native) its AsyncStorage, read straight out of'
406
+ + ' the data container. sim_ui says what is drawn; sim_storage says what the app believes — use it when the'
407
+ + ' screen and the behaviour disagree, or to check a value without driving the UI to it.'
408
+ + ' Works on a device that is NOT running, so it can answer before anything is booted.'
409
+ + ' Call with no bundleId to list the apps that have a container (match filters that list).',
410
+ inputSchema: {
411
+ type: 'object',
412
+ properties: {
413
+ bundleId: { type: 'string', description: 'The app to read, e.g. com.example.myapp. Omit to list apps instead.' },
414
+ match: { type: 'string', description: 'When listing, show only bundle ids containing this string.' },
415
+ ...deviceProp,
416
+ ...modeProps,
417
+ },
418
+ },
419
+ },
389
420
  {
390
421
  name: 'sim_devices',
391
- description: 'List the booted devices simframe can drive — iOS simulators and Android emulators.',
392
- inputSchema: { type: 'object', properties: {} },
422
+ description: 'List the devices simframe can drive — iOS simulators and Android emulators.'
423
+ + ' Booted ones by default; pass all to see every device on the host and its state.'
424
+ + ' Also reports which simframe build is answering.',
425
+ inputSchema: {
426
+ type: 'object',
427
+ properties: {
428
+ all: {
429
+ type: 'boolean',
430
+ description: 'Also account for devices that are shut down, grouped by runtime.',
431
+ },
432
+ match: {
433
+ type: 'string',
434
+ description: 'With all, list shut-down devices whose name or runtime contains this, in full.',
435
+ },
436
+ },
437
+ },
393
438
  },
394
439
  ];
395
440
 
@@ -560,7 +605,7 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
560
605
  options,
561
606
  );
562
607
  case 'sim_wait_for':
563
- return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs }, args, options);
608
+ return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs, failIfStillFor: args.failIfStillFor }, args, options);
564
609
  case 'sim_assert':
565
610
  return await oneStep(target, { assert: args.sel, is: args.is, equals: args.equals }, args, options);
566
611
  case 'sim_launch':
@@ -585,8 +630,10 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
585
630
  return await flowRun(target, args, options);
586
631
  case 'sim_capture':
587
632
  return await capture(target, args, options);
633
+ case 'sim_storage':
634
+ return await appStorage(args);
588
635
  case 'sim_devices':
589
- return await devices();
636
+ return await devices(args);
590
637
  default:
591
638
  throw new Error(`unknown tool ${req.params.name}`);
592
639
  }
@@ -935,7 +982,7 @@ function verdictLineFor(results) {
935
982
 
936
983
  function stepLines(res) {
937
984
  const lines = [
938
- `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps in ${res.totalMs}ms`,
985
+ actions.flowSummary(res),
939
986
  ];
940
987
  for (const r of res.results) {
941
988
  const settle = r.settled
@@ -1142,12 +1189,94 @@ function listStateDirs() {
1142
1189
  }
1143
1190
  }
1144
1191
 
1145
- async function devices() {
1192
+ /**
1193
+ * What an app has persisted.
1194
+ *
1195
+ * Deliberately not a daemon call and deliberately not a `simctl` call. Measured
1196
+ * on this Xcode, `simctl get_app_container` and `simctl listapps` both refuse on
1197
+ * a device that is not running — so the one property that made the field
1198
+ * reporter rate this the highest-leverage thing in their session, answering
1199
+ * *before the device is booted*, is only reachable by reading the container off
1200
+ * the host filesystem. That is what the backend does.
1201
+ */
1202
+ async function appStorage({ bundleId, match: query, device } = {}) {
1203
+ const resolved = await resolveDevice(device);
1204
+ if (!bundleId) {
1205
+ const list = await storage.apps(resolved.udid);
1206
+ const needle = query ? String(query).toLowerCase() : null;
1207
+ const shown = needle ? list.filter((a) => a.bundleId.toLowerCase().includes(needle)) : list;
1208
+ return { content: [text(storage.formatApps(shown))] };
1209
+ }
1210
+ const result = await storage.read(resolved.udid, bundleId);
1211
+ return { content: [text(storage.format(result))] };
1212
+ }
1213
+
1214
+ async function devices({ all = false, match: query } = {}) {
1146
1215
  const booted = await bootedDevices();
1147
- if (!booted.length) return { content: [text('no booted devices')] };
1148
1216
  // Noticing a name collision here is what lets every later header disambiguate
1149
1217
  // itself, and it costs nothing: this listing is already being made.
1150
1218
  const clash = noteBooted(booted);
1151
- const list = booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1152
- return { content: [text(clash ? `${clash}\n\n${list}` : list)] };
1219
+ // The build that is answering, on the one call every session starts with.
1220
+ //
1221
+ // Reported from the field: a session told to test 0.13.0 could not find out
1222
+ // what it was running. The globally installed CLI said 0.12.2 while the MCP
1223
+ // server ran from a checkout, and answering "am I on the build under test?"
1224
+ // took three shell calls and a read of `~/.claude.json`. A tool being field
1225
+ // tested should be able to state its own build, and this is the cheapest
1226
+ // place to put it.
1227
+ const head = `simframe ${packageVersion()}`;
1228
+ if (!all) {
1229
+ if (!booted.length) {
1230
+ // Never a bare "no devices". The host almost always has some, they are
1231
+ // just off, and the reporter who hit this fell out of the tool entirely
1232
+ // and went to `xcrun simctl` — for a tool whose whole job is driving
1233
+ // simulators, that is the conspicuous hole.
1234
+ const every = await listDevices().catch(() => []);
1235
+ return { content: [text(`${head}\n\nno booted devices`
1236
+ + (every.length ? ` — the host has ${every.length}, all shut down. Pass all:true to see them.` : ''))] };
1237
+ }
1238
+ const list = booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1239
+ return { content: [text([head, clash, list].filter(Boolean).join('\n\n'))] };
1240
+ }
1241
+ const every = await listDevices();
1242
+ if (!every.length) return { content: [text(`${head}\n\nno devices on this host`)] };
1243
+
1244
+ // Summarised, not dumped.
1245
+ //
1246
+ // The first version of this listed every device in full and produced **126
1247
+ // rows** on this host — two thousand tokens to answer "what else is here",
1248
+ // from a tool whose entire argument is that text beats a screenshot because
1249
+ // it is cheaper. A listing that costs more than the screenshot it replaces has
1250
+ // lost the plot.
1251
+ //
1252
+ // So: booted devices in full, because those are the ones a caller can act on,
1253
+ // and the rest grouped by runtime with counts. `match` lists in full, because
1254
+ // a caller who names what they are looking for has already narrowed it.
1255
+ const wanted = String(query ?? '').trim().toLowerCase();
1256
+ const off = every.filter((d) => d.state !== 'Booted');
1257
+ const lines = [head, clash].filter(Boolean);
1258
+ lines.push(booted.length
1259
+ ? booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
1260
+ : 'no booted devices');
1261
+
1262
+ const hits = wanted
1263
+ ? off.filter((d) => `${d.name} ${d.runtime}`.toLowerCase().includes(wanted))
1264
+ : [];
1265
+ if (wanted) {
1266
+ lines.push(hits.length
1267
+ ? `shut down, matching "${query}":\n`
1268
+ + hits.map((d) => `○ ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
1269
+ : `no shut-down device matches "${query}" (${off.length} are shut down)`);
1270
+ } else if (off.length) {
1271
+ const byRuntime = new Map();
1272
+ for (const d of off) byRuntime.set(d.runtime, (byRuntime.get(d.runtime) ?? 0) + 1);
1273
+ const summary = [...byRuntime.entries()]
1274
+ .sort((a, b) => b[1] - a[1])
1275
+ .map(([runtime, n]) => ` ${runtime} — ${n}`)
1276
+ .join('\n');
1277
+ lines.push(`${off.length} device(s) shut down, by runtime:\n${summary}\n`
1278
+ + 'pass match to list the ones you mean, e.g. match:"iPhone 17 Pro".');
1279
+ }
1280
+ lines.push('simframe cannot drive a device until it is booted.');
1281
+ return { content: [text(lines.join('\n\n'))] };
1153
1282
  }
package/src/navigate.js CHANGED
@@ -133,7 +133,10 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
133
133
  }
134
134
 
135
135
  export function knownScreens(udid) {
136
- return graph.allNodes(udid).map((n) => ({ name: graph.describe(n), hash: n.hash.slice(0, 8), edges: n.edges.length }));
136
+ // A listing needs a handle for every row, so an unnamed screen falls back to
137
+ // its own short hash here — where it is plainly the hash column's value and
138
+ // not a title in quotes.
139
+ return graph.allNodes(udid).map((n) => ({ name: graph.describe(n) ?? n.hash.slice(0, 8), hash: n.hash.slice(0, 8), edges: n.edges.length }));
137
140
  }
138
141
 
139
142
  /**
@@ -1018,6 +1018,37 @@ async function restartDevice(serial) {
1018
1018
  );
1019
1019
  }
1020
1020
 
1021
+ /**
1022
+ * Reading an app's own storage is not implemented for Android, and says so.
1023
+ *
1024
+ * Not a stub and not a borrowed answer. The iOS version reads a CoreSimulator
1025
+ * data container straight off the host filesystem, which an emulator has no
1026
+ * equivalent of: an app's files live inside the emulator's own userdata image,
1027
+ * and the way in is `adb shell run-as <package>` — which works only for a
1028
+ * debuggable build, needs the emulator running, and would be a different
1029
+ * feature with different guarantees rather than the same one.
1030
+ *
1031
+ * The standing rule is that a layer a platform does not have is declined with a
1032
+ * reason, never described in the other platform's vocabulary. Claiming a data
1033
+ * container here is how `doctor` once told an emulator its input driver was
1034
+ * idb.
1035
+ */
1036
+ const noStorage = (serial, what) => {
1037
+ throw new Error(
1038
+ `simframe cannot read ${what} on an emulator (${serial}) yet. The iOS version reads a`
1039
+ + ' simulator data container off the host filesystem and an emulator has no such thing —'
1040
+ + ' its app data lives inside the userdata image, reachable only through'
1041
+ + ' `adb shell run-as <package>` on a debuggable build, with the emulator running.'
1042
+ + ' That is a different feature and it has not been built.',
1043
+ );
1044
+ };
1045
+
1046
+ async function listApps(serial) { return noStorage(serial, 'the list of installed apps'); }
1047
+ async function appContainer(serial) { return noStorage(serial, "an app's data container"); }
1048
+ async function readPropertyList() {
1049
+ throw new Error('property lists are an iOS format; Android has no equivalent to read');
1050
+ }
1051
+
1021
1052
  /** @type {import('./index.js').Platform} */
1022
1053
  export const platform = {
1023
1054
  id: 'android',
@@ -1034,6 +1065,9 @@ export const platform = {
1034
1065
  terminateApp,
1035
1066
  openUrl,
1036
1067
  restartDevice,
1068
+ listApps,
1069
+ appContainer,
1070
+ readPropertyList,
1037
1071
  setPermission,
1038
1072
  setPasteboard,
1039
1073
  getPasteboard,
@@ -63,6 +63,7 @@ export const PLATFORM_SURFACE = Object.freeze([
63
63
  'geometry', 'inputDriver',
64
64
  'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
+ 'listApps', 'appContainer', 'readPropertyList',
66
67
  'bootedAt',
67
68
  ]);
68
69
 
@@ -209,6 +210,12 @@ export const openUrl = (udid, ...args) => platformFor(udid).openUrl(udid, ...arg
209
210
  export const setPermission = (udid, ...args) => platformFor(udid).setPermission(udid, ...args);
210
211
  export const setPasteboard = (udid, ...args) => platformFor(udid).setPasteboard(udid, ...args);
211
212
 
213
+ // Reading what an app persisted. Routed like everything else, and declined by a
214
+ // backend that has no equivalent rather than answered in the other's terms.
215
+ export const listApps = (udid, ...args) => platformFor(udid).listApps(udid, ...args);
216
+ export const appContainer = (udid, ...args) => platformFor(udid).appContainer(udid, ...args);
217
+ export const readPropertyList = (udid, file) => platformFor(udid).readPropertyList(file);
218
+
212
219
  /**
213
220
  * The permission services a device understands, or every service any backend
214
221
  * understands when no device is named.