simframe 0.12.2 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/mcp.js CHANGED
@@ -15,7 +15,8 @@ import * as api from './index.js';
15
15
  import * as input from './input.js';
16
16
  import * as metrics from './metrics.js';
17
17
  import * as navigate from './navigate.js';
18
- import { bootedDevices, permissionServices } from './platform/index.js';
18
+ import { bootedDevices, listDevices, permissionServices, resolveDevice } from './platform/index.js';
19
+ import * as storage from './storage.js';
19
20
  import * as store from './store.js';
20
21
  import * as view from './view.js';
21
22
 
@@ -132,7 +133,7 @@ const TOOLS = [
132
133
  steps: {
133
134
  type: 'array',
134
135
  description:
135
- 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
136
+ 'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}} (add "failIfStillFor":15000 to stop early once the screen has plainly stopped changing — a 180s wait once burned three minutes on an app that had logged itself out; without it a timeout still reports how long the screen had been still), {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
136
137
  items: { type: 'object' },
137
138
  },
138
139
  autoSettle: {
@@ -192,7 +193,19 @@ const TOOLS = [
192
193
  description: 'Block until something appears on screen, then return the screen map. Use this instead of pausing and re-reading.',
193
194
  inputSchema: {
194
195
  type: 'object',
195
- properties: { ...deviceProp, ...modeProps, ...selectorProp('What to wait for'), timeoutMs: { type: 'number', description: 'Default 8000.' } },
196
+ properties: {
197
+ ...deviceProp,
198
+ ...modeProps,
199
+ ...selectorProp('What to wait for'),
200
+ timeoutMs: { type: 'number', description: 'Default 8000.' },
201
+ failIfStillFor: {
202
+ type: 'number',
203
+ description: 'Give up early once the screen has not moved for this long and the target is still absent.'
204
+ + ' Off by default, because a still screen is also what a pending network call looks like —'
205
+ + ' use it when the thing you await would arrive with a visible change or not at all.'
206
+ + ' A timeout reports the stillness either way.',
207
+ },
208
+ },
196
209
  required: ['sel'],
197
210
  },
198
211
  },
@@ -386,10 +399,42 @@ const TOOLS = [
386
399
  required: ['action'],
387
400
  },
388
401
  },
402
+ {
403
+ name: 'sim_storage',
404
+ description:
405
+ 'What the app saved, as text: its UserDefaults and (for React Native) its AsyncStorage, read straight out of'
406
+ + ' the data container. sim_ui says what is drawn; sim_storage says what the app believes — use it when the'
407
+ + ' screen and the behaviour disagree, or to check a value without driving the UI to it.'
408
+ + ' Works on a device that is NOT running, so it can answer before anything is booted.'
409
+ + ' Call with no bundleId to list the apps that have a container (match filters that list).',
410
+ inputSchema: {
411
+ type: 'object',
412
+ properties: {
413
+ bundleId: { type: 'string', description: 'The app to read, e.g. com.example.myapp. Omit to list apps instead.' },
414
+ match: { type: 'string', description: 'When listing, show only bundle ids containing this string.' },
415
+ ...deviceProp,
416
+ ...modeProps,
417
+ },
418
+ },
419
+ },
389
420
  {
390
421
  name: 'sim_devices',
391
- description: 'List the booted devices simframe can drive — iOS simulators and Android emulators.',
392
- inputSchema: { type: 'object', properties: {} },
422
+ description: 'List the devices simframe can drive — iOS simulators and Android emulators.'
423
+ + ' Booted ones by default; pass all to see every device on the host and its state.'
424
+ + ' Also reports which simframe build is answering.',
425
+ inputSchema: {
426
+ type: 'object',
427
+ properties: {
428
+ all: {
429
+ type: 'boolean',
430
+ description: 'Also account for devices that are shut down, grouped by runtime.',
431
+ },
432
+ match: {
433
+ type: 'string',
434
+ description: 'With all, list shut-down devices whose name or runtime contains this, in full.',
435
+ },
436
+ },
437
+ },
393
438
  },
394
439
  ];
395
440
 
@@ -459,7 +504,13 @@ function header(device, state, ageMs, extra = '') {
459
504
  }
460
505
 
461
506
  function livenessLine(live) {
462
- return live?.ok ? null : `WARNING: ${live.note}`;
507
+ // `ok` is not the only thing worth saying. A surface that has stopped
508
+ // updating leaves every hard signal green — the loop is alive, no read
509
+ // failed, the frame is milliseconds old — and reporting nothing is how
510
+ // `sim_look` served a three-minute-old login screen while announcing it as
511
+ // 66ms old. A note that exists must reach the caller.
512
+ if (live?.note) return `WARNING: ${live.note}`;
513
+ return null;
463
514
  }
464
515
 
465
516
  function sinceLine(since) {
@@ -554,7 +605,7 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
554
605
  options,
555
606
  );
556
607
  case 'sim_wait_for':
557
- return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs }, args, options);
608
+ return await oneStep(target, { waitFor: args.sel, timeoutMs: args.timeoutMs, failIfStillFor: args.failIfStillFor }, args, options);
558
609
  case 'sim_assert':
559
610
  return await oneStep(target, { assert: args.sel, is: args.is, equals: args.equals }, args, options);
560
611
  case 'sim_launch':
@@ -579,8 +630,10 @@ export async function serve({ device: defaultDevice, options: baseOptions = {} }
579
630
  return await flowRun(target, args, options);
580
631
  case 'sim_capture':
581
632
  return await capture(target, args, options);
633
+ case 'sim_storage':
634
+ return await appStorage(args);
582
635
  case 'sim_devices':
583
- return await devices();
636
+ return await devices(args);
584
637
  default:
585
638
  throw new Error(`unknown tool ${req.params.name}`);
586
639
  }
@@ -929,7 +982,7 @@ function verdictLineFor(results) {
929
982
 
930
983
  function stepLines(res) {
931
984
  const lines = [
932
- `${res.ok ? 'flow completed' : 'FLOW FAILED'} — ${res.ranSteps}/${res.totalSteps} steps in ${res.totalMs}ms`,
985
+ actions.flowSummary(res),
933
986
  ];
934
987
  for (const r of res.results) {
935
988
  const settle = r.settled
@@ -1136,12 +1189,94 @@ function listStateDirs() {
1136
1189
  }
1137
1190
  }
1138
1191
 
1139
- async function devices() {
1192
+ /**
1193
+ * What an app has persisted.
1194
+ *
1195
+ * Deliberately not a daemon call and deliberately not a `simctl` call. Measured
1196
+ * on this Xcode, `simctl get_app_container` and `simctl listapps` both refuse on
1197
+ * a device that is not running — so the one property that made the field
1198
+ * reporter rate this the highest-leverage thing in their session, answering
1199
+ * *before the device is booted*, is only reachable by reading the container off
1200
+ * the host filesystem. That is what the backend does.
1201
+ */
1202
+ async function appStorage({ bundleId, match: query, device } = {}) {
1203
+ const resolved = await resolveDevice(device);
1204
+ if (!bundleId) {
1205
+ const list = await storage.apps(resolved.udid);
1206
+ const needle = query ? String(query).toLowerCase() : null;
1207
+ const shown = needle ? list.filter((a) => a.bundleId.toLowerCase().includes(needle)) : list;
1208
+ return { content: [text(storage.formatApps(shown))] };
1209
+ }
1210
+ const result = await storage.read(resolved.udid, bundleId);
1211
+ return { content: [text(storage.format(result))] };
1212
+ }
1213
+
1214
+ async function devices({ all = false, match: query } = {}) {
1140
1215
  const booted = await bootedDevices();
1141
- if (!booted.length) return { content: [text('no booted devices')] };
1142
1216
  // Noticing a name collision here is what lets every later header disambiguate
1143
1217
  // itself, and it costs nothing: this listing is already being made.
1144
1218
  const clash = noteBooted(booted);
1145
- const list = booted.map((d) => `${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1146
- return { content: [text(clash ? `${clash}\n\n${list}` : list)] };
1219
+ // The build that is answering, on the one call every session starts with.
1220
+ //
1221
+ // Reported from the field: a session told to test 0.13.0 could not find out
1222
+ // what it was running. The globally installed CLI said 0.12.2 while the MCP
1223
+ // server ran from a checkout, and answering "am I on the build under test?"
1224
+ // took three shell calls and a read of `~/.claude.json`. A tool being field
1225
+ // tested should be able to state its own build, and this is the cheapest
1226
+ // place to put it.
1227
+ const head = `simframe ${packageVersion()}`;
1228
+ if (!all) {
1229
+ if (!booted.length) {
1230
+ // Never a bare "no devices". The host almost always has some, they are
1231
+ // just off, and the reporter who hit this fell out of the tool entirely
1232
+ // and went to `xcrun simctl` — for a tool whose whole job is driving
1233
+ // simulators, that is the conspicuous hole.
1234
+ const every = await listDevices().catch(() => []);
1235
+ return { content: [text(`${head}\n\nno booted devices`
1236
+ + (every.length ? ` — the host has ${every.length}, all shut down. Pass all:true to see them.` : ''))] };
1237
+ }
1238
+ const list = booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n');
1239
+ return { content: [text([head, clash, list].filter(Boolean).join('\n\n'))] };
1240
+ }
1241
+ const every = await listDevices();
1242
+ if (!every.length) return { content: [text(`${head}\n\nno devices on this host`)] };
1243
+
1244
+ // Summarised, not dumped.
1245
+ //
1246
+ // The first version of this listed every device in full and produced **126
1247
+ // rows** on this host — two thousand tokens to answer "what else is here",
1248
+ // from a tool whose entire argument is that text beats a screenshot because
1249
+ // it is cheaper. A listing that costs more than the screenshot it replaces has
1250
+ // lost the plot.
1251
+ //
1252
+ // So: booted devices in full, because those are the ones a caller can act on,
1253
+ // and the rest grouped by runtime with counts. `match` lists in full, because
1254
+ // a caller who names what they are looking for has already narrowed it.
1255
+ const wanted = String(query ?? '').trim().toLowerCase();
1256
+ const off = every.filter((d) => d.state !== 'Booted');
1257
+ const lines = [head, clash].filter(Boolean);
1258
+ lines.push(booted.length
1259
+ ? booted.map((d) => `● ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
1260
+ : 'no booted devices');
1261
+
1262
+ const hits = wanted
1263
+ ? off.filter((d) => `${d.name} ${d.runtime}`.toLowerCase().includes(wanted))
1264
+ : [];
1265
+ if (wanted) {
1266
+ lines.push(hits.length
1267
+ ? `shut down, matching "${query}":\n`
1268
+ + hits.map((d) => `○ ${d.name} · ${d.runtime} · ${d.udid}`).join('\n')
1269
+ : `no shut-down device matches "${query}" (${off.length} are shut down)`);
1270
+ } else if (off.length) {
1271
+ const byRuntime = new Map();
1272
+ for (const d of off) byRuntime.set(d.runtime, (byRuntime.get(d.runtime) ?? 0) + 1);
1273
+ const summary = [...byRuntime.entries()]
1274
+ .sort((a, b) => b[1] - a[1])
1275
+ .map(([runtime, n]) => ` ${runtime} — ${n}`)
1276
+ .join('\n');
1277
+ lines.push(`${off.length} device(s) shut down, by runtime:\n${summary}\n`
1278
+ + 'pass match to list the ones you mean, e.g. match:"iPhone 17 Pro".');
1279
+ }
1280
+ lines.push('simframe cannot drive a device until it is booted.');
1281
+ return { content: [text(lines.join('\n\n'))] };
1147
1282
  }
package/src/metrics.js CHANGED
@@ -300,11 +300,30 @@ const VERIFYING_STEPS = new Set([
300
300
  */
301
301
  export function reasonForStepError(step, err) {
302
302
  const tagged = escalationOf(err);
303
- if (tagged) return tagged;
303
+ if (tagged) return { ...tagged, classified: true };
304
304
  const action = step?.action;
305
- if (action === 'confirm' || action === 'chooseAny') return { reason: 'novel_dialog', candidates: [], tried: [] };
306
- if (VERIFYING_STEPS.has(action)) return { reason: 'verification_failed', candidates: [], tried: [] };
307
- return { reason: 'verification_failed', candidates: [], tried: [] };
305
+ if (action === 'confirm' || action === 'chooseAny') {
306
+ return { reason: 'novel_dialog', candidates: [], tried: [], classified: true };
307
+ }
308
+ // Everything else is a *fallback*, and it now says so.
309
+ //
310
+ // This had two branches that returned the same value, which made it look
311
+ // like it discriminated. It does not: any step that threw without a site
312
+ // tagging it lands here. In a real field session that was **90% of all
313
+ // escalations** — and `FACULTY` then reported every one of them as evidence
314
+ // against "sense of time (Phase 11)", a claim nothing in the record supports.
315
+ //
316
+ // The tester's own first call failed with `unknown step "wait_for"` — a typo
317
+ // — and that too would be filed as evidence about which faculty to build
318
+ // next. CLAUDE.md calls this log the steering wheel; a steering wheel that
319
+ // pools typos, inert controls and slow lists into one reason is pointing
320
+ // somewhere nobody chose.
321
+ //
322
+ // No sixth reason: "unknown is not a reason" stays, and a vocabulary that
323
+ // admits "other" collects a pile of "other". What changes is that the record
324
+ // carries whether the reason was *read off the failure* or *assumed*, and
325
+ // the report declines to recommend a faculty for the assumed ones.
326
+ return { reason: 'verification_failed', candidates: [], tried: [], classified: false };
308
327
  }
309
328
 
310
329
  /**
@@ -440,6 +459,10 @@ export function recordEscalation(udid, {
440
459
  modelTurns = 1,
441
460
  wallMs = null,
442
461
  detail = null,
462
+ // Was this reason read off the failure, or assumed because nothing said?
463
+ // Default `false`, so a caller that does not think about it cannot
464
+ // accidentally claim precision it does not have.
465
+ classified = false,
443
466
  } = {}) {
444
467
  if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
445
468
  if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
@@ -455,6 +478,7 @@ export function recordEscalation(udid, {
455
478
  // What was asked for, in the caller's words. Ground truth for Phase 17's
456
479
  // go/no-go, and on its own it answers "what kind of decision is costing us".
457
480
  intent: intent ? String(intent).slice(0, 120) : null,
481
+ classified: Boolean(classified),
458
482
  step_index: stepIndex,
459
483
  screen_fingerprint: fingerprint,
460
484
  reason,
@@ -635,7 +659,9 @@ export function hpi({ flows, baselines = {} }) {
635
659
  */
636
660
  export function breakdown(records, { session = null, flow = null } = {}) {
637
661
  const byReason = {};
638
- for (const r of REASONS) byReason[r] = 0;
662
+ const classifiedByReason = {};
663
+ const assumedByReason = {};
664
+ for (const r of REASONS) { byReason[r] = 0; classifiedByReason[r] = 0; assumedByReason[r] = 0; }
639
665
  const byScreen = new Map();
640
666
  const byOutcome = {};
641
667
  const bySession = new Map();
@@ -659,6 +685,12 @@ export function breakdown(records, { session = null, flow = null } = {}) {
659
685
  }
660
686
  if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
661
687
  byReason[r.reason] += 1;
688
+ // Three states, not two. A record written before this field existed makes
689
+ // no claim either way, and folding it in with "assumed" would make an old
690
+ // log look like a diagnosis failure — a warning that cries wolf is how a
691
+ // real one gets ignored, which this file already knows in another place.
692
+ if (r.classified === true) classifiedByReason[r.reason] += 1;
693
+ else if (r.classified === false) assumedByReason[r.reason] += 1;
662
694
  byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
663
695
  // Already avoided locally, so not avoidable by anything unbuilt.
664
696
  if (r.outcome !== 'resolved_locally') avoidable += 1;
@@ -685,9 +717,20 @@ export function breakdown(records, { session = null, flow = null } = {}) {
685
717
  pooled: sessions.length > 1 || unattributed > 0,
686
718
  by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
687
719
  by_reason: byReason,
720
+ // How many of each reason were *read off the failure* rather than assumed.
721
+ //
722
+ // The breakdown above picks the next phase, so its precision has to be
723
+ // visible in it. A reason that is mostly assumed is not a finding about an
724
+ // app; it is a count of things nothing could classify, and reading it as a
725
+ // verdict on a faculty is how the instrument came to disagree with a
726
+ // tester who was right.
727
+ classified_by_reason: classifiedByReason,
728
+ assumed_by_reason: assumedByReason,
688
729
  by_outcome: byOutcome,
730
+ // Only where the reason was actually read. A faculty named against a pile
731
+ // of assumptions is advice with nothing behind it.
689
732
  faculty: Object.fromEntries(
690
- REASONS.filter((r) => byReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
733
+ REASONS.filter((r) => classifiedByReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
691
734
  ),
692
735
  avoidable,
693
736
  avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
package/src/navigate.js CHANGED
@@ -133,7 +133,10 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
133
133
  }
134
134
 
135
135
  export function knownScreens(udid) {
136
- return graph.allNodes(udid).map((n) => ({ name: graph.describe(n), hash: n.hash.slice(0, 8), edges: n.edges.length }));
136
+ // A listing needs a handle for every row, so an unnamed screen falls back to
137
+ // its own short hash here — where it is plainly the hash column's value and
138
+ // not a title in quotes.
139
+ return graph.allNodes(udid).map((n) => ({ name: graph.describe(n) ?? n.hash.slice(0, 8), hash: n.hash.slice(0, 8), edges: n.edges.length }));
137
140
  }
138
141
 
139
142
  /**
package/src/ollama.js CHANGED
@@ -49,13 +49,23 @@ export const DEFAULT_HOST = 'http://127.0.0.1:11434';
49
49
  * would produce exactly the invisible unfairness this function exists to
50
50
  * prevent.
51
51
  */
52
- export function readBrief(file = SWIFT) {
52
+ export function readBrief(file = SWIFT, { mayAbstain = false } = {}) {
53
53
  const src = fs.readFileSync(file, 'utf8');
54
- const open = src.indexOf('let instructions = """');
55
- if (open === -1) throw new Error(`no instructions block in ${file}`);
54
+ const base = block(src, 'let instructions = """', file);
55
+ if (!mayAbstain) return base;
56
+ // The addendum, from the same file and joined the way the Swift joins it.
57
+ // The fourth word has to reach both arms identically or the comparison is
58
+ // measuring two different briefs, which is the failure this whole function
59
+ // exists to prevent.
60
+ return `${base}\n\n${block(src, 'let abstainInstructions = instructions + "\\n\\n" + """', file)}`;
61
+ }
62
+
63
+ function block(src, opener, file) {
64
+ const open = src.indexOf(opener);
65
+ if (open === -1) throw new Error(`no ${JSON.stringify(opener)} block in ${file}`);
56
66
  const bodyStart = src.indexOf('\n', open) + 1;
57
67
  const close = src.indexOf('"""', bodyStart);
58
- if (close === -1) throw new Error(`unterminated instructions block in ${file}`);
68
+ if (close === -1) throw new Error(`unterminated block in ${file}`);
59
69
  return src
60
70
  .slice(bodyStart, close)
61
71
  .split('\n')
@@ -106,11 +116,28 @@ export function promptFor(s = {}) {
106
116
  * against would be a second difference between the arms, and the rationale is
107
117
  * the one thing a supervisor is explicitly not trusted for.
108
118
  */
109
- export const SCHEMA = {
119
+ export const DECISION_WORDS = ['wait', 'retry', 'stop'];
120
+
121
+ /**
122
+ * The fourth word, available on request.
123
+ *
124
+ * Not a new capability and it cannot become one: `abstain` means "behave as if
125
+ * there is no supervisor", which is the `null` every caller already handles on
126
+ * every failure path. It strictly *shrinks* what the model can cause to happen,
127
+ * which is why a component whose answer space is its safety property can afford
128
+ * to grow one.
129
+ */
130
+ export const ABSTAIN = 'abstain';
131
+
132
+ export const schemaFor = ({ mayAbstain = false } = {}) => ({
110
133
  type: 'object',
111
- properties: { decision: { type: 'string', enum: ['wait', 'retry', 'stop'] } },
134
+ properties: {
135
+ decision: { type: 'string', enum: mayAbstain ? [...DECISION_WORDS, ABSTAIN] : DECISION_WORDS },
136
+ },
112
137
  required: ['decision'],
113
- };
138
+ });
139
+
140
+ export const SCHEMA = schemaFor();
114
141
 
115
142
  /** Parse `ollama`, `ollama:qwen3:14b`, `ollama:qwen3:8b@http://host:port`. */
116
143
  export function parseTarget(raw) {
@@ -147,15 +174,16 @@ async function post(host, route, body, timeoutMs) {
147
174
  */
148
175
  export async function ask(target, situation, timeoutMs = 2500) {
149
176
  const { model, host } = target;
177
+ const mayAbstain = Boolean(situation?.mayAbstain);
150
178
  const started = Date.now();
151
179
  const out = await post(host, '/api/chat', {
152
180
  model,
153
181
  messages: [
154
- { role: 'system', content: readBrief() },
182
+ { role: 'system', content: readBrief(SWIFT, { mayAbstain }) },
155
183
  { role: 'user', content: promptFor(situation) },
156
184
  ],
157
185
  stream: false,
158
- format: SCHEMA,
186
+ format: schemaFor({ mayAbstain }),
159
187
  // Qwen3 reasons out loud by default. Turned off for two reasons and both
160
188
  // are about fairness rather than speed: the Apple arm does not deliberate
161
189
  // either, and a supervisor that takes twenty seconds to answer has already
@@ -1018,6 +1018,37 @@ async function restartDevice(serial) {
1018
1018
  );
1019
1019
  }
1020
1020
 
1021
+ /**
1022
+ * Reading an app's own storage is not implemented for Android, and says so.
1023
+ *
1024
+ * Not a stub and not a borrowed answer. The iOS version reads a CoreSimulator
1025
+ * data container straight off the host filesystem, which an emulator has no
1026
+ * equivalent of: an app's files live inside the emulator's own userdata image,
1027
+ * and the way in is `adb shell run-as <package>` — which works only for a
1028
+ * debuggable build, needs the emulator running, and would be a different
1029
+ * feature with different guarantees rather than the same one.
1030
+ *
1031
+ * The standing rule is that a layer a platform does not have is declined with a
1032
+ * reason, never described in the other platform's vocabulary. Claiming a data
1033
+ * container here is how `doctor` once told an emulator its input driver was
1034
+ * idb.
1035
+ */
1036
+ const noStorage = (serial, what) => {
1037
+ throw new Error(
1038
+ `simframe cannot read ${what} on an emulator (${serial}) yet. The iOS version reads a`
1039
+ + ' simulator data container off the host filesystem and an emulator has no such thing —'
1040
+ + ' its app data lives inside the userdata image, reachable only through'
1041
+ + ' `adb shell run-as <package>` on a debuggable build, with the emulator running.'
1042
+ + ' That is a different feature and it has not been built.',
1043
+ );
1044
+ };
1045
+
1046
+ async function listApps(serial) { return noStorage(serial, 'the list of installed apps'); }
1047
+ async function appContainer(serial) { return noStorage(serial, "an app's data container"); }
1048
+ async function readPropertyList() {
1049
+ throw new Error('property lists are an iOS format; Android has no equivalent to read');
1050
+ }
1051
+
1021
1052
  /** @type {import('./index.js').Platform} */
1022
1053
  export const platform = {
1023
1054
  id: 'android',
@@ -1034,6 +1065,9 @@ export const platform = {
1034
1065
  terminateApp,
1035
1066
  openUrl,
1036
1067
  restartDevice,
1068
+ listApps,
1069
+ appContainer,
1070
+ readPropertyList,
1037
1071
  setPermission,
1038
1072
  setPasteboard,
1039
1073
  getPasteboard,
@@ -63,6 +63,7 @@ export const PLATFORM_SURFACE = Object.freeze([
63
63
  'geometry', 'inputDriver',
64
64
  'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
+ 'listApps', 'appContainer', 'readPropertyList',
66
67
  'bootedAt',
67
68
  ]);
68
69
 
@@ -209,6 +210,12 @@ export const openUrl = (udid, ...args) => platformFor(udid).openUrl(udid, ...arg
209
210
  export const setPermission = (udid, ...args) => platformFor(udid).setPermission(udid, ...args);
210
211
  export const setPasteboard = (udid, ...args) => platformFor(udid).setPasteboard(udid, ...args);
211
212
 
213
+ // Reading what an app persisted. Routed like everything else, and declined by a
214
+ // backend that has no equivalent rather than answered in the other's terms.
215
+ export const listApps = (udid, ...args) => platformFor(udid).listApps(udid, ...args);
216
+ export const appContainer = (udid, ...args) => platformFor(udid).appContainer(udid, ...args);
217
+ export const readPropertyList = (udid, file) => platformFor(udid).readPropertyList(file);
218
+
212
219
  /**
213
220
  * The permission services a device understands, or every service any backend
214
221
  * understands when no device is named.
@@ -9,6 +9,7 @@ import fs from 'node:fs';
9
9
  import os from 'node:os';
10
10
  import path from 'node:path';
11
11
  import { promisify } from 'node:util';
12
+ import * as plist from './plist.js';
12
13
 
13
14
  const run = promisify(execFile);
14
15
 
@@ -178,17 +179,42 @@ function isBootedSync(udid) {
178
179
  return false;
179
180
  }
180
181
 
182
+ /**
183
+ * What went wrong with a `simctl io screenshot`, in one sentence.
184
+ *
185
+ * Extracted so it can be *tested* rather than reasoned about, for the same
186
+ * reason `pickDevice` and `decisionOf` were: this is the line where a wrong
187
+ * answer was expensive, and it was wrong for a day.
188
+ */
189
+ export function screenshotFailure(err) {
190
+ const killed = err.killed || err.signal === 'SIGTERM';
191
+ // `simctl` opens with `Note: No display specified …` on every run, success or
192
+ // failure. When the display surface is dead the command does not fail, it
193
+ // *hangs* — so at kill time that Note is the only thing on stderr, and the
194
+ // tool reported a benign informational line as the reason a capture failed.
195
+ // That is how this wedge stayed nameless through five CI failures.
196
+ const lines = String(err.stderr || '').trim().split('\n').map((l) => l.trim()).filter(Boolean);
197
+ const real = lines.filter((l) => !/^Note:/i.test(l)).pop();
198
+ if (killed && !real) {
199
+ // Run to completion the device names it exactly:
200
+ // NSPOSIXErrorDomain code 60 — Timeout waiting for screen surfaces
201
+ // which is CoreSimulator saying the surface is gone, and the closest thing
202
+ // to a positive test for the wedge that exists.
203
+ return 'simctl screenshot did not return within 10s. The display surface is not answering'
204
+ + ' — run to completion it reports "Timeout waiting for screen surfaces" (NSPOSIXErrorDomain 60).'
205
+ + ' This is the device, not the capture loop: `simframe revive` restarts it.';
206
+ }
207
+ const detail = real ?? lines.pop();
208
+ return detail ? `simctl screenshot failed: ${detail}` : `simctl screenshot failed: ${err.message}`;
209
+ }
210
+
181
211
  async function screenshot(udid, outFile, { mask = 'ignored' } = {}) {
182
212
  try {
183
213
  await run('xcrun', ['simctl', 'io', udid, 'screenshot', '--type=png', `--mask=${mask}`, outFile], {
184
214
  timeout: 10_000,
185
215
  });
186
216
  } catch (err) {
187
- // Same reason as launchApp: execFile's message is "Command failed: <the
188
- // whole command>" and simctl's actual complaint is in stderr. A CI failure
189
- // here reported the command and nothing about why it did not work.
190
- const detail = (err.stderr || '').trim().split('\n').filter(Boolean).pop();
191
- throw new Error(detail ? `simctl screenshot failed: ${detail}` : `simctl screenshot failed: ${err.message}`);
217
+ throw new Error(screenshotFailure(err));
192
218
  }
193
219
  }
194
220
 
@@ -309,6 +335,100 @@ function ownsUdid(udid) {
309
335
  return /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(String(udid ?? ''));
310
336
  }
311
337
 
338
+ /**
339
+ * Where CoreSimulator keeps a device's data.
340
+ *
341
+ * Read straight off disk, and that is the whole point of this family rather
342
+ * than an optimisation. Measured on this Xcode: `simctl get_app_container` and
343
+ * `simctl listapps` **both** fail on a device that is not running —
344
+ * `Unable to lookup in current state: Shutdown`. The field report that asked
345
+ * for this feature rated it highest-leverage precisely because it answered
346
+ * "what did the app save?" *before the device was booted*, and simctl cannot do
347
+ * that. The filesystem can, so this reads the filesystem.
348
+ */
349
+ const deviceRoot = (udid) =>
350
+ path.join(os.homedir(), 'Library/Developer/CoreSimulator/Devices', String(udid));
351
+
352
+ const containerRoot = (udid) => path.join(deviceRoot(udid), 'data/Containers/Data/Application');
353
+
354
+ /** The per-container metadata file that says which app owns it. */
355
+ const METADATA = '.com.apple.mobile_container_manager.metadata.plist';
356
+
357
+ /**
358
+ * Every app with a data container on this device, booted or not.
359
+ *
360
+ * The bundle id lives in `MCMMetadataIdentifier` in each container's metadata
361
+ * plist. It is *not* recoverable by grepping the file — the binary plist
362
+ * encodes strings in a way that does not leave the id as a plain substring, and
363
+ * an early version of this that tried to pre-filter that way matched nothing.
364
+ * So each metadata file is asked properly. Measured at **0.52s for 150
365
+ * containers**, which is a listing cost rather than a per-read one.
366
+ */
367
+ async function listApps(udid) {
368
+ const root = containerRoot(udid);
369
+ let entries;
370
+ try {
371
+ entries = await fs.promises.readdir(root, { withFileTypes: true });
372
+ } catch (err) {
373
+ if (err.code === 'ENOENT') {
374
+ const exists = fs.existsSync(deviceRoot(udid));
375
+ throw new Error(exists
376
+ ? `device ${udid} has no app data containers yet — nothing has been installed on it`
377
+ : `no simulator data directory for ${udid} (looked in ${root})`);
378
+ }
379
+ throw err;
380
+ }
381
+ const apps = [];
382
+ await Promise.all(entries.filter((e) => e.isDirectory()).map(async (e) => {
383
+ const dir = path.join(root, e.name);
384
+ try {
385
+ const { stdout } = await run('plutil',
386
+ ['-extract', 'MCMMetadataIdentifier', 'raw', '-o', '-', path.join(dir, METADATA)],
387
+ { timeout: 10_000 });
388
+ const bundleId = stdout.trim();
389
+ if (bundleId) apps.push({ bundleId, container: dir });
390
+ } catch {
391
+ // A container without readable metadata is not an app we can name, and
392
+ // naming it by its UUID would be offering an id nobody can use.
393
+ }
394
+ }));
395
+ return apps.sort((a, b) => a.bundleId.localeCompare(b.bundleId));
396
+ }
397
+
398
+ /** The data container for one app, or a listing of what is there instead. */
399
+ async function appContainer(udid, bundleId) {
400
+ const apps = await listApps(udid);
401
+ const hit = apps.find((a) => a.bundleId === bundleId);
402
+ if (hit) return hit.container;
403
+ // Near misses first: the id is the thing people get wrong, and a bare "not
404
+ // installed" on a device with the app under a slightly different id is the
405
+ // least useful true sentence available.
406
+ const needle = String(bundleId).toLowerCase();
407
+ const near = apps.filter((a) => a.bundleId.toLowerCase().includes(needle)
408
+ || needle.includes(a.bundleId.toLowerCase())).slice(0, 5);
409
+ throw new Error(
410
+ `"${bundleId}" has no data container on ${udid}`
411
+ + (near.length ? ` — did you mean ${near.map((a) => a.bundleId).join(', ')}?` : '')
412
+ + ` (${apps.length} app(s) have one)`,
413
+ );
414
+ }
415
+
416
+ /**
417
+ * Read a property list, whatever it contains.
418
+ *
419
+ * `-convert xml1` and not `json`: six of the twenty real preference plists on
420
+ * the bench device cannot be represented as JSON at all, because `<data>` and
421
+ * `<date>` have no JSON form and plutil refuses rather than inventing one. See
422
+ * the note at the top of plist.js.
423
+ */
424
+ async function readPropertyList(file) {
425
+ const { stdout } = await run('plutil', ['-convert', 'xml1', '-o', '-', file], {
426
+ timeout: 20_000,
427
+ maxBuffer: 64 * 1024 * 1024,
428
+ });
429
+ return plist.parse(stdout);
430
+ }
431
+
312
432
  /**
313
433
  * The prerequisites `simframe doctor` reports for this backend. Returned rather
314
434
  * than printed so doctor stays one renderer: a backend says what it needs, and
@@ -415,6 +535,9 @@ export const platform = {
415
535
  terminateApp,
416
536
  openUrl,
417
537
  restartDevice,
538
+ listApps,
539
+ appContainer,
540
+ readPropertyList,
418
541
  setPermission,
419
542
  setPasteboard,
420
543
  permissionServices: () => PERMISSION_SERVICES,