simframe 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +165 -5
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/simframed/main.swift +13 -1
  9. package/native/supervise.swift +181 -0
  10. package/package.json +4 -1
  11. package/scripts/check-package.mjs +22 -2
  12. package/scripts/check-private.mjs +143 -0
  13. package/scripts/ci-memory.mjs +104 -20
  14. package/scripts/eval-perception.mjs +281 -0
  15. package/scripts/phase17-corpus.mjs +176 -0
  16. package/skills/simframe/SKILL.md +237 -5
  17. package/src/actions.js +1825 -38
  18. package/src/analyze.js +70 -0
  19. package/src/cli.js +214 -15
  20. package/src/control.js +1 -0
  21. package/src/fingerprint.js +19 -1
  22. package/src/graph.js +193 -11
  23. package/src/index.js +428 -16
  24. package/src/input.js +115 -8
  25. package/src/localhelper.js +155 -0
  26. package/src/matching.js +119 -3
  27. package/src/mcp.js +319 -27
  28. package/src/metrics.js +134 -8
  29. package/src/navigate.js +10 -7
  30. package/src/ocr.js +18 -1
  31. package/src/planner.js +195 -0
  32. package/src/platform/android.js +3 -2
  33. package/src/platform/ios.js +2 -1
  34. package/src/png.js +26 -0
  35. package/src/refs.js +51 -8
  36. package/src/regions.js +110 -1
  37. package/src/screenmap.js +109 -10
  38. package/src/supervisor.js +117 -0
  39. package/src/view.js +396 -7
  40. package/src/vocabulary.js +134 -0
  41. package/src/wrote.js +136 -0
package/src/metrics.js CHANGED
@@ -40,8 +40,16 @@ export const FACULTY = {
40
40
  no_plan: 'exploration (Phase 14)',
41
41
  };
42
42
 
43
- /** Faculties that exist. Empty until Phase 11 lands the first one. */
44
- export const BUILT_FACULTIES = new Set();
43
+ /**
44
+ * Faculties that exist.
45
+ *
46
+ * Phase 11 landed the first one, so `verification_failed` no longer maps to
47
+ * something unbuilt — which changes what those records *mean*. Before, they
48
+ * were a queue waiting on a phase. Now they are evidence that the phase which
49
+ * shipped is not sufficient, and that is a more useful thing for the log to be
50
+ * able to say than a count of things nobody has written yet.
51
+ */
52
+ export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
45
53
 
46
54
  function metricPaths(udid) {
47
55
  const dir = store.deviceDir(udid);
@@ -111,9 +119,27 @@ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts
111
119
  * matching error strings at the boundary — a regexed message is a reason that
112
120
  * silently becomes "unknown" the day somebody rewords it.
113
121
  */
114
- export function tag(err, reason, { candidates = [], tried = [] } = {}) {
122
+ export function tag(err, reason, { candidates = [], tried = [], ambiguous = false, intent = null } = {}) {
115
123
  if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
116
- err.escalation = { reason, candidates, tried };
124
+ // `ambiguous` is narrower than the reason, and that is the point. Two very
125
+ // different failures both tag `ambiguous_intent`: the target is on screen
126
+ // several times over, and the target is not on screen at all on a screen we
127
+ // thought we knew. Only the first is resolvable by *choosing*, and only the
128
+ // first tells a waiting caller that waiting is pointless — the thing it is
129
+ // waiting for has already arrived.
130
+ // `intent` is the goal in the caller's own words, recorded as a field rather
131
+ // than left in the prose of `detail`.
132
+ //
133
+ // Phase 17's go/no-go asks whether an on-device model would pick the element
134
+ // Claude picked, given the goal and the element list. The element list is
135
+ // here as `candidates` and the eventual choice is recoverable from the
136
+ // graph — the tap that finally worked on this screen becomes a verified edge
137
+ // carrying its own step. The goal was the missing third, and it was sitting
138
+ // inside a sentence: `"X" matches 3 things on this screen — say which…`.
139
+ // Regexing it back out at export time is the exact habit this file exists to
140
+ // avoid, and it would silently return nothing the day that sentence is
141
+ // reworded.
142
+ err.escalation = { reason, candidates, tried, ambiguous, intent };
117
143
  return err;
118
144
  }
119
145
 
@@ -213,8 +239,64 @@ export function fingerprintNow(udid, screenmap) {
213
239
  * measurement's clothes, and every number in this project is supposed to say
214
240
  * where it came from.
215
241
  */
242
+ /**
243
+ * Which run of which program wrote a record.
244
+ *
245
+ * The log is per-device and, until now, anonymous — so two agents driving one
246
+ * booted simulator wrote one interleaved file with no way to separate them.
247
+ * Measured on the bench device in a single evening: 57 records to 81, a third
248
+ * of the new ones naming screens from an app the suite has never launched.
249
+ *
250
+ * That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
251
+ * the reason breakdown of this log the thing that chooses which faculty gets
252
+ * built next, and a breakdown that silently pools two sessions errs toward
253
+ * whichever of them made more mistakes — which is not the same question as
254
+ * which faculty is missing.
255
+ *
256
+ * A pid alone would not do: pids are reused, and the useful grouping is "one
257
+ * agent's run", which for the MCP server is the life of the process and for
258
+ * the CLI is a single command. So: the start time, the pid, and a random tail,
259
+ * computed once per process. `client` says what kind of process it was, since
260
+ * "the MCP server did this" and "somebody ran a CLI command" deserve different
261
+ * readings of the same reason.
262
+ *
263
+ * Deliberately not a device id, a username, or anything about the machine. This
264
+ * file is committed to a public repo in summary form, and the question it has
265
+ * to answer is "was this all one agent", which needs no identity to answer.
266
+ */
267
+ const SESSION_ID = process.env.SIMFRAME_SESSION
268
+ ? String(process.env.SIMFRAME_SESSION).slice(0, 64)
269
+ : `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
270
+
271
+ /*
272
+ * Minting the id from the pid was right for the MCP server, which is one
273
+ * long-lived process, and wrong for everything else. A CLI-driven agent starts
274
+ * a process per command, so it got one "session" per command: on the benchmark
275
+ * device, 33 session ids for 46 records, 30 of them holding a single record.
276
+ * `escalations --session` was therefore unable to answer the one question it
277
+ * exists for, and Phase 17's go/no-go step 1 — "filter to one session id" —
278
+ * had nothing to filter. `SIMFRAME_SESSION` lets a caller that knows it is one
279
+ * session say so; the per-process id stays the default.
280
+ */
281
+
282
+ /** How this process is being used, for reading a breakdown afterwards. */
283
+ function clientKind() {
284
+ const argv = process.argv.join(' ');
285
+ if (/\bmcp\b/.test(argv)) return 'mcp';
286
+ if (/bench-hpi|scripts\//.test(argv)) return 'script';
287
+ if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
288
+ return 'library';
289
+ }
290
+ const CLIENT = clientKind();
291
+
292
+ /** The session this process's records belong to. Exported for `escalations`. */
293
+ export const sessionId = () => SESSION_ID;
294
+ export const clientName = () => CLIENT;
295
+
216
296
  export function recordEscalation(udid, {
217
297
  flowId = null,
298
+ flowName = null,
299
+ intent = null,
218
300
  stepIndex = null,
219
301
  fingerprint = null,
220
302
  reason,
@@ -229,7 +311,16 @@ export function recordEscalation(udid, {
229
311
  if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
230
312
  const record = {
231
313
  timestamp: new Date().toISOString(),
314
+ // Added after the log turned out to pool two agents' work invisibly. Both
315
+ // are cheap and neither is derivable afterwards, which is the test for
316
+ // whether a field belongs in a log at all.
317
+ session_id: SESSION_ID,
318
+ client: CLIENT,
232
319
  flow_id: flowId,
320
+ flow_name: flowName,
321
+ // What was asked for, in the caller's words. Ground truth for Phase 17's
322
+ // go/no-go, and on its own it answers "what kind of decision is costing us".
323
+ intent: intent ? String(intent).slice(0, 120) : null,
233
324
  step_index: stepIndex,
234
325
  screen_fingerprint: fingerprint,
235
326
  reason,
@@ -408,14 +499,31 @@ export function hpi({ flows, baselines = {} }) {
408
499
  * rate is 1.0 by construction and says nothing. The per-reason breakdown is
409
500
  * the part that decides the next phase, and it is informative today.
410
501
  */
411
- export function breakdown(records) {
502
+ export function breakdown(records, { session = null, flow = null } = {}) {
412
503
  const byReason = {};
413
504
  for (const r of REASONS) byReason[r] = 0;
414
505
  const byScreen = new Map();
415
506
  const byOutcome = {};
507
+ const bySession = new Map();
508
+ const byFlow = new Map();
416
509
  let avoidable = 0;
417
- for (const r of records) {
510
+ let unattributed = 0;
511
+ // Filtering happens here rather than at the call site so `total` and every
512
+ // rate below it describe the same set of records.
513
+ const kept = records.filter((r) => (session ? r?.session_id === session : true))
514
+ .filter((r) => (flow ? r?.flow_name === flow : true));
515
+ for (const r of kept) {
418
516
  if (!REASONS.includes(r?.reason)) continue;
517
+ // Records written before sessions were recorded cannot be attributed, and
518
+ // saying how many there are is the difference between a breakdown that
519
+ // pools two agents and one that says it might be.
520
+ if (r.session_id) {
521
+ const key = `${r.session_id}|${r.client ?? '?'}`;
522
+ bySession.set(key, (bySession.get(key) ?? 0) + 1);
523
+ } else {
524
+ unattributed += 1;
525
+ }
526
+ if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
419
527
  byReason[r.reason] += 1;
420
528
  byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
421
529
  // Already avoided locally, so not avoidable by anything unbuilt.
@@ -423,9 +531,25 @@ export function breakdown(records) {
423
531
  const key = r.screen_fingerprint ?? '(no fingerprint)';
424
532
  byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
425
533
  }
426
- const total = records.filter((r) => REASONS.includes(r?.reason)).length;
534
+ const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
535
+ const sessions = [...bySession.entries()]
536
+ .map(([key, count]) => {
537
+ const [id, client] = key.split('|');
538
+ return { session_id: id, client, count };
539
+ })
540
+ .sort((a, b) => b.count - a.count);
427
541
  return {
428
542
  total,
543
+ // The log is per-device and shared: two agents on one booted simulator
544
+ // write one interleaved file. More than one session here means the counts
545
+ // below are a pool, and CLAUDE.md uses those counts to choose a phase.
546
+ sessions,
547
+ session_count: sessions.length,
548
+ unattributed,
549
+ // Any unattributed record at all makes this a pool: the whole point is
550
+ // that they cannot be told apart, and 92 of them is not "one session".
551
+ pooled: sessions.length > 1 || unattributed > 0,
552
+ by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
429
553
  by_reason: byReason,
430
554
  by_outcome: byOutcome,
431
555
  faculty: Object.fromEntries(
@@ -437,7 +561,9 @@ export function breakdown(records) {
437
561
  .sort((a, b) => b[1] - a[1])
438
562
  .slice(0, 10)
439
563
  .map(([fingerprint, count]) => ({ fingerprint, count })),
440
- model_turns_spent: records.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
564
+ // `kept`, not `records` — a filtered breakdown that reports the whole
565
+ // log's model turns is the same class of mistake as pooling two sessions.
566
+ model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
441
567
  };
442
568
  }
443
569
 
package/src/navigate.js CHANGED
@@ -41,12 +41,15 @@ export function stepFor(edge) {
41
41
  * count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
42
42
  * have one owner.
43
43
  */
44
- function refuse(udid, result, { detail = null } = {}) {
44
+ function refuse(udid, result, { detail = null, flowName = null } = {}) {
45
45
  try {
46
46
  const reason = metrics.PLAN_REASONS[result.reason];
47
47
  if (reason) {
48
48
  metrics.recordEscalation(udid, {
49
49
  reason,
50
+ // A refusal by `goto` is about a destination and one by `flow run` is
51
+ // about a named flow. Either is what a breakdown wants to group by.
52
+ flowName,
50
53
  fingerprint: metrics.fingerprintNow(udid, screenmap),
51
54
  outcome: 'escalated_to_model',
52
55
  detail: detail ?? result.reason,
@@ -70,8 +73,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
70
73
  const udid = device.udid;
71
74
 
72
75
  const found = graph.findScreen(udid, target);
73
- if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"` });
74
- if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens` });
76
+ if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"`, flowName: `goto:${target}` });
77
+ if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens`, flowName: `goto:${target}` });
75
78
 
76
79
  const here = await api.screenIdentity(udid, {});
77
80
  if (here.hash === found.node.hash) {
@@ -84,12 +87,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
84
87
  // `Cannot read properties of null (reading 'slice')` instead of answering.
85
88
  // Not hypothetical on Android, where README's own table puts the launcher at
86
89
  // one token.
87
- if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name });
90
+ if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name }, { flowName: `goto:${target}` });
88
91
  const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
89
- if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name });
92
+ if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name }, { flowName: `goto:${target}` });
90
93
 
91
94
  const steps = path_.map(stepFor);
92
- if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name });
95
+ if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name }, { flowName: `goto:${target}` });
93
96
 
94
97
  const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
95
98
  const arrived = await api.screenIdentity(udid, {});
@@ -148,7 +151,7 @@ export function listFlows(udid) {
148
151
  export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
149
152
  const { device } = await api.ensureDaemon(deviceQuery, options);
150
153
  const flow = loadFlow(device.udid, name);
151
- if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"` });
154
+ if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"`, flowName: name });
152
155
  // A replayed flow knows its own name, so its record can be compared against
153
156
  // a human doing the same thing. `minSteps` comes from the flow definition or
154
157
  // stays null — the step count of a recorded route is not a claim about the
package/src/ocr.js CHANGED
@@ -19,6 +19,17 @@ const BIN = path.join(BIN_DIR, 'ocr');
19
19
 
20
20
  let ready = null;
21
21
 
22
+ /**
23
+ * Which recognition level to ask Vision for.
24
+ *
25
+ * `accurate` is the default and what CLAUDE.md fixes; `fast` is the other thing
26
+ * Vision offers. Exposed so the pair can be scored against each other instead
27
+ * of one of them being a constant nobody measured.
28
+ */
29
+ export function level() {
30
+ return String(process.env.SIMFRAME_OCR ?? '').toLowerCase() === 'fast' ? 'fast' : 'accurate';
31
+ }
32
+
22
33
  /** Compile once, then reuse. Recompiles only if the source is newer than the binary. */
23
34
  export async function ensureBinary() {
24
35
  if (ready) return ready;
@@ -55,7 +66,13 @@ export async function ensureBinary() {
55
66
  export async function readText(pngFile, { density = 3 } = {}) {
56
67
  const built = await ensureBinary();
57
68
  if (!built.available) throw new Error(built.reason);
58
- const { stdout } = await run(built.binary, [pngFile], { timeout: 30_000, maxBuffer: 16 << 20 });
69
+ // The recognition level rides in the environment rather than in argv, so the
70
+ // Swift side keeps its one-argument contract and an older binary still works.
71
+ const { stdout } = await run(built.binary, [pngFile], {
72
+ timeout: 30_000,
73
+ maxBuffer: 16 << 20,
74
+ env: { ...process.env, SIMFRAME_OCR: level() },
75
+ });
59
76
  const raw = JSON.parse(stdout || '[]');
60
77
  return raw.map((r) => ({
61
78
  text: r.text,
package/src/planner.js ADDED
@@ -0,0 +1,195 @@
1
+ /**
2
+ * The local planner tier: a ranker, behind a flag, that may only reorder.
3
+ *
4
+ * **Why this exists at all, given Phase 17 was a no-go.** That phase asked a
5
+ * local model to *choose the next element*, and the answer was that the matcher
6
+ * already does — 37 of 40 real decisions. This is the complement and the one
7
+ * case a string matcher structurally cannot do: the goal matches **nothing** on
8
+ * screen, and something has to guess which container leads to it. "Change my
9
+ * username" shares no prefix, synonym or typo distance with "Account".
10
+ *
11
+ * Measured on this machine, six hand-written cases: 5 of 6 top-1, 6 of 6 top-3,
12
+ * median 564 ms warm. See `docs/BENCHMARKS.md`. Against a model round trip at
13
+ * 10–16 s that is roughly twenty times cheaper; against the honest baseline —
14
+ * breadth-first ordering, which needs no model — it won five of six.
15
+ *
16
+ * **What it is allowed to do, and it is deliberately almost nothing.** It
17
+ * reorders a list of candidates the caller has already permitted and will try
18
+ * in some order regardless. It cannot invent a label, cannot choose an action,
19
+ * cannot see pixels, and never runs on a destructive label because the caller
20
+ * filtered those out before asking (`src/vocabulary.js`). If it is wrong the
21
+ * exploration budget simply tries the next one. That is strictly weaker
22
+ * authority than Phase 17 proposed, which is what makes it safe to try.
23
+ *
24
+ * **Off unless asked.** `SIMFRAME_PLANNER=apple` turns it on; anything else,
25
+ * or any failure at all, degrades to `null` and the caller keeps its own order.
26
+ * `doctor` reports which. CI runs with it off.
27
+ */
28
+ import { spawn, execFile } from 'node:child_process';
29
+ import fs from 'node:fs';
30
+ import path from 'node:path';
31
+ import { fileURLToPath } from 'node:url';
32
+ import { promisify } from 'node:util';
33
+ import * as store from './store.js';
34
+
35
+ const run = promisify(execFile);
36
+ const SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'native', 'rank.swift');
37
+ const BIN = path.join(store.ROOT, 'bin', 'rank');
38
+
39
+ /** Which backend the caller asked for. Absent means no local planner. */
40
+ export function requested(options) {
41
+ // Per call first, then the environment, for the reason in `sensorMode`: an
42
+ // MCP server's environment is fixed when it spawns, so a tester could not
43
+ // switch backends inside one session and a round came back with one arm of
44
+ // its A/B unrun.
45
+ const raw = String(options?.planner ?? process.env.SIMFRAME_PLANNER ?? '').trim().toLowerCase();
46
+ if (!raw || raw === 'none' || raw === 'off' || raw === '0' || raw === 'false') return null;
47
+ return raw;
48
+ }
49
+
50
+ let building = null;
51
+
52
+ export async function ensureBinary() {
53
+ if (building) return building;
54
+ building = (async () => {
55
+ try {
56
+ const src = fs.statSync(SOURCE).mtimeMs;
57
+ const bin = fs.existsSync(BIN) ? fs.statSync(BIN).mtimeMs : 0;
58
+ if (bin > src) return { available: true, binary: BIN };
59
+ } catch {
60
+ return { available: false, reason: 'the ranker source is missing from this install' };
61
+ }
62
+ try {
63
+ fs.mkdirSync(path.dirname(BIN), { recursive: true });
64
+ // `swiftc`, not `xcrun swiftc`: nothing above the platform boundary may
65
+ // name a platform tool, and the boundary test catches it. `src/ocr.js`
66
+ // set this precedent — a compiler is not a device tool.
67
+ await run('swiftc', ['-O', SOURCE, '-o', BIN], { timeout: 180_000 });
68
+ return { available: true, binary: BIN };
69
+ } catch (err) {
70
+ building = null; // let a later call retry once a toolchain is present
71
+ return {
72
+ available: false,
73
+ reason: err.code === 'ENOENT'
74
+ ? 'swiftc is not installed, so the local planner cannot be built (install Xcode command line tools)'
75
+ : `could not build the local planner: ${String(err.message).split('\n')[0]}`,
76
+ };
77
+ }
78
+ })();
79
+ return building;
80
+ }
81
+
82
+ let session = null;
83
+
84
+ /** Start the helper once and keep it, because the first answer pays model load. */
85
+ async function open() {
86
+ if (session) return session;
87
+ const built = await ensureBinary();
88
+ if (!built.available) return { ok: false, reason: built.reason };
89
+ session = await new Promise((resolve) => {
90
+ const child = spawn(built.binary, [], { stdio: ['pipe', 'pipe', 'ignore'] });
91
+ // Deliberately NOT unref'd. Unreffing the child's stdout unreferences the
92
+ // very pipe every request waits on, so the process exited silently in the
93
+ // middle of an await — a flow that printed nothing and returned 0. The
94
+ // helper is closed explicitly instead, by whoever opened it.
95
+ let buffer = '';
96
+ const waiters = [];
97
+ let settled = false;
98
+ const fail = (reason) => {
99
+ if (!settled) { settled = true; resolve({ ok: false, reason }); }
100
+ while (waiters.length) waiters.shift()(null);
101
+ };
102
+ child.on('error', (err) => fail(`the local planner would not start: ${err.message}`));
103
+ child.on('exit', () => { session = null; fail('the local planner exited'); });
104
+ child.stdout.on('data', (chunk) => {
105
+ buffer += chunk;
106
+ let i = buffer.indexOf('\n');
107
+ while (i >= 0) {
108
+ const line = buffer.slice(0, i).trim();
109
+ buffer = buffer.slice(i + 1);
110
+ i = buffer.indexOf('\n');
111
+ if (!line) continue;
112
+ let msg;
113
+ try { msg = JSON.parse(line); } catch { continue; }
114
+ if (!settled) {
115
+ settled = true;
116
+ if (msg.ready) resolve({ ok: true, child, waiters });
117
+ else resolve({ ok: false, reason: msg.unavailable ?? 'the local planner did not become ready' });
118
+ continue;
119
+ }
120
+ const next = waiters.shift();
121
+ if (next) next(msg);
122
+ }
123
+ });
124
+ });
125
+ return session;
126
+ }
127
+
128
+ /**
129
+ * Reorder `options` by which is likeliest to lead to `goal`.
130
+ *
131
+ * @returns {Promise<string[]|null>} the caller's own order is correct when this
132
+ * is null, which is every failure mode: flag off, no model, a timeout, a
133
+ * parse problem, a paraphrasing answer. Never throws.
134
+ */
135
+ export async function rank(goal, options, { timeoutMs = 3000, deviceOptions } = {}) {
136
+ if (!requested(deviceOptions)) return null;
137
+ if (!goal || !Array.isArray(options) || options.length < 2) return null;
138
+ let live;
139
+ try {
140
+ live = await open();
141
+ } catch {
142
+ return null;
143
+ }
144
+ if (!live?.ok) return null;
145
+ const answer = await new Promise((resolve) => {
146
+ // A timed-out waiter has to be *retired*, not merely resolved. Leaving it in
147
+ // the queue meant the next answer went to it instead of to the next asker,
148
+ // and every call after that was off by one — which showed up as an
149
+ // exploration run that never finished rather than as an error.
150
+ let done = false;
151
+ const waiter = (msg) => {
152
+ if (done) return;
153
+ done = true;
154
+ clearTimeout(timer);
155
+ resolve(msg);
156
+ };
157
+ const timer = setTimeout(() => {
158
+ if (done) return;
159
+ done = true;
160
+ const i = live.waiters.indexOf(waiter);
161
+ if (i >= 0) live.waiters.splice(i, 1);
162
+ resolve(null);
163
+ }, timeoutMs);
164
+ live.waiters.push(waiter);
165
+ try {
166
+ live.child.stdin.write(`${JSON.stringify({ goal: String(goal), options })}\n`);
167
+ } catch {
168
+ clearTimeout(timer);
169
+ resolve(null);
170
+ }
171
+ });
172
+ if (!answer?.order?.length) return null;
173
+ // It often returns a subset, so its order comes first and ours fills the tail.
174
+ // Trusting it to be exhaustive would silently drop candidates the budget was
175
+ // going to try.
176
+ const ranked = answer.order.filter((label) => options.includes(label));
177
+ const seen = new Set(ranked);
178
+ return [...ranked, ...options.filter((o) => !seen.has(o))];
179
+ }
180
+
181
+ /** For `doctor`: what the planner layer is, in one line. */
182
+ export async function status(options) {
183
+ const want = requested(options);
184
+ if (!want) return { planner: 'none', detail: 'not requested (SIMFRAME_PLANNER is unset)' };
185
+ if (want !== 'apple') return { planner: 'none', detail: `no such planner backend: "${want}"` };
186
+ const live = await open();
187
+ if (!live?.ok) return { planner: 'none', detail: live?.reason ?? 'unavailable' };
188
+ return { planner: 'apple', detail: 'Apple Foundation Models, on-device, ranking only' };
189
+ }
190
+
191
+ /** Let a process exit without waiting on the helper. */
192
+ export function close() {
193
+ try { session?.child?.kill(); } catch { /* already gone */ }
194
+ session = null;
195
+ }
@@ -183,7 +183,8 @@ async function resolveDevice(query, opts) {
183
183
  new Error(
184
184
  `${booted.length} emulators are running and none was named: ` +
185
185
  `${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
186
- 'or set SIMFRAME_DEVICE to pick a default for this shell',
186
+ 'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
187
+ 'pass "device" once on any call and the rest of the session remembers it',
187
188
  ),
188
189
  { ambiguous: true },
189
190
  );
@@ -827,7 +828,7 @@ async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst =
827
828
  async function terminateApp(udid, bundleId) {
828
829
  // `am` reports failure on stdout and still exits 0 — the same trap launchApp
829
830
  // and openUrl already check for. Without this, terminating a package that is
830
- // not installed answered "terminated com.typo.app".
831
+ // not installed answered "terminated com.example.mistyped".
831
832
  const { stdout, stderr } = await adb(udid, ['shell', 'am', 'force-stop', bundleId]);
832
833
  const error = /^Error:.*$/m.exec(`${stdout}${stderr}`);
833
834
  if (error) {
@@ -87,7 +87,8 @@ async function resolveDevice(query, opts) {
87
87
  new Error(
88
88
  `${booted.length} simulators are booted and none was named: ` +
89
89
  `${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
90
- 'or set SIMFRAME_DEVICE to pick a default for this shell',
90
+ 'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
91
+ 'pass "device" once on any call and the rest of the session remembers it',
91
92
  ),
92
93
  { ambiguous: true },
93
94
  );
package/src/png.js CHANGED
@@ -178,6 +178,32 @@ export function grayGrid(bmp, cols, rows) {
178
178
  }
179
179
 
180
180
  /** Nearest-neighbour scale. Only used for contact sheets, where speed beats quality. */
181
+ /**
182
+ * A rectangle out of a bitmap, clamped to it.
183
+ *
184
+ * Exists because a whole screen at 1024px on the long edge cannot answer a
185
+ * question about one control. Reported from a real session: a selected filter
186
+ * chip and an unselected one are indistinguishable at that size, and selection
187
+ * state was the entire question the ticket turned on — so the agent shelled out
188
+ * to `simctl io` and PIL to crop and upscale the chip row, **for every single
189
+ * check**. Their estimate: six round trips.
190
+ *
191
+ * Coordinates are pixels; the caller converts from points, because only the
192
+ * caller knows the density it read them at.
193
+ */
194
+ export function cropBitmap(bmp, x, y, width, height) {
195
+ const left = Math.max(0, Math.min(bmp.width - 1, Math.round(x)));
196
+ const top = Math.max(0, Math.min(bmp.height - 1, Math.round(y)));
197
+ const w = Math.max(1, Math.min(bmp.width - left, Math.round(width)));
198
+ const h = Math.max(1, Math.min(bmp.height - top, Math.round(height)));
199
+ const out = Buffer.alloc(w * h * 4);
200
+ for (let row = 0; row < h; row += 1) {
201
+ const from = ((top + row) * bmp.width + left) * 4;
202
+ bmp.data.copy(out, row * w * 4, from, from + w * 4);
203
+ }
204
+ return { width: w, height: h, data: out };
205
+ }
206
+
181
207
  export function scaleBitmap(bmp, width, height) {
182
208
  const out = Buffer.allocUnsafe(width * height * 4);
183
209
  for (let y = 0; y < height; y++) {
package/src/refs.js CHANGED
@@ -104,17 +104,46 @@ export function parseSelector(query) {
104
104
  * numbering introduces that labels do not have, and a ref resolved against the
105
105
  * wrong screen taps whatever now happens to sit at those coordinates.
106
106
  */
107
- export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, tolerance = REF_TOLERANCE } = {}) {
107
+ export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, structuralDistance = 0, tolerance = REF_TOLERANCE } = {}) {
108
108
  const table = readRefs(udid);
109
109
  if (!table) throw new Error(`#${n} means nothing yet — read the screen first (sim_ui, or simframe ui)`);
110
- const stale = (was, now) =>
111
- `#${n} was numbered on a different screen (${was} → ${now}) — read the screen again before using refs`;
110
+ // The label this number was given to, when there is one. A stale ref is not
111
+ // nothing: the table records what it pointed at, which is enough for the
112
+ // caller to be offered the label instead of a bare refusal.
113
+ const labelFor = table.refs?.find((r) => r.ref === n)?.label ?? null;
114
+ // How long ago these numbers were handed out. Asked for by name: "refs
115
+ // expired (issued 4 calls ago) is actionable in a way this isn't".
116
+ const issued = Number.isFinite(table.at) ? ` refs were numbered ${Math.round((Date.now() - table.at) / 1000)}s ago;` : '';
117
+ // `staleKind` is the difference between "these numbers were drawn on a screen
118
+ // that has since shifted" and "you are somewhere else entirely", and only the
119
+ // first may be recovered by re-resolving the label the number stood for.
120
+ // Both wore the same flag once, and the caller re-resolved across an app
121
+ // switch: `#1` had been "Reminders" in Contacts, matched the status-bar
122
+ // back-to-app breadcrumb "• Reminders" at 0.64, and returned a tappable point
123
+ // in the status bar — a region the map itself refuses to offer. A refusal had
124
+ // become a confident wrong answer.
125
+ const staleError = (why, kind) => Object.assign(
126
+ new Error(`#${n} cannot be trusted here —${issued} ${why}. Read the screen again (sim_ui) to renumber`),
127
+ { staleRef: true, staleLabel: labelFor, staleKind: kind },
128
+ );
112
129
 
113
130
  // Structural identity first, because it is the question actually being asked:
114
131
  // is this the screen those numbers were assigned on? The caller gets it
115
132
  // cheaply — screen memory is a file read, not a perception pass.
116
- if (table.structuralHash && structuralHash && table.structuralHash !== structuralHash) {
117
- throw new Error(stale(table.structuralHash.slice(0, 8), structuralHash.slice(0, 8)));
133
+ //
134
+ // But only when the recall that produced it was exact. The identity arrives
135
+ // from `recallNearest`, which matches by layout within a tolerance so that a
136
+ // list with new rows stays one screen; above distance zero it is therefore a
137
+ // guess about *which* remembered screen this is, and a guess cannot be the
138
+ // sole reason to refuse. That mismatch was reported from the field as a
139
+ // refusal on unchanged state — the map had named the screen from the tolerant
140
+ // recall and printed the same header before and after, while this check read
141
+ // the same recall as exact and disagreed with it. Beyond distance zero the
142
+ // pixel backstop below is the one that decides, which is what it is for.
143
+ const exactRecall = structuralDistance === 0 || structuralDistance == null;
144
+ if (exactRecall && table.structuralHash && structuralHash && table.structuralHash !== structuralHash) {
145
+ throw staleError(`this is a different screen (${table.structuralHash.slice(0, 8)}`
146
+ + ` → ${structuralHash.slice(0, 8)})`, 'identity');
118
147
  }
119
148
  // Nothing recognises the screen we are on, so nothing can vouch for the
120
149
  // numbers. Refusing costs a re-read; guessing taps whatever is at those
@@ -128,9 +157,23 @@ export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, t
128
157
  // other — measured: refs numbered on the springboard resolved happily on a
129
158
  // different screen because both hashes were degenerate. A hash with almost
130
159
  // no bits set is not evidence of anything.
131
- if (layoutHash && table.layoutHash && informative(table.layoutHash) && informative(layoutHash)
132
- && hashDistance(table.layoutHash, layoutHash) > tolerance) {
133
- throw new Error(stale(table.layoutHash.slice(0, 8), layoutHash.slice(0, 8)));
160
+ // Reported three times in one session as `#4 was numbered on a different
161
+ // screen (03003714 → 03003714)` — a message that says the screen changed
162
+ // while showing that it did not, and left the reporter unable to tell a real
163
+ // move from a false positive. The cause was this branch printing eight
164
+ // characters of a **72-character** perceptual hash: its leading characters
165
+ // encode coarse structure, which is the very reason this comparison is a
166
+ // distance against a tolerance rather than an equality, so prefixes coincide
167
+ // routinely while the hashes differ. So this says the distance, and says that
168
+ // it is pixels rather than identity — a different thing from the branch above,
169
+ // which had been wearing the same sentence.
170
+ const drift = layoutHash && table.layoutHash && informative(table.layoutHash) && informative(layoutHash)
171
+ ? hashDistance(table.layoutHash, layoutHash)
172
+ : null;
173
+ if (drift != null && drift > tolerance) {
174
+ throw staleError(`the screen has moved too far from where these refs were numbered`
175
+ + ` (layout distance ${drift}, tolerance ${tolerance}) — the identity may be unchanged;`
176
+ + ' this is a pixel measurement, not a different screen', 'drift');
134
177
  }
135
178
  const hit = table.refs.find((r) => r.ref === n);
136
179
  if (!hit) {