simframe 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +181 -2
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
  9. package/native/simframed/Sources/simframed/main.swift +33 -3
  10. package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
  11. package/native/supervise.swift +216 -0
  12. package/package.json +4 -1
  13. package/scripts/check-package.mjs +22 -2
  14. package/scripts/check-private.mjs +9 -0
  15. package/scripts/ci-integration-local.sh +79 -0
  16. package/scripts/ci-memory.mjs +104 -20
  17. package/scripts/collect-rulings.mjs +312 -0
  18. package/scripts/eval-fingerprint.mjs +100 -23
  19. package/scripts/eval-perception.mjs +33 -0
  20. package/scripts/phase17-corpus.mjs +176 -0
  21. package/scripts/probe-network.mjs +118 -0
  22. package/scripts/soak-capture.mjs +72 -0
  23. package/skills/simframe/SKILL.md +257 -7
  24. package/src/actions.js +1791 -44
  25. package/src/cli.js +216 -14
  26. package/src/control.js +1 -0
  27. package/src/fingerprint.js +43 -1
  28. package/src/graph.js +136 -7
  29. package/src/index.js +252 -14
  30. package/src/input.js +66 -3
  31. package/src/localhelper.js +161 -0
  32. package/src/matching.js +64 -1
  33. package/src/mcp.js +333 -32
  34. package/src/metrics.js +148 -3
  35. package/src/ocr.js +18 -1
  36. package/src/planner.js +195 -0
  37. package/src/platform/android.js +25 -1
  38. package/src/platform/index.js +11 -1
  39. package/src/platform/ios.js +25 -1
  40. package/src/png.js +26 -0
  41. package/src/refs.js +51 -8
  42. package/src/regions.js +215 -1
  43. package/src/screenmap.js +89 -9
  44. package/src/supervisor.js +161 -0
  45. package/src/view.js +375 -11
  46. package/src/vocabulary.js +134 -0
  47. package/src/wrote.js +136 -0
package/src/metrics.js CHANGED
@@ -24,6 +24,21 @@ export const REASONS = [
24
24
 
25
25
  export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
26
26
 
27
+ /**
28
+ * What the executor observed after a supervisor ruling — the field that makes a
29
+ * ruling scoreable rather than merely recorded.
30
+ *
31
+ * A ruling on its own says what the supervisor thought. Three items on the
32
+ * deferred list (101's p95 replay, 106's cost of the steps a `stop` skipped,
33
+ * 96's response variable) need what happened *next*, and until now nothing
34
+ * persisted a ruling at all: they went into a `supervisions` array on the
35
+ * result and died with the process. Three rulings had ever existed anywhere.
36
+ *
37
+ * Closed, and mandatory, for the same reason `REASONS` is: a vocabulary that
38
+ * admits "other" collects a pile of "other".
39
+ */
40
+ export const RULING_OUTCOMES = ['recovered', 'still_failed', 'stopped', 'no_ruling'];
41
+
27
42
  /**
28
43
  * Which faculty would have removed this escalation.
29
44
  *
@@ -57,6 +72,7 @@ function metricPaths(udid) {
57
72
  dir,
58
73
  escalations: path.join(dir, 'escalations.jsonl'),
59
74
  flows: path.join(dir, 'flows.jsonl'),
75
+ supervisions: path.join(dir, 'supervisions.jsonl'),
60
76
  baselines: path.join(dir, 'baselines'),
61
77
  };
62
78
  }
@@ -109,6 +125,106 @@ export function readJsonl(file, { limit } = {}) {
109
125
 
110
126
  export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
111
127
  export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
128
+ export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supervisions, opts);
129
+
130
+ /**
131
+ * Write down one supervisor ruling and what came of it.
132
+ *
133
+ * The fields are chosen so the three items waiting on this can be answered
134
+ * **offline, from the log**, rather than by another device run:
135
+ *
136
+ * - `screen`/`edge` — which `(screen_hash, action)` the ruling was about, so
137
+ * rulings can be grouped by edge the way the graph groups timings.
138
+ * - `p95`/`samples` — what the graph knew about that edge **at the moment of
139
+ * the ruling**. Recording it now rather than looking it up later is the
140
+ * difference between 101 being arithmetic and 101 being archaeology: the
141
+ * graph keeps learning, so a p95 read next week is not the p95 the
142
+ * supervisor was implicitly competing with.
143
+ * - `stillMs` — the same input the supervisor got, so a replay sees what it saw.
144
+ * - `expect` — the plan's own note, because a ruling made with a briefing and
145
+ * one made without are not the same measurement (96's critical arm).
146
+ * - `outcome` — what the executor observed afterwards. Mandatory.
147
+ *
148
+ * `from` separates a deterministic rule from a model answer. Both are rulings
149
+ * and both have outcomes, but a comparison that mixed them would credit the
150
+ * model for what a two-line rule decided.
151
+ */
152
+ export function recordSupervision(udid, {
153
+ session, index, step, edge, screen, decision, from, reason, ms,
154
+ stillMs, p95, samples, expect, failure, outcome,
155
+ }) {
156
+ // Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
157
+ // difference is deliberate: this is called from inside a flow's failure
158
+ // handler, where a throw would turn a recoverable step failure into a crash.
159
+ // It still lands in `lastWriteError`, which `simframe escalations` prints.
160
+ if (!RULING_OUTCOMES.includes(outcome)) {
161
+ lastWriteError = `supervision outcome must be one of ${RULING_OUTCOMES.join('/')}, got ${JSON.stringify(outcome)}`;
162
+ return false;
163
+ }
164
+ return appendJsonl(metricPaths(udid).supervisions, {
165
+ timestamp: new Date().toISOString(),
166
+ session_id: session ?? SESSION_ID,
167
+ client: CLIENT,
168
+ step_index: index ?? null,
169
+ step: step ?? null,
170
+ edge: edge ?? null,
171
+ screen_fingerprint: screen ?? null,
172
+ decision,
173
+ from: from ?? 'model',
174
+ // Recorded, never presented as the ground for what happened: the supervisor
175
+ // has returned a correct decision with a reason citing a rule that did not
176
+ // apply. Keeping it is how that stays measurable instead of anecdotal.
177
+ reason: reason ? String(reason).slice(0, 200) : null,
178
+ latency_ms: Number.isFinite(ms) ? Math.round(ms) : null,
179
+ still_ms: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
180
+ edge_p95_ms: Number.isFinite(p95) ? Math.round(p95) : null,
181
+ edge_samples: Number.isFinite(samples) ? samples : null,
182
+ expect: expect ? String(expect).slice(0, 200) : null,
183
+ failure: failure ? String(failure).slice(0, 300) : null,
184
+ outcome,
185
+ });
186
+ }
187
+
188
+ /**
189
+ * What the ruling log currently says, and whether it can yet answer 101.
190
+ *
191
+ * Deliberately counts rather than scores. Item 101 asks which `wait`/`retry`
192
+ * rulings a p95-per-edge lookup would have got right, and that is a separate
193
+ * piece of work; what this answers is the question that comes first and was
194
+ * embarrassing to get wrong once already — **is there a population to measure
195
+ * at all, and do its rows carry the fields the measurement needs.** `p95_known`
196
+ * is that readiness check: a ruling recorded on an edge the graph had never
197
+ * timed cannot take part in the comparison, however many of them there are.
198
+ */
199
+ export function supervisionBreakdown(records) {
200
+ const by = (key) => {
201
+ const out = {};
202
+ for (const r of records) {
203
+ const k = r[key] ?? 'unknown';
204
+ out[k] = (out[k] ?? 0) + 1;
205
+ }
206
+ return out;
207
+ };
208
+ const matrix = {};
209
+ for (const r of records) {
210
+ const k = `${r.decision ?? 'unknown'} -> ${r.outcome ?? 'unknown'}`;
211
+ matrix[k] = (matrix[k] ?? 0) + 1;
212
+ }
213
+ const timed = records.filter((r) => Number.isFinite(r.edge_p95_ms));
214
+ const latencies = records.map((r) => r.latency_ms).filter((n) => Number.isFinite(n)).sort((a, b) => a - b);
215
+ return {
216
+ total: records.length,
217
+ by_decision: by('decision'),
218
+ by_outcome: by('outcome'),
219
+ by_from: by('from'),
220
+ decision_to_outcome: matrix,
221
+ // Readiness for 101, not a result for it.
222
+ p95_known: timed.length,
223
+ p95_unknown: records.length - timed.length,
224
+ sessions: [...new Set(records.map((r) => r.session_id).filter(Boolean))],
225
+ median_latency_ms: latencies.length ? latencies[latencies.length >> 1] : null,
226
+ };
227
+ }
112
228
 
113
229
  /**
114
230
  * Mark an error as an escalation with a reason, at the site that knows why.
@@ -119,7 +235,7 @@ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts
119
235
  * matching error strings at the boundary — a regexed message is a reason that
120
236
  * silently becomes "unknown" the day somebody rewords it.
121
237
  */
122
- export function tag(err, reason, { candidates = [], tried = [], ambiguous = false } = {}) {
238
+ export function tag(err, reason, { candidates = [], tried = [], ambiguous = false, intent = null } = {}) {
123
239
  if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
124
240
  // `ambiguous` is narrower than the reason, and that is the point. Two very
125
241
  // different failures both tag `ambiguous_intent`: the target is on screen
@@ -127,7 +243,19 @@ export function tag(err, reason, { candidates = [], tried = [], ambiguous = fals
127
243
  // thought we knew. Only the first is resolvable by *choosing*, and only the
128
244
  // first tells a waiting caller that waiting is pointless — the thing it is
129
245
  // waiting for has already arrived.
130
- err.escalation = { reason, candidates, tried, ambiguous };
246
+ // `intent` is the goal in the caller's own words, recorded as a field rather
247
+ // than left in the prose of `detail`.
248
+ //
249
+ // Phase 17's go/no-go asks whether an on-device model would pick the element
250
+ // Claude picked, given the goal and the element list. The element list is
251
+ // here as `candidates` and the eventual choice is recoverable from the
252
+ // graph — the tap that finally worked on this screen becomes a verified edge
253
+ // carrying its own step. The goal was the missing third, and it was sitting
254
+ // inside a sentence: `"X" matches 3 things on this screen — say which…`.
255
+ // Regexing it back out at export time is the exact habit this file exists to
256
+ // avoid, and it would silently return nothing the day that sentence is
257
+ // reworded.
258
+ err.escalation = { reason, candidates, tried, ambiguous, intent };
131
259
  return err;
132
260
  }
133
261
 
@@ -252,7 +380,20 @@ export function fingerprintNow(udid, screenmap) {
252
380
  * file is committed to a public repo in summary form, and the question it has
253
381
  * to answer is "was this all one agent", which needs no identity to answer.
254
382
  */
255
- const SESSION_ID = `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
383
+ const SESSION_ID = process.env.SIMFRAME_SESSION
384
+ ? String(process.env.SIMFRAME_SESSION).slice(0, 64)
385
+ : `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
386
+
387
+ /*
388
+ * Minting the id from the pid was right for the MCP server, which is one
389
+ * long-lived process, and wrong for everything else. A CLI-driven agent starts
390
+ * a process per command, so it got one "session" per command: on the benchmark
391
+ * device, 33 session ids for 46 records, 30 of them holding a single record.
392
+ * `escalations --session` was therefore unable to answer the one question it
393
+ * exists for, and Phase 17's go/no-go step 1 — "filter to one session id" —
394
+ * had nothing to filter. `SIMFRAME_SESSION` lets a caller that knows it is one
395
+ * session say so; the per-process id stays the default.
396
+ */
256
397
 
257
398
  /** How this process is being used, for reading a breakdown afterwards. */
258
399
  function clientKind() {
@@ -271,6 +412,7 @@ export const clientName = () => CLIENT;
271
412
  export function recordEscalation(udid, {
272
413
  flowId = null,
273
414
  flowName = null,
415
+ intent = null,
274
416
  stepIndex = null,
275
417
  fingerprint = null,
276
418
  reason,
@@ -292,6 +434,9 @@ export function recordEscalation(udid, {
292
434
  client: CLIENT,
293
435
  flow_id: flowId,
294
436
  flow_name: flowName,
437
+ // What was asked for, in the caller's words. Ground truth for Phase 17's
438
+ // go/no-go, and on its own it answers "what kind of decision is costing us".
439
+ intent: intent ? String(intent).slice(0, 120) : null,
295
440
  step_index: stepIndex,
296
441
  screen_fingerprint: fingerprint,
297
442
  reason,
package/src/ocr.js CHANGED
@@ -19,6 +19,17 @@ const BIN = path.join(BIN_DIR, 'ocr');
19
19
 
20
20
  let ready = null;
21
21
 
22
+ /**
23
+ * Which recognition level to ask Vision for.
24
+ *
25
+ * `accurate` is the default and what CLAUDE.md fixes; `fast` is the other thing
26
+ * Vision offers. Exposed so the pair can be scored against each other instead
27
+ * of one of them being a constant nobody measured.
28
+ */
29
+ export function level() {
30
+ return String(process.env.SIMFRAME_OCR ?? '').toLowerCase() === 'fast' ? 'fast' : 'accurate';
31
+ }
32
+
22
33
  /** Compile once, then reuse. Recompiles only if the source is newer than the binary. */
23
34
  export async function ensureBinary() {
24
35
  if (ready) return ready;
@@ -55,7 +66,13 @@ export async function ensureBinary() {
55
66
  export async function readText(pngFile, { density = 3 } = {}) {
56
67
  const built = await ensureBinary();
57
68
  if (!built.available) throw new Error(built.reason);
58
- const { stdout } = await run(built.binary, [pngFile], { timeout: 30_000, maxBuffer: 16 << 20 });
69
+ // The recognition level rides in the environment rather than in argv, so the
70
+ // Swift side keeps its one-argument contract and an older binary still works.
71
+ const { stdout } = await run(built.binary, [pngFile], {
72
+ timeout: 30_000,
73
+ maxBuffer: 16 << 20,
74
+ env: { ...process.env, SIMFRAME_OCR: level() },
75
+ });
59
76
  const raw = JSON.parse(stdout || '[]');
60
77
  return raw.map((r) => ({
61
78
  text: r.text,
package/src/planner.js ADDED
@@ -0,0 +1,195 @@
1
+ /**
2
+ * The local planner tier: a ranker, behind a flag, that may only reorder.
3
+ *
4
+ * **Why this exists at all, given Phase 17 was a no-go.** That phase asked a
5
+ * local model to *choose the next element*, and the answer was that the matcher
6
+ * already does — 37 of 40 real decisions. This is the complement and the one
7
+ * case a string matcher structurally cannot do: the goal matches **nothing** on
8
+ * screen, and something has to guess which container leads to it. "Change my
9
+ * username" shares no prefix, synonym or typo distance with "Account".
10
+ *
11
+ * Measured on this machine, six hand-written cases: 5 of 6 top-1, 6 of 6 top-3,
12
+ * median 564 ms warm. See `docs/BENCHMARKS.md`. Against a model round trip at
13
+ * 10–16 s that is roughly twenty times cheaper; against the honest baseline —
14
+ * breadth-first ordering, which needs no model — it won five of six.
15
+ *
16
+ * **What it is allowed to do, and it is deliberately almost nothing.** It
17
+ * reorders a list of candidates the caller has already permitted and will try
18
+ * in some order regardless. It cannot invent a label, cannot choose an action,
19
+ * cannot see pixels, and never runs on a destructive label because the caller
20
+ * filtered those out before asking (`src/vocabulary.js`). If it is wrong the
21
+ * exploration budget simply tries the next one. That is strictly weaker
22
+ * authority than Phase 17 proposed, which is what makes it safe to try.
23
+ *
24
+ * **Off unless asked.** `SIMFRAME_PLANNER=apple` turns it on; anything else,
25
+ * or any failure at all, degrades to `null` and the caller keeps its own order.
26
+ * `doctor` reports which. CI runs with it off.
27
+ */
28
+ import { spawn, execFile } from 'node:child_process';
29
+ import fs from 'node:fs';
30
+ import path from 'node:path';
31
+ import { fileURLToPath } from 'node:url';
32
+ import { promisify } from 'node:util';
33
+ import * as store from './store.js';
34
+
35
+ const run = promisify(execFile);
36
+ const SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'native', 'rank.swift');
37
+ const BIN = path.join(store.ROOT, 'bin', 'rank');
38
+
39
+ /** Which backend the caller asked for. Absent means no local planner. */
40
+ export function requested(options) {
41
+ // Per call first, then the environment, for the reason in `sensorMode`: an
42
+ // MCP server's environment is fixed when it spawns, so a tester could not
43
+ // switch backends inside one session and a round came back with one arm of
44
+ // its A/B unrun.
45
+ const raw = String(options?.planner ?? process.env.SIMFRAME_PLANNER ?? '').trim().toLowerCase();
46
+ if (!raw || raw === 'none' || raw === 'off' || raw === '0' || raw === 'false') return null;
47
+ return raw;
48
+ }
49
+
50
+ let building = null;
51
+
52
+ export async function ensureBinary() {
53
+ if (building) return building;
54
+ building = (async () => {
55
+ try {
56
+ const src = fs.statSync(SOURCE).mtimeMs;
57
+ const bin = fs.existsSync(BIN) ? fs.statSync(BIN).mtimeMs : 0;
58
+ if (bin > src) return { available: true, binary: BIN };
59
+ } catch {
60
+ return { available: false, reason: 'the ranker source is missing from this install' };
61
+ }
62
+ try {
63
+ fs.mkdirSync(path.dirname(BIN), { recursive: true });
64
+ // `swiftc`, not `xcrun swiftc`: nothing above the platform boundary may
65
+ // name a platform tool, and the boundary test catches it. `src/ocr.js`
66
+ // set this precedent — a compiler is not a device tool.
67
+ await run('swiftc', ['-O', SOURCE, '-o', BIN], { timeout: 180_000 });
68
+ return { available: true, binary: BIN };
69
+ } catch (err) {
70
+ building = null; // let a later call retry once a toolchain is present
71
+ return {
72
+ available: false,
73
+ reason: err.code === 'ENOENT'
74
+ ? 'swiftc is not installed, so the local planner cannot be built (install Xcode command line tools)'
75
+ : `could not build the local planner: ${String(err.message).split('\n')[0]}`,
76
+ };
77
+ }
78
+ })();
79
+ return building;
80
+ }
81
+
82
+ let session = null;
83
+
84
+ /** Start the helper once and keep it, because the first answer pays model load. */
85
+ async function open() {
86
+ if (session) return session;
87
+ const built = await ensureBinary();
88
+ if (!built.available) return { ok: false, reason: built.reason };
89
+ session = await new Promise((resolve) => {
90
+ const child = spawn(built.binary, [], { stdio: ['pipe', 'pipe', 'ignore'] });
91
+ // Deliberately NOT unref'd. Unreffing the child's stdout unreferences the
92
+ // very pipe every request waits on, so the process exited silently in the
93
+ // middle of an await — a flow that printed nothing and returned 0. The
94
+ // helper is closed explicitly instead, by whoever opened it.
95
+ let buffer = '';
96
+ const waiters = [];
97
+ let settled = false;
98
+ const fail = (reason) => {
99
+ if (!settled) { settled = true; resolve({ ok: false, reason }); }
100
+ while (waiters.length) waiters.shift()(null);
101
+ };
102
+ child.on('error', (err) => fail(`the local planner would not start: ${err.message}`));
103
+ child.on('exit', () => { session = null; fail('the local planner exited'); });
104
+ child.stdout.on('data', (chunk) => {
105
+ buffer += chunk;
106
+ let i = buffer.indexOf('\n');
107
+ while (i >= 0) {
108
+ const line = buffer.slice(0, i).trim();
109
+ buffer = buffer.slice(i + 1);
110
+ i = buffer.indexOf('\n');
111
+ if (!line) continue;
112
+ let msg;
113
+ try { msg = JSON.parse(line); } catch { continue; }
114
+ if (!settled) {
115
+ settled = true;
116
+ if (msg.ready) resolve({ ok: true, child, waiters });
117
+ else resolve({ ok: false, reason: msg.unavailable ?? 'the local planner did not become ready' });
118
+ continue;
119
+ }
120
+ const next = waiters.shift();
121
+ if (next) next(msg);
122
+ }
123
+ });
124
+ });
125
+ return session;
126
+ }
127
+
128
+ /**
129
+ * Reorder `options` by which is likeliest to lead to `goal`.
130
+ *
131
+ * @returns {Promise<string[]|null>} the caller's own order is correct when this
132
+ * is null, which is every failure mode: flag off, no model, a timeout, a
133
+ * parse problem, a paraphrasing answer. Never throws.
134
+ */
135
+ export async function rank(goal, options, { timeoutMs = 3000, deviceOptions } = {}) {
136
+ if (!requested(deviceOptions)) return null;
137
+ if (!goal || !Array.isArray(options) || options.length < 2) return null;
138
+ let live;
139
+ try {
140
+ live = await open();
141
+ } catch {
142
+ return null;
143
+ }
144
+ if (!live?.ok) return null;
145
+ const answer = await new Promise((resolve) => {
146
+ // A timed-out waiter has to be *retired*, not merely resolved. Leaving it in
147
+ // the queue meant the next answer went to it instead of to the next asker,
148
+ // and every call after that was off by one — which showed up as an
149
+ // exploration run that never finished rather than as an error.
150
+ let done = false;
151
+ const waiter = (msg) => {
152
+ if (done) return;
153
+ done = true;
154
+ clearTimeout(timer);
155
+ resolve(msg);
156
+ };
157
+ const timer = setTimeout(() => {
158
+ if (done) return;
159
+ done = true;
160
+ const i = live.waiters.indexOf(waiter);
161
+ if (i >= 0) live.waiters.splice(i, 1);
162
+ resolve(null);
163
+ }, timeoutMs);
164
+ live.waiters.push(waiter);
165
+ try {
166
+ live.child.stdin.write(`${JSON.stringify({ goal: String(goal), options })}\n`);
167
+ } catch {
168
+ clearTimeout(timer);
169
+ resolve(null);
170
+ }
171
+ });
172
+ if (!answer?.order?.length) return null;
173
+ // It often returns a subset, so its order comes first and ours fills the tail.
174
+ // Trusting it to be exhaustive would silently drop candidates the budget was
175
+ // going to try.
176
+ const ranked = answer.order.filter((label) => options.includes(label));
177
+ const seen = new Set(ranked);
178
+ return [...ranked, ...options.filter((o) => !seen.has(o))];
179
+ }
180
+
181
+ /** For `doctor`: what the planner layer is, in one line. */
182
+ export async function status(options) {
183
+ const want = requested(options);
184
+ if (!want) return { planner: 'none', detail: 'not requested (SIMFRAME_PLANNER is unset)' };
185
+ if (want !== 'apple') return { planner: 'none', detail: `no such planner backend: "${want}"` };
186
+ const live = await open();
187
+ if (!live?.ok) return { planner: 'none', detail: live?.reason ?? 'unavailable' };
188
+ return { planner: 'apple', detail: 'Apple Foundation Models, on-device, ranking only' };
189
+ }
190
+
191
+ /** Let a process exit without waiting on the helper. */
192
+ export function close() {
193
+ try { session?.child?.kill(); } catch { /* already gone */ }
194
+ session = null;
195
+ }
@@ -183,7 +183,8 @@ async function resolveDevice(query, opts) {
183
183
  new Error(
184
184
  `${booted.length} emulators are running and none was named: ` +
185
185
  `${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
186
- 'or set SIMFRAME_DEVICE to pick a default for this shell',
186
+ 'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
187
+ 'pass "device" once on any call and the rest of the session remembers it',
187
188
  ),
188
189
  { ambiguous: true },
189
190
  );
@@ -995,6 +996,28 @@ function capabilities() {
995
996
  };
996
997
  }
997
998
 
999
+ /**
1000
+ * Not implemented here, and it says so in its own vocabulary.
1001
+ *
1002
+ * `revive` exists for a display that has stopped rendering — a CoreSimulator
1003
+ * pathology. An emulator's failure modes are its own (the console port going
1004
+ * away, adb losing the device) and the remedy is not the same sequence, so
1005
+ * borrowing the other platform's answer would be the mistake `doctor` made when
1006
+ * it reported "input driver: idb" for an emulator: a claim about a tool that has
1007
+ * never spoken to an Android device.
1008
+ *
1009
+ * When an emulator wedge is characterised rather than guessed at, this becomes
1010
+ * a real implementation. Until then the honest answer is that this backend does
1011
+ * not have the layer.
1012
+ */
1013
+ async function restartDevice(serial) {
1014
+ throw new Error(
1015
+ `simframe cannot yet restart an emulator (${serial}) — that remedy is written for a`
1016
+ + ' simulator display that stopped rendering, and an emulator fails differently.'
1017
+ + ' Restart it from Android Studio, or `adb -s <serial> emu kill` and relaunch.',
1018
+ );
1019
+ }
1020
+
998
1021
  /** @type {import('./index.js').Platform} */
999
1022
  export const platform = {
1000
1023
  id: 'android',
@@ -1010,6 +1033,7 @@ export const platform = {
1010
1033
  launchApp,
1011
1034
  terminateApp,
1012
1035
  openUrl,
1036
+ restartDevice,
1013
1037
  setPermission,
1014
1038
  setPasteboard,
1015
1039
  getPasteboard,
@@ -61,7 +61,7 @@ export const PLATFORM_SURFACE = Object.freeze([
61
61
  'id', 'deviceNoun',
62
62
  'listDevices', 'bootedDevices', 'resolveDevice', 'isBootedSync', 'ownsUdid',
63
63
  'geometry', 'inputDriver',
64
- 'screenshot', 'launchApp', 'terminateApp', 'openUrl',
64
+ 'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
66
  'bootedAt',
67
67
  ]);
@@ -260,6 +260,16 @@ export const inputDriverFor = (udid) => platformFor(udid).inputDriver(udid);
260
260
  */
261
261
  export const capabilitiesFor = (udid) => platformFor(udid).capabilities();
262
262
 
263
+ /**
264
+ * Power-cycle a device, on the backend that owns it.
265
+ *
266
+ * Reached only from `simframe revive`, never from the capture loop: the loop
267
+ * detects a stalled display and reports it, and restarting is the operator's
268
+ * call. A backend that does not have this remedy throws in its own terms rather
269
+ * than borrowing the other's.
270
+ */
271
+ export const restartDevice = (udid) => platformFor(udid).restartDevice(udid);
272
+
263
273
  /** What `simframe doctor` should check: each registered backend's own toolchain. */
264
274
  export function toolchainChecks() {
265
275
  return backends().flatMap((backend) => backend.toolchain());
@@ -87,7 +87,8 @@ async function resolveDevice(query, opts) {
87
87
  new Error(
88
88
  `${booted.length} simulators are booted and none was named: ` +
89
89
  `${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
90
- 'or set SIMFRAME_DEVICE to pick a default for this shell',
90
+ 'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
91
+ 'pass "device" once on any call and the rest of the session remembers it',
91
92
  ),
92
93
  { ambiguous: true },
93
94
  );
@@ -188,6 +189,28 @@ async function openUrl(udid, url) {
188
189
  await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
189
190
  }
190
191
 
192
+ /**
193
+ * Shut a device down and bring it back, waiting for the boot to finish.
194
+ *
195
+ * The remedy for a display that has stopped rendering, which the capture loop
196
+ * can detect and must not perform: it reports `stalled` and stops, because a
197
+ * capture loop that rebooted the device it was watching would be a tool
198
+ * reaching for the mains when a reading looks wrong. This is the operator's
199
+ * decision, reached by `simframe revive`.
200
+ *
201
+ * `bootstatus -b` and not `boot`, for the reason it is used in CI: `boot`
202
+ * returns before the device is usable, and everything downstream then races the
203
+ * boot. Timeboxed generously — a cold boot on a busy machine is slow, and a
204
+ * boot that never finishes should fail here with a reason rather than as a
205
+ * puzzle further down.
206
+ */
207
+ async function restartDevice(udid) {
208
+ // Tolerated: a device that is already off cannot be shut down, and that is
209
+ // the state this command is most often reached from.
210
+ await run('xcrun', ['simctl', 'shutdown', udid], { timeout: 60_000 }).catch(() => null);
211
+ await run('xcrun', ['simctl', 'bootstatus', udid, '-b'], { timeout: 240_000 });
212
+ }
213
+
191
214
  const PERMISSION_SERVICES = [
192
215
  'all', 'calendar', 'contacts-limited', 'contacts', 'location', 'location-always',
193
216
  'photos-add', 'photos', 'media-library', 'microphone', 'motion', 'reminders', 'siri',
@@ -346,6 +369,7 @@ export const platform = {
346
369
  launchApp,
347
370
  terminateApp,
348
371
  openUrl,
372
+ restartDevice,
349
373
  setPermission,
350
374
  setPasteboard,
351
375
  permissionServices: () => PERMISSION_SERVICES,
package/src/png.js CHANGED
@@ -178,6 +178,32 @@ export function grayGrid(bmp, cols, rows) {
178
178
  }
179
179
 
180
180
  /** Nearest-neighbour scale. Only used for contact sheets, where speed beats quality. */
181
+ /**
182
+ * A rectangle out of a bitmap, clamped to it.
183
+ *
184
+ * Exists because a whole screen at 1024px on the long edge cannot answer a
185
+ * question about one control. Reported from a real session: a selected filter
186
+ * chip and an unselected one are indistinguishable at that size, and selection
187
+ * state was the entire question the ticket turned on — so the agent shelled out
188
+ * to `simctl io` and PIL to crop and upscale the chip row, **for every single
189
+ * check**. Their estimate: six round trips.
190
+ *
191
+ * Coordinates are pixels; the caller converts from points, because only the
192
+ * caller knows the density it read them at.
193
+ */
194
+ export function cropBitmap(bmp, x, y, width, height) {
195
+ const left = Math.max(0, Math.min(bmp.width - 1, Math.round(x)));
196
+ const top = Math.max(0, Math.min(bmp.height - 1, Math.round(y)));
197
+ const w = Math.max(1, Math.min(bmp.width - left, Math.round(width)));
198
+ const h = Math.max(1, Math.min(bmp.height - top, Math.round(height)));
199
+ const out = Buffer.alloc(w * h * 4);
200
+ for (let row = 0; row < h; row += 1) {
201
+ const from = ((top + row) * bmp.width + left) * 4;
202
+ bmp.data.copy(out, row * w * 4, from, from + w * 4);
203
+ }
204
+ return { width: w, height: h, data: out };
205
+ }
206
+
181
207
  export function scaleBitmap(bmp, width, height) {
182
208
  const out = Buffer.allocUnsafe(width * height * 4);
183
209
  for (let y = 0; y < height; y++) {
package/src/refs.js CHANGED
@@ -104,17 +104,46 @@ export function parseSelector(query) {
104
104
  * numbering introduces that labels do not have, and a ref resolved against the
105
105
  * wrong screen taps whatever now happens to sit at those coordinates.
106
106
  */
107
- export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, tolerance = REF_TOLERANCE } = {}) {
107
+ export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, structuralDistance = 0, tolerance = REF_TOLERANCE } = {}) {
108
108
  const table = readRefs(udid);
109
109
  if (!table) throw new Error(`#${n} means nothing yet — read the screen first (sim_ui, or simframe ui)`);
110
- const stale = (was, now) =>
111
- `#${n} was numbered on a different screen (${was} → ${now}) — read the screen again before using refs`;
110
+ // The label this number was given to, when there is one. A stale ref is not
111
+ // nothing: the table records what it pointed at, which is enough for the
112
+ // caller to be offered the label instead of a bare refusal.
113
+ const labelFor = table.refs?.find((r) => r.ref === n)?.label ?? null;
114
+ // How long ago these numbers were handed out. Asked for by name: "refs
115
+ // expired (issued 4 calls ago) is actionable in a way this isn't".
116
+ const issued = Number.isFinite(table.at) ? ` refs were numbered ${Math.round((Date.now() - table.at) / 1000)}s ago;` : '';
117
+ // `staleKind` is the difference between "these numbers were drawn on a screen
118
+ // that has since shifted" and "you are somewhere else entirely", and only the
119
+ // first may be recovered by re-resolving the label the number stood for.
120
+ // Both wore the same flag once, and the caller re-resolved across an app
121
+ // switch: `#1` had been "Reminders" in Contacts, matched the status-bar
122
+ // back-to-app breadcrumb "• Reminders" at 0.64, and returned a tappable point
123
+ // in the status bar — a region the map itself refuses to offer. A refusal had
124
+ // become a confident wrong answer.
125
+ const staleError = (why, kind) => Object.assign(
126
+ new Error(`#${n} cannot be trusted here —${issued} ${why}. Read the screen again (sim_ui) to renumber`),
127
+ { staleRef: true, staleLabel: labelFor, staleKind: kind },
128
+ );
112
129
 
113
130
  // Structural identity first, because it is the question actually being asked:
114
131
  // is this the screen those numbers were assigned on? The caller gets it
115
132
  // cheaply — screen memory is a file read, not a perception pass.
116
- if (table.structuralHash && structuralHash && table.structuralHash !== structuralHash) {
117
- throw new Error(stale(table.structuralHash.slice(0, 8), structuralHash.slice(0, 8)));
133
+ //
134
+ // But only when the recall that produced it was exact. The identity arrives
135
+ // from `recallNearest`, which matches by layout within a tolerance so that a
136
+ // list with new rows stays one screen; above distance zero it is therefore a
137
+ // guess about *which* remembered screen this is, and a guess cannot be the
138
+ // sole reason to refuse. That mismatch was reported from the field as a
139
+ // refusal on unchanged state — the map had named the screen from the tolerant
140
+ // recall and printed the same header before and after, while this check read
141
+ // the same recall as exact and disagreed with it. Beyond distance zero the
142
+ // pixel backstop below is the one that decides, which is what it is for.
143
+ const exactRecall = structuralDistance === 0 || structuralDistance == null;
144
+ if (exactRecall && table.structuralHash && structuralHash && table.structuralHash !== structuralHash) {
145
+ throw staleError(`this is a different screen (${table.structuralHash.slice(0, 8)}`
146
+ + ` → ${structuralHash.slice(0, 8)})`, 'identity');
118
147
  }
119
148
  // Nothing recognises the screen we are on, so nothing can vouch for the
120
149
  // numbers. Refusing costs a re-read; guessing taps whatever is at those
@@ -128,9 +157,23 @@ export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, t
128
157
  // other — measured: refs numbered on the springboard resolved happily on a
129
158
  // different screen because both hashes were degenerate. A hash with almost
130
159
  // no bits set is not evidence of anything.
131
- if (layoutHash && table.layoutHash && informative(table.layoutHash) && informative(layoutHash)
132
- && hashDistance(table.layoutHash, layoutHash) > tolerance) {
133
- throw new Error(stale(table.layoutHash.slice(0, 8), layoutHash.slice(0, 8)));
160
+ // Reported three times in one session as `#4 was numbered on a different
161
+ // screen (03003714 → 03003714)` — a message that says the screen changed
162
+ // while showing that it did not, and left the reporter unable to tell a real
163
+ // move from a false positive. The cause was this branch printing eight
164
+ // characters of a **72-character** perceptual hash: its leading characters
165
+ // encode coarse structure, which is the very reason this comparison is a
166
+ // distance against a tolerance rather than an equality, so prefixes coincide
167
+ // routinely while the hashes differ. So this says the distance, and says that
168
+ // it is pixels rather than identity — a different thing from the branch above,
169
+ // which had been wearing the same sentence.
170
+ const drift = layoutHash && table.layoutHash && informative(table.layoutHash) && informative(layoutHash)
171
+ ? hashDistance(table.layoutHash, layoutHash)
172
+ : null;
173
+ if (drift != null && drift > tolerance) {
174
+ throw staleError(`the screen has moved too far from where these refs were numbered`
175
+ + ` (layout distance ${drift}, tolerance ${tolerance}) — the identity may be unchanged;`
176
+ + ' this is a pixel measurement, not a different screen', 'drift');
134
177
  }
135
178
  const hit = table.refs.find((r) => r.ref === n);
136
179
  if (!hit) {