simframe 0.17.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/cli.js CHANGED
@@ -3,7 +3,7 @@ import fs from 'node:fs';
3
3
  import os from 'node:os';
4
4
  import path from 'node:path';
5
5
  import { runDaemon, DEFAULTS } from './daemon.js';
6
- import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, restartDevice, screenshot, toolchainChecks } from './platform/index.js';
6
+ import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
7
7
  import * as actions from './actions.js';
8
8
  import * as analyze from './analyze.js';
9
9
  import * as api from './index.js';
@@ -11,11 +11,15 @@ import * as input from './input.js';
11
11
  import * as baseline from './baseline.js';
12
12
  import * as metrics from './metrics.js';
13
13
  import * as navigate from './navigate.js';
14
+ import * as wedge from './wedge.js';
15
+
14
16
  import { decodePng } from './png.js';
15
17
  import * as storage from './storage.js';
16
18
  import * as store from './store.js';
17
19
  import * as view from './view.js';
18
20
 
21
+
22
+
19
23
  const USAGE = `simframe — always-warm iOS Simulator frames
20
24
 
21
25
  simframe mcp run the MCP server on stdio (for agents)
@@ -51,6 +55,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
51
55
  simframe escalations [device] why simframe handed decisions back, by reason
52
56
  simframe supervisions [device] local supervisor rulings, and what came of each
53
57
  simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
58
+ simframe diagnose [device] what this device is doing right now, and which failure it is
54
59
  (--session=<id> narrows to one agent; the
55
60
  ids are listed in the output. SIMFRAME_SESSION
56
61
  names one, but only at process start — an
@@ -419,31 +424,67 @@ async function main() {
419
424
  case 'revive': {
420
425
  const dev = await resolveDevice(device);
421
426
  const say = (line) => { if (!flags.json) console.log(line); };
422
- const steps = [];
423
- const did = async (what, fn) => {
424
- try { await fn(); steps.push({ step: what, ok: true }); say(` ok ${what}`); } catch (err) {
425
- steps.push({ step: what, ok: false, error: err.message });
426
- say(` .. ${what} — ${err.message.split('\n')[0]}`);
427
- }
428
- };
429
427
  say(`reviving ${dev.name}`);
430
- // Forced: the point of this command is that the device is wedged, so
431
- // something is certainly still holding it.
432
- await did('stopped the daemon', async () => { api.stopDaemon(dev.udid, { force: true }); });
433
- // Through the boundary, which is the whole point of the boundary: the
434
- // first version of this shelled out to `xcrun` from here and the test
435
- // that forbids it failed immediately, correctly.
436
- await did('restarted the device, and waited for the boot to finish',
437
- () => restartDevice(dev.udid));
438
- await did('started capture', () => api.ensureDaemon(dev.udid));
439
- await did('rebuilt the HID session', () => input.resetSession(dev.udid));
440
- const health = await api.getState(dev.udid).then((s) => s?.state ?? null).catch(() => null);
441
- const alive = Boolean(health?.hash);
442
- emit(flags, { ok: alive, device: dev.udid, steps }, alive
443
- ? `\n${dev.name} is producing frames again`
444
- : `\n${dev.name} is still not producing frames. This is past what simframe can do —`
445
- + ' check Simulator.app is not showing an error, and see docs/DEFERRED.md item 95.');
446
- if (!alive) process.exitCode = 1;
428
+ // The sequence itself lives in `wedge.js`, because `bench-hpi` needs the
429
+ // same recovery between passes and a second copy of it here is how this
430
+ // project has repeatedly ended up fixing one symptom in three places.
431
+ const revived = await wedge.revive(dev.udid, {
432
+ options,
433
+ device: dev,
434
+ onStep: ({ step, ok, error }) => say(ok ? ` ok ${step}` : ` .. ${step} — ${String(error).split('\n')[0]}`),
435
+ });
436
+ const { steps } = revived;
437
+ const diag = { verdict: revived.verdict };
438
+ const state = diag?.verdict?.state ?? 'read-failed';
439
+ const usable = !wedge.UNUSABLE.has(state);
440
+ const degraded = usable && state !== 'healthy';
441
+ emit(flags, { ok: usable, device: dev.udid, steps, verdict: diag?.verdict ?? null }, usable
442
+ ? (degraded
443
+ ? `\n${dev.name} is back and usable, but ${state}:\n ${diag.verdict.detail}`
444
+ + '\nNot a failure — it taps and reads.'
445
+ : `\n${dev.name} is healthy again — ${diag.verdict.detail}`)
446
+ : `\n${dev.name} came back ${state}, which is not usable.`
447
+ + `\n ${diag?.verdict?.detail ?? 'nothing could be read from it'}`
448
+ + '\nA second revive sometimes clears it. If it does not, this is past what simframe'
449
+ + ' can do — check Simulator.app is not showing an error, and see docs/DEFERRED.md'
450
+ + ' items 95 and 183.');
451
+ if (!usable) process.exitCode = 1;
452
+ return;
453
+ }
454
+
455
+ // Not folded into `doctor`, which answers "can this machine capture". This
456
+ // answers "what is this device doing right now", which is item 173's
457
+ // question and has never had an instrument.
458
+ case 'diagnose': {
459
+ const dev = await resolveDevice(device);
460
+ const r = await wedge.diagnose(dev.udid, { options });
461
+ emit(flags, r, [
462
+ `${r.device.name} — ${r.verdict.state}`,
463
+ ` ${r.verdict.detail}`,
464
+ '',
465
+ ` frame seq ${r.frame?.seq ?? '-'}, ${r.frame?.ageMs ?? '-'}ms old, still for ${r.frame?.stableForMs ?? '-'}ms, ${r.frame?.size ?? '-'}`,
466
+ ` elements ${r.elements.total} total — ${r.elements.ax} by tree, ${r.elements.ocr} by OCR, ${r.elements.fused} by both`,
467
+ // **No band here, and that is the second version of this fix.**
468
+ //
469
+ // A reporter saw `healthy` printed beside "measured healthy 0.66-0.93"
470
+ // over a reading of 0.567 and asked, reasonably, which to believe. The
471
+ // first fix widened the band to 0.57-0.93 — and the very next device
472
+ // read returned **0.471** on an ordinary Settings screen, which is the
473
+ // same contradiction one decimal place down. Any fixed band will be
474
+ // contradicted by the next screen, because fusion tracks how much of a
475
+ // screen's content is OCR-only and that is a property of the app.
476
+ //
477
+ // So the line prints the number and the one threshold the verdict
478
+ // actually turns on. The observed range lives in `docs/BENCHMARKS.md`,
479
+ // where it is evidence about screens rather than a standard a device is
480
+ // being held to. `HEALTHY_FUSION` is still printed by the `stale-frame`
481
+ // verdict, where a reading of 0.03 genuinely wants the contrast.
482
+ ` fusion ${r.agreement ?? 'n/a'} of elements seen by both sensors`
483
+ + ` — stale at or below ${wedge.DISAGREEMENT}, which is what this verdict turns on`,
484
+ ` frontmost ${r.frontmost?.pid ?? 'unknown'}${r.frontmost?.title ? ` (${r.frontmost.title})` : ''}`,
485
+ ...(r.verdict.revive ? ['', ' `simframe revive` is the recovery. Keep this output — item 173 needs it.'] : []),
486
+ ]);
487
+ if (r.verdict.revive) process.exitCode = 1;
447
488
  return;
448
489
  }
449
490
 
@@ -1253,8 +1294,14 @@ async function main() {
1253
1294
  'flow runs agent p50 human p50 HPI_time step_ratio turns esc',
1254
1295
  ...report.flows.map((f) => metrics.flowRow(f, { wide: true })),
1255
1296
  '',
1256
- `HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} runs, ` +
1297
+ `HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} measurable run(s), ` +
1257
1298
  `${report.overall.runs - runs.filter((r) => r.completed && !r.wrong_action_taken).length} not clean)`,
1299
+ // A denominator that quietly shrinks is worse than one that is wrong.
1300
+ report.overall.runs_lost_to_device
1301
+ ? ` ${report.overall.runs_lost_to_device} further run(s) left the denominator because the DEVICE failed,`
1302
+ + ' not the code — those are unmeasured, not inaccurate (item 173):'
1303
+ + `\n${(report.overall.device_causes ?? []).map((c) => ` ${c}`).join('\n')}`
1304
+ : null,
1258
1305
  report.overall.hpi_time == null
1259
1306
  ? `HPI_time and HPI need a human baseline — none of ${report.overall.flows_measured} measured flow(s) has one yet.`
1260
1307
  : `HPI_time ${report.overall.hpi_time} (harmonic mean over ${report.overall.flows_with_human_baseline} flow(s)), HPI ${report.overall.hpi}`,
@@ -1265,7 +1312,10 @@ async function main() {
1265
1312
  `steps per model call ${report.overall.steps_per_call ?? '—'}`
1266
1313
  + (report.overall.steps_per_call
1267
1314
  ? ` — about ${(((20000 + 1700 * report.overall.steps_per_call) / report.overall.steps_per_call) / 1000).toFixed(1)}s`
1268
- + ' per step end to end, of which simframe is ~1.7s. Raise this, not the engine.'
1315
+ + ' per step end to end. Raise this, not the engine — but note the ~1.7s'
1316
+ + ' figure for simframe\'s own work is WARM taps inside a batch: a cold app'
1317
+ + ' launch is a large one-off on top, and it dominates short flows'
1318
+ + ' (measured: 6.2s/step over 2 steps with no model at all).'
1269
1319
  : ''),
1270
1320
  flags.out ? `wrote ${flags.out}` : null,
1271
1321
  ]);
@@ -1354,13 +1404,45 @@ async function main() {
1354
1404
  if (read > 0) {
1355
1405
  verdict = named + (read < n ? ` (on the ${read} of ${n} whose reason was read)` : '');
1356
1406
  } else if (assumed > 0) {
1357
- verdict = 'reason assumed, not read — no faculty can be named from these';
1407
+ // For `verification_failed` this line used to end the story, and
1408
+ // it is no longer the whole truth: the verdict split below names
1409
+ // a faculty for some of them and takes the device's own failures
1410
+ // out altogether.
1411
+ verdict = r === 'verification_failed' && (b.verdicts?.length || b.device_total)
1412
+ ? 'reason too coarse to steer by — see the verdict split below'
1413
+ : 'reason assumed, not read — no faculty can be named from these';
1358
1414
  } else {
1359
1415
  // Neither read nor assumed: the log predates the distinction.
1360
- verdict = `${named} — but these records predate the check, so treat it as untested`;
1416
+ verdict = `${named} — but these records predate the check, so treat it as untested`
1417
+ + ' (legacy, not assumed)';
1361
1418
  }
1362
1419
  return ` ${r.padEnd(20)} ${String(n).padStart(4)} ${verdict}`;
1363
1420
  }),
1421
+ // The two things the reason alone could not say.
1422
+ //
1423
+ // Both exist because the per-reason table above was the steering wheel
1424
+ // and it pointed at one faculty for a class holding three, and counted
1425
+ // the simulator's own failures towards a perception phase.
1426
+ b.verdicts?.length ? '' : null,
1427
+ b.verdicts?.length
1428
+ ? 'verification_failed, split by the verdict that fired'
1429
+ + (b.derived?.verdicts
1430
+ ? ` (${b.derived.verdicts} of ${b.verdicts.reduce((a, v) => a + v.count, 0)} derived from the detail text, not recorded at the time):`
1431
+ : ':')
1432
+ : null,
1433
+ ...(b.verdicts ?? []).slice(0, 8).map((v) => {
1434
+ const faculty = metrics.VERDICT_FACULTY[v.name];
1435
+ return ` ${v.name.padEnd(20)} ${String(v.count).padStart(4)} `
1436
+ + (faculty
1437
+ ? `points at: ${faculty}`
1438
+ : 'no faculty follows from this verdict alone — see metrics.VERDICT_FACULTY');
1439
+ }),
1440
+ b.device_total ? '' : null,
1441
+ b.device_total
1442
+ ? `${b.device_total} escalation(s) were the DEVICE, not the code — item 173, and not evidence for any faculty`
1443
+ + `${b.derived?.device ? ` (${b.derived.device} derived from the detail text)` : ''}:`
1444
+ : null,
1445
+ ...(b.device ?? []).slice(0, 6).map((d) => ` ${String(d.count).padStart(4)} ${d.name}`),
1364
1446
  b.total ? '' : null,
1365
1447
  b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
1366
1448
  // Said out loud rather than left for someone to discover: the rate is
@@ -0,0 +1,101 @@
1
+ // Is a failure the device's fault or the code's?
2
+ //
3
+ // Moved out of `scripts/` on 2026-09-17 so `src/metrics.js` can use it. The
4
+ // escalation log needed it: 78 of the `verification_failed` records on the
5
+ // bench device are `xcrun simctl openurl` failing, an app that would not
6
+ // launch, or capture stopping — the device, filed under a code faculty and
7
+ // reported as evidence for "sense of time (Phase 11)". A steering wheel that
8
+ // counts item 173's own occurrences as a perception problem points somewhere
9
+ // nobody chose.
10
+ //
11
+ // Deliberately dependency-free. `wedge.js` imports `index.js`, and `metrics.js`
12
+ // is imported *by* `index.js`, so a shared table living in either would close a
13
+ // cycle. This module imports nothing and is imported by both.
14
+
15
+ /** Conditions that are the simulator, not the code. Each seen in a real run. */
16
+ export const DEVICE_STATE = [
17
+ [/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
18
+ [/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
19
+ [/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
20
+ [/no frames buffered|capture is wedged/i, 'capture stopped'],
21
+ [/the second app never launched|could not be dispatched/i, 'an app would not launch'],
22
+ // A launched app that never comes to the front, seen as the tour waiting for
23
+ // one of its landmarks on a screen that is showing a clock and nothing else.
24
+ //
25
+ // Measured on a runner: `ok launch — launched com.apple.Preferences
26
+ // (relaunched)` followed by `waited 8000ms for General: "General" is not on
27
+ // this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
28
+ // Two labels, one of them a clock, on a still screen — the device is not
29
+ // presenting the app, and the guard called that a check failing on its
30
+ // merits and declined to revive.
31
+ //
32
+ // Deliberately narrow. It requires the wait to have failed AND the screen to
33
+ // have been still AND almost nothing readable: a tour that genuinely asks for
34
+ // the wrong label has a screen full of other labels, and must keep failing
35
+ // rather than being retried into a pass.
36
+ [
37
+ /never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
38
+ 'a launched app never came to the front (the screen shows a clock and nothing else)',
39
+ ],
40
+ // The same condition, now said outright by the step that suffered it instead
41
+ // of inferred from the shape of the screen afterwards. Item 169 gave `launch`
42
+ // a pid to compare, so a launch that starts a process and never fronts it
43
+ // reports itself; this signature fires on the cause rather than on a
44
+ // consequence that had to be recognised by "two labels, one a clock".
45
+ //
46
+ // It cannot be triggered by a tour asking for the wrong label — only a failed
47
+ // launch emits this sentence — so it needs none of the narrowing above.
48
+ [
49
+ /never came to the front within \d+ms/i,
50
+ 'a launched app never came to the front (the launch said so itself, by pid)',
51
+ ],
52
+ // simctl itself stopped answering, and said so in simframe's own words: the
53
+ // process was killed at the timeout rather than refusing the request.
54
+ //
55
+ // Three runs of the 2026-09-17 bench died this way — 95.6 s each — and every
56
+ // one of them was written to `flows.jsonl` with `device_cause: null`, so the
57
+ // number CI gates on counted a simulator that had stopped answering as the
58
+ // code getting things wrong. That is the same fault item 173 recorded as
59
+ // fixed, still open for this signature because nothing here matched it.
60
+ [
61
+ /did not return within \d+s \(killed by simframe/i,
62
+ 'simctl stopped answering and had to be killed at the timeout',
63
+ ],
64
+ // Seen on the v0.14.3 bench run: `could not launch com.apple.Preferences:
65
+ // The system shell (SpringBoard:36454) probably crashed.` The guest's window
66
+ // server going down is the device, not the check, and nothing here matched it.
67
+ [
68
+ /system shell \(SpringBoard[^)]*\) probably crashed/i,
69
+ "the guest's SpringBoard crashed, so nothing can be fronted",
70
+ ],
71
+ ];
72
+
73
+ /** The condition this output shows, or null when the check failed on its merits. */
74
+ export function deviceCause(text) {
75
+ return DEVICE_STATE.find(([re]) => re.test(String(text ?? '')))?.[1] ?? null;
76
+ }
77
+
78
+ /**
79
+ * The one device failure that is known to heal by itself.
80
+ *
81
+ * The guest's window server dies, the launch that was in flight fails, and
82
+ * SpringBoard comes back a few seconds later. Item 173 spent a dozen
83
+ * occurrences unable to observe it for exactly that reason — by the time
84
+ * anything looked, `diagnose` said `healthy`, fusion 0.857.
85
+ *
86
+ * Measured on `326464A4` on 2026-09-18, six occurrences in one run: waiting and
87
+ * launching again recovered **6 of 6**, in 5.7-7.9 s (median ~6.3 s). The
88
+ * remedy in use until now was `simframe revive`, a ~40 s device restart — six
89
+ * times the cost, for a fault that was already over.
90
+ *
91
+ * Separate from `deviceCause` on purpose. Every entry there says "this is the
92
+ * device"; this one says "and it will be back". `simctl did not return within
93
+ * 90s` is the device too and is deliberately NOT here: it was never once
94
+ * observed to recover, and a retry costs another 90 s to find that out.
95
+ */
96
+ export const SHELL_CRASH = /system shell \(SpringBoard[^)]*\) probably crashed/i;
97
+
98
+ /** Did this failure come from the guest shell dying under us? */
99
+ export function shellCrashed(text) {
100
+ return SHELL_CRASH.test(String(text ?? ''));
101
+ }
package/src/index.js CHANGED
@@ -1397,22 +1397,76 @@ export async function getFrameAt(deviceQuery, { msAgo = 0, options } = {}) {
1397
1397
  * Wait, briefly, for a frame that is holding still. Returns whatever the newest
1398
1398
  * frame is once the screen settles or the budget runs out, saying which.
1399
1399
  */
1400
- export async function settledState(udid, { settleMs = MEMORY_SETTLE_MS, timeoutMs = 1500 } = {}) {
1400
+ /**
1401
+ * When the stillness a state reports began, or null when that cannot be told.
1402
+ *
1403
+ * `stableForMs` is measured as of the frame's capture, so the quiet period
1404
+ * started `stableForMs` before `capturedAt`. Exported because the rule built on
1405
+ * it below is the interesting part and deserves to be testable without a device.
1406
+ */
1407
+ export function stillnessBegan(state) {
1408
+ if (!Number.isFinite(state?.capturedAt) || !Number.isFinite(state?.stableForMs)) return null;
1409
+ return state.capturedAt - state.stableForMs;
1410
+ }
1411
+
1412
+ /**
1413
+ * Has this screen been still *since we acted*, or was it still before we did?
1414
+ *
1415
+ * The distinction the settle detector was missing. It answers "how long has the
1416
+ * screen been quiet" honestly and has no idea that the quiet it is describing
1417
+ * belongs to the screen the caller has just left.
1418
+ *
1419
+ * Only a stillness we can **prove** predates the action is rejected. When the
1420
+ * timestamps are missing this returns true, which is the previous behaviour —
1421
+ * this may only ever add refusals it can demonstrate, never turn an unknown
1422
+ * into a wait.
1423
+ */
1424
+ export function stillSinceActing(state, actedAt) {
1425
+ if (!Number.isFinite(actedAt)) return true;
1426
+ const began = stillnessBegan(state);
1427
+ if (began == null) return true;
1428
+ return began >= actedAt;
1429
+ }
1430
+
1431
+ export async function settledState(udid, { settleMs = MEMORY_SETTLE_MS, timeoutMs = 1500, since } = {}) {
1401
1432
  const p = store.paths(udid);
1402
1433
  const deadline = Date.now() + timeoutMs;
1434
+ // What we are settling *after*. A caller that knows may say; otherwise it is
1435
+ // the last launch, openUrl or gesture this device was given.
1436
+ const actedAt = since ?? store.lastActionAt(udid);
1403
1437
  let state = store.readJson(p.state);
1438
+ let stale = false;
1404
1439
  while (Date.now() < deadline) {
1405
1440
  state = store.readJson(p.state) ?? state;
1406
- // The daemon runs a real settle detector that can tell a spinner from a
1407
- // still screen. Prefer it; the duration check is the fallback for the
1408
- // simctl engine, which has no such thing.
1409
- if (state?.settled === true) return { state, settled: true };
1410
- if (state && state.settled === undefined && state.stableForMs >= settleMs) {
1411
- return { state, settled: true };
1441
+ // **Stillness from before the action is not settlement.** Measured on the
1442
+ // bench device: 283ms after a Settings launch the state said `settled` with
1443
+ // `stableForMs: 4427` over a screen holding **zero** elements, which filled
1444
+ // 2.3 seconds later; 408ms after `tap General` it said `settled` with
1445
+ // `stableForMs: 7753` over the 23 elements of Settings root. Both readings
1446
+ // were honest about the number and wrong about the screen.
1447
+ //
1448
+ // Field-reported independently as the highest-priority class here: a read
1449
+ // that returns chrome with the content missing and no loading marker is
1450
+ // indistinguishable from a screen that is genuinely empty, so an agent
1451
+ // reports "this filter returns zero results" and means it.
1452
+ const fresh = stillSinceActing(state, actedAt);
1453
+ if (!fresh) stale = true;
1454
+ else {
1455
+ // The daemon runs a real settle detector that can tell a spinner from a
1456
+ // still screen. Prefer it; the duration check is the fallback for the
1457
+ // simctl engine, which has no such thing.
1458
+ if (state?.settled === true) return { state, settled: true, waitedForAction: stale };
1459
+ if (state && state.settled === undefined && state.stableForMs >= settleMs) {
1460
+ return { state, settled: true, waitedForAction: stale };
1461
+ }
1412
1462
  }
1413
1463
  await sleep(40);
1414
1464
  }
1415
- return { state, settled: false };
1465
+ // `settled: false` is not a failure and never was — callers use the state and
1466
+ // decline to persist a map built from it. What is new is that this can now be
1467
+ // false because the screen has not been seen to move since the action, which
1468
+ // is a different and more honest reason than "it is still moving".
1469
+ return { state, settled: false, ...(stale ? { stillnessPredatesAction: true } : {}) };
1416
1470
  }
1417
1471
 
1418
1472
  /**
@@ -1459,6 +1513,33 @@ export function offScreenMatch(targets, query, points) {
1459
1513
  return hit.status === 'ambiguous' && hit.alternatives?.length ? hit.alternatives[0] : null;
1460
1514
  }
1461
1515
 
1516
+ /**
1517
+ * Which way a target is outside the viewport, and by which measure.
1518
+ *
1519
+ * `offViewport` has checked both axes since it was written; the sentence
1520
+ * reporting it only ever named `y` and the screen *height*. So a tab in a
1521
+ * horizontally-scrolling strip was reported as *"it is at y=143 on a 874pt
1522
+ * screen"* — a coordinate plainly inside the screen, next to a conclusion that
1523
+ * it is not in view, with four of its siblings visible. Field-reported, and the
1524
+ * reporter's summary is the right one: the honest half was right and only the
1525
+ * axis was wrong.
1526
+ *
1527
+ * Vertical is checked first because it is overwhelmingly the common case, and
1528
+ * because `scrollTo` reasons vertically — a caller told "below the fold" and a
1529
+ * caller told "past the right edge" do different things next, which is the
1530
+ * whole reason to say which.
1531
+ */
1532
+ export function offScreenAxis(target, points) {
1533
+ const { width, height } = points ?? {};
1534
+ if (Number.isFinite(height) && (target.y < 0 || target.y > height)) {
1535
+ return { axis: 'y', at: Math.round(target.y), extent: Math.round(height), edge: target.y < 0 ? 'above' : 'below' };
1536
+ }
1537
+ if (Number.isFinite(width) && (target.x < 0 || target.x > width)) {
1538
+ return { axis: 'x', at: Math.round(target.x), extent: Math.round(width), edge: target.x < 0 ? 'left of' : 'right of' };
1539
+ }
1540
+ return null;
1541
+ }
1542
+
1462
1543
  /**
1463
1544
  * Which sensors a read asks for by default.
1464
1545
  *
@@ -1500,9 +1581,52 @@ export function sensorMode(options) {
1500
1581
  */
1501
1582
  const nameFor = (t) => t.label || t.identifier || null;
1502
1583
 
1584
+ /**
1585
+ * Did screen memory alone produce this miss?
1586
+ *
1587
+ * A recall that finds the screen and not the target tags `ambiguous_intent`
1588
+ * without `ambiguous` — "I know this screen, and what you asked for is not on
1589
+ * it". That is the only miss worth re-asking, because it is the only one whose
1590
+ * answer came from a file rather than from the device. A miss off a map that
1591
+ * was just built has already read both sensors, and a genuine ambiguity is not
1592
+ * settled by reading again.
1593
+ */
1594
+ export function memoryMiss(err) {
1595
+ const why = metrics.escalationOf(err);
1596
+ return Boolean(why && why.reason === 'ambiguous_intent' && !why.ambiguous);
1597
+ }
1598
+
1503
1599
  export async function locate(deviceQuery, query, opts = {}) {
1504
1600
  if (sensorMode(opts.options) !== 'ax-first' || opts.useOcr === false || opts.escalated) {
1505
- return locateWith(deviceQuery, query, opts);
1601
+ if (opts.refresh || opts.escalated) return locateWith(deviceQuery, query, opts);
1602
+ try {
1603
+ return await locateWith(deviceQuery, query, opts);
1604
+ } catch (err) {
1605
+ // **Memory may confirm, never deny.** Measured on the bench device, 8 of
1606
+ // 8 and 4 of 4 in two separate runs: at the instant a flow's next step
1607
+ // asks, a recall says "not on this screen" in 25 ms and a fresh read
1608
+ // finds the target 1.8 s later, on the same device, without anything
1609
+ // touching it in between.
1610
+ //
1611
+ // Why the recall is wrong, and it is not staleness in the usual sense:
1612
+ // the frame it keys on was captured 47-124 ms *after* the tap — the first
1613
+ // frame of the push animation, which still looks like the screen being
1614
+ // left. Capture is damage-driven, so that frame then goes still,
1615
+ // `settledState` calls it settled at 500-700 ms of stillness, and
1616
+ // `recallNearest` — deliberately tolerant, because a list with new rows is
1617
+ // still the same screen — matches it back to the previous screen and
1618
+ // answers out of that screen's stored element list. The screen itself
1619
+ // arrives about 750 ms later.
1620
+ //
1621
+ // The cost is paid only on a miss, which today aborts the batch and buys
1622
+ // a ~20 s model round trip. 1.8 s to be sure is the cheaper mistake. The
1623
+ // hit path — the one the speed argument rests on — is untouched.
1624
+ //
1625
+ // This recovery already existed for `ax-first` (below) and had never run
1626
+ // in the default sensor mode, which is `full`.
1627
+ if (!memoryMiss(err)) throw err;
1628
+ return locateWith(deviceQuery, query, { ...opts, refresh: true, escalated: true });
1629
+ }
1506
1630
  }
1507
1631
  const options = opts;
1508
1632
  try {
@@ -1723,8 +1847,12 @@ async function locateWith(
1723
1847
  if (offScreen) {
1724
1848
  throw metrics.tag(
1725
1849
  new Error(
1726
- `"${query}" is in the tree but not in view — it is at y=${Math.round(offScreen.y)}`
1727
- + ` on a ${Math.round(points.height)}pt screen. Scroll to it (sim_scroll_to) rather than waiting;`
1850
+ `"${query}" is in the tree but not in view — ${(() => {
1851
+ const off = offScreenAxis(offScreen, points);
1852
+ if (!off) return `it is at ${Math.round(offScreen.x)},${Math.round(offScreen.y)}`;
1853
+ return `it is ${off.edge} the viewport, at ${off.axis}=${off.at}`
1854
+ + ` on a ${off.extent}pt ${off.axis === 'y' ? 'tall' : 'wide'} screen`;
1855
+ })()}. Scroll to it (sim_scroll_to) rather than waiting;`
1728
1856
  + ' waiting cannot bring it into view.',
1729
1857
  ),
1730
1858
  from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',