simframe 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +165 -5
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/simframed/main.swift +13 -1
  9. package/native/supervise.swift +181 -0
  10. package/package.json +4 -1
  11. package/scripts/check-package.mjs +22 -2
  12. package/scripts/check-private.mjs +143 -0
  13. package/scripts/ci-memory.mjs +104 -20
  14. package/scripts/eval-perception.mjs +281 -0
  15. package/scripts/phase17-corpus.mjs +176 -0
  16. package/skills/simframe/SKILL.md +237 -5
  17. package/src/actions.js +1825 -38
  18. package/src/analyze.js +70 -0
  19. package/src/cli.js +214 -15
  20. package/src/control.js +1 -0
  21. package/src/fingerprint.js +19 -1
  22. package/src/graph.js +193 -11
  23. package/src/index.js +428 -16
  24. package/src/input.js +115 -8
  25. package/src/localhelper.js +155 -0
  26. package/src/matching.js +119 -3
  27. package/src/mcp.js +319 -27
  28. package/src/metrics.js +134 -8
  29. package/src/navigate.js +10 -7
  30. package/src/ocr.js +18 -1
  31. package/src/planner.js +195 -0
  32. package/src/platform/android.js +3 -2
  33. package/src/platform/ios.js +2 -1
  34. package/src/png.js +26 -0
  35. package/src/refs.js +51 -8
  36. package/src/regions.js +110 -1
  37. package/src/screenmap.js +109 -10
  38. package/src/supervisor.js +117 -0
  39. package/src/view.js +396 -7
  40. package/src/vocabulary.js +134 -0
  41. package/src/wrote.js +136 -0
package/src/index.js CHANGED
@@ -5,10 +5,14 @@ import path from 'node:path';
5
5
  import { fileURLToPath } from 'node:url';
6
6
  import { DEFAULTS, STATE_VERSION } from './daemon.js';
7
7
  import * as engine from './engine.js';
8
+ import * as regions from './regions.js';
8
9
  import { decodePng, encodePng, scaleBitmap } from './png.js';
9
10
  import {
10
11
  REGION_COLS,
11
12
  hexToSignature,
13
+ isBlackFrame,
14
+ maxCellDelta,
15
+ CELL_CHANGE,
12
16
  regionDeltas,
13
17
  regionMap,
14
18
  signatureDiff,
@@ -398,6 +402,80 @@ export function resolveBaseline(state, since) {
398
402
  return { kind: 'unmatched', requested: key };
399
403
  }
400
404
 
405
+ /**
406
+ * Is the baseline describing a screen that had already finished moving?
407
+ *
408
+ * Pure, so the rule can be argued with in a test rather than only observed on
409
+ * a device. The full reasoning is at the call site in `waitFor`; the short
410
+ * version is that stillness cannot accumulate in the milliseconds between a
411
+ * dispatch returning and a wait beginning, so a screen that already differs
412
+ * from the baseline *and* has already been at rest for the whole stillness
413
+ * window changed for some earlier reason.
414
+ */
415
+ /** The newest frame's region signature, which the state already carries. */
416
+ function currentSig(state) {
417
+ const h = state?.history;
418
+ if (!h?.length) return null;
419
+ const newest = h.find((x) => x.seq === state.seq) ?? h[h.length - 1];
420
+ return newest?.sig ?? null;
421
+ }
422
+
423
+ /**
424
+ * The longest pause inside a transition, measured after the transition is over.
425
+ *
426
+ * This is the unbiased half of the estimator that was reverted in Phase 11, and
427
+ * the difference is entirely about *when* the measurement stops.
428
+ *
429
+ * The biased version accumulated the statistic from inside the wait: the
430
+ * longest stretch of stillness the wait itself happened to observe. Feed that
431
+ * back into how long the next wait runs and it eats itself — a wait that ends
432
+ * early never sees the pauses that come later, so the gaps read as zero, the
433
+ * window ratchets down, the next wait ends earlier still, and eventually a
434
+ * settle returns mid-transition and the graph learns a screen it never reached.
435
+ * That is not a theory; it corrupted this device's graph in one afternoon.
436
+ *
437
+ * This one reads the frame history *after* the fact, over a window whose end is
438
+ * not decided by the wait. The frames are already on disk with their timestamps
439
+ * and their hashes, so the true profile of a transition is recoverable as long
440
+ * as the history still reaches back to it.
441
+ *
442
+ * That last condition is the whole reason this returns null rather than a
443
+ * number: the ring is bounded, and during fast motion it holds a second or two.
444
+ * A partial window would produce a *shorter* gap than really occurred, which is
445
+ * the exact direction of the bias being removed. Reporting nothing is the only
446
+ * honest answer to a question the evidence cannot reach.
447
+ */
448
+ export function longestQuietGap(history, sinceMs, untilMs = Date.now()) {
449
+ if (!Array.isArray(history) || history.length < 2) return null;
450
+ const frames = history
451
+ .filter((f) => Number.isFinite(f?.at) && f.at >= sinceMs && f.at <= untilMs)
452
+ .sort((a, b) => a.at - b.at);
453
+ if (frames.length < 2) return null;
454
+ // The history has to reach back to the action itself. One frame interval of
455
+ // slack, because the frame that captures the moment of the action is not
456
+ // required to land exactly on it.
457
+ const span = frames[1].at - frames[0].at;
458
+ if (frames[0].at > sinceMs + Math.max(250, span)) return null;
459
+
460
+ let longest = 0;
461
+ let lastChangeAt = frames[0].at;
462
+ for (let i = 1; i < frames.length; i += 1) {
463
+ if (frames[i].hash !== frames[i - 1].hash) {
464
+ longest = Math.max(longest, frames[i].at - lastChangeAt);
465
+ lastChangeAt = frames[i].at;
466
+ }
467
+ }
468
+ // The quiet after the last change is not a pause *inside* the transition —
469
+ // it is the transition being over, which is what a settle already measures.
470
+ return longest;
471
+ }
472
+
473
+ export function baselineAlreadySettled({ mode, changedAtStart, stableForMs, stableMs } = {}) {
474
+ if (mode === 'stable' || !changedAtStart) return false;
475
+ if (!Number.isFinite(stableForMs) || !Number.isFinite(stableMs)) return false;
476
+ return stableForMs >= stableMs;
477
+ }
478
+
401
479
  function compareToBaseline(state, baseline) {
402
480
  if (!baseline) return null;
403
481
  if (baseline.kind === 'history') {
@@ -459,9 +537,31 @@ export async function getFrame(deviceQuery, { detail = 'normal', options } = {})
459
537
  let scaledOnRead = false;
460
538
  if (maxDim === 0) {
461
539
  file = state.fullFile;
462
- } else if (maxDim > nativeMax + 8 && fs.existsSync(state.fullFile)) {
540
+ } else if (maxDim > nativeMax + 8) {
541
+ // Asking for more detail than the ring frame holds, so it has to come from
542
+ // a full-resolution frame. The `fs.existsSync(state.fullFile)` this used to
543
+ // require is exactly the condition that fails routinely — retention thins
544
+ // full frames aggressively, and `state.fullFile` names one that is often
545
+ // already gone. When it did, this fell straight through to `latest.png` and
546
+ // returned a 322x700 image while still calling itself detail "high".
547
+ //
548
+ // That was not a small inaccuracy. Every single `sim_look` header in a
549
+ // two-agent field round read `322x700` on a 402x874pt device, so "high
550
+ // (1024px, readable small text)" was interpolating a sub-1x source the
551
+ // whole time. `fullFrameFor` has always known how to find or take a full
552
+ // frame — state named full/9660.png while the directory held eight frames
553
+ // at 1206x2622 — and this path simply never asked it.
554
+ //
555
+ // What this does *not* explain, though it is tempting: the OCR corruption
556
+ // in that round's log (`saaiad@example.com`, `suomit order`). OCR has
557
+ // always gone through `fullFrameFor`, and the daemon's own text recognition
558
+ // reads the live surface at native resolution, so neither was ever looking
559
+ // at the small frame. Those errors are a desktop-width page rendering its
560
+ // labels at a few pixels tall. Fixing what the caller *sees* does not fix
561
+ // what OCR reads, and saying so here keeps the next reader from assuming it did.
562
+ const source = await fullFrameFor(device.udid, state);
463
563
  const out = path.join(p.dir, `read-${maxDim}.png`);
464
- await resize(state.fullFile, out, maxDim);
564
+ await resize(source, out, maxDim);
465
565
  file = out;
466
566
  scaledOnRead = true;
467
567
  } else if (maxDim < nativeMax - 8) {
@@ -473,17 +573,50 @@ export async function getFrame(deviceQuery, { detail = 'normal', options } = {})
473
573
 
474
574
  const png = fs.readFileSync(file);
475
575
  const bmp = pngSize(png);
576
+ // The age must describe the *bytes*, not the state that dates them.
577
+ //
578
+ // These are two different files. `state.json` is written when a frame is
579
+ // recorded; the image is a separate write. A freshly started daemon publishes
580
+ // fresh state while `latest.png` is still the previous session's — which is
581
+ // when a caller's first look of a session happens. Reported, and it is the
582
+ // worst failure this tool can have: a header reading `frame #714 · 84ms old ·
583
+ // still for 17173ms` above an image whose status-bar clock said 6:11 when the
584
+ // real time was 11:10. `sim_ui` was right in the same session because the
585
+ // accessibility tree is read live and in-process; only the image comes from a
586
+ // file, so only the image can be hours stale while the header says otherwise.
587
+ //
588
+ // Taking the *larger* of the two ages cannot overstate freshness. It can
589
+ // overstate staleness by however long the two writes are apart, which is 6ms
590
+ // measured on a healthy daemon — the right direction to be wrong in.
591
+ let fileAgeMs = null;
592
+ try { fileAgeMs = Date.now() - fs.statSync(file).mtimeMs; } catch { /* stat is advisory */ }
593
+ const stateAgeMs = Date.now() - state.capturedAt;
594
+ const ageMs = Number.isFinite(fileAgeMs) ? Math.max(stateAgeMs, fileAgeMs) : stateAgeMs;
476
595
  return {
477
596
  device,
478
597
  state,
479
598
  png,
480
599
  width: bmp.width,
481
600
  height: bmp.height,
482
- ageMs: Date.now() - state.capturedAt,
601
+ ageMs,
602
+ // Said out loud when the image is materially older than the state, because
603
+ // "this picture is not of the screen the rest of this response describes" is
604
+ // not something a caller can work out for themselves.
605
+ frameBehindMs: Number.isFinite(fileAgeMs) && fileAgeMs - stateAgeMs > FRAME_BEHIND_MS
606
+ ? Math.round(fileAgeMs - stateAgeMs)
607
+ : null,
483
608
  scaledOnRead,
484
609
  };
485
610
  }
486
611
 
612
+ /**
613
+ * How far the image may lag the state before it is worth saying so.
614
+ *
615
+ * Measured on a healthy daemon, the two writes land 6ms apart. A second is far
616
+ * outside that and far inside the hours-stale case this exists to catch.
617
+ */
618
+ const FRAME_BEHIND_MS = 1000;
619
+
487
620
  function pngSize(png) {
488
621
  return { width: png.readUInt32BE(16), height: png.readUInt32BE(20) };
489
622
  }
@@ -497,6 +630,10 @@ export async function getState(deviceQuery, { since, options, inputHealth = fals
497
630
  map: regionMap(state.regions || [], REGION_COLS),
498
631
  since: compareToBaseline(state, resolveBaseline(state, since)),
499
632
  live: liveness(device.udid, state),
633
+ // Costs 32 integer comparisons on a signature already computed, and it is
634
+ // the difference between "the screen is calm" and "the display stopped
635
+ // rendering" — which looked identical to everything above this line.
636
+ black: isBlackFrame(currentSig(state)),
500
637
  // Off by default and asked for by the state commands only. A flow step
501
638
  // calls getState twice, and the fix for a stale session runs before every
502
639
  // action anyway (input.ensureFreshSession) — this is the report, not the
@@ -568,7 +705,7 @@ export async function waitFor(
568
705
  const p = store.paths(device.udid);
569
706
  const requested = since ?? baselineHash;
570
707
  const resolved = resolveBaseline(first, requested);
571
- const baselineHashValue =
708
+ let baselineHashValue =
572
709
  resolved?.kind === 'history' ? resolved.entry.hash : (requested ?? first.hash);
573
710
  const baselineResolved = resolved?.kind === 'history' || requested == null;
574
711
 
@@ -590,10 +727,62 @@ export async function waitFor(
590
727
  */
591
728
  let quietGapMs = 0;
592
729
  let quietRun = 0;
730
+ /**
731
+ * The baseline's own signature, for changes the mean cannot see.
732
+ *
733
+ * Only when the baseline was found in the frame history — a hash we cannot
734
+ * place has no signature to compare against, and guessing one would be worse
735
+ * than not looking.
736
+ */
737
+ const baselineSig = resolved?.kind === 'history' && resolved.entry?.sig
738
+ ? hexToSignature(resolved.entry.sig)
739
+ : (requested == null ? hexToSignature(currentSig(first) ?? '') : null);
740
+ let smallChange = false;
741
+ /** Frames the display was not rendering at all. See `isBlackFrame`. */
742
+ let blackFrames = 0;
743
+ let blackSinceStart = null;
593
744
  let lastHash = first.hash;
594
745
  let sawChange = mode === 'stable' || first.hash !== baselineHashValue;
595
746
  const changedAtStart = sawChange && mode !== 'stable';
596
747
 
748
+ /**
749
+ * The baseline describes a screen that has already finished moving.
750
+ *
751
+ * `since` means "the screen as it was before the action", and the whole
752
+ * reliability of these scripts rests on it being captured *before* rather
753
+ * than after — a baseline sampled afterwards is the commonest way to wait for
754
+ * a change that already happened. What was missing is the other end of it:
755
+ * time also passes between capturing the baseline and dispatching the action,
756
+ * and in a flow step that gap holds a `locate`, a perception pass and a
757
+ * settle wait — hundreds of milliseconds, not microseconds.
758
+ *
759
+ * So a transition can begin *and finish* in that gap, and then `sawChange` is
760
+ * true at wait start because of the previous action's animation. Measured:
761
+ * `tap Accessibility` returned `settled 124ms` against a 500 ms stillness
762
+ * window, the screen had never left the Settings root, and the graph recorded
763
+ * `root -> root` as a verified edge — count 11, changedOutcomes 5, flipping
764
+ * between the real destination and itself all day.
765
+ *
766
+ * The test is unambiguous rather than clever: the screen differs from the
767
+ * baseline *and has already been at rest for the full stillness window*.
768
+ * Stillness cannot have accumulated in the milliseconds between a dispatch
769
+ * returning and this call starting, so whatever changed, changed and settled
770
+ * before we looked, and it is not this action's doing. Re-baseline to what is
771
+ * actually on screen and wait for a further change — which is what the caller
772
+ * asked for and what a stale hash prevented.
773
+ *
774
+ * A screen that differs and is *still moving* is left alone: that is
775
+ * genuinely ambiguous, and after an action the usual reading is the right
776
+ * one.
777
+ */
778
+ const staleBaseline = baselineAlreadySettled({
779
+ mode, changedAtStart, stableForMs: first.stableForMs, stableMs,
780
+ });
781
+ if (staleBaseline) {
782
+ baselineHashValue = first.hash;
783
+ sawChange = false;
784
+ }
785
+
597
786
  const done = (satisfied, extra = {}) => ({
598
787
  device,
599
788
  state: last,
@@ -602,6 +791,15 @@ export async function waitFor(
602
791
  quietGapMs,
603
792
  sawChange,
604
793
  changedBeforeWait: changedAtStart,
794
+ // Something moved, but only in one region — a control changing state
795
+ // rather than a screen changing.
796
+ smallChange,
797
+ staleBaseline,
798
+ blackFrames,
799
+ // Said as an observation, never as a diagnosis: a screen can be black
800
+ // because the app drew black. What makes it the capture wedge is that it
801
+ // stays black while input is being delivered, and the caller knows that.
802
+ blackMs: blackSinceStart ? Date.now() - blackSinceStart : 0,
605
803
  baselineHash: baselineHashValue,
606
804
  baselineResolved,
607
805
  waitedMs: Date.now() - startedAt,
@@ -617,8 +815,44 @@ export async function waitFor(
617
815
  const live = liveness(device.udid, state);
618
816
  if (!live.ok) return done(false, { stalled: true });
619
817
 
818
+ // A black frame is not evidence, in either direction.
819
+ //
820
+ // The capture wedge leaves every frame black while the whole capture path
821
+ // reports success, so before this a settle read the black screen as a
822
+ // change (the hash differs from anything) and then as a calm one (nothing
823
+ // moves), and returned `ok` for an action nobody could see the result of.
824
+ // It self-recovers most times, so the useful behaviour is to keep waiting
825
+ // rather than to conclude.
826
+ const black = isBlackFrame(currentSig(state));
827
+ if (black) {
828
+ blackFrames += 1;
829
+ blackSinceStart = blackSinceStart ?? Date.now();
830
+ await sleep(60);
831
+ continue;
832
+ }
833
+ blackSinceStart = null;
834
+
620
835
  if (!sawChange && state.hash !== baselineHashValue) sawChange = true;
621
836
 
837
+ // A change too small for the whole-screen mean to see.
838
+ //
839
+ // A switch flipping moves one cell of thirty-two by 0.043 and the mean by
840
+ // 0.0013 — a third of the threshold — so every switch, radio dot,
841
+ // checkbox and segment highlight was an action that "changed nothing",
842
+ // and `no-visible-change` is a verdict that escalates. See
843
+ // analyze.CELL_CHANGE for the measured gap this sits in.
844
+ //
845
+ // Deliberately feeding `sawChange` and *not* stillness: `stableForMs`
846
+ // stays on the mean, because a blinking text caret is a small localised
847
+ // change and a screen with a cursor would otherwise never settle.
848
+ if (!sawChange && baselineSig) {
849
+ const sig = currentSig(state);
850
+ if (sig && maxCellDelta(hexToSignature(sig), baselineSig) > CELL_CHANGE) {
851
+ sawChange = true;
852
+ smallChange = true;
853
+ }
854
+ }
855
+
622
856
  // A pause that turned out not to be the end of the transition. Only
623
857
  // pauses followed by more movement count: the quiet at the end of a
624
858
  // settle is the answer, not a gap.
@@ -890,10 +1124,76 @@ export async function readScreenWith(deviceQuery, { useAx = true, useOcr = true,
890
1124
  return { device, entry, points: { width: geo.pointWidth, height: geo.pointHeight } };
891
1125
  }
892
1126
 
893
- export async function locate(
1127
+ /**
1128
+ * A target that answers the query but sits outside the viewport.
1129
+ *
1130
+ * The off-screen filter above is right — an element in the tree below the fold
1131
+ * is untappable in fact — but "not on this screen" was the wrong way to say so.
1132
+ * Reported: a `waitFor` spent its full 15s timeout and stopped the flow while
1133
+ * the control sat one scroll down. Waiting cannot fix that and scrolling can.
1134
+ */
1135
+ export function offScreenMatch(targets, query, points) {
1136
+ // Both axes: a horizontal row puts elements past the right edge, and checking
1137
+ // only `y` reported them as visible.
1138
+ const off = (targets ?? []).filter((t) => t.label && regions.offViewport(t, points));
1139
+ if (!off.length) return null;
1140
+ const hit = matching.resolve(off, query);
1141
+ if (hit.status === 'ok') return hit.target;
1142
+ // Ambiguous off-screen is still an answer to "why is it not here".
1143
+ return hit.status === 'ambiguous' && hit.alternatives?.length ? hit.alternatives[0] : null;
1144
+ }
1145
+
1146
+ /**
1147
+ * Which sensors a read asks for by default.
1148
+ *
1149
+ * `full` is the default and what CLAUDE.md fixes: accessibility and OCR fused
1150
+ * into one element list. `ax-first` asks for the tree alone — 85 ms against
1151
+ * 142 ms warm — and pays for OCR only when the cheap read could not answer the
1152
+ * question.
1153
+ *
1154
+ * The escalation is the whole point, and it is what makes this safe to try. A
1155
+ * mode that just dropped OCR would lose every OCR-only element, which is
1156
+ * precisely how the map cut lost discovery: nothing became untappable, but the
1157
+ * agent could no longer see what was there. Here a resolve failure — the one
1158
+ * signal that says "the cheap sensor was not enough" — triggers a full read
1159
+ * before anyone is told the target is absent. Wrong guesses cost a second read;
1160
+ * they cannot cost a wrong answer.
1161
+ */
1162
+ export function sensorMode(options) {
1163
+ // Per call first, then the environment. The environment is fixed when a
1164
+ // process starts, and an MCP server is one long-lived process — so a tester
1165
+ // asked to compare two sensor modes in one session could not do it, which is
1166
+ // exactly what happened: round 6's A was run and B and C could not be. Their
1167
+ // own suggestion was three server entries with three env blocks, and it is
1168
+ // the worse fix: three servers on one device means three writers, against the
1169
+ // one-writer-per-device rule, and it makes an A/B a configuration change
1170
+ // rather than an argument.
1171
+ const asked = options?.sensor;
1172
+ const raw = String(asked ?? process.env.SIMFRAME_SENSOR ?? '').trim().toLowerCase();
1173
+ return raw === 'ax-first' || raw === 'axfirst' ? 'ax-first' : 'full';
1174
+ }
1175
+
1176
+ export async function locate(deviceQuery, query, opts = {}) {
1177
+ if (sensorMode(opts.options) !== 'ax-first' || opts.useOcr === false || opts.escalated) {
1178
+ return locateWith(deviceQuery, query, opts);
1179
+ }
1180
+ const options = opts;
1181
+ try {
1182
+ return await locateWith(deviceQuery, query, { ...options, useOcr: false, escalated: true });
1183
+ } catch (err) {
1184
+ // Only a perception failure earns the expensive retry. A refused selector or
1185
+ // an ambiguity between two things the tree *did* see is not going to be
1186
+ // settled by reading more text.
1187
+ const why = metrics.escalationOf(err);
1188
+ if (why?.reason !== 'unknown_screen' && why?.reason !== 'ambiguous_intent') throw err;
1189
+ return locateWith(deviceQuery, query, { ...options, useOcr: true, refresh: true, escalated: true });
1190
+ }
1191
+ }
1192
+
1193
+ async function locateWith(
894
1194
  deviceQuery,
895
1195
  query,
896
- { index, refresh = false, useAx = true, useOcr = true, settleMs = MEMORY_SETTLE_MS, options } = {},
1196
+ { index, refresh = false, useAx = true, useOcr = true, settleMs = MEMORY_SETTLE_MS, options, escalated } = {},
897
1197
  ) {
898
1198
  const { device, state: firstState } = await ensureDaemon(deviceQuery, options);
899
1199
  const udid = device.udid;
@@ -916,11 +1216,47 @@ export async function locate(
916
1216
  // be checked against structural identity without paying for a perception
917
1217
  // pass — which is the whole reason a ref exists.
918
1218
  const near = screenmap.recallNearest(udid, firstState.layoutHash);
919
- const hit = refs.resolveRef(udid, selector.ref, {
920
- layoutHash: firstState.layoutHash,
921
- structuralHash: near?.entry?.structuralHash ?? null,
922
- screenKnown: Boolean(near),
923
- });
1219
+ let hit;
1220
+ try {
1221
+ hit = resolveRefHere();
1222
+ } catch (err) {
1223
+ // A stale ref carries the label it was numbered against, so it need not
1224
+ // cost the rest of the batch. Reported: `#19 was numbered on a different
1225
+ // screen (61b835b7 → 6f34c006)` because dashboard cards finished loading
1226
+ // and shifted the layout — the same screen, a new hash — and that one
1227
+ // refusal aborted the three remaining steps.
1228
+ //
1229
+ // It re-resolves by label rather than by coordinate, and it *says so*.
1230
+ // The number is not honoured; the caller's own words are, which is what
1231
+ // they would have written instead. Resolving the old coordinates would be
1232
+ // the dangerous version of this, and is not what happens.
1233
+ // Only layout drift is recoverable. `identity` means the numbers were
1234
+ // drawn somewhere else, and the label they stood for appearing here is
1235
+ // coincidence rather than evidence — measured: `#1` numbered "Reminders"
1236
+ // in Reminders re-resolved in Contacts onto the status-bar back-to-app
1237
+ // breadcrumb "• Reminders", and reported ok.
1238
+ if (!err.staleRef || !err.staleLabel || err.staleKind !== 'drift') throw err;
1239
+ let again;
1240
+ try {
1241
+ again = await locateWith(deviceQuery, err.staleLabel, {
1242
+ index, refresh: true, useAx, useOcr, settleMs, options, escalated,
1243
+ });
1244
+ } catch (second) {
1245
+ // The number was not honoured and the label it stood for is not here
1246
+ // either. That is still a refusal to tap stale coordinates, but by the
1247
+ // time it surfaces it looks like an ordinary "not on this screen" and a
1248
+ // caller cannot tell the two apart. Carrying the marker across says
1249
+ // which question was actually asked.
1250
+ second.staleRef = true;
1251
+ second.staleLabel = err.staleLabel;
1252
+ throw second;
1253
+ }
1254
+ return {
1255
+ ...again,
1256
+ from: 'ref-relabelled',
1257
+ relabelled: { ref: selector.ref, label: err.staleLabel, why: err.message },
1258
+ };
1259
+ }
924
1260
  return {
925
1261
  device,
926
1262
  state: firstState,
@@ -929,6 +1265,23 @@ export async function locate(
929
1265
  distance: 0,
930
1266
  settled: true,
931
1267
  };
1268
+
1269
+ function resolveRefHere() {
1270
+ return refs.resolveRef(udid, selector.ref, {
1271
+ layoutHash: firstState.layoutHash,
1272
+ structuralHash: near?.entry?.structuralHash ?? null,
1273
+ // How far that recall reached. `recallNearest` is deliberately tolerant —
1274
+ // a list with new rows is still the same screen — so at any distance
1275
+ // above zero it is a *guess* about which screen this is, and a guess must
1276
+ // not be the sole grounds for refusing a ref. Reported from the field: a
1277
+ // refusal reading `#5 was numbered on a different screen
1278
+ // (0f7b9e3e → 48e5c92d)` where both calls' headers printed the identical
1279
+ // screen, because the map named the screen from a tolerant recall while
1280
+ // refs treated that same recall as exact.
1281
+ structuralDistance: near?.distance ?? null,
1282
+ screenKnown: Boolean(near),
1283
+ });
1284
+ }
932
1285
  }
933
1286
  if (selector.exact) query = selector.label;
934
1287
  // Key memory off a settled frame, never off whichever frame happened to be
@@ -985,7 +1338,11 @@ export async function locate(
985
1338
  `"${query}" matches ${outcome.alternatives.length} things on this screen — say which, or pass index: ${list}`,
986
1339
  ),
987
1340
  'ambiguous_intent',
988
- { candidates: outcome.alternatives },
1341
+ // Present, several times over — as opposed to absent, which also tags
1342
+ // ambiguous_intent when the screen was one we thought we knew. A
1343
+ // waiting caller needs the difference: more time cannot make a thing
1344
+ // unique, and it can make an absent thing arrive.
1345
+ { candidates: outcome.alternatives, ambiguous: true, intent: query },
989
1346
  );
990
1347
  }
991
1348
  if (outcome.status === 'ok') {
@@ -999,8 +1356,25 @@ export async function locate(
999
1356
  // here undoes every guard above — it has no off-screen filter and no
1000
1357
  // coverage weighting, and it is what returned a scrolled-away list row for
1001
1358
  // "back". "Not found" is the correct answer.
1002
- const visible = entry.targets.filter((t) => t.label && t.y >= 0 && t.y <= points.height);
1359
+ const visible = entry.targets.filter((t) => t.label && !regions.offViewport(t, points));
1003
1360
  const sample = visible.slice(0, 12).map((t) => t.label.slice(0, 24)).join(', ');
1361
+ // "Not on this screen" and "not in view" are different answers, and giving
1362
+ // the first for the second cost a reported 15 seconds: a `waitFor REVIEW`
1363
+ // burned its whole timeout while REVIEW sat one scroll below the fold, and
1364
+ // then stopped the flow. Waiting cannot bring a thing into view, and
1365
+ // scrolling can — so the difference is the whole of what to do next.
1366
+ const offScreen = offScreenMatch(entry.targets, query, points);
1367
+ if (offScreen) {
1368
+ throw metrics.tag(
1369
+ new Error(
1370
+ `"${query}" is in the tree but not in view — it is at y=${Math.round(offScreen.y)}`
1371
+ + ` on a ${Math.round(points.height)}pt screen. Scroll to it (sim_scroll_to) rather than waiting;`
1372
+ + ' waiting cannot bring it into view.',
1373
+ ),
1374
+ from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
1375
+ { candidates: visible.slice(0, 8), intent: query },
1376
+ );
1377
+ }
1004
1378
  // Which escalation this is depends on whether the screen was recognised.
1005
1379
  // Screen memory had nothing for it (`from` is one of the built values) and
1006
1380
  // the target is missing: that is not knowing the screen. On a screen
@@ -1008,7 +1382,7 @@ export async function locate(
1008
1382
  throw metrics.tag(
1009
1383
  new Error(`"${query}" is not on this screen. Visible: ${sample || '(nothing readable)'}`),
1010
1384
  from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
1011
- { candidates: visible.slice(0, 8) },
1385
+ { candidates: visible.slice(0, 8), intent: query },
1012
1386
  );
1013
1387
  }
1014
1388
 
@@ -1020,7 +1394,7 @@ export async function locate(
1020
1394
  throw metrics.tag(
1021
1395
  new Error(`"${query}" is not on this screen. Visible: ${sample || '(nothing readable)'}`),
1022
1396
  from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
1023
- { candidates: visible.slice(0, 8) },
1397
+ { candidates: visible.slice(0, 8), intent: query },
1024
1398
  );
1025
1399
  }
1026
1400
  return { device, state: current, entry, target, from, distance, settled, screens: screenmap.stats(udid).screens };
@@ -1039,6 +1413,38 @@ export async function locate(
1039
1413
  */
1040
1414
  export const STRUCTURAL_SETTLE_MS = 300;
1041
1415
 
1416
+ /**
1417
+ * How much of the structural window is still owed, given when the last sample's
1418
+ * frame was captured.
1419
+ *
1420
+ * This was `sleep(300)` between the two readings, and unlike every other fixed
1421
+ * wait in the engine it cannot be replaced by waiting for a signal — because
1422
+ * there is no signal. The race it guards is a screen whose *pixels* have gone
1423
+ * still while its structure has not: a list whose spinner has gone and whose
1424
+ * rows have not landed is perfectly quiet and structurally wrong, so the settle
1425
+ * detector, which watches pixels, has nothing to report. Only elapsed time
1426
+ * separates the two readings.
1427
+ *
1428
+ * What can be fixed is that the wait was *additional*. The guarantee wanted is
1429
+ * 300 ms between the frames the two samples read; the code slept 300 ms after a
1430
+ * sample that had already spent an unbounded settle wait and a full perception
1431
+ * pass getting there. On a screen that took a second to go quiet the separation
1432
+ * was already there and the sleep bought nothing but a second of it. So credit
1433
+ * what has passed and wait only for the remainder — the same guarantee, and
1434
+ * usually none of the sleep.
1435
+ *
1436
+ * The window itself is per-screen learnable, and worth noting that its
1437
+ * estimator has the *opposite* feedback sign to the one that corrupted the
1438
+ * graph: a window too short produces disagreeing samples, which lengthens it.
1439
+ * Self-correcting rather than self-reinforcing. It still waits on the
1440
+ * perception eval harness, because "the samples agreed" is only evidence the
1441
+ * window was long enough if the readings themselves are trustworthy.
1442
+ */
1443
+ export function structuralSettleOwed(capturedAt, now = Date.now(), windowMs = STRUCTURAL_SETTLE_MS) {
1444
+ if (!Number.isFinite(capturedAt)) return windowMs;
1445
+ return Math.max(0, Math.min(windowMs, windowMs - (now - capturedAt)));
1446
+ }
1447
+
1042
1448
  /**
1043
1449
  * How long identity will wait for pixels to go quiet.
1044
1450
  *
@@ -1101,6 +1507,11 @@ export async function screenIdentity(deviceQuery, { options, confirmNovel = true
1101
1507
  keyboard: Boolean(entry.keyboard),
1102
1508
  layoutHash: current.layoutHash,
1103
1509
  settled,
1510
+ // Settled and incomplete are different states and used to render
1511
+ // identically. A screen awaiting a network call is perfectly still; a
1512
+ // person sees a spinner and knows to wait. The classifier already says
1513
+ // so, and nothing above this line was asking.
1514
+ loading: current.transition?.kind === 'loading',
1104
1515
  // Carried out so callers that want the elements as well as the identity
1105
1516
  // do not pay for a second perception pass to get them. The compact
1106
1517
  // screen map needs both, and reading twice was the whole cost of it.
@@ -1122,7 +1533,8 @@ export async function screenIdentity(deviceQuery, { options, confirmNovel = true
1122
1533
  // Nothing recognises this, or the pixels have not gone quiet. Either way, make
1123
1534
  // it prove it is the same screen twice running before it becomes a node.
1124
1535
  for (let i = 1; i < STRUCTURAL_SETTLE_SAMPLES; i += 1) {
1125
- await sleep(STRUCTURAL_SETTLE_MS);
1536
+ const owed = structuralSettleOwed(identity.state?.capturedAt);
1537
+ if (owed > 0) await sleep(owed);
1126
1538
  const again = await read({ fresh: true });
1127
1539
  // Two readings agree if they are the same screen — the same test identity
1128
1540
  // itself uses. Demanding an identical hash is a stricter question than the